submission 67274
vyom · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 127 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-sort-v2-67274?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:4c9249bbb579f522b413c9bfdea08e2f0f26b86151c904119b095d61dc2c84a1
license declaredunknown
license concludedunknown
authorsvyom
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
custom_kernel = torch.compile(_custom_kernel, mode="max-autotune")Kernel source
submission.py127 lines
import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t
import os
# Ensure build directory exists
os.makedirs("./cuda_build_sort", exist_ok=True)
# Inline CUDA kernel using CUB's highly optimized radix sort
cuda_source = """
#include <torch/extension.h>
#include <cuda_runtime.h>
#include <cub/device/device_radix_sort.cuh>
#include <c10/cuda/CUDAStream.h>
torch::Tensor cub_sort_kernel(torch::Tensor data, torch::Tensor output) {
const int n = data.numel();
// Get raw pointers
float* d_keys_in = data.data_ptr<float>();
float* d_keys_out = output.data_ptr<float>();
// Get the CUDA stream
cudaStream_t stream = c10::cuda::getCurrentCUDAStream();
// Allocate temporary storage
void* d_temp_storage = nullptr;
size_t temp_storage_bytes = 0;
// Determine temporary device storage requirements
cub::DeviceRadixSort::SortKeys(
d_temp_storage,
temp_storage_bytes,
d_keys_in,
d_keys_out,
n,
0, // begin_bit
32, // end_bit (all 32 bits for float)
stream
);
// Allocate temporary storage
cudaMalloc(&d_temp_storage, temp_storage_bytes);
// Run sorting operation
cub::DeviceRadixSort::SortKeys(
d_temp_storage,
temp_storage_bytes,
d_keys_in,
d_keys_out,
n,
0, // begin_bit
32, // end_bit
stream
);
// Free temporary storage
cudaFree(d_temp_storage);
return output;
}
"""
cpp_source = """
torch::Tensor cub_sort_kernel(torch::Tensor data, torch::Tensor output);
"""
# Load the CUDA kernel with proper include paths
try:
# Find CUDA toolkit path
cuda_home = (
os.environ.get("CUDA_HOME") or os.environ.get("CUDA_PATH") or "/usr/local/cuda"
)
cub_sort_module = load_inline(
name="cub_radix_sort",
cpp_sources=cpp_source,
cuda_sources=cuda_source,
functions=["cub_sort_kernel"],
with_cuda=True,
extra_cuda_cflags=[
"-O3",
"--use_fast_math",
"-std=c++17",
f"-I{cuda_home}/include",
],
extra_include_paths=[f"{cuda_home}/include"],
build_directory="./cuda_build_sort",
verbose=True,
)
def _custom_kernel(data: input_t) -> output_t:
"""
Ultra-fast sort using CUB's DeviceRadixSort.
Args:
data: Tuple of (input_tensor, output_tensor)
Returns:
Sorted output tensor
"""
input_tensor, output_tensor = data
# Ensure contiguous memory layout for optimal performance
if not input_tensor.is_contiguous():
input_tensor = input_tensor.contiguous()
if not output_tensor.is_contiguous():
output_tensor = output_tensor.contiguous()
# Call CUB sort kernel
cub_sort_module.cub_sort_kernel(input_tensor, output_tensor)
return output_tensor
custom_kernel = _custom_kernel
print("Successfully compiled CUB radix sort kernel!")
except Exception as e:
print(f"Failed to compile CUB kernel: {e}")
print("Falling back to optimized torch.sort with torch.compile")
# Fallback to optimized torch.sort
def _custom_kernel(data: input_t) -> output_t:
data, output = data
output[...] = torch.sort(data)[0]
return output
custom_kernel = torch.compile(_custom_kernel, mode="max-autotune")
scrolls · 127 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON