Skip to content
KernelIndex
Search⌘K

submission 682469

ngolhn · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 74 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-sort-v2-682469?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Sortsuite of 5 cases
NVIDIA B200
2.17ms
#9 of 23
2026-03-31

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:ae5665d7d64f12d0b177f1d0df6e9823002316ffa166e56c358830642b34bd43
license declaredunknown
license concludedunknown
authorsngolhn
imported2026-08-15

Kernel source

submission.py74 lines
#!POPCORN leaderboard sort_v2
#!POPCORN gpu B200

import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t

cuda_src = r"""
#include <torch/extension.h>
#include <cuda_runtime.h>
#include <cub/cub.cuh>

// Pre-allocated temp storage
static void* d_temp = nullptr;
static size_t d_temp_size = 0;

void sort_init(int64_t max_n) {
    // Query temp storage size for max problem size
    size_t needed = 0;
    cub::DeviceRadixSort::SortKeys(
        nullptr, needed,
        (const float*)nullptr, (float*)nullptr,
        (int)max_n);

    if (needed > d_temp_size) {
        if (d_temp) cudaFree(d_temp);
        // Allocate with some headroom
        d_temp_size = needed * 2;
        cudaMalloc(&d_temp, d_temp_size);
    }
}

void sort_float(int64_t in_ptr, int64_t out_ptr, int N) {
    size_t temp_size = d_temp_size;
    cub::DeviceRadixSort::SortKeys(
        d_temp, temp_size,
        reinterpret_cast<const float*>(in_ptr),
        reinterpret_cast<float*>(out_ptr),
        N);
}
"""

cpp_src = r"""
void sort_init(int64_t max_n);
void sort_float(int64_t in_ptr, int64_t out_ptr, int N);
"""

_ext = load_inline(
    name="cub_radix_sort",
    cpp_sources=cpp_src,
    cuda_sources=cuda_src,
    functions=["sort_init", "sort_float"],
    with_cuda=True,
    extra_cflags=["-O3"],
    extra_cuda_cflags=["-O3", "--use_fast_math", "-arch=sm_100a"],
    verbose=False,
)

# Pre-allocate temp storage for max size (100M elements)
_ext.sort_init(100_000_000)

# Warmup
_wd = torch.randn(1000, device="cuda", dtype=torch.float32)
_wo = torch.empty_like(_wd)
_ext.sort_float(_wd.data_ptr(), _wo.data_ptr(), _wd.numel())
torch.cuda.synchronize()
del _wd, _wo


def custom_kernel(data: input_t) -> output_t:
    data_in, output = data
    _ext.sort_float(data_in.data_ptr(), output.data_ptr(), data_in.numel())
    return output
scrolls · 74 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON