Skip to content
KernelIndex
Search⌘K

submission 782561

Kernel-Zhang · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 64 lines, June 9 Researcher Reciprocity License v1.0.

A100_00029.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-prefixsum-v2-782561?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Inclusive prefix sumsuite of 11 cases
NVIDIA A100
1.39ms
#5 of 25
2026-05-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:5a7a67f93d6d1a3161e6ed4849e2a612963d740e100515e0ac6c88893f764528
license declaredunknown
license concludedunknown
authorsKernel-Zhang
imported2026-08-15

Kernel source

A100_00029.py64 lines
import torch
from torch.utils.cpp_extension import load_inline

N_ELEMENTS = 268435456

_CPP_SOURCE = r"""
#include <torch/extension.h>

torch::Tensor cuda_prefixsum(std::vector<torch::Tensor> data);

PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
    m.def("cuda_prefixsum", &cuda_prefixsum, "prefixsum with custom CUDA kernel");
}
"""

_CUDA_SOURCE = r"""
#include <cuda_runtime.h>
#include <torch/extension.h>
#include <cub/cub.cuh>

constexpr size_t N_SIZE = 268435456;


float* d_temp_storage = nullptr;
size_t temp_storage_bytes = 0;

torch::Tensor cuda_prefixsum(std::vector<torch::Tensor> data) {
    if (data[0].numel() == N_SIZE) {
        if (d_temp_storage == nullptr) {
            cub::DeviceScan::InclusiveSum(
            d_temp_storage, temp_storage_bytes,
            data[0].data_ptr<float>(), data[1].data_ptr<float>(), N_SIZE, 0);
            cudaMalloc(&d_temp_storage, temp_storage_bytes);
        }
        cub::DeviceScan::InclusiveSum(
        d_temp_storage, temp_storage_bytes,
        data[0].data_ptr<float>(), data[1].data_ptr<float>(), N_SIZE, 0);

        return data[1];
    } else { 
        return torch::cumsum(data[0], 0);
    }
}
"""

_EXT = load_inline(
        name="cuda_prefixsum_extension_0023",
        cpp_sources=[_CPP_SOURCE],
        cuda_sources=[_CUDA_SOURCE],
        functions=None,
        extra_cflags=["-O3"],
        extra_cuda_cflags=[
            "-O3", 
            "-use_fast_math", 
            "-Xptxas=-v", 
            "-Xptxas=-dlcm=cg",       # L2 Cache 全局策略
            "-Xptxas=-warn-spills",   # 监控寄存器溢出
            '-gencode=arch=compute_80,code=sm_80'
        ],
        with_cuda=True,
        verbose=True,
    )

custom_kernel = _EXT.cuda_prefixsum
scrolls · 64 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 782559.

Best evidence level for this revision: reported

JSON