Skip to content
KernelIndex
Search⌘K

submission 670022

Gurkirat Singh · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 101 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-670022?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA A100
716.1µs
#83 of 96
2026-03-30

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:62d478a3bc0e35a206b883e84fd0ecd6d9131026f884173e450f7f69ddd2d38e
license declaredunknown
license concludedunknown
authorsGurkirat Singh
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

shared-memoryextern __shared__ float sdata[];

Kernel source

submission.py101 lines

import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t

cuda_source = """
#include <cuda_fp16.h>
#include <torch/extension.h>

// Shared memory for block-level reduction
__global__ void sum_reduce_kernel(
    const float* __restrict__ x,
    float* partial_sums,
    int N
) {
    extern __shared__ float sdata[];
    
    unsigned int tid = threadIdx.x;
    unsigned int idx = blockIdx.x * blockDim.x + threadIdx.x;
    
    // Load element (with boundary check)
    sdata[tid] = (idx < N) ? x[idx] : 0.0f;
    __syncthreads();
    
    // Parallel reduction in shared memory
    for (unsigned int s = blockDim.x / 2; s > 0; s >>= 1) {
        if (tid < s) {
            sdata[tid] += sdata[tid + s];
        }
        __syncthreads();
    }
    
    // Write partial sum for this block
    if (tid == 0) {
        partial_sums[blockIdx.x] = sdata[0];
    }
}

torch::Tensor sum_reduce_cuda(torch::Tensor x) {
    TORCH_CHECK(x.device().is_cuda(), "Must be CUDA tensor");
    TORCH_CHECK(x.dtype() == torch::kFloat32, "Must be float32");
    TORCH_CHECK(x.is_contiguous(), "Must be contiguous");
    
    int N = x.numel();
    const int BLOCK_SIZE = 1024;
    const int num_blocks = (N + BLOCK_SIZE - 1) / BLOCK_SIZE;
    
    // Allocate partial sums buffer
    auto partial_sums = torch::empty(num_blocks, 
                                      torch::dtype(torch::kFloat32)
                                      .device(x.device()));
    
    // Launch kernel
    sum_reduce_kernel<<<num_blocks, BLOCK_SIZE, BLOCK_SIZE * sizeof(float)>>>(
        x.data_ptr<float>(),
        partial_sums.data_ptr<float>(),
        N
    );
    
    cudaError_t err = cudaGetLastError();
    if (err != cudaSuccess) {
        throw std::runtime_error(cudaGetErrorString(err));
    }
    
    // Final reduction on GPU
    auto final_sum = partial_sums.sum();
    
    return final_sum;
}
"""

cpp_source = """
#include <torch/extension.h>
torch::Tensor sum_reduce_cuda(torch::Tensor x);
"""

# Load module once at module level
sum_module = load_inline(
    name='sum_reduce_cuda',
    cpp_sources=cpp_source,
    cuda_sources=cuda_source,
    functions=['sum_reduce_cuda'],
    verbose=False,
    extra_cuda_cflags=['-O3', '--use_fast_math'],
)


def custom_kernel(data: input_t) -> output_t:
    """
    Vector sum reduction using CUDA.
    """
    # Handle input format
    if isinstance(data, (tuple, list)):
        x = data[0]
    else:
        x = data
    
    # Ensure float32, contiguous, on GPU
    x = x.contiguous().cuda().to(torch.float32)
    
    return sum_module.sum_reduce_cuda(x)
scrolls · 101 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON