Skip to content
KernelIndex
Search⌘K

submission 118172

Nick Nuon · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 195 lines, June 9 Researcher Reciprocity License v1.0.

submissions.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-118172?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA H100
827.2µs
#43 of 44
2025-12-01

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:c5095ff4c4fcf4c159555b39560d70eb20c80204ab72e1187e312e474c3b1253
license declaredunknown
license concludedunknown
authorsNick Nuon
imported2026-08-15

Kernel source

submissions.py195 lines
# !POPCORN leaderboard vectoradd_v2
# !POPCORN gpus H100
#
# #!/usr/bin/env python3
# import torch
# from torch.utils.cpp_extension import load_inline
#
# # 1) C++ side: declaration so Python can call into the CUDA implementation
# CPP_SRC = r"""
# #include <torch/extension.h>
#
# // Forward declaration; definition will be in the CUDA translation unit.
# torch::Tensor vecadd_cuda(torch::Tensor a, torch::Tensor b);
# """
#
# # 2) CUDA side: kernel + definition (float32 math internally)
# CUDA_SRC = r"""
# #include <torch/extension.h>
# #include <cuda.h>
# #include <cuda_runtime.h>
#
# // 1D float32 vector add: c[i] = a[i] + b[i]
# __global__ void vecadd_kernel(const float* __restrict__ a,
#                               const float* __restrict__ b,
#                               float* __restrict__ c,
#                               long n) {
#     long idx = blockIdx.x * blockDim.x + threadIdx.x;
#     if (idx < n) {
#         c[idx] = a[idx] + b[idx];
#     }
# }
#
# // Definition of vecadd_cuda declared in CPP_SRC
# torch::Tensor vecadd_cuda(torch::Tensor a, torch::Tensor b) {
#     auto a_c = a.contiguous();
#     auto b_c = b.contiguous();
#
#     auto out = torch::empty_like(a_c);
#     long n = a_c.numel();
#
#     const float* a_ptr = a_c.data_ptr<float>();
#     const float* b_ptr = b_c.data_ptr<float>();
#     float* out_ptr     = out.data_ptr<float>();
#
#     constexpr int threads = 256;
#     int blocks = (int)((n + threads - 1) / threads);
#
#     vecadd_kernel<<<blocks, threads>>>(a_ptr, b_ptr, out_ptr, n);
#
#     return out;
# }
# """
#
# # 3) Build extension once at import time.
# vecadd_mod = load_inline(
#     name="vecadd_cuda_ext",
#     cpp_sources=CPP_SRC,
#     cuda_sources=CUDA_SRC,
#     functions=["vecadd_cuda"],
#     with_cuda=True,
#     verbose=False,
# )
#
#
# @torch.inference_mode()
# def custom_kernel(data):
#     """
#     GPU Mode entrypoint.
#
#     The harness passes a tuple with at least two elements, e.g.:
#         (a, b, ...)
#
#     We:
#       - take the first two tensors,
#       - upcast to float32 on CUDA,
#       - run the CUDA vecadd,
#       - cast the result back to the original dtype of a,
#       - and return a single tensor.
#     """
#     # 1) Extract a, b from whatever tuple/list the harness gives
#     a = data[0]
#     b = data[1]
#
#     orig_dtype = a.dtype
#
#     device = torch.device("cuda")
#     a32 = a.to(device=device, dtype=torch.float32, non_blocking=True)
#     b32 = b.to(device=device, dtype=torch.float32, non_blocking=True)
#
#
#     # 5) Run CUDA vecadd on float32
#     out32 = vecadd_mod.vecadd_cuda(a32, b32)
#
#     # 6) Cast back to original dtype to match the reference
#     out = out32.to(dtype=orig_dtype)
#
#     return out
#

#!POPCORN leaderboard vectoradd_v2
#!POPCORN gpus L4

#!/usr/bin/env python3
import torch
from torch.utils.cpp_extension import load_inline

# 1) C++ side: declaration so Python can call into the CUDA implementation
CPP_SRC = r"""
#include <torch/extension.h>

// Forward declaration; definition will be in the CUDA translation unit.
torch::Tensor vecadd_cuda(torch::Tensor a, torch::Tensor b);
"""

# 2) CUDA side: templated kernel + dtype dispatch
CUDA_SRC = r"""
#include <torch/extension.h>
#include <cuda.h>
#include <cuda_runtime.h>
#include <ATen/ATen.h>

// 1D vector add: c[i] = a[i] + b[i] for arbitrary floating scalar_t
template <typename scalar_t>
__global__ void vecadd_kernel(const scalar_t* __restrict__ a,
                              const scalar_t* __restrict__ b,
                              scalar_t* __restrict__ c,
                              long n) {
    long idx = blockIdx.x * blockDim.x + threadIdx.x;
    if (idx < n) {
        c[idx] = a[idx] + b[idx];
    }
}

// Definition of vecadd_cuda declared in CPP_SRC
torch::Tensor vecadd_cuda(torch::Tensor a, torch::Tensor b) {
    auto a_c = a.contiguous();
    auto b_c = b.contiguous();

    auto out = torch::empty_like(a_c);
    long n = a_c.numel();

    constexpr int threads = 256;
    int blocks = (int)((n + threads - 1) / threads);

    // Dispatch on the actual scalar type: float32, float16, bfloat16, double, ...
    AT_DISPATCH_FLOATING_TYPES_AND_HALF(a_c.scalar_type(), "vecadd_cuda", [&] {
        const scalar_t* a_ptr = a_c.data_ptr<scalar_t>();
        const scalar_t* b_ptr = b_c.data_ptr<scalar_t>();
        scalar_t* out_ptr     = out.data_ptr<scalar_t>();

        vecadd_kernel<scalar_t><<<blocks, threads>>>(a_ptr, b_ptr, out_ptr, n);
    });

    return out;
}
"""

# 3) Build extension once at import time.
vecadd_mod = load_inline(
    name="vecadd_cuda_ext",
    cpp_sources=CPP_SRC,
    cuda_sources=CUDA_SRC,
    functions=["vecadd_cuda"],
    with_cuda=True,
    verbose=False,
)


@torch.inference_mode()
def custom_kernel(data):
    """
    GPU Mode entrypoint.

    The harness passes a tuple with at least two elements, e.g.:
        (a, b, ...)

    We:
      - grab a and b,
      - move them to CUDA (keeping their original dtype),
      - run the CUDA vecadd,
      - return a single tensor.
    No checks, no dtype conversions.
    """
    a = data[0]
    b = data[1]

    # Move to CUDA; keeps dtype, so we stay in orig_dtype
    a = a.cuda(non_blocking=True)
    b = b.cuda(non_blocking=True)

    out = vecadd_mod.vecadd_cuda(a, b)

    # No explicit synchronize: let the harness force sync when it needs the result
    return out
scrolls · 195 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 117131.

- #!POPCORN leaderboard vectoradd_v2
- #!POPCORN gpus L4
+ # !POPCORN leaderboard vectoradd_v2
+ # !POPCORN gpus H100
#
# #!/usr/bin/env python3
# import torch

Best evidence level for this revision: reported

JSON