submission 642951
Nitish Naineni · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 68 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-642951?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:8e4683ae5a34d39667acba2758ec625fcc96cf726519167e47e13916c266453d
license declaredunknown
license concludedunknown
authorsNitish Naineni
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
vector-width = float4
__global__ void vec_add(const float4* A, const float4* B, float4* out, int N) {Kernel source
submission.py68 lines
#!POPCORN leaderboard vectoradd_v2
#!POPCORN gpu B200
import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t
CUDA_SRC = """#include <cuda_fp16.h>
#include <torch/extension.h>
__global__ void vec_add(const float4* A, const float4* B, float4* out, int N) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if (idx < N / 8) {
float4 a_val = A[idx];
float4 b_val = B[idx];
float4 out_val;
const half2* a = reinterpret_cast<const half2*>(&a_val);
const half2* b = reinterpret_cast<const half2*>(&b_val);
half2* o = reinterpret_cast<half2*>(&out_val);
for (int i{}; i < 4 ; i++){
o[i] = __hadd2(a[i], b[i]);
}
out[idx] = out_val;
} else if (idx == N / 8) {
const half* a = reinterpret_cast<const half*>(A);
const half* b = reinterpret_cast<const half*>(B);
half* o = reinterpret_cast<half*>(out);
for (int i{(N / 8) * 8}; i < N ; i++){
o[i] = __hadd(a[i], b[i]);
}
}
}
torch::Tensor& vecadd(const torch::Tensor& A, const torch::Tensor& B, torch::Tensor& out) {
int N = A.numel();
int threads = 256;
int blocks = (N / 8 + threads) / threads;
vec_add<<<blocks, threads>>>(
reinterpret_cast<const float4*>(A.data_ptr<at::Half>()),
reinterpret_cast<const float4*>(B.data_ptr<at::Half>()),
reinterpret_cast<float4*>(out.data_ptr<at::Half>()),
N
);
return out;
}
"""
CPP_SRC = """// Your C++ function declarations go here
torch::Tensor& vecadd(const torch::Tensor& A, const torch::Tensor& B, torch::Tensor& out);
"""
module = load_inline(
name='vecadd_module',
cpp_sources=[CPP_SRC],
cuda_sources=[CUDA_SRC],
functions=['vecadd'],
verbose=True,
extra_cuda_cflags=['-arch=sm_100', '--use_fast_math'],
)
def custom_kernel(data: input_t) -> output_t:
A, B, output = data
return module.vecadd(A, B, output)
scrolls · 68 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 636870.
#!POPCORN leaderboard vectoradd_v2- #!POPCORN gpu A100+ #!POPCORN gpu B200import torchfrom torch.utils.cpp_extension import load_inline⋯ 53 unchanged linescuda_sources=[CUDA_SRC],functions=['vecadd'],verbose=True,+ extra_cuda_cflags=['-arch=sm_100', '--use_fast_math'],)def custom_kernel(data: input_t) -> output_t:
Best evidence level for this revision: reported
JSON