submission 636870
Nitish Naineni · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 67 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-636870?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:2fe0909856e9ae462ec58580593ef4383e5657cb83358b8389574bcbb1215f09
license declaredunknown
license concludedunknown
authorsNitish Naineni
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
vector-width = float4
__global__ void vec_add(const float4* A, const float4* B, float4* out, int N) {Kernel source
submission.py67 lines
#!POPCORN leaderboard vectoradd_v2
#!POPCORN gpu A100
import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t
CUDA_SRC = """#include <cuda_fp16.h>
#include <torch/extension.h>
__global__ void vec_add(const float4* A, const float4* B, float4* out, int N) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if (idx < N / 8) {
float4 a_val = A[idx];
float4 b_val = B[idx];
float4 out_val;
const half2* a = reinterpret_cast<const half2*>(&a_val);
const half2* b = reinterpret_cast<const half2*>(&b_val);
half2* o = reinterpret_cast<half2*>(&out_val);
for (int i{}; i < 4 ; i++){
o[i] = __hadd2(a[i], b[i]);
}
out[idx] = out_val;
} else if (idx == N / 8) {
const half* a = reinterpret_cast<const half*>(A);
const half* b = reinterpret_cast<const half*>(B);
half* o = reinterpret_cast<half*>(out);
for (int i{(N / 8) * 8}; i < N ; i++){
o[i] = __hadd(a[i], b[i]);
}
}
}
torch::Tensor& vecadd(const torch::Tensor& A, const torch::Tensor& B, torch::Tensor& out) {
int N = A.numel();
int threads = 256;
int blocks = (N / 8 + threads) / threads;
vec_add<<<blocks, threads>>>(
reinterpret_cast<const float4*>(A.data_ptr<at::Half>()),
reinterpret_cast<const float4*>(B.data_ptr<at::Half>()),
reinterpret_cast<float4*>(out.data_ptr<at::Half>()),
N
);
return out;
}
"""
CPP_SRC = """// Your C++ function declarations go here
torch::Tensor& vecadd(const torch::Tensor& A, const torch::Tensor& B, torch::Tensor& out);
"""
module = load_inline(
name='vecadd_module',
cpp_sources=[CPP_SRC],
cuda_sources=[CUDA_SRC],
functions=['vecadd'],
verbose=True,
)
def custom_kernel(data: input_t) -> output_t:
A, B, output = data
return module.vecadd(A, B, output)
scrolls · 67 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON