submission 66233
dixit122 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 96 lines, June 9 Researcher Reciprocity License v1.0.
vectorAdd.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-66233?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:cf37580b60aade9972ba14ce865c6ce59dbe2eb265f93afcd752ff418f0b2466
license declaredunknown
license concludedunknown
authorsdixit122
imported2026-08-15
Kernel source
vectorAdd.py96 lines
from utils import make_match_reference, DeterministicContext
import torch
from task import input_t, output_t
from torch.utils.cpp_extension import load_inline
hip_source = """
#include <cuda_runtime.h>
#include <iostream>
#include <pybind11/buffer_info.h>
#include <pybind11/pybind11.h>
#include <stdexcept>
#include <torch/extension.h>
#include <cuda_fp16.h>
namespace py = pybind11;
__global__ void add_kernel(const __half *__restrict a, const __half *__restrict b,
__half *__restrict c, int N) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < N) {
c[i] = __hadd(a[i],b[i]);
}
}
void add_vectors(uintptr_t A, uintptr_t B, uintptr_t C, int N) {
__half *A_ptr = reinterpret_cast<__half *>(A);
__half *B_ptr = reinterpret_cast<__half *>(B);
__half *C_ptr = reinterpret_cast<__half *>(C);
int tpb = 256;
dim3 threadPerBlock(tpb);
dim3 numBlocks((N + tpb - 1) / tpb);
add_kernel<<<numBlocks, threadPerBlock>>>(A_ptr, B_ptr, C_ptr, N);
cudaDeviceSynchronize();
}
"""
# --- C++ Function Declaration (Header) ---
# This string provides the function prototype to the C++ compiler (for the bindings).
cpp_source_declaration = """
#include <torch/extension.h>
#include <pybind11/pybind11.h>
namespace py = pybind11;
void add_vectors(uintptr_t A, uintptr_t B, uintptr_t C, int N);
"""
# # Define paths and flags matching the setup
# extra_include_paths = [
# "-I/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/cuda/include/",
# "-I/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/comm_libs/nccl/include/",
# ]
# extra_library_paths = [
# "-L/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/cuda/lib64/",
# "-L/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/comm_libs/12.9/nccl/lib/",
# "-L/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/math_libs/12.9/lib64/",
# "-lnccl",
# "-lcublas",
# "-lcudart",
# ]
extra_cflags = ["-O2"]
extra_cuda_cflags = ["-O2"]
# Load the inline extension
add_kernel = load_inline(
name="add_kernel",
verbose=True,
cpp_sources=[cpp_source_declaration],
cuda_sources=[hip_source],
functions=[
"add_vectors",
],
extra_cflags=extra_cflags,
extra_cuda_cflags=extra_cuda_cflags,
# extra_include_paths=extra_include_paths,
# extra_ldflags=extra_library_paths,
)
def custom_kernel(data: input_t) -> output_t:
"""
Reference implementation of vector addition using PyTorch.
Args:
data: Tuple of tensors [A, B] to be added.
Returns:
Tensor containing element-wise sums.
"""
with DeterministicContext():
A, B, output = data
N = A.shape[0] ** 2
add_kernel.add_vectors(A.data_ptr(), B.data_ptr(), output.data_ptr(), N)
return output
scrolls · 96 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON