Skip to content
KernelIndex
Search⌘K

submission 66233

dixit122 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 96 lines, June 9 Researcher Reciprocity License v1.0.

vectorAdd.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-66233?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA A100
1.24ms
#60 of 87
2025-10-20

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:cf37580b60aade9972ba14ce865c6ce59dbe2eb265f93afcd752ff418f0b2466
license declaredunknown
license concludedunknown
authorsdixit122
imported2026-08-15

Kernel source

vectorAdd.py96 lines
from utils import make_match_reference, DeterministicContext
import torch
from task import input_t, output_t
from torch.utils.cpp_extension import load_inline

hip_source = """
#include <cuda_runtime.h>
#include <iostream>
#include <pybind11/buffer_info.h>
#include <pybind11/pybind11.h>
#include <stdexcept>
#include <torch/extension.h>
#include <cuda_fp16.h>
namespace py = pybind11;

__global__ void add_kernel(const __half *__restrict a, const __half *__restrict b,
                           __half *__restrict c, int N) {
  int i = blockIdx.x * blockDim.x + threadIdx.x;
  if (i < N) {
    c[i] = __hadd(a[i],b[i]);
  }
}

void add_vectors(uintptr_t A, uintptr_t B, uintptr_t C, int N) {

  __half *A_ptr = reinterpret_cast<__half *>(A);
  __half *B_ptr = reinterpret_cast<__half *>(B);
  __half *C_ptr = reinterpret_cast<__half *>(C);

  int tpb = 256;
  dim3 threadPerBlock(tpb);
  dim3 numBlocks((N + tpb - 1) / tpb);
  add_kernel<<<numBlocks, threadPerBlock>>>(A_ptr, B_ptr, C_ptr, N);
  cudaDeviceSynchronize();
}

"""


# --- C++ Function Declaration (Header) ---
# This string provides the function prototype to the C++ compiler (for the bindings).
cpp_source_declaration = """
#include <torch/extension.h>
#include <pybind11/pybind11.h>

namespace py = pybind11;
void add_vectors(uintptr_t A, uintptr_t B, uintptr_t C, int N);
"""

# # Define paths and flags matching the setup
# extra_include_paths = [
#     "-I/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/cuda/include/",
#     "-I/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/comm_libs/nccl/include/",
# ]
# extra_library_paths = [
#     "-L/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/cuda/lib64/",
#     "-L/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/comm_libs/12.9/nccl/lib/",
#     "-L/opt/nvidia/hpc_sdk/Linux_x86_64/25.7/math_libs/12.9/lib64/",
#     "-lnccl",
#     "-lcublas",
#     "-lcudart",
# ]
extra_cflags = ["-O2"]
extra_cuda_cflags = ["-O2"]

# Load the inline extension
add_kernel = load_inline(
    name="add_kernel",
    verbose=True,
    cpp_sources=[cpp_source_declaration],
    cuda_sources=[hip_source],
    functions=[
        "add_vectors",
    ],
    extra_cflags=extra_cflags,
    extra_cuda_cflags=extra_cuda_cflags,
    # extra_include_paths=extra_include_paths,
    # extra_ldflags=extra_library_paths,
)


def custom_kernel(data: input_t) -> output_t:
    """
    Reference implementation of vector addition using PyTorch.
    Args:
        data: Tuple of tensors [A, B] to be added.
    Returns:
        Tensor containing element-wise sums.
    """

    with DeterministicContext():
        A, B, output = data
        N = A.shape[0] ** 2
        add_kernel.add_vectors(A.data_ptr(), B.data_ptr(), output.data_ptr(), N)
        return output
scrolls · 96 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON