submission 549651
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 58 lines, June 9 Researcher Reciprocity License v1.0.
gpumode_submit_85yh0ic8.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-549651?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:71a496924f1efa490ab069d5008b014e35b9389cd7ae5b35b6f508cd6ab1ecef
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Kernel source
gpumode_submit_85yh0ic8.py58 lines
import triton
import triton.language as tl
import torch
@triton.jit
def _vector_add_kernel(ptr_a, ptr_b, ptr_c, n_elements, BLOCK_SIZE: tl.constexpr):
"""Triton kernel for elementwise addition of two float16 tensors."""
pid = tl.program_id(0)
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
a = tl.load(ptr_a + offsets, mask=mask)
b = tl.load(ptr_b + offsets, mask=mask)
c = a + b
tl.store(ptr_c + offsets, c, mask=mask)
def kernel_function(A: torch.Tensor, B: torch.Tensor, C: torch.Tensor) -> torch.Tensor:
"""Wrapper for float16 vector addition: C = A + B.
Fused stages: single elementwise add kernel (load A, load B, add, store C).
No additional stages to fuse since this is a pure elementwise operation.
Args:
A: Input tensor of shape (N, N) and dtype float16.
B: Input tensor of shape (N, N) and dtype float16.
C: Output tensor of shape (N, N) and dtype float16 (written in-place).
Returns:
C tensor with result of A + B.
"""
assert A.is_cuda and B.is_cuda and C.is_cuda, "All tensors must be on CUDA"
assert A.shape == B.shape == C.shape, "Shape mismatch"
n_elements = A.numel()
BLOCK_SIZE = 1024
grid = (triton.cdiv(n_elements, BLOCK_SIZE),)
_vector_add_kernel[grid](A, B, C, n_elements, BLOCK_SIZE)
return C
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 58 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 490594.
⋯ 3 unchanged lines@triton.jit- def _vector_add_kernel(- ptr_a, # Pointer to input tensor A- ptr_b, # Pointer to input tensor B- ptr_out, # Pointer to output tensor C- n_elements, # Total number of elements- BLOCK_SIZE: tl.constexpr, # Block size for parallelization- ):- """- Triton kernel for element-wise addition of two tensors.-- Fused operation: C = A + B-- Each program instance processes BLOCK_SIZE elements.- """- # Get the program ID (which block we're processing)- pid = tl.program_id(axis=0)-- # Calculate the starting offset for this block+ def _vector_add_kernel(ptr_a, ptr_b, ptr_c, n_elements, BLOCK_SIZE: tl.constexpr):+ """Triton kernel for elementwise addition of two float16 tensors."""+ pid = tl.program_id(0)block_start = pid * BLOCK_SIZE-- # Generate offsets for elements within this blockoffsets = block_start + tl.arange(0, BLOCK_SIZE)-- # Create mask for boundary handling (handles non-power-of-2 sizes)mask = offsets < n_elements-- # Load elements from tensor A with masking- a = tl.load(ptr_a + offsets, mask=mask, other=0.0)-- # Load elements from tensor B with masking- b = tl.load(ptr_b + offsets, mask=mask, other=0.0)-- # Perform element-wise addition using Triton operations- result = a + b-- # Store the result to output tensor with masking- tl.store(ptr_out + offsets, result, mask=mask)+ a = tl.load(ptr_a + offsets, mask=mask)+ b = tl.load(ptr_b + offsets, mask=mask)+ c = a + b++ tl.store(ptr_c + offsets, c, mask=mask)++def kernel_function(A: torch.Tensor, B: torch.Tensor, C: torch.Tensor) -> torch.Tensor:- """- Wrapper function for float16 vector addition kernel.-- Performs element-wise addition: C = A + B-- This is a single fused operation - no decomposition needed as it's- already an atomic elementwise operation.-+ """Wrapper for float16 vector addition: C = A + B.++ Fused stages: single elementwise add kernel (load A, load B, add, store C).+ No additional stages to fuse since this is a pure elementwise operation.+Args:- A: Input tensor of shape (N, N) and dtype float16- B: Input tensor of shape (N, N) and dtype float16- C: Output tensor of shape (N, N) and dtype float16 (pre-allocated)-+ A: Input tensor of shape (N, N) and dtype float16.+ B: Input tensor of shape (N, N) and dtype float16.+ C: Output tensor of shape (N, N) and dtype float16 (written in-place).+Returns:- C: The output tensor containing A + B+ C tensor with result of A + B."""- # Validate inputsassert A.is_cuda and B.is_cuda and C.is_cuda, "All tensors must be on CUDA"- assert A.dtype == torch.float16 and B.dtype == torch.float16, "Inputs must be float16"- assert C.dtype == torch.float16, "Output must be float16"- assert A.shape == B.shape == C.shape, "All tensors must have the same shape"- assert A.is_contiguous() and B.is_contiguous() and C.is_contiguous(), "Tensors must be contiguous"-- # Calculate total number of elements+ assert A.shape == B.shape == C.shape, "Shape mismatch"+n_elements = A.numel()-- # Choose block size (power of 2 for efficiency)BLOCK_SIZE = 1024-- # Calculate grid dimensions (number of blocks needed)grid = (triton.cdiv(n_elements, BLOCK_SIZE),)-- # Launch the Triton kernel- _vector_add_kernel[grid](- A, # Input tensor A- B, # Input tensor B- C, # Output tensor C- n_elements, # Total elements to process- BLOCK_SIZE, # Block size (compile-time constant)- )-++ _vector_add_kernel[grid](A, B, C, n_elements, BLOCK_SIZE)+return Cimport inspect-def custom_kernel(input):sig = inspect.signature(kernel_function)num_params = len(sig.parameters)-if len(input) == num_params:return kernel_function(*input)return kernel_function(input)-- # Ensure deterministic cuBLAS.import osif os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"-
scrolls · 128 diff lines total
Best evidence level for this revision: reported
JSON