submission 549854
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 99 lines, June 9 Researcher Reciprocity License v1.0.
gpumode_submit_290vic5t.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-549854?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:bd52da8b128c148a260e8e289dbd75740b6122a7965531d21c49c48766cc4c56
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
num_warps=8,Kernel source
gpumode_submit_290vic5t.py99 lines
# kernel.py
"""
Triton vector-add kernel (float16) for 2D tensors.
Fused stages (single pass):
1) tl.load(A) + tl.load(B)
2) elementwise add in-kernel
3) tl.store(C)
No unfused fallback is needed because the whole pipeline is a single elementwise op.
"""
import torch
import triton
import triton.language as tl
@triton.jit
def _vec_add_kernel(a_ptr, b_ptr, c_ptr, n_elements,
BLOCK_SIZE: tl.constexpr):
pid = tl.program_id(axis=0)
offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offs < n_elements
a = tl.load(a_ptr + offs, mask=mask, other=0.0)
b = tl.load(b_ptr + offs, mask=mask, other=0.0)
c = a + b
tl.store(c_ptr + offs, c, mask=mask)
def kernel_function(*args):
"""
Wrapper that validates inputs, allocates output if needed, and launches Triton.
Supported call patterns (as used by the tests):
- kernel_function(A, B) -> out
- kernel_function(A, B, C) -> out (writes into C)
- kernel_function((A, B)) -> out
- kernel_function((A, B, C)) -> out (writes into C)
"""
# Unpack tuple-style inputs
if len(args) == 1 and isinstance(args[0], (tuple, list)):
args = tuple(args[0])
if len(args) not in (2, 3):
raise TypeError(f"kernel_function expected 2 or 3 arguments, got {len(args)}")
A, B = args[0], args[1]
C = args[2] if len(args) == 3 else None
if not (isinstance(A, torch.Tensor) and isinstance(B, torch.Tensor)):
raise TypeError("A and B must be torch.Tensor")
if A.device.type != "cuda" or B.device.type != "cuda":
raise ValueError("A and B must be CUDA tensors")
if A.dtype != torch.float16 or B.dtype != torch.float16:
raise ValueError("A and B must be torch.float16")
if A.shape != B.shape:
raise ValueError(f"Shape mismatch: A.shape={tuple(A.shape)} vs B.shape={tuple(B.shape)}")
if not A.is_contiguous() or not B.is_contiguous():
raise ValueError("A and B must be contiguous")
if C is None:
C = torch.empty_like(A)
else:
if not isinstance(C, torch.Tensor):
raise TypeError("C must be a torch.Tensor when provided")
if C.device != A.device:
raise ValueError("C must be on the same device as A")
if C.dtype != torch.float16:
raise ValueError("C must be torch.float16")
if C.shape != A.shape:
raise ValueError("C must have the same shape as A")
if not C.is_contiguous():
raise ValueError("C must be contiguous")
n_elements = A.numel()
BLOCK_SIZE = 1024
grid = (triton.cdiv(n_elements, BLOCK_SIZE),)
_vec_add_kernel[grid](
A, B, C,
n_elements,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=8,
)
return C
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 99 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 549651.
+ # kernel.py+ """+ Triton vector-add kernel (float16) for 2D tensors.++ Fused stages (single pass):+ 1) tl.load(A) + tl.load(B)+ 2) elementwise add in-kernel+ 3) tl.store(C)++ No unfused fallback is needed because the whole pipeline is a single elementwise op.+ """++ import torchimport tritonimport triton.language as tl- import torch@triton.jit- def _vector_add_kernel(ptr_a, ptr_b, ptr_c, n_elements, BLOCK_SIZE: tl.constexpr):- """Triton kernel for elementwise addition of two float16 tensors."""- pid = tl.program_id(0)- block_start = pid * BLOCK_SIZE- offsets = block_start + tl.arange(0, BLOCK_SIZE)- mask = offsets < n_elements+ def _vec_add_kernel(a_ptr, b_ptr, c_ptr, n_elements,+ BLOCK_SIZE: tl.constexpr):+ pid = tl.program_id(axis=0)+ offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ mask = offs < n_elements- a = tl.load(ptr_a + offsets, mask=mask)- b = tl.load(ptr_b + offsets, mask=mask)-+ a = tl.load(a_ptr + offs, mask=mask, other=0.0)+ b = tl.load(b_ptr + offs, mask=mask, other=0.0)c = a + b+ tl.store(c_ptr + offs, c, mask=mask)- tl.store(ptr_c + offsets, c, mask=mask)+ def kernel_function(*args):+ """+ Wrapper that validates inputs, allocates output if needed, and launches Triton.- def kernel_function(A: torch.Tensor, B: torch.Tensor, C: torch.Tensor) -> torch.Tensor:- """Wrapper for float16 vector addition: C = A + B.+ Supported call patterns (as used by the tests):+ - kernel_function(A, B) -> out+ - kernel_function(A, B, C) -> out (writes into C)+ - kernel_function((A, B)) -> out+ - kernel_function((A, B, C)) -> out (writes into C)+ """+ # Unpack tuple-style inputs+ if len(args) == 1 and isinstance(args[0], (tuple, list)):+ args = tuple(args[0])- Fused stages: single elementwise add kernel (load A, load B, add, store C).- No additional stages to fuse since this is a pure elementwise operation.+ if len(args) not in (2, 3):+ raise TypeError(f"kernel_function expected 2 or 3 arguments, got {len(args)}")- Args:- A: Input tensor of shape (N, N) and dtype float16.- B: Input tensor of shape (N, N) and dtype float16.- C: Output tensor of shape (N, N) and dtype float16 (written in-place).+ A, B = args[0], args[1]+ C = args[2] if len(args) == 3 else None- Returns:- C tensor with result of A + B.- """- assert A.is_cuda and B.is_cuda and C.is_cuda, "All tensors must be on CUDA"- assert A.shape == B.shape == C.shape, "Shape mismatch"+ if not (isinstance(A, torch.Tensor) and isinstance(B, torch.Tensor)):+ raise TypeError("A and B must be torch.Tensor")+ if A.device.type != "cuda" or B.device.type != "cuda":+ raise ValueError("A and B must be CUDA tensors")+ if A.dtype != torch.float16 or B.dtype != torch.float16:+ raise ValueError("A and B must be torch.float16")+ if A.shape != B.shape:+ raise ValueError(f"Shape mismatch: A.shape={tuple(A.shape)} vs B.shape={tuple(B.shape)}")+ if not A.is_contiguous() or not B.is_contiguous():+ raise ValueError("A and B must be contiguous")+ if C is None:+ C = torch.empty_like(A)+ else:+ if not isinstance(C, torch.Tensor):+ raise TypeError("C must be a torch.Tensor when provided")+ if C.device != A.device:+ raise ValueError("C must be on the same device as A")+ if C.dtype != torch.float16:+ raise ValueError("C must be torch.float16")+ if C.shape != A.shape:+ raise ValueError("C must have the same shape as A")+ if not C.is_contiguous():+ raise ValueError("C must be contiguous")+n_elements = A.numel()BLOCK_SIZE = 1024grid = (triton.cdiv(n_elements, BLOCK_SIZE),)- _vector_add_kernel[grid](A, B, C, n_elements, BLOCK_SIZE)-+ _vec_add_kernel[grid](+ A, B, C,+ n_elements,+ BLOCK_SIZE=BLOCK_SIZE,+ num_warps=8,+ )return Cimport inspect
scrolls · 114 diff lines total
Best evidence level for this revision: reported
JSON