submission 68270
unknown_register · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 87 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-68270?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:4b630af562f462cf2e683c83b697b1770ba716e9c8ed878a17b8d87fbdca2cc2
license declaredunknown
license concludedunknown
authorsunknown_register
imported2026-08-15
Kernel source
submission.py87 lines
import torch
import triton
import triton.language as tl
@triton.jit
def vector_add_kernel(
input_a_ptr, input_b_ptr, output_ptr,
n_elements: tl.constexpr,
BLOCK_SIZE: tl.constexpr,
):
"""
Triton kernel for element-wise vector addition.
Args:
input_a_ptr: Pointer to first input tensor
input_b_ptr: Pointer to second input tensor
output_ptr: Pointer to output tensor
n_elements: Number of elements to add
BLOCK_SIZE: Number of elements per block
"""
# Get program ID (block index)
pid = tl.program_id(axis=0)
# Block start and end offsets
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
# Handle out-of-bounds
mask = offsets < n_elements
# Load input tensors with masking
a = tl.load(input_a_ptr + offsets, mask=mask)
b = tl.load(input_b_ptr + offsets, mask=mask)
# Perform addition
output = a + b
# Store result
tl.store(output_ptr + offsets, output, mask=mask)
def custom_kernel(input: tuple[torch.Tensor, torch.Tensor]) -> torch.Tensor:
"""
Entry point for vector addition. This is what popcorn-cli will call.
Renamed to match exact leaderboard expectation.
Args:
input: Tuple of two input tensors (a, b)
Returns:
Output tensor with element-wise sum
"""
# Defensive unpacking with debugging
print(f"Input type: {type(input)}") # For submission logs
print(f"Input length: {len(input) if hasattr(input, '__len__') else 'N/A'}")
try:
a, b, output = input # Attempt unpack
print(a.shape)
print(b.shape)
print(output.shape)
except ValueError as e:
# Fallback: if input is not a 3-tuple, raise detailed error
raise ValueError(f"Unpack failed: {e}. Input content: {input}, Type: {type(input)}, Length: {len(input) if hasattr(input, '__len__') else 'N/A'}") from e
# Additional validation (shapes must match, on GPU)
assert isinstance(a, torch.Tensor) and isinstance(b, torch.Tensor), f"Tensors expected, got {type(a)}, {type(b)}"
assert a.shape == b.shape, f"Input shapes mismatch: {a.shape} vs {b.shape}"
assert a.device.type == 'cuda' and b.device.type == 'cuda', "Tensors must be on CUDA device"
n = a.numel()
# output = torch.empty_like(a)
# Grid size (number of blocks)
grid = lambda meta: (triton.cdiv(n, meta['BLOCK_SIZE']),)
# Launch kernel
vector_add_kernel[grid](
a, b, output,
n,
BLOCK_SIZE=1024, # Tune based on hardware; 1024 is a good starting point
)
return output
scrolls · 87 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 68263.
Best evidence level for this revision: reported
JSON