submission 512001
Clark Kitchen · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 103 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-512001?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:9ce96f0bbb8f528fcb20fc0f287239540531746e6b856bcfcdcbfaaf2c02250c
license declaredunknown
license concludedunknown
authorsClark Kitchen
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
DEFAULT_NUM_WARPS = 8stages = 2
num_stages=2,Kernel source
submission.py103 lines
import torch
try:
import triton
import triton.language as tl
_TRITON_AVAILABLE = True
except Exception:
triton = None
tl = None
_TRITON_AVAILABLE = False
from task import input_t, output_t
BLOCK_SIZE = 1024
DEFAULT_NUM_WARPS = 8
_PARTIAL_CACHE_A = {}
_PARTIAL_CACHE_B = {}
if _TRITON_AVAILABLE:
@triton.jit
def _partial_sum_kernel(
x_ptr,
partial_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(axis=0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
values = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float64)
partial = tl.sum(values, axis=0)
tl.store(partial_ptr + pid, partial)
def _get_partial_buffer(cache: dict, device: torch.device, num_partials: int) -> torch.Tensor:
key = (device.type, device.index)
cached = cache.get(key)
if cached is None or cached.numel() < num_partials:
alloc_size = 1 << (max(1, num_partials) - 1).bit_length()
cached = torch.empty((alloc_size,), device=device, dtype=torch.float64)
cache[key] = cached
return cached[:num_partials]
def _triton_reduce_sum(x: torch.Tensor) -> torch.Tensor:
current = x
n_elements = current.numel()
use_a = True
num_warps = DEFAULT_NUM_WARPS
if n_elements < (1 << 18):
num_warps = 4
while True:
num_blocks = triton.cdiv(n_elements, BLOCK_SIZE)
if use_a:
partials = _get_partial_buffer(_PARTIAL_CACHE_A, x.device, num_blocks)
else:
partials = _get_partial_buffer(_PARTIAL_CACHE_B, x.device, num_blocks)
_partial_sum_kernel[(num_blocks,)](
current,
partials,
n_elements,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=num_warps,
num_stages=2,
)
if num_blocks == 1:
return partials[0]
current = partials
n_elements = num_blocks
use_a = not use_a
num_warps = 4
def custom_kernel(data: input_t) -> output_t:
x, _output = data
if not x.is_contiguous():
x = x.contiguous()
if x.numel() == 0:
total_fp64 = torch.zeros((), device=x.device, dtype=torch.float64)
elif x.is_cuda and _TRITON_AVAILABLE:
try:
total_fp64 = _triton_reduce_sum(x)
except Exception:
total_fp64 = x.to(torch.float64).sum()
else:
total_fp64 = x.to(torch.float64).sum()
return total_fp64.to(torch.float32)
scrolls · 103 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 511997.
⋯ 84 unchanged linesdef custom_kernel(data: input_t) -> output_t:- x, output = data+ x, _output = dataif not x.is_contiguous():x = x.contiguous()⋯ 7 unchanged lineselse:total_fp64 = x.to(torch.float64).sum()- out_scalar = output.view(())- out_scalar.copy_(total_fp64)- return out_scalar+ return total_fp64.to(torch.float32)
Best evidence level for this revision: reported
JSON