submission 511997
Clark Kitchen · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 105 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-511997?include=source"interfacepython
Compatibility
measured onNVIDIA L4
declared hardwareNVIDIA L4
architecturessm_89
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:af0e10e5af9cd3120d2ebc13a5ee4668e6a6566acf0f945cda827a43b3b55fae
license declaredunknown
license concludedunknown
authorsClark Kitchen
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
DEFAULT_NUM_WARPS = 8stages = 2
num_stages=2,Kernel source
submission.py105 lines
import torch
try:
import triton
import triton.language as tl
_TRITON_AVAILABLE = True
except Exception:
triton = None
tl = None
_TRITON_AVAILABLE = False
from task import input_t, output_t
BLOCK_SIZE = 1024
DEFAULT_NUM_WARPS = 8
_PARTIAL_CACHE_A = {}
_PARTIAL_CACHE_B = {}
if _TRITON_AVAILABLE:
@triton.jit
def _partial_sum_kernel(
x_ptr,
partial_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(axis=0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
values = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float64)
partial = tl.sum(values, axis=0)
tl.store(partial_ptr + pid, partial)
def _get_partial_buffer(cache: dict, device: torch.device, num_partials: int) -> torch.Tensor:
key = (device.type, device.index)
cached = cache.get(key)
if cached is None or cached.numel() < num_partials:
alloc_size = 1 << (max(1, num_partials) - 1).bit_length()
cached = torch.empty((alloc_size,), device=device, dtype=torch.float64)
cache[key] = cached
return cached[:num_partials]
def _triton_reduce_sum(x: torch.Tensor) -> torch.Tensor:
current = x
n_elements = current.numel()
use_a = True
num_warps = DEFAULT_NUM_WARPS
if n_elements < (1 << 18):
num_warps = 4
while True:
num_blocks = triton.cdiv(n_elements, BLOCK_SIZE)
if use_a:
partials = _get_partial_buffer(_PARTIAL_CACHE_A, x.device, num_blocks)
else:
partials = _get_partial_buffer(_PARTIAL_CACHE_B, x.device, num_blocks)
_partial_sum_kernel[(num_blocks,)](
current,
partials,
n_elements,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=num_warps,
num_stages=2,
)
if num_blocks == 1:
return partials[0]
current = partials
n_elements = num_blocks
use_a = not use_a
num_warps = 4
def custom_kernel(data: input_t) -> output_t:
x, output = data
if not x.is_contiguous():
x = x.contiguous()
if x.numel() == 0:
total_fp64 = torch.zeros((), device=x.device, dtype=torch.float64)
elif x.is_cuda and _TRITON_AVAILABLE:
try:
total_fp64 = _triton_reduce_sum(x)
except Exception:
total_fp64 = x.to(torch.float64).sum()
else:
total_fp64 = x.to(torch.float64).sum()
out_scalar = output.view(())
out_scalar.copy_(total_fp64)
return out_scalar
scrolls · 105 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON