submission 545121
rajesh0042 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 57 lines, June 9 Researcher Reciprocity License v1.0.
vectorsum_v2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-545121?include=source"interfacepython
Compatibility
measured onNVIDIA L4
declared hardwareNVIDIA L4
architecturessm_89
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:328f3648f8f9991ec6e7d97037f2ffbde3dfabc3f62284a3222708210896f280
license declaredunknown
license concludedunknown
authorsrajesh0042
imported2026-08-15
Kernel source
vectorsum_v2.py57 lines
import os
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
import torch
import triton
import triton.language as tl
from task import input_t, output_t
# Two-level reduction: GPU blocks reduce locally, then CPU sums partials
# Use float64 accumulation for accuracy matching reference
@triton.jit
def sum_reduce_kernel(
x_ptr, partial_ptr, n_elements,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
# Load as float32, accumulate in float64
x = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float64)
block_sum = tl.sum(x, axis=0)
tl.store(partial_ptr + pid, block_sum)
@triton.jit
def sum_final_kernel(
partial_ptr, out_ptr, n_partials,
BLOCK_SIZE: tl.constexpr,
):
offsets = tl.arange(0, BLOCK_SIZE)
mask = offsets < n_partials
x = tl.load(partial_ptr + offsets, mask=mask, other=0.0)
total = tl.sum(x, axis=0)
tl.store(out_ptr, total.to(tl.float32))
def custom_kernel(data: input_t) -> output_t:
data, output = data
n = data.numel()
BLOCK_SIZE = 4096
n_blocks = (n + BLOCK_SIZE - 1) // BLOCK_SIZE
partial = torch.empty(n_blocks, device=data.device, dtype=torch.float64)
sum_reduce_kernel[(n_blocks,)](data, partial, n, BLOCK_SIZE=BLOCK_SIZE)
# Second level reduction on GPU if too many partials
if n_blocks <= 65536:
# Use power-of-2 BLOCK_SIZE for final reduction
FINAL_BS = 1
while FINAL_BS < n_blocks:
FINAL_BS *= 2
if FINAL_BS > 65536:
FINAL_BS = 65536
out_scalar = torch.empty(1, device=data.device, dtype=torch.float32)
sum_final_kernel[(1,)](partial, out_scalar, n_blocks, BLOCK_SIZE=FINAL_BS)
return out_scalar[0]
else:
return partial.sum().to(torch.float32)
scrolls · 57 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 545079.
⋯ 1 unchanged linesos.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"import torch+ import triton+ import triton.language as tlfrom task import input_t, output_t+ # Two-level reduction: GPU blocks reduce locally, then CPU sums partials+ # Use float64 accumulation for accuracy matching reference++ @triton.jit+ def sum_reduce_kernel(+ x_ptr, partial_ptr, n_elements,+ BLOCK_SIZE: tl.constexpr,+ ):+ pid = tl.program_id(0)+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_elements+ # Load as float32, accumulate in float64+ x = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float64)+ block_sum = tl.sum(x, axis=0)+ tl.store(partial_ptr + pid, block_sum)++ @triton.jit+ def sum_final_kernel(+ partial_ptr, out_ptr, n_partials,+ BLOCK_SIZE: tl.constexpr,+ ):+ offsets = tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_partials+ x = tl.load(partial_ptr + offsets, mask=mask, other=0.0)+ total = tl.sum(x, axis=0)+ tl.store(out_ptr, total.to(tl.float32))+def custom_kernel(data: input_t) -> output_t:data, output = data- # Reference does: data.to(torch.float64).sum().to(torch.float32)- # and returns a scalar. We need to match that exactly.- result = data.to(torch.float64).sum().to(torch.float32)- return result+ n = data.numel()+ BLOCK_SIZE = 4096+ n_blocks = (n + BLOCK_SIZE - 1) // BLOCK_SIZE+ partial = torch.empty(n_blocks, device=data.device, dtype=torch.float64)+ sum_reduce_kernel[(n_blocks,)](data, partial, n, BLOCK_SIZE=BLOCK_SIZE)++ # Second level reduction on GPU if too many partials+ if n_blocks <= 65536:+ # Use power-of-2 BLOCK_SIZE for final reduction+ FINAL_BS = 1+ while FINAL_BS < n_blocks:+ FINAL_BS *= 2+ if FINAL_BS > 65536:+ FINAL_BS = 65536+ out_scalar = torch.empty(1, device=data.device, dtype=torch.float32)+ sum_final_kernel[(1,)](partial, out_scalar, n_blocks, BLOCK_SIZE=FINAL_BS)+ return out_scalar[0]+ else:+ return partial.sum().to(torch.float32)
scrolls · 60 diff lines total
Best evidence level for this revision: reported
JSON