submission 187025
literid · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 83 lines, June 9 Researcher Reciprocity License v1.0.
triton_my.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-187025?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:6567d169c3fcc07b6d2778500462c3b9431cec0396a040ed79cfbdf222dbba59
license declaredunknown
license concludedunknown
authorsliterid
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
num_warps=8,Kernel source
triton_my.py83 lines
import functools
import torch
import triton
import triton.language as tl
from task import input_t, output_t
@triton.jit
def _partial_sum_kernel(
x_ptr,
partial_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
x = tl.load(x_ptr + offsets, mask=mask, other=0.0)
s = tl.sum(x, axis=0)
tl.store(partial_ptr + pid, s)
@triton.jit
def _final_sum_kernel(
partial_ptr,
out_ptr,
n_partials,
BLOCK_SIZE: tl.constexpr,
):
offsets = tl.arange(0, BLOCK_SIZE)
mask = offsets < n_partials
x = tl.load(partial_ptr + offsets, mask=mask, other=0.0)
s = tl.sum(x, axis=0)
tl.store(out_ptr, s) # out_ptr is a pointer to output[0]
@functools.lru_cache(maxsize=64)
def _get_partial_buffer(num_blocks: int, device: torch.device) -> torch.Tensor:
# Cache to avoid per-call allocations during benchmarking repeats.
return torch.empty((num_blocks,), device=device, dtype=torch.float32)
def _custom_kernel(data: input_t) -> output_t:
x, out = data
n = x.numel()
# Tuned for large 1D reductions (H100): fewer blocks -> fewer partials.
BLOCK1 = 8192
grid = (triton.cdiv(n, BLOCK1),)
partial = _get_partial_buffer(grid[0], x.device)
# Pass 1: compute block partials (no atomics).
_partial_sum_kernel[grid](
x,
partial,
n,
BLOCK_SIZE=BLOCK1,
num_warps=8,
)
# Pass 2: reduce partials into the provided output buffer.
# grid[0] is <= ~6400 for the biggest benchmark when BLOCK1=8192, so 8192 covers it.
BLOCK2 = 8192
_final_sum_kernel[(1,)](
partial,
out,
grid[0],
BLOCK_SIZE=BLOCK2,
num_warps=8,
)
# Reference expects a 0-D scalar tensor (same as `data.sum()`), not a (1,) buffer.
return out[0]
custom_kernel = _custom_kernel
scrolls · 83 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 186878.
+ import functools+import torch+ import triton+ import triton.language as tl+from task import input_t, output_t+ @triton.jit+ def _partial_sum_kernel(+ x_ptr,+ partial_ptr,+ n_elements,+ BLOCK_SIZE: tl.constexpr,+ ):+ pid = tl.program_id(0)+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_elements+ x = tl.load(x_ptr + offsets, mask=mask, other=0.0)+ s = tl.sum(x, axis=0)+ tl.store(partial_ptr + pid, s)+++ @triton.jit+ def _final_sum_kernel(+ partial_ptr,+ out_ptr,+ n_partials,+ BLOCK_SIZE: tl.constexpr,+ ):+ offsets = tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_partials+ x = tl.load(partial_ptr + offsets, mask=mask, other=0.0)+ s = tl.sum(x, axis=0)+ tl.store(out_ptr, s) # out_ptr is a pointer to output[0]+++ @functools.lru_cache(maxsize=64)+ def _get_partial_buffer(num_blocks: int, device: torch.device) -> torch.Tensor:+ # Cache to avoid per-call allocations during benchmarking repeats.+ return torch.empty((num_blocks,), device=device, dtype=torch.float32)++def _custom_kernel(data: input_t) -> output_t:- data, output = data- output = data.sum()- return output+ x, out = data+ n = x.numel()+ # Tuned for large 1D reductions (H100): fewer blocks -> fewer partials.+ BLOCK1 = 8192+ grid = (triton.cdiv(n, BLOCK1),)+ partial = _get_partial_buffer(grid[0], x.device)- # Compile the kernel for better performance+ # Pass 1: compute block partials (no atomics).+ _partial_sum_kernel[grid](+ x,+ partial,+ n,+ BLOCK_SIZE=BLOCK1,+ num_warps=8,+ )++ # Pass 2: reduce partials into the provided output buffer.+ # grid[0] is <= ~6400 for the biggest benchmark when BLOCK1=8192, so 8192 covers it.+ BLOCK2 = 8192+ _final_sum_kernel[(1,)](+ partial,+ out,+ grid[0],+ BLOCK_SIZE=BLOCK2,+ num_warps=8,+ )++ # Reference expects a 0-D scalar tensor (same as `data.sum()`), not a (1,) buffer.+ return out[0]++custom_kernel = _custom_kernel++
scrolls · 86 diff lines total
Best evidence level for this revision: reported
JSON