submission 756978
ethan0027 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 97 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-prefixsum-v2-756978?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:b0063384279045ba06a701c0cb62542b9934d6e907f58762cecd60ab0719db6c
license declaredunknown
license concludedunknown
authorsethan0027
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 4
num_warps=4,stages = 2
num_stages=2,Kernel source
submission.py97 lines
# EVOLVE-BLOCK-START
import torch
import triton
import triton.language as tl
from typing import Tuple
@triton.jit
def _add_combine(a, b):
return a + b
@triton.jit
def block_sum_kernel(x_ptr, block_sum_ptr, n, BLOCK_SIZE: tl.constexpr):
pid = tl.program_id(0)
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
mask = offsets < n
x = tl.load(x_ptr + offsets, mask=mask, other=0.0)
block_sum = tl.sum(x) # sum across the block (zeros are harmless)
tl.store(block_sum_ptr + pid, block_sum)
@triton.jit
def scan_with_offset_kernel(x_ptr, out_ptr, block_prefix_ptr, n, BLOCK_SIZE: tl.constexpr):
pid = tl.program_id(0)
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
mask = offsets < n
x = tl.load(x_ptr + offsets, mask=mask, other=0.0)
# Inclusive scan within the block
out = tl.associative_scan(x, 0, _add_combine)
# Offset = sum of all previous blocks (0 for the first block)
offset = tl.load(block_prefix_ptr + pid - 1, mask=pid != 0, other=0.0)
out = out + offset
tl.store(out_ptr + offsets, out, mask=mask)
def custom_kernel(data: Tuple[torch.Tensor, torch.Tensor]) -> torch.Tensor:
"""
Inclusive prefix sum (scan) along dimension 0 using Triton.
Args:
data: Tuple (x, output) where both tensors are 1‑D float32 and contiguous on CUDA.
Returns:
The output tensor containing the inclusive prefix sum.
"""
x, output = data
# Basic validation
if not (x.is_cuda and output.is_cuda):
raise RuntimeError("Both input and output tensors must be CUDA tensors.")
if x.dtype != torch.float32 or output.dtype != torch.float32:
raise RuntimeError("Only float32 tensors are supported.")
if x.shape != output.shape:
raise RuntimeError("Input and output must have the same shape.")
if x.dim() != 1:
raise RuntimeError("Only 1‑D tensors are supported.")
n = x.numel()
if n == 0:
return output
# Ensure contiguous layout for pointer arithmetic
x = x.contiguous()
output = output.contiguous()
BLOCK_SIZE = 1024 # max threads per block on H200
num_blocks = (n + BLOCK_SIZE - 1) // BLOCK_SIZE
# Buffer for per‑block sums
block_sums = torch.empty(num_blocks, dtype=x.dtype, device=x.device)
# First pass: compute per‑block sums
block_sum_kernel[(num_blocks,)](
x,
block_sums,
n,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=4,
num_stages=2,
)
# Compute inclusive prefix of block sums on the GPU
block_prefix = torch.cumsum(block_sums, dim=0)
# Second pass: local scans with block offsets
scan_with_offset_kernel[(num_blocks,)](
x,
output,
block_prefix,
n,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=4,
num_stages=2,
)
return output
# EVOLVE-BLOCK-ENDscrolls · 97 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON