Skip to content
KernelIndex
Search⌘K

submission 612406

dannywillowliu-uchi · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 46 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-prefixsum-v2-612406?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Inclusive prefix sumsuite of 11 cases
NVIDIA B200
482.1µs
#2 of 23
2026-03-23

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:1ee0a898d71d05d691077619df78e8379243747ab20cb61e6820de11af81a908
license declaredunknown
license concludedunknown
authorsdannywillowliu-uchi
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 4_reduce[(nb,)](inp, _bsums, n, BS=BS, num_warps=4)

Kernel source

submission.py46 lines
import os
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
import torch
import triton
import triton.language as tl
from task import input_t, output_t

BS = 1024

@triton.jit
def _reduce(inp_ptr, bsums_ptr, n, BS: tl.constexpr):
	pid = tl.program_id(0)
	offs = pid * BS + tl.arange(0, BS)
	m = offs < n
	x = tl.load(inp_ptr + offs, mask=m, other=0.0)
	tl.store(bsums_ptr + pid, tl.sum(x))

@triton.jit
def _scan_add(inp_ptr, out_ptr, ps_ptr, n, BS: tl.constexpr):
	pid = tl.program_id(0)
	offs = pid * BS + tl.arange(0, BS)
	m = offs < n
	x = tl.load(inp_ptr + offs, mask=m, other=0.0)
	s = tl.cumsum(x)
	if pid > 0:
		p = tl.load(ps_ptr + pid - 1)
	else:
		p = 0.0
	tl.store(out_ptr + offs, s + p, mask=m)

_bsums = None
_psums = None

def custom_kernel(data: input_t) -> output_t:
	global _bsums, _psums
	inp, out = data
	n = inp.numel()
	nb = (n + BS - 1) // BS
	if _bsums is None or _bsums.numel() < nb:
		_bsums = torch.empty(nb, device="cuda", dtype=torch.float32)
		_psums = torch.empty(nb, device="cuda", dtype=torch.float32)
	_reduce[(nb,)](inp, _bsums, n, BS=BS, num_warps=4)
	torch.cumsum(_bsums[:nb], dim=0, out=_psums[:nb])
	_scan_add[(nb,)](inp, out, _psums, n, BS=BS, num_warps=4)
	return out
scrolls · 46 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 612308.

⋯ 4 unchanged lines
import triton.language as tl
from task import input_t, output_t
- BLOCK_SIZE = 1024
+ BS = 1024
@triton.jit
- def _reduce_kernel(input_ptr, block_sums_ptr, n, BLOCK_SIZE: tl.constexpr):
+ def _reduce(inp_ptr, bsums_ptr, n, BS: tl.constexpr):
pid = tl.program_id(0)
- offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
- mask = offsets < n
- x = tl.load(input_ptr + offsets, mask=mask, other=0.0)
- block_sum = tl.sum(x)
- tl.store(block_sums_ptr + pid, block_sum)
+ offs = pid * BS + tl.arange(0, BS)
+ m = offs < n
+ x = tl.load(inp_ptr + offs, mask=m, other=0.0)
+ tl.store(bsums_ptr + pid, tl.sum(x))
@triton.jit
- def _scan_and_add_kernel(input_ptr, output_ptr, prefix_sums_ptr, n, BLOCK_SIZE: tl.constexpr):
+ def _scan_add(inp_ptr, out_ptr, ps_ptr, n, BS: tl.constexpr):
pid = tl.program_id(0)
- offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
- mask = offsets < n
- x = tl.load(input_ptr + offsets, mask=mask, other=0.0)
- scanned = tl.cumsum(x)
+ offs = pid * BS + tl.arange(0, BS)
+ m = offs < n
+ x = tl.load(inp_ptr + offs, mask=m, other=0.0)
+ s = tl.cumsum(x)
if pid > 0:
- prefix = tl.load(prefix_sums_ptr + pid - 1)
+ p = tl.load(ps_ptr + pid - 1)
else:
- prefix = 0.0
- scanned = scanned + prefix
- tl.store(output_ptr + offsets, scanned, mask=mask)
+ p = 0.0
+ tl.store(out_ptr + offs, s + p, mask=m)
- # Pre-allocate buffers for the target size
- _block_sums = None
- _prefix_sums = None
+ _bsums = None
+ _psums = None
def custom_kernel(data: input_t) -> output_t:
- global _block_sums, _prefix_sums
+ global _bsums, _psums
inp, out = data
n = inp.numel()
- nb = (n + BLOCK_SIZE - 1) // BLOCK_SIZE
-
- if _block_sums is None or _block_sums.numel() < nb:
- _block_sums = torch.empty(nb, device="cuda", dtype=torch.float32)
- _prefix_sums = torch.empty(nb, device="cuda", dtype=torch.float32)
-
- _reduce_kernel[(nb,)](inp, _block_sums, n, BLOCK_SIZE=BLOCK_SIZE, num_warps=4)
- torch.cumsum(_block_sums[:nb], dim=0, out=_prefix_sums[:nb])
- _scan_and_add_kernel[(nb,)](inp, out, _prefix_sums, n, BLOCK_SIZE=BLOCK_SIZE, num_warps=4)
+ nb = (n + BS - 1) // BS
+ if _bsums is None or _bsums.numel() < nb:
+ _bsums = torch.empty(nb, device="cuda", dtype=torch.float32)
+ _psums = torch.empty(nb, device="cuda", dtype=torch.float32)
+ _reduce[(nb,)](inp, _bsums, n, BS=BS, num_warps=4)
+ torch.cumsum(_bsums[:nb], dim=0, out=_psums[:nb])
+ _scan_add[(nb,)](inp, out, _psums, n, BS=BS, num_warps=4)
return out
scrolls · 71 diff lines total

Best evidence level for this revision: reported

JSON