submission 763428
John · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 214 lines, June 9 Researcher Reciprocity License v1.0.
submission_v2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-763428?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:68feeb9e49942a8a838df8f667283a3386d5144727059698638aaeacd8d385d7
license declaredunknown
license concludedunknown
authorsJohn
imported2026-08-15
Kernel source
submission_v2.py214 lines
#!POPCORN leaderboard vectorsum_v2
import subprocess
import sys
for pkg in ["apache-tvm-ffi", "torch-c-dlpack-ext"]:
subprocess.check_call([sys.executable, "-m", "pip", "install", "-q", "--upgrade", pkg])
from collections.abc import Callable
from task import input_t, output_t
import cutlass
import cutlass.cute as cute
import cutlass.utils as utils
import torch
THREADS_PER_BLOCK = 256
WARPS_PER_BLOCK = THREADS_PER_BLOCK // 32
ITEMS_PER_THREAD = 8
BLOCK_ITEMS = THREADS_PER_BLOCK * ITEMS_PER_THREAD
MAX_BLOCKS_PER_SM = 8
_compiled_reduce_tiles = None
_compiled_final_reduce = None
_partial_buffer_cache: dict[int, torch.Tensor] = {}
@cute.kernel
def reduce_tiles_kernel(vector: cute.Tensor, partials: cute.Tensor):
tid_x, _, _ = cute.arch.thread_idx()
cta_x, _, _ = cute.arch.block_idx()
grid_x, _, _ = cute.arch.grid_dim()
lane = cute.arch.lane_idx()
warp = cute.arch.warp_idx()
n_elements = vector.shape[0]
grid_stride = grid_x * BLOCK_ITEMS
base_index = cta_x * BLOCK_ITEMS + tid_x
smem = utils.SmemAllocator()
warp_sums = smem.allocate_tensor(cutlass.Float32, WARPS_PER_BLOCK)
acc = cutlass.Float32(0.0)
index = base_index
while index < n_elements:
idx0 = index
idx1 = index + THREADS_PER_BLOCK
idx2 = index + 2 * THREADS_PER_BLOCK
idx3 = index + 3 * THREADS_PER_BLOCK
idx4 = index + 4 * THREADS_PER_BLOCK
idx5 = index + 5 * THREADS_PER_BLOCK
idx6 = index + 6 * THREADS_PER_BLOCK
idx7 = index + 7 * THREADS_PER_BLOCK
if idx0 < n_elements:
acc = acc + vector[idx0]
if idx1 < n_elements:
acc = acc + vector[idx1]
if idx2 < n_elements:
acc = acc + vector[idx2]
if idx3 < n_elements:
acc = acc + vector[idx3]
if idx4 < n_elements:
acc = acc + vector[idx4]
if idx5 < n_elements:
acc = acc + vector[idx5]
if idx6 < n_elements:
acc = acc + vector[idx6]
if idx7 < n_elements:
acc = acc + vector[idx7]
index = index + grid_stride
acc = cute.arch.warp_reduction_sum(acc)
if lane == 0:
warp_sums[warp] = acc
cute.arch.sync_threads()
if warp == 0:
block_sum = warp_sums[lane] if lane < WARPS_PER_BLOCK else cutlass.Float32(0.0)
block_sum = cute.arch.warp_reduction_sum(block_sum)
if lane == 0:
partials[cta_x] = block_sum
@cute.kernel
def final_reduce_kernel(partials: cute.Tensor, output: cute.Tensor):
tid_x, _, _ = cute.arch.thread_idx()
lane = cute.arch.lane_idx()
warp = cute.arch.warp_idx()
n_partials = partials.shape[0]
smem = utils.SmemAllocator()
warp_sums = smem.allocate_tensor(cutlass.Float32, WARPS_PER_BLOCK)
acc = cutlass.Float32(0.0)
index = tid_x
while index < n_partials:
acc = acc + partials[index]
index = index + THREADS_PER_BLOCK
acc = cute.arch.warp_reduction_sum(acc)
if lane == 0:
warp_sums[warp] = acc
cute.arch.sync_threads()
if warp == 0:
block_sum = warp_sums[lane] if lane < WARPS_PER_BLOCK else cutlass.Float32(0.0)
block_sum = cute.arch.warp_reduction_sum(block_sum)
if lane == 0:
output[0] = block_sum
@cute.jit
def launch_reduce_tiles(vector: cute.Tensor, partials: cute.Tensor):
reduce_tiles_kernel(vector, partials).launch(
grid=(partials.shape[0], 1, 1),
block=[THREADS_PER_BLOCK, 1, 1],
cluster=(1, 1, 1),
)
@cute.jit
def launch_final_reduce(partials: cute.Tensor, output: cute.Tensor):
final_reduce_kernel(partials, output).launch(
grid=(1, 1, 1),
block=[THREADS_PER_BLOCK, 1, 1],
cluster=(1, 1, 1),
)
def _ceil_div(x: int, y: int) -> int:
return (x + y - 1) // y
def _rounded_capacity(size: int) -> int:
capacity = 1
while capacity < size:
capacity *= 2
return capacity
def _compile_kernels() -> tuple[Callable, Callable]:
global _compiled_reduce_tiles, _compiled_final_reduce
if _compiled_reduce_tiles is not None and _compiled_final_reduce is not None:
return _compiled_reduce_tiles, _compiled_final_reduce
vector_len = cute.sym_int()
partial_len = cute.sym_int()
fake_vector = cute.runtime.make_fake_compact_tensor(cute.Float32, (vector_len,))
fake_partials = cute.runtime.make_fake_compact_tensor(cute.Float32, (partial_len,))
fake_output = cute.runtime.make_fake_compact_tensor(cute.Float32, (1,))
_compiled_reduce_tiles = cute.compile(
launch_reduce_tiles,
fake_vector,
fake_partials,
options="--enable-tvm-ffi",
)
_compiled_final_reduce = cute.compile(
launch_final_reduce,
fake_partials,
fake_output,
options="--enable-tvm-ffi",
)
return _compiled_reduce_tiles, _compiled_final_reduce
def _get_partial_buffer(device: torch.device, num_blocks: int) -> torch.Tensor:
device_index = device.index
if device_index is None:
device_index = torch.cuda.current_device()
partial_buffer = _partial_buffer_cache.get(device_index)
if partial_buffer is None or partial_buffer.numel() < num_blocks:
partial_buffer = torch.empty(
_rounded_capacity(num_blocks),
device=device,
dtype=torch.float32,
)
_partial_buffer_cache[device_index] = partial_buffer
return partial_buffer[:num_blocks]
def _num_blocks(n_elements: int, device: torch.device) -> int:
sm_count = torch.cuda.get_device_properties(device).multi_processor_count
return max(1, min(_ceil_div(n_elements, BLOCK_ITEMS), sm_count * MAX_BLOCKS_PER_SM))
def custom_kernel(data: input_t) -> output_t:
vector, output = data
n_elements = vector.numel()
if n_elements == 0:
output[0] = 0.0
return output[0]
num_blocks = _num_blocks(n_elements, vector.device)
partials = _get_partial_buffer(vector.device, num_blocks)
compiled_reduce_tiles, compiled_final_reduce = _compile_kernels()
compiled_reduce_tiles(vector, partials)
compiled_final_reduce(partials, output)
return output[0]
scrolls · 214 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON