Skip to content
KernelIndex
Search⌘K

submission 763428

John · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 214 lines, June 9 Researcher Reciprocity License v1.0.

submission_v2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-763428?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA B200
53.1µs
#39 of 88
2026-04-12

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:68feeb9e49942a8a838df8f667283a3386d5144727059698638aaeacd8d385d7
license declaredunknown
license concludedunknown
authorsJohn
imported2026-08-15

Kernel source

submission_v2.py214 lines
#!POPCORN leaderboard vectorsum_v2


import subprocess
import sys

for pkg in ["apache-tvm-ffi", "torch-c-dlpack-ext"]:
    subprocess.check_call([sys.executable, "-m", "pip", "install", "-q", "--upgrade", pkg])

from collections.abc import Callable

from task import input_t, output_t

import cutlass
import cutlass.cute as cute
import cutlass.utils as utils
import torch

THREADS_PER_BLOCK = 256
WARPS_PER_BLOCK = THREADS_PER_BLOCK // 32
ITEMS_PER_THREAD = 8
BLOCK_ITEMS = THREADS_PER_BLOCK * ITEMS_PER_THREAD
MAX_BLOCKS_PER_SM = 8

_compiled_reduce_tiles = None
_compiled_final_reduce = None
_partial_buffer_cache: dict[int, torch.Tensor] = {}



@cute.kernel
def reduce_tiles_kernel(vector: cute.Tensor, partials: cute.Tensor):
    tid_x, _, _ = cute.arch.thread_idx()
    cta_x, _, _ = cute.arch.block_idx()
    grid_x, _, _ = cute.arch.grid_dim()
    lane = cute.arch.lane_idx()
    warp = cute.arch.warp_idx()

    n_elements = vector.shape[0]
    grid_stride = grid_x * BLOCK_ITEMS
    base_index = cta_x * BLOCK_ITEMS + tid_x

    smem = utils.SmemAllocator()
    warp_sums = smem.allocate_tensor(cutlass.Float32, WARPS_PER_BLOCK)

    acc = cutlass.Float32(0.0)
    index = base_index
    while index < n_elements:
        idx0 = index
        idx1 = index + THREADS_PER_BLOCK
        idx2 = index + 2 * THREADS_PER_BLOCK
        idx3 = index + 3 * THREADS_PER_BLOCK
        idx4 = index + 4 * THREADS_PER_BLOCK
        idx5 = index + 5 * THREADS_PER_BLOCK
        idx6 = index + 6 * THREADS_PER_BLOCK
        idx7 = index + 7 * THREADS_PER_BLOCK

        if idx0 < n_elements:
            acc = acc + vector[idx0]
        if idx1 < n_elements:
            acc = acc + vector[idx1]
        if idx2 < n_elements:
            acc = acc + vector[idx2]
        if idx3 < n_elements:
            acc = acc + vector[idx3]
        if idx4 < n_elements:
            acc = acc + vector[idx4]
        if idx5 < n_elements:
            acc = acc + vector[idx5]
        if idx6 < n_elements:
            acc = acc + vector[idx6]
        if idx7 < n_elements:
            acc = acc + vector[idx7]

        index = index + grid_stride

    acc = cute.arch.warp_reduction_sum(acc)
    if lane == 0:
        warp_sums[warp] = acc

    cute.arch.sync_threads()

    if warp == 0:
        block_sum = warp_sums[lane] if lane < WARPS_PER_BLOCK else cutlass.Float32(0.0)
        block_sum = cute.arch.warp_reduction_sum(block_sum)
        if lane == 0:
            partials[cta_x] = block_sum


@cute.kernel
def final_reduce_kernel(partials: cute.Tensor, output: cute.Tensor):
    tid_x, _, _ = cute.arch.thread_idx()
    lane = cute.arch.lane_idx()
    warp = cute.arch.warp_idx()

    n_partials = partials.shape[0]

    smem = utils.SmemAllocator()
    warp_sums = smem.allocate_tensor(cutlass.Float32, WARPS_PER_BLOCK)

    acc = cutlass.Float32(0.0)
    index = tid_x
    while index < n_partials:
        acc = acc + partials[index]
        index = index + THREADS_PER_BLOCK

    acc = cute.arch.warp_reduction_sum(acc)
    if lane == 0:
        warp_sums[warp] = acc

    cute.arch.sync_threads()

    if warp == 0:
        block_sum = warp_sums[lane] if lane < WARPS_PER_BLOCK else cutlass.Float32(0.0)
        block_sum = cute.arch.warp_reduction_sum(block_sum)
        if lane == 0:
            output[0] = block_sum


@cute.jit
def launch_reduce_tiles(vector: cute.Tensor, partials: cute.Tensor):
    reduce_tiles_kernel(vector, partials).launch(
        grid=(partials.shape[0], 1, 1),
        block=[THREADS_PER_BLOCK, 1, 1],
        cluster=(1, 1, 1),
    )


@cute.jit
def launch_final_reduce(partials: cute.Tensor, output: cute.Tensor):
    final_reduce_kernel(partials, output).launch(
        grid=(1, 1, 1),
        block=[THREADS_PER_BLOCK, 1, 1],
        cluster=(1, 1, 1),
    )


def _ceil_div(x: int, y: int) -> int:
    return (x + y - 1) // y


def _rounded_capacity(size: int) -> int:
    capacity = 1
    while capacity < size:
        capacity *= 2
    return capacity


def _compile_kernels() -> tuple[Callable, Callable]:
    global _compiled_reduce_tiles, _compiled_final_reduce

    if _compiled_reduce_tiles is not None and _compiled_final_reduce is not None:
        return _compiled_reduce_tiles, _compiled_final_reduce

    vector_len = cute.sym_int()
    partial_len = cute.sym_int()

    fake_vector = cute.runtime.make_fake_compact_tensor(cute.Float32, (vector_len,))
    fake_partials = cute.runtime.make_fake_compact_tensor(cute.Float32, (partial_len,))
    fake_output = cute.runtime.make_fake_compact_tensor(cute.Float32, (1,))

    _compiled_reduce_tiles = cute.compile(
        launch_reduce_tiles,
        fake_vector,
        fake_partials,
        options="--enable-tvm-ffi",
    )
    _compiled_final_reduce = cute.compile(
        launch_final_reduce,
        fake_partials,
        fake_output,
        options="--enable-tvm-ffi",
    )
    return _compiled_reduce_tiles, _compiled_final_reduce


def _get_partial_buffer(device: torch.device, num_blocks: int) -> torch.Tensor:
    device_index = device.index
    if device_index is None:
        device_index = torch.cuda.current_device()

    partial_buffer = _partial_buffer_cache.get(device_index)
    if partial_buffer is None or partial_buffer.numel() < num_blocks:
        partial_buffer = torch.empty(
            _rounded_capacity(num_blocks),
            device=device,
            dtype=torch.float32,
        )
        _partial_buffer_cache[device_index] = partial_buffer
    return partial_buffer[:num_blocks]


def _num_blocks(n_elements: int, device: torch.device) -> int:
    sm_count = torch.cuda.get_device_properties(device).multi_processor_count
    return max(1, min(_ceil_div(n_elements, BLOCK_ITEMS), sm_count * MAX_BLOCKS_PER_SM))


def custom_kernel(data: input_t) -> output_t:
    vector, output = data
    n_elements = vector.numel()

    if n_elements == 0:
        output[0] = 0.0
        return output[0]

    num_blocks = _num_blocks(n_elements, vector.device)
    partials = _get_partial_buffer(vector.device, num_blocks)
    compiled_reduce_tiles, compiled_final_reduce = _compile_kernels()

    compiled_reduce_tiles(vector, partials)
    compiled_final_reduce(partials, output)

    return output[0]
scrolls · 214 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON