Skip to content
KernelIndex
Search⌘K

submission 187025

literid · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 83 lines, June 9 Researcher Reciprocity License v1.0.

triton_my.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-187025?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA H100
84.5µs
#14 of 37
2025-12-20

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:6567d169c3fcc07b6d2778500462c3b9431cec0396a040ed79cfbdf222dbba59
license declaredunknown
license concludedunknown
authorsliterid
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 8num_warps=8,

Kernel source

triton_my.py83 lines
import functools

import torch
import triton
import triton.language as tl

from task import input_t, output_t


@triton.jit
def _partial_sum_kernel(
    x_ptr,
    partial_ptr,
    n_elements,
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(0)
    offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_elements
    x = tl.load(x_ptr + offsets, mask=mask, other=0.0)
    s = tl.sum(x, axis=0)
    tl.store(partial_ptr + pid, s)


@triton.jit
def _final_sum_kernel(
    partial_ptr,
    out_ptr,
    n_partials,
    BLOCK_SIZE: tl.constexpr,
):
    offsets = tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_partials
    x = tl.load(partial_ptr + offsets, mask=mask, other=0.0)
    s = tl.sum(x, axis=0)
    tl.store(out_ptr, s)  # out_ptr is a pointer to output[0]


@functools.lru_cache(maxsize=64)
def _get_partial_buffer(num_blocks: int, device: torch.device) -> torch.Tensor:
    # Cache to avoid per-call allocations during benchmarking repeats.
    return torch.empty((num_blocks,), device=device, dtype=torch.float32)


def _custom_kernel(data: input_t) -> output_t:
    x, out = data
    n = x.numel()

    # Tuned for large 1D reductions (H100): fewer blocks -> fewer partials.
    BLOCK1 = 8192
    grid = (triton.cdiv(n, BLOCK1),)

    partial = _get_partial_buffer(grid[0], x.device)

    # Pass 1: compute block partials (no atomics).
    _partial_sum_kernel[grid](
        x,
        partial,
        n,
        BLOCK_SIZE=BLOCK1,
        num_warps=8,
    )

    # Pass 2: reduce partials into the provided output buffer.
    # grid[0] is <= ~6400 for the biggest benchmark when BLOCK1=8192, so 8192 covers it.
    BLOCK2 = 8192
    _final_sum_kernel[(1,)](
        partial,
        out,
        grid[0],
        BLOCK_SIZE=BLOCK2,
        num_warps=8,
    )

    # Reference expects a 0-D scalar tensor (same as `data.sum()`), not a (1,) buffer.
    return out[0]


custom_kernel = _custom_kernel



scrolls · 83 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 186878.

+ import functools
+
import torch
+ import triton
+ import triton.language as tl
+
from task import input_t, output_t
+ @triton.jit
+ def _partial_sum_kernel(
+ x_ptr,
+ partial_ptr,
+ n_elements,
+ BLOCK_SIZE: tl.constexpr,
+ ):
+ pid = tl.program_id(0)
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < n_elements
+ x = tl.load(x_ptr + offsets, mask=mask, other=0.0)
+ s = tl.sum(x, axis=0)
+ tl.store(partial_ptr + pid, s)
+
+
+ @triton.jit
+ def _final_sum_kernel(
+ partial_ptr,
+ out_ptr,
+ n_partials,
+ BLOCK_SIZE: tl.constexpr,
+ ):
+ offsets = tl.arange(0, BLOCK_SIZE)
+ mask = offsets < n_partials
+ x = tl.load(partial_ptr + offsets, mask=mask, other=0.0)
+ s = tl.sum(x, axis=0)
+ tl.store(out_ptr, s) # out_ptr is a pointer to output[0]
+
+
+ @functools.lru_cache(maxsize=64)
+ def _get_partial_buffer(num_blocks: int, device: torch.device) -> torch.Tensor:
+ # Cache to avoid per-call allocations during benchmarking repeats.
+ return torch.empty((num_blocks,), device=device, dtype=torch.float32)
+
+
def _custom_kernel(data: input_t) -> output_t:
- data, output = data
- output = data.sum()
- return output
+ x, out = data
+ n = x.numel()
+ # Tuned for large 1D reductions (H100): fewer blocks -> fewer partials.
+ BLOCK1 = 8192
+ grid = (triton.cdiv(n, BLOCK1),)
+ partial = _get_partial_buffer(grid[0], x.device)
- # Compile the kernel for better performance
+ # Pass 1: compute block partials (no atomics).
+ _partial_sum_kernel[grid](
+ x,
+ partial,
+ n,
+ BLOCK_SIZE=BLOCK1,
+ num_warps=8,
+ )
+
+ # Pass 2: reduce partials into the provided output buffer.
+ # grid[0] is <= ~6400 for the biggest benchmark when BLOCK1=8192, so 8192 covers it.
+ BLOCK2 = 8192
+ _final_sum_kernel[(1,)](
+ partial,
+ out,
+ grid[0],
+ BLOCK_SIZE=BLOCK2,
+ num_warps=8,
+ )
+
+ # Reference expects a 0-D scalar tensor (same as `data.sum()`), not a (1,) buffer.
+ return out[0]
+
+
custom_kernel = _custom_kernel
+
+
scrolls · 86 diff lines total

Best evidence level for this revision: reported

JSON