Skip to content
KernelIndex
Search⌘K

submission 775031

x3C49 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 62 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-775031?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA A100
147.5µs
#39 of 96
2026-04-19

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:1a4437de22f53718bc276212c49217cf3316ad7fab04666b93bce635ec217bf1
license declaredunknown
license concludedunknown
authorsx3C49
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 8num_warps=8,

Kernel source

submission.py62 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t

GRID = 512
BLOCK = 4096


@triton.jit
def _pass1(data_ptr, partial_ptr, N,
            BLOCK_SIZE: tl.constexpr, GRID: tl.constexpr):
    pid = tl.program_id(0)
    acc = tl.zeros((BLOCK_SIZE,), dtype=tl.float64)
    block_start = pid * BLOCK_SIZE
    step = GRID * BLOCK_SIZE
    while block_start < N:
        offs = block_start + tl.arange(0, BLOCK_SIZE)
        mask = offs < N
        vals = tl.load(data_ptr + offs, mask=mask, other=0.0)
        acc += vals.to(tl.float64)
        block_start += step
    partial = tl.sum(acc, axis=0)
    tl.store(partial_ptr + pid, partial)


@triton.jit
def _pass2(partial_ptr, output_ptr, N_PARTIALS: tl.constexpr):
    offs = tl.arange(0, N_PARTIALS)
    vals = tl.load(partial_ptr + offs)
    total = tl.sum(vals, axis=0)
    tl.store(output_ptr, total.to(tl.float32))


_scratch: torch.Tensor | None = None


def custom_kernel(data: input_t) -> output_t:
    global _scratch
    input_tensor, output_tensor = data
    N = input_tensor.numel()

    if _scratch is None:
        _scratch = torch.empty(GRID, device="cuda", dtype=torch.float64)

    _pass1[(GRID,)](
        input_tensor,
        _scratch,
        N,
        BLOCK_SIZE=BLOCK,
        GRID=GRID,
        num_warps=8,
    )

    _pass2[(1,)](
        _scratch,
        output_tensor,
        GRID,
        num_warps=4,
    )

    return output_tensor.view([])
scrolls · 62 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 774965.

⋯ 1 unchanged lines
import triton
import triton.language as tl
from task import input_t, output_t
+
GRID = 512
BLOCK = 4096
@triton.jit
- def _pass1(data_ptr, partial_ptr, N, BLOCK: tl.constexpr, GRID: tl.constexpr):
+ def _pass1(data_ptr, partial_ptr, N,
+ BLOCK_SIZE: tl.constexpr, GRID: tl.constexpr):
pid = tl.program_id(0)
- acc = tl.zeros((BLOCK,), dtype=tl.float64)
- start = pid * BLOCK
- stride = GRID * BLOCK
- while start < N:
- offs = start + tl.arange(0, BLOCK)
+ acc = tl.zeros((BLOCK_SIZE,), dtype=tl.float64)
+ block_start = pid * BLOCK_SIZE
+ step = GRID * BLOCK_SIZE
+ while block_start < N:
+ offs = block_start + tl.arange(0, BLOCK_SIZE)
mask = offs < N
- acc += tl.load(data_ptr + offs, mask=mask, other=0.0).to(tl.float64)
- start += stride
- tl.store(partial_ptr + pid, tl.sum(acc, axis=0))
+ vals = tl.load(data_ptr + offs, mask=mask, other=0.0)
+ acc += vals.to(tl.float64)
+ block_start += step
+ partial = tl.sum(acc, axis=0)
+ tl.store(partial_ptr + pid, partial)
@triton.jit
- def _pass2(partial_ptr, out_ptr, GRID: tl.constexpr):
- offs = tl.arange(0, GRID)
- total = tl.sum(tl.load(partial_ptr + offs), axis=0)
- tl.store(out_ptr, total.to(tl.float32))
+ def _pass2(partial_ptr, output_ptr, N_PARTIALS: tl.constexpr):
+ offs = tl.arange(0, N_PARTIALS)
+ vals = tl.load(partial_ptr + offs)
+ total = tl.sum(vals, axis=0)
+ tl.store(output_ptr, total.to(tl.float32))
_scratch: torch.Tensor | None = None
⋯ 11 unchanged lines
input_tensor,
_scratch,
N,
- BLOCK=BLOCK,
+ BLOCK_SIZE=BLOCK,
GRID=GRID,
num_warps=8,
)
+
_pass2[(1,)](
_scratch,
output_tensor,
- GRID=GRID,
+ GRID,
+ num_warps=4,
)
return output_tensor.view([])
No newline at end of file
scrolls · 68 diff lines total

Best evidence level for this revision: reported

JSON