Skip to content
KernelIndex
Search⌘K

submission 769395

Kernel-Zhang · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 132 lines, June 9 Researcher Reciprocity License v1.0.

triton_00003.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-769395?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA A100
150.7µs
#51 of 96
2026-04-15

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:f9cb2034462a07371c38c3e6655507771aa30220b1e0c1d4b9fae777be5cac0a
license declaredunknown
license concludedunknown
authorsKernel-Zhang
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 8num_warps = 8
stages = 3num_stages = 3

Kernel source

triton_00003.py132 lines
from utils import make_match_reference, DeterministicContext
import torch
from task import input_t, output_t

import triton
import triton.language as tl


@triton.jit
def reduce_fp32_to_fp64_kernel(
    in_ptr,
    out_ptr,
    n_elements,
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(axis=0)
    offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_elements

    x = tl.load(in_ptr + offsets, mask=mask, other=0.0).to(tl.float64)
    partial = tl.sum(x, axis=0)
    tl.store(out_ptr + pid, partial)


@triton.jit
def reduce_fp64_kernel(
    in_ptr,
    out_ptr,
    n_elements,
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(axis=0)
    offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_elements

    x = tl.load(in_ptr + offsets, mask=mask, other=0.0)
    partial = tl.sum(x, axis=0)
    tl.store(out_ptr + pid, partial)


def ref_kernel(data: input_t) -> output_t:
    """
    Reference implementation of vector sum reduction using PyTorch.
    Args:
        data: Input tensor to be reduced
    Returns:
        Tensor containing the sum of all elements
    """
    with DeterministicContext():
        data, output = data
        # Let's be on the safe side here, and do the reduction in 64 bit
        output = data.to(torch.float64).sum().to(torch.float32)
        return output


def custom_kernel(data: input_t) -> output_t:
    input_tensor, _ = data
    n_elements = input_tensor.numel()

    if n_elements == 0:
        return torch.zeros((), device=input_tensor.device, dtype=torch.float32)

    x = input_tensor.contiguous()

    BLOCK_SIZE = 1024
    num_warps = 8
    num_stages = 3

    n_partials = triton.cdiv(n_elements, BLOCK_SIZE)
    partials = torch.empty((n_partials,), device=x.device, dtype=torch.float64)
    reduce_fp32_to_fp64_kernel[(n_partials,)](
        x,
        partials,
        n_elements,
        BLOCK_SIZE=BLOCK_SIZE,
        num_warps=num_warps,
        num_stages=num_stages,
    )

    while n_partials > 1:
        next_n = triton.cdiv(n_partials, BLOCK_SIZE)
        next_partials = torch.empty((next_n,), device=x.device, dtype=torch.float64)
        reduce_fp64_kernel[(next_n,)](
            partials,
            next_partials,
            n_partials,
            BLOCK_SIZE=BLOCK_SIZE,
            num_warps=num_warps,
            num_stages=num_stages,
        )
        partials = next_partials
        n_partials = next_n

    return partials[0].to(torch.float32)


def generate_input(size: int, seed: int) -> input_t:
    """
    Generates random input tensor of specified shape with random offset and scale.
    The data is first generated as standard normal, then scaled and offset
    to prevent trivial solutions.

    Returns:
        Tensor to be reduced
    """
    gen = torch.Generator(device="cuda")
    gen.manual_seed(seed)

    # Generate base random data
    data = torch.randn(
        size, device="cuda", dtype=torch.float32, generator=gen
    ).contiguous()

    # Generate random offset and scale (using different seeds to avoid correlation)
    offset_gen = torch.Generator(device="cuda")
    offset_gen.manual_seed(seed + 1)
    scale_gen = torch.Generator(device="cuda")
    scale_gen.manual_seed(seed + 2)

    # Generate random offset between -100 and 100
    offset = (torch.rand(1, device="cuda", generator=offset_gen) * 200 - 100).item()
    # Generate random scale between 0.1 and 10
    scale = (torch.rand(1, device="cuda", generator=scale_gen) * 9.9 + 0.1).item()

    # Apply scale and offset
    input_tensor = (data * scale + offset).contiguous()
    output_tensor = torch.empty(1, device="cuda", dtype=torch.float32)
    return input_tensor, output_tensor


check_implementation = make_match_reference(ref_kernel)
scrolls · 132 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 767055.

⋯ 1 unchanged lines
import torch
from task import input_t, output_t
+ import triton
+ import triton.language as tl
+
+ @triton.jit
+ def reduce_fp32_to_fp64_kernel(
+ in_ptr,
+ out_ptr,
+ n_elements,
+ BLOCK_SIZE: tl.constexpr,
+ ):
+ pid = tl.program_id(axis=0)
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < n_elements
+
+ x = tl.load(in_ptr + offsets, mask=mask, other=0.0).to(tl.float64)
+ partial = tl.sum(x, axis=0)
+ tl.store(out_ptr + pid, partial)
+
+
+ @triton.jit
+ def reduce_fp64_kernel(
+ in_ptr,
+ out_ptr,
+ n_elements,
+ BLOCK_SIZE: tl.constexpr,
+ ):
+ pid = tl.program_id(axis=0)
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < n_elements
+
+ x = tl.load(in_ptr + offsets, mask=mask, other=0.0)
+ partial = tl.sum(x, axis=0)
+ tl.store(out_ptr + pid, partial)
+
+
def ref_kernel(data: input_t) -> output_t:
"""
Reference implementation of vector sum reduction using PyTorch.
⋯ 8 unchanged lines
output = data.to(torch.float64).sum().to(torch.float32)
return output
+
def custom_kernel(data: input_t) -> output_t:
- data, _ = data
- # Let's be on the safe side here, and do the reduction in 64 bit
- output = data.sum()
- return output
+ input_tensor, _ = data
+ n_elements = input_tensor.numel()
+ if n_elements == 0:
+ return torch.zeros((), device=input_tensor.device, dtype=torch.float32)
+
+ x = input_tensor.contiguous()
+
+ BLOCK_SIZE = 1024
+ num_warps = 8
+ num_stages = 3
+
+ n_partials = triton.cdiv(n_elements, BLOCK_SIZE)
+ partials = torch.empty((n_partials,), device=x.device, dtype=torch.float64)
+ reduce_fp32_to_fp64_kernel[(n_partials,)](
+ x,
+ partials,
+ n_elements,
+ BLOCK_SIZE=BLOCK_SIZE,
+ num_warps=num_warps,
+ num_stages=num_stages,
+ )
+
+ while n_partials > 1:
+ next_n = triton.cdiv(n_partials, BLOCK_SIZE)
+ next_partials = torch.empty((next_n,), device=x.device, dtype=torch.float64)
+ reduce_fp64_kernel[(next_n,)](
+ partials,
+ next_partials,
+ n_partials,
+ BLOCK_SIZE=BLOCK_SIZE,
+ num_warps=num_warps,
+ num_stages=num_stages,
+ )
+ partials = next_partials
+ n_partials = next_n
+
+ return partials[0].to(torch.float32)
+
+
def generate_input(size: int, seed: int) -> input_t:
"""
Generates random input tensor of specified shape with random offset and scale.
⋯ 29 unchanged lines
check_implementation = make_match_reference(ref_kernel)
-
scrolls · 101 diff lines total

Best evidence level for this revision: reported

JSON