Skip to content
KernelIndex
Search⌘K

submission 772348

Kernel-Zhang · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 133 lines, June 9 Researcher Reciprocity License v1.0.

mytri_00010.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-772348?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA A100
138.7µs
#6 of 96
2026-04-16

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:ba8928284c25cb193e3e8a830d33790b12242ce4b56379e9d37324bc014fa325
license declaredunknown
license concludedunknown
authorsKernel-Zhang
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

stages = 4NUM_STAGES = 4

Kernel source

mytri_00010.py133 lines
from utils import make_match_reference, DeterministicContext
import torch
from task import input_t, output_t

import triton
import triton.language as tl

N_ELEMENTS = 52428800
BLOCK_SIZE = 1024
# NUM_WARPS = 32
NUM_STAGES = 4
CHUNKS = 32

N1 = triton.cdiv(N_ELEMENTS, BLOCK_SIZE * CHUNKS)
_GLOBAL_REDUCE_BUF = torch.empty(N1, device="cuda", dtype=torch.float32)
BLOCK_SIZE2=triton.next_power_of_2(N1)

@triton.jit
def reduce_kernel(
    in_ptr,
    out_ptr,
    n_elements,
    BLOCK_SIZE: tl.constexpr,
    CHUNKS_PER_BLOCK: tl.constexpr,
):
    pid = tl.program_id(axis=0)
    block_start = pid * BLOCK_SIZE * CHUNKS_PER_BLOCK

    acc = tl.zeros((BLOCK_SIZE,), dtype=tl.float32)

    for c in range(CHUNKS_PER_BLOCK):
        # 计算当前 chunk 的内存偏移量
        offsets = block_start + c * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
        
        # 边界保护
        mask = offsets < n_elements

        # 加载当前 chunk 的数据
        x = tl.load(in_ptr + offsets, mask=mask, other=0.0)
        
        # 累加到 acc 中
        acc += x

    partial = tl.sum(acc, axis=0)
    tl.store(out_ptr + pid, partial)


def ref_kernel(data: input_t) -> output_t:
    """
    Reference implementation of vector sum reduction using PyTorch.
    Args:
        data: Input tensor to be reduced
    Returns:
        Tensor containing the sum of all elements
    """
    with DeterministicContext():
        data, output = data
        # Let's be on the safe side here, and do the reduction in 64 bit
        output = data.to(torch.float64).sum().to(torch.float32)
        return output


def custom_kernel(data: input_t) -> output_t:
    input_tensor, _ = data
    n_elements = input_tensor.numel()
    if n_elements != N_ELEMENTS:
        return input_tensor.sum()

    reduce_kernel[(N1,)](
        input_tensor,
        _GLOBAL_REDUCE_BUF,
        n_elements,
        BLOCK_SIZE=BLOCK_SIZE,
        CHUNKS_PER_BLOCK=CHUNKS,
        # num_warps=NUM_WARPS,
        num_stages=NUM_STAGES,
    )

    reduce_kernel[(1,)](
        _GLOBAL_REDUCE_BUF,
        input_tensor,  # 重用输入缓冲区作为中间结果存储
        N1,
        BLOCK_SIZE=BLOCK_SIZE2,
        CHUNKS_PER_BLOCK=1,  # 第二轮不需要分块了
        # num_warps=NUM_WARPS,
        num_stages=NUM_STAGES,
    )
    return input_tensor[0].to(torch.float32)
 

def generate_input(size: int, seed: int) -> input_t:
    """
    Generates random input tensor of specified shape with random offset and scale.
    The data is first generated as standard normal, then scaled and offset
    to prevent trivial solutions.

    Returns:
        Tensor to be reduced
    """
    gen = torch.Generator(device="cuda")
    gen.manual_seed(seed)

    # Generate base random data
    data = torch.randn(
        size, device="cuda", dtype=torch.float32, generator=gen
    ).contiguous()

    # Generate random offset and scale (using different seeds to avoid correlation)
    offset_gen = torch.Generator(device="cuda")
    offset_gen.manual_seed(seed + 1)
    scale_gen = torch.Generator(device="cuda")
    scale_gen.manual_seed(seed + 2)

    # Generate random offset between -100 and 100
    offset = (torch.rand(1, device="cuda", generator=offset_gen) * 200 - 100).item()
    # Generate random scale between 0.1 and 10
    scale = (torch.rand(1, device="cuda", generator=scale_gen) * 9.9 + 0.1).item()

    # Apply scale and offset
    input_tensor = (data * scale + offset).contiguous()
    output_tensor = torch.empty(1, device="cuda", dtype=torch.float32)
    return input_tensor, output_tensor


check_implementation = make_match_reference(ref_kernel)

def warmup(fn, args, n_warmup=5):
    for _ in range(n_warmup):
        _ = fn(args)
        torch.cuda.synchronize()

# 使用
warmup(custom_kernel, generate_input(N_ELEMENTS, 42))
scrolls · 133 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 771487.

⋯ 4 unchanged lines
import triton
import triton.language as tl
+ N_ELEMENTS = 52428800
+ BLOCK_SIZE = 1024
+ # NUM_WARPS = 32
+ NUM_STAGES = 4
+ CHUNKS = 32
+ N1 = triton.cdiv(N_ELEMENTS, BLOCK_SIZE * CHUNKS)
+ _GLOBAL_REDUCE_BUF = torch.empty(N1, device="cuda", dtype=torch.float32)
+ BLOCK_SIZE2=triton.next_power_of_2(N1)
+
@triton.jit
def reduce_kernel(
in_ptr,
out_ptr,
n_elements,
- BLOCK_SIZE: tl.constexpr
+ BLOCK_SIZE: tl.constexpr,
+ CHUNKS_PER_BLOCK: tl.constexpr,
):
pid = tl.program_id(axis=0)
- block_start = pid * BLOCK_SIZE
- offsets = block_start + tl.arange(0, BLOCK_SIZE)
- mask = offsets < n_elements
+ block_start = pid * BLOCK_SIZE * CHUNKS_PER_BLOCK
- # 加载数据块
- x = tl.load(in_ptr + offsets, mask=mask, other=0.0)
+ acc = tl.zeros((BLOCK_SIZE,), dtype=tl.float32)
- partial = tl.sum(x, axis=0)
+ for c in range(CHUNKS_PER_BLOCK):
+ # 计算当前 chunk 的内存偏移量
+ offsets = block_start + c * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+
+ # 边界保护
+ mask = offsets < n_elements
+
+ # 加载当前 chunk 的数据
+ x = tl.load(in_ptr + offsets, mask=mask, other=0.0)
+
+ # 累加到 acc 中
+ acc += x
+
+ partial = tl.sum(acc, axis=0)
tl.store(out_ptr + pid, partial)
⋯ 15 unchanged lines
def custom_kernel(data: input_t) -> output_t:
input_tensor, _ = data
n_elements = input_tensor.numel()
+ if n_elements != N_ELEMENTS:
+ return input_tensor.sum()
- # if n_elements == 0:
- # return torch.zeros((), device=input_tensor.device, dtype=torch.float32)
-
- # x = input_tensor.contiguous()
-
- # A100 优化参数:每个 program 处理 4 个 1024 元素块(共 4096 元素)
- # 目标是减少 launch program 数,同时保持较高并行度和内存吞吐。
- BLOCK_SIZE = 2048
- num_warps = 8
- num_stages = 4
-
- n_partials = triton.cdiv(n_elements, BLOCK_SIZE)
- partials = torch.empty((n_partials,), device="cuda", dtype=torch.float32)
- reduce_kernel[(n_partials,)](
+ reduce_kernel[(N1,)](
input_tensor,
- partials,
+ _GLOBAL_REDUCE_BUF,
n_elements,
BLOCK_SIZE=BLOCK_SIZE,
- num_warps=num_warps,
- num_stages=num_stages,
+ CHUNKS_PER_BLOCK=CHUNKS,
+ # num_warps=NUM_WARPS,
+ num_stages=NUM_STAGES,
)
- input = partials
- output = input_tensor
- while n_partials > 1:
- next_n = triton.cdiv(n_partials, BLOCK_SIZE)
- reduce_kernel[(next_n,)](
- input,
- output, # 重用输入缓冲区作为中间结果存储
- n_partials,
- BLOCK_SIZE=BLOCK_SIZE,
- num_warps=num_warps,
- num_stages=num_stages,
- )
- # 交换输入输出缓冲区
- input, output = output, input
- n_partials = next_n
+ reduce_kernel[(1,)](
+ _GLOBAL_REDUCE_BUF,
+ input_tensor, # 重用输入缓冲区作为中间结果存储
+ N1,
+ BLOCK_SIZE=BLOCK_SIZE2,
+ CHUNKS_PER_BLOCK=1, # 第二轮不需要分块了
+ # num_warps=NUM_WARPS,
+ num_stages=NUM_STAGES,
+ )
+ return input_tensor[0].to(torch.float32)
+
- return input[0].to(torch.float32)
-
-
def generate_input(size: int, seed: int) -> input_t:
"""
Generates random input tensor of specified shape with random offset and scale.
⋯ 29 unchanged lines
check_implementation = make_match_reference(ref_kernel)
+
+ def warmup(fn, args, n_warmup=5):
+ for _ in range(n_warmup):
+ _ = fn(args)
+ torch.cuda.synchronize()
+
+ # 使用
+ warmup(custom_kernel, generate_input(N_ELEMENTS, 42))
No newline at end of file
scrolls · 131 diff lines total

Best evidence level for this revision: reported

JSON