submission 769468
Kernel-Zhang · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 114 lines, June 9 Researcher Reciprocity License v1.0.
mytri_00001.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-769468?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:7996d87841baf28d94a2c70b3ec9a6faa2d461ef94ff1b7feb33499e5446e3e8
license declaredunknown
license concludedunknown
authorsKernel-Zhang
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
num_warps=8, # A100 每个 SM 最多 8 warpsstages = 4
num_stages=4, # 软件流水线深度Kernel source
mytri_00001.py114 lines
from utils import make_match_reference, DeterministicContext
import torch
from task import input_t, output_t
import triton
import triton.language as tl
@triton.jit
def sum_kernel_a100(
x_ptr,
out_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr,
):
"""
使用共享内存进行块内归约,并将最终结果原子累加到全局输出。
所有归约均在 float64 下完成,确保与参考实现数值一致。
"""
pid = tl.program_id(axis=0)
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
x = tl.load(x_ptr + offsets, mask=mask, other=0.0)
# 块内归约:使用 tl.sum 在寄存器层完成(Triton 会自动优化)
partial_sum = tl.sum(x)
# 将本块的部分和原子累加到全局双精度输出
tl.atomic_add(out_ptr, partial_sum)
def custom_kernel(data: input_t) -> output_t:
"""
A100 优化的全局求和,严格匹配 ref_kernel 的数值精度。
"""
input_tensor, _ = data
n_elements = input_tensor.numel()
if n_elements == 0:
return torch.zeros((), device=input_tensor.device, dtype=torch.float32)
# 确保输入连续
x = input_tensor.contiguous()
# 双精度临时缓冲区,用于原子累加
out_double = torch.zeros(1, device="cuda", dtype=torch.float64)
# A100 调优参数
BLOCK_SIZE = 2048
grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]),)
# 正确传递 num_warps 和 num_stages(作为 kernel 启动参数)
sum_kernel_a100[grid](
x, out_double, n_elements,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=8, # A100 每个 SM 最多 8 warps
num_stages=4, # 软件流水线深度
)
return out_double[0].to(torch.float32)
def ref_kernel(data: input_t) -> output_t:
"""
Reference implementation of vector sum reduction using PyTorch.
Args:
data: Input tensor to be reduced
Returns:
Tensor containing the sum of all elements
"""
with DeterministicContext():
data, output = data
# Let's be on the safe side here, and do the reduction in 64 bit
output = data.to(torch.float64).sum().to(torch.float32)
return output
def generate_input(size: int, seed: int) -> input_t:
"""
Generates random input tensor of specified shape with random offset and scale.
The data is first generated as standard normal, then scaled and offset
to prevent trivial solutions.
Returns:
Tensor to be reduced
"""
gen = torch.Generator(device="cuda")
gen.manual_seed(seed)
# Generate base random data
data = torch.randn(
size, device="cuda", dtype=torch.float32, generator=gen
).contiguous()
# Generate random offset and scale (using different seeds to avoid correlation)
offset_gen = torch.Generator(device="cuda")
offset_gen.manual_seed(seed + 1)
scale_gen = torch.Generator(device="cuda")
scale_gen.manual_seed(seed + 2)
# Generate random offset between -100 and 100
offset = (torch.rand(1, device="cuda", generator=offset_gen) * 200 - 100).item()
# Generate random scale between 0.1 and 10
scale = (torch.rand(1, device="cuda", generator=scale_gen) * 9.9 + 0.1).item()
# Apply scale and offset
input_tensor = (data * scale + offset).contiguous()
output_tensor = torch.empty(1, device="cuda", dtype=torch.float32)
return input_tensor, output_tensor
check_implementation = make_match_reference(ref_kernel)
scrolls · 114 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 769437.
⋯ 4 unchanged linesimport tritonimport triton.language as tl-@triton.jit- def reduce_chunked_to_fp64_kernel(- in_ptr,+ def sum_kernel_a100(+ x_ptr,out_ptr,n_elements,BLOCK_SIZE: tl.constexpr,- CHUNKS_PER_PROGRAM: tl.constexpr,):+ """+ 使用共享内存进行块内归约,并将最终结果原子累加到全局输出。+ 所有归约均在 float64 下完成,确保与参考实现数值一致。+ """pid = tl.program_id(axis=0)- base = pid * BLOCK_SIZE * CHUNKS_PER_PROGRAM- lane = tl.arange(0, BLOCK_SIZE)-- acc = tl.zeros((BLOCK_SIZE,), dtype=tl.float64)- for i in range(CHUNKS_PER_PROGRAM):- offsets = base + i * BLOCK_SIZE + lane- mask = offsets < n_elements- x = tl.load(in_ptr + offsets, mask=mask, other=0.0).to(tl.float64)- acc += x-- partial = tl.sum(acc, axis=0)- tl.store(out_ptr + pid, partial)--- @triton.jit- def reduce_fp64_kernel(- in_ptr,- out_ptr,- n_elements,- BLOCK_SIZE: tl.constexpr,- ):- pid = tl.program_id(axis=0)- offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ block_start = pid * BLOCK_SIZE+ offsets = block_start + tl.arange(0, BLOCK_SIZE)mask = offsets < n_elements- x = tl.load(in_ptr + offsets, mask=mask, other=0.0)- partial = tl.sum(x, axis=0)- tl.store(out_ptr + pid, partial)+ x = tl.load(x_ptr + offsets, mask=mask, other=0.0)+ # 块内归约:使用 tl.sum 在寄存器层完成(Triton 会自动优化)+ partial_sum = tl.sum(x)- def ref_kernel(data: input_t) -> output_t:- """- Reference implementation of vector sum reduction using PyTorch.- Args:- data: Input tensor to be reduced- Returns:- Tensor containing the sum of all elements- """- with DeterministicContext():- data, output = data- # Let's be on the safe side here, and do the reduction in 64 bit- output = data.to(torch.float64).sum().to(torch.float32)- return output+ # 将本块的部分和原子累加到全局双精度输出+ tl.atomic_add(out_ptr, partial_sum)def custom_kernel(data: input_t) -> output_t:+ """+ A100 优化的全局求和,严格匹配 ref_kernel 的数值精度。+ """input_tensor, _ = datan_elements = input_tensor.numel()if n_elements == 0:return torch.zeros((), device=input_tensor.device, dtype=torch.float32)+ # 确保输入连续x = input_tensor.contiguous()- # A100 优化参数:每个 program 处理 4 个 1024 元素块(共 4096 元素)- # 目标是减少 launch program 数,同时保持较高并行度和内存吞吐。- BLOCK_SIZE = 1024- CHUNKS_PER_PROGRAM = 2- num_warps = 8- num_stages = 4+ # 双精度临时缓冲区,用于原子累加+ out_double = torch.zeros(1, device="cuda", dtype=torch.float64)- n_partials = triton.cdiv(n_elements, BLOCK_SIZE * CHUNKS_PER_PROGRAM)- partials = torch.empty((n_partials,), device=x.device, dtype=torch.float64)- reduce_chunked_to_fp64_kernel[(n_partials,)](- x,- partials,- n_elements,+ # A100 调优参数+ BLOCK_SIZE = 2048+ grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]),)++ # 正确传递 num_warps 和 num_stages(作为 kernel 启动参数)+ sum_kernel_a100[grid](+ x, out_double, n_elements,BLOCK_SIZE=BLOCK_SIZE,- CHUNKS_PER_PROGRAM=CHUNKS_PER_PROGRAM,- num_warps=num_warps,- num_stages=num_stages,+ num_warps=8, # A100 每个 SM 最多 8 warps+ num_stages=4, # 软件流水线深度)- while n_partials > 1:- next_n = triton.cdiv(n_partials, BLOCK_SIZE * CHUNKS_PER_PROGRAM)- next_partials = torch.empty((next_n,), device=x.device, dtype=torch.float64)- reduce_chunked_to_fp64_kernel[(next_n,)](- partials,- next_partials,- n_partials,- BLOCK_SIZE=BLOCK_SIZE,- CHUNKS_PER_PROGRAM=CHUNKS_PER_PROGRAM,- num_warps=num_warps,- num_stages=num_stages,- )- partials = next_partials- n_partials = next_n+ return out_double[0].to(torch.float32)- return partials[0].to(torch.float32)+ def ref_kernel(data: input_t) -> output_t:+ """+ Reference implementation of vector sum reduction using PyTorch.+ Args:+ data: Input tensor to be reduced+ Returns:+ Tensor containing the sum of all elements+ """+ with DeterministicContext():+ data, output = data+ # Let's be on the safe side here, and do the reduction in 64 bit+ output = data.to(torch.float64).sum().to(torch.float32)+ return outputdef generate_input(size: int, seed: int) -> input_t:"""⋯ 30 unchanged linescheck_implementation = make_match_reference(ref_kernel)+
scrolls · 154 diff lines total
Best evidence level for this revision: reported
JSON