submission 769395
Kernel-Zhang · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 132 lines, June 9 Researcher Reciprocity License v1.0.
triton_00003.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-769395?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:f9cb2034462a07371c38c3e6655507771aa30220b1e0c1d4b9fae777be5cac0a
license declaredunknown
license concludedunknown
authorsKernel-Zhang
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
num_warps = 8stages = 3
num_stages = 3Kernel source
triton_00003.py132 lines
from utils import make_match_reference, DeterministicContext
import torch
from task import input_t, output_t
import triton
import triton.language as tl
@triton.jit
def reduce_fp32_to_fp64_kernel(
in_ptr,
out_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(axis=0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
x = tl.load(in_ptr + offsets, mask=mask, other=0.0).to(tl.float64)
partial = tl.sum(x, axis=0)
tl.store(out_ptr + pid, partial)
@triton.jit
def reduce_fp64_kernel(
in_ptr,
out_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(axis=0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
x = tl.load(in_ptr + offsets, mask=mask, other=0.0)
partial = tl.sum(x, axis=0)
tl.store(out_ptr + pid, partial)
def ref_kernel(data: input_t) -> output_t:
"""
Reference implementation of vector sum reduction using PyTorch.
Args:
data: Input tensor to be reduced
Returns:
Tensor containing the sum of all elements
"""
with DeterministicContext():
data, output = data
# Let's be on the safe side here, and do the reduction in 64 bit
output = data.to(torch.float64).sum().to(torch.float32)
return output
def custom_kernel(data: input_t) -> output_t:
input_tensor, _ = data
n_elements = input_tensor.numel()
if n_elements == 0:
return torch.zeros((), device=input_tensor.device, dtype=torch.float32)
x = input_tensor.contiguous()
BLOCK_SIZE = 1024
num_warps = 8
num_stages = 3
n_partials = triton.cdiv(n_elements, BLOCK_SIZE)
partials = torch.empty((n_partials,), device=x.device, dtype=torch.float64)
reduce_fp32_to_fp64_kernel[(n_partials,)](
x,
partials,
n_elements,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=num_warps,
num_stages=num_stages,
)
while n_partials > 1:
next_n = triton.cdiv(n_partials, BLOCK_SIZE)
next_partials = torch.empty((next_n,), device=x.device, dtype=torch.float64)
reduce_fp64_kernel[(next_n,)](
partials,
next_partials,
n_partials,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=num_warps,
num_stages=num_stages,
)
partials = next_partials
n_partials = next_n
return partials[0].to(torch.float32)
def generate_input(size: int, seed: int) -> input_t:
"""
Generates random input tensor of specified shape with random offset and scale.
The data is first generated as standard normal, then scaled and offset
to prevent trivial solutions.
Returns:
Tensor to be reduced
"""
gen = torch.Generator(device="cuda")
gen.manual_seed(seed)
# Generate base random data
data = torch.randn(
size, device="cuda", dtype=torch.float32, generator=gen
).contiguous()
# Generate random offset and scale (using different seeds to avoid correlation)
offset_gen = torch.Generator(device="cuda")
offset_gen.manual_seed(seed + 1)
scale_gen = torch.Generator(device="cuda")
scale_gen.manual_seed(seed + 2)
# Generate random offset between -100 and 100
offset = (torch.rand(1, device="cuda", generator=offset_gen) * 200 - 100).item()
# Generate random scale between 0.1 and 10
scale = (torch.rand(1, device="cuda", generator=scale_gen) * 9.9 + 0.1).item()
# Apply scale and offset
input_tensor = (data * scale + offset).contiguous()
output_tensor = torch.empty(1, device="cuda", dtype=torch.float32)
return input_tensor, output_tensor
check_implementation = make_match_reference(ref_kernel)
scrolls · 132 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 767055.
⋯ 1 unchanged linesimport torchfrom task import input_t, output_t+ import triton+ import triton.language as tl++ @triton.jit+ def reduce_fp32_to_fp64_kernel(+ in_ptr,+ out_ptr,+ n_elements,+ BLOCK_SIZE: tl.constexpr,+ ):+ pid = tl.program_id(axis=0)+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_elements++ x = tl.load(in_ptr + offsets, mask=mask, other=0.0).to(tl.float64)+ partial = tl.sum(x, axis=0)+ tl.store(out_ptr + pid, partial)+++ @triton.jit+ def reduce_fp64_kernel(+ in_ptr,+ out_ptr,+ n_elements,+ BLOCK_SIZE: tl.constexpr,+ ):+ pid = tl.program_id(axis=0)+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_elements++ x = tl.load(in_ptr + offsets, mask=mask, other=0.0)+ partial = tl.sum(x, axis=0)+ tl.store(out_ptr + pid, partial)++def ref_kernel(data: input_t) -> output_t:"""Reference implementation of vector sum reduction using PyTorch.⋯ 8 unchanged linesoutput = data.to(torch.float64).sum().to(torch.float32)return output+def custom_kernel(data: input_t) -> output_t:- data, _ = data- # Let's be on the safe side here, and do the reduction in 64 bit- output = data.sum()- return output+ input_tensor, _ = data+ n_elements = input_tensor.numel()+ if n_elements == 0:+ return torch.zeros((), device=input_tensor.device, dtype=torch.float32)++ x = input_tensor.contiguous()++ BLOCK_SIZE = 1024+ num_warps = 8+ num_stages = 3++ n_partials = triton.cdiv(n_elements, BLOCK_SIZE)+ partials = torch.empty((n_partials,), device=x.device, dtype=torch.float64)+ reduce_fp32_to_fp64_kernel[(n_partials,)](+ x,+ partials,+ n_elements,+ BLOCK_SIZE=BLOCK_SIZE,+ num_warps=num_warps,+ num_stages=num_stages,+ )++ while n_partials > 1:+ next_n = triton.cdiv(n_partials, BLOCK_SIZE)+ next_partials = torch.empty((next_n,), device=x.device, dtype=torch.float64)+ reduce_fp64_kernel[(next_n,)](+ partials,+ next_partials,+ n_partials,+ BLOCK_SIZE=BLOCK_SIZE,+ num_warps=num_warps,+ num_stages=num_stages,+ )+ partials = next_partials+ n_partials = next_n++ return partials[0].to(torch.float32)++def generate_input(size: int, seed: int) -> input_t:"""Generates random input tensor of specified shape with random offset and scale.⋯ 29 unchanged linescheck_implementation = make_match_reference(ref_kernel)-
scrolls · 101 diff lines total
Best evidence level for this revision: reported
JSON