submission 67720
.godelmachine · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 72 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-67720?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:a3001455debe44261096cc7b6dd18d2c138ec10fd2d0e7534b8432bd7903c304
license declaredunknown
license concludedunknown
authors.godelmachine
imported2026-08-15
Kernel source
submission.py72 lines
#!POPCORN leaderboard vectorsum_v2
#!POPCORN gpus A100
# This is a submission template for popcorn leaderboard 'vectorsum_v2'.
# Your task is as follows:
# > Implement a vector sum reduction kernel. This kernel computes the sum of all elements in the input tensor.
# >
# > Input: A tensor of shape `(N,)` with values from a normal distribution with mean 0 and variance 1.
# > Output: A scalar value equal to the sum of all elements in the input tensor.
# The deadline for this leaderboard is 2025-12-30 00:00:00+00:00
# You can automatically route this file to specific GPUs by adding a line
# `#!POPCORN gpus <GPUs>` to the header of this file.
# Happy hacking!
from task import input_t, output_t
from utils import DeterministicContext
import torch
import triton
import triton.language as tl
import time
@triton.jit
def sum_kernel(
i_ptr, # input ptr
o_ptr, # output ptr
n_elem, # total number of elements in input pointer
BLOCK_SIZE: tl.constexpr
):
pid = tl.program_id(0)
offsets = (pid * BLOCK_SIZE) + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elem
in_v = tl.load(i_ptr + offsets, mask, other=0.0)
# sum
sum_result = tl.sum(in_v)
o_offset = pid * 1
tl.store(o_ptr + o_offset, sum_result)
def custom_kernel(data: input_t) -> output_t:
data, output = data
# Let's be on the safe side here, and do the reduction in 64 bit
# output = data.to(torch.float64).sum().to(torch.float32)
n_elem = data.numel()
C_BLOCK_SIZE = 1024
c_grid_size = triton.cdiv(n_elem, C_BLOCK_SIZE)
BLOCK_SIZE = triton.cdiv(C_BLOCK_SIZE, 1)
grid_size = int(c_grid_size)
output = torch.zeros([grid_size]).cuda()
output_result = torch.zeros([1]).cuda()
grid = lambda meta: (triton.cdiv(n_elem, meta["BLOCK_SIZE"]), )
# start = time.time_ns()
sum_kernel[grid](data, output, n_elem, BLOCK_SIZE=BLOCK_SIZE)
# end = time.time_ns()
# total1 = (end - start)
# start = time.time_ns()
# result = output.sum()
new_block_size = triton.next_power_of_2(grid_size)
grid = lambda meta: (triton.cdiv(new_block_size, meta["BLOCK_SIZE"]), )
sum_kernel[grid](output, output_result, grid_size, BLOCK_SIZE=new_block_size)
# end = time.time_ns()
# total2 = (end - start)
# per = total2/(total1 + total2)
# print(n_elem," -> ", per)
# calculate sum of output
return output_result[0]scrolls · 72 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON