submission 67723
.godelmachine · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 72 lines, June 9 Researcher Reciprocity License v1.0.
sumission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-67723?include=source"interfacepython
Compatibility
measured onNVIDIA L4
declared hardwareNVIDIA L4
architecturessm_89
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:8f4e8186fc28bae7421c674907959c935046ceb9975536e7a912def67d485bad
license declaredunknown
license concludedunknown
authors.godelmachine
imported2026-08-15
Kernel source
sumission.py72 lines
#!POPCORN leaderboard vectorsum_v2
#!POPCORN gpus A100
# This is a submission template for popcorn leaderboard 'vectorsum_v2'.
# Your task is as follows:
# > Implement a vector sum reduction kernel. This kernel computes the sum of all elements in the input tensor.
# >
# > Input: A tensor of shape `(N,)` with values from a normal distribution with mean 0 and variance 1.
# > Output: A scalar value equal to the sum of all elements in the input tensor.
# The deadline for this leaderboard is 2025-12-30 00:00:00+00:00
# You can automatically route this file to specific GPUs by adding a line
# `#!POPCORN gpus <GPUs>` to the header of this file.
# Happy hacking!
from task import input_t, output_t
from utils import DeterministicContext
import torch
import triton
import triton.language as tl
import time
@triton.jit
def sum_kernel(
i_ptr, # input ptr
o_ptr, # output ptr
n_elem, # total number of elements in input pointer
BLOCK_SIZE: tl.constexpr
):
pid = tl.program_id(0)
offsets = (pid * BLOCK_SIZE) + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elem
in_v = tl.load(i_ptr + offsets, mask, other=0.0)
# sum
sum_result = tl.sum(in_v)
o_offset = pid * 1
tl.store(o_ptr + o_offset, sum_result)
def custom_kernel(data: input_t) -> output_t:
data, output = data
# Let's be on the safe side here, and do the reduction in 64 bit
# output = data.to(torch.float64).sum().to(torch.float32)
n_elem = data.numel()
C_BLOCK_SIZE = 1024
c_grid_size = triton.cdiv(n_elem, C_BLOCK_SIZE)
BLOCK_SIZE = triton.cdiv(C_BLOCK_SIZE, 1)
grid_size = int(c_grid_size)
output = torch.zeros([grid_size]).cuda()
output_result = torch.zeros([1]).cuda()
grid = lambda meta: (triton.cdiv(n_elem, meta["BLOCK_SIZE"]), )
# start = time.time_ns()
sum_kernel[grid](data, output, n_elem, BLOCK_SIZE=BLOCK_SIZE)
# end = time.time_ns()
# total1 = (end - start)
# start = time.time_ns()
# result = output.sum()
new_block_size = triton.next_power_of_2(grid_size)
grid = lambda meta: (triton.cdiv(new_block_size, meta["BLOCK_SIZE"]), )
sum_kernel[grid](output, output_result, grid_size, BLOCK_SIZE=new_block_size)
# end = time.time_ns()
# total2 = (end - start)
# per = total2/(total1 + total2)
# print(n_elem," -> ", per)
# calculate sum of output
return output_result[0]scrolls · 72 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 67722.
- #!POPCORN leaderboard vectorsum_v2- #!POPCORN gpus B200-- # This is a submission template for popcorn leaderboard 'vectorsum_v2'.- # Your task is as follows:- # > Implement a vector sum reduction kernel. This kernel computes the sum of all elements in the input tensor.- # >- # > Input: A tensor of shape `(N,)` with values from a normal distribution with mean 0 and variance 1.- # > Output: A scalar value equal to the sum of all elements in the input tensor.- # The deadline for this leaderboard is 2025-12-30 00:00:00+00:00-- # You can automatically route this file to specific GPUs by adding a line- # `#!POPCORN gpus <GPUs>` to the header of this file.- # Happy hacking!-- from task import input_t, output_t- from utils import DeterministicContext- import torch- import triton- import triton.language as tl- import time--- @triton.jit- def sum_kernel(- i_ptr, # input ptr- o_ptr, # output ptr- n_elem, # total number of elements in input pointer- BLOCK_SIZE: tl.constexpr- ):- pid = tl.program_id(0)-- offsets = (pid * BLOCK_SIZE) + tl.arange(0, BLOCK_SIZE)- mask = offsets < n_elem-- in_v = tl.load(i_ptr + offsets, mask, other=0.0)- # sum- sum_result = tl.sum(in_v)- o_offset = pid * 1- tl.store(o_ptr + o_offset, sum_result)---- def custom_kernel(data: input_t) -> output_t:- data, output = data- # Let's be on the safe side here, and do the reduction in 64 bit- # output = data.to(torch.float64).sum().to(torch.float32)-- n_elem = data.numel()- C_BLOCK_SIZE = 1024- c_grid_size = triton.cdiv(n_elem, C_BLOCK_SIZE)- BLOCK_SIZE = triton.cdiv(C_BLOCK_SIZE, 1)- grid_size = int(c_grid_size)- output = torch.zeros([grid_size]).cuda()- output_result = torch.zeros([1]).cuda()- grid = lambda meta: (triton.cdiv(n_elem, meta["BLOCK_SIZE"]), )- # start = time.time_ns()- sum_kernel[grid](data, output, n_elem, BLOCK_SIZE=BLOCK_SIZE)- # end = time.time_ns()- # total1 = (end - start)- # start = time.time_ns()- # result = output.sum()- new_block_size = triton.next_power_of_2(grid_size)- grid = lambda meta: (triton.cdiv(new_block_size, meta["BLOCK_SIZE"]), )- sum_kernel[grid](output, output_result, grid_size, BLOCK_SIZE=new_block_size)- # end = time.time_ns()- # total2 = (end - start)-- # per = total2/(total1 + total2)- # print(n_elem," -> ", per)- # calculate sum of output+ #!POPCORN leaderboard vectorsum_v2+ #!POPCORN gpus A100++ # This is a submission template for popcorn leaderboard 'vectorsum_v2'.+ # Your task is as follows:+ # > Implement a vector sum reduction kernel. This kernel computes the sum of all elements in the input tensor.+ # >+ # > Input: A tensor of shape `(N,)` with values from a normal distribution with mean 0 and variance 1.+ # > Output: A scalar value equal to the sum of all elements in the input tensor.+ # The deadline for this leaderboard is 2025-12-30 00:00:00+00:00++ # You can automatically route this file to specific GPUs by adding a line+ # `#!POPCORN gpus <GPUs>` to the header of this file.+ # Happy hacking!++ from task import input_t, output_t+ from utils import DeterministicContext+ import torch+ import triton+ import triton.language as tl+ import time+++ @triton.jit+ def sum_kernel(+ i_ptr, # input ptr+ o_ptr, # output ptr+ n_elem, # total number of elements in input pointer+ BLOCK_SIZE: tl.constexpr+ ):+ pid = tl.program_id(0)++ offsets = (pid * BLOCK_SIZE) + tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_elem++ in_v = tl.load(i_ptr + offsets, mask, other=0.0)+ # sum+ sum_result = tl.sum(in_v)+ o_offset = pid * 1+ tl.store(o_ptr + o_offset, sum_result)++++ def custom_kernel(data: input_t) -> output_t:+ data, output = data+ # Let's be on the safe side here, and do the reduction in 64 bit+ # output = data.to(torch.float64).sum().to(torch.float32)++ n_elem = data.numel()+ C_BLOCK_SIZE = 1024+ c_grid_size = triton.cdiv(n_elem, C_BLOCK_SIZE)+ BLOCK_SIZE = triton.cdiv(C_BLOCK_SIZE, 1)+ grid_size = int(c_grid_size)+ output = torch.zeros([grid_size]).cuda()+ output_result = torch.zeros([1]).cuda()+ grid = lambda meta: (triton.cdiv(n_elem, meta["BLOCK_SIZE"]), )+ # start = time.time_ns()+ sum_kernel[grid](data, output, n_elem, BLOCK_SIZE=BLOCK_SIZE)+ # end = time.time_ns()+ # total1 = (end - start)+ # start = time.time_ns()+ # result = output.sum()+ new_block_size = triton.next_power_of_2(grid_size)+ grid = lambda meta: (triton.cdiv(new_block_size, meta["BLOCK_SIZE"]), )+ sum_kernel[grid](output, output_result, grid_size, BLOCK_SIZE=new_block_size)+ # end = time.time_ns()+ # total2 = (end - start)++ # per = total2/(total1 + total2)+ # print(n_elem," -> ", per)+ # calculate sum of outputreturn output_result[0]No newline at end of file
scrolls · 144 diff lines total
Best evidence level for this revision: reported
JSON