Skip to content
KernelIndex
Search⌘K

submission 67723

.godelmachine · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 72 lines, June 9 Researcher Reciprocity License v1.0.

sumission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-67723?include=source"
interfacepython
Compatibility
measured onNVIDIA L4
declared hardwareNVIDIA L4
architecturessm_89
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA L4
1.09ms
#23 of 26
2025-11-07

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:8f4e8186fc28bae7421c674907959c935046ceb9975536e7a912def67d485bad
license declaredunknown
license concludedunknown
authors.godelmachine
imported2026-08-15

Kernel source

sumission.py72 lines
#!POPCORN leaderboard vectorsum_v2
#!POPCORN gpus A100

# This is a submission template for popcorn leaderboard 'vectorsum_v2'.
# Your task is as follows:
# > Implement a vector sum reduction kernel. This kernel computes the sum of all elements in the input tensor.
# >
# > Input: A tensor of shape `(N,)` with values from a normal distribution with mean 0 and variance 1.
# > Output: A scalar value equal to the sum of all elements in the input tensor.
# The deadline for this leaderboard is 2025-12-30 00:00:00+00:00

# You can automatically route this file to specific GPUs by adding a line
# `#!POPCORN gpus <GPUs>` to the header of this file.
# Happy hacking!

from task import input_t, output_t
from utils import DeterministicContext
import torch
import triton
import triton.language as tl
import time


@triton.jit
def sum_kernel(
    i_ptr, # input ptr
    o_ptr, # output ptr
    n_elem, # total number of elements in input pointer
    BLOCK_SIZE: tl.constexpr
):
    pid = tl.program_id(0)

    offsets = (pid * BLOCK_SIZE) + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_elem

    in_v = tl.load(i_ptr + offsets, mask, other=0.0)
    # sum
    sum_result = tl.sum(in_v)
    o_offset = pid * 1
    tl.store(o_ptr + o_offset, sum_result)



def custom_kernel(data: input_t) -> output_t:
    data, output = data
    # Let's be on the safe side here, and do the reduction in 64 bit
    # output = data.to(torch.float64).sum().to(torch.float32)

    n_elem = data.numel()
    C_BLOCK_SIZE = 1024
    c_grid_size = triton.cdiv(n_elem, C_BLOCK_SIZE)
    BLOCK_SIZE = triton.cdiv(C_BLOCK_SIZE, 1)
    grid_size = int(c_grid_size)
    output = torch.zeros([grid_size]).cuda()
    output_result = torch.zeros([1]).cuda()
    grid = lambda meta: (triton.cdiv(n_elem, meta["BLOCK_SIZE"]), )
    # start = time.time_ns()
    sum_kernel[grid](data, output, n_elem, BLOCK_SIZE=BLOCK_SIZE)
    # end = time.time_ns()
    # total1 = (end - start)
    # start = time.time_ns()
    # result = output.sum()
    new_block_size = triton.next_power_of_2(grid_size)
    grid = lambda meta: (triton.cdiv(new_block_size, meta["BLOCK_SIZE"]), )
    sum_kernel[grid](output, output_result, grid_size, BLOCK_SIZE=new_block_size)
    # end = time.time_ns()
    # total2 = (end - start)

    # per = total2/(total1 + total2)
    # print(n_elem," -> ", per)
    # calculate sum of output
    return output_result[0]
scrolls · 72 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 67722.

- #!POPCORN leaderboard vectorsum_v2
- #!POPCORN gpus B200
-
- # This is a submission template for popcorn leaderboard 'vectorsum_v2'.
- # Your task is as follows:
- # > Implement a vector sum reduction kernel. This kernel computes the sum of all elements in the input tensor.
- # >
- # > Input: A tensor of shape `(N,)` with values from a normal distribution with mean 0 and variance 1.
- # > Output: A scalar value equal to the sum of all elements in the input tensor.
- # The deadline for this leaderboard is 2025-12-30 00:00:00+00:00
-
- # You can automatically route this file to specific GPUs by adding a line
- # `#!POPCORN gpus <GPUs>` to the header of this file.
- # Happy hacking!
-
- from task import input_t, output_t
- from utils import DeterministicContext
- import torch
- import triton
- import triton.language as tl
- import time
-
-
- @triton.jit
- def sum_kernel(
- i_ptr, # input ptr
- o_ptr, # output ptr
- n_elem, # total number of elements in input pointer
- BLOCK_SIZE: tl.constexpr
- ):
- pid = tl.program_id(0)
-
- offsets = (pid * BLOCK_SIZE) + tl.arange(0, BLOCK_SIZE)
- mask = offsets < n_elem
-
- in_v = tl.load(i_ptr + offsets, mask, other=0.0)
- # sum
- sum_result = tl.sum(in_v)
- o_offset = pid * 1
- tl.store(o_ptr + o_offset, sum_result)
-
-
-
- def custom_kernel(data: input_t) -> output_t:
- data, output = data
- # Let's be on the safe side here, and do the reduction in 64 bit
- # output = data.to(torch.float64).sum().to(torch.float32)
-
- n_elem = data.numel()
- C_BLOCK_SIZE = 1024
- c_grid_size = triton.cdiv(n_elem, C_BLOCK_SIZE)
- BLOCK_SIZE = triton.cdiv(C_BLOCK_SIZE, 1)
- grid_size = int(c_grid_size)
- output = torch.zeros([grid_size]).cuda()
- output_result = torch.zeros([1]).cuda()
- grid = lambda meta: (triton.cdiv(n_elem, meta["BLOCK_SIZE"]), )
- # start = time.time_ns()
- sum_kernel[grid](data, output, n_elem, BLOCK_SIZE=BLOCK_SIZE)
- # end = time.time_ns()
- # total1 = (end - start)
- # start = time.time_ns()
- # result = output.sum()
- new_block_size = triton.next_power_of_2(grid_size)
- grid = lambda meta: (triton.cdiv(new_block_size, meta["BLOCK_SIZE"]), )
- sum_kernel[grid](output, output_result, grid_size, BLOCK_SIZE=new_block_size)
- # end = time.time_ns()
- # total2 = (end - start)
-
- # per = total2/(total1 + total2)
- # print(n_elem," -> ", per)
- # calculate sum of output
+ #!POPCORN leaderboard vectorsum_v2
+ #!POPCORN gpus A100
+
+ # This is a submission template for popcorn leaderboard 'vectorsum_v2'.
+ # Your task is as follows:
+ # > Implement a vector sum reduction kernel. This kernel computes the sum of all elements in the input tensor.
+ # >
+ # > Input: A tensor of shape `(N,)` with values from a normal distribution with mean 0 and variance 1.
+ # > Output: A scalar value equal to the sum of all elements in the input tensor.
+ # The deadline for this leaderboard is 2025-12-30 00:00:00+00:00
+
+ # You can automatically route this file to specific GPUs by adding a line
+ # `#!POPCORN gpus <GPUs>` to the header of this file.
+ # Happy hacking!
+
+ from task import input_t, output_t
+ from utils import DeterministicContext
+ import torch
+ import triton
+ import triton.language as tl
+ import time
+
+
+ @triton.jit
+ def sum_kernel(
+ i_ptr, # input ptr
+ o_ptr, # output ptr
+ n_elem, # total number of elements in input pointer
+ BLOCK_SIZE: tl.constexpr
+ ):
+ pid = tl.program_id(0)
+
+ offsets = (pid * BLOCK_SIZE) + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < n_elem
+
+ in_v = tl.load(i_ptr + offsets, mask, other=0.0)
+ # sum
+ sum_result = tl.sum(in_v)
+ o_offset = pid * 1
+ tl.store(o_ptr + o_offset, sum_result)
+
+
+
+ def custom_kernel(data: input_t) -> output_t:
+ data, output = data
+ # Let's be on the safe side here, and do the reduction in 64 bit
+ # output = data.to(torch.float64).sum().to(torch.float32)
+
+ n_elem = data.numel()
+ C_BLOCK_SIZE = 1024
+ c_grid_size = triton.cdiv(n_elem, C_BLOCK_SIZE)
+ BLOCK_SIZE = triton.cdiv(C_BLOCK_SIZE, 1)
+ grid_size = int(c_grid_size)
+ output = torch.zeros([grid_size]).cuda()
+ output_result = torch.zeros([1]).cuda()
+ grid = lambda meta: (triton.cdiv(n_elem, meta["BLOCK_SIZE"]), )
+ # start = time.time_ns()
+ sum_kernel[grid](data, output, n_elem, BLOCK_SIZE=BLOCK_SIZE)
+ # end = time.time_ns()
+ # total1 = (end - start)
+ # start = time.time_ns()
+ # result = output.sum()
+ new_block_size = triton.next_power_of_2(grid_size)
+ grid = lambda meta: (triton.cdiv(new_block_size, meta["BLOCK_SIZE"]), )
+ sum_kernel[grid](output, output_result, grid_size, BLOCK_SIZE=new_block_size)
+ # end = time.time_ns()
+ # total2 = (end - start)
+
+ # per = total2/(total1 + total2)
+ # print(n_elem," -> ", per)
+ # calculate sum of output
return output_result[0]
No newline at end of file
scrolls · 144 diff lines total

Best evidence level for this revision: reported

JSON