submission 68127
cdtmc · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 85 lines, June 9 Researcher Reciprocity License v1.0.
vectorsum_cccl.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-68127?include=source"interfacepython
Compatibility
measured onNVIDIA L4
declared hardwareNVIDIA L4
architecturessm_89
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:6641f1979a953ce46f2995985759a384dadd45641a795d2f9ced628a712e1e68
license declaredunknown
license concludedunknown
authorscdtmc
imported2026-08-15
Kernel source
vectorsum_cccl.py85 lines
#!POPCORN leaderboard vectorsum_v2
import functools
try:
import cuda.parallel.experimental.algorithms as algorithms
except:
import os
import subprocess
gpu = "a100"
if not os.path.exists("cccl"):
subprocess.check_call(
["git", "clone", "https://github.com/NaderAlAwar/cccl.git"]
)
subprocess.check_call(["git", "checkout", "gpu-mode-submissions-a100"], cwd="cccl/")
subprocess.check_call(
["git", "pull", "origin", "gpu-mode-submissions-a100"], cwd="cccl/"
)
if gpu != "l4":
subprocess.check_call(
["git", "checkout", "6092a0bead3a297666a2aba37672f67a8a1a569b"],
cwd="cccl/python/cuda_parallel",
)
else:
subprocess.check_call(
["git", "checkout", "bc36bca5c76e3e9c6bf78d6906127e8549cb551a"],
cwd="cccl/python/cuda_parallel",
)
env = os.environ.copy()
env["CC"] = "gcc"
env["CXX"] = "g++"
env["CMAKE_ARGS"] = "-DCMAKE_CXX_STANDARD=20"
subprocess.check_call(
["pip", "install", "../cuda_cccl"], cwd="cccl/python/cuda_parallel", env=env
)
subprocess.check_call(
["pip", "install", ".[test]", "-v"], cwd="cccl/python/cuda_parallel", env=env
)
subprocess.check_call(
["pip", "install", "cupy-cuda12x"], cwd="cccl/python/cuda_parallel", env=env
)
import cupy as cp
import numpy as np
import cuda.parallel.experimental.algorithms as algorithms
import functools
def add_op(a, b):
return a + b
import torch
from task import input_t, output_t
@functools.cache
def initialize(num_items):
d_in = torch.tensor(num_items, dtype=torch.float32).cuda()
d_out = torch.tensor(0, dtype=torch.float32).cuda()
h_init = np.array([0], dtype="float32")
reducer = algorithms.nondeterministic_reduce_into(d_in, d_out, add_op, h_init)
temp_storage_size = reducer(None, d_in, d_out, num_items, h_init)
d_temp_storage = cp.empty(temp_storage_size, dtype=np.uint8)
# reducer.initialize_fast(d_temp_storage, d_in, d_out, num_items, h_init)
return d_temp_storage, d_out, h_init, reducer
def custom_kernel(data: input_t) -> output_t:
data, _ = data
num_items = data.shape[0]
d_temp_storage, d_out, h_init, reducer = initialize(num_items)
reducer(d_temp_storage, data, d_out, num_items, h_init)
# _, d_out, _, reducer = initialize(num_items)
# reducer.call_fast(data)
return d_out
scrolls · 85 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON