Skip to content
KernelIndex
Search⌘K

submission 68127

cdtmc · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 85 lines, June 9 Researcher Reciprocity License v1.0.

vectorsum_cccl.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-68127?include=source"
interfacepython
Compatibility
measured onNVIDIA L4
declared hardwareNVIDIA L4
architecturessm_89
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA L4
949.6µs
#12 of 26
2025-11-08

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:6641f1979a953ce46f2995985759a384dadd45641a795d2f9ced628a712e1e68
license declaredunknown
license concludedunknown
authorscdtmc
imported2026-08-15

Kernel source

vectorsum_cccl.py85 lines
#!POPCORN leaderboard vectorsum_v2

import functools

try:
    import cuda.parallel.experimental.algorithms as algorithms
except:
    import os
    import subprocess

    gpu = "a100"

    if not os.path.exists("cccl"):
        subprocess.check_call(
            ["git", "clone", "https://github.com/NaderAlAwar/cccl.git"]
        )
    subprocess.check_call(["git", "checkout", "gpu-mode-submissions-a100"], cwd="cccl/")
    subprocess.check_call(
        ["git", "pull", "origin", "gpu-mode-submissions-a100"], cwd="cccl/"
    )
    if gpu != "l4":
        subprocess.check_call(
            ["git", "checkout", "6092a0bead3a297666a2aba37672f67a8a1a569b"],
            cwd="cccl/python/cuda_parallel",
        )
    else:
        subprocess.check_call(
            ["git", "checkout", "bc36bca5c76e3e9c6bf78d6906127e8549cb551a"],
            cwd="cccl/python/cuda_parallel",
        )

    env = os.environ.copy()
    env["CC"] = "gcc"
    env["CXX"] = "g++"
    env["CMAKE_ARGS"] = "-DCMAKE_CXX_STANDARD=20"

    subprocess.check_call(
        ["pip", "install", "../cuda_cccl"], cwd="cccl/python/cuda_parallel", env=env
    )
    subprocess.check_call(
        ["pip", "install", ".[test]", "-v"], cwd="cccl/python/cuda_parallel", env=env
    )
    subprocess.check_call(
        ["pip", "install", "cupy-cuda12x"], cwd="cccl/python/cuda_parallel", env=env
    )

import cupy as cp
import numpy as np
import cuda.parallel.experimental.algorithms as algorithms
import functools


def add_op(a, b):
    return a + b


import torch

from task import input_t, output_t


@functools.cache
def initialize(num_items):
    d_in = torch.tensor(num_items, dtype=torch.float32).cuda()
    d_out = torch.tensor(0, dtype=torch.float32).cuda()
    h_init = np.array([0], dtype="float32")
    reducer = algorithms.nondeterministic_reduce_into(d_in, d_out, add_op, h_init)
    temp_storage_size = reducer(None, d_in, d_out, num_items, h_init)

    d_temp_storage = cp.empty(temp_storage_size, dtype=np.uint8)
    # reducer.initialize_fast(d_temp_storage, d_in, d_out, num_items, h_init)

    return d_temp_storage, d_out, h_init, reducer


def custom_kernel(data: input_t) -> output_t:
    data, _ = data
    num_items = data.shape[0]
    d_temp_storage, d_out, h_init, reducer = initialize(num_items)
    reducer(d_temp_storage, data, d_out, num_items, h_init)
    # _, d_out, _, reducer = initialize(num_items)
    # reducer.call_fast(data)

    return d_out
scrolls · 85 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON