Skip to content
KernelIndex
Search⌘K

submission 68267

cdtmc · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 59 lines, June 9 Researcher Reciprocity License v1.0.

vectoradd_cccl.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-68267?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA B200
237.7µs
#39 of 66
2025-11-08

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:d6fe4778cbbbefd1b34bb190a5cdb4ec157ea011f255395bda8509d82bd35e2a
license declaredunknown
license concludedunknown
authorscdtmc
imported2026-08-15

Kernel source

vectoradd_cccl.py59 lines
#!POPCORN leaderboard vectoradd_v2
#!POPCORN gpus A100 H100 B200 L4

try:
    import cuda.parallel.experimental.algorithms as algorithms
except:
    import os
    import subprocess

    if not os.path.exists("cccl"):
        subprocess.check_call(
            ["git", "clone", "https://github.com/NaderAlAwar/cccl.git"]
        )

    subprocess.check_call(["git", "checkout", "gpu-mode-submissions-a100"], cwd="cccl")
    subprocess.check_call(
        ["git", "checkout", "49b7297dbe3abddbc25f937b132b8e6e16202100"], cwd="cccl"
    )

    env = os.environ.copy()
    env["CC"] = "gcc"
    env["CXX"] = "g++"
    env["CMAKE_ARGS"] = "-DCMAKE_CXX_STANDARD=20"

    subprocess.check_call(
        ["pip", "install", "../cuda_cccl"], cwd="cccl/python/cuda_parallel", env=env
    )
    subprocess.check_call(
        ["pip", "install", "."], cwd="cccl/python/cuda_parallel", env=env
    )

import functools
import cuda.parallel.experimental.algorithms as algorithms
import torch

from task import input_t, output_t

d_in1 = torch.tensor([1], dtype=torch.float16).cuda()
d_in2 = torch.tensor([1], dtype=torch.float16).cuda()


def op(a, b):
    return a + b


transform = algorithms.binary_transform(d_in1, d_in2, d_in2, op)


@functools.cache
def initialize(shape):
    return torch.empty(shape, dtype=torch.float16).cuda(), shape[0] * shape[1]


def custom_kernel(data: input_t) -> output_t:
    d_in1, d_in2, _ = data
    d_out, numel = initialize(d_in1.shape)
    transform(d_in1, d_in2, d_out, numel)
    return d_out
scrolls · 59 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 66948.

#!POPCORN leaderboard vectoradd_v2
+ #!POPCORN gpus A100 H100 B200 L4
+ try:
+ import cuda.parallel.experimental.algorithms as algorithms
+ except:
+ import os
+ import subprocess
+
+ if not os.path.exists("cccl"):
+ subprocess.check_call(
+ ["git", "clone", "https://github.com/NaderAlAwar/cccl.git"]
+ )
+
+ subprocess.check_call(["git", "checkout", "gpu-mode-submissions-a100"], cwd="cccl")
+ subprocess.check_call(
+ ["git", "checkout", "49b7297dbe3abddbc25f937b132b8e6e16202100"], cwd="cccl"
+ )
+
+ env = os.environ.copy()
+ env["CC"] = "gcc"
+ env["CXX"] = "g++"
+ env["CMAKE_ARGS"] = "-DCMAKE_CXX_STANDARD=20"
+
+ subprocess.check_call(
+ ["pip", "install", "../cuda_cccl"], cwd="cccl/python/cuda_parallel", env=env
+ )
+ subprocess.check_call(
+ ["pip", "install", "."], cwd="cccl/python/cuda_parallel", env=env
+ )
+
+ import functools
+ import cuda.parallel.experimental.algorithms as algorithms
import torch
- import triton
- import triton.language as tl
+
from task import input_t, output_t
- # Autotune configs: same as before (wide sweep)
- AUTOTUNE_CONFIGS = [
- triton.Config({"BLOCK_SIZE": 128, "NUM_WARPS": 1, "NUM_STAGES": 2}),
- triton.Config({"BLOCK_SIZE": 256, "NUM_WARPS": 1, "NUM_STAGES": 2}),
- triton.Config({"BLOCK_SIZE": 512, "NUM_WARPS": 2, "NUM_STAGES": 2}),
- triton.Config({"BLOCK_SIZE": 1024, "NUM_WARPS": 4, "NUM_STAGES": 2}),
- triton.Config({"BLOCK_SIZE": 2048, "NUM_WARPS": 4, "NUM_STAGES": 3}),
- triton.Config({"BLOCK_SIZE": 4096, "NUM_WARPS": 8, "NUM_STAGES": 3}),
- triton.Config({"BLOCK_SIZE": 8192, "NUM_WARPS": 8, "NUM_STAGES": 4}),
- triton.Config({"BLOCK_SIZE": 16384, "NUM_WARPS": 8, "NUM_STAGES": 4}),
- ]
+ d_in1 = torch.tensor([1], dtype=torch.float16).cuda()
+ d_in2 = torch.tensor([1], dtype=torch.float16).cuda()
- @triton.autotune(configs=AUTOTUNE_CONFIGS, key=["size"])
- @triton.jit
- def add_kernel(
- A_ptr,
- B_ptr,
- C_ptr,
- size,
- BLOCK_SIZE: tl.constexpr,
- NUM_WARPS: tl.constexpr,
- NUM_STAGES: tl.constexpr,
- ):
- pid = tl.program_id(0)
- offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
- mask = offsets < size
- a = tl.load(A_ptr + offsets, mask=mask)
- b = tl.load(B_ptr + offsets, mask=mask)
- tl.store(C_ptr + offsets, a + b, mask=mask)
+ def op(a, b):
+ return a + b
- def custom_kernel(data: input_t, deterministic: bool = False) -> output_t:
- """
- Wrapper:
- - ensures contiguity,
- - chooses BLOCK_SIZE from the autotune candidates (so grid matches compile-time BLOCK_SIZE),
- - constructs grid with that BLOCK_SIZE,
- - launches autotuned kernel.
- If deterministic=True, use a fixed config (no autotune variability) for consistent benchmarking.
- """
- A, B, C = data
- A = A.contiguous()
- B = B.contiguous()
- C = C.contiguous()
- assert A.numel() == B.numel() == C.numel(), "inputs must have same numel"
+ transform = algorithms.binary_transform(d_in1, d_in2, d_in2, op)
- size = A.numel()
- # Grid computed using the same BLOCK_SIZE we pass to the kernel — keeps coverage exact.
- grid = lambda meta: (triton.cdiv(size, meta["BLOCK_SIZE"]),)
+ @functools.cache
+ def initialize(shape):
+ return torch.empty(shape, dtype=torch.float16).cuda(), shape[0] * shape[1]
- # Prefer letting the autotuner pick NUM_WARPS/NUM_STAGES for this BLOCK_SIZE.
- # By passing BLOCK_SIZE here we ensure the grid is consistent with the compile-time BLOCK_SIZE.
- add_kernel[grid](A, B, C, size)
- return C
+ def custom_kernel(data: input_t) -> output_t:
+ d_in1, d_in2, _ = data
+ d_out, numel = initialize(d_in1.shape)
+ transform(d_in1, d_in2, d_out, numel)
+ return d_out
scrolls · 109 diff lines total

Best evidence level for this revision: reported

JSON