submission 68267
cdtmc · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 59 lines, June 9 Researcher Reciprocity License v1.0.
vectoradd_cccl.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-68267?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:d6fe4778cbbbefd1b34bb190a5cdb4ec157ea011f255395bda8509d82bd35e2a
license declaredunknown
license concludedunknown
authorscdtmc
imported2026-08-15
Kernel source
vectoradd_cccl.py59 lines
#!POPCORN leaderboard vectoradd_v2
#!POPCORN gpus A100 H100 B200 L4
try:
import cuda.parallel.experimental.algorithms as algorithms
except:
import os
import subprocess
if not os.path.exists("cccl"):
subprocess.check_call(
["git", "clone", "https://github.com/NaderAlAwar/cccl.git"]
)
subprocess.check_call(["git", "checkout", "gpu-mode-submissions-a100"], cwd="cccl")
subprocess.check_call(
["git", "checkout", "49b7297dbe3abddbc25f937b132b8e6e16202100"], cwd="cccl"
)
env = os.environ.copy()
env["CC"] = "gcc"
env["CXX"] = "g++"
env["CMAKE_ARGS"] = "-DCMAKE_CXX_STANDARD=20"
subprocess.check_call(
["pip", "install", "../cuda_cccl"], cwd="cccl/python/cuda_parallel", env=env
)
subprocess.check_call(
["pip", "install", "."], cwd="cccl/python/cuda_parallel", env=env
)
import functools
import cuda.parallel.experimental.algorithms as algorithms
import torch
from task import input_t, output_t
d_in1 = torch.tensor([1], dtype=torch.float16).cuda()
d_in2 = torch.tensor([1], dtype=torch.float16).cuda()
def op(a, b):
return a + b
transform = algorithms.binary_transform(d_in1, d_in2, d_in2, op)
@functools.cache
def initialize(shape):
return torch.empty(shape, dtype=torch.float16).cuda(), shape[0] * shape[1]
def custom_kernel(data: input_t) -> output_t:
d_in1, d_in2, _ = data
d_out, numel = initialize(d_in1.shape)
transform(d_in1, d_in2, d_out, numel)
return d_out
scrolls · 59 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 66948.
#!POPCORN leaderboard vectoradd_v2+ #!POPCORN gpus A100 H100 B200 L4+ try:+ import cuda.parallel.experimental.algorithms as algorithms+ except:+ import os+ import subprocess++ if not os.path.exists("cccl"):+ subprocess.check_call(+ ["git", "clone", "https://github.com/NaderAlAwar/cccl.git"]+ )++ subprocess.check_call(["git", "checkout", "gpu-mode-submissions-a100"], cwd="cccl")+ subprocess.check_call(+ ["git", "checkout", "49b7297dbe3abddbc25f937b132b8e6e16202100"], cwd="cccl"+ )++ env = os.environ.copy()+ env["CC"] = "gcc"+ env["CXX"] = "g++"+ env["CMAKE_ARGS"] = "-DCMAKE_CXX_STANDARD=20"++ subprocess.check_call(+ ["pip", "install", "../cuda_cccl"], cwd="cccl/python/cuda_parallel", env=env+ )+ subprocess.check_call(+ ["pip", "install", "."], cwd="cccl/python/cuda_parallel", env=env+ )++ import functools+ import cuda.parallel.experimental.algorithms as algorithmsimport torch- import triton- import triton.language as tl+from task import input_t, output_t- # Autotune configs: same as before (wide sweep)- AUTOTUNE_CONFIGS = [- triton.Config({"BLOCK_SIZE": 128, "NUM_WARPS": 1, "NUM_STAGES": 2}),- triton.Config({"BLOCK_SIZE": 256, "NUM_WARPS": 1, "NUM_STAGES": 2}),- triton.Config({"BLOCK_SIZE": 512, "NUM_WARPS": 2, "NUM_STAGES": 2}),- triton.Config({"BLOCK_SIZE": 1024, "NUM_WARPS": 4, "NUM_STAGES": 2}),- triton.Config({"BLOCK_SIZE": 2048, "NUM_WARPS": 4, "NUM_STAGES": 3}),- triton.Config({"BLOCK_SIZE": 4096, "NUM_WARPS": 8, "NUM_STAGES": 3}),- triton.Config({"BLOCK_SIZE": 8192, "NUM_WARPS": 8, "NUM_STAGES": 4}),- triton.Config({"BLOCK_SIZE": 16384, "NUM_WARPS": 8, "NUM_STAGES": 4}),- ]+ d_in1 = torch.tensor([1], dtype=torch.float16).cuda()+ d_in2 = torch.tensor([1], dtype=torch.float16).cuda()- @triton.autotune(configs=AUTOTUNE_CONFIGS, key=["size"])- @triton.jit- def add_kernel(- A_ptr,- B_ptr,- C_ptr,- size,- BLOCK_SIZE: tl.constexpr,- NUM_WARPS: tl.constexpr,- NUM_STAGES: tl.constexpr,- ):- pid = tl.program_id(0)- offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)- mask = offsets < size- a = tl.load(A_ptr + offsets, mask=mask)- b = tl.load(B_ptr + offsets, mask=mask)- tl.store(C_ptr + offsets, a + b, mask=mask)+ def op(a, b):+ return a + b- def custom_kernel(data: input_t, deterministic: bool = False) -> output_t:- """- Wrapper:- - ensures contiguity,- - chooses BLOCK_SIZE from the autotune candidates (so grid matches compile-time BLOCK_SIZE),- - constructs grid with that BLOCK_SIZE,- - launches autotuned kernel.- If deterministic=True, use a fixed config (no autotune variability) for consistent benchmarking.- """- A, B, C = data- A = A.contiguous()- B = B.contiguous()- C = C.contiguous()- assert A.numel() == B.numel() == C.numel(), "inputs must have same numel"+ transform = algorithms.binary_transform(d_in1, d_in2, d_in2, op)- size = A.numel()- # Grid computed using the same BLOCK_SIZE we pass to the kernel — keeps coverage exact.- grid = lambda meta: (triton.cdiv(size, meta["BLOCK_SIZE"]),)+ @functools.cache+ def initialize(shape):+ return torch.empty(shape, dtype=torch.float16).cuda(), shape[0] * shape[1]- # Prefer letting the autotuner pick NUM_WARPS/NUM_STAGES for this BLOCK_SIZE.- # By passing BLOCK_SIZE here we ensure the grid is consistent with the compile-time BLOCK_SIZE.- add_kernel[grid](A, B, C, size)- return C+ def custom_kernel(data: input_t) -> output_t:+ d_in1, d_in2, _ = data+ d_out, numel = initialize(d_in1.shape)+ transform(d_in1, d_in2, d_out, numel)+ return d_out
scrolls · 109 diff lines total
Best evidence level for this revision: reported
JSON