submission 68915
cdtmc · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 39 lines, June 9 Researcher Reciprocity License v1.0.
vectoradd.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-68915?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:d3cdad9d172a965322ffde2c0059674c91e9a5a7afedfee3b93dd960276b5a53
license declaredunknown
license concludedunknown
authorscdtmc
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
@triton.autotune(num-warps = 2
triton.Config(kwargs={"BLOCK_SIZE": 256}, num_warps=2, num_stages=1),stages = 1
triton.Config(kwargs={"BLOCK_SIZE": 256}, num_warps=2, num_stages=1),Kernel source
vectoradd.py39 lines
import triton
import triton.language as tl
@triton.autotune(
configs=[
# ------- Tier 1: good for L4 / small tensors -------
triton.Config(kwargs={"BLOCK_SIZE": 256}, num_warps=2, num_stages=1),
triton.Config(kwargs={"BLOCK_SIZE": 512}, num_warps=4, num_stages=1),
# ------- Tier 2: A100 general sweet spot -------
triton.Config(kwargs={"BLOCK_SIZE": 1024}, num_warps=4, num_stages=2),
triton.Config(kwargs={"BLOCK_SIZE": 2048}, num_warps=8, num_stages=2),
# ------- Tier 3: H100 / Blackwell optimized -------
# Large blocks + pipelined loads
triton.Config(kwargs={"BLOCK_SIZE": 4096}, num_warps=8, num_stages=3),
triton.Config(kwargs={"BLOCK_SIZE": 8192}, num_warps=8, num_stages=4),
],
key=["n_elements"], # autotune per problem size
)
@triton.jit
def _kernel(x_ptr, y_ptr, output_ptr, n_elements, BLOCK_SIZE: tl.constexpr):
pid = tl.program_id(axis=0)
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
x = tl.load(x_ptr + offsets, mask=mask)
y = tl.load(y_ptr + offsets, mask=mask)
tl.store(output_ptr + offsets, x + y, mask=mask)
def custom_kernel(data):
x, y, output = data
n_elements = output.numel()
grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]),)
_kernel[grid](x, y, output, n_elements)
return output
scrolls · 39 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 68267.
- #!POPCORN leaderboard vectoradd_v2- #!POPCORN gpus A100 H100 B200 L4+ import triton+ import triton.language as tl- try:- import cuda.parallel.experimental.algorithms as algorithms- except:- import os- import subprocess- if not os.path.exists("cccl"):- subprocess.check_call(- ["git", "clone", "https://github.com/NaderAlAwar/cccl.git"]- )+ @triton.autotune(+ configs=[+ # ------- Tier 1: good for L4 / small tensors -------+ triton.Config(kwargs={"BLOCK_SIZE": 256}, num_warps=2, num_stages=1),+ triton.Config(kwargs={"BLOCK_SIZE": 512}, num_warps=4, num_stages=1),+ # ------- Tier 2: A100 general sweet spot -------+ triton.Config(kwargs={"BLOCK_SIZE": 1024}, num_warps=4, num_stages=2),+ triton.Config(kwargs={"BLOCK_SIZE": 2048}, num_warps=8, num_stages=2),+ # ------- Tier 3: H100 / Blackwell optimized -------+ # Large blocks + pipelined loads+ triton.Config(kwargs={"BLOCK_SIZE": 4096}, num_warps=8, num_stages=3),+ triton.Config(kwargs={"BLOCK_SIZE": 8192}, num_warps=8, num_stages=4),+ ],+ key=["n_elements"], # autotune per problem size+ )+ @triton.jit+ def _kernel(x_ptr, y_ptr, output_ptr, n_elements, BLOCK_SIZE: tl.constexpr):+ pid = tl.program_id(axis=0)+ block_start = pid * BLOCK_SIZE+ offsets = block_start + tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_elements- subprocess.check_call(["git", "checkout", "gpu-mode-submissions-a100"], cwd="cccl")- subprocess.check_call(- ["git", "checkout", "49b7297dbe3abddbc25f937b132b8e6e16202100"], cwd="cccl"- )+ x = tl.load(x_ptr + offsets, mask=mask)+ y = tl.load(y_ptr + offsets, mask=mask)- env = os.environ.copy()- env["CC"] = "gcc"- env["CXX"] = "g++"- env["CMAKE_ARGS"] = "-DCMAKE_CXX_STANDARD=20"+ tl.store(output_ptr + offsets, x + y, mask=mask)- subprocess.check_call(- ["pip", "install", "../cuda_cccl"], cwd="cccl/python/cuda_parallel", env=env- )- subprocess.check_call(- ["pip", "install", "."], cwd="cccl/python/cuda_parallel", env=env- )- import functools- import cuda.parallel.experimental.algorithms as algorithms- import torch-- from task import input_t, output_t-- d_in1 = torch.tensor([1], dtype=torch.float16).cuda()- d_in2 = torch.tensor([1], dtype=torch.float16).cuda()--- def op(a, b):- return a + b--- transform = algorithms.binary_transform(d_in1, d_in2, d_in2, op)--- @functools.cache- def initialize(shape):- return torch.empty(shape, dtype=torch.float16).cuda(), shape[0] * shape[1]--- def custom_kernel(data: input_t) -> output_t:- d_in1, d_in2, _ = data- d_out, numel = initialize(d_in1.shape)- transform(d_in1, d_in2, d_out, numel)- return d_out+ def custom_kernel(data):+ x, y, output = data+ n_elements = output.numel()+ grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]),)+ _kernel[grid](x, y, output, n_elements)+ return output
scrolls · 90 diff lines total
Best evidence level for this revision: reported
JSON