Skip to content
KernelIndex
Search⌘K

submission 68915

cdtmc · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 39 lines, June 9 Researcher Reciprocity License v1.0.

vectoradd.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-68915?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA B200
236.2µs
#23 of 66
2025-11-10

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:d3cdad9d172a965322ffde2c0059674c91e9a5a7afedfee3b93dd960276b5a53
license declaredunknown
license concludedunknown
authorscdtmc
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotune@triton.autotune(
num-warps = 2triton.Config(kwargs={"BLOCK_SIZE": 256}, num_warps=2, num_stages=1),
stages = 1triton.Config(kwargs={"BLOCK_SIZE": 256}, num_warps=2, num_stages=1),

Kernel source

vectoradd.py39 lines
import triton
import triton.language as tl


@triton.autotune(
    configs=[
        # ------- Tier 1: good for L4 / small tensors -------
        triton.Config(kwargs={"BLOCK_SIZE": 256}, num_warps=2, num_stages=1),
        triton.Config(kwargs={"BLOCK_SIZE": 512}, num_warps=4, num_stages=1),
        # ------- Tier 2: A100 general sweet spot -------
        triton.Config(kwargs={"BLOCK_SIZE": 1024}, num_warps=4, num_stages=2),
        triton.Config(kwargs={"BLOCK_SIZE": 2048}, num_warps=8, num_stages=2),
        # ------- Tier 3: H100 / Blackwell optimized -------
        #   Large blocks + pipelined loads
        triton.Config(kwargs={"BLOCK_SIZE": 4096}, num_warps=8, num_stages=3),
        triton.Config(kwargs={"BLOCK_SIZE": 8192}, num_warps=8, num_stages=4),
    ],
    key=["n_elements"],  # autotune per problem size
)
@triton.jit
def _kernel(x_ptr, y_ptr, output_ptr, n_elements, BLOCK_SIZE: tl.constexpr):
    pid = tl.program_id(axis=0)
    block_start = pid * BLOCK_SIZE
    offsets = block_start + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_elements

    x = tl.load(x_ptr + offsets, mask=mask)
    y = tl.load(y_ptr + offsets, mask=mask)

    tl.store(output_ptr + offsets, x + y, mask=mask)


def custom_kernel(data):
    x, y, output = data
    n_elements = output.numel()
    grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]),)
    _kernel[grid](x, y, output, n_elements)
    return output
scrolls · 39 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 68267.

- #!POPCORN leaderboard vectoradd_v2
- #!POPCORN gpus A100 H100 B200 L4
+ import triton
+ import triton.language as tl
- try:
- import cuda.parallel.experimental.algorithms as algorithms
- except:
- import os
- import subprocess
- if not os.path.exists("cccl"):
- subprocess.check_call(
- ["git", "clone", "https://github.com/NaderAlAwar/cccl.git"]
- )
+ @triton.autotune(
+ configs=[
+ # ------- Tier 1: good for L4 / small tensors -------
+ triton.Config(kwargs={"BLOCK_SIZE": 256}, num_warps=2, num_stages=1),
+ triton.Config(kwargs={"BLOCK_SIZE": 512}, num_warps=4, num_stages=1),
+ # ------- Tier 2: A100 general sweet spot -------
+ triton.Config(kwargs={"BLOCK_SIZE": 1024}, num_warps=4, num_stages=2),
+ triton.Config(kwargs={"BLOCK_SIZE": 2048}, num_warps=8, num_stages=2),
+ # ------- Tier 3: H100 / Blackwell optimized -------
+ # Large blocks + pipelined loads
+ triton.Config(kwargs={"BLOCK_SIZE": 4096}, num_warps=8, num_stages=3),
+ triton.Config(kwargs={"BLOCK_SIZE": 8192}, num_warps=8, num_stages=4),
+ ],
+ key=["n_elements"], # autotune per problem size
+ )
+ @triton.jit
+ def _kernel(x_ptr, y_ptr, output_ptr, n_elements, BLOCK_SIZE: tl.constexpr):
+ pid = tl.program_id(axis=0)
+ block_start = pid * BLOCK_SIZE
+ offsets = block_start + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < n_elements
- subprocess.check_call(["git", "checkout", "gpu-mode-submissions-a100"], cwd="cccl")
- subprocess.check_call(
- ["git", "checkout", "49b7297dbe3abddbc25f937b132b8e6e16202100"], cwd="cccl"
- )
+ x = tl.load(x_ptr + offsets, mask=mask)
+ y = tl.load(y_ptr + offsets, mask=mask)
- env = os.environ.copy()
- env["CC"] = "gcc"
- env["CXX"] = "g++"
- env["CMAKE_ARGS"] = "-DCMAKE_CXX_STANDARD=20"
+ tl.store(output_ptr + offsets, x + y, mask=mask)
- subprocess.check_call(
- ["pip", "install", "../cuda_cccl"], cwd="cccl/python/cuda_parallel", env=env
- )
- subprocess.check_call(
- ["pip", "install", "."], cwd="cccl/python/cuda_parallel", env=env
- )
- import functools
- import cuda.parallel.experimental.algorithms as algorithms
- import torch
-
- from task import input_t, output_t
-
- d_in1 = torch.tensor([1], dtype=torch.float16).cuda()
- d_in2 = torch.tensor([1], dtype=torch.float16).cuda()
-
-
- def op(a, b):
- return a + b
-
-
- transform = algorithms.binary_transform(d_in1, d_in2, d_in2, op)
-
-
- @functools.cache
- def initialize(shape):
- return torch.empty(shape, dtype=torch.float16).cuda(), shape[0] * shape[1]
-
-
- def custom_kernel(data: input_t) -> output_t:
- d_in1, d_in2, _ = data
- d_out, numel = initialize(d_in1.shape)
- transform(d_in1, d_in2, d_out, numel)
- return d_out
+ def custom_kernel(data):
+ x, y, output = data
+ n_elements = output.numel()
+ grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]),)
+ _kernel[grid](x, y, output, n_elements)
+ return output
scrolls · 90 diff lines total

Best evidence level for this revision: reported

JSON