Skip to content
KernelIndex
Search⌘K

submission 66948

cdtmc · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 65 lines, June 9 Researcher Reciprocity License v1.0.

submissionv2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-66948?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA B200
239.8µs
#44 of 66
2025-11-05

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:300469c5a09b28e468b10187feaee84cc8e89633ee30db8eb8a33a2ac67d50ab
license declaredunknown
license concludedunknown
authorscdtmc
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotunetriton.Config({"BLOCK_SIZE": 128, "NUM_WARPS": 1, "NUM_STAGES": 2}),

Kernel source

submissionv2.py65 lines
#!POPCORN leaderboard vectoradd_v2

import torch
import triton
import triton.language as tl
from task import input_t, output_t

# Autotune configs: same as before (wide sweep)
AUTOTUNE_CONFIGS = [
    triton.Config({"BLOCK_SIZE": 128, "NUM_WARPS": 1, "NUM_STAGES": 2}),
    triton.Config({"BLOCK_SIZE": 256, "NUM_WARPS": 1, "NUM_STAGES": 2}),
    triton.Config({"BLOCK_SIZE": 512, "NUM_WARPS": 2, "NUM_STAGES": 2}),
    triton.Config({"BLOCK_SIZE": 1024, "NUM_WARPS": 4, "NUM_STAGES": 2}),
    triton.Config({"BLOCK_SIZE": 2048, "NUM_WARPS": 4, "NUM_STAGES": 3}),
    triton.Config({"BLOCK_SIZE": 4096, "NUM_WARPS": 8, "NUM_STAGES": 3}),
    triton.Config({"BLOCK_SIZE": 8192, "NUM_WARPS": 8, "NUM_STAGES": 4}),
    triton.Config({"BLOCK_SIZE": 16384, "NUM_WARPS": 8, "NUM_STAGES": 4}),
]


@triton.autotune(configs=AUTOTUNE_CONFIGS, key=["size"])
@triton.jit
def add_kernel(
    A_ptr,
    B_ptr,
    C_ptr,
    size,
    BLOCK_SIZE: tl.constexpr,
    NUM_WARPS: tl.constexpr,
    NUM_STAGES: tl.constexpr,
):
    pid = tl.program_id(0)
    offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offsets < size
    a = tl.load(A_ptr + offsets, mask=mask)
    b = tl.load(B_ptr + offsets, mask=mask)
    tl.store(C_ptr + offsets, a + b, mask=mask)


def custom_kernel(data: input_t, deterministic: bool = False) -> output_t:
    """
    Wrapper:
      - ensures contiguity,
      - chooses BLOCK_SIZE from the autotune candidates (so grid matches compile-time BLOCK_SIZE),
      - constructs grid with that BLOCK_SIZE,
      - launches autotuned kernel.
    If deterministic=True, use a fixed config (no autotune variability) for consistent benchmarking.
    """
    A, B, C = data
    A = A.contiguous()
    B = B.contiguous()
    C = C.contiguous()
    assert A.numel() == B.numel() == C.numel(), "inputs must have same numel"

    size = A.numel()

    # Grid computed using the same BLOCK_SIZE we pass to the kernel — keeps coverage exact.
    grid = lambda meta: (triton.cdiv(size, meta["BLOCK_SIZE"]),)

    # Prefer letting the autotuner pick NUM_WARPS/NUM_STAGES for this BLOCK_SIZE.
    # By passing BLOCK_SIZE here we ensure the grid is consistent with the compile-time BLOCK_SIZE.
    add_kernel[grid](A, B, C, size)

    return C
scrolls · 65 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 66933.

⋯ 4 unchanged lines
import triton.language as tl
from task import input_t, output_t
+ # Autotune configs: same as before (wide sweep)
+ AUTOTUNE_CONFIGS = [
+ triton.Config({"BLOCK_SIZE": 128, "NUM_WARPS": 1, "NUM_STAGES": 2}),
+ triton.Config({"BLOCK_SIZE": 256, "NUM_WARPS": 1, "NUM_STAGES": 2}),
+ triton.Config({"BLOCK_SIZE": 512, "NUM_WARPS": 2, "NUM_STAGES": 2}),
+ triton.Config({"BLOCK_SIZE": 1024, "NUM_WARPS": 4, "NUM_STAGES": 2}),
+ triton.Config({"BLOCK_SIZE": 2048, "NUM_WARPS": 4, "NUM_STAGES": 3}),
+ triton.Config({"BLOCK_SIZE": 4096, "NUM_WARPS": 8, "NUM_STAGES": 3}),
+ triton.Config({"BLOCK_SIZE": 8192, "NUM_WARPS": 8, "NUM_STAGES": 4}),
+ triton.Config({"BLOCK_SIZE": 16384, "NUM_WARPS": 8, "NUM_STAGES": 4}),
+ ]
+
+ @triton.autotune(configs=AUTOTUNE_CONFIGS, key=["size"])
@triton.jit
- def add_kernel(A_ptr, B_ptr, C_ptr, size, BLOCK_SIZE: tl.constexpr):
+ def add_kernel(
+ A_ptr,
+ B_ptr,
+ C_ptr,
+ size,
+ BLOCK_SIZE: tl.constexpr,
+ NUM_WARPS: tl.constexpr,
+ NUM_STAGES: tl.constexpr,
+ ):
pid = tl.program_id(0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < size
- A = tl.load(A_ptr + offsets, mask=mask)
- B = tl.load(B_ptr + offsets, mask=mask)
- tl.store(C_ptr + offsets, A + B, mask=mask)
+ a = tl.load(A_ptr + offsets, mask=mask)
+ b = tl.load(B_ptr + offsets, mask=mask)
+ tl.store(C_ptr + offsets, a + b, mask=mask)
- def custom_kernel(data: input_t) -> output_t:
+ def custom_kernel(data: input_t, deterministic: bool = False) -> output_t:
+ """
+ Wrapper:
+ - ensures contiguity,
+ - chooses BLOCK_SIZE from the autotune candidates (so grid matches compile-time BLOCK_SIZE),
+ - constructs grid with that BLOCK_SIZE,
+ - launches autotuned kernel.
+ If deterministic=True, use a fixed config (no autotune variability) for consistent benchmarking.
+ """
A, B, C = data
- A, B, C = A.contiguous(), B.contiguous(), C.contiguous()
+ A = A.contiguous()
+ B = B.contiguous()
+ C = C.contiguous()
+ assert A.numel() == B.numel() == C.numel(), "inputs must have same numel"
+
size = A.numel()
- BLOCK_SIZE = 1024
- grid = (triton.cdiv(size, BLOCK_SIZE),)
- add_kernel[grid](A, B, C, size, BLOCK_SIZE=BLOCK_SIZE)
+
+ # Grid computed using the same BLOCK_SIZE we pass to the kernel — keeps coverage exact.
+ grid = lambda meta: (triton.cdiv(size, meta["BLOCK_SIZE"]),)
+
+ # Prefer letting the autotuner pick NUM_WARPS/NUM_STAGES for this BLOCK_SIZE.
+ # By passing BLOCK_SIZE here we ensure the grid is consistent with the compile-time BLOCK_SIZE.
+ add_kernel[grid](A, B, C, size)
+
return C
scrolls · 70 diff lines total

Best evidence level for this revision: reported

JSON