submission 66948
cdtmc · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 65 lines, June 9 Researcher Reciprocity License v1.0.
submissionv2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-66948?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:300469c5a09b28e468b10187feaee84cc8e89633ee30db8eb8a33a2ac67d50ab
license declaredunknown
license concludedunknown
authorscdtmc
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
triton.Config({"BLOCK_SIZE": 128, "NUM_WARPS": 1, "NUM_STAGES": 2}),Kernel source
submissionv2.py65 lines
#!POPCORN leaderboard vectoradd_v2
import torch
import triton
import triton.language as tl
from task import input_t, output_t
# Autotune configs: same as before (wide sweep)
AUTOTUNE_CONFIGS = [
triton.Config({"BLOCK_SIZE": 128, "NUM_WARPS": 1, "NUM_STAGES": 2}),
triton.Config({"BLOCK_SIZE": 256, "NUM_WARPS": 1, "NUM_STAGES": 2}),
triton.Config({"BLOCK_SIZE": 512, "NUM_WARPS": 2, "NUM_STAGES": 2}),
triton.Config({"BLOCK_SIZE": 1024, "NUM_WARPS": 4, "NUM_STAGES": 2}),
triton.Config({"BLOCK_SIZE": 2048, "NUM_WARPS": 4, "NUM_STAGES": 3}),
triton.Config({"BLOCK_SIZE": 4096, "NUM_WARPS": 8, "NUM_STAGES": 3}),
triton.Config({"BLOCK_SIZE": 8192, "NUM_WARPS": 8, "NUM_STAGES": 4}),
triton.Config({"BLOCK_SIZE": 16384, "NUM_WARPS": 8, "NUM_STAGES": 4}),
]
@triton.autotune(configs=AUTOTUNE_CONFIGS, key=["size"])
@triton.jit
def add_kernel(
A_ptr,
B_ptr,
C_ptr,
size,
BLOCK_SIZE: tl.constexpr,
NUM_WARPS: tl.constexpr,
NUM_STAGES: tl.constexpr,
):
pid = tl.program_id(0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offsets < size
a = tl.load(A_ptr + offsets, mask=mask)
b = tl.load(B_ptr + offsets, mask=mask)
tl.store(C_ptr + offsets, a + b, mask=mask)
def custom_kernel(data: input_t, deterministic: bool = False) -> output_t:
"""
Wrapper:
- ensures contiguity,
- chooses BLOCK_SIZE from the autotune candidates (so grid matches compile-time BLOCK_SIZE),
- constructs grid with that BLOCK_SIZE,
- launches autotuned kernel.
If deterministic=True, use a fixed config (no autotune variability) for consistent benchmarking.
"""
A, B, C = data
A = A.contiguous()
B = B.contiguous()
C = C.contiguous()
assert A.numel() == B.numel() == C.numel(), "inputs must have same numel"
size = A.numel()
# Grid computed using the same BLOCK_SIZE we pass to the kernel — keeps coverage exact.
grid = lambda meta: (triton.cdiv(size, meta["BLOCK_SIZE"]),)
# Prefer letting the autotuner pick NUM_WARPS/NUM_STAGES for this BLOCK_SIZE.
# By passing BLOCK_SIZE here we ensure the grid is consistent with the compile-time BLOCK_SIZE.
add_kernel[grid](A, B, C, size)
return C
scrolls · 65 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 66933.
⋯ 4 unchanged linesimport triton.language as tlfrom task import input_t, output_t+ # Autotune configs: same as before (wide sweep)+ AUTOTUNE_CONFIGS = [+ triton.Config({"BLOCK_SIZE": 128, "NUM_WARPS": 1, "NUM_STAGES": 2}),+ triton.Config({"BLOCK_SIZE": 256, "NUM_WARPS": 1, "NUM_STAGES": 2}),+ triton.Config({"BLOCK_SIZE": 512, "NUM_WARPS": 2, "NUM_STAGES": 2}),+ triton.Config({"BLOCK_SIZE": 1024, "NUM_WARPS": 4, "NUM_STAGES": 2}),+ triton.Config({"BLOCK_SIZE": 2048, "NUM_WARPS": 4, "NUM_STAGES": 3}),+ triton.Config({"BLOCK_SIZE": 4096, "NUM_WARPS": 8, "NUM_STAGES": 3}),+ triton.Config({"BLOCK_SIZE": 8192, "NUM_WARPS": 8, "NUM_STAGES": 4}),+ triton.Config({"BLOCK_SIZE": 16384, "NUM_WARPS": 8, "NUM_STAGES": 4}),+ ]++ @triton.autotune(configs=AUTOTUNE_CONFIGS, key=["size"])@triton.jit- def add_kernel(A_ptr, B_ptr, C_ptr, size, BLOCK_SIZE: tl.constexpr):+ def add_kernel(+ A_ptr,+ B_ptr,+ C_ptr,+ size,+ BLOCK_SIZE: tl.constexpr,+ NUM_WARPS: tl.constexpr,+ NUM_STAGES: tl.constexpr,+ ):pid = tl.program_id(0)offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)mask = offsets < size- A = tl.load(A_ptr + offsets, mask=mask)- B = tl.load(B_ptr + offsets, mask=mask)- tl.store(C_ptr + offsets, A + B, mask=mask)+ a = tl.load(A_ptr + offsets, mask=mask)+ b = tl.load(B_ptr + offsets, mask=mask)+ tl.store(C_ptr + offsets, a + b, mask=mask)- def custom_kernel(data: input_t) -> output_t:+ def custom_kernel(data: input_t, deterministic: bool = False) -> output_t:+ """+ Wrapper:+ - ensures contiguity,+ - chooses BLOCK_SIZE from the autotune candidates (so grid matches compile-time BLOCK_SIZE),+ - constructs grid with that BLOCK_SIZE,+ - launches autotuned kernel.+ If deterministic=True, use a fixed config (no autotune variability) for consistent benchmarking.+ """A, B, C = data- A, B, C = A.contiguous(), B.contiguous(), C.contiguous()+ A = A.contiguous()+ B = B.contiguous()+ C = C.contiguous()+ assert A.numel() == B.numel() == C.numel(), "inputs must have same numel"+size = A.numel()- BLOCK_SIZE = 1024- grid = (triton.cdiv(size, BLOCK_SIZE),)- add_kernel[grid](A, B, C, size, BLOCK_SIZE=BLOCK_SIZE)++ # Grid computed using the same BLOCK_SIZE we pass to the kernel — keeps coverage exact.+ grid = lambda meta: (triton.cdiv(size, meta["BLOCK_SIZE"]),)++ # Prefer letting the autotuner pick NUM_WARPS/NUM_STAGES for this BLOCK_SIZE.+ # By passing BLOCK_SIZE here we ensure the grid is consistent with the compile-time BLOCK_SIZE.+ add_kernel[grid](A, B, C, size)+return C
scrolls · 70 diff lines total
Best evidence level for this revision: reported
JSON