submission 649002
JNJYan · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 91 lines, June 9 Researcher Reciprocity License v1.0.
l4.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-649002?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:a9dc258a57b624603f2cfed58bd9477293ae9afc9a790aff0196245e97660a67
license declaredunknown
license concludedunknown
authorsJNJYan
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
_triton.Config({"BLOCK": 128}, num_warps=2),num-warps = 2
_triton.Config({"BLOCK": 128}, num_warps=2),Kernel source
l4.py91 lines
#!POPCORN leaderboard vectoradd_v2
#!POPCORN gpu L4
from task import input_t, output_t
import triton as _triton
import triton.language as _tl
_VECADD_CONFIGS = [
_triton.Config({"BLOCK": 128}, num_warps=2),
_triton.Config({"BLOCK": 256}, num_warps=2),
_triton.Config({"BLOCK": 256}, num_warps=4),
_triton.Config({"BLOCK": 512}, num_warps=4),
_triton.Config({"BLOCK": 1024}, num_warps=4),
_triton.Config({"BLOCK": 1024}, num_warps=8),
]
@_triton.autotune(configs=_VECADD_CONFIGS, key=["n_elements"])
@_triton.jit
def _vecadd_kernel(x_ptr, y_ptr, out_ptr, n_elements, BLOCK: _tl.constexpr):
pid = _tl.program_id(axis=0)
offsets = pid * BLOCK + _tl.arange(0, BLOCK)
mask = offsets < n_elements
x = _tl.load(x_ptr + offsets, mask=mask, other=0)
y = _tl.load(y_ptr + offsets, mask=mask, other=0)
_tl.store(out_ptr + offsets, x + y, mask=mask)
def custom_kernel(data: input_t) -> output_t:
A, B, output = data
# if not (hasattr(A, "is_cuda") and A.is_cuda and hasattr(B, "is_cuda") and B.is_cuda):
# output[...] = A + B
# return output
# if A.dtype != B.dtype or A.dtype != output.dtype:
# output[...] = A + B
# return output
# if A.numel() != B.numel() or A.numel() != output.numel():
# output[...] = A + B
# return output
# if not (A.is_contiguous() and B.is_contiguous() and output.is_contiguous()):
# output[...] = A + B
# return output
A_1d = A.view(-1)
B_1d = B.view(-1)
O_1d = output.view(-1)
n_elements = O_1d.numel()
grid = lambda meta: (_triton.cdiv(n_elements, meta["BLOCK"]),)
_vecadd_kernel[grid](A_1d, B_1d, O_1d, n_elements)
return output
_WARMUP_SHAPES = (1024, 2048, 4096, 8192, 16384)
def _run_autotune_warmup() -> None:
try:
import torch
except Exception:
return
if not (hasattr(torch, "cuda") and torch.cuda.is_available()):
return
try:
device = torch.device("cuda")
dtype = torch.float32
for n_elements in _WARMUP_SHAPES:
a = torch.empty((n_elements,), device=device, dtype=dtype)
b = torch.empty((n_elements,), device=device, dtype=dtype)
o = torch.empty((n_elements,), device=device, dtype=dtype)
grid = lambda meta, n=n_elements: (_triton.cdiv(n, meta["BLOCK"]),)
_vecadd_kernel[grid](a, b, o, n_elements)
torch.cuda.synchronize()
except Exception:
return
_run_autotune_warmup()
scrolls · 91 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 648869.
⋯ 3 unchanged linesfrom task import input_t, output_t- try:- import triton as _triton- import triton.language as _tl+ import triton as _triton+ import triton.language as _tl- _TRITON_AVAILABLE = True+ _VECADD_CONFIGS = [+ _triton.Config({"BLOCK": 128}, num_warps=2),+ _triton.Config({"BLOCK": 256}, num_warps=2),+ _triton.Config({"BLOCK": 256}, num_warps=4),+ _triton.Config({"BLOCK": 512}, num_warps=4),+ _triton.Config({"BLOCK": 1024}, num_warps=4),+ _triton.Config({"BLOCK": 1024}, num_warps=8),+ ]- @_triton.jit- def _vecadd_kernel(x_ptr, y_ptr, out_ptr, n_elements, BLOCK: _tl.constexpr):- pid = _tl.program_id(axis=0)- offsets = pid * BLOCK + _tl.arange(0, BLOCK)- mask = offsets < n_elements- x = _tl.load(x_ptr + offsets, mask=mask, other=0)- y = _tl.load(y_ptr + offsets, mask=mask, other=0)- _tl.store(out_ptr + offsets, x + y, mask=mask)- except Exception:- _TRITON_AVAILABLE = False- _triton = None- _vecadd_kernel = None+ @_triton.autotune(configs=_VECADD_CONFIGS, key=["n_elements"])+ @_triton.jit+ def _vecadd_kernel(x_ptr, y_ptr, out_ptr, n_elements, BLOCK: _tl.constexpr):+ pid = _tl.program_id(axis=0)+ offsets = pid * BLOCK + _tl.arange(0, BLOCK)+ mask = offsets < n_elements+ x = _tl.load(x_ptr + offsets, mask=mask, other=0)+ y = _tl.load(y_ptr + offsets, mask=mask, other=0)+ _tl.store(out_ptr + offsets, x + y, mask=mask)+++def custom_kernel(data: input_t) -> output_t:A, B, output = data- if not _TRITON_AVAILABLE:- output[...] = A + B- return output- if not (hasattr(A, "is_cuda") and A.is_cuda and hasattr(B, "is_cuda") and B.is_cuda):- output[...] = A + B- return output+ # if not (hasattr(A, "is_cuda") and A.is_cuda and hasattr(B, "is_cuda") and B.is_cuda):+ # output[...] = A + B+ # return output- if A.dtype != B.dtype or A.dtype != output.dtype:- output[...] = A + B- return output+ # if A.dtype != B.dtype or A.dtype != output.dtype:+ # output[...] = A + B+ # return output- if A.numel() != B.numel() or A.numel() != output.numel():- output[...] = A + B- return output+ # if A.numel() != B.numel() or A.numel() != output.numel():+ # output[...] = A + B+ # return output- if not (A.is_contiguous() and B.is_contiguous() and output.is_contiguous()):- output[...] = A + B- return output+ # if not (A.is_contiguous() and B.is_contiguous() and output.is_contiguous()):+ # output[...] = A + B+ # return outputA_1d = A.view(-1)B_1d = B.view(-1)O_1d = output.view(-1)n_elements = O_1d.numel()- grid = (_triton.cdiv(n_elements, 1024),)- _vecadd_kernel[grid](A_1d, B_1d, O_1d, n_elements, BLOCK=1024, num_warps=8)++ grid = lambda meta: (_triton.cdiv(n_elements, meta["BLOCK"]),)+ _vecadd_kernel[grid](A_1d, B_1d, O_1d, n_elements)return output+++ _WARMUP_SHAPES = (1024, 2048, 4096, 8192, 16384)+++ def _run_autotune_warmup() -> None:+ try:+ import torch+ except Exception:+ return++ if not (hasattr(torch, "cuda") and torch.cuda.is_available()):+ return++ try:+ device = torch.device("cuda")+ dtype = torch.float32+ for n_elements in _WARMUP_SHAPES:+ a = torch.empty((n_elements,), device=device, dtype=dtype)+ b = torch.empty((n_elements,), device=device, dtype=dtype)+ o = torch.empty((n_elements,), device=device, dtype=dtype)++ grid = lambda meta, n=n_elements: (_triton.cdiv(n, meta["BLOCK"]),)+ _vecadd_kernel[grid](a, b, o, n_elements)++ torch.cuda.synchronize()+ except Exception:+ return+++ _run_autotune_warmup()
scrolls · 121 diff lines total
Best evidence level for this revision: reported
JSON