Skip to content
KernelIndex
Search⌘K

submission 649002

JNJYan · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 91 lines, June 9 Researcher Reciprocity License v1.0.

l4.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-649002?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA B200
250.5µs
#53 of 66
2026-03-27

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:a9dc258a57b624603f2cfed58bd9477293ae9afc9a790aff0196245e97660a67
license declaredunknown
license concludedunknown
authorsJNJYan
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotune_triton.Config({"BLOCK": 128}, num_warps=2),
num-warps = 2_triton.Config({"BLOCK": 128}, num_warps=2),

Kernel source

l4.py91 lines
#!POPCORN leaderboard vectoradd_v2
#!POPCORN gpu L4


from task import input_t, output_t

import triton as _triton
import triton.language as _tl

_VECADD_CONFIGS = [
    _triton.Config({"BLOCK": 128}, num_warps=2),
    _triton.Config({"BLOCK": 256}, num_warps=2),
    _triton.Config({"BLOCK": 256}, num_warps=4),
    _triton.Config({"BLOCK": 512}, num_warps=4),
    _triton.Config({"BLOCK": 1024}, num_warps=4),
    _triton.Config({"BLOCK": 1024}, num_warps=8),
]

@_triton.autotune(configs=_VECADD_CONFIGS, key=["n_elements"])
@_triton.jit
def _vecadd_kernel(x_ptr, y_ptr, out_ptr, n_elements, BLOCK: _tl.constexpr):
    pid = _tl.program_id(axis=0)
    offsets = pid * BLOCK + _tl.arange(0, BLOCK)
    mask = offsets < n_elements
    x = _tl.load(x_ptr + offsets, mask=mask, other=0)
    y = _tl.load(y_ptr + offsets, mask=mask, other=0)
    _tl.store(out_ptr + offsets, x + y, mask=mask)





def custom_kernel(data: input_t) -> output_t:
    A, B, output = data

    # if not (hasattr(A, "is_cuda") and A.is_cuda and hasattr(B, "is_cuda") and B.is_cuda):
    #     output[...] = A + B
    #     return output

    # if A.dtype != B.dtype or A.dtype != output.dtype:
    #     output[...] = A + B
    #     return output

    # if A.numel() != B.numel() or A.numel() != output.numel():
    #     output[...] = A + B
    #     return output

    # if not (A.is_contiguous() and B.is_contiguous() and output.is_contiguous()):
    #     output[...] = A + B
    #     return output

    A_1d = A.view(-1)
    B_1d = B.view(-1)
    O_1d = output.view(-1)
    n_elements = O_1d.numel()

    grid = lambda meta: (_triton.cdiv(n_elements, meta["BLOCK"]),)
    _vecadd_kernel[grid](A_1d, B_1d, O_1d, n_elements)
    return output


_WARMUP_SHAPES = (1024, 2048, 4096, 8192, 16384)


def _run_autotune_warmup() -> None:
    try:
        import torch
    except Exception:
        return

    if not (hasattr(torch, "cuda") and torch.cuda.is_available()):
        return

    try:
        device = torch.device("cuda")
        dtype = torch.float32
        for n_elements in _WARMUP_SHAPES:
            a = torch.empty((n_elements,), device=device, dtype=dtype)
            b = torch.empty((n_elements,), device=device, dtype=dtype)
            o = torch.empty((n_elements,), device=device, dtype=dtype)

            grid = lambda meta, n=n_elements: (_triton.cdiv(n, meta["BLOCK"]),)
            _vecadd_kernel[grid](a, b, o, n_elements)

        torch.cuda.synchronize()
    except Exception:
        return


_run_autotune_warmup()
scrolls · 91 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 648869.

⋯ 3 unchanged lines
from task import input_t, output_t
- try:
- import triton as _triton
- import triton.language as _tl
+ import triton as _triton
+ import triton.language as _tl
- _TRITON_AVAILABLE = True
+ _VECADD_CONFIGS = [
+ _triton.Config({"BLOCK": 128}, num_warps=2),
+ _triton.Config({"BLOCK": 256}, num_warps=2),
+ _triton.Config({"BLOCK": 256}, num_warps=4),
+ _triton.Config({"BLOCK": 512}, num_warps=4),
+ _triton.Config({"BLOCK": 1024}, num_warps=4),
+ _triton.Config({"BLOCK": 1024}, num_warps=8),
+ ]
- @_triton.jit
- def _vecadd_kernel(x_ptr, y_ptr, out_ptr, n_elements, BLOCK: _tl.constexpr):
- pid = _tl.program_id(axis=0)
- offsets = pid * BLOCK + _tl.arange(0, BLOCK)
- mask = offsets < n_elements
- x = _tl.load(x_ptr + offsets, mask=mask, other=0)
- y = _tl.load(y_ptr + offsets, mask=mask, other=0)
- _tl.store(out_ptr + offsets, x + y, mask=mask)
- except Exception:
- _TRITON_AVAILABLE = False
- _triton = None
- _vecadd_kernel = None
+ @_triton.autotune(configs=_VECADD_CONFIGS, key=["n_elements"])
+ @_triton.jit
+ def _vecadd_kernel(x_ptr, y_ptr, out_ptr, n_elements, BLOCK: _tl.constexpr):
+ pid = _tl.program_id(axis=0)
+ offsets = pid * BLOCK + _tl.arange(0, BLOCK)
+ mask = offsets < n_elements
+ x = _tl.load(x_ptr + offsets, mask=mask, other=0)
+ y = _tl.load(y_ptr + offsets, mask=mask, other=0)
+ _tl.store(out_ptr + offsets, x + y, mask=mask)
+
+
+
def custom_kernel(data: input_t) -> output_t:
A, B, output = data
- if not _TRITON_AVAILABLE:
- output[...] = A + B
- return output
- if not (hasattr(A, "is_cuda") and A.is_cuda and hasattr(B, "is_cuda") and B.is_cuda):
- output[...] = A + B
- return output
+ # if not (hasattr(A, "is_cuda") and A.is_cuda and hasattr(B, "is_cuda") and B.is_cuda):
+ # output[...] = A + B
+ # return output
- if A.dtype != B.dtype or A.dtype != output.dtype:
- output[...] = A + B
- return output
+ # if A.dtype != B.dtype or A.dtype != output.dtype:
+ # output[...] = A + B
+ # return output
- if A.numel() != B.numel() or A.numel() != output.numel():
- output[...] = A + B
- return output
+ # if A.numel() != B.numel() or A.numel() != output.numel():
+ # output[...] = A + B
+ # return output
- if not (A.is_contiguous() and B.is_contiguous() and output.is_contiguous()):
- output[...] = A + B
- return output
+ # if not (A.is_contiguous() and B.is_contiguous() and output.is_contiguous()):
+ # output[...] = A + B
+ # return output
A_1d = A.view(-1)
B_1d = B.view(-1)
O_1d = output.view(-1)
n_elements = O_1d.numel()
- grid = (_triton.cdiv(n_elements, 1024),)
- _vecadd_kernel[grid](A_1d, B_1d, O_1d, n_elements, BLOCK=1024, num_warps=8)
+
+ grid = lambda meta: (_triton.cdiv(n_elements, meta["BLOCK"]),)
+ _vecadd_kernel[grid](A_1d, B_1d, O_1d, n_elements)
return output
+
+
+ _WARMUP_SHAPES = (1024, 2048, 4096, 8192, 16384)
+
+
+ def _run_autotune_warmup() -> None:
+ try:
+ import torch
+ except Exception:
+ return
+
+ if not (hasattr(torch, "cuda") and torch.cuda.is_available()):
+ return
+
+ try:
+ device = torch.device("cuda")
+ dtype = torch.float32
+ for n_elements in _WARMUP_SHAPES:
+ a = torch.empty((n_elements,), device=device, dtype=dtype)
+ b = torch.empty((n_elements,), device=device, dtype=dtype)
+ o = torch.empty((n_elements,), device=device, dtype=dtype)
+
+ grid = lambda meta, n=n_elements: (_triton.cdiv(n, meta["BLOCK"]),)
+ _vecadd_kernel[grid](a, b, o, n_elements)
+
+ torch.cuda.synchronize()
+ except Exception:
+ return
+
+
+ _run_autotune_warmup()
scrolls · 121 diff lines total

Best evidence level for this revision: reported

JSON