Skip to content
KernelIndex
Search⌘K

submission 549854

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 99 lines, June 9 Researcher Reciprocity License v1.0.

gpumode_submit_290vic5t.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-549854?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA H100
523.9µs
#6 of 44
2026-03-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:bd52da8b128c148a260e8e289dbd75740b6122a7965531d21c49c48766cc4c56
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 8num_warps=8,

Kernel source

gpumode_submit_290vic5t.py99 lines
# kernel.py
"""
Triton vector-add kernel (float16) for 2D tensors.

Fused stages (single pass):
  1) tl.load(A) + tl.load(B)
  2) elementwise add in-kernel
  3) tl.store(C)

No unfused fallback is needed because the whole pipeline is a single elementwise op.
"""

import torch
import triton
import triton.language as tl


@triton.jit
def _vec_add_kernel(a_ptr, b_ptr, c_ptr, n_elements,
                    BLOCK_SIZE: tl.constexpr):
    pid = tl.program_id(axis=0)
    offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offs < n_elements

    a = tl.load(a_ptr + offs, mask=mask, other=0.0)
    b = tl.load(b_ptr + offs, mask=mask, other=0.0)
    c = a + b
    tl.store(c_ptr + offs, c, mask=mask)


def kernel_function(*args):
    """
    Wrapper that validates inputs, allocates output if needed, and launches Triton.

    Supported call patterns (as used by the tests):
      - kernel_function(A, B) -> out
      - kernel_function(A, B, C) -> out (writes into C)
      - kernel_function((A, B)) -> out
      - kernel_function((A, B, C)) -> out (writes into C)
    """
    # Unpack tuple-style inputs
    if len(args) == 1 and isinstance(args[0], (tuple, list)):
        args = tuple(args[0])

    if len(args) not in (2, 3):
        raise TypeError(f"kernel_function expected 2 or 3 arguments, got {len(args)}")

    A, B = args[0], args[1]
    C = args[2] if len(args) == 3 else None

    if not (isinstance(A, torch.Tensor) and isinstance(B, torch.Tensor)):
        raise TypeError("A and B must be torch.Tensor")
    if A.device.type != "cuda" or B.device.type != "cuda":
        raise ValueError("A and B must be CUDA tensors")
    if A.dtype != torch.float16 or B.dtype != torch.float16:
        raise ValueError("A and B must be torch.float16")
    if A.shape != B.shape:
        raise ValueError(f"Shape mismatch: A.shape={tuple(A.shape)} vs B.shape={tuple(B.shape)}")
    if not A.is_contiguous() or not B.is_contiguous():
        raise ValueError("A and B must be contiguous")

    if C is None:
        C = torch.empty_like(A)
    else:
        if not isinstance(C, torch.Tensor):
            raise TypeError("C must be a torch.Tensor when provided")
        if C.device != A.device:
            raise ValueError("C must be on the same device as A")
        if C.dtype != torch.float16:
            raise ValueError("C must be torch.float16")
        if C.shape != A.shape:
            raise ValueError("C must have the same shape as A")
        if not C.is_contiguous():
            raise ValueError("C must be contiguous")

    n_elements = A.numel()
    BLOCK_SIZE = 1024
    grid = (triton.cdiv(n_elements, BLOCK_SIZE),)

    _vec_add_kernel[grid](
        A, B, C,
        n_elements,
        BLOCK_SIZE=BLOCK_SIZE,
        num_warps=8,
    )
    return C

import inspect
def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)
    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)

import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 99 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 549651.

+ # kernel.py
+ """
+ Triton vector-add kernel (float16) for 2D tensors.
+
+ Fused stages (single pass):
+ 1) tl.load(A) + tl.load(B)
+ 2) elementwise add in-kernel
+ 3) tl.store(C)
+
+ No unfused fallback is needed because the whole pipeline is a single elementwise op.
+ """
+
+ import torch
import triton
import triton.language as tl
- import torch
@triton.jit
- def _vector_add_kernel(ptr_a, ptr_b, ptr_c, n_elements, BLOCK_SIZE: tl.constexpr):
- """Triton kernel for elementwise addition of two float16 tensors."""
- pid = tl.program_id(0)
- block_start = pid * BLOCK_SIZE
- offsets = block_start + tl.arange(0, BLOCK_SIZE)
- mask = offsets < n_elements
+ def _vec_add_kernel(a_ptr, b_ptr, c_ptr, n_elements,
+ BLOCK_SIZE: tl.constexpr):
+ pid = tl.program_id(axis=0)
+ offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = offs < n_elements
- a = tl.load(ptr_a + offsets, mask=mask)
- b = tl.load(ptr_b + offsets, mask=mask)
-
+ a = tl.load(a_ptr + offs, mask=mask, other=0.0)
+ b = tl.load(b_ptr + offs, mask=mask, other=0.0)
c = a + b
+ tl.store(c_ptr + offs, c, mask=mask)
- tl.store(ptr_c + offsets, c, mask=mask)
+ def kernel_function(*args):
+ """
+ Wrapper that validates inputs, allocates output if needed, and launches Triton.
- def kernel_function(A: torch.Tensor, B: torch.Tensor, C: torch.Tensor) -> torch.Tensor:
- """Wrapper for float16 vector addition: C = A + B.
+ Supported call patterns (as used by the tests):
+ - kernel_function(A, B) -> out
+ - kernel_function(A, B, C) -> out (writes into C)
+ - kernel_function((A, B)) -> out
+ - kernel_function((A, B, C)) -> out (writes into C)
+ """
+ # Unpack tuple-style inputs
+ if len(args) == 1 and isinstance(args[0], (tuple, list)):
+ args = tuple(args[0])
- Fused stages: single elementwise add kernel (load A, load B, add, store C).
- No additional stages to fuse since this is a pure elementwise operation.
+ if len(args) not in (2, 3):
+ raise TypeError(f"kernel_function expected 2 or 3 arguments, got {len(args)}")
- Args:
- A: Input tensor of shape (N, N) and dtype float16.
- B: Input tensor of shape (N, N) and dtype float16.
- C: Output tensor of shape (N, N) and dtype float16 (written in-place).
+ A, B = args[0], args[1]
+ C = args[2] if len(args) == 3 else None
- Returns:
- C tensor with result of A + B.
- """
- assert A.is_cuda and B.is_cuda and C.is_cuda, "All tensors must be on CUDA"
- assert A.shape == B.shape == C.shape, "Shape mismatch"
+ if not (isinstance(A, torch.Tensor) and isinstance(B, torch.Tensor)):
+ raise TypeError("A and B must be torch.Tensor")
+ if A.device.type != "cuda" or B.device.type != "cuda":
+ raise ValueError("A and B must be CUDA tensors")
+ if A.dtype != torch.float16 or B.dtype != torch.float16:
+ raise ValueError("A and B must be torch.float16")
+ if A.shape != B.shape:
+ raise ValueError(f"Shape mismatch: A.shape={tuple(A.shape)} vs B.shape={tuple(B.shape)}")
+ if not A.is_contiguous() or not B.is_contiguous():
+ raise ValueError("A and B must be contiguous")
+ if C is None:
+ C = torch.empty_like(A)
+ else:
+ if not isinstance(C, torch.Tensor):
+ raise TypeError("C must be a torch.Tensor when provided")
+ if C.device != A.device:
+ raise ValueError("C must be on the same device as A")
+ if C.dtype != torch.float16:
+ raise ValueError("C must be torch.float16")
+ if C.shape != A.shape:
+ raise ValueError("C must have the same shape as A")
+ if not C.is_contiguous():
+ raise ValueError("C must be contiguous")
+
n_elements = A.numel()
BLOCK_SIZE = 1024
grid = (triton.cdiv(n_elements, BLOCK_SIZE),)
- _vector_add_kernel[grid](A, B, C, n_elements, BLOCK_SIZE)
-
+ _vec_add_kernel[grid](
+ A, B, C,
+ n_elements,
+ BLOCK_SIZE=BLOCK_SIZE,
+ num_warps=8,
+ )
return C
import inspect
scrolls · 114 diff lines total

Best evidence level for this revision: reported

JSON