Skip to content
KernelIndex
Search⌘K

submission 780463

Cookie 🍪 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 157 lines, June 9 Researcher Reciprocity License v1.0.

vectoradd_v2_H100_claude-opus-4.5_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-780463?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA H100
525.5µs
#18 of 44
2026-04-29

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:26ac785a5aefdf1d9f77e8936e9bfb30dd1e2162808f1c1f9b0885ee54683d66
license declaredunknown
license concludedunknown
authorsCookie 🍪
imported2026-08-15

Kernel source

vectoradd_v2_H100_claude-opus-4.5_ka_submission.py157 lines
import torch
import triton
import triton.language as tl


@triton.jit
def _add_kernel(
    x0_ptr,      # Pointer to first input tensor
    x1_ptr,      # Pointer to second input tensor
    out_ptr,     # Pointer to output tensor
    n_elements,  # Total number of elements
    BLOCK_SIZE: tl.constexpr,  # Number of elements per block
):
    """
    Triton kernel for elementwise addition of two tensors.

    Each program instance handles BLOCK_SIZE elements.
    """
    # Get the program ID (which block we're processing)
    pid = tl.program_id(axis=0)

    # Calculate the starting offset for this block
    block_start = pid * BLOCK_SIZE

    # Create offsets for elements within this block
    offsets = block_start + tl.arange(0, BLOCK_SIZE)

    # Create mask to handle boundary conditions (last block may be partial)
    mask = offsets < n_elements

    # Load elements from both input tensors
    x0 = tl.load(x0_ptr + offsets, mask=mask, other=0.0)
    x1 = tl.load(x1_ptr + offsets, mask=mask, other=0.0)

    # Perform elementwise addition
    result = x0 + x1

    # Store the result
    tl.store(out_ptr + offsets, result, mask=mask)


def kernel_function(x0: torch.Tensor, x1: torch.Tensor, output: torch.Tensor = None) -> torch.Tensor:
    """
    Wrapper function for elementwise addition of two tensors using Triton.

    Fused operation: Single kernel performs load + add + store in one pass.
    No separate stages needed as this is a simple elementwise operation.

    Args:
        x0: First input tensor of shape [N, N], dtype float16
        x1: Second input tensor of shape [N, N], dtype float16
        output: Optional output tensor of shape [N, N], dtype float16

    Returns:
        Output tensor of shape [N, N], dtype float16, containing x0 + x1
    """
    # Validate inputs
    assert x0.is_cuda and x1.is_cuda, "Both tensors must be on CUDA device"
    assert x0.shape == x1.shape, "Input tensors must have the same shape"
    assert x0.dtype == x1.dtype, "Input tensors must have the same dtype"

    # Ensure tensors are contiguous for proper memory access
    x0 = x0.contiguous()
    x1 = x1.contiguous()

    # Allocate output tensor with same shape and dtype if not provided
    if output is None:
        output = torch.empty_like(x0)

    # Calculate total number of elements
    n_elements = x0.numel()

    # Choose block size (power of 2, common choice for good performance)
    BLOCK_SIZE = 1024

    # Calculate grid size (number of blocks needed)
    grid = (triton.cdiv(n_elements, BLOCK_SIZE),)

    # Launch the Triton kernel
    _add_kernel[grid](
        x0,          # First input pointer
        x1,          # Second input pointer
        output,      # Output pointer
        n_elements,  # Total elements
        BLOCK_SIZE,  # Block size (compile-time constant)
    )

    return output


def test_kernel():
    """
    Test the Triton kernel against PyTorch reference implementation.
    """
    test_cases = [
        {"seed": 4242, "size": 127},
        {"seed": 5236, "size": 128},
        {"seed": 1001, "size": 129},
        {"seed": 5531, "size": 256},
        {"seed": 9173, "size": 512},
    ]

    all_passed = True

    for test in test_cases:
        size = test["size"]
        seed = test["seed"]

        # Generate input tensors
        gen = torch.Generator(device="cuda")
        gen.manual_seed(seed)
        A = torch.randn(size, size, device="cuda", dtype=torch.float16, generator=gen).contiguous()
        B = torch.randn(size, size, device="cuda", dtype=torch.float16, generator=gen).contiguous()

        # Compute reference result using PyTorch
        ref_output = A + B

        # Compute result using Triton kernel
        triton_output = kernel_function(A, B)

        # Compare results
        if torch.allclose(triton_output, ref_output, rtol=2e-2, atol=2e-2):
            print(f"Test size={size}, seed={seed}: PASS")
        else:
            print(f"Test size={size}, seed={seed}: FAIL")
            max_diff = (triton_output - ref_output).abs().max().item()
            print(f"  Max difference: {max_diff}")
            all_passed = False

    if all_passed:
        print("PASS")
    else:
        print("FAIL")
        exit(1)


if __name__ == "__main__":
    test_kernel()


import inspect

def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)

    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)


# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"

scrolls · 157 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 780043.

Best evidence level for this revision: reported

JSON