Skip to content
KernelIndex
Search⌘K

submission 549651

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 58 lines, June 9 Researcher Reciprocity License v1.0.

gpumode_submit_85yh0ic8.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectoradd-v2-549651?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp16

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
FP16 vector additionsuite of 5 cases
NVIDIA H100
523.9µs
#7 of 44
2026-03-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:71a496924f1efa490ab069d5008b014e35b9389cd7ae5b35b6f508cd6ab1ecef
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Kernel source

gpumode_submit_85yh0ic8.py58 lines
import triton
import triton.language as tl
import torch


@triton.jit
def _vector_add_kernel(ptr_a, ptr_b, ptr_c, n_elements, BLOCK_SIZE: tl.constexpr):
    """Triton kernel for elementwise addition of two float16 tensors."""
    pid = tl.program_id(0)
    block_start = pid * BLOCK_SIZE
    offsets = block_start + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_elements

    a = tl.load(ptr_a + offsets, mask=mask)
    b = tl.load(ptr_b + offsets, mask=mask)

    c = a + b

    tl.store(ptr_c + offsets, c, mask=mask)


def kernel_function(A: torch.Tensor, B: torch.Tensor, C: torch.Tensor) -> torch.Tensor:
    """Wrapper for float16 vector addition: C = A + B.

    Fused stages: single elementwise add kernel (load A, load B, add, store C).
    No additional stages to fuse since this is a pure elementwise operation.

    Args:
        A: Input tensor of shape (N, N) and dtype float16.
        B: Input tensor of shape (N, N) and dtype float16.
        C: Output tensor of shape (N, N) and dtype float16 (written in-place).

    Returns:
        C tensor with result of A + B.
    """
    assert A.is_cuda and B.is_cuda and C.is_cuda, "All tensors must be on CUDA"
    assert A.shape == B.shape == C.shape, "Shape mismatch"

    n_elements = A.numel()
    BLOCK_SIZE = 1024
    grid = (triton.cdiv(n_elements, BLOCK_SIZE),)

    _vector_add_kernel[grid](A, B, C, n_elements, BLOCK_SIZE)

    return C

import inspect
def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)
    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)

import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 58 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 490594.

⋯ 3 unchanged lines
@triton.jit
- def _vector_add_kernel(
- ptr_a, # Pointer to input tensor A
- ptr_b, # Pointer to input tensor B
- ptr_out, # Pointer to output tensor C
- n_elements, # Total number of elements
- BLOCK_SIZE: tl.constexpr, # Block size for parallelization
- ):
- """
- Triton kernel for element-wise addition of two tensors.
-
- Fused operation: C = A + B
-
- Each program instance processes BLOCK_SIZE elements.
- """
- # Get the program ID (which block we're processing)
- pid = tl.program_id(axis=0)
-
- # Calculate the starting offset for this block
+ def _vector_add_kernel(ptr_a, ptr_b, ptr_c, n_elements, BLOCK_SIZE: tl.constexpr):
+ """Triton kernel for elementwise addition of two float16 tensors."""
+ pid = tl.program_id(0)
block_start = pid * BLOCK_SIZE
-
- # Generate offsets for elements within this block
offsets = block_start + tl.arange(0, BLOCK_SIZE)
-
- # Create mask for boundary handling (handles non-power-of-2 sizes)
mask = offsets < n_elements
-
- # Load elements from tensor A with masking
- a = tl.load(ptr_a + offsets, mask=mask, other=0.0)
-
- # Load elements from tensor B with masking
- b = tl.load(ptr_b + offsets, mask=mask, other=0.0)
-
- # Perform element-wise addition using Triton operations
- result = a + b
-
- # Store the result to output tensor with masking
- tl.store(ptr_out + offsets, result, mask=mask)
+ a = tl.load(ptr_a + offsets, mask=mask)
+ b = tl.load(ptr_b + offsets, mask=mask)
+ c = a + b
+
+ tl.store(ptr_c + offsets, c, mask=mask)
+
+
def kernel_function(A: torch.Tensor, B: torch.Tensor, C: torch.Tensor) -> torch.Tensor:
- """
- Wrapper function for float16 vector addition kernel.
-
- Performs element-wise addition: C = A + B
-
- This is a single fused operation - no decomposition needed as it's
- already an atomic elementwise operation.
-
+ """Wrapper for float16 vector addition: C = A + B.
+
+ Fused stages: single elementwise add kernel (load A, load B, add, store C).
+ No additional stages to fuse since this is a pure elementwise operation.
+
Args:
- A: Input tensor of shape (N, N) and dtype float16
- B: Input tensor of shape (N, N) and dtype float16
- C: Output tensor of shape (N, N) and dtype float16 (pre-allocated)
-
+ A: Input tensor of shape (N, N) and dtype float16.
+ B: Input tensor of shape (N, N) and dtype float16.
+ C: Output tensor of shape (N, N) and dtype float16 (written in-place).
+
Returns:
- C: The output tensor containing A + B
+ C tensor with result of A + B.
"""
- # Validate inputs
assert A.is_cuda and B.is_cuda and C.is_cuda, "All tensors must be on CUDA"
- assert A.dtype == torch.float16 and B.dtype == torch.float16, "Inputs must be float16"
- assert C.dtype == torch.float16, "Output must be float16"
- assert A.shape == B.shape == C.shape, "All tensors must have the same shape"
- assert A.is_contiguous() and B.is_contiguous() and C.is_contiguous(), "Tensors must be contiguous"
-
- # Calculate total number of elements
+ assert A.shape == B.shape == C.shape, "Shape mismatch"
+
n_elements = A.numel()
-
- # Choose block size (power of 2 for efficiency)
BLOCK_SIZE = 1024
-
- # Calculate grid dimensions (number of blocks needed)
grid = (triton.cdiv(n_elements, BLOCK_SIZE),)
-
- # Launch the Triton kernel
- _vector_add_kernel[grid](
- A, # Input tensor A
- B, # Input tensor B
- C, # Output tensor C
- n_elements, # Total elements to process
- BLOCK_SIZE, # Block size (compile-time constant)
- )
-
+
+ _vector_add_kernel[grid](A, B, C, n_elements, BLOCK_SIZE)
+
return C
import inspect
-
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
-
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
-
- # Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
-
scrolls · 128 diff lines total

Best evidence level for this revision: reported

JSON