Skip to content
KernelIndex
Search⌘K

submission 489489

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 177 lines, June 9 Researcher Reciprocity License v1.0.

conv2d_py_H100_claude-opus-4.5_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-489489?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
2D convolutionsuite of 5 cases
NVIDIA H100
1.07s
#34 of 35
2026-02-12

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:d0e888c92b0066dad67e54e9a9af0fab5e6e8cb53ace04e57703babdad5edde7
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Kernel source

conv2d_py_H100_claude-opus-4.5_ka_submission.py177 lines
import triton
import triton.language as tl
import torch


@triton.jit
def conv2d_kernel(
    input_ptr,      # Pointer to input tensor [B, C, H, W]
    kernel_ptr,     # Pointer to kernel tensor [C_out, C_in, KH, KW]
    output_ptr,     # Pointer to output tensor [B, C_out, OH, OW]
    # Dimensions
    batch: tl.constexpr,
    in_channels: tl.constexpr,
    out_channels: tl.constexpr,
    in_height: tl.constexpr,
    in_width: tl.constexpr,
    kernel_size: tl.constexpr,
    out_height: tl.constexpr,
    out_width: tl.constexpr,
    # Block sizes
    BLOCK_OH: tl.constexpr,
    BLOCK_OW: tl.constexpr,
):
    """
    Fused 2D convolution kernel.
    Each program computes a block of output pixels for one (batch, out_channel) pair.
    The convolution sum over in_channels and kernel spatial dimensions is computed inline.
    """
    # Program IDs
    pid_b = tl.program_id(0)  # batch index
    pid_oc = tl.program_id(1)  # output channel index
    pid_spatial = tl.program_id(2)  # spatial block index
    
    # Calculate spatial block position
    num_blocks_ow = tl.cdiv(out_width, BLOCK_OW)
    pid_oh = pid_spatial // num_blocks_ow
    pid_ow = pid_spatial % num_blocks_ow
    
    # Output pixel offsets within this block
    offs_oh = pid_oh * BLOCK_OH + tl.arange(0, BLOCK_OH)
    offs_ow = pid_ow * BLOCK_OW + tl.arange(0, BLOCK_OW)
    
    # Masks for valid output positions
    mask_oh = offs_oh < out_height
    mask_ow = offs_ow < out_width
    
    # Initialize accumulator for this block of output pixels
    acc = tl.zeros((BLOCK_OH, BLOCK_OW), dtype=tl.float32)
    
    # Loop over input channels
    for ic in range(in_channels):
        # Loop over kernel height
        for kh in range(kernel_size):
            # Loop over kernel width
            for kw in range(kernel_size):
                # Load kernel weight for this (oc, ic, kh, kw)
                # Kernel layout: [out_channels, in_channels, kernel_size, kernel_size]
                kernel_offset = (pid_oc * in_channels * kernel_size * kernel_size +
                                ic * kernel_size * kernel_size +
                                kh * kernel_size + kw)
                w = tl.load(kernel_ptr + kernel_offset)
                
                # Input positions: ih = oh + kh, iw = ow + kw
                # Input layout: [batch, channels, height, width]
                # offs_ih = offs_oh + kh (shape: BLOCK_OH)
                # offs_iw = offs_ow + kw (shape: BLOCK_OW)
                
                # Load input values for this block
                # We need to load input[pid_b, ic, offs_oh + kh, offs_ow + kw]
                input_base = (pid_b * in_channels * in_height * in_width +
                             ic * in_height * in_width)
                
                # Calculate input offsets for each output position
                # input_offset[i, j] = input_base + (offs_oh[i] + kh) * in_width + (offs_ow[j] + kw)
                offs_ih = offs_oh + kh  # [BLOCK_OH]
                offs_iw = offs_ow + kw  # [BLOCK_OW]
                
                # Create 2D offset grid
                input_offsets = input_base + offs_ih[:, None] * in_width + offs_iw[None, :]
                
                # Create mask (input positions are always valid since we only compute valid output positions)
                mask = mask_oh[:, None] & mask_ow[None, :]
                
                # Load input values
                x = tl.load(input_ptr + input_offsets, mask=mask, other=0.0)
                
                # Accumulate: acc += x * w
                acc += x * w
    
    # Store output
    # Output layout: [batch, out_channels, out_height, out_width]
    output_base = (pid_b * out_channels * out_height * out_width +
                   pid_oc * out_height * out_width)
    output_offsets = output_base + offs_oh[:, None] * out_width + offs_ow[None, :]
    output_mask = mask_oh[:, None] & mask_ow[None, :]
    
    tl.store(output_ptr + output_offsets, acc, mask=output_mask)


def kernel_function(input_tensor: torch.Tensor, kernel: torch.Tensor, output_tensor: torch.Tensor) -> torch.Tensor:
    """
    Wrapper for 2D convolution using Triton.
    
    This is a fused implementation that computes the entire convolution in a single kernel:
    - For each output position, accumulates over all input channels and kernel spatial positions
    - No separate im2col or matrix multiplication steps
    
    Args:
        input_tensor: Input tensor of shape [batch, in_channels, height, width]
        kernel: Convolution kernel of shape [out_channels, in_channels, kH, kW]
        output_tensor: Pre-allocated output tensor of shape [batch, out_channels, oH, oW]
        
    Returns:
        output_tensor filled with convolution result
    """
    # Extract dimensions
    batch, in_channels, in_height, in_width = input_tensor.shape
    out_channels, _, kernel_h, kernel_w = kernel.shape
    
    # For this problem, kernel is square and in_channels == out_channels
    assert kernel_h == kernel_w, "Only square kernels supported"
    kernel_size = kernel_h
    
    # Output dimensions (stride=1, padding=0)
    out_height = in_height - kernel_size + 1
    out_width = in_width - kernel_size + 1
    
    # Verify output shape
    assert output_tensor.shape == (batch, out_channels, out_height, out_width)
    
    # Block sizes for output spatial dimensions
    BLOCK_OH = 8
    BLOCK_OW = 8
    
    # Grid dimensions
    num_blocks_oh = triton.cdiv(out_height, BLOCK_OH)
    num_blocks_ow = triton.cdiv(out_width, BLOCK_OW)
    num_spatial_blocks = num_blocks_oh * num_blocks_ow
    
    grid = (batch, out_channels, num_spatial_blocks)
    
    # Launch kernel
    conv2d_kernel[grid](
        input_tensor,
        kernel,
        output_tensor,
        batch,
        in_channels,
        out_channels,
        in_height,
        in_width,
        kernel_size,
        out_height,
        out_width,
        BLOCK_OH,
        BLOCK_OW,
    )
    
    return output_tensor

import inspect

def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)

    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)


# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"

scrolls · 177 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON