submission 489489
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 177 lines, June 9 Researcher Reciprocity License v1.0.
conv2d_py_H100_claude-opus-4.5_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-conv2d-v2-489489?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:d0e888c92b0066dad67e54e9a9af0fab5e6e8cb53ace04e57703babdad5edde7
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Kernel source
conv2d_py_H100_claude-opus-4.5_ka_submission.py177 lines
import triton
import triton.language as tl
import torch
@triton.jit
def conv2d_kernel(
input_ptr, # Pointer to input tensor [B, C, H, W]
kernel_ptr, # Pointer to kernel tensor [C_out, C_in, KH, KW]
output_ptr, # Pointer to output tensor [B, C_out, OH, OW]
# Dimensions
batch: tl.constexpr,
in_channels: tl.constexpr,
out_channels: tl.constexpr,
in_height: tl.constexpr,
in_width: tl.constexpr,
kernel_size: tl.constexpr,
out_height: tl.constexpr,
out_width: tl.constexpr,
# Block sizes
BLOCK_OH: tl.constexpr,
BLOCK_OW: tl.constexpr,
):
"""
Fused 2D convolution kernel.
Each program computes a block of output pixels for one (batch, out_channel) pair.
The convolution sum over in_channels and kernel spatial dimensions is computed inline.
"""
# Program IDs
pid_b = tl.program_id(0) # batch index
pid_oc = tl.program_id(1) # output channel index
pid_spatial = tl.program_id(2) # spatial block index
# Calculate spatial block position
num_blocks_ow = tl.cdiv(out_width, BLOCK_OW)
pid_oh = pid_spatial // num_blocks_ow
pid_ow = pid_spatial % num_blocks_ow
# Output pixel offsets within this block
offs_oh = pid_oh * BLOCK_OH + tl.arange(0, BLOCK_OH)
offs_ow = pid_ow * BLOCK_OW + tl.arange(0, BLOCK_OW)
# Masks for valid output positions
mask_oh = offs_oh < out_height
mask_ow = offs_ow < out_width
# Initialize accumulator for this block of output pixels
acc = tl.zeros((BLOCK_OH, BLOCK_OW), dtype=tl.float32)
# Loop over input channels
for ic in range(in_channels):
# Loop over kernel height
for kh in range(kernel_size):
# Loop over kernel width
for kw in range(kernel_size):
# Load kernel weight for this (oc, ic, kh, kw)
# Kernel layout: [out_channels, in_channels, kernel_size, kernel_size]
kernel_offset = (pid_oc * in_channels * kernel_size * kernel_size +
ic * kernel_size * kernel_size +
kh * kernel_size + kw)
w = tl.load(kernel_ptr + kernel_offset)
# Input positions: ih = oh + kh, iw = ow + kw
# Input layout: [batch, channels, height, width]
# offs_ih = offs_oh + kh (shape: BLOCK_OH)
# offs_iw = offs_ow + kw (shape: BLOCK_OW)
# Load input values for this block
# We need to load input[pid_b, ic, offs_oh + kh, offs_ow + kw]
input_base = (pid_b * in_channels * in_height * in_width +
ic * in_height * in_width)
# Calculate input offsets for each output position
# input_offset[i, j] = input_base + (offs_oh[i] + kh) * in_width + (offs_ow[j] + kw)
offs_ih = offs_oh + kh # [BLOCK_OH]
offs_iw = offs_ow + kw # [BLOCK_OW]
# Create 2D offset grid
input_offsets = input_base + offs_ih[:, None] * in_width + offs_iw[None, :]
# Create mask (input positions are always valid since we only compute valid output positions)
mask = mask_oh[:, None] & mask_ow[None, :]
# Load input values
x = tl.load(input_ptr + input_offsets, mask=mask, other=0.0)
# Accumulate: acc += x * w
acc += x * w
# Store output
# Output layout: [batch, out_channels, out_height, out_width]
output_base = (pid_b * out_channels * out_height * out_width +
pid_oc * out_height * out_width)
output_offsets = output_base + offs_oh[:, None] * out_width + offs_ow[None, :]
output_mask = mask_oh[:, None] & mask_ow[None, :]
tl.store(output_ptr + output_offsets, acc, mask=output_mask)
def kernel_function(input_tensor: torch.Tensor, kernel: torch.Tensor, output_tensor: torch.Tensor) -> torch.Tensor:
"""
Wrapper for 2D convolution using Triton.
This is a fused implementation that computes the entire convolution in a single kernel:
- For each output position, accumulates over all input channels and kernel spatial positions
- No separate im2col or matrix multiplication steps
Args:
input_tensor: Input tensor of shape [batch, in_channels, height, width]
kernel: Convolution kernel of shape [out_channels, in_channels, kH, kW]
output_tensor: Pre-allocated output tensor of shape [batch, out_channels, oH, oW]
Returns:
output_tensor filled with convolution result
"""
# Extract dimensions
batch, in_channels, in_height, in_width = input_tensor.shape
out_channels, _, kernel_h, kernel_w = kernel.shape
# For this problem, kernel is square and in_channels == out_channels
assert kernel_h == kernel_w, "Only square kernels supported"
kernel_size = kernel_h
# Output dimensions (stride=1, padding=0)
out_height = in_height - kernel_size + 1
out_width = in_width - kernel_size + 1
# Verify output shape
assert output_tensor.shape == (batch, out_channels, out_height, out_width)
# Block sizes for output spatial dimensions
BLOCK_OH = 8
BLOCK_OW = 8
# Grid dimensions
num_blocks_oh = triton.cdiv(out_height, BLOCK_OH)
num_blocks_ow = triton.cdiv(out_width, BLOCK_OW)
num_spatial_blocks = num_blocks_oh * num_blocks_ow
grid = (batch, out_channels, num_spatial_blocks)
# Launch kernel
conv2d_kernel[grid](
input_tensor,
kernel,
output_tensor,
batch,
in_channels,
out_channels,
in_height,
in_width,
kernel_size,
out_height,
out_width,
BLOCK_OH,
BLOCK_OW,
)
return output_tensor
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 177 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON