submission 490598
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 138 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_py_H100_claude-opus-4.5_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-490598?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:9369cafd70d891c716c23c0993946fcdf3bc97ac1b25c3d5b53ab1ce08992bb2
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Kernel source
grayscale_py_H100_claude-opus-4.5_ka_submission.py138 lines
import triton
import triton.language as tl
import torch
@triton.jit
def _rgb_to_grayscale_kernel(
rgb_ptr, # Pointer to input RGB tensor (H, W, 3)
gray_ptr, # Pointer to output grayscale tensor (H, W)
H, # Height of image
W, # Width of image
stride_h, # Stride for height dimension in RGB tensor
stride_w, # Stride for width dimension in RGB tensor
stride_c, # Stride for channel dimension in RGB tensor
out_stride_h, # Stride for height dimension in output tensor
out_stride_w, # Stride for width dimension in output tensor
BLOCK_SIZE: tl.constexpr, # Number of pixels to process per block
):
"""
Triton kernel for RGB to grayscale conversion.
Fused operation: For each pixel, loads R, G, B values and computes
Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass.
"""
# Standard RGB to grayscale coefficients
R_COEFF = 0.2989
G_COEFF = 0.5870
B_COEFF = 0.1140
# Get program ID - each program handles BLOCK_SIZE pixels
pid = tl.program_id(0)
# Total number of pixels
n_pixels = H * W
# Calculate pixel indices this block will process
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
# Mask for out-of-bounds pixels
mask = offsets < n_pixels
# Convert linear index to 2D coordinates (row, col)
row = offsets // W
col = offsets % W
# Calculate input pointer offsets for each channel
# Input layout is (H, W, 3) so offset = row * stride_h + col * stride_w + channel * stride_c
base_offset = row * stride_h + col * stride_w
# Load R, G, B values
r_vals = tl.load(rgb_ptr + base_offset + 0 * stride_c, mask=mask, other=0.0)
g_vals = tl.load(rgb_ptr + base_offset + 1 * stride_c, mask=mask, other=0.0)
b_vals = tl.load(rgb_ptr + base_offset + 2 * stride_c, mask=mask, other=0.0)
# Compute grayscale using standard coefficients
gray_vals = R_COEFF * r_vals + G_COEFF * g_vals + B_COEFF * b_vals
# Calculate output pointer offsets
out_offset = row * out_stride_h + col * out_stride_w
# Store result
tl.store(gray_ptr + out_offset, gray_vals, mask=mask)
def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
"""
RGB to grayscale conversion using Triton kernel.
Fused operation: Single kernel pass that loads RGB values and computes
grayscale output using Y = 0.2989*R + 0.5870*G + 0.1140*B.
Args:
rgb_input: Input RGB tensor of shape (H, W, 3) with dtype float32
output: Output grayscale tensor of shape (H, W) with dtype float32
Returns:
The output tensor containing grayscale values
"""
# Validate inputs
assert rgb_input.is_cuda, "Input must be on CUDA device"
assert output.is_cuda, "Output must be on CUDA device"
assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
assert rgb_input.is_contiguous(), "Input must be contiguous"
assert output.is_contiguous(), "Output must be contiguous"
H, W, C = rgb_input.shape
assert output.shape == (H, W), f"Output shape must be ({H}, {W})"
# Total number of pixels
n_pixels = H * W
# Block size - number of pixels per block
BLOCK_SIZE = 1024
# Calculate grid size
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
# Get strides (in elements, not bytes)
stride_h = rgb_input.stride(0)
stride_w = rgb_input.stride(1)
stride_c = rgb_input.stride(2)
out_stride_h = output.stride(0)
out_stride_w = output.stride(1)
# Launch kernel
_rgb_to_grayscale_kernel[grid](
rgb_input,
output,
H,
W,
stride_h,
stride_w,
stride_c,
out_stride_h,
out_stride_w,
BLOCK_SIZE=BLOCK_SIZE,
)
return output
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 138 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON