submission 510650
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 144 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_v2_H100_claude-opus-4.5_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-510650?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:f3c3aaa70ada425518a67013c325aa80d7db199b0380e6e37674fd4936c975c8
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Kernel source
grayscale_v2_H100_claude-opus-4.5_ka_submission.py144 lines
import triton
import triton.language as tl
import torch
@triton.jit
def _rgb_to_grayscale_kernel(
rgb_ptr, # Pointer to input RGB image (H, W, 3)
out_ptr, # Pointer to output grayscale image (H, W)
H, # Image height
W, # Image width
stride_h, # Stride for height dimension in RGB
stride_w, # Stride for width dimension in RGB
stride_c, # Stride for channel dimension in RGB
out_stride_h, # Stride for height dimension in output
out_stride_w, # Stride for width dimension in output
BLOCK_SIZE: tl.constexpr, # Number of pixels to process per block
):
"""
Triton kernel for RGB to grayscale conversion.
Fused operation: For each pixel, loads R, G, B values and computes
the weighted sum Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass.
"""
# Standard RGB to grayscale coefficients
R_COEFF = 0.2989
G_COEFF = 0.5870
B_COEFF = 0.1140
# Each program processes BLOCK_SIZE pixels
pid = tl.program_id(0)
# Total number of pixels
n_pixels = H * W
# Calculate pixel indices this block will process
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
# Mask for valid pixels
mask = offsets < n_pixels
# Convert linear index to 2D coordinates (row-major order)
# pixel_idx = h * W + w
h_idx = offsets // W
w_idx = offsets % W
# Calculate base offset for each pixel in RGB image
# RGB layout is (H, W, 3), so offset = h * stride_h + w * stride_w + c * stride_c
base_offset = h_idx * stride_h + w_idx * stride_w
# Load R, G, B values for each pixel
r_offset = base_offset + 0 * stride_c # Channel 0 = R
g_offset = base_offset + 1 * stride_c # Channel 1 = G
b_offset = base_offset + 2 * stride_c # Channel 2 = B
r = tl.load(rgb_ptr + r_offset, mask=mask, other=0.0)
g = tl.load(rgb_ptr + g_offset, mask=mask, other=0.0)
b = tl.load(rgb_ptr + b_offset, mask=mask, other=0.0)
# Compute grayscale value using standard coefficients
gray = R_COEFF * r + G_COEFF * g + B_COEFF * b
# Calculate output offset
out_offset = h_idx * out_stride_h + w_idx * out_stride_w
# Store the result
tl.store(out_ptr + out_offset, gray, mask=mask)
def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
"""
Wrapper function for RGB to grayscale conversion.
This function performs a fused RGB to grayscale conversion where:
- Loading of R, G, B channels
- Weighted sum computation (Y = 0.2989*R + 0.5870*G + 0.1140*B)
- Storage of grayscale result
are all done in a single kernel pass, minimizing memory traffic.
Args:
rgb_input: Input RGB tensor of shape (H, W, 3), dtype float32
output: Output buffer of shape (H, W), dtype float32
Returns:
The output tensor containing grayscale values
"""
# Validate inputs
assert rgb_input.is_cuda, "Input must be on CUDA device"
assert output.is_cuda, "Output must be on CUDA device"
assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
assert output.ndim == 2, "Output must be (H, W)"
assert rgb_input.shape[0] == output.shape[0] and rgb_input.shape[1] == output.shape[1], \
"Input and output spatial dimensions must match"
H, W, C = rgb_input.shape
n_pixels = H * W
# Get strides (in elements, not bytes)
stride_h = rgb_input.stride(0)
stride_w = rgb_input.stride(1)
stride_c = rgb_input.stride(2)
out_stride_h = output.stride(0)
out_stride_w = output.stride(1)
# Choose block size - power of 2 for efficiency
BLOCK_SIZE = 1024
# Calculate grid size
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
# Launch kernel
_rgb_to_grayscale_kernel[grid](
rgb_input,
output,
H,
W,
stride_h,
stride_w,
stride_c,
out_stride_h,
out_stride_w,
BLOCK_SIZE=BLOCK_SIZE,
)
return output
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 144 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 490599.
- import torchimport tritonimport triton.language as tl+ import torch@triton.jit- def _rgb_to_grayscale_kernel(x_ptr, y_ptr, n_pixels, #- w_r, w_g, w_b, #- BLOCK_SIZE: tl.constexpr):+ def _rgb_to_grayscale_kernel(+ rgb_ptr, # Pointer to input RGB image (H, W, 3)+ out_ptr, # Pointer to output grayscale image (H, W)+ H, # Image height+ W, # Image width+ stride_h, # Stride for height dimension in RGB+ stride_w, # Stride for width dimension in RGB+ stride_c, # Stride for channel dimension in RGB+ out_stride_h, # Stride for height dimension in output+ out_stride_w, # Stride for width dimension in output+ BLOCK_SIZE: tl.constexpr, # Number of pixels to process per block+ ):"""- Elementwise RGB -> Grayscale conversion for a flat list of pixels.-- Each program processes up to BLOCK_SIZE pixels. Pixels are laid out as- contiguous triplets [R, G, B] in memory, i.e., the original image is (H, W, 3) contiguous.+ Triton kernel for RGB to grayscale conversion.++ Fused operation: For each pixel, loads R, G, B values and computes+ the weighted sum Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass."""- pid = tl.program_id(axis=0)+ # Standard RGB to grayscale coefficients+ R_COEFF = 0.2989+ G_COEFF = 0.5870+ B_COEFF = 0.1140++ # Each program processes BLOCK_SIZE pixels+ pid = tl.program_id(0)++ # Total number of pixels+ n_pixels = H * W++ # Calculate pixel indices this block will processblock_start = pid * BLOCK_SIZEoffsets = block_start + tl.arange(0, BLOCK_SIZE)++ # Mask for valid pixelsmask = offsets < n_pixels++ # Convert linear index to 2D coordinates (row-major order)+ # pixel_idx = h * W + w+ h_idx = offsets // W+ w_idx = offsets % W++ # Calculate base offset for each pixel in RGB image+ # RGB layout is (H, W, 3), so offset = h * stride_h + w * stride_w + c * stride_c+ base_offset = h_idx * stride_h + w_idx * stride_w++ # Load R, G, B values for each pixel+ r_offset = base_offset + 0 * stride_c # Channel 0 = R+ g_offset = base_offset + 1 * stride_c # Channel 1 = G+ b_offset = base_offset + 2 * stride_c # Channel 2 = B++ r = tl.load(rgb_ptr + r_offset, mask=mask, other=0.0)+ g = tl.load(rgb_ptr + g_offset, mask=mask, other=0.0)+ b = tl.load(rgb_ptr + b_offset, mask=mask, other=0.0)++ # Compute grayscale value using standard coefficients+ gray = R_COEFF * r + G_COEFF * g + B_COEFF * b++ # Calculate output offset+ out_offset = h_idx * out_stride_h + w_idx * out_stride_w++ # Store the result+ tl.store(out_ptr + out_offset, gray, mask=mask)- # Input layout: (H, W, 3) contiguous => pixel i starts at index (i * 3)- base = offsets * 3- r = tl.load(x_ptr + base + 0, mask=mask, other=0.0)- g = tl.load(x_ptr + base + 1, mask=mask, other=0.0)- b = tl.load(x_ptr + base + 2, mask=mask, other=0.0)-- # Compute in fp32 for numerical stability- r32 = r.to(tl.float32)- g32 = g.to(tl.float32)- b32 = b.to(tl.float32)-- # Y = 0.2989 R + 0.5870 G + 0.1140 B- y32 = r32 * w_r + g32 * w_g + b32 * w_b-- # Cast back to output dtype on store- y = y32.to(y_ptr.dtype.element_ty)- tl.store(y_ptr + offsets, y, mask=mask)--- def kernel_function(x: torch.Tensor, y_out: torch.Tensor = None):+ def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:"""- Triton RGB -> Grayscale conversion (fused in a single pass).-- What is fused:- - Load R, G, B for each pixel- - Convert to fp32- - Apply grayscale weights and accumulate- - Cast to output dtype and store- All steps are fused within one Triton kernel, avoiding intermediate global-memory traffic.-+ Wrapper function for RGB to grayscale conversion.++ This function performs a fused RGB to grayscale conversion where:+ - Loading of R, G, B channels+ - Weighted sum computation (Y = 0.2989*R + 0.5870*G + 0.1140*B)+ - Storage of grayscale result+ are all done in a single kernel pass, minimizing memory traffic.+Args:- x: Input tensor of shape (H, W, 3), contiguous, on CUDA. dtypes supported: float32, bfloat16, float16.- y_out: Optional preallocated output tensor of shape (H, W) and same dtype/device as x.-+ rgb_input: Input RGB tensor of shape (H, W, 3), dtype float32+ output: Output buffer of shape (H, W), dtype float32+Returns:- Tensor of shape (H, W) with grayscale values in the same dtype/device as x.+ The output tensor containing grayscale values"""- # Argument validation (no math in wrapper)- assert x.is_cuda, "Input x must be on CUDA device"- assert x.ndim == 3 and x.shape[-1] == 3, "Input must have shape (H, W, 3)"- assert x.is_contiguous(), "Input must be contiguous (H, W, 3) layout"- assert x.dtype in (torch.float32, torch.bfloat16, torch.float16), "Supported dtypes: float32, bfloat16, float16"-- H, W, C = x.shape+ # Validate inputs+ assert rgb_input.is_cuda, "Input must be on CUDA device"+ assert output.is_cuda, "Output must be on CUDA device"+ assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"+ assert output.ndim == 2, "Output must be (H, W)"+ assert rgb_input.shape[0] == output.shape[0] and rgb_input.shape[1] == output.shape[1], \+ "Input and output spatial dimensions must match"++ H, W, C = rgb_input.shapen_pixels = H * W-- # Prepare output buffer- if y_out is None:- y_out = torch.empty((H, W), device=x.device, dtype=x.dtype)- else:- assert y_out.is_cuda, "Output must be on CUDA"- assert y_out.shape == (H, W), f"Output must have shape {(H, W)}"- assert y_out.dtype == x.dtype, "Output dtype must match input dtype"- assert y_out.is_contiguous(), "Output must be contiguous"-- # Weights for grayscale conversion- w_r = 0.2989- w_g = 0.5870- w_b = 0.1140-- # Launch configuration- # Use a 1D grid over pixels; BLOCK_SIZE chosen as a power of two for good occupancy++ # Get strides (in elements, not bytes)+ stride_h = rgb_input.stride(0)+ stride_w = rgb_input.stride(1)+ stride_c = rgb_input.stride(2)+ out_stride_h = output.stride(0)+ out_stride_w = output.stride(1)++ # Choose block size - power of 2 for efficiencyBLOCK_SIZE = 1024++ # Calculate grid sizegrid = (triton.cdiv(n_pixels, BLOCK_SIZE),)-- # Launch Triton kernel++ # Launch kernel_rgb_to_grayscale_kernel[grid](- x, y_out, n_pixels,- w_r, w_g, w_b,+ rgb_input,+ output,+ H,+ W,+ stride_h,+ stride_w,+ stride_c,+ out_stride_h,+ out_stride_w,BLOCK_SIZE=BLOCK_SIZE,- num_warps=4, # reasonable default for elementwise workloads- num_stages=2,)++ return output- return y_out-import inspectdef custom_kernel(input):
scrolls · 204 diff lines total
Best evidence level for this revision: reported
JSON