submission 512144
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 89 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_v2_H100_claude-opus-4.6_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-512144?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:488759fca43ec2e025e0b9c5c290795c7e42e5a52ac149176a2f760fcaefe5d1
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Kernel source
grayscale_v2_H100_claude-opus-4.6_ka_submission.py89 lines
import triton
import triton.language as tl
import torch
@triton.jit
def _rgb_to_grayscale_kernel(
input_ptr, # Pointer to input RGB tensor (H, W, 3)
output_ptr, # Pointer to output grayscale tensor (H, W)
n_pixels, # Total number of pixels (H * W)
BLOCK_SIZE: tl.constexpr,
):
"""
Fused RGB to grayscale conversion kernel.
Computes Y = 0.2989 * R + 0.5870 * G + 0.1140 * B in a single pass.
Each program handles BLOCK_SIZE pixels. For each pixel, we load 3 channels,
multiply by the luminance weights, sum, and store the result.
"""
pid = tl.program_id(0)
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_pixels
# Each pixel has 3 channels stored contiguously: [R, G, B, R, G, B, ...]
# Input layout is (H, W, 3), so pixel i starts at index i * 3
base = offsets * 3
# Load R, G, B channels for each pixel in the block
r = tl.load(input_ptr + base, mask=mask, other=0.0)
g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)
b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)
# Compute grayscale using standard luminance weights
# Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
gray = 0.2989 * r + 0.5870 * g + 0.1140 * b
# Store result
tl.store(output_ptr + offsets, gray, mask=mask)
def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
"""
Wrapper for RGB to grayscale conversion.
Fusion note: The entire computation (loading 3 channels, weighted sum, store)
is fused into a single Triton kernel pass. No intermediate buffers needed.
Args:
rgb_input: Input tensor of shape (H, W, 3), dtype float32, on CUDA
output: Pre-allocated output tensor of shape (H, W), dtype float32, on CUDA
Returns:
The output tensor with grayscale values written in-place.
"""
assert rgb_input.is_cuda and output.is_cuda, "Tensors must be on CUDA"
assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
assert output.shape == rgb_input.shape[:2], "Output must be (H, W)"
assert rgb_input.is_contiguous(), "Input must be contiguous"
H, W, _ = rgb_input.shape
n_pixels = H * W
BLOCK_SIZE = 1024
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
_rgb_to_grayscale_kernel[grid](
rgb_input, output, n_pixels, BLOCK_SIZE=BLOCK_SIZE
)
return output
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 89 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 510652.
- # kernel.py- import torchimport tritonimport triton.language as tl+ import torch- @triton.autotune(- configs=[- triton.Config({"BLOCK": 256}, num_warps=4),- triton.Config({"BLOCK": 512}, num_warps=8),- triton.Config({"BLOCK": 1024}, num_warps=8),- ],- key=["N"],- )@triton.jit- def _rgb_to_gray_kernel(- x_ptr, # *[H, W, 3]- y_ptr, # *[H, W]- N, # H*W- BLOCK: tl.constexpr,+ def _rgb_to_grayscale_kernel(+ input_ptr, # Pointer to input RGB tensor (H, W, 3)+ output_ptr, # Pointer to output grayscale tensor (H, W)+ n_pixels, # Total number of pixels (H * W)+ BLOCK_SIZE: tl.constexpr,):- pid = tl.program_id(axis=0)- offs = pid * BLOCK + tl.arange(0, BLOCK)- mask = offs < N+ """+ Fused RGB to grayscale conversion kernel.+ Computes Y = 0.2989 * R + 0.5870 * G + 0.1140 * B in a single pass.++ Each program handles BLOCK_SIZE pixels. For each pixel, we load 3 channels,+ multiply by the luminance weights, sum, and store the result.+ """+ pid = tl.program_id(0)+ block_start = pid * BLOCK_SIZE+ offsets = block_start + tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_pixels- # x is contiguous [H, W, 3] => flattened pixels are contiguous triplets- base = offs * 3+ # Each pixel has 3 channels stored contiguously: [R, G, B, R, G, B, ...]+ # Input layout is (H, W, 3), so pixel i starts at index i * 3+ base = offsets * 3- r = tl.load(x_ptr + base + 0, mask=mask, other=0).to(tl.float32)- g = tl.load(x_ptr + base + 1, mask=mask, other=0).to(tl.float32)- b = tl.load(x_ptr + base + 2, mask=mask, other=0).to(tl.float32)+ # Load R, G, B channels for each pixel in the block+ r = tl.load(input_ptr + base, mask=mask, other=0.0)+ g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)+ b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)- # Match reference semantics more closely by quantizing weights to x dtype first.- x_ty = x_ptr.dtype.element_ty- w_r = tl.full((), 0.2989, tl.float32).to(x_ty).to(tl.float32)- w_g = tl.full((), 0.5870, tl.float32).to(x_ty).to(tl.float32)- w_b = tl.full((), 0.1140, tl.float32).to(x_ty).to(tl.float32)+ # Compute grayscale using standard luminance weights+ # Y = 0.2989 * R + 0.5870 * G + 0.1140 * B+ gray = 0.2989 * r + 0.5870 * g + 0.1140 * b- y_f32 = r * w_r + g * w_g + b * w_b- tl.store(y_ptr + offs, y_f32.to(y_ptr.dtype.element_ty), mask=mask)+ # Store result+ tl.store(output_ptr + offsets, gray, mask=mask)- def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:- # Validation / allocation only (no compute here).- if not isinstance(x, torch.Tensor):- raise TypeError("x must be a torch.Tensor")- if not x.is_cuda:- raise ValueError("x must be a CUDA tensor")- if x.ndim != 3 or x.shape[-1] != 3:- raise ValueError(f"x must have shape [H, W, 3], got {tuple(x.shape)}")- if not x.is_contiguous():- raise ValueError("x must be contiguous (expected contiguous [H, W, 3])")+ def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:+ """+ Wrapper for RGB to grayscale conversion.++ Fusion note: The entire computation (loading 3 channels, weighted sum, store)+ is fused into a single Triton kernel pass. No intermediate buffers needed.++ Args:+ rgb_input: Input tensor of shape (H, W, 3), dtype float32, on CUDA+ output: Pre-allocated output tensor of shape (H, W), dtype float32, on CUDA++ Returns:+ The output tensor with grayscale values written in-place.+ """+ assert rgb_input.is_cuda and output.is_cuda, "Tensors must be on CUDA"+ assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"+ assert output.shape == rgb_input.shape[:2], "Output must be (H, W)"+ assert rgb_input.is_contiguous(), "Input must be contiguous"- H, W, _ = x.shape- if y is None:- y = torch.empty((H, W), device=x.device, dtype=x.dtype)- else:- if not isinstance(y, torch.Tensor):- raise TypeError("y must be a torch.Tensor")- if not y.is_cuda:- raise ValueError("y must be a CUDA tensor")- if tuple(y.shape) != (H, W):- raise ValueError(f"y must have shape {(H, W)}, got {tuple(y.shape)}")- if not y.is_contiguous():- raise ValueError("y must be contiguous")+ H, W, _ = rgb_input.shape+ n_pixels = H * W- N = H * W+ BLOCK_SIZE = 1024+ grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)- grid = lambda META: (triton.cdiv(N, META["BLOCK"]),)- _rgb_to_gray_kernel[grid](x, y, N)+ _rgb_to_grayscale_kernel[grid](+ rgb_input, output, n_pixels, BLOCK_SIZE=BLOCK_SIZE+ )- return y+ return outputimport inspect
scrolls · 130 diff lines total
Best evidence level for this revision: reported
JSON