submission 549697
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 73 lines, June 9 Researcher Reciprocity License v1.0.
gpumode_submit_q9stbrl6.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-549697?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:46e49f5cd40ffff1521c3bf9194152815172a59daece28a61137c547d40b13f0
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Kernel source
gpumode_submit_q9stbrl6.py73 lines
import triton
import triton.language as tl
import torch
@triton.jit
def _rgb_to_grayscale_kernel(
rgb_ptr,
gray_ptr,
n_pixels,
BLOCK_SIZE: tl.constexpr,
):
"""
Fused RGB to grayscale conversion kernel.
Single-pass fusion: load R,G,B channels -> weighted sum -> store grayscale.
Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
Input layout: (H, W, 3) contiguous, pixel i has R at 3*i, G at 3*i+1, B at 3*i+2.
Output layout: (H, W) contiguous, grayscale value at index i.
"""
pid = tl.program_id(0)
offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offs < n_pixels
# Base indices into interleaved RGB buffer
base = offs * 3
# Load R, G, B channels with masking
r = tl.load(rgb_ptr + base, mask=mask, other=0.0)
g = tl.load(rgb_ptr + base + 1, mask=mask, other=0.0)
b = tl.load(rgb_ptr + base + 2, mask=mask, other=0.0)
# Fused weighted sum
gray = r * 0.2989 + g * 0.5870 + b * 0.1140
# Store result
tl.store(gray_ptr + offs, gray, mask=mask)
def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
"""
Wrapper: validates inputs, computes grid, launches the Triton kernel.
No PyTorch compute ops — all math is inside the Triton kernel.
"""
assert rgb_input.is_cuda and output.is_cuda
assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3
assert rgb_input.is_contiguous()
H, W, _ = rgb_input.shape
n_pixels = H * W
BLOCK_SIZE = 512
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
_rgb_to_grayscale_kernel[grid](
rgb_input, output, n_pixels,
BLOCK_SIZE=BLOCK_SIZE,
)
return output
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 73 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 543368.
- # kernel.py- # Triton RGB(H,W,3) -> Grayscale(H,W) kernel:- # Y = 0.2989 * R + 0.5870 * G + 0.1140 * B- #- # Single fused kernel:- # load RGB -> weighted sum in fp32 -> store Y-- import torchimport tritonimport triton.language as tl+ import torch@triton.jit- def _rgb_to_gray_kernel(- x_ptr, # *fp32, flattened H*W*3 (HWC contiguous)- y_ptr, # *fp32, flattened H*W- n_pixels: tl.int32,- BLOCK: tl.constexpr,+ def _rgb_to_grayscale_kernel(+ rgb_ptr,+ gray_ptr,+ n_pixels,+ BLOCK_SIZE: tl.constexpr,):- pid = tl.program_id(axis=0)- offs = pid * BLOCK + tl.arange(0, BLOCK) # pixel indices+ """+ Fused RGB to grayscale conversion kernel.+ Single-pass fusion: load R,G,B channels -> weighted sum -> store grayscale.+ Y = 0.2989 * R + 0.5870 * G + 0.1140 * B++ Input layout: (H, W, 3) contiguous, pixel i has R at 3*i, G at 3*i+1, B at 3*i+2.+ Output layout: (H, W) contiguous, grayscale value at index i.+ """+ pid = tl.program_id(0)+ offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)mask = offs < n_pixels- # NOTE: tl.arange(0, 3) is invalid in this Triton version (range must be power-of-2),- # so we load R/G/B as three separate strided loads.+ # Base indices into interleaved RGB bufferbase = offs * 3- r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)- g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)- b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)- wr = tl.full((), 0.2989, tl.float32)- wg = tl.full((), 0.5870, tl.float32)- wb = tl.full((), 0.1140, tl.float32)+ # Load R, G, B channels with masking+ r = tl.load(rgb_ptr + base, mask=mask, other=0.0)+ g = tl.load(rgb_ptr + base + 1, mask=mask, other=0.0)+ b = tl.load(rgb_ptr + base + 2, mask=mask, other=0.0)- gray = r * wr + g * wg + b * wb- tl.store(y_ptr + offs, gray, mask=mask)+ # Fused weighted sum+ gray = r * 0.2989 + g * 0.5870 + b * 0.1140+ # Store result+ tl.store(gray_ptr + offs, gray, mask=mask)- def kernel_function(x: torch.Tensor, y: torch.Tensor):- """- Launch RGB(H,W,3)->Grayscale(H,W) Triton kernel.- Wrapper does only validation + launch (no PyTorch math).+ def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:"""- assert isinstance(x, torch.Tensor) and isinstance(y, torch.Tensor)- assert x.is_cuda and y.is_cuda, "Inputs must be CUDA tensors"- assert x.dtype == torch.float32 and y.dtype == torch.float32, "x and y must be float32"- assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"- assert y.ndim == 2 and y.shape[0] == x.shape[0] and y.shape[1] == x.shape[1], "y must have shape (H, W)"- assert x.is_contiguous(), "x must be contiguous (HWC)"- assert y.is_contiguous(), "y must be contiguous"+ Wrapper: validates inputs, computes grid, launches the Triton kernel.+ No PyTorch compute ops — all math is inside the Triton kernel.+ """+ assert rgb_input.is_cuda and output.is_cuda+ assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3+ assert rgb_input.is_contiguous()- n_pixels = x.shape[0] * x.shape[1]- BLOCK = 1024 # power-of-2, good for tl.arange- grid = (triton.cdiv(n_pixels, BLOCK),)+ H, W, _ = rgb_input.shape+ n_pixels = H * W- _rgb_to_gray_kernel[grid](- x,- y,- n_pixels,- BLOCK=BLOCK,- num_warps=8,- num_stages=1,+ BLOCK_SIZE = 512+ grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)++ _rgb_to_grayscale_kernel[grid](+ rgb_input, output, n_pixels,+ BLOCK_SIZE=BLOCK_SIZE,)- return y+ return output+import inspectdef custom_kernel(input):sig = inspect.signature(kernel_function)
scrolls · 111 diff lines total
Best evidence level for this revision: reported
JSON