submission 543368
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 78 lines, June 9 Researcher Reciprocity License v1.0.
gpumode_submit_0y1czc34.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-543368?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:9fbdd72cbd3f24bb1300d8833a1c031a62ca8abf550eb5e9706141117f9ce920
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
num_warps=8,stages = 1
num_stages=1,Kernel source
gpumode_submit_0y1czc34.py78 lines
# kernel.py
# Triton RGB(H,W,3) -> Grayscale(H,W) kernel:
# Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
#
# Single fused kernel:
# load RGB -> weighted sum in fp32 -> store Y
import torch
import triton
import triton.language as tl
@triton.jit
def _rgb_to_gray_kernel(
x_ptr, # *fp32, flattened H*W*3 (HWC contiguous)
y_ptr, # *fp32, flattened H*W
n_pixels: tl.int32,
BLOCK: tl.constexpr,
):
pid = tl.program_id(axis=0)
offs = pid * BLOCK + tl.arange(0, BLOCK) # pixel indices
mask = offs < n_pixels
# NOTE: tl.arange(0, 3) is invalid in this Triton version (range must be power-of-2),
# so we load R/G/B as three separate strided loads.
base = offs * 3
r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)
g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)
b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)
wr = tl.full((), 0.2989, tl.float32)
wg = tl.full((), 0.5870, tl.float32)
wb = tl.full((), 0.1140, tl.float32)
gray = r * wr + g * wg + b * wb
tl.store(y_ptr + offs, gray, mask=mask)
def kernel_function(x: torch.Tensor, y: torch.Tensor):
"""
Launch RGB(H,W,3)->Grayscale(H,W) Triton kernel.
Wrapper does only validation + launch (no PyTorch math).
"""
assert isinstance(x, torch.Tensor) and isinstance(y, torch.Tensor)
assert x.is_cuda and y.is_cuda, "Inputs must be CUDA tensors"
assert x.dtype == torch.float32 and y.dtype == torch.float32, "x and y must be float32"
assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"
assert y.ndim == 2 and y.shape[0] == x.shape[0] and y.shape[1] == x.shape[1], "y must have shape (H, W)"
assert x.is_contiguous(), "x must be contiguous (HWC)"
assert y.is_contiguous(), "y must be contiguous"
n_pixels = x.shape[0] * x.shape[1]
BLOCK = 1024 # power-of-2, good for tl.arange
grid = (triton.cdiv(n_pixels, BLOCK),)
_rgb_to_gray_kernel[grid](
x,
y,
n_pixels,
BLOCK=BLOCK,
num_warps=8,
num_stages=1,
)
return y
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 78 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 543210.
+ # kernel.py+ # Triton RGB(H,W,3) -> Grayscale(H,W) kernel:+ # Y = 0.2989 * R + 0.5870 * G + 0.1140 * B+ #+ # Single fused kernel:+ # load RGB -> weighted sum in fp32 -> store Y++ import torchimport tritonimport triton.language as tl- import torch@triton.jit- def _rgb_to_grayscale_kernel(- input_ptr,- output_ptr,- n_pixels,- BLOCK_SIZE: tl.constexpr,+ def _rgb_to_gray_kernel(+ x_ptr, # *fp32, flattened H*W*3 (HWC contiguous)+ y_ptr, # *fp32, flattened H*W+ n_pixels: tl.int32,+ BLOCK: tl.constexpr,):- # Fused RGB->grayscale: load 3 channels, weighted sum, store- # Y = 0.2989*R + 0.5870*G + 0.1140*B- pid = tl.program_id(0)- offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ pid = tl.program_id(axis=0)+ offs = pid * BLOCK + tl.arange(0, BLOCK) # pixel indicesmask = offs < n_pixels+ # NOTE: tl.arange(0, 3) is invalid in this Triton version (range must be power-of-2),+ # so we load R/G/B as three separate strided loads.base = offs * 3- r = tl.load(input_ptr + base, mask=mask, other=0.0)- g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)- b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)+ r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)+ g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)+ b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)- gray = r * 0.2989 + g * 0.5870 + b * 0.1140+ wr = tl.full((), 0.2989, tl.float32)+ wg = tl.full((), 0.5870, tl.float32)+ wb = tl.full((), 0.1140, tl.float32)- tl.store(output_ptr + offs, gray, mask=mask)+ gray = r * wr + g * wg + b * wb+ tl.store(y_ptr + offs, gray, mask=mask)- def kernel_function(rgb_input, output):- assert rgb_input.is_cuda- assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3- assert rgb_input.is_contiguous()- assert output.is_cuda+ def kernel_function(x: torch.Tensor, y: torch.Tensor):+ """+ Launch RGB(H,W,3)->Grayscale(H,W) Triton kernel.- H, W, _ = rgb_input.shape- n_pixels = H * W+ Wrapper does only validation + launch (no PyTorch math).+ """+ assert isinstance(x, torch.Tensor) and isinstance(y, torch.Tensor)+ assert x.is_cuda and y.is_cuda, "Inputs must be CUDA tensors"+ assert x.dtype == torch.float32 and y.dtype == torch.float32, "x and y must be float32"+ assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"+ assert y.ndim == 2 and y.shape[0] == x.shape[0] and y.shape[1] == x.shape[1], "y must have shape (H, W)"+ assert x.is_contiguous(), "x must be contiguous (HWC)"+ assert y.is_contiguous(), "y must be contiguous"- BLOCK_SIZE = 1024- grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)+ n_pixels = x.shape[0] * x.shape[1]+ BLOCK = 1024 # power-of-2, good for tl.arange+ grid = (triton.cdiv(n_pixels, BLOCK),)- _rgb_to_grayscale_kernel[grid](- rgb_input, output, n_pixels,- BLOCK_SIZE=BLOCK_SIZE,+ _rgb_to_gray_kernel[grid](+ x,+ y,+ n_pixels,+ BLOCK=BLOCK,+ num_warps=8,+ num_stages=1,)+ return y- return output-import inspectdef custom_kernel(input):sig = inspect.signature(kernel_function)
scrolls · 98 diff lines total
Best evidence level for this revision: reported
JSON