Skip to content
KernelIndex
Search⌘K

submission 549697

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 73 lines, June 9 Researcher Reciprocity License v1.0.

gpumode_submit_q9stbrl6.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-549697?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.37ms
#4 of 36
2026-03-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:46e49f5cd40ffff1521c3bf9194152815172a59daece28a61137c547d40b13f0
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Kernel source

gpumode_submit_q9stbrl6.py73 lines
import triton
import triton.language as tl
import torch


@triton.jit
def _rgb_to_grayscale_kernel(
    rgb_ptr,
    gray_ptr,
    n_pixels,
    BLOCK_SIZE: tl.constexpr,
):
    """
    Fused RGB to grayscale conversion kernel.
    Single-pass fusion: load R,G,B channels -> weighted sum -> store grayscale.
    Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
    
    Input layout: (H, W, 3) contiguous, pixel i has R at 3*i, G at 3*i+1, B at 3*i+2.
    Output layout: (H, W) contiguous, grayscale value at index i.
    """
    pid = tl.program_id(0)
    offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offs < n_pixels

    # Base indices into interleaved RGB buffer
    base = offs * 3

    # Load R, G, B channels with masking
    r = tl.load(rgb_ptr + base, mask=mask, other=0.0)
    g = tl.load(rgb_ptr + base + 1, mask=mask, other=0.0)
    b = tl.load(rgb_ptr + base + 2, mask=mask, other=0.0)

    # Fused weighted sum
    gray = r * 0.2989 + g * 0.5870 + b * 0.1140

    # Store result
    tl.store(gray_ptr + offs, gray, mask=mask)


def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
    """
    Wrapper: validates inputs, computes grid, launches the Triton kernel.
    No PyTorch compute ops — all math is inside the Triton kernel.
    """
    assert rgb_input.is_cuda and output.is_cuda
    assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3
    assert rgb_input.is_contiguous()

    H, W, _ = rgb_input.shape
    n_pixels = H * W

    BLOCK_SIZE = 512
    grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)

    _rgb_to_grayscale_kernel[grid](
        rgb_input, output, n_pixels,
        BLOCK_SIZE=BLOCK_SIZE,
    )

    return output

import inspect
def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)
    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)

import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 73 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 543368.

- # kernel.py
- # Triton RGB(H,W,3) -> Grayscale(H,W) kernel:
- # Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
- #
- # Single fused kernel:
- # load RGB -> weighted sum in fp32 -> store Y
-
- import torch
import triton
import triton.language as tl
+ import torch
@triton.jit
- def _rgb_to_gray_kernel(
- x_ptr, # *fp32, flattened H*W*3 (HWC contiguous)
- y_ptr, # *fp32, flattened H*W
- n_pixels: tl.int32,
- BLOCK: tl.constexpr,
+ def _rgb_to_grayscale_kernel(
+ rgb_ptr,
+ gray_ptr,
+ n_pixels,
+ BLOCK_SIZE: tl.constexpr,
):
- pid = tl.program_id(axis=0)
- offs = pid * BLOCK + tl.arange(0, BLOCK) # pixel indices
+ """
+ Fused RGB to grayscale conversion kernel.
+ Single-pass fusion: load R,G,B channels -> weighted sum -> store grayscale.
+ Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
+
+ Input layout: (H, W, 3) contiguous, pixel i has R at 3*i, G at 3*i+1, B at 3*i+2.
+ Output layout: (H, W) contiguous, grayscale value at index i.
+ """
+ pid = tl.program_id(0)
+ offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offs < n_pixels
- # NOTE: tl.arange(0, 3) is invalid in this Triton version (range must be power-of-2),
- # so we load R/G/B as three separate strided loads.
+ # Base indices into interleaved RGB buffer
base = offs * 3
- r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)
- g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)
- b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)
- wr = tl.full((), 0.2989, tl.float32)
- wg = tl.full((), 0.5870, tl.float32)
- wb = tl.full((), 0.1140, tl.float32)
+ # Load R, G, B channels with masking
+ r = tl.load(rgb_ptr + base, mask=mask, other=0.0)
+ g = tl.load(rgb_ptr + base + 1, mask=mask, other=0.0)
+ b = tl.load(rgb_ptr + base + 2, mask=mask, other=0.0)
- gray = r * wr + g * wg + b * wb
- tl.store(y_ptr + offs, gray, mask=mask)
+ # Fused weighted sum
+ gray = r * 0.2989 + g * 0.5870 + b * 0.1140
+ # Store result
+ tl.store(gray_ptr + offs, gray, mask=mask)
- def kernel_function(x: torch.Tensor, y: torch.Tensor):
- """
- Launch RGB(H,W,3)->Grayscale(H,W) Triton kernel.
- Wrapper does only validation + launch (no PyTorch math).
+ def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
"""
- assert isinstance(x, torch.Tensor) and isinstance(y, torch.Tensor)
- assert x.is_cuda and y.is_cuda, "Inputs must be CUDA tensors"
- assert x.dtype == torch.float32 and y.dtype == torch.float32, "x and y must be float32"
- assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"
- assert y.ndim == 2 and y.shape[0] == x.shape[0] and y.shape[1] == x.shape[1], "y must have shape (H, W)"
- assert x.is_contiguous(), "x must be contiguous (HWC)"
- assert y.is_contiguous(), "y must be contiguous"
+ Wrapper: validates inputs, computes grid, launches the Triton kernel.
+ No PyTorch compute ops — all math is inside the Triton kernel.
+ """
+ assert rgb_input.is_cuda and output.is_cuda
+ assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3
+ assert rgb_input.is_contiguous()
- n_pixels = x.shape[0] * x.shape[1]
- BLOCK = 1024 # power-of-2, good for tl.arange
- grid = (triton.cdiv(n_pixels, BLOCK),)
+ H, W, _ = rgb_input.shape
+ n_pixels = H * W
- _rgb_to_gray_kernel[grid](
- x,
- y,
- n_pixels,
- BLOCK=BLOCK,
- num_warps=8,
- num_stages=1,
+ BLOCK_SIZE = 512
+ grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
+
+ _rgb_to_grayscale_kernel[grid](
+ rgb_input, output, n_pixels,
+ BLOCK_SIZE=BLOCK_SIZE,
)
- return y
+ return output
+
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
scrolls · 111 diff lines total

Best evidence level for this revision: reported

JSON