Skip to content
KernelIndex
Search⌘K

submission 584793

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 76 lines, June 9 Researcher Reciprocity License v1.0.

gpumode_submit_d_k9ivzl.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-584793?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.37ms
#3 of 36
2026-03-18

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:960db82bb6d522ad8fad4eed57e9a9a8ea449de00ac212fd36d9d7a4d8a631b7
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Kernel source

gpumode_submit_d_k9ivzl.py76 lines
import triton
import triton.language as tl
import torch


@triton.jit
def _rgb_to_grayscale_kernel(
    input_ptr,
    output_ptr,
    n_pixels,
    BLOCK_SIZE: tl.constexpr,
):
    """
    Fused RGB to grayscale conversion kernel.
    Computes Y = 0.2989 * R + 0.5870 * G + 0.1140 * B in a single pass.
    
    Fusion: channel loads, weighted multiply-accumulate, and store are all
    fused into one kernel. No intermediate buffers needed.
    """
    pid = tl.program_id(0)
    pixel_offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = pixel_offsets < n_pixels

    # Input layout is (H, W, 3) contiguous: pixel i has R,G,B at i*3+0,1,2
    base_offsets = pixel_offsets * 3

    # Load R, G, B channels
    r = tl.load(input_ptr + base_offsets, mask=mask, other=0.0)
    g = tl.load(input_ptr + base_offsets + 1, mask=mask, other=0.0)
    b = tl.load(input_ptr + base_offsets + 2, mask=mask, other=0.0)

    # Compute grayscale with standard NTSC/PAL luminance weights
    gray = r * 0.2989 + g * 0.5870 + b * 0.1140

    # Store result
    tl.store(output_ptr + pixel_offsets, gray, mask=mask)


def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
    """
    Wrapper for RGB to grayscale conversion.

    Args:
        rgb_input: Input tensor of shape (H, W, 3), dtype float32, on CUDA.
        output: Pre-allocated output tensor of shape (H, W), dtype float32, on CUDA.

    Returns:
        The output tensor containing grayscale values.
    """
    H, W, C = rgb_input.shape
    n_pixels = H * W

    BLOCK_SIZE = 1024
    grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)

    _rgb_to_grayscale_kernel[grid](
        rgb_input,
        output,
        n_pixels,
        BLOCK_SIZE=BLOCK_SIZE,
    )

    return output

import inspect
def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)
    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)

import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 76 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 549697.

⋯ 4 unchanged lines
@triton.jit
def _rgb_to_grayscale_kernel(
- rgb_ptr,
- gray_ptr,
+ input_ptr,
+ output_ptr,
n_pixels,
BLOCK_SIZE: tl.constexpr,
):
"""
Fused RGB to grayscale conversion kernel.
- Single-pass fusion: load R,G,B channels -> weighted sum -> store grayscale.
- Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
+ Computes Y = 0.2989 * R + 0.5870 * G + 0.1140 * B in a single pass.
- Input layout: (H, W, 3) contiguous, pixel i has R at 3*i, G at 3*i+1, B at 3*i+2.
- Output layout: (H, W) contiguous, grayscale value at index i.
+ Fusion: channel loads, weighted multiply-accumulate, and store are all
+ fused into one kernel. No intermediate buffers needed.
"""
pid = tl.program_id(0)
- offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
- mask = offs < n_pixels
+ pixel_offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = pixel_offsets < n_pixels
- # Base indices into interleaved RGB buffer
- base = offs * 3
+ # Input layout is (H, W, 3) contiguous: pixel i has R,G,B at i*3+0,1,2
+ base_offsets = pixel_offsets * 3
- # Load R, G, B channels with masking
- r = tl.load(rgb_ptr + base, mask=mask, other=0.0)
- g = tl.load(rgb_ptr + base + 1, mask=mask, other=0.0)
- b = tl.load(rgb_ptr + base + 2, mask=mask, other=0.0)
+ # Load R, G, B channels
+ r = tl.load(input_ptr + base_offsets, mask=mask, other=0.0)
+ g = tl.load(input_ptr + base_offsets + 1, mask=mask, other=0.0)
+ b = tl.load(input_ptr + base_offsets + 2, mask=mask, other=0.0)
- # Fused weighted sum
+ # Compute grayscale with standard NTSC/PAL luminance weights
gray = r * 0.2989 + g * 0.5870 + b * 0.1140
# Store result
- tl.store(gray_ptr + offs, gray, mask=mask)
+ tl.store(output_ptr + pixel_offsets, gray, mask=mask)
def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
"""
- Wrapper: validates inputs, computes grid, launches the Triton kernel.
- No PyTorch compute ops — all math is inside the Triton kernel.
- """
- assert rgb_input.is_cuda and output.is_cuda
- assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3
- assert rgb_input.is_contiguous()
+ Wrapper for RGB to grayscale conversion.
- H, W, _ = rgb_input.shape
+ Args:
+ rgb_input: Input tensor of shape (H, W, 3), dtype float32, on CUDA.
+ output: Pre-allocated output tensor of shape (H, W), dtype float32, on CUDA.
+
+ Returns:
+ The output tensor containing grayscale values.
+ """
+ H, W, C = rgb_input.shape
n_pixels = H * W
- BLOCK_SIZE = 512
+ BLOCK_SIZE = 1024
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
_rgb_to_grayscale_kernel[grid](
- rgb_input, output, n_pixels,
+ rgb_input,
+ output,
+ n_pixels,
BLOCK_SIZE=BLOCK_SIZE,
)
scrolls · 84 diff lines total

Best evidence level for this revision: reported

JSON