Skip to content
KernelIndex
Search⌘K

submission 543210

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 59 lines, June 9 Researcher Reciprocity License v1.0.

gpumode_submit_rglqa5th.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-543210?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.37ms
#10 of 36
2026-03-13

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:89c11e19593683d5113ca7cdd9a34a8ff6adff342a8458d4d956e1169d6dd1f0
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Kernel source

gpumode_submit_rglqa5th.py59 lines
import triton
import triton.language as tl
import torch


@triton.jit
def _rgb_to_grayscale_kernel(
    input_ptr,
    output_ptr,
    n_pixels,
    BLOCK_SIZE: tl.constexpr,
):
    # Fused RGB->grayscale: load 3 channels, weighted sum, store
    # Y = 0.2989*R + 0.5870*G + 0.1140*B
    pid = tl.program_id(0)
    offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offs < n_pixels

    base = offs * 3
    r = tl.load(input_ptr + base, mask=mask, other=0.0)
    g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)
    b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)

    gray = r * 0.2989 + g * 0.5870 + b * 0.1140

    tl.store(output_ptr + offs, gray, mask=mask)


def kernel_function(rgb_input, output):
    assert rgb_input.is_cuda
    assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3
    assert rgb_input.is_contiguous()
    assert output.is_cuda

    H, W, _ = rgb_input.shape
    n_pixels = H * W

    BLOCK_SIZE = 1024
    grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)

    _rgb_to_grayscale_kernel[grid](
        rgb_input, output, n_pixels,
        BLOCK_SIZE=BLOCK_SIZE,
    )

    return output

import inspect
def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)
    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)

import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 59 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 512144.

⋯ 4 unchanged lines
@triton.jit
def _rgb_to_grayscale_kernel(
- input_ptr, # Pointer to input RGB tensor (H, W, 3)
- output_ptr, # Pointer to output grayscale tensor (H, W)
- n_pixels, # Total number of pixels (H * W)
+ input_ptr,
+ output_ptr,
+ n_pixels,
BLOCK_SIZE: tl.constexpr,
):
- """
- Fused RGB to grayscale conversion kernel.
- Computes Y = 0.2989 * R + 0.5870 * G + 0.1140 * B in a single pass.
-
- Each program handles BLOCK_SIZE pixels. For each pixel, we load 3 channels,
- multiply by the luminance weights, sum, and store the result.
- """
+ # Fused RGB->grayscale: load 3 channels, weighted sum, store
+ # Y = 0.2989*R + 0.5870*G + 0.1140*B
pid = tl.program_id(0)
- block_start = pid * BLOCK_SIZE
- offsets = block_start + tl.arange(0, BLOCK_SIZE)
- mask = offsets < n_pixels
+ offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = offs < n_pixels
- # Each pixel has 3 channels stored contiguously: [R, G, B, R, G, B, ...]
- # Input layout is (H, W, 3), so pixel i starts at index i * 3
- base = offsets * 3
-
- # Load R, G, B channels for each pixel in the block
+ base = offs * 3
r = tl.load(input_ptr + base, mask=mask, other=0.0)
g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)
b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)
- # Compute grayscale using standard luminance weights
- # Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
- gray = 0.2989 * r + 0.5870 * g + 0.1140 * b
+ gray = r * 0.2989 + g * 0.5870 + b * 0.1140
- # Store result
- tl.store(output_ptr + offsets, gray, mask=mask)
+ tl.store(output_ptr + offs, gray, mask=mask)
- def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
- """
- Wrapper for RGB to grayscale conversion.
-
- Fusion note: The entire computation (loading 3 channels, weighted sum, store)
- is fused into a single Triton kernel pass. No intermediate buffers needed.
-
- Args:
- rgb_input: Input tensor of shape (H, W, 3), dtype float32, on CUDA
- output: Pre-allocated output tensor of shape (H, W), dtype float32, on CUDA
-
- Returns:
- The output tensor with grayscale values written in-place.
- """
- assert rgb_input.is_cuda and output.is_cuda, "Tensors must be on CUDA"
- assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
- assert output.shape == rgb_input.shape[:2], "Output must be (H, W)"
- assert rgb_input.is_contiguous(), "Input must be contiguous"
+ def kernel_function(rgb_input, output):
+ assert rgb_input.is_cuda
+ assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3
+ assert rgb_input.is_contiguous()
+ assert output.is_cuda
H, W, _ = rgb_input.shape
n_pixels = H * W
⋯ 2 unchanged lines
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
_rgb_to_grayscale_kernel[grid](
- rgb_input, output, n_pixels, BLOCK_SIZE=BLOCK_SIZE
+ rgb_input, output, n_pixels,
+ BLOCK_SIZE=BLOCK_SIZE,
)
return output
import inspect
-
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
-
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
-
- # Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
-
scrolls · 101 diff lines total

Best evidence level for this revision: reported

JSON