Skip to content
KernelIndex
Search⌘K

submission 543368

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 78 lines, June 9 Researcher Reciprocity License v1.0.

gpumode_submit_0y1czc34.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-543368?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.37ms
#6 of 36
2026-03-13

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:9fbdd72cbd3f24bb1300d8833a1c031a62ca8abf550eb5e9706141117f9ce920
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 8num_warps=8,
stages = 1num_stages=1,

Kernel source

gpumode_submit_0y1czc34.py78 lines
# kernel.py
# Triton RGB(H,W,3) -> Grayscale(H,W) kernel:
#   Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
#
# Single fused kernel:
#   load RGB -> weighted sum in fp32 -> store Y

import torch
import triton
import triton.language as tl


@triton.jit
def _rgb_to_gray_kernel(
    x_ptr,  # *fp32, flattened H*W*3 (HWC contiguous)
    y_ptr,  # *fp32, flattened H*W
    n_pixels: tl.int32,
    BLOCK: tl.constexpr,
):
    pid = tl.program_id(axis=0)
    offs = pid * BLOCK + tl.arange(0, BLOCK)  # pixel indices
    mask = offs < n_pixels

    # NOTE: tl.arange(0, 3) is invalid in this Triton version (range must be power-of-2),
    # so we load R/G/B as three separate strided loads.
    base = offs * 3
    r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)
    g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)
    b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)

    wr = tl.full((), 0.2989, tl.float32)
    wg = tl.full((), 0.5870, tl.float32)
    wb = tl.full((), 0.1140, tl.float32)

    gray = r * wr + g * wg + b * wb
    tl.store(y_ptr + offs, gray, mask=mask)


def kernel_function(x: torch.Tensor, y: torch.Tensor):
    """
    Launch RGB(H,W,3)->Grayscale(H,W) Triton kernel.

    Wrapper does only validation + launch (no PyTorch math).
    """
    assert isinstance(x, torch.Tensor) and isinstance(y, torch.Tensor)
    assert x.is_cuda and y.is_cuda, "Inputs must be CUDA tensors"
    assert x.dtype == torch.float32 and y.dtype == torch.float32, "x and y must be float32"
    assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"
    assert y.ndim == 2 and y.shape[0] == x.shape[0] and y.shape[1] == x.shape[1], "y must have shape (H, W)"
    assert x.is_contiguous(), "x must be contiguous (HWC)"
    assert y.is_contiguous(), "y must be contiguous"

    n_pixels = x.shape[0] * x.shape[1]
    BLOCK = 1024  # power-of-2, good for tl.arange
    grid = (triton.cdiv(n_pixels, BLOCK),)

    _rgb_to_gray_kernel[grid](
        x,
        y,
        n_pixels,
        BLOCK=BLOCK,
        num_warps=8,
        num_stages=1,
    )
    return y

import inspect
def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)
    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)

import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 78 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 543210.

+ # kernel.py
+ # Triton RGB(H,W,3) -> Grayscale(H,W) kernel:
+ # Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
+ #
+ # Single fused kernel:
+ # load RGB -> weighted sum in fp32 -> store Y
+
+ import torch
import triton
import triton.language as tl
- import torch
@triton.jit
- def _rgb_to_grayscale_kernel(
- input_ptr,
- output_ptr,
- n_pixels,
- BLOCK_SIZE: tl.constexpr,
+ def _rgb_to_gray_kernel(
+ x_ptr, # *fp32, flattened H*W*3 (HWC contiguous)
+ y_ptr, # *fp32, flattened H*W
+ n_pixels: tl.int32,
+ BLOCK: tl.constexpr,
):
- # Fused RGB->grayscale: load 3 channels, weighted sum, store
- # Y = 0.2989*R + 0.5870*G + 0.1140*B
- pid = tl.program_id(0)
- offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ pid = tl.program_id(axis=0)
+ offs = pid * BLOCK + tl.arange(0, BLOCK) # pixel indices
mask = offs < n_pixels
+ # NOTE: tl.arange(0, 3) is invalid in this Triton version (range must be power-of-2),
+ # so we load R/G/B as three separate strided loads.
base = offs * 3
- r = tl.load(input_ptr + base, mask=mask, other=0.0)
- g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)
- b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)
+ r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)
+ g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)
+ b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)
- gray = r * 0.2989 + g * 0.5870 + b * 0.1140
+ wr = tl.full((), 0.2989, tl.float32)
+ wg = tl.full((), 0.5870, tl.float32)
+ wb = tl.full((), 0.1140, tl.float32)
- tl.store(output_ptr + offs, gray, mask=mask)
+ gray = r * wr + g * wg + b * wb
+ tl.store(y_ptr + offs, gray, mask=mask)
- def kernel_function(rgb_input, output):
- assert rgb_input.is_cuda
- assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3
- assert rgb_input.is_contiguous()
- assert output.is_cuda
+ def kernel_function(x: torch.Tensor, y: torch.Tensor):
+ """
+ Launch RGB(H,W,3)->Grayscale(H,W) Triton kernel.
- H, W, _ = rgb_input.shape
- n_pixels = H * W
+ Wrapper does only validation + launch (no PyTorch math).
+ """
+ assert isinstance(x, torch.Tensor) and isinstance(y, torch.Tensor)
+ assert x.is_cuda and y.is_cuda, "Inputs must be CUDA tensors"
+ assert x.dtype == torch.float32 and y.dtype == torch.float32, "x and y must be float32"
+ assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"
+ assert y.ndim == 2 and y.shape[0] == x.shape[0] and y.shape[1] == x.shape[1], "y must have shape (H, W)"
+ assert x.is_contiguous(), "x must be contiguous (HWC)"
+ assert y.is_contiguous(), "y must be contiguous"
- BLOCK_SIZE = 1024
- grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
+ n_pixels = x.shape[0] * x.shape[1]
+ BLOCK = 1024 # power-of-2, good for tl.arange
+ grid = (triton.cdiv(n_pixels, BLOCK),)
- _rgb_to_grayscale_kernel[grid](
- rgb_input, output, n_pixels,
- BLOCK_SIZE=BLOCK_SIZE,
+ _rgb_to_gray_kernel[grid](
+ x,
+ y,
+ n_pixels,
+ BLOCK=BLOCK,
+ num_warps=8,
+ num_stages=1,
)
+ return y
- return output
-
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
scrolls · 98 diff lines total

Best evidence level for this revision: reported

JSON