Skip to content
KernelIndex
Search⌘K

submission 512144

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 89 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_v2_H100_claude-opus-4.6_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-512144?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.37ms
#13 of 36
2026-03-01

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:488759fca43ec2e025e0b9c5c290795c7e42e5a52ac149176a2f760fcaefe5d1
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Kernel source

grayscale_v2_H100_claude-opus-4.6_ka_submission.py89 lines
import triton
import triton.language as tl
import torch


@triton.jit
def _rgb_to_grayscale_kernel(
    input_ptr,      # Pointer to input RGB tensor (H, W, 3)
    output_ptr,     # Pointer to output grayscale tensor (H, W)
    n_pixels,       # Total number of pixels (H * W)
    BLOCK_SIZE: tl.constexpr,
):
    """
    Fused RGB to grayscale conversion kernel.
    Computes Y = 0.2989 * R + 0.5870 * G + 0.1140 * B in a single pass.
    
    Each program handles BLOCK_SIZE pixels. For each pixel, we load 3 channels,
    multiply by the luminance weights, sum, and store the result.
    """
    pid = tl.program_id(0)
    block_start = pid * BLOCK_SIZE
    offsets = block_start + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_pixels

    # Each pixel has 3 channels stored contiguously: [R, G, B, R, G, B, ...]
    # Input layout is (H, W, 3), so pixel i starts at index i * 3
    base = offsets * 3

    # Load R, G, B channels for each pixel in the block
    r = tl.load(input_ptr + base, mask=mask, other=0.0)
    g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)
    b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)

    # Compute grayscale using standard luminance weights
    # Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
    gray = 0.2989 * r + 0.5870 * g + 0.1140 * b

    # Store result
    tl.store(output_ptr + offsets, gray, mask=mask)


def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
    """
    Wrapper for RGB to grayscale conversion.
    
    Fusion note: The entire computation (loading 3 channels, weighted sum, store)
    is fused into a single Triton kernel pass. No intermediate buffers needed.
    
    Args:
        rgb_input: Input tensor of shape (H, W, 3), dtype float32, on CUDA
        output: Pre-allocated output tensor of shape (H, W), dtype float32, on CUDA
        
    Returns:
        The output tensor with grayscale values written in-place.
    """
    assert rgb_input.is_cuda and output.is_cuda, "Tensors must be on CUDA"
    assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
    assert output.shape == rgb_input.shape[:2], "Output must be (H, W)"
    assert rgb_input.is_contiguous(), "Input must be contiguous"

    H, W, _ = rgb_input.shape
    n_pixels = H * W

    BLOCK_SIZE = 1024
    grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)

    _rgb_to_grayscale_kernel[grid](
        rgb_input, output, n_pixels, BLOCK_SIZE=BLOCK_SIZE
    )

    return output

import inspect

def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)

    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)


# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"

scrolls · 89 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 510652.

- # kernel.py
- import torch
import triton
import triton.language as tl
+ import torch
- @triton.autotune(
- configs=[
- triton.Config({"BLOCK": 256}, num_warps=4),
- triton.Config({"BLOCK": 512}, num_warps=8),
- triton.Config({"BLOCK": 1024}, num_warps=8),
- ],
- key=["N"],
- )
@triton.jit
- def _rgb_to_gray_kernel(
- x_ptr, # *[H, W, 3]
- y_ptr, # *[H, W]
- N, # H*W
- BLOCK: tl.constexpr,
+ def _rgb_to_grayscale_kernel(
+ input_ptr, # Pointer to input RGB tensor (H, W, 3)
+ output_ptr, # Pointer to output grayscale tensor (H, W)
+ n_pixels, # Total number of pixels (H * W)
+ BLOCK_SIZE: tl.constexpr,
):
- pid = tl.program_id(axis=0)
- offs = pid * BLOCK + tl.arange(0, BLOCK)
- mask = offs < N
+ """
+ Fused RGB to grayscale conversion kernel.
+ Computes Y = 0.2989 * R + 0.5870 * G + 0.1140 * B in a single pass.
+
+ Each program handles BLOCK_SIZE pixels. For each pixel, we load 3 channels,
+ multiply by the luminance weights, sum, and store the result.
+ """
+ pid = tl.program_id(0)
+ block_start = pid * BLOCK_SIZE
+ offsets = block_start + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < n_pixels
- # x is contiguous [H, W, 3] => flattened pixels are contiguous triplets
- base = offs * 3
+ # Each pixel has 3 channels stored contiguously: [R, G, B, R, G, B, ...]
+ # Input layout is (H, W, 3), so pixel i starts at index i * 3
+ base = offsets * 3
- r = tl.load(x_ptr + base + 0, mask=mask, other=0).to(tl.float32)
- g = tl.load(x_ptr + base + 1, mask=mask, other=0).to(tl.float32)
- b = tl.load(x_ptr + base + 2, mask=mask, other=0).to(tl.float32)
+ # Load R, G, B channels for each pixel in the block
+ r = tl.load(input_ptr + base, mask=mask, other=0.0)
+ g = tl.load(input_ptr + base + 1, mask=mask, other=0.0)
+ b = tl.load(input_ptr + base + 2, mask=mask, other=0.0)
- # Match reference semantics more closely by quantizing weights to x dtype first.
- x_ty = x_ptr.dtype.element_ty
- w_r = tl.full((), 0.2989, tl.float32).to(x_ty).to(tl.float32)
- w_g = tl.full((), 0.5870, tl.float32).to(x_ty).to(tl.float32)
- w_b = tl.full((), 0.1140, tl.float32).to(x_ty).to(tl.float32)
+ # Compute grayscale using standard luminance weights
+ # Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
+ gray = 0.2989 * r + 0.5870 * g + 0.1140 * b
- y_f32 = r * w_r + g * w_g + b * w_b
- tl.store(y_ptr + offs, y_f32.to(y_ptr.dtype.element_ty), mask=mask)
+ # Store result
+ tl.store(output_ptr + offsets, gray, mask=mask)
- def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:
- # Validation / allocation only (no compute here).
- if not isinstance(x, torch.Tensor):
- raise TypeError("x must be a torch.Tensor")
- if not x.is_cuda:
- raise ValueError("x must be a CUDA tensor")
- if x.ndim != 3 or x.shape[-1] != 3:
- raise ValueError(f"x must have shape [H, W, 3], got {tuple(x.shape)}")
- if not x.is_contiguous():
- raise ValueError("x must be contiguous (expected contiguous [H, W, 3])")
+ def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
+ """
+ Wrapper for RGB to grayscale conversion.
+
+ Fusion note: The entire computation (loading 3 channels, weighted sum, store)
+ is fused into a single Triton kernel pass. No intermediate buffers needed.
+
+ Args:
+ rgb_input: Input tensor of shape (H, W, 3), dtype float32, on CUDA
+ output: Pre-allocated output tensor of shape (H, W), dtype float32, on CUDA
+
+ Returns:
+ The output tensor with grayscale values written in-place.
+ """
+ assert rgb_input.is_cuda and output.is_cuda, "Tensors must be on CUDA"
+ assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
+ assert output.shape == rgb_input.shape[:2], "Output must be (H, W)"
+ assert rgb_input.is_contiguous(), "Input must be contiguous"
- H, W, _ = x.shape
- if y is None:
- y = torch.empty((H, W), device=x.device, dtype=x.dtype)
- else:
- if not isinstance(y, torch.Tensor):
- raise TypeError("y must be a torch.Tensor")
- if not y.is_cuda:
- raise ValueError("y must be a CUDA tensor")
- if tuple(y.shape) != (H, W):
- raise ValueError(f"y must have shape {(H, W)}, got {tuple(y.shape)}")
- if not y.is_contiguous():
- raise ValueError("y must be contiguous")
+ H, W, _ = rgb_input.shape
+ n_pixels = H * W
- N = H * W
+ BLOCK_SIZE = 1024
+ grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
- grid = lambda META: (triton.cdiv(N, META["BLOCK"]),)
- _rgb_to_gray_kernel[grid](x, y, N)
+ _rgb_to_grayscale_kernel[grid](
+ rgb_input, output, n_pixels, BLOCK_SIZE=BLOCK_SIZE
+ )
- return y
+ return output
import inspect
scrolls · 130 diff lines total

Best evidence level for this revision: reported

JSON