Skip to content
KernelIndex
Search⌘K

submission 490598

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 138 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_py_H100_claude-opus-4.5_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-490598?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.39ms
#27 of 36
2026-02-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:9369cafd70d891c716c23c0993946fcdf3bc97ac1b25c3d5b53ab1ce08992bb2
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Kernel source

grayscale_py_H100_claude-opus-4.5_ka_submission.py138 lines
import triton
import triton.language as tl
import torch


@triton.jit
def _rgb_to_grayscale_kernel(
    rgb_ptr,      # Pointer to input RGB tensor (H, W, 3)
    gray_ptr,     # Pointer to output grayscale tensor (H, W)
    H,            # Height of image
    W,            # Width of image
    stride_h,     # Stride for height dimension in RGB tensor
    stride_w,     # Stride for width dimension in RGB tensor
    stride_c,     # Stride for channel dimension in RGB tensor
    out_stride_h, # Stride for height dimension in output tensor
    out_stride_w, # Stride for width dimension in output tensor
    BLOCK_SIZE: tl.constexpr,  # Number of pixels to process per block
):
    """
    Triton kernel for RGB to grayscale conversion.
    
    Fused operation: For each pixel, loads R, G, B values and computes
    Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass.
    """
    # Standard RGB to grayscale coefficients
    R_COEFF = 0.2989
    G_COEFF = 0.5870
    B_COEFF = 0.1140
    
    # Get program ID - each program handles BLOCK_SIZE pixels
    pid = tl.program_id(0)
    
    # Total number of pixels
    n_pixels = H * W
    
    # Calculate pixel indices this block will process
    block_start = pid * BLOCK_SIZE
    offsets = block_start + tl.arange(0, BLOCK_SIZE)
    
    # Mask for out-of-bounds pixels
    mask = offsets < n_pixels
    
    # Convert linear index to 2D coordinates (row, col)
    row = offsets // W
    col = offsets % W
    
    # Calculate input pointer offsets for each channel
    # Input layout is (H, W, 3) so offset = row * stride_h + col * stride_w + channel * stride_c
    base_offset = row * stride_h + col * stride_w
    
    # Load R, G, B values
    r_vals = tl.load(rgb_ptr + base_offset + 0 * stride_c, mask=mask, other=0.0)
    g_vals = tl.load(rgb_ptr + base_offset + 1 * stride_c, mask=mask, other=0.0)
    b_vals = tl.load(rgb_ptr + base_offset + 2 * stride_c, mask=mask, other=0.0)
    
    # Compute grayscale using standard coefficients
    gray_vals = R_COEFF * r_vals + G_COEFF * g_vals + B_COEFF * b_vals
    
    # Calculate output pointer offsets
    out_offset = row * out_stride_h + col * out_stride_w
    
    # Store result
    tl.store(gray_ptr + out_offset, gray_vals, mask=mask)


def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
    """
    RGB to grayscale conversion using Triton kernel.
    
    Fused operation: Single kernel pass that loads RGB values and computes
    grayscale output using Y = 0.2989*R + 0.5870*G + 0.1140*B.
    
    Args:
        rgb_input: Input RGB tensor of shape (H, W, 3) with dtype float32
        output: Output grayscale tensor of shape (H, W) with dtype float32
        
    Returns:
        The output tensor containing grayscale values
    """
    # Validate inputs
    assert rgb_input.is_cuda, "Input must be on CUDA device"
    assert output.is_cuda, "Output must be on CUDA device"
    assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
    assert rgb_input.is_contiguous(), "Input must be contiguous"
    assert output.is_contiguous(), "Output must be contiguous"
    
    H, W, C = rgb_input.shape
    assert output.shape == (H, W), f"Output shape must be ({H}, {W})"
    
    # Total number of pixels
    n_pixels = H * W
    
    # Block size - number of pixels per block
    BLOCK_SIZE = 1024
    
    # Calculate grid size
    grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
    
    # Get strides (in elements, not bytes)
    stride_h = rgb_input.stride(0)
    stride_w = rgb_input.stride(1)
    stride_c = rgb_input.stride(2)
    out_stride_h = output.stride(0)
    out_stride_w = output.stride(1)
    
    # Launch kernel
    _rgb_to_grayscale_kernel[grid](
        rgb_input,
        output,
        H,
        W,
        stride_h,
        stride_w,
        stride_c,
        out_stride_h,
        out_stride_w,
        BLOCK_SIZE=BLOCK_SIZE,
    )
    
    return output

import inspect

def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)

    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)


# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"

scrolls · 138 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON