Skip to content
KernelIndex
Search⌘K

submission 510650

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 144 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_v2_H100_claude-opus-4.5_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-510650?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.38ms
#24 of 36
2026-02-27

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:f3c3aaa70ada425518a67013c325aa80d7db199b0380e6e37674fd4936c975c8
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Kernel source

grayscale_v2_H100_claude-opus-4.5_ka_submission.py144 lines
import triton
import triton.language as tl
import torch


@triton.jit
def _rgb_to_grayscale_kernel(
    rgb_ptr,        # Pointer to input RGB image (H, W, 3)
    out_ptr,        # Pointer to output grayscale image (H, W)
    H,              # Image height
    W,              # Image width
    stride_h,       # Stride for height dimension in RGB
    stride_w,       # Stride for width dimension in RGB
    stride_c,       # Stride for channel dimension in RGB
    out_stride_h,   # Stride for height dimension in output
    out_stride_w,   # Stride for width dimension in output
    BLOCK_SIZE: tl.constexpr,  # Number of pixels to process per block
):
    """
    Triton kernel for RGB to grayscale conversion.
    
    Fused operation: For each pixel, loads R, G, B values and computes
    the weighted sum Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass.
    """
    # Standard RGB to grayscale coefficients
    R_COEFF = 0.2989
    G_COEFF = 0.5870
    B_COEFF = 0.1140
    
    # Each program processes BLOCK_SIZE pixels
    pid = tl.program_id(0)
    
    # Total number of pixels
    n_pixels = H * W
    
    # Calculate pixel indices this block will process
    block_start = pid * BLOCK_SIZE
    offsets = block_start + tl.arange(0, BLOCK_SIZE)
    
    # Mask for valid pixels
    mask = offsets < n_pixels
    
    # Convert linear index to 2D coordinates (row-major order)
    # pixel_idx = h * W + w
    h_idx = offsets // W
    w_idx = offsets % W
    
    # Calculate base offset for each pixel in RGB image
    # RGB layout is (H, W, 3), so offset = h * stride_h + w * stride_w + c * stride_c
    base_offset = h_idx * stride_h + w_idx * stride_w
    
    # Load R, G, B values for each pixel
    r_offset = base_offset + 0 * stride_c  # Channel 0 = R
    g_offset = base_offset + 1 * stride_c  # Channel 1 = G
    b_offset = base_offset + 2 * stride_c  # Channel 2 = B
    
    r = tl.load(rgb_ptr + r_offset, mask=mask, other=0.0)
    g = tl.load(rgb_ptr + g_offset, mask=mask, other=0.0)
    b = tl.load(rgb_ptr + b_offset, mask=mask, other=0.0)
    
    # Compute grayscale value using standard coefficients
    gray = R_COEFF * r + G_COEFF * g + B_COEFF * b
    
    # Calculate output offset
    out_offset = h_idx * out_stride_h + w_idx * out_stride_w
    
    # Store the result
    tl.store(out_ptr + out_offset, gray, mask=mask)


def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
    """
    Wrapper function for RGB to grayscale conversion.
    
    This function performs a fused RGB to grayscale conversion where:
    - Loading of R, G, B channels
    - Weighted sum computation (Y = 0.2989*R + 0.5870*G + 0.1140*B)
    - Storage of grayscale result
    are all done in a single kernel pass, minimizing memory traffic.
    
    Args:
        rgb_input: Input RGB tensor of shape (H, W, 3), dtype float32
        output: Output buffer of shape (H, W), dtype float32
        
    Returns:
        The output tensor containing grayscale values
    """
    # Validate inputs
    assert rgb_input.is_cuda, "Input must be on CUDA device"
    assert output.is_cuda, "Output must be on CUDA device"
    assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
    assert output.ndim == 2, "Output must be (H, W)"
    assert rgb_input.shape[0] == output.shape[0] and rgb_input.shape[1] == output.shape[1], \
        "Input and output spatial dimensions must match"
    
    H, W, C = rgb_input.shape
    n_pixels = H * W
    
    # Get strides (in elements, not bytes)
    stride_h = rgb_input.stride(0)
    stride_w = rgb_input.stride(1)
    stride_c = rgb_input.stride(2)
    out_stride_h = output.stride(0)
    out_stride_w = output.stride(1)
    
    # Choose block size - power of 2 for efficiency
    BLOCK_SIZE = 1024
    
    # Calculate grid size
    grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
    
    # Launch kernel
    _rgb_to_grayscale_kernel[grid](
        rgb_input,
        output,
        H,
        W,
        stride_h,
        stride_w,
        stride_c,
        out_stride_h,
        out_stride_w,
        BLOCK_SIZE=BLOCK_SIZE,
    )
    
    return output

import inspect

def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)

    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)


# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"

scrolls · 144 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 490599.

- import torch
import triton
import triton.language as tl
+ import torch
@triton.jit
- def _rgb_to_grayscale_kernel(x_ptr, y_ptr, n_pixels, #
- w_r, w_g, w_b, #
- BLOCK_SIZE: tl.constexpr):
+ def _rgb_to_grayscale_kernel(
+ rgb_ptr, # Pointer to input RGB image (H, W, 3)
+ out_ptr, # Pointer to output grayscale image (H, W)
+ H, # Image height
+ W, # Image width
+ stride_h, # Stride for height dimension in RGB
+ stride_w, # Stride for width dimension in RGB
+ stride_c, # Stride for channel dimension in RGB
+ out_stride_h, # Stride for height dimension in output
+ out_stride_w, # Stride for width dimension in output
+ BLOCK_SIZE: tl.constexpr, # Number of pixels to process per block
+ ):
"""
- Elementwise RGB -> Grayscale conversion for a flat list of pixels.
-
- Each program processes up to BLOCK_SIZE pixels. Pixels are laid out as
- contiguous triplets [R, G, B] in memory, i.e., the original image is (H, W, 3) contiguous.
+ Triton kernel for RGB to grayscale conversion.
+
+ Fused operation: For each pixel, loads R, G, B values and computes
+ the weighted sum Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass.
"""
- pid = tl.program_id(axis=0)
+ # Standard RGB to grayscale coefficients
+ R_COEFF = 0.2989
+ G_COEFF = 0.5870
+ B_COEFF = 0.1140
+
+ # Each program processes BLOCK_SIZE pixels
+ pid = tl.program_id(0)
+
+ # Total number of pixels
+ n_pixels = H * W
+
+ # Calculate pixel indices this block will process
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
+
+ # Mask for valid pixels
mask = offsets < n_pixels
+
+ # Convert linear index to 2D coordinates (row-major order)
+ # pixel_idx = h * W + w
+ h_idx = offsets // W
+ w_idx = offsets % W
+
+ # Calculate base offset for each pixel in RGB image
+ # RGB layout is (H, W, 3), so offset = h * stride_h + w * stride_w + c * stride_c
+ base_offset = h_idx * stride_h + w_idx * stride_w
+
+ # Load R, G, B values for each pixel
+ r_offset = base_offset + 0 * stride_c # Channel 0 = R
+ g_offset = base_offset + 1 * stride_c # Channel 1 = G
+ b_offset = base_offset + 2 * stride_c # Channel 2 = B
+
+ r = tl.load(rgb_ptr + r_offset, mask=mask, other=0.0)
+ g = tl.load(rgb_ptr + g_offset, mask=mask, other=0.0)
+ b = tl.load(rgb_ptr + b_offset, mask=mask, other=0.0)
+
+ # Compute grayscale value using standard coefficients
+ gray = R_COEFF * r + G_COEFF * g + B_COEFF * b
+
+ # Calculate output offset
+ out_offset = h_idx * out_stride_h + w_idx * out_stride_w
+
+ # Store the result
+ tl.store(out_ptr + out_offset, gray, mask=mask)
- # Input layout: (H, W, 3) contiguous => pixel i starts at index (i * 3)
- base = offsets * 3
- r = tl.load(x_ptr + base + 0, mask=mask, other=0.0)
- g = tl.load(x_ptr + base + 1, mask=mask, other=0.0)
- b = tl.load(x_ptr + base + 2, mask=mask, other=0.0)
-
- # Compute in fp32 for numerical stability
- r32 = r.to(tl.float32)
- g32 = g.to(tl.float32)
- b32 = b.to(tl.float32)
-
- # Y = 0.2989 R + 0.5870 G + 0.1140 B
- y32 = r32 * w_r + g32 * w_g + b32 * w_b
-
- # Cast back to output dtype on store
- y = y32.to(y_ptr.dtype.element_ty)
- tl.store(y_ptr + offsets, y, mask=mask)
-
-
- def kernel_function(x: torch.Tensor, y_out: torch.Tensor = None):
+ def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
"""
- Triton RGB -> Grayscale conversion (fused in a single pass).
-
- What is fused:
- - Load R, G, B for each pixel
- - Convert to fp32
- - Apply grayscale weights and accumulate
- - Cast to output dtype and store
- All steps are fused within one Triton kernel, avoiding intermediate global-memory traffic.
-
+ Wrapper function for RGB to grayscale conversion.
+
+ This function performs a fused RGB to grayscale conversion where:
+ - Loading of R, G, B channels
+ - Weighted sum computation (Y = 0.2989*R + 0.5870*G + 0.1140*B)
+ - Storage of grayscale result
+ are all done in a single kernel pass, minimizing memory traffic.
+
Args:
- x: Input tensor of shape (H, W, 3), contiguous, on CUDA. dtypes supported: float32, bfloat16, float16.
- y_out: Optional preallocated output tensor of shape (H, W) and same dtype/device as x.
-
+ rgb_input: Input RGB tensor of shape (H, W, 3), dtype float32
+ output: Output buffer of shape (H, W), dtype float32
+
Returns:
- Tensor of shape (H, W) with grayscale values in the same dtype/device as x.
+ The output tensor containing grayscale values
"""
- # Argument validation (no math in wrapper)
- assert x.is_cuda, "Input x must be on CUDA device"
- assert x.ndim == 3 and x.shape[-1] == 3, "Input must have shape (H, W, 3)"
- assert x.is_contiguous(), "Input must be contiguous (H, W, 3) layout"
- assert x.dtype in (torch.float32, torch.bfloat16, torch.float16), "Supported dtypes: float32, bfloat16, float16"
-
- H, W, C = x.shape
+ # Validate inputs
+ assert rgb_input.is_cuda, "Input must be on CUDA device"
+ assert output.is_cuda, "Output must be on CUDA device"
+ assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
+ assert output.ndim == 2, "Output must be (H, W)"
+ assert rgb_input.shape[0] == output.shape[0] and rgb_input.shape[1] == output.shape[1], \
+ "Input and output spatial dimensions must match"
+
+ H, W, C = rgb_input.shape
n_pixels = H * W
-
- # Prepare output buffer
- if y_out is None:
- y_out = torch.empty((H, W), device=x.device, dtype=x.dtype)
- else:
- assert y_out.is_cuda, "Output must be on CUDA"
- assert y_out.shape == (H, W), f"Output must have shape {(H, W)}"
- assert y_out.dtype == x.dtype, "Output dtype must match input dtype"
- assert y_out.is_contiguous(), "Output must be contiguous"
-
- # Weights for grayscale conversion
- w_r = 0.2989
- w_g = 0.5870
- w_b = 0.1140
-
- # Launch configuration
- # Use a 1D grid over pixels; BLOCK_SIZE chosen as a power of two for good occupancy
+
+ # Get strides (in elements, not bytes)
+ stride_h = rgb_input.stride(0)
+ stride_w = rgb_input.stride(1)
+ stride_c = rgb_input.stride(2)
+ out_stride_h = output.stride(0)
+ out_stride_w = output.stride(1)
+
+ # Choose block size - power of 2 for efficiency
BLOCK_SIZE = 1024
+
+ # Calculate grid size
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
-
- # Launch Triton kernel
+
+ # Launch kernel
_rgb_to_grayscale_kernel[grid](
- x, y_out, n_pixels,
- w_r, w_g, w_b,
+ rgb_input,
+ output,
+ H,
+ W,
+ stride_h,
+ stride_w,
+ stride_c,
+ out_stride_h,
+ out_stride_w,
BLOCK_SIZE=BLOCK_SIZE,
- num_warps=4, # reasonable default for elementwise workloads
- num_stages=2,
)
+
+ return output
- return y_out
-
import inspect
def custom_kernel(input):
scrolls · 204 diff lines total

Best evidence level for this revision: reported

JSON