Skip to content
KernelIndex
Search⌘K

submission 490599

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 114 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_py_H100_gpt-5_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-490599?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.38ms
#25 of 36
2026-02-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:13962b423e0017e6830228a0033fa44a16be46e59c5554d38e440e7093a8d68b
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 4num_warps=4, # reasonable default for elementwise workloads
stages = 2num_stages=2,

Kernel source

grayscale_py_H100_gpt-5_ka_submission.py114 lines
import torch
import triton
import triton.language as tl


@triton.jit
def _rgb_to_grayscale_kernel(x_ptr, y_ptr, n_pixels,  #
                             w_r, w_g, w_b,  #
                             BLOCK_SIZE: tl.constexpr):
    """
    Elementwise RGB -> Grayscale conversion for a flat list of pixels.

    Each program processes up to BLOCK_SIZE pixels. Pixels are laid out as
    contiguous triplets [R, G, B] in memory, i.e., the original image is (H, W, 3) contiguous.
    """
    pid = tl.program_id(axis=0)
    block_start = pid * BLOCK_SIZE
    offsets = block_start + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_pixels

    # Input layout: (H, W, 3) contiguous => pixel i starts at index (i * 3)
    base = offsets * 3

    r = tl.load(x_ptr + base + 0, mask=mask, other=0.0)
    g = tl.load(x_ptr + base + 1, mask=mask, other=0.0)
    b = tl.load(x_ptr + base + 2, mask=mask, other=0.0)

    # Compute in fp32 for numerical stability
    r32 = r.to(tl.float32)
    g32 = g.to(tl.float32)
    b32 = b.to(tl.float32)

    # Y = 0.2989 R + 0.5870 G + 0.1140 B
    y32 = r32 * w_r + g32 * w_g + b32 * w_b

    # Cast back to output dtype on store
    y = y32.to(y_ptr.dtype.element_ty)
    tl.store(y_ptr + offsets, y, mask=mask)


def kernel_function(x: torch.Tensor, y_out: torch.Tensor = None):
    """
    Triton RGB -> Grayscale conversion (fused in a single pass).

    What is fused:
    - Load R, G, B for each pixel
    - Convert to fp32
    - Apply grayscale weights and accumulate
    - Cast to output dtype and store
    All steps are fused within one Triton kernel, avoiding intermediate global-memory traffic.

    Args:
        x: Input tensor of shape (H, W, 3), contiguous, on CUDA. dtypes supported: float32, bfloat16, float16.
        y_out: Optional preallocated output tensor of shape (H, W) and same dtype/device as x.

    Returns:
        Tensor of shape (H, W) with grayscale values in the same dtype/device as x.
    """
    # Argument validation (no math in wrapper)
    assert x.is_cuda, "Input x must be on CUDA device"
    assert x.ndim == 3 and x.shape[-1] == 3, "Input must have shape (H, W, 3)"
    assert x.is_contiguous(), "Input must be contiguous (H, W, 3) layout"
    assert x.dtype in (torch.float32, torch.bfloat16, torch.float16), "Supported dtypes: float32, bfloat16, float16"

    H, W, C = x.shape
    n_pixels = H * W

    # Prepare output buffer
    if y_out is None:
        y_out = torch.empty((H, W), device=x.device, dtype=x.dtype)
    else:
        assert y_out.is_cuda, "Output must be on CUDA"
        assert y_out.shape == (H, W), f"Output must have shape {(H, W)}"
        assert y_out.dtype == x.dtype, "Output dtype must match input dtype"
        assert y_out.is_contiguous(), "Output must be contiguous"

    # Weights for grayscale conversion
    w_r = 0.2989
    w_g = 0.5870
    w_b = 0.1140

    # Launch configuration
    # Use a 1D grid over pixels; BLOCK_SIZE chosen as a power of two for good occupancy
    BLOCK_SIZE = 1024
    grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)

    # Launch Triton kernel
    _rgb_to_grayscale_kernel[grid](
        x, y_out, n_pixels,
        w_r, w_g, w_b,
        BLOCK_SIZE=BLOCK_SIZE,
        num_warps=4,  # reasonable default for elementwise workloads
        num_stages=2,
    )

    return y_out

import inspect

def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)

    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)


# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"

scrolls · 114 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 490598.

+ import torch
import triton
import triton.language as tl
- import torch
@triton.jit
- def _rgb_to_grayscale_kernel(
- rgb_ptr, # Pointer to input RGB tensor (H, W, 3)
- gray_ptr, # Pointer to output grayscale tensor (H, W)
- H, # Height of image
- W, # Width of image
- stride_h, # Stride for height dimension in RGB tensor
- stride_w, # Stride for width dimension in RGB tensor
- stride_c, # Stride for channel dimension in RGB tensor
- out_stride_h, # Stride for height dimension in output tensor
- out_stride_w, # Stride for width dimension in output tensor
- BLOCK_SIZE: tl.constexpr, # Number of pixels to process per block
- ):
+ def _rgb_to_grayscale_kernel(x_ptr, y_ptr, n_pixels, #
+ w_r, w_g, w_b, #
+ BLOCK_SIZE: tl.constexpr):
"""
- Triton kernel for RGB to grayscale conversion.
-
- Fused operation: For each pixel, loads R, G, B values and computes
- Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass.
+ Elementwise RGB -> Grayscale conversion for a flat list of pixels.
+
+ Each program processes up to BLOCK_SIZE pixels. Pixels are laid out as
+ contiguous triplets [R, G, B] in memory, i.e., the original image is (H, W, 3) contiguous.
"""
- # Standard RGB to grayscale coefficients
- R_COEFF = 0.2989
- G_COEFF = 0.5870
- B_COEFF = 0.1140
-
- # Get program ID - each program handles BLOCK_SIZE pixels
- pid = tl.program_id(0)
-
- # Total number of pixels
- n_pixels = H * W
-
- # Calculate pixel indices this block will process
+ pid = tl.program_id(axis=0)
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
-
- # Mask for out-of-bounds pixels
mask = offsets < n_pixels
-
- # Convert linear index to 2D coordinates (row, col)
- row = offsets // W
- col = offsets % W
-
- # Calculate input pointer offsets for each channel
- # Input layout is (H, W, 3) so offset = row * stride_h + col * stride_w + channel * stride_c
- base_offset = row * stride_h + col * stride_w
-
- # Load R, G, B values
- r_vals = tl.load(rgb_ptr + base_offset + 0 * stride_c, mask=mask, other=0.0)
- g_vals = tl.load(rgb_ptr + base_offset + 1 * stride_c, mask=mask, other=0.0)
- b_vals = tl.load(rgb_ptr + base_offset + 2 * stride_c, mask=mask, other=0.0)
-
- # Compute grayscale using standard coefficients
- gray_vals = R_COEFF * r_vals + G_COEFF * g_vals + B_COEFF * b_vals
-
- # Calculate output pointer offsets
- out_offset = row * out_stride_h + col * out_stride_w
-
- # Store result
- tl.store(gray_ptr + out_offset, gray_vals, mask=mask)
+ # Input layout: (H, W, 3) contiguous => pixel i starts at index (i * 3)
+ base = offsets * 3
- def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
+ r = tl.load(x_ptr + base + 0, mask=mask, other=0.0)
+ g = tl.load(x_ptr + base + 1, mask=mask, other=0.0)
+ b = tl.load(x_ptr + base + 2, mask=mask, other=0.0)
+
+ # Compute in fp32 for numerical stability
+ r32 = r.to(tl.float32)
+ g32 = g.to(tl.float32)
+ b32 = b.to(tl.float32)
+
+ # Y = 0.2989 R + 0.5870 G + 0.1140 B
+ y32 = r32 * w_r + g32 * w_g + b32 * w_b
+
+ # Cast back to output dtype on store
+ y = y32.to(y_ptr.dtype.element_ty)
+ tl.store(y_ptr + offsets, y, mask=mask)
+
+
+ def kernel_function(x: torch.Tensor, y_out: torch.Tensor = None):
"""
- RGB to grayscale conversion using Triton kernel.
-
- Fused operation: Single kernel pass that loads RGB values and computes
- grayscale output using Y = 0.2989*R + 0.5870*G + 0.1140*B.
-
+ Triton RGB -> Grayscale conversion (fused in a single pass).
+
+ What is fused:
+ - Load R, G, B for each pixel
+ - Convert to fp32
+ - Apply grayscale weights and accumulate
+ - Cast to output dtype and store
+ All steps are fused within one Triton kernel, avoiding intermediate global-memory traffic.
+
Args:
- rgb_input: Input RGB tensor of shape (H, W, 3) with dtype float32
- output: Output grayscale tensor of shape (H, W) with dtype float32
-
+ x: Input tensor of shape (H, W, 3), contiguous, on CUDA. dtypes supported: float32, bfloat16, float16.
+ y_out: Optional preallocated output tensor of shape (H, W) and same dtype/device as x.
+
Returns:
- The output tensor containing grayscale values
+ Tensor of shape (H, W) with grayscale values in the same dtype/device as x.
"""
- # Validate inputs
- assert rgb_input.is_cuda, "Input must be on CUDA device"
- assert output.is_cuda, "Output must be on CUDA device"
- assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
- assert rgb_input.is_contiguous(), "Input must be contiguous"
- assert output.is_contiguous(), "Output must be contiguous"
-
- H, W, C = rgb_input.shape
- assert output.shape == (H, W), f"Output shape must be ({H}, {W})"
-
- # Total number of pixels
+ # Argument validation (no math in wrapper)
+ assert x.is_cuda, "Input x must be on CUDA device"
+ assert x.ndim == 3 and x.shape[-1] == 3, "Input must have shape (H, W, 3)"
+ assert x.is_contiguous(), "Input must be contiguous (H, W, 3) layout"
+ assert x.dtype in (torch.float32, torch.bfloat16, torch.float16), "Supported dtypes: float32, bfloat16, float16"
+
+ H, W, C = x.shape
n_pixels = H * W
-
- # Block size - number of pixels per block
+
+ # Prepare output buffer
+ if y_out is None:
+ y_out = torch.empty((H, W), device=x.device, dtype=x.dtype)
+ else:
+ assert y_out.is_cuda, "Output must be on CUDA"
+ assert y_out.shape == (H, W), f"Output must have shape {(H, W)}"
+ assert y_out.dtype == x.dtype, "Output dtype must match input dtype"
+ assert y_out.is_contiguous(), "Output must be contiguous"
+
+ # Weights for grayscale conversion
+ w_r = 0.2989
+ w_g = 0.5870
+ w_b = 0.1140
+
+ # Launch configuration
+ # Use a 1D grid over pixels; BLOCK_SIZE chosen as a power of two for good occupancy
BLOCK_SIZE = 1024
-
- # Calculate grid size
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
-
- # Get strides (in elements, not bytes)
- stride_h = rgb_input.stride(0)
- stride_w = rgb_input.stride(1)
- stride_c = rgb_input.stride(2)
- out_stride_h = output.stride(0)
- out_stride_w = output.stride(1)
-
- # Launch kernel
+
+ # Launch Triton kernel
_rgb_to_grayscale_kernel[grid](
- rgb_input,
- output,
- H,
- W,
- stride_h,
- stride_w,
- stride_c,
- out_stride_h,
- out_stride_w,
+ x, y_out, n_pixels,
+ w_r, w_g, w_b,
BLOCK_SIZE=BLOCK_SIZE,
+ num_warps=4, # reasonable default for elementwise workloads
+ num_stages=2,
)
-
- return output
+ return y_out
+
import inspect
def custom_kernel(input):
scrolls · 198 diff lines total

Best evidence level for this revision: reported

JSON