Skip to content
KernelIndex
Search⌘K

submission 510652

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 90 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_v2_H100_gpt-5-2_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-510652?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.37ms
#19 of 36
2026-02-27

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:5264ca76d6fe9150bac46844e38cf238895ca85a2fdb26ee53bb4c3831fe4ad5
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotune@triton.autotune(
num-warps = 4triton.Config({"BLOCK": 256}, num_warps=4),

Kernel source

grayscale_v2_H100_gpt-5-2_ka_submission.py90 lines
# kernel.py
import torch
import triton
import triton.language as tl


@triton.autotune(
    configs=[
        triton.Config({"BLOCK": 256}, num_warps=4),
        triton.Config({"BLOCK": 512}, num_warps=8),
        triton.Config({"BLOCK": 1024}, num_warps=8),
    ],
    key=["N"],
)
@triton.jit
def _rgb_to_gray_kernel(
    x_ptr,  # *[H, W, 3]
    y_ptr,  # *[H, W]
    N,      # H*W
    BLOCK: tl.constexpr,
):
    pid = tl.program_id(axis=0)
    offs = pid * BLOCK + tl.arange(0, BLOCK)
    mask = offs < N

    # x is contiguous [H, W, 3] => flattened pixels are contiguous triplets
    base = offs * 3

    r = tl.load(x_ptr + base + 0, mask=mask, other=0).to(tl.float32)
    g = tl.load(x_ptr + base + 1, mask=mask, other=0).to(tl.float32)
    b = tl.load(x_ptr + base + 2, mask=mask, other=0).to(tl.float32)

    # Match reference semantics more closely by quantizing weights to x dtype first.
    x_ty = x_ptr.dtype.element_ty
    w_r = tl.full((), 0.2989, tl.float32).to(x_ty).to(tl.float32)
    w_g = tl.full((), 0.5870, tl.float32).to(x_ty).to(tl.float32)
    w_b = tl.full((), 0.1140, tl.float32).to(x_ty).to(tl.float32)

    y_f32 = r * w_r + g * w_g + b * w_b
    tl.store(y_ptr + offs, y_f32.to(y_ptr.dtype.element_ty), mask=mask)


def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:
    # Validation / allocation only (no compute here).
    if not isinstance(x, torch.Tensor):
        raise TypeError("x must be a torch.Tensor")
    if not x.is_cuda:
        raise ValueError("x must be a CUDA tensor")
    if x.ndim != 3 or x.shape[-1] != 3:
        raise ValueError(f"x must have shape [H, W, 3], got {tuple(x.shape)}")
    if not x.is_contiguous():
        raise ValueError("x must be contiguous (expected contiguous [H, W, 3])")

    H, W, _ = x.shape
    if y is None:
        y = torch.empty((H, W), device=x.device, dtype=x.dtype)
    else:
        if not isinstance(y, torch.Tensor):
            raise TypeError("y must be a torch.Tensor")
        if not y.is_cuda:
            raise ValueError("y must be a CUDA tensor")
        if tuple(y.shape) != (H, W):
            raise ValueError(f"y must have shape {(H, W)}, got {tuple(y.shape)}")
        if not y.is_contiguous():
            raise ValueError("y must be contiguous")

    N = H * W

    grid = lambda META: (triton.cdiv(N, META["BLOCK"]),)
    _rgb_to_gray_kernel[grid](x, y, N)

    return y

import inspect

def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)

    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)


# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"

scrolls · 90 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 510650.

+ # kernel.py
+ import torch
import triton
import triton.language as tl
- import torch
+ @triton.autotune(
+ configs=[
+ triton.Config({"BLOCK": 256}, num_warps=4),
+ triton.Config({"BLOCK": 512}, num_warps=8),
+ triton.Config({"BLOCK": 1024}, num_warps=8),
+ ],
+ key=["N"],
+ )
@triton.jit
- def _rgb_to_grayscale_kernel(
- rgb_ptr, # Pointer to input RGB image (H, W, 3)
- out_ptr, # Pointer to output grayscale image (H, W)
- H, # Image height
- W, # Image width
- stride_h, # Stride for height dimension in RGB
- stride_w, # Stride for width dimension in RGB
- stride_c, # Stride for channel dimension in RGB
- out_stride_h, # Stride for height dimension in output
- out_stride_w, # Stride for width dimension in output
- BLOCK_SIZE: tl.constexpr, # Number of pixels to process per block
+ def _rgb_to_gray_kernel(
+ x_ptr, # *[H, W, 3]
+ y_ptr, # *[H, W]
+ N, # H*W
+ BLOCK: tl.constexpr,
):
- """
- Triton kernel for RGB to grayscale conversion.
-
- Fused operation: For each pixel, loads R, G, B values and computes
- the weighted sum Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass.
- """
- # Standard RGB to grayscale coefficients
- R_COEFF = 0.2989
- G_COEFF = 0.5870
- B_COEFF = 0.1140
-
- # Each program processes BLOCK_SIZE pixels
- pid = tl.program_id(0)
-
- # Total number of pixels
- n_pixels = H * W
-
- # Calculate pixel indices this block will process
- block_start = pid * BLOCK_SIZE
- offsets = block_start + tl.arange(0, BLOCK_SIZE)
-
- # Mask for valid pixels
- mask = offsets < n_pixels
-
- # Convert linear index to 2D coordinates (row-major order)
- # pixel_idx = h * W + w
- h_idx = offsets // W
- w_idx = offsets % W
-
- # Calculate base offset for each pixel in RGB image
- # RGB layout is (H, W, 3), so offset = h * stride_h + w * stride_w + c * stride_c
- base_offset = h_idx * stride_h + w_idx * stride_w
-
- # Load R, G, B values for each pixel
- r_offset = base_offset + 0 * stride_c # Channel 0 = R
- g_offset = base_offset + 1 * stride_c # Channel 1 = G
- b_offset = base_offset + 2 * stride_c # Channel 2 = B
-
- r = tl.load(rgb_ptr + r_offset, mask=mask, other=0.0)
- g = tl.load(rgb_ptr + g_offset, mask=mask, other=0.0)
- b = tl.load(rgb_ptr + b_offset, mask=mask, other=0.0)
-
- # Compute grayscale value using standard coefficients
- gray = R_COEFF * r + G_COEFF * g + B_COEFF * b
-
- # Calculate output offset
- out_offset = h_idx * out_stride_h + w_idx * out_stride_w
-
- # Store the result
- tl.store(out_ptr + out_offset, gray, mask=mask)
+ pid = tl.program_id(axis=0)
+ offs = pid * BLOCK + tl.arange(0, BLOCK)
+ mask = offs < N
+ # x is contiguous [H, W, 3] => flattened pixels are contiguous triplets
+ base = offs * 3
- def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
- """
- Wrapper function for RGB to grayscale conversion.
-
- This function performs a fused RGB to grayscale conversion where:
- - Loading of R, G, B channels
- - Weighted sum computation (Y = 0.2989*R + 0.5870*G + 0.1140*B)
- - Storage of grayscale result
- are all done in a single kernel pass, minimizing memory traffic.
-
- Args:
- rgb_input: Input RGB tensor of shape (H, W, 3), dtype float32
- output: Output buffer of shape (H, W), dtype float32
-
- Returns:
- The output tensor containing grayscale values
- """
- # Validate inputs
- assert rgb_input.is_cuda, "Input must be on CUDA device"
- assert output.is_cuda, "Output must be on CUDA device"
- assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"
- assert output.ndim == 2, "Output must be (H, W)"
- assert rgb_input.shape[0] == output.shape[0] and rgb_input.shape[1] == output.shape[1], \
- "Input and output spatial dimensions must match"
-
- H, W, C = rgb_input.shape
- n_pixels = H * W
-
- # Get strides (in elements, not bytes)
- stride_h = rgb_input.stride(0)
- stride_w = rgb_input.stride(1)
- stride_c = rgb_input.stride(2)
- out_stride_h = output.stride(0)
- out_stride_w = output.stride(1)
-
- # Choose block size - power of 2 for efficiency
- BLOCK_SIZE = 1024
-
- # Calculate grid size
- grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
-
- # Launch kernel
- _rgb_to_grayscale_kernel[grid](
- rgb_input,
- output,
- H,
- W,
- stride_h,
- stride_w,
- stride_c,
- out_stride_h,
- out_stride_w,
- BLOCK_SIZE=BLOCK_SIZE,
- )
-
- return output
+ r = tl.load(x_ptr + base + 0, mask=mask, other=0).to(tl.float32)
+ g = tl.load(x_ptr + base + 1, mask=mask, other=0).to(tl.float32)
+ b = tl.load(x_ptr + base + 2, mask=mask, other=0).to(tl.float32)
+ # Match reference semantics more closely by quantizing weights to x dtype first.
+ x_ty = x_ptr.dtype.element_ty
+ w_r = tl.full((), 0.2989, tl.float32).to(x_ty).to(tl.float32)
+ w_g = tl.full((), 0.5870, tl.float32).to(x_ty).to(tl.float32)
+ w_b = tl.full((), 0.1140, tl.float32).to(x_ty).to(tl.float32)
+
+ y_f32 = r * w_r + g * w_g + b * w_b
+ tl.store(y_ptr + offs, y_f32.to(y_ptr.dtype.element_ty), mask=mask)
+
+
+ def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:
+ # Validation / allocation only (no compute here).
+ if not isinstance(x, torch.Tensor):
+ raise TypeError("x must be a torch.Tensor")
+ if not x.is_cuda:
+ raise ValueError("x must be a CUDA tensor")
+ if x.ndim != 3 or x.shape[-1] != 3:
+ raise ValueError(f"x must have shape [H, W, 3], got {tuple(x.shape)}")
+ if not x.is_contiguous():
+ raise ValueError("x must be contiguous (expected contiguous [H, W, 3])")
+
+ H, W, _ = x.shape
+ if y is None:
+ y = torch.empty((H, W), device=x.device, dtype=x.dtype)
+ else:
+ if not isinstance(y, torch.Tensor):
+ raise TypeError("y must be a torch.Tensor")
+ if not y.is_cuda:
+ raise ValueError("y must be a CUDA tensor")
+ if tuple(y.shape) != (H, W):
+ raise ValueError(f"y must have shape {(H, W)}, got {tuple(y.shape)}")
+ if not y.is_contiguous():
+ raise ValueError("y must be contiguous")
+
+ N = H * W
+
+ grid = lambda META: (triton.cdiv(N, META["BLOCK"]),)
+ _rgb_to_gray_kernel[grid](x, y, N)
+
+ return y
+
import inspect
def custom_kernel(input):
scrolls · 194 diff lines total

Best evidence level for this revision: reported

JSON