Skip to content
KernelIndex
Search⌘K

submission 640270

KernelAgent · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 93 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_v2_H100_gpt-5-2_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-640270?include=source"
interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA H100
1.19ms
#2 of 36
2026-03-26

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:d74ca7b97fd6581a3924053db8c72162b016794b040399fc5e1548dfaf89b8b3
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 4num_warps=4,

Kernel source

grayscale_v2_H100_gpt-5-2_ka_submission.py93 lines
# kernel.py
"""
RGB -> Grayscale Triton kernel.

Fused pipeline (single pass):
  1) Load RGB (float32) pixels from contiguous (H, W, 3) tensor
  2) Compute grayscale: Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
  3) Store to contiguous (H, W) float32 output

No PyTorch math is used in the wrapper; all numerical work happens in the Triton kernel.
"""

from __future__ import annotations

import torch
import triton
import triton.language as tl


@triton.jit
def _rgb_to_gray_kernel(
    x_ptr,  # *fp32, flattened RGB: length = n_pixels * 3
    y_ptr,  # *fp32, flattened gray: length = n_pixels
    n_pixels: tl.int32,
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(axis=0)
    offs_p = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offs_p < n_pixels

    base = offs_p * 3
    r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)
    g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)
    b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)

    y = r * 0.2989 + g * 0.5870 + b * 0.1140
    tl.store(y_ptr + offs_p, y, mask=mask)


def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:
    """
    Args:
        x: (H, W, 3) float32 CUDA tensor, contiguous
        y: optional (H, W) float32 CUDA tensor, contiguous (written in-place)

    Returns:
        (H, W) float32 CUDA tensor (same object as `y` if provided)
    """
    assert isinstance(x, torch.Tensor)
    assert x.is_cuda, "x must be a CUDA tensor"
    assert x.dtype == torch.float32, "x must be float32"
    assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"
    assert x.is_contiguous(), "x must be contiguous"

    H, W, _ = x.shape
    n_pixels = H * W

    if y is None:
        y = torch.empty((H, W), device=x.device, dtype=torch.float32)
    else:
        assert isinstance(y, torch.Tensor)
        assert y.is_cuda and y.device == x.device, "y must be on same CUDA device as x"
        assert y.dtype == torch.float32, "y must be float32"
        assert y.shape == (H, W), "y must have shape (H, W)"
        assert y.is_contiguous(), "y must be contiguous"

    BLOCK_SIZE = 1024
    grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
    _rgb_to_gray_kernel[grid](
        x, y,
        n_pixels,
        BLOCK_SIZE=BLOCK_SIZE,
        num_warps=4,
    )
    return y

import inspect

def custom_kernel(input):
    sig = inspect.signature(kernel_function)
    num_params = len(sig.parameters)

    if len(input) == num_params:
        return kernel_function(*input)
    return kernel_function(input)


# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
    os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"

scrolls · 93 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 584793.

+ # kernel.py
+ """
+ RGB -> Grayscale Triton kernel.
+
+ Fused pipeline (single pass):
+ 1) Load RGB (float32) pixels from contiguous (H, W, 3) tensor
+ 2) Compute grayscale: Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
+ 3) Store to contiguous (H, W) float32 output
+
+ No PyTorch math is used in the wrapper; all numerical work happens in the Triton kernel.
+ """
+
+ from __future__ import annotations
+
+ import torch
import triton
import triton.language as tl
- import torch
@triton.jit
- def _rgb_to_grayscale_kernel(
- input_ptr,
- output_ptr,
- n_pixels,
+ def _rgb_to_gray_kernel(
+ x_ptr, # *fp32, flattened RGB: length = n_pixels * 3
+ y_ptr, # *fp32, flattened gray: length = n_pixels
+ n_pixels: tl.int32,
BLOCK_SIZE: tl.constexpr,
):
- """
- Fused RGB to grayscale conversion kernel.
- Computes Y = 0.2989 * R + 0.5870 * G + 0.1140 * B in a single pass.
-
- Fusion: channel loads, weighted multiply-accumulate, and store are all
- fused into one kernel. No intermediate buffers needed.
- """
- pid = tl.program_id(0)
- pixel_offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
- mask = pixel_offsets < n_pixels
+ pid = tl.program_id(axis=0)
+ offs_p = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
+ mask = offs_p < n_pixels
- # Input layout is (H, W, 3) contiguous: pixel i has R,G,B at i*3+0,1,2
- base_offsets = pixel_offsets * 3
+ base = offs_p * 3
+ r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)
+ g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)
+ b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)
- # Load R, G, B channels
- r = tl.load(input_ptr + base_offsets, mask=mask, other=0.0)
- g = tl.load(input_ptr + base_offsets + 1, mask=mask, other=0.0)
- b = tl.load(input_ptr + base_offsets + 2, mask=mask, other=0.0)
+ y = r * 0.2989 + g * 0.5870 + b * 0.1140
+ tl.store(y_ptr + offs_p, y, mask=mask)
- # Compute grayscale with standard NTSC/PAL luminance weights
- gray = r * 0.2989 + g * 0.5870 + b * 0.1140
- # Store result
- tl.store(output_ptr + pixel_offsets, gray, mask=mask)
-
-
- def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
+ def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:
"""
- Wrapper for RGB to grayscale conversion.
-
Args:
- rgb_input: Input tensor of shape (H, W, 3), dtype float32, on CUDA.
- output: Pre-allocated output tensor of shape (H, W), dtype float32, on CUDA.
+ x: (H, W, 3) float32 CUDA tensor, contiguous
+ y: optional (H, W) float32 CUDA tensor, contiguous (written in-place)
Returns:
- The output tensor containing grayscale values.
+ (H, W) float32 CUDA tensor (same object as `y` if provided)
"""
- H, W, C = rgb_input.shape
+ assert isinstance(x, torch.Tensor)
+ assert x.is_cuda, "x must be a CUDA tensor"
+ assert x.dtype == torch.float32, "x must be float32"
+ assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"
+ assert x.is_contiguous(), "x must be contiguous"
+
+ H, W, _ = x.shape
n_pixels = H * W
+ if y is None:
+ y = torch.empty((H, W), device=x.device, dtype=torch.float32)
+ else:
+ assert isinstance(y, torch.Tensor)
+ assert y.is_cuda and y.device == x.device, "y must be on same CUDA device as x"
+ assert y.dtype == torch.float32, "y must be float32"
+ assert y.shape == (H, W), "y must have shape (H, W)"
+ assert y.is_contiguous(), "y must be contiguous"
+
BLOCK_SIZE = 1024
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
-
- _rgb_to_grayscale_kernel[grid](
- rgb_input,
- output,
+ _rgb_to_gray_kernel[grid](
+ x, y,
n_pixels,
BLOCK_SIZE=BLOCK_SIZE,
+ num_warps=4,
)
+ return y
- return output
-
import inspect
+
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
+
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
+
+ # Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
+
scrolls · 132 diff lines total

Best evidence level for this revision: reported

JSON