submission 640270
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 93 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_v2_H100_gpt-5-2_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-640270?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:d74ca7b97fd6581a3924053db8c72162b016794b040399fc5e1548dfaf89b8b3
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 4
num_warps=4,Kernel source
grayscale_v2_H100_gpt-5-2_ka_submission.py93 lines
# kernel.py
"""
RGB -> Grayscale Triton kernel.
Fused pipeline (single pass):
1) Load RGB (float32) pixels from contiguous (H, W, 3) tensor
2) Compute grayscale: Y = 0.2989 * R + 0.5870 * G + 0.1140 * B
3) Store to contiguous (H, W) float32 output
No PyTorch math is used in the wrapper; all numerical work happens in the Triton kernel.
"""
from __future__ import annotations
import torch
import triton
import triton.language as tl
@triton.jit
def _rgb_to_gray_kernel(
x_ptr, # *fp32, flattened RGB: length = n_pixels * 3
y_ptr, # *fp32, flattened gray: length = n_pixels
n_pixels: tl.int32,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(axis=0)
offs_p = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offs_p < n_pixels
base = offs_p * 3
r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)
g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)
b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)
y = r * 0.2989 + g * 0.5870 + b * 0.1140
tl.store(y_ptr + offs_p, y, mask=mask)
def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:
"""
Args:
x: (H, W, 3) float32 CUDA tensor, contiguous
y: optional (H, W) float32 CUDA tensor, contiguous (written in-place)
Returns:
(H, W) float32 CUDA tensor (same object as `y` if provided)
"""
assert isinstance(x, torch.Tensor)
assert x.is_cuda, "x must be a CUDA tensor"
assert x.dtype == torch.float32, "x must be float32"
assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"
assert x.is_contiguous(), "x must be contiguous"
H, W, _ = x.shape
n_pixels = H * W
if y is None:
y = torch.empty((H, W), device=x.device, dtype=torch.float32)
else:
assert isinstance(y, torch.Tensor)
assert y.is_cuda and y.device == x.device, "y must be on same CUDA device as x"
assert y.dtype == torch.float32, "y must be float32"
assert y.shape == (H, W), "y must have shape (H, W)"
assert y.is_contiguous(), "y must be contiguous"
BLOCK_SIZE = 1024
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
_rgb_to_gray_kernel[grid](
x, y,
n_pixels,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=4,
)
return y
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 93 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 584793.
+ # kernel.py+ """+ RGB -> Grayscale Triton kernel.++ Fused pipeline (single pass):+ 1) Load RGB (float32) pixels from contiguous (H, W, 3) tensor+ 2) Compute grayscale: Y = 0.2989 * R + 0.5870 * G + 0.1140 * B+ 3) Store to contiguous (H, W) float32 output++ No PyTorch math is used in the wrapper; all numerical work happens in the Triton kernel.+ """++ from __future__ import annotations++ import torchimport tritonimport triton.language as tl- import torch@triton.jit- def _rgb_to_grayscale_kernel(- input_ptr,- output_ptr,- n_pixels,+ def _rgb_to_gray_kernel(+ x_ptr, # *fp32, flattened RGB: length = n_pixels * 3+ y_ptr, # *fp32, flattened gray: length = n_pixels+ n_pixels: tl.int32,BLOCK_SIZE: tl.constexpr,):- """- Fused RGB to grayscale conversion kernel.- Computes Y = 0.2989 * R + 0.5870 * G + 0.1140 * B in a single pass.-- Fusion: channel loads, weighted multiply-accumulate, and store are all- fused into one kernel. No intermediate buffers needed.- """- pid = tl.program_id(0)- pixel_offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)- mask = pixel_offsets < n_pixels+ pid = tl.program_id(axis=0)+ offs_p = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ mask = offs_p < n_pixels- # Input layout is (H, W, 3) contiguous: pixel i has R,G,B at i*3+0,1,2- base_offsets = pixel_offsets * 3+ base = offs_p * 3+ r = tl.load(x_ptr + base + 0, mask=mask, other=0.0).to(tl.float32)+ g = tl.load(x_ptr + base + 1, mask=mask, other=0.0).to(tl.float32)+ b = tl.load(x_ptr + base + 2, mask=mask, other=0.0).to(tl.float32)- # Load R, G, B channels- r = tl.load(input_ptr + base_offsets, mask=mask, other=0.0)- g = tl.load(input_ptr + base_offsets + 1, mask=mask, other=0.0)- b = tl.load(input_ptr + base_offsets + 2, mask=mask, other=0.0)+ y = r * 0.2989 + g * 0.5870 + b * 0.1140+ tl.store(y_ptr + offs_p, y, mask=mask)- # Compute grayscale with standard NTSC/PAL luminance weights- gray = r * 0.2989 + g * 0.5870 + b * 0.1140- # Store result- tl.store(output_ptr + pixel_offsets, gray, mask=mask)--- def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:+ def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:"""- Wrapper for RGB to grayscale conversion.-Args:- rgb_input: Input tensor of shape (H, W, 3), dtype float32, on CUDA.- output: Pre-allocated output tensor of shape (H, W), dtype float32, on CUDA.+ x: (H, W, 3) float32 CUDA tensor, contiguous+ y: optional (H, W) float32 CUDA tensor, contiguous (written in-place)Returns:- The output tensor containing grayscale values.+ (H, W) float32 CUDA tensor (same object as `y` if provided)"""- H, W, C = rgb_input.shape+ assert isinstance(x, torch.Tensor)+ assert x.is_cuda, "x must be a CUDA tensor"+ assert x.dtype == torch.float32, "x must be float32"+ assert x.ndim == 3 and x.shape[2] == 3, "x must have shape (H, W, 3)"+ assert x.is_contiguous(), "x must be contiguous"++ H, W, _ = x.shapen_pixels = H * W+ if y is None:+ y = torch.empty((H, W), device=x.device, dtype=torch.float32)+ else:+ assert isinstance(y, torch.Tensor)+ assert y.is_cuda and y.device == x.device, "y must be on same CUDA device as x"+ assert y.dtype == torch.float32, "y must be float32"+ assert y.shape == (H, W), "y must have shape (H, W)"+ assert y.is_contiguous(), "y must be contiguous"+BLOCK_SIZE = 1024grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)-- _rgb_to_grayscale_kernel[grid](- rgb_input,- output,+ _rgb_to_gray_kernel[grid](+ x, y,n_pixels,BLOCK_SIZE=BLOCK_SIZE,+ num_warps=4,)+ return y- return output-import inspect+def custom_kernel(input):sig = inspect.signature(kernel_function)num_params = len(sig.parameters)+if len(input) == num_params:return kernel_function(*input)return kernel_function(input)++ # Ensure deterministic cuBLAS.import osif os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"+
scrolls · 132 diff lines total
Best evidence level for this revision: reported
JSON