submission 490599
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 114 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_py_H100_gpt-5_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-490599?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:13962b423e0017e6830228a0033fa44a16be46e59c5554d38e440e7093a8d68b
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 4
num_warps=4, # reasonable default for elementwise workloadsstages = 2
num_stages=2,Kernel source
grayscale_py_H100_gpt-5_ka_submission.py114 lines
import torch
import triton
import triton.language as tl
@triton.jit
def _rgb_to_grayscale_kernel(x_ptr, y_ptr, n_pixels, #
w_r, w_g, w_b, #
BLOCK_SIZE: tl.constexpr):
"""
Elementwise RGB -> Grayscale conversion for a flat list of pixels.
Each program processes up to BLOCK_SIZE pixels. Pixels are laid out as
contiguous triplets [R, G, B] in memory, i.e., the original image is (H, W, 3) contiguous.
"""
pid = tl.program_id(axis=0)
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_pixels
# Input layout: (H, W, 3) contiguous => pixel i starts at index (i * 3)
base = offsets * 3
r = tl.load(x_ptr + base + 0, mask=mask, other=0.0)
g = tl.load(x_ptr + base + 1, mask=mask, other=0.0)
b = tl.load(x_ptr + base + 2, mask=mask, other=0.0)
# Compute in fp32 for numerical stability
r32 = r.to(tl.float32)
g32 = g.to(tl.float32)
b32 = b.to(tl.float32)
# Y = 0.2989 R + 0.5870 G + 0.1140 B
y32 = r32 * w_r + g32 * w_g + b32 * w_b
# Cast back to output dtype on store
y = y32.to(y_ptr.dtype.element_ty)
tl.store(y_ptr + offsets, y, mask=mask)
def kernel_function(x: torch.Tensor, y_out: torch.Tensor = None):
"""
Triton RGB -> Grayscale conversion (fused in a single pass).
What is fused:
- Load R, G, B for each pixel
- Convert to fp32
- Apply grayscale weights and accumulate
- Cast to output dtype and store
All steps are fused within one Triton kernel, avoiding intermediate global-memory traffic.
Args:
x: Input tensor of shape (H, W, 3), contiguous, on CUDA. dtypes supported: float32, bfloat16, float16.
y_out: Optional preallocated output tensor of shape (H, W) and same dtype/device as x.
Returns:
Tensor of shape (H, W) with grayscale values in the same dtype/device as x.
"""
# Argument validation (no math in wrapper)
assert x.is_cuda, "Input x must be on CUDA device"
assert x.ndim == 3 and x.shape[-1] == 3, "Input must have shape (H, W, 3)"
assert x.is_contiguous(), "Input must be contiguous (H, W, 3) layout"
assert x.dtype in (torch.float32, torch.bfloat16, torch.float16), "Supported dtypes: float32, bfloat16, float16"
H, W, C = x.shape
n_pixels = H * W
# Prepare output buffer
if y_out is None:
y_out = torch.empty((H, W), device=x.device, dtype=x.dtype)
else:
assert y_out.is_cuda, "Output must be on CUDA"
assert y_out.shape == (H, W), f"Output must have shape {(H, W)}"
assert y_out.dtype == x.dtype, "Output dtype must match input dtype"
assert y_out.is_contiguous(), "Output must be contiguous"
# Weights for grayscale conversion
w_r = 0.2989
w_g = 0.5870
w_b = 0.1140
# Launch configuration
# Use a 1D grid over pixels; BLOCK_SIZE chosen as a power of two for good occupancy
BLOCK_SIZE = 1024
grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)
# Launch Triton kernel
_rgb_to_grayscale_kernel[grid](
x, y_out, n_pixels,
w_r, w_g, w_b,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=4, # reasonable default for elementwise workloads
num_stages=2,
)
return y_out
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 114 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 490598.
+ import torchimport tritonimport triton.language as tl- import torch@triton.jit- def _rgb_to_grayscale_kernel(- rgb_ptr, # Pointer to input RGB tensor (H, W, 3)- gray_ptr, # Pointer to output grayscale tensor (H, W)- H, # Height of image- W, # Width of image- stride_h, # Stride for height dimension in RGB tensor- stride_w, # Stride for width dimension in RGB tensor- stride_c, # Stride for channel dimension in RGB tensor- out_stride_h, # Stride for height dimension in output tensor- out_stride_w, # Stride for width dimension in output tensor- BLOCK_SIZE: tl.constexpr, # Number of pixels to process per block- ):+ def _rgb_to_grayscale_kernel(x_ptr, y_ptr, n_pixels, #+ w_r, w_g, w_b, #+ BLOCK_SIZE: tl.constexpr):"""- Triton kernel for RGB to grayscale conversion.-- Fused operation: For each pixel, loads R, G, B values and computes- Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass.+ Elementwise RGB -> Grayscale conversion for a flat list of pixels.++ Each program processes up to BLOCK_SIZE pixels. Pixels are laid out as+ contiguous triplets [R, G, B] in memory, i.e., the original image is (H, W, 3) contiguous."""- # Standard RGB to grayscale coefficients- R_COEFF = 0.2989- G_COEFF = 0.5870- B_COEFF = 0.1140-- # Get program ID - each program handles BLOCK_SIZE pixels- pid = tl.program_id(0)-- # Total number of pixels- n_pixels = H * W-- # Calculate pixel indices this block will process+ pid = tl.program_id(axis=0)block_start = pid * BLOCK_SIZEoffsets = block_start + tl.arange(0, BLOCK_SIZE)-- # Mask for out-of-bounds pixelsmask = offsets < n_pixels-- # Convert linear index to 2D coordinates (row, col)- row = offsets // W- col = offsets % W-- # Calculate input pointer offsets for each channel- # Input layout is (H, W, 3) so offset = row * stride_h + col * stride_w + channel * stride_c- base_offset = row * stride_h + col * stride_w-- # Load R, G, B values- r_vals = tl.load(rgb_ptr + base_offset + 0 * stride_c, mask=mask, other=0.0)- g_vals = tl.load(rgb_ptr + base_offset + 1 * stride_c, mask=mask, other=0.0)- b_vals = tl.load(rgb_ptr + base_offset + 2 * stride_c, mask=mask, other=0.0)-- # Compute grayscale using standard coefficients- gray_vals = R_COEFF * r_vals + G_COEFF * g_vals + B_COEFF * b_vals-- # Calculate output pointer offsets- out_offset = row * out_stride_h + col * out_stride_w-- # Store result- tl.store(gray_ptr + out_offset, gray_vals, mask=mask)+ # Input layout: (H, W, 3) contiguous => pixel i starts at index (i * 3)+ base = offsets * 3- def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:+ r = tl.load(x_ptr + base + 0, mask=mask, other=0.0)+ g = tl.load(x_ptr + base + 1, mask=mask, other=0.0)+ b = tl.load(x_ptr + base + 2, mask=mask, other=0.0)++ # Compute in fp32 for numerical stability+ r32 = r.to(tl.float32)+ g32 = g.to(tl.float32)+ b32 = b.to(tl.float32)++ # Y = 0.2989 R + 0.5870 G + 0.1140 B+ y32 = r32 * w_r + g32 * w_g + b32 * w_b++ # Cast back to output dtype on store+ y = y32.to(y_ptr.dtype.element_ty)+ tl.store(y_ptr + offsets, y, mask=mask)+++ def kernel_function(x: torch.Tensor, y_out: torch.Tensor = None):"""- RGB to grayscale conversion using Triton kernel.-- Fused operation: Single kernel pass that loads RGB values and computes- grayscale output using Y = 0.2989*R + 0.5870*G + 0.1140*B.-+ Triton RGB -> Grayscale conversion (fused in a single pass).++ What is fused:+ - Load R, G, B for each pixel+ - Convert to fp32+ - Apply grayscale weights and accumulate+ - Cast to output dtype and store+ All steps are fused within one Triton kernel, avoiding intermediate global-memory traffic.+Args:- rgb_input: Input RGB tensor of shape (H, W, 3) with dtype float32- output: Output grayscale tensor of shape (H, W) with dtype float32-+ x: Input tensor of shape (H, W, 3), contiguous, on CUDA. dtypes supported: float32, bfloat16, float16.+ y_out: Optional preallocated output tensor of shape (H, W) and same dtype/device as x.+Returns:- The output tensor containing grayscale values+ Tensor of shape (H, W) with grayscale values in the same dtype/device as x."""- # Validate inputs- assert rgb_input.is_cuda, "Input must be on CUDA device"- assert output.is_cuda, "Output must be on CUDA device"- assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"- assert rgb_input.is_contiguous(), "Input must be contiguous"- assert output.is_contiguous(), "Output must be contiguous"-- H, W, C = rgb_input.shape- assert output.shape == (H, W), f"Output shape must be ({H}, {W})"-- # Total number of pixels+ # Argument validation (no math in wrapper)+ assert x.is_cuda, "Input x must be on CUDA device"+ assert x.ndim == 3 and x.shape[-1] == 3, "Input must have shape (H, W, 3)"+ assert x.is_contiguous(), "Input must be contiguous (H, W, 3) layout"+ assert x.dtype in (torch.float32, torch.bfloat16, torch.float16), "Supported dtypes: float32, bfloat16, float16"++ H, W, C = x.shapen_pixels = H * W-- # Block size - number of pixels per block++ # Prepare output buffer+ if y_out is None:+ y_out = torch.empty((H, W), device=x.device, dtype=x.dtype)+ else:+ assert y_out.is_cuda, "Output must be on CUDA"+ assert y_out.shape == (H, W), f"Output must have shape {(H, W)}"+ assert y_out.dtype == x.dtype, "Output dtype must match input dtype"+ assert y_out.is_contiguous(), "Output must be contiguous"++ # Weights for grayscale conversion+ w_r = 0.2989+ w_g = 0.5870+ w_b = 0.1140++ # Launch configuration+ # Use a 1D grid over pixels; BLOCK_SIZE chosen as a power of two for good occupancyBLOCK_SIZE = 1024-- # Calculate grid sizegrid = (triton.cdiv(n_pixels, BLOCK_SIZE),)-- # Get strides (in elements, not bytes)- stride_h = rgb_input.stride(0)- stride_w = rgb_input.stride(1)- stride_c = rgb_input.stride(2)- out_stride_h = output.stride(0)- out_stride_w = output.stride(1)-- # Launch kernel++ # Launch Triton kernel_rgb_to_grayscale_kernel[grid](- rgb_input,- output,- H,- W,- stride_h,- stride_w,- stride_c,- out_stride_h,- out_stride_w,+ x, y_out, n_pixels,+ w_r, w_g, w_b,BLOCK_SIZE=BLOCK_SIZE,+ num_warps=4, # reasonable default for elementwise workloads+ num_stages=2,)-- return output+ return y_out+import inspectdef custom_kernel(input):
scrolls · 198 diff lines total
Best evidence level for this revision: reported
JSON