submission 510652
KernelAgent · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 90 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_v2_H100_gpt-5-2_ka_submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-510652?include=source"interfacepython
Compatibility
measured onNVIDIA H100
declared hardwareNVIDIA H100
architecturessm_90
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:5264ca76d6fe9150bac46844e38cf238895ca85a2fdb26ee53bb4c3831fe4ad5
license declaredunknown
license concludedunknown
authorsKernelAgent
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
@triton.autotune(num-warps = 4
triton.Config({"BLOCK": 256}, num_warps=4),Kernel source
grayscale_v2_H100_gpt-5-2_ka_submission.py90 lines
# kernel.py
import torch
import triton
import triton.language as tl
@triton.autotune(
configs=[
triton.Config({"BLOCK": 256}, num_warps=4),
triton.Config({"BLOCK": 512}, num_warps=8),
triton.Config({"BLOCK": 1024}, num_warps=8),
],
key=["N"],
)
@triton.jit
def _rgb_to_gray_kernel(
x_ptr, # *[H, W, 3]
y_ptr, # *[H, W]
N, # H*W
BLOCK: tl.constexpr,
):
pid = tl.program_id(axis=0)
offs = pid * BLOCK + tl.arange(0, BLOCK)
mask = offs < N
# x is contiguous [H, W, 3] => flattened pixels are contiguous triplets
base = offs * 3
r = tl.load(x_ptr + base + 0, mask=mask, other=0).to(tl.float32)
g = tl.load(x_ptr + base + 1, mask=mask, other=0).to(tl.float32)
b = tl.load(x_ptr + base + 2, mask=mask, other=0).to(tl.float32)
# Match reference semantics more closely by quantizing weights to x dtype first.
x_ty = x_ptr.dtype.element_ty
w_r = tl.full((), 0.2989, tl.float32).to(x_ty).to(tl.float32)
w_g = tl.full((), 0.5870, tl.float32).to(x_ty).to(tl.float32)
w_b = tl.full((), 0.1140, tl.float32).to(x_ty).to(tl.float32)
y_f32 = r * w_r + g * w_g + b * w_b
tl.store(y_ptr + offs, y_f32.to(y_ptr.dtype.element_ty), mask=mask)
def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:
# Validation / allocation only (no compute here).
if not isinstance(x, torch.Tensor):
raise TypeError("x must be a torch.Tensor")
if not x.is_cuda:
raise ValueError("x must be a CUDA tensor")
if x.ndim != 3 or x.shape[-1] != 3:
raise ValueError(f"x must have shape [H, W, 3], got {tuple(x.shape)}")
if not x.is_contiguous():
raise ValueError("x must be contiguous (expected contiguous [H, W, 3])")
H, W, _ = x.shape
if y is None:
y = torch.empty((H, W), device=x.device, dtype=x.dtype)
else:
if not isinstance(y, torch.Tensor):
raise TypeError("y must be a torch.Tensor")
if not y.is_cuda:
raise ValueError("y must be a CUDA tensor")
if tuple(y.shape) != (H, W):
raise ValueError(f"y must have shape {(H, W)}, got {tuple(y.shape)}")
if not y.is_contiguous():
raise ValueError("y must be contiguous")
N = H * W
grid = lambda META: (triton.cdiv(N, META["BLOCK"]),)
_rgb_to_gray_kernel[grid](x, y, N)
return y
import inspect
def custom_kernel(input):
sig = inspect.signature(kernel_function)
num_params = len(sig.parameters)
if len(input) == num_params:
return kernel_function(*input)
return kernel_function(input)
# Ensure deterministic cuBLAS.
import os
if os.environ.get("CUBLAS_WORKSPACE_CONFIG", "") not in (":4096:8", ":16:8"):
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
scrolls · 90 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 510650.
+ # kernel.py+ import torchimport tritonimport triton.language as tl- import torch+ @triton.autotune(+ configs=[+ triton.Config({"BLOCK": 256}, num_warps=4),+ triton.Config({"BLOCK": 512}, num_warps=8),+ triton.Config({"BLOCK": 1024}, num_warps=8),+ ],+ key=["N"],+ )@triton.jit- def _rgb_to_grayscale_kernel(- rgb_ptr, # Pointer to input RGB image (H, W, 3)- out_ptr, # Pointer to output grayscale image (H, W)- H, # Image height- W, # Image width- stride_h, # Stride for height dimension in RGB- stride_w, # Stride for width dimension in RGB- stride_c, # Stride for channel dimension in RGB- out_stride_h, # Stride for height dimension in output- out_stride_w, # Stride for width dimension in output- BLOCK_SIZE: tl.constexpr, # Number of pixels to process per block+ def _rgb_to_gray_kernel(+ x_ptr, # *[H, W, 3]+ y_ptr, # *[H, W]+ N, # H*W+ BLOCK: tl.constexpr,):- """- Triton kernel for RGB to grayscale conversion.-- Fused operation: For each pixel, loads R, G, B values and computes- the weighted sum Y = 0.2989*R + 0.5870*G + 0.1140*B in a single pass.- """- # Standard RGB to grayscale coefficients- R_COEFF = 0.2989- G_COEFF = 0.5870- B_COEFF = 0.1140-- # Each program processes BLOCK_SIZE pixels- pid = tl.program_id(0)-- # Total number of pixels- n_pixels = H * W-- # Calculate pixel indices this block will process- block_start = pid * BLOCK_SIZE- offsets = block_start + tl.arange(0, BLOCK_SIZE)-- # Mask for valid pixels- mask = offsets < n_pixels-- # Convert linear index to 2D coordinates (row-major order)- # pixel_idx = h * W + w- h_idx = offsets // W- w_idx = offsets % W-- # Calculate base offset for each pixel in RGB image- # RGB layout is (H, W, 3), so offset = h * stride_h + w * stride_w + c * stride_c- base_offset = h_idx * stride_h + w_idx * stride_w-- # Load R, G, B values for each pixel- r_offset = base_offset + 0 * stride_c # Channel 0 = R- g_offset = base_offset + 1 * stride_c # Channel 1 = G- b_offset = base_offset + 2 * stride_c # Channel 2 = B-- r = tl.load(rgb_ptr + r_offset, mask=mask, other=0.0)- g = tl.load(rgb_ptr + g_offset, mask=mask, other=0.0)- b = tl.load(rgb_ptr + b_offset, mask=mask, other=0.0)-- # Compute grayscale value using standard coefficients- gray = R_COEFF * r + G_COEFF * g + B_COEFF * b-- # Calculate output offset- out_offset = h_idx * out_stride_h + w_idx * out_stride_w-- # Store the result- tl.store(out_ptr + out_offset, gray, mask=mask)+ pid = tl.program_id(axis=0)+ offs = pid * BLOCK + tl.arange(0, BLOCK)+ mask = offs < N+ # x is contiguous [H, W, 3] => flattened pixels are contiguous triplets+ base = offs * 3- def kernel_function(rgb_input: torch.Tensor, output: torch.Tensor) -> torch.Tensor:- """- Wrapper function for RGB to grayscale conversion.-- This function performs a fused RGB to grayscale conversion where:- - Loading of R, G, B channels- - Weighted sum computation (Y = 0.2989*R + 0.5870*G + 0.1140*B)- - Storage of grayscale result- are all done in a single kernel pass, minimizing memory traffic.-- Args:- rgb_input: Input RGB tensor of shape (H, W, 3), dtype float32- output: Output buffer of shape (H, W), dtype float32-- Returns:- The output tensor containing grayscale values- """- # Validate inputs- assert rgb_input.is_cuda, "Input must be on CUDA device"- assert output.is_cuda, "Output must be on CUDA device"- assert rgb_input.ndim == 3 and rgb_input.shape[2] == 3, "Input must be (H, W, 3)"- assert output.ndim == 2, "Output must be (H, W)"- assert rgb_input.shape[0] == output.shape[0] and rgb_input.shape[1] == output.shape[1], \- "Input and output spatial dimensions must match"-- H, W, C = rgb_input.shape- n_pixels = H * W-- # Get strides (in elements, not bytes)- stride_h = rgb_input.stride(0)- stride_w = rgb_input.stride(1)- stride_c = rgb_input.stride(2)- out_stride_h = output.stride(0)- out_stride_w = output.stride(1)-- # Choose block size - power of 2 for efficiency- BLOCK_SIZE = 1024-- # Calculate grid size- grid = (triton.cdiv(n_pixels, BLOCK_SIZE),)-- # Launch kernel- _rgb_to_grayscale_kernel[grid](- rgb_input,- output,- H,- W,- stride_h,- stride_w,- stride_c,- out_stride_h,- out_stride_w,- BLOCK_SIZE=BLOCK_SIZE,- )-- return output+ r = tl.load(x_ptr + base + 0, mask=mask, other=0).to(tl.float32)+ g = tl.load(x_ptr + base + 1, mask=mask, other=0).to(tl.float32)+ b = tl.load(x_ptr + base + 2, mask=mask, other=0).to(tl.float32)+ # Match reference semantics more closely by quantizing weights to x dtype first.+ x_ty = x_ptr.dtype.element_ty+ w_r = tl.full((), 0.2989, tl.float32).to(x_ty).to(tl.float32)+ w_g = tl.full((), 0.5870, tl.float32).to(x_ty).to(tl.float32)+ w_b = tl.full((), 0.1140, tl.float32).to(x_ty).to(tl.float32)++ y_f32 = r * w_r + g * w_g + b * w_b+ tl.store(y_ptr + offs, y_f32.to(y_ptr.dtype.element_ty), mask=mask)+++ def kernel_function(x: torch.Tensor, y: torch.Tensor | None = None) -> torch.Tensor:+ # Validation / allocation only (no compute here).+ if not isinstance(x, torch.Tensor):+ raise TypeError("x must be a torch.Tensor")+ if not x.is_cuda:+ raise ValueError("x must be a CUDA tensor")+ if x.ndim != 3 or x.shape[-1] != 3:+ raise ValueError(f"x must have shape [H, W, 3], got {tuple(x.shape)}")+ if not x.is_contiguous():+ raise ValueError("x must be contiguous (expected contiguous [H, W, 3])")++ H, W, _ = x.shape+ if y is None:+ y = torch.empty((H, W), device=x.device, dtype=x.dtype)+ else:+ if not isinstance(y, torch.Tensor):+ raise TypeError("y must be a torch.Tensor")+ if not y.is_cuda:+ raise ValueError("y must be a CUDA tensor")+ if tuple(y.shape) != (H, W):+ raise ValueError(f"y must have shape {(H, W)}, got {tuple(y.shape)}")+ if not y.is_contiguous():+ raise ValueError("y must be contiguous")++ N = H * W++ grid = lambda META: (triton.cdiv(N, META["BLOCK"]),)+ _rgb_to_gray_kernel[grid](x, y, N)++ return y+import inspectdef custom_kernel(input):
scrolls · 194 diff lines total
Best evidence level for this revision: reported
JSON