submission 514597
HankBO · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 79 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-514597?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:c9f5ef7745eaab774ae8f796609fb5e4cac89e41ab4dfc95cfcf13d51a9ed5b0
license declaredunknown
license concludedunknown
authorsHankBO
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
num_warps=8,Kernel source
submission.py79 lines
from task import input_t, output_t
import torch
try:
import triton
import triton.language as tl
HAS_TRITON = True
except Exception:
HAS_TRITON = False
if HAS_TRITON:
@triton.jit
def _rgb_to_gray_kernel(
rgb_ptr,
out_ptr,
n_pixels,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(axis=0)
offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offs < n_pixels
# Input is packed as [..., 3], so each pixel starts at 3 * idx.
base = offs * 3
r = tl.load(rgb_ptr + base + 0, mask=mask, other=0.0)
g = tl.load(rgb_ptr + base + 1, mask=mask, other=0.0)
b = tl.load(rgb_ptr + base + 2, mask=mask, other=0.0)
y = r * 0.2989 + g * 0.5870 + b * 0.1140
tl.store(out_ptr + offs, y, mask=mask)
def _fallback_pytorch(rgb: torch.Tensor, output: torch.Tensor) -> torch.Tensor:
# Older (pre-Triton) solution for learning:
# 1) Split channels from (H, W, 3) into three (H, W) tensors.
# 2) Reuse preallocated output buffer to avoid extra allocations.
# 3) Accumulate weighted channels in place.
#
# Equivalent math:
# output[...] = 0.2989 * rgb[..., 0] + 0.5870 * rgb[..., 1] + 0.1140 * rgb[..., 2]
r, g, b = rgb.unbind(-1)
output.copy_(r)
output.mul_(0.2989)
output.add_(g, alpha=0.5870)
output.add_(b, alpha=0.1140)
return output
def custom_kernel(data: input_t) -> output_t:
rgb, output = data
if (
HAS_TRITON
and rgb.is_cuda
and output.is_cuda
and rgb.dtype == torch.float32
and output.dtype == torch.float32
and rgb.is_contiguous()
and output.is_contiguous()
):
n_pixels = output.numel()
rgb_flat = rgb.view(-1)
out_flat = output.view(-1)
grid = (triton.cdiv(n_pixels, 1024),)
_rgb_to_gray_kernel[grid](
rgb_flat,
out_flat,
n_pixels,
BLOCK_SIZE=1024,
num_warps=8,
)
return output
return _fallback_pytorch(rgb, output)
scrolls · 79 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON