submission 67540
yue · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 68 lines, June 9 Researcher Reciprocity License v1.0.
submission2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-67540?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:cc3657b4740851a0c807cc970a940cfafad5d7b0b258775a61b827964b03486d
license declaredunknown
license concludedunknown
authorsyue
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 4
num_warps=4,stages = 1
num_stages=1,Kernel source
submission2.py68 lines
#!POPCORN leaderboard grayscale_v2
from task import input_t, output_t
from utils import DeterministicContext
import triton
import triton.language as tl
@triton.jit
def _grayscale_kernel(
rgb_ptr,
output_ptr,
height: tl.int32,
width: tl.int32,
stride_h: tl.int32,
stride_w: tl.int32,
stride_c: tl.int32,
out_stride_h: tl.int32,
out_stride_w: tl.int32,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(0)
offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
n_pixels = height * width
mask = offsets < n_pixels
rows = offsets // width
cols = offsets % width
rgb_index = rows * stride_h + cols * stride_w
r = tl.load(rgb_ptr + rgb_index + 0 * stride_c, mask=mask, other=0.0)
g = tl.load(rgb_ptr + rgb_index + 1 * stride_c, mask=mask, other=0.0)
b = tl.load(rgb_ptr + rgb_index + 2 * stride_c, mask=mask, other=0.0)
grayscale = 0.2989 * r + 0.5870 * g + 0.1140 * b
out_index = rows * out_stride_h + cols * out_stride_w
tl.store(output_ptr + out_index, grayscale, mask=mask)
def custom_kernel(data: input_t) -> output_t:
with DeterministicContext():
rgb, output = data
height, width, channels = rgb.shape
if channels != 3:
raise ValueError(f"Expected last dimension to be 3, got {channels}")
BLOCK_SIZE = 1024
grid = (triton.cdiv(height * width, BLOCK_SIZE),)
_grayscale_kernel[grid](
rgb,
output,
height,
width,
rgb.stride(0),
rgb.stride(1),
rgb.stride(2),
output.stride(0),
output.stride(1),
BLOCK_SIZE=BLOCK_SIZE,
num_warps=4,
num_stages=1,
)
return outputscrolls · 68 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 67528.
⋯ 16 unchanged linesstride_c: tl.int32,out_stride_h: tl.int32,out_stride_w: tl.int32,- BLOCK_H: tl.constexpr,- BLOCK_W: tl.constexpr,+ BLOCK_SIZE: tl.constexpr,):- pid_h = tl.program_id(0)- pid_w = tl.program_id(1)+ pid = tl.program_id(0)- offs_h = pid_h * BLOCK_H + tl.arange(0, BLOCK_H)- offs_w = pid_w * BLOCK_W + tl.arange(0, BLOCK_W)+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ n_pixels = height * width+ mask = offsets < n_pixels- mask_h = offs_h < height- mask_w = offs_w < width+ rows = offsets // width+ cols = offsets % width- hh = offs_h[:, None]- ww = offs_w[None, :]- mask = mask_h[:, None] & mask_w[None, :]+ rgb_index = rows * stride_h + cols * stride_w- base = hh * stride_h + ww * stride_w+ r = tl.load(rgb_ptr + rgb_index + 0 * stride_c, mask=mask, other=0.0)+ g = tl.load(rgb_ptr + rgb_index + 1 * stride_c, mask=mask, other=0.0)+ b = tl.load(rgb_ptr + rgb_index + 2 * stride_c, mask=mask, other=0.0)- r = tl.load(rgb_ptr + base + 0 * stride_c, mask=mask, other=0.0)- g = tl.load(rgb_ptr + base + 1 * stride_c, mask=mask, other=0.0)- b = tl.load(rgb_ptr + base + 2 * stride_c, mask=mask, other=0.0)-grayscale = 0.2989 * r + 0.5870 * g + 0.1140 * b- tl.store(output_ptr + hh * out_stride_h + ww * out_stride_w, grayscale, mask=mask)+ out_index = rows * out_stride_h + cols * out_stride_w+ tl.store(output_ptr + out_index, grayscale, mask=mask)def custom_kernel(data: input_t) -> output_t:⋯ 3 unchanged linesif channels != 3:raise ValueError(f"Expected last dimension to be 3, got {channels}")- BLOCK_H = 32- BLOCK_W = 32+ BLOCK_SIZE = 1024- grid = (triton.cdiv(height, BLOCK_H), triton.cdiv(width, BLOCK_W))+ grid = (triton.cdiv(height * width, BLOCK_SIZE),)_grayscale_kernel[grid](rgb,⋯ 5 unchanged linesrgb.stride(2),output.stride(0),output.stride(1),- BLOCK_H=BLOCK_H,- BLOCK_W=BLOCK_W,+ BLOCK_SIZE=BLOCK_SIZE,+ num_warps=4,+ num_stages=1,)return outputNo newline at end of file
scrolls · 70 diff lines total
Best evidence level for this revision: reported
JSON