submission 66583
albert9823 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 62 lines, June 9 Researcher Reciprocity License v1.0.
submission_v4.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-66583?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:0f3eeb01112beb85b23e0106315e0ea2560b25d012101dcbc3b4fbec41506add
license declaredunknown
license concludedunknown
authorsalbert9823
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 4
num_warps=4, # tune: 4 or 8stages = 2
num_stages=2 # tune: 2 or 3Kernel source
submission_v4.py62 lines
import torch, triton, triton.language as tl
from task import input_t, output_t
@triton.jit
def rgb2gray_kernel(
x_ptr, y_ptr,
H, W,
stride_h, stride_w, stride_c,
y_stride_h, y_stride_w,
w_r, w_g, w_b,
BLOCK_H: tl.constexpr, # <-- constexpr tile sizes
BLOCK_W: tl.constexpr
):
pid_h = tl.program_id(0)
pid_w = tl.program_id(1)
h0 = pid_h * BLOCK_H
w0 = pid_w * BLOCK_W
hs = h0 + tl.arange(0, BLOCK_H)
ws = w0 + tl.arange(0, BLOCK_W)
Hs = hs[:, None] # (BH, 1)
Ws = ws[None, :] # (1, BW)
mask = (Hs < H) & (Ws < W)
base = Hs * stride_h + Ws * stride_w # offsets in elements
r = tl.load(x_ptr + base + 0 * stride_c, mask=mask, other=0.0)
g = tl.load(x_ptr + base + 1 * stride_c, mask=mask, other=0.0)
b = tl.load(x_ptr + base + 2 * stride_c, mask=mask, other=0.0)
y = w_r * r + w_g * g + w_b * b
tl.store(y_ptr + Hs * y_stride_h + Ws * y_stride_w, y, mask=mask)
def custom_kernel(data: input_t) -> output_t:
x, out = data # x: (H,W,3) float32 cuda contiguous; out: (H,W) float32 cuda
H, W, C = x.shape
assert C == 3
# PyTorch gives strides in elements for contiguous NHWC: (W*3, 3, 1)
s_h, s_w, s_c = x.stride()
ys_h, ys_w = out.stride()
# Tile/grid setup. Try (64, 64) first; (32,128) or (128,64) may be faster depending on size.
BLOCK_H, BLOCK_W = 64, 64
grid = (triton.cdiv(H, BLOCK_H), triton.cdiv(W, BLOCK_W))
rgb2gray_kernel[grid](
x, out,
H, W,
s_h, s_w, s_c,
ys_h, ys_w,
0.2989, 0.5870, 0.1140,
BLOCK_H=BLOCK_H, BLOCK_W=BLOCK_W, # <-- pass constexprs
num_warps=4, # tune: 4 or 8
num_stages=2 # tune: 2 or 3
)
return out
scrolls · 62 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 66580.
+ import torch, triton, triton.language as tlfrom task import input_t, output_t- import torch- def custom_kernel(data: input_t) -> output_t:- x, out = data # x: (H, W, 3), out: (H, W)+ @triton.jit+ def rgb2gray_kernel(+ x_ptr, y_ptr,+ H, W,+ stride_h, stride_w, stride_c,+ y_stride_h, y_stride_w,+ w_r, w_g, w_b,+ BLOCK_H: tl.constexpr, # <-- constexpr tile sizes+ BLOCK_W: tl.constexpr+ ):+ pid_h = tl.program_id(0)+ pid_w = tl.program_id(1)- # out = 0.2989 * R- torch.mul(x[..., 0], 0.2989, out=out)+ h0 = pid_h * BLOCK_H+ w0 = pid_w * BLOCK_W- # out += 0.5870 * G- torch.add(out, x[..., 1], alpha=0.5870, out=out)+ hs = h0 + tl.arange(0, BLOCK_H)+ ws = w0 + tl.arange(0, BLOCK_W)- # out += 0.1140 * B- torch.add(out, x[..., 2], alpha=0.1140, out=out)+ Hs = hs[:, None] # (BH, 1)+ Ws = ws[None, :] # (1, BW)+ mask = (Hs < H) & (Ws < W)++ base = Hs * stride_h + Ws * stride_w # offsets in elements++ r = tl.load(x_ptr + base + 0 * stride_c, mask=mask, other=0.0)+ g = tl.load(x_ptr + base + 1 * stride_c, mask=mask, other=0.0)+ b = tl.load(x_ptr + base + 2 * stride_c, mask=mask, other=0.0)++ y = w_r * r + w_g * g + w_b * b++ tl.store(y_ptr + Hs * y_stride_h + Ws * y_stride_w, y, mask=mask)++ def custom_kernel(data: input_t) -> output_t:+ x, out = data # x: (H,W,3) float32 cuda contiguous; out: (H,W) float32 cuda+ H, W, C = x.shape+ assert C == 3+ # PyTorch gives strides in elements for contiguous NHWC: (W*3, 3, 1)+ s_h, s_w, s_c = x.stride()+ ys_h, ys_w = out.stride()++ # Tile/grid setup. Try (64, 64) first; (32,128) or (128,64) may be faster depending on size.+ BLOCK_H, BLOCK_W = 64, 64+ grid = (triton.cdiv(H, BLOCK_H), triton.cdiv(W, BLOCK_W))++ rgb2gray_kernel[grid](+ x, out,+ H, W,+ s_h, s_w, s_c,+ ys_h, ys_w,+ 0.2989, 0.5870, 0.1140,+ BLOCK_H=BLOCK_H, BLOCK_W=BLOCK_W, # <-- pass constexprs+ num_warps=4, # tune: 4 or 8+ num_stages=2 # tune: 2 or 3+ )return out
scrolls · 70 diff lines total
Best evidence level for this revision: reported
JSON