Skip to content
KernelIndex
Search⌘K

submission 66616

albert9823 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 71 lines, June 9 Researcher Reciprocity License v1.0.

submission_v7.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-66616?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
3.80ms
#41 of 137
2025-11-04

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:793554691b1a421d70cd297d26c895bf144333fd3d5492b671012ebbf430c6d3
license declaredunknown
license concludedunknown
authorsalbert9823
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 8num_warps=8, num_stages=2
stages = 2num_warps=8, num_stages=2

Kernel source

submission_v7.py71 lines
import torch, triton, triton.language as tl
from task import input_t, output_t

@triton.jit
def rgb2gray_kernel(
    x_ptr, y_ptr,
    H, W,
    stride_h, stride_w, stride_c,
    y_stride_h, y_stride_w,
    WR, WG, WB,                       # weights as scalars
    BLOCK_H: tl.constexpr,
    BLOCK_W: tl.constexpr,
):
    pid_h = tl.program_id(0)
    pid_w = tl.program_id(1)

    h0 = pid_h * BLOCK_H
    w0 = pid_w * BLOCK_W

    hs = h0 + tl.arange(0, BLOCK_H)          # (BH,)
    ws = w0 + tl.arange(0, BLOCK_W)          # (BW,)
    Hs = hs[:, None]                          # (BH,1)
    Ws = ws[None, :]                          # (1,BW)

    mask = (Hs < H) & (Ws < W)
    base = Hs * stride_h + Ws * stride_w      # (BH,BW)

    # Vectorized 4-lane per-pixel load (last lane is dummy)
    ch = tl.arange(0, 4)                      # (4,)
    ptrs = x_ptr + base[:, :, None] + ch[None, None, :] * stride_c  # (BH,BW,4)

    rgb4 = tl.load(
        ptrs,
        mask=mask[:, :, None] & (ch[None, None, :] < 3),
        other=0.0,
        cache_modifier=".ca",
    )  # (BH,BW,4) = [R,G,B,0]

    # Build weights per lane without slicing
    # w_lane[k] = [WR, WG, WB, 0][k]
    w_lane = (tl.where(ch == 0, WR, 0.0)
            + tl.where(ch == 1, WG, 0.0)
            + tl.where(ch == 2, WB, 0.0))      # shape (4,)
    w_broadcast = w_lane[None, None, :]        # (1,1,4)

    # Weighted sum across channel axis
    y = tl.sum(rgb4 * w_broadcast, axis=2)     # (BH,BW)

    tl.store(y_ptr + Hs * y_stride_h + Ws * y_stride_w, y, mask=mask)

def custom_kernel(data: input_t) -> output_t:
    x, out = data
    H, W, C = x.shape
    s_h, s_w, s_c = x.stride()
    ys_h, ys_w = out.stride()

    BLOCK_H, BLOCK_W = 64, 128
    grid = (triton.cdiv(H, BLOCK_H), triton.cdiv(W, BLOCK_W))

    rgb2gray_kernel[grid](
        x, out,
        H, W,
        s_h, s_w, s_c,
        ys_h, ys_w,
        0.2989, 0.5870, 0.1140,          # WR, WG, WB
        BLOCK_H=BLOCK_H, BLOCK_W=BLOCK_W,
        num_warps=8, num_stages=2
    )
    return out

scrolls · 71 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 66607.

⋯ 6 unchanged lines
H, W,
stride_h, stride_w, stride_c,
y_stride_h, y_stride_w,
- w_r, w_g, w_b,
- BLOCK_H: tl.constexpr, # <-- constexpr tile sizes
- BLOCK_W: tl.constexpr
+ WR, WG, WB, # weights as scalars
+ BLOCK_H: tl.constexpr,
+ BLOCK_W: tl.constexpr,
):
pid_h = tl.program_id(0)
pid_w = tl.program_id(1)
⋯ 1 unchanged lines
h0 = pid_h * BLOCK_H
w0 = pid_w * BLOCK_W
- hs = h0 + tl.arange(0, BLOCK_H)
- ws = w0 + tl.arange(0, BLOCK_W)
+ hs = h0 + tl.arange(0, BLOCK_H) # (BH,)
+ ws = w0 + tl.arange(0, BLOCK_W) # (BW,)
+ Hs = hs[:, None] # (BH,1)
+ Ws = ws[None, :] # (1,BW)
- Hs = hs[:, None] # (BH, 1)
- Ws = ws[None, :] # (1, BW)
-
mask = (Hs < H) & (Ws < W)
+ base = Hs * stride_h + Ws * stride_w # (BH,BW)
- base = Hs * stride_h + Ws * stride_w # offsets in elements
+ # Vectorized 4-lane per-pixel load (last lane is dummy)
+ ch = tl.arange(0, 4) # (4,)
+ ptrs = x_ptr + base[:, :, None] + ch[None, None, :] * stride_c # (BH,BW,4)
- r = tl.load(x_ptr + base + 0 * stride_c, mask=mask, other=0.0)
- g = tl.load(x_ptr + base + 1 * stride_c, mask=mask, other=0.0)
- b = tl.load(x_ptr + base + 2 * stride_c, mask=mask, other=0.0)
+ rgb4 = tl.load(
+ ptrs,
+ mask=mask[:, :, None] & (ch[None, None, :] < 3),
+ other=0.0,
+ cache_modifier=".ca",
+ ) # (BH,BW,4) = [R,G,B,0]
- #y = w_r * r + w_g * g + w_b * b
- y = tl.math.fma(w_b, b, tl.math.fma(w_g, g, w_r * r))
+ # Build weights per lane without slicing
+ # w_lane[k] = [WR, WG, WB, 0][k]
+ w_lane = (tl.where(ch == 0, WR, 0.0)
+ + tl.where(ch == 1, WG, 0.0)
+ + tl.where(ch == 2, WB, 0.0)) # shape (4,)
+ w_broadcast = w_lane[None, None, :] # (1,1,4)
+ # Weighted sum across channel axis
+ y = tl.sum(rgb4 * w_broadcast, axis=2) # (BH,BW)
+
tl.store(y_ptr + Hs * y_stride_h + Ws * y_stride_w, y, mask=mask)
def custom_kernel(data: input_t) -> output_t:
- x, out = data # x: (H,W,3) float32 cuda contiguous; out: (H,W) float32 cuda
+ x, out = data
H, W, C = x.shape
- # assert C == 3
- # PyTorch gives strides in elements for contiguous NHWC: (W*3, 3, 1)
s_h, s_w, s_c = x.stride()
ys_h, ys_w = out.stride()
- # Tile/grid setup. Try (64, 64) first; (32,128) or (128,64) may be faster depending on size.
- BLOCK_H, BLOCK_W = 64, 64
+ BLOCK_H, BLOCK_W = 64, 128
grid = (triton.cdiv(H, BLOCK_H), triton.cdiv(W, BLOCK_W))
rgb2gray_kernel[grid](
⋯ 1 unchanged lines
H, W,
s_h, s_w, s_c,
ys_h, ys_w,
- 0.2989, 0.5870, 0.1140,
- BLOCK_H=BLOCK_H, BLOCK_W=BLOCK_W, # <-- pass constexprs
- num_warps=8, # tune: 4 or 8
- num_stages=2 # tune: 2 or 3
+ 0.2989, 0.5870, 0.1140, # WR, WG, WB
+ BLOCK_H=BLOCK_H, BLOCK_W=BLOCK_W,
+ num_warps=8, num_stages=2
)
return out
scrolls · 88 diff lines total

Best evidence level for this revision: reported

JSON