Skip to content
KernelIndex
Search⌘K

submission 227417

HayatoFujihara · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 111 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_v2_9.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-227417?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.47ms
#14 of 137
2025-12-28

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:6d8c38b759f8a2e8926b38fa2b23107f13f706a603bbb764c6bfe74d52490b85
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotune@triton.autotune(
num-warps = 4triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),
stages = 3triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),

Kernel source

grayscale_v2_9.py111 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t

# =============================================================================
# Phase 1.3: 明示的分解版(並列乗算 + 最後に加算)
# =============================================================================
# 目標: 2.44ms → 2.38ms
#
# 仮説: 累積加算(gray += x * coef)はFMA依存チェーンを形成
#       並列乗算(term_x = x * coef)後に加算することでILP向上
#       3つの乗算が並列実行可能になる
#
# Phase 1.1 結果: G→R→B = 2.47ms(効果なし)
# Phase 1.2 結果: B→R→G = 2.47ms(効果なし)
#
# 変更点:
# - 3つのチャンネルを並列でロード(独立Block Pointer)
# - 3つの乗算を並列で実行
# - 最後に3項を加算

@triton.autotune(
    configs=[
        # 中ブロック
        triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),
        triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),
        triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),

        # 大ブロック
        triton.Config({'BLOCK_SIZE': 4096}, num_warps=4, num_stages=4),
        triton.Config({'BLOCK_SIZE': 4096}, num_warps=4, num_stages=5),
        triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=3),
        triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),
        triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5),

        # 超大ブロック
        triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=3),
        triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),
        triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=5),
        triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=3),
        triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=4),
    ],
    key=['n_elements'],
)
@triton.jit
def grayscale_kernel(
    input_ptr, output_ptr,
    n_elements,
    BLOCK_SIZE: tl.constexpr
):
    pid = tl.program_id(axis=0)

    # 独立した3つのBlock Pointer(並列ロード可能)
    ptr_r = tl.make_block_ptr(
        base=input_ptr,
        shape=(n_elements, 3),
        strides=(3, 1),
        offsets=(pid * BLOCK_SIZE, 0),
        block_shape=(BLOCK_SIZE, 1),
        order=(1, 0)
    )
    ptr_g = tl.make_block_ptr(
        base=input_ptr,
        shape=(n_elements, 3),
        strides=(3, 1),
        offsets=(pid * BLOCK_SIZE, 1),
        block_shape=(BLOCK_SIZE, 1),
        order=(1, 0)
    )
    ptr_b = tl.make_block_ptr(
        base=input_ptr,
        shape=(n_elements, 3),
        strides=(3, 1),
        offsets=(pid * BLOCK_SIZE, 2),
        block_shape=(BLOCK_SIZE, 1),
        order=(1, 0)
    )

    # 並列ロード(依存関係なし)
    r = tl.load(ptr_r, boundary_check=(0,), padding_option="zero")
    g = tl.load(ptr_g, boundary_check=(0,), padding_option="zero")
    b = tl.load(ptr_b, boundary_check=(0,), padding_option="zero")

    # 並列乗算(依存関係なし、ILP最大化)
    term_r = r * 0.2989
    term_g = g * 0.5870
    term_b = b * 0.1140

    # 最後に加算
    gray = term_r + term_g + term_b

    # Output (1Dポインタ)
    offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
    mask = offs < n_elements
    tl.store(output_ptr + offs, tl.reshape(gray, (BLOCK_SIZE,)), mask=mask)


def custom_kernel(data: input_t) -> output_t:
    input_tensor, output_tensor = data
    n_elements = output_tensor.numel()

    grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']),)

    grayscale_kernel[grid](
        input_tensor, output_tensor,
        n_elements
    )

    return output_tensor
scrolls · 111 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 227239.

⋯ 3 unchanged lines
from task import input_t, output_t
# =============================================================================
- # Block Pointer + advance() 版(ベースライン: 2.44ms)
+ # Phase 1.3: 明示的分解版(並列乗算 + 最後に加算)
# =============================================================================
- # 連続ロード+reshape方式は tl.arange() の2のべき乗制約により不可
- # 入力側コンパイラヒントは正確性エラー
- # → ベースラインに戻して安定性を確保
+ # 目標: 2.44ms → 2.38ms
+ #
+ # 仮説: 累積加算(gray += x * coef)はFMA依存チェーンを形成
+ # 並列乗算(term_x = x * coef)後に加算することでILP向上
+ # 3つの乗算が並列実行可能になる
+ #
+ # Phase 1.1 結果: G→R→B = 2.47ms(効果なし)
+ # Phase 1.2 結果: B→R→G = 2.47ms(効果なし)
+ #
+ # 変更点:
+ # - 3つのチャンネルを並列でロード(独立Block Pointer)
+ # - 3つの乗算を並列で実行
+ # - 最後に3項を加算
@triton.autotune(
configs=[
⋯ 26 unchanged lines
):
pid = tl.program_id(axis=0)
- # Block Pointer
- in_ptr = tl.make_block_ptr(
+ # 独立した3つのBlock Pointer(並列ロード可能)
+ ptr_r = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
⋯ 1 unchanged lines
block_shape=(BLOCK_SIZE, 1),
order=(1, 0)
)
+ ptr_g = tl.make_block_ptr(
+ base=input_ptr,
+ shape=(n_elements, 3),
+ strides=(3, 1),
+ offsets=(pid * BLOCK_SIZE, 1),
+ block_shape=(BLOCK_SIZE, 1),
+ order=(1, 0)
+ )
+ ptr_b = tl.make_block_ptr(
+ base=input_ptr,
+ shape=(n_elements, 3),
+ strides=(3, 1),
+ offsets=(pid * BLOCK_SIZE, 2),
+ block_shape=(BLOCK_SIZE, 1),
+ order=(1, 0)
+ )
- # R Channel
- r = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
- gray = r * 0.2989
+ # 並列ロード(依存関係なし)
+ r = tl.load(ptr_r, boundary_check=(0,), padding_option="zero")
+ g = tl.load(ptr_g, boundary_check=(0,), padding_option="zero")
+ b = tl.load(ptr_b, boundary_check=(0,), padding_option="zero")
- # G Channel
- in_ptr = tl.advance(in_ptr, (0, 1))
- g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
- gray += g * 0.5870
+ # 並列乗算(依存関係なし、ILP最大化)
+ term_r = r * 0.2989
+ term_g = g * 0.5870
+ term_b = b * 0.1140
- # B Channel
- in_ptr = tl.advance(in_ptr, (0, 1))
- b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
- gray += b * 0.1140
+ # 最後に加算
+ gray = term_r + term_g + term_b
# Output (1Dポインタ)
offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
scrolls · 84 diff lines total

Best evidence level for this revision: reported

JSON