submission 227417
HayatoFujihara · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 111 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_v2_9.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-227417?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:6d8c38b759f8a2e8926b38fa2b23107f13f706a603bbb764c6bfe74d52490b85
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
@triton.autotune(num-warps = 4
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),stages = 3
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),Kernel source
grayscale_v2_9.py111 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t
# =============================================================================
# Phase 1.3: 明示的分解版(並列乗算 + 最後に加算)
# =============================================================================
# 目標: 2.44ms → 2.38ms
#
# 仮説: 累積加算(gray += x * coef)はFMA依存チェーンを形成
# 並列乗算(term_x = x * coef)後に加算することでILP向上
# 3つの乗算が並列実行可能になる
#
# Phase 1.1 結果: G→R→B = 2.47ms(効果なし)
# Phase 1.2 結果: B→R→G = 2.47ms(効果なし)
#
# 変更点:
# - 3つのチャンネルを並列でロード(独立Block Pointer)
# - 3つの乗算を並列で実行
# - 最後に3項を加算
@triton.autotune(
configs=[
# 中ブロック
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),
# 大ブロック
triton.Config({'BLOCK_SIZE': 4096}, num_warps=4, num_stages=4),
triton.Config({'BLOCK_SIZE': 4096}, num_warps=4, num_stages=5),
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=3),
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5),
# 超大ブロック
triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=3),
triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),
triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=5),
triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=3),
triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=4),
],
key=['n_elements'],
)
@triton.jit
def grayscale_kernel(
input_ptr, output_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr
):
pid = tl.program_id(axis=0)
# 独立した3つのBlock Pointer(並列ロード可能)
ptr_r = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
offsets=(pid * BLOCK_SIZE, 0),
block_shape=(BLOCK_SIZE, 1),
order=(1, 0)
)
ptr_g = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
offsets=(pid * BLOCK_SIZE, 1),
block_shape=(BLOCK_SIZE, 1),
order=(1, 0)
)
ptr_b = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
offsets=(pid * BLOCK_SIZE, 2),
block_shape=(BLOCK_SIZE, 1),
order=(1, 0)
)
# 並列ロード(依存関係なし)
r = tl.load(ptr_r, boundary_check=(0,), padding_option="zero")
g = tl.load(ptr_g, boundary_check=(0,), padding_option="zero")
b = tl.load(ptr_b, boundary_check=(0,), padding_option="zero")
# 並列乗算(依存関係なし、ILP最大化)
term_r = r * 0.2989
term_g = g * 0.5870
term_b = b * 0.1140
# 最後に加算
gray = term_r + term_g + term_b
# Output (1Dポインタ)
offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offs < n_elements
tl.store(output_ptr + offs, tl.reshape(gray, (BLOCK_SIZE,)), mask=mask)
def custom_kernel(data: input_t) -> output_t:
input_tensor, output_tensor = data
n_elements = output_tensor.numel()
grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']),)
grayscale_kernel[grid](
input_tensor, output_tensor,
n_elements
)
return output_tensor
scrolls · 111 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 227239.
⋯ 3 unchanged linesfrom task import input_t, output_t# =============================================================================- # Block Pointer + advance() 版(ベースライン: 2.44ms)+ # Phase 1.3: 明示的分解版(並列乗算 + 最後に加算)# =============================================================================- # 連続ロード+reshape方式は tl.arange() の2のべき乗制約により不可- # 入力側コンパイラヒントは正確性エラー- # → ベースラインに戻して安定性を確保+ # 目標: 2.44ms → 2.38ms+ #+ # 仮説: 累積加算(gray += x * coef)はFMA依存チェーンを形成+ # 並列乗算(term_x = x * coef)後に加算することでILP向上+ # 3つの乗算が並列実行可能になる+ #+ # Phase 1.1 結果: G→R→B = 2.47ms(効果なし)+ # Phase 1.2 結果: B→R→G = 2.47ms(効果なし)+ #+ # 変更点:+ # - 3つのチャンネルを並列でロード(独立Block Pointer)+ # - 3つの乗算を並列で実行+ # - 最後に3項を加算@triton.autotune(configs=[⋯ 26 unchanged lines):pid = tl.program_id(axis=0)- # Block Pointer- in_ptr = tl.make_block_ptr(+ # 独立した3つのBlock Pointer(並列ロード可能)+ ptr_r = tl.make_block_ptr(base=input_ptr,shape=(n_elements, 3),strides=(3, 1),⋯ 1 unchanged linesblock_shape=(BLOCK_SIZE, 1),order=(1, 0))+ ptr_g = tl.make_block_ptr(+ base=input_ptr,+ shape=(n_elements, 3),+ strides=(3, 1),+ offsets=(pid * BLOCK_SIZE, 1),+ block_shape=(BLOCK_SIZE, 1),+ order=(1, 0)+ )+ ptr_b = tl.make_block_ptr(+ base=input_ptr,+ shape=(n_elements, 3),+ strides=(3, 1),+ offsets=(pid * BLOCK_SIZE, 2),+ block_shape=(BLOCK_SIZE, 1),+ order=(1, 0)+ )- # R Channel- r = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")- gray = r * 0.2989+ # 並列ロード(依存関係なし)+ r = tl.load(ptr_r, boundary_check=(0,), padding_option="zero")+ g = tl.load(ptr_g, boundary_check=(0,), padding_option="zero")+ b = tl.load(ptr_b, boundary_check=(0,), padding_option="zero")- # G Channel- in_ptr = tl.advance(in_ptr, (0, 1))- g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")- gray += g * 0.5870+ # 並列乗算(依存関係なし、ILP最大化)+ term_r = r * 0.2989+ term_g = g * 0.5870+ term_b = b * 0.1140- # B Channel- in_ptr = tl.advance(in_ptr, (0, 1))- b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")- gray += b * 0.1140+ # 最後に加算+ gray = term_r + term_g + term_b# Output (1Dポインタ)offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
scrolls · 84 diff lines total
Best evidence level for this revision: reported
JSON