submission 153879
HayatoFujihara · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 133 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_v2_6.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-153879?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:da0eeb57f1d40fc1a78860d75dbe4544d315410b260d76270855ace9e13d5788
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
@triton.autotune(num-warps = 8
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),stages = 4
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),Kernel source
grayscale_v2_6.py133 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t
# -----------------------------------------------------------------------------
# Final Stable & Fast: "Independent Block Pointers + Parallel FMA"
# -----------------------------------------------------------------------------
# 過去の実験で最も高速かつ安定していた2つの要素を統合します。
# 1. Block Pointer: アドレス計算負荷をゼロにする (2.47msの実績)
# 2. Parallel Math: R,G,Bの計算を独立させ、パイプラインを埋める (2.46msの実績)
#
# これらに対し、コンパイルエラーの原因となる「スライシング」や「変なReshape」を
# 一切行わないことで、確実に動作させます。
@triton.autotune(
configs=[
# A100のスイートスポット (4096 x 8warps)
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5),
# 巨大ブロック (ループ回数削減)
# Warps=16 を投入し、計算レイテンシの隠蔽能力を最大化します
triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),
triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=4),
# 回転率重視
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),
],
key=['n_elements'],
)
@triton.jit
def grayscale_parallel_independent_kernel(
input_ptr, output_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr
):
pid = tl.program_id(axis=0)
# -----------------------------------------------------------
# 1. Independent Input Pointers (No Dependency Chain)
# -----------------------------------------------------------
# advance() を使わず、最初からオフセット済みのポインタを3つ生成します。
# これにより、GPUは R, G, B のロード命令を任意の順序で・同時に発行できます。
# R Pointer (Column 0)
ptr_r = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
offsets=(pid * BLOCK_SIZE, 0),
block_shape=(BLOCK_SIZE, 1),
order=(1, 0)
)
# G Pointer (Column 1)
ptr_g = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
offsets=(pid * BLOCK_SIZE, 1),
block_shape=(BLOCK_SIZE, 1),
order=(1, 0)
)
# B Pointer (Column 2)
ptr_b = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
offsets=(pid * BLOCK_SIZE, 2),
block_shape=(BLOCK_SIZE, 1),
order=(1, 0)
)
# -----------------------------------------------------------
# 2. Parallel Load
# -----------------------------------------------------------
# 3つのロードを連続記述。依存関係がないため即座に発行されます。
# boundary_check は (0,) = 行方向のみチェック (列は1なので安全)
# Load (BLOCK_SIZE, 1)
r = tl.load(ptr_r, boundary_check=(0,), padding_option="zero")
g = tl.load(ptr_g, boundary_check=(0,), padding_option="zero")
b = tl.load(ptr_b, boundary_check=(0,), padding_option="zero")
# -----------------------------------------------------------
# 3. Parallel Computation (ILP)
# -----------------------------------------------------------
# 以前のエラー原因だった「複雑なスライシング」は不要です。
# 単純な (BLOCK_SIZE, 1) 同士の計算なので絶対に落ちません。
# 積和演算の依存を切り、3並列で計算させます
term_r = r * 0.2989
term_g = g * 0.5870
term_b = b * 0.1140
# 合算
gray = term_r + term_g + term_b
# -----------------------------------------------------------
# 4. Output
# -----------------------------------------------------------
# Output Block Pointer
out_ptr = tl.make_block_ptr(
base=output_ptr,
shape=(n_elements,),
strides=(1,),
offsets=(pid * BLOCK_SIZE,),
block_shape=(BLOCK_SIZE,),
order=(0,)
)
# gray は (BLOCK_SIZE, 1) なので、(BLOCK_SIZE,) に戻してストア
# reshape はデータ移動を伴わないメタデータ操作なのでコストゼロです
tl.store(out_ptr, tl.reshape(gray, (BLOCK_SIZE,)), boundary_check=(0,))
def custom_kernel(data: input_t) -> output_t:
input_tensor, output_tensor = data
n_elements = output_tensor.numel()
# Python Overhead Optimization
# 関数呼び出しを避け、直接計算でグリッドを渡します
grid = lambda META: ((n_elements + META['BLOCK_SIZE'] - 1) // META['BLOCK_SIZE'], )
grayscale_parallel_independent_kernel[grid](
input_tensor, output_tensor,
n_elements
)
return output_tensorscrolls · 133 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 153868.
⋯ 3 unchanged linesfrom task import input_t, output_t# ------------------------------------------------------------------------------ # Configuration Search Space+ # Final Stable & Fast: "Independent Block Pointers + Parallel FMA"# ------------------------------------------------------------------------------ # 人手による絞り込みをやめ、機械的に全探索します。- # A100で可能性のある範囲を網羅します。+ # 過去の実験で最も高速かつ安定していた2つの要素を統合します。+ # 1. Block Pointer: アドレス計算負荷をゼロにする (2.47msの実績)+ # 2. Parallel Math: R,G,Bの計算を独立させ、パイプラインを埋める (2.46msの実績)+ #+ # これらに対し、コンパイルエラーの原因となる「スライシング」や「変なReshape」を+ # 一切行わないことで、確実に動作させます。- block_sizes = [1024, 2048, 4096, 8192, 16384]- num_warps_list = [4, 8, 16]- num_stages_list = [3, 4, 5, 6]-- configs = [- triton.Config({'BLOCK_SIZE': bs}, num_warps=w, num_stages=s)- for bs in block_sizes- for w in num_warps_list- for s in num_stages_list- ]-- # ------------------------------------------------------------------------------ # Kernel Implementation: Full Block Pointer (Best Logic)- # -----------------------------------------------------------------------------@triton.autotune(- configs=configs,+ configs=[+ # A100のスイートスポット (4096 x 8warps)+ triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),+ triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5),++ # 巨大ブロック (ループ回数削減)+ # Warps=16 を投入し、計算レイテンシの隠蔽能力を最大化します+ triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),+ triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=4),++ # 回転率重視+ triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),+ triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),+ ],key=['n_elements'],)@triton.jit- def grayscale_full_search_kernel(+ def grayscale_parallel_independent_kernel(input_ptr, output_ptr,n_elements,BLOCK_SIZE: tl.constexpr):pid = tl.program_id(axis=0)- # --- Input Block Pointer ---- # Shape: (N, 3), Block: (BS, 1)- in_ptr = tl.make_block_ptr(+ # -----------------------------------------------------------+ # 1. Independent Input Pointers (No Dependency Chain)+ # -----------------------------------------------------------+ # advance() を使わず、最初からオフセット済みのポインタを3つ生成します。+ # これにより、GPUは R, G, B のロード命令を任意の順序で・同時に発行できます。++ # R Pointer (Column 0)+ ptr_r = tl.make_block_ptr(base=input_ptr,shape=(n_elements, 3),strides=(3, 1),⋯ 2 unchanged linesorder=(1, 0))- # --- Load & Compute (Sequential) ----- # R Channel- r = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")- gray = r * 0.2989+ # G Pointer (Column 1)+ ptr_g = tl.make_block_ptr(+ base=input_ptr,+ shape=(n_elements, 3),+ strides=(3, 1),+ offsets=(pid * BLOCK_SIZE, 1),+ block_shape=(BLOCK_SIZE, 1),+ order=(1, 0)+ )- # G Channel- in_ptr = tl.advance(in_ptr, (0, 1))- g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")- gray = tl.fma(g, 0.5870, gray)+ # B Pointer (Column 2)+ ptr_b = tl.make_block_ptr(+ base=input_ptr,+ shape=(n_elements, 3),+ strides=(3, 1),+ offsets=(pid * BLOCK_SIZE, 2),+ block_shape=(BLOCK_SIZE, 1),+ order=(1, 0)+ )- # B Channel- in_ptr = tl.advance(in_ptr, (0, 1))- b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")- gray = tl.fma(b, 0.1140, gray)+ # -----------------------------------------------------------+ # 2. Parallel Load+ # -----------------------------------------------------------+ # 3つのロードを連続記述。依存関係がないため即座に発行されます。+ # boundary_check は (0,) = 行方向のみチェック (列は1なので安全)++ # Load (BLOCK_SIZE, 1)+ r = tl.load(ptr_r, boundary_check=(0,), padding_option="zero")+ g = tl.load(ptr_g, boundary_check=(0,), padding_option="zero")+ b = tl.load(ptr_b, boundary_check=(0,), padding_option="zero")- # --- Output Block Pointer ---- # Shape: (N), Block: (BS)+ # -----------------------------------------------------------+ # 3. Parallel Computation (ILP)+ # -----------------------------------------------------------+ # 以前のエラー原因だった「複雑なスライシング」は不要です。+ # 単純な (BLOCK_SIZE, 1) 同士の計算なので絶対に落ちません。++ # 積和演算の依存を切り、3並列で計算させます+ term_r = r * 0.2989+ term_g = g * 0.5870+ term_b = b * 0.1140++ # 合算+ gray = term_r + term_g + term_b++ # -----------------------------------------------------------+ # 4. Output+ # -----------------------------------------------------------+ # Output Block Pointerout_ptr = tl.make_block_ptr(base=output_ptr,shape=(n_elements,),⋯ 2 unchanged linesblock_shape=(BLOCK_SIZE,),order=(0,))++ # gray は (BLOCK_SIZE, 1) なので、(BLOCK_SIZE,) に戻してストア+ # reshape はデータ移動を伴わないメタデータ操作なのでコストゼロです+ tl.store(out_ptr, tl.reshape(gray, (BLOCK_SIZE,)), boundary_check=(0,))- # --- Reshape & Store ---- # (BS, 1) -> (BS)- gray_flat = tl.reshape(gray, (BLOCK_SIZE,))- tl.store(out_ptr, gray_flat, boundary_check=(0,))-def custom_kernel(data: input_t) -> output_t:input_tensor, output_tensor = datan_elements = output_tensor.numel()- grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']), )+ # Python Overhead Optimization+ # 関数呼び出しを避け、直接計算でグリッドを渡します+ grid = lambda META: ((n_elements + META['BLOCK_SIZE'] - 1) // META['BLOCK_SIZE'], )- grayscale_full_search_kernel[grid](+ grayscale_parallel_independent_kernel[grid](input_tensor, output_tensor,n_elements)
scrolls · 171 diff lines total
Best evidence level for this revision: reported
JSON