submission 227239
HayatoFujihara · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 86 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_v2_9.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-227239?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:397f6d21e580070d8de9138bc07a7caf3d4b76836b777b2a47213cd63f25a21c
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
@triton.autotune(num-warps = 4
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),stages = 3
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),Kernel source
grayscale_v2_9.py86 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t
# =============================================================================
# Block Pointer + advance() 版(ベースライン: 2.44ms)
# =============================================================================
# 連続ロード+reshape方式は tl.arange() の2のべき乗制約により不可
# 入力側コンパイラヒントは正確性エラー
# → ベースラインに戻して安定性を確保
@triton.autotune(
configs=[
# 中ブロック
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),
triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),
# 大ブロック
triton.Config({'BLOCK_SIZE': 4096}, num_warps=4, num_stages=4),
triton.Config({'BLOCK_SIZE': 4096}, num_warps=4, num_stages=5),
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=3),
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),
triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5),
# 超大ブロック
triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=3),
triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),
triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=5),
triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=3),
triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=4),
],
key=['n_elements'],
)
@triton.jit
def grayscale_kernel(
input_ptr, output_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr
):
pid = tl.program_id(axis=0)
# Block Pointer
in_ptr = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
offsets=(pid * BLOCK_SIZE, 0),
block_shape=(BLOCK_SIZE, 1),
order=(1, 0)
)
# R Channel
r = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
gray = r * 0.2989
# G Channel
in_ptr = tl.advance(in_ptr, (0, 1))
g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
gray += g * 0.5870
# B Channel
in_ptr = tl.advance(in_ptr, (0, 1))
b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
gray += b * 0.1140
# Output (1Dポインタ)
offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
mask = offs < n_elements
tl.store(output_ptr + offs, tl.reshape(gray, (BLOCK_SIZE,)), mask=mask)
def custom_kernel(data: input_t) -> output_t:
input_tensor, output_tensor = data
n_elements = output_tensor.numel()
grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']),)
grayscale_kernel[grid](
input_tensor, output_tensor,
n_elements
)
return output_tensor
scrolls · 86 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 153879.
⋯ 2 unchanged linesimport triton.language as tlfrom task import input_t, output_t- # ------------------------------------------------------------------------------ # Final Stable & Fast: "Independent Block Pointers + Parallel FMA"- # ------------------------------------------------------------------------------ # 過去の実験で最も高速かつ安定していた2つの要素を統合します。- # 1. Block Pointer: アドレス計算負荷をゼロにする (2.47msの実績)- # 2. Parallel Math: R,G,Bの計算を独立させ、パイプラインを埋める (2.46msの実績)- #- # これらに対し、コンパイルエラーの原因となる「スライシング」や「変なReshape」を- # 一切行わないことで、確実に動作させます。+ # =============================================================================+ # Block Pointer + advance() 版(ベースライン: 2.44ms)+ # =============================================================================+ # 連続ロード+reshape方式は tl.arange() の2のべき乗制約により不可+ # 入力側コンパイラヒントは正確性エラー+ # → ベースラインに戻して安定性を確保@triton.autotune(configs=[- # A100のスイートスポット (4096 x 8warps)+ # 中ブロック+ triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=3),+ triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),+ triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),++ # 大ブロック+ triton.Config({'BLOCK_SIZE': 4096}, num_warps=4, num_stages=4),+ triton.Config({'BLOCK_SIZE': 4096}, num_warps=4, num_stages=5),+ triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=3),triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5),-- # 巨大ブロック (ループ回数削減)- # Warps=16 を投入し、計算レイテンシの隠蔽能力を最大化します++ # 超大ブロック+ triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=3),triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),+ triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=5),+ triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=3),triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=4),-- # 回転率重視- triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),- triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),],key=['n_elements'],)@triton.jit- def grayscale_parallel_independent_kernel(+ def grayscale_kernel(input_ptr, output_ptr,n_elements,BLOCK_SIZE: tl.constexpr):pid = tl.program_id(axis=0)-- # ------------------------------------------------------------ # 1. Independent Input Pointers (No Dependency Chain)- # ------------------------------------------------------------ # advance() を使わず、最初からオフセット済みのポインタを3つ生成します。- # これにより、GPUは R, G, B のロード命令を任意の順序で・同時に発行できます。-- # R Pointer (Column 0)- ptr_r = tl.make_block_ptr(++ # Block Pointer+ in_ptr = tl.make_block_ptr(base=input_ptr,shape=(n_elements, 3),strides=(3, 1),⋯ 2 unchanged linesorder=(1, 0))- # G Pointer (Column 1)- ptr_g = tl.make_block_ptr(- base=input_ptr,- shape=(n_elements, 3),- strides=(3, 1),- offsets=(pid * BLOCK_SIZE, 1),- block_shape=(BLOCK_SIZE, 1),- order=(1, 0)- )+ # R Channel+ r = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")+ gray = r * 0.2989- # B Pointer (Column 2)- ptr_b = tl.make_block_ptr(- base=input_ptr,- shape=(n_elements, 3),- strides=(3, 1),- offsets=(pid * BLOCK_SIZE, 2),- block_shape=(BLOCK_SIZE, 1),- order=(1, 0)- )+ # G Channel+ in_ptr = tl.advance(in_ptr, (0, 1))+ g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")+ gray += g * 0.5870- # ------------------------------------------------------------ # 2. Parallel Load- # ------------------------------------------------------------ # 3つのロードを連続記述。依存関係がないため即座に発行されます。- # boundary_check は (0,) = 行方向のみチェック (列は1なので安全)-- # Load (BLOCK_SIZE, 1)- r = tl.load(ptr_r, boundary_check=(0,), padding_option="zero")- g = tl.load(ptr_g, boundary_check=(0,), padding_option="zero")- b = tl.load(ptr_b, boundary_check=(0,), padding_option="zero")+ # B Channel+ in_ptr = tl.advance(in_ptr, (0, 1))+ b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")+ gray += b * 0.1140- # ------------------------------------------------------------ # 3. Parallel Computation (ILP)- # ------------------------------------------------------------ # 以前のエラー原因だった「複雑なスライシング」は不要です。- # 単純な (BLOCK_SIZE, 1) 同士の計算なので絶対に落ちません。-- # 積和演算の依存を切り、3並列で計算させます- term_r = r * 0.2989- term_g = g * 0.5870- term_b = b * 0.1140-- # 合算- gray = term_r + term_g + term_b+ # Output (1Dポインタ)+ offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)+ mask = offs < n_elements+ tl.store(output_ptr + offs, tl.reshape(gray, (BLOCK_SIZE,)), mask=mask)- # ------------------------------------------------------------ # 4. Output- # ------------------------------------------------------------ # Output Block Pointer- out_ptr = tl.make_block_ptr(- base=output_ptr,- shape=(n_elements,),- strides=(1,),- offsets=(pid * BLOCK_SIZE,),- block_shape=(BLOCK_SIZE,),- order=(0,)- )-- # gray は (BLOCK_SIZE, 1) なので、(BLOCK_SIZE,) に戻してストア- # reshape はデータ移動を伴わないメタデータ操作なのでコストゼロです- tl.store(out_ptr, tl.reshape(gray, (BLOCK_SIZE,)), boundary_check=(0,))-def custom_kernel(data: input_t) -> output_t:input_tensor, output_tensor = datan_elements = output_tensor.numel()-- # Python Overhead Optimization- # 関数呼び出しを避け、直接計算でグリッドを渡します- grid = lambda META: ((n_elements + META['BLOCK_SIZE'] - 1) // META['BLOCK_SIZE'], )-- grayscale_parallel_independent_kernel[grid](++ grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']),)++ grayscale_kernel[grid](input_tensor, output_tensor,n_elements)- return output_tensorNo newline at end of file+ return output_tensor
scrolls · 176 diff lines total
Best evidence level for this revision: reported
JSON