submission 153868
HayatoFujihara · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 93 lines, June 9 Researcher Reciprocity License v1.0.
grayscale_v2_5.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-153868?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:5415efaa56b814e2271edf4a00a71ae5dc3f9de792e23fabf6ac0cf509e720b3
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
autotune
triton.Config({'BLOCK_SIZE': bs}, num_warps=w, num_stages=s)Kernel source
grayscale_v2_5.py93 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t
# -----------------------------------------------------------------------------
# Configuration Search Space
# -----------------------------------------------------------------------------
# 人手による絞り込みをやめ、機械的に全探索します。
# A100で可能性のある範囲を網羅します。
block_sizes = [1024, 2048, 4096, 8192, 16384]
num_warps_list = [4, 8, 16]
num_stages_list = [3, 4, 5, 6]
configs = [
triton.Config({'BLOCK_SIZE': bs}, num_warps=w, num_stages=s)
for bs in block_sizes
for w in num_warps_list
for s in num_stages_list
]
# -----------------------------------------------------------------------------
# Kernel Implementation: Full Block Pointer (Best Logic)
# -----------------------------------------------------------------------------
@triton.autotune(
configs=configs,
key=['n_elements'],
)
@triton.jit
def grayscale_full_search_kernel(
input_ptr, output_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr
):
pid = tl.program_id(axis=0)
# --- Input Block Pointer ---
# Shape: (N, 3), Block: (BS, 1)
in_ptr = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
offsets=(pid * BLOCK_SIZE, 0),
block_shape=(BLOCK_SIZE, 1),
order=(1, 0)
)
# --- Load & Compute (Sequential) ---
# R Channel
r = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
gray = r * 0.2989
# G Channel
in_ptr = tl.advance(in_ptr, (0, 1))
g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
gray = tl.fma(g, 0.5870, gray)
# B Channel
in_ptr = tl.advance(in_ptr, (0, 1))
b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
gray = tl.fma(b, 0.1140, gray)
# --- Output Block Pointer ---
# Shape: (N), Block: (BS)
out_ptr = tl.make_block_ptr(
base=output_ptr,
shape=(n_elements,),
strides=(1,),
offsets=(pid * BLOCK_SIZE,),
block_shape=(BLOCK_SIZE,),
order=(0,)
)
# --- Reshape & Store ---
# (BS, 1) -> (BS)
gray_flat = tl.reshape(gray, (BLOCK_SIZE,))
tl.store(out_ptr, gray_flat, boundary_check=(0,))
def custom_kernel(data: input_t) -> output_t:
input_tensor, output_tensor = data
n_elements = output_tensor.numel()
grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']), )
grayscale_full_search_kernel[grid](
input_tensor, output_tensor,
n_elements
)
return output_tensorscrolls · 93 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 153816.
⋯ 3 unchanged linesfrom task import input_t, output_t# ------------------------------------------------------------------------------ # Final Optimization: "Parallel Arithmetic & Lean Launch"+ # Configuration Search Space# ------------------------------------------------------------------------------ # 1. Parallel FMA: 積和演算の依存チェーンを断ち切り、R,G,Bの計算を同時に行わせます。- # 2. Optimized Launch: triton.cdiv などの関数呼び出しをやめ、直接計算します。+ # 人手による絞り込みをやめ、機械的に全探索します。+ # A100で可能性のある範囲を網羅します。+ block_sizes = [1024, 2048, 4096, 8192, 16384]+ num_warps_list = [4, 8, 16]+ num_stages_list = [3, 4, 5, 6]++ configs = [+ triton.Config({'BLOCK_SIZE': bs}, num_warps=w, num_stages=s)+ for bs in block_sizes+ for w in num_warps_list+ for s in num_stages_list+ ]++ # -----------------------------------------------------------------------------+ # Kernel Implementation: Full Block Pointer (Best Logic)+ # -----------------------------------------------------------------------------@triton.autotune(- configs=[- # A100の鉄板設定- triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),- triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5), # 深いパイプライン- triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),- triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),- # 巨大ブロックも一応入れておく- triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),- ],+ configs=configs,key=['n_elements'],)@triton.jit- def grayscale_parallel_math_kernel(+ def grayscale_full_search_kernel(input_ptr, output_ptr,n_elements,BLOCK_SIZE: tl.constexpr⋯ 1 unchanged linespid = tl.program_id(axis=0)# --- Input Block Pointer ---+ # Shape: (N, 3), Block: (BS, 1)in_ptr = tl.make_block_ptr(base=input_ptr,shape=(n_elements, 3),⋯ 3 unchanged linesorder=(1, 0))- # --- Load (Issuing 3 loads back-to-back) ---- # ポインタを複製して advance させることで、独立したロード命令として発行- # コンパイラがスケジューリングしやすくなります+ # --- Load & Compute (Sequential) ---# R Channelr = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")-+ gray = r * 0.2989+# G Channel- in_ptr_g = tl.advance(in_ptr, (0, 1))- g = tl.load(in_ptr_g, boundary_check=(0,), padding_option="zero")-+ in_ptr = tl.advance(in_ptr, (0, 1))+ g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")+ gray = tl.fma(g, 0.5870, gray)+# B Channel- in_ptr_b = tl.advance(in_ptr_g, (0, 1))- b = tl.load(in_ptr_b, boundary_check=(0,), padding_option="zero")+ in_ptr = tl.advance(in_ptr, (0, 1))+ b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")+ gray = tl.fma(b, 0.1140, gray)- # --- Parallel Computation (ILP) ---- # 以前: gray = r*c1; gray += g*c2; gray += b*c3 (直列依存)- # 今回: term1, term2, term3 を並列計算 -> 最後に合算-- term1 = r * 0.2989- term2 = g * 0.5870- term3 = b * 0.1140-- # 合算- gray = term1 + term2 + term3-- # --- Output Block Pointer & Store ---+ # --- Output Block Pointer ---+ # Shape: (N), Block: (BS)out_ptr = tl.make_block_ptr(base=output_ptr,shape=(n_elements,),⋯ 3 unchanged linesorder=(0,))+ # --- Reshape & Store ---+ # (BS, 1) -> (BS)gray_flat = tl.reshape(gray, (BLOCK_SIZE,))tl.store(out_ptr, gray_flat, boundary_check=(0,))⋯ 2 unchanged linesinput_tensor, output_tensor = datan_elements = output_tensor.numel()- # Python Overhead Optimization:- # triton.cdiv(n, b) は (n + b - 1) // b と等価ですが、- # 関数呼び出しオーバーヘッドを嫌って直接書きます。+ grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']), )- # ベストな BLOCK_SIZE は autotune で選ばれますが、- # 起動時にはその値を使ってグリッドを計算する必要があります。- grid = lambda META: ((n_elements + META['BLOCK_SIZE'] - 1) // META['BLOCK_SIZE'], )-- grayscale_parallel_math_kernel[grid](+ grayscale_full_search_kernel[grid](input_tensor, output_tensor,n_elements)
scrolls · 126 diff lines total
Best evidence level for this revision: reported
JSON