Skip to content
KernelIndex
Search⌘K

submission 153879

HayatoFujihara · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 133 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_v2_6.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-153879?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA A100
2.48ms
#16 of 137
2025-12-13

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:da0eeb57f1d40fc1a78860d75dbe4544d315410b260d76270855ace9e13d5788
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotune@triton.autotune(
num-warps = 8triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),
stages = 4triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),

Kernel source

grayscale_v2_6.py133 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t

# -----------------------------------------------------------------------------
# Final Stable & Fast: "Independent Block Pointers + Parallel FMA"
# -----------------------------------------------------------------------------
# 過去の実験で最も高速かつ安定していた2つの要素を統合します。
# 1. Block Pointer: アドレス計算負荷をゼロにする (2.47msの実績)
# 2. Parallel Math: R,G,Bの計算を独立させ、パイプラインを埋める (2.46msの実績)
# 
# これらに対し、コンパイルエラーの原因となる「スライシング」や「変なReshape」を
# 一切行わないことで、確実に動作させます。

@triton.autotune(
    configs=[
        # A100のスイートスポット (4096 x 8warps)
        triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),
        triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5),
        
        # 巨大ブロック (ループ回数削減)
        # Warps=16 を投入し、計算レイテンシの隠蔽能力を最大化します
        triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),
        triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=4),
        
        # 回転率重視
        triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),
        triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),
    ],
    key=['n_elements'],
)
@triton.jit
def grayscale_parallel_independent_kernel(
    input_ptr, output_ptr,
    n_elements,
    BLOCK_SIZE: tl.constexpr
):
    pid = tl.program_id(axis=0)
    
    # -----------------------------------------------------------
    # 1. Independent Input Pointers (No Dependency Chain)
    # -----------------------------------------------------------
    # advance() を使わず、最初からオフセット済みのポインタを3つ生成します。
    # これにより、GPUは R, G, B のロード命令を任意の順序で・同時に発行できます。
    
    # R Pointer (Column 0)
    ptr_r = tl.make_block_ptr(
        base=input_ptr,
        shape=(n_elements, 3),
        strides=(3, 1),
        offsets=(pid * BLOCK_SIZE, 0),
        block_shape=(BLOCK_SIZE, 1),
        order=(1, 0)
    )

    # G Pointer (Column 1)
    ptr_g = tl.make_block_ptr(
        base=input_ptr,
        shape=(n_elements, 3),
        strides=(3, 1),
        offsets=(pid * BLOCK_SIZE, 1),
        block_shape=(BLOCK_SIZE, 1),
        order=(1, 0)
    )

    # B Pointer (Column 2)
    ptr_b = tl.make_block_ptr(
        base=input_ptr,
        shape=(n_elements, 3),
        strides=(3, 1),
        offsets=(pid * BLOCK_SIZE, 2),
        block_shape=(BLOCK_SIZE, 1),
        order=(1, 0)
    )

    # -----------------------------------------------------------
    # 2. Parallel Load
    # -----------------------------------------------------------
    # 3つのロードを連続記述。依存関係がないため即座に発行されます。
    # boundary_check は (0,) = 行方向のみチェック (列は1なので安全)
    
    # Load (BLOCK_SIZE, 1)
    r = tl.load(ptr_r, boundary_check=(0,), padding_option="zero")
    g = tl.load(ptr_g, boundary_check=(0,), padding_option="zero")
    b = tl.load(ptr_b, boundary_check=(0,), padding_option="zero")

    # -----------------------------------------------------------
    # 3. Parallel Computation (ILP)
    # -----------------------------------------------------------
    # 以前のエラー原因だった「複雑なスライシング」は不要です。
    # 単純な (BLOCK_SIZE, 1) 同士の計算なので絶対に落ちません。
    
    # 積和演算の依存を切り、3並列で計算させます
    term_r = r * 0.2989
    term_g = g * 0.5870
    term_b = b * 0.1140
    
    # 合算
    gray = term_r + term_g + term_b

    # -----------------------------------------------------------
    # 4. Output
    # -----------------------------------------------------------
    # Output Block Pointer
    out_ptr = tl.make_block_ptr(
        base=output_ptr,
        shape=(n_elements,),
        strides=(1,),
        offsets=(pid * BLOCK_SIZE,),
        block_shape=(BLOCK_SIZE,),
        order=(0,)
    )
    
    # gray は (BLOCK_SIZE, 1) なので、(BLOCK_SIZE,) に戻してストア
    # reshape はデータ移動を伴わないメタデータ操作なのでコストゼロです
    tl.store(out_ptr, tl.reshape(gray, (BLOCK_SIZE,)), boundary_check=(0,))


def custom_kernel(data: input_t) -> output_t:
    input_tensor, output_tensor = data
    n_elements = output_tensor.numel()
    
    # Python Overhead Optimization
    # 関数呼び出しを避け、直接計算でグリッドを渡します
    grid = lambda META: ((n_elements + META['BLOCK_SIZE'] - 1) // META['BLOCK_SIZE'], )
    
    grayscale_parallel_independent_kernel[grid](
        input_tensor, output_tensor,
        n_elements
    )

    return output_tensor
scrolls · 133 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 153868.

⋯ 3 unchanged lines
from task import input_t, output_t
# -----------------------------------------------------------------------------
- # Configuration Search Space
+ # Final Stable & Fast: "Independent Block Pointers + Parallel FMA"
# -----------------------------------------------------------------------------
- # 人手による絞り込みをやめ、機械的に全探索します。
- # A100で可能性のある範囲を網羅します。
+ # 過去の実験で最も高速かつ安定していた2つの要素を統合します。
+ # 1. Block Pointer: アドレス計算負荷をゼロにする (2.47msの実績)
+ # 2. Parallel Math: R,G,Bの計算を独立させ、パイプラインを埋める (2.46msの実績)
+ #
+ # これらに対し、コンパイルエラーの原因となる「スライシング」や「変なReshape」を
+ # 一切行わないことで、確実に動作させます。
- block_sizes = [1024, 2048, 4096, 8192, 16384]
- num_warps_list = [4, 8, 16]
- num_stages_list = [3, 4, 5, 6]
-
- configs = [
- triton.Config({'BLOCK_SIZE': bs}, num_warps=w, num_stages=s)
- for bs in block_sizes
- for w in num_warps_list
- for s in num_stages_list
- ]
-
- # -----------------------------------------------------------------------------
- # Kernel Implementation: Full Block Pointer (Best Logic)
- # -----------------------------------------------------------------------------
@triton.autotune(
- configs=configs,
+ configs=[
+ # A100のスイートスポット (4096 x 8warps)
+ triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),
+ triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5),
+
+ # 巨大ブロック (ループ回数削減)
+ # Warps=16 を投入し、計算レイテンシの隠蔽能力を最大化します
+ triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),
+ triton.Config({'BLOCK_SIZE': 8192}, num_warps=16, num_stages=4),
+
+ # 回転率重視
+ triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),
+ triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),
+ ],
key=['n_elements'],
)
@triton.jit
- def grayscale_full_search_kernel(
+ def grayscale_parallel_independent_kernel(
input_ptr, output_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr
):
pid = tl.program_id(axis=0)
- # --- Input Block Pointer ---
- # Shape: (N, 3), Block: (BS, 1)
- in_ptr = tl.make_block_ptr(
+ # -----------------------------------------------------------
+ # 1. Independent Input Pointers (No Dependency Chain)
+ # -----------------------------------------------------------
+ # advance() を使わず、最初からオフセット済みのポインタを3つ生成します。
+ # これにより、GPUは R, G, B のロード命令を任意の順序で・同時に発行できます。
+
+ # R Pointer (Column 0)
+ ptr_r = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
strides=(3, 1),
⋯ 2 unchanged lines
order=(1, 0)
)
- # --- Load & Compute (Sequential) ---
-
- # R Channel
- r = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
- gray = r * 0.2989
+ # G Pointer (Column 1)
+ ptr_g = tl.make_block_ptr(
+ base=input_ptr,
+ shape=(n_elements, 3),
+ strides=(3, 1),
+ offsets=(pid * BLOCK_SIZE, 1),
+ block_shape=(BLOCK_SIZE, 1),
+ order=(1, 0)
+ )
- # G Channel
- in_ptr = tl.advance(in_ptr, (0, 1))
- g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
- gray = tl.fma(g, 0.5870, gray)
+ # B Pointer (Column 2)
+ ptr_b = tl.make_block_ptr(
+ base=input_ptr,
+ shape=(n_elements, 3),
+ strides=(3, 1),
+ offsets=(pid * BLOCK_SIZE, 2),
+ block_shape=(BLOCK_SIZE, 1),
+ order=(1, 0)
+ )
- # B Channel
- in_ptr = tl.advance(in_ptr, (0, 1))
- b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
- gray = tl.fma(b, 0.1140, gray)
+ # -----------------------------------------------------------
+ # 2. Parallel Load
+ # -----------------------------------------------------------
+ # 3つのロードを連続記述。依存関係がないため即座に発行されます。
+ # boundary_check は (0,) = 行方向のみチェック (列は1なので安全)
+
+ # Load (BLOCK_SIZE, 1)
+ r = tl.load(ptr_r, boundary_check=(0,), padding_option="zero")
+ g = tl.load(ptr_g, boundary_check=(0,), padding_option="zero")
+ b = tl.load(ptr_b, boundary_check=(0,), padding_option="zero")
- # --- Output Block Pointer ---
- # Shape: (N), Block: (BS)
+ # -----------------------------------------------------------
+ # 3. Parallel Computation (ILP)
+ # -----------------------------------------------------------
+ # 以前のエラー原因だった「複雑なスライシング」は不要です。
+ # 単純な (BLOCK_SIZE, 1) 同士の計算なので絶対に落ちません。
+
+ # 積和演算の依存を切り、3並列で計算させます
+ term_r = r * 0.2989
+ term_g = g * 0.5870
+ term_b = b * 0.1140
+
+ # 合算
+ gray = term_r + term_g + term_b
+
+ # -----------------------------------------------------------
+ # 4. Output
+ # -----------------------------------------------------------
+ # Output Block Pointer
out_ptr = tl.make_block_ptr(
base=output_ptr,
shape=(n_elements,),
⋯ 2 unchanged lines
block_shape=(BLOCK_SIZE,),
order=(0,)
)
+
+ # gray は (BLOCK_SIZE, 1) なので、(BLOCK_SIZE,) に戻してストア
+ # reshape はデータ移動を伴わないメタデータ操作なのでコストゼロです
+ tl.store(out_ptr, tl.reshape(gray, (BLOCK_SIZE,)), boundary_check=(0,))
- # --- Reshape & Store ---
- # (BS, 1) -> (BS)
- gray_flat = tl.reshape(gray, (BLOCK_SIZE,))
- tl.store(out_ptr, gray_flat, boundary_check=(0,))
-
def custom_kernel(data: input_t) -> output_t:
input_tensor, output_tensor = data
n_elements = output_tensor.numel()
- grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']), )
+ # Python Overhead Optimization
+ # 関数呼び出しを避け、直接計算でグリッドを渡します
+ grid = lambda META: ((n_elements + META['BLOCK_SIZE'] - 1) // META['BLOCK_SIZE'], )
- grayscale_full_search_kernel[grid](
+ grayscale_parallel_independent_kernel[grid](
input_tensor, output_tensor,
n_elements
)
scrolls · 171 diff lines total

Best evidence level for this revision: reported

JSON