Skip to content
KernelIndex
Search⌘K

submission 153868

HayatoFujihara · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 93 lines, June 9 Researcher Reciprocity License v1.0.

grayscale_v2_5.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-grayscale-v2-153868?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RGB to grayscalesuite of 6 cases
NVIDIA B200
621.3µs
#52 of 84
2025-12-13

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:5415efaa56b814e2271edf4a00a71ae5dc3f9de792e23fabf6ac0cf509e720b3
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotunetriton.Config({'BLOCK_SIZE': bs}, num_warps=w, num_stages=s)

Kernel source

grayscale_v2_5.py93 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t

# -----------------------------------------------------------------------------
# Configuration Search Space
# -----------------------------------------------------------------------------
# 人手による絞り込みをやめ、機械的に全探索します。
# A100で可能性のある範囲を網羅します。

block_sizes = [1024, 2048, 4096, 8192, 16384]
num_warps_list = [4, 8, 16]
num_stages_list = [3, 4, 5, 6]

configs = [
    triton.Config({'BLOCK_SIZE': bs}, num_warps=w, num_stages=s)
    for bs in block_sizes
    for w in num_warps_list
    for s in num_stages_list
]

# -----------------------------------------------------------------------------
# Kernel Implementation: Full Block Pointer (Best Logic)
# -----------------------------------------------------------------------------
@triton.autotune(
    configs=configs,
    key=['n_elements'],
)
@triton.jit
def grayscale_full_search_kernel(
    input_ptr, output_ptr,
    n_elements,
    BLOCK_SIZE: tl.constexpr
):
    pid = tl.program_id(axis=0)
    
    # --- Input Block Pointer ---
    # Shape: (N, 3), Block: (BS, 1)
    in_ptr = tl.make_block_ptr(
        base=input_ptr,
        shape=(n_elements, 3),
        strides=(3, 1),
        offsets=(pid * BLOCK_SIZE, 0),
        block_shape=(BLOCK_SIZE, 1),
        order=(1, 0)
    )

    # --- Load & Compute (Sequential) ---
    
    # R Channel
    r = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
    gray = r * 0.2989

    # G Channel
    in_ptr = tl.advance(in_ptr, (0, 1))
    g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
    gray = tl.fma(g, 0.5870, gray)

    # B Channel
    in_ptr = tl.advance(in_ptr, (0, 1))
    b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
    gray = tl.fma(b, 0.1140, gray)

    # --- Output Block Pointer ---
    # Shape: (N), Block: (BS)
    out_ptr = tl.make_block_ptr(
        base=output_ptr,
        shape=(n_elements,),
        strides=(1,),
        offsets=(pid * BLOCK_SIZE,),
        block_shape=(BLOCK_SIZE,),
        order=(0,)
    )

    # --- Reshape & Store ---
    # (BS, 1) -> (BS)
    gray_flat = tl.reshape(gray, (BLOCK_SIZE,))
    tl.store(out_ptr, gray_flat, boundary_check=(0,))


def custom_kernel(data: input_t) -> output_t:
    input_tensor, output_tensor = data
    n_elements = output_tensor.numel()
    
    grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']), )
    
    grayscale_full_search_kernel[grid](
        input_tensor, output_tensor,
        n_elements
    )

    return output_tensor
scrolls · 93 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 153816.

⋯ 3 unchanged lines
from task import input_t, output_t
# -----------------------------------------------------------------------------
- # Final Optimization: "Parallel Arithmetic & Lean Launch"
+ # Configuration Search Space
# -----------------------------------------------------------------------------
- # 1. Parallel FMA: 積和演算の依存チェーンを断ち切り、R,G,Bの計算を同時に行わせます。
- # 2. Optimized Launch: triton.cdiv などの関数呼び出しをやめ、直接計算します。
+ # 人手による絞り込みをやめ、機械的に全探索します。
+ # A100で可能性のある範囲を網羅します。
+ block_sizes = [1024, 2048, 4096, 8192, 16384]
+ num_warps_list = [4, 8, 16]
+ num_stages_list = [3, 4, 5, 6]
+
+ configs = [
+ triton.Config({'BLOCK_SIZE': bs}, num_warps=w, num_stages=s)
+ for bs in block_sizes
+ for w in num_warps_list
+ for s in num_stages_list
+ ]
+
+ # -----------------------------------------------------------------------------
+ # Kernel Implementation: Full Block Pointer (Best Logic)
+ # -----------------------------------------------------------------------------
@triton.autotune(
- configs=[
- # A100の鉄板設定
- triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=4),
- triton.Config({'BLOCK_SIZE': 4096}, num_warps=8, num_stages=5), # 深いパイプライン
- triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=4),
- triton.Config({'BLOCK_SIZE': 2048}, num_warps=4, num_stages=5),
- # 巨大ブロックも一応入れておく
- triton.Config({'BLOCK_SIZE': 8192}, num_warps=8, num_stages=4),
- ],
+ configs=configs,
key=['n_elements'],
)
@triton.jit
- def grayscale_parallel_math_kernel(
+ def grayscale_full_search_kernel(
input_ptr, output_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr
⋯ 1 unchanged lines
pid = tl.program_id(axis=0)
# --- Input Block Pointer ---
+ # Shape: (N, 3), Block: (BS, 1)
in_ptr = tl.make_block_ptr(
base=input_ptr,
shape=(n_elements, 3),
⋯ 3 unchanged lines
order=(1, 0)
)
- # --- Load (Issuing 3 loads back-to-back) ---
- # ポインタを複製して advance させることで、独立したロード命令として発行
- # コンパイラがスケジューリングしやすくなります
+ # --- Load & Compute (Sequential) ---
# R Channel
r = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
-
+ gray = r * 0.2989
+
# G Channel
- in_ptr_g = tl.advance(in_ptr, (0, 1))
- g = tl.load(in_ptr_g, boundary_check=(0,), padding_option="zero")
-
+ in_ptr = tl.advance(in_ptr, (0, 1))
+ g = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
+ gray = tl.fma(g, 0.5870, gray)
+
# B Channel
- in_ptr_b = tl.advance(in_ptr_g, (0, 1))
- b = tl.load(in_ptr_b, boundary_check=(0,), padding_option="zero")
+ in_ptr = tl.advance(in_ptr, (0, 1))
+ b = tl.load(in_ptr, boundary_check=(0,), padding_option="zero")
+ gray = tl.fma(b, 0.1140, gray)
- # --- Parallel Computation (ILP) ---
- # 以前: gray = r*c1; gray += g*c2; gray += b*c3 (直列依存)
- # 今回: term1, term2, term3 を並列計算 -> 最後に合算
-
- term1 = r * 0.2989
- term2 = g * 0.5870
- term3 = b * 0.1140
-
- # 合算
- gray = term1 + term2 + term3
-
- # --- Output Block Pointer & Store ---
+ # --- Output Block Pointer ---
+ # Shape: (N), Block: (BS)
out_ptr = tl.make_block_ptr(
base=output_ptr,
shape=(n_elements,),
⋯ 3 unchanged lines
order=(0,)
)
+ # --- Reshape & Store ---
+ # (BS, 1) -> (BS)
gray_flat = tl.reshape(gray, (BLOCK_SIZE,))
tl.store(out_ptr, gray_flat, boundary_check=(0,))
⋯ 2 unchanged lines
input_tensor, output_tensor = data
n_elements = output_tensor.numel()
- # Python Overhead Optimization:
- # triton.cdiv(n, b) は (n + b - 1) // b と等価ですが、
- # 関数呼び出しオーバーヘッドを嫌って直接書きます。
+ grid = lambda META: (triton.cdiv(n_elements, META['BLOCK_SIZE']), )
- # ベストな BLOCK_SIZE は autotune で選ばれますが、
- # 起動時にはその値を使ってグリッドを計算する必要があります。
- grid = lambda META: ((n_elements + META['BLOCK_SIZE'] - 1) // META['BLOCK_SIZE'], )
-
- grayscale_parallel_math_kernel[grid](
+ grayscale_full_search_kernel[grid](
input_tensor, output_tensor,
n_elements
)
scrolls · 126 diff lines total

Best evidence level for this revision: reported

JSON