Skip to content
KernelIndex
Search⌘K

submission 128264

HayatoFujihara · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 89 lines, June 9 Researcher Reciprocity License v1.0.

submission2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-128264?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA A100
139.4µs
#9 of 96
2025-12-07

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:414288d1527b1dbb84d86d39157a01d79fbece853b9dd6393a28ee367a67fb89
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotune@triton.autotune(
num-warps = 16triton.Config({}, num_warps=16, num_stages=2),
stages = 2triton.Config({}, num_warps=16, num_stages=2),

Kernel source

submission2.py89 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t

# BLOCK_SIZEは固定しますが、内部の実行パラメータを徹底的にチューニングします
# pre_hook不要(Storeは上書きなので何度実行しても安全)
@triton.autotune(
    configs=[
        # BLOCK_SIZE=32768 に対する最適設定を探る
        # Warpsを増やすことで、大量の要素処理時のレイテンシを隠蔽
        triton.Config({}, num_warps=16, num_stages=2),
        triton.Config({}, num_warps=32, num_stages=2),
        triton.Config({}, num_warps=16, num_stages=4),
        triton.Config({}, num_warps=32, num_stages=4),
        # 環境によってはWarp数が多すぎるとレジスタ溢れするため、少なめの設定も保険に入れる
        triton.Config({}, num_warps=8, num_stages=2),
    ],
    key=['n_elements'],
)
@triton.jit
def _sum_kernel_map_fixed_block(
    x_ptr,
    temp_ptr,
    n_elements,
    # BLOCK_SIZEはコンパイル時定数として渡すが、値は固定
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(axis=0)
    
    # 担当領域計算
    block_start = pid * BLOCK_SIZE
    offsets = block_start + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_elements

    # Load & Cast
    # 32768要素を一気にロード。Tritonが自動でベクトル化・分割して最適化します
    x = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float32)

    # ブロック内リダクション
    block_sum = tl.sum(x, axis=0)

    # Store (Atomicを使わず上書き)
    # これにより "null" 落ちやテスト失敗(非決定性)を回避
    tl.store(temp_ptr + pid, block_sum)

def custom_kernel(data: input_t) -> output_t:
    """
    Triton Optimized Map-Reduce (Fixed Huge Block).
    - Fixes BLOCK_SIZE to Maximize Bandwidth & Determine Grid Size.
    - Autotunes Warps/Stages for Hardware Optimization.
    - Uses torch.empty for zero-overhead allocation.
    """
    input_tensor, _ = data
    x = input_tensor.view(-1)
    
    if not x.is_contiguous():
        x = x.contiguous()
        
    n_elements = x.numel()
    
    # 【高速化の鍵】BLOCK_SIZEを32768に固定
    # 5000万要素 ÷ 32768 ≒ 1526 ブロック
    # これによりAtomic競合を無くしつつ、ループオーバーヘッドも最小化
    BLOCK_SIZE = 32768
    
    # グリッドサイズを確定
    grid_size = triton.cdiv(n_elements, BLOCK_SIZE)
    
    # 【高速化の鍵】torch.emptyを使用
    # グリッドサイズ分だけ確保。初期化しない(カーネルが全要素を上書きするため安全)
    # zerosの初期化コスト(数µs)をカット
    temp_buffer = torch.empty(grid_size, device=input_tensor.device, dtype=torch.float32)
    
    # Grid定義
    grid = (grid_size, )
    
    # カーネル実行(内部でAutotuneが走り、最適なnum_warpsが選ばれる)
    # kwargsでBLOCK_SIZEを渡す
    _sum_kernel_map_fixed_block[grid](
        x, 
        temp_buffer, 
        n_elements, 
        BLOCK_SIZE=BLOCK_SIZE
    )
    
    # Phase 2: わずか~1500要素の足し算
    # ここはPyTorchのC++実装が一瞬で処理する
    return torch.sum(temp_buffer)
scrolls · 89 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 128154.

⋯ 2 unchanged lines
import triton.language as tl
from task import input_t, output_t
- # 固定パラメータ版のカーネル定義
+ # BLOCK_SIZEは固定しますが、内部の実行パラメータを徹底的にチューニングします
+ # pre_hook不要(Storeは上書きなので何度実行しても安全)
+ @triton.autotune(
+ configs=[
+ # BLOCK_SIZE=32768 に対する最適設定を探る
+ # Warpsを増やすことで、大量の要素処理時のレイテンシを隠蔽
+ triton.Config({}, num_warps=16, num_stages=2),
+ triton.Config({}, num_warps=32, num_stages=2),
+ triton.Config({}, num_warps=16, num_stages=4),
+ triton.Config({}, num_warps=32, num_stages=4),
+ # 環境によってはWarp数が多すぎるとレジスタ溢れするため、少なめの設定も保険に入れる
+ triton.Config({}, num_warps=8, num_stages=2),
+ ],
+ key=['n_elements'],
+ )
@triton.jit
- def _sum_kernel_fixed(
+ def _sum_kernel_map_fixed_block(
x_ptr,
temp_ptr,
n_elements,
+ # BLOCK_SIZEはコンパイル時定数として渡すが、値は固定
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(axis=0)
+
+ # 担当領域計算
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
-
- # 読み込み & float32キャスト
+
+ # Load & Cast
+ # 32768要素を一気にロード。Tritonが自動でベクトル化・分割して最適化します
x = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
-
- # ブロック内リダクション (競合なし)
+
+ # ブロック内リダクション
block_sum = tl.sum(x, axis=0)
-
- # 各ブロック専用の場所に書き込み (競合なし)
+
+ # Store (Atomicを使わず上書き)
+ # これにより "null" 落ちやテスト失敗(非決定性)を回避
tl.store(temp_ptr + pid, block_sum)
def custom_kernel(data: input_t) -> output_t:
"""
- Triton Two-Pass Reduction (Fixed Configuration)
- 1. GPU: 各ブロック(4096要素)ごとの小計を計算し、一時バッファに書き込む(Atomicなしで高速)
- 2. CPU/GPU: 一時バッファの総和をPyTorchで計算する
+ Triton Optimized Map-Reduce (Fixed Huge Block).
+ - Fixes BLOCK_SIZE to Maximize Bandwidth & Determine Grid Size.
+ - Autotunes Warps/Stages for Hardware Optimization.
+ - Uses torch.empty for zero-overhead allocation.
"""
input_tensor, _ = data
-
- # 連続メモリ化
x = input_tensor.view(-1)
+
if not x.is_contiguous():
x = x.contiguous()
n_elements = x.numel()
-
- # パラメータ設定 (チューニング済み想定)
- # 4096はA100/H100等の最新GPUでメモリ帯域を使い切りやすいサイズ
- BLOCK_SIZE = 4096
- # グリッドサイズ計算 (切り上げ除算)
- # triton.cdiv は (a + b - 1) // b
- actual_blocks = triton.cdiv(n_elements, BLOCK_SIZE)
+ # 【高速化の鍵】BLOCK_SIZEを32768に固定
+ # 5000万要素 ÷ 32768 ≒ 1526 ブロック
+ # これによりAtomic競合を無くしつつ、ループオーバーヘッドも最小化
+ BLOCK_SIZE = 32768
- # 一時バッファ(各ブロックの合計値が入る)
- temp_buffer = torch.empty(actual_blocks, device=input_tensor.device, dtype=torch.float32)
-
- # カーネル起動
- # ここに [grid] 指定を追加しました。これが抜けていたのがエラーの原因です。
- grid = (actual_blocks, )
+ # グリッドサイズを確定
+ grid_size = triton.cdiv(n_elements, BLOCK_SIZE)
- # num_warps=8 を指定して並列度を上げます (BLOCK_SIZEが大きい場合に有効)
- _sum_kernel_fixed[grid](
+ # 【高速化の鍵】torch.emptyを使用
+ # グリッドサイズ分だけ確保。初期化しない(カーネルが全要素を上書きするため安全)
+ # zerosの初期化コスト(数µs)をカット
+ temp_buffer = torch.empty(grid_size, device=input_tensor.device, dtype=torch.float32)
+
+ # Grid定義
+ grid = (grid_size, )
+
+ # カーネル実行(内部でAutotuneが走り、最適なnum_warpsが選ばれる)
+ # kwargsでBLOCK_SIZEを渡す
+ _sum_kernel_map_fixed_block[grid](
x,
temp_buffer,
n_elements,
- BLOCK_SIZE=BLOCK_SIZE,
- num_warps=8
+ BLOCK_SIZE=BLOCK_SIZE
)
-
- # Phase 2: 残った小計(数千〜数万個)を合計する
- # このコストは無視できるほど小さい
+
+ # Phase 2: わずか~1500要素の足し算
+ # ここはPyTorchのC++実装が一瞬で処理する
return torch.sum(temp_buffer)
No newline at end of file
scrolls · 122 diff lines total

Best evidence level for this revision: reported

JSON