Skip to content
KernelIndex
Search⌘K

submission 128154

HayatoFujihara · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 69 lines, June 9 Researcher Reciprocity License v1.0.

submission2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-128154?include=source"
interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Vector sum reductionsuite of 6 cases
NVIDIA A100
150.5µs
#48 of 96
2025-12-07

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:e21070b6115e162455c8b2a0e98656a9f6ce635c0772b55957de7467cea931e3
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 8num_warps=8

Kernel source

submission2.py69 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t

# 固定パラメータ版のカーネル定義
@triton.jit
def _sum_kernel_fixed(
    x_ptr,
    temp_ptr,
    n_elements,
    BLOCK_SIZE: tl.constexpr,
):
    pid = tl.program_id(axis=0)
    block_start = pid * BLOCK_SIZE
    offsets = block_start + tl.arange(0, BLOCK_SIZE)
    mask = offsets < n_elements
    
    # 読み込み & float32キャスト
    x = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
    
    # ブロック内リダクション (競合なし)
    block_sum = tl.sum(x, axis=0)
    
    # 各ブロック専用の場所に書き込み (競合なし)
    tl.store(temp_ptr + pid, block_sum)

def custom_kernel(data: input_t) -> output_t:
    """
    Triton Two-Pass Reduction (Fixed Configuration)
    1. GPU: 各ブロック(4096要素)ごとの小計を計算し、一時バッファに書き込む(Atomicなしで高速)
    2. CPU/GPU: 一時バッファの総和をPyTorchで計算する
    """
    input_tensor, _ = data
    
    # 連続メモリ化
    x = input_tensor.view(-1)
    if not x.is_contiguous():
        x = x.contiguous()
        
    n_elements = x.numel()

    # パラメータ設定 (チューニング済み想定)
    # 4096はA100/H100等の最新GPUでメモリ帯域を使い切りやすいサイズ
    BLOCK_SIZE = 4096 
    
    # グリッドサイズ計算 (切り上げ除算)
    # triton.cdiv は (a + b - 1) // b
    actual_blocks = triton.cdiv(n_elements, BLOCK_SIZE)
    
    # 一時バッファ(各ブロックの合計値が入る)
    temp_buffer = torch.empty(actual_blocks, device=input_tensor.device, dtype=torch.float32)

    # カーネル起動
    # ここに [grid] 指定を追加しました。これが抜けていたのがエラーの原因です。
    grid = (actual_blocks, )
    
    # num_warps=8 を指定して並列度を上げます (BLOCK_SIZEが大きい場合に有効)
    _sum_kernel_fixed[grid](
        x, 
        temp_buffer, 
        n_elements, 
        BLOCK_SIZE=BLOCK_SIZE,
        num_warps=8 
    )

    # Phase 2: 残った小計(数千〜数万個)を合計する
    # このコストは無視できるほど小さい
    return torch.sum(temp_buffer)
scrolls · 69 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 128138.

import torch
+ import triton
+ import triton.language as tl
from task import input_t, output_t
+ # 固定パラメータ版のカーネル定義
+ @triton.jit
+ def _sum_kernel_fixed(
+ x_ptr,
+ temp_ptr,
+ n_elements,
+ BLOCK_SIZE: tl.constexpr,
+ ):
+ pid = tl.program_id(axis=0)
+ block_start = pid * BLOCK_SIZE
+ offsets = block_start + tl.arange(0, BLOCK_SIZE)
+ mask = offsets < n_elements
+
+ # 読み込み & float32キャスト
+ x = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
+
+ # ブロック内リダクション (競合なし)
+ block_sum = tl.sum(x, axis=0)
+
+ # 各ブロック専用の場所に書き込み (競合なし)
+ tl.store(temp_ptr + pid, block_sum)
+
def custom_kernel(data: input_t) -> output_t:
"""
- Fastest candidate: single-shot reduction.
- - 全体を一発で float32 集約
- - 出力は float32 スカラー
+ Triton Two-Pass Reduction (Fixed Configuration)
+ 1. GPU: 各ブロック(4096要素)ごとの小計を計算し、一時バッファに書き込む(Atomicなしで高速)
+ 2. CPU/GPU: 一時バッファの総和をPyTorchで計算する
"""
input_tensor, _ = data
+
+ # 連続メモリ化
+ x = input_tensor.view(-1)
+ if not x.is_contiguous():
+ x = x.contiguous()
+
+ n_elements = x.numel()
- # 一発で float32 集約
- result = torch.sum(input_tensor, dtype=torch.float32)
+ # パラメータ設定 (チューニング済み想定)
+ # 4096はA100/H100等の最新GPUでメモリ帯域を使い切りやすいサイズ
+ BLOCK_SIZE = 4096
+
+ # グリッドサイズ計算 (切り上げ除算)
+ # triton.cdiv は (a + b - 1) // b
+ actual_blocks = triton.cdiv(n_elements, BLOCK_SIZE)
+
+ # 一時バッファ(各ブロックの合計値が入る)
+ temp_buffer = torch.empty(actual_blocks, device=input_tensor.device, dtype=torch.float32)
- return result
+ # カーネル起動
+ # ここに [grid] 指定を追加しました。これが抜けていたのがエラーの原因です。
+ grid = (actual_blocks, )
+
+ # num_warps=8 を指定して並列度を上げます (BLOCK_SIZEが大きい場合に有効)
+ _sum_kernel_fixed[grid](
+ x,
+ temp_buffer,
+ n_elements,
+ BLOCK_SIZE=BLOCK_SIZE,
+ num_warps=8
+ )
+
+ # Phase 2: 残った小計(数千〜数万個)を合計する
+ # このコストは無視できるほど小さい
+ return torch.sum(temp_buffer)
No newline at end of file
scrolls · 76 diff lines total

Best evidence level for this revision: reported

JSON