submission 128154
HayatoFujihara · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 69 lines, June 9 Researcher Reciprocity License v1.0.
submission2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-vectorsum-v2-128154?include=source"interfacepython
Compatibility
measured onNVIDIA A100
declared hardwareNVIDIA A100
architecturessm_80
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:e21070b6115e162455c8b2a0e98656a9f6ce635c0772b55957de7467cea931e3
license declaredunknown
license concludedunknown
authorsHayatoFujihara
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 8
num_warps=8Kernel source
submission2.py69 lines
import torch
import triton
import triton.language as tl
from task import input_t, output_t
# 固定パラメータ版のカーネル定義
@triton.jit
def _sum_kernel_fixed(
x_ptr,
temp_ptr,
n_elements,
BLOCK_SIZE: tl.constexpr,
):
pid = tl.program_id(axis=0)
block_start = pid * BLOCK_SIZE
offsets = block_start + tl.arange(0, BLOCK_SIZE)
mask = offsets < n_elements
# 読み込み & float32キャスト
x = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
# ブロック内リダクション (競合なし)
block_sum = tl.sum(x, axis=0)
# 各ブロック専用の場所に書き込み (競合なし)
tl.store(temp_ptr + pid, block_sum)
def custom_kernel(data: input_t) -> output_t:
"""
Triton Two-Pass Reduction (Fixed Configuration)
1. GPU: 各ブロック(4096要素)ごとの小計を計算し、一時バッファに書き込む(Atomicなしで高速)
2. CPU/GPU: 一時バッファの総和をPyTorchで計算する
"""
input_tensor, _ = data
# 連続メモリ化
x = input_tensor.view(-1)
if not x.is_contiguous():
x = x.contiguous()
n_elements = x.numel()
# パラメータ設定 (チューニング済み想定)
# 4096はA100/H100等の最新GPUでメモリ帯域を使い切りやすいサイズ
BLOCK_SIZE = 4096
# グリッドサイズ計算 (切り上げ除算)
# triton.cdiv は (a + b - 1) // b
actual_blocks = triton.cdiv(n_elements, BLOCK_SIZE)
# 一時バッファ(各ブロックの合計値が入る)
temp_buffer = torch.empty(actual_blocks, device=input_tensor.device, dtype=torch.float32)
# カーネル起動
# ここに [grid] 指定を追加しました。これが抜けていたのがエラーの原因です。
grid = (actual_blocks, )
# num_warps=8 を指定して並列度を上げます (BLOCK_SIZEが大きい場合に有効)
_sum_kernel_fixed[grid](
x,
temp_buffer,
n_elements,
BLOCK_SIZE=BLOCK_SIZE,
num_warps=8
)
# Phase 2: 残った小計(数千〜数万個)を合計する
# このコストは無視できるほど小さい
return torch.sum(temp_buffer)scrolls · 69 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 128138.
import torch+ import triton+ import triton.language as tlfrom task import input_t, output_t+ # 固定パラメータ版のカーネル定義+ @triton.jit+ def _sum_kernel_fixed(+ x_ptr,+ temp_ptr,+ n_elements,+ BLOCK_SIZE: tl.constexpr,+ ):+ pid = tl.program_id(axis=0)+ block_start = pid * BLOCK_SIZE+ offsets = block_start + tl.arange(0, BLOCK_SIZE)+ mask = offsets < n_elements++ # 読み込み & float32キャスト+ x = tl.load(x_ptr + offsets, mask=mask, other=0.0).to(tl.float32)++ # ブロック内リダクション (競合なし)+ block_sum = tl.sum(x, axis=0)++ # 各ブロック専用の場所に書き込み (競合なし)+ tl.store(temp_ptr + pid, block_sum)+def custom_kernel(data: input_t) -> output_t:"""- Fastest candidate: single-shot reduction.- - 全体を一発で float32 集約- - 出力は float32 スカラー+ Triton Two-Pass Reduction (Fixed Configuration)+ 1. GPU: 各ブロック(4096要素)ごとの小計を計算し、一時バッファに書き込む(Atomicなしで高速)+ 2. CPU/GPU: 一時バッファの総和をPyTorchで計算する"""input_tensor, _ = data++ # 連続メモリ化+ x = input_tensor.view(-1)+ if not x.is_contiguous():+ x = x.contiguous()++ n_elements = x.numel()- # 一発で float32 集約- result = torch.sum(input_tensor, dtype=torch.float32)+ # パラメータ設定 (チューニング済み想定)+ # 4096はA100/H100等の最新GPUでメモリ帯域を使い切りやすいサイズ+ BLOCK_SIZE = 4096++ # グリッドサイズ計算 (切り上げ除算)+ # triton.cdiv は (a + b - 1) // b+ actual_blocks = triton.cdiv(n_elements, BLOCK_SIZE)++ # 一時バッファ(各ブロックの合計値が入る)+ temp_buffer = torch.empty(actual_blocks, device=input_tensor.device, dtype=torch.float32)- return result+ # カーネル起動+ # ここに [grid] 指定を追加しました。これが抜けていたのがエラーの原因です。+ grid = (actual_blocks, )++ # num_warps=8 を指定して並列度を上げます (BLOCK_SIZEが大きい場合に有効)+ _sum_kernel_fixed[grid](+ x,+ temp_buffer,+ n_elements,+ BLOCK_SIZE=BLOCK_SIZE,+ num_warps=8+ )++ # Phase 2: 残った小計(数千〜数万個)を合計する+ # このコストは無視できるほど小さい+ return torch.sum(temp_buffer)No newline at end of file
scrolls · 76 diff lines total
Best evidence level for this revision: reported
JSON