submission 553794
ramizzik · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 84 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-gated-deltanet-recompute-w-u-553794?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:22d03cc398c59f7fb5fff173ac24f22b42c6682cb5bbf2f61dc9bebdac9d193b
license declaredunknown
license concludedunknown
authorsramizzik
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 32
…s=[[0, 1]], maxnreg=32, num_sm_multiplier=16, num_stages=1, num_warps=32, pid_type='persistent_blocked', range_flattens=[None], range_multi_buffers=[False], range_num_stages=[3], r…persistent-kernel
…num_sm_multiplier=16, num_stages=1, num_warps=32, pid_type='persistent_blocked', range_flattens=[None], range_multi_buffers=[False], range_num_stages=[3], range_unroll_factors=[4],…stages = 1
…'], loop_orders=[[0, 1]], maxnreg=32, num_sm_multiplier=16, num_stages=1, num_warps=32, pid_type='persistent_blocked', range_flattens=[None], range_multi_buffers=[False], range_num…warp-specialization
…lse], range_num_stages=[3], range_unroll_factors=[4], range_warp_specializes=[None])…Kernel source
submission.py84 lines
#!POPCORN leaderboard gated_deltanet_recompute_w_u
#!POPCORN gpu B200_Nebius
from task import input_t, output_t
import torch
import helion
import helion.language as hl
# Autotuned config from reference (B200, full effort)
_TUNED = helion.Config(block_sizes=[], indexing=['tensor_descriptor', 'pointer', 'tensor_descriptor', 'tensor_descriptor', 'tensor_descriptor', 'tensor_descriptor', 'pointer'], l2_groupings=[16], load_eviction_policies=['', 'first', '', 'first', ''], loop_orders=[[0, 1]], maxnreg=32, num_sm_multiplier=16, num_stages=1, num_warps=32, pid_type='persistent_blocked', range_flattens=[None], range_multi_buffers=[False], range_num_stages=[3], range_unroll_factors=[4], range_warp_specializes=[None])
SHAPE_CONFIGS: dict[tuple, helion.Config] = {
# Test shapes
(1, 64, 2, 64, 64): _TUNED,
(2, 128, 4, 64, 64): _TUNED,
(1, 256, 4, 64, 128): _TUNED,
(1, 64, 1, 128, 128): _TUNED,
(2, 128, 2, 100, 100): _TUNED,
# Benchmark shapes
(1, 64, 1, 64, 64): _TUNED,
(2, 512, 3, 64, 64): _TUNED,
(2, 1024, 3, 64, 64): _TUNED,
(3, 1024, 4, 100, 100): _TUNED,
(4, 1024, 4, 128, 128): _TUNED,
(2, 1536, 4, 128, 128): _TUNED,
(4, 2048, 8, 64, 64): _TUNED,
}
def _make_kernel(config: helion.Config):
@helion.kernel(static_shapes=True, dot_precision="ieee", config=config)
def kernel(
k: torch.Tensor,
v: torch.Tensor,
beta: torch.Tensor,
A: torch.Tensor,
g: torch.Tensor,
) -> tuple[torch.Tensor, torch.Tensor]:
B, T, H, K = k.shape
V = v.shape[-1]
C = hl.specialize(A.shape[-1])
K = hl.specialize(K)
V = hl.specialize(V)
w_out = torch.empty_like(k)
u_out = torch.empty_like(v)
BH = B * H
for flat_bh, rt in hl.tile([BH, T], block_size=[1, C]):
b_idx = flat_bh.begin // H
h_idx = flat_bh.begin % H
beta_vals = beta[b_idx, rt, h_idx].to(torch.float32)
g_vals = g[b_idx, rt, h_idx].to(torch.float32)
k_chunk = k[b_idx, rt, h_idx, :].to(torch.float32)
v_chunk = v[b_idx, rt, h_idx, :].to(torch.float32)
A_chunk = A[b_idx, rt, h_idx, :].to(torch.float32)
k_scaled = k_chunk * (beta_vals * torch.exp(g_vals))[:, None]
v_scaled = v_chunk * beta_vals[:, None]
w_result = hl.dot(A_chunk, k_scaled, out_dtype=torch.float32)
u_result = hl.dot(A_chunk, v_scaled, out_dtype=torch.float32)
w_out[b_idx, rt, h_idx, :] = w_result.to(k.dtype)
u_out[b_idx, rt, h_idx, :] = u_result.to(v.dtype)
return w_out, u_out
return kernel
_KERNELS = {shape: _make_kernel(cfg) for shape, cfg in SHAPE_CONFIGS.items()}
def custom_kernel(data: input_t) -> output_t:
k, v, beta, A, g = data
B, T, H, K = k.shape
V = v.shape[-1]
kernel = _KERNELS[(B, T, H, K, V)]
return kernel(k, v, beta, A, g)
scrolls · 84 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 553122.
+ #!POPCORN leaderboard gated_deltanet_recompute_w_u+ #!POPCORN gpu B200_Nebius+from task import input_t, output_timport torchimport helionimport helion.language as hl- from pathlib import Path- # ACF: find best recompute_w_u ACF on B200- def _find_acf(pattern):- bp = Path("/opt/booster_pack")- if not bp.exists():- return None- for p in sorted(bp.glob(pattern)):- return str(p)- return None- _acf = _find_acf("recompute_w_u_fwd_*.acf")- _cfg = {"block_sizes": [], "num_warps": 4, "num_stages": 1}- if _acf:- _cfg["advanced_controls_file"] = _acf+ # Autotuned config from reference (B200, full effort)+ _TUNED = helion.Config(block_sizes=[], indexing=['tensor_descriptor', 'pointer', 'tensor_descriptor', 'tensor_descriptor', 'tensor_descriptor', 'tensor_descriptor', 'pointer'], l2_groupings=[16], load_eviction_policies=['', 'first', '', 'first', ''], loop_orders=[[0, 1]], maxnreg=32, num_sm_multiplier=16, num_stages=1, num_warps=32, pid_type='persistent_blocked', range_flattens=[None], range_multi_buffers=[False], range_num_stages=[3], range_unroll_factors=[4], range_warp_specializes=[None])- @helion.kernel(- static_shapes=True,- dot_precision="ieee",- config=helion.Config(**_cfg),- )- def project_kv(- k: torch.Tensor, # [B, T, H, K] -- key vectors- v: torch.Tensor, # [B, T, H, V] -- value vectors- beta: torch.Tensor, # [B, T, H] -- writing strength (scalar per position)- A: torch.Tensor, # [B, T, H, BT] -- WY transform matrix (from UT transform)- g: torch.Tensor, # [B, T, H] -- gating/decay values (negative, so exp(g) in (0,1])- ) -> tuple[torch.Tensor, torch.Tensor]:- B, T, H, K = k.shape- V = v.shape[-1]- # Specialize chunk size, K, V as compile-time constants- C = hl.specialize(A.shape[-1]) # 64- K = hl.specialize(K)- V = hl.specialize(V)+ SHAPE_CONFIGS: dict[tuple, helion.Config] = {+ # Test shapes+ (1, 64, 2, 64, 64): _TUNED,+ (2, 128, 4, 64, 64): _TUNED,+ (1, 256, 4, 64, 128): _TUNED,+ (1, 64, 1, 128, 128): _TUNED,+ (2, 128, 2, 100, 100): _TUNED,+ # Benchmark shapes+ (1, 64, 1, 64, 64): _TUNED,+ (2, 512, 3, 64, 64): _TUNED,+ (2, 1024, 3, 64, 64): _TUNED,+ (3, 1024, 4, 100, 100): _TUNED,+ (4, 1024, 4, 128, 128): _TUNED,+ (2, 1536, 4, 128, 128): _TUNED,+ (4, 2048, 8, 64, 64): _TUNED,+ }- w_out = torch.empty_like(k)- u_out = torch.empty_like(v)- # Flatten batch*head for parallelization- BH = B * H- for flat_bh, rt in hl.tile([BH, T], block_size=[1, C]):- b_idx = flat_bh.begin // H- h_idx = flat_bh.begin % H+ def _make_kernel(config: helion.Config):+ @helion.kernel(static_shapes=True, dot_precision="ieee", config=config)+ def kernel(+ k: torch.Tensor,+ v: torch.Tensor,+ beta: torch.Tensor,+ A: torch.Tensor,+ g: torch.Tensor,+ ) -> tuple[torch.Tensor, torch.Tensor]:+ B, T, H, K = k.shape+ V = v.shape[-1]+ C = hl.specialize(A.shape[-1])+ K = hl.specialize(K)+ V = hl.specialize(V)- # Load A matrix [C, C], scalars, and vectors for this chunk- a_chunk = A[b_idx, rt, h_idx, :].to(torch.float32)- beta_chunk = beta[b_idx, rt, h_idx].to(torch.float32)- g_chunk = g[b_idx, rt, h_idx].to(torch.float32)- k_chunk = k[b_idx, rt, h_idx, :].to(torch.float32)- v_chunk = v[b_idx, rt, h_idx, :].to(torch.float32)+ w_out = torch.empty_like(k)+ u_out = torch.empty_like(v)- # Scale: v * beta and k * beta * exp(g)- v_scaled = v_chunk * beta_chunk[:, None]- k_scaled = k_chunk * (beta_chunk * torch.exp(g_chunk))[:, None]+ BH = B * H+ for flat_bh, rt in hl.tile([BH, T], block_size=[1, C]):+ b_idx = flat_bh.begin // H+ h_idx = flat_bh.begin % H- # Two matmuls: u = A @ v_scaled, w = A @ k_scaled- u_out[b_idx, rt, h_idx, :] = torch.matmul(a_chunk, v_scaled).to(v.dtype)- w_out[b_idx, rt, h_idx, :] = torch.matmul(a_chunk, k_scaled).to(k.dtype)+ beta_vals = beta[b_idx, rt, h_idx].to(torch.float32)+ g_vals = g[b_idx, rt, h_idx].to(torch.float32)+ k_chunk = k[b_idx, rt, h_idx, :].to(torch.float32)+ v_chunk = v[b_idx, rt, h_idx, :].to(torch.float32)+ A_chunk = A[b_idx, rt, h_idx, :].to(torch.float32)- return w_out, u_out+ k_scaled = k_chunk * (beta_vals * torch.exp(g_vals))[:, None]+ v_scaled = v_chunk * beta_vals[:, None]+ w_result = hl.dot(A_chunk, k_scaled, out_dtype=torch.float32)+ u_result = hl.dot(A_chunk, v_scaled, out_dtype=torch.float32)+ w_out[b_idx, rt, h_idx, :] = w_result.to(k.dtype)+ u_out[b_idx, rt, h_idx, :] = u_result.to(v.dtype)++ return w_out, u_out++ return kernel+++ _KERNELS = {shape: _make_kernel(cfg) for shape, cfg in SHAPE_CONFIGS.items()}++def custom_kernel(data: input_t) -> output_t:k, v, beta, A, g = data- return project_kv(k, v, beta, A, g)+ B, T, H, K = k.shape+ V = v.shape[-1]+ kernel = _KERNELS[(B, T, H, K, V)]+ return kernel(k, v, beta, A, g)
scrolls · 135 diff lines total
Best evidence level for this revision: reported
JSON