submission 643450
SwordHoly · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 103 lines, June 9 Researcher Reciprocity License v1.0.
v10_tuned_csv.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-mxfp4-mm-643450?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:3733ec8f792e19079fa571ef79d10380e52efc4257f8c1f87352501ada74cb35
license declaredunknown
license concludedunknown
authorsSwordHoly
imported2026-08-26
Kernel source
v10_tuned_csv.py103 lines
"""
GEMM v10: Custom tuned CSV with entries for ALL benchmark shapes.
The existing a4w4_blockscale_tuned_gemm.csv misses N=2880 (small M) and N=4096/K=512.
From v8 diagnostic, available BpreShuffle kernels:
32x128, 32x256, 64x128, 64x256, 64x640, 64x768,
96x128, 96x256, 96x384, 96x512, 96x640,
128x128, 128x256, 128x384, 128x512,
160x128, 160x256, 160x384,
192x128, 192x256,
224x128, 224x256,
256x128, 256x256
Existing CSV matches for our shapes:
M=8,N=2112,K=7168: 32x128, 12.8µs
M=64,N=7168,K=2048: 32x128, 6.8µs
M=256,N=3072,K=1536: 32x128, 6.2µs
Missing shapes: M=4/32 with N=2880/K=512, M=32 with N=4096/K=512, M=16 with N=2112/K=7168
Strategy: Merge existing tuned CSV with custom entries for missing shapes.
Try different kernels for small M: 32x128 vs 64x128.
"""
from task import input_t, output_t
def custom_kernel(data: input_t) -> output_t:
import os
import sys
import torch
import aiter
from aiter import dtypes
from aiter.ops.triton.quant import dynamic_mxfp4_quant
from aiter.utility.fp4_utils import e8m0_shuffle
if not hasattr(custom_kernel, "_setup_done"):
custom_kernel._setup_done = True
# Find and extend the existing tuned CSV
try:
aiter_root = os.path.dirname(os.path.dirname(aiter.__file__))
base_csv = os.path.join(aiter_root, "aiter", "configs", "a4w4_blockscale_tuned_gemm.csv")
custom_csv = "/tmp/a4w4_tuned_v10.csv"
if not os.path.exists(custom_csv):
# Read existing tuned CSV
with open(base_csv) as f:
existing_lines = f.readlines()
# Add custom entries for missing shapes
# Format: cu_num,M,N,K,kernelId,splitK,us,kernelName,tflops,bw,errRatio
k32 = "_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E"
k64 = "_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_64x128E"
custom_entries = [
# M=4, N=2880, K=512 - very small M, use 64x128 (better for small M)
f"256,4,2880,512,29,0,10.0,{k64},0,0,0",
# M=16, N=2112, K=7168 - small M, large K, splitK=2
f"256,16,2112,7168,29,2,10.0,{k64},0,0,0",
# M=32, N=4096, K=512 - medium M, small K
f"256,32,4096,512,29,0,10.0,{k64},0,0,0",
# M=32, N=2880, K=512 - medium M, small K
f"256,32,2880,512,29,0,10.0,{k64},0,0,0",
# M=256, N=2880, K=512 (test shape)
f"256,256,2880,512,29,0,10.0,{k64},0,0,0",
]
# Write merged CSV (custom entries first for priority)
with open(custom_csv, "w") as f:
f.write(existing_lines[0]) # header
for entry in custom_entries:
f.write(entry + "\n")
for line in existing_lines[1:]:
f.write(line)
print(f"[v10] Merged CSV: {len(existing_lines)} existing + {len(custom_entries)} custom entries", file=sys.stderr)
os.environ["AITER_CONFIG_GEMM_A4W4"] = custom_csv
print(f"[v10] Set AITER_CONFIG_GEMM_A4W4={custom_csv}", file=sys.stderr)
except Exception as e:
print(f"[v10] Setup error: {e}", file=sys.stderr)
def _quant_mxfp4(x, shuffle=True):
x_fp4, bs_e8m0 = dynamic_mxfp4_quant(x)
if shuffle:
bs_e8m0 = e8m0_shuffle(bs_e8m0)
return x_fp4.view(dtypes.fp4x2), bs_e8m0.view(dtypes.fp8_e8m0)
A, B, B_q, B_shuffle, B_scale_sh = data
A = A.contiguous()
A_q, A_scale_sh = _quant_mxfp4(A, shuffle=True)
out = aiter.gemm_a4w4(
A_q,
B_shuffle,
A_scale_sh,
B_scale_sh,
dtype=dtypes.bf16,
bpreshuffle=True,
)
return out
scrolls · 103 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON