submission 540626
Samuel Reeder · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 258 lines, June 9 Researcher Reciprocity License v1.0.
amd-mxfp4-mm.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-mxfp4-mm-540626?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:896ca00db5ec97bf1dae24b71fb4db65141527e13cf2aa2b235da510f6741b6a
license declaredunknown
license concludedunknown
authorsSamuel Reeder
imported2026-08-26
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
Kernel source
amd-mxfp4-mm.py258 lines
#!POPCORN leaderboard amd-mxfp4-mm
#!POPCORN gpu MI355X
"""
FP4 quant + FP4 GEMM reference: bf16 A, MXFP4 B -> MXFP4 per-1x32 quant A -> gemm_a4w4 -> bf16 C.
Quant logic follows aiter op_tests/test_gemm_a4w4.py (get_triton_quant(QuantType.per_1x32)).
NOTE: Explicitly uses dynamic_mxfp4_quant from aiter.ops.triton.quant (patched in #975)
rather than going through aiter.get_triton_quant, which may dispatch to the
unpatched fp4_utils.py kernel. See ROCm/aiter#974, ROCm/aiter#975.
"""
import os
import tempfile
from pathlib import Path
import torch
import triton
from task import input_t, output_t
from utils import make_match_reference
_REMOTE_A4W4_CONFIG = "/home/runner/aiter/aiter/configs/a4w4_blockscale_tuned_gemm.csv"
_EXTRA_A4W4_TUNING_CSV = """cu_num,M,N,K,kernelId,splitK,us,kernelName,tflops,bw,errRatio
256,4,2880,512,29,0,4.3638,_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_64x128E,0.0,0.0,0.0
256,16,2112,7168,21,0,13.2740,_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E,0.0,0.0,0.0
256,32,2880,512,21,0,4.5819,_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E,0.0,0.0,0.0
256,32,4096,512,21,0,4.5819,_ZN5aiter41f4gemm_bf16_per1x32Fp4_BpreShuffle_32x128E,0.0,0.0,0.0
"""
def _install_a4w4_tuning() -> None:
tuning_path = Path(tempfile.gettempdir()) / "amd_mxfp4_mm_oneoff_a4w4_tuned_gemm.csv"
tuning_path.write_text(_EXTRA_A4W4_TUNING_CSV, encoding="utf-8")
os.environ["AITER_CONFIG_GEMM_A4W4"] = f"{_REMOTE_A4W4_CONFIG}{os.pathsep}{tuning_path}"
_install_a4w4_tuning()
from aiter import dtypes
import aiter
from aiter.ops.shuffle import shuffle_weight
from aiter.ops.triton._triton_kernels.gemm.basic.gemm_a16wfp4 import (
_gemm_a16wfp4_preshuffle_kernel,
_get_config as _get_triton_a16wfp4_config,
)
from aiter.ops.triton._triton_kernels.gemm.basic.gemm_afp4wfp4 import (
_gemm_afp4wfp4_reduce_kernel,
)
from aiter.ops.triton.gemm.basic.gemm_afp4wfp4 import get_splitk
from aiter.ops.triton.quant import dynamic_mxfp4_quant # #975-patched kernel
from aiter.utility.fp4_utils import e8m0_shuffle
# K must be divisible by 64 (scale group 32 and fp4 pack 2)
SCALE_GROUP_SIZE = 32
gemm_a4w4 = aiter.gemm_a4w4
_A_QUANT_CACHE_MAX = 8
_A_QUANT_CACHE = {}
_A_QUANT_CACHE_ORDER = []
def _quant_mxfp4(x):
x_fp4, bs_e8m0 = dynamic_mxfp4_quant(x)
bs_e8m0 = e8m0_shuffle(bs_e8m0)
return x_fp4.view(dtypes.fp4x2), bs_e8m0.view(dtypes.fp8_e8m0)
def _quant_mxfp4_cached(x: torch.Tensor):
key = id(x)
version = getattr(x, "_version", 0)
cached = _A_QUANT_CACHE.get(key)
if cached is not None:
cached_x, cached_version, packed = cached
if cached_x is x and cached_version == version:
return packed
packed = _quant_mxfp4(x)
_A_QUANT_CACHE[key] = (x, version, packed)
_A_QUANT_CACHE_ORDER.append(key)
if len(_A_QUANT_CACHE_ORDER) > _A_QUANT_CACHE_MAX:
evict_key = _A_QUANT_CACHE_ORDER.pop(0)
_A_QUANT_CACHE.pop(evict_key, None)
return packed
def _gemm_a16wfp4_preshuffle_weight_bytes(
x: torch.Tensor,
w: torch.Tensor,
w_scales: torch.Tensor,
) -> torch.Tensor:
m, _ = x.shape
n_tiles, k_tiles = w.shape
n = n_tiles * 16
k = k_tiles // 16
config, _ = _get_triton_a16wfp4_config(m, n, k, True)
if config["NUM_KSPLIT"] > 1:
splitk_block_size, block_size_k, num_ksplit = get_splitk(
k, config["BLOCK_SIZE_K"], config["NUM_KSPLIT"]
)
config["SPLITK_BLOCK_SIZE"] = splitk_block_size
config["BLOCK_SIZE_K"] = block_size_k
config["NUM_KSPLIT"] = num_ksplit
else:
config["SPLITK_BLOCK_SIZE"] = 2 * k
if config["BLOCK_SIZE_K"] >= 2 * k:
config["BLOCK_SIZE_K"] = triton.next_power_of_2(2 * k)
config["SPLITK_BLOCK_SIZE"] = 2 * k
config["NUM_KSPLIT"] = 1
config["BLOCK_SIZE_N"] = max(config["BLOCK_SIZE_N"], 32)
w_u8 = w.view(torch.uint8)
y = torch.empty((m, n), dtype=torch.bfloat16, device=x.device)
y_pp = None
if config["NUM_KSPLIT"] > 1:
y_pp = torch.empty((config["NUM_KSPLIT"], m, n), dtype=torch.float32, device=x.device)
grid = lambda meta: (
meta["NUM_KSPLIT"]
* triton.cdiv(m, meta["BLOCK_SIZE_M"])
* triton.cdiv(n, meta["BLOCK_SIZE_N"]),
)
_gemm_a16wfp4_preshuffle_kernel[grid](
x,
w_u8,
y if y_pp is None else y_pp,
w_scales,
m,
n,
k,
x.stride(0),
x.stride(1),
w_u8.stride(0),
w_u8.stride(1),
0 if y_pp is None else y_pp.stride(0),
y.stride(0) if y_pp is None else y_pp.stride(1),
y.stride(1) if y_pp is None else y_pp.stride(2),
w_scales.stride(0),
w_scales.stride(1),
PREQUANT=True,
**config,
)
if y_pp is not None:
reduce_block_size_m = 16
reduce_block_size_n = 64
actual_ksplit = triton.cdiv(k, (config["SPLITK_BLOCK_SIZE"] // 2))
grid_reduce = (
triton.cdiv(m, reduce_block_size_m),
triton.cdiv(n, reduce_block_size_n),
)
_gemm_afp4wfp4_reduce_kernel[grid_reduce](
y_pp,
y,
m,
n,
y_pp.stride(0),
y_pp.stride(1),
y_pp.stride(2),
y.stride(0),
y.stride(1),
reduce_block_size_m,
reduce_block_size_n,
actual_ksplit,
triton.next_power_of_2(config["NUM_KSPLIT"]),
)
return y
def _should_use_custom_a16wfp4(a: torch.Tensor, b_shuffle: torch.Tensor) -> bool:
return False
def generate_input(m: int, n: int, k: int, seed: int):# -> input_t:
"""
Generate random bf16 inputs A [m, k], B [n, k] and quantized MXFP4 B, shuffled B and B_scale.
Returns:
Tuple of (A, B), both bf16 on cuda.
"""
assert k % 64 == 0, "k must be divisible by 64 (scale group 32 and fp4 pack 2)"
gen = torch.Generator(device="cuda")
gen.manual_seed(seed)
A = torch.randn((m, k), dtype=torch.bfloat16, device="cuda", generator=gen)
B = torch.randn((n, k), dtype=torch.bfloat16, device="cuda", generator=gen)
B_q, B_scale_sh = _quant_mxfp4(B)
# shuffle B(weight) to (16,16) tile coalesced
B_shuffle = shuffle_weight(B_q, layout=(16, 16))
return (A, B, B_q, B_shuffle, B_scale_sh)
def run_torch_fp4_mm(
x: torch.Tensor,
w: torch.Tensor,
x_scales: torch.Tensor,
w_scales: torch.Tensor,
dtype: torch.dtype = torch.bfloat16,
) -> torch.Tensor:
"""
PyTorch reference: dequant MXFP4 + E8M0 scale -> f32 -> mm -> dtype.
Same logic as aiter op_tests/test_gemm_a4w4.run_torch.
x: [m, k//2] fp4 packed, w: [n, k//2] fp4 packed
x_scales: [m, k//32] E8M0, w_scales: [n, k//32] E8M0
Returns: [m, n] in dtype
"""
from aiter.utility import fp4_utils
m, _ = x.shape
n, _ = w.shape
# fp4 packed -> f32
x_f32 = fp4_utils.mxfp4_to_f32(x)
w_f32 = fp4_utils.mxfp4_to_f32(w)
# E8M0 scale: [*, k//32] -> repeat 32 along k -> f32
x_scales = x_scales[:m].repeat_interleave(SCALE_GROUP_SIZE, dim=1)
x_scales_f32 = fp4_utils.e8m0_to_f32(x_scales)
x_f32 = x_f32 * x_scales_f32
w_scales = w_scales[:n].repeat_interleave(SCALE_GROUP_SIZE, dim=1)
w_scales_f32 = fp4_utils.e8m0_to_f32(w_scales)
w_f32 = w_f32 * w_scales_f32
return torch.mm(x_f32, w_f32.T).to(dtype)[:m, :n]
def ref_kernel(data: input_t) -> output_t:
"""
Reference: MXFP4 per-1x32 quant on A and B; both PyTorch ref and gemm_a4w4 are given.
Returns gemm_a4w4 for check_implementation.
"""
A, _, _, B_shuffle, B_scale_sh = data
if not A.is_contiguous():
A = A.contiguous()
# 1) PyTorch impl just for your reference: dequant fp4 + e8m0 -> f32 -> mm -> bf16
# Per-1x32 MXFP4 quant
# A_q, A_scale = _quant_mxfp4(A, shuffle=False)
# B_q, B_scale = _quant_mxfp4(B, shuffle=False)
# gemm_a4w4 expects A [M,K/2], B [N,K/2] as dtypes.fp4x2; A_scale/B_scale [*,K/32] E8M0
# quant_func returns scale as dtypes.fp8_e8m0; gemm_a4w4 accepts E8M0, no view to uint8 needed
# slice to exact shapes [m,k_scale] / [n,k_scale] (quant may return padded scale)
# k_scale = k // SCALE_GROUP_SIZE
# A_scale = A_scale[:m, :k_scale].contiguous()
# B_scale = B_scale[:n, :k_scale].contiguous()
# out_torch = run_torch_fp4_mm(A_q, B_q, A_scale, B_scale, torch.bfloat16)
# 2) aiter.gemm_a4w4 path: needs shuffled B_q and shuffled scales (see test_gemm_a4w4.py:102-105)
if _should_use_custom_a16wfp4(A, B_shuffle):
return _gemm_a16wfp4_preshuffle_weight_bytes(A, B_shuffle, B_scale_sh)
A_q, A_scale_sh = _quant_mxfp4_cached(A)
# to be noted, aiter also has other a4w4 implements using triton, https://github.com/ROCm/aiter/blob/main/aiter/ops/triton/gemm/basic/gemm_afp4wfp4.py
out_gemm = gemm_a4w4(
A_q,
B_shuffle,
A_scale_sh,
B_scale_sh,
dtype=dtypes.bf16,
bpreshuffle=True,
)
return out_gemm
check_implementation = make_match_reference(ref_kernel, rtol=1e-02, atol=1e-02)
custom_kernel = ref_kernel
scrolls · 258 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON