submission 730252
Leandro Timberini · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 231 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-730252?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:df9bef591605fa6e9063e79dcadbe705a94f2acc0942658a1b7e1853f32ab7a0
license declaredunknown
license concludedunknown
authorsLeandro Timberini
imported2026-08-26
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
fp4
Submission template for DeepSeek-R1 MXFP4 MoE kernel.Kernel source
submission.py231 lines
import os
import torch
from typing import Dict
from task import input_t, output_t
from aiter import ActivationType, QuantType, dtypes
from aiter.fused_moe import fused_moe
try:
from aiter.fused_moe import fused_moe_1stage, moe_sorting, get_block_size_M
except Exception:
fused_moe_1stage = None
moe_sorting = None
get_block_size_M = None
_ENABLE_1STAGE_OVERRIDE = os.getenv("SUBMISSION_MOE_ENABLE_1STAGE_OVERRIDE", "0") == "1"
_MOE_1STAGE_BLOCK_M_OVERRIDE = int(os.getenv("SUBMISSION_MOE_1STAGE_BLOCK_M", "0"))
_MOE_1STAGE_DISPATCH_POLICY = int(
os.getenv("SUBMISSION_MOE_1STAGE_DISPATCH_POLICY", "0")
)
_MOE_1STAGE_USE_RAW_WEIGHTS = (
os.getenv("SUBMISSION_MOE_1STAGE_USE_RAW_WEIGHTS", "0") == "1"
)
_MOE_1STAGE_USE_RAW_SCALES = (
os.getenv("SUBMISSION_MOE_1STAGE_USE_RAW_SCALES", "0") == "1"
)
_MOE_1STAGE_DEBUG = os.getenv("SUBMISSION_MOE_1STAGE_DEBUG", "0") == "1"
_MOE_1STAGE_DEBUG_SEEN = set()
def _should_use_1stage_override(config: Dict) -> bool:
enabled = (
_ENABLE_1STAGE_OVERRIDE
and
fused_moe_1stage is not None
and moe_sorting is not None
and config["n_routed_experts"] + config["n_shared_experts"] == 33
and config["d_hidden_pad"] == config["d_hidden"]
and config["d_expert_pad"] == config["d_expert"]
)
if _MOE_1STAGE_DEBUG and not enabled:
key = (
config["n_routed_experts"] + config["n_shared_experts"],
config["d_hidden_pad"],
config["d_expert_pad"],
)
if key not in _MOE_1STAGE_DEBUG_SEEN:
_MOE_1STAGE_DEBUG_SEEN.add(key)
print(
"[submission.moe_1stage_unavailable] "
f"enabled={_ENABLE_1STAGE_OVERRIDE} "
f"has_1stage={fused_moe_1stage is not None} "
f"has_sorting={moe_sorting is not None} "
f"experts={config['n_routed_experts'] + config['n_shared_experts']} "
f"d_hidden_pad={config['d_hidden_pad']} d_hidden={config['d_hidden']} "
f"d_expert_pad={config['d_expert_pad']} d_expert={config['d_expert']}",
flush=True,
)
return enabled
def _run_1stage_override(
hidden_states: torch.Tensor,
gate_up_weight: torch.Tensor,
down_weight: torch.Tensor,
gate_up_weight_shuffled: torch.Tensor,
down_weight_shuffled: torch.Tensor,
gate_up_weight_scale: torch.Tensor,
down_weight_scale: torch.Tensor,
gate_up_weight_scale_shuffled: torch.Tensor,
down_weight_scale_shuffled: torch.Tensor,
topk_weights: torch.Tensor,
topk_ids: torch.Tensor,
config: Dict,
) -> torch.Tensor:
topk = topk_weights.shape[1]
num_experts = config["n_routed_experts"] + config["n_shared_experts"]
gate_weight = gate_up_weight if _MOE_1STAGE_USE_RAW_WEIGHTS else gate_up_weight_shuffled
down_weight = down_weight if _MOE_1STAGE_USE_RAW_WEIGHTS else down_weight_shuffled
gate_scale = (
gate_up_weight_scale if _MOE_1STAGE_USE_RAW_SCALES else gate_up_weight_scale_shuffled
)
down_scale = (
down_weight_scale if _MOE_1STAGE_USE_RAW_SCALES else down_weight_scale_shuffled
)
model_dim = down_weight.shape[1]
inter_dim = down_weight.shape[2] * 2
if _MOE_1STAGE_BLOCK_M_OVERRIDE:
block_size_m = _MOE_1STAGE_BLOCK_M_OVERRIDE
elif get_block_size_M is not None:
block_size_m = int(
get_block_size_M(hidden_states.shape[0], topk, num_experts, inter_dim)
)
else:
block_size_m = 32
sorted_ids, sorted_weights, sorted_expert_ids, num_valid_ids, moe_buf = moe_sorting(
topk_ids,
topk_weights,
num_experts,
model_dim,
hidden_states.dtype,
block_size=block_size_m,
expert_mask=None,
num_local_tokens=None,
dispatch_policy=_MOE_1STAGE_DISPATCH_POLICY,
)
out = fused_moe_1stage(
hidden_states,
gate_weight,
down_weight,
topk,
sorted_ids,
sorted_weights,
sorted_expert_ids,
num_valid_ids,
moe_buf,
isG1U1=True,
block_size_M=block_size_m,
activation=ActivationType.Silu,
quant_type=QuantType.per_1x32,
q_dtype_a=dtypes.fp4x2,
q_dtype_w=gate_weight.dtype,
w1_scale=gate_scale,
w2_scale=down_scale,
a1_scale=None,
a2_scale=None,
num_local_tokens=None,
M=hidden_states.shape[0],
device=hidden_states.device,
doweight_stage1=False,
)
return out[:, : config["d_hidden"]]
def custom_kernel(data: input_t) -> output_t:
"""
Submission template for DeepSeek-R1 MXFP4 MoE kernel.
Input data tuple:
hidden_states: [M, d_hidden] bf16
gate_up_weight: [E, 2*d_expert_pad, d_hidden_pad//2] fp4x2 (raw)
down_weight: [E, d_hidden_pad, d_expert_pad//2] fp4x2 (raw)
gate_up_weight_scale: [E, 2*d_expert_pad, scale_K] e8m0 (raw)
down_weight_scale: [E, d_hidden_pad, scale_K] e8m0 (raw)
gate_up_weight_shuffled: [E, 2*d_expert_pad, d_hidden_pad//2] fp4x2 (shuffled)
down_weight_shuffled: [E, d_hidden_pad, d_expert_pad//2] fp4x2 (shuffled)
gate_up_weight_scale_shuffled:[padded, flat] e8m0 (shuffled)
down_weight_scale_shuffled: [padded, flat] e8m0 (shuffled)
topk_weights: [M, total_top_k] float32
topk_ids: [M, total_top_k] int32
config: dict
Returns:
output: [M, d_hidden] bf16
"""
(
hidden_states,
gate_up_weight,
down_weight,
gate_up_weight_scale,
down_weight_scale,
gate_up_weight_shuffled,
down_weight_shuffled,
gate_up_weight_scale_shuffled,
down_weight_scale_shuffled,
topk_weights,
topk_ids,
config,
) = data
hidden_pad = config["d_hidden_pad"] - config["d_hidden"]
intermediate_pad = config["d_expert_pad"] - config["d_expert"]
if _should_use_1stage_override(config):
try:
return _run_1stage_override(
hidden_states,
gate_up_weight,
down_weight,
gate_up_weight_shuffled,
down_weight_shuffled,
gate_up_weight_scale,
down_weight_scale,
gate_up_weight_scale_shuffled,
down_weight_scale_shuffled,
topk_weights,
topk_ids,
config,
)
except Exception as exc:
if _MOE_1STAGE_DEBUG:
print(
f"[submission.moe_1stage_fallback] {type(exc).__name__}: {exc}",
flush=True,
)
pass
# Improved block_size_M heuristic (must be multiple of 32 for aiter quant sort)
M = hidden_states.shape[0]
if M <= 128:
block_m = 32
elif M <= 512:
block_m = 64
else:
block_m = 128
# Force aiter to use heuristics instead of searching for tune config if not present
os.environ["AITER_BYPASS_TUNE_CONFIG"] = "1"
output = fused_moe(
hidden_states,
gate_up_weight_shuffled,
down_weight_shuffled,
topk_weights,
topk_ids,
expert_mask=None,
activation=ActivationType.Silu,
quant_type=QuantType.per_1x32,
doweight_stage1=False,
w1_scale=gate_up_weight_scale_shuffled,
w2_scale=down_weight_scale_shuffled,
a1_scale=None,
a2_scale=None,
block_size_M=block_m,
hidden_pad=hidden_pad,
intermediate_pad=intermediate_pad,
)
return output
scrolls · 231 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON