submission 692250
.jonnss · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 100 lines, June 9 Researcher Reciprocity License v1.0.
Submission_v234.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-amd-moe-mxfp4-692250?include=source"interfacepython
Compatibility
measured onAMD Instinct MI355X
declared hardwareAMD Instinct MI355X
architecturesgfx950
dtypesbf16, fp32, fp8_e8m0, int32, mxfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:4688d8e98ada4453d402ebb0ade66c1114c038c6286349bb1208c1470e99c983
license declaredunknown
license concludedunknown
authors.jonnss
imported2026-08-15
Kernel source
Submission_v234.py100 lines
import functools
import importlib
import os
import torch
from task import input_t, output_t
os.environ["VLLM_MOE_CHUNK_SIZE"] = "512"
from aiter import ActivationType, QuantType
from aiter.fused_moe import fused_moe
_PATCH_DONE = False
def _install_patch() -> None:
global _PATCH_DONE
if _PATCH_DONE:
return
_PATCH_DONE = True
try:
fused_moe_module = importlib.import_module("aiter.fused_moe")
original = getattr(fused_moe_module, "get_2stage_cfgs", None)
if original is None:
return
def patched_get_2stage_cfgs(*args, **kwargs):
meta = original(*args, **kwargs)
token = kwargs.get("token", args[0] if len(args) > 0 else None)
model_dim = kwargs.get("model_dim", args[1] if len(args) > 1 else None)
inter_dim = kwargs.get("inter_dim", args[2] if len(args) > 2 else None)
expert = kwargs.get("expert", args[3] if len(args) > 3 else None)
topk = kwargs.get("topk", args[4] if len(args) > 4 else None)
if token == 512 and model_dim == 7168 and inter_dim == 512 and expert == 33 and topk == 9:
try:
meta.block_m = 128
except Exception:
pass
for attr in ("stage1", "stage2"):
fn = getattr(meta, attr, None)
if isinstance(fn, functools.partial):
keywords = dict(fn.keywords or {})
keywords["block_m"] = 128
setattr(meta, attr, functools.partial(fn.func, *(fn.args or ()), **keywords))
return meta
fused_moe_module.get_2stage_cfgs = patched_get_2stage_cfgs
fused_moe_globals = getattr(fused_moe, "__globals__", None)
if isinstance(fused_moe_globals, dict) and fused_moe_globals.get("get_2stage_cfgs") is original:
fused_moe_globals["get_2stage_cfgs"] = patched_get_2stage_cfgs
except Exception:
pass
@torch.inference_mode()
def custom_kernel(data: input_t) -> output_t:
(
hidden_states,
_gate_up_weight,
_down_weight,
_gate_up_weight_scale,
_down_weight_scale,
gate_up_weight_shuffled,
down_weight_shuffled,
gate_up_weight_scale_shuffled,
down_weight_scale_shuffled,
topk_weights,
topk_ids,
config,
) = data
_install_patch()
hidden_pad = int(config["d_hidden_pad"]) - int(config["d_hidden"])
intermediate_pad = int(config["d_expert_pad"]) - int(config["d_expert"])
return fused_moe(
hidden_states,
gate_up_weight_shuffled,
down_weight_shuffled,
topk_weights,
topk_ids,
expert_mask=None,
activation=ActivationType.Silu,
quant_type=QuantType.per_1x32,
doweight_stage1=False,
w1_scale=gate_up_weight_scale_shuffled,
w2_scale=down_weight_scale_shuffled,
a1_scale=None,
a2_scale=None,
hidden_pad=hidden_pad,
intermediate_pad=intermediate_pad,
)
scrolls · 100 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 646374.
+ import functools+ import importlibimport os+import torch+from task import input_t, output_t- os.environ["VLLM_MOE_WTYPE"] = "fp8"- os.environ["VLLM_QUANT_OVERRIDE"] = "0"+ os.environ["VLLM_MOE_CHUNK_SIZE"] = "512"from aiter import ActivationType, QuantTypefrom aiter.fused_moe import fused_moe- @torch.inference_mode()- def custom_kernel(data: input_t) -> output_t:- """- Submission template for DeepSeek-R1 MXFP4 MoE kernel.+ _PATCH_DONE = False- Input data tuple:- hidden_states: [M, d_hidden] bf16- gate_up_weight: [E, 2*d_expert_pad, d_hidden_pad//2] fp4x2 (raw)- down_weight: [E, d_hidden_pad, d_expert_pad//2] fp4x2 (raw)- gate_up_weight_scale: [E, 2*d_expert_pad, scale_K] e8m0 (raw)- down_weight_scale: [E, d_hidden_pad, scale_K] e8m0 (raw)- gate_up_weight_shuffled: [E, 2*d_expert_pad, d_hidden_pad//2] fp4x2 (shuffled)- down_weight_shuffled: [E, d_hidden_pad, d_expert_pad//2] fp4x2 (shuffled)- gate_up_weight_scale_shuffled:[padded, flat] e8m0 (shuffled)- down_weight_scale_shuffled: [padded, flat] e8m0 (shuffled)- topk_weights: [M, total_top_k] float32- topk_ids: [M, total_top_k] int32- config: dict- Returns:- output: [M, d_hidden] bf16- """+ def _install_patch() -> None:+ global _PATCH_DONE+ if _PATCH_DONE:+ return+ _PATCH_DONE = True++ try:+ fused_moe_module = importlib.import_module("aiter.fused_moe")+ original = getattr(fused_moe_module, "get_2stage_cfgs", None)+ if original is None:+ return++ def patched_get_2stage_cfgs(*args, **kwargs):+ meta = original(*args, **kwargs)++ token = kwargs.get("token", args[0] if len(args) > 0 else None)+ model_dim = kwargs.get("model_dim", args[1] if len(args) > 1 else None)+ inter_dim = kwargs.get("inter_dim", args[2] if len(args) > 2 else None)+ expert = kwargs.get("expert", args[3] if len(args) > 3 else None)+ topk = kwargs.get("topk", args[4] if len(args) > 4 else None)++ if token == 512 and model_dim == 7168 and inter_dim == 512 and expert == 33 and topk == 9:+ try:+ meta.block_m = 128+ except Exception:+ pass+ for attr in ("stage1", "stage2"):+ fn = getattr(meta, attr, None)+ if isinstance(fn, functools.partial):+ keywords = dict(fn.keywords or {})+ keywords["block_m"] = 128+ setattr(meta, attr, functools.partial(fn.func, *(fn.args or ()), **keywords))+ return meta++ fused_moe_module.get_2stage_cfgs = patched_get_2stage_cfgs++ fused_moe_globals = getattr(fused_moe, "__globals__", None)+ if isinstance(fused_moe_globals, dict) and fused_moe_globals.get("get_2stage_cfgs") is original:+ fused_moe_globals["get_2stage_cfgs"] = patched_get_2stage_cfgs+ except Exception:+ pass+++ @torch.inference_mode()+ def custom_kernel(data: input_t) -> output_t:(hidden_states,- gate_up_weight,- down_weight,- gate_up_weight_scale,- down_weight_scale,+ _gate_up_weight,+ _down_weight,+ _gate_up_weight_scale,+ _down_weight_scale,gate_up_weight_shuffled,down_weight_shuffled,gate_up_weight_scale_shuffled,⋯ 3 unchanged linesconfig,) = data- hidden_pad = config["d_hidden_pad"] - config["d_hidden"]- intermediate_pad = config["d_expert_pad"] - config["d_expert"]+ _install_patch()- output = fused_moe(+ hidden_pad = int(config["d_hidden_pad"]) - int(config["d_hidden"])+ intermediate_pad = int(config["d_expert_pad"]) - int(config["d_expert"])++ return fused_moe(hidden_states,gate_up_weight_shuffled,down_weight_shuffled,⋯ 10 unchanged lineshidden_pad=hidden_pad,intermediate_pad=intermediate_pad,)-- return output
scrolls · 119 diff lines total
Best evidence level for this revision: reported
JSON