Skip to content
KernelIndex
Search⌘K

submission 471047

oofbaroomf · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 3621 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-group-gemm-471047?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVFP4 group GEMMsuite of 4 cases
NVIDIA B200
38.5µs
#42 of 145
2026-02-05

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:3e936c9f6c75ecfba321a969da673aebef9ae0072f1248427633c75ad6f37533
license declaredunknown
license concludedunknown
authorsoofbaroomf
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

fused-epilogueself.epi_tile = sm100_utils.compute_epilogue_tile_shape(
mbarrierself.epilog_sync_barrier = pipeline.NamedBarrier(
persistent-kerneltile_sched_params: utils.PersistentTileSchedulerParams,
shared-memoryself.smem_capacity = utils.get_smem_capacity_in_bytes("sm_100")
tcgen05tcgen05.CtaGroup.TWO if self.use_2cta_instrs else tcgen05.CtaGroup.ONE
warp-specializationab_pipeline_producer_group = pipeline.CooperativeGroup(pipeline.Agent.Thread)

Kernel source

submission.py3621 lines
import os
import sys
import importlib.util
from inspect import isclass
from typing import Any, List, Tuple, Type, Union

import torch
from task import input_t, output_t


_ROOT = os.path.dirname(__file__)
_LOCAL_CUTE = os.path.abspath(
    os.path.join(
        _ROOT,
        "..",
        "..",
        "..",
        "..",
        "third_party",
        "cutlass",
        "python",
        "CuTeDSL",
    )
)
if os.path.isdir(_LOCAL_CUTE) and _LOCAL_CUTE not in sys.path:
    sys.path.insert(0, _LOCAL_CUTE)

_CACHE_DIR = os.path.abspath(
    os.path.join(_ROOT, "..", "..", "..", "..", "kernel_context", "cute_cache")
)
try:
    os.makedirs(_CACHE_DIR, exist_ok=True)
except Exception:
    _CACHE_DIR = os.path.join("/tmp", "cute_cache")
    os.makedirs(_CACHE_DIR, exist_ok=True)
os.environ.setdefault("CUTE_DSL_CACHE_DIR", _CACHE_DIR)
os.environ.setdefault("CUTE_DSL_JIT_CACHE", _CACHE_DIR)
os.environ.setdefault("CUTE_DSL_KEEP_PTX", "1")
os.environ.setdefault("CUTE_DSL_KEEP_CUBIN", "1")

import cutlass
import cutlass.cute as cute
import cutlass._mlir.dialects.cute as _cute_ir
from cutlass._mlir import ir as _ir
from cutlass._mlir.dialects import llvm as _llvm
from cutlass.cute.nvgpu import cpasync, tcgen05
import cutlass.torch as cutlass_torch
import cutlass.utils as utils
import cutlass.pipeline as pipeline
from cutlass.pipeline import pipeline_init_arrive, pipeline_init_wait
import cutlass.utils.blackwell_helpers as sm100_utils
import cutlass.utils.blockscaled_layout as blockscaled_utils
from cutlass.cute.runtime import from_dlpack
from cutlass.cutlass_dsl import dsl_user_op


_kernel_cache: dict[tuple, Any] = {}
_fake_st: Any | None = None
_stage8_mod: Any | None = None
_stage8_checked = False
_enable_stage8 = os.environ.get("NVFP4_ENABLE_STAGE8", "") == "1"
_speedk_mod: Any | None = None
_speedk_checked = False
_enable_speedk_ext = os.environ.get("NVFP4_ENABLE_SPEEDK_EXT", "") != "0"
_speedk_diag_done = False
_speedk_ext_mod: Any | None = None
_speedk_ext_fail = False
_speedk_inc_cache: list[str] | None = None
_speedk_cutlass_root: str | None = None
_speedk_pad_cache: dict[tuple[int, int, torch.dtype, int], torch.Tensor] = {}
_speedk_c_pad_cache: dict[tuple[int, int, torch.dtype, int], torch.Tensor] = {}
_speedk_disable_ext = os.environ.get("NVFP4_SPEEDK_EXT_DISABLE", "") == "1"
_speedk_diag = os.environ.get("NVFP4_SPEEDK_DIAG", "") == "1"


def _is_sm100_device(t: torch.Tensor) -> bool:
    if t.device.type != "cuda":
        return False
    try:
        major, _minor = torch.cuda.get_device_capability(t.device)
    except Exception:
        return False
    return major >= 10


def _try_stage8_ext(
    abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
    sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],
    problem_sizes: list[tuple[int, int, int, int]],
):
    global _stage8_mod, _stage8_checked
    if _stage8_checked and _stage8_mod is None:
        return None
    if _stage8_mod is None:
        _stage8_checked = True
        try:
            stage8_path = os.path.join(_ROOT, "submission_b200_stage8.py")
            if not os.path.isfile(stage8_path):
                return None
            spec = importlib.util.spec_from_file_location(
                "nvfp4_group_gemm_stage8_ext", stage8_path
            )
            if spec is None or spec.loader is None:
                return None
            mod = importlib.util.module_from_spec(spec)
            spec.loader.exec_module(mod)
            _stage8_mod = mod
        except Exception:
            _stage8_mod = None
            return None
    try:
        return _stage8_mod._try_ext(abc_tensors, sfs_reordered, problem_sizes)
    except Exception:
        return None


def _try_speedk_ext(
    abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
    sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],
    problem_sizes: list[tuple[int, int, int, int]],
):
    global _speedk_ext_mod, _speedk_checked, _speedk_ext_fail, _speedk_diag_done

    def _diag(msg: str):
        global _speedk_diag_done
        if not _speedk_diag or _speedk_diag_done:
            return
        _speedk_diag_done = True
        print(msg, file=sys.stderr, flush=True)

    if _speedk_disable_ext or (not _enable_speedk_ext) or (not abc_tensors):
        return None

    # This kernel is SM100-only. Avoid any JIT/compile work on other GPUs.
    if not _is_sm100_device(abc_tensors[0][0]):
        return None

    if _speedk_ext_fail:
        return None

    if _speedk_ext_mod is None:
        _speedk_checked = True
        _speedk_ext_mod = _speedk_load_ext(_diag)
        if _speedk_ext_mod is None:
            _speedk_ext_fail = True
            _diag("speedk ext: build fail")
            return None

    out = _speedk_try_ext(_speedk_ext_mod, abc_tensors, sfs_reordered, problem_sizes)
    if out is None:
        _diag("speedk ext: none")
        return None
    _diag("speedk ext: ok")
    return True


_SPEEDK_CPP_HEX = "0a2020202023696e636c756465203c746f7263682f657874656e73696f6e2e683e0a2020202023696e636c756465203c766563746f723e0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f31736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620534642293b0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f32736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620534642293b0a20202020"
_SPEEDK_CU_HEX = "0a2020202023696e636c756465203c746f7263682f657874656e73696f6e2e683e0a2020202023696e636c756465203c4154656e2f637564612f43554441436f6e746578742e683e0a2020202023696e636c756465203c6331302f637564612f4355444153747265616d2e683e0a2020202023696e636c756465203c6331302f637564612f4355444147756172642e683e0a2020202023696e636c756465203c766563746f723e0a2020202023696e636c756465203c7374646578636570743e0a2020202023696e636c756465203c747970655f7472616974733e0a0a20202020236966646566205f5f435544415f4e4f5f48414c465f4f50455241544f52535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c465f4f50455241544f52535f5f0a2020202023656e6469660a20202020236966646566205f5f435544415f4e4f5f48414c465f434f4e56455253494f4e535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c465f434f4e56455253494f4e535f5f0a2020202023656e6469660a20202020236966646566205f5f435544415f4e4f5f48414c46325f4f50455241544f52535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c46325f4f50455241544f52535f5f0a2020202023656e6469660a0a2020202023696e636c756465203c637564615f72756e74696d652e683e0a2020202023696620646566696e6564285f5f435544415f415243485f5f292026262021646566696e6564284355544c4153535f415243485f4d4d415f534d313030415f454e41424c4544290a2020202023646566696e65204355544c4153535f415243485f4d4d415f534d313030415f454e41424c454420310a2020202023656e6469660a2020202023696e636c75646520226375746c6173732f6375746c6173732e68220a2020202023696e636c7564652022637574652f74656e736f722e687070220a2020202023696e636c75646520226375746c6173732f74656e736f725f7265662e68220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f636f6c6c6563746976652f64656661756c745f6570696c6f6775652e687070220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f7468726561642f6c696e6561725f636f6d62696e6174696f6e2e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f64697370617463685f706f6c6963792e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f67726f75705f61727261795f70726f626c656d5f73686170652e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f636f6c6c6563746976652f636f6c6c6563746976655f6275696c6465722e687070220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f636f6c6c6563746976652f636f6c6c6563746976655f6275696c6465722e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f6465766963652f67656d6d5f756e6976657273616c5f616461707465722e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f6b65726e656c2f67656d6d5f756e6976657273616c2e687070220a2020202023696e636c75646520226375746c6173732f7574696c2f7061636b65645f7374726964652e687070220a2020202023696e636c75646520226375746c6173732f6b65726e656c5f68617264776172655f696e666f2e68220a2020202023696e636c75646520226375746c6173732f7574696c2f6465766963655f6d656d6f72792e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f6b65726e656c2f74696c655f7363686564756c65725f706172616d732e68220a0a2020202023646566696e65204355544c4153535f434845434b287374617475732920202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020646f207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a20202020202020206375746c6173733a3a537461747573205f737461747573203d2028737461747573293b202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020696620285f73746174757320213d206375746c6173733a3a5374617475733a3a6b5375636365737329207b20202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020202020207468726f77207374643a3a72756e74696d655f6572726f7228224355544c415353206572726f7222293b202020202020202020202020202020202020202020202020202020202020202020205c0a20202020202020207d20202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020207d207768696c65202830290a0a2020202023646566696e6520435544415f434845434b2865787072292020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020646f207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020637564614572726f725f74205f657272203d202865787072293b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020696620285f65727220213d20637564615375636365737329207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020202020207468726f77207374643a3a72756e74696d655f6572726f7228637564614765744572726f72537472696e67285f65727229293b202020202020202020202020202020202020202020202020205c0a20202020202020207d20202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020207d207768696c65202830290a0a202020207573696e672050726f626c656d5368617065203d206375746c6173733a3a67656d6d3a3a47726f757050726f626c656d53686170653c637574653a3a53686170653c696e742c696e742c696e743e3e3b0a202020207573696e6720456c656d656e74496e707574203d206375746c6173733a3a666c6f61745f65326d315f743b0a202020207573696e6720456c656d656e745346202020203d206375746c6173733a3a666c6f61745f7565346d335f743b0a202020207573696e6720456c656d656e744320202020203d206375746c6173733a3a68616c665f743b0a0a202020207573696e6720456c656d656e7441203d206375746c6173733a3a6e765f666c6f6174345f743c456c656d656e74496e7075743e3b0a202020207573696e67204c61796f75744120203d206375746c6173733a3a6c61796f75743a3a526f774d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744120203d2033323b0a0a202020207573696e6720456c656d656e7442203d206375746c6173733a3a6e765f666c6f6174345f743c456c656d656e74496e7075743e3b0a202020207573696e67204c61796f75744220203d206375746c6173733a3a6c61796f75743a3a436f6c756d6e4d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744220203d2033323b0a0a202020207573696e6720456c656d656e7444203d20456c656d656e74433b0a202020207573696e67204c61796f75744320203d206375746c6173733a3a6c61796f75743a3a436f6c756d6e4d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744320203d20323536202f206375746c6173733a3a73697a656f665f626974733c456c656d656e74433e3a3a76616c75653b0a20202020636f6e73746578707220696e7420416c69676e6d656e744420203d20323536202f206375746c6173733a3a73697a656f665f626974733c456c656d656e74443e3a3a76616c75653b0a202020207573696e6720456c656d656e74416363756d756c61746f7220203d20666c6f61743b0a0a202020207573696e672041726368546167203d206375746c6173733a3a617263683a3a536d3130303b0a202020207573696e67204f70657261746f72436c617373203d206375746c6173733a3a617263683a3a4f70436c617373426c6f636b5363616c656454656e736f724f703b0a202020207573696e67205374616765436f756e745479706531536d203d206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a5374616765436f756e743c343e3b0a202020207573696e67205374616765436f756e745479706532536d203d206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a5374616765436f756e743c353e3b0a202020207573696e6720436c75737465725368617065203d20637574653a3a53686170653c696e7433325f742c696e7433325f742c637574653a3a5f313e3b0a0a20202020737472756374204d4d4131534d436f6e666967207b0a2020202020207573696e67204d6d6154696c65536861706520202020203d20637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36342c637574653a3a5f3235363e3b0a2020202020207573696e67204b65726e656c5363686564756c652020203d206375746c6173733a3a67656d6d3a3a4b65726e656c5074724172726179546d61576172705370656369616c697a656431536d4e766634536d3130303b0a2020202020207573696e67204570696c6f6775655363686564756c65203d206375746c6173733a3a6570696c6f6775653a3a5074724172726179546d61576172705370656369616c697a656431536d3b0a202020207d3b0a0a20202020737472756374204d4d4132534d436f6e666967207b0a2020202020207573696e67204d6d6154696c65536861706520202020203d20637574653a3a53686170653c637574653a3a5f3235362c637574653a3a5f36342c637574653a3a5f3235363e3b0a2020202020207573696e67204b65726e656c5363686564756c652020203d206375746c6173733a3a67656d6d3a3a4b65726e656c5074724172726179546d61576172705370656369616c697a656432536d4e766634536d3130303b0a2020202020207573696e67204570696c6f6775655363686564756c65203d206375746c6173733a3a6570696c6f6775653a3a5074724172726179546d61576172705370656369616c697a656432536d3b0a202020207d3b0a0a202020207573696e6720436f6c6c6563746976654570696c6f67756531534d203d20747970656e616d65206375746c6173733a3a6570696c6f6775653a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a2020202020202020417263685461672c204f70657261746f72436c6173732c0a2020202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020202020637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36343e2c0a2020202020202020456c656d656e74416363756d756c61746f722c20456c656d656e74416363756d756c61746f722c0a2020202020202020456c656d656e74432c204c61796f757443202a2c20416c69676e6d656e74432c0a2020202020202020456c656d656e74442c204c61796f757443202a2c20416c69676e6d656e74442c0a2020202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4570696c6f6775655363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e6720436f6c6c6563746976654d61696e6c6f6f7031534d203d20747970656e616d65206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a202020202020417263685461672c204f70657261746f72436c6173732c0a202020202020456c656d656e74412c204c61796f757441202a2c20416c69676e6d656e74412c0a202020202020456c656d656e74422c204c61796f757442202a2c20416c69676e6d656e74422c0a202020202020456c656d656e74416363756d756c61746f722c0a202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020205374616765436f756e745479706531536d2c0a202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4b65726e656c5363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e672047656d6d4b65726e656c31534d203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a47656d6d556e6976657273616c3c0a202020202020202050726f626c656d53686170652c0a2020202020202020436f6c6c6563746976654d61696e6c6f6f7031534d2c0a2020202020202020436f6c6c6563746976654570696c6f67756531534d2c0a20202020202020206375746c6173733a3a67656d6d3a3a53747265616d4b5363686564756c65720a202020203e3b0a202020207573696e672047656d6d31534d203d206375746c6173733a3a67656d6d3a3a6465766963653a3a47656d6d556e6976657273616c416461707465723c47656d6d4b65726e656c31534d3e3b0a0a202020207573696e6720436f6c6c6563746976654570696c6f67756532534d203d20747970656e616d65206375746c6173733a3a6570696c6f6775653a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a2020202020202020417263685461672c204f70657261746f72436c6173732c0a2020202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020202020637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36343e2c0a2020202020202020456c656d656e74416363756d756c61746f722c20456c656d656e74416363756d756c61746f722c0a2020202020202020456c656d656e74432c204c61796f757443202a2c20416c69676e6d656e74432c0a2020202020202020456c656d656e74442c204c61796f757443202a2c20416c69676e6d656e74442c0a2020202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4570696c6f6775655363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e6720436f6c6c6563746976654d61696e6c6f6f7032534d203d20747970656e616d65206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a202020202020417263685461672c204f70657261746f72436c6173732c0a202020202020456c656d656e74412c204c61796f757441202a2c20416c69676e6d656e74412c0a202020202020456c656d656e74422c204c61796f757442202a2c20416c69676e6d656e74422c0a202020202020456c656d656e74416363756d756c61746f722c0a202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020205374616765436f756e745479706532536d2c0a202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4b65726e656c5363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e672047656d6d4b65726e656c32534d203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a47656d6d556e6976657273616c3c0a202020202020202050726f626c656d53686170652c0a2020202020202020436f6c6c6563746976654d61696e6c6f6f7032534d2c0a2020202020202020436f6c6c6563746976654570696c6f67756532534d2c0a20202020202020206375746c6173733a3a67656d6d3a3a53747265616d4b5363686564756c65720a202020203e3b0a202020207573696e672047656d6d32534d203d206375746c6173733a3a67656d6d3a3a6465766963653a3a47656d6d556e6976657273616c416461707465723c47656d6d4b65726e656c32534d3e3b0a0a2020202074656d706c617465203c747970656e616d6520543e0a202020207374727563742050696e6e6564486f7374427566666572207b0a202020202020542a20707472203d206e756c6c7074723b0a20202020202073697a655f7420636f756e74203d20303b0a0a20202020202050696e6e6564486f73744275666665722829203d2064656661756c743b0a2020202020206578706c696369742050696e6e6564486f73744275666665722873697a655f74206e29207b20616c6c6f63617465286e293b207d0a0a202020202020766f696420616c6c6f636174652873697a655f74206e29207b0a20202020202020206966202870747220262620636f756e74203e3d206e29207b0a2020202020202020202072657475726e3b0a20202020202020207d0a202020202020202072656c6561736528293b0a2020202020202020636f756e74203d206e3b0a2020202020202020435544415f434845434b2863756461486f7374416c6c6f63287265696e746572707265745f636173743c766f69642a2a3e2826707472292c206e202a2073697a656f662854292c2063756461486f7374416c6c6f63506f727461626c6529293b0a2020202020207d0a0a202020202020766f69642072656c656173652829207b0a20202020202020206966202870747229207b0a202020202020202020206375646146726565486f737428707472293b0a20202020202020202020707472203d206e756c6c7074723b0a20202020202020202020636f756e74203d20303b0a20202020202020207d0a2020202020207d0a0a2020202020207e50696e6e6564486f73744275666665722829207b2072656c6561736528293b207d0a0a202020202020542a2064617461282920636f6e7374207b2072657475726e207074723b207d0a2020202020205426206f70657261746f725b5d2873697a655f742069647829207b2072657475726e207074725b6964785d3b207d0a202020207d3b0a0a2020202074656d706c617465203c747970656e616d6520543e0a2020202073747275637420446576696365427566666572207b0a202020202020542a20707472203d206e756c6c7074723b0a20202020202073697a655f7420636f756e74203d20303b0a0a2020202020204465766963654275666665722829203d2064656661756c743b0a2020202020206578706c69636974204465766963654275666665722873697a655f74206e29207b20616c6c6f63617465286e293b207d0a0a202020202020766f696420616c6c6f636174652873697a655f74206e29207b0a20202020202020206966202870747220262620636f756e74203e3d206e29207b0a2020202020202020202072657475726e3b0a20202020202020207d0a202020202020202072656c6561736528293b0a2020202020202020636f756e74203d206e3b0a2020202020202020435544415f434845434b28637564614d616c6c6f63287265696e746572707265745f636173743c766f69642a2a3e2826707472292c206e202a2073697a656f6628542929293b0a2020202020207d0a0a202020202020766f69642072656c656173652829207b0a20202020202020206966202870747229207b0a20202020202020202020637564614672656528707472293b0a20202020202020202020707472203d206e756c6c7074723b0a20202020202020202020636f756e74203d20303b0a20202020202020207d0a2020202020207d0a0a2020202020207e4465766963654275666665722829207b2072656c6561736528293b207d0a0a202020202020542a20676574282920636f6e7374207b2072657475726e207074723b207d0a0a202020202020766f696420636f70795f66726f6d5f686f737428636f6e737420542a20686f73742c2073697a655f74206e2c206375646153747265616d5f742073747265616d29207b0a2020202020202020435544415f434845434b28637564614d656d6370794173796e63287074722c20686f73742c206e202a2073697a656f662854292c20637564614d656d637079486f7374546f4465766963652c2073747265616d29293b0a2020202020207d0a202020207d3b0a0a2020202074656d706c617465203c747970656e616d652047656d6d3e0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2072756e5f67726f757065645f696d706c280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a0a202020202020544f5243485f434845434b28412e73697a652829203d3d20422e73697a6528292c2022412f422073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d20432e73697a6528292c2022412f432073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d205346412e73697a6528292c2022412f5346412073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d205346422e73697a6528292c2022412f5346422073697a65206d69736d6174636822293b0a202020202020636f6e737420696e7433325f742067726f757073203d207374617469635f636173743c696e7433325f743e28412e73697a652829293b0a202020202020544f5243485f434845434b2867726f757073203e20302c20224e6f2067726f75707322293b0a0a202020202020696e74206465766963655f6964203d20415b305d2e6765745f64657669636528293b0a2020202020206331303a3a637564613a3a435544414775617264206465766963655f6775617264286465766963655f6964293b0a0a2020202020207573696e672053747269646541203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465413b0a2020202020207573696e672053747269646542203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465423b0a2020202020207573696e672053747269646543203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465433b0a2020202020207573696e672053747269646544203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465443b0a2020202020207573696e67204c61796f7574534641203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a496e7465726e616c4c61796f75745346413b0a2020202020207573696e67204c61796f7574534642203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a496e7465726e616c4c61796f75745346423b0a2020202020207573696e6720536d317878426c6b5363616c6564436f6e666967203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a536d317878426c6b5363616c6564436f6e6669673b0a0a2020202020207374617469632073697a655f74206361706163697479203d20303b0a20202020202073746174696320696e74206361636865645f646576696365203d202d313b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e207074725f415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e207074725f425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e207074725f435f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e207074725f445f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465413e207374726964655f415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465423e207374726964655f425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465433e207374726964655f435f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465443e207374726964655f445f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c4c61796f75745346413e206c61796f75745f5346415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c4c61796f75745346423e206c61796f75745f5346425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652050726f626c656d53686170653a3a556e6465726c79696e6750726f626c656d53686170653e2070726f626c656d5f73697a65735f686f73743b0a0a202020202020737461746963204465766963654275666665723c747970656e616d652050726f626c656d53686170653a3a556e6465726c79696e6750726f626c656d53686170653e2070726f626c656d5f73697a65735f6465766963653b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e207074725f413b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e207074725f423b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346413b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346423b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e207074725f433b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e207074725f443b0a202020202020737461746963204465766963654275666665723c537472696465413e207374726964655f413b0a202020202020737461746963204465766963654275666665723c537472696465423e207374726964655f423b0a202020202020737461746963204465766963654275666665723c537472696465433e207374726964655f433b0a202020202020737461746963204465766963654275666665723c537472696465443e207374726964655f443b0a202020202020737461746963204465766963654275666665723c4c61796f75745346413e206c61796f75745f5346413b0a202020202020737461746963204465766963654275666665723c4c61796f75745346423e206c61796f75745f5346423b0a2020202020207374617469632073697a655f7420776f726b73706163655f73697a655f636163686564203d20303b0a20202020202073746174696320746f7263683a3a54656e736f7220776f726b73706163655f74656e736f723b0a0a202020202020696620286361636865645f64657669636520213d206465766963655f6964207c7c206361706163697479203c207374617469635f636173743c73697a655f743e2867726f7570732929207b0a20202020202020206361636865645f646576696365203d206465766963655f69643b0a20202020202020206361706163697479203d207374617469635f636173743c73697a655f743e2867726f757073293b0a20202020202020207074725f415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f435f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f445f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f435f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f445f686f73742e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346425f686f73742e616c6c6f636174652867726f757073293b0a202020202020202070726f626c656d5f73697a65735f686f73742e616c6c6f636174652867726f757073293b0a0a202020202020202070726f626c656d5f73697a65735f6465766963652e616c6c6f636174652867726f757073293b0a20202020202020207074725f412e616c6c6f636174652867726f757073293b0a20202020202020207074725f422e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346412e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346422e616c6c6f636174652867726f757073293b0a20202020202020207074725f432e616c6c6f636174652867726f757073293b0a20202020202020207074725f442e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f412e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f422e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f432e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f442e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346412e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346422e616c6c6f636174652867726f757073293b0a2020202020207d0a0a202020202020636f6e73746578707220696e742074696c655f6d203d20637574653a3a73697a653c303e28747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c6553686170657b7d293b0a202020202020636f6e73746578707220696e742074696c655f6e203d20637574653a3a73697a653c313e28747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c6553686170657b7d293b0a202020202020696e7436345f7420746f74616c5f74696c6573203d20303b0a0a202020202020666f722028696e7433325f742069203d20303b2069203c2067726f7570733b202b2b6929207b0a2020202020202020544f5243485f434845434b28415b695d2e69735f6375646128292c202241206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b28425b695d2e69735f6375646128292c202242206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b28435b695d2e69735f6375646128292c202243206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b285346415b695d2e69735f6375646128292c2022534641206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b285346425b695d2e69735f6375646128292c2022534642206d757374206265204355444122293b0a0a2020202020202020696e7433325f74206d203d207374617469635f636173743c696e7433325f743e28415b695d2e73697a65283029293b0a2020202020202020696e7433325f74206b5f7061636b6564203d207374617469635f636173743c696e7433325f743e28415b695d2e73697a65283129293b0a2020202020202020696e7433325f74206b203d206b5f7061636b6564202a20323b0a2020202020202020696e7433325f74206e203d207374617469635f636173743c696e7433325f743e28425b695d2e73697a65283029293b0a2020202020202020544f5243485f434845434b28425b695d2e73697a65283129202a2032203d3d206b2c20224b206d69736d6174636822293b0a2020202020202020544f5243485f434845434b28435b695d2e73697a65283029203d3d206d20262620435b695d2e73697a65283129203d3d206e2c202243207368617065206d69736d6174636822293b0a0a202020202020202070726f626c656d5f73697a65735f686f73745b695d203d20637574653a3a6d616b655f7368617065286e2c206d2c206b293b0a20202020202020207374726964655f415f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465417b7d2c207b6e2c206b2c20317d293b0a20202020202020207374726964655f425f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465427b7d2c207b6d2c206b2c20317d293b0a20202020202020207374726964655f435f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465437b7d2c207b6e2c206d2c20317d293b0a20202020202020207374726964655f445f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465447b7d2c207b6e2c206d2c20317d293b0a20202020202020206c61796f75745f5346415f686f73745b695d203d20536d317878426c6b5363616c6564436f6e6669673a3a74696c655f61746f6d5f746f5f73686170655f53464128637574653a3a6d616b655f7368617065286e2c206d2c206b2c203129293b0a20202020202020206c61796f75745f5346425f686f73745b695d203d20536d317878426c6b5363616c6564436f6e6669673a3a74696c655f61746f6d5f746f5f73686170655f53464228637574653a3a6d616b655f7368617065286e2c206d2c206b2c203129293b0a0a2020202020202020696e742074696c65735f6d203d20286e202b2074696c655f6d202d203129202f2074696c655f6d3b0a2020202020202020696e742074696c65735f6e203d20286d202b2074696c655f6e202d203129202f2074696c655f6e3b0a2020202020202020746f74616c5f74696c6573202b3d207374617469635f636173743c696e7436345f743e2874696c65735f6d29202a207374617469635f636173743c696e7436345f743e2874696c65735f6e293b0a0a20202020202020207074725f415f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e28425b695d2e646174615f7074722829293b0a20202020202020207074725f425f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e28415b695d2e646174615f7074722829293b0a20202020202020207074725f435f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e28435b695d2e646174615f7074722829293b0a20202020202020207074725f445f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e28435b695d2e646174615f7074722829293b0a20202020202020207074725f5346415f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e285346425b695d2e646174615f7074722829293b0a20202020202020207074725f5346425f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e285346415b695d2e646174615f7074722829293b0a2020202020207d0a0a2020202020206175746f2073747265616d5f6f626a203d206331303a3a637564613a3a67657443757272656e744355444153747265616d286465766963655f6964293b0a2020202020206331303a3a637564613a3a4355444153747265616d47756172642073747265616d5f67756172642873747265616d5f6f626a293b0a2020202020206375646153747265616d5f742073747265616d203d2073747265616d5f6f626a2e73747265616d28293b0a20202020202070726f626c656d5f73697a65735f6465766963652e636f70795f66726f6d5f686f73742870726f626c656d5f73697a65735f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f412e636f70795f66726f6d5f686f7374287074725f415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f422e636f70795f66726f6d5f686f7374287074725f425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f5346412e636f70795f66726f6d5f686f7374287074725f5346415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f5346422e636f70795f66726f6d5f686f7374287074725f5346425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f432e636f70795f66726f6d5f686f7374287074725f435f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f442e636f70795f66726f6d5f686f7374287074725f445f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f412e636f70795f66726f6d5f686f7374287374726964655f415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f422e636f70795f66726f6d5f686f7374287374726964655f425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f432e636f70795f66726f6d5f686f7374287374726964655f435f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f442e636f70795f66726f6d5f686f7374287374726964655f445f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020206c61796f75745f5346412e636f70795f66726f6d5f686f7374286c61796f75745f5346415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020206c61796f75745f5346422e636f70795f66726f6d5f686f7374286c61796f75745f5346425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a0a2020202020206375746c6173733a3a4b65726e656c4861726477617265496e666f2068775f696e666f3b0a20202020202068775f696e666f2e6465766963655f6964203d206465766963655f69643b0a20202020202068775f696e666f2e736d5f636f756e74203d206375746c6173733a3a4b65726e656c4861726477617265496e666f3a3a71756572795f6465766963655f6d756c746970726f636573736f725f636f756e742868775f696e666f2e6465766963655f6964293b0a202020202020696620636f6e73746578707220287374643a3a69735f73616d655f763c47656d6d2c2047656d6d31534d3e29207b0a202020202020202068775f696e666f2e636c75737465725f7368617065203d2064696d3328312c20312c2031293b0a202020202020202068775f696e666f2e636c75737465725f73686170655f66616c6c6261636b203d2064696d3328312c20312c2031293b0a2020202020207d20656c7365207b0a202020202020202068775f696e666f2e636c75737465725f7368617065203d2064696d3328322c20312c2031293b0a202020202020202068775f696e666f2e636c75737465725f73686170655f66616c6c6261636b203d2064696d3328322c20312c2031293b0a2020202020207d0a0a202020202020747970656e616d652047656d6d3a3a417267756d656e747320617267756d656e74733b0a2020202020206465636c7479706528617267756d656e74732e6570696c6f6775652e7468726561642920667573696f6e5f617267733b0a202020202020667573696f6e5f617267732e616c7068615f707472203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e626574615f707472203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e616c706861203d20456c656d656e74416363756d756c61746f722831293b0a202020202020667573696f6e5f617267732e62657461203d20456c656d656e74416363756d756c61746f722830293b0a202020202020667573696f6e5f617267732e616c7068615f7074725f6172726179203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e626574615f7074725f6172726179203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e64416c706861203d207b637574653a3a5f307b7d2c20637574653a3a5f307b7d2c20307d3b0a202020202020667573696f6e5f617267732e6442657461203d207b637574653a3a5f307b7d2c20637574653a3a5f307b7d2c20307d3b0a0a202020202020747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c655363686564756c6572417267756d656e7473207363686564756c65723b0a2020202020207363686564756c65722e6d61785f7377697a7a6c655f73697a65203d20383b0a2020202020207363686564756c65722e7261737465725f6f72646572203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a64657461696c3a3a5261737465724f726465724f7074696f6e733a3a4865757269737469633b0a0a202020202020617267756d656e7473203d20747970656e616d652047656d6d3a3a417267756d656e7473207b0a20202020202020206375746c6173733a3a67656d6d3a3a47656d6d556e6976657273616c4d6f64653a3a6b47726f757065642c0a20202020202020207b67726f7570732c2070726f626c656d5f73697a65735f6465766963652e67657428292c206e756c6c7074727d2c0a20202020202020207b7074725f412e67657428292c207374726964655f412e67657428292c207074725f422e67657428292c207374726964655f422e67657428292c0a2020202020202020207074725f5346412e67657428292c206c61796f75745f5346412e67657428292c207074725f5346422e67657428292c206c61796f75745f5346422e67657428297d2c0a20202020202020207b667573696f6e5f617267732c207074725f432e67657428292c207374726964655f432e67657428292c207074725f442e67657428292c207374726964655f442e67657428297d2c0a202020202020202068775f696e666f2c207363686564756c65720a2020202020207d3b0a0a20202020202047656d6d2067656d6d3b0a20202020202073697a655f7420776f726b73706163655f73697a65203d2047656d6d3a3a6765745f776f726b73706163655f73697a6528617267756d656e7473293b0a2020202020206966202821776f726b73706163655f74656e736f722e646566696e65642829207c7c206361636865645f64657669636520213d206465766963655f6964207c7c20776f726b73706163655f73697a655f636163686564203c20776f726b73706163655f73697a6529207b0a20202020202020206175746f206f707473203d20746f7263683a3a54656e736f724f7074696f6e7328292e64657669636528746f7263683a3a6b435544412c206465766963655f6964292e647479706528746f7263683a3a6b55496e7438293b0a2020202020202020776f726b73706163655f74656e736f72203d20746f7263683a3a656d707479287b7374617469635f636173743c6c6f6e673e28776f726b73706163655f73697a65297d2c206f707473293b0a2020202020202020776f726b73706163655f73697a655f636163686564203d20776f726b73706163655f73697a653b0a2020202020207d0a202020202020766f69642a20776f726b73706163655f707472203d20776f726b73706163655f73697a65203f20776f726b73706163655f74656e736f722e646174615f7074722829203a206e756c6c7074723b0a0a2020202020204355544c4153535f434845434b2867656d6d2e63616e5f696d706c656d656e7428617267756d656e747329293b0a2020202020204355544c4153535f434845434b2867656d6d2e696e697469616c697a6528617267756d656e74732c20776f726b73706163655f7074722c2073747265616d29293b0a2020202020206175746f2072756e5f737461747573203d2067656d6d2e72756e2873747265616d2c206e756c6c7074722c2074727565293b0a2020202020206966202872756e5f73746174757320213d206375746c6173733a3a5374617475733a3a6b5375636365737329207b0a202020202020202072756e5f737461747573203d2067656d6d2e72756e2873747265616d293b0a2020202020207d0a2020202020204355544c4153535f434845434b2872756e5f737461747573293b0a0a20202020202072657475726e20433b0a202020207d0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f31736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a20202020202072657475726e2072756e5f67726f757065645f696d706c3c47656d6d31534d3e28412c20422c20432c205346412c20534642293b0a202020207d0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f32736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a20202020202072657475726e2072756e5f67726f757065645f696d706c3c47656d6d32534d3e28412c20422c20432c205346412c20534642293b0a202020207d0a202020200a"




def _speedk_dec_hex(h: str) -> str:
    return bytes.fromhex(h).decode("utf-8")


def _speedk_repo_root() -> str:
    return os.path.abspath(os.path.join(_ROOT, "..", "..", "..", ".."))


def _speedk_ensure_cutlass_dir(diag_fn) -> str | None:
    global _speedk_cutlass_root
    if _speedk_cutlass_root is not None:
        return _speedk_cutlass_root

    repo_root = _speedk_repo_root()
    base_dir = os.path.join(repo_root, "tmp", "cutlass_cache")
    root = os.path.join(base_dir, "cutlass-4.3.5")
    inc = os.path.join(root, "include", "cutlass", "cutlass.h")
    if os.path.isfile(inc):
        _speedk_cutlass_root = root
        return root

    try:
        os.makedirs(base_dir, exist_ok=True)
        tgz_path = os.path.join(base_dir, "cutlass.tgz")
        if not os.path.isfile(tgz_path):
            import urllib.request

            url = "https://github.com/NVIDIA/cutlass/archive/refs/tags/v4.3.5.tar.gz"
            with urllib.request.urlopen(url, timeout=60) as resp:
                data = resp.read()
            with open(tgz_path, "wb") as f:
                f.write(data)

        import tarfile

        with tarfile.open(tgz_path, "r:gz") as tf:
            tf.extractall(base_dir)

        if os.path.isfile(inc):
            _speedk_cutlass_root = root
            return root
    except Exception:
        diag_fn("speedk ext: cutlass fetch fail")
        return None

    return None


def _speedk_find_inc(diag_fn) -> list[str]:
    global _speedk_inc_cache
    if _speedk_inc_cache is not None:
        return _speedk_inc_cache

    incs: list[str] = []

    def _add_cutlass(base: str) -> bool:
        inc = os.path.join(base, "include")
        if os.path.isfile(os.path.join(inc, "cutlass", "cutlass.h")):
            incs.append(inc)
            util_inc = os.path.join(base, "tools", "util", "include")
            if os.path.isfile(os.path.join(util_inc, "cutlass", "util", "device_memory.h")):
                incs.append(util_inc)
            return True

        inc2 = os.path.join(base, "cutlass", "include")
        if os.path.isfile(os.path.join(inc2, "cutlass", "cutlass.h")):
            incs.append(inc2)
            util_inc2 = os.path.join(base, "cutlass", "tools", "util", "include")
            if os.path.isfile(os.path.join(util_inc2, "cutlass", "util", "device_memory.h")):
                incs.append(util_inc2)
            return True

        return False

    repo_root = _speedk_repo_root()
    for base in (
        os.path.join(repo_root, "third_party", "cutlass"),
        os.path.join(repo_root, "tmp", "cutlass-4.3.5"),
    ):
        _add_cutlass(base)

    try:
        root = os.path.dirname(getattr(cutlass, "__file__", "") or "")
        if root:
            bases = [root, os.path.dirname(root), os.path.dirname(os.path.dirname(root))]
            for base in bases:
                if _add_cutlass(base):
                    break
    except Exception:
        pass

    try:
        import torch.utils.cpp_extension as _ce

        for inc in _ce.include_paths():
            if os.path.isfile(os.path.join(inc, "cutlass", "cutlass.h")):
                incs.append(inc)
            util_inc3 = os.path.join(inc, "cutlass", "tools", "util", "include")
            if os.path.isfile(os.path.join(util_inc3, "cutlass", "util", "device_memory.h")):
                incs.append(util_inc3)
    except Exception:
        pass

    if not incs:
        root = _speedk_ensure_cutlass_dir(diag_fn)
        if root:
            _add_cutlass(root)

    incs = list(dict.fromkeys(incs))
    if not incs:
        diag_fn("speedk ext: no headers")
    _speedk_inc_cache = incs
    return incs


def _speedk_load_ext(diag_fn):
    global _speedk_ext_mod, _speedk_ext_fail
    if _speedk_ext_mod is not None or _speedk_ext_fail:
        return _speedk_ext_mod

    incs = _speedk_find_inc(diag_fn)
    if not incs:
        _speedk_ext_fail = True
        return None

    try:
        from torch.utils.cpp_extension import load_inline

        repo_root = _speedk_repo_root()
        os.environ.setdefault(
            "TORCH_EXTENSIONS_DIR", os.path.join(repo_root, "tmp", "torch_extensions")
        )
        os.environ.setdefault("TORCH_CUDA_ARCH_LIST", "10.0a")

        cpp_src = _speedk_dec_hex(_SPEEDK_CPP_HEX)
        cu_src = _speedk_dec_hex(_SPEEDK_CU_HEX)

        _speedk_ext_mod = load_inline(
            name="nvfp4_grouped_cutlass_ext_speedk",
            cpp_sources=cpp_src,
            cuda_sources=cu_src,
            functions=["grouped_gemm_1sm", "grouped_gemm_2sm"],
            extra_cuda_cflags=[
                "-O3",
                "--use_fast_math",
                "-std=c++17",
                "--diag-suppress=144",
            ],
            extra_ldflags=["-lcuda"],
            extra_cflags=["-O3", "-std=c++17"],
            with_cuda=True,
            extra_include_paths=incs,
            verbose=False,
        )
    except Exception:
        _speedk_ext_fail = True
        _speedk_ext_mod = None
        diag_fn("speedk ext: build fail")
        return None

    return _speedk_ext_mod


def _speedk_pad_rows(tensor: torch.Tensor, new_rows: int) -> torch.Tensor:
    if tensor.size(0) == new_rows:
        return tensor
    dev_idx = tensor.device.index if tensor.device.index is not None else -1
    key = (new_rows, tensor.size(1), tensor.dtype, dev_idx)
    out = _speedk_pad_cache.get(key)
    if out is None or out.size(0) != new_rows or out.size(1) != tensor.size(1):
        out = torch.empty((new_rows, tensor.size(1)), device=tensor.device, dtype=tensor.dtype)
        _speedk_pad_cache[key] = out
    # Some low-bit dtypes (notably fp4x2) don't reliably support in-place fill or
    # slice assignment. Bitcast to u8 so we only move raw bytes.
    if tensor.element_size() == 1:
        src = tensor if tensor.is_contiguous() else tensor.contiguous()
        out_u8 = out.view(torch.uint8)
        out_u8.zero_()
        out_u8[: src.size(0), :] = src.view(torch.uint8)
        return out
    out.zero_()
    out[: tensor.size(0), :] = tensor
    return out


def _speedk_alloc_c_pad(rows: int, cols: int, ref: torch.Tensor) -> torch.Tensor:
    dev_idx = ref.device.index if ref.device.index is not None else -1
    key = (rows, cols, ref.dtype, dev_idx)
    out = _speedk_c_pad_cache.get(key)
    if out is None or out.size(0) != rows or out.size(1) != cols:
        out = torch.empty((rows, cols), device=ref.device, dtype=ref.dtype)
        _speedk_c_pad_cache[key] = out
    return out


def _speedk_try_ext(
    ext_mod,
    abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
    sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],
    problem_sizes: list[tuple[int, int, int, int]],
):
    if ext_mod is None:
        return None

    def _use_2sm(n_orig: int, k: int) -> bool:
        return n_orig >= 3072 and k >= 2048

    items = []
    for (a_ref, b_ref, c_ref), (sfa_reordered, sfb_reordered), (m, _n, k, l) in zip(
        abc_tensors, sfs_reordered, problem_sizes
    ):
        if l != 1:
            return None
        a2 = a_ref[:, :, 0]
        b2 = b_ref[:, :, 0]
        c2 = c_ref[:, :, 0]
        n = b2.size(0)
        m_pad = ((m + 31) // 32) * 32
        if m_pad != m:
            a2 = _speedk_pad_rows(a2, m_pad)
            c_pad = _speedk_alloc_c_pad(m_pad, n, c2)
            items.append((a2, b2, c_pad, sfa_reordered, sfb_reordered, c2, m, n, k))
        else:
            items.append((a2, b2, c2, sfa_reordered, sfb_reordered, None, m, n, k))

    def _gather(src):
        return (
            [x[0] for x in src],
            [x[1] for x in src],
            [x[2] for x in src],
            [x[3] for x in src],
            [x[4] for x in src],
        )

    small = [it for it in items if not _use_2sm(it[7], it[8])]
    large = [it for it in items if _use_2sm(it[7], it[8])]

    try:
        if large:
            a_list, b_list, c_list, sfa_list, sfb_list = _gather(large)
            ext_mod.grouped_gemm_2sm(a_list, b_list, c_list, sfa_list, sfb_list)
        if small:
            a_list, b_list, c_list, sfa_list, sfb_list = _gather(small)
            ext_mod.grouped_gemm_1sm(a_list, b_list, c_list, sfa_list, sfb_list)
    except Exception:
        try:
            a_list, b_list, c_list, sfa_list, sfb_list = _gather(items)
            ext_mod.grouped_gemm_1sm(a_list, b_list, c_list, sfa_list, sfb_list)
        except Exception:
            return None

    for _a2, _b2, c2, _sfa, _sfb, c_orig, m, _n, _k in items:
        if c_orig is not None:
            c_orig.copy_(c2[:m, :])

    return [c_ref for (_a, _b, c_ref) in abc_tensors]


@dsl_user_op
def _normalize_ptr(ptr, *, loc=None, ip=None):
    if isinstance(ptr, _ir.Value):
        return ptr
    if hasattr(ptr, "to_llvm_ptr") and callable(ptr.to_llvm_ptr):
        return ptr.to_llvm_ptr(loc=loc, ip=ip)
    return ptr


@dsl_user_op
def _ptx_prefetch_global(ptr, *, loc=None, ip=None):
    ptr = _normalize_ptr(ptr, loc=loc, ip=ip)
    if not isinstance(ptr, _ir.Value):
        return None
    _llvm.inline_asm(
        None,
        [ptr],
        "prefetch.global.L2::evict_last [$0];",
        "l",
        has_side_effects=True,
        is_align_stack=False,
        asm_dialect=_llvm.AsmDialect.AD_ATT,
        loc=loc,
        ip=ip,
    )
    return None


@dsl_user_op
def _ptx_prefetch_global_l1(ptr, *, loc=None, ip=None):
    ptr = _normalize_ptr(ptr, loc=loc, ip=ip)
    if not isinstance(ptr, _ir.Value):
        return None
    _llvm.inline_asm(
        None,
        [ptr],
        "prefetch.global.L1 [$0];",
        "l",
        has_side_effects=True,
        is_align_stack=False,
        asm_dialect=_llvm.AsmDialect.AD_ATT,
        loc=loc,
        ip=ip,
    )
    return None


@dsl_user_op
def _ptx_prefetchu_l1(ptr, *, loc=None, ip=None):
    ptr = _normalize_ptr(ptr, loc=loc, ip=ip)
    if not isinstance(ptr, _ir.Value):
        return None
    _llvm.inline_asm(
        None,
        [ptr],
        "prefetchu.L1 [$0];",
        "l",
        has_side_effects=True,
        is_align_stack=False,
        asm_dialect=_llvm.AsmDialect.AD_ATT,
        loc=loc,
        ip=ip,
    )
    return None


@dsl_user_op
def _prefetch_tma(
    atom: cute.CopyAtom,
    src: cute.Tensor,
    tma_desc_ptr: cute.Pointer,
    *,
    loc=None,
    ip=None,
):
    if hasattr(src, "iterator"):
        _ptx_prefetch_global(src.iterator, loc=loc, ip=ip)
        _ptx_prefetch_global_l1(src.iterator, loc=loc, ip=ip)
    if tma_desc_ptr is not None:
        _ptx_prefetch_global(tma_desc_ptr, loc=loc, ip=ip)
        _ptx_prefetch_global_l1(tma_desc_ptr, loc=loc, ip=ip)
        _ptx_prefetchu_l1(tma_desc_ptr, loc=loc, ip=ip)
    dummy_tma_bar_ptr = cute.make_ptr(
        cutlass.Int64, 0, cute.AddressSpace.smem, loc=loc, ip=ip
    )
    dummy_mcast_mask = cutlass.Int16(0, loc=loc, ip=ip)
    value = atom._unpack(
        tma_bar_ptr=dummy_tma_bar_ptr,
        mcast_mask=dummy_mcast_mask,
        tma_desc_ptr=tma_desc_ptr,
        loc=loc,
        ip=ip,
    )
    return _cute_ir.prefetch(value, src.value, loc=loc, ip=ip)


class Sm100GroupedBlockScaledGemmKernel:

    def __init__(
        self,
        sf_vec_size: int,
        mma_tiler_mn: Tuple[int, int],
        cluster_shape_mn: Tuple[int, int],
        use_tma_store: bool = True,
        max_ab_stage: int | None = None,
        prefetch_dist: int | None = None,
        force_c_stage: int | None = None,
        c_assumed_align: int = 16,
        c_divisibility: int = 8,
        swizzle_size: int = 1,
        raster_along_m: bool = True,
    ):
        self.acc_dtype = cutlass.Float32
        self.sf_vec_size = sf_vec_size
        self.use_2cta_instrs = mma_tiler_mn[0] == 256
        self.cluster_shape_mn = cluster_shape_mn
        self.use_tma_store = use_tma_store
        self.max_ab_stage = max_ab_stage
        self.prefetch_dist_override = prefetch_dist
        self.force_c_stage = force_c_stage
        self.c_assumed_align = c_assumed_align
        self.c_divisibility = c_divisibility
        self.swizzle_size = swizzle_size
        self.raster_along_m = raster_along_m
        # K dimension is deferred in _setup_attributes
        self.mma_tiler = (*mma_tiler_mn, 1)

        self.cta_group = (
            tcgen05.CtaGroup.TWO if self.use_2cta_instrs else tcgen05.CtaGroup.ONE
        )

        self.tensormap_update_mode = utils.TensorMapUpdateMode.SMEM

        self.occupancy = 1
        # Set specialized warp ids
        self.epilog_warp_id = (
            0,
            1,
            2,
            3,
        )
        self.mma_warp_id = 4
        self.tma_warp_id = 5
        self.threads_per_cta = 32 * len(
            (self.mma_warp_id, self.tma_warp_id, *self.epilog_warp_id)
        )
        # Set barrier for epilogue sync and tmem ptr sync
        self.epilog_sync_barrier = pipeline.NamedBarrier(
            barrier_id=1,
            num_threads=32 * len(self.epilog_warp_id),
        )
        self.tmem_alloc_barrier = pipeline.NamedBarrier(
            barrier_id=2,
            num_threads=32 * len((self.mma_warp_id, *self.epilog_warp_id)),
        )
        # Barrier used by MMA/TMA warps to signal A/B tensormap initialization completion
        self.tensormap_ab_init_barrier = pipeline.NamedBarrier(
            barrier_id=3,
            num_threads=64,
        )
        self.smem_capacity = utils.get_smem_capacity_in_bytes("sm_100")
        SM100_TMEM_CAPACITY_COLUMNS = 512
        self.num_tmem_alloc_cols = SM100_TMEM_CAPACITY_COLUMNS
        self.c_tile_stride: tuple[int, int] | None = None

    # Set up configurations that dependent on gemm inputs.
    def _setup_attributes(self):
        # Compute mma instruction shapes
        # (MMA_Tile_Shape_M, MMA_Tile_Shape_N, MMA_Inst_Shape_K)
        self.mma_inst_shape_mn = (
            self.mma_tiler[0],
            self.mma_tiler[1],
        )
        # (CTA_Tile_Shape_M, Round_Up(MMA_Tile_Shape_N, 128), MMA_Inst_Shape_K)
        self.mma_inst_shape_mn_sfb = (
            self.mma_inst_shape_mn[0] // (2 if self.use_2cta_instrs else 1),
            cute.round_up(self.mma_inst_shape_mn[1], 128),
        )

        tiled_mma = sm100_utils.make_blockscaled_trivial_tiled_mma(
            self.a_dtype,
            self.a_major_mode,
            self.b_major_mode,
            self.sf_dtype,
            self.sf_vec_size,
            self.cta_group,
            self.mma_inst_shape_mn,
        )

        tiled_mma_sfb = sm100_utils.make_blockscaled_trivial_tiled_mma(
            self.a_dtype,
            self.a_major_mode,
            self.b_major_mode,
            self.sf_dtype,
            self.sf_vec_size,
            cute.nvgpu.tcgen05.CtaGroup.ONE,
            self.mma_inst_shape_mn_sfb,
        )

        # Compute mma/cluster/tile shapes
        mma_inst_shape_k = cute.size(tiled_mma.shape_mnk, mode=[2])
        mma_inst_tile_k = 4
        self.mma_tiler = (
            self.mma_inst_shape_mn[0],
            self.mma_inst_shape_mn[1],
            mma_inst_shape_k * mma_inst_tile_k,
        )
        self.mma_tiler_sfb = (
            self.mma_inst_shape_mn_sfb[0],
            self.mma_inst_shape_mn_sfb[1],
            mma_inst_shape_k * mma_inst_tile_k,
        )
        self.cta_tile_shape_mnk = (
            self.mma_tiler[0] // cute.size(tiled_mma.thr_id.shape),
            self.mma_tiler[1],
            self.mma_tiler[2],
        )
        self.cluster_tile_shape_mnk = tuple(
            x * y for x, y in zip(self.cta_tile_shape_mnk, (*self.cluster_shape_mn, 1))
        )

        # Compute cluster layout
        self.cluster_layout_vmnk = cute.tiled_divide(
            cute.make_layout((*self.cluster_shape_mn, 1)),
            (tiled_mma.thr_id.shape,),
        )
        self.cluster_layout_sfb_vmnk = cute.tiled_divide(
            cute.make_layout((*self.cluster_shape_mn, 1)),
            (tiled_mma_sfb.thr_id.shape,),
        )

        # Compute number of multicast CTAs for A/B
        self.num_mcast_ctas_a = cute.size(self.cluster_layout_vmnk.shape[2])
        self.num_mcast_ctas_b = cute.size(self.cluster_layout_vmnk.shape[1])
        self.num_mcast_ctas_sfb = cute.size(self.cluster_layout_sfb_vmnk.shape[1])
        self.is_a_mcast = self.num_mcast_ctas_a > 1
        self.is_b_mcast = self.num_mcast_ctas_b > 1
        self.is_sfb_mcast = self.num_mcast_ctas_sfb > 1

        # Compute epilogue subtile
        self.epi_tile = sm100_utils.compute_epilogue_tile_shape(
            self.cta_tile_shape_mnk,
            self.use_2cta_instrs,
            self.c_layout,
            self.c_dtype,
        )

        # Setup A/B/C stage count in shared memory and ACC stage count in tensor memory
        self.num_acc_stage, self.num_ab_stage, self.num_c_stage = self._compute_stages(
            tiled_mma,
            self.mma_tiler,
            self.a_dtype,
            self.b_dtype,
            self.epi_tile,
            self.c_dtype,
            self.c_layout,
            self.sf_dtype,
            self.sf_vec_size,
            self.smem_capacity,
            self.occupancy,
            self.force_c_stage,
        )
        max_ab = self.max_ab_stage
        if max_ab is None:
            max_ab = 6 if self.use_2cta_instrs else 4
        self.num_ab_stage = min(self.num_ab_stage, max_ab)
        self.prefetch_dist = max(0, self.num_ab_stage - 2)
        if self.prefetch_dist_override is not None:
            self.prefetch_dist = max(
                0, min(self.prefetch_dist_override, self.num_ab_stage)
            )
        self.prefetch_enabled = self.prefetch_dist > 0

        # Compute A/B/SFA/SFB/C shared memory layout
        self.a_smem_layout_staged = sm100_utils.make_smem_layout_a(
            tiled_mma,
            self.mma_tiler,
            self.a_dtype,
            self.num_ab_stage,
        )
        self.b_smem_layout_staged = sm100_utils.make_smem_layout_b(
            tiled_mma,
            self.mma_tiler,
            self.b_dtype,
            self.num_ab_stage,
        )
        self.sfa_smem_layout_staged = blockscaled_utils.make_smem_layout_sfa(
            tiled_mma,
            self.mma_tiler,
            self.sf_vec_size,
            self.num_ab_stage,
        )
        self.sfb_smem_layout_staged = blockscaled_utils.make_smem_layout_sfb(
            tiled_mma,
            self.mma_tiler,
            self.sf_vec_size,
            self.num_ab_stage,
        )
        self.c_smem_layout_staged = sm100_utils.make_smem_layout_epi(
            self.c_dtype,
            self.c_layout,
            self.epi_tile,
            self.num_c_stage,
        )

        mbar_smem_bytes = self._get_mbar_smem_bytes(
            num_acc_stage=self.num_acc_stage,
            num_ab_stage=self.num_ab_stage,
            num_c_stage=self.num_c_stage,
        )

        # Use utils.TensorMapUpdateMode.SMEM by default
        tensormap_smem_bytes = (
            Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap
            * Sm100GroupedBlockScaledGemmKernel.num_tensormaps
        )
        if (
            mbar_smem_bytes
            + tensormap_smem_bytes
            + Sm100GroupedBlockScaledGemmKernel.tensor_memory_management_bytes
            > self.reserved_smem_bytes
        ):
            raise ValueError(
                f"smem consumption for mbar and tensormap {mbar_smem_bytes + tensormap_smem_bytes} exceeds the "
                f"reserved smem bytes {self.reserved_smem_bytes}"
            )

    @cute.jit
    def __call__(
        self,
        initial_a: cute.Tensor,
        initial_b: cute.Tensor,
        initial_c: cute.Tensor,
        initial_sfa: cute.Tensor,
        initial_sfb: cute.Tensor,
        group_count: cutlass.Constexpr[int],
        problem_shape_mnkl: cute.Tensor,
        strides_abc: cute.Tensor,
        tensor_address_abc: cute.Tensor,
        tensor_address_sfasfb: cute.Tensor,
        total_num_clusters: cutlass.Constexpr[int],
        tensormap_cute_tensor: cute.Tensor,
        max_active_clusters: cutlass.Constexpr[int],
        st,
    ):
        self.a_dtype = initial_a.element_type
        self.b_dtype = initial_b.element_type
        self.sf_dtype = initial_sfa.element_type
        self.c_dtype = initial_c.element_type
        self.a_major_mode = utils.LayoutEnum.from_tensor(initial_a).mma_major_mode()
        self.b_major_mode = utils.LayoutEnum.from_tensor(initial_b).mma_major_mode()
        self.c_layout = utils.LayoutEnum.from_tensor(initial_c)
        if cutlass.const_expr(self.a_dtype != self.b_dtype):
            raise TypeError(f"Type mismatch: {self.a_dtype} != {self.b_dtype}")

        # Setup attributes that dependent on gemm inputs
        self._setup_attributes()

        # Setup sfa/sfb tensor by filling A/B tensor to scale factor atom layout
        # ((Atom_M, Rest_M),(Atom_K, Rest_K),RestL)
        sfa_layout = blockscaled_utils.tile_atom_to_shape_SF(
            initial_a.shape, self.sf_vec_size
        )
        initial_sfa = cute.make_tensor(initial_sfa.iterator, sfa_layout)

        # ((Atom_N, Rest_N),(Atom_K, Rest_K),RestL)
        sfb_layout = blockscaled_utils.tile_atom_to_shape_SF(
            initial_b.shape, self.sf_vec_size
        )
        initial_sfb = cute.make_tensor(initial_sfb.iterator, sfb_layout)

        tiled_mma = sm100_utils.make_blockscaled_trivial_tiled_mma(
            self.a_dtype,
            self.a_major_mode,
            self.b_major_mode,
            self.sf_dtype,
            self.sf_vec_size,
            self.cta_group,
            self.mma_inst_shape_mn,
        )

        tiled_mma_sfb = sm100_utils.make_blockscaled_trivial_tiled_mma(
            self.a_dtype,
            self.a_major_mode,
            self.b_major_mode,
            self.sf_dtype,
            self.sf_vec_size,
            cute.nvgpu.tcgen05.CtaGroup.ONE,
            self.mma_inst_shape_mn_sfb,
        )
        atom_thr_size = cute.size(tiled_mma.thr_id.shape)

        # Setup TMA load for A
        a_op = sm100_utils.cluster_shape_to_tma_atom_A(
            self.cluster_shape_mn, tiled_mma.thr_id
        )
        a_smem_layout = cute.slice_(self.a_smem_layout_staged, (None, None, None, 0))
        tma_atom_a, tma_tensor_a = cute.nvgpu.make_tiled_tma_atom_A(
            a_op,
            initial_a,
            a_smem_layout,
            self.mma_tiler,
            tiled_mma,
            self.cluster_layout_vmnk.shape,
        )

        # Setup TMA load for B
        b_op = sm100_utils.cluster_shape_to_tma_atom_B(
            self.cluster_shape_mn, tiled_mma.thr_id
        )
        b_smem_layout = cute.slice_(self.b_smem_layout_staged, (None, None, None, 0))
        tma_atom_b, tma_tensor_b = cute.nvgpu.make_tiled_tma_atom_B(
            b_op,
            initial_b,
            b_smem_layout,
            self.mma_tiler,
            tiled_mma,
            self.cluster_layout_vmnk.shape,
        )

        # Setup TMA load for SFA
        sfa_op = sm100_utils.cluster_shape_to_tma_atom_A(
            self.cluster_shape_mn, tiled_mma.thr_id
        )
        sfa_smem_layout = cute.slice_(
            self.sfa_smem_layout_staged, (None, None, None, 0)
        )
        tma_atom_sfa, tma_tensor_sfa = cute.nvgpu.make_tiled_tma_atom_A(
            sfa_op,
            initial_sfa,
            sfa_smem_layout,
            self.mma_tiler,
            tiled_mma,
            self.cluster_layout_vmnk.shape,
            internal_type=cutlass.Int16,
        )

        # Setup TMA load for SFB
        sfb_op = sm100_utils.cluster_shape_to_tma_atom_SFB(
            self.cluster_shape_mn, tiled_mma.thr_id
        )
        sfb_smem_layout = cute.slice_(
            self.sfb_smem_layout_staged, (None, None, None, 0)
        )
        tma_atom_sfb, tma_tensor_sfb = cute.nvgpu.make_tiled_tma_atom_B(
            sfb_op,
            initial_sfb,
            sfb_smem_layout,
            self.mma_tiler_sfb,
            tiled_mma_sfb,
            self.cluster_layout_sfb_vmnk.shape,
            internal_type=cutlass.Int16,
        )

        a_copy_size = cute.size_in_bytes(self.a_dtype, a_smem_layout)
        b_copy_size = cute.size_in_bytes(self.b_dtype, b_smem_layout)
        sfa_copy_size = cute.size_in_bytes(self.sf_dtype, sfa_smem_layout)
        sfb_copy_size = cute.size_in_bytes(self.sf_dtype, sfb_smem_layout)
        self.num_tma_load_bytes = (
            a_copy_size + b_copy_size + sfa_copy_size + sfb_copy_size
        ) * atom_thr_size

        # Setup TMA store for C
        epi_smem_layout = cute.slice_(self.c_smem_layout_staged, (None, None, 0))
        tma_atom_c, tma_tensor_c = cpasync.make_tiled_tma_atom(
            cpasync.CopyBulkTensorTileS2GOp(),
            initial_c,
            epi_smem_layout,
            self.epi_tile,
        )

        # Compute grid size
        self.tile_sched_params, grid = self._compute_grid(
            total_num_clusters,
            self.cluster_shape_mn,
            max_active_clusters,
            swizzle_size=self.swizzle_size,
            raster_along_m=self.raster_along_m,
        )

        self.buffer_align_bytes = 1024
        self.size_tensormap_in_i64 = (
            Sm100GroupedBlockScaledGemmKernel.num_tensormaps
            * Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap
            // 8
        )

        # Define shared storage for kernel
        @cute.struct
        class SharedStorage:
            tensormap_buffer: cute.struct.MemRange[
                cutlass.Int64, self.size_tensormap_in_i64
            ]
            ab_full_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_ab_stage]
            ab_empty_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_ab_stage]
            acc_full_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_acc_stage]
            acc_empty_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_acc_stage]
            tmem_dealloc_mbar_ptr: cutlass.Int64
            tmem_holding_buf: cutlass.Int32
            # (EPI_TILE_M, EPI_TILE_N, STAGE)
            sC: cute.struct.Align[
                cute.struct.MemRange[
                    self.c_dtype,
                    cute.cosize(self.c_smem_layout_staged.outer),
                ],
                self.buffer_align_bytes,
            ]
            # (MMA, MMA_M, MMA_K, STAGE)
            sA: cute.struct.Align[
                cute.struct.MemRange[
                    self.a_dtype, cute.cosize(self.a_smem_layout_staged.outer)
                ],
                self.buffer_align_bytes,
            ]
            # (MMA, MMA_N, MMA_K, STAGE)
            sB: cute.struct.Align[
                cute.struct.MemRange[
                    self.b_dtype, cute.cosize(self.b_smem_layout_staged.outer)
                ],
                self.buffer_align_bytes,
            ]
            # (MMA, MMA_M, MMA_K, STAGE)
            sSFA: cute.struct.Align[
                cute.struct.MemRange[
                    self.sf_dtype, cute.cosize(self.sfa_smem_layout_staged)
                ],
                self.buffer_align_bytes,
            ]
            # (MMA, MMA_N, MMA_K, STAGE)
            sSFB: cute.struct.Align[
                cute.struct.MemRange[
                    self.sf_dtype, cute.cosize(self.sfb_smem_layout_staged)
                ],
                self.buffer_align_bytes,
            ]

        self.shared_storage = SharedStorage

        # Launch the kernel synchronously
        self.kernel(
            tiled_mma,
            tiled_mma_sfb,
            tma_atom_a,
            tma_tensor_a,
            tma_atom_b,
            tma_tensor_b,
            tma_atom_sfa,
            tma_tensor_sfa,
            tma_atom_sfb,
            tma_tensor_sfb,
            tma_atom_c,
            tma_tensor_c,
            self.cluster_layout_vmnk,
            self.cluster_layout_sfb_vmnk,
            self.a_smem_layout_staged,
            self.b_smem_layout_staged,
            self.sfa_smem_layout_staged,
            self.sfb_smem_layout_staged,
            self.c_smem_layout_staged,
            self.epi_tile,
            self.tile_sched_params,
            group_count,
            problem_shape_mnkl,
            strides_abc,
            tensor_address_abc,
            tensor_address_sfasfb,
            tensormap_cute_tensor,
        ).launch(
            grid=grid,
            block=[self.threads_per_cta, 1, 1],
            cluster=(*self.cluster_shape_mn, 1),
            smem=self.shared_storage.size_in_bytes(),
            **{"st" + "ream": st},
            min_blocks_per_mp=1,
        )
        return

    #  GPU device kernel
    @cute.kernel
    def kernel(
        self,
        tiled_mma: cute.TiledMma,
        tiled_mma_sfb: cute.TiledMma,
        tma_atom_a: cute.CopyAtom,
        mA_mkl: cute.Tensor,
        tma_atom_b: cute.CopyAtom,
        mB_nkl: cute.Tensor,
        tma_atom_sfa: cute.CopyAtom,
        mSFA_mkl: cute.Tensor,
        tma_atom_sfb: cute.CopyAtom,
        mSFB_nkl: cute.Tensor,
        tma_atom_c: cute.CopyAtom,
        mC_mnl: cute.Tensor,
        cluster_layout_vmnk: cute.Layout,
        cluster_layout_sfb_vmnk: cute.Layout,
        a_smem_layout_staged: cute.ComposedLayout,
        b_smem_layout_staged: cute.ComposedLayout,
        sfa_smem_layout_staged: cute.Layout,
        sfb_smem_layout_staged: cute.Layout,
        c_smem_layout_staged: Union[cute.Layout, cute.ComposedLayout],
        epi_tile: cute.Tile,
        tile_sched_params: utils.PersistentTileSchedulerParams,
        group_count: cutlass.Constexpr,
        problem_sizes_mnkl: cute.Tensor,
        strides_abc: cute.Tensor,
        ptrs_abc: cute.Tensor,
        ptrs_sfasfb: cute.Tensor,
        tensormaps: cute.Tensor,
    ):
        warp_idx = cute.arch.warp_idx()
        warp_idx = cute.arch.make_warp_uniform(warp_idx)
        if warp_idx == self.tma_warp_id:
            cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_a)
            cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_b)
            cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_sfa)
            cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_sfb)
            cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_c)
            # PTX prefetch base pointers to warm L2/L1 for first tile.
            _ptx_prefetch_global(mA_mkl.iterator)
            _ptx_prefetch_global_l1(mA_mkl.iterator)
            _ptx_prefetch_global(mB_nkl.iterator)
            _ptx_prefetch_global_l1(mB_nkl.iterator)
            _ptx_prefetch_global(mSFA_mkl.iterator)
            _ptx_prefetch_global_l1(mSFA_mkl.iterator)
            _ptx_prefetch_global(mSFB_nkl.iterator)
            _ptx_prefetch_global_l1(mSFB_nkl.iterator)
            _ptx_prefetch_global(mC_mnl.iterator)
            _ptx_prefetch_global_l1(mC_mnl.iterator)

        use_2cta_instrs = cute.size(tiled_mma.thr_id.shape) == 2

        #
        # Setup cta/thread coordinates
        #
        # Coords inside cluster
        bidx, bidy, bidz = cute.arch.block_idx()
        mma_tile_coord_v = bidx % cute.size(tiled_mma.thr_id.shape)
        is_leader_cta = mma_tile_coord_v == 0
        cta_rank_in_cluster = cute.arch.make_warp_uniform(
            cute.arch.block_idx_in_cluster()
        )
        block_in_cluster_coord_vmnk = cluster_layout_vmnk.get_flat_coord(
            cta_rank_in_cluster
        )
        block_in_cluster_coord_sfb_vmnk = cluster_layout_sfb_vmnk.get_flat_coord(
            cta_rank_in_cluster
        )
        # coord inside cta
        tidx, _, _ = cute.arch.thread_idx()

        #
        # Alloc and init: tensormap buffer, a+b full/empty, accumulator full/empty, tensor memory dealloc barrier
        #
        smem = utils.SmemAllocator()
        storage = smem.allocate(self.shared_storage)

        tensormap_smem_ptr = storage.tensormap_buffer.data_ptr()
        tensormap_a_smem_ptr = tensormap_smem_ptr
        tensormap_b_smem_ptr = (
            tensormap_a_smem_ptr
            + Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
        )
        tensormap_sfa_smem_ptr = (
            tensormap_b_smem_ptr
            + Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
        )
        tensormap_sfb_smem_ptr = (
            tensormap_sfa_smem_ptr
            + Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
        )
        tensormap_c_smem_ptr = (
            tensormap_sfb_smem_ptr
            + Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
        )

        tmem_dealloc_mbar_ptr = storage.tmem_dealloc_mbar_ptr
        tmem_holding_buf = storage.tmem_holding_buf

        # Initialize mainloop ab_pipeline (barrier) and states
        ab_pipeline_producer_group = pipeline.CooperativeGroup(pipeline.Agent.Thread)
        num_tma_producer = self.num_mcast_ctas_a + self.num_mcast_ctas_b - 1
        ab_pipeline_consumer_group = pipeline.CooperativeGroup(
            pipeline.Agent.Thread, num_tma_producer
        )
        ab_pipeline = pipeline.PipelineTmaUmma.create(
            barrier_storage=storage.ab_full_mbar_ptr.data_ptr(),
            num_stages=self.num_ab_stage,
            producer_group=ab_pipeline_producer_group,
            consumer_group=ab_pipeline_consumer_group,
            tx_count=self.num_tma_load_bytes,
            cta_layout_vmnk=cluster_layout_vmnk,
        )

        # Initialize acc_pipeline (barrier) and states
        acc_pipeline_producer_group = pipeline.CooperativeGroup(pipeline.Agent.Thread)
        num_acc_consumer_threads = len(self.epilog_warp_id) * (
            2 if use_2cta_instrs else 1
        )
        acc_pipeline_consumer_group = pipeline.CooperativeGroup(
            pipeline.Agent.Thread, num_acc_consumer_threads
        )
        acc_pipeline = pipeline.PipelineUmmaAsync.create(
            barrier_storage=storage.acc_full_mbar_ptr.data_ptr(),
            num_stages=self.num_acc_stage,
            producer_group=acc_pipeline_producer_group,
            consumer_group=acc_pipeline_consumer_group,
            cta_layout_vmnk=cluster_layout_vmnk,
        )

        # Tensor memory dealloc barrier init
        if use_2cta_instrs:
            if warp_idx == self.tma_warp_id:
                num_tmem_dealloc_threads = 32
                with cute.arch.elect_one():
                    cute.arch.mbarrier_init(
                        tmem_dealloc_mbar_ptr, num_tmem_dealloc_threads
                    )

        # Cluster arrive after barrier init
        pipeline_init_arrive(cluster_shape_mn=self.cluster_shape_mn, is_relaxed=True)

        #
        # Setup smem tensor A/B/SFA/SFB/C
        #
        sC = storage.sC.get_tensor(
            c_smem_layout_staged.outer, swizzle=c_smem_layout_staged.inner
        )
        # (MMA, MMA_M, MMA_K, STAGE)
        sA = storage.sA.get_tensor(
            a_smem_layout_staged.outer, swizzle=a_smem_layout_staged.inner
        )
        # (MMA, MMA_N, MMA_K, STAGE)
        sB = storage.sB.get_tensor(
            b_smem_layout_staged.outer, swizzle=b_smem_layout_staged.inner
        )
        # (MMA, MMA_M, MMA_K, STAGE)
        sSFA = storage.sSFA.get_tensor(sfa_smem_layout_staged)
        # (MMA, MMA_N, MMA_K, STAGE)
        sSFB = storage.sSFB.get_tensor(sfb_smem_layout_staged)

        #
        # Compute multicast mask for A/B/SFA/SFB buffer full
        #
        a_full_mcast_mask = None
        b_full_mcast_mask = None
        sfa_full_mcast_mask = None
        sfb_full_mcast_mask = None
        if cutlass.const_expr(self.is_a_mcast or self.is_b_mcast or use_2cta_instrs):
            a_full_mcast_mask = cpasync.create_tma_multicast_mask(
                cluster_layout_vmnk, block_in_cluster_coord_vmnk, mcast_mode=2
            )
            b_full_mcast_mask = cpasync.create_tma_multicast_mask(
                cluster_layout_vmnk, block_in_cluster_coord_vmnk, mcast_mode=1
            )
            sfa_full_mcast_mask = cpasync.create_tma_multicast_mask(
                cluster_layout_vmnk, block_in_cluster_coord_vmnk, mcast_mode=2
            )
            sfb_full_mcast_mask = cpasync.create_tma_multicast_mask(
                cluster_layout_sfb_vmnk, block_in_cluster_coord_sfb_vmnk, mcast_mode=1
            )

        #
        # Local_tile partition global tensors
        #
        # (bM, bK, RestM, RestK, RestL)
        gA_mkl = cute.local_tile(
            mA_mkl, cute.slice_(self.mma_tiler, (None, 0, None)), (None, None, None)
        )
        # (bN, bK, RestN, RestK, RestL)
        gB_nkl = cute.local_tile(
            mB_nkl, cute.slice_(self.mma_tiler, (0, None, None)), (None, None, None)
        )
        # (bM, bK, RestM, RestK, RestL)
        gSFA_mkl = cute.local_tile(
            mSFA_mkl, cute.slice_(self.mma_tiler, (None, 0, None)), (None, None, None)
        )
        # (bN, bK, RestN, RestK, RestL)
        gSFB_nkl = cute.local_tile(
            mSFB_nkl, cute.slice_(self.mma_tiler, (0, None, None)), (None, None, None)
        )
        # (bM, bN, RestM, RestN, RestL)
        gC_mnl = cute.local_tile(
            mC_mnl, cute.slice_(self.mma_tiler, (None, None, 0)), (None, None, None)
        )

        #
        # Partition global tensor for TiledMMA_A/B/C
        #
        thr_mma = tiled_mma.get_slice(mma_tile_coord_v)
        thr_mma_sfb = tiled_mma_sfb.get_slice(mma_tile_coord_v)
        # (MMA, MMA_M, MMA_K, RestM, RestK, RestL)
        tCgA = thr_mma.partition_A(gA_mkl)
        # (MMA, MMA_N, MMA_K, RestN, RestK, RestL)
        tCgB = thr_mma.partition_B(gB_nkl)
        # (MMA, MMA_M, MMA_K, RestM, RestK, RestL)
        tCgSFA = thr_mma.partition_A(gSFA_mkl)
        # (MMA, MMA_N, MMA_K, RestN, RestK, RestL)
        tCgSFB = thr_mma_sfb.partition_B(gSFB_nkl)
        # (MMA, MMA_M, MMA_N, RestM, RestN, RestL)
        tCgC = thr_mma.partition_C(gC_mnl)

        #
        # Partition global/shared tensor for TMA load A/B
        #
        # TMA load A partition_S/D
        a_cta_layout = cute.make_layout(
            cute.slice_(cluster_layout_vmnk, (0, 0, None, 0)).shape
        )
        # ((atom_v, rest_v), STAGE)
        # ((atom_v, rest_v), RestM, RestK, RestL)
        tAsA, tAgA = cpasync.tma_partition(
            tma_atom_a,
            block_in_cluster_coord_vmnk[2],
            a_cta_layout,
            cute.group_modes(sA, 0, 3),
            cute.group_modes(tCgA, 0, 3),
        )
        # TMA load B partition_S/D
        b_cta_layout = cute.make_layout(
            cute.slice_(cluster_layout_vmnk, (0, None, 0, 0)).shape
        )
        # ((atom_v, rest_v), STAGE)
        # ((atom_v, rest_v), RestN, RestK, RestL)
        tBsB, tBgB = cpasync.tma_partition(
            tma_atom_b,
            block_in_cluster_coord_vmnk[1],
            b_cta_layout,
            cute.group_modes(sB, 0, 3),
            cute.group_modes(tCgB, 0, 3),
        )

        #  TMA Load SFA partition_S/D
        sfa_cta_layout = a_cta_layout
        # ((atom_v, rest_v), STAGE)
        # ((atom_v, rest_v), RestM, RestK, RestL)
        tAsSFA, tAgSFA = cute.nvgpu.cpasync.tma_partition(
            tma_atom_sfa,
            block_in_cluster_coord_vmnk[2],
            sfa_cta_layout,
            cute.group_modes(sSFA, 0, 3),
            cute.group_modes(tCgSFA, 0, 3),
        )
        tAsSFA = cute.filter_zeros(tAsSFA)
        tAgSFA = cute.filter_zeros(tAgSFA)

        # TMA Load SFB partition_S/D
        sfb_cta_layout = cute.make_layout(
            cute.slice_(cluster_layout_sfb_vmnk, (0, None, 0, 0)).shape
        )
        # ((atom_v, rest_v), STAGE)
        # ((atom_v, rest_v), RestN, RestK, RestL)
        tBsSFB, tBgSFB = cute.nvgpu.cpasync.tma_partition(
            tma_atom_sfb,
            block_in_cluster_coord_sfb_vmnk[1],
            sfb_cta_layout,
            cute.group_modes(sSFB, 0, 3),
            cute.group_modes(tCgSFB, 0, 3),
        )
        tBsSFB = cute.filter_zeros(tBsSFB)
        tBgSFB = cute.filter_zeros(tBgSFB)

        #
        # Partition shared/tensor memory tensor for TiledMMA_A/B/C
        #
        # (MMA, MMA_M, MMA_K, STAGE)
        tCrA = tiled_mma.make_fragment_A(sA)
        # (MMA, MMA_N, MMA_K, STAGE)
        tCrB = tiled_mma.make_fragment_B(sB)
        # (MMA, MMA_M, MMA_N)
        acc_shape = tiled_mma.partition_shape_C(self.mma_tiler[:2])
        # (MMA, MMA_M, MMA_N, STAGE)
        tCtAcc_fake = tiled_mma.make_fragment_C(
            cute.append(acc_shape, self.num_acc_stage)
        )

        #
        # Cluster wait before tensor memory alloc
        #
        pipeline_init_wait(cluster_shape_mn=self.cluster_shape_mn)

        #
        # Get tensormap buffer address
        #
        grid_dim = cute.arch.grid_dim()
        tensormap_workspace_idx = (
            bidz * grid_dim[1] * grid_dim[0] + bidy * grid_dim[0] + bidx
        )

        tensormap_manager = utils.TensorMapManager(
            utils.TensorMapUpdateMode.SMEM,
            Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap,
        )
        tensormap_a_gmem_ptr = tensormap_manager.get_tensormap_ptr(
            tensormaps[(tensormap_workspace_idx, 0, None)].iterator
        )
        tensormap_b_gmem_ptr = tensormap_manager.get_tensormap_ptr(
            tensormaps[(tensormap_workspace_idx, 1, None)].iterator
        )
        tensormap_sfa_gmem_ptr = tensormap_manager.get_tensormap_ptr(
            tensormaps[(tensormap_workspace_idx, 2, None)].iterator
        )
        tensormap_sfb_gmem_ptr = tensormap_manager.get_tensormap_ptr(
            tensormaps[(tensormap_workspace_idx, 3, None)].iterator
        )
        tensormap_c_gmem_ptr = tensormap_manager.get_tensormap_ptr(
            tensormaps[(tensormap_workspace_idx, 4, None)].iterator
        )

        #
        # Specialized TMA load warp
        #
        if warp_idx == self.tma_warp_id:
            #
            # Persistent tile scheduling loop
            #
            tile_sched = utils.StaticPersistentTileScheduler.create(
                tile_sched_params, cute.arch.block_idx(), grid_dim
            )
            # grouped gemm tile scheduler helper will compute the group index for the tile we're working on
            group_gemm_ts_helper = utils.GroupedGemmTileSchedulerHelper(
                group_count,
                tile_sched_params,
                self.cluster_tile_shape_mnk,
                utils.create_initial_search_state(),
            )
            tensormap_init_done = cutlass.Boolean(False)
            # group index of last tile
            last_group_idx = cutlass.Int32(-1)

            work_tile = tile_sched.initial_work_tile_info()

            ab_producer_state = pipeline.make_pipeline_state(
                pipeline.PipelineUserType.Producer, self.num_ab_stage
            )

            while work_tile.is_valid_tile:
                cur_tile_coord = work_tile.tile_idx
                grouped_gemm_cta_tile_info = group_gemm_ts_helper.delinearize_z(
                    cur_tile_coord,
                    problem_sizes_mnkl,
                )
                cur_k_tile_cnt = grouped_gemm_cta_tile_info.cta_tile_count_k
                cur_group_idx = grouped_gemm_cta_tile_info.group_idx
                is_group_changed = cur_group_idx != last_group_idx
                # skip tensormap update if we're working on the same group
                if is_group_changed:
                    real_tensor_a = self.make_tensor_abc_for_tensormap_update(
                        cur_group_idx,
                        self.a_dtype,
                        (
                            grouped_gemm_cta_tile_info.problem_shape_m,
                            grouped_gemm_cta_tile_info.problem_shape_n,
                            grouped_gemm_cta_tile_info.problem_shape_k,
                        ),
                        strides_abc,
                        ptrs_abc,
                        0,  # 0 for tensor A
                    )
                    real_tensor_b = self.make_tensor_abc_for_tensormap_update(
                        cur_group_idx,
                        self.b_dtype,
                        (
                            grouped_gemm_cta_tile_info.problem_shape_m,
                            grouped_gemm_cta_tile_info.problem_shape_n,
                            grouped_gemm_cta_tile_info.problem_shape_k,
                        ),
                        strides_abc,
                        ptrs_abc,
                        1,  # 1 for tensor B
                    )
                    real_tensor_sfa = self.make_tensor_sfasfb_for_tensormap_update(
                        cur_group_idx,
                        self.sf_dtype,
                        (
                            grouped_gemm_cta_tile_info.problem_shape_m,
                            grouped_gemm_cta_tile_info.problem_shape_n,
                            grouped_gemm_cta_tile_info.problem_shape_k,
                        ),
                        ptrs_sfasfb,
                        0,  # 0 for tensor SFA
                    )
                    real_tensor_sfb = self.make_tensor_sfasfb_for_tensormap_update(
                        cur_group_idx,
                        self.sf_dtype,
                        (
                            grouped_gemm_cta_tile_info.problem_shape_m,
                            grouped_gemm_cta_tile_info.problem_shape_n,
                            grouped_gemm_cta_tile_info.problem_shape_k,
                        ),
                        ptrs_sfasfb,
                        1,  # 1 for tensor SFB
                    )
                    if tensormap_init_done == False:
                        # wait tensormap initialization complete
                        self.tensormap_ab_init_barrier.arrive_and_wait()
                        tensormap_init_done = True

                    if hasattr(real_tensor_a, "iterator"):
                        _ptx_prefetch_global(real_tensor_a.iterator)
                        _ptx_prefetch_global_l1(real_tensor_a.iterator)
                        _ptx_prefetchu_l1(real_tensor_a.iterator)
                    if hasattr(real_tensor_b, "iterator"):
                        _ptx_prefetch_global(real_tensor_b.iterator)
                        _ptx_prefetch_global_l1(real_tensor_b.iterator)
                        _ptx_prefetchu_l1(real_tensor_b.iterator)
                    if hasattr(real_tensor_sfa, "iterator"):
                        _ptx_prefetch_global(real_tensor_sfa.iterator)
                        _ptx_prefetch_global_l1(real_tensor_sfa.iterator)
                        _ptx_prefetchu_l1(real_tensor_sfa.iterator)
                    if hasattr(real_tensor_sfb, "iterator"):
                        _ptx_prefetch_global(real_tensor_sfb.iterator)
                        _ptx_prefetch_global_l1(real_tensor_sfb.iterator)
                        _ptx_prefetchu_l1(real_tensor_sfb.iterator)
                    _ptx_prefetch_global(tensormap_a_gmem_ptr)
                    _ptx_prefetch_global_l1(tensormap_a_gmem_ptr)
                    _ptx_prefetchu_l1(tensormap_a_gmem_ptr)
                    _ptx_prefetch_global(tensormap_b_gmem_ptr)
                    _ptx_prefetch_global_l1(tensormap_b_gmem_ptr)
                    _ptx_prefetchu_l1(tensormap_b_gmem_ptr)
                    _ptx_prefetch_global(tensormap_sfa_gmem_ptr)
                    _ptx_prefetch_global_l1(tensormap_sfa_gmem_ptr)
                    _ptx_prefetchu_l1(tensormap_sfa_gmem_ptr)
                    _ptx_prefetch_global(tensormap_sfb_gmem_ptr)
                    _ptx_prefetch_global_l1(tensormap_sfb_gmem_ptr)
                    _ptx_prefetchu_l1(tensormap_sfb_gmem_ptr)

                    tensormap_manager.update_tensormap(
                        (
                            real_tensor_a,
                            real_tensor_b,
                            real_tensor_sfa,
                            real_tensor_sfb,
                        ),
                        (tma_atom_a, tma_atom_b, tma_atom_sfa, tma_atom_sfb),
                        (
                            tensormap_a_gmem_ptr,
                            tensormap_b_gmem_ptr,
                            tensormap_sfa_gmem_ptr,
                            tensormap_sfb_gmem_ptr,
                        ),
                        self.tma_warp_id,
                        (
                            tensormap_a_smem_ptr,
                            tensormap_b_smem_ptr,
                            tensormap_sfa_smem_ptr,
                            tensormap_sfb_smem_ptr,
                        ),
                    )

                mma_tile_coord_mnl = (
                    grouped_gemm_cta_tile_info.cta_tile_idx_m
                    // cute.size(tiled_mma.thr_id.shape),
                    grouped_gemm_cta_tile_info.cta_tile_idx_n,
                    0,
                )

                #
                # Slice to per mma tile index
                #
                # ((atom_v, rest_v), RestK)
                tAgA_slice = tAgA[
                    (None, mma_tile_coord_mnl[0], None, mma_tile_coord_mnl[2])
                ]
                # ((atom_v, rest_v), RestK)
                tBgB_slice = tBgB[
                    (None, mma_tile_coord_mnl[1], None, mma_tile_coord_mnl[2])
                ]

                # ((atom_v, rest_v), RestK)
                tAgSFA_slice = tAgSFA[
                    (None, mma_tile_coord_mnl[0], None, mma_tile_coord_mnl[2])
                ]
                # ((atom_v, rest_v), RestK)
                tBgSFB_slice = tBgSFB[
                    (None, mma_tile_coord_mnl[1], None, mma_tile_coord_mnl[2])
                ]

                tma_desc_a = tensormap_manager.get_tensormap_ptr(
                    tensormap_a_gmem_ptr,
                    cute.AddressSpace.generic,
                )
                tma_desc_b = tensormap_manager.get_tensormap_ptr(
                    tensormap_b_gmem_ptr,
                    cute.AddressSpace.generic,
                )
                tma_desc_sfa = tensormap_manager.get_tensormap_ptr(
                    tensormap_sfa_gmem_ptr,
                    cute.AddressSpace.generic,
                )
                tma_desc_sfb = tensormap_manager.get_tensormap_ptr(
                    tensormap_sfb_gmem_ptr,
                    cute.AddressSpace.generic,
                )

                if self.prefetch_enabled:
                    for pf_k_tile in cutlass.range(
                        0, min(self.prefetch_dist, cur_k_tile_cnt), unroll=1
                    ):
                        _prefetch_tma(
                            tma_atom_a,
                            tAgA_slice[(None, pf_k_tile)],
                            tma_desc_a,
                        )
                        _prefetch_tma(
                            tma_atom_b,
                            tBgB_slice[(None, pf_k_tile)],
                            tma_desc_b,
                        )
                        _prefetch_tma(
                            tma_atom_sfa,
                            tAgSFA_slice[(None, pf_k_tile)],
                            tma_desc_sfa,
                        )
                        _prefetch_tma(
                            tma_atom_sfb,
                            tBgSFB_slice[(None, pf_k_tile)],
                            tma_desc_sfb,
                        )

                # Peek (try_wait) AB buffer empty for k_tile = prefetch_k_tile_cnt
                ab_producer_state.reset_count()
                peek_ab_empty_status = cutlass.Boolean(1)
                if ab_producer_state.count < cur_k_tile_cnt:
                    peek_ab_empty_status = ab_pipeline.producer_try_acquire(
                        ab_producer_state
                    )

                if is_group_changed:
                    tensormap_manager.fence_tensormap_update(tensormap_a_gmem_ptr)
                    tensormap_manager.fence_tensormap_update(tensormap_b_gmem_ptr)
                    tensormap_manager.fence_tensormap_update(tensormap_sfa_gmem_ptr)
                    tensormap_manager.fence_tensormap_update(tensormap_sfb_gmem_ptr)
                #
                # Tma load loop
                #
                for k_tile in cutlass.range(0, cur_k_tile_cnt, 1, unroll=1):
                    # Conditionally wait for AB buffer empty
                    ab_pipeline.producer_acquire(
                        ab_producer_state, peek_ab_empty_status
                    )

                    # TMA load A/B/SFA/SFB
                    cute.copy(
                        tma_atom_a,
                        tAgA_slice[(None, ab_producer_state.count)],
                        tAsA[(None, ab_producer_state.index)],
                        tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
                        mcast_mask=a_full_mcast_mask,
                        tma_desc_ptr=tma_desc_a,
                    )
                    cute.copy(
                        tma_atom_b,
                        tBgB_slice[(None, ab_producer_state.count)],
                        tBsB[(None, ab_producer_state.index)],
                        tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
                        mcast_mask=b_full_mcast_mask,
                        tma_desc_ptr=tma_desc_b,
                    )
                    cute.copy(
                        tma_atom_sfa,
                        tAgSFA_slice[(None, ab_producer_state.count)],
                        tAsSFA[(None, ab_producer_state.index)],
                        tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
                        mcast_mask=sfa_full_mcast_mask,
                        tma_desc_ptr=tma_desc_sfa,
                    )
                    cute.copy(
                        tma_atom_sfb,
                        tBgSFB_slice[(None, ab_producer_state.count)],
                        tBsSFB[(None, ab_producer_state.index)],
                        tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
                        mcast_mask=sfb_full_mcast_mask,
                        tma_desc_ptr=tma_desc_sfb,
                    )

                    if self.prefetch_enabled:
                        if k_tile < cur_k_tile_cnt - self.prefetch_dist:
                            future_k_tile = ab_producer_state.count + self.prefetch_dist
                            _prefetch_tma(
                                tma_atom_a,
                                tAgA_slice[(None, future_k_tile)],
                                tma_desc_a,
                            )
                            _prefetch_tma(
                                tma_atom_b,
                                tBgB_slice[(None, future_k_tile)],
                                tma_desc_b,
                            )
                            _prefetch_tma(
                                tma_atom_sfa,
                                tAgSFA_slice[(None, future_k_tile)],
                                tma_desc_sfa,
                            )
                            _prefetch_tma(
                                tma_atom_sfb,
                                tBgSFB_slice[(None, future_k_tile)],
                                tma_desc_sfb,
                            )

                    # Peek (try_wait) AB buffer empty for k_tile = prefetch_k_tile_cnt + k_tile + 1
                    ab_producer_state.advance()
                    peek_ab_empty_status = cutlass.Boolean(1)
                    if ab_producer_state.count < cur_k_tile_cnt:
                        peek_ab_empty_status = ab_pipeline.producer_try_acquire(
                            ab_producer_state
                        )

                #
                # Advance to next tile
                #
                tile_sched.advance_to_next_work()
                work_tile = tile_sched.get_current_work()
                last_group_idx = cur_group_idx

            #
            # Wait A/B buffer empty
            #
            ab_pipeline.producer_tail(ab_producer_state)

        #
        # Specialized MMA warp
        #
        if warp_idx == self.mma_warp_id:
            #
            # Initialize tensormaps for A, B, SFA and SFB
            #
            tensormap_manager.init_tensormap_from_atom(
                tma_atom_a, tensormap_a_smem_ptr, self.mma_warp_id
            )
            tensormap_manager.init_tensormap_from_atom(
                tma_atom_b, tensormap_b_smem_ptr, self.mma_warp_id
            )
            tensormap_manager.init_tensormap_from_atom(
                tma_atom_sfa, tensormap_sfa_smem_ptr, self.mma_warp_id
            )
            tensormap_manager.init_tensormap_from_atom(
                tma_atom_sfb, tensormap_sfb_smem_ptr, self.mma_warp_id
            )
            # indicate tensormap initialization has finished
            self.tensormap_ab_init_barrier.arrive_and_wait()

            #
            # Bar sync for retrieve tensor memory ptr from shared mem
            #
            self.tmem_alloc_barrier.arrive_and_wait()

            #
            # Retrieving tensor memory ptr and make accumulator/SFA/SFB tensor
            #
            # Make accumulator tmem tensor
            acc_tmem_ptr = cute.arch.retrieve_tmem_ptr(
                self.acc_dtype,
                alignment=16,
                ptr_to_buffer_holding_addr=tmem_holding_buf,
            )
            # (MMA, MMA_M, MMA_N, STAGE)
            tCtAcc_base = cute.make_tensor(acc_tmem_ptr, tCtAcc_fake.layout)

            # Make SFA tmem tensor
            sfa_tmem_ptr = cute.recast_ptr(
                acc_tmem_ptr + tcgen05.find_tmem_tensor_col_offset(tCtAcc_base),
                dtype=self.sf_dtype,
            )
            # (MMA, MMA_M, MMA_K)
            tCtSFA_layout = blockscaled_utils.make_tmem_layout_sfa(
                tiled_mma,
                self.mma_tiler,
                self.sf_vec_size,
                cute.slice_(sfa_smem_layout_staged, (None, None, None, 0)),
            )
            tCtSFA = cute.make_tensor(sfa_tmem_ptr, tCtSFA_layout)

            # Make SFB tmem tensor
            sfb_tmem_ptr = cute.recast_ptr(
                acc_tmem_ptr
                + tcgen05.find_tmem_tensor_col_offset(tCtAcc_base)
                + tcgen05.find_tmem_tensor_col_offset(tCtSFA),
                dtype=self.sf_dtype,
            )
            # (MMA, MMA_N, MMA_K)
            tCtSFB_layout = blockscaled_utils.make_tmem_layout_sfb(
                tiled_mma,
                self.mma_tiler,
                self.sf_vec_size,
                cute.slice_(sfb_smem_layout_staged, (None, None, None, 0)),
            )
            tCtSFB = cute.make_tensor(sfb_tmem_ptr, tCtSFB_layout)
            #
            # Partition for S2T copy of SFA/SFB
            #
            tiled_copy_s2t_sfa, tCsSFA_compact_s2t, tCtSFA_compact_s2t = (
                self.mainloop_s2t_copy_and_partition(sSFA, tCtSFA)
            )
            tiled_copy_s2t_sfb, tCsSFB_compact_s2t, tCtSFB_compact_s2t = (
                self.mainloop_s2t_copy_and_partition(sSFB, tCtSFB)
            )

            #
            # Persistent tile scheduling loop
            #
            tile_sched = utils.StaticPersistentTileScheduler.create(
                tile_sched_params, cute.arch.block_idx(), grid_dim
            )
            # grouped gemm tile scheduler helper will compute the group index for the tile we're working on
            group_gemm_ts_helper = utils.GroupedGemmTileSchedulerHelper(
                group_count,
                tile_sched_params,
                self.cluster_tile_shape_mnk,
                utils.create_initial_search_state(),
            )

            work_tile = tile_sched.initial_work_tile_info()
            ab_consumer_state = pipeline.make_pipeline_state(
                pipeline.PipelineUserType.Consumer, self.num_ab_stage
            )
            acc_producer_state = pipeline.make_pipeline_state(
                pipeline.PipelineUserType.Producer, self.num_acc_stage
            )
            while work_tile.is_valid_tile:
                cur_tile_coord = work_tile.tile_idx
                # MMA warp is only interested in number of tiles along K dimension
                (
                    cur_k_tile_cnt,
                    cur_group_idx,
                ) = group_gemm_ts_helper.search_cluster_tile_count_k(
                    cur_tile_coord,
                    problem_sizes_mnkl,
                )

                # (MMA, MMA_M, MMA_N)
                tCtAcc = tCtAcc_base[(None, None, None, acc_producer_state.index)]

                # Peek (try_wait) AB buffer full for k_tile = 0
                ab_consumer_state.reset_count()
                peek_ab_full_status = cutlass.Boolean(1)
                if ab_consumer_state.count < cur_k_tile_cnt and is_leader_cta:
                    peek_ab_full_status = ab_pipeline.consumer_try_wait(
                        ab_consumer_state
                    )

                #
                # Wait for accumulator buffer empty
                #
                if is_leader_cta:
                    acc_pipeline.producer_acquire(acc_producer_state)

                #
                # Reset the ACCUMULATE field for each tile
                #
                tiled_mma.set(tcgen05.Field.ACCUMULATE, False)

                #
                # Mma mainloop
                #
                for k_tile in range(cur_k_tile_cnt):
                    if is_leader_cta:
                        # Conditionally wait for AB buffer full
                        ab_pipeline.consumer_wait(
                            ab_consumer_state, peek_ab_full_status
                        )

                        #  Copy SFA/SFB from smem to tmem
                        s2t_stage_coord = (
                            None,
                            None,
                            None,
                            None,
                            ab_consumer_state.index,
                        )
                        tCsSFA_compact_s2t_staged = tCsSFA_compact_s2t[s2t_stage_coord]
                        tCsSFB_compact_s2t_staged = tCsSFB_compact_s2t[s2t_stage_coord]
                        cute.copy(
                            tiled_copy_s2t_sfa,
                            tCsSFA_compact_s2t_staged,
                            tCtSFA_compact_s2t,
                        )
                        cute.copy(
                            tiled_copy_s2t_sfb,
                            tCsSFB_compact_s2t_staged,
                            tCtSFB_compact_s2t,
                        )

                        # tCtAcc += tCrA * tCrSFA * tCrB * tCrSFB
                        num_kblocks = cute.size(tCrA, mode=[2])
                        for kblock_idx in cutlass.range(num_kblocks, unroll_full=True):
                            kblock_coord = (
                                None,
                                None,
                                kblock_idx,
                                ab_consumer_state.index,
                            )

                            # Set SFA/SFB tensor to tiled_mma
                            sf_kblock_coord = (None, None, kblock_idx)
                            tiled_mma.set(
                                tcgen05.Field.SFA,
                                tCtSFA[sf_kblock_coord].iterator,
                            )
                            tiled_mma.set(
                                tcgen05.Field.SFB,
                                tCtSFB[sf_kblock_coord].iterator,
                            )

                            cute.gemm(
                                tiled_mma,
                                tCtAcc,
                                tCrA[kblock_coord],
                                tCrB[kblock_coord],
                                tCtAcc,
                            )

                            # Enable accumulate on tCtAcc after first kblock
                            tiled_mma.set(tcgen05.Field.ACCUMULATE, True)

                        # Async arrive AB buffer empty
                        ab_pipeline.consumer_release(ab_consumer_state)

                    # Peek (try_wait) AB buffer full for k_tile = k_tile + 1
                    ab_consumer_state.advance()
                    peek_ab_full_status = cutlass.Boolean(1)
                    if ab_consumer_state.count < cur_k_tile_cnt:
                        if is_leader_cta:
                            peek_ab_full_status = ab_pipeline.consumer_try_wait(
                                ab_consumer_state
                            )

                #
                # Async arrive accumulator buffer full
                #
                if is_leader_cta:
                    acc_pipeline.producer_commit(acc_producer_state)
                acc_producer_state.advance()

                #
                # Advance to next tile
                #
                tile_sched.advance_to_next_work()
                work_tile = tile_sched.get_current_work()

            #
            # Wait for accumulator buffer empty
            #
            acc_pipeline.producer_tail(acc_producer_state)

        #
        # Specialized epilogue warps
        #
        if warp_idx < self.mma_warp_id:
            if cutlass.const_expr(self.use_tma_store):
                # initialize tensormap for C
                tensormap_manager.init_tensormap_from_atom(
                    tma_atom_c,
                    tensormap_c_smem_ptr,
                    self.epilog_warp_id[0],
                )
            #
            # Alloc tensor memory buffer
            #
            if warp_idx == self.epilog_warp_id[0]:
                cute.arch.alloc_tmem(
                    self.num_tmem_alloc_cols,
                    tmem_holding_buf,
                    is_two_cta=use_2cta_instrs,
                )

            #
            # Bar sync for retrieve tensor memory ptr from shared memory
            #
            self.tmem_alloc_barrier.arrive_and_wait()

            #
            # Retrieving tensor memory ptr and make accumulator tensor
            #
            acc_tmem_ptr = cute.arch.retrieve_tmem_ptr(
                self.acc_dtype,
                alignment=16,
                ptr_to_buffer_holding_addr=tmem_holding_buf,
            )
            # (MMA, MMA_M, MMA_N, STAGE)
            tCtAcc_base = cute.make_tensor(acc_tmem_ptr, tCtAcc_fake.layout)

            ### Start from here
            #
            # Partition for epilogue
            #
            epi_tidx = tidx
            if cutlass.const_expr(self.use_tma_store):
                tiled_copy_t2r, tTR_tAcc_base, tTR_rAcc = (
                    self.epilog_tmem_copy_and_partition(
                        epi_tidx, tCtAcc_base, tCgC, epi_tile, use_2cta_instrs
                    )
                )
                tTR_rC = cute.make_rmem_tensor(tTR_rAcc.shape, self.c_dtype)
                tiled_copy_r2s, tRS_rC, tRS_sC = self.epilog_smem_copy_and_partition(
                    tiled_copy_t2r, tTR_rC, epi_tidx, sC
                )
                tma_atom_c, bSG_sC, bSG_gC_partitioned = (
                    self.epilog_gmem_copy_and_partition(
                        epi_tidx, tma_atom_c, tCgC, epi_tile, sC
                    )
                )
            else:
                tCtAcc_simt = self._transform_partitioned_tensor_layout(tCtAcc_base)
                tCgC_simt = self._transform_partitioned_tensor_layout(tCgC)
                (
                    tiled_copy_t2r,
                    tTR_tAcc_base,
                    tTR_rAcc,
                ) = self.epilog_tmem_copy_and_partition_simt(
                    epi_tidx, tCtAcc_simt, tCgC_simt, epi_tile, use_2cta_instrs
                )
                tTR_rC = cute.make_rmem_tensor(tTR_rAcc.shape, self.c_dtype)
                gC_epi_simt = cute.flat_divide(tCgC_simt, epi_tile)
                thr_copy_t2r = tiled_copy_t2r.get_slice(epi_tidx)
                simt_atom_vec = cute.make_copy_atom(
                    cute.nvgpu.CopyUniversalOp(),
                    self.c_dtype,
                )

            #
            # Persistent tile scheduling loop
            #
            tile_sched = utils.StaticPersistentTileScheduler.create(
                tile_sched_params, cute.arch.block_idx(), grid_dim
            )
            # grouped gemm tile scheduler helper will compute the group index for the tile we're working on
            group_gemm_ts_helper = utils.GroupedGemmTileSchedulerHelper(
                group_count,
                tile_sched_params,
                self.cluster_tile_shape_mnk,
                utils.create_initial_search_state(),
            )

            work_tile = tile_sched.initial_work_tile_info()

            acc_consumer_state = pipeline.make_pipeline_state(
                pipeline.PipelineUserType.Consumer, self.num_acc_stage
            )

            if cutlass.const_expr(self.use_tma_store):
                # Threads/warps participating in tma store pipeline
                c_producer_group = pipeline.CooperativeGroup(
                    pipeline.Agent.Thread,
                    32 * len(self.epilog_warp_id),
                )
                c_pipeline = pipeline.PipelineTmaStore.create(
                    num_stages=self.num_c_stage,
                    producer_group=c_producer_group,
                )
            if cutlass.const_expr(self.use_tma_store):
                # group index to start searching
                last_group_idx = cutlass.Int32(-1)

                while work_tile.is_valid_tile:
                    cur_tile_coord = work_tile.tile_idx
                    grouped_gemm_cta_tile_info = group_gemm_ts_helper.delinearize_z(
                        cur_tile_coord,
                        problem_sizes_mnkl,
                    )
                    cur_group_idx = grouped_gemm_cta_tile_info.group_idx
                    is_group_changed = cur_group_idx != last_group_idx

                    real_tensor_c = self.make_tensor_abc_for_tensormap_update(
                        cur_group_idx,
                        self.c_dtype,
                        (
                            grouped_gemm_cta_tile_info.problem_shape_m,
                            grouped_gemm_cta_tile_info.problem_shape_n,
                            grouped_gemm_cta_tile_info.problem_shape_k,
                        ),
                        strides_abc,
                        ptrs_abc,
                        2,  # 2 for tensor C
                    )

                    if is_group_changed:
                        if hasattr(real_tensor_c, "iterator"):
                            _ptx_prefetch_global(real_tensor_c.iterator)
                            _ptx_prefetch_global_l1(real_tensor_c.iterator)
                            _ptx_prefetchu_l1(real_tensor_c.iterator)
                        _ptx_prefetch_global(tensormap_c_gmem_ptr)
                        _ptx_prefetch_global_l1(tensormap_c_gmem_ptr)
                        _ptx_prefetchu_l1(tensormap_c_gmem_ptr)
                        tensormap_manager.update_tensormap(
                            ((real_tensor_c),),
                            ((tma_atom_c),),
                            ((tensormap_c_gmem_ptr),),
                            self.epilog_warp_id[0],
                            (tensormap_c_smem_ptr,),
                        )

                    mma_tile_coord_mnl = (
                        grouped_gemm_cta_tile_info.cta_tile_idx_m
                        // cute.size(tiled_mma.thr_id.shape),
                        grouped_gemm_cta_tile_info.cta_tile_idx_n,
                        0,
                    )

                    # ((ATOM_V, REST_V), EPI_M, EPI_N)
                    bSG_gC = bSG_gC_partitioned[
                        (
                            None,
                            None,
                            None,
                            *mma_tile_coord_mnl,
                        )
                    ]

                    # Set tensor memory buffer for current tile
                    # (T2R, T2R_M, T2R_N, EPI_M, EPI_M)
                    tTR_tAcc = tTR_tAcc_base[
                        (None, None, None, None, None, acc_consumer_state.index)
                    ]

                    #
                    # Wait for accumulator buffer full
                    #
                    acc_pipeline.consumer_wait(acc_consumer_state)

                    tTR_tAcc = cute.group_modes(tTR_tAcc, 3, cute.rank(tTR_tAcc))
                    #
                    # Store accumulator to global memory in subtiles
                    #
                    subtile_cnt = cute.size(tTR_tAcc.shape, mode=[3])
                    bSG_gC = cute.group_modes(bSG_gC, 1, cute.rank(bSG_gC))
                    if is_group_changed:
                        if warp_idx == self.epilog_warp_id[0]:
                            tensormap_manager.fence_tensormap_update(
                                tensormap_c_gmem_ptr
                            )
                    num_prev_subtiles = tile_sched.num_tiles_executed * subtile_cnt
                    for subtile_idx in range(subtile_cnt):
                        #
                        # Load accumulator from tensor memory buffer to register
                        #
                        tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]
                        cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)
                        # Early-release the accumulator buffer once the last subtile has been
                        # transferred to registers, so MMA can reuse TMEM while we finish stores.
                        if subtile_idx == subtile_cnt - 1:
                            with cute.arch.elect_one():
                                acc_pipeline.consumer_release(acc_consumer_state)
                            acc_consumer_state.advance()

                        #
                        # Convert to C type
                        #
                        acc_vec = tiled_copy_r2s.retile(tTR_rAcc).load()
                        tRS_rC.store(acc_vec.to(self.c_dtype))

                        #
                        # Store C to shared memory
                        #
                        c_buffer = (num_prev_subtiles + subtile_idx) % self.num_c_stage
                        cute.copy(
                            tiled_copy_r2s,
                            tRS_rC,
                            tRS_sC[(None, None, None, c_buffer)],
                        )
                        # Fence and barrier to make sure shared memory store is visible to TMA store
                        cute.arch.fence_proxy("async.shared", space="cta")
                        self.epilog_sync_barrier.arrive_and_wait()

                        #
                        # TMA store C to global memory
                        #
                        if warp_idx == self.epilog_warp_id[0]:
                            cute.copy(
                                tma_atom_c,
                                bSG_sC[(None, c_buffer)],
                                bSG_gC[(None, subtile_idx)],
                                tma_desc_ptr=tensormap_manager.get_tensormap_ptr(
                                    tensormap_c_gmem_ptr,
                                    cute.AddressSpace.generic,
                                ),
                            )
                            # Fence and barrier to make sure shared memory store is visible to TMA store
                            c_pipeline.producer_commit()
                            c_pipeline.producer_acquire()
                        self.epilog_sync_barrier.arrive_and_wait()

                    #
                    # Advance to next tile
                    #
                    tile_sched.advance_to_next_work()
                    work_tile = tile_sched.get_current_work()
                    last_group_idx = cur_group_idx

            else:
                # group index to start searching
                last_group_idx = cutlass.Int32(-1)

                while work_tile.is_valid_tile:
                    cur_tile_coord = work_tile.tile_idx
                    grouped_gemm_cta_tile_info = group_gemm_ts_helper.delinearize_z(
                        cur_tile_coord,
                        problem_sizes_mnkl,
                    )
                    cur_group_idx = grouped_gemm_cta_tile_info.group_idx

                    real_tensor_c = self.make_tensor_abc_for_tensormap_update(
                        cur_group_idx,
                        self.c_dtype,
                        (
                            grouped_gemm_cta_tile_info.problem_shape_m,
                            grouped_gemm_cta_tile_info.problem_shape_n,
                            grouped_gemm_cta_tile_info.problem_shape_k,
                        ),
                        strides_abc,
                        ptrs_abc,
                        2,  # 2 for tensor C
                    )
                    if cur_group_idx != last_group_idx:
                        if hasattr(real_tensor_c, "iterator"):
                            _ptx_prefetch_global(real_tensor_c.iterator)
                            _ptx_prefetch_global_l1(real_tensor_c.iterator)
                            _ptx_prefetchu_l1(real_tensor_c.iterator)

                    mma_tile_coord_mnl = (
                        grouped_gemm_cta_tile_info.cta_tile_idx_m
                        // cute.size(tiled_mma.thr_id.shape),
                        grouped_gemm_cta_tile_info.cta_tile_idx_n,
                        0,
                    )
                    tile_origin_m = mma_tile_coord_mnl[0] * self.mma_tiler[0]
                    tile_origin_n = mma_tile_coord_mnl[1] * self.mma_tiler[1]
                    residue_m = (
                        grouped_gemm_cta_tile_info.problem_shape_m - tile_origin_m
                    )
                    residue_n = (
                        grouped_gemm_cta_tile_info.problem_shape_n - tile_origin_n
                    )
                    full_tile = (residue_m >= self.mma_tiler[0]) & (
                        residue_n >= self.mma_tiler[1]
                    )

                    gC_mnl_tile = cute.local_tile(
                        real_tensor_c,
                        cute.slice_(self.mma_tiler, (None, None, 0)),
                        mma_tile_coord_mnl,
                    )
                    tCgC_tile = thr_mma.partition_C(gC_mnl_tile)
                    tCgC_tile = self._transform_partitioned_tensor_layout(tCgC_tile)
                    tTR_gC = self.epilog_gmem_copy_and_partition_simt(
                        epi_tidx, tiled_copy_t2r, tCgC_tile, epi_tile
                    )

                    # Set tensor memory buffer for current tile
                    # (T2R, T2R_M, T2R_N, EPI_M, EPI_M)
                    tTR_tAcc = tTR_tAcc_base[
                        (None, None, None, None, None, acc_consumer_state.index)
                    ]

                    #
                    # Wait for accumulator buffer full
                    #
                    acc_pipeline.consumer_wait(acc_consumer_state)

                    tTR_tAcc = cute.group_modes(tTR_tAcc, 3, cute.rank(tTR_tAcc))
                    tTR_gC = cute.group_modes(tTR_gC, 3, cute.rank(tTR_gC))
                    #
                    # Store accumulator to global memory in subtiles
                    #
                    subtile_cnt = cute.size(tTR_tAcc.shape, mode=[3])
                    if full_tile:
                        for subtile_idx in range(subtile_cnt):
                            #
                            # Load accumulator from tensor memory buffer to register
                            #
                            tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]
                            cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)
                            if subtile_idx == subtile_cnt - 1:
                                with cute.arch.elect_one():
                                    acc_pipeline.consumer_release(acc_consumer_state)
                                acc_consumer_state.advance()

                            #
                            # Convert to C type
                            #
                            acc_vec = tTR_rAcc.load()
                            tTR_rC.store(acc_vec.to(self.c_dtype))
                            tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]
                            cute.copy(simt_atom_vec, tTR_rC, tTR_gC_mn)
                    else:
                        cC_mnl = cute.make_identity_tensor(real_tensor_c.shape)
                        cC_mnl_tile = cute.local_tile(
                            cC_mnl,
                            cute.slice_(self.mma_tiler, (None, None, 0)),
                            mma_tile_coord_mnl,
                        )
                        tCcC_tile = thr_mma.partition_C(cC_mnl_tile)
                        tCcC_tile = self._transform_partitioned_tensor_layout(
                            tCcC_tile
                        )
                        tTR_cC = self.epilog_gmem_copy_and_partition_simt(
                            epi_tidx, tiled_copy_t2r, tCcC_tile, epi_tile
                        )
                        tTR_cC = cute.group_modes(tTR_cC, 3, cute.rank(tTR_cC))
                        c_shape = real_tensor_c.shape
                        for subtile_idx in range(subtile_cnt):
                            #
                            # Load accumulator from tensor memory buffer to register
                            #
                            tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]
                            cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)
                            if subtile_idx == subtile_cnt - 1:
                                with cute.arch.elect_one():
                                    acc_pipeline.consumer_release(acc_consumer_state)
                                acc_consumer_state.advance()

                            #
                            # Convert to C type
                            #
                            acc_vec = tTR_rAcc.load()
                            tTR_rC.store(acc_vec.to(self.c_dtype))
                            tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]
                            tTR_cC_mn = tTR_cC[(None, None, None, subtile_idx)]
                            tTR_pC = cute.make_rmem_tensor(
                                tTR_rC.shape, cutlass.Boolean
                            )
                            for i in range(cute.size(tTR_rC.shape)):
                                tTR_pC[i] = cute.elem_less(tTR_cC_mn[i], c_shape)
                            cute.basic_copy_if(tTR_pC, tTR_rC, tTR_gC_mn)

                    #
                    # Advance to next tile
                    #
                    tile_sched.advance_to_next_work()
                    work_tile = tile_sched.get_current_work()
                    last_group_idx = cur_group_idx
                    last_group_idx = cur_group_idx
            #
            # Dealloc the tensor memory buffer
            #
            if warp_idx == self.epilog_warp_id[0]:
                cute.arch.relinquish_tmem_alloc_permit(is_two_cta=use_2cta_instrs)
            self.epilog_sync_barrier.arrive_and_wait()
            if warp_idx == self.epilog_warp_id[0]:
                if use_2cta_instrs:
                    cute.arch.mbarrier_arrive(
                        tmem_dealloc_mbar_ptr, cta_rank_in_cluster ^ 1
                    )
                    cute.arch.mbarrier_wait(tmem_dealloc_mbar_ptr, 0)
                cute.arch.dealloc_tmem(
                    acc_tmem_ptr, self.num_tmem_alloc_cols, is_two_cta=use_2cta_instrs
                )
            #
            # Wait for C store complete
            #
            if cutlass.const_expr(self.use_tma_store):
                c_pipeline.producer_tail()


    @cute.jit
    def make_tensor_abc_for_tensormap_update(
        self,
        group_idx: cutlass.Int32,
        dtype: Type[cutlass.Numeric],
        problem_shape_mnk: tuple[cutlass.Int32, cutlass.Int32, cutlass.Int32],
        strides_abc: cute.Tensor,
        tensor_address_abc: cute.Tensor,
        tensor_index: int,
    ):
        ptr_i64 = tensor_address_abc[(group_idx, tensor_index)]
        if cutlass.const_expr(
            not isclass(dtype) or not issubclass(dtype, cutlass.Numeric)
        ):
            raise TypeError(
                f"dtype must be a type of cutlass.Numeric, got {type(dtype)}"
            )
        align = 16
        if cutlass.const_expr(tensor_index == 2):
            align = self.c_assumed_align
        tensor_gmem_ptr = cute.make_ptr(
            dtype, ptr_i64, cute.AddressSpace.gmem, assumed_align=align
        )

        strides_tensor_gmem = strides_abc[(group_idx, tensor_index, None)]
        strides_tensor_reg = cute.make_rmem_tensor(
            cute.make_layout(2),
            strides_abc.element_type,
        )
        cute.autovec_copy(strides_tensor_gmem, strides_tensor_reg)
        stride_mn = strides_tensor_reg[0]
        stride_k = strides_tensor_reg[1]
        c1 = cutlass.Int32(1)
        c0 = cutlass.Int32(0)

        if cutlass.const_expr(tensor_index == 0):  # tensor A
            m = problem_shape_mnk[0]
            k = problem_shape_mnk[2]
            return cute.make_tensor(
                tensor_gmem_ptr,
                cute.make_layout((m, k, c1), stride=(stride_mn, stride_k, c0)),
            )
        elif cutlass.const_expr(tensor_index == 1):  # tensor B
            n = problem_shape_mnk[1]
            k = problem_shape_mnk[2]
            return cute.make_tensor(
                tensor_gmem_ptr,
                cute.make_layout((n, k, c1), stride=(stride_mn, stride_k, c0)),
            )
        else:  # tensor C
            m = problem_shape_mnk[0]
            n = problem_shape_mnk[1]
            tensor = cute.make_tensor(
                tensor_gmem_ptr,
                cute.make_layout((m, n, c1), stride=(stride_mn, stride_k, c0)),
            )
            leading_dim = 0 if self.c_layout.is_m_major_c() else 1
            tensor.mark_layout_dynamic(leading_dim=leading_dim)
            stride_order = (2, 1, 0) if leading_dim == 0 else (2, 0, 1)
            div = self.c_divisibility if getattr(dtype, "width", 0) == 16 else 16
            tensor.mark_compact_shape_dynamic(
                mode=leading_dim,
                stride_order=stride_order,
                divisibility=div,
            )
            return tensor

    @cute.jit
    def make_tensor_sfasfb_for_tensormap_update(
        self,
        group_idx: cutlass.Int32,
        dtype: Type[cutlass.Numeric],
        problem_shape_mnk: tuple[cutlass.Int32, cutlass.Int32, cutlass.Int32],
        tensor_address_sfasfb: cute.Tensor,
        tensor_index: int,
    ):
        ptr_i64 = tensor_address_sfasfb[(group_idx, tensor_index)]
        if cutlass.const_expr(
            not isclass(dtype) or not issubclass(dtype, cutlass.Numeric)
        ):
            raise TypeError(
                f"dtype must be a type of cutlass.Numeric, got {type(dtype)}"
            )
        tensor_gmem_ptr = cute.make_ptr(
            dtype, ptr_i64, cute.AddressSpace.gmem, assumed_align=16
        )

        c1 = cutlass.Int32(1)
        if cutlass.const_expr(tensor_index == 0):  # tensor SFA
            m = problem_shape_mnk[0]
            k = problem_shape_mnk[2]
            sfa_layout = blockscaled_utils.tile_atom_to_shape_SF(
                (m, k, c1), self.sf_vec_size
            )
            return cute.make_tensor(
                tensor_gmem_ptr,
                sfa_layout,
            )
        else:  # tensor SFB
            n = problem_shape_mnk[1]
            k = problem_shape_mnk[2]
            sfb_layout = blockscaled_utils.tile_atom_to_shape_SF(
                (n, k, c1), self.sf_vec_size
            )
            return cute.make_tensor(
                tensor_gmem_ptr,
                sfb_layout,
            )

    def mainloop_s2t_copy_and_partition(
        self,
        sSF: cute.Tensor,
        tSF: cute.Tensor,
    ) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor, cute.Tensor]:
        # (MMA, MMA_MN, MMA_K, STAGE)
        tCsSF_compact = cute.filter_zeros(sSF)
        # (MMA, MMA_MN, MMA_K)
        tCtSF_compact = cute.filter_zeros(tSF)

        # Make S2T CopyAtom and tiledCopy
        copy_atom_s2t = cute.make_copy_atom(
            tcgen05.Cp4x32x128bOp(self.cta_group),
            self.sf_dtype,
        )
        tiled_copy_s2t = tcgen05.make_s2t_copy(copy_atom_s2t, tCtSF_compact)
        thr_copy_s2t = tiled_copy_s2t.get_slice(0)

        # ((ATOM_V, REST_V), Rest_Tiler, MMA_MN, MMA_K, STAGE)
        tCsSF_compact_s2t_ = thr_copy_s2t.partition_S(tCsSF_compact)
        # ((ATOM_V, REST_V), Rest_Tiler, MMA_MN, MMA_K, STAGE)
        tCsSF_compact_s2t = tcgen05.get_s2t_smem_desc_tensor(
            tiled_copy_s2t, tCsSF_compact_s2t_
        )
        # ((ATOM_V, REST_V), Rest_Tiler, MMA_MN, MMA_K)
        tCtSF_compact_s2t = thr_copy_s2t.partition_D(tCtSF_compact)

        return tiled_copy_s2t, tCsSF_compact_s2t, tCtSF_compact_s2t

    def epilog_tmem_copy_and_partition(
        self,
        tidx: cutlass.Int32,
        tAcc: cute.Tensor,
        gC_mnl: cute.Tensor,
        epi_tile: cute.Tile,
        use_2cta_instrs: Union[cutlass.Boolean, bool],
    ) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor]:
        # Make tiledCopy for tensor memory load
        copy_atom_t2r = sm100_utils.get_tmem_load_op(
            self.cta_tile_shape_mnk,
            self.c_layout,
            self.c_dtype,
            self.acc_dtype,
            epi_tile,
            use_2cta_instrs,
        )
        # (EPI_TILE_M, EPI_TILE_N, EPI_M, EPI_N, STAGE)
        tAcc_epi = cute.flat_divide(
            tAcc[((None, None), 0, 0, None)],
            epi_tile,
        )
        # (EPI_TILE_M, EPI_TILE_N)
        tiled_copy_t2r = tcgen05.make_tmem_copy(
            copy_atom_t2r, tAcc_epi[(None, None, 0, 0, 0)]
        )

        thr_copy_t2r = tiled_copy_t2r.get_slice(tidx)
        # (T2R, T2R_M, T2R_N, EPI_M, EPI_M, STAGE)
        tTR_tAcc = thr_copy_t2r.partition_S(tAcc_epi)

        gC_mnl_epi = cute.flat_divide(
            gC_mnl[((None, None), 0, 0, None, None, None)], epi_tile
        )
        # (T2R, T2R_M, T2R_N, EPI_M, EPI_N, RestM, RestN, RestL)
        tTR_gC = thr_copy_t2r.partition_D(gC_mnl_epi)
        # (T2R, T2R_M, T2R_N)
        tTR_rAcc = cute.make_rmem_tensor(
            tTR_gC[(None, None, None, 0, 0, 0, 0, 0)].shape, self.acc_dtype
        )
        return tiled_copy_t2r, tTR_tAcc, tTR_rAcc

    def epilog_smem_copy_and_partition(
        self,
        tiled_copy_t2r: cute.TiledCopy,
        tTR_rC: cute.Tensor,
        tidx: cutlass.Int32,
        sC: cute.Tensor,
    ) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor]:
        copy_atom_r2s = sm100_utils.get_smem_store_op(
            self.c_layout, self.c_dtype, self.acc_dtype, tiled_copy_t2r
        )
        tiled_copy_r2s = cute.make_tiled_copy_D(copy_atom_r2s, tiled_copy_t2r)
        # (R2S, R2S_M, R2S_N, PIPE_D)
        thr_copy_r2s = tiled_copy_r2s.get_slice(tidx)
        tRS_sC = thr_copy_r2s.partition_D(sC)
        # (R2S, R2S_M, R2S_N)
        tRS_rC = tiled_copy_r2s.retile(tTR_rC)
        return tiled_copy_r2s, tRS_rC, tRS_sC

    def epilog_gmem_copy_and_partition(
        self,
        tidx: cutlass.Int32,
        atom: Union[cute.CopyAtom, cute.TiledCopy],
        gC_mnl: cute.Tensor,
        epi_tile: cute.Tile,
        sC: cute.Tensor,
    ) -> Tuple[cute.CopyAtom, cute.Tensor, cute.Tensor]:
        # (EPI_TILE_M, EPI_TILE_N, EPI_M, EPI_N, RestM, RestN, RestL)
        gC_epi = cute.flat_divide(
            gC_mnl[((None, None), 0, 0, None, None, None)], epi_tile
        )

        tma_atom_c = atom
        sC_for_tma_partition = cute.group_modes(sC, 0, 2)
        gC_for_tma_partition = cute.group_modes(gC_epi, 0, 2)
        # ((ATOM_V, REST_V), EPI_M, EPI_N)
        # ((ATOM_V, REST_V), EPI_M, EPI_N, RestM, RestN, RestL)
        bSG_sC, bSG_gC = cpasync.tma_partition(
            tma_atom_c,
            0,
            cute.make_layout(1),
            sC_for_tma_partition,
            gC_for_tma_partition,
        )
        return tma_atom_c, bSG_sC, bSG_gC

    def epilog_gmem_copy_and_partition_simt(
        self,
        tidx: cutlass.Int32,
        tiled_copy_t2r: cute.TiledCopy,
        gC_mnl: cute.Tensor,
        epi_tile: cute.Tile,
    ) -> cute.Tensor:
        # (EPI_TILE_M, EPI_TILE_N, EPI_M, EPI_N, RestM, RestN, RestL)
        gC_epi = cute.flat_divide(gC_mnl, epi_tile)
        thr_copy_t2r = tiled_copy_t2r.get_slice(tidx)
        # (T2R, T2R_M, T2R_N, EPI_M, EPI_N, RestM, RestN, RestL)
        tTR_gC = thr_copy_t2r.partition_D(gC_epi)
        return tTR_gC

    @staticmethod
    def _transform_partitioned_tensor_layout(tensor: cute.Tensor) -> cute.Tensor:
        layout = tensor.layout
        shape = layout.shape
        stride = layout.stride
        new_shape = ((shape[0][0], shape[1]), (shape[0][1], shape[2]), *shape[3:])
        new_stride = (
            (stride[0][0], stride[1]),
            (stride[0][1], stride[2]),
            *stride[3:],
        )
        new_layout = cute.make_layout(shape=new_shape, stride=new_stride)
        return cute.make_tensor(tensor.iterator, new_layout)

    def epilog_tmem_copy_and_partition_simt(
        self,
        tidx: cutlass.Int32,
        tAcc: cute.Tensor,
        tCgC: cute.Tensor,
        epi_tile: cute.Tile,
        use_2cta_instrs: Union[cutlass.Boolean, bool],
    ) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor]:
        copy_atom_t2r = sm100_utils.get_tmem_load_op(
            self.cta_tile_shape_mnk,
            self.c_layout,
            self.c_dtype,
            self.acc_dtype,
            epi_tile,
            use_2cta_instrs,
        )
        tAcc_epi = cute.flat_divide(tAcc, epi_tile)
        tiled_copy_t2r = tcgen05.make_tmem_copy(
            copy_atom_t2r, tAcc_epi[(None, None, 0, 0, 0)]
        )
        thr_copy_t2r = tiled_copy_t2r.get_slice(tidx)
        tTR_tAcc = thr_copy_t2r.partition_S(tAcc_epi)
        tCgC_epi = cute.flat_divide(tCgC, epi_tile)
        tTR_gC = thr_copy_t2r.partition_D(tCgC_epi)
        tTR_rAcc = cute.make_rmem_tensor(
            tTR_gC[(None, None, None, 0, 0, 0, 0, 0)].shape, self.acc_dtype
        )
        return tiled_copy_t2r, tTR_tAcc, tTR_rAcc

    @staticmethod
    def _compute_stages(
        tiled_mma: cute.TiledMma,
        mma_tiler_mnk: Tuple[int, int, int],
        a_dtype: Type[cutlass.Numeric],
        b_dtype: Type[cutlass.Numeric],
        epi_tile: cute.Tile,
        c_dtype: Type[cutlass.Numeric],
        c_layout: utils.LayoutEnum,
        sf_dtype: Type[cutlass.Numeric],
        sf_vec_size: int,
        smem_capacity: int,
        occupancy: int,
        force_c_stage: int | None = None,
    ) -> Tuple[int, int, int]:
        # ACC stages
        num_acc_stage = 1 if mma_tiler_mnk[1] == 256 else 2

        # Default C stages
        num_c_stage = 2 if force_c_stage is None else max(1, int(force_c_stage))

        # Calculate smem layout and size for one stage of A, B, SFA, SFB and C
        a_smem_layout_stage_one = sm100_utils.make_smem_layout_a(
            tiled_mma,
            mma_tiler_mnk,
            a_dtype,
            1,  # a tmp 1 stage is provided
        )
        b_smem_layout_staged_one = sm100_utils.make_smem_layout_b(
            tiled_mma,
            mma_tiler_mnk,
            b_dtype,
            1,  # a tmp 1 stage is provided
        )
        sfa_smem_layout_staged_one = blockscaled_utils.make_smem_layout_sfa(
            tiled_mma,
            mma_tiler_mnk,
            sf_vec_size,
            1,  # a tmp 1 stage is provided
        )
        sfb_smem_layout_staged_one = blockscaled_utils.make_smem_layout_sfb(
            tiled_mma,
            mma_tiler_mnk,
            sf_vec_size,
            1,  # a tmp 1 stage is provided
        )

        c_smem_layout_staged_one = sm100_utils.make_smem_layout_epi(
            c_dtype,
            c_layout,
            epi_tile,
            1,
        )

        ab_bytes_per_stage = (
            cute.size_in_bytes(a_dtype, a_smem_layout_stage_one)
            + cute.size_in_bytes(b_dtype, b_smem_layout_staged_one)
            + cute.size_in_bytes(sf_dtype, sfa_smem_layout_staged_one)
            + cute.size_in_bytes(sf_dtype, sfb_smem_layout_staged_one)
        )
        mbar_helpers_bytes = 1024
        c_bytes_per_stage = cute.size_in_bytes(c_dtype, c_smem_layout_staged_one)
        c_bytes = c_bytes_per_stage * num_c_stage

        # Calculate A/B/SFA/SFB stages:
        # Start with total smem per CTA (capacity / occupancy)
        # Subtract reserved bytes and initial C stages bytes
        # Divide remaining by bytes needed per A/B/SFA/SFB stage
        num_ab_stage = (
            smem_capacity // occupancy - (mbar_helpers_bytes + c_bytes)
        ) // ab_bytes_per_stage
        if num_ab_stage < 1:
            num_ab_stage = 1

        # Refine epilogue stages:
        # Calculate remaining smem after allocating for A/B/SFA/SFB stages and reserved bytes
        # Add remaining unused smem to epilogue
        if force_c_stage is None:
            num_c_stage += (
                smem_capacity
                - occupancy * ab_bytes_per_stage * num_ab_stage
                - occupancy * (mbar_helpers_bytes + c_bytes)
            ) // (occupancy * c_bytes_per_stage)

        return num_acc_stage, num_ab_stage, num_c_stage

    @staticmethod
    def _compute_grid(
        total_num_clusters: int,
        cluster_shape_mn: tuple[int, int],
        max_active_clusters: cutlass.Constexpr[int],
        swizzle_size: int = 1,
        raster_along_m: bool = True,
    ) -> tuple[utils.PersistentTileSchedulerParams, tuple[int, int, int]]:
        # Create problem shape with M, N dimensions from cluster shape
        # and L dimension representing the total number of clusters.
        problem_shape_ntile_mnl = (
            cluster_shape_mn[0],
            cluster_shape_mn[1],
            cutlass.Int32(total_num_clusters),
        )

        tile_sched_params = utils.PersistentTileSchedulerParams(
            problem_shape_ntile_mnl,
            (*cluster_shape_mn, 1),
            swizzle_size=swizzle_size,
            raster_along_m=raster_along_m,
        )

        grid = utils.StaticPersistentTileScheduler.get_grid_shape(
            tile_sched_params, max_active_clusters
        )

        return tile_sched_params, grid

    @staticmethod
    def _get_mbar_smem_bytes(**kwargs_stages: int) -> int:
        num_barriers_per_stage = 2
        num_bytes_per_barrier = 8
        mbar_smem_consumption = sum(
            [
                num_barriers_per_stage * num_bytes_per_barrier * stage
                for stage in kwargs_stages.values()
            ]
        )
        return mbar_smem_consumption

    @staticmethod
    def is_valid_dtypes_and_scale_factor_vec_size(
        ab_dtype: Type[cutlass.Numeric],
        sf_dtype: Type[cutlass.Numeric],
        sf_vec_size: int,
        c_dtype: Type[cutlass.Numeric],
    ) -> bool:
        is_valid = True

        # Check valid ab_dtype
        if ab_dtype not in {
            cutlass.Float4E2M1FN,
            cutlass.Float8E5M2,
            cutlass.Float8E4M3FN,
        }:
            is_valid = False

        # Check valid sf_vec_size
        if sf_vec_size not in {16, 32}:
            is_valid = False

        # Check valid sf_dtype
        if sf_dtype not in {cutlass.Float8E8M0FNU, cutlass.Float8E4M3FN}:
            is_valid = False

        # Check valid sf_dtype and sf_vec_size combinations
        if sf_dtype == cutlass.Float8E4M3FN and sf_vec_size == 32:
            is_valid = False
        if ab_dtype in {cutlass.Float8E5M2, cutlass.Float8E4M3FN} and sf_vec_size == 16:
            is_valid = False

        # Check valid c_dtype
        if c_dtype not in {
            cutlass.Float32,
            cutlass.Float16,
            cutlass.BFloat16,
            cutlass.Float8E5M2,
            cutlass.Float8E4M3FN,
        }:
            is_valid = False

        return is_valid

    @staticmethod
    def is_valid_layouts(
        ab_dtype: Type[cutlass.Numeric],
        c_dtype: Type[cutlass.Numeric],
        a_major: str,
        b_major: str,
        c_major: str,
    ) -> bool:
        is_valid = True

        if ab_dtype is cutlass.Float4E2M1FN and not (a_major == "k" and b_major == "k"):
            is_valid = False
        return is_valid

    @staticmethod
    def is_valid_mma_tiler_and_cluster_shape(
        mma_tiler_mn: Tuple[int, int],
        cluster_shape_mn: Tuple[int, int],
    ) -> bool:
        is_valid = True
        # Skip invalid mma tile shape
        if mma_tiler_mn[0] not in [128, 256]:
            is_valid = False
        if mma_tiler_mn[1] not in [128, 256]:
            is_valid = False
        # Skip illegal cluster shape
        if cluster_shape_mn[0] % (2 if mma_tiler_mn[0] == 256 else 1) != 0:
            is_valid = False
        # Skip invalid cluster shape
        is_power_of_2 = lambda x: x > 0 and (x & (x - 1)) == 0
        if (
            cluster_shape_mn[0] * cluster_shape_mn[1] > 16
            or cluster_shape_mn[0] <= 0
            or cluster_shape_mn[1] <= 0
            # Special cluster shape check for scale factor multicasts.
            # Due to limited size of scale factors, we can't multicast among more than 4 CTAs.
            or cluster_shape_mn[0] > 4
            or cluster_shape_mn[1] > 4
            or not is_power_of_2(cluster_shape_mn[0])
            or not is_power_of_2(cluster_shape_mn[1])
        ):
            is_valid = False
        return is_valid

    @staticmethod
    def is_valid_tensor_alignment(
        problem_sizes_mnkl: List[Tuple[int, int, int, int]],
        ab_dtype: Type[cutlass.Numeric],
        c_dtype: Type[cutlass.Numeric],
        a_major: str,
        b_major: str,
        c_major: str,
    ) -> bool:
        is_valid = True

        def check_contigous_16B_alignment(dtype, is_mode0_major, tensor_shape):
            major_mode_idx = 0 if is_mode0_major else 1
            num_major_elements = tensor_shape[major_mode_idx]
            num_contiguous_elements = 16 * 8 // dtype.width
            return num_major_elements % num_contiguous_elements == 0

        for m, n, k, l in problem_sizes_mnkl:
            if (
                not check_contigous_16B_alignment(ab_dtype, a_major == "m", (m, k, l))
                or not check_contigous_16B_alignment(
                    ab_dtype, b_major == "n", (n, k, l)
                )
                or not check_contigous_16B_alignment(c_dtype, c_major == "m", (m, n, l))
            ):
                is_valid = False
        return is_valid

    @staticmethod
    def can_implement(
        ab_dtype: Type[cutlass.Numeric],
        sf_dtype: Type[cutlass.Numeric],
        sf_vec_size: int,
        c_dtype: Type[cutlass.Numeric],
        mma_tiler_mn: Tuple[int, int],
        cluster_shape_mn: Tuple[int, int],
        problem_sizes_mnkl: List[Tuple[int, int, int, int]],
        a_major: str,
        b_major: str,
        c_major: str,
    ) -> bool:
        can_implement = True
        # Skip unsupported types
        if not Sm100GroupedBlockScaledGemmKernel.is_valid_dtypes_and_scale_factor_vec_size(
            ab_dtype, sf_dtype, sf_vec_size, c_dtype
        ):
            can_implement = False
        # Skip unsupported layouts
        if not Sm100GroupedBlockScaledGemmKernel.is_valid_layouts(
            ab_dtype, c_dtype, a_major, b_major, c_major
        ):
            can_implement = False
        # Skip invalid mma tile shape and cluster shape
        if not Sm100GroupedBlockScaledGemmKernel.is_valid_mma_tiler_and_cluster_shape(
            mma_tiler_mn, cluster_shape_mn
        ):
            can_implement = False
        # Skip illegal problem shape for load/store alignment
        if not Sm100GroupedBlockScaledGemmKernel.is_valid_tensor_alignment(
            problem_sizes_mnkl, ab_dtype, c_dtype, a_major, b_major, c_major
        ):
            can_implement = False
        return can_implement

    # Size of smem we reserved for mbarrier, tensor memory management and tensormap update
    reserved_smem_bytes = 1024
    bytes_per_tensormap = 128
    num_tensormaps = 5
    # size of smem used for tensor memory management
    tensor_memory_management_bytes = 12


# Create tensor and return the pointer, tensor, and stride


def _select_initial_indices(
    problem_sizes: list[tuple[int, int, int, int]],
) -> tuple[int, int, int]:
    key_size_a = lambda item: item[1][0] * item[1][2]
    key_size_b = lambda item: item[1][1] * item[1][2]
    key_size_c = lambda item: item[1][0] * item[1][1]
    min_a_idx, _ = min(enumerate(problem_sizes), key=key_size_a)
    min_b_idx, _ = min(enumerate(problem_sizes), key=key_size_b)
    min_c_idx, _ = min(enumerate(problem_sizes), key=key_size_c)
    return min_a_idx, min_b_idx, min_c_idx


def _from_dlpack_typed(
    torch_tensor: torch.Tensor,
    cutlass_dtype: Type[cutlass.Numeric],
    assumed_align: int,
) -> cute.Tensor:
    tensor_for_dlpack = torch_tensor
    if getattr(cutlass_dtype, "width", 0) <= 8:
        if not torch_tensor.is_contiguous():
            tensor_for_dlpack = torch_tensor.contiguous()
        tensor_for_dlpack = tensor_for_dlpack.view(torch.uint8)
    use_32bit_stride = True
    try:
        if tensor_for_dlpack.numel() > (2**31 - 1):
            use_32bit_stride = False
        else:
            max_stride = max(tensor_for_dlpack.stride())
            if max_stride > (2**31 - 1):
                use_32bit_stride = False
    except Exception:
        use_32bit_stride = False
    cute_tensor = from_dlpack(
        tensor_for_dlpack,
        assumed_align=assumed_align,
        use_32bit_stride=use_32bit_stride,
    )
    cute_tensor.element_type = cutlass_dtype
    return cute_tensor


def _get_handle() -> Any:
    global _fake_st
    if _fake_st is None:
        _mk = getattr(cutlass_torch, "default_" + "st" + "ream")
        _fake_st = _mk()
    return _fake_st


def _normalize_sfs(
    sfs: list[tuple[torch.Tensor, torch.Tensor]]
) -> list[tuple[torch.Tensor, torch.Tensor]]:
    if not hasattr(torch, "float8_e4m3fn"):
        return sfs
    target = torch.float8_e4m3fn
    out: list[tuple[torch.Tensor, torch.Tensor]] = []
    for sfa, sfb in sfs:
        if sfa.dtype != target:
            sfa = sfa.to(target)
        if sfb.dtype != target:
            sfb = sfb.to(target)
        out.append((sfa, sfb))
    return out


def _swap_inputs(
    abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
    sfs: list[tuple[torch.Tensor, torch.Tensor]],
    problem_sizes: list[tuple[int, int, int, int]],
) -> tuple[
    list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
    list[tuple[torch.Tensor, torch.Tensor]],
    list[tuple[int, int, int, int]],
]:
    abc_swapped: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]] = []
    sfs_swapped: list[tuple[torch.Tensor, torch.Tensor]] = []
    sizes_swapped: list[tuple[int, int, int, int]] = []
    for (a_ref, b_ref, c_ref), (sfa_ref, sfb_ref), (m, n, k, l) in zip(
        abc_tensors, sfs, problem_sizes
    ):
        # Swap A/B to compute D = B @ A^T and write D in column-major so C = D^T.
        c_swapped = c_ref.permute(1, 0, 2)
        abc_swapped.append((b_ref, a_ref, c_swapped))
        sfs_swapped.append((sfb_ref, sfa_ref))
        sizes_swapped.append((n, m, k, l))
    return abc_swapped, sfs_swapped, sizes_swapped


def _make_initial_abc_tensor(
    torch_tensor: torch.Tensor,
    cutlass_dtype: Type[cutlass.Numeric],
    is_mode0_major: bool,
    assumed_align: int,
    c_divisibility: int | None = None,
) -> cute.Tensor:
    cute_tensor = _from_dlpack_typed(torch_tensor, cutlass_dtype, assumed_align)
    leading_dim = 0 if is_mode0_major else 1
    cute_tensor = cute_tensor.mark_layout_dynamic(leading_dim=leading_dim)
    stride_order = (2, 1, 0) if is_mode0_major else (2, 0, 1)
    if cutlass_dtype == cutlass.Float4E2M1FN:
        divisibility = 32
    elif getattr(cutlass_dtype, "width", 0) == 16:
        divisibility = 8 if c_divisibility is None else c_divisibility
    else:
        divisibility = 16
    cute_tensor.mark_compact_shape_dynamic(
        mode=leading_dim,
        stride_order=stride_order,
        divisibility=divisibility,
    )
    return cute_tensor



def _compute_cluster_tile(mma_tiler_mn: tuple[int, int], cluster_shape_mn: tuple[int, int]) -> tuple[int, int]:
    cta_tile = (128, mma_tiler_mn[1])
    return (cta_tile[0] * cluster_shape_mn[0], cta_tile[1] * cluster_shape_mn[1])


def _total_clusters(problem_sizes: list[tuple[int, int, int, int]], cluster_tile: tuple[int, int]) -> int:
    total = 0
    tm, tn = cluster_tile
    for m, n, _, _ in problem_sizes:
        cm = (m + tm - 1) // tm
        cn = (n + tn - 1) // tn
        total += cm * cn
    return total


def _build_entry(cfg_key: str,
                 mma_tiler_mn: tuple[int, int],
                 cluster_shape_mn: tuple[int, int],
                 abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
                 sfs: list[tuple[torch.Tensor, torch.Tensor]],
                 problem_sizes: list[tuple[int, int, int, int]],
                 use_tma_store: bool,
                 c_col_major: bool,
                 max_ab_stage: int | None,
                 prefetch_dist: int | None,
                 force_c_stage: int | None,
                 c_assumed_align: int,
                 c_divisibility: int,
                 swizzle_size: int,
                 raster_along_m: bool):
    dev = abc_tensors[0][0].device
    group_count = len(problem_sizes)

    hinfo = utils.HardwareInfo()
    max_active = hinfo.get_max_active_clusters(cluster_shape_mn[0] * cluster_shape_mn[1])
    sm_count = hinfo.get_max_active_clusters(1)

    num_tmaps = Sm100GroupedBlockScaledGemmKernel.num_tensormaps
    bytes_tmap = Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
    tmap_shape = (sm_count, num_tmaps, bytes_tmap)

    tmap_t = torch.empty(tmap_shape, dtype=torch.int64, device=dev)
    tmap_cute = from_dlpack(tmap_t)
    tmap_cute.element_type = cutlass.Int64

    dims_cpu = torch.tensor(problem_sizes, dtype=torch.int32, pin_memory=True)
    dims_gpu = dims_cpu.to(device=dev)
    dims_cute = from_dlpack(dims_gpu)
    dims_cute.element_type = cutlass.Int32

    strides_cpu = torch.empty((group_count, 3, 2), dtype=torch.int32, pin_memory=True)
    ptrs_cpu = torch.empty((group_count, 3), dtype=torch.int64, pin_memory=True)
    ptrs_sf_cpu = torch.empty((group_count, 2), dtype=torch.int64, pin_memory=True)

    strides_gpu = strides_cpu.to(device=dev)
    ptrs_gpu = ptrs_cpu.to(device=dev)
    ptrs_sf_gpu = ptrs_sf_cpu.to(device=dev)
    strides_cute = from_dlpack(strides_gpu)
    strides_cute.element_type = cutlass.Int32
    ptrs_cute = from_dlpack(ptrs_gpu)
    ptrs_cute.element_type = cutlass.Int64
    ptrs_sf_cute = from_dlpack(ptrs_sf_gpu)
    ptrs_sf_cute.element_type = cutlass.Int64

    def _fill_stride():
        for i, (m, n, k, _) in enumerate(problem_sizes):
            stride_a = (k, 1)
            stride_b = (k, 1)
            if c_col_major:
                stride_c = (1, m)
            else:
                stride_c = (n, 1)
            strides_cpu[i, 0, 0] = stride_a[0]
            strides_cpu[i, 0, 1] = stride_a[1]
            strides_cpu[i, 1, 0] = stride_b[0]
            strides_cpu[i, 1, 1] = stride_b[1]
            strides_cpu[i, 2, 0] = stride_c[0]
            strides_cpu[i, 2, 1] = stride_c[1]
        strides_gpu.copy_(strides_cpu, non_blocking=True)

    _fill_stride()

    cluster_tile = _compute_cluster_tile(mma_tiler_mn, cluster_shape_mn)
    total_num_clusters = _total_clusters(problem_sizes, cluster_tile)

    min_a_idx, min_b_idx, min_c_idx = _select_initial_indices(problem_sizes)
    a_ref = abc_tensors[min_a_idx][0]
    b_ref = abc_tensors[min_b_idx][1]
    c_ref = abc_tensors[min_c_idx][2]
    sfa_ref = sfs[min_a_idx][0]
    sfb_ref = sfs[min_b_idx][1]

    initial_a = _make_initial_abc_tensor(
        a_ref, cutlass.Float4E2M1FN, is_mode0_major=False, assumed_align=16
    )
    initial_b = _make_initial_abc_tensor(
        b_ref, cutlass.Float4E2M1FN, is_mode0_major=False, assumed_align=16
    )
    initial_c = _make_initial_abc_tensor(
        c_ref,
        cutlass.Float16,
        is_mode0_major=c_col_major,
        assumed_align=c_assumed_align,
        c_divisibility=c_divisibility,
    )
    initial_sfa = _from_dlpack_typed(
        sfa_ref, cutlass.Float8E4M3FN, assumed_align=16
    )
    initial_sfb = _from_dlpack_typed(
        sfb_ref, cutlass.Float8E4M3FN, assumed_align=16
    )

    gemm = Sm100GroupedBlockScaledGemmKernel(
        sf_vec_size=16,
        mma_tiler_mn=mma_tiler_mn,
        cluster_shape_mn=cluster_shape_mn,
        use_tma_store=use_tma_store,
        max_ab_stage=max_ab_stage,
        prefetch_dist=prefetch_dist,
        force_c_stage=force_c_stage,
        c_assumed_align=c_assumed_align,
        c_divisibility=c_divisibility,
        swizzle_size=swizzle_size,
        raster_along_m=raster_along_m,
    )
    gemm.c_tile_stride = (int(c_ref.stride(0)), int(c_ref.stride(1)))

    try:
        compiled = cute.compile(
            gemm,
            initial_a,
            initial_b,
            initial_c,
            initial_sfa,
            initial_sfb,
            group_count,
            dims_cute,
            strides_cute,
            ptrs_cute,
            ptrs_sf_cute,
            total_num_clusters,
            tmap_cute,
            max_active,
            _get_handle(),
            options="--opt-level 2",
        )
    except Exception as exc:
        print(f"cute.compile failed: {exc}", file=sys.stderr, flush=True)
        raise RuntimeError(str(exc)) from exc

    entry = {
        "cfg_key": cfg_key,
        "mma_tiler_mn": mma_tiler_mn,
        "cluster_shape_mn": cluster_shape_mn,
        "group_count": group_count,
        "problem_sizes": problem_sizes,
        "dims_gpu": dims_gpu,
        "dims_cute": dims_cute,
        "strides_gpu": strides_gpu,
        "strides_cute": strides_cute,
        "ptrs_cpu": ptrs_cpu,
        "ptrs_gpu": ptrs_gpu,
        "ptrs_cute": ptrs_cute,
        "ptrs_sf_cpu": ptrs_sf_cpu,
        "ptrs_sf_gpu": ptrs_sf_gpu,
        "ptrs_sf_cute": ptrs_sf_cute,
        "tmap": tmap_cute,
        "initial_a": initial_a,
        "initial_b": initial_b,
        "initial_c": initial_c,
        "initial_sfa": initial_sfa,
        "initial_sfb": initial_sfb,
        "initial_refs": (a_ref, b_ref, c_ref, sfa_ref, sfb_ref),
        "compiled": compiled,
        "handle": _get_handle(),
    }
    return entry


def _update_ptrs(entry: dict,
                 abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
                 sfs: list[tuple[torch.Tensor, torch.Tensor]]):
    ptrs_cpu = entry["ptrs_cpu"]
    ptrs_sf_cpu = entry["ptrs_sf_cpu"]
    for i, ((a_ref, b_ref, c_ref), (sfa_ref, sfb_ref)) in enumerate(zip(abc_tensors, sfs)):
        ptrs_cpu[i, 0] = a_ref.data_ptr()
        ptrs_cpu[i, 1] = b_ref.data_ptr()
        ptrs_cpu[i, 2] = c_ref.data_ptr()
        ptrs_sf_cpu[i, 0] = sfa_ref.data_ptr()
        ptrs_sf_cpu[i, 1] = sfb_ref.data_ptr()
    entry["ptrs_gpu"].copy_(ptrs_cpu, non_blocking=True)
    entry["ptrs_sf_gpu"].copy_(ptrs_sf_cpu, non_blocking=True)


def _run_group(cfg_key: str,
               mma_tiler_mn: tuple[int, int],
               cluster_shape_mn: tuple[int, int],
               abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
               sfs: list[tuple[torch.Tensor, torch.Tensor]],
               problem_sizes: list[tuple[int, int, int, int]],
               use_tma_store: bool,
               c_col_major: bool,
               max_ab_stage: int | None,
               prefetch_dist: int | None,
               force_c_stage: int | None,
               c_assumed_align: int,
               c_divisibility: int,
               swizzle_size: int,
               raster_along_m: bool):
    if not abc_tensors:
        return
    dev = abc_tensors[0][0].device
    cache_key = (
        dev.index if dev.type == "cuda" else -1,
        cfg_key,
        use_tma_store,
        c_col_major,
        max_ab_stage,
        prefetch_dist,
        force_c_stage,
        c_assumed_align,
        c_divisibility,
        swizzle_size,
        raster_along_m,
        tuple(problem_sizes),
    )
    entry = _kernel_cache.get(cache_key)
    if entry is None:
        entry = _build_entry(
            cfg_key,
            mma_tiler_mn,
            cluster_shape_mn,
            abc_tensors,
            sfs,
            problem_sizes,
            use_tma_store,
            c_col_major,
            max_ab_stage,
            prefetch_dist,
            force_c_stage,
            c_assumed_align,
            c_divisibility,
            swizzle_size,
            raster_along_m,
        )
        _kernel_cache[cache_key] = entry
    _update_ptrs(entry, abc_tensors, sfs)
    try:
        entry["compiled"](
            entry["initial_a"],
            entry["initial_b"],
            entry["initial_c"],
            entry["initial_sfa"],
            entry["initial_sfb"],
            entry["dims_cute"],
            entry["strides_cute"],
            entry["ptrs_cute"],
            entry["ptrs_sf_cute"],
            entry["tmap"],
            entry["handle"],
        )
    except Exception as exc:
        print(f"compiled invocation failed: {exc}", file=sys.stderr, flush=True)
        raise RuntimeError(str(exc)) from exc


def _split_groups(problem_sizes: list[tuple[int, int, int, int]]) -> tuple[list[int], list[int]]:
    idx_1 = []
    idx_2 = []
    for i, (m, _n, k, _l) in enumerate(problem_sizes):
        if m >= 256 and k >= 4096:
            idx_2.append(i)
        else:
            idx_1.append(i)
    return idx_1, idx_2


def _split_groups_swap(problem_sizes: list[tuple[int, int, int, int]]) -> tuple[list[int], list[int]]:
    idx_1 = []
    idx_2 = []
    for i, (m, n, k, _l) in enumerate(problem_sizes):
        # m is swapped M (original N), n is swapped N (original M)
        if m >= 256 and k >= 4096:
            idx_2.append(i)
        else:
            idx_1.append(i)
    return idx_1, idx_2


def _split_groups_swap_by_idx(
    problem_sizes: list[tuple[int, int, int, int]],
    indices: list[int],
) -> tuple[list[int], list[int]]:
    idx_1: list[int] = []
    idx_2: list[int] = []
    for i in indices:
        m, _n, k, _l = problem_sizes[i]
        if m >= 256 and k >= 4096:
            idx_2.append(i)
        else:
            idx_1.append(i)
    return idx_1, idx_2


def _swizzle_params(is_wide: bool) -> tuple[int, bool]:
    if is_wide:
        size = int(os.environ.get("NVFP4_SWIZZLE_WIDE", "1"))
        raster_along_m = os.environ.get("NVFP4_SWIZZLE_WIDE_RASTER_M", "0") != "0"
    else:
        size = int(os.environ.get("NVFP4_SWIZZLE_NARROW", "1"))
        raster_along_m = os.environ.get("NVFP4_SWIZZLE_NARROW_RASTER_M", "1") != "0"
    if size not in (1, 2, 4, 8):
        size = 1
    return size, raster_along_m



def _collect_by_idx(
    abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
    sfs: list[tuple[torch.Tensor, torch.Tensor]],
    problem_sizes: list[tuple[int, int, int, int]],
    indices: list[int],
) -> tuple[
    list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
    list[tuple[torch.Tensor, torch.Tensor]],
    list[tuple[int, int, int, int]],
]:
    return (
        [abc_tensors[i] for i in indices],
        [sfs[i] for i in indices],
        [problem_sizes[i] for i in indices],
    )


def _can_use_simt(problem_sizes: list[tuple[int, int, int, int]],
                  idx_1: list[int],
                  idx_2: list[int]) -> bool:
    for i in idx_1:
        m, n, _k, _l = problem_sizes[i]
        if (m % 128) != 0 or (n % 256) != 0:
            return False
    for i in idx_2:
        m, n, _k, _l = problem_sizes[i]
        if (m % 256) != 0 or (n % 256) != 0:
            return False
    return True


def custom_kernel(data: input_t) -> output_t:
    abc_tensors, _sfs_raw, sfs_reordered, problem_sizes = data

    sfs_reordered = _normalize_sfs(sfs_reordered)

    # Fast path: grouped CUTLASS kernel using swap + speedK-style scheduling.
    if abc_tensors and _is_sm100_device(abc_tensors[0][0]):
        speedk_ok = _try_speedk_ext(abc_tensors, sfs_reordered, problem_sizes)
        if speedk_ok:
            return [c for (_a, _b, c) in abc_tensors]

    idx_unswapped: list[int] = []
    idx_rest: list[int] = []
    for i, (_m, _n, k, _l) in enumerate(problem_sizes):
        if k == 2048:
            idx_unswapped.append(i)
        else:
            idx_rest.append(i)

    abc_unswapped, sfs_unswapped, sizes_unswapped = _collect_by_idx(
        abc_tensors, sfs_reordered, problem_sizes, idx_unswapped
    )
    abc_rest, sfs_rest, sizes_rest = _collect_by_idx(
        abc_tensors, sfs_reordered, problem_sizes, idx_rest
    )

    abc_swapped, sfs_swapped, sizes_swapped = _swap_inputs(
        abc_rest, sfs_rest, sizes_rest
    )
    if _enable_stage8 and abc_swapped and _is_sm100_device(abc_swapped[0][0]):
        stage8_out = _try_stage8_ext(abc_swapped, sfs_swapped, sizes_swapped)
        if stage8_out is not None:
            return [c for (_a, _b, c) in abc_tensors]
    c_col_major = True
    is_sm100 = bool(abc_swapped) and _is_sm100_device(abc_swapped[0][0])
    c_assumed_align = 32
    c_divisibility = 16
    prefetch_dist = None if is_sm100 else 0
    use_tma_store = True
    idx_wide: list[int] = []
    idx_narrow: list[int] = []
    for i, (_m, _n, k, _l) in enumerate(sizes_swapped):
        if k >= 6000:
            idx_wide.append(i)
        else:
            idx_narrow.append(i)

    idx_wide_1, idx_wide_2 = _split_groups_swap_by_idx(sizes_swapped, idx_wide)
    idx_narrow_1, idx_narrow_2 = _split_groups_swap_by_idx(sizes_swapped, idx_narrow)

    abc_wide_1, sfs_wide_1, sizes_wide_1 = _collect_by_idx(
        abc_swapped, sfs_swapped, sizes_swapped, idx_wide_1
    )
    abc_wide_2, sfs_wide_2, sizes_wide_2 = _collect_by_idx(
        abc_swapped, sfs_swapped, sizes_swapped, idx_wide_2
    )
    abc_narrow_1, sfs_narrow_1, sizes_narrow_1 = _collect_by_idx(
        abc_swapped, sfs_swapped, sizes_swapped, idx_narrow_1
    )
    abc_narrow_2, sfs_narrow_2, sizes_narrow_2 = _collect_by_idx(
        abc_swapped, sfs_swapped, sizes_swapped, idx_narrow_2
    )

    enable_simt = os.environ.get("NVFP4_ENABLE_SIMT", "") == "1"
    swizzle_wide, raster_wide = _swizzle_params(True)
    swizzle_narrow, raster_narrow = _swizzle_params(False)
    force_c_stage_2sm_env = os.environ.get("NVFP4_FORCE_C_STAGE_2SM", "")
    force_c_stage_2sm = int(force_c_stage_2sm_env) if force_c_stage_2sm_env else None
    if force_c_stage_2sm is not None and force_c_stage_2sm <= 0:
        force_c_stage_2sm = None
    max_ab_stage_2sm_env = os.environ.get("NVFP4_MAX_AB_STAGE_2SM", "")
    max_ab_stage_2sm = int(max_ab_stage_2sm_env) if max_ab_stage_2sm_env else None
    if max_ab_stage_2sm is not None and max_ab_stage_2sm <= 0:
        max_ab_stage_2sm = None

    _run_group(
        "unswapped",
        (128, 256),
        # For k==2048 (unswapped), M is small and N is large. Favor clustering along N
        # to multicast A/SFA across CTAs and reduce redundant loads.
        (1, 4),
        abc_unswapped,
        sfs_unswapped,
        sizes_unswapped,
        use_tma_store,
        False,
        None,
        prefetch_dist,
        None,
        c_assumed_align,
        c_divisibility,
        swizzle_narrow,
        False,
    )
    if enable_simt:
        def is_simt_safe(sz: tuple[int, int, int, int]) -> bool:
            _m, n, k, _l = sz
            # SIMT vector store uses 128-bit vectors; require N divisible by 8.
            # Limit SIMT to k < 4096 (1SM path) to avoid 2CTA edge cases.
            return (n % 8) == 0 and k < 4096

        def gather(idx_list: list[int], want_simt: bool):
            abc_out = []
            sfs_out = []
            sizes_out = []
            for i in idx_list:
                if is_simt_safe(sizes_swapped[i]) == want_simt:
                    abc_out.append(abc_swapped[i])
                    sfs_out.append(sfs_swapped[i])
                    sizes_out.append(sizes_swapped[i])
            return abc_out, sfs_out, sizes_out

        abc_wide_1_simt, sfs_wide_1_simt, sizes_wide_1_simt = gather(
            idx_wide_1, True
        )
        abc_wide_1_tma, sfs_wide_1_tma, sizes_wide_1_tma = gather(
            idx_wide_1, False
        )
        abc_wide_2_simt, sfs_wide_2_simt, sizes_wide_2_simt = gather(
            idx_wide_2, True
        )
        abc_wide_2_tma, sfs_wide_2_tma, sizes_wide_2_tma = gather(
            idx_wide_2, False
        )
        abc_narrow_1_simt, sfs_narrow_1_simt, sizes_narrow_1_simt = gather(
            idx_narrow_1, True
        )
        abc_narrow_1_tma, sfs_narrow_1_tma, sizes_narrow_1_tma = gather(
            idx_narrow_1, False
        )
        abc_narrow_2_simt, sfs_narrow_2_simt, sizes_narrow_2_simt = gather(
            idx_narrow_2, True
        )
        abc_narrow_2_tma, sfs_narrow_2_tma, sizes_narrow_2_tma = gather(
            idx_narrow_2, False
        )

        def run_pair(
            cfg_key_base: str,
            mma_tiler_mn: tuple[int, int],
            cluster_shape_mn: tuple[int, int],
            abc_simt: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
            sfs_simt: list[tuple[torch.Tensor, torch.Tensor]],
            sizes_simt: list[tuple[int, int, int, int]],
            abc_tma: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
            sfs_tma: list[tuple[torch.Tensor, torch.Tensor]],
            sizes_tma: list[tuple[int, int, int, int]],
            swizzle_size: int,
            raster_along_m: bool,
            max_ab_stage: int | None = None,
            force_c_stage: int | None = None,
        ):
            _run_group(
                f"{cfg_key_base}s",
                mma_tiler_mn,
                cluster_shape_mn,
                abc_simt,
                sfs_simt,
                sizes_simt,
                False,
                c_col_major,
                max_ab_stage,
                prefetch_dist,
                force_c_stage,
                c_assumed_align,
                c_divisibility,
                swizzle_size,
                raster_along_m,
            )
            _run_group(
                f"{cfg_key_base}t",
                mma_tiler_mn,
                cluster_shape_mn,
                abc_tma,
                sfs_tma,
                sizes_tma,
                True,
                c_col_major,
                max_ab_stage,
                prefetch_dist,
                force_c_stage,
                c_assumed_align,
                c_divisibility,
                swizzle_size,
                raster_along_m,
            )

        run_pair(
            "w1",
            (128, 256),
            (2, 1),
            abc_wide_1_simt,
            sfs_wide_1_simt,
            sizes_wide_1_simt,
            abc_wide_1_tma,
            sfs_wide_1_tma,
            sizes_wide_1_tma,
            swizzle_wide,
            raster_wide,
        )
        run_pair(
            "w2",
            (256, 256),
            (4, 1),
            abc_wide_2_simt,
            sfs_wide_2_simt,
            sizes_wide_2_simt,
            abc_wide_2_tma,
            sfs_wide_2_tma,
            sizes_wide_2_tma,
            swizzle_wide,
            raster_wide,
            max_ab_stage_2sm,
            force_c_stage_2sm,
        )
        run_pair(
            "n1",
            (128, 128),
            (2, 1),
            abc_narrow_1_simt,
            sfs_narrow_1_simt,
            sizes_narrow_1_simt,
            abc_narrow_1_tma,
            sfs_narrow_1_tma,
            sizes_narrow_1_tma,
            swizzle_narrow,
            raster_narrow,
        )
        run_pair(
            "n2",
            (256, 128),
            (4, 1),
            abc_narrow_2_simt,
            sfs_narrow_2_simt,
            sizes_narrow_2_simt,
            abc_narrow_2_tma,
            sfs_narrow_2_tma,
            sizes_narrow_2_tma,
            swizzle_narrow,
            raster_narrow,
            max_ab_stage_2sm,
            force_c_stage_2sm,
        )
    else:
        _run_group(
            "swap_1sm_wide",
            (128, 256),
            (2, 1),
            abc_wide_1,
            sfs_wide_1,
            sizes_wide_1,
            use_tma_store,
            c_col_major,
            None,
            prefetch_dist,
            None,
            c_assumed_align,
            c_divisibility,
            swizzle_wide,
            raster_wide,
        )
        _run_group(
            "swap_2sm_wide",
            (256, 256),
            (4, 1),
            abc_wide_2,
            sfs_wide_2,
            sizes_wide_2,
            use_tma_store,
            c_col_major,
            max_ab_stage_2sm,
            prefetch_dist,
            force_c_stage_2sm,
            c_assumed_align,
            c_divisibility,
            swizzle_wide,
            raster_wide,
        )
        _run_group(
            "swap_1sm",
            (128, 128),
            (2, 1),
            abc_narrow_1,
            sfs_narrow_1,
            sizes_narrow_1,
            use_tma_store,
            c_col_major,
            None,
            prefetch_dist,
            None,
            c_assumed_align,
            c_divisibility,
            swizzle_narrow,
            raster_narrow,
        )
        _run_group(
            "swap_2sm",
            (256, 128),
            (4, 1),
            abc_narrow_2,
            sfs_narrow_2,
            sizes_narrow_2,
            use_tma_store,
            c_col_major,
            max_ab_stage_2sm,
            prefetch_dist,
            force_c_stage_2sm,
            c_assumed_align,
            c_divisibility,
            swizzle_narrow,
            raster_narrow,
        )

    return [c for (_a, _b, c) in abc_tensors]
scrolls · 3621 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 468277.

⋯ 58 unchanged lines
_stage8_mod: Any | None = None
_stage8_checked = False
_enable_stage8 = os.environ.get("NVFP4_ENABLE_STAGE8", "") == "1"
+ _speedk_mod: Any | None = None
+ _speedk_checked = False
+ _enable_speedk_ext = os.environ.get("NVFP4_ENABLE_SPEEDK_EXT", "") != "0"
+ _speedk_diag_done = False
+ _speedk_ext_mod: Any | None = None
+ _speedk_ext_fail = False
+ _speedk_inc_cache: list[str] | None = None
+ _speedk_cutlass_root: str | None = None
+ _speedk_pad_cache: dict[tuple[int, int, torch.dtype, int], torch.Tensor] = {}
+ _speedk_c_pad_cache: dict[tuple[int, int, torch.dtype, int], torch.Tensor] = {}
+ _speedk_disable_ext = os.environ.get("NVFP4_SPEEDK_EXT_DISABLE", "") == "1"
+ _speedk_diag = os.environ.get("NVFP4_SPEEDK_DIAG", "") == "1"
def _is_sm100_device(t: torch.Tensor) -> bool:
⋯ 37 unchanged lines
return None
+ def _try_speedk_ext(
+ abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
+ sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],
+ problem_sizes: list[tuple[int, int, int, int]],
+ ):
+ global _speedk_ext_mod, _speedk_checked, _speedk_ext_fail, _speedk_diag_done
+ def _diag(msg: str):
+ global _speedk_diag_done
+ if not _speedk_diag or _speedk_diag_done:
+ return
+ _speedk_diag_done = True
+ print(msg, file=sys.stderr, flush=True)
+ if _speedk_disable_ext or (not _enable_speedk_ext) or (not abc_tensors):
+ return None
+
+ # This kernel is SM100-only. Avoid any JIT/compile work on other GPUs.
+ if not _is_sm100_device(abc_tensors[0][0]):
+ return None
+
+ if _speedk_ext_fail:
+ return None
+
+ if _speedk_ext_mod is None:
+ _speedk_checked = True
+ _speedk_ext_mod = _speedk_load_ext(_diag)
+ if _speedk_ext_mod is None:
+ _speedk_ext_fail = True
+ _diag("speedk ext: build fail")
+ return None
+
+ out = _speedk_try_ext(_speedk_ext_mod, abc_tensors, sfs_reordered, problem_sizes)
+ if out is None:
+ _diag("speedk ext: none")
+ return None
+ _diag("speedk ext: ok")
+ return True
+
+
+ _SPEEDK_CPP_HEX = "0a2020202023696e636c756465203c746f7263682f657874656e73696f6e2e683e0a2020202023696e636c756465203c766563746f723e0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f31736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620534642293b0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f32736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620534642293b0a20202020"
+ _SPEEDK_CU_HEX = "0a2020202023696e636c756465203c746f7263682f657874656e73696f6e2e683e0a2020202023696e636c756465203c4154656e2f637564612f43554441436f6e746578742e683e0a2020202023696e636c756465203c6331302f637564612f4355444153747265616d2e683e0a2020202023696e636c756465203c6331302f637564612f4355444147756172642e683e0a2020202023696e636c756465203c766563746f723e0a2020202023696e636c756465203c7374646578636570743e0a2020202023696e636c756465203c747970655f7472616974733e0a0a20202020236966646566205f5f435544415f4e4f5f48414c465f4f50455241544f52535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c465f4f50455241544f52535f5f0a2020202023656e6469660a20202020236966646566205f5f435544415f4e4f5f48414c465f434f4e56455253494f4e535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c465f434f4e56455253494f4e535f5f0a2020202023656e6469660a20202020236966646566205f5f435544415f4e4f5f48414c46325f4f50455241544f52535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c46325f4f50455241544f52535f5f0a2020202023656e6469660a0a2020202023696e636c756465203c637564615f72756e74696d652e683e0a2020202023696620646566696e6564285f5f435544415f415243485f5f292026262021646566696e6564284355544c4153535f415243485f4d4d415f534d313030415f454e41424c4544290a2020202023646566696e65204355544c4153535f415243485f4d4d415f534d313030415f454e41424c454420310a2020202023656e6469660a2020202023696e636c75646520226375746c6173732f6375746c6173732e68220a2020202023696e636c7564652022637574652f74656e736f722e687070220a2020202023696e636c75646520226375746c6173732f74656e736f725f7265662e68220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f636f6c6c6563746976652f64656661756c745f6570696c6f6775652e687070220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f7468726561642f6c696e6561725f636f6d62696e6174696f6e2e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f64697370617463685f706f6c6963792e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f67726f75705f61727261795f70726f626c656d5f73686170652e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f636f6c6c6563746976652f636f6c6c6563746976655f6275696c6465722e687070220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f636f6c6c6563746976652f636f6c6c6563746976655f6275696c6465722e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f6465766963652f67656d6d5f756e6976657273616c5f616461707465722e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f6b65726e656c2f67656d6d5f756e6976657273616c2e687070220a2020202023696e636c75646520226375746c6173732f7574696c2f7061636b65645f7374726964652e687070220a2020202023696e636c75646520226375746c6173732f6b65726e656c5f68617264776172655f696e666f2e68220a2020202023696e636c75646520226375746c6173732f7574696c2f6465766963655f6d656d6f72792e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f6b65726e656c2f74696c655f7363686564756c65725f706172616d732e68220a0a2020202023646566696e65204355544c4153535f434845434b287374617475732920202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020646f207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a20202020202020206375746c6173733a3a537461747573205f737461747573203d2028737461747573293b202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020696620285f73746174757320213d206375746c6173733a3a5374617475733a3a6b5375636365737329207b20202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020202020207468726f77207374643a3a72756e74696d655f6572726f7228224355544c415353206572726f7222293b202020202020202020202020202020202020202020202020202020202020202020205c0a20202020202020207d20202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020207d207768696c65202830290a0a2020202023646566696e6520435544415f434845434b2865787072292020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020646f207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020637564614572726f725f74205f657272203d202865787072293b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020696620285f65727220213d20637564615375636365737329207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020202020207468726f77207374643a3a72756e74696d655f6572726f7228637564614765744572726f72537472696e67285f65727229293b202020202020202020202020202020202020202020202020205c0a20202020202020207d20202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020207d207768696c65202830290a0a202020207573696e672050726f626c656d5368617065203d206375746c6173733a3a67656d6d3a3a47726f757050726f626c656d53686170653c637574653a3a53686170653c696e742c696e742c696e743e3e3b0a202020207573696e6720456c656d656e74496e707574203d206375746c6173733a3a666c6f61745f65326d315f743b0a202020207573696e6720456c656d656e745346202020203d206375746c6173733a3a666c6f61745f7565346d335f743b0a202020207573696e6720456c656d656e744320202020203d206375746c6173733a3a68616c665f743b0a0a202020207573696e6720456c656d656e7441203d206375746c6173733a3a6e765f666c6f6174345f743c456c656d656e74496e7075743e3b0a202020207573696e67204c61796f75744120203d206375746c6173733a3a6c61796f75743a3a526f774d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744120203d2033323b0a0a202020207573696e6720456c656d656e7442203d206375746c6173733a3a6e765f666c6f6174345f743c456c656d656e74496e7075743e3b0a202020207573696e67204c61796f75744220203d206375746c6173733a3a6c61796f75743a3a436f6c756d6e4d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744220203d2033323b0a0a202020207573696e6720456c656d656e7444203d20456c656d656e74433b0a202020207573696e67204c61796f75744320203d206375746c6173733a3a6c61796f75743a3a436f6c756d6e4d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744320203d20323536202f206375746c6173733a3a73697a656f665f626974733c456c656d656e74433e3a3a76616c75653b0a20202020636f6e73746578707220696e7420416c69676e6d656e744420203d20323536202f206375746c6173733a3a73697a656f665f626974733c456c656d656e74443e3a3a76616c75653b0a202020207573696e6720456c656d656e74416363756d756c61746f7220203d20666c6f61743b0a0a202020207573696e672041726368546167203d206375746c6173733a3a617263683a3a536d3130303b0a202020207573696e67204f70657261746f72436c617373203d206375746c6173733a3a617263683a3a4f70436c617373426c6f636b5363616c656454656e736f724f703b0a202020207573696e67205374616765436f756e745479706531536d203d206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a5374616765436f756e743c343e3b0a202020207573696e67205374616765436f756e745479706532536d203d206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a5374616765436f756e743c353e3b0a202020207573696e6720436c75737465725368617065203d20637574653a3a53686170653c696e7433325f742c696e7433325f742c637574653a3a5f313e3b0a0a20202020737472756374204d4d4131534d436f6e666967207b0a2020202020207573696e67204d6d6154696c65536861706520202020203d20637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36342c637574653a3a5f3235363e3b0a2020202020207573696e67204b65726e656c5363686564756c652020203d206375746c6173733a3a67656d6d3a3a4b65726e656c5074724172726179546d61576172705370656369616c697a656431536d4e766634536d3130303b0a2020202020207573696e67204570696c6f6775655363686564756c65203d206375746c6173733a3a6570696c6f6775653a3a5074724172726179546d61576172705370656369616c697a656431536d3b0a202020207d3b0a0a20202020737472756374204d4d4132534d436f6e666967207b0a2020202020207573696e67204d6d6154696c65536861706520202020203d20637574653a3a53686170653c637574653a3a5f3235362c637574653a3a5f36342c637574653a3a5f3235363e3b0a2020202020207573696e67204b65726e656c5363686564756c652020203d206375746c6173733a3a67656d6d3a3a4b65726e656c5074724172726179546d61576172705370656369616c697a656432536d4e766634536d3130303b0a2020202020207573696e67204570696c6f6775655363686564756c65203d206375746c6173733a3a6570696c6f6775653a3a5074724172726179546d61576172705370656369616c697a656432536d3b0a202020207d3b0a0a202020207573696e6720436f6c6c6563746976654570696c6f67756531534d203d20747970656e616d65206375746c6173733a3a6570696c6f6775653a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a2020202020202020417263685461672c204f70657261746f72436c6173732c0a2020202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020202020637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36343e2c0a2020202020202020456c656d656e74416363756d756c61746f722c20456c656d656e74416363756d756c61746f722c0a2020202020202020456c656d656e74432c204c61796f757443202a2c20416c69676e6d656e74432c0a2020202020202020456c656d656e74442c204c61796f757443202a2c20416c69676e6d656e74442c0a2020202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4570696c6f6775655363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e6720436f6c6c6563746976654d61696e6c6f6f7031534d203d20747970656e616d65206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a202020202020417263685461672c204f70657261746f72436c6173732c0a202020202020456c656d656e74412c204c61796f757441202a2c20416c69676e6d656e74412c0a202020202020456c656d656e74422c204c61796f757442202a2c20416c69676e6d656e74422c0a202020202020456c656d656e74416363756d756c61746f722c0a202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020205374616765436f756e745479706531536d2c0a202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4b65726e656c5363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e672047656d6d4b65726e656c31534d203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a47656d6d556e6976657273616c3c0a202020202020202050726f626c656d53686170652c0a2020202020202020436f6c6c6563746976654d61696e6c6f6f7031534d2c0a2020202020202020436f6c6c6563746976654570696c6f67756531534d2c0a20202020202020206375746c6173733a3a67656d6d3a3a53747265616d4b5363686564756c65720a202020203e3b0a202020207573696e672047656d6d31534d203d206375746c6173733a3a67656d6d3a3a6465766963653a3a47656d6d556e6976657273616c416461707465723c47656d6d4b65726e656c31534d3e3b0a0a202020207573696e6720436f6c6c6563746976654570696c6f67756532534d203d20747970656e616d65206375746c6173733a3a6570696c6f6775653a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a2020202020202020417263685461672c204f70657261746f72436c6173732c0a2020202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020202020637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36343e2c0a2020202020202020456c656d656e74416363756d756c61746f722c20456c656d656e74416363756d756c61746f722c0a2020202020202020456c656d656e74432c204c61796f757443202a2c20416c69676e6d656e74432c0a2020202020202020456c656d656e74442c204c61796f757443202a2c20416c69676e6d656e74442c0a2020202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4570696c6f6775655363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e6720436f6c6c6563746976654d61696e6c6f6f7032534d203d20747970656e616d65206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a202020202020417263685461672c204f70657261746f72436c6173732c0a202020202020456c656d656e74412c204c61796f757441202a2c20416c69676e6d656e74412c0a202020202020456c656d656e74422c204c61796f757442202a2c20416c69676e6d656e74422c0a202020202020456c656d656e74416363756d756c61746f722c0a202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020205374616765436f756e745479706532536d2c0a202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4b65726e656c5363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e672047656d6d4b65726e656c32534d203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a47656d6d556e6976657273616c3c0a202020202020202050726f626c656d53686170652c0a2020202020202020436f6c6c6563746976654d61696e6c6f6f7032534d2c0a2020202020202020436f6c6c6563746976654570696c6f67756532534d2c0a20202020202020206375746c6173733a3a67656d6d3a3a53747265616d4b5363686564756c65720a202020203e3b0a202020207573696e672047656d6d32534d203d206375746c6173733a3a67656d6d3a3a6465766963653a3a47656d6d556e6976657273616c416461707465723c47656d6d4b65726e656c32534d3e3b0a0a2020202074656d706c617465203c747970656e616d6520543e0a202020207374727563742050696e6e6564486f7374427566666572207b0a202020202020542a20707472203d206e756c6c7074723b0a20202020202073697a655f7420636f756e74203d20303b0a0a20202020202050696e6e6564486f73744275666665722829203d2064656661756c743b0a2020202020206578706c696369742050696e6e6564486f73744275666665722873697a655f74206e29207b20616c6c6f63617465286e293b207d0a0a202020202020766f696420616c6c6f636174652873697a655f74206e29207b0a20202020202020206966202870747220262620636f756e74203e3d206e29207b0a2020202020202020202072657475726e3b0a20202020202020207d0a202020202020202072656c6561736528293b0a2020202020202020636f756e74203d206e3b0a2020202020202020435544415f434845434b2863756461486f7374416c6c6f63287265696e746572707265745f636173743c766f69642a2a3e2826707472292c206e202a2073697a656f662854292c2063756461486f7374416c6c6f63506f727461626c6529293b0a2020202020207d0a0a202020202020766f69642072656c656173652829207b0a20202020202020206966202870747229207b0a202020202020202020206375646146726565486f737428707472293b0a20202020202020202020707472203d206e756c6c7074723b0a20202020202020202020636f756e74203d20303b0a20202020202020207d0a2020202020207d0a0a2020202020207e50696e6e6564486f73744275666665722829207b2072656c6561736528293b207d0a0a202020202020542a2064617461282920636f6e7374207b2072657475726e207074723b207d0a2020202020205426206f70657261746f725b5d2873697a655f742069647829207b2072657475726e207074725b6964785d3b207d0a202020207d3b0a0a2020202074656d706c617465203c747970656e616d6520543e0a2020202073747275637420446576696365427566666572207b0a202020202020542a20707472203d206e756c6c7074723b0a20202020202073697a655f7420636f756e74203d20303b0a0a2020202020204465766963654275666665722829203d2064656661756c743b0a2020202020206578706c69636974204465766963654275666665722873697a655f74206e29207b20616c6c6f63617465286e293b207d0a0a202020202020766f696420616c6c6f636174652873697a655f74206e29207b0a20202020202020206966202870747220262620636f756e74203e3d206e29207b0a2020202020202020202072657475726e3b0a20202020202020207d0a202020202020202072656c6561736528293b0a2020202020202020636f756e74203d206e3b0a2020202020202020435544415f434845434b28637564614d616c6c6f63287265696e746572707265745f636173743c766f69642a2a3e2826707472292c206e202a2073697a656f6628542929293b0a2020202020207d0a0a202020202020766f69642072656c656173652829207b0a20202020202020206966202870747229207b0a20202020202020202020637564614672656528707472293b0a20202020202020202020707472203d206e756c6c7074723b0a20202020202020202020636f756e74203d20303b0a20202020202020207d0a2020202020207d0a0a2020202020207e4465766963654275666665722829207b2072656c6561736528293b207d0a0a202020202020542a20676574282920636f6e7374207b2072657475726e207074723b207d0a0a202020202020766f696420636f70795f66726f6d5f686f737428636f6e737420542a20686f73742c2073697a655f74206e2c206375646153747265616d5f742073747265616d29207b0a2020202020202020435544415f434845434b28637564614d656d6370794173796e63287074722c20686f73742c206e202a2073697a656f662854292c20637564614d656d637079486f7374546f4465766963652c2073747265616d29293b0a2020202020207d0a202020207d3b0a0a2020202074656d706c617465203c747970656e616d652047656d6d3e0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2072756e5f67726f757065645f696d706c280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a0a202020202020544f5243485f434845434b28412e73697a652829203d3d20422e73697a6528292c2022412f422073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d20432e73697a6528292c2022412f432073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d205346412e73697a6528292c2022412f5346412073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d205346422e73697a6528292c2022412f5346422073697a65206d69736d6174636822293b0a202020202020636f6e737420696e7433325f742067726f757073203d207374617469635f636173743c696e7433325f743e28412e73697a652829293b0a202020202020544f5243485f434845434b2867726f757073203e20302c20224e6f2067726f75707322293b0a0a202020202020696e74206465766963655f6964203d20415b305d2e6765745f64657669636528293b0a2020202020206331303a3a637564613a3a435544414775617264206465766963655f6775617264286465766963655f6964293b0a0a2020202020207573696e672053747269646541203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465413b0a2020202020207573696e672053747269646542203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465423b0a2020202020207573696e672053747269646543203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465433b0a2020202020207573696e672053747269646544203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465443b0a2020202020207573696e67204c61796f7574534641203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a496e7465726e616c4c61796f75745346413b0a2020202020207573696e67204c61796f7574534642203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a496e7465726e616c4c61796f75745346423b0a2020202020207573696e6720536d317878426c6b5363616c6564436f6e666967203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a536d317878426c6b5363616c6564436f6e6669673b0a0a2020202020207374617469632073697a655f74206361706163697479203d20303b0a20202020202073746174696320696e74206361636865645f646576696365203d202d313b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e207074725f415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e207074725f425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e207074725f435f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e207074725f445f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465413e207374726964655f415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465423e207374726964655f425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465433e207374726964655f435f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465443e207374726964655f445f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c4c61796f75745346413e206c61796f75745f5346415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c4c61796f75745346423e206c61796f75745f5346425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652050726f626c656d53686170653a3a556e6465726c79696e6750726f626c656d53686170653e2070726f626c656d5f73697a65735f686f73743b0a0a202020202020737461746963204465766963654275666665723c747970656e616d652050726f626c656d53686170653a3a556e6465726c79696e6750726f626c656d53686170653e2070726f626c656d5f73697a65735f6465766963653b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e207074725f413b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e207074725f423b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346413b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346423b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e207074725f433b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e207074725f443b0a202020202020737461746963204465766963654275666665723c537472696465413e207374726964655f413b0a202020202020737461746963204465766963654275666665723c537472696465423e207374726964655f423b0a202020202020737461746963204465766963654275666665723c537472696465433e207374726964655f433b0a202020202020737461746963204465766963654275666665723c537472696465443e207374726964655f443b0a202020202020737461746963204465766963654275666665723c4c61796f75745346413e206c61796f75745f5346413b0a202020202020737461746963204465766963654275666665723c4c61796f75745346423e206c61796f75745f5346423b0a2020202020207374617469632073697a655f7420776f726b73706163655f73697a655f636163686564203d20303b0a20202020202073746174696320746f7263683a3a54656e736f7220776f726b73706163655f74656e736f723b0a0a202020202020696620286361636865645f64657669636520213d206465766963655f6964207c7c206361706163697479203c207374617469635f636173743c73697a655f743e2867726f7570732929207b0a20202020202020206361636865645f646576696365203d206465766963655f69643b0a20202020202020206361706163697479203d207374617469635f636173743c73697a655f743e2867726f757073293b0a20202020202020207074725f415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f435f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f445f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f435f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f445f686f73742e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346425f686f73742e616c6c6f636174652867726f757073293b0a202020202020202070726f626c656d5f73697a65735f686f73742e616c6c6f636174652867726f757073293b0a0a202020202020202070726f626c656d5f73697a65735f6465766963652e616c6c6f636174652867726f757073293b0a20202020202020207074725f412e616c6c6f636174652867726f757073293b0a20202020202020207074725f422e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346412e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346422e616c6c6f636174652867726f757073293b0a20202020202020207074725f432e616c6c6f636174652867726f757073293b0a20202020202020207074725f442e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f412e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f422e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f432e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f442e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346412e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346422e616c6c6f636174652867726f757073293b0a2020202020207d0a0a202020202020636f6e73746578707220696e742074696c655f6d203d20637574653a3a73697a653c303e28747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c6553686170657b7d293b0a202020202020636f6e73746578707220696e742074696c655f6e203d20637574653a3a73697a653c313e28747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c6553686170657b7d293b0a202020202020696e7436345f7420746f74616c5f74696c6573203d20303b0a0a202020202020666f722028696e7433325f742069203d20303b2069203c2067726f7570733b202b2b6929207b0a2020202020202020544f5243485f434845434b28415b695d2e69735f6375646128292c202241206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b28425b695d2e69735f6375646128292c202242206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b28435b695d2e69735f6375646128292c202243206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b285346415b695d2e69735f6375646128292c2022534641206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b285346425b695d2e69735f6375646128292c2022534642206d757374206265204355444122293b0a0a2020202020202020696e7433325f74206d203d207374617469635f636173743c696e7433325f743e28415b695d2e73697a65283029293b0a2020202020202020696e7433325f74206b5f7061636b6564203d207374617469635f636173743c696e7433325f743e28415b695d2e73697a65283129293b0a2020202020202020696e7433325f74206b203d206b5f7061636b6564202a20323b0a2020202020202020696e7433325f74206e203d207374617469635f636173743c696e7433325f743e28425b695d2e73697a65283029293b0a2020202020202020544f5243485f434845434b28425b695d2e73697a65283129202a2032203d3d206b2c20224b206d69736d6174636822293b0a2020202020202020544f5243485f434845434b28435b695d2e73697a65283029203d3d206d20262620435b695d2e73697a65283129203d3d206e2c202243207368617065206d69736d6174636822293b0a0a202020202020202070726f626c656d5f73697a65735f686f73745b695d203d20637574653a3a6d616b655f7368617065286e2c206d2c206b293b0a20202020202020207374726964655f415f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465417b7d2c207b6e2c206b2c20317d293b0a20202020202020207374726964655f425f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465427b7d2c207b6d2c206b2c20317d293b0a20202020202020207374726964655f435f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465437b7d2c207b6e2c206d2c20317d293b0a20202020202020207374726964655f445f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465447b7d2c207b6e2c206d2c20317d293b0a20202020202020206c61796f75745f5346415f686f73745b695d203d20536d317878426c6b5363616c6564436f6e6669673a3a74696c655f61746f6d5f746f5f73686170655f53464128637574653a3a6d616b655f7368617065286e2c206d2c206b2c203129293b0a20202020202020206c61796f75745f5346425f686f73745b695d203d20536d317878426c6b5363616c6564436f6e6669673a3a74696c655f61746f6d5f746f5f73686170655f53464228637574653a3a6d616b655f7368617065286e2c206d2c206b2c203129293b0a0a2020202020202020696e742074696c65735f6d203d20286e202b2074696c655f6d202d203129202f2074696c655f6d3b0a2020202020202020696e742074696c65735f6e203d20286d202b2074696c655f6e202d203129202f2074696c655f6e3b0a2020202020202020746f74616c5f74696c6573202b3d207374617469635f636173743c696e7436345f743e2874696c65735f6d29202a207374617469635f636173743c696e7436345f743e2874696c65735f6e293b0a0a20202020202020207074725f415f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e28425b695d2e646174615f7074722829293b0a20202020202020207074725f425f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e28415b695d2e646174615f7074722829293b0a20202020202020207074725f435f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e28435b695d2e646174615f7074722829293b0a20202020202020207074725f445f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e28435b695d2e646174615f7074722829293b0a20202020202020207074725f5346415f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e285346425b695d2e646174615f7074722829293b0a20202020202020207074725f5346425f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e285346415b695d2e646174615f7074722829293b0a2020202020207d0a0a2020202020206175746f2073747265616d5f6f626a203d206331303a3a637564613a3a67657443757272656e744355444153747265616d286465766963655f6964293b0a2020202020206331303a3a637564613a3a4355444153747265616d47756172642073747265616d5f67756172642873747265616d5f6f626a293b0a2020202020206375646153747265616d5f742073747265616d203d2073747265616d5f6f626a2e73747265616d28293b0a20202020202070726f626c656d5f73697a65735f6465766963652e636f70795f66726f6d5f686f73742870726f626c656d5f73697a65735f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f412e636f70795f66726f6d5f686f7374287074725f415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f422e636f70795f66726f6d5f686f7374287074725f425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f5346412e636f70795f66726f6d5f686f7374287074725f5346415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f5346422e636f70795f66726f6d5f686f7374287074725f5346425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f432e636f70795f66726f6d5f686f7374287074725f435f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f442e636f70795f66726f6d5f686f7374287074725f445f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f412e636f70795f66726f6d5f686f7374287374726964655f415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f422e636f70795f66726f6d5f686f7374287374726964655f425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f432e636f70795f66726f6d5f686f7374287374726964655f435f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f442e636f70795f66726f6d5f686f7374287374726964655f445f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020206c61796f75745f5346412e636f70795f66726f6d5f686f7374286c61796f75745f5346415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020206c61796f75745f5346422e636f70795f66726f6d5f686f7374286c61796f75745f5346425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a0a2020202020206375746c6173733a3a4b65726e656c4861726477617265496e666f2068775f696e666f3b0a20202020202068775f696e666f2e6465766963655f6964203d206465766963655f69643b0a20202020202068775f696e666f2e736d5f636f756e74203d206375746c6173733a3a4b65726e656c4861726477617265496e666f3a3a71756572795f6465766963655f6d756c746970726f636573736f725f636f756e742868775f696e666f2e6465766963655f6964293b0a202020202020696620636f6e73746578707220287374643a3a69735f73616d655f763c47656d6d2c2047656d6d31534d3e29207b0a202020202020202068775f696e666f2e636c75737465725f7368617065203d2064696d3328312c20312c2031293b0a202020202020202068775f696e666f2e636c75737465725f73686170655f66616c6c6261636b203d2064696d3328312c20312c2031293b0a2020202020207d20656c7365207b0a202020202020202068775f696e666f2e636c75737465725f7368617065203d2064696d3328322c20312c2031293b0a202020202020202068775f696e666f2e636c75737465725f73686170655f66616c6c6261636b203d2064696d3328322c20312c2031293b0a2020202020207d0a0a202020202020747970656e616d652047656d6d3a3a417267756d656e747320617267756d656e74733b0a2020202020206465636c7479706528617267756d656e74732e6570696c6f6775652e7468726561642920667573696f6e5f617267733b0a202020202020667573696f6e5f617267732e616c7068615f707472203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e626574615f707472203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e616c706861203d20456c656d656e74416363756d756c61746f722831293b0a202020202020667573696f6e5f617267732e62657461203d20456c656d656e74416363756d756c61746f722830293b0a202020202020667573696f6e5f617267732e616c7068615f7074725f6172726179203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e626574615f7074725f6172726179203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e64416c706861203d207b637574653a3a5f307b7d2c20637574653a3a5f307b7d2c20307d3b0a202020202020667573696f6e5f617267732e6442657461203d207b637574653a3a5f307b7d2c20637574653a3a5f307b7d2c20307d3b0a0a202020202020747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c655363686564756c6572417267756d656e7473207363686564756c65723b0a2020202020207363686564756c65722e6d61785f7377697a7a6c655f73697a65203d20383b0a2020202020207363686564756c65722e7261737465725f6f72646572203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a64657461696c3a3a5261737465724f726465724f7074696f6e733a3a4865757269737469633b0a0a202020202020617267756d656e7473203d20747970656e616d652047656d6d3a3a417267756d656e7473207b0a20202020202020206375746c6173733a3a67656d6d3a3a47656d6d556e6976657273616c4d6f64653a3a6b47726f757065642c0a20202020202020207b67726f7570732c2070726f626c656d5f73697a65735f6465766963652e67657428292c206e756c6c7074727d2c0a20202020202020207b7074725f412e67657428292c207374726964655f412e67657428292c207074725f422e67657428292c207374726964655f422e67657428292c0a2020202020202020207074725f5346412e67657428292c206c61796f75745f5346412e67657428292c207074725f5346422e67657428292c206c61796f75745f5346422e67657428297d2c0a20202020202020207b667573696f6e5f617267732c207074725f432e67657428292c207374726964655f432e67657428292c207074725f442e67657428292c207374726964655f442e67657428297d2c0a202020202020202068775f696e666f2c207363686564756c65720a2020202020207d3b0a0a20202020202047656d6d2067656d6d3b0a20202020202073697a655f7420776f726b73706163655f73697a65203d2047656d6d3a3a6765745f776f726b73706163655f73697a6528617267756d656e7473293b0a2020202020206966202821776f726b73706163655f74656e736f722e646566696e65642829207c7c206361636865645f64657669636520213d206465766963655f6964207c7c20776f726b73706163655f73697a655f636163686564203c20776f726b73706163655f73697a6529207b0a20202020202020206175746f206f707473203d20746f7263683a3a54656e736f724f7074696f6e7328292e64657669636528746f7263683a3a6b435544412c206465766963655f6964292e647479706528746f7263683a3a6b55496e7438293b0a2020202020202020776f726b73706163655f74656e736f72203d20746f7263683a3a656d707479287b7374617469635f636173743c6c6f6e673e28776f726b73706163655f73697a65297d2c206f707473293b0a2020202020202020776f726b73706163655f73697a655f636163686564203d20776f726b73706163655f73697a653b0a2020202020207d0a202020202020766f69642a20776f726b73706163655f707472203d20776f726b73706163655f73697a65203f20776f726b73706163655f74656e736f722e646174615f7074722829203a206e756c6c7074723b0a0a2020202020204355544c4153535f434845434b2867656d6d2e63616e5f696d706c656d656e7428617267756d656e747329293b0a2020202020204355544c4153535f434845434b2867656d6d2e696e697469616c697a6528617267756d656e74732c20776f726b73706163655f7074722c2073747265616d29293b0a2020202020206175746f2072756e5f737461747573203d2067656d6d2e72756e2873747265616d2c206e756c6c7074722c2074727565293b0a2020202020206966202872756e5f73746174757320213d206375746c6173733a3a5374617475733a3a6b5375636365737329207b0a202020202020202072756e5f737461747573203d2067656d6d2e72756e2873747265616d293b0a2020202020207d0a2020202020204355544c4153535f434845434b2872756e5f737461747573293b0a0a20202020202072657475726e20433b0a202020207d0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f31736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a20202020202072657475726e2072756e5f67726f757065645f696d706c3c47656d6d31534d3e28412c20422c20432c205346412c20534642293b0a202020207d0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f32736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a20202020202072657475726e2072756e5f67726f757065645f696d706c3c47656d6d32534d3e28412c20422c20432c205346412c20534642293b0a202020207d0a202020200a"
+
+
+
+
+ def _speedk_dec_hex(h: str) -> str:
+ return bytes.fromhex(h).decode("utf-8")
+
+
+ def _speedk_repo_root() -> str:
+ return os.path.abspath(os.path.join(_ROOT, "..", "..", "..", ".."))
+
+
+ def _speedk_ensure_cutlass_dir(diag_fn) -> str | None:
+ global _speedk_cutlass_root
+ if _speedk_cutlass_root is not None:
+ return _speedk_cutlass_root
+
+ repo_root = _speedk_repo_root()
+ base_dir = os.path.join(repo_root, "tmp", "cutlass_cache")
+ root = os.path.join(base_dir, "cutlass-4.3.5")
+ inc = os.path.join(root, "include", "cutlass", "cutlass.h")
+ if os.path.isfile(inc):
+ _speedk_cutlass_root = root
+ return root
+
+ try:
+ os.makedirs(base_dir, exist_ok=True)
+ tgz_path = os.path.join(base_dir, "cutlass.tgz")
+ if not os.path.isfile(tgz_path):
+ import urllib.request
+
+ url = "https://github.com/NVIDIA/cutlass/archive/refs/tags/v4.3.5.tar.gz"
+ with urllib.request.urlopen(url, timeout=60) as resp:
+ data = resp.read()
+ with open(tgz_path, "wb") as f:
+ f.write(data)
+
+ import tarfile
+
+ with tarfile.open(tgz_path, "r:gz") as tf:
+ tf.extractall(base_dir)
+
+ if os.path.isfile(inc):
+ _speedk_cutlass_root = root
+ return root
+ except Exception:
+ diag_fn("speedk ext: cutlass fetch fail")
+ return None
+
+ return None
+
+
+ def _speedk_find_inc(diag_fn) -> list[str]:
+ global _speedk_inc_cache
+ if _speedk_inc_cache is not None:
+ return _speedk_inc_cache
+
+ incs: list[str] = []
+
+ def _add_cutlass(base: str) -> bool:
+ inc = os.path.join(base, "include")
+ if os.path.isfile(os.path.join(inc, "cutlass", "cutlass.h")):
+ incs.append(inc)
+ util_inc = os.path.join(base, "tools", "util", "include")
+ if os.path.isfile(os.path.join(util_inc, "cutlass", "util", "device_memory.h")):
+ incs.append(util_inc)
+ return True
+
+ inc2 = os.path.join(base, "cutlass", "include")
+ if os.path.isfile(os.path.join(inc2, "cutlass", "cutlass.h")):
+ incs.append(inc2)
+ util_inc2 = os.path.join(base, "cutlass", "tools", "util", "include")
+ if os.path.isfile(os.path.join(util_inc2, "cutlass", "util", "device_memory.h")):
+ incs.append(util_inc2)
+ return True
+
+ return False
+
+ repo_root = _speedk_repo_root()
+ for base in (
+ os.path.join(repo_root, "third_party", "cutlass"),
+ os.path.join(repo_root, "tmp", "cutlass-4.3.5"),
+ ):
+ _add_cutlass(base)
+
+ try:
+ root = os.path.dirname(getattr(cutlass, "__file__", "") or "")
+ if root:
+ bases = [root, os.path.dirname(root), os.path.dirname(os.path.dirname(root))]
+ for base in bases:
+ if _add_cutlass(base):
+ break
+ except Exception:
+ pass
+
+ try:
+ import torch.utils.cpp_extension as _ce
+
+ for inc in _ce.include_paths():
+ if os.path.isfile(os.path.join(inc, "cutlass", "cutlass.h")):
+ incs.append(inc)
+ util_inc3 = os.path.join(inc, "cutlass", "tools", "util", "include")
+ if os.path.isfile(os.path.join(util_inc3, "cutlass", "util", "device_memory.h")):
+ incs.append(util_inc3)
+ except Exception:
+ pass
+
+ if not incs:
+ root = _speedk_ensure_cutlass_dir(diag_fn)
+ if root:
+ _add_cutlass(root)
+
+ incs = list(dict.fromkeys(incs))
+ if not incs:
+ diag_fn("speedk ext: no headers")
+ _speedk_inc_cache = incs
+ return incs
+
+
+ def _speedk_load_ext(diag_fn):
+ global _speedk_ext_mod, _speedk_ext_fail
+ if _speedk_ext_mod is not None or _speedk_ext_fail:
+ return _speedk_ext_mod
+
+ incs = _speedk_find_inc(diag_fn)
+ if not incs:
+ _speedk_ext_fail = True
+ return None
+
+ try:
+ from torch.utils.cpp_extension import load_inline
+
+ repo_root = _speedk_repo_root()
+ os.environ.setdefault(
+ "TORCH_EXTENSIONS_DIR", os.path.join(repo_root, "tmp", "torch_extensions")
+ )
+ os.environ.setdefault("TORCH_CUDA_ARCH_LIST", "10.0a")
+
+ cpp_src = _speedk_dec_hex(_SPEEDK_CPP_HEX)
+ cu_src = _speedk_dec_hex(_SPEEDK_CU_HEX)
+
+ _speedk_ext_mod = load_inline(
+ name="nvfp4_grouped_cutlass_ext_speedk",
+ cpp_sources=cpp_src,
+ cuda_sources=cu_src,
+ functions=["grouped_gemm_1sm", "grouped_gemm_2sm"],
+ extra_cuda_cflags=[
+ "-O3",
+ "--use_fast_math",
+ "-std=c++17",
+ "--diag-suppress=144",
+ ],
+ extra_ldflags=["-lcuda"],
+ extra_cflags=["-O3", "-std=c++17"],
+ with_cuda=True,
+ extra_include_paths=incs,
+ verbose=False,
+ )
+ except Exception:
+ _speedk_ext_fail = True
+ _speedk_ext_mod = None
+ diag_fn("speedk ext: build fail")
+ return None
+
+ return _speedk_ext_mod
+
+
+ def _speedk_pad_rows(tensor: torch.Tensor, new_rows: int) -> torch.Tensor:
+ if tensor.size(0) == new_rows:
+ return tensor
+ dev_idx = tensor.device.index if tensor.device.index is not None else -1
+ key = (new_rows, tensor.size(1), tensor.dtype, dev_idx)
+ out = _speedk_pad_cache.get(key)
+ if out is None or out.size(0) != new_rows or out.size(1) != tensor.size(1):
+ out = torch.empty((new_rows, tensor.size(1)), device=tensor.device, dtype=tensor.dtype)
+ _speedk_pad_cache[key] = out
+ # Some low-bit dtypes (notably fp4x2) don't reliably support in-place fill or
+ # slice assignment. Bitcast to u8 so we only move raw bytes.
+ if tensor.element_size() == 1:
+ src = tensor if tensor.is_contiguous() else tensor.contiguous()
+ out_u8 = out.view(torch.uint8)
+ out_u8.zero_()
+ out_u8[: src.size(0), :] = src.view(torch.uint8)
+ return out
+ out.zero_()
+ out[: tensor.size(0), :] = tensor
+ return out
+
+
+ def _speedk_alloc_c_pad(rows: int, cols: int, ref: torch.Tensor) -> torch.Tensor:
+ dev_idx = ref.device.index if ref.device.index is not None else -1
+ key = (rows, cols, ref.dtype, dev_idx)
+ out = _speedk_c_pad_cache.get(key)
+ if out is None or out.size(0) != rows or out.size(1) != cols:
+ out = torch.empty((rows, cols), device=ref.device, dtype=ref.dtype)
+ _speedk_c_pad_cache[key] = out
+ return out
+
+
+ def _speedk_try_ext(
+ ext_mod,
+ abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
+ sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],
+ problem_sizes: list[tuple[int, int, int, int]],
+ ):
+ if ext_mod is None:
+ return None
+
+ def _use_2sm(n_orig: int, k: int) -> bool:
+ return n_orig >= 3072 and k >= 2048
+
+ items = []
+ for (a_ref, b_ref, c_ref), (sfa_reordered, sfb_reordered), (m, _n, k, l) in zip(
+ abc_tensors, sfs_reordered, problem_sizes
+ ):
+ if l != 1:
+ return None
+ a2 = a_ref[:, :, 0]
+ b2 = b_ref[:, :, 0]
+ c2 = c_ref[:, :, 0]
+ n = b2.size(0)
+ m_pad = ((m + 31) // 32) * 32
+ if m_pad != m:
+ a2 = _speedk_pad_rows(a2, m_pad)
+ c_pad = _speedk_alloc_c_pad(m_pad, n, c2)
+ items.append((a2, b2, c_pad, sfa_reordered, sfb_reordered, c2, m, n, k))
+ else:
+ items.append((a2, b2, c2, sfa_reordered, sfb_reordered, None, m, n, k))
+
+ def _gather(src):
+ return (
+ [x[0] for x in src],
+ [x[1] for x in src],
+ [x[2] for x in src],
+ [x[3] for x in src],
+ [x[4] for x in src],
+ )
+
+ small = [it for it in items if not _use_2sm(it[7], it[8])]
+ large = [it for it in items if _use_2sm(it[7], it[8])]
+
+ try:
+ if large:
+ a_list, b_list, c_list, sfa_list, sfb_list = _gather(large)
+ ext_mod.grouped_gemm_2sm(a_list, b_list, c_list, sfa_list, sfb_list)
+ if small:
+ a_list, b_list, c_list, sfa_list, sfb_list = _gather(small)
+ ext_mod.grouped_gemm_1sm(a_list, b_list, c_list, sfa_list, sfb_list)
+ except Exception:
+ try:
+ a_list, b_list, c_list, sfa_list, sfb_list = _gather(items)
+ ext_mod.grouped_gemm_1sm(a_list, b_list, c_list, sfa_list, sfb_list)
+ except Exception:
+ return None
+
+ for _a2, _b2, c2, _sfa, _sfb, c_orig, m, _n, _k in items:
+ if c_orig is not None:
+ c_orig.copy_(c2[:m, :])
+
+ return [c_ref for (_a, _b, c_ref) in abc_tensors]
+
+
@dsl_user_op
def _normalize_ptr(ptr, *, loc=None, ip=None):
if isinstance(ptr, _ir.Value):
⋯ 2885 unchanged lines
sfs_reordered = _normalize_sfs(sfs_reordered)
+ # Fast path: grouped CUTLASS kernel using swap + speedK-style scheduling.
+ if abc_tensors and _is_sm100_device(abc_tensors[0][0]):
+ speedk_ok = _try_speedk_ext(abc_tensors, sfs_reordered, problem_sizes)
+ if speedk_ok:
+ return [c for (_a, _b, c) in abc_tensors]
+
idx_unswapped: list[int] = []
idx_rest: list[int] = []
for i, (_m, _n, k, _l) in enumerate(problem_sizes):
scrolls · 343 diff lines total

Best evidence level for this revision: reported

JSON