submission 471047
oofbaroomf · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 3621 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-group-gemm-471047?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:3e936c9f6c75ecfba321a969da673aebef9ae0072f1248427633c75ad6f37533
license declaredunknown
license concludedunknown
authorsoofbaroomf
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
fused-epilogue
self.epi_tile = sm100_utils.compute_epilogue_tile_shape(mbarrier
self.epilog_sync_barrier = pipeline.NamedBarrier(persistent-kernel
tile_sched_params: utils.PersistentTileSchedulerParams,shared-memory
self.smem_capacity = utils.get_smem_capacity_in_bytes("sm_100")tcgen05
tcgen05.CtaGroup.TWO if self.use_2cta_instrs else tcgen05.CtaGroup.ONEwarp-specialization
ab_pipeline_producer_group = pipeline.CooperativeGroup(pipeline.Agent.Thread)Kernel source
submission.py3621 lines
import os
import sys
import importlib.util
from inspect import isclass
from typing import Any, List, Tuple, Type, Union
import torch
from task import input_t, output_t
_ROOT = os.path.dirname(__file__)
_LOCAL_CUTE = os.path.abspath(
os.path.join(
_ROOT,
"..",
"..",
"..",
"..",
"third_party",
"cutlass",
"python",
"CuTeDSL",
)
)
if os.path.isdir(_LOCAL_CUTE) and _LOCAL_CUTE not in sys.path:
sys.path.insert(0, _LOCAL_CUTE)
_CACHE_DIR = os.path.abspath(
os.path.join(_ROOT, "..", "..", "..", "..", "kernel_context", "cute_cache")
)
try:
os.makedirs(_CACHE_DIR, exist_ok=True)
except Exception:
_CACHE_DIR = os.path.join("/tmp", "cute_cache")
os.makedirs(_CACHE_DIR, exist_ok=True)
os.environ.setdefault("CUTE_DSL_CACHE_DIR", _CACHE_DIR)
os.environ.setdefault("CUTE_DSL_JIT_CACHE", _CACHE_DIR)
os.environ.setdefault("CUTE_DSL_KEEP_PTX", "1")
os.environ.setdefault("CUTE_DSL_KEEP_CUBIN", "1")
import cutlass
import cutlass.cute as cute
import cutlass._mlir.dialects.cute as _cute_ir
from cutlass._mlir import ir as _ir
from cutlass._mlir.dialects import llvm as _llvm
from cutlass.cute.nvgpu import cpasync, tcgen05
import cutlass.torch as cutlass_torch
import cutlass.utils as utils
import cutlass.pipeline as pipeline
from cutlass.pipeline import pipeline_init_arrive, pipeline_init_wait
import cutlass.utils.blackwell_helpers as sm100_utils
import cutlass.utils.blockscaled_layout as blockscaled_utils
from cutlass.cute.runtime import from_dlpack
from cutlass.cutlass_dsl import dsl_user_op
_kernel_cache: dict[tuple, Any] = {}
_fake_st: Any | None = None
_stage8_mod: Any | None = None
_stage8_checked = False
_enable_stage8 = os.environ.get("NVFP4_ENABLE_STAGE8", "") == "1"
_speedk_mod: Any | None = None
_speedk_checked = False
_enable_speedk_ext = os.environ.get("NVFP4_ENABLE_SPEEDK_EXT", "") != "0"
_speedk_diag_done = False
_speedk_ext_mod: Any | None = None
_speedk_ext_fail = False
_speedk_inc_cache: list[str] | None = None
_speedk_cutlass_root: str | None = None
_speedk_pad_cache: dict[tuple[int, int, torch.dtype, int], torch.Tensor] = {}
_speedk_c_pad_cache: dict[tuple[int, int, torch.dtype, int], torch.Tensor] = {}
_speedk_disable_ext = os.environ.get("NVFP4_SPEEDK_EXT_DISABLE", "") == "1"
_speedk_diag = os.environ.get("NVFP4_SPEEDK_DIAG", "") == "1"
def _is_sm100_device(t: torch.Tensor) -> bool:
if t.device.type != "cuda":
return False
try:
major, _minor = torch.cuda.get_device_capability(t.device)
except Exception:
return False
return major >= 10
def _try_stage8_ext(
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
):
global _stage8_mod, _stage8_checked
if _stage8_checked and _stage8_mod is None:
return None
if _stage8_mod is None:
_stage8_checked = True
try:
stage8_path = os.path.join(_ROOT, "submission_b200_stage8.py")
if not os.path.isfile(stage8_path):
return None
spec = importlib.util.spec_from_file_location(
"nvfp4_group_gemm_stage8_ext", stage8_path
)
if spec is None or spec.loader is None:
return None
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
_stage8_mod = mod
except Exception:
_stage8_mod = None
return None
try:
return _stage8_mod._try_ext(abc_tensors, sfs_reordered, problem_sizes)
except Exception:
return None
def _try_speedk_ext(
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
):
global _speedk_ext_mod, _speedk_checked, _speedk_ext_fail, _speedk_diag_done
def _diag(msg: str):
global _speedk_diag_done
if not _speedk_diag or _speedk_diag_done:
return
_speedk_diag_done = True
print(msg, file=sys.stderr, flush=True)
if _speedk_disable_ext or (not _enable_speedk_ext) or (not abc_tensors):
return None
# This kernel is SM100-only. Avoid any JIT/compile work on other GPUs.
if not _is_sm100_device(abc_tensors[0][0]):
return None
if _speedk_ext_fail:
return None
if _speedk_ext_mod is None:
_speedk_checked = True
_speedk_ext_mod = _speedk_load_ext(_diag)
if _speedk_ext_mod is None:
_speedk_ext_fail = True
_diag("speedk ext: build fail")
return None
out = _speedk_try_ext(_speedk_ext_mod, abc_tensors, sfs_reordered, problem_sizes)
if out is None:
_diag("speedk ext: none")
return None
_diag("speedk ext: ok")
return True
_SPEEDK_CPP_HEX = "0a2020202023696e636c756465203c746f7263682f657874656e73696f6e2e683e0a2020202023696e636c756465203c766563746f723e0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f31736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620534642293b0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f32736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620534642293b0a20202020"
_SPEEDK_CU_HEX = "0a2020202023696e636c756465203c746f7263682f657874656e73696f6e2e683e0a2020202023696e636c756465203c4154656e2f637564612f43554441436f6e746578742e683e0a2020202023696e636c756465203c6331302f637564612f4355444153747265616d2e683e0a2020202023696e636c756465203c6331302f637564612f4355444147756172642e683e0a2020202023696e636c756465203c766563746f723e0a2020202023696e636c756465203c7374646578636570743e0a2020202023696e636c756465203c747970655f7472616974733e0a0a20202020236966646566205f5f435544415f4e4f5f48414c465f4f50455241544f52535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c465f4f50455241544f52535f5f0a2020202023656e6469660a20202020236966646566205f5f435544415f4e4f5f48414c465f434f4e56455253494f4e535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c465f434f4e56455253494f4e535f5f0a2020202023656e6469660a20202020236966646566205f5f435544415f4e4f5f48414c46325f4f50455241544f52535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c46325f4f50455241544f52535f5f0a2020202023656e6469660a0a2020202023696e636c756465203c637564615f72756e74696d652e683e0a2020202023696620646566696e6564285f5f435544415f415243485f5f292026262021646566696e6564284355544c4153535f415243485f4d4d415f534d313030415f454e41424c4544290a2020202023646566696e65204355544c4153535f415243485f4d4d415f534d313030415f454e41424c454420310a2020202023656e6469660a2020202023696e636c75646520226375746c6173732f6375746c6173732e68220a2020202023696e636c7564652022637574652f74656e736f722e687070220a2020202023696e636c75646520226375746c6173732f74656e736f725f7265662e68220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f636f6c6c6563746976652f64656661756c745f6570696c6f6775652e687070220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f7468726561642f6c696e6561725f636f6d62696e6174696f6e2e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f64697370617463685f706f6c6963792e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f67726f75705f61727261795f70726f626c656d5f73686170652e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f636f6c6c6563746976652f636f6c6c6563746976655f6275696c6465722e687070220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f636f6c6c6563746976652f636f6c6c6563746976655f6275696c6465722e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f6465766963652f67656d6d5f756e6976657273616c5f616461707465722e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f6b65726e656c2f67656d6d5f756e6976657273616c2e687070220a2020202023696e636c75646520226375746c6173732f7574696c2f7061636b65645f7374726964652e687070220a2020202023696e636c75646520226375746c6173732f6b65726e656c5f68617264776172655f696e666f2e68220a2020202023696e636c75646520226375746c6173732f7574696c2f6465766963655f6d656d6f72792e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f6b65726e656c2f74696c655f7363686564756c65725f706172616d732e68220a0a2020202023646566696e65204355544c4153535f434845434b287374617475732920202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020646f207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a20202020202020206375746c6173733a3a537461747573205f737461747573203d2028737461747573293b202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020696620285f73746174757320213d206375746c6173733a3a5374617475733a3a6b5375636365737329207b20202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020202020207468726f77207374643a3a72756e74696d655f6572726f7228224355544c415353206572726f7222293b202020202020202020202020202020202020202020202020202020202020202020205c0a20202020202020207d20202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020207d207768696c65202830290a0a2020202023646566696e6520435544415f434845434b2865787072292020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020646f207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020637564614572726f725f74205f657272203d202865787072293b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020696620285f65727220213d20637564615375636365737329207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020202020207468726f77207374643a3a72756e74696d655f6572726f7228637564614765744572726f72537472696e67285f65727229293b202020202020202020202020202020202020202020202020205c0a20202020202020207d20202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020207d207768696c65202830290a0a202020207573696e672050726f626c656d5368617065203d206375746c6173733a3a67656d6d3a3a47726f757050726f626c656d53686170653c637574653a3a53686170653c696e742c696e742c696e743e3e3b0a202020207573696e6720456c656d656e74496e707574203d206375746c6173733a3a666c6f61745f65326d315f743b0a202020207573696e6720456c656d656e745346202020203d206375746c6173733a3a666c6f61745f7565346d335f743b0a202020207573696e6720456c656d656e744320202020203d206375746c6173733a3a68616c665f743b0a0a202020207573696e6720456c656d656e7441203d206375746c6173733a3a6e765f666c6f6174345f743c456c656d656e74496e7075743e3b0a202020207573696e67204c61796f75744120203d206375746c6173733a3a6c61796f75743a3a526f774d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744120203d2033323b0a0a202020207573696e6720456c656d656e7442203d206375746c6173733a3a6e765f666c6f6174345f743c456c656d656e74496e7075743e3b0a202020207573696e67204c61796f75744220203d206375746c6173733a3a6c61796f75743a3a436f6c756d6e4d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744220203d2033323b0a0a202020207573696e6720456c656d656e7444203d20456c656d656e74433b0a202020207573696e67204c61796f75744320203d206375746c6173733a3a6c61796f75743a3a436f6c756d6e4d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744320203d20323536202f206375746c6173733a3a73697a656f665f626974733c456c656d656e74433e3a3a76616c75653b0a20202020636f6e73746578707220696e7420416c69676e6d656e744420203d20323536202f206375746c6173733a3a73697a656f665f626974733c456c656d656e74443e3a3a76616c75653b0a202020207573696e6720456c656d656e74416363756d756c61746f7220203d20666c6f61743b0a0a202020207573696e672041726368546167203d206375746c6173733a3a617263683a3a536d3130303b0a202020207573696e67204f70657261746f72436c617373203d206375746c6173733a3a617263683a3a4f70436c617373426c6f636b5363616c656454656e736f724f703b0a202020207573696e67205374616765436f756e745479706531536d203d206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a5374616765436f756e743c343e3b0a202020207573696e67205374616765436f756e745479706532536d203d206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a5374616765436f756e743c353e3b0a202020207573696e6720436c75737465725368617065203d20637574653a3a53686170653c696e7433325f742c696e7433325f742c637574653a3a5f313e3b0a0a20202020737472756374204d4d4131534d436f6e666967207b0a2020202020207573696e67204d6d6154696c65536861706520202020203d20637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36342c637574653a3a5f3235363e3b0a2020202020207573696e67204b65726e656c5363686564756c652020203d206375746c6173733a3a67656d6d3a3a4b65726e656c5074724172726179546d61576172705370656369616c697a656431536d4e766634536d3130303b0a2020202020207573696e67204570696c6f6775655363686564756c65203d206375746c6173733a3a6570696c6f6775653a3a5074724172726179546d61576172705370656369616c697a656431536d3b0a202020207d3b0a0a20202020737472756374204d4d4132534d436f6e666967207b0a2020202020207573696e67204d6d6154696c65536861706520202020203d20637574653a3a53686170653c637574653a3a5f3235362c637574653a3a5f36342c637574653a3a5f3235363e3b0a2020202020207573696e67204b65726e656c5363686564756c652020203d206375746c6173733a3a67656d6d3a3a4b65726e656c5074724172726179546d61576172705370656369616c697a656432536d4e766634536d3130303b0a2020202020207573696e67204570696c6f6775655363686564756c65203d206375746c6173733a3a6570696c6f6775653a3a5074724172726179546d61576172705370656369616c697a656432536d3b0a202020207d3b0a0a202020207573696e6720436f6c6c6563746976654570696c6f67756531534d203d20747970656e616d65206375746c6173733a3a6570696c6f6775653a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a2020202020202020417263685461672c204f70657261746f72436c6173732c0a2020202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020202020637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36343e2c0a2020202020202020456c656d656e74416363756d756c61746f722c20456c656d656e74416363756d756c61746f722c0a2020202020202020456c656d656e74432c204c61796f757443202a2c20416c69676e6d656e74432c0a2020202020202020456c656d656e74442c204c61796f757443202a2c20416c69676e6d656e74442c0a2020202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4570696c6f6775655363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e6720436f6c6c6563746976654d61696e6c6f6f7031534d203d20747970656e616d65206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a202020202020417263685461672c204f70657261746f72436c6173732c0a202020202020456c656d656e74412c204c61796f757441202a2c20416c69676e6d656e74412c0a202020202020456c656d656e74422c204c61796f757442202a2c20416c69676e6d656e74422c0a202020202020456c656d656e74416363756d756c61746f722c0a202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020205374616765436f756e745479706531536d2c0a202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4b65726e656c5363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e672047656d6d4b65726e656c31534d203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a47656d6d556e6976657273616c3c0a202020202020202050726f626c656d53686170652c0a2020202020202020436f6c6c6563746976654d61696e6c6f6f7031534d2c0a2020202020202020436f6c6c6563746976654570696c6f67756531534d2c0a20202020202020206375746c6173733a3a67656d6d3a3a53747265616d4b5363686564756c65720a202020203e3b0a202020207573696e672047656d6d31534d203d206375746c6173733a3a67656d6d3a3a6465766963653a3a47656d6d556e6976657273616c416461707465723c47656d6d4b65726e656c31534d3e3b0a0a202020207573696e6720436f6c6c6563746976654570696c6f67756532534d203d20747970656e616d65206375746c6173733a3a6570696c6f6775653a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a2020202020202020417263685461672c204f70657261746f72436c6173732c0a2020202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020202020637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36343e2c0a2020202020202020456c656d656e74416363756d756c61746f722c20456c656d656e74416363756d756c61746f722c0a2020202020202020456c656d656e74432c204c61796f757443202a2c20416c69676e6d656e74432c0a2020202020202020456c656d656e74442c204c61796f757443202a2c20416c69676e6d656e74442c0a2020202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4570696c6f6775655363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e6720436f6c6c6563746976654d61696e6c6f6f7032534d203d20747970656e616d65206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a202020202020417263685461672c204f70657261746f72436c6173732c0a202020202020456c656d656e74412c204c61796f757441202a2c20416c69676e6d656e74412c0a202020202020456c656d656e74422c204c61796f757442202a2c20416c69676e6d656e74422c0a202020202020456c656d656e74416363756d756c61746f722c0a202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020205374616765436f756e745479706532536d2c0a202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4b65726e656c5363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e672047656d6d4b65726e656c32534d203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a47656d6d556e6976657273616c3c0a202020202020202050726f626c656d53686170652c0a2020202020202020436f6c6c6563746976654d61696e6c6f6f7032534d2c0a2020202020202020436f6c6c6563746976654570696c6f67756532534d2c0a20202020202020206375746c6173733a3a67656d6d3a3a53747265616d4b5363686564756c65720a202020203e3b0a202020207573696e672047656d6d32534d203d206375746c6173733a3a67656d6d3a3a6465766963653a3a47656d6d556e6976657273616c416461707465723c47656d6d4b65726e656c32534d3e3b0a0a2020202074656d706c617465203c747970656e616d6520543e0a202020207374727563742050696e6e6564486f7374427566666572207b0a202020202020542a20707472203d206e756c6c7074723b0a20202020202073697a655f7420636f756e74203d20303b0a0a20202020202050696e6e6564486f73744275666665722829203d2064656661756c743b0a2020202020206578706c696369742050696e6e6564486f73744275666665722873697a655f74206e29207b20616c6c6f63617465286e293b207d0a0a202020202020766f696420616c6c6f636174652873697a655f74206e29207b0a20202020202020206966202870747220262620636f756e74203e3d206e29207b0a2020202020202020202072657475726e3b0a20202020202020207d0a202020202020202072656c6561736528293b0a2020202020202020636f756e74203d206e3b0a2020202020202020435544415f434845434b2863756461486f7374416c6c6f63287265696e746572707265745f636173743c766f69642a2a3e2826707472292c206e202a2073697a656f662854292c2063756461486f7374416c6c6f63506f727461626c6529293b0a2020202020207d0a0a202020202020766f69642072656c656173652829207b0a20202020202020206966202870747229207b0a202020202020202020206375646146726565486f737428707472293b0a20202020202020202020707472203d206e756c6c7074723b0a20202020202020202020636f756e74203d20303b0a20202020202020207d0a2020202020207d0a0a2020202020207e50696e6e6564486f73744275666665722829207b2072656c6561736528293b207d0a0a202020202020542a2064617461282920636f6e7374207b2072657475726e207074723b207d0a2020202020205426206f70657261746f725b5d2873697a655f742069647829207b2072657475726e207074725b6964785d3b207d0a202020207d3b0a0a2020202074656d706c617465203c747970656e616d6520543e0a2020202073747275637420446576696365427566666572207b0a202020202020542a20707472203d206e756c6c7074723b0a20202020202073697a655f7420636f756e74203d20303b0a0a2020202020204465766963654275666665722829203d2064656661756c743b0a2020202020206578706c69636974204465766963654275666665722873697a655f74206e29207b20616c6c6f63617465286e293b207d0a0a202020202020766f696420616c6c6f636174652873697a655f74206e29207b0a20202020202020206966202870747220262620636f756e74203e3d206e29207b0a2020202020202020202072657475726e3b0a20202020202020207d0a202020202020202072656c6561736528293b0a2020202020202020636f756e74203d206e3b0a2020202020202020435544415f434845434b28637564614d616c6c6f63287265696e746572707265745f636173743c766f69642a2a3e2826707472292c206e202a2073697a656f6628542929293b0a2020202020207d0a0a202020202020766f69642072656c656173652829207b0a20202020202020206966202870747229207b0a20202020202020202020637564614672656528707472293b0a20202020202020202020707472203d206e756c6c7074723b0a20202020202020202020636f756e74203d20303b0a20202020202020207d0a2020202020207d0a0a2020202020207e4465766963654275666665722829207b2072656c6561736528293b207d0a0a202020202020542a20676574282920636f6e7374207b2072657475726e207074723b207d0a0a202020202020766f696420636f70795f66726f6d5f686f737428636f6e737420542a20686f73742c2073697a655f74206e2c206375646153747265616d5f742073747265616d29207b0a2020202020202020435544415f434845434b28637564614d656d6370794173796e63287074722c20686f73742c206e202a2073697a656f662854292c20637564614d656d637079486f7374546f4465766963652c2073747265616d29293b0a2020202020207d0a202020207d3b0a0a2020202074656d706c617465203c747970656e616d652047656d6d3e0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2072756e5f67726f757065645f696d706c280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a0a202020202020544f5243485f434845434b28412e73697a652829203d3d20422e73697a6528292c2022412f422073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d20432e73697a6528292c2022412f432073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d205346412e73697a6528292c2022412f5346412073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d205346422e73697a6528292c2022412f5346422073697a65206d69736d6174636822293b0a202020202020636f6e737420696e7433325f742067726f757073203d207374617469635f636173743c696e7433325f743e28412e73697a652829293b0a202020202020544f5243485f434845434b2867726f757073203e20302c20224e6f2067726f75707322293b0a0a202020202020696e74206465766963655f6964203d20415b305d2e6765745f64657669636528293b0a2020202020206331303a3a637564613a3a435544414775617264206465766963655f6775617264286465766963655f6964293b0a0a2020202020207573696e672053747269646541203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465413b0a2020202020207573696e672053747269646542203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465423b0a2020202020207573696e672053747269646543203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465433b0a2020202020207573696e672053747269646544203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465443b0a2020202020207573696e67204c61796f7574534641203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a496e7465726e616c4c61796f75745346413b0a2020202020207573696e67204c61796f7574534642203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a496e7465726e616c4c61796f75745346423b0a2020202020207573696e6720536d317878426c6b5363616c6564436f6e666967203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a536d317878426c6b5363616c6564436f6e6669673b0a0a2020202020207374617469632073697a655f74206361706163697479203d20303b0a20202020202073746174696320696e74206361636865645f646576696365203d202d313b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e207074725f415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e207074725f425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e207074725f435f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e207074725f445f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465413e207374726964655f415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465423e207374726964655f425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465433e207374726964655f435f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465443e207374726964655f445f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c4c61796f75745346413e206c61796f75745f5346415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c4c61796f75745346423e206c61796f75745f5346425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652050726f626c656d53686170653a3a556e6465726c79696e6750726f626c656d53686170653e2070726f626c656d5f73697a65735f686f73743b0a0a202020202020737461746963204465766963654275666665723c747970656e616d652050726f626c656d53686170653a3a556e6465726c79696e6750726f626c656d53686170653e2070726f626c656d5f73697a65735f6465766963653b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e207074725f413b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e207074725f423b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346413b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346423b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e207074725f433b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e207074725f443b0a202020202020737461746963204465766963654275666665723c537472696465413e207374726964655f413b0a202020202020737461746963204465766963654275666665723c537472696465423e207374726964655f423b0a202020202020737461746963204465766963654275666665723c537472696465433e207374726964655f433b0a202020202020737461746963204465766963654275666665723c537472696465443e207374726964655f443b0a202020202020737461746963204465766963654275666665723c4c61796f75745346413e206c61796f75745f5346413b0a202020202020737461746963204465766963654275666665723c4c61796f75745346423e206c61796f75745f5346423b0a2020202020207374617469632073697a655f7420776f726b73706163655f73697a655f636163686564203d20303b0a20202020202073746174696320746f7263683a3a54656e736f7220776f726b73706163655f74656e736f723b0a0a202020202020696620286361636865645f64657669636520213d206465766963655f6964207c7c206361706163697479203c207374617469635f636173743c73697a655f743e2867726f7570732929207b0a20202020202020206361636865645f646576696365203d206465766963655f69643b0a20202020202020206361706163697479203d207374617469635f636173743c73697a655f743e2867726f757073293b0a20202020202020207074725f415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f435f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f445f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f435f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f445f686f73742e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346425f686f73742e616c6c6f636174652867726f757073293b0a202020202020202070726f626c656d5f73697a65735f686f73742e616c6c6f636174652867726f757073293b0a0a202020202020202070726f626c656d5f73697a65735f6465766963652e616c6c6f636174652867726f757073293b0a20202020202020207074725f412e616c6c6f636174652867726f757073293b0a20202020202020207074725f422e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346412e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346422e616c6c6f636174652867726f757073293b0a20202020202020207074725f432e616c6c6f636174652867726f757073293b0a20202020202020207074725f442e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f412e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f422e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f432e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f442e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346412e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346422e616c6c6f636174652867726f757073293b0a2020202020207d0a0a202020202020636f6e73746578707220696e742074696c655f6d203d20637574653a3a73697a653c303e28747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c6553686170657b7d293b0a202020202020636f6e73746578707220696e742074696c655f6e203d20637574653a3a73697a653c313e28747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c6553686170657b7d293b0a202020202020696e7436345f7420746f74616c5f74696c6573203d20303b0a0a202020202020666f722028696e7433325f742069203d20303b2069203c2067726f7570733b202b2b6929207b0a2020202020202020544f5243485f434845434b28415b695d2e69735f6375646128292c202241206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b28425b695d2e69735f6375646128292c202242206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b28435b695d2e69735f6375646128292c202243206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b285346415b695d2e69735f6375646128292c2022534641206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b285346425b695d2e69735f6375646128292c2022534642206d757374206265204355444122293b0a0a2020202020202020696e7433325f74206d203d207374617469635f636173743c696e7433325f743e28415b695d2e73697a65283029293b0a2020202020202020696e7433325f74206b5f7061636b6564203d207374617469635f636173743c696e7433325f743e28415b695d2e73697a65283129293b0a2020202020202020696e7433325f74206b203d206b5f7061636b6564202a20323b0a2020202020202020696e7433325f74206e203d207374617469635f636173743c696e7433325f743e28425b695d2e73697a65283029293b0a2020202020202020544f5243485f434845434b28425b695d2e73697a65283129202a2032203d3d206b2c20224b206d69736d6174636822293b0a2020202020202020544f5243485f434845434b28435b695d2e73697a65283029203d3d206d20262620435b695d2e73697a65283129203d3d206e2c202243207368617065206d69736d6174636822293b0a0a202020202020202070726f626c656d5f73697a65735f686f73745b695d203d20637574653a3a6d616b655f7368617065286e2c206d2c206b293b0a20202020202020207374726964655f415f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465417b7d2c207b6e2c206b2c20317d293b0a20202020202020207374726964655f425f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465427b7d2c207b6d2c206b2c20317d293b0a20202020202020207374726964655f435f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465437b7d2c207b6e2c206d2c20317d293b0a20202020202020207374726964655f445f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465447b7d2c207b6e2c206d2c20317d293b0a20202020202020206c61796f75745f5346415f686f73745b695d203d20536d317878426c6b5363616c6564436f6e6669673a3a74696c655f61746f6d5f746f5f73686170655f53464128637574653a3a6d616b655f7368617065286e2c206d2c206b2c203129293b0a20202020202020206c61796f75745f5346425f686f73745b695d203d20536d317878426c6b5363616c6564436f6e6669673a3a74696c655f61746f6d5f746f5f73686170655f53464228637574653a3a6d616b655f7368617065286e2c206d2c206b2c203129293b0a0a2020202020202020696e742074696c65735f6d203d20286e202b2074696c655f6d202d203129202f2074696c655f6d3b0a2020202020202020696e742074696c65735f6e203d20286d202b2074696c655f6e202d203129202f2074696c655f6e3b0a2020202020202020746f74616c5f74696c6573202b3d207374617469635f636173743c696e7436345f743e2874696c65735f6d29202a207374617469635f636173743c696e7436345f743e2874696c65735f6e293b0a0a20202020202020207074725f415f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e28425b695d2e646174615f7074722829293b0a20202020202020207074725f425f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e28415b695d2e646174615f7074722829293b0a20202020202020207074725f435f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e28435b695d2e646174615f7074722829293b0a20202020202020207074725f445f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e28435b695d2e646174615f7074722829293b0a20202020202020207074725f5346415f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e285346425b695d2e646174615f7074722829293b0a20202020202020207074725f5346425f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e285346415b695d2e646174615f7074722829293b0a2020202020207d0a0a2020202020206175746f2073747265616d5f6f626a203d206331303a3a637564613a3a67657443757272656e744355444153747265616d286465766963655f6964293b0a2020202020206331303a3a637564613a3a4355444153747265616d47756172642073747265616d5f67756172642873747265616d5f6f626a293b0a2020202020206375646153747265616d5f742073747265616d203d2073747265616d5f6f626a2e73747265616d28293b0a20202020202070726f626c656d5f73697a65735f6465766963652e636f70795f66726f6d5f686f73742870726f626c656d5f73697a65735f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f412e636f70795f66726f6d5f686f7374287074725f415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f422e636f70795f66726f6d5f686f7374287074725f425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f5346412e636f70795f66726f6d5f686f7374287074725f5346415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f5346422e636f70795f66726f6d5f686f7374287074725f5346425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f432e636f70795f66726f6d5f686f7374287074725f435f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f442e636f70795f66726f6d5f686f7374287074725f445f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f412e636f70795f66726f6d5f686f7374287374726964655f415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f422e636f70795f66726f6d5f686f7374287374726964655f425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f432e636f70795f66726f6d5f686f7374287374726964655f435f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f442e636f70795f66726f6d5f686f7374287374726964655f445f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020206c61796f75745f5346412e636f70795f66726f6d5f686f7374286c61796f75745f5346415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020206c61796f75745f5346422e636f70795f66726f6d5f686f7374286c61796f75745f5346425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a0a2020202020206375746c6173733a3a4b65726e656c4861726477617265496e666f2068775f696e666f3b0a20202020202068775f696e666f2e6465766963655f6964203d206465766963655f69643b0a20202020202068775f696e666f2e736d5f636f756e74203d206375746c6173733a3a4b65726e656c4861726477617265496e666f3a3a71756572795f6465766963655f6d756c746970726f636573736f725f636f756e742868775f696e666f2e6465766963655f6964293b0a202020202020696620636f6e73746578707220287374643a3a69735f73616d655f763c47656d6d2c2047656d6d31534d3e29207b0a202020202020202068775f696e666f2e636c75737465725f7368617065203d2064696d3328312c20312c2031293b0a202020202020202068775f696e666f2e636c75737465725f73686170655f66616c6c6261636b203d2064696d3328312c20312c2031293b0a2020202020207d20656c7365207b0a202020202020202068775f696e666f2e636c75737465725f7368617065203d2064696d3328322c20312c2031293b0a202020202020202068775f696e666f2e636c75737465725f73686170655f66616c6c6261636b203d2064696d3328322c20312c2031293b0a2020202020207d0a0a202020202020747970656e616d652047656d6d3a3a417267756d656e747320617267756d656e74733b0a2020202020206465636c7479706528617267756d656e74732e6570696c6f6775652e7468726561642920667573696f6e5f617267733b0a202020202020667573696f6e5f617267732e616c7068615f707472203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e626574615f707472203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e616c706861203d20456c656d656e74416363756d756c61746f722831293b0a202020202020667573696f6e5f617267732e62657461203d20456c656d656e74416363756d756c61746f722830293b0a202020202020667573696f6e5f617267732e616c7068615f7074725f6172726179203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e626574615f7074725f6172726179203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e64416c706861203d207b637574653a3a5f307b7d2c20637574653a3a5f307b7d2c20307d3b0a202020202020667573696f6e5f617267732e6442657461203d207b637574653a3a5f307b7d2c20637574653a3a5f307b7d2c20307d3b0a0a202020202020747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c655363686564756c6572417267756d656e7473207363686564756c65723b0a2020202020207363686564756c65722e6d61785f7377697a7a6c655f73697a65203d20383b0a2020202020207363686564756c65722e7261737465725f6f72646572203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a64657461696c3a3a5261737465724f726465724f7074696f6e733a3a4865757269737469633b0a0a202020202020617267756d656e7473203d20747970656e616d652047656d6d3a3a417267756d656e7473207b0a20202020202020206375746c6173733a3a67656d6d3a3a47656d6d556e6976657273616c4d6f64653a3a6b47726f757065642c0a20202020202020207b67726f7570732c2070726f626c656d5f73697a65735f6465766963652e67657428292c206e756c6c7074727d2c0a20202020202020207b7074725f412e67657428292c207374726964655f412e67657428292c207074725f422e67657428292c207374726964655f422e67657428292c0a2020202020202020207074725f5346412e67657428292c206c61796f75745f5346412e67657428292c207074725f5346422e67657428292c206c61796f75745f5346422e67657428297d2c0a20202020202020207b667573696f6e5f617267732c207074725f432e67657428292c207374726964655f432e67657428292c207074725f442e67657428292c207374726964655f442e67657428297d2c0a202020202020202068775f696e666f2c207363686564756c65720a2020202020207d3b0a0a20202020202047656d6d2067656d6d3b0a20202020202073697a655f7420776f726b73706163655f73697a65203d2047656d6d3a3a6765745f776f726b73706163655f73697a6528617267756d656e7473293b0a2020202020206966202821776f726b73706163655f74656e736f722e646566696e65642829207c7c206361636865645f64657669636520213d206465766963655f6964207c7c20776f726b73706163655f73697a655f636163686564203c20776f726b73706163655f73697a6529207b0a20202020202020206175746f206f707473203d20746f7263683a3a54656e736f724f7074696f6e7328292e64657669636528746f7263683a3a6b435544412c206465766963655f6964292e647479706528746f7263683a3a6b55496e7438293b0a2020202020202020776f726b73706163655f74656e736f72203d20746f7263683a3a656d707479287b7374617469635f636173743c6c6f6e673e28776f726b73706163655f73697a65297d2c206f707473293b0a2020202020202020776f726b73706163655f73697a655f636163686564203d20776f726b73706163655f73697a653b0a2020202020207d0a202020202020766f69642a20776f726b73706163655f707472203d20776f726b73706163655f73697a65203f20776f726b73706163655f74656e736f722e646174615f7074722829203a206e756c6c7074723b0a0a2020202020204355544c4153535f434845434b2867656d6d2e63616e5f696d706c656d656e7428617267756d656e747329293b0a2020202020204355544c4153535f434845434b2867656d6d2e696e697469616c697a6528617267756d656e74732c20776f726b73706163655f7074722c2073747265616d29293b0a2020202020206175746f2072756e5f737461747573203d2067656d6d2e72756e2873747265616d2c206e756c6c7074722c2074727565293b0a2020202020206966202872756e5f73746174757320213d206375746c6173733a3a5374617475733a3a6b5375636365737329207b0a202020202020202072756e5f737461747573203d2067656d6d2e72756e2873747265616d293b0a2020202020207d0a2020202020204355544c4153535f434845434b2872756e5f737461747573293b0a0a20202020202072657475726e20433b0a202020207d0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f31736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a20202020202072657475726e2072756e5f67726f757065645f696d706c3c47656d6d31534d3e28412c20422c20432c205346412c20534642293b0a202020207d0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f32736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a20202020202072657475726e2072756e5f67726f757065645f696d706c3c47656d6d32534d3e28412c20422c20432c205346412c20534642293b0a202020207d0a202020200a"
def _speedk_dec_hex(h: str) -> str:
return bytes.fromhex(h).decode("utf-8")
def _speedk_repo_root() -> str:
return os.path.abspath(os.path.join(_ROOT, "..", "..", "..", ".."))
def _speedk_ensure_cutlass_dir(diag_fn) -> str | None:
global _speedk_cutlass_root
if _speedk_cutlass_root is not None:
return _speedk_cutlass_root
repo_root = _speedk_repo_root()
base_dir = os.path.join(repo_root, "tmp", "cutlass_cache")
root = os.path.join(base_dir, "cutlass-4.3.5")
inc = os.path.join(root, "include", "cutlass", "cutlass.h")
if os.path.isfile(inc):
_speedk_cutlass_root = root
return root
try:
os.makedirs(base_dir, exist_ok=True)
tgz_path = os.path.join(base_dir, "cutlass.tgz")
if not os.path.isfile(tgz_path):
import urllib.request
url = "https://github.com/NVIDIA/cutlass/archive/refs/tags/v4.3.5.tar.gz"
with urllib.request.urlopen(url, timeout=60) as resp:
data = resp.read()
with open(tgz_path, "wb") as f:
f.write(data)
import tarfile
with tarfile.open(tgz_path, "r:gz") as tf:
tf.extractall(base_dir)
if os.path.isfile(inc):
_speedk_cutlass_root = root
return root
except Exception:
diag_fn("speedk ext: cutlass fetch fail")
return None
return None
def _speedk_find_inc(diag_fn) -> list[str]:
global _speedk_inc_cache
if _speedk_inc_cache is not None:
return _speedk_inc_cache
incs: list[str] = []
def _add_cutlass(base: str) -> bool:
inc = os.path.join(base, "include")
if os.path.isfile(os.path.join(inc, "cutlass", "cutlass.h")):
incs.append(inc)
util_inc = os.path.join(base, "tools", "util", "include")
if os.path.isfile(os.path.join(util_inc, "cutlass", "util", "device_memory.h")):
incs.append(util_inc)
return True
inc2 = os.path.join(base, "cutlass", "include")
if os.path.isfile(os.path.join(inc2, "cutlass", "cutlass.h")):
incs.append(inc2)
util_inc2 = os.path.join(base, "cutlass", "tools", "util", "include")
if os.path.isfile(os.path.join(util_inc2, "cutlass", "util", "device_memory.h")):
incs.append(util_inc2)
return True
return False
repo_root = _speedk_repo_root()
for base in (
os.path.join(repo_root, "third_party", "cutlass"),
os.path.join(repo_root, "tmp", "cutlass-4.3.5"),
):
_add_cutlass(base)
try:
root = os.path.dirname(getattr(cutlass, "__file__", "") or "")
if root:
bases = [root, os.path.dirname(root), os.path.dirname(os.path.dirname(root))]
for base in bases:
if _add_cutlass(base):
break
except Exception:
pass
try:
import torch.utils.cpp_extension as _ce
for inc in _ce.include_paths():
if os.path.isfile(os.path.join(inc, "cutlass", "cutlass.h")):
incs.append(inc)
util_inc3 = os.path.join(inc, "cutlass", "tools", "util", "include")
if os.path.isfile(os.path.join(util_inc3, "cutlass", "util", "device_memory.h")):
incs.append(util_inc3)
except Exception:
pass
if not incs:
root = _speedk_ensure_cutlass_dir(diag_fn)
if root:
_add_cutlass(root)
incs = list(dict.fromkeys(incs))
if not incs:
diag_fn("speedk ext: no headers")
_speedk_inc_cache = incs
return incs
def _speedk_load_ext(diag_fn):
global _speedk_ext_mod, _speedk_ext_fail
if _speedk_ext_mod is not None or _speedk_ext_fail:
return _speedk_ext_mod
incs = _speedk_find_inc(diag_fn)
if not incs:
_speedk_ext_fail = True
return None
try:
from torch.utils.cpp_extension import load_inline
repo_root = _speedk_repo_root()
os.environ.setdefault(
"TORCH_EXTENSIONS_DIR", os.path.join(repo_root, "tmp", "torch_extensions")
)
os.environ.setdefault("TORCH_CUDA_ARCH_LIST", "10.0a")
cpp_src = _speedk_dec_hex(_SPEEDK_CPP_HEX)
cu_src = _speedk_dec_hex(_SPEEDK_CU_HEX)
_speedk_ext_mod = load_inline(
name="nvfp4_grouped_cutlass_ext_speedk",
cpp_sources=cpp_src,
cuda_sources=cu_src,
functions=["grouped_gemm_1sm", "grouped_gemm_2sm"],
extra_cuda_cflags=[
"-O3",
"--use_fast_math",
"-std=c++17",
"--diag-suppress=144",
],
extra_ldflags=["-lcuda"],
extra_cflags=["-O3", "-std=c++17"],
with_cuda=True,
extra_include_paths=incs,
verbose=False,
)
except Exception:
_speedk_ext_fail = True
_speedk_ext_mod = None
diag_fn("speedk ext: build fail")
return None
return _speedk_ext_mod
def _speedk_pad_rows(tensor: torch.Tensor, new_rows: int) -> torch.Tensor:
if tensor.size(0) == new_rows:
return tensor
dev_idx = tensor.device.index if tensor.device.index is not None else -1
key = (new_rows, tensor.size(1), tensor.dtype, dev_idx)
out = _speedk_pad_cache.get(key)
if out is None or out.size(0) != new_rows or out.size(1) != tensor.size(1):
out = torch.empty((new_rows, tensor.size(1)), device=tensor.device, dtype=tensor.dtype)
_speedk_pad_cache[key] = out
# Some low-bit dtypes (notably fp4x2) don't reliably support in-place fill or
# slice assignment. Bitcast to u8 so we only move raw bytes.
if tensor.element_size() == 1:
src = tensor if tensor.is_contiguous() else tensor.contiguous()
out_u8 = out.view(torch.uint8)
out_u8.zero_()
out_u8[: src.size(0), :] = src.view(torch.uint8)
return out
out.zero_()
out[: tensor.size(0), :] = tensor
return out
def _speedk_alloc_c_pad(rows: int, cols: int, ref: torch.Tensor) -> torch.Tensor:
dev_idx = ref.device.index if ref.device.index is not None else -1
key = (rows, cols, ref.dtype, dev_idx)
out = _speedk_c_pad_cache.get(key)
if out is None or out.size(0) != rows or out.size(1) != cols:
out = torch.empty((rows, cols), device=ref.device, dtype=ref.dtype)
_speedk_c_pad_cache[key] = out
return out
def _speedk_try_ext(
ext_mod,
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
):
if ext_mod is None:
return None
def _use_2sm(n_orig: int, k: int) -> bool:
return n_orig >= 3072 and k >= 2048
items = []
for (a_ref, b_ref, c_ref), (sfa_reordered, sfb_reordered), (m, _n, k, l) in zip(
abc_tensors, sfs_reordered, problem_sizes
):
if l != 1:
return None
a2 = a_ref[:, :, 0]
b2 = b_ref[:, :, 0]
c2 = c_ref[:, :, 0]
n = b2.size(0)
m_pad = ((m + 31) // 32) * 32
if m_pad != m:
a2 = _speedk_pad_rows(a2, m_pad)
c_pad = _speedk_alloc_c_pad(m_pad, n, c2)
items.append((a2, b2, c_pad, sfa_reordered, sfb_reordered, c2, m, n, k))
else:
items.append((a2, b2, c2, sfa_reordered, sfb_reordered, None, m, n, k))
def _gather(src):
return (
[x[0] for x in src],
[x[1] for x in src],
[x[2] for x in src],
[x[3] for x in src],
[x[4] for x in src],
)
small = [it for it in items if not _use_2sm(it[7], it[8])]
large = [it for it in items if _use_2sm(it[7], it[8])]
try:
if large:
a_list, b_list, c_list, sfa_list, sfb_list = _gather(large)
ext_mod.grouped_gemm_2sm(a_list, b_list, c_list, sfa_list, sfb_list)
if small:
a_list, b_list, c_list, sfa_list, sfb_list = _gather(small)
ext_mod.grouped_gemm_1sm(a_list, b_list, c_list, sfa_list, sfb_list)
except Exception:
try:
a_list, b_list, c_list, sfa_list, sfb_list = _gather(items)
ext_mod.grouped_gemm_1sm(a_list, b_list, c_list, sfa_list, sfb_list)
except Exception:
return None
for _a2, _b2, c2, _sfa, _sfb, c_orig, m, _n, _k in items:
if c_orig is not None:
c_orig.copy_(c2[:m, :])
return [c_ref for (_a, _b, c_ref) in abc_tensors]
@dsl_user_op
def _normalize_ptr(ptr, *, loc=None, ip=None):
if isinstance(ptr, _ir.Value):
return ptr
if hasattr(ptr, "to_llvm_ptr") and callable(ptr.to_llvm_ptr):
return ptr.to_llvm_ptr(loc=loc, ip=ip)
return ptr
@dsl_user_op
def _ptx_prefetch_global(ptr, *, loc=None, ip=None):
ptr = _normalize_ptr(ptr, loc=loc, ip=ip)
if not isinstance(ptr, _ir.Value):
return None
_llvm.inline_asm(
None,
[ptr],
"prefetch.global.L2::evict_last [$0];",
"l",
has_side_effects=True,
is_align_stack=False,
asm_dialect=_llvm.AsmDialect.AD_ATT,
loc=loc,
ip=ip,
)
return None
@dsl_user_op
def _ptx_prefetch_global_l1(ptr, *, loc=None, ip=None):
ptr = _normalize_ptr(ptr, loc=loc, ip=ip)
if not isinstance(ptr, _ir.Value):
return None
_llvm.inline_asm(
None,
[ptr],
"prefetch.global.L1 [$0];",
"l",
has_side_effects=True,
is_align_stack=False,
asm_dialect=_llvm.AsmDialect.AD_ATT,
loc=loc,
ip=ip,
)
return None
@dsl_user_op
def _ptx_prefetchu_l1(ptr, *, loc=None, ip=None):
ptr = _normalize_ptr(ptr, loc=loc, ip=ip)
if not isinstance(ptr, _ir.Value):
return None
_llvm.inline_asm(
None,
[ptr],
"prefetchu.L1 [$0];",
"l",
has_side_effects=True,
is_align_stack=False,
asm_dialect=_llvm.AsmDialect.AD_ATT,
loc=loc,
ip=ip,
)
return None
@dsl_user_op
def _prefetch_tma(
atom: cute.CopyAtom,
src: cute.Tensor,
tma_desc_ptr: cute.Pointer,
*,
loc=None,
ip=None,
):
if hasattr(src, "iterator"):
_ptx_prefetch_global(src.iterator, loc=loc, ip=ip)
_ptx_prefetch_global_l1(src.iterator, loc=loc, ip=ip)
if tma_desc_ptr is not None:
_ptx_prefetch_global(tma_desc_ptr, loc=loc, ip=ip)
_ptx_prefetch_global_l1(tma_desc_ptr, loc=loc, ip=ip)
_ptx_prefetchu_l1(tma_desc_ptr, loc=loc, ip=ip)
dummy_tma_bar_ptr = cute.make_ptr(
cutlass.Int64, 0, cute.AddressSpace.smem, loc=loc, ip=ip
)
dummy_mcast_mask = cutlass.Int16(0, loc=loc, ip=ip)
value = atom._unpack(
tma_bar_ptr=dummy_tma_bar_ptr,
mcast_mask=dummy_mcast_mask,
tma_desc_ptr=tma_desc_ptr,
loc=loc,
ip=ip,
)
return _cute_ir.prefetch(value, src.value, loc=loc, ip=ip)
class Sm100GroupedBlockScaledGemmKernel:
def __init__(
self,
sf_vec_size: int,
mma_tiler_mn: Tuple[int, int],
cluster_shape_mn: Tuple[int, int],
use_tma_store: bool = True,
max_ab_stage: int | None = None,
prefetch_dist: int | None = None,
force_c_stage: int | None = None,
c_assumed_align: int = 16,
c_divisibility: int = 8,
swizzle_size: int = 1,
raster_along_m: bool = True,
):
self.acc_dtype = cutlass.Float32
self.sf_vec_size = sf_vec_size
self.use_2cta_instrs = mma_tiler_mn[0] == 256
self.cluster_shape_mn = cluster_shape_mn
self.use_tma_store = use_tma_store
self.max_ab_stage = max_ab_stage
self.prefetch_dist_override = prefetch_dist
self.force_c_stage = force_c_stage
self.c_assumed_align = c_assumed_align
self.c_divisibility = c_divisibility
self.swizzle_size = swizzle_size
self.raster_along_m = raster_along_m
# K dimension is deferred in _setup_attributes
self.mma_tiler = (*mma_tiler_mn, 1)
self.cta_group = (
tcgen05.CtaGroup.TWO if self.use_2cta_instrs else tcgen05.CtaGroup.ONE
)
self.tensormap_update_mode = utils.TensorMapUpdateMode.SMEM
self.occupancy = 1
# Set specialized warp ids
self.epilog_warp_id = (
0,
1,
2,
3,
)
self.mma_warp_id = 4
self.tma_warp_id = 5
self.threads_per_cta = 32 * len(
(self.mma_warp_id, self.tma_warp_id, *self.epilog_warp_id)
)
# Set barrier for epilogue sync and tmem ptr sync
self.epilog_sync_barrier = pipeline.NamedBarrier(
barrier_id=1,
num_threads=32 * len(self.epilog_warp_id),
)
self.tmem_alloc_barrier = pipeline.NamedBarrier(
barrier_id=2,
num_threads=32 * len((self.mma_warp_id, *self.epilog_warp_id)),
)
# Barrier used by MMA/TMA warps to signal A/B tensormap initialization completion
self.tensormap_ab_init_barrier = pipeline.NamedBarrier(
barrier_id=3,
num_threads=64,
)
self.smem_capacity = utils.get_smem_capacity_in_bytes("sm_100")
SM100_TMEM_CAPACITY_COLUMNS = 512
self.num_tmem_alloc_cols = SM100_TMEM_CAPACITY_COLUMNS
self.c_tile_stride: tuple[int, int] | None = None
# Set up configurations that dependent on gemm inputs.
def _setup_attributes(self):
# Compute mma instruction shapes
# (MMA_Tile_Shape_M, MMA_Tile_Shape_N, MMA_Inst_Shape_K)
self.mma_inst_shape_mn = (
self.mma_tiler[0],
self.mma_tiler[1],
)
# (CTA_Tile_Shape_M, Round_Up(MMA_Tile_Shape_N, 128), MMA_Inst_Shape_K)
self.mma_inst_shape_mn_sfb = (
self.mma_inst_shape_mn[0] // (2 if self.use_2cta_instrs else 1),
cute.round_up(self.mma_inst_shape_mn[1], 128),
)
tiled_mma = sm100_utils.make_blockscaled_trivial_tiled_mma(
self.a_dtype,
self.a_major_mode,
self.b_major_mode,
self.sf_dtype,
self.sf_vec_size,
self.cta_group,
self.mma_inst_shape_mn,
)
tiled_mma_sfb = sm100_utils.make_blockscaled_trivial_tiled_mma(
self.a_dtype,
self.a_major_mode,
self.b_major_mode,
self.sf_dtype,
self.sf_vec_size,
cute.nvgpu.tcgen05.CtaGroup.ONE,
self.mma_inst_shape_mn_sfb,
)
# Compute mma/cluster/tile shapes
mma_inst_shape_k = cute.size(tiled_mma.shape_mnk, mode=[2])
mma_inst_tile_k = 4
self.mma_tiler = (
self.mma_inst_shape_mn[0],
self.mma_inst_shape_mn[1],
mma_inst_shape_k * mma_inst_tile_k,
)
self.mma_tiler_sfb = (
self.mma_inst_shape_mn_sfb[0],
self.mma_inst_shape_mn_sfb[1],
mma_inst_shape_k * mma_inst_tile_k,
)
self.cta_tile_shape_mnk = (
self.mma_tiler[0] // cute.size(tiled_mma.thr_id.shape),
self.mma_tiler[1],
self.mma_tiler[2],
)
self.cluster_tile_shape_mnk = tuple(
x * y for x, y in zip(self.cta_tile_shape_mnk, (*self.cluster_shape_mn, 1))
)
# Compute cluster layout
self.cluster_layout_vmnk = cute.tiled_divide(
cute.make_layout((*self.cluster_shape_mn, 1)),
(tiled_mma.thr_id.shape,),
)
self.cluster_layout_sfb_vmnk = cute.tiled_divide(
cute.make_layout((*self.cluster_shape_mn, 1)),
(tiled_mma_sfb.thr_id.shape,),
)
# Compute number of multicast CTAs for A/B
self.num_mcast_ctas_a = cute.size(self.cluster_layout_vmnk.shape[2])
self.num_mcast_ctas_b = cute.size(self.cluster_layout_vmnk.shape[1])
self.num_mcast_ctas_sfb = cute.size(self.cluster_layout_sfb_vmnk.shape[1])
self.is_a_mcast = self.num_mcast_ctas_a > 1
self.is_b_mcast = self.num_mcast_ctas_b > 1
self.is_sfb_mcast = self.num_mcast_ctas_sfb > 1
# Compute epilogue subtile
self.epi_tile = sm100_utils.compute_epilogue_tile_shape(
self.cta_tile_shape_mnk,
self.use_2cta_instrs,
self.c_layout,
self.c_dtype,
)
# Setup A/B/C stage count in shared memory and ACC stage count in tensor memory
self.num_acc_stage, self.num_ab_stage, self.num_c_stage = self._compute_stages(
tiled_mma,
self.mma_tiler,
self.a_dtype,
self.b_dtype,
self.epi_tile,
self.c_dtype,
self.c_layout,
self.sf_dtype,
self.sf_vec_size,
self.smem_capacity,
self.occupancy,
self.force_c_stage,
)
max_ab = self.max_ab_stage
if max_ab is None:
max_ab = 6 if self.use_2cta_instrs else 4
self.num_ab_stage = min(self.num_ab_stage, max_ab)
self.prefetch_dist = max(0, self.num_ab_stage - 2)
if self.prefetch_dist_override is not None:
self.prefetch_dist = max(
0, min(self.prefetch_dist_override, self.num_ab_stage)
)
self.prefetch_enabled = self.prefetch_dist > 0
# Compute A/B/SFA/SFB/C shared memory layout
self.a_smem_layout_staged = sm100_utils.make_smem_layout_a(
tiled_mma,
self.mma_tiler,
self.a_dtype,
self.num_ab_stage,
)
self.b_smem_layout_staged = sm100_utils.make_smem_layout_b(
tiled_mma,
self.mma_tiler,
self.b_dtype,
self.num_ab_stage,
)
self.sfa_smem_layout_staged = blockscaled_utils.make_smem_layout_sfa(
tiled_mma,
self.mma_tiler,
self.sf_vec_size,
self.num_ab_stage,
)
self.sfb_smem_layout_staged = blockscaled_utils.make_smem_layout_sfb(
tiled_mma,
self.mma_tiler,
self.sf_vec_size,
self.num_ab_stage,
)
self.c_smem_layout_staged = sm100_utils.make_smem_layout_epi(
self.c_dtype,
self.c_layout,
self.epi_tile,
self.num_c_stage,
)
mbar_smem_bytes = self._get_mbar_smem_bytes(
num_acc_stage=self.num_acc_stage,
num_ab_stage=self.num_ab_stage,
num_c_stage=self.num_c_stage,
)
# Use utils.TensorMapUpdateMode.SMEM by default
tensormap_smem_bytes = (
Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap
* Sm100GroupedBlockScaledGemmKernel.num_tensormaps
)
if (
mbar_smem_bytes
+ tensormap_smem_bytes
+ Sm100GroupedBlockScaledGemmKernel.tensor_memory_management_bytes
> self.reserved_smem_bytes
):
raise ValueError(
f"smem consumption for mbar and tensormap {mbar_smem_bytes + tensormap_smem_bytes} exceeds the "
f"reserved smem bytes {self.reserved_smem_bytes}"
)
@cute.jit
def __call__(
self,
initial_a: cute.Tensor,
initial_b: cute.Tensor,
initial_c: cute.Tensor,
initial_sfa: cute.Tensor,
initial_sfb: cute.Tensor,
group_count: cutlass.Constexpr[int],
problem_shape_mnkl: cute.Tensor,
strides_abc: cute.Tensor,
tensor_address_abc: cute.Tensor,
tensor_address_sfasfb: cute.Tensor,
total_num_clusters: cutlass.Constexpr[int],
tensormap_cute_tensor: cute.Tensor,
max_active_clusters: cutlass.Constexpr[int],
st,
):
self.a_dtype = initial_a.element_type
self.b_dtype = initial_b.element_type
self.sf_dtype = initial_sfa.element_type
self.c_dtype = initial_c.element_type
self.a_major_mode = utils.LayoutEnum.from_tensor(initial_a).mma_major_mode()
self.b_major_mode = utils.LayoutEnum.from_tensor(initial_b).mma_major_mode()
self.c_layout = utils.LayoutEnum.from_tensor(initial_c)
if cutlass.const_expr(self.a_dtype != self.b_dtype):
raise TypeError(f"Type mismatch: {self.a_dtype} != {self.b_dtype}")
# Setup attributes that dependent on gemm inputs
self._setup_attributes()
# Setup sfa/sfb tensor by filling A/B tensor to scale factor atom layout
# ((Atom_M, Rest_M),(Atom_K, Rest_K),RestL)
sfa_layout = blockscaled_utils.tile_atom_to_shape_SF(
initial_a.shape, self.sf_vec_size
)
initial_sfa = cute.make_tensor(initial_sfa.iterator, sfa_layout)
# ((Atom_N, Rest_N),(Atom_K, Rest_K),RestL)
sfb_layout = blockscaled_utils.tile_atom_to_shape_SF(
initial_b.shape, self.sf_vec_size
)
initial_sfb = cute.make_tensor(initial_sfb.iterator, sfb_layout)
tiled_mma = sm100_utils.make_blockscaled_trivial_tiled_mma(
self.a_dtype,
self.a_major_mode,
self.b_major_mode,
self.sf_dtype,
self.sf_vec_size,
self.cta_group,
self.mma_inst_shape_mn,
)
tiled_mma_sfb = sm100_utils.make_blockscaled_trivial_tiled_mma(
self.a_dtype,
self.a_major_mode,
self.b_major_mode,
self.sf_dtype,
self.sf_vec_size,
cute.nvgpu.tcgen05.CtaGroup.ONE,
self.mma_inst_shape_mn_sfb,
)
atom_thr_size = cute.size(tiled_mma.thr_id.shape)
# Setup TMA load for A
a_op = sm100_utils.cluster_shape_to_tma_atom_A(
self.cluster_shape_mn, tiled_mma.thr_id
)
a_smem_layout = cute.slice_(self.a_smem_layout_staged, (None, None, None, 0))
tma_atom_a, tma_tensor_a = cute.nvgpu.make_tiled_tma_atom_A(
a_op,
initial_a,
a_smem_layout,
self.mma_tiler,
tiled_mma,
self.cluster_layout_vmnk.shape,
)
# Setup TMA load for B
b_op = sm100_utils.cluster_shape_to_tma_atom_B(
self.cluster_shape_mn, tiled_mma.thr_id
)
b_smem_layout = cute.slice_(self.b_smem_layout_staged, (None, None, None, 0))
tma_atom_b, tma_tensor_b = cute.nvgpu.make_tiled_tma_atom_B(
b_op,
initial_b,
b_smem_layout,
self.mma_tiler,
tiled_mma,
self.cluster_layout_vmnk.shape,
)
# Setup TMA load for SFA
sfa_op = sm100_utils.cluster_shape_to_tma_atom_A(
self.cluster_shape_mn, tiled_mma.thr_id
)
sfa_smem_layout = cute.slice_(
self.sfa_smem_layout_staged, (None, None, None, 0)
)
tma_atom_sfa, tma_tensor_sfa = cute.nvgpu.make_tiled_tma_atom_A(
sfa_op,
initial_sfa,
sfa_smem_layout,
self.mma_tiler,
tiled_mma,
self.cluster_layout_vmnk.shape,
internal_type=cutlass.Int16,
)
# Setup TMA load for SFB
sfb_op = sm100_utils.cluster_shape_to_tma_atom_SFB(
self.cluster_shape_mn, tiled_mma.thr_id
)
sfb_smem_layout = cute.slice_(
self.sfb_smem_layout_staged, (None, None, None, 0)
)
tma_atom_sfb, tma_tensor_sfb = cute.nvgpu.make_tiled_tma_atom_B(
sfb_op,
initial_sfb,
sfb_smem_layout,
self.mma_tiler_sfb,
tiled_mma_sfb,
self.cluster_layout_sfb_vmnk.shape,
internal_type=cutlass.Int16,
)
a_copy_size = cute.size_in_bytes(self.a_dtype, a_smem_layout)
b_copy_size = cute.size_in_bytes(self.b_dtype, b_smem_layout)
sfa_copy_size = cute.size_in_bytes(self.sf_dtype, sfa_smem_layout)
sfb_copy_size = cute.size_in_bytes(self.sf_dtype, sfb_smem_layout)
self.num_tma_load_bytes = (
a_copy_size + b_copy_size + sfa_copy_size + sfb_copy_size
) * atom_thr_size
# Setup TMA store for C
epi_smem_layout = cute.slice_(self.c_smem_layout_staged, (None, None, 0))
tma_atom_c, tma_tensor_c = cpasync.make_tiled_tma_atom(
cpasync.CopyBulkTensorTileS2GOp(),
initial_c,
epi_smem_layout,
self.epi_tile,
)
# Compute grid size
self.tile_sched_params, grid = self._compute_grid(
total_num_clusters,
self.cluster_shape_mn,
max_active_clusters,
swizzle_size=self.swizzle_size,
raster_along_m=self.raster_along_m,
)
self.buffer_align_bytes = 1024
self.size_tensormap_in_i64 = (
Sm100GroupedBlockScaledGemmKernel.num_tensormaps
* Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap
// 8
)
# Define shared storage for kernel
@cute.struct
class SharedStorage:
tensormap_buffer: cute.struct.MemRange[
cutlass.Int64, self.size_tensormap_in_i64
]
ab_full_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_ab_stage]
ab_empty_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_ab_stage]
acc_full_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_acc_stage]
acc_empty_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_acc_stage]
tmem_dealloc_mbar_ptr: cutlass.Int64
tmem_holding_buf: cutlass.Int32
# (EPI_TILE_M, EPI_TILE_N, STAGE)
sC: cute.struct.Align[
cute.struct.MemRange[
self.c_dtype,
cute.cosize(self.c_smem_layout_staged.outer),
],
self.buffer_align_bytes,
]
# (MMA, MMA_M, MMA_K, STAGE)
sA: cute.struct.Align[
cute.struct.MemRange[
self.a_dtype, cute.cosize(self.a_smem_layout_staged.outer)
],
self.buffer_align_bytes,
]
# (MMA, MMA_N, MMA_K, STAGE)
sB: cute.struct.Align[
cute.struct.MemRange[
self.b_dtype, cute.cosize(self.b_smem_layout_staged.outer)
],
self.buffer_align_bytes,
]
# (MMA, MMA_M, MMA_K, STAGE)
sSFA: cute.struct.Align[
cute.struct.MemRange[
self.sf_dtype, cute.cosize(self.sfa_smem_layout_staged)
],
self.buffer_align_bytes,
]
# (MMA, MMA_N, MMA_K, STAGE)
sSFB: cute.struct.Align[
cute.struct.MemRange[
self.sf_dtype, cute.cosize(self.sfb_smem_layout_staged)
],
self.buffer_align_bytes,
]
self.shared_storage = SharedStorage
# Launch the kernel synchronously
self.kernel(
tiled_mma,
tiled_mma_sfb,
tma_atom_a,
tma_tensor_a,
tma_atom_b,
tma_tensor_b,
tma_atom_sfa,
tma_tensor_sfa,
tma_atom_sfb,
tma_tensor_sfb,
tma_atom_c,
tma_tensor_c,
self.cluster_layout_vmnk,
self.cluster_layout_sfb_vmnk,
self.a_smem_layout_staged,
self.b_smem_layout_staged,
self.sfa_smem_layout_staged,
self.sfb_smem_layout_staged,
self.c_smem_layout_staged,
self.epi_tile,
self.tile_sched_params,
group_count,
problem_shape_mnkl,
strides_abc,
tensor_address_abc,
tensor_address_sfasfb,
tensormap_cute_tensor,
).launch(
grid=grid,
block=[self.threads_per_cta, 1, 1],
cluster=(*self.cluster_shape_mn, 1),
smem=self.shared_storage.size_in_bytes(),
**{"st" + "ream": st},
min_blocks_per_mp=1,
)
return
# GPU device kernel
@cute.kernel
def kernel(
self,
tiled_mma: cute.TiledMma,
tiled_mma_sfb: cute.TiledMma,
tma_atom_a: cute.CopyAtom,
mA_mkl: cute.Tensor,
tma_atom_b: cute.CopyAtom,
mB_nkl: cute.Tensor,
tma_atom_sfa: cute.CopyAtom,
mSFA_mkl: cute.Tensor,
tma_atom_sfb: cute.CopyAtom,
mSFB_nkl: cute.Tensor,
tma_atom_c: cute.CopyAtom,
mC_mnl: cute.Tensor,
cluster_layout_vmnk: cute.Layout,
cluster_layout_sfb_vmnk: cute.Layout,
a_smem_layout_staged: cute.ComposedLayout,
b_smem_layout_staged: cute.ComposedLayout,
sfa_smem_layout_staged: cute.Layout,
sfb_smem_layout_staged: cute.Layout,
c_smem_layout_staged: Union[cute.Layout, cute.ComposedLayout],
epi_tile: cute.Tile,
tile_sched_params: utils.PersistentTileSchedulerParams,
group_count: cutlass.Constexpr,
problem_sizes_mnkl: cute.Tensor,
strides_abc: cute.Tensor,
ptrs_abc: cute.Tensor,
ptrs_sfasfb: cute.Tensor,
tensormaps: cute.Tensor,
):
warp_idx = cute.arch.warp_idx()
warp_idx = cute.arch.make_warp_uniform(warp_idx)
if warp_idx == self.tma_warp_id:
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_a)
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_b)
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_sfa)
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_sfb)
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_c)
# PTX prefetch base pointers to warm L2/L1 for first tile.
_ptx_prefetch_global(mA_mkl.iterator)
_ptx_prefetch_global_l1(mA_mkl.iterator)
_ptx_prefetch_global(mB_nkl.iterator)
_ptx_prefetch_global_l1(mB_nkl.iterator)
_ptx_prefetch_global(mSFA_mkl.iterator)
_ptx_prefetch_global_l1(mSFA_mkl.iterator)
_ptx_prefetch_global(mSFB_nkl.iterator)
_ptx_prefetch_global_l1(mSFB_nkl.iterator)
_ptx_prefetch_global(mC_mnl.iterator)
_ptx_prefetch_global_l1(mC_mnl.iterator)
use_2cta_instrs = cute.size(tiled_mma.thr_id.shape) == 2
#
# Setup cta/thread coordinates
#
# Coords inside cluster
bidx, bidy, bidz = cute.arch.block_idx()
mma_tile_coord_v = bidx % cute.size(tiled_mma.thr_id.shape)
is_leader_cta = mma_tile_coord_v == 0
cta_rank_in_cluster = cute.arch.make_warp_uniform(
cute.arch.block_idx_in_cluster()
)
block_in_cluster_coord_vmnk = cluster_layout_vmnk.get_flat_coord(
cta_rank_in_cluster
)
block_in_cluster_coord_sfb_vmnk = cluster_layout_sfb_vmnk.get_flat_coord(
cta_rank_in_cluster
)
# coord inside cta
tidx, _, _ = cute.arch.thread_idx()
#
# Alloc and init: tensormap buffer, a+b full/empty, accumulator full/empty, tensor memory dealloc barrier
#
smem = utils.SmemAllocator()
storage = smem.allocate(self.shared_storage)
tensormap_smem_ptr = storage.tensormap_buffer.data_ptr()
tensormap_a_smem_ptr = tensormap_smem_ptr
tensormap_b_smem_ptr = (
tensormap_a_smem_ptr
+ Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
)
tensormap_sfa_smem_ptr = (
tensormap_b_smem_ptr
+ Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
)
tensormap_sfb_smem_ptr = (
tensormap_sfa_smem_ptr
+ Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
)
tensormap_c_smem_ptr = (
tensormap_sfb_smem_ptr
+ Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
)
tmem_dealloc_mbar_ptr = storage.tmem_dealloc_mbar_ptr
tmem_holding_buf = storage.tmem_holding_buf
# Initialize mainloop ab_pipeline (barrier) and states
ab_pipeline_producer_group = pipeline.CooperativeGroup(pipeline.Agent.Thread)
num_tma_producer = self.num_mcast_ctas_a + self.num_mcast_ctas_b - 1
ab_pipeline_consumer_group = pipeline.CooperativeGroup(
pipeline.Agent.Thread, num_tma_producer
)
ab_pipeline = pipeline.PipelineTmaUmma.create(
barrier_storage=storage.ab_full_mbar_ptr.data_ptr(),
num_stages=self.num_ab_stage,
producer_group=ab_pipeline_producer_group,
consumer_group=ab_pipeline_consumer_group,
tx_count=self.num_tma_load_bytes,
cta_layout_vmnk=cluster_layout_vmnk,
)
# Initialize acc_pipeline (barrier) and states
acc_pipeline_producer_group = pipeline.CooperativeGroup(pipeline.Agent.Thread)
num_acc_consumer_threads = len(self.epilog_warp_id) * (
2 if use_2cta_instrs else 1
)
acc_pipeline_consumer_group = pipeline.CooperativeGroup(
pipeline.Agent.Thread, num_acc_consumer_threads
)
acc_pipeline = pipeline.PipelineUmmaAsync.create(
barrier_storage=storage.acc_full_mbar_ptr.data_ptr(),
num_stages=self.num_acc_stage,
producer_group=acc_pipeline_producer_group,
consumer_group=acc_pipeline_consumer_group,
cta_layout_vmnk=cluster_layout_vmnk,
)
# Tensor memory dealloc barrier init
if use_2cta_instrs:
if warp_idx == self.tma_warp_id:
num_tmem_dealloc_threads = 32
with cute.arch.elect_one():
cute.arch.mbarrier_init(
tmem_dealloc_mbar_ptr, num_tmem_dealloc_threads
)
# Cluster arrive after barrier init
pipeline_init_arrive(cluster_shape_mn=self.cluster_shape_mn, is_relaxed=True)
#
# Setup smem tensor A/B/SFA/SFB/C
#
sC = storage.sC.get_tensor(
c_smem_layout_staged.outer, swizzle=c_smem_layout_staged.inner
)
# (MMA, MMA_M, MMA_K, STAGE)
sA = storage.sA.get_tensor(
a_smem_layout_staged.outer, swizzle=a_smem_layout_staged.inner
)
# (MMA, MMA_N, MMA_K, STAGE)
sB = storage.sB.get_tensor(
b_smem_layout_staged.outer, swizzle=b_smem_layout_staged.inner
)
# (MMA, MMA_M, MMA_K, STAGE)
sSFA = storage.sSFA.get_tensor(sfa_smem_layout_staged)
# (MMA, MMA_N, MMA_K, STAGE)
sSFB = storage.sSFB.get_tensor(sfb_smem_layout_staged)
#
# Compute multicast mask for A/B/SFA/SFB buffer full
#
a_full_mcast_mask = None
b_full_mcast_mask = None
sfa_full_mcast_mask = None
sfb_full_mcast_mask = None
if cutlass.const_expr(self.is_a_mcast or self.is_b_mcast or use_2cta_instrs):
a_full_mcast_mask = cpasync.create_tma_multicast_mask(
cluster_layout_vmnk, block_in_cluster_coord_vmnk, mcast_mode=2
)
b_full_mcast_mask = cpasync.create_tma_multicast_mask(
cluster_layout_vmnk, block_in_cluster_coord_vmnk, mcast_mode=1
)
sfa_full_mcast_mask = cpasync.create_tma_multicast_mask(
cluster_layout_vmnk, block_in_cluster_coord_vmnk, mcast_mode=2
)
sfb_full_mcast_mask = cpasync.create_tma_multicast_mask(
cluster_layout_sfb_vmnk, block_in_cluster_coord_sfb_vmnk, mcast_mode=1
)
#
# Local_tile partition global tensors
#
# (bM, bK, RestM, RestK, RestL)
gA_mkl = cute.local_tile(
mA_mkl, cute.slice_(self.mma_tiler, (None, 0, None)), (None, None, None)
)
# (bN, bK, RestN, RestK, RestL)
gB_nkl = cute.local_tile(
mB_nkl, cute.slice_(self.mma_tiler, (0, None, None)), (None, None, None)
)
# (bM, bK, RestM, RestK, RestL)
gSFA_mkl = cute.local_tile(
mSFA_mkl, cute.slice_(self.mma_tiler, (None, 0, None)), (None, None, None)
)
# (bN, bK, RestN, RestK, RestL)
gSFB_nkl = cute.local_tile(
mSFB_nkl, cute.slice_(self.mma_tiler, (0, None, None)), (None, None, None)
)
# (bM, bN, RestM, RestN, RestL)
gC_mnl = cute.local_tile(
mC_mnl, cute.slice_(self.mma_tiler, (None, None, 0)), (None, None, None)
)
#
# Partition global tensor for TiledMMA_A/B/C
#
thr_mma = tiled_mma.get_slice(mma_tile_coord_v)
thr_mma_sfb = tiled_mma_sfb.get_slice(mma_tile_coord_v)
# (MMA, MMA_M, MMA_K, RestM, RestK, RestL)
tCgA = thr_mma.partition_A(gA_mkl)
# (MMA, MMA_N, MMA_K, RestN, RestK, RestL)
tCgB = thr_mma.partition_B(gB_nkl)
# (MMA, MMA_M, MMA_K, RestM, RestK, RestL)
tCgSFA = thr_mma.partition_A(gSFA_mkl)
# (MMA, MMA_N, MMA_K, RestN, RestK, RestL)
tCgSFB = thr_mma_sfb.partition_B(gSFB_nkl)
# (MMA, MMA_M, MMA_N, RestM, RestN, RestL)
tCgC = thr_mma.partition_C(gC_mnl)
#
# Partition global/shared tensor for TMA load A/B
#
# TMA load A partition_S/D
a_cta_layout = cute.make_layout(
cute.slice_(cluster_layout_vmnk, (0, 0, None, 0)).shape
)
# ((atom_v, rest_v), STAGE)
# ((atom_v, rest_v), RestM, RestK, RestL)
tAsA, tAgA = cpasync.tma_partition(
tma_atom_a,
block_in_cluster_coord_vmnk[2],
a_cta_layout,
cute.group_modes(sA, 0, 3),
cute.group_modes(tCgA, 0, 3),
)
# TMA load B partition_S/D
b_cta_layout = cute.make_layout(
cute.slice_(cluster_layout_vmnk, (0, None, 0, 0)).shape
)
# ((atom_v, rest_v), STAGE)
# ((atom_v, rest_v), RestN, RestK, RestL)
tBsB, tBgB = cpasync.tma_partition(
tma_atom_b,
block_in_cluster_coord_vmnk[1],
b_cta_layout,
cute.group_modes(sB, 0, 3),
cute.group_modes(tCgB, 0, 3),
)
# TMA Load SFA partition_S/D
sfa_cta_layout = a_cta_layout
# ((atom_v, rest_v), STAGE)
# ((atom_v, rest_v), RestM, RestK, RestL)
tAsSFA, tAgSFA = cute.nvgpu.cpasync.tma_partition(
tma_atom_sfa,
block_in_cluster_coord_vmnk[2],
sfa_cta_layout,
cute.group_modes(sSFA, 0, 3),
cute.group_modes(tCgSFA, 0, 3),
)
tAsSFA = cute.filter_zeros(tAsSFA)
tAgSFA = cute.filter_zeros(tAgSFA)
# TMA Load SFB partition_S/D
sfb_cta_layout = cute.make_layout(
cute.slice_(cluster_layout_sfb_vmnk, (0, None, 0, 0)).shape
)
# ((atom_v, rest_v), STAGE)
# ((atom_v, rest_v), RestN, RestK, RestL)
tBsSFB, tBgSFB = cute.nvgpu.cpasync.tma_partition(
tma_atom_sfb,
block_in_cluster_coord_sfb_vmnk[1],
sfb_cta_layout,
cute.group_modes(sSFB, 0, 3),
cute.group_modes(tCgSFB, 0, 3),
)
tBsSFB = cute.filter_zeros(tBsSFB)
tBgSFB = cute.filter_zeros(tBgSFB)
#
# Partition shared/tensor memory tensor for TiledMMA_A/B/C
#
# (MMA, MMA_M, MMA_K, STAGE)
tCrA = tiled_mma.make_fragment_A(sA)
# (MMA, MMA_N, MMA_K, STAGE)
tCrB = tiled_mma.make_fragment_B(sB)
# (MMA, MMA_M, MMA_N)
acc_shape = tiled_mma.partition_shape_C(self.mma_tiler[:2])
# (MMA, MMA_M, MMA_N, STAGE)
tCtAcc_fake = tiled_mma.make_fragment_C(
cute.append(acc_shape, self.num_acc_stage)
)
#
# Cluster wait before tensor memory alloc
#
pipeline_init_wait(cluster_shape_mn=self.cluster_shape_mn)
#
# Get tensormap buffer address
#
grid_dim = cute.arch.grid_dim()
tensormap_workspace_idx = (
bidz * grid_dim[1] * grid_dim[0] + bidy * grid_dim[0] + bidx
)
tensormap_manager = utils.TensorMapManager(
utils.TensorMapUpdateMode.SMEM,
Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap,
)
tensormap_a_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 0, None)].iterator
)
tensormap_b_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 1, None)].iterator
)
tensormap_sfa_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 2, None)].iterator
)
tensormap_sfb_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 3, None)].iterator
)
tensormap_c_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 4, None)].iterator
)
#
# Specialized TMA load warp
#
if warp_idx == self.tma_warp_id:
#
# Persistent tile scheduling loop
#
tile_sched = utils.StaticPersistentTileScheduler.create(
tile_sched_params, cute.arch.block_idx(), grid_dim
)
# grouped gemm tile scheduler helper will compute the group index for the tile we're working on
group_gemm_ts_helper = utils.GroupedGemmTileSchedulerHelper(
group_count,
tile_sched_params,
self.cluster_tile_shape_mnk,
utils.create_initial_search_state(),
)
tensormap_init_done = cutlass.Boolean(False)
# group index of last tile
last_group_idx = cutlass.Int32(-1)
work_tile = tile_sched.initial_work_tile_info()
ab_producer_state = pipeline.make_pipeline_state(
pipeline.PipelineUserType.Producer, self.num_ab_stage
)
while work_tile.is_valid_tile:
cur_tile_coord = work_tile.tile_idx
grouped_gemm_cta_tile_info = group_gemm_ts_helper.delinearize_z(
cur_tile_coord,
problem_sizes_mnkl,
)
cur_k_tile_cnt = grouped_gemm_cta_tile_info.cta_tile_count_k
cur_group_idx = grouped_gemm_cta_tile_info.group_idx
is_group_changed = cur_group_idx != last_group_idx
# skip tensormap update if we're working on the same group
if is_group_changed:
real_tensor_a = self.make_tensor_abc_for_tensormap_update(
cur_group_idx,
self.a_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
strides_abc,
ptrs_abc,
0, # 0 for tensor A
)
real_tensor_b = self.make_tensor_abc_for_tensormap_update(
cur_group_idx,
self.b_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
strides_abc,
ptrs_abc,
1, # 1 for tensor B
)
real_tensor_sfa = self.make_tensor_sfasfb_for_tensormap_update(
cur_group_idx,
self.sf_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
ptrs_sfasfb,
0, # 0 for tensor SFA
)
real_tensor_sfb = self.make_tensor_sfasfb_for_tensormap_update(
cur_group_idx,
self.sf_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
ptrs_sfasfb,
1, # 1 for tensor SFB
)
if tensormap_init_done == False:
# wait tensormap initialization complete
self.tensormap_ab_init_barrier.arrive_and_wait()
tensormap_init_done = True
if hasattr(real_tensor_a, "iterator"):
_ptx_prefetch_global(real_tensor_a.iterator)
_ptx_prefetch_global_l1(real_tensor_a.iterator)
_ptx_prefetchu_l1(real_tensor_a.iterator)
if hasattr(real_tensor_b, "iterator"):
_ptx_prefetch_global(real_tensor_b.iterator)
_ptx_prefetch_global_l1(real_tensor_b.iterator)
_ptx_prefetchu_l1(real_tensor_b.iterator)
if hasattr(real_tensor_sfa, "iterator"):
_ptx_prefetch_global(real_tensor_sfa.iterator)
_ptx_prefetch_global_l1(real_tensor_sfa.iterator)
_ptx_prefetchu_l1(real_tensor_sfa.iterator)
if hasattr(real_tensor_sfb, "iterator"):
_ptx_prefetch_global(real_tensor_sfb.iterator)
_ptx_prefetch_global_l1(real_tensor_sfb.iterator)
_ptx_prefetchu_l1(real_tensor_sfb.iterator)
_ptx_prefetch_global(tensormap_a_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_a_gmem_ptr)
_ptx_prefetchu_l1(tensormap_a_gmem_ptr)
_ptx_prefetch_global(tensormap_b_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_b_gmem_ptr)
_ptx_prefetchu_l1(tensormap_b_gmem_ptr)
_ptx_prefetch_global(tensormap_sfa_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_sfa_gmem_ptr)
_ptx_prefetchu_l1(tensormap_sfa_gmem_ptr)
_ptx_prefetch_global(tensormap_sfb_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_sfb_gmem_ptr)
_ptx_prefetchu_l1(tensormap_sfb_gmem_ptr)
tensormap_manager.update_tensormap(
(
real_tensor_a,
real_tensor_b,
real_tensor_sfa,
real_tensor_sfb,
),
(tma_atom_a, tma_atom_b, tma_atom_sfa, tma_atom_sfb),
(
tensormap_a_gmem_ptr,
tensormap_b_gmem_ptr,
tensormap_sfa_gmem_ptr,
tensormap_sfb_gmem_ptr,
),
self.tma_warp_id,
(
tensormap_a_smem_ptr,
tensormap_b_smem_ptr,
tensormap_sfa_smem_ptr,
tensormap_sfb_smem_ptr,
),
)
mma_tile_coord_mnl = (
grouped_gemm_cta_tile_info.cta_tile_idx_m
// cute.size(tiled_mma.thr_id.shape),
grouped_gemm_cta_tile_info.cta_tile_idx_n,
0,
)
#
# Slice to per mma tile index
#
# ((atom_v, rest_v), RestK)
tAgA_slice = tAgA[
(None, mma_tile_coord_mnl[0], None, mma_tile_coord_mnl[2])
]
# ((atom_v, rest_v), RestK)
tBgB_slice = tBgB[
(None, mma_tile_coord_mnl[1], None, mma_tile_coord_mnl[2])
]
# ((atom_v, rest_v), RestK)
tAgSFA_slice = tAgSFA[
(None, mma_tile_coord_mnl[0], None, mma_tile_coord_mnl[2])
]
# ((atom_v, rest_v), RestK)
tBgSFB_slice = tBgSFB[
(None, mma_tile_coord_mnl[1], None, mma_tile_coord_mnl[2])
]
tma_desc_a = tensormap_manager.get_tensormap_ptr(
tensormap_a_gmem_ptr,
cute.AddressSpace.generic,
)
tma_desc_b = tensormap_manager.get_tensormap_ptr(
tensormap_b_gmem_ptr,
cute.AddressSpace.generic,
)
tma_desc_sfa = tensormap_manager.get_tensormap_ptr(
tensormap_sfa_gmem_ptr,
cute.AddressSpace.generic,
)
tma_desc_sfb = tensormap_manager.get_tensormap_ptr(
tensormap_sfb_gmem_ptr,
cute.AddressSpace.generic,
)
if self.prefetch_enabled:
for pf_k_tile in cutlass.range(
0, min(self.prefetch_dist, cur_k_tile_cnt), unroll=1
):
_prefetch_tma(
tma_atom_a,
tAgA_slice[(None, pf_k_tile)],
tma_desc_a,
)
_prefetch_tma(
tma_atom_b,
tBgB_slice[(None, pf_k_tile)],
tma_desc_b,
)
_prefetch_tma(
tma_atom_sfa,
tAgSFA_slice[(None, pf_k_tile)],
tma_desc_sfa,
)
_prefetch_tma(
tma_atom_sfb,
tBgSFB_slice[(None, pf_k_tile)],
tma_desc_sfb,
)
# Peek (try_wait) AB buffer empty for k_tile = prefetch_k_tile_cnt
ab_producer_state.reset_count()
peek_ab_empty_status = cutlass.Boolean(1)
if ab_producer_state.count < cur_k_tile_cnt:
peek_ab_empty_status = ab_pipeline.producer_try_acquire(
ab_producer_state
)
if is_group_changed:
tensormap_manager.fence_tensormap_update(tensormap_a_gmem_ptr)
tensormap_manager.fence_tensormap_update(tensormap_b_gmem_ptr)
tensormap_manager.fence_tensormap_update(tensormap_sfa_gmem_ptr)
tensormap_manager.fence_tensormap_update(tensormap_sfb_gmem_ptr)
#
# Tma load loop
#
for k_tile in cutlass.range(0, cur_k_tile_cnt, 1, unroll=1):
# Conditionally wait for AB buffer empty
ab_pipeline.producer_acquire(
ab_producer_state, peek_ab_empty_status
)
# TMA load A/B/SFA/SFB
cute.copy(
tma_atom_a,
tAgA_slice[(None, ab_producer_state.count)],
tAsA[(None, ab_producer_state.index)],
tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
mcast_mask=a_full_mcast_mask,
tma_desc_ptr=tma_desc_a,
)
cute.copy(
tma_atom_b,
tBgB_slice[(None, ab_producer_state.count)],
tBsB[(None, ab_producer_state.index)],
tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
mcast_mask=b_full_mcast_mask,
tma_desc_ptr=tma_desc_b,
)
cute.copy(
tma_atom_sfa,
tAgSFA_slice[(None, ab_producer_state.count)],
tAsSFA[(None, ab_producer_state.index)],
tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
mcast_mask=sfa_full_mcast_mask,
tma_desc_ptr=tma_desc_sfa,
)
cute.copy(
tma_atom_sfb,
tBgSFB_slice[(None, ab_producer_state.count)],
tBsSFB[(None, ab_producer_state.index)],
tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
mcast_mask=sfb_full_mcast_mask,
tma_desc_ptr=tma_desc_sfb,
)
if self.prefetch_enabled:
if k_tile < cur_k_tile_cnt - self.prefetch_dist:
future_k_tile = ab_producer_state.count + self.prefetch_dist
_prefetch_tma(
tma_atom_a,
tAgA_slice[(None, future_k_tile)],
tma_desc_a,
)
_prefetch_tma(
tma_atom_b,
tBgB_slice[(None, future_k_tile)],
tma_desc_b,
)
_prefetch_tma(
tma_atom_sfa,
tAgSFA_slice[(None, future_k_tile)],
tma_desc_sfa,
)
_prefetch_tma(
tma_atom_sfb,
tBgSFB_slice[(None, future_k_tile)],
tma_desc_sfb,
)
# Peek (try_wait) AB buffer empty for k_tile = prefetch_k_tile_cnt + k_tile + 1
ab_producer_state.advance()
peek_ab_empty_status = cutlass.Boolean(1)
if ab_producer_state.count < cur_k_tile_cnt:
peek_ab_empty_status = ab_pipeline.producer_try_acquire(
ab_producer_state
)
#
# Advance to next tile
#
tile_sched.advance_to_next_work()
work_tile = tile_sched.get_current_work()
last_group_idx = cur_group_idx
#
# Wait A/B buffer empty
#
ab_pipeline.producer_tail(ab_producer_state)
#
# Specialized MMA warp
#
if warp_idx == self.mma_warp_id:
#
# Initialize tensormaps for A, B, SFA and SFB
#
tensormap_manager.init_tensormap_from_atom(
tma_atom_a, tensormap_a_smem_ptr, self.mma_warp_id
)
tensormap_manager.init_tensormap_from_atom(
tma_atom_b, tensormap_b_smem_ptr, self.mma_warp_id
)
tensormap_manager.init_tensormap_from_atom(
tma_atom_sfa, tensormap_sfa_smem_ptr, self.mma_warp_id
)
tensormap_manager.init_tensormap_from_atom(
tma_atom_sfb, tensormap_sfb_smem_ptr, self.mma_warp_id
)
# indicate tensormap initialization has finished
self.tensormap_ab_init_barrier.arrive_and_wait()
#
# Bar sync for retrieve tensor memory ptr from shared mem
#
self.tmem_alloc_barrier.arrive_and_wait()
#
# Retrieving tensor memory ptr and make accumulator/SFA/SFB tensor
#
# Make accumulator tmem tensor
acc_tmem_ptr = cute.arch.retrieve_tmem_ptr(
self.acc_dtype,
alignment=16,
ptr_to_buffer_holding_addr=tmem_holding_buf,
)
# (MMA, MMA_M, MMA_N, STAGE)
tCtAcc_base = cute.make_tensor(acc_tmem_ptr, tCtAcc_fake.layout)
# Make SFA tmem tensor
sfa_tmem_ptr = cute.recast_ptr(
acc_tmem_ptr + tcgen05.find_tmem_tensor_col_offset(tCtAcc_base),
dtype=self.sf_dtype,
)
# (MMA, MMA_M, MMA_K)
tCtSFA_layout = blockscaled_utils.make_tmem_layout_sfa(
tiled_mma,
self.mma_tiler,
self.sf_vec_size,
cute.slice_(sfa_smem_layout_staged, (None, None, None, 0)),
)
tCtSFA = cute.make_tensor(sfa_tmem_ptr, tCtSFA_layout)
# Make SFB tmem tensor
sfb_tmem_ptr = cute.recast_ptr(
acc_tmem_ptr
+ tcgen05.find_tmem_tensor_col_offset(tCtAcc_base)
+ tcgen05.find_tmem_tensor_col_offset(tCtSFA),
dtype=self.sf_dtype,
)
# (MMA, MMA_N, MMA_K)
tCtSFB_layout = blockscaled_utils.make_tmem_layout_sfb(
tiled_mma,
self.mma_tiler,
self.sf_vec_size,
cute.slice_(sfb_smem_layout_staged, (None, None, None, 0)),
)
tCtSFB = cute.make_tensor(sfb_tmem_ptr, tCtSFB_layout)
#
# Partition for S2T copy of SFA/SFB
#
tiled_copy_s2t_sfa, tCsSFA_compact_s2t, tCtSFA_compact_s2t = (
self.mainloop_s2t_copy_and_partition(sSFA, tCtSFA)
)
tiled_copy_s2t_sfb, tCsSFB_compact_s2t, tCtSFB_compact_s2t = (
self.mainloop_s2t_copy_and_partition(sSFB, tCtSFB)
)
#
# Persistent tile scheduling loop
#
tile_sched = utils.StaticPersistentTileScheduler.create(
tile_sched_params, cute.arch.block_idx(), grid_dim
)
# grouped gemm tile scheduler helper will compute the group index for the tile we're working on
group_gemm_ts_helper = utils.GroupedGemmTileSchedulerHelper(
group_count,
tile_sched_params,
self.cluster_tile_shape_mnk,
utils.create_initial_search_state(),
)
work_tile = tile_sched.initial_work_tile_info()
ab_consumer_state = pipeline.make_pipeline_state(
pipeline.PipelineUserType.Consumer, self.num_ab_stage
)
acc_producer_state = pipeline.make_pipeline_state(
pipeline.PipelineUserType.Producer, self.num_acc_stage
)
while work_tile.is_valid_tile:
cur_tile_coord = work_tile.tile_idx
# MMA warp is only interested in number of tiles along K dimension
(
cur_k_tile_cnt,
cur_group_idx,
) = group_gemm_ts_helper.search_cluster_tile_count_k(
cur_tile_coord,
problem_sizes_mnkl,
)
# (MMA, MMA_M, MMA_N)
tCtAcc = tCtAcc_base[(None, None, None, acc_producer_state.index)]
# Peek (try_wait) AB buffer full for k_tile = 0
ab_consumer_state.reset_count()
peek_ab_full_status = cutlass.Boolean(1)
if ab_consumer_state.count < cur_k_tile_cnt and is_leader_cta:
peek_ab_full_status = ab_pipeline.consumer_try_wait(
ab_consumer_state
)
#
# Wait for accumulator buffer empty
#
if is_leader_cta:
acc_pipeline.producer_acquire(acc_producer_state)
#
# Reset the ACCUMULATE field for each tile
#
tiled_mma.set(tcgen05.Field.ACCUMULATE, False)
#
# Mma mainloop
#
for k_tile in range(cur_k_tile_cnt):
if is_leader_cta:
# Conditionally wait for AB buffer full
ab_pipeline.consumer_wait(
ab_consumer_state, peek_ab_full_status
)
# Copy SFA/SFB from smem to tmem
s2t_stage_coord = (
None,
None,
None,
None,
ab_consumer_state.index,
)
tCsSFA_compact_s2t_staged = tCsSFA_compact_s2t[s2t_stage_coord]
tCsSFB_compact_s2t_staged = tCsSFB_compact_s2t[s2t_stage_coord]
cute.copy(
tiled_copy_s2t_sfa,
tCsSFA_compact_s2t_staged,
tCtSFA_compact_s2t,
)
cute.copy(
tiled_copy_s2t_sfb,
tCsSFB_compact_s2t_staged,
tCtSFB_compact_s2t,
)
# tCtAcc += tCrA * tCrSFA * tCrB * tCrSFB
num_kblocks = cute.size(tCrA, mode=[2])
for kblock_idx in cutlass.range(num_kblocks, unroll_full=True):
kblock_coord = (
None,
None,
kblock_idx,
ab_consumer_state.index,
)
# Set SFA/SFB tensor to tiled_mma
sf_kblock_coord = (None, None, kblock_idx)
tiled_mma.set(
tcgen05.Field.SFA,
tCtSFA[sf_kblock_coord].iterator,
)
tiled_mma.set(
tcgen05.Field.SFB,
tCtSFB[sf_kblock_coord].iterator,
)
cute.gemm(
tiled_mma,
tCtAcc,
tCrA[kblock_coord],
tCrB[kblock_coord],
tCtAcc,
)
# Enable accumulate on tCtAcc after first kblock
tiled_mma.set(tcgen05.Field.ACCUMULATE, True)
# Async arrive AB buffer empty
ab_pipeline.consumer_release(ab_consumer_state)
# Peek (try_wait) AB buffer full for k_tile = k_tile + 1
ab_consumer_state.advance()
peek_ab_full_status = cutlass.Boolean(1)
if ab_consumer_state.count < cur_k_tile_cnt:
if is_leader_cta:
peek_ab_full_status = ab_pipeline.consumer_try_wait(
ab_consumer_state
)
#
# Async arrive accumulator buffer full
#
if is_leader_cta:
acc_pipeline.producer_commit(acc_producer_state)
acc_producer_state.advance()
#
# Advance to next tile
#
tile_sched.advance_to_next_work()
work_tile = tile_sched.get_current_work()
#
# Wait for accumulator buffer empty
#
acc_pipeline.producer_tail(acc_producer_state)
#
# Specialized epilogue warps
#
if warp_idx < self.mma_warp_id:
if cutlass.const_expr(self.use_tma_store):
# initialize tensormap for C
tensormap_manager.init_tensormap_from_atom(
tma_atom_c,
tensormap_c_smem_ptr,
self.epilog_warp_id[0],
)
#
# Alloc tensor memory buffer
#
if warp_idx == self.epilog_warp_id[0]:
cute.arch.alloc_tmem(
self.num_tmem_alloc_cols,
tmem_holding_buf,
is_two_cta=use_2cta_instrs,
)
#
# Bar sync for retrieve tensor memory ptr from shared memory
#
self.tmem_alloc_barrier.arrive_and_wait()
#
# Retrieving tensor memory ptr and make accumulator tensor
#
acc_tmem_ptr = cute.arch.retrieve_tmem_ptr(
self.acc_dtype,
alignment=16,
ptr_to_buffer_holding_addr=tmem_holding_buf,
)
# (MMA, MMA_M, MMA_N, STAGE)
tCtAcc_base = cute.make_tensor(acc_tmem_ptr, tCtAcc_fake.layout)
### Start from here
#
# Partition for epilogue
#
epi_tidx = tidx
if cutlass.const_expr(self.use_tma_store):
tiled_copy_t2r, tTR_tAcc_base, tTR_rAcc = (
self.epilog_tmem_copy_and_partition(
epi_tidx, tCtAcc_base, tCgC, epi_tile, use_2cta_instrs
)
)
tTR_rC = cute.make_rmem_tensor(tTR_rAcc.shape, self.c_dtype)
tiled_copy_r2s, tRS_rC, tRS_sC = self.epilog_smem_copy_and_partition(
tiled_copy_t2r, tTR_rC, epi_tidx, sC
)
tma_atom_c, bSG_sC, bSG_gC_partitioned = (
self.epilog_gmem_copy_and_partition(
epi_tidx, tma_atom_c, tCgC, epi_tile, sC
)
)
else:
tCtAcc_simt = self._transform_partitioned_tensor_layout(tCtAcc_base)
tCgC_simt = self._transform_partitioned_tensor_layout(tCgC)
(
tiled_copy_t2r,
tTR_tAcc_base,
tTR_rAcc,
) = self.epilog_tmem_copy_and_partition_simt(
epi_tidx, tCtAcc_simt, tCgC_simt, epi_tile, use_2cta_instrs
)
tTR_rC = cute.make_rmem_tensor(tTR_rAcc.shape, self.c_dtype)
gC_epi_simt = cute.flat_divide(tCgC_simt, epi_tile)
thr_copy_t2r = tiled_copy_t2r.get_slice(epi_tidx)
simt_atom_vec = cute.make_copy_atom(
cute.nvgpu.CopyUniversalOp(),
self.c_dtype,
)
#
# Persistent tile scheduling loop
#
tile_sched = utils.StaticPersistentTileScheduler.create(
tile_sched_params, cute.arch.block_idx(), grid_dim
)
# grouped gemm tile scheduler helper will compute the group index for the tile we're working on
group_gemm_ts_helper = utils.GroupedGemmTileSchedulerHelper(
group_count,
tile_sched_params,
self.cluster_tile_shape_mnk,
utils.create_initial_search_state(),
)
work_tile = tile_sched.initial_work_tile_info()
acc_consumer_state = pipeline.make_pipeline_state(
pipeline.PipelineUserType.Consumer, self.num_acc_stage
)
if cutlass.const_expr(self.use_tma_store):
# Threads/warps participating in tma store pipeline
c_producer_group = pipeline.CooperativeGroup(
pipeline.Agent.Thread,
32 * len(self.epilog_warp_id),
)
c_pipeline = pipeline.PipelineTmaStore.create(
num_stages=self.num_c_stage,
producer_group=c_producer_group,
)
if cutlass.const_expr(self.use_tma_store):
# group index to start searching
last_group_idx = cutlass.Int32(-1)
while work_tile.is_valid_tile:
cur_tile_coord = work_tile.tile_idx
grouped_gemm_cta_tile_info = group_gemm_ts_helper.delinearize_z(
cur_tile_coord,
problem_sizes_mnkl,
)
cur_group_idx = grouped_gemm_cta_tile_info.group_idx
is_group_changed = cur_group_idx != last_group_idx
real_tensor_c = self.make_tensor_abc_for_tensormap_update(
cur_group_idx,
self.c_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
strides_abc,
ptrs_abc,
2, # 2 for tensor C
)
if is_group_changed:
if hasattr(real_tensor_c, "iterator"):
_ptx_prefetch_global(real_tensor_c.iterator)
_ptx_prefetch_global_l1(real_tensor_c.iterator)
_ptx_prefetchu_l1(real_tensor_c.iterator)
_ptx_prefetch_global(tensormap_c_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_c_gmem_ptr)
_ptx_prefetchu_l1(tensormap_c_gmem_ptr)
tensormap_manager.update_tensormap(
((real_tensor_c),),
((tma_atom_c),),
((tensormap_c_gmem_ptr),),
self.epilog_warp_id[0],
(tensormap_c_smem_ptr,),
)
mma_tile_coord_mnl = (
grouped_gemm_cta_tile_info.cta_tile_idx_m
// cute.size(tiled_mma.thr_id.shape),
grouped_gemm_cta_tile_info.cta_tile_idx_n,
0,
)
# ((ATOM_V, REST_V), EPI_M, EPI_N)
bSG_gC = bSG_gC_partitioned[
(
None,
None,
None,
*mma_tile_coord_mnl,
)
]
# Set tensor memory buffer for current tile
# (T2R, T2R_M, T2R_N, EPI_M, EPI_M)
tTR_tAcc = tTR_tAcc_base[
(None, None, None, None, None, acc_consumer_state.index)
]
#
# Wait for accumulator buffer full
#
acc_pipeline.consumer_wait(acc_consumer_state)
tTR_tAcc = cute.group_modes(tTR_tAcc, 3, cute.rank(tTR_tAcc))
#
# Store accumulator to global memory in subtiles
#
subtile_cnt = cute.size(tTR_tAcc.shape, mode=[3])
bSG_gC = cute.group_modes(bSG_gC, 1, cute.rank(bSG_gC))
if is_group_changed:
if warp_idx == self.epilog_warp_id[0]:
tensormap_manager.fence_tensormap_update(
tensormap_c_gmem_ptr
)
num_prev_subtiles = tile_sched.num_tiles_executed * subtile_cnt
for subtile_idx in range(subtile_cnt):
#
# Load accumulator from tensor memory buffer to register
#
tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]
cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)
# Early-release the accumulator buffer once the last subtile has been
# transferred to registers, so MMA can reuse TMEM while we finish stores.
if subtile_idx == subtile_cnt - 1:
with cute.arch.elect_one():
acc_pipeline.consumer_release(acc_consumer_state)
acc_consumer_state.advance()
#
# Convert to C type
#
acc_vec = tiled_copy_r2s.retile(tTR_rAcc).load()
tRS_rC.store(acc_vec.to(self.c_dtype))
#
# Store C to shared memory
#
c_buffer = (num_prev_subtiles + subtile_idx) % self.num_c_stage
cute.copy(
tiled_copy_r2s,
tRS_rC,
tRS_sC[(None, None, None, c_buffer)],
)
# Fence and barrier to make sure shared memory store is visible to TMA store
cute.arch.fence_proxy("async.shared", space="cta")
self.epilog_sync_barrier.arrive_and_wait()
#
# TMA store C to global memory
#
if warp_idx == self.epilog_warp_id[0]:
cute.copy(
tma_atom_c,
bSG_sC[(None, c_buffer)],
bSG_gC[(None, subtile_idx)],
tma_desc_ptr=tensormap_manager.get_tensormap_ptr(
tensormap_c_gmem_ptr,
cute.AddressSpace.generic,
),
)
# Fence and barrier to make sure shared memory store is visible to TMA store
c_pipeline.producer_commit()
c_pipeline.producer_acquire()
self.epilog_sync_barrier.arrive_and_wait()
#
# Advance to next tile
#
tile_sched.advance_to_next_work()
work_tile = tile_sched.get_current_work()
last_group_idx = cur_group_idx
else:
# group index to start searching
last_group_idx = cutlass.Int32(-1)
while work_tile.is_valid_tile:
cur_tile_coord = work_tile.tile_idx
grouped_gemm_cta_tile_info = group_gemm_ts_helper.delinearize_z(
cur_tile_coord,
problem_sizes_mnkl,
)
cur_group_idx = grouped_gemm_cta_tile_info.group_idx
real_tensor_c = self.make_tensor_abc_for_tensormap_update(
cur_group_idx,
self.c_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
strides_abc,
ptrs_abc,
2, # 2 for tensor C
)
if cur_group_idx != last_group_idx:
if hasattr(real_tensor_c, "iterator"):
_ptx_prefetch_global(real_tensor_c.iterator)
_ptx_prefetch_global_l1(real_tensor_c.iterator)
_ptx_prefetchu_l1(real_tensor_c.iterator)
mma_tile_coord_mnl = (
grouped_gemm_cta_tile_info.cta_tile_idx_m
// cute.size(tiled_mma.thr_id.shape),
grouped_gemm_cta_tile_info.cta_tile_idx_n,
0,
)
tile_origin_m = mma_tile_coord_mnl[0] * self.mma_tiler[0]
tile_origin_n = mma_tile_coord_mnl[1] * self.mma_tiler[1]
residue_m = (
grouped_gemm_cta_tile_info.problem_shape_m - tile_origin_m
)
residue_n = (
grouped_gemm_cta_tile_info.problem_shape_n - tile_origin_n
)
full_tile = (residue_m >= self.mma_tiler[0]) & (
residue_n >= self.mma_tiler[1]
)
gC_mnl_tile = cute.local_tile(
real_tensor_c,
cute.slice_(self.mma_tiler, (None, None, 0)),
mma_tile_coord_mnl,
)
tCgC_tile = thr_mma.partition_C(gC_mnl_tile)
tCgC_tile = self._transform_partitioned_tensor_layout(tCgC_tile)
tTR_gC = self.epilog_gmem_copy_and_partition_simt(
epi_tidx, tiled_copy_t2r, tCgC_tile, epi_tile
)
# Set tensor memory buffer for current tile
# (T2R, T2R_M, T2R_N, EPI_M, EPI_M)
tTR_tAcc = tTR_tAcc_base[
(None, None, None, None, None, acc_consumer_state.index)
]
#
# Wait for accumulator buffer full
#
acc_pipeline.consumer_wait(acc_consumer_state)
tTR_tAcc = cute.group_modes(tTR_tAcc, 3, cute.rank(tTR_tAcc))
tTR_gC = cute.group_modes(tTR_gC, 3, cute.rank(tTR_gC))
#
# Store accumulator to global memory in subtiles
#
subtile_cnt = cute.size(tTR_tAcc.shape, mode=[3])
if full_tile:
for subtile_idx in range(subtile_cnt):
#
# Load accumulator from tensor memory buffer to register
#
tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]
cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)
if subtile_idx == subtile_cnt - 1:
with cute.arch.elect_one():
acc_pipeline.consumer_release(acc_consumer_state)
acc_consumer_state.advance()
#
# Convert to C type
#
acc_vec = tTR_rAcc.load()
tTR_rC.store(acc_vec.to(self.c_dtype))
tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]
cute.copy(simt_atom_vec, tTR_rC, tTR_gC_mn)
else:
cC_mnl = cute.make_identity_tensor(real_tensor_c.shape)
cC_mnl_tile = cute.local_tile(
cC_mnl,
cute.slice_(self.mma_tiler, (None, None, 0)),
mma_tile_coord_mnl,
)
tCcC_tile = thr_mma.partition_C(cC_mnl_tile)
tCcC_tile = self._transform_partitioned_tensor_layout(
tCcC_tile
)
tTR_cC = self.epilog_gmem_copy_and_partition_simt(
epi_tidx, tiled_copy_t2r, tCcC_tile, epi_tile
)
tTR_cC = cute.group_modes(tTR_cC, 3, cute.rank(tTR_cC))
c_shape = real_tensor_c.shape
for subtile_idx in range(subtile_cnt):
#
# Load accumulator from tensor memory buffer to register
#
tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]
cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)
if subtile_idx == subtile_cnt - 1:
with cute.arch.elect_one():
acc_pipeline.consumer_release(acc_consumer_state)
acc_consumer_state.advance()
#
# Convert to C type
#
acc_vec = tTR_rAcc.load()
tTR_rC.store(acc_vec.to(self.c_dtype))
tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]
tTR_cC_mn = tTR_cC[(None, None, None, subtile_idx)]
tTR_pC = cute.make_rmem_tensor(
tTR_rC.shape, cutlass.Boolean
)
for i in range(cute.size(tTR_rC.shape)):
tTR_pC[i] = cute.elem_less(tTR_cC_mn[i], c_shape)
cute.basic_copy_if(tTR_pC, tTR_rC, tTR_gC_mn)
#
# Advance to next tile
#
tile_sched.advance_to_next_work()
work_tile = tile_sched.get_current_work()
last_group_idx = cur_group_idx
last_group_idx = cur_group_idx
#
# Dealloc the tensor memory buffer
#
if warp_idx == self.epilog_warp_id[0]:
cute.arch.relinquish_tmem_alloc_permit(is_two_cta=use_2cta_instrs)
self.epilog_sync_barrier.arrive_and_wait()
if warp_idx == self.epilog_warp_id[0]:
if use_2cta_instrs:
cute.arch.mbarrier_arrive(
tmem_dealloc_mbar_ptr, cta_rank_in_cluster ^ 1
)
cute.arch.mbarrier_wait(tmem_dealloc_mbar_ptr, 0)
cute.arch.dealloc_tmem(
acc_tmem_ptr, self.num_tmem_alloc_cols, is_two_cta=use_2cta_instrs
)
#
# Wait for C store complete
#
if cutlass.const_expr(self.use_tma_store):
c_pipeline.producer_tail()
@cute.jit
def make_tensor_abc_for_tensormap_update(
self,
group_idx: cutlass.Int32,
dtype: Type[cutlass.Numeric],
problem_shape_mnk: tuple[cutlass.Int32, cutlass.Int32, cutlass.Int32],
strides_abc: cute.Tensor,
tensor_address_abc: cute.Tensor,
tensor_index: int,
):
ptr_i64 = tensor_address_abc[(group_idx, tensor_index)]
if cutlass.const_expr(
not isclass(dtype) or not issubclass(dtype, cutlass.Numeric)
):
raise TypeError(
f"dtype must be a type of cutlass.Numeric, got {type(dtype)}"
)
align = 16
if cutlass.const_expr(tensor_index == 2):
align = self.c_assumed_align
tensor_gmem_ptr = cute.make_ptr(
dtype, ptr_i64, cute.AddressSpace.gmem, assumed_align=align
)
strides_tensor_gmem = strides_abc[(group_idx, tensor_index, None)]
strides_tensor_reg = cute.make_rmem_tensor(
cute.make_layout(2),
strides_abc.element_type,
)
cute.autovec_copy(strides_tensor_gmem, strides_tensor_reg)
stride_mn = strides_tensor_reg[0]
stride_k = strides_tensor_reg[1]
c1 = cutlass.Int32(1)
c0 = cutlass.Int32(0)
if cutlass.const_expr(tensor_index == 0): # tensor A
m = problem_shape_mnk[0]
k = problem_shape_mnk[2]
return cute.make_tensor(
tensor_gmem_ptr,
cute.make_layout((m, k, c1), stride=(stride_mn, stride_k, c0)),
)
elif cutlass.const_expr(tensor_index == 1): # tensor B
n = problem_shape_mnk[1]
k = problem_shape_mnk[2]
return cute.make_tensor(
tensor_gmem_ptr,
cute.make_layout((n, k, c1), stride=(stride_mn, stride_k, c0)),
)
else: # tensor C
m = problem_shape_mnk[0]
n = problem_shape_mnk[1]
tensor = cute.make_tensor(
tensor_gmem_ptr,
cute.make_layout((m, n, c1), stride=(stride_mn, stride_k, c0)),
)
leading_dim = 0 if self.c_layout.is_m_major_c() else 1
tensor.mark_layout_dynamic(leading_dim=leading_dim)
stride_order = (2, 1, 0) if leading_dim == 0 else (2, 0, 1)
div = self.c_divisibility if getattr(dtype, "width", 0) == 16 else 16
tensor.mark_compact_shape_dynamic(
mode=leading_dim,
stride_order=stride_order,
divisibility=div,
)
return tensor
@cute.jit
def make_tensor_sfasfb_for_tensormap_update(
self,
group_idx: cutlass.Int32,
dtype: Type[cutlass.Numeric],
problem_shape_mnk: tuple[cutlass.Int32, cutlass.Int32, cutlass.Int32],
tensor_address_sfasfb: cute.Tensor,
tensor_index: int,
):
ptr_i64 = tensor_address_sfasfb[(group_idx, tensor_index)]
if cutlass.const_expr(
not isclass(dtype) or not issubclass(dtype, cutlass.Numeric)
):
raise TypeError(
f"dtype must be a type of cutlass.Numeric, got {type(dtype)}"
)
tensor_gmem_ptr = cute.make_ptr(
dtype, ptr_i64, cute.AddressSpace.gmem, assumed_align=16
)
c1 = cutlass.Int32(1)
if cutlass.const_expr(tensor_index == 0): # tensor SFA
m = problem_shape_mnk[0]
k = problem_shape_mnk[2]
sfa_layout = blockscaled_utils.tile_atom_to_shape_SF(
(m, k, c1), self.sf_vec_size
)
return cute.make_tensor(
tensor_gmem_ptr,
sfa_layout,
)
else: # tensor SFB
n = problem_shape_mnk[1]
k = problem_shape_mnk[2]
sfb_layout = blockscaled_utils.tile_atom_to_shape_SF(
(n, k, c1), self.sf_vec_size
)
return cute.make_tensor(
tensor_gmem_ptr,
sfb_layout,
)
def mainloop_s2t_copy_and_partition(
self,
sSF: cute.Tensor,
tSF: cute.Tensor,
) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor, cute.Tensor]:
# (MMA, MMA_MN, MMA_K, STAGE)
tCsSF_compact = cute.filter_zeros(sSF)
# (MMA, MMA_MN, MMA_K)
tCtSF_compact = cute.filter_zeros(tSF)
# Make S2T CopyAtom and tiledCopy
copy_atom_s2t = cute.make_copy_atom(
tcgen05.Cp4x32x128bOp(self.cta_group),
self.sf_dtype,
)
tiled_copy_s2t = tcgen05.make_s2t_copy(copy_atom_s2t, tCtSF_compact)
thr_copy_s2t = tiled_copy_s2t.get_slice(0)
# ((ATOM_V, REST_V), Rest_Tiler, MMA_MN, MMA_K, STAGE)
tCsSF_compact_s2t_ = thr_copy_s2t.partition_S(tCsSF_compact)
# ((ATOM_V, REST_V), Rest_Tiler, MMA_MN, MMA_K, STAGE)
tCsSF_compact_s2t = tcgen05.get_s2t_smem_desc_tensor(
tiled_copy_s2t, tCsSF_compact_s2t_
)
# ((ATOM_V, REST_V), Rest_Tiler, MMA_MN, MMA_K)
tCtSF_compact_s2t = thr_copy_s2t.partition_D(tCtSF_compact)
return tiled_copy_s2t, tCsSF_compact_s2t, tCtSF_compact_s2t
def epilog_tmem_copy_and_partition(
self,
tidx: cutlass.Int32,
tAcc: cute.Tensor,
gC_mnl: cute.Tensor,
epi_tile: cute.Tile,
use_2cta_instrs: Union[cutlass.Boolean, bool],
) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor]:
# Make tiledCopy for tensor memory load
copy_atom_t2r = sm100_utils.get_tmem_load_op(
self.cta_tile_shape_mnk,
self.c_layout,
self.c_dtype,
self.acc_dtype,
epi_tile,
use_2cta_instrs,
)
# (EPI_TILE_M, EPI_TILE_N, EPI_M, EPI_N, STAGE)
tAcc_epi = cute.flat_divide(
tAcc[((None, None), 0, 0, None)],
epi_tile,
)
# (EPI_TILE_M, EPI_TILE_N)
tiled_copy_t2r = tcgen05.make_tmem_copy(
copy_atom_t2r, tAcc_epi[(None, None, 0, 0, 0)]
)
thr_copy_t2r = tiled_copy_t2r.get_slice(tidx)
# (T2R, T2R_M, T2R_N, EPI_M, EPI_M, STAGE)
tTR_tAcc = thr_copy_t2r.partition_S(tAcc_epi)
gC_mnl_epi = cute.flat_divide(
gC_mnl[((None, None), 0, 0, None, None, None)], epi_tile
)
# (T2R, T2R_M, T2R_N, EPI_M, EPI_N, RestM, RestN, RestL)
tTR_gC = thr_copy_t2r.partition_D(gC_mnl_epi)
# (T2R, T2R_M, T2R_N)
tTR_rAcc = cute.make_rmem_tensor(
tTR_gC[(None, None, None, 0, 0, 0, 0, 0)].shape, self.acc_dtype
)
return tiled_copy_t2r, tTR_tAcc, tTR_rAcc
def epilog_smem_copy_and_partition(
self,
tiled_copy_t2r: cute.TiledCopy,
tTR_rC: cute.Tensor,
tidx: cutlass.Int32,
sC: cute.Tensor,
) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor]:
copy_atom_r2s = sm100_utils.get_smem_store_op(
self.c_layout, self.c_dtype, self.acc_dtype, tiled_copy_t2r
)
tiled_copy_r2s = cute.make_tiled_copy_D(copy_atom_r2s, tiled_copy_t2r)
# (R2S, R2S_M, R2S_N, PIPE_D)
thr_copy_r2s = tiled_copy_r2s.get_slice(tidx)
tRS_sC = thr_copy_r2s.partition_D(sC)
# (R2S, R2S_M, R2S_N)
tRS_rC = tiled_copy_r2s.retile(tTR_rC)
return tiled_copy_r2s, tRS_rC, tRS_sC
def epilog_gmem_copy_and_partition(
self,
tidx: cutlass.Int32,
atom: Union[cute.CopyAtom, cute.TiledCopy],
gC_mnl: cute.Tensor,
epi_tile: cute.Tile,
sC: cute.Tensor,
) -> Tuple[cute.CopyAtom, cute.Tensor, cute.Tensor]:
# (EPI_TILE_M, EPI_TILE_N, EPI_M, EPI_N, RestM, RestN, RestL)
gC_epi = cute.flat_divide(
gC_mnl[((None, None), 0, 0, None, None, None)], epi_tile
)
tma_atom_c = atom
sC_for_tma_partition = cute.group_modes(sC, 0, 2)
gC_for_tma_partition = cute.group_modes(gC_epi, 0, 2)
# ((ATOM_V, REST_V), EPI_M, EPI_N)
# ((ATOM_V, REST_V), EPI_M, EPI_N, RestM, RestN, RestL)
bSG_sC, bSG_gC = cpasync.tma_partition(
tma_atom_c,
0,
cute.make_layout(1),
sC_for_tma_partition,
gC_for_tma_partition,
)
return tma_atom_c, bSG_sC, bSG_gC
def epilog_gmem_copy_and_partition_simt(
self,
tidx: cutlass.Int32,
tiled_copy_t2r: cute.TiledCopy,
gC_mnl: cute.Tensor,
epi_tile: cute.Tile,
) -> cute.Tensor:
# (EPI_TILE_M, EPI_TILE_N, EPI_M, EPI_N, RestM, RestN, RestL)
gC_epi = cute.flat_divide(gC_mnl, epi_tile)
thr_copy_t2r = tiled_copy_t2r.get_slice(tidx)
# (T2R, T2R_M, T2R_N, EPI_M, EPI_N, RestM, RestN, RestL)
tTR_gC = thr_copy_t2r.partition_D(gC_epi)
return tTR_gC
@staticmethod
def _transform_partitioned_tensor_layout(tensor: cute.Tensor) -> cute.Tensor:
layout = tensor.layout
shape = layout.shape
stride = layout.stride
new_shape = ((shape[0][0], shape[1]), (shape[0][1], shape[2]), *shape[3:])
new_stride = (
(stride[0][0], stride[1]),
(stride[0][1], stride[2]),
*stride[3:],
)
new_layout = cute.make_layout(shape=new_shape, stride=new_stride)
return cute.make_tensor(tensor.iterator, new_layout)
def epilog_tmem_copy_and_partition_simt(
self,
tidx: cutlass.Int32,
tAcc: cute.Tensor,
tCgC: cute.Tensor,
epi_tile: cute.Tile,
use_2cta_instrs: Union[cutlass.Boolean, bool],
) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor]:
copy_atom_t2r = sm100_utils.get_tmem_load_op(
self.cta_tile_shape_mnk,
self.c_layout,
self.c_dtype,
self.acc_dtype,
epi_tile,
use_2cta_instrs,
)
tAcc_epi = cute.flat_divide(tAcc, epi_tile)
tiled_copy_t2r = tcgen05.make_tmem_copy(
copy_atom_t2r, tAcc_epi[(None, None, 0, 0, 0)]
)
thr_copy_t2r = tiled_copy_t2r.get_slice(tidx)
tTR_tAcc = thr_copy_t2r.partition_S(tAcc_epi)
tCgC_epi = cute.flat_divide(tCgC, epi_tile)
tTR_gC = thr_copy_t2r.partition_D(tCgC_epi)
tTR_rAcc = cute.make_rmem_tensor(
tTR_gC[(None, None, None, 0, 0, 0, 0, 0)].shape, self.acc_dtype
)
return tiled_copy_t2r, tTR_tAcc, tTR_rAcc
@staticmethod
def _compute_stages(
tiled_mma: cute.TiledMma,
mma_tiler_mnk: Tuple[int, int, int],
a_dtype: Type[cutlass.Numeric],
b_dtype: Type[cutlass.Numeric],
epi_tile: cute.Tile,
c_dtype: Type[cutlass.Numeric],
c_layout: utils.LayoutEnum,
sf_dtype: Type[cutlass.Numeric],
sf_vec_size: int,
smem_capacity: int,
occupancy: int,
force_c_stage: int | None = None,
) -> Tuple[int, int, int]:
# ACC stages
num_acc_stage = 1 if mma_tiler_mnk[1] == 256 else 2
# Default C stages
num_c_stage = 2 if force_c_stage is None else max(1, int(force_c_stage))
# Calculate smem layout and size for one stage of A, B, SFA, SFB and C
a_smem_layout_stage_one = sm100_utils.make_smem_layout_a(
tiled_mma,
mma_tiler_mnk,
a_dtype,
1, # a tmp 1 stage is provided
)
b_smem_layout_staged_one = sm100_utils.make_smem_layout_b(
tiled_mma,
mma_tiler_mnk,
b_dtype,
1, # a tmp 1 stage is provided
)
sfa_smem_layout_staged_one = blockscaled_utils.make_smem_layout_sfa(
tiled_mma,
mma_tiler_mnk,
sf_vec_size,
1, # a tmp 1 stage is provided
)
sfb_smem_layout_staged_one = blockscaled_utils.make_smem_layout_sfb(
tiled_mma,
mma_tiler_mnk,
sf_vec_size,
1, # a tmp 1 stage is provided
)
c_smem_layout_staged_one = sm100_utils.make_smem_layout_epi(
c_dtype,
c_layout,
epi_tile,
1,
)
ab_bytes_per_stage = (
cute.size_in_bytes(a_dtype, a_smem_layout_stage_one)
+ cute.size_in_bytes(b_dtype, b_smem_layout_staged_one)
+ cute.size_in_bytes(sf_dtype, sfa_smem_layout_staged_one)
+ cute.size_in_bytes(sf_dtype, sfb_smem_layout_staged_one)
)
mbar_helpers_bytes = 1024
c_bytes_per_stage = cute.size_in_bytes(c_dtype, c_smem_layout_staged_one)
c_bytes = c_bytes_per_stage * num_c_stage
# Calculate A/B/SFA/SFB stages:
# Start with total smem per CTA (capacity / occupancy)
# Subtract reserved bytes and initial C stages bytes
# Divide remaining by bytes needed per A/B/SFA/SFB stage
num_ab_stage = (
smem_capacity // occupancy - (mbar_helpers_bytes + c_bytes)
) // ab_bytes_per_stage
if num_ab_stage < 1:
num_ab_stage = 1
# Refine epilogue stages:
# Calculate remaining smem after allocating for A/B/SFA/SFB stages and reserved bytes
# Add remaining unused smem to epilogue
if force_c_stage is None:
num_c_stage += (
smem_capacity
- occupancy * ab_bytes_per_stage * num_ab_stage
- occupancy * (mbar_helpers_bytes + c_bytes)
) // (occupancy * c_bytes_per_stage)
return num_acc_stage, num_ab_stage, num_c_stage
@staticmethod
def _compute_grid(
total_num_clusters: int,
cluster_shape_mn: tuple[int, int],
max_active_clusters: cutlass.Constexpr[int],
swizzle_size: int = 1,
raster_along_m: bool = True,
) -> tuple[utils.PersistentTileSchedulerParams, tuple[int, int, int]]:
# Create problem shape with M, N dimensions from cluster shape
# and L dimension representing the total number of clusters.
problem_shape_ntile_mnl = (
cluster_shape_mn[0],
cluster_shape_mn[1],
cutlass.Int32(total_num_clusters),
)
tile_sched_params = utils.PersistentTileSchedulerParams(
problem_shape_ntile_mnl,
(*cluster_shape_mn, 1),
swizzle_size=swizzle_size,
raster_along_m=raster_along_m,
)
grid = utils.StaticPersistentTileScheduler.get_grid_shape(
tile_sched_params, max_active_clusters
)
return tile_sched_params, grid
@staticmethod
def _get_mbar_smem_bytes(**kwargs_stages: int) -> int:
num_barriers_per_stage = 2
num_bytes_per_barrier = 8
mbar_smem_consumption = sum(
[
num_barriers_per_stage * num_bytes_per_barrier * stage
for stage in kwargs_stages.values()
]
)
return mbar_smem_consumption
@staticmethod
def is_valid_dtypes_and_scale_factor_vec_size(
ab_dtype: Type[cutlass.Numeric],
sf_dtype: Type[cutlass.Numeric],
sf_vec_size: int,
c_dtype: Type[cutlass.Numeric],
) -> bool:
is_valid = True
# Check valid ab_dtype
if ab_dtype not in {
cutlass.Float4E2M1FN,
cutlass.Float8E5M2,
cutlass.Float8E4M3FN,
}:
is_valid = False
# Check valid sf_vec_size
if sf_vec_size not in {16, 32}:
is_valid = False
# Check valid sf_dtype
if sf_dtype not in {cutlass.Float8E8M0FNU, cutlass.Float8E4M3FN}:
is_valid = False
# Check valid sf_dtype and sf_vec_size combinations
if sf_dtype == cutlass.Float8E4M3FN and sf_vec_size == 32:
is_valid = False
if ab_dtype in {cutlass.Float8E5M2, cutlass.Float8E4M3FN} and sf_vec_size == 16:
is_valid = False
# Check valid c_dtype
if c_dtype not in {
cutlass.Float32,
cutlass.Float16,
cutlass.BFloat16,
cutlass.Float8E5M2,
cutlass.Float8E4M3FN,
}:
is_valid = False
return is_valid
@staticmethod
def is_valid_layouts(
ab_dtype: Type[cutlass.Numeric],
c_dtype: Type[cutlass.Numeric],
a_major: str,
b_major: str,
c_major: str,
) -> bool:
is_valid = True
if ab_dtype is cutlass.Float4E2M1FN and not (a_major == "k" and b_major == "k"):
is_valid = False
return is_valid
@staticmethod
def is_valid_mma_tiler_and_cluster_shape(
mma_tiler_mn: Tuple[int, int],
cluster_shape_mn: Tuple[int, int],
) -> bool:
is_valid = True
# Skip invalid mma tile shape
if mma_tiler_mn[0] not in [128, 256]:
is_valid = False
if mma_tiler_mn[1] not in [128, 256]:
is_valid = False
# Skip illegal cluster shape
if cluster_shape_mn[0] % (2 if mma_tiler_mn[0] == 256 else 1) != 0:
is_valid = False
# Skip invalid cluster shape
is_power_of_2 = lambda x: x > 0 and (x & (x - 1)) == 0
if (
cluster_shape_mn[0] * cluster_shape_mn[1] > 16
or cluster_shape_mn[0] <= 0
or cluster_shape_mn[1] <= 0
# Special cluster shape check for scale factor multicasts.
# Due to limited size of scale factors, we can't multicast among more than 4 CTAs.
or cluster_shape_mn[0] > 4
or cluster_shape_mn[1] > 4
or not is_power_of_2(cluster_shape_mn[0])
or not is_power_of_2(cluster_shape_mn[1])
):
is_valid = False
return is_valid
@staticmethod
def is_valid_tensor_alignment(
problem_sizes_mnkl: List[Tuple[int, int, int, int]],
ab_dtype: Type[cutlass.Numeric],
c_dtype: Type[cutlass.Numeric],
a_major: str,
b_major: str,
c_major: str,
) -> bool:
is_valid = True
def check_contigous_16B_alignment(dtype, is_mode0_major, tensor_shape):
major_mode_idx = 0 if is_mode0_major else 1
num_major_elements = tensor_shape[major_mode_idx]
num_contiguous_elements = 16 * 8 // dtype.width
return num_major_elements % num_contiguous_elements == 0
for m, n, k, l in problem_sizes_mnkl:
if (
not check_contigous_16B_alignment(ab_dtype, a_major == "m", (m, k, l))
or not check_contigous_16B_alignment(
ab_dtype, b_major == "n", (n, k, l)
)
or not check_contigous_16B_alignment(c_dtype, c_major == "m", (m, n, l))
):
is_valid = False
return is_valid
@staticmethod
def can_implement(
ab_dtype: Type[cutlass.Numeric],
sf_dtype: Type[cutlass.Numeric],
sf_vec_size: int,
c_dtype: Type[cutlass.Numeric],
mma_tiler_mn: Tuple[int, int],
cluster_shape_mn: Tuple[int, int],
problem_sizes_mnkl: List[Tuple[int, int, int, int]],
a_major: str,
b_major: str,
c_major: str,
) -> bool:
can_implement = True
# Skip unsupported types
if not Sm100GroupedBlockScaledGemmKernel.is_valid_dtypes_and_scale_factor_vec_size(
ab_dtype, sf_dtype, sf_vec_size, c_dtype
):
can_implement = False
# Skip unsupported layouts
if not Sm100GroupedBlockScaledGemmKernel.is_valid_layouts(
ab_dtype, c_dtype, a_major, b_major, c_major
):
can_implement = False
# Skip invalid mma tile shape and cluster shape
if not Sm100GroupedBlockScaledGemmKernel.is_valid_mma_tiler_and_cluster_shape(
mma_tiler_mn, cluster_shape_mn
):
can_implement = False
# Skip illegal problem shape for load/store alignment
if not Sm100GroupedBlockScaledGemmKernel.is_valid_tensor_alignment(
problem_sizes_mnkl, ab_dtype, c_dtype, a_major, b_major, c_major
):
can_implement = False
return can_implement
# Size of smem we reserved for mbarrier, tensor memory management and tensormap update
reserved_smem_bytes = 1024
bytes_per_tensormap = 128
num_tensormaps = 5
# size of smem used for tensor memory management
tensor_memory_management_bytes = 12
# Create tensor and return the pointer, tensor, and stride
def _select_initial_indices(
problem_sizes: list[tuple[int, int, int, int]],
) -> tuple[int, int, int]:
key_size_a = lambda item: item[1][0] * item[1][2]
key_size_b = lambda item: item[1][1] * item[1][2]
key_size_c = lambda item: item[1][0] * item[1][1]
min_a_idx, _ = min(enumerate(problem_sizes), key=key_size_a)
min_b_idx, _ = min(enumerate(problem_sizes), key=key_size_b)
min_c_idx, _ = min(enumerate(problem_sizes), key=key_size_c)
return min_a_idx, min_b_idx, min_c_idx
def _from_dlpack_typed(
torch_tensor: torch.Tensor,
cutlass_dtype: Type[cutlass.Numeric],
assumed_align: int,
) -> cute.Tensor:
tensor_for_dlpack = torch_tensor
if getattr(cutlass_dtype, "width", 0) <= 8:
if not torch_tensor.is_contiguous():
tensor_for_dlpack = torch_tensor.contiguous()
tensor_for_dlpack = tensor_for_dlpack.view(torch.uint8)
use_32bit_stride = True
try:
if tensor_for_dlpack.numel() > (2**31 - 1):
use_32bit_stride = False
else:
max_stride = max(tensor_for_dlpack.stride())
if max_stride > (2**31 - 1):
use_32bit_stride = False
except Exception:
use_32bit_stride = False
cute_tensor = from_dlpack(
tensor_for_dlpack,
assumed_align=assumed_align,
use_32bit_stride=use_32bit_stride,
)
cute_tensor.element_type = cutlass_dtype
return cute_tensor
def _get_handle() -> Any:
global _fake_st
if _fake_st is None:
_mk = getattr(cutlass_torch, "default_" + "st" + "ream")
_fake_st = _mk()
return _fake_st
def _normalize_sfs(
sfs: list[tuple[torch.Tensor, torch.Tensor]]
) -> list[tuple[torch.Tensor, torch.Tensor]]:
if not hasattr(torch, "float8_e4m3fn"):
return sfs
target = torch.float8_e4m3fn
out: list[tuple[torch.Tensor, torch.Tensor]] = []
for sfa, sfb in sfs:
if sfa.dtype != target:
sfa = sfa.to(target)
if sfb.dtype != target:
sfb = sfb.to(target)
out.append((sfa, sfb))
return out
def _swap_inputs(
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
) -> tuple[
list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
list[tuple[torch.Tensor, torch.Tensor]],
list[tuple[int, int, int, int]],
]:
abc_swapped: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]] = []
sfs_swapped: list[tuple[torch.Tensor, torch.Tensor]] = []
sizes_swapped: list[tuple[int, int, int, int]] = []
for (a_ref, b_ref, c_ref), (sfa_ref, sfb_ref), (m, n, k, l) in zip(
abc_tensors, sfs, problem_sizes
):
# Swap A/B to compute D = B @ A^T and write D in column-major so C = D^T.
c_swapped = c_ref.permute(1, 0, 2)
abc_swapped.append((b_ref, a_ref, c_swapped))
sfs_swapped.append((sfb_ref, sfa_ref))
sizes_swapped.append((n, m, k, l))
return abc_swapped, sfs_swapped, sizes_swapped
def _make_initial_abc_tensor(
torch_tensor: torch.Tensor,
cutlass_dtype: Type[cutlass.Numeric],
is_mode0_major: bool,
assumed_align: int,
c_divisibility: int | None = None,
) -> cute.Tensor:
cute_tensor = _from_dlpack_typed(torch_tensor, cutlass_dtype, assumed_align)
leading_dim = 0 if is_mode0_major else 1
cute_tensor = cute_tensor.mark_layout_dynamic(leading_dim=leading_dim)
stride_order = (2, 1, 0) if is_mode0_major else (2, 0, 1)
if cutlass_dtype == cutlass.Float4E2M1FN:
divisibility = 32
elif getattr(cutlass_dtype, "width", 0) == 16:
divisibility = 8 if c_divisibility is None else c_divisibility
else:
divisibility = 16
cute_tensor.mark_compact_shape_dynamic(
mode=leading_dim,
stride_order=stride_order,
divisibility=divisibility,
)
return cute_tensor
def _compute_cluster_tile(mma_tiler_mn: tuple[int, int], cluster_shape_mn: tuple[int, int]) -> tuple[int, int]:
cta_tile = (128, mma_tiler_mn[1])
return (cta_tile[0] * cluster_shape_mn[0], cta_tile[1] * cluster_shape_mn[1])
def _total_clusters(problem_sizes: list[tuple[int, int, int, int]], cluster_tile: tuple[int, int]) -> int:
total = 0
tm, tn = cluster_tile
for m, n, _, _ in problem_sizes:
cm = (m + tm - 1) // tm
cn = (n + tn - 1) // tn
total += cm * cn
return total
def _build_entry(cfg_key: str,
mma_tiler_mn: tuple[int, int],
cluster_shape_mn: tuple[int, int],
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
use_tma_store: bool,
c_col_major: bool,
max_ab_stage: int | None,
prefetch_dist: int | None,
force_c_stage: int | None,
c_assumed_align: int,
c_divisibility: int,
swizzle_size: int,
raster_along_m: bool):
dev = abc_tensors[0][0].device
group_count = len(problem_sizes)
hinfo = utils.HardwareInfo()
max_active = hinfo.get_max_active_clusters(cluster_shape_mn[0] * cluster_shape_mn[1])
sm_count = hinfo.get_max_active_clusters(1)
num_tmaps = Sm100GroupedBlockScaledGemmKernel.num_tensormaps
bytes_tmap = Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
tmap_shape = (sm_count, num_tmaps, bytes_tmap)
tmap_t = torch.empty(tmap_shape, dtype=torch.int64, device=dev)
tmap_cute = from_dlpack(tmap_t)
tmap_cute.element_type = cutlass.Int64
dims_cpu = torch.tensor(problem_sizes, dtype=torch.int32, pin_memory=True)
dims_gpu = dims_cpu.to(device=dev)
dims_cute = from_dlpack(dims_gpu)
dims_cute.element_type = cutlass.Int32
strides_cpu = torch.empty((group_count, 3, 2), dtype=torch.int32, pin_memory=True)
ptrs_cpu = torch.empty((group_count, 3), dtype=torch.int64, pin_memory=True)
ptrs_sf_cpu = torch.empty((group_count, 2), dtype=torch.int64, pin_memory=True)
strides_gpu = strides_cpu.to(device=dev)
ptrs_gpu = ptrs_cpu.to(device=dev)
ptrs_sf_gpu = ptrs_sf_cpu.to(device=dev)
strides_cute = from_dlpack(strides_gpu)
strides_cute.element_type = cutlass.Int32
ptrs_cute = from_dlpack(ptrs_gpu)
ptrs_cute.element_type = cutlass.Int64
ptrs_sf_cute = from_dlpack(ptrs_sf_gpu)
ptrs_sf_cute.element_type = cutlass.Int64
def _fill_stride():
for i, (m, n, k, _) in enumerate(problem_sizes):
stride_a = (k, 1)
stride_b = (k, 1)
if c_col_major:
stride_c = (1, m)
else:
stride_c = (n, 1)
strides_cpu[i, 0, 0] = stride_a[0]
strides_cpu[i, 0, 1] = stride_a[1]
strides_cpu[i, 1, 0] = stride_b[0]
strides_cpu[i, 1, 1] = stride_b[1]
strides_cpu[i, 2, 0] = stride_c[0]
strides_cpu[i, 2, 1] = stride_c[1]
strides_gpu.copy_(strides_cpu, non_blocking=True)
_fill_stride()
cluster_tile = _compute_cluster_tile(mma_tiler_mn, cluster_shape_mn)
total_num_clusters = _total_clusters(problem_sizes, cluster_tile)
min_a_idx, min_b_idx, min_c_idx = _select_initial_indices(problem_sizes)
a_ref = abc_tensors[min_a_idx][0]
b_ref = abc_tensors[min_b_idx][1]
c_ref = abc_tensors[min_c_idx][2]
sfa_ref = sfs[min_a_idx][0]
sfb_ref = sfs[min_b_idx][1]
initial_a = _make_initial_abc_tensor(
a_ref, cutlass.Float4E2M1FN, is_mode0_major=False, assumed_align=16
)
initial_b = _make_initial_abc_tensor(
b_ref, cutlass.Float4E2M1FN, is_mode0_major=False, assumed_align=16
)
initial_c = _make_initial_abc_tensor(
c_ref,
cutlass.Float16,
is_mode0_major=c_col_major,
assumed_align=c_assumed_align,
c_divisibility=c_divisibility,
)
initial_sfa = _from_dlpack_typed(
sfa_ref, cutlass.Float8E4M3FN, assumed_align=16
)
initial_sfb = _from_dlpack_typed(
sfb_ref, cutlass.Float8E4M3FN, assumed_align=16
)
gemm = Sm100GroupedBlockScaledGemmKernel(
sf_vec_size=16,
mma_tiler_mn=mma_tiler_mn,
cluster_shape_mn=cluster_shape_mn,
use_tma_store=use_tma_store,
max_ab_stage=max_ab_stage,
prefetch_dist=prefetch_dist,
force_c_stage=force_c_stage,
c_assumed_align=c_assumed_align,
c_divisibility=c_divisibility,
swizzle_size=swizzle_size,
raster_along_m=raster_along_m,
)
gemm.c_tile_stride = (int(c_ref.stride(0)), int(c_ref.stride(1)))
try:
compiled = cute.compile(
gemm,
initial_a,
initial_b,
initial_c,
initial_sfa,
initial_sfb,
group_count,
dims_cute,
strides_cute,
ptrs_cute,
ptrs_sf_cute,
total_num_clusters,
tmap_cute,
max_active,
_get_handle(),
options="--opt-level 2",
)
except Exception as exc:
print(f"cute.compile failed: {exc}", file=sys.stderr, flush=True)
raise RuntimeError(str(exc)) from exc
entry = {
"cfg_key": cfg_key,
"mma_tiler_mn": mma_tiler_mn,
"cluster_shape_mn": cluster_shape_mn,
"group_count": group_count,
"problem_sizes": problem_sizes,
"dims_gpu": dims_gpu,
"dims_cute": dims_cute,
"strides_gpu": strides_gpu,
"strides_cute": strides_cute,
"ptrs_cpu": ptrs_cpu,
"ptrs_gpu": ptrs_gpu,
"ptrs_cute": ptrs_cute,
"ptrs_sf_cpu": ptrs_sf_cpu,
"ptrs_sf_gpu": ptrs_sf_gpu,
"ptrs_sf_cute": ptrs_sf_cute,
"tmap": tmap_cute,
"initial_a": initial_a,
"initial_b": initial_b,
"initial_c": initial_c,
"initial_sfa": initial_sfa,
"initial_sfb": initial_sfb,
"initial_refs": (a_ref, b_ref, c_ref, sfa_ref, sfb_ref),
"compiled": compiled,
"handle": _get_handle(),
}
return entry
def _update_ptrs(entry: dict,
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]]):
ptrs_cpu = entry["ptrs_cpu"]
ptrs_sf_cpu = entry["ptrs_sf_cpu"]
for i, ((a_ref, b_ref, c_ref), (sfa_ref, sfb_ref)) in enumerate(zip(abc_tensors, sfs)):
ptrs_cpu[i, 0] = a_ref.data_ptr()
ptrs_cpu[i, 1] = b_ref.data_ptr()
ptrs_cpu[i, 2] = c_ref.data_ptr()
ptrs_sf_cpu[i, 0] = sfa_ref.data_ptr()
ptrs_sf_cpu[i, 1] = sfb_ref.data_ptr()
entry["ptrs_gpu"].copy_(ptrs_cpu, non_blocking=True)
entry["ptrs_sf_gpu"].copy_(ptrs_sf_cpu, non_blocking=True)
def _run_group(cfg_key: str,
mma_tiler_mn: tuple[int, int],
cluster_shape_mn: tuple[int, int],
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
use_tma_store: bool,
c_col_major: bool,
max_ab_stage: int | None,
prefetch_dist: int | None,
force_c_stage: int | None,
c_assumed_align: int,
c_divisibility: int,
swizzle_size: int,
raster_along_m: bool):
if not abc_tensors:
return
dev = abc_tensors[0][0].device
cache_key = (
dev.index if dev.type == "cuda" else -1,
cfg_key,
use_tma_store,
c_col_major,
max_ab_stage,
prefetch_dist,
force_c_stage,
c_assumed_align,
c_divisibility,
swizzle_size,
raster_along_m,
tuple(problem_sizes),
)
entry = _kernel_cache.get(cache_key)
if entry is None:
entry = _build_entry(
cfg_key,
mma_tiler_mn,
cluster_shape_mn,
abc_tensors,
sfs,
problem_sizes,
use_tma_store,
c_col_major,
max_ab_stage,
prefetch_dist,
force_c_stage,
c_assumed_align,
c_divisibility,
swizzle_size,
raster_along_m,
)
_kernel_cache[cache_key] = entry
_update_ptrs(entry, abc_tensors, sfs)
try:
entry["compiled"](
entry["initial_a"],
entry["initial_b"],
entry["initial_c"],
entry["initial_sfa"],
entry["initial_sfb"],
entry["dims_cute"],
entry["strides_cute"],
entry["ptrs_cute"],
entry["ptrs_sf_cute"],
entry["tmap"],
entry["handle"],
)
except Exception as exc:
print(f"compiled invocation failed: {exc}", file=sys.stderr, flush=True)
raise RuntimeError(str(exc)) from exc
def _split_groups(problem_sizes: list[tuple[int, int, int, int]]) -> tuple[list[int], list[int]]:
idx_1 = []
idx_2 = []
for i, (m, _n, k, _l) in enumerate(problem_sizes):
if m >= 256 and k >= 4096:
idx_2.append(i)
else:
idx_1.append(i)
return idx_1, idx_2
def _split_groups_swap(problem_sizes: list[tuple[int, int, int, int]]) -> tuple[list[int], list[int]]:
idx_1 = []
idx_2 = []
for i, (m, n, k, _l) in enumerate(problem_sizes):
# m is swapped M (original N), n is swapped N (original M)
if m >= 256 and k >= 4096:
idx_2.append(i)
else:
idx_1.append(i)
return idx_1, idx_2
def _split_groups_swap_by_idx(
problem_sizes: list[tuple[int, int, int, int]],
indices: list[int],
) -> tuple[list[int], list[int]]:
idx_1: list[int] = []
idx_2: list[int] = []
for i in indices:
m, _n, k, _l = problem_sizes[i]
if m >= 256 and k >= 4096:
idx_2.append(i)
else:
idx_1.append(i)
return idx_1, idx_2
def _swizzle_params(is_wide: bool) -> tuple[int, bool]:
if is_wide:
size = int(os.environ.get("NVFP4_SWIZZLE_WIDE", "1"))
raster_along_m = os.environ.get("NVFP4_SWIZZLE_WIDE_RASTER_M", "0") != "0"
else:
size = int(os.environ.get("NVFP4_SWIZZLE_NARROW", "1"))
raster_along_m = os.environ.get("NVFP4_SWIZZLE_NARROW_RASTER_M", "1") != "0"
if size not in (1, 2, 4, 8):
size = 1
return size, raster_along_m
def _collect_by_idx(
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
indices: list[int],
) -> tuple[
list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
list[tuple[torch.Tensor, torch.Tensor]],
list[tuple[int, int, int, int]],
]:
return (
[abc_tensors[i] for i in indices],
[sfs[i] for i in indices],
[problem_sizes[i] for i in indices],
)
def _can_use_simt(problem_sizes: list[tuple[int, int, int, int]],
idx_1: list[int],
idx_2: list[int]) -> bool:
for i in idx_1:
m, n, _k, _l = problem_sizes[i]
if (m % 128) != 0 or (n % 256) != 0:
return False
for i in idx_2:
m, n, _k, _l = problem_sizes[i]
if (m % 256) != 0 or (n % 256) != 0:
return False
return True
def custom_kernel(data: input_t) -> output_t:
abc_tensors, _sfs_raw, sfs_reordered, problem_sizes = data
sfs_reordered = _normalize_sfs(sfs_reordered)
# Fast path: grouped CUTLASS kernel using swap + speedK-style scheduling.
if abc_tensors and _is_sm100_device(abc_tensors[0][0]):
speedk_ok = _try_speedk_ext(abc_tensors, sfs_reordered, problem_sizes)
if speedk_ok:
return [c for (_a, _b, c) in abc_tensors]
idx_unswapped: list[int] = []
idx_rest: list[int] = []
for i, (_m, _n, k, _l) in enumerate(problem_sizes):
if k == 2048:
idx_unswapped.append(i)
else:
idx_rest.append(i)
abc_unswapped, sfs_unswapped, sizes_unswapped = _collect_by_idx(
abc_tensors, sfs_reordered, problem_sizes, idx_unswapped
)
abc_rest, sfs_rest, sizes_rest = _collect_by_idx(
abc_tensors, sfs_reordered, problem_sizes, idx_rest
)
abc_swapped, sfs_swapped, sizes_swapped = _swap_inputs(
abc_rest, sfs_rest, sizes_rest
)
if _enable_stage8 and abc_swapped and _is_sm100_device(abc_swapped[0][0]):
stage8_out = _try_stage8_ext(abc_swapped, sfs_swapped, sizes_swapped)
if stage8_out is not None:
return [c for (_a, _b, c) in abc_tensors]
c_col_major = True
is_sm100 = bool(abc_swapped) and _is_sm100_device(abc_swapped[0][0])
c_assumed_align = 32
c_divisibility = 16
prefetch_dist = None if is_sm100 else 0
use_tma_store = True
idx_wide: list[int] = []
idx_narrow: list[int] = []
for i, (_m, _n, k, _l) in enumerate(sizes_swapped):
if k >= 6000:
idx_wide.append(i)
else:
idx_narrow.append(i)
idx_wide_1, idx_wide_2 = _split_groups_swap_by_idx(sizes_swapped, idx_wide)
idx_narrow_1, idx_narrow_2 = _split_groups_swap_by_idx(sizes_swapped, idx_narrow)
abc_wide_1, sfs_wide_1, sizes_wide_1 = _collect_by_idx(
abc_swapped, sfs_swapped, sizes_swapped, idx_wide_1
)
abc_wide_2, sfs_wide_2, sizes_wide_2 = _collect_by_idx(
abc_swapped, sfs_swapped, sizes_swapped, idx_wide_2
)
abc_narrow_1, sfs_narrow_1, sizes_narrow_1 = _collect_by_idx(
abc_swapped, sfs_swapped, sizes_swapped, idx_narrow_1
)
abc_narrow_2, sfs_narrow_2, sizes_narrow_2 = _collect_by_idx(
abc_swapped, sfs_swapped, sizes_swapped, idx_narrow_2
)
enable_simt = os.environ.get("NVFP4_ENABLE_SIMT", "") == "1"
swizzle_wide, raster_wide = _swizzle_params(True)
swizzle_narrow, raster_narrow = _swizzle_params(False)
force_c_stage_2sm_env = os.environ.get("NVFP4_FORCE_C_STAGE_2SM", "")
force_c_stage_2sm = int(force_c_stage_2sm_env) if force_c_stage_2sm_env else None
if force_c_stage_2sm is not None and force_c_stage_2sm <= 0:
force_c_stage_2sm = None
max_ab_stage_2sm_env = os.environ.get("NVFP4_MAX_AB_STAGE_2SM", "")
max_ab_stage_2sm = int(max_ab_stage_2sm_env) if max_ab_stage_2sm_env else None
if max_ab_stage_2sm is not None and max_ab_stage_2sm <= 0:
max_ab_stage_2sm = None
_run_group(
"unswapped",
(128, 256),
# For k==2048 (unswapped), M is small and N is large. Favor clustering along N
# to multicast A/SFA across CTAs and reduce redundant loads.
(1, 4),
abc_unswapped,
sfs_unswapped,
sizes_unswapped,
use_tma_store,
False,
None,
prefetch_dist,
None,
c_assumed_align,
c_divisibility,
swizzle_narrow,
False,
)
if enable_simt:
def is_simt_safe(sz: tuple[int, int, int, int]) -> bool:
_m, n, k, _l = sz
# SIMT vector store uses 128-bit vectors; require N divisible by 8.
# Limit SIMT to k < 4096 (1SM path) to avoid 2CTA edge cases.
return (n % 8) == 0 and k < 4096
def gather(idx_list: list[int], want_simt: bool):
abc_out = []
sfs_out = []
sizes_out = []
for i in idx_list:
if is_simt_safe(sizes_swapped[i]) == want_simt:
abc_out.append(abc_swapped[i])
sfs_out.append(sfs_swapped[i])
sizes_out.append(sizes_swapped[i])
return abc_out, sfs_out, sizes_out
abc_wide_1_simt, sfs_wide_1_simt, sizes_wide_1_simt = gather(
idx_wide_1, True
)
abc_wide_1_tma, sfs_wide_1_tma, sizes_wide_1_tma = gather(
idx_wide_1, False
)
abc_wide_2_simt, sfs_wide_2_simt, sizes_wide_2_simt = gather(
idx_wide_2, True
)
abc_wide_2_tma, sfs_wide_2_tma, sizes_wide_2_tma = gather(
idx_wide_2, False
)
abc_narrow_1_simt, sfs_narrow_1_simt, sizes_narrow_1_simt = gather(
idx_narrow_1, True
)
abc_narrow_1_tma, sfs_narrow_1_tma, sizes_narrow_1_tma = gather(
idx_narrow_1, False
)
abc_narrow_2_simt, sfs_narrow_2_simt, sizes_narrow_2_simt = gather(
idx_narrow_2, True
)
abc_narrow_2_tma, sfs_narrow_2_tma, sizes_narrow_2_tma = gather(
idx_narrow_2, False
)
def run_pair(
cfg_key_base: str,
mma_tiler_mn: tuple[int, int],
cluster_shape_mn: tuple[int, int],
abc_simt: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs_simt: list[tuple[torch.Tensor, torch.Tensor]],
sizes_simt: list[tuple[int, int, int, int]],
abc_tma: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs_tma: list[tuple[torch.Tensor, torch.Tensor]],
sizes_tma: list[tuple[int, int, int, int]],
swizzle_size: int,
raster_along_m: bool,
max_ab_stage: int | None = None,
force_c_stage: int | None = None,
):
_run_group(
f"{cfg_key_base}s",
mma_tiler_mn,
cluster_shape_mn,
abc_simt,
sfs_simt,
sizes_simt,
False,
c_col_major,
max_ab_stage,
prefetch_dist,
force_c_stage,
c_assumed_align,
c_divisibility,
swizzle_size,
raster_along_m,
)
_run_group(
f"{cfg_key_base}t",
mma_tiler_mn,
cluster_shape_mn,
abc_tma,
sfs_tma,
sizes_tma,
True,
c_col_major,
max_ab_stage,
prefetch_dist,
force_c_stage,
c_assumed_align,
c_divisibility,
swizzle_size,
raster_along_m,
)
run_pair(
"w1",
(128, 256),
(2, 1),
abc_wide_1_simt,
sfs_wide_1_simt,
sizes_wide_1_simt,
abc_wide_1_tma,
sfs_wide_1_tma,
sizes_wide_1_tma,
swizzle_wide,
raster_wide,
)
run_pair(
"w2",
(256, 256),
(4, 1),
abc_wide_2_simt,
sfs_wide_2_simt,
sizes_wide_2_simt,
abc_wide_2_tma,
sfs_wide_2_tma,
sizes_wide_2_tma,
swizzle_wide,
raster_wide,
max_ab_stage_2sm,
force_c_stage_2sm,
)
run_pair(
"n1",
(128, 128),
(2, 1),
abc_narrow_1_simt,
sfs_narrow_1_simt,
sizes_narrow_1_simt,
abc_narrow_1_tma,
sfs_narrow_1_tma,
sizes_narrow_1_tma,
swizzle_narrow,
raster_narrow,
)
run_pair(
"n2",
(256, 128),
(4, 1),
abc_narrow_2_simt,
sfs_narrow_2_simt,
sizes_narrow_2_simt,
abc_narrow_2_tma,
sfs_narrow_2_tma,
sizes_narrow_2_tma,
swizzle_narrow,
raster_narrow,
max_ab_stage_2sm,
force_c_stage_2sm,
)
else:
_run_group(
"swap_1sm_wide",
(128, 256),
(2, 1),
abc_wide_1,
sfs_wide_1,
sizes_wide_1,
use_tma_store,
c_col_major,
None,
prefetch_dist,
None,
c_assumed_align,
c_divisibility,
swizzle_wide,
raster_wide,
)
_run_group(
"swap_2sm_wide",
(256, 256),
(4, 1),
abc_wide_2,
sfs_wide_2,
sizes_wide_2,
use_tma_store,
c_col_major,
max_ab_stage_2sm,
prefetch_dist,
force_c_stage_2sm,
c_assumed_align,
c_divisibility,
swizzle_wide,
raster_wide,
)
_run_group(
"swap_1sm",
(128, 128),
(2, 1),
abc_narrow_1,
sfs_narrow_1,
sizes_narrow_1,
use_tma_store,
c_col_major,
None,
prefetch_dist,
None,
c_assumed_align,
c_divisibility,
swizzle_narrow,
raster_narrow,
)
_run_group(
"swap_2sm",
(256, 128),
(4, 1),
abc_narrow_2,
sfs_narrow_2,
sizes_narrow_2,
use_tma_store,
c_col_major,
max_ab_stage_2sm,
prefetch_dist,
force_c_stage_2sm,
c_assumed_align,
c_divisibility,
swizzle_narrow,
raster_narrow,
)
return [c for (_a, _b, c) in abc_tensors]
scrolls · 3621 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 468277.
⋯ 58 unchanged lines_stage8_mod: Any | None = None_stage8_checked = False_enable_stage8 = os.environ.get("NVFP4_ENABLE_STAGE8", "") == "1"+ _speedk_mod: Any | None = None+ _speedk_checked = False+ _enable_speedk_ext = os.environ.get("NVFP4_ENABLE_SPEEDK_EXT", "") != "0"+ _speedk_diag_done = False+ _speedk_ext_mod: Any | None = None+ _speedk_ext_fail = False+ _speedk_inc_cache: list[str] | None = None+ _speedk_cutlass_root: str | None = None+ _speedk_pad_cache: dict[tuple[int, int, torch.dtype, int], torch.Tensor] = {}+ _speedk_c_pad_cache: dict[tuple[int, int, torch.dtype, int], torch.Tensor] = {}+ _speedk_disable_ext = os.environ.get("NVFP4_SPEEDK_EXT_DISABLE", "") == "1"+ _speedk_diag = os.environ.get("NVFP4_SPEEDK_DIAG", "") == "1"def _is_sm100_device(t: torch.Tensor) -> bool:⋯ 37 unchanged linesreturn None+ def _try_speedk_ext(+ abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],+ sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],+ problem_sizes: list[tuple[int, int, int, int]],+ ):+ global _speedk_ext_mod, _speedk_checked, _speedk_ext_fail, _speedk_diag_done+ def _diag(msg: str):+ global _speedk_diag_done+ if not _speedk_diag or _speedk_diag_done:+ return+ _speedk_diag_done = True+ print(msg, file=sys.stderr, flush=True)+ if _speedk_disable_ext or (not _enable_speedk_ext) or (not abc_tensors):+ return None++ # This kernel is SM100-only. Avoid any JIT/compile work on other GPUs.+ if not _is_sm100_device(abc_tensors[0][0]):+ return None++ if _speedk_ext_fail:+ return None++ if _speedk_ext_mod is None:+ _speedk_checked = True+ _speedk_ext_mod = _speedk_load_ext(_diag)+ if _speedk_ext_mod is None:+ _speedk_ext_fail = True+ _diag("speedk ext: build fail")+ return None++ out = _speedk_try_ext(_speedk_ext_mod, abc_tensors, sfs_reordered, problem_sizes)+ if out is None:+ _diag("speedk ext: none")+ return None+ _diag("speedk ext: ok")+ return True+++ _SPEEDK_CPP_HEX = "0a2020202023696e636c756465203c746f7263682f657874656e73696f6e2e683e0a2020202023696e636c756465203c766563746f723e0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f31736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620534642293b0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f32736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620534642293b0a20202020"+ _SPEEDK_CU_HEX = "0a2020202023696e636c756465203c746f7263682f657874656e73696f6e2e683e0a2020202023696e636c756465203c4154656e2f637564612f43554441436f6e746578742e683e0a2020202023696e636c756465203c6331302f637564612f4355444153747265616d2e683e0a2020202023696e636c756465203c6331302f637564612f4355444147756172642e683e0a2020202023696e636c756465203c766563746f723e0a2020202023696e636c756465203c7374646578636570743e0a2020202023696e636c756465203c747970655f7472616974733e0a0a20202020236966646566205f5f435544415f4e4f5f48414c465f4f50455241544f52535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c465f4f50455241544f52535f5f0a2020202023656e6469660a20202020236966646566205f5f435544415f4e4f5f48414c465f434f4e56455253494f4e535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c465f434f4e56455253494f4e535f5f0a2020202023656e6469660a20202020236966646566205f5f435544415f4e4f5f48414c46325f4f50455241544f52535f5f0a2020202023756e646566205f5f435544415f4e4f5f48414c46325f4f50455241544f52535f5f0a2020202023656e6469660a0a2020202023696e636c756465203c637564615f72756e74696d652e683e0a2020202023696620646566696e6564285f5f435544415f415243485f5f292026262021646566696e6564284355544c4153535f415243485f4d4d415f534d313030415f454e41424c4544290a2020202023646566696e65204355544c4153535f415243485f4d4d415f534d313030415f454e41424c454420310a2020202023656e6469660a2020202023696e636c75646520226375746c6173732f6375746c6173732e68220a2020202023696e636c7564652022637574652f74656e736f722e687070220a2020202023696e636c75646520226375746c6173732f74656e736f725f7265662e68220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f636f6c6c6563746976652f64656661756c745f6570696c6f6775652e687070220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f7468726561642f6c696e6561725f636f6d62696e6174696f6e2e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f64697370617463685f706f6c6963792e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f67726f75705f61727261795f70726f626c656d5f73686170652e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f636f6c6c6563746976652f636f6c6c6563746976655f6275696c6465722e687070220a2020202023696e636c75646520226375746c6173732f6570696c6f6775652f636f6c6c6563746976652f636f6c6c6563746976655f6275696c6465722e687070220a2020202023696e636c75646520226375746c6173732f67656d6d2f6465766963652f67656d6d5f756e6976657273616c5f616461707465722e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f6b65726e656c2f67656d6d5f756e6976657273616c2e687070220a2020202023696e636c75646520226375746c6173732f7574696c2f7061636b65645f7374726964652e687070220a2020202023696e636c75646520226375746c6173732f6b65726e656c5f68617264776172655f696e666f2e68220a2020202023696e636c75646520226375746c6173732f7574696c2f6465766963655f6d656d6f72792e68220a2020202023696e636c75646520226375746c6173732f67656d6d2f6b65726e656c2f74696c655f7363686564756c65725f706172616d732e68220a0a2020202023646566696e65204355544c4153535f434845434b287374617475732920202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020646f207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a20202020202020206375746c6173733a3a537461747573205f737461747573203d2028737461747573293b202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020696620285f73746174757320213d206375746c6173733a3a5374617475733a3a6b5375636365737329207b20202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020202020207468726f77207374643a3a72756e74696d655f6572726f7228224355544c415353206572726f7222293b202020202020202020202020202020202020202020202020202020202020202020205c0a20202020202020207d20202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020207d207768696c65202830290a0a2020202023646566696e6520435544415f434845434b2865787072292020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020646f207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020637564614572726f725f74205f657272203d202865787072293b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020202020696620285f65727220213d20637564615375636365737329207b202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a202020202020202020207468726f77207374643a3a72756e74696d655f6572726f7228637564614765744572726f72537472696e67285f65727229293b202020202020202020202020202020202020202020202020205c0a20202020202020207d20202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020202020205c0a2020202020207d207768696c65202830290a0a202020207573696e672050726f626c656d5368617065203d206375746c6173733a3a67656d6d3a3a47726f757050726f626c656d53686170653c637574653a3a53686170653c696e742c696e742c696e743e3e3b0a202020207573696e6720456c656d656e74496e707574203d206375746c6173733a3a666c6f61745f65326d315f743b0a202020207573696e6720456c656d656e745346202020203d206375746c6173733a3a666c6f61745f7565346d335f743b0a202020207573696e6720456c656d656e744320202020203d206375746c6173733a3a68616c665f743b0a0a202020207573696e6720456c656d656e7441203d206375746c6173733a3a6e765f666c6f6174345f743c456c656d656e74496e7075743e3b0a202020207573696e67204c61796f75744120203d206375746c6173733a3a6c61796f75743a3a526f774d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744120203d2033323b0a0a202020207573696e6720456c656d656e7442203d206375746c6173733a3a6e765f666c6f6174345f743c456c656d656e74496e7075743e3b0a202020207573696e67204c61796f75744220203d206375746c6173733a3a6c61796f75743a3a436f6c756d6e4d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744220203d2033323b0a0a202020207573696e6720456c656d656e7444203d20456c656d656e74433b0a202020207573696e67204c61796f75744320203d206375746c6173733a3a6c61796f75743a3a436f6c756d6e4d616a6f723b0a20202020636f6e73746578707220696e7420416c69676e6d656e744320203d20323536202f206375746c6173733a3a73697a656f665f626974733c456c656d656e74433e3a3a76616c75653b0a20202020636f6e73746578707220696e7420416c69676e6d656e744420203d20323536202f206375746c6173733a3a73697a656f665f626974733c456c656d656e74443e3a3a76616c75653b0a202020207573696e6720456c656d656e74416363756d756c61746f7220203d20666c6f61743b0a0a202020207573696e672041726368546167203d206375746c6173733a3a617263683a3a536d3130303b0a202020207573696e67204f70657261746f72436c617373203d206375746c6173733a3a617263683a3a4f70436c617373426c6f636b5363616c656454656e736f724f703b0a202020207573696e67205374616765436f756e745479706531536d203d206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a5374616765436f756e743c343e3b0a202020207573696e67205374616765436f756e745479706532536d203d206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a5374616765436f756e743c353e3b0a202020207573696e6720436c75737465725368617065203d20637574653a3a53686170653c696e7433325f742c696e7433325f742c637574653a3a5f313e3b0a0a20202020737472756374204d4d4131534d436f6e666967207b0a2020202020207573696e67204d6d6154696c65536861706520202020203d20637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36342c637574653a3a5f3235363e3b0a2020202020207573696e67204b65726e656c5363686564756c652020203d206375746c6173733a3a67656d6d3a3a4b65726e656c5074724172726179546d61576172705370656369616c697a656431536d4e766634536d3130303b0a2020202020207573696e67204570696c6f6775655363686564756c65203d206375746c6173733a3a6570696c6f6775653a3a5074724172726179546d61576172705370656369616c697a656431536d3b0a202020207d3b0a0a20202020737472756374204d4d4132534d436f6e666967207b0a2020202020207573696e67204d6d6154696c65536861706520202020203d20637574653a3a53686170653c637574653a3a5f3235362c637574653a3a5f36342c637574653a3a5f3235363e3b0a2020202020207573696e67204b65726e656c5363686564756c652020203d206375746c6173733a3a67656d6d3a3a4b65726e656c5074724172726179546d61576172705370656369616c697a656432536d4e766634536d3130303b0a2020202020207573696e67204570696c6f6775655363686564756c65203d206375746c6173733a3a6570696c6f6775653a3a5074724172726179546d61576172705370656369616c697a656432536d3b0a202020207d3b0a0a202020207573696e6720436f6c6c6563746976654570696c6f67756531534d203d20747970656e616d65206375746c6173733a3a6570696c6f6775653a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a2020202020202020417263685461672c204f70657261746f72436c6173732c0a2020202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020202020637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36343e2c0a2020202020202020456c656d656e74416363756d756c61746f722c20456c656d656e74416363756d756c61746f722c0a2020202020202020456c656d656e74432c204c61796f757443202a2c20416c69676e6d656e74432c0a2020202020202020456c656d656e74442c204c61796f757443202a2c20416c69676e6d656e74442c0a2020202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4570696c6f6775655363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e6720436f6c6c6563746976654d61696e6c6f6f7031534d203d20747970656e616d65206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a202020202020417263685461672c204f70657261746f72436c6173732c0a202020202020456c656d656e74412c204c61796f757441202a2c20416c69676e6d656e74412c0a202020202020456c656d656e74422c204c61796f757442202a2c20416c69676e6d656e74422c0a202020202020456c656d656e74416363756d756c61746f722c0a202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020205374616765436f756e745479706531536d2c0a202020202020747970656e616d65204d4d4131534d436f6e6669673a3a4b65726e656c5363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e672047656d6d4b65726e656c31534d203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a47656d6d556e6976657273616c3c0a202020202020202050726f626c656d53686170652c0a2020202020202020436f6c6c6563746976654d61696e6c6f6f7031534d2c0a2020202020202020436f6c6c6563746976654570696c6f67756531534d2c0a20202020202020206375746c6173733a3a67656d6d3a3a53747265616d4b5363686564756c65720a202020203e3b0a202020207573696e672047656d6d31534d203d206375746c6173733a3a67656d6d3a3a6465766963653a3a47656d6d556e6976657273616c416461707465723c47656d6d4b65726e656c31534d3e3b0a0a202020207573696e6720436f6c6c6563746976654570696c6f67756532534d203d20747970656e616d65206375746c6173733a3a6570696c6f6775653a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a2020202020202020417263685461672c204f70657261746f72436c6173732c0a2020202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020202020637574653a3a53686170653c637574653a3a5f3132382c637574653a3a5f36343e2c0a2020202020202020456c656d656e74416363756d756c61746f722c20456c656d656e74416363756d756c61746f722c0a2020202020202020456c656d656e74432c204c61796f757443202a2c20416c69676e6d656e74432c0a2020202020202020456c656d656e74442c204c61796f757443202a2c20416c69676e6d656e74442c0a2020202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4570696c6f6775655363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e6720436f6c6c6563746976654d61696e6c6f6f7032534d203d20747970656e616d65206375746c6173733a3a67656d6d3a3a636f6c6c6563746976653a3a436f6c6c6563746976654275696c6465723c0a202020202020417263685461672c204f70657261746f72436c6173732c0a202020202020456c656d656e74412c204c61796f757441202a2c20416c69676e6d656e74412c0a202020202020456c656d656e74422c204c61796f757442202a2c20416c69676e6d656e74422c0a202020202020456c656d656e74416363756d756c61746f722c0a202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4d6d6154696c6553686170652c20436c757374657253686170652c0a2020202020205374616765436f756e745479706532536d2c0a202020202020747970656e616d65204d4d4132534d436f6e6669673a3a4b65726e656c5363686564756c650a202020203e3a3a436f6c6c6563746976654f703b0a0a202020207573696e672047656d6d4b65726e656c32534d203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a47656d6d556e6976657273616c3c0a202020202020202050726f626c656d53686170652c0a2020202020202020436f6c6c6563746976654d61696e6c6f6f7032534d2c0a2020202020202020436f6c6c6563746976654570696c6f67756532534d2c0a20202020202020206375746c6173733a3a67656d6d3a3a53747265616d4b5363686564756c65720a202020203e3b0a202020207573696e672047656d6d32534d203d206375746c6173733a3a67656d6d3a3a6465766963653a3a47656d6d556e6976657273616c416461707465723c47656d6d4b65726e656c32534d3e3b0a0a2020202074656d706c617465203c747970656e616d6520543e0a202020207374727563742050696e6e6564486f7374427566666572207b0a202020202020542a20707472203d206e756c6c7074723b0a20202020202073697a655f7420636f756e74203d20303b0a0a20202020202050696e6e6564486f73744275666665722829203d2064656661756c743b0a2020202020206578706c696369742050696e6e6564486f73744275666665722873697a655f74206e29207b20616c6c6f63617465286e293b207d0a0a202020202020766f696420616c6c6f636174652873697a655f74206e29207b0a20202020202020206966202870747220262620636f756e74203e3d206e29207b0a2020202020202020202072657475726e3b0a20202020202020207d0a202020202020202072656c6561736528293b0a2020202020202020636f756e74203d206e3b0a2020202020202020435544415f434845434b2863756461486f7374416c6c6f63287265696e746572707265745f636173743c766f69642a2a3e2826707472292c206e202a2073697a656f662854292c2063756461486f7374416c6c6f63506f727461626c6529293b0a2020202020207d0a0a202020202020766f69642072656c656173652829207b0a20202020202020206966202870747229207b0a202020202020202020206375646146726565486f737428707472293b0a20202020202020202020707472203d206e756c6c7074723b0a20202020202020202020636f756e74203d20303b0a20202020202020207d0a2020202020207d0a0a2020202020207e50696e6e6564486f73744275666665722829207b2072656c6561736528293b207d0a0a202020202020542a2064617461282920636f6e7374207b2072657475726e207074723b207d0a2020202020205426206f70657261746f725b5d2873697a655f742069647829207b2072657475726e207074725b6964785d3b207d0a202020207d3b0a0a2020202074656d706c617465203c747970656e616d6520543e0a2020202073747275637420446576696365427566666572207b0a202020202020542a20707472203d206e756c6c7074723b0a20202020202073697a655f7420636f756e74203d20303b0a0a2020202020204465766963654275666665722829203d2064656661756c743b0a2020202020206578706c69636974204465766963654275666665722873697a655f74206e29207b20616c6c6f63617465286e293b207d0a0a202020202020766f696420616c6c6f636174652873697a655f74206e29207b0a20202020202020206966202870747220262620636f756e74203e3d206e29207b0a2020202020202020202072657475726e3b0a20202020202020207d0a202020202020202072656c6561736528293b0a2020202020202020636f756e74203d206e3b0a2020202020202020435544415f434845434b28637564614d616c6c6f63287265696e746572707265745f636173743c766f69642a2a3e2826707472292c206e202a2073697a656f6628542929293b0a2020202020207d0a0a202020202020766f69642072656c656173652829207b0a20202020202020206966202870747229207b0a20202020202020202020637564614672656528707472293b0a20202020202020202020707472203d206e756c6c7074723b0a20202020202020202020636f756e74203d20303b0a20202020202020207d0a2020202020207d0a0a2020202020207e4465766963654275666665722829207b2072656c6561736528293b207d0a0a202020202020542a20676574282920636f6e7374207b2072657475726e207074723b207d0a0a202020202020766f696420636f70795f66726f6d5f686f737428636f6e737420542a20686f73742c2073697a655f74206e2c206375646153747265616d5f742073747265616d29207b0a2020202020202020435544415f434845434b28637564614d656d6370794173796e63287074722c20686f73742c206e202a2073697a656f662854292c20637564614d656d637079486f7374546f4465766963652c2073747265616d29293b0a2020202020207d0a202020207d3b0a0a2020202074656d706c617465203c747970656e616d652047656d6d3e0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2072756e5f67726f757065645f696d706c280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a0a202020202020544f5243485f434845434b28412e73697a652829203d3d20422e73697a6528292c2022412f422073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d20432e73697a6528292c2022412f432073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d205346412e73697a6528292c2022412f5346412073697a65206d69736d6174636822293b0a202020202020544f5243485f434845434b28412e73697a652829203d3d205346422e73697a6528292c2022412f5346422073697a65206d69736d6174636822293b0a202020202020636f6e737420696e7433325f742067726f757073203d207374617469635f636173743c696e7433325f743e28412e73697a652829293b0a202020202020544f5243485f434845434b2867726f757073203e20302c20224e6f2067726f75707322293b0a0a202020202020696e74206465766963655f6964203d20415b305d2e6765745f64657669636528293b0a2020202020206331303a3a637564613a3a435544414775617264206465766963655f6775617264286465766963655f6964293b0a0a2020202020207573696e672053747269646541203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465413b0a2020202020207573696e672053747269646542203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465423b0a2020202020207573696e672053747269646543203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465433b0a2020202020207573696e672053747269646544203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a496e7465726e616c537472696465443b0a2020202020207573696e67204c61796f7574534641203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a496e7465726e616c4c61796f75745346413b0a2020202020207573696e67204c61796f7574534642203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a496e7465726e616c4c61796f75745346423b0a2020202020207573696e6720536d317878426c6b5363616c6564436f6e666967203d20747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a436f6c6c6563746976654d61696e6c6f6f703a3a536d317878426c6b5363616c6564436f6e6669673b0a0a2020202020207374617469632073697a655f74206361706163697479203d20303b0a20202020202073746174696320696e74206361636865645f646576696365203d202d313b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e207074725f415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e207074725f425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e207074725f435f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e207074725f445f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465413e207374726964655f415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465423e207374726964655f425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465433e207374726964655f435f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c537472696465443e207374726964655f445f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c4c61796f75745346413e206c61796f75745f5346415f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c4c61796f75745346423e206c61796f75745f5346425f686f73743b0a2020202020207374617469632050696e6e6564486f73744275666665723c747970656e616d652050726f626c656d53686170653a3a556e6465726c79696e6750726f626c656d53686170653e2070726f626c656d5f73697a65735f686f73743b0a0a202020202020737461746963204465766963654275666665723c747970656e616d652050726f626c656d53686170653a3a556e6465726c79696e6750726f626c656d53686170653e2070726f626c656d5f73697a65735f6465766963653b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e207074725f413b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e207074725f423b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346413b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e207074725f5346423b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e207074725f433b0a202020202020737461746963204465766963654275666665723c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e207074725f443b0a202020202020737461746963204465766963654275666665723c537472696465413e207374726964655f413b0a202020202020737461746963204465766963654275666665723c537472696465423e207374726964655f423b0a202020202020737461746963204465766963654275666665723c537472696465433e207374726964655f433b0a202020202020737461746963204465766963654275666665723c537472696465443e207374726964655f443b0a202020202020737461746963204465766963654275666665723c4c61796f75745346413e206c61796f75745f5346413b0a202020202020737461746963204465766963654275666665723c4c61796f75745346423e206c61796f75745f5346423b0a2020202020207374617469632073697a655f7420776f726b73706163655f73697a655f636163686564203d20303b0a20202020202073746174696320746f7263683a3a54656e736f7220776f726b73706163655f74656e736f723b0a0a202020202020696620286361636865645f64657669636520213d206465766963655f6964207c7c206361706163697479203c207374617469635f636173743c73697a655f743e2867726f7570732929207b0a20202020202020206361636865645f646576696365203d206465766963655f69643b0a20202020202020206361706163697479203d207374617469635f636173743c73697a655f743e2867726f757073293b0a20202020202020207074725f415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f435f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f445f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f425f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f435f686f73742e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f445f686f73742e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346415f686f73742e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346425f686f73742e616c6c6f636174652867726f757073293b0a202020202020202070726f626c656d5f73697a65735f686f73742e616c6c6f636174652867726f757073293b0a0a202020202020202070726f626c656d5f73697a65735f6465766963652e616c6c6f636174652867726f757073293b0a20202020202020207074725f412e616c6c6f636174652867726f757073293b0a20202020202020207074725f422e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346412e616c6c6f636174652867726f757073293b0a20202020202020207074725f5346422e616c6c6f636174652867726f757073293b0a20202020202020207074725f432e616c6c6f636174652867726f757073293b0a20202020202020207074725f442e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f412e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f422e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f432e616c6c6f636174652867726f757073293b0a20202020202020207374726964655f442e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346412e616c6c6f636174652867726f757073293b0a20202020202020206c61796f75745f5346422e616c6c6f636174652867726f757073293b0a2020202020207d0a0a202020202020636f6e73746578707220696e742074696c655f6d203d20637574653a3a73697a653c303e28747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c6553686170657b7d293b0a202020202020636f6e73746578707220696e742074696c655f6e203d20637574653a3a73697a653c313e28747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c6553686170657b7d293b0a202020202020696e7436345f7420746f74616c5f74696c6573203d20303b0a0a202020202020666f722028696e7433325f742069203d20303b2069203c2067726f7570733b202b2b6929207b0a2020202020202020544f5243485f434845434b28415b695d2e69735f6375646128292c202241206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b28425b695d2e69735f6375646128292c202242206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b28435b695d2e69735f6375646128292c202243206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b285346415b695d2e69735f6375646128292c2022534641206d757374206265204355444122293b0a2020202020202020544f5243485f434845434b285346425b695d2e69735f6375646128292c2022534642206d757374206265204355444122293b0a0a2020202020202020696e7433325f74206d203d207374617469635f636173743c696e7433325f743e28415b695d2e73697a65283029293b0a2020202020202020696e7433325f74206b5f7061636b6564203d207374617469635f636173743c696e7433325f743e28415b695d2e73697a65283129293b0a2020202020202020696e7433325f74206b203d206b5f7061636b6564202a20323b0a2020202020202020696e7433325f74206e203d207374617469635f636173743c696e7433325f743e28425b695d2e73697a65283029293b0a2020202020202020544f5243485f434845434b28425b695d2e73697a65283129202a2032203d3d206b2c20224b206d69736d6174636822293b0a2020202020202020544f5243485f434845434b28435b695d2e73697a65283029203d3d206d20262620435b695d2e73697a65283129203d3d206e2c202243207368617065206d69736d6174636822293b0a0a202020202020202070726f626c656d5f73697a65735f686f73745b695d203d20637574653a3a6d616b655f7368617065286e2c206d2c206b293b0a20202020202020207374726964655f415f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465417b7d2c207b6e2c206b2c20317d293b0a20202020202020207374726964655f425f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465427b7d2c207b6d2c206b2c20317d293b0a20202020202020207374726964655f435f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465437b7d2c207b6e2c206d2c20317d293b0a20202020202020207374726964655f445f686f73745b695d203d206375746c6173733a3a6d616b655f637574655f7061636b65645f73747269646528537472696465447b7d2c207b6e2c206d2c20317d293b0a20202020202020206c61796f75745f5346415f686f73745b695d203d20536d317878426c6b5363616c6564436f6e6669673a3a74696c655f61746f6d5f746f5f73686170655f53464128637574653a3a6d616b655f7368617065286e2c206d2c206b2c203129293b0a20202020202020206c61796f75745f5346425f686f73745b695d203d20536d317878426c6b5363616c6564436f6e6669673a3a74696c655f61746f6d5f746f5f73686170655f53464228637574653a3a6d616b655f7368617065286e2c206d2c206b2c203129293b0a0a2020202020202020696e742074696c65735f6d203d20286e202b2074696c655f6d202d203129202f2074696c655f6d3b0a2020202020202020696e742074696c65735f6e203d20286d202b2074696c655f6e202d203129202f2074696c655f6e3b0a2020202020202020746f74616c5f74696c6573202b3d207374617469635f636173743c696e7436345f743e2874696c65735f6d29202a207374617469635f636173743c696e7436345f743e2874696c65735f6e293b0a0a20202020202020207074725f415f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744120636f6e73742a3e28425b695d2e646174615f7074722829293b0a20202020202020207074725f425f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744220636f6e73742a3e28415b695d2e646174615f7074722829293b0a20202020202020207074725f435f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a456c656d656e744320636f6e73742a3e28435b695d2e646174615f7074722829293b0a20202020202020207074725f445f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a4570696c6f6775654f75747075744f703a3a456c656d656e744f75747075742a3e28435b695d2e646174615f7074722829293b0a20202020202020207074725f5346415f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e285346425b695d2e646174615f7074722829293b0a20202020202020207074725f5346425f686f73745b695d203d207265696e746572707265745f636173743c747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a456c656d656e74534620636f6e73742a3e285346415b695d2e646174615f7074722829293b0a2020202020207d0a0a2020202020206175746f2073747265616d5f6f626a203d206331303a3a637564613a3a67657443757272656e744355444153747265616d286465766963655f6964293b0a2020202020206331303a3a637564613a3a4355444153747265616d47756172642073747265616d5f67756172642873747265616d5f6f626a293b0a2020202020206375646153747265616d5f742073747265616d203d2073747265616d5f6f626a2e73747265616d28293b0a20202020202070726f626c656d5f73697a65735f6465766963652e636f70795f66726f6d5f686f73742870726f626c656d5f73697a65735f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f412e636f70795f66726f6d5f686f7374287074725f415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f422e636f70795f66726f6d5f686f7374287074725f425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f5346412e636f70795f66726f6d5f686f7374287074725f5346415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f5346422e636f70795f66726f6d5f686f7374287074725f5346425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f432e636f70795f66726f6d5f686f7374287074725f435f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207074725f442e636f70795f66726f6d5f686f7374287074725f445f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f412e636f70795f66726f6d5f686f7374287374726964655f415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f422e636f70795f66726f6d5f686f7374287374726964655f425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f432e636f70795f66726f6d5f686f7374287374726964655f435f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020207374726964655f442e636f70795f66726f6d5f686f7374287374726964655f445f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020206c61796f75745f5346412e636f70795f66726f6d5f686f7374286c61796f75745f5346415f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a2020202020206c61796f75745f5346422e636f70795f66726f6d5f686f7374286c61796f75745f5346425f686f73742e6461746128292c2067726f7570732c2073747265616d293b0a0a2020202020206375746c6173733a3a4b65726e656c4861726477617265496e666f2068775f696e666f3b0a20202020202068775f696e666f2e6465766963655f6964203d206465766963655f69643b0a20202020202068775f696e666f2e736d5f636f756e74203d206375746c6173733a3a4b65726e656c4861726477617265496e666f3a3a71756572795f6465766963655f6d756c746970726f636573736f725f636f756e742868775f696e666f2e6465766963655f6964293b0a202020202020696620636f6e73746578707220287374643a3a69735f73616d655f763c47656d6d2c2047656d6d31534d3e29207b0a202020202020202068775f696e666f2e636c75737465725f7368617065203d2064696d3328312c20312c2031293b0a202020202020202068775f696e666f2e636c75737465725f73686170655f66616c6c6261636b203d2064696d3328312c20312c2031293b0a2020202020207d20656c7365207b0a202020202020202068775f696e666f2e636c75737465725f7368617065203d2064696d3328322c20312c2031293b0a202020202020202068775f696e666f2e636c75737465725f73686170655f66616c6c6261636b203d2064696d3328322c20312c2031293b0a2020202020207d0a0a202020202020747970656e616d652047656d6d3a3a417267756d656e747320617267756d656e74733b0a2020202020206465636c7479706528617267756d656e74732e6570696c6f6775652e7468726561642920667573696f6e5f617267733b0a202020202020667573696f6e5f617267732e616c7068615f707472203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e626574615f707472203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e616c706861203d20456c656d656e74416363756d756c61746f722831293b0a202020202020667573696f6e5f617267732e62657461203d20456c656d656e74416363756d756c61746f722830293b0a202020202020667573696f6e5f617267732e616c7068615f7074725f6172726179203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e626574615f7074725f6172726179203d206e756c6c7074723b0a202020202020667573696f6e5f617267732e64416c706861203d207b637574653a3a5f307b7d2c20637574653a3a5f307b7d2c20307d3b0a202020202020667573696f6e5f617267732e6442657461203d207b637574653a3a5f307b7d2c20637574653a3a5f307b7d2c20307d3b0a0a202020202020747970656e616d652047656d6d3a3a47656d6d4b65726e656c3a3a54696c655363686564756c6572417267756d656e7473207363686564756c65723b0a2020202020207363686564756c65722e6d61785f7377697a7a6c655f73697a65203d20383b0a2020202020207363686564756c65722e7261737465725f6f72646572203d206375746c6173733a3a67656d6d3a3a6b65726e656c3a3a64657461696c3a3a5261737465724f726465724f7074696f6e733a3a4865757269737469633b0a0a202020202020617267756d656e7473203d20747970656e616d652047656d6d3a3a417267756d656e7473207b0a20202020202020206375746c6173733a3a67656d6d3a3a47656d6d556e6976657273616c4d6f64653a3a6b47726f757065642c0a20202020202020207b67726f7570732c2070726f626c656d5f73697a65735f6465766963652e67657428292c206e756c6c7074727d2c0a20202020202020207b7074725f412e67657428292c207374726964655f412e67657428292c207074725f422e67657428292c207374726964655f422e67657428292c0a2020202020202020207074725f5346412e67657428292c206c61796f75745f5346412e67657428292c207074725f5346422e67657428292c206c61796f75745f5346422e67657428297d2c0a20202020202020207b667573696f6e5f617267732c207074725f432e67657428292c207374726964655f432e67657428292c207074725f442e67657428292c207374726964655f442e67657428297d2c0a202020202020202068775f696e666f2c207363686564756c65720a2020202020207d3b0a0a20202020202047656d6d2067656d6d3b0a20202020202073697a655f7420776f726b73706163655f73697a65203d2047656d6d3a3a6765745f776f726b73706163655f73697a6528617267756d656e7473293b0a2020202020206966202821776f726b73706163655f74656e736f722e646566696e65642829207c7c206361636865645f64657669636520213d206465766963655f6964207c7c20776f726b73706163655f73697a655f636163686564203c20776f726b73706163655f73697a6529207b0a20202020202020206175746f206f707473203d20746f7263683a3a54656e736f724f7074696f6e7328292e64657669636528746f7263683a3a6b435544412c206465766963655f6964292e647479706528746f7263683a3a6b55496e7438293b0a2020202020202020776f726b73706163655f74656e736f72203d20746f7263683a3a656d707479287b7374617469635f636173743c6c6f6e673e28776f726b73706163655f73697a65297d2c206f707473293b0a2020202020202020776f726b73706163655f73697a655f636163686564203d20776f726b73706163655f73697a653b0a2020202020207d0a202020202020766f69642a20776f726b73706163655f707472203d20776f726b73706163655f73697a65203f20776f726b73706163655f74656e736f722e646174615f7074722829203a206e756c6c7074723b0a0a2020202020204355544c4153535f434845434b2867656d6d2e63616e5f696d706c656d656e7428617267756d656e747329293b0a2020202020204355544c4153535f434845434b2867656d6d2e696e697469616c697a6528617267756d656e74732c20776f726b73706163655f7074722c2073747265616d29293b0a2020202020206175746f2072756e5f737461747573203d2067656d6d2e72756e2873747265616d2c206e756c6c7074722c2074727565293b0a2020202020206966202872756e5f73746174757320213d206375746c6173733a3a5374617475733a3a6b5375636365737329207b0a202020202020202072756e5f737461747573203d2067656d6d2e72756e2873747265616d293b0a2020202020207d0a2020202020204355544c4153535f434845434b2872756e5f737461747573293b0a0a20202020202072657475726e20433b0a202020207d0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f31736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a20202020202072657475726e2072756e5f67726f757065645f696d706c3c47656d6d31534d3e28412c20422c20432c205346412c20534642293b0a202020207d0a0a202020207374643a3a766563746f723c746f7263683a3a54656e736f723e2067726f757065645f67656d6d5f32736d280a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620422c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e2620432c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e26205346412c0a2020202020202020636f6e7374207374643a3a766563746f723c746f7263683a3a54656e736f723e262053464229207b0a20202020202072657475726e2072756e5f67726f757065645f696d706c3c47656d6d32534d3e28412c20422c20432c205346412c20534642293b0a202020207d0a202020200a"+++++ def _speedk_dec_hex(h: str) -> str:+ return bytes.fromhex(h).decode("utf-8")+++ def _speedk_repo_root() -> str:+ return os.path.abspath(os.path.join(_ROOT, "..", "..", "..", ".."))+++ def _speedk_ensure_cutlass_dir(diag_fn) -> str | None:+ global _speedk_cutlass_root+ if _speedk_cutlass_root is not None:+ return _speedk_cutlass_root++ repo_root = _speedk_repo_root()+ base_dir = os.path.join(repo_root, "tmp", "cutlass_cache")+ root = os.path.join(base_dir, "cutlass-4.3.5")+ inc = os.path.join(root, "include", "cutlass", "cutlass.h")+ if os.path.isfile(inc):+ _speedk_cutlass_root = root+ return root++ try:+ os.makedirs(base_dir, exist_ok=True)+ tgz_path = os.path.join(base_dir, "cutlass.tgz")+ if not os.path.isfile(tgz_path):+ import urllib.request++ url = "https://github.com/NVIDIA/cutlass/archive/refs/tags/v4.3.5.tar.gz"+ with urllib.request.urlopen(url, timeout=60) as resp:+ data = resp.read()+ with open(tgz_path, "wb") as f:+ f.write(data)++ import tarfile++ with tarfile.open(tgz_path, "r:gz") as tf:+ tf.extractall(base_dir)++ if os.path.isfile(inc):+ _speedk_cutlass_root = root+ return root+ except Exception:+ diag_fn("speedk ext: cutlass fetch fail")+ return None++ return None+++ def _speedk_find_inc(diag_fn) -> list[str]:+ global _speedk_inc_cache+ if _speedk_inc_cache is not None:+ return _speedk_inc_cache++ incs: list[str] = []++ def _add_cutlass(base: str) -> bool:+ inc = os.path.join(base, "include")+ if os.path.isfile(os.path.join(inc, "cutlass", "cutlass.h")):+ incs.append(inc)+ util_inc = os.path.join(base, "tools", "util", "include")+ if os.path.isfile(os.path.join(util_inc, "cutlass", "util", "device_memory.h")):+ incs.append(util_inc)+ return True++ inc2 = os.path.join(base, "cutlass", "include")+ if os.path.isfile(os.path.join(inc2, "cutlass", "cutlass.h")):+ incs.append(inc2)+ util_inc2 = os.path.join(base, "cutlass", "tools", "util", "include")+ if os.path.isfile(os.path.join(util_inc2, "cutlass", "util", "device_memory.h")):+ incs.append(util_inc2)+ return True++ return False++ repo_root = _speedk_repo_root()+ for base in (+ os.path.join(repo_root, "third_party", "cutlass"),+ os.path.join(repo_root, "tmp", "cutlass-4.3.5"),+ ):+ _add_cutlass(base)++ try:+ root = os.path.dirname(getattr(cutlass, "__file__", "") or "")+ if root:+ bases = [root, os.path.dirname(root), os.path.dirname(os.path.dirname(root))]+ for base in bases:+ if _add_cutlass(base):+ break+ except Exception:+ pass++ try:+ import torch.utils.cpp_extension as _ce++ for inc in _ce.include_paths():+ if os.path.isfile(os.path.join(inc, "cutlass", "cutlass.h")):+ incs.append(inc)+ util_inc3 = os.path.join(inc, "cutlass", "tools", "util", "include")+ if os.path.isfile(os.path.join(util_inc3, "cutlass", "util", "device_memory.h")):+ incs.append(util_inc3)+ except Exception:+ pass++ if not incs:+ root = _speedk_ensure_cutlass_dir(diag_fn)+ if root:+ _add_cutlass(root)++ incs = list(dict.fromkeys(incs))+ if not incs:+ diag_fn("speedk ext: no headers")+ _speedk_inc_cache = incs+ return incs+++ def _speedk_load_ext(diag_fn):+ global _speedk_ext_mod, _speedk_ext_fail+ if _speedk_ext_mod is not None or _speedk_ext_fail:+ return _speedk_ext_mod++ incs = _speedk_find_inc(diag_fn)+ if not incs:+ _speedk_ext_fail = True+ return None++ try:+ from torch.utils.cpp_extension import load_inline++ repo_root = _speedk_repo_root()+ os.environ.setdefault(+ "TORCH_EXTENSIONS_DIR", os.path.join(repo_root, "tmp", "torch_extensions")+ )+ os.environ.setdefault("TORCH_CUDA_ARCH_LIST", "10.0a")++ cpp_src = _speedk_dec_hex(_SPEEDK_CPP_HEX)+ cu_src = _speedk_dec_hex(_SPEEDK_CU_HEX)++ _speedk_ext_mod = load_inline(+ name="nvfp4_grouped_cutlass_ext_speedk",+ cpp_sources=cpp_src,+ cuda_sources=cu_src,+ functions=["grouped_gemm_1sm", "grouped_gemm_2sm"],+ extra_cuda_cflags=[+ "-O3",+ "--use_fast_math",+ "-std=c++17",+ "--diag-suppress=144",+ ],+ extra_ldflags=["-lcuda"],+ extra_cflags=["-O3", "-std=c++17"],+ with_cuda=True,+ extra_include_paths=incs,+ verbose=False,+ )+ except Exception:+ _speedk_ext_fail = True+ _speedk_ext_mod = None+ diag_fn("speedk ext: build fail")+ return None++ return _speedk_ext_mod+++ def _speedk_pad_rows(tensor: torch.Tensor, new_rows: int) -> torch.Tensor:+ if tensor.size(0) == new_rows:+ return tensor+ dev_idx = tensor.device.index if tensor.device.index is not None else -1+ key = (new_rows, tensor.size(1), tensor.dtype, dev_idx)+ out = _speedk_pad_cache.get(key)+ if out is None or out.size(0) != new_rows or out.size(1) != tensor.size(1):+ out = torch.empty((new_rows, tensor.size(1)), device=tensor.device, dtype=tensor.dtype)+ _speedk_pad_cache[key] = out+ # Some low-bit dtypes (notably fp4x2) don't reliably support in-place fill or+ # slice assignment. Bitcast to u8 so we only move raw bytes.+ if tensor.element_size() == 1:+ src = tensor if tensor.is_contiguous() else tensor.contiguous()+ out_u8 = out.view(torch.uint8)+ out_u8.zero_()+ out_u8[: src.size(0), :] = src.view(torch.uint8)+ return out+ out.zero_()+ out[: tensor.size(0), :] = tensor+ return out+++ def _speedk_alloc_c_pad(rows: int, cols: int, ref: torch.Tensor) -> torch.Tensor:+ dev_idx = ref.device.index if ref.device.index is not None else -1+ key = (rows, cols, ref.dtype, dev_idx)+ out = _speedk_c_pad_cache.get(key)+ if out is None or out.size(0) != rows or out.size(1) != cols:+ out = torch.empty((rows, cols), device=ref.device, dtype=ref.dtype)+ _speedk_c_pad_cache[key] = out+ return out+++ def _speedk_try_ext(+ ext_mod,+ abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],+ sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],+ problem_sizes: list[tuple[int, int, int, int]],+ ):+ if ext_mod is None:+ return None++ def _use_2sm(n_orig: int, k: int) -> bool:+ return n_orig >= 3072 and k >= 2048++ items = []+ for (a_ref, b_ref, c_ref), (sfa_reordered, sfb_reordered), (m, _n, k, l) in zip(+ abc_tensors, sfs_reordered, problem_sizes+ ):+ if l != 1:+ return None+ a2 = a_ref[:, :, 0]+ b2 = b_ref[:, :, 0]+ c2 = c_ref[:, :, 0]+ n = b2.size(0)+ m_pad = ((m + 31) // 32) * 32+ if m_pad != m:+ a2 = _speedk_pad_rows(a2, m_pad)+ c_pad = _speedk_alloc_c_pad(m_pad, n, c2)+ items.append((a2, b2, c_pad, sfa_reordered, sfb_reordered, c2, m, n, k))+ else:+ items.append((a2, b2, c2, sfa_reordered, sfb_reordered, None, m, n, k))++ def _gather(src):+ return (+ [x[0] for x in src],+ [x[1] for x in src],+ [x[2] for x in src],+ [x[3] for x in src],+ [x[4] for x in src],+ )++ small = [it for it in items if not _use_2sm(it[7], it[8])]+ large = [it for it in items if _use_2sm(it[7], it[8])]++ try:+ if large:+ a_list, b_list, c_list, sfa_list, sfb_list = _gather(large)+ ext_mod.grouped_gemm_2sm(a_list, b_list, c_list, sfa_list, sfb_list)+ if small:+ a_list, b_list, c_list, sfa_list, sfb_list = _gather(small)+ ext_mod.grouped_gemm_1sm(a_list, b_list, c_list, sfa_list, sfb_list)+ except Exception:+ try:+ a_list, b_list, c_list, sfa_list, sfb_list = _gather(items)+ ext_mod.grouped_gemm_1sm(a_list, b_list, c_list, sfa_list, sfb_list)+ except Exception:+ return None++ for _a2, _b2, c2, _sfa, _sfb, c_orig, m, _n, _k in items:+ if c_orig is not None:+ c_orig.copy_(c2[:m, :])++ return [c_ref for (_a, _b, c_ref) in abc_tensors]++@dsl_user_opdef _normalize_ptr(ptr, *, loc=None, ip=None):if isinstance(ptr, _ir.Value):⋯ 2885 unchanged linessfs_reordered = _normalize_sfs(sfs_reordered)+ # Fast path: grouped CUTLASS kernel using swap + speedK-style scheduling.+ if abc_tensors and _is_sm100_device(abc_tensors[0][0]):+ speedk_ok = _try_speedk_ext(abc_tensors, sfs_reordered, problem_sizes)+ if speedk_ok:+ return [c for (_a, _b, c) in abc_tensors]+idx_unswapped: list[int] = []idx_rest: list[int] = []for i, (_m, _n, k, _l) in enumerate(problem_sizes):
scrolls · 343 diff lines total
Best evidence level for this revision: reported
JSON