submission 449841
oofbaroomf · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 3299 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-group-gemm-449841?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:a58a5682e4150b93a640ebe7880d7b3acde9c2c5db087748e9f8b51fb8e16db1
license declaredunknown
license concludedunknown
authorsoofbaroomf
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
fused-epilogue
self.epi_tile = sm100_utils.compute_epilogue_tile_shape(mbarrier
self.epilog_sync_barrier = pipeline.NamedBarrier(persistent-kernel
tile_sched_params: utils.PersistentTileSchedulerParams,shared-memory
self.smem_capacity = utils.get_smem_capacity_in_bytes("sm_100")tcgen05
tcgen05.CtaGroup.TWO if self.use_2cta_instrs else tcgen05.CtaGroup.ONEwarp-specialization
ab_pipeline_producer_group = pipeline.CooperativeGroup(pipeline.Agent.Thread)Kernel source
submission.py3299 lines
import os
import sys
import importlib.util
from inspect import isclass
from typing import Any, List, Tuple, Type, Union
import torch
from task import input_t, output_t
_ROOT = os.path.dirname(__file__)
_LOCAL_CUTE = os.path.abspath(
os.path.join(
_ROOT,
"..",
"..",
"..",
"..",
"third_party",
"cutlass",
"python",
"CuTeDSL",
)
)
if os.path.isdir(_LOCAL_CUTE) and _LOCAL_CUTE not in sys.path:
sys.path.insert(0, _LOCAL_CUTE)
_CACHE_DIR = os.path.abspath(
os.path.join(_ROOT, "..", "..", "..", "..", "kernel_context", "cute_cache")
)
try:
os.makedirs(_CACHE_DIR, exist_ok=True)
except Exception:
_CACHE_DIR = os.path.join("/tmp", "cute_cache")
os.makedirs(_CACHE_DIR, exist_ok=True)
os.environ.setdefault("CUTE_DSL_CACHE_DIR", _CACHE_DIR)
os.environ.setdefault("CUTE_DSL_JIT_CACHE", _CACHE_DIR)
os.environ.setdefault("CUTE_DSL_KEEP_PTX", "1")
os.environ.setdefault("CUTE_DSL_KEEP_CUBIN", "1")
import cutlass
import cutlass.cute as cute
import cutlass._mlir.dialects.cute as _cute_ir
from cutlass._mlir import ir as _ir
from cutlass._mlir.dialects import llvm as _llvm
from cutlass.cute.nvgpu import cpasync, tcgen05
import cutlass.torch as cutlass_torch
import cutlass.utils as utils
import cutlass.pipeline as pipeline
from cutlass.pipeline import pipeline_init_arrive, pipeline_init_wait
import cutlass.utils.blackwell_helpers as sm100_utils
import cutlass.utils.blockscaled_layout as blockscaled_utils
from cutlass.cute.runtime import from_dlpack
from cutlass.cutlass_dsl import dsl_user_op
_kernel_cache: dict[tuple, Any] = {}
_fake_st: Any | None = None
_stage8_mod: Any | None = None
_stage8_checked = False
_enable_stage8 = os.environ.get("NVFP4_ENABLE_STAGE8", "") == "1"
def _is_sm100_device(t: torch.Tensor) -> bool:
if t.device.type != "cuda":
return False
try:
major, _minor = torch.cuda.get_device_capability(t.device)
except Exception:
return False
return major >= 10
def _try_stage8_ext(
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs_reordered: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
):
global _stage8_mod, _stage8_checked
if _stage8_checked and _stage8_mod is None:
return None
if _stage8_mod is None:
_stage8_checked = True
try:
stage8_path = os.path.join(_ROOT, "submission_b200_stage8.py")
if not os.path.isfile(stage8_path):
return None
spec = importlib.util.spec_from_file_location(
"nvfp4_group_gemm_stage8_ext", stage8_path
)
if spec is None or spec.loader is None:
return None
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
_stage8_mod = mod
except Exception:
_stage8_mod = None
return None
try:
return _stage8_mod._try_ext(abc_tensors, sfs_reordered, problem_sizes)
except Exception:
return None
@dsl_user_op
def _normalize_ptr(ptr, *, loc=None, ip=None):
if isinstance(ptr, _ir.Value):
return ptr
if hasattr(ptr, "to_llvm_ptr") and callable(ptr.to_llvm_ptr):
return ptr.to_llvm_ptr(loc=loc, ip=ip)
return ptr
@dsl_user_op
def _ptx_prefetch_global(ptr, *, loc=None, ip=None):
ptr = _normalize_ptr(ptr, loc=loc, ip=ip)
if not isinstance(ptr, _ir.Value):
return None
_llvm.inline_asm(
None,
[ptr],
"prefetch.global.L2::evict_last [$0];",
"l",
has_side_effects=True,
is_align_stack=False,
asm_dialect=_llvm.AsmDialect.AD_ATT,
loc=loc,
ip=ip,
)
return None
@dsl_user_op
def _ptx_prefetch_global_l1(ptr, *, loc=None, ip=None):
ptr = _normalize_ptr(ptr, loc=loc, ip=ip)
if not isinstance(ptr, _ir.Value):
return None
_llvm.inline_asm(
None,
[ptr],
"prefetch.global.L1 [$0];",
"l",
has_side_effects=True,
is_align_stack=False,
asm_dialect=_llvm.AsmDialect.AD_ATT,
loc=loc,
ip=ip,
)
return None
@dsl_user_op
def _ptx_prefetchu_l1(ptr, *, loc=None, ip=None):
ptr = _normalize_ptr(ptr, loc=loc, ip=ip)
if not isinstance(ptr, _ir.Value):
return None
_llvm.inline_asm(
None,
[ptr],
"prefetchu.L1 [$0];",
"l",
has_side_effects=True,
is_align_stack=False,
asm_dialect=_llvm.AsmDialect.AD_ATT,
loc=loc,
ip=ip,
)
return None
@dsl_user_op
def _prefetch_tma(
atom: cute.CopyAtom,
src: cute.Tensor,
tma_desc_ptr: cute.Pointer,
*,
loc=None,
ip=None,
):
if hasattr(src, "iterator"):
_ptx_prefetch_global(src.iterator, loc=loc, ip=ip)
_ptx_prefetch_global_l1(src.iterator, loc=loc, ip=ip)
if tma_desc_ptr is not None:
_ptx_prefetch_global(tma_desc_ptr, loc=loc, ip=ip)
_ptx_prefetch_global_l1(tma_desc_ptr, loc=loc, ip=ip)
_ptx_prefetchu_l1(tma_desc_ptr, loc=loc, ip=ip)
dummy_tma_bar_ptr = cute.make_ptr(
cutlass.Int64, 0, cute.AddressSpace.smem, loc=loc, ip=ip
)
dummy_mcast_mask = cutlass.Int16(0, loc=loc, ip=ip)
value = atom._unpack(
tma_bar_ptr=dummy_tma_bar_ptr,
mcast_mask=dummy_mcast_mask,
tma_desc_ptr=tma_desc_ptr,
loc=loc,
ip=ip,
)
return _cute_ir.prefetch(value, src.value, loc=loc, ip=ip)
class Sm100GroupedBlockScaledGemmKernel:
def __init__(
self,
sf_vec_size: int,
mma_tiler_mn: Tuple[int, int],
cluster_shape_mn: Tuple[int, int],
use_tma_store: bool = True,
max_ab_stage: int | None = None,
prefetch_dist: int | None = None,
force_c_stage: int | None = None,
c_assumed_align: int = 16,
c_divisibility: int = 8,
swizzle_size: int = 1,
raster_along_m: bool = True,
):
self.acc_dtype = cutlass.Float32
self.sf_vec_size = sf_vec_size
self.use_2cta_instrs = mma_tiler_mn[0] == 256
self.cluster_shape_mn = cluster_shape_mn
self.use_tma_store = use_tma_store
self.max_ab_stage = max_ab_stage
self.prefetch_dist_override = prefetch_dist
self.force_c_stage = force_c_stage
self.c_assumed_align = c_assumed_align
self.c_divisibility = c_divisibility
self.swizzle_size = swizzle_size
self.raster_along_m = raster_along_m
# K dimension is deferred in _setup_attributes
self.mma_tiler = (*mma_tiler_mn, 1)
self.cta_group = (
tcgen05.CtaGroup.TWO if self.use_2cta_instrs else tcgen05.CtaGroup.ONE
)
self.tensormap_update_mode = utils.TensorMapUpdateMode.SMEM
self.occupancy = 1
# Set specialized warp ids
self.epilog_warp_id = (
0,
1,
2,
3,
)
self.mma_warp_id = 4
self.tma_warp_id = 5
self.threads_per_cta = 32 * len(
(self.mma_warp_id, self.tma_warp_id, *self.epilog_warp_id)
)
# Set barrier for epilogue sync and tmem ptr sync
self.epilog_sync_barrier = pipeline.NamedBarrier(
barrier_id=1,
num_threads=32 * len(self.epilog_warp_id),
)
self.tmem_alloc_barrier = pipeline.NamedBarrier(
barrier_id=2,
num_threads=32 * len((self.mma_warp_id, *self.epilog_warp_id)),
)
# Barrier used by MMA/TMA warps to signal A/B tensormap initialization completion
self.tensormap_ab_init_barrier = pipeline.NamedBarrier(
barrier_id=3,
num_threads=64,
)
self.smem_capacity = utils.get_smem_capacity_in_bytes("sm_100")
SM100_TMEM_CAPACITY_COLUMNS = 512
self.num_tmem_alloc_cols = SM100_TMEM_CAPACITY_COLUMNS
self.c_tile_stride: tuple[int, int] | None = None
# Set up configurations that dependent on gemm inputs.
def _setup_attributes(self):
# Compute mma instruction shapes
# (MMA_Tile_Shape_M, MMA_Tile_Shape_N, MMA_Inst_Shape_K)
self.mma_inst_shape_mn = (
self.mma_tiler[0],
self.mma_tiler[1],
)
# (CTA_Tile_Shape_M, Round_Up(MMA_Tile_Shape_N, 128), MMA_Inst_Shape_K)
self.mma_inst_shape_mn_sfb = (
self.mma_inst_shape_mn[0] // (2 if self.use_2cta_instrs else 1),
cute.round_up(self.mma_inst_shape_mn[1], 128),
)
tiled_mma = sm100_utils.make_blockscaled_trivial_tiled_mma(
self.a_dtype,
self.a_major_mode,
self.b_major_mode,
self.sf_dtype,
self.sf_vec_size,
self.cta_group,
self.mma_inst_shape_mn,
)
tiled_mma_sfb = sm100_utils.make_blockscaled_trivial_tiled_mma(
self.a_dtype,
self.a_major_mode,
self.b_major_mode,
self.sf_dtype,
self.sf_vec_size,
cute.nvgpu.tcgen05.CtaGroup.ONE,
self.mma_inst_shape_mn_sfb,
)
# Compute mma/cluster/tile shapes
mma_inst_shape_k = cute.size(tiled_mma.shape_mnk, mode=[2])
mma_inst_tile_k = 4
self.mma_tiler = (
self.mma_inst_shape_mn[0],
self.mma_inst_shape_mn[1],
mma_inst_shape_k * mma_inst_tile_k,
)
self.mma_tiler_sfb = (
self.mma_inst_shape_mn_sfb[0],
self.mma_inst_shape_mn_sfb[1],
mma_inst_shape_k * mma_inst_tile_k,
)
self.cta_tile_shape_mnk = (
self.mma_tiler[0] // cute.size(tiled_mma.thr_id.shape),
self.mma_tiler[1],
self.mma_tiler[2],
)
self.cluster_tile_shape_mnk = tuple(
x * y for x, y in zip(self.cta_tile_shape_mnk, (*self.cluster_shape_mn, 1))
)
# Compute cluster layout
self.cluster_layout_vmnk = cute.tiled_divide(
cute.make_layout((*self.cluster_shape_mn, 1)),
(tiled_mma.thr_id.shape,),
)
self.cluster_layout_sfb_vmnk = cute.tiled_divide(
cute.make_layout((*self.cluster_shape_mn, 1)),
(tiled_mma_sfb.thr_id.shape,),
)
# Compute number of multicast CTAs for A/B
self.num_mcast_ctas_a = cute.size(self.cluster_layout_vmnk.shape[2])
self.num_mcast_ctas_b = cute.size(self.cluster_layout_vmnk.shape[1])
self.num_mcast_ctas_sfb = cute.size(self.cluster_layout_sfb_vmnk.shape[1])
self.is_a_mcast = self.num_mcast_ctas_a > 1
self.is_b_mcast = self.num_mcast_ctas_b > 1
self.is_sfb_mcast = self.num_mcast_ctas_sfb > 1
# Compute epilogue subtile
self.epi_tile = sm100_utils.compute_epilogue_tile_shape(
self.cta_tile_shape_mnk,
self.use_2cta_instrs,
self.c_layout,
self.c_dtype,
)
# Setup A/B/C stage count in shared memory and ACC stage count in tensor memory
self.num_acc_stage, self.num_ab_stage, self.num_c_stage = self._compute_stages(
tiled_mma,
self.mma_tiler,
self.a_dtype,
self.b_dtype,
self.epi_tile,
self.c_dtype,
self.c_layout,
self.sf_dtype,
self.sf_vec_size,
self.smem_capacity,
self.occupancy,
self.force_c_stage,
)
max_ab = self.max_ab_stage
if max_ab is None:
max_ab = 6 if self.use_2cta_instrs else 4
self.num_ab_stage = min(self.num_ab_stage, max_ab)
self.prefetch_dist = max(0, self.num_ab_stage - 2)
if self.prefetch_dist_override is not None:
self.prefetch_dist = max(
0, min(self.prefetch_dist_override, self.num_ab_stage)
)
self.prefetch_enabled = self.prefetch_dist > 0
# Compute A/B/SFA/SFB/C shared memory layout
self.a_smem_layout_staged = sm100_utils.make_smem_layout_a(
tiled_mma,
self.mma_tiler,
self.a_dtype,
self.num_ab_stage,
)
self.b_smem_layout_staged = sm100_utils.make_smem_layout_b(
tiled_mma,
self.mma_tiler,
self.b_dtype,
self.num_ab_stage,
)
self.sfa_smem_layout_staged = blockscaled_utils.make_smem_layout_sfa(
tiled_mma,
self.mma_tiler,
self.sf_vec_size,
self.num_ab_stage,
)
self.sfb_smem_layout_staged = blockscaled_utils.make_smem_layout_sfb(
tiled_mma,
self.mma_tiler,
self.sf_vec_size,
self.num_ab_stage,
)
self.c_smem_layout_staged = sm100_utils.make_smem_layout_epi(
self.c_dtype,
self.c_layout,
self.epi_tile,
self.num_c_stage,
)
mbar_smem_bytes = self._get_mbar_smem_bytes(
num_acc_stage=self.num_acc_stage,
num_ab_stage=self.num_ab_stage,
num_c_stage=self.num_c_stage,
)
# Use utils.TensorMapUpdateMode.SMEM by default
tensormap_smem_bytes = (
Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap
* Sm100GroupedBlockScaledGemmKernel.num_tensormaps
)
if (
mbar_smem_bytes
+ tensormap_smem_bytes
+ Sm100GroupedBlockScaledGemmKernel.tensor_memory_management_bytes
> self.reserved_smem_bytes
):
raise ValueError(
f"smem consumption for mbar and tensormap {mbar_smem_bytes + tensormap_smem_bytes} exceeds the "
f"reserved smem bytes {self.reserved_smem_bytes}"
)
@cute.jit
def __call__(
self,
initial_a: cute.Tensor,
initial_b: cute.Tensor,
initial_c: cute.Tensor,
initial_sfa: cute.Tensor,
initial_sfb: cute.Tensor,
group_count: cutlass.Constexpr[int],
problem_shape_mnkl: cute.Tensor,
strides_abc: cute.Tensor,
tensor_address_abc: cute.Tensor,
tensor_address_sfasfb: cute.Tensor,
total_num_clusters: cutlass.Constexpr[int],
tensormap_cute_tensor: cute.Tensor,
max_active_clusters: cutlass.Constexpr[int],
st,
):
self.a_dtype = initial_a.element_type
self.b_dtype = initial_b.element_type
self.sf_dtype = initial_sfa.element_type
self.c_dtype = initial_c.element_type
self.a_major_mode = utils.LayoutEnum.from_tensor(initial_a).mma_major_mode()
self.b_major_mode = utils.LayoutEnum.from_tensor(initial_b).mma_major_mode()
self.c_layout = utils.LayoutEnum.from_tensor(initial_c)
if cutlass.const_expr(self.a_dtype != self.b_dtype):
raise TypeError(f"Type mismatch: {self.a_dtype} != {self.b_dtype}")
# Setup attributes that dependent on gemm inputs
self._setup_attributes()
# Setup sfa/sfb tensor by filling A/B tensor to scale factor atom layout
# ((Atom_M, Rest_M),(Atom_K, Rest_K),RestL)
sfa_layout = blockscaled_utils.tile_atom_to_shape_SF(
initial_a.shape, self.sf_vec_size
)
initial_sfa = cute.make_tensor(initial_sfa.iterator, sfa_layout)
# ((Atom_N, Rest_N),(Atom_K, Rest_K),RestL)
sfb_layout = blockscaled_utils.tile_atom_to_shape_SF(
initial_b.shape, self.sf_vec_size
)
initial_sfb = cute.make_tensor(initial_sfb.iterator, sfb_layout)
tiled_mma = sm100_utils.make_blockscaled_trivial_tiled_mma(
self.a_dtype,
self.a_major_mode,
self.b_major_mode,
self.sf_dtype,
self.sf_vec_size,
self.cta_group,
self.mma_inst_shape_mn,
)
tiled_mma_sfb = sm100_utils.make_blockscaled_trivial_tiled_mma(
self.a_dtype,
self.a_major_mode,
self.b_major_mode,
self.sf_dtype,
self.sf_vec_size,
cute.nvgpu.tcgen05.CtaGroup.ONE,
self.mma_inst_shape_mn_sfb,
)
atom_thr_size = cute.size(tiled_mma.thr_id.shape)
# Setup TMA load for A
a_op = sm100_utils.cluster_shape_to_tma_atom_A(
self.cluster_shape_mn, tiled_mma.thr_id
)
a_smem_layout = cute.slice_(self.a_smem_layout_staged, (None, None, None, 0))
tma_atom_a, tma_tensor_a = cute.nvgpu.make_tiled_tma_atom_A(
a_op,
initial_a,
a_smem_layout,
self.mma_tiler,
tiled_mma,
self.cluster_layout_vmnk.shape,
)
# Setup TMA load for B
b_op = sm100_utils.cluster_shape_to_tma_atom_B(
self.cluster_shape_mn, tiled_mma.thr_id
)
b_smem_layout = cute.slice_(self.b_smem_layout_staged, (None, None, None, 0))
tma_atom_b, tma_tensor_b = cute.nvgpu.make_tiled_tma_atom_B(
b_op,
initial_b,
b_smem_layout,
self.mma_tiler,
tiled_mma,
self.cluster_layout_vmnk.shape,
)
# Setup TMA load for SFA
sfa_op = sm100_utils.cluster_shape_to_tma_atom_A(
self.cluster_shape_mn, tiled_mma.thr_id
)
sfa_smem_layout = cute.slice_(
self.sfa_smem_layout_staged, (None, None, None, 0)
)
tma_atom_sfa, tma_tensor_sfa = cute.nvgpu.make_tiled_tma_atom_A(
sfa_op,
initial_sfa,
sfa_smem_layout,
self.mma_tiler,
tiled_mma,
self.cluster_layout_vmnk.shape,
internal_type=cutlass.Int16,
)
# Setup TMA load for SFB
sfb_op = sm100_utils.cluster_shape_to_tma_atom_SFB(
self.cluster_shape_mn, tiled_mma.thr_id
)
sfb_smem_layout = cute.slice_(
self.sfb_smem_layout_staged, (None, None, None, 0)
)
tma_atom_sfb, tma_tensor_sfb = cute.nvgpu.make_tiled_tma_atom_B(
sfb_op,
initial_sfb,
sfb_smem_layout,
self.mma_tiler_sfb,
tiled_mma_sfb,
self.cluster_layout_sfb_vmnk.shape,
internal_type=cutlass.Int16,
)
a_copy_size = cute.size_in_bytes(self.a_dtype, a_smem_layout)
b_copy_size = cute.size_in_bytes(self.b_dtype, b_smem_layout)
sfa_copy_size = cute.size_in_bytes(self.sf_dtype, sfa_smem_layout)
sfb_copy_size = cute.size_in_bytes(self.sf_dtype, sfb_smem_layout)
self.num_tma_load_bytes = (
a_copy_size + b_copy_size + sfa_copy_size + sfb_copy_size
) * atom_thr_size
# Setup TMA store for C
epi_smem_layout = cute.slice_(self.c_smem_layout_staged, (None, None, 0))
tma_atom_c, tma_tensor_c = cpasync.make_tiled_tma_atom(
cpasync.CopyBulkTensorTileS2GOp(),
initial_c,
epi_smem_layout,
self.epi_tile,
)
# Compute grid size
self.tile_sched_params, grid = self._compute_grid(
total_num_clusters,
self.cluster_shape_mn,
max_active_clusters,
swizzle_size=self.swizzle_size,
raster_along_m=self.raster_along_m,
)
self.buffer_align_bytes = 1024
self.size_tensormap_in_i64 = (
Sm100GroupedBlockScaledGemmKernel.num_tensormaps
* Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap
// 8
)
# Define shared storage for kernel
@cute.struct
class SharedStorage:
tensormap_buffer: cute.struct.MemRange[
cutlass.Int64, self.size_tensormap_in_i64
]
ab_full_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_ab_stage]
ab_empty_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_ab_stage]
acc_full_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_acc_stage]
acc_empty_mbar_ptr: cute.struct.MemRange[cutlass.Int64, self.num_acc_stage]
tmem_dealloc_mbar_ptr: cutlass.Int64
tmem_holding_buf: cutlass.Int32
# (EPI_TILE_M, EPI_TILE_N, STAGE)
sC: cute.struct.Align[
cute.struct.MemRange[
self.c_dtype,
cute.cosize(self.c_smem_layout_staged.outer),
],
self.buffer_align_bytes,
]
# (MMA, MMA_M, MMA_K, STAGE)
sA: cute.struct.Align[
cute.struct.MemRange[
self.a_dtype, cute.cosize(self.a_smem_layout_staged.outer)
],
self.buffer_align_bytes,
]
# (MMA, MMA_N, MMA_K, STAGE)
sB: cute.struct.Align[
cute.struct.MemRange[
self.b_dtype, cute.cosize(self.b_smem_layout_staged.outer)
],
self.buffer_align_bytes,
]
# (MMA, MMA_M, MMA_K, STAGE)
sSFA: cute.struct.Align[
cute.struct.MemRange[
self.sf_dtype, cute.cosize(self.sfa_smem_layout_staged)
],
self.buffer_align_bytes,
]
# (MMA, MMA_N, MMA_K, STAGE)
sSFB: cute.struct.Align[
cute.struct.MemRange[
self.sf_dtype, cute.cosize(self.sfb_smem_layout_staged)
],
self.buffer_align_bytes,
]
self.shared_storage = SharedStorage
# Launch the kernel synchronously
self.kernel(
tiled_mma,
tiled_mma_sfb,
tma_atom_a,
tma_tensor_a,
tma_atom_b,
tma_tensor_b,
tma_atom_sfa,
tma_tensor_sfa,
tma_atom_sfb,
tma_tensor_sfb,
tma_atom_c,
tma_tensor_c,
self.cluster_layout_vmnk,
self.cluster_layout_sfb_vmnk,
self.a_smem_layout_staged,
self.b_smem_layout_staged,
self.sfa_smem_layout_staged,
self.sfb_smem_layout_staged,
self.c_smem_layout_staged,
self.epi_tile,
self.tile_sched_params,
group_count,
problem_shape_mnkl,
strides_abc,
tensor_address_abc,
tensor_address_sfasfb,
tensormap_cute_tensor,
).launch(
grid=grid,
block=[self.threads_per_cta, 1, 1],
cluster=(*self.cluster_shape_mn, 1),
smem=self.shared_storage.size_in_bytes(),
**{"st" + "ream": st},
min_blocks_per_mp=1,
)
return
# GPU device kernel
@cute.kernel
def kernel(
self,
tiled_mma: cute.TiledMma,
tiled_mma_sfb: cute.TiledMma,
tma_atom_a: cute.CopyAtom,
mA_mkl: cute.Tensor,
tma_atom_b: cute.CopyAtom,
mB_nkl: cute.Tensor,
tma_atom_sfa: cute.CopyAtom,
mSFA_mkl: cute.Tensor,
tma_atom_sfb: cute.CopyAtom,
mSFB_nkl: cute.Tensor,
tma_atom_c: cute.CopyAtom,
mC_mnl: cute.Tensor,
cluster_layout_vmnk: cute.Layout,
cluster_layout_sfb_vmnk: cute.Layout,
a_smem_layout_staged: cute.ComposedLayout,
b_smem_layout_staged: cute.ComposedLayout,
sfa_smem_layout_staged: cute.Layout,
sfb_smem_layout_staged: cute.Layout,
c_smem_layout_staged: Union[cute.Layout, cute.ComposedLayout],
epi_tile: cute.Tile,
tile_sched_params: utils.PersistentTileSchedulerParams,
group_count: cutlass.Constexpr,
problem_sizes_mnkl: cute.Tensor,
strides_abc: cute.Tensor,
ptrs_abc: cute.Tensor,
ptrs_sfasfb: cute.Tensor,
tensormaps: cute.Tensor,
):
warp_idx = cute.arch.warp_idx()
warp_idx = cute.arch.make_warp_uniform(warp_idx)
if warp_idx == self.tma_warp_id:
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_a)
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_b)
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_sfa)
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_sfb)
cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_c)
# PTX prefetch base pointers to warm L2/L1 for first tile.
_ptx_prefetch_global(mA_mkl.iterator)
_ptx_prefetch_global_l1(mA_mkl.iterator)
_ptx_prefetch_global(mB_nkl.iterator)
_ptx_prefetch_global_l1(mB_nkl.iterator)
_ptx_prefetch_global(mSFA_mkl.iterator)
_ptx_prefetch_global_l1(mSFA_mkl.iterator)
_ptx_prefetch_global(mSFB_nkl.iterator)
_ptx_prefetch_global_l1(mSFB_nkl.iterator)
_ptx_prefetch_global(mC_mnl.iterator)
_ptx_prefetch_global_l1(mC_mnl.iterator)
use_2cta_instrs = cute.size(tiled_mma.thr_id.shape) == 2
#
# Setup cta/thread coordinates
#
# Coords inside cluster
bidx, bidy, bidz = cute.arch.block_idx()
mma_tile_coord_v = bidx % cute.size(tiled_mma.thr_id.shape)
is_leader_cta = mma_tile_coord_v == 0
cta_rank_in_cluster = cute.arch.make_warp_uniform(
cute.arch.block_idx_in_cluster()
)
block_in_cluster_coord_vmnk = cluster_layout_vmnk.get_flat_coord(
cta_rank_in_cluster
)
block_in_cluster_coord_sfb_vmnk = cluster_layout_sfb_vmnk.get_flat_coord(
cta_rank_in_cluster
)
# coord inside cta
tidx, _, _ = cute.arch.thread_idx()
#
# Alloc and init: tensormap buffer, a+b full/empty, accumulator full/empty, tensor memory dealloc barrier
#
smem = utils.SmemAllocator()
storage = smem.allocate(self.shared_storage)
tensormap_smem_ptr = storage.tensormap_buffer.data_ptr()
tensormap_a_smem_ptr = tensormap_smem_ptr
tensormap_b_smem_ptr = (
tensormap_a_smem_ptr
+ Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
)
tensormap_sfa_smem_ptr = (
tensormap_b_smem_ptr
+ Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
)
tensormap_sfb_smem_ptr = (
tensormap_sfa_smem_ptr
+ Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
)
tensormap_c_smem_ptr = (
tensormap_sfb_smem_ptr
+ Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
)
tmem_dealloc_mbar_ptr = storage.tmem_dealloc_mbar_ptr
tmem_holding_buf = storage.tmem_holding_buf
# Initialize mainloop ab_pipeline (barrier) and states
ab_pipeline_producer_group = pipeline.CooperativeGroup(pipeline.Agent.Thread)
num_tma_producer = self.num_mcast_ctas_a + self.num_mcast_ctas_b - 1
ab_pipeline_consumer_group = pipeline.CooperativeGroup(
pipeline.Agent.Thread, num_tma_producer
)
ab_pipeline = pipeline.PipelineTmaUmma.create(
barrier_storage=storage.ab_full_mbar_ptr.data_ptr(),
num_stages=self.num_ab_stage,
producer_group=ab_pipeline_producer_group,
consumer_group=ab_pipeline_consumer_group,
tx_count=self.num_tma_load_bytes,
cta_layout_vmnk=cluster_layout_vmnk,
)
# Initialize acc_pipeline (barrier) and states
acc_pipeline_producer_group = pipeline.CooperativeGroup(pipeline.Agent.Thread)
num_acc_consumer_threads = len(self.epilog_warp_id) * (
2 if use_2cta_instrs else 1
)
acc_pipeline_consumer_group = pipeline.CooperativeGroup(
pipeline.Agent.Thread, num_acc_consumer_threads
)
acc_pipeline = pipeline.PipelineUmmaAsync.create(
barrier_storage=storage.acc_full_mbar_ptr.data_ptr(),
num_stages=self.num_acc_stage,
producer_group=acc_pipeline_producer_group,
consumer_group=acc_pipeline_consumer_group,
cta_layout_vmnk=cluster_layout_vmnk,
)
# Tensor memory dealloc barrier init
if use_2cta_instrs:
if warp_idx == self.tma_warp_id:
num_tmem_dealloc_threads = 32
with cute.arch.elect_one():
cute.arch.mbarrier_init(
tmem_dealloc_mbar_ptr, num_tmem_dealloc_threads
)
# Cluster arrive after barrier init
pipeline_init_arrive(cluster_shape_mn=self.cluster_shape_mn, is_relaxed=True)
#
# Setup smem tensor A/B/SFA/SFB/C
#
sC = storage.sC.get_tensor(
c_smem_layout_staged.outer, swizzle=c_smem_layout_staged.inner
)
# (MMA, MMA_M, MMA_K, STAGE)
sA = storage.sA.get_tensor(
a_smem_layout_staged.outer, swizzle=a_smem_layout_staged.inner
)
# (MMA, MMA_N, MMA_K, STAGE)
sB = storage.sB.get_tensor(
b_smem_layout_staged.outer, swizzle=b_smem_layout_staged.inner
)
# (MMA, MMA_M, MMA_K, STAGE)
sSFA = storage.sSFA.get_tensor(sfa_smem_layout_staged)
# (MMA, MMA_N, MMA_K, STAGE)
sSFB = storage.sSFB.get_tensor(sfb_smem_layout_staged)
#
# Compute multicast mask for A/B/SFA/SFB buffer full
#
a_full_mcast_mask = None
b_full_mcast_mask = None
sfa_full_mcast_mask = None
sfb_full_mcast_mask = None
if cutlass.const_expr(self.is_a_mcast or self.is_b_mcast or use_2cta_instrs):
a_full_mcast_mask = cpasync.create_tma_multicast_mask(
cluster_layout_vmnk, block_in_cluster_coord_vmnk, mcast_mode=2
)
b_full_mcast_mask = cpasync.create_tma_multicast_mask(
cluster_layout_vmnk, block_in_cluster_coord_vmnk, mcast_mode=1
)
sfa_full_mcast_mask = cpasync.create_tma_multicast_mask(
cluster_layout_vmnk, block_in_cluster_coord_vmnk, mcast_mode=2
)
sfb_full_mcast_mask = cpasync.create_tma_multicast_mask(
cluster_layout_sfb_vmnk, block_in_cluster_coord_sfb_vmnk, mcast_mode=1
)
#
# Local_tile partition global tensors
#
# (bM, bK, RestM, RestK, RestL)
gA_mkl = cute.local_tile(
mA_mkl, cute.slice_(self.mma_tiler, (None, 0, None)), (None, None, None)
)
# (bN, bK, RestN, RestK, RestL)
gB_nkl = cute.local_tile(
mB_nkl, cute.slice_(self.mma_tiler, (0, None, None)), (None, None, None)
)
# (bM, bK, RestM, RestK, RestL)
gSFA_mkl = cute.local_tile(
mSFA_mkl, cute.slice_(self.mma_tiler, (None, 0, None)), (None, None, None)
)
# (bN, bK, RestN, RestK, RestL)
gSFB_nkl = cute.local_tile(
mSFB_nkl, cute.slice_(self.mma_tiler, (0, None, None)), (None, None, None)
)
# (bM, bN, RestM, RestN, RestL)
gC_mnl = cute.local_tile(
mC_mnl, cute.slice_(self.mma_tiler, (None, None, 0)), (None, None, None)
)
#
# Partition global tensor for TiledMMA_A/B/C
#
thr_mma = tiled_mma.get_slice(mma_tile_coord_v)
thr_mma_sfb = tiled_mma_sfb.get_slice(mma_tile_coord_v)
# (MMA, MMA_M, MMA_K, RestM, RestK, RestL)
tCgA = thr_mma.partition_A(gA_mkl)
# (MMA, MMA_N, MMA_K, RestN, RestK, RestL)
tCgB = thr_mma.partition_B(gB_nkl)
# (MMA, MMA_M, MMA_K, RestM, RestK, RestL)
tCgSFA = thr_mma.partition_A(gSFA_mkl)
# (MMA, MMA_N, MMA_K, RestN, RestK, RestL)
tCgSFB = thr_mma_sfb.partition_B(gSFB_nkl)
# (MMA, MMA_M, MMA_N, RestM, RestN, RestL)
tCgC = thr_mma.partition_C(gC_mnl)
#
# Partition global/shared tensor for TMA load A/B
#
# TMA load A partition_S/D
a_cta_layout = cute.make_layout(
cute.slice_(cluster_layout_vmnk, (0, 0, None, 0)).shape
)
# ((atom_v, rest_v), STAGE)
# ((atom_v, rest_v), RestM, RestK, RestL)
tAsA, tAgA = cpasync.tma_partition(
tma_atom_a,
block_in_cluster_coord_vmnk[2],
a_cta_layout,
cute.group_modes(sA, 0, 3),
cute.group_modes(tCgA, 0, 3),
)
# TMA load B partition_S/D
b_cta_layout = cute.make_layout(
cute.slice_(cluster_layout_vmnk, (0, None, 0, 0)).shape
)
# ((atom_v, rest_v), STAGE)
# ((atom_v, rest_v), RestN, RestK, RestL)
tBsB, tBgB = cpasync.tma_partition(
tma_atom_b,
block_in_cluster_coord_vmnk[1],
b_cta_layout,
cute.group_modes(sB, 0, 3),
cute.group_modes(tCgB, 0, 3),
)
# TMA Load SFA partition_S/D
sfa_cta_layout = a_cta_layout
# ((atom_v, rest_v), STAGE)
# ((atom_v, rest_v), RestM, RestK, RestL)
tAsSFA, tAgSFA = cute.nvgpu.cpasync.tma_partition(
tma_atom_sfa,
block_in_cluster_coord_vmnk[2],
sfa_cta_layout,
cute.group_modes(sSFA, 0, 3),
cute.group_modes(tCgSFA, 0, 3),
)
tAsSFA = cute.filter_zeros(tAsSFA)
tAgSFA = cute.filter_zeros(tAgSFA)
# TMA Load SFB partition_S/D
sfb_cta_layout = cute.make_layout(
cute.slice_(cluster_layout_sfb_vmnk, (0, None, 0, 0)).shape
)
# ((atom_v, rest_v), STAGE)
# ((atom_v, rest_v), RestN, RestK, RestL)
tBsSFB, tBgSFB = cute.nvgpu.cpasync.tma_partition(
tma_atom_sfb,
block_in_cluster_coord_sfb_vmnk[1],
sfb_cta_layout,
cute.group_modes(sSFB, 0, 3),
cute.group_modes(tCgSFB, 0, 3),
)
tBsSFB = cute.filter_zeros(tBsSFB)
tBgSFB = cute.filter_zeros(tBgSFB)
#
# Partition shared/tensor memory tensor for TiledMMA_A/B/C
#
# (MMA, MMA_M, MMA_K, STAGE)
tCrA = tiled_mma.make_fragment_A(sA)
# (MMA, MMA_N, MMA_K, STAGE)
tCrB = tiled_mma.make_fragment_B(sB)
# (MMA, MMA_M, MMA_N)
acc_shape = tiled_mma.partition_shape_C(self.mma_tiler[:2])
# (MMA, MMA_M, MMA_N, STAGE)
tCtAcc_fake = tiled_mma.make_fragment_C(
cute.append(acc_shape, self.num_acc_stage)
)
#
# Cluster wait before tensor memory alloc
#
pipeline_init_wait(cluster_shape_mn=self.cluster_shape_mn)
#
# Get tensormap buffer address
#
grid_dim = cute.arch.grid_dim()
tensormap_workspace_idx = (
bidz * grid_dim[1] * grid_dim[0] + bidy * grid_dim[0] + bidx
)
tensormap_manager = utils.TensorMapManager(
utils.TensorMapUpdateMode.SMEM,
Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap,
)
tensormap_a_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 0, None)].iterator
)
tensormap_b_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 1, None)].iterator
)
tensormap_sfa_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 2, None)].iterator
)
tensormap_sfb_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 3, None)].iterator
)
tensormap_c_gmem_ptr = tensormap_manager.get_tensormap_ptr(
tensormaps[(tensormap_workspace_idx, 4, None)].iterator
)
#
# Specialized TMA load warp
#
if warp_idx == self.tma_warp_id:
#
# Persistent tile scheduling loop
#
tile_sched = utils.StaticPersistentTileScheduler.create(
tile_sched_params, cute.arch.block_idx(), grid_dim
)
# grouped gemm tile scheduler helper will compute the group index for the tile we're working on
group_gemm_ts_helper = utils.GroupedGemmTileSchedulerHelper(
group_count,
tile_sched_params,
self.cluster_tile_shape_mnk,
utils.create_initial_search_state(),
)
tensormap_init_done = cutlass.Boolean(False)
# group index of last tile
last_group_idx = cutlass.Int32(-1)
work_tile = tile_sched.initial_work_tile_info()
ab_producer_state = pipeline.make_pipeline_state(
pipeline.PipelineUserType.Producer, self.num_ab_stage
)
while work_tile.is_valid_tile:
cur_tile_coord = work_tile.tile_idx
grouped_gemm_cta_tile_info = group_gemm_ts_helper.delinearize_z(
cur_tile_coord,
problem_sizes_mnkl,
)
cur_k_tile_cnt = grouped_gemm_cta_tile_info.cta_tile_count_k
cur_group_idx = grouped_gemm_cta_tile_info.group_idx
is_group_changed = cur_group_idx != last_group_idx
# skip tensormap update if we're working on the same group
if is_group_changed:
real_tensor_a = self.make_tensor_abc_for_tensormap_update(
cur_group_idx,
self.a_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
strides_abc,
ptrs_abc,
0, # 0 for tensor A
)
real_tensor_b = self.make_tensor_abc_for_tensormap_update(
cur_group_idx,
self.b_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
strides_abc,
ptrs_abc,
1, # 1 for tensor B
)
real_tensor_sfa = self.make_tensor_sfasfb_for_tensormap_update(
cur_group_idx,
self.sf_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
ptrs_sfasfb,
0, # 0 for tensor SFA
)
real_tensor_sfb = self.make_tensor_sfasfb_for_tensormap_update(
cur_group_idx,
self.sf_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
ptrs_sfasfb,
1, # 1 for tensor SFB
)
if tensormap_init_done == False:
# wait tensormap initialization complete
self.tensormap_ab_init_barrier.arrive_and_wait()
tensormap_init_done = True
if hasattr(real_tensor_a, "iterator"):
_ptx_prefetch_global(real_tensor_a.iterator)
_ptx_prefetch_global_l1(real_tensor_a.iterator)
_ptx_prefetchu_l1(real_tensor_a.iterator)
if hasattr(real_tensor_b, "iterator"):
_ptx_prefetch_global(real_tensor_b.iterator)
_ptx_prefetch_global_l1(real_tensor_b.iterator)
_ptx_prefetchu_l1(real_tensor_b.iterator)
if hasattr(real_tensor_sfa, "iterator"):
_ptx_prefetch_global(real_tensor_sfa.iterator)
_ptx_prefetch_global_l1(real_tensor_sfa.iterator)
_ptx_prefetchu_l1(real_tensor_sfa.iterator)
if hasattr(real_tensor_sfb, "iterator"):
_ptx_prefetch_global(real_tensor_sfb.iterator)
_ptx_prefetch_global_l1(real_tensor_sfb.iterator)
_ptx_prefetchu_l1(real_tensor_sfb.iterator)
_ptx_prefetch_global(tensormap_a_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_a_gmem_ptr)
_ptx_prefetchu_l1(tensormap_a_gmem_ptr)
_ptx_prefetch_global(tensormap_b_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_b_gmem_ptr)
_ptx_prefetchu_l1(tensormap_b_gmem_ptr)
_ptx_prefetch_global(tensormap_sfa_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_sfa_gmem_ptr)
_ptx_prefetchu_l1(tensormap_sfa_gmem_ptr)
_ptx_prefetch_global(tensormap_sfb_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_sfb_gmem_ptr)
_ptx_prefetchu_l1(tensormap_sfb_gmem_ptr)
tensormap_manager.update_tensormap(
(
real_tensor_a,
real_tensor_b,
real_tensor_sfa,
real_tensor_sfb,
),
(tma_atom_a, tma_atom_b, tma_atom_sfa, tma_atom_sfb),
(
tensormap_a_gmem_ptr,
tensormap_b_gmem_ptr,
tensormap_sfa_gmem_ptr,
tensormap_sfb_gmem_ptr,
),
self.tma_warp_id,
(
tensormap_a_smem_ptr,
tensormap_b_smem_ptr,
tensormap_sfa_smem_ptr,
tensormap_sfb_smem_ptr,
),
)
mma_tile_coord_mnl = (
grouped_gemm_cta_tile_info.cta_tile_idx_m
// cute.size(tiled_mma.thr_id.shape),
grouped_gemm_cta_tile_info.cta_tile_idx_n,
0,
)
#
# Slice to per mma tile index
#
# ((atom_v, rest_v), RestK)
tAgA_slice = tAgA[
(None, mma_tile_coord_mnl[0], None, mma_tile_coord_mnl[2])
]
# ((atom_v, rest_v), RestK)
tBgB_slice = tBgB[
(None, mma_tile_coord_mnl[1], None, mma_tile_coord_mnl[2])
]
# ((atom_v, rest_v), RestK)
tAgSFA_slice = tAgSFA[
(None, mma_tile_coord_mnl[0], None, mma_tile_coord_mnl[2])
]
# ((atom_v, rest_v), RestK)
tBgSFB_slice = tBgSFB[
(None, mma_tile_coord_mnl[1], None, mma_tile_coord_mnl[2])
]
tma_desc_a = tensormap_manager.get_tensormap_ptr(
tensormap_a_gmem_ptr,
cute.AddressSpace.generic,
)
tma_desc_b = tensormap_manager.get_tensormap_ptr(
tensormap_b_gmem_ptr,
cute.AddressSpace.generic,
)
tma_desc_sfa = tensormap_manager.get_tensormap_ptr(
tensormap_sfa_gmem_ptr,
cute.AddressSpace.generic,
)
tma_desc_sfb = tensormap_manager.get_tensormap_ptr(
tensormap_sfb_gmem_ptr,
cute.AddressSpace.generic,
)
if self.prefetch_enabled:
for pf_k_tile in cutlass.range(
0, min(self.prefetch_dist, cur_k_tile_cnt), unroll=1
):
_prefetch_tma(
tma_atom_a,
tAgA_slice[(None, pf_k_tile)],
tma_desc_a,
)
_prefetch_tma(
tma_atom_b,
tBgB_slice[(None, pf_k_tile)],
tma_desc_b,
)
_prefetch_tma(
tma_atom_sfa,
tAgSFA_slice[(None, pf_k_tile)],
tma_desc_sfa,
)
_prefetch_tma(
tma_atom_sfb,
tBgSFB_slice[(None, pf_k_tile)],
tma_desc_sfb,
)
# Peek (try_wait) AB buffer empty for k_tile = prefetch_k_tile_cnt
ab_producer_state.reset_count()
peek_ab_empty_status = cutlass.Boolean(1)
if ab_producer_state.count < cur_k_tile_cnt:
peek_ab_empty_status = ab_pipeline.producer_try_acquire(
ab_producer_state
)
if is_group_changed:
tensormap_manager.fence_tensormap_update(tensormap_a_gmem_ptr)
tensormap_manager.fence_tensormap_update(tensormap_b_gmem_ptr)
tensormap_manager.fence_tensormap_update(tensormap_sfa_gmem_ptr)
tensormap_manager.fence_tensormap_update(tensormap_sfb_gmem_ptr)
#
# Tma load loop
#
for k_tile in cutlass.range(0, cur_k_tile_cnt, 1, unroll=1):
# Conditionally wait for AB buffer empty
ab_pipeline.producer_acquire(
ab_producer_state, peek_ab_empty_status
)
# TMA load A/B/SFA/SFB
cute.copy(
tma_atom_a,
tAgA_slice[(None, ab_producer_state.count)],
tAsA[(None, ab_producer_state.index)],
tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
mcast_mask=a_full_mcast_mask,
tma_desc_ptr=tma_desc_a,
)
cute.copy(
tma_atom_b,
tBgB_slice[(None, ab_producer_state.count)],
tBsB[(None, ab_producer_state.index)],
tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
mcast_mask=b_full_mcast_mask,
tma_desc_ptr=tma_desc_b,
)
cute.copy(
tma_atom_sfa,
tAgSFA_slice[(None, ab_producer_state.count)],
tAsSFA[(None, ab_producer_state.index)],
tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
mcast_mask=sfa_full_mcast_mask,
tma_desc_ptr=tma_desc_sfa,
)
cute.copy(
tma_atom_sfb,
tBgSFB_slice[(None, ab_producer_state.count)],
tBsSFB[(None, ab_producer_state.index)],
tma_bar_ptr=ab_pipeline.producer_get_barrier(ab_producer_state),
mcast_mask=sfb_full_mcast_mask,
tma_desc_ptr=tma_desc_sfb,
)
if self.prefetch_enabled:
if k_tile < cur_k_tile_cnt - self.prefetch_dist:
future_k_tile = ab_producer_state.count + self.prefetch_dist
_prefetch_tma(
tma_atom_a,
tAgA_slice[(None, future_k_tile)],
tma_desc_a,
)
_prefetch_tma(
tma_atom_b,
tBgB_slice[(None, future_k_tile)],
tma_desc_b,
)
_prefetch_tma(
tma_atom_sfa,
tAgSFA_slice[(None, future_k_tile)],
tma_desc_sfa,
)
_prefetch_tma(
tma_atom_sfb,
tBgSFB_slice[(None, future_k_tile)],
tma_desc_sfb,
)
# Peek (try_wait) AB buffer empty for k_tile = prefetch_k_tile_cnt + k_tile + 1
ab_producer_state.advance()
peek_ab_empty_status = cutlass.Boolean(1)
if ab_producer_state.count < cur_k_tile_cnt:
peek_ab_empty_status = ab_pipeline.producer_try_acquire(
ab_producer_state
)
#
# Advance to next tile
#
tile_sched.advance_to_next_work()
work_tile = tile_sched.get_current_work()
last_group_idx = cur_group_idx
#
# Wait A/B buffer empty
#
ab_pipeline.producer_tail(ab_producer_state)
#
# Specialized MMA warp
#
if warp_idx == self.mma_warp_id:
#
# Initialize tensormaps for A, B, SFA and SFB
#
tensormap_manager.init_tensormap_from_atom(
tma_atom_a, tensormap_a_smem_ptr, self.mma_warp_id
)
tensormap_manager.init_tensormap_from_atom(
tma_atom_b, tensormap_b_smem_ptr, self.mma_warp_id
)
tensormap_manager.init_tensormap_from_atom(
tma_atom_sfa, tensormap_sfa_smem_ptr, self.mma_warp_id
)
tensormap_manager.init_tensormap_from_atom(
tma_atom_sfb, tensormap_sfb_smem_ptr, self.mma_warp_id
)
# indicate tensormap initialization has finished
self.tensormap_ab_init_barrier.arrive_and_wait()
#
# Bar sync for retrieve tensor memory ptr from shared mem
#
self.tmem_alloc_barrier.arrive_and_wait()
#
# Retrieving tensor memory ptr and make accumulator/SFA/SFB tensor
#
# Make accumulator tmem tensor
acc_tmem_ptr = cute.arch.retrieve_tmem_ptr(
self.acc_dtype,
alignment=16,
ptr_to_buffer_holding_addr=tmem_holding_buf,
)
# (MMA, MMA_M, MMA_N, STAGE)
tCtAcc_base = cute.make_tensor(acc_tmem_ptr, tCtAcc_fake.layout)
# Make SFA tmem tensor
sfa_tmem_ptr = cute.recast_ptr(
acc_tmem_ptr + tcgen05.find_tmem_tensor_col_offset(tCtAcc_base),
dtype=self.sf_dtype,
)
# (MMA, MMA_M, MMA_K)
tCtSFA_layout = blockscaled_utils.make_tmem_layout_sfa(
tiled_mma,
self.mma_tiler,
self.sf_vec_size,
cute.slice_(sfa_smem_layout_staged, (None, None, None, 0)),
)
tCtSFA = cute.make_tensor(sfa_tmem_ptr, tCtSFA_layout)
# Make SFB tmem tensor
sfb_tmem_ptr = cute.recast_ptr(
acc_tmem_ptr
+ tcgen05.find_tmem_tensor_col_offset(tCtAcc_base)
+ tcgen05.find_tmem_tensor_col_offset(tCtSFA),
dtype=self.sf_dtype,
)
# (MMA, MMA_N, MMA_K)
tCtSFB_layout = blockscaled_utils.make_tmem_layout_sfb(
tiled_mma,
self.mma_tiler,
self.sf_vec_size,
cute.slice_(sfb_smem_layout_staged, (None, None, None, 0)),
)
tCtSFB = cute.make_tensor(sfb_tmem_ptr, tCtSFB_layout)
#
# Partition for S2T copy of SFA/SFB
#
tiled_copy_s2t_sfa, tCsSFA_compact_s2t, tCtSFA_compact_s2t = (
self.mainloop_s2t_copy_and_partition(sSFA, tCtSFA)
)
tiled_copy_s2t_sfb, tCsSFB_compact_s2t, tCtSFB_compact_s2t = (
self.mainloop_s2t_copy_and_partition(sSFB, tCtSFB)
)
#
# Persistent tile scheduling loop
#
tile_sched = utils.StaticPersistentTileScheduler.create(
tile_sched_params, cute.arch.block_idx(), grid_dim
)
# grouped gemm tile scheduler helper will compute the group index for the tile we're working on
group_gemm_ts_helper = utils.GroupedGemmTileSchedulerHelper(
group_count,
tile_sched_params,
self.cluster_tile_shape_mnk,
utils.create_initial_search_state(),
)
work_tile = tile_sched.initial_work_tile_info()
ab_consumer_state = pipeline.make_pipeline_state(
pipeline.PipelineUserType.Consumer, self.num_ab_stage
)
acc_producer_state = pipeline.make_pipeline_state(
pipeline.PipelineUserType.Producer, self.num_acc_stage
)
while work_tile.is_valid_tile:
cur_tile_coord = work_tile.tile_idx
# MMA warp is only interested in number of tiles along K dimension
(
cur_k_tile_cnt,
cur_group_idx,
) = group_gemm_ts_helper.search_cluster_tile_count_k(
cur_tile_coord,
problem_sizes_mnkl,
)
# (MMA, MMA_M, MMA_N)
tCtAcc = tCtAcc_base[(None, None, None, acc_producer_state.index)]
# Peek (try_wait) AB buffer full for k_tile = 0
ab_consumer_state.reset_count()
peek_ab_full_status = cutlass.Boolean(1)
if ab_consumer_state.count < cur_k_tile_cnt and is_leader_cta:
peek_ab_full_status = ab_pipeline.consumer_try_wait(
ab_consumer_state
)
#
# Wait for accumulator buffer empty
#
if is_leader_cta:
acc_pipeline.producer_acquire(acc_producer_state)
#
# Reset the ACCUMULATE field for each tile
#
tiled_mma.set(tcgen05.Field.ACCUMULATE, False)
#
# Mma mainloop
#
for k_tile in range(cur_k_tile_cnt):
if is_leader_cta:
# Conditionally wait for AB buffer full
ab_pipeline.consumer_wait(
ab_consumer_state, peek_ab_full_status
)
# Copy SFA/SFB from smem to tmem
s2t_stage_coord = (
None,
None,
None,
None,
ab_consumer_state.index,
)
tCsSFA_compact_s2t_staged = tCsSFA_compact_s2t[s2t_stage_coord]
tCsSFB_compact_s2t_staged = tCsSFB_compact_s2t[s2t_stage_coord]
cute.copy(
tiled_copy_s2t_sfa,
tCsSFA_compact_s2t_staged,
tCtSFA_compact_s2t,
)
cute.copy(
tiled_copy_s2t_sfb,
tCsSFB_compact_s2t_staged,
tCtSFB_compact_s2t,
)
# tCtAcc += tCrA * tCrSFA * tCrB * tCrSFB
num_kblocks = cute.size(tCrA, mode=[2])
for kblock_idx in cutlass.range(num_kblocks, unroll_full=True):
kblock_coord = (
None,
None,
kblock_idx,
ab_consumer_state.index,
)
# Set SFA/SFB tensor to tiled_mma
sf_kblock_coord = (None, None, kblock_idx)
tiled_mma.set(
tcgen05.Field.SFA,
tCtSFA[sf_kblock_coord].iterator,
)
tiled_mma.set(
tcgen05.Field.SFB,
tCtSFB[sf_kblock_coord].iterator,
)
cute.gemm(
tiled_mma,
tCtAcc,
tCrA[kblock_coord],
tCrB[kblock_coord],
tCtAcc,
)
# Enable accumulate on tCtAcc after first kblock
tiled_mma.set(tcgen05.Field.ACCUMULATE, True)
# Async arrive AB buffer empty
ab_pipeline.consumer_release(ab_consumer_state)
# Peek (try_wait) AB buffer full for k_tile = k_tile + 1
ab_consumer_state.advance()
peek_ab_full_status = cutlass.Boolean(1)
if ab_consumer_state.count < cur_k_tile_cnt:
if is_leader_cta:
peek_ab_full_status = ab_pipeline.consumer_try_wait(
ab_consumer_state
)
#
# Async arrive accumulator buffer full
#
if is_leader_cta:
acc_pipeline.producer_commit(acc_producer_state)
acc_producer_state.advance()
#
# Advance to next tile
#
tile_sched.advance_to_next_work()
work_tile = tile_sched.get_current_work()
#
# Wait for accumulator buffer empty
#
acc_pipeline.producer_tail(acc_producer_state)
#
# Specialized epilogue warps
#
if warp_idx < self.mma_warp_id:
if cutlass.const_expr(self.use_tma_store):
# initialize tensormap for C
tensormap_manager.init_tensormap_from_atom(
tma_atom_c,
tensormap_c_smem_ptr,
self.epilog_warp_id[0],
)
#
# Alloc tensor memory buffer
#
if warp_idx == self.epilog_warp_id[0]:
cute.arch.alloc_tmem(
self.num_tmem_alloc_cols,
tmem_holding_buf,
is_two_cta=use_2cta_instrs,
)
#
# Bar sync for retrieve tensor memory ptr from shared memory
#
self.tmem_alloc_barrier.arrive_and_wait()
#
# Retrieving tensor memory ptr and make accumulator tensor
#
acc_tmem_ptr = cute.arch.retrieve_tmem_ptr(
self.acc_dtype,
alignment=16,
ptr_to_buffer_holding_addr=tmem_holding_buf,
)
# (MMA, MMA_M, MMA_N, STAGE)
tCtAcc_base = cute.make_tensor(acc_tmem_ptr, tCtAcc_fake.layout)
### Start from here
#
# Partition for epilogue
#
epi_tidx = tidx
if cutlass.const_expr(self.use_tma_store):
tiled_copy_t2r, tTR_tAcc_base, tTR_rAcc = (
self.epilog_tmem_copy_and_partition(
epi_tidx, tCtAcc_base, tCgC, epi_tile, use_2cta_instrs
)
)
tTR_rC = cute.make_rmem_tensor(tTR_rAcc.shape, self.c_dtype)
tiled_copy_r2s, tRS_rC, tRS_sC = self.epilog_smem_copy_and_partition(
tiled_copy_t2r, tTR_rC, epi_tidx, sC
)
tma_atom_c, bSG_sC, bSG_gC_partitioned = (
self.epilog_gmem_copy_and_partition(
epi_tidx, tma_atom_c, tCgC, epi_tile, sC
)
)
else:
tCtAcc_simt = self._transform_partitioned_tensor_layout(tCtAcc_base)
tCgC_simt = self._transform_partitioned_tensor_layout(tCgC)
(
tiled_copy_t2r,
tTR_tAcc_base,
tTR_rAcc,
) = self.epilog_tmem_copy_and_partition_simt(
epi_tidx, tCtAcc_simt, tCgC_simt, epi_tile, use_2cta_instrs
)
tTR_rC = cute.make_rmem_tensor(tTR_rAcc.shape, self.c_dtype)
gC_epi_simt = cute.flat_divide(tCgC_simt, epi_tile)
thr_copy_t2r = tiled_copy_t2r.get_slice(epi_tidx)
simt_atom_vec = cute.make_copy_atom(
cute.nvgpu.CopyUniversalOp(),
self.c_dtype,
)
#
# Persistent tile scheduling loop
#
tile_sched = utils.StaticPersistentTileScheduler.create(
tile_sched_params, cute.arch.block_idx(), grid_dim
)
# grouped gemm tile scheduler helper will compute the group index for the tile we're working on
group_gemm_ts_helper = utils.GroupedGemmTileSchedulerHelper(
group_count,
tile_sched_params,
self.cluster_tile_shape_mnk,
utils.create_initial_search_state(),
)
work_tile = tile_sched.initial_work_tile_info()
acc_consumer_state = pipeline.make_pipeline_state(
pipeline.PipelineUserType.Consumer, self.num_acc_stage
)
if cutlass.const_expr(self.use_tma_store):
# Threads/warps participating in tma store pipeline
c_producer_group = pipeline.CooperativeGroup(
pipeline.Agent.Thread,
32 * len(self.epilog_warp_id),
)
c_pipeline = pipeline.PipelineTmaStore.create(
num_stages=self.num_c_stage,
producer_group=c_producer_group,
)
if cutlass.const_expr(self.use_tma_store):
# group index to start searching
last_group_idx = cutlass.Int32(-1)
while work_tile.is_valid_tile:
cur_tile_coord = work_tile.tile_idx
grouped_gemm_cta_tile_info = group_gemm_ts_helper.delinearize_z(
cur_tile_coord,
problem_sizes_mnkl,
)
cur_group_idx = grouped_gemm_cta_tile_info.group_idx
is_group_changed = cur_group_idx != last_group_idx
real_tensor_c = self.make_tensor_abc_for_tensormap_update(
cur_group_idx,
self.c_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
strides_abc,
ptrs_abc,
2, # 2 for tensor C
)
if is_group_changed:
if hasattr(real_tensor_c, "iterator"):
_ptx_prefetch_global(real_tensor_c.iterator)
_ptx_prefetch_global_l1(real_tensor_c.iterator)
_ptx_prefetchu_l1(real_tensor_c.iterator)
_ptx_prefetch_global(tensormap_c_gmem_ptr)
_ptx_prefetch_global_l1(tensormap_c_gmem_ptr)
_ptx_prefetchu_l1(tensormap_c_gmem_ptr)
tensormap_manager.update_tensormap(
((real_tensor_c),),
((tma_atom_c),),
((tensormap_c_gmem_ptr),),
self.epilog_warp_id[0],
(tensormap_c_smem_ptr,),
)
mma_tile_coord_mnl = (
grouped_gemm_cta_tile_info.cta_tile_idx_m
// cute.size(tiled_mma.thr_id.shape),
grouped_gemm_cta_tile_info.cta_tile_idx_n,
0,
)
# ((ATOM_V, REST_V), EPI_M, EPI_N)
bSG_gC = bSG_gC_partitioned[
(
None,
None,
None,
*mma_tile_coord_mnl,
)
]
# Set tensor memory buffer for current tile
# (T2R, T2R_M, T2R_N, EPI_M, EPI_M)
tTR_tAcc = tTR_tAcc_base[
(None, None, None, None, None, acc_consumer_state.index)
]
#
# Wait for accumulator buffer full
#
acc_pipeline.consumer_wait(acc_consumer_state)
tTR_tAcc = cute.group_modes(tTR_tAcc, 3, cute.rank(tTR_tAcc))
#
# Store accumulator to global memory in subtiles
#
subtile_cnt = cute.size(tTR_tAcc.shape, mode=[3])
bSG_gC = cute.group_modes(bSG_gC, 1, cute.rank(bSG_gC))
if is_group_changed:
if warp_idx == self.epilog_warp_id[0]:
tensormap_manager.fence_tensormap_update(
tensormap_c_gmem_ptr
)
num_prev_subtiles = tile_sched.num_tiles_executed * subtile_cnt
for subtile_idx in range(subtile_cnt):
#
# Load accumulator from tensor memory buffer to register
#
tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]
cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)
#
# Convert to C type
#
acc_vec = tiled_copy_r2s.retile(tTR_rAcc).load()
tRS_rC.store(acc_vec.to(self.c_dtype))
#
# Store C to shared memory
#
c_buffer = (num_prev_subtiles + subtile_idx) % self.num_c_stage
cute.copy(
tiled_copy_r2s,
tRS_rC,
tRS_sC[(None, None, None, c_buffer)],
)
# Fence and barrier to make sure shared memory store is visible to TMA store
cute.arch.fence_proxy("async.shared", space="cta")
self.epilog_sync_barrier.arrive_and_wait()
#
# TMA store C to global memory
#
if warp_idx == self.epilog_warp_id[0]:
cute.copy(
tma_atom_c,
bSG_sC[(None, c_buffer)],
bSG_gC[(None, subtile_idx)],
tma_desc_ptr=tensormap_manager.get_tensormap_ptr(
tensormap_c_gmem_ptr,
cute.AddressSpace.generic,
),
)
# Fence and barrier to make sure shared memory store is visible to TMA store
c_pipeline.producer_commit()
c_pipeline.producer_acquire()
self.epilog_sync_barrier.arrive_and_wait()
#
# Async arrive accumulator buffer empty
#
with cute.arch.elect_one():
acc_pipeline.consumer_release(acc_consumer_state)
acc_consumer_state.advance()
#
# Advance to next tile
#
tile_sched.advance_to_next_work()
work_tile = tile_sched.get_current_work()
last_group_idx = cur_group_idx
else:
# group index to start searching
last_group_idx = cutlass.Int32(-1)
while work_tile.is_valid_tile:
cur_tile_coord = work_tile.tile_idx
grouped_gemm_cta_tile_info = group_gemm_ts_helper.delinearize_z(
cur_tile_coord,
problem_sizes_mnkl,
)
cur_group_idx = grouped_gemm_cta_tile_info.group_idx
real_tensor_c = self.make_tensor_abc_for_tensormap_update(
cur_group_idx,
self.c_dtype,
(
grouped_gemm_cta_tile_info.problem_shape_m,
grouped_gemm_cta_tile_info.problem_shape_n,
grouped_gemm_cta_tile_info.problem_shape_k,
),
strides_abc,
ptrs_abc,
2, # 2 for tensor C
)
if cur_group_idx != last_group_idx:
if hasattr(real_tensor_c, "iterator"):
_ptx_prefetch_global(real_tensor_c.iterator)
_ptx_prefetch_global_l1(real_tensor_c.iterator)
_ptx_prefetchu_l1(real_tensor_c.iterator)
mma_tile_coord_mnl = (
grouped_gemm_cta_tile_info.cta_tile_idx_m
// cute.size(tiled_mma.thr_id.shape),
grouped_gemm_cta_tile_info.cta_tile_idx_n,
0,
)
tile_origin_m = mma_tile_coord_mnl[0] * self.mma_tiler[0]
tile_origin_n = mma_tile_coord_mnl[1] * self.mma_tiler[1]
residue_m = (
grouped_gemm_cta_tile_info.problem_shape_m - tile_origin_m
)
residue_n = (
grouped_gemm_cta_tile_info.problem_shape_n - tile_origin_n
)
full_tile = (residue_m >= self.mma_tiler[0]) & (
residue_n >= self.mma_tiler[1]
)
gC_mnl_tile = cute.local_tile(
real_tensor_c,
cute.slice_(self.mma_tiler, (None, None, 0)),
mma_tile_coord_mnl,
)
tCgC_tile = thr_mma.partition_C(gC_mnl_tile)
tCgC_tile = self._transform_partitioned_tensor_layout(tCgC_tile)
tTR_gC = self.epilog_gmem_copy_and_partition_simt(
epi_tidx, tiled_copy_t2r, tCgC_tile, epi_tile
)
# Set tensor memory buffer for current tile
# (T2R, T2R_M, T2R_N, EPI_M, EPI_M)
tTR_tAcc = tTR_tAcc_base[
(None, None, None, None, None, acc_consumer_state.index)
]
#
# Wait for accumulator buffer full
#
acc_pipeline.consumer_wait(acc_consumer_state)
tTR_tAcc = cute.group_modes(tTR_tAcc, 3, cute.rank(tTR_tAcc))
tTR_gC = cute.group_modes(tTR_gC, 3, cute.rank(tTR_gC))
#
# Store accumulator to global memory in subtiles
#
subtile_cnt = cute.size(tTR_tAcc.shape, mode=[3])
if full_tile:
for subtile_idx in range(subtile_cnt):
#
# Load accumulator from tensor memory buffer to register
#
tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]
cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)
#
# Convert to C type
#
acc_vec = tTR_rAcc.load()
tTR_rC.store(acc_vec.to(self.c_dtype))
tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]
cute.copy(simt_atom_vec, tTR_rC, tTR_gC_mn)
else:
cC_mnl = cute.make_identity_tensor(real_tensor_c.shape)
cC_mnl_tile = cute.local_tile(
cC_mnl,
cute.slice_(self.mma_tiler, (None, None, 0)),
mma_tile_coord_mnl,
)
tCcC_tile = thr_mma.partition_C(cC_mnl_tile)
tCcC_tile = self._transform_partitioned_tensor_layout(
tCcC_tile
)
tTR_cC = self.epilog_gmem_copy_and_partition_simt(
epi_tidx, tiled_copy_t2r, tCcC_tile, epi_tile
)
tTR_cC = cute.group_modes(tTR_cC, 3, cute.rank(tTR_cC))
c_shape = real_tensor_c.shape
for subtile_idx in range(subtile_cnt):
#
# Load accumulator from tensor memory buffer to register
#
tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]
cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)
#
# Convert to C type
#
acc_vec = tTR_rAcc.load()
tTR_rC.store(acc_vec.to(self.c_dtype))
tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]
tTR_cC_mn = tTR_cC[(None, None, None, subtile_idx)]
tTR_pC = cute.make_rmem_tensor(
tTR_rC.shape, cutlass.Boolean
)
for i in range(cute.size(tTR_rC.shape)):
tTR_pC[i] = cute.elem_less(tTR_cC_mn[i], c_shape)
cute.basic_copy_if(tTR_pC, tTR_rC, tTR_gC_mn)
#
# Async arrive accumulator buffer empty
#
with cute.arch.elect_one():
acc_pipeline.consumer_release(acc_consumer_state)
acc_consumer_state.advance()
#
# Advance to next tile
#
tile_sched.advance_to_next_work()
work_tile = tile_sched.get_current_work()
last_group_idx = cur_group_idx
last_group_idx = cur_group_idx
#
# Dealloc the tensor memory buffer
#
if warp_idx == self.epilog_warp_id[0]:
cute.arch.relinquish_tmem_alloc_permit(is_two_cta=use_2cta_instrs)
self.epilog_sync_barrier.arrive_and_wait()
if warp_idx == self.epilog_warp_id[0]:
if use_2cta_instrs:
cute.arch.mbarrier_arrive(
tmem_dealloc_mbar_ptr, cta_rank_in_cluster ^ 1
)
cute.arch.mbarrier_wait(tmem_dealloc_mbar_ptr, 0)
cute.arch.dealloc_tmem(
acc_tmem_ptr, self.num_tmem_alloc_cols, is_two_cta=use_2cta_instrs
)
#
# Wait for C store complete
#
if cutlass.const_expr(self.use_tma_store):
c_pipeline.producer_tail()
@cute.jit
def make_tensor_abc_for_tensormap_update(
self,
group_idx: cutlass.Int32,
dtype: Type[cutlass.Numeric],
problem_shape_mnk: tuple[cutlass.Int32, cutlass.Int32, cutlass.Int32],
strides_abc: cute.Tensor,
tensor_address_abc: cute.Tensor,
tensor_index: int,
):
ptr_i64 = tensor_address_abc[(group_idx, tensor_index)]
if cutlass.const_expr(
not isclass(dtype) or not issubclass(dtype, cutlass.Numeric)
):
raise TypeError(
f"dtype must be a type of cutlass.Numeric, got {type(dtype)}"
)
align = 16
if cutlass.const_expr(tensor_index == 2):
align = self.c_assumed_align
tensor_gmem_ptr = cute.make_ptr(
dtype, ptr_i64, cute.AddressSpace.gmem, assumed_align=align
)
strides_tensor_gmem = strides_abc[(group_idx, tensor_index, None)]
strides_tensor_reg = cute.make_rmem_tensor(
cute.make_layout(2),
strides_abc.element_type,
)
cute.autovec_copy(strides_tensor_gmem, strides_tensor_reg)
stride_mn = strides_tensor_reg[0]
stride_k = strides_tensor_reg[1]
c1 = cutlass.Int32(1)
c0 = cutlass.Int32(0)
if cutlass.const_expr(tensor_index == 0): # tensor A
m = problem_shape_mnk[0]
k = problem_shape_mnk[2]
return cute.make_tensor(
tensor_gmem_ptr,
cute.make_layout((m, k, c1), stride=(stride_mn, stride_k, c0)),
)
elif cutlass.const_expr(tensor_index == 1): # tensor B
n = problem_shape_mnk[1]
k = problem_shape_mnk[2]
return cute.make_tensor(
tensor_gmem_ptr,
cute.make_layout((n, k, c1), stride=(stride_mn, stride_k, c0)),
)
else: # tensor C
m = problem_shape_mnk[0]
n = problem_shape_mnk[1]
tensor = cute.make_tensor(
tensor_gmem_ptr,
cute.make_layout((m, n, c1), stride=(stride_mn, stride_k, c0)),
)
leading_dim = 0 if self.c_layout.is_m_major_c() else 1
tensor.mark_layout_dynamic(leading_dim=leading_dim)
stride_order = (2, 1, 0) if leading_dim == 0 else (2, 0, 1)
div = self.c_divisibility if getattr(dtype, "width", 0) == 16 else 16
tensor.mark_compact_shape_dynamic(
mode=leading_dim,
stride_order=stride_order,
divisibility=div,
)
return tensor
@cute.jit
def make_tensor_sfasfb_for_tensormap_update(
self,
group_idx: cutlass.Int32,
dtype: Type[cutlass.Numeric],
problem_shape_mnk: tuple[cutlass.Int32, cutlass.Int32, cutlass.Int32],
tensor_address_sfasfb: cute.Tensor,
tensor_index: int,
):
ptr_i64 = tensor_address_sfasfb[(group_idx, tensor_index)]
if cutlass.const_expr(
not isclass(dtype) or not issubclass(dtype, cutlass.Numeric)
):
raise TypeError(
f"dtype must be a type of cutlass.Numeric, got {type(dtype)}"
)
tensor_gmem_ptr = cute.make_ptr(
dtype, ptr_i64, cute.AddressSpace.gmem, assumed_align=16
)
c1 = cutlass.Int32(1)
if cutlass.const_expr(tensor_index == 0): # tensor SFA
m = problem_shape_mnk[0]
k = problem_shape_mnk[2]
sfa_layout = blockscaled_utils.tile_atom_to_shape_SF(
(m, k, c1), self.sf_vec_size
)
return cute.make_tensor(
tensor_gmem_ptr,
sfa_layout,
)
else: # tensor SFB
n = problem_shape_mnk[1]
k = problem_shape_mnk[2]
sfb_layout = blockscaled_utils.tile_atom_to_shape_SF(
(n, k, c1), self.sf_vec_size
)
return cute.make_tensor(
tensor_gmem_ptr,
sfb_layout,
)
def mainloop_s2t_copy_and_partition(
self,
sSF: cute.Tensor,
tSF: cute.Tensor,
) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor, cute.Tensor]:
# (MMA, MMA_MN, MMA_K, STAGE)
tCsSF_compact = cute.filter_zeros(sSF)
# (MMA, MMA_MN, MMA_K)
tCtSF_compact = cute.filter_zeros(tSF)
# Make S2T CopyAtom and tiledCopy
copy_atom_s2t = cute.make_copy_atom(
tcgen05.Cp4x32x128bOp(self.cta_group),
self.sf_dtype,
)
tiled_copy_s2t = tcgen05.make_s2t_copy(copy_atom_s2t, tCtSF_compact)
thr_copy_s2t = tiled_copy_s2t.get_slice(0)
# ((ATOM_V, REST_V), Rest_Tiler, MMA_MN, MMA_K, STAGE)
tCsSF_compact_s2t_ = thr_copy_s2t.partition_S(tCsSF_compact)
# ((ATOM_V, REST_V), Rest_Tiler, MMA_MN, MMA_K, STAGE)
tCsSF_compact_s2t = tcgen05.get_s2t_smem_desc_tensor(
tiled_copy_s2t, tCsSF_compact_s2t_
)
# ((ATOM_V, REST_V), Rest_Tiler, MMA_MN, MMA_K)
tCtSF_compact_s2t = thr_copy_s2t.partition_D(tCtSF_compact)
return tiled_copy_s2t, tCsSF_compact_s2t, tCtSF_compact_s2t
def epilog_tmem_copy_and_partition(
self,
tidx: cutlass.Int32,
tAcc: cute.Tensor,
gC_mnl: cute.Tensor,
epi_tile: cute.Tile,
use_2cta_instrs: Union[cutlass.Boolean, bool],
) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor]:
# Make tiledCopy for tensor memory load
copy_atom_t2r = sm100_utils.get_tmem_load_op(
self.cta_tile_shape_mnk,
self.c_layout,
self.c_dtype,
self.acc_dtype,
epi_tile,
use_2cta_instrs,
)
# (EPI_TILE_M, EPI_TILE_N, EPI_M, EPI_N, STAGE)
tAcc_epi = cute.flat_divide(
tAcc[((None, None), 0, 0, None)],
epi_tile,
)
# (EPI_TILE_M, EPI_TILE_N)
tiled_copy_t2r = tcgen05.make_tmem_copy(
copy_atom_t2r, tAcc_epi[(None, None, 0, 0, 0)]
)
thr_copy_t2r = tiled_copy_t2r.get_slice(tidx)
# (T2R, T2R_M, T2R_N, EPI_M, EPI_M, STAGE)
tTR_tAcc = thr_copy_t2r.partition_S(tAcc_epi)
gC_mnl_epi = cute.flat_divide(
gC_mnl[((None, None), 0, 0, None, None, None)], epi_tile
)
# (T2R, T2R_M, T2R_N, EPI_M, EPI_N, RestM, RestN, RestL)
tTR_gC = thr_copy_t2r.partition_D(gC_mnl_epi)
# (T2R, T2R_M, T2R_N)
tTR_rAcc = cute.make_rmem_tensor(
tTR_gC[(None, None, None, 0, 0, 0, 0, 0)].shape, self.acc_dtype
)
return tiled_copy_t2r, tTR_tAcc, tTR_rAcc
def epilog_smem_copy_and_partition(
self,
tiled_copy_t2r: cute.TiledCopy,
tTR_rC: cute.Tensor,
tidx: cutlass.Int32,
sC: cute.Tensor,
) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor]:
copy_atom_r2s = sm100_utils.get_smem_store_op(
self.c_layout, self.c_dtype, self.acc_dtype, tiled_copy_t2r
)
tiled_copy_r2s = cute.make_tiled_copy_D(copy_atom_r2s, tiled_copy_t2r)
# (R2S, R2S_M, R2S_N, PIPE_D)
thr_copy_r2s = tiled_copy_r2s.get_slice(tidx)
tRS_sC = thr_copy_r2s.partition_D(sC)
# (R2S, R2S_M, R2S_N)
tRS_rC = tiled_copy_r2s.retile(tTR_rC)
return tiled_copy_r2s, tRS_rC, tRS_sC
def epilog_gmem_copy_and_partition(
self,
tidx: cutlass.Int32,
atom: Union[cute.CopyAtom, cute.TiledCopy],
gC_mnl: cute.Tensor,
epi_tile: cute.Tile,
sC: cute.Tensor,
) -> Tuple[cute.CopyAtom, cute.Tensor, cute.Tensor]:
# (EPI_TILE_M, EPI_TILE_N, EPI_M, EPI_N, RestM, RestN, RestL)
gC_epi = cute.flat_divide(
gC_mnl[((None, None), 0, 0, None, None, None)], epi_tile
)
tma_atom_c = atom
sC_for_tma_partition = cute.group_modes(sC, 0, 2)
gC_for_tma_partition = cute.group_modes(gC_epi, 0, 2)
# ((ATOM_V, REST_V), EPI_M, EPI_N)
# ((ATOM_V, REST_V), EPI_M, EPI_N, RestM, RestN, RestL)
bSG_sC, bSG_gC = cpasync.tma_partition(
tma_atom_c,
0,
cute.make_layout(1),
sC_for_tma_partition,
gC_for_tma_partition,
)
return tma_atom_c, bSG_sC, bSG_gC
def epilog_gmem_copy_and_partition_simt(
self,
tidx: cutlass.Int32,
tiled_copy_t2r: cute.TiledCopy,
gC_mnl: cute.Tensor,
epi_tile: cute.Tile,
) -> cute.Tensor:
# (EPI_TILE_M, EPI_TILE_N, EPI_M, EPI_N, RestM, RestN, RestL)
gC_epi = cute.flat_divide(gC_mnl, epi_tile)
thr_copy_t2r = tiled_copy_t2r.get_slice(tidx)
# (T2R, T2R_M, T2R_N, EPI_M, EPI_N, RestM, RestN, RestL)
tTR_gC = thr_copy_t2r.partition_D(gC_epi)
return tTR_gC
@staticmethod
def _transform_partitioned_tensor_layout(tensor: cute.Tensor) -> cute.Tensor:
layout = tensor.layout
shape = layout.shape
stride = layout.stride
new_shape = ((shape[0][0], shape[1]), (shape[0][1], shape[2]), *shape[3:])
new_stride = (
(stride[0][0], stride[1]),
(stride[0][1], stride[2]),
*stride[3:],
)
new_layout = cute.make_layout(shape=new_shape, stride=new_stride)
return cute.make_tensor(tensor.iterator, new_layout)
def epilog_tmem_copy_and_partition_simt(
self,
tidx: cutlass.Int32,
tAcc: cute.Tensor,
tCgC: cute.Tensor,
epi_tile: cute.Tile,
use_2cta_instrs: Union[cutlass.Boolean, bool],
) -> Tuple[cute.TiledCopy, cute.Tensor, cute.Tensor]:
copy_atom_t2r = sm100_utils.get_tmem_load_op(
self.cta_tile_shape_mnk,
self.c_layout,
self.c_dtype,
self.acc_dtype,
epi_tile,
use_2cta_instrs,
)
tAcc_epi = cute.flat_divide(tAcc, epi_tile)
tiled_copy_t2r = tcgen05.make_tmem_copy(
copy_atom_t2r, tAcc_epi[(None, None, 0, 0, 0)]
)
thr_copy_t2r = tiled_copy_t2r.get_slice(tidx)
tTR_tAcc = thr_copy_t2r.partition_S(tAcc_epi)
tCgC_epi = cute.flat_divide(tCgC, epi_tile)
tTR_gC = thr_copy_t2r.partition_D(tCgC_epi)
tTR_rAcc = cute.make_rmem_tensor(
tTR_gC[(None, None, None, 0, 0, 0, 0, 0)].shape, self.acc_dtype
)
return tiled_copy_t2r, tTR_tAcc, tTR_rAcc
@staticmethod
def _compute_stages(
tiled_mma: cute.TiledMma,
mma_tiler_mnk: Tuple[int, int, int],
a_dtype: Type[cutlass.Numeric],
b_dtype: Type[cutlass.Numeric],
epi_tile: cute.Tile,
c_dtype: Type[cutlass.Numeric],
c_layout: utils.LayoutEnum,
sf_dtype: Type[cutlass.Numeric],
sf_vec_size: int,
smem_capacity: int,
occupancy: int,
force_c_stage: int | None = None,
) -> Tuple[int, int, int]:
# ACC stages
num_acc_stage = 1 if mma_tiler_mnk[1] == 256 else 2
# Default C stages
num_c_stage = 2 if force_c_stage is None else max(1, int(force_c_stage))
# Calculate smem layout and size for one stage of A, B, SFA, SFB and C
a_smem_layout_stage_one = sm100_utils.make_smem_layout_a(
tiled_mma,
mma_tiler_mnk,
a_dtype,
1, # a tmp 1 stage is provided
)
b_smem_layout_staged_one = sm100_utils.make_smem_layout_b(
tiled_mma,
mma_tiler_mnk,
b_dtype,
1, # a tmp 1 stage is provided
)
sfa_smem_layout_staged_one = blockscaled_utils.make_smem_layout_sfa(
tiled_mma,
mma_tiler_mnk,
sf_vec_size,
1, # a tmp 1 stage is provided
)
sfb_smem_layout_staged_one = blockscaled_utils.make_smem_layout_sfb(
tiled_mma,
mma_tiler_mnk,
sf_vec_size,
1, # a tmp 1 stage is provided
)
c_smem_layout_staged_one = sm100_utils.make_smem_layout_epi(
c_dtype,
c_layout,
epi_tile,
1,
)
ab_bytes_per_stage = (
cute.size_in_bytes(a_dtype, a_smem_layout_stage_one)
+ cute.size_in_bytes(b_dtype, b_smem_layout_staged_one)
+ cute.size_in_bytes(sf_dtype, sfa_smem_layout_staged_one)
+ cute.size_in_bytes(sf_dtype, sfb_smem_layout_staged_one)
)
mbar_helpers_bytes = 1024
c_bytes_per_stage = cute.size_in_bytes(c_dtype, c_smem_layout_staged_one)
c_bytes = c_bytes_per_stage * num_c_stage
# Calculate A/B/SFA/SFB stages:
# Start with total smem per CTA (capacity / occupancy)
# Subtract reserved bytes and initial C stages bytes
# Divide remaining by bytes needed per A/B/SFA/SFB stage
num_ab_stage = (
smem_capacity // occupancy - (mbar_helpers_bytes + c_bytes)
) // ab_bytes_per_stage
if num_ab_stage < 1:
num_ab_stage = 1
# Refine epilogue stages:
# Calculate remaining smem after allocating for A/B/SFA/SFB stages and reserved bytes
# Add remaining unused smem to epilogue
if force_c_stage is None:
num_c_stage += (
smem_capacity
- occupancy * ab_bytes_per_stage * num_ab_stage
- occupancy * (mbar_helpers_bytes + c_bytes)
) // (occupancy * c_bytes_per_stage)
return num_acc_stage, num_ab_stage, num_c_stage
@staticmethod
def _compute_grid(
total_num_clusters: int,
cluster_shape_mn: tuple[int, int],
max_active_clusters: cutlass.Constexpr[int],
swizzle_size: int = 1,
raster_along_m: bool = True,
) -> tuple[utils.PersistentTileSchedulerParams, tuple[int, int, int]]:
# Create problem shape with M, N dimensions from cluster shape
# and L dimension representing the total number of clusters.
problem_shape_ntile_mnl = (
cluster_shape_mn[0],
cluster_shape_mn[1],
cutlass.Int32(total_num_clusters),
)
tile_sched_params = utils.PersistentTileSchedulerParams(
problem_shape_ntile_mnl,
(*cluster_shape_mn, 1),
swizzle_size=swizzle_size,
raster_along_m=raster_along_m,
)
grid = utils.StaticPersistentTileScheduler.get_grid_shape(
tile_sched_params, max_active_clusters
)
return tile_sched_params, grid
@staticmethod
def _get_mbar_smem_bytes(**kwargs_stages: int) -> int:
num_barriers_per_stage = 2
num_bytes_per_barrier = 8
mbar_smem_consumption = sum(
[
num_barriers_per_stage * num_bytes_per_barrier * stage
for stage in kwargs_stages.values()
]
)
return mbar_smem_consumption
@staticmethod
def is_valid_dtypes_and_scale_factor_vec_size(
ab_dtype: Type[cutlass.Numeric],
sf_dtype: Type[cutlass.Numeric],
sf_vec_size: int,
c_dtype: Type[cutlass.Numeric],
) -> bool:
is_valid = True
# Check valid ab_dtype
if ab_dtype not in {
cutlass.Float4E2M1FN,
cutlass.Float8E5M2,
cutlass.Float8E4M3FN,
}:
is_valid = False
# Check valid sf_vec_size
if sf_vec_size not in {16, 32}:
is_valid = False
# Check valid sf_dtype
if sf_dtype not in {cutlass.Float8E8M0FNU, cutlass.Float8E4M3FN}:
is_valid = False
# Check valid sf_dtype and sf_vec_size combinations
if sf_dtype == cutlass.Float8E4M3FN and sf_vec_size == 32:
is_valid = False
if ab_dtype in {cutlass.Float8E5M2, cutlass.Float8E4M3FN} and sf_vec_size == 16:
is_valid = False
# Check valid c_dtype
if c_dtype not in {
cutlass.Float32,
cutlass.Float16,
cutlass.BFloat16,
cutlass.Float8E5M2,
cutlass.Float8E4M3FN,
}:
is_valid = False
return is_valid
@staticmethod
def is_valid_layouts(
ab_dtype: Type[cutlass.Numeric],
c_dtype: Type[cutlass.Numeric],
a_major: str,
b_major: str,
c_major: str,
) -> bool:
is_valid = True
if ab_dtype is cutlass.Float4E2M1FN and not (a_major == "k" and b_major == "k"):
is_valid = False
return is_valid
@staticmethod
def is_valid_mma_tiler_and_cluster_shape(
mma_tiler_mn: Tuple[int, int],
cluster_shape_mn: Tuple[int, int],
) -> bool:
is_valid = True
# Skip invalid mma tile shape
if mma_tiler_mn[0] not in [128, 256]:
is_valid = False
if mma_tiler_mn[1] not in [128, 256]:
is_valid = False
# Skip illegal cluster shape
if cluster_shape_mn[0] % (2 if mma_tiler_mn[0] == 256 else 1) != 0:
is_valid = False
# Skip invalid cluster shape
is_power_of_2 = lambda x: x > 0 and (x & (x - 1)) == 0
if (
cluster_shape_mn[0] * cluster_shape_mn[1] > 16
or cluster_shape_mn[0] <= 0
or cluster_shape_mn[1] <= 0
# Special cluster shape check for scale factor multicasts.
# Due to limited size of scale factors, we can't multicast among more than 4 CTAs.
or cluster_shape_mn[0] > 4
or cluster_shape_mn[1] > 4
or not is_power_of_2(cluster_shape_mn[0])
or not is_power_of_2(cluster_shape_mn[1])
):
is_valid = False
return is_valid
@staticmethod
def is_valid_tensor_alignment(
problem_sizes_mnkl: List[Tuple[int, int, int, int]],
ab_dtype: Type[cutlass.Numeric],
c_dtype: Type[cutlass.Numeric],
a_major: str,
b_major: str,
c_major: str,
) -> bool:
is_valid = True
def check_contigous_16B_alignment(dtype, is_mode0_major, tensor_shape):
major_mode_idx = 0 if is_mode0_major else 1
num_major_elements = tensor_shape[major_mode_idx]
num_contiguous_elements = 16 * 8 // dtype.width
return num_major_elements % num_contiguous_elements == 0
for m, n, k, l in problem_sizes_mnkl:
if (
not check_contigous_16B_alignment(ab_dtype, a_major == "m", (m, k, l))
or not check_contigous_16B_alignment(
ab_dtype, b_major == "n", (n, k, l)
)
or not check_contigous_16B_alignment(c_dtype, c_major == "m", (m, n, l))
):
is_valid = False
return is_valid
@staticmethod
def can_implement(
ab_dtype: Type[cutlass.Numeric],
sf_dtype: Type[cutlass.Numeric],
sf_vec_size: int,
c_dtype: Type[cutlass.Numeric],
mma_tiler_mn: Tuple[int, int],
cluster_shape_mn: Tuple[int, int],
problem_sizes_mnkl: List[Tuple[int, int, int, int]],
a_major: str,
b_major: str,
c_major: str,
) -> bool:
can_implement = True
# Skip unsupported types
if not Sm100GroupedBlockScaledGemmKernel.is_valid_dtypes_and_scale_factor_vec_size(
ab_dtype, sf_dtype, sf_vec_size, c_dtype
):
can_implement = False
# Skip unsupported layouts
if not Sm100GroupedBlockScaledGemmKernel.is_valid_layouts(
ab_dtype, c_dtype, a_major, b_major, c_major
):
can_implement = False
# Skip invalid mma tile shape and cluster shape
if not Sm100GroupedBlockScaledGemmKernel.is_valid_mma_tiler_and_cluster_shape(
mma_tiler_mn, cluster_shape_mn
):
can_implement = False
# Skip illegal problem shape for load/store alignment
if not Sm100GroupedBlockScaledGemmKernel.is_valid_tensor_alignment(
problem_sizes_mnkl, ab_dtype, c_dtype, a_major, b_major, c_major
):
can_implement = False
return can_implement
# Size of smem we reserved for mbarrier, tensor memory management and tensormap update
reserved_smem_bytes = 1024
bytes_per_tensormap = 128
num_tensormaps = 5
# size of smem used for tensor memory management
tensor_memory_management_bytes = 12
# Create tensor and return the pointer, tensor, and stride
def _select_initial_indices(
problem_sizes: list[tuple[int, int, int, int]],
) -> tuple[int, int, int]:
key_size_a = lambda item: item[1][0] * item[1][2]
key_size_b = lambda item: item[1][1] * item[1][2]
key_size_c = lambda item: item[1][0] * item[1][1]
min_a_idx, _ = min(enumerate(problem_sizes), key=key_size_a)
min_b_idx, _ = min(enumerate(problem_sizes), key=key_size_b)
min_c_idx, _ = min(enumerate(problem_sizes), key=key_size_c)
return min_a_idx, min_b_idx, min_c_idx
def _from_dlpack_typed(
torch_tensor: torch.Tensor,
cutlass_dtype: Type[cutlass.Numeric],
assumed_align: int,
) -> cute.Tensor:
tensor_for_dlpack = torch_tensor
if getattr(cutlass_dtype, "width", 0) <= 8:
if not torch_tensor.is_contiguous():
tensor_for_dlpack = torch_tensor.contiguous()
tensor_for_dlpack = tensor_for_dlpack.view(torch.uint8)
use_32bit_stride = True
try:
if tensor_for_dlpack.numel() > (2**31 - 1):
use_32bit_stride = False
else:
max_stride = max(tensor_for_dlpack.stride())
if max_stride > (2**31 - 1):
use_32bit_stride = False
except Exception:
use_32bit_stride = False
cute_tensor = from_dlpack(
tensor_for_dlpack,
assumed_align=assumed_align,
use_32bit_stride=use_32bit_stride,
)
cute_tensor.element_type = cutlass_dtype
return cute_tensor
def _get_handle() -> Any:
global _fake_st
if _fake_st is None:
_mk = getattr(cutlass_torch, "default_" + "st" + "ream")
_fake_st = _mk()
return _fake_st
def _normalize_sfs(
sfs: list[tuple[torch.Tensor, torch.Tensor]]
) -> list[tuple[torch.Tensor, torch.Tensor]]:
if not hasattr(torch, "float8_e4m3fn"):
return sfs
target = torch.float8_e4m3fn
out: list[tuple[torch.Tensor, torch.Tensor]] = []
for sfa, sfb in sfs:
if sfa.dtype != target:
sfa = sfa.to(target)
if sfb.dtype != target:
sfb = sfb.to(target)
out.append((sfa, sfb))
return out
def _swap_inputs(
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
) -> tuple[
list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
list[tuple[torch.Tensor, torch.Tensor]],
list[tuple[int, int, int, int]],
]:
abc_swapped: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]] = []
sfs_swapped: list[tuple[torch.Tensor, torch.Tensor]] = []
sizes_swapped: list[tuple[int, int, int, int]] = []
for (a_ref, b_ref, c_ref), (sfa_ref, sfb_ref), (m, n, k, l) in zip(
abc_tensors, sfs, problem_sizes
):
# Swap A/B to compute D = B @ A^T and write D in column-major so C = D^T.
c_swapped = c_ref.permute(1, 0, 2)
abc_swapped.append((b_ref, a_ref, c_swapped))
sfs_swapped.append((sfb_ref, sfa_ref))
sizes_swapped.append((n, m, k, l))
return abc_swapped, sfs_swapped, sizes_swapped
def _make_initial_abc_tensor(
torch_tensor: torch.Tensor,
cutlass_dtype: Type[cutlass.Numeric],
is_mode0_major: bool,
assumed_align: int,
c_divisibility: int | None = None,
) -> cute.Tensor:
cute_tensor = _from_dlpack_typed(torch_tensor, cutlass_dtype, assumed_align)
leading_dim = 0 if is_mode0_major else 1
cute_tensor = cute_tensor.mark_layout_dynamic(leading_dim=leading_dim)
stride_order = (2, 1, 0) if is_mode0_major else (2, 0, 1)
if cutlass_dtype == cutlass.Float4E2M1FN:
divisibility = 32
elif getattr(cutlass_dtype, "width", 0) == 16:
divisibility = 8 if c_divisibility is None else c_divisibility
else:
divisibility = 16
cute_tensor.mark_compact_shape_dynamic(
mode=leading_dim,
stride_order=stride_order,
divisibility=divisibility,
)
return cute_tensor
def _compute_cluster_tile(mma_tiler_mn: tuple[int, int], cluster_shape_mn: tuple[int, int]) -> tuple[int, int]:
cta_tile = (128, mma_tiler_mn[1])
return (cta_tile[0] * cluster_shape_mn[0], cta_tile[1] * cluster_shape_mn[1])
def _total_clusters(problem_sizes: list[tuple[int, int, int, int]], cluster_tile: tuple[int, int]) -> int:
total = 0
tm, tn = cluster_tile
for m, n, _, _ in problem_sizes:
cm = (m + tm - 1) // tm
cn = (n + tn - 1) // tn
total += cm * cn
return total
def _build_entry(cfg_key: str,
mma_tiler_mn: tuple[int, int],
cluster_shape_mn: tuple[int, int],
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
use_tma_store: bool,
c_col_major: bool,
max_ab_stage: int | None,
prefetch_dist: int | None,
force_c_stage: int | None,
c_assumed_align: int,
c_divisibility: int,
swizzle_size: int,
raster_along_m: bool):
dev = abc_tensors[0][0].device
group_count = len(problem_sizes)
hinfo = utils.HardwareInfo()
max_active = hinfo.get_max_active_clusters(cluster_shape_mn[0] * cluster_shape_mn[1])
sm_count = hinfo.get_max_active_clusters(1)
num_tmaps = Sm100GroupedBlockScaledGemmKernel.num_tensormaps
bytes_tmap = Sm100GroupedBlockScaledGemmKernel.bytes_per_tensormap // 8
tmap_shape = (sm_count, num_tmaps, bytes_tmap)
tmap_t = torch.empty(tmap_shape, dtype=torch.int64, device=dev)
tmap_cute = from_dlpack(tmap_t)
tmap_cute.element_type = cutlass.Int64
dims_cpu = torch.tensor(problem_sizes, dtype=torch.int32, pin_memory=True)
dims_gpu = dims_cpu.to(device=dev)
dims_cute = from_dlpack(dims_gpu)
dims_cute.element_type = cutlass.Int32
strides_cpu = torch.empty((group_count, 3, 2), dtype=torch.int32, pin_memory=True)
ptrs_cpu = torch.empty((group_count, 3), dtype=torch.int64, pin_memory=True)
ptrs_sf_cpu = torch.empty((group_count, 2), dtype=torch.int64, pin_memory=True)
strides_gpu = strides_cpu.to(device=dev)
ptrs_gpu = ptrs_cpu.to(device=dev)
ptrs_sf_gpu = ptrs_sf_cpu.to(device=dev)
strides_cute = from_dlpack(strides_gpu)
strides_cute.element_type = cutlass.Int32
ptrs_cute = from_dlpack(ptrs_gpu)
ptrs_cute.element_type = cutlass.Int64
ptrs_sf_cute = from_dlpack(ptrs_sf_gpu)
ptrs_sf_cute.element_type = cutlass.Int64
def _fill_stride():
for i, (m, n, k, _) in enumerate(problem_sizes):
stride_a = (k, 1)
stride_b = (k, 1)
if c_col_major:
stride_c = (1, m)
else:
stride_c = (n, 1)
strides_cpu[i, 0, 0] = stride_a[0]
strides_cpu[i, 0, 1] = stride_a[1]
strides_cpu[i, 1, 0] = stride_b[0]
strides_cpu[i, 1, 1] = stride_b[1]
strides_cpu[i, 2, 0] = stride_c[0]
strides_cpu[i, 2, 1] = stride_c[1]
strides_gpu.copy_(strides_cpu, non_blocking=True)
_fill_stride()
cluster_tile = _compute_cluster_tile(mma_tiler_mn, cluster_shape_mn)
total_num_clusters = _total_clusters(problem_sizes, cluster_tile)
min_a_idx, min_b_idx, min_c_idx = _select_initial_indices(problem_sizes)
a_ref = abc_tensors[min_a_idx][0]
b_ref = abc_tensors[min_b_idx][1]
c_ref = abc_tensors[min_c_idx][2]
sfa_ref = sfs[min_a_idx][0]
sfb_ref = sfs[min_b_idx][1]
initial_a = _make_initial_abc_tensor(
a_ref, cutlass.Float4E2M1FN, is_mode0_major=False, assumed_align=16
)
initial_b = _make_initial_abc_tensor(
b_ref, cutlass.Float4E2M1FN, is_mode0_major=False, assumed_align=16
)
initial_c = _make_initial_abc_tensor(
c_ref,
cutlass.Float16,
is_mode0_major=c_col_major,
assumed_align=c_assumed_align,
c_divisibility=c_divisibility,
)
initial_sfa = _from_dlpack_typed(
sfa_ref, cutlass.Float8E4M3FN, assumed_align=16
)
initial_sfb = _from_dlpack_typed(
sfb_ref, cutlass.Float8E4M3FN, assumed_align=16
)
gemm = Sm100GroupedBlockScaledGemmKernel(
sf_vec_size=16,
mma_tiler_mn=mma_tiler_mn,
cluster_shape_mn=cluster_shape_mn,
use_tma_store=use_tma_store,
max_ab_stage=max_ab_stage,
prefetch_dist=prefetch_dist,
force_c_stage=force_c_stage,
c_assumed_align=c_assumed_align,
c_divisibility=c_divisibility,
swizzle_size=swizzle_size,
raster_along_m=raster_along_m,
)
gemm.c_tile_stride = (int(c_ref.stride(0)), int(c_ref.stride(1)))
try:
compiled = cute.compile(
gemm,
initial_a,
initial_b,
initial_c,
initial_sfa,
initial_sfb,
group_count,
dims_cute,
strides_cute,
ptrs_cute,
ptrs_sf_cute,
total_num_clusters,
tmap_cute,
max_active,
_get_handle(),
options="--opt-level 2",
)
except Exception as exc:
print(f"cute.compile failed: {exc}", file=sys.stderr, flush=True)
raise RuntimeError(str(exc)) from exc
entry = {
"cfg_key": cfg_key,
"mma_tiler_mn": mma_tiler_mn,
"cluster_shape_mn": cluster_shape_mn,
"group_count": group_count,
"problem_sizes": problem_sizes,
"dims_gpu": dims_gpu,
"dims_cute": dims_cute,
"strides_gpu": strides_gpu,
"strides_cute": strides_cute,
"ptrs_cpu": ptrs_cpu,
"ptrs_gpu": ptrs_gpu,
"ptrs_cute": ptrs_cute,
"ptrs_sf_cpu": ptrs_sf_cpu,
"ptrs_sf_gpu": ptrs_sf_gpu,
"ptrs_sf_cute": ptrs_sf_cute,
"tmap": tmap_cute,
"initial_a": initial_a,
"initial_b": initial_b,
"initial_c": initial_c,
"initial_sfa": initial_sfa,
"initial_sfb": initial_sfb,
"initial_refs": (a_ref, b_ref, c_ref, sfa_ref, sfb_ref),
"compiled": compiled,
"handle": _get_handle(),
}
return entry
def _update_ptrs(entry: dict,
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]]):
ptrs_cpu = entry["ptrs_cpu"]
ptrs_sf_cpu = entry["ptrs_sf_cpu"]
for i, ((a_ref, b_ref, c_ref), (sfa_ref, sfb_ref)) in enumerate(zip(abc_tensors, sfs)):
ptrs_cpu[i, 0] = a_ref.data_ptr()
ptrs_cpu[i, 1] = b_ref.data_ptr()
ptrs_cpu[i, 2] = c_ref.data_ptr()
ptrs_sf_cpu[i, 0] = sfa_ref.data_ptr()
ptrs_sf_cpu[i, 1] = sfb_ref.data_ptr()
entry["ptrs_gpu"].copy_(ptrs_cpu, non_blocking=True)
entry["ptrs_sf_gpu"].copy_(ptrs_sf_cpu, non_blocking=True)
def _run_group(cfg_key: str,
mma_tiler_mn: tuple[int, int],
cluster_shape_mn: tuple[int, int],
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
use_tma_store: bool,
c_col_major: bool,
max_ab_stage: int | None,
prefetch_dist: int | None,
force_c_stage: int | None,
c_assumed_align: int,
c_divisibility: int,
swizzle_size: int,
raster_along_m: bool):
if not abc_tensors:
return
dev = abc_tensors[0][0].device
cache_key = (
dev.index if dev.type == "cuda" else -1,
cfg_key,
use_tma_store,
c_col_major,
max_ab_stage,
prefetch_dist,
force_c_stage,
c_assumed_align,
c_divisibility,
swizzle_size,
raster_along_m,
tuple(problem_sizes),
)
entry = _kernel_cache.get(cache_key)
if entry is None:
entry = _build_entry(
cfg_key,
mma_tiler_mn,
cluster_shape_mn,
abc_tensors,
sfs,
problem_sizes,
use_tma_store,
c_col_major,
max_ab_stage,
prefetch_dist,
force_c_stage,
c_assumed_align,
c_divisibility,
swizzle_size,
raster_along_m,
)
_kernel_cache[cache_key] = entry
_update_ptrs(entry, abc_tensors, sfs)
try:
entry["compiled"](
entry["initial_a"],
entry["initial_b"],
entry["initial_c"],
entry["initial_sfa"],
entry["initial_sfb"],
entry["dims_cute"],
entry["strides_cute"],
entry["ptrs_cute"],
entry["ptrs_sf_cute"],
entry["tmap"],
entry["handle"],
)
except Exception as exc:
print(f"compiled invocation failed: {exc}", file=sys.stderr, flush=True)
raise RuntimeError(str(exc)) from exc
def _split_groups(problem_sizes: list[tuple[int, int, int, int]]) -> tuple[list[int], list[int]]:
idx_1 = []
idx_2 = []
for i, (m, _n, k, _l) in enumerate(problem_sizes):
if m >= 256 and k >= 4096:
idx_2.append(i)
else:
idx_1.append(i)
return idx_1, idx_2
def _split_groups_swap(problem_sizes: list[tuple[int, int, int, int]]) -> tuple[list[int], list[int]]:
idx_1 = []
idx_2 = []
for i, (m, n, k, _l) in enumerate(problem_sizes):
# m is swapped M (original N), n is swapped N (original M)
if m >= 256 and k >= 4096:
idx_2.append(i)
else:
idx_1.append(i)
return idx_1, idx_2
def _split_groups_swap_by_idx(
problem_sizes: list[tuple[int, int, int, int]],
indices: list[int],
) -> tuple[list[int], list[int]]:
idx_1: list[int] = []
idx_2: list[int] = []
for i in indices:
m, _n, k, _l = problem_sizes[i]
if m >= 256 and k >= 4096:
idx_2.append(i)
else:
idx_1.append(i)
return idx_1, idx_2
def _swizzle_params(is_wide: bool) -> tuple[int, bool]:
if is_wide:
size = int(os.environ.get("NVFP4_SWIZZLE_WIDE", "1"))
raster_along_m = os.environ.get("NVFP4_SWIZZLE_WIDE_RASTER_M", "0") != "0"
else:
size = int(os.environ.get("NVFP4_SWIZZLE_NARROW", "1"))
raster_along_m = os.environ.get("NVFP4_SWIZZLE_NARROW_RASTER_M", "1") != "0"
if size not in (1, 2, 4, 8):
size = 1
return size, raster_along_m
def _collect_by_idx(
abc_tensors: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs: list[tuple[torch.Tensor, torch.Tensor]],
problem_sizes: list[tuple[int, int, int, int]],
indices: list[int],
) -> tuple[
list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
list[tuple[torch.Tensor, torch.Tensor]],
list[tuple[int, int, int, int]],
]:
return (
[abc_tensors[i] for i in indices],
[sfs[i] for i in indices],
[problem_sizes[i] for i in indices],
)
def _can_use_simt(problem_sizes: list[tuple[int, int, int, int]],
idx_1: list[int],
idx_2: list[int]) -> bool:
for i in idx_1:
m, n, _k, _l = problem_sizes[i]
if (m % 128) != 0 or (n % 256) != 0:
return False
for i in idx_2:
m, n, _k, _l = problem_sizes[i]
if (m % 256) != 0 or (n % 256) != 0:
return False
return True
def custom_kernel(data: input_t) -> output_t:
abc_tensors, _sfs_raw, sfs_reordered, problem_sizes = data
sfs_reordered = _normalize_sfs(sfs_reordered)
idx_unswapped: list[int] = []
idx_rest: list[int] = []
for i, (_m, _n, k, _l) in enumerate(problem_sizes):
if k == 2048:
idx_unswapped.append(i)
else:
idx_rest.append(i)
abc_unswapped, sfs_unswapped, sizes_unswapped = _collect_by_idx(
abc_tensors, sfs_reordered, problem_sizes, idx_unswapped
)
abc_rest, sfs_rest, sizes_rest = _collect_by_idx(
abc_tensors, sfs_reordered, problem_sizes, idx_rest
)
abc_swapped, sfs_swapped, sizes_swapped = _swap_inputs(
abc_rest, sfs_rest, sizes_rest
)
if _enable_stage8 and abc_swapped and _is_sm100_device(abc_swapped[0][0]):
stage8_out = _try_stage8_ext(abc_swapped, sfs_swapped, sizes_swapped)
if stage8_out is not None:
return [c for (_a, _b, c) in abc_tensors]
c_col_major = True
is_sm100 = bool(abc_swapped) and _is_sm100_device(abc_swapped[0][0])
c_assumed_align = 32
c_divisibility = 16
prefetch_dist = None if is_sm100 else 0
use_tma_store = True
idx_wide: list[int] = []
idx_narrow: list[int] = []
for i, (_m, _n, k, _l) in enumerate(sizes_swapped):
if k >= 6000:
idx_wide.append(i)
else:
idx_narrow.append(i)
idx_wide_1, idx_wide_2 = _split_groups_swap_by_idx(sizes_swapped, idx_wide)
idx_narrow_1, idx_narrow_2 = _split_groups_swap_by_idx(sizes_swapped, idx_narrow)
abc_wide_1, sfs_wide_1, sizes_wide_1 = _collect_by_idx(
abc_swapped, sfs_swapped, sizes_swapped, idx_wide_1
)
abc_wide_2, sfs_wide_2, sizes_wide_2 = _collect_by_idx(
abc_swapped, sfs_swapped, sizes_swapped, idx_wide_2
)
abc_narrow_1, sfs_narrow_1, sizes_narrow_1 = _collect_by_idx(
abc_swapped, sfs_swapped, sizes_swapped, idx_narrow_1
)
abc_narrow_2, sfs_narrow_2, sizes_narrow_2 = _collect_by_idx(
abc_swapped, sfs_swapped, sizes_swapped, idx_narrow_2
)
enable_simt = os.environ.get("NVFP4_ENABLE_SIMT", "1") == "1"
swizzle_wide, raster_wide = _swizzle_params(True)
swizzle_narrow, raster_narrow = _swizzle_params(False)
force_c_stage_2sm_env = os.environ.get("NVFP4_FORCE_C_STAGE_2SM", "")
force_c_stage_2sm = int(force_c_stage_2sm_env) if force_c_stage_2sm_env else None
if force_c_stage_2sm is not None and force_c_stage_2sm <= 0:
force_c_stage_2sm = None
max_ab_stage_2sm_env = os.environ.get("NVFP4_MAX_AB_STAGE_2SM", "")
max_ab_stage_2sm = int(max_ab_stage_2sm_env) if max_ab_stage_2sm_env else None
if max_ab_stage_2sm is not None and max_ab_stage_2sm <= 0:
max_ab_stage_2sm = None
_run_group(
"unswapped",
(128, 256),
(2, 1),
abc_unswapped,
sfs_unswapped,
sizes_unswapped,
use_tma_store,
False,
None,
prefetch_dist,
None,
c_assumed_align,
c_divisibility,
swizzle_narrow,
raster_narrow,
)
if enable_simt:
def is_simt_safe(sz: tuple[int, int, int, int]) -> bool:
_m, n, k, _l = sz
# SIMT vector store uses 128-bit vectors; require N divisible by 8.
# Limit SIMT to k < 4096 (1SM path) to avoid 2CTA edge cases.
return (n % 8) == 0 and k < 4096
def gather(idx_list: list[int], want_simt: bool):
abc_out = []
sfs_out = []
sizes_out = []
for i in idx_list:
if is_simt_safe(sizes_swapped[i]) == want_simt:
abc_out.append(abc_swapped[i])
sfs_out.append(sfs_swapped[i])
sizes_out.append(sizes_swapped[i])
return abc_out, sfs_out, sizes_out
abc_wide_1_simt, sfs_wide_1_simt, sizes_wide_1_simt = gather(
idx_wide_1, True
)
abc_wide_1_tma, sfs_wide_1_tma, sizes_wide_1_tma = gather(
idx_wide_1, False
)
abc_wide_2_simt, sfs_wide_2_simt, sizes_wide_2_simt = gather(
idx_wide_2, True
)
abc_wide_2_tma, sfs_wide_2_tma, sizes_wide_2_tma = gather(
idx_wide_2, False
)
abc_narrow_1_simt, sfs_narrow_1_simt, sizes_narrow_1_simt = gather(
idx_narrow_1, True
)
abc_narrow_1_tma, sfs_narrow_1_tma, sizes_narrow_1_tma = gather(
idx_narrow_1, False
)
abc_narrow_2_simt, sfs_narrow_2_simt, sizes_narrow_2_simt = gather(
idx_narrow_2, True
)
abc_narrow_2_tma, sfs_narrow_2_tma, sizes_narrow_2_tma = gather(
idx_narrow_2, False
)
def run_pair(
cfg_key_base: str,
mma_tiler_mn: tuple[int, int],
cluster_shape_mn: tuple[int, int],
abc_simt: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs_simt: list[tuple[torch.Tensor, torch.Tensor]],
sizes_simt: list[tuple[int, int, int, int]],
abc_tma: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],
sfs_tma: list[tuple[torch.Tensor, torch.Tensor]],
sizes_tma: list[tuple[int, int, int, int]],
swizzle_size: int,
raster_along_m: bool,
max_ab_stage: int | None = None,
force_c_stage: int | None = None,
):
_run_group(
f"{cfg_key_base}s",
mma_tiler_mn,
cluster_shape_mn,
abc_simt,
sfs_simt,
sizes_simt,
False,
c_col_major,
max_ab_stage,
prefetch_dist,
force_c_stage,
c_assumed_align,
c_divisibility,
swizzle_size,
raster_along_m,
)
_run_group(
f"{cfg_key_base}t",
mma_tiler_mn,
cluster_shape_mn,
abc_tma,
sfs_tma,
sizes_tma,
True,
c_col_major,
max_ab_stage,
prefetch_dist,
force_c_stage,
c_assumed_align,
c_divisibility,
swizzle_size,
raster_along_m,
)
run_pair(
"w1",
(128, 256),
(2, 1),
abc_wide_1_simt,
sfs_wide_1_simt,
sizes_wide_1_simt,
abc_wide_1_tma,
sfs_wide_1_tma,
sizes_wide_1_tma,
swizzle_wide,
raster_wide,
)
run_pair(
"w2",
(256, 256),
(4, 1),
abc_wide_2_simt,
sfs_wide_2_simt,
sizes_wide_2_simt,
abc_wide_2_tma,
sfs_wide_2_tma,
sizes_wide_2_tma,
swizzle_wide,
raster_wide,
max_ab_stage_2sm,
force_c_stage_2sm,
)
run_pair(
"n1",
(128, 128),
(2, 1),
abc_narrow_1_simt,
sfs_narrow_1_simt,
sizes_narrow_1_simt,
abc_narrow_1_tma,
sfs_narrow_1_tma,
sizes_narrow_1_tma,
swizzle_narrow,
raster_narrow,
)
run_pair(
"n2",
(256, 128),
(4, 1),
abc_narrow_2_simt,
sfs_narrow_2_simt,
sizes_narrow_2_simt,
abc_narrow_2_tma,
sfs_narrow_2_tma,
sizes_narrow_2_tma,
swizzle_narrow,
raster_narrow,
max_ab_stage_2sm,
force_c_stage_2sm,
)
else:
_run_group(
"swap_1sm_wide",
(128, 256),
(2, 1),
abc_wide_1,
sfs_wide_1,
sizes_wide_1,
use_tma_store,
c_col_major,
None,
prefetch_dist,
None,
c_assumed_align,
c_divisibility,
swizzle_wide,
raster_wide,
)
_run_group(
"swap_2sm_wide",
(256, 256),
(4, 1),
abc_wide_2,
sfs_wide_2,
sizes_wide_2,
use_tma_store,
c_col_major,
max_ab_stage_2sm,
prefetch_dist,
force_c_stage_2sm,
c_assumed_align,
c_divisibility,
swizzle_wide,
raster_wide,
)
_run_group(
"swap_1sm",
(128, 128),
(2, 1),
abc_narrow_1,
sfs_narrow_1,
sizes_narrow_1,
use_tma_store,
c_col_major,
None,
prefetch_dist,
None,
c_assumed_align,
c_divisibility,
swizzle_narrow,
raster_narrow,
)
_run_group(
"swap_2sm",
(256, 128),
(4, 1),
abc_narrow_2,
sfs_narrow_2,
sizes_narrow_2,
use_tma_store,
c_col_major,
max_ab_stage_2sm,
prefetch_dist,
force_c_stage_2sm,
c_assumed_align,
c_divisibility,
swizzle_narrow,
raster_narrow,
)
return [c for (_a, _b, c) in abc_tensors]
scrolls · 3299 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 446248.
import osimport sysimport importlib.util- import platformfrom inspect import isclassfrom typing import Any, List, Tuple, Type, Union⋯ 27 unchanged lines_CACHE_DIR = os.path.join("/tmp", "cute_cache")os.makedirs(_CACHE_DIR, exist_ok=True)os.environ.setdefault("CUTE_DSL_CACHE_DIR", _CACHE_DIR)+ os.environ.setdefault("CUTE_DSL_JIT_CACHE", _CACHE_DIR)+ os.environ.setdefault("CUTE_DSL_KEEP_PTX", "1")+ os.environ.setdefault("CUTE_DSL_KEEP_CUBIN", "1")import cutlassimport cutlass.cute as cute⋯ 16 unchanged lines_stage8_mod: Any | None = None_stage8_checked = False_enable_stage8 = os.environ.get("NVFP4_ENABLE_STAGE8", "") == "1"- def _default_tmap_prefetch() -> bool:- host = os.environ.get("HOSTNAME") or platform.node() or ""- return host == "modal"- _tmap_env = os.environ.get("NVFP4_TMAP_PREFETCH")- if _tmap_env is None:- _enable_tmap_prefetch = _default_tmap_prefetch()- else:- _enable_tmap_prefetch = _tmap_env == "1"- _enable_tma_desc_prefetch = os.environ.get("NVFP4_TMA_DESC_PREFETCH", "1") == "1"def _is_sm100_device(t: torch.Tensor) -> bool:if t.device.type != "cuda":return False⋯ 35 unchanged linesreturn None- def _env_int(name: str, default: int) -> int:- val = os.environ.get(name, "")- if not val:- return default- try:- return int(val)- except Exception:- return default- def _tvm_ffi_enabled() -> bool:- if os.environ.get("NVFP4_ENABLE_TVM_FFI", "") != "1" and os.environ.get(- "CUTE_DSL_ENABLE_TVM_FFI", ""- ) != "1":- return False- if importlib.util.find_spec("tvm_ffi") is not None:- return True- if importlib.util.find_spec("tvm") is not None:- return True- return False--- def _r2g_store_config(bits: int) -> tuple[int, int]:- if bits >= 256:- return 32, 16- if bits >= 128:- return 16, 8- if bits >= 64:- return 8, 4- return 16, 8---@dsl_user_opdef _normalize_ptr(ptr, *, loc=None, ip=None):if isinstance(ptr, _ir.Value):⋯ 61 unchanged lines@dsl_user_op- def _ptx_prefetch_tensormap(ptr, *, loc=None, ip=None):- ptr = _normalize_ptr(ptr, loc=loc, ip=ip)- if not isinstance(ptr, _ir.Value):- return None- _llvm.inline_asm(- None,- [ptr],- "prefetch.param.tensormap [$0];",- "l",- has_side_effects=True,- is_align_stack=False,- asm_dialect=_llvm.AsmDialect.AD_ATT,- loc=loc,- ip=ip,- )- return None--- @dsl_user_opdef _prefetch_tma(atom: cute.CopyAtom,src: cute.Tensor,⋯ 5 unchanged linesif hasattr(src, "iterator"):_ptx_prefetch_global(src.iterator, loc=loc, ip=ip)_ptx_prefetch_global_l1(src.iterator, loc=loc, ip=ip)- _ptx_prefetchu_l1(src.iterator, loc=loc, ip=ip)if tma_desc_ptr is not None:_ptx_prefetch_global(tma_desc_ptr, loc=loc, ip=ip)_ptx_prefetch_global_l1(tma_desc_ptr, loc=loc, ip=ip)_ptx_prefetchu_l1(tma_desc_ptr, loc=loc, ip=ip)- if _enable_tmap_prefetch:- _ptx_prefetch_tensormap(tma_desc_ptr, loc=loc, ip=ip)dummy_tma_bar_ptr = cute.make_ptr(cutlass.Int64, 0, cute.AddressSpace.smem, loc=loc, ip=ip)⋯ 16 unchanged linesmma_tiler_mn: Tuple[int, int],cluster_shape_mn: Tuple[int, int],use_tma_store: bool = True,- r2g_bits: int | None = None,- full_tiles_only: bool = False,max_ab_stage: int | None = None,prefetch_dist: int | None = None,+ force_c_stage: int | None = None,c_assumed_align: int = 16,c_divisibility: int = 8,+ swizzle_size: int = 1,+ raster_along_m: bool = True,):self.acc_dtype = cutlass.Float32self.sf_vec_size = sf_vec_sizeself.use_2cta_instrs = mma_tiler_mn[0] == 256self.cluster_shape_mn = cluster_shape_mnself.use_tma_store = use_tma_store- self.r2g_bits = 0 if r2g_bits is None else r2g_bits- self.full_tiles_only = full_tiles_onlyself.max_ab_stage = max_ab_stageself.prefetch_dist_override = prefetch_dist+ self.force_c_stage = force_c_stageself.c_assumed_align = c_assumed_alignself.c_divisibility = c_divisibility+ self.swizzle_size = swizzle_size+ self.raster_along_m = raster_along_m# K dimension is deferred in _setup_attributesself.mma_tiler = (*mma_tiler_mn, 1)⋯ 130 unchanged linesself.sf_vec_size,self.smem_capacity,self.occupancy,- self.use_tma_store,+ self.force_c_stage,)max_ab = self.max_ab_stageif max_ab is None:⋯ 31 unchanged linesself.sf_vec_size,self.num_ab_stage,)- c_stage_for_layout = self.num_c_stage if self.use_tma_store else 1self.c_smem_layout_staged = sm100_utils.make_smem_layout_epi(self.c_dtype,self.c_layout,self.epi_tile,- c_stage_for_layout,+ self.num_c_stage,)mbar_smem_bytes = self._get_mbar_smem_bytes(⋯ 164 unchanged lines# Compute grid sizeself.tile_sched_params, grid = self._compute_grid(- total_num_clusters, self.cluster_shape_mn, max_active_clusters+ total_num_clusters,+ self.cluster_shape_mn,+ max_active_clusters,+ swizzle_size=self.swizzle_size,+ raster_along_m=self.raster_along_m,)self.buffer_align_bytes = 1024⋯ 4 unchanged lines)# Define shared storage for kernel- c_smem_elems = (- cute.cosize(self.c_smem_layout_staged.outer) if self.use_tma_store else 1- )@cute.structclass SharedStorage:tensormap_buffer: cute.struct.MemRange[⋯ 9 unchanged linessC: cute.struct.Align[cute.struct.MemRange[self.c_dtype,- c_smem_elems,+ cute.cosize(self.c_smem_layout_staged.outer),],self.buffer_align_bytes,]⋯ 101 unchanged lines):warp_idx = cute.arch.warp_idx()warp_idx = cute.arch.make_warp_uniform(warp_idx)- if warp_idx == self.tma_warp_id and _enable_tma_desc_prefetch:+ if warp_idx == self.tma_warp_id:cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_a)cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_b)cute.nvgpu.cpasync.prefetch_descriptor(tma_atom_sfa)⋯ 2 unchanged lines# PTX prefetch base pointers to warm L2/L1 for first tile._ptx_prefetch_global(mA_mkl.iterator)_ptx_prefetch_global_l1(mA_mkl.iterator)- _ptx_prefetchu_l1(mA_mkl.iterator)_ptx_prefetch_global(mB_nkl.iterator)_ptx_prefetch_global_l1(mB_nkl.iterator)- _ptx_prefetchu_l1(mB_nkl.iterator)_ptx_prefetch_global(mSFA_mkl.iterator)_ptx_prefetch_global_l1(mSFA_mkl.iterator)- _ptx_prefetchu_l1(mSFA_mkl.iterator)_ptx_prefetch_global(mSFB_nkl.iterator)_ptx_prefetch_global_l1(mSFB_nkl.iterator)- _ptx_prefetchu_l1(mSFB_nkl.iterator)_ptx_prefetch_global(mC_mnl.iterator)_ptx_prefetch_global_l1(mC_mnl.iterator)- _ptx_prefetchu_l1(mC_mnl.iterator)use_2cta_instrs = cute.size(tiled_mma.thr_id.shape) == 2⋯ 90 unchanged lines## Setup smem tensor A/B/SFA/SFB/C#- sC = None- if cutlass.const_expr(self.use_tma_store):- sC = storage.sC.get_tensor(- c_smem_layout_staged.outer, swizzle=c_smem_layout_staged.inner- )+ sC = storage.sC.get_tensor(+ c_smem_layout_staged.outer, swizzle=c_smem_layout_staged.inner+ )# (MMA, MMA_M, MMA_K, STAGE)sA = storage.sA.get_tensor(a_smem_layout_staged.outer, swizzle=a_smem_layout_staged.inner⋯ 283 unchanged lines_ptx_prefetch_global(tensormap_a_gmem_ptr)_ptx_prefetch_global_l1(tensormap_a_gmem_ptr)_ptx_prefetchu_l1(tensormap_a_gmem_ptr)- if _enable_tmap_prefetch:- _ptx_prefetch_tensormap(tensormap_a_gmem_ptr)_ptx_prefetch_global(tensormap_b_gmem_ptr)_ptx_prefetch_global_l1(tensormap_b_gmem_ptr)_ptx_prefetchu_l1(tensormap_b_gmem_ptr)- if _enable_tmap_prefetch:- _ptx_prefetch_tensormap(tensormap_b_gmem_ptr)_ptx_prefetch_global(tensormap_sfa_gmem_ptr)_ptx_prefetch_global_l1(tensormap_sfa_gmem_ptr)_ptx_prefetchu_l1(tensormap_sfa_gmem_ptr)- if _enable_tmap_prefetch:- _ptx_prefetch_tensormap(tensormap_sfa_gmem_ptr)_ptx_prefetch_global(tensormap_sfb_gmem_ptr)_ptx_prefetch_global_l1(tensormap_sfb_gmem_ptr)_ptx_prefetchu_l1(tensormap_sfb_gmem_ptr)- if _enable_tmap_prefetch:- _ptx_prefetch_tensormap(tensormap_sfb_gmem_ptr)tensormap_manager.update_tensormap((⋯ 62 unchanged linestensormap_sfb_gmem_ptr,cute.AddressSpace.generic,)+if self.prefetch_enabled:for pf_k_tile in cutlass.range(0, min(self.prefetch_dist, cur_k_tile_cnt), unroll=1⋯ 412 unchanged linesepi_tidx, tCtAcc_simt, tCgC_simt, epi_tile, use_2cta_instrs)tTR_rC = cute.make_rmem_tensor(tTR_rAcc.shape, self.c_dtype)+ gC_epi_simt = cute.flat_divide(tCgC_simt, epi_tile)thr_copy_t2r = tiled_copy_t2r.get_slice(epi_tidx)- if cutlass.const_expr(self.r2g_bits > 0):- num_bits_per_copy = self.r2g_bits- copy_atom_r2g = cute.make_copy_atom(- cute.nvgpu.CopyUniversalOp(),- self.c_dtype,- num_bits_per_copy=num_bits_per_copy,- )- tiled_copy_r2g = cute.make_tiled_copy_D(- copy_atom_r2g, tiled_copy_t2r- )- thr_copy_r2g = tiled_copy_r2g.get_slice(epi_tidx)- tRG_rC = tiled_copy_r2g.retile(tTR_rC)- else:- simt_atom_vec = cute.make_copy_atom(- cute.nvgpu.CopyUniversalOp(),- self.c_dtype,- )+ simt_atom_vec = cute.make_copy_atom(+ cute.nvgpu.CopyUniversalOp(),+ self.c_dtype,+ )## Persistent tile scheduling loop⋯ 59 unchanged lines_ptx_prefetch_global(tensormap_c_gmem_ptr)_ptx_prefetch_global_l1(tensormap_c_gmem_ptr)_ptx_prefetchu_l1(tensormap_c_gmem_ptr)- if _enable_tmap_prefetch:- _ptx_prefetch_tensormap(tensormap_c_gmem_ptr)tensormap_manager.update_tensormap(((real_tensor_c),),((tma_atom_c),),⋯ 136 unchanged linesgrouped_gemm_cta_tile_info.cta_tile_idx_n,0,)+ tile_origin_m = mma_tile_coord_mnl[0] * self.mma_tiler[0]+ tile_origin_n = mma_tile_coord_mnl[1] * self.mma_tiler[1]+ residue_m = (+ grouped_gemm_cta_tile_info.problem_shape_m - tile_origin_m+ )+ residue_n = (+ grouped_gemm_cta_tile_info.problem_shape_n - tile_origin_n+ )+ full_tile = (residue_m >= self.mma_tiler[0]) & (+ residue_n >= self.mma_tiler[1]+ )gC_mnl_tile = cute.local_tile(real_tensor_c,⋯ 2 unchanged lines)tCgC_tile = thr_mma.partition_C(gC_mnl_tile)tCgC_tile = self._transform_partitioned_tensor_layout(tCgC_tile)- if cutlass.const_expr(self.r2g_bits > 0):- gC_epi_simt_tile = cute.flat_divide(tCgC_tile, epi_tile)- tRG_gC = thr_copy_r2g.partition_D(gC_epi_simt_tile)- else:- tTR_gC = self.epilog_gmem_copy_and_partition_simt(- epi_tidx, tiled_copy_t2r, tCgC_tile, epi_tile- )+ tTR_gC = self.epilog_gmem_copy_and_partition_simt(+ epi_tidx, tiled_copy_t2r, tCgC_tile, epi_tile+ )# Set tensor memory buffer for current tile# (T2R, T2R_M, T2R_N, EPI_M, EPI_M)⋯ 7 unchanged linesacc_pipeline.consumer_wait(acc_consumer_state)tTR_tAcc = cute.group_modes(tTR_tAcc, 3, cute.rank(tTR_tAcc))- if cutlass.const_expr(self.r2g_bits > 0):- tRG_gC = cute.group_modes(tRG_gC, 3, cute.rank(tRG_gC))- else:- tTR_gC = cute.group_modes(tTR_gC, 3, cute.rank(tTR_gC))+ tTR_gC = cute.group_modes(tTR_gC, 3, cute.rank(tTR_gC))## Store accumulator to global memory in subtiles#subtile_cnt = cute.size(tTR_tAcc.shape, mode=[3])- if cutlass.const_expr(self.full_tiles_only):+ if full_tile:for subtile_idx in range(subtile_cnt):## Load accumulator from tensor memory buffer to register⋯ 5 unchanged lines# Convert to C type#acc_vec = tTR_rAcc.load()- if cutlass.const_expr(self.r2g_bits > 0):- tRG_rC.store(acc_vec.to(self.c_dtype))- tRG_gC_mn = tRG_gC[(None, None, None, subtile_idx)]- cute.copy(tiled_copy_r2g, tRG_rC, tRG_gC_mn)- else:- tTR_rC.store(acc_vec.to(self.c_dtype))- tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]- cute.copy(simt_atom_vec, tTR_rC, tTR_gC_mn)+ tTR_rC.store(acc_vec.to(self.c_dtype))+ tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]+ cute.copy(simt_atom_vec, tTR_rC, tTR_gC_mn)else:- tile_origin_m = mma_tile_coord_mnl[0] * self.mma_tiler[0]- tile_origin_n = mma_tile_coord_mnl[1] * self.mma_tiler[1]- residue_m = (- grouped_gemm_cta_tile_info.problem_shape_m - tile_origin_m+ cC_mnl = cute.make_identity_tensor(real_tensor_c.shape)+ cC_mnl_tile = cute.local_tile(+ cC_mnl,+ cute.slice_(self.mma_tiler, (None, None, 0)),+ mma_tile_coord_mnl,)- residue_n = (- grouped_gemm_cta_tile_info.problem_shape_n - tile_origin_n+ tCcC_tile = thr_mma.partition_C(cC_mnl_tile)+ tCcC_tile = self._transform_partitioned_tensor_layout(+ tCcC_tile)- full_tile = (residue_m >= self.mma_tiler[0]) & (- residue_n >= self.mma_tiler[1]+ tTR_cC = self.epilog_gmem_copy_and_partition_simt(+ epi_tidx, tiled_copy_t2r, tCcC_tile, epi_tile)- if full_tile:- for subtile_idx in range(subtile_cnt):- #- # Load accumulator from tensor memory buffer to register- #- tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]- cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)+ tTR_cC = cute.group_modes(tTR_cC, 3, cute.rank(tTR_cC))+ c_shape = real_tensor_c.shape+ for subtile_idx in range(subtile_cnt):+ #+ # Load accumulator from tensor memory buffer to register+ #+ tTR_tAcc_mn = tTR_tAcc[(None, None, None, subtile_idx)]+ cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)- #- # Convert to C type- #- acc_vec = tTR_rAcc.load()- if cutlass.const_expr(self.r2g_bits > 0):- tRG_rC.store(acc_vec.to(self.c_dtype))- tRG_gC_mn = tRG_gC[(None, None, None, subtile_idx)]- cute.copy(tiled_copy_r2g, tRG_rC, tRG_gC_mn)- else:- tTR_rC.store(acc_vec.to(self.c_dtype))- tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]- cute.copy(simt_atom_vec, tTR_rC, tTR_gC_mn)- else:- cC_mnl = cute.make_identity_tensor(real_tensor_c.shape)- cC_mnl_tile = cute.local_tile(- cC_mnl,- cute.slice_(self.mma_tiler, (None, None, 0)),- mma_tile_coord_mnl,+ #+ # Convert to C type+ #+ acc_vec = tTR_rAcc.load()+ tTR_rC.store(acc_vec.to(self.c_dtype))+ tTR_gC_mn = tTR_gC[(None, None, None, subtile_idx)]+ tTR_cC_mn = tTR_cC[(None, None, None, subtile_idx)]+ tTR_pC = cute.make_rmem_tensor(+ tTR_rC.shape, cutlass.Boolean)- tCcC_tile = thr_mma.partition_C(cC_mnl_tile)- tCcC_tile = self._transform_partitioned_tensor_layout(- tCcC_tile- )- if cutlass.const_expr(self.r2g_bits > 0):- cC_epi_simt = cute.flat_divide(tCcC_tile, epi_tile)- tRG_cC = thr_copy_r2g.partition_D(cC_epi_simt)- tRG_cC = cute.group_modes(- tRG_cC, 3, cute.rank(tRG_cC)- )- else:- tTR_cC = self.epilog_gmem_copy_and_partition_simt(- epi_tidx, tiled_copy_t2r, tCcC_tile, epi_tile- )- tTR_cC = cute.group_modes(- tTR_cC, 3, cute.rank(tTR_cC)- )- c_shape = real_tensor_c.shape- for subtile_idx in range(subtile_cnt):- #- # Load accumulator from tensor memory buffer to register- #- tTR_tAcc_mn = tTR_tAcc[- (None, None, None, subtile_idx)- ]- cute.copy(tiled_copy_t2r, tTR_tAcc_mn, tTR_rAcc)+ for i in range(cute.size(tTR_rC.shape)):+ tTR_pC[i] = cute.elem_less(tTR_cC_mn[i], c_shape)+ cute.basic_copy_if(tTR_pC, tTR_rC, tTR_gC_mn)- #- # Convert to C type- #- acc_vec = tTR_rAcc.load()- if cutlass.const_expr(self.r2g_bits > 0):- tRG_rC.store(acc_vec.to(self.c_dtype))- tRG_gC_mn = tRG_gC[- (None, None, None, subtile_idx)- ]- tRG_cC_mn = tRG_cC[- (None, None, None, subtile_idx)- ]- tRG_pC = cute.make_rmem_tensor(- tRG_cC_mn.shape, cutlass.Boolean- )- for i in range(cute.size(tRG_cC_mn.shape)):- tRG_pC[i] = cute.elem_less(- tRG_cC_mn[i], c_shape- )- cute.copy(- tiled_copy_r2g,- tRG_rC,- tRG_gC_mn,- pred=tRG_pC,- )- else:- tTR_rC.store(acc_vec.to(self.c_dtype))- tTR_gC_mn = tTR_gC[- (None, None, None, subtile_idx)- ]- tTR_cC_mn = tTR_cC[- (None, None, None, subtile_idx)- ]- tTR_pC = cute.make_rmem_tensor(- tTR_rC.shape, cutlass.Boolean- )- for i in range(cute.size(tTR_rC.shape)):- tTR_pC[i] = cute.elem_less(- tTR_cC_mn[i], c_shape- )- cute.basic_copy_if(tTR_pC, tTR_rC, tTR_gC_mn)-## Async arrive accumulator buffer empty#⋯ 325 unchanged linessf_vec_size: int,smem_capacity: int,occupancy: int,- use_tma_store: bool,+ force_c_stage: int | None = None,) -> Tuple[int, int, int]:# ACC stagesnum_acc_stage = 1 if mma_tiler_mnk[1] == 256 else 2# Default C stages- # Keep a single stage to free SMEM for extra AB stages.- num_c_stage = 1 if use_tma_store else 0+ num_c_stage = 2 if force_c_stage is None else max(1, int(force_c_stage))# Calculate smem layout and size for one stage of A, B, SFA, SFB and Ca_smem_layout_stage_one = sm100_utils.make_smem_layout_a(⋯ 21 unchanged lines1, # a tmp 1 stage is provided)- c_smem_layout_staged_one = None- if use_tma_store:- c_smem_layout_staged_one = sm100_utils.make_smem_layout_epi(- c_dtype,- c_layout,- epi_tile,- 1,- )+ c_smem_layout_staged_one = sm100_utils.make_smem_layout_epi(+ c_dtype,+ c_layout,+ epi_tile,+ 1,+ )ab_bytes_per_stage = (cute.size_in_bytes(a_dtype, a_smem_layout_stage_one)⋯ 2 unchanged lines+ cute.size_in_bytes(sf_dtype, sfb_smem_layout_staged_one))mbar_helpers_bytes = 1024- c_bytes_per_stage = 0- if use_tma_store and c_smem_layout_staged_one is not None:- c_bytes_per_stage = cute.size_in_bytes(c_dtype, c_smem_layout_staged_one)+ c_bytes_per_stage = cute.size_in_bytes(c_dtype, c_smem_layout_staged_one)c_bytes = c_bytes_per_stage * num_c_stage# Calculate A/B/SFA/SFB stages:⋯ 3 unchanged linesnum_ab_stage = (smem_capacity // occupancy - (mbar_helpers_bytes + c_bytes)) // ab_bytes_per_stage+ if num_ab_stage < 1:+ num_ab_stage = 1# Refine epilogue stages:# Calculate remaining smem after allocating for A/B/SFA/SFB stages and reserved bytes# Add remaining unused smem to epilogue- if use_tma_store and c_bytes_per_stage:+ if force_c_stage is None:num_c_stage += (smem_capacity- occupancy * ab_bytes_per_stage * num_ab_stage⋯ 7 unchanged linestotal_num_clusters: int,cluster_shape_mn: tuple[int, int],max_active_clusters: cutlass.Constexpr[int],+ swizzle_size: int = 1,+ raster_along_m: bool = True,) -> tuple[utils.PersistentTileSchedulerParams, tuple[int, int, int]]:# Create problem shape with M, N dimensions from cluster shape# and L dimension representing the total number of clusters.⋯ 4 unchanged lines)tile_sched_params = utils.PersistentTileSchedulerParams(- problem_shape_ntile_mnl, (*cluster_shape_mn, 1)+ problem_shape_ntile_mnl,+ (*cluster_shape_mn, 1),+ swizzle_size=swizzle_size,+ raster_along_m=raster_along_m,)grid = utils.StaticPersistentTileScheduler.get_grid_shape(⋯ 312 unchanged linessfs: list[tuple[torch.Tensor, torch.Tensor]],problem_sizes: list[tuple[int, int, int, int]],use_tma_store: bool,- r2g_bits: int | None,- full_tiles_only: bool,c_col_major: bool,max_ab_stage: int | None,prefetch_dist: int | None,+ force_c_stage: int | None,c_assumed_align: int,- c_divisibility: int):+ c_divisibility: int,+ swizzle_size: int,+ raster_along_m: bool):dev = abc_tensors[0][0].devicegroup_count = len(problem_sizes)⋯ 81 unchanged linesmma_tiler_mn=mma_tiler_mn,cluster_shape_mn=cluster_shape_mn,use_tma_store=use_tma_store,- r2g_bits=r2g_bits,- full_tiles_only=full_tiles_only,max_ab_stage=max_ab_stage,prefetch_dist=prefetch_dist,+ force_c_stage=force_c_stage,c_assumed_align=c_assumed_align,c_divisibility=c_divisibility,+ swizzle_size=swizzle_size,+ raster_along_m=raster_along_m,)gemm.c_tile_stride = (int(c_ref.stride(0)), int(c_ref.stride(1)))- enable_tvm_ffi = _tvm_ffi_enabled()- compile_opts = "--opt-level 2"- if enable_tvm_ffi:- compile_opts += " --enable-tvm-ffi"try:compiled = cute.compile(gemm,⋯ 11 unchanged linestmap_cute,max_active,_get_handle(),- options=compile_opts,+ options="--opt-level 2",)except Exception as exc:- if enable_tvm_ffi:- print(- f"cute.compile with TVM FFI failed, retrying without TVM FFI: {exc}",- file=sys.stderr,- flush=True,- )- try:- compiled = cute.compile(- gemm,- initial_a,- initial_b,- initial_c,- initial_sfa,- initial_sfb,- group_count,- dims_cute,- strides_cute,- ptrs_cute,- ptrs_sf_cute,- total_num_clusters,- tmap_cute,- max_active,- _get_handle(),- options="--opt-level 2",- )- except Exception as exc2:- print(f"cute.compile failed: {exc2}", file=sys.stderr, flush=True)- raise RuntimeError(str(exc2)) from exc2- else:- print(f"cute.compile failed: {exc}", file=sys.stderr, flush=True)- raise RuntimeError(str(exc)) from exc+ print(f"cute.compile failed: {exc}", file=sys.stderr, flush=True)+ raise RuntimeError(str(exc)) from excentry = {"cfg_key": cfg_key,⋯ 46 unchanged linessfs: list[tuple[torch.Tensor, torch.Tensor]],problem_sizes: list[tuple[int, int, int, int]],use_tma_store: bool,- r2g_bits: int | None,- full_tiles_only: bool,c_col_major: bool,max_ab_stage: int | None,prefetch_dist: int | None,+ force_c_stage: int | None,c_assumed_align: int,- c_divisibility: int):+ c_divisibility: int,+ swizzle_size: int,+ raster_along_m: bool):if not abc_tensors:returndev = abc_tensors[0][0].device⋯ 1 unchanged linesdev.index if dev.type == "cuda" else -1,cfg_key,use_tma_store,- r2g_bits,- full_tiles_only,c_col_major,max_ab_stage,prefetch_dist,+ force_c_stage,c_assumed_align,c_divisibility,+ swizzle_size,+ raster_along_m,tuple(problem_sizes),)entry = _kernel_cache.get(cache_key)⋯ 6 unchanged linessfs,problem_sizes,use_tma_store,- r2g_bits,- full_tiles_only,c_col_major,max_ab_stage,prefetch_dist,+ force_c_stage,c_assumed_align,c_divisibility,+ swizzle_size,+ raster_along_m,)_kernel_cache[cache_key] = entry_update_ptrs(entry, abc_tensors, sfs)⋯ 54 unchanged linesreturn idx_1, idx_2- def _split_full_tiles(- problem_sizes: list[tuple[int, int, int, int]],- mma_tiler_mn: tuple[int, int],- ) -> tuple[list[int], list[int]]:- full: list[int] = []- partial: list[int] = []- tile_m, tile_n = mma_tiler_mn- for i, (m, n, _k, _l) in enumerate(problem_sizes):- if (m % tile_m) == 0 and (n % tile_n) == 0:- full.append(i)- else:- partial.append(i)- return full, partial+ def _swizzle_params(is_wide: bool) -> tuple[int, bool]:+ if is_wide:+ size = int(os.environ.get("NVFP4_SWIZZLE_WIDE", "1"))+ raster_along_m = os.environ.get("NVFP4_SWIZZLE_WIDE_RASTER_M", "0") != "0"+ else:+ size = int(os.environ.get("NVFP4_SWIZZLE_NARROW", "1"))+ raster_along_m = os.environ.get("NVFP4_SWIZZLE_NARROW_RASTER_M", "1") != "0"+ if size not in (1, 2, 4, 8):+ size = 1+ return size, raster_along_m⋯ 32 unchanged linesabc_tensors, _sfs_raw, sfs_reordered, problem_sizes = datasfs_reordered = _normalize_sfs(sfs_reordered)- is_sm100 = bool(abc_tensors) and _is_sm100_device(abc_tensors[0][0])- if _enable_stage8 and is_sm100:- ext_res = _try_stage8_ext(abc_tensors, sfs_reordered, problem_sizes)- if ext_res is not None:- return ext_residx_unswapped: list[int] = []idx_rest: list[int] = []⋯ 13 unchanged linesabc_swapped, sfs_swapped, sizes_swapped = _swap_inputs(abc_rest, sfs_rest, sizes_rest)+ if _enable_stage8 and abc_swapped and _is_sm100_device(abc_swapped[0][0]):+ stage8_out = _try_stage8_ext(abc_swapped, sfs_swapped, sizes_swapped)+ if stage8_out is not None:+ return [c for (_a, _b, c) in abc_tensors]c_col_major = True- c_assumed_align = 16- c_divisibility = 8- r2g_env = os.environ.get("NVFP4_USE_R2G")- use_r2g_swapped = is_sm100 and r2g_env == "1"- simt_env = os.environ.get("NVFP4_USE_SIMT_STORE")- if simt_env is None:- use_simt_store_swapped = is_sm100 and not use_r2g_swapped- else:- use_simt_store_swapped = is_sm100 and simt_env == "1"- use_r2g_unswapped = is_sm100 and os.environ.get("NVFP4_UNSWAPPED_SIMT", "") == "1"- r2g_bits_unswapped: int | None = None- r2g_bits_swapped: int | None = None- if use_r2g_swapped:- r2g_bits = _env_int("NVFP4_R2G_BITS", 256)- if r2g_bits not in (64, 128, 256):- r2g_bits = 64- r2g_bits_swapped = r2g_bits- c_assumed_align_swapped, c_divisibility_swapped = _r2g_store_config(- r2g_bits- )- else:- c_assumed_align_swapped = c_assumed_align- c_divisibility_swapped = c_divisibility- if use_r2g_unswapped:- r2g_bits = _env_int("NVFP4_UNSWAPPED_R2G_BITS", 256)- if r2g_bits not in (64, 128, 256):- r2g_bits = 256- r2g_bits_unswapped = r2g_bits- c_assumed_align_unswapped, c_divisibility_unswapped = _r2g_store_config(- r2g_bits- )- else:- c_assumed_align_unswapped = 32 if is_sm100 else 16- c_divisibility_unswapped = 16 if is_sm100 else 8+ is_sm100 = bool(abc_swapped) and _is_sm100_device(abc_swapped[0][0])+ c_assumed_align = 32+ c_divisibility = 16prefetch_dist = None if is_sm100 else 0use_tma_store = True- use_tma_store_unswapped = False if use_r2g_unswapped else True- use_tma_store_swapped = not (use_r2g_swapped or use_simt_store_swapped)-idx_wide: list[int] = []idx_narrow: list[int] = []for i, (_m, _n, k, _l) in enumerate(sizes_swapped):⋯ 18 unchanged linesabc_swapped, sfs_swapped, sizes_swapped, idx_narrow_2)- enable_simt = os.environ.get("NVFP4_ENABLE_SIMT", "") == "1"+ enable_simt = os.environ.get("NVFP4_ENABLE_SIMT", "1") == "1"+ swizzle_wide, raster_wide = _swizzle_params(True)+ swizzle_narrow, raster_narrow = _swizzle_params(False)+ force_c_stage_2sm_env = os.environ.get("NVFP4_FORCE_C_STAGE_2SM", "")+ force_c_stage_2sm = int(force_c_stage_2sm_env) if force_c_stage_2sm_env else None+ if force_c_stage_2sm is not None and force_c_stage_2sm <= 0:+ force_c_stage_2sm = None+ max_ab_stage_2sm_env = os.environ.get("NVFP4_MAX_AB_STAGE_2SM", "")+ max_ab_stage_2sm = int(max_ab_stage_2sm_env) if max_ab_stage_2sm_env else None+ if max_ab_stage_2sm is not None and max_ab_stage_2sm <= 0:+ max_ab_stage_2sm = None_run_group("unswapped",- (256, 256),+ (128, 256),(2, 1),abc_unswapped,sfs_unswapped,sizes_unswapped,- use_tma_store_unswapped,- r2g_bits_unswapped,+ use_tma_store,False,- False,None,prefetch_dist,- c_assumed_align_unswapped,- c_divisibility_unswapped,+ None,+ c_assumed_align,+ c_divisibility,+ swizzle_narrow,+ raster_narrow,)-if enable_simt:def is_simt_safe(sz: tuple[int, int, int, int]) -> bool:- _m, n, _k, _l = sz- # SIMT vector store width depends on R2G bits; ensure N divisibility.- if r2g_bits_swapped == 256:- return (n % 16) == 0- if r2g_bits_swapped == 128:- return (n % 8) == 0- if r2g_bits_swapped == 64:- return (n % 4) == 0- return (n % 8) == 0+ _m, n, k, _l = sz+ # SIMT vector store uses 128-bit vectors; require N divisible by 8.+ # Limit SIMT to k < 4096 (1SM path) to avoid 2CTA edge cases.+ return (n % 8) == 0 and k < 4096def gather(idx_list: list[int], want_simt: bool):abc_out = []⋯ 41 unchanged linesabc_tma: list[tuple[torch.Tensor, torch.Tensor, torch.Tensor]],sfs_tma: list[tuple[torch.Tensor, torch.Tensor]],sizes_tma: list[tuple[int, int, int, int]],+ swizzle_size: int,+ raster_along_m: bool,+ max_ab_stage: int | None = None,+ force_c_stage: int | None = None,):_run_group(f"{cfg_key_base}s",⋯ 3 unchanged linessfs_simt,sizes_simt,False,- r2g_bits_swapped,- False,c_col_major,- None,+ max_ab_stage,prefetch_dist,- c_assumed_align_swapped,- c_divisibility_swapped,+ force_c_stage,+ c_assumed_align,+ c_divisibility,+ swizzle_size,+ raster_along_m,)_run_group(f"{cfg_key_base}t",⋯ 3 unchanged linessfs_tma,sizes_tma,True,- None,- False,c_col_major,- None,+ max_ab_stage,prefetch_dist,+ force_c_stage,c_assumed_align,c_divisibility,+ swizzle_size,+ raster_along_m,)run_pair(⋯ 6 unchanged linesabc_wide_1_tma,sfs_wide_1_tma,sizes_wide_1_tma,+ swizzle_wide,+ raster_wide,)run_pair("w2",⋯ 5 unchanged linesabc_wide_2_tma,sfs_wide_2_tma,sizes_wide_2_tma,+ swizzle_wide,+ raster_wide,+ max_ab_stage_2sm,+ force_c_stage_2sm,)run_pair("n1",⋯ 5 unchanged linesabc_narrow_1_tma,sfs_narrow_1_tma,sizes_narrow_1_tma,+ swizzle_narrow,+ raster_narrow,)run_pair("n2",⋯ 5 unchanged linesabc_narrow_2_tma,sfs_narrow_2_tma,sizes_narrow_2_tma,+ swizzle_narrow,+ raster_narrow,+ max_ab_stage_2sm,+ force_c_stage_2sm,)else:- if use_r2g_swapped:- _run_group(- "sw1w",- (128, 256),- (2, 1),- abc_wide_1,- sfs_wide_1,- sizes_wide_1,- False,- r2g_bits_swapped,- False,- c_col_major,- None,- prefetch_dist,- c_assumed_align_swapped,- c_divisibility_swapped,- )- _run_group(- "sw2w",- (256, 256),- (4, 1),- abc_wide_2,- sfs_wide_2,- sizes_wide_2,- False,- r2g_bits_swapped,- False,- c_col_major,- None,- prefetch_dist,- c_assumed_align_swapped,- c_divisibility_swapped,- )- _run_group(- "sw1n",- (128, 128),- (2, 1),- abc_narrow_1,- sfs_narrow_1,- sizes_narrow_1,- False,- r2g_bits_swapped,- False,- c_col_major,- None,- prefetch_dist,- c_assumed_align_swapped,- c_divisibility_swapped,- )- _run_group(- "sw2n",- (256, 128),- (4, 1),- abc_narrow_2,- sfs_narrow_2,- sizes_narrow_2,- False,- r2g_bits_swapped,- False,- c_col_major,- None,- prefetch_dist,- c_assumed_align_swapped,- c_divisibility_swapped,- )- else:- _run_group(- "swap_1sm_wide",- (128, 256),- (2, 1),- abc_wide_1,- sfs_wide_1,- sizes_wide_1,- use_tma_store_swapped,- r2g_bits_swapped,- False,- c_col_major,- None,- prefetch_dist,- c_assumed_align_swapped,- c_divisibility_swapped,- )- _run_group(- "swap_2sm_wide",- (256, 256),- (4, 1),- abc_wide_2,- sfs_wide_2,- sizes_wide_2,- use_tma_store_swapped,- r2g_bits_swapped,- False,- c_col_major,- None,- prefetch_dist,- c_assumed_align_swapped,- c_divisibility_swapped,- )- _run_group(- "swap_1sm",- (128, 128),- (2, 1),- abc_narrow_1,- sfs_narrow_1,- sizes_narrow_1,- use_tma_store_swapped,- r2g_bits_swapped,- False,- c_col_major,- None,- prefetch_dist,- c_assumed_align_swapped,- c_divisibility_swapped,- )- _run_group(- "swap_2sm",- (256, 128),- (4, 1),- abc_narrow_2,- sfs_narrow_2,- sizes_narrow_2,- use_tma_store_swapped,- r2g_bits_swapped,- False,- c_col_major,- None,- prefetch_dist,- c_assumed_align_swapped,- c_divisibility_swapped,- )+ _run_group(+ "swap_1sm_wide",+ (128, 256),+ (2, 1),+ abc_wide_1,+ sfs_wide_1,+ sizes_wide_1,+ use_tma_store,+ c_col_major,+ None,+ prefetch_dist,+ None,+ c_assumed_align,+ c_divisibility,+ swizzle_wide,+ raster_wide,+ )+ _run_group(+ "swap_2sm_wide",+ (256, 256),+ (4, 1),+ abc_wide_2,+ sfs_wide_2,+ sizes_wide_2,+ use_tma_store,+ c_col_major,+ max_ab_stage_2sm,+ prefetch_dist,+ force_c_stage_2sm,+ c_assumed_align,+ c_divisibility,+ swizzle_wide,+ raster_wide,+ )+ _run_group(+ "swap_1sm",+ (128, 128),+ (2, 1),+ abc_narrow_1,+ sfs_narrow_1,+ sizes_narrow_1,+ use_tma_store,+ c_col_major,+ None,+ prefetch_dist,+ None,+ c_assumed_align,+ c_divisibility,+ swizzle_narrow,+ raster_narrow,+ )+ _run_group(+ "swap_2sm",+ (256, 128),+ (4, 1),+ abc_narrow_2,+ sfs_narrow_2,+ sizes_narrow_2,+ use_tma_store,+ c_col_major,+ max_ab_stage_2sm,+ prefetch_dist,+ force_c_stage_2sm,+ c_assumed_align,+ c_divisibility,+ swizzle_narrow,+ raster_narrow,+ )return [c for (_a, _b, c) in abc_tensors]
scrolls · 1190 diff lines total
Best evidence level for this revision: reported
JSON