Skip to content
KernelIndex
Search⌘K

submission 117316

shiyegao · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 110 lines, June 9 Researcher Reciprocity License v1.0.

template.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemm-117316?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVFP4 GEMMsuite of 3 cases
NVIDIA B200
17.6µs
#173 of 369
2025-12-01

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:cfc5e9555af203866e878d2285171b329c06b101b7094e77c96b1da305f78100
license declaredunknown
license concludedunknown
authorsshiyegao
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

fp4NVFP4 块缩放 GEMM - 极致性能版

Kernel source

template.py110 lines
"""
NVFP4 块缩放 GEMM - 极致性能版
优化策略:
1. 批量处理:将 L 维度的 Scale 重排合并为单次大内存操作,避免循环内的小内存分配与拷贝。
2. 零拷贝视图:循环内部仅做 View 操作,消除 Python 端内存开销。
3. 快速路径:针对官方预处理数据 (7元组) 提供极速路径。
"""

from __future__ import annotations

from typing import Tuple, Optional

import torch

def _ceil_div(a: int, b: int) -> int:
    return (a + b - 1) // b

def _to_blocked(input_matrix: torch.Tensor) -> torch.Tensor:
    """
    回退路径:将原始缩放因子转换为分块布局。
    仅在未提供预重排数据时使用。
    """
    rows, cols = input_matrix.shape
    n_row_blocks = _ceil_div(rows, 128)
    n_col_blocks = _ceil_div(cols, 4)
    # 原始逻辑保持不变,确保正确性
    blocks = input_matrix.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
    rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
    return rearranged.flatten()

def _prepare_scales_batch(scale_perm: torch.Tensor) -> torch.Tensor:
    """
    极致优化路径:
    输入形状: (32, 4, BlockM, 4, BlockK, L)
    目标形状: (L, Flattened_Blocked_Scale)
    
    操作:
    1. 将 L 维提到第 0 维。
    2. 调整其余维度以匹配 _scaled_mm 要求的 (BlockM, BlockK, 32, 4, 4)。
    3. 执行一次性 contiguous 拷贝,消除循环内的所有内存搬运。
    """
    # 原始维度索引: 0:32, 1:4, 2:BM, 3:4, 4:BK, 5:L
    # 目标维度顺序: 5(L), 2(BM), 4(BK), 0(32), 1(4), 3(4)
    # 对应 _scaled_mm 需要的 blocked layout
    permuted = scale_perm.permute(5, 2, 4, 0, 1, 3)
    
    # 这里的 contiguous 是关键:它将所有 L 层的重排合并为一次 GPU Kernel 调用
    return permuted.contiguous().view(scale_perm.size(-1), -1)

@torch.inference_mode()
def custom_kernel(data: Tuple[torch.Tensor, ...]) -> torch.Tensor:
    # 快速解包,避免 len() 检查的微小开销(假设输入总是合法的 5 或 7)
    if len(data) >= 7:
        a, b, sfa, sfb, sfa_perm, sfb_perm, c = data
    else:
        a, b, sfa, sfb, c = data
        sfa_perm = sfb_perm = None

    _, _, l = c.shape

    # --- 缩放因子准备阶段 ---
    
    scale_a_batch: Optional[torch.Tensor] = None
    scale_b_batch: Optional[torch.Tensor] = None

    # 策略 A: 极速路径 (利用官方预重排数据)
    # 通过一次性重排所有 L 层,将复杂度从 O(L) 降低到 O(1) 的 Kernel Launch
    if sfa_perm is not None and sfa_perm.dim() == 6:
        scale_a_batch = _prepare_scales_batch(sfa_perm)
    
    if sfb_perm is not None and sfb_perm.dim() == 6:
        scale_b_batch = _prepare_scales_batch(sfb_perm)

    # --- 计算循环阶段 ---
    
    # 预取 transpose,b 在内存中通常是 (N, K, L)
    # 如果 b 是 (N, K, L),transpose(0, 1) 只是 stride 变换,开销极小
    
    for i in range(l):
        # 1. 获取 Scale A
        if scale_a_batch is not None:
            # 这里的切片是 Zero-Copy 的 View,极快
            scale_a = scale_a_batch[i]
        else:
            # 回退路径 (慢)
            scale_a = _to_blocked(sfa[:, :, i])

        # 2. 获取 Scale B
        if scale_b_batch is not None:
            scale_b = scale_b_batch[i]
        else:
            scale_b = _to_blocked(sfb[:, :, i])

        # 3. 执行核心计算
        # 注意:out_dtype=torch.float16 是必须的,bias=None
        res = torch._scaled_mm(
            a[:, :, i],
            b[:, :, i].transpose(0, 1),
            scale_a,
            scale_b,
            bias=None,
            out_dtype=torch.float16
        )
        
        # 4. 写回结果
        c[:, :, i].copy_(res)

    return c

__all__ = ["custom_kernel"]
scrolls · 110 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 117310.

"""
- 使用 PyTorch 内置 `torch._scaled_mm` 完成 NVFP4 块缩放 GEMM。
- 优先利用评测侧提供的预重排缩放因子,减少 Python 端重排开销;若未提供则退回参考重排。
+ NVFP4 块缩放 GEMM - 极致性能版
+ 优化策略:
+ 1. 批量处理:将 L 维度的 Scale 重排合并为单次大内存操作,避免循环内的小内存分配与拷贝。
+ 2. 零拷贝视图:循环内部仅做 View 操作,消除 Python 端内存开销。
+ 3. 快速路径:针对官方预处理数据 (7元组) 提供极速路径。
"""
from __future__ import annotations
- from typing import Tuple
+ from typing import Tuple, Optional
import torch
-
def _ceil_div(a: int, b: int) -> int:
return (a + b - 1) // b
-
def _to_blocked(input_matrix: torch.Tensor) -> torch.Tensor:
- """将缩放因子转换为 torch._scaled_mm 期望的分块布局。"""
+ """
+ 回退路径:将原始缩放因子转换为分块布局。
+ 仅在未提供预重排数据时使用。
+ """
rows, cols = input_matrix.shape
n_row_blocks = _ceil_div(rows, 128)
n_col_blocks = _ceil_div(cols, 4)
+ # 原始逻辑保持不变,确保正确性
blocks = input_matrix.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
return rearranged.flatten()
-
- def _permuted_to_blocked(scale_permuted: torch.Tensor, l_idx: int) -> torch.Tensor:
+ def _prepare_scales_batch(scale_perm: torch.Tensor) -> torch.Tensor:
"""
- 将评测侧预重排的缩放因子恢复到 torch._scaled_mm 可接受的扁平布局。
- 预重排形状约为 [32, 4, ceil(m/128), 4, ceil(k/16/4), L]。
+ 极致优化路径:
+ 输入形状: (32, 4, BlockM, 4, BlockK, L)
+ 目标形状: (L, Flattened_Blocked_Scale)
+
+ 操作:
+ 1. 将 L 维提到第 0 维。
+ 2. 调整其余维度以匹配 _scaled_mm 要求的 (BlockM, BlockK, 32, 4, 4)。
+ 3. 执行一次性 contiguous 拷贝,消除循环内的所有内存搬运。
"""
- # 先取出指定 batch,再调整维度顺序使得 block_m、block_k 成为前两维,确保与参考重排一致。
- sliced = scale_permuted[..., l_idx] # (32, 4, block_m, 4, block_k)
- blocked = sliced.permute(2, 4, 0, 1, 3).contiguous().reshape(-1, 32, 16)
- return blocked.flatten()
+ # 原始维度索引: 0:32, 1:4, 2:BM, 3:4, 4:BK, 5:L
+ # 目标维度顺序: 5(L), 2(BM), 4(BK), 0(32), 1(4), 3(4)
+ # 对应 _scaled_mm 需要的 blocked layout
+ permuted = scale_perm.permute(5, 2, 4, 0, 1, 3)
+
+ # 这里的 contiguous 是关键:它将所有 L 层的重排合并为一次 GPU Kernel 调用
+ return permuted.contiguous().view(scale_perm.size(-1), -1)
-
+ @torch.inference_mode()
def custom_kernel(data: Tuple[torch.Tensor, ...]) -> torch.Tensor:
- """
- 兼容五元组 (a, b, sfa, sfb, c) 与七元组 (a, b, sfa, sfb, sfa_perm, sfb_perm, c)。
- 优先使用预重排缩放因子以减少重排成本。
- """
- if len(data) == 5:
- a, b, sfa, sfb, c = data
- sfa_perm = sfb_perm = None
- elif len(data) >= 7:
+ # 快速解包,避免 len() 检查的微小开销(假设输入总是合法的 5 或 7)
+ if len(data) >= 7:
a, b, sfa, sfb, sfa_perm, sfb_perm, c = data
else:
- raise ValueError("data tuple size must be 5 or 7")
+ a, b, sfa, sfb, c = data
+ sfa_perm = sfb_perm = None
_, _, l = c.shape
- for l_idx in range(l):
- # 缩放优先走预重排路径,缺失时回退参考重排。
- if sfa_perm is not None and sfa_perm.dim() == 6:
- scale_a = _permuted_to_blocked(sfa_perm, l_idx)
+
+ # --- 缩放因子准备阶段 ---
+
+ scale_a_batch: Optional[torch.Tensor] = None
+ scale_b_batch: Optional[torch.Tensor] = None
+
+ # 策略 A: 极速路径 (利用官方预重排数据)
+ # 通过一次性重排所有 L 层,将复杂度从 O(L) 降低到 O(1) 的 Kernel Launch
+ if sfa_perm is not None and sfa_perm.dim() == 6:
+ scale_a_batch = _prepare_scales_batch(sfa_perm)
+
+ if sfb_perm is not None and sfb_perm.dim() == 6:
+ scale_b_batch = _prepare_scales_batch(sfb_perm)
+
+ # --- 计算循环阶段 ---
+
+ # 预取 transpose,b 在内存中通常是 (N, K, L)
+ # 如果 b 是 (N, K, L),transpose(0, 1) 只是 stride 变换,开销极小
+
+ for i in range(l):
+ # 1. 获取 Scale A
+ if scale_a_batch is not None:
+ # 这里的切片是 Zero-Copy 的 View,极快
+ scale_a = scale_a_batch[i]
else:
- scale_a = _to_blocked(sfa[:, :, l_idx])
+ # 回退路径 (慢)
+ scale_a = _to_blocked(sfa[:, :, i])
- if sfb_perm is not None and sfb_perm.dim() == 6:
- scale_b = _permuted_to_blocked(sfb_perm, l_idx)
+ # 2. 获取 Scale B
+ if scale_b_batch is not None:
+ scale_b = scale_b_batch[i]
else:
- scale_b = _to_blocked(sfb[:, :, l_idx])
+ scale_b = _to_blocked(sfb[:, :, i])
- result = torch._scaled_mm(
- a[:, :, l_idx],
- b[:, :, l_idx].transpose(0, 1),
+ # 3. 执行核心计算
+ # 注意:out_dtype=torch.float16 是必须的,bias=None
+ res = torch._scaled_mm(
+ a[:, :, i],
+ b[:, :, i].transpose(0, 1),
scale_a,
scale_b,
bias=None,
- out_dtype=torch.float16,
+ out_dtype=torch.float16
)
- c[:, :, l_idx].copy_(result)
+
+ # 4. 写回结果
+ c[:, :, i].copy_(res)
+
return c
-
- __all__ = ["custom_kernel"]
+ __all__ = ["custom_kernel"]
No newline at end of file
scrolls · 150 diff lines total

Best evidence level for this revision: reported

JSON