Skip to content
KernelIndex
Search⌘K

submission 385049

novo_force · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 72 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-group-gemm-385049?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVFP4 group GEMMsuite of 4 cases
NVIDIA B200
1.03ms
#136 of 145
2026-01-20

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:2664789c53681c8ffe469ea930e1e415f0820e79d50699984b9d38e06fc81c46
license declaredunknown
license concludedunknown
authorsnovo_force
imported2026-08-15

Kernel source

submission.py72 lines
import torch


_SF_VEC_SIZE = 16


def _ceil_div(a: int, b: int) -> int:
    return (a + b - 1) // b


def _to_blocked(scale_2d: torch.Tensor) -> torch.Tensor:
    
    rows, cols = scale_2d.shape

    n_row_blocks = _ceil_div(rows, 128)
    n_col_blocks = _ceil_div(cols, 4)
    padded_rows = n_row_blocks * 128
    padded_cols = n_col_blocks * 4

    if padded_rows != rows or padded_cols != cols:
        padded = scale_2d.new_zeros((padded_rows, padded_cols))
        padded[:rows, :cols] = scale_2d
    else:
        padded = scale_2d

    blocks = padded.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
    rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
    return rearranged.flatten()


def _scaled_fp4_gemm(a: torch.Tensor, b: torch.Tensor, sfa: torch.Tensor, sfb: torch.Tensor) -> torch.Tensor:
    
    scale_a = _to_blocked(sfa)
    scale_b = _to_blocked(sfb)
    return torch._scaled_mm(
        a.view(torch.float4_e2m1fn_x2),
        b.transpose(0, 1).view(torch.float4_e2m1fn_x2),
        scale_a,
        scale_b,
        bias=None,
        out_dtype=torch.float16,
    )


def custom_kernel(data):
    
    if len(data) == 3:
        abc_tensors, sfasfb_tensors, problem_sizes = data
    else:
        abc_tensors, sfasfb_tensors, _, problem_sizes = data

    outputs = []
    for (a, b, c), (sfa, sfb), (m, n, k, l) in zip(abc_tensors, sfasfb_tensors, problem_sizes):
        device = a.device
        if b.device != device:
            b = b.to(device)
        if c.device != device:
            c = c.to(device)
        if sfa.device != device:
            sfa = sfa.to(device)
        if sfb.device != device:
            sfb = sfb.to(device)

        
        for l_idx in range(l):
            c[:, :, l_idx] = _scaled_fp4_gemm(a[:, :, l_idx], b[:, :, l_idx], sfa[:, :, l_idx], sfb[:, :, l_idx])
        outputs.append(c)
    return outputs


__all__ = ["custom_kernel"]
scrolls · 72 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON