submission 385049
novo_force · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 72 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-group-gemm-385049?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:2664789c53681c8ffe469ea930e1e415f0820e79d50699984b9d38e06fc81c46
license declaredunknown
license concludedunknown
authorsnovo_force
imported2026-08-15
Kernel source
submission.py72 lines
import torch
_SF_VEC_SIZE = 16
def _ceil_div(a: int, b: int) -> int:
return (a + b - 1) // b
def _to_blocked(scale_2d: torch.Tensor) -> torch.Tensor:
rows, cols = scale_2d.shape
n_row_blocks = _ceil_div(rows, 128)
n_col_blocks = _ceil_div(cols, 4)
padded_rows = n_row_blocks * 128
padded_cols = n_col_blocks * 4
if padded_rows != rows or padded_cols != cols:
padded = scale_2d.new_zeros((padded_rows, padded_cols))
padded[:rows, :cols] = scale_2d
else:
padded = scale_2d
blocks = padded.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
return rearranged.flatten()
def _scaled_fp4_gemm(a: torch.Tensor, b: torch.Tensor, sfa: torch.Tensor, sfb: torch.Tensor) -> torch.Tensor:
scale_a = _to_blocked(sfa)
scale_b = _to_blocked(sfb)
return torch._scaled_mm(
a.view(torch.float4_e2m1fn_x2),
b.transpose(0, 1).view(torch.float4_e2m1fn_x2),
scale_a,
scale_b,
bias=None,
out_dtype=torch.float16,
)
def custom_kernel(data):
if len(data) == 3:
abc_tensors, sfasfb_tensors, problem_sizes = data
else:
abc_tensors, sfasfb_tensors, _, problem_sizes = data
outputs = []
for (a, b, c), (sfa, sfb), (m, n, k, l) in zip(abc_tensors, sfasfb_tensors, problem_sizes):
device = a.device
if b.device != device:
b = b.to(device)
if c.device != device:
c = c.to(device)
if sfa.device != device:
sfa = sfa.to(device)
if sfb.device != device:
sfb = sfb.to(device)
for l_idx in range(l):
c[:, :, l_idx] = _scaled_fp4_gemm(a[:, :, l_idx], b[:, :, l_idx], sfa[:, :, l_idx], sfb[:, :, l_idx])
outputs.append(c)
return outputs
__all__ = ["custom_kernel"]
scrolls · 72 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON