Skip to content
KernelIndex
Search⌘K

submission 71049

mdouglas · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 59 lines, June 9 Researcher Reciprocity License v1.0.

submission2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-71049?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVFP4 GEMVsuite of 3 cases
NVIDIA B200
88.2µs
#367 of 678
2025-11-11

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:3aba0c7b9c36b4f187e4112f6f48975b0c7c930a2e07fc787a39638287f875c7
license declaredunknown
license concludedunknown
authorsmdouglas
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

autotune@torch.compile(dynamic=False, mode="max-autotune-no-cudagraphs")
fp4PyTorch reference implementation of NVFP4 block-scaled GEMV.

Kernel source

submission2.py59 lines
import torch
from task import input_t, output_t

torch._dynamo.config.cache_size_limit = 32

# Convert all scale factors to blocked formats

@torch.compile(dynamic=False, fullgraph=True)
def to_blocked_3d(input_matrix):
    # input_matrix is rows x cols x l
    rows, cols, l = input_matrix.shape

    data = input_matrix.permute(2, 0, 1)

    return data.view(l, rows // 128, 128, cols // 4, 4) \
        .transpose(2, 3) \
        .reshape(l, -1, 4, 32, 4) \
        .transpose(2, 3) \
        .flatten(1)

@torch.compile(dynamic=False, mode="max-autotune-no-cudagraphs")
def batched_gemv_impl(a_ref, b_ref, c_ref, sfa, sfb):
    _, _, l = b_ref.shape

    sfa_blocked = to_blocked_3d(sfa)
    sfb_blocked = to_blocked_3d(sfb)

    for l_idx in range(l):
        c_ref[:, 0, l_idx] = torch._scaled_mm(
            a_ref[..., l_idx],
            b_ref[..., l_idx].t(),
            sfa_blocked[l_idx, ...],
            sfb_blocked[l_idx, ...],
            bias=None,
            out_dtype=torch.float16,
        )[:, 0]

    return c_ref

def custom_kernel(
    data: input_t,
) -> output_t:
    """
    PyTorch reference implementation of NVFP4 block-scaled GEMV.
    """
    a_ref, b_ref, sfa, sfb, _, _, c_ref = data

    # a_ref is [m, k//2, l]
    # b_ref is [n, k//2, l], n=1 padded to n=128
    # c_ref is [m, 1, l]
    return batched_gemv_impl(
        a_ref,
        b_ref,
        c_ref,
        sfa,
        sfb,
    )

scrolls · 59 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 69558.

⋯ 2 unchanged lines
torch._dynamo.config.cache_size_limit = 32
- # Helper function to convert scale factor tensor to blocked format
- @torch.compile(dynamic=False, mode="reduce-overhead", fullgraph=True)
- def to_blocked(input_matrix):
- rows, cols = input_matrix.shape
- blocks = input_matrix.view(rows // 128, 128, cols // 4, 4).permute(0, 2, 1, 3)
- #rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
- #return rearranged.flatten()
- return blocks.reshape(-1, 4, 32, 4).transpose(1, 2).flatten()
+ # Convert all scale factors to blocked formats
- @torch.compile(dynamic=False, mode="reduce-overhead", fullgraph=True)
- def _inner(a_ref, b_ref, sfa_ref, sfb_ref, l_idx):
- scale_a = to_blocked(sfa_ref[..., l_idx])
- scale_b = to_blocked(sfb_ref[..., l_idx])
+ @torch.compile(dynamic=False, fullgraph=True)
+ def to_blocked_3d(input_matrix):
+ # input_matrix is rows x cols x l
+ rows, cols, l = input_matrix.shape
- # (m, k) @ (n, k).T -> (m, n)
- return torch._scaled_mm(
- a_ref[..., l_idx],
- b_ref[..., l_idx].transpose(0, 1),
- scale_a,
- scale_b,
- bias=None,
- out_dtype=torch.float16,
- )[:, 0]
+ data = input_matrix.permute(2, 0, 1)
+ return data.view(l, rows // 128, 128, cols // 4, 4) \
+ .transpose(2, 3) \
+ .reshape(l, -1, 4, 32, 4) \
+ .transpose(2, 3) \
+ .flatten(1)
- @torch.compile(mode="max-autotune-no-cudagraphs")
- def batched_gemv_impl(a_ref, b_ref, c_ref, sfa_ref, sfb_ref):
+ @torch.compile(dynamic=False, mode="max-autotune-no-cudagraphs")
+ def batched_gemv_impl(a_ref, b_ref, c_ref, sfa, sfb):
_, _, l = b_ref.shape
+ sfa_blocked = to_blocked_3d(sfa)
+ sfb_blocked = to_blocked_3d(sfb)
+
for l_idx in range(l):
- c_ref[:, 0, l_idx] = _inner(a_ref, b_ref, sfa_ref, sfb_ref, l_idx)
+ c_ref[:, 0, l_idx] = torch._scaled_mm(
+ a_ref[..., l_idx],
+ b_ref[..., l_idx].t(),
+ sfa_blocked[l_idx, ...],
+ sfb_blocked[l_idx, ...],
+ bias=None,
+ out_dtype=torch.float16,
+ )[:, 0]
return c_ref
⋯ 3 unchanged lines
"""
PyTorch reference implementation of NVFP4 block-scaled GEMV.
"""
- a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, _, _, c_ref = data
+ a_ref, b_ref, sfa, sfb, _, _, c_ref = data
- sfa_ref_gpu = sfa_ref_cpu.cuda()
- sfb_ref_gpu = sfb_ref_cpu.cuda()
-
+ # a_ref is [m, k//2, l]
+ # b_ref is [n, k//2, l], n=1 padded to n=128
+ # c_ref is [m, 1, l]
return batched_gemv_impl(
a_ref,
b_ref,
c_ref,
- sfa_ref_gpu,
- sfb_ref_gpu,
+ sfa,
+ sfb,
)
scrolls · 85 diff lines total

Best evidence level for this revision: reported

JSON