submission 71049
mdouglas · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 59 lines, June 9 Researcher Reciprocity License v1.0.
submission2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-71049?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:3aba0c7b9c36b4f187e4112f6f48975b0c7c930a2e07fc787a39638287f875c7
license declaredunknown
license concludedunknown
authorsmdouglas
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
Kernel source
submission2.py59 lines
import torch
from task import input_t, output_t
torch._dynamo.config.cache_size_limit = 32
# Convert all scale factors to blocked formats
@torch.compile(dynamic=False, fullgraph=True)
def to_blocked_3d(input_matrix):
# input_matrix is rows x cols x l
rows, cols, l = input_matrix.shape
data = input_matrix.permute(2, 0, 1)
return data.view(l, rows // 128, 128, cols // 4, 4) \
.transpose(2, 3) \
.reshape(l, -1, 4, 32, 4) \
.transpose(2, 3) \
.flatten(1)
@torch.compile(dynamic=False, mode="max-autotune-no-cudagraphs")
def batched_gemv_impl(a_ref, b_ref, c_ref, sfa, sfb):
_, _, l = b_ref.shape
sfa_blocked = to_blocked_3d(sfa)
sfb_blocked = to_blocked_3d(sfb)
for l_idx in range(l):
c_ref[:, 0, l_idx] = torch._scaled_mm(
a_ref[..., l_idx],
b_ref[..., l_idx].t(),
sfa_blocked[l_idx, ...],
sfb_blocked[l_idx, ...],
bias=None,
out_dtype=torch.float16,
)[:, 0]
return c_ref
def custom_kernel(
data: input_t,
) -> output_t:
"""
PyTorch reference implementation of NVFP4 block-scaled GEMV.
"""
a_ref, b_ref, sfa, sfb, _, _, c_ref = data
# a_ref is [m, k//2, l]
# b_ref is [n, k//2, l], n=1 padded to n=128
# c_ref is [m, 1, l]
return batched_gemv_impl(
a_ref,
b_ref,
c_ref,
sfa,
sfb,
)
scrolls · 59 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 69558.
⋯ 2 unchanged linestorch._dynamo.config.cache_size_limit = 32- # Helper function to convert scale factor tensor to blocked format- @torch.compile(dynamic=False, mode="reduce-overhead", fullgraph=True)- def to_blocked(input_matrix):- rows, cols = input_matrix.shape- blocks = input_matrix.view(rows // 128, 128, cols // 4, 4).permute(0, 2, 1, 3)- #rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)- #return rearranged.flatten()- return blocks.reshape(-1, 4, 32, 4).transpose(1, 2).flatten()+ # Convert all scale factors to blocked formats- @torch.compile(dynamic=False, mode="reduce-overhead", fullgraph=True)- def _inner(a_ref, b_ref, sfa_ref, sfb_ref, l_idx):- scale_a = to_blocked(sfa_ref[..., l_idx])- scale_b = to_blocked(sfb_ref[..., l_idx])+ @torch.compile(dynamic=False, fullgraph=True)+ def to_blocked_3d(input_matrix):+ # input_matrix is rows x cols x l+ rows, cols, l = input_matrix.shape- # (m, k) @ (n, k).T -> (m, n)- return torch._scaled_mm(- a_ref[..., l_idx],- b_ref[..., l_idx].transpose(0, 1),- scale_a,- scale_b,- bias=None,- out_dtype=torch.float16,- )[:, 0]+ data = input_matrix.permute(2, 0, 1)+ return data.view(l, rows // 128, 128, cols // 4, 4) \+ .transpose(2, 3) \+ .reshape(l, -1, 4, 32, 4) \+ .transpose(2, 3) \+ .flatten(1)- @torch.compile(mode="max-autotune-no-cudagraphs")- def batched_gemv_impl(a_ref, b_ref, c_ref, sfa_ref, sfb_ref):+ @torch.compile(dynamic=False, mode="max-autotune-no-cudagraphs")+ def batched_gemv_impl(a_ref, b_ref, c_ref, sfa, sfb):_, _, l = b_ref.shape+ sfa_blocked = to_blocked_3d(sfa)+ sfb_blocked = to_blocked_3d(sfb)+for l_idx in range(l):- c_ref[:, 0, l_idx] = _inner(a_ref, b_ref, sfa_ref, sfb_ref, l_idx)+ c_ref[:, 0, l_idx] = torch._scaled_mm(+ a_ref[..., l_idx],+ b_ref[..., l_idx].t(),+ sfa_blocked[l_idx, ...],+ sfb_blocked[l_idx, ...],+ bias=None,+ out_dtype=torch.float16,+ )[:, 0]return c_ref⋯ 3 unchanged lines"""PyTorch reference implementation of NVFP4 block-scaled GEMV."""- a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, _, _, c_ref = data+ a_ref, b_ref, sfa, sfb, _, _, c_ref = data- sfa_ref_gpu = sfa_ref_cpu.cuda()- sfb_ref_gpu = sfb_ref_cpu.cuda()-+ # a_ref is [m, k//2, l]+ # b_ref is [n, k//2, l], n=1 padded to n=128+ # c_ref is [m, 1, l]return batched_gemv_impl(a_ref,b_ref,c_ref,- sfa_ref_gpu,- sfb_ref_gpu,+ sfa,+ sfb,)
scrolls · 85 diff lines total
Best evidence level for this revision: reported
JSON