submission 114558
JB Gage · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 49 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-114558?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:0f9c6ab96100f0990746f3a5b4800485cf098721075224a47a99853eba7e6b21
license declaredunknown
license concludedunknown
authorsJB Gage
imported2026-08-15
Kernel source
submission.py49 lines
import torch
from typing import TypeVar
input_t = TypeVar("input_t", bound=tuple)
output_t = TypeVar("output_t", bound=torch.Tensor)
def custom_kernel(data: input_t) -> output_t:
"""
OPTIMIZED: Use pre-permuted scales directly!
Avoids calling to_blocked() and CPU→GPU transfer in the loop.
Pre-permuted format just needs: permute(2,4,0,1,3).flatten()
"""
a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, sfa_permuted, sfb_permuted, c_ref = data
m, k_packed, l = a_ref.shape
# Extract scales from pre-permuted format
# sfa_permuted is already on GPU and in (32,4,rest_m,4,rest_k,L) format
# We just need to permute and flatten per batch
scales_a = []
scales_b = []
for l_idx in range(l):
# Extract batch slice
sfa_slice = sfa_permuted[:, :, :, :, :, l_idx] # (32, 4, rest_m, 4, rest_k)
sfb_slice = sfb_permuted[:, :, :, :, :, l_idx] # (32, 4, 1, 4, rest_k)
# Apply the magic permutation: (2, 4, 0, 1, 3)
# This reorders to match what to_blocked() produces
scale_a = sfa_slice.permute(2, 4, 0, 1, 3).flatten()
scale_b = sfb_slice.permute(2, 4, 0, 1, 3).flatten()
scales_a.append(scale_a)
scales_b.append(scale_b)
# Main compute loop
for l_idx in range(l):
res = torch._scaled_mm(
a_ref[:, :, l_idx],
b_ref[:, :, l_idx].transpose(0, 1),
scales_a[l_idx],
scales_b[l_idx],
bias=None,
out_dtype=torch.float16,
)
c_ref[:, 0, l_idx] = res[:, 0]
return c_refscrolls · 49 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 114152.
⋯ 3 unchanged linesinput_t = TypeVar("input_t", bound=tuple)output_t = TypeVar("output_t", bound=torch.Tensor)- def ceil_div(a, b):- return (a + b - 1) // b-- def to_blocked(input_matrix):- """Convert scale factor tensor to blocked format required by torch._scaled_mm"""- rows, cols = input_matrix.shape- n_row_blocks = ceil_div(rows, 128)- n_col_blocks = ceil_div(cols, 4)+ def custom_kernel(data: input_t) -> output_t:+ """+ OPTIMIZED: Use pre-permuted scales directly!- padded = input_matrix- blocks = padded.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)- rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)+ Avoids calling to_blocked() and CPU→GPU transfer in the loop.+ Pre-permuted format just needs: permute(2,4,0,1,3).flatten()+ """+ a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, sfa_permuted, sfb_permuted, c_ref = data+ m, k_packed, l = a_ref.shape- return rearranged.flatten()-- def custom_kernel(data: input_t) -> output_t:-- a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, _, _, c_ref = data- _, _, l = c_ref.shape+ # Extract scales from pre-permuted format+ # sfa_permuted is already on GPU and in (32,4,rest_m,4,rest_k,L) format+ # We just need to permute and flatten per batch- # Pre-convert all scales to blocked format (CPU)- # This minimizes overhead in the main compute loop- scales_a = [to_blocked(sfa_ref_cpu[:, :, l_idx]) for l_idx in range(l)]- scales_b = [to_blocked(sfb_ref_cpu[:, :, l_idx]) for l_idx in range(l)]+ scales_a = []+ scales_b = []- # Batch transfer to GPU- scales_a_gpu = [s.cuda() for s in scales_a]- scales_b_gpu = [s.cuda() for s in scales_b]+ for l_idx in range(l):+ # Extract batch slice+ sfa_slice = sfa_permuted[:, :, :, :, :, l_idx] # (32, 4, rest_m, 4, rest_k)+ sfb_slice = sfb_permuted[:, :, :, :, :, l_idx] # (32, 4, 1, 4, rest_k)++ # Apply the magic permutation: (2, 4, 0, 1, 3)+ # This reorders to match what to_blocked() produces+ scale_a = sfa_slice.permute(2, 4, 0, 1, 3).flatten()+ scale_b = sfb_slice.permute(2, 4, 0, 1, 3).flatten()++ scales_a.append(scale_a)+ scales_b.append(scale_b)- # Process each batch using cuBLAS (fastest available FP4 GEMV)+ # Main compute loopfor l_idx in range(l):res = torch._scaled_mm(a_ref[:, :, l_idx],b_ref[:, :, l_idx].transpose(0, 1),- scales_a_gpu[l_idx],- scales_b_gpu[l_idx],+ scales_a[l_idx],+ scales_b[l_idx],bias=None,out_dtype=torch.float16,)
scrolls · 71 diff lines total
Best evidence level for this revision: reported
JSON