submission 96789
Jasmin Bogatinovski · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 127 lines, June 9 Researcher Reciprocity License v1.0.
fourth_sub.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-96789?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:7d04c63e4a3d6cd4713f97bffd38870e5b654217cd2c39dc1be67a63da6101fd
license declaredunknown
license concludedunknown
authorsJasmin Bogatinovski
imported2026-08-26
Kernel source
fourth_sub.py127 lines
import torch
from task import input_t, output_t
sf_vec_size = 16
def ceil_div(a, b):
return (a + b - 1) // b
def to_blocked(input_matrix: torch.Tensor) -> torch.Tensor:
"""
Optimized CPU-side to_blocked function.
"""
rows, cols = input_matrix.shape
n_row_blocks = ceil_div(rows, 128)
n_col_blocks = ceil_div(cols, 4)
if rows != n_row_blocks * 128 or cols != n_col_blocks * 4:
padded = torch.zeros((n_row_blocks * 128, n_col_blocks * 4), dtype=input_matrix.dtype, device=input_matrix.device)
padded[:rows, :cols] = input_matrix
else:
padded = input_matrix
blocks = padded.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
return rearranged.flatten()
def _to_blocked_gpu(input_matrix: torch.Tensor) -> torch.Tensor:
"""
Optimized GPU-side version of to_blocked.
"""
x = input_matrix
rows, cols = x.shape
n_row_blocks = ceil_div(rows, 128)
n_col_blocks = ceil_div(cols, 4)
target_rows = n_row_blocks * 128
target_cols = n_col_blocks * 4
if rows != target_rows or cols != target_cols:
padded = torch.zeros((target_rows, target_cols), dtype=x.dtype, device=x.device)
padded[:rows, :cols] = x
x = padded
blocks = x.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
return rearranged.flatten()
def _blocked_from_permuted_slice(sfp_slice: torch.Tensor) -> torch.Tensor:
"""
Optimized fast path for permuted layout (GPU side).
"""
if sfp_slice.dim() != 5:
raise RuntimeError(f"_blocked_from_permuted_slice expects 5D tensor, got {sfp_slice.shape}")
x = sfp_slice
x = x.permute(2, 4, 0, 1, 3) # Optimized permutation for better GPU access pattern
mm, kk, _, _, _ = x.shape
x = x.reshape(mm * kk, 32, 16)
return x.flatten()
def custom_kernel(data: input_t) -> output_t:
"""
Optimized custom kernel for matrix-vector multiplication using GPU.
"""
if len(data) == 7:
a, b, sfa_ref_cpu, sfb_ref_cpu, sfa_permuted, sfb_permuted, c = data
elif len(data) == 5:
a, b, sfa_ref_cpu, sfb_ref_cpu, c = data
sfa_permuted = None
sfb_permuted = None
else:
raise RuntimeError(f"Expected tuple length 5 or 7, got {len(data)}")
if a.dim() != 3 or b.dim() != 3 or sfa_ref_cpu.dim() != 3 or sfb_ref_cpu.dim() != 3 or c.dim() != 3:
raise RuntimeError("Unexpected tensor ranks for a/b/sfa/sfb/c")
m, k, l = a.shape[0], a.shape[1], a.shape[2]
if c.shape != (m, 1, l):
raise RuntimeError(f"Unexpected c shape {c.shape}, expected ({m},1,{l})")
device = a.device
blocked_sfa = [None] * l
blocked_sfb = [None] * l
if sfa_permuted is not None and sfb_permuted is not None:
for li in range(l):
sfa_slice = sfa_permuted[..., li]
sfb_slice = sfb_permuted[..., li]
blocked_sfa[li] = _blocked_from_permuted_slice(sfa_slice)
blocked_sfb[li] = _blocked_from_permuted_slice(sfb_slice)
if blocked_sfa[li].device != device:
blocked_sfa[li] = blocked_sfa[li].to(device)
if blocked_sfb[li].device != device:
blocked_sfb[li] = blocked_sfb[li].to(device)
else:
for li in range(l):
sfa_slice_cpu = sfa_ref_cpu[:, :, li]
sfb_slice_cpu = sfb_ref_cpu[:, :, li]
blocked_sfa[li] = _to_blocked_gpu(sfa_slice_cpu)
blocked_sfb[li] = _to_blocked_gpu(sfb_slice_cpu)
for li in range(l):
a_mat = a[:, :, li]
b_slice = b[:, :, li].transpose(0, 1)
res = torch._scaled_mm(
a_mat,
b_slice,
blocked_sfa[li],
blocked_sfb[li],
bias=None,
out_dtype=torch.float16,
)
c[:, 0, li] = res[:, 0]
return c
scrolls · 127 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON