Skip to content
KernelIndex
Search⌘K

submission 114558

JB Gage · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 49 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-114558?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVFP4 GEMVsuite of 3 cases
NVIDIA B200
65.1µs
#316 of 678
2025-11-29

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:0f9c6ab96100f0990746f3a5b4800485cf098721075224a47a99853eba7e6b21
license declaredunknown
license concludedunknown
authorsJB Gage
imported2026-08-15

Kernel source

submission.py49 lines
import torch
from typing import TypeVar

input_t = TypeVar("input_t", bound=tuple)
output_t = TypeVar("output_t", bound=torch.Tensor)

def custom_kernel(data: input_t) -> output_t:
    """
    OPTIMIZED: Use pre-permuted scales directly!
    
    Avoids calling to_blocked() and CPU→GPU transfer in the loop.
    Pre-permuted format just needs: permute(2,4,0,1,3).flatten()
    """
    a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, sfa_permuted, sfb_permuted, c_ref = data
    m, k_packed, l = a_ref.shape
    
    # Extract scales from pre-permuted format
    # sfa_permuted is already on GPU and in (32,4,rest_m,4,rest_k,L) format
    # We just need to permute and flatten per batch
    
    scales_a = []
    scales_b = []
    
    for l_idx in range(l):
        # Extract batch slice
        sfa_slice = sfa_permuted[:, :, :, :, :, l_idx]  # (32, 4, rest_m, 4, rest_k)
        sfb_slice = sfb_permuted[:, :, :, :, :, l_idx]  # (32, 4, 1, 4, rest_k)
        
        # Apply the magic permutation: (2, 4, 0, 1, 3)
        # This reorders to match what to_blocked() produces
        scale_a = sfa_slice.permute(2, 4, 0, 1, 3).flatten()
        scale_b = sfb_slice.permute(2, 4, 0, 1, 3).flatten()
        
        scales_a.append(scale_a)
        scales_b.append(scale_b)
    
    # Main compute loop
    for l_idx in range(l):
        res = torch._scaled_mm(
            a_ref[:, :, l_idx],
            b_ref[:, :, l_idx].transpose(0, 1),
            scales_a[l_idx],
            scales_b[l_idx],
            bias=None,
            out_dtype=torch.float16,
        )
        c_ref[:, 0, l_idx] = res[:, 0]
    
    return c_ref
scrolls · 49 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 114152.

⋯ 3 unchanged lines
input_t = TypeVar("input_t", bound=tuple)
output_t = TypeVar("output_t", bound=torch.Tensor)
- def ceil_div(a, b):
- return (a + b - 1) // b
-
- def to_blocked(input_matrix):
- """Convert scale factor tensor to blocked format required by torch._scaled_mm"""
- rows, cols = input_matrix.shape
- n_row_blocks = ceil_div(rows, 128)
- n_col_blocks = ceil_div(cols, 4)
+ def custom_kernel(data: input_t) -> output_t:
+ """
+ OPTIMIZED: Use pre-permuted scales directly!
- padded = input_matrix
- blocks = padded.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
- rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
+ Avoids calling to_blocked() and CPU→GPU transfer in the loop.
+ Pre-permuted format just needs: permute(2,4,0,1,3).flatten()
+ """
+ a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, sfa_permuted, sfb_permuted, c_ref = data
+ m, k_packed, l = a_ref.shape
- return rearranged.flatten()
-
- def custom_kernel(data: input_t) -> output_t:
-
- a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, _, _, c_ref = data
- _, _, l = c_ref.shape
+ # Extract scales from pre-permuted format
+ # sfa_permuted is already on GPU and in (32,4,rest_m,4,rest_k,L) format
+ # We just need to permute and flatten per batch
- # Pre-convert all scales to blocked format (CPU)
- # This minimizes overhead in the main compute loop
- scales_a = [to_blocked(sfa_ref_cpu[:, :, l_idx]) for l_idx in range(l)]
- scales_b = [to_blocked(sfb_ref_cpu[:, :, l_idx]) for l_idx in range(l)]
+ scales_a = []
+ scales_b = []
- # Batch transfer to GPU
- scales_a_gpu = [s.cuda() for s in scales_a]
- scales_b_gpu = [s.cuda() for s in scales_b]
+ for l_idx in range(l):
+ # Extract batch slice
+ sfa_slice = sfa_permuted[:, :, :, :, :, l_idx] # (32, 4, rest_m, 4, rest_k)
+ sfb_slice = sfb_permuted[:, :, :, :, :, l_idx] # (32, 4, 1, 4, rest_k)
+
+ # Apply the magic permutation: (2, 4, 0, 1, 3)
+ # This reorders to match what to_blocked() produces
+ scale_a = sfa_slice.permute(2, 4, 0, 1, 3).flatten()
+ scale_b = sfb_slice.permute(2, 4, 0, 1, 3).flatten()
+
+ scales_a.append(scale_a)
+ scales_b.append(scale_b)
- # Process each batch using cuBLAS (fastest available FP4 GEMV)
+ # Main compute loop
for l_idx in range(l):
res = torch._scaled_mm(
a_ref[:, :, l_idx],
b_ref[:, :, l_idx].transpose(0, 1),
- scales_a_gpu[l_idx],
- scales_b_gpu[l_idx],
+ scales_a[l_idx],
+ scales_b[l_idx],
bias=None,
out_dtype=torch.float16,
)
scrolls · 71 diff lines total

Best evidence level for this revision: reported

JSON