Skip to content
KernelIndex
Search⌘K

submission 96789

Jasmin Bogatinovski · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 127 lines, June 9 Researcher Reciprocity License v1.0.

fourth_sub.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-96789?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVFP4 GEMVsuite of 3 cases
NVIDIA B200
64.5µs
#311 of 678
2025-11-22

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:7d04c63e4a3d6cd4713f97bffd38870e5b654217cd2c39dc1be67a63da6101fd
license declaredunknown
license concludedunknown
authorsJasmin Bogatinovski
imported2026-08-26

Kernel source

fourth_sub.py127 lines
import torch
from task import input_t, output_t

sf_vec_size = 16


def ceil_div(a, b):
    return (a + b - 1) // b


def to_blocked(input_matrix: torch.Tensor) -> torch.Tensor:
    """
    Optimized CPU-side to_blocked function.
    """
    rows, cols = input_matrix.shape
    n_row_blocks = ceil_div(rows, 128)
    n_col_blocks = ceil_div(cols, 4)

    if rows != n_row_blocks * 128 or cols != n_col_blocks * 4:
        padded = torch.zeros((n_row_blocks * 128, n_col_blocks * 4), dtype=input_matrix.dtype, device=input_matrix.device)
        padded[:rows, :cols] = input_matrix
    else:
        padded = input_matrix

    blocks = padded.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
    rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
    return rearranged.flatten()


def _to_blocked_gpu(input_matrix: torch.Tensor) -> torch.Tensor:
    """
    Optimized GPU-side version of to_blocked.
    """
    x = input_matrix
    rows, cols = x.shape

    n_row_blocks = ceil_div(rows, 128)
    n_col_blocks = ceil_div(cols, 4)

    target_rows = n_row_blocks * 128
    target_cols = n_col_blocks * 4

    if rows != target_rows or cols != target_cols:
        padded = torch.zeros((target_rows, target_cols), dtype=x.dtype, device=x.device)
        padded[:rows, :cols] = x
        x = padded

    blocks = x.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
    rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
    return rearranged.flatten()


def _blocked_from_permuted_slice(sfp_slice: torch.Tensor) -> torch.Tensor:
    """
    Optimized fast path for permuted layout (GPU side).
    """
    if sfp_slice.dim() != 5:
        raise RuntimeError(f"_blocked_from_permuted_slice expects 5D tensor, got {sfp_slice.shape}")

    x = sfp_slice
    x = x.permute(2, 4, 0, 1, 3)  # Optimized permutation for better GPU access pattern
    mm, kk, _, _, _ = x.shape
    x = x.reshape(mm * kk, 32, 16)
    return x.flatten()


def custom_kernel(data: input_t) -> output_t:
    """
    Optimized custom kernel for matrix-vector multiplication using GPU.
    """
    if len(data) == 7:
        a, b, sfa_ref_cpu, sfb_ref_cpu, sfa_permuted, sfb_permuted, c = data
    elif len(data) == 5:
        a, b, sfa_ref_cpu, sfb_ref_cpu, c = data
        sfa_permuted = None
        sfb_permuted = None
    else:
        raise RuntimeError(f"Expected tuple length 5 or 7, got {len(data)}")

    if a.dim() != 3 or b.dim() != 3 or sfa_ref_cpu.dim() != 3 or sfb_ref_cpu.dim() != 3 or c.dim() != 3:
        raise RuntimeError("Unexpected tensor ranks for a/b/sfa/sfb/c")

    m, k, l = a.shape[0], a.shape[1], a.shape[2]
    if c.shape != (m, 1, l):
        raise RuntimeError(f"Unexpected c shape {c.shape}, expected ({m},1,{l})")

    device = a.device

    blocked_sfa = [None] * l
    blocked_sfb = [None] * l

    if sfa_permuted is not None and sfb_permuted is not None:
        for li in range(l):
            sfa_slice = sfa_permuted[..., li]  
            sfb_slice = sfb_permuted[..., li]
            blocked_sfa[li] = _blocked_from_permuted_slice(sfa_slice)
            blocked_sfb[li] = _blocked_from_permuted_slice(sfb_slice)

            if blocked_sfa[li].device != device:
                blocked_sfa[li] = blocked_sfa[li].to(device)
            if blocked_sfb[li].device != device:
                blocked_sfb[li] = blocked_sfb[li].to(device)
    else:
        for li in range(l):
            sfa_slice_cpu = sfa_ref_cpu[:, :, li]
            sfb_slice_cpu = sfb_ref_cpu[:, :, li]
            
            blocked_sfa[li] = _to_blocked_gpu(sfa_slice_cpu)
            blocked_sfb[li] = _to_blocked_gpu(sfb_slice_cpu)

    for li in range(l):
        a_mat = a[:, :, li]
        b_slice = b[:, :, li].transpose(0, 1)

        res = torch._scaled_mm(
            a_mat,
            b_slice,
            blocked_sfa[li],
            blocked_sfb[li],
            bias=None,
            out_dtype=torch.float16,
        )

        c[:, 0, li] = res[:, 0]

    return c
scrolls · 127 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON