Skip to content
KernelIndex
Search⌘K

submission 116691

qinking · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 411 lines, June 9 Researcher Reciprocity License v1.0.

p1_v2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-116691?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVFP4 GEMVsuite of 3 cases
NVIDIA B200
28.1µs
#136 of 678
2025-11-30

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:213865fef4837af35031d115fa411fda2fd87d98f2c9b9ad0c348fc137a88ddc
license declaredunknown
license concludedunknown
authorsqinking
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

fp4ab_dtype = cutlass.Float4E2M1FN # FP4 data type for A and B

Kernel source

p1_v2.py411 lines
#!POPCORN leaderboard nvfp4_gemv

from typing import Callable
import torch
from task import input_t, output_t

import cutlass
import cutlass.cute as cute
from cutlass.cute.runtime import make_ptr
import cutlass.utils.blockscaled_layout as blockscaled_utils

# Kernel configuration parameters
# mma_tiler_mnk = (16, 1, 256)  # Tile sizes for M, N, K dimensions
ab_dtype = cutlass.Float4E2M1FN  # FP4 data type for A and B
sf_dtype = cutlass.Float8E4M3FN  # FP8 data type for scale factors
c_dtype = cutlass.Float16  # FP16 output type
sf_vec_size = 16  # Scale factor block size (16 elements share one scale)

threadsPerRow = 16
threadsPerCol = 8
elementsPerAccess = 128
scalesPerThread = elementsPerAccess // sf_vec_size

my_res = {}
my_cc = {}


# Helper function for ceiling division
def ceil_div(a, b):
    return (a + b - 1) // b


def to_blocked(input_matrix):
    rows, cols = input_matrix.shape

    # Please ensure rows and cols are multiples of 128 and 4 respectively
    n_row_blocks = ceil_div(rows, 128)
    n_col_blocks = ceil_div(cols, 4)

    padded = input_matrix
    blocks = padded.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
    rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)

    return rearranged.flatten()


def custom_kernel2(
    data: input_t,
) -> output_t:
    """
    PyTorch reference implementation of NVFP4 block-scaled GEMV.
    """
    a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, _, _, c_ref = data

    # Get dimensions from MxNxL layout
    _, _, l = c_ref.shape

    # Call torch._scaled_mm to compute the GEMV result
    for l_idx in range(l):
        # Convert the scale factor tensor to blocked format
        scale_a = to_blocked(sfa_ref_cpu[:, :, l_idx])
        scale_b = to_blocked(sfb_ref_cpu[:, :, l_idx])
        # (m, k) @ (n, k).T -> (m, n)
        res = torch._scaled_mm(
            a_ref[:, :, l_idx],
            b_ref[:, :, l_idx].transpose(0, 1),
            scale_a.cuda(),
            scale_b.cuda(),
            bias=None,
            out_dtype=torch.float16,
        )
        c_ref[:, 0, l_idx] = res[:, 0]
    return c_ref


def _warp_reduce_8way(val: cutlass.Float32, op: Callable) -> cutlass.Float32:
    val = op(
        val, cute.arch.shuffle_sync_bfly(val, offset=1, mask=-1, mask_and_clamp=31)
    )
    val = op(
        val, cute.arch.shuffle_sync_bfly(val, offset=2, mask=-1, mask_and_clamp=31)
    )
    val = op(
        val, cute.arch.shuffle_sync_bfly(val, offset=4, mask=-1, mask_and_clamp=31)
    )
    val = op(
        val, cute.arch.shuffle_sync_bfly(val, offset=8, mask=-1, mask_and_clamp=31)
    )
    # val = op(val, cute.arch.shuffle_sync_bfly(val, offset=16, mask=-1, mask_and_clamp=31))
    return val


def _warp_reduce_sum_8way(val: cutlass.Float32) -> cutlass.Float32:
    return _warp_reduce_8way(val, lambda x, y: x + y)


# The CuTe reference implementation for NVFP4 block-scaled GEMV
@cute.kernel
def kernel(
    mA_mkl: cute.Tensor,
    mB_nkl: cute.Tensor,
    mBB_nkl: cute.Tensor,
    mSFA_mkl: cute.Tensor,
    mSFB_nkl: cute.Tensor,
    mC_mnl: cute.Tensor,
):
    # Get CUDA block and thread indices
    bidx, bidy, bidz = cute.arch.block_idx()
    tidx, tidy, _ = cute.arch.thread_idx()
    bdx, bdy, _ = cute.arch.block_dim()
    tid = tidx + tidy * bdx

    # Extract the local tile for input matrix A (shape: [block_M, block_K, rest_M, rest_K, rest_L])
    gA_mkl = cute.local_tile(
        mA_mkl, (threadsPerCol, (elementsPerAccess, threadsPerRow)), (None, None, None)
    )
    # # Extract the local tile for scale factor tensor for A (same shape as gA_mkl)
    # # Here, block_M = (32, 4); block_K = (16, 4)
    gSFA_mkl = cute.local_tile(
        mSFA_mkl,
        (threadsPerCol, ((16, scalesPerThread), threadsPerRow)),
        (None, None, None),
    )
    # # Extract the local tile for input matrix B (shape: [block_N, block_K, rest_N, rest_K, rest_L])
    gB_nkl = cute.local_tile(
        mB_nkl,
        cute.slice_(
            (threadsPerCol, 1, (elementsPerAccess, threadsPerRow)), (0, None, None)
        ),
        (None, None, None),
    )
    # Extract the local tile for scale factor tensor for B (same shape as gB_nkl)
    gSFB_nkl = cute.local_tile(
        mSFB_nkl,
        cute.slice_(
            (threadsPerCol, 1, ((16, scalesPerThread), threadsPerRow)), (0, None, None)
        ),
        (None, None, None),
    )
    # Extract the local tile for output matrix C (shape: [block_M, block_N, rest_M, rest_N, rest_L])
    gC_mnl = cute.local_tile(
        mC_mnl,
        cute.slice_(
            (threadsPerCol, 1, (elementsPerAccess, threadsPerRow)), (None, None, 0)
        ),
        (None, None, None),
    )

    local_sum = cutlass.Float32(0.0)
    k_tile_cnt = cute.size(gA_mkl, mode=[3])
    N_step = mA_mkl.layout[1].shape[1]

    for k_tile in cutlass.range(k_tile_cnt):
        if (tidx + k_tile * threadsPerRow) < N_step:
            tAgA = gA_mkl[tidy, (None, tidx), bidx, k_tile, bidz]
            tBgB = gB_nkl[0, (None, tidx), bidy, k_tile, bidz]
            tAgSFA = gSFA_mkl[tidy, (None, tidx), bidx, k_tile, bidz]
            tBgSFB = gSFB_nkl[0, (None, tidx), bidy, k_tile, bidz]

            tArA = cute.make_rmem_tensor_like(tAgA, cutlass.Float16)
            # tBrB = cute.make_rmem_tensor_like(tBgB, cutlass.Float16)
            tArSFA = cute.make_rmem_tensor_like(tAgSFA, cutlass.Float32)
            # tBrSFB = cute.make_rmem_tensor_like(tBgSFB, cutlass.Float16)

            # Load NVFP4 or FP8 values from global memory
            a_val_nvfp4 = tAgA.load()
            b_val_nvfp4 = tBgB.load()
            sfa_val_fp8 = tAgSFA.load()
            sfb_val_fp8 = tBgSFB.load()

            # Convert loaded values to float32 for computation (FFMA)
            a_val = a_val_nvfp4.to(cutlass.Float16)
            b_val = b_val_nvfp4.to(cutlass.Float16)
            sfa_val = sfa_val_fp8.to(cutlass.Float32)
            sfb_val = sfb_val_fp8.to(cutlass.Float32)

            # Store the converted values to RMEM CuTe tensors
            tArA.store(a_val * b_val)
            # tBrB.store(b_val)
            tArSFA.store(sfa_val * sfb_val)
            # tBrSFB.store(sfb_val)

            # Iterate over SF vector tiles and compute the scale&matmul accumulation
            for k in cutlass.range_constexpr(elementsPerAccess // 16):
                offset = k * 16

                group_sum = cutlass.Float16(0.0)
                for i in cutlass.range_constexpr(16):
                    idxx = i + offset
                    group_sum += tArA[idxx]

                # group_sum *= tArA[offset] * tBrB[offset]
                local_sum += group_sum * tArSFA[offset]
                # * tArSFA[i] * tBrB[i] * tBrSFB[i]

    row_sum = _warp_reduce_sum_8way(local_sum)

    tCgC = gC_mnl[tidy, None, bidx, bidy, bidz]
    tCgC = cute.make_tensor(tCgC.iterator, 1)
    res = cute.zeros_like(tCgC, cutlass.Float32)

    res += row_sum

    # cute.printf(row_sum)

    if tidx == 0:
        tCgC.store(res.to(cutlass.Float16))
    return


@cute.jit
def my_kernel(
    a_ptr: cute.Pointer,
    b_ptr: cute.Pointer,
    sfa_ptr: cute.Pointer,
    sfb_ptr: cute.Pointer,
    c_ptr: cute.Pointer,
    problem_size: tuple,
):
    """
    Host-side JIT function to prepare tensors and launch GPU kernel.
    """
    m, _, k, l = problem_size
    # Create CuTe Tensor via pointer and problem size.
    a_tensor = cute.make_tensor(
        a_ptr,
        cute.make_layout(
            (
                m,
                (
                    elementsPerAccess,
                    cute.ceil_div(cute.assume(k, 32), elementsPerAccess),
                ),
                l,
            ),
            stride=(cute.assume(k, 32), (1, elementsPerAccess), cute.assume(m * k, 32)),
        ),
    )
    sfa_tensor = cute.make_tensor(
        sfa_ptr,
        cute.make_layout(
            (
                m,
                (
                    (16, scalesPerThread),
                    cute.ceil_div(cute.assume(k, 32), elementsPerAccess),
                ),
                l,
            ),
            stride=(
                cute.ceil_div(cute.assume(k, 32), 16),
                ((0, 1), scalesPerThread),
                cute.ceil_div(cute.assume(m * k, 32), 16),
            ),
        ),
    )

    # We use n=128 to create the torch tensor to do fp4 computation via torch._scaled_mm
    # then copy torch tensor to cute tensor for cute customize kernel computation
    # therefore we need to ensure b_tensor has the right stride with this 128 padded size on n.
    n_padded_128 = 128
    b_tensor = cute.make_tensor(
        b_ptr,
        cute.make_layout(
            (
                n_padded_128,
                (
                    elementsPerAccess,
                    cute.ceil_div(cute.assume(k, 32), elementsPerAccess),
                ),
                l,
            ),
            stride=(
                cute.assume(k, 32),
                (1, elementsPerAccess),
                cute.assume(n_padded_128 * k, 32),
            ),
        ),
    )
    sfb_tensor = cute.make_tensor(
        sfb_ptr,
        cute.make_layout(
            (
                n_padded_128,
                (
                    (16, scalesPerThread),
                    cute.ceil_div(cute.assume(k, 32), elementsPerAccess),
                ),
                l,
            ),
            stride=(
                cute.ceil_div(cute.assume(k, 32), 16),
                ((0, 1), scalesPerThread),
                cute.ceil_div(cute.assume(n_padded_128 * k, 32), 16),
            ),
        ),
    )
    bb_tensor = cute.make_tensor(
        b_ptr,
        cute.make_layout(
            (n_padded_128, (8, cute.ceil_div(cute.assume(k, 32), 8)), l),
            stride=(cute.assume(k, 32), (1, 8), cute.assume(n_padded_128 * k, 32)),
        ),
    )
    c_tensor = cute.make_tensor(
        c_ptr, cute.make_layout((cute.assume(m, 32), 1, l), stride=(1, 1, m))
    )

    grid = (
        cute.ceil_div(c_tensor.shape[0], threadsPerCol),
        1,
        c_tensor.shape[2],
    )

    # Launch the CUDA kernel
    kernel(a_tensor, b_tensor, bb_tensor, sfa_tensor, sfb_tensor, c_tensor).launch(
        grid=grid,
        block=[threadsPerRow, threadsPerCol, 1],
        cluster=(1, 1, 1),
    )
    return


# Global cache for compiled kernel
_compiled_kernel_cache = None


# This function is used to compile the kernel once and cache it and then allow users to
# run the kernel multiple times to get more accurate timing results.
def compile_kernel():
    """
    Compile the kernel once and cache it.
    This should be called before any timing measurements.

    Returns:
        The compiled kernel function
    """
    global _compiled_kernel_cache

    if _compiled_kernel_cache is not None:
        return _compiled_kernel_cache

    # Create CuTe pointers for A/B/C/SFA/SFB via torch tensor data pointer
    a_ptr = make_ptr(ab_dtype, 0, cute.AddressSpace.gmem, assumed_align=16)
    b_ptr = make_ptr(ab_dtype, 0, cute.AddressSpace.gmem, assumed_align=16)
    c_ptr = make_ptr(c_dtype, 0, cute.AddressSpace.gmem, assumed_align=16)
    sfa_ptr = make_ptr(sf_dtype, 0, cute.AddressSpace.gmem, assumed_align=32)
    sfb_ptr = make_ptr(sf_dtype, 0, cute.AddressSpace.gmem, assumed_align=32)

    # Compile the kernel
    _compiled_kernel_cache = cute.compile(
        my_kernel, a_ptr, b_ptr, sfa_ptr, sfb_ptr, c_ptr, (0, 0, 0, 0)
    )

    return _compiled_kernel_cache


def custom_kernel(data: input_t) -> output_t:
    """
    Execute the block-scaled GEMV kernel.

    This is the main entry point called by the evaluation framework.
    It converts PyTorch tensors to CuTe tensors, launches the kernel,
    and returns the result.

    Args:
        data: Tuple of (a, b, sfa_cpu, sfb_cpu, c) PyTorch tensors
            a: [m, k, l] - Input matrix in float4e2m1fn
            b: [1, k, l] - Input vector in float4e2m1fn
            sfa_cpu: [m, k, l] - Scale factors in float8_e4m3fn
            sfb_cpu: [1, k, l] - Scale factors in float8_e4m3fn
            sfa_permuted: [32, 4, rest_m, 4, rest_k, l] - Scale factors in float8_e4m3fn
            sfb_permuted: [32, 4, rest_n, 4, rest_k, l] - Scale factors in float8_e4m3fn
            c: [m, 1, l] - Output vector in float16

    Returns:
        Output tensor c with computed GEMV results
    """
    a, b, sfa, sfb, sfa_permuted, sfb_permuted, c = data

    # Ensure kernel is compiled (will use cached version if available)
    # To avoid the compilation overhead, we compile the kernel once and cache it.
    compiled_func = compile_kernel()

    # Get dimensions from MxKxL layout
    m, k, l = a.shape
    # Torch use e2m1_x2 data type, thus k is halved
    # if l not in my_res:
    #     my_res[l] = custom_kernel2(data)
    #     my_cc[l] = c.clone()

    k = k * 2
    # GEMV N dimension is always 1
    n = 1

    # Create CuTe pointers for A/B/C/SFA/SFB via torch tensor data pointer
    a_ptr = make_ptr(ab_dtype, a.data_ptr(), cute.AddressSpace.gmem, assumed_align=16)
    b_ptr = make_ptr(ab_dtype, b.data_ptr(), cute.AddressSpace.gmem, assumed_align=16)
    c_ptr = make_ptr(c_dtype, c.data_ptr(), cute.AddressSpace.gmem, assumed_align=16)
    sfa_ptr = make_ptr(
        sf_dtype, sfa.data_ptr(), cute.AddressSpace.gmem, assumed_align=32
    )
    sfb_ptr = make_ptr(
        sf_dtype, sfb.data_ptr(), cute.AddressSpace.gmem, assumed_align=32
    )

    # Execute the compiled kernel
    compiled_func(a_ptr, b_ptr, sfa_ptr, sfb_ptr, c_ptr, (m, n, k, l))

    return c
scrolls · 411 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON