Skip to content
KernelIndex
Search⌘K

submission 116512

guaguabear · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 395 lines, June 9 Researcher Reciprocity License v1.0.

template_cute_bk_best.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-116512?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVFP4 GEMVsuite of 3 cases
NVIDIA B200
27.9µs
#133 of 678
2025-11-30

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:808f1c9f69d78e693cf01e534e87a6aa73ae6b3c5096668d60c2983ed0872c0d
license declaredunknown
license concludedunknown
authorsguaguabear
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

fp4ab_dtype = cutlass.Float4E2M1FN # FP4 data type for A and B

Kernel source

template_cute_bk_best.py395 lines
from typing import Callable
import torch
from task import input_t, output_t

import cutlass
import cutlass.cute as cute
from cutlass.cute.runtime import make_ptr
import cutlass.utils.blockscaled_layout as blockscaled_utils

# Kernel configuration parameters
# mma_tiler_mnk = (16, 1, 256)  # Tile sizes for M, N, K dimensions
ab_dtype = cutlass.Float4E2M1FN  # FP4 data type for A and B
sf_dtype = cutlass.Float8E4M3FN  # FP8 data type for scale factors
c_dtype = cutlass.Float16  # FP16 output type
sf_vec_size = 16  # Scale factor block size (16 elements share one scale)

threadsPerRow = 16
threadsPerCol = 8
elementsPerAccess = 128
scalesPerThread = elementsPerAccess // sf_vec_size

my_res = {}
my_cc = {}

# Helper function for ceiling division
def ceil_div(a, b):
    return (a + b - 1) // b


def to_blocked(input_matrix):
    rows, cols = input_matrix.shape

    # Please ensure rows and cols are multiples of 128 and 4 respectively
    n_row_blocks = ceil_div(rows, 128)
    n_col_blocks = ceil_div(cols, 4)

    padded = input_matrix
    blocks = padded.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
    rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)

    return rearranged.flatten()

def custom_kernel2(
    data: input_t,
) -> output_t:
    """
    PyTorch reference implementation of NVFP4 block-scaled GEMV.
    """
    a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, _, _, c_ref = data

    # Get dimensions from MxNxL layout
    _, _, l = c_ref.shape

    # Call torch._scaled_mm to compute the GEMV result
    for l_idx in range(l):
        # Convert the scale factor tensor to blocked format
        scale_a = to_blocked(sfa_ref_cpu[:, :, l_idx])
        scale_b = to_blocked(sfb_ref_cpu[:, :, l_idx])
        # (m, k) @ (n, k).T -> (m, n)
        res = torch._scaled_mm(
            a_ref[:, :, l_idx],
            b_ref[:, :, l_idx].transpose(0, 1),
            scale_a.cuda(),
            scale_b.cuda(),
            bias=None,
            out_dtype=torch.float16,
        )
        c_ref[:, 0, l_idx] = res[:, 0]
    return c_ref


def _warp_reduce_8way(val: cutlass.Float32, op: Callable) -> cutlass.Float32:
    val = op(val, cute.arch.shuffle_sync_bfly(val, offset=1, mask=-1, mask_and_clamp=31))
    val = op(val, cute.arch.shuffle_sync_bfly(val, offset=2, mask=-1, mask_and_clamp=31))
    val = op(val, cute.arch.shuffle_sync_bfly(val, offset=4, mask=-1, mask_and_clamp=31))
    val = op(val, cute.arch.shuffle_sync_bfly(val, offset=8, mask=-1, mask_and_clamp=31))
    # val = op(val, cute.arch.shuffle_sync_bfly(val, offset=16, mask=-1, mask_and_clamp=31))
    return val

def _warp_reduce_sum_8way(val: cutlass.Float32) -> cutlass.Float32:
    return _warp_reduce_8way(val, lambda x, y : x + y)

# The CuTe reference implementation for NVFP4 block-scaled GEMV
@cute.kernel
def kernel(
    mA_mkl: cute.Tensor,
    mB_nkl: cute.Tensor,
    mBB_nkl: cute.Tensor,
    mSFA_mkl: cute.Tensor,
    mSFB_nkl: cute.Tensor,
    mC_mnl: cute.Tensor,
):
    # Get CUDA block and thread indices
    bidx, bidy, bidz = cute.arch.block_idx()
    tidx, tidy, _ = cute.arch.thread_idx()
    bdx, bdy, _ = cute.arch.block_dim()
    tid = tidx + tidy * bdx

    # Extract the local tile for input matrix A (shape: [block_M, block_K, rest_M, rest_K, rest_L])
    gA_mkl = cute.local_tile(
        mA_mkl, (threadsPerCol, (elementsPerAccess, threadsPerRow)), (None, None, None)
    )
    # # Extract the local tile for scale factor tensor for A (same shape as gA_mkl)
    # # Here, block_M = (32, 4); block_K = (16, 4)
    gSFA_mkl = cute.local_tile(
        mSFA_mkl, (threadsPerCol, ((16, scalesPerThread), threadsPerRow)), (None, None, None)
    )
    # # Extract the local tile for input matrix B (shape: [block_N, block_K, rest_N, rest_K, rest_L])
    gB_nkl = cute.local_tile(
        mB_nkl, cute.slice_((threadsPerCol, 1, (elementsPerAccess, threadsPerRow)), (0, None, None)), (None, None, None)
    )
    # Extract the local tile for scale factor tensor for B (same shape as gB_nkl)
    gSFB_nkl = cute.local_tile(
        mSFB_nkl, cute.slice_((threadsPerCol, 1, ((16, scalesPerThread), threadsPerRow)), (0, None, None)), (None, None, None)
    )
    # Extract the local tile for output matrix C (shape: [block_M, block_N, rest_M, rest_N, rest_L])
    gC_mnl = cute.local_tile(
        mC_mnl, cute.slice_((threadsPerCol, 1, (elementsPerAccess, threadsPerRow)), (None, None, 0)), (None, None, None)
    )

    local_sum = cutlass.Float32(0.0)
    k_tile_cnt = cute.size(gA_mkl, mode=[3])
    N_step = mA_mkl.layout[1].shape[1]

    # allocator = cutlass.utils.SmemAllocator()
    # B_shm = allocator.allocate_tensor(
    #     element_type=cutlass.Float4E2M1FN,
    #     layout=cute.make_layout((16384,)),
    #     byte_alignment=16,
    #     swizzle=None
    # )
    # B_shm_store_global = cute.make_tensor(B_shm.iterator, cute.make_layout((8, 128, 16)))
    # B_shm_read_global = cute.make_tensor(B_shm.iterator, cute.make_layout((32, 8, 64)))

    # gBB_nkl = cute.local_tile(
    #     mBB_nkl, cute.slice_((16, 1, (8, 128)), (0, None, None)), (None, None, None)
    # )

    # for k_tile in cutlass.range(cute.ceil_div(cute.size(mA_mkl, mode=[1]), 1024)):
    #     if (k_tile * 128 + tid) < cute.ceil_div(cute.size(mA_mkl, mode=[1]), 8):
    #         gB_t = gBB_nkl[0, (None, tid), bidy, k_tile, bidz]
    #         B_shm_store_thread = B_shm_store_global[None, tid, k_tile]
    #         B_shm_store_thread.store(gB_t.load())
    # cute.arch.sync_threads()
    # if (bidx == 0 and bidy == 0 and bidz == 0 and tidx ==0 and tidy == 0):
    #     # cute.printf(mA_mkl.shape)
    #     cute.printf(mA_mkl.layout)
    #     cute.printf(gA_mkl.layout)
    #     cute.printf(mA_mkl.layout[1].shape[1])
    
    for k_tile in cutlass.range(k_tile_cnt): 
        if (tidx + k_tile * threadsPerRow) < N_step:
            tAgA = gA_mkl[tidy, (None, tidx), bidx, k_tile, bidz]
            tBgB = gB_nkl[0, (None, tidx), bidy, k_tile, bidz]
            tAgSFA = gSFA_mkl[tidy, (None, tidx), bidx, k_tile, bidz]
            tBgSFB = gSFB_nkl[0, (None, tidx), bidy, k_tile, bidz]

            tArA = cute.make_rmem_tensor_like(tAgA, cutlass.Float16)
            tBrB = cute.make_rmem_tensor_like(tBgB, cutlass.Float16)
            tArSFA = cute.make_rmem_tensor_like(tAgSFA, cutlass.Float32)
            tBrSFB = cute.make_rmem_tensor_like(tBgSFB, cutlass.Float32)

            # Load NVFP4 or FP8 values from global memory
            a_val_nvfp4 = tAgA.load()
            b_val_nvfp4 = tBgB.load()
            sfa_val_fp8 = tAgSFA.load()
            sfb_val_fp8 = tBgSFB.load()

            # Convert loaded values to float32 for computation (FFMA)
            a_val = a_val_nvfp4.to(cutlass.Float16)
            b_val = b_val_nvfp4.to(cutlass.Float16)
            sfa_val = sfa_val_fp8.to(cutlass.Float32)
            sfb_val = sfb_val_fp8.to(cutlass.Float32)

            # Store the converted values to RMEM CuTe tensors
            tArA.store(a_val)
            tBrB.store(b_val)
            tArSFA.store(sfa_val)
            tBrSFB.store(sfb_val)

            # Iterate over SF vector tiles and compute the scale&matmul accumulation
            for k in cutlass.range_constexpr(elementsPerAccess//16):
                offset = k * 16

                group_sum = cutlass.Float16(0.0)
                for i in cutlass.range_constexpr(16):
                    idxx = i + offset
                    group_sum += tArA[idxx] * tBrB[idxx] 

                # group_sum *= tArA[offset] * tBrB[offset] 
                local_sum += group_sum * tArSFA[offset] * tBrSFB[offset] 
                # * tArSFA[i] * tBrB[i] * tBrSFB[i]
            # local_sum += tmp
            # if (bidx == 0 and bidy == 0 and bidz == 0 and tidx ==0 and tidy == 0 and k_tile== 0):
            #     if (i % 16 == 0):
            #         cute.printf(f"---{i+224}---")
            #         cute.printf(mSFA_mkl.layout)
            #         cute.printf(gSFA_mkl.layout)
            #         cute.printf(tArA[i])
            #         cute.printf(tArSFA[i])
                # cute.printf(gB_nkl.layout)
                # cute.printf(tBrB[1])
            #         cute.printf(tBrSFB[i+32])
            #         cute.printf(tmp)

    row_sum = _warp_reduce_sum_8way(local_sum)

    tCgC = gC_mnl[tidy, None, bidx, bidy, bidz]
    tCgC = cute.make_tensor(tCgC.iterator, 1)
    res = cute.zeros_like(tCgC, cutlass.Float32)

    res += row_sum

    # cute.printf(row_sum)

    if tidx == 0:
        tCgC.store(res.to(cutlass.Float16))
    return


@cute.jit
def my_kernel(
    a_ptr: cute.Pointer,
    b_ptr: cute.Pointer,
    sfa_ptr: cute.Pointer,
    sfb_ptr: cute.Pointer,
    c_ptr: cute.Pointer,
    problem_size: tuple,
):
    """
    Host-side JIT function to prepare tensors and launch GPU kernel.
    """
    m, _, k, l = problem_size
    # Create CuTe Tensor via pointer and problem size.
    a_tensor = cute.make_tensor(
        a_ptr,
        cute.make_layout(
            (m, (elementsPerAccess, cute.ceil_div(cute.assume(k, 32), elementsPerAccess)), l),
            stride=(cute.assume(k, 32), (1, elementsPerAccess), cute.assume(m * k, 32)),
        ),
    )
    sfa_tensor = cute.make_tensor(
        sfa_ptr,
        cute.make_layout(
            (m, ((16, scalesPerThread), cute.ceil_div(cute.assume(k, 32), elementsPerAccess)), l),
            stride=(cute.ceil_div(cute.assume(k, 32),16), ((0, 1), scalesPerThread), cute.ceil_div(cute.assume(m * k, 32),16)),
        ),
    )

    # We use n=128 to create the torch tensor to do fp4 computation via torch._scaled_mm
    # then copy torch tensor to cute tensor for cute customize kernel computation
    # therefore we need to ensure b_tensor has the right stride with this 128 padded size on n.
    n_padded_128 = 128
    b_tensor = cute.make_tensor(
        b_ptr,
        cute.make_layout(
            (n_padded_128, (elementsPerAccess, cute.ceil_div(cute.assume(k, 32), elementsPerAccess)), l),
            stride=(cute.assume(k, 32), (1,elementsPerAccess), cute.assume(n_padded_128 * k, 32)),
        ),
    )
    sfb_tensor = cute.make_tensor(
        sfb_ptr,
        cute.make_layout(
            (n_padded_128, ((16, scalesPerThread), cute.ceil_div(cute.assume(k, 32), elementsPerAccess)), l),
            stride=(cute.ceil_div(cute.assume(k, 32),16), ((0, 1), scalesPerThread), cute.ceil_div(cute.assume(n_padded_128 * k, 32),16)),
        ),
    )
    bb_tensor = cute.make_tensor(
        b_ptr,
        cute.make_layout(
            (n_padded_128, (8, cute.ceil_div(cute.assume(k, 32), 8)), l),
            stride=(cute.assume(k, 32), (1,8), cute.assume(n_padded_128 * k, 32)),
        ),
    )
    c_tensor = cute.make_tensor(
        c_ptr, cute.make_layout((cute.assume(m, 32), 1, l), stride=(1, 1, m))
    )
    # Convert scale factor tensors to MMA layout
    # The layout matches Tensor Core requirements: (((32, 4), REST_M), ((SF_K, 4), REST_K), (1, REST_L))
    # sfa_layout = blockscaled_utils.tile_atom_to_shape_SF(a_tensor.shape, sf_vec_size)
    # sfa_tensor = cute.make_tensor(sfa_ptr, sfa_layout)

    # cute.printf(a_tensor.layout)
    # cute.printf(sfa_layout)


    # sfb_layout = blockscaled_utils.tile_atom_to_shape_SF(b_tensor.shape, sf_vec_size)
    # sfb_tensor = cute.make_tensor(sfb_ptr, sfb_layout)

    # Compute grid dimensions
    # Grid is (M_blocks, 1, L) where:
    # - M_blocks = ceil(M / 128) to cover all output rows
    # - L = batch size
    grid = (
        cute.ceil_div(c_tensor.shape[0], threadsPerCol),
        1,
        c_tensor.shape[2],
    )

    # Launch the CUDA kernel
    kernel(a_tensor, b_tensor, bb_tensor, sfa_tensor, sfb_tensor, c_tensor).launch(
        grid=grid,
        block=[threadsPerRow, threadsPerCol, 1],
        cluster=(1, 1, 1),
    )
    return


# Global cache for compiled kernel
_compiled_kernel_cache = None


# This function is used to compile the kernel once and cache it and then allow users to
# run the kernel multiple times to get more accurate timing results.
def compile_kernel():
    """
    Compile the kernel once and cache it.
    This should be called before any timing measurements.

    Returns:
        The compiled kernel function
    """
    global _compiled_kernel_cache

    if _compiled_kernel_cache is not None:
        return _compiled_kernel_cache

    # Create CuTe pointers for A/B/C/SFA/SFB via torch tensor data pointer
    a_ptr = make_ptr(ab_dtype, 0, cute.AddressSpace.gmem, assumed_align=16)
    b_ptr = make_ptr(ab_dtype, 0, cute.AddressSpace.gmem, assumed_align=16)
    c_ptr = make_ptr(c_dtype, 0, cute.AddressSpace.gmem, assumed_align=16)
    sfa_ptr = make_ptr(sf_dtype, 0, cute.AddressSpace.gmem, assumed_align=32)
    sfb_ptr = make_ptr(sf_dtype, 0, cute.AddressSpace.gmem, assumed_align=32)

    # Compile the kernel
    _compiled_kernel_cache = cute.compile(
        my_kernel, a_ptr, b_ptr, sfa_ptr, sfb_ptr, c_ptr, (0, 0, 0, 0)
    )

    return _compiled_kernel_cache


def custom_kernel(data: input_t) -> output_t:
    """
    Execute the block-scaled GEMV kernel.

    This is the main entry point called by the evaluation framework.
    It converts PyTorch tensors to CuTe tensors, launches the kernel,
    and returns the result.

    Args:
        data: Tuple of (a, b, sfa_cpu, sfb_cpu, c) PyTorch tensors
            a: [m, k, l] - Input matrix in float4e2m1fn
            b: [1, k, l] - Input vector in float4e2m1fn
            sfa_cpu: [m, k, l] - Scale factors in float8_e4m3fn
            sfb_cpu: [1, k, l] - Scale factors in float8_e4m3fn
            sfa_permuted: [32, 4, rest_m, 4, rest_k, l] - Scale factors in float8_e4m3fn
            sfb_permuted: [32, 4, rest_n, 4, rest_k, l] - Scale factors in float8_e4m3fn
            c: [m, 1, l] - Output vector in float16

    Returns:
        Output tensor c with computed GEMV results
    """
    a, b, sfa, sfb, sfa_permuted, sfb_permuted, c = data

    # Ensure kernel is compiled (will use cached version if available)
    # To avoid the compilation overhead, we compile the kernel once and cache it.
    compiled_func = compile_kernel()

    # Get dimensions from MxKxL layout
    m, k, l = a.shape
    # Torch use e2m1_x2 data type, thus k is halved
    # if l not in my_res:
    #     my_res[l] = custom_kernel2(data)
    #     my_cc[l] = c.clone()

    k = k * 2
    # GEMV N dimension is always 1
    n = 1

    # Create CuTe pointers for A/B/C/SFA/SFB via torch tensor data pointer
    a_ptr = make_ptr(ab_dtype, a.data_ptr(), cute.AddressSpace.gmem, assumed_align=16)
    b_ptr = make_ptr(ab_dtype, b.data_ptr(), cute.AddressSpace.gmem, assumed_align=16)
    c_ptr = make_ptr(c_dtype, c.data_ptr(), cute.AddressSpace.gmem, assumed_align=16)
    sfa_ptr = make_ptr(
        sf_dtype, sfa.data_ptr(), cute.AddressSpace.gmem, assumed_align=32
    )
    sfb_ptr = make_ptr(
        sf_dtype, sfb.data_ptr(), cute.AddressSpace.gmem, assumed_align=32
    )

    # Execute the compiled kernel
    compiled_func(a_ptr, b_ptr, sfa_ptr, sfb_ptr, c_ptr, (m, n, k, l))

    return c
scrolls · 395 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON