submission 116691
qinking · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 411 lines, June 9 Researcher Reciprocity License v1.0.
p1_v2.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-116691?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:213865fef4837af35031d115fa411fda2fd87d98f2c9b9ad0c348fc137a88ddc
license declaredunknown
license concludedunknown
authorsqinking
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
fp4
ab_dtype = cutlass.Float4E2M1FN # FP4 data type for A and BKernel source
p1_v2.py411 lines
#!POPCORN leaderboard nvfp4_gemv
from typing import Callable
import torch
from task import input_t, output_t
import cutlass
import cutlass.cute as cute
from cutlass.cute.runtime import make_ptr
import cutlass.utils.blockscaled_layout as blockscaled_utils
# Kernel configuration parameters
# mma_tiler_mnk = (16, 1, 256) # Tile sizes for M, N, K dimensions
ab_dtype = cutlass.Float4E2M1FN # FP4 data type for A and B
sf_dtype = cutlass.Float8E4M3FN # FP8 data type for scale factors
c_dtype = cutlass.Float16 # FP16 output type
sf_vec_size = 16 # Scale factor block size (16 elements share one scale)
threadsPerRow = 16
threadsPerCol = 8
elementsPerAccess = 128
scalesPerThread = elementsPerAccess // sf_vec_size
my_res = {}
my_cc = {}
# Helper function for ceiling division
def ceil_div(a, b):
return (a + b - 1) // b
def to_blocked(input_matrix):
rows, cols = input_matrix.shape
# Please ensure rows and cols are multiples of 128 and 4 respectively
n_row_blocks = ceil_div(rows, 128)
n_col_blocks = ceil_div(cols, 4)
padded = input_matrix
blocks = padded.view(n_row_blocks, 128, n_col_blocks, 4).permute(0, 2, 1, 3)
rearranged = blocks.reshape(-1, 4, 32, 4).transpose(1, 2).reshape(-1, 32, 16)
return rearranged.flatten()
def custom_kernel2(
data: input_t,
) -> output_t:
"""
PyTorch reference implementation of NVFP4 block-scaled GEMV.
"""
a_ref, b_ref, sfa_ref_cpu, sfb_ref_cpu, _, _, c_ref = data
# Get dimensions from MxNxL layout
_, _, l = c_ref.shape
# Call torch._scaled_mm to compute the GEMV result
for l_idx in range(l):
# Convert the scale factor tensor to blocked format
scale_a = to_blocked(sfa_ref_cpu[:, :, l_idx])
scale_b = to_blocked(sfb_ref_cpu[:, :, l_idx])
# (m, k) @ (n, k).T -> (m, n)
res = torch._scaled_mm(
a_ref[:, :, l_idx],
b_ref[:, :, l_idx].transpose(0, 1),
scale_a.cuda(),
scale_b.cuda(),
bias=None,
out_dtype=torch.float16,
)
c_ref[:, 0, l_idx] = res[:, 0]
return c_ref
def _warp_reduce_8way(val: cutlass.Float32, op: Callable) -> cutlass.Float32:
val = op(
val, cute.arch.shuffle_sync_bfly(val, offset=1, mask=-1, mask_and_clamp=31)
)
val = op(
val, cute.arch.shuffle_sync_bfly(val, offset=2, mask=-1, mask_and_clamp=31)
)
val = op(
val, cute.arch.shuffle_sync_bfly(val, offset=4, mask=-1, mask_and_clamp=31)
)
val = op(
val, cute.arch.shuffle_sync_bfly(val, offset=8, mask=-1, mask_and_clamp=31)
)
# val = op(val, cute.arch.shuffle_sync_bfly(val, offset=16, mask=-1, mask_and_clamp=31))
return val
def _warp_reduce_sum_8way(val: cutlass.Float32) -> cutlass.Float32:
return _warp_reduce_8way(val, lambda x, y: x + y)
# The CuTe reference implementation for NVFP4 block-scaled GEMV
@cute.kernel
def kernel(
mA_mkl: cute.Tensor,
mB_nkl: cute.Tensor,
mBB_nkl: cute.Tensor,
mSFA_mkl: cute.Tensor,
mSFB_nkl: cute.Tensor,
mC_mnl: cute.Tensor,
):
# Get CUDA block and thread indices
bidx, bidy, bidz = cute.arch.block_idx()
tidx, tidy, _ = cute.arch.thread_idx()
bdx, bdy, _ = cute.arch.block_dim()
tid = tidx + tidy * bdx
# Extract the local tile for input matrix A (shape: [block_M, block_K, rest_M, rest_K, rest_L])
gA_mkl = cute.local_tile(
mA_mkl, (threadsPerCol, (elementsPerAccess, threadsPerRow)), (None, None, None)
)
# # Extract the local tile for scale factor tensor for A (same shape as gA_mkl)
# # Here, block_M = (32, 4); block_K = (16, 4)
gSFA_mkl = cute.local_tile(
mSFA_mkl,
(threadsPerCol, ((16, scalesPerThread), threadsPerRow)),
(None, None, None),
)
# # Extract the local tile for input matrix B (shape: [block_N, block_K, rest_N, rest_K, rest_L])
gB_nkl = cute.local_tile(
mB_nkl,
cute.slice_(
(threadsPerCol, 1, (elementsPerAccess, threadsPerRow)), (0, None, None)
),
(None, None, None),
)
# Extract the local tile for scale factor tensor for B (same shape as gB_nkl)
gSFB_nkl = cute.local_tile(
mSFB_nkl,
cute.slice_(
(threadsPerCol, 1, ((16, scalesPerThread), threadsPerRow)), (0, None, None)
),
(None, None, None),
)
# Extract the local tile for output matrix C (shape: [block_M, block_N, rest_M, rest_N, rest_L])
gC_mnl = cute.local_tile(
mC_mnl,
cute.slice_(
(threadsPerCol, 1, (elementsPerAccess, threadsPerRow)), (None, None, 0)
),
(None, None, None),
)
local_sum = cutlass.Float32(0.0)
k_tile_cnt = cute.size(gA_mkl, mode=[3])
N_step = mA_mkl.layout[1].shape[1]
for k_tile in cutlass.range(k_tile_cnt):
if (tidx + k_tile * threadsPerRow) < N_step:
tAgA = gA_mkl[tidy, (None, tidx), bidx, k_tile, bidz]
tBgB = gB_nkl[0, (None, tidx), bidy, k_tile, bidz]
tAgSFA = gSFA_mkl[tidy, (None, tidx), bidx, k_tile, bidz]
tBgSFB = gSFB_nkl[0, (None, tidx), bidy, k_tile, bidz]
tArA = cute.make_rmem_tensor_like(tAgA, cutlass.Float16)
# tBrB = cute.make_rmem_tensor_like(tBgB, cutlass.Float16)
tArSFA = cute.make_rmem_tensor_like(tAgSFA, cutlass.Float32)
# tBrSFB = cute.make_rmem_tensor_like(tBgSFB, cutlass.Float16)
# Load NVFP4 or FP8 values from global memory
a_val_nvfp4 = tAgA.load()
b_val_nvfp4 = tBgB.load()
sfa_val_fp8 = tAgSFA.load()
sfb_val_fp8 = tBgSFB.load()
# Convert loaded values to float32 for computation (FFMA)
a_val = a_val_nvfp4.to(cutlass.Float16)
b_val = b_val_nvfp4.to(cutlass.Float16)
sfa_val = sfa_val_fp8.to(cutlass.Float32)
sfb_val = sfb_val_fp8.to(cutlass.Float32)
# Store the converted values to RMEM CuTe tensors
tArA.store(a_val * b_val)
# tBrB.store(b_val)
tArSFA.store(sfa_val * sfb_val)
# tBrSFB.store(sfb_val)
# Iterate over SF vector tiles and compute the scale&matmul accumulation
for k in cutlass.range_constexpr(elementsPerAccess // 16):
offset = k * 16
group_sum = cutlass.Float16(0.0)
for i in cutlass.range_constexpr(16):
idxx = i + offset
group_sum += tArA[idxx]
# group_sum *= tArA[offset] * tBrB[offset]
local_sum += group_sum * tArSFA[offset]
# * tArSFA[i] * tBrB[i] * tBrSFB[i]
row_sum = _warp_reduce_sum_8way(local_sum)
tCgC = gC_mnl[tidy, None, bidx, bidy, bidz]
tCgC = cute.make_tensor(tCgC.iterator, 1)
res = cute.zeros_like(tCgC, cutlass.Float32)
res += row_sum
# cute.printf(row_sum)
if tidx == 0:
tCgC.store(res.to(cutlass.Float16))
return
@cute.jit
def my_kernel(
a_ptr: cute.Pointer,
b_ptr: cute.Pointer,
sfa_ptr: cute.Pointer,
sfb_ptr: cute.Pointer,
c_ptr: cute.Pointer,
problem_size: tuple,
):
"""
Host-side JIT function to prepare tensors and launch GPU kernel.
"""
m, _, k, l = problem_size
# Create CuTe Tensor via pointer and problem size.
a_tensor = cute.make_tensor(
a_ptr,
cute.make_layout(
(
m,
(
elementsPerAccess,
cute.ceil_div(cute.assume(k, 32), elementsPerAccess),
),
l,
),
stride=(cute.assume(k, 32), (1, elementsPerAccess), cute.assume(m * k, 32)),
),
)
sfa_tensor = cute.make_tensor(
sfa_ptr,
cute.make_layout(
(
m,
(
(16, scalesPerThread),
cute.ceil_div(cute.assume(k, 32), elementsPerAccess),
),
l,
),
stride=(
cute.ceil_div(cute.assume(k, 32), 16),
((0, 1), scalesPerThread),
cute.ceil_div(cute.assume(m * k, 32), 16),
),
),
)
# We use n=128 to create the torch tensor to do fp4 computation via torch._scaled_mm
# then copy torch tensor to cute tensor for cute customize kernel computation
# therefore we need to ensure b_tensor has the right stride with this 128 padded size on n.
n_padded_128 = 128
b_tensor = cute.make_tensor(
b_ptr,
cute.make_layout(
(
n_padded_128,
(
elementsPerAccess,
cute.ceil_div(cute.assume(k, 32), elementsPerAccess),
),
l,
),
stride=(
cute.assume(k, 32),
(1, elementsPerAccess),
cute.assume(n_padded_128 * k, 32),
),
),
)
sfb_tensor = cute.make_tensor(
sfb_ptr,
cute.make_layout(
(
n_padded_128,
(
(16, scalesPerThread),
cute.ceil_div(cute.assume(k, 32), elementsPerAccess),
),
l,
),
stride=(
cute.ceil_div(cute.assume(k, 32), 16),
((0, 1), scalesPerThread),
cute.ceil_div(cute.assume(n_padded_128 * k, 32), 16),
),
),
)
bb_tensor = cute.make_tensor(
b_ptr,
cute.make_layout(
(n_padded_128, (8, cute.ceil_div(cute.assume(k, 32), 8)), l),
stride=(cute.assume(k, 32), (1, 8), cute.assume(n_padded_128 * k, 32)),
),
)
c_tensor = cute.make_tensor(
c_ptr, cute.make_layout((cute.assume(m, 32), 1, l), stride=(1, 1, m))
)
grid = (
cute.ceil_div(c_tensor.shape[0], threadsPerCol),
1,
c_tensor.shape[2],
)
# Launch the CUDA kernel
kernel(a_tensor, b_tensor, bb_tensor, sfa_tensor, sfb_tensor, c_tensor).launch(
grid=grid,
block=[threadsPerRow, threadsPerCol, 1],
cluster=(1, 1, 1),
)
return
# Global cache for compiled kernel
_compiled_kernel_cache = None
# This function is used to compile the kernel once and cache it and then allow users to
# run the kernel multiple times to get more accurate timing results.
def compile_kernel():
"""
Compile the kernel once and cache it.
This should be called before any timing measurements.
Returns:
The compiled kernel function
"""
global _compiled_kernel_cache
if _compiled_kernel_cache is not None:
return _compiled_kernel_cache
# Create CuTe pointers for A/B/C/SFA/SFB via torch tensor data pointer
a_ptr = make_ptr(ab_dtype, 0, cute.AddressSpace.gmem, assumed_align=16)
b_ptr = make_ptr(ab_dtype, 0, cute.AddressSpace.gmem, assumed_align=16)
c_ptr = make_ptr(c_dtype, 0, cute.AddressSpace.gmem, assumed_align=16)
sfa_ptr = make_ptr(sf_dtype, 0, cute.AddressSpace.gmem, assumed_align=32)
sfb_ptr = make_ptr(sf_dtype, 0, cute.AddressSpace.gmem, assumed_align=32)
# Compile the kernel
_compiled_kernel_cache = cute.compile(
my_kernel, a_ptr, b_ptr, sfa_ptr, sfb_ptr, c_ptr, (0, 0, 0, 0)
)
return _compiled_kernel_cache
def custom_kernel(data: input_t) -> output_t:
"""
Execute the block-scaled GEMV kernel.
This is the main entry point called by the evaluation framework.
It converts PyTorch tensors to CuTe tensors, launches the kernel,
and returns the result.
Args:
data: Tuple of (a, b, sfa_cpu, sfb_cpu, c) PyTorch tensors
a: [m, k, l] - Input matrix in float4e2m1fn
b: [1, k, l] - Input vector in float4e2m1fn
sfa_cpu: [m, k, l] - Scale factors in float8_e4m3fn
sfb_cpu: [1, k, l] - Scale factors in float8_e4m3fn
sfa_permuted: [32, 4, rest_m, 4, rest_k, l] - Scale factors in float8_e4m3fn
sfb_permuted: [32, 4, rest_n, 4, rest_k, l] - Scale factors in float8_e4m3fn
c: [m, 1, l] - Output vector in float16
Returns:
Output tensor c with computed GEMV results
"""
a, b, sfa, sfb, sfa_permuted, sfb_permuted, c = data
# Ensure kernel is compiled (will use cached version if available)
# To avoid the compilation overhead, we compile the kernel once and cache it.
compiled_func = compile_kernel()
# Get dimensions from MxKxL layout
m, k, l = a.shape
# Torch use e2m1_x2 data type, thus k is halved
# if l not in my_res:
# my_res[l] = custom_kernel2(data)
# my_cc[l] = c.clone()
k = k * 2
# GEMV N dimension is always 1
n = 1
# Create CuTe pointers for A/B/C/SFA/SFB via torch tensor data pointer
a_ptr = make_ptr(ab_dtype, a.data_ptr(), cute.AddressSpace.gmem, assumed_align=16)
b_ptr = make_ptr(ab_dtype, b.data_ptr(), cute.AddressSpace.gmem, assumed_align=16)
c_ptr = make_ptr(c_dtype, c.data_ptr(), cute.AddressSpace.gmem, assumed_align=16)
sfa_ptr = make_ptr(
sf_dtype, sfa.data_ptr(), cute.AddressSpace.gmem, assumed_align=32
)
sfb_ptr = make_ptr(
sf_dtype, sfb.data_ptr(), cute.AddressSpace.gmem, assumed_align=32
)
# Execute the compiled kernel
compiled_func(a_ptr, b_ptr, sfa_ptr, sfb_ptr, c_ptr, (m, n, k, l))
return c
scrolls · 411 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON