submission 72202
data_pi · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 84 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-nvfp4-gemv-72202?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp8_e4m3, nvfp4
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:b61980dfccdce6c817c16650e3a45d46f69b71887c45d23452e3bea04e726110
license declaredunknown
license concludedunknown
authorsdata_pi
imported2026-08-26
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
fp4
Custom implementation of the NVFP4 block-scaled GEMV using C++ inline code.Kernel source
submission.py84 lines
import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t
# Embedded C++ source code
mm_cpp_source = """
#include <ATen/ATen.h>
#include <c10/util/Optional.h>
#include <torch/extension.h>
// Helper function for ceiling division
long ceil_div(long a, long b) { return (a + b - 1) / b; }
// Helper function to convert scale factor tensor to blocked format
torch::Tensor to_blocked(const torch::Tensor &input_matrix) {
long rows = input_matrix.size(0);
long cols = input_matrix.size(1);
long n_row_blocks = ceil_div(rows, 128);
long n_col_blocks = ceil_div(cols, 4);
torch::Tensor blocks = input_matrix.view({n_row_blocks, 128, n_col_blocks, 4})
.permute({0, 2, 1, 3});
torch::Tensor rearranged =
blocks.reshape({-1, 4, 32, 4}).transpose(1, 2).reshape({-1, 32, 16});
return rearranged.flatten();
}
torch::Tensor scaled_mm_cuda(torch::Tensor a_ref, torch::Tensor b_ref,
torch::Tensor sfa_ref, torch::Tensor sfb_ref,
torch::Tensor c_ref) {
TORCH_CHECK(a_ref.is_cuda() && b_ref.is_cuda() && c_ref.is_cuda(),
"a, b, and c must be CUDA tensors.");
long L = c_ref.size(2);
for (long l_idx = 0; l_idx < L; ++l_idx) {
torch::Tensor A = a_ref.select(2, l_idx);
torch::Tensor B_t = b_ref.select(2, l_idx).transpose(0, 1);
torch::Tensor sfa_slice = sfa_ref.select(2, l_idx);
torch::Tensor sfb_slice = sfb_ref.select(2, l_idx);
torch::Tensor scale_a = to_blocked(sfa_slice);
torch::Tensor scale_b = to_blocked(sfb_slice);
if (!scale_a.is_cuda())
scale_a = scale_a.cuda();
if (!scale_b.is_cuda())
scale_b = scale_b.cuda();
torch::Tensor res =
torch::_scaled_mm(A, B_t, scale_a, scale_b, c10::nullopt, c10::nullopt,
torch::kHalf, false);
c_ref.select(2, l_idx).select(1, 0).copy_(res.select(1, 0));
}
return c_ref;
}
"""
mm_cuda_source = ""
# Load the extension
mm_module = load_inline(
name="scaled_mm_kernel",
cpp_sources=mm_cpp_source,
cuda_sources=mm_cuda_source,
functions=["scaled_mm_cuda"],
verbose=True,
extra_cflags=["-g", "-DCUDA_HAS_FP8"],
extra_cuda_cflags=["-g", "-DCUDA_HAS_FP8", "-gencode=arch=compute_90,code=sm_90"],
)
def custom_kernel(data: input_t) -> output_t:
"""
Custom implementation of the NVFP4 block-scaled GEMV using C++ inline code.
"""
a_ref, b_ref, sfa_ref, sfb_ref, _, _, c_ref = data
return mm_module.scaled_mm_cuda(a_ref, b_ref, sfa_ref, sfb_ref, c_ref)
scrolls · 84 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON