gpt-5 / cuda95c7fe
gpt-5_cuda_95c7fe · gpt-5-2025-08-07 · cuda · Apache-2.0
Kernel source · 85 lines ↓holds 1 record
Use it
Vendorable · source mirrored · Apache-2.0View source →
No package. Vendor the mirrored source: 85 lines, Apache-2.0, pinned at da91508.
main.cpp
curl "https://kernelindex.com/api/v1/implementations/flashinfer-gpt-5-cuda-95c7fe?include=source"interfacecuda
revisionda915083d4c7
symbolrun
pathmain.cpp
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesbf16, fp32, int32
Benchmark evidence
48 measurements across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=18 · num_kv_indices=2
NVIDIA B200
8.19µs
#1 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=11 · num_kv_indices=10
NVIDIA B200
13.7µs
#3 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=15 · num_kv_indices=14
NVIDIA B200
16.5µs
#4 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=55 · num_kv_indices=38
NVIDIA B200
32.8µs
#5 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=59 · num_kv_indices=42
NVIDIA B200
34.9µs
#5 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=63 · num_kv_indices=46
NVIDIA B200
38.8µs
#5 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=67 · num_kv_indices=50
NVIDIA B200
40.6µs
#5 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=71 · num_kv_indices=54
NVIDIA B200
43.1µs
#5 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=74 · num_kv_indices=57
NVIDIA B200
45.6µs
#5 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=78 · num_kv_indices=61
NVIDIA B200
48.2µs
#5 of 7
2025-10-16
Show all 48 measurements ›Showing all 48 measurements ⌄
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=82 · num_kv_indices=65
NVIDIA B200
51.1µs
#5 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=9316 · num_kv_indices=73
NVIDIA B200
56.9µs
#5 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=9341 · num_kv_indices=98
NVIDIA B200
71.1µs
#5 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=223 · num_kv_indices=173
NVIDIA B200
124.8µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=1220 · num_kv_indices=1193
NVIDIA B200
235.4µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=1069 · num_kv_indices=1034
NVIDIA B200
238.0µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=406 · num_kv_indices=356
NVIDIA B200
244.7µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=1396 · num_kv_indices=1369
NVIDIA B200
255.3µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=1556 · num_kv_indices=1529
NVIDIA B200
257.5µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=1732 · num_kv_indices=1705
NVIDIA B200
260.0µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=1892 · num_kv_indices=1865
NVIDIA B200
265.0µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=2052 · num_kv_indices=2025
NVIDIA B200
275.6µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=2212 · num_kv_indices=2185
NVIDIA B200
277.2µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=2372 · num_kv_indices=2345
NVIDIA B200
285.0µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=2548 · num_kv_indices=2521
NVIDIA B200
299.6µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=2708 · num_kv_indices=2681
NVIDIA B200
306.6µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=3044 · num_kv_indices=3017
NVIDIA B200
308.8µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=2868 · num_kv_indices=2841
NVIDIA B200
311.3µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=4333 · num_kv_indices=4298
NVIDIA B200
367.7µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [1, 32, 128] · num_pages=597 · num_kv_indices=547
NVIDIA B200
370.7µs
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=81390 · num_kv_indices=12942
NVIDIA B200
1.71ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [16, 32, 128] · num_pages=30163 · num_kv_indices=20911
NVIDIA B200
2.01ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=60071 · num_kv_indices=50902
NVIDIA B200
2.03ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=62605 · num_kv_indices=53334
NVIDIA B200
2.05ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=63821 · num_kv_indices=54550
NVIDIA B200
2.05ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=65421 · num_kv_indices=56150
NVIDIA B200
2.08ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=64653 · num_kv_indices=55382
NVIDIA B200
2.08ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=63437 · num_kv_indices=54166
NVIDIA B200
2.09ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=63053 · num_kv_indices=53782
NVIDIA B200
2.10ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=67405 · num_kv_indices=58134
NVIDIA B200
2.10ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=67853 · num_kv_indices=58582
NVIDIA B200
2.11ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=68237 · num_kv_indices=58966
NVIDIA B200
2.11ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=66637 · num_kv_indices=57366
NVIDIA B200
2.12ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=65805 · num_kv_indices=56534
NVIDIA B200
2.12ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=64205 · num_kv_indices=54934
NVIDIA B200
2.14ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=65037 · num_kv_indices=55766
NVIDIA B200
2.15ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=67021 · num_kv_indices=57750
NVIDIA B200
2.17ms
#6 of 7
2025-10-16
GQA paged decode h32 kv8 d128 ps1bf16 · [64, 32, 128] · num_pages=66253 · num_kv_indices=56982
NVIDIA B200
2.18ms
#6 of 7
2025-10-16
Reproduction-ready · How evidence levels are derived →
Source and license
sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:14185d3afa8e6a2d18582f2f64a63f0d5843645303df60306ca2020142cba687
license declaredApache-2.0
license concludedApache-2.0
authorsgpt-5-2025-08-07
imported2026-08-20
Kernel source
main.cpp85 lines
#include <torch/extension.h>
#include "kernel.h"
#include <vector>
#include <cmath>
#include <limits>
#include <stdexcept>
#include <ATen/cuda/CUDAContext.h>
namespace py = pybind11;
static inline torch::Tensor to_device_contig(torch::Tensor t, c10::Device device, c10::ScalarType dtype) {
if (t.device() == device && t.scalar_type() == dtype && t.is_contiguous()) {
return t;
}
return t.to(device, dtype, /*non_blocking=*/false, /*copy=*/true).contiguous();
}
std::vector<torch::Tensor> run(
torch::Tensor q, // [B, 32, 128] bfloat16
torch::Tensor k_cache, // [num_pages, 1, 8, 128] bfloat16
torch::Tensor v_cache, // [num_pages, 1, 8, 128] bfloat16
torch::Tensor kv_indptr, // [B+1] int32
torch::Tensor kv_indices, // [num_kv_indices] int32
double sm_scale_double // default 1/sqrt(128)
) {
TORCH_CHECK(q.dim() == 3, "q must be [B, 32, 128]");
TORCH_CHECK(q.size(1) == 32 && q.size(2) == 128, "q must be [B, 32, 128]");
TORCH_CHECK(k_cache.dim() == 4 && k_cache.size(1) == 1 && k_cache.size(2) == 8 && k_cache.size(3) == 128,
"k_cache must be [num_pages, 1, 8, 128]");
TORCH_CHECK(v_cache.dim() == 4 && v_cache.size(1) == 1 && v_cache.size(2) == 8 && v_cache.size(3) == 128,
"v_cache must be [num_pages, 1, 8, 128]");
TORCH_CHECK(kv_indptr.dim() == 1, "kv_indptr must be 1D");
TORCH_CHECK(kv_indices.dim() == 1, "kv_indices must be 1D");
const auto B = q.size(0);
TORCH_CHECK(kv_indptr.size(0) == B + 1, "len_indptr must be batch_size + 1");
// Determine target device: use CUDA
c10::Device cuda_device = c10::Device(torch::kCUDA, at::cuda::current_device());
// Move inputs to CUDA if needed and ensure correct dtype/contiguity
auto q_dev = to_device_contig(q, cuda_device, at::kBFloat16);
auto k_cache_dev = to_device_contig(k_cache, cuda_device, at::kBFloat16);
auto v_cache_dev = to_device_contig(v_cache, cuda_device, at::kBFloat16);
auto kv_indptr_dev = to_device_contig(kv_indptr, cuda_device, at::kInt);
auto kv_indices_dev= to_device_contig(kv_indices, cuda_device, at::kInt);
// Allocate outputs on CUDA
auto output_dev = torch::empty_like(q_dev, q_dev.options().dtype(at::kBFloat16));
auto lse_dev = torch::empty({B, 32}, q_dev.options().dtype(at::kFloat));
// sm_scale
float sm_scale = static_cast<float>(sm_scale_double);
// Launch CUDA kernel
gqa_paged_decode_h32_kv8_d128_ps1_cuda(
q_dev, k_cache_dev, v_cache_dev, kv_indptr_dev, kv_indices_dev,
sm_scale, output_dev, lse_dev
);
// If original inputs were on CPU, return CPU tensors; otherwise return CUDA tensors
if (!q.is_cuda()) {
auto output_cpu = output_dev.to(torch::kCPU, at::kBFloat16);
auto lse_cpu = lse_dev.to(torch::kCPU, at::kFloat);
return {output_cpu, lse_cpu};
} else {
return {output_dev, lse_dev};
}
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
double default_sm_scale = 1.0 / std::sqrt(128.0);
m.def(
"run",
&run,
py::arg("q"),
py::arg("k_cache"),
py::arg("v_cache"),
py::arg("kv_indptr"),
py::arg("kv_indices"),
py::arg("sm_scale") = default_sm_scale,
"GQA paged decode kernel (h32, kv8, d128, page_size=1) optimized for B200"
);
}scrolls · 85 lines total
Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0
Best evidence level for this revision: reproducible
JSON