Skip to content
KernelIndex
Search⌘K

gpt-5 / cuda8ba217

gpt-5_cuda_8ba217 · gpt-5-2025-08-07 · cuda · Apache-2.0

Use it

Vendorable · source mirrored · Apache-2.0View source →

No package. Vendor the mirrored source: 52 lines, Apache-2.0, pinned at da91508.

main.cpp
curl "https://kernelindex.com/api/v1/implementations/flashinfer-gpt-5-cuda-8ba217?include=source"
interfacecuda
revisionda915083d4c7
symbolrun
pathmain.cpp
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

43 measurements across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
GEMM n28672 k4096fp16 · [1, 4096]
NVIDIA B200
67.8µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [4, 4096]
NVIDIA B200
67.9µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [16, 4096]
NVIDIA B200
67.9µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [15, 4096]
NVIDIA B200
67.9µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [2, 4096]
NVIDIA B200
68.2µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [7, 4096]
NVIDIA B200
68.3µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [32, 4096]
NVIDIA B200
68.3µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [8, 4096]
NVIDIA B200
68.3µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [35, 4096]
NVIDIA B200
68.5µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [48, 4096]
NVIDIA B200
68.6µs
#4 of 8
2025-10-16
Show all 43 measurements ›
GEMM n28672 k4096fp16 · [24, 4096]
NVIDIA B200
68.6µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [64, 4096]
NVIDIA B200
68.7µs
#4 of 8
2025-10-16
GEMM n28672 k4096fp16 · [70, 4096]
NVIDIA B200
69.2µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [40, 4096]
NVIDIA B200
69.3µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [80, 4096]
NVIDIA B200
69.4µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [56, 4096]
NVIDIA B200
69.8µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [72, 4096]
NVIDIA B200
69.8µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [144, 4096]
NVIDIA B200
70.7µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [160, 4096]
NVIDIA B200
71.3µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [136, 4096]
NVIDIA B200
71.5µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [104, 4096]
NVIDIA B200
74.5µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [120, 4096]
NVIDIA B200
74.6µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [88, 4096]
NVIDIA B200
74.6µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [128, 4096]
NVIDIA B200
74.9µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [96, 4096]
NVIDIA B200
74.9µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [112, 4096]
NVIDIA B200
74.9µs
#5 of 8
2025-10-16
GEMM n28672 k4096fp16 · [168, 4096]
NVIDIA B200
75.8µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [176, 4096]
NVIDIA B200
75.8µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [152, 4096]
NVIDIA B200
76.7µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [184, 4096]
NVIDIA B200
76.9µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [216, 4096]
NVIDIA B200
77.8µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [200, 4096]
NVIDIA B200
77.8µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [232, 4096]
NVIDIA B200
77.9µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [192, 4096]
NVIDIA B200
78.0µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [248, 4096]
NVIDIA B200
78.1µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [224, 4096]
NVIDIA B200
78.2µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [208, 4096]
NVIDIA B200
78.3µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [240, 4096]
NVIDIA B200
78.3µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [256, 4096]
NVIDIA B200
78.4µs
#3 of 8
2025-10-16
GEMM n28672 k4096fp16 · [972, 4096]
NVIDIA B200
180.0µs
#1 of 8
2025-10-16
GEMM n28672 k4096fp16 · [2053, 4096]
NVIDIA B200
360.2µs
#2 of 8
2025-10-16
GEMM n28672 k4096fp16 · [2379, 4096]
NVIDIA B200
414.3µs
#1 of 8
2025-10-16
GEMM n28672 k4096fp16 · [8192, 4096]
NVIDIA B200
1.36ms
#2 of 8
2025-10-16

Reproduction-ready · How evidence levels are derived →

Source and license

sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:5403fa34448dd4b18ceaf939fd50ff8eedf46ec6e2099e0e3aa06cc72b24de7a
license declaredApache-2.0
license concludedApache-2.0
authorsgpt-5-2025-08-07
imported2026-08-20

Kernel source

main.cpp52 lines
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include <cuda_runtime.h>
#include <cuda_fp16.h>
#include <stdexcept>
#include <string>
#include "kernel.h"

namespace py = pybind11;

static void validate_inputs(const torch::Tensor& A, const torch::Tensor& B) {
    if (!A.is_cuda() || !B.is_cuda())
        throw std::invalid_argument("A and B must be CUDA tensors");
    if (A.scalar_type() != at::kHalf || B.scalar_type() != at::kHalf)
        throw std::invalid_argument("A and B must be float16 (Half) tensors");
    if (A.dim() != 2 || B.dim() != 2)
        throw std::invalid_argument("A and B must be 2D tensors");
    if (A.size(1) != CONST_K)
        throw std::invalid_argument("A.shape[1] must be 4096");
    if (B.size(0) != CONST_N || B.size(1) != CONST_K)
        throw std::invalid_argument("B must have shape [28672, 4096]");
    if (A.device().index() != B.device().index())
        throw std::invalid_argument("A and B must be on the same CUDA device");
}

torch::Tensor run(torch::Tensor A, torch::Tensor B) {
    validate_inputs(A, B);

    if (!A.is_contiguous()) A = A.contiguous();
    if (!B.is_contiguous()) B = B.contiguous();

    const int64_t M = A.size(0);
    auto options = A.options();
    torch::Tensor C = torch::empty({M, (int64_t)CONST_N}, options);

    const __half* A_ptr = reinterpret_cast<const __half*>(A.data_ptr<at::Half>());
    const __half* B_ptr = reinterpret_cast<const __half*>(B.data_ptr<at::Half>());
    __half* C_ptr = reinterpret_cast<__half*>(C.data_ptr<at::Half>());

    cudaStream_t stream = at::cuda::getCurrentCUDAStream().stream();

    gemm_n_28672_k_4096(A_ptr, B_ptr, C_ptr, M, stream);

    CUDA_CHECK(cudaGetLastError());

    return C;
}

PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
    m.def("run", &run, "gemm_n_28672_k_4096 (CUDA, cuBLASLt if available)",
          py::arg("A"), py::arg("B"));
}
scrolls · 52 lines total

Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0

Best evidence level for this revision: reproducible

JSON