Skip to content
KernelIndex
Search⌘K

gpt-o3 / cuda5a050d

gpt-o3_cuda_5a050d · gpt-o3 · cuda · Apache-2.0

Use it

Vendorable · source mirrored · Apache-2.0View source →

No package. Vendor the mirrored source: 48 lines, Apache-2.0, pinned at da91508.

main.cpp
curl "https://kernelindex.com/api/v1/implementations/flashinfer-gpt-o3-cuda-5a050d?include=source"
interfacecuda
revisionda915083d4c7
symbolrun
pathmain.cpp
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

29 measurements across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
GEMM n2048 k4096fp16 · [1, 4096]
NVIDIA B200
404.8µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [2, 4096]
NVIDIA B200
405.2µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [32, 4096]
NVIDIA B200
407.1µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [4, 4096]
NVIDIA B200
410.8µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [5, 4096]
NVIDIA B200
415.2µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [6, 4096]
NVIDIA B200
416.4µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [8, 4096]
NVIDIA B200
419.6µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [172, 4096]
NVIDIA B200
426.8µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [15, 4096]
NVIDIA B200
428.3µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [16, 4096]
NVIDIA B200
429.4µs
#7 of 7
2025-10-16
Show all 29 measurements ›
GEMM n2048 k4096fp16 · [17, 4096]
NVIDIA B200
431.8µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [34, 4096]
NVIDIA B200
437.1µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [25, 4096]
NVIDIA B200
441.1µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [128, 4096]
NVIDIA B200
442.2µs
#6 of 7
2025-10-16
GEMM n2048 k4096fp16 · [93, 4096]
NVIDIA B200
445.2µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [64, 4096]
NVIDIA B200
452.4µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [63, 4096]
NVIDIA B200
465.0µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [492, 4096]
NVIDIA B200
493.7µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [289, 4096]
NVIDIA B200
494.3µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [969, 4096]
NVIDIA B200
821.0µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [952, 4096]
NVIDIA B200
825.7µs
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [8828, 4096]
NVIDIA B200
5.94ms
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [11006, 4096]
NVIDIA B200
7.49ms
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [11938, 4096]
NVIDIA B200
8.04ms
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [12251, 4096]
NVIDIA B200
8.32ms
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [12853, 4096]
NVIDIA B200
8.65ms
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [14915, 4096]
NVIDIA B200
10.0ms
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [15813, 4096]
NVIDIA B200
10.6ms
#7 of 7
2025-10-16
GEMM n2048 k4096fp16 · [16294, 4096]
NVIDIA B200
10.9ms
#7 of 7
2025-10-16

Reproduction-ready · How evidence levels are derived →

Source and license

sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:0fba8780ddd4a7400ef0dbf6cf0029c119fc1e4c4e22de902852f1d4c1092d94
license declaredApache-2.0
license concludedApache-2.0
authorsgpt-o3
imported2026-08-20

Kernel source

main.cpp48 lines
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include "kernel.h"

/* -------------------------------------------------------------------------- */
/*                       Python-visible entry point                           */
/* -------------------------------------------------------------------------- */
torch::Tensor run(torch::Tensor A, torch::Tensor B)
{
    TORCH_CHECK(A.is_cuda() && B.is_cuda(),
                "Input tensors must be on CUDA device");
    TORCH_CHECK(A.dtype() == torch::kFloat16 &&
                B.dtype() == torch::kFloat16,
                "Only fp16 tensors are supported");
    TORCH_CHECK(A.dim() == 2 && B.dim() == 2,
                "Inputs must be 2-D matrices");
    TORCH_CHECK(A.size(1) == 4096,
                "A must have shape (M,4096)");
    TORCH_CHECK(B.size(0) == 2048 && B.size(1) == 4096,
                "B must have shape (2048,4096)");

    const int64_t M = A.size(0);

    /* Output tensor ---------------------------------------------------- */
    auto options = torch::TensorOptions()
                     .dtype(torch::kFloat16)
                     .device(torch::kCUDA, A.device().index());
    torch::Tensor C = torch::empty({M, 2048}, options);

    /* Raw device pointers ---------------------------------------------- */
    const __half *A_ptr = reinterpret_cast<const __half*>(A.data_ptr<at::Half>());
    const __half *B_ptr = reinterpret_cast<const __half*>(B.data_ptr<at::Half>());
    __half       *C_ptr = reinterpret_cast<__half*>(C.data_ptr<at::Half>());

    cudaStream_t stream = at::cuda::getCurrentCUDAStream();
    gemm_n2048_k4096_launcher(A_ptr, B_ptr, C_ptr,
                              static_cast<int>(M), stream);

    return C;
}

/* ------------------------------ PyBind11 ---------------------------------- */
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m)
{
    m.def("run", &run,
          "GEMM  (A[M,4096] · B[2048,4096]^T → C[M,2048])  "
          "optimised for NVIDIA B200");
}
scrolls · 48 lines total

Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0

Best evidence level for this revision: reproducible

JSON