Skip to content
KernelIndex
Search⌘K

gpt-o3_cuda_270394

gpt-o3 · cuda · Apache-2.0

Use it

Vendorable · source mirrored · Apache-2.0View source →

No package. Vendor the mirrored source: 69 lines, Apache-2.0, pinned at da91508.

main.cpp
curl "https://kernelindex.com/api/v1/implementations/flashinfer-gpt-o3-cuda-270394?include=source"
interfacecuda
revisionda915083d4c7
symbolrun
pathmain.cpp
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

25 measurements across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
GEMM n128 k2048fp16 · [2, 2048]
NVIDIA B200
347.0µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [6, 2048]
NVIDIA B200
347.5µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [5, 2048]
NVIDIA B200
347.6µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [4, 2048]
NVIDIA B200
349.1µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [1, 2048]
NVIDIA B200
349.2µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [8, 2048]
NVIDIA B200
359.9µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [25, 2048]
NVIDIA B200
364.8µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [17, 2048]
NVIDIA B200
365.0µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [32, 2048]
NVIDIA B200
365.8µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [16, 2048]
NVIDIA B200
366.4µs
#7 of 7
2025-10-16
Show all 25 measurements ›
GEMM n128 k2048fp16 · [34, 2048]
NVIDIA B200
369.1µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [93, 2048]
NVIDIA B200
382.2µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [172, 2048]
NVIDIA B200
384.4µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [289, 2048]
NVIDIA B200
385.0µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [128, 2048]
NVIDIA B200
385.3µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [492, 2048]
NVIDIA B200
385.4µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [952, 2048]
NVIDIA B200
386.9µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [63, 2048]
NVIDIA B200
387.8µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [64, 2048]
NVIDIA B200
389.3µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [8828, 2048]
NVIDIA B200
396.2µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [12251, 2048]
NVIDIA B200
624.2µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [11006, 2048]
NVIDIA B200
625.7µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [16294, 2048]
NVIDIA B200
626.6µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [12853, 2048]
NVIDIA B200
628.2µs
#7 of 7
2025-10-16
GEMM n128 k2048fp16 · [14915, 2048]
NVIDIA B200
632.9µs
#7 of 7
2025-10-16

Reproduction-ready · How evidence levels are derived →

Source and license

sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:9e6ba9df338e62697b671f0f369957c49a7c795b46cac1a899b056a5c4c24ce3
license declaredApache-2.0
license concludedApache-2.0
authorsgpt-o3
imported2026-08-20

Kernel source

main.cpp69 lines
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include <cuda_runtime.h>
#include <cuda_fp16.h>

#include "kernel.h"

/******************************************************************************
 *  Python-visible entry point
 *****************************************************************************/
torch::Tensor run(torch::Tensor A, torch::Tensor B)
{
    /* ---------------- argument checking ---------------- */
    TORCH_CHECK(A.dim() == 2 && B.dim() == 2,
                "A and B must be 2-D tensors");
    TORCH_CHECK(A.size(1) == 2048 &&
                B.size(0) == 128  && B.size(1) == 2048,
                "Shapes must be  A[M,2048]  and  B[128,2048]");
    TORCH_CHECK(A.scalar_type() == at::kHalf &&
                B.scalar_type() == at::kHalf,
                "Tensors must be float16");
    TORCH_CHECK(A.is_cuda() && B.is_cuda(),
                "Tensors have to live on CUDA");

    /* make contiguous (no-op if already so) */
    auto A_c = A.contiguous();
    auto B_c = B.contiguous();

    const int64_t M = A_c.size(0);

    /* output tensor */
    auto options = torch::TensorOptions()
                     .dtype(at::kHalf)
                     .device(A.device());
    auto C = torch::empty({M, 128}, options);

    /* temporary buffer for transposed B */
    auto B_col = torch::empty({2048, 128}, options);

    cudaStream_t stream = at::cuda::getCurrentCUDAStream();

    /* 1. transpose B  -------------------------------------------------------- */
    launch_transpose_B(
        reinterpret_cast<const __half *>(B_c.data_ptr<at::Half>()),
        reinterpret_cast<      __half *>(B_col.data_ptr<at::Half>()),
        stream);

    /* 2. GEMM ---------------------------------------------------------------- */
    launch_gemm_n128_k2048(
        reinterpret_cast<const __half *>(A_c.data_ptr<at::Half>()),
        reinterpret_cast<const __half *>(B_col.data_ptr<at::Half>()),
        reinterpret_cast<      __half *>(C.data_ptr<at::Half>()),
        static_cast<int>(M),
        stream);

    /* make sure the kernel finished before returning to Python */
    CUDA_CHECK(cudaStreamSynchronize(stream));

    return C;
}

/******************************************************************************
 *  PyBind11 module definition
 *****************************************************************************/
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m)
{
    m.def("run", &run,
          "gemm_n128_k2048 (CUDA, FP16)  –  C = A @ B.T");
}
scrolls · 69 lines total

Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0

Best evidence level for this revision: reproducible

JSON