Skip to content
KernelIndex
Search⌘K

gpt-o3 / cuda7a2145

gpt-o3_cuda_7a2145 · gpt-o3 · cuda · Apache-2.0

Use it

Vendorable · source mirrored · Apache-2.0View source →

No package. Vendor the mirrored source: 66 lines, Apache-2.0, pinned at da91508.

main.cpp
curl "https://kernelindex.com/api/v1/implementations/flashinfer-gpt-o3-cuda-7a2145?include=source"
interfacecuda
revisionda915083d4c7
symbolrun
pathmain.cpp
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

17 measurements across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
GEMM n256 k7168fp16 · [1, 7168]
NVIDIA B200
523.3µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [4, 7168]
NVIDIA B200
542.4µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [15, 7168]
NVIDIA B200
549.0µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [14, 7168]
NVIDIA B200
565.8µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [16, 7168]
NVIDIA B200
575.0µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [32, 7168]
NVIDIA B200
625.5µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [80, 7168]
NVIDIA B200
764.8µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [901, 7168]
NVIDIA B200
802.5µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [56, 7168]
NVIDIA B200
804.0µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [55, 7168]
NVIDIA B200
831.4µs
#6 of 7
2025-10-16
Show all 17 measurements ›
GEMM n256 k7168fp16 · [53, 7168]
NVIDIA B200
836.3µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [63, 7168]
NVIDIA B200
838.4µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [57, 7168]
NVIDIA B200
840.5µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [54, 7168]
NVIDIA B200
843.2µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [58, 7168]
NVIDIA B200
846.8µs
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [14104, 7168]
NVIDIA B200
1.75ms
#6 of 7
2025-10-16
GEMM n256 k7168fp16 · [11948, 7168]
NVIDIA B200
1.80ms
#6 of 7
2025-10-16

Reproduction-ready · How evidence levels are derived →

Source and license

sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:a7bb0b05ac9d56a9b28d8afff5ec54a658236bc92ce67a449985cc5d52fd8b17
license declaredApache-2.0
license concludedApache-2.0
authorsgpt-o3
imported2026-08-20

Kernel source

main.cpp66 lines
#include "kernel.h"
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>

/*
 *  Python interface
 *
 *      C = run(A, B)
 *
 *  A : [M, 7168]   torch.float16 (CUDA)  – row-major
 *  B : [256,7168]  torch.float16 (CUDA)  – row-major
 *  C : [M, 256 ]   torch.float16 (CUDA)  – row-major
 */

torch::Tensor run(torch::Tensor A, torch::Tensor B)
{
    /*  sanity checks -------------------------------------------------------- */
    TORCH_CHECK(A.is_cuda(), "A must reside on CUDA device");
    TORCH_CHECK(B.is_cuda(), "B must reside on CUDA device");
    TORCH_CHECK(A.scalar_type() == at::kHalf, "A must be float16");
    TORCH_CHECK(B.scalar_type() == at::kHalf, "B must be float16");
    TORCH_CHECK(A.dim() == 2 && B.dim() == 2, "Inputs must be rank-2 tensors");
    TORCH_CHECK(A.size(1) == 7168,
                "A has wrong second dimension (expected 7168)");
    TORCH_CHECK(B.size(0) == 256 && B.size(1) == 7168,
                "B must have shape [256, 7168]");

    /*  make the inputs contiguous (no-op if already) ----------------------- */
    auto A_c = A.contiguous();
    auto B_c = B.contiguous();

    const int64_t M = A_c.size(0);

    /*  allocate output ------------------------------------------------------ */
    auto C = torch::empty({M, 256},
                          torch::TensorOptions()
                              .dtype(at::kHalf)
                              .device(A.device()));

    /*  raw pointers --------------------------------------------------------- */
    const __half* d_A = reinterpret_cast<const __half*>(A_c.data_ptr<at::Half>());
    const __half* d_B = reinterpret_cast<const __half*>(B_c.data_ptr<at::Half>());
    __half*       d_C = reinterpret_cast<__half*>(C.data_ptr<at::Half>());

    /*  current CUDA stream -------------------------------------------------- */
    cudaStream_t stream = at::cuda::getCurrentCUDAStream();

    /*  launch specialised kernel ------------------------------------------- */
    launch_gemm_n256_k7168(d_A, d_B, d_C, static_cast<int>(M), stream);

    /*  ensure completion ---------------------------------------------------- */
    cudaError_t err = cudaStreamSynchronize(stream);
    TORCH_CHECK(err == cudaSuccess,
                "CUDA kernel failed : ",
                cudaGetErrorString(err));

    return C;
}

/* -------------------------- PyBind registration --------------------------- */
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
    m.def("run", &run,
          "Optimised GEMM  C = A·B^T  (A[M,7168]  ·  B[256,7168]^T)",
          py::arg("A"),
          py::arg("B"));
}
scrolls · 66 lines total

Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0

Best evidence level for this revision: reproducible

JSON