Skip to content
KernelIndex
Search⌘K

gpt-5 / cuda5c1f52

gpt-5_cuda_5c1f52 · gpt-5-2025-08-07 · cuda · Apache-2.0

Use it

Vendorable · source mirrored · Apache-2.0View source →

No package. Vendor the mirrored source: 64 lines, Apache-2.0, pinned at da91508.

main.cpp
curl "https://kernelindex.com/api/v1/implementations/flashinfer-gpt-5-cuda-5c1f52?include=source"
interfacecuda
revisionda915083d4c7
symbolrun
pathmain.cpp
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp16

Benchmark evidence

43 measurements across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
GEMM n4096 k4096fp16 · [972, 4096]
NVIDIA B200
484.8µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [160, 4096]
NVIDIA B200
504.5µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [144, 4096]
NVIDIA B200
504.6µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [152, 4096]
NVIDIA B200
504.6µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [176, 4096]
NVIDIA B200
504.8µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [136, 4096]
NVIDIA B200
504.8µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [168, 4096]
NVIDIA B200
504.8µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [192, 4096]
NVIDIA B200
504.9µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [184, 4096]
NVIDIA B200
505.0µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [208, 4096]
NVIDIA B200
505.1µs
#8 of 9
2025-10-16
Show all 43 measurements ›
GEMM n4096 k4096fp16 · [240, 4096]
NVIDIA B200
505.2µs
#7 of 8
2025-10-16
GEMM n4096 k4096fp16 · [224, 4096]
NVIDIA B200
505.3µs
#7 of 8
2025-10-16
GEMM n4096 k4096fp16 · [200, 4096]
NVIDIA B200
505.3µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [232, 4096]
NVIDIA B200
505.5µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [216, 4096]
NVIDIA B200
505.5µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [256, 4096]
NVIDIA B200
505.6µs
#8 of 9
2025-10-16
GEMM n4096 k4096fp16 · [248, 4096]
NVIDIA B200
505.8µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [72, 4096]
NVIDIA B200
534.4µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [40, 4096]
NVIDIA B200
534.5µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [56, 4096]
NVIDIA B200
534.5µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [88, 4096]
NVIDIA B200
534.6µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [64, 4096]
NVIDIA B200
534.6µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [104, 4096]
NVIDIA B200
534.7µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [112, 4096]
NVIDIA B200
534.9µs
#7 of 8
2025-10-16
GEMM n4096 k4096fp16 · [120, 4096]
NVIDIA B200
534.9µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [24, 4096]
NVIDIA B200
534.9µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [96, 4096]
NVIDIA B200
534.9µs
#7 of 8
2025-10-16
GEMM n4096 k4096fp16 · [80, 4096]
NVIDIA B200
534.9µs
#7 of 8
2025-10-16
GEMM n4096 k4096fp16 · [8, 4096]
NVIDIA B200
535.0µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [48, 4096]
NVIDIA B200
535.0µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [70, 4096]
NVIDIA B200
535.0µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [128, 4096]
NVIDIA B200
535.1µs
#8 of 9
2025-10-16
GEMM n4096 k4096fp16 · [7, 4096]
NVIDIA B200
535.2µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [32, 4096]
NVIDIA B200
535.3µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [2, 4096]
NVIDIA B200
535.3µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [16, 4096]
NVIDIA B200
535.4µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [35, 4096]
NVIDIA B200
535.6µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [4, 4096]
NVIDIA B200
535.7µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [15, 4096]
NVIDIA B200
535.8µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [1, 4096]
NVIDIA B200
535.9µs
#8 of 8
2025-10-16
GEMM n4096 k4096fp16 · [2053, 4096]
NVIDIA B200
952.1µs
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [2379, 4096]
NVIDIA B200
1.42ms
#7 of 7
2025-10-16
GEMM n4096 k4096fp16 · [8192, 4096]
NVIDIA B200
3.30ms
#7 of 8
2025-10-16

Reproduction-ready · How evidence levels are derived →

Source and license

sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:c06413387387964f7742957a369ef48c356881c6bb4aa3b68453be0e6bfdb438
license declaredApache-2.0
license concludedApache-2.0
authorsgpt-5-2025-08-07
imported2026-08-20

Kernel source

main.cpp64 lines
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include <c10/cuda/CUDAStream.h>
#include <cuda_fp16.h>
#include <vector>
#include <stdexcept>
#include <sstream>
#include "kernel.h"

static void check_inputs(const torch::Tensor& A, const torch::Tensor& B) {
  // Shapes: A [M, 4096], B [4096, 4096], dtype float16
  TORCH_CHECK(A.dim() == 2, "A must be 2D [M, 4096]");
  TORCH_CHECK(B.dim() == 2, "B must be 2D [4096, 4096]");
  TORCH_CHECK(A.size(1) == GEMM_K_CONST, "A.shape[1] must be 4096 (K)");
  TORCH_CHECK(B.size(0) == GEMM_N_CONST && B.size(1) == GEMM_K_CONST,
              "B must be [4096, 4096] (N=4096, K=4096)");
  TORCH_CHECK(A.dtype() == torch::kFloat16, "A must be torch.float16");
  TORCH_CHECK(B.dtype() == torch::kFloat16, "B must be torch.float16");
  TORCH_CHECK(A.is_contiguous(), "A must be contiguous");
  TORCH_CHECK(B.is_contiguous(), "B must be contiguous");
}

torch::Tensor run(torch::Tensor A, torch::Tensor B) {
  check_inputs(A, B);
  const int64_t M = A.size(0);

  // Decide device placement
  bool inputs_on_cuda = A.is_cuda() && B.is_cuda();

  torch::Tensor A_cuda = A;
  torch::Tensor B_cuda = B;

  if (!inputs_on_cuda) {
    // Move to CUDA with dtype preserved (float16)
    A_cuda = A.contiguous().to(torch::kCUDA);
    B_cuda = B.contiguous().to(torch::kCUDA);
  } else {
    A_cuda = A.contiguous();
    B_cuda = B.contiguous();
  }

  // Allocate output on CUDA
  auto options = torch::TensorOptions().device(A_cuda.device()).dtype(torch::kFloat16);
  torch::Tensor C_cuda = torch::empty({M, (int64_t)GEMM_N_CONST}, options);

  // Launch kernel on current stream
  auto stream = at::cuda::getCurrentCUDAStream();
  const __half* A_ptr = reinterpret_cast<const __half*>(A_cuda.data_ptr<at::Half>());
  const __half* B_ptr = reinterpret_cast<const __half*>(B_cuda.data_ptr<at::Half>());
  __half* C_ptr = reinterpret_cast<__half*>(C_cuda.data_ptr<at::Half>());

  gemm_n_4096_k_4096_launch(A_ptr, B_ptr, C_ptr, static_cast<int>(M), stream.stream());

  // If inputs were CPU tensors, return result to CPU to match requirement
  if (!inputs_on_cuda) {
    return C_cuda.to(torch::kCPU);
  }
  return C_cuda;
}

PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
  m.def("run", &run, "gemm_n_4096_k_4096 (A[M,4096], B[4096,4096]) -> C[M,4096] (float16)",
        py::arg("A"), py::arg("B"));
}
scrolls · 64 lines total

Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0

Best evidence level for this revision: reproducible

JSON