claude-opus-4-1 / cudafbec80
claude-opus-4-1_cuda_fbec80 · claude-opus-4-1-20250805 · cuda · Apache-2.0
Use it
Vendorable · source mirrored · Apache-2.0View source →
No package. Vendor the mirrored source: 80 lines, Apache-2.0, pinned at da91508.
main.cpp
curl "https://kernelindex.com/api/v1/implementations/flashinfer-claude-opus-4-1-cuda-fbec80?include=source"interfacecuda
revisionda915083d4c7
symbolrun
pathmain.cpp
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesbf16
Benchmark evidence
7 measurements across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reproduction-ready · How evidence levels are derived →
Source and license
sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:75e5a9e54edc5791f85c0011d295c36427ef65c20d4903b1152c431bb279c539
license declaredApache-2.0
license concludedApache-2.0
authorsclaude-opus-4-1-20250805
imported2026-08-20
Kernel source
main.cpp80 lines
#include <torch/extension.h>
#include <cuda_runtime.h>
#include "kernel.h"
#include <ATen/cuda/CUDAContext.h>
#include <c10/cuda/CUDAStream.h>
#include <stdexcept>
#include <vector>
// Helper function to check CUDA errors
#define CUDA_CHECK(call) \
do { \
cudaError_t error = call; \
if (error != cudaSuccess) { \
throw std::runtime_error(std::string("CUDA error at ") + __FILE__ + ":" + \
std::to_string(__LINE__) + " - " + cudaGetErrorString(error)); \
} \
} while(0)
torch::Tensor run(
torch::Tensor hidden_states,
torch::Tensor weight
) {
// Input validation
TORCH_CHECK(hidden_states.dim() == 2, "hidden_states must be 2D tensor");
TORCH_CHECK(weight.dim() == 1, "weight must be 1D tensor");
TORCH_CHECK(hidden_states.size(1) == 2048, "hidden_size must be 2048");
TORCH_CHECK(weight.size(0) == 2048, "weight size must be 2048");
TORCH_CHECK(hidden_states.scalar_type() == torch::ScalarType::BFloat16,
"hidden_states must be BFloat16");
TORCH_CHECK(weight.scalar_type() == torch::ScalarType::BFloat16,
"weight must be BFloat16");
TORCH_CHECK(hidden_states.is_cuda(), "hidden_states must be on CUDA device");
TORCH_CHECK(weight.is_cuda(), "weight must be on CUDA device");
TORCH_CHECK(hidden_states.is_contiguous(), "hidden_states must be contiguous");
TORCH_CHECK(weight.is_contiguous(), "weight must be contiguous");
// Get dimensions
const int batch_size = hidden_states.size(0);
// Create output tensor with the same properties as input
auto output = torch::empty_like(hidden_states);
// Get CUDA stream from PyTorch
c10::cuda::CUDAStream torch_stream = c10::cuda::getCurrentCUDAStream();
cudaStream_t stream = torch_stream.stream();
// Get data pointers - use correct casting for bfloat16
const __nv_bfloat16* hidden_states_ptr = reinterpret_cast<const __nv_bfloat16*>(
hidden_states.data_ptr<at::BFloat16>()
);
const __nv_bfloat16* weight_ptr = reinterpret_cast<const __nv_bfloat16*>(
weight.data_ptr<at::BFloat16>()
);
__nv_bfloat16* output_ptr = reinterpret_cast<__nv_bfloat16*>(
output.data_ptr<at::BFloat16>()
);
// Launch kernel
launch_rmsnorm_h2048(
hidden_states_ptr,
weight_ptr,
output_ptr,
batch_size,
stream
);
// Check for errors after kernel launch
CUDA_CHECK(cudaGetLastError());
// PyTorch manages stream synchronization, so no need for explicit sync
return output;
}
// Python bindings
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("run", &run, "RMSNorm kernel for hidden_size=2048",
py::arg("hidden_states"),
py::arg("weight"));
}scrolls · 80 lines total
Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0
Best evidence level for this revision: reproducible
JSON