Skip to content
KernelIndex
Search⌘K

claude-opus-4-1 / cudafbec80

claude-opus-4-1_cuda_fbec80 · claude-opus-4-1-20250805 · cuda · Apache-2.0

Use it

Vendorable · source mirrored · Apache-2.0View source →

No package. Vendor the mirrored source: 80 lines, Apache-2.0, pinned at da91508.

main.cpp
curl "https://kernelindex.com/api/v1/implementations/flashinfer-claude-opus-4-1-cuda-fbec80?include=source"
interfacecuda
revisionda915083d4c7
symbolrun
pathmain.cpp
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesbf16

Benchmark evidence

7 measurements across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
RMSNorm h2048bf16 · [2048] · batch_size=1
NVIDIA B200
6.46µs
#2 of 7
2025-10-16
RMSNorm h2048bf16 · [2048] · batch_size=6
NVIDIA B200
6.81µs
#2 of 7
2025-10-16
RMSNorm h2048bf16 · [2048] · batch_size=34
NVIDIA B200
7.10µs
#3 of 7
2025-10-16
RMSNorm h2048bf16 · [2048] · batch_size=64
NVIDIA B200
7.56µs
#3 of 7
2025-10-16
RMSNorm h2048bf16 · [2048] · batch_size=79
NVIDIA B200
7.79µs
#3 of 7
2025-10-16
RMSNorm h2048bf16 · [2048] · batch_size=12383
NVIDIA B200
33.1µs
#3 of 7
2025-10-16
RMSNorm h2048bf16 · [2048] · batch_size=16254
NVIDIA B200
41.2µs
#3 of 7
2025-10-16

Reproduction-ready · How evidence levels are derived →

Source and license

sourcehttps://huggingface.co/datasets/flashinfer-ai/flashinfer-trace
commitda915083d4c7c5e61aa3005e3d17ae488e0fc71c
revision digestsha256:75e5a9e54edc5791f85c0011d295c36427ef65c20d4903b1152c431bb279c539
license declaredApache-2.0
license concludedApache-2.0
authorsclaude-opus-4-1-20250805
imported2026-08-20

Kernel source

main.cpp80 lines
#include <torch/extension.h>
#include <cuda_runtime.h>
#include "kernel.h"
#include <ATen/cuda/CUDAContext.h>
#include <c10/cuda/CUDAStream.h>
#include <stdexcept>
#include <vector>

// Helper function to check CUDA errors
#define CUDA_CHECK(call) \
    do { \
        cudaError_t error = call; \
        if (error != cudaSuccess) { \
            throw std::runtime_error(std::string("CUDA error at ") + __FILE__ + ":" + \
                std::to_string(__LINE__) + " - " + cudaGetErrorString(error)); \
        } \
    } while(0)

torch::Tensor run(
    torch::Tensor hidden_states,
    torch::Tensor weight
) {
    // Input validation
    TORCH_CHECK(hidden_states.dim() == 2, "hidden_states must be 2D tensor");
    TORCH_CHECK(weight.dim() == 1, "weight must be 1D tensor");
    TORCH_CHECK(hidden_states.size(1) == 2048, "hidden_size must be 2048");
    TORCH_CHECK(weight.size(0) == 2048, "weight size must be 2048");
    TORCH_CHECK(hidden_states.scalar_type() == torch::ScalarType::BFloat16, 
                "hidden_states must be BFloat16");
    TORCH_CHECK(weight.scalar_type() == torch::ScalarType::BFloat16, 
                "weight must be BFloat16");
    TORCH_CHECK(hidden_states.is_cuda(), "hidden_states must be on CUDA device");
    TORCH_CHECK(weight.is_cuda(), "weight must be on CUDA device");
    TORCH_CHECK(hidden_states.is_contiguous(), "hidden_states must be contiguous");
    TORCH_CHECK(weight.is_contiguous(), "weight must be contiguous");
    
    // Get dimensions
    const int batch_size = hidden_states.size(0);
    
    // Create output tensor with the same properties as input
    auto output = torch::empty_like(hidden_states);
    
    // Get CUDA stream from PyTorch
    c10::cuda::CUDAStream torch_stream = c10::cuda::getCurrentCUDAStream();
    cudaStream_t stream = torch_stream.stream();
    
    // Get data pointers - use correct casting for bfloat16
    const __nv_bfloat16* hidden_states_ptr = reinterpret_cast<const __nv_bfloat16*>(
        hidden_states.data_ptr<at::BFloat16>()
    );
    const __nv_bfloat16* weight_ptr = reinterpret_cast<const __nv_bfloat16*>(
        weight.data_ptr<at::BFloat16>()
    );
    __nv_bfloat16* output_ptr = reinterpret_cast<__nv_bfloat16*>(
        output.data_ptr<at::BFloat16>()
    );
    
    // Launch kernel
    launch_rmsnorm_h2048(
        hidden_states_ptr,
        weight_ptr,
        output_ptr,
        batch_size,
        stream
    );
    
    // Check for errors after kernel launch
    CUDA_CHECK(cudaGetLastError());
    
    // PyTorch manages stream synchronization, so no need for explicit sync
    
    return output;
}

// Python bindings
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
    m.def("run", &run, "RMSNorm kernel for hidden_size=2048",
          py::arg("hidden_states"),
          py::arg("weight"));
}
scrolls · 80 lines total

Source code from FlashInfer-Bench (flashinfer-ai/flashinfer-trace) · Apache-2.0

Best evidence level for this revision: reproducible

JSON