Skip to content
KernelIndex
Search⌘K

submission 843442

dannywillowliu-uchi · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 79 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-843442?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVIDIA B200
51.0ms
#186 of 286
2026-06-29

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:6486ff121294963523b1ab131d858785be7e2129aa7e5fde7376d3557231caca
license declaredunknown
license concludedunknown
authorsdannywillowliu-uchi
imported2026-08-26

Kernel source

submission.py79 lines
import os
import sys
import torch
from task import input_t, output_t
from torch.utils.cpp_extension import load_inline

if sys.stdout is None:
    sys.stdout = open(os.devnull, "w")
if sys.stderr is None:
    sys.stderr = open(os.devnull, "w")

_CPP = r"""
#include <torch/extension.h>
#include <cusolverDn.h>
#include <vector>

static cusolverDnHandle_t g_handle = nullptr;

// Returns (A_overwritten_with_eigenvectors_colmajor, W_eigenvalues, status)
std::vector<torch::Tensor> bsyev(torch::Tensor A) {
    int64_t batch = A.size(0);
    int64_t n = A.size(1);
    auto W = torch::empty({batch, n}, A.options());
    if (g_handle == nullptr) cusolverDnCreate(&g_handle);
    auto handle = g_handle;
    cusolverDnParams_t params;
    cusolverDnCreateParams(&params);
    cusolverEigMode_t jobz = CUSOLVER_EIG_MODE_VECTOR;
    cublasFillMode_t uplo = CUBLAS_FILL_MODE_LOWER;
    size_t dwork = 0, hwork = 0;
    cusolverStatus_t st = cusolverDnXsyevBatched_bufferSize(
        handle, params, jobz, uplo, n,
        CUDA_R_32F, A.data_ptr(), n,
        CUDA_R_32F, W.data_ptr(),
        CUDA_R_32F, &dwork, &hwork, batch);
    if (st != CUSOLVER_STATUS_SUCCESS) {
        cusolverDnDestroyParams(params);
        auto status = torch::full({1}, (int)st, torch::dtype(torch::kInt32));
        return {A, W, status};
    }
    auto dbuf = torch::empty({(int64_t)dwork},
                  torch::dtype(torch::kUInt8).device(A.device()));
    std::vector<uint8_t> hbuf(hwork);
    auto info = torch::empty({batch}, torch::dtype(torch::kInt32).device(A.device()));
    st = cusolverDnXsyevBatched(
        handle, params, jobz, uplo, n,
        CUDA_R_32F, A.data_ptr(), n,
        CUDA_R_32F, W.data_ptr(),
        CUDA_R_32F, dbuf.data_ptr(), dwork, hbuf.data(), hwork,
        info.data_ptr<int>(), batch);
    cusolverDnDestroyParams(params);
    auto status = torch::full({1}, (int)st, torch::dtype(torch::kInt32));
    return {A, W, status};
}

PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
    m.def("bsyev", &bsyev, "batched syev");
}
"""

_mod = load_inline(
    name="bsyev_ext3",
    cpp_sources=_CPP,
    extra_cflags=["-O2"],
    extra_include_paths=["/usr/local/cuda/include"],
    extra_ldflags=["-L/usr/local/cuda/lib64", "-lcusolver", "-lcudart"],
    verbose=False,
)


def custom_kernel(data: input_t) -> output_t:
    Ac = data.clone()
    Q_cm, W, status = _mod.bsyev(Ac)
    if int(status.item()) != 0:
        values, vectors = torch.linalg.eigh(data)
        return vectors, values
    Q = Q_cm.transpose(-1, -2)
    return Q, W
scrolls · 79 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON