submission 843442
dannywillowliu-uchi · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 79 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-843442?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:6486ff121294963523b1ab131d858785be7e2129aa7e5fde7376d3557231caca
license declaredunknown
license concludedunknown
authorsdannywillowliu-uchi
imported2026-08-26
Kernel source
submission.py79 lines
import os
import sys
import torch
from task import input_t, output_t
from torch.utils.cpp_extension import load_inline
if sys.stdout is None:
sys.stdout = open(os.devnull, "w")
if sys.stderr is None:
sys.stderr = open(os.devnull, "w")
_CPP = r"""
#include <torch/extension.h>
#include <cusolverDn.h>
#include <vector>
static cusolverDnHandle_t g_handle = nullptr;
// Returns (A_overwritten_with_eigenvectors_colmajor, W_eigenvalues, status)
std::vector<torch::Tensor> bsyev(torch::Tensor A) {
int64_t batch = A.size(0);
int64_t n = A.size(1);
auto W = torch::empty({batch, n}, A.options());
if (g_handle == nullptr) cusolverDnCreate(&g_handle);
auto handle = g_handle;
cusolverDnParams_t params;
cusolverDnCreateParams(¶ms);
cusolverEigMode_t jobz = CUSOLVER_EIG_MODE_VECTOR;
cublasFillMode_t uplo = CUBLAS_FILL_MODE_LOWER;
size_t dwork = 0, hwork = 0;
cusolverStatus_t st = cusolverDnXsyevBatched_bufferSize(
handle, params, jobz, uplo, n,
CUDA_R_32F, A.data_ptr(), n,
CUDA_R_32F, W.data_ptr(),
CUDA_R_32F, &dwork, &hwork, batch);
if (st != CUSOLVER_STATUS_SUCCESS) {
cusolverDnDestroyParams(params);
auto status = torch::full({1}, (int)st, torch::dtype(torch::kInt32));
return {A, W, status};
}
auto dbuf = torch::empty({(int64_t)dwork},
torch::dtype(torch::kUInt8).device(A.device()));
std::vector<uint8_t> hbuf(hwork);
auto info = torch::empty({batch}, torch::dtype(torch::kInt32).device(A.device()));
st = cusolverDnXsyevBatched(
handle, params, jobz, uplo, n,
CUDA_R_32F, A.data_ptr(), n,
CUDA_R_32F, W.data_ptr(),
CUDA_R_32F, dbuf.data_ptr(), dwork, hbuf.data(), hwork,
info.data_ptr<int>(), batch);
cusolverDnDestroyParams(params);
auto status = torch::full({1}, (int)st, torch::dtype(torch::kInt32));
return {A, W, status};
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("bsyev", &bsyev, "batched syev");
}
"""
_mod = load_inline(
name="bsyev_ext3",
cpp_sources=_CPP,
extra_cflags=["-O2"],
extra_include_paths=["/usr/local/cuda/include"],
extra_ldflags=["-L/usr/local/cuda/lib64", "-lcusolver", "-lcudart"],
verbose=False,
)
def custom_kernel(data: input_t) -> output_t:
Ac = data.clone()
Q_cm, W, status = _mod.bsyev(Ac)
if int(status.item()) != 0:
values, vectors = torch.linalg.eigh(data)
return vectors, values
Q = Q_cm.transpose(-1, -2)
return Q, W
scrolls · 79 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON