submission 875325
obito092430 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 90 lines, June 9 Researcher Reciprocity License v1.0.
submission_cusolver.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-875325?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:98776899413821e7160ac3148fd40af75ed228ab5b27b22d175cb4da9d5d37ed
license declaredunknown
license concludedunknown
authorsobito092430
imported2026-08-26
Kernel source
submission_cusolver.py90 lines
#!POPCORN leaderboard eigh
#!POPCORN gpu B200
import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t
CPP_SRC = """
torch::Tensor syev_batched(torch::Tensor a, torch::Tensor w);
"""
CUDA_SRC = """
#include <cusolverDn.h>
#include <stdexcept>
#define CUSOLVER_CHECK(expr) \\
do { \\
cusolverStatus_t st_ = (expr); \\
if (st_ != CUSOLVER_STATUS_SUCCESS) { \\
throw std::runtime_error("cusolver error " + std::to_string(st_));\\
} \\
} while (0)
static cusolverDnHandle_t get_handle() {
static cusolverDnHandle_t handle = [] {
cusolverDnHandle_t h;
CUSOLVER_CHECK(cusolverDnCreate(&h));
return h;
}();
return handle;
}
static cusolverDnParams_t get_params() {
static cusolverDnParams_t params = [] {
cusolverDnParams_t p;
CUSOLVER_CHECK(cusolverDnCreateParams(&p));
return p;
}();
return params;
}
// In-place batched symmetric eigendecomposition.
// `a` is (batch, n, n) fp32 contiguous; on exit it holds eigenvectors
// (column-major per matrix => row-major tensor is Q^T).
// `w` is (batch, n) fp32, eigenvalues ascending.
torch::Tensor syev_batched(torch::Tensor a, torch::Tensor w) {
TORCH_CHECK(a.is_cuda() && a.is_contiguous());
const int64_t batch = a.size(0);
const int64_t n = a.size(1);
cusolverDnHandle_t handle = get_handle();
cusolverDnParams_t params = get_params();
size_t d_bytes = 0, h_bytes = 0;
CUSOLVER_CHECK(cusolverDnXsyevBatched_bufferSize(
handle, params, CUSOLVER_EIG_MODE_VECTOR, CUBLAS_FILL_MODE_LOWER,
n, CUDA_R_32F, a.data_ptr(), n, CUDA_R_32F, w.data_ptr(),
CUDA_R_32F, &d_bytes, &h_bytes, (int)batch));
auto opts = torch::TensorOptions().device(a.device()).dtype(torch::kUInt8);
torch::Tensor d_work = torch::empty({(int64_t)std::max<size_t>(d_bytes, 1)}, opts);
std::vector<uint8_t> h_work(std::max<size_t>(h_bytes, 1));
torch::Tensor info = torch::empty({batch}, torch::TensorOptions().device(a.device()).dtype(torch::kInt32));
CUSOLVER_CHECK(cusolverDnXsyevBatched(
handle, params, CUSOLVER_EIG_MODE_VECTOR, CUBLAS_FILL_MODE_LOWER,
n, CUDA_R_32F, a.data_ptr(), n, CUDA_R_32F, w.data_ptr(),
CUDA_R_32F, d_work.data_ptr(), d_bytes, h_work.data(), h_bytes,
info.data_ptr<int>(), (int)batch));
return a;
}
"""
_mod = load_inline(
name="eigh_cusolver_batched",
cpp_sources=[CPP_SRC],
cuda_sources=[CUDA_SRC],
functions=["syev_batched"],
extra_ldflags=["-lcusolver"],
verbose=False,
)
def custom_kernel(data: input_t) -> output_t:
a = data.clone()
w = torch.empty(a.shape[0], a.shape[1], device=a.device, dtype=torch.float32)
_mod.syev_batched(a, w)
return a.transpose(-1, -2), w
scrolls · 90 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON