submission 873181
d_lolo_ · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 94 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-873181?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:6b88459aec0625bca6937a4a9b012ec6c38764397c90d4e5ccba8e0bdf000a82
license declaredunknown
license concludedunknown
authorsd_lolo_
imported2026-08-26
Kernel source
submission.py94 lines
#!POPCORN leaderboard eigh
#!POPCORN gpu B200
"""Batched symmetric eigendecomposition.
torch.linalg.eigh serial-loops the batch for n>32 (a ~200x cliff), because it does
not use the true batched cuSOLVER API. This calls `cusolverDnXsyevBatched` (CUDA 13)
in one shot via a native (load_inline) extension linking libcusolver — computing the
whole batch as a single operation. Returns (Q, L) in eigh convention (Q columns are
eigenvectors, L ascending). Falls back to torch.linalg.eigh if anything is unsupported.
"""
import os
import torch
# Point the JIT compiler at the CUDA toolkit if it isn't already discoverable.
for _c in ("/usr/local/cuda", "/usr/local/cuda-13.0", "/usr/local/cuda-13"):
if os.path.isdir(_c):
os.environ.setdefault("CUDA_HOME", _c)
break
_CU = r'''
#include <torch/extension.h>
#include <cusolverDn.h>
#include <cuda_runtime.h>
#include <vector>
// cuSOLVER runs on the default execution queue (queue 0), which is what the
// harness times and where torch places its work by default — so results and
// timing stay consistent without any explicit queue management.
static cusolverDnHandle_t g_handle = nullptr;
static cusolverDnParams_t g_params = nullptr;
void xsyev_batched(torch::Tensor A, torch::Tensor W, torch::Tensor info) {
const int64_t B = A.size(0);
const int64_t n = A.size(1);
if (g_handle == nullptr) {
cusolverDnCreate(&g_handle);
cusolverDnCreateParams(&g_params);
}
cusolverEigMode_t jobz = CUSOLVER_EIG_MODE_VECTOR;
cublasFillMode_t uplo = CUBLAS_FILL_MODE_LOWER;
size_t ws_dev = 0, ws_host = 0;
cusolverDnXsyevBatched_bufferSize(
g_handle, g_params, jobz, uplo, n,
CUDA_R_32F, A.data_ptr(), n, CUDA_R_32F, W.data_ptr(),
CUDA_R_32F, &ws_dev, &ws_host, B);
auto dev_buf = torch::empty({(int64_t)ws_dev}, A.options().dtype(torch::kUInt8));
std::vector<uint8_t> host_buf(ws_host > 0 ? ws_host : 1);
cusolverStatus_t st = cusolverDnXsyevBatched(
g_handle, g_params, jobz, uplo, n,
CUDA_R_32F, A.data_ptr(), n, CUDA_R_32F, W.data_ptr(),
CUDA_R_32F, dev_buf.data_ptr(), ws_dev, host_buf.data(), ws_host,
info.data_ptr<int>(), B);
TORCH_CHECK(st == CUSOLVER_STATUS_SUCCESS, "xsyev_batched failed: ", (int)st);
}
'''
_ext = None
def _get_ext():
global _ext
if _ext is None:
from torch.utils.cpp_extension import load_inline
_ext = load_inline(
name="eigh_xsyev_ext",
cpp_sources="void xsyev_batched(torch::Tensor A, torch::Tensor W, torch::Tensor info);",
cuda_sources=_CU,
functions=["xsyev_batched"],
extra_ldflags=["-lcusolver"],
verbose=False,
)
return _ext
def _batched_eigh(A):
B, n, _ = A.shape
a = A.contiguous().clone() # overwritten with eigenvectors
W = torch.empty(B, n, dtype=torch.float32, device=A.device)
info = torch.empty(B, dtype=torch.int32, device=A.device)
_get_ext().xsyev_batched(a, W, info)
return a.transpose(-1, -2), W # col-major cols -> our columns
def custom_kernel(data):
A = data
if not (A.is_cuda and A.dtype == torch.float32 and A.dim() == 3 and A.size(1) == A.size(2)):
values, vectors = torch.linalg.eigh(A)
return vectors, values
try:
return _batched_eigh(A)
except Exception:
values, vectors = torch.linalg.eigh(A)
return vectors, values
scrolls · 94 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON