submission 845307
Lorenzo · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 113 lines, June 9 Researcher Reciprocity License v1.0.
submission_7.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-845307?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:d58a197b46635387198ac598faabccede681c138d88fb71213d8b60db6c7cfb3
license declaredunknown
license concludedunknown
authorsLorenzo
imported2026-08-26
Kernel source
submission_7.py113 lines
#!POPCORN leaderboard eigh
#!POPCORN gpu B200
"""eigh_py submission_7 — BATCHED-SYEV PROBE.
Hypothesis: torch.linalg.eigh loops cusolverDnSsyevd per matrix (SMs idle at big
batch). CUDA 12.6+ ships cusolverDnXsyevBatched — a *batched* dense symmetric
eigensolver with NO n<=32 limit. If it is genuinely batch-parallel, calling it
directly should beat the looped path with ZERO custom math.
This file routes EVERY shape through cusolverDnXsyevBatched (jobz=VECTOR), with
a per-call finite-guard + exception fallback to torch.linalg.eigh. If the symbol
is unavailable (older toolkit) the build fails and we fall back everywhere -> the
probe still runs (== baseline) and we learn the API isn't present.
Output contract: (Q, L), Q = eigenvector COLUMNS, L ascending (syev gives both).
"""
import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t
torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.allow_tf32 = False
_CUDA_SRC = r"""
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include <cusolverDn.h>
#include <vector>
#include <stdexcept>
static cusolverDnHandle_t solverHandle() {
static cusolverDnHandle_t h = nullptr;
if (h == nullptr) {
if (cusolverDnCreate(&h) != CUSOLVER_STATUS_SUCCESS)
throw std::runtime_error("cusolverDnCreate failed");
}
return h;
}
// Batched symmetric eigensolver (any n) via cusolverDnXsyevBatched.
// A [N,n,n] FP32 symmetric (row-major == col-major up to triangle choice).
// Returns {V [N,n,n] eigenvectors as COLUMNS, W [N,n] ascending}.
std::vector<torch::Tensor> batched_syev(torch::Tensor A) {
int64_t N = A.size(0), n = A.size(1);
auto Ac = A.clone().contiguous(); // overwritten with eigenvectors (col-major)
auto W = torch::empty({N, n}, A.options());
auto info = torch::empty({N}, torch::dtype(torch::kInt32).device(A.device()));
cusolverDnHandle_t h = solverHandle();
cusolverDnParams_t params;
cusolverDnCreateParams(¶ms);
size_t dWorkBytes = 0, hWorkBytes = 0;
cusolverStatus_t st = cusolverDnXsyevBatched_bufferSize(
h, params, CUSOLVER_EIG_MODE_VECTOR, CUBLAS_FILL_MODE_LOWER, n,
CUDA_R_32F, Ac.data_ptr<float>(), n,
CUDA_R_32F, W.data_ptr<float>(),
CUDA_R_32F, &dWorkBytes, &hWorkBytes, N);
if (st != CUSOLVER_STATUS_SUCCESS)
throw std::runtime_error("Xsyev_bufferSize failed: " + std::to_string((int)st));
auto dbuf = torch::empty({(int64_t)dWorkBytes},
torch::dtype(torch::kUInt8).device(A.device()));
std::vector<uint8_t> hbuf(hWorkBytes);
st = cusolverDnXsyevBatched(
h, params, CUSOLVER_EIG_MODE_VECTOR, CUBLAS_FILL_MODE_LOWER, n,
CUDA_R_32F, Ac.data_ptr<float>(), n,
CUDA_R_32F, W.data_ptr<float>(),
CUDA_R_32F, dbuf.data_ptr(), dWorkBytes,
hWorkBytes ? hbuf.data() : nullptr, hWorkBytes,
info.data_ptr<int>(), N);
cusolverDnDestroyParams(params);
if (st != CUSOLVER_STATUS_SUCCESS)
throw std::runtime_error("Xsyev failed: " + std::to_string((int)st));
auto V = Ac.transpose(-1, -2).contiguous(); // col-major eigvec cols -> our columns
return {V, W};
}
"""
_CPP_SRC = "std::vector<torch::Tensor> batched_syev(torch::Tensor A);"
_module = None
try:
_module = load_inline(
name="eigh_v7_xsyev", cpp_sources=[_CPP_SRC], cuda_sources=[_CUDA_SRC],
functions=["batched_syev"], extra_cuda_cflags=["-O3"],
extra_ldflags=["-lcusolver"], verbose=True)
print("[eigh] load_inline build OK (Xsyev available)")
except Exception as exc:
print(f"[eigh] build FAILED, fallback everywhere: {exc!r}")
def _fallback(data):
L, Q = torch.linalg.eigh(data)
return Q, L
def custom_kernel(data: input_t) -> output_t:
# Dispatch: Xsyev batched wins for 176<=n<=1024 (big batch amortizes its
# per-matrix Jacobi cost). At n<=32 looped syevd launch is cheaper (Xsyev 2x
# slower); at n>=2048 batch is too small to benefit. Fall back there.
n = data.size(-1)
if _module is not None and 176 <= n <= 1024:
try:
V, W = _module.batched_syev(data.contiguous())
if torch.isfinite(V).all() and torch.isfinite(W).all():
return V, W
print("[eigh] non-finite from Xsyev, fallback")
except Exception as exc:
print(f"[eigh] Xsyev raised, fallback: {exc!r}")
return _fallback(data)
scrolls · 113 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON