Skip to content
KernelIndex
Search⌘K

submission 845307

Lorenzo · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 113 lines, June 9 Researcher Reciprocity License v1.0.

submission_7.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-845307?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVIDIA B200
48.4ms
#143 of 286
2026-06-30

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:d58a197b46635387198ac598faabccede681c138d88fb71213d8b60db6c7cfb3
license declaredunknown
license concludedunknown
authorsLorenzo
imported2026-08-26

Kernel source

submission_7.py113 lines
#!POPCORN leaderboard eigh
#!POPCORN gpu B200
"""eigh_py submission_7 — BATCHED-SYEV PROBE.

Hypothesis: torch.linalg.eigh loops cusolverDnSsyevd per matrix (SMs idle at big
batch). CUDA 12.6+ ships cusolverDnXsyevBatched — a *batched* dense symmetric
eigensolver with NO n<=32 limit. If it is genuinely batch-parallel, calling it
directly should beat the looped path with ZERO custom math.

This file routes EVERY shape through cusolverDnXsyevBatched (jobz=VECTOR), with
a per-call finite-guard + exception fallback to torch.linalg.eigh. If the symbol
is unavailable (older toolkit) the build fails and we fall back everywhere -> the
probe still runs (== baseline) and we learn the API isn't present.

Output contract: (Q, L), Q = eigenvector COLUMNS, L ascending (syev gives both).
"""
import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t

torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.allow_tf32 = False

_CUDA_SRC = r"""
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include <cusolverDn.h>
#include <vector>
#include <stdexcept>

static cusolverDnHandle_t solverHandle() {
    static cusolverDnHandle_t h = nullptr;
    if (h == nullptr) {
        if (cusolverDnCreate(&h) != CUSOLVER_STATUS_SUCCESS)
            throw std::runtime_error("cusolverDnCreate failed");
    }
    return h;
}

// Batched symmetric eigensolver (any n) via cusolverDnXsyevBatched.
// A [N,n,n] FP32 symmetric (row-major == col-major up to triangle choice).
// Returns {V [N,n,n] eigenvectors as COLUMNS, W [N,n] ascending}.
std::vector<torch::Tensor> batched_syev(torch::Tensor A) {
    int64_t N = A.size(0), n = A.size(1);
    auto Ac = A.clone().contiguous();          // overwritten with eigenvectors (col-major)
    auto W = torch::empty({N, n}, A.options());
    auto info = torch::empty({N}, torch::dtype(torch::kInt32).device(A.device()));
    cusolverDnHandle_t h = solverHandle();
    cusolverDnParams_t params;
    cusolverDnCreateParams(&params);

    size_t dWorkBytes = 0, hWorkBytes = 0;
    cusolverStatus_t st = cusolverDnXsyevBatched_bufferSize(
        h, params, CUSOLVER_EIG_MODE_VECTOR, CUBLAS_FILL_MODE_LOWER, n,
        CUDA_R_32F, Ac.data_ptr<float>(), n,
        CUDA_R_32F, W.data_ptr<float>(),
        CUDA_R_32F, &dWorkBytes, &hWorkBytes, N);
    if (st != CUSOLVER_STATUS_SUCCESS)
        throw std::runtime_error("Xsyev_bufferSize failed: " + std::to_string((int)st));

    auto dbuf = torch::empty({(int64_t)dWorkBytes},
                             torch::dtype(torch::kUInt8).device(A.device()));
    std::vector<uint8_t> hbuf(hWorkBytes);

    st = cusolverDnXsyevBatched(
        h, params, CUSOLVER_EIG_MODE_VECTOR, CUBLAS_FILL_MODE_LOWER, n,
        CUDA_R_32F, Ac.data_ptr<float>(), n,
        CUDA_R_32F, W.data_ptr<float>(),
        CUDA_R_32F, dbuf.data_ptr(), dWorkBytes,
        hWorkBytes ? hbuf.data() : nullptr, hWorkBytes,
        info.data_ptr<int>(), N);
    cusolverDnDestroyParams(params);
    if (st != CUSOLVER_STATUS_SUCCESS)
        throw std::runtime_error("Xsyev failed: " + std::to_string((int)st));

    auto V = Ac.transpose(-1, -2).contiguous();   // col-major eigvec cols -> our columns
    return {V, W};
}
"""

_CPP_SRC = "std::vector<torch::Tensor> batched_syev(torch::Tensor A);"

_module = None
try:
    _module = load_inline(
        name="eigh_v7_xsyev", cpp_sources=[_CPP_SRC], cuda_sources=[_CUDA_SRC],
        functions=["batched_syev"], extra_cuda_cflags=["-O3"],
        extra_ldflags=["-lcusolver"], verbose=True)
    print("[eigh] load_inline build OK (Xsyev available)")
except Exception as exc:
    print(f"[eigh] build FAILED, fallback everywhere: {exc!r}")


def _fallback(data):
    L, Q = torch.linalg.eigh(data)
    return Q, L


def custom_kernel(data: input_t) -> output_t:
    # Dispatch: Xsyev batched wins for 176<=n<=1024 (big batch amortizes its
    # per-matrix Jacobi cost). At n<=32 looped syevd launch is cheaper (Xsyev 2x
    # slower); at n>=2048 batch is too small to benefit. Fall back there.
    n = data.size(-1)
    if _module is not None and 176 <= n <= 1024:
        try:
            V, W = _module.batched_syev(data.contiguous())
            if torch.isfinite(V).all() and torch.isfinite(W).all():
                return V, W
            print("[eigh] non-finite from Xsyev, fallback")
        except Exception as exc:
            print(f"[eigh] Xsyev raised, fallback: {exc!r}")
    return _fallback(data)
scrolls · 113 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON