Skip to content
KernelIndex
Search⌘K

submission 873181

d_lolo_ · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 94 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-873181?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVIDIA B200
48.0ms
#133 of 286
2026-07-13

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:6b88459aec0625bca6937a4a9b012ec6c38764397c90d4e5ccba8e0bdf000a82
license declaredunknown
license concludedunknown
authorsd_lolo_
imported2026-08-26

Kernel source

submission.py94 lines
#!POPCORN leaderboard eigh
#!POPCORN gpu B200

"""Batched symmetric eigendecomposition.

torch.linalg.eigh serial-loops the batch for n>32 (a ~200x cliff), because it does
not use the true batched cuSOLVER API. This calls `cusolverDnXsyevBatched` (CUDA 13)
in one shot via a native (load_inline) extension linking libcusolver — computing the
whole batch as a single operation. Returns (Q, L) in eigh convention (Q columns are
eigenvectors, L ascending). Falls back to torch.linalg.eigh if anything is unsupported.
"""

import os
import torch

# Point the JIT compiler at the CUDA toolkit if it isn't already discoverable.
for _c in ("/usr/local/cuda", "/usr/local/cuda-13.0", "/usr/local/cuda-13"):
    if os.path.isdir(_c):
        os.environ.setdefault("CUDA_HOME", _c)
        break

_CU = r'''
#include <torch/extension.h>
#include <cusolverDn.h>
#include <cuda_runtime.h>
#include <vector>

// cuSOLVER runs on the default execution queue (queue 0), which is what the
// harness times and where torch places its work by default — so results and
// timing stay consistent without any explicit queue management.
static cusolverDnHandle_t g_handle = nullptr;
static cusolverDnParams_t g_params = nullptr;

void xsyev_batched(torch::Tensor A, torch::Tensor W, torch::Tensor info) {
    const int64_t B = A.size(0);
    const int64_t n = A.size(1);
    if (g_handle == nullptr) {
        cusolverDnCreate(&g_handle);
        cusolverDnCreateParams(&g_params);
    }
    cusolverEigMode_t jobz = CUSOLVER_EIG_MODE_VECTOR;
    cublasFillMode_t uplo = CUBLAS_FILL_MODE_LOWER;
    size_t ws_dev = 0, ws_host = 0;
    cusolverDnXsyevBatched_bufferSize(
        g_handle, g_params, jobz, uplo, n,
        CUDA_R_32F, A.data_ptr(), n, CUDA_R_32F, W.data_ptr(),
        CUDA_R_32F, &ws_dev, &ws_host, B);
    auto dev_buf = torch::empty({(int64_t)ws_dev}, A.options().dtype(torch::kUInt8));
    std::vector<uint8_t> host_buf(ws_host > 0 ? ws_host : 1);
    cusolverStatus_t st = cusolverDnXsyevBatched(
        g_handle, g_params, jobz, uplo, n,
        CUDA_R_32F, A.data_ptr(), n, CUDA_R_32F, W.data_ptr(),
        CUDA_R_32F, dev_buf.data_ptr(), ws_dev, host_buf.data(), ws_host,
        info.data_ptr<int>(), B);
    TORCH_CHECK(st == CUSOLVER_STATUS_SUCCESS, "xsyev_batched failed: ", (int)st);
}
'''

_ext = None
def _get_ext():
    global _ext
    if _ext is None:
        from torch.utils.cpp_extension import load_inline
        _ext = load_inline(
            name="eigh_xsyev_ext",
            cpp_sources="void xsyev_batched(torch::Tensor A, torch::Tensor W, torch::Tensor info);",
            cuda_sources=_CU,
            functions=["xsyev_batched"],
            extra_ldflags=["-lcusolver"],
            verbose=False,
        )
    return _ext


def _batched_eigh(A):
    B, n, _ = A.shape
    a = A.contiguous().clone()                       # overwritten with eigenvectors
    W = torch.empty(B, n, dtype=torch.float32, device=A.device)
    info = torch.empty(B, dtype=torch.int32, device=A.device)
    _get_ext().xsyev_batched(a, W, info)
    return a.transpose(-1, -2), W                     # col-major cols -> our columns


def custom_kernel(data):
    A = data
    if not (A.is_cuda and A.dtype == torch.float32 and A.dim() == 3 and A.size(1) == A.size(2)):
        values, vectors = torch.linalg.eigh(A)
        return vectors, values
    try:
        return _batched_eigh(A)
    except Exception:
        values, vectors = torch.linalg.eigh(A)
        return vectors, values
scrolls · 94 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON