Skip to content
KernelIndex
Search⌘K

submission 850559

kookiesnkareem · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 139 lines, June 9 Researcher Reciprocity License v1.0.

sub_002.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-850559?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVIDIA B200
49.4ms
#166 of 286
2026-07-02

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:694440561d13498423d26f7a09701f82f9d6a2956a2ce9bfd763a5d16ccea30b
license declaredunknown
license concludedunknown
authorskookiesnkareem
imported2026-08-26

Kernel source

sub_002.py139 lines
#!POPCORN leaderboard eigh
#!POPCORN gpu B200

import ctypes
from ctypes import POINTER, byref, c_int, c_int64, c_size_t, c_void_p

import torch
from task import input_t, output_t

JOBZ_VECTOR = 1   # CUSOLVER_EIG_MODE_VECTOR
UPLO_LOWER = 0    # CUBLAS_FILL_MODE_LOWER
CUDA_R_32F = 0

_solver = None
_solver_failed = False


class _XsyevBatched:
    def __init__(self):
        torch.linalg.eigh(torch.eye(2, device="cuda"))
        torch.cuda.synchronize()
        self.lib = self._load_lib()
        self._declare()
        self.handle = c_void_p()
        self._check(self.lib.cusolverDnCreate(byref(self.handle)), "create")
        self.params = c_void_p()
        self._check(self.lib.cusolverDnCreateParams(byref(self.params)), "params")
        self._buffers = {}  # (n, batch) -> (device_ws, host_ws, dev_bytes, host_bytes)
        self._info = {}     # batch -> device int32 tensor (never returned)

    @staticmethod
    def _load_lib():
        paths = []
        try:
            with open("/proc/self/maps") as f:
                for line in f:
                    if "libcusolver.so" in line:
                        idx = line.find("/")
                        if idx != -1:
                            path = line[idx:].strip()
                            if path not in paths:
                                paths.append(path)
        except OSError:
            pass
        for path in paths:
            try:
                return ctypes.CDLL(path)
            except OSError:
                continue
        return ctypes.CDLL("libcusolver.so")

    def _declare(self):
        lib = self.lib
        lib.cusolverDnCreate.argtypes = [POINTER(c_void_p)]
        lib.cusolverDnCreate.restype = c_int
        lib.cusolverDnCreateParams.argtypes = [POINTER(c_void_p)]
        lib.cusolverDnCreateParams.restype = c_int
        lib.cusolverDnXsyevBatched_bufferSize.argtypes = [
            c_void_p, c_void_p, c_int, c_int, c_int64,
            c_int, c_void_p, c_int64,
            c_int, c_void_p, c_int,
            POINTER(c_size_t), POINTER(c_size_t), c_int64,
        ]
        lib.cusolverDnXsyevBatched_bufferSize.restype = c_int
        lib.cusolverDnXsyevBatched.argtypes = [
            c_void_p, c_void_p, c_int, c_int, c_int64,
            c_int, c_void_p, c_int64,
            c_int, c_void_p, c_int,
            c_void_p, c_size_t, c_void_p, c_size_t,
            c_void_p, c_int64,
        ]
        lib.cusolverDnXsyevBatched.restype = c_int

    @staticmethod
    def _check(status, what):
        if status != 0:
            raise RuntimeError(f"cusolver {what} failed with status {status}")

    def solve(self, a):
        batch, n, _ = a.shape
        v = a.clone()
        w = torch.empty((batch, n), dtype=torch.float32, device=a.device)
        info = self._info.get(batch)
        if info is None:
            info = torch.empty(batch, dtype=torch.int32, device=a.device)
            self._info[batch] = info

        key = (n, batch)
        cached = self._buffers.get(key)
        if cached is None:
            dev_bytes = c_size_t(0)
            host_bytes = c_size_t(0)
            self._check(
                self.lib.cusolverDnXsyevBatched_bufferSize(
                    self.handle, self.params, JOBZ_VECTOR, UPLO_LOWER, n,
                    CUDA_R_32F, c_void_p(v.data_ptr()), n,
                    CUDA_R_32F, c_void_p(w.data_ptr()), CUDA_R_32F,
                    byref(dev_bytes), byref(host_bytes), batch,
                ),
                "buffer_size",
            )
            dev_ws = torch.empty(
                max(int(dev_bytes.value), 1), dtype=torch.uint8, device=a.device
            )
            host_ws = torch.empty(max(int(host_bytes.value), 1), dtype=torch.uint8)
            cached = (dev_ws, host_ws, int(dev_bytes.value), int(host_bytes.value))
            self._buffers[key] = cached
        dev_ws, host_ws, dev_bytes_val, host_bytes_val = cached

        self._check(
            self.lib.cusolverDnXsyevBatched(
                self.handle, self.params, JOBZ_VECTOR, UPLO_LOWER, n,
                CUDA_R_32F, c_void_p(v.data_ptr()), n,
                CUDA_R_32F, c_void_p(w.data_ptr()), CUDA_R_32F,
                c_void_p(dev_ws.data_ptr()), dev_bytes_val,
                c_void_p(host_ws.data_ptr()), host_bytes_val,
                c_void_p(info.data_ptr()), batch,
            ),
            "xsyev_batched",
        )
        return v.mT, w


def custom_kernel(data: input_t) -> output_t:
    global _solver, _solver_failed
    if data.is_cuda and not _solver_failed:
        try:
            if _solver is None:
                _solver = _XsyevBatched()
        except Exception:
            _solver_failed = True
        if _solver is not None:
            try:
                return _solver.solve(data)
            except Exception:
                _solver_failed = True
    values, vectors = torch.linalg.eigh(data)
    return vectors, values
scrolls · 139 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON