submission 850559
kookiesnkareem · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 139 lines, June 9 Researcher Reciprocity License v1.0.
sub_002.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-850559?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:694440561d13498423d26f7a09701f82f9d6a2956a2ce9bfd763a5d16ccea30b
license declaredunknown
license concludedunknown
authorskookiesnkareem
imported2026-08-26
Kernel source
sub_002.py139 lines
#!POPCORN leaderboard eigh
#!POPCORN gpu B200
import ctypes
from ctypes import POINTER, byref, c_int, c_int64, c_size_t, c_void_p
import torch
from task import input_t, output_t
JOBZ_VECTOR = 1 # CUSOLVER_EIG_MODE_VECTOR
UPLO_LOWER = 0 # CUBLAS_FILL_MODE_LOWER
CUDA_R_32F = 0
_solver = None
_solver_failed = False
class _XsyevBatched:
def __init__(self):
torch.linalg.eigh(torch.eye(2, device="cuda"))
torch.cuda.synchronize()
self.lib = self._load_lib()
self._declare()
self.handle = c_void_p()
self._check(self.lib.cusolverDnCreate(byref(self.handle)), "create")
self.params = c_void_p()
self._check(self.lib.cusolverDnCreateParams(byref(self.params)), "params")
self._buffers = {} # (n, batch) -> (device_ws, host_ws, dev_bytes, host_bytes)
self._info = {} # batch -> device int32 tensor (never returned)
@staticmethod
def _load_lib():
paths = []
try:
with open("/proc/self/maps") as f:
for line in f:
if "libcusolver.so" in line:
idx = line.find("/")
if idx != -1:
path = line[idx:].strip()
if path not in paths:
paths.append(path)
except OSError:
pass
for path in paths:
try:
return ctypes.CDLL(path)
except OSError:
continue
return ctypes.CDLL("libcusolver.so")
def _declare(self):
lib = self.lib
lib.cusolverDnCreate.argtypes = [POINTER(c_void_p)]
lib.cusolverDnCreate.restype = c_int
lib.cusolverDnCreateParams.argtypes = [POINTER(c_void_p)]
lib.cusolverDnCreateParams.restype = c_int
lib.cusolverDnXsyevBatched_bufferSize.argtypes = [
c_void_p, c_void_p, c_int, c_int, c_int64,
c_int, c_void_p, c_int64,
c_int, c_void_p, c_int,
POINTER(c_size_t), POINTER(c_size_t), c_int64,
]
lib.cusolverDnXsyevBatched_bufferSize.restype = c_int
lib.cusolverDnXsyevBatched.argtypes = [
c_void_p, c_void_p, c_int, c_int, c_int64,
c_int, c_void_p, c_int64,
c_int, c_void_p, c_int,
c_void_p, c_size_t, c_void_p, c_size_t,
c_void_p, c_int64,
]
lib.cusolverDnXsyevBatched.restype = c_int
@staticmethod
def _check(status, what):
if status != 0:
raise RuntimeError(f"cusolver {what} failed with status {status}")
def solve(self, a):
batch, n, _ = a.shape
v = a.clone()
w = torch.empty((batch, n), dtype=torch.float32, device=a.device)
info = self._info.get(batch)
if info is None:
info = torch.empty(batch, dtype=torch.int32, device=a.device)
self._info[batch] = info
key = (n, batch)
cached = self._buffers.get(key)
if cached is None:
dev_bytes = c_size_t(0)
host_bytes = c_size_t(0)
self._check(
self.lib.cusolverDnXsyevBatched_bufferSize(
self.handle, self.params, JOBZ_VECTOR, UPLO_LOWER, n,
CUDA_R_32F, c_void_p(v.data_ptr()), n,
CUDA_R_32F, c_void_p(w.data_ptr()), CUDA_R_32F,
byref(dev_bytes), byref(host_bytes), batch,
),
"buffer_size",
)
dev_ws = torch.empty(
max(int(dev_bytes.value), 1), dtype=torch.uint8, device=a.device
)
host_ws = torch.empty(max(int(host_bytes.value), 1), dtype=torch.uint8)
cached = (dev_ws, host_ws, int(dev_bytes.value), int(host_bytes.value))
self._buffers[key] = cached
dev_ws, host_ws, dev_bytes_val, host_bytes_val = cached
self._check(
self.lib.cusolverDnXsyevBatched(
self.handle, self.params, JOBZ_VECTOR, UPLO_LOWER, n,
CUDA_R_32F, c_void_p(v.data_ptr()), n,
CUDA_R_32F, c_void_p(w.data_ptr()), CUDA_R_32F,
c_void_p(dev_ws.data_ptr()), dev_bytes_val,
c_void_p(host_ws.data_ptr()), host_bytes_val,
c_void_p(info.data_ptr()), batch,
),
"xsyev_batched",
)
return v.mT, w
def custom_kernel(data: input_t) -> output_t:
global _solver, _solver_failed
if data.is_cuda and not _solver_failed:
try:
if _solver is None:
_solver = _XsyevBatched()
except Exception:
_solver_failed = True
if _solver is not None:
try:
return _solver.solve(data)
except Exception:
_solver_failed = True
values, vectors = torch.linalg.eigh(data)
return vectors, values
scrolls · 139 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON