Skip to content
KernelIndex
Search⌘K

submission 855396

misha_antonenko · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 120 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-eigh-855396?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVIDIA B200
51.0ms
#185 of 286
2026-07-04

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:33a802ee756096a22cdef51fc87729f15b4bf6a577cb10543a2b6c11858320a0
license declaredunknown
license concludedunknown
authorsmisha_antonenko
imported2026-08-26

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

shared-memory__shared__ float tile[TILE][TILE + 1];

Kernel source

submission.py120 lines
#!POPCORN leaderboard eigh
#!POPCORN gpu B200

import torch
from torch.utils.cpp_extension import load_inline
from task import input_t, output_t

CUDA_SRC = r"""
#include <cusolverDn.h>
#include <cublas_v2.h>
#include <cuda_runtime.h>
#include <torch/extension.h>
#include <stdexcept>
#include <algorithm>

#define TILE 32

__global__ void transpose_col2row_kernel(
    const float* __restrict__ src,
    float* __restrict__ dst,
    int n, int batch)
{
    __shared__ float tile[TILE][TILE + 1];
    int b  = blockIdx.z;
    int bi = blockIdx.y * TILE;
    int bj = blockIdx.x * TILE;
    int tx = threadIdx.x;
    int ty = threadIdx.y;

    int src_i = bi + tx;
    int src_j = bj + ty;
    if (src_i < n && src_j < n)
        tile[ty][tx] = src[b * n * n + src_j * n + src_i];
    __syncthreads();

    int dst_i = bi + ty;
    int dst_j = bj + tx;
    if (dst_i < n && dst_j < n)
        dst[b * n * n + dst_i * n + dst_j] = tile[tx][ty];
}

static cusolverDnHandle_t g_handle = nullptr;

static cusolverDnHandle_t get_handle() {
    if (!g_handle) cusolverDnCreate(&g_handle);
    return g_handle;
}

std::tuple<torch::Tensor, torch::Tensor> eigh_solve(torch::Tensor input) {
    auto A = input.contiguous().clone();
    int batch = A.size(0);
    int n     = A.size(1);
    auto eigenvalues = torch::empty({batch, n}, A.options());
    auto handle = get_handle();

    cusolverDnParams_t params;
    cusolverDnCreateParams(&params);

    size_t lwork_device = 0;
    size_t lwork_host = 0;

    cusolverDnXsyevBatched_bufferSize(
        handle, params,
        CUSOLVER_EIG_MODE_VECTOR, CUBLAS_FILL_MODE_UPPER,
        (int64_t)n, CUDA_R_32F, A.data_ptr<float>(), (int64_t)n,
        CUDA_R_32F, eigenvalues.data_ptr<float>(), CUDA_R_32F,
        &lwork_device, &lwork_host, (int64_t)batch);

    auto work_device = torch::empty({(int64_t)lwork_device}, A.options().dtype(torch::kUInt8));
    std::vector<uint8_t> work_host(lwork_host);
    auto info = torch::zeros({batch}, A.options().dtype(torch::kInt32));

    cusolverDnXsyevBatched(
        handle, params,
        CUSOLVER_EIG_MODE_VECTOR, CUBLAS_FILL_MODE_UPPER,
        (int64_t)n, CUDA_R_32F, A.data_ptr<float>(), (int64_t)n,
        CUDA_R_32F, eigenvalues.data_ptr<float>(), CUDA_R_32F,
        work_device.data_ptr(), lwork_device,
        work_host.data(), lwork_host,
        info.data_ptr<int>(), (int64_t)batch);

    cusolverDnDestroyParams(params);

    auto eigenvectors = torch::empty({batch, n, n}, A.options());
    dim3 block(TILE, TILE);
    dim3 grid((n + TILE - 1) / TILE, (n + TILE - 1) / TILE, batch);
    transpose_col2row_kernel<<<grid, block>>>(
        A.data_ptr<float>(), eigenvectors.data_ptr<float>(), n, batch);
    return std::make_tuple(eigenvectors, eigenvalues);
}

"""

CPP_SRC = """
std::tuple<torch::Tensor, torch::Tensor> eigh_solve(torch::Tensor input);
"""

_module = None
_load_err = ""
import hashlib as _hl
_mod_name = "eigh_" + _hl.md5(CUDA_SRC.encode()).hexdigest()[:8]
try:
    _module = load_inline(
        name=_mod_name,
        cpp_sources=[CPP_SRC],
        cuda_sources=[CUDA_SRC],
        functions=["eigh_solve"],
        verbose=True,
        extra_cuda_cflags=["-O2", "-std=c++17", "-I/usr/local/cuda-12.9/targets/x86_64-linux/include"],
    extra_ldflags=["-lcusolver", "-L/usr/local/cuda-12.9/targets/x86_64-linux/lib"],
    )
except Exception as _e:
    _load_err = str(_e)


def custom_kernel(data: input_t) -> output_t:
    if _module is None:
        raise RuntimeError(f"load_inline failed: {_load_err}")
    return _module.eigh_solve(data)
scrolls · 120 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON