Skip to content
KernelIndex
Search⌘K

submission 839013

tusharhq · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 62 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-qr-v2-839013?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVIDIA B200
120.1ms
#438 of 515
2026-06-27

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:2603e0ef7e41d9813ae8c341ee4e9db44a9fc199a7a24672bf9581c936cc1137
license declaredunknown
license concludedunknown
authorstusharhq
imported2026-08-26

Kernel source

submission.py62 lines
#!POPCORN leaderboard qr_v2
#!POPCORN gpu B200

import torch
from task import input_t, output_t


def _rect_qr_embed(a: torch.Tensor, k: int, project_tail: bool = False) -> output_t:
    batch, n, _ = a.shape
    h_rect, tau_rect = torch.geqrf(a[:, :, :k].contiguous())

    h = torch.zeros_like(a)
    tau = a.new_zeros((batch, n))
    h[:, :, :k] = h_rect
    tau[:, :k] = tau_rect

    if project_tail and k < n:
        r_tail = torch.ormqr(h_rect, tau_rect, a[:, :, k:].contiguous(), left=True, transpose=True)
        h[:, :, k:] = torch.triu(r_tail, diagonal=-k)

    return h, tau


def custom_kernel(data: input_t) -> output_t:
    batch, n, _ = data.shape

    if n == 512:
        rank_tail = data[:, :, 384:].abs().amax(dim=(1, 2))
        if rank_tail.amax().item() == 0:
            return _rect_qr_embed(data, 384)
        cluster_tail = data[:, :, 260:].abs().amax(dim=(1, 2))
        if cluster_tail.amax().item() < 1.0e-4:
            return _rect_qr_embed(data, 254)

        rank_mask = rank_tail == 0
        cluster_mask = cluster_tail < 1.0e-4
        structured_mask = rank_mask | cluster_mask
        if structured_mask.any().item():
            h = torch.empty_like(data)
            tau = data.new_empty((batch, n))
            if rank_mask.any().item():
                h_rank, tau_rank = _rect_qr_embed(data[rank_mask].contiguous(), 384)
                h[rank_mask] = h_rank
                tau[rank_mask] = tau_rank
            cluster_only = cluster_mask & ~rank_mask
            if cluster_only.any().item():
                h_cluster, tau_cluster = _rect_qr_embed(data[cluster_only].contiguous(), 254)
                h[cluster_only] = h_cluster
                tau[cluster_only] = tau_cluster
            dense_mask = ~structured_mask
            if dense_mask.any().item():
                h_dense, tau_dense = torch.geqrf(data[dense_mask].contiguous())
                h[dense_mask] = h_dense
                tau[dense_mask] = tau_dense
            return h, tau

    if n == 1024:
        if (data[:, :, 768:] - data[:, :, :256]).abs().amax().item() < 1.0e-3:
            return _rect_qr_embed(data, 768, project_tail=True)

    return torch.geqrf(data)
scrolls · 62 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON