Skip to content
KernelIndex
Search⌘K

submission 837255

ayushnangia · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 105 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-qr-v2-837255?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVIDIA B200
125.3ms
#449 of 515
2026-06-26

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:5ed5e745bc6eb450725ad2995f500ef9fbe3bb836958f59afcfd4c52a8246efd
license declaredunknown
license concludedunknown
authorsayushnangia
imported2026-08-26

Kernel source

submission.py105 lines
#!POPCORN leaderboard qr_v2
#!POPCORN gpu B200

import torch
from task import input_t, output_t


def _larft_from_panel(panel: torch.Tensor, tau: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]:
    """
    Build V,T for a compact Householder panel.

    panel: (batch, m, b), output from torch.geqrf on the panel.
           The lower triangle stores Householder tails; diagonal is R.
    tau:   (batch, b)

    Returns:
      V: (batch, m, b), explicit Householder vectors with implicit 1s inserted.
      T: (batch, b, b), upper triangular WY factor such that
         H = I - V T V.T = H_0 H_1 ... H_{b-1}
    """
    batch, m, b = panel.shape
    device = panel.device
    dtype = panel.dtype

    V = torch.tril(panel, diagonal=-1)
    idx = torch.arange(b, device=device)
    V[:, idx, idx] = 1.0

    T = torch.zeros((batch, b, b), device=device, dtype=dtype)

    for j in range(b):
        tau_j = tau[:, j]
        T[:, j, j] = tau_j

        if j == 0:
            continue

        # t = -tau_j * V_j_previous.T @ v_j
        vj = V[:, j:, j:j + 1]          # (batch, m-j, 1)
        Vprev = V[:, j:, :j]            # (batch, m-j, j)

        t = torch.bmm(Vprev.transpose(1, 2), vj).squeeze(-1)
        t = -tau_j[:, None] * t

        # t = T_previous @ t
        t = torch.bmm(T[:, :j, :j], t.unsqueeze(-1)).squeeze(-1)
        T[:, :j, j] = t

    return V, T


def _blocked_geqrf(data: torch.Tensor, block: int) -> output_t:
    A = data.clone()
    batch, n, _ = A.shape
    tau_all = torch.empty((batch, n), device=A.device, dtype=A.dtype)

    for k in range(0, n, block):
        b = min(block, n - k)

        # Factor the current panel.
        # This produces exactly the compact Householder convention the checker expects.
        panel, tau = torch.geqrf(A[:, k:, k:k + b].contiguous())

        A[:, k:, k:k + b] = panel
        tau_all[:, k:k + b] = tau

        # Apply block reflector to the trailing matrix:
        # C <- C - V T.T V.T C
        if k + b < n:
            V, T = _larft_from_panel(panel, tau)
            C = A[:, k:, k + b:].contiguous()

            W = torch.bmm(V.transpose(1, 2), C)
            W = torch.bmm(T.transpose(1, 2), W)
            C = C - torch.bmm(V, W)

            A[:, k:, k + b:] = C

    return A, tau_all


def custom_kernel(data: input_t) -> output_t:
    batch, n, _ = data.shape

    # Small cases: cuSOLVER/PyTorch overhead is acceptable, and this path is safest.
    # For a top submission, replace this with one-CTA custom Householder kernels.
    if n <= 352:
        return torch.geqrf(data)

    # Main QR-v2 shapes.
    # B=64 gives good arithmetic intensity for the WY update while keeping panel
    # work stable and not too sequential.
    if n == 512:
        return _blocked_geqrf(data, 64)

    if n == 1024:
        return _blocked_geqrf(data, 64)

    # Larger matrices can use B=64 or B=128.
    # B=64 is safer numerically; B=128 reduces launches but makes panel work heavier.
    if n >= 2048:
        return _blocked_geqrf(data, 64)

    return torch.geqrf(data)
scrolls · 105 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON