submission 837255
ayushnangia · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 105 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-qr-v2-837255?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:5ed5e745bc6eb450725ad2995f500ef9fbe3bb836958f59afcfd4c52a8246efd
license declaredunknown
license concludedunknown
authorsayushnangia
imported2026-08-26
Kernel source
submission.py105 lines
#!POPCORN leaderboard qr_v2
#!POPCORN gpu B200
import torch
from task import input_t, output_t
def _larft_from_panel(panel: torch.Tensor, tau: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]:
"""
Build V,T for a compact Householder panel.
panel: (batch, m, b), output from torch.geqrf on the panel.
The lower triangle stores Householder tails; diagonal is R.
tau: (batch, b)
Returns:
V: (batch, m, b), explicit Householder vectors with implicit 1s inserted.
T: (batch, b, b), upper triangular WY factor such that
H = I - V T V.T = H_0 H_1 ... H_{b-1}
"""
batch, m, b = panel.shape
device = panel.device
dtype = panel.dtype
V = torch.tril(panel, diagonal=-1)
idx = torch.arange(b, device=device)
V[:, idx, idx] = 1.0
T = torch.zeros((batch, b, b), device=device, dtype=dtype)
for j in range(b):
tau_j = tau[:, j]
T[:, j, j] = tau_j
if j == 0:
continue
# t = -tau_j * V_j_previous.T @ v_j
vj = V[:, j:, j:j + 1] # (batch, m-j, 1)
Vprev = V[:, j:, :j] # (batch, m-j, j)
t = torch.bmm(Vprev.transpose(1, 2), vj).squeeze(-1)
t = -tau_j[:, None] * t
# t = T_previous @ t
t = torch.bmm(T[:, :j, :j], t.unsqueeze(-1)).squeeze(-1)
T[:, :j, j] = t
return V, T
def _blocked_geqrf(data: torch.Tensor, block: int) -> output_t:
A = data.clone()
batch, n, _ = A.shape
tau_all = torch.empty((batch, n), device=A.device, dtype=A.dtype)
for k in range(0, n, block):
b = min(block, n - k)
# Factor the current panel.
# This produces exactly the compact Householder convention the checker expects.
panel, tau = torch.geqrf(A[:, k:, k:k + b].contiguous())
A[:, k:, k:k + b] = panel
tau_all[:, k:k + b] = tau
# Apply block reflector to the trailing matrix:
# C <- C - V T.T V.T C
if k + b < n:
V, T = _larft_from_panel(panel, tau)
C = A[:, k:, k + b:].contiguous()
W = torch.bmm(V.transpose(1, 2), C)
W = torch.bmm(T.transpose(1, 2), W)
C = C - torch.bmm(V, W)
A[:, k:, k + b:] = C
return A, tau_all
def custom_kernel(data: input_t) -> output_t:
batch, n, _ = data.shape
# Small cases: cuSOLVER/PyTorch overhead is acceptable, and this path is safest.
# For a top submission, replace this with one-CTA custom Householder kernels.
if n <= 352:
return torch.geqrf(data)
# Main QR-v2 shapes.
# B=64 gives good arithmetic intensity for the WY update while keeping panel
# work stable and not too sequential.
if n == 512:
return _blocked_geqrf(data, 64)
if n == 1024:
return _blocked_geqrf(data, 64)
# Larger matrices can use B=64 or B=128.
# B=64 is safer numerically; B=128 reduces launches but makes panel work heavier.
if n >= 2048:
return _blocked_geqrf(data, 64)
return torch.geqrf(data)
scrolls · 105 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON