submission 839013
tusharhq · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 62 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-qr-v2-839013?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:2603e0ef7e41d9813ae8c341ee4e9db44a9fc199a7a24672bf9581c936cc1137
license declaredunknown
license concludedunknown
authorstusharhq
imported2026-08-26
Kernel source
submission.py62 lines
#!POPCORN leaderboard qr_v2
#!POPCORN gpu B200
import torch
from task import input_t, output_t
def _rect_qr_embed(a: torch.Tensor, k: int, project_tail: bool = False) -> output_t:
batch, n, _ = a.shape
h_rect, tau_rect = torch.geqrf(a[:, :, :k].contiguous())
h = torch.zeros_like(a)
tau = a.new_zeros((batch, n))
h[:, :, :k] = h_rect
tau[:, :k] = tau_rect
if project_tail and k < n:
r_tail = torch.ormqr(h_rect, tau_rect, a[:, :, k:].contiguous(), left=True, transpose=True)
h[:, :, k:] = torch.triu(r_tail, diagonal=-k)
return h, tau
def custom_kernel(data: input_t) -> output_t:
batch, n, _ = data.shape
if n == 512:
rank_tail = data[:, :, 384:].abs().amax(dim=(1, 2))
if rank_tail.amax().item() == 0:
return _rect_qr_embed(data, 384)
cluster_tail = data[:, :, 260:].abs().amax(dim=(1, 2))
if cluster_tail.amax().item() < 1.0e-4:
return _rect_qr_embed(data, 254)
rank_mask = rank_tail == 0
cluster_mask = cluster_tail < 1.0e-4
structured_mask = rank_mask | cluster_mask
if structured_mask.any().item():
h = torch.empty_like(data)
tau = data.new_empty((batch, n))
if rank_mask.any().item():
h_rank, tau_rank = _rect_qr_embed(data[rank_mask].contiguous(), 384)
h[rank_mask] = h_rank
tau[rank_mask] = tau_rank
cluster_only = cluster_mask & ~rank_mask
if cluster_only.any().item():
h_cluster, tau_cluster = _rect_qr_embed(data[cluster_only].contiguous(), 254)
h[cluster_only] = h_cluster
tau[cluster_only] = tau_cluster
dense_mask = ~structured_mask
if dense_mask.any().item():
h_dense, tau_dense = torch.geqrf(data[dense_mask].contiguous())
h[dense_mask] = h_dense
tau[dense_mask] = tau_dense
return h, tau
if n == 1024:
if (data[:, :, 768:] - data[:, :, :256]).abs().amax().item() < 1.0e-3:
return _rect_qr_embed(data, 768, project_tail=True)
return torch.geqrf(data)
scrolls · 62 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON