submission 833218
.satan_99 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 102 lines, June 9 Researcher Reciprocity License v1.0.
submission_v32.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-qr-v2-833218?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:2ee44e5345df5e685c2deddfb949f5902059027b17d1f4aec7d4508736ae7092
license declaredunknown
license concludedunknown
authors.satan_99
imported2026-08-26
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
stages = 1
num_stages=1,Kernel source
submission_v32.py102 lines
import torch
import triton
import triton.language as tl
# Exact shape -> launch params.
# N is compile-time specialized per case.
LAUNCH = {
32: (32, 4),
176: (64, 4),
352: (128, 4),
512: (128, 8),
1024: (128, 8),
2048: (256, 8),
4096: (256, 8),
}
@triton.jit
def _geqrf_2xn_kernel(
x_ptr,
out_ptr,
tau_ptr,
stride_b,
N: tl.constexpr,
BLOCK_N: tl.constexpr,
):
pid = tl.program_id(0)
in_base = x_ptr + pid * stride_b
out_base = out_ptr + pid * stride_b
tau_base = tau_ptr + pid * 2
row0_in = in_base
row1_in = in_base + N
row0_out = out_base
row1_out = out_base + N
# First column reflector.
a00 = tl.load(row0_in).to(tl.float32)
a10 = tl.load(row1_in).to(tl.float32)
norm = tl.sqrt(a00 * a00 + a10 * a10)
beta = tl.where(a00 >= 0.0, -norm, norm)
zero = a10 == 0.0
t = tl.where(zero, 0.0, (beta - a00) / beta)
v2 = tl.where(zero, 0.0, a10 / (a00 - beta))
tl.store(row0_out, beta)
tl.store(row1_out, v2)
tl.store(tau_base + 0, t)
tl.store(tau_base + 1, 0.0)
tv2 = t * v2
offs = tl.arange(0, BLOCK_N)
# Exact-N tile loop, aggressively unrolled by the compiler.
for c0 in tl.static_range(1, N, BLOCK_N):
cols = c0 + offs
mask = cols < N
x0 = tl.load(row0_in + cols, mask=mask, other=0.0).to(tl.float32)
x1 = tl.load(row1_in + cols, mask=mask, other=0.0).to(tl.float32)
dot = x0 + v2 * x1
y0 = x0 - t * dot
y1 = x1 - tv2 * dot
tl.store(row0_out + cols, y0, mask=mask)
tl.store(row1_out + cols, y1, mask=mask)
def _launch_2xn(x, out, tau, n):
block_n, num_warps = LAUNCH[n]
_geqrf_2xn_kernel[(x.shape[0],)](
x, out, tau,
x.stride(0),
N=n,
BLOCK_N=block_n,
num_warps=num_warps,
num_stages=1,
)
def custom_kernel(data: torch.Tensor):
x = data
batch, cond, n = x.shape
if cond <= 1:
return x.clone(), x.new_zeros((batch, cond))
if cond == 2:
out = torch.empty_like(x)
tau = torch.empty((batch, 2), device=x.device, dtype=x.dtype)
_launch_2xn(x, out, tau, n)
return out, tau
return torch.geqrf(x)
scrolls · 102 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON