submission 554279
Ayush10 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 105 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-causal-conv1d-554279?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:e05d8c8993d616743bf9d4d1e6ea07c14cd5efdf1955e8be46861192e400fd9a
license declaredunknown
license concludedunknown
authorsAyush10
imported2026-08-15
Techniques
Extracted from the mirrored source by pattern, never inferred. Each row cites its line.
num-warps = 4
3: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),stages = 2
3: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),Kernel source
submission.py105 lines
#!POPCORN leaderboard causal_conv1d
#!POPCORN gpu B200_Nebius
# Team: Kernal Forge
# Optimizations: per-W tuned configs, lazy compilation
from task import input_t, output_t
import torch
import helion
import helion.language as hl
W_CONFIGS: dict[int, helion.Config] = {
3: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
4: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
8: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
}
def _make_kernel(config: helion.Config):
@helion.kernel(config=config)
def kernel(
x: torch.Tensor,
w: torch.Tensor,
b: torch.Tensor,
) -> torch.Tensor:
B, D, S = x.shape
W = hl.specialize(w.size(1))
y = torch.empty(B, D, S, dtype=x.dtype, device=x.device)
for rb, rd, rs in hl.tile([B, D, S], block_size=[1, None, None]):
bi = rb.begin
rs_idx = rs.index
in_bounds = rs_idx < S
bias_tile = hl.zeros([rd, rs], dtype=torch.float32)
bias_tile = bias_tile + b[rd][:, None]
if W == 4:
ib_base = in_bounds[None, :]
ge1 = (rs_idx >= 1)
ge2 = (rs_idx >= 2)
ge3 = (rs_idx >= 3)
mask0 = ib_base
mask1 = (in_bounds & ge1)[None, :]
mask2 = (in_bounds & ge2)[None, :]
mask3 = (in_bounds & ge3)[None, :]
x0 = hl.load(x, [bi, rd, rs_idx - 3], extra_mask=mask3)
x1 = hl.load(x, [bi, rd, rs_idx - 2], extra_mask=mask2)
x2 = hl.load(x, [bi, rd, rs_idx - 1], extra_mask=mask1)
x3 = hl.load(x, [bi, rd, rs_idx], extra_mask=mask0)
acc = bias_tile
acc = acc + x0 * w[rd, 0][:, None]
acc = acc + x1 * w[rd, 1][:, None]
acc = acc + x2 * w[rd, 2][:, None]
acc = acc + x3 * w[rd, 3][:, None]
elif W == 3:
ib_base = in_bounds[None, :]
ge1 = (rs_idx >= 1)
ge2 = (rs_idx >= 2)
mask0 = ib_base
mask1 = (in_bounds & ge1)[None, :]
mask2 = (in_bounds & ge2)[None, :]
x0 = hl.load(x, [bi, rd, rs_idx - 2], extra_mask=mask2)
x1 = hl.load(x, [bi, rd, rs_idx - 1], extra_mask=mask1)
x2 = hl.load(x, [bi, rd, rs_idx], extra_mask=mask0)
acc = bias_tile
acc = acc + x0 * w[rd, 0][:, None]
acc = acc + x1 * w[rd, 1][:, None]
acc = acc + x2 * w[rd, 2][:, None]
else:
acc = bias_tile
for tap in range(W):
shift = W - 1 - tap
x_tap = hl.load(
x,
[bi, rd, rs_idx - shift],
extra_mask=(in_bounds & (rs_idx >= shift))[None, :],
)
acc = acc + x_tap * w[rd, tap][:, None]
y[rb, rd, rs] = acc[None, :, :]
return y
return kernel
_KERNEL_CACHE: dict[int, callable] = {}
def _get_kernel(w_val: int):
kernel = _KERNEL_CACHE.get(w_val)
if kernel is None:
kernel = _make_kernel(W_CONFIGS[w_val])
_KERNEL_CACHE[w_val] = kernel
return kernel
def custom_kernel(data: input_t) -> output_t:
x, weight, bias = data
W = weight.shape[1]
kernel = _get_kernel(W)
return kernel(x, weight, bias)
scrolls · 105 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON