submission 555480
Ayush10 · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 75 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-causal-conv1d-555480?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:909167aa105dda6c72d3f720784d871bb9ba7589fa9423c9991c533bf8504129
license declaredunknown
license concludedunknown
authorsAyush10
imported2026-08-15
Kernel source
submission.py75 lines
#!POPCORN leaderboard causal_conv1d
#!POPCORN gpu B200_Nebius
from task import input_t, output_t
import torch
import triton
import triton.language as tl
@triton.jit
def _causal_conv1d(
x_ptr, w_ptr, b_ptr, y_ptr,
D, S,
W: tl.constexpr,
BLOCK_S: tl.constexpr,
):
pid_s = tl.program_id(0)
pid_bd = tl.program_id(1) # batch*D index
d = pid_bd % D
b_val = tl.load(b_ptr + d)
x_row = x_ptr + pid_bd * S
y_row = y_ptr + pid_bd * S
s_off = pid_s * BLOCK_S + tl.arange(0, BLOCK_S)
s_mask = s_off < S
acc = tl.where(s_mask, b_val, 0.0)
for k in tl.static_range(W):
shift = W - 1 - k
wk = tl.load(w_ptr + d * W + k)
s_shifted = s_off - shift
xk = tl.load(
x_row + s_shifted,
mask=s_mask & (s_shifted >= 0),
other=0.0,
)
acc += wk * xk
tl.store(y_row + s_off, acc, mask=s_mask)
# Per-shape tuned configs: (BLOCK_S, num_warps, num_stages)
_CONFIGS = {
(1, 768, 512, 4): (256, 2, 2),
(1, 768, 2048, 4): (2048, 16, 4),
(1, 1536, 2048, 4): (1024, 8, 2),
(1, 2560, 2048, 4): (512, 1, 2),
(1, 2560, 4096, 4): (4096, 8, 2),
}
_DEFAULT = (256, 4, 2)
def custom_kernel(data: input_t) -> output_t:
x, weight, bias = data
B, D, S = x.shape
W = weight.shape[1]
y = torch.empty_like(x)
bs, nw, ns = _CONFIGS.get((B, D, S, W), _DEFAULT)
grid = (triton.cdiv(S, bs), B * D)
_causal_conv1d[grid](
x, weight, bias, y,
D, S,
W=W,
BLOCK_S=bs,
num_warps=nw,
num_stages=ns,
)
return y
scrolls · 75 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Changes from previous submission
Against this author's previous submission submission 554279.
- #!POPCORN leaderboard causal_conv1d- #!POPCORN gpu B200_Nebius- # Team: Kernal Forge- # Optimizations: per-W tuned configs, lazy compilation- from task import input_t, output_t-- import torch- import helion- import helion.language as hl--- W_CONFIGS: dict[int, helion.Config] = {- 3: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),- 4: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),- 8: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),- }--- def _make_kernel(config: helion.Config):- @helion.kernel(config=config)- def kernel(- x: torch.Tensor,- w: torch.Tensor,- b: torch.Tensor,- ) -> torch.Tensor:- B, D, S = x.shape- W = hl.specialize(w.size(1))- y = torch.empty(B, D, S, dtype=x.dtype, device=x.device)-- for rb, rd, rs in hl.tile([B, D, S], block_size=[1, None, None]):- bi = rb.begin- rs_idx = rs.index- in_bounds = rs_idx < S- bias_tile = hl.zeros([rd, rs], dtype=torch.float32)- bias_tile = bias_tile + b[rd][:, None]-- if W == 4:- ib_base = in_bounds[None, :]- ge1 = (rs_idx >= 1)- ge2 = (rs_idx >= 2)- ge3 = (rs_idx >= 3)- mask0 = ib_base- mask1 = (in_bounds & ge1)[None, :]- mask2 = (in_bounds & ge2)[None, :]- mask3 = (in_bounds & ge3)[None, :]-- x0 = hl.load(x, [bi, rd, rs_idx - 3], extra_mask=mask3)- x1 = hl.load(x, [bi, rd, rs_idx - 2], extra_mask=mask2)- x2 = hl.load(x, [bi, rd, rs_idx - 1], extra_mask=mask1)- x3 = hl.load(x, [bi, rd, rs_idx], extra_mask=mask0)- acc = bias_tile- acc = acc + x0 * w[rd, 0][:, None]- acc = acc + x1 * w[rd, 1][:, None]- acc = acc + x2 * w[rd, 2][:, None]- acc = acc + x3 * w[rd, 3][:, None]- elif W == 3:- ib_base = in_bounds[None, :]- ge1 = (rs_idx >= 1)- ge2 = (rs_idx >= 2)- mask0 = ib_base- mask1 = (in_bounds & ge1)[None, :]- mask2 = (in_bounds & ge2)[None, :]-- x0 = hl.load(x, [bi, rd, rs_idx - 2], extra_mask=mask2)- x1 = hl.load(x, [bi, rd, rs_idx - 1], extra_mask=mask1)- x2 = hl.load(x, [bi, rd, rs_idx], extra_mask=mask0)- acc = bias_tile- acc = acc + x0 * w[rd, 0][:, None]- acc = acc + x1 * w[rd, 1][:, None]- acc = acc + x2 * w[rd, 2][:, None]- else:- acc = bias_tile- for tap in range(W):- shift = W - 1 - tap- x_tap = hl.load(- x,- [bi, rd, rs_idx - shift],- extra_mask=(in_bounds & (rs_idx >= shift))[None, :],- )- acc = acc + x_tap * w[rd, tap][:, None]-- y[rb, rd, rs] = acc[None, :, :]-- return y-- return kernel--- _KERNEL_CACHE: dict[int, callable] = {}--- def _get_kernel(w_val: int):- kernel = _KERNEL_CACHE.get(w_val)- if kernel is None:- kernel = _make_kernel(W_CONFIGS[w_val])- _KERNEL_CACHE[w_val] = kernel- return kernel--- def custom_kernel(data: input_t) -> output_t:- x, weight, bias = data- W = weight.shape[1]- kernel = _get_kernel(W)- return kernel(x, weight, bias)+ #!POPCORN leaderboard causal_conv1d+ #!POPCORN gpu B200_Nebius+ from task import input_t, output_t++ import torch+ import triton+ import triton.language as tl+++ @triton.jit+ def _causal_conv1d(+ x_ptr, w_ptr, b_ptr, y_ptr,+ D, S,+ W: tl.constexpr,+ BLOCK_S: tl.constexpr,+ ):+ pid_s = tl.program_id(0)+ pid_bd = tl.program_id(1) # batch*D index+ d = pid_bd % D++ b_val = tl.load(b_ptr + d)+ x_row = x_ptr + pid_bd * S+ y_row = y_ptr + pid_bd * S++ s_off = pid_s * BLOCK_S + tl.arange(0, BLOCK_S)+ s_mask = s_off < S++ acc = tl.where(s_mask, b_val, 0.0)++ for k in tl.static_range(W):+ shift = W - 1 - k+ wk = tl.load(w_ptr + d * W + k)+ s_shifted = s_off - shift+ xk = tl.load(+ x_row + s_shifted,+ mask=s_mask & (s_shifted >= 0),+ other=0.0,+ )+ acc += wk * xk++ tl.store(y_row + s_off, acc, mask=s_mask)+++ # Per-shape tuned configs: (BLOCK_S, num_warps, num_stages)+ _CONFIGS = {+ (1, 768, 512, 4): (256, 2, 2),+ (1, 768, 2048, 4): (2048, 16, 4),+ (1, 1536, 2048, 4): (1024, 8, 2),+ (1, 2560, 2048, 4): (512, 1, 2),+ (1, 2560, 4096, 4): (4096, 8, 2),+ }++ _DEFAULT = (256, 4, 2)+++ def custom_kernel(data: input_t) -> output_t:+ x, weight, bias = data+ B, D, S = x.shape+ W = weight.shape[1]++ y = torch.empty_like(x)++ bs, nw, ns = _CONFIGS.get((B, D, S, W), _DEFAULT)+ grid = (triton.cdiv(S, bs), B * D)++ _causal_conv1d[grid](+ x, weight, bias, y,+ D, S,+ W=W,+ BLOCK_S=bs,+ num_warps=nw,+ num_stages=ns,+ )+ return y
scrolls · 178 diff lines total
Best evidence level for this revision: reported
JSON