Skip to content
KernelIndex
Search⌘K

submission 555480

Ayush10 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 75 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-causal-conv1d-555480?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Causal depthwise conv1dsuite of 3 cases
NVIDIA B200
12.6µs
#2 of 36
2026-03-15

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:909167aa105dda6c72d3f720784d871bb9ba7589fa9423c9991c533bf8504129
license declaredunknown
license concludedunknown
authorsAyush10
imported2026-08-15

Kernel source

submission.py75 lines
#!POPCORN leaderboard causal_conv1d
#!POPCORN gpu B200_Nebius
from task import input_t, output_t

import torch
import triton
import triton.language as tl


@triton.jit
def _causal_conv1d(
    x_ptr, w_ptr, b_ptr, y_ptr,
    D, S,
    W: tl.constexpr,
    BLOCK_S: tl.constexpr,
):
    pid_s = tl.program_id(0)
    pid_bd = tl.program_id(1)  # batch*D index
    d = pid_bd % D

    b_val = tl.load(b_ptr + d)
    x_row = x_ptr + pid_bd * S
    y_row = y_ptr + pid_bd * S

    s_off = pid_s * BLOCK_S + tl.arange(0, BLOCK_S)
    s_mask = s_off < S

    acc = tl.where(s_mask, b_val, 0.0)

    for k in tl.static_range(W):
        shift = W - 1 - k
        wk = tl.load(w_ptr + d * W + k)
        s_shifted = s_off - shift
        xk = tl.load(
            x_row + s_shifted,
            mask=s_mask & (s_shifted >= 0),
            other=0.0,
        )
        acc += wk * xk

    tl.store(y_row + s_off, acc, mask=s_mask)


# Per-shape tuned configs: (BLOCK_S, num_warps, num_stages)
_CONFIGS = {
    (1, 768, 512, 4):   (256, 2, 2),
    (1, 768, 2048, 4):  (2048, 16, 4),
    (1, 1536, 2048, 4): (1024, 8, 2),
    (1, 2560, 2048, 4): (512, 1, 2),
    (1, 2560, 4096, 4): (4096, 8, 2),
}

_DEFAULT = (256, 4, 2)


def custom_kernel(data: input_t) -> output_t:
    x, weight, bias = data
    B, D, S = x.shape
    W = weight.shape[1]

    y = torch.empty_like(x)

    bs, nw, ns = _CONFIGS.get((B, D, S, W), _DEFAULT)
    grid = (triton.cdiv(S, bs), B * D)

    _causal_conv1d[grid](
        x, weight, bias, y,
        D, S,
        W=W,
        BLOCK_S=bs,
        num_warps=nw,
        num_stages=ns,
    )
    return y
scrolls · 75 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Changes from previous submission

Against this author's previous submission submission 554279.

- #!POPCORN leaderboard causal_conv1d
- #!POPCORN gpu B200_Nebius
- # Team: Kernal Forge
- # Optimizations: per-W tuned configs, lazy compilation
- from task import input_t, output_t
-
- import torch
- import helion
- import helion.language as hl
-
-
- W_CONFIGS: dict[int, helion.Config] = {
- 3: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
- 4: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
- 8: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
- }
-
-
- def _make_kernel(config: helion.Config):
- @helion.kernel(config=config)
- def kernel(
- x: torch.Tensor,
- w: torch.Tensor,
- b: torch.Tensor,
- ) -> torch.Tensor:
- B, D, S = x.shape
- W = hl.specialize(w.size(1))
- y = torch.empty(B, D, S, dtype=x.dtype, device=x.device)
-
- for rb, rd, rs in hl.tile([B, D, S], block_size=[1, None, None]):
- bi = rb.begin
- rs_idx = rs.index
- in_bounds = rs_idx < S
- bias_tile = hl.zeros([rd, rs], dtype=torch.float32)
- bias_tile = bias_tile + b[rd][:, None]
-
- if W == 4:
- ib_base = in_bounds[None, :]
- ge1 = (rs_idx >= 1)
- ge2 = (rs_idx >= 2)
- ge3 = (rs_idx >= 3)
- mask0 = ib_base
- mask1 = (in_bounds & ge1)[None, :]
- mask2 = (in_bounds & ge2)[None, :]
- mask3 = (in_bounds & ge3)[None, :]
-
- x0 = hl.load(x, [bi, rd, rs_idx - 3], extra_mask=mask3)
- x1 = hl.load(x, [bi, rd, rs_idx - 2], extra_mask=mask2)
- x2 = hl.load(x, [bi, rd, rs_idx - 1], extra_mask=mask1)
- x3 = hl.load(x, [bi, rd, rs_idx], extra_mask=mask0)
- acc = bias_tile
- acc = acc + x0 * w[rd, 0][:, None]
- acc = acc + x1 * w[rd, 1][:, None]
- acc = acc + x2 * w[rd, 2][:, None]
- acc = acc + x3 * w[rd, 3][:, None]
- elif W == 3:
- ib_base = in_bounds[None, :]
- ge1 = (rs_idx >= 1)
- ge2 = (rs_idx >= 2)
- mask0 = ib_base
- mask1 = (in_bounds & ge1)[None, :]
- mask2 = (in_bounds & ge2)[None, :]
-
- x0 = hl.load(x, [bi, rd, rs_idx - 2], extra_mask=mask2)
- x1 = hl.load(x, [bi, rd, rs_idx - 1], extra_mask=mask1)
- x2 = hl.load(x, [bi, rd, rs_idx], extra_mask=mask0)
- acc = bias_tile
- acc = acc + x0 * w[rd, 0][:, None]
- acc = acc + x1 * w[rd, 1][:, None]
- acc = acc + x2 * w[rd, 2][:, None]
- else:
- acc = bias_tile
- for tap in range(W):
- shift = W - 1 - tap
- x_tap = hl.load(
- x,
- [bi, rd, rs_idx - shift],
- extra_mask=(in_bounds & (rs_idx >= shift))[None, :],
- )
- acc = acc + x_tap * w[rd, tap][:, None]
-
- y[rb, rd, rs] = acc[None, :, :]
-
- return y
-
- return kernel
-
-
- _KERNEL_CACHE: dict[int, callable] = {}
-
-
- def _get_kernel(w_val: int):
- kernel = _KERNEL_CACHE.get(w_val)
- if kernel is None:
- kernel = _make_kernel(W_CONFIGS[w_val])
- _KERNEL_CACHE[w_val] = kernel
- return kernel
-
-
- def custom_kernel(data: input_t) -> output_t:
- x, weight, bias = data
- W = weight.shape[1]
- kernel = _get_kernel(W)
- return kernel(x, weight, bias)
+ #!POPCORN leaderboard causal_conv1d
+ #!POPCORN gpu B200_Nebius
+ from task import input_t, output_t
+
+ import torch
+ import triton
+ import triton.language as tl
+
+
+ @triton.jit
+ def _causal_conv1d(
+ x_ptr, w_ptr, b_ptr, y_ptr,
+ D, S,
+ W: tl.constexpr,
+ BLOCK_S: tl.constexpr,
+ ):
+ pid_s = tl.program_id(0)
+ pid_bd = tl.program_id(1) # batch*D index
+ d = pid_bd % D
+
+ b_val = tl.load(b_ptr + d)
+ x_row = x_ptr + pid_bd * S
+ y_row = y_ptr + pid_bd * S
+
+ s_off = pid_s * BLOCK_S + tl.arange(0, BLOCK_S)
+ s_mask = s_off < S
+
+ acc = tl.where(s_mask, b_val, 0.0)
+
+ for k in tl.static_range(W):
+ shift = W - 1 - k
+ wk = tl.load(w_ptr + d * W + k)
+ s_shifted = s_off - shift
+ xk = tl.load(
+ x_row + s_shifted,
+ mask=s_mask & (s_shifted >= 0),
+ other=0.0,
+ )
+ acc += wk * xk
+
+ tl.store(y_row + s_off, acc, mask=s_mask)
+
+
+ # Per-shape tuned configs: (BLOCK_S, num_warps, num_stages)
+ _CONFIGS = {
+ (1, 768, 512, 4): (256, 2, 2),
+ (1, 768, 2048, 4): (2048, 16, 4),
+ (1, 1536, 2048, 4): (1024, 8, 2),
+ (1, 2560, 2048, 4): (512, 1, 2),
+ (1, 2560, 4096, 4): (4096, 8, 2),
+ }
+
+ _DEFAULT = (256, 4, 2)
+
+
+ def custom_kernel(data: input_t) -> output_t:
+ x, weight, bias = data
+ B, D, S = x.shape
+ W = weight.shape[1]
+
+ y = torch.empty_like(x)
+
+ bs, nw, ns = _CONFIGS.get((B, D, S, W), _DEFAULT)
+ grid = (triton.cdiv(S, bs), B * D)
+
+ _causal_conv1d[grid](
+ x, weight, bias, y,
+ D, S,
+ W=W,
+ BLOCK_S=bs,
+ num_warps=nw,
+ num_stages=ns,
+ )
+ return y
scrolls · 178 diff lines total

Best evidence level for this revision: reported

JSON