Skip to content
KernelIndex
Search⌘K

submission 554279

Ayush10 · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 105 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-causal-conv1d-554279?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
Causal depthwise conv1dsuite of 3 cases
NVIDIA B200
13.9µs
#4 of 36
2026-03-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:e05d8c8993d616743bf9d4d1e6ea07c14cd5efdf1955e8be46861192e400fd9a
license declaredunknown
license concludedunknown
authorsAyush10
imported2026-08-15

Techniques

Extracted from the mirrored source by pattern, never inferred. Each row cites its line.

num-warps = 43: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
stages = 23: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),

Kernel source

submission.py105 lines
#!POPCORN leaderboard causal_conv1d
#!POPCORN gpu B200_Nebius
# Team: Kernal Forge
# Optimizations: per-W tuned configs, lazy compilation
from task import input_t, output_t

import torch
import helion
import helion.language as hl


W_CONFIGS: dict[int, helion.Config] = {
    3: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
    4: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
    8: helion.Config(block_sizes=[32, 64], num_warps=4, num_stages=2),
}


def _make_kernel(config: helion.Config):
    @helion.kernel(config=config)
    def kernel(
        x: torch.Tensor,
        w: torch.Tensor,
        b: torch.Tensor,
    ) -> torch.Tensor:
        B, D, S = x.shape
        W = hl.specialize(w.size(1))
        y = torch.empty(B, D, S, dtype=x.dtype, device=x.device)

        for rb, rd, rs in hl.tile([B, D, S], block_size=[1, None, None]):
            bi = rb.begin
            rs_idx = rs.index
            in_bounds = rs_idx < S
            bias_tile = hl.zeros([rd, rs], dtype=torch.float32)
            bias_tile = bias_tile + b[rd][:, None]

            if W == 4:
                ib_base = in_bounds[None, :]
                ge1 = (rs_idx >= 1)
                ge2 = (rs_idx >= 2)
                ge3 = (rs_idx >= 3)
                mask0 = ib_base
                mask1 = (in_bounds & ge1)[None, :]
                mask2 = (in_bounds & ge2)[None, :]
                mask3 = (in_bounds & ge3)[None, :]

                x0 = hl.load(x, [bi, rd, rs_idx - 3], extra_mask=mask3)
                x1 = hl.load(x, [bi, rd, rs_idx - 2], extra_mask=mask2)
                x2 = hl.load(x, [bi, rd, rs_idx - 1], extra_mask=mask1)
                x3 = hl.load(x, [bi, rd, rs_idx], extra_mask=mask0)
                acc = bias_tile
                acc = acc + x0 * w[rd, 0][:, None]
                acc = acc + x1 * w[rd, 1][:, None]
                acc = acc + x2 * w[rd, 2][:, None]
                acc = acc + x3 * w[rd, 3][:, None]
            elif W == 3:
                ib_base = in_bounds[None, :]
                ge1 = (rs_idx >= 1)
                ge2 = (rs_idx >= 2)
                mask0 = ib_base
                mask1 = (in_bounds & ge1)[None, :]
                mask2 = (in_bounds & ge2)[None, :]

                x0 = hl.load(x, [bi, rd, rs_idx - 2], extra_mask=mask2)
                x1 = hl.load(x, [bi, rd, rs_idx - 1], extra_mask=mask1)
                x2 = hl.load(x, [bi, rd, rs_idx], extra_mask=mask0)
                acc = bias_tile
                acc = acc + x0 * w[rd, 0][:, None]
                acc = acc + x1 * w[rd, 1][:, None]
                acc = acc + x2 * w[rd, 2][:, None]
            else:
                acc = bias_tile
                for tap in range(W):
                    shift = W - 1 - tap
                    x_tap = hl.load(
                        x,
                        [bi, rd, rs_idx - shift],
                        extra_mask=(in_bounds & (rs_idx >= shift))[None, :],
                    )
                    acc = acc + x_tap * w[rd, tap][:, None]

            y[rb, rd, rs] = acc[None, :, :]

        return y

    return kernel


_KERNEL_CACHE: dict[int, callable] = {}


def _get_kernel(w_val: int):
    kernel = _KERNEL_CACHE.get(w_val)
    if kernel is None:
        kernel = _make_kernel(W_CONFIGS[w_val])
        _KERNEL_CACHE[w_val] = kernel
    return kernel


def custom_kernel(data: input_t) -> output_t:
    x, weight, bias = data
    W = weight.shape[1]
    kernel = _get_kernel(W)
    return kernel(x, weight, bias)
scrolls · 105 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON