Skip to content
KernelIndex
Search⌘K

submission 553151

ramizzik · python · License unknown

Use it

Vendorable · source mirrored · license unknownView source →

No package. Vendor the mirrored source: 109 lines, June 9 Researcher Reciprocity License v1.0.

submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-fp8-quant-553151?include=source"
interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32

Benchmark evidence

1 measurement across 1 GPU, fastest first.

Operation / workload
Hardware
Latency
Rank
Observed
NVIDIA B200
24.0µs
#13 of 17
2026-03-14

Reported · How evidence levels are derived →

Source and license

sourceavailable
revision digestsha256:5844ee65713762cdea02649a3d5d840f19ead30c1023e473e1c2f04e038ee5f5
license declaredunknown
license concludedunknown
authorsramizzik
imported2026-08-15

Kernel source

submission.py109 lines
# Import the type aliases for input (3 tensors) and output (2 tensors)
from task import input_t, output_t

import torch
import helion                 # Helion: Python-embedded DSL for writing GPU kernels on top of Triton
import helion.language as hl  # hl contains Helion primitives like tile(), specialize(), etc.
from pathlib import Path

# ACF files on B200 at /opt/booster_pack/ — try each, use first found
def _find_acf(pattern):
    """Find first matching ACF file on B200."""
    bp = Path("/opt/booster_pack")
    if not bp.exists():
        return None
    for p in sorted(bp.glob(pattern)):
        return str(p)
    return None

_acf = _find_acf("fp8_group_quant_*.acf")
CONFIG_DICT = {
    "block_sizes": [1],
    "num_warps": 1,
    "num_stages": 1,
}
if _acf:
    CONFIG_DICT["advanced_controls_file"] = _acf

@helion.kernel(
    static_shapes=True,
    config=helion.Config(**CONFIG_DICT),
)
def normalize_to_range(
    data: torch.Tensor,       # [N, G] input: each row is one group of `group_size` elements
    scales_out: torch.Tensor,  # [N] output: one scale factor per row/group
) -> torch.Tensor:
    # Total number of rows = total number of groups across all tokens
    nrows = data.size(0)

    # hl.specialize() tells Helion this dimension is known at compile time,
    # so it can generate optimized code with the exact column count baked in
    ncols = hl.specialize(data.size(1))

    # FP8 E4M3 max representable value -- we scale inputs so the largest fits within [-448, 448]
    MAX_VAL = 448.0

    # Allocate the output tensor for quantized values, same shape as input
    qout = torch.empty(nrows, ncols, dtype=torch.float32, device=data.device)

    # hl.tile(nrows) partitions the row indices across GPU thread blocks
    # Each iteration of this loop runs on a different block, processing a tile of rows
    for rr in hl.tile(nrows):
        # Load one tile of rows from global memory and ensure float32 precision
        row = data[rr, :].to(torch.float32)

        # Per-group absmax: find the largest absolute value in each row/group
        amax = torch.amax(torch.abs(row), -1)

        # Clamp to at least 1e-10 to prevent division by zero
        amax = torch.clamp(amax, min=1e-10)

        # Scale: maps the largest value to FP8_MAX (448.0)
        scale = amax / MAX_VAL

        # Quantize: divide by scale, then clamp to FP8 representable range [-448, 448]
        q = torch.clamp(row / scale[:, None], -448.0, 448.0)

        # Store quantized values and per-group scale factors
        qout[rr, :] = q
        scales_out[rr] = scale

    return qout


def custom_kernel(data: input_t) -> output_t:
    """Entry point called by the evaluation harness. Reshapes, calls the GPU kernel, reshapes back."""
    # Unpack the 3 input tensors: raw data, pre-allocated quantized output, pre-allocated scales output
    x, x_q, x_s = data

    # T = num_tokens (rows), H = hidden_dim (columns per row)
    T, H = x.shape

    # G = number of groups per row (e.g., hidden_dim=7168, group_size=128 -> G=56)
    G = x_s.shape[1]

    # gsz = group_size: how many elements in each group (e.g., 128)
    gsz = H // G

    # N = total number of groups across all tokens (flatten tokens and groups into one axis)
    N = T * G

    # Reshape x from [T, H] to [N, gsz]: each row is now exactly one group of `gsz` elements
    # This lets the kernel treat each group as an independent row to quantize
    flat_in = x.reshape(N, gsz)

    # Reshape x_s from [T, G] to [N]: one scale per flattened group-row
    flat_s = x_s.reshape(N)

    # Launch the Helion GPU kernel: quantizes each group-row and computes its scale
    flat_q = normalize_to_range(flat_in, flat_s)

    # Reshape quantized output back from [N, gsz] to [T, H] and copy into pre-allocated buffer
    x_q[...] = flat_q.reshape(T, H)

    # Reshape scales back from [N] to [T, G] and copy into pre-allocated buffer
    x_s[...] = flat_s.reshape(T, G)

    # Return the filled-in output buffers
    return x_q, x_s
scrolls · 109 lines total

Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0

Best evidence level for this revision: reported

JSON