submission 553151
ramizzik · python · License unknown
Use it
Vendorable · source mirrored · license unknownView source →
No package. Vendor the mirrored source: 109 lines, June 9 Researcher Reciprocity License v1.0.
submission.py
curl "https://kernelindex.com/api/v1/implementations/kernelbot-fp8-quant-553151?include=source"interfacepython
Compatibility
measured onNVIDIA B200
declared hardwareNVIDIA B200
architecturessm_100
dtypesfp32
Benchmark evidence
1 measurement across 1 GPU, fastest first.
Operation / workload
Hardware
Latency
Rank
Observed
Reported · How evidence levels are derived →
Source and license
sourceavailable
revision digestsha256:5844ee65713762cdea02649a3d5d840f19ead30c1023e473e1c2f04e038ee5f5
license declaredunknown
license concludedunknown
authorsramizzik
imported2026-08-15
Kernel source
submission.py109 lines
# Import the type aliases for input (3 tensors) and output (2 tensors)
from task import input_t, output_t
import torch
import helion # Helion: Python-embedded DSL for writing GPU kernels on top of Triton
import helion.language as hl # hl contains Helion primitives like tile(), specialize(), etc.
from pathlib import Path
# ACF files on B200 at /opt/booster_pack/ — try each, use first found
def _find_acf(pattern):
"""Find first matching ACF file on B200."""
bp = Path("/opt/booster_pack")
if not bp.exists():
return None
for p in sorted(bp.glob(pattern)):
return str(p)
return None
_acf = _find_acf("fp8_group_quant_*.acf")
CONFIG_DICT = {
"block_sizes": [1],
"num_warps": 1,
"num_stages": 1,
}
if _acf:
CONFIG_DICT["advanced_controls_file"] = _acf
@helion.kernel(
static_shapes=True,
config=helion.Config(**CONFIG_DICT),
)
def normalize_to_range(
data: torch.Tensor, # [N, G] input: each row is one group of `group_size` elements
scales_out: torch.Tensor, # [N] output: one scale factor per row/group
) -> torch.Tensor:
# Total number of rows = total number of groups across all tokens
nrows = data.size(0)
# hl.specialize() tells Helion this dimension is known at compile time,
# so it can generate optimized code with the exact column count baked in
ncols = hl.specialize(data.size(1))
# FP8 E4M3 max representable value -- we scale inputs so the largest fits within [-448, 448]
MAX_VAL = 448.0
# Allocate the output tensor for quantized values, same shape as input
qout = torch.empty(nrows, ncols, dtype=torch.float32, device=data.device)
# hl.tile(nrows) partitions the row indices across GPU thread blocks
# Each iteration of this loop runs on a different block, processing a tile of rows
for rr in hl.tile(nrows):
# Load one tile of rows from global memory and ensure float32 precision
row = data[rr, :].to(torch.float32)
# Per-group absmax: find the largest absolute value in each row/group
amax = torch.amax(torch.abs(row), -1)
# Clamp to at least 1e-10 to prevent division by zero
amax = torch.clamp(amax, min=1e-10)
# Scale: maps the largest value to FP8_MAX (448.0)
scale = amax / MAX_VAL
# Quantize: divide by scale, then clamp to FP8 representable range [-448, 448]
q = torch.clamp(row / scale[:, None], -448.0, 448.0)
# Store quantized values and per-group scale factors
qout[rr, :] = q
scales_out[rr] = scale
return qout
def custom_kernel(data: input_t) -> output_t:
"""Entry point called by the evaluation harness. Reshapes, calls the GPU kernel, reshapes back."""
# Unpack the 3 input tensors: raw data, pre-allocated quantized output, pre-allocated scales output
x, x_q, x_s = data
# T = num_tokens (rows), H = hidden_dim (columns per row)
T, H = x.shape
# G = number of groups per row (e.g., hidden_dim=7168, group_size=128 -> G=56)
G = x_s.shape[1]
# gsz = group_size: how many elements in each group (e.g., 128)
gsz = H // G
# N = total number of groups across all tokens (flatten tokens and groups into one axis)
N = T * G
# Reshape x from [T, H] to [N, gsz]: each row is now exactly one group of `gsz` elements
# This lets the kernel treat each group as an independent row to quantize
flat_in = x.reshape(N, gsz)
# Reshape x_s from [T, G] to [N]: one scale per flattened group-row
flat_s = x_s.reshape(N)
# Launch the Helion GPU kernel: quantizes each group-row and computes its scale
flat_q = normalize_to_range(flat_in, flat_s)
# Reshape quantized output back from [N, gsz] to [T, H] and copy into pre-allocated buffer
x_q[...] = flat_q.reshape(T, H)
# Reshape scales back from [N] to [T, G] and copy into pre-allocated buffer
x_s[...] = flat_s.reshape(T, G)
# Return the filled-in output buffers
return x_q, x_s
scrolls · 109 lines total
Source code from GPU Mode and the KernelBot dataset · June 9 Researcher Reciprocity License v1.0
Best evidence level for this revision: reported
JSON