Files
AstrAI/astrai/extension/backend/linear.py
T
0z5a 1c3515714f perf: vectorize bf16 gemv and extend M support to 1-8
- Replace per-element loads with 128-bit uint4 vectorized loads (8 halves per access), improving every measured shape: q/k/v at M=2 from 6.0us to 5.4us, q_proj speedup 2.28-2.45x, mlp_down at M=4 2.76x, lm_head at M=1 +6-8%
- Extend kernel M support from {1,2,4,8} to all M in 1-8 via new BLOCK_M cases 3,5,6,7, since cuBLAS wmma templates pad small M to 8/16 rows and waste compute
- Keep the auto-dispatch allowlist unchanged: a 64-step greedy-walk probe on the real decode path showed mlp_down (K=6912) divergence at step 1 and argmax flips for every candidate odd-M band, the same noise class already present in the merged M=2/4 entries, so no entry has the stability evidence the gate requires
- Rejected alternatives with measurements: split-K accumulation (k/v shapes regress 6.0us to 9.2us, code removed) and MMA tiles (small M is DRAM-bound at ~1 FLOP/byte vs the ~138 needed)
- Update test_gemv M-rejection case to M=9 and test_linear_dispatch multirow fallback to M=9 for the widened range

Benchmark: 8x L20 (sm_89, CUDA 12.8), single-GPU microbench, 200 iters after 20 warmup, weights L2-resident; q(1536x1536) M=3 8.9->5.3us, kv(256x1536) M=3 8.7->3.0us, down(1536x6912) M=3 53.5->10.3us; full gate 691 passed, test_bf16_gemv_uses_current_stream passes in isolation after GPU contention rerun
2026-09-02 14:19:02 +08:00

265 lines
8.1 KiB
Python

"""Inference-only dispatch for AstrAI linear layers.
The CUDA GEMV path is deliberately narrow: automatic selection is enabled
only for single-row BF16 shapes measured to beat ``F.linear`` on a supported
architecture. Every training, prefill, unsupported-layout, and unmeasured
call falls back to PyTorch.
"""
import logging
import os
from functools import lru_cache
from typing import Optional
import torch
import torch.nn.functional as F
from torch import Tensor
from astrai.extension.dispatch import (
ImplRecord,
Spec,
axis,
get_override,
register_family,
resolve,
tensor_axes,
)
from astrai.extension.loader import is_available
from astrai.extension.ops.gemv import bf16_gemv
logger = logging.getLogger(__name__)
# Shape keys are (N, K) for Y[M, N] = X[M, K] @ W[N, K].T. A band is
# automatic only after both the per-shape >=5% and end-to-end decode >=3%
# gates pass and checkpoint greedy output remains stable. M=1 and M=8 remain
# empty on SM89; the safe M=2/4 bands improve real-engine throughput by
# 11.8-14.0%.
_AUTO_GEMV_SHAPES: dict[tuple[int, int], dict[int, frozenset[tuple[int, int]]]] = {
(8, 9): {
2: frozenset(
{
(256, 1536),
(1536, 1536),
(100000, 1536),
}
),
4: frozenset({(256, 1536), (1536, 1536)}),
}
}
_AUTO_GEMV_M = frozenset(
m for architecture in _AUTO_GEMV_SHAPES.values() for m in architecture
)
_VALID_MODES = {"0", "1", "auto"}
_WARNED_MODES: set[str] = set()
def _gemv_mode() -> str:
mode = os.environ.get("ASTRAI_GEMV", "auto").strip().lower()
if mode in _VALID_MODES:
return mode
if mode not in _WARNED_MODES:
_WARNED_MODES.add(mode)
logger.warning(
"ASTRAI_GEMV=%r is invalid; expected 0, 1, or auto; using auto",
mode,
)
return "auto"
def _axes(
x: Tensor, weight: Tensor, bias: Optional[Tensor] = None
) -> dict[str, object]:
x_shape = tuple(x.shape)
weight_shape = tuple(weight.shape)
m = 1 if x.ndim == 1 else (x.shape[0] if x.ndim == 2 else None)
supported_m = m is not None and 1 <= m <= 8
shape_matches = (
weight.ndim == 2
and x.ndim in (1, 2)
and bool(x_shape)
and x_shape[-1] == weight_shape[-1]
)
same_device = x.device == weight.device and (
bias is None or bias.device == x.device
)
bias_supported = bias is None or (
bias.ndim == 1
and weight.ndim == 2
and bias.shape[0] == weight.shape[0]
and bias.dtype == torch.bfloat16
and bias.is_contiguous()
)
capability = torch.cuda.get_device_capability(x.device) if x.is_cuda else None
n = weight_shape[0] if weight.ndim == 2 else None
k = weight_shape[1] if weight.ndim == 2 else None
return tensor_axes(
x,
mode=_gemv_mode(),
capability=capability,
n=n,
k=k,
m=m,
supported_m=supported_m,
shape_matches=shape_matches,
same_device=same_device,
weight_dtype=weight.dtype,
x_contiguous=x.is_contiguous(),
weight_contiguous=weight.is_contiguous(),
bias_supported=bias_supported,
k_even=k is not None and k % 2 == 0,
)
_SPEC_CAPABLE = (
axis("device_cuda").truthy()
& axis("dtype").in_(torch.bfloat16)
& axis("weight_dtype").in_(torch.bfloat16)
& axis("grad_enabled").eq(False)
& axis("supported_m").truthy()
& axis("shape_matches").truthy()
& axis("same_device").truthy()
& axis("x_contiguous").truthy()
& axis("weight_contiguous").truthy()
& axis("bias_supported").truthy()
& axis("k_even").truthy()
)
_SPEC_AUTO = _SPEC_CAPABLE & Spec.of(
lambda ax: (
(ax.get("n"), ax.get("k"))
in _AUTO_GEMV_SHAPES.get(ax.get("capability"), {}).get(ax.get("m"), ())
),
"shape is a measured winner for this architecture",
)
def _torch_linear(x: Tensor, weight: Tensor, bias: Optional[Tensor] = None) -> Tensor:
return F.linear(x, weight, bias)
def _inference_bf16_gemv(
x: Tensor, weight: Tensor, bias: Optional[Tensor] = None
) -> Tensor:
# Model parameters retain requires_grad=True after eval(). Dispatch is
# already restricted to no-grad, so detached views preserve storage and
# layout while satisfying the primitive's explicit autograd guard.
return bf16_gemv(
x.detach(),
weight.detach(),
bias.detach() if bias is not None else None,
)
@lru_cache(maxsize=None)
def _device_capability(device_index: int) -> tuple[int, int]:
return torch.cuda.get_device_capability(device_index)
def _gemv_capable(x: Tensor, weight: Tensor, bias: Optional[Tensor]) -> bool:
if (
torch.is_grad_enabled()
or not x.is_cuda
or x.dtype != torch.bfloat16
or weight.dtype != torch.bfloat16
or weight.ndim != 2
or x.ndim not in (1, 2)
or (x.ndim == 2 and not 1 <= x.shape[0] <= 8)
or x.shape[-1] != weight.shape[1]
or weight.shape[1] % 2 != 0
or x.device != weight.device
or not x.is_contiguous()
or not weight.is_contiguous()
or not is_available("bf16_gemv")
):
return False
return bias is None or (
bias.device == x.device
and bias.dtype == torch.bfloat16
and bias.ndim == 1
and bias.shape[0] == weight.shape[0]
and bias.is_contiguous()
)
def _auto_gemv_shape(x: Tensor, weight: Tensor) -> bool:
capability = _device_capability(x.get_device())
m = 1 if x.ndim == 1 else x.shape[0]
return (weight.shape[0], weight.shape[1]) in _AUTO_GEMV_SHAPES.get(
capability, {}
).get(m, ())
def _linear_records() -> list[ImplRecord]:
mode = _gemv_mode()
gemv_priority = 0 if mode == "1" else 100
auto_priority = 0 if mode == "auto" else 90
torch_priority = 0 if mode == "0" else 50
return [
ImplRecord(
family="linear",
name="gemv",
obj=_inference_bf16_gemv,
spec=_SPEC_CAPABLE,
available=lambda: is_available("bf16_gemv"),
priority=gemv_priority,
),
ImplRecord(
family="linear",
name="auto_gemv",
obj=_inference_bf16_gemv,
spec=_SPEC_AUTO,
available=lambda: is_available("bf16_gemv"),
priority=auto_priority,
),
ImplRecord(
family="linear",
name="torch",
obj=_torch_linear,
spec=Spec.always(),
priority=torch_priority,
),
]
def _fallback_record() -> ImplRecord:
return ImplRecord(
family="linear",
name="torch",
obj=_torch_linear,
spec=Spec.always(),
priority=999,
)
register_family("linear", _axes, _linear_records, _fallback_record)
def linear(x: Tensor, weight: Tensor, bias: Optional[Tensor] = None) -> Tensor:
"""Apply a linear projection with safe inference-only GEMV dispatch.
``ASTRAI_GEMV=0`` always uses PyTorch, ``1`` forces GEMV whenever the
primitive can safely handle any M in ``{1, ..., 8}``, and ``auto`` (the
default) uses only architecture/shape bands backed by benchmark evidence.
"""
# Preserve the shared dispatcher for explicit/context selection and
# ASTR_OPS diagnostics, while keeping the default per-layer hot path free
# of axes dictionaries, record sorting, and repeated capability queries.
if get_override("linear") is not None or "linear" in os.environ.get("ASTR_OPS", ""):
return resolve("linear", x, weight, bias).record.obj(x, weight, bias)
mode = _gemv_mode()
if mode == "0" or (mode == "auto" and not _AUTO_GEMV_SHAPES):
return _torch_linear(x, weight, bias)
if mode == "auto":
m = 1 if x.ndim == 1 else (x.shape[0] if x.ndim == 2 else None)
if m not in _AUTO_GEMV_M:
return _torch_linear(x, weight, bias)
if mode != "0" and _gemv_capable(x, weight, bias):
if mode == "1" or _auto_gemv_shape(x, weight):
return _inference_bf16_gemv(x, weight, bias)
return _torch_linear(x, weight, bias)
__all__ = ["linear"]