perf: accelerate decode linear with bf16 gemv

- add decode-shape benchmark harness
- add bf16 GEMV CUDA primitive with head-dim generic kernel
- dispatch decode-time linear layers to gemv for M=1
- extend gemv coverage to small decode batches
This commit is contained in:
0z5a
2026-09-02 13:11:25 +08:00
parent 9c3ef0c2a1
commit a144d7f306
14 changed files with 1242 additions and 3 deletions
+4
View File
@@ -26,6 +26,7 @@ from astrai.extension.backend import (
attention,
attn_backend,
get_backend,
linear,
)
from astrai.extension.dispatch import (
Axes,
@@ -49,6 +50,7 @@ from astrai.extension.ops import (
attn_decode,
attn_paged_decode,
attn_prefill,
bf16_gemv,
)
__all__ = [
@@ -62,9 +64,11 @@ __all__ = [
"attention",
"attn_backend",
"get_backend",
"linear",
"attn_decode",
"attn_paged_decode",
"attn_prefill",
"bf16_gemv",
"is_available",
"KERNEL_NAMES",
"apply_rotary_emb",