refactor: extract QTileMapper for prefill tile dispatch

- wrap one-thread map + shared broadcast + early exit
- both scalar and MMA prefill kernels use the shared helper
This commit is contained in:
2026-08-09 23:18:03 +08:00
parent c5fba9c238
commit 9b58fef222
3 changed files with 30 additions and 22 deletions
+4 -11
View File
@@ -24,18 +24,11 @@ __global__ void attn_prefill_split_q_mma_kernel(AttentionParams<bf16> p) {
const int tid4 = lane & 3; // 0..3
const int q_head = blockIdx.y;
__shared__ int mapped_batch;
__shared__ int mapped_q_tile;
if (threadIdx.x == 0) {
mapped_batch = -1;
KV::template map_q_tile<Traits::BR * Traits::WARPS>(
p, blockIdx.x, blockIdx.z, mapped_batch, mapped_q_tile);
}
__syncthreads();
if (mapped_batch < 0)
__shared__ QTileMapper<Traits::BR * Traits::WARPS, KV> qmap;
if (!qmap.init(p, blockIdx.x, blockIdx.z))
return;
const int batch = mapped_batch;
const int q_tile = mapped_q_tile;
const int batch = qmap.batch;
const int q_tile = qmap.q_tile;
const int kv_head = q_head / (p.q_head / p.kv_head);
const int qrow0 = (q_tile * Traits::WARPS + warp) * Traits::BR;