refactor: remove ineffective __launch_bounds__ from prefill kernel
- Remove MIN_BLOCKS template param and __launch_bounds__ attribute - Profiling shows smem (not registers) is the occupancy bottleneck for D>=64, making the hint a no-op - D=64 sees 2-4% speedup, D=128 unchanged (smem-capped at 1 block/SM) - Update comment blocks in kernel header and both dispatch sites - Verified correctness via standalone CUDA test (max_err ~1e-4)
This commit is contained in:
@@ -15,15 +15,11 @@ static void dispatch_prefill(AttentionParams<bf16>& p) {
|
||||
// D<=128. D=256 stays at 16: BC=32 double-buffered would need 64KB smem,
|
||||
// over the 48KB static cap. Both keep 3 blocks/SM (2 for D=256).
|
||||
constexpr int BC = (HEAD_DIM <= 128) ? 32 : 16;
|
||||
// Register-hint MIN_BLOCKS tuned per HEAD_DIM's (BC=32) smem+register
|
||||
// footprint: the largest blocks/SM that avoids register spills.
|
||||
constexpr int MIN_BLOCKS = (HEAD_DIM <= 32) ? 6 : (HEAD_DIM <= 64) ? 4
|
||||
: (HEAD_DIM <= 128) ? 3 : 2;
|
||||
dim3 grid((p.q_len + BR * WARPS - 1) / (BR * WARPS), p.q_head, p.batch);
|
||||
dim3 block(WARPS * 32, 1, 1);
|
||||
// Static shared memory — no dynamic smem or cudaFuncSetAttribute needed.
|
||||
// sK[BC*LD] + sV[BC*LD] + sQ[BR*LD], all sized by template params.
|
||||
attn_prefill_split_q_mma_kernel<HEAD_DIM, WARPS, BC, MIN_BLOCKS><<<grid, block>>>(p);
|
||||
attn_prefill_split_q_mma_kernel<HEAD_DIM, WARPS, BC><<<grid, block>>>(p);
|
||||
#else
|
||||
constexpr int G = 8, ROWS = 32, P_BC = 32;
|
||||
dim3 grid((p.q_len + ROWS - 1) / ROWS, p.q_head, p.batch);
|
||||
|
||||
Reference in New Issue
Block a user