refactor: clean up inference design patterns

1. KVCache base: add default task_cached/task_record_hashes, remove getattr from scheduler
2. Remove page_size param from scheduler constructor (ContiguousCache-only)
3. InferenceEngine expose cache param for KVCache injection
4. Rename page_cache -> kv_cache in Executor
5. Move stream_callback from Task to TaskManager._callbacks dict
6. TaskManager.clear_queues clears callbacks
This commit is contained in:
2026-07-05 11:41:54 +08:00
parent 5416c2e8fb
commit 849e1e00a3
5 changed files with 39 additions and 50 deletions
+4 -8
View File
@@ -19,13 +19,13 @@ class Executor:
self,
model: AutoModel,
tokenizer: AutoTokenizer,
page_cache: KVCache,
kv_cache: KVCache,
device: Optional[str] = None,
dtype: Optional[torch.dtype] = None,
):
self.model = model
self.tokenizer = tokenizer
self.page_cache = page_cache
self.kv_cache = kv_cache
self.device = device or next(model.parameters()).device
self.dtype = dtype or next(model.parameters()).dtype
@@ -52,9 +52,7 @@ class Executor:
)
.unsqueeze(0)
.expand(batch_sz, -1),
paged_cache=self.page_cache.bind_tasks(
task_ids, prompt_len, self.device
),
paged_cache=self.kv_cache.bind_tasks(task_ids, prompt_len, self.device),
)
def execute_decode(self, tasks: List[Task]) -> List[int]:
@@ -81,9 +79,7 @@ class Executor:
with torch.inference_mode():
outputs = self.model(
input_ids.unsqueeze(1),
paged_cache=self.page_cache.bind_tasks(
task_ids, total_len, self.device
),
paged_cache=self.kv_cache.bind_tasks(task_ids, total_len, self.device),
position_ids=position_ids.unsqueeze(1),
)
logits = outputs["logits"][:, -1, :]