[spec decoding] fix inkling multi layer mtp draft extend cuda graph (#32254)
This commit is contained in:
@@ -676,8 +676,12 @@ class MultiLayerEagleDraftWorker(EagleDraftWorkerBase):
|
|||||||
# Batch 2: Draft extend
|
# Batch 2: Draft extend
|
||||||
draft_extend_input = EagleDraftExtendInput(
|
draft_extend_input = EagleDraftExtendInput(
|
||||||
hidden_states=batch_result.logits_output.hidden_states,
|
hidden_states=batch_result.logits_output.hidden_states,
|
||||||
# Actual width: the multi-layer chain fills num_steps + 1 rows/req.
|
# Actual width: the multi-layer chain fills num_steps + 1 rows/req,
|
||||||
num_tokens_per_req=self.speculative_num_steps + 1,
|
# plus the boundary-KV front rows when the widened window is active
|
||||||
|
# (must match the capture width in the draft-extend graph runner).
|
||||||
|
num_tokens_per_req=self.speculative_num_steps
|
||||||
|
+ 1
|
||||||
|
+ self.draft_extend_num_front_tokens,
|
||||||
num_tokens_for_logprob_per_req=1,
|
num_tokens_for_logprob_per_req=1,
|
||||||
num_front_tokens=self.draft_extend_num_front_tokens,
|
num_front_tokens=self.draft_extend_num_front_tokens,
|
||||||
)
|
)
|
||||||
|
|||||||
Reference in New Issue
Block a user