[Perf]Remove H2D for Qwen3.5 SpecV2 (#20864)

This commit is contained in:
Jincong Chen
2026-03-31 11:54:58 +08:00
committed by GitHub
parent 33e725b052
commit 03e4f2858d
2 changed files with 17 additions and 13 deletions
@@ -715,9 +715,15 @@ class ForwardBatch(ForwardBatchDeepSeekMHAMixin):
else: # target_verify or draft_decode
seq_positions = batch.spec_info.positions.view(batch_size, -1)
# Split text-only and mixed batches here because SpecV2 text-only batches can avoid an extra D2H.
if all(mm_input is None for mm_input in mm_inputs):
mrope_delta_tensor = torch.zeros(
(batch_size, 1), dtype=torch.int64, device=device
)
else:
mrope_deltas = [
(
torch.tensor([0], dtype=torch.int64)
torch.zeros(1, dtype=torch.int64)
if mm_inputs[i] is None
else mm_inputs[i].mrope_position_delta.squeeze(0)
)
@@ -234,14 +234,12 @@ class EagleVerifyInputV2Mixin:
# Set mamba_track_indices for mamba prefix-cache state tracking
if get_global_server_args().enable_mamba_extra_buffer():
batch.mamba_track_indices = torch.tensor(
batch.mamba_track_indices = torch.stack(
[
req.mamba_ping_pong_track_buffer[req.mamba_next_track_idx]
for req in batch.reqs
],
dtype=torch.int64,
device=device,
)
]
).to(torch.int64)
batch.mamba_track_mask = None
batch.mamba_track_seqlens = None