[Fix] DO NOT skip save_kv_cache for dllm (#19020)
This commit is contained in:
@@ -812,7 +812,7 @@ class FlashInferAttnBackend(AttentionBackend):
|
|||||||
or layer.attn_type == AttentionType.ENCODER_ONLY
|
or layer.attn_type == AttentionType.ENCODER_ONLY
|
||||||
):
|
):
|
||||||
causal = False
|
causal = False
|
||||||
if save_kv_cache and layer.attn_type == AttentionType.ENCODER_ONLY:
|
if not self.is_dllm_model and layer.attn_type == AttentionType.ENCODER_ONLY:
|
||||||
save_kv_cache = False
|
save_kv_cache = False
|
||||||
|
|
||||||
if self.forward_metadata.extend_no_prefix:
|
if self.forward_metadata.extend_no_prefix:
|
||||||
|
|||||||
Reference in New Issue
Block a user