Fix DeepSeek V4 multistream QKV buffer lifetime (#36547)
This commit is contained in:
@@ -1059,8 +1059,6 @@ class MQALayer(MqaAttentionBase):
|
|||||||
x_linear, positions, forward_batch, attn_backend, qkv_a=qkv_a
|
x_linear, positions, forward_batch, attn_backend, qkv_a=qkv_a
|
||||||
)
|
)
|
||||||
|
|
||||||
del qkv_a
|
|
||||||
|
|
||||||
if self.compressor is not None:
|
if self.compressor is not None:
|
||||||
with torch.cuda.stream(stream_compressor):
|
with torch.cuda.stream(stream_compressor):
|
||||||
attn_backend.forward_core_compressor(
|
attn_backend.forward_core_compressor(
|
||||||
@@ -1071,6 +1069,7 @@ class MQALayer(MqaAttentionBase):
|
|||||||
current_stream.wait_stream(stream_kv)
|
current_stream.wait_stream(stream_kv)
|
||||||
current_stream.wait_stream(stream_compressor)
|
current_stream.wait_stream(stream_compressor)
|
||||||
current_stream.wait_stream(stream_indexer)
|
current_stream.wait_stream(stream_indexer)
|
||||||
|
del qkv_a
|
||||||
|
|
||||||
return q
|
return q
|
||||||
|
|
||||||
@@ -1153,8 +1152,6 @@ class MQALayer(MqaAttentionBase):
|
|||||||
q_out.copy_(q)
|
q_out.copy_(q)
|
||||||
q.record_stream(stream_q)
|
q.record_stream(stream_q)
|
||||||
|
|
||||||
del qkv_a
|
|
||||||
|
|
||||||
# Indexer + compressor: serial on current.
|
# Indexer + compressor: serial on current.
|
||||||
if self.indexer is not None:
|
if self.indexer is not None:
|
||||||
self.indexer(
|
self.indexer(
|
||||||
@@ -1174,6 +1171,7 @@ class MQALayer(MqaAttentionBase):
|
|||||||
# Join stream_kv + stream_q before downstream attention.
|
# Join stream_kv + stream_q before downstream attention.
|
||||||
current_stream.wait_stream(stream_kv)
|
current_stream.wait_stream(stream_kv)
|
||||||
current_stream.wait_stream(stream_q)
|
current_stream.wait_stream(stream_q)
|
||||||
|
del qkv_a
|
||||||
return q
|
return q
|
||||||
|
|
||||||
def _forward_prepare_multi_stream_hip(
|
def _forward_prepare_multi_stream_hip(
|
||||||
|
|||||||
Reference in New Issue
Block a user