[Tokenizer] Fix abort racing server crash when large amount of aborts (#28341)
This commit is contained in:
@@ -2654,7 +2654,19 @@ class TokenizerManager(TokenizerControlMixin, TokenizerManagerScoreMixin):
|
|||||||
def _handle_abort_req(self, recv_obj: AbortReq):
|
def _handle_abort_req(self, recv_obj: AbortReq):
|
||||||
if is_health_check_generate_req(recv_obj):
|
if is_health_check_generate_req(recv_obj):
|
||||||
return
|
return
|
||||||
state = self.rid_to_state[recv_obj.rid]
|
# Two scheduler messages can race in handle_loop for the same rid: a
|
||||||
|
# batch output that finishes it normally (deletes rid_to_state[rid])
|
||||||
|
# and this abort echo. If the finish wins, the rid is already gone and
|
||||||
|
# there is nothing left to abort. Common under mass client
|
||||||
|
# disconnects, amplified by prefix / abort_all fan-out.
|
||||||
|
state = self.rid_to_state.get(recv_obj.rid)
|
||||||
|
if state is None:
|
||||||
|
logger.info(
|
||||||
|
"Abort request for rid=%s not found in rid_to_state; "
|
||||||
|
"likely already finished/removed.",
|
||||||
|
recv_obj.rid,
|
||||||
|
)
|
||||||
|
return
|
||||||
state.finished = True
|
state.finished = True
|
||||||
state.time_stats.set_finished_time()
|
state.time_stats.set_finished_time()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user