[Refactor] Share chat encoding dispatch between serving and offline tools (#30623)

This commit is contained in:
Liangsheng Yin
2026-07-09 02:45:42 -07:00
committed by GitHub
parent c02b032da0
commit 078f06fbf4
3 changed files with 109 additions and 41 deletions
+21 -25
View File
@@ -483,25 +483,21 @@ def _flush_cache_with_retry(url: str, endpoint: str, max_retries: int = 3):
time.sleep(2)
# FIXME: this mirrors the chat-encoding dispatch in
# serving_chat._resolve_chat_encoding_spec (DeepSeek-V4 custom encoding vs HF
# chat template) so the benchmark reproduces the serving token stream. Unify
# the dispatch into a shared resolver instead of duplicating it client-side.
@lru_cache(maxsize=None)
def _is_deepseek_v4_model(name_or_path: str) -> bool:
def _load_hf_config(name_or_path: str):
if not name_or_path:
return None
from transformers import AutoConfig
from sglang.srt.configs.model_config import is_deepseek_v4
try:
hf_config = AutoConfig.from_pretrained(name_or_path, trust_remote_code=True)
return AutoConfig.from_pretrained(name_or_path, trust_remote_code=True)
except Exception as e:
print(
f"Warning: could not load config for {name_or_path!r} ({e}); "
"assuming a non-DeepSeek-V4 model for --apply-chat-template."
"falling back to the HF chat template for --apply-chat-template."
)
return False
return is_deepseek_v4(hf_config)
return None
def _encode_fixed_prompt(
@@ -510,21 +506,21 @@ def _encode_fixed_prompt(
if not apply_chat_template:
return tok_inner.encode(prompt_text)
messages = [{"role": "user", "content": prompt_text}]
# DeepSeek-V4 chat encoding does not go through the HF chat template; use
# its own encoder so the token stream matches /v1/chat/completions.
if _is_deepseek_v4_model(getattr(tok_inner, "name_or_path", "") or ""):
from sglang.srt.entrypoints.openai import encoding_dsv4
from sglang.srt.entrypoints.openai.chat_encoding import (
encode_simple_chat,
resolve_chat_encoding_spec,
)
real_input = encoding_dsv4.encode_messages(messages, thinking_mode="chat")
return tok_inner.encode(real_input)
if getattr(tok_inner, "chat_template", None) is None:
raise ValueError(
"--apply-chat-template requires a tokenizer with a chat template, "
f"but {getattr(tok_inner, 'name_or_path', tok_inner)!r} has none."
)
return tok_inner.apply_chat_template(
messages, add_generation_prompt=True, tokenize=True
hf_config = _load_hf_config(getattr(tok_inner, "name_or_path", "") or "")
spec = (
resolve_chat_encoding_spec(hf_config=hf_config, tokenizer=tok_inner)
if hf_config is not None
else None
)
return encode_simple_chat(
tokenizer=tok_inner,
spec=spec,
messages=[{"role": "user", "content": prompt_text}],
)
@@ -0,0 +1,81 @@
"""Single home for the chat-encoding dispatch.
Which encoder turns chat messages into prompt tokens is a property of the
model, so the serving path and offline tools (benchmarks, evals) must resolve
it here instead of re-deriving it from model architectures themselves.
"""
from __future__ import annotations
from typing import Any, Dict, List, Optional
def resolve_chat_encoding_spec(
*,
hf_config: Any,
tokenizer: Any,
tool_call_parser: Optional[str] = None,
) -> Optional[str]:
"""Return the chat encoding spec for a model: "dsv4", "dsv32", or None.
None means the default path (HF chat template).
"""
if tool_call_parser == "deepseekv4":
return "dsv4"
if tool_call_parser == "deepseekv32":
return "dsv32"
architectures = hf_config.architectures
arch = architectures[0] if architectures else ""
if "DeepseekV4" in arch:
return "dsv4"
has_chat_template = tokenizer is not None and tokenizer.chat_template is not None
if "DeepseekV3" in arch and not has_chat_template:
return "dsv32"
return None
def encode_simple_chat(
*,
tokenizer: Any,
spec: Optional[str],
messages: List[Dict[str, Any]],
thinking_mode: str = "chat",
) -> List[int]:
"""Encode a plain-text chat conversation into prompt token ids.
Minimal encode for offline tools: no tools, no multimodal content, no
continue_final_message; the serving path keeps its full request-level
pipeline in ``serving_chat``. Like
``serving_chat``, an empty system message is prepended when the
conversation does not start with one (for the dsv4/dsv32 encoders this
currently renders to zero tokens, but keeping the insertion explicit ties
this helper to the serving semantics rather than to that coincidence).
"""
if spec in ("dsv4", "dsv32"):
if messages and messages[0]["role"] != "system":
messages = [{"role": "system", "content": ""}] + list(messages)
if spec == "dsv4":
from sglang.srt.entrypoints.openai import encoding_dsv4
real_input = encoding_dsv4.encode_messages(
messages, thinking_mode=thinking_mode
)
else:
from sglang.srt.entrypoints.openai import encoding_dsv32
real_input = encoding_dsv32.encode_messages(
messages, thinking_mode=thinking_mode
)
return tokenizer.encode(real_input)
if getattr(tokenizer, "chat_template", None) is None:
raise ValueError(
"This model has no HF chat template and no custom chat encoder; "
f"cannot encode chat messages with {getattr(tokenizer, 'name_or_path', tokenizer)!r}."
)
return tokenizer.apply_chat_template(
messages, add_generation_prompt=True, tokenize=True
)
@@ -293,24 +293,15 @@ class OpenAIServingChat(OpenAIServingBase):
Override in subclass to add custom encoding specs.
"""
if self.tool_call_parser == "deepseekv4":
return "dsv4"
if self.tool_call_parser == "deepseekv32":
return "dsv32"
from sglang.srt.entrypoints.openai.chat_encoding import (
resolve_chat_encoding_spec,
)
architectures = self.tokenizer_manager.model_config.hf_config.architectures
arch = architectures[0] if architectures else ""
if "DeepseekV4" in arch:
return "dsv4"
has_chat_template = (
self.tokenizer_manager.tokenizer is not None
and self.tokenizer_manager.tokenizer.chat_template is not None
return resolve_chat_encoding_spec(
hf_config=self.tokenizer_manager.model_config.hf_config,
tokenizer=self.tokenizer_manager.tokenizer,
tool_call_parser=self.tool_call_parser,
)
if "DeepseekV3" in arch and not has_chat_template:
return "dsv32"
return None
def _request_id_prefix(self) -> str:
return "chatcmpl-"