Fix corrupted chat prompts on mistral_common tokenizers (tool_choice auto never fires) (#39773)
Co-authored-by: Xinyuan Tong <xinyuantong.cs@gmail.com>
This commit is contained in:
co-authored by
Xinyuan Tong
parent
2394b231c2
commit
21e6c98ccb
@@ -366,9 +366,42 @@ class OpenAIServingChat(OpenAIServingBase):
|
||||
)
|
||||
except Exception:
|
||||
self._tokenizer_auto_adds_specials = True
|
||||
self._chat_template_cache: OrderedDict[
|
||||
bytes, tuple[str, tuple[int, ...], str]
|
||||
] = OrderedDict()
|
||||
self._prompt_text_round_trip_is_lossy = self._probe_prompt_text_round_trip()
|
||||
self._chat_template_cache: OrderedDict[bytes, tuple[tuple[int, ...], str]] = (
|
||||
OrderedDict()
|
||||
)
|
||||
|
||||
def _probe_prompt_text_round_trip(self) -> bool:
|
||||
"""Does rendering the chat template to text and re-encoding lose anything?
|
||||
|
||||
mistral_common tokenizers emit control tokens ([INST],
|
||||
[AVAILABLE_TOOLS], ...) that have no text form. Rendering to a string
|
||||
turns them into literal characters and re-encoding also prepends a
|
||||
second BOS, so the model sees the letters "AVAILABLE_TOOLS" instead of
|
||||
the control token that frames the tool block. Encoding straight to ids
|
||||
is the only faithful route on such tokenizers, so compare the two here
|
||||
once and remember which to trust.
|
||||
"""
|
||||
probe = [{"role": "user", "content": "x"}]
|
||||
try:
|
||||
tokenizer = self.tokenizer_manager.tokenizer
|
||||
rendered = tokenizer.apply_chat_template(
|
||||
probe, tokenize=False, add_generation_prompt=True, return_dict=False
|
||||
)
|
||||
encode_kwargs = (
|
||||
{"add_special_tokens": False}
|
||||
if self._tokenizer_auto_adds_specials
|
||||
else {}
|
||||
)
|
||||
via_text = tokenizer.encode(rendered, **encode_kwargs)
|
||||
via_ids = tokenizer.apply_chat_template(
|
||||
probe, tokenize=True, add_generation_prompt=True, return_dict=False
|
||||
)
|
||||
return list(via_text) != list(via_ids)
|
||||
except Exception:
|
||||
# A template that needs kwargs this probe does not supply tells us
|
||||
# nothing; keep the long-standing text path.
|
||||
return False
|
||||
|
||||
def _handle_last_assistant_message(
|
||||
self,
|
||||
@@ -1084,8 +1117,23 @@ class OpenAIServingChat(OpenAIServingBase):
|
||||
pre-rendered input_ids with single placeholder ids and leave the text
|
||||
empty; pass those through rather than re-tokenizing an empty prompt.
|
||||
"""
|
||||
if is_multimodal and not chat_encoding.spec_renders_prompt_ids(
|
||||
self.chat_encoding_spec
|
||||
# A lossy text round-trip makes the rendered prompt unusable, so send the
|
||||
# ids instead. Only when nothing needs placeholder expansion: with media
|
||||
# attached the MM processor still has to tokenize the text itself.
|
||||
prefers_prompt_ids = (
|
||||
self._prompt_text_round_trip_is_lossy
|
||||
and isinstance(processed_messages.prompt_ids, list)
|
||||
and processed_messages.prompt_ids
|
||||
and not (
|
||||
processed_messages.image_data
|
||||
or processed_messages.video_data
|
||||
or processed_messages.audio_data
|
||||
)
|
||||
)
|
||||
if (
|
||||
is_multimodal
|
||||
and not chat_encoding.spec_renders_prompt_ids(self.chat_encoding_spec)
|
||||
and not prefers_prompt_ids
|
||||
):
|
||||
return "text", processed_messages.prompt
|
||||
if isinstance(processed_messages.prompt_ids, str):
|
||||
@@ -1603,14 +1651,12 @@ class OpenAIServingChat(OpenAIServingBase):
|
||||
else {}
|
||||
)
|
||||
try:
|
||||
rendered_prompt, prompt_ids, decoded_prompt = (
|
||||
self._render_and_encode_chat_template(
|
||||
openai_compatible_messages,
|
||||
tools=tools,
|
||||
template_kwargs=extra_template_kwargs,
|
||||
encode_kwargs=encode_kwargs,
|
||||
use_cache=is_multimodal,
|
||||
)
|
||||
prompt_ids, decoded_prompt = self._render_and_encode_chat_template(
|
||||
openai_compatible_messages,
|
||||
tools=tools,
|
||||
template_kwargs=extra_template_kwargs,
|
||||
encode_kwargs=encode_kwargs,
|
||||
use_cache=is_multimodal,
|
||||
)
|
||||
except Exception:
|
||||
# If the first attempt fails, try with flat function-only format.
|
||||
@@ -1621,14 +1667,12 @@ class OpenAIServingChat(OpenAIServingBase):
|
||||
else None
|
||||
)
|
||||
try:
|
||||
rendered_prompt, prompt_ids, decoded_prompt = (
|
||||
self._render_and_encode_chat_template(
|
||||
openai_compatible_messages,
|
||||
tools=tools,
|
||||
template_kwargs=extra_template_kwargs,
|
||||
encode_kwargs=encode_kwargs,
|
||||
use_cache=is_multimodal,
|
||||
)
|
||||
prompt_ids, decoded_prompt = self._render_and_encode_chat_template(
|
||||
openai_compatible_messages,
|
||||
tools=tools,
|
||||
template_kwargs=extra_template_kwargs,
|
||||
encode_kwargs=encode_kwargs,
|
||||
use_cache=is_multimodal,
|
||||
)
|
||||
except _CHAT_TEMPLATE_CLIENT_ERRORS as template_error:
|
||||
# Template errors (e.g., from raise_exception in Jinja templates)
|
||||
@@ -1674,7 +1718,7 @@ class OpenAIServingChat(OpenAIServingBase):
|
||||
template_kwargs: Dict[str, Any],
|
||||
encode_kwargs: Dict[str, Any],
|
||||
use_cache: bool,
|
||||
) -> tuple[str, List[int], Optional[str]]:
|
||||
) -> tuple[List[int], Optional[str]]:
|
||||
cache_key = None
|
||||
if use_cache:
|
||||
try:
|
||||
@@ -1699,20 +1743,31 @@ class OpenAIServingChat(OpenAIServingBase):
|
||||
cached = self._chat_template_cache.get(cache_key)
|
||||
if cached is not None:
|
||||
self._chat_template_cache.move_to_end(cache_key)
|
||||
rendered_prompt, prompt_ids, decoded_prompt = cached
|
||||
return rendered_prompt, list(prompt_ids), decoded_prompt
|
||||
prompt_ids, decoded_prompt = cached
|
||||
return list(prompt_ids), decoded_prompt
|
||||
|
||||
rendered_prompt = self.tokenizer_manager.tokenizer.apply_chat_template(
|
||||
messages,
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
tools=tools,
|
||||
return_dict=False,
|
||||
**template_kwargs,
|
||||
)
|
||||
prompt_ids = self.tokenizer_manager.tokenizer.encode(
|
||||
rendered_prompt, **encode_kwargs
|
||||
)
|
||||
if self._prompt_text_round_trip_is_lossy:
|
||||
# Re-encoding rendered text would drop the template's control tokens.
|
||||
prompt_ids = self.tokenizer_manager.tokenizer.apply_chat_template(
|
||||
messages,
|
||||
tokenize=True,
|
||||
add_generation_prompt=True,
|
||||
tools=tools,
|
||||
return_dict=False,
|
||||
**template_kwargs,
|
||||
)
|
||||
else:
|
||||
rendered_prompt = self.tokenizer_manager.tokenizer.apply_chat_template(
|
||||
messages,
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
tools=tools,
|
||||
return_dict=False,
|
||||
**template_kwargs,
|
||||
)
|
||||
prompt_ids = self.tokenizer_manager.tokenizer.encode(
|
||||
rendered_prompt, **encode_kwargs
|
||||
)
|
||||
decoded_prompt = (
|
||||
self.tokenizer_manager.tokenizer.decode(prompt_ids)
|
||||
if cache_key is not None
|
||||
@@ -1720,15 +1775,11 @@ class OpenAIServingChat(OpenAIServingBase):
|
||||
)
|
||||
|
||||
if cache_key is not None:
|
||||
self._chat_template_cache[cache_key] = (
|
||||
rendered_prompt,
|
||||
tuple(prompt_ids),
|
||||
decoded_prompt,
|
||||
)
|
||||
self._chat_template_cache[cache_key] = (tuple(prompt_ids), decoded_prompt)
|
||||
if len(self._chat_template_cache) > _CHAT_TEMPLATE_CACHE_MAX_SIZE:
|
||||
self._chat_template_cache.popitem(last=False)
|
||||
|
||||
return rendered_prompt, prompt_ids, decoded_prompt
|
||||
return prompt_ids, decoded_prompt
|
||||
|
||||
def _apply_conversation_template(
|
||||
self,
|
||||
|
||||
Reference in New Issue
Block a user