diff --git a/experimental/sgl-router/src/policies/mod.rs b/experimental/sgl-router/src/policies/mod.rs index 5b0474c93..e473584b1 100644 --- a/experimental/sgl-router/src/policies/mod.rs +++ b/experimental/sgl-router/src/policies/mod.rs @@ -34,8 +34,9 @@ use std::sync::Arc; pub struct RequestTokens { /// The prompt token ids. pub ids: Vec, - /// Whether the token ids are safe to forward as engine `input_ids`. - pub engine_equivalent: bool, + /// Whether the IDs came from rendered chat messages. + /// Forwarding also requires the request safety guard. + pub rendered_from_chat: bool, } /// External indexer answer prepared by the async ingress path for the @@ -45,19 +46,18 @@ pub struct ExternalPrefixSignal { pub query_blocks: usize, } -/// Tokenizes a request for routing. Chat-encoder tokens are engine-equivalent; -/// raw prompt tokens are used only for routing. +/// Tokenizes requests for routing, preferring chat rendering over raw text. pub fn request_tokens_for( tokenizers: &TokenizerRegistry, model_id: &ModelId, value: &serde_json::Value, ) -> Option { - if tokenizers.has_chat_encoder(&model_id.0) { + if tokenizers.has_chat_formatter(&model_id.0) { if let Some(messages) = value.get("messages").filter(|m| m.is_array()) { if let Some(ids) = tokenizers.encode_chat(&model_id.0, messages) { return Some(RequestTokens { ids, - engine_equivalent: true, + rendered_from_chat: true, }); } } @@ -66,7 +66,7 @@ pub fn request_tokens_for( let ids = tokenize_text(tokenizers, model_id, &text)?; Some(RequestTokens { ids, - engine_equivalent: false, + rendered_from_chat: false, }) } @@ -511,11 +511,11 @@ pub trait Policy: Send + Sync + std::fmt::Debug { /// Whether this policy's routing decision needs request tokens (i.e. /// it routes by prompt prefix). Ingress tokenization itself is no longer - /// gated on this — that is a model property (`has_chat_encoder`) decided at + /// gated on this — that is a model property (`has_chat_formatter`) decided at /// ingress via [`request_tokens_for`]. This flag is the EXTRA gate that /// keeps the cache-aware policy's RAW-prompt routing path alive: a - /// cache-aware model with no chat encoder still wants its `/v1/completions` - /// /`text` prompt tokenized for tree matching, which `has_chat_encoder` + /// cache-aware model with no chat formatter still wants its `/v1/completions` + /// /`text` prompt tokenized for tree matching, which `has_chat_formatter` /// alone would not trigger. Default `false` for load-only and sticky /// routes; only the cache-aware policy overrides it. fn needs_request_tokens(&self) -> bool { diff --git a/experimental/sgl-router/src/server/metrics.rs b/experimental/sgl-router/src/server/metrics.rs index 429d15543..4c86acaf7 100644 --- a/experimental/sgl-router/src/server/metrics.rs +++ b/experimental/sgl-router/src/server/metrics.rs @@ -735,7 +735,7 @@ impl MetricsRegistry { /// Bump `sgl_router_ingress_tokenize_errors_total{model_id}`. /// - /// Count encoder failures only for chats eligible for `input_ids` + /// Count formatter failures only for chats eligible for `input_ids` /// forwarding. Requests excluded by the guard are expected fallbacks. /// Pairs with the per-model WARN log in `encode_chat`. pub fn record_ingress_tokenize_error(&self, model_id: &str) { @@ -1191,7 +1191,7 @@ impl MetricsRegistry { // ingress_tokenize_errors_total out.push_str( - "# HELP sgl_router_ingress_tokenize_errors_total Plain text chat requests on a chat-encoder model whose ingress rendering or tokenization failed, silently falling back to engine-side tokenization (the input_ids offload was defeated).\n", + "# HELP sgl_router_ingress_tokenize_errors_total Plain text chat requests on a chat-formatter model whose ingress rendering or tokenization failed, silently falling back to engine-side tokenization (the input_ids offload was defeated).\n", ); out.push_str("# TYPE sgl_router_ingress_tokenize_errors_total counter\n"); let guard = self.ingress_tokenize_errors_total.lock(); diff --git a/experimental/sgl-router/src/server/routes/chat.rs b/experimental/sgl-router/src/server/routes/chat.rs index 524520775..1dc19fb1d 100644 --- a/experimental/sgl-router/src/server/routes/chat.rs +++ b/experimental/sgl-router/src/server/routes/chat.rs @@ -504,16 +504,16 @@ pub async fn chat_completions( // Tokenize once at ingress whenever it can pay off — decoupled from the // routing policy, because forwarding `input_ids` is a property of the - // MODEL (does it have a chat encoder so the router can produce + // MODEL (does it have a chat formatter so the router can produce // engine-equivalent tokens?), not of how we pick the worker. Two gates: // - // * `has_chat_encoder` → a chat request on this model yields + // * `has_chat_formatter` → a chat request on this model yields // engine-equivalent ids we can forward as `input_ids` so the engine // skips re-tokenizing. This enables the offload for EVERY policy — // sticky and round-robin included — not just cache-aware. // * `needs_request_tokens()` → the cache-aware policy ALSO wants the // raw-prompt path tokenized for tree matching even on a model with no - // chat encoder (`/v1/completions` / `text`), which the first gate + // chat formatter (`/v1/completions` / `text`), which the first gate // alone wouldn't trigger. // // * Bucket routing also needs the prompt token count. @@ -524,7 +524,7 @@ pub async fn chat_completions( // tokenization and the outgoing-body injection below (and PD bootstrap // injection). `parse_probe` already validated the object shape. let want_tokens = should_tokenize_request( - ctx.tokenizers.has_chat_encoder(&model_str), + ctx.tokenizers.has_chat_formatter(&model_str), policy.needs_request_tokens(), ctx.bucket_selector.is_enabled(), ); @@ -798,8 +798,8 @@ pub async fn chat_completions( // Forward the router-computed tokens to the engine as `input_ids` so it // skips re-tokenizing the same prompt — but only when they are - // engine-equivalent (chat-encoder path) AND the request contains nothing - // the router's encoder didn't replicate (see `input_ids_safe_to_forward`). + // engine-equivalent (chat-formatter path) AND the request contains nothing + // the router's formatter didn't replicate (see `input_ids_safe_to_forward`). // Otherwise omit them and the engine tokenizes from `messages` as usual — // a transparent, always-correct fallback (`messages` are always retained // in the forwarded body). `forward_input_ids` is `Some` only when @@ -807,19 +807,19 @@ pub async fn chat_completions( // predicate always has a parsed body to inspect. let forward_input_ids: Option<&[u32]> = match (request_tokens.as_ref(), request_value.as_ref()) { - (Some(t), Some(v)) if t.engine_equivalent && input_ids_safe_to_forward(v) => { + (Some(t), Some(v)) if t.rendered_from_chat && input_ids_safe_to_forward(v) => { Some(t.ids.as_slice()) } _ => None, }; - // Surface a broken offload: when the encoder SHOULD have produced + // Surface a broken offload: when the formatter SHOULD have produced // engine-equivalent ids but didn't, the chat request silently fell back to // engine-side tokenization. Count only that case (see // `ingress_tokenize_offload_failed`); successful forwards and expected // omissions are not problems. if ingress_tokenize_offload_failed( - ctx.tokenizers.has_chat_encoder(&model_str), + ctx.tokenizers.has_chat_formatter(&model_str), request_value.as_ref(), request_tokens.as_ref(), ) { @@ -1198,11 +1198,11 @@ fn parse_optional_positive_f64_header( } fn should_tokenize_request( - has_chat_encoder: bool, + has_chat_formatter: bool, policy_needs_request_tokens: bool, bucket_enabled: bool, ) -> bool { - has_chat_encoder || policy_needs_request_tokens || bucket_enabled + has_chat_formatter || policy_needs_request_tokens || bucket_enabled } /// Estimate prefill-token count from the raw request body for use as @@ -1339,28 +1339,28 @@ fn build_outgoing_body( /// uses it verbatim and ignores everything that would otherwise steer its /// `messages`-side tokenization (only stop tokens / tool-call constraint are /// still taken from `messages`). So any request field that changes that -/// tokenization but which the router's chat encoder does not replicate makes +/// tokenization but which the router's chat formatter does not replicate makes /// the forwarded ids wrong. This predicate is conservative by construction — /// any such signal returns `false` and the engine tokenizes from `messages` /// (always correct). /// /// Replicated-and-safe: plain text `messages` with a string `content`. /// Not replicated → omit: -/// * `tools` / `functions` — the encoder doesn't render tool schemas. +/// * `tools` / `functions` — the formatter doesn't render tool schemas. /// * non-string or missing `content` (arrays, `null`): the engine normalizes -/// these before rendering; the router's encoder renders them verbatim. +/// these before rendering; the router's formatter renders them verbatim. /// * `chat_template` — an OpenAI-compatible per-request template override /// (e.g. vLLM); the router renders with the model's default template, so a /// custom one would diverge. (SGLang ignores it today, but block it so the /// offload stays correct across engines / future versions.) /// * `chat_template_kwargs` (carries `enable_thinking`/`thinking`), /// `reasoning` / `reasoning_effort`, `task` — thinking/mode toggles the -/// encoder renders in the engine's default mode only. +/// formatter renders in the engine's default mode only. /// * `continue_final_message: true`, or a trailing `assistant` message — the -/// engine rewrites/strips the final assistant turn; the encoder renders it +/// engine rewrites/strips the final assistant turn; the formatter renders it /// verbatim. /// -/// NOTE: the router's chat encoder renders in the engine's default +/// NOTE: the router's chat formatter renders in the engine's default /// (non-thinking) mode. Current sglang derives thinking from the request /// (`chat_template_kwargs`), which this guard already omits, so a plain request /// the router rendered matches the engine. The only way to diverge is an engine @@ -1376,7 +1376,7 @@ fn input_ids_safe_to_forward(value: &serde_json::Value) -> bool { return false; } // Fields that steer the engine's template tokenization but which the - // router's encoder does not thread through. + // router's formatter does not thread through. for key in [ "chat_template", "chat_template_kwargs", @@ -1400,15 +1400,15 @@ fn input_ids_safe_to_forward(value: &serde_json::Value) -> bool { /// Whether to increment `sgl_router_ingress_tokenize_errors_total`. /// -/// Count chats with a configured encoder that pass the forwarding guard -/// but lack engine-equivalent tokens. Excluded requests are expected fallbacks, +/// Count chats with a configured formatter that pass the forwarding guard +/// but lack chat-rendered tokens. Excluded requests are expected fallbacks, /// even when rendering fails. fn ingress_tokenize_offload_failed( - has_chat_encoder: bool, + has_chat_formatter: bool, request_value: Option<&serde_json::Value>, request_tokens: Option<&RequestTokens>, ) -> bool { - if !has_chat_encoder { + if !has_chat_formatter { return false; } let chat_request = request_value.is_some_and(|v| { @@ -1417,7 +1417,7 @@ fn ingress_tokenize_offload_failed( if !chat_request { return false; } - !request_tokens.is_some_and(|t| t.engine_equivalent) + !request_tokens.is_some_and(|t| t.rendered_from_chat) } /// Whether the final chat message has `role: "assistant"` (a prefix / @@ -1433,7 +1433,7 @@ fn last_message_is_assistant(value: &serde_json::Value) -> bool { } /// Whether the request carries tool / function definitions. The router's chat -/// encoder renders only `messages`, so its `input_ids` would omit the tool +/// formatter renders only `messages`, so its `input_ids` would omit the tool /// schemas the engine's template injects into the prompt — the caller must let /// the engine tokenize these itself. fn request_has_tools(value: &serde_json::Value) -> bool { @@ -1448,7 +1448,7 @@ fn request_has_tools(value: &serde_json::Value) -> bool { } /// Detect non-string or missing content, which requires engine tokenization: -/// the engine normalizes arrays and nulls differently from the router's encoder. +/// the engine normalizes arrays and nulls differently from the router's formatter. fn request_has_non_text_content(value: &serde_json::Value) -> bool { value .get("messages") @@ -1747,7 +1747,7 @@ mod tests { } /// Tool / function requests are detected so the caller omits `input_ids` - /// (the router's encoder doesn't render tools). + /// (the router's formatter doesn't render tools). #[test] fn request_has_tools_detects_tools_and_functions() { assert!(request_has_tools( @@ -1792,7 +1792,7 @@ mod tests { } /// Every field the engine honors on the `messages` path but which the - /// router's encoder does not replicate must block forwarding — otherwise + /// router's formatter does not replicate must block forwarding — otherwise /// the engine uses the router's ids verbatim and silently runs a different /// prompt than the request asked for. #[test] @@ -1848,14 +1848,14 @@ mod tests { assert!(parsed.get("input_ids").is_none()); } - /// A chat request on a chat-encoder model that yields engine-equivalent + /// A chat request on a chat-formatter model that yields engine-equivalent /// ids (encode succeeded) is NOT a failure — the offload worked. #[test] - fn offload_failed_false_when_tokens_engine_equivalent() { + fn offload_failed_false_when_tokens_rendered_from_chat() { let value = serde_json::json!({"messages":[{"role":"user","content":"hi"}]}); let tokens = RequestTokens { ids: vec![1, 2, 3], - engine_equivalent: true, + rendered_from_chat: true, }; assert!(!ingress_tokenize_offload_failed( true, @@ -1874,22 +1874,22 @@ mod tests { assert!(!ingress_tokenize_offload_failed(true, Some(&value), None)); } - /// Missing tokens count as a failure for an eligible chat with an encoder. + /// Missing tokens count as a failure for an eligible chat with a formatter. #[test] - fn offload_failed_true_when_chat_encoder_request_has_no_tokens() { + fn offload_failed_true_when_chat_formatter_request_has_no_tokens() { let value = serde_json::json!({"messages":[{"role":"user","content":"hi"}]}); assert!(ingress_tokenize_offload_failed(true, Some(&value), None)); } - /// Encode produced ids but NOT via the chat encoder (raw fallback, - /// `engine_equivalent = false`) on a chat-encoder model + chat request → + /// Encode produced ids but NOT via the chat formatter (raw fallback, + /// `rendered_from_chat = false`) on a chat-formatter model + chat request → /// the chat-encode render/encode failed and fell through to the raw path. #[test] - fn offload_failed_true_when_tokens_not_engine_equivalent() { + fn offload_failed_true_when_tokens_not_rendered_from_chat() { let value = serde_json::json!({"messages":[{"role":"user","content":"hi"}]}); let tokens = RequestTokens { ids: vec![1, 2, 3], - engine_equivalent: false, + rendered_from_chat: false, }; assert!(ingress_tokenize_offload_failed( true, @@ -1898,15 +1898,15 @@ mod tests { )); } - /// Non-chat-encoder models never expected the offload → not a failure even + /// Non-chat-formatter models never expected the offload → not a failure even /// with no tokens. #[test] - fn offload_failed_false_without_chat_encoder() { + fn offload_failed_false_without_chat_formatter() { let value = serde_json::json!({"messages":[{"role":"user","content":"hi"}]}); assert!(!ingress_tokenize_offload_failed(false, Some(&value), None)); } - /// A non-chat (no `messages`) request on a chat-encoder model — e.g. + /// A non-chat (no `messages`) request on a chat-formatter model — e.g. /// `/v1/completions` `prompt` — never expected the chat-encode offload, so /// the absence of engine-equivalent ids is not a failure. #[test] diff --git a/experimental/sgl-router/src/tokenizer/mod.rs b/experimental/sgl-router/src/tokenizer/mod.rs index f9f80044c..c246524e2 100644 --- a/experimental/sgl-router/src/tokenizer/mod.rs +++ b/experimental/sgl-router/src/tokenizer/mod.rs @@ -15,7 +15,7 @@ use std::sync::Arc; /// How to turn a chat request's `messages` into the prompt the engine tokenizes /// and caches. Cache-aware routing renders this before hashing so its query /// tokens match the engine's stored blocks. -pub enum ChatEncoder { +pub enum ChatFormatter { /// HuggingFace Jinja chat template from `tokenizer_config.json` (most /// models). Boxed: it holds a minijinja `Environment`, far larger than the /// other variants. @@ -24,26 +24,26 @@ pub enum ChatEncoder { DeepSeekV4, } -impl ChatEncoder { +impl ChatFormatter { /// Render `messages` into the engine-equivalent prompt text. fn render(&self, messages: &serde_json::Value) -> Result { match self { - ChatEncoder::Jinja(t) => t.render(messages), - ChatEncoder::DeepSeekV4 => Ok(dsv4::render_messages(messages)), + ChatFormatter::Jinja(t) => t.render(messages), + ChatFormatter::DeepSeekV4 => Ok(dsv4::render_messages(messages)), } } } -/// A model's chat encoder plus its fallback-logging state. -struct ChatEncoderEntry { - encoder: ChatEncoder, +/// A model's chat formatter plus its fallback-logging state. +struct ChatFormatterEntry { + formatter: ChatFormatter, fallback_warned: AtomicBool, } -impl ChatEncoderEntry { - fn new(encoder: ChatEncoder) -> Self { +impl ChatFormatterEntry { + fn new(formatter: ChatFormatter) -> Self { Self { - encoder, + formatter, fallback_warned: AtomicBool::new(false), } } @@ -56,11 +56,11 @@ impl ChatEncoderEntry { fn log_fallback(&self, model_id: &str, cause: &str) { if !self.fallback_warned.swap(true, Ordering::Relaxed) { tracing::warn!(model = %model_id, %cause, - "chat-encoder failed; falling back to raw prompt-text hashing \ + "chat-formatter failed; falling back to raw prompt-text hashing \ (cache-aware overlap degrades for this model; further failures log at debug)"); } else { tracing::debug!(model = %model_id, %cause, - "chat-encoder failed; falling back to raw prompt-text hashing"); + "chat-formatter failed; falling back to raw prompt-text hashing"); } } } @@ -68,12 +68,12 @@ impl ChatEncoderEntry { #[derive(Default)] pub struct TokenizerRegistry { inner: DashMap>, - /// Per-model chat encoder, present only when the model's prompt format is - /// known (a `tokenizer_config.json` chat template, or a built-in encoder + /// Per-model chat formatter, present only when the model's prompt format is + /// known (a `tokenizer_config.json` chat template, or a built-in formatter /// like DeepSeek-V4's). Cache-aware routing uses it to tokenize chat /// requests the way the engine does; models without one fall back to raw /// prompt-text tokenization. - encoders: DashMap>, + formatters: DashMap>, } impl std::fmt::Debug for TokenizerRegistry { @@ -90,28 +90,32 @@ impl TokenizerRegistry { let m = &cfg.model; let t = adapter::load(&m.tokenizer_path)?; me.inner.insert(m.id.clone(), t); - // Resolve the chat encoder, best-effort: a Jinja template from - // tokenizer_config.json, else a built-in encoder for a recognized model + // Resolve the chat formatter, best-effort: a Jinja template from + // tokenizer_config.json, else a built-in formatter for a recognized model // (DeepSeek-V4), else none (chat traffic routes via raw text). Every // path logs its outcome — whether chat-aware routing is live for this // model is the single most useful signal for diagnosing "cache-aware // routing degraded to overlap=0 on chat traffic", so it must never be // silent. - if let Some(encoder) = me.resolve_chat_encoder(&m.id, &m.tokenizer_path) { - me.encoders - .insert(m.id.clone(), Arc::new(ChatEncoderEntry::new(encoder))); + if let Some(formatter) = me.resolve_chat_formatter(&m.id, &m.tokenizer_path) { + me.formatters + .insert(m.id.clone(), Arc::new(ChatFormatterEntry::new(formatter))); } Ok(me) } - /// Pick the chat encoder for a model, logging the outcome on every branch. - fn resolve_chat_encoder(&self, model_id: &str, tokenizer_path: &str) -> Option { + /// Pick the chat formatter for a model, logging the outcome on every branch. + fn resolve_chat_formatter( + &self, + model_id: &str, + tokenizer_path: &str, + ) -> Option { match adapter::load_tokenizer_config(tokenizer_path) { Ok(Some(cfg_json)) => match ChatTemplate::from_tokenizer_config(&cfg_json) { Ok(Some(tmpl)) => { tracing::info!(model = %model_id, "chat-template routing enabled; chat requests route by templated tokens"); - return Some(ChatEncoder::Jinja(Box::new(tmpl))); + return Some(ChatFormatter::Jinja(Box::new(tmpl))); } Ok(None) => {} // no template — fall through to built-in detection Err(e) => tracing::warn!(model = %model_id, error = %e, @@ -123,11 +127,11 @@ impl TokenizerRegistry { } if is_deepseek_v4(model_id) { tracing::info!(model = %model_id, - "DeepSeek-V4 routing enabled; chat requests route via the built-in V4 encoder"); - return Some(ChatEncoder::DeepSeekV4); + "DeepSeek-V4 routing enabled; chat requests route via the built-in V4 formatter"); + return Some(ChatFormatter::DeepSeekV4); } tracing::info!(model = %model_id, - "no chat template or built-in encoder; chat traffic routes via raw prompt text"); + "no chat template or built-in formatter; chat traffic routes via raw prompt text"); None } @@ -135,24 +139,24 @@ impl TokenizerRegistry { self.inner.get(model_id).map(|r| Arc::clone(&*r)) } - /// Whether this model has a chat encoder (and thus the chat-aware + /// Whether this model has a chat formatter (and thus the chat-aware /// tokenization path is available for it). - pub fn has_chat_encoder(&self, model_id: &str) -> bool { - self.encoders.contains_key(model_id) + pub fn has_chat_formatter(&self, model_id: &str) -> bool { + self.formatters.contains_key(model_id) } - /// Render `messages` through the model's chat encoder, then tokenize the + /// Render `messages` through the model's chat formatter, then tokenize the /// result the same way the engine does (`add_special_tokens = false`, so the - /// encoder's literal `bos_token`/role markers carry the specials). Returns + /// formatter's literal `bos_token`/role markers carry the specials). Returns /// `None` — caller falls back to raw routing — when the model has no - /// encoder, no tokenizer, or rendering/encoding fails or yields no tokens. + /// formatter, no tokenizer, or rendering/encoding fails or yields no tokens. pub fn encode_chat(&self, model_id: &str, messages: &serde_json::Value) -> Option> { // Clone the Arc and drop the DashMap guard before the CPU-bound // render+encode (mirrors `get`), so no shard read-lock is held across it. - let entry = Arc::clone(&*self.encoders.get(model_id)?); + let entry = Arc::clone(&*self.formatters.get(model_id)?); let tokenizer = self.get(model_id)?; let rendered = entry - .encoder + .formatter .render(messages) .inspect_err(|e| { // `{e:#}` prints the full anyhow chain, so the underlying @@ -178,18 +182,18 @@ impl TokenizerRegistry { self.inner.iter().map(|kv| kv.key().clone()).collect() } - /// Attach a chat encoder to an already-loaded model. Lets policy tests in + /// Attach a chat formatter to an already-loaded model. Lets policy tests in /// other modules exercise the chat-aware routing path without a co-located /// fixture. #[cfg(test)] - pub(crate) fn attach_chat_encoder_for_test(&self, model_id: &str, encoder: ChatEncoder) { - self.encoders.insert( + pub(crate) fn attach_chat_formatter_for_test(&self, model_id: &str, formatter: ChatFormatter) { + self.formatters.insert( model_id.to_string(), - Arc::new(ChatEncoderEntry::new(encoder)), + Arc::new(ChatFormatterEntry::new(formatter)), ); } - /// Convenience: attach a Jinja chat encoder built from an inline + /// Convenience: attach a Jinja chat formatter built from an inline /// `tokenizer_config.json` value. #[cfg(test)] pub(crate) fn attach_chat_template_for_test( @@ -200,12 +204,12 @@ impl TokenizerRegistry { let template = ChatTemplate::from_tokenizer_config(tokenizer_config) .expect("valid test chat template") .expect("test tokenizer_config has a chat_template"); - self.attach_chat_encoder_for_test(model_id, ChatEncoder::Jinja(Box::new(template))); + self.attach_chat_formatter_for_test(model_id, ChatFormatter::Jinja(Box::new(template))); } } /// Whether `model_id` denotes a DeepSeek-V4 model, which the engine encodes via -/// the built-in [`dsv4`] encoder rather than a Jinja template. Heuristic on the +/// the built-in [`dsv4`] formatter rather than a Jinja template. Heuristic on the /// served model id (the router has no model architecture from `/server_info`); /// scoped to "deepseek" + "v4" so it doesn't claim V3-family models, whose /// encoding differs. @@ -420,7 +424,7 @@ mod tests { "bos_token": "", }); reg.attach_chat_template_for_test("tiny", &cfg); - assert!(reg.has_chat_encoder("tiny")); + assert!(reg.has_chat_formatter("tiny")); let messages = serde_json::json!([{"role":"user","content":"hi"}]); let chat_ids = reg.encode_chat("tiny", &messages).expect("encode_chat"); @@ -435,10 +439,10 @@ mod tests { // encode_chat is exactly tokenize(render(messages)). let rendered = reg - .encoders + .formatters .get("tiny") .unwrap() - .encoder + .formatter .render(&messages) .unwrap(); assert_eq!(chat_ids, adapter::encode(&tok, &rendered).unwrap()); @@ -451,7 +455,7 @@ mod tests { "tiny".into(), adapter::load("tests/fixtures/tiny_tokenizer.json").unwrap(), ); - assert!(!reg.has_chat_encoder("tiny")); + assert!(!reg.has_chat_formatter("tiny")); let messages = serde_json::json!([{"role":"user","content":"hi"}]); assert!(reg.encode_chat("tiny", &messages).is_none()); } @@ -473,7 +477,7 @@ mod tests { "bos_token": "", }), ); - assert!(reg.has_chat_encoder("tiny")); + assert!(reg.has_chat_formatter("tiny")); let messages = serde_json::json!([{"role":"user","content":"hi"}]); assert!( reg.encode_chat("tiny", &messages).is_none(), diff --git a/experimental/sgl-router/tests/proxy/cache_aware_input_ids.rs b/experimental/sgl-router/tests/proxy/cache_aware_input_ids.rs index 816825ddd..d4b7cda17 100644 --- a/experimental/sgl-router/tests/proxy/cache_aware_input_ids.rs +++ b/experimental/sgl-router/tests/proxy/cache_aware_input_ids.rs @@ -6,7 +6,7 @@ //! skips re-tokenizing the same prompt). Asserts the gating contract through //! the real chat handler + a MockWorker backend: //! -//! * A plain text chat request on the engine-equivalent chat-encoder path → +//! * A plain text chat request on the engine-equivalent chat-formatter path → //! the forwarded body carries `input_ids` AND retains `messages`. //! * A request carrying `tools` → `input_ids` omitted (the router's encoder //! doesn't render tool schemas, so its ids would diverge from the engine). @@ -35,8 +35,8 @@ fn build_ctx(url: String) -> Arc { let cfg = config(); let tokenizers = Arc::new(TokenizerRegistry::load_from_config(&cfg).unwrap()); assert!( - tokenizers.has_chat_encoder(MODEL), - "deepseek-v4 model id must auto-attach the built-in chat encoder" + tokenizers.has_chat_formatter(MODEL), + "deepseek-v4 model id must auto-attach the built-in chat formatter" ); let registry = Arc::new(WorkerRegistry::default()); let _ = registry.add(WorkerSpec { diff --git a/experimental/sgl-router/tests/proxy/common/cache_aware_fixture.rs b/experimental/sgl-router/tests/proxy/common/cache_aware_fixture.rs index 67feee682..72cdc4ab5 100644 --- a/experimental/sgl-router/tests/proxy/common/cache_aware_fixture.rs +++ b/experimental/sgl-router/tests/proxy/common/cache_aware_fixture.rs @@ -4,7 +4,7 @@ //! Shared router config for the cache-aware proxy tests. //! //! The model id contains `deepseek-v4` so the tokenizer registry auto-attaches the -//! built-in V4 chat encoder — the engine-equivalent path — with no template fixture. +//! built-in V4 chat formatter — the engine-equivalent path — with no template fixture. use sgl_router::config::{ ActiveLoadConfig, CacheAwareConfig, Config, DiscoveryBackend, ModelConfig, ObservabilityConfig, diff --git a/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs b/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs index 4c970b2bd..cbe84b272 100644 --- a/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs +++ b/experimental/sgl-router/tests/proxy/roundrobin_input_ids.rs @@ -2,10 +2,10 @@ // SPDX-License-Identifier: Apache-2.0 //! `input_ids` forwarding is policy-independent: a load-only **round-robin** -//! policy on a chat-encoder model still forwards `input_ids` to the engine +//! policy on a chat-formatter model still forwards `input_ids` to the engine //! (the engine-tokenization offload), even though it picks workers round-robin //! and ignores the tokens for routing. Tokenization is gated on the model's -//! chat encoder at ingress, not on the policy. +//! chat formatter at ingress, not on the policy. use axum::body::Body; use axum::http::{Request, StatusCode}; @@ -66,7 +66,7 @@ fn build_ctx(url: String) -> Arc { // The handler tokenizes via the AppContext's registry (which carries the V4 // encoder); the RoundRobin policy itself needs no tokenizer. let tokenizers = Arc::new(TokenizerRegistry::load_from_config(&cfg).unwrap()); - assert!(tokenizers.has_chat_encoder(MODEL)); + assert!(tokenizers.has_chat_formatter(MODEL)); let registry = Arc::new(WorkerRegistry::default()); let _ = registry.add(WorkerSpec { id: WorkerId(url.clone()), @@ -103,7 +103,7 @@ fn captured(mock: &MockWorker) -> Value { } /// A round-robin (load-only) policy still forwards `input_ids` on a -/// chat-encoder model — the offload is decoupled from routing. +/// chat-formatter model — the offload is decoupled from routing. #[tokio::test] async fn round_robin_plain_chat_forwards_input_ids() { let mock = MockWorker::start(vec![]).await; @@ -122,7 +122,7 @@ async fn round_robin_plain_chat_forwards_input_ids() { let ids = body.get("input_ids").and_then(|v| v.as_array()); assert!( ids.is_some_and(|a| !a.is_empty()), - "round-robin must forward input_ids on a chat-encoder model; got {body}" + "round-robin must forward input_ids on a chat-formatter model; got {body}" ); assert!( body.get("messages").is_some(), @@ -154,7 +154,7 @@ async fn round_robin_tool_request_omits_input_ids() { ); } -/// A successful plain-chat forward on a chat-encoder model must NOT emit +/// A successful plain-chat forward on a chat-formatter model must NOT emit /// `sgl_router_ingress_tokenize_errors_total` — that counter fires only when the /// offload was expected but the encoder failed. A tool request on the same model /// is an *expected* omission (its ids are still engine-equivalent; the diff --git a/experimental/sgl-router/tests/proxy/sticky_input_ids.rs b/experimental/sgl-router/tests/proxy/sticky_input_ids.rs index 237a0f5fe..1f22f38cc 100644 --- a/experimental/sgl-router/tests/proxy/sticky_input_ids.rs +++ b/experimental/sgl-router/tests/proxy/sticky_input_ids.rs @@ -3,8 +3,8 @@ //! Tokenize-once at ingress under the STICKY policy. The engine-tokenization //! offload (`input_ids` forwarding) is a property of the MODEL — does it have a -//! chat encoder? — not of the routing policy, so a sticky-routed request on a -//! chat-encoder model must forward `input_ids` exactly like cache-aware does, +//! chat formatter? — not of the routing policy, so a sticky-routed request on a +//! chat-formatter model must forward `input_ids` exactly like cache-aware does, //! while still pinning sessions O(1) by header. //! //! Asserts through the real chat handler + `MockWorker` backends: @@ -17,7 +17,7 @@ //! routing is unchanged by the added tokenization). //! //! The model id contains `deepseek-v4` so the tokenizer registry auto-attaches -//! the built-in V4 chat encoder — the engine-equivalent path — without a +//! the built-in V4 chat formatter — the engine-equivalent path — without a //! template fixture. use axum::body::Body; @@ -82,14 +82,14 @@ fn config() -> Config { /// Build an `AppContext` running the sticky policy over the given workers. /// The tokenizer registry is loaded from config (real tiny tokenizer + the -/// auto-attached V4 chat encoder) so the ingress can tokenize — the sticky +/// auto-attached V4 chat formatter) so the ingress can tokenize — the sticky /// policy itself holds no tokenizer. fn build_ctx(worker_urls: &[String]) -> Arc { let cfg = config(); let tokenizers = Arc::new(TokenizerRegistry::load_from_config(&cfg).unwrap()); assert!( - tokenizers.has_chat_encoder(MODEL), - "deepseek-v4 model id must auto-attach the built-in chat encoder" + tokenizers.has_chat_formatter(MODEL), + "deepseek-v4 model id must auto-attach the built-in chat formatter" ); let registry = Arc::new(WorkerRegistry::default()); for (i, url) in worker_urls.iter().enumerate() {