[sgl-router] Forward input_ids only for string content; count tokenize errors only when forwardable (#39458)
Co-authored-by: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5.1
parent
cc171fbad0
commit
e41026f434
@@ -735,15 +735,9 @@ impl MetricsRegistry {
|
|||||||
|
|
||||||
/// Bump `sgl_router_ingress_tokenize_errors_total{model_id}`.
|
/// Bump `sgl_router_ingress_tokenize_errors_total{model_id}`.
|
||||||
///
|
///
|
||||||
/// Recorded ONLY when the tokenization offload SHOULD have fired but the
|
/// Count encoder failures only for chats eligible for `input_ids`
|
||||||
/// router's chat encoder failed: a chat request (`messages`) on a model with
|
/// forwarding. Requests excluded by the guard are expected fallbacks.
|
||||||
/// a chat encoder that did not yield engine-equivalent ids. That request
|
/// Pairs with the per-model WARN log in `encode_chat`.
|
||||||
/// silently fell back to engine-side tokenization, defeating the offload —
|
|
||||||
/// the actionable "offload broken" signal. It stays at ~0 in healthy
|
|
||||||
/// operation and climbs only on a real tokenizer problem; successful
|
|
||||||
/// forwards and expected omissions (tools / multimodal / thinking, whose
|
|
||||||
/// ids are engine-equivalent but withheld by the safe-predicate) are NOT
|
|
||||||
/// counted. Pairs with the per-occurrence WARN log in `tokenize_text`.
|
|
||||||
pub fn record_ingress_tokenize_error(&self, model_id: &str) {
|
pub fn record_ingress_tokenize_error(&self, model_id: &str) {
|
||||||
let mut guard = self.ingress_tokenize_errors_total.lock();
|
let mut guard = self.ingress_tokenize_errors_total.lock();
|
||||||
let counter = guard
|
let counter = guard
|
||||||
@@ -1197,7 +1191,7 @@ impl MetricsRegistry {
|
|||||||
|
|
||||||
// ingress_tokenize_errors_total
|
// ingress_tokenize_errors_total
|
||||||
out.push_str(
|
out.push_str(
|
||||||
"# HELP sgl_router_ingress_tokenize_errors_total Chat requests on a chat-encoder model whose ingress tokenization failed, silently falling back to engine-side tokenization (the input_ids offload was defeated).\n",
|
"# HELP sgl_router_ingress_tokenize_errors_total Plain text chat requests on a chat-encoder model whose ingress rendering or tokenization failed, silently falling back to engine-side tokenization (the input_ids offload was defeated).\n",
|
||||||
);
|
);
|
||||||
out.push_str("# TYPE sgl_router_ingress_tokenize_errors_total counter\n");
|
out.push_str("# TYPE sgl_router_ingress_tokenize_errors_total counter\n");
|
||||||
let guard = self.ingress_tokenize_errors_total.lock();
|
let guard = self.ingress_tokenize_errors_total.lock();
|
||||||
|
|||||||
@@ -1347,7 +1347,8 @@ fn build_outgoing_body(
|
|||||||
/// Replicated-and-safe: plain text `messages` with a string `content`.
|
/// Replicated-and-safe: plain text `messages` with a string `content`.
|
||||||
/// Not replicated → omit:
|
/// Not replicated → omit:
|
||||||
/// * `tools` / `functions` — the encoder doesn't render tool schemas.
|
/// * `tools` / `functions` — the encoder doesn't render tool schemas.
|
||||||
/// * multimodal (array) `content` — a text tokenizer can't represent images.
|
/// * non-string or missing `content` (arrays, `null`): the engine normalizes
|
||||||
|
/// these before rendering; the router's encoder renders them verbatim.
|
||||||
/// * `chat_template` — an OpenAI-compatible per-request template override
|
/// * `chat_template` — an OpenAI-compatible per-request template override
|
||||||
/// (e.g. vLLM); the router renders with the model's default template, so a
|
/// (e.g. vLLM); the router renders with the model's default template, so a
|
||||||
/// custom one would diverge. (SGLang ignores it today, but block it so the
|
/// custom one would diverge. (SGLang ignores it today, but block it so the
|
||||||
@@ -1371,7 +1372,7 @@ fn build_outgoing_body(
|
|||||||
/// tokenizer that does not would diverge by a leading special, again undetectable
|
/// tokenizer that does not would diverge by a leading special, again undetectable
|
||||||
/// from the request.
|
/// from the request.
|
||||||
fn input_ids_safe_to_forward(value: &serde_json::Value) -> bool {
|
fn input_ids_safe_to_forward(value: &serde_json::Value) -> bool {
|
||||||
if request_has_tools(value) || request_is_multimodal(value) {
|
if request_has_tools(value) || request_has_non_text_content(value) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
// Fields that steer the engine's template tokenization but which the
|
// Fields that steer the engine's template tokenization but which the
|
||||||
@@ -1397,22 +1398,11 @@ fn input_ids_safe_to_forward(value: &serde_json::Value) -> bool {
|
|||||||
!last_message_is_assistant(value)
|
!last_message_is_assistant(value)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether the ingress tokenization offload was expected to fire but failed —
|
/// Whether to increment `sgl_router_ingress_tokenize_errors_total`.
|
||||||
/// the condition behind `sgl_router_ingress_tokenize_errors_total`.
|
|
||||||
///
|
///
|
||||||
/// True only when ALL of:
|
/// Count chats with a configured encoder that pass the forwarding guard
|
||||||
/// * the model has a chat encoder (`has_chat_encoder`), so a chat request
|
/// but lack engine-equivalent tokens. Excluded requests are expected fallbacks,
|
||||||
/// on it SHOULD have produced engine-equivalent ids;
|
/// even when rendering fails.
|
||||||
/// * the request is a chat request (`messages` array present);
|
|
||||||
/// * the tokens are absent OR not engine-equivalent — i.e. `encode_chat`
|
|
||||||
/// render/encode failed and the request silently fell back to engine-side
|
|
||||||
/// tokenization.
|
|
||||||
///
|
|
||||||
/// Non-chat-encoder / non-`messages` requests never expected the offload, so
|
|
||||||
/// they are not failures. A tools / multimodal / thinking request on a
|
|
||||||
/// chat-encoder model still gets engine-equivalent ids (`encode_chat`
|
|
||||||
/// succeeded; the safe-predicate withholds forwarding for other reasons), so it
|
|
||||||
/// is an expected omission, not a failure.
|
|
||||||
fn ingress_tokenize_offload_failed(
|
fn ingress_tokenize_offload_failed(
|
||||||
has_chat_encoder: bool,
|
has_chat_encoder: bool,
|
||||||
request_value: Option<&serde_json::Value>,
|
request_value: Option<&serde_json::Value>,
|
||||||
@@ -1421,8 +1411,9 @@ fn ingress_tokenize_offload_failed(
|
|||||||
if !has_chat_encoder {
|
if !has_chat_encoder {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
let chat_request =
|
let chat_request = request_value.is_some_and(|v| {
|
||||||
request_value.is_some_and(|v| v.get("messages").is_some_and(|m| m.is_array()));
|
v.get("messages").is_some_and(|m| m.is_array()) && input_ids_safe_to_forward(v)
|
||||||
|
});
|
||||||
if !chat_request {
|
if !chat_request {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
@@ -1456,16 +1447,15 @@ fn request_has_tools(value: &serde_json::Value) -> bool {
|
|||||||
nonempty("tools") || nonempty("functions")
|
nonempty("tools") || nonempty("functions")
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether any message carries non-string (array / multimodal) content. A text
|
/// Detect non-string or missing content, which requires engine tokenization:
|
||||||
/// tokenizer cannot represent image content, so the router's `input_ids` would
|
/// the engine normalizes arrays and nulls differently from the router's encoder.
|
||||||
/// drop it — the caller must let the engine handle these requests.
|
fn request_has_non_text_content(value: &serde_json::Value) -> bool {
|
||||||
fn request_is_multimodal(value: &serde_json::Value) -> bool {
|
|
||||||
value
|
value
|
||||||
.get("messages")
|
.get("messages")
|
||||||
.and_then(|m| m.as_array())
|
.and_then(|m| m.as_array())
|
||||||
.is_some_and(|msgs| {
|
.is_some_and(|msgs| {
|
||||||
msgs.iter()
|
msgs.iter()
|
||||||
.any(|m| matches!(m.get("content"), Some(serde_json::Value::Array(_))))
|
.any(|m| !matches!(m.get("content"), Some(serde_json::Value::String(_))))
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1770,14 +1760,25 @@ mod tests {
|
|||||||
assert!(!request_has_tools(&serde_json::json!({"messages":[]})));
|
assert!(!request_has_tools(&serde_json::json!({"messages":[]})));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Array (multimodal) message content is detected so the caller omits
|
/// Arrays, nulls, and missing content block `input_ids` forwarding.
|
||||||
/// `input_ids` (a text tokenizer can't represent image content).
|
|
||||||
#[test]
|
#[test]
|
||||||
fn request_is_multimodal_detects_array_content() {
|
fn request_has_non_text_content_detects_non_string_content() {
|
||||||
assert!(request_is_multimodal(&serde_json::json!({
|
for content in [
|
||||||
"messages":[{"role":"user","content":[{"type":"image_url","image_url":"x"}]}]
|
serde_json::json!([{"type":"image_url","image_url":"x"}]),
|
||||||
|
serde_json::json!([{"type":"text","text":"a"},{"type":"text","text":"b"}]),
|
||||||
|
serde_json::Value::Null,
|
||||||
|
] {
|
||||||
|
assert!(
|
||||||
|
request_has_non_text_content(&serde_json::json!({
|
||||||
|
"messages":[{"role":"user","content":"hi"},{"role":"assistant","content":content}]
|
||||||
|
})),
|
||||||
|
"content {content} must block"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
assert!(request_has_non_text_content(&serde_json::json!({
|
||||||
|
"messages":[{"role":"assistant","tool_calls":[]}]
|
||||||
})));
|
})));
|
||||||
assert!(!request_is_multimodal(&serde_json::json!({
|
assert!(!request_has_non_text_content(&serde_json::json!({
|
||||||
"messages":[{"role":"user","content":"hello"}]
|
"messages":[{"role":"user","content":"hello"}]
|
||||||
})));
|
})));
|
||||||
}
|
}
|
||||||
@@ -1863,9 +1864,17 @@ mod tests {
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A chat request on a chat-encoder model whose tokenization yielded NO
|
/// Excluded requests are expected fallbacks, even without rendered tokens.
|
||||||
/// tokens (encode_chat returned None → request_tokens None) IS a failure:
|
#[test]
|
||||||
/// the encoder should have fired but didn't.
|
fn offload_failed_false_for_unforwardable_request() {
|
||||||
|
let value = serde_json::json!({
|
||||||
|
"messages":[{"role":"user","content":"hi"}],
|
||||||
|
"tools":[{"type":"function","function":{"name":"f"}}]
|
||||||
|
});
|
||||||
|
assert!(!ingress_tokenize_offload_failed(true, Some(&value), None));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Missing tokens count as a failure for an eligible chat with an encoder.
|
||||||
#[test]
|
#[test]
|
||||||
fn offload_failed_true_when_chat_encoder_request_has_no_tokens() {
|
fn offload_failed_true_when_chat_encoder_request_has_no_tokens() {
|
||||||
let value = serde_json::json!({"messages":[{"role":"user","content":"hi"}]});
|
let value = serde_json::json!({"messages":[{"role":"user","content":"hi"}]});
|
||||||
|
|||||||
Reference in New Issue
Block a user