[sgl-router] Add SGLang-compatible DeepSeek V4.1 Flash rendering (#40532)

This commit is contained in:
Kan Wu
2026-09-21 18:35:09 +08:00
committed by GitHub
parent 0abb251a20
commit 0f6761b54f
9 changed files with 247 additions and 44 deletions
+2 -2
View File
@@ -918,9 +918,9 @@ dependencies = [
[[package]] [[package]]
name = "dynamo-renderer" name = "dynamo-renderer"
version = "5.1.2" version = "5.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3ae2eaa139651c535aeaad8856c5546709608931ccd4d24d3a529f6c5233c4b6" checksum = "d4609cdbd65f8bd15c74f4531c1560d04496e8737dfb6b65b45487644b878963"
dependencies = [ dependencies = [
"anyhow", "anyhow",
"chrono", "chrono",
+1 -1
View File
@@ -27,7 +27,7 @@ dynamo-tokenizers = "=1.8.1"
# Renders the model's chat template (HF Jinja, or Dynamo's built-in encoder for # Renders the model's chat template (HF Jinja, or Dynamo's built-in encoder for
# template-less models like DeepSeek-V4) so cache-aware routing hashes the same # template-less models like DeepSeek-V4) so cache-aware routing hashes the same
# tokens the engine caches. # tokens the engine caches.
dynamo-renderer = "=5.1.2" dynamo-renderer = "=5.2.0"
# `OAIChatLikeRequest` speaks minijinja values. # `OAIChatLikeRequest` speaks minijinja values.
minijinja = "2.24" minijinja = "2.24"
+6 -1
View File
@@ -204,7 +204,7 @@ settings are not inferred from the router's environment. Disabling forwarding pr
engine behavior but does not establish parity for local routing hashes. engine behavior but does not establish parity for local routing hashes.
Also set `--disable-input-ids-forwarding` for array-only templates: Dynamo may wrap Also set `--disable-input-ids-forwarding` for array-only templates: Dynamo may wrap
string content into arrays differently from the worker. Dynamo 5.1.2 does not expose string content into arrays differently from the worker. The pinned Dynamo renderer does not expose
its conversion flag, so the router cannot automatically block these templates. its conversion flag, so the router cannot automatically block these templates.
Detailed content-format parity coverage follows in #39133. Detailed content-format parity coverage follows in #39133.
@@ -221,6 +221,11 @@ official/preview effort profile is detected from the checkpoint's
`config.json`, as in SGLang. Reference prompts live in `tests/fixtures/deepseek/` `config.json`, as in SGLang. Reference prompts live in `tests/fixtures/deepseek/`
and are regenerated by `tests/scripts/generate_deepseek_parity.py`. and are regenerated by `tests/scripts/generate_deepseek_parity.py`.
V4.1 Flash uses Dynamo's separate V4.1 encoder with SGLang's numeric reasoning
budgets, tool payloads, and `<|System|>` markers. Developer messages and media
are left to the worker (the pinned encoder renders them differently), and a
non-default `SGLANG_DSV41_REASONING_EFFORT` needs the forwarding precautions above.
## Kimi-K3 ## Kimi-K3
Kimi-K3 renders through dynamo-render's native XTML formatter with SGLang's Kimi-K3 renders through dynamo-render's native XTML formatter with SGLang's
@@ -40,7 +40,7 @@ pub struct ChatFormatter {
defaults: ChatTemplateKwargs, defaults: ChatTemplateKwargs,
/// Stripped from a separately tokenized continuation prefix, as SGLang does. /// Stripped from a separately tokenized continuation prefix, as SGLang does.
bos_token: Option<String>, bos_token: Option<String>,
deepseek_v4: Option<super::deepseek::V4Profile>, deepseek: Option<super::deepseek::Encoder>,
is_kimi_k3: bool, is_kimi_k3: bool,
} }
@@ -49,7 +49,14 @@ impl ChatFormatter {
pub fn load(model_id: &str, tokenizer_path: &str) -> Result<Option<Self>> { pub fn load(model_id: &str, tokenizer_path: &str) -> Result<Option<Self>> {
let files = super::adapter::ModelFiles::open(tokenizer_path); let files = super::adapter::ModelFiles::open(tokenizer_path);
let config = files.json("config.json")?.unwrap_or_default(); let config = files.json("config.json")?.unwrap_or_default();
let model_type = config["model_type"].as_str().map(str::to_owned); let mut model_type = config["model_type"].as_str().map(str::to_owned);
// SGLang also recognizes V4.1 checkpoints that retain a V4 model_type.
if config["architectures"][0]
.as_str()
.is_some_and(|a| a.contains("DeepseekV41"))
{
model_type = Some("deepseek_v41".into());
}
if let Some(kimi) = Self::kimi_native(model_type.as_deref(), model_id) { if let Some(kimi) = Self::kimi_native(model_type.as_deref(), model_id) {
return Ok(Some(kimi)); return Ok(Some(kimi));
} }
@@ -59,8 +66,7 @@ impl ChatFormatter {
Some(t) if t.starts_with("deepseek_v4") => { Some(t) if t.starts_with("deepseek_v4") => {
let mut formatter = Self::deepseek_native(model_type.as_deref(), model_id); let mut formatter = Self::deepseek_native(model_type.as_deref(), model_id);
if let Some(formatter) = &mut formatter { if let Some(formatter) = &mut formatter {
formatter.deepseek_v4 = formatter.configure_deepseek(&files, &config)?;
Some(super::deepseek::V4Profile::load(&files, &config)?);
} }
return Ok(formatter); return Ok(formatter);
} }
@@ -73,13 +79,22 @@ impl ChatFormatter {
let mut formatter = Self::from_tokenizer_config(cfg, jinja.as_deref())? let mut formatter = Self::from_tokenizer_config(cfg, jinja.as_deref())?
.or_else(|| Self::deepseek_native(model_type.as_deref(), model_id)); .or_else(|| Self::deepseek_native(model_type.as_deref(), model_id));
if let Some(formatter) = &mut formatter { if let Some(formatter) = &mut formatter {
if formatter.deepseek_v4.is_some() { formatter.configure_deepseek(&files, &config)?;
formatter.deepseek_v4 = Some(super::deepseek::V4Profile::load(&files, &config)?);
}
} }
Ok(formatter) Ok(formatter)
} }
fn configure_deepseek(
&mut self,
files: &super::adapter::ModelFiles,
config: &JsonValue,
) -> Result<()> {
if let Some(super::deepseek::Encoder::V4(profile)) = &mut self.deepseek {
*profile = super::deepseek::V4Profile::load(files, config)?;
}
Ok(())
}
/// HF Jinja template from `tokenizer_config.json`, overridden by a sibling /// HF Jinja template from `tokenizer_config.json`, overridden by a sibling
/// `chat_template.jinja` when present (transformers' precedence); `Ok(None)` /// `chat_template.jinja` when present (transformers' precedence); `Ok(None)`
/// when the model ships neither. /// when the model ships neither.
@@ -146,7 +161,7 @@ impl ChatFormatter {
formatter, formatter,
defaults, defaults,
bos_token, bos_token,
deepseek_v4: None, deepseek: None,
is_kimi_k3: false, is_kimi_k3: false,
})) }))
} }
@@ -161,23 +176,18 @@ impl ChatFormatter {
formatter, formatter,
defaults: HashMap::new(), defaults: HashMap::new(),
bos_token: Some("[BOS]".into()), bos_token: Some("[BOS]".into()),
deepseek_v4: None, deepseek: None,
is_kimi_k3: true, is_kimi_k3: true,
}) })
} }
/// Native V4 and V3.2 formatters. Other V4-family encoders must be /// Native DeepSeek encoders; V4.1 uses its own wire format.
/// supported explicitly; a V4.1 checkpoint cannot use the V4 wire format.
pub fn deepseek_native(model_type: Option<&str>, model_id: &str) -> Option<Self> { pub fn deepseek_native(model_type: Option<&str>, model_id: &str) -> Option<Self> {
let name = model_name(model_id); let name = model_name(model_id);
let model_type = model_type.map(str::to_lowercase); let model_type = model_type.map(str::to_lowercase);
if model_type.is_none() && (name.contains("v4.1") || name.contains("v41")) { if model_type.as_deref().is_some_and(|t| {
return None; t.starts_with("deepseek_v4") && !matches!(t, "deepseek_v4" | "deepseek_v41")
} }) {
if model_type
.as_deref()
.is_some_and(|t| t.starts_with("deepseek_v4") && t != "deepseek_v4")
{
return None; return None;
} }
let PromptFormatter::OAI(formatter) = deepseek_formatter_for(&model_type, &name)?; let PromptFormatter::OAI(formatter) = deepseek_formatter_for(&model_type, &name)?;
@@ -190,6 +200,11 @@ impl ChatFormatter {
.map_or(version.split(['-', '_', '.']).next() == Some("v4"), |t| { .map_or(version.split(['-', '_', '.']).next() == Some("v4"), |t| {
t == "deepseek_v4" t == "deepseek_v4"
}); });
let is_deepseek_v41 = model_type
.as_deref()
.map_or(name.contains("v4.1") || name.contains("v41"), |t| {
t == "deepseek_v41"
});
// Engine defaults: chat mode (`SGLANG_DEFAULT_THINKING=false`) and no // Engine defaults: chat mode (`SGLANG_DEFAULT_THINKING=false`) and no
// reasoning-effort preamble; dynamo-render defaults to thinking at high effort. // reasoning-effort preamble; dynamo-render defaults to thinking at high effort.
let defaults = HashMap::from([ let defaults = HashMap::from([
@@ -200,7 +215,11 @@ impl ChatFormatter {
formatter, formatter,
defaults, defaults,
bos_token: Some("<|begin▁of▁sentence|>".into()), bos_token: Some("<|begin▁of▁sentence|>".into()),
deepseek_v4: is_deepseek_v4.then(Default::default), deepseek: if is_deepseek_v41 {
Some(super::deepseek::Encoder::V41)
} else {
is_deepseek_v4.then(|| super::deepseek::Encoder::V4(Default::default()))
},
is_kimi_k3: false, is_kimi_k3: false,
}) })
} }
@@ -280,8 +299,8 @@ impl ChatFormatter {
if self.is_kimi_k3 { if self.is_kimi_k3 {
super::kimi::normalize(request, &mut messages, &mut kwargs)?; super::kimi::normalize(request, &mut messages, &mut kwargs)?;
} }
if self.deepseek_v4.is_some() { if let Some(encoder) = self.deepseek {
super::deepseek::normalize_messages(&mut messages)?; encoder.normalize(&mut messages)?;
} }
let mut prefix = String::new(); let mut prefix = String::new();
if let Some(last) = messages.last_mut().filter(|m| m["role"] == "assistant") { if let Some(last) = messages.last_mut().filter(|m| m["role"] == "assistant") {
@@ -294,17 +313,25 @@ impl ChatFormatter {
} }
} }
} }
if self.deepseek_v4.is_some() { if self.deepseek.is_some() {
if let Some(task) = request.get("task").filter(|v| !v.is_null()).cloned() { if let Some(task) = request.get("task").filter(|v| !v.is_null()).cloned() {
// V4.1 also treats a mid-conversation system turn as the task target.
let v41 = matches!(self.deepseek, Some(super::deepseek::Encoder::V41));
let message = messages let message = messages
.iter_mut() .iter_mut()
.enumerate()
.rev() .rev()
.find(|m| matches!(m["role"].as_str(), Some("user" | "developer"))) .find(|(i, m)| match m["role"].as_str() {
Some("user" | "developer") => true,
Some("system") => v41 && *i > 0,
_ => false,
})
.map(|(_, m)| m)
.context("task requires a user or developer message")?; .context("task requires a user or developer message")?;
message["task"] = task; message["task"] = task;
} }
} }
if let Some(profile) = self.deepseek_v4 { if let Some(profile) = self.deepseek {
return Ok(( return Ok((
RenderedPrompt::text(profile.render(request, messages, &kwargs)?), RenderedPrompt::text(profile.render(request, messages, &kwargs)?),
prefix, prefix,
@@ -657,7 +684,8 @@ mod tests {
assert!(ChatFormatter::deepseek_native(None, "deepseek-ai/DeepSeek-V4-Flash").is_some()); assert!(ChatFormatter::deepseek_native(None, "deepseek-ai/DeepSeek-V4-Flash").is_some());
assert!(ChatFormatter::deepseek_native(None, "deepseek-v4-tiny").is_some()); assert!(ChatFormatter::deepseek_native(None, "deepseek-v4-tiny").is_some());
assert!(ChatFormatter::deepseek_native(Some("deepseek_v4"), "alias").is_some()); assert!(ChatFormatter::deepseek_native(Some("deepseek_v4"), "alias").is_some());
assert!(ChatFormatter::deepseek_native(Some("deepseek_v41"), "alias").is_none()); assert!(ChatFormatter::deepseek_native(Some("deepseek_v41"), "alias").is_some());
assert!(ChatFormatter::deepseek_native(None, "deepseek-ai/DeepSeek-V4.1-Flash").is_some());
assert!(ChatFormatter::deepseek_native(Some("deepseek_v32"), "DeepSeek-V3.2").is_some()); assert!(ChatFormatter::deepseek_native(Some("deepseek_v32"), "DeepSeek-V3.2").is_some());
assert!(ChatFormatter::deepseek_native(Some("inkling_mm_model"), "inkling").is_none()); assert!(ChatFormatter::deepseek_native(Some("inkling_mm_model"), "inkling").is_none());
assert!(ChatFormatter::deepseek_native(Some("llama"), "deepseek-v4").is_none()); assert!(ChatFormatter::deepseek_native(Some("llama"), "deepseek-v4").is_none());
+125 -12
View File
@@ -4,11 +4,47 @@
//! SGLang's native DeepSeek serving semantics around Dynamo's V4 encoder. //! SGLang's native DeepSeek serving semantics around Dynamo's V4 encoder.
use anyhow::{bail, ensure, Context, Result}; use anyhow::{bail, ensure, Context, Result};
use dynamo_renderer::deepseek::v4::{self, ReasoningEffort, ThinkingMode}; use dynamo_renderer::deepseek::{
v4::{self, ReasoningEffort, ThinkingMode},
v41,
};
use serde_json::{json, Map, Value}; use serde_json::{json, Map, Value};
use super::{adapter::ModelFiles, chat_formatter::ChatTemplateKwargs}; use super::{adapter::ModelFiles, chat_formatter::ChatTemplateKwargs};
#[derive(Clone, Copy)]
pub(super) enum Encoder {
V4(V4Profile),
V41,
}
impl Encoder {
pub fn normalize(self, messages: &mut [Value]) -> Result<()> {
normalize_messages(messages, matches!(self, Self::V4(_)))
}
pub fn render(
self,
request: &Value,
mut messages: Vec<Value>,
kwargs: &ChatTemplateKwargs,
) -> Result<String> {
// SGLang drops a later user's task when merging it into an existing
// user/tool-result turn. Dynamo otherwise preserves that field.
for index in 1..messages.len() {
if messages[index]["role"] == "user"
&& matches!(messages[index - 1]["role"].as_str(), Some("user" | "tool"))
{
messages[index].as_object_mut().unwrap().remove("task");
}
}
match self {
Self::V4(profile) => profile.render(request, messages, kwargs),
Self::V41 => render_v41(request, messages, kwargs),
}
}
}
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] #[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
pub(super) enum V4Profile { pub(super) enum V4Profile {
#[default] #[default]
@@ -77,15 +113,6 @@ impl V4Profile {
if let Some(tools) = request["tools"].as_array().filter(|t| !t.is_empty()) { if let Some(tools) = request["tools"].as_array().filter(|t| !t.is_empty()) {
messages[0]["tools"] = normalize_tools(tools)?.into(); messages[0]["tools"] = normalize_tools(tools)?.into();
} }
// SGLang drops a later user's task when merging it into an existing
// user/tool-result turn. Dynamo otherwise preserves that field.
for index in 1..messages.len() {
if messages[index]["role"] == "user"
&& matches!(messages[index - 1]["role"].as_str(), Some("user" | "tool"))
{
messages[index].as_object_mut().unwrap().remove("task");
}
}
let effort = match ( let effort = match (
self, self,
request_effort(request).as_ref().and_then(Value::as_str), request_effort(request).as_ref().and_then(Value::as_str),
@@ -145,9 +172,11 @@ pub(super) fn request_effort(request: &Value) -> Option<Value> {
/// Before continuation extraction, SGLang flattens V4 parts with spaces and /// Before continuation extraction, SGLang flattens V4 parts with spaces and
/// parses assistant arguments as JSON objects. Dynamo expects those arguments /// parses assistant arguments as JSON objects. Dynamo expects those arguments
/// serialized, but must not apply its permissive malformed-JSON fallback here. /// serialized, but must not apply its permissive malformed-JSON fallback here.
pub(super) fn normalize_messages(messages: &mut [Value]) -> Result<()> { fn normalize_messages(messages: &mut [Value], flatten_content: bool) -> Result<()> {
for message in messages { for message in messages {
if let Some(parts) = message["content"].as_array() { // V4.1 keeps parts lists (its encoder joins them), so a parts-list
// final assistant turn is not a continuation prefix there, as in SGLang.
if let Some(parts) = message["content"].as_array().filter(|_| flatten_content) {
message["content"] = parts message["content"] = parts
.iter() .iter()
.filter(|p| matches!(p["type"].as_str(), Some("text" | "input_text"))) .filter(|p| matches!(p["type"].as_str(), Some("text" | "input_text")))
@@ -213,6 +242,90 @@ fn normalize_tools(tools: &[Value]) -> Result<Vec<Value>> {
.collect() .collect()
} }
/// Use Dynamo's low-level encoder so SGLang, rather than Dynamo's OpenAI
/// defaults, controls tool selection and the numeric reasoning budget.
fn render_v41(
request: &Value,
mut messages: Vec<Value>,
kwargs: &ChatTemplateKwargs,
) -> Result<String> {
ensure!(
!messages.is_empty(),
"DeepSeek requires messages after continuation extraction"
);
let thinking = kwargs
.get("thinking")
.is_some_and(|v| minijinja::Value::from_serialize(v).is_true());
// Dynamo maps developer to system, unlike this SGLang encoder. Leave
// that unsupported shape to the worker instead of forwarding wrong IDs.
ensure!(
!messages.iter().any(|m| m["role"] == "developer"),
"V4.1 developer messages require engine-side rendering"
);
if let Some(tools) = request["tools"].as_array().filter(|t| !t.is_empty()) {
if messages[0]["role"] != "system" {
messages.insert(0, json!({"role":"system", "content":""}));
}
// dsv41_tool_payload: only supplied fields, in the OpenAI field order.
messages[0]["tools"] = tools
.iter()
.map(|tool| {
let f = &tool["function"];
let mut function: Map<String, Value> = [
"name",
"description",
"parameters",
"strict",
"defer_loading",
]
.into_iter()
.filter_map(|key| {
f.get(key)
.filter(|v| !v.is_null())
.map(|v| (key.into(), v.clone()))
})
.collect();
if !function.contains_key("defer_loading") {
if let Some(v) = tool.get("defer_loading").filter(|v| !v.is_null()) {
function.insert("defer_loading".into(), v.clone());
}
}
json!({"type":"function", "function":function})
})
.collect();
}
// chat_encoding.parse_dsv41_reasoning_effort; unsupported values use the
// engine default (`high`), matching SGLANG_DSV41_REASONING_EFFORT unset.
let budget = match request_effort(request) {
Some(Value::String(s)) => match s.as_str() {
"low" => Some(25),
"high" => Some(50),
"xhigh" => Some(75),
"max" => Some(100),
_ => None,
},
Some(Value::Number(n)) => match (n.as_i64(), n.as_f64()) {
(Some(i), _) => u8::try_from(i).ok().filter(|i| (1..=100).contains(i)),
(None, Some(f)) => (0.0..=0.99)
.contains(&f)
.then(|| (f * 100.0).round_ties_even().max(1.0) as u8),
_ => None,
},
_ => None,
}
.unwrap_or(50);
v41::encode_messages(
&messages,
if thinking {
ThinkingMode::Thinking
} else {
ThinkingMode::Chat
},
true,
budget,
)
}
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
+3 -2
View File
@@ -85,7 +85,7 @@ impl TokenizerRegistry {
tracing::warn!(model = %m.id, tracing::warn!(model = %m.id,
"router-generated input_ids forwarding enabled: requires matching worker model \ "router-generated input_ids forwarding enabled: requires matching worker model \
files and template defaults; native DeepSeek assumes SGLANG_DEFAULT_THINKING=false \ files and template defaults; native DeepSeek assumes SGLANG_DEFAULT_THINKING=false \
and no SGLANG_DSV4_REASONING_EFFORT preamble; worker parser overrides \ and default SGLANG_DSV4_REASONING_EFFORT / SGLANG_DSV41_REASONING_EFFORT; worker parser overrides \
(including --tool-call-parser deepseekv32), content-format detection, and \ (including --tool-call-parser deepseekv32), content-format detection, and \
conversation-template stop strings are not replicated. Use \ conversation-template stop strings are not replicated. Use \
--disable-input-ids-forwarding for array-only templates or when these assumptions do not hold"); --disable-input-ids-forwarding for array-only templates or when these assumptions do not hold");
@@ -357,7 +357,8 @@ mod tests {
.render(&request) .render(&request)
.unwrap() .unwrap()
.contains("<|open|>message")); .contains("<|open|>message"));
assert!(resolve("deepseek_v41").is_none()); assert_eq!(resolve("deepseek_v41").unwrap().render(&serde_json::json!({"messages":[{"role":"system","content":"S"},{"role":"user","content":"hi"}]})).unwrap(),
"<|begin▁of▁sentence|><|System|>S<|User|>hi<|Assistant|></think>");
assert_eq!( assert_eq!(
resolve("deepseek_v4").unwrap().render(&request).unwrap(), resolve("deepseek_v4").unwrap().render(&request).unwrap(),
"<|begin▁of▁sentence|><|User|>hi<|Assistant|></think>" "<|begin▁of▁sentence|><|User|>hi<|Assistant|></think>"
@@ -77,3 +77,22 @@ fn v4_matches_sglang_serving() {
"deepseek_v4", "deepseek_v4",
); );
} }
#[test]
fn v41_matches_sglang_serving() {
check_fixture(
include_str!("../../fixtures/deepseek/v41.json"),
"deepseek_v41",
);
}
#[test]
fn v41_leaves_unsupported_shapes_to_the_worker() {
let formatter = ChatFormatter::deepseek_native(Some("deepseek_v41"), "alias").unwrap();
for request in [
json!({"messages":[{"role":"developer","content":"rule"}]}),
json!({"messages":[{"role":"user","content":[{"type":"image_url","image_url":{"url":"https://example.com/a.png"}}]}]}),
] {
assert!(formatter.render(&request).is_err(), "{request}");
}
}
@@ -0,0 +1,27 @@
{"model": "deepseek-ai/DeepSeek-V4.1-Flash", "revision": "dba1be0a40aa45a94ad051997016db3960a90277", "cases": [
{"name":"user","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}]},"prompt":"<|begin▁of▁sentence|><|User|>Hi<|Assistant|></think>","token_count":5,"token_sha256":"95aeed952d0173ec8849b13faec66eb61ec4412cd9c41dae0c405257559365a6"},
{"name":"system","profile":"v41","request":{"model":"m","messages":[{"role":"system","content":"Be terse"},{"role":"user","content":"Hi"}]},"prompt":"<|begin▁of▁sentence|><|System|>Be terse<|User|>Hi<|Assistant|></think>","token_count":9,"token_sha256":"01befc17a5a1030ed5de3c3d7924514de09fc4fb9724e396586458bdcc396c1d"},
{"name":"multi_turn","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":"Answer","reasoning_content":"Prior reasoning"},{"role":"user","content":"Again"}]},"prompt":"<|begin▁of▁sentence|><|User|>Hi<|Assistant|></think>Answer<|end▁of▁sentence|><|User|>Again<|Assistant|></think>","token_count":11,"token_sha256":"5ab68a07b5a4b1fa1811d68fe156ece6b5284af15cac9b76933c28d9909e4ee2"},
{"name":"parts","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":[{"type":"text","text":"Hello"},{"type":"text","text":"world"}]}]},"prompt":"<|begin▁of▁sentence|><|User|>Hello\n\nworld<|Assistant|></think>","token_count":7,"token_sha256":"1e631e7763d21541ad375b0ff431453be6515189b7588b30a285dec4da726ae7"},
{"name":"continuation_parts","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":[{"type":"text","text":"Hello"},{"type":"text","text":"world"}]}],"continue_final_message":true},"prompt":"<|begin▁of▁sentence|><|User|>Hi<|Assistant|></think>Hello\n\nworld<|end▁of▁sentence|>","token_count":9,"token_sha256":"bbbd4e92e321aa42394288a7575735657c094ab48ea6aea8f609b37f84d70bd5"},
{"name":"continuation_bos","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":"<|begin▁of▁sentence|>abc"}],"continue_final_message":true},"prompt":"<|begin▁of▁sentence|><|User|>Hi<|Assistant|></think><|begin▁of▁sentence|>abc","token_count":6,"token_sha256":"defef25e49dca2301b583fb67cabe122b8aa1bdd889727ea1e8b083be7149c14"},
{"name":"tools","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}]},"prompt":"<|begin▁of▁sentence|><|System|>\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<|DSML| calls>\" block like the following:\n\n<|DSML| calls>\n<|DSML| invoke name=\"$TOOL_NAME\">\n<|DSML| parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</|DSML| parameter>\n...\n</|DSML| invoke>\n<|DSML| invoke name=\"$TOOL_NAME2\">\n...\n</|DSML| invoke>\n</|DSML| calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}}\n{\"name\": \"bar\", \"description\": \"第二个\", \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<|User|>Hi<|Assistant|></think>","token_count":256,"token_sha256":"89eaf9e9642f765c5f244da76ce14d680a6c22bf97f6a859d1946d528649d458"},
{"name":"tools_none","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}],"tool_choice":"none"},"prompt":"<|begin▁of▁sentence|><|System|>\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<|DSML| calls>\" block like the following:\n\n<|DSML| calls>\n<|DSML| invoke name=\"$TOOL_NAME\">\n<|DSML| parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</|DSML| parameter>\n...\n</|DSML| invoke>\n<|DSML| invoke name=\"$TOOL_NAME2\">\n...\n</|DSML| invoke>\n</|DSML| calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}}\n{\"name\": \"bar\", \"description\": \"第二个\", \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<|User|>Hi<|Assistant|></think>","token_count":256,"token_sha256":"89eaf9e9642f765c5f244da76ce14d680a6c22bf97f6a859d1946d528649d458"},
{"name":"tools_named","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}],"tool_choice":{"type":"function","function":{"name":"bar"}}},"prompt":"<|begin▁of▁sentence|><|System|>\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<|DSML| calls>\" block like the following:\n\n<|DSML| calls>\n<|DSML| invoke name=\"$TOOL_NAME\">\n<|DSML| parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</|DSML| parameter>\n...\n</|DSML| invoke>\n<|DSML| invoke name=\"$TOOL_NAME2\">\n...\n</|DSML| invoke>\n</|DSML| calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}}\n{\"name\": \"bar\", \"description\": \"第二个\", \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<|User|>Hi<|Assistant|></think>","token_count":256,"token_sha256":"89eaf9e9642f765c5f244da76ce14d680a6c22bf97f6a859d1946d528649d458"},
{"name":"message_tools","profile":"v41","request":{"model":"m","messages":[{"role":"system","content":"SYS","tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}]},{"role":"user","content":"Hi"}]},"prompt":"<|begin▁of▁sentence|><|System|>SYS\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<|DSML| calls>\" block like the following:\n\n<|DSML| calls>\n<|DSML| invoke name=\"$TOOL_NAME\">\n<|DSML| parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</|DSML| parameter>\n...\n</|DSML| invoke>\n<|DSML| invoke name=\"$TOOL_NAME2\">\n...\n</|DSML| invoke>\n</|DSML| calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"description\": null, \"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}, \"strict\": false}\n{\"description\": \"第二个\", \"name\": \"bar\", \"parameters\": null, \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<|User|>Hi<|Assistant|></think>","token_count":272,"token_sha256":"d6ebecca3f9e9a498e18f1a5140ff8d2e833951d64329ff970ff8ef6157f0a0c"},
{"name":"empty_message_tools","profile":"v41","request":{"model":"m","messages":[{"role":"system","content":"SYS","tools":[]},{"role":"user","content":"Hi"}]},"prompt":"<|begin▁of▁sentence|><|System|>SYS<|User|>Hi<|Assistant|></think>","token_count":8,"token_sha256":"0c718f5b123ecfacb6fc0d0091dfc63b1b7959c9122b2365fe823b270b6c3d35"},
{"name":"tool_results","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":null,"tool_calls":[{"id":"a","type":"function","function":{"name":"foo","arguments":"{\"x\":\"雪\", \"y\":[1, true]}"}},{"id":"b","type":"function","function":{"name":"bar","arguments":{"z":2}}}],"reasoning_content":"Call both"},{"role":"tool","content":[{"type":"text","text":"Hello"},{"type":"text","text":"world"}],"tool_call_id":"b"},{"role":"tool","content":null,"tool_call_id":"a"}],"tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}],"chat_template_kwargs":{"thinking":true}},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 50 (range 1-100, the higher the value, the more thorough the reasoning)\n\n\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<|DSML| calls>\" block like the following:\n\n<|DSML| calls>\n<|DSML| invoke name=\"$TOOL_NAME\">\n<|DSML| parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</|DSML| parameter>\n...\n</|DSML| invoke>\n<|DSML| invoke name=\"$TOOL_NAME2\">\n...\n</|DSML| invoke>\n</|DSML| calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}}\n{\"name\": \"bar\", \"description\": \"第二个\", \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<|User|>Hi<|Assistant|><think>Call both</think>\n\n<|DSML| calls>\n<|DSML| invoke name=\"foo\">\n<|DSML| parameter name=\"x\" string=\"true\">雪</|DSML| parameter>\n<|DSML| parameter name=\"y\" string=\"false\">[1, true]</|DSML| parameter>\n</|DSML| invoke>\n<|DSML| invoke name=\"bar\">\n<|DSML| parameter name=\"z\" string=\"false\">2</|DSML| parameter>\n</|DSML| invoke>\n</|DSML| calls><|end▁of▁sentence|><|User|><tool_result></tool_result>\n\n<tool_result>Hello\n\nworld</tool_result><|Assistant|><think>","token_count":388,"token_sha256":"f9fb80a884084408d01041c452a9ece38a9c3a959a5fd3ed0289db3f6611c203"},
{"name":"kwargs_effort","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"chat_template_kwargs":{"thinking":true,"reasoning_effort":"max"}},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 100 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|><think>","token_count":31,"token_sha256":"a29f18c6e65c09656d098c43e548224b67ced6a674f65186b7e132fb6d997d75"},
{"name":"effort_conflict","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"high","chat_template_kwargs":{"reasoning_effort":"low"}},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 25 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|><think>","token_count":31,"token_sha256":"80aa41f1cbf4d847db08af7c5f4dde508ea75aed6e2cb6b94f3a0a806e275d46"},
{"name":"drop_thinking_ignored","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":"Answer","reasoning_content":"Prior reasoning"},{"role":"user","content":"Again"}],"chat_template_kwargs":{"thinking":true,"drop_thinking":false}},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 50 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|></think>Answer<|end▁of▁sentence|><|User|>Again<|Assistant|><think>","token_count":37,"token_sha256":"f6cb0ab0ad4a220297622920684237a208098dd2b2cd368599bc49ebba7994f9"},
{"name":"thinking_false","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"high","chat_template_kwargs":{"thinking":false}},"prompt":"<|begin▁of▁sentence|><|User|>Hi<|Assistant|></think>","token_count":5,"token_sha256":"95aeed952d0173ec8849b13faec66eb61ec4412cd9c41dae0c405257559365a6"},
{"name":"thinking_none","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"none","chat_template_kwargs":{"thinking":true}},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 50 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|><think>","token_count":31,"token_sha256":"4add8297343cd581f7ded6ecba6c06c5f70f402802db9a021d469efdc2138a15"},
{"name":"consecutive_task","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"user","content":"Again"}],"task":"domain"},"prompt":"<|begin▁of▁sentence|><|User|>Hi\n\nAgain<|Assistant|></think>","token_count":7,"token_sha256":"d05e248d8a06f2c06df20b7a71af0d4dfae1cd91ce9f614719b5842674042d6b"},
{"name":"effort_high","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"high"},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 50 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|><think>","token_count":31,"token_sha256":"4add8297343cd581f7ded6ecba6c06c5f70f402802db9a021d469efdc2138a15"},
{"name":"effort_max","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"max"},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 100 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|><think>","token_count":31,"token_sha256":"a29f18c6e65c09656d098c43e548224b67ced6a674f65186b7e132fb6d997d75"},
{"name":"task_action","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"task":"action"},"prompt":"<|begin▁of▁sentence|><|User|>Hi<|Assistant|></think><|action|>","token_count":6,"token_sha256":"925bcaae248cc41efabd3e5d8f660f17513e43174f1d34aba51a1399aba76e8e"},
{"name":"effort_low","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"low"},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 25 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|><think>","token_count":31,"token_sha256":"80aa41f1cbf4d847db08af7c5f4dde508ea75aed6e2cb6b94f3a0a806e275d46"},
{"name":"effort_xhigh","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"xhigh"},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 75 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|><think>","token_count":31,"token_sha256":"54a0e744ad23d9a75775f4a71851980f51a778a1011d73de2f370b1d5499a5b7"},
{"name":"effort_0.75","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":0.75},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 75 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|><think>","token_count":31,"token_sha256":"54a0e744ad23d9a75775f4a71851980f51a778a1011d73de2f370b1d5499a5b7"},
{"name":"kwargs_budget","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"chat_template_kwargs":{"thinking":true,"reasoning_effort":80}},"prompt":"<|begin▁of▁sentence|><|System|>Reasoning Effort: 80 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<|User|>Hi<|Assistant|><think>","token_count":31,"token_sha256":"32a4462bbf34bce35dd5c163574188ebd71896ac46d58966e0a6c2481b1b4081"}
]}
@@ -25,6 +25,10 @@ from sglang.srt.entrypoints.openai.serving_chat import OpenAIServingChat
ROOT = Path(__file__).resolve().parents[1] / "fixtures/deepseek" ROOT = Path(__file__).resolve().parents[1] / "fixtures/deepseek"
MODELS = { MODELS = {
"v41": (
"deepseek-ai/DeepSeek-V4.1-Flash",
"dba1be0a40aa45a94ad051997016db3960a90277",
),
"v4": ("deepseek-ai/DeepSeek-V4-Flash", "60d8d70770c6776ff598c94bb586a859a38244f1"), "v4": ("deepseek-ai/DeepSeek-V4-Flash", "60d8d70770c6776ff598c94bb586a859a38244f1"),
} }
@@ -45,12 +49,18 @@ class RecordingTokenizer:
def main(): def main():
ROOT.mkdir(exist_ok=True) ROOT.mkdir(exist_ok=True)
for family, (model, revision) in MODELS.items(): for family, (model, revision) in MODELS.items():
path = snapshot_download(model, revision=revision, local_files_only=True) path = snapshot_download(
model,
revision=revision,
local_files_only=True,
allow_patterns=["config.json", "tokenizer*.json"],
)
tok = RecordingTokenizer( tok = RecordingTokenizer(
AutoTokenizer.from_pretrained(path, local_files_only=True) AutoTokenizer.from_pretrained(path, local_files_only=True)
) )
server = object.__new__(OpenAIServingChat) server = object.__new__(OpenAIServingChat)
server.chat_encoding_spec = "ds" + family server.chat_encoding_spec = "ds" + family
server._dsv41_default_reasoning_effort = "high"
server.tokenizer_manager = SimpleNamespace(tokenizer=tok) server.tokenizer_manager = SimpleNamespace(tokenizer=tok)
server.template_manager = SimpleNamespace( server.template_manager = SimpleNamespace(
jinja_template_content_format="string" jinja_template_content_format="string"