[sgl-router] Add SGLang-compatible DeepSeek V4.1 Flash rendering (#40532)

This commit is contained in:
Kan Wu
2026-09-21 18:35:09 +08:00
committed by GitHub
parent 0abb251a20
commit 0f6761b54f
9 changed files with 247 additions and 44 deletions
+2 -2
View File
@@ -918,9 +918,9 @@ dependencies = [
[[package]]
name = "dynamo-renderer"
version = "5.1.2"
version = "5.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3ae2eaa139651c535aeaad8856c5546709608931ccd4d24d3a529f6c5233c4b6"
checksum = "d4609cdbd65f8bd15c74f4531c1560d04496e8737dfb6b65b45487644b878963"
dependencies = [
"anyhow",
"chrono",
+1 -1
View File
@@ -27,7 +27,7 @@ dynamo-tokenizers = "=1.8.1"
# Renders the model's chat template (HF Jinja, or Dynamo's built-in encoder for
# template-less models like DeepSeek-V4) so cache-aware routing hashes the same
# tokens the engine caches.
dynamo-renderer = "=5.1.2"
dynamo-renderer = "=5.2.0"
# `OAIChatLikeRequest` speaks minijinja values.
minijinja = "2.24"
+6 -1
View File
@@ -204,7 +204,7 @@ settings are not inferred from the router's environment. Disabling forwarding pr
engine behavior but does not establish parity for local routing hashes.
Also set `--disable-input-ids-forwarding` for array-only templates: Dynamo may wrap
string content into arrays differently from the worker. Dynamo 5.1.2 does not expose
string content into arrays differently from the worker. The pinned Dynamo renderer does not expose
its conversion flag, so the router cannot automatically block these templates.
Detailed content-format parity coverage follows in #39133.
@@ -221,6 +221,11 @@ official/preview effort profile is detected from the checkpoint's
`config.json`, as in SGLang. Reference prompts live in `tests/fixtures/deepseek/`
and are regenerated by `tests/scripts/generate_deepseek_parity.py`.
V4.1 Flash uses Dynamo's separate V4.1 encoder with SGLang's numeric reasoning
budgets, tool payloads, and `<System>` markers. Developer messages and media
are left to the worker (the pinned encoder renders them differently), and a
non-default `SGLANG_DSV41_REASONING_EFFORT` needs the forwarding precautions above.
## Kimi-K3
Kimi-K3 renders through dynamo-render's native XTML formatter with SGLang's
@@ -40,7 +40,7 @@ pub struct ChatFormatter {
defaults: ChatTemplateKwargs,
/// Stripped from a separately tokenized continuation prefix, as SGLang does.
bos_token: Option<String>,
deepseek_v4: Option<super::deepseek::V4Profile>,
deepseek: Option<super::deepseek::Encoder>,
is_kimi_k3: bool,
}
@@ -49,7 +49,14 @@ impl ChatFormatter {
pub fn load(model_id: &str, tokenizer_path: &str) -> Result<Option<Self>> {
let files = super::adapter::ModelFiles::open(tokenizer_path);
let config = files.json("config.json")?.unwrap_or_default();
let model_type = config["model_type"].as_str().map(str::to_owned);
let mut model_type = config["model_type"].as_str().map(str::to_owned);
// SGLang also recognizes V4.1 checkpoints that retain a V4 model_type.
if config["architectures"][0]
.as_str()
.is_some_and(|a| a.contains("DeepseekV41"))
{
model_type = Some("deepseek_v41".into());
}
if let Some(kimi) = Self::kimi_native(model_type.as_deref(), model_id) {
return Ok(Some(kimi));
}
@@ -59,8 +66,7 @@ impl ChatFormatter {
Some(t) if t.starts_with("deepseek_v4") => {
let mut formatter = Self::deepseek_native(model_type.as_deref(), model_id);
if let Some(formatter) = &mut formatter {
formatter.deepseek_v4 =
Some(super::deepseek::V4Profile::load(&files, &config)?);
formatter.configure_deepseek(&files, &config)?;
}
return Ok(formatter);
}
@@ -73,13 +79,22 @@ impl ChatFormatter {
let mut formatter = Self::from_tokenizer_config(cfg, jinja.as_deref())?
.or_else(|| Self::deepseek_native(model_type.as_deref(), model_id));
if let Some(formatter) = &mut formatter {
if formatter.deepseek_v4.is_some() {
formatter.deepseek_v4 = Some(super::deepseek::V4Profile::load(&files, &config)?);
}
formatter.configure_deepseek(&files, &config)?;
}
Ok(formatter)
}
fn configure_deepseek(
&mut self,
files: &super::adapter::ModelFiles,
config: &JsonValue,
) -> Result<()> {
if let Some(super::deepseek::Encoder::V4(profile)) = &mut self.deepseek {
*profile = super::deepseek::V4Profile::load(files, config)?;
}
Ok(())
}
/// HF Jinja template from `tokenizer_config.json`, overridden by a sibling
/// `chat_template.jinja` when present (transformers' precedence); `Ok(None)`
/// when the model ships neither.
@@ -146,7 +161,7 @@ impl ChatFormatter {
formatter,
defaults,
bos_token,
deepseek_v4: None,
deepseek: None,
is_kimi_k3: false,
}))
}
@@ -161,23 +176,18 @@ impl ChatFormatter {
formatter,
defaults: HashMap::new(),
bos_token: Some("[BOS]".into()),
deepseek_v4: None,
deepseek: None,
is_kimi_k3: true,
})
}
/// Native V4 and V3.2 formatters. Other V4-family encoders must be
/// supported explicitly; a V4.1 checkpoint cannot use the V4 wire format.
/// Native DeepSeek encoders; V4.1 uses its own wire format.
pub fn deepseek_native(model_type: Option<&str>, model_id: &str) -> Option<Self> {
let name = model_name(model_id);
let model_type = model_type.map(str::to_lowercase);
if model_type.is_none() && (name.contains("v4.1") || name.contains("v41")) {
return None;
}
if model_type
.as_deref()
.is_some_and(|t| t.starts_with("deepseek_v4") && t != "deepseek_v4")
{
if model_type.as_deref().is_some_and(|t| {
t.starts_with("deepseek_v4") && !matches!(t, "deepseek_v4" | "deepseek_v41")
}) {
return None;
}
let PromptFormatter::OAI(formatter) = deepseek_formatter_for(&model_type, &name)?;
@@ -190,6 +200,11 @@ impl ChatFormatter {
.map_or(version.split(['-', '_', '.']).next() == Some("v4"), |t| {
t == "deepseek_v4"
});
let is_deepseek_v41 = model_type
.as_deref()
.map_or(name.contains("v4.1") || name.contains("v41"), |t| {
t == "deepseek_v41"
});
// Engine defaults: chat mode (`SGLANG_DEFAULT_THINKING=false`) and no
// reasoning-effort preamble; dynamo-render defaults to thinking at high effort.
let defaults = HashMap::from([
@@ -200,7 +215,11 @@ impl ChatFormatter {
formatter,
defaults,
bos_token: Some("<begin▁of▁sentence>".into()),
deepseek_v4: is_deepseek_v4.then(Default::default),
deepseek: if is_deepseek_v41 {
Some(super::deepseek::Encoder::V41)
} else {
is_deepseek_v4.then(|| super::deepseek::Encoder::V4(Default::default()))
},
is_kimi_k3: false,
})
}
@@ -280,8 +299,8 @@ impl ChatFormatter {
if self.is_kimi_k3 {
super::kimi::normalize(request, &mut messages, &mut kwargs)?;
}
if self.deepseek_v4.is_some() {
super::deepseek::normalize_messages(&mut messages)?;
if let Some(encoder) = self.deepseek {
encoder.normalize(&mut messages)?;
}
let mut prefix = String::new();
if let Some(last) = messages.last_mut().filter(|m| m["role"] == "assistant") {
@@ -294,17 +313,25 @@ impl ChatFormatter {
}
}
}
if self.deepseek_v4.is_some() {
if self.deepseek.is_some() {
if let Some(task) = request.get("task").filter(|v| !v.is_null()).cloned() {
// V4.1 also treats a mid-conversation system turn as the task target.
let v41 = matches!(self.deepseek, Some(super::deepseek::Encoder::V41));
let message = messages
.iter_mut()
.enumerate()
.rev()
.find(|m| matches!(m["role"].as_str(), Some("user" | "developer")))
.find(|(i, m)| match m["role"].as_str() {
Some("user" | "developer") => true,
Some("system") => v41 && *i > 0,
_ => false,
})
.map(|(_, m)| m)
.context("task requires a user or developer message")?;
message["task"] = task;
}
}
if let Some(profile) = self.deepseek_v4 {
if let Some(profile) = self.deepseek {
return Ok((
RenderedPrompt::text(profile.render(request, messages, &kwargs)?),
prefix,
@@ -657,7 +684,8 @@ mod tests {
assert!(ChatFormatter::deepseek_native(None, "deepseek-ai/DeepSeek-V4-Flash").is_some());
assert!(ChatFormatter::deepseek_native(None, "deepseek-v4-tiny").is_some());
assert!(ChatFormatter::deepseek_native(Some("deepseek_v4"), "alias").is_some());
assert!(ChatFormatter::deepseek_native(Some("deepseek_v41"), "alias").is_none());
assert!(ChatFormatter::deepseek_native(Some("deepseek_v41"), "alias").is_some());
assert!(ChatFormatter::deepseek_native(None, "deepseek-ai/DeepSeek-V4.1-Flash").is_some());
assert!(ChatFormatter::deepseek_native(Some("deepseek_v32"), "DeepSeek-V3.2").is_some());
assert!(ChatFormatter::deepseek_native(Some("inkling_mm_model"), "inkling").is_none());
assert!(ChatFormatter::deepseek_native(Some("llama"), "deepseek-v4").is_none());
+125 -12
View File
@@ -4,11 +4,47 @@
//! SGLang's native DeepSeek serving semantics around Dynamo's V4 encoder.
use anyhow::{bail, ensure, Context, Result};
use dynamo_renderer::deepseek::v4::{self, ReasoningEffort, ThinkingMode};
use dynamo_renderer::deepseek::{
v4::{self, ReasoningEffort, ThinkingMode},
v41,
};
use serde_json::{json, Map, Value};
use super::{adapter::ModelFiles, chat_formatter::ChatTemplateKwargs};
#[derive(Clone, Copy)]
pub(super) enum Encoder {
V4(V4Profile),
V41,
}
impl Encoder {
pub fn normalize(self, messages: &mut [Value]) -> Result<()> {
normalize_messages(messages, matches!(self, Self::V4(_)))
}
pub fn render(
self,
request: &Value,
mut messages: Vec<Value>,
kwargs: &ChatTemplateKwargs,
) -> Result<String> {
// SGLang drops a later user's task when merging it into an existing
// user/tool-result turn. Dynamo otherwise preserves that field.
for index in 1..messages.len() {
if messages[index]["role"] == "user"
&& matches!(messages[index - 1]["role"].as_str(), Some("user" | "tool"))
{
messages[index].as_object_mut().unwrap().remove("task");
}
}
match self {
Self::V4(profile) => profile.render(request, messages, kwargs),
Self::V41 => render_v41(request, messages, kwargs),
}
}
}
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
pub(super) enum V4Profile {
#[default]
@@ -77,15 +113,6 @@ impl V4Profile {
if let Some(tools) = request["tools"].as_array().filter(|t| !t.is_empty()) {
messages[0]["tools"] = normalize_tools(tools)?.into();
}
// SGLang drops a later user's task when merging it into an existing
// user/tool-result turn. Dynamo otherwise preserves that field.
for index in 1..messages.len() {
if messages[index]["role"] == "user"
&& matches!(messages[index - 1]["role"].as_str(), Some("user" | "tool"))
{
messages[index].as_object_mut().unwrap().remove("task");
}
}
let effort = match (
self,
request_effort(request).as_ref().and_then(Value::as_str),
@@ -145,9 +172,11 @@ pub(super) fn request_effort(request: &Value) -> Option<Value> {
/// Before continuation extraction, SGLang flattens V4 parts with spaces and
/// parses assistant arguments as JSON objects. Dynamo expects those arguments
/// serialized, but must not apply its permissive malformed-JSON fallback here.
pub(super) fn normalize_messages(messages: &mut [Value]) -> Result<()> {
fn normalize_messages(messages: &mut [Value], flatten_content: bool) -> Result<()> {
for message in messages {
if let Some(parts) = message["content"].as_array() {
// V4.1 keeps parts lists (its encoder joins them), so a parts-list
// final assistant turn is not a continuation prefix there, as in SGLang.
if let Some(parts) = message["content"].as_array().filter(|_| flatten_content) {
message["content"] = parts
.iter()
.filter(|p| matches!(p["type"].as_str(), Some("text" | "input_text")))
@@ -213,6 +242,90 @@ fn normalize_tools(tools: &[Value]) -> Result<Vec<Value>> {
.collect()
}
/// Use Dynamo's low-level encoder so SGLang, rather than Dynamo's OpenAI
/// defaults, controls tool selection and the numeric reasoning budget.
fn render_v41(
request: &Value,
mut messages: Vec<Value>,
kwargs: &ChatTemplateKwargs,
) -> Result<String> {
ensure!(
!messages.is_empty(),
"DeepSeek requires messages after continuation extraction"
);
let thinking = kwargs
.get("thinking")
.is_some_and(|v| minijinja::Value::from_serialize(v).is_true());
// Dynamo maps developer to system, unlike this SGLang encoder. Leave
// that unsupported shape to the worker instead of forwarding wrong IDs.
ensure!(
!messages.iter().any(|m| m["role"] == "developer"),
"V4.1 developer messages require engine-side rendering"
);
if let Some(tools) = request["tools"].as_array().filter(|t| !t.is_empty()) {
if messages[0]["role"] != "system" {
messages.insert(0, json!({"role":"system", "content":""}));
}
// dsv41_tool_payload: only supplied fields, in the OpenAI field order.
messages[0]["tools"] = tools
.iter()
.map(|tool| {
let f = &tool["function"];
let mut function: Map<String, Value> = [
"name",
"description",
"parameters",
"strict",
"defer_loading",
]
.into_iter()
.filter_map(|key| {
f.get(key)
.filter(|v| !v.is_null())
.map(|v| (key.into(), v.clone()))
})
.collect();
if !function.contains_key("defer_loading") {
if let Some(v) = tool.get("defer_loading").filter(|v| !v.is_null()) {
function.insert("defer_loading".into(), v.clone());
}
}
json!({"type":"function", "function":function})
})
.collect();
}
// chat_encoding.parse_dsv41_reasoning_effort; unsupported values use the
// engine default (`high`), matching SGLANG_DSV41_REASONING_EFFORT unset.
let budget = match request_effort(request) {
Some(Value::String(s)) => match s.as_str() {
"low" => Some(25),
"high" => Some(50),
"xhigh" => Some(75),
"max" => Some(100),
_ => None,
},
Some(Value::Number(n)) => match (n.as_i64(), n.as_f64()) {
(Some(i), _) => u8::try_from(i).ok().filter(|i| (1..=100).contains(i)),
(None, Some(f)) => (0.0..=0.99)
.contains(&f)
.then(|| (f * 100.0).round_ties_even().max(1.0) as u8),
_ => None,
},
_ => None,
}
.unwrap_or(50);
v41::encode_messages(
&messages,
if thinking {
ThinkingMode::Thinking
} else {
ThinkingMode::Chat
},
true,
budget,
)
}
#[cfg(test)]
mod tests {
use super::*;
+3 -2
View File
@@ -85,7 +85,7 @@ impl TokenizerRegistry {
tracing::warn!(model = %m.id,
"router-generated input_ids forwarding enabled: requires matching worker model \
files and template defaults; native DeepSeek assumes SGLANG_DEFAULT_THINKING=false \
and no SGLANG_DSV4_REASONING_EFFORT preamble; worker parser overrides \
and default SGLANG_DSV4_REASONING_EFFORT / SGLANG_DSV41_REASONING_EFFORT; worker parser overrides \
(including --tool-call-parser deepseekv32), content-format detection, and \
conversation-template stop strings are not replicated. Use \
--disable-input-ids-forwarding for array-only templates or when these assumptions do not hold");
@@ -357,7 +357,8 @@ mod tests {
.render(&request)
.unwrap()
.contains("<|open|>message"));
assert!(resolve("deepseek_v41").is_none());
assert_eq!(resolve("deepseek_v41").unwrap().render(&serde_json::json!({"messages":[{"role":"system","content":"S"},{"role":"user","content":"hi"}]})).unwrap(),
"<begin▁of▁sentence><System>S<User>hi<Assistant></think>");
assert_eq!(
resolve("deepseek_v4").unwrap().render(&request).unwrap(),
"<begin▁of▁sentence><User>hi<Assistant></think>"
@@ -77,3 +77,22 @@ fn v4_matches_sglang_serving() {
"deepseek_v4",
);
}
#[test]
fn v41_matches_sglang_serving() {
check_fixture(
include_str!("../../fixtures/deepseek/v41.json"),
"deepseek_v41",
);
}
#[test]
fn v41_leaves_unsupported_shapes_to_the_worker() {
let formatter = ChatFormatter::deepseek_native(Some("deepseek_v41"), "alias").unwrap();
for request in [
json!({"messages":[{"role":"developer","content":"rule"}]}),
json!({"messages":[{"role":"user","content":[{"type":"image_url","image_url":{"url":"https://example.com/a.png"}}]}]}),
] {
assert!(formatter.render(&request).is_err(), "{request}");
}
}
@@ -0,0 +1,27 @@
{"model": "deepseek-ai/DeepSeek-V4.1-Flash", "revision": "dba1be0a40aa45a94ad051997016db3960a90277", "cases": [
{"name":"user","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}]},"prompt":"<begin▁of▁sentence><User>Hi<Assistant></think>","token_count":5,"token_sha256":"95aeed952d0173ec8849b13faec66eb61ec4412cd9c41dae0c405257559365a6"},
{"name":"system","profile":"v41","request":{"model":"m","messages":[{"role":"system","content":"Be terse"},{"role":"user","content":"Hi"}]},"prompt":"<begin▁of▁sentence><System>Be terse<User>Hi<Assistant></think>","token_count":9,"token_sha256":"01befc17a5a1030ed5de3c3d7924514de09fc4fb9724e396586458bdcc396c1d"},
{"name":"multi_turn","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":"Answer","reasoning_content":"Prior reasoning"},{"role":"user","content":"Again"}]},"prompt":"<begin▁of▁sentence><User>Hi<Assistant></think>Answer<end▁of▁sentence><User>Again<Assistant></think>","token_count":11,"token_sha256":"5ab68a07b5a4b1fa1811d68fe156ece6b5284af15cac9b76933c28d9909e4ee2"},
{"name":"parts","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":[{"type":"text","text":"Hello"},{"type":"text","text":"world"}]}]},"prompt":"<begin▁of▁sentence><User>Hello\n\nworld<Assistant></think>","token_count":7,"token_sha256":"1e631e7763d21541ad375b0ff431453be6515189b7588b30a285dec4da726ae7"},
{"name":"continuation_parts","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":[{"type":"text","text":"Hello"},{"type":"text","text":"world"}]}],"continue_final_message":true},"prompt":"<begin▁of▁sentence><User>Hi<Assistant></think>Hello\n\nworld<end▁of▁sentence>","token_count":9,"token_sha256":"bbbd4e92e321aa42394288a7575735657c094ab48ea6aea8f609b37f84d70bd5"},
{"name":"continuation_bos","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":"<begin▁of▁sentence>abc"}],"continue_final_message":true},"prompt":"<begin▁of▁sentence><User>Hi<Assistant></think><begin▁of▁sentence>abc","token_count":6,"token_sha256":"defef25e49dca2301b583fb67cabe122b8aa1bdd889727ea1e8b083be7149c14"},
{"name":"tools","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}]},"prompt":"<begin▁of▁sentence><System>\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<DSML calls>\" block like the following:\n\n<DSML calls>\n<DSML invoke name=\"$TOOL_NAME\">\n<DSML parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</DSML parameter>\n...\n</DSML invoke>\n<DSML invoke name=\"$TOOL_NAME2\">\n...\n</DSML invoke>\n</DSML calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}}\n{\"name\": \"bar\", \"description\": \"第二个\", \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<User>Hi<Assistant></think>","token_count":256,"token_sha256":"89eaf9e9642f765c5f244da76ce14d680a6c22bf97f6a859d1946d528649d458"},
{"name":"tools_none","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}],"tool_choice":"none"},"prompt":"<begin▁of▁sentence><System>\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<DSML calls>\" block like the following:\n\n<DSML calls>\n<DSML invoke name=\"$TOOL_NAME\">\n<DSML parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</DSML parameter>\n...\n</DSML invoke>\n<DSML invoke name=\"$TOOL_NAME2\">\n...\n</DSML invoke>\n</DSML calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}}\n{\"name\": \"bar\", \"description\": \"第二个\", \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<User>Hi<Assistant></think>","token_count":256,"token_sha256":"89eaf9e9642f765c5f244da76ce14d680a6c22bf97f6a859d1946d528649d458"},
{"name":"tools_named","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}],"tool_choice":{"type":"function","function":{"name":"bar"}}},"prompt":"<begin▁of▁sentence><System>\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<DSML calls>\" block like the following:\n\n<DSML calls>\n<DSML invoke name=\"$TOOL_NAME\">\n<DSML parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</DSML parameter>\n...\n</DSML invoke>\n<DSML invoke name=\"$TOOL_NAME2\">\n...\n</DSML invoke>\n</DSML calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}}\n{\"name\": \"bar\", \"description\": \"第二个\", \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<User>Hi<Assistant></think>","token_count":256,"token_sha256":"89eaf9e9642f765c5f244da76ce14d680a6c22bf97f6a859d1946d528649d458"},
{"name":"message_tools","profile":"v41","request":{"model":"m","messages":[{"role":"system","content":"SYS","tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}]},{"role":"user","content":"Hi"}]},"prompt":"<begin▁of▁sentence><System>SYS\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<DSML calls>\" block like the following:\n\n<DSML calls>\n<DSML invoke name=\"$TOOL_NAME\">\n<DSML parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</DSML parameter>\n...\n</DSML invoke>\n<DSML invoke name=\"$TOOL_NAME2\">\n...\n</DSML invoke>\n</DSML calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"description\": null, \"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}, \"strict\": false}\n{\"description\": \"第二个\", \"name\": \"bar\", \"parameters\": null, \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<User>Hi<Assistant></think>","token_count":272,"token_sha256":"d6ebecca3f9e9a498e18f1a5140ff8d2e833951d64329ff970ff8ef6157f0a0c"},
{"name":"empty_message_tools","profile":"v41","request":{"model":"m","messages":[{"role":"system","content":"SYS","tools":[]},{"role":"user","content":"Hi"}]},"prompt":"<begin▁of▁sentence><System>SYS<User>Hi<Assistant></think>","token_count":8,"token_sha256":"0c718f5b123ecfacb6fc0d0091dfc63b1b7959c9122b2365fe823b270b6c3d35"},
{"name":"tool_results","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":null,"tool_calls":[{"id":"a","type":"function","function":{"name":"foo","arguments":"{\"x\":\"雪\", \"y\":[1, true]}"}},{"id":"b","type":"function","function":{"name":"bar","arguments":{"z":2}}}],"reasoning_content":"Call both"},{"role":"tool","content":[{"type":"text","text":"Hello"},{"type":"text","text":"world"}],"tool_call_id":"b"},{"role":"tool","content":null,"tool_call_id":"a"}],"tools":[{"type":"function","function":{"name":"foo","parameters":{"type":"object","properties":{}}}},{"type":"function","function":{"name":"bar","description":"第二个","strict":true}}],"chat_template_kwargs":{"thinking":true}},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 50 (range 1-100, the higher the value, the more thorough the reasoning)\n\n\n\n## Tools\n\nYou have access to a set of tools to help answer the user's question. You can invoke tools by writing a \"<DSML calls>\" block like the following:\n\n<DSML calls>\n<DSML invoke name=\"$TOOL_NAME\">\n<DSML parameter name=\"$PARAMETER_NAME\" string=\"true|false\">$PARAMETER_VALUE</DSML parameter>\n...\n</DSML invoke>\n<DSML invoke name=\"$TOOL_NAME2\">\n...\n</DSML invoke>\n</DSML calls>\n\nString parameters should be specified as is and set `string=\"true\"`. For all other types (numbers, booleans, arrays, objects), pass the value in JSON format and set `string=\"false\"`.\n\nIf thinking_mode is enabled (triggered by <think>), you MUST output your complete reasoning inside <think>...</think> BEFORE any tool calls or final response.\n\nOtherwise, output directly after </think> with tool calls or final response.\n\n### Available Tool Schemas\n\n{\"name\": \"foo\", \"parameters\": {\"type\": \"object\", \"properties\": {}}}\n{\"name\": \"bar\", \"description\": \"第二个\", \"strict\": true}\n\nYou MUST strictly follow the above defined tool name and parameter schemas to invoke tool calls.\n<User>Hi<Assistant><think>Call both</think>\n\n<DSML calls>\n<DSML invoke name=\"foo\">\n<DSML parameter name=\"x\" string=\"true\">雪</DSML parameter>\n<DSML parameter name=\"y\" string=\"false\">[1, true]</DSML parameter>\n</DSML invoke>\n<DSML invoke name=\"bar\">\n<DSML parameter name=\"z\" string=\"false\">2</DSML parameter>\n</DSML invoke>\n</DSML calls><end▁of▁sentence><User><tool_result></tool_result>\n\n<tool_result>Hello\n\nworld</tool_result><Assistant><think>","token_count":388,"token_sha256":"f9fb80a884084408d01041c452a9ece38a9c3a959a5fd3ed0289db3f6611c203"},
{"name":"kwargs_effort","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"chat_template_kwargs":{"thinking":true,"reasoning_effort":"max"}},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 100 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant><think>","token_count":31,"token_sha256":"a29f18c6e65c09656d098c43e548224b67ced6a674f65186b7e132fb6d997d75"},
{"name":"effort_conflict","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"high","chat_template_kwargs":{"reasoning_effort":"low"}},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 25 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant><think>","token_count":31,"token_sha256":"80aa41f1cbf4d847db08af7c5f4dde508ea75aed6e2cb6b94f3a0a806e275d46"},
{"name":"drop_thinking_ignored","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"assistant","content":"Answer","reasoning_content":"Prior reasoning"},{"role":"user","content":"Again"}],"chat_template_kwargs":{"thinking":true,"drop_thinking":false}},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 50 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant></think>Answer<end▁of▁sentence><User>Again<Assistant><think>","token_count":37,"token_sha256":"f6cb0ab0ad4a220297622920684237a208098dd2b2cd368599bc49ebba7994f9"},
{"name":"thinking_false","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"high","chat_template_kwargs":{"thinking":false}},"prompt":"<begin▁of▁sentence><User>Hi<Assistant></think>","token_count":5,"token_sha256":"95aeed952d0173ec8849b13faec66eb61ec4412cd9c41dae0c405257559365a6"},
{"name":"thinking_none","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"none","chat_template_kwargs":{"thinking":true}},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 50 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant><think>","token_count":31,"token_sha256":"4add8297343cd581f7ded6ecba6c06c5f70f402802db9a021d469efdc2138a15"},
{"name":"consecutive_task","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"},{"role":"user","content":"Again"}],"task":"domain"},"prompt":"<begin▁of▁sentence><User>Hi\n\nAgain<Assistant></think>","token_count":7,"token_sha256":"d05e248d8a06f2c06df20b7a71af0d4dfae1cd91ce9f614719b5842674042d6b"},
{"name":"effort_high","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"high"},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 50 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant><think>","token_count":31,"token_sha256":"4add8297343cd581f7ded6ecba6c06c5f70f402802db9a021d469efdc2138a15"},
{"name":"effort_max","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"max"},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 100 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant><think>","token_count":31,"token_sha256":"a29f18c6e65c09656d098c43e548224b67ced6a674f65186b7e132fb6d997d75"},
{"name":"task_action","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"task":"action"},"prompt":"<begin▁of▁sentence><User>Hi<Assistant></think><action>","token_count":6,"token_sha256":"925bcaae248cc41efabd3e5d8f660f17513e43174f1d34aba51a1399aba76e8e"},
{"name":"effort_low","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"low"},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 25 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant><think>","token_count":31,"token_sha256":"80aa41f1cbf4d847db08af7c5f4dde508ea75aed6e2cb6b94f3a0a806e275d46"},
{"name":"effort_xhigh","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":"xhigh"},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 75 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant><think>","token_count":31,"token_sha256":"54a0e744ad23d9a75775f4a71851980f51a778a1011d73de2f370b1d5499a5b7"},
{"name":"effort_0.75","profile":"v41","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"reasoning_effort":0.75},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 75 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant><think>","token_count":31,"token_sha256":"54a0e744ad23d9a75775f4a71851980f51a778a1011d73de2f370b1d5499a5b7"},
{"name":"kwargs_budget","request":{"model":"m","messages":[{"role":"user","content":"Hi"}],"chat_template_kwargs":{"thinking":true,"reasoning_effort":80}},"prompt":"<begin▁of▁sentence><System>Reasoning Effort: 80 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<User>Hi<Assistant><think>","token_count":31,"token_sha256":"32a4462bbf34bce35dd5c163574188ebd71896ac46d58966e0a6c2481b1b4081"}
]}
@@ -25,6 +25,10 @@ from sglang.srt.entrypoints.openai.serving_chat import OpenAIServingChat
ROOT = Path(__file__).resolve().parents[1] / "fixtures/deepseek"
MODELS = {
"v41": (
"deepseek-ai/DeepSeek-V4.1-Flash",
"dba1be0a40aa45a94ad051997016db3960a90277",
),
"v4": ("deepseek-ai/DeepSeek-V4-Flash", "60d8d70770c6776ff598c94bb586a859a38244f1"),
}
@@ -45,12 +49,18 @@ class RecordingTokenizer:
def main():
ROOT.mkdir(exist_ok=True)
for family, (model, revision) in MODELS.items():
path = snapshot_download(model, revision=revision, local_files_only=True)
path = snapshot_download(
model,
revision=revision,
local_files_only=True,
allow_patterns=["config.json", "tokenizer*.json"],
)
tok = RecordingTokenizer(
AutoTokenizer.from_pretrained(path, local_files_only=True)
)
server = object.__new__(OpenAIServingChat)
server.chat_encoding_spec = "ds" + family
server._dsv41_default_reasoning_effort = "high"
server.tokenizer_manager = SimpleNamespace(tokenizer=tok)
server.template_manager = SimpleNamespace(
jinja_template_content_format="string"