[sgl-router] Add Kimi-K3 rendering with SGLang parity (#40390)

Co-authored-by: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
Kan Wu
2026-09-21 17:38:49 +08:00
committed by GitHub
co-authored by Claude Fable 5.1
parent b86a30afba
commit 2016f5e7a1
16 changed files with 473 additions and 49 deletions
+1
View File
@@ -3283,6 +3283,7 @@ version = "0.1.0"
dependencies = [
"anyhow",
"axum",
"base64 0.22.1",
"bytes",
"clap",
"criterion",
+1
View File
@@ -94,6 +94,7 @@ url = "2"
zeromq = { version = "0.6", default-features = false, features = ["tokio-runtime", "tcp-transport"] }
[dev-dependencies]
base64 = "0.22"
dirs = "5"
http-body-util = "0.1"
# HTTP/2-only mock server for the proxy h2c forwarding tests (tests/proxy/h2c_forward.rs).
+11 -1
View File
@@ -176,7 +176,7 @@ HF Jinja template from `tokenizer_config.json` or a sibling
`chat_template.jinja`, or dynamo-render's built-in DeepSeek encoder (V4 family, V3.2)
for template-less models. Cache-aware routing hashes the rendered tokens so its
prefix queries match the blocks the engine caches. Models the engine encodes in
code but dynamo-render cannot tokenize here (Inkling, Kimi K3) route via raw prompt
code but dynamo-render cannot tokenize here (Inkling) route via raw prompt
text, as does any model whose template fails to load or render.
Plain text chat requests (string `content`, no tools, no template kwargs or
@@ -211,6 +211,16 @@ Detailed content-format parity coverage follows in #39133.
The Dynamo crates are pinned exactly and `Cargo.lock` is committed; CI builds
with `--locked`, so rendered bytes cannot change without a reviewed diff.
## Kimi-K3
Kimi-K3 renders through dynamo-render's native XTML formatter with SGLang's
request semantics (reasoning controls, tools, `response_format`, continuations)
and the checkpoint's chunked tiktoken encoding. `--tokenizer-path` accepts a
local `tiktoken.model` or an HF repo id, whose `tiktoken.model`, `config.json`
and `tokenizer_config.json` are downloaded when it has no `tokenizer.json`.
An explicit null `thinking_effort` with thinking enabled is not representable
in the pinned formatter and falls back to engine-side rendering.
## HTTP/2
There is nothing to configure. The router negotiates per connection inbound and
@@ -6,22 +6,15 @@ use dynamo_tokenizers::{traits::DecodeResult, Tokenizer};
use std::path::Path;
use std::sync::Arc;
/// Load a tokenizer from `source`, which is either a local `tokenizer.json`
/// path or a HuggingFace repo id.
///
/// An existing local file (or anything with a filesystem-path shape) is
/// loaded directly via `Tokenizer::from_file`. Otherwise `source` is treated
/// as a HuggingFace repo id and its `tokenizer.json` is downloaded (once, at
/// startup) into the HF cache, honoring `HF_TOKEN` / `HF_HOME` /
/// `HF_HUB_OFFLINE`. `dynamo_tokenizers` itself has no HF-download path, so
/// the fetch is done here via `hf-hub`.
/// Load a local tokenizer file or Hugging Face repo, honoring HF cache/auth settings.
/// Tiktoken `.model` files also require sibling config.json and tokenizer_config.json.
pub fn load(source: &str) -> Result<Arc<Tokenizer>> {
if Path::new(source).is_file() || looks_like_path(source) {
return Tokenizer::from_file(source)
.map(Arc::new)
.with_context(|| format!("load tokenizer from {source}"));
}
let downloaded = download_tokenizer_json(source)?;
let downloaded = download_tokenizer(source)?;
let path = downloaded
.to_str()
.context("downloaded tokenizer path is not valid UTF-8")?;
@@ -31,8 +24,8 @@ pub fn load(source: &str) -> Result<Arc<Tokenizer>> {
}
/// Treat `source` as a filesystem path (rather than a HuggingFace repo id)
/// when it has a path-like shape — an absolute/relative prefix or a `.json`
/// suffix. HF repo ids are `namespace/name` with none of these markers, so a
/// when it has a path-like shape — an absolute/relative prefix or a
/// tokenizer-file suffix. HF repo ids are `namespace/name` with none of these markers, so a
/// missing local file like `/models/tok.json` reports a load error instead of
/// silently attempting a (doomed) network fetch.
fn looks_like_path(source: &str) -> bool {
@@ -41,20 +34,26 @@ fn looks_like_path(source: &str) -> bool {
|| source.starts_with("../")
|| source.starts_with('~')
|| source.ends_with(".json")
|| source.ends_with(".model")
}
/// Download `tokenizer.json` for a HuggingFace repo id and return the cached
/// local path, adding an actionable error context. The actual fetch (blocking
/// `ureq`, `from_env` so `HF_TOKEN` / `HF_HOME` / endpoint overrides apply)
/// lives in [`download_repo_file`].
fn download_tokenizer_json(repo_id: &str) -> Result<std::path::PathBuf> {
download_repo_file(repo_id, "tokenizer.json").with_context(|| {
/// Keep the tokenizer.json path unchanged; tiktoken models additionally need
/// their configuration siblings in the same HF snapshot directory.
fn download_tokenizer(repo_id: &str) -> Result<std::path::PathBuf> {
if let Ok(path) = download_repo_file(repo_id, "tokenizer.json") {
return Ok(path);
}
let path = download_repo_file(repo_id, "tiktoken.model").with_context(|| {
format!(
"download tokenizer.json for HuggingFace repo {repo_id:?} \
(pass --tokenizer-path with a local tokenizer.json, or set HF_TOKEN \
"download tokenizer.json or tiktoken.model for HuggingFace repo {repo_id:?} \
(pass --tokenizer-path with a local tokenizer file, or set HF_TOKEN \
for a gated/private repo)"
)
})
})?;
for sibling in ["config.json", "tokenizer_config.json"] {
download_repo_file(repo_id, sibling)?;
}
Ok(path)
}
/// Download `file` from a HuggingFace repo id and return the cached local path.
@@ -12,9 +12,10 @@ use std::sync::Arc;
use anyhow::{Context, Result};
use dynamo_renderer::{
deepseek_formatter_for, may_be_fix_tool_schema, ChatTemplate, ContextMixins,
OAIChatLikeRequest, OAIPromptFormatter, PromptFormatter,
deepseek_formatter_for, kimi_k3_formatter_for, may_be_fix_tool_schema, ChatTemplate,
ContextMixins, OAIChatLikeRequest, OAIPromptFormatter, PromptFormatter, RenderedPrompt,
};
use dynamo_tokenizers::{EncodeSegment, Tokenizer};
use minijinja::Value;
use serde_json::Value as JsonValue;
@@ -40,6 +41,7 @@ pub struct ChatFormatter {
/// Stripped from a separately tokenized continuation prefix, as SGLang does.
bos_token: Option<String>,
is_deepseek_v4: bool,
is_kimi_k3: bool,
}
impl ChatFormatter {
@@ -49,9 +51,12 @@ impl ChatFormatter {
let model_type = files
.json("config.json")?
.and_then(|cfg| cfg["model_type"].as_str().map(str::to_owned));
if let Some(kimi) = Self::kimi_native(model_type.as_deref(), model_id) {
return Ok(Some(kimi));
}
match model_type.as_deref() {
// These require tokenization paths not yet supported by this adapter.
Some("inkling_mm_model" | "kimi_k3") => return Ok(None),
Some("inkling_mm_model") => return Ok(None),
Some(t) if t.starts_with("deepseek_v4") => {
return Ok(Self::deepseek_native(model_type.as_deref(), model_id));
}
@@ -132,19 +137,31 @@ impl ChatFormatter {
defaults,
bos_token,
is_deepseek_v4: false,
is_kimi_k3: false,
}))
}
/// dynamo-render's native Kimi-K3 XTML formatter, wrapped in SGLang's request
/// semantics (`kimi::normalize`) and the checkpoint's chunked tokenization.
pub fn kimi_native(model_type: Option<&str>, model_id: &str) -> Option<Self> {
let model_type = model_type.map(str::to_lowercase);
let PromptFormatter::OAI(formatter) =
kimi_k3_formatter_for(&model_type, &model_name(model_id), false)?;
Some(Self {
formatter,
defaults: HashMap::new(),
bos_token: Some("[BOS]".into()),
is_deepseek_v4: false,
is_kimi_k3: true,
})
}
/// dynamo-render's code-based DeepSeek encoders, for V4 (including variants
/// such as V4.1) and V3.2 non-Exp: the only built-in formatters verified
/// against the engine. `model_type` (from `config.json`) is authoritative;
/// the model id's last path segment is the fallback.
pub fn deepseek_native(model_type: Option<&str>, model_id: &str) -> Option<Self> {
let name = model_id
.rsplit('/')
.next()
.unwrap_or(model_id)
.to_lowercase();
let name = model_name(model_id);
// The engine treats every `deepseek_v4*` variant (e.g. V4.1) as V4.
let model_type = model_type.map(str::to_lowercase).map(|t| {
if t.starts_with("deepseek_v4") {
@@ -174,6 +191,7 @@ impl ChatFormatter {
defaults,
bos_token: Some("<begin▁of▁sentence>".into()),
is_deepseek_v4,
is_kimi_k3: false,
})
}
@@ -229,7 +247,13 @@ impl ChatFormatter {
if self.is_deepseek_v4 && !matches!(effort.as_str(), Some("low" | "high" | "max")) {
effort = "low".into();
}
kwargs.entry("reasoning_effort".into()).or_insert(effort);
if self.is_kimi_k3 {
if matches!(effort.as_str(), Some("low" | "high" | "max")) {
kwargs.entry("thinking_effort".into()).or_insert(effort);
}
} else {
kwargs.entry("reasoning_effort".into()).or_insert(effort);
}
}
for (key, value) in &self.defaults {
kwargs.entry(key.clone()).or_insert_with(|| value.clone());
@@ -239,8 +263,8 @@ impl ChatFormatter {
/// Rendered prompt plus the assistant continuation prefix SGLang tokenizes
/// separately (`_handle_last_assistant_message`).
fn render_parts(&self, request: &JsonValue) -> Result<(String, String)> {
let kwargs = self.template_kwargs(request)?;
fn render_parts(&self, request: &JsonValue) -> Result<(RenderedPrompt, String)> {
let mut kwargs = self.template_kwargs(request)?;
let continuing = request["continue_final_message"] == true;
let mut messages: Vec<JsonValue> = request["messages"]
.as_array()
@@ -248,13 +272,16 @@ impl ChatFormatter {
.iter()
.map(engine_message)
.collect();
if self.is_kimi_k3 {
super::kimi::normalize(request, &mut messages, &mut kwargs)?;
}
let mut prefix = String::new();
if let Some(last) = messages.last_mut().filter(|m| m["role"] == "assistant") {
if let Some(content) = last["content"].as_str() {
if continuing {
prefix = content.to_owned();
messages.pop();
} else {
} else if !self.is_kimi_k3 {
*last = serde_json::json!({"role": "user", "content": content});
}
}
@@ -271,25 +298,30 @@ impl ChatFormatter {
}
let prompt = self
.formatter
.render(&ChatRequest {
.render_prompt(&ChatRequest {
request,
messages,
kwargs,
is_kimi_k3: self.is_kimi_k3,
})
.context("render chat template")?;
Ok((prompt, prefix))
}
pub fn encode(
&self,
tokenizer: &dynamo_tokenizers::Tokenizer,
request: &JsonValue,
) -> Result<Vec<u32>> {
pub fn encode(&self, tokenizer: &Tokenizer, request: &JsonValue) -> Result<Vec<u32>> {
let (prompt, prefix) = self.render_parts(request)?;
let mut ids = super::adapter::encode(tokenizer, &prompt)?;
let mut ids = match prompt.encode_segments() {
Some(segments) if self.is_kimi_k3 => super::kimi::encode(tokenizer, &segments)?,
Some(segments) => tokenizer.encode_segments(&segments)?.token_ids().to_vec(),
None => super::adapter::encode(tokenizer, prompt.as_str())?,
};
if !prefix.is_empty() {
// SGLang encodes the assistant prefix separately and removes its leading BOS.
let mut suffix = super::adapter::encode(tokenizer, &prefix)?;
let mut suffix = if self.is_kimi_k3 {
super::kimi::encode(tokenizer, &[EncodeSegment::control(&prefix)])?
} else {
super::adapter::encode(tokenizer, &prefix)?
};
if let Some(bos) = self.bos_token.as_deref().filter(|s| !s.is_empty()) {
let bos = super::adapter::encode(tokenizer, bos)?;
if bos.len() == 1 && suffix.first() == bos.first() {
@@ -304,7 +336,7 @@ impl ChatFormatter {
/// Use `encode` for token ids to preserve continuation boundaries.
pub fn render(&self, request: &JsonValue) -> Result<String> {
let (prompt, prefix) = self.render_parts(request)?;
Ok(prompt + &prefix)
Ok(prompt.into_text() + &prefix)
}
}
@@ -341,6 +373,15 @@ fn engine_message(message: &JsonValue) -> JsonValue {
out.into()
}
/// Lowercased last path segment of a model id, dynamo-render's name fallback.
fn model_name(model_id: &str) -> String {
model_id
.rsplit('/')
.next()
.unwrap_or(model_id)
.to_lowercase()
}
/// `content` of an HF `AddedToken` object (`{"content": "<s>", "lstrip": ...}`).
fn added_token_content(token: &JsonValue) -> Option<String> {
token
@@ -355,6 +396,7 @@ struct ChatRequest<'a> {
/// Normalized copy of `request["messages"]`.
messages: Vec<JsonValue>,
kwargs: ChatTemplateKwargs,
is_kimi_k3: bool,
}
impl OAIChatLikeRequest for ChatRequest<'_> {
@@ -375,6 +417,9 @@ impl OAIChatLikeRequest for ChatRequest<'_> {
if tools.as_array().is_none_or(|t| t.is_empty()) {
return None;
}
if self.is_kimi_k3 {
return Some(Value::from_serialize(tools));
}
let mut tools = tools.clone();
// SGLang renders only the named tool for a function `tool_choice`.
if let Some(name) = self.request["tool_choice"]["function"]["name"].as_str() {
@@ -385,18 +430,24 @@ impl OAIChatLikeRequest for ChatRequest<'_> {
may_be_fix_tool_schema(tools)
}
fn tool_choice(&self) -> Option<Value> {
self.request.get("tool_choice").map(Value::from_serialize)
if self.is_kimi_k3 {
self.kwargs.get("tool_choice").map(Value::from_serialize)
} else {
self.request.get("tool_choice").map(Value::from_serialize)
}
}
fn reasoning_effort(&self) -> Option<Value> {
self.kwargs
.get("reasoning_effort")
.map(Value::from_serialize)
}
/// Withheld: the engine enforces `response_format` by constrained decoding
/// and never renders it, while dynamo-render's DeepSeek formatters would
/// append a "## Response Format" schema preamble to the system turn.
/// Only Kimi-K3 renders `response_format`; elsewhere the engine enforces
/// it by constrained decoding and never renders it.
fn response_format(&self) -> Option<Value> {
None
self.is_kimi_k3
.then(|| self.kwargs.get("response_format"))
.flatten()
.map(Value::from_serialize)
}
fn should_add_generation_prompt(&self) -> bool {
true
@@ -0,0 +1,130 @@
// SPDX-FileCopyrightText: Copyright (c) 2026 The SGLang Authors
// SPDX-License-Identifier: Apache-2.0
//! Kimi-K3 request semantics from `serving_chat.py` and the checkpoint's
//! `tokenization_kimi.py`, applied around dynamo-render's native formatter.
use anyhow::{ensure, Result};
use dynamo_tokenizers::{EncodeSegment, Tokenizer};
use serde_json::{json, Value};
use super::chat_formatter::ChatTemplateKwargs;
pub(super) fn normalize(
request: &Value,
messages: &mut [Value],
kwargs: &mut ChatTemplateKwargs,
) -> Result<()> {
// Dynamo reads `reasoning_effort` and treats a non-bool `thinking` as true;
// the checkpoint ignores the former and uses Python truthiness for the latter.
kwargs.remove("reasoning_effort");
let thinking = kwargs
.get("thinking")
.is_none_or(|v| minijinja::Value::from_serialize(v).is_true());
kwargs.insert("thinking".into(), thinking.into());
ensure!(
!thinking || !kwargs.get("thinking_effort").is_some_and(Value::is_null),
"Kimi null thinking_effort requires engine-side rendering"
);
for message in messages.iter_mut() {
if message["role"] == "developer" {
message["role"] = "system".into();
}
if let Some(parts) = message["content"].as_array_mut() {
parts.retain(|part| matches!(part["type"].as_str(), Some("text" | "image_url")));
}
for call in message["tool_calls"].as_array_mut().into_iter().flatten() {
let args = &mut call["function"]["arguments"];
if let Some(parsed) = args
.as_str()
.and_then(|s| serde_json::from_str::<Value>(s).ok())
.filter(Value::is_object)
{
*args = parsed;
}
neutralize(args);
}
neutralize(&mut message["content"]);
if let Some(reasoning) = message.get_mut("reasoning_content") {
neutralize(reasoning);
}
}
let has_tools = std::iter::once(request)
.chain(messages.iter().filter(|m| m["role"] == "system"))
.any(|m| m["tools"].as_array().is_some_and(|t| !t.is_empty()));
if has_tools && matches!(request["tool_choice"].as_str(), Some("none" | "required")) {
kwargs
.entry("tool_choice".into())
.or_insert_with(|| request["tool_choice"].clone());
}
// The checkpoint renders only these; a named choice is constrained decoding.
if !matches!(
kwargs.get("tool_choice").and_then(Value::as_str),
Some("none" | "required")
) {
kwargs.remove("tool_choice");
}
if let Some(mut format) = request
.get("response_format")
.filter(|v| !v.is_null())
.cloned()
{
// protocol.py lifts a legacy top-level `schema` into `json_schema`.
if format["type"] == "json_schema" && format["json_schema"].is_null() {
if let Some(mut schema) = format.as_object_mut().and_then(|f| f.remove("schema")) {
if let Some(props) = schema.get_mut("properties").and_then(Value::as_object_mut) {
props.remove("strict");
}
format["json_schema"] = json!({"schema": schema});
}
}
kwargs.entry("response_format".into()).or_insert(format);
}
if let Some(schema) = kwargs.get("response_schema").cloned() {
if let Some(format) = kwargs
.get_mut("response_format")
.filter(|f| f["type"] == "json_schema")
{
format["json_schema"] = json!({"schema": schema});
}
}
Ok(())
}
fn neutralize(value: &mut Value) {
match value {
Value::String(text) => {
*text = text.replace("<|kimi_image_placeholder|>", "<| kimi_image_placeholder |>")
}
Value::Array(values) => values.iter_mut().for_each(neutralize),
Value::Object(values) => values.values_mut().for_each(neutralize),
_ => {}
}
}
/// `tokenization_kimi.py` encodes 400k-char windows, each split after 25k
/// consecutive (non-)whitespace chars, and BPE is not chunk-invariant.
pub(super) fn encode(tokenizer: &Tokenizer, segments: &[EncodeSegment<'_>]) -> Result<Vec<u32>> {
let mut chunks = Vec::new();
for segment in segments {
let (mut start, mut run, mut was_space) = (0, 0, false);
for (count, (offset, ch)) in segment.text.char_indices().enumerate() {
// Python's `str.isspace` also covers U+001C..U+001F.
let space = ch.is_whitespace() || ('\u{1c}'..='\u{1f}').contains(&ch);
run = if space == was_space { run + 1 } else { 1 };
if (count > 0 && count % 400_000 == 0) || run > 25_000 {
chunks.push(EncodeSegment::new(
&segment.text[start..offset],
segment.allow_special,
));
(start, run) = (offset, 1);
}
was_space = space;
}
chunks.push(EncodeSegment::new(
&segment.text[start..],
segment.allow_special,
));
}
Ok(tokenizer.encode_segments(&chunks)?.token_ids().to_vec())
}
+6 -1
View File
@@ -3,6 +3,7 @@
pub mod adapter;
pub mod chat_formatter;
mod kimi;
use anyhow::Result;
use chat_formatter::ChatFormatter;
@@ -350,7 +351,11 @@ mod tests {
assert_eq!(resolve(model_type).unwrap().render(&request).unwrap(), "T");
}
assert!(resolve("inkling_mm_model").is_none());
assert!(resolve("kimi_k3").is_none());
assert!(resolve("kimi_k3")
.unwrap()
.render(&request)
.unwrap()
.contains("<|open|>message"));
assert_eq!(
resolve("deepseek_v41").unwrap().render(&request).unwrap(),
"<begin▁of▁sentence><User>hi<Assistant></think>"
@@ -0,0 +1,44 @@
// SPDX-FileCopyrightText: Copyright (c) 2026 The SGLang Authors
// SPDX-License-Identifier: Apache-2.0
use serde_json::{json, Value};
use sgl_router::tokenizer::{adapter, chat_formatter::ChatFormatter};
use sha2::{Digest, Sha256};
#[path = "../../fixtures/kimi_k3.rs"]
mod fixture;
#[test]
fn kimi_tokens_match_sglang() {
let fixture = fixture::tokenizer();
let path = fixture.path().join("tiktoken.model");
let path = path.to_str().unwrap();
let tokenizer = adapter::load(path).unwrap();
let formatter = ChatFormatter::load("served-alias", path).unwrap().unwrap();
let cases: Vec<Value> =
serde_json::from_str(include_str!("../../fixtures/kimi_k3/prompts.json")).unwrap();
for case in cases {
let mut request = case["request"].clone();
if let Some(repeat) = case["repeat"].as_u64() {
request["messages"][0]["content"] = request["messages"][0]["content"]
.as_str()
.unwrap()
.repeat(repeat as usize)
.into();
}
let ids = formatter.encode(&tokenizer, &request).unwrap();
let mut hash = Sha256::new();
for id in &ids {
hash.update(id.to_le_bytes());
}
assert_eq!(json!(ids.len()), case["token_count"], "{}", case["name"]);
assert_eq!(
format!("{:x}", hash.finalize()),
case["sha256"],
"{}",
case["name"]
);
}
let request = json!({"messages":[{"role":"user","content":"hi"}], "chat_template_kwargs":{"thinking_effort":null}});
assert!(formatter.encode(&tokenizer, &request).is_err());
}
@@ -1,5 +1,6 @@
// SPDX-FileCopyrightText: Copyright (c) 2026 The SGLang Authors
// SPDX-License-Identifier: Apache-2.0
mod kimi;
mod parity;
mod render_parity;
+28
View File
@@ -0,0 +1,28 @@
// SPDX-FileCopyrightText: Copyright (c) 2026 The SGLang Authors
// SPDX-License-Identifier: Apache-2.0
use base64::{engine::general_purpose::STANDARD, Engine};
pub fn tokenizer() -> tempfile::TempDir {
let dir = tempfile::tempdir().unwrap();
let bytes = (0..=255u8).map(|byte| vec![byte]);
let merges = include_str!("kimi_k3/merges.txt")
.split_whitespace()
.map(|s| s.as_bytes().to_vec());
let vocab: String = bytes
.chain(merges)
.enumerate()
.map(|(rank, token)| format!("{} {rank}\n", STANDARD.encode(token)))
.collect();
std::fs::write(dir.path().join("tiktoken.model"), vocab).unwrap();
for (name, contents) in [
("config.json", r#"{"model_type":"kimi_k3"}"#),
(
"tokenizer_config.json",
include_str!("kimi_k3/tokenizer_config.json"),
),
] {
std::fs::write(dir.path().join(name), contents).unwrap();
}
dir
}
@@ -0,0 +1,9 @@
Synthetic Kimi vocabulary: the shared fixture generates 256 byte tokens, then
adds `merges.txt` in rank order and the configured protocol markers.
`prompts.json` records token counts and SHA-256 of little-endian u32 token IDs
from SGLang's `_encode_messages` and `moonshotai/Kimi-K3` revision
`f831ab66814297da540d832a5235f8e904f29d06`. Regenerate in a SGLang Python environment:
```sh
python tests/scripts/generate_kimi_parity.py
```
@@ -0,0 +1,4 @@
me ss ag message ro le role us er user as si st
ant assistant sy em system to ol tool th in think res pon
se response he ll hello wo rld world =" ca call ar gu
ment argument js on json ty pe type ke key de cl are
@@ -0,0 +1,22 @@
[
{"name":"default","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}]},"token_count":275,"sha256":"43f237fef9c404e15ba9ac9a4fc948b6301b8a0bf5336576dd5e90c46d5c1cb8"},
{"name":"effort_none","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"reasoning_effort":"none"},"token_count":58,"sha256":"d9764be8b6a9a04114759d162f5a1c4e67109a7af2d62e72c5faf8f150c05d59"},
{"name":"effort_precedence","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"reasoning_effort":"high","chat_template_kwargs":{"reasoning_effort":"low"}},"token_count":276,"sha256":"54e51363cdb9f32dd2a11d805c3b5797b969526c338c7a93d0ae7740457073dc"},
{"name":"reasoning_alias","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"reasoning":{"effort":"low"}},"token_count":275,"sha256":"8da270ec9559dfa53515591fe37b480e68f6b70275265d5086a2ff43a747d32a"},
{"name":"enable_thinking_only","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"chat_template_kwargs":{"enable_thinking":false}},"token_count":275,"sha256":"43f237fef9c404e15ba9ac9a4fc948b6301b8a0bf5336576dd5e90c46d5c1cb8"},
{"name":"unsupported_effort","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"reasoning_effort":"medium"},"token_count":275,"sha256":"43f237fef9c404e15ba9ac9a4fc948b6301b8a0bf5336576dd5e90c46d5c1cb8"},
{"name":"tool_none","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"tools":[{"type":"function","function":{"name":"lookup","parameters":{"type":"object"}}}],"tool_choice":"none"},"token_count":523,"sha256":"925e846f2a2686ecbb6ee899b15155305459bd01b31833c8e4c522d6655a5bae"},
{"name":"named_tool","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"tools":[{"type":"function","function":{"name":"lookup","parameters":{"type":"object"}}}],"tool_choice":{"type":"function","function":{"name":"lookup"}}},"token_count":426,"sha256":"6d87edaec38985be9cbbba896e68ceb6c6cf6c9fa0b0d3addf0b05045511347c"},
{"name":"choice_without_tools","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"tool_choice":"required"},"token_count":275,"sha256":"43f237fef9c404e15ba9ac9a4fc948b6301b8a0bf5336576dd5e90c46d5c1cb8"},
{"name":"developer_tools","request":{"messages":[{"role":"developer","content":"policy","tools":[{"type":"function","function":{"name":"lookup","parameters":{"type":"object"}}}]},{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"tool_choice":"required"},"token_count":614,"sha256":"16ee5d4d7cf852ce4a5cb7a079f42fcae517200704a7a99f5e12572976e6c50c"},
{"name":"schema_alias","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"response_format":{"type":"json_schema","schema":{"type":"object","properties":{"answer":{"type":"string"},"strict":{"default":true}}}}},"token_count":530,"sha256":"4978c30e0da93740589ec2d193f06ade089301d6af32ec6f2b39a4792cad5e17"},
{"name":"format_override","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"}],"response_format":{"type":"json_object"},"chat_template_kwargs":{"response_format":{"type":"json_schema","json_schema":{"name":"answer","schema":{"type":"string"}}}}},"token_count":494,"sha256":"d78ae98b5fd5998c75a7cbcf7ca253a169c61fdea7b6549c0d959c44e9f486d5"},
{"name":"continuation","request":{"messages":[{"role":"user","content":"hello <|open|> <|kimi_image_placeholder|>"},{"role":"assistant","content":"[BOS]partial <|open|> <|kimi_image_placeholder|>"}],"continue_final_message":true},"token_count":309,"sha256":"583362e9c2e28226e52ac6ca3eb4fb2d2c312653d995bcac291ff94154cdd1a7"},
{"name":"assistant_history","request":{"messages":[{"role":"user","content":"hello","name":"ignored"},{"role":"assistant","reasoning_content":"why <|kimi_image_placeholder|>","content":"answer"}]},"token_count":299,"sha256":"0480fa42372ee2cd1bc0653a1d3bebef5eee0b2acc7aac47960b05d14768dc8a"},
{"name":"tool_result","request":{"messages":[{"role":"assistant","tool_calls":[{"id":"c1","type":"function","function":{"name":"lookup","arguments":"{\"x\":\"<|kimi_image_placeholder|>\"}"}}]},{"role":"tool","tool_call_id":"c1","content":"result"}]},"token_count":362,"sha256":"4f85e03e0f4e9e7909040afbac4432a57fc1b01ce1c90b13d0c365cf4155336e"},
{"name":"image","request":{"messages":[{"role":"user","content":[{"type":"text","text":"hi <|kimi_image_placeholder|>"},{"type":"image_url","image_url":{"url":"https://example.com/a.png"}}]}]},"token_count":269,"sha256":"c22f63a7138b758acf9925aea4264e6a3469b0ea50c4b1d16325f825476e966d"},
{"name":"long_run","request":{"messages":[{"role":"user","content":"界hello"}]},"repeat":4168,"token_count":16915,"sha256":"05651835990fb23205bd5731718edd7c0b246d30e4f84ad345671156d0c0be4c"},
{"name":"long_segment","request":{"messages":[{"role":"user","content":"hello "}]},"repeat":66668,"token_count":266910,"sha256":"0d1ec109069b74c82c0af95b2685972dc56f6c381639a62f9993384407354ec3"},
{"name":"null_thinking","request":{"messages":[{"role":"user","content":"hello"}],"chat_template_kwargs":{"thinking":null,"thinking_effort":null,"tool_choice":"specified"}},"token_count":24,"sha256":"e7a5f51a4704edb223f44493ca6f637d008177112d612fb57267a9e647480d3f"},
{"name":"scalar_schema_alias","request":{"messages":[{"role":"user","content":"hello"}],"response_format":{"type":"json_schema","schema":{"type":"string"}}},"token_count":460,"sha256":"13b51dde795aa1bb91428316cd6d6db66402c39ba649405e8d8d7adfe3817c0b"}
]
@@ -0,0 +1,19 @@
{
"added_tokens_decoder": {
"308": {"content": "[BOS]"},
"309": {"content": "[EOS]"},
"310": {"content": "<|end_of_msg|>"},
"311": {"content": "<|open|>"},
"312": {"content": "<|close|>"},
"313": {"content": "<|sep|>"},
"329": {"content": "<|media_pad|>"},
"562": {"content": "[UNK]"},
"563": {"content": "[PAD]"}
},
"bos_token": "[BOS]",
"eos_token": "[EOS]",
"pad_token": "[PAD]",
"unk_token": "[UNK]",
"additional_special_tokens": ["<|end_of_msg|>"],
"tokenizer_class": "TikTokenTokenizer"
}
@@ -399,3 +399,35 @@ async fn role_rewrites_preserve_messages_without_forwarding_ids() {
assert_eq!(send(ctx, request).await, StatusCode::OK);
assert!(captured(&mock).get("input_ids").is_some());
}
#[path = "../fixtures/kimi_k3.rs"]
mod kimi_fixture;
#[tokio::test]
async fn kimi_ids_forward_with_engine_rendering_fallback() {
let mock = MockWorker::start(vec![]).await;
let fixture = kimi_fixture::tokenizer();
let mut cfg = config();
let path = fixture.path().join("tiktoken.model");
cfg.model.tokenizer_path = path.display().to_string();
let ctx = build_ctx_with_config(mock.url.clone(), cfg);
for (content, kwargs) in [
("literal <|open|> text", None),
("hi", Some(json!({"thinking_effort": null}))),
] {
let mut request =
json!({"model": MODEL, "messages": [{"role": "user", "content": content}]});
let forward = kwargs.is_none();
if let Some(kwargs) = kwargs {
request["chat_template_kwargs"] = kwargs;
}
let ids = ctx.tokenizers.encode_chat(MODEL, &request);
assert_eq!(send(ctx.clone(), request.clone()).await, StatusCode::OK);
if forward {
request["input_ids"] = json!(ids.unwrap());
} else {
assert!(ids.is_none());
}
assert_eq!(captured(&mock), request);
}
}
@@ -0,0 +1,68 @@
"""Regenerate Kimi IDs with SGLang and the pinned checkpoint tokenizer.
Run from experimental/sgl-router in a SGLang Python environment.
"""
import base64
import copy
import hashlib
import json
import pathlib
import sys
import tempfile
from types import SimpleNamespace
from huggingface_hub import hf_hub_download
from tokenizers import AddedToken
from sglang.srt.entrypoints.openai.protocol import ChatCompletionRequest
from sglang.srt.entrypoints.openai.serving_chat import (
OpenAIServingChat,
ThinkingMode,
normalize_assistant_tool_call_arguments,
)
REVISION = "f831ab66814297da540d832a5235f8e904f29d06"
for name in ("encoding_k3.py", "tokenization_kimi.py"):
source = hf_hub_download("moonshotai/Kimi-K3", name, revision=REVISION)
sys.path.insert(0, str(pathlib.Path(source).parent))
from tokenization_kimi import TikTokenTokenizer # noqa: E402
fixture = pathlib.Path(__file__).resolve().parents[1] / "fixtures/kimi_k3"
config = json.loads((fixture / "tokenizer_config.json").read_text())
config["added_tokens_decoder"] = {
int(k): AddedToken(**v) for k, v in config["added_tokens_decoder"].items()
}
tokens = [bytes([b]) for b in range(256)]
tokens += [s.encode() for s in (fixture / "merges.txt").read_text().split()]
vocab = "".join(
f"{base64.b64encode(token).decode()} {rank}\n" for rank, token in enumerate(tokens)
)
with tempfile.NamedTemporaryFile(suffix=".model", mode="w+") as model:
model.write(vocab)
model.flush()
tokenizer = TikTokenTokenizer(model.name, **config)
server = object.__new__(OpenAIServingChat)
server.chat_encoding_spec = "kimi_k3"
server.tokenizer_manager = SimpleNamespace(tokenizer=tokenizer)
cases = json.loads((fixture / "prompts.json").read_text())
for case in cases:
data = copy.deepcopy(case["request"])
if "repeat" in case:
data["messages"][0]["content"] *= case["repeat"]
request = ChatCompletionRequest(**data)
messages = [message.model_dump() for message in request.messages]
for message in messages:
normalize_assistant_tool_call_arguments(message, strict=False)
ids = server._encode_messages(messages, request, ThinkingMode.THINKING)
case["token_count"] = len(ids)
case["sha256"] = hashlib.sha256(
b"".join(token.to_bytes(4, "little") for token in ids)
).hexdigest()
(fixture / "prompts.json").write_text(
"[\n"
+ ",\n".join(
json.dumps(c, ensure_ascii=False, separators=(",", ":")) for c in cases
)
+ "\n]\n"
)