Fix Anthropic Messages API compatibility (#25876)
Co-authored-by: Jairo David Campaña Rosero <jairocampana10001@gmail.com> Co-authored-by: Karan Bansal <3264937+karanb192@users.noreply.github.com> Co-authored-by: eason <85663565+mango766@users.noreply.github.com> Co-authored-by: Yufeng He <40085740+he-yufeng@users.noreply.github.com> Co-authored-by: qingchanghan <17794466+qingchanghan@users.noreply.github.com> Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com> Co-authored-by: Ajay Anubolu <124525760+AjAnubolu@users.noreply.github.com> Co-authored-by: Ravitez Dondeti <13931987+dondetir@users.noreply.github.com> Co-authored-by: Ratish P <114130421+Ratish1@users.noreply.github.com> Co-authored-by: Xiaoshuai Zhang <15795935+jetd1@users.noreply.github.com> Co-authored-by: Ricardo-M-L <69202550+Ricardo-M-L@users.noreply.github.com> Co-authored-by: Xinyuan Tong <xinyuan.tong@radixark.ai>
This commit is contained in:
co-authored by
Jairo David Campaña Rosero
Karan Bansal
eason
Yufeng He
qingchanghan
Claude Opus 4.7
Ajay Anubolu
Ravitez Dondeti
Ratish P
Xiaoshuai Zhang
Ricardo-M-L
Xinyuan Tong
parent
caf59759ea
commit
b3270264e4
@@ -1,71 +1,140 @@
|
||||
"""Pydantic models for Anthropic Messages API protocol"""
|
||||
"""Pydantic models for Anthropic Messages API protocol.
|
||||
|
||||
Mirrors the shape of the official Anthropic Python SDK
|
||||
(``anthropic-sdk-python``): ``ContentBlock``, ``Tool``, ``MessageStreamEvent``
|
||||
and ``ContentBlockDelta`` are discriminated unions over the ``type`` field,
|
||||
so each variant carries only the fields it actually uses.
|
||||
"""
|
||||
|
||||
import uuid
|
||||
from typing import Any, Literal, Optional
|
||||
from typing import Annotated, Any, Literal, Optional, Union
|
||||
|
||||
from pydantic import BaseModel, Field, field_validator
|
||||
from pydantic import (
|
||||
BaseModel,
|
||||
Discriminator,
|
||||
Field,
|
||||
NonNegativeInt,
|
||||
Tag,
|
||||
field_validator,
|
||||
model_validator,
|
||||
)
|
||||
|
||||
|
||||
class AnthropicError(BaseModel):
|
||||
"""Error structure for Anthropic API"""
|
||||
"""Error structure for Anthropic API."""
|
||||
|
||||
type: str
|
||||
message: str
|
||||
|
||||
|
||||
class AnthropicErrorResponse(BaseModel):
|
||||
"""Error response structure for Anthropic API"""
|
||||
"""Error response structure for Anthropic API."""
|
||||
|
||||
type: Literal["error"] = "error"
|
||||
error: AnthropicError
|
||||
|
||||
|
||||
class AnthropicUsage(BaseModel):
|
||||
"""Token usage information"""
|
||||
"""Token usage information.
|
||||
|
||||
input_tokens: int
|
||||
output_tokens: int
|
||||
cache_creation_input_tokens: Optional[int] = None
|
||||
cache_read_input_tokens: Optional[int] = None
|
||||
``input_tokens``/``output_tokens`` are ``Optional`` because Anthropic's
|
||||
streaming ``message_delta`` event omits ``input_tokens`` (the spec
|
||||
requires it only on ``message_start``). Non-streaming responses set both.
|
||||
"""
|
||||
|
||||
input_tokens: Optional[NonNegativeInt] = None
|
||||
output_tokens: Optional[NonNegativeInt] = None
|
||||
cache_creation_input_tokens: Optional[NonNegativeInt] = None
|
||||
cache_read_input_tokens: Optional[NonNegativeInt] = None
|
||||
|
||||
|
||||
class AnthropicContentBlock(BaseModel):
|
||||
"""Content block in message"""
|
||||
# ---------- Content blocks (discriminated by ``type``) ----------
|
||||
|
||||
type: Literal[
|
||||
"text",
|
||||
"image",
|
||||
"tool_use",
|
||||
"tool_result",
|
||||
"tool_reference",
|
||||
"thinking",
|
||||
"redacted_thinking",
|
||||
]
|
||||
text: Optional[str] = None
|
||||
# For image content
|
||||
source: Optional[dict[str, Any]] = None
|
||||
# For tool use/result
|
||||
id: Optional[str] = None
|
||||
|
||||
class TextBlock(BaseModel):
|
||||
type: Literal["text"] = "text"
|
||||
text: str
|
||||
|
||||
|
||||
class ImageBlock(BaseModel):
|
||||
type: Literal["image"] = "image"
|
||||
# Kept loosely typed for compat with both base64 and URL sources; the
|
||||
# serving layer normalises to OpenAI ``image_url`` parts.
|
||||
source: Optional[Union[dict[str, Any], str]] = None
|
||||
|
||||
|
||||
class ToolUseBlock(BaseModel):
|
||||
type: Literal["tool_use"] = "tool_use"
|
||||
id: str
|
||||
name: str
|
||||
input: dict[str, Any] = Field(default_factory=dict)
|
||||
|
||||
|
||||
class ToolResultBlock(BaseModel):
|
||||
type: Literal["tool_result"] = "tool_result"
|
||||
tool_use_id: Optional[str] = None
|
||||
name: Optional[str] = None
|
||||
input: Optional[dict[str, Any]] = None
|
||||
content: Optional[str | list[dict[str, Any]]] = None
|
||||
# Some legacy payloads use ``id`` instead of ``tool_use_id``.
|
||||
id: Optional[str] = None
|
||||
content: Optional[Union[str, list["AnthropicContentBlock"]]] = None
|
||||
is_error: Optional[bool] = None
|
||||
# For thinking content
|
||||
thinking: Optional[str] = None
|
||||
|
||||
|
||||
class ToolReferenceBlock(BaseModel):
|
||||
"""sglang extension: references a deferred-loaded tool by name."""
|
||||
|
||||
type: Literal["tool_reference"] = "tool_reference"
|
||||
name: Optional[str] = None
|
||||
# Anthropic-style payloads sometimes use ``tool_name``; accept both.
|
||||
tool_name: Optional[str] = None
|
||||
id: Optional[str] = None
|
||||
|
||||
|
||||
class SearchResultBlock(BaseModel):
|
||||
type: Literal["search_result"] = "search_result"
|
||||
# ``source`` here is a URL/identifier string (unlike ImageBlock.source).
|
||||
source: Optional[Union[str, dict[str, Any]]] = None
|
||||
title: Optional[str] = None
|
||||
content: Optional[list[dict[str, Any]]] = None
|
||||
|
||||
|
||||
class ThinkingBlock(BaseModel):
|
||||
type: Literal["thinking"] = "thinking"
|
||||
thinking: str
|
||||
signature: Optional[str] = None
|
||||
|
||||
|
||||
class RedactedThinkingBlock(BaseModel):
|
||||
type: Literal["redacted_thinking"] = "redacted_thinking"
|
||||
data: Optional[str] = None
|
||||
|
||||
|
||||
AnthropicContentBlock = Annotated[
|
||||
Union[
|
||||
TextBlock,
|
||||
ImageBlock,
|
||||
ToolUseBlock,
|
||||
ToolResultBlock,
|
||||
ToolReferenceBlock,
|
||||
SearchResultBlock,
|
||||
ThinkingBlock,
|
||||
RedactedThinkingBlock,
|
||||
],
|
||||
Field(discriminator="type"),
|
||||
]
|
||||
|
||||
|
||||
class AnthropicMessage(BaseModel):
|
||||
"""Message structure"""
|
||||
|
||||
role: Literal["user", "assistant"]
|
||||
content: str | list[AnthropicContentBlock]
|
||||
content: Union[str, list[AnthropicContentBlock]]
|
||||
|
||||
|
||||
class AnthropicTool(BaseModel):
|
||||
"""Tool definition"""
|
||||
# ---------- Tools (discriminated by ``type`` family) ----------
|
||||
|
||||
|
||||
class AnthropicCustomTool(BaseModel):
|
||||
"""Custom tool defined by the API user — requires ``input_schema``."""
|
||||
|
||||
type: Optional[Literal["custom"]] = None # absent or explicit "custom"
|
||||
name: str
|
||||
description: Optional[str] = None
|
||||
input_schema: dict[str, Any]
|
||||
@@ -73,7 +142,7 @@ class AnthropicTool(BaseModel):
|
||||
|
||||
@field_validator("input_schema")
|
||||
@classmethod
|
||||
def validate_input_schema(cls, v):
|
||||
def _ensure_object_schema(cls, v):
|
||||
if not isinstance(v, dict):
|
||||
raise ValueError("input_schema must be a dictionary")
|
||||
if "type" not in v:
|
||||
@@ -81,31 +150,216 @@ class AnthropicTool(BaseModel):
|
||||
return v
|
||||
|
||||
|
||||
class AnthropicWebSearchTool(BaseModel):
|
||||
"""Anthropic ``web_search_*`` server tool family.
|
||||
|
||||
No client-side ``input_schema`` — Anthropic provides the backing
|
||||
search implementation. Tag format is ``web_search_YYYYMMDD``.
|
||||
"""
|
||||
|
||||
type: str = Field(pattern=r"^web_search_\d{8}$")
|
||||
name: Literal["web_search"] = "web_search"
|
||||
description: Optional[str] = None
|
||||
defer_loading: Optional[bool] = None
|
||||
max_uses: Optional[int] = None
|
||||
allowed_domains: Optional[list[str]] = None
|
||||
blocked_domains: Optional[list[str]] = None
|
||||
|
||||
|
||||
class AnthropicComputerTool(BaseModel):
|
||||
"""Anthropic ``computer_*`` server tool family."""
|
||||
|
||||
type: str = Field(pattern=r"^computer_\d{8}$")
|
||||
name: Literal["computer"] = "computer"
|
||||
description: Optional[str] = None
|
||||
defer_loading: Optional[bool] = None
|
||||
display_width_px: Optional[int] = None
|
||||
display_height_px: Optional[int] = None
|
||||
display_number: Optional[int] = None
|
||||
|
||||
|
||||
class AnthropicBashTool(BaseModel):
|
||||
"""Anthropic ``bash_*`` server tool family."""
|
||||
|
||||
type: str = Field(pattern=r"^bash_\d{8}$")
|
||||
name: Literal["bash"] = "bash"
|
||||
description: Optional[str] = None
|
||||
defer_loading: Optional[bool] = None
|
||||
|
||||
|
||||
class AnthropicTextEditorTool(BaseModel):
|
||||
"""Anthropic ``text_editor_*`` server tool family."""
|
||||
|
||||
type: str = Field(pattern=r"^text_editor_\d{8}$")
|
||||
name: Literal["str_replace_editor", "str_replace_based_edit_tool"]
|
||||
description: Optional[str] = None
|
||||
defer_loading: Optional[bool] = None
|
||||
|
||||
|
||||
def _tool_discriminator(v) -> str:
|
||||
"""Pick the right tool variant from a dict or model instance.
|
||||
|
||||
Pydantic discriminators don't accept ``None`` as a tag, and custom
|
||||
tools allow ``type`` to be absent. Map missing/``custom`` to
|
||||
``"custom"`` and prefix-match server-tool families.
|
||||
"""
|
||||
if isinstance(v, dict):
|
||||
t = v.get("type")
|
||||
else:
|
||||
t = getattr(v, "type", None)
|
||||
if not t or t == "custom":
|
||||
return "custom"
|
||||
if t.startswith("web_search_"):
|
||||
return "web_search"
|
||||
if t.startswith("computer_"):
|
||||
return "computer"
|
||||
if t.startswith("bash_"):
|
||||
return "bash"
|
||||
if t.startswith("text_editor_"):
|
||||
return "text_editor"
|
||||
return "custom"
|
||||
|
||||
|
||||
AnthropicTool = Annotated[
|
||||
Union[
|
||||
Annotated[AnthropicCustomTool, Tag("custom")],
|
||||
Annotated[AnthropicWebSearchTool, Tag("web_search")],
|
||||
Annotated[AnthropicComputerTool, Tag("computer")],
|
||||
Annotated[AnthropicBashTool, Tag("bash")],
|
||||
Annotated[AnthropicTextEditorTool, Tag("text_editor")],
|
||||
],
|
||||
Discriminator(_tool_discriminator),
|
||||
]
|
||||
|
||||
|
||||
def is_server_tool(tool) -> bool:
|
||||
"""Return True for Anthropic built-in server-side tools."""
|
||||
return isinstance(
|
||||
tool,
|
||||
(
|
||||
AnthropicWebSearchTool,
|
||||
AnthropicComputerTool,
|
||||
AnthropicBashTool,
|
||||
AnthropicTextEditorTool,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
class AnthropicToolChoice(BaseModel):
|
||||
"""Tool Choice definition"""
|
||||
"""Tool choice strategy."""
|
||||
|
||||
type: Literal["auto", "any", "tool", "none"]
|
||||
name: Optional[str] = None
|
||||
|
||||
|
||||
class AnthropicThinkingParam(BaseModel):
|
||||
"""Anthropic extended-thinking control on the request.
|
||||
|
||||
Mirrors the Anthropic SDK's ``ThinkingConfigParam`` discriminated
|
||||
union of three variants — see ``anthropic-sdk-python``'s
|
||||
``thinking_config_{enabled,disabled,adaptive}_param.py``:
|
||||
|
||||
* ``enabled`` requires ``budget_tokens`` (≥1024) and accepts
|
||||
``display``.
|
||||
* ``disabled`` accepts no other fields.
|
||||
* ``adaptive`` (Claude 4.7) accepts ``display`` but not
|
||||
``budget_tokens``.
|
||||
|
||||
The serving layer treats ``adaptive`` identically to ``enabled``
|
||||
because the local OpenAI-compatible backend has no auto-throttle
|
||||
equivalent. ``budget_tokens`` is accepted on ``enabled`` for SDK
|
||||
compatibility but the backend has no hard-cap knob to honor it; the
|
||||
serving layer logs a WARNING so operators see that the requested
|
||||
budget is not enforced. ``display="omitted"`` is accepted but
|
||||
similarly cannot suppress reasoning mid-stream and is logged.
|
||||
"""
|
||||
|
||||
type: Literal["enabled", "disabled", "adaptive"]
|
||||
budget_tokens: Optional[int] = None
|
||||
display: Optional[Literal["summarized", "omitted"]] = None
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _validate_thinking_shape(self):
|
||||
# Cross-field rules mirror the SDK's three discriminated variants.
|
||||
if self.type == "enabled":
|
||||
if self.budget_tokens is None:
|
||||
raise ValueError(
|
||||
"thinking.budget_tokens is required when "
|
||||
"thinking.type is 'enabled'"
|
||||
)
|
||||
if self.budget_tokens < 1024:
|
||||
raise ValueError(
|
||||
"thinking.budget_tokens must be >= 1024 "
|
||||
"(got {})".format(self.budget_tokens)
|
||||
)
|
||||
elif self.type == "disabled":
|
||||
if self.budget_tokens is not None:
|
||||
raise ValueError(
|
||||
"thinking.budget_tokens is not allowed when "
|
||||
"thinking.type is 'disabled'"
|
||||
)
|
||||
if self.display is not None:
|
||||
raise ValueError(
|
||||
"thinking.display is not allowed when "
|
||||
"thinking.type is 'disabled'"
|
||||
)
|
||||
elif self.type == "adaptive":
|
||||
if self.budget_tokens is not None:
|
||||
raise ValueError(
|
||||
"thinking.budget_tokens is not allowed when "
|
||||
"thinking.type is 'adaptive'"
|
||||
)
|
||||
return self
|
||||
|
||||
|
||||
class AnthropicTaskBudget(BaseModel):
|
||||
"""Claude 4.7 ``output_config.task_budget`` — soft hint, not a hard cap.
|
||||
|
||||
Mirrors ``BetaTokenTaskBudgetParam`` in the Anthropic SDK: ``total``
|
||||
and ``type`` are required; ``remaining`` is the client-tracked
|
||||
countdown used for compaction. The hard cap on generation is still
|
||||
``max_tokens``; we never enforce ``task_budget`` ourselves.
|
||||
"""
|
||||
|
||||
type: Literal["tokens"]
|
||||
total: int = Field(gt=0)
|
||||
remaining: Optional[int] = Field(default=None, ge=0)
|
||||
|
||||
|
||||
class AnthropicOutputConfig(BaseModel):
|
||||
"""Claude 4.7 ``output_config`` block.
|
||||
|
||||
``effort`` maps to the OpenAI ``reasoning_effort`` knob (``xhigh`` →
|
||||
``max`` because the OpenAI Literal does not include ``xhigh``).
|
||||
``task_budget`` is propagated as a custom-param hint.
|
||||
"""
|
||||
|
||||
effort: Optional[Literal["low", "medium", "high", "xhigh", "max"]] = None
|
||||
task_budget: Optional[AnthropicTaskBudget] = None
|
||||
|
||||
|
||||
class AnthropicCountTokensRequest(BaseModel):
|
||||
"""Anthropic Count Tokens API request"""
|
||||
"""Anthropic count_tokens API request."""
|
||||
|
||||
model: str
|
||||
messages: list[AnthropicMessage]
|
||||
system: Optional[str | list[AnthropicContentBlock]] = None
|
||||
system: Optional[Union[str, list[AnthropicContentBlock]]] = None
|
||||
thinking: Optional[AnthropicThinkingParam] = None
|
||||
tool_choice: Optional[AnthropicToolChoice] = None
|
||||
tools: Optional[list[AnthropicTool]] = None
|
||||
# Claude 4.7 / SDK-compatibility fields. Accepted but no-op on count.
|
||||
output_config: Optional[AnthropicOutputConfig] = None
|
||||
betas: Optional[list[str]] = None
|
||||
|
||||
|
||||
class AnthropicCountTokensResponse(BaseModel):
|
||||
"""Anthropic Count Tokens API response"""
|
||||
"""Anthropic count_tokens API response."""
|
||||
|
||||
input_tokens: int
|
||||
|
||||
|
||||
class AnthropicMessagesRequest(BaseModel):
|
||||
"""Anthropic Messages API request"""
|
||||
"""Anthropic Messages API request."""
|
||||
|
||||
model: str
|
||||
messages: list[AnthropicMessage]
|
||||
@@ -113,65 +367,139 @@ class AnthropicMessagesRequest(BaseModel):
|
||||
metadata: Optional[dict[str, Any]] = None
|
||||
stop_sequences: Optional[list[str]] = None
|
||||
stream: Optional[bool] = False
|
||||
system: Optional[str | list[AnthropicContentBlock]] = None
|
||||
system: Optional[Union[str, list[AnthropicContentBlock]]] = None
|
||||
temperature: Optional[float] = None
|
||||
thinking: Optional[AnthropicThinkingParam] = None
|
||||
tool_choice: Optional[AnthropicToolChoice] = None
|
||||
tools: Optional[list[AnthropicTool]] = None
|
||||
top_k: Optional[int] = None
|
||||
top_p: Optional[float] = None
|
||||
# Claude 4.7 fields. The Anthropic SDK / Claude Code attach these even
|
||||
# when targeting non-Anthropic backends, so the schema must accept them.
|
||||
output_config: Optional[AnthropicOutputConfig] = None
|
||||
betas: Optional[list[str]] = None
|
||||
|
||||
@field_validator("model")
|
||||
@classmethod
|
||||
def validate_model(cls, v):
|
||||
def _validate_model(cls, v):
|
||||
if not v:
|
||||
raise ValueError("Model is required")
|
||||
return v
|
||||
|
||||
@field_validator("max_tokens")
|
||||
@classmethod
|
||||
def validate_max_tokens(cls, v):
|
||||
def _validate_max_tokens(cls, v):
|
||||
if v <= 0:
|
||||
raise ValueError("max_tokens must be positive")
|
||||
return v
|
||||
|
||||
|
||||
class AnthropicDelta(BaseModel):
|
||||
"""Delta for streaming responses"""
|
||||
# ---------- Stream deltas ----------
|
||||
# Content-block deltas (discriminated by ``type``) vs message-end delta
|
||||
# (separate model; the wire format does not put ``type`` inside its payload).
|
||||
|
||||
type: Optional[Literal["text_delta", "input_json_delta"]] = None
|
||||
text: Optional[str] = None
|
||||
partial_json: Optional[str] = None
|
||||
|
||||
# Message delta fields
|
||||
class TextDelta(BaseModel):
|
||||
type: Literal["text_delta"] = "text_delta"
|
||||
text: str
|
||||
|
||||
|
||||
class InputJsonDelta(BaseModel):
|
||||
type: Literal["input_json_delta"] = "input_json_delta"
|
||||
partial_json: str
|
||||
|
||||
|
||||
class ThinkingDelta(BaseModel):
|
||||
type: Literal["thinking_delta"] = "thinking_delta"
|
||||
thinking: str
|
||||
|
||||
|
||||
class SignatureDelta(BaseModel):
|
||||
type: Literal["signature_delta"] = "signature_delta"
|
||||
signature: str
|
||||
|
||||
|
||||
AnthropicContentDelta = Annotated[
|
||||
Union[TextDelta, InputJsonDelta, ThinkingDelta, SignatureDelta],
|
||||
Field(discriminator="type"),
|
||||
]
|
||||
|
||||
|
||||
class AnthropicMessageEndDelta(BaseModel):
|
||||
"""Delta carried on ``message_delta`` events.
|
||||
|
||||
Anthropic's protocol does NOT put a ``type`` field inside this delta
|
||||
payload — the SSE ``event:`` header already says ``message_delta``.
|
||||
Stop reason and stop sequence are the only fields.
|
||||
"""
|
||||
|
||||
stop_reason: Optional[
|
||||
Literal["end_turn", "max_tokens", "stop_sequence", "tool_use"]
|
||||
] = None
|
||||
stop_sequence: Optional[str] = None
|
||||
|
||||
|
||||
class AnthropicStreamEvent(BaseModel):
|
||||
"""Streaming event"""
|
||||
# ---------- Stream events (discriminated by ``type``) ----------
|
||||
|
||||
type: Literal[
|
||||
"message_start",
|
||||
"message_delta",
|
||||
"message_stop",
|
||||
"content_block_start",
|
||||
"content_block_delta",
|
||||
"content_block_stop",
|
||||
"ping",
|
||||
"error",
|
||||
]
|
||||
message: Optional["AnthropicMessagesResponse"] = None
|
||||
delta: Optional[AnthropicDelta] = None
|
||||
content_block: Optional[AnthropicContentBlock] = None
|
||||
index: Optional[int] = None
|
||||
error: Optional[AnthropicError] = None
|
||||
usage: Optional[AnthropicUsage] = None
|
||||
|
||||
class MessageStartEvent(BaseModel):
|
||||
type: Literal["message_start"] = "message_start"
|
||||
message: "AnthropicMessagesResponse"
|
||||
|
||||
|
||||
class MessageDeltaEvent(BaseModel):
|
||||
type: Literal["message_delta"] = "message_delta"
|
||||
delta: AnthropicMessageEndDelta
|
||||
usage: AnthropicUsage
|
||||
|
||||
|
||||
class MessageStopEvent(BaseModel):
|
||||
type: Literal["message_stop"] = "message_stop"
|
||||
|
||||
|
||||
class ContentBlockStartEvent(BaseModel):
|
||||
type: Literal["content_block_start"] = "content_block_start"
|
||||
index: int
|
||||
content_block: AnthropicContentBlock
|
||||
|
||||
|
||||
class ContentBlockDeltaEvent(BaseModel):
|
||||
type: Literal["content_block_delta"] = "content_block_delta"
|
||||
index: int
|
||||
delta: AnthropicContentDelta
|
||||
|
||||
|
||||
class ContentBlockStopEvent(BaseModel):
|
||||
type: Literal["content_block_stop"] = "content_block_stop"
|
||||
index: int
|
||||
|
||||
|
||||
class PingEvent(BaseModel):
|
||||
type: Literal["ping"] = "ping"
|
||||
|
||||
|
||||
class ErrorEvent(BaseModel):
|
||||
type: Literal["error"] = "error"
|
||||
error: AnthropicError
|
||||
|
||||
|
||||
AnthropicStreamEvent = Annotated[
|
||||
Union[
|
||||
MessageStartEvent,
|
||||
MessageDeltaEvent,
|
||||
MessageStopEvent,
|
||||
ContentBlockStartEvent,
|
||||
ContentBlockDeltaEvent,
|
||||
ContentBlockStopEvent,
|
||||
PingEvent,
|
||||
ErrorEvent,
|
||||
],
|
||||
Field(discriminator="type"),
|
||||
]
|
||||
|
||||
|
||||
class AnthropicMessagesResponse(BaseModel):
|
||||
"""Anthropic Messages API response"""
|
||||
"""Anthropic Messages API response."""
|
||||
|
||||
id: str = Field(default_factory=lambda: f"msg_{uuid.uuid4().hex}")
|
||||
type: Literal["message"] = "message"
|
||||
@@ -183,3 +511,8 @@ class AnthropicMessagesResponse(BaseModel):
|
||||
] = None
|
||||
stop_sequence: Optional[str] = None
|
||||
usage: Optional[AnthropicUsage] = None
|
||||
|
||||
|
||||
# Resolve forward references for nested types.
|
||||
ToolResultBlock.model_rebuild()
|
||||
MessageStartEvent.model_rebuild()
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -425,13 +425,77 @@ from sglang.srt.entrypoints.v1_loads import router as v1_loads_router
|
||||
app.include_router(v1_loads_router)
|
||||
|
||||
|
||||
def _anthropic_validation_message(raw_errors) -> str:
|
||||
"""Render Pydantic-style errors for an Anthropic /v1/messages route.
|
||||
|
||||
Builds a short ``loc: msg`` digest that names the offending fields without
|
||||
leaking file paths or Python internals (the default ``str(exc)`` includes
|
||||
the dispatcher's ``File "/.../http_server.py"`` line).
|
||||
"""
|
||||
parts: list[str] = []
|
||||
for err in raw_errors or []:
|
||||
loc = err.get("loc") or ()
|
||||
if loc:
|
||||
loc_str = ".".join(str(p) for p in loc if p not in ("body",))
|
||||
else:
|
||||
loc_str = ""
|
||||
msg = (err.get("msg") or "").strip()
|
||||
if loc_str and msg:
|
||||
parts.append(f"{loc_str}: {msg}")
|
||||
elif msg:
|
||||
parts.append(msg)
|
||||
text = "; ".join(parts) or "Invalid request"
|
||||
if len(text) > 500:
|
||||
text = text[:500] + "…"
|
||||
return text
|
||||
|
||||
|
||||
def _anthropic_error_response(*, status_code: int, error_type: str, message: str):
|
||||
"""Anthropic-format error envelope: {"type":"error","error":{"type":...,"message":...}}."""
|
||||
return ORJSONResponse(
|
||||
status_code=status_code,
|
||||
content={
|
||||
"type": "error",
|
||||
"error": {"type": error_type, "message": message},
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
@app.exception_handler(HTTPException)
|
||||
async def validation_exception_handler(request: Request, exc: HTTPException):
|
||||
"""Enrich HTTP exception with status code and other details.
|
||||
|
||||
For /v1/responses, emit OpenAI-style nested error envelope:
|
||||
{"error": {"message": "...", "type": "...", "param": null, "code": <status>}}
|
||||
For /v1/messages, emit Anthropic-style envelope so SDK clients can parse it.
|
||||
"""
|
||||
if request.url.path.startswith("/v1/messages"):
|
||||
# Map HTTP status to Anthropic error.type; fall back to api_error.
|
||||
anthropic_type = {
|
||||
400: "invalid_request_error",
|
||||
401: "authentication_error",
|
||||
403: "permission_error",
|
||||
404: "not_found_error",
|
||||
413: "request_too_large",
|
||||
422: "invalid_request_error",
|
||||
429: "rate_limit_error",
|
||||
500: "api_error",
|
||||
502: "api_error",
|
||||
503: "overloaded_error",
|
||||
504: "api_error",
|
||||
}.get(exc.status_code, "api_error")
|
||||
# 5xx must never echo upstream detail (may contain stack/PII).
|
||||
message = (
|
||||
"Internal server error"
|
||||
if exc.status_code >= 500
|
||||
else (str(exc.detail) if exc.detail else "Request failed")
|
||||
)
|
||||
return _anthropic_error_response(
|
||||
status_code=exc.status_code,
|
||||
error_type=anthropic_type,
|
||||
message=message,
|
||||
)
|
||||
|
||||
# adjust fmt for responses api
|
||||
if request.url.path.startswith("/v1/responses"):
|
||||
nested_error = {
|
||||
@@ -458,8 +522,18 @@ async def validation_exception_handler(request: Request, exc: HTTPException):
|
||||
async def validation_exception_handler(request: Request, exc: RequestValidationError):
|
||||
"""Override FastAPI's default 422 validation error with 400.
|
||||
|
||||
For /v1/responses, emit OpenAI-style nested error envelope; for other endpoints keep legacy format.
|
||||
For /v1/messages, emit Anthropic-style envelope and scrub the message so
|
||||
file paths or Python internals from the default ``str(exc)`` representation
|
||||
never reach the client. For /v1/responses, keep OpenAI-style. Otherwise
|
||||
use the legacy ErrorResponse shape.
|
||||
"""
|
||||
if request.url.path.startswith("/v1/messages"):
|
||||
return _anthropic_error_response(
|
||||
status_code=HTTPStatus.BAD_REQUEST.value,
|
||||
error_type="invalid_request_error",
|
||||
message=_anthropic_validation_message(exc.errors()),
|
||||
)
|
||||
|
||||
exc_str = str(exc)
|
||||
errors_str = str(exc.errors())
|
||||
|
||||
@@ -1067,9 +1141,11 @@ async def dump_expert_distribution_record_async():
|
||||
@auth_level(AuthLevel.ADMIN_OPTIONAL)
|
||||
async def update_weights_from_disk(obj: UpdateWeightFromDiskReqInput, request: Request):
|
||||
"""Update the weights from disk inplace without re-launching the server."""
|
||||
success, message, num_paused_requests = (
|
||||
await _global_state.tokenizer_manager.update_weights_from_disk(obj, request)
|
||||
)
|
||||
(
|
||||
success,
|
||||
message,
|
||||
num_paused_requests,
|
||||
) = await _global_state.tokenizer_manager.update_weights_from_disk(obj, request)
|
||||
|
||||
content = {
|
||||
"success": success,
|
||||
@@ -1093,10 +1169,11 @@ async def update_weights_from_disk(obj: UpdateWeightFromDiskReqInput, request: R
|
||||
async def init_weights_send_group_for_remote_instance(
|
||||
obj: InitWeightsSendGroupForRemoteInstanceReqInput, request: Request
|
||||
):
|
||||
success, message = (
|
||||
await _global_state.tokenizer_manager.init_weights_send_group_for_remote_instance(
|
||||
obj, request
|
||||
)
|
||||
(
|
||||
success,
|
||||
message,
|
||||
) = await _global_state.tokenizer_manager.init_weights_send_group_for_remote_instance(
|
||||
obj, request
|
||||
)
|
||||
content = {"success": success, "message": message}
|
||||
if success:
|
||||
@@ -1110,10 +1187,11 @@ async def init_weights_send_group_for_remote_instance(
|
||||
async def send_weights_to_remote_instance(
|
||||
obj: SendWeightsToRemoteInstanceReqInput, request: Request
|
||||
):
|
||||
success, message = (
|
||||
await _global_state.tokenizer_manager.send_weights_to_remote_instance(
|
||||
obj, request
|
||||
)
|
||||
(
|
||||
success,
|
||||
message,
|
||||
) = await _global_state.tokenizer_manager.send_weights_to_remote_instance(
|
||||
obj, request
|
||||
)
|
||||
content = {"success": success, "message": message}
|
||||
if success:
|
||||
@@ -1182,9 +1260,10 @@ async def destroy_weights_update_group(
|
||||
obj: DestroyWeightsUpdateGroupReqInput, request: Request
|
||||
):
|
||||
"""Destroy the parameter update group."""
|
||||
success, message = (
|
||||
await _global_state.tokenizer_manager.destroy_weights_update_group(obj, request)
|
||||
)
|
||||
(
|
||||
success,
|
||||
message,
|
||||
) = await _global_state.tokenizer_manager.destroy_weights_update_group(obj, request)
|
||||
content = {"success": success, "message": message}
|
||||
return ORJSONResponse(
|
||||
content, status_code=200 if success else HTTPStatus.BAD_REQUEST
|
||||
@@ -1219,10 +1298,11 @@ async def update_weights_from_distributed(
|
||||
obj: UpdateWeightsFromDistributedReqInput, request: Request
|
||||
):
|
||||
"""Update model parameter from distributed online."""
|
||||
success, message = (
|
||||
await _global_state.tokenizer_manager.update_weights_from_distributed(
|
||||
obj, request
|
||||
)
|
||||
(
|
||||
success,
|
||||
message,
|
||||
) = await _global_state.tokenizer_manager.update_weights_from_distributed(
|
||||
obj, request
|
||||
)
|
||||
|
||||
content = {"success": success, "message": message}
|
||||
|
||||
@@ -1564,6 +1564,111 @@ class OpenAIServingChat(OpenAIServingBase):
|
||||
):
|
||||
request.skip_special_tokens = False
|
||||
|
||||
def wrap_reasoning_history(self, reasoning_text: str) -> str:
|
||||
"""Wrap prior-turn reasoning in the detector's own start/end tokens.
|
||||
|
||||
Pulling the delimiters from the detector keeps adapters in lockstep
|
||||
with any future parser that ships non-``<think>`` markers — Mistral's
|
||||
``[THINK]``, Gemma4's ``think_start_self_label = "thought\\n"``, etc.
|
||||
Falling back to a plain string is unsafe: it would let prior
|
||||
thinking text reach a non-reasoning model as ordinary assistant
|
||||
content, so the caller must surface this state, not paper over it.
|
||||
"""
|
||||
if self._reasoning_detector is None:
|
||||
raise ValueError(
|
||||
"Cannot rewrap thinking history: no reasoning detector is "
|
||||
"configured for this model"
|
||||
)
|
||||
d = self._reasoning_detector
|
||||
return (
|
||||
f"{d.think_start_token}{d.think_start_self_label}"
|
||||
f"{reasoning_text}\n{d.think_end_token}"
|
||||
)
|
||||
|
||||
def _reasoning_default_mode(self) -> Optional[str]:
|
||||
if self._reasoning_detector is None:
|
||||
return None
|
||||
return self._reasoning_detector.reasoning_default
|
||||
|
||||
def _get_reasoning_toggle_param(self) -> Optional[str]:
|
||||
"""Resolve the chat-template kwarg that toggles reasoning, if any."""
|
||||
config = self.template_manager.reasoning_config
|
||||
if config is not None:
|
||||
return config.toggle_param
|
||||
|
||||
mode = self._reasoning_default_mode()
|
||||
if mode in ("thinking", "enable_thinking"):
|
||||
return mode
|
||||
if mode in ("explicit_thinking", "explicit_enable_thinking"):
|
||||
return mode.replace("explicit_", "")
|
||||
return None
|
||||
|
||||
def apply_reasoning_enabled(
|
||||
self, request: ChatCompletionRequest, enabled: bool
|
||||
) -> None:
|
||||
"""Force the request into the requested reasoning-on/off mode.
|
||||
|
||||
Mirrors the read-side logic in ``_get_reasoning_from_request``;
|
||||
the two must stay in sync. Always-on models cannot be disabled,
|
||||
so explicit ``enabled=False`` raises rather than silently leaving
|
||||
reasoning on.
|
||||
"""
|
||||
if not self.reasoning_parser:
|
||||
if enabled:
|
||||
raise ValueError(
|
||||
"Anthropic thinking is not supported for models without "
|
||||
"a reasoning parser"
|
||||
)
|
||||
return
|
||||
|
||||
if self.reasoning_parser == "hunyuan":
|
||||
request.reasoning_effort = "medium" if enabled else "no_think"
|
||||
return
|
||||
|
||||
config = self.template_manager.reasoning_config
|
||||
is_mistral = (config is not None and config.special_case == "mistral") or (
|
||||
config is None and self._reasoning_default_mode() == "mistral"
|
||||
)
|
||||
if is_mistral:
|
||||
request.reasoning_effort = "medium" if enabled else "none"
|
||||
return
|
||||
|
||||
is_always_on = (config is not None and config.special_case == "always") or (
|
||||
config is None and self._reasoning_default_mode() == "always"
|
||||
)
|
||||
if is_always_on:
|
||||
if not enabled:
|
||||
raise ValueError(
|
||||
f"Reasoning parser '{self.reasoning_parser}' is always-on "
|
||||
f"and cannot be disabled via Anthropic thinking"
|
||||
)
|
||||
return
|
||||
|
||||
toggle_param = self._get_reasoning_toggle_param()
|
||||
# The read side (``_get_reasoning_from_request``) returns False
|
||||
# whenever ``config.toggle_param is None`` OR
|
||||
# ``config.default_enabled is None``. The write side must mirror
|
||||
# both conditions: if ``default_enabled`` is unset we cannot
|
||||
# actually honor an ``enabled=True`` request even when the toggle
|
||||
# name itself is resolvable, so writing the kwarg would set up the
|
||||
# template to emit reasoning tokens while the parser ignores them
|
||||
# (literal ``<think>`` markers leak into the assistant text).
|
||||
config = self.template_manager.reasoning_config
|
||||
read_side_supported = toggle_param is not None and (
|
||||
config is None or config.default_enabled is not None
|
||||
)
|
||||
if not read_side_supported:
|
||||
if not enabled:
|
||||
return
|
||||
raise ValueError(
|
||||
f"Anthropic thinking is not supported for reasoning parser "
|
||||
f"'{self.reasoning_parser}'"
|
||||
)
|
||||
|
||||
chat_template_kwargs = dict(request.chat_template_kwargs or {})
|
||||
chat_template_kwargs[toggle_param] = enabled
|
||||
request.chat_template_kwargs = chat_template_kwargs
|
||||
|
||||
def _get_reasoning_from_request(self, request: ChatCompletionRequest) -> bool:
|
||||
"""Determine whether reasoning mode should be enabled for this request.
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user