Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 12 additions & 2 deletions app/adapters/anthropic_adapter.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
import time
from typing import Any

from app.reasoning import extract_reasoning_text, map_reasoning_controls
from app.upstream_io import (StreamOutputBudget, merge_tool_call_delta, new_tool_state,
seal_tool_identities, tool_identity_complete)

Expand Down Expand Up @@ -93,6 +94,7 @@ def anthropic_request_to_chat(body: dict) -> dict:
raise ValueError("stop_sequences must be an array of strings")
chat["stop"] = stop_sequences

map_reasoning_controls(body, chat, protocol="messages")
return chat


Expand Down Expand Up @@ -124,6 +126,9 @@ def _convert_anthropic_message(msg: dict) -> list[dict]:

# Structured content blocks
blocks = content
if role != "assistant" and any(isinstance(block, dict) and block.get("type") in
("thinking", "redacted_thinking") for block in blocks):
raise ValueError("thinking requires an assistant message")

# User messages may contain tool results.
if role == "user":
Expand Down Expand Up @@ -157,11 +162,14 @@ def _convert_anthropic_message(msg: dict) -> list[dict]:
if role == "assistant":
content_out = _convert_content_blocks(blocks)
tool_calls: list[dict] = []
thoughts: list[str] = []
for block in blocks:
if not isinstance(block, dict):
continue
bt = block.get("type", "")
if bt == "tool_use":
thought = extract_reasoning_text(block)
if thought is not None:
thoughts.append(thought)
elif block.get("type") == "tool_use":
tc = {
"id": block.get("id", _rand_id("call_")),
"type": "function",
Expand All @@ -176,6 +184,8 @@ def _convert_anthropic_message(msg: dict) -> list[dict]:
msg_out["content"] = content_out if content_out or has_text else None
if tool_calls:
msg_out["tool_calls"] = tool_calls
if thoughts:
msg_out["reasoning_content"] = "".join(thoughts)
return [msg_out]

content_out = _convert_content_blocks(blocks)
Expand Down
14 changes: 7 additions & 7 deletions app/adapters/chat_input.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,8 @@

from fastapi import HTTPException

from app.reasoning import ReasoningInputError, extract_reasoning_text


_ANTHROPIC_BLOCKS = ("tool_use", "tool_result", "thinking", "redacted_thinking", "image")

Expand Down Expand Up @@ -116,17 +118,15 @@ def _convert_message(message, index, pending):
raise _invalid(location + ".tool_use_id", "tool_result must match one preceding, unanswered tool call")
result_ids.add(identifier)
results.append(result)
elif kind == "thinking":
elif kind in ("thinking", "redacted_thinking"):
if role != "assistant":
raise _invalid(location, "thinking requires an assistant message")
if message.get("reasoning_content") not in (None, ""):
raise _invalid(location, "thinking conflicts with existing reasoning_content")
if not isinstance(block.get("thinking"), str):
raise _invalid(location + ".thinking", "Thinking content must be a string")
# Anthropic signatures have no Chat equivalent and must not become visible text.
thoughts.append(block["thinking"])
elif kind == "redacted_thinking":
raise _invalid(location, "redacted_thinking cannot be converted to Chat; use the Messages protocol")
try:
thoughts.append(extract_reasoning_text(block))
except ReasoningInputError as error:
raise _invalid(location + ("." + error.field if error.field else ""), str(error)) from None
else:
parts.append(_content_part(block, location))
if results:
Expand Down
43 changes: 32 additions & 11 deletions app/adapters/responses_adapter.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
import time
from typing import Any

from app.reasoning import extract_reasoning_text, map_reasoning_controls
from app.upstream_io import (StreamOutputBudget, merge_tool_call_delta, new_tool_state,
seal_tool_identities, tool_identity_complete)

Expand Down Expand Up @@ -78,16 +79,12 @@ def responses_request_to_chat(body: dict) -> dict:
# Forward supported parameters.
for key in ("temperature", "top_p", "stop", "seed",
"presence_penalty", "frequency_penalty",
"response_format", "reasoning_effort", "parallel_tool_calls", "prompt_cache_key"):
"response_format", "parallel_tool_calls", "prompt_cache_key"):
if key in body:
chat[key] = body[key]

# Explicit top-level values override equivalent nested fields.
reasoning = body.get("reasoning")
if isinstance(reasoning, dict) and "reasoning_effort" not in chat:
effort = reasoning.get("effort")
if isinstance(effort, str) and effort.strip():
chat["reasoning_effort"] = effort
map_reasoning_controls(body, chat, protocol="responses")
text = body.get("text")
if isinstance(text, dict) and "response_format" not in chat:
mapped = _text_format_to_response_format(text.get("format"))
Expand All @@ -108,17 +105,36 @@ def _convert_input_items(items: list) -> list[dict]:
# Buffer adjacent assistant text and function calls.
pending_assistant_content: str | list[dict] | None = None
pending_tool_calls: list[dict] = []
pending_reasoning: list[str] = []

def _flush_assistant():
nonlocal pending_assistant_content, pending_tool_calls
if pending_assistant_content is not None or pending_tool_calls:
if pending_assistant_content is not None or pending_tool_calls or pending_reasoning:
msg: dict[str, Any] = {"role": "assistant",
"content": pending_assistant_content or ""}
if pending_tool_calls:
msg["tool_calls"] = pending_tool_calls[:]
if pending_reasoning:
msg["reasoning_content"] = "".join(pending_reasoning)
messages.append(msg)
pending_assistant_content = None
pending_tool_calls.clear()
pending_reasoning.clear()

def _set_assistant_content(content):
nonlocal pending_assistant_content
if pending_assistant_content is not None and not pending_tool_calls:
_flush_assistant()
# Realtime output can place text after a tool call within the same turn.
if pending_tool_calls and pending_assistant_content:
if isinstance(pending_assistant_content, str) and isinstance(content, str):
content = pending_assistant_content + content
else:
previous = (pending_assistant_content if isinstance(pending_assistant_content, list)
else [{"type": "text", "text": pending_assistant_content}])
following = content if isinstance(content, list) else [{"type": "text", "text": content}]
content = previous + following
pending_assistant_content = content

for item in items:
if not isinstance(item, dict):
Expand All @@ -127,6 +143,13 @@ def _flush_assistant():
item_type = item.get("type")
role = item.get("role", "")

if item_type == "reasoning":
if role not in ("", "assistant"):
raise ValueError("reasoning requires an assistant item")
text = extract_reasoning_text(item)
pending_reasoning.append(text)
continue

# Untyped role messages
if item_type is None and role in ("user", "system", "developer"):
_flush_assistant()
Expand All @@ -145,17 +168,15 @@ def _flush_assistant():

# Assistant output from history
if item_type == "message" and role == "assistant":
_flush_assistant()
content_parts = item.get("content", [])
text = _extract_output_text(content_parts) if isinstance(content_parts, list) else str(content_parts)
pending_assistant_content = text
_set_assistant_content(text)
continue

# Untyped assistant messages
if item_type is None and role == "assistant":
_flush_assistant()
content = _extract_content(item.get("content", ""))
pending_assistant_content = content
_set_assistant_content(content)
continue

# Merge calls into the preceding assistant message.
Expand Down
13 changes: 7 additions & 6 deletions app/model_capabilities.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
from app.safe_logging import sanitize_log_text

from app.model_catalog_view import SharedModel
from app.reasoning import resolve_reasoning_effort, thinking_mode
PROFILES = frozenset(("cn-cli", "cn-work", "intl-cli", "intl-work"))
_TEXT = frozenset(("id", "name", "vendor", "description", "descriptionZh", "descriptionEn", "credits", "summary"))
_BOOL = frozenset(("supportsImages", "disabledMultimodal", "supportsToolCall", "supportsReasoning",
Expand Down Expand Up @@ -209,8 +210,7 @@ def from_request(cls, body, payload=None, protocol="chat"):
tools = bool(body.get("tools")) or any(isinstance(message, dict) and
(message.get("role") == "tool" or message.get("tool_calls")) for message in messages)
effort = body.get("reasoning_effort")
thinking = (payload or {}).get("thinking") if protocol == "messages" else None
thinking = thinking.get("type") if isinstance(thinking, dict) else None
thinking = thinking_mode(payload or {}) if protocol == "messages" else None
output = body.get("max_tokens")
param = "max_output_tokens" if protocol == "responses" else "max_tokens"
if output is not None and _positive(output) is None:
Expand All @@ -227,16 +227,17 @@ def violations(self, model):
failures.append(("unsupported_image_input", self.image_param, "当前路由的模型声明不支持图片输入"))
if self.tools and caps["tools"] is False:
failures.append(("unsupported_tools", "tools", "当前路由的模型声明不支持工具调用"))
enabled = self.effort not in (None, "none") or self.thinking in ("enabled", "adaptive")
disabled = self.effort == "none" or self.thinking == "disabled"
effort = resolve_reasoning_effort(self.effort, self.thinking, model)
enabled = effort not in (None, "none")
disabled = effort == "none"
if enabled and caps["reasoning"] is False:
failures.append(("unsupported_reasoning", "reasoning_effort", "当前路由的模型声明不支持思考"))
if disabled and caps["thinking_disable"] is False:
failures.append(("reasoning_required", "reasoning_effort", "当前路由的模型声明不能关闭思考"))
reasoning = model.get("reasoning") if isinstance(model.get("reasoning"), dict) else {}
efforts = reasoning.get("supportedEfforts")
if (self.effort not in (None, "none") and isinstance(efforts, list) and (efforts or isinstance(model, SharedModel))
and all(isinstance(value, str) for value in efforts) and self.effort not in efforts):
if (effort not in (None, "none") and isinstance(efforts, list) and (efforts or isinstance(model, SharedModel))
and all(isinstance(value, str) for value in efforts) and effort not in efforts):
failures.append(("unsupported_reasoning_effort", "reasoning_effort", "思考强度不在当前模型声明的选项中"))
maximum = _positive(model.get("maxOutputTokens"))
if maximum is not None and self.max_output is not None and self.max_output > maximum:
Expand Down
121 changes: 121 additions & 0 deletions app/reasoning.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,121 @@
"""Normalize readable reasoning and request controls for Chat upstreams."""


class ReasoningInputError(ValueError):
"""Identify an unsupported reasoning field without exposing its contents."""

def __init__(self, message, field=""):
self.field = field
super().__init__(f"{field}: {message}" if field else message)


def _text_parts(parts, kinds, field):
if parts is None:
return []
if not isinstance(parts, list):
raise ReasoningInputError("must be an array", field)
texts = []
for index, part in enumerate(parts):
location = f"{field}[{index}]"
if not isinstance(part, dict) or part.get("type") not in kinds:
raise ReasoningInputError("unsupported reasoning text block", location)
if not isinstance(part.get("text"), str):
raise ReasoningInputError("must be a string", location + ".text")
texts.append(part["text"])
return texts


def extract_reasoning_text(block):
"""Return readable thinking without interpreting signatures or encrypted state."""
kind = block.get("type")
if kind == "redacted_thinking":
raise ReasoningInputError("encrypted reasoning cannot be converted to Chat", "data")
if kind == "thinking":
text = block.get("thinking")
if not isinstance(text, str):
raise ReasoningInputError("must be a string", "thinking")
if not text and block.get("signature") not in (None, ""):
raise ReasoningInputError("encrypted-only thinking cannot be converted to Chat", "signature")
return text
if kind == "reasoning":
if block.get("encrypted_content") not in (None, ""):
raise ReasoningInputError("encrypted reasoning cannot be converted to Chat", "encrypted_content")
content = _text_parts(block.get("content"), ("reasoning_text", "text"), "content")
summary = _text_parts(block.get("summary"), ("summary_text",), "summary")
return "".join(content if content else summary)
return None


def thinking_mode(body):
"""Read the already-validated Messages thinking mode for account selection."""
thinking = body.get("thinking")
return thinking.get("type") if isinstance(thinking, dict) else None


def _object(body, field):
value = body.get(field)
if value is None:
return {}
if not isinstance(value, dict):
raise ReasoningInputError("must be an object", field)
return value


def _effort(value, field, allowed=None):
if value is not None and (not isinstance(value, str) or not value.strip()
or (allowed is not None and value not in allowed)):
raise ReasoningInputError("unsupported reasoning effort", field)
return value


def map_reasoning_controls(body, chat, *, protocol):
"""Map explicit controls, leaving implicit activation to the selected account."""
explicit = "reasoning_effort" in body
effort = _effort(body.get("reasoning_effort"), "reasoning_effort")
if protocol == "responses":
reasoning = _object(body, "reasoning")
if not explicit and "effort" in reasoning:
effort = _effort(reasoning["effort"], "reasoning.effort")
explicit = True
elif protocol == "messages":
thinking = _object(body, "thinking")
mode = thinking_mode(body)
if body.get("thinking") is not None and mode not in ("enabled", "adaptive", "disabled"):
raise ReasoningInputError("must be enabled, adaptive or disabled", "thinking.type")
if mode == "enabled":
budget = thinking.get("budget_tokens")
if type(budget) is not int or budget < 1024:
raise ReasoningInputError("must be an integer >= 1024", "thinking.budget_tokens")
elif "budget_tokens" in thinking:
raise ReasoningInputError("requires thinking.type=enabled", "thinking.budget_tokens")
if thinking.get("display") not in (None, "summarized"):
raise ReasoningInputError("only summarized display is supported by Chat upstreams", "thinking.display")
output = _object(body, "output_config")
if not explicit and "effort" in output:
effort = _effort(output["effort"], "output_config.effort", ("low", "medium", "high", "xhigh", "max"))
explicit = True
if mode == "disabled":
effort, explicit = "none", True
else:
raise ValueError("unsupported reasoning protocol")
if explicit:
chat["reasoning_effort"] = effort


def resolve_reasoning_effort(effort, mode, model):
"""Use the same account-owned default for capability checks and upstream requests."""
if mode == "disabled":
return "none"
if effort is not None or mode not in ("enabled", "adaptive"):
return effort
model = model or {}
reasoning = model.get("reasoning") if isinstance(model.get("reasoning"), dict) else {}
supported = reasoning.get("supportedEfforts")
supported = supported if isinstance(supported, list) and supported and all(isinstance(v, str) for v in supported) else None
candidates = [reasoning.get("defaultEffort"), reasoning.get("effort"),
"high", "medium", "low", "xhigh", "max", "minimal", *(supported or [])]
for candidate in candidates:
if (isinstance(candidate, str) and candidate.strip() and candidate != "none"
and (supported is None or candidate in supported)):
return candidate
return "high"
16 changes: 12 additions & 4 deletions converter.py
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,7 @@ def desensitize_body(body, roles=("system",), desensitize_harness_user=False,
from app.inference_resources import (AccountCapacity, InferenceResourcesMiddleware, inference_lifespan,
request_resources, release_credential)
from app.request_context import SessionIdentifierError, current_context
from app.reasoning import resolve_reasoning_effort, thinking_mode
from app import model_capabilities
from app.message_normalization import merge_intl_user_images
from app.adapters.chat_input import normalize_chat_messages
Expand Down Expand Up @@ -1931,21 +1932,28 @@ def _cred_for(payload: dict, model: str | None = None, *, region=None, tried=(),
def _route_chat(payload, body, rid, *, tried=()):
"""Validate account capabilities and derive each routed body from canonical input."""
context = current_context()
protocol = context.protocol if context is not None else "chat"
mode = thinking_mode(payload) if protocol == "messages" else None
enabled = context.capability_guard if context is not None else CONFIG.get("model_capability_guard", True)
cred = None
try:
requirements = (model_capabilities.Requirements.from_request(
body, payload, context.protocol if context is not None else "chat") if enabled else None)
requirements = model_capabilities.Requirements.from_request(body, payload, protocol) if enabled else None
cred, headers = _cred_for(payload, body.get("model"), tried=tried, requirements=requirements)
profile = profile_for_headers(headers)
routed_model = _upstream_model(body.get("model"), profile)
if requirements is not None and CONFIG.get("cred_pool") is None:
metadata = None
if mode in ("enabled", "adaptive") or (requirements is not None and CONFIG.get("cred_pool") is None):
entry = {"profile": profile, "account_key": account_key(
profile, headers.get("X-User-Id"), headers.get("X-Enterprise-Id"))}
failures = requirements.violations(model_capabilities.entry_model(sys.modules[__name__], entry, body.get("model")))
metadata = model_capabilities.entry_model(sys.modules[__name__], entry, body.get("model"))
if requirements is not None and CONFIG.get("cred_pool") is None:
failures = requirements.violations(metadata)
if failures:
raise model_capabilities.capability_error(failures)
canonical = body
effort = resolve_reasoning_effort(body.get("reasoning_effort"), mode, metadata)
if effort != body.get("reasoning_effort"):
body = {**body, "reasoning_effort": effort}
if routed_model != body.get("model"):
body = {**body, "model": routed_model}
body, merged_runs, merged_messages = merge_intl_user_images(body, profile)
Expand Down
Loading
Loading