fix: drop replayed reasoning_details on every chat-completions route that does not read it

OpenRouter and the Nous Portal replay reasoning_details for multi-turn reasoning
continuity; every other OpenAI-compatible route either ignores the field or, when
its schema is strict (Groq, Mistral, Cerebras, opencode relays), rejects the whole
request with 400/422 once an earlier reasoning turn is in history — wedging the
session after an in-session model switch (#70233). Strip the field from the wire
copy in ChatCompletionsTransport.convert_messages (keyed on the target base_url),
mirror it in the auxiliary wire boundary and the iteration-summary path; state.db
history keeps the field so switching back to OpenRouter/Nous replays it again.
This commit is contained in:
teknium1
2026-09-19 09:28:45 -07:00
committed by Teknium
parent 0837f549be
commit c8ecc3db64
5 changed files with 56 additions and 8 deletions
+1 -1
View File
@@ -15,6 +15,6 @@ def prepare_chat_messages(client, kwargs: dict) -> dict:
if not isinstance(client, (OpenAI, AsyncOpenAI)) or "messages" not in kwargs:
return kwargs
messages = ChatCompletionsTransport().convert_messages(
kwargs["messages"], model=kwargs.get("model")
kwargs["messages"], model=kwargs.get("model"), base_url=str(getattr(client, "base_url", "") or ""),
)
return {**kwargs, "messages": messages}
+4
View File
@@ -2059,6 +2059,8 @@ _EMPTY_SUMMARY_RESPONSE = "I reached the iteration limit and couldn't generate a
def _iteration_summary_api_messages(agent, messages: list) -> list:
"""Wire-ready messages for the summary call, mirroring the main loop's api_messages build
(sidecar substitution, tool-call repair, thinking-only drop, underscore-key sweep)."""
from agent.transports.chat_completions import _route_replays_reasoning_details
needs_sanitize = agent._should_sanitize_tool_calls()
sanitize_model = agent.model
if needs_sanitize and agent.provider == "moa":
@@ -2071,6 +2073,8 @@ def _iteration_summary_api_messages(agent, messages: list) -> list:
agent._copy_reasoning_content_for_api(msg, api_msg)
for key in _SUMMARY_FOREIGN_MESSAGE_KEYS:
api_msg.pop(key, None)
if not _route_replays_reasoning_details(getattr(agent, "base_url", None)):
api_msg.pop("reasoning_details", None)
# Mirror of the transport's role-qualified strip: ``name`` is
# schema-foreign on tool results only (strict providers reject with
# "contains item with unknown key name"); it stays on user/assistant.
+25 -5
View File
@@ -216,6 +216,22 @@ def _model_consumes_thought_signature(model: Any) -> bool:
return "gemini" in m or "gemma" in m
def _route_replays_reasoning_details(base_url: Any) -> bool:
"""True when the target route reads replayed ``reasoning_details`` (OpenRouter's unified
reasoning array, also consumed by the Nous Portal).
Every other chat-completions endpoint either ignores the field or, when its schema is
strict (Groq, Mistral, Cerebras, opencode relays: ``property 'reasoning_details' is
unsupported`` / ``Extra inputs are not permitted`` / ``no such field``), rejects the whole
request with HTTP 400/422 — so a reasoning turn produced earlier in the session wedges every
later turn once the model is switched (#70233). The stored history keeps the field; only the
wire copy drops it.
"""
from utils import base_url_host_matches
return base_url_host_matches(base_url, "openrouter.ai") or base_url_host_matches(base_url, "nousresearch.com")
def _has_replayable_thought_signature(extra_content: Any) -> bool:
"""Whether OpenRouter's Gemini sidecar contains a usable thought signature.
@@ -317,18 +333,21 @@ def _finish_kwargs(api_kwargs: dict[str, Any], sanitized: list, params: dict, *,
return api_kwargs
def _sanitize_message(msg: Any, strip_extra_content: bool) -> dict | None:
def _sanitize_message(msg: Any, strip_extra_content: bool, strip_reasoning_details: bool = False) -> dict | None:
"""Sanitized copy of ``msg``, or None when nothing needs stripping.
Drops persistence sidecars, ``_``-prefixed scaffolding markers, tool-call ``call_id`` /
``response_item_id`` (and ``extra_content`` unless Gemini), an assistant
``tool_calls: []`` / ``null`` (strict providers reject both), and ``name``
``tool_calls: []`` / ``null`` (strict providers reject both), ``name``
on tool results (schema-valid only on user/assistant messages; strict
providers reject it with ``contains item with unknown key name``).
providers reject it with ``contains item with unknown key name``), and
``reasoning_details`` unless the route replays it (``_route_replays_reasoning_details``).
"""
if not isinstance(msg, dict):
return None
strip_keys = [k for k in msg if k in _STRIP_MSG_KEYS or (isinstance(k, str) and k.startswith("_"))]
if strip_reasoning_details and "reasoning_details" in msg:
strip_keys.append("reasoning_details")
# ``name`` is schema-valid on user/assistant messages, so the removal is
# role-qualified: only tool results carry it illegally (strict providers
# reject with "contains item with unknown key name").
@@ -375,7 +394,8 @@ class ChatCompletionsTransport(ProviderTransport):
Returns the input list unchanged when nothing needs sanitizing.
"""
strip_extra_content = not _model_consumes_thought_signature(kwargs.get("model"))
sanitized_pairs = [(m, _sanitize_message(m, strip_extra_content)) for m in messages]
strip_reasoning_details = not _route_replays_reasoning_details(kwargs.get("base_url"))
sanitized_pairs = [(m, _sanitize_message(m, strip_extra_content, strip_reasoning_details)) for m in messages]
if all(s is None for _, s in sanitized_pairs):
return messages
return [m if s is None else s for m, s in sanitized_pairs]
@@ -392,7 +412,7 @@ class ChatCompletionsTransport(ProviderTransport):
With ``provider_profile`` every quirk comes from the profile; the legacy flag
path below (is_kimi, is_openrouter, ...) is only reached for unregistered providers.
"""
sanitized = self.convert_messages(messages, model=model)
sanitized = self.convert_messages(messages, model=model, base_url=params.get("base_url"))
_profile = params.get("provider_profile")
if _profile:
return self._build_kwargs_from_profile(_profile, model, sanitized, tools, params)
+2 -2
View File
@@ -1157,8 +1157,8 @@ def build_api_messages(
agent._sanitize_tool_calls_for_strict_api(
api_msg, model=_sanitize_model_for(agent, moa_config)
)
# 'reasoning_details' is kept: OpenRouter uses it for multi-turn reasoning
# continuity.
# 'reasoning_details' is kept here; the chat-completions transport drops it on the
# wire for every route that does not replay it (OpenRouter/Nous do).
api_messages.append(api_msg)
# Final system message = cached prompt + ephemeral additions (API-time only).
@@ -0,0 +1,24 @@
"""``reasoning_details`` replay is route-scoped: OpenRouter/Nous read it, every other
chat-completions route gets a wire copy without it (strict schemas 400/422 on the field,
wedging the session after an in-session model switch — hermes-agent#70233)."""
from agent.transports import get_transport
_HISTORY = [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": "ok", "reasoning_details": [{"type": "reasoning.text", "text": "x", "signature": "E"}]},
{"role": "user", "content": "again"},
]
def test_non_replaying_route_drops_reasoning_details_only_on_the_wire_copy():
kwargs = get_transport("chat_completions").build_kwargs("qwen/qwen3.6-27b", _HISTORY, base_url="https://api.groq.com/openai/v1")
assert all("reasoning_details" not in m for m in kwargs["messages"])
assert "reasoning_details" in _HISTORY[1] # durable history is untouched
def test_openrouter_and_nous_routes_keep_reasoning_details():
transport = get_transport("chat_completions")
for base_url in ("https://openrouter.ai/api/v1", "https://inference-api.nousresearch.com/v1"):
kwargs = transport.build_kwargs("m", _HISTORY, base_url=base_url)
assert any("reasoning_details" in m for m in kwargs["messages"]), base_url