Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions agent_core/components/agent_bus/fan_in.py
Original file line number Diff line number Diff line change
Expand Up @@ -58,6 +58,10 @@
# ``no_tool``: the agent never chose to stop, so its report is unfinished
# rather than merely answer-less.
"response_truncated",
# Every reply in a turn, and the loop-level recovery after it, spent the
# whole output budget on private reasoning. Distinct from ``no_tool`` for
# the same reason as ``response_truncated``: the agent was stopped.
"reasoning_runaway",
"exception",
"thrash_no_progress",
})
Expand All @@ -72,6 +76,10 @@
"response_truncated": (
"agent's replies kept hitting the output token limit; report is partial"
),
"reasoning_runaway": (
"agent's replies kept spending the whole output budget on reasoning "
"without a visible answer; report is partial"
),
"budget_exhausted": "agent exhausted its token budget; report is partial",
"wall_deadline": (
"agent ran out of wall-clock time; report is partial"
Expand Down
6 changes: 6 additions & 0 deletions agent_core/loop_types.py
Original file line number Diff line number Diff line change
Expand Up @@ -172,6 +172,12 @@ class LoopConfig:
# a tool-less turn is the model choosing to stop, a truncated one is the
# model being stopped, so a truncation must not spend the nudge budget.
truncation_max_continuations: int = 2
# Loop-level recoveries for a turn that ``call_llm`` returned as a reasoning
# runaway (cap hit, no text, no tool call) after its own resample ladder ran
# out. Like truncation, this is the model being stopped, not choosing to
# stop, so it must not reach the ``no_tool`` exit; when these run out too the
# run ends with ``stop_reason="reasoning_runaway"`` so the failure is named.
runaway_max_loop_recoveries: int = 1
# Same-leg resamples after a wholly empty response, in both transports.
# Applies per logical call and per serving fallback leg, independently of
# max_llm_retries (the generic failure allowance). All resamples and chain
Expand Down
30 changes: 29 additions & 1 deletion agent_core/providers/anthropic.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
stream_terminator_required,
)
from agent_core.providers.finish_reason import normalize_finish_reason
from agent_core.runtime.llm_request_overrides import current_thinking_retry_override

logger = logging.getLogger(__name__)

Expand Down Expand Up @@ -108,6 +109,33 @@ def __init__(
default_headers=default_headers,
)

def _retry_effort(self) -> str:
"""The construction-time effort, unless a runaway retry overrides it.

The runaway ladder in ``_call.py`` steps thinking down by setting a
task-local :class:`ThinkingRetryOverride`. This adapter used to ignore it
entirely: every rung went out with the profile's ``effort`` and only a
smaller ``max_tokens``, so a model that spent its whole cap thinking at
``effort=max`` was asked to do the same thing in less room, and failed
the same way on every rung (GDPval 2026-10-08: three tasks lost whole
turns to ``8192 → 4096 → 2048``, every one pure thinking).

Effort is the only lever: ``claude-opus-5-5`` rejects
``thinking={"type": "disabled"}`` outright (400, "Use thinking.type.adaptive
and output_config.effort to control thinking behavior"), so the
``disabled`` rung is expressed through the effort the ladder pairs with
it, and the ``thinking`` field itself is never touched. Measured on the
same runaway prompt at an 8192 cap: ``max`` ended at the cap with no text;
``high``/``medium``/``low`` all finished with a full answer.

Like the OpenAI adapter, this only *replaces* an effort the profile
already opted into; a client built without one keeps sending none.
"""
override = current_thinking_retry_override()
if override is not None and override.reasoning_effort:
return override.reasoning_effort
return self._effort

def _build_kwargs(
self,
messages: list[Message],
Expand Down Expand Up @@ -151,7 +179,7 @@ def _build_kwargs(
# Current Claude models think adaptively even when thinking is omitted.
# Keep effort independent of that optional display/configuration field.
if self._effort:
kwargs["extra_body"] = {"output_config": {"effort": self._effort}}
kwargs["extra_body"] = {"output_config": {"effort": self._retry_effort()}}
if tools:
kwargs["tools"] = [_to_anthropic_tool(t) for t in tools]
if extra_headers:
Expand Down
11 changes: 9 additions & 2 deletions agent_core/runtime/loop/_call.py
Original file line number Diff line number Diff line change
Expand Up @@ -968,8 +968,8 @@ async def _recover_empty_completion(exc: LLMEmptyCompletion) -> None:
if runaway_state is not None:
runaway_state["last_call_reason"] = "reasoning_runaway"
# DELIVERED, not failed: this response is returned below, so
# the loop appends it to history, bills it, and salvages the
# turn with its no-tool nudge. Marking it ``failed`` would
# the loop appends it to history, bills it, and tries bounded
# runaway recovery. Marking it ``failed`` would
# make consumers drop bytes the loop actually used and would
# flip the enclosing trace call to ``status="failed"`` even
# though it produced a turn. ``reason`` carries the health.
Expand Down Expand Up @@ -1101,6 +1101,13 @@ async def _recover_empty_completion(exc: LLMEmptyCompletion) -> None:
"loop-level nudge handling",
turn, exc.trigger, exc.elapsed_s, exc.estimated_tokens,
)
# The stream ended before its terminal finish reason and usage
# arrived. Carry the watchdog's diagnosis to the loop rather than
# letting this empty response look like a chosen no-tool finish.
partial_response.response_metadata = {
**(partial_response.response_metadata or {}),
"reasoning_runaway_early": True,
}
await _finish_attempt(
outcome=ATTEMPT_ACCEPTED_DEGRADED,
reason="reasoning_runaway_early",
Expand Down
44 changes: 37 additions & 7 deletions agent_core/runtime/loop/_runaway.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,9 +26,8 @@
# runaway switches the retry to a reduced ``max_tokens`` and appends a
# transient, throwaway reminder asking for concise reasoning plus visible
# output/tool use. The reminder never enters durable message history.
# If the retry budget is exhausted the response is returned as-is so the
# loop's existing no-tool nudge stays the behavioural floor — a runaway must
# never escalate into a fatal ``llm_error`` stop past turn 1.
# If the retry budget is exhausted the response is returned to the loop for
# bounded recovery; a runaway must never become a clean ``no_tool`` finish.

_RUNAWAY_MIN_OUTPUT_TOKENS = 1024
# Ceiling for retry caps after a confirmed runaway. The actual bound cap
Expand Down Expand Up @@ -128,16 +127,20 @@ def _env_float(suffix: str, default: float) -> float:
RUNAWAY_STATE_KEY = "_runaway_state"

def _is_runaway_response(response: Any) -> bool:
"""True for a successful response whose budget went entirely to
reasoning: no visible content, no tool calls, and either
``finish_reason="length"`` or a completion-token count too large to
be a plain empty reply (gateways that drop ``finish_reason``)."""
"""True when a response was stopped after producing only reasoning.

Completed streams have a length finish reason or high token usage. An
early-stopped stream has neither terminal field, so ``call_llm`` marks the
partial response explicitly when its retry allowance is exhausted.
"""
if not isinstance(response, LLMResponse):
return False
if response.tool_calls:
return False
if _visible_response_text(response):
return False
if (response.response_metadata or {}).get("reasoning_runaway_early"):
return True
if response.finish_reason == "length":
return True
usage = extract_usage(response) or {}
Expand All @@ -155,6 +158,33 @@ def _is_runaway_response(response: Any) -> bool:
)


# Asked of a model after ``call_llm`` already spent its whole resample ladder on
# the turn and every rung came back as pure reasoning. Unlike the per-call
# guidance this lands in the conversation, so it can name the usual cause: a
# model with no network trying to *recall* data it was told to look up.
RUNAWAY_LOOP_RECOVERY_GUIDANCE = (
"[system reminder] Your last several replies spent the entire output budget "
"on private reasoning and produced no visible text or tool call. Do not "
"try to reconstruct large amounts of external data (prices, listings, "
"constituents, product catalogues) from memory. Act now: either make one "
"tool call that advances the task, or write the deliverable with what you "
"can support and clearly mark what could not be verified."
)


def is_runaway_response(response: Any) -> bool:
"""True for a reply whose whole budget went to reasoning (see below).

The loop-facing name for :func:`_is_runaway_response`. ``call_llm`` returns
such a response once its own resample ladder is exhausted, logging that it
is "returning empty response for loop-level nudge handling" — but the loop
had no branch for it. With no text and no tool call it fell into
``if not parsed_calls``, and under ``no_tool_behavior="stop"`` ended the run
as an ordinary ``no_tool`` finish with nothing delivered.
"""
return _is_runaway_response(response)


def is_truncated_with_text(response: Any) -> bool:
"""True for a reply the token cap cut off *after* it had produced text.

Expand Down
29 changes: 29 additions & 0 deletions agent_core/runtime/loop/agent_loop.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,7 @@
from agent_core.runtime.loop.image_attach import attach_images, evict_old_images
from agent_core.runtime.loop.llm_client import (
DEFAULT_SESSION_HEADER_NAMES,
RUNAWAY_LOOP_RECOVERY_GUIDANCE,
RUNAWAY_STATE_KEY,
TRUNCATION_CONTINUATION_GUIDANCE,
LLMCallExhausted,
Expand All @@ -65,6 +66,7 @@
extract_final_content,
extract_leaked_reasoning,
extract_usage,
is_runaway_response,
is_truncated_with_text,
)
from agent_core.runtime.loop.model_profile import (
Expand Down Expand Up @@ -444,6 +446,7 @@ async def _run_loop_inner(
total_tool_calls = 0
no_tool_retries = 0
truncation_continuations = 0
runaway_recoveries = 0
truncated_text_parts: list[str] = []

last_input_tokens = 0
Expand Down Expand Up @@ -607,6 +610,31 @@ async def _run_loop_inner(
)
break

# The empty half of the same "the model was stopped" signal: ``call_llm``
# spent its resample ladder and every rung was pure reasoning. Falling
# through to ``no_tool`` would end the run with nothing delivered (GDPval
# 2026-10-08: three whole tasks, all asked to look up data offline).
if not parsed_calls and is_runaway_response(response):
runaway_recoveries += 1
if runaway_recoveries <= cfg.runaway_max_loop_recoveries:
logger.warning(
"turn=%d reasoning runaway survived the call-level resamples "
"— recovering at loop level (%d/%d)",
turn, runaway_recoveries, cfg.runaway_max_loop_recoveries,
)
messages.append(user_msg(RUNAWAY_LOOP_RECOVERY_GUIDANCE))
# This is a retry of the interrupted turn, including when the
# runaway happened on the final allowed turn.
turn -= 1
continue
stop_reason = "reasoning_runaway"
logger.warning(
"turn=%d reasoning runaway after %d loop-level recover%s — stopping",
turn, cfg.runaway_max_loop_recoveries,
"y" if cfg.runaway_max_loop_recoveries == 1 else "ies",
)
break

if not parsed_calls:
no_tool_retries += 1
stops_for_no_tool = (
Expand Down Expand Up @@ -634,6 +662,7 @@ async def _run_loop_inner(

no_tool_retries = 0
truncation_continuations = 0
runaway_recoveries = 0
truncated_text_parts.clear()

if not skip_tool_execution:
Expand Down
4 changes: 4 additions & 0 deletions agent_core/runtime/loop/llm_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,8 +36,10 @@
usage_output_tokens,
)
from agent_core.runtime.loop._runaway import (
RUNAWAY_LOOP_RECOVERY_GUIDANCE,
RUNAWAY_STATE_KEY,
TRUNCATION_CONTINUATION_GUIDANCE,
is_runaway_response,
is_truncated_with_text,
)
from agent_core.runtime.loop._streaming import ThinkTagSplitter
Expand All @@ -48,6 +50,7 @@

__all__ = [
"DEFAULT_SESSION_HEADER_NAMES",
"RUNAWAY_LOOP_RECOVERY_GUIDANCE",
"RUNAWAY_STATE_KEY",
"TRUNCATION_CONTINUATION_GUIDANCE",
"LLMCallExhausted",
Expand All @@ -66,6 +69,7 @@
"extract_leaked_reasoning",
"extract_model_name",
"extract_usage",
"is_runaway_response",
"is_truncated_with_text",
"is_wholly_empty_response",
"usage_input_tokens",
Expand Down
3 changes: 3 additions & 0 deletions changes/77.feature.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
Reasoning-runaway recovery now works on the native Anthropic adapter. Its retry ladder applies `ThinkingRetryOverride.reasoning_effort` to `output_config.effort` without sending unsupported `thinking={"type": "disabled"}` to adaptive-only models.

When all call-level retries fail, the loop gives a reasoning-only response one bounded recovery attempt instead of ending it as `no_tool`. This applies both to completions that exhaust the output cap and to streams stopped early by the reasoning watchdog, whose terminal usage and finish reason may never arrive. Recovery remains available on the final allowed turn without consuming an additional logical turn. If recovery also fails, the loop stops with `stop_reason="reasoning_runaway"`; AgentBus fan-in treats that reason as an incomplete report. Consumers that enumerate stop reasons should add it.
Loading
Loading