From 60a7be813ca81cc6269c73e35010e418770a6b81 Mon Sep 17 00:00:00 2001 From: Zhang Handuo Date: Fri, 9 Oct 2026 21:08:58 +0800 Subject: [PATCH 1/3] fix(loop): make runaway recovery reach Anthropic and stop treating it as no_tool Two defects turned a reasoning runaway on claude-opus-5-5 into a lost task (GDPval 2026-10-08: 40a99a31 / 8079e27d / 99ac6944, all asked to look up live data offline): - providers/anthropic.py never read the ladder's ThinkingRetryOverride, so 8192 -> 4096 -> 2048 went out at effort=max every time. It now maps the override's reasoning_effort onto output_config.effort. Thinking itself is left alone: opus-5-5 400s on thinking.type=disabled. Measured on a runaway prompt at an 8192 cap: max -> 0 text at the cap; high/medium/low -> full answers. - call_llm returned the runaway "for loop-level nudge handling", but the loop had no such branch, so no text + no tool call reached no_tool and ended the run under no_tool_behavior="stop". The loop now recovers once (LoopConfig.runaway_max_loop_recoveries) with explicit guidance, then stops with stop_reason="reasoning_runaway". Co-Authored-By: Claude Opus 5.5 (1M context) --- agent_core/components/agent_bus/fan_in.py | 8 + agent_core/loop_types.py | 6 + agent_core/providers/anthropic.py | 30 ++- agent_core/runtime/loop/_runaway.py | 27 +++ agent_core/runtime/loop/agent_loop.py | 26 +++ agent_core/runtime/loop/llm_client.py | 4 + .../runaway-effort-and-loop-exit.feature.md | 1 + tests/test_runaway_loop_exit.py | 190 ++++++++++++++++++ 8 files changed, 291 insertions(+), 1 deletion(-) create mode 100644 changes/runaway-effort-and-loop-exit.feature.md create mode 100644 tests/test_runaway_loop_exit.py diff --git a/agent_core/components/agent_bus/fan_in.py b/agent_core/components/agent_bus/fan_in.py index c577024..9e2dee5 100644 --- a/agent_core/components/agent_bus/fan_in.py +++ b/agent_core/components/agent_bus/fan_in.py @@ -58,6 +58,10 @@ # ``no_tool``: the agent never chose to stop, so its report is unfinished # rather than merely answer-less. "response_truncated", + # Every reply in a turn, and the loop-level recovery after it, spent the + # whole output budget on private reasoning. Distinct from ``no_tool`` for + # the same reason as ``response_truncated``: the agent was stopped. + "reasoning_runaway", "exception", "thrash_no_progress", }) @@ -72,6 +76,10 @@ "response_truncated": ( "agent's replies kept hitting the output token limit; report is partial" ), + "reasoning_runaway": ( + "agent's replies kept spending the whole output budget on reasoning " + "without a visible answer; report is partial" + ), "budget_exhausted": "agent exhausted its token budget; report is partial", "wall_deadline": ( "agent ran out of wall-clock time; report is partial" diff --git a/agent_core/loop_types.py b/agent_core/loop_types.py index 54379aa..8d93808 100644 --- a/agent_core/loop_types.py +++ b/agent_core/loop_types.py @@ -172,6 +172,12 @@ class LoopConfig: # a tool-less turn is the model choosing to stop, a truncated one is the # model being stopped, so a truncation must not spend the nudge budget. truncation_max_continuations: int = 2 + # Loop-level recoveries for a turn that ``call_llm`` returned as a reasoning + # runaway (cap hit, no text, no tool call) after its own resample ladder ran + # out. Like truncation, this is the model being stopped, not choosing to + # stop, so it must not reach the ``no_tool`` exit; when these run out too the + # run ends with ``stop_reason="reasoning_runaway"`` so the failure is named. + runaway_max_loop_recoveries: int = 1 # Same-leg resamples after a wholly empty response, in both transports. # Applies per logical call and per serving fallback leg, independently of # max_llm_retries (the generic failure allowance). All resamples and chain diff --git a/agent_core/providers/anthropic.py b/agent_core/providers/anthropic.py index c415bd7..7fb1330 100644 --- a/agent_core/providers/anthropic.py +++ b/agent_core/providers/anthropic.py @@ -40,6 +40,7 @@ stream_terminator_required, ) from agent_core.providers.finish_reason import normalize_finish_reason +from agent_core.runtime.llm_request_overrides import current_thinking_retry_override logger = logging.getLogger(__name__) @@ -108,6 +109,33 @@ def __init__( default_headers=default_headers, ) + def _retry_effort(self) -> str: + """The construction-time effort, unless a runaway retry overrides it. + + The runaway ladder in ``_call.py`` steps thinking down by setting a + task-local :class:`ThinkingRetryOverride`. This adapter used to ignore it + entirely: every rung went out with the profile's ``effort`` and only a + smaller ``max_tokens``, so a model that spent its whole cap thinking at + ``effort=max`` was asked to do the same thing in less room, and failed + the same way on every rung (GDPval 2026-10-08: three tasks lost whole + turns to ``8192 → 4096 → 2048``, every one pure thinking). + + Effort is the only lever: ``claude-opus-5-5`` rejects + ``thinking={"type": "disabled"}`` outright (400, "Use thinking.type.adaptive + and output_config.effort to control thinking behavior"), so the + ``disabled`` rung is expressed through the effort the ladder pairs with + it, and the ``thinking`` field itself is never touched. Measured on the + same runaway prompt at an 8192 cap: ``max`` ended at the cap with no text; + ``high``/``medium``/``low`` all finished with a full answer. + + Like the OpenAI adapter, this only *replaces* an effort the profile + already opted into; a client built without one keeps sending none. + """ + override = current_thinking_retry_override() + if override is not None and override.reasoning_effort: + return override.reasoning_effort + return self._effort + def _build_kwargs( self, messages: list[Message], @@ -151,7 +179,7 @@ def _build_kwargs( # Current Claude models think adaptively even when thinking is omitted. # Keep effort independent of that optional display/configuration field. if self._effort: - kwargs["extra_body"] = {"output_config": {"effort": self._effort}} + kwargs["extra_body"] = {"output_config": {"effort": self._retry_effort()}} if tools: kwargs["tools"] = [_to_anthropic_tool(t) for t in tools] if extra_headers: diff --git a/agent_core/runtime/loop/_runaway.py b/agent_core/runtime/loop/_runaway.py index 7fb7eca..9b382f0 100644 --- a/agent_core/runtime/loop/_runaway.py +++ b/agent_core/runtime/loop/_runaway.py @@ -155,6 +155,33 @@ def _is_runaway_response(response: Any) -> bool: ) +# Asked of a model after ``call_llm`` already spent its whole resample ladder on +# the turn and every rung came back as pure reasoning. Unlike the per-call +# guidance this lands in the conversation, so it can name the usual cause: a +# model with no network trying to *recall* data it was told to look up. +RUNAWAY_LOOP_RECOVERY_GUIDANCE = ( + "[system reminder] Your last several replies spent the entire output budget " + "on private reasoning and produced no visible text or tool call. Do not " + "try to reconstruct large amounts of external data (prices, listings, " + "constituents, product catalogues) from memory. Act now: either make one " + "tool call that advances the task, or write the deliverable with what you " + "can support and clearly mark what could not be verified." +) + + +def is_runaway_response(response: Any) -> bool: + """True for a reply whose whole budget went to reasoning (see below). + + The loop-facing name for :func:`_is_runaway_response`. ``call_llm`` returns + such a response once its own resample ladder is exhausted, logging that it + is "returning empty response for loop-level nudge handling" — but the loop + had no branch for it. With no text and no tool call it fell into + ``if not parsed_calls``, and under ``no_tool_behavior="stop"`` ended the run + as an ordinary ``no_tool`` finish with nothing delivered. + """ + return _is_runaway_response(response) + + def is_truncated_with_text(response: Any) -> bool: """True for a reply the token cap cut off *after* it had produced text. diff --git a/agent_core/runtime/loop/agent_loop.py b/agent_core/runtime/loop/agent_loop.py index 132d766..8651dad 100644 --- a/agent_core/runtime/loop/agent_loop.py +++ b/agent_core/runtime/loop/agent_loop.py @@ -53,6 +53,7 @@ from agent_core.runtime.loop.image_attach import attach_images, evict_old_images from agent_core.runtime.loop.llm_client import ( DEFAULT_SESSION_HEADER_NAMES, + RUNAWAY_LOOP_RECOVERY_GUIDANCE, RUNAWAY_STATE_KEY, TRUNCATION_CONTINUATION_GUIDANCE, LLMCallExhausted, @@ -65,6 +66,7 @@ extract_final_content, extract_leaked_reasoning, extract_usage, + is_runaway_response, is_truncated_with_text, ) from agent_core.runtime.loop.model_profile import ( @@ -444,6 +446,7 @@ async def _run_loop_inner( total_tool_calls = 0 no_tool_retries = 0 truncation_continuations = 0 + runaway_recoveries = 0 truncated_text_parts: list[str] = [] last_input_tokens = 0 @@ -607,6 +610,28 @@ async def _run_loop_inner( ) break + # The empty half of the same "the model was stopped" signal: ``call_llm`` + # spent its resample ladder and every rung was pure reasoning. Falling + # through to ``no_tool`` would end the run with nothing delivered (GDPval + # 2026-10-08: three whole tasks, all asked to look up data offline). + if not parsed_calls and is_runaway_response(response): + runaway_recoveries += 1 + if runaway_recoveries <= cfg.runaway_max_loop_recoveries: + logger.warning( + "turn=%d reasoning runaway survived the call-level resamples " + "— recovering at loop level (%d/%d)", + turn, runaway_recoveries, cfg.runaway_max_loop_recoveries, + ) + messages.append(user_msg(RUNAWAY_LOOP_RECOVERY_GUIDANCE)) + continue + stop_reason = "reasoning_runaway" + logger.warning( + "turn=%d reasoning runaway after %d loop-level recover%s — stopping", + turn, cfg.runaway_max_loop_recoveries, + "y" if cfg.runaway_max_loop_recoveries == 1 else "ies", + ) + break + if not parsed_calls: no_tool_retries += 1 stops_for_no_tool = ( @@ -634,6 +659,7 @@ async def _run_loop_inner( no_tool_retries = 0 truncation_continuations = 0 + runaway_recoveries = 0 truncated_text_parts.clear() if not skip_tool_execution: diff --git a/agent_core/runtime/loop/llm_client.py b/agent_core/runtime/loop/llm_client.py index 914a3e8..c36e44d 100644 --- a/agent_core/runtime/loop/llm_client.py +++ b/agent_core/runtime/loop/llm_client.py @@ -36,8 +36,10 @@ usage_output_tokens, ) from agent_core.runtime.loop._runaway import ( + RUNAWAY_LOOP_RECOVERY_GUIDANCE, RUNAWAY_STATE_KEY, TRUNCATION_CONTINUATION_GUIDANCE, + is_runaway_response, is_truncated_with_text, ) from agent_core.runtime.loop._streaming import ThinkTagSplitter @@ -48,6 +50,7 @@ __all__ = [ "DEFAULT_SESSION_HEADER_NAMES", + "RUNAWAY_LOOP_RECOVERY_GUIDANCE", "RUNAWAY_STATE_KEY", "TRUNCATION_CONTINUATION_GUIDANCE", "LLMCallExhausted", @@ -66,6 +69,7 @@ "extract_leaked_reasoning", "extract_model_name", "extract_usage", + "is_runaway_response", "is_truncated_with_text", "is_wholly_empty_response", "usage_input_tokens", diff --git a/changes/runaway-effort-and-loop-exit.feature.md b/changes/runaway-effort-and-loop-exit.feature.md new file mode 100644 index 0000000..f3af6bd --- /dev/null +++ b/changes/runaway-effort-and-loop-exit.feature.md @@ -0,0 +1 @@ +Reasoning-runaway recovery now works on the native Anthropic adapter, and a turn the resample ladder could not recover no longer ends the run as an ordinary `no_tool` finish. The Anthropic adapter previously ignored the ladder's task-local `ThinkingRetryOverride`, so every rung went out with the profile's `effort` and only a smaller `max_tokens`; it now sends the override's `reasoning_effort` as `output_config.effort` (replacing an effort the profile set, never adding one) and never sends `thinking={"type": "disabled"}`, which adaptive-only models such as `claude-opus-5-5` reject with a 400. When `call_llm` still returns a runaway (output cap reached with no visible text or tool call), the loop appends a recovery reminder and continues, up to the new `LoopConfig.runaway_max_loop_recoveries` (default 1, reset by any turn with a tool call); after that the run stops with the new `stop_reason="reasoning_runaway"` instead of `no_tool`. Consumers that enumerate stop reasons should add `reasoning_runaway` (AgentBus fan-in already treats it as an incomplete report); runs that hit this path spend at most one extra turn per runaway. diff --git a/tests/test_runaway_loop_exit.py b/tests/test_runaway_loop_exit.py new file mode 100644 index 0000000..d24a216 --- /dev/null +++ b/tests/test_runaway_loop_exit.py @@ -0,0 +1,190 @@ +"""A reasoning runaway must be steerable and must never look like a finished run. + +Measured on ApodexHarness's 2026-10-08 GDPval batch over claude-opus-5-5 at +``effort=max``: three tasks asked to look up live data offline (S&P 500 closing +prices, a priced equipment list, a hardware bill of materials) spent a turn +entirely in private reasoning. The resample ladder went ``8192 → 4096 → 2048`` +with ``next_thinking_mode=disabled`` logged on the last rungs, and every rung was +still pure reasoning. Two independent defects made that a lost task: + +* the Anthropic adapter never read the ladder's override, so every rung went out + with the profile's ``effort=max`` and only a smaller ``max_tokens``; +* ``call_llm`` returned the runaway "for loop-level nudge handling", but the loop + had no such branch: no text and no tool call reached ``no_tool``, and under + ``no_tool_behavior="stop"`` the run ended with nothing delivered. +""" +from __future__ import annotations + +import json +from typing import Any +from unittest.mock import AsyncMock, MagicMock + +import pytest + +from agent_core.llm import LLMResponse +from agent_core.loop_types import LoopConfig, LoopPolicy +from agent_core.providers import anthropic as ac +from agent_core.runtime.llm_request_overrides import ( + ThinkingRetryOverride, + thinking_retry_override, +) +from agent_core.runtime.loop import _call +from agent_core.runtime.loop._runaway import RUNAWAY_LOOP_RECOVERY_GUIDANCE +from agent_core.runtime.loop.agent_loop import run_agent_loop +from agent_core.runtime.loop.model_profile import ModelProfile + +# ── provider: the ladder's effort reaches the request ───────────────────── + + +def _request(client: ac.AnthropicClient) -> dict[str, Any]: + return client._build_kwargs( + [{"role": "user", "content": "hi"}], tools=None, temperature=None, + max_tokens=2048, extra_headers=None, timeout=None, + ) + + +def _opus(effort: str = "max") -> ac.AnthropicClient: + return ac.AnthropicClient( + "claude-opus-5-5", api_key="x", max_tokens=128000, + thinking={"type": "adaptive", "display": "summarized"}, effort=effort, + ) + + +def test_profile_effort_is_sent_without_an_override() -> None: + assert _request(_opus())["extra_body"] == {"output_config": {"effort": "max"}} + + +@pytest.mark.parametrize("mode", ["expanded", "reduced", "disabled"]) +def test_retry_override_replaces_effort_and_never_touches_thinking(mode: str) -> None: + """``disabled`` is expressed through effort: opus-5-5 rejects + ``thinking={"type": "disabled"}`` with a 400.""" + client = _opus() + with thinking_retry_override(ThinkingRetryOverride(mode=mode, reasoning_effort="low")): + request = _request(client) + + assert request["extra_body"] == {"output_config": {"effort": "low"}} + assert request["thinking"] == {"type": "adaptive", "display": "summarized"} + + +def test_override_without_effort_keeps_the_profile_effort() -> None: + with thinking_retry_override(ThinkingRetryOverride(mode="reduced", thinking_budget=512)): + request = _request(_opus()) + assert request["extra_body"] == {"output_config": {"effort": "max"}} + + +def test_client_without_effort_does_not_gain_one() -> None: + """Like the OpenAI adapter: replace an effort the profile opted into, never add one.""" + client = ac.AnthropicClient("claude-x", api_key="x") + with thinking_retry_override(ThinkingRetryOverride(mode="disabled", reasoning_effort="low")): + assert "extra_body" not in _request(client) + + +# ── loop: a runaway turn is recovered, and a persistent one is named ────── + + +def _runaway() -> LLMResponse: + return LLMResponse( + content="", reasoning_content="x" * 5_000, finish_reason="length", + usage={"prompt_tokens": 10, "completion_tokens": 2_048}, + ) + + +def _tool_call(n: int) -> LLMResponse: + return LLMResponse( + content="", finish_reason="tool_calls", + tool_calls=[{"type": "function", "id": f"c{n}", + "function": {"name": "bash", "arguments": json.dumps({"command": "ls"})}}], + usage={"prompt_tokens": 10, "completion_tokens": 20}, + ) + + +def _final() -> LLMResponse: + return LLMResponse(content="done", finish_reason="stop", + usage={"prompt_tokens": 10, "completion_tokens": 2}) + + +class _Scripted: + model = "fake" + + def __init__(self, *responses: LLMResponse) -> None: + self._pending = list(responses) + self.requests: list[list[Any]] = [] + + async def chat(self, messages: list[Any], **_kwargs: Any) -> LLMResponse: + self.requests.append(list(messages)) + return self._pending.pop(0) + + +@pytest.fixture +def no_call_ladder(monkeypatch: pytest.MonkeyPatch) -> None: + """Hand every runaway straight back to the loop, as an exhausted ladder does.""" + monkeypatch.setattr(_call, "_RUNAWAY_MAX_RETRIES", 0) + monkeypatch.setattr(_call, "_RUNAWAY_BACKOFF_S", 0.0) + + +def _tool() -> MagicMock: + tool = MagicMock() + tool.name = "bash" + tool.ainvoke = AsyncMock(return_value="ok") + return tool + + +async def _run(llm: _Scripted, tool: MagicMock, **cfg: Any) -> Any: + return await run_agent_loop( + system_prompt="system", user_message="start", llm=llm, tools=[tool], + config=LoopConfig( + max_turns=8, max_llm_retries=2, retry_wait_fixed=0, + stream_llm_tokens=False, + loop_policy=LoopPolicy(no_tool_behavior="stop"), + **cfg, + ), + model_profile=ModelProfile(model_id="fake", provider="p", protocol="openai"), + ) + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("no_call_ladder") +async def test_runaway_turn_is_recovered_instead_of_ending_as_no_tool() -> None: + llm, tool = _Scripted(_runaway(), _tool_call(1), _final()), _tool() + + result = await _run(llm, tool) + + tool.ainvoke.assert_awaited_once() + assert result.final_content == "done" + assert result.stopped_by == "no_tool" # the real, chosen finish + assert any(m.get("content") == RUNAWAY_LOOP_RECOVERY_GUIDANCE for m in llm.requests[1]) + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("no_call_ladder") +async def test_persistent_runaway_stops_with_its_own_reason() -> None: + llm, tool = _Scripted(_runaway(), _runaway()), _tool() + + result = await _run(llm, tool) + + assert len(llm.requests) == 2 # one turn + one loop-level recovery + tool.ainvoke.assert_not_awaited() + assert result.stopped_by == "reasoning_runaway" + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("no_call_ladder") +async def test_recovery_allowance_resets_after_progress() -> None: + llm = _Scripted(_runaway(), _tool_call(1), _runaway(), _tool_call(2), _final()) + tool = _tool() + + result = await _run(llm, tool) + + assert tool.ainvoke.await_count == 2 + assert result.final_content == "done" + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("no_call_ladder") +async def test_zero_allowance_still_names_the_failure() -> None: + llm, tool = _Scripted(_runaway()), _tool() + + result = await _run(llm, tool, runaway_max_loop_recoveries=0) + + assert len(llm.requests) == 1 + assert result.stopped_by == "reasoning_runaway" From 78c85b081048e12ad8532743b1ba8172765f2254 Mon Sep 17 00:00:00 2001 From: Zhang Handuo Date: Fri, 9 Oct 2026 21:09:36 +0800 Subject: [PATCH 2/3] chore: name the change fragment after its PR Co-Authored-By: Claude Opus 5.5 (1M context) --- .../{runaway-effort-and-loop-exit.feature.md => 77.feature.md} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename changes/{runaway-effort-and-loop-exit.feature.md => 77.feature.md} (100%) diff --git a/changes/runaway-effort-and-loop-exit.feature.md b/changes/77.feature.md similarity index 100% rename from changes/runaway-effort-and-loop-exit.feature.md rename to changes/77.feature.md From c2f7b20a73a52979f3ffc8b1e1649967ed31a63d Mon Sep 17 00:00:00 2001 From: zhanghanduo Date: Fri, 9 Oct 2026 21:29:21 +0800 Subject: [PATCH 3/3] fix(loop): preserve runaway recovery on final and early-stream turns --- agent_core/runtime/loop/_call.py | 11 ++++- agent_core/runtime/loop/_runaway.py | 17 ++++--- agent_core/runtime/loop/agent_loop.py | 3 ++ changes/77.feature.md | 4 +- tests/test_runaway_loop_exit.py | 68 +++++++++++++++++++++++++-- 5 files changed, 89 insertions(+), 14 deletions(-) diff --git a/agent_core/runtime/loop/_call.py b/agent_core/runtime/loop/_call.py index b214366..34adbe9 100644 --- a/agent_core/runtime/loop/_call.py +++ b/agent_core/runtime/loop/_call.py @@ -968,8 +968,8 @@ async def _recover_empty_completion(exc: LLMEmptyCompletion) -> None: if runaway_state is not None: runaway_state["last_call_reason"] = "reasoning_runaway" # DELIVERED, not failed: this response is returned below, so - # the loop appends it to history, bills it, and salvages the - # turn with its no-tool nudge. Marking it ``failed`` would + # the loop appends it to history, bills it, and tries bounded + # runaway recovery. Marking it ``failed`` would # make consumers drop bytes the loop actually used and would # flip the enclosing trace call to ``status="failed"`` even # though it produced a turn. ``reason`` carries the health. @@ -1101,6 +1101,13 @@ async def _recover_empty_completion(exc: LLMEmptyCompletion) -> None: "loop-level nudge handling", turn, exc.trigger, exc.elapsed_s, exc.estimated_tokens, ) + # The stream ended before its terminal finish reason and usage + # arrived. Carry the watchdog's diagnosis to the loop rather than + # letting this empty response look like a chosen no-tool finish. + partial_response.response_metadata = { + **(partial_response.response_metadata or {}), + "reasoning_runaway_early": True, + } await _finish_attempt( outcome=ATTEMPT_ACCEPTED_DEGRADED, reason="reasoning_runaway_early", diff --git a/agent_core/runtime/loop/_runaway.py b/agent_core/runtime/loop/_runaway.py index 9b382f0..a9f9cad 100644 --- a/agent_core/runtime/loop/_runaway.py +++ b/agent_core/runtime/loop/_runaway.py @@ -26,9 +26,8 @@ # runaway switches the retry to a reduced ``max_tokens`` and appends a # transient, throwaway reminder asking for concise reasoning plus visible # output/tool use. The reminder never enters durable message history. -# If the retry budget is exhausted the response is returned as-is so the -# loop's existing no-tool nudge stays the behavioural floor — a runaway must -# never escalate into a fatal ``llm_error`` stop past turn 1. +# If the retry budget is exhausted the response is returned to the loop for +# bounded recovery; a runaway must never become a clean ``no_tool`` finish. _RUNAWAY_MIN_OUTPUT_TOKENS = 1024 # Ceiling for retry caps after a confirmed runaway. The actual bound cap @@ -128,16 +127,20 @@ def _env_float(suffix: str, default: float) -> float: RUNAWAY_STATE_KEY = "_runaway_state" def _is_runaway_response(response: Any) -> bool: - """True for a successful response whose budget went entirely to - reasoning: no visible content, no tool calls, and either - ``finish_reason="length"`` or a completion-token count too large to - be a plain empty reply (gateways that drop ``finish_reason``).""" + """True when a response was stopped after producing only reasoning. + + Completed streams have a length finish reason or high token usage. An + early-stopped stream has neither terminal field, so ``call_llm`` marks the + partial response explicitly when its retry allowance is exhausted. + """ if not isinstance(response, LLMResponse): return False if response.tool_calls: return False if _visible_response_text(response): return False + if (response.response_metadata or {}).get("reasoning_runaway_early"): + return True if response.finish_reason == "length": return True usage = extract_usage(response) or {} diff --git a/agent_core/runtime/loop/agent_loop.py b/agent_core/runtime/loop/agent_loop.py index 8651dad..5a1e6a5 100644 --- a/agent_core/runtime/loop/agent_loop.py +++ b/agent_core/runtime/loop/agent_loop.py @@ -623,6 +623,9 @@ async def _run_loop_inner( turn, runaway_recoveries, cfg.runaway_max_loop_recoveries, ) messages.append(user_msg(RUNAWAY_LOOP_RECOVERY_GUIDANCE)) + # This is a retry of the interrupted turn, including when the + # runaway happened on the final allowed turn. + turn -= 1 continue stop_reason = "reasoning_runaway" logger.warning( diff --git a/changes/77.feature.md b/changes/77.feature.md index f3af6bd..635d3b9 100644 --- a/changes/77.feature.md +++ b/changes/77.feature.md @@ -1 +1,3 @@ -Reasoning-runaway recovery now works on the native Anthropic adapter, and a turn the resample ladder could not recover no longer ends the run as an ordinary `no_tool` finish. The Anthropic adapter previously ignored the ladder's task-local `ThinkingRetryOverride`, so every rung went out with the profile's `effort` and only a smaller `max_tokens`; it now sends the override's `reasoning_effort` as `output_config.effort` (replacing an effort the profile set, never adding one) and never sends `thinking={"type": "disabled"}`, which adaptive-only models such as `claude-opus-5-5` reject with a 400. When `call_llm` still returns a runaway (output cap reached with no visible text or tool call), the loop appends a recovery reminder and continues, up to the new `LoopConfig.runaway_max_loop_recoveries` (default 1, reset by any turn with a tool call); after that the run stops with the new `stop_reason="reasoning_runaway"` instead of `no_tool`. Consumers that enumerate stop reasons should add `reasoning_runaway` (AgentBus fan-in already treats it as an incomplete report); runs that hit this path spend at most one extra turn per runaway. +Reasoning-runaway recovery now works on the native Anthropic adapter. Its retry ladder applies `ThinkingRetryOverride.reasoning_effort` to `output_config.effort` without sending unsupported `thinking={"type": "disabled"}` to adaptive-only models. + +When all call-level retries fail, the loop gives a reasoning-only response one bounded recovery attempt instead of ending it as `no_tool`. This applies both to completions that exhaust the output cap and to streams stopped early by the reasoning watchdog, whose terminal usage and finish reason may never arrive. Recovery remains available on the final allowed turn without consuming an additional logical turn. If recovery also fails, the loop stops with `stop_reason="reasoning_runaway"`; AgentBus fan-in treats that reason as an incomplete report. Consumers that enumerate stop reasons should add it. diff --git a/tests/test_runaway_loop_exit.py b/tests/test_runaway_loop_exit.py index d24a216..a2ea08a 100644 --- a/tests/test_runaway_loop_exit.py +++ b/tests/test_runaway_loop_exit.py @@ -21,7 +21,7 @@ import pytest -from agent_core.llm import LLMResponse +from agent_core.llm import LLMResponse, StreamDelta from agent_core.loop_types import LoopConfig, LoopPolicy from agent_core.providers import anthropic as ac from agent_core.runtime.llm_request_overrides import ( @@ -129,12 +129,15 @@ def _tool() -> MagicMock: return tool -async def _run(llm: _Scripted, tool: MagicMock, **cfg: Any) -> Any: +async def _run( + llm: Any, tool: MagicMock, *, max_turns: int = 8, + max_llm_retries: int = 2, stream_llm_tokens: bool = False, **cfg: Any, +) -> Any: return await run_agent_loop( system_prompt="system", user_message="start", llm=llm, tools=[tool], config=LoopConfig( - max_turns=8, max_llm_retries=2, retry_wait_fixed=0, - stream_llm_tokens=False, + max_turns=max_turns, max_llm_retries=max_llm_retries, + retry_wait_fixed=0, stream_llm_tokens=stream_llm_tokens, loop_policy=LoopPolicy(no_tool_behavior="stop"), **cfg, ), @@ -155,6 +158,30 @@ async def test_runaway_turn_is_recovered_instead_of_ending_as_no_tool() -> None: assert any(m.get("content") == RUNAWAY_LOOP_RECOVERY_GUIDANCE for m in llm.requests[1]) +@pytest.mark.asyncio +@pytest.mark.usefixtures("no_call_ladder") +async def test_runaway_on_last_turn_still_gets_its_recovery() -> None: + llm, tool = _Scripted(_runaway(), _final()), _tool() + + result = await _run(llm, tool, max_turns=1) + + assert len(llm.requests) == 2 + assert result.final_content == "done" + assert result.stopped_by == "no_tool" + assert result.turns_used == 1 + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("no_call_ladder") +async def test_persistent_runaway_on_last_turn_names_the_failure() -> None: + llm, tool = _Scripted(_runaway(), _runaway()), _tool() + + result = await _run(llm, tool, max_turns=1) + + assert len(llm.requests) == 2 + assert result.stopped_by == "reasoning_runaway" + + @pytest.mark.asyncio @pytest.mark.usefixtures("no_call_ladder") async def test_persistent_runaway_stops_with_its_own_reason() -> None: @@ -188,3 +215,36 @@ async def test_zero_allowance_still_names_the_failure() -> None: assert len(llm.requests) == 1 assert result.stopped_by == "reasoning_runaway" + + +class _EarlyRunawayStream: + model = "fake" + + def __init__(self, *, recover: bool) -> None: + self.recover = recover + self.requests: list[list[Any]] = [] + + async def stream(self, messages: list[Any], **_kwargs: Any) -> Any: + self.requests.append(list(messages)) + if self.recover and len(self.requests) == 2: + yield StreamDelta(content="done", finish_reason="stop") + else: + # The guard fires before the provider's terminal usage/finish chunk. + yield StreamDelta(reasoning_content="x" * 400) + + +@pytest.mark.asyncio +@pytest.mark.usefixtures("no_call_ladder") +@pytest.mark.parametrize("recover", [True, False]) +async def test_early_stream_runaway_reaches_loop_recovery(recover: bool) -> None: + llm, tool = _EarlyRunawayStream(recover=recover), _tool() + + result = await _run( + llm, tool, max_turns=1, max_llm_retries=1, + stream_llm_tokens=True, reasoning_only_max_tokens=100, + ) + + assert len(llm.requests) == 2 + assert any(m.get("content") == RUNAWAY_LOOP_RECOVERY_GUIDANCE for m in llm.requests[1]) + assert result.stopped_by == ("no_tool" if recover else "reasoning_runaway") + assert result.final_content == ("done" if recover else "")