From d1a221c257f4bd5df416e597f2f6ed567672dd7a Mon Sep 17 00:00:00 2001 From: yannickmonney Date: Fri, 9 Oct 2026 17:14:08 +0200 Subject: [PATCH] fix(platform): bound retry and worker cleanup --- .../backend/core/automations/agent_retry.ts | 10 ++++- .../backend/core/automations/stepper.ts | 8 ++-- .../backend/core/chat/external_turn_shared.ts | 2 +- .../core/tasks/task_input_mirrors.test.ts | 34 +++++++++++++-- .../backend/core/tasks/task_input_mirrors.ts | 21 +++++++--- .../backend/domains/tasks/agent-runs.test.ts | 42 +++++++++++++++++++ .../backend/domains/tasks/agent-runs.ts | 26 ++++++++---- services/platform/messages/de/tasks.yml | 2 +- services/platform/messages/en/tasks.yml | 2 +- services/platform/messages/fr/tasks.yml | 2 +- .../reference/automation/automations.md | 2 +- services/sandbox/README.md | 10 +++-- tools/cli/scripts/cli-workflow.test.ts | 1 + 13 files changed, 131 insertions(+), 31 deletions(-) diff --git a/services/platform/backend/core/automations/agent_retry.ts b/services/platform/backend/core/automations/agent_retry.ts index f43b17b1d2..12e4fff927 100644 --- a/services/platform/backend/core/automations/agent_retry.ts +++ b/services/platform/backend/core/automations/agent_retry.ts @@ -30,8 +30,8 @@ export type WorkflowAgentFailureCode = | 'turn_stalled' /** The sandbox ran out of memory and the kernel's OOM killer ended the * harness or its session. Re-kicked like any failure, but only after - * `resourceExhaustedRetryDelayMs` for the node's attempt: at once it - * would meet the same limit. */ + * {@link RESOURCE_EXHAUSTED_REKICK_DELAY_MS}: at once it would meet the + * same limit. */ | 'resource_exhausted' | 'session_gone' | 'start_failed' @@ -76,6 +76,12 @@ export const SANDBOX_ROOM_MAX_WAIT_MS = 2 * 60 * 60_000; * past it, a node still waiting asks about once a minute on average. */ export const SANDBOX_ROOM_RETRY_CEILING_MS = 2 * 60_000; +/** How long the re-kick of a node whose sandbox ran out of memory is held: + * as long as the kick may hold a start (its op row must stay inside the + * stalled-turn sweep's window), so the 10 and 30 minutes a task run waits + * are not available here. */ +export const RESOURCE_EXHAUSTED_REKICK_DELAY_MS = SANDBOX_ROOM_RETRY_CEILING_MS; + /** The most a start that holds a place in the spawner's line comes back * after its hint: enough to keep waiters refused together apart. */ const QUEUED_RETRY_JITTER_MS = 1_000; diff --git a/services/platform/backend/core/automations/stepper.ts b/services/platform/backend/core/automations/stepper.ts index e351bd01a7..06fcc41967 100644 --- a/services/platform/backend/core/automations/stepper.ts +++ b/services/platform/backend/core/automations/stepper.ts @@ -33,13 +33,13 @@ import { harnessResumesConversations } from '../chat/external_turn_shared'; import type { ActionCtx } from '../lib/ctx'; import { internal } from '../lib/handler_names'; import type { Id } from '../lib/rows'; -import { resourceExhaustedRetryDelayMs } from '../tasks/task_auto_retry'; import { automationAgentHost, type AutomationAgentHost, type WorkflowAgentRequest, } from './agent_host'; import { + RESOURCE_EXHAUSTED_REKICK_DELAY_MS, SANDBOX_ROOM_MAX_WAIT_MS, isWorkflowAgentRetryable, planWorkflowAgentRetry, @@ -1720,8 +1720,8 @@ async function stepAgentNode(args: AgentStepArgs): Promise { // more with each refusal in a row; any other refusal with a hint (a // broker pool cooling down) waits for exactly that. const now = Date.now(); - // One whose sandbox ran out of memory waits 2, 10, then 30 minutes: - // at once it would meet the same limit. + // One whose sandbox ran out of memory waits as long as a start may be + // held (two minutes): at once it would meet the same limit. const notBefore = waitingForRoom ? sandboxRoomRetryAtMs({ now, @@ -1732,7 +1732,7 @@ async function stepAgentNode(args: AgentStepArgs): Promise { queued: settled.roomQueued === true, }) : settled.failureCode === 'resource_exhausted' - ? now + resourceExhaustedRetryDelayMs(parked.attempt) + ? now + RESOURCE_EXHAUSTED_REKICK_DELAY_MS : settled.retryAtMs; const kicked = await run.agent.kick({ runId: run.runId, diff --git a/services/platform/backend/core/chat/external_turn_shared.ts b/services/platform/backend/core/chat/external_turn_shared.ts index 2a24ee93da..eec97c5372 100644 --- a/services/platform/backend/core/chat/external_turn_shared.ts +++ b/services/platform/backend/core/chat/external_turn_shared.ts @@ -1136,7 +1136,7 @@ const OUT_OF_MEMORY_CODES: ReadonlySet = new Set([ /** The reason a turn settles failed with when its sandbox ran out of * memory. */ export const OUT_OF_MEMORY_TURN_REASON = - "The agent's sandbox ran out of memory: the kernel's OOM killer ended the agent. A retry follows after a pause; if it keeps happening, the agent sessions need a larger memory limit (SANDBOX_AGENT_MEMORY)."; + "The agent's sandbox ran out of memory: the kernel's OOM killer ended the agent. If it keeps happening, the agent sessions need a larger memory limit (SANDBOX_AGENT_MEMORY)."; /** The reason a turn settles failed with when the sandbox ended its harness * as stalled. */ diff --git a/services/platform/backend/core/tasks/task_input_mirrors.test.ts b/services/platform/backend/core/tasks/task_input_mirrors.test.ts index 4de1fb4805..365ffcff6c 100644 --- a/services/platform/backend/core/tasks/task_input_mirrors.test.ts +++ b/services/platform/backend/core/tasks/task_input_mirrors.test.ts @@ -146,7 +146,7 @@ describe('the start’s pass over the worker’s copies of task inputs', () => { ); }); - it('looks at the oldest copies first, a bounded number per start', async () => { + it('removes the oldest stale copies first, a bounded number per start', async () => { io.dirs['/agent/inputs'] = Array.from( { length: MAX_INPUT_MIRRORS_PER_PASS + 10 }, (_, n) => dir(`task-${n}`, 10_000 - n), @@ -158,16 +158,42 @@ describe('the start’s pass over the worker’s copies of task inputs', () => { await pruneStaleTaskInputMirrors(ctx, ARGS); + // Every copy is asked about, the least recently changed first. const checked = asked[0]?.taskIds as string[]; - expect(checked).toHaveLength(MAX_INPUT_MIRRORS_PER_PASS); - // The least recently changed first: task-59 down to task-10. + expect(checked).toHaveLength(MAX_INPUT_MIRRORS_PER_PASS + 10); expect(checked[0]).toBe(`task-${MAX_INPUT_MIRRORS_PER_PASS + 9}`); - expect(checked).not.toContain('task-0'); + // A bounded number goes: the oldest, task-59 down to task-10. expect(io.deletes[0]).toHaveLength(MAX_INPUT_MIRRORS_PER_PASS); + expect(io.deletes[0]?.[0]).toBe( + `/agent/inputs/task-${MAX_INPUT_MIRRORS_PER_PASS + 9}`, + ); + expect(io.deletes[0]).not.toContain('/agent/inputs/task-0'); // No reviews directory, so none is listed. expect(io.listings).toEqual(['/agent/inputs']); }); + it('old copies that are still needed never hide the stale ones behind them', async () => { + // The oldest 60 copies belong to tasks still open; the 5 newer ones are + // stale. + io.dirs['/agent/inputs'] = Array.from({ length: 65 }, (_, n) => + dir(`task-${n}`, n < 60 ? n : 10_000 + n), + ); + const { ctx } = makeCtx(({ taskIds }) => ({ + taskIds: taskIds.filter((id) => Number(id.slice('task-'.length)) >= 60), + reviewHashes: [], + })); + + await pruneStaleTaskInputMirrors(ctx, ARGS); + + expect(io.deletes[0]).toEqual([ + '/agent/inputs/task-60', + '/agent/inputs/task-61', + '/agent/inputs/task-62', + '/agent/inputs/task-63', + '/agent/inputs/task-64', + ]); + }); + it('asks nothing and removes nothing when the worker holds no other copy', async () => { io.dirs['/agent/inputs'] = [dir('task-current')]; const { ctx, asked } = makeCtx(() => ({ taskIds: [], reviewHashes: [] })); diff --git a/services/platform/backend/core/tasks/task_input_mirrors.ts b/services/platform/backend/core/tasks/task_input_mirrors.ts index 247b0976e6..c465361ae1 100644 --- a/services/platform/backend/core/tasks/task_input_mirrors.ts +++ b/services/platform/backend/core/tasks/task_input_mirrors.ts @@ -31,11 +31,17 @@ const REVIEWS_DIR_NAME = 'reviews'; const REVIEW_INPUTS_ROOT = `${TASK_INPUTS_ROOT}/${REVIEWS_DIR_NAME}`; -/** The most copies of each kind one pass looks at, the oldest first, so a +/** The most copies of each kind one pass removes, the oldest first, so a * worker that gathered many is cleared over several starts rather than * holding up one. */ export const MAX_INPUT_MIRRORS_PER_PASS = 50; +/** The most copies of each kind one pass asks about, the oldest first: far + * more than it removes, so copies that are still needed — the oldest are + * often long-open tasks — never hide the stale ones behind them. One + * indexed lookup answers them all. */ +const MAX_INPUT_MIRROR_CANDIDATES = 1000; + /** A directory name a task id can be; anything else under the root is left * alone, since nothing the platform stages is named so. */ const TASK_DIR_NAME_RE = /^[A-Za-z0-9_-]{1,128}$/; @@ -52,7 +58,7 @@ export function reviewInputsDir(taskId: string): string { return `${REVIEW_INPUTS_ROOT}/${hash}`; } -/** The names of up to {@link MAX_INPUT_MIRRORS_PER_PASS} directories that +/** The names of up to {@link MAX_INPUT_MIRROR_CANDIDATES} directories that * `accept` takes, the least recently changed first. */ function oldestDirs( entries: readonly SessionFsEntry[], @@ -61,7 +67,7 @@ function oldestDirs( return entries .filter((entry) => entry.type === 'dir' && accept(entry.name)) .sort((a, b) => a.mtimeMs - b.mtimeMs) - .slice(0, MAX_INPUT_MIRRORS_PER_PASS) + .slice(0, MAX_INPUT_MIRROR_CANDIDATES) .map((entry) => entry.name); } @@ -119,9 +125,14 @@ export async function pruneStaleTaskInputMirrors( taskIds, reviewHashes, }); + // The answer keeps the candidates' order, oldest first. const paths = [ - ...stale.taskIds.map((id) => `${TASK_INPUTS_ROOT}/${id}`), - ...stale.reviewHashes.map((hash) => `${REVIEW_INPUTS_ROOT}/${hash}`), + ...stale.taskIds + .slice(0, MAX_INPUT_MIRRORS_PER_PASS) + .map((id) => `${TASK_INPUTS_ROOT}/${id}`), + ...stale.reviewHashes + .slice(0, MAX_INPUT_MIRRORS_PER_PASS) + .map((hash) => `${REVIEW_INPUTS_ROOT}/${hash}`), ]; if (paths.length === 0) return; const removed = await sessionDeleteFiles(args.sessionId, paths); diff --git a/services/platform/backend/domains/tasks/agent-runs.test.ts b/services/platform/backend/domains/tasks/agent-runs.test.ts index 4540c63b61..f39e39fe63 100644 --- a/services/platform/backend/domains/tasks/agent-runs.test.ts +++ b/services/platform/backend/domains/tasks/agent-runs.test.ts @@ -315,6 +315,48 @@ describe('the turn host’s terminal marks write the provenance entry', () => { ); }); + it.each([ + [undefined, 2 * 60_000], + [1, 10 * 60_000], + [2, 30 * 60_000], + [5, 30 * 60_000], + ])( + 'holds the retry decision of a run out of memory (attempt %p) for %p ms', + async (autoRetryAttempt, waitMs) => { + const { sql } = fakeSql((text) => + text.startsWith('UPDATE app.project_agent_runs') + ? [ + { + organizationId: 'org-1', + taskId: 'task-1', + agentId: 'agent-1', + autoRetryAttempt: autoRetryAttempt ?? null, + }, + ] + : [], + ); + await failAgentRunFromTurn(sql, { + runId: 'run-1', + execId: 'exec-1', + error: "the agent's sandbox ran out of memory", + failureCode: 'resource_exhausted', + }); + // The job itself waits: no queued run sits out the wait for the + // stranded-queued-run sweep to start early. + expect(addJobInTx).toHaveBeenCalledExactlyOnceWith( + expect.anything(), + 'task.agent_retry', + { + organizationId: 'org-1', + taskId: 'task-1', + agentId: 'agent-1', + expectedRunId: 'run-1', + }, + { startAfter: new Date(NOW + waitMs) }, + ); + }, + ); + it('starts the model-capacity floor after a terminal-update lock wait, not before it', async () => { const { sql } = fakeSql((text) => { if (!text.startsWith('UPDATE app.project_agent_runs')) return []; diff --git a/services/platform/backend/domains/tasks/agent-runs.ts b/services/platform/backend/domains/tasks/agent-runs.ts index a65bb1db29..f4bd03aaa2 100644 --- a/services/platform/backend/domains/tasks/agent-runs.ts +++ b/services/platform/backend/domains/tasks/agent-runs.ts @@ -706,18 +706,30 @@ export async function failAgentRunFromTurn( const startAfterMs = args.failureCode === 'model_capacity' ? Date.now() + MODEL_CAPACITY_RETRY_DELAY_MS - : args.failureCode === 'resource_exhausted' - ? Date.now() + resourceExhaustedRetryDelayMs(run.autoRetryAttempt) - : args.retryAtMs !== undefined && args.retryAtMs > now - ? Math.min(args.retryAtMs, now + BROKER_RATE_LIMIT_COOLDOWN_MS) - : undefined; - await addJobInTx(tx, 'task.agent_retry', { + : args.retryAtMs !== undefined && args.retryAtMs > now + ? Math.min(args.retryAtMs, now + BROKER_RATE_LIMIT_COOLDOWN_MS) + : undefined; + // A run its sandbox's memory limit ended waits 2, 10, then 30 + // minutes — longer than the stranded-queued-run sweep lets a queued + // run wait, so the retry DECISION is held instead: no queued run + // exists meanwhile, the failed run shows its retry pending, and every + // guard is re-derived when the wait ends. + const decideAfter = + args.failureCode === 'resource_exhausted' + ? new Date( + Date.now() + resourceExhaustedRetryDelayMs(run.autoRetryAttempt), + ) + : undefined; + const retry = { organizationId: run.organizationId, taskId: run.taskId, agentId: run.agentId, expectedRunId: args.runId, ...(startAfterMs !== undefined && { startAfterMs }), - }); + }; + await (decideAfter !== undefined + ? addJobInTx(tx, 'task.agent_retry', retry, { startAfter: decideAfter }) + : addJobInTx(tx, 'task.agent_retry', retry)); } else { await announceAgentRunFailed(tx, { organizationId: run.organizationId, diff --git a/services/platform/messages/de/tasks.yml b/services/platform/messages/de/tasks.yml index b0709b357b..e46ce2119e 100644 --- a/services/platform/messages/de/tasks.yml +++ b/services/platform/messages/de/tasks.yml @@ -121,7 +121,7 @@ agentRun: model: Das KI-Modell hinter diesem Agenten ist ausgefallen, bevor die Arbeit erledigt war. Starte den Agenten erneut — schlägt er wieder fehl, bitte einen Admin, den KI-Anbieter zu prüfen. start: 'Der Lauf konnte nicht starten. Versuche es erneut — schlägt er wieder fehl, zeig einem Admin, was der Lauf unter "Details" gemeldet hat.' interrupted: Der Lauf wurde unterbrochen, bevor der Agent fertig war. Starte den Agenten erneut. - out_of_memory: Der Sandbox des Agenten ist der Arbeitsspeicher ausgegangen, deshalb wurde seine Arbeit gestoppt. Tale versucht es nach einer Pause erneut — passiert das wieder, bitte einen Admin, Agenten mehr Arbeitsspeicher zu geben. + out_of_memory: Der Sandbox des Agenten ist der Arbeitsspeicher ausgegangen, deshalb wurde seine Arbeit gestoppt. Starte den Agenten erneut — passiert das wieder, bitte einen Admin, Agenten mehr Arbeitsspeicher zu geben. stalled: 'Der Agent hat nicht mehr reagiert: Er hat lange nichts ausgegeben und kaum gearbeitet, deshalb hat seine Sandbox ihn gestoppt. Starte den Agenten erneut — bleibt er wieder stehen, zeig einem Admin, was der Lauf unter "Details" gemeldet hat.' unknown: 'Der Lauf ist fehlgeschlagen. Starte den Agenten erneut — schlägt er wieder fehl, zeig einem Admin, was der Lauf unter "Details" gemeldet hat.' standardAgent: diff --git a/services/platform/messages/en/tasks.yml b/services/platform/messages/en/tasks.yml index 97055d766a..1909275d14 100644 --- a/services/platform/messages/en/tasks.yml +++ b/services/platform/messages/en/tasks.yml @@ -123,7 +123,7 @@ agentRun: model: The AI model behind this agent failed before the work was done. Start the agent again — if it fails again, ask an Admin to check the AI provider. start: The run couldn't start. Try again — if it fails again, show an Admin what the run reported under Details. interrupted: The run was interrupted before the agent finished. Start the agent again. - out_of_memory: The agent's sandbox ran out of memory, so its work was stopped. Tale tries again after a pause — if it keeps happening, ask an Admin to give agents more memory. + out_of_memory: The agent's sandbox ran out of memory, so its work was stopped. Start the agent again — if it keeps happening, ask an Admin to give agents more memory. stalled: The agent stopped responding — it printed nothing and did almost no work for a long time — so its sandbox stopped it. Start the agent again; if it stops again, show an Admin what the run reported under Details. unknown: The run failed. Start the agent again — if it fails again, show an Admin what the run reported under Details. standardAgent: diff --git a/services/platform/messages/fr/tasks.yml b/services/platform/messages/fr/tasks.yml index 95455cae98..7fb6026c0b 100644 --- a/services/platform/messages/fr/tasks.yml +++ b/services/platform/messages/fr/tasks.yml @@ -122,7 +122,7 @@ agentRun: model: Le modèle d’IA derrière cet agent a échoué avant la fin du travail. Relance l’agent — s’il échoue encore, demande à un admin de vérifier le fournisseur d’IA. start: L’exécution n’a pas pu démarrer. Réessaie — si elle échoue encore, montre à un admin ce que l’exécution a signalé dans « Détails ». interrupted: L’exécution a été interrompue avant que l’agent ait terminé. Relance l’agent. - out_of_memory: La sandbox de l’agent a manqué de mémoire, son travail a donc été arrêté. Tale réessaie après une pause — si cela se reproduit, demande à un admin de donner plus de mémoire aux agents. + out_of_memory: La sandbox de l’agent a manqué de mémoire, son travail a donc été arrêté. Relance l’agent — si cela se reproduit, demande à un admin de donner plus de mémoire aux agents. stalled: L’agent ne répondait plus — il n’a rien affiché et n’a presque rien fait pendant longtemps —, sa sandbox l’a donc arrêté. Relance l’agent — s’il s’arrête encore, montre à un admin ce que l’exécution a signalé dans « Détails ». unknown: L’exécution a échoué. Relance l’agent — si elle échoue encore, montre à un admin ce que l’exécution a signalé dans « Détails ». standardAgent: diff --git a/services/platform/tests/manual/reference/automation/automations.md b/services/platform/tests/manual/reference/automation/automations.md index 0289639003..7b45b3a86e 100644 --- a/services/platform/tests/manual/reference/automation/automations.md +++ b/services/platform/tests/manual/reference/automation/automations.md @@ -6,7 +6,7 @@ Rows for [`automations`](../../suites/automations.md) boxes that moved out of th | Suite | Automated slice | Coverage | Specs | | --- | --- | --- | --- | | [automations](../../suites/automations.md) | Version picker remains at the right of the tabs at 320, 390 and 1280px, shows messages and test results, marks the selected version, and opens a version from General | ✅ automated | `app/features/automations/components/automation-editor.browser.test.tsx`; draft confirmation remains covered by `automation-editor.test.tsx` | -| [automations](../../suites/automations.md) / [tasks](../../suites/tasks.md) | An agent the sandbox's memory limit ended is named so: runnerd marks an exec that died of an unsent SIGKILL while the session's `memory.events` `oom_kill` rose as `oomKilled` (`OOM_KILLED`), the spawner reports a container Docker recorded as `OOMKilled` as `SESSION_OOM` on the exec it took down, and the platform settles both as `resource_exhausted` (the task card's out-of-memory sentence in EN/DE/FR), retried after 2, 10, then 30 minutes; `/healthz` reports the session's memory, peak and OOM kills | ✅ unit | `services/sandbox-runtime/daemon/src/exec-manager.stall.test.ts` (`ExecManager OOM attribution`), `services/sandbox-runtime/daemon/src/session-memory.test.ts`, `services/sandbox/src/session/session-routes.test.ts` (`OOM_KILLED`, `mid-exec`), `services/sandbox/src/backend/docker/docker-session-backend.test.ts` (`OOM killer hit`), `services/platform/backend/core/{tasks/agent_run_host,automations/agent_host,chat/external_turn_shared}.sandbox_ended.test.ts` | +| [automations](../../suites/automations.md) / [tasks](../../suites/tasks.md) | An agent the sandbox's memory limit ended is named so: runnerd marks an exec that died of an unsent SIGKILL while the session's `memory.events` `oom_kill` rose as `oomKilled` (`OOM_KILLED`), the spawner reports a container Docker recorded as `OOMKilled` as `SESSION_OOM` on the exec it took down, and the platform settles both as `resource_exhausted` (the task card's out-of-memory sentence in EN/DE/FR), a task run retried after 2, 10, then 30 minutes (the retry decision itself held, so the stranded-queued-run sweep cannot start it early) and an automation's agent node after 2 minutes, the longest a start may be held; an attach that outlives the container reports `SESSION_OOM` too; `/healthz` reports the session's memory, peak and OOM kills | ✅ unit | `services/sandbox-runtime/daemon/src/exec-manager.stall.test.ts` (`ExecManager OOM attribution`), `services/sandbox-runtime/daemon/src/session-memory.test.ts`, `services/sandbox/src/session/session-routes.test.ts` (`OOM_KILLED`, `mid-exec`), `services/sandbox/src/backend/docker/docker-session-backend.test.ts` (`OOM killer hit`), `services/platform/backend/core/{tasks/agent_run_host,automations/agent_host,chat/external_turn_shared}.sandbox_ended.test.ts` | | [automations](../../suites/automations.md) / [tasks](../../suites/tasks.md) | A running agent turn persists its live transcript at most once per 2 s (`LIVE_TRANSCRIPT_WRITE_FLOOR_MS`), the cadence its readers poll at: notifications arriving meanwhile fold into the one pending snapshot, and the settle's flush writes it at once | ✅ unit | `services/platform/backend/core/automations/agent_host.progress.test.ts` (`liveProgressSink cadence`) | | [automations](../../suites/automations.md) / [tasks](../../suites/tasks.md) | A session short of memory starts no new exec: runnerd reads the session cgroup's `memory.current`, `memory.max` and `memory.stat` and, once the working set (current minus `inactive_file`) reaches `TALE_EXEC_ADMISSION_MEMORY_PERCENT` (90, set by the spawner) of the limit, refuses `POST /execs` with 429 `session_memory_busy` and `retry-after: 5`, never touching running execs and never refusing without a limit; the spawner starts the exec before it answers and forwards the refusal as an HTTP 429 with the exec id free again; the platform's drain starts the exec again after the hint within its consecutive-failure budget, then hands the lanes a session-room refusal — a task run parks as **Waiting for room**, an automation turn waits as `sandbox_capacity` | ✅ unit + HTTP (fake cgroup files and spawner) | `services/sandbox-runtime/daemon/src/session-memory.test.ts`, `services/sandbox-runtime/daemon/src/main.test.ts`, `services/sandbox/src/session/runnerd-client.test.ts`, `services/sandbox/src/session/session-routes.test.ts`, `backend/core/node_only/sandbox/helpers/session_client.memory.test.ts`, `backend/core/tasks/run_park_reason.test.ts`, `backend/core/automations/agent_host.start_gate.test.ts`; a real session at its memory limit stays a deployment observation | | [automations](../../suites/automations.md) / [tasks](../../suites/tasks.md) | A hung agent ends: runnerd ends an exec that printed nothing and whose processes (its `/proc` tree plus the inner engine's containers) used under 1% of one CPU for `TALE_EXEC_STALL_MS` (the spawner's `SANDBOX_EXEC_STALL_MINUTES`, 45 by default, `0` off) through the cancel path and marks its exit `EXEC_STALLED`; the spawner reports a failed `EXEC_STALLED` result on the exec and attach streams, even after a clean exit; a task run settles `turn_stalled` with its localized "stopped responding" notice, no automatic retry and no fresh relaunch of a resume, and an automation turn settles `turn_stalled`, never re-kicked and reported as `turn_crashed` | ✅ unit (fake process table and clock) + real process | `services/sandbox-runtime/daemon/src/exec-stall.test.ts`, `services/sandbox-runtime/daemon/src/exec-manager.stall.test.ts`, `services/sandbox/src/session/runnerd-protocol.test.ts`, `services/sandbox/src/session/session-routes.test.ts`, `services/sandbox/src/config.test.ts`, `backend/core/chat/external_turn_shared.sandbox_ended.test.ts`, `backend/core/tasks/agent_run_host.sandbox_ended.test.ts`, `backend/core/automations/agent_host.sandbox_ended.test.ts`, `lib/shared/task-run-failure.test.ts`; a real hang in a live session stays a deployment observation | diff --git a/services/sandbox/README.md b/services/sandbox/README.md index 5f5c2b3ff1..ad42824fe0 100644 --- a/services/sandbox/README.md +++ b/services/sandbox/README.md @@ -155,10 +155,12 @@ An exec the kernel's OOM killer ended reads `failed` with the error code `OOM_KILLED` (runnerd saw its SIGKILL while the session counted a new OOM kill), and an exec whose session container died with it reads `SESSION_OOM` instead of `SESSION_LOST` when Docker recorded that the OOM killer hit the -container (`State.OOMKilled`, read by the inspect that evicts the dead -session). The platform settles such a run as `resource_exhausted` and retries -it after 2, 10, then 30 minutes rather than at once. Kubernetes restarts an -OOM-killed runner inside its Pod, so a session there ends as `SESSION_LOST`. +container (`State.OOMKilled`, read by the eviction's inspect — on the exec's +first stream and on every later attach). The platform settles such a run as +`resource_exhausted` and retries it after a pause rather than at once: a task +run 2, 10, then 30 minutes later, an automation's agent node 2 minutes later +(the longest a start may be held). Kubernetes restarts an OOM-killed runner +inside its Pod, so a session there ends as `SESSION_LOST`. Every session container has a CPU quota (`SANDBOX_AGENT_CPUS` for agents, one CPU for the `default` profile) and a CPU weight below the control plane's: diff --git a/tools/cli/scripts/cli-workflow.test.ts b/tools/cli/scripts/cli-workflow.test.ts index 1c7b3c0148..a1b386b169 100644 --- a/tools/cli/scripts/cli-workflow.test.ts +++ b/tools/cli/scripts/cli-workflow.test.ts @@ -492,6 +492,7 @@ describe('CLI command test targets', () => { 'deployment.test.ts', 'doctor.test.ts', 'hash-password.test.ts', + 'observation.test.ts', 'platform-configuration.test.ts', 'provision.test.ts', ]);