From 41021e3dbcbc1d2dede777d0cb721342e31e538a Mon Sep 17 00:00:00 2001 From: Louis Choquel Date: Sat, 26 Sep 2026 19:08:31 +0200 Subject: [PATCH 1/4] pin `@pipelex/sdk` at 98260ff0 for the sprint `pipelex-method-apps` takes `@pipelex/sdk` from `pipelex-sdk-js`'s feature/Failed-run-report-js, written at the dependency entry of webapp-js/package.json with the lock regenerated in the same commit (P1). The collapse before this branch merges is `wt unpin _pipelex-method-apps--failed-run-reason-shown pipelex-sdk-js --to ` (P2, P7). --- webapp-js/package-lock.json | 64 ++++++++++++++++++++++++++++++++++--- webapp-js/package.json | 2 +- 2 files changed, 61 insertions(+), 5 deletions(-) diff --git a/webapp-js/package-lock.json b/webapp-js/package-lock.json index 16c9d81..f9a0cef 100644 --- a/webapp-js/package-lock.json +++ b/webapp-js/package-lock.json @@ -9,7 +9,7 @@ "version": "0.5.5", "dependencies": { "@pipelex/mthds-form": "^0.10.0", - "@pipelex/sdk": "^0.25.1", + "@pipelex/sdk": "github:Pipelex/pipelex-sdk-js#98260ff0c1b25c8e0d8bea3ee8366094f9f30a65", "next": "^16.3.5", "react": "^19.2.4", "react-dom": "^19.2.4", @@ -1929,11 +1929,11 @@ }, "node_modules/@pipelex/sdk": { "version": "0.25.1", - "resolved": "https://registry.npmjs.org/@pipelex/sdk/-/sdk-0.25.1.tgz", - "integrity": "sha512-QjSx2/UAjrowvrspcNUhMTmOmiHkWqlB0duF/Xwl4fZynV+KbTX+Dub1W53zC9SmbEAEIOqwLvABxoxJTvCq6A==", + "resolved": "git+ssh://git@github.com/Pipelex/pipelex-sdk-js.git#98260ff0c1b25c8e0d8bea3ee8366094f9f30a65", + "integrity": "sha512-HiXBAfBtA6xd/I5KpYjEsbfZrINHQEldPvD8mUq1I9B4/ssCEPHkvZZpaGpt+3EdNZZF6Zln8veHw+wGPq/mcQ==", "license": "MIT", "dependencies": { - "mthds": "^0.25.0", + "mthds": "github:mthds-ai/mthds-js#d65d2e52070c6dac500a223c0fb3ba5377dfcd59", "smol-toml": "^1.6.0", "undici": "^7.29.1" }, @@ -1941,6 +1941,62 @@ "node": ">=22.12.0" } }, + "node_modules/@pipelex/sdk/node_modules/chalk": { + "version": "5.6.2", + "resolved": "https://registry.npmjs.org/chalk/-/chalk-5.6.2.tgz", + "integrity": "sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA==", + "license": "MIT", + "engines": { + "node": "^12.17.0 || ^14.13 || >=16.0.0" + }, + "funding": { + "url": "https://github.com/chalk/chalk?sponsor=1" + } + }, + "node_modules/@pipelex/sdk/node_modules/commander": { + "version": "13.1.0", + "resolved": "https://registry.npmjs.org/commander/-/commander-13.1.0.tgz", + "integrity": "sha512-/rFeCpNJQbhSZjGVwO9RFV3xPqbnERS8MmIQzCtD/zl6gpJuV/bMLuN92oG3F7d8oDEHHRrujSXNUr8fpjntKw==", + "license": "MIT", + "engines": { + "node": ">=18" + } + }, + "node_modules/@pipelex/sdk/node_modules/mthds": { + "version": "0.27.0", + "resolved": "git+ssh://git@github.com/mthds-ai/mthds-js.git#d65d2e52070c6dac500a223c0fb3ba5377dfcd59", + "integrity": "sha512-GGfu3yzo+Mlyj/Op/XQtoX50cy9ihAThE8K2QO1Ya1q3iwfiJ561IDdSpeAfO/hNPx+T3eon6ptQ+stWVCeFbw==", + "license": "MIT", + "dependencies": { + "@clack/prompts": "^1.0.0", + "chalk": "^5.4.1", + "commander": "^13.1.0", + "ora": "^8.2.0", + "posthog-node": "^4.4.0", + "semver": "^7.7.4", + "smol-toml": "^1.6.0", + "zod": "^4.3.6" + }, + "bin": { + "mthds": "dist/cli.js", + "mthds-agent": "dist/agent-cli.js" + }, + "engines": { + "node": ">=22" + } + }, + "node_modules/@pipelex/sdk/node_modules/semver": { + "version": "7.8.5", + "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.5.tgz", + "integrity": "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA==", + "license": "ISC", + "bin": { + "semver": "bin/semver.js" + }, + "engines": { + "node": ">=10" + } + }, "node_modules/@playwright/test": { "version": "1.59.1", "resolved": "https://registry.npmjs.org/@playwright/test/-/test-1.59.1.tgz", diff --git a/webapp-js/package.json b/webapp-js/package.json index 5cbb8c1..7646d17 100644 --- a/webapp-js/package.json +++ b/webapp-js/package.json @@ -27,7 +27,7 @@ }, "dependencies": { "@pipelex/mthds-form": "^0.10.0", - "@pipelex/sdk": "^0.25.1", + "@pipelex/sdk": "github:Pipelex/pipelex-sdk-js#98260ff0c1b25c8e0d8bea3ee8366094f9f30a65", "next": "^16.3.5", "react": "^19.2.4", "react-dom": "^19.2.4", From 7f4b76723181acc0c43f19b4c890b0459b982bf9 Mon Sep 17 00:00:00 2001 From: Louis Choquel Date: Sat, 26 Sep 2026 19:23:59 +0200 Subject: [PATCH 2/4] fix: show a failed durable run's reason, next step and retry advice from its stored report `pollDurableRun` now puts the run's stored error report (the result lookup's, else the status read's) on the `RunFailedError` it classifies, and passes the status read's `finished_at`. `classifyRunFailed` reads the display from the report: its title in the headline, its message with the provider's raw text cut out, its user action's detail as the hint, its `retryable` verdict as a retry line, and a support line with the run id, the error type and when the run ended. A run with no report keeps the SDK's sentence. `` renders the retry line and the support line, and `docs/errors.md` describes where a failed run's reason comes from. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01FigDssaJrvNcmbnBedi7oq --- CHANGELOG.md | 6 + .../skills/bootstrap/scripts/bootstrap.mjs | 1 + webapp-js/CLAUDE.md | 8 +- webapp-js/README.md | 1 + webapp-js/docs/errors.md | 36 ++++ .../src/components/ErrorDisplay.test.tsx | 79 +++++++++ webapp-js/src/components/ErrorDisplay.tsx | 22 ++- webapp-js/src/lib/durableRun.test.ts | 70 ++++++++ webapp-js/src/lib/durableRun.ts | 17 +- webapp-js/src/lib/errors.test.ts | 128 +++++++++++++++ webapp-js/src/lib/errors.ts | 155 +++++++++++++++++- webapp-js/src/test/fixtures/runReports.ts | 89 ++++++++++ 12 files changed, 596 insertions(+), 16 deletions(-) create mode 100644 webapp-js/docs/errors.md create mode 100644 webapp-js/src/components/ErrorDisplay.test.tsx create mode 100644 webapp-js/src/test/fixtures/runReports.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 51bf449..97f0cb2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,11 @@ # Changelog +## [Unreleased] + +### Fixed + +- **A failed durable run says why**: the web app template's failure display now reads a failed run's stored error report and shows the runtime's reason with the failing pipe, its advice as the next step, whether running it again can help, and a line to quote to support with the run id, the error type and when the run ended, where it used to show only "Run finished with status FAILED; no result available". A model provider's raw error text is kept out of everything the person can read, and a run that ended with no stored report keeps the old sentence. + ## [v0.5.5] - 2026-09-25 ### Fixed diff --git a/webapp-js/.claude/skills/bootstrap/scripts/bootstrap.mjs b/webapp-js/.claude/skills/bootstrap/scripts/bootstrap.mjs index 560c5d6..491de11 100644 --- a/webapp-js/.claude/skills/bootstrap/scripts/bootstrap.mjs +++ b/webapp-js/.claude/skills/bootstrap/scripts/bootstrap.mjs @@ -319,6 +319,7 @@ The server listens on loopback only, because anyone who can reach it runs method - [\`docs/add-method.md\`](docs/add-method.md) — adding a method, and removing one. - [\`docs/codegen.md\`](docs/codegen.md) — the generated types and the checks that keep them current. - [\`docs/input-form.md\`](docs/input-form.md) — how the input form and the result view are rendered from a method's contract. +- [\`docs/errors.md\`](docs/errors.md) — what a person reads when something fails, and where a failed run's reason comes from. - [\`CLAUDE.md\`](CLAUDE.md) — the project guide for coding agents. ## License diff --git a/webapp-js/CLAUDE.md b/webapp-js/CLAUDE.md index 7125cf5..20dcb39 100644 --- a/webapp-js/CLAUDE.md +++ b/webapp-js/CLAUDE.md @@ -77,10 +77,11 @@ src/ ResultEnv.tsx # client component — the kernel's ResultEnvProvider with this app's resolvers, mounted in the layout HydrationMark.tsx # client component — sets html[data-hydrated] for a browser script to wait on, mounted in the layout CostReport.tsx # per-run token usage + cost breakdown, inside RunDetails' disclosure - ErrorDisplay.tsx # server component (renders classified PipelineError) + ErrorDisplay.tsx # server component (renders classified PipelineError: reason, next step, retry line, support line) types/ pipelineError.ts # BadPipelineOutputError (tagged) test/fixtures/contracts/ # recorded codegen output the shared code's tests run against + test/fixtures/runReports.ts # failed runs' stored error reports the failure display's tests run against e2e/ home.spec.ts # offline — the page renders its title and either the empty state or a form liveApi.ts # requireLiveApi() — the key guard every live spec calls @@ -121,7 +122,7 @@ The slice is the same for every source kind; only `methods//` and the way - **`src/hooks/`** — `useRun`, the unified client state machine (`idle → running → done|error`) that dispatches blocking vs durable by `mode`. Holds the durable poll loop, the staleness token, the elapsed ticker, the wall-clock ceiling, the `classifyTransportError` wrapping, and the **transient-failure budget** (a momentary 5xx/network blip on one poll tick — flagged `transient` by `pollDurableRun` — or a rejected poll await is retried up to `MAX_TRANSIENT_POLL_FAILURES`, surfacing `health: "retrying"` meanwhile, rather than abandoning a run that's still completing server-side). The running state's `health` field (`RunHealth | null`) names _why_ the poll loop is in a resilient state so `` can show reassuring, cause-specific copy instead of one alarming "degraded" note: `"reconnecting"` when the **server** reported `degraded` (its status endpoint served a last-known DB status because Temporal was unreachable), `"retrying"` for a **client-side** poll blip, `null` when polling cleanly. A durable `start` that returns `lifecycle_unavailable` (the configured URL doesn't serve the durable run lifecycle) surfaces as an explicit error — `useRun` never silently downgrades durable to blocking. Forms never branch on mode — they just call `run(input)`. `useFileInputs` is the file seam — drop, size early-exit, ask the method's grant action, send the file to storage with the SDK's `uploadWithGrant` (from the browser-safe `@pipelex/sdk/upload`), write the stored reference back at the field's dotted path, and hold the id in the set the kernel reads as `uploadingIds` meanwhile — so a form with a file input composes it rather than restating it. - **`src/components/`** — React components. `"use client"` only when the component uses hooks, event handlers, or browser APIs (`RunDetails` does; `RunStatus` is a pure render). `RunResult` is the one kernel composition on the output side — `RunInputsForm`'s twin — and every method's form renders it: the kernel's `` under the same `presentation="app"` the form uses, inside a labelled `
` so the result is a region a screen reader can name and jump to. There is no per-output-shape component, and that is the point: a result view stops being a design decision the app has to take once the method's own declaration of what it produces is a committed artifact. `ResultEnv` is the kernel's `ResultEnvProvider` carrying this app's two resolvers, mounted once in the root layout — display through `assetPath`, sharing through `resolveShareUrl` — so every file arm in every result view paints a stored reference through the assets route. - **`src/types/`** — the **adapter layer over `src/generated/`**, not a place where shapes are declared. Each `parseXxx(results: RunResults)` hands `wireOutput(results)` to the binder generated from that method's own contract, and translates a thrown `ZodError` into the app's tagged error model; the type itself is re-exported from the generated `types.ts`. Hand-written validation belongs here only where it adds semantics the concept does not declare. Narrowers throw on mismatch; that's deliberate (system boundary). -- **`src/test/fixtures/contracts/`** — recorded `contracts.ts` files, real codegen output for methods this app does not ship, so the shared code's tests (`runInputs`, `resultField`, `RunResult`, `ResultEnv`) run against the shapes a real method produces. They are test data, never imported by app code. +- **`src/test/fixtures/contracts/`** — recorded `contracts.ts` files, real codegen output for methods this app does not ship, so the shared code's tests (`runInputs`, `resultField`, `RunResult`, `ResultEnv`) run against the shapes a real method produces. They are test data, never imported by app code. `src/test/fixtures/runReports.ts` holds failed runs' stored error reports, one recorded and two in the runtime's shapes, for the tests of the failed-run display (`errors`, `durableRun`, `ErrorDisplay`). ## Generated types (`src/generated/`) @@ -237,7 +238,8 @@ Conventions: - **One client**: instantiate `PipelexApiClient` once via `getPipelexClient()`. Never `new PipelexApiClient()` directly in actions or components. - **Narrow at the boundary, but never re-declare the shape**: the SDK returns loosely-typed output, so always pass the whole `RunResults` through a `parseXxx(results)` narrower in `src/types/`. That narrower hands `wireOutput(results)` to the generated binder and translates the thrown `ZodError` into a tagged subclass of `Error` (`BadPipelineOutputError`) via `describeSchemaFailure`. Do not `as` your way through, and do not hand-write the shape it validates — the method already declares it and `npm run codegen` projects it. - **Return classified errors, don't throw across the server→client boundary**: the shared helpers return `{ ok: true, ... } | { ok: false, error: PipelineError }`. Throwing works in dev but Next.js production builds strip server-action error messages to opaque digests, which destroys the developer-facing error UX. `executeBlockingRun` / `startDurableRun` / `pollDurableRun` wrap the SDK call in `try/catch`, hand the caught value to `classifyPipelineError(err, env)`, and return the structured error. Render it client-side with ``. -- **Classification stays server-side, in the helpers.** `classifyPipelineError` `instanceof`-matches SDK error classes, which only exist server-side (they're stripped to opaque digests crossing the boundary) — so it runs inside the helpers, never on a poll/blocking result the client received. The durable `failed` poll constructs a `RunFailedError` from the result lookup and classifies it there too. +- **Classification stays server-side, in the helpers.** `classifyPipelineError` `instanceof`-matches SDK error classes, which only exist server-side (they're stripped to opaque digests crossing the boundary) — so it runs inside the helpers, never on a poll/blocking result the client received. The durable `failed` poll constructs a `RunFailedError` carrying the run's **stored error report** — the result lookup's `error`, else the status read's, which the hosted platform serves even where its result lookup does not yet — and classifies it there too, passing the status read's `finished_at`. +- **A failed run is shown from its report, never from the lookup's sentence.** `classifyRunFailed` puts the report's `title` in the headline, its `message` as what happened, its `user_action.detail` as the hint, its `retryable` verdict as `retry` (absent when the report has none, so nothing is claimed on a guess), and a `support` line with the run id, the `error_type` and when the run ended, which `` shows in place of the bare run id. **The provider's raw text never reaches the person**: the report's message quotes the provider SDK's text verbatim, which can be a raw error body or a whole HTML page, and the report names it as `provider_metadata.message`, so `visibleReportMessage` cuts it out and `provider_metadata` is left out of the technical details too. A run with no report — cancelled, terminated, timed out, finalized by the platform, or on a platform without the report — keeps the SDK's sentence. [`docs/errors.md`](docs/errors.md) is the reference. - **Inputs are gated, never hand-guarded.** Every action starts with `gateRunInputs(CONTRACT, data)` over the method's committed contract — the same gate the browser ran for the Run button — and returns its `{ ok: false, error }` unchanged. Do not add a per-input `if (!x) return badRequest()` beside it. What legitimately sits _after_ the gate is a check the contract cannot express — the file-reference scheme check is the one example, and it runs over the _gated_ inputs. A file's type and size are checked earlier still, by the grant action, before the file is stored. - **Add new error kinds in `src/lib/errors.ts`**: extend `PipelineErrorKind`, add an `instanceof` branch in `classifyPipelineError` (import the class from `@pipelex/sdk`), and cover it in `src/lib/errors.test.ts`. Keep `classifyPipelineError` pure — env passed in by caller, no `process.env` reads inside. The dual-mode kinds (`execute_timeout`, `run_still_running`, `run_failed`, `run_timeout`, `lifecycle_unavailable`) follow this pattern. Two exceptions build a `PipelineError` inline (no thrown error to classify): pre-flight validation (`file_too_large`, `unsupported_file_type`, `bad_request`) in a Server Action or in `useFileInputs`'s size check, and the client-side poll ceiling (`buildClientTimeoutError`, kind `run_timeout`) in `useRun`. - **`lifecycle_unavailable` has two sources.** A 404 from a URL that doesn't serve the run-lifecycle routes arrives as the SDK's `RunLifecycleUnavailableError` (`instanceof` branch → `classifyLifecycleUnavailable`); a `/start` against a deployment whose orchestrator is blocking-only (the in-process `direct` mode) arrives as a 400 `ApiResponseError` with `error_type: "StartRequiresAsyncOrchestration"`, matched by an `errorType` branch in `classifyResponse` → `classifyStartRequiresAsync`. Both restate the runtime's vocabulary ("orchestration mode", "fire-and-forget") in this app's term — **durable execution** — and frame the configured URL as the problem, steering to `PIPELEX_BASE_URL`; the messages differ because the root causes differ. diff --git a/webapp-js/README.md b/webapp-js/README.md index 7063b98..76ecd6e 100644 --- a/webapp-js/README.md +++ b/webapp-js/README.md @@ -112,6 +112,7 @@ Next.js 16 (App Router), React 19, TypeScript 5 (strict), Tailwind CSS 4 (config - [`docs/add-method.md`](docs/add-method.md) — adding a method, and removing one. - [`docs/codegen.md`](docs/codegen.md) — the generated types and the checks that keep them current. - [`docs/input-form.md`](docs/input-form.md) — how the input form and the result view are rendered from a method's contract. +- [`docs/errors.md`](docs/errors.md) — what a person reads when something fails, and where a failed run's reason comes from. - [`docs/ci.md`](docs/ci.md) — what the pull-request checks prove, and how `make create` is proven against the live API. - [`docs/chrome-lineage.md`](docs/chrome-lineage.md) — what this template took from the gallery, and what it changed. - [`CLAUDE.md`](CLAUDE.md) — the project guide for coding agents. diff --git a/webapp-js/docs/errors.md b/webapp-js/docs/errors.md new file mode 100644 index 0000000..47d53e6 --- /dev/null +++ b/webapp-js/docs/errors.md @@ -0,0 +1,36 @@ +# Errors: what a person reads when something goes wrong + +Every failure this app can meet reaches the person using it as one `PipelineError`, rendered by ``. The error is classified on the server, inside the helpers that call the SDK (`executeBlockingRun`, `startDurableRun`, `pollDurableRun`, the grant action), by `classifyPipelineError` in `src/lib/errors.ts`, because the SDK's error classes only match with `instanceof` there. The helpers return the classified error instead of throwing it, since a production build of Next.js turns a thrown Server Action error into an opaque digest. + +## The fields of a classified error + +| Field | What `` does with it | +| ------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `title` | The headline. | +| `message` | One or two sentences on what happened, in plain language. | +| `apiMessage` | The API's own words, in a block of their own, when `message` re-frames them. | +| `hint` | The next step, with an optional snippet to copy and a link. | +| `retry` | Whether running it again can succeed, as the runtime judged it: a line offering a re-run when it can, and saying a re-run unchanged will fail when it cannot. Absent when nobody said, and then nothing is claimed. | +| `support` | One line to quote to support, shown selectable in place of the bare run id. | +| `details` | The raw technical facts, in a "Technical details" disclosure that starts closed. | + +## A failed durable run + +A durable run that ends without a result is read from **its stored error report**, the runtime's own account of the failure. The platform stores the report on the run and serves it on the status read (`RunRead.error`) and in the result lookup's refusal, which `@pipelex/sdk` hands back on the failed arm of `getRunResult`. `pollDurableRun` takes the result lookup's report, or the status read's when the lookup has none (a platform that does not serve it there yet), puts it on the `RunFailedError` it classifies, and passes the status read's `finished_at` along. The classification reads each part of the display from the report: + +| The display | From the report | +| ----------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| The headline | "The pipeline run failed: " and the report's `title`, the stable label of the error class. | +| What happened | The report's `message`, which names the failing pipe, without the provider's raw text (below). | +| The next step | `user_action.detail`, the runtime's advice: change an input, choose another model, check billing, wait and retry. | +| The retry line | `retryable`. True offers a re-run; false says a re-run unchanged fails the same way; absent says nothing. | +| The support line | The run id, `error_type` and the time the run ended, as `run · · failed `. | +| Technical details | The run's status and the report's classification (`error_type`, `error_domain`, `error_category`, `retryable`, the user action's kind, `model`, `type_uri`), the message shown above, and the validation items of a method that failed validation. | + +**The provider's raw text never reaches the person.** When a model provider refused a call, the runtime writes the provider SDK's text into the report's message verbatim, and that text can be the raw body of the provider's error, naming the deployment's own provider account, or a whole HTML page from an edge in front of the provider. The report names that text as `provider_metadata.message`, so the classification cuts it out of the message wherever it appears, and leaves `provider_metadata` out of the technical details as well. What remains still names the pipe, the provider, the model and the HTTP status: `Pipe 'summarize' (path: two_steps > summarize) failed: openai inference failed for model 'claude-4.8-opus' (HTTP 412)`. + +**A run with no report keeps the SDK's sentence**, "Run finished with status FAILED; no result available", with the bare run id. That is every cancelled, terminated or timed-out run, every run the platform finalized itself, and every failed run on a platform that serves no report; the absence of a report says nothing about why. + +## Other failures + +A run that fails on the blocking path comes back as a refused request, an `ApiResponseError`, and is classified by its HTTP status like any other refusal. Every other kind (an unreachable API, a missing key, a durable-run lifecycle the URL does not serve, a file upload that failed, an output that does not match the method's contract) has its own branch in `classifyPipelineError`, and [`CLAUDE.md`](../CLAUDE.md) says how to add one. diff --git a/webapp-js/src/components/ErrorDisplay.test.tsx b/webapp-js/src/components/ErrorDisplay.test.tsx new file mode 100644 index 0000000..171cae9 --- /dev/null +++ b/webapp-js/src/components/ErrorDisplay.test.tsx @@ -0,0 +1,79 @@ +import { describe, it, expect } from "vitest"; +import { render, screen } from "@testing-library/react"; +import { RunFailedError, type RunErrorReport } from "@pipelex/sdk"; +import { classifyPipelineError } from "@/lib/errors"; +import { + MODEL_NOT_ENABLED, + MODEL_NOT_ENABLED_PROVIDER_TEXT, + RATE_LIMITED, + WRONG_ITEM_COUNT, +} from "@/test/fixtures/runReports"; +import { ErrorDisplay } from "./ErrorDisplay"; + +const ENV = { apiUrl: undefined, hasApiKey: true }; +const FINISHED_AT = "2026-09-26T10:01:00+00:00"; + +/** A failed durable run as `pollDurableRun` classifies it, then shows it. */ +function showFailedRun(report: RunErrorReport | null, message: string) { + const error = classifyPipelineError( + new RunFailedError(message, "run-1", "FAILED", { error: report }), + ENV, + { finishedAt: FINISHED_AT }, + ); + return render(); +} + +describe("ErrorDisplay — a failed run", () => { + it("shows the report's reason, next step and support line, and offers no re-run for a change_input fault", () => { + showFailedRun(WRONG_ITEM_COUNT, `Run finished with status FAILED: ${WRONG_ITEM_COUNT.message}`); + const alert = screen.getByRole("alert"); + expect( + screen.getByRole("heading", { + name: "The pipeline run failed: Multiplicity count mismatch", + }), + ).toBeInTheDocument(); + expect(screen.getByText(WRONG_ITEM_COUNT.message as string)).toBeInTheDocument(); + expect(screen.getByText("Provide exactly 2 items for input 'pages'.")).toBeInTheDocument(); + expect( + screen.getByText("Running it again unchanged will fail the same way."), + ).toBeInTheDocument(); + expect(alert).not.toHaveTextContent(/run it again/); + expect( + screen.getByText("run run-1 · MultiplicityCountMismatchError · failed 2026-09-26T10:01:00Z"), + ).toBeInTheDocument(); + expect(alert).toHaveTextContent(/For support:/); + expect(alert).not.toHaveTextContent(/no result available/); + }); + + it("offers a re-run for a retryable failure", () => { + showFailedRun(RATE_LIMITED, `Run finished with status FAILED: ${RATE_LIMITED.message}`); + expect( + screen.getByText("This failure can pass on a second try: run it again."), + ).toBeInTheDocument(); + }); + + it("never shows the provider's raw text, even under the technical details", () => { + showFailedRun( + MODEL_NOT_ENABLED, + `Run finished with status FAILED: ${MODEL_NOT_ENABLED.message}`, + ); + const alert = screen.getByRole("alert"); + expect(alert).toHaveTextContent( + "This model is not enabled on the inference gateway; choose another model for the pipe.", + ); + expect(alert.textContent).not.toContain(MODEL_NOT_ENABLED_PROVIDER_TEXT); + expect(alert).not.toHaveTextContent(/not allowed for this integration/); + }); + + it("keeps today's sentence and the bare run id for a run with no report", () => { + showFailedRun(null, "Run finished with status FAILED; no result available"); + const alert = screen.getByRole("alert"); + expect(screen.getByRole("heading", { name: "The pipeline run failed" })).toBeInTheDocument(); + expect( + screen.getByText("Run finished with status FAILED; no result available"), + ).toBeInTheDocument(); + expect(alert).not.toHaveTextContent(/For support:/); + expect(alert).not.toHaveTextContent(/run it again|fail the same way/); + expect(alert).toHaveTextContent(/Run run-1/); + }); +}); diff --git a/webapp-js/src/components/ErrorDisplay.tsx b/webapp-js/src/components/ErrorDisplay.tsx index 5e1550f..25bd148 100644 --- a/webapp-js/src/components/ErrorDisplay.tsx +++ b/webapp-js/src/components/ErrorDisplay.tsx @@ -10,6 +10,12 @@ interface ErrorDisplayProps { runId?: string | null; } +/** + * A classified error, as the person using the app reads it: what happened, + * the next step, whether running it again can help, and what to quote to + * support — the run's id alone, or the error's own support line when it has + * one, which names the run with what failed and when. + */ export function ErrorDisplay({ error, runId }: ErrorDisplayProps) { return (
)} - {runId && ( + {error.retry && ( +

+ {error.retry.summary} +

+ )} + + {error.support ? (

- Run {runId} + For support: {error.support}

+ ) : ( + runId && ( +

+ Run {runId} +

+ ) )}
diff --git a/webapp-js/src/lib/durableRun.test.ts b/webapp-js/src/lib/durableRun.test.ts index 176d972..75205b5 100644 --- a/webapp-js/src/lib/durableRun.test.ts +++ b/webapp-js/src/lib/durableRun.test.ts @@ -15,6 +15,7 @@ vi.mock("@/lib/pipelexClient", () => ({ })); import { pollDurableRun, startDurableRun } from "./durableRun"; +import { MODEL_NOT_ENABLED, RATE_LIMITED, WRONG_ITEM_COUNT } from "@/test/fixtures/runReports"; import { BadPipelineOutputError } from "@/types/pipelineError"; beforeEach(() => { @@ -164,6 +165,75 @@ describe("pollDurableRun", () => { expect(result.transient).toBe(false); // a real run failure is terminal }); + it("classifies a failed run from the report the results read carries, dated by the status read", async () => { + getRunStatus.mockResolvedValueOnce({ + status: "FAILED", + degraded: false, + finished_at: "2026-09-26T10:01:00+00:00", + error: WRONG_ITEM_COUNT, + }); + getRunResult.mockResolvedValueOnce({ + state: "failed", + pipeline_run_id: "run-1", + status: "FAILED", + message: `Run finished with status FAILED: ${MODEL_NOT_ENABLED.message}`, + error: MODEL_NOT_ENABLED, + }); + const result = await pollDurableRun("run-1", parseFixture); + expect(result.ok).toBe(false); + if (result.ok) return; + expect(result.transient).toBe(false); + expect(result.error.title).toBe("The pipeline run failed: LLM completion"); + expect(result.error.hint?.summary).toBe(MODEL_NOT_ENABLED.user_action?.detail); + expect(result.error.retry?.retryable).toBe(false); + expect(result.error.support).toBe( + "run run-1 · LLMCompletionError · failed 2026-09-26T10:01:00Z", + ); + expect(JSON.stringify(result.error)).not.toContain("not allowed for this integration"); + }); + + it("takes the report from the status read on a platform whose results read has none", async () => { + getRunStatus.mockResolvedValueOnce({ + status: "FAILED", + degraded: false, + finished_at: null, + error: RATE_LIMITED, + }); + getRunResult.mockResolvedValueOnce({ + state: "failed", + pipeline_run_id: "run-1", + status: "FAILED", + message: "Run finished with status FAILED; no result available", + error: null, + }); + const result = await pollDurableRun("run-1", parseFixture); + expect(result.ok).toBe(false); + if (result.ok) return; + expect(result.error.message).not.toContain("no result available"); + expect(result.error.hint?.summary).toBe(RATE_LIMITED.user_action?.detail); + expect(result.error.retry?.retryable).toBe(true); + expect(result.error.support).toBe("run run-1 · LLMCompletionError"); + }); + + it("keeps the lookup's sentence for a run that ended with no report", async () => { + getRunStatus.mockResolvedValueOnce({ status: "CANCELLED", degraded: false, error: null }); + getRunResult.mockResolvedValueOnce({ + state: "failed", + pipeline_run_id: "run-1", + status: "CANCELLED", + message: "Run finished with status CANCELLED; no result available", + error: null, + }); + const result = await pollDurableRun("run-1", parseFixture); + expect(result.ok).toBe(false); + if (result.ok) return; + expect(result.error.title).toBe("The pipeline run failed"); + expect(result.error.message).toBe("Run finished with status CANCELLED; no result available"); + expect(result.error.hint).toBeUndefined(); + expect(result.error.retry).toBeUndefined(); + expect(result.error.support).toBeUndefined(); + }); + it("re-reports running on the mid-write race (terminal status, result still running)", async () => { getRunStatus.mockResolvedValueOnce({ status: "COMPLETED", degraded: false }); getRunResult.mockResolvedValueOnce({ diff --git a/webapp-js/src/lib/durableRun.ts b/webapp-js/src/lib/durableRun.ts index 80b6009..377efc4 100644 --- a/webapp-js/src/lib/durableRun.ts +++ b/webapp-js/src/lib/durableRun.ts @@ -76,8 +76,10 @@ export async function startDurableRun( * status + degraded flag + the server's `Retry-After` hint. * 2. On a terminal status, `getRunResult`: * - `completed` → narrow `result` and report `completed`. - * - `failed` → classify a constructed `RunFailedError` (the status read - * has no failure message; the result lookup does). + * - `failed` → classify a constructed `RunFailedError` carrying the run's + * stored error report (the result lookup's, else the status + * read's), so the person sees the runtime's reason, next + * step and retry verdict rather than the lookup's sentence. * - `running` → mid-write race (status flipped terminal but `main_stuff` / * `graph_spec` aren't written yet) → report `running` so the * client polls once more. @@ -116,12 +118,19 @@ export async function pollDurableRun( }; } if (res.state === "failed") { - // A genuine run failure is terminal — never retried. + // A genuine run failure is terminal — never retried. Its stored error + // report says why: the results read carries it on a platform that serves + // it there, and the status read just made carries it on any hosted + // platform, even one whose results read does not yet, so the reason + // reaches the person either way. The status read also says when the run + // ended, for the support line. + const report = res.error ?? read.error ?? null; return { ok: false, error: classifyPipelineError( - new RunFailedError(res.message, runId, res.status), + new RunFailedError(res.message, runId, res.status, { error: report }), readClassifyEnv(), + { finishedAt: read.finished_at }, ), transient: false, }; diff --git a/webapp-js/src/lib/errors.test.ts b/webapp-js/src/lib/errors.test.ts index fcf76a5..29ff943 100644 --- a/webapp-js/src/lib/errors.test.ts +++ b/webapp-js/src/lib/errors.test.ts @@ -14,8 +14,16 @@ import { UnsupportedUploadCapabilityError, UploadAuthenticationError, UploadTransportError, + type RunErrorReport, type UploadTransportCode, } from "@pipelex/sdk"; +import { + MODEL_NOT_ENABLED, + MODEL_NOT_ENABLED_PROVIDER_TEXT, + RATE_LIMITED, + RATE_LIMITED_PROVIDER_TEXT, + WRONG_ITEM_COUNT, +} from "@/test/fixtures/runReports"; import { BadPipelineOutputError } from "@/types/pipelineError"; import { buildClientTimeoutError, @@ -457,6 +465,126 @@ describe("classifyPipelineError — run-lifecycle errors", () => { }); }); +describe("classifyPipelineError — a failed run's stored error report", () => { + // The results read's `detail`, as a platform serving the report writes it: + // the status, then the report's message — provider text included — and the + // sentence it writes for a run with no report. + const detailFor = (report: RunErrorReport) => + `Run finished with status FAILED: ${report.message}`; + const NO_REPORT_SENTENCE = "Run finished with status FAILED; no result available"; + const failed = (report: RunErrorReport | null) => + new RunFailedError(report ? detailFor(report) : NO_REPORT_SENTENCE, "run-1", "FAILED", { + error: report, + }); + + it("reads a change_input failure from its report: reason, next step, no re-run, support line", () => { + const result = classifyPipelineError(failed(WRONG_ITEM_COUNT), OVERRIDE_ENV, { + finishedAt: "2026-09-26T10:01:00.482913+00:00", + }); + expect(result.kind).toBe("run_failed"); + expect(result.title).toBe("The pipeline run failed: Multiplicity count mismatch"); + expect(result.message).toBe(WRONG_ITEM_COUNT.message); + expect(result.hint).toEqual({ summary: "Provide exactly 2 items for input 'pages'." }); + expect(result.retry).toEqual({ + retryable: false, + summary: "Running it again unchanged will fail the same way.", + }); + expect(result.support).toBe( + "run run-1 · MultiplicityCountMismatchError · failed 2026-09-26T10:01:00Z", + ); + expect(result.details).toContain("run run-1 ended FAILED"); + expect(result.details).toContain("error_domain: input"); + expect(result.details).toContain("user_action: change_input"); + expect(result.message).not.toContain("no result available"); + }); + + it("offers a re-run when the report says the failure is retryable", () => { + const result = classifyPipelineError(failed(RATE_LIMITED), OVERRIDE_ENV); + expect(result.retry).toEqual({ + retryable: true, + summary: "This failure can pass on a second try: run it again.", + }); + expect(result.hint?.summary).toBe(RATE_LIMITED.user_action?.detail); + }); + + it("keeps the provider's raw text out of everything the person can read", () => { + const result = classifyPipelineError(failed(MODEL_NOT_ENABLED), OVERRIDE_ENV); + expect(result.title).toBe("The pipeline run failed: LLM completion"); + // What remains of the runtime's message still names the pipe, the model and the status. + expect(result.message).toBe( + "Pipe 'summarize' (path: two_steps > summarize) failed: openai inference failed for model 'claude-4.8-opus' (HTTP 412)", + ); + expect(result.hint?.summary).toBe( + "This model is not enabled on the inference gateway; choose another model for the pipe.", + ); + expect(result.retry?.retryable).toBe(false); + const shown = JSON.stringify(result); + expect(shown).not.toContain(MODEL_NOT_ENABLED_PROVIDER_TEXT); + expect(shown).not.toContain("not allowed for this integration"); + expect(shown).not.toContain("req_gw_5f1c"); + expect(result.details).toContain("model: claude-4.8-opus"); + }); + + it("claims nothing about retrying when the report has no verdict", () => { + // `null` and an absent field both mean unknown, never "no". + const result = classifyPipelineError( + failed({ ...WRONG_ITEM_COUNT, retryable: null }), + OVERRIDE_ENV, + ); + expect(result.retry).toBeUndefined(); + expect(result.details).not.toContain("retryable"); + }); + + it("leaves the time out of the support line when the run's end is unknown", () => { + const result = classifyPipelineError(failed(WRONG_ITEM_COUNT), OVERRIDE_ENV); + expect(result.support).toBe("run run-1 · MultiplicityCountMismatchError"); + }); + + it("says the report is silent rather than showing the provider's text when nothing else is left", () => { + const bare: RunErrorReport = { + error_type: "LLMCompletionError", + message: RATE_LIMITED_PROVIDER_TEXT, + provider_metadata: { message: RATE_LIMITED_PROVIDER_TEXT }, + }; + const result = classifyPipelineError(failed(bare), OVERRIDE_ENV); + expect(result.title).toBe("The pipeline run failed"); + expect(result.message).toBe("The run ended FAILED, and its error report does not say why."); + expect(result.hint).toBeUndefined(); + expect(JSON.stringify(result)).not.toContain("rate_limit_exceeded"); + }); + + it("lists a method's validation items under the technical details", () => { + const report: RunErrorReport = { + error_type: "PipeValidationError", + message: "The method failed validation.", + title: "Pipe validation", + validation_errors: [ + { + category: "pipe_validation", + message: "Output multiplicity does not match the parallel branches.", + pipe_code: "fan_out", + }, + ], + }; + const result = classifyPipelineError(failed(report), OVERRIDE_ENV); + expect(result.details).toContain( + "validation: fan_out: Output multiplicity does not match the parallel branches.", + ); + }); + + it("keeps today's wording for a failed run with no stored report", () => { + const result = classifyPipelineError(failed(null), OVERRIDE_ENV, { + finishedAt: "2026-09-26T10:01:00+00:00", + }); + expect(result).toEqual({ + kind: "run_failed", + title: "The pipeline run failed", + message: NO_REPORT_SENTENCE, + details: `RunFailedError: run run-1 ended FAILED\n${NO_REPORT_SENTENCE}`, + }); + }); +}); + describe("classifyPipelineError — input-preparation (upload) errors", () => { it("classifies UnsupportedUploadCapabilityError into upload_failed pointing at the hosted API", () => { const err = new UnsupportedUploadCapabilityError("no /v1/upload route"); diff --git a/webapp-js/src/lib/errors.ts b/webapp-js/src/lib/errors.ts index caaeff7..25152dd 100644 --- a/webapp-js/src/lib/errors.ts +++ b/webapp-js/src/lib/errors.ts @@ -20,6 +20,7 @@ import { UnsupportedUploadCapabilityError, UploadAuthenticationError, UploadTransportError, + type RunErrorReport, } from "@pipelex/sdk"; import { BadPipelineOutputError } from "@/types/pipelineError"; @@ -93,10 +94,28 @@ export interface PipelineError { */ apiMessage?: string; hint?: ErrorHint; + /** + * Whether running it again can succeed, as the runtime judged it. Set only + * from a verdict — a failed run's report carries `retryable` — and absent + * when nobody said, so the display never claims either way on a guess. + */ + retry?: RetryAdvice; + /** + * One line a person quotes to support: the run's id, what failed and when. + * `` shows it selectable in place of the bare run id. + */ + support?: string; /** Raw technical info for the collapsible "Technical details" section. */ details: string; } +export interface RetryAdvice { + /** True when running it again can succeed: the display offers a re-run. */ + retryable: boolean; + /** The sentence that says so. */ + summary: string; +} + export interface ClassifyEnv { apiUrl: string | undefined; hasApiKey: boolean; @@ -120,6 +139,12 @@ export interface ClassifyOptions { * a declared size over its limit, which is the authority on "too large". */ uploadGrant?: boolean; + /** + * Set by `pollDurableRun` for a run that ended without a result: when it + * ended, from the run's status read (`finished_at`). A failed run's support + * line carries it, because it is what finds the run in the server's logs. + */ + finishedAt?: string | null; } export function classifyPipelineError( @@ -141,7 +166,7 @@ export function classifyPipelineError( // are distinct concrete classes, so order among them is irrelevant). if (err instanceof PipelineExecuteTimeoutError) return classifyExecuteTimeout(err); if (err instanceof RunStillRunningError) return classifyRunStillRunning(err); - if (err instanceof RunFailedError) return classifyRunFailed(err); + if (err instanceof RunFailedError) return classifyRunFailed(err, opts?.finishedAt); if (err instanceof RunTimeoutError) return classifyRunTimeout(err); if (err instanceof RunLifecycleUnavailableError) return classifyLifecycleUnavailable(err, env); if (err instanceof InputPreparationError) return classifyInputPreparationError(err, env); @@ -395,15 +420,131 @@ function classifyRunStillRunning(err: RunStillRunningError): PipelineError { }; } -function classifyRunFailed(err: RunFailedError): PipelineError { +/** + * A run that ended without a result. Its stored error report, when it has one, + * is the runtime's own account of the failure, and the classification is read + * from it: the report's title in the headline, its message as what happened + * (with the provider's raw text taken out, see `visibleReportMessage`), its + * user action's detail as the next step, its `retryable` verdict as the retry + * advice, and a support line naming the run, the error type and when it ended. + * + * A run with no report keeps the SDK's sentence, which is all there is: a + * cancelled, terminated or timed-out run, one the platform finalized itself, + * and any failed run on a platform that does not serve the report. + */ +function classifyRunFailed(err: RunFailedError, finishedAt?: string | null): PipelineError { + const report = err.error; + if (!report) { + return { + kind: "run_failed", + title: "The pipeline run failed", + message: + err.message || + `The run finished in a non-successful state (${err.status}). Check the technical details below.`, + details: `${err.name}: run ${err.runId} ended ${err.status}\n${err.message}`, + }; + } + + const title = nonEmpty(report.title); + const message = + visibleReportMessage(report) ?? + `The run ended ${err.status}, and its error report does not say why.`; + const nextStep = nonEmpty(report.user_action?.detail); + const endedAt = formatInstant(finishedAt); return { kind: "run_failed", - title: "The pipeline run failed", - message: - err.message || - `The run finished in a non-successful state (${err.status}). Check the technical details below.`, - details: `${err.name}: run ${err.runId} ended ${err.status}\n${err.message}`, + title: title ? `The pipeline run failed: ${title}` : "The pipeline run failed", + message, + ...(nextStep ? { hint: { summary: nextStep } } : {}), + ...(typeof report.retryable === "boolean" + ? { retry: report.retryable ? RETRYABLE : NOT_RETRYABLE } + : {}), + support: [`run ${err.runId}`, nonEmpty(report.error_type), endedAt && `failed ${endedAt}`] + .filter(Boolean) + .join(" · "), + details: reportDetails(err, report, message), + }; +} + +const RETRYABLE: RetryAdvice = { + retryable: true, + summary: "This failure can pass on a second try: run it again.", +}; + +const NOT_RETRYABLE: RetryAdvice = { + retryable: false, + summary: "Running it again unchanged will fail the same way.", +}; + +/** + * The report's message as a person may read it. A failure that came back from + * a model provider carries the provider SDK's own text inside the runtime's + * message (` inference failed for model '' (HTTP 412): `), and that text is raw: a repr of the provider's error + * body, or a whole HTML page from an edge in front of it, naming the + * deployment's own provider account. The report says which text is the + * provider's (`provider_metadata.message`), so it is cut out wherever it + * appears, with the separator before it; what remains names the failing pipe, + * the provider, the model and the HTTP status. Undefined when nothing is left. + */ +function visibleReportMessage(report: RunErrorReport): string | undefined { + const message = nonEmpty(report.message); + if (!message) return undefined; + const providerText = nonEmpty(report.provider_metadata?.message); + if (!providerText) return message; + const cut = message.split(`: ${providerText}`).join("").split(providerText).join(""); + return nonEmpty(cut.replace(/[\s:]+$/, "")); +} + +/** + * The report's classification for the "Technical details" section: every field + * a developer reads to place the failure, and the validation items of a method + * that failed validation. The provider's metadata is left out, for the reason + * `visibleReportMessage` gives, and the message is the one shown above. + */ +function reportDetails(err: RunFailedError, report: RunErrorReport, message: string): string { + const field = (name: string, value: unknown) => { + const text = typeof value === "boolean" ? String(value) : nonEmpty(value); + return text === undefined ? null : `${name}: ${text}`; }; + const items = Array.isArray(report.validation_errors) ? report.validation_errors : []; + return [ + `${err.name}: run ${err.runId} ended ${err.status}`, + field("error_type", report.error_type), + field("error_domain", report.error_domain), + field("error_category", report.error_category), + field("retryable", report.retryable), + field("user_action", report.user_action?.kind), + field("model", report.model), + field("type_uri", report.type_uri), + `message: ${message}`, + ...items.map((item) => { + const where = nonEmpty(item?.pipe_code) ?? nonEmpty(item?.concept_code); + return `validation: ${where ? `${where}: ` : ""}${nonEmpty(item?.message) ?? "(no message)"}`; + }), + ] + .filter(Boolean) + .join("\n"); +} + +/** A string with something in it, trimmed; anything else is undefined. */ +function nonEmpty(value: unknown): string | undefined { + if (typeof value !== "string") return undefined; + const text = value.trim(); + return text === "" ? undefined : text; +} + +/** + * An instant as a support desk reads it: UTC to the second when the value + * carries its zone, and the value as given otherwise, since a timestamp + * without one cannot be converted without guessing. + */ +function formatInstant(value: string | null | undefined): string | undefined { + const text = nonEmpty(value); + if (text === undefined) return undefined; + const time = new Date(text); + if (!/(Z|[+-]\d{2}:?\d{2})$/i.test(text) || Number.isNaN(time.getTime())) return text; + return time.toISOString().replace(/\.\d{3}Z$/, "Z"); } function classifyRunTimeout(err: RunTimeoutError): PipelineError { diff --git a/webapp-js/src/test/fixtures/runReports.ts b/webapp-js/src/test/fixtures/runReports.ts new file mode 100644 index 0000000..d47bd17 --- /dev/null +++ b/webapp-js/src/test/fixtures/runReports.ts @@ -0,0 +1,89 @@ +// --------------------------------------------------------------------------- +// TEST FIXTURE — the stored error reports of failed runs. +// +// A failed durable run's report is the runtime's `ErrorReport`, stored on the +// run and served whole by the platform (`RunRead.error`, and the `error` member +// of the results read's 409). `MODEL_NOT_ENABLED` is recorded verbatim: it is +// the report in `tests/fixtures/problems/results-409-failed.json` of +// pipelex-sdk-js at 98260ff0c1b25c8e0d8bea3ee8366094f9f30a65, which that repo +// recorded from pipelex's `LLMCompletionError` through the platform's handler. +// The other two follow the runtime's shapes for their error classes +// (`MultiplicityCountMismatchError`, and an `LLMCompletionError` classified as +// transient), with the wording of `pipelex/core/memory/exceptions.py` and +// `pipelex/cogt/inference/error_render.py`. +// --------------------------------------------------------------------------- + +import type { RunErrorReport } from "@pipelex/sdk"; + +/** The provider's raw text inside `MODEL_NOT_ENABLED`, which a person must never read. */ +export const MODEL_NOT_ENABLED_PROVIDER_TEXT = + "Error code: 412 - {'error': {'message': 'Model global.anthropic.claude-opus-4-8 is not allowed for this integration'}}"; + +/** A configuration failure from a model provider: not retryable, `change_model`. */ +export const MODEL_NOT_ENABLED: RunErrorReport = { + error_type: "LLMCompletionError", + message: `Pipe 'summarize' (path: two_steps > summarize) failed: openai inference failed for model 'claude-4.8-opus' (HTTP 412): ${MODEL_NOT_ENABLED_PROVIDER_TEXT}`, + title: "LLM completion", + type_uri: "https://docs.pipelex.com/latest/errors/llm-completion-error/", + error_category: "configuration", + error_domain: "config", + retryable: false, + user_action: { + kind: "change_model", + detail: + "This model is not enabled on the inference gateway; choose another model for the pipe.", + }, + model: "claude-4.8-opus", + provider: "pipelex_gateway", + provider_metadata: { + provider: "openai", + sdk_exception_type: "APIStatusError", + message: MODEL_NOT_ENABLED_PROVIDER_TEXT, + status_code: "412", + request_id: "req_gw_5f1c", + retry_after_seconds: "1.5", + }, +}; + +/** A caller's own input fault: not retryable, `change_input`, no provider involved. */ +export const WRONG_ITEM_COUNT: RunErrorReport = { + error_type: "MultiplicityCountMismatchError", + message: + "Input 'pages' declares exactly 2 items of 'native.Image' ('native.Image[2]'), but you provided 3.", + title: "Multiplicity count mismatch", + type_uri: "https://docs.pipelex.com/latest/errors/multiplicity-count-mismatch-error/", + error_domain: "input", + retryable: false, + user_action: { + kind: "change_input", + detail: "Provide exactly 2 items for input 'pages'.", + }, +}; + +/** The provider's raw text inside `RATE_LIMITED`. */ +export const RATE_LIMITED_PROVIDER_TEXT = + "Error code: 429 - {'error': {'message': 'Rate limit reached for gpt-4o in organization org-internal on tokens per min (TPM): Limit 30000, Used 29000.', 'type': 'tokens', 'code': 'rate_limit_exceeded'}}"; + +/** A transient provider failure: retryable, `wait_and_retry`. */ +export const RATE_LIMITED: RunErrorReport = { + error_type: "LLMCompletionError", + message: `Pipe 'summarize' (path: two_steps > summarize) failed: openai inference failed for model 'gpt-4o' (HTTP 429): ${RATE_LIMITED_PROVIDER_TEXT}`, + title: "LLM completion", + type_uri: "https://docs.pipelex.com/latest/errors/llm-completion-error/", + error_category: "transient", + error_domain: "runtime", + retryable: true, + user_action: { + kind: "wait_and_retry", + detail: "Transient provider error — the system will retry automatically.", + }, + model: "gpt-4o", + provider: "pipelex_gateway", + provider_metadata: { + provider: "openai", + sdk_exception_type: "RateLimitError", + message: RATE_LIMITED_PROVIDER_TEXT, + status_code: "429", + request_id: "req_rl_77a0", + }, +}; From ef6e9fbaf2c3d263aa2974be5e3c24da45861dbc Mon Sep 17 00:00:00 2001 From: Louis Choquel Date: Sat, 26 Sep 2026 19:41:04 +0200 Subject: [PATCH 3/4] fix: drop a failed run's stale wait_and_retry advice A `wait_and_retry` user action says the system will retry automatically, which is untrue of a run that has ended: nothing retries it. The failed-run display now drops that advice and lets the retry line say what to do. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01FigDssaJrvNcmbnBedi7oq --- webapp-js/CLAUDE.md | 2 +- webapp-js/docs/errors.md | 16 ++++++++-------- webapp-js/src/components/ErrorDisplay.test.tsx | 1 + webapp-js/src/lib/durableRun.test.ts | 2 +- webapp-js/src/lib/errors.test.ts | 4 +++- webapp-js/src/lib/errors.ts | 9 ++++++++- 6 files changed, 22 insertions(+), 12 deletions(-) diff --git a/webapp-js/CLAUDE.md b/webapp-js/CLAUDE.md index 20dcb39..a7563ff 100644 --- a/webapp-js/CLAUDE.md +++ b/webapp-js/CLAUDE.md @@ -239,7 +239,7 @@ Conventions: - **Narrow at the boundary, but never re-declare the shape**: the SDK returns loosely-typed output, so always pass the whole `RunResults` through a `parseXxx(results)` narrower in `src/types/`. That narrower hands `wireOutput(results)` to the generated binder and translates the thrown `ZodError` into a tagged subclass of `Error` (`BadPipelineOutputError`) via `describeSchemaFailure`. Do not `as` your way through, and do not hand-write the shape it validates — the method already declares it and `npm run codegen` projects it. - **Return classified errors, don't throw across the server→client boundary**: the shared helpers return `{ ok: true, ... } | { ok: false, error: PipelineError }`. Throwing works in dev but Next.js production builds strip server-action error messages to opaque digests, which destroys the developer-facing error UX. `executeBlockingRun` / `startDurableRun` / `pollDurableRun` wrap the SDK call in `try/catch`, hand the caught value to `classifyPipelineError(err, env)`, and return the structured error. Render it client-side with ``. - **Classification stays server-side, in the helpers.** `classifyPipelineError` `instanceof`-matches SDK error classes, which only exist server-side (they're stripped to opaque digests crossing the boundary) — so it runs inside the helpers, never on a poll/blocking result the client received. The durable `failed` poll constructs a `RunFailedError` carrying the run's **stored error report** — the result lookup's `error`, else the status read's, which the hosted platform serves even where its result lookup does not yet — and classifies it there too, passing the status read's `finished_at`. -- **A failed run is shown from its report, never from the lookup's sentence.** `classifyRunFailed` puts the report's `title` in the headline, its `message` as what happened, its `user_action.detail` as the hint, its `retryable` verdict as `retry` (absent when the report has none, so nothing is claimed on a guess), and a `support` line with the run id, the `error_type` and when the run ended, which `` shows in place of the bare run id. **The provider's raw text never reaches the person**: the report's message quotes the provider SDK's text verbatim, which can be a raw error body or a whole HTML page, and the report names it as `provider_metadata.message`, so `visibleReportMessage` cuts it out and `provider_metadata` is left out of the technical details too. A run with no report — cancelled, terminated, timed out, finalized by the platform, or on a platform without the report — keeps the SDK's sentence. [`docs/errors.md`](docs/errors.md) is the reference. +- **A failed run is shown from its report, never from the lookup's sentence.** `classifyRunFailed` puts the report's `title` in the headline, its `message` as what happened, its `user_action.detail` as the hint (except a `wait_and_retry` detail, which promises an automatic retry a failed run never gets), its `retryable` verdict as `retry` (absent when the report has none, so nothing is claimed on a guess), and a `support` line with the run id, the `error_type` and when the run ended, which `` shows in place of the bare run id. **The provider's raw text never reaches the person**: the report's message quotes the provider SDK's text verbatim, which can be a raw error body or a whole HTML page, and the report names it as `provider_metadata.message`, so `visibleReportMessage` cuts it out and `provider_metadata` is left out of the technical details too. A run with no report — cancelled, terminated, timed out, finalized by the platform, or on a platform without the report — keeps the SDK's sentence. [`docs/errors.md`](docs/errors.md) is the reference. - **Inputs are gated, never hand-guarded.** Every action starts with `gateRunInputs(CONTRACT, data)` over the method's committed contract — the same gate the browser ran for the Run button — and returns its `{ ok: false, error }` unchanged. Do not add a per-input `if (!x) return badRequest()` beside it. What legitimately sits _after_ the gate is a check the contract cannot express — the file-reference scheme check is the one example, and it runs over the _gated_ inputs. A file's type and size are checked earlier still, by the grant action, before the file is stored. - **Add new error kinds in `src/lib/errors.ts`**: extend `PipelineErrorKind`, add an `instanceof` branch in `classifyPipelineError` (import the class from `@pipelex/sdk`), and cover it in `src/lib/errors.test.ts`. Keep `classifyPipelineError` pure — env passed in by caller, no `process.env` reads inside. The dual-mode kinds (`execute_timeout`, `run_still_running`, `run_failed`, `run_timeout`, `lifecycle_unavailable`) follow this pattern. Two exceptions build a `PipelineError` inline (no thrown error to classify): pre-flight validation (`file_too_large`, `unsupported_file_type`, `bad_request`) in a Server Action or in `useFileInputs`'s size check, and the client-side poll ceiling (`buildClientTimeoutError`, kind `run_timeout`) in `useRun`. - **`lifecycle_unavailable` has two sources.** A 404 from a URL that doesn't serve the run-lifecycle routes arrives as the SDK's `RunLifecycleUnavailableError` (`instanceof` branch → `classifyLifecycleUnavailable`); a `/start` against a deployment whose orchestrator is blocking-only (the in-process `direct` mode) arrives as a 400 `ApiResponseError` with `error_type: "StartRequiresAsyncOrchestration"`, matched by an `errorType` branch in `classifyResponse` → `classifyStartRequiresAsync`. Both restate the runtime's vocabulary ("orchestration mode", "fire-and-forget") in this app's term — **durable execution** — and frame the configured URL as the problem, steering to `PIPELEX_BASE_URL`; the messages differ because the root causes differ. diff --git a/webapp-js/docs/errors.md b/webapp-js/docs/errors.md index 47d53e6..912b954 100644 --- a/webapp-js/docs/errors.md +++ b/webapp-js/docs/errors.md @@ -18,14 +18,14 @@ Every failure this app can meet reaches the person using it as one `PipelineErro A durable run that ends without a result is read from **its stored error report**, the runtime's own account of the failure. The platform stores the report on the run and serves it on the status read (`RunRead.error`) and in the result lookup's refusal, which `@pipelex/sdk` hands back on the failed arm of `getRunResult`. `pollDurableRun` takes the result lookup's report, or the status read's when the lookup has none (a platform that does not serve it there yet), puts it on the `RunFailedError` it classifies, and passes the status read's `finished_at` along. The classification reads each part of the display from the report: -| The display | From the report | -| ----------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| The headline | "The pipeline run failed: " and the report's `title`, the stable label of the error class. | -| What happened | The report's `message`, which names the failing pipe, without the provider's raw text (below). | -| The next step | `user_action.detail`, the runtime's advice: change an input, choose another model, check billing, wait and retry. | -| The retry line | `retryable`. True offers a re-run; false says a re-run unchanged fails the same way; absent says nothing. | -| The support line | The run id, `error_type` and the time the run ended, as `run · · failed `. | -| Technical details | The run's status and the report's classification (`error_type`, `error_domain`, `error_category`, `retryable`, the user action's kind, `model`, `type_uri`), the message shown above, and the validation items of a method that failed validation. | +| The display | From the report | +| ----------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| The headline | "The pipeline run failed: " and the report's `title`, the stable label of the error class. | +| What happened | The report's `message`, which names the failing pipe, without the provider's raw text (below). | +| The next step | `user_action.detail`, the runtime's advice: change an input, choose another model, check billing. A `wait_and_retry` advice is dropped, because it says the system will retry automatically and a failed run is retried by nothing; the retry line speaks instead. | +| The retry line | `retryable`. True offers a re-run; false says a re-run unchanged fails the same way; absent says nothing. | +| The support line | The run id, `error_type` and the time the run ended, as `run · · failed `. | +| Technical details | The run's status and the report's classification (`error_type`, `error_domain`, `error_category`, `retryable`, the user action's kind, `model`, `type_uri`), the message shown above, and the validation items of a method that failed validation. | **The provider's raw text never reaches the person.** When a model provider refused a call, the runtime writes the provider SDK's text into the report's message verbatim, and that text can be the raw body of the provider's error, naming the deployment's own provider account, or a whole HTML page from an edge in front of the provider. The report names that text as `provider_metadata.message`, so the classification cuts it out of the message wherever it appears, and leaves `provider_metadata` out of the technical details as well. What remains still names the pipe, the provider, the model and the HTTP status: `Pipe 'summarize' (path: two_steps > summarize) failed: openai inference failed for model 'claude-4.8-opus' (HTTP 412)`. diff --git a/webapp-js/src/components/ErrorDisplay.test.tsx b/webapp-js/src/components/ErrorDisplay.test.tsx index 171cae9..a913669 100644 --- a/webapp-js/src/components/ErrorDisplay.test.tsx +++ b/webapp-js/src/components/ErrorDisplay.test.tsx @@ -47,6 +47,7 @@ describe("ErrorDisplay — a failed run", () => { it("offers a re-run for a retryable failure", () => { showFailedRun(RATE_LIMITED, `Run finished with status FAILED: ${RATE_LIMITED.message}`); + expect(screen.getByRole("alert")).not.toHaveTextContent(/retry automatically/); expect( screen.getByText("This failure can pass on a second try: run it again."), ).toBeInTheDocument(); diff --git a/webapp-js/src/lib/durableRun.test.ts b/webapp-js/src/lib/durableRun.test.ts index 75205b5..ebdacdb 100644 --- a/webapp-js/src/lib/durableRun.test.ts +++ b/webapp-js/src/lib/durableRun.test.ts @@ -210,7 +210,7 @@ describe("pollDurableRun", () => { expect(result.ok).toBe(false); if (result.ok) return; expect(result.error.message).not.toContain("no result available"); - expect(result.error.hint?.summary).toBe(RATE_LIMITED.user_action?.detail); + expect(result.error.hint).toBeUndefined(); // wait_and_retry advice is stale on a failed run expect(result.error.retry?.retryable).toBe(true); expect(result.error.support).toBe("run run-1 · LLMCompletionError"); }); diff --git a/webapp-js/src/lib/errors.test.ts b/webapp-js/src/lib/errors.test.ts index 29ff943..76bbfb4 100644 --- a/webapp-js/src/lib/errors.test.ts +++ b/webapp-js/src/lib/errors.test.ts @@ -504,7 +504,9 @@ describe("classifyPipelineError — a failed run's stored error report", () => { retryable: true, summary: "This failure can pass on a second try: run it again.", }); - expect(result.hint?.summary).toBe(RATE_LIMITED.user_action?.detail); + // "The system will retry automatically" is untrue of a run that ended, so + // the retry line is the only advice. + expect(result.hint).toBeUndefined(); }); it("keeps the provider's raw text out of everything the person can read", () => { diff --git a/webapp-js/src/lib/errors.ts b/webapp-js/src/lib/errors.ts index 25152dd..d0391fb 100644 --- a/webapp-js/src/lib/errors.ts +++ b/webapp-js/src/lib/errors.ts @@ -449,7 +449,14 @@ function classifyRunFailed(err: RunFailedError, finishedAt?: string | null): Pip const message = visibleReportMessage(report) ?? `The run ended ${err.status}, and its error report does not say why.`; - const nextStep = nonEmpty(report.user_action?.detail); + // `wait_and_retry` advice is written while the runtime is still retrying + // ("the system will retry automatically"). A failed run has stopped, and + // nothing retries it, so that advice is dropped and the retry line says + // what to do instead. + const nextStep = + report.user_action?.kind === "wait_and_retry" + ? undefined + : nonEmpty(report.user_action?.detail); const endedAt = formatInstant(finishedAt); return { kind: "run_failed", From 51f0b3b7da74d129f74b04a5f93274648eaca785 Mon Sep 17 00:00:00 2001 From: Louis Choquel Date: Sun, 27 Sep 2026 12:44:56 +0200 Subject: [PATCH 4/4] collapse the `@pipelex/sdk` pin onto 0.26.0 `pipelex-method-apps` takes `@pipelex/sdk` from the registry again, with the lock regenerated in the same commit. The pin stood at 98260ff0 of `pipelex-sdk-js` (P7). --- webapp-js/package-lock.json | 16 ++++++++-------- webapp-js/package.json | 2 +- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/webapp-js/package-lock.json b/webapp-js/package-lock.json index f9a0cef..2fe9e62 100644 --- a/webapp-js/package-lock.json +++ b/webapp-js/package-lock.json @@ -9,7 +9,7 @@ "version": "0.5.5", "dependencies": { "@pipelex/mthds-form": "^0.10.0", - "@pipelex/sdk": "github:Pipelex/pipelex-sdk-js#98260ff0c1b25c8e0d8bea3ee8366094f9f30a65", + "@pipelex/sdk": "^0.26.0", "next": "^16.3.5", "react": "^19.2.4", "react-dom": "^19.2.4", @@ -1928,12 +1928,12 @@ "license": "MIT" }, "node_modules/@pipelex/sdk": { - "version": "0.25.1", - "resolved": "git+ssh://git@github.com/Pipelex/pipelex-sdk-js.git#98260ff0c1b25c8e0d8bea3ee8366094f9f30a65", - "integrity": "sha512-HiXBAfBtA6xd/I5KpYjEsbfZrINHQEldPvD8mUq1I9B4/ssCEPHkvZZpaGpt+3EdNZZF6Zln8veHw+wGPq/mcQ==", + "version": "0.26.0", + "resolved": "https://registry.npmjs.org/@pipelex/sdk/-/sdk-0.26.0.tgz", + "integrity": "sha512-4gQKoZVdRKWb1dVO4TZFvOPPcnfUCsuh1fy+dbNGyTbjUgoFDAuvcPZXPUhY78E0gtO/yN7FEdUF+0CGOeFVkQ==", "license": "MIT", "dependencies": { - "mthds": "github:mthds-ai/mthds-js#d65d2e52070c6dac500a223c0fb3ba5377dfcd59", + "mthds": "^0.28.0", "smol-toml": "^1.6.0", "undici": "^7.29.1" }, @@ -1963,9 +1963,9 @@ } }, "node_modules/@pipelex/sdk/node_modules/mthds": { - "version": "0.27.0", - "resolved": "git+ssh://git@github.com/mthds-ai/mthds-js.git#d65d2e52070c6dac500a223c0fb3ba5377dfcd59", - "integrity": "sha512-GGfu3yzo+Mlyj/Op/XQtoX50cy9ihAThE8K2QO1Ya1q3iwfiJ561IDdSpeAfO/hNPx+T3eon6ptQ+stWVCeFbw==", + "version": "0.28.0", + "resolved": "https://registry.npmjs.org/mthds/-/mthds-0.28.0.tgz", + "integrity": "sha512-5ShF4ckgavb4rnHaoiJX2wXnOk0Y3JwzKc1N26f8bszGHFVjv1bAkdYsQRrJKrefZsmIVy5G8MJli4Von1iDJQ==", "license": "MIT", "dependencies": { "@clack/prompts": "^1.0.0", diff --git a/webapp-js/package.json b/webapp-js/package.json index 7646d17..265a6e7 100644 --- a/webapp-js/package.json +++ b/webapp-js/package.json @@ -27,7 +27,7 @@ }, "dependencies": { "@pipelex/mthds-form": "^0.10.0", - "@pipelex/sdk": "github:Pipelex/pipelex-sdk-js#98260ff0c1b25c8e0d8bea3ee8366094f9f30a65", + "@pipelex/sdk": "^0.26.0", "next": "^16.3.5", "react": "^19.2.4", "react-dom": "^19.2.4",