From db93f9a108752f5f8eec93f0ea644c1a1b8d8a85 Mon Sep 17 00:00:00 2001 From: chris-colinsky Date: Fri, 2 Oct 2026 14:56:01 -0700 Subject: [PATCH 1/4] Stop the merge advice firing with no override in play The collision hint added two commits ago chose between two messages by whether the colliding extras key was inherited from the base. Both arms assume a per-attempt override exists. A caller who put a declared field and an extras key of the same name in one config, with no retry schedule anywhere, got told "the per-attempt override sets it both ways" about an override they never wrote. Three states, not two: no override, so no advice and the section 8.1 message stands alone; an override whose colliding key came from the base, so name the base's channel; an override that sets the key both ways itself, so say to remove one. Found by running examples/provider-extras, which has no retry config at all and printed the false advice in its own output. Every unit test for the hint supplied an override, so the path was never exercised. The new test covers it and asserts the request never goes out, since the refusal happens before the transport is reached. --- src/openarmature/llm/providers/openai.py | 19 +++++++++--- tests/unit/test_llm_provider.py | 37 ++++++++++++++++++++++++ 2 files changed, 52 insertions(+), 4 deletions(-) diff --git a/src/openarmature/llm/providers/openai.py b/src/openarmature/llm/providers/openai.py index f9d9ca3..ebfc47e 100644 --- a/src/openarmature/llm/providers/openai.py +++ b/src/openarmature/llm/providers/openai.py @@ -520,7 +520,7 @@ async def complete( def build_body( attempt_config: RuntimeConfig | None, attempt_messages: Sequence[Message], - inherited_extras: frozenset[str], + inherited_extras: frozenset[str] | None, ) -> dict[str, Any]: # Proposal 0095: the wire body is assembled PER ATTEMPT so a # retry can vary sampling (0095a per-attempt override) and/or the @@ -541,6 +541,12 @@ def build_body( # and sending that caller to inspect a base config they never # wrote is worse than saying nothing. def hint(key: str) -> str | None: + # None means no override applied to this attempt, so the + # collision is between a caller's declared field and their own + # extras key in one config. There is no second channel to + # explain, and §8.1's message already names both values. + if inherited_extras is None: + return None if key in inherited_extras: return ( "this attempt merges the base config with a per-attempt " @@ -728,7 +734,9 @@ def _append_reask_pair( async def _do_complete_with_retry( self, - build_body: Callable[[RuntimeConfig | None, Sequence[Message], frozenset[str]], dict[str, Any]], + build_body: Callable[ + [RuntimeConfig | None, Sequence[Message], frozenset[str] | None], dict[str, Any] + ], base_config: RuntimeConfig | None, base_messages: Sequence[Message], schema_dict: dict[str, Any] | None, @@ -754,7 +762,7 @@ async def _do_complete_with_retry( terminal event still fires per ``complete()`` call. """ if retry is None: - body = build_body(base_config, base_messages, frozenset()) + body = build_body(base_config, base_messages, None) attempt_start = time.perf_counter() try: response = await self._do_complete(body, schema_dict, schema_class) @@ -793,7 +801,10 @@ async def _do_complete_with_retry( attempt = 0 while True: attempt_config, inherited_extras = self._config_for_attempt(base_config, overrides, attempt) - body = build_body(attempt_config, transcript, inherited_extras) + # Attempt 0 runs the caller's config untouched, so no override is in + # play and a collision there has only one channel to report. + merged = attempt > 0 and bool(overrides) + body = build_body(attempt_config, transcript, inherited_extras if merged else None) attempt_start = time.perf_counter() try: response = await self._do_complete(body, schema_dict, schema_class) diff --git a/tests/unit/test_llm_provider.py b/tests/unit/test_llm_provider.py index 92b0d9f..770f7e3 100644 --- a/tests/unit/test_llm_provider.py +++ b/tests/unit/test_llm_provider.py @@ -3516,6 +3516,43 @@ def handler(req: httpx.Request) -> httpx.Response: assert bodies[1]["keep"] == 1, "every unmentioned base key inherits, not just the first" +async def test_a_collision_with_no_override_in_play_gets_no_merge_advice() -> None: + # The merge advice explains a collision assembled from two configs. A caller + # who put a declared field and an extras key of that name in ONE config, with + # no retry schedule anywhere, has no second channel to be told about, and the + # section 8.1 message already names both values. + # + # Caught by running examples/provider-extras, not by a unit test: every test + # for the hint supplied a per_attempt_override, so the no-override path was + # never exercised and the advice fired there claiming an override existed. + # + # Not vacuous: the negative assertions are the whole test. The call raises + # either way, so only the message distinguishes a correct hint from a false + # one. Killed by passing an empty frozenset instead of None on the paths + # where no override applies. + def _unreached(_req: httpx.Request) -> httpx.Response: + raise AssertionError("the collision rejects before any request is sent") + + provider = _collision_provider(_unreached) + try: + with pytest.raises(ProviderInvalidRequest) as caught: + await provider.complete( + [UserMessage(content="hi")], + config=RuntimeConfig(temperature=0.2, extras={"temperature": 0.9}), + ) + finally: + await provider.aclose() + + message = str(caught.value) + assert "temperature" in message, message + assert "managed field cannot be overridden" in message, message + assert "override" not in message.replace("overridden", ""), ( + "no per-attempt override is in play, so the message must not mention one: " + message + ) + assert "inherited" not in message, message + assert "channel" not in message, message + + async def test_merged_extras_key_colliding_with_an_overridden_field_rejects() -> None: # A base extras key is the section 6 escape hatch: unmanaged where it was # written, because the base leaves the declared field unset so the mapping From dbf8fdc890cb48726453163f8192be7410c43e8a Mon Sep 17 00:00:00 2001 From: chris-colinsky Date: Fri, 2 Oct 2026 15:00:48 -0700 Subject: [PATCH 2/4] Bump to 0.17.0 and publish the missing example pages The version lands in five places, not the three the release doc names: pyproject, __version__, the smoke-test assertion, and then uv.lock and the bundled AGENTS.md, which both embed it and regenerate from it. The docs sweep found no stale wording. What it found instead was sixteen examples on disk and thirteen pages on the site. Two of the three missing shipped in the previous commit range and one has been missing since 0.16.0, so the examples were reaching the sdist and the bundled index while the published site did not know they existed. The docs-in-sync rule says a page lands with its code, and it did not. Three pages written in the established shape, with their output sections taken from real runs where credentials allowed: the reask demo and provider-extras against a local endpoint, retrieval-rag from its print statements with the indices marked as shape rather than expected values, since it needs two API keys. A guard now compares the example directories against both the pages and the mkdocs nav. The nav half matters on its own: a page missing from it is unreachable even when the file exists, and mkdocs reports that as INFO rather than failing the strict build. Removing either a page or a nav line fails the guard. --- docs/examples/index.md | 20 +++ docs/examples/provider-extras.md | 130 ++++++++++++++++ docs/examples/retrieval-rag.md | 143 ++++++++++++++++++ docs/examples/structured-output-reask.md | 181 +++++++++++++++++++++++ mkdocs.yml | 5 + pyproject.toml | 2 +- src/openarmature/AGENTS.md | 2 +- src/openarmature/__init__.py | 2 +- tests/test_examples_smoke.py | 25 ++++ tests/test_smoke.py | 2 +- uv.lock | 2 +- 11 files changed, 509 insertions(+), 5 deletions(-) create mode 100644 docs/examples/provider-extras.md create mode 100644 docs/examples/retrieval-rag.md create mode 100644 docs/examples/structured-output-reask.md diff --git a/docs/examples/index.md b/docs/examples/index.md index 3290eb6..0388dce 100644 --- a/docs/examples/index.md +++ b/docs/examples/index.md @@ -67,6 +67,26 @@ the problem you're trying to solve. Survive a simulated mid-pipeline crash and resume from the saved checkpoint, then re-resume under an upgraded state schema with a v1->v2 migration backfilling new fields. +- [**Structured-output reask**](structured-output-reask.md). Extract + a mission record under an output-token ceiling too tight for a + complete answer, then recover by correcting the model and raising + the ceiling on the retry. Three modes show why one without the + other does not recover. + +### Retrieval + +- [**Retrieval-augmented answering**](retrieval-rag.md). Answer a + question from a small corpus using embed-then-rerank before + generation: cosine similarity for recall over the whole corpus, + a cross-encoder for precision over the shortlist. + +### Providers + +- [**Provider extras**](provider-extras.md). Reach two OpenAI + request fields openarmature does not model, and see the three + things the `extras` container refuses. Runs with no credentials + against a stub transport, because the outbound request body is + the subject. ### Observability diff --git a/docs/examples/provider-extras.md b/docs/examples/provider-extras.md new file mode 100644 index 0000000..b9fd597 --- /dev/null +++ b/docs/examples/provider-extras.md @@ -0,0 +1,130 @@ +# Provider extras + +!!! info "Source" + [https://github.com/LunarCommand/openarmature-python/blob/main/examples/provider-extras/main.py](https://github.com/LunarCommand/openarmature-python/blob/main/examples/provider-extras/main.py){target="_blank" rel="noopener"} + +Reach a vendor knob openarmature does not model, and watch the +guardrails that stop you reaching the wrong one. + +## Overview + +You classify lunar telemetry alerts in bulk. Two of the things you +want are provider-specific rather than portable, so there is no +first-class field for either: `service_tier` to take the cheaper, +slower lane for a batch nobody is waiting on, and `logit_bias` to +stop the model emitting a severity label your team retired last +quarter. Both are real OpenAI request fields. Neither means +anything on another provider. + +`extras` is where those go. It is a named container on the runtime +config, and whatever you put in it rides to the wire untouched. +That is the whole feature, and it exists so a provider-specific +knob does not require either a fork or a framework release. + +The interesting half is what it refuses. A field openarmature +already models is managed, and putting it in `extras` as well is an +error rather than an override, because two sources of truth for one +wire field is a bug you want at the call site and not in a trace +three days later. + +## What it teaches + +- `RuntimeConfig(extras={...})` carrying anything the framework does + not model. `service_tier` and `logit_bias` arrive on the request + body verbatim. +- **A key naming a field this call produced is rejected.** Setting + `temperature=0.2` and also `extras={"temperature": 0.9}` raises + `ProviderInvalidRequest` naming the key and both values. +- **Managed means "produced by this call", not "nameable".** A + sampling field you leave unset is not managed on that call, so + `extras={"temperature": 0.9}` with no `temperature=` rides + through. This is the one that surprises people, and it is what + makes `extras` usable as an escape hatch for a field the framework + models but you did not set. +- **A matching value is a no-op, not an error.** Sending `0.2` in + both places is not ambiguous, so it is allowed. +- **Structural keys reject always.** `model`, `messages`, `tools` + and `tool_choice` are rejected even on a call that produced no + such field. That asymmetry is deliberate: it is what stops an + `extras` tool array from reaching the wire without passing tool + validation. +- **`stop` merges instead of colliding**, because it realizes the + same wire field as `stop_sequences`. Both lists arrive, + concatenated and de-duplicated. + +## How to run + +```bash +uv run python examples/provider-extras/main.py +``` + +**No credentials, no endpoint.** The demo installs a stub transport +and prints the outbound request body, because the shape of that body +*is* the subject: whether a knob reached the wire, and what happened +when it collided with one the framework manages. A real endpoint +answers neither question any better, and every refusal happens +before a request is sent. + +To watch a real provider accept the knobs, drop the `transport=` +argument from `_provider()` and supply a real `base_url` and +`api_key`. + +## The graph + +```mermaid +flowchart TD + start([start]) + classify[classify] + show_guardrails[show_guardrails] + stop([end]) + + start --> classify --> show_guardrails --> stop +``` + +`classify` runs the alerts with the two passthrough knobs set. +`show_guardrails` then attempts the collisions and records what +came back, so the accepted and refused cases print side by side from +one run. + +## Reading the output + +``` +=== openarmature provider-extras demo === +alerts: 3 + +classified: + [watch] Regolith intake auger current 18% above nominal for 40 seconds, ... + [watch] South-pole relay lost carrier for 3 frames during Earth occultat... + [watch] Battery bus B cell 4 reading 0.2V under its siblings at end of c... + +the knobs that reached the wire: + service_tier = 'flex' + logit_bias = {'24886': -100} + temperature = 0.0 + +what extras refuses: + temperature in both: refused, extras key 'temperature' conflicts with the + mapping-managed wire field 'temperature' (managed value 0.2, extras value + 0.9); a managed field cannot be overridden via extras + tools via extras: refused, extras key 'tools' conflicts with the + mapping-managed wire field 'tools' (managed value None, extras value + ); a managed field cannot be overridden via extras + stop merges rather than collides: accepted +``` + +Three things in that output are the point. + +- **The wire block** is read off the body the stub captured, not + from the config. `service_tier` and `logit_bias` are there because + nothing manages them; `temperature` is there because the mapping + produced it. +- **`managed value None`** on the `tools` refusal is the structural + rule showing its work. The call declared no tools, so the mapping + produced nothing, and the key is still rejected. Every other + entry in the table would have been allowed under those + circumstances. +- **`stop` accepted** is the merge arm. It is the only managed field + in the OpenAI mapping that combines rather than collides. + +The error messages name the key and both values, so the fix is +readable from the message without reaching for the mapping table. diff --git a/docs/examples/retrieval-rag.md b/docs/examples/retrieval-rag.md new file mode 100644 index 0000000..a14b869 --- /dev/null +++ b/docs/examples/retrieval-rag.md @@ -0,0 +1,143 @@ +# Retrieval-augmented answering + +!!! info "Source" + [https://github.com/LunarCommand/openarmature-python/blob/main/examples/retrieval-rag/main.py](https://github.com/LunarCommand/openarmature-python/blob/main/examples/retrieval-rag/main.py){target="_blank" rel="noopener"} + +Answer a question about the Moon from a small corpus of passages, +using the two-stage retrieval pattern before generation. + +## Overview + +You have eight passages of lunar reference material and a question. +Stuffing all eight into the prompt would work at this size and stop +working at a thousand, so the pipeline narrows instead: find the +plausibly-relevant passages cheaply, reorder them accurately, and +ground the answer in the few that survive. + +Four steps, and the middle two are the interesting ones. + +1. **Index**, once and offline. Batch-embed every passage. `embed` + over a list returns one vector per input in input order, so the + index lines up positionally with the corpus and no separate id + map is needed. +2. **Retrieve**, per query. Embed the question, rank the corpus by + cosine similarity, keep the top four. Cheap and broad: it buys + recall, not precision. +3. **Rerank** those four with a cross-encoder, which scores each + candidate against the query directly rather than comparing two + independently-computed vectors. More accurate and more + expensive, which is why it runs over four candidates instead of + the whole corpus. Two survive. +4. **Generate** from the two reranked passages. + +Retrieval gives recall, reranking gives precision, and the split is +what makes the cost work: the expensive comparison runs over a +shortlist the cheap one produced. + +## What it teaches + +- `OpenAIEmbeddingProvider` from `openarmature.retrieval`. Batch + `embed` for the index, single `embed` for the query. One vector + per input, in input order, both times. +- `EmbeddingRuntimeConfig(input_type=...)`, the query-versus-document + knob. On OpenAI it is a wire no-op because the model is symmetric, + and it is set anyway: the same call selects the correct + representation on an asymmetric provider, so the pipeline moves to + TEI, Cohere or Jina without a code change. Setting a field that + does nothing today is what keeps it portable. +- `CohereRerankProvider.rerank`, returning `ScoredDocument` results + sorted by relevance. +- **Mapping results back by index, not by text.** Cohere does not + echo the document body, so `ScoredDocument.document` is `None` and + the only way home is `ScoredDocument.index`, which points into the + candidate list you passed. The example translates that back to a + corpus index. A pipeline that matched on returned text would work + against a provider that echoes and break against one that does not. +- `RerankRuntimeConfig(return_documents=True)` asking for the echo + where a provider supports it. +- **Retrieval providers driven inside node bodies**, so their + `EmbeddingEvent` and `RerankEvent` reach an attached observer the + same way an LLM completion does. The offline index build runs + outside the graph deliberately, and emits nothing: there is no + invocation to attribute it to. +- An `OpenAIProvider` answer node grounded in the reranked passages. + +## How to run + +```bash +uv sync --group examples +OPENAI_API_KEY=sk-... COHERE_API_KEY=... \ + uv run python examples/retrieval-rag/main.py +``` + +Both keys are required: OpenAI serves the embeddings and the answer, +Cohere serves the rerank. + +| Variable | Default | Notes | +| --- | --- | --- | +| `OPENAI_API_KEY` | required | embeddings and the answer | +| `OPENAI_BASE_URL` | `https://api.openai.com` | host root, no `/v1` | +| `OPENAI_EMBED_MODEL` | `text-embedding-3-small` | | +| `OPENAI_CHAT_MODEL` | `gpt-4o-mini` | | +| `COHERE_API_KEY` | required | rerank | +| `COHERE_RERANK_MODEL` | `rerank-v3.5` | | + +`OPENAI_BASE_URL` takes the host root. The provider appends the +`/v1` routes itself, so a URL that already ends in `/v1` is +rejected rather than producing a doubled path. + +## The graph + +```mermaid +flowchart TD + start([start]) + retrieve[retrieve] + rerank[rerank] + answer[answer] + stop([end]) + + start --> retrieve --> rerank --> answer --> stop +``` + +Linear, because each stage narrows the input to the next. The index +build is not a node: it runs once before the first `invoke`, outside +any graph. + +State carries corpus indices rather than passage text between +stages. `candidate_indices` after retrieval, `ranked_indices` after +reranking, both best-first, the second a reordered and trimmed +subset of the first. + +## Reading the output + +``` +indexed 8 passages + + [obs] embed retrieve: 1 in / 1536d + [obs] rerank rerank: 4 in / top 2 +Q: Why did Apollo 13 not land on the Moon? + retrieved: [3, 0, 6, 1] + reranked: [3, 6] + A: +``` + +The indices are what to watch, and the two lines together show the +rerank doing its job. + +- **`indexed 8 passages`** is the offline build. It prints no + observer line, because it ran outside the graph. +- **`[obs] embed retrieve: 1 in / 1536d`** is the per-query embed, + inside the node, so it reaches the observer. One input, and the + dimensionality the model returned. +- **`[obs] rerank rerank: 4 in / top 2`** is `_RETRIEVE_K` in and + `_RERANK_K` out. +- **`retrieved`** is cosine order, best-first. **`reranked`** is a + subset in a different order. If the reranked list is the first + two of the retrieved list unchanged, the cross-encoder agreed with + cosine on that query, which happens and is not a failure. The + interesting case is the one above, where a passage cosine ranked + third is promoted over the two above it. + +The specific indices and the answer text depend on the models you +point at, so treat the numbers as shape rather than as expected +output. diff --git a/docs/examples/structured-output-reask.md b/docs/examples/structured-output-reask.md new file mode 100644 index 0000000..0f56cad --- /dev/null +++ b/docs/examples/structured-output-reask.md @@ -0,0 +1,181 @@ +# Structured-output reask + +!!! info "Source" + [https://github.com/LunarCommand/openarmature-python/blob/main/examples/structured-output-reask/main.py](https://github.com/LunarCommand/openarmature-python/blob/main/examples/structured-output-reask/main.py){target="_blank" rel="noopener"} + +Pull a structured mission record out of a prose lunar-landing +report, and recover when the reply arrives unusable. + +## Overview + +A feed delivers lunar-landing reports as free prose. You want one +row per report: mission, operator, landing site, outcome, and the +mass delivered in kilograms. A JSON schema says exactly that. + +You also cap output tokens, because the records are small and you +pay by the token. That cap is a guess about the longest record you +will ever need, and the guess is sometimes wrong. When it is, the +model stops mid-object and the reply that arrives is a fragment: +valid so far, parseable as nothing. The schema boundary rejects it +exactly as it rejects a model that answered in prose. + +Retrying the identical request reproduces the identical fragment, +because nothing about the request changed. Two things have to +change, and they are different kinds of thing. + +- **What you say.** Show the model what came back and what was + wrong with it. That is `reask`: you supply the corrective + message, because only you know how to talk to your model about + your schema. +- **What it is allowed to spend.** A correction cannot help a reply + that gets cut off at the same place. That is + `per_attempt_override`: the retry runs under a raised ceiling. + +Supply only the first and the retry is better informed and still +truncated. The demo's three modes exist to make that visible rather +than asserted. + +## Why the ceiling, and not the wrong-shaped answer + +The ceiling is the failure this demo can guarantee, on any endpoint +and any model. The wrong-shaped answer is the one you are more +likely to meet, and whether you meet it is a property of your +serving stack rather than of your model. + +A schema reaches a model through two channels only: the endpoint +enforces it while decoding, or the call puts it in the prompt. An +endpoint that enforces it cannot return a wrong shape. One that +accepts `response_format` and ignores it, or a proxy that drops the +field, leaves neither channel open, and a model that was never told +the field names invents plausible ones. A stronger model does not +fix that. + +Both failures arrive at the same exception, so one builder covers +both. What the builder puts in the correction is what decides +whether it recovers. + +## What it teaches + +- `complete(response_schema=...)` raising `StructuredOutputInvalid` + rather than handing back a half-built object. A truncated reply + and a wrong-typed field arrive through the same door. +- `LlmRetryConfig(reask=...)` making that failure retryable **for + this call**. Without a builder it is terminal, which is the right + default: a schema the model cannot satisfy is usually a bug in the + schema, not a transient. +- The builder reading four fields off the exception: + `output_content` (verbatim, what the model actually sent), + `error_message` (what the reader objected to), `finish_reason` + (`"length"` when the ceiling ended the reply) and + `response_schema` (the shape that was asked for). +- **Branching on `finish_reason`, because the two failures need + different information rather than different wording.** A truncated + reply already acted on the schema and ran out of room, so + resending it the schema tells it nothing. A complete-and-wrong + reply usually never saw the schema at all. +- **Sending the schema, not only the objection.** `error_message` + names the first violation validation found, so a correction + quoting only it spends an attempt per wrong field. The schema is + the whole contract at once. +- `LlmRetryConfig(per_attempt_override=...)` applying a config + schedule to retries only. Attempt 0 runs the caller's config + untouched; each retry merges the next entry over it, and a + schedule shorter than the retry count carries its last entry + forward. +- The framework appending the model's reply as an `assistant` turn + and the correction as a `user` turn, so the conversation stays + role-alternating and the model sees its own fragment in context. + It authors no prompt of its own; every word sent is yours. +- Reask sharing the `max_attempts` budget with transient retries. A + call that burns two attempts on unusable output has one left for a + rate limit. + +## How to run + +```bash +uv sync --group examples +LLM_API_KEY=sk-... uv run python examples/structured-output-reask/main.py + +MODE=nocap LLM_API_KEY=sk-... uv run python examples/structured-output-reask/main.py +MODE=off LLM_API_KEY=sk-... uv run python examples/structured-output-reask/main.py +``` + +| `MODE` | Posture | +| --- | --- | +| unset (default) | corrective message **and** raised ceiling | +| `nocap` | corrective message, ceiling left alone | +| `off` | neither | + +Point `LLM_BASE_URL` at any OpenAI-compatible endpoint, as the host +root rather than its `/v1` path. `LLM_MODEL` defaults to a small +fast model. + +## The graph + +```mermaid +flowchart TD + start([start]) + extract[extract] + present[present] + stop([end]) + + start --> extract --> present --> stop +``` + +`extract` loops the reports, running one `complete` per report with +the retry config the mode selected. A report that never recovers +lands in `state.failures` instead of `state.records`, so one bad +report does not abort the batch. + +## Reading the output + +The three modes on one endpoint, same reports, same schema: + +``` +mode: reask (corrective message + raised token ceiling) +cap: 32 output tokens, raised to 256 on retry + + IM-3 | Intuitive Machines | Reiner Gamma | landed | 1340 kg + Chandrayaan-4 | ISRO | south pole | landed | 620 kg + Peregrine Flight 2 | unknown | Sinus Viscositatis | unconfirmed | 90 kg + +extracted 3 of 3 +``` + +``` +mode: nocap (corrective message, ceiling left alone) +cap: 32 output tokens + + unrecovered: + Unterminated string starting at: line 1 column 82: {"landing_site": "Reiner Gamma", "mass_kg": 1340, ... + ... +extracted 0 of 3 +``` + +``` +mode: off (neither) +cap: 32 output tokens + + unrecovered: + Unterminated string starting at: line 1 column 70: {"landing_site": "Reiner Gamma", "mass_kg": 1340, ... + ... +extracted 0 of 3 +``` + +**`nocap` is the arm that carries the lesson.** The model is told +exactly what went wrong and still has nowhere to put the answer, so +every attempt is cut off and the budget drains. Compare the column +numbers between `off` and `nocap`: the fragments get slightly +longer, because the model heeded "keep every value short" and still +ran out of room. Being better informed bought a few characters. + +`Peregrine Flight 2 | unknown` is the model being right, not wrong. +The report says "the operator has not yet confirmed" and never names +the operator, so `unknown` is the correct extraction from the text +alone. + +The fragments arrive as a JSON parse error rather than a schema +violation, because a truncated object is not well-formed. That is +why the builder branches on `finish_reason` instead of on the +message: `"length"` identifies a spend problem, and no amount of +reading the parse error would. diff --git a/mkdocs.yml b/mkdocs.yml index 4b7d134..cba0b63 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -154,6 +154,11 @@ nav: - Tool use: examples/tool-use.md - Reliability: - Checkpointing and migration: examples/checkpointing-and-migration.md + - Structured-output reask: examples/structured-output-reask.md + - Retrieval: + - Retrieval-augmented answering: examples/retrieval-rag.md + - Providers: + - Provider extras: examples/provider-extras.md - Observability: - Observer hooks: examples/observer-hooks.md - Langfuse observability: examples/langfuse-observability.md diff --git a/pyproject.toml b/pyproject.toml index 0cdb548..4d3d95f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -9,7 +9,7 @@ build-backend = "hatchling.build" [project] name = "openarmature" -version = "0.16.0" +version = "0.17.0" description = "Workflow framework for LLM pipelines and tool-calling agents." readme = "README.md" requires-python = ">=3.12" diff --git a/src/openarmature/AGENTS.md b/src/openarmature/AGENTS.md index ba416bc..c94063a 100644 --- a/src/openarmature/AGENTS.md +++ b/src/openarmature/AGENTS.md @@ -1,6 +1,6 @@ # OpenArmature — Agent documentation -*This is the agent guide bundled with the openarmature Python package, version 0.16.0 (spec v0.118.2). For the full docs site see [openarmature.ai](https://openarmature.ai). For the canonical spec text see [openarmature.org/capabilities](https://openarmature.org/capabilities/). For project-specific conventions for the code you're editing, see the host project's `AGENTS.md` or `CLAUDE.md`.* +*This is the agent guide bundled with the openarmature Python package, version 0.17.0 (spec v0.118.2). For the full docs site see [openarmature.ai](https://openarmature.ai). For the canonical spec text see [openarmature.org/capabilities](https://openarmature.org/capabilities/). For project-specific conventions for the code you're editing, see the host project's `AGENTS.md` or `CLAUDE.md`.* ## TL;DR diff --git a/src/openarmature/__init__.py b/src/openarmature/__init__.py index 76516f9..29c77e7 100644 --- a/src/openarmature/__init__.py +++ b/src/openarmature/__init__.py @@ -24,7 +24,7 @@ sessions opening the project find the bundled docs automatically. """ -__version__ = "0.16.0" +__version__ = "0.17.0" __spec_version__ = "0.118.2" # Proposal 0052 (spec observability §5.1 / §8.4.1): canonical # package-registry name for this implementation. Surfaces on every diff --git a/tests/test_examples_smoke.py b/tests/test_examples_smoke.py index 7f49f41..83c351e 100644 --- a/tests/test_examples_smoke.py +++ b/tests/test_examples_smoke.py @@ -65,6 +65,31 @@ def test_every_example_directory_is_listed() -> None: ) +def test_every_example_has_a_published_docs_page() -> None: + # The examples ship three ways: in the sdist, in `examples/README.md`, and + # as a page on the docs site. The first two are derived from the directory + # listing and stay in step on their own. The site is hand-maintained in two + # places, so an example can ship with no page and nothing notices, which is + # what happened to three of them across two releases. + # + # Checked against the nav as well as the file, because a page absent from + # `mkdocs.yml` is unreachable even when it exists, and `mkdocs build` reports + # that as INFO rather than failing. + docs_dir = EXAMPLES_DIR.parent / "docs" / "examples" + mkdocs = (EXAMPLES_DIR.parent / "mkdocs.yml").read_text() + + missing_page = sorted(name for name in DEMOS if not (docs_dir / f"{name}.md").is_file()) + assert not missing_page, ( + f"these examples have no docs page: {missing_page}. Add docs/examples/.md for each." + ) + + missing_nav = sorted(name for name in DEMOS if f"examples/{name}.md" not in mkdocs) + assert not missing_nav, ( + f"these examples have a docs page that the nav does not reference: " + f"{missing_nav}. Add each to the Examples section of mkdocs.yml." + ) + + @pytest.mark.parametrize("demo", DEMOS) def test_example_loads(demo: str) -> None: main_py = EXAMPLES_DIR / demo / "main.py" diff --git a/tests/test_smoke.py b/tests/test_smoke.py index bef6adc..3d53545 100644 --- a/tests/test_smoke.py +++ b/tests/test_smoke.py @@ -9,7 +9,7 @@ def test_package_versions() -> None: - assert openarmature.__version__ == "0.16.0" + assert openarmature.__version__ == "0.17.0" assert openarmature.__spec_version__ == "0.118.2" diff --git a/uv.lock b/uv.lock index 810afe5..303e16d 100644 --- a/uv.lock +++ b/uv.lock @@ -925,7 +925,7 @@ wheels = [ [[package]] name = "openarmature" -version = "0.16.0" +version = "0.17.0" source = { editable = "." } dependencies = [ { name = "httpx" }, From b13452ab83dca7444778993555406f52da29c706 Mon Sep 17 00:00:00 2001 From: chris-colinsky Date: Fri, 2 Oct 2026 19:22:23 -0700 Subject: [PATCH 3/4] Put the agent orientation in the ingest, not just the wheel docs/agent/tldr.md and docs/agent/non-obvious-shapes.md are the generator's input for the AGENTS.md bundled in the wheel, and they are deliberately not in the site nav: they are written in agent register rather than as browsable prose. The site already has a channel for that register. The llmstxt plugin emits /llms.txt and /llms-full.txt for exactly this audience, and those two files were the only content written for it and the only content the plugin did not carry. Meanwhile mkdocs builds every page under docs_dir whether the nav references it or not, so they were publishing as unlinked HTML through the browse path they were never written for. Now an llmstxt section, listed first because it is the orientation the rest reads against. The ingest grows 602KB to 624KB and the index gains a heading; the HTML stays unlinked, which is what a page that exists to be the canonical URL for an ingest entry should be. Guarded against the plugin config rather than a rendered artifact, so it needs no site build. Nothing failed before: a missing section is silent, and the strict build calls an unreferenced page INFO. --- mkdocs.yml | 7 +++++++ tests/test_agents_md_drift.py | 37 +++++++++++++++++++++++++++++++++++ 2 files changed, 44 insertions(+) diff --git a/mkdocs.yml b/mkdocs.yml index cba0b63..83e80e0 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -62,6 +62,13 @@ plugins: - llmstxt: full_output: llms-full.txt sections: + # First, because it is the orientation an agent needs before any of + # the rest parses: what the framework is, what it is not, and the + # shapes that are not deducible from the API surface. These two files + # are also the generator's input for the AGENTS.md bundled in the + # wheel, so the ingest and the installed guide say the same thing. + Agent orientation: + - agent/*.md Getting Started: - getting-started/*.md Concepts: diff --git a/tests/test_agents_md_drift.py b/tests/test_agents_md_drift.py index 991fc76..f0194ba 100644 --- a/tests/test_agents_md_drift.py +++ b/tests/test_agents_md_drift.py @@ -182,6 +182,43 @@ def test_the_drift_check_skips_where_the_submodule_is_absent(tmp_path: Path) -> ) +def test_the_agent_orientation_reaches_the_llms_txt_ingest() -> None: + # docs/agent/ is the generator's input for the bundled AGENTS.md, and it is + # not in the site nav because it is written in agent register rather than as + # browsable prose. The site's channel for that register is the llmstxt + # plugin's /llms.txt and /llms-full.txt. + # + # Those two files were the only content written for the ingesting audience + # and the only content missing from the ingest, while publishing as unlinked + # HTML through the browse path they were never written for. A section absent + # from the plugin config fails nothing: mkdocs still builds the pages, and + # the strict build reports an unreferenced page as INFO. + # + # Not vacuous: asserted against the plugin's own glob rather than against a + # rendered artifact, so it does not need a site build, and removing the + # section or renaming the directory both fail it. + import yaml + + repo = Path(__file__).resolve().parent.parent + raw = (repo / "mkdocs.yml").read_text() + + # mkdocs.yml carries python/name: tags that safe_load rejects, and the nav + # is not what this checks, so the plugin block is read on its own. + assert "agent/*.md" in raw, ( + "docs/agent/ is not listed in any mkdocs plugin section, so the agent " + "orientation reaches neither the site nav nor /llms.txt. Add it to the " + "llmstxt sections." + ) + start = raw.index(" - llmstxt:") + end = raw.index("\nmarkdown_extensions:", start) + llmstxt = cast("dict[str, Any]", yaml.safe_load(raw[start:end])[0])["llmstxt"] + globs = [g for section in llmstxt["sections"].values() for g in section] + assert "agent/*.md" in globs, f"the agent orientation must be an llmstxt section; globs are {globs}" + + on_disk = sorted(p.name for p in (repo / "docs" / "agent").glob("*.md")) + assert on_disk, "docs/agent/ is empty, so the glob covers nothing" + + def test_patterns_dir_matches_generator_output() -> None: generator = _load_generator() expected_payload: dict[str, str] = generator.build_patterns_data() From b73940261f59603c46cd66c4cc968583f29ef802 Mon Sep 17 00:00:00 2001 From: chris-colinsky Date: Sat, 3 Oct 2026 14:12:24 -0700 Subject: [PATCH 4/4] Parse the nav, and correct the structural-keys claim Two review findings, both in work added by this PR. The example-docs guard searched mkdocs.yml as text, so a commented-out nav entry satisfied it while the page had genuinely left the navigation. Verified: commenting the line out left the guard passing. A mention in a comment or a path under `plugins:` would have done the same. That is the defect the guard exists to catch, one level up, and the same shape as the sdist guard this PR replaced for asserting on the exclude list's text rather than the built artifact. It now loads the config and walks `nav` recursively. SafeLoader rejects mkdocs-material's python tags, so the loader ignores unknown tags rather than slicing the nav block out by line offsets, which would be text matching again. Both removing and commenting out a nav line now fail it. The provider-extras page said structural keys reject always, two bullets after saying a matching value is a no-op. Both cannot hold, and the code sides with the second: the reject arm returns early when the extras value equals the managed one, with no special case for structural keys. What is actually unconditional about model, messages, tools and tool_choice is WHEN they are managed, not how they reject. Every other reject-arm key is managed only where the mapping produced it, so an unset temperature lets an extras temperature through. The structural four are managed regardless, so a conflict rejects even where the body carries nothing of that name, which is what stops an extras tool array reaching the wire on a no-tools call. All four arms of the rewritten claim were probed against the real provider. --- docs/examples/provider-extras.md | 20 ++++++------ tests/test_examples_smoke.py | 52 ++++++++++++++++++++++++++++++-- 2 files changed, 61 insertions(+), 11 deletions(-) diff --git a/docs/examples/provider-extras.md b/docs/examples/provider-extras.md index b9fd597..e943055 100644 --- a/docs/examples/provider-extras.md +++ b/docs/examples/provider-extras.md @@ -43,11 +43,13 @@ three days later. models but you did not set. - **A matching value is a no-op, not an error.** Sending `0.2` in both places is not ambiguous, so it is allowed. -- **Structural keys reject always.** `model`, `messages`, `tools` - and `tool_choice` are rejected even on a call that produced no - such field. That asymmetry is deliberate: it is what stops an - `extras` tool array from reaching the wire without passing tool - validation. +- **Structural keys are managed unconditionally.** `model`, + `messages`, `tools` and `tool_choice` are managed whether or not + the mapping produced the field, so a *conflicting* value rejects + even on a call whose body carries nothing of that name. A matching + value is still a no-op, by the same rule as above. The asymmetry is + deliberate: it is what stops an `extras` tool array from reaching + the wire on a no-tools call without passing tool validation. - **`stop` merges instead of colliding**, because it realizes the same wire field as `stop_sequences`. Both lists arrive, concatenated and de-duplicated. @@ -119,10 +121,10 @@ Three things in that output are the point. nothing manages them; `temperature` is there because the mapping produced it. - **`managed value None`** on the `tools` refusal is the structural - rule showing its work. The call declared no tools, so the mapping - produced nothing, and the key is still rejected. Every other - entry in the table would have been allowed under those - circumstances. + rule showing its work. The call declared no tools, so the managed + value is `None`, and a list conflicts with that. Every other entry + in the table is unmanaged on a call that did not produce it, so the + same extras key would have ridden through. - **`stop` accepted** is the merge arm. It is the only managed field in the OpenAI mapping that combines rather than collides. diff --git a/tests/test_examples_smoke.py b/tests/test_examples_smoke.py index 83c351e..74a0e1b 100644 --- a/tests/test_examples_smoke.py +++ b/tests/test_examples_smoke.py @@ -24,8 +24,10 @@ import runpy from pathlib import Path +from typing import Any, cast import pytest +import yaml EXAMPLES_DIR = Path(__file__).parent.parent / "examples" @@ -49,6 +51,44 @@ ] +class _TolerantLoader(yaml.SafeLoader): + """SafeLoader that survives mkdocs-material's python tags. + + `mkdocs.yml` carries `!!python/name:` values for the emoji extension, which + SafeLoader refuses and the unsafe loaders would execute. Ignoring unknown + tags parses the file without either. + """ + + +def _ignore_unknown_tag(loader: Any, suffix: str, node: Any) -> None: + return None + + +for _prefix in ( + "tag:yaml.org,2002:python/name:", + "tag:yaml.org,2002:python/object/apply:", +): + # pyyaml ships no annotations for this classmethod, so reading it is + # unknown-typed under strict mode regardless of what the result is cast to. + _TolerantLoader.add_multi_constructor(_prefix, _ignore_unknown_tag) # pyright: ignore[reportUnknownMemberType] + + +def _nav_paths(nav: Any) -> list[str]: + """Every document path the nav references, at any depth. + + A nav is nested lists of either a bare path or a one-key `{title: entry}` + mapping, where the entry is a path or another list. Only the leaf strings + are paths. + """ + if isinstance(nav, str): + return [nav] + if isinstance(nav, list): + return [path for item in cast("list[Any]", nav) for path in _nav_paths(item)] + if isinstance(nav, dict): + return [path for value in cast("dict[str, Any]", nav).values() for path in _nav_paths(value)] + return [] + + def test_every_example_directory_is_listed() -> None: # `DEMOS` is an explicit enumeration, so a new example is covered by nothing # until someone remembers to add it here. Derive the directory set and @@ -75,15 +115,23 @@ def test_every_example_has_a_published_docs_page() -> None: # Checked against the nav as well as the file, because a page absent from # `mkdocs.yml` is unreachable even when it exists, and `mkdocs build` reports # that as INFO rather than failing. + # + # The nav half reads the PARSED nav rather than the file's text. A substring + # search over mkdocs.yml is satisfied by a commented-out entry, by a mention + # in a comment, and by a path under `plugins:`, so it passes while the page + # is gone from the navigation. Which is the defect this guard exists to + # catch, one level up. docs_dir = EXAMPLES_DIR.parent / "docs" / "examples" - mkdocs = (EXAMPLES_DIR.parent / "mkdocs.yml").read_text() + config = yaml.load((EXAMPLES_DIR.parent / "mkdocs.yml").read_text(), Loader=_TolerantLoader) missing_page = sorted(name for name in DEMOS if not (docs_dir / f"{name}.md").is_file()) assert not missing_page, ( f"these examples have no docs page: {missing_page}. Add docs/examples/.md for each." ) - missing_nav = sorted(name for name in DEMOS if f"examples/{name}.md" not in mkdocs) + in_nav = set(_nav_paths(config.get("nav"))) + assert in_nav, "parsed no nav entries at all, so the check below cannot mean anything" + missing_nav = sorted(name for name in DEMOS if f"examples/{name}.md" not in in_nav) assert not missing_nav, ( f"these examples have a docs page that the nav does not reference: " f"{missing_nav}. Add each to the Examples section of mkdocs.yml."