diff --git a/.env.example b/.env.example
index 2bccaba..f1264dc 100644
--- a/.env.example
+++ b/.env.example
@@ -28,10 +28,20 @@ API_KEY=
# MODEL_BASE_URL=http://172.17.0.1:8000/v1 # serving model only (agent, A/B, rollouts)
# BASE_URL=http://172.17.0.1:11434/v1 # or everything: teacher + judge too
-# Routing model + thresholds. The defaults are calibrated TOGETHER on a drafted routing eval;
+# Routing model + thresholds. The defaults are calibrated TOGETHER on a drafted routing eval.
+# CPU stays the safe default. NVIDIA hosts can opt into the larger Apache-2.0 Qwen3-Embedding-8B
+# GPU server with:
+# docker compose -f docker-compose.yml -f compose.gpu-embeddings.yml \
+# --profile gpu-embeddings up -d embed-gpu mcp
+# The overlay sets these four values; set them directly only for an external compatible server:
+# EMBED_BACKEND=remote
+# EMBED_BASE_URL=http://embed-gpu:8080/v1
+# EMBED_REMOTE_MODEL=Qwen/Qwen3-Embedding-8B-GGUF
+# EMBED_TIMEOUT_SECONDS=120
# if you override EMBED_MODEL, recalibrate the three scores with it (cosine distributions differ
# per model. For the previous default BAAI/bge-small-en-v1.5 use 0.65 / 0.45 / 0.93):
# EMBED_MODEL=onnx-community/Qwen3-Embedding-0.6B-ONNX # or any fastembed model name
+# ROUTER_BODY_CHARS=1000 # approved body prefix embedded beside name + description; max 4000
# MIN_SCORE=0.53 # at/above -> routable match; below -> related band or novel
# RELATED_SCORE=0.37 # floor of the "related" (compose/extend) band; below it a task
# # is novel (weak/strong escalation)
@@ -57,6 +67,51 @@ API_KEY=
# MINE_MAX_JUDGE_CALLS=24 # new representative judge calls per run; <=0 removes the cap
# MINE_CLUSTER_THRESHOLD=0.90 # task cosine at/above this shares a representative verdict
+# Vault publisher handoff. Approval writes runs/publications at mode 0700, and the publisher
+# (ops/systemd/ingot-publisher.service) reads those receipts from the host as an ordinary user.
+# When both run on one machine, set these to that user so the two agree on who owns the receipts.
+# Leaving them unset keeps the UI container as root, which is right for every deployment with no
+# host publisher. Get them wrong and the failure is silent in the console: approvals queue, the
+# publisher lists an empty directory, and nothing publishes. The publisher logs it at each poll.
+# INGOT_UID=1000
+# INGOT_GID=1000
+
+# Where mutable state lives: the served library, the review queue, publication receipts, evidence,
+# snapshots, and eval task sets. Defaults to $XDG_STATE_HOME/ingot (else ~/.local/state/ingot), so
+# a pip-installed Ingot keeps nothing inside site-packages where an upgrade would discard it.
+# Compose sets each path explicitly to what it mounted. `ingot status` prints every resolved path,
+# where it came from, and whether it is writable.
+# INGOT_HOME moves all of them at once; the specific settings override it one at a time.
+# INGOT_HOME=~/.local/state/ingot
+# INGOT_LIBRARY=/srv/ingot/library # SKILLS_DIR is the deprecated name for this
+# INGOT_RUNS=/srv/ingot/runs
+# INGOT_TASKS=/srv/ingot/tasks
+
+# Published host ports. Both stay on loopback; only the host side moves. Set them when this box
+# already runs something on 8000 or 8080, including a second Ingot stack. `0` asks the kernel for
+# a free port, which is what scripts/managed_smoke.sh does: it reaches every container through
+# `docker compose exec` and needs no host port at all.
+# INGOT_MCP_PORT=8000
+# INGOT_UI_PORT=8080
+
+# Publication backend. `local` (the default) publishes into ./vault, a Git repository on this
+# machine: no network, no GitHub account, no `gh`. `forge` makes a merged pull request the
+# publication authority instead, which anchors activation off-box at the cost of the air gap and
+# requires compose.forge.yaml plus a ./vault that is a clone of the repository below. Setting the
+# forge variables without selecting the backend is inert, and the publisher says so at startup.
+# INGOT_PUBLISH_BACKEND=local
+# INGOT_FORGE_REPOSITORY=owner/repo
+# INGOT_FORGE_REMOTE=origin
+# INGOT_FORGE_BRANCH=main
+
+# Delivery targets: where an approved revision is installed once the vault carries it. The vault is
+# the managed-MCP library, so agents using Ingot's MCP server are already served; this is for an
+# agent that reads a native skill directory on disk instead. Comma-separated `name=kind:path`;
+# `filesystem` is the only kind you configure (the vault target is always present). The publisher
+# creates each root at startup and refuses to start if it cannot. Names are yours -- Ingot knows
+# nothing about what reads the directory.
+# INGOT_DELIVERY_TARGETS=claude=filesystem:~/.claude/skills,codex=filesystem:~/.codex/skills
+
# Change-control UI login. Three modes; AUTH_MODE picks one (compose default: password).
# See docs/sso.md. To share the UI beyond this machine, use the TLS front door:
# `docker compose --profile lan up -d proxy` (docs/security.md "Network exposure").
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 20efcc9..3a1ccb1 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -13,6 +13,39 @@ jobs:
- name: Build image
run: docker build -t ingot-mcp .
- name: Run tests
- run: docker run --rm -v "$PWD:/app" -w /app ingot-mcp python -m pytest tests -q
+ # git comes from the image now: the publisher container drives real repositories, so it is
+ # a runtime dependency rather than something only the tests need.
+ run: >
+ docker run --rm -v "$PWD:/app" -w /app ingot-mcp
+ python -m pytest tests -q
- name: Smoke test Compose and Langfuse TLS
run: ./scripts/compose_smoke.sh
+
+ # The control-plane claim is that no non-publisher service can change what is served. Everything
+ # else that checks it -- tests/test_compose_managed.py, `docker compose config` -- reads the
+ # configuration. This job is the only one that watches the kernel refuse the write, so it is the
+ # one that has to pass before that claim is repeated anywhere. Make it a required check.
+ managed:
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+ - name: One writer, and it is the publisher
+ run: ./scripts/managed_smoke.sh
+ - name: A removed read-only mount must fail the smoke test
+ # A check that cannot fail proves nothing. This deliberately breaks the invariant and
+ # requires the script to notice, so a later edit that quietly drops a `:ro` cannot leave a
+ # green job behind it.
+ run: |
+ python3 - <<'PY'
+ import pathlib
+ path = pathlib.Path("docker-compose.yml")
+ text = path.read_text()
+ broken = text.replace("./vault:/app/skills:ro", "./vault:/app/skills", 1)
+ assert broken != text, "no read-only served mount left to break"
+ path.write_text(broken)
+ PY
+ if MANAGED_SMOKE_PROJECT=ingot-managed-negative ./scripts/managed_smoke.sh; then
+ echo "the smoke test passed with a writable served mount; it is checking nothing"
+ exit 1
+ fi
+ git checkout -- docker-compose.yml
diff --git a/.gitignore b/.gitignore
index ee200b8..2133f17 100644
--- a/.gitignore
+++ b/.gitignore
@@ -18,10 +18,51 @@ __pycache__/
.pytest_cache/
.hf_cache/
.venv/
+/.venv-harbor-gateway/
+.worktrees/
+
+# packaging build artifacts (`pip install -e .`)
+*.egg-info/
+build/
+dist/
# fetched skills are optional & not redistributed here (scripts/fetch_skills.sh)
skills/*
!skills/.gitkeep
+# the managed stack's skill vault: its own Git repository, created by `ingot vault init`
+vault/
+
# eval task sets are runtime artifacts (auto-drafted or user-authored), not shipped opinions
optimize/tasks/*.yaml
+# ...except a hand-authored one, which is the measuring instrument rather than its output: the
+# seeded working trees and weighted checklists are the experiment's design, and a matrix produced
+# by a task set that no longer exists in the tree cannot be reproduced or argued with.
+!optimize/tasks/build-loop.yaml
+!optimize/tasks/adversarial-council-review.yaml
+!optimize/tasks/assumption-audit.yaml
+!optimize/tasks/auditing-economic-claims.yaml
+!optimize/tasks/auditing-system-claims.yaml
+!optimize/tasks/decomposing-skill-libraries.yaml
+!optimize/tasks/forward-intro.yaml
+!optimize/tasks/isolated-integration-fixtures.yaml
+!optimize/tasks/live-caller-gate.yaml
+!optimize/tasks/measurement-integrity.yaml
+!optimize/tasks/memory-defrag.yaml
+!optimize/tasks/memory-notes.yaml
+!optimize/tasks/memory-reflect.yaml
+!optimize/tasks/operating-accountability-loop.yaml
+!optimize/tasks/oss-ready.yaml
+!optimize/tasks/product-marketing.yaml
+!optimize/tasks/prose-style-hemingway.yaml
+!optimize/tasks/copywriting.yaml
+!optimize/tasks/routing-economic-evidence.yaml
+!optimize/tasks/skill-security.yaml
+!optimize/tasks/skill-retrospective.yaml
+!optimize/tasks/agentic-action-safety.yaml
+!optimize/tasks/turning-buyer-notes-into-decisions.yaml
+!optimize/tasks/unattended-overnight-ops.yaml
+!optimize/tasks/writing-clearly-and-concisely.yaml
+!optimize/tasks/linting-implementation-plans.yaml
+!optimize/tasks/op-credentials.yaml
+!optimize/tasks/slancha-cred.yaml
diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md
index 73d89b7..11b6e97 100644
--- a/ARCHITECTURE.md
+++ b/ARCHITECTURE.md
@@ -1,10 +1,10 @@
# Architecture
-Ingot is a local-first change-control system for agent instructions. A skill folder is the unit of
-change; every version of it is content-addressed, every proposed change is quarantined until a human
-approves it, and every promotion is atomic and reversible. Routing exists to serve the approved
-revision to an agent. Ingot is not a multi-tenant service, and the default Compose deployment
-exposes only loopback ports.
+Ingot is a local-first, air-gappable control plane for a team's skill library. A skill folder is the
+unit of change; every version of it is content-addressed, every proposed change is quarantined until
+a human approves it, and every promotion is atomic and reversible. Routing exists to serve the
+approved revision to an agent. Ingot is not a multi-tenant service, and the default Compose
+deployment exposes only loopback ports.
## The change-control pipeline
@@ -36,11 +36,14 @@ exposes only loopback ports.
1. The bundled agent sends task and execution context to `route_and_load` over MCP.
2. The router refreshes the skill registry when files change, filters by harness, platform, scope,
- tools, MCPs, activation, and trust, then ranks compatible descriptions. Description embeddings
- are cached across refreshes.
+ tools, MCPs, activation, and trust, then ranks compatible skills by the stronger cosine score
+ from the description or a bounded document containing name, description, and approved
+ harness-specific body. Both representations share a 4,096-vector least-recently-used cache
+ across refreshes; description-only vectors remain authoritative for collision detection.
3. One response is authoritative for the direct `match` or explicit `related_match`, loaded body,
- revision, root, body-free alternatives, and `novel` escalation signal. A related match is loaded
- for compose-or-extend use. The agent uses the weak model unless `novel` is true.
+ revision, root, body-free alternatives, component scores, and `novel` escalation signal. A
+ related match is loaded for compose-or-extend use. The agent uses the weak model unless `novel`
+ is true.
4. The run is recorded to Langfuse (the default evals backend, or a Langfuse-compatible endpoint
`LANGFUSE_*` points at); mining reads it back and has no local fallback. Hosted model calls use
the configured OpenAI-compatible endpoint. OpenRouter calls always request ZDR providers.
@@ -48,7 +51,7 @@ exposes only loopback ports.
## SkillOpt optimization
SkillOpt optimization is a core product capability. It proposes changes but never activates them.
-Runs start in the background (`optimize.loop`) or on demand from the UI, and every result enters the
+Runs start in the background (`ingot.optimize.loop`) or on demand from the UI, and every result enters the
same human review path.
Mining reads every usable Langfuse trace by default, with `--limit N` available only as an explicit
@@ -102,7 +105,7 @@ text components (`OPTIMIZE_COMPONENTS=body,file:`), diffed for review and
## Stores and ownership
-`skills/` contains active skills. `optimize/tasks/` contains eval sets. `runs/pending/` contains one
+`skills/` contains active skills. `ingot/optimize/tasks/` contains eval sets. `runs/pending/` contains one
active review slot per skill, with displaced candidates archived. `runs/revisions/` contains
rollback snapshots, plus a `.snapshots.json` index per skill that records when each revision was
last snapshotted; it sits beside the snapshot directories, never inside one, so a rollback restores
@@ -163,13 +166,13 @@ Promotion stages changes and restores the prior directory if the swap fails. Eve
the displaced revision. Restore it from the UI's History section, or with:
```bash
-docker compose run --rm --entrypoint python optimize -m optimize.promote rollback SKILL REVISION
+docker compose run --rm --entrypoint python optimize -m ingot.optimize.promote rollback SKILL REVISION
```
-The `optimize` service's entrypoint is `python -m optimize.ab`, so the entrypoint override is what
-makes the arguments reach `optimize.promote`.
+The `optimize` service's entrypoint is `python -m ingot.optimize.ab`, so the entrypoint override is what
+makes the arguments reach `ingot.optimize.promote`.
-Operators should back up `skills/`, `runs/`, and `optimize/tasks/`. Container databases require
+Operators should back up `skills/`, `runs/`, and `ingot/optimize/tasks/`. Container databases require
normal volume backup procedures.
## Trust boundaries
diff --git a/Dockerfile b/Dockerfile
index 6f0977e..0f40294 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -2,6 +2,13 @@ FROM python:3.12-slim-bookworm
WORKDIR /app
+# The publisher's vault is a Git repository and every publication is a worktree, a commit and a
+# fast-forward. slim does not ship git, so without this the one service that owns the served
+# library cannot start.
+RUN apt-get update \
+ && apt-get install -y --no-install-recommends git \
+ && rm -rf /var/lib/apt/lists/*
+
COPY requirements.txt .
RUN pip install --no-cache-dir pip==26.1.2 \
&& pip install --no-cache-dir -r requirements.txt
@@ -32,12 +39,19 @@ ARG FALLBACK_EMBED_MODEL=BAAI/bge-small-en-v1.5
RUN python -c "from fastembed import TextEmbedding; TextEmbedding('${FALLBACK_EMBED_MODEL}')"
ENV BAKED_FALLBACK_EMBED_MODEL=${FALLBACK_EMBED_MODEL}
-COPY mcp_server ./mcp_server
+# `mcp_server` and `optimize` live under `ingot/` and arrive with the COPY below.
COPY agent ./agent
-COPY optimize ./optimize
COPY ui ./ui
COPY skills ./skills
+# The `ingot` console script. `--no-deps` keeps the installed set exactly the pinned
+# requirements.txt above instead of re-resolving it from pyproject.toml, and `-e` points the script
+# at the /app copies the services already run with `python -m`, so there is only ever one copy of
+# the code in the image.
+COPY pyproject.toml README.md ./
+COPY ingot ./ingot
+RUN pip install --no-cache-dir --no-deps -e .
+
ENV PYTHONUNBUFFERED=1
-CMD ["python", "-m", "mcp_server.server"]
+CMD ["python", "-m", "ingot.mcp_server.server"]
diff --git a/PRODUCTION_SETUP.md b/PRODUCTION_SETUP.md
index 2de7273..d442d22 100644
--- a/PRODUCTION_SETUP.md
+++ b/PRODUCTION_SETUP.md
@@ -209,7 +209,7 @@ Codex, with the exact rule in [Make skill loading part of the agent instructions
## Operations
Back up all named Langfuse datastore volumes and the repository's `skills/`, `runs/`, and
-`optimize/tasks/` directories. Pin image versions, review upgrades before applying them, and test
+`ingot/optimize/tasks/` directories. Pin image versions, review upgrades before applying them, and test
restore procedures. Monitor container health and disk usage:
```bash
diff --git a/README.md b/README.md
index 53bea85..e327160 100644
--- a/README.md
+++ b/README.md
@@ -4,14 +4,18 @@
-**Evidence-gated change control for agent instructions.**
+**Open-source release control for agent skills.** Quarantine, prove, approve, publish,
+and roll back exact skill revisions.
[](https://github.com/SlanchaAI/ingot/actions/workflows/ci.yml) [](LICENSE) [](Dockerfile) [](docker-compose.yml)
-An agent's [skills](https://github.com/anthropics/skills) are instructions it follows. **Ingot** is a
-local-first library and MCP server for individual developers who serve versioned skills, evaluate
-optimizer-generated challengers, and require human promotion before those challengers replace live
-instructions.
+Every tool installs a skill. Ingot quarantines it.
+
+An agent's [skills](https://github.com/anthropics/skills) are instructions it follows, and they
+arrive from anywhere: a marketplace, a teammate, an optimizer. **Ingot** is a local-first,
+air-gappable control plane for a team's skill library. It serves an exact revision of each skill
+over MCP, holds every proposed change in quarantine, and requires a human approval backed by
+evidence before one reaches what agents load.
**[SkillOpt integration](https://github.com/microsoft/SkillOpt)** learns from real agent traces,
trains bounded instruction edits, compares them on held-out tasks, and produces an evidence-backed
@@ -35,10 +39,19 @@ proposal. SkillOpt can propose a change but cannot activate one.
installation fails. Restore any snapshot from the UI or CLI.
- **Decisions produce a local audit trail.** Approvals, rejections, and rollbacks attempt to append
metadata-only records after the transition. Audit-write failures are logged and do not roll back
- the decision; the local trail is not tamper-proof.
+ the decision. Publication history is Git-backed, revision-bound, and externally anchorable in
+ forge mode; anyone with a shell on the machine can still rewrite the local trail.
+
+**One writer.** In the tracked stack the served library is a Git vault mounted read-only into every
+service except the publisher, and the publisher only acts on an approved receipt. Approval does not
+change what is served: it queues a receipt the publisher then commits and activates. `ingot status`
+answers whether that actually holds for a given deployment, by asking the filesystem rather than
+reading a claim back out of the configuration.
-MCP serves the current contents of `skills/`. Direct edits, fetched skills, and copied or restored
-folders bypass the proposal workflow and become active. Review them as trusted code before use.
+The writable stack still exists, as `compose.dev.yaml`, and it reports `UNMANAGED`. Anything running
+as that user can change what is served without an approval, so the guarantees above do not apply to
+it. Skills that arrive by direct edit, `cp`, or a restored folder are trusted code either way —
+review them as such.
## A recorded gated change
@@ -69,7 +82,13 @@ through evidence review, deliberate promotion, and rollback.
- **Local development by default.** Compose binds the public ports to localhost and password-gates
the UI. The MCP endpoint has no built-in authentication, and Ingot is not a hardened multi-tenant
service. Follow the production guide before sharing it beyond one trusted machine.
-- **Easy.** A skill is a folder with a `SKILL.md`. Drop one in and it is live on the next request.
+- **Offline by default.** The default publication backend is `local`: the vault is a Git repository
+ on this machine and publishing needs no network, no GitHub account, and no `gh`. Set
+ `INGOT_PUBLISH_BACKEND=forge` (see `compose.forge.yaml`) to make a merged pull request the
+ publication authority instead, which anchors activation off-box at the cost of the air gap.
+- **Easy.** A skill is a folder with a `SKILL.md`. `ingot add file:./that-folder` quarantines a
+ local package. `ingot add github:OWNER/REPO --skill path/to/skill` fetches a public repository at
+ an exact commit and quarantines that package. Neither command publishes it.
## Quickstart
@@ -79,20 +98,41 @@ Prerequisites:
- Free localhost ports `8000`, `8080`, and `3100`.
- An OpenRouter API key, or a reachable OpenAI-compatible Ollama or vLLM endpoint.
-`scripts/fetch_skills.sh` copies unpinned third-party skills into the live `skills/` directory. For
-this PDF demo, fetch only Anthropic's document skills. Review their instructions and per-skill
-licenses in [Skill sources](docs/skill-sources.md) before running the fetch; add other sources after
-the first run.
+`scripts/fetch_skills.sh` clones unpinned third-party skills and quarantines each one for review. It
+serves nothing: the library stays byte-identical until you approve a package. For this PDF demo,
+fetch only Anthropic's document skills. Review their instructions and per-skill licenses in
+[Skill sources](docs/skill-sources.md) before running the fetch; add other sources after the first
+run.
```bash
git clone https://github.com/SlanchaAI/ingot.git && cd ingot
cp .env.example .env # set API_KEY, or point BASE_URL at Ollama or vLLM
-scripts/fetch_skills.sh anthropics # fetch the document skills used by this demo
-docker compose up -d --build # router (:8000), UI (:8080), Langfuse (:3100)
+pip install -e . # the `ingot` command (Python 3.12+, PyYAML only)
+ingot vault init vault # the Git vault the publisher owns
+scripts/fetch_skills.sh anthropics # quarantine the document skills used by this demo
+docker compose up -d --build # router (:8000), UI (:8080), publisher, Langfuse (:3100)
docker compose ps
+open http://localhost:8080 # approve `pdf`; the publisher commits and activates it
docker compose run --rm agent "How do I merge several PDFs into one and add page numbers?"
```
+`ingot vault init` is idempotent and the publisher runs it on every start, so skipping it only
+means the vault appears when the stack does.
+
+The whole loop also runs from a terminal, and none of it writes a served byte:
+
+```bash
+ingot pending # what is waiting on a decision
+ingot approve pdf # queue a publication receipt; the publisher activates it
+ingot history pdf # snapshots, receipts, and the decision trail
+ingot rollback pdf
+ingot status # MANAGED, PENDING, DRIFTED, or UNMANAGED
+```
+
+`ingot status` compares what is served against what the last release receipt says should be served,
+so an out-of-band edit to the library reports `DRIFTED` and the command exits non-zero. See
+[Managed deployment](docs/managed-deployment.md).
+
A successful run names the route and the exact skill revision loaded before the answer. This is an
excerpt from the recorded tutorial run; scores, hashes, token counts, and model output vary:
@@ -115,9 +155,12 @@ The change-control UI at `localhost:8080` asks for a login; the compose default
set `AUTH_MODE=open` explicitly. See [Privacy & security](docs/security.md#network-exposure).
`docker compose up` brings up a self-hosted Langfuse (traces + experiment UI) alongside the router
-and UI; trace mining reads from it and has no local fallback, so it fails loudly if no
-Langfuse-compatible backend is reachable. To send traces to your own Langfuse without starting the
-bundled containers, set `LANGFUSE_*` and use `docker-compose.external-langfuse.yml` as documented in
+and UI. Langfuse remains the default mining source and fails loudly when unreachable. Historical
+Claude Code and Codex transcripts can be normalized locally as a separate, explicit source;
+external judging stays disabled until a mining run supplies an affirmative flag. See
+[Local coding-agent transcripts](docs/mcp-integration.md#local-coding-agent-transcripts). To send
+traces to your own Langfuse without starting the bundled containers, set `LANGFUSE_*` and use
+`docker-compose.external-langfuse.yml` as documented in
[Configuration](docs/configuration.md#using-your-own-langfuse-project). Backend, model, and gate
settings live in [Configuration](docs/configuration.md).
@@ -132,10 +175,10 @@ shows `SKILL.md` plus any bundled resources for the selected version. Browsing d
revision being served. Each skill row shows the total number of active, pending, and snapshotted
versions available in that explorer.
-Find a skill with an eval set and click **Optimize with SkillOpt**. The UI immediately explains
-that optimization can take a few minutes, disables the button, and opens the live generation log
-directly beneath that skill. The log updates automatically, so progress stays attached to the
-change you started instead of appearing in a page-level activity panel.
+For a skill without measured tasks, click **Create eval set**. The teacher drafts a separate
+train/holdout set and the live log stays attached to that skill. When the draft finishes, the card
+enables **Optimize with SkillOpt**. Optimization can take a few minutes; its attached log updates
+automatically and the resulting challenger remains quarantined.

@@ -219,7 +262,8 @@ Ingot does three things around your skill library:
human promotion. Promotion is snapshotted and recoverable.
- **Improve.** SkillOpt integration mines real traces for failing skills, trains bounded instruction
edits with its reflective optimizer, and A/Bs the result on held-out tasks, leaving a reviewable
- proposal that only a human can activate.
+ proposal that only a human can activate. Agents using `skill-retrospective` can also submit a
+ verified, revision-bound update through the MCP; it lands in the same inert review queue.
The component map is in [docs/how-it-works.md](docs/how-it-works.md); deeper design in
[ARCHITECTURE.md](ARCHITECTURE.md).
@@ -232,6 +276,7 @@ The component map is in [docs/how-it-works.md](docs/how-it-works.md); deeper des
| [How it works](docs/how-it-works.md) | Component map (MCP server, agent, optimizer, UI) |
| [Configuration](docs/configuration.md) | Env reference, SkillOpt optimization, cross-model compatibility, eval task sets, Langfuse |
| [The evidence gate](docs/evidence-gate.md) | The anti reward-hacking checks a reviewer relies on |
+| [Managed deployment](docs/managed-deployment.md) | One writer, publication backends, recovery, and what the audit trail does not guarantee |
| [Privacy & security](docs/security.md) | Zero-data-retention, network exposure, threat model |
| [Sign in with Google (SSO)](docs/sso.md) | Domain-restricted login and roles for a shared deployment |
| [Bring your own agent](docs/mcp-integration.md) | Use the MCP server from your own harness; tracing |
diff --git a/agent/run.py b/agent/run.py
index 0b34cc2..1c138b6 100644
--- a/agent/run.py
+++ b/agent/run.py
@@ -20,8 +20,8 @@
# Endpoint + ZDR handling is shared with the optimizer (single source of truth): OpenRouter
# endpoints get the hardcoded zero-data-retention provider preference; MODEL_BASE_URL points this
# serving role at a local vLLM/Ollama server instead (README: Privacy).
-from optimize import (ZDR_PROVIDER, agent_model, api_key, client_kwargs, model_api_key, # noqa: E402,F401
- model_base_url, skillopt_model, teacher_base_url)
+from ingot.optimize import (ZDR_PROVIDER, agent_model, api_key, client_kwargs, model_api_key, # noqa: E402,F401
+ model_base_url, skillopt_model, teacher_base_url)
MODEL = agent_model()
@@ -276,7 +276,7 @@ async def main(task: str):
routed = await _route(task, connected_tools, serving_tools)
_print_route(routed)
- from optimize import openrouter_key_missing
+ from ingot.optimize import openrouter_key_missing
if openrouter_key_missing():
print("\n[agent] OPENROUTER_API_KEY not set, showing router proposals only.")
print(" Set OPENROUTER_API_KEY in .env to run the deep agent (or point")
diff --git a/compose.dev.yaml b/compose.dev.yaml
new file mode 100644
index 0000000..cf798c1
--- /dev/null
+++ b/compose.dev.yaml
@@ -0,0 +1,52 @@
+# Development mode: an explicitly UNMANAGED stack.
+#
+# docker compose -f docker-compose.yml -f compose.dev.yaml up
+#
+# Every service gets the served library read-write again and the publisher is switched off. That is
+# convenient and it is not the product: nothing here prevents a service, a script, or a shell from
+# changing what is served without an approval, so quarantine and publication guarantees DO NOT
+# APPLY to a stack started this way. It must never be the configuration used to substantiate the
+# control-plane claim. `ingot status` reports UNMANAGED, and the `unmanaged` service below runs it
+# once at startup so the reason is in the log rather than in a document nobody reads.
+#
+# The managed default is `docker compose up` with no -f at all.
+services:
+ unmanaged:
+ build: .
+ command: ["ingot", "status"]
+ environment:
+ INGOT_MODE: dev # `ingot status` reports UNMANAGED for the deployment, not per skill
+ volumes:
+ - ./skills:/app/skills
+
+ publisher:
+ deploy:
+ replicas: 0 # there is no single writer in development mode
+
+ mcp:
+ environment:
+ INGOT_MODE: dev
+ volumes:
+ - ./skills:/app/skills
+
+ optimize:
+ volumes:
+ - ./skills:/app/skills
+
+ optimize-mine:
+ volumes:
+ - ./skills:/app/skills
+
+ optimize-compat:
+ volumes:
+ - ./skills:/app/skills
+
+ optimize-loop:
+ volumes:
+ - ./skills:/app/skills
+
+ ui:
+ environment:
+ INGOT_MODE: dev
+ volumes:
+ - ./skills:/app/skills
diff --git a/compose.forge.yaml b/compose.forge.yaml
new file mode 100644
index 0000000..ed44140
--- /dev/null
+++ b/compose.forge.yaml
@@ -0,0 +1,26 @@
+# The opt-in GitHub publication lane.
+#
+# INGOT_FORGE_REPOSITORY=owner/repo docker compose -f docker-compose.yml -f compose.forge.yaml up
+#
+# Publication authority becomes a merged pull request in the configured repository rather than the
+# local commit, so a receipt sits at `awaiting_merge` until the merge lands and the activation is
+# anchored somewhere a local administrator cannot quietly rewrite. It also means the deployment is
+# no longer air-gapped: the publisher needs the network, an authenticated `gh`, and a vault whose
+# remote is the configured repository.
+#
+# The publisher refuses to start unless `gh` is present, authenticated, and the repository
+# resolves — loudly, at startup, rather than on the first approval.
+#
+# ./vault must already be a clone of INGOT_FORGE_REPOSITORY. `ingot vault init` creates a local
+# vault with no remote and cannot make one for you; clone it yourself first.
+services:
+ publisher:
+ environment:
+ INGOT_PUBLISH_BACKEND: forge
+ INGOT_FORGE_REPOSITORY: ${INGOT_FORGE_REPOSITORY:?set INGOT_FORGE_REPOSITORY=owner/repo}
+ INGOT_FORGE_REMOTE: ${INGOT_FORGE_REMOTE:-origin}
+ INGOT_FORGE_BRANCH: ${INGOT_FORGE_BRANCH:-main}
+ volumes:
+ # `gh` reuses the host's credentials rather than taking a token of its own. Read-only: the
+ # publisher authenticates with them and must never rewrite them.
+ - ${GH_CONFIG_DIR:-${HOME}/.config/gh}:/root/.config/gh:ro
diff --git a/compose.gpu-embeddings.yml b/compose.gpu-embeddings.yml
new file mode 100644
index 0000000..c4a9266
--- /dev/null
+++ b/compose.gpu-embeddings.yml
@@ -0,0 +1,67 @@
+# Opt-in larger router model for NVIDIA hosts. Keep docker-compose.yml as the CPU-safe default:
+# docker compose -f docker-compose.yml -f compose.gpu-embeddings.yml \
+# --profile gpu-embeddings up -d embed-gpu mcp
+#
+# Qwen3-Embedding-8B is the highest-ranked permissively licensed text model supported by the
+# serving stack as checked on 2026-07-28. Q4_K_M keeps its weights inside the spare GPU capacity
+# on the production multi-tenant host; routing quality still needs the project eval before this
+# overlay becomes the production default.
+services:
+ embed-gpu:
+ image: ghcr.io/ggml-org/llama.cpp:server-cuda13-b9445@sha256:f92150249e1913ef96e744b5d78f6291f0e4399a7925ffc7b1d0680d82506551
+ command:
+ - --hf-repo
+ - Qwen/Qwen3-Embedding-8B-GGUF:Q4_K_M
+ - --embedding
+ - --pooling
+ - last
+ - --ctx-size
+ - "2048"
+ - --batch-size
+ - "2048"
+ - --ubatch-size
+ - "2048"
+ # Routing batches are short-lived and MCP serializes initialization; extra slots multiply
+ # context buffers on the shared production GPU without improving this caller.
+ - --parallel
+ - "1"
+ - --cache-ram
+ - "0"
+ - --n-gpu-layers
+ - "99"
+ - --host
+ - 0.0.0.0
+ - --port
+ - "8080"
+ - --no-webui
+ deploy:
+ resources:
+ reservations:
+ devices:
+ # Fully-qualified CDI avoids Compose v5 probing absent AMD/Intel vendors for `gpus:
+ # all`. NVIDIA Container Toolkit 1.18+ keeps this device spec current on the host.
+ - driver: cdi
+ device_ids: ["nvidia.com/gpu=all"]
+ capabilities: [gpu]
+ healthcheck:
+ test: ["CMD", "curl", "-f", "http://localhost:8080/health"]
+ interval: 5s
+ timeout: 3s
+ retries: 360
+ start_period: 30s
+ volumes:
+ - embed_models:/root/.cache/huggingface
+ profiles: ["gpu-embeddings"]
+
+ mcp:
+ depends_on:
+ embed-gpu:
+ condition: service_healthy
+ environment:
+ EMBED_BACKEND: remote
+ EMBED_BASE_URL: http://embed-gpu:8080/v1
+ EMBED_REMOTE_MODEL: Qwen/Qwen3-Embedding-8B-GGUF
+ EMBED_TIMEOUT_SECONDS: "120"
+
+volumes:
+ embed_models:
diff --git a/docker-compose.yml b/docker-compose.yml
index ec610cd..5035f9f 100644
--- a/docker-compose.yml
+++ b/docker-compose.yml
@@ -32,16 +32,51 @@ services:
- ./ops/keycloak:/opt/keycloak/data/import:ro
profiles: ["sso"]
+ # The one writer of the served library. Every other service mounts ./vault read-only, so an
+ # approved change reaches what is served through this process or not at all. The default backend
+ # is `local`: the vault is a Git repository on this machine and no network is involved. See
+ # compose.forge.yaml for the opt-in GitHub lane, and compose.dev.yaml for the writable stack.
+ #
+ # `ingot vault init` runs on every start and is idempotent, so a first `docker compose up` on an
+ # empty checkout is not a failure. Approval writes runs/publications at mode 0700, so this and
+ # the ui must run as the same user; both default to root and both follow INGOT_UID/INGOT_GID.
+ publisher:
+ build: .
+ user: "${INGOT_UID:-0}:${INGOT_GID:-0}"
+ command: ["sh", "-c", "ingot vault init /app/vault && exec python -m ingot.optimize.publisher --watch"]
+ environment:
+ INGOT_PUBLISH_BACKEND: ${INGOT_PUBLISH_BACKEND:-local}
+ # State is configuration, not a directory next to the code. Every service names what it
+ # mounted; a default would put the review queue and the receipts inside the image.
+ INGOT_VAULT_PATH: /app/vault
+ INGOT_LIBRARY: /app/vault
+ INGOT_RUNS: /app/runs
+ # INGOT_DELIVERY_TARGETS: claude=filesystem:/app/targets/claude
+ volumes:
+ - ./vault:/app/vault # the only writable mount of the served library in this stack
+ - ./runs:/app/runs # the approval receipts it consumes
+ # Native delivery: mount an agent's skill directory here and name it in
+ # INGOT_DELIVERY_TARGETS, and the publisher installs each approved revision into it as well
+ # as the vault (docs/managed-deployment.md, "Delivery targets"). Writable on purpose -- this
+ # is the destination, and the publisher is still the only thing that writes it.
+ # - ~/.claude/skills:/app/targets/claude
+
mcp:
build: .
- command: python -m mcp_server.server
+ # Creation proposals are mode 0600 in the same runs/ queue the UI reads. Keep both services on
+ # one configured host identity or MCP-authored proposals become invisible to review.
+ user: "${INGOT_UID:-0}:${INGOT_GID:-0}"
+ command: python -m ingot.mcp_server.server
environment:
HOST: 0.0.0.0 # bind the container interface (needed for the port publish);
# host access stays localhost-only via the 127.0.0.1 mapping
+ INGOT_LIBRARY: /app/skills
+ INGOT_RUNS: /app/runs
+ INGOT_TASKS: /app/ingot/optimize/tasks
ports:
- - "127.0.0.1:8000:8000" # MCP streamable-HTTP, localhost only (see README: Network exposure)
+ - "127.0.0.1:${INGOT_MCP_PORT:-8000}:8000" # MCP streamable-HTTP, localhost only (see README: Network exposure)
volumes:
- - ./skills:/app/skills # promoted skills must reach the running server (hot reload)
+ - ./vault:/app/skills:ro # the served library, read-only: only the publisher may change it
- ./runs:/app/runs # usage counts (runs/skill_usage.json) must reach the host runs/ the ui reads
agent:
@@ -52,6 +87,9 @@ services:
- path: .env # OPENROUTER_API_KEY (see .env.example)
required: false # a missing .env/key gets a friendly in-app message instead
environment:
+ INGOT_LIBRARY: /app/skills
+ INGOT_RUNS: /app/runs
+ INGOT_TASKS: /app/ingot/optimize/tasks
MCP_URL: http://mcp:8000/mcp
LANGFUSE_BASE_URL: ${LANGFUSE_BASE_URL:-http://langfuse-web:3000}
LANGFUSE_PUBLIC_KEY: ${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo}
@@ -68,19 +106,22 @@ services:
# Writes a quarantined pending record; it can never activate a skill.
optimize:
build: .
- entrypoint: ["python", "-m", "optimize.ab"]
+ entrypoint: ["python", "-m", "ingot.optimize.ab"]
env_file:
- path: .env
required: false
environment:
+ INGOT_LIBRARY: /app/skills
+ INGOT_RUNS: /app/runs
+ INGOT_TASKS: /app/ingot/optimize/tasks
MCP_URL: http://mcp:8000/mcp
LANGFUSE_BASE_URL: ${LANGFUSE_BASE_URL:-http://langfuse-web:3000}
LANGFUSE_PUBLIC_KEY: ${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo}
LANGFUSE_SECRET_KEY: ${LANGFUSE_SECRET_KEY:-sk-lf-local-demo}
volumes:
- - ./skills:/app/skills
+ - ./vault:/app/skills:ro # read-only: only the publisher may change what is served
- ./runs:/app/runs
- - ./optimize/tasks:/app/optimize/tasks # auto-drafted eval sets must survive the container
+ - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # auto-drafted eval sets must survive the container
- /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers (README: execution-grounded judging)
depends_on:
mcp:
@@ -92,18 +133,21 @@ services:
# Success/failure mining over real Langfuse traces: `docker compose run --rm optimize-mine pdf`
optimize-mine:
build: .
- entrypoint: ["python", "-m", "optimize.mine"]
+ entrypoint: ["python", "-m", "ingot.optimize.mine"]
env_file:
- path: .env
required: false
environment:
+ INGOT_LIBRARY: /app/skills
+ INGOT_RUNS: /app/runs
+ INGOT_TASKS: /app/ingot/optimize/tasks
LANGFUSE_BASE_URL: ${LANGFUSE_BASE_URL:-http://langfuse-web:3000}
LANGFUSE_PUBLIC_KEY: ${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo}
LANGFUSE_SECRET_KEY: ${LANGFUSE_SECRET_KEY:-sk-lf-local-demo}
volumes:
- - ./skills:/app/skills
+ - ./vault:/app/skills:ro # read-only: only the publisher may change what is served
- ./runs:/app/runs # usage counts + mined-candidate output
- - ./optimize/tasks:/app/optimize/tasks # train sets feed the mined-candidate leakage guard
+ - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # train sets feed the mined-candidate leakage guard
- /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers (README: execution-grounded judging)
depends_on:
langfuse-web:
@@ -115,14 +159,18 @@ services:
# Set COMPAT_MODELS=modelA,modelB,... in .env. Langfuse-free (local rollout + judge).
optimize-compat:
build: .
- entrypoint: ["python", "-m", "optimize.compat"]
+ entrypoint: ["python", "-m", "ingot.optimize.compat"]
env_file:
- path: .env
required: false
+ environment:
+ INGOT_LIBRARY: /app/skills
+ INGOT_RUNS: /app/runs
+ INGOT_TASKS: /app/ingot/optimize/tasks
volumes:
- - ./skills:/app/skills
+ - ./vault:/app/skills:ro # read-only: only the publisher may change what is served
- ./runs:/app/runs # writes the compatibility matrix to runs/compat/
- - ./optimize/tasks:/app/optimize/tasks # held-out task sets (auto-drafted if missing)
+ - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # held-out task sets (auto-drafted if missing)
- /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers
profiles: ["optimize"]
@@ -130,19 +178,22 @@ services:
# for review: `docker compose run --rm optimize-loop`
optimize-loop:
build: .
- entrypoint: ["python", "-m", "optimize.loop"]
+ entrypoint: ["python", "-m", "ingot.optimize.loop"]
env_file:
- path: .env
required: false
environment:
+ INGOT_LIBRARY: /app/skills
+ INGOT_RUNS: /app/runs
+ INGOT_TASKS: /app/ingot/optimize/tasks
MCP_URL: http://mcp:8000/mcp
LANGFUSE_BASE_URL: ${LANGFUSE_BASE_URL:-http://langfuse-web:3000}
LANGFUSE_PUBLIC_KEY: ${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo}
LANGFUSE_SECRET_KEY: ${LANGFUSE_SECRET_KEY:-sk-lf-local-demo}
volumes:
- - ./skills:/app/skills
+ - ./vault:/app/skills:ro # read-only: only the publisher may change what is served
- ./runs:/app/runs
- - ./optimize/tasks:/app/optimize/tasks # auto-drafted eval sets must survive the container
+ - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # auto-drafted eval sets must survive the container
- /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers (README: execution-grounded judging)
depends_on:
mcp:
@@ -154,13 +205,21 @@ services:
# Change-control UI (review evidence, promote, roll back): http://localhost:8080
ui:
build: .
+ # Approval writes runs/publications at mode 0700, and the vault publisher reads those receipts
+ # from the host as an ordinary user. Where both run on one machine, set INGOT_UID/INGOT_GID in
+ # .env to that user, or the publisher silently sees an empty queue and approvals never publish.
+ # The default keeps the container as root, which is every deployment that has no host publisher.
+ user: "${INGOT_UID:-0}:${INGOT_GID:-0}"
command: python -m uvicorn ui.app:app --host 0.0.0.0 --port 8080
ports:
- - "127.0.0.1:8080:8080" # localhost only; LAN access goes through the TLS proxy (--profile lan)
+ - "127.0.0.1:${INGOT_UI_PORT:-8080}:8080" # localhost only; LAN access goes through the TLS proxy (--profile lan)
env_file:
- path: .env
required: false
environment:
+ INGOT_LIBRARY: /app/skills
+ INGOT_RUNS: /app/runs
+ INGOT_TASKS: /app/ingot/optimize/tasks
MCP_URL: http://mcp:8000/mcp
# Change-control UI login. AUTH_MODE defaults to password (HTTP Basic) with the documented
# default credentials admin/ingot, CHANGE AUTH_PASSWORD in .env before exposing the UI beyond
@@ -184,9 +243,9 @@ services:
LANGFUSE_SECRET_KEY: ${LANGFUSE_SECRET_KEY:-sk-lf-local-demo}
LANGFUSE_PUBLIC_URL: ${LANGFUSE_PUBLIC_URL:-http://localhost:3100} # where the browser reaches Langfuse
volumes:
- - ./skills:/app/skills
+ - ./vault:/app/skills:ro # read-only: only the publisher may change what is served
- ./runs:/app/runs
- - ./optimize/tasks:/app/optimize/tasks # auto-drafted eval sets must survive the container
+ - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # auto-drafted eval sets must survive the container
- /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers (README: execution-grounded judging)
depends_on:
mcp:
diff --git a/docs/configuration.md b/docs/configuration.md
index 25518f4..0c8e8fa 100644
--- a/docs/configuration.md
+++ b/docs/configuration.md
@@ -16,6 +16,11 @@ Set in `.env` (never committed):
| `RELATED_SCORE` | `0.37` | floor of the `related` band; below it a task is novel (weak/strong escalation). Calibrated to `EMBED_MODEL` (0.45 for bge-small) |
| `EMBED_MODEL` | `onnx-community/Qwen3-Embedding-0.6B-ONNX` | router embedding model (q4 ONNX, ~15 ms/query on CPU; +7 top-1 over the former bge-small default on a 297-query eval). The default Qwen files are baked into the image and loaded cache-only. Ingot does not use BGE-M3. Any fastembed name also works, but an unbaked override may download at first use and requires recalibrating the three score thresholds. Keep in sync with the Dockerfile's build arg |
| `EMBED_ONNX_FILE` | `onnx/model_q4.onnx` | which ONNX weight file to load inside the `EMBED_MODEL` repo; only relevant for ONNX exports that ship multiple quantizations |
+| `EMBED_BACKEND` | `local` | `local` uses the in-process CPU backend; `remote` uses an OpenAI-compatible embedding server and fails if it cannot be reached |
+| `EMBED_BASE_URL` | unset | required for `remote`, including `/v1` (for example `http://embed-gpu:8080/v1`) |
+| `EMBED_REMOTE_MODEL` | unset | required for `remote`; model identifier sent to the server, separate from the local `EMBED_MODEL` so GPU mode cannot accidentally request the CPU ONNX export |
+| `EMBED_TIMEOUT_SECONDS` | `60` | remote embedding request timeout |
+| `ROUTER_BODY_CHARS` | `1000` | approved body-prefix characters in each body-aware routing document. Name and description are always included; harness variants are embedded separately. Must be 1–4000. The hard ceiling prevents the decoder-style CPU ONNX attention tensor from exceeding 4 GB |
| `BODY_TARGET_CHARS` | `6000` | length penalty starts past this body size |
| `LENGTH_PENALTY` | `0.10` | max score subtracted for a very long body |
| `LOOP_HEALTH_THRESHOLD` | `0.7` | the background loop proposes a change for skills whose mined mean score is below this |
@@ -52,6 +57,12 @@ OIDC/SSO variables (`OIDC_ISSUER`, `OIDC_CLIENT_ID`, `OIDC_CLIENT_SECRET`, `OIDC
Evidence-gate knobs (`PROMOTE_MIN_MARGIN`, `PROMOTE_MIN_SAMPLES`, `COLLISION_SCORE`,
`JUDGE_MODELS`) are covered in [The evidence gate](evidence-gate.md).
+Routing responses report `matched_on` (`description` or `content`) and both
+cosine values under `score_components`. The aggregate is their maximum, so
+existing description scores cannot decrease. Body evidence can create new
+matches, however, so run a representative held-out routing suite before
+changing thresholds or enabling a new library at scale.
+
### SkillOpt optimization
The body pass trains the skill body with **[SkillOpt](https://github.com/microsoft/SkillOpt)**'s
@@ -99,12 +110,12 @@ the local rollout + judge, so it needs no Langfuse.
### Writing eval task sets
Task sets are runtime artifacts, not shipped opinions; the repo commits none. They live in
-`optimize/tasks/.yaml` (gitignored). Create one by hand or let `SKILLOPT_MODEL` auto-draft one
+`ingot/optimize/tasks/.yaml` (gitignored). Create one by hand or let `SKILLOPT_MODEL` auto-draft one
on the first CLI optimize run.
To author one manually:
-1. Create `optimize/tasks/.yaml`, where `` exactly matches the directory name under
+1. Create `ingot/optimize/tasks/.yaml`, where `` exactly matches the directory name under
`skills/`.
2. Add separate `train:` and `holdout:` lists. The candidate search sees only `train`; the evidence
gate sees only `holdout`. A flat `tasks:` list is treated as train/holdout leakage and cannot
diff --git a/docs/how-it-works.md b/docs/how-it-works.md
index 8323459..41eb6f6 100644
--- a/docs/how-it-works.md
+++ b/docs/how-it-works.md
@@ -4,7 +4,7 @@
-- **`mcp_server/`**: [FastMCP](https://github.com/jlowin/fastmcp) v3 server (HTTP transport), five tools:
+- **`mcp_server/`**: [FastMCP](https://github.com/jlowin/fastmcp) v3 server (HTTP transport), six tools:
- `suggest_skills(task, k)`: routable matches by embedding similarity (Qwen3-Embedding-0.6B
q4 on CPU ONNX Runtime, no GPU; any fastembed model via `EMBED_MODEL`); near-misses come back flagged
`related`; empty = truly novel
@@ -14,6 +14,8 @@
- `route_and_load(task, harness, cwd, available_tools, available_mcps)`: one-round-trip
selection and loading for direct or related compatible routes (see
[Bring your own agent](mcp-integration.md#bring-your-own-agent-mcp-only))
+ - `propose_skill_update(...)`: revision-bound `skill-retrospective` update submission; creates
+ one inert pending challenger and never activates or displaces instructions
- **`agent/run.py`**: [deepagents](https://github.com/langchain-ai/deepagents) LangGraph agent
wired to those tools, traced to Langfuse. Serves routed tasks on the weak `AGENT_MODEL` and
escalates truly novel tasks to `STRONG_MODEL`.
@@ -21,12 +23,15 @@
loads. Its folder's content hash is its revision.
- **`optimize/promote.py`**: the change-control core, and the only module that writes under
`skills/`: the pending queue, the evidence check, revision snapshots, the atomic promotion and
- rollback swaps, and the approval-audit append.
+ rollback swaps, and the approval-audit append. When the serving revision comes from a read-only
+ mounted library, approval snapshots that source and atomically installs the challenger in the
+ first-precedence writable `skills/` root; the source mount remains unchanged.
- **`optimize/`**: the SkillOpt integration and evaluation pipeline: trace mining (`mine.py`),
multi-dimensional LLM judge
(`judge.py`), the SkillOpt candidate search (`skillopt_loop.py` + `skillopt_bridge.py`) and its
rollout/teacher plumbing (`rollout.py`),
- held-out A/B (`ab.py`), the portable evidence bundle (`evidence.py`), the routing pass
+ held-out A/B (`ab.py`), the portable evidence bundle (`evidence.py`), retrospective proposal
+ ingestion (`retrospective.py`), the routing pass
(`routing.py`), the background loop (`loop.py`), the library-wide routing health check
(`routing_health.py`, embedding-only, cron/CI-friendly, read-only), token ledger (`usage.py`).
None of these can activate anything: most write pending records; `routing_health.py` writes
diff --git a/docs/managed-deployment.md b/docs/managed-deployment.md
new file mode 100644
index 0000000..3dc73fe
--- /dev/null
+++ b/docs/managed-deployment.md
@@ -0,0 +1,350 @@
+# Managed deployment
+
+The claim is that only an approved change reaches what is served. This page is how that is
+enforced, how to check it on your own machine, and exactly what it does and does not protect
+against.
+
+## One writer
+
+The served skill library is a Git repository — the vault. Every service mounts it read-only except
+the publisher, which is the only process allowed to change it and only acts on an approved receipt.
+
+```text
+ingot add file:./pkg quarantine the library is byte-identical
+ingot add github:OWNER/REPO --skill path/to/pkg
+ quarantine the library is byte-identical
+approve receipt written the library is byte-identical
+publisher commit, activate the library now serves the approved revision
+```
+
+The whole loop from a terminal, none of which writes a served byte:
+
+```bash
+ingot add file:./pkg # quarantine a package for review
+ingot add github:OWNER/REPO --skill path/to/pkg
+ingot review ./pkg # what is wrong with it, offline
+ingot pending # what is waiting on a decision
+ingot approve csv-tidy # queue a publication receipt
+ingot history csv-tidy # snapshots, receipts, and the decision trail
+ingot rollback csv-tidy
+ingot status # is this deployment still what was approved?
+```
+
+`approve`, `reject`, and `rollback` call the same services the console calls. There is no second
+approval path and no command that activates a skill directly.
+
+Approval is a human gate that writes a receipt. It does not touch the library. The publisher reads
+the receipt, materializes exactly the components the receipt names, runs the vault's validator,
+commits, snapshots the revision it is about to displace, fast-forwards the served checkout, and
+re-verifies that what is served is the revision the receipt named. Any mismatch at any step fails
+the receipt and changes nothing.
+
+`docker compose up` with no `-f` is the managed stack. That is deliberate: a default that launched
+a writable stack while this page described controlled activation would make the claim untrue for
+almost every reader.
+
+## Is it actually managed here?
+
+```bash
+ingot status
+ingot status --json
+```
+
+Four answers, decided per skill by comparing what is served against what the last successful
+release receipt says should be served:
+
+| | Meaning |
+|---|---|
+| `MANAGED` | Every served skill is exactly the revision its release receipt names. |
+| `PENDING` | A proposal or publication is in flight. Nothing has drifted. |
+| `DRIFTED` | Served bytes differ from the last successful release. Something changed them outside the publisher. |
+| `UNMANAGED` | Some served bytes have no release receipt behind them, or the deployment is in development mode. |
+
+The deployment reports the worst of them, and exits non-zero for anything but `MANAGED`, so a check
+can assert it. This is an observation, not a configuration flag: a flag would have agreed with the
+claim rather than tested it, which is the failure this command exists to catch.
+
+A skill with no release receipt is `UNMANAGED`, not `DRIFTED`. Fetched, copied, and hand-committed
+skills are real and common; there is no release for them to have drifted from, and calling that
+drift would make the alarm mean nothing.
+
+Whether the library is writable by the calling process is reported alongside the verdict but does
+not decide it. The administrator who owns the vault can always write it, and a status command that
+answered `UNMANAGED` from their shell would hide the drift they most need to see.
+
+## Drift, and what to do about it
+
+A read-only mount does not stop the machine owner from editing the host directory. Rather than
+claim it does, Ingot detects it:
+
+```text
+DRIFTED (uid 1000)
+
+ Served bytes differ from the last successful release. Something changed them outside the publisher.
+ DRIFTED csv-tidy served 35e6c7a05313 != released e660639d81da
+```
+
+Two ways back to `MANAGED`:
+
+- **Restore the released bytes.** In the local backend the served checkout is the vault, so the
+ edit is an uncommitted change: `git -C vault checkout -- `. This is an explicit
+ administrator action on the vault, not something Ingot does behind the publisher's back.
+- **Keep the change and get it approved.** Copy the edited directory somewhere else, restore the
+ vault, and submit the copy: `ingot add file:./that-copy`. It goes through review and approval
+ like any other proposal.
+
+Note the ordering. Until the vault checkout is clean the publisher refuses to run at all — a dirty
+vault is exactly the state it must not build on — so a drifted deployment cannot publish its way
+out. That is a deliberate refusal, not a deadlock: restoring first is one command.
+
+**Deliberately not built:** a single `ingot reconcile` that quarantines the drifted bytes for you.
+It needs an answer to a question this design has not settled — a reconcile proposal's champion is
+the last release while the disk holds the drifted bytes, so the existing freshness check refuses
+it, and every way past that either fabricates evidence or opens a second approval path. The two
+steps above do the same work with no new mutation path.
+
+## Development mode
+
+```bash
+docker compose -f docker-compose.yml -f compose.dev.yaml up
+```
+
+Every service gets the library read-write again and the publisher is switched off. It is convenient
+for working on Ingot itself. **Quarantine and publication guarantees do not apply to a stack
+started this way**, and it must not be the configuration used to substantiate the control-plane
+claim. The stack runs `ingot status` at startup so the reason is in the log.
+
+## Where state lives
+
+Nothing mutable is kept beside the code. The served library, the review queue, publication
+receipts, evidence bundles, snapshots, and eval task sets all resolve through one setting:
+
+```text
+INGOT_HOME # $XDG_STATE_HOME/ingot, else ~/.local/state/ingot
+├── library/ # INGOT_LIBRARY (SKILLS_DIR is the deprecated name)
+├── runs/ # INGOT_RUNS pending, publications, evidence, revisions
+├── tasks/ # INGOT_TASKS
+└── vault/ # INGOT_VAULT_PATH, defaults to the library
+```
+
+`INGOT_HOME` moves all of them; the specific settings override it one at a time, which is what the
+compose stack does — every service names the paths it mounted rather than relying on a default.
+
+```bash
+ingot status # prints every resolved path, where it came from, and whether it is writable
+```
+
+This is not cosmetic. A `pip install ingot` used to keep its review queue and its receipts inside
+`site-packages`, which meant an upgrade discarded them, a read-only or system Python could not
+start, and two deployments sharing one installation shared one queue. If `ingot status` finds state
+left there by an earlier version it says so and does nothing else: moving a review queue on your
+behalf is a change to controlled state made by a process nobody asked to make it.
+
+## Publication backends
+
+Selected explicitly with `INGOT_PUBLISH_BACKEND`, never inferred. A vault that later gains a remote
+does not start opening pull requests on its own.
+
+### `local` (default)
+
+The vault is a Git repository on this machine. No network, no GitHub account, no `gh`, no remote
+origin required. `ingot vault init ` creates one; the managed compose runs it on every start
+and it is idempotent.
+
+States: `approved_publishing → publishing → active`. There is no `awaiting_merge`, because there is
+nothing external to wait for — the human gate is the approval.
+
+The vault may have other legitimate writers; a person committing to it directly is fine and the
+publisher fast-forwards onto their work. What the publisher will never do is rebase, merge, or
+force. If a publication branch can no longer fast-forward, the receipt fails with an inspectable
+error and nothing moves; the next attempt re-cuts the branch from the vault as it now stands and
+re-checks the champion, so an unrelated commit resolves itself and a conflicting one is refused.
+
+### `forge` (opt-in)
+
+```bash
+INGOT_FORGE_REPOSITORY=owner/repo \
+ docker compose -f docker-compose.yml -f compose.forge.yaml up
+```
+
+Publication authority becomes a merged pull request. States:
+`approved_publishing → publishing → awaiting_merge → active`. The vault must already be a clone of
+the configured repository. The publisher verifies `gh` is present, authenticated, and that the
+repository resolves — at startup, loudly, rather than on the first approval.
+
+This anchors activation somewhere a local administrator cannot quietly rewrite. It also ends the
+air gap.
+
+| | `local` | `forge` |
+|---|---|---|
+| Network | none | required |
+| Authority | the approval | a merged pull request |
+| Activation record | local Git history | Git history, mirrored off-box |
+| Air-gappable | yes | no |
+
+## Delivery targets
+
+The vault is the managed-MCP library: agents that load skills through Ingot's MCP server read the
+same checkout the publisher commits into. An agent that reads a native skill directory on disk
+reads nothing at all. A delivery target is that second destination.
+
+Configure them with `INGOT_DELIVERY_TARGETS`, a comma-separated list of `name=kind:path`:
+
+```sh
+INGOT_DELIVERY_TARGETS=claude=filesystem:~/.claude/skills,codex=filesystem:~/.codex/skills
+```
+
+Two kinds:
+
+| Kind | What it is |
+|---|---|
+| `managed-mcp` | the vault itself, always present, always named `vault` unless you name it |
+| `filesystem` | a directory the publisher installs approved revisions into |
+
+Ingot knows nothing about Codex or Claude beyond those names being yours to choose. A filesystem
+target is a directory; what reads it is not Ingot's business.
+
+**What delivery does not change.** Publication stays receipt-driven and human-approved, and the
+publisher stays the only supported writer. Delivery runs *after* the vault serves the approved
+revision and *before* the receipt is marked `active`, so a target that cannot be written leaves a
+release that retries rather than one that reports itself finished in places it never reached. Each
+target's outcome is recorded on the receipt separately, under `delivery`.
+
+**The managed target is a deliberate no-op.** It has a name, a status, and a line on every receipt,
+but the publication commit and the fast-forward are the only things that write the vault. A second
+writer there is the one thing this control plane exists to prevent.
+
+**Installing is atomic.** The approved revision is staged beside the destination and swapped in with
+same-filesystem renames. A failure between the two renames puts the displaced directory back, so an
+agent never loads a skill folder that is neither the old revision nor the new one. Whatever the
+target held is snapshotted first — keyed by the revision of the bytes actually there, so a target
+someone edited by hand is recoverable too.
+
+**Rollback needs nothing extra.** It travels the ordinary publication queue, so every target returns
+to the prior approved revision on the way through.
+
+**Drift is per target.** `ingot status` reports each one separately, and a drifted target counts
+toward the overall verdict — a status that answered MANAGED while a native skill root served the
+wrong bytes would be the lie the command exists to prevent. A target is graded only on the skills
+Ingot released there: a native skill root is shared with whatever its owner put in it, and those are
+not Ingot's to judge.
+
+`route_and_load` is unchanged and stays the managed-MCP delivery contract. A native agent activates
+from its own skill directory in whatever way it already does; Ingot's router is not mandatory.
+
+## Recovery
+
+`process()` is re-entrant, and a kill at any point leaves a state the next pass resolves:
+
+| Killed | On restart |
+|---|---|
+| before the worktree is cut | re-prepared from scratch |
+| worktree cut, before the commit | the stale worktree is destroyed and recut |
+| after the commit, before activation | the branch is reused, re-authorized, activated |
+| after activation, before the receipt | the receipt is finalized; **nothing is re-snapshotted** |
+| after the receipt | no work; the stored state is returned |
+
+The fourth row is the one that matters. The served bytes already equal the candidate, so the
+champion a second snapshot would capture is gone; re-snapshotting would refuse a publication that
+has in fact already activated.
+
+## Artifact fidelity
+
+A revision names the exact package. `ingot add` stages every regular file byte-for-byte into a
+**candidate tree** under `runs/candidates//`, and the receipt carries a manifest recording
+each file's relative path, mode, size, and SHA-256 of its raw bytes. Publication copies that staged
+tree into the vault worktree and verifies every hash on the way, so a file whose bytes moved between
+the approval and the publication stops the publication instead of being served.
+
+Hashes are of bytes, never of decoded text: a file that is not valid UTF-8 has no decoded form, and
+one that is would hash differently after a round trip — which is exactly how files used to go
+missing.
+
+Two behaviours are deliberate, and both are visible rather than silent:
+
+- **SKILL.md is normalized, not preserved.** Its frontmatter is the routing interface, so the name
+ is forced to the skill's identity, the description is collapsed to one line, and the file is
+ re-emitted through a safe YAML dump. The manifest still records the source file's real hash and
+ size, so the normalization is auditable. The approved revision is computed by performing exactly
+ this materialization, so it is the revision the library serves.
+- **Symlinks are refused.** `ingot review` reports `symlink-unsupported` and `ingot add` stops.
+ Preserving a link puts a path into the vault that leads a reader back out of the library;
+ flattening it into its target silently changes the artifact's shape. Neither is a decision
+ admission should make on an operator's behalf.
+
+For `github:`, acquisition resolves the public repository's `HEAD` to a commit before cloning and
+refuses if the cloned commit differs. It inspects the selected Git tree before fetching its blobs,
+rejects gitlinks and symlinks, then reads each blob without a checkout. Repository attributes and
+checkout filters cannot rewrite or execute while those bytes enter quarantine. The candidate
+records the repository, requested ref, commit, subdirectory, and tree digest.
+
+Assets a reviewer cannot read are reported rather than refused:
+
+```console
+$ ingot review ./csv-tidy
+structural
+ warning binary-asset: 1 file(s) are not text and cannot be read before approval; they will be
+ published byte-for-byte: assets/logo.png
+```
+
+Decodability decides, not the file extension — an extension is a claim about a file, and the point
+is to check the file. Editor and VCS metadata (`.git/`, `__pycache__/`, `.DS_Store`) is not skill
+content and is not reported; a finding that fires on `.DS_Store` is one people learn to scroll past.
+
+Modes are clamped to `0644` or `0755`, the two a Git checkout reproduces. A package is capped at
+256 files and 20 MB; both refusals name the limit.
+
+Staged trees are named by their digest, so resubmitting the same package reuses one directory
+rather than making a second copy. Nothing removes them afterwards: a rejected proposal leaves its
+tree in `runs/candidates/`, bounded by the per-package cap, and deleting them is a housekeeping
+decision rather than something publication should make on its own.
+
+## What the audit trail actually guarantees
+
+Publication history is Git-backed, revision-bound, and externally anchorable in `forge` mode.
+Stated plainly, because a control plane that overstates this is worse than one that has none:
+
+- **Revision-bound.** Every revision is a digest of the exact package — the parsed SKILL.md plus the
+ raw bytes of every other file — so the receipt names specific bytes and a moved tag cannot stand
+ in for them.
+- **Git-backed.** Each publication is a commit with the receipt id in its message, so what was
+ served when is reconstructable from history.
+- **Detects normal inconsistency.** A champion that changed under a publication, a materialization
+ that does not match the approved revision, a staged candidate file whose bytes moved since
+ approval, a served checkout that does not match after activation, and a pending review that no
+ longer matches its receipt are all refused.
+- **Externally anchorable** in `forge` mode, where the activation record exists somewhere the local
+ machine does not control.
+
+Git is not by itself proof that history did not change. A local repository can be rewritten by
+anyone with a shell in it; a GitHub repository can be force-pushed or administratively altered, and
+`forge` mode is an external anchor rather than an immutable transparency log. Someone with root can
+rewrite the vault history, the receipts, and the audit log together. A signed log or a real
+transparency log is the answer to that threat, and it waits for a concrete threat model rather than
+being guessed at now. The records deliberately carry no signature field: a local record an
+administrator can rewrite must not carry anything shaped like proof that they did not.
+
+## Proving it on your own machine
+
+```bash
+scripts/managed_smoke.sh
+```
+
+Starts the managed stack and checks what Docker actually enforces: `mcp` and `ui` must fail to
+write the served library, the publisher must succeed on the vault, what `mcp` serves must be the
+commit the publisher's vault is at, and `ingot status` inside the stack must report `MANAGED`.
+
+`tests/test_compose_managed.py` checks the same invariant against the tracked YAML on every test
+run, which catches a regression in the configuration but cannot prove the containers behave.
+
+CI runs `managed_smoke.sh` on a Linux runner as a required check, and then runs it again with one
+`:ro` deliberately removed and requires it to fail. A check that cannot fail proves nothing.
+
+## Running the publisher on the host instead
+
+`ops/systemd/ingot-publisher.service` runs it as a user unit. That sidesteps the uid mismatch
+between a container writing receipts at mode 0700 and a host process reading them, and in `forge`
+mode it reuses the host's already authenticated `git` and `gh` so no credential has to live in a
+container. Copy `ops/systemd/publisher.env.example` to `~/.config/ingot/publisher.env` first.
+
+Whichever you run — the compose service or the unit — exactly one must.
diff --git a/docs/mcp-integration.md b/docs/mcp-integration.md
index 19f7c93..fe35ba2 100644
--- a/docs/mcp-integration.md
+++ b/docs/mcp-integration.md
@@ -136,6 +136,11 @@ returned skill_body while completing the request. If it returns novel, continue
Do not merely list or suggest the skill: load it and apply it before doing the task.
```
+When a loaded `skill-retrospective` produces a verified update for an existing skill, agents may
+call `ingot.propose_skill_update`. The tool only files a revision-bound challenger in Ingot's
+review queue. A `quarantined` response does not change the served revision: do not reload skills or
+claim the update is active. Human approval in the console remains a separate action.
+
Use the same rule in organization-managed agent instructions if repositories should not carry local
agent files. After enrollment, verify behavior with a harmless request and confirm both the
`route_and_load` tool call and final answer appear in Langfuse. A successful `--doctor` result proves
@@ -206,13 +211,46 @@ unchanged. Two caveats: mining re-judges traffic with `JUDGE_MODEL` (on your API
candidate rollouts still execute on the bundled scaffold, so set `AGENT_MODEL` to your production
serving model.
+### Local coding-agent transcripts
+
+Existing Claude Code and Codex JSONL transcripts can feed the same miner without first uploading
+them to Langfuse:
+
+```bash
+python -m ingot.optimize.local_traces
+python -m ingot.optimize.mine --source local --allow-external-judge
+```
+
+The scan writes `runs/local_traces.json`. It keeps completed human turns, final answers, observed
+skill names, exact revisions returned by `ingot.route_and_load`, timing, token counts, and tool
+error counts. It excludes reasoning, attachments, tool arguments and results, injected agent
+instructions, hook output, compaction records, aborted turns, and subagent threads. A historical
+skill use without a served revision stays unpinned; the scanner never substitutes the current
+revision.
+
+Scanning is local and makes no model call. Repeated scans reuse unchanged transcript files. Bound
+the snapshot at import time with repeatable `--project `, `--since YYYY-MM-DD`, and
+`--until YYYY-MM-DD`; use `--force` after a same-size transcript rewrite whose mtime was preserved.
+The console also filters the imported snapshot by project, agent, and date.
+
+Local mining fails closed unless `--allow-external-judge` is present. That flag is the explicit
+paid/data-egress boundary: selected task/answer pairs go to `JUDGE_MODEL` under the normal usage
+cap. Langfuse and local snapshots are separate sources rather than an implicitly merged corpus.
+The console's Traces view reads a safe projection of the normalized store and never returns answer
+text. Task previews are also hidden until the reviewer enables **Show task previews**; previews
+are capped at 280 characters.
+
+When the console runs on another host, transfer the normalized file through the deployment's
+existing trusted channel into that checkout's `runs/local_traces.json`. Do not mount or copy raw
+home-directory transcripts into the UI container.
+
## Using your own evals platform
-Langfuse is the **default and required** evals backend: it comes up with `docker compose up`, and
-trace mining has no local fallback (`optimize-mine` fails loudly if no Langfuse-compatible endpoint
-is reachable, rather than returning an empty result that would read as "nothing failing"). You have
-three options:
+Langfuse is the **default online** evals backend: it comes up with `docker compose up`, and the
+default trace source fails loudly if no Langfuse-compatible endpoint is reachable, rather than
+returning an empty result that would read as "nothing failing". The local transcript source above
+is an explicit historical backfill path, not an online backend. You have three online options:
1. **Bundled Langfuse** (default): self-hosted in the compose stack, nothing to configure. Secure
its demo credentials before exposing it: [Securing the Langfuse deployment](security.md#securing-the-langfuse-deployment).
diff --git a/docs/security.md b/docs/security.md
index 1101944..4866921 100644
--- a/docs/security.md
+++ b/docs/security.md
@@ -52,6 +52,13 @@ Write paths, and what guards each:
- **Generated rewrites** land in `runs/pending/` and cannot activate themselves. They also require
evidence whose champion and challenger revisions still match the skill on disk before UI approval.
+- **Retrospective MCP submissions** may create one bounded pending update for an existing skill.
+ They require passed pressure verification, bind to the exact loaded champion revision, refuse an
+ occupied review slot, treat candidate text and verification commands as inert data, and cannot
+ approve, reject, or reload anything. Its gate explicitly identifies retrospective evidence and
+ warns that no held-out A/B quality comparison ran. Exact retries are idempotent. MCP remains
+ unauthenticated, so this reversible proposal action is available only inside the same trusted
+ network boundary as the read tools.
- **Approval and rollback** are the only application paths that write under `skills/`. Both go
through `optimize/promote.py`, both snapshot what they displace, and both append an audit record
on a best-effort basis (a failed append is logged and does not undo the committed change).
@@ -70,7 +77,8 @@ change-control UI is password-gated by Compose, using the local demo login `admi
supports OIDC for shared deployments. Loopback binding remains the first protection layer:
- `docker-compose.yml` publishes every port on loopback only (`127.0.0.1:8000` MCP,
- `127.0.0.1:8080` UI, `127.0.0.1:3100` Langfuse).
+ `127.0.0.1:8080` UI, `127.0.0.1:3100` Langfuse). `INGOT_MCP_PORT` and `INGOT_UI_PORT` move the
+ host port when the box already serves one of them; the loopback binding is not theirs to change.
- Run outside Docker, the MCP server also binds `127.0.0.1` by default.
To expose MCP, use a private interface override as shown in [Production setup](../PRODUCTION_SETUP.md)
diff --git a/docs/tutorial.md b/docs/tutorial.md
index b1b491d..0bdfea6 100644
--- a/docs/tutorial.md
+++ b/docs/tutorial.md
@@ -154,7 +154,7 @@ At this point you know what is wrong and could fix the body by hand. SkillOpt in
other half of Ingot's value: it trains a bounded instruction revision from real failures and
attaches measured evidence without activating the result.
-Write an eval task set for the skill (`optimize/tasks/tailwind.yaml`) with train and holdout tasks
+Write an eval task set for the skill (`ingot/optimize/tasks/tailwind.yaml`) with train and holdout tasks
whose rubrics carry the v4 ground truth (the teacher can also auto-draft one on a skill's first CLI
run). Then run it headless, which is how it is meant to run:
@@ -266,8 +266,8 @@ That snapshot is the undo. It appears in the UI's **History** section, and resto
click, or one command:
```bash
-# --entrypoint python replaces the service's own `python -m optimize.ab` entrypoint
-docker compose run --rm --entrypoint python optimize -m optimize.promote rollback tailwind
+# --entrypoint python replaces the service's own `python -m ingot.optimize.ab` entrypoint
+docker compose run --rm --entrypoint python optimize -m ingot.optimize.promote rollback tailwind
```
Rollback snapshots the revision it displaces too, so the round trip is symmetric, and it writes its
@@ -285,7 +285,7 @@ displaced candidate is archived beside the slot (the run tells you where) rather
The body is fixed, but step 3's third request still misroutes: the routing key is the
`description`, so routing gets its own pass with its own metric, run against the `routing:` cases
-in `optimize/tasks/tailwind.yaml`: realistic positive phrasings plus `expected: null` negatives.
+in `ingot/optimize/tasks/tailwind.yaml`: realistic positive phrasings plus `expected: null` negatives.
The cases that matter are the real misses, so put your mined traffic in the suite (we added the
node_modules request verbatim, plus a "classes disappear in the production build" variant):
@@ -364,7 +364,7 @@ against the real router plus a description-collision scan, embedding-only, no LL
exits non-zero on problems, so it slots into cron or CI:
```bash
-docker compose run --rm --entrypoint "python -m optimize.routing_health" optimize
+docker compose run --rm --entrypoint "python -m ingot.optimize.routing_health" optimize
# [health] tailwind: top1 1.000 · recall@3 1.000 · no-route precision 0.333 (7 cases)
# [health] ✓ routing healthy: every suite passes and no descriptions collide.
```
diff --git a/evals/fixtures/skills/billing-runbook/SKILL.md b/evals/fixtures/skills/billing-runbook/SKILL.md
new file mode 100644
index 0000000..8d49e51
--- /dev/null
+++ b/evals/fixtures/skills/billing-runbook/SKILL.md
@@ -0,0 +1,6 @@
+---
+name: billing-runbook
+description: Operate a production service.
+---
+Investigate invoice charges, subscription renewals, payment failures, credits,
+refunds, and billing-account ownership.
diff --git a/evals/fixtures/skills/kubernetes-runbook/SKILL.md b/evals/fixtures/skills/kubernetes-runbook/SKILL.md
new file mode 100644
index 0000000..07d5ed2
--- /dev/null
+++ b/evals/fixtures/skills/kubernetes-runbook/SKILL.md
@@ -0,0 +1,6 @@
+---
+name: kubernetes-runbook
+description: Operate a production service.
+---
+Diagnose a Kubernetes pod stuck in CrashLoopBackOff. Inspect pod events,
+container logs, probes, resource limits, and recent deployment changes.
diff --git a/evals/routing.yaml b/evals/routing.yaml
index 2ce0b65..5ce17b4 100644
--- a/evals/routing.yaml
+++ b/evals/routing.yaml
@@ -56,6 +56,10 @@ cases:
expected: null
harness: claude
min_score: 0.99
+ - task: Diagnose a Kubernetes pod stuck in CrashLoopBackOff.
+ expected: kubernetes-runbook
+ harness: codex
+ parity: true
- task: Thanks, that answers my question.
expected: null
harness: codex
diff --git a/ingot/__init__.py b/ingot/__init__.py
new file mode 100644
index 0000000..6ab7110
--- /dev/null
+++ b/ingot/__init__.py
@@ -0,0 +1,6 @@
+"""Ingot's command surface.
+
+Deliberately empty of imports. `ingot.cli` must stay runnable with nothing installed beyond the
+skill loader's own dependency, so anything that reaches for the server, the optimizer, or a model
+belongs in the subcommand that needs it, imported inside the function."""
+__version__ = "0.2.0"
diff --git a/ingot/acquire.py b/ingot/acquire.py
new file mode 100644
index 0000000..a4e3b2c
--- /dev/null
+++ b/ingot/acquire.py
@@ -0,0 +1,121 @@
+"""Fetch remote package bytes without admitting, reviewing, or executing them."""
+from __future__ import annotations
+
+import os
+import re
+import subprocess
+from pathlib import Path
+
+from ingot.optimize.tree import MAX_FILES, MAX_TREE_BYTES, portable_path
+
+_REPOSITORY = re.compile(
+ r"^[A-Za-z0-9](?:[A-Za-z0-9-]{0,38})/[A-Za-z0-9](?:[A-Za-z0-9._-]{0,99})$")
+_COMMIT = re.compile(r"^[0-9a-f]{40,64}$")
+
+
+def _remote_url(repository: str) -> str:
+ return f"https://github.com/{repository}.git"
+
+
+def _git(*args: str) -> bytes:
+ environment = {**os.environ, "GIT_TERMINAL_PROMPT": "0"}
+ try:
+ result = subprocess.run(["git", *args], capture_output=True, env=environment,
+ timeout=120)
+ except (OSError, subprocess.TimeoutExpired) as error:
+ raise ValueError(f"Git acquisition failed: {error}") from error
+ if result.returncode:
+ detail = result.stderr.decode("utf-8", errors="replace").strip()
+ raise ValueError(f"Git acquisition failed: {detail or 'git exited non-zero'}")
+ return result.stdout
+
+
+def _resolved_commit(remote: str, ref: str) -> str:
+ output = _git("ls-remote", remote, ref)
+ rows = [line.split(b"\t", 1) for line in output.splitlines()]
+ matches = [sha.decode("ascii") for sha, name in rows
+ if name.decode("utf-8", errors="replace") == ref]
+ if len(matches) != 1 or not _COMMIT.fullmatch(matches[0]):
+ raise ValueError(f"Git acquisition failed: ref {ref!r} did not resolve to one commit")
+ return matches[0]
+
+
+def _bounded_tree(repository: Path, subdirectory: str) -> list[tuple[str, str, str, int]]:
+ """Return safe entries after enforcing bounds, before asking Git for blob contents."""
+ output = _git("-C", str(repository), "ls-tree", "-r", "-l", "-z", "HEAD", "--",
+ subdirectory)
+ entries, total = [], 0
+ prefix = f"{subdirectory}/"
+ for record in output.split(b"\0"):
+ if not record:
+ continue
+ try:
+ metadata, raw_path = record.split(b"\t", 1)
+ mode, kind, object_id, raw_size = metadata.split()
+ path = raw_path.decode("utf-8")
+ except (UnicodeDecodeError, ValueError) as error:
+ raise ValueError("Git acquisition failed: the selected tree has an invalid entry") \
+ from error
+ if not path.startswith(prefix):
+ raise ValueError(f"Git acquisition failed: {path!r} escapes the selected package")
+ relative = path[len(prefix):]
+ portable_path(relative, allow_skill_md=True)
+ if mode == b"120000":
+ raise ValueError(f"symlinks are not admissible: {relative}")
+ if kind != b"blob" or raw_size == b"-":
+ raise ValueError(f"not a regular file: {relative}")
+ size = int(raw_size)
+ entries.append((relative, mode.decode("ascii"), object_id.decode("ascii"), size))
+ total += size
+
+ if not entries:
+ raise ValueError(f"Git acquisition failed: {subdirectory!r} is not a package directory")
+ if len(entries) > MAX_FILES:
+ raise ValueError(f"a package may hold at most {MAX_FILES} files; this one holds "
+ f"{len(entries)}")
+ if total > MAX_TREE_BYTES:
+ raise ValueError(f"a package may hold at most {MAX_TREE_BYTES} bytes; this one holds {total}")
+ return entries
+
+
+def _materialize(repository: Path, package: Path,
+ entries: list[tuple[str, str, str, int]]) -> None:
+ """Write raw Git blobs, bypassing checkout hooks and attribute-selected filters."""
+ for relative, mode, object_id, expected_size in entries:
+ content = _git("-C", str(repository), "cat-file", "blob", object_id)
+ if len(content) != expected_size:
+ raise ValueError(f"Git acquisition failed: {relative} changed while it was fetched")
+ target = package / relative
+ target.parent.mkdir(parents=True, exist_ok=True)
+ target.write_bytes(content)
+ target.chmod(0o755 if mode == "100755" else 0o644)
+
+
+def github(repository: str, *, ref: str, subdirectory: str,
+ destination: Path) -> tuple[Path, dict]:
+ """Fetch one public GitHub repository subdirectory at an exact resolved commit."""
+ if not isinstance(repository, str) or not _REPOSITORY.fullmatch(repository) or \
+ repository.casefold().endswith(".git"):
+ raise ValueError(f"GitHub repository must be OWNER/REPO, found {repository!r}")
+ if ref != "HEAD":
+ raise ValueError("GitHub acquisition currently supports the remote HEAD ref")
+ selected = portable_path(subdirectory).as_posix()
+ remote = _remote_url(repository)
+ commit = _resolved_commit(remote, ref)
+
+ destination = Path(destination)
+ clone = destination / "repository"
+ destination.mkdir(parents=True, exist_ok=True)
+ _git("-c", "core.hooksPath=/dev/null", "-c", "core.symlinks=false", "clone",
+ "--depth", "1", "--filter=blob:none", "--no-checkout", "--no-tags",
+ "--single-branch", remote, str(clone))
+ checked_out = _git("-C", str(clone), "rev-parse", "HEAD").decode("ascii").strip()
+ if checked_out != commit:
+ raise ValueError(
+ f"Git acquisition failed: ref moved from {commit} to {checked_out} during acquisition")
+
+ entries = _bounded_tree(clone, selected)
+ package = destination / "package"
+ _materialize(clone, package, entries)
+ return package, {"repository": repository, "ref": ref, "commit": commit,
+ "subdirectory": selected}
diff --git a/ingot/admission.py b/ingot/admission.py
new file mode 100644
index 0000000..1330a07
--- /dev/null
+++ b/ingot/admission.py
@@ -0,0 +1,133 @@
+"""One complete local ingest path: a directory on disk becomes a quarantined proposal.
+
+The whole product claim lives in this file's one guarantee -- **`ingot add` never activates
+anything**. It runs the deterministic review, computes the exact revision the library would serve,
+records where the package came from, and takes the review slot. The served library is not touched.
+
+This is an adapter, not a second admission service. Path validation, component assembly, content
+hashing, evidence writing, pending-record creation, and the atomic slot claim all belong to
+`ingot.optimize.ingress` and are called, not reimplemented. What is new here is only the part that is
+genuinely new: turning a directory into the fields that service already takes, and binding a
+provenance manifest to the result.
+
+`optimize` is imported inside the function rather than at module scope. `ingot list` and
+`ingot review` promise to run in a bare virtualenv, and a module-level import here would put the
+optimizer on their import path."""
+from __future__ import annotations
+
+import time
+from pathlib import Path
+
+from . import records
+from .parse import ERROR, WARNING, parse_raw
+from .review import REVIEW_SCHEMA, review_package
+
+_SUPPORTED_SCHEMES = ("file", "github")
+
+
+class AdmissionRefused(Exception):
+ """The package cannot be represented as a candidate. Nothing was written."""
+
+
+def parse_locator(locator: str) -> tuple[str, Path | str]:
+ """A file path or GitHub repository. Unknown schemes are refused by name."""
+ if locator.startswith("file:"):
+ return "file", Path(locator[len("file:"):]).expanduser().resolve()
+ if locator.startswith("github:"):
+ return "github", locator[len("github:"):]
+
+ head, separator, _ = locator.partition(":")
+ if separator and head.isalpha() and len(head) > 1:
+ raise ValueError(
+ f"unsupported source scheme {head!r}; this version supports "
+ f"{', '.join(f'{s}:' for s in _SUPPORTED_SCHEMES)} and bare paths")
+ return "file", Path(locator).expanduser().resolve()
+
+
+def _codes(result: dict, level: str) -> list[str]:
+ return [finding["code"]
+ for section in result["sections"].values()
+ for finding in section["findings"]
+ if finding["level"] == level]
+
+
+def add_package(package: Path, *, actor: str, producer: str = "ingot-cli",
+ source_type: str = "file", locator: str | None = None,
+ provenance: dict | None = None) -> dict:
+ """Review, quarantine, and report. Leaves the served library byte-identical."""
+ from ingot.mcp_server import registry
+ from ingot.mcp_server.registry import read_components, skill_revision
+ from ingot.optimize import ingress, tree
+
+ package = Path(package).expanduser().resolve()
+ if not package.is_dir():
+ raise AdmissionRefused(f"{package} is not a directory")
+ source_locator = locator or str(package)
+
+ # `registry.library_dir()` resolved per call, never bound at import: a frozen copy would
+ # check collisions
+ # against a different library than the one this process actually serves.
+ report = review_package(package, library_root=registry.library_dir())
+ errors, warnings = _codes(report, ERROR), _codes(report, WARNING)
+ if not report["valid"]:
+ raise AdmissionRefused(
+ f"{package.name} is not admissible: {', '.join(errors)}")
+
+ raw = parse_raw((package / "SKILL.md").read_text(encoding="utf-8", errors="replace"))
+ frontmatter = raw.frontmatter or {}
+ skill = str(frontmatter.get("name") or package.name)
+
+ # No `file:` components. The package's files travel as a staged tree of exact bytes; carrying
+ # decoded copies of the text ones beside it would be a second description of the same files,
+ # and the two would eventually disagree about which is authoritative.
+ read = read_components(package)
+ components, metadata = ingress.build_components(
+ skill, read["description"], read["body"], {}, frontmatter)
+ try:
+ candidate_tree = tree.build(package)
+ # Staged before the revision is computed, because the revision *is* the result of
+ # materializing the staged tree -- deriving it any other way would be a second description
+ # of the same bytes, and the two would eventually disagree. Staging is named by the tree
+ # digest and so is idempotent: a submission refused further down leaves nothing behind but
+ # a directory the next identical one reuses.
+ tree.stage(package, candidate_tree)
+ except ValueError as error:
+ raise AdmissionRefused(f"{package.name} is not admissible: {error}") from error
+
+ manifest = records.candidate_manifest(
+ kind="creation",
+ skill=skill,
+ source_type=source_type,
+ locator=source_locator,
+ # What the source resolved to, and what the library will serve. Equal for a package that is
+ # already canonical, and deliberately separate fields because they are not always equal --
+ # admission collapses whitespace in a description, and then the two diverge.
+ resolved_revision=skill_revision(package),
+ candidate_revision=tree.revision(skill, candidate_tree, components),
+ review={"schema_version": REVIEW_SCHEMA,
+ "valid": report["valid"],
+ "errors": errors,
+ "warnings": warnings,
+ "report_digest": records.digest(report)},
+ created_at=int(time.time()),
+ provenance=({**(provenance or {}), "content_digest": candidate_tree["digest"]}
+ if provenance is not None else None))
+
+ problems = records.validate_candidate(manifest)
+ if problems:
+ raise AdmissionRefused("the candidate manifest is malformed: " + "; ".join(problems))
+
+ outcome = ingress.submit_package_ingest(
+ skill=skill,
+ components=components,
+ candidate_tree=candidate_tree,
+ metadata=metadata,
+ revision=manifest["candidate_revision"],
+ source=(source_locator if source_locator.startswith(f"{source_type}:")
+ else f"{source_type}:{source_locator}"),
+ candidate=manifest,
+ identity=records.candidate_identity(manifest),
+ review_summary=warnings,
+ producer=producer,
+ caller=actor)
+ return {**outcome, "candidate": manifest}
diff --git a/ingot/cli.py b/ingot/cli.py
new file mode 100644
index 0000000..4338edd
--- /dev/null
+++ b/ingot/cli.py
@@ -0,0 +1,360 @@
+"""The `ingot` command line.
+
+Nothing here writes a served byte. `add` quarantines, `approve` and `rollback` queue a publication
+receipt, `reject` discards a quarantined change: the publisher is the only writer of the served
+library, and every mutating verb calls the same service the console calls rather than a second
+approval path of its own.
+
+Every import stays inside the function that needs it. Importing this module must not pull in
+FastAPI, ONNX, LangGraph, Langfuse, or the optimizer, because the first thing a developer runs has
+to work in a bare virtualenv with no services, no model, and no key."""
+from __future__ import annotations
+
+import argparse
+import getpass
+import json
+import os
+import sys
+from pathlib import Path
+
+LIST_SCHEMA = "ingot/list/v1"
+
+
+def _default_actor() -> str:
+ """Who a proposal is attributed to. Best effort, and never blank: an unattributed proposal in
+ the review queue is one nobody can ask about."""
+ return os.environ.get("INGOT_ACTOR") or getpass.getuser()
+
+
+def list_library(root: Path | None = None) -> dict:
+ """The skills a server would serve, with the roots they came from.
+
+ `root` is passed to the loader rather than replacing its configuration: `configured_roots`
+ always puts the local authoring root first, even ahead of an explicit root, so the answer can
+ legitimately include skills from somewhere the caller did not name. Reporting `roots` is what
+ keeps that honest -- a caller who sees an unexpected skill can see which library it came from."""
+ from ingot.mcp_server.registry import configured_roots, load_skills
+
+ explicit = [root] if root is not None else None
+ return {
+ "schema_version": LIST_SCHEMA,
+ "roots": [str(path) for path in configured_roots(explicit)],
+ "skills": [{"name": skill.name,
+ "description": skill.description,
+ "revision": skill.revision,
+ "root": skill.root}
+ for skill in load_skills(roots=explicit)],
+ }
+
+
+def _render(result: dict) -> str:
+ roots = ", ".join(result["roots"])
+ skills = result["skills"]
+ if not skills:
+ return f"No skills in {roots}"
+ width = max(len(skill["name"]) for skill in skills)
+ lines = [f"{len(skills)} skill{'s' if len(skills) != 1 else ''} in {roots}", ""]
+ lines += [f" {skill['name']:<{width}} {skill['revision'][:8]} {skill['description']}"
+ for skill in skills]
+ return "\n".join(lines)
+
+
+def _list(args: argparse.Namespace) -> int:
+ result = list_library(args.root)
+ print(json.dumps(result, indent=2) if args.json else _render(result))
+ return 0
+
+
+def _review(args: argparse.Namespace) -> int:
+ """Exit non-zero only for deterministic validity errors. Warnings are advice, and a command
+ that fails on advice teaches people to stop reading it."""
+ from . import review as review_module
+
+ package = args.path
+ if not package.is_dir():
+ print(f"ingot review: {package} is not a directory", file=sys.stderr)
+ return 2
+
+ result = review_module.review_package(package, library_root=args.root)
+ print(json.dumps(result, indent=2) if args.json else review_module.render(result))
+ return 0 if result["valid"] else 1
+
+
+def _add(args: argparse.Namespace) -> int:
+ """Quarantine a package. Never activates anything, so the only failures are refusals."""
+ from . import admission
+
+ try:
+ kind, resolved = admission.parse_locator(args.locator)
+ if kind == "file":
+ if args.skill:
+ raise admission.AdmissionRefused("--skill is only valid for github: sources")
+ result = admission.add_package(resolved, actor=args.actor)
+ else:
+ if not args.skill:
+ raise admission.AdmissionRefused("--skill is required for github: sources")
+ import tempfile
+ from pathlib import Path
+ from . import acquire
+
+ with tempfile.TemporaryDirectory() as temporary:
+ package, provenance = acquire.github(
+ resolved, ref="HEAD", subdirectory=args.skill,
+ destination=Path(temporary))
+ result = admission.add_package(
+ package, actor=args.actor, source_type="github", locator=args.locator,
+ provenance=provenance)
+ except (admission.AdmissionRefused, ValueError) as refusal:
+ print(f"ingot add: {refusal}", file=sys.stderr)
+ return 1
+
+ if args.json:
+ print(json.dumps(result, indent=2))
+ return 0
+
+ verb = "already quarantined" if result["status"] == "duplicate" else "quarantined"
+ review_hint = (f" ingot review {result['candidate']['source']['locator']}\n"
+ if result["candidate"]["source"]["type"] == "file" else "")
+ print(f"{verb} '{result['skill']}' as proposal {result['proposal_id']}\n"
+ f" revision {result['candidate']['candidate_revision'][:16]}\n"
+ f" source {result['candidate']['source']['locator']}\n"
+ f"\nThe served library is unchanged. Review and approve it in the console, or:\n"
+ f"{review_hint}"
+ f" ingot approve {result['skill']}")
+ return 0
+
+
+def _vault_init(args: argparse.Namespace) -> int:
+ from . import vault
+
+ try:
+ result = vault.init_vault(args.path)
+ except ValueError as refusal:
+ print(f"ingot vault init: {refusal}", file=sys.stderr)
+ return 1
+ if args.json:
+ print(json.dumps(result, indent=2))
+ return 0
+ print(f"{result['status']} vault at {result['path']}\n"
+ f" branch {result['branch']}\n"
+ f" head {result['head'][:12]}")
+ if result["added"]:
+ print(f" added {', '.join(result['added'])}")
+ return 0
+
+
+def _status(args: argparse.Namespace) -> int:
+ """Exit non-zero when the served library is writable, so a managed deployment can assert it."""
+ from . import status as status_module
+
+ result = status_module.library_status(args.root)
+ print(json.dumps(result, indent=2) if args.json else status_module.render(result))
+ return 0 if result["mode"] == status_module.MANAGED else 1
+
+
+def _when(seconds: object) -> str:
+ """Unix seconds as a local timestamp, or blank. A record written before the field existed must
+ print as an empty column rather than a traceback."""
+ if not isinstance(seconds, (int, float)) or isinstance(seconds, bool) or seconds <= 0:
+ return ""
+ import datetime
+ return datetime.datetime.fromtimestamp(seconds).strftime("%Y-%m-%d %H:%M")
+
+
+def _pending(args: argparse.Namespace) -> int:
+ from . import decisions
+
+ result = decisions.pending_view()
+ if args.json:
+ print(json.dumps(result, indent=2))
+ return 0
+ if result["unreadable"]:
+ print(f"WARNING {len(result['unreadable'])} quarantined change(s) cannot be read and are "
+ f"not listed: {', '.join(result['unreadable'])}", file=sys.stderr)
+ if not result["pending"] and not result["publishing"]:
+ print("Nothing waiting.")
+ return 0
+ for entry in result["pending"]:
+ verdict = "ready" if entry["promotable"] else "BLOCKED"
+ print(f" {verdict:<8} {entry['skill']:<20} {entry['kind']:<12} "
+ f"{entry['revision'][:12]}")
+ for reason in entry["blocked"]:
+ print(f" {reason}")
+ if entry["publication"]:
+ print(f" publication {entry['publication']['status']}")
+ for entry in result["publishing"]:
+ print(f" {entry['status']:<8} {entry['skill']:<20} {entry['action']:<12} "
+ f"{entry['revision'][:12]}")
+ if entry["error"]:
+ print(f" {entry['error']}")
+ return 0
+
+
+def _decide(args: argparse.Namespace) -> int:
+ """Approve, reject, or roll back. Every one queues or discards; none writes a served byte."""
+ from . import decisions
+
+ try:
+ if args.command == "approve":
+ result = decisions.approve(args.skill, actor=args.actor)
+ elif args.command == "reject":
+ result = decisions.reject(args.skill, actor=args.actor, reason=args.reason)
+ else:
+ result = decisions.rollback(args.skill, args.revision, actor=args.actor)
+ except ValueError as refusal:
+ print(f"ingot {args.command}: {refusal}", file=sys.stderr)
+ return 1
+
+ if args.json:
+ print(json.dumps(result, indent=2))
+ return 0
+ print(result["result"])
+ receipt = result["publication"]
+ if receipt:
+ print(f" publication {receipt['id']} {receipt['status']}")
+ if receipt["error"]:
+ print(f" {receipt['error']}")
+ if receipt["status"] != "published":
+ print(" The served library is unchanged until the publisher activates it.")
+ return 0
+
+
+def _history(args: argparse.Namespace) -> int:
+ from . import decisions
+
+ try:
+ result = decisions.history_view(args.skill)
+ except ValueError as refusal:
+ print(f"ingot history: {refusal}", file=sys.stderr)
+ return 1
+ if args.json:
+ print(json.dumps(result, indent=2))
+ return 0
+ print(f"{result['skill']}\n")
+ print(" snapshots (rollback targets)")
+ for revision in result["revisions"] or []:
+ print(f" {revision['revision'][:16]} {_when(revision.get('created'))}")
+ if not result["revisions"]:
+ print(" none")
+ print("\n publications")
+ for receipt in result["publications"] or []:
+ print(f" {receipt['status']:<10} {receipt['action']:<9} "
+ f"{(receipt['revision'] or '')[:16]} {receipt['id']}")
+ if not result["publications"]:
+ print(" none")
+ print("\n decisions")
+ for record in result["audit"] or []:
+ print(f" {record.get('action', ''):<10} {record.get('actor', ''):<16} "
+ f"{(record.get('revision') or '')[:16]} {_when(record.get('ts'))}"
+ + (f" {record['reason']}" if record.get("reason") else ""))
+ if not result["audit"]:
+ print(" none")
+ return 0
+
+
+def build_parser() -> argparse.ArgumentParser:
+ parser = argparse.ArgumentParser(
+ prog="ingot",
+ description="Release control for agent skills.")
+ sub = parser.add_subparsers(dest="command", required=True)
+
+ listing = sub.add_parser("list", help="list the skills in a library")
+ listing.add_argument("--root", type=Path, default=None,
+ help="library root to read instead of the configured one")
+ listing.add_argument("--json", action="store_true", help="emit the versioned JSON payload")
+ listing.set_defaults(handler=_list)
+
+ reviewing = sub.add_parser(
+ "review", help="report what is wrong with a skill package, offline",
+ description="Deterministic, model-free, network-free, read-only review of one skill "
+ "package. Reports six sections and no composite score; questions it cannot "
+ "answer offline are reported UNMEASURED with the command that answers them.")
+ reviewing.add_argument("path", type=Path, help="the skill package directory to review")
+ reviewing.add_argument("--root", type=Path, default=None,
+ help="library root to check for collisions against")
+ reviewing.add_argument("--json", action="store_true", help="emit the versioned JSON payload")
+ reviewing.set_defaults(handler=_review)
+
+ adding = sub.add_parser(
+ "add", help="quarantine a skill package for review",
+ description="Review a package and place it in quarantine. The served library is left "
+ "byte-identical; a human must approve the proposal before anything is served.")
+ adding.add_argument("locator", help="file:./path/to/skill, a bare path, or github:OWNER/REPO")
+ adding.add_argument("--skill", help="package subdirectory inside a github: repository")
+ adding.add_argument("--actor", default=_default_actor(),
+ help="who is submitting this (defaults to the current user)")
+ adding.add_argument("--json", action="store_true", help="emit the versioned JSON payload")
+ adding.set_defaults(handler=_add)
+
+ queue = sub.add_parser(
+ "pending", help="list quarantined changes waiting on a decision",
+ description="Everything waiting on a person, plus anything already travelling to the "
+ "vault. Read-only.")
+ queue.add_argument("--json", action="store_true", help="emit the versioned JSON payload")
+ queue.set_defaults(handler=_pending)
+
+ approving = sub.add_parser(
+ "approve", help="approve a quarantined change for publication",
+ description="Queues a publication receipt. It does not activate anything: the publisher "
+ "commits the approved revision and only then does the library serve it.")
+ approving.add_argument("skill")
+
+ rejecting = sub.add_parser(
+ "reject", help="discard a quarantined change",
+ description="Deletes the pending record and records the decision in the approval trail.")
+ rejecting.add_argument("skill")
+ rejecting.add_argument("--reason", default="", help="why, for the approval trail")
+
+ reverting = sub.add_parser(
+ "rollback", help="queue a stored snapshot for publication",
+ description="Takes the same lane as an approval: the snapshot is published through the "
+ "publisher rather than copied over the served library.")
+ reverting.add_argument("skill")
+ reverting.add_argument("revision", help="a revision from `ingot history SKILL`")
+
+ for decision in (approving, rejecting, reverting):
+ decision.add_argument("--actor", default=_default_actor(),
+ help="who is deciding (defaults to the current user)")
+ decision.add_argument("--json", action="store_true",
+ help="emit the versioned JSON payload")
+ decision.set_defaults(handler=_decide)
+
+ past = sub.add_parser(
+ "history", help="what a skill has been, and what was decided about it",
+ description="Rollback targets, publication receipts, and the approval trail for one "
+ "skill. Read-only.")
+ past.add_argument("skill")
+ past.add_argument("--json", action="store_true", help="emit the versioned JSON payload")
+ past.set_defaults(handler=_history)
+
+ vault = sub.add_parser(
+ "vault", help="the Git vault the publisher owns",
+ description="The local publication backend publishes into a Git repository on this "
+ "machine. This creates one, and is idempotent against an existing vault.")
+ vault_sub = vault.add_subparsers(dest="vault_command", required=True)
+ initialize = vault_sub.add_parser("init", help="create or complete a local vault")
+ initialize.add_argument("path", type=Path, nargs="?", default=Path("vault"),
+ help="where the vault lives (default: ./vault)")
+ initialize.add_argument("--json", action="store_true", help="emit the versioned JSON payload")
+ initialize.set_defaults(handler=_vault_init)
+
+ reporting = sub.add_parser(
+ "status", help="report whether this deployment's guarantees hold",
+ description="MANAGED when the served library is read-only to this process, so only the "
+ "publisher can change what is served. UNMANAGED otherwise, and the exit code "
+ "says so: 0 for MANAGED, 1 for UNMANAGED.")
+ reporting.add_argument("--root", type=Path, default=None,
+ help="library root to inspect instead of the configured one")
+ reporting.add_argument("--json", action="store_true", help="emit the versioned JSON payload")
+ reporting.set_defaults(handler=_status)
+
+ return parser
+
+
+def main(argv: list[str] | None = None) -> int:
+ args = build_parser().parse_args(argv)
+ return args.handler(args)
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/ingot/decisions.py b/ingot/decisions.py
new file mode 100644
index 0000000..b3ddda9
--- /dev/null
+++ b/ingot/decisions.py
@@ -0,0 +1,107 @@
+"""Read models and decision verbs for the command line.
+
+Every mutating verb here calls the same service the console calls. There is no second approval
+path, no direct activation helper, and nothing that writes a served byte: approval and rollback
+queue a publication receipt, and the publisher is what acts on it."""
+from __future__ import annotations
+
+PENDING_SCHEMA = "ingot/pending/v1"
+HISTORY_SCHEMA = "ingot/history/v1"
+DECISION_SCHEMA = "ingot/decision/v1"
+
+APPROVED = "approved" # a receipt exists; the publisher has not picked it up yet
+PUBLISHING = "publishing" # in flight, including waiting on a forge merge
+PUBLISHED = "published" # the served library carries it
+FAILED = "failed" # the last attempt errored; the receipt is retryable
+
+
+def release_status(record: dict | None) -> dict | None:
+ """One receipt, in the four words a person making a decision needs.
+
+ `failed` reads off `last_error` rather than the state, because a failed attempt leaves the
+ receipt in whatever state it was working through. A receipt that reports only its state hides
+ the one thing an operator has to act on."""
+ if not record:
+ return None
+ state = record.get("state")
+ if state == "active":
+ status = PUBLISHED
+ elif record.get("last_error"):
+ status = FAILED
+ elif state in {"publishing", "awaiting_merge"}:
+ status = PUBLISHING
+ else:
+ status = APPROVED
+ return {"id": record.get("id"), "status": status, "state": state,
+ "action": record.get("action"), "attempts": record.get("attempts", 0),
+ "error": record.get("last_error") or "", "revision": record.get("candidate_revision")}
+
+
+def pending_view() -> dict:
+ """Everything waiting on a person, with whatever is already travelling to the vault."""
+ from ingot.optimize.promote import challenger_revision, list_pending, unreadable_pending
+ from ingot.optimize.publication import publication_for_skill, recent_publications
+
+ entries = []
+ for record in sorted(list_pending(), key=lambda item: item.get("skill", "")):
+ skill = record.get("skill", "")
+ gate = record.get("gate") or {}
+ entries.append({
+ "skill": skill,
+ "kind": record.get("kind", "quality"),
+ "promotable": gate.get("promotable") is True,
+ "blocked": list(gate.get("blocked") or []),
+ "revision": challenger_revision(record),
+ "publication": release_status(publication_for_skill(skill)),
+ })
+ # A receipt outlives the pending record it came from, so an approved change is invisible to the
+ # queue above for exactly the window in which it is travelling and someone might be waiting.
+ queued = {entry["skill"] for entry in entries}
+ travelling = [release_status(record) | {"skill": record.get("skill")}
+ for record in recent_publications()
+ if record.get("skill") not in queued and record.get("state") != "active"]
+ return {"schema_version": PENDING_SCHEMA, "pending": entries, "publishing": travelling,
+ "unreadable": unreadable_pending()}
+
+
+def history_view(skill: str, limit: int = 50) -> dict:
+ """What this skill has been, what it is travelling toward, and what was decided about it."""
+ from ingot.optimize.promote import check_slug, list_revisions, read_audit
+ from ingot.optimize.publication import recent_publications
+
+ skill = check_slug(skill)
+ audit = [record for record in read_audit(limit=10_000)["records"]
+ if record.get("skill") == skill][:limit]
+ return {
+ "schema_version": HISTORY_SCHEMA,
+ "skill": skill,
+ "revisions": list_revisions(skill),
+ "publications": [release_status(record) for record in recent_publications(10_000)
+ if record.get("skill") == skill][:limit],
+ "audit": audit,
+ }
+
+
+def _decision(skill: str, message: str) -> dict:
+ from ingot.optimize.publication import publication_for_skill
+
+ return {"schema_version": DECISION_SCHEMA, "skill": skill, "result": message,
+ "publication": release_status(publication_for_skill(skill))}
+
+
+def approve(skill: str, actor: str) -> dict:
+ from ingot.optimize.promote import approve_pending
+
+ return _decision(skill, approve_pending(skill, actor=actor))
+
+
+def reject(skill: str, actor: str, reason: str = "") -> dict:
+ from ingot.optimize.promote import reject_pending
+
+ return _decision(skill, reject_pending(skill, actor=actor, reason=reason))
+
+
+def rollback(skill: str, revision: str, actor: str) -> dict:
+ from ingot.optimize.promote import rollback as rollback_pending
+
+ return _decision(skill, rollback_pending(skill, revision, actor=actor))
diff --git a/ingot/delivery.py b/ingot/delivery.py
new file mode 100644
index 0000000..75dc708
--- /dev/null
+++ b/ingot/delivery.py
@@ -0,0 +1,219 @@
+"""Where an approved revision is installed once the vault already carries it.
+
+The vault is the publication authority and the managed-MCP library at the same time: the publisher
+commits into it, fast-forwards it, and every other service mounts it read-only. That covers agents
+that load skills through Ingot's MCP server. It does not cover an agent that reads a native skill
+directory on disk, which is most of them.
+
+A delivery target closes that gap without opening a second way to approve anything. Targets are
+configured on the publisher, never named in a receipt's identity, so what is approved does not
+depend on where a particular deployment happens to install it. Delivery runs *after* the vault
+serves the approved revision and *before* the receipt is marked active, which is what makes a
+failed delivery a retryable release rather than a finished one.
+
+Two kinds:
+
+- `managed-mcp` is the vault itself. Delivering to it is a deliberate no-op -- the publication
+ commit and the fast-forward already put the bytes there, and a second writer in the vault is the
+ one thing this control plane exists to prevent. It appears in the target list so it has a name,
+ a status, and a per-target line on the receipt like any other destination.
+- `filesystem` copies the skill directory out of the vault into a configured root.
+
+The source of every delivery is the vault checkout, already verified to be at the approved
+revision. Nothing here re-derives the bytes from components or from a staged tree: a second
+materialization is a second thing that can disagree with the first.
+"""
+from __future__ import annotations
+
+import os
+import shutil
+import uuid
+from dataclasses import dataclass
+from pathlib import Path
+
+from ingot.mcp_server.registry import SLUG_RE, skill_revision
+from ingot.optimize import promote
+
+DELIVERY_SCHEMA = "ingot/delivery/v1"
+TARGETS = "INGOT_DELIVERY_TARGETS"
+
+MANAGED_MCP = "managed-mcp"
+FILESYSTEM = "filesystem"
+KINDS = (MANAGED_MCP, FILESYSTEM)
+
+VAULT_TARGET = "vault"
+
+
+@dataclass(frozen=True)
+class Target:
+ name: str
+ kind: str
+ root: Path
+
+ def path(self, skill: str) -> Path:
+ return self.root / skill
+
+
+def _parse_one(entry: str, *, vault: Path) -> Target:
+ name, separator, remainder = entry.partition("=")
+ kind, kind_separator, raw_path = remainder.partition(":")
+ if not separator or not kind_separator:
+ raise ValueError(f"invalid delivery target {entry!r}: expected name=kind:path")
+ name, kind, raw_path = name.strip(), kind.strip(), raw_path.strip()
+ if not SLUG_RE.fullmatch(name):
+ raise ValueError(f"invalid delivery target name {name!r}")
+ if kind not in KINDS:
+ raise ValueError(f"unknown delivery kind {kind!r}; expected one of {', '.join(KINDS)}")
+ path = Path(raw_path).expanduser()
+ if not path.is_absolute():
+ # A relative root resolves against whatever directory the publisher happened to start in,
+ # which for a systemd unit is not a directory anyone chose.
+ raise ValueError(f"delivery target {name!r} must be an absolute path, not {raw_path!r}")
+ return Target(name, kind, path)
+
+
+def _contained(path: Path, root: Path) -> bool:
+ return path == root or root in path.parents
+
+
+def parse_targets(spec: str, *, vault: Path) -> tuple[Target, ...]:
+ """Read the configured target list, refusing anything that cannot work.
+
+ Every check here is one the publisher must make before it starts. A delivery target that fails
+ on the first approval strands a change that has already been approved, which is the stalled-lane
+ failure the backend configuration validation exists to prevent."""
+ vault = vault.expanduser()
+ targets: list[Target] = []
+ seen: dict[str, Target] = {}
+ roots: dict[Path, str] = {}
+ for entry in (piece.strip() for piece in spec.split(",")):
+ if not entry:
+ continue
+ target = _parse_one(entry, vault=vault)
+ if target.name in seen:
+ raise ValueError(f"duplicate delivery target {target.name!r}")
+ if target.kind == MANAGED_MCP and target.root != vault:
+ raise ValueError(f"the managed-mcp target must be the vault ({vault}), "
+ f"not {target.root}")
+ if target.kind == FILESYSTEM and _contained(target.root, vault):
+ raise ValueError(f"delivery target {target.name!r} is inside the vault; it would write "
+ f"the checkout the publisher just committed")
+ if target.root in roots:
+ raise ValueError(f"delivery targets {roots[target.root]!r} and {target.name!r} name the "
+ f"same directory")
+ seen[target.name] = target
+ roots[target.root] = target.name
+ targets.append(target)
+ # The vault is not optional. It is what the MCP server serves and what publication authority is
+ # measured against, so a configuration that omits it gets it anyway rather than silently
+ # switching off managed delivery.
+ if not any(target.kind == MANAGED_MCP for target in targets):
+ if VAULT_TARGET in seen:
+ raise ValueError(f"delivery target {VAULT_TARGET!r} is reserved for the managed-mcp "
+ f"vault target")
+ targets.insert(0, Target(VAULT_TARGET, MANAGED_MCP, vault))
+ return tuple(targets)
+
+
+def load_targets(env: dict | None = None, *, vault: Path) -> tuple[Target, ...]:
+ env = os.environ if env is None else env
+ return parse_targets(env.get(TARGETS) or "", vault=vault)
+
+
+def observed(target: Target, skill: str) -> str:
+ """The revision the target actually holds, asked of the filesystem.
+
+ Absence is a revision: a skill a target does not carry is a real, checkable state, and the
+ publisher delivers it deliberately when a rollback restores one."""
+ path = target.path(skill)
+ if not path.is_dir():
+ return promote.ABSENT_REVISION
+ return skill_revision(path)
+
+
+def _snapshot_displaced(target: Target, skill: str) -> None:
+ """Preserve whatever the target held, under the revision of the bytes actually there.
+
+ Keyed by observation rather than by the receipt's champion: a target someone edited by hand
+ holds bytes no release describes, and those are the ones worth keeping."""
+ path = target.path(skill)
+ if not path.is_dir():
+ return
+ promote._snapshot(path, skill, skill_revision(path))
+
+
+def _refuse_symlinks(source: Path) -> None:
+ """A delivered tree carries no symlinks, for the same reason an admitted one carries none.
+
+ `copytree` without `symlinks=True` copies what a link points at, so a link to somewhere outside
+ the skill would land in a native agent's skill root as an ordinary file holding those bytes.
+ Copying the link instead is no better: it puts a path in an agent's library leading out of it.
+ Admission already refuses symlinks, so reaching this means the vault acquired one some other
+ way, and delivering it is not this code's decision to make."""
+ stack = [source]
+ while stack:
+ for entry in stack.pop().iterdir():
+ if entry.is_symlink():
+ raise ValueError(
+ f"symlink-unsupported: {entry.relative_to(source)} is a symbolic link")
+ if entry.is_dir():
+ stack.append(entry)
+
+
+def _swap(staged: Path, destination: Path) -> None:
+ """Replace one directory with another using same-filesystem renames only.
+
+ POSIX has no atomic directory swap, so this is two renames with the displaced directory kept
+ until the second one succeeds. The window between them is the only moment the destination is
+ absent, and a failure inside it puts the original back rather than leaving a half-installed
+ skill an agent could load."""
+ if not destination.exists():
+ os.replace(staged, destination)
+ return
+ displaced = destination.with_name(f".{destination.name}.{uuid.uuid4().hex}.old")
+ os.replace(destination, displaced)
+ try:
+ os.replace(staged, destination)
+ except BaseException:
+ os.replace(displaced, destination)
+ raise
+ shutil.rmtree(displaced, ignore_errors=True)
+
+
+def _remove(destination: Path) -> None:
+ if not destination.exists():
+ return
+ displaced = destination.with_name(f".{destination.name}.{uuid.uuid4().hex}.old")
+ os.replace(destination, displaced)
+ shutil.rmtree(displaced, ignore_errors=True)
+
+
+def install(target: Target, skill: str, source: Path | None, revision: str) -> bool:
+ """Install one approved revision at one target. True when the target was written.
+
+ `source` is the vault's copy of the skill, already verified to be the approved revision, or
+ None to deliver an absence. `revision` is what the receipt names, and it is the staged copy --
+ the bytes that will actually be installed -- that is checked against it, so a source altered
+ between the vault check and this call cannot install itself."""
+ promote.check_slug(skill)
+ if target.kind == MANAGED_MCP:
+ return False
+ destination = target.path(skill)
+ if observed(target, skill) == revision:
+ return False
+ _snapshot_displaced(target, skill)
+ if source is None or revision == promote.ABSENT_REVISION:
+ _remove(destination)
+ return True
+ _refuse_symlinks(source)
+ target.root.mkdir(parents=True, exist_ok=True)
+ staged = target.root / f".{skill}.{uuid.uuid4().hex}.tmp"
+ try:
+ shutil.copytree(source, staged, symlinks=False)
+ if skill_revision(staged) != revision:
+ raise ValueError(f"delivered '{skill}' does not match the approved revision")
+ _swap(staged, destination)
+ except BaseException:
+ shutil.rmtree(staged, ignore_errors=True)
+ raise
+ return True
diff --git a/mcp_server/__init__.py b/ingot/mcp_server/__init__.py
similarity index 100%
rename from mcp_server/__init__.py
rename to ingot/mcp_server/__init__.py
diff --git a/mcp_server/embedding.py b/ingot/mcp_server/embedding.py
similarity index 66%
rename from mcp_server/embedding.py
rename to ingot/mcp_server/embedding.py
index c20565b..faeebeb 100644
--- a/mcp_server/embedding.py
+++ b/ingot/mcp_server/embedding.py
@@ -1,4 +1,4 @@
-"""Embedding backends for the router, both CPU-only ONNX (no GPU, no torch).
+"""Embedding backends for the router: local CPU ONNX or a remote GPU embedding server.
Default: the Qwen3-Embedding-0.6B q4 export, chosen on a 297-query drafted routing eval. It is an
asymmetric retrieval model: queries carry the official instruction prefix, documents (skill
@@ -19,12 +19,16 @@
COLLISION_SCORE together with the model."""
from __future__ import annotations
+import json
import os
+from urllib.error import HTTPError, URLError
+from urllib.request import Request, urlopen
import numpy as np
EMBED_MODEL = os.environ.get("EMBED_MODEL", "onnx-community/Qwen3-Embedding-0.6B-ONNX")
EMBED_ONNX_FILE = os.environ.get("EMBED_ONNX_FILE", "onnx/model_q4.onnx")
+EMBED_BACKEND = os.environ.get("EMBED_BACKEND", "local")
# The official Qwen3-Embedding retrieval instruction: applied to queries only, never documents.
QUERY_PREFIX = ("Instruct: Given a web search query, retrieve relevant passages that answer "
"the query\nQuery: ")
@@ -119,5 +123,69 @@ def embed(self, texts):
embed_query = embed
+class RemoteQwenEmbedding:
+ """OpenAI-compatible Qwen embedding server.
+
+ The server stays outside the MCP process so the CPU image remains small and portable. GPU
+ selection is explicit: requesting this backend without a reachable server fails instead of
+ silently changing model quality and invalidating calibrated routing thresholds.
+ """
+
+ def __init__(self, model: str, base_url: str, timeout: float = 60):
+ self._model = model
+ self._url = base_url.rstrip("/") + "/embeddings"
+ self._timeout = timeout
+ self.identity = f"remote:{model}@{base_url.rstrip('/')}"
+
+ def _run(self, texts: list[str]) -> list[np.ndarray]:
+ if not texts:
+ return []
+ body = json.dumps({
+ "model": self._model,
+ "input": texts,
+ "encoding_format": "float",
+ }).encode()
+ request = Request(
+ self._url,
+ data=body,
+ headers={"Content-Type": "application/json", "Authorization": "Bearer no-key"},
+ method="POST",
+ )
+ try:
+ with urlopen(request, timeout=self._timeout) as response:
+ payload = json.loads(response.read())
+ except (HTTPError, URLError, TimeoutError, json.JSONDecodeError) as exc:
+ raise RuntimeError(f"embedding server unavailable at {self._url}: {exc}") from exc
+ try:
+ ordered = sorted(payload["data"], key=lambda item: item["index"])
+ vectors = [np.asarray(item["embedding"], dtype=np.float32) for item in ordered]
+ except (KeyError, TypeError, ValueError) as exc:
+ raise RuntimeError("embedding server returned an invalid response") from exc
+ if len(vectors) != len(texts):
+ raise RuntimeError("embedding server returned the wrong vector count")
+ return vectors
+
+ def embed(self, texts) -> list[np.ndarray]:
+ return self._run(list(texts))
+
+ def embed_query(self, texts) -> list[np.ndarray]:
+ return self._run([QUERY_PREFIX + text for text in texts])
+
+
def build_embedding(model: str = EMBED_MODEL):
+ backend = os.environ.get("EMBED_BACKEND", EMBED_BACKEND).lower()
+ if backend == "remote":
+ base_url = os.environ.get("EMBED_BASE_URL", "").strip()
+ if not base_url:
+ raise ValueError("EMBED_BASE_URL is required when EMBED_BACKEND=remote")
+ remote_model = os.environ.get("EMBED_REMOTE_MODEL", "").strip()
+ if not remote_model:
+ raise ValueError("EMBED_REMOTE_MODEL is required when EMBED_BACKEND=remote")
+ try:
+ timeout = float(os.environ.get("EMBED_TIMEOUT_SECONDS", "60"))
+ except ValueError as exc:
+ raise ValueError("EMBED_TIMEOUT_SECONDS must be a number") from exc
+ return RemoteQwenEmbedding(remote_model, base_url, timeout)
+ if backend != "local":
+ raise ValueError("EMBED_BACKEND must be local or remote")
return QwenOnnxEmbedding(model) if is_qwen_onnx(model) else FastembedEmbedding(model)
diff --git a/ingot/mcp_server/provenance.py b/ingot/mcp_server/provenance.py
new file mode 100644
index 0000000..6777462
--- /dev/null
+++ b/ingot/mcp_server/provenance.py
@@ -0,0 +1,140 @@
+"""Where each served skill came from.
+
+A merged library hides its own history. Once several roots are mounted together the UI shows
+one flat list, so a skill written here looks exactly like one pulled from a third-party repo.
+That matters for two questions an operator actually asks: what have I contributed, and whose
+licence governs this text.
+
+Four provenances:
+
+- ``authored`` — written in this library.
+- ``vendored`` — copied into this library from an upstream repo. The library records these in
+ a ``VENDORED.md`` ledger beside the skills; a vendored skill lives in the same root as an
+ authored one, so the root alone cannot tell them apart.
+- ``fetched`` — pulled by ``scripts/fetch_skills.sh`` into the local ``skills/`` directory.
+- ``external`` — served from any other configured root.
+
+The ledger is the only source of truth for ``vendored``. A skill copied in but never recorded
+reads as ``authored``, which overstates authorship — the fix is to record it, not to guess here.
+"""
+
+from __future__ import annotations
+
+import os
+import re
+from pathlib import Path
+from ingot import paths
+
+AUTHORED = "authored"
+VENDORED = "vendored"
+FETCHED = "fetched"
+EXTERNAL = "external"
+
+LEDGER_NAME = "VENDORED.md"
+
+# `## threejs-{animation,fundamentals}` and `## a, b` both name several skills in one heading.
+_BRACE = re.compile(r"^(?P[^{]*)\{(?P[^}]*)\}(?P.*)$")
+
+
+def _split_top_level(heading: str) -> list[str]:
+ """Split on commas that separate entries, not on commas inside a brace group.
+
+ `threejs-{a,b}, other` is two entries, not three: the first two commas belong to the
+ brace group. Splitting the raw string first would tear `threejs-{a` off `b}`.
+ """
+ parts, depth, current = [], 0, []
+ for ch in heading:
+ if ch == "{":
+ depth += 1
+ elif ch == "}":
+ depth = max(0, depth - 1)
+ if ch == "," and depth == 0:
+ parts.append("".join(current))
+ current = []
+ else:
+ current.append(ch)
+ parts.append("".join(current))
+ return [p.strip() for p in parts if p.strip()]
+
+
+def _expand(heading: str) -> list[str]:
+ """Skill names named by one ledger heading.
+
+ Handles both shapes the ledger uses: brace expansion (`threejs-{a,b}` -> `threejs-a`,
+ `threejs-b`) and a plain comma-separated list. Anything else is one name.
+ """
+ names: list[str] = []
+ for part in _split_top_level(heading):
+ match = _BRACE.match(part)
+ if match:
+ stem, tail = match.group("stem").strip(), match.group("tail").strip()
+ names.extend(f"{stem}{opt.strip()}{tail}"
+ for opt in match.group("options").split(",") if opt.strip())
+ else:
+ names.append(part)
+ return names
+
+
+def vendored_names(root: Path) -> set[str]:
+ """Skill names the root's ``VENDORED.md`` declares as copied in from upstream.
+
+ Returns an empty set when the root publishes no ledger, which is the common case for a
+ fetched or external root.
+ """
+ ledger = root / LEDGER_NAME
+ try:
+ text = ledger.read_text(encoding="utf-8", errors="ignore")
+ except (OSError, ValueError):
+ return set()
+ names: set[str] = set()
+ for line in text.splitlines():
+ if not line.startswith("## "):
+ continue
+ heading = line[3:].strip()
+ # Prose sections ("Local divergences from upstream") are not skill names. A real skill
+ # slug has no spaces once the comma/brace forms are expanded.
+ for name in _expand(heading):
+ if name and " " not in name:
+ names.add(name)
+ return names
+
+
+def local_root() -> Path:
+ """The repository's own ``skills/`` directory, where fetch_skills.sh writes."""
+ return paths.library()
+
+
+def classify(name: str, skill_dir: str | Path, *,
+ ledgers: dict[Path, set[str]] | None = None) -> str:
+ """Provenance of one skill.
+
+ ``skill_dir`` is ``Skill.root``, which is the skill's OWN directory
+ (``/srv/skills/dotfiles/game-dev``), not the library root. The ledger lives one level up,
+ beside its sibling skills, so the library root is the parent.
+
+ ``ledgers`` caches each library root's parsed ledger, so a whole inventory costs one read
+ per root rather than one per skill.
+ """
+ library = Path(skill_dir).parent
+ if ledgers is None:
+ ledgers = {}
+ if library not in ledgers:
+ ledgers[library] = vendored_names(library)
+ if name in ledgers[library]:
+ return VENDORED
+ try:
+ if library.resolve() == local_root().resolve():
+ return FETCHED
+ except OSError:
+ pass
+ return AUTHORED if (library / LEDGER_NAME).exists() else EXTERNAL
+
+
+def label(provenance: str) -> str:
+ """Human-facing group name."""
+ return {
+ AUTHORED: "Authored here",
+ VENDORED: "Vendored in",
+ FETCHED: "Fetched",
+ EXTERNAL: "External root",
+ }.get(provenance, provenance)
diff --git a/mcp_server/registry.py b/ingot/mcp_server/registry.py
similarity index 74%
rename from mcp_server/registry.py
rename to ingot/mcp_server/registry.py
index 4ea59bc..8d05b17 100644
--- a/mcp_server/registry.py
+++ b/ingot/mcp_server/registry.py
@@ -2,17 +2,25 @@
key; the markdown body is what an agent loads. No compilation, no DB, a skill is just its SKILL.md."""
from __future__ import annotations
import hashlib
+import json
import os
import re
import uuid
import warnings
from dataclasses import dataclass, field
from pathlib import Path
+from pathlib import PurePosixPath
from typing import Iterable
import yaml
+from ingot import paths
+
+def library_dir() -> Path:
+ """The local authoring root. A function, not a constant: it is configuration, and a value
+ frozen at import cannot follow a process that is told where its state lives."""
+ return paths.library()
+
-SKILLS_DIR = Path(__file__).resolve().parent.parent / "skills"
_FRONTMATTER = re.compile(r"^---\s*\n(.*?)\n---\s*\n(.*)$", re.DOTALL)
# One slug rule for every layer (promotion, UI), a name one layer accepts
@@ -85,7 +93,7 @@ def configured_roots(explicit: Iterable[str | Path] | None = None) -> list[Path]
values = list(explicit) if explicit is not None else [
p for p in os.environ.get("SKILL_ROUTER_PATHS", "").split(os.pathsep) if p
]
- values = [SKILLS_DIR, *values] if values else [SKILLS_DIR]
+ values = [library_dir(), *values] if values else [library_dir()]
roots: list[Path] = []
seen: set[Path] = set()
for value in values:
@@ -110,6 +118,11 @@ def _router_metadata(meta: dict) -> dict:
for key in list_fields:
if not isinstance(result[key], list) or not all(isinstance(item, str) for item in result[key]):
raise ValueError(f"metadata.skill-router.{key} must be a list of strings")
+ for pattern in result["path_patterns"]:
+ path = PurePosixPath(pattern)
+ if (not pattern or path.is_absolute() or ".." in path.parts or "\\" in pattern or
+ any(ord(char) < 32 for char in pattern)):
+ raise ValueError("metadata.skill-router.path_patterns must be relative POSIX globs")
try:
result["priority"] = int(result["priority"])
except (TypeError, ValueError):
@@ -144,15 +157,32 @@ def skill_revision(skill_root: Path, components: dict[str, str] | None = None) -
raise ValueError(f"component escapes skill root: {relative}")
_contained_file(skill_root, skill_root / relative)
files.setdefault(relative.as_posix(), skill_root / relative)
+ # A quarantined creation has no directory or SKILL.md yet. Include its logical SKILL.md in
+ # the same digest shape an activated skill will use, so approval can bind the exact candidate
+ # before any filesystem mutation occurs.
+ if components is not None and "SKILL.md" not in files:
+ files["SKILL.md"] = skill_root / "SKILL.md"
digest = hashlib.sha256()
for relative, path in sorted(files.items()):
digest.update(relative.encode())
digest.update(b"\0")
if relative == "SKILL.md":
- meta, body = parse_skill(path.read_text(encoding="utf-8", errors="ignore"), skill_root.name)
+ if path.exists():
+ meta, body = parse_skill(
+ path.read_text(encoding="utf-8", errors="ignore"), skill_root.name)
+ else:
+ meta, body = {"name": skill_root.name, "description": ""}, ""
if components is not None:
- meta["description"] = components["description"]
+ if "frontmatter" in components:
+ try:
+ supplied = json.loads(components["frontmatter"])
+ except (TypeError, ValueError) as exc:
+ raise ValueError("frontmatter component is not valid JSON") from exc
+ meta = normalized_frontmatter(skill_root.name,
+ components["description"], supplied)
+ else:
+ meta["description"] = components["description"]
body = components["body"]
digest.update(yaml.safe_dump(meta, sort_keys=True, allow_unicode=True).encode())
digest.update(b"\0")
@@ -202,6 +232,31 @@ def load_skills(skills_dir: Path | None = None, *, roots: Iterable[str | Path] |
return skills
+def resolve_skill_dir(name: str) -> Path:
+ """The directory `name` actually lives in, across every configured root.
+
+ `library_dir() / name` is only correct for the one *writable* authoring root. A merged library
+ serves most of its skills from read-only mounts, so that path finds them by luck or not at
+ all — and every optimize entry point used to build it by hand, which meant the optimizer
+ silently refused to touch anything it did not itself author."""
+ for item in load_skills():
+ if item.name == name:
+ return Path(item.root)
+ raise LookupError(f"no indexed skill named '{name}'; check SKILL_ROUTER_PATHS")
+
+
+def writable_skill_dir(name: str) -> Path:
+ """The activation destination in the first-precedence writable authoring root.
+
+ This is not a lookup: callers must still use ``resolve_skill_dir`` to read the serving
+ revision. Promotion uses this destination only after resolving and validating that revision,
+ because merged libraries may supply it from a read-only mount.
+ """
+ if not SLUG_RE.fullmatch(name):
+ raise ValueError(f"invalid skill name: {name!r}")
+ return library_dir().expanduser().resolve() / name
+
+
# --- writing / full-skill components (used by candidate generation and promotion) ---
_TEXT_SUFFIXES = {".md", ".txt", ".py", ".sh", ".js", ".ts", ".json", ".yaml", ".yml", ".toml", ".cfg"}
@@ -217,6 +272,27 @@ def write_skill_md(path: Path, meta: dict, body: str) -> None:
path.write_text(f"---\n{dumped}---\n\n{body.strip()}\n", encoding="utf-8")
+def normalized_frontmatter(name: str, description: str, frontmatter: dict | None = None) -> dict:
+ """Canonical, router-valid metadata for a proposed skill before revision hashing.
+
+ Creation cannot hash caller bytes and normalize them only during activation: that would let
+ the reviewed revision differ from what routing serves. A safe YAML round trip also rejects
+ object types that could not be persisted as portable frontmatter.
+ """
+ if frontmatter is not None and not isinstance(frontmatter, dict):
+ raise ValueError("frontmatter must be an object")
+ try:
+ meta = yaml.safe_load(yaml.safe_dump(frontmatter or {}, allow_unicode=True)) or {}
+ except yaml.YAMLError as exc:
+ raise ValueError("frontmatter must contain YAML-safe values") from exc
+ if not isinstance(meta, dict):
+ raise ValueError("frontmatter must be an object")
+ meta["name"] = name
+ meta["description"] = " ".join(str(description).split())
+ _router_metadata(meta)
+ return meta
+
+
def read_components(skill_dir: Path) -> dict[str, str]:
"""Every optimizable text component of a skill: its routing `description`, its SKILL.md `body`,
and each bundled text file as `file:`. This is the unit a candidate rewrite works on."""
diff --git a/ingot/mcp_server/router.py b/ingot/mcp_server/router.py
new file mode 100644
index 0000000..030d5a4
--- /dev/null
+++ b/ingot/mcp_server/router.py
@@ -0,0 +1,298 @@
+"""Embedding router: cache description and bounded approved-content vectors, then suggest the
+top-k skills for a task by cosine similarity. The default remains CPU-only; the tracked GPU
+Compose overlay selects the larger remote model.
+
+Model is `EMBED_MODEL` (default Qwen3-Embedding-0.6B q4, ~15 ms/query on CPU; queries get the
+retrieval instruction prefix, descriptions don't). Any fastembed model name also works (e.g. the
+previous default `BAAI/bge-small-en-v1.5`, ~4 ms/query), but recalibrate MIN_SCORE /
+RELATED_SCORE / COLLISION_SCORE with the model (mcp_server/embedding.py)."""
+from __future__ import annotations
+import os
+import sys
+import threading
+from collections import OrderedDict
+from dataclasses import dataclass
+from pathlib import Path
+
+import numpy as np
+
+from .embedding import EMBED_MODEL as _MODEL, build_embedding
+from .registry import Skill
+
+
+@dataclass(frozen=True)
+class _RankedSkill:
+ skill: Skill
+ score: float
+ description_score: float
+ content_score: float
+
+ @property
+ def matched_on(self) -> str:
+ return "description" if self.description_score >= self.content_score else "content"
+
+ def explanation(self) -> dict:
+ return {
+ "score_components": {
+ "description": round(self.description_score, 3),
+ "content": round(self.content_score, 3),
+ },
+ "matched_on": self.matched_on,
+ }
+
+
+class Router:
+ _vector_cache: OrderedDict[tuple[str, str, str, str], np.ndarray] = OrderedDict()
+ _vector_cache_limit = 4096
+ _cache_lock = threading.Lock()
+
+ def __init__(self, skills: list[Skill]):
+ self.skills = skills
+ try:
+ self._body_chars = int(os.environ.get("ROUTER_BODY_CHARS", "1000"))
+ except ValueError as exc:
+ raise ValueError("ROUTER_BODY_CHARS must be an integer from 1 to 4000") from exc
+ if not 1 <= self._body_chars <= 4000:
+ raise ValueError("ROUTER_BODY_CHARS must be an integer from 1 to 4000")
+ if not skills: # empty library, don't normalize an empty matrix
+ self._embed = None
+ self._mat = np.zeros((0, 0), dtype=np.float32)
+ return
+ self._embed = build_embedding()
+ backend = type(self._embed)
+ self._embedding_identity = getattr(
+ self._embed, "identity", f"{backend.__module__}.{backend.__qualname__}")
+ self._mat = self._matrix("description", [skill.description for skill in skills])
+
+ def _matrix(self, representation: str, texts: list[str]) -> np.ndarray:
+ keys = [(_MODEL, self._embedding_identity, representation, text) for text in texts]
+ resolved = {}
+ with self._cache_lock:
+ missing = []
+ for key in dict.fromkeys(keys):
+ vector = self._vector_cache.get(key)
+ if vector is None:
+ missing.append(key)
+ continue
+ self._vector_cache.move_to_end(key)
+ resolved[key] = vector
+ if missing:
+ vectors = list(self._embed.embed([text for _, _, _, text in missing]))
+ if len(vectors) != len(missing):
+ raise RuntimeError("embedding backend returned the wrong vector count")
+ generated = {
+ key: np.asarray(vector, dtype=np.float32)
+ for key, vector in zip(missing, vectors)
+ }
+ resolved.update(generated)
+ with self._cache_lock:
+ for key, vector in generated.items():
+ self._vector_cache[key] = vector
+ self._vector_cache.move_to_end(key)
+ while len(self._vector_cache) > self._vector_cache_limit:
+ self._vector_cache.popitem(last=False)
+ mat = np.array([resolved[key] for key in keys], dtype=np.float32)
+ return mat / (np.linalg.norm(mat, axis=1, keepdims=True) + 1e-8)
+
+ def _content_text(self, skill: Skill, harness: str) -> str:
+ return (
+ f"Skill: {skill.name}\n"
+ f"Description: {skill.description}\n"
+ f"Instructions:\n{skill.body_for(harness)[:self._body_chars]}"
+ )
+
+ def _ranked(self, task: str, harness: str, skills: list[Skill]) -> list[_RankedSkill]:
+ if not skills:
+ return []
+ query = np.array(next(iter(self._embed.embed_query([task]))), dtype=np.float32)
+ query = query / (np.linalg.norm(query) + 1e-8)
+ index_by_name = {skill.name: index for index, skill in enumerate(self.skills)}
+ content = self._matrix(
+ f"content:{harness or 'default'}",
+ [self._content_text(skill, harness) for skill in skills],
+ )
+ ranked = []
+ for content_index, skill in enumerate(skills):
+ description_score = float(self._mat[index_by_name[skill.name]] @ query)
+ content_score = float(content[content_index] @ query)
+ ranked.append(_RankedSkill(
+ skill=skill,
+ score=max(description_score, content_score),
+ description_score=description_score,
+ content_score=content_score,
+ ))
+ return sorted(
+ ranked,
+ key=lambda item: (
+ -item.score, -int(item.skill.metadata.get("priority", 50)), item.skill.name
+ ),
+ )
+
+ def nearest(self, text: str) -> tuple[str, float]:
+ """The most similar existing skill to `text` and its cosine score, used to reject a new
+ skill whose description near-duplicates (shadows) an existing one's routing."""
+ if not self.skills:
+ return "", 0.0
+ q = np.array(next(iter(self._embed.embed([text]))), dtype=np.float32)
+ q = q / (np.linalg.norm(q) + 1e-8)
+ scores = self._mat @ q
+ i = int(np.argmax(scores))
+ return self.skills[i].name, float(scores[i])
+
+ def suggest(self, task: str, k: int = 5, min_score: float = 0.0) -> list[dict]:
+ if not self.skills:
+ return []
+ return [
+ {
+ "name": item.skill.name,
+ "description": item.skill.description,
+ "score": round(item.score, 3),
+ **item.explanation(),
+ }
+ for item in self._ranked(task, "", self.skills)[:k] if item.score >= min_score
+ ]
+
+ @staticmethod
+ def _platform(value: str | None) -> str:
+ value = (value or sys.platform).lower()
+ if value.startswith("darwin") or value == "macos":
+ return "macos"
+ if value.startswith("win"):
+ return "windows"
+ return "linux" if value.startswith("linux") else value
+
+ @staticmethod
+ def _compatible(skill: Skill, harness: str, cwd: str, available_tools: set[str],
+ available_mcps: set[str], platform: str) -> bool:
+ meta = skill.metadata or {}
+ if harness not in meta.get("harnesses", ["claude", "codex"]):
+ return False
+ if platform not in meta.get("platforms", ["macos", "linux", "windows"]):
+ return False
+ if meta.get("activation", "automatic") != "automatic" or meta.get("trust") == "blocked":
+ return False
+ if not set(meta.get("required_tools", [])).issubset(available_tools):
+ return False
+ if not set(meta.get("required_mcps", [])).issubset(available_mcps):
+ return False
+ scopes = meta.get("scopes", ["global"])
+ if "global" not in scopes:
+ patterns = meta.get("path_patterns", [])
+ if "project" not in scopes or not patterns:
+ return False
+ project = Path(cwd).expanduser().resolve()
+ try:
+ matched = any(any(project.glob(pattern)) for pattern in patterns)
+ except (NotImplementedError, OSError, ValueError):
+ return False
+ if not project.is_dir() or not matched:
+ return False
+ return True
+
+ def _eligible_ranking(self, task: str, harness: str, cwd: str,
+ available_tools: set[str], available_mcps: set[str],
+ platform: str) -> list[_RankedSkill]:
+ eligible = [skill for skill in self.skills if self._compatible(
+ skill, harness, cwd, available_tools, available_mcps, platform
+ )]
+ return self._ranked(task, harness, eligible)
+
+ @staticmethod
+ def _without_conflicts(ranked: list[_RankedSkill]) -> list[_RankedSkill]:
+ selected = []
+ for candidate in ranked:
+ skill = candidate.skill
+ if any(skill.name in set(existing.skill.metadata.get("conflicts", [])) or
+ existing.skill.name in set(skill.metadata.get("conflicts", []))
+ for existing in selected):
+ continue
+ selected.append(candidate)
+ return selected
+
+ @staticmethod
+ def _alternatives(ranked: list[_RankedSkill]) -> list[dict]:
+ return [
+ {
+ "name": item.skill.name,
+ "score": round(item.score, 3),
+ "reason": f"compatible alternative; {item.matched_on} cosine {item.score:.3f}",
+ **item.explanation(),
+ }
+ for item in ranked[1:3]
+ ]
+
+ @staticmethod
+ def _novel_response(score: float = 0.0, reason: str = "no compatible skill candidates",
+ alternatives: list[dict] | None = None,
+ candidate: _RankedSkill | None = None) -> dict:
+ explanation = (
+ candidate.explanation()
+ if candidate is not None
+ else {
+ "matched_on": None,
+ "score_components": {"description": 0.0, "content": 0.0},
+ }
+ )
+ return {
+ "match": None, "related_match": None, "score": round(score, 3),
+ "reason": reason, "skill_body": "", "skill_root": None, "revision": None,
+ "alternatives": alternatives or [], "novel": True, **explanation,
+ }
+
+ @staticmethod
+ def _related_response(item: _RankedSkill, harness: str, min_score: float,
+ alternatives: list[dict]) -> dict:
+ skill, score = item.skill, item.score
+ return {
+ "match": None, "related_match": skill.name, "score": round(score, 3),
+ "reason": (f"best compatible score {score:.3f} below direct threshold "
+ f"{min_score:.3f}; matched on {item.matched_on}; "
+ "loaded for compose or extend"),
+ "skill_body": skill.body_for(harness),
+ "skill_root": skill.root or str(os.path.dirname(skill.path)),
+ "revision": skill.revision or None, "alternatives": alternatives, "novel": False,
+ **item.explanation(),
+ }
+
+ @staticmethod
+ def _direct_response(item: _RankedSkill, harness: str,
+ alternatives: list[dict]) -> dict:
+ skill, score = item.skill, item.score
+ return {
+ "match": skill.name, "related_match": None, "score": round(score, 3),
+ "reason": f"compatible {harness} skill; {item.matched_on} cosine {score:.3f}",
+ "skill_body": skill.body_for(harness),
+ "skill_root": skill.root or str(os.path.dirname(skill.path)),
+ "revision": skill.revision or None, "alternatives": alternatives, "novel": False,
+ **item.explanation(),
+ }
+
+ def route(self, task: str, harness: str, cwd: str, available_tools=(), available_mcps=(),
+ platform: str | None = None, min_score: float = 0.53,
+ related_score: float = 0.37) -> dict:
+ """Filter compatible skills, rank them locally, and return at most one instruction body.
+ `novel` is the escalation signal for the calling harness: True when nothing compatible is
+ even related (best score below `related_score`), the case where a weak/strong setup should
+ serve with the strong model, then queue a candidate for human review."""
+ harness = harness.lower()
+ ranked = self._eligible_ranking(
+ task, harness, cwd, set(available_tools), set(available_mcps), self._platform(platform)
+ )
+ if not ranked:
+ return self._novel_response()
+ ranked = self._without_conflicts(ranked)
+ top = ranked[0]
+ score = top.score
+ alternatives = self._alternatives(ranked)
+ if score < min_score:
+ if score < related_score:
+ related = [{"name": top.skill.name, "score": round(score, 3),
+ "reason": (f"best compatible candidate; {top.matched_on} "
+ f"cosine {score:.3f}"),
+ **top.explanation()},
+ *alternatives]
+ reason = (f"best compatible score {score:.3f} below related threshold "
+ f"{related_score:.3f}")
+ return self._novel_response(score, reason, related[:3], top)
+ return self._related_response(top, harness, min_score, alternatives)
+ return self._direct_response(top, harness, alternatives)
diff --git a/mcp_server/routing_eval.py b/ingot/mcp_server/routing_eval.py
similarity index 100%
rename from mcp_server/routing_eval.py
rename to ingot/mcp_server/routing_eval.py
diff --git a/mcp_server/server.py b/ingot/mcp_server/server.py
similarity index 67%
rename from mcp_server/server.py
rename to ingot/mcp_server/server.py
index 5d90356..ab60ae9 100644
--- a/mcp_server/server.py
+++ b/ingot/mcp_server/server.py
@@ -13,7 +13,7 @@
MIN_SCORE = float(os.environ.get("MIN_SCORE", "0.53"))
RELATED_SCORE = float(os.environ.get("RELATED_SCORE", "0.37"))
PORT = int(os.environ.get("PORT", "8000"))
-# Loopback by default: the tools are unauthenticated, so a bare `python -m mcp_server.server` must
+# Loopback by default: the tools are unauthenticated, so a bare `python -m ingot.mcp_server.server` must
# not listen on the network. The compose mcp service sets HOST=0.0.0.0 (required for Docker port
# publishing); host access stays localhost-only via the 127.0.0.1 port mapping.
HOST = os.environ.get("HOST", "127.0.0.1")
@@ -118,6 +118,81 @@ def route_and_load(task: str, harness: str, cwd: str, available_tools: list[str]
return result
+@mcp.tool()
+def propose_skill_update(
+ skill: str,
+ champion_revision: str,
+ challenger_body: str,
+ summary: str,
+ trigger: str,
+ minimal_content: str,
+ producer: str,
+ caller: str,
+ evidence: list[str],
+ pressure_scenario: str,
+ risk: str,
+ verification_status: str,
+ verification_command: str,
+ verification_result: str,
+ challenger_description: str = "",
+) -> dict:
+ """Quarantine an evidence-backed update proposed by skill-retrospective.
+
+ This tool never activates, rejects, or replaces instructions. It accepts one full candidate
+ for an existing skill after pressure verification passes, binds it to the exact loaded champion
+ revision, and refuses an occupied review slot. Human approval in the Ingot console remains
+ required.
+ """
+ from ingot.optimize.retrospective import submit_skill_update
+ return submit_skill_update(
+ skill=skill,
+ champion_revision=champion_revision,
+ challenger_body=challenger_body,
+ challenger_description=challenger_description,
+ summary=summary,
+ trigger=trigger,
+ minimal_content=minimal_content,
+ producer=producer,
+ caller=caller,
+ evidence=evidence,
+ pressure_scenario=pressure_scenario,
+ risk=risk,
+ verification_status=verification_status,
+ verification_command=verification_command,
+ verification_result=verification_result,
+ )
+
+
+@mcp.tool()
+def propose_skill_create(
+ skill: str,
+ description: str,
+ body: str,
+ files: dict[str, str],
+ frontmatter: dict,
+ summary: str,
+ source: str,
+ producer: str,
+ caller: str,
+ evidence: list[str],
+ pressure_scenario: str,
+ risk: str,
+ verification_status: str,
+ verification_command: str,
+ verification_result: str,
+) -> dict:
+ """Quarantine a vetted new skill package as “to be added”; never activate it."""
+ from ingot.optimize.ingress import submit_skill_create
+ return submit_skill_create(
+ skill=skill, description=description, body=body, files=files, frontmatter=frontmatter,
+ summary=summary,
+ source=source, producer=producer, caller=caller, evidence=evidence,
+ pressure_scenario=pressure_scenario, risk=risk,
+ verification_status=verification_status, verification_command=verification_command,
+ verification_result=verification_result,
+ )
+
+
if __name__ == "__main__":
print(f"[ingot] {len(STATE.skills)} skills loaded; serving MCP on :{PORT}/mcp", flush=True)
mcp.run(transport="http", host=HOST, port=PORT, path="/mcp",
diff --git a/mcp_server/usage_counts.py b/ingot/mcp_server/usage_counts.py
similarity index 90%
rename from mcp_server/usage_counts.py
rename to ingot/mcp_server/usage_counts.py
index 2eac885..0b55e30 100644
--- a/mcp_server/usage_counts.py
+++ b/ingot/mcp_server/usage_counts.py
@@ -6,10 +6,10 @@
import os
import threading
from pathlib import Path
+from ingot import paths
_LOCK = threading.Lock()
-_PATH = Path(os.environ.get("SKILL_USAGE_FILE") or
- Path(__file__).resolve().parent.parent / "runs" / "skill_usage.json")
+_PATH = Path(os.environ.get("SKILL_USAGE_FILE") or paths.runs() / "skill_usage.json")
def load_counts() -> dict[str, int]:
diff --git a/optimize/__init__.py b/ingot/optimize/__init__.py
similarity index 89%
rename from optimize/__init__.py
rename to ingot/optimize/__init__.py
index 35cc94a..b73e3fa 100644
--- a/optimize/__init__.py
+++ b/ingot/optimize/__init__.py
@@ -193,3 +193,30 @@ def preflight_provider_pins() -> list[str]:
# Loaded skill
{body}"""
+
+
+def resolve_skill_dir(name: str):
+ """Where a skill lives, as a clean CLI exit rather than a traceback when it is not indexed.
+
+ Every optimize entry point needs this and none of them should hand-build `library_dir() / name`:
+ only the authoring root is writable, so that path misses every skill served from a read-only
+ mount."""
+ from ingot.mcp_server.registry import resolve_skill_dir as _resolve
+ try:
+ return _resolve(name)
+ except LookupError as e:
+ raise SystemExit(str(e)) from e
+
+
+def configured_models(setting: str, fallback: str = "") -> list[str]:
+ """Parse one comma-separated model setting and refuse accidental duplicate votes/spend."""
+ configured = os.environ.get(setting)
+ raw = configured if configured and any(part.strip() for part in configured.split(",")) \
+ else fallback
+ models = [model.strip() for model in raw.split(",") if model.strip()]
+ seen: set[str] = set()
+ for model in models:
+ if model in seen:
+ raise SystemExit(f"{setting} contains duplicate model '{model}'; list each model once")
+ seen.add(model)
+ return models
diff --git a/optimize/ab.py b/ingot/optimize/ab.py
similarity index 94%
rename from optimize/ab.py
rename to ingot/optimize/ab.py
index e2bfb23..bb1700b 100644
--- a/optimize/ab.py
+++ b/ingot/optimize/ab.py
@@ -1,12 +1,12 @@
"""Generate a candidate change for one skill and produce the evidence a reviewer needs.
-The candidate search (optimize.skillopt_loop, driven per-component by `_greedy_search`) turns the
+The candidate search (ingot.optimize.skillopt_loop, driven per-component by `_greedy_search`) turns the
skill's components into a challenger on the train tasks. Champion and challenger then run through the full agent with a local `route_and_load` on the
held-out tasks; each variant is a Langfuse dataset run (side-by-side in the UI) when the stack is
up. The result is a quarantined record in runs/pending/.json plus a portable evidence bundle
in runs/evidence/. Nothing here activates anything: promotion is a human action in the UI.
-Usage: python -m optimize.ab [--description | --scripts] [--skip-search]
+Usage: python -m ingot.optimize.ab [--description | --scripts] [--skip-search]
"""
import argparse
import asyncio
@@ -22,16 +22,17 @@
from langchain_core.tools import tool
from agent.run import build_agent, langfuse_config, run_task
-from mcp_server.registry import SKILLS_DIR, load_skills, optimizable_components, skill_revision
+from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision
-from . import agent_model, langfuse_available
+from . import agent_model, langfuse_available, resolve_skill_dir
from . import usage as usage_ledger
from .acceptance import classify as acceptance_classify, load_criteria as load_acceptance
from .judge import MODELS as JUDGE_MODELS, judge
from .promote import save_pending
from .evidence import build_evidence, recorded_path, write_evidence
+from ingot import paths
-TASKS_DIR = Path(__file__).resolve().parent / "tasks"
+TASKS_DIR = paths.tasks()
# --- promotion gate (anti reward-hacking / overfitting) ---
PROMOTE_MIN_MARGIN = float(os.environ.get("PROMOTE_MIN_MARGIN", "0.15")) # mean holdout lift required
@@ -57,7 +58,7 @@
OPTIMIZE_COMPONENTS = [c.strip() for c in os.environ.get("OPTIMIZE_COMPONENTS", "body").split(",")
if c.strip()]
-EVAL_CACHE_DIR = Path(__file__).resolve().parent.parent / "runs" / "eval-cache"
+EVAL_CACHE_DIR = paths.runs() / "eval-cache"
def _champion_cache_key(revision: str, holdout: list[dict], components: list[str]) -> str:
@@ -112,8 +113,8 @@ def retention_warnings(champion: dict, challenger: dict, changed: list[str], sam
def _description_shadows(skill: str, new_description: str) -> tuple[str, float]:
"""Nearest OTHER skill to a rewritten description (route-shadow check). ("",0.0) if none."""
- from mcp_server.registry import load_skills
- from mcp_server.router import Router
+ from ingot.mcp_server.registry import load_skills
+ from ingot.mcp_server.router import Router
others = [s for s in load_skills() if s.name != skill]
if not others:
return "", 0.0
@@ -184,17 +185,20 @@ def promotion_gate(skill: str, champ_scores: list[float], chall_scores: list[flo
from . import SERVE_TEMPLATE as EVAL_SERVE_TEMPLATE # noqa: E402
-def load_tasks(skill: str, log=print) -> tuple[list[dict], list[dict], dict]:
+def load_tasks(skill: str, log=print,
+ draft_components: dict[str, str] | None = None) -> tuple[list[dict], list[dict], dict]:
"""Return train, holdout, and split metadata. The candidate search sees train; the gate is
judged on holdout, a leakage-clean split so a challenger has to *generalize*, not memorize.
A flat `tasks:` list is marked leaky and cannot produce a promotable gate. If no task set exists,
the teacher drafts one; that draft must include a real holdout before promotion."""
p = TASKS_DIR / f"{skill}.yaml"
if not p.exists():
- from mcp_server.registry import SKILLS_DIR as _SD, read_components
- comps = read_components(_SD / skill)
+ if draft_components is None:
+ from ingot.mcp_server.registry import read_components
+ draft_components = read_components(resolve_skill_dir(skill))
from .draft import draft_and_save
- draft_and_save(skill, comps["description"], comps["body"], TASKS_DIR, log=log)
+ draft_and_save(skill, draft_components["description"], draft_components["body"],
+ TASKS_DIR, log=log)
data = yaml.safe_load(p.read_text())
train = data.get("train") or data.get("tasks") or []
explicit_holdout = bool(data.get("holdout"))
@@ -205,7 +209,7 @@ def load_tasks(skill: str, log=print) -> tuple[list[dict], list[dict], dict]:
def _variant_tools(skill: str, body: str, description: str):
"""One read-only route tool backed by the variant description and body."""
- from mcp_server.router import Router
+ from ingot.mcp_server.router import Router
skills = load_skills()
variants = [replace(item, description=description, body=body) if item.name == skill else item
for item in skills]
@@ -221,7 +225,7 @@ async def route_and_load(task: str, harness: str, cwd: str, available_tools: lis
def _routing_failures(skill: str, challenger: dict, tasks: list[dict]) -> list[str]:
- from mcp_server.router import Router
+ from ingot.mcp_server.router import Router
skills = load_skills()
variants = [replace(item, description=challenger["description"], body=challenger["body"])
if item.name == skill else item for item in skills]
@@ -235,8 +239,8 @@ def _routing_metrics(skill: str, champion: dict, challenger: dict) -> dict | Non
cases = data.get("routing") or []
if not cases:
return None
- from mcp_server.router import Router
- from mcp_server.routing_eval import evaluate_cases, evaluate_parity
+ from ingot.mcp_server.router import Router
+ from ingot.mcp_server.routing_eval import evaluate_cases, evaluate_parity
skills = load_skills()
def variant(components):
@@ -273,7 +277,7 @@ async def task_fn(*, item, **kwargs):
def judge_evaluator(*, input, output, **kwargs):
j = judge(input["task"], input["rubric"], str(output), check=input.get("check"),
- deliverable=input.get("deliverable"))
+ deliverable=input.get("deliverable"), checklist=input.get("checklist"))
scores_by_task[input["task"]] = j["score"]
return Evaluation(name="judge_score", value=j["score"], comment=j["feedback"])
@@ -309,7 +313,8 @@ async def rollout_all():
with ThreadPoolExecutor(max_workers=max(1, len(ok))) as pool:
judgments = list(pool.map(
lambda x: judge(x[1]["task"], x[1]["rubric"], str(x[2][0]),
- check=x[1].get("check"), deliverable=x[1].get("deliverable")), ok))
+ check=x[1].get("check"), deliverable=x[1].get("deliverable"),
+ checklist=x[1].get("checklist")), ok))
for (i, _, _), j in zip(ok, judgments):
scores[i] = j["score"]
zero = {"input_tokens": 0, "output_tokens": 0}
@@ -436,16 +441,14 @@ def run_ab(skill: str, skip_search: bool = False, challenger_file: str | None =
usage_ledger.reset()
components = components if components is not None else OPTIMIZE_COMPONENTS
train, holdout, split = load_tasks(skill)
- skill_dir = SKILLS_DIR / skill
- if not (skill_dir / "SKILL.md").exists():
- raise SystemExit(f"No skill named '{skill}' in skills/.")
+ skill_dir = resolve_skill_dir(skill)
# description + body always; bundled file components join only when `components` names
# them (they then also render into rollouts and the A/B serving). Everything else stays
# untouched on disk.
champion = optimizable_components(skill_dir)
file_components = [c for c in components if c.startswith("file:")]
if file_components:
- from mcp_server.registry import read_components
+ from ingot.mcp_server.registry import read_components
everything = read_components(skill_dir)
champion.update({k: everything[k] for k in file_components if k in everything})
@@ -464,7 +467,7 @@ def run_ab(skill: str, skip_search: bool = False, challenger_file: str | None =
return {"skill": skill, "improved": False}
log(f"[opt] components changed: {changed}")
# checkpoint the candidate so an A/B failure doesn't cost the whole search
- ckpt = Path(__file__).resolve().parent.parent / "runs" / f"challenger-{skill}.json"
+ ckpt = paths.runs() / f"challenger-{skill}.json"
ckpt.parent.mkdir(parents=True, exist_ok=True)
ckpt.write_text(json.dumps({"components": challenger, "seed_score": seed_score,
"best_score": best_score}, indent=2))
@@ -539,7 +542,7 @@ def run_ab(skill: str, skip_search: bool = False, challenger_file: str | None =
summary["optimization_usage"] = usage_ledger.report()
evidence = build_evidence(summary, champion_skill.revision,
skill_revision(Path(champion_skill.root), challenger))
- evidence_root = Path(__file__).resolve().parent.parent / "runs" / "evidence" / skill / str(ts)
+ evidence_root = paths.runs() / "evidence" / skill / str(ts)
evidence_json, evidence_markdown = write_evidence(evidence, evidence_root)
summary["evidence"] = evidence
summary["evidence_paths"] = {"json": recorded_path(evidence_json),
@@ -575,10 +578,8 @@ def script_pass_components(skill: str) -> list[str]:
bundles no scripts, or when no holdout task carries an execution-grounded `check:` (the LLM
judge alone cannot tell a broken script from a working one), so a scripts pass without checks
would produce evidence worth nothing."""
- from mcp_server.registry import read_components
- skill_dir = SKILLS_DIR / skill
- if not (skill_dir / "SKILL.md").exists():
- raise SystemExit(f"No skill named '{skill}' in skills/.")
+ from ingot.mcp_server.registry import read_components
+ skill_dir = resolve_skill_dir(skill)
scripts = sorted(k for k in read_components(skill_dir) if k.startswith("file:scripts/"))
if not scripts:
raise SystemExit(f"'{skill}' bundles no scripts/ files, nothing for the scripts pass "
@@ -590,14 +591,14 @@ def script_pass_components(skill: str) -> list[str]:
f"'{skill}' has no execution-grounded holdout checks. The scripts pass needs at "
f"least one holdout task with a check: {{fixture, assert}} entry so a broken script "
f"fails objectively instead of being waved through by the judge. Add one to "
- f"optimize/tasks/{skill}.yaml first.")
+ f"ingot/optimize/tasks/{skill}.yaml first.")
return scripts
def build_parser() -> argparse.ArgumentParser:
"""The candidate-generation CLI. Kept out of `__main__` so its rejections are testable: a flag
for a pass that does not exist has to fail loudly, not be quietly accepted or ignored."""
- ap = argparse.ArgumentParser(prog="python -m optimize.ab")
+ ap = argparse.ArgumentParser(prog="python -m ingot.optimize.ab")
ap.add_argument("skill")
passes = ap.add_mutually_exclusive_group()
passes.add_argument("--body", action="store_true",
diff --git a/optimize/acceptance.py b/ingot/optimize/acceptance.py
similarity index 100%
rename from optimize/acceptance.py
rename to ingot/optimize/acceptance.py
diff --git a/ingot/optimize/agy_judge.py b/ingot/optimize/agy_judge.py
new file mode 100644
index 0000000..4d61ef1
--- /dev/null
+++ b/ingot/optimize/agy_judge.py
@@ -0,0 +1,256 @@
+"""Fail-closed subprocess adapter for the subscription-backed Agy judge."""
+from __future__ import annotations
+
+import json
+import os
+import subprocess
+import threading
+import time
+from pathlib import Path
+from typing import Mapping, Sequence
+
+
+AGY_MODEL = "gemini-3.6-flash-medium"
+AGY_IDENTITY = f"agy/{AGY_MODEL}"
+
+_VERDICTS = ("pass", "partial", "fail")
+_PROVIDER_PREFIXES = (
+ "OPENROUTER_",
+ "OPENAI_",
+ "ANTHROPIC_",
+ "GEMINI_",
+ "GOOGLE_",
+ "VERTEX_",
+)
+_PROVIDER_KEYS = frozenset({"BASE_URL", "API_KEY", "MODEL_API_KEY"})
+_LAUNCH_INTERVAL_SECONDS = 3.2
+_RESOURCE_EXHAUSTED_RETRIES = 3
+_launch_lock = threading.Lock()
+_next_launch_at = 0.0
+
+
+class AgyJudgeError(RuntimeError):
+ """Agy did not produce one trustworthy checklist grade."""
+
+
+def agy_process_env(parent: Mapping[str, str]) -> dict[str, str]:
+ """Copy the parent environment without provider credentials or routing controls."""
+ child = {
+ key: value
+ for key, value in parent.items()
+ if key not in _PROVIDER_KEYS and not key.startswith(_PROVIDER_PREFIXES)
+ }
+ child["AGY_CLI_DISABLE_AUTO_UPDATE"] = "true"
+ return child
+
+
+def _checklist_ids(checklist: Sequence[Mapping[str, object]]) -> list[str]:
+ ids = [item.get("id") for item in checklist]
+ if not ids or any(not isinstance(item_id, str) or not item_id for item_id in ids):
+ raise AgyJudgeError("Agy checklist requires non-empty string IDs")
+ if len(set(ids)) != len(ids):
+ raise AgyJudgeError("Agy checklist IDs must be unique")
+ return ids
+
+
+def judge_schema(checklist: Sequence[Mapping[str, object]]) -> dict:
+ """Build the strict Agy output schema for the exact checklist IDs."""
+ ids = _checklist_ids(checklist)
+ item_schema = {
+ "type": "object",
+ "additionalProperties": False,
+ "required": ["verdict", "note"],
+ "properties": {
+ "verdict": {"enum": list(_VERDICTS)},
+ "note": {"type": "string"},
+ },
+ }
+ return {
+ "type": "object",
+ "additionalProperties": False,
+ "required": ["items", "feedback"],
+ "properties": {
+ "items": {
+ "type": "object",
+ "additionalProperties": False,
+ "required": ids,
+ "properties": {item_id: item_schema.copy() for item_id in ids},
+ },
+ "feedback": {"type": "string"},
+ },
+ }
+
+
+def _validate_grade(raw: object, checklist: Sequence[Mapping[str, object]]) -> dict:
+ ids = _checklist_ids(checklist)
+ if not isinstance(raw, dict):
+ raise AgyJudgeError("Agy result has no structured output")
+ if set(raw) != {"items", "feedback"} or not isinstance(raw.get("feedback"), str):
+ raise AgyJudgeError("Agy structured output has an invalid top-level shape")
+ items = raw.get("items")
+ if not isinstance(items, dict) or set(items) != set(ids):
+ raise AgyJudgeError("Agy structured output does not match the checklist IDs")
+ for item_id in ids:
+ item = items[item_id]
+ if not isinstance(item, dict) or set(item) != {"verdict", "note"}:
+ raise AgyJudgeError(f"Agy checklist item {item_id!r} has an invalid shape")
+ if item["verdict"] not in _VERDICTS:
+ raise AgyJudgeError(f"Agy checklist item {item_id!r} has an unknown verdict")
+ if not isinstance(item["note"], str):
+ raise AgyJudgeError(f"Agy checklist item {item_id!r} has an invalid note")
+ return raw
+
+
+def parse_stream(
+ stdout: str,
+ checklist: Sequence[Mapping[str, object]],
+) -> tuple[dict, dict]:
+ """Return the sole successful terminal grade and usage from an Agy JSONL stream."""
+ terminal = []
+ for line_number, line in enumerate(stdout.splitlines(), start=1):
+ if not line.strip():
+ continue
+ try:
+ event = json.loads(line)
+ except json.JSONDecodeError as exc:
+ raise AgyJudgeError(f"Agy emitted malformed JSONL on line {line_number}") from exc
+ if not isinstance(event, dict):
+ raise AgyJudgeError(f"Agy emitted a non-object event on line {line_number}")
+ if event.get("event") == "result":
+ terminal.append(event)
+ if len(terminal) != 1:
+ raise AgyJudgeError(f"Agy emitted {len(terminal)} terminal results; expected one")
+
+ result = terminal[0].get("result")
+ if not isinstance(result, dict):
+ raise AgyJudgeError("Agy terminal result has an invalid shape")
+ if result.get("status") != "SUCCESS":
+ raise AgyJudgeError("Agy terminal result was not successful")
+ grade = _validate_grade(result.get("structured_output"), checklist)
+ usage = result.get("usage")
+ if not isinstance(usage, dict):
+ raise AgyJudgeError("Agy terminal result has no usage object")
+ return grade, usage
+
+
+def _runtime_paths() -> tuple[Path, Path]:
+ agy_value = os.environ.get("AGY_BIN", "").strip()
+ workspace_value = os.environ.get("AGY_JUDGE_WORKSPACE", "").strip()
+ if not agy_value or not workspace_value:
+ raise AgyJudgeError("AGY_BIN and AGY_JUDGE_WORKSPACE must be set")
+ agy_bin = Path(agy_value)
+ workspace = Path(workspace_value)
+ if not agy_bin.is_absolute() or not workspace.is_absolute():
+ raise AgyJudgeError("Agy runtime paths must be absolute")
+ if not agy_bin.is_file() or not os.access(agy_bin, os.X_OK):
+ raise AgyJudgeError("AGY_BIN is not an executable file")
+ if not workspace.is_dir():
+ raise AgyJudgeError("AGY_JUDGE_WORKSPACE is not a directory")
+ return agy_bin, workspace
+
+
+def invoke(
+ prompt: str,
+ checklist: Sequence[Mapping[str, object]],
+ *,
+ timeout: float = 120.0,
+) -> tuple[dict, dict]:
+ """Invoke Agy once and return no grade unless its complete contract validates."""
+ agy_bin, workspace = _runtime_paths()
+ schema = judge_schema(checklist)
+ argv = [
+ str(agy_bin),
+ "--model", AGY_MODEL,
+ "--print", prompt,
+ "--output-format", "stream-json",
+ "--json-schema", json.dumps(schema),
+ "--sandbox",
+ "--mode", "plan",
+ "--disable-slash-commands",
+ "--print-timeout", f"{int(timeout)}s",
+ ]
+ for attempt in range(_RESOURCE_EXHAUSTED_RETRIES + 1):
+ _wait_for_launch_slot()
+ try:
+ done = subprocess.run(
+ argv,
+ cwd=workspace,
+ env=agy_process_env(os.environ),
+ text=True,
+ capture_output=True,
+ timeout=timeout + 10,
+ check=False,
+ )
+ except subprocess.TimeoutExpired as exc:
+ raise AgyJudgeError("Agy judge timed out") from exc
+ except OSError as exc:
+ raise AgyJudgeError("Agy judge process could not start") from exc
+ if done.returncode == 0:
+ return parse_stream(done.stdout, checklist)
+ if not _resource_exhausted(done.stdout) or attempt == _RESOURCE_EXHAUSTED_RETRIES:
+ raise AgyJudgeError(f"Agy judge exited with status {done.returncode}")
+ raise AssertionError("unreachable")
+
+
+def _wait_for_launch_slot() -> None:
+ """Keep the subscription eligibility gate below its observed burst limit."""
+ global _next_launch_at
+ with _launch_lock:
+ now = time.monotonic()
+ delay = max(0.0, _next_launch_at - now)
+ if delay:
+ time.sleep(delay)
+ now = time.monotonic()
+ _next_launch_at = now + _LAUNCH_INTERVAL_SECONDS
+
+
+def _resource_exhausted(stdout: str) -> bool:
+ for line in stdout.splitlines():
+ try:
+ event = json.loads(line)
+ except json.JSONDecodeError:
+ continue
+ result = event.get("result") if isinstance(event, dict) else None
+ if not isinstance(result, dict) or result.get("status") != "ERROR":
+ continue
+ error = str(result.get("error", ""))
+ if "RESOURCE_EXHAUSTED" in error and "429" in error:
+ return True
+ return False
+
+
+def _run_preflight(argv: list[str], workspace: Path) -> str:
+ try:
+ done = subprocess.run(
+ argv,
+ cwd=workspace,
+ env=agy_process_env(os.environ),
+ text=True,
+ capture_output=True,
+ timeout=20.0,
+ check=False,
+ )
+ except subprocess.TimeoutExpired as exc:
+ raise AgyJudgeError("Agy preflight timed out") from exc
+ except OSError as exc:
+ raise AgyJudgeError("Agy preflight process could not start") from exc
+ if done.returncode != 0:
+ raise AgyJudgeError(f"Agy preflight exited with status {done.returncode}")
+ if not done.stdout.strip():
+ raise AgyJudgeError("Agy preflight returned no output")
+ return done.stdout.strip()
+
+
+def preflight() -> dict[str, object]:
+ """Verify the explicit runtime and return its fixed judge provenance."""
+ agy_bin, workspace = _runtime_paths()
+ version = _run_preflight([str(agy_bin), "--version"], workspace)
+ models = _run_preflight([str(agy_bin), "models"], workspace)
+ if AGY_MODEL not in models.split():
+ raise AgyJudgeError(f"Agy model {AGY_MODEL!r} is unavailable")
+ return {
+ "identity": AGY_IDENTITY,
+ "model": AGY_MODEL,
+ "version": version,
+ "billing_mode": "subscription",
+ }
diff --git a/ingot/optimize/cluster.py b/ingot/optimize/cluster.py
new file mode 100644
index 0000000..57b458a
--- /dev/null
+++ b/ingot/optimize/cluster.py
@@ -0,0 +1,205 @@
+"""Group the library into category buckets from the embeddings the router already uses.
+
+A merged library is a flat list of names. Provenance folders say where a skill came from, which is
+a fact about its origin and not about what it does — "authored here" holds testing skills next to
+finance skills. This clusters on meaning instead, so the console can show what the library is
+actually about and where it is thin.
+
+Not UMAP + HDBSCAN, which is what the same view uses over 100k prompts elsewhere — but the reduction
+those bring is not optional. Clustering the raw 1024-d embeddings scored silhouette +0.05 and put
+half the library in one bucket, because at that width every pair of a hundred points sits at
+roughly the same distance. Over the top 10 principal components the same run scores +0.28. So:
+cosine k-means on an SVD projection, numpy only, with the reduction doing the job UMAP does there.
+
+Clustering is not cheap enough to run inside a request (it loads the embedding model), so this is a
+command that writes runs/clusters.json and the UI serves the file.
+
+ python -m ingot.optimize.cluster
+"""
+import json
+import os
+import re
+from collections import Counter
+from pathlib import Path
+
+import numpy as np
+from ingot import paths
+
+CLUSTER_PATH = paths.runs() / "clusters.json"
+# Enough buckets to separate concerns, few enough that each one still means something. Bounded
+# because both ends degenerate: k=2 says nothing, k=n gives every skill its own bucket.
+K_MIN, K_MAX = 4, 12
+# Components to cluster over. Measured on the 102-skill library: raw 1024d scores silhouette
+# +0.05, 10d +0.28, 20d +0.19, 40d +0.13. Keep enough signal to separate topics, few enough
+# dimensions that distances still mean something.
+CLUSTER_DIMS = 10
+_WORD = re.compile(r"[a-z][a-z0-9+-]{2,}")
+# Words that describe every skill in a library of instructions and so distinguish none of them.
+_STOP = {
+ "the", "and", "for", "when", "with", "this", "that", "you", "your", "use", "uses", "used",
+ "using", "user", "from", "into", "not", "any", "are", "its", "has", "have", "was", "will",
+ "can", "should", "must", "need", "needs", "want", "wants", "asks", "ask", "them", "they",
+ "what", "why", "how", "who", "which", "than", "then", "there", "here", "over", "under",
+ "before", "after", "each", "every", "some", "all", "one", "two", "new", "own", "out",
+ "run", "runs", "get", "gets", "set", "sets", "make", "makes", "does", "done", "via",
+ "skill", "skills", "task", "tasks", "work", "works", "write", "writes", "write-up",
+ # Trigger-phrase vocabulary. A routing description is written to be matched ("Use whenever the
+ # user asks…"), so this register appears in most of them and named a bucket "whenever · api".
+ "whenever", "asking", "request", "requests", "mentions", "triggers", "trigger", "wants",
+ "needed", "including", "instead", "rather", "across", "against", "within", "about",
+}
+
+
+def skill_texts(skills) -> list[str]:
+ """What gets embedded. The name carries real signal in this library (`aws-cdk`,
+ `writing-clearly`), so it leads, and the description is the routing trigger the router itself
+ matches on — clustering the same text keeps the buckets consistent with routing behaviour."""
+ return [f"{s.name.replace('-', ' ')}. {s.description}" for s in skills]
+
+
+def _normalise(matrix: np.ndarray) -> np.ndarray:
+ norms = np.linalg.norm(matrix, axis=1, keepdims=True)
+ return matrix / np.maximum(norms, 1e-12)
+
+
+def kmeans(vectors: np.ndarray, k: int, seed: int = 0, iters: int = 60):
+ """k-means++ init, Lloyd iterations, on unit vectors — so squared euclidean ranks the same as
+ cosine. Seeded: the same library must produce the same buckets twice, or the view reshuffles
+ under a poll and nobody can trust what they are looking at."""
+ rng = np.random.default_rng(seed)
+ n = len(vectors)
+ centres = [vectors[rng.integers(n)]]
+ for _ in range(k - 1): # k-means++: sample far from what is already chosen
+ d2 = np.min(((vectors[:, None, :] - np.array(centres)[None]) ** 2).sum(-1), axis=1)
+ total = d2.sum()
+ centres.append(vectors[rng.choice(n, p=d2 / total) if total > 0 else rng.integers(n)])
+ centres = np.array(centres)
+ labels = np.zeros(n, dtype=int)
+ for _ in range(iters):
+ labels = np.argmin(((vectors[:, None, :] - centres[None]) ** 2).sum(-1), axis=1)
+ moved = False
+ for j in range(k):
+ members = vectors[labels == j]
+ if not len(members): # an emptied centre is re-seeded, never left to
+ members = vectors[rng.integers(n)][None] # collapse the run to k-1 buckets
+ centre = _normalise(members.mean(0, keepdims=True))[0]
+ if not np.allclose(centre, centres[j]):
+ centres[j], moved = centre, True
+ if not moved:
+ break
+ return labels, centres
+
+
+def silhouette(vectors: np.ndarray, labels: np.ndarray) -> float:
+ """Mean silhouette, used only to pick k. A bucket count chosen by hand is a guess that ages
+ badly as the library grows."""
+ unique = np.unique(labels)
+ if len(unique) < 2:
+ return -1.0
+ dist = np.sqrt(np.maximum(((vectors[:, None, :] - vectors[None]) ** 2).sum(-1), 0))
+ scores = []
+ for i in range(len(vectors)):
+ same = labels == labels[i]
+ same[i] = False
+ if not same.any():
+ continue # a singleton has no cohesion to measure
+ a = dist[i][same].mean()
+ b = min(dist[i][labels == other].mean() for other in unique if other != labels[i])
+ scores.append((b - a) / max(a, b))
+ return float(np.mean(scores)) if scores else -1.0
+
+
+def project(vectors: np.ndarray, dims: int) -> np.ndarray:
+ """Top `dims` principal components."""
+ centred = vectors - vectors.mean(0, keepdims=True)
+ _, _, vt = np.linalg.svd(centred, full_matrices=False)
+ return centred @ vt[:dims].T
+
+
+def project_2d(vectors: np.ndarray) -> np.ndarray:
+ return project(vectors, 2)
+
+
+def top_terms(texts: list[str], members: list[int], k: int = 6) -> list[str]:
+ """Terms frequent inside the bucket and rare outside it. Raw frequency returns the words every
+ skill uses, which names nothing."""
+ inside, outside = Counter(), Counter()
+ member_set = set(members)
+ for i, text in enumerate(texts):
+ words = {w for w in _WORD.findall(text.lower()) if w not in _STOP}
+ (inside if i in member_set else outside).update(words)
+ n_in, n_out = max(len(members), 1), max(len(texts) - len(members), 1)
+ scored = {w: (c / n_in) - (outside[w] / n_out) for w, c in inside.items() if c > 1 or n_in < 3}
+ return [w for w, _ in sorted(scored.items(), key=lambda kv: -kv[1])[:k]]
+
+
+def label_for(terms: list[str]) -> str:
+ return " · ".join(terms[:3]) if terms else "unlabelled"
+
+
+def build(log=print) -> dict:
+ from ingot.mcp_server.embedding import EMBED_MODEL, build_embedding
+ from ingot.mcp_server.registry import load_skills
+
+ skills = load_skills()
+ if len(skills) < K_MIN:
+ raise SystemExit(f"only {len(skills)} skills indexed; clustering needs at least {K_MIN}. "
+ f"Check SKILL_ROUTER_PATHS.")
+ texts = skill_texts(skills)
+ log(f"[cluster] embedding {len(skills)} skills with {EMBED_MODEL}…")
+ vectors = _normalise(np.array([np.asarray(v, dtype=float)
+ for v in build_embedding().embed(texts)]))
+
+ # Cluster in the reduced space, not the raw one. At 1024 dimensions over a hundred points every
+ # pair sits at a similar distance (measured on this library: cosine p25 0.42, median 0.51,
+ # p75 0.60) and k-means has nothing to bite on — it scored silhouette +0.05 and put half the
+ # library in one bucket. The same run over the top 10 components scores +0.28. This is what UMAP
+ # is for in the version of this view that runs over 100k prompts; at this size the projection
+ # the layout already needs is enough, taken a few more components deep.
+ reduced = _normalise(project(vectors, min(CLUSTER_DIMS, len(skills) - 1)))
+
+ upper = min(K_MAX, len(skills) // 2)
+ best = None
+ for k in range(K_MIN, max(K_MIN, upper) + 1):
+ labels, centres = kmeans(reduced, k)
+ score = silhouette(reduced, labels)
+ log(f"[cluster] k={k:<3} silhouette {score:+.3f}")
+ if best is None or score > best[0]:
+ best = (score, k, labels, centres)
+ score, k, labels, centres = best
+ log(f"[cluster] chose k={k} (silhouette {score:+.3f})")
+
+ xy = project_2d(vectors)
+ clusters = []
+ for j in range(k):
+ members = [i for i in range(len(skills)) if labels[i] == j]
+ if not members:
+ continue
+ terms = top_terms(texts, members)
+ clusters.append({
+ "id": j, "label": label_for(terms), "size": len(members), "top_terms": terms,
+ "centroid_2d": [float(xy[members, 0].mean()), float(xy[members, 1].mean())],
+ "members": sorted(({"name": skills[i].name,
+ "xy": [float(xy[i, 0]), float(xy[i, 1])],
+ "provenance": (skills[i].metadata or {}).get("provenance", "")}
+ for i in members), key=lambda m: m["name"]),
+ })
+ clusters.sort(key=lambda c: -c["size"])
+ return {"version": 1, "n_skills": len(skills), "k": k, "silhouette": round(score, 4),
+ "embedder": EMBED_MODEL, "clusterer": f"cosine k-means (k={k}, seeded)",
+ "reducer": f"pca-{CLUSTER_DIMS}d cluster / pca-2d layout", "clusters": clusters}
+
+
+def build_and_save(log=print) -> Path:
+ data = build(log=log)
+ CLUSTER_PATH.parent.mkdir(parents=True, exist_ok=True)
+ CLUSTER_PATH.write_text(json.dumps(data, indent=2))
+ log(f"[cluster] {data['k']} buckets over {data['n_skills']} skills → {CLUSTER_PATH}")
+ for c in data["clusters"]:
+ log(f" {c['size']:>3} {c['label']}")
+ return CLUSTER_PATH
+
+
+if __name__ == "__main__":
+ os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
+ build_and_save()
diff --git a/ingot/optimize/compat.py b/ingot/optimize/compat.py
new file mode 100644
index 0000000..76fabdc
--- /dev/null
+++ b/ingot/optimize/compat.py
@@ -0,0 +1,155 @@
+"""Cross-model skill compatibility, how well a skill's body transfers across serving models.
+
+A skill body is tuned for one serving model (`AGENT_MODEL`); SkillOpt's own result is that good
+skills transfer, but not always. For each model in `COMPAT_MODELS`, this runs the skill's held-out
+tasks through the one serving contract twice, once with the skill body, once with an empty body
+(the no-skill baseline), judges both with the FIXED judge, and reports per-model **lift**
+(skill mean − baseline mean). Positive lift = the body helps that model; ~0 = the model already
+knows this and the body is dead weight there.
+
+Langfuse-free: it reuses the direct rollout + judge (the same path the inner loop uses), so it needs
+no trace backend or experiment logging. Only the *serving* model varies, the judge stays fixed so
+scores are comparable across models.
+
+Usage: python -m ingot.optimize.compat
+Config: COMPAT_MODELS=qwen/qwen3-32b,openai/gpt-5.5,anthropic/claude-sonnet-... (default: AGENT_MODEL)
+"""
+import hashlib
+import json
+import os
+import statistics
+from concurrent.futures import ThreadPoolExecutor
+from pathlib import Path
+
+from langchain_openai import ChatOpenAI
+
+from ingot.mcp_server.registry import optimizable_components
+
+from . import (SERVE_TEMPLATE, agent_model, api_key, client_kwargs, configured_models,
+ model_api_key, model_base_url, resolve_skill_dir, teacher_base_url)
+from . import usage as usage_ledger
+from .ab import load_tasks
+from .judge import invoke_retry, judge
+from .rollout import assemble
+from ingot import paths
+
+_MAX_WORKERS = 8
+COMPAT_DIR = paths.runs() / "compat"
+BASELINE_CACHE_DIR = paths.runs() / "compat-baseline"
+# The no-skill baseline: the identical serving contract with no skill body, so `lift` isolates the
+# body's contribution rather than the difference between two different prompts.
+NO_SKILL_BODY = "(no skill loaded, answer the task from your own knowledge)"
+
+
+def compat_models() -> list[str]:
+ """Models to sweep: COMPAT_MODELS (comma-separated), else just the configured AGENT_MODEL."""
+ models = configured_models("COMPAT_MODELS")
+ return models or [agent_model()]
+
+
+def _llm(model: str):
+ # Route each row to the endpoint that actually serves it. The model this box serves
+ # (AGENT_MODEL) comes from MODEL_BASE_URL — a local vLLM, and therefore a free row in the grid;
+ # every other slug goes to the hosted endpoint. Sending the whole sweep to one endpoint meant
+ # either the local row was impossible, or MODEL_BASE_URL had to be overridden by hand for the
+ # run, which loses the free reference row exactly when you want to compare against it.
+ # reasoning is left at the provider default, some models reject the flag.
+ if model == agent_model():
+ base, key = model_base_url(), model_api_key()
+ else:
+ base, key = teacher_base_url(), api_key()
+ return ChatOpenAI(model=model, temperature=0, **client_kwargs(base, key=key))
+
+
+def _baseline_cache_key(model: str, holdout: list[dict]) -> str:
+ """The no-skill baseline is a pure function of (serving model, held-out tasks, judge): the skill
+ body is precisely what it leaves out, so editing the skill cannot change it. Uncached, every
+ re-sweep paid again for byte-identical work — half the cost of every run after the first."""
+ payload = json.dumps({"v": 1, "model": model, "holdout": holdout,
+ "judge": os.environ.get("JUDGE_MODELS") or os.environ.get("JUDGE_MODEL", "")},
+ sort_keys=True)
+ return hashlib.sha256(payload.encode()).hexdigest()[:16]
+
+
+def _score(llm, system: str, task: dict, role: str = "compat") -> float:
+ msg = invoke_retry(llm, [("system", system), ("user", task["task"])])
+ usage_ledger.add(role, getattr(msg, "usage_metadata", None))
+ return judge(task["task"], task["rubric"], msg.content,
+ check=task.get("check"), deliverable=task.get("deliverable"),
+ checklist=task.get("checklist"))["score"]
+
+
+def _run_arm(llm, system: str, tasks: list[dict], role: str = "compat") -> list[float]:
+ with ThreadPoolExecutor(max_workers=min(_MAX_WORKERS, len(tasks))) as pool:
+ return list(pool.map(lambda t: _score(llm, system, t, role), tasks))
+
+
+def _sweep_model(model: str, skill_system: str, base_system: str, holdout: list[dict],
+ log=print) -> dict:
+ """One row of the matrix: this model with the skill body, and without it."""
+ llm = _llm(model)
+ # Bill each arm to its own model: one "compat" bucket cannot be priced, because the whole
+ # point of the sweep is that the serving model changes underneath it.
+ role = f"compat:{model}"
+ skill_scores = _run_arm(llm, skill_system, holdout, role)
+ cache_path = BASELINE_CACHE_DIR / f"{_baseline_cache_key(model, holdout)}.json"
+ if cache_path.exists():
+ base_scores = json.loads(cache_path.read_text())
+ log(f"[compat] {model:<34} baseline reused from cache (no spend)")
+ else:
+ base_scores = _run_arm(llm, base_system, holdout, role)
+ BASELINE_CACHE_DIR.mkdir(parents=True, exist_ok=True)
+ cache_path.write_text(json.dumps(base_scores))
+ s_mean, b_mean = statistics.mean(skill_scores), statistics.mean(base_scores)
+ verdict = "helps" if s_mean - b_mean > 0.05 else "no lift" if s_mean - b_mean >= -0.05 else "HURTS"
+ log(f"[compat] {model:<34} skill {s_mean:.3f} baseline {b_mean:.3f} "
+ f"lift {s_mean - b_mean:+.3f} ({verdict})")
+ return {"skill_mean": s_mean, "baseline_mean": b_mean, "lift": s_mean - b_mean,
+ "skill_scores": skill_scores, "baseline_scores": base_scores}
+
+
+def run_compat(skill: str, log=print) -> dict:
+ """Sweep COMPAT_MODELS over the skill's held-out tasks (skill vs no-skill) and write the matrix
+ to runs/compat/.json. Returns the summary."""
+ usage_ledger.reset()
+ skill_dir = resolve_skill_dir(skill)
+ _, holdout, _ = load_tasks(skill)
+ if not holdout:
+ raise SystemExit(f"'{skill}' has no held-out eval tasks to run.")
+ skill_system = SERVE_TEMPLATE.format(body=assemble(optimizable_components(skill_dir)))
+ base_system = SERVE_TEMPLATE.format(body=NO_SKILL_BODY)
+ models = compat_models()
+ log(f"[compat] '{skill}': {len(holdout)} held-out tasks × {len(models)} model(s); "
+ f"judge fixed, serving model varies")
+
+ models_out = {}
+ for model in models:
+ # One model the endpoint cannot serve must not discard the rows already paid for. A slug
+ # with no ZDR-qualified endpoint 404s on the first call, and before this the whole sweep
+ # died there — losing every earlier model's scores and writing no matrix at all.
+ try:
+ models_out[model] = _sweep_model(model, skill_system, base_system, holdout, log)
+ except Exception as error: # noqa: BLE001 - any provider failure is one unusable row
+ models_out[model] = {"error": f"{type(error).__name__}: {error}"[:400]}
+ log(f"[compat] {model:<34} UNAVAILABLE ({type(error).__name__}), skipped")
+ if not any("error" not in row for row in models_out.values()):
+ raise SystemExit(f"[compat] no model in COMPAT_MODELS could be reached for '{skill}'; "
+ f"nothing was measured.")
+
+ summary = {"skill": skill, "tasks": len(holdout),
+ "judge": os.environ.get("JUDGE_MODELS") or os.environ.get("JUDGE_MODEL", ""),
+ "models": models_out, "usage": usage_ledger.report()}
+ COMPAT_DIR.mkdir(parents=True, exist_ok=True)
+ path = COMPAT_DIR / f"{skill}.json"
+ path.write_text(json.dumps(summary, indent=2))
+ log(f"[compat] matrix written to {path}")
+ log(usage_ledger.format_report())
+ return summary
+
+
+if __name__ == "__main__":
+ import sys
+
+ from . import require_openrouter_key
+ require_openrouter_key()
+ run_compat(sys.argv[1] if len(sys.argv) > 1 else "tailwind")
diff --git a/optimize/draft.py b/ingot/optimize/draft.py
similarity index 62%
rename from optimize/draft.py
rename to ingot/optimize/draft.py
index 59c68c2..5528baa 100644
--- a/optimize/draft.py
+++ b/ingot/optimize/draft.py
@@ -2,7 +2,7 @@
immediately optimizable. The authoring model (SKILLOPT_MODEL) reads the skill's description + body and
writes train/holdout tasks with judge rubrics, split by *operation* so the holdout tests
generalization, not recall. Kept out of the MCP server (no LLM in its hot serving path); the
-optimizer calls it on demand when `optimize/tasks/.yaml` is missing."""
+optimizer calls it on demand when `ingot/optimize/tasks/.yaml` is missing."""
import json
import os
import re
@@ -27,21 +27,64 @@
LLM judge (what a correct answer must contain). Phrase tasks so the answer is the deliverable itself
(e.g. "Write Python code that…"), not a request to go find files.
-Return ONLY JSON: {{"tasks": [{{"task": "...", "rubric": "..."}}, ...]}} with exactly {n} items,
-each covering a different operation/capability."""
+Each task also carries a CHECKLIST of {items} independent checks that decide its score. Write checks
+a grader can answer without re-reading the whole answer, and that a good and a bad answer would
+genuinely split on:
+
+- Each check tests ONE observable property. "Handles the empty input case" is a check; "is high
+ quality" is not.
+- Make them specific to THIS task, not generic writing advice. Prefer things the skill body says
+ matter.
+- id: short snake_case, unique within the task. weight: 1 (minor) to 5 (the point of the task).
+- dimension: one of correctness, completeness, instruction_following, efficiency.
+
+Return ONLY JSON with exactly {n} items, each covering a different operation/capability:
+{{"tasks": [{{"task": "...", "rubric": "...",
+ "checklist": [{{"id": "...", "criterion": "...", "weight": 3,
+ "dimension": "correctness"}}, ...]}}, ...]}}"""
def _llm():
return ChatOpenAI(model=MODEL, temperature=0.4, **client_kwargs(teacher_base_url()))
-def draft_tasks(name: str, description: str, body: str, n: int = 8) -> dict:
- """Draft n tasks and split them evenly into train/holdout (disjoint operations)."""
- msg = _llm().invoke(_PROMPT.format(name=name, description=description, body=body[:6000], n=n))
+_ID_RE = re.compile(r"^[a-z][a-z0-9_]{1,39}$")
+
+
+def _clean_checklist(raw) -> list[dict]:
+ """Keep only checks a grader can actually apply, and drop the rest rather than shipping a
+ rubric with unusable items in it. An empty result is fine: judge() falls back to its default
+ checklist, which still grades four dimensions independently."""
+ from .judge import DIMENSIONS
+ out, seen = [], set()
+ for item in raw if isinstance(raw, list) else []:
+ if not isinstance(item, dict):
+ continue
+ item_id, criterion = str(item.get("id", "")).strip().lower(), str(item.get("criterion", "")).strip()
+ if not _ID_RE.match(item_id) or item_id in seen or len(criterion) < 8:
+ continue
+ try:
+ weight = min(5, max(1, int(item.get("weight", 1))))
+ except (TypeError, ValueError):
+ weight = 1
+ dimension = str(item.get("dimension", "")).strip().lower()
+ seen.add(item_id)
+ out.append({"id": item_id, "criterion": criterion, "weight": weight,
+ "dimension": dimension if dimension in DIMENSIONS else "correctness"})
+ return out
+
+
+def draft_tasks(name: str, description: str, body: str, n: int = 8, items: int = 6) -> dict:
+ """Draft n tasks, each with its own weighted checklist, split evenly into train/holdout
+ (disjoint operations)."""
+ msg = _llm().invoke(_PROMPT.format(name=name, description=description, body=body[:6000],
+ n=n, items=items))
usage_ledger.add("draft", getattr(msg, "usage_metadata", None))
m = re.search(r"\{.*\}", msg.content, re.DOTALL)
tasks = (json.loads(m.group(0)) if m else {}).get("tasks", [])
- tasks = [{"task": str(t["task"]), "rubric": str(t.get("rubric", ""))} for t in tasks if t.get("task")]
+ tasks = [{"task": str(t["task"]), "rubric": str(t.get("rubric", "")),
+ "checklist": _clean_checklist(t.get("checklist"))}
+ for t in tasks if t.get("task")]
if len(tasks) < 4:
raise SystemExit(f"draft_tasks: teacher returned only {len(tasks)} usable tasks for '{name}'.")
half = len(tasks) // 2
@@ -55,7 +98,13 @@ def draft_and_save(name: str, description: str, body: str, tasks_dir, n: int = 8
data = draft_tasks(name, description, body, n=n)
path = Path(tasks_dir) / f"{name}.yaml"
path.write_text(yaml.safe_dump(data, sort_keys=False, allow_unicode=True, width=100000))
- log(f"[draft] wrote {len(data['train'])} train + {len(data['holdout'])} holdout tasks → {path}")
+ checks = sum(len(t["checklist"]) for t in data["train"] + data["holdout"])
+ bare = [t for t in data["train"] + data["holdout"] if not t["checklist"]]
+ log(f"[draft] wrote {len(data['train'])} train + {len(data['holdout'])} holdout tasks, "
+ f"{checks} graded checks → {path}")
+ if bare: # silently falling back to the default checklist would read as a richer set than it is
+ log(f"[draft] {len(bare)} task(s) got no usable checklist and will grade on the default "
+ f"four dimensions; edit {path} to add checks that matter for them.")
return path
@@ -89,7 +138,7 @@ def draft_routing_cases(name: str, description: str, body: str,
if len(pos) < 2 or len(neg) < 1:
raise SystemExit(f"draft_routing_cases: teacher returned {len(pos)} positive / {len(neg)} "
f"negative cases for '{name}', need at least 2/1. Re-run or hand-write "
- f"a routing: block in optimize/tasks/{name}.yaml.")
+ f"a routing: block in ingot/optimize/tasks/{name}.yaml.")
cases = []
for i, task in enumerate(pos):
case = {"task": task, "expected": name, "harness": "claude" if i == 1 else "codex"}
diff --git a/optimize/evidence.py b/ingot/optimize/evidence.py
similarity index 92%
rename from optimize/evidence.py
rename to ingot/optimize/evidence.py
index 5070c8c..009d3fe 100644
--- a/optimize/evidence.py
+++ b/ingot/optimize/evidence.py
@@ -4,21 +4,28 @@
import json
from dataclasses import dataclass
from pathlib import Path
+from ingot import paths
SCHEMA = "skill-router/evidence/v1"
ROUTING_SCHEMA = "skill-router/evidence/routing/v1"
-_REPO_ROOT = Path(__file__).resolve().parent.parent
+_PACKAGE_ROOT = Path(__file__).resolve().parent.parent
def recorded_path(path: Path) -> str:
- """How an evidence location is written into a pending record: relative to the repo root.
- A bundle written inside a container is then still resolvable from the host checkout, and the
- review surface has a path it can contain to runs/evidence."""
- try:
- return path.resolve().relative_to(_REPO_ROOT).as_posix()
- except ValueError:
- return str(path)
+ """How an evidence location is written into a pending record: relative to the state root.
+
+ A bundle written inside a container is then still resolvable from the host, and the review
+ surface has a path it can contain to runs/evidence. The package root is tried second so a
+ record written before state moved out of the code directory still reads as `runs/evidence/...`
+ rather than an absolute path from someone else's machine."""
+ resolved = path.resolve()
+ for anchor in (paths.runs().parent, _PACKAGE_ROOT):
+ try:
+ return resolved.relative_to(anchor).as_posix()
+ except ValueError:
+ continue
+ return str(resolved)
def first_divergence(champion: list[dict], challenger: list[dict]) -> dict | None:
diff --git a/optimize/execcheck.py b/ingot/optimize/execcheck.py
similarity index 99%
rename from optimize/execcheck.py
rename to ingot/optimize/execcheck.py
index 271bee1..e1ad339 100644
--- a/optimize/execcheck.py
+++ b/ingot/optimize/execcheck.py
@@ -92,7 +92,7 @@ def _sandbox(spec: dict, timeout: int) -> dict | None:
"--env", "PYTHONPATH=/app"]
if SANDBOX_RUNTIME:
cmd += ["--runtime", SANDBOX_RUNTIME]
- cmd += [SANDBOX_IMAGE, "python", "-m", "optimize.sandbox_driver"]
+ cmd += [SANDBOX_IMAGE, "python", "-m", "ingot.optimize.sandbox_driver"]
try:
run = subprocess.run(cmd, input=json.dumps({**spec, "timeout": timeout}),
capture_output=True, text=True, timeout=timeout * 3 + 30)
diff --git a/ingot/optimize/harbor-shared-network.compose.yml b/ingot/optimize/harbor-shared-network.compose.yml
new file mode 100644
index 0000000..ac1ae0f
--- /dev/null
+++ b/ingot/optimize/harbor-shared-network.compose.yml
@@ -0,0 +1,4 @@
+networks:
+ default:
+ external: true
+ name: ingot-harbor-trials
diff --git a/ingot/optimize/harbor_catalog.py b/ingot/optimize/harbor_catalog.py
new file mode 100644
index 0000000..54e26d4
--- /dev/null
+++ b/ingot/optimize/harbor_catalog.py
@@ -0,0 +1,386 @@
+"""Durable skill-catalog caller for :func:`ingot.optimize.harbor_eval.run_local_sweep`.
+
+One controller owns this filesystem queue. Each item invokes the restart-safe native Harbor
+one-skill path; Harbor trial, telemetry, grade, and publication receipts remain the source of truth.
+The catalog files schedule those calls and never duplicate their evidence state.
+"""
+from __future__ import annotations
+
+import argparse
+import fcntl
+import hashlib
+import json
+import os
+import shutil
+import subprocess
+import time
+from dataclasses import asdict, dataclass
+from pathlib import Path
+from typing import Iterable, Mapping, Sequence
+
+import yaml
+
+from ingot.mcp_server.registry import load_skills, skill_revision
+from ingot.optimize import resolve_skill_dir
+from ingot.optimize.ab import TASKS_DIR
+from ingot.optimize.harbor_eval import HARBOR_DIR, SCORING_REVISION, _task_fingerprint, run_local_sweep
+from ingot.optimize.harbor_native import NATIVE_RUNNER_REVISION, native_trial_identity
+from ingot.optimize.harbor_targets import HARNESS_PROTOCOLS, LocalTarget, discover_target, parse_target
+
+
+_SCHEMA = 1
+CATALOG_OWNER = HARBOR_DIR / "catalog.controller.lock"
+
+
+def _atomic_json(path: Path, payload: dict) -> None:
+ path.parent.mkdir(parents=True, exist_ok=True)
+ temporary = path.with_name(f".{path.name}.{os.getpid()}.tmp")
+ try:
+ with temporary.open("w", encoding="utf-8") as handle:
+ json.dump(payload, handle, sort_keys=True, separators=(",", ":"))
+ handle.write("\n")
+ handle.flush()
+ os.fsync(handle.fileno())
+ os.replace(temporary, path)
+ finally:
+ temporary.unlink(missing_ok=True)
+
+
+@dataclass(frozen=True)
+class CatalogIntent:
+ skill: str
+ skill_sha256: str
+ task_fingerprint: str
+ target_specs: tuple[str, ...]
+ target_fingerprints: tuple[str, ...]
+ harnesses: tuple[str, ...]
+ runtime_revisions: tuple[tuple[str, str], ...]
+ publish_root: str = str(HARBOR_DIR)
+ global_concurrency: int = 16
+ endpoint_concurrency: int = 2
+ priority: int = 100
+
+ def __post_init__(self) -> None:
+ if not self.skill or len(self.skill_sha256) != 64 or len(self.task_fingerprint) != 64:
+ raise ValueError("catalog intent requires skill and full SHA-256 identities")
+ if not self.target_specs or len(self.target_specs) != len(self.target_fingerprints):
+ raise ValueError("catalog intent target specs and fingerprints must align")
+ if not self.harnesses or any(item not in HARNESS_PROTOCOLS for item in self.harnesses):
+ raise ValueError("catalog intent contains no harnesses or an unknown harness")
+ if len(set(self.target_specs)) != len(self.target_specs) or len(set(self.harnesses)) != len(self.harnesses):
+ raise ValueError("catalog intent contains duplicate targets or harnesses")
+ if (self.global_concurrency < 1 or self.endpoint_concurrency < 1
+ or self.endpoint_concurrency > self.global_concurrency):
+ raise ValueError("catalog concurrency requires 1 <= endpoint <= global")
+ if not self.publish_root or not Path(self.publish_root).is_absolute():
+ raise ValueError("catalog publish root must be an absolute path")
+
+ def identity_payload(self) -> dict:
+ payload = asdict(self)
+ payload.pop("priority")
+ payload["target_specs"] = list(self.target_specs)
+ payload["target_fingerprints"] = list(self.target_fingerprints)
+ payload["harnesses"] = list(self.harnesses)
+ payload["runtime_revisions"] = [list(item) for item in self.runtime_revisions]
+ return payload
+
+ @property
+ def digest(self) -> str:
+ encoded = json.dumps(self.identity_payload(), sort_keys=True,
+ separators=(",", ":")).encode()
+ return hashlib.sha256(encoded).hexdigest()
+
+
+def _heldout(skill: str) -> list[dict] | None:
+ path = TASKS_DIR / f"{skill}.yaml"
+ if not path.is_file():
+ return None
+ data = yaml.safe_load(path.read_text())
+ if not isinstance(data, dict):
+ return None
+ holdout = data.get("holdout") or data.get("train") or data.get("tasks") or []
+ return holdout if isinstance(holdout, list) and holdout else None
+
+
+def _skill_sha(skill: str) -> str:
+ return hashlib.sha256((resolve_skill_dir(skill) / "SKILL.md").read_bytes()).hexdigest()
+
+
+def _intent_for_skill(skill: str, target_specs: Sequence[str], harnesses: Sequence[str],
+ *, priority: int = 100, global_concurrency: int = 16,
+ endpoint_concurrency: int = 2,
+ publish_root: Path | str = HARBOR_DIR) -> CatalogIntent | None:
+ provisional = tuple(parse_target(spec) for spec in target_specs)
+ targets = tuple(discover_target(target.alias, target.base_url) for target in provisional)
+ holdout = _heldout(skill)
+ if not holdout:
+ return None
+ source = resolve_skill_dir(skill)
+ revisions = [("harbor", "0.20.0"), ("runner", NATIVE_RUNNER_REVISION),
+ ("scoring", SCORING_REVISION), ("skill-tree", skill_revision(source))]
+ revisions.extend(
+ (f"route:{target.fingerprint}:{harness}",
+ f"{native_trial_identity(target, harness, 'skill').protocol}/"
+ f"{native_trial_identity(target, harness, 'skill').gateway_revision}/"
+ f"context={target.context_length}")
+ for target in targets for harness in harnesses
+ )
+ return CatalogIntent(
+ skill=skill,
+ skill_sha256=_skill_sha(skill),
+ task_fingerprint=_task_fingerprint(holdout),
+ target_specs=tuple(target_specs),
+ target_fingerprints=tuple(target.fingerprint for target in targets),
+ harnesses=tuple(harnesses),
+ runtime_revisions=tuple(revisions),
+ publish_root=str(Path(publish_root)),
+ global_concurrency=global_concurrency,
+ endpoint_concurrency=endpoint_concurrency,
+ priority=priority,
+ )
+
+
+def _intent_document(intent: CatalogIntent) -> dict:
+ return {"schema": _SCHEMA, "digest": intent.digest, "identity": intent.identity_payload()}
+
+
+def _state_path(root: Path, digest: str) -> Path:
+ return root / "state" / f"{digest}.json"
+
+
+def _read_json(path: Path) -> dict:
+ value = json.loads(path.read_text())
+ if not isinstance(value, dict):
+ raise RuntimeError(f"catalog record is not an object: {path}")
+ return value
+
+
+def enqueue_catalog(root: Path | str, intents: Iterable[CatalogIntent]) -> list[Path]:
+ """Persist measurement intents; scheduling priority never changes content identity."""
+ root = Path(root)
+ intent_dir = root / "intents"
+ known = []
+ if intent_dir.is_dir():
+ known = [_read_json(path) for path in intent_dir.glob("*.json")]
+ written: list[Path] = []
+ for supplied in intents:
+ changed = any(item.get("identity", {}).get("skill") == supplied.skill
+ and item.get("identity", {}).get("skill_sha256") != supplied.skill_sha256
+ for item in known)
+ path = intent_dir / f"{supplied.digest}.json"
+ state_path = _state_path(root, supplied.digest)
+ if not path.exists():
+ _atomic_json(path, _intent_document(supplied))
+ known.append(_intent_document(supplied))
+ if state_path.exists():
+ state = _read_json(state_path)
+ if state.get("status") != "complete":
+ if state.get("status") == "superseded":
+ state["status"] = "pending"
+ state.pop("error", None)
+ state.pop("finished_at", None)
+ state["priority"] = max(int(state.get("priority", 0)),
+ 300 if changed else 200)
+ _atomic_json(state_path, state)
+ else:
+ _atomic_json(state_path, {"schema": _SCHEMA, "intent_digest": supplied.digest,
+ "status": "pending",
+ "priority": max(supplied.priority, 300 if changed else 100)})
+ written.append(path)
+ return written
+
+
+def _process_start_token(pid: int) -> str | None:
+ try:
+ return Path(f"/proc/{pid}/stat").read_text().split()[21]
+ except (OSError, IndexError):
+ done = subprocess.run(["ps", "-o", "lstart=", "-p", str(pid)], capture_output=True,
+ text=True)
+ return done.stdout.strip() or None
+
+
+def _claim_controller(path: Path):
+ """Hold one kernel lock across all catalog roots; no stale-receipt unlink race exists."""
+ path.parent.mkdir(parents=True, exist_ok=True)
+ handle = path.open("a+", encoding="utf-8")
+ try:
+ fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
+ except BlockingIOError as error:
+ handle.seek(0)
+ try:
+ current = json.load(handle)
+ except (ValueError, OSError):
+ current = {}
+ handle.close()
+ raise RuntimeError(f"catalog already has live controller PID {current.get('pid', 'unknown')}") from error
+ handle.seek(0)
+ handle.truncate()
+ json.dump({"schema": _SCHEMA, "pid": os.getpid(),
+ "start_token": _process_start_token(os.getpid())}, handle, sort_keys=True)
+ handle.flush()
+ os.fsync(handle.fileno())
+ return handle
+
+
+def _load_intent(path: Path, priority: int) -> CatalogIntent:
+ document = _read_json(path)
+ identity = document.get("identity")
+ if not isinstance(identity, dict) or document.get("digest") != path.stem:
+ raise RuntimeError(f"catalog intent identity is invalid: {path}")
+ revisions = tuple(tuple(item) for item in identity["runtime_revisions"])
+ intent = CatalogIntent(**{**identity, "target_specs": tuple(identity["target_specs"]),
+ "target_fingerprints": tuple(identity["target_fingerprints"]),
+ "harnesses": tuple(identity["harnesses"]),
+ "runtime_revisions": revisions, "priority": priority})
+ if intent.digest != path.stem or document.get("digest") != intent.digest:
+ raise RuntimeError(f"catalog intent digest is invalid: {path}")
+ return intent
+
+
+def _prepare_execution(root: Path, intent: CatalogIntent) -> tuple[Path, Path]:
+ execution_root = root / "runs" / intent.digest
+ source = execution_root / "staged"
+ staged_skill = source / intent.skill
+ revisions = dict(intent.runtime_revisions)
+ expected_tree = revisions.get("skill-tree")
+ if not expected_tree:
+ raise RuntimeError("catalog intent lacks full skill-tree revision")
+ if not staged_skill.exists():
+ execution_root.mkdir(parents=True, exist_ok=True)
+ temporary = execution_root / f".staged.{os.getpid()}.tmp"
+ shutil.rmtree(temporary, ignore_errors=True)
+ try:
+ copied = temporary / intent.skill
+ shutil.copytree(resolve_skill_dir(intent.skill), copied)
+ if skill_revision(copied) != expected_tree:
+ raise RuntimeError(f"catalog skill tree changed while staging: {intent.skill}")
+ os.replace(temporary, source)
+ finally:
+ shutil.rmtree(temporary, ignore_errors=True)
+ if skill_revision(staged_skill) != expected_tree:
+ raise RuntimeError(f"catalog staged skill tree identity changed: {intent.skill}")
+ return source, execution_root
+
+
+def _refuse_live_harbor(execution_root: Path) -> None:
+ """Do not launch probes/canaries over a surviving Harbor child for this exact intent."""
+ listing = subprocess.run(["ps", "-axo", "pid=,args="], capture_output=True, text=True,
+ check=True).stdout
+ marker = str(execution_root)
+ live = []
+ for line in listing.splitlines():
+ pid, separator, command = line.strip().partition(" ")
+ if separator and pid.isdigit() and marker in command and "harbor run" in command:
+ live.append(pid)
+ if live:
+ raise RuntimeError(f"catalog intent already has live Harbor child PID {','.join(live)}")
+
+
+def run_catalog(root: Path | str, *, max_skills: int | None = None,
+ stop_file: Path | str | None = None, controller_owner=None,
+ process_env: Mapping[str, str] | None = None) -> None:
+ """Run queued skills serially; native Harbor owns all within-skill parallelism."""
+ root = Path(root)
+ root.mkdir(parents=True, exist_ok=True)
+ stop = Path(stop_file) if stop_file is not None else root / "STOP"
+ owner = controller_owner or _claim_controller(CATALOG_OWNER)
+ owns_controller = controller_owner is None
+ attempted = 0
+ try:
+ candidates = []
+ for state_path in (root / "state").glob("*.json") if (root / "state").is_dir() else ():
+ state = _read_json(state_path)
+ if state_path.stem != state.get("intent_digest"):
+ raise RuntimeError(f"catalog state digest is invalid: {state_path}")
+ if state.get("status") in {"pending", "running", "failed"}:
+ candidates.append((int(state.get("priority", 0)), state_path, state))
+ for _, state_path, state in sorted(candidates, key=lambda item: (-item[0], item[1].name)):
+ if stop.exists() or (max_skills is not None and attempted >= max_skills):
+ break
+ digest = state_path.stem
+ intent = _load_intent(root / "intents" / f"{digest}.json",
+ int(state.get("priority", 0)))
+ current = _intent_for_skill(intent.skill, intent.target_specs, intent.harnesses,
+ priority=300,
+ global_concurrency=intent.global_concurrency,
+ endpoint_concurrency=intent.endpoint_concurrency,
+ publish_root=intent.publish_root)
+ if current is None:
+ state.update(status="failed", finished_at=time.time(), error="MissingTasks")
+ _atomic_json(state_path, state)
+ attempted += 1
+ continue
+ if current.digest != intent.digest:
+ enqueue_catalog(root, [current])
+ state.update(status="superseded", finished_at=time.time(), error="IdentityChanged")
+ _atomic_json(state_path, state)
+ continue
+ source, execution_root = _prepare_execution(root, intent)
+ _refuse_live_harbor(execution_root)
+ state.update(status="running", started_at=state.get("started_at") or time.time(),
+ run_root=str(execution_root))
+ _atomic_json(state_path, state)
+ attempted += 1
+ try:
+ targets: list[LocalTarget] = [parse_target(spec) for spec in intent.target_specs]
+ manifest = run_local_sweep(
+ intent.skill, targets, harnesses=intent.harnesses, attempts=3,
+ native_parallel=True, skill_source=str(source), evidence_root=execution_root,
+ expected_task_fingerprint=intent.task_fingerprint,
+ expected_runtime_revisions=dict(intent.runtime_revisions),
+ global_concurrency=intent.global_concurrency,
+ endpoint_concurrency=intent.endpoint_concurrency,
+ publish_root=Path(intent.publish_root), content_addressed_resume=True,
+ process_env=process_env)
+ if manifest.get("aborted"):
+ raise RuntimeError("sweep aborted")
+ if manifest.get("telemetry_pending"):
+ raise RuntimeError("sweep telemetry is pending verification")
+ except Exception as error: # noqa: BLE001 - record one failed item, continue catalog
+ state.update(status="failed", finished_at=time.time(), error=type(error).__name__)
+ else:
+ state.update(status="complete", finished_at=time.time(),
+ utilization=manifest.get("utilization"),
+ combinations=len(manifest.get("combinations", {})))
+ _atomic_json(state_path, state)
+ finally:
+ if owns_controller:
+ fcntl.flock(owner.fileno(), fcntl.LOCK_UN)
+ owner.close()
+
+
+def _parser() -> argparse.ArgumentParser:
+ parser = argparse.ArgumentParser(description="Queue restart-safe native Harbor skill sweeps")
+ parser.add_argument("--root", type=Path, required=True)
+ selection = parser.add_mutually_exclusive_group(required=True)
+ selection.add_argument("--skill", action="append")
+ selection.add_argument("--all", action="store_true")
+ parser.add_argument("--target", action="append", required=True,
+ help="repeat ALIAS=BASE_URL")
+ parser.add_argument("--harness", action="append", choices=tuple(HARNESS_PROTOCOLS),
+ default=[])
+ parser.add_argument("--max-skills", type=int)
+ parser.add_argument("--global-concurrency", type=int, default=16)
+ parser.add_argument("--endpoint-concurrency", type=int, default=2)
+ parser.add_argument("--publish-root", type=Path, default=HARBOR_DIR,
+ help="absolute directory consumed by the UI for final matrices")
+ parser.add_argument("--enqueue-only", action="store_true")
+ return parser
+
+
+def main(argv: Sequence[str] | None = None) -> int:
+ args = _parser().parse_args(argv)
+ skills = [item.name for item in load_skills()] if args.all else args.skill
+ harnesses = tuple(args.harness or HARNESS_PROTOCOLS)
+ intents = [intent for skill in skills if (intent := _intent_for_skill(
+ skill, args.target, harnesses, global_concurrency=args.global_concurrency,
+ endpoint_concurrency=args.endpoint_concurrency,
+ publish_root=args.publish_root)) is not None]
+ enqueue_catalog(args.root, intents)
+ if not args.enqueue_only:
+ run_catalog(args.root, max_skills=args.max_skills)
+ return 0
+
+
+if __name__ == "__main__": # pragma: no cover - exercised through the installed module CLI
+ raise SystemExit(main())
diff --git a/ingot/optimize/harbor_codex_gateway.py b/ingot/optimize/harbor_codex_gateway.py
new file mode 100644
index 0000000..547d957
--- /dev/null
+++ b/ingot/optimize/harbor_codex_gateway.py
@@ -0,0 +1,22 @@
+"""Harbor Codex adapter variant for the local LiteLLM compatibility gateway."""
+from __future__ import annotations
+
+from harbor.agents.installed.codex import Codex
+
+from .harbor_gateway import codex_gateway_setup_command
+
+
+class GatewayCodex(Codex):
+ """Use HTTP Responses without the native-only reasoning parameter."""
+ # Harbor's parent Codex adapter defaults this flag to "high". LiteLLM custom_openai
+ # correctly rejects it for the local Qwen endpoint, while its remaining CLI flags still carry
+ # the normal tool-call behavior.
+ CLI_FLAGS = [flag for flag in Codex.CLI_FLAGS if flag.kwarg != "reasoning_effort"]
+
+ async def run(self, instruction, environment, context): # type: ignore[no-untyped-def]
+ await self.exec_as_agent(
+ environment,
+ command=codex_gateway_setup_command(),
+ env={"CODEX_HOME": self._REMOTE_CODEX_HOME.as_posix()},
+ )
+ await super().run(instruction, environment, context)
diff --git a/ingot/optimize/harbor_eval.py b/ingot/optimize/harbor_eval.py
new file mode 100644
index 0000000..813e695
--- /dev/null
+++ b/ingot/optimize/harbor_eval.py
@@ -0,0 +1,1567 @@
+"""Sandboxed cross-harness skill evaluation, built on Harbor.
+
+`compat.py` answers "does this skill body help this *model*", using one bare completion per task.
+This answers "does it help this *harness*" — the real CLI agent, with its own system prompt, tool
+loop and configured model, running unrestricted inside a fresh container that Harbor provisions,
+injects the agent into, and tears down.
+
+Harbor (https://github.com/harbor-framework/harbor, Apache-2.0) owns the parts that are not ours:
+the per-task container, ~30 CLI agent adapters (claude-code, codex, pi, goose, gemini-cli, aider,
+opencode, cursor-cli, openhands, …), concurrency, and trajectory capture. It also ships Terminus-2,
+a neutral harness that gives any model the same shell loop — which is the only way to vary the model
+without also varying the harness, since `claude` serves only Anthropic models and `codex` only
+OpenAI.
+
+What is ours is the experiment: a skill body is a *treatment*. Every task runs twice, once with the
+body injected into the harness's system prompt and once without it, and lift is the difference. The
+same fixed Ingot judge grades both arms, so a harness cannot flatter itself.
+
+The judge runs OUTSIDE the sandbox, over artifacts the verifier copies into `/logs/verifier/`.
+That keeps the judge prompt in one place instead of duplicated into every task image, and keeps the
+judge's API key out of a container that is running an agent in yolo mode.
+
+Usage: python -m ingot.optimize.harbor_eval --agent claude-code [--agent codex] [--model M]
+"""
+from __future__ import annotations
+
+import argparse
+from concurrent.futures import ThreadPoolExecutor
+import contextlib
+import hashlib
+import json
+import os
+import shlex
+import shutil
+import stat
+import subprocess
+import tempfile
+import threading
+import time
+from pathlib import Path
+from typing import Any, Mapping, Sequence
+
+from . import resolve_skill_dir
+from .ab import load_tasks
+from .harbor_targets import (LocalTarget, discover_target, harbor_agent_kwargs, harbor_model,
+ local_agent_env, parse_target, probe_chat_tool_round_trip,
+ probe_protocol, protocol_for,
+ scrub_provider_env)
+from .harbor_gateway import (GatewaySession, gateway_agent_env, gateway_agent_name, gateway_process_env,
+ gateway_metadata, gateway_route)
+from .harbor_langfuse import EXPORTER_REVISION, export_job_attempts
+from .harbor_native import (NativeCell, NativeTrialIdentity, compile_canary_job,
+ compile_measurement_job,
+ identity_env, identity_from_env, iter_attempt_dirs, native_trial_identity,
+ NATIVE_TRIAL_MEMORY_MB,
+ select_measurement_cells, write_job_config)
+from .harbor_redaction import _redact_harbor_receipt_output, _redact_persisted
+from .judge import judge
+from ingot import paths
+
+HARBOR_DIR = paths.runs() / "harbor"
+BUILD_DIR = HARBOR_DIR / "datasets"
+HARBOR_BIN = os.environ.get("HARBOR_BIN", "harbor")
+LOCAL_HARNESSES = (
+ "claude-code", "terminus-2", "goose", "opencode", "openclaw",
+ "mini-swe-agent", "codex", "aider", "pi",
+)
+
+# The agent works in a container, not a chat window, so the deliverable is a file it produced. This
+# is the whole reason to run in a sandbox rather than judge a completion: the harness has to
+# actually do the work.
+SOLUTION_DIR = "/app/solution"
+REPO_DIR = "/app/repo"
+_INSTRUCTION_SUFFIX = f"""
+
+---
+
+Write your complete deliverable into `{SOLUTION_DIR}/` (create the directory if it does not exist).
+Anything outside that directory is discarded and will not be graded.
+"""
+
+# A process skill has nothing to bite on in an empty container. "Verify in the execution context",
+# "feed the guard the input it must reject" and "wire it to the real caller" are all unanswerable
+# without existing code to read, run, and change — which is why the first task set could not
+# separate any arm from any other. A task may seed a working tree; the agent edits it in place.
+_SEEDED_SUFFIX = f"""
+
+---
+
+An existing project is checked out at `{REPO_DIR}/`. Work in it directly.
+
+When you are done, copy every file you changed or created, plus your evidence, into
+`{SOLUTION_DIR}/` (create it if needed), preserving the paths they have in the project.
+Only `{SOLUTION_DIR}/` is graded.
+"""
+
+# tmux and asciinema are here because terminus-2 installs them into the container itself when they
+# are missing, and that apt-get overran the 120s exec budget on a cold cache — surfacing as a bare
+# "RuntimeError: Command timed out after 120 seconds" from _install_recording_tools, with an empty
+# verifier directory, counted as a broken task and dropped. Its installer skips the work entirely
+# when both are already present.
+#
+# It is also a fairness fix. Harnesses differ in how much they install before they can start, and a
+# harness whose setup is heavier was losing whole tasks for it. That is a measurement of apt, not of
+# the harness. pytest is here for the same reason: the seeded tasks' own READMEs tell the agent to
+# run it, and Ubuntu 24.04 refuses `pip install` under PEP 668, so every agent would otherwise spend
+# its budget discovering that.
+_DOCKERFILE = """FROM ubuntu:24.04
+RUN apt-get update && apt-get install -y --no-install-recommends \\
+ python3 python3-pip python3-pytest git curl ca-certificates tmux asciinema \\
+ && rm -rf /var/lib/apt/lists/*
+WORKDIR /app
+"""
+
+_SEEDED_DOCKERFILE = _DOCKERFILE + f"""COPY seed/ {REPO_DIR}/
+RUN find {REPO_DIR} -name '*.sh' -exec chmod +x {{}} +
+"""
+
+# The verifier does not grade. It copies what the agent produced somewhere Harbor persists, and
+# always returns 1: a real reward here would be a second, unfixed grader competing with the Ingot
+# judge, and the two would disagree.
+_TEST_SH = f"""#!/bin/bash
+mkdir -p /logs/verifier/solution
+cp -r {SOLUTION_DIR}/. /logs/verifier/solution/ 2>/dev/null || true
+echo 1 > /logs/verifier/reward.txt
+"""
+
+# An agent's own evidence log is a claim about what it ran, and a claim is exactly what a skill that
+# rewards writing evidence logs teaches it to produce. `verify` runs the project's real check after
+# the agent is gone and captures the result, so the judge has one outcome signal from outside the
+# answer being graded. It is captured evidence, not the reward: the reward stays fixed at 1 so the
+# Ingot judge remains the only grader.
+_VERIFY_SH = """#!/bin/bash
+mkdir -p /logs/verifier/solution
+cp -r {solution}/. /logs/verifier/solution/ 2>/dev/null || true
+out=/logs/verifier/solution/_objective_check.txt
+{{
+ echo "Ran by the harness after the agent finished, in {repo}, not by the agent:"
+ printf ' $ %s\\n' {quoted}
+ echo "---"
+ cd {repo} 2>/dev/null && timeout 120 bash -lc {quoted}
+ echo "--- exit=$?"
+}} > "$out" 2>&1
+echo 1 > /logs/verifier/reward.txt
+"""
+
+
+def _task_name(skill: str, index: int) -> str:
+ return f"{skill}-h{index}"
+
+
+def _write_seed(files: dict | None, seed_dir: Path) -> bool:
+ """Write a task's seeded working tree under the image build context. True if anything was written.
+
+ Paths are confined to `seed_dir`: a task file is authored data, but `../..` in a key would write
+ outside the dataset and silently corrupt this checkout rather than the container's."""
+ if not isinstance(files, dict) or not files:
+ return False
+ seed_dir.mkdir(parents=True, exist_ok=True)
+ root = seed_dir.resolve()
+ for relative, content in files.items():
+ target = (seed_dir / str(relative)).resolve()
+ if not target.is_relative_to(root):
+ raise ValueError(f"seeded file path escapes the task directory: {relative!r}")
+ target.parent.mkdir(parents=True, exist_ok=True)
+ target.write_text(str(content), encoding="utf-8")
+ return True
+
+
+def build_dataset(skill: str, holdout: list[dict], out_dir: Path) -> Path:
+ """Write a Harbor dataset with one task per held-out task. Returns the dataset directory.
+
+ Rebuilt from scratch each time: a stale task left behind from an earlier holdout would be run
+ and scored as though it were part of this skill's current eval set."""
+ dataset = out_dir / skill
+ if dataset.exists():
+ shutil.rmtree(dataset)
+ (dataset).mkdir(parents=True)
+
+ entries = []
+ for index, task in enumerate(holdout):
+ name = _task_name(skill, index)
+ root = dataset / name
+ (root / "environment").mkdir(parents=True)
+ (root / "tests").mkdir(parents=True)
+ seeded = _write_seed(task.get("files"), root / "environment" / "seed")
+ (root / "instruction.md").write_text(
+ task["task"] + (_SEEDED_SUFFIX if seeded else _INSTRUCTION_SUFFIX), encoding="utf-8")
+ (root / "environment" / "Dockerfile").write_text(
+ _SEEDED_DOCKERFILE if seeded else _DOCKERFILE, encoding="utf-8")
+ test_sh = root / "tests" / "test.sh"
+ verify = str(task.get("verify") or "").strip()
+ test_sh.write_text(
+ _VERIFY_SH.format(solution=SOLUTION_DIR, repo=REPO_DIR, quoted=shlex.quote(verify))
+ if seeded and verify else _TEST_SH, encoding="utf-8")
+ test_sh.chmod(0o755)
+ (root / "task.toml").write_text(
+ 'schema_version = "1.3"\n'
+ "artifacts = []\n\n"
+ "[task]\n"
+ f'name = "ingot/{name}"\n'
+ f'description = "held-out eval task {index} for skill {skill}"\n'
+ "authors = []\n"
+ "keywords = []\n\n"
+ "[metadata]\n\n"
+ "[verifier]\n"
+ "timeout_sec = 300.0\n"
+ "collect = []\n\n"
+ "[verifier.env]\n\n"
+ "[agent]\n"
+ "timeout_sec = 900.0\n\n"
+ "[environment]\n"
+ # The agent needs the network to reach its own model provider. Containment here is the
+ # container, not the network: that is what makes yolo mode acceptable.
+ 'network_mode = "public"\n'
+ "build_timeout_sec = 600.0\n"
+ 'os = "linux"\n'
+ "mcp_servers = []\n\n"
+ "[environment.env]\n\n"
+ "[solution.env]\n",
+ encoding="utf-8")
+ entries.append(f'[[tasks]]\nname = "ingot/{name}"\n')
+
+ (dataset / "dataset.toml").write_text(
+ "[dataset]\n"
+ f'name = "ingot/{skill}"\n'
+ f'description = "Ingot held-out eval tasks for skill {skill}"\n'
+ "authors = []\n"
+ "keywords = []\n\n" + "\n".join(entries), encoding="utf-8")
+ return dataset
+
+
+def stage_skill(skill: str) -> Path:
+ """A directory holding exactly one `/SKILL.md`, for Harbor's Agent Skills loader.
+
+ Not the skill's parent directory: the vault holds every other skill beside it, and handing the
+ loader that whole tree would put 70-odd unrelated skills in front of the agent. The two arms
+ have to differ by exactly one skill or lift measures the library, not the skill."""
+ staged = BUILD_DIR / "staged" / skill
+ if staged.exists():
+ shutil.rmtree(staged)
+ staged.mkdir(parents=True)
+ shutil.copytree(resolve_skill_dir(skill), staged / skill)
+ return staged
+
+
+# Harnesses that are a CLI with its own subscription login, and the flag that makes Harbor use it.
+# Harbor's adapters default to the API key and only take the subscription when told: claude-code
+# keeps ANTHROPIC_API_KEY unless CLAUDE_FORCE_OAUTH is set (and prefers the key when both are
+# present), codex keeps OPENAI_API_KEY unless CODEX_FORCE_AUTH_JSON is. Every other harness here is
+# a generic model-caller with no CLI to harness, so an API key is inherent to running it at all.
+SUBSCRIPTION_HARNESSES = {
+ "claude-code": ("CLAUDE_FORCE_OAUTH", "ANTHROPIC_API_KEY", "claude setup-token"),
+ "codex": ("CODEX_FORCE_AUTH_JSON", "OPENAI_API_KEY", "codex login"),
+}
+ALLOW_API_BILLING = "HARBOR_ALLOW_API_BILLING"
+
+# Headroom for the image build and the agent install, both of which a seeded dataset makes heavier.
+# Overridable because the right value depends on the host's network and how cold its build cache is.
+BUILD_TIMEOUT_MULTIPLIER = float(os.environ.get("HARBOR_BUILD_TIMEOUT_MULTIPLIER", "4"))
+SETUP_TIMEOUT_MULTIPLIER = float(os.environ.get("HARBOR_SETUP_TIMEOUT_MULTIPLIER", "3"))
+
+
+def billing_refusals(agents: list[str]) -> list[str]:
+ """Harnesses about to bill per token when a subscription login was available.
+
+ Fail-closed on purpose. The default is silent and expensive: a whole grid ran on metered API
+ keys with both subscription credentials sitting unused on the same host, and nothing in the
+ output said so — the per-arm dollar figure Harbor prints is a computed estimate and reads the
+ same either way. Set HARBOR_ALLOW_API_BILLING=1 to opt in deliberately."""
+ if os.environ.get(ALLOW_API_BILLING, "").strip().lower() in ("1", "true", "yes"):
+ return []
+ refusals = []
+ for entry in agents:
+ harness = entry.partition("@")[0]
+ pair = SUBSCRIPTION_HARNESSES.get(harness)
+ if not pair:
+ continue
+ flag, key, how = pair
+ if os.environ.get(flag, "").strip() or not os.environ.get(key, "").strip():
+ continue
+ refusals.append(f"{entry}: would run on {key} (metered) rather than its subscription. "
+ f"Set {flag}=1 after `{how}`.")
+ return refusals
+
+
+def _without_langfuse_env(values: Mapping[str, str] | None) -> dict[str, str]:
+ return {key: value for key, value in (values or {}).items() if not key.startswith("LANGFUSE_")}
+
+
+def _write_json_atomic(path: Path, payload: Mapping[str, object]) -> None:
+ """Avoid exposing a partially written receipt or canary manifest to an interrupted reader."""
+ path.parent.mkdir(parents=True, exist_ok=True)
+ with tempfile.NamedTemporaryFile("w", encoding="utf-8", dir=path.parent,
+ prefix=f".{path.name}.", suffix=".tmp", delete=False) as handle:
+ json.dump(payload, handle, indent=2)
+ handle.flush()
+ os.fsync(handle.fileno())
+ temporary = Path(handle.name)
+ os.replace(temporary, path)
+
+
+def _write_harbor_invocation_receipt(jobs_dir: Path, job_name: str, done: subprocess.CompletedProcess,
+ parent: Mapping[str, str]) -> None:
+ """Persist only non-sensitive Harbor boundary facts, including zero exits before a trial."""
+ job = jobs_dir / job_name
+ # Harbor normally creates this directory. An early CLI failure has no job state, so create
+ # only this expected receipt location rather than fabricating a trial or measurement.
+ _write_json_atomic(job / "harbor-invocation.json", {
+ "returncode": done.returncode,
+ # Arbitrary Harbor output can contain a provider header, argv, or raw endpoint in forms a
+ # redactor cannot soundly enumerate. Counts prove the process boundary without persisting
+ # any of that material.
+ "stdout_bytes": len(str(done.stdout or "").encode("utf-8")),
+ "stderr_bytes": len(str(done.stderr or "").encode("utf-8")),
+ "stdout_excerpt": _redact_harbor_receipt_output(done.stdout, parent),
+ "stderr_excerpt": _redact_harbor_receipt_output(done.stderr, parent),
+ })
+
+
+def run_arm(dataset: Path, agent: str, skill_source: str | None, jobs_dir: Path, job_name: str,
+ model: str | None = None, concurrency: int = 2, attempts: int = 1, *,
+ agent_env: Mapping[str, str] | None = None,
+ agent_kwargs: Mapping[str, str] | None = None,
+ task_name: str | None = None,
+ process_env: Mapping[str, str] | None = None,
+ log=print) -> Path:
+ """Run every task in the dataset through one harness, with or without the skill. Returns job dir.
+
+ The treatment is Harbor's own `--skill`, which implements the Agent Skills spec: the skill
+ directory is mounted into the environment and the harness discovers `SKILL.md` itself. That is
+ how a skill actually reaches an agent in production, and it applies identically to every
+ adapter — pasting the body into a system prompt for one harness and a recipe for another would
+ make the comparison measure the injection channel as much as the skill.
+
+ `skill_source` is a local path or a git source (`org/name[@ref]`), so the benchmark can be
+ pointed at exactly the bytes the canonical vault publishes.
+ """
+ argv = [HARBOR_BIN, "run", "--path", str(dataset), "--agent", agent,
+ "--n-concurrent", str(concurrency), "--jobs-dir", str(jobs_dir),
+ "--job-name", job_name]
+ if attempts > 1:
+ argv += ["--n-attempts", str(attempts)]
+ # Seeded tasks each build their own image, because their seed differs. Before seeding, every
+ # task in a dataset shared one identical Dockerfile and so one cached image built once; now
+ # there are as many builds as tasks, and `apt-get update && install` on an uncached image
+ # overran the 120s compose budget. Observed as `RuntimeError: Command timed out after 120
+ # seconds` with an empty verifier directory and `docker inspect returned 1` in the trial log —
+ # a build failure that looks nothing like one.
+ argv += ["--environment-build-timeout-multiplier", str(BUILD_TIMEOUT_MULTIPLIER),
+ "--agent-setup-timeout-multiplier", str(SETUP_TIMEOUT_MULTIPLIER)]
+ if skill_source:
+ argv += ["--skill", skill_source]
+ if model:
+ argv += ["--model", model]
+ # Harbor forwards these repeated options to the adapter. Sort keys so an identical local
+ # target produces identical command evidence regardless of mapping insertion order.
+ agent_env = _without_langfuse_env(agent_env)
+ for key in sorted(agent_env):
+ argv += ["--ae", f"{key}={agent_env[key]}"]
+ for key in sorted(agent_kwargs or {}):
+ value = agent_kwargs[key]
+ rendered = value if isinstance(value, str) else json.dumps(value, separators=(",", ":"))
+ argv += ["--ak", f"{key}={rendered}"]
+ if task_name:
+ # Harbor filters local datasets by the task directory basename, not dataset.toml's
+ # namespaced task label. The latter looks right but matches no local task.
+ argv += ["--include-task-name", task_name]
+ log(f"[harbor] {agent:<14} running {dataset.name} ({job_name} arm)")
+ run_kwargs = {"capture_output": True, "text": True}
+ if process_env is None:
+ # Legacy/nonlocal runs need inherited provider auth, but Langfuse remains parent-only.
+ run_kwargs["env"] = _without_langfuse_env(os.environ)
+ else:
+ # Some Harbor adapters read their routing settings before they construct the
+ # container command. Preserve only the explicit, local adapter settings here;
+ # inherited provider credentials remain stripped at this process boundary.
+ run_kwargs["env"] = _without_langfuse_env(scrub_provider_env(process_env))
+ run_kwargs["env"].update(agent_env)
+ done = subprocess.run(argv, **run_kwargs)
+ _write_harbor_invocation_receipt(jobs_dir, job_name, done,
+ process_env if process_env is not None else os.environ)
+ if done.returncode != 0:
+ raise RuntimeError(f"harbor run failed for {agent}: {done.stderr.strip()[-600:]}")
+ job = jobs_dir / job_name
+ _refuse_broken_job(job, agent, job_name)
+ return job
+
+
+def watch_native_job(job: Path, expected: Mapping[NativeTrialIdentity, Mapping[str, int]], on_ready,
+ *, released: set[NativeTrialIdentity] | None = None
+ ) -> set[NativeTrialIdentity]:
+ """Release exact identities only after their complete terminal attempt set is persisted."""
+ released = set(released or ())
+ if not job.is_dir():
+ return released
+ for identity, required in expected.items():
+ observed = {task: 0 for task in required}
+ for attempt in iter_attempt_dirs(job, identity=identity):
+ try:
+ record = json.loads((attempt / "result.json").read_text())
+ except (OSError, ValueError):
+ continue
+ if not isinstance(record, dict):
+ continue
+ if not record.get("finished_at"):
+ continue
+ task = str(record.get("task_name") or "").split("/")[-1].split("__")[0]
+ if task not in observed:
+ raise RuntimeError(f"{identity.combination_id} {identity.arm} wrote unexpected task {task}")
+ observed[task] += 1
+ if any(observed[task] > count for task, count in required.items()):
+ raise RuntimeError(f"{identity.combination_id} {identity.arm} exceeded expected attempts")
+ if observed == dict(required) and identity not in released:
+ if on_ready(identity) is not False:
+ released.add(identity)
+ return released
+
+
+def _process_start_token(pid: int) -> str | None:
+ """Bind an owner receipt to one process lifetime, not a reusable PID."""
+ proc_stat = Path(f"/proc/{pid}/stat")
+ try:
+ return proc_stat.read_text().split()[21]
+ except (OSError, IndexError):
+ done = subprocess.run(["ps", "-o", "lstart=", "-p", str(pid)], capture_output=True,
+ text=True)
+ token = done.stdout.strip()
+ return token or None
+
+
+def _claim_native_owner(path: Path, config: Path) -> None:
+ owner = {"pid": os.getpid(), "start_token": _process_start_token(os.getpid()),
+ "config": str(config)}
+ while True:
+ try:
+ descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
+ except FileExistsError:
+ try:
+ existing = json.loads(path.read_text())
+ pid = existing.get("pid")
+ token = existing.get("start_token")
+ except (OSError, ValueError, AttributeError):
+ raise RuntimeError("native Harbor owner receipt is unreadable")
+ if (isinstance(pid, int) and isinstance(token, str)
+ and _process_start_token(pid) == token):
+ raise RuntimeError(f"native Harbor job already has live owner PID {pid}")
+ path.unlink()
+ continue
+ with os.fdopen(descriptor, "w") as handle:
+ json.dump(owner, handle)
+ handle.flush()
+ os.fsync(handle.fileno())
+ return
+
+
+def run_native_job(config: Path, jobs_dir: Path, job_name: str,
+ expected: Mapping[NativeTrialIdentity, Mapping[str, int]], *, on_ready,
+ process_env: Mapping[str, str], poll_seconds: float = 0.5,
+ allow_completed_reuse: bool = False) -> Path:
+ """Run one Harbor config and publish complete identity slices while siblings continue."""
+ jobs_dir.mkdir(parents=True, exist_ok=True)
+ job = jobs_dir / job_name
+ log_path = jobs_dir / f"{job_name}.harbor.log"
+ owner_path = jobs_dir / f"{job_name}.owner.json"
+ state_path = jobs_dir / f"{job_name}.released.json"
+ _claim_native_owner(owner_path, config)
+ argv = [HARBOR_BIN, "run", "--config", str(config),
+ "--override-memory-mb", str(NATIVE_TRIAL_MEMORY_MB), "--job-name", job_name]
+ env = _without_langfuse_env(scrub_provider_env(process_env))
+ extra_compose = env.pop("HARBOR_EXTRA_DOCKER_COMPOSE", None)
+ if extra_compose:
+ overlay = Path(extra_compose)
+ if not overlay.is_file():
+ raise RuntimeError("Harbor Docker Compose overlay is missing")
+ argv.extend(["--extra-docker-compose", str(overlay)])
+ # Aider checks provider presence in the Harbor parent before building its container command.
+ # These are local sentinels; endpoint URLs remain isolated in each agent configuration.
+ env.update({"OPENAI_API_KEY": "local", "ANTHROPIC_API_KEY": "local",
+ "CODEX_API_KEY": "local"})
+ process = None
+ try:
+ released: set[NativeTrialIdentity] = set()
+ if allow_completed_reuse:
+ if state_path.is_file():
+ state = json.loads(state_path.read_text())
+ released = {identity_from_env(item) for item in state.get("released", [])}
+ # Harbor 0.20 redacts credential-shaped agent env values in persisted TrialConfigs. A
+ # second `harbor run` then compares those placeholders with the resolved plan and
+ # rejects an otherwise identical completed job. Released state is not enough on its
+ # own: require the complete terminal artifact set before skipping the subprocess.
+ terminal = watch_native_job(job, expected, lambda _identity: True)
+ if released == terminal == set(expected):
+ return job
+ if terminal == set(expected):
+ released = watch_native_job(job, expected, on_ready, released=released)
+ _write_json_atomic(state_path, {"released": [identity_env(item)
+ for item in sorted(released, key=repr)]})
+ if released == terminal:
+ return job
+ raise RuntimeError("native Harbor finalization is pending")
+ with log_path.open("a", encoding="utf-8") as output:
+ process = subprocess.Popen(argv, stdout=output, stderr=subprocess.STDOUT, env=env,
+ text=True)
+ try:
+ if not allow_completed_reuse and state_path.is_file():
+ state = json.loads(state_path.read_text())
+ released = {identity_from_env(item) for item in state.get("released", [])}
+ while process.poll() is None:
+ before = set(released)
+ released = watch_native_job(job, expected, on_ready, released=released)
+ if released != before:
+ _write_json_atomic(state_path, {"released": [identity_env(item)
+ for item in sorted(released, key=repr)]})
+ time.sleep(poll_seconds)
+ returncode = process.wait()
+ released = watch_native_job(job, expected, on_ready, released=released)
+ _write_json_atomic(state_path, {"released": [identity_env(item)
+ for item in sorted(released, key=repr)]})
+ finally:
+ if process.poll() is None:
+ process.terminate()
+ process.wait()
+ finally:
+ owner_path.unlink(missing_ok=True)
+ if returncode != 0:
+ raise RuntimeError(f"native Harbor job exited {returncode}; see {log_path}")
+ missing = set(expected) - released
+ if missing:
+ raise RuntimeError(f"native Harbor job ended before {len(missing)} identity slice(s) completed")
+ return job
+
+
+def _refuse_broken_job(job: Path, agent: str, arm: str) -> None:
+ """Fail an arm only when nothing in it ran.
+
+ `harbor run` exits 0 even when trials error, and a crashed trial leaves an empty solution
+ directory that `score` reads as a legitimate 0.0. Observed live: a control arm whose four trials
+ were all killed during `docker compose up` scored 0.000 against a skill arm's 0.750 and reported
+ `lift +0.750` — fabricated, and exactly the failure `compat.py` already guards against.
+
+ Failing the whole arm on *any* broken trial is the opposite mistake: a single transient
+ container failure then discards three good trials and the paid-for opposite arm. Individual
+ broken tasks are dropped instead, by `broken_tasks`, from both arms at once."""
+ result = job / "result.json"
+ if not result.is_file():
+ raise RuntimeError(f"{agent} {arm} arm wrote no result.json at {job}")
+ stats = (json.loads(result.read_text()) or {}).get("stats") or {}
+ ran = (stats.get("n_completed_trials", 0) or 0)
+ broken = (stats.get("n_errored_trials", 0) or 0) + (stats.get("n_cancelled_trials", 0) or 0)
+ if ran and broken >= ran:
+ raise RuntimeError(f"{agent} {arm} arm had every one of its {ran} trial(s) error or "
+ f"cancel; refusing to score it")
+
+
+def _trial_outcomes(job: Path, identity: NativeTrialIdentity | None = None) -> list[tuple[str, str, bool]]:
+ """(trial directory name, task name, ok) for every trial in this arm."""
+ out = []
+ results = ([attempt / "result.json" for attempt in iter_attempt_dirs(job, identity=identity)]
+ if identity is not None else job.glob("*/result.json"))
+ for result in results:
+ try:
+ record = json.loads(result.read_text()) or {}
+ except (OSError, ValueError):
+ continue
+ # Harbor nests this as exception_info.exception_type, and names the task in `task_name`
+ # (as "ingot/"). Reading a top-level `exception_type` finds nothing, which made this
+ # guard a silent no-op: opencode's skill arm lost two tasks to AgentSetupTimeoutError and
+ # they were scored as two 0.0s, turning an install timeout into "lift -0.375".
+ failure = ((record.get("exception_info") or {}).get("exception_type") or "").strip()
+ name = str(record.get("task_name") or result.parent.name).split("/")[-1]
+ out.append((result.parent.name, name.split("__")[0], not failure))
+ return out
+
+
+def broken_trials(job: Path, identity: NativeTrialIdentity | None = None) -> set[str]:
+ """Trial directory names that errored or were cancelled.
+
+ With more than one attempt per task these have to be excluded individually. A crashed attempt
+ leaves an empty solution directory, and an empty directory is scored as a real zero — so one
+ flaky attempt out of three would pull the task's mean down by a third and read as the skill
+ performing worse."""
+ return {trial for trial, _, ok in _trial_outcomes(job, identity) if not ok}
+
+
+def broken_tasks(job: Path, identity: NativeTrialIdentity | None = None) -> set[str]:
+ """Task names with no surviving attempt in this arm.
+
+ A task that crashed in one arm has to be dropped from *both*, or the arms are scored on
+ different task sets and the difference between them stops being lift. But with several attempts
+ per task, dropping the task because one attempt broke discards the attempts that did run — and
+ they are the whole reason for paying for repeats."""
+ outcomes = _trial_outcomes(job, identity)
+ survivors = {task for _, task, ok in outcomes if ok}
+ return {task for _, task, _ in outcomes} - survivors
+
+
+def collect_answers(job_dir: Path, skip_trials: set[str] | None = None, *,
+ identity: NativeTrialIdentity | None = None) -> dict[str, list[str]]:
+ """The text each task's agent left in the solution directory, keyed by task name.
+
+ `skip_trials` drops individual crashed attempts, whose workspaces are empty through no fault of
+ the agent and would otherwise be averaged in as zeros.
+
+ A list per task, not a string: with `--n-attempts` above 1 a task has several trials, and
+ keying a single answer by task name silently kept only whichever was read last — throwing away
+ exactly the repeated measurements that were paid for to average the agent's own variance out.
+
+ A task that produced nothing maps to "" rather than being dropped: an empty workspace is a real
+ result (the harness ran and delivered nothing), and silently omitting it would raise the arm's
+ mean by removing its own failures."""
+ skip_trials = skip_trials or set()
+ answers: dict[str, list[str]] = {}
+ solutions = ([attempt / "verifier" / "solution"
+ for attempt in iter_attempt_dirs(job_dir, identity=identity)]
+ if identity is not None else sorted(job_dir.rglob("verifier/solution")))
+ for solution in solutions:
+ if identity is not None:
+ verifier = solution.parent
+ try:
+ verifier_info = verifier.lstat()
+ solution_info = solution.lstat()
+ except OSError:
+ continue
+ if (not stat.S_ISDIR(verifier_info.st_mode)
+ or not stat.S_ISDIR(solution_info.st_mode)):
+ raise ValueError("native Harbor solution directory is not a real directory")
+ elif not solution.is_dir():
+ continue
+ if solution.parent.parent.name in skip_trials:
+ continue
+ name = _trial_task_name(solution)
+ parts = []
+ for path in sorted(p for p in solution.rglob("*") if p.is_file()):
+ if identity is not None:
+ relative = path.relative_to(solution)
+ current = solution
+ for part in relative.parts:
+ current = current / part
+ if stat.S_ISLNK(current.lstat().st_mode):
+ raise ValueError("native Harbor solution contains a symlink")
+ if not _is_deliverable(path.relative_to(solution)):
+ continue
+ try:
+ text = path.read_text(encoding="utf-8")
+ except (OSError, UnicodeDecodeError):
+ continue # a binary the agent happened to leave behind is not the deliverable
+ parts.append(f"--- {path.relative_to(solution)} ---\n{text}")
+ answers.setdefault(name, []).append("\n\n".join(parts)[:60000])
+ return answers
+
+
+# Build leavings, not deliverables. The first real container run wrote __pycache__/*.pyc beside
+# solution.py, and those bytes went into the text handed to the judge — noise the judge pays for
+# and can be misled by.
+_IGNORED_DIRS = {"__pycache__", ".git", "node_modules", ".venv", ".pytest_cache", ".mypy_cache"}
+
+
+def _is_deliverable(relative: Path) -> bool:
+ return not set(relative.parts) & _IGNORED_DIRS
+
+
+def _trial_task_name(solution: Path) -> str:
+ """The task name for a `/verifier/solution` directory.
+
+ Harbor names the trial `__` (observed: `probe-h0__suyygRM`) so repeated
+ attempts at one task cannot collide. The suffix has to come off, or no trial ever matches the
+ task it came from and every score silently reads as a zero."""
+ return solution.parent.parent.name.split("__")[0]
+
+
+def score(answers: dict[str, list[str]], skill: str, holdout: list[dict],
+ skip: set[str] | None = None, concurrency: int = 1) -> list[float]:
+ """Judge each held-out task's collected artifacts with the fixed Ingot judge.
+
+ A task's score is the mean over its attempts. Measured directly on this eval: re-judging one
+ fixed answer three times returned an identical 0.278 every time, while re-running the same
+ agent on the same task under the same model moved the score from 0.278 to 0.556. The variance
+ is the agent's, not the judge's, so the remedy is repeated attempts rather than a better grader.
+
+ An arm that delivered nothing for *every* task is refused rather than scored. A trial can
+ "complete" while its agent never worked: the verifier always reports success, so an agent that
+ died on its first API call still counts as a completed trial with an empty workspace. Observed
+ live: aider v0.86.2 sends `temperature`, claude-sonnet-5 rejects it as deprecated, and the arm
+ came back completed-and-empty — which would have scored a clean 0.000 and read as "aider is
+ terrible at this skill" rather than "aider never ran". Some tasks empty is a real failure and
+ still scores zero; all tasks empty is a broken combination."""
+ skip = skip or set()
+ kept = [i for i in range(len(holdout)) if _task_name(skill, i) not in skip]
+ if not kept:
+ raise RuntimeError("every task crashed in one arm or the other; nothing comparable is left")
+ if not any(any(answers.get(_task_name(skill, i)) or []) for i in kept):
+ raise RuntimeError(f"every task returned an empty workspace; the harness produced no "
+ f"deliverable at all, refusing to score it as zeros")
+ if not isinstance(concurrency, int) or isinstance(concurrency, bool) or concurrency < 1:
+ raise ValueError("score concurrency must be a positive integer")
+ graded: dict[int, list[float]] = {index: [] for index in kept}
+ jobs = []
+ for index in kept:
+ task = holdout[index]
+ attempts = answers.get(_task_name(skill, index)) or [""]
+ for answer in attempts:
+ if not answer:
+ graded[index].append(0.0) # ran, produced nothing: a real zero
+ continue
+ jobs.append((index, task, answer))
+
+ def grade(item) -> tuple[int, float]:
+ index, task, answer = item
+ # The task's own checklist, not the judge's generic four. Dropping it here is what made
+ # the first build-loop matrix unreadable: controls piled up at 0.849.
+ value = judge(task["task"], task["rubric"], answer,
+ check=task.get("check"), deliverable=task.get("deliverable"),
+ checklist=task.get("checklist"))["score"]
+ return index, value
+
+ if concurrency == 1 or len(jobs) < 2:
+ results = map(grade, jobs)
+ for index, value in results:
+ graded[index].append(value)
+ else:
+ with ThreadPoolExecutor(max_workers=min(concurrency, len(jobs))) as pool:
+ for index, value in pool.map(grade, jobs):
+ graded[index].append(value)
+ return [sum(graded[index]) / len(graded[index]) for index in kept]
+
+
+# Changes to how persisted Harbor artifacts are interpreted must change this identifier. Rescore
+# uses it to refuse evidence made under different scoring semantics rather than mixing the rows.
+SCORING_REVISION = "harbor-rubric-v2-agy"
+
+
+def _task_fingerprint(holdout: list[dict]) -> str:
+ """Stable identity of the exact held-out task set, independent of dict insertion order."""
+ canonical = json.dumps(holdout, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
+ return hashlib.sha256(canonical.encode()).hexdigest()
+
+
+def _combination_id(harness: str, target: LocalTarget) -> str:
+ """Stable local evidence identity, including the endpoint fingerprint through its job slug."""
+ return f"{harness}@{target.served_model}--{target.job_slug}"
+
+
+def _combination_job_slug(harness: str, target: LocalTarget) -> str:
+ """Keep the evidence identity exact without putting Docker-unsafe model IDs in bind paths."""
+ identity = _combination_id(harness, target)
+ return identity if ":" not in identity else f"{harness}@{target.job_slug}"
+
+
+def _native_full_job_name(cell: NativeCell) -> str:
+ """Stable one-cell Harbor boundary; changing the rest of a matrix never changes its config."""
+ return f"native-full--{cell.harness}--{cell.target.job_slug}"
+
+
+def _round_robin_endpoints(cells: Sequence[NativeCell]) -> list[NativeCell]:
+ """Keep the next bounded cell jobs on different physical endpoint identities."""
+ buckets: dict[str, list[NativeCell]] = {}
+ for cell in cells:
+ buckets.setdefault(cell.target.fingerprint, []).append(cell)
+ ordered = []
+ while any(buckets.values()):
+ for bucket in buckets.values():
+ if bucket:
+ ordered.append(bucket.pop(0))
+ return ordered
+
+
+def _canary_artifact(job: Path, task_name: str,
+ identity: NativeTrialIdentity | None = None) -> str | None:
+ """Return a diagnostic when the one-task canary did not yield usable Harbor evidence."""
+ if identity is not None:
+ attempts = list(iter_attempt_dirs(job, identity=identity))
+ if len(attempts) != 1:
+ return "canary wrote no completed trial result"
+ result = attempts[0] / "result.json"
+ try:
+ record = json.loads(result.read_text()) or {}
+ except (OSError, ValueError):
+ return "canary trial result was unreadable"
+ recorded_task = str(record.get("task_name") or "").split("/")[-1].split("__")[0]
+ exception = ((record.get("exception_info") or {}).get("exception_type") or "").strip()
+ if recorded_task != task_name or exception:
+ return "canary trial did not complete without an exception"
+ exception_path = result.parent / "exception.txt"
+ if exception_path.is_file() and exception_path.read_bytes().strip():
+ return "canary trial wrote exception evidence"
+ answers = collect_answers(job, identity=identity)
+ if not any(answer.strip() for values in answers.values() for answer in values):
+ return "canary produced no nonempty verifier solution artifact"
+ return None
+ try:
+ summary = json.loads((job / "result.json").read_text()) or {}
+ completed = ((summary.get("stats") or {}).get("n_completed_trials", 0) or 0)
+ except (OSError, ValueError):
+ return "canary wrote no completed trial result"
+ if not isinstance(completed, int) or isinstance(completed, bool) or completed < 1:
+ return "canary wrote no completed trial"
+ trials = list(job.glob("*/result.json"))
+ if not trials:
+ return "canary wrote no completed trial result"
+ for result in trials:
+ try:
+ record = json.loads(result.read_text()) or {}
+ except (OSError, ValueError):
+ return "canary trial result was unreadable"
+ recorded_task = str(record.get("task_name") or "").split("/")[-1].split("__")[0]
+ exception = ((record.get("exception_info") or {}).get("exception_type") or "").strip()
+ if recorded_task != task_name or exception:
+ return "canary trial did not complete without an exception"
+ exception_path = result.parent / "exception.txt"
+ if exception_path.is_file() and exception_path.read_bytes().strip():
+ return "canary trial wrote exception evidence"
+ solution = result.parent / "verifier" / "solution"
+ if not solution.is_dir() or not any(path.is_file() and path.stat().st_size > 0
+ for path in solution.rglob("*")):
+ return "canary produced no nonempty verifier solution artifact"
+ return None
+ return "canary wrote no matching held-out trial"
+
+
+def run_canary(skill: str, dataset: Path, holdout: list[dict], source: str, harness: str,
+ target: LocalTarget, canary_root: Path, *, exploratory: bool = False, log=print) -> dict:
+ """Run the first held-out task once and retain the diagnostic evidence for this seam."""
+ task_name = _task_name(skill, 0)
+ jobs_dir = canary_root / skill / target.job_slug
+ route = gateway_route(target, harness)
+ # Preserve failed native/gateway evidence. A translation revision changes the gateway model
+ # but Harbor's fixed harness job name otherwise reopens the prior one-task job.
+ job_name = f"{harness}--{route.identity}" if route else harness
+ model = route.model if route else harbor_model(target, harness)
+ record = {"combination": _combination_id(harness, target), "harness": harness,
+ "model": model, "target_alias": target.alias,
+ "endpoint_fingerprint": target.fingerprint, "protocol": protocol_for(harness),
+ "job": str(jobs_dir / job_name), "family": target.family,
+ "parameter_billions": target.parameter_billions,
+ "quantization": target.quantization, "tool_parser": target.tool_parser,
+ "exploratory": exploratory, "rankable": not exploratory}
+ if route:
+ record.update(gateway_metadata(route))
+ try:
+ job = run_arm(
+ dataset, gateway_agent_name(route) if route else harness, source, jobs_dir, job_name, model=model, concurrency=1,
+ attempts=1,
+ agent_env=_without_langfuse_env(
+ gateway_agent_env(target, route) if route else local_agent_env(target, harness)),
+ agent_kwargs={} if route else harbor_agent_kwargs(target, harness), task_name=task_name,
+ process_env=gateway_process_env(os.environ) if route and route.harness == "codex" else os.environ, log=log,
+ )
+ except Exception as error: # noqa: BLE001 - retain failed seam evidence and stop before full arms
+ record["error"] = f"{type(error).__name__}: {error}"[:400]
+ return record
+ try:
+ telemetry_metadata = {key: value for key, value in record.items() if key != "job"}
+ telemetry_metadata.update(_telemetry_provenance(skill, holdout, source))
+ _write_json_atomic(job / "combo.json", telemetry_metadata)
+ export_job_attempts(job, {**telemetry_metadata, "arm": "canary"})
+ except Exception as error: # noqa: BLE001 - measurement survives telemetry repair work
+ record["telemetry_error"] = _redact_harbor_receipt_output(
+ f"{type(error).__name__}: {error}", os.environ)
+ try:
+ if diagnostic := _canary_artifact(job, task_name):
+ record["error"] = diagnostic
+ else:
+ record["ok"] = True
+ except Exception as error: # noqa: BLE001 - retain failed seam evidence and stop before full arms
+ record["error"] = f"{type(error).__name__}: {error}"[:400]
+ return record
+
+
+def _run_native_canaries(skill: str, dataset: Path, targets: Sequence[LocalTarget],
+ harnesses: Sequence[str], holdout: list[dict], source: str,
+ canary_root: Path, *, global_limit: int, endpoint_limit: int,
+ allow_completed_reuse: bool = False,
+ process_env: Mapping[str, str] = os.environ, log=print) -> dict:
+ """Run every skill-specific model×harness canary through one bounded Harbor job."""
+ cells = [NativeCell(target, harness) for target in targets for harness in harnesses]
+ jobs_root = canary_root / skill
+ job_name = "native-canaries"
+ config = compile_canary_job(
+ dataset, _task_name(skill, 0), cells, Path(source), jobs_root,
+ global_limit=global_limit,
+ endpoint_limits={cell.target.fingerprint: endpoint_limit for cell in cells})
+ config_path = jobs_root / "native-canaries.config.json"
+ write_job_config(config_path, config)
+ job = jobs_root / job_name
+ records = {}
+ provenance = _telemetry_provenance(skill, holdout, source)
+ expected = {}
+ for cell in cells:
+ identity = native_trial_identity(cell.target, cell.harness, "canary")
+ expected[identity] = {_task_name(skill, 0): 1}
+ route = gateway_route(cell.target, cell.harness)
+ record = {**_combo_metadata(holdout, cell.harness, cell.target, 1),
+ "model": route.model if route else harbor_model(cell.target, cell.harness),
+ "job": str(job), "exploratory": False, "rankable": True}
+ if route:
+ record.update(gateway_metadata(route))
+ record["gateway_revision"] = identity.gateway_revision
+ records[cell.combination_id] = record
+
+ def on_ready(identity: NativeTrialIdentity) -> bool:
+ record = records[identity.combination_id]
+ metadata = {key: value for key, value in record.items() if key != "job"}
+ metadata.update(provenance)
+ try:
+ export_job_attempts(job, {**metadata, "arm": "canary"}, identity=identity)
+ except Exception as error: # noqa: BLE001 - retain canary evidence
+ record["telemetry_error"] = _redact_harbor_receipt_output(
+ f"{type(error).__name__}: {error}", os.environ)
+ record["error"] = "canary telemetry receipt was not verified"
+ return False
+ record.pop("telemetry_error", None)
+ record.pop("error", None)
+ diagnostic = _canary_artifact(job, _task_name(skill, 0), identity)
+ if diagnostic:
+ record["error"] = diagnostic
+ else:
+ record["ok"] = True
+ return True
+
+ process_env = gateway_process_env(process_env) if any(
+ cell.harness == "codex" and gateway_route(cell.target, cell.harness) for cell in cells
+ ) else process_env
+ run_native_job(config_path, jobs_root, job_name, expected, on_ready=on_ready,
+ process_env=process_env, allow_completed_reuse=allow_completed_reuse)
+ # A prior controller may have persisted terminal/released trials before it returned the
+ # manifest. Re-finalize those exact identities from disk; exporter receipts are idempotent.
+ for identity in expected:
+ record = records[identity.combination_id]
+ if "ok" not in record and "error" not in record:
+ on_ready(identity)
+ return records
+
+
+def _telemetry_provenance(skill: str, holdout: list[dict], source: str) -> dict:
+ skill_file = Path(source) / skill / "SKILL.md"
+ if not skill_file.is_file():
+ skill_file = Path(source) / "SKILL.md"
+ skill_bytes = skill_file.read_bytes()
+ skill_body = skill_bytes.decode("utf-8")
+ return {
+ "skill": skill,
+ "skill_body": skill_body,
+ "skill_sha256": hashlib.sha256(skill_bytes).hexdigest(),
+ "task_texts": {_task_name(skill, index): _redact_harbor_receipt_output(
+ str(task.get("task") or ""), {})
+ for index, task in enumerate(holdout)},
+ }
+
+
+def _combo_metadata(holdout: list[dict], harness: str, target: LocalTarget,
+ attempts: int, exploratory: bool = False) -> dict:
+ metadata = {
+ "combination": _combination_id(harness, target),
+ "harness": harness,
+ "model": target.served_model,
+ "target_alias": target.alias,
+ "endpoint_fingerprint": target.fingerprint,
+ "protocol": protocol_for(harness),
+ "task_fingerprint": _task_fingerprint(holdout),
+ "attempts": attempts,
+ "family": target.family,
+ "parameter_billions": target.parameter_billions,
+ "quantization": target.quantization,
+ "tool_parser": target.tool_parser,
+ "exploratory": exploratory,
+ "rankable": not exploratory,
+ }
+ if route := gateway_route(target, harness):
+ metadata.update(gateway_metadata(route))
+ return metadata
+
+
+def _run_local_full_arms(skill: str, dataset: Path, targets: Sequence[LocalTarget],
+ harnesses: Sequence[str], holdout: list[dict], source: str,
+ attempts: int, concurrency: int,
+ jobs_root: Path, manifest: dict, canaries: dict | None = None,
+ exploratory: bool = False, log=print) -> None:
+ """Run full arms for passed canaries and persist failed seams as unmeasured rows."""
+ for target in targets:
+ for harness in harnesses:
+ route = gateway_route(target, harness)
+ metadata = _combo_metadata(holdout, harness, target, attempts, exploratory)
+ key = _combination_id(harness, target)
+ jobs_dir = jobs_root / _combination_job_slug(harness, target)
+ canary = (canaries or {}).get(key, {})
+ if "error" in canary:
+ error = _redact_harbor_receipt_output(str(canary["error"]), os.environ)[:400]
+ jobs_dir.mkdir(parents=True, exist_ok=True)
+ _write_json_atomic(jobs_dir / "combo.json", {**metadata, "canary_error": error})
+ manifest["combinations"][key] = {**metadata, "error": error}
+ log(f"[harbor] {key:<54} UNMEASURED: {error[:300]}")
+ continue
+ try:
+ jobs_dir.mkdir(parents=True, exist_ok=True)
+ jobs = {}
+ routing = {
+ "model": route.model if route else harbor_model(target, harness),
+ "concurrency": concurrency, "attempts": attempts,
+ "agent_env": _without_langfuse_env(
+ gateway_agent_env(target, route) if route else local_agent_env(target, harness)),
+ "agent_kwargs": {} if route else harbor_agent_kwargs(target, harness),
+ "process_env": (gateway_process_env(os.environ)
+ if route and route.harness == "codex" else os.environ), "log": log,
+ }
+ telemetry_errors = {}
+ telemetry_ready: bool | None = None
+ for arm in ("skill", "control"):
+ jobs[arm] = run_arm(dataset, gateway_agent_name(route) if route else harness,
+ source if arm == "skill" else None,
+ jobs_dir, arm, **routing)
+ if telemetry_ready is None:
+ try:
+ metadata.update(_telemetry_provenance(skill, holdout, source))
+ _write_json_atomic(jobs_dir / "combo.json", metadata)
+ telemetry_ready = True
+ except Exception as error: # noqa: BLE001 - retain paid-for arms
+ telemetry_ready = False
+ telemetry_errors["provenance"] = _redact_harbor_receipt_output(
+ f"{type(error).__name__}: {error}", os.environ)
+ if telemetry_ready:
+ try:
+ export_job_attempts(jobs[arm], {**metadata, "arm": arm})
+ except Exception as error: # noqa: BLE001 - publication gates later
+ telemetry_errors[arm] = _redact_harbor_receipt_output(
+ f"{type(error).__name__}: {error}", os.environ)
+ skipped = broken_tasks(jobs["skill"]) | broken_tasks(jobs["control"])
+ manifest["combinations"][key] = {
+ **metadata, "raw_evidence": True,
+ "skill_job": str(jobs["skill"]), "control_job": str(jobs["control"]),
+ "tasks_dropped": sorted(skipped),
+ }
+ if telemetry_errors:
+ manifest["combinations"][key]["telemetry_errors"] = telemetry_errors
+ except Exception as error: # noqa: BLE001 - other combinations remain useful evidence
+ manifest["combinations"][key] = {
+ **metadata, "error": f"{type(error).__name__}: {error}"[:400],
+ }
+ log(f"[harbor] {key:<54} UNAVAILABLE: {str(error)[:300]}")
+
+
+def _run_native_full_arms(skill: str, dataset: Path, targets: Sequence[LocalTarget],
+ harnesses: Sequence[str], holdout: list[dict], source: str,
+ jobs_root: Path, manifest: dict,
+ canaries: Mapping[str, Mapping[str, object]], *,
+ global_limit: int, endpoint_limit: int,
+ publish_root: Path = HARBOR_DIR,
+ allow_completed_reuse: bool = False,
+ process_env: Mapping[str, str] = os.environ, log=print) -> None:
+ """Run approved cells in bounded independent Harbor jobs and publish each complete pair."""
+ from .harbor_rescore import current_scoring_identity, rescore
+
+ cells = [NativeCell(target, harness) for target in targets for harness in harnesses]
+ selected, unmeasured = select_measurement_cells(cells, canaries)
+ manifest["combinations"].update(unmeasured)
+ provenance = _telemetry_provenance(skill, holdout, source)
+ unmeasured_paths = []
+ for cell in cells:
+ failed = unmeasured.get(cell.combination_id)
+ if failed is None:
+ continue
+ combo = jobs_root / _combination_job_slug(cell.harness, cell.target)
+ _write_json_atomic(combo / "combo.json", {
+ **_combo_metadata(holdout, cell.harness, cell.target, 3),
+ **provenance,
+ "canary_error": failed["error"],
+ })
+ unmeasured_paths.append(combo)
+ if not selected:
+ return
+ scoring = current_scoring_identity()
+ combos = {}
+ agent_identity = {
+ "skill_sha256": provenance["skill_sha256"],
+ "task_fingerprint": _task_fingerprint(holdout),
+ "attempts": 3,
+ "exporter_revision": EXPORTER_REVISION,
+ "cells": sorted((cell.combination_id,
+ native_trial_identity(cell.target, cell.harness, "skill").gateway_revision)
+ for cell in selected),
+ }
+ pipeline_path = jobs_root / "native-full.pipeline.json"
+ pipeline = {"agent_identity": agent_identity, "scoring_identity": scoring,
+ "exported": {}, "graded": [], "published": []}
+ if pipeline_path.is_file():
+ saved = json.loads(pipeline_path.read_text())
+ if isinstance(saved, dict):
+ if saved.get("agent_identity") == agent_identity:
+ pipeline["exported"] = saved.get("exported", {})
+ if saved.get("scoring_identity") == scoring:
+ pipeline["graded"] = saved.get("graded", [])
+ pipeline["published"] = saved.get("published", [])
+ ready = {key: set(value) for key, value in pipeline["exported"].items()}
+ failed_this_run: set[tuple[str, str]] = set()
+ unmeasured_pending = bool(unmeasured_paths)
+ prepared = []
+ for cell in _round_robin_endpoints(selected):
+ combo = jobs_root / _combination_job_slug(cell.harness, cell.target)
+ job_name = _native_full_job_name(cell)
+ identities = {}
+ expected = {}
+ for arm in ("skill", "control"):
+ identity = native_trial_identity(cell.target, cell.harness, arm)
+ identities[arm] = identity
+ expected[identity] = {_task_name(skill, index): 3 for index in range(len(holdout))}
+ prepared.append((cell, combo, job_name, identities, expected))
+
+ legacy_job_name = "native-full"
+ legacy_expected = {identity: required for _cell, _combo, _job, _identities, expected in prepared
+ for identity, required in expected.items()}
+ legacy_terminal = watch_native_job(
+ jobs_root / legacy_job_name, legacy_expected, lambda _identity: True)
+ adopted: list[NativeTrialIdentity] = []
+ for cell, combo, job_name, identities, expected in prepared:
+ try:
+ prior = json.loads((combo / "combo.json").read_text())
+ except (OSError, ValueError):
+ prior = {}
+ prior = prior if isinstance(prior, dict) else {}
+ metadata = {**prior, **_combo_metadata(holdout, cell.harness, cell.target, 3),
+ **provenance, "native_identities": {
+ arm: identity_env(identity) for arm, identity in identities.items()}}
+ metadata["gateway_revision"] = identities["skill"].gateway_revision
+ native_jobs = dict(metadata.get("native_jobs") or {})
+ source_jobs = {}
+ missing = {}
+ for arm, identity in identities.items():
+ if identity in legacy_terminal:
+ source_jobs[arm] = legacy_job_name
+ native_jobs[arm] = legacy_job_name
+ adopted.append(identity)
+ else:
+ source_jobs[arm] = job_name
+ missing[identity] = expected[identity]
+ if native_jobs:
+ metadata["native_jobs"] = native_jobs
+ _write_json_atomic(combo / "combo.json", metadata)
+ config_path = None
+ if missing:
+ config = compile_measurement_job(
+ dataset, [_task_name(skill, index) for index in range(len(holdout))], [cell],
+ Path(source), jobs_root, attempts=3, global_limit=endpoint_limit,
+ endpoint_limits={cell.target.fingerprint: endpoint_limit},
+ arms=tuple(identity.arm for identity in missing))
+ config_path = jobs_root / f"{job_name}.config.json"
+ write_job_config(config_path, config)
+ combos[cell.combination_id] = (
+ combo, metadata, identities, job_name, config_path, missing, source_jobs)
+
+ def has_recovered_lift(identity: NativeTrialIdentity, rows: Mapping[str, Any]) -> bool:
+ return any(
+ isinstance(row, dict) and row.get("combination") == identity.combination_id
+ and row.get("endpoint_fingerprint") == identity.endpoint_fingerprint
+ and row.get("skill_sha256") == agent_identity["skill_sha256"]
+ and row.get("task_fingerprint") == agent_identity["task_fingerprint"]
+ and row.get("attempts") == agent_identity["attempts"]
+ and row.get("harness") == identity.harness
+ and row.get("protocol") == identity.protocol
+ and row.get("gateway_revision", "direct") == identity.gateway_revision
+ and all(row.get(key) == value for key, value in scoring.items())
+ and isinstance(row.get("lift"), (int, float))
+ and not isinstance(row.get("lift"), bool)
+ for row in rows.values()
+ )
+
+ matrix_output = publish_root / f"{skill}.rescored.json"
+ if unmeasured_pending and matrix_output.is_file():
+ try:
+ existing = json.loads(matrix_output.read_text())
+ except (OSError, ValueError):
+ existing = {}
+ rows = existing.get("combinations", {}) if isinstance(existing, dict) else {}
+ if any(has_recovered_lift(identities["skill"], rows)
+ for _combo, _metadata, identities, _job, _config, _expected, _sources
+ in combos.values()):
+ rescore(skill, jobs_roots=[jobs_root], combination_paths=unmeasured_paths,
+ output=matrix_output, scoring_identity=scoring, log=log)
+ unmeasured_pending = False
+
+ state_lock = threading.Lock()
+ rescore_lock = threading.Lock()
+ combo_locks = {key: threading.Lock() for key in combos}
+
+ def on_ready(identity: NativeTrialIdentity) -> None:
+ nonlocal unmeasured_pending
+ with combo_locks[identity.combination_id]:
+ combo, metadata, _identities, job_name, _config, _expected, source_jobs = combos[
+ identity.combination_id]
+ native_jobs = metadata.setdefault("native_jobs", {})
+ native_jobs[identity.arm] = source_jobs[identity.arm]
+ if set(native_jobs) >= {"skill", "control"}:
+ metadata.pop("native_job", None)
+ _write_json_atomic(combo / "combo.json", metadata)
+ with state_lock:
+ arms = ready.setdefault(identity.combination_id, set())
+ if identity.arm not in arms:
+ stage = (identity.combination_id, f"export:{identity.arm}")
+ with state_lock:
+ if stage in failed_this_run:
+ return False
+ try:
+ export_job_attempts(jobs_root / source_jobs[identity.arm],
+ {**metadata, "arm": identity.arm},
+ identity=identity)
+ except Exception as error: # noqa: BLE001 - retry on a later controller run
+ with state_lock:
+ failed_this_run.add(stage)
+ manifest["combinations"].setdefault(identity.combination_id, {}).update(
+ telemetry_error=f"{type(error).__name__}: {error}"[:400])
+ return False
+ with state_lock:
+ arms.add(identity.arm)
+ pipeline["exported"][identity.combination_id] = sorted(arms)
+ _write_json_atomic(pipeline_path, pipeline)
+ if arms == {"skill", "control"}:
+ with state_lock:
+ needs_grade = identity.combination_id not in pipeline["graded"]
+ if needs_grade:
+ stage = (identity.combination_id, "grade")
+ with state_lock:
+ if stage in failed_this_run:
+ return False
+ try:
+ with rescore_lock:
+ existing = (json.loads(matrix_output.read_text())
+ if matrix_output.is_file() else {})
+ rows = (existing.get("combinations", {})
+ if isinstance(existing, dict) else {})
+ recovered = has_recovered_lift(identity, rows)
+ with state_lock:
+ include_unmeasured = unmeasured_pending
+ if not recovered or include_unmeasured:
+ selected_paths = ([*unmeasured_paths, combo]
+ if include_unmeasured else [combo])
+ rescore(skill, jobs_roots=[jobs_root],
+ combination_paths=selected_paths,
+ output=matrix_output, scoring_identity=scoring, log=log)
+ with state_lock:
+ unmeasured_pending = False
+ except Exception as error: # noqa: BLE001 - retry later without rerunning agents
+ with state_lock:
+ failed_this_run.add(stage)
+ manifest["combinations"].setdefault(identity.combination_id, {}).update(
+ scoring_error=f"{type(error).__name__}: {error}"[:400])
+ return False
+ with state_lock:
+ pipeline["graded"].append(identity.combination_id)
+ _write_json_atomic(pipeline_path, pipeline)
+ with state_lock:
+ needs_publish = identity.combination_id not in pipeline["published"]
+ if needs_publish:
+ try:
+ with state_lock:
+ manifest["combinations"][identity.combination_id] = {
+ **metadata, "raw_evidence": True,
+ "native_jobs": {arm: str(jobs_root / source_job)
+ for arm, source_job in source_jobs.items()}}
+ _write_json_atomic(jobs_root / "progress.json", manifest)
+ except Exception as error: # noqa: BLE001 - grade receipt prevents repeated billing
+ with state_lock:
+ manifest["combinations"].setdefault(identity.combination_id, {}).update(
+ publication_error=f"{type(error).__name__}: {error}"[:400])
+ return False
+ with state_lock:
+ pipeline["published"].append(identity.combination_id)
+ _write_json_atomic(pipeline_path, pipeline)
+ return True
+
+ process_env = gateway_process_env(process_env) if any(
+ cell.harness == "codex" and gateway_route(cell.target, cell.harness) for cell in selected
+ ) else process_env
+ endpoint_locks = {cell.target.fingerprint: threading.Lock() for cell in selected}
+
+ for identity in adopted:
+ if on_ready(identity) is False:
+ raise RuntimeError("legacy native Harbor finalization is pending")
+
+ def run_cell(item) -> None:
+ _combo, _metadata, identities, job_name, config_path, expected, _sources = item
+ if not expected:
+ return
+ assert config_path is not None
+ fingerprint = identities["skill"].endpoint_fingerprint
+ with endpoint_locks[fingerprint]:
+ run_native_job(config_path, jobs_root, job_name, expected, on_ready=on_ready,
+ process_env=process_env, allow_completed_reuse=allow_completed_reuse)
+
+ # Each one-cell Harbor job can consume at most endpoint_limit slots. Bound the number of live
+ # jobs so their aggregate cannot exceed the caller's global limit. Harbor then schedules only
+ # that cell's 24 trials, producing publishable evidence before later cells finish.
+ workers = max(1, min(len(combos), global_limit // endpoint_limit))
+ with ThreadPoolExecutor(max_workers=workers) as pool:
+ list(pool.map(run_cell, combos.values()))
+
+
+def run_local_sweep(skill: str, targets: list[LocalTarget], *,
+ harnesses: Sequence[str] = LOCAL_HARNESSES, concurrency: int = 2,
+ attempts: int = 3, skill_source: str | None = None, canary_only: bool = False,
+ exploratory: bool = False, native_parallel: bool = False,
+ evidence_root: Path | None = None,
+ expected_task_fingerprint: str | None = None,
+ expected_runtime_revisions: Mapping[str, str] | None = None,
+ global_concurrency: int | None = None,
+ endpoint_concurrency: int | None = None,
+ publish_root: Path = HARBOR_DIR,
+ content_addressed_resume: bool = False,
+ process_env: Mapping[str, str] | None = None, log=print) -> dict:
+ """Evaluate every local target/harness pair whose routing canary passes.
+
+ This deliberately returns raw-evidence manifest only. `harbor_rescore` owns publication of
+ the visible matrix, so a half-finished local sweep cannot replace a known-good matrix.
+ """
+ if attempts != 3 and not (exploratory and attempts == 1):
+ raise ValueError("local Harbor sweeps require 3 attempts, or 1 with exploratory=True")
+ if not targets:
+ raise ValueError("local Harbor sweep needs at least one target")
+ harnesses = tuple(harnesses)
+ for harness in harnesses:
+ protocol_for(harness)
+ _, holdout, _ = load_tasks(skill)
+ if not holdout:
+ raise SystemExit(f"'{skill}' has no held-out eval tasks to run.")
+ task_fingerprint = _task_fingerprint(holdout)
+ if expected_task_fingerprint is not None and task_fingerprint != expected_task_fingerprint:
+ raise RuntimeError(f"{skill} held-out tasks changed after catalog enqueue")
+ if not shutil.which(HARBOR_BIN):
+ raise SystemExit(f"'{HARBOR_BIN}' is not on PATH; install with `uv tool install harbor`.")
+ source = skill_source or str(stage_skill(skill))
+ dataset = build_dataset(skill, holdout, BUILD_DIR)
+ manifest = {"skill": skill, "tasks": len(holdout), "attempts": attempts,
+ "canary_only": canary_only, "canaries": {}, "combinations": {},
+ "aborted": False, "exploratory": exploratory, "rankable": not exploratory}
+ run_root = evidence_root or HARBOR_DIR
+ canary_root = run_root / ("canaries-k1" if exploratory else "canaries")
+ manifest_path = canary_root / skill / "manifest.json"
+ global_limit = global_concurrency if global_concurrency is not None else max(16, concurrency)
+ endpoint_limit = endpoint_concurrency if endpoint_concurrency is not None else max(1, concurrency)
+ native_process_env = process_env if process_env is not None else os.environ
+ if (not isinstance(global_limit, int) or isinstance(global_limit, bool) or global_limit < 1
+ or not isinstance(endpoint_limit, int) or isinstance(endpoint_limit, bool)
+ or endpoint_limit < 1 or endpoint_limit > global_limit):
+ raise ValueError("native concurrency limits must be positive and endpoint <= global")
+
+ def finish() -> dict:
+ _write_json_atomic(manifest_path, _redact_persisted(manifest, os.environ))
+ return manifest
+
+ # Re-discover even targets supplied through the Python API. CLI callers already do this while
+ # parsing `--target`, but the sweep is also a public orchestration interface and must not trust
+ # a hand-constructed LocalTarget to have passed the `/v1/models` identity/context preflight.
+ # No container trial or full job directory exists yet.
+ try:
+ targets = [discover_target(target.alias, target.base_url) for target in targets]
+ if expected_runtime_revisions is not None:
+ for target in targets:
+ for harness in harnesses:
+ identity = native_trial_identity(target, harness, "skill")
+ key = f"route:{target.fingerprint}:{harness}"
+ observed = f"{identity.protocol}/{identity.gateway_revision}/context={target.context_length}"
+ if expected_runtime_revisions.get(key) != observed:
+ raise RuntimeError(f"{key} changed after catalog enqueue")
+ # Probe every required adapter protocol for every target before spending one container
+ # trial. A local endpoint is part of the treatment identity; provider fallback is forbidden.
+ required_protocols = sorted({protocol_for(harness) for harness in harnesses})
+ for target in targets:
+ for protocol in required_protocols:
+ probe_protocol(target, protocol)
+ if "chat" in required_protocols:
+ probe_chat_tool_round_trip(target)
+ except Exception as error: # noqa: BLE001 - endpoint preflight is diagnostic, not a measurement
+ manifest["aborted"] = True
+ manifest["preflight_error"] = f"{type(error).__name__}: {error}"[:400]
+ return finish()
+
+ routes = [(route, target) for target in targets for harness in harnesses
+ if (route := gateway_route(target, harness)) is not None]
+ try:
+ gateway_context = (GatewaySession(routes, canary_root / "gateway" / skill)
+ if routes else contextlib.nullcontext())
+ with gateway_context:
+ if native_parallel and attempts == 3 and not exploratory:
+ manifest["canaries"].update(_run_native_canaries(
+ skill, dataset, targets, harnesses, holdout, source, canary_root,
+ global_limit=global_limit, endpoint_limit=endpoint_limit,
+ allow_completed_reuse=content_addressed_resume,
+ process_env=native_process_env, log=log))
+ if any("telemetry_error" in record for record in manifest["canaries"].values()):
+ manifest["telemetry_pending"] = True
+ else:
+ for target in targets:
+ for harness in harnesses:
+ key = _combination_id(harness, target)
+ manifest["canaries"][key] = run_canary(
+ skill, dataset, holdout, source, harness, target, canary_root,
+ exploratory=exploratory, log=log)
+ if canary_only:
+ return finish()
+ jobs_root = run_root / "jobs" / f"{skill}-k{attempts}"
+ if native_parallel and attempts == 3 and not exploratory:
+ _run_native_full_arms(
+ skill, dataset, targets, harnesses, holdout, source, jobs_root, manifest,
+ manifest["canaries"], global_limit=global_limit,
+ endpoint_limit=endpoint_limit, publish_root=publish_root,
+ allow_completed_reuse=content_addressed_resume,
+ process_env=native_process_env, log=log)
+ else:
+ _run_local_full_arms(skill, dataset, targets, harnesses, holdout, source,
+ attempts, concurrency, jobs_root, manifest,
+ canaries=manifest["canaries"], exploratory=exploratory, log=log)
+ except Exception as error: # noqa: BLE001 - fail before Harbor if the fixed gateway is stale/unreachable
+ manifest["aborted"] = True
+ manifest["gateway_error"] = f"{type(error).__name__}: {error}"[:400]
+ return finish()
+ return finish()
+
+
+def run_harbor_eval(skill: str, agents: list[str], model: str | None = None,
+ concurrency: int = 2, skill_source: str | None = None, attempts: int = 1,
+ log=print) -> dict:
+ """Skill-vs-control lift for one skill across several harnesses. Writes runs/harbor/.json.
+
+ `skill_source` overrides where the skill is read from — a git source pins the benchmark to the
+ bytes the canonical vault publishes rather than whatever this checkout happens to hold.
+
+ `attempts` runs each task that many times per arm and averages. One attempt per task is not
+ enough to see an effect this size: two control-arm runs of an identical configuration moved a
+ task's score by 0.278 and swapped the ranking of two harnesses, which is larger than any lift
+ the first grid reported."""
+ _, holdout, _ = load_tasks(skill)
+ if not holdout:
+ raise SystemExit(f"'{skill}' has no held-out eval tasks to run.")
+ source = skill_source or str(stage_skill(skill))
+ if not shutil.which(HARBOR_BIN):
+ raise SystemExit(f"'{HARBOR_BIN}' is not on PATH; install with `uv tool install harbor`.")
+ # Before anything is spent, not per-row after: a grid is hours long and the bill is already run
+ # up by the time a row would report it.
+ if refusals := billing_refusals(agents):
+ raise SystemExit("[harbor] refusing to start; these would bill per token:\n "
+ + "\n ".join(refusals)
+ + f"\nOr set {ALLOW_API_BILLING}=1 to accept metered billing.")
+ dataset = build_dataset(skill, holdout, BUILD_DIR)
+ # Runs at different attempt counts keep separate roots. Harbor refuses a job directory whose
+ # config has changed ("cannot be resumed with a different config"), so re-running an existing
+ # skill at a new -k failed every combination before a single container started; and the earlier
+ # run's trials are the raw evidence a rescore reads, so overwriting them is worse than the
+ # collision. Both runs now sit side by side.
+ jobs_root = HARBOR_DIR / "jobs" / (skill if attempts == 1 else f"{skill}-k{attempts}")
+ log(f"[harbor] '{skill}': {len(holdout)} held-out tasks × {len(agents)} harness(es), "
+ f"two arms each, {attempts} attempt(s) per task → {jobs_root}")
+
+ rows: dict[str, dict] = {}
+ for entry in agents:
+ # "agent" or "agent@model": the question is which *combination* serves a skill best, so a
+ # row is one harness paired with one model, not a harness alone.
+ agent, _, combo_model = entry.partition("@")
+ row_model = combo_model or model
+ # One broken harness must not discard the arms already paid for.
+ try:
+ jobs_dir = jobs_root / entry.replace("/", "_")
+ # The directory name has had its slashes flattened, so "openai/gpt-5.5" and a model
+ # genuinely named "openai_gpt-5.5" are indistinguishable once written. Rescoring reads
+ # only these directories, and a co-occurrence grid that cannot say which model a row
+ # used is not a co-occurrence grid. Record the pair beside the arms.
+ jobs_dir.mkdir(parents=True, exist_ok=True)
+ (jobs_dir / "combo.json").write_text(json.dumps(
+ {"combination": entry, "harness": agent, "model": row_model or "harness default"}))
+ jobs, answers = {}, {}
+ for arm in ("skill", "control"):
+ jobs[arm] = run_arm(dataset, agent, source if arm == "skill" else None,
+ jobs_dir, arm, row_model, concurrency, attempts, log)
+ answers[arm] = collect_answers(jobs[arm], broken_trials(jobs[arm]))
+ # A task that crashed in either arm is dropped from both, so the two means are always
+ # over the same tasks. Otherwise the difference between them is not lift.
+ skipped = broken_tasks(jobs["skill"]) | broken_tasks(jobs["control"])
+ arms = {arm: score(answers[arm], skill, holdout, skipped) for arm in jobs}
+ s_mean = sum(arms["skill"]) / len(arms["skill"])
+ c_mean = sum(arms["control"]) / len(arms["control"])
+ verdict = ("helps" if s_mean - c_mean > 0.05
+ else "no lift" if s_mean - c_mean >= -0.05 else "HURTS")
+ note = f" [{len(skipped)} task(s) dropped]" if skipped else ""
+ log(f"[harbor] {entry:<38} skill {s_mean:.3f} control {c_mean:.3f} "
+ f"lift {s_mean - c_mean:+.3f} ({verdict}){note}")
+ rows[entry] = {"skill_mean": s_mean, "control_mean": c_mean, "lift": s_mean - c_mean,
+ "skill_scores": arms["skill"], "control_scores": arms["control"],
+ "harness": agent, "model": row_model or "harness default",
+ # Not cosmetic: a lift over 2 tasks and one over 4 are different claims,
+ # and a reader comparing rows has to be able to see which is which.
+ "tasks_scored": len(arms["skill"]), "tasks_dropped": sorted(skipped),
+ "attempts": attempts}
+ except Exception as error: # noqa: BLE001 - any harness failure is one unusable row
+ rows[entry] = {"error": f"{type(error).__name__}: {error}"[:400], "harness": agent,
+ "model": row_model or "harness default"}
+ # The message, not just the type: when every row fails there is no matrix to read the
+ # detail out of, and a log saying only "RuntimeError" cannot be diagnosed at all.
+ log(f"[harbor] {entry:<38} UNAVAILABLE: {str(error)[:300]}")
+ if not any("error" not in row for row in rows.values()):
+ raise SystemExit(f"[harbor] no harness could be run for '{skill}'; nothing was measured.")
+
+ summary = {"skill": skill, "tasks": len(holdout), "pinned_model": model,
+ "judge": os.environ.get("JUDGE_MODELS") or os.environ.get("JUDGE_MODEL", ""),
+ "harnesses": rows}
+ HARBOR_DIR.mkdir(parents=True, exist_ok=True)
+ path = HARBOR_DIR / f"{skill}.json"
+ path.write_text(json.dumps(summary, indent=2))
+ log(f"[harbor] matrix written to {path}")
+ return summary
+
+
+def build_parser() -> argparse.ArgumentParser:
+ parser = argparse.ArgumentParser(description="Sandboxed cross-harness skill evaluation.")
+ parser.add_argument("skill")
+ parser.add_argument("--agent", action="append", default=None,
+ help="harness to run (repeatable); default claude-code")
+ parser.add_argument("--model", default=None, help="pin the model where the harness allows it")
+ parser.add_argument("--target", action="append", default=None, metavar="ALIAS=URL",
+ help="local allowlisted endpoint (repeatable); enables local model sweep")
+ parser.add_argument("-n", "--concurrent", type=int, default=2)
+ parser.add_argument("--global-concurrency", type=int,
+ help="native Harbor global trial cap (default max(16, --concurrent))")
+ parser.add_argument("--endpoint-concurrency", type=int,
+ help="native Harbor per-endpoint cap (default --concurrent)")
+ parser.add_argument("-k", "--attempts", type=int, default=None,
+ help="attempts per task per arm, averaged; 1 is below this eval's noise")
+ parser.add_argument("--canary-only", action="store_true",
+ help="run endpoint preflights and one-task routing canaries without full arms")
+ parser.add_argument("--exploratory", action="store_true",
+ help="allow a one-attempt local sweep that is explicitly not rankable")
+ return parser
+
+
+def main(argv: Sequence[str] | None = None) -> int:
+ parser = build_parser()
+ args = parser.parse_args(argv)
+ if args.canary_only and not args.target:
+ parser.error("--canary-only requires --target")
+ if args.exploratory and not args.target:
+ parser.error("--exploratory requires --target")
+ if args.exploratory and args.attempts != 1:
+ parser.error("--exploratory requires --attempts 1")
+ if args.target:
+ if args.model is not None:
+ parser.error("--target cannot be combined with --model; targets pin their served model")
+ attempts = 3 if args.attempts is None else args.attempts
+ if attempts == 1 and not args.exploratory:
+ parser.error("--attempts 1 requires --exploratory")
+ if attempts not in (1, 3):
+ parser.error("--target requires --attempts 3, or 1 with --exploratory")
+ targets = []
+ for spec in args.target:
+ provisional = parse_target(spec)
+ # parse_target enforces the allowlist and canonical URL before discovery makes a request.
+ targets.append(discover_target(provisional.alias, provisional.base_url))
+ manifest = run_local_sweep(args.skill, targets,
+ harnesses=args.agent or list(LOCAL_HARNESSES),
+ concurrency=args.concurrent, attempts=attempts,
+ global_concurrency=args.global_concurrency,
+ endpoint_concurrency=args.endpoint_concurrency,
+ canary_only=args.canary_only,
+ exploratory=args.exploratory, native_parallel=True, log=print)
+ return 1 if manifest.get("aborted") else 0
+ run_harbor_eval(args.skill, args.agent or ["claude-code"], model=args.model,
+ concurrency=args.concurrent, attempts=args.attempts or 1, log=print)
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/ingot/optimize/harbor_gateway.py b/ingot/optimize/harbor_gateway.py
new file mode 100644
index 0000000..c0d82fd
--- /dev/null
+++ b/ingot/optimize/harbor_gateway.py
@@ -0,0 +1,380 @@
+"""Narrow, runner-owned LiteLLM compatibility gateway for observed role rejections.
+
+The gateway is deliberately not a general provider proxy. It exists only while Harbor evaluates
+the three local harness/target combinations whose native endpoints rejected system/developer
+roles. Its model list has no fallback deployment or provider credential.
+"""
+from __future__ import annotations
+
+import hashlib
+import json
+import os
+import socket
+import subprocess
+import time
+import urllib.error
+import urllib.request
+from dataclasses import dataclass
+from pathlib import Path
+from typing import Any, Callable, Sequence
+
+from .harbor_targets import LocalTarget, scrub_provider_env
+
+
+# Docker's bridge gateway is reachable from Harbor's per-trial bridge networks but not published
+# outside the Dell host. A fixed address/port makes it possible to health-check before Harbor.
+GATEWAY_HOST = "172.17.0.1"
+GATEWAY_PORT = 4865
+GATEWAY_URL = f"http://{GATEWAY_HOST}:{GATEWAY_PORT}"
+GATEWAY_REVISION = "litellm-1.93-role-user-v1"
+DELL_CLAUDE_OUTPUT_CAP_REVISION = "litellm-1.93-role-user-output-v4"
+DELL_CODEX_HTTP_REVISION = "litellm-1.93-role-user-codex-http-catalog-v8"
+# This is an ignored Dell-only environment, created from requirements-harbor-gateway.txt. Harbor
+# and Ingot's primary environment deliberately remain untouched.
+_LITELLM_BIN = str(Path(__file__).resolve().parents[2] / ".venv-harbor-gateway/bin/litellm")
+_CLAUDE_GATEWAY_TARGETS = frozenset({"dell-qwen", "spark-deepseek"})
+
+
+@dataclass(frozen=True)
+class GatewayRoute:
+ harness: str
+ target_alias: str
+ model: str
+ served_model: str
+ upstream_env: str
+ output_limit: int | None = None
+ revision: str = GATEWAY_REVISION
+
+ @property
+ def identity(self) -> str:
+ payload = json.dumps(
+ {"harness": self.harness, "target": self.target_alias, "model": self.model,
+ "served_model": self.served_model, "output_limit": self.output_limit,
+ "revision": self.revision}, sort_keys=True, separators=(",", ":")
+ ).encode()
+ return hashlib.sha256(payload).hexdigest()[:12]
+
+
+def gateway_route(target: LocalTarget, harness: str) -> GatewayRoute | None:
+ """Return a route only for the three observed local role-protocol failures."""
+ if ((harness == "claude-code" and target.alias in _CLAUDE_GATEWAY_TARGETS)
+ or (harness == "codex" and target.alias == "dell-qwen")):
+ # Put the translation revision in the gateway's served model name. A surviving process
+ # from an old run then fails the model-list health check instead of silently reusing
+ # evidence made under a different role conversion.
+ output_limit = None
+ revision = GATEWAY_REVISION
+ if harness == "claude-code" and target.alias == "dell-qwen":
+ # A 1/4 cap left a 24,577-token trajectory one token beyond this target's 32,768
+ # window. Reserve seven eighths for the accumulated prompt and tool history instead.
+ output_limit = target.context_length // 8
+ revision = f"{DELL_CLAUDE_OUTPUT_CAP_REVISION}-{output_limit}"
+ elif harness == "codex":
+ revision = DELL_CODEX_HTTP_REVISION
+ revision_hash = hashlib.sha256(revision.encode()).hexdigest()[:8]
+ slug = f"harbor-compat-{target.alias}-{harness}-{revision_hash}"
+ return GatewayRoute(harness, target.alias, slug, target.served_model,
+ f"HARBOR_GATEWAY_UPSTREAM_{target.alias.upper().replace('-', '_')}",
+ output_limit=output_limit, revision=revision)
+ return None
+
+
+def _system_user(content: Any, label: str) -> dict[str, Any]:
+ if isinstance(content, str):
+ content = f"[{label}]\n{content}"
+ return {"role": "user", "content": content}
+
+
+def normalize_role_request(data: dict[str, Any], call_type: str,
+ output_limits: dict[str, int] | None = None) -> dict[str, Any]:
+ """Convert only roles rejected by the local endpoints, retaining every tool object verbatim."""
+ data = dict(data)
+ if call_type == "anthropic_messages":
+ messages = list(data.get("messages") or [])
+ system = data.pop("system", None)
+ if system not in (None, "", []):
+ messages.insert(0, _system_user(system, "system"))
+ data["messages"] = [
+ _system_user(message.get("content"), message.get("role"))
+ if isinstance(message, dict) and message.get("role") in {"system", "developer"}
+ else message
+ for message in messages
+ ]
+ elif call_type == "aresponses":
+ items = list(data.get("input") or [])
+ instructions = data.pop("instructions", None)
+ if instructions not in (None, "", []):
+ items.insert(0, _system_user(
+ [{"type": "input_text", "text": f"[instructions]\n{instructions}"}], "instructions"))
+ data["input"] = [
+ {**item, "role": "user"}
+ if isinstance(item, dict) and item.get("role") in {"system", "developer"}
+ else item
+ for item in items
+ ]
+ # Codex attaches its model-default reasoning object even when Harbor omits the CLI flag.
+ # LiteLLM's Responses bridge derives the custom backend's rejected reasoning_effort from it.
+ data.pop("reasoning", None)
+ data.pop("reasoning_effort", None)
+ # LiteLLM 1.93's custom_openai Chat Completions route never accepts this Responses-style
+ # key. Every Claude Messages request uses that route; only Dell gets a max_tokens cap.
+ if call_type == "anthropic_messages":
+ data.pop("max_output_tokens", None)
+ output_limit = (output_limits or {}).get(str(data.get("model")))
+ if output_limit is not None:
+ keys = ("max_tokens",) if call_type == "anthropic_messages" else ("max_tokens", "max_output_tokens")
+ for key in keys:
+ value = data.get(key)
+ if value is None or isinstance(value, int) and value > output_limit:
+ data[key] = output_limit
+ return data
+
+
+def gateway_agent_env(target: LocalTarget, route: GatewayRoute) -> dict[str, str]:
+ """Explicit Harbor child settings for a gateway route, never a provider key."""
+ env = {"ANTHROPIC_API_KEY": "local", "OPENAI_API_KEY": "local", "CODEX_API_KEY": "local"}
+ if route.harness == "claude-code":
+ env.update({"ANTHROPIC_BASE_URL": GATEWAY_URL, "ANTHROPIC_MODEL": route.model})
+ elif route.harness == "codex":
+ env.update({"OPENAI_BASE_URL": f"{GATEWAY_URL}/v1", "OPENAI_API_BASE": f"{GATEWAY_URL}/v1",
+ "OPENAI_HOST": GATEWAY_URL, "HARBOR_GATEWAY_CODEX_PROVIDER": "1",
+ "HARBOR_GATEWAY_CODEX_MODEL": route.model,
+ "HARBOR_GATEWAY_CODEX_SERVED_MODEL": route.served_model,
+ "HARBOR_GATEWAY_CODEX_CONTEXT": str(target.context_length)})
+ else: # defensive: callers must not route an unrelated harness through this service
+ raise ValueError(f"unsupported compatibility gateway harness: {route.harness}")
+ return env
+
+
+def gateway_process_env(parent: Mapping[str, str]) -> dict[str, str]:
+ """Expose this checkout only to Harbor when it must import the custom Codex adapter."""
+ root = str(Path(__file__).resolve().parents[2])
+ inherited = parent.get("PYTHONPATH", "")
+ return {**parent, "PYTHONPATH": os.pathsep.join(item for item in (root, inherited) if item)}
+
+
+def gateway_metadata(route: GatewayRoute) -> dict[str, str]:
+ return {"gateway_revision": route.revision, "gateway_identity": route.identity,
+ "gateway_agent": gateway_agent_name(route)}
+
+
+def codex_gateway_setup_command() -> str:
+ """Write the HTTP provider and a truthful local catalog using Codex's own instructions."""
+ return '''set -eu
+test "${HARBOR_GATEWAY_CODEX_PROVIDER:-}" = 1
+test -n "${HARBOR_GATEWAY_CODEX_MODEL:-}"
+test -n "${HARBOR_GATEWAY_CODEX_SERVED_MODEL:-}"
+case "${HARBOR_GATEWAY_CODEX_CONTEXT:-}" in *[!0-9]*|'') exit 1;; esac
+mkdir -p "$CODEX_HOME"
+codex debug models --bundled >"$CODEX_HOME/bundled-models.json"
+python3 <<'PY'
+import json
+import os
+from pathlib import Path
+
+home = Path(os.environ["CODEX_HOME"])
+bundled = json.loads((home / "bundled-models.json").read_text())
+models = bundled.get("models")
+first = models[0] if isinstance(models, list) and models else None
+messages = first.get("model_messages") if isinstance(first, dict) else None
+if not isinstance(messages, dict) or not messages.get("instructions_template"):
+ raise ValueError("Codex bundled catalog has no reusable instruction template")
+context = int(os.environ["HARBOR_GATEWAY_CODEX_CONTEXT"])
+if context < 32768 or context > 9_007_199_254_740_991:
+ raise ValueError("invalid Codex local-model context")
+model = dict(first)
+model.update({
+ "slug": os.environ["HARBOR_GATEWAY_CODEX_MODEL"],
+ "display_name": os.environ["HARBOR_GATEWAY_CODEX_SERVED_MODEL"],
+ "description": "Local Qwen model through the Ingot compatibility gateway",
+ "default_reasoning_level": None,
+ "supported_reasoning_levels": [],
+ "shell_type": "default",
+ "visibility": "none",
+ "supported_in_api": True,
+ "priority": 99,
+ "additional_speed_tiers": [],
+ "service_tiers": [],
+ "default_service_tier": None,
+ "availability_nux": None,
+ "upgrade": None,
+ "include_skills_usage_instructions": False,
+ "include_plugin_usage_instructions": False,
+ "include_apps_usage_instructions": False,
+ "supports_reasoning_summary_parameter": False,
+ "default_reasoning_summary": "none",
+ "support_verbosity": False,
+ "default_verbosity": None,
+ "apply_patch_tool_type": None,
+ "web_search_tool_type": "text",
+ "truncation_policy": {"mode": "bytes", "limit": 10000},
+ "supports_parallel_tool_calls": False,
+ "supports_image_detail_original": False,
+ "context_window": context,
+ "max_context_window": context,
+ "auto_compact_token_limit": None,
+ "comp_hash": os.environ["HARBOR_GATEWAY_CODEX_MODEL"],
+ "effective_context_window_percent": 95,
+ "experimental_supported_tools": [],
+ "input_modalities": ["text"],
+ "supports_search_tool": False,
+ "use_responses_lite": False,
+ "auto_review_model_override": None,
+ "model_specialty": None,
+ "tool_mode": None,
+ "multi_agent_version": None,
+})
+(home / "model-catalog.json").write_text(json.dumps({"models": [model]}, separators=(",", ":")))
+PY
+cat >>"$CODEX_HOME/config.toml" < str:
+ return "ingot.optimize.harbor_codex_gateway:GatewayCodex" if route.harness == "codex" else route.harness
+
+
+def _callback_source(routes: Sequence[GatewayRoute]) -> str:
+ output_limits = {route.model: route.output_limit for route in routes if route.output_limit is not None}
+ return '''from litellm.integrations.custom_logger import CustomLogger
+from ingot.optimize.harbor_gateway import normalize_role_request
+
+OUTPUT_LIMITS = ''' + repr(output_limits) + '''
+
+class GatewayRoleNormalizer(CustomLogger):
+ async def async_pre_call_hook(self, user_api_key_dict, cache, data, call_type):
+ return normalize_role_request(data, call_type, OUTPUT_LIMITS)
+
+gateway_role_normalizer = GatewayRoleNormalizer()
+'''
+
+
+def _gateway_config(routes: Sequence[GatewayRoute]) -> dict[str, Any]:
+ return {
+ "model_list": [
+ {"model_name": route.model,
+ "litellm_params": {"model": f"custom_openai/{route.served_model}",
+ "api_base": f"os.environ/{route.upstream_env}", "api_key": "local"}}
+ for route in routes
+ ],
+ "router_settings": {"num_retries": 0, "allowed_fails": 0},
+ "litellm_settings": {"callbacks": ["role_normalizer.gateway_role_normalizer"]},
+ "general_settings": {"master_key": "local"},
+ }
+
+
+def _fixed_bind_occupied() -> bool:
+ """Bound the stale-listener probe; any prior listener invalidates this gateway session."""
+ try:
+ connection = socket.create_connection((GATEWAY_HOST, GATEWAY_PORT), timeout=0.2)
+ except OSError:
+ return False
+ connection.close()
+ return True
+
+
+class GatewaySession:
+ """One LiteLLM process per canary invocation; writes an untracked cleanup receipt on exit."""
+ def __init__(self, routes: Sequence[tuple[GatewayRoute, LocalTarget]], runtime_dir: Path, *,
+ litellm_bin: str = _LITELLM_BIN,
+ popen: Callable[..., subprocess.Popen] = subprocess.Popen,
+ health_timeout: float = 20.0):
+ self.routes = tuple(routes)
+ self.runtime_dir = runtime_dir
+ self.litellm_bin = litellm_bin
+ self.popen = popen
+ self.health_timeout = health_timeout
+ self.process: subprocess.Popen | None = None
+
+ def _write_files(self) -> Path:
+ self.runtime_dir.mkdir(parents=True, exist_ok=True)
+ (self.runtime_dir / "role_normalizer.py").write_text(
+ _callback_source([route for route, _ in self.routes]), encoding="utf-8")
+ config = self.runtime_dir / "config.json"
+ config.write_text(json.dumps(_gateway_config([route for route, _ in self.routes]), indent=2), encoding="utf-8")
+ return config
+
+ def _healthy_models(self) -> bool:
+ request = urllib.request.Request(
+ f"{GATEWAY_URL}/v1/models", headers={"Authorization": "Bearer local"}, method="GET")
+ try:
+ with urllib.request.urlopen(request, timeout=1.0) as response:
+ if not 200 <= int(getattr(response, "status", 200)) < 300:
+ return False
+ payload = json.loads(response.read())
+ except (urllib.error.URLError, TimeoutError, OSError, ValueError):
+ return False
+ expected = {route.model for route, _ in self.routes}
+ actual = {str(item.get("id")) for item in payload.get("data", [])
+ if isinstance(item, dict)}
+ return expected == actual
+
+ def start(self) -> None:
+ if self.process is not None:
+ return
+ try:
+ if not self.routes:
+ raise ValueError("compatibility gateway needs at least one route")
+ if not Path(self.litellm_bin).is_file():
+ raise RuntimeError(
+ "compatibility gateway runtime is missing; create .venv-harbor-gateway from "
+ "ingot/optimize/requirements-harbor-gateway.txt")
+ if _fixed_bind_occupied():
+ raise RuntimeError("compatibility gateway bind is already occupied")
+ config = self._write_files()
+ env = gateway_process_env(scrub_provider_env(os.environ))
+ for route, target in self.routes:
+ env[route.upstream_env] = f"{target.base_url}/v1"
+ self.process = self.popen(
+ [self.litellm_bin, "--config", str(config), "--host", GATEWAY_HOST,
+ "--port", str(GATEWAY_PORT)],
+ cwd=Path.cwd(), env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
+ )
+ deadline = time.monotonic() + self.health_timeout
+ while time.monotonic() < deadline:
+ if self.process.poll() is not None:
+ raise RuntimeError("compatibility gateway exited before health check")
+ try:
+ with urllib.request.urlopen(f"{GATEWAY_URL}/health/liveliness", timeout=1.0) as response:
+ if (200 <= int(getattr(response, "status", 200)) < 300
+ and self._healthy_models() and self.process.poll() is None):
+ return
+ except (urllib.error.URLError, TimeoutError, OSError):
+ time.sleep(0.2)
+ raise RuntimeError("compatibility gateway did not become healthy")
+ except Exception:
+ self.close("failed-start")
+ raise
+
+ def close(self, reason: str = "completed") -> None:
+ if self.process is not None and self.process.poll() is None:
+ self.process.terminate()
+ try:
+ self.process.wait(timeout=10)
+ except subprocess.TimeoutExpired:
+ self.process.kill()
+ self.process.wait(timeout=5)
+ self.runtime_dir.mkdir(parents=True, exist_ok=True)
+ (self.runtime_dir / "cleanup.json").write_text(json.dumps({
+ "gateway_revisions": sorted({route.revision for route, _ in self.routes}),
+ "routes": [route.identity for route, _ in self.routes],
+ "reason": reason,
+ "stopped": True,
+ }, indent=2), encoding="utf-8")
+
+ def __enter__(self) -> "GatewaySession":
+ self.start()
+ return self
+
+ def __exit__(self, *unused: object) -> None:
+ self.close()
diff --git a/ingot/optimize/harbor_langfuse.py b/ingot/optimize/harbor_langfuse.py
new file mode 100644
index 0000000..9c09ac8
--- /dev/null
+++ b/ingot/optimize/harbor_langfuse.py
@@ -0,0 +1,944 @@
+"""Export persisted Harbor attempts to Langfuse with public read-back receipts."""
+from __future__ import annotations
+
+import argparse
+import hashlib
+import json
+import os
+import re
+import stat
+import tempfile
+import time
+from collections.abc import Mapping, Sequence
+from pathlib import Path
+
+import httpx
+
+from .harbor_redaction import _redact_harbor_receipt_output, _redact_persisted
+from .harbor_native import NativeTrialIdentity, read_trial_identity
+
+
+EXPORTER_REVISION = "harbor-langfuse-v3"
+RECEIPT_NAME = "langfuse-receipt.json"
+LEGACY_V2_RECEIPT_NAME = "langfuse-receipt.v2.json"
+_METADATA_FIELDS = (
+ "combination", "harness", "model", "target_alias", "endpoint_fingerprint", "protocol",
+ "task_fingerprint", "attempts", "gateway_revision", "gateway_identity", "gateway_agent",
+ "arm", "canary", "skill", "skill_sha256",
+)
+_MAX_TEXT_BYTES = 64 * 1024
+# A retained Goose full-arm trajectory reached 3,368,055 bytes. Four MiB leaves measured headroom;
+# the outbound projection still retains only allowlisted timestamps, sources, and numeric totals.
+_MAX_TRAJECTORY_BYTES = 4 * 1024 * 1024
+_MAX_ARTIFACT_PATHS = 128
+_MAX_ARTIFACT_SCAN_PATHS = 4096
+_MAX_ARTIFACT_DEPTH = 16
+_MAX_ARTIFACT_BYTES = 128 * 1024
+_READBACK_ATTEMPTS = 5
+_READBACK_POLL_SECONDS = 1.0
+_TEXT_SUFFIXES = {
+ ".c", ".cc", ".cpp", ".css", ".csv", ".h", ".html", ".ini", ".java", ".js",
+ ".json", ".jsx", ".log", ".md", ".py", ".rb", ".rs", ".rst", ".sh", ".sql",
+ ".toml", ".ts", ".tsx", ".txt", ".xml", ".yaml", ".yml",
+}
+_TEXT_NAMES = {"Dockerfile", "Makefile"}
+_STRUCTURAL_EXCLUSIONS = {".env", "config", "credentials", "env", "environment", "secrets"}
+_SENSITIVE_ARTIFACT_STEMS = {
+ "config", "configuration", "credential", "credentials", "endpoint", "endpoints",
+ "env", "environment", "secret", "secrets",
+}
+
+
+class TelemetryReceiptError(RuntimeError):
+ """Persisted Harbor evidence lacks a matching, publicly readable trace receipt."""
+
+
+_DIRECTORY_FLAGS = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0)
+
+
+def _same_file(first: os.stat_result, second: os.stat_result) -> bool:
+ return (first.st_dev, first.st_ino, stat.S_IFMT(first.st_mode)) == (
+ second.st_dev, second.st_ino, stat.S_IFMT(second.st_mode))
+
+
+class _AttemptReader:
+ """Read only the fixed Harbor attempt layout through job-relative descriptors."""
+
+ def __init__(self, trial: Path) -> None:
+ self.trial = trial
+ self._job_fd = -1
+ self._trial_fd = -1
+ self._artifact_paths = 0
+ self._artifact_files = 0
+ self._artifact_bytes = 0
+ self._artifact_omitted_files = 0
+ self._artifact_omitted_hidden = 0
+ self._artifact_scan_truncated = False
+
+ def __enter__(self) -> "_AttemptReader":
+ try:
+ job_info = self.trial.parent.lstat()
+ if stat.S_ISLNK(job_info.st_mode) or not stat.S_ISDIR(job_info.st_mode):
+ raise TelemetryReceiptError(f"{self.trial.parent} is not a real job directory")
+ self._job_fd = os.open(self.trial.parent, _DIRECTORY_FLAGS)
+ if not _same_file(job_info, os.fstat(self._job_fd)):
+ raise TelemetryReceiptError(f"{self.trial.parent} changed during admission")
+ trial_info = os.stat(self.trial.name, dir_fd=self._job_fd, follow_symlinks=False)
+ if stat.S_ISLNK(trial_info.st_mode) or not stat.S_ISDIR(trial_info.st_mode):
+ raise TelemetryReceiptError(f"{self.trial} is a symlink or non-directory attempt")
+ self._trial_fd = os.open(self.trial.name, _DIRECTORY_FLAGS, dir_fd=self._job_fd)
+ if not _same_file(trial_info, os.fstat(self._trial_fd)):
+ raise TelemetryReceiptError(f"{self.trial} changed during admission")
+ return self
+ except (OSError, TelemetryReceiptError):
+ self.close()
+ raise
+
+ def __exit__(self, *_args) -> None:
+ self.close()
+
+ def close(self) -> None:
+ for descriptor in (self._trial_fd, self._job_fd):
+ if descriptor >= 0:
+ os.close(descriptor)
+ self._trial_fd = self._job_fd = -1
+
+ def _open_dir(self, parent_fd: int, name: str, *, missing_ok: bool = False) -> tuple[int, os.stat_result] | None:
+ try:
+ before = os.stat(name, dir_fd=parent_fd, follow_symlinks=False)
+ except FileNotFoundError:
+ if missing_ok:
+ return None
+ raise
+ if stat.S_ISLNK(before.st_mode) or not stat.S_ISDIR(before.st_mode):
+ raise TelemetryReceiptError(f"attempt directory {name!r} failed admission")
+ try:
+ descriptor = os.open(name, _DIRECTORY_FLAGS, dir_fd=parent_fd)
+ except OSError as error:
+ raise TelemetryReceiptError(f"attempt directory {name!r} failed admission") from error
+ if not _same_file(before, os.fstat(descriptor)):
+ os.close(descriptor)
+ raise TelemetryReceiptError(f"attempt directory {name!r} changed during admission")
+ return descriptor, before
+
+ @staticmethod
+ def _verify_entry(parent_fd: int, name: str, opened: os.stat_result) -> None:
+ try:
+ current = os.stat(name, dir_fd=parent_fd, follow_symlinks=False)
+ except OSError as error:
+ raise TelemetryReceiptError(f"attempt entry {name!r} changed during admission") from error
+ if not _same_file(opened, current):
+ raise TelemetryReceiptError(f"attempt entry {name!r} changed during admission")
+
+ def _read_at(self, parent_fd: int, name: str, *, artifact: bool = False,
+ limit: int = _MAX_TEXT_BYTES) -> bytes | None:
+ try:
+ before = os.stat(name, dir_fd=parent_fd, follow_symlinks=False)
+ except FileNotFoundError:
+ return None
+ if stat.S_ISLNK(before.st_mode) or not stat.S_ISREG(before.st_mode):
+ raise TelemetryReceiptError(f"attempt entry {name!r} is non-regular")
+ if before.st_size > limit:
+ raise TelemetryReceiptError(f"attempt artifact byte budget exceeded by {name!r}")
+ try:
+ descriptor = os.open(
+ name, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0), dir_fd=parent_fd)
+ except OSError as error:
+ raise TelemetryReceiptError(f"attempt entry {name!r} failed admission") from error
+ try:
+ opened = os.fstat(descriptor)
+ if not stat.S_ISREG(opened.st_mode) or not _same_file(before, opened):
+ raise TelemetryReceiptError(f"attempt entry {name!r} changed during admission")
+ chunks = []
+ remaining = limit + 1
+ while remaining:
+ chunk = os.read(descriptor, remaining)
+ if not chunk:
+ break
+ chunks.append(chunk)
+ remaining -= len(chunk)
+ data = b"".join(chunks)
+ finally:
+ os.close(descriptor)
+ self._verify_entry(parent_fd, name, before)
+ if len(data) > limit:
+ raise TelemetryReceiptError(f"attempt artifact byte budget exceeded by {name!r}")
+ return data
+
+ def read(self, *parts: str, limit: int = _MAX_TEXT_BYTES) -> bytes | None:
+ descriptor = os.dup(self._trial_fd)
+ opened_dirs: list[tuple[int, str, os.stat_result]] = []
+ try:
+ for part in parts[:-1]:
+ opened = self._open_dir(descriptor, part, missing_ok=True)
+ if opened is None:
+ return None
+ child_fd, child_info = opened
+ opened_dirs.append((descriptor, part, child_info))
+ descriptor = child_fd
+ return self._read_at(descriptor, parts[-1], limit=limit)
+ finally:
+ os.close(descriptor)
+ for parent_fd, name, opened in reversed(opened_dirs):
+ self._verify_entry(parent_fd, name, opened)
+ os.close(parent_fd)
+
+ def read_json(self, *parts: str, limit: int = _MAX_TEXT_BYTES) -> dict:
+ data = self.read(*parts, limit=limit)
+ if data is None:
+ return {}
+ try:
+ value = json.loads(data.decode("utf-8"))
+ except (UnicodeDecodeError, ValueError):
+ return {}
+ return value if isinstance(value, dict) else {}
+
+ def artifacts(self, *parts: str, skip_solution: bool = False,
+ skill_body: bytes | None = None,
+ first: Sequence[str] = ()) -> dict[str, str]:
+ descriptor = os.dup(self._trial_fd)
+ opened_dirs: list[tuple[int, str, os.stat_result]] = []
+ try:
+ for part in parts:
+ opened = self._open_dir(descriptor, part, missing_ok=True)
+ if opened is None:
+ return {}
+ child_fd, child_info = opened
+ opened_dirs.append((descriptor, part, child_info))
+ descriptor = child_fd
+ found: list[tuple[str, str]] = []
+ for name in first:
+ try:
+ info = os.stat(name, dir_fd=descriptor, follow_symlinks=False)
+ except FileNotFoundError:
+ continue
+ self._project_artifact(descriptor, name, Path(name), info, found, skill_body)
+ self._walk_artifacts(descriptor, (), found, skip_solution, skill_body, depth=0,
+ excluded=frozenset(first))
+ return dict(sorted(found))
+ finally:
+ os.close(descriptor)
+ for parent_fd, name, opened in reversed(opened_dirs):
+ self._verify_entry(parent_fd, name, opened)
+ os.close(parent_fd)
+
+ def artifact_projection(self) -> dict[str, int | bool]:
+ """Describe a bounded omission without turning optional telemetry into failed evidence."""
+ return {
+ "exported_files": self._artifact_files,
+ "exported_bytes": self._artifact_bytes,
+ "omitted_files": self._artifact_omitted_files,
+ "omitted_hidden_paths": self._artifact_omitted_hidden,
+ "scan_truncated": self._artifact_scan_truncated,
+ }
+
+ def _project_artifact(self, descriptor: int, name: str, path: Path,
+ info: os.stat_result, found: list[tuple[str, str]],
+ skill_body: bytes | None) -> None:
+ if stat.S_ISLNK(info.st_mode):
+ raise TelemetryReceiptError(f"attempt artifact {name!r} failed admission")
+ if not stat.S_ISREG(info.st_mode):
+ raise TelemetryReceiptError(f"attempt artifact {name!r} is non-regular")
+ if name == RECEIPT_NAME:
+ return
+ lowered = {part.lower() for part in path.parts}
+ if (lowered & _STRUCTURAL_EXCLUSIONS
+ or path.stem.lower() in _SENSITIVE_ARTIFACT_STEMS):
+ return
+ if path.suffix.lower() not in _TEXT_SUFFIXES and name not in _TEXT_NAMES:
+ return
+ if (info.st_size > _MAX_TEXT_BYTES
+ or self._artifact_files >= _MAX_ARTIFACT_PATHS
+ or info.st_size > _MAX_ARTIFACT_BYTES - self._artifact_bytes):
+ self._artifact_omitted_files += 1
+ return
+ data = self._read_at(descriptor, name)
+ assert data is not None
+ if skill_body and skill_body in data:
+ self._artifact_omitted_files += 1
+ return
+ if any(byte < 32 and byte not in (9, 10, 13) or byte == 127 for byte in data):
+ self._artifact_omitted_files += 1
+ return
+ try:
+ value = data.decode("utf-8")
+ except UnicodeDecodeError:
+ self._artifact_omitted_files += 1
+ return
+ found.append((path.as_posix(), _redact_harbor_receipt_output(value, {})))
+ self._artifact_files += 1
+ self._artifact_bytes += len(data)
+
+ def _walk_artifacts(self, descriptor: int, relative: tuple[str, ...],
+ found: list[tuple[str, str]], skip_solution: bool,
+ skill_body: bytes | None, *, depth: int,
+ excluded: frozenset[str] = frozenset()) -> None:
+ if self._artifact_scan_truncated:
+ return
+ if depth > _MAX_ARTIFACT_DEPTH:
+ self._artifact_omitted_files += 1
+ return
+ names = []
+ with os.scandir(descriptor) as entries:
+ for entry in entries:
+ if self._artifact_paths >= _MAX_ARTIFACT_SCAN_PATHS:
+ self._artifact_scan_truncated = True
+ break
+ self._artifact_paths += 1
+ # Harbor and harnesses may retain caches/backups beside the submitted solution.
+ # They are never evidence. Skip hidden entries at every depth before stat or
+ # traversal, so even a hidden symlink cannot redirect the exporter.
+ if entry.name.startswith("."):
+ self._artifact_omitted_hidden += 1
+ continue
+ names.append(entry.name)
+ for name in sorted(names, key=lambda item: (item != "answer.md", item)):
+ if self._artifact_scan_truncated:
+ return
+ if depth == 0 and name in excluded:
+ continue
+ path_parts = (*relative, name)
+ info = os.stat(name, dir_fd=descriptor, follow_symlinks=False)
+ if stat.S_ISLNK(info.st_mode):
+ raise TelemetryReceiptError(f"attempt artifact {name!r} failed admission")
+ if stat.S_ISDIR(info.st_mode):
+ if skip_solution and path_parts == ("solution",):
+ continue
+ opened = self._open_dir(descriptor, name)
+ assert opened is not None
+ child_fd, child_info = opened
+ try:
+ self._walk_artifacts(child_fd, path_parts, found, skip_solution, skill_body,
+ depth=depth + 1)
+ finally:
+ os.close(child_fd)
+ self._verify_entry(descriptor, name, child_info)
+ continue
+ self._project_artifact(descriptor, name, Path(*path_parts), info, found, skill_body)
+
+
+def _path_json(path: Path, limit: int = _MAX_TEXT_BYTES) -> dict:
+ try:
+ info = path.lstat()
+ if stat.S_ISLNK(info.st_mode) or not stat.S_ISREG(info.st_mode) or info.st_size > limit:
+ return {}
+ descriptor = os.open(path, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0))
+ try:
+ data = os.read(descriptor, limit + 1)
+ finally:
+ os.close(descriptor)
+ value = json.loads(data.decode("utf-8"))
+ except (OSError, UnicodeDecodeError, ValueError):
+ return {}
+ return value if isinstance(value, dict) else {}
+
+
+def load_provenance_metadata(path: Path) -> dict:
+ """Load one explicit bounded migration document used at export and publication."""
+ metadata = _path_json(path)
+ if not metadata:
+ raise TelemetryReceiptError(f"{path} has no readable provenance metadata")
+ return metadata
+
+
+def _clean_string(value: object) -> str:
+ # Canonical payload identity cannot depend on which credentials happen to be active now.
+ return _redact_harbor_receipt_output(value, {})
+
+
+def _metadata(metadata: Mapping[str, object]) -> dict:
+ clean = {}
+ for key in _METADATA_FIELDS:
+ value = metadata.get(key)
+ if isinstance(value, str):
+ clean[key] = _clean_string(value)
+ elif isinstance(value, (int, float, bool)) or value is None:
+ if key in metadata:
+ clean[key] = value
+ return clean
+
+
+def _trajectory(reader: _AttemptReader) -> dict:
+ record = reader.read_json(
+ "agent", "trajectory.json", limit=_MAX_TRAJECTORY_BYTES)
+ if not record:
+ return {}
+ result = {"steps": []}
+ raw_steps = record.get("steps", [])
+ if not isinstance(raw_steps, list):
+ raise TelemetryReceiptError("trajectory steps must be a list")
+ for raw in raw_steps:
+ if not isinstance(raw, Mapping):
+ continue
+ step = {}
+ if raw.get("source") in ("user", "agent"):
+ step["source"] = raw["source"]
+ timestamp = raw.get("timestamp")
+ if isinstance(timestamp, str) and re.fullmatch(
+ r"\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d(?:\.\d+)?Z", timestamp):
+ step["timestamp"] = timestamp
+ result["steps"].append(step)
+ if isinstance(record.get("final_metrics"), Mapping):
+ final_metrics = record["final_metrics"]
+ result["final_metrics"] = {}
+ for key in (
+ "total_prompt_tokens", "total_cached_tokens",
+ "total_completion_tokens", "total_cost_usd",
+ ):
+ value = final_metrics.get(key)
+ if isinstance(value, (int, float)) and not isinstance(value, bool):
+ result["final_metrics"][key] = value
+ return result
+
+
+def _exception_evidence(reader: _AttemptReader,
+ result: Mapping[str, object]) -> tuple[str | None, str | None]:
+ info = result.get("exception_info")
+ if isinstance(info, Mapping) and info.get("exception_type"):
+ detail = _clean_string(info.get("exception_message")) or None
+ return _clean_string(info["exception_type"]), detail
+ data = reader.read("exception.txt")
+ if data:
+ try:
+ exception = _redact_harbor_receipt_output(data.decode("utf-8"), {})
+ except UnicodeDecodeError:
+ return None, None
+ matches = re.findall(r"(?:^|\n)([A-Za-z_][A-Za-z0-9_.]*)(?=:\s)", exception)
+ detail = exception.rsplit("\n", 1)[-1]
+ return (matches[-1] if matches else "Exception"), detail
+ return None, None
+
+
+def _task_slug(task_name: object) -> str:
+ return str(task_name or "").split("/")[-1].split("__", 1)[0]
+
+
+def _task_text(metadata: Mapping[str, object], task_name: object) -> str:
+ texts = metadata.get("task_texts")
+ if not isinstance(texts, Mapping):
+ return ""
+ return _clean_string(texts.get(_task_slug(task_name)))
+
+
+def _timestamps(result: Mapping[str, object], trajectory: Mapping[str, object]) -> dict[str, str]:
+ timestamps = {}
+ for source, destination in (("started_at", "started_at"), ("finished_at", "finished_at")):
+ value = result.get(source)
+ if isinstance(value, str) and re.fullmatch(r"\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d(?:\.\d+)?Z", value):
+ timestamps[destination] = value
+ info = result.get("exception_info")
+ occurred_at = info.get("occurred_at") if isinstance(info, Mapping) else None
+ if (isinstance(occurred_at, str)
+ and re.fullmatch(r"\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d(?:\.\d+)?Z", occurred_at)):
+ timestamps["error_at"] = occurred_at
+ step_times = [step.get("timestamp") for step in trajectory.get("steps") or []
+ if isinstance(step, Mapping) and isinstance(step.get("timestamp"), str)]
+ if step_times:
+ timestamps.setdefault("started_at", step_times[0])
+ timestamps.setdefault("finished_at", step_times[-1])
+ return timestamps
+
+
+def _known_skill_body(metadata: Mapping[str, object]) -> bytes | None:
+ body = metadata.get("skill_body")
+ digest = metadata.get("skill_sha256")
+ if body in (None, "") and digest in (None, ""):
+ return None
+ if not isinstance(body, str) or not isinstance(digest, str):
+ raise TelemetryReceiptError("skill provenance is incomplete")
+ encoded = body.encode("utf-8")
+ if hashlib.sha256(encoded).hexdigest() != digest:
+ raise TelemetryReceiptError("skill provenance SHA-256 does not match its body")
+ return encoded
+
+
+def _usage(result: Mapping[str, object], trajectory: Mapping[str, object]) -> dict:
+ agent_result = result.get("agent_result")
+ agent_result = agent_result if isinstance(agent_result, Mapping) else {}
+ fields = {
+ "input_tokens": agent_result.get("n_input_tokens"),
+ "cache_tokens": agent_result.get("n_cache_tokens"),
+ "output_tokens": agent_result.get("n_output_tokens"),
+ "cost_usd": agent_result.get("cost_usd"),
+ }
+ final = trajectory.get("final_metrics")
+ if isinstance(final, Mapping):
+ fields = {
+ "input_tokens": fields["input_tokens"] or final.get("total_prompt_tokens"),
+ "cache_tokens": fields["cache_tokens"] or final.get("total_cached_tokens"),
+ "output_tokens": fields["output_tokens"] or final.get("total_completion_tokens"),
+ "cost_usd": fields["cost_usd"] or final.get("total_cost_usd"),
+ }
+ return {key: value for key, value in fields.items()
+ if isinstance(value, (int, float)) and not isinstance(value, bool)}
+
+
+def build_attempt_payload(trial: Path, metadata: Mapping[str, object]) -> dict:
+ """Return the bounded, normalized telemetry payload for one persisted Harbor trial."""
+ with _AttemptReader(trial) as reader:
+ result = reader.read_json("result.json")
+ trajectory = _trajectory(reader)
+ exception_category, error_detail = _exception_evidence(reader, result)
+ skill_body = _known_skill_body(metadata)
+ solution_artifacts = reader.artifacts(
+ "verifier", "solution", skill_body=skill_body, first=("answer.md",))
+ verifier_output = reader.artifacts(
+ "verifier", skip_solution=True, skill_body=skill_body)
+ artifact_projection = reader.artifact_projection()
+ agent_info = result.get("agent_info")
+ agent_info = agent_info if isinstance(agent_info, Mapping) else {}
+ model_info = agent_info.get("model_info")
+ model_info = model_info if isinstance(model_info, Mapping) else {}
+ model = metadata.get("model") or model_info.get("name")
+ task_name = result.get("task_name") or trial.name.split("__", 1)[0]
+ terminal_success = (bool(result)
+ and isinstance(result.get("finished_at"), str)
+ and bool(result["finished_at"].strip())
+ and result.get("exception_info") is None)
+ payload = {
+ "exporter_revision": EXPORTER_REVISION,
+ "attempt": {
+ "id": _clean_string(result.get("id") or trial.name),
+ "trial_name": _clean_string(result.get("trial_name") or trial.name),
+ },
+ "task": {
+ "name": _clean_string(task_name),
+ "checksum": _clean_string(result.get("task_checksum")),
+ "source": _clean_string(result.get("source")),
+ "text": _task_text(metadata, task_name),
+ },
+ "skill": _clean_string(metadata.get("skill")),
+ "trajectory": trajectory,
+ "verifier_output": verifier_output,
+ "solution_artifacts": solution_artifacts,
+ "status": "failed" if exception_category else "succeeded" if terminal_success else "incomplete",
+ "exception_category": exception_category,
+ "error_detail": error_detail,
+ "timestamps": _timestamps(result, trajectory),
+ "usage": _usage(result, trajectory),
+ "model": _clean_string(model),
+ "metadata": _metadata(metadata),
+ }
+ if (artifact_projection["omitted_files"] or artifact_projection["omitted_hidden_paths"]
+ or artifact_projection["scan_truncated"]):
+ payload["artifact_projection"] = artifact_projection
+ return payload
+
+
+def _payload_sha256(payload: Mapping[str, object]) -> str:
+ encoded = json.dumps(payload, sort_keys=True, separators=(",", ":"),
+ ensure_ascii=False).encode("utf-8")
+ return hashlib.sha256(encoded).hexdigest()
+
+
+def _atomic_receipt(path: Path, receipt: Mapping[str, object]) -> None:
+ path.parent.mkdir(parents=True, exist_ok=True)
+ descriptor, name = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=path.parent)
+ temporary = Path(name)
+ try:
+ with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
+ json.dump(receipt, handle, indent=2)
+ handle.flush()
+ os.fsync(handle.fileno())
+ os.replace(temporary, path)
+ except BaseException:
+ temporary.unlink(missing_ok=True)
+ raise
+
+
+def _expected_trace_id(digest: str) -> str:
+ from langfuse import Langfuse
+ return Langfuse.create_trace_id(seed=f"{EXPORTER_REVISION}:{digest}")
+
+
+def _expected_receipt(digest: str) -> dict:
+ return {
+ "status": "verified",
+ "trace_id": _expected_trace_id(digest),
+ "payload_sha256": digest,
+ "exporter_revision": EXPORTER_REVISION,
+ }
+
+
+def _pending_receipt(digest: str, observation_id: str | None = None) -> dict:
+ receipt = {
+ "status": "pending",
+ "trace_id": _expected_trace_id(digest),
+ "payload_sha256": digest,
+ "exporter_revision": EXPORTER_REVISION,
+ }
+ if observation_id:
+ receipt["observation_id"] = observation_id
+ return receipt
+
+
+def _read_receipt(trial: Path) -> dict:
+ with _AttemptReader(trial) as reader:
+ return reader.read_json(RECEIPT_NAME)
+
+
+def _verified_receipt(trial: Path, digest: str) -> dict | None:
+ receipt = _read_receipt(trial)
+ if receipt == _expected_receipt(digest):
+ return receipt
+ return None
+
+
+def _create_pending_receipt(path: Path, receipt: Mapping[str, object]) -> bool:
+ encoded = json.dumps(receipt, separators=(",", ":")).encode("utf-8")
+ try:
+ descriptor = os.open(
+ path,
+ os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0),
+ 0o600,
+ )
+ except FileExistsError:
+ return False
+ try:
+ os.write(descriptor, encoded)
+ os.fsync(descriptor)
+ finally:
+ os.close(descriptor)
+ return True
+
+
+def _read_back(trace_id: str, client) -> dict | None:
+ if hasattr(client, "read_trace"):
+ value = client.read_trace(trace_id)
+ return value if isinstance(value, dict) else None
+ public_key = os.environ.get("LANGFUSE_PUBLIC_KEY", "").strip()
+ secret_key = os.environ.get("LANGFUSE_SECRET_KEY", "").strip()
+ base_url = os.environ.get("LANGFUSE_BASE_URL", "http://langfuse-web:3000").rstrip("/")
+ if not public_key or not secret_key:
+ raise TelemetryReceiptError("Langfuse public read-back credentials are missing")
+ response = httpx.get(f"{base_url}/api/public/traces/{trace_id}",
+ auth=(public_key, secret_key), timeout=15)
+ if response.status_code == 404:
+ return None
+ if response.status_code != 200:
+ raise TelemetryReceiptError(
+ f"Langfuse public read-back returned status {response.status_code}")
+ value = response.json()
+ return value if isinstance(value, dict) else None
+
+
+def _telemetry_evidence(payload: Mapping[str, object]) -> dict:
+ usage = payload.get("usage") if isinstance(payload.get("usage"), Mapping) else {}
+ return {
+ "revision": "harbor-agent-evidence-v1",
+ "model": payload.get("model") or "",
+ "status": payload.get("status") or "incomplete",
+ "tokens": {key: usage[key] for key in (
+ "input_tokens", "cache_tokens", "output_tokens") if key in usage},
+ }
+
+
+def _matching_observation_id(value: dict | None, trace_id: str, digest: str,
+ evidence: Mapping[str, object],
+ observation_id: str | None = None, *,
+ exporter_revision: str = EXPORTER_REVISION) -> str | None:
+ if not value or value.get("id") != trace_id:
+ return None
+ observations = value.get("observations")
+ if not isinstance(observations, list) or len(observations) != 1:
+ return None
+ observation = observations[0]
+ if not isinstance(observation, Mapping):
+ return None
+ metadata = observation.get("metadata")
+ actual_id = observation.get("id")
+ if (not isinstance(actual_id, str) or not actual_id.strip()
+ or (observation_id is not None and actual_id != observation_id)
+ or observation.get("name") != "harbor-attempt"
+ or str(observation.get("type") or "").lower() != "agent"
+ or not isinstance(metadata, Mapping)
+ or metadata.get("payload_sha256") != digest
+ or metadata.get("exporter_revision") != exporter_revision
+ or metadata.get("telemetry_evidence") != evidence):
+ return None
+ return actual_id
+
+
+def _wait_for_matching_observation(trace_id: str, digest: str,
+ evidence: Mapping[str, object],
+ observation_id: str, client) -> bool:
+ for attempt in range(_READBACK_ATTEMPTS):
+ if _matching_observation_id(
+ _read_back(trace_id, client), trace_id, digest, evidence, observation_id):
+ return True
+ if attempt + 1 < _READBACK_ATTEMPTS:
+ time.sleep(_READBACK_POLL_SECONDS)
+ return False
+
+
+def _client():
+ from langfuse import Langfuse
+ return Langfuse()
+
+
+def _resume_pending(trial: Path, current: Mapping[str, object], pending: dict,
+ trace_id: str, digest: str, evidence: Mapping[str, object], client) -> dict:
+ if current == pending:
+ raise TelemetryReceiptError(
+ f"Langfuse trace {trace_id} has unknown pending observation state")
+ observation_id = current.get("observation_id")
+ if not (current.get("status") == "pending"
+ and current.get("trace_id") == trace_id
+ and current.get("payload_sha256") == digest
+ and current.get("exporter_revision") == EXPORTER_REVISION
+ and isinstance(observation_id, str)
+ and bool(observation_id.strip())):
+ raise TelemetryReceiptError(f"{trial} has a stale or malformed Langfuse receipt")
+ if _wait_for_matching_observation(trace_id, digest, evidence, observation_id, client):
+ receipt = _expected_receipt(digest)
+ _atomic_receipt(trial / RECEIPT_NAME, receipt)
+ return receipt
+ raise TelemetryReceiptError(
+ f"Langfuse trace {trace_id} remains pending public read-back")
+
+
+def _migrate_verified_v2_receipt(trial: Path, current: Mapping[str, object], pending: dict,
+ evidence: Mapping[str, object], client) -> bool:
+ """Preserve one exact verified v2 receipt before authorizing telemetry-only v3 export."""
+ from langfuse import Langfuse
+
+ digest = current.get("payload_sha256")
+ trace_id = current.get("trace_id")
+ if not (current.get("status") == "verified"
+ and current.get("exporter_revision") == "harbor-langfuse-v2"
+ and isinstance(digest, str) and re.fullmatch(r"[0-9a-f]{64}", digest)
+ and isinstance(trace_id, str)
+ and trace_id == Langfuse.create_trace_id(seed=f"harbor-langfuse-v2:{digest}")):
+ return False
+ if not _matching_observation_id(
+ _read_back(trace_id, client), trace_id, digest, evidence,
+ exporter_revision="harbor-langfuse-v2"):
+ raise TelemetryReceiptError(f"{trial} has no matching verified v2 Langfuse trace")
+ archive = trial / LEGACY_V2_RECEIPT_NAME
+ if not _create_pending_receipt(archive, current):
+ if _path_json(archive) != dict(current):
+ raise TelemetryReceiptError(f"{trial} has a conflicting archived v2 receipt")
+ _atomic_receipt(trial / RECEIPT_NAME, pending)
+ return True
+
+
+def _export_attempt(trial: Path, metadata: Mapping[str, object], client=None) -> dict:
+ payload = build_attempt_payload(trial, metadata)
+ digest = _payload_sha256(payload)
+ evidence = _telemetry_evidence(payload)
+ if receipt := _verified_receipt(trial, digest):
+ return receipt
+ client = client or _client()
+ trace_id = client.create_trace_id(seed=f"{EXPORTER_REVISION}:{digest}")
+ if trace_id != _expected_trace_id(digest):
+ raise TelemetryReceiptError("Langfuse returned a non-deterministic trace ID")
+ pending = _pending_receipt(digest)
+ current = _read_receipt(trial)
+ migrated = False
+ existing_checked = False
+ existing = None
+ if current:
+ if current.get("exporter_revision") == "harbor-langfuse-v2":
+ # Keep the verified v2 receipt active until the side-effect-free v3 collision/read
+ # check succeeds. A transient read error then leaves an exact retryable state.
+ existing = _read_back(trace_id, client)
+ existing_checked = True
+ migrated = _migrate_verified_v2_receipt(trial, current, pending, evidence, client)
+ if not migrated:
+ return _resume_pending(trial, current, pending, trace_id, digest, evidence, client)
+ if not existing_checked:
+ existing = _read_back(trace_id, client)
+ if existing is not None:
+ if not _matching_observation_id(existing, trace_id, digest, evidence):
+ raise TelemetryReceiptError(f"existing deterministic trace {trace_id} is not one expected attempt")
+ receipt = _expected_receipt(digest)
+ _atomic_receipt(trial / RECEIPT_NAME, receipt)
+ return receipt
+ if not migrated and not _create_pending_receipt(trial / RECEIPT_NAME, pending):
+ current = _read_receipt(trial)
+ if current == _expected_receipt(digest):
+ return current
+ return _resume_pending(trial, current, pending, trace_id, digest, evidence, client)
+ outbound = _redact_persisted(payload, os.environ)
+ # ISO timestamps are validated structurally before this final free-text scrub. The diagnostic
+ # host pattern also matches clock fragments, so retain the already-validated canonical values.
+ outbound["timestamps"] = payload["timestamps"]
+ for outbound_step, canonical_step in zip(
+ outbound["trajectory"].get("steps") or [], payload["trajectory"].get("steps") or []):
+ if "timestamp" in canonical_step:
+ outbound_step["timestamp"] = canonical_step["timestamp"]
+ observation_metadata = {
+ **outbound["metadata"],
+ "attempt": outbound["attempt"],
+ "payload_sha256": digest,
+ "exporter_revision": EXPORTER_REVISION,
+ "telemetry_evidence": evidence,
+ }
+ observation_output = {key: outbound[key] for key in (
+ "skill", "status", "exception_category", "error_detail", "timestamps",
+ "trajectory", "verifier_output", "solution_artifacts")}
+ if "artifact_projection" in outbound:
+ observation_output["artifact_projection"] = outbound["artifact_projection"]
+ kwargs = {
+ "trace_context": {"trace_id": trace_id},
+ "name": "harbor-attempt",
+ "as_type": "agent",
+ "input": outbound["task"],
+ "output": observation_output,
+ "metadata": observation_metadata,
+ "level": "ERROR" if outbound["status"] == "failed" else "DEFAULT",
+ "status_message": outbound["status"],
+ }
+ observation = client.start_observation(**kwargs)
+ observation_id = getattr(observation, "id", None)
+ if not isinstance(observation_id, str) or not observation_id.strip():
+ raise TelemetryReceiptError("Langfuse returned no observation ID")
+ _atomic_receipt(trial / RECEIPT_NAME, _pending_receipt(digest, observation_id))
+ observation.end()
+ client.flush()
+ if not _wait_for_matching_observation(trace_id, digest, evidence, observation_id, client):
+ raise TelemetryReceiptError(f"Langfuse trace {trace_id} failed public read-back verification")
+ receipt = _expected_receipt(digest)
+ _atomic_receipt(trial / RECEIPT_NAME, receipt)
+ return receipt
+
+
+def _is_attempt(path: Path) -> bool:
+ with _AttemptReader(path) as reader:
+ if (reader.read(
+ "agent", "trajectory.json", limit=_MAX_TRAJECTORY_BYTES) is not None
+ or reader.read("exception.txt") is not None):
+ return True
+ opened = reader._open_dir(reader._trial_fd, "verifier", missing_ok=True)
+ if opened is not None:
+ descriptor, _info = opened
+ os.close(descriptor)
+ return True
+ return isinstance(reader.read_json("result.json").get("task_name"), str)
+
+
+def _job_attempts(job: Path, identity: NativeTrialIdentity | None = None) -> list[Path]:
+ try:
+ job_info = job.lstat()
+ except OSError:
+ return []
+ if stat.S_ISLNK(job_info.st_mode) or not stat.S_ISDIR(job_info.st_mode):
+ raise TelemetryReceiptError(f"{job} is not a real job directory")
+ attempts = []
+ for path in sorted(job.iterdir()):
+ info = path.lstat()
+ if stat.S_ISLNK(info.st_mode):
+ raise TelemetryReceiptError(f"{path} is a symlinked attempt entry")
+ if not stat.S_ISDIR(info.st_mode):
+ continue
+ if identity is not None:
+ try:
+ if read_trial_identity(path) != identity:
+ continue
+ except ValueError:
+ continue
+ if _is_attempt(path):
+ attempts.append(path)
+ return attempts
+
+
+def export_job_attempts(job: Path, metadata: Mapping[str, object], *,
+ identity: NativeTrialIdentity | None = None, client=None) -> list[dict]:
+ """Export every persisted attempt directly beneath one Harbor job directory."""
+ attempts = _job_attempts(job, identity)
+ if not attempts:
+ raise TelemetryReceiptError(f"{job} has no persisted Harbor attempts")
+ return [_export_attempt(trial, metadata, client) for trial in attempts]
+
+
+def _evidence_attempts(root: Path) -> list[Path]:
+ try:
+ root_info = root.lstat()
+ except OSError:
+ return []
+ if stat.S_ISLNK(root_info.st_mode) or not stat.S_ISDIR(root_info.st_mode):
+ raise TelemetryReceiptError(f"{root} is not a real evidence root")
+ candidates = {path.parent.parent for path in root.rglob("agent/trajectory.json")}
+ candidates.update(path.parent for path in root.rglob("exception.txt"))
+ candidates.update(path.parent for path in root.rglob("verifier") if path.is_dir())
+ for path in root.rglob("result.json"):
+ if isinstance(_path_json(path).get("task_name"), str):
+ candidates.add(path.parent)
+ return sorted(path for path in candidates if _is_attempt(path))
+
+
+def _trial_metadata(trial: Path, root: Path,
+ migration: Mapping[str, object] | None = None) -> dict:
+ metadata = {}
+ for ancestor in trial.parents:
+ combo = ancestor / "combo.json"
+ if combo.is_file():
+ metadata.update(_path_json(combo))
+ break
+ if ancestor == root:
+ break
+ if migration:
+ metadata.update(migration)
+ metadata["arm"] = "canary" if "canaries" in trial.parts else trial.parent.name
+ return metadata
+
+
+def _validate_preserved_provenance(trial: Path, metadata: Mapping[str, object]) -> None:
+ skill = metadata.get("skill")
+ task_texts = metadata.get("task_texts")
+ if not isinstance(skill, str) or not skill.strip() or not isinstance(task_texts, Mapping):
+ raise TelemetryReceiptError(f"{trial} lacks explicit preserved provenance")
+ body = _known_skill_body(metadata)
+ with _AttemptReader(trial) as reader:
+ task_name = _task_slug(reader.read_json("result.json").get("task_name") or trial.name)
+ task_text = task_texts.get(task_name)
+ if body is None or not isinstance(task_text, str) or not task_text.strip():
+ raise TelemetryReceiptError(f"{trial} lacks matching preserved provenance")
+
+
+def export_evidence_root(root: Path, *, metadata: Mapping[str, object] | None = None,
+ client=None) -> list[dict]:
+ """Export attempts recursively from preserved Harbor evidence without invoking an agent."""
+ attempts = _evidence_attempts(root)
+ if not attempts:
+ raise TelemetryReceiptError(f"{root} has no persisted Harbor attempts")
+ exports = []
+ for trial in attempts:
+ trial_metadata = _trial_metadata(trial, root, metadata)
+ _validate_preserved_provenance(trial, trial_metadata)
+ exports.append(_export_attempt(trial, trial_metadata, client))
+ return exports
+
+
+def validate_job_receipts(job: Path, metadata: Mapping[str, object], *,
+ identity: NativeTrialIdentity | None = None) -> None:
+ """Fail closed unless every consumed attempt has a current verified receipt."""
+ attempts = _job_attempts(job, identity)
+ if not attempts:
+ raise TelemetryReceiptError(f"{job} has no attempt receipt evidence")
+ for trial in attempts:
+ _validate_preserved_provenance(trial, metadata)
+ digest = _payload_sha256(build_attempt_payload(trial, metadata))
+ with _AttemptReader(trial) as reader:
+ receipt = reader.read_json(RECEIPT_NAME)
+ if receipt != _expected_receipt(digest):
+ raise TelemetryReceiptError(f"{trial} has a stale or withheld Langfuse receipt")
+
+
+def main(argv: Sequence[str] | None = None) -> int:
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--root", action="append", required=True, type=Path,
+ help="preserved Harbor evidence root (repeatable)")
+ parser.add_argument("--metadata", action="append", required=True, type=Path,
+ help="bounded provenance migration JSON paired with each --root")
+ args = parser.parse_args(argv)
+ if len(args.root) != len(args.metadata):
+ parser.error("each --root requires one paired --metadata document")
+ for root, metadata_path in zip(args.root, args.metadata):
+ metadata = load_provenance_metadata(metadata_path)
+ export_evidence_root(root, metadata=metadata)
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/ingot/optimize/harbor_native.py b/ingot/optimize/harbor_native.py
new file mode 100644
index 0000000..e2e2e61
--- /dev/null
+++ b/ingot/optimize/harbor_native.py
@@ -0,0 +1,306 @@
+"""Compile and recover exact Ingot identities in native Harbor jobs."""
+from __future__ import annotations
+
+import json
+import os
+import re
+import stat
+import tempfile
+from dataclasses import dataclass
+from pathlib import Path
+from typing import Iterator, Mapping, Sequence
+
+from .harbor_gateway import (gateway_agent_env, gateway_agent_name, gateway_route)
+from .harbor_targets import (HARNESS_PROTOCOLS, LocalTarget, harbor_agent_kwargs,
+ harbor_model, local_agent_env, protocol_for)
+
+NATIVE_TRIAL_MEMORY_MB = 2_048
+NATIVE_RUNNER_REVISION = f"native-v4-memory{NATIVE_TRIAL_MEMORY_MB}mb"
+
+_IDENTITY_KEYS = {
+ "INGOT_COMBINATION_ID": "combination_id",
+ "INGOT_ENDPOINT_FINGERPRINT": "endpoint_fingerprint",
+ "INGOT_HARNESS": "harness",
+ "INGOT_PROTOCOL": "protocol",
+ "INGOT_GATEWAY_REVISION": "gateway_revision",
+ "INGOT_ARM": "arm",
+}
+_FINGERPRINT = re.compile(r"^[0-9a-f]{12}$")
+_REVISION = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*$")
+
+
+@dataclass(frozen=True)
+class NativeTrialIdentity:
+ combination_id: str
+ endpoint_fingerprint: str
+ harness: str
+ protocol: str
+ gateway_revision: str
+ arm: str
+
+ def __post_init__(self) -> None:
+ values = tuple(self.__dict__.values())
+ if not all(isinstance(value, str) and value for value in values):
+ raise ValueError("Harbor trial identity fields must be nonempty strings")
+ if any(token in self.combination_id for token in ("/", "\\", "\x00", "..")):
+ raise ValueError("Harbor trial identity combination is not path-safe")
+ if not _FINGERPRINT.fullmatch(self.endpoint_fingerprint):
+ raise ValueError("Harbor trial identity endpoint fingerprint is invalid")
+ if self.harness not in HARNESS_PROTOCOLS:
+ raise ValueError("Harbor trial identity harness is unsupported")
+ if self.protocol != protocol_for(self.harness):
+ raise ValueError("Harbor trial identity protocol does not match its harness")
+ if not _REVISION.fullmatch(self.gateway_revision):
+ raise ValueError("Harbor trial identity gateway revision is invalid")
+ if self.arm not in {"canary", "skill", "control"}:
+ raise ValueError("Harbor trial identity arm is invalid")
+ if not self.combination_id.startswith(f"{self.harness}@"):
+ raise ValueError("Harbor trial identity combination does not match its harness")
+ if self.endpoint_fingerprint not in self.combination_id:
+ raise ValueError("Harbor trial identity combination does not match its endpoint")
+
+
+@dataclass(frozen=True)
+class NativeCell:
+ target: LocalTarget
+ harness: str
+
+ @property
+ def combination_id(self) -> str:
+ return native_trial_identity(self.target, self.harness, "skill").combination_id
+
+
+def identity_env(identity: NativeTrialIdentity) -> dict[str, str]:
+ """Return the exact non-secret fields Harbor persists on ``agent.env``."""
+ return {key: getattr(identity, field) for key, field in _IDENTITY_KEYS.items()}
+
+
+def identity_from_env(env: Mapping[str, object]) -> NativeTrialIdentity:
+ """Parse exactly the six persisted Ingot identity fields from an agent environment."""
+ ingot_keys = {key for key in env if isinstance(key, str) and key.startswith("INGOT_")}
+ if ingot_keys != set(_IDENTITY_KEYS):
+ raise ValueError("Harbor trial identity fields are incomplete or ambiguous")
+ values = {field: env[key] for key, field in _IDENTITY_KEYS.items()}
+ try:
+ return NativeTrialIdentity(**values)
+ except (TypeError, ValueError) as exc:
+ raise ValueError(f"Harbor trial identity is invalid: {exc}") from exc
+
+
+def read_trial_identity(trial_dir: Path) -> NativeTrialIdentity:
+ """Recover identity from Harbor's resolved per-trial lock and fail closed."""
+ lock_path = Path(trial_dir) / "lock.json"
+ try:
+ lock_info = lock_path.lstat()
+ if not stat.S_ISREG(lock_info.st_mode):
+ raise ValueError("Harbor trial identity lock is not a regular file")
+ payload = json.loads(lock_path.read_text())
+ env = payload["agent"]["env"]
+ except (OSError, ValueError, TypeError, KeyError) as exc:
+ raise ValueError("Harbor trial identity lock is unreadable") from exc
+ if not isinstance(env, dict):
+ raise ValueError("Harbor trial identity environment is invalid")
+ return identity_from_env(env)
+
+
+def iter_attempt_dirs(job_dir: Path, *, identity: NativeTrialIdentity | None = None
+ ) -> Iterator[Path]:
+ """Yield direct Harbor attempt directories, optionally by exact persisted identity."""
+ job_dir = Path(job_dir)
+ try:
+ job_info = job_dir.lstat()
+ except OSError as exc:
+ raise ValueError("native Harbor job is not a real directory") from exc
+ if stat.S_ISLNK(job_info.st_mode) or not stat.S_ISDIR(job_info.st_mode):
+ raise ValueError("native Harbor job is not a real directory")
+ for attempt in sorted(job_dir.iterdir()):
+ info = attempt.lstat()
+ if stat.S_ISLNK(info.st_mode):
+ raise ValueError("native Harbor entry is not a real attempt directory")
+ if not stat.S_ISDIR(info.st_mode):
+ continue
+ result = attempt / "result.json"
+ try:
+ result_info = result.lstat()
+ except OSError:
+ continue
+ if not stat.S_ISREG(result_info.st_mode):
+ raise ValueError("native Harbor result is not a regular file")
+ if identity is None:
+ yield attempt
+ continue
+ try:
+ lock_info = (attempt / "lock.json").lstat()
+ except OSError:
+ continue
+ if not stat.S_ISREG(lock_info.st_mode):
+ raise ValueError("native Harbor lock is not a regular file")
+ try:
+ observed = read_trial_identity(attempt)
+ except ValueError:
+ continue
+ if observed == identity:
+ yield attempt
+
+
+def native_trial_identity(target: LocalTarget, harness: str, arm: str) -> NativeTrialIdentity:
+ """Derive the only valid persisted identity for one target, harness, and arm."""
+ route = gateway_route(target, harness)
+ return NativeTrialIdentity(
+ combination_id=f"{harness}@{target.served_model}--{target.job_slug}",
+ endpoint_fingerprint=target.fingerprint,
+ harness=harness,
+ protocol=protocol_for(harness),
+ gateway_revision=f"{route.revision if route else 'direct'}-{NATIVE_RUNNER_REVISION}",
+ arm=arm,
+ )
+
+
+def compile_agent_config(target: LocalTarget, harness: str, arm: str, skill_source: Path,
+ endpoint_limit: int) -> dict:
+ """Compile one Harbor agent entry whose resolved lock retains exact cell identity."""
+ if not isinstance(endpoint_limit, int) or isinstance(endpoint_limit, bool) or endpoint_limit < 1:
+ raise ValueError("endpoint concurrency must be a positive integer")
+
+ identity = native_trial_identity(target, harness, arm)
+ route = gateway_route(target, harness)
+ agent = gateway_agent_name(route) if route else harness
+ env = gateway_agent_env(target, route) if route else local_agent_env(target, harness)
+ config = {
+ "model_name": route.model if route else harbor_model(target, harness),
+ "n_concurrent": endpoint_limit,
+ "concurrency_group": f"endpoint:{target.fingerprint}",
+ "skills": [str(skill_source)] if arm in {"skill", "canary"} else [],
+ "kwargs": {} if route else harbor_agent_kwargs(target, harness),
+ "env": {**env, **identity_env(identity)},
+ }
+ if ":" in agent:
+ config["import_path"] = agent
+ else:
+ config["name"] = agent
+ return config
+
+
+def _job_base(dataset: Path, task_names: Sequence[str], jobs_dir: Path, attempts: int,
+ global_limit: int, agents: list[dict]) -> dict:
+ if (not isinstance(global_limit, int) or isinstance(global_limit, bool)
+ or global_limit < 1):
+ raise ValueError("global concurrency must be a positive integer")
+ names = list(task_names)
+ if not names or not all(isinstance(name, str) and name for name in names):
+ raise ValueError("task names must be nonempty strings")
+ if len(names) != len(set(names)):
+ raise ValueError("task names must be unique")
+ return {
+ "jobs_dir": str(jobs_dir),
+ "n_attempts": attempts,
+ "n_concurrent_trials": global_limit,
+ "agents": agents,
+ "datasets": [{"path": str(dataset), "task_names": names}],
+ }
+
+
+def _limits(cells: Sequence[NativeCell], endpoint_limits: Mapping[str, int]) -> None:
+ required = {cell.target.fingerprint for cell in cells}
+ if set(endpoint_limits) != required:
+ raise ValueError("endpoint concurrency must cover exactly the requested targets")
+ for value in endpoint_limits.values():
+ if not isinstance(value, int) or isinstance(value, bool) or value < 1:
+ raise ValueError("endpoint concurrency must be a positive integer")
+ identities = [cell.combination_id for cell in cells]
+ if len(identities) != len(set(identities)):
+ raise ValueError("native Harbor cells must have unique combination identities")
+
+
+def compile_canary_job(dataset: Path, task_name: str, cells: Sequence[NativeCell],
+ skill_source: Path, jobs_dir: Path, *, global_limit: int,
+ endpoint_limits: Mapping[str, int]) -> dict:
+ """Compile one skill-bearing attempt for every requested model/harness cell."""
+ cells = list(cells)
+ if not cells:
+ raise ValueError("native Harbor job requires at least one cell")
+ _limits(cells, endpoint_limits)
+ agents = [compile_agent_config(cell.target, cell.harness, "canary", skill_source,
+ endpoint_limits[cell.target.fingerprint])
+ for cell in cells]
+ return _job_base(dataset, [task_name], jobs_dir, 1, global_limit, agents)
+
+
+def compile_measurement_job(dataset: Path, task_names: Sequence[str], cells: Sequence[NativeCell],
+ skill_source: Path, jobs_dir: Path, *, attempts: int,
+ global_limit: int,
+ endpoint_limits: Mapping[str, int],
+ arms: Sequence[str] = ("skill", "control")) -> dict:
+ """Compile the requested exact arms for every canary-approved cell."""
+ if not isinstance(attempts, int) or isinstance(attempts, bool) or attempts != 3:
+ raise ValueError("full measurement requires exactly three attempts")
+ names = list(task_names)
+ if len(names) != 4:
+ raise ValueError("full measurement requires exactly four tasks")
+ cells = list(cells)
+ if not cells:
+ raise ValueError("native Harbor job requires at least one cell")
+ arms = tuple(arms)
+ if not arms or len(arms) != len(set(arms)) or not set(arms) <= {"skill", "control"}:
+ raise ValueError("measurement arms must be a unique nonempty skill/control subset")
+ _limits(cells, endpoint_limits)
+ agents = []
+ buckets: dict[str, list[NativeCell]] = {}
+ for cell in cells:
+ buckets.setdefault(cell.target.fingerprint, []).append(cell)
+ ordered = []
+ while any(buckets.values()):
+ for fingerprint in buckets:
+ if buckets[fingerprint]:
+ ordered.append(buckets[fingerprint].pop(0))
+ for cell in ordered:
+ limit = endpoint_limits[cell.target.fingerprint]
+ for arm in arms:
+ agents.append(compile_agent_config(cell.target, cell.harness, arm, skill_source, limit))
+ return _job_base(dataset, names, jobs_dir, attempts, global_limit, agents)
+
+
+def select_measurement_cells(cells: Sequence[NativeCell], canaries: Mapping[str, Mapping[str, object]]
+ ) -> tuple[list[NativeCell], dict[str, dict[str, object]]]:
+ """Partition exact passed canaries from explicit unmeasured cell records."""
+ selected = []
+ unmeasured = {}
+ for cell in cells:
+ record = canaries.get(cell.combination_id)
+ if isinstance(record, Mapping) and record.get("ok") is True and "error" not in record:
+ selected.append(cell)
+ continue
+ error = str((record or {}).get("error") or "canary did not pass")[:400]
+ unmeasured[cell.combination_id] = {
+ "combination": cell.combination_id,
+ "harness": cell.harness,
+ "target_alias": cell.target.alias,
+ "endpoint_fingerprint": cell.target.fingerprint,
+ "state": "unmeasured",
+ "error": error,
+ }
+ return selected, unmeasured
+
+
+def write_job_config(path: Path, config: Mapping[str, object]) -> None:
+ """Write one deterministic Harbor job document without exposing a partial file."""
+ path = Path(path)
+ if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]*", path.name):
+ raise ValueError("Harbor config filename must be a safe slug")
+ path.parent.mkdir(parents=True, exist_ok=True)
+ with tempfile.NamedTemporaryFile("w", encoding="utf-8", dir=path.parent,
+ prefix=f".{path.name}.", suffix=".tmp",
+ delete=False) as handle:
+ temporary = Path(handle.name)
+ try:
+ json.dump(config, handle, indent=2, sort_keys=True)
+ handle.write("\n")
+ handle.flush()
+ os.fsync(handle.fileno())
+ except BaseException:
+ temporary.unlink(missing_ok=True)
+ raise
+ try:
+ os.replace(temporary, path)
+ finally:
+ temporary.unlink(missing_ok=True)
diff --git a/ingot/optimize/harbor_redaction.py b/ingot/optimize/harbor_redaction.py
new file mode 100644
index 0000000..1e2719d
--- /dev/null
+++ b/ingot/optimize/harbor_redaction.py
@@ -0,0 +1,41 @@
+"""Shared redaction for persisted Harbor diagnostics and exported text."""
+from __future__ import annotations
+
+import re
+from collections.abc import Mapping
+
+
+_HARBOR_RECEIPT_EXCERPT_LIMIT = 2000
+
+
+def _receipt_secret_values(parent: Mapping[str, str]) -> tuple[str, ...]:
+ """Known ambient secrets, longest first so a shorter value cannot leave a suffix behind."""
+ return tuple(sorted({value for key, value in parent.items() if value and
+ (key.endswith(("API_KEY", "TOKEN", "SECRET", "PASSWORD")) or key == "API_KEY")},
+ key=len, reverse=True))
+
+
+def _redact_harbor_receipt_output(value: object, parent: Mapping[str, str]) -> str:
+ """Return a bounded Harbor diagnostic excerpt without endpoint or credential material."""
+ text = str(value or "")
+ for secret in _receipt_secret_values(parent):
+ text = text.replace(secret, "")
+ text = re.sub(r"https?://[^\s\"']+", "", text, flags=re.IGNORECASE)
+ text = re.sub(r"(?im)\b(authorization\s*:\s*)[^\r\n]+", r"\1", text)
+ text = re.sub(r"(?i)\b(x-api-key\s*:\s*)[^\s,;\"']+", r"\1", text)
+ text = re.sub(r"\b(?:sk|pk)[_-][A-Za-z0-9_-]+\b", "", text)
+ text = re.sub(r"(?i)([\"']?[A-Z][A-Z0-9_]*(?:API_KEY|TOKEN|SECRET|PASSWORD)[\"']?\s*[:=]\s*[\"']?)[^\s,;\"'}]+",
+ r"\1", text)
+ text = re.sub(r"\b(?:localhost|[A-Za-z0-9.-]+):[0-9]{2,5}\b", "", text)
+ return text[-_HARBOR_RECEIPT_EXCERPT_LIMIT:]
+
+
+def _redact_persisted(value: object, parent: Mapping[str, str]) -> object:
+ """Copy diagnostics for disk; runtime callers keep their original exception objects."""
+ if isinstance(value, str):
+ return _redact_harbor_receipt_output(value, parent)
+ if isinstance(value, Mapping):
+ return {key: _redact_persisted(item, parent) for key, item in value.items()}
+ if isinstance(value, list):
+ return [_redact_persisted(item, parent) for item in value]
+ return value
diff --git a/ingot/optimize/harbor_report.py b/ingot/optimize/harbor_report.py
new file mode 100644
index 0000000..fd8a665
--- /dev/null
+++ b/ingot/optimize/harbor_report.py
@@ -0,0 +1,195 @@
+"""Read a harness x model matrix off disk for display.
+
+`harbor_eval` writes `runs/harbor/.json`; `harbor_rescore` writes `.rescored.json`
+from the same job directories when the scoring rules change. The two carry the same rows under
+different keys (`harnesses` and `combinations`), and the rescored file is the corrected one wherever
+it exists — the first grid shipped two fabricated cells that only rescoring removed.
+
+Stdlib only, and no judging: this reads results, it does not produce them.
+
+The reporting rules here are the ones the matrix violated the first time it was run:
+
+- a combination that failed carries no `lift` at all, so nothing downstream can render it as a zero
+ in a column of measurements;
+- every row carries the `n` it was measured over, because a lift over two tasks and one over four
+ are different claims;
+- and if the control arms sit near the top of the scale, the instrument could not discriminate and
+ the ranking means nothing, so the ceiling is reported instead of the winner.
+"""
+from __future__ import annotations
+
+import json
+import math
+from pathlib import Path
+
+from ingot import paths
+
+from .harbor_targets import TARGETS
+
+HARBOR_DIR = paths.runs() / "harbor"
+
+# Above this, the controls are close enough to the top of the scale that the remaining headroom is
+# smaller than the judge's own noise (~0.10 observed across re-runs of an unchanged arm), so a
+# treatment cannot separate from its control whatever the skill does. Reporting a "best combination"
+# off a matrix in this state is reporting the noise.
+CEILING = 0.75
+
+
+def matrix_path(skill: str, root: Path | None = None) -> Path | None:
+ """The newest of the rescored and raw matrices for `skill`, or None if neither exists."""
+ base = root or HARBOR_DIR
+ found = [p for p in (base / f"{skill}.rescored.json", base / f"{skill}.json") if p.is_file()]
+ return max(found, key=lambda p: p.stat().st_mtime) if found else None
+
+
+def _split(combo: str) -> tuple[str, str]:
+ harness, _, model = combo.partition("@")
+ return harness, model or "harness default"
+
+
+def read_matrix(skill: str, root: Path | None = None) -> dict | None:
+ """Normalized rows for one skill, or None when the skill has never been run.
+
+ Raises ValueError on an unreadable or malformed file rather than returning an empty matrix: a
+ page saying "no combinations" when the file is actually corrupt reads as "nothing helps"."""
+ path = matrix_path(skill, root)
+ if path is None:
+ return None
+ try:
+ raw = json.loads(path.read_text(encoding="utf-8"))
+ except (OSError, ValueError) as error:
+ raise ValueError(f"{path.name} is unreadable ({error})") from error
+ if not isinstance(raw, dict):
+ raise ValueError(f"{path.name} does not contain a matrix")
+
+ source = raw.get("combinations") if isinstance(raw.get("combinations"), dict) else raw.get("harnesses")
+ rows = []
+ for combo, record in sorted((source or {}).items()):
+ if not isinstance(record, dict):
+ continue
+ harness, model = _split(combo)
+ target = TARGETS.get(str(record.get("target_alias") or ""), {})
+ family = str(record.get("family") or "")
+ display_model = (target.get("display_name")
+ if family.startswith("Qwen") and target.get("family") == family else None)
+ row = {"combination": combo,
+ "harness": record.get("harness") or harness,
+ # Keep wire IDs in provenance and job identity, but present the canonical model
+ # name whenever a pinned target alias provides one.
+ "model": display_model or record.get("model") or model}
+ # Endpoint identity is evidence, not presentation decoration. Legacy matrices do not
+ # have it, so only copy fields that the row actually recorded; a fabricated null identity
+ # would falsely suggest that an endpoint was checked.
+ for field in ("target_alias", "endpoint_fingerprint", "protocol", "family",
+ "quantization", "tool_parser", "exploratory",
+ "rankable"):
+ if field in record:
+ row[field] = record[field]
+ size = record.get("parameter_billions")
+ if isinstance(size, (int, float)) and not isinstance(size, bool) and math.isfinite(size) \
+ and size > 0:
+ row["parameter_billions"] = size
+ if "lift" not in record:
+ # No lift, no means, no scores. A row shaped like a measurement is how "the container
+ # died" became "lift +0.750" the first time this ran.
+ row["error"] = str(record.get("error") or "not measured")
+ rows.append(row)
+ continue
+ row.update({"lift": record["lift"],
+ "skill_mean": record.get("skill_mean"),
+ "control_mean": record.get("control_mean"),
+ "n": record.get("tasks_scored"),
+ "attempts": record.get("attempts") or 1,
+ "dropped": record.get("tasks_dropped") or []})
+ rows.append(row)
+
+ summary = summarize(rows)
+ # A global winner is not meaningful once columns have independent evidence fitness. Keep the
+ # reusable `summarize` result intact for callers that need it, but do not expose its global
+ # `best` to the UI alongside per-model winners.
+ summary.pop("best", None)
+ models = sorted({row["model"] for row in rows})
+ harnesses = sorted({row["harness"] for row in rows})
+ exploratory = raw.get("exploratory") is True or any(
+ row.get("exploratory") is True for row in rows)
+ rankable = raw.get("rankable") is not False and all(
+ row.get("rankable") is not False for row in rows)
+ return {"skill": skill, "source": path.name, "rescored": path.name.endswith(".rescored.json"),
+ "generated": int(path.stat().st_mtime), "judge": raw.get("judge") or "",
+ "exploratory": exploratory, "rankable": rankable,
+ "rows": rows, "models": models, "harnesses": harnesses,
+ "model_summaries": summarize_models(rows), **summary}
+
+
+def summarize(rows: list[dict]) -> dict:
+ """Headline numbers, plus whether the matrix is fit to be read as a ranking at all."""
+ measured = [r for r in rows if "lift" in r]
+ if not measured:
+ return {"measured": 0, "unmeasured": len(rows), "mean_lift": None,
+ "control_mean": None, "best": None,
+ "warning": "No combination produced a measurement." if rows else ""}
+
+ controls = [r["control_mean"] for r in measured if isinstance(r.get("control_mean"), (int, float))]
+ control_mean = sum(controls) / len(controls) if controls else None
+ best = max(measured, key=lambda r: r["lift"])
+
+ warning = ""
+ if any(r.get("rankable") is False for r in measured):
+ warning = ("Exploratory evidence is shown for inspection but is not readable as a "
+ "ranking. Re-run with the full measurement contract before naming a winner.")
+ elif control_mean is not None and control_mean >= CEILING:
+ # Deliberately not "the best combination is X". At this control level the ranking is noise,
+ # and naming a winner off it is the failure mode this whole module exists to prevent.
+ warning = (f"Controls average {control_mean:.3f} of 1.00, at or above the {CEILING:.2f} "
+ f"ceiling. There is less headroom left than the judge's own run-to-run spread, "
+ f"so these combinations cannot be ranked against each other — the held-out tasks "
+ f"are too easy, whatever the lift column says.")
+ elif all((r.get("n") or 0) < 3 for r in measured):
+ warning = ("Every combination was measured over fewer than 3 tasks; these differences are "
+ "not distinguishable from noise.")
+ elif all((r.get("attempts") or 1) < 2 for r in measured):
+ # Measured, not assumed: two control-arm runs of an identical configuration moved a task's
+ # score by 0.278 and swapped two harnesses' rank, while re-judging one fixed answer three
+ # times was identical. A single attempt per task sits under that, so a ranking built from
+ # these rows is a ranking of the agents' own run-to-run variance.
+ warning = ("Every combination was run with one attempt per task. Repeat runs of an "
+ "identical configuration have moved a task's score by 0.28 and swapped the rank "
+ "of two harnesses, so single-attempt differences are agent variance rather than "
+ "an effect. Re-run with -k 3 or more before reading this as a ranking.")
+
+ return {"measured": len(measured), "unmeasured": len(rows) - len(measured),
+ "mean_lift": sum(r["lift"] for r in measured) / len(measured),
+ "control_mean": control_mean,
+ "best": None if warning else {"combination": best["combination"], "lift": best["lift"],
+ "n": best.get("n")},
+ "warning": warning}
+
+
+def summarize_models(rows: list[dict]) -> dict[str, dict]:
+ """Summarize each recorded model column without filling absent intersections."""
+ summaries = {}
+ for model in sorted({row["model"] for row in rows}):
+ summary = summarize([row for row in rows if row["model"] == model])
+ best = summary.pop("best")
+ summaries[model] = {
+ "model": model,
+ "measured": summary["measured"],
+ "unmeasured": summary["unmeasured"],
+ "mean_lift": summary["mean_lift"],
+ "control_mean": summary["control_mean"],
+ "warning": summary["warning"],
+ "best_harness": None if best is None else next(
+ row["harness"] for row in rows
+ if row["model"] == model and row.get("combination") == best["combination"]
+ ),
+ }
+ return summaries
+
+
+def available(root: Path | None = None) -> list[str]:
+ """Skills that have a matrix on disk, newest first."""
+ base = root or HARBOR_DIR
+ if not base.is_dir():
+ return []
+ skills = {p.name.split(".")[0] for p in base.glob("*.json")}
+ return sorted((s for s in skills if s), key=lambda s: -(matrix_path(s, base).stat().st_mtime))
diff --git a/ingot/optimize/harbor_rescore.py b/ingot/optimize/harbor_rescore.py
new file mode 100644
index 0000000..a15840b
--- /dev/null
+++ b/ingot/optimize/harbor_rescore.py
@@ -0,0 +1,524 @@
+"""Recompute compatible Harbor matrices from job directories already on disk.
+
+The agents have already run; rescoring only re-reads their persisted verifier artifacts. A matrix
+may combine the preserved proprietary jobs and endpoint-qualified local jobs only when the evidence
+states that they used the same task set and retry count. Every answer is then judged under the
+current scoring provenance.
+
+Usage: python -m ingot.optimize.harbor_rescore [--jobs runs/harbor/jobs/]
+"""
+from __future__ import annotations
+
+import json
+import os
+import tempfile
+from pathlib import Path
+from typing import Sequence
+
+from .ab import load_tasks
+from .agy_judge import AGY_IDENTITY, preflight
+from .harbor_eval import (HARBOR_DIR, _task_fingerprint, _task_name,
+ broken_tasks, broken_trials, collect_answers, score)
+from .harbor_langfuse import load_provenance_metadata, validate_job_receipts
+from .harbor_native import NATIVE_RUNNER_REVISION, NativeTrialIdentity, identity_from_env
+
+
+_EVIDENCE_FIELDS = ("task_fingerprint", "attempts")
+_CURRENT_SCORING = {
+ "judge": "agy/gemini-3.6-flash-medium",
+ "scoring_revision": "harbor-rubric-v2-agy",
+ "judge_billing_mode": "subscription",
+}
+_HISTORICAL_SCORING_FIELDS = {
+ "judge", "scoring_revision", "judge_runtime", "judge_billing_mode", "billing_mode",
+ "cost_usd",
+}
+_MEASUREMENT_FIELDS = {
+ "error", "score", "scores", "skill_mean", "control_mean", "lift", "skill_scores",
+ "control_scores", "tasks_scored", "tasks_dropped", "mean_lift", "scored", "unscorable",
+ "n", "dropped", "best", "measured", "unmeasured",
+ "canary_error",
+}
+_AGY_SCORE_CONCURRENCY = 4
+
+
+def _arm_evidence(combo: Path, arm: str) -> tuple[Path, NativeTrialIdentity | None]:
+ """Resolve legacy arm directories or one exact arm view inside a native sibling job."""
+ metadata = _combo_identity(combo)
+ native_jobs = metadata.get("native_jobs")
+ if native_jobs is not None and not isinstance(native_jobs, dict):
+ raise ValueError("native jobs must map arms to sibling slugs")
+ native_job = native_jobs.get(arm) if isinstance(native_jobs, dict) else metadata.get("native_job")
+ if native_job is None:
+ return combo / arm, None
+ if (not isinstance(native_job, str)
+ or not native_job
+ or not all(character.isalnum() or character in "._-" for character in native_job)):
+ raise ValueError("native job must be a safe sibling slug")
+ identities = metadata.get("native_identities")
+ if not isinstance(identities, dict) or arm not in identities:
+ raise ValueError(f"native job is missing {arm} identity")
+ env = identities[arm]
+ if not isinstance(env, dict):
+ raise ValueError(f"native job {arm} identity is invalid")
+ identity = identity_from_env(env)
+ if identity.arm != arm:
+ raise ValueError(f"native job {arm} identity has the wrong arm")
+ expected = {
+ "combination_id": metadata.get("combination"),
+ "harness": metadata.get("harness"),
+ "endpoint_fingerprint": metadata.get("endpoint_fingerprint"),
+ "protocol": metadata.get("protocol"),
+ "gateway_revision": metadata.get("gateway_revision", "direct"),
+ }
+ for field, value in expected.items():
+ if getattr(identity, field) != value:
+ raise ValueError(f"native job {arm} identity does not match combo metadata")
+ return combo.parent / native_job, identity
+
+
+def _combo_identity(combo: Path) -> dict:
+ """Return every identity field the live run recorded beside this combination."""
+ try:
+ record = json.loads((combo / "combo.json").read_text(encoding="utf-8"))
+ except (OSError, ValueError) as error:
+ raise ValueError(f"{combo} has no readable combo.json identity") from error
+ if not isinstance(record, dict) or not record:
+ raise ValueError(f"{combo} has no combo identity")
+ return dict(record)
+
+
+def _identity_key(identity: dict) -> str:
+ """Deduplicate exact recorded identities, never lossy job-directory names."""
+ try:
+ return json.dumps(identity, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
+ except (TypeError, ValueError) as error:
+ raise ValueError("combo identity is not JSON data") from error
+
+
+def discover_combinations(roots: Sequence[Path]) -> list[Path]:
+ """Return unique combinations across roots, keyed by complete combo metadata."""
+ found: list[Path] = []
+ identities: set[str] = set()
+ for root in roots:
+ if not root.is_dir():
+ raise SystemExit(f"no job directories at {root}")
+ combos = sorted(path for path in root.iterdir()
+ if path.is_dir() and (path / "combo.json").is_file())
+ for combo in combos:
+ record = _combo_identity(combo)
+ key = _identity_key(record)
+ if key not in identities:
+ found.append(combo)
+ identities.add(key)
+ return found
+
+
+def select_combinations(roots: Sequence[Path], paths: Sequence[Path]) -> list[Path]:
+ """Select exact real directories without inspecting sibling combinations."""
+ allowed = {root.resolve() for root in roots}
+ found: list[Path] = []
+ identities: set[str] = set()
+ for path in paths:
+ if path.is_symlink() or not path.is_dir() or path.parent.resolve() not in allowed:
+ raise ValueError(f"selected combination must be a real immediate child of a jobs root: {path}")
+ resolved = path.resolve()
+ key = _identity_key(_combo_identity(resolved))
+ if key not in identities:
+ found.append(resolved)
+ identities.add(key)
+ if not found:
+ raise ValueError("no completed combination paths selected")
+ return found
+
+
+def validate_compatibility(records: Sequence[dict]) -> None:
+ """Refuse raw evidence whose task set or retry count differs."""
+ if not records:
+ raise ValueError("no combination evidence found")
+ for field in _EVIDENCE_FIELDS:
+ values = [record.get(field) for record in records]
+ if field == "attempts":
+ invalid = any(not isinstance(value, int) or isinstance(value, bool) or value < 1
+ for value in values)
+ else:
+ invalid = any(not isinstance(value, str) or not value.strip() for value in values)
+ if invalid:
+ raise ValueError(f"combination metadata is missing or invalid {field}")
+ if len(set(values)) != 1:
+ raise ValueError(f"incompatible combination metadata: {field}")
+
+
+def _validate_current_compatibility(records: Sequence[dict], holdout: list[dict]) -> None:
+ """Require otherwise-compatible evidence to describe this exact raw task run."""
+ if records[0]["task_fingerprint"] != _task_fingerprint(holdout):
+ raise ValueError("combination metadata does not match current task_fingerprint")
+ attempts = records[0]["attempts"]
+ exploratory = all(record.get("exploratory") is True and record.get("rankable") is False
+ for record in records)
+ if attempts != 3 and not (attempts == 1 and exploratory):
+ raise ValueError("combination metadata attempts do not match current measurement contract")
+
+
+def _load_legacy_manifest(path: Path) -> dict:
+ """Read the user-supplied historical metadata; never fill missing history from today."""
+ try:
+ manifest = json.loads(path.read_text(encoding="utf-8"))
+ except (OSError, ValueError) as error:
+ raise ValueError(f"legacy metadata manifest is unreadable: {path}") from error
+ if not isinstance(manifest, dict) or not isinstance(manifest.get("root"), str):
+ raise ValueError("legacy metadata manifest needs a root")
+ missing = [field for field in _EVIDENCE_FIELDS if field not in manifest]
+ if missing:
+ raise ValueError(f"legacy metadata manifest is missing {', '.join(missing)}")
+ return manifest
+
+
+def _legacy_metadata(skill: str, root: Path, combo: Path, identity: dict,
+ manifest: dict | None) -> dict:
+ """Attach only explicit history to an old proprietary combination after cross-checking it."""
+ if manifest is None:
+ raise ValueError(f"{combo} needs an explicit legacy metadata manifest")
+ if Path(manifest["root"]).resolve() != root.resolve():
+ raise ValueError(f"{combo} legacy metadata manifest names a different root")
+ try:
+ matrix = json.loads((HARBOR_DIR / f"{skill}.json").read_text(encoding="utf-8"))
+ except (OSError, ValueError) as error:
+ raise ValueError(f"{combo} lacks explicit matrix metadata") from error
+ if not isinstance(matrix, dict):
+ raise ValueError(f"{combo} lacks explicit matrix metadata")
+
+ key = identity.get("combination") or combo.name
+ rows = matrix.get("harnesses") or {}
+ row = rows.get(key) if isinstance(rows, dict) else None
+ if matrix.get("skill") != skill or not isinstance(row, dict):
+ raise ValueError(f"{combo} lacks explicit matrix metadata")
+ metadata = {field: manifest[field] for field in _EVIDENCE_FIELDS}
+ matrix_checks = {"attempts": row.get("attempts")}
+ conflicting_matrix = [field for field, value in matrix_checks.items()
+ if value not in (None, "") and value != metadata[field]]
+ if conflicting_matrix:
+ raise ValueError(f"{combo} conflicts with legacy matrix metadata: {', '.join(conflicting_matrix)}")
+ conflicting = [field for field in _EVIDENCE_FIELDS
+ if identity.get(field) not in (None, "")
+ and identity[field] != metadata[field]]
+ if conflicting:
+ raise ValueError(f"{combo} conflicts with legacy metadata manifest: {', '.join(conflicting)}")
+ return {**identity, **metadata}
+
+
+def _evidence_metadata(skill: str, root: Path, combo: Path, manifest: dict | None) -> dict:
+ identity = _combo_identity(combo)
+ if all(identity.get(field) not in (None, "") for field in _EVIDENCE_FIELDS):
+ return identity
+ return _legacy_metadata(skill, root, combo, identity, manifest)
+
+
+def _current_scoring(runtime: dict) -> dict:
+ """Scoring provenance for this one rescore, without inventing metered cost."""
+ return {**_CURRENT_SCORING, "judge_runtime": str(runtime["version"])}
+
+
+def current_scoring_identity() -> dict:
+ """Resolve the exact current subscription scorer identity once for a controller run."""
+ return _current_scoring(preflight())
+
+
+def _output_identity(identity: dict, scoring: dict) -> dict:
+ """Replace any historical scoring receipt while preserving raw evidence identity."""
+ raw = {key: value for key, value in identity.items()
+ if key not in _HISTORICAL_SCORING_FIELDS | _MEASUREMENT_FIELDS}
+ return {**raw, **scoring}
+
+
+def _validate_attempt_artifacts(arm: Path, skill: str, holdout: list[dict], attempts: int,
+ identity: NativeTrialIdentity | None = None) -> None:
+ """Require one readable Harbor trial record for every expected task attempt."""
+ observed = {_task_name(skill, index): 0 for index in range(len(holdout))}
+ from .harbor_native import iter_attempt_dirs
+ results = ([attempt / "result.json" for attempt in iter_attempt_dirs(arm, identity=identity)]
+ if identity is not None else sorted(arm.glob("*/result.json")))
+ for result in results:
+ try:
+ record = json.loads(result.read_text(encoding="utf-8"))
+ except (OSError, ValueError) as error:
+ raise RuntimeError(f"{arm.name} arm has an unreadable attempt record") from error
+ task_name = record.get("task_name") if isinstance(record, dict) else None
+ if not isinstance(task_name, str) or not task_name.strip():
+ raise RuntimeError(f"{arm.name} arm has an attempt record without a task name")
+ task = task_name.split("/")[-1].split("__")[0]
+ if task not in observed:
+ raise RuntimeError(f"{arm.name} arm has an unexpected attempt for {task}")
+ observed[task] += 1
+ mismatched = [f"{task}={count}" for task, count in observed.items() if count != attempts]
+ if mismatched:
+ raise RuntimeError(f"{arm.name} arm needs exactly {attempts} attempt records per task; "
+ f"found {', '.join(mismatched)}")
+
+
+def _row_key(identity: dict, rows: dict[str, dict]) -> str:
+ """Keep a malformed repeated combination label from overwriting distinct endpoint evidence."""
+ key = str(identity.get("combination") or "")
+ if not key:
+ raise ValueError("combo identity is missing combination")
+ if key not in rows:
+ return key
+ fingerprint = str(identity.get("endpoint_fingerprint") or "unknown")
+ qualified = f"{key}--{fingerprint}"
+ if qualified not in rows:
+ return qualified
+ suffix = 2
+ while f"{qualified}-{suffix}" in rows:
+ suffix += 1
+ return f"{qualified}-{suffix}"
+
+
+def atomic_write_json(path: Path, payload: dict) -> None:
+ """Publish JSON with one replacement; every earlier failure leaves ``path`` untouched."""
+ encoded = json.dumps(payload, indent=2).encode("utf-8")
+ path.parent.mkdir(parents=True, exist_ok=True)
+ descriptor, temporary_name = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=path.parent)
+ temporary = Path(temporary_name)
+ try:
+ with os.fdopen(descriptor, "wb") as handle:
+ handle.write(encoded)
+ handle.flush()
+ os.fsync(handle.fileno())
+ os.replace(temporary, path)
+ except BaseException:
+ temporary.unlink(missing_ok=True)
+ raise
+
+
+def _summary(skill: str, rows: dict[str, dict], shared: dict, scoring: dict) -> dict | None:
+ scored = {name: row for name, row in rows.items() if "lift" in row}
+ if not scored:
+ return None
+ return {
+ "skill": skill,
+ "combinations": rows,
+ "exploratory": any(row.get("exploratory") is True for row in rows.values()),
+ "rankable": all(row.get("rankable") is not False for row in rows.values()),
+ "mean_lift": sum(row["lift"] for row in scored.values()) / len(scored),
+ "scored": len(scored),
+ "unscorable": len(rows) - len(scored),
+ **shared,
+ **scoring,
+ }
+
+
+def _prior_rows(path: Path, skill: str, shared: dict, scoring: dict,
+ skill_sha256: str) -> dict[str, dict]:
+ if not path.is_file():
+ return {}
+ try:
+ payload = json.loads(path.read_text(encoding="utf-8"))
+ except (OSError, ValueError) as error:
+ raise ValueError(f"incompatible existing rescore output: {path}") from error
+ expected = {"skill": skill, **shared, **scoring}
+ if (not isinstance(payload, dict)
+ or any(payload.get(key) != value for key, value in expected.items())
+ or not isinstance(payload.get("combinations"), dict)):
+ raise ValueError(f"incompatible existing rescore output: {path}")
+ rows = payload["combinations"]
+ if any(not isinstance(key, str) or not isinstance(row, dict) for key, row in rows.items()):
+ raise ValueError(f"incompatible existing rescore output: {path}")
+ return {key: row for key, row in rows.items() if row.get("skill_sha256") == skill_sha256}
+
+
+def _matching_row_key(identity: dict, rows: dict[str, dict]) -> str | None:
+ fields = ("combination", "harness", "model", "target_alias", "endpoint_fingerprint",
+ "protocol", "task_fingerprint", "skill_sha256", "attempts",
+ "gateway_revision", "gateway_identity", "family", "parameter_billions",
+ "quantization", "tool_parser")
+ for key, row in rows.items():
+ if all(row.get(field) == identity.get(field) for field in fields):
+ return key
+ return None
+
+
+def _same_logical_cell(left: dict, right: dict) -> bool:
+ fields = ("combination", "harness", "target_alias", "endpoint_fingerprint", "protocol",
+ "task_fingerprint", "skill_sha256", "attempts")
+ return all(left.get(field) == right.get(field) for field in fields)
+
+
+def _prefer_current_native_evidence(evidence: list[tuple[Path, dict]]) -> list[tuple[Path, dict]]:
+ current = [identity for _, identity in evidence
+ if str(identity.get("gateway_revision") or "").endswith(
+ f"-{NATIVE_RUNNER_REVISION}")]
+ if not current:
+ return evidence
+ return [
+ item for item in evidence if (
+ str(item[1].get("gateway_revision") or "").endswith(f"-{NATIVE_RUNNER_REVISION}")
+ or not any(_same_logical_cell(item[1], identity) for identity in current)
+ )
+ ]
+
+
+def rescore(skill: str, jobs_roots: Sequence[Path] | None = None,
+ legacy_metadata: Path | None = None,
+ provenance_metadata: Sequence[Path] | None = None,
+ combination_paths: Sequence[Path] | None = None,
+ output: Path | None = None, scoring_identity: dict | None = None, log=print) -> dict:
+ """Recompute compatible combinations and atomically publish their matrix when measured."""
+ if os.environ.get("JUDGE_BACKEND", "").strip().lower() != "agy":
+ raise RuntimeError("Harbor rescore requires JUDGE_BACKEND=agy; refusing scorer fallback")
+ selected_mode = combination_paths is not None
+ roots = list(jobs_roots) if jobs_roots is not None else [HARBOR_DIR / "jobs" / skill]
+ provenance_paths = list(provenance_metadata or [])
+ if provenance_paths and len(provenance_paths) != len(roots):
+ raise ValueError("each jobs root requires one paired provenance metadata document")
+ provenance_by_root = {
+ root.resolve(): load_provenance_metadata(path)
+ for root, path in zip(roots, provenance_paths)
+ }
+ _, holdout, _ = load_tasks(skill)
+ manifest = _load_legacy_manifest(legacy_metadata) if legacy_metadata else None
+ discovered = (select_combinations(roots, combination_paths or [])
+ if selected_mode else discover_combinations(roots))
+ evidence = _prefer_current_native_evidence([
+ (combo, _evidence_metadata(skill, combo.parent, combo, manifest))
+ for combo in discovered
+ ])
+ records = [identity for _, identity in evidence]
+ validate_compatibility(records)
+ _validate_current_compatibility(records, holdout)
+ skill_sha256 = ""
+ if selected_mode:
+ skill_revisions = {identity.get("skill_sha256") for identity in records}
+ if (len(skill_revisions) != 1 or not isinstance(next(iter(skill_revisions)), str)
+ or not next(iter(skill_revisions)).strip()):
+ raise ValueError("selected combinations need one non-empty skill_sha256")
+ skill_sha256 = next(iter(skill_revisions))
+ for combo, identity in evidence:
+ if identity.get("canary_error") or identity.get("measurement_error"):
+ continue
+ for arm in ("skill", "control"):
+ job, arm_identity = _arm_evidence(combo, arm)
+ if not job.is_dir():
+ raise RuntimeError(f"selected combination is incomplete (missing {arm} arm): {combo}")
+ _validate_attempt_artifacts(job, skill, holdout, identity["attempts"], arm_identity)
+ for combo, identity in evidence:
+ if identity.get("canary_error") or identity.get("measurement_error"):
+ continue
+ receipt_metadata = {
+ **_combo_identity(combo),
+ **provenance_by_root.get(combo.parent.resolve(), {}),
+ }
+ for arm in ("skill", "control"):
+ job, arm_identity = _arm_evidence(combo, arm)
+ if job.is_dir():
+ validate_job_receipts(job, {**receipt_metadata, "arm": arm},
+ identity=arm_identity)
+ scoring = dict(scoring_identity or current_scoring_identity())
+ shared = {field: evidence[0][1][field] for field in _EVIDENCE_FIELDS}
+ out = output or HARBOR_DIR / f"{skill}.rescored.json"
+
+ rows = (_prior_rows(out, skill, shared, scoring, skill_sha256) if selected_mode else {})
+ last_checkpoint = None
+ for combo, identity in evidence:
+ matching = _matching_row_key(identity, rows)
+ for prior_key, prior_row in list(rows.items()):
+ if prior_key != matching and _same_logical_cell(identity, prior_row):
+ del rows[prior_key]
+ key = matching or _row_key(identity, rows)
+ # Selected progressive runs carry earlier paths so pre-measurement failures can enter the
+ # first published matrix. Compatible prior lifts are durable grade receipts, not an order
+ # to pay Agy again or replace a stochastic score.
+ output_identity = _output_identity(identity, scoring)
+ if identity.get("canary_error") or identity.get("measurement_error"):
+ error = str(identity.get("canary_error") or identity["measurement_error"])[:300]
+ rows[key] = {**output_identity, "error": error}
+ log(f"[rescore] {key:<44} {error}")
+ continue
+ if "lift" in rows.get(key, {}):
+ continue
+ skill_arm, skill_identity = _arm_evidence(combo, "skill")
+ control_arm, control_identity = _arm_evidence(combo, "control")
+ if not (skill_arm.is_dir() and control_arm.is_dir()):
+ error = str(identity.get("canary_error") or identity.get("measurement_error")
+ or "incomplete (missing an arm)")[:300]
+ if "lift" not in rows.get(key, {}):
+ rows[key] = {**output_identity, "error": error}
+ log(f"[rescore] {key:<44} {error}")
+ continue
+ try:
+ _validate_attempt_artifacts(skill_arm, skill, holdout, identity["attempts"],
+ skill_identity)
+ _validate_attempt_artifacts(control_arm, skill, holdout, identity["attempts"],
+ control_identity)
+ skipped = (broken_tasks(skill_arm, skill_identity)
+ | broken_tasks(control_arm, control_identity))
+ arms = {
+ "skill": score(collect_answers(
+ skill_arm, broken_trials(skill_arm, skill_identity), identity=skill_identity),
+ skill, holdout, skipped, _AGY_SCORE_CONCURRENCY),
+ "control": score(collect_answers(
+ control_arm, broken_trials(control_arm, control_identity), identity=control_identity),
+ skill, holdout, skipped, _AGY_SCORE_CONCURRENCY),
+ }
+ except RuntimeError as error:
+ if "lift" not in rows.get(key, {}):
+ rows[key] = {**output_identity, "error": str(error)[:300]}
+ log(f"[rescore] {key:<44} UNSCORABLE: {str(error)[:90]}")
+ continue
+ skill_mean = sum(arms["skill"]) / len(arms["skill"])
+ control_mean = sum(arms["control"]) / len(arms["control"])
+ lift = skill_mean - control_mean
+ verdict = "helps" if lift > 0.05 else "no lift" if lift >= -0.05 else "HURTS"
+ note = f" [{len(skipped)} dropped]" if skipped else ""
+ log(f"[rescore] {key:<44} skill {skill_mean:.3f} control {control_mean:.3f} "
+ f"lift {lift:+.3f} ({verdict}) n={len(arms['skill'])}{note}")
+ rows[key] = {
+ **output_identity,
+ "skill_mean": skill_mean,
+ "control_mean": control_mean,
+ "lift": lift,
+ "skill_scores": arms["skill"],
+ "control_scores": arms["control"],
+ "tasks_scored": len(arms["skill"]),
+ "tasks_dropped": sorted(skipped),
+ }
+ checkpoint = _summary(skill, rows, shared, scoring)
+ if checkpoint is not None:
+ atomic_write_json(out, checkpoint)
+ last_checkpoint = checkpoint
+
+ summary = _summary(skill, rows, shared, scoring)
+ if summary is None:
+ log("[rescore] nothing scorable; no matrix published")
+ raise SystemExit(f"[rescore] no combination for '{skill}' was measured; nothing was measured.")
+ if summary != last_checkpoint:
+ atomic_write_json(out, summary)
+ log(f"[rescore] {summary['scored']} scorable of {len(rows)}; "
+ f"mean lift {summary['mean_lift']:+.4f}")
+ log(f"[rescore] written to {out}")
+ return summary
+
+
+if __name__ == "__main__":
+ import argparse
+
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("skill")
+ parser.add_argument("--jobs", action="append", default=None, metavar="PATH",
+ help="job root (repeatable; default runs/harbor/jobs/)")
+ parser.add_argument("--legacy-metadata", type=Path, default=None, metavar="PATH",
+ help="authoritative metadata manifest for preserved proprietary jobs")
+ parser.add_argument("--provenance-metadata", action="append", type=Path, default=None,
+ metavar="PATH",
+ help="provenance migration JSON paired with each repeated --jobs root")
+ parser.add_argument("--combination-path", action="append", type=Path, default=None,
+ metavar="PATH", help="exact completed combination directory (repeatable)")
+ parser.add_argument("--output", type=Path, default=None, metavar="PATH",
+ help="atomic matrix destination (default runs/harbor/.rescored.json)")
+ args = parser.parse_args()
+ rescore(
+ args.skill, [Path(root) for root in args.jobs] if args.jobs else None,
+ legacy_metadata=args.legacy_metadata,
+ provenance_metadata=args.provenance_metadata,
+ combination_paths=args.combination_path,
+ output=args.output,
+ )
diff --git a/ingot/optimize/harbor_targets.py b/ingot/optimize/harbor_targets.py
new file mode 100644
index 0000000..3c8f857
--- /dev/null
+++ b/ingot/optimize/harbor_targets.py
@@ -0,0 +1,425 @@
+"""Identity and safe routing helpers for Harbor's local model targets.
+
+Endpoint addresses are runtime inputs. They are retained only on the immutable target used to make
+the request; the stable job identity uses a fingerprint of the normalized address and served model.
+"""
+from __future__ import annotations
+
+import hashlib
+import json
+import os
+import urllib.error
+import urllib.parse
+import urllib.request
+from dataclasses import dataclass
+from typing import Mapping
+
+
+MIN_CONTEXT_LENGTH = 32_768
+PROTOCOLS = frozenset({"chat", "responses", "messages"})
+PROTOCOL_PATHS = {
+ "chat": "/v1/chat/completions",
+ "responses": "/v1/responses",
+ "messages": "/v1/messages",
+}
+
+# This is the complete local matrix allowlist. Keep it in one mapping so routing cannot silently
+# fall back to a provider's default protocol when a new Harbor adapter is added.
+HARNESS_PROTOCOLS = {
+ "claude-code": "messages",
+ "terminus-2": "chat",
+ "goose": "chat",
+ "opencode": "chat",
+ "openclaw": "chat",
+ "mini-swe-agent": "chat",
+ "codex": "responses",
+ "aider": "chat",
+ "pi": "chat",
+}
+
+# Alias configuration pins served identity and display/scale provenance. Context length alone comes
+# from discovery; parse_target uses the minimum accepted value until that live check completes.
+TARGETS = {
+ "dell-qwen": {"display_name": "Qwen/Qwen3.6-27B", "served_model": "dot-backbone",
+ "family": "Qwen3.6", "parameter_billions": 27.0,
+ "quantization": "fp8-published", "tool_parser": "qwen3_xml"},
+ # Alias names avoid punctuation; the server's exact model ID retains the decimal point.
+ "qwen35-08b": {"display_name": "Qwen/Qwen3.5-0.8B", "served_model": "qwen35-0.8b",
+ "family": "Qwen3.5", "parameter_billions": 0.8,
+ "quantization": "fp8-load", "tool_parser": "qwen3_coder"},
+ "qwen35-2b": {"display_name": "Qwen/Qwen3.5-2B", "served_model": "qwen35-2b",
+ "family": "Qwen3.5", "parameter_billions": 2.0,
+ "quantization": "fp8-load", "tool_parser": "qwen3_coder"},
+ "qwen35-4b": {"display_name": "Qwen/Qwen3.5-4B", "served_model": "qwen35-4b",
+ "family": "Qwen3.5", "parameter_billions": 4.0,
+ "quantization": "fp8-load", "tool_parser": "qwen3_coder"},
+ "qwen35-9b": {"display_name": "Qwen/Qwen3.5-9B", "served_model": "qwen35-9b",
+ "family": "Qwen3.5", "parameter_billions": 9.0,
+ "quantization": "fp8-load", "tool_parser": "qwen3_coder"},
+ "orin-qwen35-9b": {"display_name": "Qwen/Qwen3.5-9B (Q4_K_M)",
+ "served_model": "qwen3.5:9b", "family": "Qwen3.5",
+ "parameter_billions": 9.7, "quantization": "Q4_K_M",
+ "tool_parser": "ollama", "context_api": "ollama-ps"},
+ "spark-deepseek": {"display_name": "deepseek-ai/DeepSeek-V4-Flash-0731 (NVFP4)", "served_model": "deepseek-v4-flash",
+ "family": "DeepSeek V4 Flash", "parameter_billions": None,
+ "quantization": "nvfp4", "tool_parser": "deepseek_v3"},
+ "orin-abliterated": {"display_name": "ablit35b (34.66B, Q4_K_M)", "served_model": "ablit35b",
+ "family": "Abliterated 35B", "parameter_billions": 35.0,
+ "quantization": "gguf", "tool_parser": "llama.cpp"},
+}
+
+_PROVIDER_PREFIXES = (
+ "ANTHROPIC_",
+ "CLAUDE_",
+ "OPENAI_",
+ "OPENROUTER_",
+ "CODEX_",
+ "LITELLM_",
+ "GEMINI_",
+ "GOOSE_",
+ "AIDER_",
+ "OPENCODE_",
+ "OPENCLAW_",
+ "MINI_SWE_AGENT_",
+ "PI_",
+)
+_EXACT_SECRET_KEYS = frozenset({"API_KEY", "MODEL_API_KEY"})
+
+
+def _canonical_url(base_url: str) -> str:
+ value = str(base_url).strip()
+ if value.endswith("/"):
+ value = value[:-1]
+ parsed = urllib.parse.urlsplit(value)
+ if parsed.scheme not in {"http", "https"} or not parsed.netloc:
+ raise ValueError("base URL must be an http(s) URL")
+ if parsed.username or parsed.password:
+ raise ValueError("base URL must not contain credentials")
+ if parsed.query or parsed.fragment:
+ raise ValueError("base URL must not contain a query or fragment")
+ return value
+
+
+def _alias_config(alias: str) -> dict[str, str]:
+ try:
+ return TARGETS[alias]
+ except KeyError as exc:
+ raise ValueError(f"unknown local target alias: {alias}") from exc
+
+
+@dataclass(frozen=True)
+class LocalTarget:
+ alias: str
+ display_name: str
+ base_url: str
+ served_model: str
+ context_length: int
+ protocols: frozenset[str]
+ family: str = ""
+ parameter_billions: float | None = None
+ quantization: str = ""
+ tool_parser: str = ""
+
+ @property
+ def fingerprint(self) -> str:
+ payload = json.dumps(
+ {"base_url": _canonical_url(self.base_url), "served_model": self.served_model},
+ sort_keys=True,
+ separators=(",", ":"),
+ ).encode()
+ return hashlib.sha256(payload).hexdigest()[:12]
+
+ @property
+ def job_slug(self) -> str:
+ return f"{self.alias}-{self.fingerprint}"
+
+
+def parse_target(spec: str) -> LocalTarget:
+ """Parse ``alias=base_url`` into a provisional configured target."""
+ alias, separator, base_url = str(spec).partition("=")
+ if not separator or not alias.strip() or not base_url.strip():
+ raise ValueError("target must be specified as ALIAS=BASE_URL")
+ config = _alias_config(alias.strip())
+ return LocalTarget(
+ alias=alias.strip(),
+ display_name=config["display_name"],
+ base_url=_canonical_url(base_url),
+ served_model=config["served_model"],
+ context_length=MIN_CONTEXT_LENGTH,
+ protocols=PROTOCOLS,
+ family=config["family"],
+ parameter_billions=config["parameter_billions"],
+ quantization=config["quantization"],
+ tool_parser=config["tool_parser"],
+ )
+
+
+def _ollama_runtime_context(base_url: str, served_model: str, timeout: float) -> object:
+ """Read the loaded Ollama runtime context when its OpenAI model list omits that field."""
+ request = urllib.request.Request(f"{base_url}/api/ps", method="GET")
+ try:
+ with urllib.request.urlopen(request, timeout=timeout) as response:
+ status = getattr(response, "status", None)
+ if status is not None and not 200 <= int(status) < 300:
+ raise RuntimeError(f"Ollama process list returned HTTP {status}")
+ payload = json.loads(response.read())
+ except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, ValueError) as exc:
+ raise RuntimeError(f"Ollama process list failed: {exc}") from exc
+ models = payload.get("models") if isinstance(payload, dict) else None
+ if not isinstance(models, list):
+ return None
+ loaded = next((item for item in models
+ if isinstance(item, dict) and item.get("name") == served_model), None)
+ return loaded.get("context_length") if loaded else None
+
+
+def discover_target(alias: str, base_url: str, *, timeout: float = 10.0) -> LocalTarget:
+ """Fetch ``/v1/models`` and return a target after identity/context validation."""
+ config = _alias_config(alias)
+ canonical = _canonical_url(base_url)
+ request = urllib.request.Request(f"{canonical}/v1/models", method="GET")
+ try:
+ with urllib.request.urlopen(request, timeout=timeout) as response:
+ status = getattr(response, "status", None)
+ if status is not None and int(status) >= 400:
+ raise RuntimeError(f"model discovery returned HTTP {status}")
+ payload = json.loads(response.read())
+ except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, ValueError) as exc:
+ if isinstance(exc, ValueError) and str(exc).startswith("model discovery"):
+ raise
+ raise RuntimeError(f"model discovery failed for {alias}: {exc}") from exc
+ if not isinstance(payload, dict) or not isinstance(payload.get("data"), list):
+ raise RuntimeError("model discovery returned an invalid response object")
+ model = next((entry for entry in payload["data"]
+ if isinstance(entry, dict) and entry.get("id") == config["served_model"]), None)
+ if model is None:
+ ids = [entry.get("id") for entry in payload["data"] if isinstance(entry, dict)]
+ raise ValueError(f"served model {config['served_model']!r} not found; received {ids!r}")
+ context = model.get("max_model_len", model.get("context_length"))
+ if context is None and isinstance(model.get("meta"), dict):
+ context = model["meta"].get("n_ctx")
+ if context is None and config.get("context_api") == "ollama-ps":
+ context = _ollama_runtime_context(canonical, config["served_model"], timeout)
+ if not isinstance(context, int) or isinstance(context, bool):
+ raise ValueError(f"served model {config['served_model']!r} has no context length")
+ if context < MIN_CONTEXT_LENGTH:
+ raise ValueError(f"served model {config['served_model']!r} context length {context} is below "
+ f"the minimum {MIN_CONTEXT_LENGTH}")
+ return LocalTarget(
+ alias=alias,
+ display_name=config["display_name"],
+ base_url=canonical,
+ served_model=config["served_model"],
+ context_length=context,
+ protocols=PROTOCOLS,
+ family=config["family"],
+ parameter_billions=config["parameter_billions"],
+ quantization=config["quantization"],
+ tool_parser=config["tool_parser"],
+ )
+
+
+def protocol_for(harness: str) -> str:
+ try:
+ return HARNESS_PROTOCOLS[harness]
+ except KeyError as exc:
+ raise ValueError(f"unsupported harness for local target: {harness}") from exc
+
+
+def harbor_model(target: LocalTarget, harness: str) -> str:
+ protocol_for(harness)
+ # Harbor's local CLI adapters use provider/model identifiers; Claude Code consumes the
+ # Anthropic-compatible model ID directly.
+ if harness == "claude-code":
+ return target.served_model
+ if harness == "opencode":
+ return f"local/{target.served_model}"
+ return f"openai/{target.served_model}"
+
+
+def harbor_agent_kwargs(target: LocalTarget, harness: str) -> dict[str, str]:
+ """Return only adapter kwargs that cannot be supplied through the child environment."""
+ protocol_for(harness)
+ if harness == "terminus-2":
+ return {"api_base": f"{_canonical_url(target.base_url)}/v1"}
+ if harness == "openclaw":
+ # Harbor defaults this adapter to "high", which both local models reject.
+ return {"thinking": "off"}
+ if harness == "opencode":
+ model = target.served_model
+ # OpenCode otherwise requests 32k output tokens. Reserve three quarters of the
+ # discovered context for its prompt, tool calls, and accumulated transcript.
+ output_limit = target.context_length // 4
+ return {"opencode_config": {"provider": {"local": {
+ "npm": "@ai-sdk/openai-compatible",
+ "options": {"baseURL": f"{_canonical_url(target.base_url)}/v1", "apiKey": "local"},
+ "models": {model: {"limit": {"context": target.context_length,
+ "output": output_limit}}},
+ }}}}
+ return {}
+
+
+def scrub_provider_env(parent: Mapping[str, str]) -> dict[str, str]:
+ """Copy a parent environment without provider credentials or routing controls."""
+ return {key: value for key, value in parent.items()
+ if not key.startswith(_PROVIDER_PREFIXES) and key not in _EXACT_SECRET_KEYS}
+
+
+def local_agent_env(target: LocalTarget, harness: str) -> dict[str, str]:
+ """Build a complete child environment with a literal local credential sentinel."""
+ protocol = protocol_for(harness)
+ # This mapping is repeated as Harbor --ae values. Do not copy ambient process
+ # variables into it: only local sentinels and routing settings belong here.
+ env: dict[str, str] = {}
+ base_url = _canonical_url(target.base_url)
+ openai_base_url = f"{base_url}/v1"
+ # Keep every provider key at the non-secret sentinel. Adapters choose their own protocol
+ # below, but a generic adapter must never discover a real inherited key and fall back to it.
+ env.update({
+ "ANTHROPIC_API_KEY": "local",
+ "OPENAI_API_KEY": "local",
+ "CODEX_API_KEY": "local",
+ })
+ if protocol == "messages":
+ env.update({
+ "ANTHROPIC_BASE_URL": base_url,
+ "ANTHROPIC_MODEL": target.served_model,
+ })
+ elif protocol == "responses":
+ env.update({
+ "OPENAI_BASE_URL": openai_base_url,
+ "OPENAI_API_BASE": openai_base_url,
+ "OPENAI_HOST": base_url,
+ })
+ else:
+ env.update({
+ "OPENAI_BASE_URL": openai_base_url,
+ "OPENAI_API_BASE": openai_base_url,
+ # Goose reads OPENAI_HOST rather than the OpenAI SDK spelling.
+ "OPENAI_HOST": base_url,
+ })
+ return env
+
+
+def probe_protocol(target: LocalTarget, protocol: str, *, timeout: float = 20.0) -> None:
+ """Send a one-token request and require a successful, non-empty JSON object response."""
+ if protocol not in PROTOCOL_PATHS:
+ raise ValueError(f"unsupported protocol: {protocol}")
+ if protocol not in target.protocols:
+ raise ValueError(f"target does not support protocol: {protocol}")
+ body: dict[str, object]
+ headers = {"Content-Type": "application/json", "Authorization": "Bearer local"}
+ if protocol == "messages":
+ body = {"model": target.served_model, "max_tokens": 1,
+ "messages": [{"role": "user", "content": "ping"}]}
+ headers["anthropic-version"] = "2023-06-01"
+ headers["x-api-key"] = "local"
+ elif protocol == "responses":
+ body = {"model": target.served_model, "input": "ping", "max_output_tokens": 1}
+ else:
+ body = {"model": target.served_model, "messages": [{"role": "user", "content": "ping"}],
+ "max_tokens": 1}
+ if target.family.startswith("Qwen"):
+ body["messages"] = [{"role": "user", "content": "ping /no_think"}]
+ request = urllib.request.Request(
+ f"{_canonical_url(target.base_url)}{PROTOCOL_PATHS[protocol]}",
+ data=json.dumps(body).encode(),
+ headers=headers,
+ method="POST",
+ )
+ try:
+ with urllib.request.urlopen(request, timeout=timeout) as response:
+ status = getattr(response, "status", None)
+ if status is not None and not 200 <= int(status) < 300:
+ raise RuntimeError(f"{protocol} probe returned HTTP {status}")
+ payload = json.loads(response.read())
+ except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, ValueError) as exc:
+ if isinstance(exc, ValueError) and str(exc).startswith(protocol + " probe returned"):
+ raise
+ raise RuntimeError(f"{protocol} probe failed: {exc}") from exc
+ if not isinstance(payload, dict) or not payload:
+ raise RuntimeError(f"{protocol} probe returned an empty response object; expected non-empty")
+
+
+def probe_chat_tool_round_trip(target: LocalTarget, *, timeout: float = 20.0) -> None:
+ """Require one parsed function call and a successful tool-result continuation."""
+ url = f"{_canonical_url(target.base_url)}{PROTOCOL_PATHS['chat']}"
+ headers = {"Content-Type": "application/json", "Authorization": "Bearer local"}
+ tool = {
+ "type": "function",
+ "function": {
+ "name": "ingot_echo",
+ "description": "Return the supplied string unchanged.",
+ "parameters": {
+ "type": "object",
+ "properties": {"value": {"type": "string"}},
+ "required": ["value"],
+ },
+ },
+ }
+
+ def post(body: dict[str, object]) -> dict:
+ request = urllib.request.Request(
+ url, data=json.dumps(body).encode(), headers=headers, method="POST")
+ try:
+ with urllib.request.urlopen(request, timeout=timeout) as response:
+ status = getattr(response, "status", None)
+ if status is not None and not 200 <= int(status) < 300:
+ raise RuntimeError(f"chat tool probe returned HTTP {status}")
+ payload = json.loads(response.read())
+ except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError,
+ ValueError) as exc:
+ raise RuntimeError(f"chat tool probe failed: {exc}") from exc
+ if not isinstance(payload, dict):
+ raise RuntimeError("chat tool probe returned an invalid response object")
+ return payload
+
+ common = {
+ "model": target.served_model,
+ "chat_template_kwargs": {"enable_thinking": False},
+ "reasoning_effort": "none",
+ "temperature": 0,
+ }
+ first = post({
+ **common,
+ "messages": [
+ {"role": "system", "content": "Call ingot_echo with value cutover-ok. /no_think"},
+ {"role": "user", "content": "Use the tool now."},
+ ],
+ "tools": [tool],
+ "tool_choice": "required",
+ "max_tokens": 128,
+ })
+ try:
+ assistant = first["choices"][0]["message"]
+ call = assistant["tool_calls"][0]
+ arguments = json.loads(call["function"]["arguments"])
+ if call["function"]["name"] != "ingot_echo" or arguments != {"value": "cutover-ok"}:
+ raise ValueError
+ call_id = call["id"]
+ except (KeyError, IndexError, TypeError, ValueError, json.JSONDecodeError) as exc:
+ raise RuntimeError("chat tool probe did not return the required parsed tool call") from exc
+
+ assistant_turn = {
+ "role": "assistant",
+ "content": assistant.get("content") or "",
+ "tool_calls": assistant["tool_calls"],
+ }
+ second = post({
+ **common,
+ "messages": [
+ {"role": "system", "content":
+ "After the tool result, reply exactly cutover-ok. /no_think"},
+ {"role": "user", "content": "Use the tool now."},
+ assistant_turn,
+ {"role": "tool", "tool_call_id": call_id, "content": "cutover-ok"},
+ ],
+ "tools": [tool],
+ "max_tokens": 64,
+ })
+ try:
+ content = second["choices"][0]["message"]["content"]
+ except (KeyError, IndexError, TypeError) as exc:
+ raise RuntimeError("chat tool probe returned an invalid continuation") from exc
+ if str(content).strip() != "cutover-ok":
+ raise RuntimeError("chat tool probe did not consume the tool result")
diff --git a/ingot/optimize/ingress.py b/ingot/optimize/ingress.py
new file mode 100644
index 0000000..44934ca
--- /dev/null
+++ b/ingot/optimize/ingress.py
@@ -0,0 +1,309 @@
+"""Quarantine a vetted new skill for human review without adding it to the served library."""
+from __future__ import annotations
+
+import hashlib
+import json
+import logging
+import os
+import shutil
+import threading
+import time
+import uuid
+from pathlib import Path
+
+from ingot.mcp_server.registry import (SLUG_RE, load_skills, normalized_frontmatter, skill_revision,
+ writable_skill_dir)
+from ingot.optimize import promote, tree
+from ingot.optimize.evidence import recorded_path
+from ingot import paths
+
+
+
+def evidence_dir() -> Path:
+ return paths.runs() / "evidence"
+
+
+def audit_file() -> Path:
+ return paths.runs() / "ingress-audit.jsonl"
+
+MAX_FILES = 32
+MAX_TOTAL_CHARS = 1_000_000
+MAX_FIELD_CHARS = 4_000
+MAX_EVIDENCE_ITEMS = 12
+TEXT_SUFFIXES = {".md", ".txt", ".py", ".sh", ".js", ".ts", ".json", ".yaml", ".yml",
+ ".toml", ".cfg"}
+logger = logging.getLogger(__name__)
+_SUBMIT_LOCK = threading.Lock()
+
+
+def _text(name: str, value: object, *, limit: int = MAX_FIELD_CHARS) -> str:
+ if not isinstance(value, str) or not value.strip():
+ raise ValueError(f"{name} is required")
+ value = value.strip()
+ if len(value) > limit:
+ raise ValueError(f"{name} exceeds {limit} characters")
+ return value
+
+
+def _files(value: object) -> dict[str, str]:
+ if not isinstance(value, dict) or len(value) > MAX_FILES:
+ raise ValueError(f"files must be an object with at most {MAX_FILES} entries")
+ result = {}
+ portable_paths = set()
+ for raw_path, content in value.items():
+ if not isinstance(raw_path, str) or not isinstance(content, str):
+ raise ValueError("file paths and contents must be strings")
+ path = tree.portable_path(raw_path)
+ folded = path.as_posix().casefold()
+ if folded in portable_paths:
+ raise ValueError(f"component path collides case-insensitively: {raw_path}")
+ portable_paths.add(folded)
+ if path.suffix.lower() not in TEXT_SUFFIXES:
+ raise ValueError(f"unsupported skill file type: {raw_path}")
+ result[f"file:{path.as_posix()}"] = content
+ if sum(len(item) for item in result.values()) > MAX_TOTAL_CHARS:
+ raise ValueError(f"files exceed {MAX_TOTAL_CHARS} total characters")
+ return result
+
+
+def _proposal_id(identity: dict) -> str:
+ canonical = json.dumps(identity, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
+ return hashlib.sha256(canonical.encode()).hexdigest()[:24]
+
+
+def _write_evidence(skill: str, proposal: dict, gate: dict) -> dict[str, str]:
+ """The bundle a reviewer reads.
+
+ Narrative sections are optional. An agent proposing a skill it authored can state a pressure
+ scenario and a verification it ran; an operator ingesting a third-party package cannot, and
+ inventing one on their behalf would put a fabricated claim in front of the person whose job is
+ to check claims. Absent sections are omitted rather than filled in."""
+ root = evidence_dir() / skill / f"creation-{proposal['proposal_id']}"
+ root.mkdir(parents=True, exist_ok=True)
+ bundle = {"schema_version": "ingot/skill-create/v1", "skill": skill,
+ "created": proposal["created"], "proposal": proposal, "gate": gate}
+ verification = proposal.get("verification") or {}
+ markdown = "\n".join([
+ f"# New skill proposal: {skill}", "", f"**Source:** `{proposal['source']}`", "",
+ proposal["summary"], "", "## Evidence", "",
+ *[f"- {item}" for item in proposal["evidence"]], "",
+ *(["## Pressure scenario", "", proposal["pressure_scenario"], ""]
+ if proposal.get("pressure_scenario") else []),
+ *(["## Verification", "",
+ f"- Command: `{verification.get('command', '')}`",
+ f"- Result: {verification.get('result', '')}", ""] if verification else []),
+ "Human approval is required before this skill enters the served library.", "",
+ ])
+ paths = ((root / "evidence.json", json.dumps(bundle, indent=2) + "\n"),
+ (root / "EVIDENCE.md", markdown))
+ for path, content in paths:
+ temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp")
+ temporary.write_text(content, encoding="utf-8")
+ temporary.replace(path)
+ return {"json": recorded_path(paths[0][0]), "markdown": recorded_path(paths[1][0])}
+
+
+def _publish(skill: str, record: dict) -> bool:
+ promote.pending_dir().mkdir(parents=True, exist_ok=True)
+ destination = promote.pending_path(skill)
+ temporary = destination.with_name(f".{destination.name}.{uuid.uuid4().hex}.tmp")
+ payload = (json.dumps(record, indent=2) + "\n").encode()
+ fd = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
+ try:
+ view = memoryview(payload)
+ while view:
+ written = os.write(fd, view)
+ if written <= 0:
+ raise OSError("pending proposal write made no progress")
+ view = view[written:]
+ os.fsync(fd)
+ finally:
+ os.close(fd)
+ try:
+ os.link(temporary, destination)
+ except FileExistsError as exc:
+ existing = promote.load_pending(skill)
+ existing_id = (existing.get("creation") or {}).get("proposal_id") if existing else None
+ if existing_id == record["creation"]["proposal_id"]:
+ return False
+ raise ValueError(f"review slot is occupied for '{skill}'") from exc
+ finally:
+ temporary.unlink(missing_ok=True)
+ return True
+
+
+def _audit(record: dict) -> None:
+ audit_file().parent.mkdir(parents=True, exist_ok=True)
+ fd = os.open(audit_file(), os.O_APPEND | os.O_CREAT | os.O_WRONLY, 0o600)
+ try:
+ view = memoryview((json.dumps(record, separators=(",", ":")) + "\n").encode())
+ while view:
+ written = os.write(fd, view)
+ if written <= 0:
+ raise OSError("ingress audit write made no progress")
+ view = view[written:]
+ os.fsync(fd)
+ finally:
+ os.close(fd)
+
+
+def _accept_slot(skill: str) -> Path:
+ """The skill must be new. Returns the directory it would occupy.
+
+ Shared by every admission front door, so a package cannot enter through one path what another
+ would refuse."""
+ if not isinstance(skill, str) or len(skill) > 80 or not SLUG_RE.fullmatch(skill):
+ raise ValueError(f"invalid skill name: {skill!r}")
+ target = writable_skill_dir(skill)
+ if target.exists() or target.is_symlink() or any(item.name == skill for item in load_skills()):
+ raise ValueError(f"skill '{skill}' already exists; propose an update instead")
+ return target
+
+
+def build_components(skill: str, description: str, body: str, files: dict[str, str],
+ frontmatter: dict) -> tuple[dict[str, str], dict]:
+ """The component map a candidate is made of, plus its normalized frontmatter.
+
+ One implementation for every source: path validation, portability, and the field limits are
+ properties of what Ingot will serve, not of who proposed it. Public because an ingest adapter
+ has to know the candidate revision -- which is a property of these components, not of the
+ directory they were read from -- before it can build a candidate manifest."""
+ description = " ".join(_text("description", description, limit=2_000).split())
+ body = _text("body", body, limit=200_000)
+ metadata = normalized_frontmatter(skill, description, frontmatter)
+ canonical_frontmatter = json.dumps(metadata, sort_keys=True, separators=(",", ":"),
+ ensure_ascii=False)
+ if len(canonical_frontmatter) > 20_000:
+ raise ValueError("frontmatter exceeds 20000 characters")
+ return ({"description": description, "body": body,
+ "frontmatter": canonical_frontmatter, **_files(files)}, metadata)
+
+
+def _quarantine(skill: str, record: dict, proposal: dict, gate: dict) -> dict:
+ """Claim the review slot, write the evidence bundle, publish the pending record, audit.
+
+ The only path that creates a pending proposal. `_publish` links the record into place, which is
+ atomic across processes, so two submitters racing for one slot cannot both win -- the in-process
+ lock below narrows the window but is not what makes it safe."""
+ with _SUBMIT_LOCK:
+ proposal_id = proposal["proposal_id"]
+ existing = promote.load_pending(skill)
+ if existing:
+ existing_id = (existing.get("creation") or {}).get("proposal_id")
+ if existing_id == proposal_id:
+ return {"status": "duplicate", "skill": skill, "proposal_id": proposal_id,
+ "promotable": True}
+ raise ValueError(f"review slot is occupied for '{skill}'")
+ record["evidence_paths"] = _write_evidence(skill, proposal, gate)
+ evidence_root = evidence_dir() / skill / f"creation-{proposal_id}"
+ try:
+ published = _publish(skill, record)
+ except Exception:
+ shutil.rmtree(evidence_root, ignore_errors=True)
+ raise
+ if not published:
+ return {"status": "duplicate", "skill": skill, "proposal_id": proposal_id,
+ "promotable": True}
+ try:
+ _audit({"schema_version": 1, "ts": int(time.time()), "action": "quarantine",
+ "skill": skill, "proposal_id": proposal_id,
+ "challenger_revision": proposal["revision"],
+ "producer": proposal["producer"]})
+ except Exception:
+ logger.warning("Quarantined creation proposal %s, but its audit write failed",
+ proposal_id, exc_info=True)
+ return {"status": "quarantined", "skill": skill, "proposal_id": proposal_id,
+ "promotable": True}
+
+
+def submit_package_ingest(*, skill: str, components: dict[str, str], metadata: dict,
+ revision: str, source: str, candidate: dict, identity: str,
+ review_summary: list[str], producer: str, caller: str,
+ candidate_tree: dict) -> dict:
+ """Quarantine a package an operator ingested from somewhere else.
+
+ Distinct from `submit_skill_create` in what it is allowed to claim, not in what it does. An
+ agent proposing a skill it wrote can attest to a pressure scenario and a verification run; an
+ operator pointing at a third-party directory can attest only to where it came from and what the
+ deterministic review found. Both produce the same record, take the same slot, and are equally
+ inert until a human approves them.
+
+ Also distinct in what it carries. This path has real bytes on disk, so the candidate is a
+ staged tree covering every file in the package; `submit_skill_create` receives strings over
+ MCP and has no bytes to preserve."""
+ _accept_slot(skill)
+ tree.verify_manifest(candidate_tree)
+
+ proposal = {
+ "skill": skill,
+ "revision": revision,
+ "source": _text("source", source),
+ "summary": f"Ingested package '{skill}' from {candidate['source']['type']}",
+ # Real, machine-produced findings. Nothing here is a claim a person made.
+ "evidence": review_summary or ["Deterministic review found no findings."],
+ "created": candidate["created_at"],
+ "producer": _text("producer", producer),
+ "caller": _text("caller", caller),
+ "candidate": candidate,
+ "frontmatter": metadata,
+ # Identity is the candidate's, so the same bytes ingested twice are one proposal even
+ # though the manifest's timestamp differs.
+ "proposal_id": _proposal_id({"candidate_identity": identity}),
+ }
+ gate = {"promotable": True, "blocked": [],
+ "warnings": ["Ingested package; no behavioural evidence and no held-out A/B exists.",
+ *review_summary],
+ "kind": "package_ingest"}
+ record = {"skill": skill, "kind": "creation", "created": proposal["created"],
+ "changed_components": list(components), "champion_components": {},
+ "challenger_components": components, "tree": candidate_tree, "gate": gate,
+ "evidence": {"schema_version": "ingot/skill-create/v1",
+ "challenger": {"revision": revision}, "gate": gate},
+ "creation": proposal}
+ return _quarantine(skill, record, proposal, gate)
+
+
+def submit_skill_create(*, skill: str, description: str, body: str, files: dict[str, str],
+ frontmatter: dict,
+ summary: str, source: str, producer: str, caller: str,
+ evidence: list[str], pressure_scenario: str, risk: str,
+ verification_status: str, verification_command: str,
+ verification_result: str) -> dict:
+ """Validate and quarantine one new skill package; never add it to the active registry."""
+ target = _accept_slot(skill)
+ components, metadata = build_components(skill, description, body, files, frontmatter)
+ description = components["description"]
+ if not isinstance(verification_status, str) or verification_status.strip().lower() != "passed":
+ raise ValueError("verification_status must be passed before proposing")
+ if not isinstance(evidence, list) or not 2 <= len(evidence) <= MAX_EVIDENCE_ITEMS:
+ raise ValueError(f"evidence must contain 2 to {MAX_EVIDENCE_ITEMS} concrete items")
+ evidence = [_text("evidence item", item) for item in evidence]
+ if len(set(evidence)) != len(evidence):
+ raise ValueError("evidence items must be distinct")
+
+ revision = skill_revision(target, components)
+ proposal = {"skill": skill, "revision": revision, "source": _text("source", source),
+ "summary": _text("summary", summary), "evidence": evidence,
+ "created": int(time.time()),
+ "producer": _text("producer", producer), "caller": _text("caller", caller),
+ "pressure_scenario": _text("pressure_scenario", pressure_scenario),
+ "risk": _text("risk", risk),
+ "verification": {"status": "passed",
+ "command": _text("verification_command", verification_command),
+ "result": _text("verification_result", verification_result)},
+ "frontmatter": metadata}
+ identity = {key: value for key, value in proposal.items()
+ if key not in {"created", "producer", "caller"}}
+ proposal_id = _proposal_id(identity)
+ proposal["proposal_id"] = proposal_id
+ gate = {"promotable": True, "blocked": [],
+ "warnings": ["New skill admission only; no active champion or held-out A/B exists."],
+ "kind": "new_skill_admission"}
+ record = {"skill": skill, "kind": "creation", "created": proposal["created"],
+ "changed_components": list(components), "champion_components": {},
+ "challenger_components": components, "gate": gate,
+ "evidence": {"schema_version": "ingot/skill-create/v1",
+ "challenger": {"revision": revision}, "gate": gate},
+ "creation": proposal}
+
+ return _quarantine(skill, record, proposal, gate)
diff --git a/ingot/optimize/judge.py b/ingot/optimize/judge.py
new file mode 100644
index 0000000..d0b0a58
--- /dev/null
+++ b/ingot/optimize/judge.py
@@ -0,0 +1,244 @@
+"""LLM judge: scores an answer 0..1 and, following the SkillForge paper's multi-dimensional Failure
+Analyzer (Liu et al., "SkillForge", arXiv:2604.08618), classifies each failure across fixed
+dimensions so the search gets *categorized* feedback, not one opaque score. The dimension labels
+also drive success/failure mining (optimize/mine.py) and the candidate search's diagnosis.
+
+Judges against a task `rubric` when given one; with no rubric it grades reference-free (used when
+mining real traces). If a task supplies a `reference` answer, consistency-against-reference is added
+to the prompt (the paper's Consistency-Rate signal, lower variance than a rubric alone)."""
+import json
+import os
+import re
+import time
+
+from langchain_openai import ChatOpenAI
+
+from . import agy_judge
+from . import configured_models
+from . import usage as usage_ledger
+
+# Reward-hacking guard: the judge must NOT be the same model as SKILLOPT_MODEL.
+# If the author and the grader share blind spots, the search learns to please the judge instead of
+# improving the skill. Default judge is a model distinct from both the reflection LM (GLM) and the
+# student (Qwen).
+# JUDGE_MODELS (comma-separated) runs an ensemble and averages, harder still to game. Repeating one
+# ID is not an ensemble: it silently gives one grader multiple votes and pays for every duplicate.
+MODELS = configured_models(
+ "JUDGE_MODELS", os.environ.get("JUDGE_MODEL", "google/gemini-2.5-flash"))
+from . import ZDR_PROVIDER, client_kwargs, skillopt_model, teacher_base_url # noqa: E402
+
+if skillopt_model() in MODELS:
+ print(f"[judge] WARNING: judge model {MODELS} includes the teacher model, which invites "
+ f"reward-hacking (author == grader). Set JUDGE_MODEL to a different model.", flush=True)
+
+# Failure dimensions (the general-purpose analogue of the paper's Knowledge/Tool/Clarification/Style).
+DIMENSIONS = ["correctness", "completeness", "instruction_following", "efficiency"]
+
+# The score is the weighted mean of independent checks, never a number the model picks holistically.
+# A single 0..1 judgement is one noisy measurement: in practice models emit it on a coarse ladder
+# (0.90 / 0.95 / 1.00), so a mean can move by a whole rung because two tasks crossed a boundary, and
+# re-running an unchanged arm moves it by that much. k independent checks average toward the truth
+# roughly as sqrt(k). This is also what a graded checklist buys commercially -- Tessl grades ~38
+# weighted items per run against our 4.
+#
+# A task may declare its own `checklist` (see optimize/draft.py); with none, every task is still
+# graded on these four rather than on one holistic guess.
+DEFAULT_CHECKLIST = [
+ {"id": "correctness", "dimension": "correctness", "weight": 3,
+ "criterion": "The core logic, API usage, and factual claims are right."},
+ {"id": "completeness", "dimension": "completeness", "weight": 2,
+ "criterion": "It covers the whole request, including any edge cases the task or rubric names."},
+ {"id": "instruction_following", "dimension": "instruction_following", "weight": 2,
+ "criterion": "It did what was asked, in the form asked for (e.g. complete runnable code, "
+ "not a description of code)."},
+ {"id": "efficiency", "dimension": "efficiency", "weight": 1,
+ "criterion": "It is concise, with no padded, repeated, or irrelevant output."},
+]
+
+# pass / partial / fail rather than a free float: a three-way verdict per item is a judgement a
+# model makes reliably, and the resolution comes from having many of them, not from pretending a
+# single one is precise to two decimals.
+VERDICT_VALUES = {"pass": 1.0, "partial": 0.5, "fail": 0.0}
+
+_PROMPT = """You are grading an AI assistant's answer to a task against a checklist.
+
+TASK: {task}
+{rubric_block}{reference_block}
+ASSISTANT'S ANSWER:
+{answer}
+{code_block}
+Grade EVERY checklist item independently. Judge each item only on what it asks about, and do not
+let a good or bad impression of the answer overall carry across items. Treat any OBJECTIVE CODE
+CHECK above as ground truth: do not pass a code item whose code is broken or absent.
+
+CHECKLIST:
+{checklist_block}
+For each item give a verdict of "pass", "partial", or "fail", plus a note of at most 12 words.
+The note is required when the verdict is not "pass" and must say what is wrong, not restate the
+criterion. Then write one short paragraph of concrete, actionable feedback on the answer overall.
+
+Respond with ONLY a JSON object:
+{{"items": {{{item_shape}}}, "feedback": ""}}"""
+
+_llms: dict[str, ChatOpenAI] = {}
+
+
+def _get_llm(model: str):
+ if model not in _llms: # built once per model, reuses the HTTP pool across many judge calls
+ _llms[model] = ChatOpenAI(model=model, temperature=0, **client_kwargs(teacher_base_url()))
+ return _llms[model]
+
+
+# OpenRouter phrasings that mean "your model/provider configuration can never work", retrying
+# only burns time, so explain and stop instead.
+_PERMANENT = ("no allowed providers", "no providers are available", "not a valid model",
+ "no endpoints found", "is not available")
+
+
+def _config_error(exc: Exception) -> str | None:
+ text = str(exc).lower()
+ if any(marker in text for marker in _PERMANENT):
+ pins = os.environ.get("OPENROUTER_PROVIDERS", "")
+ hint = (f" You have OPENROUTER_PROVIDERS={pins}, the pinned provider may not serve this "
+ f"model, or may not be ZDR-qualified for it; unset the pin or change the model."
+ if pins else
+ " No ZDR-qualified endpoint may exist for this model; try another model.")
+ return f"OpenRouter cannot route this request: {exc}.{hint}"
+ return None
+
+
+def invoke_retry(llm, messages, tries: int = 3):
+ """Retry transient provider failures (corrupted responses, 5xx) with a short backoff.
+ Permanent configuration errors (model/provider mismatch) fail immediately with an explanation
+ instead of retrying."""
+ for i in range(tries):
+ try:
+ return llm.invoke(messages)
+ except Exception as exc:
+ explained = _config_error(exc)
+ if explained:
+ raise SystemExit(explained) from exc
+ if i == tries - 1:
+ raise
+ time.sleep(5 * (i + 1))
+
+
+def _extract_json(text: str, key: str = "score") -> dict:
+ """First valid JSON object carrying `key`, robust to prose/braces around the JSON."""
+ dec = json.JSONDecoder()
+ for m in re.finditer(r"\{", text):
+ try:
+ obj, _ = dec.raw_decode(text[m.start():])
+ except json.JSONDecodeError:
+ continue
+ if isinstance(obj, dict) and key in obj:
+ return obj
+ return {}
+
+
+def _verdict(raw) -> tuple[float, str]:
+ """(value, note) from one item's grade. Accepts the strict {verdict, note} shape and the bare
+ string a model sometimes emits instead. An unrecognized verdict reads as a fail with the raw
+ text as its note: silently scoring it 1.0 would let a malformed grade inflate the result."""
+ note = ""
+ if isinstance(raw, dict):
+ note = str(raw.get("note", "")).strip()
+ raw = raw.get("verdict", "")
+ word = str(raw).strip().lower()
+ if word in VERDICT_VALUES:
+ return VERDICT_VALUES[word], note
+ return 0.0, note or f"ungraded ({str(raw)[:40]})"
+
+
+def _judge_one(model: str, prompt: str, checklist: list[dict]) -> dict:
+ if os.environ.get("JUDGE_BACKEND", "").strip().lower() == "agy":
+ out, usage = agy_judge.invoke(prompt, checklist)
+ usage_ledger.add("judge", usage, billing_mode="subscription")
+ raw = json.dumps(out)
+ else:
+ msg = invoke_retry(_get_llm(model), prompt)
+ usage_ledger.add("judge", getattr(msg, "usage_metadata", None))
+ raw = msg.content
+ out = _extract_json(raw, "items")
+ items = out.get("items")
+ if not isinstance(items, dict) or not items:
+ return {"items": {}, "feedback": f"Judge output unparseable: {raw[:200]}",
+ "unparseable": True}
+ graded = {}
+ for item in checklist:
+ # a missing item is not a pass; the judge was asked for it and did not answer
+ value, note = _verdict(items.get(item["id"], "")) if item["id"] in items else (0.0, "not graded")
+ graded[item["id"]] = {"value": value, "note": note}
+ return {"items": graded, "feedback": str(out.get("feedback", "")), "unparseable": False}
+
+
+def _weighted(graded: dict, checklist: list[dict]) -> float:
+ total = sum(float(i.get("weight", 1)) for i in checklist) or 1.0
+ return sum(float(i.get("weight", 1)) * graded[i["id"]]["value"]
+ for i in checklist if i["id"] in graded) / total
+
+
+def judge(task: str, rubric: str = "", answer: str = "", reference: str = "",
+ check: dict | None = None, deliverable: str | None = None,
+ checklist: list[dict] | None = None) -> dict:
+ """Return {score, feedback, dimensions, checklist}.
+
+ `score` is the weighted mean of the checklist verdicts, not a number the judge chose. `checklist`
+ carries the per-item verdicts so a reviewer can see which checks moved rather than only that the
+ mean did. `dimensions` keeps its old prose shape for the candidate search and trace mining.
+
+ With multiple JUDGE_MODELS this is an ensemble: each item's value is averaged across judges
+ before weighting, which is smoother than averaging whole-answer scores, and a dimension counts
+ as failed only when a majority of judges failed an item mapped to it.
+
+ `deliverable` (task yaml) declares the expected answer kind; non-code values skip the static
+ Python check, see execcheck.judge_note."""
+ checklist = [i for i in (checklist or DEFAULT_CHECKLIST) if i.get("id") and i.get("criterion")]
+ if not checklist:
+ checklist = DEFAULT_CHECKLIST
+ rubric_block = f"GRADING RUBRIC: {rubric}\n" if rubric else ""
+ reference_block = f"KNOWN-GOOD REFERENCE ANSWER (judge consistency against it): {reference}\n" if reference else ""
+ from . import execcheck # objective code-validity signal to ground the judge
+ code_note = execcheck.judge_note(answer, task, rubric, check_spec=check, deliverable=deliverable)
+ code_block = f"\n{code_note}\n" if code_note else ""
+ checklist_block = "\n".join(f"- {i['id']} (weight {i.get('weight', 1)}): {i['criterion']}"
+ for i in checklist)
+ item_shape = ", ".join(f'"{i["id"]}": {{"verdict": "pass|partial|fail", "note": "..."}}'
+ for i in checklist)
+ prompt = _PROMPT.format(task=task, answer=answer, rubric_block=rubric_block,
+ reference_block=reference_block, code_block=code_block,
+ checklist_block=checklist_block, item_shape=item_shape)
+ models = ([agy_judge.AGY_IDENTITY]
+ if os.environ.get("JUDGE_BACKEND", "").strip().lower() == "agy" else MODELS)
+ results = [_judge_one(m, prompt, checklist) for m in models]
+
+ usable = [r for r in results if not r["unparseable"]]
+ if not usable: # a parse failure is not a skill failure, but it must not read as a clean pass
+ return {"score": 0.0, "feedback": results[0]["feedback"],
+ "dimensions": {d: "pass" for d in DIMENSIONS},
+ "checklist": {i["id"]: {"value": 0.0, "note": "judge output unparseable"}
+ for i in checklist}}
+
+ merged = {i["id"]: {"value": sum(r["items"][i["id"]]["value"] for r in usable) / len(usable),
+ "note": next((r["items"][i["id"]]["note"] for r in usable
+ if r["items"][i["id"]]["value"] < 1.0
+ and r["items"][i["id"]]["note"]), "")}
+ for i in checklist}
+ score = _weighted(merged, checklist)
+
+ # A dimension fails when the items mapped to it did not clean-pass across a majority of judges.
+ dims = {d: "pass" for d in DIMENSIONS}
+ for item in checklist:
+ d = item.get("dimension")
+ entry = merged[item["id"]]
+ if d in dims and entry["value"] < 0.5 and dims[d] == "pass":
+ dims[d] = entry["note"] or f"failed check '{item['id']}'"
+ feedback = " | ".join(f"[{m.split('/')[-1]}] {r['feedback']}"
+ for m, r in zip(models, results) if r["feedback"]) \
+ if len(results) > 1 else usable[0]["feedback"]
+ return {"score": score, "feedback": feedback, "dimensions": dims, "checklist": merged}
+
+
+def failed_dimensions(dimensions: dict) -> list[str]:
+ """Dimension names the judge did NOT mark as a clean pass."""
+ return [d for d, v in dimensions.items() if str(v).strip().lower() not in ("pass", "ok", "", "n/a")]
diff --git a/ingot/optimize/local_traces.py b/ingot/optimize/local_traces.py
new file mode 100644
index 0000000..1193519
--- /dev/null
+++ b/ingot/optimize/local_traces.py
@@ -0,0 +1,739 @@
+"""Normalize local Claude Code and Codex transcripts into trace roots Ingot can mine.
+
+Only the human task, final answer, observed skill identity, timing, token counts, and error count
+cross this boundary. Reasoning, tool arguments, tool results, attachments, and injected agent
+context stay in the source transcript. Historical revisions are recorded only when an Ingot
+`route_and_load` result supplied one; the current skill hash is not evidence of what ran before.
+
+Usage:
+ python -m ingot.optimize.local_traces
+"""
+from __future__ import annotations
+
+import argparse
+import hashlib
+import json
+import os
+import re
+import tempfile
+import time
+from collections import Counter
+from datetime import date
+from functools import lru_cache
+from pathlib import Path
+from typing import Iterable
+from ingot import paths
+
+SCHEMA = "ingot/local-traces/v1"
+# Cursor reuse is valid only while parser semantics are unchanged. Bump this when accepted record
+# shapes, attribution, or usage/error extraction changes; source mtimes cannot invalidate code.
+PARSER_VERSION = 3
+RUNS_DIR = paths.runs()
+LOCAL_TRACE_FILE = Path(os.environ.get(
+ "LOCAL_TRACE_FILE", RUNS_DIR / "local_traces.json"))
+CODEX_DIR = Path(os.environ.get("CODEX_SESSIONS_DIR", Path.home() / ".codex" / "sessions"))
+CLAUDE_DIR = Path(os.environ.get("CLAUDE_PROJECTS_DIR", Path.home() / ".claude" / "projects"))
+
+_INJECTED_CODEX_PREFIXES = (
+ "# AGENTS.md instructions",
+ "",
+ "",
+)
+_SYNTHETIC_CLAUDE_PREFIXES = (
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+ "",
+)
+_SKILL_PATH = re.compile(r"(?:^|[\s\"'=:(])[^\s\"']*/skills/([^/\s\"']+)/SKILL\.md")
+_SKILL_NAME = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:-]{0,127}$")
+_REVISION = re.compile(r"^[0-9a-f]{8,64}$", re.IGNORECASE)
+_MAX_ROUTE_ENVELOPE = 1_000_000
+_MAX_ROUTE_DEPTH = 6
+_MAX_ROUTE_NODES = 64
+_ROUTE_TOOL_NAMES = {"route_and_load", "mcp__ingot__route_and_load"}
+_ROUTE_MARKERS = {"revision", "skill_body", "related_match"}
+_USAGE_KEYS = (
+ "input_tokens", "output_tokens", "cached_input_tokens",
+ "cache_write_input_tokens", "reasoning_output_tokens",
+)
+
+
+def _records(path: Path) -> Iterable[dict]:
+ try:
+ with path.open(errors="replace") as handle:
+ for line in handle:
+ try:
+ record = json.loads(line)
+ except (TypeError, ValueError):
+ continue
+ if isinstance(record, dict):
+ yield record
+ except OSError:
+ return
+
+
+def _content_text(content, block_types: set[str]) -> str:
+ if isinstance(content, str):
+ return content.strip()
+ if not isinstance(content, list):
+ return ""
+ parts = [block.get("text", "") for block in content
+ if isinstance(block, dict) and block.get("type") in block_types
+ and isinstance(block.get("text"), str)]
+ return "\n".join(part.strip() for part in parts if part.strip()).strip()
+
+
+def _codex_task(payload: dict) -> str:
+ content = payload.get("content")
+ if not isinstance(content, list):
+ return ""
+ parts = []
+ for block in content:
+ if not isinstance(block, dict) or block.get("type") != "input_text":
+ continue
+ text = block.get("text")
+ if not isinstance(text, str) or not text.strip():
+ continue
+ stripped = text.lstrip()
+ if stripped.startswith(_INJECTED_CODEX_PREFIXES + _SYNTHETIC_CLAUDE_PREFIXES):
+ continue
+ parts.append(text.strip())
+ return "\n".join(parts).strip()
+
+
+def _valid_skill(value) -> str | None:
+ value = value.strip() if isinstance(value, str) else ""
+ return value if _SKILL_NAME.fullmatch(value) else None
+
+
+def _skills_from_command(command: str) -> list[str]:
+ return list(dict.fromkeys(
+ name for match in _SKILL_PATH.finditer(command)
+ if (name := _valid_skill(match.group(1)))
+ ))
+
+
+def _json_value(value):
+ if not isinstance(value, str):
+ return value
+ try:
+ return json.loads(value)
+ except ValueError:
+ return value
+
+
+def _route_identity(value, depth: int = 0,
+ budget: list[int] | None = None) -> tuple[str, str | None] | None:
+ """Extract only route identity from nested MCP result shapes; discard the served body."""
+ budget = [_MAX_ROUTE_NODES] if budget is None else budget
+ if depth > _MAX_ROUTE_DEPTH or budget[0] <= 0:
+ return None
+ budget[0] -= 1
+ if isinstance(value, str) and len(value) > _MAX_ROUTE_ENVELOPE:
+ return None
+ value = _json_value(value)
+ if isinstance(value, dict):
+ selected = _valid_skill(value.get("match") or value.get("related_match"))
+ if selected and _ROUTE_MARKERS.intersection(value):
+ revision = value.get("revision")
+ revision = (revision if isinstance(revision, str)
+ and _REVISION.fullmatch(revision) else None)
+ return selected, revision
+ for key in ("result", "structuredContent", "content", "output"):
+ if key in value and (identity := _route_identity(
+ value[key], depth + 1, budget)):
+ return identity
+ elif isinstance(value, list):
+ for item in value:
+ if isinstance(item, dict) and item.get("type") == "text":
+ item = item.get("text")
+ if identity := _route_identity(item, depth + 1, budget):
+ return identity
+ return None
+
+
+def _merge_skill(skills: dict[str, str | None], name: str, revision: str | None = None) -> None:
+ if name not in skills or revision:
+ skills[name] = revision
+
+
+def _tags(skills: dict[str, str | None]) -> list[str]:
+ out = []
+ for name, revision in skills.items():
+ out.append(f"skill:{name}")
+ if revision:
+ out.append(f"revision={name}@{revision}")
+ return out
+
+
+def _add_usage(total: dict[str, int], usage) -> None:
+ if not isinstance(usage, dict):
+ return
+ values = {
+ "input_tokens": usage.get("input_tokens", 0),
+ "output_tokens": usage.get("output_tokens", 0),
+ "cached_input_tokens": (
+ usage.get("cached_input_tokens", 0) + usage.get("cache_read_input_tokens", 0)),
+ "cache_write_input_tokens": (
+ usage.get("cache_write_input_tokens", 0)
+ + usage.get("cache_creation_input_tokens", 0)),
+ "reasoning_output_tokens": usage.get("reasoning_output_tokens", 0),
+ }
+ for key, value in values.items():
+ try:
+ amount = int(value or 0)
+ except (TypeError, ValueError):
+ continue
+ if amount:
+ total[key] = total.get(key, 0) + amount
+
+
+def _tool_failed(value) -> bool:
+ """Count only structured failure signals; free-form output text is not an error contract."""
+ value = _json_value(value)
+ if not isinstance(value, dict):
+ return False
+ exit_code = value.get("exit_code")
+ try:
+ failed_exit = exit_code is not None and int(exit_code) != 0
+ except (TypeError, ValueError):
+ failed_exit = False
+ return failed_exit or value.get("success") is False or value.get("is_error") is True
+
+
+def _trace(*, harness: str, session_id: str, turn_id: str, timestamp: str, cwd: str,
+ task: str, answer: str, skills: dict[str, str | None], usage: dict[str, int],
+ duration_ms: int | None = None, tool_errors: int = 0) -> dict:
+ identity = json.dumps([harness, session_id, turn_id, task, answer],
+ ensure_ascii=False, separators=(",", ":"))
+ result = {
+ "id": hashlib.sha256(identity.encode()).hexdigest(),
+ "timestamp": timestamp,
+ "harness": harness,
+ "session_id": session_id,
+ "turn_id": turn_id,
+ "cwd": cwd,
+ "project": Path(cwd).name if cwd else "",
+ "task": task,
+ "rubric": "",
+ "answer": answer,
+ "skills": [{"name": name, "revision": revision} for name, revision in skills.items()],
+ "tags": _tags(skills),
+ "usage": {key: usage[key] for key in _USAGE_KEYS if usage.get(key)},
+ "tool_errors": tool_errors,
+ }
+ if duration_ms is not None:
+ result["duration_ms"] = duration_ms
+ return result
+
+
+def parse_codex_session(path: Path) -> list[dict]:
+ session_id = path.stem
+ cwd = ""
+ thread_source = ""
+ pending_task = ""
+ pending_timestamp = ""
+ active = None
+ traces = []
+
+ for record in _records(path):
+ kind, payload = record.get("type"), record.get("payload")
+ payload = payload if isinstance(payload, dict) else {}
+ if kind == "session_meta":
+ session_id = str(payload.get("id") or payload.get("session_id") or session_id)
+ cwd = str(payload.get("cwd") or cwd)
+ source = payload.get("thread_source")
+ if not source and isinstance(payload.get("source"), dict):
+ source = "subagent" if payload["source"].get("subagent") else ""
+ thread_source = str(source or "")
+ continue
+ if thread_source and thread_source != "user":
+ continue
+ if (kind == "response_item" and payload.get("type") == "message"
+ and payload.get("role") == "user"):
+ task = _codex_task(payload)
+ if task:
+ pending_task = task
+ pending_timestamp = str(record.get("timestamp") or "")
+ continue
+ if kind == "event_msg" and payload.get("type") == "task_started":
+ active = ({
+ "task": pending_task, "timestamp": pending_timestamp,
+ "turn_id": str(payload.get("turn_id") or ""),
+ "skills": {}, "usage": {}, "route_calls": set(), "tool_errors": 0,
+ } if pending_task else None)
+ pending_task = pending_timestamp = ""
+ continue
+ if active is None:
+ continue
+ if kind == "response_item" and payload.get("type") in {
+ "function_call", "custom_tool_call"}:
+ name = str(payload.get("name") or "")
+ arguments = _json_value(payload.get("arguments"))
+ if isinstance(arguments, dict):
+ if isinstance(arguments.get("cmd"), str):
+ for skill in _skills_from_command(arguments["cmd"]):
+ _merge_skill(active["skills"], skill)
+ if name in _ROUTE_TOOL_NAMES:
+ call_id = str(payload.get("call_id") or "")
+ if call_id:
+ active["route_calls"].add(call_id)
+ continue
+ if kind == "response_item" and payload.get("type") in {
+ "function_call_output", "custom_tool_call_output"}:
+ call_id = str(payload.get("call_id") or "")
+ output = payload.get("output")
+ if (call_id and call_id in active["route_calls"]
+ and (identity := _route_identity(output))):
+ _merge_skill(active["skills"], *identity)
+ active["tool_errors"] += int(_tool_failed(output))
+ continue
+ if kind == "event_msg" and payload.get("type") == "token_count":
+ info = payload.get("info")
+ if isinstance(info, dict):
+ _add_usage(active["usage"], info.get("last_token_usage"))
+ continue
+ if kind == "event_msg" and payload.get("type") == "turn_aborted":
+ active = None
+ continue
+ if kind == "event_msg" and payload.get("type") == "task_complete":
+ answer = payload.get("last_agent_message")
+ if isinstance(answer, str) and answer.strip():
+ duration = payload.get("duration_ms")
+ try:
+ duration = int(duration) if duration is not None else None
+ except (TypeError, ValueError):
+ duration = None
+ traces.append(_trace(
+ harness="codex", session_id=session_id, turn_id=active["turn_id"],
+ timestamp=active["timestamp"], cwd=cwd, task=active["task"],
+ answer=answer.strip(), skills=active["skills"], usage=active["usage"],
+ duration_ms=duration, tool_errors=active["tool_errors"],
+ ))
+ active = None
+ return traces
+
+
+def _claude_human_task(record: dict) -> str:
+ if record.get("type") != "user" or record.get("isSidechain") is True:
+ return ""
+ # Claude serializes skill-hook output, compaction summaries, local command echoes, and some
+ # system injections as role=user. Provenance fields, not message wording, distinguish those
+ # records from typed/queued/SDK requests; accepting every user role turns tool output into
+ # fabricated tasks and severs the skill call from its real turn.
+ if (record.get("isMeta") or record.get("isCompactSummary")
+ or record.get("isVisibleInTranscriptOnly") or record.get("sourceToolUseID")):
+ return ""
+ source = record.get("promptSource")
+ if source == "system":
+ return ""
+ if not record.get("origin") and source not in {
+ "typed", "queued", "suggestion_accepted", "sdk"}:
+ return ""
+ message = record.get("message")
+ if not isinstance(message, dict):
+ return ""
+ content = message.get("content")
+ if isinstance(content, list) and any(
+ isinstance(block, dict) and block.get("type") == "tool_result" for block in content):
+ return ""
+ task = _content_text(content, {"text"})
+ return "" if task.lstrip().startswith(_SYNTHETIC_CLAUDE_PREFIXES) else task
+
+
+def parse_claude_session(path: Path) -> list[dict]:
+ traces = []
+ active = None
+ sequence = 0
+
+ for record in _records(path):
+ task = _claude_human_task(record)
+ if task:
+ active = {
+ "task": task,
+ "timestamp": str(record.get("timestamp") or ""),
+ "session_id": str(record.get("sessionId") or path.stem),
+ "turn_id": str(record.get("uuid") or f"turn-{sequence}"),
+ "cwd": str(record.get("cwd") or ""),
+ "skills": {}, "usage": {}, "route_calls": set(), "tool_errors": 0,
+ }
+ sequence += 1
+ continue
+ if active is None or record.get("isSidechain") is True:
+ continue
+ message = record.get("message")
+ if not isinstance(message, dict):
+ continue
+ content = message.get("content")
+ if record.get("type") == "assistant":
+ _add_usage(active["usage"], message.get("usage"))
+ for block in content if isinstance(content, list) else ():
+ if not isinstance(block, dict) or block.get("type") != "tool_use":
+ continue
+ name, inputs = str(block.get("name") or ""), block.get("input")
+ inputs = inputs if isinstance(inputs, dict) else {}
+ if name == "Skill" and (skill := _valid_skill(
+ inputs.get("skill") or inputs.get("name"))):
+ _merge_skill(active["skills"], skill)
+ if name in _ROUTE_TOOL_NAMES:
+ call_id = str(block.get("id") or "")
+ if call_id:
+ active["route_calls"].add(call_id)
+ if message.get("stop_reason") == "end_turn":
+ answer = _content_text(content, {"text"})
+ if answer:
+ traces.append(_trace(
+ harness="claude", session_id=active["session_id"],
+ turn_id=active["turn_id"], timestamp=active["timestamp"],
+ cwd=active["cwd"], task=active["task"], answer=answer,
+ skills=active["skills"], usage=active["usage"],
+ tool_errors=active["tool_errors"],
+ ))
+ active = None
+ elif record.get("type") == "user" and isinstance(content, list):
+ for block in content:
+ if not isinstance(block, dict) or block.get("type") != "tool_result":
+ continue
+ if block.get("is_error"):
+ active["tool_errors"] += 1
+ call_id = str(block.get("tool_use_id") or "")
+ if call_id and call_id in active["route_calls"]:
+ if identity := _route_identity(block.get("content")):
+ _merge_skill(active["skills"], *identity)
+ return traces
+
+
+def _jsonl_files(root: Path) -> list[Path]:
+ if not root.exists():
+ return []
+ return sorted(path for path in root.rglob("*.jsonl") if path.is_file())
+
+
+def _date_value(value: str) -> str:
+ value = value.strip()
+ if not value:
+ return ""
+ try:
+ return date.fromisoformat(value).isoformat()
+ except ValueError as error:
+ raise ValueError(f"expected an ISO date (YYYY-MM-DD), got {value!r}") from error
+
+
+def _filters(projects: Iterable[str] = (), since: str = "", until: str = "") -> dict:
+ result = {
+ "projects": sorted({str(project).strip() for project in projects if str(project).strip()}),
+ "since": _date_value(since),
+ "until": _date_value(until),
+ }
+ if result["since"] and result["until"] and result["since"] > result["until"]:
+ raise ValueError("--since must be on or before --until")
+ return result
+
+
+def _selected(trace: dict, filters: dict) -> bool:
+ projects = filters["projects"]
+ if projects and trace.get("project") not in projects:
+ return False
+ day = str(trace.get("timestamp") or "")[:10]
+ if (filters["since"] or filters["until"]) and not re.fullmatch(r"\d{4}-\d{2}-\d{2}", day):
+ return False
+ if filters["since"] and day < filters["since"]:
+ return False
+ if filters["until"] and day > filters["until"]:
+ return False
+ return True
+
+
+def _source_key(path: Path) -> str:
+ return hashlib.sha256(str(path.resolve()).encode()).hexdigest()
+
+
+def _prior_store(output: Path, filters: dict) -> dict:
+ try:
+ payload = json.loads(output.read_text())
+ except (OSError, ValueError):
+ return {}
+ if (payload.get("schema_version") != SCHEMA
+ or payload.get("parser_version") != PARSER_VERSION
+ or payload.get("filters", _filters()) != filters
+ or not isinstance(payload.get("traces"), list)
+ or not isinstance(payload.get("source_index"), dict)):
+ return {}
+ for trace in payload["traces"]:
+ if (not isinstance(trace, dict) or not isinstance(trace.get("id"), str)
+ or not isinstance(trace.get("_sources"), list)
+ or not trace["_sources"]
+ or not all(isinstance(source, str) for source in trace["_sources"])):
+ return {}
+ return payload
+
+
+def scan(*, codex_dir: Path = CODEX_DIR, claude_dir: Path = CLAUDE_DIR,
+ output: Path = LOCAL_TRACE_FILE, projects: Iterable[str] = (),
+ since: str = "", until: str = "", force: bool = False) -> dict:
+ selected_filters = _filters(projects, since, until)
+ files = {"codex": _jsonl_files(codex_dir), "claude": _jsonl_files(claude_dir)}
+ prior = {} if force else _prior_store(output, selected_filters)
+ prior_index = prior.get("source_index", {})
+ by_source: dict[str, list[dict]] = {}
+ for trace in prior.get("traces", []):
+ if not isinstance(trace, dict):
+ continue
+ for source in trace.get("_sources", []):
+ if isinstance(source, str):
+ by_source.setdefault(source, []).append(trace)
+
+ source_index = {"codex": {}, "claude": {}}
+ collected = []
+ reused = parsed = 0
+ parsers = {"codex": parse_codex_session, "claude": parse_claude_session}
+ for harness, paths in files.items():
+ old = prior_index.get(harness, {}) if isinstance(prior_index, dict) else {}
+ old = old if isinstance(old, dict) else {}
+ for path in paths:
+ try:
+ stat = path.stat()
+ except OSError:
+ continue
+ key = _source_key(path)
+ # This cheap cursor fits append-only agent logs. --force is the escape hatch for a
+ # same-size rewrite whose mtime was deliberately preserved.
+ fingerprint = {"size": stat.st_size, "mtime_ns": stat.st_mtime_ns}
+ source_index[harness][key] = fingerprint
+ if old.get(key) == fingerprint:
+ file_traces = by_source.get(key, [])
+ fresh = False
+ reused += 1
+ else:
+ file_traces = parsers[harness](path)
+ fresh = True
+ parsed += 1
+ for trace in file_traces:
+ if (not isinstance(trace, dict) or not isinstance(trace.get("id"), str)
+ or not _selected(trace, selected_filters)):
+ continue
+ copied = dict(trace)
+ # Rebuild provenance from files that still exist. Carrying the old list forward
+ # would keep a removed duplicate as a live source forever.
+ copied["_sources"] = [key]
+ copied["_fresh"] = fresh
+ collected.append(copied)
+
+ deduped = {}
+ for trace in collected:
+ existing = deduped.get(trace["id"])
+ if existing is None:
+ deduped[trace["id"]] = trace
+ else:
+ sources = sorted(
+ set(existing.get("_sources", ())) | set(trace.get("_sources", ())))
+ existing_source = min(existing.get("_sources", ("",)))
+ candidate_source = min(trace.get("_sources", ("",)))
+ if ((trace["_fresh"] and not existing["_fresh"])
+ or (trace["_fresh"] == existing["_fresh"]
+ and candidate_source < existing_source)):
+ deduped[trace["id"]] = trace
+ existing = trace
+ existing["_sources"] = sources
+ for trace in deduped.values():
+ trace.pop("_fresh", None)
+ ordered = sorted(deduped.values(), key=lambda trace: (
+ trace.get("timestamp", ""), trace["id"]))
+ payload = {
+ "schema_version": SCHEMA,
+ "parser_version": PARSER_VERSION,
+ "generated_at": int(time.time()),
+ "filters": selected_filters,
+ "sources": {name: {"files": len(paths)} for name, paths in files.items()},
+ "source_index": source_index,
+ "traces": ordered,
+ }
+ output.parent.mkdir(parents=True, exist_ok=True)
+ staged_path = None
+ try:
+ with tempfile.NamedTemporaryFile(
+ mode="w", dir=output.parent, prefix=f".{output.name}.", delete=False) as staged:
+ staged_path = Path(staged.name)
+ json.dump(payload, staged, ensure_ascii=False, separators=(",", ":"))
+ staged.write("\n")
+ staged.flush()
+ os.fsync(staged.fileno())
+ os.replace(staged_path, output)
+ staged_path = None
+ directory = os.open(output.parent, os.O_RDONLY)
+ try:
+ os.fsync(directory)
+ finally:
+ os.close(directory)
+ finally:
+ if staged_path is not None:
+ staged_path.unlink(missing_ok=True)
+ return {"output": str(output), "traces": len(ordered),
+ "files": sum(len(paths) for paths in files.values()),
+ "parsed_files": parsed, "reused_files": reused}
+
+
+def _empty_summary(status: str = "missing") -> dict:
+ return {
+ "configured": False, "status": status, "generated_at": None,
+ "total": 0, "harnesses": {},
+ "available_total": 0, "available_projects": {}, "available_harnesses": {},
+ "available_skills": {},
+ "filters": {"projects": [], "harness": "", "skill": "", "since": "", "until": "",
+ "include_tasks": False},
+ "skill_uses": {}, "revision_pinned": 0, "unattributed": 0,
+ "usage": {}, "tool_errors": 0, "recent": [],
+ }
+
+
+@lru_cache(maxsize=1)
+def _read_store(path: str, mtime_ns: int, size: int) -> dict:
+ del mtime_ns, size # cache-key material; the file path is the only read target
+ return json.loads(Path(path).read_text())
+
+
+def _observed_skills(trace: dict) -> set[str]:
+ """Validated skill names attributed to one turn."""
+ entries = trace.get("skills") if isinstance(trace.get("skills"), list) else []
+ return {name for skill in entries if isinstance(skill, dict)
+ and (name := _valid_skill(skill.get("name")))}
+
+
+def store_summary(path: Path = LOCAL_TRACE_FILE, recent: int = 20, *, project: str = "",
+ harness: str = "", skill: str = "", since: str = "", until: str = "",
+ include_tasks: bool = False) -> dict:
+ try:
+ stat = path.stat()
+ except OSError:
+ return _empty_summary()
+ try:
+ payload = _read_store(str(path), stat.st_mtime_ns, stat.st_size)
+ except (OSError, ValueError):
+ return _empty_summary("unreadable")
+ if payload.get("schema_version") != SCHEMA or not isinstance(payload.get("traces"), list):
+ return _empty_summary("unreadable")
+ all_traces = [trace for trace in payload["traces"] if isinstance(trace, dict)]
+ available_projects = Counter(
+ str(trace.get("project")) for trace in all_traces if trace.get("project"))
+ available_harnesses = Counter(
+ str(trace.get("harness")) for trace in all_traces if trace.get("harness"))
+ # Counted over every turn, not the filtered set: the picker has to keep offering a skill after
+ # you select it, and offering only skills that survive the current filter would empty itself.
+ available_skills = Counter(
+ name for trace in all_traces for name in _observed_skills(trace))
+ selected_filters = _filters([project] if project else (), since, until)
+ traces = [trace for trace in all_traces
+ if _selected(trace, selected_filters)
+ and (not harness or trace.get("harness") == harness)
+ and (not skill or skill in _observed_skills(trace))]
+ harnesses = Counter()
+ skill_uses = Counter()
+ usage = Counter()
+ pinned = tool_errors = unattributed = 0
+ for trace in traces:
+ if not isinstance(trace, dict):
+ continue
+ harnesses[str(trace.get("harness") or "unknown")] += 1
+ skills = trace.get("skills") if isinstance(trace.get("skills"), list) else []
+ if not skills:
+ unattributed += 1
+ for observed in skills: # not `skill`: that name is the filter parameter
+ if not isinstance(observed, dict) or not _valid_skill(observed.get("name")):
+ continue
+ skill_uses[observed["name"]] += 1
+ pinned += int(bool(observed.get("revision")))
+ for key in _USAGE_KEYS:
+ try:
+ usage[key] += int((trace.get("usage") or {}).get(key, 0))
+ except (AttributeError, TypeError, ValueError):
+ pass
+ try:
+ tool_errors += int(trace.get("tool_errors") or 0)
+ except (TypeError, ValueError):
+ pass
+ newest = sorted(
+ (trace for trace in traces if isinstance(trace, dict)),
+ key=lambda trace: (trace.get("timestamp", ""), trace.get("id", "")), reverse=True,
+ )[:recent]
+ previews = []
+ for trace in newest:
+ preview_skills = []
+ for observed in trace.get("skills", []) if isinstance(trace.get("skills"), list) else ():
+ if not isinstance(observed, dict) or not (name := _valid_skill(observed.get("name"))):
+ continue
+ revision = observed.get("revision")
+ revision = (revision if isinstance(revision, str)
+ and _REVISION.fullmatch(revision) else None)
+ preview_skills.append({"name": name, "revision": revision})
+ preview_usage = {}
+ stored_usage = trace.get("usage") if isinstance(trace.get("usage"), dict) else {}
+ for key in _USAGE_KEYS:
+ try:
+ amount = int(stored_usage.get(key, 0))
+ except (TypeError, ValueError):
+ continue
+ if amount > 0:
+ preview_usage[key] = amount
+ try:
+ preview_errors = max(0, int(trace.get("tool_errors") or 0))
+ except (TypeError, ValueError):
+ preview_errors = 0
+ preview = {
+ "id": str(trace.get("id") or ""),
+ "timestamp": str(trace.get("timestamp") or ""),
+ "harness": str(trace.get("harness") or ""),
+ "project": str(trace.get("project") or ""),
+ "skills": preview_skills,
+ "usage": preview_usage,
+ "tool_errors": preview_errors,
+ }
+ if include_tasks:
+ task = str(trace.get("task") or "")
+ preview["task"] = task[:280] + ("…" if len(task) > 280 else "")
+ previews.append(preview)
+ return {
+ "configured": True, "status": "ready", "generated_at": payload.get("generated_at"),
+ "available_total": len(all_traces),
+ "available_projects": dict(available_projects.most_common()),
+ "available_harnesses": dict(available_harnesses.most_common()),
+ "available_skills": dict(available_skills.most_common()),
+ "filters": {**selected_filters, "harness": harness, "skill": skill,
+ "include_tasks": include_tasks},
+ "total": len(traces), "harnesses": dict(harnesses),
+ "skill_uses": dict(skill_uses.most_common()), "revision_pinned": pinned,
+ "unattributed": unattributed, "usage": dict(usage),
+ "tool_errors": tool_errors, "recent": previews,
+ }
+
+
+def main() -> None:
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--codex-dir", type=Path, default=CODEX_DIR)
+ parser.add_argument("--claude-dir", type=Path, default=CLAUDE_DIR)
+ parser.add_argument("--output", type=Path, default=LOCAL_TRACE_FILE)
+ parser.add_argument("--project", action="append", default=[],
+ help="keep only this cwd basename; repeat for more than one project")
+ parser.add_argument("--since", default="", help="keep turns on or after YYYY-MM-DD")
+ parser.add_argument("--until", default="", help="keep turns on or before YYYY-MM-DD")
+ parser.add_argument("--force", action="store_true",
+ help="reparse every transcript instead of reusing the cursor")
+ args = parser.parse_args()
+ try:
+ result = scan(codex_dir=args.codex_dir, claude_dir=args.claude_dir, output=args.output,
+ projects=args.project, since=args.since, until=args.until, force=args.force)
+ except ValueError as error:
+ parser.error(str(error))
+ print(f"[traces] normalized {result['traces']} completed turn(s) from "
+ f"{result['files']} transcript file(s) "
+ f"({result['parsed_files']} parsed, {result['reused_files']} unchanged) "
+ f"→ {result['output']}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/optimize/loop.py b/ingot/optimize/loop.py
similarity index 98%
rename from optimize/loop.py
rename to ingot/optimize/loop.py
index 3ab8740..d688061 100644
--- a/optimize/loop.py
+++ b/ingot/optimize/loop.py
@@ -4,7 +4,7 @@
unattended, off the review path. Every surviving candidate lands quarantined in the review queue and
still requires human approval.
-Usage: python -m optimize.loop [skill ...] (default: every skill with an eval task set)
+Usage: python -m ingot.optimize.loop [skill ...] (default: every skill with an eval task set)
"""
import argparse
import os
diff --git a/optimize/mine.py b/ingot/optimize/mine.py
similarity index 83%
rename from optimize/mine.py
rename to ingot/optimize/mine.py
index e9436c5..eed4e40 100644
--- a/optimize/mine.py
+++ b/ingot/optimize/mine.py
@@ -9,7 +9,7 @@
the signal that drives targeted optimization (the paper: Liu et al., "SkillForge: Forging
Domain-Specific, Self-Evolving Agent Skills", arXiv:2604.08618).
-Usage: python -m optimize.mine [--limit N]
+Usage: python -m ingot.optimize.mine [--limit N]
"""
import argparse
import base64
@@ -23,11 +23,13 @@
from urllib.parse import urlencode, urlparse
from .judge import DIMENSIONS, MODELS, _PROMPT, failed_dimensions, judge
+from .local_traces import LOCAL_TRACE_FILE, SCHEMA as LOCAL_TRACE_SCHEMA
+from ingot import paths
LF_URL = os.environ.get("LANGFUSE_BASE_URL", "http://langfuse-web:3000")
LF_PK = os.environ.get("LANGFUSE_PUBLIC_KEY", "pk-lf-local-demo")
LF_SK = os.environ.get("LANGFUSE_SECRET_KEY", "sk-lf-local-demo")
-RUNS_DIR = Path(__file__).resolve().parent.parent / "runs"
+RUNS_DIR = paths.runs()
JUDGE_CACHE_FILE = RUNS_DIR / "mine-cache" / "judgments.json"
TRACE_PAGE_SIZE = 100
CLUSTER_THRESHOLD = float(os.environ.get("MINE_CLUSTER_THRESHOLD", "0.90"))
@@ -97,6 +99,38 @@ def fetch_traces(limit: int = 0) -> list[dict]:
return out
+def fetch_local_traces(limit: int = 0) -> list[dict]:
+ """Read normalized local coding-agent turns. Transcript parsing is a separate, local command
+ so the paid miner never gains permission to crawl a home directory as a side effect."""
+ if not LOCAL_TRACE_FILE.exists():
+ raise SystemExit(
+ f"No local trace store at {LOCAL_TRACE_FILE}; run `python -m ingot.optimize.local_traces` "
+ "on the machine that owns the Codex and Claude transcripts.")
+ try:
+ payload = json.loads(LOCAL_TRACE_FILE.read_text())
+ except (OSError, ValueError) as e:
+ raise SystemExit(f"Local trace store is unreadable ({e}); scan it again.") from e
+ if (payload.get("schema_version") != LOCAL_TRACE_SCHEMA
+ or not isinstance(payload.get("traces"), list)):
+ raise SystemExit(
+ f"unsupported local trace store; expected {LOCAL_TRACE_SCHEMA}, scan it again")
+ ordered = sorted(
+ (trace for trace in payload["traces"] if isinstance(trace, dict)),
+ key=lambda trace: (trace.get("timestamp", ""), trace.get("id", "")), reverse=True,
+ )
+ out = []
+ for trace in ordered:
+ parsed = _task_answer(trace.get("task"), trace.get("answer"))
+ if not parsed:
+ continue
+ task, rubric, answer = parsed
+ out.append({"task": task, "rubric": rubric, "answer": answer,
+ "tags": trace.get("tags", [])})
+ if limit and len(out) >= limit:
+ break
+ return out
+
+
def _task_answer(inp, ans):
"""(task, rubric, answer) from supported trace roots: explicit eval dictionaries, LangGraph
state, and the root shapes emitted by the Claude Code and Codex Langfuse connectors."""
@@ -121,16 +155,34 @@ def _task_answer(inp, ans):
return None
+def _tagged_with(tags, skill: str) -> bool:
+ """Does any tag name `skill`? Harnesses spell it three ways, and only the first is bare:
+ our own agent writes `pdf` plus a revision pin `revision=pdf@`, while Claude Code
+ writes `skill:`, namespaced by plugin when the skill came from one
+ (`skill:superpowers:systematic-debugging`). Matching the bare form alone drops every trace an
+ external harness produced — which is nearly all real traffic — and mining then reports
+ 'no traces relevant to X' as though the skill were never used."""
+ for tag in tags or []:
+ tag = str(tag)
+ if tag == skill:
+ return True
+ if tag.startswith("revision=") and tag[len("revision="):].split("@", 1)[0] == skill:
+ return True
+ if tag.startswith("skill:") and tag[len("skill:"):].rsplit(":", 1)[-1] == skill:
+ return True
+ return False
+
+
def relevant_traces(traces: list[dict], skill: str, k: int = 5) -> list[dict]:
"""Traces attributable to `skill`: tagged with it (external harnesses tag the
routed skill), or ranking it in the embedding top-k for the task text. The rank check is what
catches traffic the skill *should* have served but didn't route, under-triggering, the common
routing failure, which a tag filter alone would attribute to the wrong skill."""
- from mcp_server.registry import load_skills
- from mcp_server.router import Router
+ from ingot.mcp_server.registry import load_skills
+ from ingot.mcp_server.router import Router
router = Router(load_skills())
return [t for t in traces
- if skill in t.get("tags", [])
+ if _tagged_with(t.get("tags"), skill)
or any(s["name"] == skill for s in router.suggest(t["task"], k=k, min_score=0.0))]
@@ -182,7 +234,7 @@ def _normalized_embedder():
"""Return an embedding function that produces row-normalized float vectors. Task-to-task
similarity (diversity + train-dup checks), so both sides use the document embedding."""
import numpy as np
- from mcp_server.embedding import build_embedding
+ from ingot.mcp_server.embedding import build_embedding
embedder = build_embedding()
@@ -360,10 +412,23 @@ def _select_candidates(traces: list[dict], scores: list[float], skill: str, log=
return _rank_candidates(traces, scores, indices, context)
-def mine(skill: str, limit: int = 0, log=print) -> dict:
+def _source_traces(source: str, limit: int) -> list[dict]:
+ if source == "langfuse":
+ return fetch_traces(limit)
+ if source == "local":
+ return fetch_local_traces(limit)
+ raise SystemExit(f"unknown trace source {source!r}; choose langfuse or local")
+
+
+def mine(skill: str, limit: int = 0, log=print, source: str = "langfuse",
+ allow_external_judge: bool = False) -> dict:
+ if source == "local" and not allow_external_judge:
+ raise SystemExit(
+ "Local transcript content stays local by default. Re-run with "
+ "`--allow-external-judge` to send selected task/answer pairs to JUDGE_MODEL.")
scope = f"the newest {limit}" if limit else "all"
- log(f"[mine] pulling {scope} traces from Langfuse for '{skill}'…")
- traces = fetch_traces(limit)
+ log(f"[mine] pulling {scope} {source} traces for '{skill}'…")
+ traces = _source_traces(source, limit)
if not traces:
raise SystemExit("No usable traces found; run the agent or a candidate pass first to "
"generate some.")
@@ -438,7 +503,14 @@ def mine(skill: str, limit: int = 0, log=print) -> dict:
ap.add_argument("skill")
ap.add_argument("--limit", type=int, default=0,
help="newest traces to inspect; 0 (default) paginates through all uses")
+ ap.add_argument("--source", choices=("langfuse", "local"), default="langfuse",
+ help="trace store to mine; local requires an explicit transcript scan")
+ ap.add_argument("--allow-external-judge", action="store_true",
+ help="allow local task/answer pairs to be sent to JUDGE_MODEL")
args = ap.parse_args()
+ if args.source == "local" and not args.allow_external_judge:
+ ap.error("--source local requires --allow-external-judge")
from . import require_openrouter_key
require_openrouter_key()
- mine(args.skill, limit=args.limit)
+ mine(args.skill, limit=args.limit, source=args.source,
+ allow_external_judge=args.allow_external_judge)
diff --git a/optimize/promote.py b/ingot/optimize/promote.py
similarity index 57%
rename from optimize/promote.py
rename to ingot/optimize/promote.py
index 7b9b930..a81a0c8 100644
--- a/optimize/promote.py
+++ b/ingot/optimize/promote.py
@@ -1,20 +1,35 @@
"""Gate-enforced, revisioned, atomic promotion and rollback for quarantined skill changes."""
from __future__ import annotations
+import ctypes
+import errno
import json
import logging
import os
import shutil
+import sys
import time
import uuid
from pathlib import Path
-from mcp_server.registry import (SLUG_RE, load_skills, read_components, skill_revision,
- write_components)
+from ingot.mcp_server.registry import (SLUG_RE, load_skills, read_components, skill_revision,
+ writable_skill_dir, write_components)
+from ingot import paths
-RUNS_DIR = Path(__file__).resolve().parent.parent / "runs"
-PENDING_DIR = RUNS_DIR / "pending"
-REVISIONS_DIR = RUNS_DIR / "revisions"
+
+
+def pending_dir() -> Path:
+ """The one review slot per skill. Resolved per call: it is configuration, and a value frozen at
+ import cannot follow a process that is told where its state lives."""
+ return paths.runs() / "pending"
+
+
+def revisions_dir() -> Path:
+ """Stored snapshots, the rollback targets."""
+ return paths.runs() / "revisions"
+
+ABSENT_REVISION = "absent"
+ABSENT_MARKER = ".ingot-absent"
logger = logging.getLogger(__name__)
@@ -25,11 +40,11 @@ def check_slug(skill: str) -> str:
def pending_path(skill: str) -> Path:
- return PENDING_DIR / f"{check_slug(skill)}.json"
+ return pending_dir() / f"{check_slug(skill)}.json"
def save_pending(skill: str, data: dict) -> Path:
- PENDING_DIR.mkdir(parents=True, exist_ok=True)
+ pending_dir().mkdir(parents=True, exist_ok=True)
path = pending_path(skill)
_archive_displaced(skill, path, data)
temporary = path.with_suffix(f".{uuid.uuid4().hex}.tmp")
@@ -47,7 +62,7 @@ def _archive_displaced(skill: str, path: Path, data: dict) -> None:
existing = json.loads(path.read_text())
if sorted(existing.get("changed_components", [])) == sorted(data.get("changed_components", [])):
return
- archived = PENDING_DIR / f"{skill}.displaced-{existing.get('created', uuid.uuid4().hex)}.json"
+ archived = pending_dir() / f"{skill}.displaced-{existing.get('created', uuid.uuid4().hex)}.json"
shutil.copy(path, archived)
print(f"[pending] one review slot per skill: the pending "
f"{existing.get('changed_components')} challenger was displaced by this "
@@ -60,17 +75,42 @@ def load_pending(skill: str) -> dict | None:
return json.loads(path.read_text()) if path.exists() else None
+def unreadable_pending() -> list[str]:
+ """Pending files this process can see but cannot use.
+
+ `list_pending` skips them so one corrupt record cannot break review — which also means a
+ proposal this process cannot read is indistinguishable from no proposal at all, and the console
+ reports CLEAR over it. Observed live: the MCP container writes records as root 0600 while the UI
+ runs as uid 1000, so an approved-and-waiting skill was invisible for hours."""
+ if not pending_dir().exists():
+ return []
+ blocked = []
+ for path in sorted(pending_dir().glob("*.json")):
+ if not SLUG_RE.fullmatch(path.stem):
+ continue
+ try:
+ record = json.loads(path.read_text())
+ except (OSError, ValueError): # ValueError covers JSONDecodeError and UnicodeDecodeError
+ blocked.append(path.name)
+ continue
+ if not (isinstance(record, dict) and record.get("skill") == path.stem):
+ blocked.append(path.name)
+ return blocked
+
+
def list_pending() -> list[dict]:
"""Return valid pending records without letting a malformed queue file break the review UI."""
- if not PENDING_DIR.exists():
+ if not pending_dir().exists():
return []
records = []
- for path in sorted(PENDING_DIR.glob("*.json")):
+ for path in sorted(pending_dir().glob("*.json")):
if not SLUG_RE.fullmatch(path.stem):
continue
try:
record = json.loads(path.read_text())
- except (OSError, json.JSONDecodeError):
+ except (OSError, ValueError):
+ # ValueError, not json.JSONDecodeError: a non-UTF-8 file raises UnicodeDecodeError,
+ # which escaped this handler and took the whole review page down with it.
continue
if isinstance(record, dict) and record.get("skill") == path.stem:
records.append(record)
@@ -81,7 +121,7 @@ def snapshot_index_path(skill: str) -> Path:
"""When each snapshot was last taken. It lives beside the snapshot directories, never inside
one, so a rollback copies back the skill and nothing else. The leading dot also keeps it out of
the slug-matched snapshot listing."""
- return REVISIONS_DIR / check_slug(skill) / ".snapshots.json"
+ return revisions_dir() / check_slug(skill) / ".snapshots.json"
def _read_snapshot_index(skill: str) -> dict:
@@ -145,7 +185,7 @@ def list_revisions(skill: str) -> list[dict]:
entry is `{"revision": , "created": }`. Snapshots taken before the index
existed, and snapshots whose index entry is unusable, fall back to directory mtime and sort
below stamped ones."""
- root = REVISIONS_DIR / check_slug(skill)
+ root = revisions_dir() / check_slug(skill)
if not root.is_dir():
return []
index = _read_snapshot_index(skill)
@@ -171,9 +211,9 @@ def load_snapshot_components(skill: str, revision: str) -> dict[str, str]:
def list_snapshotted_skills() -> list[str]:
"""Skills with at least one snapshot. Reading the snapshot store directly keeps the history
view off the skill-library hash scan that the skills listing already pays for."""
- if not REVISIONS_DIR.is_dir():
+ if not revisions_dir().is_dir():
return []
- return sorted(path.name for path in REVISIONS_DIR.iterdir()
+ return sorted(path.name for path in revisions_dir().iterdir()
if path.is_dir() and SLUG_RE.fullmatch(path.name))
@@ -213,6 +253,25 @@ def stale_evidence_reason(skill: str, pending: dict) -> str | None:
evidence = pending.get("evidence")
if not isinstance(evidence, dict) or not evidence:
return "evidence is required for promotion"
+ if pending.get("kind") == "creation":
+ if any(item.name == skill for item in load_skills()):
+ return "a skill with this name appeared after the proposal; review it as an update"
+ expected = evidence.get("challenger", {}).get("revision")
+ components = pending.get("challenger_components", {})
+ manifest = pending.get("tree")
+ # An ingested package's revision is what materializing its staged tree produces, not what
+ # its decoded text components produce -- the tree carries files no component describes.
+ # Recomputing from the components alone would report every package with a second file as
+ # permanently stale: quarantined, reviewable, and impossible to approve.
+ if manifest is not None:
+ from ingot.optimize import tree as candidate_tree
+ try:
+ actual = candidate_tree.revision(skill, manifest, components)
+ except (ValueError, OSError) as exc:
+ return f"the staged candidate tree is no longer usable: {exc}"
+ else:
+ actual = skill_revision(writable_skill_dir(skill), components)
+ return None if expected == actual else "challenger revision does not match the recorded evidence"
try:
current = _current_skill(skill)
except ValueError as exc:
@@ -224,7 +283,7 @@ def _snapshot(skill_dir: Path, skill: str, revision: str) -> Path:
"""Preserve a revision as a rollback target. Re-snapshotting a revision that is already stored
is a no-op on disk but still restamps it: that is what a rollback followed by a promotion does,
and the restored revision is then the most recent thing a promotion displaced."""
- destination = REVISIONS_DIR / skill / revision
+ destination = revisions_dir() / skill / revision
if not destination.exists():
destination.parent.mkdir(parents=True, exist_ok=True)
temporary = destination.with_name(f".{revision}.{uuid.uuid4().hex}.tmp")
@@ -240,7 +299,7 @@ def _snapshot(skill_dir: Path, skill: str, revision: str) -> Path:
def audit_path() -> Path:
"""The approval trail lives beside the review queue, so a relocated queue moves both."""
- return PENDING_DIR.parent / "approval-audit.jsonl"
+ return pending_dir().parent / "approval-audit.jsonl"
def read_audit(limit: int = 50) -> dict:
@@ -344,6 +403,38 @@ def _sweep_staging(skill_dir: Path) -> None:
logger.warning("Could not remove the stale staging directory %s", stale, exc_info=True)
+def _rename_no_replace(source: Path, target: Path) -> None:
+ """Atomically publish a new skill directory without replacing a target that won a race.
+
+ Python's POSIX rename replaces an empty directory, so an existence check followed by rename is
+ not a guard. Linux and macOS expose the needed no-replace operation under different names; fail
+ closed on other POSIX platforms rather than quietly restore the race.
+ """
+ if sys.platform.startswith("linux"):
+ libc = ctypes.CDLL(None, use_errno=True)
+ rename = getattr(libc, "renameat2", None)
+ if rename is None:
+ raise OSError(errno.ENOTSUP, "atomic no-replace rename is unavailable")
+ rename.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p,
+ ctypes.c_uint]
+ rename.restype = ctypes.c_int
+ result = rename(-100, os.fsencode(source), -100, os.fsencode(target), 1)
+ elif sys.platform == "darwin":
+ libc = ctypes.CDLL(None, use_errno=True)
+ rename = libc.renamex_np
+ rename.argtypes = [ctypes.c_char_p, ctypes.c_char_p, ctypes.c_uint]
+ rename.restype = ctypes.c_int
+ result = rename(os.fsencode(source), os.fsencode(target), 0x00000004)
+ elif os.name == "nt":
+ source.rename(target) # Windows rename already refuses an existing destination.
+ return
+ else:
+ raise OSError(errno.ENOTSUP, "atomic no-replace rename is unavailable")
+ if result:
+ error = ctypes.get_errno()
+ raise OSError(error, os.strerror(error), target)
+
+
def _activate_rewrite(skill: str, pending: dict) -> str:
components = pending["challenger_components"]
evidence = pending.get("evidence")
@@ -352,22 +443,35 @@ def _activate_rewrite(skill: str, pending: dict) -> str:
current = _current_skill(skill)
_validate_evidence(current, components, evidence)
- skill_dir = Path(current.root)
- _snapshot(skill_dir, skill, current.revision)
+ source_dir = Path(current.root)
+ skill_dir = writable_skill_dir(skill)
+ if source_dir != skill_dir and (skill_dir.exists() or skill_dir.is_symlink()):
+ raise ValueError(
+ f"writable activation target already exists but is not the serving skill: {skill_dir}")
+ _snapshot(source_dir, skill, current.revision)
_sweep_staging(skill_dir)
stage = skill_dir.with_name(f".{skill_dir.name}.{uuid.uuid4().hex}.stage")
previous = skill_dir.with_name(f".{skill_dir.name}.{uuid.uuid4().hex}.previous")
try:
- shutil.copytree(skill_dir, stage, symlinks=True)
+ shutil.copytree(source_dir, stage, symlinks=True)
write_components(stage, components)
- skill_dir.rename(previous)
- try:
- stage.rename(skill_dir)
- except BaseException:
- previous.rename(skill_dir)
- raise
- shutil.rmtree(previous, ignore_errors=True)
+ if source_dir != skill_dir:
+ try:
+ _rename_no_replace(stage, skill_dir)
+ except OSError as exc:
+ if exc.errno in (errno.EEXIST, errno.ENOTEMPTY):
+ raise ValueError(
+ f"writable activation target appeared during promotion: {skill_dir}") from exc
+ raise
+ else:
+ skill_dir.rename(previous)
+ try:
+ stage.rename(skill_dir)
+ except BaseException:
+ previous.rename(skill_dir)
+ raise
+ shutil.rmtree(previous, ignore_errors=True)
except BaseException:
shutil.rmtree(stage, ignore_errors=True)
raise
@@ -376,17 +480,96 @@ def _activate_rewrite(skill: str, pending: dict) -> str:
return f"Promoted '{skill}' from revision {current.revision}; previous revision snapshotted."
+def _snapshot_absence(skill: str) -> None:
+ destination = revisions_dir() / skill / ABSENT_REVISION
+ destination.mkdir(parents=True, exist_ok=True)
+ (destination / ABSENT_MARKER).touch(exist_ok=True)
+ _stamp_snapshot_best_effort(skill, ABSENT_REVISION)
+
+
+def _activate_creation(skill: str, pending: dict) -> str:
+ """Atomically add a reviewed skill while preserving absence as its rollback target."""
+ if any(item.name == skill for item in load_skills()):
+ raise ValueError(f"skill '{skill}' already exists; review it as an update")
+ problem = stale_evidence_reason(skill, pending)
+ if problem:
+ raise ValueError(problem)
+ components = pending["challenger_components"]
+ skill_dir = writable_skill_dir(skill)
+ if skill_dir.exists() or skill_dir.is_symlink():
+ raise ValueError(f"writable activation target already exists: {skill_dir}")
+ skill_dir.parent.mkdir(parents=True, exist_ok=True)
+ _snapshot_absence(skill)
+ _sweep_staging(skill_dir)
+ stage = skill_dir.with_name(f".{skill_dir.name}.{uuid.uuid4().hex}.stage")
+ try:
+ stage.mkdir()
+ from ingot.mcp_server.registry import write_skill_md
+ try:
+ frontmatter = json.loads(components.get("frontmatter", "{}"))
+ except (TypeError, ValueError) as exc:
+ raise ValueError("creation frontmatter is not valid JSON") from exc
+ write_skill_md(stage / "SKILL.md", frontmatter, components["body"])
+ write_components(stage, components)
+ expected = pending.get("evidence", {}).get("challenger", {}).get("revision")
+ actual = skill_revision(stage)
+ if actual != expected:
+ raise ValueError("staged creation revision does not match the recorded evidence")
+ _rename_no_replace(stage, skill_dir)
+ except BaseException:
+ shutil.rmtree(stage, ignore_errors=True)
+ raise
+ pending_path(skill).unlink(missing_ok=True)
+ return f"Added '{skill}' to the served library; prior state was absence."
+
+
+def _activate_approved(skill: str, pending: dict, actor: str = "local-operator") -> str:
+ """Complete a publisher-verified approval. The UI approval path must never call this."""
+ skill = check_slug(skill)
+ _require_promotable(pending)
+ result = (_activate_creation(skill, pending) if pending.get("kind") == "creation"
+ else _activate_rewrite(skill, pending))
+ revision = _current_skill(skill).revision
+ _audit_best_effort("approve", skill, revision, actor)
+ return result
+
+
def approve_pending(skill: str, actor: str = "local-operator") -> str:
- """Activate one tested rewrite after an explicit approval action."""
+ """Approve one tested challenger for vault publication without activating it."""
skill = check_slug(skill)
pending = load_pending(skill)
if not pending:
raise ValueError(f"no pending challenger for '{skill}'")
_require_promotable(pending)
- result = _activate_rewrite(skill, pending)
- revision = _current_skill(skill).revision
- _audit_best_effort("approve", skill, revision, actor)
- return result
+ problem = stale_evidence_reason(skill, pending)
+ if problem:
+ raise ValueError(problem)
+ from ingot.optimize.publication import queue_publication
+ queue_publication(skill, pending, actor, "promote")
+ return f"Approved '{skill}'; publishing to vault."
+
+
+def challenger_revision(pending: dict) -> str:
+ """The challenger's revision from a pending record's evidence, or '' when none is recorded."""
+ revision = ((pending.get("evidence") or {}).get("challenger") or {}).get("revision")
+ return revision if isinstance(revision, str) else ""
+
+
+def reject_pending(skill: str, actor: str = "local-operator", reason: str = "") -> str:
+ """Discard one quarantined change and record why.
+
+ Shared by the console and the command line so a rejection means the same thing and lands in the
+ same trail whichever one a reviewer used. Callers that run concurrently -- the console's request
+ threads -- serialize around this themselves; loading, deleting and auditing must happen under
+ one lock or a second rejection re-deletes and double-audits after the first releases."""
+ skill = check_slug(skill)
+ pending = load_pending(skill)
+ if pending is None:
+ raise ValueError(f"no pending change for '{skill}'")
+ revision = challenger_revision(pending)
+ pending_path(skill).unlink(missing_ok=True)
+ _audit_best_effort("reject", skill, revision, actor, reason=" ".join(reason.split()))
+ return f"rejected the pending change for '{skill}'"
def _rollback_source(skill: str, revision: str) -> Path:
@@ -394,7 +577,7 @@ def _rollback_source(skill: str, revision: str) -> Path:
skill = check_slug(skill)
if not SLUG_RE.fullmatch(revision):
raise ValueError(f"invalid revision: {revision!r}")
- source = REVISIONS_DIR / skill / revision
+ source = revisions_dir() / skill / revision
if not source.is_dir():
raise ValueError(f"no snapshot for '{skill}' at revision {revision}")
return source
@@ -430,12 +613,70 @@ def _swap_rollback(skill_dir: Path, stage: Path) -> None:
def rollback(skill: str, revision: str, actor: str = "local-operator") -> str:
- """Atomically restore a snapshot while preserving the displaced current revision."""
+ """Queue a stored snapshot for vault publication without changing any served byte.
+
+ Rollback takes the same Git lane as approval: the served library is a read-only checkout of the
+ canonical vault, so restoring a revision has to travel through a merged vault commit rather
+ than through the filesystem beneath it."""
+ _rollback_source(skill, revision)
+ try:
+ expected = _current_skill(skill).revision
+ except ValueError:
+ if revision == ABSENT_REVISION:
+ raise ValueError(f"skill '{skill}' is already absent") from None
+ expected = ABSENT_REVISION
+ if expected == revision:
+ raise ValueError(f"'{skill}' already serves revision {revision}")
+ from ingot.optimize.publication import queue_publication
+ queue_publication(skill, {
+ "skill": skill,
+ "kind": "rollback",
+ "challenger_components": {},
+ "evidence": {"champion": {"revision": expected},
+ "challenger": {"revision": revision}},
+ }, actor, "rollback")
+ return f"Approved rollback of '{skill}' to {revision}; publishing to vault."
+
+
+def _activate_rollback(skill: str, revision: str, actor: str = "local-operator") -> str:
+ """Complete a publisher-verified rollback against a writable library. The UI must never call
+ this: with the canonical vault mounted read-only, the merged vault commit is what restores a
+ revision. It has no production caller today (see the session log)."""
source = _rollback_source(skill, revision)
- current = _current_skill(skill)
+ try:
+ current = _current_skill(skill)
+ except ValueError:
+ if (source / ABSENT_MARKER).is_file():
+ raise ValueError(f"skill '{skill}' is already absent")
+ skill_dir = writable_skill_dir(skill)
+ if skill_dir.exists() or skill_dir.is_symlink():
+ raise ValueError(f"inactive skill target already exists: {skill_dir}")
+ skill_dir.parent.mkdir(parents=True, exist_ok=True)
+ _sweep_staging(skill_dir)
+ stage = _stage_rollback(source, skill_dir)
+ try:
+ _rename_no_replace(stage, skill_dir)
+ except BaseException:
+ shutil.rmtree(stage, ignore_errors=True)
+ raise
+ restored = _current_skill(skill)
+ _audit_best_effort("rollback", skill, restored.revision, actor)
+ return f"Restored absent skill '{skill}' at revision {restored.revision}."
skill_dir = Path(current.root)
_snapshot(skill_dir, skill, current.revision)
_sweep_staging(skill_dir)
+ if (source / ABSENT_MARKER).is_file():
+ if skill_dir != writable_skill_dir(skill):
+ raise ValueError(f"cannot remove mounted skill '{skill}' through an absence rollback")
+ previous = skill_dir.with_name(f".{skill_dir.name}.{uuid.uuid4().hex}.previous")
+ skill_dir.rename(previous)
+ try:
+ shutil.rmtree(previous)
+ except BaseException:
+ previous.rename(skill_dir)
+ raise
+ _audit_best_effort("rollback", skill, ABSENT_REVISION, actor)
+ return f"Rolled back '{skill}' from {current.revision} to absence."
stage = _stage_rollback(source, skill_dir)
_swap_rollback(skill_dir, stage)
restored = _current_skill(skill)
diff --git a/ingot/optimize/publication.py b/ingot/optimize/publication.py
new file mode 100644
index 0000000..3e9d268
--- /dev/null
+++ b/ingot/optimize/publication.py
@@ -0,0 +1,303 @@
+"""Durable, inert approvals waiting for publication into the canonical skill vault."""
+from __future__ import annotations
+
+import fcntl
+import hashlib
+import json
+import os
+import time
+import uuid
+from dataclasses import dataclass
+from pathlib import Path, PurePosixPath
+
+from ingot.mcp_server.registry import SLUG_RE
+from ingot.optimize import tree
+from ingot import paths
+
+
+
+
+def publications_dir() -> Path:
+ """Approved receipts waiting on the publisher. Resolved per call, never bound at import."""
+ return paths.runs() / "publications"
+
+ACTIVE_STATES = {"approved_publishing", "publishing", "awaiting_merge", "merged"}
+ACTIONS = {"promote", "rollback"}
+
+
+@dataclass(frozen=True)
+class PublicationReceipt:
+ id: str
+ state: str
+ path: Path
+
+
+def _canonical(value: object) -> str:
+ return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
+
+
+def _publication_id(identity: dict) -> str:
+ return hashlib.sha256(_canonical(identity).encode()).hexdigest()[:24]
+
+
+def _proposal_id(pending: dict) -> str:
+ for key in ("creation", "retrospective"):
+ value = (pending.get(key) or {}).get("proposal_id")
+ if isinstance(value, str) and value:
+ return value
+ value = pending.get("proposal_id")
+ return value if isinstance(value, str) else ""
+
+
+def _components(pending: dict, action: str) -> dict[str, str]:
+ """A promotion carries the exact approved bytes; a rollback republishes a stored snapshot.
+
+ The snapshot is the authority for a rollback, so carrying components beside it would create a
+ second description of the same target that could disagree with it."""
+ raw = pending.get("challenger_components")
+ if action == "rollback":
+ if raw is not None and not isinstance(raw, dict):
+ raise ValueError("challenger components must be an object")
+ if raw:
+ raise ValueError("a rollback republishes a stored snapshot and carries no components")
+ return {}
+ if not isinstance(raw, dict) or not raw:
+ raise ValueError("challenger components are required for publication")
+ result = {}
+ for key, value in raw.items():
+ if not isinstance(key, str) or not isinstance(value, str):
+ raise ValueError("component names and contents must be strings")
+ if key.startswith("file:"):
+ path = PurePosixPath(key[5:])
+ if path.is_absolute() or ".." in path.parts or path.as_posix() in {".", "SKILL.md"}:
+ raise ValueError(f"component escapes skill root: {path}")
+ elif key not in {"description", "body", "frontmatter"}:
+ raise ValueError(f"unsupported component: {key}")
+ result[key] = value
+ if "description" not in result or "body" not in result:
+ raise ValueError("description and body components are required")
+ return result
+
+
+def _tree(pending: dict, action: str) -> dict | None:
+ """The staged bytes an ingested package publishes, if it carries any.
+
+ Absent for an optimizer promotion, which rewrites text in a skill the vault already holds, and
+ for a rollback, which restores a stored snapshot. Present for an ingested package, where it is
+ the authority for every file: the receipt binds the tree's digest, and the publisher refuses
+ any staged file whose hash has moved since approval."""
+ raw = pending.get("tree")
+ if raw is None:
+ return None
+ if action == "rollback":
+ raise ValueError("a rollback republishes a stored snapshot and carries no candidate tree")
+ return tree.verify_manifest(raw)
+
+
+def _read(path: Path) -> dict:
+ value = json.loads(path.read_text(encoding="utf-8"))
+ if not isinstance(value, dict):
+ raise ValueError(f"invalid publication record: {path}")
+ return value
+
+
+def load_publication(publication_id: str) -> dict | None:
+ if not SLUG_RE.fullmatch(publication_id):
+ raise ValueError(f"invalid publication id: {publication_id!r}")
+ path = publications_dir() / f"{publication_id}.json"
+ return _read(path) if path.is_file() else None
+
+
+def update_publication(publication_id: str, **changes) -> dict:
+ """Replace one receipt durably; the publication id and candidate identity cannot change."""
+ path = publications_dir() / f"{publication_id}.json"
+ record = _read(path)
+ immutable = {"id", "skill", "action", "expected_champion", "candidate_revision", "components",
+ "tree"}
+ if immutable.intersection(changes):
+ raise ValueError("publication identity is immutable")
+ record.update(changes)
+ temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp")
+ fd = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
+ try:
+ os.fchmod(fd, 0o600)
+ _write_all(fd, (_canonical(record) + "\n").encode())
+ os.fsync(fd)
+ finally:
+ os.close(fd)
+ try:
+ temporary.replace(path)
+ directory_fd = os.open(path.parent, os.O_RDONLY)
+ try:
+ os.fsync(directory_fd)
+ finally:
+ os.close(directory_fd)
+ finally:
+ temporary.unlink(missing_ok=True)
+ return record
+
+
+def _all_records() -> list[dict]:
+ """Every readable receipt, newest first. A receipt this process cannot parse is skipped
+ rather than fatal: one corrupt file must not blank the whole lane."""
+ if not publications_dir().is_dir():
+ return []
+ records = []
+ for path in publications_dir().glob("*.json"):
+ try:
+ records.append(_read(path))
+ except (OSError, ValueError, json.JSONDecodeError):
+ continue
+ records.sort(key=_ordering, reverse=True)
+ return records
+
+
+def publication_for_skill(skill: str) -> dict | None:
+ for record in _all_records():
+ if record.get("skill") == skill:
+ return record
+ return None
+
+
+LIVE_STATES = ("approved_publishing", "publishing", "awaiting_merge")
+
+
+def publishing_skills() -> set[str]:
+ """Skills whose newest receipt is still travelling to the vault.
+
+ One pass over the store, not one per skill: the board asks this for every skill it lists.
+ Only the newest receipt counts, or a skill would look like it were publishing forever on the
+ strength of some earlier attempt."""
+ newest: dict[str, dict] = {}
+ for record in _all_records():
+ newest.setdefault(record.get("skill", ""), record)
+ return {skill for skill, record in newest.items() if record.get("state") in LIVE_STATES}
+
+
+def latest_releases() -> dict[str, dict]:
+ """The newest successful release per skill: what the deployment is supposed to be serving.
+
+ One pass over the store, like `publishing_skills`, because status asks this for every skill it
+ lists. Only an `active` receipt counts — a receipt that failed on its way to the vault never
+ described served bytes and must not be mistaken for a release."""
+ releases: dict[str, dict] = {}
+ for record in _all_records():
+ if record.get("state") == "active":
+ releases.setdefault(record.get("skill", ""), record)
+ return releases
+
+
+def recent_publications(limit: int = 12) -> list[dict]:
+ """The publication lane as a whole, newest first.
+
+ `publication_for_skill` only answers for a skill that still has a pending record, so once a
+ change is approved the console loses sight of it — which is exactly the window where it is
+ waiting on a vault merge and a person needs to see it."""
+ return _all_records()[:max(0, limit)]
+
+
+def _ordering(record: dict) -> tuple:
+ """Newest first. `created` is whole seconds, so a rollback queued in the same second as the
+ approval it displaces would otherwise be ordered by the hash in its id — and the review surface
+ would render whichever receipt happened to sort higher."""
+ created = record.get("created_ns")
+ if not isinstance(created, int) or isinstance(created, bool):
+ created = int(record.get("created", 0) or 0) * 1_000_000_000
+ return (created, record.get("id", ""))
+
+
+def _write_all(fd: int, payload: bytes) -> None:
+ view = memoryview(payload)
+ while view:
+ written = os.write(fd, view)
+ if written <= 0:
+ raise OSError("publication write made no progress")
+ view = view[written:]
+
+
+def _publish(path: Path, record: dict) -> None:
+ temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp")
+ fd = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
+ try:
+ os.fchmod(fd, 0o600)
+ _write_all(fd, (_canonical(record) + "\n").encode())
+ os.fsync(fd)
+ finally:
+ os.close(fd)
+ try:
+ os.link(temporary, path)
+ directory_fd = os.open(path.parent, os.O_RDONLY)
+ try:
+ os.fsync(directory_fd)
+ finally:
+ os.close(directory_fd)
+ finally:
+ temporary.unlink(missing_ok=True)
+
+
+def queue_publication(skill: str, pending: dict, actor: str, action: str) -> PublicationReceipt:
+ """Record one approved candidate without changing any served skill or pending review slot."""
+ if not SLUG_RE.fullmatch(skill):
+ raise ValueError(f"invalid skill name: {skill!r}")
+ if not isinstance(actor, str) or not actor.strip():
+ raise ValueError("approval actor is required")
+ if action not in ACTIONS:
+ raise ValueError(f"invalid publication action: {action!r}")
+ evidence = pending.get("evidence") or {}
+ expected = ((evidence.get("champion") or {}).get("revision")
+ if pending.get("kind") != "creation" else "absent")
+ candidate = (evidence.get("challenger") or {}).get("revision")
+ if not isinstance(expected, str) or not expected:
+ raise ValueError("expected champion revision is required")
+ if not isinstance(candidate, str) or not candidate:
+ raise ValueError("candidate revision is required")
+ components = _components(pending, action)
+ candidate_tree = _tree(pending, action)
+ proposal_id = _proposal_id(pending)
+ identity = {"skill": skill, "action": action, "expected_champion": expected,
+ "candidate_revision": candidate, "components": components}
+ if candidate_tree:
+ identity["tree"] = candidate_tree["digest"]
+ publication_id = _publication_id(identity)
+ path = publications_dir() / f"{publication_id}.json"
+ record = {
+ "schema_version": "ingot/publication/v1",
+ "id": publication_id,
+ "skill": skill,
+ "state": "approved_publishing",
+ "proposal_id": proposal_id,
+ "actor": actor.strip(),
+ "action": action,
+ "kind": pending.get("kind", "quality"),
+ "expected_champion": expected,
+ "candidate_revision": candidate,
+ "components": components,
+ "created": int(time.time()),
+ "created_ns": time.time_ns(),
+ "attempts": 0,
+ "last_error": "",
+ }
+ if candidate_tree:
+ record["tree"] = candidate_tree
+
+ publications_dir().mkdir(parents=True, exist_ok=True, mode=0o700)
+ os.chmod(publications_dir(), 0o700)
+ lock_path = publications_dir() / ".queue.lock"
+ lock_fd = os.open(lock_path, os.O_CREAT | os.O_RDWR, 0o600)
+ try:
+ fcntl.flock(lock_fd, fcntl.LOCK_EX)
+ if path.is_file():
+ existing = _read(path)
+ return PublicationReceipt(publication_id, existing["state"], path)
+ occupied = publication_for_skill(skill)
+ if occupied and occupied.get("state") in ACTIVE_STATES:
+ raise ValueError(f"publication is already in progress for '{skill}'")
+ try:
+ _publish(path, record)
+ except FileExistsError:
+ existing = _read(path)
+ return PublicationReceipt(publication_id, existing["state"], path)
+ finally:
+ fcntl.flock(lock_fd, fcntl.LOCK_UN)
+ os.close(lock_fd)
+ return PublicationReceipt(publication_id, record["state"], path)
diff --git a/ingot/optimize/publisher.py b/ingot/optimize/publisher.py
new file mode 100644
index 0000000..e03a37f
--- /dev/null
+++ b/ingot/optimize/publisher.py
@@ -0,0 +1,713 @@
+"""Publish approved skill bytes through Git, then finalize only the authorized vault revision.
+
+Two backends sit behind one publication contract. `local` is the default: the vault is a Git
+repository on this machine, nothing leaves it, and the approval that queued the receipt is the only
+gate. `forge` is opt-in and preserves the GitHub lane — push, pull request, merge — for deployments
+that want the vault mirrored and the activation anchored somewhere a local administrator cannot
+rewrite. The backend is always selected explicitly; a vault that happens to have an `origin` must
+never start opening pull requests on its own.
+"""
+from __future__ import annotations
+
+import argparse
+import json
+import os
+import shutil
+import subprocess
+import sys
+import time
+from dataclasses import dataclass, field
+from pathlib import Path
+
+from ingot import delivery
+from ingot.mcp_server.registry import skill_revision, write_components, write_skill_md
+from ingot.optimize import promote
+from ingot.optimize import publication
+from ingot.optimize import tree
+
+
+BACKENDS = ("local", "forge")
+DEFAULT_BACKEND = "local"
+DEFAULT_REMOTE = "origin"
+DEFAULT_BRANCH = "main"
+PUBLISHER_IDENTITY = ("-c", "user.name=Ingot Publisher", "-c", "user.email=ingot@local.invalid")
+
+
+class ConfigurationError(ValueError):
+ """A publisher that must refuse to start rather than queue approvals it can never publish."""
+
+
+def _run(command: list[str], *, cwd: Path) -> str:
+ result = subprocess.run(command, cwd=cwd, capture_output=True, text=True)
+ if result.returncode:
+ label = " ".join(command[:2])
+ detail = result.stderr.strip().splitlines()[-1] if result.stderr.strip() else "command failed"
+ raise RuntimeError(f"{label}: {detail}")
+ return result.stdout.strip()
+
+
+def _forge_urls(repository: str) -> set[str]:
+ """Every spelling a checkout of one GitHub repository legitimately uses for its remote."""
+ return {f"https://github.com/{repository}.git", f"https://github.com/{repository}",
+ f"git@github.com:{repository}.git", f"git@github.com:{repository}",
+ f"ssh://git@github.com/{repository}.git", f"ssh://git@github.com/{repository}"}
+
+
+class VaultRepo:
+ """The vault checkout, which is also the served checkout.
+
+ `remote` is None in local mode, where this repository is the whole history there is: no origin
+ to verify, nothing to fetch, and `main` is the only authority."""
+
+ def __init__(self, path: Path, *, remote: str | None = None, branch: str = DEFAULT_BRANCH):
+ self.path = path
+ self.remote = remote
+ self.branch = branch
+
+ @property
+ def base(self) -> str:
+ """The ref publications are cut from and fast-forwarded onto."""
+ return f"{self.remote}/{self.branch}" if self.remote else self.branch
+
+ def git(self, *args: str) -> str:
+ return _run(["git", *args], cwd=self.path)
+
+ def _is_ancestor(self, commit: str, of: str) -> bool:
+ result = subprocess.run(["git", "merge-base", "--is-ancestor", commit, of],
+ cwd=self.path, capture_output=True)
+ return result.returncode == 0
+
+ def contains(self, commit: str) -> bool:
+ """Whether the base ref already carries this commit, directly or beneath a later one."""
+ return self._is_ancestor(commit, self.base)
+
+ def start_point(self, branch: str) -> str:
+ """Where this publication's worktree is cut from.
+
+ An existing publication branch is reused so a retry is idempotent — the absence rollback
+ depends on it, because its second attempt must start from a tree where the skill is already
+ gone. In local mode a branch that no longer descends from `main` (someone committed to the
+ vault directly in between) can never fast-forward, so it is abandoned and recut from `main`
+ rather than left to wedge the receipt on every retry. The publication is re-materialized
+ against the new champion and refused if it no longer matches."""
+ if self.remote:
+ found = self.git("ls-remote", "--heads", self.remote, f"refs/heads/{branch}")
+ return f"{self.remote}/{branch}" if found else self.base
+ exists = subprocess.run(["git", "rev-parse", "--verify", "--quiet", f"refs/heads/{branch}"],
+ cwd=self.path, capture_output=True).returncode == 0
+ return branch if exists and self._is_ancestor(self.base, branch) else self.base
+
+ @classmethod
+ def open(cls, path: Path, *, remote: str | None = None, expected_remotes: set[str] | None = None,
+ branch: str = DEFAULT_BRANCH, allow_behind: bool = False,
+ sync: bool = False) -> "VaultRepo":
+ """Open the served vault checkout, refusing anything the publisher must not build on.
+
+ `sync` fast-forwards a checkout that is merely behind its remote. The vault has other
+ writers — a person committing directly, another machine — and the served library is by
+ definition a mirror of vault `main`, so falling behind is normal and must not block
+ publishing. Only `_prepare` may sync: in `_finalize` the fast-forward is the activation step
+ itself, and syncing early would discard the champion bytes before they are snapshotted as
+ the rollback target. In local mode there is no remote, so both are no-ops."""
+ resolved = path.expanduser().resolve(strict=True)
+ repo = cls(resolved, remote=remote, branch=branch)
+ if remote is not None and expected_remotes is not None:
+ if repo.git("remote", "get-url", remote) not in expected_remotes:
+ raise ValueError(f"vault {remote} is not the configured vault repository")
+ try:
+ head = repo.git("symbolic-ref", "--short", "HEAD")
+ except RuntimeError as exc:
+ raise ValueError("vault checkout is detached") from exc
+ if head != branch:
+ raise ValueError(f"vault checkout must remain on {branch}")
+ if repo.git("status", "--porcelain"):
+ raise ValueError("vault checkout is dirty")
+ if remote is None:
+ return repo
+ repo.git("fetch", remote, branch)
+ head_commit = repo.git("rev-parse", "HEAD")
+ upstream = repo.git("rev-parse", repo.base)
+ if head_commit != upstream:
+ if not repo._is_ancestor(head_commit, repo.base):
+ raise ValueError(f"vault {branch} diverged from {repo.base}")
+ if sync:
+ repo.git("merge", "--ff-only", repo.base)
+ elif not allow_behind:
+ raise ValueError(f"vault HEAD must equal {repo.base}")
+ return repo
+
+
+class LocalBackend:
+ """The default. The vault is a local Git repository and no network is involved at any point.
+
+ There is no `awaiting_merge` state: the commit is authorized the moment it exists, because the
+ human gate is the approval that queued the receipt, not a second review of the same bytes."""
+
+ name = "local"
+ remote = None
+ expected_remotes = None
+
+ def __init__(self, *, branch: str = DEFAULT_BRANCH):
+ self.branch = branch
+
+ def authorize(self, repo: VaultRepo, record: dict, branch: str,
+ workspace: Path | None = None) -> str | None:
+ return repo.git("rev-parse", f"refs/heads/{branch}")
+
+ def advance(self, repo: VaultRepo, record: dict, commit: str) -> None:
+ """Fast-forward only. A vault whose `main` moved under the publication is a refusal, never
+ a rebase, a merge commit, or a reset: the operator resolves it and the receipt retries."""
+ repo.git("merge", "--ff-only", commit)
+
+ def retire(self, repo: VaultRepo, record: dict) -> None:
+ _retire(repo, record, (("branch", "-D", record.get("branch", "")),))
+
+
+class GitHub:
+ """`gh` against the vault checkout, whose authenticated host credentials this process reuses."""
+
+ def __init__(self, vault_dir: Path, repository: str, *, base: str = DEFAULT_BRANCH):
+ self.vault_dir = vault_dir
+ self.repository = repository
+ self.base = base
+
+ def _json(self, args: list[str]):
+ output = _run(["gh", *args], cwd=self.vault_dir)
+ return json.loads(output) if output else None
+
+ def create_or_find(self, branch: str, publication_id: str) -> int:
+ found = self._json(["pr", "list", "--repo", self.repository, "--head", branch,
+ "--state", "all", "--json", "number,state"]) or []
+ if found:
+ # Someone closing the vault pull request has rejected the publication. Opening a second
+ # one for the same branch would overrule that, so this refuses and leaves the reason on
+ # the receipt instead.
+ if found[0].get("state") == "CLOSED":
+ raise ValueError(f"the vault pull request for {branch} was closed without merging")
+ return int(found[0]["number"])
+ output = _run(["gh", "pr", "create", "--repo", self.repository, "--base", self.base,
+ "--head", branch, "--title", f"Publish Ingot approval {publication_id}",
+ "--body", f"Evidence-gated publication `{publication_id}`."],
+ cwd=self.vault_dir)
+ return int(output.rstrip("/").split("/")[-1])
+
+ def enable_auto_merge(self, pr: int) -> None:
+ _run(["gh", "pr", "merge", str(pr), "--repo", self.repository, "--auto", "--squash"],
+ cwd=self.vault_dir)
+
+ def merged_commit(self, pr: int) -> str | None:
+ data = self._json(["pr", "view", str(pr), "--repo", self.repository,
+ "--json", "state,mergeCommit"])
+ if not data or data.get("state") != "MERGED":
+ return None
+ return (data.get("mergeCommit") or {}).get("oid")
+
+
+class ForgeBackend:
+ """Opt-in. Publication authority is a merged pull request in a configured Git forge."""
+
+ name = "forge"
+
+ def __init__(self, github=None, *, repository: str, remote: str = DEFAULT_REMOTE,
+ branch: str = DEFAULT_BRANCH, expected_remotes: set[str] | None = None):
+ if not repository:
+ raise ConfigurationError("the forge backend requires a repository "
+ "(INGOT_FORGE_REPOSITORY)")
+ self.repository = repository
+ self.remote = remote
+ self.branch = branch
+ self.expected_remotes = expected_remotes or _forge_urls(repository)
+ self.github = github
+
+ def _forge(self, repo: VaultRepo) -> GitHub:
+ if self.github is None:
+ self.github = GitHub(repo.path, self.repository, base=self.branch)
+ return self.github
+
+ def authorize(self, repo: VaultRepo, record: dict, branch: str,
+ workspace: Path | None = None) -> str | None:
+ """With a workspace, submit the branch for authority; without one, ask whether it landed.
+
+ A vault that does not allow auto-merge is not a failure: the pull request is open and a
+ human merging it is the gate this whole lane exists for. Wedging the receipt here would
+ strand an approval that is already waiting on GitHub, and nothing downstream acts until the
+ merge is visible, so the guarantee is unchanged either way."""
+ github = self._forge(repo)
+ if workspace is not None:
+ _run(["git", "push", self.remote, f"HEAD:refs/heads/{branch}"], cwd=workspace)
+ pr = github.create_or_find(branch, record["id"])
+ publication.update_publication(record["id"], branch=branch, pr=pr)
+ auto_merge, note = True, ""
+ try:
+ github.enable_auto_merge(pr)
+ except RuntimeError as exc:
+ auto_merge, note = False, f"auto-merge unavailable, waiting on a human merge: {exc}"
+ publication.update_publication(record["id"], pr=pr, auto_merge=auto_merge, note=note)
+ return None
+ merged = github.merged_commit(int(record["pr"]))
+ if not merged:
+ return None
+ # Fetch here rather than relying on the caller having opened the vault: a merge that landed
+ # a second ago is not an ancestor of a stale `origin/main`, and refusing it would fail the
+ # receipt for the one outcome this lane is waiting for.
+ repo.git("fetch", self.remote, self.branch)
+ if not repo.contains(merged):
+ raise ValueError(f"the approved merge commit is not part of {repo.base}")
+ return merged
+
+ def advance(self, repo: VaultRepo, record: dict, commit: str) -> None:
+ # The authorized commit is already on the base ref, verified above, so the served checkout
+ # advances by fast-forwarding onto it rather than onto the branch.
+ repo.git("merge", "--ff-only", repo.base)
+
+ def retire(self, repo: VaultRepo, record: dict) -> None:
+ branch = record.get("branch", "")
+ _retire(repo, record, (("push", self.remote, "--delete", branch), ("branch", "-D", branch)))
+
+
+def _retire(repo: VaultRepo, record: dict, commands) -> None:
+ """Drop the publication branch once its commit is served.
+
+ Every publication opens one, so leaving them behind grows the vault's branch list without
+ bound. This runs after the receipt is durable and never raises: the publication is already
+ active, and an uncleaned branch must not turn a completed activation into a retry."""
+ if not record.get("branch"):
+ return
+ for command in commands:
+ try:
+ repo.git(*command)
+ except RuntimeError as exc:
+ print(f"[publisher] {record['id']}: could not remove {record['branch']}: {exc}",
+ flush=True)
+
+
+class Publisher:
+ def __init__(self, vault_dir: Path, *, backend=None, targets=None):
+ self.vault_dir = vault_dir.expanduser().resolve()
+ self.backend = backend if backend is not None else LocalBackend()
+ # An unconfigured publisher delivers to the vault and nowhere else, which is what every
+ # deployment before delivery targets existed already did.
+ self.targets = (tuple(targets) if targets is not None
+ else delivery.parse_targets("", vault=self.vault_dir))
+
+ def _repo(self) -> VaultRepo:
+ """The vault checkout, unvalidated. For questions that do not build on it."""
+ return VaultRepo(self.vault_dir, remote=self.backend.remote, branch=self.backend.branch)
+
+ def _open(self, **kwargs) -> VaultRepo:
+ return VaultRepo.open(self.vault_dir, remote=self.backend.remote,
+ expected_remotes=self.backend.expected_remotes,
+ branch=self.backend.branch, **kwargs)
+
+ @staticmethod
+ def _branch(record: dict) -> str:
+ return f"ingot/{record['id']}"
+
+ def _workspace(self, record: dict) -> Path:
+ return publication.publications_dir() / "worktrees" / record["id"]
+
+ @staticmethod
+ def _register(root: Path, skill: str, *, present: bool) -> None:
+ """Keep the vault registry in step with what the publication adds or removes.
+
+ A stale entry left by an earlier removal would otherwise survive: the skill would land in
+ the vault still marked for removal, and the projection would drop what was just approved. A
+ curated `keep` entry is left exactly as the operator wrote it."""
+ path = root / "registry.json"
+ registry = json.loads(path.read_text(encoding="utf-8"))
+ if present:
+ entry = registry.get(skill)
+ if not isinstance(entry, dict) or entry.get("disposition") != "keep":
+ registry[skill] = {"disposition": "keep", "reason": "Approved through Ingot."}
+ else:
+ registry.pop(skill, None)
+ path.write_text(json.dumps(registry, indent=2, sort_keys=True) + "\n")
+
+ def _materialize_rollback(self, root: Path, record: dict) -> None:
+ """Restore a stored snapshot wholesale.
+
+ Writing the recorded components back would only restore text the optimizer can rewrite: a
+ file the displaced revision added, or any non-text file, would survive the rollback and the
+ revision check below would then refuse it. The snapshot is a complete copy of the skill, so
+ replacing the directory with it is both simpler and exact."""
+ skill = root / record["skill"]
+ if skill.is_dir():
+ shutil.rmtree(skill)
+ if record["candidate_revision"] == promote.ABSENT_REVISION:
+ self._register(root, record["skill"], present=False)
+ return
+ source = promote._rollback_source(record["skill"], record["candidate_revision"])
+ shutil.copytree(source, skill, symlinks=True)
+ self._register(root, record["skill"], present=True)
+ if skill_revision(skill) != record["candidate_revision"]:
+ raise ValueError("restored vault package does not match the approved revision")
+
+ def _materialize(self, root: Path, record: dict) -> None:
+ if record["action"] == "rollback":
+ return self._materialize_rollback(root, record)
+ skill = root / record["skill"]
+ components = record["components"]
+ candidate_tree = record.get("tree")
+ if record["kind"] == "creation":
+ if skill.is_dir() and skill_revision(skill) == record["candidate_revision"]:
+ return
+ if skill.exists() or skill.is_symlink():
+ raise ValueError(f"skill '{record['skill']}' already exists in the vault")
+ if candidate_tree:
+ # Every file the operator ingested, byte for byte, verified against the receipt on
+ # the way in. `write_components` is deliberately not called afterwards: it would
+ # rewrite each text file from a decoded copy, and a decoded copy is exactly what
+ # this exists to stop the vault from serving.
+ tree.materialize_creation(candidate_tree, components, record["skill"], skill)
+ self._register(root, record["skill"], present=True)
+ if skill_revision(skill) != record["candidate_revision"]:
+ raise ValueError(
+ "materialized vault package does not match the approved revision")
+ return
+ skill.mkdir(parents=True)
+ try:
+ metadata = json.loads(components.get("frontmatter", "{}"))
+ except (TypeError, ValueError) as exc:
+ raise ValueError("creation frontmatter is not valid JSON") from exc
+ metadata["name"] = record["skill"]
+ metadata["description"] = components["description"]
+ write_skill_md(skill / "SKILL.md", metadata, components["body"])
+ self._register(root, record["skill"], present=True)
+ else:
+ if not skill.is_dir():
+ raise ValueError(f"skill '{record['skill']}' is absent from the vault")
+ current = skill_revision(skill)
+ if current == record["candidate_revision"]:
+ return
+ if current != record["expected_champion"]:
+ raise ValueError("vault champion does not match the approved revision")
+ write_components(skill, components)
+ if skill_revision(skill) != record["candidate_revision"]:
+ raise ValueError("materialized vault package does not match the approved revision")
+
+ def _prepare(self, record: dict) -> str:
+ repo = self._open(sync=True)
+ branch = record.get("branch") or self._branch(record)
+ publication.update_publication(record["id"], state="publishing", branch=branch,
+ attempts=record.get("attempts", 0) + 1, last_error="")
+ workspace = self._workspace(record)
+ workspace.parent.mkdir(parents=True, exist_ok=True)
+ start = repo.start_point(branch)
+ # A run killed mid-publication leaves its worktree behind. Reusing it would let whatever it
+ # holds — a half-written materialization, an unrelated edit — ride into the publication,
+ # so every attempt starts from a worktree cut fresh from the recorded start point.
+ if workspace.exists():
+ try:
+ repo.git("worktree", "remove", "--force", str(workspace))
+ except RuntimeError:
+ shutil.rmtree(workspace, ignore_errors=True)
+ repo.git("worktree", "prune")
+ repo.git("worktree", "add", "-B", branch, str(workspace), start)
+ self._materialize(workspace, record)
+ _run([sys.executable, "scripts/validate.py"], cwd=workspace)
+ # Stage the whole worktree rather than naming the skill: a retry of an absence rollback
+ # starts from the publication branch, where the skill directory is already gone, and naming
+ # a pathspec that matches nothing is a fatal `git add`. The worktree is cut fresh from the
+ # start ref and `_materialize` is its only writer, so there is nothing else in it to stage.
+ _run(["git", "add", "-A", "."], cwd=workspace)
+ # An empty diff is a no-op, not an error: re-publishing a revision the vault already
+ # carries must converge on the existing branch tip.
+ if _run(["git", "status", "--porcelain"], cwd=workspace):
+ _run(["git", *PUBLISHER_IDENTITY, "commit", "-m",
+ f"Publish {record['skill']} via Ingot {record['id']}"], cwd=workspace)
+ authorized = self.backend.authorize(repo, record, branch, workspace)
+ repo.git("worktree", "remove", "--force", str(workspace))
+ branch_commit = repo.git("rev-parse", f"refs/heads/{branch}")
+ if authorized is None:
+ publication.update_publication(record["id"], state="awaiting_merge", branch=branch,
+ branch_commit=branch_commit, last_error="")
+ return "awaiting_merge"
+ # The local backend has no external gate to wait on, so one pass runs both halves. A crash
+ # in between leaves the receipt at `publishing` with the branch committed, and the next
+ # pass re-enters here, finds the branch, re-authorizes, and finalizes.
+ publication.update_publication(record["id"], branch=branch, branch_commit=branch_commit,
+ last_error="")
+ return self._finalize(publication.load_publication(record["id"]))
+
+ def _serves(self, record: dict, revision: str) -> bool:
+ """Whether the served checkout is exactly at one recorded revision. Absence is a revision:
+ a creation's champion and an absence rollback's candidate are both 'no directory at all'."""
+ skill = self.vault_dir / record["skill"]
+ if revision == promote.ABSENT_REVISION:
+ return not skill.exists()
+ return skill.is_dir() and skill_revision(skill) == revision
+
+ def _deliver(self, record: dict) -> None:
+ """Install the approved revision at every configured target.
+
+ Runs after the vault serves the candidate and before the receipt is marked active, so a
+ target that cannot be written leaves a release that retries rather than one that claims to
+ be finished. Each target's outcome is recorded separately: a deployment with two targets
+ where one failed must be able to say which one.
+
+ The source is the vault's own copy, already checked against the receipt by `_serves`. A
+ failure raises, and `process` records it on the receipt as `last_error`."""
+ skill, revision = record["skill"], record["candidate_revision"]
+ source = None if revision == promote.ABSENT_REVISION else self.vault_dir / skill
+ if source is not None and not source.is_dir():
+ raise ValueError(f"the vault does not hold '{skill}' to deliver")
+ delivered = dict(record.get("delivery") or {})
+ for target in self.targets:
+ entry = {"kind": target.kind, "root": str(target.root), "at": int(time.time())}
+ try:
+ delivery.install(target, skill, source, revision)
+ except Exception as exc:
+ delivered[target.name] = {**entry, "state": "failed", "error": str(exc),
+ "revision": delivery.observed(target, skill)}
+ publication.update_publication(record["id"], delivery=delivered)
+ raise
+ delivered[target.name] = {**entry, "state": "delivered", "revision": revision}
+ publication.update_publication(record["id"], delivery=delivered)
+
+ def _finalize(self, record: dict) -> str:
+ branch = record.get("branch") or self._branch(record)
+ # Ask whether authority exists before validating the checkout. A receipt waiting on a forge
+ # merge is polled every few seconds and opening the vault fetches, so validating first put
+ # a network round trip -- and a failure mode -- in front of a question that needs neither.
+ authorized = self.backend.authorize(self._repo(), record, branch)
+ if not authorized:
+ return "awaiting_merge"
+ repo = self._open(allow_behind=True)
+ # Check the review slot before anything is snapshotted or fast-forwarded. Validating it
+ # afterwards would leave a receipt that no longer matches its review having already moved
+ # the served bytes, which is the one thing this lane exists to prevent. Only an approval
+ # owns a review slot: a rollback publishes a stored snapshot and must leave an unrelated
+ # pending challenger for that skill reviewable.
+ pending = promote.load_pending(record["skill"]) if record["action"] == "promote" else None
+ if pending is not None:
+ recorded = ((pending.get("evidence") or {}).get("challenger") or {}).get("revision")
+ if recorded != record["candidate_revision"]:
+ raise ValueError("pending review no longer matches the publication")
+ # A crash between the fast-forward and this receipt leaves the approved revision already
+ # served. Re-snapshotting and re-merging then would refuse a champion that is legitimately
+ # gone, so a checkout already at the candidate skips straight to finalizing the receipt.
+ if not self._serves(record, record["candidate_revision"]):
+ if not self._serves(record, record["expected_champion"]):
+ raise ValueError("served champion changed before vault activation")
+ if record["expected_champion"] == promote.ABSENT_REVISION:
+ promote._snapshot_absence(record["skill"])
+ else:
+ promote._snapshot(self.vault_dir / record["skill"], record["skill"],
+ record["expected_champion"])
+ self.backend.advance(repo, record, authorized)
+ if not self._serves(record, record["candidate_revision"]):
+ raise ValueError("the fast-forwarded vault does not serve the approved revision")
+ self._deliver(record)
+ if pending is not None:
+ promote.pending_path(record["skill"]).unlink()
+ promote._audit_best_effort(record["action"] if record["action"] == "rollback" else "approve",
+ record["skill"], record["candidate_revision"], record["actor"])
+ publication.update_publication(record["id"], state="active", merged_commit=authorized,
+ activated=int(time.time()), last_error="")
+ self.backend.retire(repo, record)
+ return "active"
+
+ def process(self, publication_id: str) -> str:
+ record = publication.load_publication(publication_id)
+ if not record:
+ raise ValueError(f"unknown publication: {publication_id}")
+ try:
+ # The queue validates both of these when it writes a receipt. Re-checking them here
+ # keeps a receipt that was edited or corrupted on disk from steering a filesystem path
+ # or reaching an activation branch it was never approved for.
+ promote.check_slug(record["skill"])
+ if record.get("action") not in publication.ACTIONS:
+ raise ValueError(f"invalid publication action: {record.get('action')!r}")
+ if record["state"] in {"approved_publishing", "publishing"}:
+ return self._prepare(record)
+ if record["state"] == "awaiting_merge":
+ return self._finalize(record)
+ return record["state"]
+ except Exception as exc:
+ current = publication.load_publication(publication_id) or record
+ publication.update_publication(
+ publication_id, attempts=current.get("attempts", 0) + 1, last_error=str(exc))
+ if isinstance(exc, RuntimeError):
+ raise
+ raise RuntimeError(f"publication {publication_id}: {exc}") from exc
+
+
+@dataclass(frozen=True)
+class PublisherConfig:
+ """What the publisher was told to be, before anything checks whether it can be."""
+
+ backend: str
+ vault_dir: Path
+ forge_repository: str | None = None
+ forge_remote: str = DEFAULT_REMOTE
+ forge_branch: str = DEFAULT_BRANCH
+ delivery_targets: tuple = field(default=())
+ warnings: tuple[str, ...] = field(default=())
+
+ def build(self) -> Publisher:
+ targets = self.delivery_targets or None
+ if self.backend == "forge":
+ return Publisher(self.vault_dir, targets=targets, backend=ForgeBackend(
+ repository=self.forge_repository, remote=self.forge_remote,
+ branch=self.forge_branch))
+ return Publisher(self.vault_dir, targets=targets, backend=LocalBackend())
+
+
+FORGE_KEYS = ("INGOT_FORGE_REPOSITORY", "INGOT_FORGE_REMOTE", "INGOT_FORGE_BRANCH")
+
+
+def load_config(env: dict | None = None, *, backend: str | None = None,
+ vault: Path | None = None) -> PublisherConfig:
+ """Explicit argument, then environment. There is no fallback to a writable default path.
+
+ A publisher that guesses its vault would happily own the demo directory, which is the one
+ outcome managed mode exists to prevent."""
+ env = os.environ if env is None else env
+ chosen = backend or env.get("INGOT_PUBLISH_BACKEND") or DEFAULT_BACKEND
+ if chosen not in BACKENDS:
+ raise ConfigurationError(f"unknown publication backend {chosen!r}; "
+ f"expected one of {', '.join(BACKENDS)}")
+ warnings = []
+ path = vault or env.get("INGOT_VAULT_PATH") or env.get("VAULT_DIR")
+ if not path:
+ raise ConfigurationError("no vault configured; set INGOT_VAULT_PATH to the checkout this "
+ "publisher owns, or pass --vault")
+ if not vault and not env.get("INGOT_VAULT_PATH") and env.get("VAULT_DIR"):
+ warnings.append("VAULT_DIR is the pre-backend name for INGOT_VAULT_PATH; prefer the latter")
+ repository = env.get("INGOT_FORGE_REPOSITORY")
+ if chosen == "local":
+ # A half-finished forge configuration under the local backend looks active and is not. Say
+ # so rather than letting someone believe their pull requests are being opened.
+ stray = [key for key in FORGE_KEYS if env.get(key)]
+ if stray:
+ warnings.append(f"{', '.join(stray)} set but the backend is local; forge settings are "
+ f"inert until INGOT_PUBLISH_BACKEND=forge")
+ elif not repository:
+ raise ConfigurationError("the forge backend requires INGOT_FORGE_REPOSITORY=owner/repo")
+ vault_dir = Path(path).expanduser()
+ try:
+ targets = delivery.parse_targets(env.get(delivery.TARGETS) or "", vault=vault_dir)
+ except ValueError as exc:
+ raise ConfigurationError(f"{delivery.TARGETS}: {exc}") from exc
+ return PublisherConfig(
+ backend=chosen, vault_dir=Path(path), forge_repository=repository,
+ forge_remote=env.get("INGOT_FORGE_REMOTE") or DEFAULT_REMOTE,
+ forge_branch=env.get("INGOT_FORGE_BRANCH") or DEFAULT_BRANCH,
+ delivery_targets=targets, warnings=tuple(warnings))
+
+
+def validate(config: PublisherConfig) -> Publisher:
+ """Refuse to start rather than fail on the first approval.
+
+ An approval queued against a publisher that cannot publish is the stalled-lane failure this
+ deployment has already hit once: the console reports the change accepted, the receipt sits at
+ `approved_publishing` forever, and nothing says why."""
+ publisher = config.build()
+ if config.backend == "forge":
+ if not shutil.which("gh"):
+ raise ConfigurationError("the forge backend needs the GitHub CLI (`gh`) on PATH")
+ try:
+ _run(["gh", "auth", "status"], cwd=Path.cwd())
+ except RuntimeError as exc:
+ raise ConfigurationError(f"`gh` is not authenticated: {exc}") from exc
+ try:
+ _run(["gh", "repo", "view", config.forge_repository, "--json", "name"], cwd=Path.cwd())
+ except RuntimeError as exc:
+ raise ConfigurationError(
+ f"the configured forge repository {config.forge_repository!r} does not "
+ f"resolve: {exc}") from exc
+ try:
+ publisher._open(allow_behind=True)
+ except (ValueError, OSError) as exc:
+ raise ConfigurationError(f"vault at {config.vault_dir}: {exc}") from exc
+ for target in config.delivery_targets:
+ if target.kind != delivery.FILESYSTEM:
+ continue
+ # Created here rather than on the first approval. A delivery root that cannot be made or
+ # written strands a change that a human has already approved, and this is the one place
+ # that can say so before anybody approves anything.
+ try:
+ target.root.mkdir(parents=True, exist_ok=True)
+ except OSError as exc:
+ raise ConfigurationError(
+ f"delivery target {target.name!r}: cannot create {target.root}: {exc}") from exc
+ if not os.access(target.root, os.W_OK | os.X_OK):
+ raise ConfigurationError(f"delivery target {target.name!r}: {target.root} is not "
+ f"writable by uid {os.getuid()}")
+ validator = config.vault_dir / "scripts" / "validate.py"
+ if not validator.is_file():
+ raise ConfigurationError(
+ f"the vault has no validator at {validator}; every publication runs it before "
+ f"committing, so a missing one is a configuration error and not a silent skip "
+ f"(`ingot vault init {config.vault_dir}` writes a default)")
+ return publisher
+
+
+def unreadable_queue(directory: Path) -> str | None:
+ """Why the receipt store cannot be read, or None.
+
+ `Path.glob` swallows `PermissionError`, so a queue this process cannot list is indistinguishable
+ from an empty one: approvals pile up in the console while the publisher reports nothing at all.
+ That is the exact shape of the deployment failure on the host — receipts written by a container
+ running as root, read by a publisher running as an ordinary user."""
+ if not directory.exists():
+ return None
+ if not os.access(directory, os.R_OK | os.X_OK):
+ return (f"cannot read the receipt store at {directory} as uid {os.getuid()}; approvals will "
+ f"queue and never publish (the console writes it at mode 0700, so both must run as "
+ f"the same user)")
+ return None
+
+
+def watch(publisher: Publisher, interval: float = 5.0) -> None:
+ reported = None
+ while True:
+ blocked = unreadable_queue(publication.publications_dir())
+ if blocked != reported: # report each transition once, not every poll
+ print(f"[publisher] {blocked}" if blocked else "[publisher] receipt store readable again",
+ flush=True)
+ reported = blocked
+ if not blocked:
+ for path in sorted(publication.publications_dir().glob("*.json")):
+ try:
+ publisher.process(path.stem)
+ except Exception as exc:
+ print(f"[publisher] {path.stem}: {exc}", flush=True)
+ time.sleep(interval)
+
+
+def main(argv: list[str] | None = None) -> int:
+ parser = argparse.ArgumentParser(
+ prog="python -m ingot.optimize.publisher",
+ description="The one writer of the served skill library.")
+ parser.add_argument("--watch", action="store_true")
+ parser.add_argument("--backend", choices=BACKENDS, default=None,
+ help="publication backend (default: $INGOT_PUBLISH_BACKEND, else local)")
+ parser.add_argument("--vault", type=Path, default=None,
+ help="the vault checkout this publisher owns "
+ "(default: $INGOT_VAULT_PATH)")
+ parser.add_argument("publication_id", nargs="?")
+ args = parser.parse_args(argv)
+ try:
+ config = load_config(backend=args.backend, vault=args.vault)
+ for warning in config.warnings:
+ print(f"[publisher] warning: {warning}", flush=True)
+ publisher = validate(config)
+ except ConfigurationError as exc:
+ print(f"[publisher] refusing to start: {exc}", file=sys.stderr, flush=True)
+ return 2
+ print(f"[publisher] backend={config.backend} vault={config.vault_dir}", flush=True)
+ for target in config.delivery_targets:
+ print(f"[publisher] delivering to {target.name} ({target.kind}) at {target.root}",
+ flush=True)
+ if args.watch:
+ watch(publisher)
+ return 0
+ if not args.publication_id:
+ parser.error("publication_id is required without --watch")
+ print(publisher.process(args.publication_id))
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/ingot/optimize/requirements-harbor-gateway.txt b/ingot/optimize/requirements-harbor-gateway.txt
new file mode 100644
index 0000000..4a37abd
--- /dev/null
+++ b/ingot/optimize/requirements-harbor-gateway.txt
@@ -0,0 +1,3 @@
+# Dedicated, ignored Dell-only runtime for Harbor's local role compatibility gateway.
+# Keep the proxy extra explicit: LiteLLM's base package does not install its proxy server.
+litellm[proxy]==1.93.0
diff --git a/ingot/optimize/retrospective.py b/ingot/optimize/retrospective.py
new file mode 100644
index 0000000..222efa2
--- /dev/null
+++ b/ingot/optimize/retrospective.py
@@ -0,0 +1,306 @@
+"""Turn a skill-retrospective finding into an inert, revision-bound Ingot challenger."""
+from __future__ import annotations
+
+import hashlib
+import json
+import logging
+import os
+import shutil
+import threading
+import time
+import uuid
+from pathlib import Path
+
+from ingot.mcp_server.registry import (load_skills, optimizable_components, resolve_skill_dir,
+ skill_revision)
+from ingot.optimize import promote
+from ingot.optimize.evidence import recorded_path
+from ingot import paths
+
+SCHEMA = "ingot/retrospective-proposal/v1"
+
+
+def evidence_dir() -> Path:
+ return paths.runs() / "evidence"
+
+
+def audit_file() -> Path:
+ return paths.runs() / "retrospective-audit.jsonl"
+
+MAX_BODY_CHARS = 200_000
+MAX_FIELD_CHARS = 4_000
+MAX_EVIDENCE_ITEMS = 12
+
+logger = logging.getLogger(__name__)
+_SUBMIT_LOCK = threading.Lock()
+
+
+def _text(name: str, value: object, *, required: bool = True,
+ limit: int = MAX_FIELD_CHARS) -> str:
+ if not isinstance(value, str):
+ raise ValueError(f"{name} must be a string")
+ value = value.strip()
+ if required and not value:
+ raise ValueError(f"{name} is required")
+ if len(value) > limit:
+ raise ValueError(f"{name} exceeds {limit} characters")
+ return value
+
+
+def _evidence_items(items: object) -> list[str]:
+ if not isinstance(items, list) or len(items) < 2:
+ raise ValueError("evidence must contain at least two concrete repeat items")
+ if len(items) > MAX_EVIDENCE_ITEMS:
+ raise ValueError(f"evidence exceeds {MAX_EVIDENCE_ITEMS} items")
+ evidence = [_text(f"evidence[{index}]", item) for index, item in enumerate(items)]
+ if len(set(evidence)) != len(evidence):
+ raise ValueError("evidence items must be distinct")
+ return evidence
+
+
+def _current_skill(skill: str):
+ try:
+ skill_dir = resolve_skill_dir(skill)
+ except LookupError as exc:
+ raise ValueError(f"no indexed skill named '{skill}'") from exc
+ current = next((item for item in load_skills() if item.name == skill), None)
+ if current is None:
+ raise ValueError(f"no indexed skill named '{skill}'")
+ return current, skill_dir
+
+
+def _proposal_id(payload: dict) -> str:
+ canonical = json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
+ return hashlib.sha256(canonical.encode()).hexdigest()[:24]
+
+
+def _gate(evidence: list[str]) -> dict:
+ """Record the admission basis without presenting it as held-out quality evidence."""
+ return {
+ "promotable": True,
+ "blocked": [],
+ "warnings": ["Retrospective evidence only; no held-out A/B comparison was run."],
+ "kind": "retrospective_admission",
+ "admission": {"pressure_verification": "passed", "evidence_items": len(evidence)},
+ }
+
+
+def _render_markdown(skill: str, proposal: dict, gate: dict) -> str:
+ verification = proposal["verification"]
+ lines = [
+ f"# Retrospective proposal: {skill}",
+ "",
+ f"**Schema:** `{SCHEMA}`",
+ f"**Proposal:** `{proposal['proposal_id']}`",
+ f"**Gate:** {'PASS' if gate['promotable'] else 'BLOCKED'}",
+ f"**Producer:** {proposal['producer']}",
+ f"**Live caller:** {proposal['caller']}",
+ "",
+ "## Finding",
+ "",
+ proposal["summary"],
+ "",
+ f"- Trigger: {proposal['trigger']}",
+ f"- Minimal reusable content: {proposal['minimal_content']}",
+ f"- Risk: {proposal['risk']}",
+ "",
+ "## Evidence",
+ "",
+ *[f"- {item}" for item in proposal["evidence"]],
+ "",
+ "## Pressure scenario",
+ "",
+ proposal["pressure_scenario"],
+ "",
+ "## Verification",
+ "",
+ f"- Status: **{verification['status'].upper()}**",
+ f"- Command: `{verification['command']}`",
+ f"- Result: {verification['result']}",
+ "",
+ "Submission only quarantines this challenger. Human approval remains required.",
+ "",
+ ]
+ return "\n".join(lines)
+
+
+def _write_evidence(skill: str, proposal: dict, gate: dict) -> dict[str, str]:
+ root = evidence_dir() / skill / f"retrospective-{proposal['proposal_id']}"
+ root.mkdir(parents=True, exist_ok=True)
+ bundle = {
+ "schema_version": SCHEMA,
+ "skill": skill,
+ "created": proposal["created"],
+ "proposal": proposal,
+ "gate": gate,
+ }
+ targets = (
+ (root / "evidence.json", json.dumps(bundle, indent=2, ensure_ascii=False) + "\n"),
+ (root / "EVIDENCE.md", _render_markdown(skill, proposal, gate)),
+ )
+ for path, content in targets:
+ temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp")
+ temporary.write_text(content, encoding="utf-8")
+ temporary.replace(path)
+ return {"json": recorded_path(targets[0][0]), "markdown": recorded_path(targets[1][0])}
+
+
+def _publish_pending(skill: str, record: dict) -> None:
+ """Create the review slot without replacing a writer that wins the race.
+
+ `save_pending` intentionally archives and replaces another pass. Retrospective submissions have
+ weaker authority: an agent may propose, but it may not displace something a human can review.
+ A hard link gives that policy an atomic create-if-absent operation on the queue filesystem.
+ """
+ promote.pending_dir().mkdir(parents=True, exist_ok=True)
+ destination = promote.pending_path(skill)
+ temporary = destination.with_name(f".{destination.name}.{uuid.uuid4().hex}.tmp")
+ with temporary.open("w", encoding="utf-8") as output:
+ output.write(json.dumps(record, indent=2, ensure_ascii=False) + "\n")
+ output.flush()
+ os.fsync(output.fileno())
+ try:
+ os.link(temporary, destination)
+ except FileExistsError:
+ raise ValueError(f"review slot is occupied for '{skill}'; resolve it before proposing")
+ finally:
+ temporary.unlink(missing_ok=True)
+
+
+def _audit(record: dict) -> None:
+ audit_file().parent.mkdir(parents=True, exist_ok=True)
+ fd = os.open(audit_file(), os.O_APPEND | os.O_CREAT | os.O_WRONLY, 0o600)
+ try:
+ os.fchmod(fd, 0o600)
+ data = (json.dumps(record, separators=(",", ":")) + "\n").encode()
+ written = 0
+ while written < len(data):
+ written += os.write(fd, data[written:])
+ os.fsync(fd)
+ finally:
+ os.close(fd)
+
+
+def submit_skill_update(
+ *,
+ skill: str,
+ champion_revision: str,
+ challenger_body: str,
+ challenger_description: str = "",
+ summary: str,
+ trigger: str,
+ minimal_content: str,
+ producer: str,
+ caller: str,
+ evidence: list[str],
+ pressure_scenario: str,
+ risk: str,
+ verification_status: str,
+ verification_command: str,
+ verification_result: str,
+) -> dict:
+ """Validate and quarantine one retrospective update; never activate or replace another slot."""
+ skill = _text("skill", skill, limit=80)
+ champion_revision = _text("champion_revision", champion_revision, limit=128)
+ challenger_body = _text("challenger_body", challenger_body, limit=MAX_BODY_CHARS)
+ challenger_description = _text(
+ "challenger_description", challenger_description, required=False, limit=2_000)
+ verification_status = _text(
+ "verification_status", verification_status, limit=20).lower()
+ if verification_status != "passed":
+ raise ValueError("verification_status must be passed before proposing")
+
+ current, skill_dir = _current_skill(skill)
+ if current.revision != champion_revision:
+ raise ValueError("champion revision changed; reload the skill before proposing")
+
+ champion = optimizable_components(skill_dir)
+ challenger = dict(champion)
+ challenger["body"] = challenger_body
+ if challenger_description:
+ challenger["description"] = challenger_description
+ changed = sorted(key for key in challenger if challenger.get(key) != champion.get(key))
+ if not changed:
+ raise ValueError("retrospective proposal does not change the skill")
+
+ created = int(time.time())
+ proposal = {
+ "schema_version": SCHEMA,
+ "created": created,
+ "summary": _text("summary", summary),
+ "trigger": _text("trigger", trigger),
+ "minimal_content": _text("minimal_content", minimal_content),
+ "producer": _text("producer", producer, limit=200),
+ "caller": _text("caller", caller, limit=500),
+ "evidence": _evidence_items(evidence),
+ "pressure_scenario": _text("pressure_scenario", pressure_scenario),
+ "risk": _text("risk", risk),
+ "verification": {
+ "status": verification_status,
+ "command": _text("verification_command", verification_command),
+ "result": _text("verification_result", verification_result),
+ },
+ }
+ # Submission time is evidence metadata, not proposal identity. Excluding it makes an exact
+ # retry idempotent even when a transport retries after the clock advances.
+ identity = {
+ **{key: value for key, value in proposal.items()
+ if key not in {"created", "producer", "caller"}},
+ "skill": skill,
+ "champion_revision": champion_revision,
+ "challenger_revision": skill_revision(skill_dir, challenger),
+ }
+ proposal["proposal_id"] = _proposal_id(identity)
+ gate = _gate(proposal["evidence"])
+ record = {
+ "skill": skill,
+ "kind": "retrospective",
+ "created": created,
+ "changed_components": changed,
+ "champion_components": champion,
+ "challenger_components": challenger,
+ "gate": gate,
+ "evidence": {
+ "schema_version": SCHEMA,
+ "champion": {"revision": champion_revision},
+ "challenger": {"revision": identity["challenger_revision"]},
+ "gate": gate,
+ },
+ "retrospective": proposal,
+ }
+
+ with _SUBMIT_LOCK:
+ existing = promote.load_pending(skill)
+ if existing:
+ existing_id = (existing.get("retrospective") or {}).get("proposal_id")
+ if existing_id == proposal["proposal_id"]:
+ return {"status": "duplicate", "skill": skill,
+ "proposal_id": proposal["proposal_id"], "promotable": gate["promotable"]}
+ raise ValueError(f"review slot is occupied for '{skill}'; resolve it before proposing")
+ record["evidence_paths"] = _write_evidence(skill, proposal, gate)
+ evidence_root = evidence_dir() / skill / f"retrospective-{proposal['proposal_id']}"
+ try:
+ _publish_pending(skill, record)
+ except ValueError:
+ current = promote.load_pending(skill)
+ current_id = ((current or {}).get("retrospective") or {}).get("proposal_id")
+ if current_id == proposal["proposal_id"]:
+ return {"status": "duplicate", "skill": skill,
+ "proposal_id": proposal["proposal_id"], "promotable": gate["promotable"]}
+ shutil.rmtree(evidence_root, ignore_errors=True)
+ raise
+ except OSError as exc:
+ shutil.rmtree(evidence_root, ignore_errors=True)
+ raise RuntimeError(
+ f"cannot atomically publish the review slot for '{skill}'") from exc
+ try:
+ _audit({"schema_version": 1, "ts": int(time.time()), "action": "quarantine",
+ "skill": skill, "proposal_id": proposal["proposal_id"],
+ "challenger_revision": identity["challenger_revision"],
+ "producer": proposal["producer"]})
+ except Exception:
+ logger.warning("Quarantined retrospective proposal %s, but its audit write failed",
+ proposal["proposal_id"], exc_info=True)
+
+ return {"status": "quarantined", "skill": skill,
+ "proposal_id": proposal["proposal_id"], "promotable": gate["promotable"]}
diff --git a/ingot/optimize/review.py b/ingot/optimize/review.py
new file mode 100644
index 0000000..7dec89b
--- /dev/null
+++ b/ingot/optimize/review.py
@@ -0,0 +1,139 @@
+"""Score one skill as it stands and report which checks it fails, without changing anything.
+
+The optimize loop answers "is this candidate better than the champion". That is the wrong question
+when you have just written a skill and want to know whether it is any good, and it is unreachable
+anyway until an eval set exists. This runs the skill against its own tasks, grades every answer
+against that task's checklist, and ranks the failures by how much they cost -- so the output is a
+list of things to fix, not a number.
+
+It never writes a pending record, never promotes, and never touches the skill directory. Drafting an
+eval set is the one side effect, and only when the skill has none (see optimize/draft.py).
+
+ python -m ingot.optimize.review
+"""
+import argparse
+import json
+import os
+import shutil
+import tempfile
+from concurrent.futures import ThreadPoolExecutor
+from pathlib import Path
+
+from ingot.mcp_server.registry import optimizable_components, skill_revision
+
+from . import SERVE_TEMPLATE, agent_model, resolve_skill_dir
+from . import usage as usage_ledger
+from .ab import load_tasks
+from .compat import _llm
+from .judge import DIMENSIONS, invoke_retry, judge
+from .rollout import assemble
+from ingot import paths
+
+REVIEW_DIR = paths.runs() / "reviews"
+_MAX_WORKERS = 8
+
+
+def _review_one(llm, system: str, task: dict) -> dict:
+ """Answer one task with the skill loaded, then grade that answer against the task's checklist."""
+ msg = invoke_retry(llm, [("system", system), ("user", task["task"])])
+ usage_ledger.add("review", getattr(msg, "usage_metadata", None))
+ verdict = judge(task["task"], task.get("rubric", ""), msg.content,
+ check=task.get("check"), deliverable=task.get("deliverable"),
+ checklist=task.get("checklist"))
+ return {"task": task["task"], "score": verdict["score"], "answer": msg.content,
+ "feedback": verdict["feedback"], "checklist": verdict["checklist"],
+ "spec": {i["id"]: i for i in (task.get("checklist") or [])}}
+
+
+def findings(results: list[dict]) -> list[dict]:
+ """Every check that did not clean-pass, worst first.
+
+ Ranked by weight x shortfall rather than by score: a weight-5 check scraping a partial matters
+ more than a weight-1 check failing outright, and sorting by the raw value buries it."""
+ out = []
+ for i, result in enumerate(results):
+ for check_id, graded in result["checklist"].items():
+ if graded["value"] >= 1.0:
+ continue
+ spec = result["spec"].get(check_id, {})
+ weight = float(spec.get("weight", 1))
+ out.append({"task_index": i, "task": result["task"], "check": check_id,
+ "criterion": spec.get("criterion", ""), "weight": weight,
+ "dimension": spec.get("dimension", ""), "value": graded["value"],
+ "note": graded["note"], "cost": weight * (1.0 - graded["value"])})
+ return sorted(out, key=lambda f: -f["cost"])
+
+
+def by_dimension(found: list[dict]) -> dict:
+ """Where the losses concentrate. A skill failing everything on completeness needs a different
+ edit from one failing scattered correctness checks."""
+ totals = {d: 0.0 for d in DIMENSIONS}
+ for f in found:
+ if f["dimension"] in totals:
+ totals[f["dimension"]] += f["cost"]
+ return {d: round(v, 3) for d, v in sorted(totals.items(), key=lambda kv: -kv[1]) if v}
+
+
+def run_review(skill: str, log=print) -> dict:
+ usage_ledger.reset()
+ skill_dir = resolve_skill_dir(skill)
+ # A review lasts about a minute. Copy once so a concurrent promotion or trusted filesystem edit
+ # cannot make the revision describe one read while the prompt grades a later read.
+ with tempfile.TemporaryDirectory(prefix=f"ingot-review-{skill}-") as temporary:
+ snapshot = Path(temporary) / skill_dir.name
+ shutil.copytree(skill_dir, snapshot, symlinks=True)
+ revision = skill_revision(snapshot)
+ components = optimizable_components(snapshot)
+ train, holdout, _ = load_tasks(
+ skill, draft_components=components) # draft against the same revision being graded
+ tasks = list(train) + list(holdout) # nothing is being generalized to; grade on all
+ if not tasks:
+ raise SystemExit(f"'{skill}' has no eval tasks to run.")
+ model = agent_model()
+ system = SERVE_TEMPLATE.format(body=assemble(components))
+ graded = sum(len(t.get("checklist") or []) for t in tasks)
+ log(f"[review] '{skill}': {len(tasks)} tasks, {graded or len(tasks) * 4} checks, on {model}")
+
+ llm = _llm(model)
+ with ThreadPoolExecutor(max_workers=min(_MAX_WORKERS, len(tasks))) as pool:
+ results = list(pool.map(lambda t: _review_one(llm, system, t), tasks))
+
+ found = findings(results)
+ score = sum(r["score"] for r in results) / len(results)
+ summary = {
+ "skill": skill, "model": model, "revision": revision,
+ "score": round(score, 4), "tasks": len(tasks),
+ "checks": sum(len(r["checklist"]) for r in results),
+ "failed_checks": len(found),
+ "by_dimension": by_dimension(found),
+ "findings": [{k: v for k, v in f.items() if k != "task_index"} for f in found],
+ "per_task": [{"task": r["task"], "score": round(r["score"], 4)} for r in results],
+ }
+ REVIEW_DIR.mkdir(parents=True, exist_ok=True)
+ path = REVIEW_DIR / f"{skill}.json"
+ path.write_text(json.dumps(summary, indent=2))
+
+ log(f"\n[review] {skill}: {score:.3f} over {len(tasks)} tasks "
+ f"({summary['checks'] - len(found)}/{summary['checks']} checks clean)")
+ if summary["by_dimension"]:
+ log("[review] losses by dimension: " +
+ ", ".join(f"{d} {v}" for d, v in summary["by_dimension"].items()))
+ for f in found[:10]:
+ log(f" - {f['check']} (weight {f['weight']:g}, {'partial' if f['value'] else 'fail'})"
+ f" — {f['note'] or f['criterion']}")
+ if len(found) > 10:
+ log(f" … {len(found) - 10} more in {path}")
+ log(f"[review] wrote {path}")
+ log(usage_ledger.format_report())
+ return summary
+
+
+if __name__ == "__main__":
+ ap = argparse.ArgumentParser(description="Score a skill against its own eval tasks and report "
+ "which checks it fails. Changes nothing.")
+ ap.add_argument("skill")
+ args = ap.parse_args()
+ if os.environ.get("REVIEW_REQUIRE_KEY", "1") != "0":
+ from . import require_openrouter_key
+ require_openrouter_key()
+ run_review(args.skill)
diff --git a/optimize/rollout.py b/ingot/optimize/rollout.py
similarity index 91%
rename from optimize/rollout.py
rename to ingot/optimize/rollout.py
index 8f3f949..dbb211c 100644
--- a/optimize/rollout.py
+++ b/ingot/optimize/rollout.py
@@ -4,7 +4,7 @@
judge's textual feedback is what the candidate search reads.
This module owns the rollout and the teacher (reflection) client only. The candidate search lives in
-`optimize.skillopt_loop`; the held-out A/B that produces promotion evidence lives in `optimize.ab`.
+`ingot.optimize.skillopt_loop`; the held-out A/B that produces promotion evidence lives in `ingot.optimize.ab`.
"""
import os
@@ -83,7 +83,8 @@ def _rollout(self, system, ex):
usage_ledger.add("rollout", getattr(msg, "usage_metadata", None))
answer = msg.content
j = judge(ex["task"], ex["rubric"], answer, reference=ex.get("reference", ""),
- check=ex.get("check"), deliverable=ex.get("deliverable"))
+ check=ex.get("check"), deliverable=ex.get("deliverable"),
+ checklist=ex.get("checklist"))
return answer, j["score"], {"task": ex["task"], "output": answer,
"feedback": j["feedback"], "dimensions": j["dimensions"]}
@@ -112,6 +113,11 @@ def make_reflection_lm():
local OPENROUTER_BASE_URL (vLLM/Ollama) is honored through litellm's generic openai provider.
Shared by SkillOpt's body and description passes."""
import litellm
+ # We pass a provider-prefixed model, but litellm resolves the provider again after the call
+ # for cost tracking, from the response's bare slug — which it cannot map, so it prints a red
+ # "Provider List: ..." banner around calls that succeeded. Nine of them in a 38-line run made
+ # a healthy loop read as broken. This suppresses the hint only; real errors still raise.
+ litellm.suppress_debug_info = True
litellm.success_callback = [_track_reflection]
base = teacher_base_url()
diff --git a/optimize/routing.py b/ingot/optimize/routing.py
similarity index 96%
rename from optimize/routing.py
rename to ingot/optimize/routing.py
index 0874e19..f608b22 100644
--- a/optimize/routing.py
+++ b/ingot/optimize/routing.py
@@ -12,17 +12,18 @@
import yaml
-from mcp_server.registry import SKILLS_DIR, load_skills, optimizable_components, skill_revision
-from mcp_server.router import Router
+from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision
+from ingot.mcp_server.router import Router
+from . import resolve_skill_dir
from . import usage as usage_ledger
from .ab import COLLISION_SCORE, TASKS_DIR, _description_shadows
from .evidence import RoutingRun, build_routing_evidence, recorded_path, write_evidence
from .promote import save_pending
from .rollout import make_reflection_lm
from . import skillopt_bridge as sk
+from ingot import paths
-EVIDENCE_DIR = Path(__file__).resolve().parent.parent / "runs" / "evidence"
_DIAGNOSIS = ("The `description` is a routing trigger matched by embedding similarity against the "
"user's task. Adjust trigger phrases so expected tasks match and unrelated ones "
@@ -161,9 +162,7 @@ def routing_gate(skill: str, metrics: dict, challenger: dict) -> tuple[bool, lis
def run_routing(skill: str, budget: int = 60, log=print) -> dict:
usage_ledger.reset()
- skill_dir = SKILLS_DIR / skill
- if not (skill_dir / "SKILL.md").exists():
- raise SystemExit(f"No skill named '{skill}' in skills/.")
+ skill_dir = resolve_skill_dir(skill)
tasks_path = TASKS_DIR / f"{skill}.yaml"
cases = (yaml.safe_load(tasks_path.read_text()) or {}).get("routing") if tasks_path.exists() else None
champion = optimizable_components(skill_dir)
@@ -207,7 +206,7 @@ def run_routing(skill: str, budget: int = 60, log=print) -> dict:
skill=skill, created=created, dataset=dataset, metrics=metrics,
champion_revision=current.revision, challenger_revision=challenger_revision,
inner_loop=inner_loop, gate=gate))
- evidence_json, evidence_markdown = write_evidence(evidence, EVIDENCE_DIR / skill / str(created))
+ evidence_json, evidence_markdown = write_evidence(evidence, paths.runs() / "evidence" / skill / str(created))
log(f"[ci] routing evidence: {evidence_json} and {evidence_markdown}")
pending = {
diff --git a/optimize/routing_health.py b/ingot/optimize/routing_health.py
similarity index 89%
rename from optimize/routing_health.py
rename to ingot/optimize/routing_health.py
index 4ec21a3..956ed3c 100644
--- a/optimize/routing_health.py
+++ b/ingot/optimize/routing_health.py
@@ -6,18 +6,18 @@
plus a pairwise description-collision scan at the same COLLISION_SCORE cutoff the promotion gate
uses. Read-only; it proposes nothing and never touches the review queue.
-Usage: python -m optimize.routing_health [skill ...] (default: every skill with routing cases)
+Usage: python -m ingot.optimize.routing_health [skill ...] (default: every skill with routing cases)
Exit status is non-zero when any suite case fails or any collision is found, so it can run
-unattended from cron or CI: docker compose run --rm optimize python -m optimize.routing_health
+unattended from cron or CI: docker compose run --rm optimize python -m ingot.optimize.routing_health
"""
import argparse
from pathlib import Path
import yaml
-from mcp_server.registry import load_skills
-from mcp_server.router import Router
-from mcp_server.routing_eval import evaluate_cases, evaluate_parity
+from ingot.mcp_server.registry import load_skills
+from ingot.mcp_server.router import Router
+from ingot.mcp_server.routing_eval import evaluate_cases, evaluate_parity
from .ab import COLLISION_SCORE, TASKS_DIR
@@ -82,14 +82,14 @@ def run_health(skills: list[str] | None = None, log=print) -> list[str]:
for p in problems:
log(f"[health] - {p}")
log("[health] fix: refine the colliding/regressed description by hand, or run the "
- "routing pass (python -m optimize.ab --description) and review the result.")
+ "routing pass (python -m ingot.optimize.ab --description) and review the result.")
else:
log("\n[health] ✓ routing healthy: every suite passes and no descriptions collide.")
return problems
if __name__ == "__main__":
- ap = argparse.ArgumentParser(prog="python -m optimize.routing_health")
+ ap = argparse.ArgumentParser(prog="python -m ingot.optimize.routing_health")
ap.add_argument("skills", nargs="*",
help="skills to check (default: every skill with an eval task set)")
raise SystemExit(1 if run_health(ap.parse_args().skills or None) else 0)
diff --git a/optimize/sandbox_driver.py b/ingot/optimize/sandbox_driver.py
similarity index 100%
rename from optimize/sandbox_driver.py
rename to ingot/optimize/sandbox_driver.py
diff --git a/optimize/skillopt_bridge.py b/ingot/optimize/skillopt_bridge.py
similarity index 100%
rename from optimize/skillopt_bridge.py
rename to ingot/optimize/skillopt_bridge.py
diff --git a/optimize/skillopt_loop.py b/ingot/optimize/skillopt_loop.py
similarity index 98%
rename from optimize/skillopt_loop.py
rename to ingot/optimize/skillopt_loop.py
index c5e8f96..fde2b98 100644
--- a/optimize/skillopt_loop.py
+++ b/ingot/optimize/skillopt_loop.py
@@ -2,7 +2,7 @@
adapted to skill_router's rollout + judge.
Treats the skill `body` as trainable state and improves it with SkillOpt's four disciplines,
-each driven through `optimize.skillopt_bridge` (the single dependency seam):
+each driven through `ingot.optimize.skillopt_bridge` (the single dependency seam):
1. a step buffer of prior failures + *rejected* edits, fed back into each reflection so the
optimizer stops re-proposing what the gate already threw out;
@@ -14,7 +14,7 @@
Drop-in for the inner loop: `run_skillopt(seed, tasks, frozen, ...) -> (best_components, seed_score,
best_score)`, scores being the penalized mean judge over the train set (comparable to what the outer
-A/B logs). The leakage-clean held-out promotion gate in optimize.ab is unchanged and remains the
+A/B logs). The leakage-clean held-out promotion gate in ingot.optimize.ab is unchanged and remains the
promotion authority; this only replaces how the challenger body is produced.
"""
import os
diff --git a/optimize/skillopt_prompts/README.md b/ingot/optimize/skillopt_prompts/README.md
similarity index 100%
rename from optimize/skillopt_prompts/README.md
rename to ingot/optimize/skillopt_prompts/README.md
diff --git a/optimize/skillopt_prompts/analyst_error.md b/ingot/optimize/skillopt_prompts/analyst_error.md
similarity index 100%
rename from optimize/skillopt_prompts/analyst_error.md
rename to ingot/optimize/skillopt_prompts/analyst_error.md
diff --git a/optimize/skillopt_prompts/lr_autonomous.md b/ingot/optimize/skillopt_prompts/lr_autonomous.md
similarity index 100%
rename from optimize/skillopt_prompts/lr_autonomous.md
rename to ingot/optimize/skillopt_prompts/lr_autonomous.md
diff --git a/optimize/skillopt_prompts/ranking.md b/ingot/optimize/skillopt_prompts/ranking.md
similarity index 100%
rename from optimize/skillopt_prompts/ranking.md
rename to ingot/optimize/skillopt_prompts/ranking.md
diff --git a/optimize/skillopt_prompts/slow_update.md b/ingot/optimize/skillopt_prompts/slow_update.md
similarity index 100%
rename from optimize/skillopt_prompts/slow_update.md
rename to ingot/optimize/skillopt_prompts/slow_update.md
diff --git a/optimize/tasks/.gitkeep b/ingot/optimize/tasks/.gitkeep
similarity index 100%
rename from optimize/tasks/.gitkeep
rename to ingot/optimize/tasks/.gitkeep
diff --git a/ingot/optimize/tree.py b/ingot/optimize/tree.py
new file mode 100644
index 0000000..67b76b0
--- /dev/null
+++ b/ingot/optimize/tree.py
@@ -0,0 +1,272 @@
+"""The exact bytes of an admitted package, staged for publication.
+
+Admission used to reduce a package to a dictionary of decoded text components. Anything that
+dictionary could not hold -- an image, a PDF, a wheel, a file whose bytes are not valid UTF-8 --
+was dropped on the way in, and the revision the reviewer approved then named a package that was
+missing them. A content-addressed release controller has exactly one promise to keep: a revision
+names the exact package. Dropping a file breaks it silently, which is the worst way to break it.
+
+So a candidate is a *tree*, not a dictionary. Every regular file is copied byte-for-byte into a
+staging directory named by the tree's digest, and the manifest records each file's relative path,
+mode, size, and SHA-256 of its raw bytes. Publication copies that staged tree into the vault
+worktree and verifies every hash on the way. The reviewer, the receipt, and the vault are then
+looking at the same bytes.
+
+Two deliberate exceptions, both visible rather than silent:
+
+- **SKILL.md is normalized, not preserved.** Its frontmatter is the routing interface: the name is
+ forced to the skill's identity, the description is collapsed to one line, and the whole thing is
+ re-emitted through a safe YAML dump so a stray `---` in a model's output cannot corrupt it. The
+ manifest still records the source file's real hash and size, so the normalization is auditable,
+ but what gets served is the normalized file. `revision()` below hashes the result of doing
+ exactly that, so the approved revision is the served revision.
+- **Symlinks are refused.** Preserving one means the vault commits a link that a reader follows out
+ of the library; flattening one into its target silently changes the artifact's shape. Neither is
+ a decision admission should make on an operator's behalf, so a package containing a symlink is
+ refused by name until there is a reason to build one of them.
+
+Modes are clamped to 0o644 or 0o755 because those are the only two a Git checkout reproduces, and
+the vault is a Git repository. Recording the raw mode would describe bytes the vault cannot serve.
+"""
+from __future__ import annotations
+
+import hashlib
+import json
+import os
+import shutil
+import tempfile
+import unicodedata
+import uuid
+from pathlib import Path, PurePosixPath
+
+from ingot import paths
+
+TREE_SCHEMA = "ingot/candidate-tree/v1"
+
+# Bounds on what admission will stage. The byte limit does the real work; the file count stops a
+# package of a million empty files from turning one approval into a filesystem problem. Both sit
+# above the advisory thresholds `ingot review` warns at, so a package it merely warns about is
+# still admissible.
+MAX_FILES = 256
+MAX_TREE_BYTES = 20_000_000
+
+_CHUNK = 1 << 20
+_WINDOWS_RESERVED = {"con", "prn", "aux", "nul", *(f"com{i}" for i in range(1, 10)),
+ *(f"lpt{i}" for i in range(1, 10))}
+
+
+def candidates_dir() -> Path:
+ """Staged candidate trees. Resolved per call, never bound at import."""
+ return paths.runs() / "candidates"
+
+
+def staged_dir(digest: str) -> Path:
+ if not isinstance(digest, str) or len(digest) != 64 or not all(
+ char in "0123456789abcdef" for char in digest):
+ raise ValueError(f"invalid candidate tree digest: {digest!r}")
+ return candidates_dir() / digest
+
+
+def portable_path(raw: str, *, allow_skill_md: bool = False) -> PurePosixPath:
+ """One relative path every filesystem in the deployment agrees on.
+
+ Shared with the text-component path in `ingot.optimize.ingress` so a package cannot enter through one
+ door what the other would refuse."""
+ if not isinstance(raw, str) or "\\" in raw or any(ord(char) < 32 for char in raw):
+ raise ValueError(f"component is not a portable POSIX path: {raw}")
+ path = PurePosixPath(raw)
+ parts = path.parts
+ forbidden = {"."} if allow_skill_md else {".", "SKILL.md"}
+ if path.is_absolute() or ".." in parts or path.as_posix() in forbidden:
+ raise ValueError(f"component escapes skill root: {raw}")
+ if any(":" in part or part.endswith((".", " ")) or
+ part.split(".", 1)[0].casefold() in _WINDOWS_RESERVED for part in parts):
+ raise ValueError(f"component is not a portable POSIX path: {raw}")
+ if any(unicodedata.normalize("NFC", part) != part for part in parts):
+ raise ValueError(f"component path must be NFC-normalized: {raw}")
+ return path
+
+
+def _hash(path: Path) -> tuple[str, int]:
+ """SHA-256 of the raw bytes, and the byte count. Never a decoded string: a file that is not
+ valid UTF-8 has no decoded form, and one that is would hash differently after a round trip."""
+ digest, size = hashlib.sha256(), 0
+ with path.open("rb") as handle:
+ while chunk := handle.read(_CHUNK):
+ digest.update(chunk)
+ size += len(chunk)
+ return digest.hexdigest(), size
+
+
+def _mode(raw: int) -> int:
+ return 0o755 if raw & 0o111 else 0o644
+
+
+def _digest(files: list[dict]) -> str:
+ canonical = json.dumps({"schema_version": TREE_SCHEMA, "files": files},
+ sort_keys=True, separators=(",", ":"), ensure_ascii=False)
+ return hashlib.sha256(canonical.encode()).hexdigest()
+
+
+def _walk(root: Path, current: Path) -> list[Path]:
+ """Every regular file under `current`, refusing anything that is not one.
+
+ An explicit descent rather than `rglob`, which follows directory symlinks: a package could
+ otherwise pull in a whole tree from outside itself and this would never see the link."""
+ found = []
+ for item in sorted(current.iterdir()):
+ relative = item.relative_to(root).as_posix()
+ if item.is_symlink():
+ raise ValueError(f"symlinks are not admissible: {relative}")
+ if item.is_dir():
+ found.extend(_walk(root, item))
+ elif item.is_file():
+ found.append(item)
+ else:
+ raise ValueError(f"not a regular file: {relative}")
+ return found
+
+
+def build(package: Path) -> dict:
+ """Describe every regular file in `package` exactly as it will be served."""
+ package = Path(package).resolve()
+ files, total, folded = [], 0, set()
+ for path in _walk(package, package):
+ raw = path.relative_to(package).as_posix()
+ relative = portable_path(raw, allow_skill_md=True).as_posix()
+ if relative.casefold() in folded:
+ raise ValueError(f"component path collides case-insensitively: {raw}")
+ folded.add(relative.casefold())
+ checksum, size = _hash(path)
+ total += size
+ files.append({"path": relative, "mode": _mode(path.stat().st_mode),
+ "size": size, "sha256": checksum})
+
+ if not files:
+ raise ValueError("the package holds no files")
+ if len(files) > MAX_FILES:
+ raise ValueError(f"a package may hold at most {MAX_FILES} files; this one holds "
+ f"{len(files)}")
+ if total > MAX_TREE_BYTES:
+ raise ValueError(f"a package may hold at most {MAX_TREE_BYTES} bytes; this one holds "
+ f"{total}")
+ files.sort(key=lambda entry: entry["path"])
+ return {"schema_version": TREE_SCHEMA, "files": files, "size": total,
+ "digest": _digest(files)}
+
+
+def verify_manifest(manifest: object) -> dict:
+ """The manifest is well formed and its digest covers the file list it arrived with.
+
+ Recomputed rather than trusted, because the digest is what binds a receipt to a staged tree: a
+ receipt whose file list was edited without its digest would otherwise publish a tree nobody
+ approved."""
+ if not isinstance(manifest, dict) or manifest.get("schema_version") != TREE_SCHEMA:
+ raise ValueError("unsupported candidate tree manifest")
+ files = manifest.get("files")
+ if not isinstance(files, list) or not files:
+ raise ValueError("a candidate tree manifest lists at least one file")
+ for entry in files:
+ if not isinstance(entry, dict):
+ raise ValueError("candidate tree entries must be objects")
+ if set(entry) != {"path", "mode", "size", "sha256"}:
+ raise ValueError(f"unexpected candidate tree entry: {sorted(entry)}")
+ portable_path(entry["path"], allow_skill_md=True)
+ if entry["mode"] not in (0o644, 0o755):
+ raise ValueError(f"unsupported mode for {entry['path']}: {entry['mode']}")
+ if not isinstance(entry["size"], int) or isinstance(entry["size"], bool) \
+ or entry["size"] < 0:
+ raise ValueError(f"invalid size for {entry['path']}")
+ if _digest(files) != manifest.get("digest"):
+ raise ValueError("candidate tree digest does not cover its file list")
+ return manifest
+
+
+def stage(package: Path, manifest: dict) -> Path:
+ """Copy the described bytes somewhere the publisher can reach them.
+
+ The source directory belongs to whoever ran `ingot add`; it can be edited or deleted between
+ the approval and the publication, and a publisher that read from it would publish whatever it
+ found. Staging is named by the tree digest, so ingesting identical bytes twice converges on one
+ directory instead of racing."""
+ package = Path(package).resolve()
+ destination = staged_dir(manifest["digest"])
+ if destination.is_dir():
+ return destination
+
+ candidates_dir().mkdir(parents=True, exist_ok=True, mode=0o700)
+ temporary = candidates_dir() / f".{manifest['digest']}.{uuid.uuid4().hex}.tmp"
+ try:
+ for entry in manifest["files"]:
+ target = temporary / entry["path"]
+ target.parent.mkdir(parents=True, exist_ok=True)
+ shutil.copyfile(package / entry["path"], target)
+ os.chmod(target, entry["mode"])
+ checksum, size = _hash(target)
+ if (checksum, size) != (entry["sha256"], entry["size"]):
+ raise ValueError(f"{entry['path']} changed while it was being staged")
+ try:
+ os.rename(temporary, destination)
+ except OSError:
+ # Another submitter staged the identical tree first. Same digest, same bytes.
+ if not destination.is_dir():
+ raise
+ shutil.rmtree(temporary, ignore_errors=True)
+ except BaseException:
+ shutil.rmtree(temporary, ignore_errors=True)
+ raise
+ return destination
+
+
+def materialize(manifest: dict, destination: Path) -> None:
+ """Write the staged tree into `destination`, checking every file against the manifest."""
+ verify_manifest(manifest)
+ source = staged_dir(manifest["digest"])
+ if not source.is_dir():
+ raise ValueError(f"the staged candidate tree {manifest['digest'][:16]} is missing")
+ destination = Path(destination)
+ destination.mkdir(parents=True, exist_ok=True)
+ for entry in manifest["files"]:
+ relative = portable_path(entry["path"], allow_skill_md=True)
+ staged = source / relative
+ checksum, size = _hash(staged)
+ if (checksum, size) != (entry["sha256"], entry["size"]):
+ raise ValueError(f"staged candidate file does not match the receipt: {entry['path']}")
+ target = destination / relative
+ target.parent.mkdir(parents=True, exist_ok=True)
+ shutil.copyfile(staged, target)
+ os.chmod(target, entry["mode"])
+
+
+def materialize_creation(manifest: dict, components: dict, skill: str, destination: Path) -> None:
+ """The whole tree, then the normalized SKILL.md over the top.
+
+ The one implementation of what an admitted package becomes. `revision()` runs it into a
+ temporary directory to compute the revision a reviewer approves, and the publisher runs it into
+ the vault worktree to produce the revision the library serves. They cannot disagree, because
+ there is nothing here for them to disagree about."""
+ from ingot.mcp_server.registry import write_skill_md
+
+ materialize(manifest, destination)
+ try:
+ metadata = json.loads(components.get("frontmatter") or "{}")
+ except (TypeError, ValueError) as exc:
+ raise ValueError("creation frontmatter is not valid JSON") from exc
+ if not isinstance(metadata, dict):
+ raise ValueError("creation frontmatter is not an object")
+ metadata["name"] = skill
+ metadata["description"] = components["description"]
+ write_skill_md(Path(destination) / "SKILL.md", metadata, components["body"])
+
+
+def revision(skill: str, manifest: dict, components: dict) -> str:
+ """The revision the library will serve once this tree is published.
+
+ Computed by materializing it, because a revision derived some other way would be a second
+ description of the same bytes and the two would eventually disagree."""
+ from ingot.mcp_server.registry import skill_revision
+
+ with tempfile.TemporaryDirectory() as temporary:
+ root = Path(temporary) / skill
+ materialize_creation(manifest, components, skill, root)
+ return skill_revision(root)
diff --git a/ingot/optimize/usage.py b/ingot/optimize/usage.py
new file mode 100644
index 0000000..1205edf
--- /dev/null
+++ b/ingot/optimize/usage.py
@@ -0,0 +1,162 @@
+"""Token ledger for an optimize run: every LLM call is attributed to a role
+(rollout / judge / reflection / agent_ab) so the run can report what it actually cost ,
+including a best-effort USD estimate from OpenRouter list prices, and an optional hard
+spend cap (MAX_RUN_USD) that aborts a run before it exceeds the budget."""
+import os
+import threading
+from collections import defaultdict
+
+COUNTS: dict[str, dict[str, int]] = defaultdict(lambda: {"input": 0, "output": 0, "calls": 0})
+_SUBSCRIPTION_COUNTS: dict[str, dict[str, int]] = defaultdict(
+ lambda: {"input": 0, "output": 0, "calls": 0}
+)
+_LOCK = threading.RLock() # the search fans rollout+judge across a thread pool; add() re-enters for the cap
+_PRICES: dict[str, tuple[float, float]] | None = None
+
+
+def reset():
+ """Start a fresh ledger, the UI process runs many optimizations; counts must not leak across runs."""
+ with _LOCK:
+ COUNTS.clear()
+ _SUBSCRIPTION_COUNTS.clear()
+
+
+def add(role: str, usage: dict | None, *, billing_mode: str = "metered"):
+ """usage: langchain usage_metadata ({'input_tokens','output_tokens'}) or equivalent dict."""
+ if not usage:
+ return
+ if billing_mode not in {"metered", "subscription"}:
+ raise ValueError(f"unknown billing mode: {billing_mode}")
+ with _LOCK:
+ c = COUNTS[role]
+ c["input"] += int(usage.get("input_tokens", 0))
+ c["output"] += int(usage.get("output_tokens", 0))
+ c["calls"] += 1
+ if billing_mode == "subscription":
+ subscription = _SUBSCRIPTION_COUNTS[role]
+ subscription["input"] += int(usage.get("input_tokens", 0))
+ subscription["output"] += int(usage.get("output_tokens", 0))
+ subscription["calls"] += 1
+ if billing_mode == "metered":
+ _enforce_cap()
+
+
+def _enforce_cap():
+ cap = float(os.environ.get("MAX_RUN_USD", "0") or 0)
+ if not cap:
+ return
+ cost = estimated_cost()
+ if cost is not None and cost > cap:
+ raise SystemExit(f"MAX_RUN_USD exceeded: estimated ${cost:.2f} > cap ${cap:.2f}, "
+ f"aborting before spending more.\n{format_report()}")
+
+
+def _openrouter_prices() -> dict[str, tuple[float, float]]:
+ """model id -> (prompt, completion) USD per token from OpenRouter's public models API;
+ {} on any failure (cost reporting is best-effort, never a gate on offline work)."""
+ import json
+ import urllib.request
+ try:
+ with urllib.request.urlopen("https://openrouter.ai/api/v1/models", timeout=10) as r:
+ data = json.loads(r.read())["data"]
+ return {m["id"]: (float(m["pricing"]["prompt"]), float(m["pricing"]["completion"]))
+ for m in data if m.get("pricing")}
+ except Exception:
+ return {}
+
+
+def _role_models() -> dict[str, str]:
+ """Which model each ledger role runs on (first judge only, for ensemble setups)."""
+ from . import agent_model, skillopt_model
+ teacher = skillopt_model()
+ judge = (os.environ.get("JUDGE_MODELS") or
+ os.environ.get("JUDGE_MODEL", "google/gemini-2.5-flash")).split(",")[0].strip()
+ return {"rollout": agent_model(), "agent_ab": agent_model(), "review": agent_model(),
+ "judge": judge, "reflection": teacher, "draft": teacher}
+
+
+def _model_for(role: str) -> str:
+ """The model a ledger role ran on. A role may name its own model after a colon
+ (`compat:anthropic/claude-sonnet-4.5`): the compatibility sweep varies the *serving* model by
+ design, so unlike the fixed roles it cannot be mapped to one model up front."""
+ name, _, explicit = role.partition(":")
+ return explicit or _role_models().get(name, "")
+
+
+def _metered_counts() -> dict[str, dict[str, int]]:
+ """Per-role usage remaining after subscription-backed calls are removed."""
+ with _LOCK:
+ return {
+ role: {key: count - _SUBSCRIPTION_COUNTS[role][key] for key, count in counts.items()}
+ for role, counts in COUNTS.items()
+ }
+
+
+def unpriced_roles() -> list[str]:
+ """Roles that spent tokens but contribute nothing to the estimate, because their model carries
+ no OpenRouter list price — a local endpoint (genuinely free) or a slug we could not resolve
+ (not free at all). Surfaced rather than swallowed: an unpriced role silently counts as $0, and
+ that is how the entire compatibility sweep once vanished from both the cost line and the
+ MAX_RUN_USD cap, reporting $0.04 against $1.42 actually spent."""
+ if _PRICES is None:
+ return []
+ return sorted(role for role, c in _metered_counts().items()
+ if (c["input"] or c["output"]) and not _PRICES.get(_model_for(role)))
+
+
+def estimated_cost() -> float | None:
+ """Best-effort USD estimate for the current ledger, from OpenRouter list prices. None when
+ the endpoint isn't OpenRouter or pricing is unavailable (local endpoints cost nothing).
+ Roles whose model has no list price contribute nothing — ask unpriced_roles() which those are
+ before trusting this as a total."""
+ global _PRICES
+ from . import is_openrouter, teacher_base_url
+ if not is_openrouter(teacher_base_url()):
+ return None
+ if _PRICES is None:
+ _PRICES = _openrouter_prices()
+ if not _PRICES:
+ return None
+ metered = _metered_counts()
+ if not any(c["input"] or c["output"] for c in metered.values()):
+ return None
+ return sum(c["input"] * p[0] + c["output"] * p[1]
+ for role, c in metered.items()
+ if (p := _PRICES.get(_model_for(role))))
+
+
+def report() -> dict:
+ out = {role: dict(c) for role, c in COUNTS.items()}
+ out["total"] = {
+ "input": sum(c["input"] for c in COUNTS.values()),
+ "output": sum(c["output"] for c in COUNTS.values()),
+ "calls": sum(c["calls"] for c in COUNTS.values()),
+ }
+ return out
+
+
+def format_report() -> str:
+ r = report()
+ with _LOCK:
+ subscriptions = {role: dict(c) for role, c in _SUBSCRIPTION_COUNTS.items()}
+ lines = []
+ for role, c in r.items():
+ if role == "total":
+ continue
+ subscription = subscriptions.get(role)
+ suffix = (f" ({subscription['calls']} subscription calls, "
+ f"{subscription['input']:,} in, {subscription['output']:,} out)"
+ if subscription and subscription["calls"] else "")
+ lines.append(f" {role:<12} {c['calls']:>4} calls {c['input']:>9,} in "
+ f"{c['output']:>8,} out{suffix}")
+ t = r["total"]
+ lines.append(f" {'TOTAL':<12} {t['calls']:>4} calls {t['input']:>9,} in {t['output']:>8,} out")
+ cost = estimated_cost()
+ if cost is not None:
+ lines.append(f" estimated cost: ${cost:.2f} (OpenRouter list prices)")
+ if (blind := unpriced_roles()):
+ lines.append(f" NOT in that estimate: {', '.join(blind)} — no list price for "
+ f"{', '.join(sorted({_model_for(r) or '?' for r in blind}))}. "
+ f"Free if that is a local endpoint; otherwise the figure above is low, "
+ f"and so is any MAX_RUN_USD cap resting on it.")
+ return "\n".join(lines)
diff --git a/ingot/parse.py b/ingot/parse.py
new file mode 100644
index 0000000..2ff58c9
--- /dev/null
+++ b/ingot/parse.py
@@ -0,0 +1,67 @@
+"""A SKILL.md parser that reports what is wrong instead of repairing it.
+
+`ingot.mcp_server.registry.parse_skill` is deliberately tolerant: absent frontmatter, unparseable YAML,
+and a frontmatter that is not a mapping all normalize to empty metadata so the server keeps
+serving. That is correct for serving and useless for diagnosis -- silent normalization is not a
+diagnostic API. This parser answers the same question in the opposite direction: it keeps the
+malformed input and returns findings.
+
+The two must agree on the shape of a well-formed document, so the frontmatter pattern is imported
+from the serving parser rather than restated here."""
+from __future__ import annotations
+
+from dataclasses import dataclass, field
+
+import yaml
+
+from ingot.mcp_server.registry import _FRONTMATTER
+
+ERROR = "error"
+WARNING = "warning"
+INFO = "info"
+
+
+@dataclass(frozen=True)
+class Finding:
+ code: str
+ level: str
+ message: str
+ path: str | None = None
+
+ def as_dict(self) -> dict:
+ found = {"code": self.code, "level": self.level, "message": self.message}
+ if self.path is not None:
+ found["path"] = self.path
+ return found
+
+
+@dataclass
+class RawSkill:
+ """`frontmatter` is None when the document has none that can be read as a mapping. That is the
+ distinction the serving parser erases, and every caller here depends on it."""
+ frontmatter: dict | None
+ body: str
+ findings: list[Finding] = field(default_factory=list)
+
+
+def parse_raw(text: str) -> RawSkill:
+ match = _FRONTMATTER.match(text)
+ if not match:
+ return RawSkill(None, text.strip(), [Finding(
+ "frontmatter-missing", ERROR,
+ "no YAML frontmatter: a SKILL.md must open with a '---' delimited block")])
+
+ try:
+ loaded = yaml.safe_load(match.group(1))
+ except yaml.YAMLError as error:
+ detail = str(error).replace("\n", " ").strip()
+ return RawSkill(None, match.group(2).strip(), [Finding(
+ "frontmatter-invalid", ERROR, f"frontmatter is not valid YAML: {detail}")])
+
+ if not isinstance(loaded, dict):
+ kind = type(loaded).__name__ if loaded is not None else "nothing"
+ return RawSkill(None, match.group(2).strip(), [Finding(
+ "frontmatter-not-a-mapping", ERROR,
+ f"frontmatter must be a mapping of fields, found {kind}")])
+
+ return RawSkill(loaded, match.group(2).strip(), [])
diff --git a/ingot/paths.py b/ingot/paths.py
new file mode 100644
index 0000000..473a1b7
--- /dev/null
+++ b/ingot/paths.py
@@ -0,0 +1,128 @@
+"""Where mutable state lives. The one place that answers, so no module derives it from `__file__`.
+
+State that was package-relative worked for exactly two deployments — a container and a checkout —
+and broke everywhere else: a read-only or system Python, an upgrade that replaces the package
+directory, a recreated environment, two deployments sharing one installation, and any backup that
+expects application code and controlled state to be separable. A release controller that keeps its
+own review queue inside `site-packages` cannot claim to control anything.
+
+Every function reads the environment when called. Modules that still expose a module-level constant
+derive it from here, so the default is right for an installation; the resolution rule lives here.
+"""
+from __future__ import annotations
+
+import os
+from pathlib import Path
+
+HOME = "INGOT_HOME"
+LIBRARY = "INGOT_LIBRARY"
+RUNS = "INGOT_RUNS"
+TASKS = "INGOT_TASKS"
+VAULT = "INGOT_VAULT_PATH"
+# Names that predate INGOT_HOME. Honoured so an existing deployment keeps working, and reported by
+# `ingot status` so it is visible rather than load-bearing and forgotten.
+LEGACY = {LIBRARY: "SKILLS_DIR", VAULT: "VAULT_DIR"}
+
+PACKAGE_ROOT = Path(__file__).resolve().parent.parent
+
+
+def _env(name: str, env: dict | None = None) -> str:
+ env = os.environ if env is None else env
+ value = env.get(name) or env.get(LEGACY.get(name, ""), "")
+ return value.strip()
+
+
+def home(env: dict | None = None) -> Path:
+ """The root of everything mutable.
+
+ XDG on Unix by default, so a plain `pip install ingot` puts state somewhere a package upgrade
+ cannot take with it. Containers and managed deployments override it, or override each path
+ below individually."""
+ env = os.environ if env is None else env
+ explicit = _env(HOME, env)
+ if explicit:
+ return Path(explicit).expanduser()
+ state = (env.get("XDG_STATE_HOME") or "").strip()
+ if state:
+ return Path(state).expanduser() / "ingot"
+ return Path.home() / ".local" / "state" / "ingot"
+
+
+def _under(name: str, default: str, env: dict | None = None) -> Path:
+ explicit = _env(name, env)
+ return Path(explicit).expanduser() if explicit else home(env) / default
+
+
+def library(env: dict | None = None) -> Path:
+ """The served skill library."""
+ return _under(LIBRARY, "library", env)
+
+
+def runs(env: dict | None = None) -> Path:
+ """Review queue, publication receipts, evidence, snapshots, and audit trails."""
+ return _under(RUNS, "runs", env)
+
+
+def tasks(env: dict | None = None) -> Path:
+ """Held-out evaluation task sets."""
+ return _under(TASKS, "tasks", env)
+
+
+def vault(env: dict | None = None) -> Path:
+ """The Git repository the publisher owns.
+
+ Defaults to the library, because in the local backend the served checkout *is* the vault. A
+ deployment that wants them separate says so."""
+ explicit = _env(VAULT, env)
+ return Path(explicit).expanduser() if explicit else library(env)
+
+
+def _source(name: str, env: dict | None = None) -> str:
+ """Which setting decided a path, for a report that has to be checkable rather than believed."""
+ env = os.environ if env is None else env
+ if env.get(name):
+ return name
+ legacy = LEGACY.get(name)
+ if legacy and env.get(legacy):
+ return f"{legacy} (deprecated; prefer {name})"
+ if env.get(HOME):
+ return HOME
+ if name == VAULT:
+ return _source(LIBRARY, env)
+ return "XDG_STATE_HOME" if env.get("XDG_STATE_HOME") else "default"
+
+
+def resolved(env: dict | None = None) -> list[dict]:
+ """Every path this process would use, where it came from, and whether it can be written."""
+ entries = [("home", home(env), HOME), ("library", library(env), LIBRARY),
+ ("runs", runs(env), RUNS), ("tasks", tasks(env), TASKS),
+ ("vault", vault(env), VAULT)]
+ return [{"name": name, "path": str(path), "source": _source(setting, env),
+ "exists": path.exists(),
+ # An absent path is not a problem: it is created on first write. An absent path whose
+ # parent cannot be written is, and reporting only `exists` would hide it.
+ "writable": os.access(path if path.exists() else _nearest(path), os.W_OK | os.X_OK)}
+ for name, path, setting in entries]
+
+
+def _nearest(path: Path) -> Path:
+ for candidate in path.parents:
+ if candidate.exists():
+ return candidate
+ return path
+
+
+def legacy_state() -> list[str]:
+ """Package-relative state directories left by a version that kept them next to the code.
+
+ Reported, never migrated. Moving a review queue or a receipt store on someone's behalf is a
+ change to controlled state made by a process that was not asked to make it."""
+ found = []
+ for name in ("runs", "skills"):
+ directory = PACKAGE_ROOT / name
+ if not directory.is_dir():
+ continue
+ contents = [item for item in directory.iterdir() if item.name != ".gitkeep"]
+ if contents:
+ found.append(str(directory))
+ return found
diff --git a/ingot/records.py b/ingot/records.py
new file mode 100644
index 0000000..e3fbabd
--- /dev/null
+++ b/ingot/records.py
@@ -0,0 +1,224 @@
+"""The two versioned records the control plane is built on.
+
+A **candidate manifest** names exactly what is being proposed, where it came from, and what the
+deterministic review said about it. A **release receipt** names exactly what was published, against
+which champion, on whose authority.
+
+Three properties matter more than the field list:
+
+- **A revision names exact bytes.** Every revision here is a content digest or the absence marker.
+ A semantic version is not accepted anywhere: a tag can be moved and a digest cannot.
+- **Candidate identity is deterministic.** Two submissions of the same bytes, from different
+ checkouts at different times, are the same candidate. That is what makes an identical resubmission
+ idempotent instead of a duplicate proposal.
+- **A receipt is not a proof.** There is no signature field and there will not be one until there is
+ a threat model. A local record that a machine administrator can rewrite must not carry anything
+ shaped like evidence that they did not.
+
+Consumers: `ingot add` writes candidate manifests; the publisher writes release receipts once
+publication succeeds -- never when approval merely queues it.
+"""
+from __future__ import annotations
+
+import hashlib
+import json
+import re
+
+from ingot.mcp_server.registry import SLUG_RE
+
+CANDIDATE_SCHEMA = "ingot/candidate/v1"
+RELEASE_SCHEMA = "ingot/release/v1"
+
+# Absence is a revision. A creation displaces nothing; a rollback can restore nothing. The existing
+# publisher already treats it this way (optimize/promote.py ABSENT_REVISION), and the two spellings
+# must not drift apart.
+ABSENT_REVISION = "absent"
+
+SOURCE_TYPES = frozenset({"file", "github", "optimizer", "mcp"})
+CANDIDATE_KINDS = frozenset({"creation", "update"})
+ACTIONS = frozenset({"promote", "rollback"})
+RESULTS = frozenset({"published", "failed"})
+
+_DIGEST = re.compile(r"^[0-9a-f]{64}$")
+
+# Kept for file candidates created before GitHub acquisition. Their proposal IDs already include
+# the complete bound review record except circumstance metadata, and changing that basis would
+# turn an identical resubmission into a slot conflict during upgrade.
+_NOT_IDENTITY = ("created_at", "actor")
+
+def digest(payload: object) -> str:
+ """SHA-256 over a canonical JSON encoding. Key order must not change the answer."""
+ canonical = json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
+ return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
+
+
+def _is_revision(value: object, *, allow_absent: bool = True) -> bool:
+ if allow_absent and value == ABSENT_REVISION:
+ return True
+ return isinstance(value, str) and bool(_DIGEST.fullmatch(value))
+
+
+def candidate_manifest(*, kind: str, skill: str, source_type: str, locator: str,
+ resolved_revision: str, candidate_revision: str, review: dict,
+ created_at: int, provenance: dict | None = None) -> dict:
+ """Build a candidate manifest. Validation is separate -- callers validate what they build."""
+ return {
+ "schema_version": CANDIDATE_SCHEMA,
+ "kind": kind,
+ "skill": skill,
+ "source": {**(provenance or {}),
+ "type": source_type,
+ "locator": locator,
+ "resolved_revision": resolved_revision},
+ "candidate_revision": candidate_revision,
+ # `digest` binds this summary; `report_digest`, when the caller supplies one, binds the full
+ # report the summary was drawn from. Both travel, so neither the summary nor the report it
+ # came from can be edited afterwards without the manifest noticing.
+ "review": {"schema_version": review.get("schema_version"),
+ "digest": digest(review),
+ "valid": review.get("valid"),
+ "errors": list(review.get("errors") or []),
+ "warnings": list(review.get("warnings") or []),
+ **({"report_digest": review["report_digest"]}
+ if review.get("report_digest") else {})},
+ "created_at": created_at,
+ }
+
+
+def candidate_identity(manifest: dict) -> str:
+ """The digest that decides whether two submissions are the same candidate.
+
+ Drops timestamps, actor metadata, and the local path the package happened to be read from.
+ Keeps the skill, the kind, the source type, both revisions, and the review outcome -- a review
+ is evidence, and evidence is revision-bound, so the same bytes reviewed clean and reviewed with
+ errors are not interchangeable proposals."""
+ source = manifest.get("source") or {}
+ review = manifest.get("review") or {}
+ identity_review = ({key: review.get(key) for key in
+ ("schema_version", "valid", "errors", "warnings")}
+ if source.get("type") == "github" else
+ {key: value for key, value in review.items() if key not in _NOT_IDENTITY})
+ return digest({
+ "schema_version": manifest.get("schema_version"),
+ "kind": manifest.get("kind"),
+ "skill": manifest.get("skill"),
+ "source_type": source.get("type"),
+ "resolved_revision": source.get("resolved_revision"),
+ "candidate_revision": manifest.get("candidate_revision"),
+ # A GitHub report includes its temporary clone path. That path is provenance rather than
+ # outcome, so GitHub identity keeps the deterministic result while its manifest keeps both
+ # digests. File candidates retain the original formula above for upgrade compatibility.
+ "review": identity_review,
+ })
+
+
+def validate_candidate(manifest: dict) -> list[str]:
+ """Every problem with a manifest, or an empty list. Never raises: a caller reporting to a person
+ wants all of them at once, not the first."""
+ problems = []
+ if manifest.get("schema_version") != CANDIDATE_SCHEMA:
+ problems.append(f"schema_version must be {CANDIDATE_SCHEMA}, "
+ f"found {manifest.get('schema_version')!r}")
+ if manifest.get("kind") not in CANDIDATE_KINDS:
+ problems.append(f"kind must be one of {sorted(CANDIDATE_KINDS)}, "
+ f"found {manifest.get('kind')!r}")
+
+ skill = manifest.get("skill")
+ if not isinstance(skill, str) or not SLUG_RE.fullmatch(skill):
+ problems.append(f"skill must be a valid slug, found {skill!r}")
+
+ source = manifest.get("source")
+ if not isinstance(source, dict):
+ problems.append("source is missing")
+ else:
+ if source.get("type") not in SOURCE_TYPES:
+ problems.append(f"source.type must be one of {sorted(SOURCE_TYPES)}, "
+ f"found {source.get('type')!r}")
+ if not isinstance(source.get("locator"), str) or not source.get("locator"):
+ problems.append("source.locator is missing")
+ if not _is_revision(source.get("resolved_revision"), allow_absent=False):
+ problems.append("source.resolved_revision must be a content digest, not a version: "
+ f"found {source.get('resolved_revision')!r}")
+
+ if not _is_revision(manifest.get("candidate_revision"), allow_absent=False):
+ problems.append("candidate_revision must be a content digest, found "
+ f"{manifest.get('candidate_revision')!r}")
+
+ review = manifest.get("review")
+ if not isinstance(review, dict):
+ problems.append("review is missing")
+ elif not _is_revision(review.get("digest"), allow_absent=False):
+ problems.append("review.digest must be a content digest")
+
+ if not isinstance(manifest.get("created_at"), int):
+ problems.append("created_at must be an integer timestamp")
+ return problems
+
+
+def release_receipt(*, skill: str, action: str, proposal_id: str, publication_id: str,
+ expected_champion: str, candidate_revision: str, evidence_digests: list[str],
+ actor: str, publisher: str, target: str, published_at: int,
+ result: str, error: str | None = None) -> dict:
+ """Build a release receipt.
+
+ Emitted only once publication has actually succeeded or failed. Approval that merely queues a
+ publication is a different state and does not produce one of these -- a receipt that appeared at
+ approval time would claim a skill was released while the served bytes were unchanged."""
+ receipt = {
+ "schema_version": RELEASE_SCHEMA,
+ "skill": skill,
+ "action": action,
+ "proposal_id": proposal_id,
+ "publication_id": publication_id,
+ "expected_champion": expected_champion,
+ "candidate_revision": candidate_revision,
+ "evidence_digests": list(evidence_digests),
+ "actor": actor,
+ "publisher": publisher,
+ "target": target,
+ "published_at": published_at,
+ "result": result,
+ }
+ if error is not None:
+ receipt["error"] = error
+ return receipt
+
+
+def validate_release(receipt: dict) -> list[str]:
+ problems = []
+ if receipt.get("schema_version") != RELEASE_SCHEMA:
+ problems.append(f"schema_version must be {RELEASE_SCHEMA}, "
+ f"found {receipt.get('schema_version')!r}")
+
+ skill = receipt.get("skill")
+ if not isinstance(skill, str) or not SLUG_RE.fullmatch(skill):
+ problems.append(f"skill must be a valid slug, found {skill!r}")
+ if receipt.get("action") not in ACTIONS:
+ problems.append(f"action must be one of {sorted(ACTIONS)}, found {receipt.get('action')!r}")
+ if receipt.get("result") not in RESULTS:
+ problems.append(f"result must be one of {sorted(RESULTS)}, found {receipt.get('result')!r}")
+
+ for field in ("expected_champion", "candidate_revision"):
+ if not _is_revision(receipt.get(field)):
+ problems.append(f"{field} must be a content digest or {ABSENT_REVISION!r}, "
+ f"found {receipt.get(field)!r}")
+
+ for field in ("proposal_id", "publication_id", "actor", "publisher", "target"):
+ if not isinstance(receipt.get(field), str) or not receipt.get(field):
+ problems.append(f"{field} is missing")
+
+ if not isinstance(receipt.get("evidence_digests"), list):
+ problems.append("evidence_digests must be a list")
+ elif not all(_is_revision(item, allow_absent=False) for item in receipt["evidence_digests"]):
+ problems.append("every entry in evidence_digests must be a content digest")
+
+ if not isinstance(receipt.get("published_at"), int):
+ problems.append("published_at must be an integer timestamp")
+
+ # A failure that erased its reason leaves an operator with a stalled lane and nothing to read;
+ # a success carrying an error is a receipt that disagrees with itself.
+ if receipt.get("result") == "failed" and not receipt.get("error"):
+ problems.append("a failed receipt must carry an error explaining why")
+ if receipt.get("result") == "published" and receipt.get("error"):
+ problems.append("a published receipt must not carry an error")
+ return problems
diff --git a/ingot/review.py b/ingot/review.py
new file mode 100644
index 0000000..e99d12b
--- /dev/null
+++ b/ingot/review.py
@@ -0,0 +1,418 @@
+"""Deterministic, offline, read-only review of one skill package.
+
+Six sections, never one number. A composite score invites exactly the reward-hacking the evidence
+gate exists to prevent, and it hides which of six unrelated questions actually failed.
+
+What this command will not do:
+
+- run a model, read a key, reach the network, or start a service;
+- convert similarity to another skill into an "activation score" -- that measures collision, not
+ whether the router loads this skill at the right moment;
+- pattern-match for markers and call the result prompt-injection safety;
+- write anything, anywhere.
+
+Where a question needs evidence this command cannot produce, it reports UNMEASURED and names the
+command that can. An honest gap beats a confident guess."""
+from __future__ import annotations
+
+import codecs
+import json
+import re
+import unicodedata
+from pathlib import Path
+
+import yaml
+
+from ingot.mcp_server.registry import SLUG_RE, skill_revision, skill_sources
+
+from .parse import ERROR, INFO, WARNING, Finding, parse_raw
+
+REVIEW_SCHEMA = "ingot/review/v1"
+
+MEASURED = "measured"
+UNMEASURED = "unmeasured"
+
+# Thresholds are advisory and deliberately loose: they exist to surface a package that will surprise
+# someone, not to impose a house style.
+LARGE_BODY_BYTES = 40_000
+LARGE_PACKAGE_BYTES = 10_000_000
+MANY_FILES = 200
+
+# A relative link in the body that points at something in the package. Skips URLs and anchors.
+_LOCAL_LINK = re.compile(r"\[[^\]]*\]\(\s*(?!\w+:|#)([^)\s]+)")
+_URL = re.compile(r"https?://[^\s<>\"')\]]+")
+# Mutable by construction: a branch ref rather than a commit or tag.
+_MUTABLE_REF = re.compile(r"https?://[^\s]*/(?:blob|tree|raw)/(?:main|master|HEAD)/")
+_WINDOWS_RESERVED = set('<>:"|?*')
+
+
+def _section(status: str = MEASURED, **extra) -> dict:
+ return {"status": status, "findings": [], **extra}
+
+
+def _add(section: dict, *findings: Finding) -> None:
+ section["findings"].extend(finding.as_dict() for finding in findings)
+
+
+def _package_files(package: Path) -> tuple[list[Path], list[Finding]]:
+ """Every regular file in the package, plus a finding for any symlink.
+
+ Symlinks are refused rather than resolved. Admission stages a package as exact bytes, and there
+ are only two things it could do with a link: preserve it, which puts a path into the vault that
+ a reader follows back out of the library, or flatten it into a copy of its target, which
+ silently changes the artifact's shape. Neither is a call to make on an operator's behalf, so a
+ package containing one is refused by name. This walks explicitly instead of using `rglob`,
+ which follows directory symlinks and would pull in a subtree without ever reporting the link."""
+ files, findings = [], []
+ stack = [package]
+ while stack:
+ for path in sorted(stack.pop().iterdir()):
+ if path.is_symlink():
+ findings.append(Finding(
+ "symlink-unsupported", ERROR,
+ "symlinks are not admissible: admission stages exact bytes, and a link is "
+ "neither preserved nor followed",
+ path.relative_to(package).as_posix()))
+ elif path.is_dir():
+ stack.append(path)
+ elif path.is_file():
+ files.append(path)
+ return sorted(files), findings
+
+
+# Metadata no packaging format treats as skill content, and nothing a reviewer needs flagged.
+IGNORED_PARTS = {".git", "__pycache__", ".pytest_cache"}
+IGNORED_NAMES = {".DS_Store", "Thumbs.db"}
+
+
+def binary_assets(package: Path, files: list[Path]) -> list[str]:
+ """Files whose bytes are not text, so a reviewer knows what they are approving unread.
+
+ Admission preserves these byte-for-byte, which is why this is a note rather than a refusal. It
+ is still the fact a reviewer most needs in front of them: a skill's text can be read before it
+ is approved, and a compiled binary or an image cannot. Decodability, not the file extension,
+ decides -- an extension is a claim about a file, and the point here is to check the file."""
+ found = []
+ for path in files:
+ relative = path.relative_to(package)
+ if relative.name in IGNORED_NAMES or IGNORED_PARTS.intersection(relative.parts):
+ continue
+ # Decoded in chunks rather than read whole: this runs before any size limit applies, and a
+ # review command that a large file can exhaust memory on is one nobody runs on the packages
+ # that most need reviewing.
+ decoder = codecs.getincrementaldecoder("utf-8")()
+ try:
+ with path.open("rb") as handle:
+ while chunk := handle.read(1 << 20):
+ decoder.decode(chunk)
+ decoder.decode(b"", final=True)
+ except (UnicodeDecodeError, OSError):
+ found.append(relative.as_posix())
+ return found
+
+
+def _structural(package: Path, files: list[Path], escapes: list[Finding]) -> tuple[dict, dict | None]:
+ section = _section()
+ _add(section, *escapes)
+
+ skill_md = package / "SKILL.md"
+ if not skill_md.is_file():
+ _add(section, Finding("skill-md-missing", ERROR,
+ "no SKILL.md: a skill package is a directory containing one"))
+ return section, None
+
+ raw = parse_raw(skill_md.read_text(encoding="utf-8", errors="replace"))
+ _add(section, *raw.findings)
+ frontmatter = raw.frontmatter or {}
+
+ name = str(frontmatter.get("name") or "").strip()
+ if raw.frontmatter is not None:
+ if not name:
+ _add(section, Finding("name-missing", ERROR, "frontmatter declares no name"))
+ elif not SLUG_RE.fullmatch(name):
+ _add(section, Finding(
+ "name-invalid", ERROR,
+ f"name {name!r} is not a valid slug (lowercase letters, digits and hyphens, "
+ f"starting with a letter or digit)"))
+ elif name != package.name:
+ _add(section, Finding(
+ "name-directory-mismatch", WARNING,
+ f"frontmatter name {name!r} differs from the directory name {package.name!r}; "
+ f"the frontmatter name is the identity that will be served"))
+
+ if not str(frontmatter.get("description") or "").strip():
+ _add(section, Finding(
+ "description-empty", ERROR,
+ "no description: the router keys on it, so a skill without one is never loaded"))
+
+ body_bytes = len(raw.body.encode("utf-8"))
+ if not raw.body.strip():
+ _add(section, Finding("body-empty", WARNING, "the body is empty"))
+ elif body_bytes > LARGE_BODY_BYTES:
+ _add(section, Finding(
+ "body-large", WARNING,
+ f"body is {body_bytes} bytes; every load pays for it in the agent's context"))
+
+ total_bytes = sum(path.stat().st_size for path in files)
+ if len(files) > MANY_FILES:
+ _add(section, Finding("file-count-high", WARNING,
+ f"{len(files)} files in the package"))
+ if total_bytes > LARGE_PACKAGE_BYTES:
+ _add(section, Finding("package-large", WARNING,
+ f"package is {total_bytes} bytes"))
+
+ _add(section, *_path_findings(package, files))
+ _add(section, *_reference_findings(package, raw.body))
+ binaries = binary_assets(package, files)
+ if binaries:
+ _add(section, Finding(
+ "binary-asset", WARNING,
+ f"{len(binaries)} file(s) are not text and cannot be read before approval; they will "
+ f"be published byte-for-byte: {', '.join(binaries[:10])}"
+ + (f", and {len(binaries) - 10} more" if len(binaries) > 10 else "")))
+
+ section.update(
+ name=name or package.name,
+ description=str(frontmatter.get("description") or "").strip(),
+ body_bytes=body_bytes,
+ file_count=len(files),
+ total_bytes=total_bytes,
+ file_types=sorted({path.suffix.lower() or "(none)" for path in files}),
+ binary_assets=binaries,
+ )
+ return section, frontmatter if raw.frontmatter is not None else None
+
+
+def _path_findings(package: Path, files: list[Path]) -> list[Finding]:
+ """Portability of the paths themselves: characters, case, and Unicode form.
+
+ A pair of files that differ only by case or only by Unicode normalization is two files on Linux
+ and one on macOS or Windows. That makes the package's content revision depend on who checked it
+ out, which is the one property a content-addressed system cannot tolerate."""
+ findings, by_fold, by_norm = [], {}, {}
+ for path in files:
+ relative = path.relative_to(package).as_posix()
+ if _WINDOWS_RESERVED.intersection(relative) or "\\" in relative:
+ findings.append(Finding(
+ "path-not-portable", WARNING,
+ "path contains characters that are not portable across filesystems", relative))
+ by_fold.setdefault(relative.casefold(), []).append(relative)
+ by_norm.setdefault(unicodedata.normalize("NFC", relative), []).append(relative)
+
+ for group in by_fold.values():
+ if len(group) > 1:
+ findings.append(Finding(
+ "path-case-collision", WARNING,
+ f"paths differ only by case and collapse on a case-insensitive filesystem: "
+ f"{', '.join(sorted(group))}"))
+ for group in by_norm.values():
+ if len(group) > 1 and len({unicodedata.normalize("NFC", p) for p in group}) == 1 \
+ and len(set(group)) > 1:
+ findings.append(Finding(
+ "path-unicode-collision", WARNING,
+ f"paths differ only by Unicode normalization form: {', '.join(sorted(group))}"))
+ return findings
+
+
+def _reference_findings(package: Path, body: str) -> list[Finding]:
+ """Local links in the body: do they escape the package, and do they resolve?"""
+ findings = []
+ for target in _LOCAL_LINK.findall(body):
+ cleaned = target.split("#", 1)[0].strip()
+ if not cleaned:
+ continue
+ candidate = Path(cleaned)
+ if candidate.is_absolute() or ".." in candidate.parts:
+ findings.append(Finding(
+ "path-traversal", ERROR,
+ "reference points outside the package", cleaned))
+ continue
+ if not (package / candidate).exists():
+ findings.append(Finding(
+ "file-reference-missing", WARNING,
+ "referenced file is not in the package", cleaned))
+ return findings
+
+
+def _supply_chain(package: Path, files: list[Path], frontmatter: dict | None) -> dict:
+ """What the package declares about where it came from, and what it reaches for.
+
+ Everything here is advisory and nothing here fails a package. These are the facts a reviewer
+ needs in front of them; none of them is a security verdict, and this command does not pretend
+ to detect prompt injection -- that is a semantic property no deterministic scan establishes."""
+ section = _section()
+ declared = frontmatter or {}
+
+ if not declared.get("source"):
+ _add(section, Finding("source-metadata-missing", WARNING,
+ "no source declared: the package does not record where it came from"))
+ if not declared.get("license"):
+ _add(section, Finding("license-metadata-missing", WARNING,
+ "no license declared"))
+
+ executables = [path.relative_to(package).as_posix() for path in files
+ if path.stat().st_mode & 0o111 and not path.is_symlink()]
+ for relative in executables:
+ _add(section, Finding("executable-asset", WARNING,
+ "asset is executable", relative))
+
+ urls, mutable = set(), set()
+ for path in files:
+ try:
+ text = path.read_text(encoding="utf-8")
+ except (UnicodeDecodeError, OSError):
+ continue
+ urls.update(_URL.findall(text))
+ mutable.update(_MUTABLE_REF.findall(text))
+
+ # One aggregate finding, not one per URL: a documentation-heavy skill has dozens of links, and
+ # a wall of identical warnings is how a reviewer learns to stop reading them. The full list
+ # stays in `remote_references` for anything consuming the JSON.
+ if urls:
+ _add(section, Finding(
+ "remote-reference", WARNING,
+ f"package references {len(urls)} remote URL(s): {', '.join(sorted(urls)[:3])}"
+ + (" …" if len(urls) > 3 else "")))
+ if mutable:
+ _add(section, Finding(
+ "reference-unpinned", WARNING,
+ "reference points at a moving branch rather than a commit or tag; what it returns "
+ "today is not what it will return later"))
+
+ section.update(source=declared.get("source"), license=declared.get("license"),
+ executables=executables, remote_references=sorted(urls))
+ return section
+
+
+def _collision(package: Path, name: str, library_root: Path | None) -> dict:
+ """Deterministic collision only.
+
+ Description shadowing -- the thing that actually steals traffic -- is a cosine comparison that
+ needs the embedding router, which needs a model. Approximating it here would be worse than not
+ answering, so it reports UNMEASURED and names `ingot.optimize.routing_health`, which already does the
+ real library-wide scan."""
+ semantic = {"status": UNMEASURED,
+ "reason": "description shadowing needs the embedding router",
+ "measure_with": "python -m ingot.optimize.routing_health"}
+
+ if library_root is None:
+ return _section(UNMEASURED, reason="no library root supplied", semantic=semantic)
+
+ # `skill_sources`, not a bare glob: it skips the dot-prefixed staging directories promotion and
+ # rollback leave beside a live skill, each of which carries a complete SKILL.md. Globbing
+ # directly would report an abandoned `.pdf..stage` as a colliding skill.
+ section = _section(semantic=semantic, library_root=str(library_root))
+ existing = {path.parent.name for path in skill_sources(library_root)
+ if path.parent.resolve() != package.resolve()}
+ if name in existing:
+ _add(section, Finding(
+ "name-collision", WARNING,
+ f"the library already has a skill named {name!r}; one would shadow the other"))
+ for other in sorted(existing):
+ if other != name and other.casefold() == name.casefold():
+ _add(section, Finding(
+ "name-case-collision", WARNING,
+ f"the library has {other!r}, which differs from {name!r} only by case"))
+ return section
+
+
+def _activation(name: str, evidence_root: Path) -> dict:
+ """Whether a routing suite exists is a fact available offline. Whether the router loads this
+ skill at the right time is not, so it is never reported as a score."""
+ suite = evidence_root / "ingot" / "optimize" / "tasks" / f"{name}.yaml"
+ cases = 0
+ if suite.is_file():
+ try:
+ loaded = yaml.safe_load(suite.read_text(encoding="utf-8")) or {}
+ except yaml.YAMLError:
+ loaded = {}
+ routing = loaded.get("routing") if isinstance(loaded, dict) else None
+ cases = len(routing) if isinstance(routing, list) else 0
+
+ return _section(
+ UNMEASURED,
+ routing_cases=cases,
+ reason=("a routing suite exists but scoring it needs the embedding router"
+ if cases else "no routing suite for this skill"),
+ measure_with=f"python -m ingot.optimize.routing_health {name}",
+ )
+
+
+def _behavioral(name: str, evidence_root: Path) -> dict:
+ """Surface what `ingot.optimize.compat` already measured. Never recompute it: that costs a model, a
+ key, and money, and this command promises none of the three."""
+ path = evidence_root / "runs" / "compat" / f"{name}.json"
+ if not path.is_file():
+ return _section(UNMEASURED,
+ reason="no compatibility evidence for this skill",
+ measure_with=f"python -m ingot.optimize.compat {name}")
+ try:
+ summary = json.loads(path.read_text(encoding="utf-8"))
+ except (json.JSONDecodeError, OSError) as error:
+ return _section(UNMEASURED, reason=f"compatibility evidence is unreadable: {error}")
+
+ return _section(MEASURED, source=str(path), tasks=summary.get("tasks"),
+ judge=summary.get("judge"), models=summary.get("models") or {})
+
+
+def review_package(package: Path, *, library_root: Path | None = None,
+ evidence_root: Path | None = None) -> dict:
+ """Review one skill package. Reads only; writes nothing, anywhere."""
+ package = Path(package)
+ evidence_root = Path(evidence_root) if evidence_root is not None \
+ else Path(__file__).resolve().parent.parent
+
+ files, escapes = _package_files(package)
+ structural, frontmatter = _structural(package, files, escapes)
+ name = structural.get("name", package.name)
+
+ try:
+ revision = skill_revision(package)
+ except (ValueError, OSError) as error:
+ revision = ""
+ _add(structural, Finding("revision-unavailable", ERROR,
+ f"cannot compute a content revision: {error}"))
+
+ sections = {
+ "structural": structural,
+ "supply_chain": _supply_chain(package, files, frontmatter),
+ "collision": _collision(package, name, library_root),
+ "activation": _activation(name, evidence_root),
+ "behavioral": _behavioral(name, evidence_root),
+ }
+
+ findings = [finding for section in sections.values() for finding in section["findings"]]
+ errors = sum(1 for finding in findings if finding["level"] == ERROR)
+ return {
+ "schema_version": REVIEW_SCHEMA,
+ "package": str(package),
+ "skill": name,
+ "revision": revision,
+ "valid": errors == 0,
+ "errors": errors,
+ "warnings": sum(1 for finding in findings if finding["level"] == WARNING),
+ "sections": sections,
+ }
+
+
+def render(result: dict) -> str:
+ """One block per section, so a reader sees which question failed rather than a number."""
+ verdict = "VALID" if result["valid"] else "INVALID"
+ lines = [f"{result['skill']} {verdict} {result['errors']} error(s), "
+ f"{result['warnings']} warning(s)",
+ f" package {result['package']}",
+ f" revision {result['revision'][:16] or '(unavailable)'}"]
+
+ for title, section in result["sections"].items():
+ status = "" if section["status"] == MEASURED else f" [{section['status'].upper()}]"
+ lines.append(f"\n{title.replace('_', ' ')}{status}")
+ if section["status"] == UNMEASURED and section.get("reason"):
+ lines.append(f" {section['reason']}")
+ if section.get("measure_with"):
+ lines.append(f" measure with: {section['measure_with']}")
+ for finding in section["findings"]:
+ where = f" ({finding['path']})" if finding.get("path") else ""
+ lines.append(f" {finding['level']:<7} {finding['code']}: {finding['message']}{where}")
+ if not section["findings"] and section["status"] == MEASURED:
+ lines.append(" no findings")
+ return "\n".join(lines)
diff --git a/ingot/status.py b/ingot/status.py
new file mode 100644
index 0000000..89b88db
--- /dev/null
+++ b/ingot/status.py
@@ -0,0 +1,221 @@
+"""Whether this deployment's guarantees actually hold, reported as an observation.
+
+The verdict is not read back out of configuration. It compares what is served against what the
+last successful release receipt says should be served, per skill, and reports the worst answer.
+A configuration flag would have agreed with the claim rather than checked it, which is the failure
+this command exists to catch."""
+from __future__ import annotations
+
+import os
+from pathlib import Path
+
+STATUS_SCHEMA = "ingot/status/v1"
+
+MANAGED = "MANAGED" # served bytes are the last successful release
+PENDING = "PENDING" # a proposal or publication is in flight
+DRIFTED = "DRIFTED" # served bytes differ from the last successful release
+UNMANAGED = "UNMANAGED" # no release receipt covers these bytes, or development mode
+
+# Worst first. Drift is the alarm; an unmanaged skill is one the guarantee never covered; a
+# publication in flight is expected and transient.
+SEVERITY = (DRIFTED, UNMANAGED, PENDING, MANAGED)
+ABSENT = "absent"
+
+
+def _writable(path: Path) -> bool:
+ """Whether this process could change what is served.
+
+ `os.access` rather than a permission-bit reading: it accounts for the read-only mount managed
+ mode relies on, which no mode bit describes. Reported as a fact rather than folded into the
+ verdict — the administrator who owns the vault can always write it, and a status command that
+ answered UNMANAGED from their shell would hide the drift they most need to see."""
+ return path.is_dir() and os.access(path, os.W_OK | os.X_OK)
+
+
+def _worst(states) -> str:
+ for state in SEVERITY:
+ if state in states:
+ return state
+ return MANAGED
+
+
+def skill_states(root: Path | None = None) -> list[dict]:
+ """One verdict per skill, plus every skill a release receipt names but nothing serves."""
+ from ingot.mcp_server.registry import load_skills
+ from ingot.optimize.promote import list_pending
+ from ingot.optimize.publication import latest_releases, publishing_skills
+
+ explicit = [root] if root is not None else None
+ served = {skill.name: skill.revision for skill in load_skills(roots=explicit)}
+ releases = latest_releases()
+ in_flight = publishing_skills() | {record.get("skill") for record in list_pending()}
+
+ states = []
+ # In-flight names join the union: a creation that is quarantined or travelling is served by
+ # nothing and released by nothing, so a status built from those two sets alone would report an
+ # empty library as fully MANAGED while a publication was in progress.
+ for name in sorted(set(served) | set(releases) | {name for name in in_flight if name}):
+ current = served.get(name, ABSENT)
+ release = releases.get(name)
+ released = release.get("candidate_revision") if release else None
+ if released is None:
+ # Nothing Ingot published is responsible for these bytes: they were fetched, copied,
+ # or committed to the vault by hand. Real, common, and not drift — there is no release
+ # to have drifted from.
+ state = PENDING if name in in_flight else UNMANAGED
+ elif current != released:
+ state = DRIFTED
+ elif name in in_flight:
+ state = PENDING
+ else:
+ state = MANAGED
+ states.append({"skill": name, "state": state, "revision": current,
+ "released": released,
+ "publication": release.get("id") if release else None})
+ return states
+
+
+def target_states(env: dict | None = None) -> list[dict]:
+ """One verdict per delivery target, decided the same way as the library's: by looking.
+
+ A target is graded only on the skills Ingot released there or has in flight for it. A native
+ skill root is shared with whatever its owner put in it, and those skills are not Ingot's to
+ judge -- grading them would report every real deployment as permanently UNMANAGED and bury the
+ one line that matters. A released skill that has been deleted from a target is still drift:
+ absence is a revision, and it is not the released one."""
+ from ingot import delivery
+ from ingot.optimize.promote import list_pending
+ from ingot.optimize.publication import latest_releases, publishing_skills
+
+ from . import paths
+
+ env = os.environ if env is None else env
+ targets = delivery.load_targets(env, vault=paths.vault(env))
+ releases = latest_releases()
+ in_flight = {name for name in
+ publishing_skills() | {record.get("skill") for record in list_pending()} if name}
+
+ reported = []
+ for target in targets:
+ skills = []
+ for name in sorted(set(releases) | in_flight):
+ released = (releases.get(name) or {}).get("candidate_revision")
+ current = delivery.observed(target, name)
+ if released is None:
+ state = PENDING if name in in_flight else UNMANAGED
+ elif current != released:
+ state = DRIFTED
+ elif name in in_flight:
+ state = PENDING
+ else:
+ state = MANAGED
+ skills.append({"skill": name, "state": state, "revision": current,
+ "released": released})
+ reported.append({"name": target.name, "kind": target.kind, "root": str(target.root),
+ "state": _worst({entry["state"] for entry in skills}), "skills": skills})
+ return reported
+
+
+def library_status(root: Path | None = None, env: dict | None = None) -> dict:
+ from ingot.mcp_server.registry import configured_roots
+
+ from . import paths
+
+ env = os.environ if env is None else env
+ # An explicit root means exactly that root. `configured_roots` always prepends the local
+ # authoring library even ahead of one, which is right for serving and wrong here: asked whether
+ # a particular library is managed, this must not answer about a different one.
+ roots = ([Path(root).expanduser().resolve()] if root is not None
+ else [Path(path) for path in configured_roots()])
+ development = (env.get("INGOT_MODE") or "").strip().lower() in {"dev", "unmanaged"}
+ states = skill_states(root)
+ # A misconfigured target list must not take the command down. Status is what an operator runs
+ # when something is already wrong, so a broken variable is a finding to report, not a crash.
+ try:
+ targets, delivery_error = target_states(env), None
+ except (ValueError, OSError) as exc:
+ targets, delivery_error = [], str(exc)
+ verdicts = {entry["state"] for entry in states}
+ verdicts |= {entry["state"] for entry in targets}
+ if delivery_error:
+ verdicts.add(DRIFTED)
+ mode = UNMANAGED if development else _worst(verdicts)
+ return {
+ "schema_version": STATUS_SCHEMA,
+ "mode": mode,
+ "development_mode": development,
+ "uid": os.getuid(),
+ "roots": [str(path) for path in roots],
+ "writable_roots": [str(path) for path in roots if _writable(path)],
+ "skills": states,
+ "targets": targets,
+ "delivery_error": delivery_error,
+ "publish_backend": env.get("INGOT_PUBLISH_BACKEND") or "local",
+ "vault_path": str(paths.vault(env)),
+ "forge_repository": env.get("INGOT_FORGE_REPOSITORY"),
+ "paths": paths.resolved(env),
+ # Never migrated, only reported: moving a review queue or a receipt store on someone's
+ # behalf is a change to controlled state made by a process nobody asked to make it.
+ "legacy_state": paths.legacy_state(),
+ }
+
+
+_EXPLANATION = {
+ MANAGED: "Every served skill is exactly the revision its last release receipt names.",
+ PENDING: "A proposal or publication is in flight. Nothing has drifted.",
+ DRIFTED: "Served bytes differ from the last successful release. Something changed them "
+ "outside the publisher.",
+ UNMANAGED: "Some served bytes have no release receipt behind them, so the quarantine and "
+ "publication guarantees do not describe them.",
+}
+
+
+def render(result: dict) -> str:
+ lines = [f"{result['mode']} (uid {result['uid']})", ""]
+ if result["development_mode"]:
+ lines.append(" Development mode (INGOT_MODE). The served library is writable by every")
+ lines.append(" service and control-plane guarantees do not apply to this deployment.")
+ else:
+ lines.append(f" {_EXPLANATION[result['mode']]}")
+ for entry in result["skills"]:
+ if entry["state"] == MANAGED:
+ continue
+ detail = (f"served {entry['revision'][:12]} != released {entry['released'][:12]}"
+ if entry["state"] == DRIFTED else
+ "in flight" if entry["state"] == PENDING else "no release receipt")
+ lines.append(f" {entry['state']:<10} {entry['skill']:<20} {detail}")
+ if result["writable_roots"] and not result["development_mode"]:
+ lines += ["", " Writable by this process, so nothing here stops this user from changing",
+ " what is served without an approval:"]
+ lines += [f" {path}" for path in result["writable_roots"]]
+ if result.get("delivery_error"):
+ lines += ["", " The delivery target configuration cannot be read, so nothing here knows",
+ f" what this deployment installs or where: {result['delivery_error']}"]
+ for target in result.get("targets") or []:
+ drifted = [entry for entry in target["skills"] if entry["state"] != MANAGED]
+ if target["state"] == MANAGED and not drifted:
+ continue
+ lines += ["", f" {target['state']:<10} target {target['name']} ({target['kind']}) "
+ f"{target['root']}"]
+ for entry in drifted:
+ detail = (f"holds {entry['revision'][:12]} != released {entry['released'][:12]}"
+ if entry["state"] == DRIFTED else
+ "in flight" if entry["state"] == PENDING else "no release receipt")
+ lines.append(f" {entry['state']:<10} {entry['skill']:<20} {detail}")
+ lines += ["", f" backend {result['publish_backend']}"]
+ if result["forge_repository"]:
+ lines.append(f" forge {result['forge_repository']}")
+ lines.append(f" roots {', '.join(result['roots'])}")
+ for target in result.get("targets") or []:
+ lines.append(f" deliver {target['name']:<12} {target['kind']:<12} {target['root']} "
+ f"{target['state']}")
+ lines += ["", " state"]
+ width = max(len(entry["name"]) for entry in result["paths"])
+ for entry in result["paths"]:
+ note = "" if entry["writable"] else " NOT WRITABLE"
+ lines.append(f" {entry['name']:<{width}} {entry['path']} [{entry['source']}]{note}")
+ if result["legacy_state"]:
+ lines += ["", " State left beside the code by an earlier version. Nothing here reads it;",
+ " move what you want to keep under the paths above, then delete it:"]
+ lines += [f" {path}" for path in result["legacy_state"]]
+ return "\n".join(lines)
diff --git a/ingot/vault.py b/ingot/vault.py
new file mode 100644
index 0000000..79b3067
--- /dev/null
+++ b/ingot/vault.py
@@ -0,0 +1,138 @@
+"""Create the local Git vault the publisher owns.
+
+The local backend needs a Git repository before it can publish anything, and a deployment whose
+very first start fails because the directory is empty is a deployment nobody gets past. This is the
+one bootstrap command, and it is idempotent so the managed compose can call it on every start."""
+from __future__ import annotations
+
+import json
+import subprocess
+from pathlib import Path
+
+VAULT_SCHEMA = "ingot/vault/v1"
+
+# Written into the vault, not imported from it: the publisher runs this with the vault as the
+# working directory and nothing but the interpreter guaranteed to be there. It refuses the tree
+# rather than the change, so a vault a server could not load can never be committed.
+VALIDATOR = '''#!/usr/bin/env python3
+"""Refuse a vault tree the skill server could not serve.
+
+Every publication runs this against a fresh worktree before the commit is made."""
+import json
+import re
+import sys
+from pathlib import Path
+
+import yaml
+
+FRONTMATTER = re.compile(r"\\A---\\r?\\n(.*?)\\r?\\n---\\r?\\n?", re.DOTALL)
+SLUG = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*\\Z")
+
+root = Path(__file__).resolve().parent.parent
+problems = []
+
+registry = root / "registry.json"
+if registry.is_file():
+ try:
+ if not isinstance(json.loads(registry.read_text(encoding="utf-8")), dict):
+ problems.append("registry.json is not a JSON object")
+ except ValueError as exc:
+ problems.append(f"registry.json is not valid JSON: {exc}")
+
+for skill_md in sorted(root.glob("*/SKILL.md")):
+ name = skill_md.parent.name
+ where = f"{name}/SKILL.md"
+ if not SLUG.fullmatch(name):
+ problems.append(f"{name}: directory name is not a slug")
+ text = skill_md.read_text(encoding="utf-8")
+ match = FRONTMATTER.match(text)
+ if not match:
+ problems.append(f"{where}: no YAML frontmatter")
+ continue
+ try:
+ metadata = yaml.safe_load(match.group(1))
+ except yaml.YAMLError as exc:
+ problems.append(f"{where}: frontmatter is not valid YAML: {exc}")
+ continue
+ if not isinstance(metadata, dict):
+ problems.append(f"{where}: frontmatter is not a mapping")
+ continue
+ if metadata.get("name") != name:
+ problems.append(f"{where}: frontmatter name {metadata.get('name')!r} != directory {name!r}")
+ if not str(metadata.get("description", "")).strip():
+ problems.append(f"{where}: description is empty")
+
+for problem in problems:
+ print(f"invalid: {problem}", file=sys.stderr)
+sys.exit(1 if problems else 0)
+'''
+
+
+def _git(path: Path, *args: str) -> str:
+ result = subprocess.run(["git", "-C", str(path), *args], capture_output=True, text=True)
+ if result.returncode:
+ detail = result.stderr.strip().splitlines()[-1] if result.stderr.strip() else "failed"
+ verb = next((arg for arg in args if not arg.startswith("-") and "=" not in arg), args[0])
+ raise ValueError(f"git {verb}: {detail}")
+ return result.stdout.strip()
+
+
+def _is_repository(path: Path) -> bool:
+ result = subprocess.run(["git", "-C", str(path), "rev-parse", "--git-dir"],
+ capture_output=True, text=True)
+ return result.returncode == 0
+
+
+def init_vault(path: Path, *, branch: str = "main") -> dict:
+ """Create or complete a vault. Idempotent against one that is already valid.
+
+ A non-empty directory that is not a Git repository is refused rather than adopted: pointing the
+ publisher at a directory of loose skills would make its first commit look like a publication
+ nobody approved."""
+ path = Path(path).expanduser()
+ existed = path.exists()
+ if existed and not path.is_dir():
+ raise ValueError(f"{path} is not a directory")
+ if existed and not _is_repository(path) and any(path.iterdir()):
+ raise ValueError(f"{path} is not empty and is not a Git repository; "
+ f"initialize an empty directory or point at an existing vault")
+ path.mkdir(parents=True, exist_ok=True)
+ path = path.resolve()
+
+ created = not _is_repository(path)
+ if created:
+ _git(path, "init", "-b", branch)
+
+ written = []
+ for relative, contents in ((Path("registry.json"), "{}\n"),
+ (Path("scripts") / "validate.py", VALIDATOR)):
+ target = path / relative
+ if target.exists():
+ continue
+ target.parent.mkdir(parents=True, exist_ok=True)
+ target.write_text(contents, encoding="utf-8")
+ written.append(relative.as_posix())
+
+ head = None
+ if written or created:
+ _git(path, "add", "-A", ".")
+ if _git(path, "status", "--porcelain"):
+ _git(path, "-c", "user.name=Ingot Publisher", "-c", "user.email=ingot@local.invalid",
+ "commit", "-m", "Initialize the Ingot skill vault")
+ try:
+ head = _git(path, "rev-parse", "HEAD")
+ except ValueError:
+ # An adopted repository with no commits at all: give it one so a publication branch has a
+ # base to be cut from.
+ _git(path, "-c", "user.name=Ingot Publisher", "-c", "user.email=ingot@local.invalid",
+ "commit", "--allow-empty", "-m", "Initialize the Ingot skill vault")
+ head = _git(path, "rev-parse", "HEAD")
+
+ return {
+ "schema_version": VAULT_SCHEMA,
+ "path": str(path),
+ "status": "created" if created else ("updated" if written else "unchanged"),
+ "branch": _git(path, "symbolic-ref", "--short", "HEAD"),
+ "head": head,
+ "added": written,
+ }
diff --git a/mcp_server/router.py b/mcp_server/router.py
deleted file mode 100644
index b9be636..0000000
--- a/mcp_server/router.py
+++ /dev/null
@@ -1,192 +0,0 @@
-"""Tier-1 embedding router: embed every skill's description once, then suggest the top-k skills for a
-task by cosine similarity. CPU-only ONNX, no GPU, so the demo is `docker compose up`.
-
-Model is `EMBED_MODEL` (default Qwen3-Embedding-0.6B q4, ~15 ms/query on CPU; queries get the
-retrieval instruction prefix, descriptions don't). Any fastembed model name also works (e.g. the
-previous default `BAAI/bge-small-en-v1.5`, ~4 ms/query), but recalibrate MIN_SCORE /
-RELATED_SCORE / COLLISION_SCORE with the model (mcp_server/embedding.py)."""
-from __future__ import annotations
-import os
-import sys
-import threading
-from pathlib import Path
-
-import numpy as np
-
-from .embedding import EMBED_MODEL as _MODEL, build_embedding
-from .registry import Skill
-
-
-class Router:
- _vector_cache: dict[tuple[str, str], np.ndarray] = {}
- _cache_lock = threading.Lock()
-
- def __init__(self, skills: list[Skill]):
- self.skills = skills
- self._embed = build_embedding()
- if not skills: # empty library, don't normalize an empty matrix
- self._mat = np.zeros((0, 0), dtype=np.float32)
- return
- keys = [(_MODEL, skill.description) for skill in skills]
- with self._cache_lock:
- missing = list(dict.fromkeys(key for key in keys if key not in self._vector_cache))
- if missing:
- vectors = self._embed.embed([description for _, description in missing])
- with self._cache_lock:
- for key, vector in zip(missing, vectors):
- self._vector_cache[key] = np.asarray(vector, dtype=np.float32)
- with self._cache_lock:
- mat = np.array([self._vector_cache[key] for key in keys], dtype=np.float32)
- self._mat = mat / (np.linalg.norm(mat, axis=1, keepdims=True) + 1e-8)
-
- def nearest(self, text: str) -> tuple[str, float]:
- """The most similar existing skill to `text` and its cosine score, used to reject a new
- skill whose description near-duplicates (shadows) an existing one's routing."""
- if not self.skills:
- return "", 0.0
- q = np.array(next(iter(self._embed.embed([text]))), dtype=np.float32)
- q = q / (np.linalg.norm(q) + 1e-8)
- scores = self._mat @ q
- i = int(np.argmax(scores))
- return self.skills[i].name, float(scores[i])
-
- def suggest(self, task: str, k: int = 5, min_score: float = 0.0) -> list[dict]:
- if not self.skills:
- return []
- q = np.array(next(iter(self._embed.embed_query([task]))), dtype=np.float32)
- q = q / (np.linalg.norm(q) + 1e-8)
- scores = self._mat @ q
- top = np.argsort(-scores)[:k]
- return [
- {"name": self.skills[i].name, "description": self.skills[i].description,
- "score": round(float(scores[i]), 3)}
- for i in top if scores[i] >= min_score
- ]
-
- @staticmethod
- def _platform(value: str | None) -> str:
- value = (value or sys.platform).lower()
- if value.startswith("darwin") or value == "macos":
- return "macos"
- if value.startswith("win"):
- return "windows"
- return "linux" if value.startswith("linux") else value
-
- @staticmethod
- def _compatible(skill: Skill, harness: str, cwd: str, available_tools: set[str],
- available_mcps: set[str], platform: str) -> bool:
- meta = skill.metadata or {}
- if harness not in meta.get("harnesses", ["claude", "codex"]):
- return False
- if platform not in meta.get("platforms", ["macos", "linux", "windows"]):
- return False
- if meta.get("activation", "automatic") != "automatic" or meta.get("trust") == "blocked":
- return False
- if not set(meta.get("required_tools", [])).issubset(available_tools):
- return False
- if not set(meta.get("required_mcps", [])).issubset(available_mcps):
- return False
- scopes = meta.get("scopes", ["global"])
- if "global" not in scopes:
- patterns = meta.get("path_patterns", [])
- if "project" not in scopes or not patterns:
- return False
- project = Path(cwd).expanduser().resolve()
- if not project.is_dir() or not any(any(project.glob(pattern)) for pattern in patterns):
- return False
- return True
-
- def _eligible_ranking(self, task: str, harness: str, cwd: str,
- available_tools: set[str], available_mcps: set[str],
- platform: str) -> list[tuple[Skill, float]]:
- eligible = [skill for skill in self.skills if self._compatible(
- skill, harness, cwd, available_tools, available_mcps, platform
- )]
- if not eligible:
- return []
- query = np.array(next(iter(self._embed.embed_query([task]))), dtype=np.float32)
- query = query / (np.linalg.norm(query) + 1e-8)
- index_by_name = {skill.name: index for index, skill in enumerate(self.skills)}
- return sorted(
- ((skill, float(self._mat[index_by_name[skill.name]] @ query)) for skill in eligible),
- key=lambda pair: (-pair[1], -int(pair[0].metadata.get("priority", 50)), pair[0].name),
- )
-
- @staticmethod
- def _without_conflicts(ranked: list[tuple[Skill, float]]) -> list[tuple[Skill, float]]:
- selected = []
- for candidate in ranked:
- skill = candidate[0]
- if any(skill.name in set(existing.metadata.get("conflicts", [])) or
- existing.name in set(skill.metadata.get("conflicts", []))
- for existing, _ in selected):
- continue
- selected.append(candidate)
- return selected
-
- @staticmethod
- def _alternatives(ranked: list[tuple[Skill, float]]) -> list[dict]:
- return [
- {"name": skill.name, "score": round(score, 3),
- "reason": f"compatible alternative; cosine {score:.3f}"}
- for skill, score in ranked[1:3]
- ]
-
- @staticmethod
- def _novel_response(score: float = 0.0, reason: str = "no compatible skill candidates",
- alternatives: list[dict] | None = None) -> dict:
- return {
- "match": None, "related_match": None, "score": round(score, 3),
- "reason": reason, "skill_body": "", "skill_root": None, "revision": None,
- "alternatives": alternatives or [], "novel": True,
- }
-
- @staticmethod
- def _related_response(skill: Skill, score: float, harness: str, min_score: float,
- alternatives: list[dict]) -> dict:
- return {
- "match": None, "related_match": skill.name, "score": round(score, 3),
- "reason": (f"best compatible score {score:.3f} below direct threshold "
- f"{min_score:.3f}; loaded for compose or extend"),
- "skill_body": skill.body_for(harness),
- "skill_root": skill.root or str(os.path.dirname(skill.path)),
- "revision": skill.revision or None, "alternatives": alternatives, "novel": False,
- }
-
- @staticmethod
- def _direct_response(skill: Skill, score: float, harness: str,
- alternatives: list[dict]) -> dict:
- return {
- "match": skill.name, "related_match": None, "score": round(score, 3),
- "reason": f"compatible {harness} skill; cosine {score:.3f}",
- "skill_body": skill.body_for(harness),
- "skill_root": skill.root or str(os.path.dirname(skill.path)),
- "revision": skill.revision or None, "alternatives": alternatives, "novel": False,
- }
-
- def route(self, task: str, harness: str, cwd: str, available_tools=(), available_mcps=(),
- platform: str | None = None, min_score: float = 0.53,
- related_score: float = 0.37) -> dict:
- """Filter compatible skills, rank them locally, and return at most one instruction body.
- `novel` is the escalation signal for the calling harness: True when nothing compatible is
- even related (best score below `related_score`), the case where a weak/strong setup should
- serve with the strong model, then queue a candidate for human review."""
- harness = harness.lower()
- ranked = self._eligible_ranking(
- task, harness, cwd, set(available_tools), set(available_mcps), self._platform(platform)
- )
- if not ranked:
- return self._novel_response()
- ranked = self._without_conflicts(ranked)
- top, score = ranked[0]
- alternatives = self._alternatives(ranked)
- if score < min_score:
- if score < related_score:
- related = [{"name": top.name, "score": round(score, 3),
- "reason": f"best compatible candidate; cosine {score:.3f}"},
- *alternatives]
- reason = (f"best compatible score {score:.3f} below related threshold "
- f"{related_score:.3f}")
- return self._novel_response(score, reason, related[:3])
- return self._related_response(top, score, harness, min_score, alternatives)
- return self._direct_response(top, score, harness, alternatives)
diff --git a/ops/systemd/ingot-publisher.service b/ops/systemd/ingot-publisher.service
new file mode 100644
index 0000000..03b23ca
--- /dev/null
+++ b/ops/systemd/ingot-publisher.service
@@ -0,0 +1,33 @@
+[Unit]
+# The one writer of the served skill library. It publishes approved receipts into the vault and
+# activates only the revision the receipt names.
+#
+# Running it on the host rather than in Docker sidesteps the uid mismatch the compose file warns
+# about — the receipts it reads are written at mode 0700 by the console — and, in `forge` mode, it
+# reuses the host's already authenticated git and gh so no GitHub credential has to live in a
+# container or in .env. The compose stack ships an equivalent `publisher` service for the
+# all-in-one case. Whichever you run, exactly one must run.
+#
+# Install (as the user who owns the vault):
+# mkdir -p ~/.config/ingot && cp ops/systemd/publisher.env.example ~/.config/ingot/publisher.env
+# $EDITOR ~/.config/ingot/publisher.env # at minimum INGOT_VAULT_PATH
+# systemctl --user enable --now ingot-publisher
+Description=Ingot skill vault publisher
+After=network-online.target
+Wants=network-online.target
+
+[Service]
+Type=simple
+# Where this checkout lives. Override with a drop-in rather than editing this unit:
+# systemctl --user edit ingot-publisher
+WorkingDirectory=%h/Source/ingot
+Environment=PYTHONUNBUFFERED=1
+# Backend, vault path, and any forge settings. The publisher refuses to start without a vault
+# rather than falling back to a writable default, so this file is required.
+EnvironmentFile=%h/.config/ingot/publisher.env
+ExecStart=/usr/bin/python3 -m ingot.optimize.publisher --watch
+Restart=always
+RestartSec=10
+
+[Install]
+WantedBy=default.target
diff --git a/ops/systemd/publisher.env.example b/ops/systemd/publisher.env.example
new file mode 100644
index 0000000..f9a7a4f
--- /dev/null
+++ b/ops/systemd/publisher.env.example
@@ -0,0 +1,16 @@
+# Copy to ~/.config/ingot/publisher.env. Read by ops/systemd/ingot-publisher.service.
+
+# local (default): the vault is a Git repository on this machine and nothing leaves it.
+# forge: publication authority is a merged pull request; needs the network and an authenticated gh.
+INGOT_PUBLISH_BACKEND=local
+
+# The vault checkout this publisher owns, and the library the server serves. Required: the
+# publisher refuses to start rather than adopt a writable default. `ingot vault init `
+# creates one. Absolute: systemd does not expand `~` or its own specifiers inside an
+# EnvironmentFile.
+INGOT_VAULT_PATH=/home/you/ingot-vault
+
+# forge only. Setting these under the local backend is inert and the publisher says so at startup.
+# INGOT_FORGE_REPOSITORY=owner/repo
+# INGOT_FORGE_REMOTE=origin
+# INGOT_FORGE_BRANCH=main
diff --git a/optimize/compat.py b/optimize/compat.py
deleted file mode 100644
index 36d39bc..0000000
--- a/optimize/compat.py
+++ /dev/null
@@ -1,108 +0,0 @@
-"""Cross-model skill compatibility, how well a skill's body transfers across serving models.
-
-A skill body is tuned for one serving model (`AGENT_MODEL`); SkillOpt's own result is that good
-skills transfer, but not always. For each model in `COMPAT_MODELS`, this runs the skill's held-out
-tasks through the one serving contract twice, once with the skill body, once with an empty body
-(the no-skill baseline), judges both with the FIXED judge, and reports per-model **lift**
-(skill mean − baseline mean). Positive lift = the body helps that model; ~0 = the model already
-knows this and the body is dead weight there.
-
-Langfuse-free: it reuses the direct rollout + judge (the same path the inner loop uses), so it needs
-no trace backend or experiment logging. Only the *serving* model varies, the judge stays fixed so
-scores are comparable across models.
-
-Usage: python -m optimize.compat
-Config: COMPAT_MODELS=qwen/qwen3-32b,openai/gpt-5.5,anthropic/claude-sonnet-... (default: AGENT_MODEL)
-"""
-import json
-import os
-import statistics
-from concurrent.futures import ThreadPoolExecutor
-from pathlib import Path
-
-from langchain_openai import ChatOpenAI
-
-from mcp_server.registry import SKILLS_DIR, optimizable_components
-
-from . import SERVE_TEMPLATE, agent_model, client_kwargs, model_api_key, model_base_url
-from . import usage as usage_ledger
-from .ab import load_tasks
-from .judge import invoke_retry, judge
-from .rollout import assemble
-
-_MAX_WORKERS = 8
-COMPAT_DIR = Path(__file__).resolve().parent.parent / "runs" / "compat"
-# The no-skill baseline: the identical serving contract with no skill body, so `lift` isolates the
-# body's contribution rather than the difference between two different prompts.
-NO_SKILL_BODY = "(no skill loaded, answer the task from your own knowledge)"
-
-
-def compat_models() -> list[str]:
- """Models to sweep: COMPAT_MODELS (comma-separated), else just the configured AGENT_MODEL."""
- models = [m.strip() for m in os.environ.get("COMPAT_MODELS", "").split(",") if m.strip()]
- return models or [agent_model()]
-
-
-def _llm(model: str):
- # OpenRouter (default) selects the model by slug over one endpoint, with ZDR routing applied;
- # a local MODEL_BASE_URL serves a single model, so a multi-model sweep only makes sense on a
- # multi-model endpoint. reasoning is left at the provider default, some models reject the flag.
- return ChatOpenAI(model=model, temperature=0, **client_kwargs(model_base_url(), key=model_api_key()))
-
-
-def _score(llm, system: str, task: dict) -> float:
- msg = invoke_retry(llm, [("system", system), ("user", task["task"])])
- usage_ledger.add("compat", getattr(msg, "usage_metadata", None))
- return judge(task["task"], task["rubric"], msg.content,
- check=task.get("check"), deliverable=task.get("deliverable"))["score"]
-
-
-def _run_arm(llm, system: str, tasks: list[dict]) -> list[float]:
- with ThreadPoolExecutor(max_workers=min(_MAX_WORKERS, len(tasks))) as pool:
- return list(pool.map(lambda t: _score(llm, system, t), tasks))
-
-
-def run_compat(skill: str, log=print) -> dict:
- """Sweep COMPAT_MODELS over the skill's held-out tasks (skill vs no-skill) and write the matrix
- to runs/compat/.json. Returns the summary."""
- usage_ledger.reset()
- if not (SKILLS_DIR / skill / "SKILL.md").exists():
- raise SystemExit(f"No skill named '{skill}' in skills/.")
- _, holdout, _ = load_tasks(skill)
- if not holdout:
- raise SystemExit(f"'{skill}' has no held-out eval tasks to run.")
- skill_system = SERVE_TEMPLATE.format(body=assemble(optimizable_components(SKILLS_DIR / skill)))
- base_system = SERVE_TEMPLATE.format(body=NO_SKILL_BODY)
- models = compat_models()
- log(f"[compat] '{skill}': {len(holdout)} held-out tasks × {len(models)} model(s); "
- f"judge fixed, serving model varies")
-
- models_out = {}
- for model in models:
- llm = _llm(model)
- skill_scores = _run_arm(llm, skill_system, holdout)
- base_scores = _run_arm(llm, base_system, holdout)
- s_mean, b_mean = statistics.mean(skill_scores), statistics.mean(base_scores)
- models_out[model] = {"skill_mean": s_mean, "baseline_mean": b_mean, "lift": s_mean - b_mean,
- "skill_scores": skill_scores, "baseline_scores": base_scores}
- verdict = "helps" if s_mean - b_mean > 0.05 else "no lift" if s_mean - b_mean >= -0.05 else "HURTS"
- log(f"[compat] {model:<34} skill {s_mean:.3f} baseline {b_mean:.3f} "
- f"lift {s_mean - b_mean:+.3f} ({verdict})")
-
- summary = {"skill": skill, "tasks": len(holdout),
- "judge": os.environ.get("JUDGE_MODELS") or os.environ.get("JUDGE_MODEL", ""),
- "models": models_out, "usage": usage_ledger.report()}
- COMPAT_DIR.mkdir(parents=True, exist_ok=True)
- path = COMPAT_DIR / f"{skill}.json"
- path.write_text(json.dumps(summary, indent=2))
- log(f"[compat] matrix written to {path}")
- log(usage_ledger.format_report())
- return summary
-
-
-if __name__ == "__main__":
- import sys
-
- from . import require_openrouter_key
- require_openrouter_key()
- run_compat(sys.argv[1] if len(sys.argv) > 1 else "tailwind")
diff --git a/optimize/judge.py b/optimize/judge.py
deleted file mode 100644
index 2c08b97..0000000
--- a/optimize/judge.py
+++ /dev/null
@@ -1,148 +0,0 @@
-"""LLM judge: scores an answer 0..1 and, following the SkillForge paper's multi-dimensional Failure
-Analyzer (Liu et al., "SkillForge", arXiv:2604.08618), classifies each failure across fixed
-dimensions so the search gets *categorized* feedback, not one opaque score. The dimension labels
-also drive success/failure mining (optimize/mine.py) and the candidate search's diagnosis.
-
-Judges against a task `rubric` when given one; with no rubric it grades reference-free (used when
-mining real traces). If a task supplies a `reference` answer, consistency-against-reference is added
-to the prompt (the paper's Consistency-Rate signal, lower variance than a rubric alone)."""
-import json
-import os
-import re
-import time
-
-from langchain_openai import ChatOpenAI
-
-from . import usage as usage_ledger
-
-# Reward-hacking guard: the judge must NOT be the same model as SKILLOPT_MODEL.
-# If the author and the grader share blind spots, the search learns to please the judge instead of
-# improving the skill. Default judge is a model distinct from both the reflection LM (GLM) and the
-# student (Qwen).
-# JUDGE_MODELS (comma-separated) runs an ensemble and averages, harder still to game.
-MODELS = [m.strip() for m in os.environ.get(
- "JUDGE_MODELS", os.environ.get("JUDGE_MODEL", "google/gemini-2.5-flash")).split(",") if m.strip()]
-from . import ZDR_PROVIDER, client_kwargs, skillopt_model, teacher_base_url # noqa: E402
-
-if skillopt_model() in MODELS:
- print(f"[judge] WARNING: judge model {MODELS} includes the teacher model, which invites "
- f"reward-hacking (author == grader). Set JUDGE_MODEL to a different model.", flush=True)
-
-# Failure dimensions (the general-purpose analogue of the paper's Knowledge/Tool/Clarification/Style).
-DIMENSIONS = ["correctness", "completeness", "instruction_following", "efficiency"]
-
-_PROMPT = """You are grading an AI assistant's answer to a task.
-
-TASK: {task}
-{rubric_block}{reference_block}
-ASSISTANT'S ANSWER:
-{answer}
-{code_block}
-Score the answer from 0.0 to 1.0, and write one short paragraph of concrete, actionable feedback.
-Treat any OBJECTIVE CODE CHECK above as ground truth, do not rate broken or absent code highly.
-Then classify each failure dimension as "pass" or a short (<=12 word) note on what's wrong:
-- correctness: is the core logic / API usage right?
-- completeness: does it cover the whole request, including edge cases named above?
-- instruction_following: did it do what was asked (e.g. output complete runnable code, not a description)?
-- efficiency: is it concise, without wasted or padded output?
-
-Respond with ONLY a JSON object:
-{{"score": , "feedback": "", "dimensions": {{"correctness": "...", "completeness": "...", "instruction_following": "...", "efficiency": "..."}}}}"""
-
-_llms: dict[str, ChatOpenAI] = {}
-
-
-def _get_llm(model: str):
- if model not in _llms: # built once per model, reuses the HTTP pool across many judge calls
- _llms[model] = ChatOpenAI(model=model, temperature=0, **client_kwargs(teacher_base_url()))
- return _llms[model]
-
-
-# OpenRouter phrasings that mean "your model/provider configuration can never work", retrying
-# only burns time, so explain and stop instead.
-_PERMANENT = ("no allowed providers", "no providers are available", "not a valid model",
- "no endpoints found", "is not available")
-
-
-def _config_error(exc: Exception) -> str | None:
- text = str(exc).lower()
- if any(marker in text for marker in _PERMANENT):
- pins = os.environ.get("OPENROUTER_PROVIDERS", "")
- hint = (f" You have OPENROUTER_PROVIDERS={pins}, the pinned provider may not serve this "
- f"model, or may not be ZDR-qualified for it; unset the pin or change the model."
- if pins else
- " No ZDR-qualified endpoint may exist for this model; try another model.")
- return f"OpenRouter cannot route this request: {exc}.{hint}"
- return None
-
-
-def invoke_retry(llm, messages, tries: int = 3):
- """Retry transient provider failures (corrupted responses, 5xx) with a short backoff.
- Permanent configuration errors (model/provider mismatch) fail immediately with an explanation
- instead of retrying."""
- for i in range(tries):
- try:
- return llm.invoke(messages)
- except Exception as exc:
- explained = _config_error(exc)
- if explained:
- raise SystemExit(explained) from exc
- if i == tries - 1:
- raise
- time.sleep(5 * (i + 1))
-
-
-def _extract_json(text: str) -> dict:
- """First valid JSON object with a 'score' key, robust to prose/braces around the JSON."""
- dec = json.JSONDecoder()
- for m in re.finditer(r"\{", text):
- try:
- obj, _ = dec.raw_decode(text[m.start():])
- except json.JSONDecodeError:
- continue
- if isinstance(obj, dict) and "score" in obj:
- return obj
- return {}
-
-
-def _judge_one(model: str, prompt: str) -> dict:
- msg = invoke_retry(_get_llm(model), prompt)
- usage_ledger.add("judge", getattr(msg, "usage_metadata", None))
- out = _extract_json(msg.content)
- try:
- dims = out.get("dimensions") or {}
- return {"score": max(0.0, min(1.0, float(out["score"]))), "feedback": str(out.get("feedback", "")),
- "dimensions": {d: str(dims.get(d, "pass")) for d in DIMENSIONS}}
- except (KeyError, TypeError, ValueError):
- return {"score": 0.0, "feedback": f"Judge output unparseable: {msg.content[:200]}",
- "dimensions": {d: "pass" for d in DIMENSIONS}}
-
-
-def judge(task: str, rubric: str = "", answer: str = "", reference: str = "",
- check: dict | None = None, deliverable: str | None = None) -> dict:
- """Return {score, feedback, dimensions}. With multiple JUDGE_MODELS this is an ensemble: score is
- the mean, and a dimension counts as failed if a majority of judges flag it (harder to game).
- `deliverable` (task yaml) declares the expected answer kind; non-code values skip the static
- Python check, see execcheck.judge_note."""
- rubric_block = f"GRADING RUBRIC: {rubric}\n" if rubric else ""
- reference_block = f"KNOWN-GOOD REFERENCE ANSWER (judge consistency against it): {reference}\n" if reference else ""
- from . import execcheck # objective code-validity signal to ground the judge
- code_note = execcheck.judge_note(answer, task, rubric, check_spec=check, deliverable=deliverable)
- code_block = f"\n{code_note}\n" if code_note else ""
- prompt = _PROMPT.format(task=task, answer=answer, rubric_block=rubric_block,
- reference_block=reference_block, code_block=code_block)
- results = [_judge_one(m, prompt) for m in MODELS]
- if len(results) == 1:
- return results[0]
- score = sum(r["score"] for r in results) / len(results)
- dims = {}
- for d in DIMENSIONS:
- notes = [r["dimensions"][d] for r in results if d in failed_dimensions(r["dimensions"])]
- dims[d] = notes[0] if len(notes) * 2 > len(results) else "pass" # fail only on majority
- feedback = " | ".join(f"[{m.split('/')[-1]}] {r['feedback']}" for m, r in zip(MODELS, results))
- return {"score": score, "feedback": feedback, "dimensions": dims}
-
-
-def failed_dimensions(dimensions: dict) -> list[str]:
- """Dimension names the judge did NOT mark as a clean pass."""
- return [d for d, v in dimensions.items() if str(v).strip().lower() not in ("pass", "ok", "", "n/a")]
diff --git a/optimize/usage.py b/optimize/usage.py
deleted file mode 100644
index 21beb58..0000000
--- a/optimize/usage.py
+++ /dev/null
@@ -1,103 +0,0 @@
-"""Token ledger for an optimize run: every LLM call is attributed to a role
-(rollout / judge / reflection / agent_ab) so the run can report what it actually cost ,
-including a best-effort USD estimate from OpenRouter list prices, and an optional hard
-spend cap (MAX_RUN_USD) that aborts a run before it exceeds the budget."""
-import os
-import threading
-from collections import defaultdict
-
-COUNTS: dict[str, dict[str, int]] = defaultdict(lambda: {"input": 0, "output": 0, "calls": 0})
-_LOCK = threading.RLock() # the search fans rollout+judge across a thread pool; add() re-enters for the cap
-_PRICES: dict[str, tuple[float, float]] | None = None
-
-
-def reset():
- """Start a fresh ledger, the UI process runs many optimizations; counts must not leak across runs."""
- with _LOCK:
- COUNTS.clear()
-
-
-def add(role: str, usage: dict | None):
- """usage: langchain usage_metadata ({'input_tokens','output_tokens'}) or equivalent dict."""
- if not usage:
- return
- with _LOCK:
- c = COUNTS[role]
- c["input"] += int(usage.get("input_tokens", 0))
- c["output"] += int(usage.get("output_tokens", 0))
- c["calls"] += 1
- _enforce_cap()
-
-
-def _enforce_cap():
- cap = float(os.environ.get("MAX_RUN_USD", "0") or 0)
- if not cap:
- return
- cost = estimated_cost()
- if cost is not None and cost > cap:
- raise SystemExit(f"MAX_RUN_USD exceeded: estimated ${cost:.2f} > cap ${cap:.2f}, "
- f"aborting before spending more.\n{format_report()}")
-
-
-def _openrouter_prices() -> dict[str, tuple[float, float]]:
- """model id -> (prompt, completion) USD per token from OpenRouter's public models API;
- {} on any failure (cost reporting is best-effort, never a gate on offline work)."""
- import json
- import urllib.request
- try:
- with urllib.request.urlopen("https://openrouter.ai/api/v1/models", timeout=10) as r:
- data = json.loads(r.read())["data"]
- return {m["id"]: (float(m["pricing"]["prompt"]), float(m["pricing"]["completion"]))
- for m in data if m.get("pricing")}
- except Exception:
- return {}
-
-
-def _role_models() -> dict[str, str]:
- """Which model each ledger role runs on (first judge only, for ensemble setups)."""
- from . import agent_model, skillopt_model
- teacher = skillopt_model()
- judge = (os.environ.get("JUDGE_MODELS") or
- os.environ.get("JUDGE_MODEL", "google/gemini-2.5-flash")).split(",")[0].strip()
- return {"rollout": agent_model(), "agent_ab": agent_model(),
- "judge": judge, "reflection": teacher}
-
-
-def estimated_cost() -> float | None:
- """Best-effort USD estimate for the current ledger, from OpenRouter list prices. None when
- the endpoint isn't OpenRouter or pricing is unavailable (local endpoints cost nothing)."""
- global _PRICES
- from . import is_openrouter, teacher_base_url
- if not is_openrouter(teacher_base_url()):
- return None
- if _PRICES is None:
- _PRICES = _openrouter_prices()
- if not _PRICES:
- return None
- models = _role_models()
- with _LOCK:
- return sum(c["input"] * p[0] + c["output"] * p[1]
- for role, c in COUNTS.items()
- if (p := _PRICES.get(models.get(role, ""))))
-
-
-def report() -> dict:
- out = {role: dict(c) for role, c in COUNTS.items()}
- out["total"] = {
- "input": sum(c["input"] for c in COUNTS.values()),
- "output": sum(c["output"] for c in COUNTS.values()),
- "calls": sum(c["calls"] for c in COUNTS.values()),
- }
- return out
-
-
-def format_report() -> str:
- r = report()
- lines = [f" {role:<12} {c['calls']:>4} calls {c['input']:>9,} in {c['output']:>8,} out"
- for role, c in r.items() if role != "total"]
- t = r["total"]
- lines.append(f" {'TOTAL':<12} {t['calls']:>4} calls {t['input']:>9,} in {t['output']:>8,} out")
- cost = estimated_cost()
- if cost is not None:
- lines.append(f" estimated cost: ${cost:.2f} (OpenRouter list prices)")
- return "\n".join(lines)
diff --git a/pyproject.toml b/pyproject.toml
new file mode 100644
index 0000000..bfbd524
--- /dev/null
+++ b/pyproject.toml
@@ -0,0 +1,40 @@
+[build-system]
+requires = ["setuptools>=77"]
+build-backend = "setuptools.build_meta"
+
+[project]
+name = "ingot"
+version = "0.2.0"
+description = "Release control for AI agent skills"
+readme = "README.md"
+requires-python = ">=3.12"
+license = "Apache-2.0"
+
+# Deliberately minimal. `ingot list` reads SKILL.md folders through ingot.mcp_server.registry, whose only
+# non-stdlib import is PyYAML, and the deterministic commands that follow must keep that property:
+# a developer inspecting a skill should not install FastAPI, ONNX, LangGraph, Langfuse, or the
+# optimizer to do it. The server, UI, and optimizer keep installing from requirements.txt, which
+# stays the pinned set CI exercises -- this is a packaging shim over that, not a replacement for it.
+dependencies = [
+ "pyyaml==6.0.3",
+]
+
+[project.scripts]
+ingot = "ingot.cli:main"
+
+[tool.setuptools]
+# One installed name. `ingot add` reaches the quarantine through `ingot.optimize.ingress`, so the
+# optimizer package is installed too -- without it the command works under pytest, where the repo
+# root is the working directory, and fails for anyone running the installed script from anywhere
+# else.
+#
+# `mcp_server` and `optimize` were top-level packages until they moved under `ingot`. Both names are
+# far too generic to claim on PyPI, and a compatibility shim would have been self-defeating: a shim
+# named `optimize` still claims `optimize`. Nothing outside this repository imported them, so they
+# were moved rather than aliased.
+#
+# `ui` and `agent` stay repo-root modules run with `python -m` from /app, as the Dockerfile and
+# Compose already do. They are not installed, so they claim no name; moving them under `ingot` would
+# make them installed and drag FastAPI and the agent scaffold into a CLI whose whole point is that
+# `ingot list` works in a bare virtualenv.
+packages = ["ingot", "ingot.mcp_server", "ingot.optimize"]
diff --git a/scripts/claude_langfuse_smoke.sh b/scripts/claude_langfuse_smoke.sh
index 2b772bf..85a13e3 100755
--- a/scripts/claude_langfuse_smoke.sh
+++ b/scripts/claude_langfuse_smoke.sh
@@ -3,6 +3,7 @@ set -eu
LF_URL=${LANGFUSE_BASE_URL:-http://localhost:3100}
LF_DOCKER_URL=${SMOKE_LANGFUSE_DOCKER_URL:-http://host.docker.internal:3100}
+MCP_URL=${INGOT_MCP_URL:-http://localhost:8000/mcp}
LF_PK=${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo}
LF_SK=${LANGFUSE_SECRET_KEY:-sk-lf-local-demo}
MARKER="ingot-claude-smoke-$(date +%s)"
@@ -12,7 +13,7 @@ trap 'rm -rf "$WORKDIR"' EXIT INT TERM
command -v claude >/dev/null 2>&1 || { echo "error: claude is required" >&2; exit 1; }
curl -sSf "$LF_URL/api/public/health" >/dev/null
-curl -sS -o /dev/null http://localhost:8000/mcp || [ "$?" -eq 52 ]
+curl -sS -o /dev/null "$MCP_URL" || [ "$?" -eq 52 ]
(cd "$WORKDIR" && CC_LANGFUSE_DEBUG=1 claude -p --output-format stream-json --verbose \
--permission-mode bypassPermissions --allowedTools=mcp__ingot__route_and_load \
@@ -65,7 +66,7 @@ docker run --rm --add-host host.docker.internal:host-gateway -v "$PWD:/app" -w /
-e LANGFUSE_SECRET_KEY="$LF_SK" \
-e SMOKE_MARKER="$MARKER" ingot-mcp python -c '
import os
-from optimize.mine import fetch_traces
+from ingot.optimize.mine import fetch_traces
marker = os.environ["SMOKE_MARKER"]
for trace in fetch_traces(50):
diff --git a/scripts/codex_langfuse_smoke.sh b/scripts/codex_langfuse_smoke.sh
index 1e643a2..97aaf41 100755
--- a/scripts/codex_langfuse_smoke.sh
+++ b/scripts/codex_langfuse_smoke.sh
@@ -56,7 +56,7 @@ docker run --rm --add-host host.docker.internal:host-gateway -v "$PWD:/app" -w /
-e LANGFUSE_SECRET_KEY="$LF_SK" \
-e SMOKE_MARKER="$MARKER" ingot-mcp python -c '
import os
-from optimize.mine import fetch_traces
+from ingot.optimize.mine import fetch_traces
marker = os.environ["SMOKE_MARKER"]
for trace in fetch_traces(50):
diff --git a/scripts/codex_setup.sh b/scripts/codex_setup.sh
index aff0a93..8d5dc11 100755
--- a/scripts/codex_setup.sh
+++ b/scripts/codex_setup.sh
@@ -36,9 +36,13 @@ require_command node
require_command python3
CODEX_VERSION=$(codex --version | awk '{print $2}')
-CODEX_MINOR=$(printf '%s' "$CODEX_VERSION" | awk -F. '{print $2}')
-if [ "${CODEX_VERSION%%.*}" = "0" ] && [ "${CODEX_MINOR:-0}" -lt 128 ]; then
- echo "error: Codex 0.128 or newer is required by the Langfuse plugin" >&2
+# The 0.128 floor stands for one thing: the `codex plugin` subcommand this script
+# installs through. Probe for that directly. A locally built codex stamps no
+# version — `codex-cli 0.0.0` parses as older than every release while carrying
+# the capability, so a version comparison rejects a build that works.
+if ! codex plugin --help >/dev/null 2>&1; then
+ echo "error: this codex build has no 'plugin' subcommand, which the Langfuse plugin needs" >&2
+ echo " (released builds carry it from 0.128; found version $CODEX_VERSION)" >&2
exit 1
fi
NODE_MAJOR=$(node -p 'process.versions.node.split(".")[0]')
@@ -54,6 +58,12 @@ MARKETPLACE_OK=0
codex plugin marketplace list --json 2>/dev/null | grep -F 'codex-observability-plugin' >/dev/null && MARKETPLACE_OK=1
PLUGIN_OK=0
codex plugin list --json 2>/dev/null | grep -F 'tracing@codex-observability-plugin' >/dev/null && PLUGIN_OK=1
+# Codex runs a hook only after a human has reviewed it once and Codex has persisted
+# the approval. Until then the Stop hook is skipped in silence: the plugin reports
+# installed and enabled, and no trace is ever written.
+TRUST_OK=0
+grep -q 'tracing@codex-observability-plugin:hooks/hooks.json:stop' \
+ "${CODEX_HOME:-$HOME/.codex}/config.toml" 2>/dev/null && TRUST_OK=1
if [ "$MODE" = "doctor" ]; then
echo "Codex version: $CODEX_VERSION"
@@ -62,13 +72,29 @@ if [ "$MODE" = "doctor" ]; then
[ "$MARKETPLACE_OK" = "1" ] && echo "Langfuse marketplace: installed" || echo "Langfuse marketplace: missing"
[ "$PLUGIN_OK" = "1" ] && echo "Langfuse plugin: installed" || echo "Langfuse plugin: missing"
[ -f "$LANGFUSE_CONFIG" ] && echo "Langfuse config: present at $LANGFUSE_CONFIG" || echo "Langfuse config: missing"
- if command -v curl >/dev/null 2>&1 && curl -sSf "$LF_URL/api/public/health" >/dev/null 2>&1; then
+ [ "$TRUST_OK" = "1" ] && echo "Stop hook: trusted" || echo "Stop hook: not yet trusted — run codex interactively once and approve it"
+ # Probe with Node rather than curl. Node is what uploads the traces, and it reads a
+ # CA store of its own: against a private CA, curl can report a healthy endpoint that
+ # Node cannot reach. Whichever curl happens to sit first on PATH answers for a third
+ # trust store, so it speaks for nothing here.
+ if node -e '
+const url = process.argv[1] + "/api/public/health";
+const lib = url.startsWith("https:") ? require("https") : require("http");
+const req = lib.get(url, { timeout: 10000 }, (res) => {
+ res.resume();
+ process.exit(res.statusCode === 200 ? 0 : 1);
+});
+req.on("timeout", () => { req.destroy(); process.exit(1); });
+req.on("error", () => process.exit(1));
+' "$LF_URL" 2>/dev/null; then
echo "Langfuse endpoint: healthy at $LF_URL"
else
- echo "Langfuse endpoint: unreachable at $LF_URL"
+ echo "Langfuse endpoint: unreachable from Node at $LF_URL"
+ echo " (if curl reaches it, Node is missing the CA — set NODE_EXTRA_CA_CERTS to the root)"
fi
trap - 0
- [ "$MCP_OK" = "1" ] && [ "$MARKETPLACE_OK" = "1" ] && [ "$PLUGIN_OK" = "1" ] && [ -f "$LANGFUSE_CONFIG" ]
+ [ "$MCP_OK" = "1" ] && [ "$MARKETPLACE_OK" = "1" ] && [ "$PLUGIN_OK" = "1" ] \
+ && [ -f "$LANGFUSE_CONFIG" ] && [ "$TRUST_OK" = "1" ]
exit
fi
diff --git a/scripts/fetch_skills.sh b/scripts/fetch_skills.sh
index c4100b7..7a7bd14 100755
--- a/scripts/fetch_skills.sh
+++ b/scripts/fetch_skills.sh
@@ -1,7 +1,13 @@
#!/usr/bin/env bash
-# Fetch example Agent Skills (SKILL.md format) into ./skills/. Every source is OPTIONAL and nothing
-# is redistributed in this repo, each source is cloned from upstream, its skills copied in, and the
-# clone deleted, so skills stay under their own upstream licenses.
+# Fetch example Agent Skills (SKILL.md format) and QUARANTINE them for review. Every source is
+# OPTIONAL and nothing is redistributed in this repo, each source is cloned from upstream, its
+# skills submitted for review, and the clone deleted, so skills stay under their own upstream
+# licenses.
+#
+# Nothing here is served. This used to copy third-party directories straight into ./skills, where
+# the server picked them up on the next restart — an install, with no review and no record of what
+# changed. Every package now goes through `ingot add`, which reviews it, records where it came
+# from, and leaves the served library byte-identical until a human approves it.
#
# Usage:
# scripts/fetch_skills.sh all # every source below
@@ -15,24 +21,29 @@
# trailofbits trailofbits/skills security analysis CC-BY-SA-4.0
set -euo pipefail
-SKILLS="$(cd "$(dirname "$0")/.." && pwd)/skills"
-mkdir -p "$SKILLS"
+if ! command -v ingot >/dev/null 2>&1; then
+ echo "fetch_skills.sh needs the \`ingot\` command (pip install -e .)." >&2
+ echo "It quarantines each package for review instead of copying it into the served library." >&2
+ exit 1
+fi
-# copy up to $cap skill dirs (0 = no cap) from a freshly-cloned repo, skipping any that already exist
+# quarantine up to $cap skill dirs (0 = no cap) from a freshly-cloned repo
fetch() { # repo cap license
- local repo="$1" cap="$2" license="$3" tmp added=0
+ local repo="$1" cap="$2" license="$3" tmp added=0 refused=0
tmp="$(mktemp -d)"
echo "[fetch] cloning $repo …"
git clone --depth 1 -q "https://github.com/$repo" "$tmp/repo"
while IFS= read -r skill_md; do
- local dir name; dir="$(dirname "$skill_md")"; name="$(basename "$dir")"
- [ -e "$SKILLS/$name" ] && continue # never clobber
+ local dir; dir="$(dirname "$skill_md")"
[ "$cap" -ne 0 ] && [ "$added" -ge "$cap" ] && break # respect the cap
- cp -r "$dir" "$SKILLS/$name" # whole dir: SKILL.md + bundled files
- added=$((added + 1))
+ if ingot add "file:$dir" >/dev/null; then # whole dir: SKILL.md + bundled files
+ added=$((added + 1))
+ else
+ refused=$((refused + 1)) # already present, or refused on review
+ fi
done < <(find "$tmp/repo" -name SKILL.md | sort)
rm -rf "$tmp" # remove the clone
- echo "[fetch] $repo: added $added skills (license: $license)"
+ echo "[fetch] $repo: quarantined $added, refused or skipped $refused (license: $license)"
}
# source lookup as a case statement (not `declare -A`): macOS ships bash 3.2, which has no
@@ -59,5 +70,6 @@ for t in "${targets[@]}"; do
fetch $spec
done
-echo "[fetch] $(find "$SKILLS" -name SKILL.md | wc -l | tr -d ' ') skills now in $SKILLS"
-echo "[fetch] restart the server to pick them up: docker compose restart mcp"
+echo "[fetch] nothing is served yet. Review what arrived and approve what you want:"
+echo "[fetch] ingot list # unchanged until an approval publishes"
+echo "[fetch] open http://localhost:8080 # the change-control console"
diff --git a/scripts/managed_smoke.sh b/scripts/managed_smoke.sh
new file mode 100755
index 0000000..fc2f2e4
--- /dev/null
+++ b/scripts/managed_smoke.sh
@@ -0,0 +1,82 @@
+#!/bin/sh
+# Prove the managed stack's one-writer invariant against real containers.
+#
+# scripts/managed_smoke.sh
+#
+# The static checks (tests/test_compose_managed.py, `docker compose config`) verify what the
+# tracked configuration declares. This verifies what Docker actually enforces, which is the only
+# thing that makes the claim true: every non-publisher service must fail to write the served
+# library, and the publisher must succeed.
+#
+# Run on dellpromax 2026-08-19 (Docker 29.7.2), passing, with the negative control failing as it
+# must. The Mac it was written on has no daemon; run it anywhere that does before repeating the
+# control-plane claim.
+set -eu
+
+PROJECT=${MANAGED_SMOKE_PROJECT:-ingot-managed-smoke}
+COMPOSE="docker compose -p $PROJECT"
+
+# Every check below reaches its container through `docker compose exec`, so this stack needs no
+# host port. Asking for one anyway makes the run fail on any machine that already serves something
+# on 8000 or 8080 -- a shared Docker host, or a box already running Ingot. `0` takes whatever the
+# kernel has free.
+INGOT_MCP_PORT=${INGOT_MCP_PORT:-0}
+INGOT_UI_PORT=${INGOT_UI_PORT:-0}
+export INGOT_MCP_PORT INGOT_UI_PORT
+
+cleanup() {
+ if [ "${MANAGED_SMOKE_KEEP:-0}" != "1" ]; then
+ $COMPOSE down -v --remove-orphans
+ fi
+}
+trap cleanup EXIT INT TERM
+
+# `ui` pulls in `mcp`; the publisher owns the vault. Langfuse is not part of this invariant.
+$COMPOSE up -d --build publisher mcp ui
+
+# The publisher initializes the vault on start; wait for the repository rather than racing it.
+elapsed=0
+until $COMPOSE exec -T publisher git -C /app/vault rev-parse HEAD >/dev/null 2>&1; do
+ elapsed=$((elapsed + 2))
+ [ "$elapsed" -ge 60 ] && { echo "error: the publisher never initialized /app/vault" >&2; exit 1; }
+ sleep 2
+done
+
+failed=0
+
+for service in mcp ui; do
+ if $COMPOSE exec -T "$service" sh -c 'touch /app/skills/.smoke-write' 2>/dev/null; then
+ echo "FAIL: $service can write the served library; it is not read-only" >&2
+ failed=1
+ else
+ echo "ok: $service cannot write /app/skills"
+ fi
+done
+
+if $COMPOSE exec -T publisher sh -c 'touch /app/vault/.smoke-write && rm /app/vault/.smoke-write'; then
+ echo "ok: the publisher can write /app/vault"
+else
+ echo "FAIL: the publisher cannot write its own vault" >&2
+ failed=1
+fi
+
+# The served library the other services read is the vault the publisher writes. A stack where they
+# are different directories serves stale bytes and reports no drift, because nothing compares them.
+vault_head=$($COMPOSE exec -T publisher git -C /app/vault rev-parse HEAD)
+served_head=$($COMPOSE exec -T mcp sh -c 'git -C /app/skills rev-parse HEAD' 2>/dev/null || echo "")
+if [ "$vault_head" != "$served_head" ]; then
+ echo "FAIL: mcp serves $served_head but the publisher's vault is at $vault_head" >&2
+ failed=1
+else
+ echo "ok: mcp serves the publisher's vault at $vault_head"
+fi
+
+if $COMPOSE exec -T mcp ingot status; then
+ echo "ok: ingot status reports MANAGED inside the stack"
+else
+ echo "FAIL: ingot status did not report MANAGED inside the stack" >&2
+ failed=1
+fi
+
+[ "$failed" -eq 0 ] || { echo "managed smoke FAILED" >&2; exit 1; }
+echo "managed smoke passed: one writer, and it is the publisher"
diff --git a/tests/conftest.py b/tests/conftest.py
index 2bce5cd..5abe890 100644
--- a/tests/conftest.py
+++ b/tests/conftest.py
@@ -11,8 +11,23 @@
@pytest.fixture(autouse=True)
-def _isolated_local_skills_root(tmp_path_factory, monkeypatch):
- monkeypatch.setattr("mcp_server.registry.SKILLS_DIR", tmp_path_factory.mktemp("local-skills"))
+def _isolated_state(tmp_path_factory, monkeypatch):
+ """No test may touch the developer's real state directory, or the checkout's.
+
+ State resolves through `ingot.paths` at call time and defaults to an XDG directory, so a test
+ that reaches a default instead of a fixture would write a real review queue and real receipts
+ into `~/.local/state/ingot` and pass while doing it. One per-test INGOT_HOME contains every one
+ of them; specific overrides are cleared so an environment variable the developer happens to
+ have exported cannot reach in either.
+
+ This also replaces the old SKILLS_DIR patch. `configured_roots` always puts the local authoring
+ root first, even ahead of an explicit root, so a test that loads skills would otherwise see
+ whatever `scripts/fetch_skills.sh` left in the checkout (first caught on a machine with 72
+ fetched skills: 9 failures and a multi-minute embedding stall)."""
+ from ingot import paths
+ monkeypatch.setenv(paths.HOME, str(tmp_path_factory.mktemp("state")))
+ for name in (paths.LIBRARY, paths.RUNS, paths.TASKS, paths.VAULT, *paths.LEGACY.values()):
+ monkeypatch.delenv(name, raising=False)
monkeypatch.delenv("SKILL_ROUTER_PATHS", raising=False)
diff --git a/tests/fixtures/agy/judge-stream.jsonl b/tests/fixtures/agy/judge-stream.jsonl
new file mode 100644
index 0000000..60ce466
--- /dev/null
+++ b/tests/fixtures/agy/judge-stream.jsonl
@@ -0,0 +1,9 @@
+{"event":"init","conversation_id":"","init":{"model":"gemini-3.6-flash-medium","cwd":"","tools":["ask_permission","ask_question","browser_click_element","browser_drag_pixel_to_pixel","browser_get_dom","browser_get_network_request","browser_input","browser_list_network_requests","browser_mouse_down","browser_mouse_up","browser_move_mouse","browser_press_key","browser_refresh_page","browser_resize_window","browser_scroll","browser_scroll_dom","browser_select_option","browser_subagent","call_mcp_tool","capture_browser_console_logs","capture_browser_screenshot","click_browser_pixel","command_status","define_subagent","delete_knowledge","execute_browser_javascript","find_by_name","finish","generate_image","grep_search","invoke_subagent","list_browser_pages","list_permissions","list_resources","manage_inbox","manage_subagents","manage_task","multi_replace_file_content","notebook_edit","notebook_execution","open_browser_url","read_browser_page","read_resource","read_url_content","replace_file_content","run_command","schedule","search_web","sed_file","send_command_input","send_message","view_file","wait","wait_5_seconds","write_to_file"],"permission_mode":"request-review","json_schema":{"type":"object","additionalProperties":false,"required":["items","feedback"],"properties":{"items":{"type":"object","additionalProperties":false,"required":["probe"],"properties":{"probe":{"type":"object","additionalProperties":false,"required":["verdict","note"],"properties":{"verdict":{"enum":["pass","partial","fail"]},"note":{"type":"string"}}}}},"feedback":{"type":"string"}}}}}
+{"event":"step_update","step_update":{"conversation_id":"","step_index":0,"state":"DONE","step_type":"user_input"}}
+{"event":"step_update","step_update":{"conversation_id":"","step_index":1,"state":"DONE","step_type":"unknown","duration_seconds":0.000501423}}
+{"event":"step_update","step_update":{"conversation_id":"","step_index":2,"state":"DONE","step_type":"agent_response","text_delta":"{\"feedback\":\"ok\",\"items\":[{\"probe\":{\"note\":\"\",\"verdict\":\"pass\"}}]}\n","duration_seconds":2.94894296,"usage":{"input_tokens":18403,"output_tokens":1374,"thinking_tokens":1342,"cache_read_tokens":0,"total_tokens":19777}}}
+{"event":"step_update","step_update":{"conversation_id":"","step_index":3,"state":"DONE","step_type":"error_message"}}
+{"event":"step_update","step_update":{"conversation_id":"","step_index":4,"state":"DONE","step_type":"agent_response","text_delta":"{\"feedback\":\"ok\",\"items\":{\"probe\":{\"note\":\"\",\"verdict\":\"pass\"}}}\n","duration_seconds":1.035306864,"usage":{"input_tokens":3680,"output_tokens":464,"thinking_tokens":434,"cache_read_tokens":16287,"total_tokens":4144}}}
+{"event":"step_update","step_update":{"conversation_id":"","step_index":5,"state":"DONE","step_type":"finish","duration_seconds":0.055575545}}
+{"event":"step_update","step_update":{"conversation_id":"","step_index":6,"state":"DONE","step_type":"checkpoint","duration_seconds":0.565171204,"usage":{"input_tokens":142,"output_tokens":4,"thinking_tokens":0,"cache_read_tokens":0,"total_tokens":146}}}
+{"event":"result","result":{"conversation_id":"","status":"SUCCESS","response":"{\"feedback\":\"ok\",\"items\":[{\"probe\":{\"note\":\"\",\"verdict\":\"pass\"}}]}\n{\"feedback\":\"ok\",\"items\":{\"probe\":{\"note\":\"\",\"verdict\":\"pass\"}}}\n","duration_seconds":4.58409492,"num_turns":1,"structured_output":{"feedback":"ok","items":{"probe":{"note":"","verdict":"pass"}}},"json_schema":{"type":"object","additionalProperties":false,"required":["items","feedback"],"properties":{"items":{"type":"object","additionalProperties":false,"required":["probe"],"properties":{"probe":{"type":"object","additionalProperties":false,"required":["verdict","note"],"properties":{"verdict":{"enum":["pass","partial","fail"]},"note":{"type":"string"}}}}},"feedback":{"type":"string"}}},"usage":{"input_tokens":22225,"output_tokens":1842,"thinking_tokens":1776,"cache_read_tokens":16287,"total_tokens":24067}}}
diff --git a/tests/fixtures/harbor/deepseek-models.json b/tests/fixtures/harbor/deepseek-models.json
new file mode 100644
index 0000000..b14f102
--- /dev/null
+++ b/tests/fixtures/harbor/deepseek-models.json
@@ -0,0 +1,28 @@
+{
+ "object": "list",
+ "data": [
+ {
+ "id": "deepseek-v4-flash",
+ "object": "model",
+ "owned_by": "vllm",
+ "root": "deepseek-ai/DeepSeek-V4-Flash-0731",
+ "parent": null,
+ "max_model_len": 1048576,
+ "permission": [
+ {
+ "id": "modelperm-a8f876e158347e2a",
+ "object": "model_permission",
+ "allow_create_engine": false,
+ "allow_sampling": true,
+ "allow_logprobs": true,
+ "allow_search_indices": false,
+ "allow_view": true,
+ "allow_fine_tuning": false,
+ "organization": "*",
+ "group": null,
+ "is_blocking": false
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/harbor/langfuse-trial/agent/trajectory.json b/tests/fixtures/harbor/langfuse-trial/agent/trajectory.json
new file mode 100644
index 0000000..73ef6b1
--- /dev/null
+++ b/tests/fixtures/harbor/langfuse-trial/agent/trajectory.json
@@ -0,0 +1,33 @@
+{
+ "schema_version": "ATIF-v1.1",
+ "session_id": "fixture-session",
+ "agent": {
+ "name": "codex",
+ "version": "fixture-version",
+ "model_name": "fixture-model",
+ "extra": {
+ "originator": "fixture",
+ "cwd": "/fixtures/workspace"
+ }
+ },
+ "steps": [
+ {
+ "step_id": 1,
+ "timestamp": "2000-01-01T00:00:00Z",
+ "source": "user",
+ "message": "Create the deterministic fixture deliverable."
+ },
+ {
+ "step_id": 2,
+ "timestamp": "2000-01-01T00:00:00Z",
+ "source": "agent",
+ "message": "I will inspect the fixture workspace."
+ },
+ {
+ "step_id": 3,
+ "timestamp": "2000-01-01T00:00:01Z",
+ "source": "agent",
+ "message": "The fixture deliverable is ready."
+ }
+ ]
+}
diff --git a/tests/fixtures/harbor/langfuse-trial/exception.txt b/tests/fixtures/harbor/langfuse-trial/exception.txt
new file mode 100644
index 0000000..4c98819
--- /dev/null
+++ b/tests/fixtures/harbor/langfuse-trial/exception.txt
@@ -0,0 +1,4 @@
+Traceback (most recent call last):
+ File "/fixtures/harbor/runner.py", line 1, in run
+ raise AgentSetupTimeoutError("sanitized fixture failure")
+AgentSetupTimeoutError: sanitized fixture failure
diff --git a/tests/fixtures/harbor/langfuse-trial/result.json b/tests/fixtures/harbor/langfuse-trial/result.json
new file mode 100644
index 0000000..3a1e3f6
--- /dev/null
+++ b/tests/fixtures/harbor/langfuse-trial/result.json
@@ -0,0 +1,124 @@
+{
+ "id": "00000000-0000-0000-0000-000000000001",
+ "task_name": "ingot/fixture-h0",
+ "trial_name": "fixture-h0__attempt-1",
+ "trial_uri": "file:///fixtures/harbor/langfuse-trial",
+ "task_id": {
+ "path": "/fixtures/tasks/fixture-h0"
+ },
+ "source": "fixture",
+ "task_checksum": "fixture-task-checksum",
+ "config": {
+ "task": {
+ "path": "/fixtures/tasks/fixture-h0",
+ "git_url": null,
+ "git_commit_id": null,
+ "name": null,
+ "ref": null,
+ "overwrite": false,
+ "download_dir": null,
+ "source": "fixture"
+ },
+ "trial_name": "fixture-h0__attempt-1",
+ "trials_dir": "/fixtures/trials",
+ "install_only": false,
+ "timeout_multiplier": 1.0,
+ "agent_timeout_multiplier": null,
+ "verifier_timeout_multiplier": null,
+ "agent_setup_timeout_multiplier": 3.0,
+ "environment_build_timeout_multiplier": 4.0,
+ "agent": {
+ "name": "codex",
+ "import_path": null,
+ "model_name": "fixture-model",
+ "n_concurrent": null,
+ "concurrency_group": null,
+ "skills": [
+ "/fixtures/skills/private-skill/SKILL.md"
+ ],
+ "override_timeout_sec": null,
+ "override_setup_timeout_sec": null,
+ "max_timeout_sec": null,
+ "resume_trajectory": false,
+ "load_trajectory": null,
+ "extra_allowed_hosts": [],
+ "kwargs": {},
+ "env": {
+ "OPENAI_API_KEY": "fixture-secret-must-not-export",
+ "OPENAI_BASE_URL": "https://fixture-endpoint.invalid/v1"
+ },
+ "mcp_servers": []
+ },
+ "environment": {
+ "type": "docker",
+ "import_path": null,
+ "force_build": false,
+ "delete": true,
+ "cpu_enforcement_policy": "none",
+ "memory_enforcement_policy": "none",
+ "override_cpus": null,
+ "override_memory_mb": null,
+ "override_storage_mb": null,
+ "override_gpus": null,
+ "override_tpu": null,
+ "mounts": null,
+ "extra_docker_compose": [],
+ "kwargs": {},
+ "extra_allowed_hosts": []
+ },
+ "verifier": {
+ "override_timeout_sec": null,
+ "max_timeout_sec": null,
+ "disable": false
+ },
+ "artifacts": [],
+ "extra_instruction_paths": [],
+ "job_id": "fixture-job"
+ },
+ "agent_info": {
+ "name": "codex",
+ "version": "fixture-version",
+ "model_info": {
+ "name": "fixture-model",
+ "provider": null
+ }
+ },
+ "agent_result": {
+ "n_input_tokens": 120,
+ "n_cache_tokens": 40,
+ "n_output_tokens": 30,
+ "cost_usd": 0.01,
+ "rollout_details": null,
+ "metadata": null
+ },
+ "verifier_result": {
+ "rewards": {
+ "reward": 1.0
+ }
+ },
+ "exception_info": {
+ "exception_type": "AgentSetupTimeoutError",
+ "exception_message": "sanitized fixture failure",
+ "exception_traceback": "sanitized fixture traceback",
+ "occurred_at": "2000-01-01T00:00:00Z"
+ },
+ "started_at": "2000-01-01T00:00:00Z",
+ "finished_at": "2000-01-01T00:00:01Z",
+ "environment_setup": {
+ "started_at": "2000-01-01T00:00:00Z",
+ "finished_at": "2000-01-01T00:00:00Z"
+ },
+ "agent_setup": {
+ "started_at": "2000-01-01T00:00:00Z",
+ "finished_at": "2000-01-01T00:00:01Z"
+ },
+ "agent_execution": {
+ "started_at": null,
+ "finished_at": null
+ },
+ "verifier": {
+ "started_at": "2000-01-01T00:00:01Z",
+ "finished_at": "2000-01-01T00:00:01Z"
+ },
+ "step_results": null
+}
diff --git a/tests/fixtures/harbor/langfuse-trial/verifier/solution/_objective_check.txt b/tests/fixtures/harbor/langfuse-trial/verifier/solution/_objective_check.txt
new file mode 100644
index 0000000..71f7e19
--- /dev/null
+++ b/tests/fixtures/harbor/langfuse-trial/verifier/solution/_objective_check.txt
@@ -0,0 +1,7 @@
+Fixture objective check
+
+- deterministic output exists
+- no external endpoint was contacted
+- no credential was used
+
+PASS
diff --git a/tests/fixtures/harbor/langfuse-trial/verifier/test-stdout.txt b/tests/fixtures/harbor/langfuse-trial/verifier/test-stdout.txt
new file mode 100644
index 0000000..e69de29
diff --git a/tests/fixtures/harbor/orin-models.json b/tests/fixtures/harbor/orin-models.json
new file mode 100644
index 0000000..7f96b26
--- /dev/null
+++ b/tests/fixtures/harbor/orin-models.json
@@ -0,0 +1,17 @@
+{
+ "object": "list",
+ "data": [
+ {
+ "id": "ablit35b",
+ "aliases": ["ablit35b"],
+ "object": "model",
+ "owned_by": "llamacpp",
+ "meta": {
+ "n_ctx": 65536,
+ "n_ctx_train": 262144,
+ "n_params": 34660610688,
+ "ftype": "Q4_K - Medium"
+ }
+ }
+ ]
+}
diff --git a/tests/fixtures/harbor/qwen-models.json b/tests/fixtures/harbor/qwen-models.json
new file mode 100644
index 0000000..0ff0fef
--- /dev/null
+++ b/tests/fixtures/harbor/qwen-models.json
@@ -0,0 +1,28 @@
+{
+ "object": "list",
+ "data": [
+ {
+ "id": "dot-backbone",
+ "object": "model",
+ "owned_by": "vllm",
+ "root": "Qwen/Qwen3.6-27B-FP8",
+ "parent": null,
+ "max_model_len": 163840,
+ "permission": [
+ {
+ "id": "modelperm-bf3337a2a0df3e11",
+ "object": "model_permission",
+ "allow_create_engine": false,
+ "allow_sampling": true,
+ "allow_logprobs": true,
+ "allow_search_indices": false,
+ "allow_view": true,
+ "allow_fine_tuning": false,
+ "organization": "*",
+ "group": null,
+ "is_blocking": false
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/harbor/qwen-size-matrix.json b/tests/fixtures/harbor/qwen-size-matrix.json
new file mode 100644
index 0000000..2120f8f
--- /dev/null
+++ b/tests/fixtures/harbor/qwen-size-matrix.json
@@ -0,0 +1,15 @@
+{
+ "judge": "agy/gemini-3.6-flash-medium",
+ "exploratory": true,
+ "rankable": false,
+ "harnesses": {
+ "aider@qwen35-0.8b": {"harness":"aider","model":"qwen35-0.8b","family":"Qwen3.5","parameter_billions":0.8,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":-0.04,"skill_mean":0.42,"control_mean":0.46,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false},
+ "pi@qwen35-0.8b": {"harness":"pi","model":"qwen35-0.8b","family":"Qwen3.5","parameter_billions":0.8,"quantization":"fp8-load","tool_parser":"qwen3_coder","error":"canary failed","exploratory":true,"rankable":false},
+ "terminus-2@qwen35-2b": {"harness":"terminus-2","model":"qwen35-2b","family":"Qwen3.5","parameter_billions":2.0,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":0.03,"skill_mean":0.51,"control_mean":0.48,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false},
+ "aider@qwen35-4b": {"harness":"aider","model":"qwen35-4b","family":"Qwen3.5","parameter_billions":4.0,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":0.08,"skill_mean":0.57,"control_mean":0.49,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false},
+ "pi@qwen35-4b": {"harness":"pi","model":"qwen35-4b","family":"Qwen3.5","parameter_billions":4.0,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":0.05,"skill_mean":0.55,"control_mean":0.50,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false},
+ "terminus-2@qwen35-9b": {"harness":"terminus-2","model":"qwen35-9b","family":"Qwen3.5","parameter_billions":9.0,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":0.12,"skill_mean":0.63,"control_mean":0.51,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false},
+ "aider@dot-backbone": {"harness":"aider","model":"dot-backbone","family":"Qwen3.6","parameter_billions":27.0,"quantization":"fp8-published","tool_parser":"qwen3_xml","lift":0.16,"skill_mean":0.68,"control_mean":0.52,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false},
+ "pi@dot-backbone": {"harness":"pi","model":"dot-backbone","family":"Qwen3.6","parameter_billions":27.0,"quantization":"fp8-published","tool_parser":"qwen3_xml","lift":0.14,"skill_mean":0.66,"control_mean":0.52,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false}
+ }
+}
diff --git a/tests/test_acceptance.py b/tests/test_acceptance.py
index 85d0843..4ff49ae 100644
--- a/tests/test_acceptance.py
+++ b/tests/test_acceptance.py
@@ -3,7 +3,7 @@
import pytest
-from optimize.acceptance import classify, evaluate, load_criteria
+from ingot.optimize.acceptance import classify, evaluate, load_criteria
def test_load_criteria_parses_forbid_and_skips_empty(tmp_path):
diff --git a/tests/test_acquire.py b/tests/test_acquire.py
new file mode 100644
index 0000000..f1b3eaa
--- /dev/null
+++ b/tests/test_acquire.py
@@ -0,0 +1,181 @@
+"""Offline Git fixtures for public GitHub acquisition.
+
+The tests replace only the OWNER/REPO-to-URL mapping. Resolution, clone, tree inspection, and raw
+blob reads all run through real Git against a local repository.
+"""
+import subprocess
+from pathlib import Path
+
+import pytest
+
+from ingot import acquire
+
+
+def _git(path: Path, *args: str) -> str:
+ result = subprocess.run(["git", "-C", str(path), *args], check=True,
+ capture_output=True, text=True)
+ return result.stdout.strip()
+
+
+def _remote(tmp_path: Path, *, asset: bytes = b"first bytes") -> tuple[Path, str]:
+ remote = tmp_path / "remote"
+ subprocess.run(["git", "init", "-b", "main", str(remote)], check=True,
+ capture_output=True)
+ _git(remote, "config", "user.email", "fixture@example.test")
+ _git(remote, "config", "user.name", "Fixture")
+ skill = remote / "skills" / "pdf"
+ skill.mkdir(parents=True)
+ (skill / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\n\nUse this skill.\n",
+ encoding="utf-8")
+ (skill / "asset.bin").write_bytes(asset)
+ (remote / "outside.txt").write_text("not part of the package", encoding="utf-8")
+ _git(remote, "add", ".")
+ _git(remote, "commit", "-m", "Create fixture")
+ return remote, _git(remote, "rev-parse", "HEAD")
+
+
+def test_github_acquisition_resolves_commit_and_checks_out_only_the_selected_package(
+ tmp_path, monkeypatch):
+ """Break caught: recording an unverified ref, or admitting the repository root."""
+ remote, commit = _remote(tmp_path)
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+
+ package, provenance = acquire.github(
+ "acme/skills", ref="HEAD", subdirectory="skills/pdf", destination=tmp_path / "clone")
+
+ assert package == tmp_path / "clone" / "package"
+ assert (package / "asset.bin").read_bytes() == b"first bytes"
+ assert not (package / "outside.txt").exists()
+ assert provenance == {
+ "repository": "acme/skills",
+ "ref": "HEAD",
+ "commit": commit,
+ "subdirectory": "skills/pdf",
+ }
+
+
+@pytest.mark.parametrize("subdirectory", ["../pdf", "/skills/pdf", "skills\\pdf", "skills/\npdf"])
+def test_github_acquisition_refuses_a_subdirectory_that_can_escape_or_change_spelling(
+ tmp_path, subdirectory):
+ """Break caught: joining an unchecked remote-controlled path onto the clone root."""
+ with pytest.raises(ValueError, match="path|root|portable"):
+ acquire.github("acme/skills", ref="HEAD", subdirectory=subdirectory,
+ destination=tmp_path / "clone")
+
+
+@pytest.mark.parametrize("repository", ["acme", "acme/skills/extra", "../skills", "acme\\skills"])
+def test_github_acquisition_refuses_a_non_owner_repository_name(tmp_path, repository):
+ """Break caught: accepting a URL or option where the public OWNER/REPO locator is required."""
+ with pytest.raises(ValueError, match="OWNER/REPO"):
+ acquire.github(repository, ref="HEAD", subdirectory="skills/pdf",
+ destination=tmp_path / "clone")
+
+
+def test_github_acquisition_refuses_a_repository_with_a_git_suffix(tmp_path, monkeypatch):
+ """Break caught: accepting repo.git and then fetching the unintended repo.git.git URL."""
+ monkeypatch.setattr(
+ acquire, "_resolved_commit",
+ lambda remote, ref: pytest.fail("repository validation must precede resolution"))
+
+ with pytest.raises(ValueError, match="OWNER/REPO"):
+ acquire.github("acme/skills.git", ref="HEAD", subdirectory="skills/pdf",
+ destination=tmp_path / "clone")
+
+
+def test_github_acquisition_refuses_a_symlink_recorded_by_git(tmp_path, monkeypatch):
+ """Break caught: core.symlinks=false flattening a Git symlink before tree.build can see it."""
+ remote, _ = _remote(tmp_path)
+ (remote / "skills" / "pdf" / "link.md").symlink_to("SKILL.md")
+ _git(remote, "add", ".")
+ _git(remote, "commit", "-m", "Add a symlink")
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+
+ with pytest.raises(ValueError, match="symlink"):
+ acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf",
+ destination=tmp_path / "clone")
+
+
+def test_github_acquisition_refuses_a_submodule_recorded_by_git(tmp_path, monkeypatch):
+ """Break caught: treating a gitlink as package bytes or recursively acquiring another repo."""
+ child = tmp_path / "child"
+ subprocess.run(["git", "init", "-b", "main", str(child)], check=True,
+ capture_output=True)
+ _git(child, "config", "user.email", "fixture@example.test")
+ _git(child, "config", "user.name", "Fixture")
+ (child / "payload.txt").write_text("submodule bytes", encoding="utf-8")
+ _git(child, "add", ".")
+ _git(child, "commit", "-m", "Create child")
+ remote, _ = _remote(tmp_path)
+ subprocess.run(
+ ["git", "-c", "protocol.file.allow=always", "-C", str(remote), "submodule", "add",
+ child.as_uri(), "skills/pdf/vendor"], check=True, capture_output=True)
+ _git(remote, "commit", "-am", "Add a submodule")
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+
+ with pytest.raises(ValueError, match="regular file"):
+ acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf",
+ destination=tmp_path / "clone")
+
+
+def test_github_acquisition_refuses_an_oversized_tree_before_checkout(tmp_path, monkeypatch):
+ """Break caught: downloading selected blobs before enforcing the admission byte budget."""
+ remote, _ = _remote(tmp_path, asset=b"0123456789")
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+ monkeypatch.setattr(acquire, "MAX_TREE_BYTES", 8)
+
+ with pytest.raises(ValueError, match="at most 8 bytes"):
+ acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf",
+ destination=tmp_path / "clone")
+
+ assert not (tmp_path / "clone" / "package" / "asset.bin").exists()
+
+
+def test_github_acquisition_refuses_too_many_files_before_checkout(tmp_path, monkeypatch):
+ """Break caught: enforcing only byte size lets an empty-file tree exhaust the filesystem."""
+ remote, _ = _remote(tmp_path)
+ (remote / "skills" / "pdf" / "extra.txt").touch()
+ _git(remote, "add", ".")
+ _git(remote, "commit", "-m", "Add another file")
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+ monkeypatch.setattr(acquire, "MAX_FILES", 2)
+
+ with pytest.raises(ValueError, match="at most 2 files"):
+ acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf",
+ destination=tmp_path / "clone")
+
+ assert not (tmp_path / "clone" / "package").exists()
+
+
+def test_github_acquisition_refuses_when_the_ref_moves_before_clone(tmp_path, monkeypatch):
+ """Break caught: recording the ls-remote commit while admitting a later checkout."""
+ remote, old_commit = _remote(tmp_path)
+ (remote / "skills" / "pdf" / "asset.bin").write_bytes(b"new bytes")
+ _git(remote, "add", ".")
+ _git(remote, "commit", "-m", "Move HEAD")
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+ monkeypatch.setattr(acquire, "_resolved_commit", lambda remote, ref: old_commit)
+
+ with pytest.raises(ValueError, match="ref moved"):
+ acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf",
+ destination=tmp_path / "clone")
+
+
+def test_github_acquisition_does_not_run_checkout_filters(tmp_path, monkeypatch):
+ """Break caught: checkout executing a configured smudge command selected by .gitattributes."""
+ remote, _ = _remote(tmp_path)
+ (remote / "skills" / "pdf" / ".gitattributes").write_text(
+ "*.bin filter=fixture\n", encoding="utf-8")
+ _git(remote, "add", ".")
+ _git(remote, "commit", "-m", "Select a checkout filter")
+ marker = tmp_path / "filter-ran"
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+ monkeypatch.setenv("GIT_CONFIG_COUNT", "1")
+ monkeypatch.setenv("GIT_CONFIG_KEY_0", "filter.fixture.smudge")
+ monkeypatch.setenv("GIT_CONFIG_VALUE_0", f"touch {marker}; cat")
+
+ package, _ = acquire.github(
+ "acme/skills", ref="HEAD", subdirectory="skills/pdf", destination=tmp_path / "clone")
+
+ assert (package / "asset.bin").read_bytes() == b"first bytes"
+ assert not marker.exists()
diff --git a/tests/test_admission.py b/tests/test_admission.py
new file mode 100644
index 0000000..12e9084
--- /dev/null
+++ b/tests/test_admission.py
@@ -0,0 +1,619 @@
+"""`ingot add file:` — one complete local ingest path.
+
+The invariant under test is the product's whole claim: a package can be submitted, reviewed,
+quarantined and made visible to a reviewer without a single byte of the served library changing.
+
+Real filesystem throughout. The exclusivity test uses real processes, because `os.link` is a
+cross-process guarantee and a thread-only test would not exercise it."""
+import hashlib
+import json
+import multiprocessing
+import os
+import subprocess
+import sys
+from pathlib import Path
+
+import pytest
+
+from ingot import admission, cli, records
+from ingot.mcp_server import registry
+from ingot.mcp_server.registry import skill_revision
+from ingot.optimize import ingress, tree
+from ingot.optimize import promote as P
+
+
+def _store(tmp_path, monkeypatch):
+ """The served library plus every store admission writes to, all under tmp_path."""
+ root = tmp_path / "skills"
+ root.mkdir()
+ monkeypatch.setenv("INGOT_LIBRARY", str(root))
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs"))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root))
+ return root
+
+
+def _package(root, name="pdf", description="Merge and split PDF files.",
+ body="Use this to combine PDFs.", files=None):
+ directory = root / name
+ directory.mkdir(parents=True, exist_ok=True)
+ (directory / "SKILL.md").write_text(
+ f"---\nname: {name}\ndescription: {description}\n---\n\n{body}\n", encoding="utf-8")
+ for relative, content in (files or {}).items():
+ target = directory / relative
+ target.parent.mkdir(parents=True, exist_ok=True)
+ target.write_text(content, encoding="utf-8")
+ return directory
+
+
+def _tree_hash(root):
+ """Every path and its bytes under a root. The library must be identical after admission, and
+ 'identical' means content, not just the directory listing."""
+ entries = {}
+ for path in sorted(root.rglob("*")):
+ entries[str(path.relative_to(root))] = path.read_bytes() if path.is_file() else None
+ return entries
+
+
+def _github_remote(root, *, body="Use this skill."):
+ """A real Git remote; only URL construction is replaced in GitHub command tests."""
+ subprocess.run(["git", "init", "-b", "main", str(root)], check=True,
+ capture_output=True)
+ subprocess.run(["git", "-C", str(root), "config", "user.email", "fixture@example.test"],
+ check=True)
+ subprocess.run(["git", "-C", str(root), "config", "user.name", "Fixture"], check=True)
+ _package(root / "packages", "pdf", body=body, files={"assets/data.bin": "bytes"})
+ subprocess.run(["git", "-C", str(root), "add", "."], check=True)
+ subprocess.run(["git", "-C", str(root), "commit", "-m", "Create fixture"], check=True,
+ capture_output=True)
+ return subprocess.run(["git", "-C", str(root), "rev-parse", "HEAD"], check=True,
+ capture_output=True, text=True).stdout.strip()
+
+
+# --- locators -------------------------------------------------------------------------------
+
+def test_a_file_locator_resolves_to_its_directory(tmp_path):
+ package = _package(tmp_path / "src")
+
+ kind, resolved = admission.parse_locator(f"file:{package}")
+
+ assert kind == "file"
+ assert resolved == package
+
+
+def test_a_bare_path_is_accepted_as_a_file_locator(tmp_path):
+ package = _package(tmp_path / "src")
+
+ assert admission.parse_locator(str(package)) == ("file", package)
+
+
+def test_a_github_locator_keeps_the_repository_name_unresolved():
+ assert admission.parse_locator("github:acme/skills") == ("github", "acme/skills")
+
+
+def test_an_unsupported_scheme_is_refused():
+ with pytest.raises(ValueError, match="tessl"):
+ admission.parse_locator("tessl:acme/skills")
+
+
+# --- the ingest path ------------------------------------------------------------------------
+
+def test_admission_leaves_the_served_library_byte_identical(tmp_path, monkeypatch):
+ """The claim, tested directly."""
+ library = _store(tmp_path, monkeypatch)
+ _package(library, "docx", description="Edit Word documents.")
+ package = _package(tmp_path / "src", "pdf")
+ before = _tree_hash(library)
+
+ admission.add_package(package, actor="operator")
+
+ assert _tree_hash(library) == before
+
+
+def test_admission_writes_a_pending_proposal(tmp_path, monkeypatch):
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf")
+
+ result = admission.add_package(package, actor="operator")
+
+ pending = P.load_pending("pdf")
+ assert pending is not None
+ assert pending["kind"] == "creation"
+ assert result["proposal_id"] == pending["creation"]["proposal_id"]
+
+
+def test_the_pending_revision_matches_a_canonical_package_on_disk(tmp_path, monkeypatch):
+ """For a package that is already canonical the two revisions agree, bundled files included."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", files={"references/flags.md": "# Flags\n"})
+
+ admission.add_package(package, actor="operator")
+
+ pending = P.load_pending("pdf")
+ assert pending["evidence"]["challenger"]["revision"] == skill_revision(package)
+
+
+def test_a_non_canonical_package_binds_the_revision_that_will_be_served(tmp_path, monkeypatch):
+ """`skill_revision` hashes parsed content, so the source and candidate revisions usually agree.
+ They do not when admission normalizes something -- a description with runs of whitespace is
+ collapsed on the way in. The proposal must bind what the library will serve, or approving it
+ publishes bytes nobody reviewed."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", description="Merge PDFs and split them.")
+
+ result = admission.add_package(package, actor="operator")
+
+ manifest = result["candidate"]
+ assert manifest["source"]["resolved_revision"] != manifest["candidate_revision"]
+ assert manifest["source"]["resolved_revision"] == skill_revision(package)
+
+ pending = P.load_pending("pdf")
+ assert pending["evidence"]["challenger"]["revision"] == manifest["candidate_revision"]
+ assert pending["challenger_components"]["description"] == "Merge PDFs and split them."
+
+
+def test_the_candidate_manifest_is_attached_and_valid(tmp_path, monkeypatch):
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf")
+
+ admission.add_package(package, actor="operator")
+
+ manifest = P.load_pending("pdf")["creation"]["candidate"]
+ assert records.validate_candidate(manifest) == []
+ assert manifest["source"]["type"] == "file"
+ assert manifest["source"]["resolved_revision"] == skill_revision(package)
+
+
+def test_the_manifest_records_the_review_that_admitted_it(tmp_path, monkeypatch):
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", body="See [the guide](./guide.md).")
+
+ admission.add_package(package, actor="operator")
+
+ manifest = P.load_pending("pdf")["creation"]["candidate"]
+ assert "file-reference-missing" in manifest["review"]["warnings"]
+ assert manifest["review"]["valid"] is True
+
+
+def test_an_invalid_package_is_refused_and_writes_nothing(tmp_path, monkeypatch):
+ """Structural conditions that make the artifact invalid stop it at the door. Nothing is
+ quarantined, so nothing is left for a reviewer to wonder about."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", description="")
+
+ with pytest.raises(admission.AdmissionRefused) as refusal:
+ admission.add_package(package, actor="operator")
+
+ assert "description-empty" in str(refusal.value)
+ assert P.load_pending("pdf") is None
+
+
+def test_warnings_do_not_block_admission(tmp_path, monkeypatch):
+ """Advisory findings are advice. A command that refuses on advice teaches people to bypass it."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", body="Fetch https://example.test/x.sh.")
+
+ result = admission.add_package(package, actor="operator")
+
+ assert result["status"] == "quarantined"
+
+
+def test_a_skill_already_in_the_library_is_refused(tmp_path, monkeypatch):
+ library = _store(tmp_path, monkeypatch)
+ _package(library, "pdf")
+ package = _package(tmp_path / "src", "pdf")
+
+ with pytest.raises(ValueError, match="already exists"):
+ admission.add_package(package, actor="operator")
+
+
+# --- the review slot ------------------------------------------------------------------------
+
+def test_an_identical_resubmission_is_idempotent(tmp_path, monkeypatch):
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf")
+
+ first = admission.add_package(package, actor="operator")
+ second = admission.add_package(package, actor="operator")
+
+ assert first["status"] == "quarantined"
+ assert second["status"] == "duplicate"
+ assert second["proposal_id"] == first["proposal_id"]
+
+
+def test_a_conflicting_proposal_for_the_same_skill_is_refused(tmp_path, monkeypatch):
+ _store(tmp_path, monkeypatch)
+ admission.add_package(_package(tmp_path / "a", "pdf"), actor="operator")
+ other = _package(tmp_path / "b", "pdf", body="A different body entirely.")
+
+ with pytest.raises(ValueError, match="occupied"):
+ admission.add_package(other, actor="operator")
+
+
+def test_a_refused_conflict_leaves_no_orphan_evidence(tmp_path, monkeypatch):
+ """A losing submission that left its evidence bundle behind would accumulate directories
+ describing proposals that do not exist."""
+ _store(tmp_path, monkeypatch)
+ admission.add_package(_package(tmp_path / "a", "pdf"), actor="operator")
+ winner = P.load_pending("pdf")["creation"]["proposal_id"]
+ other = _package(tmp_path / "b", "pdf", body="A different body entirely.")
+
+ with pytest.raises(ValueError):
+ admission.add_package(other, actor="operator")
+
+ bundles = sorted(p.name for p in (tmp_path / "runs" / "evidence" / "pdf").iterdir())
+ assert bundles == [f"creation-{winner}"]
+
+
+def test_the_evidence_bundle_is_written(tmp_path, monkeypatch):
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf")
+
+ admission.add_package(package, actor="operator")
+
+ proposal_id = P.load_pending("pdf")["creation"]["proposal_id"]
+ bundle = tmp_path / "runs" / "evidence" / "pdf" / f"creation-{proposal_id}" / "evidence.json"
+ assert json.loads(bundle.read_text())["skill"] == "pdf"
+
+
+# --- cross-process exclusivity ----------------------------------------------------------------
+
+def _submit_in_child(package_dir, library, runs, queue):
+ """Module scope and explicit configuration on purpose: under macOS `spawn` a nested function is
+ unpicklable and a parent's monkeypatch never reaches the child, so the child is told where its
+ state lives through the environment -- the same mechanism a real deployment uses. Setting it
+ before the import is what makes it stick, and getting it wrong would point both children at the
+ developer's own state directory while the test still passed."""
+ import pathlib
+
+ os.environ["INGOT_LIBRARY"] = str(library)
+ os.environ["INGOT_RUNS"] = str(runs)
+ os.environ["SKILL_ROUTER_PATHS"] = str(library)
+
+ from ingot import admission as child_admission
+ try:
+ queue.put(("ok", child_admission.add_package(pathlib.Path(package_dir),
+ actor="operator")["status"]))
+ except Exception as error: # noqa: BLE001 - the refusal is the result under test
+ queue.put(("refused", str(error)))
+
+
+def test_two_processes_cannot_both_claim_the_review_slot(tmp_path, monkeypatch):
+ """`_publish` creates the pending record with `os.link`, which is atomic across processes.
+ A threading lock alone would not survive the publisher and the console running separately."""
+ _store(tmp_path, monkeypatch)
+ first = _package(tmp_path / "a", "pdf", body="First body.")
+ second = _package(tmp_path / "b", "pdf", body="Second body, different bytes.")
+
+ context = multiprocessing.get_context("spawn")
+ queue = context.Queue()
+ stores = (tmp_path / "skills", tmp_path / "runs")
+ workers = [context.Process(target=_submit_in_child,
+ args=(str(package), *[str(s) for s in stores], queue))
+ for package in (first, second)]
+ for worker in workers:
+ worker.start()
+ for worker in workers:
+ worker.join(timeout=120)
+
+ outcomes = [queue.get(timeout=10) for _ in workers]
+ assert sorted(kind for kind, _ in outcomes) == ["ok", "refused"]
+ assert P.load_pending("pdf") is not None
+
+
+# --- the command ----------------------------------------------------------------------------
+
+def test_command_quarantines_and_names_the_next_action(tmp_path, monkeypatch, capsys):
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf")
+
+ code = cli.main(["add", f"file:{package}"])
+
+ output = capsys.readouterr().out
+ assert code == 0
+ assert P.load_pending("pdf") is not None
+ assert "pdf" in output
+ assert "approve" in output.lower() or "review" in output.lower()
+
+
+def test_command_reports_a_refusal_without_a_traceback(tmp_path, monkeypatch, capsys):
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", description="")
+
+ code = cli.main(["add", f"file:{package}"])
+
+ assert code != 0
+ assert "description-empty" in capsys.readouterr().err
+
+
+def test_command_json_reports_the_proposal(tmp_path, monkeypatch, capsys):
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf")
+
+ code = cli.main(["add", f"file:{package}", "--json"])
+
+ payload = json.loads(capsys.readouterr().out)
+ assert code == 0
+ assert payload["status"] == "quarantined"
+ assert records.validate_candidate(payload["candidate"]) == []
+
+
+def test_file_command_refuses_a_github_skill_subdirectory(tmp_path, monkeypatch, capsys):
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf")
+
+ code = cli.main(["add", f"file:{package}", "--skill", "skills/pdf"])
+
+ assert code == 1
+ assert "--skill is only valid" in capsys.readouterr().err
+
+
+def test_github_command_requires_a_skill_subdirectory(capsys):
+ code = cli.main(["add", "github:acme/skills"])
+
+ assert code == 1
+ assert "--skill is required" in capsys.readouterr().err
+
+
+def test_github_command_records_exact_provenance_and_leaves_active_targets_unchanged(
+ tmp_path, monkeypatch, capsys):
+ """Break caught: Git metadata lost at the shared admission seam, or acquisition delivering."""
+ from ingot import acquire
+
+ library = _store(tmp_path, monkeypatch)
+ native = tmp_path / "native"
+ _package(library, "docx", description="Edit Word documents.")
+ _package(native, "existing", description="Existing native skill.")
+ monkeypatch.setenv("INGOT_DELIVERY_TARGETS", f"native=filesystem:{native}")
+ before = (_tree_hash(library), _tree_hash(native))
+ remote = tmp_path / "remote"
+ commit = _github_remote(remote)
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+
+ code = cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"])
+
+ payload = json.loads(capsys.readouterr().out)
+ source = payload["candidate"]["source"]
+ assert code == 0
+ assert source["type"] == "github"
+ assert source["repository"] == "acme/skills"
+ assert source["ref"] == "HEAD"
+ assert source["commit"] == commit
+ assert source["subdirectory"] == "packages/pdf"
+ assert source["content_digest"] == P.load_pending("pdf")["tree"]["digest"]
+ assert P.load_pending("pdf")["creation"]["source"] == "github:acme/skills"
+ assert records.validate_candidate(payload["candidate"]) == []
+ assert (_tree_hash(library), _tree_hash(native)) == before
+
+
+def test_moved_github_head_with_identical_bytes_is_idempotent(tmp_path, monkeypatch, capsys):
+ """Break caught: commit or temporary clone paths leaking into content identity."""
+ from ingot import acquire
+
+ _store(tmp_path, monkeypatch)
+ remote = tmp_path / "remote"
+ _github_remote(remote)
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+
+ assert cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"]) == 0
+ first = json.loads(capsys.readouterr().out)
+ subprocess.run(["git", "-C", str(remote), "commit", "--allow-empty", "-m", "Move HEAD"],
+ check=True, capture_output=True)
+ assert cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"]) == 0
+ second = json.loads(capsys.readouterr().out)
+
+ assert second["status"] == "duplicate"
+ assert second["proposal_id"] == first["proposal_id"]
+
+
+def test_changed_github_bytes_produce_a_different_candidate_revision(tmp_path, monkeypatch, capsys):
+ """Break caught: commit provenance changing while the candidate still names old package bytes."""
+ from ingot import acquire
+
+ _store(tmp_path, monkeypatch)
+ remote = tmp_path / "remote"
+ _github_remote(remote)
+ monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri())
+ assert cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"]) == 0
+ first = json.loads(capsys.readouterr().out)["candidate"]["candidate_revision"]
+ P.pending_path("pdf").unlink()
+
+ skill_md = remote / "packages" / "pdf" / "SKILL.md"
+ skill_md.write_text(skill_md.read_text(encoding="utf-8") + "Changed upstream.\n",
+ encoding="utf-8")
+ subprocess.run(["git", "-C", str(remote), "add", "."], check=True)
+ subprocess.run(["git", "-C", str(remote), "commit", "-m", "Change bytes"], check=True,
+ capture_output=True)
+ assert cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"]) == 0
+ second = json.loads(capsys.readouterr().out)["candidate"]["candidate_revision"]
+
+ assert second != first
+
+
+def test_the_installed_script_works_outside_the_repository(tmp_path):
+ """Run from elsewhere, with the repo root off `sys.path`.
+
+ Every other test in this file runs with the repository as the working directory, which puts
+ `optimize` on the import path whether or not it was ever installed. That masked a real break:
+ the console script resolved `ingot` and `mcp_server` from site-packages and then failed on
+ `import optimize` for anyone who ran it from their own directory. Only a test that leaves the
+ repository can see it."""
+ script = Path(sys.executable).parent / "ingot"
+ if not script.exists():
+ pytest.skip("console script not installed; run `pip install -e .`")
+
+ library, package = tmp_path / "lib", tmp_path / "src" / "pdf"
+ library.mkdir()
+ package.mkdir(parents=True)
+ (package / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\n\nBody.\n", encoding="utf-8")
+
+ environment = {**os.environ, "SKILL_ROUTER_PATHS": str(library),
+ "INGOT_ACTOR": "integration-test"}
+ environment.pop("PYTHONPATH", None)
+ result = subprocess.run([str(script), "review", str(package), "--json"],
+ cwd=tmp_path, capture_output=True, text=True, env=environment)
+
+ assert result.returncode == 0, result.stderr
+ assert json.loads(result.stdout)["skill"] == "pdf"
+
+
+def test_the_installed_script_can_reach_the_quarantine_from_outside_the_repository(tmp_path):
+ """The same escape, for the verb that needs `optimize`. `ingot add` is the whole point of this
+ PR, so a version of it that only runs inside a checkout is not shipped."""
+ script = Path(sys.executable).parent / "ingot"
+ if not script.exists():
+ pytest.skip("console script not installed; run `pip install -e .`")
+
+ package = tmp_path / "src" / "pdf"
+ package.mkdir(parents=True)
+ (package / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\n\nBody.\n", encoding="utf-8")
+
+ environment = {**os.environ, "INGOT_ACTOR": "integration-test"}
+ environment.pop("PYTHONPATH", None)
+ result = subprocess.run([str(script), "add", "--help"],
+ cwd=tmp_path, capture_output=True, text=True, env=environment)
+
+ assert result.returncode == 0, result.stderr
+ # Importing the admission path is what actually proves `optimize` resolves from site-packages.
+ probe = subprocess.run([sys.executable, "-c", "import ingot.admission; from ingot.optimize import "
+ "ingress; print(ingress.submit_package_ingest.__name__)"],
+ cwd=tmp_path, capture_output=True, text=True, env=environment)
+ assert probe.returncode == 0, probe.stderr
+ assert probe.stdout.strip() == "submit_package_ingest"
+
+
+def test_listing_the_cli_stays_light_after_admission_exists():
+ """`ingot list` and `ingot review` must still import nothing heavy. Admission pulls in
+ `optimize`, so it has to stay behind a function-local import rather than riding along at module
+ import time."""
+ heavy = ["fastapi", "onnxruntime", "langgraph", "langfuse", "litellm", "fastembed", "ingot.optimize"]
+ program = f"import sys, ingot.cli; print([m for m in {heavy!r} if m in sys.modules])"
+
+ result = subprocess.run([sys.executable, "-c", program], capture_output=True, text=True)
+
+ assert result.returncode == 0, result.stderr
+ assert result.stdout.strip() == "[]"
+
+
+# --- artifact fidelity ----------------------------------------------------------------------
+
+PNG = b"\x89PNG\r\n\x1a\n" + bytes(range(256)) + b"\xff\xfe\xfd"
+
+
+def test_a_binary_asset_survives_admission_byte_for_byte(tmp_path, monkeypatch):
+ """The defect this exists to stop: admission used to reduce a package to decoded text, so an
+ image was reviewed as part of the candidate and then was not in it."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf")
+ (package / "assets").mkdir()
+ (package / "assets" / "logo.png").write_bytes(PNG)
+
+ admission.add_package(package, actor="operator")
+
+ entry = _entry(P.load_pending("pdf"), "assets/logo.png")
+ assert entry["size"] == len(PNG)
+ assert entry["sha256"] == hashlib.sha256(PNG).hexdigest()
+ staged = tree.staged_dir(P.load_pending("pdf")["tree"]["digest"])
+ assert (staged / "assets" / "logo.png").read_bytes() == PNG
+
+
+def test_the_approved_revision_changes_when_an_asset_changes(tmp_path, monkeypatch):
+ """The whole point of binding a revision. While assets were dropped, two packages differing
+ only in an image hashed identically, so approving one approved the other."""
+ _store(tmp_path, monkeypatch)
+ first = _package(tmp_path / "one", "pdf")
+ (first / "logo.png").write_bytes(PNG)
+ second = _package(tmp_path / "two", "pdf")
+ (second / "logo.png").write_bytes(PNG[:-1] + b"\x00")
+
+ one = admission.add_package(first, actor="operator")["candidate"]["candidate_revision"]
+ P.pending_path("pdf").unlink()
+ two = admission.add_package(second, actor="operator")["candidate"]["candidate_revision"]
+
+ assert one != two
+
+
+def test_the_executable_bit_is_recorded_and_nothing_else_is(tmp_path, monkeypatch):
+ """Git stores two modes. Recording the raw one would describe a file the vault cannot serve."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", files={"run.sh": "#!/bin/sh\necho hi\n",
+ "notes.md": "# Notes\n"})
+ (package / "run.sh").chmod(0o777)
+
+ admission.add_package(package, actor="operator")
+
+ pending = P.load_pending("pdf")
+ assert _entry(pending, "run.sh")["mode"] == 0o755
+ assert _entry(pending, "notes.md")["mode"] == 0o644
+
+
+def test_a_symlink_is_refused_and_the_source_is_left_alone(tmp_path, monkeypatch):
+ """Admission stages exact bytes. Preserving a link puts a path into the vault that leads out of
+ the library; flattening it changes the artifact's shape. Neither is admission's call."""
+ root = _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"})
+ (package / "link.md").symlink_to(package / "notes.md")
+ before = sorted(item.name for item in package.iterdir())
+
+ with pytest.raises(admission.AdmissionRefused, match="symlink-unsupported"):
+ admission.add_package(package, actor="operator")
+
+ assert sorted(item.name for item in package.iterdir()) == before
+ assert P.load_pending("pdf") is None
+ assert not (root / "pdf").exists()
+
+
+def test_a_binary_asset_does_not_arrive_as_a_decoded_component(tmp_path, monkeypatch):
+ """Two descriptions of one file is one description too many: the receipt would carry a decoded
+ copy beside the exact bytes, and the publisher would have to pick."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"})
+
+ admission.add_package(package, actor="operator")
+
+ components = P.load_pending("pdf")["challenger_components"]
+ assert sorted(components) == ["body", "description", "frontmatter"]
+
+
+def test_an_ingested_package_is_approvable_the_moment_it_is_quarantined(tmp_path, monkeypatch):
+ """The freshness check has to recompute the revision the way admission computed it. Deriving it
+ from the decoded components instead makes every package with a second file permanently stale:
+ quarantined, reviewable, and impossible to approve."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"})
+ admission.add_package(package, actor="operator")
+
+ assert P.stale_evidence_reason("pdf", P.load_pending("pdf")) is None
+
+
+def test_an_ingested_package_whose_staged_bytes_moved_is_stale(tmp_path, monkeypatch):
+ """The other direction. A freshness check that cannot go stale is not checking anything."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"})
+ admission.add_package(package, actor="operator")
+ pending = P.load_pending("pdf")
+ staged = tree.staged_dir(pending["tree"]["digest"])
+ (staged / "notes.md").write_text("# Substituted\n", encoding="utf-8")
+
+ assert P.stale_evidence_reason("pdf", pending) is not None
+
+
+def test_an_ingested_package_survives_the_whole_cli_lane(tmp_path, monkeypatch, capsys):
+ """add, then approve, on the real verbs. The unit tests queue publications directly, so this is
+ the only place the approval path itself is exercised on an ingested package."""
+ _store(tmp_path, monkeypatch)
+ package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"})
+ assert cli.main(["add", f"file:{package}"]) == 0
+ capsys.readouterr()
+
+ assert cli.main(["approve", "pdf", "--actor", "operator"]) == 0
+
+ from ingot.optimize import publication as Q
+ receipt = Q.publication_for_skill("pdf")
+ assert receipt["state"] == "approved_publishing"
+ assert receipt["candidate_revision"] == P.load_pending("pdf")["evidence"]["challenger"]["revision"]
+
+
+def _entry(pending, path):
+ return next(item for item in pending["tree"]["files"] if item["path"] == path)
diff --git a/tests/test_agy_judge.py b/tests/test_agy_judge.py
new file mode 100644
index 0000000..82969ae
--- /dev/null
+++ b/tests/test_agy_judge.py
@@ -0,0 +1,408 @@
+"""Contract tests for the subscription-backed Agy judge process boundary."""
+from __future__ import annotations
+
+import json
+import subprocess
+from pathlib import Path
+
+import pytest
+
+from ingot.optimize import agy_judge as A
+
+
+FIXTURE = Path(__file__).parent / "fixtures" / "agy" / "judge-stream.jsonl"
+CHECKLIST = [{"id": "probe", "criterion": "Return the requested token.", "weight": 1}]
+TWO_ITEM_CHECKLIST = [
+ *CHECKLIST,
+ {"id": "second", "criterion": "Return the second requested token.", "weight": 1},
+]
+VALID_GRADE = {
+ "items": {"probe": {"verdict": "pass", "note": ""}},
+ "feedback": "ok",
+}
+VALID_USAGE = {
+ "input_tokens": 1,
+ "output_tokens": 1,
+ "thinking_tokens": 0,
+ "cache_read_tokens": 0,
+ "total_tokens": 2,
+}
+
+
+def _terminal(**changes: object) -> dict:
+ result = {
+ "status": "SUCCESS",
+ "structured_output": VALID_GRADE,
+ "usage": VALID_USAGE,
+ }
+ result.update(changes)
+ return {"event": "result", "result": result}
+
+
+def _jsonl(*events: dict) -> str:
+ return "\n".join(json.dumps(event) for event in events)
+
+
+def _runtime(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> tuple[Path, Path]:
+ agy_bin = tmp_path / "agy"
+ agy_bin.write_text("fixture executable")
+ agy_bin.chmod(0o700)
+ workspace = tmp_path / "workspace"
+ workspace.mkdir()
+ monkeypatch.setenv("AGY_BIN", str(agy_bin))
+ monkeypatch.setenv("AGY_JUDGE_WORKSPACE", str(workspace))
+ return agy_bin, workspace
+
+
+def test_captured_stream_supplies_the_fixed_judge_contract():
+ grade, usage = A.parse_stream(FIXTURE.read_text(), CHECKLIST)
+ assert A.AGY_MODEL == "gemini-3.6-flash-medium"
+ assert A.AGY_IDENTITY == "agy/gemini-3.6-flash-medium"
+ assert grade["items"]["probe"]["verdict"] == "pass"
+ assert usage["total_tokens"] == 24067
+
+
+def test_schema_requires_exact_checklist_items_and_feedback():
+ assert A.judge_schema(CHECKLIST) == {
+ "type": "object",
+ "additionalProperties": False,
+ "required": ["items", "feedback"],
+ "properties": {
+ "items": {
+ "type": "object",
+ "additionalProperties": False,
+ "required": ["probe"],
+ "properties": {
+ "probe": {
+ "type": "object",
+ "additionalProperties": False,
+ "required": ["verdict", "note"],
+ "properties": {
+ "verdict": {"enum": ["pass", "partial", "fail"]},
+ "note": {"type": "string"},
+ },
+ }
+ },
+ },
+ "feedback": {"type": "string"},
+ },
+ }
+
+
+def test_child_environment_scrubs_provider_routing_and_disables_updates():
+ parent = {
+ "PATH": "/bin",
+ "BASE_URL": "https://provider.invalid",
+ "API_KEY": "secret-a",
+ "MODEL_API_KEY": "secret-b",
+ "OPENROUTER_API_KEY": "secret-c",
+ "OPENAI_API_KEY": "secret-d",
+ "ANTHROPIC_API_KEY": "secret-e",
+ "GEMINI_API_KEY": "secret-f",
+ "GOOGLE_API_KEY": "secret-g",
+ "VERTEX_API_KEY": "secret-h",
+ "AGY_CLI_DISABLE_AUTO_UPDATE": "false",
+ }
+ child = A.agy_process_env(parent)
+ assert child == {"PATH": "/bin", "AGY_CLI_DISABLE_AUTO_UPDATE": "true"}
+ assert "OPENROUTER_API_KEY" not in child
+
+
+@pytest.mark.parametrize(
+ "case,stdout",
+ [
+ ("zero terminal results", _jsonl({"event": "init"})),
+ ("two terminal results", _jsonl(_terminal(), _terminal())),
+ ("non-success status", _jsonl(_terminal(status="FAILED"))),
+ (
+ "missing structured output",
+ _jsonl(_terminal(structured_output=None, response=json.dumps(VALID_GRADE))),
+ ),
+ (
+ "missing checklist item",
+ _jsonl(_terminal(structured_output={"items": {}, "feedback": "bad"})),
+ ),
+ (
+ "extra checklist item",
+ _jsonl(_terminal(structured_output={
+ "items": {
+ **VALID_GRADE["items"],
+ "other": {"verdict": "fail", "note": "not requested"},
+ },
+ "feedback": "bad",
+ })),
+ ),
+ (
+ "unknown verdict",
+ _jsonl(_terminal(structured_output={
+ "items": {"probe": {"verdict": "excellent", "note": ""}},
+ "feedback": "bad",
+ })),
+ ),
+ (
+ "invalid note",
+ _jsonl(_terminal(structured_output={
+ "items": {"probe": {"verdict": "pass", "note": ["not", "text"]}},
+ "feedback": "bad",
+ })),
+ ),
+ ("non-object event", "[]"),
+ ("non-object result", _jsonl({"event": "result", "result": []})),
+ (
+ "missing usage",
+ _jsonl({
+ "event": "result",
+ "result": {"status": "SUCCESS", "structured_output": VALID_GRADE},
+ }),
+ ),
+ ("non-object usage", _jsonl(_terminal(usage=[]))),
+ (
+ "invalid feedback",
+ _jsonl(_terminal(structured_output={
+ "items": VALID_GRADE["items"],
+ "feedback": ["not", "text"],
+ })),
+ ),
+ (
+ "extra grade key",
+ _jsonl(_terminal(structured_output={**VALID_GRADE, "score": 1.0})),
+ ),
+ (
+ "extra item key",
+ _jsonl(_terminal(structured_output={
+ "items": {"probe": {"verdict": "pass", "note": "", "score": 1.0}},
+ "feedback": "bad",
+ })),
+ ),
+ ("malformed jsonl", '{"event":"result"'),
+ ],
+)
+def test_invalid_streams_raise_without_manufacturing_a_grade(case: str, stdout: str):
+ with pytest.raises(A.AgyJudgeError, match="."):
+ A.parse_stream(stdout, CHECKLIST)
+
+
+def test_stream_parser_accepts_every_item_in_a_two_id_checklist():
+ grade = {
+ "items": {
+ "probe": {"verdict": "pass", "note": ""},
+ "second": {"verdict": "partial", "note": "one mismatch"},
+ },
+ "feedback": "Fix the second item.",
+ }
+ parsed, usage = A.parse_stream(
+ _jsonl(_terminal(structured_output=grade, usage=VALID_USAGE)),
+ TWO_ITEM_CHECKLIST,
+ )
+ assert parsed == grade
+ assert usage == VALID_USAGE
+
+
+def test_invoke_uses_only_the_explicit_sandboxed_process_boundary(monkeypatch, tmp_path):
+ agy_bin, workspace = _runtime(monkeypatch, tmp_path)
+ monkeypatch.setenv("OPENROUTER_API_KEY", "must-not-leak")
+ seen = {}
+
+ def fake_run(argv, **kwargs):
+ seen.update(argv=argv, **kwargs)
+ return subprocess.CompletedProcess(argv, 0, stdout=FIXTURE.read_text(), stderr="ignored")
+
+ monkeypatch.setattr(A.subprocess, "run", fake_run)
+ grade, usage = A.invoke("grade this", CHECKLIST, timeout=17.9)
+
+ assert grade["items"]["probe"]["verdict"] == "pass"
+ assert usage["total_tokens"] == 24067
+ assert seen["argv"] == [
+ str(agy_bin),
+ "--model", A.AGY_MODEL,
+ "--print", "grade this",
+ "--output-format", "stream-json",
+ "--json-schema", json.dumps(A.judge_schema(CHECKLIST)),
+ "--sandbox",
+ "--mode", "plan",
+ "--disable-slash-commands",
+ "--print-timeout", "17s",
+ ]
+ assert seen["cwd"] == workspace
+ assert seen["text"] is True
+ assert seen["capture_output"] is True
+ assert seen["timeout"] == pytest.approx(27.9)
+ assert seen["check"] is False
+ assert "OPENROUTER_API_KEY" not in seen["env"]
+ assert seen["env"]["AGY_CLI_DISABLE_AUTO_UPDATE"] == "true"
+
+
+@pytest.mark.parametrize("failure", ["timeout", "nonzero"])
+def test_process_failures_raise_without_using_stdout_or_stderr_as_a_grade(
+ failure, monkeypatch, tmp_path
+):
+ _runtime(monkeypatch, tmp_path)
+
+ def fake_run(argv, **kwargs):
+ if failure == "timeout":
+ raise subprocess.TimeoutExpired(argv, kwargs["timeout"], output=FIXTURE.read_text())
+ return subprocess.CompletedProcess(
+ argv,
+ 9,
+ stdout=FIXTURE.read_text(),
+ stderr=json.dumps(VALID_GRADE),
+ )
+
+ monkeypatch.setattr(A.subprocess, "run", fake_run)
+ with pytest.raises(A.AgyJudgeError, match="."):
+ A.invoke("grade this", CHECKLIST)
+
+
+def test_resource_exhaustion_is_rate_limited_and_retried(monkeypatch, tmp_path):
+ _runtime(monkeypatch, tmp_path)
+ calls = []
+ sleeps = []
+ clock = iter([0.0, 0.0, 3.2])
+
+ def fake_run(argv, **kwargs):
+ calls.append(argv)
+ if len(calls) == 1:
+ return subprocess.CompletedProcess(
+ argv, 1,
+ stdout=_jsonl({
+ "event": "result",
+ "result": {
+ "status": "ERROR",
+ "error": "Eligibility check failed: RESOURCE_EXHAUSTED (code 429)",
+ },
+ }),
+ stderr="",
+ )
+ return subprocess.CompletedProcess(argv, 0, stdout=FIXTURE.read_text(), stderr="")
+
+ monkeypatch.setattr(A.subprocess, "run", fake_run)
+ monkeypatch.setattr(A.time, "monotonic", lambda: next(clock))
+ monkeypatch.setattr(A.time, "sleep", sleeps.append)
+ A._next_launch_at = 0.0
+
+ grade, _ = A.invoke("grade this", CHECKLIST)
+
+ assert grade["items"]["probe"]["verdict"] == "pass"
+ assert len(calls) == 2
+ assert sleeps == [pytest.approx(3.2)]
+
+
+def test_non_quota_process_failure_is_not_retried(monkeypatch, tmp_path):
+ _runtime(monkeypatch, tmp_path)
+ calls = []
+
+ def fake_run(argv, **kwargs):
+ calls.append(argv)
+ return subprocess.CompletedProcess(argv, 1, stdout="not a quota result", stderr="")
+
+ monkeypatch.setattr(A.subprocess, "run", fake_run)
+ monkeypatch.setattr(A.time, "monotonic", lambda: 0.0)
+ A._next_launch_at = 0.0
+
+ with pytest.raises(A.AgyJudgeError, match="status 1"):
+ A.invoke("grade this", CHECKLIST)
+ assert len(calls) == 1
+
+
+def test_preflight_records_version_and_requires_the_fixed_model(monkeypatch, tmp_path):
+ agy_bin, workspace = _runtime(monkeypatch, tmp_path)
+ calls = []
+
+ def fake_run(argv, **kwargs):
+ calls.append((argv, kwargs))
+ stdout = "agy 1.1.11\n" if argv[-1] == "--version" else f"{A.AGY_MODEL}\n"
+ return subprocess.CompletedProcess(argv, 0, stdout=stdout, stderr="")
+
+ monkeypatch.setattr(A.subprocess, "run", fake_run)
+ assert A.preflight() == {
+ "identity": A.AGY_IDENTITY,
+ "model": A.AGY_MODEL,
+ "version": "agy 1.1.11",
+ "billing_mode": "subscription",
+ }
+ assert [call[0] for call in calls] == [
+ [str(agy_bin), "--version"],
+ [str(agy_bin), "models"],
+ ]
+ assert all(call[1]["cwd"] == workspace for call in calls)
+ assert all("OPENROUTER_API_KEY" not in call[1]["env"] for call in calls)
+
+
+@pytest.mark.parametrize(
+ "invalid_runtime",
+ ["missing", "relative", "non_executable", "bad_workspace"],
+)
+def test_preflight_rejects_invalid_runtime_before_starting_a_process(
+ invalid_runtime, monkeypatch, tmp_path
+):
+ agy_bin = tmp_path / "agy"
+ agy_bin.write_text("fixture executable")
+ agy_bin.chmod(0o700)
+ workspace = tmp_path / "workspace"
+ workspace.mkdir()
+ monkeypatch.setenv("AGY_BIN", str(agy_bin))
+ monkeypatch.setenv("AGY_JUDGE_WORKSPACE", str(workspace))
+
+ if invalid_runtime == "missing":
+ monkeypatch.delenv("AGY_BIN")
+ elif invalid_runtime == "relative":
+ monkeypatch.setenv("AGY_BIN", "agy")
+ elif invalid_runtime == "non_executable":
+ agy_bin.chmod(0o600)
+ else:
+ bad_workspace = tmp_path / "workspace-file"
+ bad_workspace.write_text("not a directory")
+ monkeypatch.setenv("AGY_JUDGE_WORKSPACE", str(bad_workspace))
+
+ calls = []
+ monkeypatch.setattr(A.subprocess, "run", lambda *args, **kwargs: calls.append(args))
+ with pytest.raises(A.AgyJudgeError):
+ A.preflight()
+ assert calls == []
+
+
+@pytest.mark.parametrize(
+ "failed_command,failure,expected_calls",
+ [
+ ("version", "timeout", 1),
+ ("version", "empty", 1),
+ ("models", "nonzero", 2),
+ ],
+)
+def test_preflight_stops_at_a_failed_command(
+ failed_command, failure, expected_calls, monkeypatch, tmp_path
+):
+ _runtime(monkeypatch, tmp_path)
+ calls = []
+
+ def fake_run(argv, **kwargs):
+ calls.append(argv)
+ command = "version" if argv[-1] == "--version" else "models"
+ stdout = "agy 1.1.11\n" if command == "version" else f"{A.AGY_MODEL}\n"
+ if command != failed_command:
+ return subprocess.CompletedProcess(argv, 0, stdout=stdout, stderr="")
+ if failure == "timeout":
+ raise subprocess.TimeoutExpired(argv, kwargs["timeout"], output=stdout)
+ if failure == "nonzero":
+ return subprocess.CompletedProcess(argv, 7, stdout=stdout, stderr="grade-like text")
+ return subprocess.CompletedProcess(argv, 0, stdout="", stderr="")
+
+ monkeypatch.setattr(A.subprocess, "run", fake_run)
+ with pytest.raises(A.AgyJudgeError):
+ A.preflight()
+ assert len(calls) == expected_calls
+
+
+def test_preflight_requires_the_exact_model_token(monkeypatch, tmp_path):
+ _runtime(monkeypatch, tmp_path)
+ calls = []
+
+ def fake_run(argv, **kwargs):
+ calls.append(argv)
+ stdout = "agy 1.1.11\n" if argv[-1] == "--version" else f"{A.AGY_MODEL}-preview\n"
+ return subprocess.CompletedProcess(argv, 0, stdout=stdout, stderr="")
+
+ monkeypatch.setattr(A.subprocess, "run", fake_run)
+ with pytest.raises(A.AgyJudgeError):
+ A.preflight()
+ assert len(calls) == 2
diff --git a/tests/test_auth.py b/tests/test_auth.py
index 82af674..f701ef7 100644
--- a/tests/test_auth.py
+++ b/tests/test_auth.py
@@ -5,7 +5,7 @@
from fastapi.testclient import TestClient
-from optimize import promote as P
+from ingot.optimize import promote as P
from ui import auth
from ui.app import app
@@ -107,7 +107,6 @@ def _promotable(skill):
def test_promote_endpoint_threads_the_authenticated_actor(tmp_path, monkeypatch):
import ui.app as ui_app
- monkeypatch.setattr(P, "PENDING_DIR", tmp_path / "pending")
f = tmp_path / "auth.json"
f.write_text(json.dumps({"alice": auth.hash_password("pw")}))
monkeypatch.setattr(auth, "AUTH_FILE", f)
diff --git a/tests/test_cli.py b/tests/test_cli.py
new file mode 100644
index 0000000..462f74f
--- /dev/null
+++ b/tests/test_cli.py
@@ -0,0 +1,150 @@
+"""The `ingot` console entry point.
+
+The first product experience must not require the full stack, so these tests care about two things
+the rest of the suite cannot see: that the CLI runs as an installed console script, and that
+importing it does not drag in FastAPI, ONNX, LangGraph, Langfuse, or the optimizer. A CLI that only
+works from a repo checkout with every server dependency present is not the lightweight install this
+milestone exists to deliver."""
+import json
+import shutil
+import subprocess
+import sys
+import textwrap
+from pathlib import Path
+
+import pytest
+
+from ingot import cli
+
+
+def _skill(root, name, description, body="Do the thing."):
+ directory = root / name
+ directory.mkdir(parents=True)
+ (directory / "SKILL.md").write_text(
+ textwrap.dedent(f"""\
+ ---
+ name: {name}
+ description: {description}
+ ---
+
+ {body}
+ """),
+ encoding="utf-8")
+ return directory
+
+
+def test_list_reports_a_skill_in_an_explicit_root(tmp_path):
+ _skill(tmp_path, "pdf", "Merge and split PDF files.")
+
+ result = cli.list_library(tmp_path)
+
+ assert [s["name"] for s in result["skills"]] == ["pdf"]
+ assert result["skills"][0]["description"] == "Merge and split PDF files."
+ assert len(result["skills"][0]["revision"]) == 64
+
+
+def test_list_reports_the_roots_it_actually_read(tmp_path):
+ """`configured_roots` always prepends the local authoring root, even ahead of an explicit one,
+ so a caller who passes `--root` can still be shown skills from somewhere else. Naming every root
+ in the payload is what makes that visible instead of baffling."""
+ _skill(tmp_path, "pdf", "Merge and split PDF files.")
+
+ result = cli.list_library(tmp_path)
+
+ assert str(tmp_path.resolve()) in result["roots"]
+
+
+def test_list_of_an_empty_root_is_not_an_error(tmp_path):
+ result = cli.list_library(tmp_path)
+
+ assert result["skills"] == []
+ assert result["schema_version"] == cli.LIST_SCHEMA
+
+
+def test_list_skips_a_skill_with_no_description(tmp_path):
+ """The router keys on description; a skill without one is unroutable and the loader drops it.
+ The CLI must report the same library the server would serve, not a more generous one."""
+ _skill(tmp_path, "pdf", "Merge and split PDF files.")
+ empty = tmp_path / "blank"
+ empty.mkdir()
+ (empty / "SKILL.md").write_text("---\nname: blank\n---\n\nbody\n", encoding="utf-8")
+
+ result = cli.list_library(tmp_path)
+
+ assert [s["name"] for s in result["skills"]] == ["pdf"]
+
+
+def test_main_list_json_emits_the_versioned_payload(tmp_path, capsys):
+ _skill(tmp_path, "pdf", "Merge and split PDF files.")
+
+ code = cli.main(["list", "--root", str(tmp_path), "--json"])
+
+ assert code == 0
+ payload = json.loads(capsys.readouterr().out)
+ assert payload["schema_version"] == cli.LIST_SCHEMA
+ assert [s["name"] for s in payload["skills"]] == ["pdf"]
+
+
+def test_main_list_human_output_names_the_skill(tmp_path, capsys):
+ _skill(tmp_path, "pdf", "Merge and split PDF files.")
+
+ code = cli.main(["list", "--root", str(tmp_path)])
+
+ assert code == 0
+ assert "pdf" in capsys.readouterr().out
+
+
+def test_main_with_no_command_fails_rather_than_doing_something(capsys):
+ with pytest.raises(SystemExit) as exit_info:
+ cli.main([])
+
+ assert exit_info.value.code != 0
+
+
+def _console_script() -> str | None:
+ """The `ingot` script installed beside the interpreter running these tests.
+
+ Not `shutil.which`: a virtualenv that has not been activated is not on PATH, so searching PATH
+ silently skips the one test that proves the entry point exists. Fall back to PATH for a system
+ install."""
+ beside = Path(sys.executable).parent / "ingot"
+ return str(beside) if beside.exists() else shutil.which("ingot")
+
+
+@pytest.mark.skipif(_console_script() is None,
+ reason="console script not installed; run `pip install -e .`")
+def test_installed_console_script_runs(tmp_path):
+ """In situ: the real entry point, not an in-process call. `--help` is the one command that must
+ work before anything else is configured."""
+ result = subprocess.run([_console_script(), "--help"], capture_output=True, text=True)
+
+ assert result.returncode == 0
+ assert "list" in result.stdout
+
+
+def test_installed_console_script_lists_a_real_library(tmp_path):
+ """The acceptance criterion for this PR, run the way a user runs it: the installed script, a
+ real directory, no services and no Docker."""
+ script = _console_script()
+ if script is None:
+ pytest.skip("console script not installed; run `pip install -e .`")
+ _skill(tmp_path, "pdf", "Merge and split PDF files.")
+
+ result = subprocess.run([script, "list", "--root", str(tmp_path), "--json"],
+ capture_output=True, text=True)
+
+ assert result.returncode == 0, result.stderr
+ assert "pdf" in [skill["name"] for skill in json.loads(result.stdout)["skills"]]
+
+
+def test_importing_the_cli_stays_light():
+ """A subprocess, because the rest of the suite has already imported the heavy stack in-process.
+ This is the whole point of the milestone: `ingot list` must not need the server's dependencies."""
+ heavy = ["fastapi", "onnxruntime", "langgraph", "langfuse", "litellm", "fastembed", "ingot.optimize"]
+ program = ("import sys, ingot.cli; "
+ f"print([m for m in {heavy!r} if m in sys.modules])")
+
+ result = subprocess.run([sys.executable, "-c", program], capture_output=True, text=True)
+
+ assert result.returncode == 0, result.stderr
+ assert result.stdout.strip() == "[]"
diff --git a/tests/test_cluster.py b/tests/test_cluster.py
new file mode 100644
index 0000000..e25f8b6
--- /dev/null
+++ b/tests/test_cluster.py
@@ -0,0 +1,132 @@
+"""Unit tests for skill clustering (embeddings mocked; the numpy maths is exercised for real)."""
+import numpy as np
+import pytest
+
+from ingot.optimize import cluster as C
+
+
+def _blobs(seed=0):
+ """Three well-separated groups on the unit sphere — clustering that cannot recover these is
+ not going to recover anything subtler."""
+ rng = np.random.default_rng(seed)
+ centres = np.array([[1.0, 0, 0], [0, 1.0, 0], [0, 0, 1.0]])
+ pts = np.vstack([c + rng.normal(0, 0.05, (8, 3)) for c in centres])
+ return C._normalise(pts)
+
+
+def test_normalise_gives_unit_vectors():
+ v = C._normalise(np.array([[3.0, 4.0], [0.0, 2.0]]))
+ assert np.allclose(np.linalg.norm(v, axis=1), 1.0)
+
+
+def test_normalise_survives_a_zero_vector():
+ """A zero row would divide by zero and poison every downstream distance with nan."""
+ assert np.isfinite(C._normalise(np.zeros((1, 4)))).all()
+
+
+def test_kmeans_recovers_separated_groups():
+ labels, _ = C.kmeans(_blobs(), 3)
+ groups = [set(np.where(labels == j)[0]) for j in range(3)]
+ assert sorted(len(g) for g in groups) == [8, 8, 8]
+
+
+def test_kmeans_is_deterministic_for_one_library():
+ """The view polls. Buckets that reshuffle between identical runs are not something a reader
+ can build a mental model of."""
+ v = _blobs()
+ assert np.array_equal(C.kmeans(v, 3)[0], C.kmeans(v, 3)[0])
+
+
+def test_kmeans_never_returns_fewer_buckets_than_asked():
+ """An emptied centre must be re-seeded. Left alone it collapses the run to k-1 and the caller
+ silently gets a different k than the silhouette was computed for."""
+ v = C._normalise(np.array([[1.0, 0.0], [1.0, 0.001], [1.0, 0.002], [1.0, 0.003]]))
+ labels, centres = C.kmeans(v, 3)
+ assert len(centres) == 3 and np.isfinite(centres).all()
+
+
+def test_silhouette_prefers_the_true_group_count():
+ v = _blobs()
+ scores = {k: C.silhouette(v, C.kmeans(v, k)[0]) for k in (2, 3, 5)}
+ assert scores[3] == max(scores.values())
+
+
+def test_silhouette_is_undefined_for_one_bucket():
+ assert C.silhouette(_blobs(), np.zeros(24, dtype=int)) == -1.0
+
+
+def test_project_2d_returns_two_dimensions_and_is_centred():
+ xy = C.project_2d(_blobs())
+ assert xy.shape == (24, 2)
+ assert np.allclose(xy.mean(0), 0, atol=1e-9)
+
+
+def test_project_keeps_the_requested_number_of_components():
+ assert C.project(_blobs(), 3).shape == (24, 3)
+
+
+def test_reducing_before_clustering_beats_clustering_raw():
+ """The reason CLUSTER_DIMS exists. In high dimensions over few points every pair sits at a
+ similar distance and k-means has nothing to bite on — measured on the real 102-skill library,
+ raw 1024d scored +0.05 against +0.28 over 10 components. Reproduced here with padded noise
+ dimensions so the property is checked, not just remembered."""
+ rng = np.random.default_rng(3)
+ signal = _blobs(seed=2)
+ noise = rng.normal(0, 0.6, (len(signal), 300))
+ wide = C._normalise(np.hstack([signal, noise]))
+ raw = max(C.silhouette(wide, C.kmeans(wide, k)[0]) for k in (3, 4, 5))
+ reduced = C._normalise(C.project(wide, 10))
+ cut = max(C.silhouette(reduced, C.kmeans(reduced, k)[0]) for k in (3, 4, 5))
+ assert cut > raw
+
+
+def test_top_terms_favours_what_is_distinctive_not_what_is_common():
+ """Every skill says "use"; only one group says "invoice". Raw frequency returns the first."""
+ texts = ["use invoice billing ledger", "use invoice payment ledger",
+ "use pytest fixture mock", "use pytest assert mock"]
+ assert "invoice" in C.top_terms(texts, [0, 1], k=3)
+ assert "use" not in C.top_terms(texts, [0, 1], k=3)
+
+
+def test_top_terms_drops_stopwords_and_short_tokens():
+ texts = ["the and for a bb kubernetes", "the and for a bb postgres"]
+ assert C.top_terms(texts, [0], k=5) == ["kubernetes"]
+
+
+def test_label_for_names_a_bucket_from_its_terms():
+ assert C.label_for(["aws", "lambda", "deploy", "iam"]) == "aws · lambda · deploy"
+ assert C.label_for([]) == "unlabelled"
+
+
+def test_skill_texts_lead_with_the_name_as_words():
+ """Hyphenated names carry the topic in this library, and an embedder splits them better as
+ words than as one token."""
+ s = type("S", (), {"name": "aws-cdk", "description": "Infra as code."})()
+ assert C.skill_texts([s]) == ["aws cdk. Infra as code."]
+
+
+def test_build_refuses_a_library_too_small_to_cluster(monkeypatch):
+ monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda *a, **k: [])
+ with pytest.raises(SystemExit, match="SKILL_ROUTER_PATHS"):
+ C.build(log=lambda *a: None)
+
+
+def test_build_shapes_the_payload_the_ui_reads(monkeypatch, tmp_path):
+ names = ["billing-runbook", "invoice-audit", "pytest-fixtures", "mock-patterns",
+ "k8s-deploy", "helm-charts", "prose-edit", "headline-cuts"]
+ skills = [type("S", (), {"name": n, "description": n.replace("-", " "), "metadata": {}})()
+ for n in names]
+ monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda *a, **k: skills)
+ rng = np.random.default_rng(1)
+ vecs = {n: rng.normal(size=6) + (i // 2) * 3 for i, n in enumerate(names)}
+ monkeypatch.setattr("ingot.mcp_server.embedding.build_embedding",
+ lambda *a, **k: type("E", (), {"embed": lambda self, t: [
+ vecs[n] for n in names]})())
+ monkeypatch.setattr(C, "K_MAX", 4)
+ data = C.build(log=lambda *a: None)
+ assert data["n_skills"] == 8 and data["clusters"]
+ assert sum(c["size"] for c in data["clusters"]) == 8 # every skill lands somewhere
+ assert data["clusters"] == sorted(data["clusters"], key=lambda c: -c["size"])
+ member = data["clusters"][0]["members"][0]
+ assert set(member) == {"name", "xy", "provenance"} and len(member["xy"]) == 2
+ assert {m["name"] for c in data["clusters"] for m in c["members"]} == set(names)
diff --git a/tests/test_compat.py b/tests/test_compat.py
index 5eb6516..7a828ff 100644
--- a/tests/test_compat.py
+++ b/tests/test_compat.py
@@ -4,7 +4,7 @@
import pytest
-from optimize import compat
+from ingot.optimize import compat
def test_compat_models_parses_env_else_defaults(monkeypatch):
@@ -15,10 +15,28 @@ def test_compat_models_parses_env_else_defaults(monkeypatch):
assert compat.compat_models() == ["solo/model"]
+def test_the_tasks_own_checklist_reaches_the_judge(monkeypatch):
+ """Without it every task grades on judge()'s generic four, which any capable model passes on an
+ easy task — so the sweep measures the tasks' difficulty rather than the skill."""
+ checklist = [{"id": "cites_success_criterion", "criterion": "cites a criterion", "weight": 3,
+ "dimension": "correctness"}]
+ seen = {}
+
+ def fake_judge(task, rubric, answer, **kwargs):
+ seen.update(kwargs)
+ return {"score": 1.0}
+
+ monkeypatch.setattr(compat, "judge", fake_judge)
+ monkeypatch.setattr(compat, "invoke_retry",
+ lambda llm, messages: type("M", (), {"content": "answer"})())
+ compat._score(object(), "system", {"task": "t", "rubric": "r", "checklist": checklist})
+ assert seen["checklist"] == checklist
+
+
def test_run_compat_sweeps_models_and_computes_lift(tmp_path, monkeypatch):
(tmp_path / "tailwind").mkdir()
(tmp_path / "tailwind" / "SKILL.md").write_text("x")
- monkeypatch.setattr(compat, "SKILLS_DIR", tmp_path)
+ monkeypatch.setattr(compat, "resolve_skill_dir", lambda name: tmp_path / name)
monkeypatch.setattr(compat, "COMPAT_DIR", tmp_path / "out")
monkeypatch.setenv("COMPAT_MODELS", "m1,m2")
monkeypatch.setattr(compat, "load_tasks",
@@ -26,7 +44,8 @@ def test_run_compat_sweeps_models_and_computes_lift(tmp_path, monkeypatch):
monkeypatch.setattr(compat, "optimizable_components", lambda d: {"description": "d", "body": "THEBODY"})
monkeypatch.setattr(compat, "_llm", lambda model: model) # pass the model name through as the "llm"
# the skill arm serves THEBODY (score 0.9); the no-skill baseline does not (0.2)
- monkeypatch.setattr(compat, "_score", lambda llm, system, task: 0.9 if "THEBODY" in system else 0.2)
+ monkeypatch.setattr(compat, "_score",
+ lambda llm, system, task, role="compat": 0.9 if "THEBODY" in system else 0.2)
out = compat.run_compat("tailwind", log=lambda *a: None)
assert set(out["models"]) == {"m1", "m2"}
@@ -39,7 +58,100 @@ def test_run_compat_sweeps_models_and_computes_lift(tmp_path, monkeypatch):
assert written["skill"] == "tailwind" and set(written["models"]) == {"m1", "m2"}
+def test_baseline_arm_is_cached_across_runs(tmp_path, monkeypatch):
+ """The no-skill baseline does not depend on the skill body, so a re-sweep after editing a skill
+ must not pay for it twice — that was half the cost of every run after the first."""
+ (tmp_path / "tailwind").mkdir()
+ (tmp_path / "tailwind" / "SKILL.md").write_text("x")
+ monkeypatch.setattr(compat, "resolve_skill_dir", lambda name: tmp_path / name)
+ monkeypatch.setattr(compat, "COMPAT_DIR", tmp_path / "out")
+ monkeypatch.setattr(compat, "BASELINE_CACHE_DIR", tmp_path / "cache")
+ monkeypatch.setenv("COMPAT_MODELS", "m1")
+ holdout = [{"task": "t", "rubric": "r"}]
+ monkeypatch.setattr(compat, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(compat, "_llm", lambda model: model)
+
+ served = []
+
+ def fake_score(llm, system, task, role="compat"):
+ served.append("skill" if "THEBODY" in system else "baseline")
+ return 0.9 if "THEBODY" in system else 0.2
+
+ monkeypatch.setattr(compat, "optimizable_components", lambda d: {"description": "d", "body": "THEBODY"})
+ monkeypatch.setattr(compat, "_score", fake_score)
+
+ first = compat.run_compat("tailwind", log=lambda *a: None)
+ assert served == ["skill", "baseline"]
+
+ served.clear()
+ second = compat.run_compat("tailwind", log=lambda *a: None)
+ assert served == ["skill"], "the baseline arm was re-run instead of reused"
+ assert second["models"]["m1"]["baseline_mean"] == first["models"]["m1"]["baseline_mean"]
+
+
+def test_compat_serves_the_local_model_from_the_local_endpoint(monkeypatch):
+ """AGENT_MODEL is served by this box's own endpoint — a free row in the grid. Every other slug
+ is hosted. Routing the whole sweep through one endpoint is what forced a by-hand override."""
+ monkeypatch.setenv("AGENT_MODEL", "dot-backbone")
+ monkeypatch.setenv("MODEL_BASE_URL", "http://local:8011/v1")
+ monkeypatch.setenv("BASE_URL", "https://openrouter.ai/api/v1")
+ monkeypatch.setenv("MODEL_API_KEY", "local-key")
+ monkeypatch.setenv("API_KEY", "hosted-key")
+
+ assert str(compat._llm("dot-backbone").openai_api_base) == "http://local:8011/v1"
+ assert str(compat._llm("anthropic/claude-sonnet-4.5").openai_api_base) == "https://openrouter.ai/api/v1"
+
+
+def _stub_sweep(tmp_path, monkeypatch, models: str, score):
+ (tmp_path / "tailwind").mkdir(exist_ok=True)
+ (tmp_path / "tailwind" / "SKILL.md").write_text("x")
+ monkeypatch.setattr(compat, "resolve_skill_dir", lambda name: tmp_path / name)
+ monkeypatch.setattr(compat, "COMPAT_DIR", tmp_path / "out")
+ monkeypatch.setattr(compat, "BASELINE_CACHE_DIR", tmp_path / "cache")
+ monkeypatch.setenv("COMPAT_MODELS", models)
+ monkeypatch.setattr(compat, "load_tasks",
+ lambda skill: ([], [{"task": "t", "rubric": "r"}], {}))
+ monkeypatch.setattr(compat, "optimizable_components",
+ lambda d: {"description": "d", "body": "THEBODY"})
+ monkeypatch.setattr(compat, "_llm", lambda model: model)
+ monkeypatch.setattr(compat, "_score", score)
+
+
+def test_an_unreachable_model_does_not_discard_the_rows_already_paid_for(tmp_path, monkeypatch):
+ """A slug with no ZDR-qualified endpoint 404s on its first call. Before this the exception
+ escaped the sweep, so an earlier model's scores were computed, billed, and then thrown away
+ with no matrix written at all."""
+ def score(llm, system, task, role="compat"):
+ if llm == "gone/model":
+ raise RuntimeError("Error code: 404 - No endpoints found for gone/model.")
+ return 0.9 if "THEBODY" in system else 0.2
+
+ _stub_sweep(tmp_path, monkeypatch, "ok/model,gone/model", score)
+ out = compat.run_compat("tailwind", log=lambda *a: None)
+
+ assert out["models"]["ok/model"]["lift"] == pytest.approx(0.7)
+ assert "404" in out["models"]["gone/model"]["error"]
+ # An unavailable row must not read as a measured zero: that is the reading a reader most
+ # wants to make, and it is the opposite of the truth.
+ assert "lift" not in out["models"]["gone/model"]
+ written = json.loads((tmp_path / "out" / "tailwind.json").read_text())
+ assert set(written["models"]) == {"ok/model", "gone/model"}
+
+
+def test_a_sweep_where_no_model_could_be_reached_fails_loudly(tmp_path, monkeypatch):
+ """Writing an all-error matrix would leave a file that looks like a completed measurement."""
+ def score(llm, system, task, role="compat"):
+ raise RuntimeError("Error code: 401 - no key")
+
+ _stub_sweep(tmp_path, monkeypatch, "a/model,b/model", score)
+ with pytest.raises(SystemExit, match="nothing was measured"):
+ compat.run_compat("tailwind", log=lambda *a: None)
+ assert not (tmp_path / "out" / "tailwind.json").exists()
+
+
def test_run_compat_rejects_unknown_skill(tmp_path, monkeypatch):
- monkeypatch.setattr(compat, "SKILLS_DIR", tmp_path)
- with pytest.raises(SystemExit, match="No skill named"):
+ """An unresolvable name is a roots misconfiguration, and the message has to say so — the old
+ text pointed at skills/, a directory the operator may never have configured."""
+ monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda *a, **k: [])
+ with pytest.raises(SystemExit, match="SKILL_ROUTER_PATHS"):
compat.run_compat("nope", log=lambda *a: None)
diff --git a/tests/test_compose_managed.py b/tests/test_compose_managed.py
new file mode 100644
index 0000000..c4f6846
--- /dev/null
+++ b/tests/test_compose_managed.py
@@ -0,0 +1,103 @@
+"""The tracked deployment must declare the invariant it claims.
+
+`docker compose up` with no `-f` is the configuration a reader gets, and it must not launch a
+writable stack while the README describes controlled activation. These are static reads of the
+tracked YAML: they cannot prove the containers behave, which needs a daemon and the compose smoke
+script, but they do catch the failure that actually happened — an invariant that lived only in one
+machine's untracked overlay."""
+import yaml
+import pytest
+from pathlib import Path
+
+ROOT = Path(__file__).resolve().parent.parent
+SERVED = "/app/skills"
+
+
+def _compose(name):
+ return yaml.safe_load((ROOT / name).read_text(encoding="utf-8"))
+
+
+def _mounts(service, target):
+ for mount in service.get("volumes") or []:
+ if isinstance(mount, str) and mount.split(":")[1:2] == [target]:
+ yield mount
+
+
+@pytest.fixture(scope="module")
+def managed():
+ return _compose("docker-compose.yml")
+
+
+def test_the_default_stack_has_exactly_one_writer_of_the_served_library(managed):
+ writers = [name for name, service in managed["services"].items()
+ if any(not mount.endswith(":ro") for mount in _mounts(service, SERVED))]
+
+ assert writers == [], f"these services can change what is served without an approval: {writers}"
+
+
+def test_every_service_that_serves_the_library_mounts_the_vault(managed):
+ """A service reading a different directory than the publisher writes serves stale bytes and
+ reports no drift, because nothing is comparing the two."""
+ sources = {mount.split(":")[0]
+ for service in managed["services"].values()
+ for mount in _mounts(service, SERVED)}
+
+ assert sources == {"./vault"}
+
+
+def test_the_publisher_is_in_the_default_stack_and_owns_the_vault(managed):
+ publisher = managed["services"]["publisher"]
+
+ assert "profiles" not in publisher, "the one writer must not be opt-in"
+ assert publisher["environment"]["INGOT_VAULT_PATH"] == "/app/vault"
+ writable = [mount for mount in publisher["volumes"]
+ if mount.startswith("./vault:") and not mount.endswith(":ro")]
+ assert writable, "the publisher must be able to write the vault"
+
+
+def test_the_publisher_and_the_console_run_as_the_same_user(managed):
+ """Approval writes receipts at mode 0700. A publisher running as a different user sees an
+ empty queue and approvals never publish — the exact stall this deployment already hit."""
+ assert managed["services"]["publisher"]["user"] == managed["services"]["ui"]["user"]
+
+
+def test_every_service_that_touches_state_names_where_it_lives(managed):
+ """State no longer defaults to a directory beside the code. A service that mounts a state
+ volume without naming the paths would write its review queue and receipts inside the image,
+ where the next `docker compose build` discards them."""
+ for name, service in managed["services"].items():
+ mounts = [mount for mount in (service.get("volumes") or []) if isinstance(mount, str)
+ and mount.split(":")[1:2] and mount.split(":")[1].startswith(("/app/skills",
+ "/app/runs",
+ "/app/vault"))]
+ if not mounts:
+ continue
+ environment = service.get("environment") or {}
+ assert "INGOT_RUNS" in environment, f"{name} mounts state without naming INGOT_RUNS"
+ assert environment.get("INGOT_LIBRARY"), f"{name} mounts state without naming INGOT_LIBRARY"
+
+
+def test_the_default_backend_is_local(managed):
+ backend = managed["services"]["publisher"]["environment"]["INGOT_PUBLISH_BACKEND"]
+
+ assert backend.startswith("${INGOT_PUBLISH_BACKEND:-local}")
+
+
+def test_the_development_override_is_explicit_about_being_unmanaged():
+ dev = _compose("compose.dev.yaml")
+
+ assert dev["services"]["publisher"]["deploy"]["replicas"] == 0
+ writable = {name for name, service in dev["services"].items()
+ if any(not mount.endswith(":ro") for mount in _mounts(service, SERVED))}
+ assert writable, "the development stack is the writable one; that is its whole purpose"
+ assert dev["services"]["unmanaged"]["command"] == ["ingot", "status"]
+ assert dev["services"]["unmanaged"]["environment"]["INGOT_MODE"] == "dev"
+ assert dev["services"]["ui"]["environment"]["INGOT_MODE"] == "dev"
+
+
+def test_the_forge_override_never_infers_the_repository():
+ forge = _compose("compose.forge.yaml")
+ environment = forge["services"]["publisher"]["environment"]
+
+ assert environment["INGOT_PUBLISH_BACKEND"] == "forge"
+ assert environment["INGOT_FORGE_REPOSITORY"].startswith("${INGOT_FORGE_REPOSITORY:?")
diff --git a/tests/test_decisions.py b/tests/test_decisions.py
new file mode 100644
index 0000000..42aaccb
--- /dev/null
+++ b/tests/test_decisions.py
@@ -0,0 +1,271 @@
+"""`ingot pending`, `approve`, `reject`, `history`, `rollback`.
+
+The command line wraps the services the console already calls. What these tests protect is that it
+stays a wrapper: no second approval path, no direct activation helper, and no verb that changes a
+served byte. Approval and rollback queue a receipt; the publisher is what acts on it."""
+import json
+
+import pytest
+
+from ingot import cli, decisions
+from ingot.mcp_server.registry import optimizable_components, skill_revision
+from ingot.optimize import promote as P
+from ingot.optimize import publication as Q
+
+
+def _library(tmp_path, monkeypatch):
+ root = tmp_path / "skills"
+ root.mkdir()
+ monkeypatch.setenv("INGOT_LIBRARY", str(root))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root))
+ return root
+
+
+def _skill(root, name="pdf", body="old body"):
+ directory = root / name
+ directory.mkdir(exist_ok=True)
+ (directory / "SKILL.md").write_text(
+ f"---\nname: {name}\ndescription: Merge PDFs.\n---\n{body}\n")
+ return directory
+
+
+def _quarantine(directory, *, promotable=True, blocked=()):
+ skill = directory.name
+ champion = optimizable_components(directory)
+ challenger = {**champion, "body": "new body"}
+ pending = {
+ "skill": skill,
+ "champion_components": champion,
+ "challenger_components": challenger,
+ "gate": {"promotable": promotable, "blocked": list(blocked)},
+ "evidence": {"champion": {"revision": skill_revision(directory)},
+ "challenger": {"revision": skill_revision(directory, challenger)}},
+ }
+ P.save_pending(skill, pending)
+ return pending
+
+
+# --------------------------------------------------------------------------- the four words
+
+@pytest.mark.parametrize("record,expected", [
+ ({"state": "approved_publishing"}, decisions.APPROVED),
+ ({"state": "publishing"}, decisions.PUBLISHING),
+ ({"state": "awaiting_merge"}, decisions.PUBLISHING),
+ ({"state": "active"}, decisions.PUBLISHED),
+ ({"state": "publishing", "last_error": "git push: rejected"}, decisions.FAILED),
+ ({"state": "approved_publishing", "last_error": "vault is dirty"}, decisions.FAILED),
+])
+def test_a_receipt_reports_the_word_an_operator_has_to_act_on(record, expected):
+ """`failed` reads off `last_error`, not the state: a failed attempt leaves the receipt in
+ whatever state it was working through, so a status taken from the state alone hides it."""
+ assert decisions.release_status(record)["status"] == expected
+
+
+def test_an_active_receipt_that_once_failed_reads_as_published():
+ """A receipt that recovered is published. `last_error` is cleared on success, and reporting a
+ stale error over a completed activation would send someone chasing a resolved failure."""
+ assert decisions.release_status(
+ {"state": "active", "last_error": ""})["status"] == decisions.PUBLISHED
+
+
+def test_no_receipt_is_not_a_status():
+ assert decisions.release_status(None) is None
+
+
+# --------------------------------------------------------------------------- pending
+
+def test_pending_lists_what_is_waiting_with_its_gate_verdict(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ _quarantine(_skill(root, "pdf"))
+ _quarantine(_skill(root, "tailwind"), promotable=False, blocked=["holdout regression"])
+
+ result = decisions.pending_view()
+
+ assert [entry["skill"] for entry in result["pending"]] == ["pdf", "tailwind"]
+ assert result["pending"][0]["promotable"] is True
+ assert result["pending"][1]["promotable"] is False
+ assert result["pending"][1]["blocked"] == ["holdout regression"]
+
+
+def test_pending_still_shows_a_change_after_approval_consumes_its_record(tmp_path, monkeypatch):
+ """Approval consumes the pending record at publication, so the window in which someone is most
+ likely to ask where their change went is exactly the window the queue cannot see."""
+ root = _library(tmp_path, monkeypatch)
+ _quarantine(_skill(root))
+ decisions.approve("pdf", actor="admin")
+ P.pending_path("pdf").unlink()
+
+ result = decisions.pending_view()
+
+ assert result["pending"] == []
+ assert [entry["skill"] for entry in result["publishing"]] == ["pdf"]
+ assert result["publishing"][0]["status"] == decisions.APPROVED
+
+
+def test_pending_reports_an_empty_queue_rather_than_nothing(tmp_path, monkeypatch, capsys):
+ _library(tmp_path, monkeypatch)
+
+ assert cli.main(["pending"]) == 0
+ assert "Nothing waiting." in capsys.readouterr().out
+
+
+# --------------------------------------------------------------------------- decisions
+
+def test_approve_queues_a_receipt_and_changes_no_served_byte(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ directory = _skill(root)
+ _quarantine(directory)
+ before = (directory / "SKILL.md").read_bytes()
+
+ result = decisions.approve("pdf", actor="admin")
+
+ assert result["publication"]["status"] == decisions.APPROVED
+ assert result["publication"]["action"] == "promote"
+ assert (directory / "SKILL.md").read_bytes() == before
+ assert Q.load_publication(result["publication"]["id"])["state"] == "approved_publishing"
+
+
+def test_approve_refuses_a_change_the_evidence_gate_blocked(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ _quarantine(_skill(root), promotable=False, blocked=["holdout regression"])
+
+ with pytest.raises(ValueError, match="evidence gate blocked"):
+ decisions.approve("pdf", actor="admin")
+
+ assert Q.publication_for_skill("pdf") is None
+
+
+def test_approve_refuses_a_second_publication_for_the_same_skill(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ _quarantine(_skill(root))
+ P._snapshot_absence("pdf")
+ decisions.approve("pdf", actor="admin")
+
+ with pytest.raises(ValueError, match="already in progress"):
+ decisions.rollback("pdf", P.ABSENT_REVISION, actor="admin")
+
+
+def test_reject_discards_the_change_and_records_the_reason(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ directory = _skill(root)
+ _quarantine(directory)
+ before = (directory / "SKILL.md").read_bytes()
+
+ result = decisions.reject("pdf", actor="admin", reason=" the body loses the API note ")
+
+ assert not P.pending_path("pdf").exists()
+ assert (directory / "SKILL.md").read_bytes() == before
+ assert result["publication"] is None
+ record = P.read_audit()["records"][0]
+ assert record["action"] == "reject" and record["actor"] == "admin"
+ assert record["reason"] == "the body loses the API note"
+
+
+def test_reject_without_a_pending_change_is_a_refusal_not_a_crash(tmp_path, monkeypatch):
+ _library(tmp_path, monkeypatch)
+
+ with pytest.raises(ValueError, match="no pending change"):
+ decisions.reject("pdf", actor="admin")
+
+
+def test_rollback_queues_the_snapshot_without_restoring_it(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ directory = _skill(root)
+ champion = skill_revision(directory)
+ P._snapshot(directory, "pdf", champion)
+ _skill(root, body="a later revision")
+
+ result = decisions.rollback("pdf", champion, actor="admin")
+
+ assert result["publication"]["action"] == "rollback"
+ assert result["publication"]["revision"] == champion
+ assert "a later revision" in (directory / "SKILL.md").read_text()
+
+
+def test_rollback_to_a_revision_with_no_snapshot_is_refused(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ _skill(root)
+
+ with pytest.raises(ValueError, match="no snapshot"):
+ decisions.rollback("pdf", "f" * 64, actor="admin")
+
+
+# --------------------------------------------------------------------------- history
+
+def test_history_gathers_snapshots_receipts_and_decisions_for_one_skill(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ directory = _skill(root)
+ champion = skill_revision(directory)
+ P._snapshot(directory, "pdf", champion)
+ _quarantine(directory)
+ decisions.approve("pdf", actor="admin")
+ _quarantine(_skill(root, "tailwind"))
+ decisions.reject("tailwind", actor="admin", reason="not this one")
+
+ result = decisions.history_view("pdf")
+
+ assert [entry["revision"] for entry in result["revisions"]] == [champion]
+ assert [entry["status"] for entry in result["publications"]] == [decisions.APPROVED]
+ assert [record["skill"] for record in result["audit"]] == [] # approval audits on activation
+
+
+def test_history_never_reports_another_skills_decisions(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ _quarantine(_skill(root, "tailwind"))
+ decisions.reject("tailwind", actor="admin", reason="not this one")
+
+ assert decisions.history_view("pdf")["audit"] == []
+ assert decisions.history_view("tailwind")["audit"][0]["reason"] == "not this one"
+
+
+def test_history_refuses_an_invalid_skill_name(tmp_path, monkeypatch):
+ _library(tmp_path, monkeypatch)
+
+ with pytest.raises(ValueError):
+ decisions.history_view("../etc")
+
+
+# --------------------------------------------------------------------------- the command line
+
+def test_the_command_line_reports_the_receipt_and_that_nothing_is_served_yet(tmp_path, monkeypatch,
+ capsys):
+ root = _library(tmp_path, monkeypatch)
+ _quarantine(_skill(root))
+
+ assert cli.main(["approve", "pdf", "--actor", "admin"]) == 0
+
+ out = capsys.readouterr().out
+ assert "Approved 'pdf'" in out
+ assert "approved" in out
+ assert "The served library is unchanged" in out
+
+
+def test_a_refusal_exits_one_without_a_traceback(tmp_path, monkeypatch, capsys):
+ _library(tmp_path, monkeypatch)
+
+ assert cli.main(["approve", "pdf"]) == 1
+ assert "ingot approve: no pending challenger" in capsys.readouterr().err
+
+
+def test_every_decision_payload_is_versioned(tmp_path, monkeypatch, capsys):
+ root = _library(tmp_path, monkeypatch)
+ _quarantine(_skill(root))
+
+ assert cli.main(["approve", "pdf", "--actor", "admin", "--json"]) == 0
+
+ assert json.loads(capsys.readouterr().out)["schema_version"] == decisions.DECISION_SCHEMA
+
+
+def test_an_explicit_actor_outranks_the_environment(tmp_path, monkeypatch):
+ """A decision nobody can be asked about is not much of an audit trail, so the actor is never
+ blank: an explicit flag, then INGOT_ACTOR, then whoever is running the command."""
+ root = _library(tmp_path, monkeypatch)
+ _quarantine(_skill(root, "pdf"))
+ _quarantine(_skill(root, "tailwind"))
+ monkeypatch.setenv("INGOT_ACTOR", "from-the-environment")
+
+ assert cli.main(["reject", "pdf", "--actor", "reviewer", "--reason", "no"]) == 0
+ assert cli.main(["reject", "tailwind", "--reason", "no"]) == 0
+
+ actors = {record["skill"]: record["actor"] for record in P.read_audit()["records"]}
+ assert actors == {"pdf": "reviewer", "tailwind": "from-the-environment"}
diff --git a/tests/test_delivery.py b/tests/test_delivery.py
new file mode 100644
index 0000000..a7e7822
--- /dev/null
+++ b/tests/test_delivery.py
@@ -0,0 +1,290 @@
+"""Delivery targets: the approved revision reaching more than one place, without a second writer.
+
+Everything here guards the same boundary. A delivery target is somewhere the publisher installs an
+approved revision *after* the vault already carries it -- a native agent's skill root, say. It is
+not a second authority: it never decides what is approved, it cannot activate anything, and a
+target the publisher fails to write must not leave the release looking finished. The tests below
+are the difference between that and a `cp` in a cron job."""
+import json
+import os
+import stat
+
+import pytest
+
+from ingot import delivery
+from ingot.mcp_server.registry import skill_revision
+from ingot.optimize import promote
+
+PNG = b"\x89PNG\r\n\x1a\n" + bytes(range(256))
+
+
+@pytest.fixture
+def vault(tmp_path):
+ root = tmp_path / "vault"
+ root.mkdir()
+ return root
+
+
+def _package(root, name="pdf", body="Combine PDFs.", asset=PNG):
+ """A skill directory with a binary asset and an executable, the two things a text-only copy
+ would quietly damage."""
+ directory = root / name
+ directory.mkdir(parents=True, exist_ok=True)
+ (directory / "SKILL.md").write_text(
+ f"---\nname: {name}\ndescription: Merge and split PDF files.\n---\n\n{body}\n",
+ encoding="utf-8")
+ if asset is not None:
+ (directory / "assets").mkdir(exist_ok=True)
+ (directory / "assets" / "logo.png").write_bytes(asset)
+ runner = directory / "run.sh"
+ runner.write_text("#!/bin/sh\necho hi\n", encoding="utf-8")
+ runner.chmod(0o755)
+ return directory
+
+
+# --- configuration ------------------------------------------------------------------------------
+
+def test_an_unconfigured_deployment_delivers_to_the_vault_and_nowhere_else(vault):
+ targets = delivery.load_targets({}, vault=vault)
+ assert [(target.name, target.kind, target.root) for target in targets] == [
+ ("vault", delivery.MANAGED_MCP, vault)]
+
+
+def test_a_filesystem_target_joins_the_vault_rather_than_replacing_it(tmp_path, vault):
+ native = tmp_path / "claude" / "skills"
+ targets = delivery.load_targets(
+ {delivery.TARGETS: f"vault=managed-mcp:{vault},claude=filesystem:{native}"}, vault=vault)
+ assert [target.name for target in targets] == ["vault", "claude"]
+ assert targets[1].kind == delivery.FILESYSTEM
+ assert targets[1].root == native
+
+
+def test_the_vault_is_delivered_to_even_when_the_configuration_forgets_it(tmp_path, vault):
+ """Dropping the managed target from the list must not stop serving MCP. The vault is where
+ publication authority lives; a delivery list is not the place to switch it off."""
+ native = tmp_path / "native"
+ targets = delivery.load_targets({delivery.TARGETS: f"claude=filesystem:{native}"}, vault=vault)
+ assert [(target.name, target.kind) for target in targets] == [
+ ("vault", delivery.MANAGED_MCP), ("claude", delivery.FILESYSTEM)]
+
+
+@pytest.mark.parametrize("spec, reason", [
+ ("claude=carrier-pigeon:/tmp/x", "unknown delivery kind"),
+ ("=filesystem:/tmp/x", "invalid delivery target name"),
+ ("Claude Skills=filesystem:/tmp/x", "invalid delivery target name"),
+ ("claude=filesystem:relative/path", "must be an absolute path"),
+ ("claude=filesystem:/tmp/x,claude=filesystem:/tmp/y", "duplicate delivery target"),
+ ("a=filesystem:/tmp/x,b=filesystem:/tmp/x", "same directory"),
+ ("claude=filesystem", "expected name=kind:path"),
+])
+def test_an_unusable_target_is_refused_at_configuration_time(spec, reason, vault):
+ """The publisher validates its configuration before it will start. A delivery target that
+ cannot work has to fail there, not on the first approval that tries to use it."""
+ with pytest.raises(ValueError, match=reason):
+ delivery.load_targets({delivery.TARGETS: spec}, vault=vault)
+
+
+def test_a_managed_target_that_is_not_the_vault_is_refused(tmp_path, vault):
+ with pytest.raises(ValueError, match="managed-mcp target must be the vault"):
+ delivery.load_targets({delivery.TARGETS: f"x=managed-mcp:{tmp_path / 'elsewhere'}"},
+ vault=vault)
+
+
+def test_a_filesystem_target_inside_the_vault_is_refused(vault):
+ """It would write into the checkout the publisher just committed, so the vault would drift from
+ its own release the moment delivery finished."""
+ with pytest.raises(ValueError, match="inside the vault"):
+ delivery.load_targets({delivery.TARGETS: f"x=filesystem:{vault / 'nested'}"}, vault=vault)
+
+
+# --- installing ---------------------------------------------------------------------------------
+
+def test_the_delivered_target_is_byte_identical_to_the_source(tmp_path, vault):
+ source = _package(vault)
+ native = tmp_path / "native"
+ target = delivery.Target("claude", delivery.FILESYSTEM, native)
+
+ delivery.install(target, "pdf", source, skill_revision(source))
+
+ delivered = native / "pdf"
+ assert (delivered / "assets" / "logo.png").read_bytes() == PNG
+ assert (delivered / "SKILL.md").read_bytes() == (source / "SKILL.md").read_bytes()
+ assert skill_revision(delivered) == skill_revision(source)
+
+
+def test_the_executable_bit_survives_delivery(tmp_path, vault):
+ source = _package(vault)
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+
+ delivery.install(target, "pdf", source, skill_revision(source))
+
+ mode = (tmp_path / "native" / "pdf" / "run.sh").stat().st_mode
+ assert mode & stat.S_IXUSR
+
+
+def test_delivering_a_revision_the_target_already_holds_changes_nothing(tmp_path, vault):
+ source = _package(vault)
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+ revision = skill_revision(source)
+ delivery.install(target, "pdf", source, revision)
+ before = (tmp_path / "native" / "pdf").stat().st_mtime_ns
+
+ delivery.install(target, "pdf", source, revision)
+
+ assert (tmp_path / "native" / "pdf").stat().st_mtime_ns == before
+
+
+def test_the_displaced_target_is_snapshotted_before_it_is_replaced(tmp_path, vault):
+ """Whatever was there is recoverable, including bytes that never came from a release: a target
+ someone edited by hand is exactly the case where the previous content matters most."""
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+ displaced = _package(tmp_path / "old", body="The bytes that were there.")
+ delivery.install(target, "pdf", displaced, skill_revision(displaced))
+ displaced_revision = skill_revision(tmp_path / "native" / "pdf")
+
+ replacement = _package(vault, body="The approved bytes.")
+ delivery.install(target, "pdf", replacement, skill_revision(replacement))
+
+ stored = promote.revisions_dir() / "pdf" / displaced_revision
+ assert (stored / "SKILL.md").read_text(encoding="utf-8").endswith("The bytes that were there.\n")
+
+
+def test_a_delivery_that_cannot_finish_leaves_the_previous_revision_in_place(tmp_path, vault,
+ monkeypatch):
+ """The half-written target is the failure this exists to prevent: an agent loading a skill
+ directory that is neither the old revision nor the new one."""
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+ original = _package(tmp_path / "old", body="The bytes that were there.")
+ delivery.install(target, "pdf", original, skill_revision(original))
+ original_revision = skill_revision(tmp_path / "native" / "pdf")
+
+ replacement = _package(vault, body="The approved bytes.")
+ monkeypatch.setattr(delivery.os, "replace", _explode)
+ with pytest.raises(OSError):
+ delivery.install(target, "pdf", replacement, skill_revision(replacement))
+
+ assert skill_revision(tmp_path / "native" / "pdf") == original_revision
+ assert [entry for entry in (tmp_path / "native").iterdir() if entry.name != "pdf"] == []
+
+
+def _explode(*args, **kwargs):
+ raise OSError("no space left on device")
+
+
+def test_a_delivery_that_fails_after_displacing_the_target_puts_it_back(tmp_path, vault,
+ monkeypatch):
+ """The dangerous window. Replacing a directory takes two renames, and a failure between them
+ leaves the destination missing unless the displaced copy is moved back."""
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+ original = _package(tmp_path / "old", body="The bytes that were there.")
+ delivery.install(target, "pdf", original, skill_revision(original))
+ original_revision = skill_revision(tmp_path / "native" / "pdf")
+
+ real_replace = os.replace
+ restored = []
+
+ def fail_installing_the_staged_copy(source, destination, *args, **kwargs):
+ # Precisely the second rename of the swap: the displaced original is already out of the
+ # way and the staged copy is going in. Counting calls would catch the snapshot index write
+ # instead, which is not the window being tested.
+ if str(source).endswith(".tmp"):
+ raise OSError("no space left on device")
+ if str(source).endswith(".old"):
+ restored.append(destination)
+ return real_replace(source, destination, *args, **kwargs)
+
+ replacement = _package(vault, body="The approved bytes.")
+ monkeypatch.setattr(delivery.os, "replace", fail_installing_the_staged_copy)
+ with pytest.raises(OSError):
+ delivery.install(target, "pdf", replacement, skill_revision(replacement))
+
+ assert restored == [tmp_path / "native" / "pdf"], "the displaced directory was never moved back"
+ assert skill_revision(tmp_path / "native" / "pdf") == original_revision
+ assert sorted(entry.name for entry in (tmp_path / "native").iterdir()) == ["pdf"]
+
+
+def test_delivering_bytes_that_do_not_match_the_receipt_is_refused(tmp_path, vault):
+ """The check is against the revision the receipt names, not against the source directory, so a
+ source that was tampered with between approval and delivery cannot install itself."""
+ source = _package(vault)
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+
+ with pytest.raises(ValueError, match="does not match the approved revision"):
+ delivery.install(target, "pdf", source, "0" * 16)
+
+ assert not (tmp_path / "native" / "pdf").exists()
+
+
+def test_a_symlink_in_the_source_is_refused_rather_than_followed(tmp_path, vault):
+ """Following it would copy whatever it points at -- possibly from outside the vault entirely --
+ into an agent's skill root as an ordinary file."""
+ source = _package(vault)
+ # A link pointing *out* of the skill is already refused upstream: `skill_revision` will not
+ # hash one. A contained link is the case that reaches here, and copying it without `symlinks`
+ # silently turns it into a second regular file holding the same bytes.
+ (source / "readme.md").symlink_to(source / "SKILL.md")
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+
+ with pytest.raises(ValueError, match="symlink-unsupported"):
+ delivery.install(target, "pdf", source, skill_revision(source))
+
+ assert not (tmp_path / "native" / "pdf").exists()
+
+
+def test_delivering_an_absence_removes_the_skill_and_snapshots_it(tmp_path, vault):
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+ source = _package(vault)
+ delivery.install(target, "pdf", source, skill_revision(source))
+ revision = skill_revision(source)
+
+ delivery.install(target, "pdf", None, promote.ABSENT_REVISION)
+
+ assert not (tmp_path / "native" / "pdf").exists()
+ assert (promote.revisions_dir() / "pdf" / revision).is_dir()
+
+
+def test_removing_a_skill_the_target_never_held_is_not_an_error(tmp_path, vault):
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+ delivery.install(target, "pdf", None, promote.ABSENT_REVISION)
+ assert not (tmp_path / "native" / "pdf").exists()
+
+
+def test_delivery_never_writes_a_managed_target(tmp_path, vault):
+ """The vault is written by the publication commit and the fast-forward, and by nothing else.
+ A second writer there is the invariant this whole control plane exists to hold."""
+ _package(vault, body="What the vault holds.")
+ # Deliberately a *different* revision. Asking the managed target to install bytes it does not
+ # hold is the only way to tell the refusal apart from delivery being a no-op by luck.
+ elsewhere = _package(tmp_path / "other", body="Something else entirely.")
+ target = delivery.Target("vault", delivery.MANAGED_MCP, vault)
+ before = skill_revision(vault / "pdf")
+
+ assert delivery.install(target, "pdf", elsewhere, skill_revision(elsewhere)) is False
+
+ assert skill_revision(vault / "pdf") == before
+ assert sorted(path.name for path in vault.iterdir()) == ["pdf"]
+
+
+# --- observing ----------------------------------------------------------------------------------
+
+def test_a_target_reports_the_revision_it_actually_holds(tmp_path, vault):
+ source = _package(vault)
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+ delivery.install(target, "pdf", source, skill_revision(source))
+
+ assert delivery.observed(target, "pdf") == skill_revision(source)
+
+
+def test_a_target_edited_out_of_band_reports_the_new_bytes_not_the_old_ones(tmp_path, vault):
+ source = _package(vault)
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+ delivery.install(target, "pdf", source, skill_revision(source))
+
+ (tmp_path / "native" / "pdf" / "assets" / "logo.png").write_bytes(PNG + b"tampered")
+
+ assert delivery.observed(target, "pdf") != skill_revision(source)
+
+
+def test_a_missing_skill_reads_as_absent_rather_than_as_an_error(tmp_path, vault):
+ target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native")
+ assert delivery.observed(target, "pdf") == promote.ABSENT_REVISION
diff --git a/tests/test_distribution.py b/tests/test_distribution.py
new file mode 100644
index 0000000..2352498
--- /dev/null
+++ b/tests/test_distribution.py
@@ -0,0 +1,69 @@
+"""What the repository ships.
+
+A release controller must not arrive with an unexplained pending change already in its queue, and
+must not ship a served skill nobody approved. These read what git tracks rather than what happens
+to be in one working tree, because that is what a clone gets."""
+import subprocess
+from pathlib import Path
+
+import pytest
+
+ROOT = Path(__file__).resolve().parent.parent
+
+
+def _tracked(pathspec: str) -> list[str]:
+ result = subprocess.run(["git", "-C", str(ROOT), "ls-files", "--", pathspec],
+ capture_output=True, text=True)
+ if result.returncode:
+ pytest.skip("not a git checkout")
+ return [line for line in result.stdout.splitlines() if line]
+
+
+def test_a_clean_checkout_begins_with_no_live_pending_proposal():
+ """A tracked pending record would arrive as a real quarantined change in every clone: it would
+ show in `ingot pending`, be approvable, and publish something nobody proposed."""
+ assert _tracked("runs") == []
+
+
+def test_a_clean_checkout_serves_no_skill_it_did_not_publish():
+ """The demo library ships empty. A tracked skill would be served with no release receipt behind
+ it, which is precisely the UNMANAGED state `ingot status` exists to report."""
+ assert _tracked("skills") == ["skills/.gitkeep"]
+
+
+def test_the_vault_is_never_tracked():
+ """`ingot vault init` makes it a Git repository of its own; a tracked copy would be a second,
+ stale answer to what is published."""
+ assert _tracked("vault") == []
+
+
+@pytest.mark.parametrize("pathspec", ["runs", "skills/*/", "vault"])
+def test_state_paths_are_ignored_so_a_live_deployment_cannot_commit_itself(pathspec):
+ """A checkout used as a deployment writes into these. Without ignore rules the first `git add`
+ would commit a review queue and a set of receipts into the product."""
+ probe = {"runs": "runs/pending/probe.json",
+ "skills/*/": "skills/probe/SKILL.md",
+ "vault": "vault/registry.json"}[pathspec]
+ result = subprocess.run(["git", "-c", f"safe.directory={ROOT}", "-C", str(ROOT),
+ "check-ignore", "-q", probe],
+ capture_output=True)
+
+ assert result.returncode == 0, f"{probe} is not ignored"
+
+
+def test_the_package_claims_one_installed_name():
+ """`mcp_server` and `optimize` are far too generic to own on PyPI. They now live under `ingot`,
+ and a shim would have been self-defeating: a shim named `optimize` still claims `optimize`."""
+ import tomllib
+
+ configured = tomllib.loads((ROOT / "pyproject.toml").read_text())
+ packages = configured["tool"]["setuptools"]["packages"]
+
+ assert [name for name in packages if not name.startswith("ingot")] == []
+
+
+@pytest.mark.parametrize("name", ["mcp_server", "optimize"])
+def test_no_generic_package_reappears_at_the_repository_root(name):
+ """A stale `build/` directory from before the move will happily reinstall the old top-level
+ packages, so this checks what git tracks rather than what happens to be on disk."""
+ assert _tracked(name) == []
diff --git a/tests/test_draft.py b/tests/test_draft.py
index 83d7caa..d3c4581 100644
--- a/tests/test_draft.py
+++ b/tests/test_draft.py
@@ -1,9 +1,9 @@
-"""Unit tests for auto-drafted eval task sets (optimize.draft), LLM mocked."""
+"""Unit tests for auto-drafted eval task sets (ingot.optimize.draft), LLM mocked."""
import json
import pytest
-from optimize import draft as D
+from ingot.optimize import draft as D
class _FakeMsg:
@@ -41,6 +41,54 @@ def test_draft_raises_if_too_few_usable_tasks(monkeypatch):
D.draft_tasks("pdf", "d", "b", n=8)
+def test_draft_carries_a_weighted_checklist_per_task(monkeypatch):
+ """The checklist is what gives a task more than one measurement, so a drafted set has to carry
+ it -- otherwise only hand-written tasks ever get graded finely."""
+ check = {"id": "handles_empty", "criterion": "Handles an empty input file without raising.",
+ "weight": 4, "dimension": "completeness"}
+ _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r", "checklist": [check]} for i in range(4)])
+ out = D.draft_tasks("pdf", "d", "b", n=4)
+ assert out["train"][0]["checklist"] == [check]
+
+
+@pytest.mark.parametrize("bad, why", [
+ ({"id": "Has Spaces", "criterion": "a valid criterion here"}, "id is not snake_case"),
+ ({"id": "ok_id", "criterion": "short"}, "criterion too short to grade"),
+ ("not a dict", "not an object"),
+])
+def test_draft_drops_ungradeable_checks(monkeypatch, bad, why):
+ good = {"id": "keeps_this", "criterion": "A criterion long enough to grade.", "weight": 2,
+ "dimension": "correctness"}
+ _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r", "checklist": [good, bad]}
+ for i in range(4)])
+ out = D.draft_tasks("pdf", "d", "b", n=4)
+ assert [c["id"] for c in out["train"][0]["checklist"]] == ["keeps_this"], why
+
+
+def test_draft_deduplicates_check_ids(monkeypatch):
+ """Two checks with one id would collapse in the judge's JSON response, silently dropping a
+ check while its weight still counted against the total."""
+ dup = [{"id": "same", "criterion": "The first criterion text."},
+ {"id": "same", "criterion": "A different criterion, same id."}]
+ _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r", "checklist": dup} for i in range(4)])
+ assert len(D.draft_tasks("pdf", "d", "b", n=4)["train"][0]["checklist"]) == 1
+
+
+def test_draft_clamps_weights_and_defaults_unknown_dimensions(monkeypatch):
+ wild = {"id": "wild", "criterion": "A criterion long enough to grade.", "weight": 99,
+ "dimension": "vibes"}
+ _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r", "checklist": [wild]} for i in range(4)])
+ got = D.draft_tasks("pdf", "d", "b", n=4)["train"][0]["checklist"][0]
+ assert got["weight"] == 5 and got["dimension"] == "correctness"
+
+
+def test_draft_tolerates_a_task_with_no_checklist(monkeypatch):
+ """judge() falls back to its default checklist, so a missing one degrades to four dimensions
+ rather than failing the draft."""
+ _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r"} for i in range(4)])
+ assert D.draft_tasks("pdf", "d", "b", n=4)["train"][0]["checklist"] == []
+
+
def _mock_routing_llm(monkeypatch, positive, negative):
payload = json.dumps({"positive": positive, "negative": negative})
monkeypatch.setattr(D, "_llm", lambda: type("L", (), {"invoke": lambda self, p: _FakeMsg(payload)})())
diff --git a/tests/test_embedding.py b/tests/test_embedding.py
index 9c04c00..93ec32f 100644
--- a/tests/test_embedding.py
+++ b/tests/test_embedding.py
@@ -1,11 +1,13 @@
"""Embedding backends: model selection, and the Qwen ONNX pooling/prefix contract with a stubbed
session (no model download, no onnxruntime inference)."""
+import json
import sys
from types import SimpleNamespace
+from urllib.error import URLError
import numpy as np
-from mcp_server import embedding as E
+from ingot.mcp_server import embedding as E
def test_backend_selection_by_model_name():
@@ -144,3 +146,70 @@ def test_qwen_tolerates_empty_text():
def test_fastembed_backend_has_no_query_prefix():
# symmetric embedding is the pre-Qwen contract fastembed overrides rely on
assert E.FastembedEmbedding.embed_query is E.FastembedEmbedding.embed
+
+
+class _Response:
+ def __init__(self, payload):
+ self._payload = json.dumps(payload).encode()
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *_args):
+ pass
+
+ def read(self):
+ return self._payload
+
+
+def test_remote_qwen_backend_batches_and_prefixes_only_queries(monkeypatch):
+ requests = []
+
+ def open_request(request, timeout):
+ requests.append((request, timeout))
+ body = json.loads(request.data)
+ return _Response({
+ "data": [
+ {"index": index, "embedding": [float(index), 1.0]}
+ for index, _ in reversed(list(enumerate(body["input"])))
+ ]
+ })
+
+ monkeypatch.setattr(E, "urlopen", open_request)
+ backend = E.RemoteQwenEmbedding(
+ "Qwen/Qwen3-Embedding-8B-GGUF", "http://embed-gpu:8080/v1", timeout=7)
+
+ documents = backend.embed(["first", "second"])
+ queries = backend.embed_query(["route this"])
+
+ assert [vector.tolist() for vector in documents] == [[0.0, 1.0], [1.0, 1.0]]
+ assert queries[0].tolist() == [0.0, 1.0]
+ assert requests[0][1] == 7
+ assert requests[0][0].full_url == "http://embed-gpu:8080/v1/embeddings"
+ assert json.loads(requests[0][0].data)["input"] == ["first", "second"]
+ assert json.loads(requests[1][0].data)["input"] == [E.QUERY_PREFIX + "route this"]
+ assert backend.identity == "remote:Qwen/Qwen3-Embedding-8B-GGUF@http://embed-gpu:8080/v1"
+
+
+def test_remote_backend_reports_unavailable_server(monkeypatch):
+ monkeypatch.setattr(E, "urlopen", lambda *_args, **_kwargs: (_ for _ in ()).throw(
+ URLError("connection refused")))
+ backend = E.RemoteQwenEmbedding("model", "http://embed-gpu:8080/v1")
+
+ with np.testing.assert_raises_regex(RuntimeError, "embedding server unavailable"):
+ backend.embed(["hello"])
+
+
+def test_build_embedding_requires_remote_url(monkeypatch):
+ monkeypatch.setenv("EMBED_BACKEND", "remote")
+ monkeypatch.delenv("EMBED_BASE_URL", raising=False)
+ with np.testing.assert_raises_regex(ValueError, "EMBED_BASE_URL"):
+ E.build_embedding()
+
+
+def test_build_embedding_requires_distinct_remote_model(monkeypatch):
+ monkeypatch.setenv("EMBED_BACKEND", "remote")
+ monkeypatch.setenv("EMBED_BASE_URL", "http://embed-gpu:8080/v1")
+ monkeypatch.delenv("EMBED_REMOTE_MODEL", raising=False)
+ with np.testing.assert_raises_regex(ValueError, "EMBED_REMOTE_MODEL"):
+ E.build_embedding()
diff --git a/tests/test_env_check.py b/tests/test_env_check.py
index b4e0f90..8765a26 100644
--- a/tests/test_env_check.py
+++ b/tests/test_env_check.py
@@ -6,7 +6,7 @@
import pytest
-from optimize import require_openrouter_key
+from ingot.optimize import require_openrouter_key
ROOT = Path(__file__).resolve().parent.parent
@@ -35,7 +35,7 @@ def test_set_key_passes(monkeypatch):
def test_fully_local_setup_needs_no_key(monkeypatch):
- from optimize import openrouter_key_missing, require_openrouter_key
+ from ingot.optimize import openrouter_key_missing, require_openrouter_key
monkeypatch.delenv("OPENROUTER_API_KEY", raising=False)
monkeypatch.setenv("OPENROUTER_BASE_URL", "http://localhost:11434/v1") # e.g. Ollama
monkeypatch.delenv("MODEL_BASE_URL", raising=False)
@@ -45,7 +45,7 @@ def test_fully_local_setup_needs_no_key(monkeypatch):
def test_local_model_but_openrouter_teacher_still_needs_key(monkeypatch):
import pytest
- from optimize import require_openrouter_key
+ from ingot.optimize import require_openrouter_key
monkeypatch.delenv("OPENROUTER_API_KEY", raising=False)
monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False) # teacher on OpenRouter
monkeypatch.setenv("MODEL_BASE_URL", "http://localhost:8000/v1") # agent on local vLLM
@@ -54,7 +54,7 @@ def test_local_model_but_openrouter_teacher_still_needs_key(monkeypatch):
def test_client_kwargs_openrouter_gets_zdr_local_does_not(monkeypatch):
- from optimize import ZDR_PROVIDER, client_kwargs
+ from ingot.optimize import ZDR_PROVIDER, client_kwargs
monkeypatch.setenv("OPENROUTER_API_KEY", "sk-or-test")
kw = client_kwargs("https://openrouter.ai/api/v1")
assert kw == {"base_url": "https://openrouter.ai/api/v1", "api_key": "sk-or-test",
@@ -66,7 +66,7 @@ def test_client_kwargs_openrouter_gets_zdr_local_does_not(monkeypatch):
def test_model_base_url_overrides_only_the_serving_role(monkeypatch):
- from optimize import model_base_url, teacher_base_url
+ from ingot.optimize import model_base_url, teacher_base_url
monkeypatch.delenv("MODEL_BASE_URL", raising=False)
monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False)
assert model_base_url() == teacher_base_url() == "https://openrouter.ai/api/v1"
@@ -76,7 +76,7 @@ def test_model_base_url_overrides_only_the_serving_role(monkeypatch):
def test_provider_priority_composes_with_zdr(monkeypatch):
- from optimize import ZDR_PROVIDER, client_kwargs, openrouter_extra_body
+ from ingot.optimize import ZDR_PROVIDER, client_kwargs, openrouter_extra_body
monkeypatch.delenv("OPENROUTER_PROVIDERS", raising=False)
assert openrouter_extra_body() == ZDR_PROVIDER # default: ZDR only, no pin
monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq, deepinfra")
@@ -106,7 +106,7 @@ def __exit__(self, *a):
def test_provider_conflict_loose_name_matching(monkeypatch):
import urllib.request
- from optimize import provider_conflict
+ from ingot.optimize import provider_conflict
monkeypatch.setattr(urllib.request, "urlopen",
lambda url, timeout=10: _FakeEndpoints(["DeepInfra", "Io Net"]))
assert provider_conflict("qwen/x", ["deep-infra"]) is None # display-name vs slug
@@ -118,7 +118,7 @@ def test_provider_conflict_loose_name_matching(monkeypatch):
def test_provider_conflict_unknown_model_and_network_failure(monkeypatch):
import urllib.request
- from optimize import provider_conflict
+ from ingot.optimize import provider_conflict
monkeypatch.setattr(urllib.request, "urlopen", lambda url, timeout=10: _FakeEndpoints([]))
assert "no endpoints on OpenRouter" in provider_conflict("qwen/typo-27b", ["groq"])
def boom(url, timeout=10):
@@ -129,7 +129,7 @@ def boom(url, timeout=10):
def test_preflight_no_pins_makes_no_network_calls(monkeypatch):
import urllib.request
- from optimize import preflight_provider_pins
+ from ingot.optimize import preflight_provider_pins
monkeypatch.delenv("OPENROUTER_PROVIDERS", raising=False)
def forbidden(url, timeout=10):
raise AssertionError("network call without pins")
@@ -138,7 +138,7 @@ def forbidden(url, timeout=10):
def test_preflight_warns_on_every_uncovered_role(monkeypatch):
- import optimize
+ import ingot.optimize as optimize
monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq")
monkeypatch.delenv("MODEL_BASE_URL", raising=False)
monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False)
@@ -150,7 +150,7 @@ def test_preflight_warns_on_every_uncovered_role(monkeypatch):
def test_preflight_reports_agent_model_alias_value(monkeypatch):
# the pin check must validate the model the agent will actually use, whichever alias set it
- import optimize
+ import ingot.optimize as optimize
monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq")
for var in ("MODEL_BASE_URL", "BASE_URL", "OPENROUTER_BASE_URL", "MODEL"):
monkeypatch.delenv(var, raising=False)
@@ -162,7 +162,7 @@ def test_preflight_reports_agent_model_alias_value(monkeypatch):
def test_agent_model_resolution(monkeypatch):
# AGENT_MODEL wins; MODEL is the legacy alias; then the literal default
- from optimize import agent_model
+ from ingot.optimize import agent_model
monkeypatch.delenv("AGENT_MODEL", raising=False)
monkeypatch.delenv("MODEL", raising=False)
assert agent_model() == "qwen/qwen3-32b"
@@ -173,7 +173,7 @@ def test_agent_model_resolution(monkeypatch):
def test_skillopt_model_resolution(monkeypatch):
- from optimize import skillopt_model
+ from ingot.optimize import skillopt_model
monkeypatch.delenv("SKILLOPT_MODEL", raising=False)
assert skillopt_model() == "z-ai/glm-5.2"
monkeypatch.setenv("SKILLOPT_MODEL", "author/model")
@@ -186,9 +186,9 @@ def test_skillopt_model_reaches_every_authoring_role():
env = {**os.environ, "SKILLOPT_MODEL": "author/model",
"JUDGE_MODEL": "different/judge"}
code = """
-import optimize.draft as draft
-import optimize.rollout as rollout
-from optimize.usage import _role_models
+import ingot.optimize.draft as draft
+import ingot.optimize.rollout as rollout
+from ingot.optimize.usage import _role_models
assert draft.MODEL == 'author/model'
assert rollout.SKILLOPT_MODEL == 'author/model'
assert _role_models()['reflection'] == 'author/model'
@@ -201,14 +201,37 @@ def test_judge_warns_when_skillopt_model_is_the_grader():
env = {**os.environ, "SKILLOPT_MODEL": "same/model", "JUDGE_MODEL": "same/model"}
env.pop("JUDGE_MODELS", None)
- result = subprocess.run([sys.executable, "-c", "import optimize.judge"], env=env,
+ result = subprocess.run([sys.executable, "-c", "import ingot.optimize.judge"], env=env,
cwd=ROOT, text=True, capture_output=True, check=True)
assert "author == grader" in result.stdout
+def test_duplicate_judge_models_fail_closed_before_spend():
+ env = {**os.environ, "JUDGE_MODELS": "alpha/model,alpha/model"}
+ result = subprocess.run([sys.executable, "-c", "import ingot.optimize.judge"], env=env,
+ cwd=ROOT, text=True, capture_output=True)
+ assert result.returncode != 0
+ assert "JUDGE_MODELS contains duplicate model 'alpha/model'" in result.stderr
+
+
+def test_blank_judge_ensemble_falls_back_to_single_judge():
+ env = {**os.environ, "JUDGE_MODELS": " , ", "JUDGE_MODEL": "backup/model"}
+ result = subprocess.run(
+ [sys.executable, "-c", "import ingot.optimize.judge as j; print(j.MODELS)"], env=env,
+ cwd=ROOT, text=True, capture_output=True, check=True)
+ assert result.stdout.strip() == "['backup/model']"
+
+
+def test_duplicate_compat_models_fail_closed_before_spend(monkeypatch):
+ from ingot.optimize.compat import compat_models
+ monkeypatch.setenv("COMPAT_MODELS", "alpha/model,alpha/model")
+ with pytest.raises(SystemExit, match="COMPAT_MODELS contains duplicate model 'alpha/model'"):
+ compat_models()
+
+
def test_preflight_checks_strong_model_only_when_explicitly_set(monkeypatch):
- import optimize
+ import ingot.optimize as optimize
monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq")
monkeypatch.delenv("MODEL_BASE_URL", raising=False)
monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False)
@@ -222,7 +245,7 @@ def test_preflight_checks_strong_model_only_when_explicitly_set(monkeypatch):
def test_preflight_skips_roles_on_local_endpoints(monkeypatch):
- import optimize
+ import ingot.optimize as optimize
monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq")
monkeypatch.setenv("OPENROUTER_BASE_URL", "http://localhost:11434/v1") # fully local
monkeypatch.setattr(optimize, "provider_conflict",
@@ -233,7 +256,7 @@ def test_preflight_skips_roles_on_local_endpoints(monkeypatch):
def test_invoke_retry_fails_fast_on_permanent_config_errors(monkeypatch):
import pytest
- from optimize import judge as judge_mod
+ from ingot.optimize import judge as judge_mod
monkeypatch.setattr(judge_mod.time, "sleep", lambda s: (_ for _ in ()).throw(AssertionError("slept")))
class Doomed:
@@ -252,7 +275,7 @@ def invoke(self, messages):
def test_invoke_retry_still_retries_transient_errors(monkeypatch):
- from optimize import judge as judge_mod
+ from ingot.optimize import judge as judge_mod
monkeypatch.setattr(judge_mod.time, "sleep", lambda s: None)
class Flaky:
@@ -267,7 +290,7 @@ def invoke(self, messages):
def test_generic_base_url_and_api_key_win_with_legacy_fallback(monkeypatch):
- from optimize import api_key, model_api_key, teacher_base_url
+ from ingot.optimize import api_key, model_api_key, teacher_base_url
monkeypatch.delenv("BASE_URL", raising=False)
monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False)
monkeypatch.setenv("OPENROUTER_API_KEY", "sk-or-legacy")
@@ -284,7 +307,7 @@ def test_generic_base_url_and_api_key_win_with_legacy_fallback(monkeypatch):
def test_hosted_https_endpoint_requires_a_key(monkeypatch):
- from optimize import openrouter_key_missing
+ from ingot.optimize import openrouter_key_missing
for var in ("API_KEY", "OPENROUTER_API_KEY", "MODEL_API_KEY", "MODEL_BASE_URL"):
monkeypatch.delenv(var, raising=False)
monkeypatch.setenv("BASE_URL", "https://openrouter.ai/api/v1")
@@ -294,7 +317,7 @@ def test_hosted_https_endpoint_requires_a_key(monkeypatch):
def test_local_endpoint_gets_clean_openai_request(monkeypatch):
- from optimize import client_kwargs
+ from ingot.optimize import client_kwargs
kw = client_kwargs("http://ollama:11434/v1")
assert kw == {"base_url": "http://ollama:11434/v1", "api_key": "local",
"extra_body": {}} # no OpenRouter provider prefs
diff --git a/tests/test_evidence.py b/tests/test_evidence.py
index 37965f8..368c085 100644
--- a/tests/test_evidence.py
+++ b/tests/test_evidence.py
@@ -1,6 +1,6 @@
import json
-from optimize.evidence import build_evidence, first_divergence, render_markdown, write_evidence
+from ingot.optimize.evidence import build_evidence, first_divergence, render_markdown, write_evidence
SUMMARY = {
@@ -81,7 +81,7 @@ def test_markdown_surfaces_gate_warnings():
def _routing_evidence(gate=None, parity=True):
- from optimize.evidence import RoutingRun, build_routing_evidence
+ from ingot.optimize.evidence import RoutingRun, build_routing_evidence
metrics = dict(ROUTING_METRICS)
if not parity:
metrics["parity"] = {"rate": 0.0, "total": 0}
@@ -93,7 +93,7 @@ def _routing_evidence(gate=None, parity=True):
def test_routing_evidence_carries_revisions_and_router_metrics():
- from optimize.evidence import ROUTING_SCHEMA
+ from ingot.optimize.evidence import ROUTING_SCHEMA
evidence = _routing_evidence()
assert evidence["schema_version"] == ROUTING_SCHEMA
assert evidence["champion"]["revision"] == "champ-rev"
@@ -103,7 +103,7 @@ def test_routing_evidence_carries_revisions_and_router_metrics():
def test_write_evidence_renders_the_routing_report(tmp_path):
- from optimize.evidence import render_routing_markdown
+ from ingot.optimize.evidence import render_routing_markdown
evidence = _routing_evidence()
json_path, md_path = write_evidence(evidence, tmp_path)
assert json.loads(json_path.read_text()) == evidence
@@ -117,7 +117,7 @@ def test_write_evidence_renders_the_routing_report(tmp_path):
def test_routing_report_marks_a_blocked_gate_and_unexercised_parity():
- from optimize.evidence import render_routing_markdown
+ from ingot.optimize.evidence import render_routing_markdown
blocked = {"promotable": False, "blocked": ["routing top1 regressed"], "warnings": []}
text = render_routing_markdown(_routing_evidence(gate=blocked, parity=False))
assert "BLOCKED" in text
@@ -125,11 +125,27 @@ def test_routing_report_marks_a_blocked_gate_and_unexercised_parity():
assert "Cross-harness parity: not exercised" in text
-def test_recorded_path_is_repo_relative_and_leaves_outside_paths_alone(tmp_path):
+def test_recorded_path_is_state_relative_and_leaves_outside_paths_alone(tmp_path, monkeypatch):
+ """A bundle written inside a container has to be resolvable from the host, so the location is
+ recorded relative to the state root rather than absolutely."""
from pathlib import Path
- from optimize.evidence import _REPO_ROOT, recorded_path
- inside = _REPO_ROOT / "runs" / "evidence" / "pdf" / "1" / "EVIDENCE.md"
+ from ingot import paths
+ from ingot.optimize.evidence import recorded_path
+ monkeypatch.setenv("INGOT_HOME", str(tmp_path))
+ inside = paths.runs() / "evidence" / "pdf" / "1" / "EVIDENCE.md"
assert recorded_path(inside) == "runs/evidence/pdf/1/EVIDENCE.md"
outside = Path("/somewhere/else/EVIDENCE.md")
assert recorded_path(outside) == "/somewhere/else/EVIDENCE.md"
+
+
+def test_a_bundle_recorded_before_state_moved_out_of_the_package_still_reads_as_relative(
+ tmp_path, monkeypatch):
+ """Records written when `runs/` lived beside the code carry `runs/evidence/...`. Falling back
+ to an absolute path from whoever's machine wrote it would make them unresolvable."""
+ from ingot.optimize.evidence import _PACKAGE_ROOT, recorded_path
+ monkeypatch.setenv("INGOT_HOME", str(tmp_path))
+
+ legacy = _PACKAGE_ROOT / "runs" / "evidence" / "pdf" / "1" / "EVIDENCE.md"
+
+ assert recorded_path(legacy) == "runs/evidence/pdf/1/EVIDENCE.md"
diff --git a/tests/test_execcheck.py b/tests/test_execcheck.py
index e0aa13b..fbdb9f3 100644
--- a/tests/test_execcheck.py
+++ b/tests/test_execcheck.py
@@ -1,9 +1,9 @@
-"""Unit tests for execution-based code validation (optimize.execcheck), static path (no EXEC_SANDBOX)."""
+"""Unit tests for execution-based code validation (ingot.optimize.execcheck), static path (no EXEC_SANDBOX)."""
import os
import subprocess
import sys
-from optimize import execcheck as E
+from ingot.optimize import execcheck as E
def test_expects_code_gates_on_task_shape():
@@ -79,7 +79,7 @@ def test_exec_sandbox_env_modes():
# fresh-import subprocesses exercise the real env surface: the sandbox is the DEFAULT,
# "1" is the legacy bare opt-in, anything else turns execution off entirely
base = {k: v for k, v in os.environ.items() if k != "EXEC_SANDBOX"}
- probe = "from optimize.execcheck import EXEC_MODE, EXEC_SANDBOX, check; "
+ probe = "from ingot.optimize.execcheck import EXEC_MODE, EXEC_SANDBOX, check; "
default = subprocess.run([sys.executable, "-c", probe + "print(EXEC_MODE)"],
capture_output=True, text=True, env=base)
assert default.stdout.strip() == "docker"
@@ -209,12 +209,13 @@ def test_judge_note_execution_verdicts(monkeypatch):
def test_judge_threads_check_spec_into_the_prompt(monkeypatch):
monkeypatch.setattr(E, "EXEC_MODE", "1")
monkeypatch.setattr(E, "EXEC_SANDBOX", True) # legacy bare path
- from optimize import judge as judge_mod
+ from ingot.optimize import judge as judge_mod
seen = {}
monkeypatch.setattr(judge_mod, "MODELS", ["m"])
- def capture(model, prompt):
+ def capture(model, prompt, checklist):
seen["prompt"] = prompt
- return {"score": 1.0, "feedback": "f", "dimensions": {d: "pass" for d in judge_mod.DIMENSIONS}}
+ return {"items": {i["id"]: {"value": 1.0, "note": ""} for i in checklist},
+ "feedback": "f", "unparseable": False}
monkeypatch.setattr(judge_mod, "_judge_one", capture)
judge_mod.judge("t", "r", ANSWER_OK, check=CHECK)
assert "EXECUTION CHECK, PASSED" in seen["prompt"]
@@ -250,7 +251,7 @@ def test_docker_sandbox_container_is_actually_locked_down(monkeypatch):
assert " ".join(flag) in joined, f"missing {flag} in {cmd}"
assert "--pids-limit" in cmd and "--memory" in cmd
assert "-v" not in cmd and "--volume" not in cmd # nothing mounted in
- assert cmd[-4:] == [E.SANDBOX_IMAGE, "python", "-m", "optimize.sandbox_driver"]
+ assert cmd[-4:] == [E.SANDBOX_IMAGE, "python", "-m", "ingot.optimize.sandbox_driver"]
def test_docker_sandbox_runtime_flag_enables_gvisor(monkeypatch):
@@ -315,12 +316,12 @@ def test_sandbox_driver_end_to_end_without_docker(tmp_path):
import json as _json
spec = {"fixture": CHECK["fixture"], "code": 'text = open("input.txt").read()\n'
'open("output.txt", "w").write(text.upper())', "assertion": CHECK["assert"]}
- run = subprocess.run([sys.executable, "-m", "optimize.sandbox_driver"],
+ run = subprocess.run([sys.executable, "-m", "ingot.optimize.sandbox_driver"],
input=_json.dumps(spec), capture_output=True, text=True,
cwd=str(tmp_path), env={**os.environ, "PYTHONPATH": os.getcwd()})
assert _json.loads(run.stdout) == {"ok": True}
bad = {**spec, "code": spec["code"].replace(".upper()", ".lower()")}
- run = subprocess.run([sys.executable, "-m", "optimize.sandbox_driver"],
+ run = subprocess.run([sys.executable, "-m", "ingot.optimize.sandbox_driver"],
input=_json.dumps(bad), capture_output=True, text=True,
cwd=str(tmp_path), env={**os.environ, "PYTHONPATH": os.getcwd()})
verdict = _json.loads(run.stdout)
diff --git a/tests/test_harbor_catalog.py b/tests/test_harbor_catalog.py
new file mode 100644
index 0000000..f1b4725
--- /dev/null
+++ b/tests/test_harbor_catalog.py
@@ -0,0 +1,361 @@
+from __future__ import annotations
+
+import json
+from pathlib import Path
+
+import pytest
+
+import ingot.optimize.harbor_catalog as HC
+from ingot.optimize.harbor_catalog import CatalogIntent, enqueue_catalog, run_catalog
+
+_REAL_PREPARE_EXECUTION = HC._prepare_execution
+_REAL_INTENT_FOR_SKILL = HC._intent_for_skill
+
+
+def _intent(skill: str = "demo", sha: str = "a" * 64, *, priority: int = 100,
+ publish_root: str = "/srv/ingot/runs/harbor") -> CatalogIntent:
+ return CatalogIntent(
+ skill=skill,
+ skill_sha256=sha,
+ task_fingerprint="b" * 64,
+ target_specs=("dell-qwen=http://127.0.0.1:8011",),
+ target_fingerprints=("434045372e7c",),
+ harnesses=("aider", "pi"),
+ runtime_revisions=(("harbor", "0.20.0"), ("runner", "native-v1")),
+ publish_root=publish_root,
+ priority=priority,
+ )
+
+
+def _read(path: Path) -> dict:
+ return json.loads(path.read_text())
+
+
+@pytest.fixture(autouse=True)
+def _catalog_runtime(tmp_path, monkeypatch):
+ monkeypatch.setattr(HC, "CATALOG_OWNER", tmp_path / "global-controller.lock")
+ monkeypatch.setattr(HC, "_prepare_execution",
+ lambda root, intent: (root / "source", root / "runs" / intent.digest))
+ monkeypatch.setattr(HC, "_intent_for_skill",
+ lambda skill, targets, harnesses, priority=100, **_kwargs:
+ _intent(skill, priority=priority))
+ monkeypatch.setattr(HC, "_refuse_live_harbor", lambda _root: None)
+
+
+def test_enqueue_uses_content_identity_and_leaves_compatible_completion_untouched(tmp_path):
+ intent = _intent()
+ [path] = enqueue_catalog(tmp_path, [intent])
+ assert path.name == f"{intent.digest}.json"
+ assert _read(path)["identity"] == intent.identity_payload()
+ state = tmp_path / "state" / path.name
+ state.write_text(json.dumps({"schema": 1, "status": "complete", "priority": 100,
+ "intent_digest": intent.digest, "marker": "keep"}))
+
+ [again] = enqueue_catalog(tmp_path, [intent])
+
+ assert again == path
+ assert _read(state)["marker"] == "keep"
+
+
+def test_enqueue_prioritizes_changed_then_incomplete_current_revision(tmp_path):
+ old = _intent(sha="a" * 64)
+ [old_path] = enqueue_catalog(tmp_path, [old])
+ old_state = tmp_path / "state" / old_path.name
+ old_state.write_text(json.dumps({"schema": 1, "status": "complete", "priority": 100,
+ "intent_digest": old.digest}))
+
+ changed = _intent(sha="c" * 64)
+ [changed_path] = enqueue_catalog(tmp_path, [changed])
+ assert _read(tmp_path / "state" / changed_path.name)["priority"] == 300
+
+ current = _intent(skill="other")
+ [current_path] = enqueue_catalog(tmp_path, [current])
+ current_state = tmp_path / "state" / current_path.name
+ state = _read(current_state)
+ state["status"] = "running"
+ current_state.write_text(json.dumps(state))
+ enqueue_catalog(tmp_path, [current])
+ assert _read(current_state)["priority"] == 200
+
+
+def test_enqueue_reactivates_a_superseded_identity_that_becomes_current_again(tmp_path):
+ intent = _intent()
+ [intent_path] = enqueue_catalog(tmp_path, [intent])
+ state_path = tmp_path / "state" / intent_path.name
+ state = _read(state_path)
+ state.update(status="superseded", error="IdentityChanged", finished_at=123.0)
+ state_path.write_text(json.dumps(state))
+
+ enqueue_catalog(tmp_path, [intent])
+
+ assert _read(state_path) == {
+ "schema": 1,
+ "intent_digest": intent.digest,
+ "status": "pending",
+ "priority": 200,
+ }
+
+
+def test_stop_file_prevents_runner_and_preserves_pending(tmp_path):
+ [path] = enqueue_catalog(tmp_path, [_intent()])
+ stop = tmp_path / "STOP"
+ stop.touch()
+
+ monkeypatch = pytest.MonkeyPatch()
+ monkeypatch.setattr(HC, "run_local_sweep",
+ lambda *_args, **_kwargs: pytest.fail("runner called"))
+ run_catalog(tmp_path, stop_file=stop)
+ monkeypatch.undo()
+
+ assert _read(tmp_path / "state" / path.name)["status"] == "pending"
+
+
+def test_catalog_resumes_running_intent_and_advances_without_recreating_completion(tmp_path,
+ monkeypatch):
+ first, second = _intent("first", priority=200), _intent("second", priority=100)
+ first_path, second_path = enqueue_catalog(tmp_path, [first, second])
+ first_state = tmp_path / "state" / first_path.name
+ state = _read(first_state)
+ state.update(status="running", run_root="same-root")
+ first_state.write_text(json.dumps(state))
+ calls = []
+
+ def runner(skill, targets, **kwargs):
+ calls.append((skill, tuple(target.alias for target in targets), kwargs["native_parallel"],
+ kwargs["evidence_root"], kwargs["publish_root"],
+ kwargs["content_addressed_resume"]))
+ return {"skill": skill, "aborted": False, "combinations": {"one": {"lift": 0.1}}}
+
+ monkeypatch.setattr(HC, "run_local_sweep", runner)
+ run_catalog(tmp_path, max_skills=2)
+ assert [item[0] for item in calls] == ["first", "second"]
+ assert all(item[2] is True for item in calls)
+ assert _read(first_state)["run_root"].endswith(first.digest)
+ assert calls[0][3] == Path(_read(first_state)["run_root"])
+ assert calls[0][4] == Path("/srv/ingot/runs/harbor")
+ assert calls[0][5] is True
+ assert _read(first_state)["status"] == "complete"
+
+ run_catalog(tmp_path, max_skills=2)
+ assert len(calls) == 2
+ assert _read(tmp_path / "state" / second_path.name)["status"] == "complete"
+
+
+def test_catalog_forwards_native_process_environment_to_sweep(tmp_path, monkeypatch):
+ enqueue_catalog(tmp_path, [_intent()])
+ captured = []
+
+ def runner(_skill, _targets, **kwargs):
+ captured.append(kwargs["process_env"])
+ return {"aborted": False}
+
+ monkeypatch.setattr(HC, "run_local_sweep", runner)
+ process_env = {"PATH": "/bin", "HARBOR_EXTRA_DOCKER_COMPOSE": "/tmp/network.yml"}
+
+ run_catalog(tmp_path, max_skills=1, process_env=process_env)
+
+ assert captured == [process_env]
+
+
+def test_live_controller_refuses_before_runner(tmp_path, monkeypatch):
+ enqueue_catalog(tmp_path, [_intent()])
+ owner = HC._claim_controller(HC.CATALOG_OWNER)
+ monkeypatch.setattr(HC, "run_local_sweep",
+ lambda *_args, **_kwargs: pytest.fail("runner called"))
+
+ try:
+ with pytest.raises(RuntimeError, match="live controller"):
+ run_catalog(tmp_path)
+ finally:
+ owner.close()
+
+
+def test_stale_controller_receipt_is_reused_safely(tmp_path, monkeypatch):
+ enqueue_catalog(tmp_path, [_intent()])
+ HC.CATALOG_OWNER.write_text(json.dumps({"pid": 123, "start_token": "stale"}))
+ monkeypatch.setattr(HC, "run_local_sweep", lambda *_args, **_kwargs: {"aborted": False})
+
+ run_catalog(tmp_path, max_skills=1)
+ assert _read(next((tmp_path / "state").glob("*.json")))["status"] == "complete"
+
+
+def test_failed_skill_is_recorded_and_does_not_abort_sibling(tmp_path, monkeypatch):
+ enqueue_catalog(tmp_path, [_intent("bad", priority=200), _intent("good", priority=100)])
+
+ def runner(skill, *_args, **_kwargs):
+ if skill == "bad":
+ raise RuntimeError("boom")
+ return {"aborted": False}
+
+ monkeypatch.setattr(HC, "run_local_sweep", runner)
+ run_catalog(tmp_path, max_skills=2)
+ states = [_read(path) for path in (tmp_path / "state").glob("*.json")]
+ assert {state["status"] for state in states} == {"complete", "failed"}
+ assert next(state for state in states if state["status"] == "failed")["error"] == "RuntimeError"
+
+ monkeypatch.setattr(HC, "run_local_sweep", lambda *_args, **_kwargs: {"aborted": False})
+ run_catalog(tmp_path, max_skills=1)
+ assert {state["status"] for state in
+ (_read(path) for path in (tmp_path / "state").glob("*.json"))} == {"complete"}
+
+
+def test_identity_change_enqueues_current_and_runs_no_model(tmp_path, monkeypatch):
+ old = _intent(sha="a" * 64)
+ enqueue_catalog(tmp_path, [old])
+ current = _intent(sha="c" * 64)
+ monkeypatch.setattr(HC, "_intent_for_skill", lambda *_args, **_kwargs: current)
+ monkeypatch.setattr(HC, "run_local_sweep",
+ lambda *_args, **_kwargs: pytest.fail("runner called"))
+
+ run_catalog(tmp_path, max_skills=1)
+
+ states = [_read(path) for path in (tmp_path / "state").glob("*.json")]
+ assert {state["status"] for state in states} == {"superseded", "pending"}
+ assert next(state for state in states if state["status"] == "pending")["priority"] == 300
+
+
+def test_tampered_intent_or_state_digest_refuses_before_runner(tmp_path, monkeypatch):
+ [intent_path] = enqueue_catalog(tmp_path, [_intent()])
+ document = _read(intent_path)
+ document["identity"]["skill"] = "tampered"
+ intent_path.write_text(json.dumps(document))
+ monkeypatch.setattr(HC, "run_local_sweep",
+ lambda *_args, **_kwargs: pytest.fail("runner called"))
+ with pytest.raises(RuntimeError, match="intent digest"):
+ run_catalog(tmp_path)
+
+ intent_path.write_text(json.dumps(HC._intent_document(_intent())))
+ state_path = tmp_path / "state" / intent_path.name
+ state = _read(state_path)
+ state["intent_digest"] = "0" * 64
+ state_path.write_text(json.dumps(state))
+ with pytest.raises(RuntimeError, match="state digest"):
+ run_catalog(tmp_path)
+
+
+def test_live_harbor_child_refuses_before_runner(tmp_path, monkeypatch):
+ intent = _intent()
+ enqueue_catalog(tmp_path, [intent])
+ monkeypatch.setattr(HC, "_refuse_live_harbor",
+ lambda _root: (_ for _ in ()).throw(RuntimeError("live Harbor child")))
+ monkeypatch.setattr(HC, "run_local_sweep",
+ lambda *_args, **_kwargs: pytest.fail("runner called"))
+
+ with pytest.raises(RuntimeError, match="live Harbor child"):
+ run_catalog(tmp_path)
+
+
+def test_low_utilization_is_reported_not_used_to_overlap_skills(tmp_path, monkeypatch):
+ enqueue_catalog(tmp_path, [_intent("first"), _intent("second")])
+ active = 0
+ peak = 0
+
+ def runner(*_args, **_kwargs):
+ nonlocal active, peak
+ active += 1
+ peak = max(peak, active)
+ active -= 1
+ return {"aborted": False, "utilization": 0.1}
+
+ monkeypatch.setattr(HC, "run_local_sweep", runner)
+ run_catalog(tmp_path, max_skills=2)
+ assert peak == 1
+ assert all("utilization" in _read(path) for path in (tmp_path / "state").glob("*.json"))
+
+
+def test_telemetry_pending_state_retries_on_next_controller_without_marking_complete(tmp_path,
+ monkeypatch):
+ enqueue_catalog(tmp_path, [_intent()])
+ calls = 0
+
+ def runner(*_args, **_kwargs):
+ nonlocal calls
+ calls += 1
+ return {"aborted": False, "telemetry_pending": calls == 1}
+
+ monkeypatch.setattr(HC, "run_local_sweep", runner)
+ run_catalog(tmp_path, max_skills=1)
+ state_path = next((tmp_path / "state").glob("*.json"))
+ assert _read(state_path)["status"] == "failed"
+ run_catalog(tmp_path, max_skills=1)
+ assert _read(state_path)["status"] == "complete"
+ assert calls == 2
+
+
+def test_all_skips_skills_without_tasks_and_enqueues_eligible_siblings(tmp_path, monkeypatch):
+ class Skill:
+ def __init__(self, name):
+ self.name = name
+
+ monkeypatch.setattr(HC, "load_skills", lambda: [Skill("missing"), Skill("eligible")])
+ monkeypatch.setattr(HC, "_intent_for_skill",
+ lambda skill, *_args, **_kwargs: None if skill == "missing" else _intent(skill))
+
+ assert HC.main(["--root", str(tmp_path), "--all", "--target",
+ "dell-qwen=http://127.0.0.1:8011", "--enqueue-only"]) == 0
+ documents = [_read(path) for path in (tmp_path / "intents").glob("*.json")]
+ assert [item["identity"]["skill"] for item in documents] == ["eligible"]
+
+
+def test_intent_route_revision_uses_discovered_target_context(tmp_path, monkeypatch):
+ monkeypatch.setattr(HC, "_intent_for_skill", _REAL_INTENT_FOR_SKILL)
+ skill = tmp_path / "skills" / "demo"
+ skill.mkdir(parents=True)
+ (skill / "SKILL.md").write_text("---\nname: demo\ndescription: x\n---\nbody")
+ monkeypatch.setattr(HC, "resolve_skill_dir", lambda _name: skill)
+ monkeypatch.setattr(HC, "_heldout", lambda _name: [{"task": "x"}] * 4)
+ live = HC.parse_target("dell-qwen=http://host:8011")
+ live = HC.LocalTarget(**{**live.__dict__, "context_length": 163840})
+ monkeypatch.setattr(HC, "discover_target", lambda *_args: live)
+
+ intent = HC._intent_for_skill("demo", ("dell-qwen=http://host:8011",),
+ ("claude-code",))
+
+ revisions = dict(intent.runtime_revisions)
+ assert any(key.startswith(f"route:{live.fingerprint}:claude-code")
+ and "20480" in value and "context=163840" in value
+ for key, value in revisions.items())
+
+
+def test_execution_stages_full_tree_atomically_and_rejects_partial_reuse(tmp_path, monkeypatch):
+ monkeypatch.setattr(HC, "_prepare_execution", _REAL_PREPARE_EXECUTION)
+ source = tmp_path / "library" / "demo"
+ source.mkdir(parents=True)
+ (source / "SKILL.md").write_text("---\nname: demo\ndescription: x\n---\nbody")
+ (source / "reference.md").write_text("evidence")
+ revision = HC.skill_revision(source)
+ intent = _intent()
+ intent = CatalogIntent(**{**intent.__dict__,
+ "runtime_revisions": (("skill-tree", revision),)})
+ monkeypatch.setattr(HC, "resolve_skill_dir", lambda _skill: source)
+
+ staged, execution = HC._prepare_execution(tmp_path, intent)
+ assert (staged / "demo" / "reference.md").read_text() == "evidence"
+ assert not list(execution.glob(".staged.*.tmp"))
+
+ (staged / "demo" / "reference.md").unlink()
+ with pytest.raises(RuntimeError, match="skill tree identity"):
+ HC._prepare_execution(tmp_path, intent)
+
+
+def test_execution_rejects_source_tree_change_during_copy(tmp_path, monkeypatch):
+ monkeypatch.setattr(HC, "_prepare_execution", _REAL_PREPARE_EXECUTION)
+ source = tmp_path / "library" / "demo"
+ source.mkdir(parents=True)
+ (source / "SKILL.md").write_text("---\nname: demo\ndescription: x\n---\nbody")
+ (source / "reference.md").write_text("before")
+ revision = HC.skill_revision(source)
+ intent = CatalogIntent(**{**_intent().__dict__,
+ "runtime_revisions": (("skill-tree", revision),)})
+ monkeypatch.setattr(HC, "resolve_skill_dir", lambda _skill: source)
+ original = HC.shutil.copytree
+
+ def changed_copy(source_path, destination):
+ result = original(source_path, destination)
+ (destination / "reference.md").write_text("after")
+ return result
+
+ monkeypatch.setattr(HC.shutil, "copytree", changed_copy)
+ with pytest.raises(RuntimeError, match="changed while staging"):
+ HC._prepare_execution(tmp_path, intent)
+ assert not (tmp_path / "runs" / intent.digest / "staged").exists()
diff --git a/tests/test_harbor_eval.py b/tests/test_harbor_eval.py
new file mode 100644
index 0000000..86e72e2
--- /dev/null
+++ b/tests/test_harbor_eval.py
@@ -0,0 +1,1538 @@
+"""Unit tests for the sandboxed cross-harness skill eval (no Docker, no network, no harness).
+
+The Harbor subprocess and the judge are stubbed; what is exercised here is the part that is ours:
+the generated dataset, the treatment/control arms, the trial-name parsing that reads Harbor's
+output layout, and the scoring of what a harness actually produced."""
+import hashlib
+import json
+import os
+import subprocess
+import threading
+import time
+from pathlib import Path
+
+import pytest
+
+from ingot.optimize import agy_judge as A
+from ingot.optimize import harbor_eval as H
+from ingot.optimize.harbor_targets import LocalTarget
+
+
+HOLDOUT = [{"task": "Write add(a, b).", "rubric": "defines add"},
+ {"task": "Write sub(a, b).", "rubric": "defines sub"}]
+
+
+@pytest.fixture(autouse=True)
+def _avoid_starting_a_gateway_process_in_harbor_unit_tests(monkeypatch):
+ """Gateway lifecycle has its own unit tests; these tests only assert orchestration order."""
+ class Session:
+ def __init__(self, *args, **kwargs):
+ pass
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *args):
+ return None
+
+ monkeypatch.setattr(H, "GatewaySession", Session)
+
+
+def test_dataset_has_one_task_per_holdout_task(tmp_path):
+ dataset = H.build_dataset("demo", HOLDOUT, tmp_path)
+ names = sorted(p.name for p in dataset.iterdir() if p.is_dir())
+ assert names == ["demo-h0", "demo-h1"]
+ for index, name in enumerate(names):
+ root = dataset / name
+ assert (root / "environment" / "Dockerfile").is_file()
+ assert (root / "tests" / "test.sh").is_file()
+ instruction = (root / "instruction.md").read_text()
+ assert HOLDOUT[index]["task"] in instruction
+ # Without this the agent answers in chat and the workspace stays empty, which the collector
+ # would then score as a zero for every task.
+ assert H.SOLUTION_DIR in instruction
+ assert 'name = "ingot/demo"' in (dataset / "dataset.toml").read_text()
+
+
+SEEDED = [{"task": "Fix the backup.", "rubric": "fixes it",
+ "files": {"backup.sh": "#!/bin/bash\necho hi\n", "tools/check.py": "print(1)\n"},
+ "verify": 'cd /tmp && python3 -c "print(\'ok\')"'}]
+
+
+def test_a_seeded_task_ships_its_working_tree_into_the_image(tmp_path):
+ """A process skill has nothing to act on in an empty container. Verifying in the execution
+ context, feeding a guard its reject input and wiring a real caller all need existing code."""
+ root = H.build_dataset("demo", SEEDED, tmp_path) / "demo-h0"
+ seed = root / "environment" / "seed"
+ assert (seed / "backup.sh").read_text() == "#!/bin/bash\necho hi\n"
+ assert (seed / "tools" / "check.py").read_text() == "print(1)\n"
+ dockerfile = (root / "environment" / "Dockerfile").read_text()
+ assert f"COPY seed/ {H.REPO_DIR}/" in dockerfile
+ # The agent is told where the tree is; without this it starts from a blank directory and the
+ # seeded defect is never seen.
+ assert H.REPO_DIR in (root / "instruction.md").read_text()
+
+
+def test_an_unseeded_task_keeps_the_blank_environment(tmp_path):
+ """Seeding is opt-in per task: a task with no files must build the image it always built."""
+ root = H.build_dataset("demo", HOLDOUT, tmp_path) / "demo-h0"
+ assert not (root / "environment" / "seed").exists()
+ assert "COPY seed/" not in (root / "environment" / "Dockerfile").read_text()
+ assert "_objective_check" not in (root / "tests" / "test.sh").read_text()
+
+
+def test_the_objective_check_runs_after_the_agent_and_survives_quoting(tmp_path):
+ """The agent's own evidence log is a claim about what it ran, and a skill that rewards writing
+ evidence logs is exactly what teaches it to produce one. This is the referent from outside."""
+ root = H.build_dataset("demo", SEEDED, tmp_path) / "demo-h0"
+ test_sh = (root / "tests" / "test.sh").read_text()
+ assert "_objective_check.txt" in test_sh
+ # The command embeds both quote kinds. Interpolating it raw produced unrunnable shell.
+ assert subprocess.run(["bash", "-n", str(root / "tests" / "test.sh")],
+ capture_output=True).returncode == 0
+ # It must stay captured evidence, never a second grader: the Ingot judge owns the score.
+ assert "echo 1 > /logs/verifier/reward.txt" in test_sh
+
+
+def test_a_seeded_path_cannot_escape_the_task_directory(tmp_path):
+ """Task files are authored data, but a traversing key would write into this checkout rather
+ than the container's."""
+ escaping = [{"task": "t", "rubric": "r", "files": {"../../pwned": "x"}}]
+ with pytest.raises(ValueError, match="escapes"):
+ H.build_dataset("demo", escaping, tmp_path)
+ assert not (tmp_path.parent / "pwned").exists()
+
+
+def test_rebuilding_a_dataset_drops_tasks_from_a_shorter_holdout(tmp_path):
+ """A holdout that shrinks must not leave last run's extra task behind to be run and scored."""
+ H.build_dataset("demo", HOLDOUT, tmp_path)
+ dataset = H.build_dataset("demo", HOLDOUT[:1], tmp_path)
+ assert sorted(p.name for p in dataset.iterdir() if p.is_dir()) == ["demo-h0"]
+
+
+def test_the_skill_arm_passes_a_skill_and_the_control_arm_does_not(tmp_path, monkeypatch):
+ """Lift is the difference between these two commands. If the control also carried --skill there
+ would be no control at all, and every lift would come out at zero."""
+ seen = []
+
+ class Done:
+ returncode, stderr, stdout = 0, "", ""
+
+ monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: seen.append(argv) or Done())
+ monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None) # covered by its own tests
+ H.run_arm(tmp_path / "ds", "claude-code", "/skills/demo", tmp_path / "jobs", "skill",
+ log=lambda *a: None)
+ H.run_arm(tmp_path / "ds", "claude-code", None, tmp_path / "jobs", "control",
+ log=lambda *a: None)
+
+ assert "--skill" in seen[0] and seen[0][seen[0].index("--skill") + 1] == "/skills/demo"
+ assert "--skill" not in seen[1]
+ assert "--path" in seen[0], "a local dataset must be passed with --path, not --dataset"
+
+
+def test_run_arm_forwards_agent_env_agent_kwargs_task_name_without_provider_leak(
+ tmp_path, monkeypatch
+):
+ """Local adapter settings cross the Harbor boundary without exposing the parent credentials."""
+ seen = []
+ logs = []
+
+ class Done:
+ returncode, stderr, stdout = 0, "", ""
+
+ def fake_run(argv, **kwargs):
+ seen.append((argv, kwargs))
+ return Done()
+
+ monkeypatch.setattr(H.subprocess, "run", fake_run)
+ monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None)
+ secret = "sk-parent-secret"
+ child_env = {"PATH": "/usr/bin", "OPENAI_API_KEY": secret,
+ "ANTHROPIC_API_KEY": "sk-anthropic-parent", "SAFE": "yes",
+ "SAFE_SECRET": secret,
+ "LANGFUSE_PUBLIC_KEY": "pk-parent",
+ "LANGFUSE_SECRET_KEY": "sk-parent",
+ "LANGFUSE_BASE_URL": "https://langfuse-parent.invalid",
+ "LANGFUSE_ENCRYPTION_KEY": "encryption-parent",
+ "LANGFUSE_INIT_USER_PASSWORD": "password-parent",
+ "LANGFUSE_PUBLIC_URL": "https://public-parent.invalid"}
+ H.run_arm(
+ tmp_path / "ds", "terminus-2", None, tmp_path / "jobs", "local",
+ model="deepseek-v4-flash", concurrency=3, attempts=2,
+ agent_env={"Z_KEY": "z-value", "A_KEY": "a-value", "OPENAI_API_KEY": "local",
+ "LANGFUSE_PUBLIC_KEY": "pk-agent",
+ "LANGFUSE_SECRET_KEY": "sk-agent",
+ "LANGFUSE_BASE_URL": "https://langfuse-agent.invalid",
+ "LANGFUSE_ENCRYPTION_KEY": "encryption-agent",
+ "LANGFUSE_INIT_USER_PASSWORD": "password-agent",
+ "LANGFUSE_PUBLIC_URL": "https://public-agent.invalid"},
+ agent_kwargs={"zeta": "z-value", "alpha": "a-value"},
+ task_name="demo-h0", process_env=child_env, log=logs.append,
+ )
+
+ argv, kwargs = seen[0]
+ assert [argv[i + 1] for i, value in enumerate(argv) if value == "--ae"] == [
+ "A_KEY=a-value", "OPENAI_API_KEY=local", "Z_KEY=z-value"
+ ]
+ assert [argv[i + 1] for i, value in enumerate(argv) if value == "--ak"] == [
+ "alpha=a-value", "zeta=z-value"
+ ]
+ assert argv[argv.index("--include-task-name") + 1] == "demo-h0"
+ assert kwargs["env"] is not child_env
+ assert kwargs["env"]["PATH"] == "/usr/bin" and kwargs["env"]["SAFE"] == "yes"
+ assert kwargs["env"]["OPENAI_API_KEY"] == "local"
+ assert "ANTHROPIC_API_KEY" not in kwargs["env"]
+ assert not any(key.startswith("LANGFUSE_") for key in kwargs["env"])
+ assert not any("LANGFUSE_" in value for value in argv)
+ assert kwargs["capture_output"] is True and kwargs["text"] is True
+ assert "OPENAI_API_KEY=local" in argv
+ assert secret not in " ".join(argv)
+ assert secret not in "\n".join(logs)
+ assert secret not in str(tmp_path / "jobs" / "local")
+
+
+def test_run_arm_serializes_nested_agent_kwargs_as_json(tmp_path, monkeypatch):
+ seen = []
+
+ class Done:
+ returncode, stderr, stdout = 0, "", ""
+
+ monkeypatch.setattr(H.subprocess, "run", lambda argv, **kwargs: seen.append(argv) or Done())
+ monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None)
+ H.run_arm(tmp_path / "ds", "opencode", None, tmp_path / "jobs", "local",
+ agent_kwargs={"opencode_config": {"provider": {"local": {"npm": "x"}}}})
+ value = seen[0][seen[0].index("--ak") + 1]
+ assert value == 'opencode_config={"provider":{"local":{"npm":"x"}}}'
+
+
+def test_run_arm_without_local_overrides_keeps_legacy_command_and_process_call(
+ tmp_path, monkeypatch
+):
+ """Existing proprietary callers keep the old command and inherited process environment."""
+ seen = []
+
+ class Done:
+ returncode, stderr, stdout = 0, "", ""
+
+ monkeypatch.setenv("LANGFUSE_ENCRYPTION_KEY", "parent-encryption")
+ monkeypatch.setenv("LANGFUSE_PUBLIC_URL", "https://public.invalid")
+ monkeypatch.setenv("OPENAI_API_KEY", "provider-key-must-remain")
+ monkeypatch.setattr(H.subprocess, "run", lambda argv, **kwargs: seen.append((argv, kwargs)) or Done())
+ monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None)
+ H.run_arm(tmp_path / "ds", "claude-code", None, tmp_path / "jobs", "control",
+ log=lambda *a: None)
+ argv, kwargs = seen[0]
+ assert argv == [
+ H.HARBOR_BIN, "run", "--path", str(tmp_path / "ds"), "--agent", "claude-code",
+ "--n-concurrent", "2", "--jobs-dir", str(tmp_path / "jobs"), "--job-name", "control",
+ "--environment-build-timeout-multiplier", str(H.BUILD_TIMEOUT_MULTIPLIER),
+ "--agent-setup-timeout-multiplier", str(H.SETUP_TIMEOUT_MULTIPLIER),
+ ]
+ assert kwargs["capture_output"] is True and kwargs["text"] is True
+ assert not any(key.startswith("LANGFUSE_") for key in kwargs["env"])
+ assert kwargs["env"]["OPENAI_API_KEY"] == "provider-key-must-remain"
+
+
+def test_a_failing_harbor_run_is_raised_not_silently_scored(tmp_path, monkeypatch):
+ class Done:
+ returncode, stderr, stdout = 1, "boom", "outer stdout"
+
+ monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: Done())
+ with pytest.raises(RuntimeError, match="boom"):
+ H.run_arm(tmp_path / "ds", "codex", None, tmp_path / "jobs", "control", log=lambda *a: None)
+ assert json.loads((tmp_path / "jobs" / "control" / "harbor-invocation.json").read_text()) == {
+ "returncode": 1, "stdout_bytes": 12, "stderr_bytes": 4,
+ "stdout_excerpt": "outer stdout", "stderr_excerpt": "boom",
+ }
+
+
+def test_run_arm_keeps_outer_harbor_exit_evidence_when_a_pending_job_is_refused(tmp_path, monkeypatch):
+ """A Harbor zero exit can still leave a pending job; preserve the only outer diagnostic."""
+ class Done:
+ returncode = 0
+ stdout = "Authorization: Bearer bearer-value http://target.invalid:8001/sk-path"
+ stderr = ('x-api-key: header-value Authorization: Basic basic-value '
+ 'sk-live-value pk_live_value known-secret-value '
+ '"OPENAI_API_KEY": "json-value" LITELLM_API_KEY: colon-value '
+ 'MODEL_API_KEY space-value internal.example:8001')
+
+ parent = {"OPENAI_API_KEY": "known-secret-value", "CUSTOM_API_KEY": "mapping-secret"}
+ monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: Done())
+
+ def refuse_pending(*args):
+ raise RuntimeError("canary wrote no completed trial")
+
+ monkeypatch.setattr(H, "_refuse_broken_job", refuse_pending)
+ job = tmp_path / "jobs" / "codex"
+ with pytest.raises(RuntimeError, match="canary wrote no completed trial"):
+ H.run_arm(tmp_path / "ds", "codex", None, tmp_path / "jobs", "codex",
+ process_env=parent, log=lambda *a: None)
+ receipt_path = job / "harbor-invocation.json"
+ receipt = json.loads(receipt_path.read_text())
+ assert receipt["returncode"] == 0
+ assert receipt["stdout_bytes"] == len(Done.stdout.encode())
+ assert receipt["stderr_bytes"] == len(Done.stderr.encode())
+ persisted = receipt_path.read_text()
+ for raw in ("target.invalid", "bearer-value", "header-value", "basic-value", "sk-live-value",
+ "pk_live_value", "known-secret-value", "mapping-secret", "json-value",
+ "colon-value", "space-value", "internal.example:8001"):
+ assert raw not in persisted
+ assert receipt["stdout_excerpt"] and receipt["stderr_excerpt"]
+
+
+def test_trial_name_drops_the_run_suffix_harbor_appends(tmp_path):
+ """Observed layout: jobs//__/verifier/solution. Keeping the suffix means no
+ trial ever matches its task and every score reads as a zero."""
+ solution = tmp_path / "demo-h1__abc123" / "verifier" / "solution"
+ solution.mkdir(parents=True)
+ assert H._trial_task_name(solution) == "demo-h1"
+
+
+def test_collect_reads_what_the_agent_left_in_the_solution_directory(tmp_path):
+ solution = tmp_path / "demo-h0__xy" / "verifier" / "solution"
+ (solution / "pkg").mkdir(parents=True)
+ (solution / "answer.py").write_text("def add(a, b): return a + b")
+ (solution / "pkg" / "notes.md").write_text("reasoning")
+ answers = H.collect_answers(tmp_path)
+ assert set(answers) == {"demo-h0"}
+ assert "def add" in answers["demo-h0"][0] and "reasoning" in answers["demo-h0"][0]
+
+
+def test_repeated_attempts_at_one_task_are_all_kept(tmp_path):
+ """Keying a single answer by task name silently kept only whichever trial was read last, so
+ --n-attempts above 1 paid for repeated measurements and then threw all but one away. Repetition
+ is the whole remedy for this eval's noise: re-judging a fixed answer three times returned an
+ identical score, while re-running the agent on the same task moved it by 0.278."""
+ for suffix, body in (("aa", "first attempt"), ("bb", "second attempt")):
+ solution = tmp_path / f"demo-h0__{suffix}" / "verifier" / "solution"
+ solution.mkdir(parents=True)
+ (solution / "answer.py").write_text(body)
+ answers = H.collect_answers(tmp_path)
+ assert len(answers["demo-h0"]) == 2
+ assert {"first attempt", "second attempt"} == {a.split("\n", 1)[1] for a in answers["demo-h0"]}
+
+
+def test_a_tasks_score_is_the_mean_over_its_attempts(monkeypatch):
+ scores = iter([1.0, 0.0])
+ monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": next(scores)})
+ got = H.score({"demo-h0": ["one", "two"]}, "demo", HOLDOUT[:1])
+ assert got == [0.5]
+
+
+def test_arm_scoring_runs_independent_grades_with_bounded_concurrency(monkeypatch):
+ active = 0
+ peak = 0
+ lock = threading.Lock()
+
+ def fake_judge(*_args, **_kwargs):
+ nonlocal active, peak
+ with lock:
+ active += 1
+ peak = max(peak, active)
+ time.sleep(0.02)
+ with lock:
+ active -= 1
+ return {"score": 0.5}
+
+ monkeypatch.setattr(H, "judge", fake_judge)
+ answers = {f"demo-h{index}": ["one", "two", "three"] for index in range(2)}
+
+ scores = H.score(answers, "demo", HOLDOUT, concurrency=4)
+
+ assert scores == [0.5, 0.5]
+ assert 1 < peak <= 4
+
+
+def test_a_harness_that_produced_nothing_scores_zero_rather_than_being_dropped(monkeypatch):
+ """Dropping the empty task would raise the arm's mean by removing its own failure. A harness
+ that ran and delivered nothing has a real score, and it is zero."""
+ monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.8})
+ scores = H.score({"demo-h0": ["some code"]}, "demo", HOLDOUT)
+ assert scores == [0.8, 0.0]
+
+
+def test_the_tasks_own_checklist_reaches_the_judge(monkeypatch):
+ """A task's checklist is what gives its score any resolution.
+
+ `judge()` grades on four generic items unless a task supplies its own, and a frontier model
+ passes all four on any easy task. Dropping the checklist here is what flattened the first
+ build-loop matrix: every control landed near 0.85 and no arm could separate from any other."""
+ checklist = [{"id": "declares_stakes_tier", "criterion": "names a tier", "weight": 3,
+ "dimension": "instruction_following"}]
+ holdout = [{"task": "Write add(a, b).", "rubric": "defines add", "checklist": checklist}]
+ seen = {}
+
+ def fake_judge(task, rubric, answer, **kwargs):
+ seen.update(kwargs)
+ return {"score": 1.0}
+
+ monkeypatch.setattr(H, "judge", fake_judge)
+ H.score({"demo-h0": ["def add(a, b): return a + b"]}, "demo", holdout)
+ assert seen["checklist"] == checklist
+
+
+def test_agy_failure_propagates_out_of_arm_scoring(monkeypatch):
+ monkeypatch.setenv("JUDGE_BACKEND", "agy")
+ monkeypatch.setattr(
+ A,
+ "invoke",
+ lambda *_args: (_ for _ in ()).throw(A.AgyJudgeError("agy stopped")),
+ )
+
+ with pytest.raises(A.AgyJudgeError, match="agy stopped"):
+ H.score({"demo-h0": ["answer"]}, "demo", HOLDOUT[:1])
+
+
+def test_matrix_records_a_broken_harness_without_discarding_the_others(tmp_path, monkeypatch):
+ """One missing harness must not throw away the arms already paid for, and its row must carry no
+ lift: a blank zero would read as 'measured, no effect'."""
+ monkeypatch.setattr(H, "load_tasks", lambda skill: ([], HOLDOUT, {}))
+ monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "source")
+ monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "staged")
+ monkeypatch.setattr(H.shutil, "which", lambda binary: "/usr/bin/harbor")
+ monkeypatch.setattr(H, "build_dataset", lambda *a, **k: tmp_path / "ds")
+ monkeypatch.setattr(H, "HARBOR_DIR", tmp_path / "out")
+ monkeypatch.setattr(H, "collect_answers", lambda job, skip=None: {})
+
+ def fake_run_arm(dataset, agent, source, jobs_dir, job_name, *a, **k):
+ if agent == "broken":
+ raise RuntimeError("no such agent")
+ return jobs_dir / job_name
+
+ monkeypatch.setattr(H, "run_arm", fake_run_arm)
+ monkeypatch.setattr(H, "broken_tasks", lambda job: set())
+ monkeypatch.setattr(H, "score", lambda answers, skill, holdout, skip=None: (
+ [1.0, 1.0] if "skill" in str(answers) else [0.5, 0.5]))
+
+ out = H.run_harbor_eval("demo", ["claude-code", "broken"], log=lambda *a: None)
+ assert "lift" in out["harnesses"]["claude-code"]
+ assert "no such agent" in out["harnesses"]["broken"]["error"]
+ assert "lift" not in out["harnesses"]["broken"]
+ assert json.loads((tmp_path / "out" / "demo.json").read_text())["skill"] == "demo"
+
+
+def test_a_run_where_no_harness_worked_fails_loudly(tmp_path, monkeypatch):
+ monkeypatch.setattr(H, "load_tasks", lambda skill: ([], HOLDOUT, {}))
+ monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "staged")
+ monkeypatch.setattr(H.shutil, "which", lambda binary: "/usr/bin/harbor")
+ monkeypatch.setattr(H, "build_dataset", lambda *a, **k: tmp_path / "ds")
+ monkeypatch.setattr(H, "HARBOR_DIR", tmp_path / "out")
+ monkeypatch.setattr(H, "run_arm", lambda *a, **k: (_ for _ in ()).throw(RuntimeError("down")))
+ with pytest.raises(SystemExit, match="nothing was measured"):
+ H.run_harbor_eval("demo", ["a", "b"], log=lambda *a: None)
+ assert not (tmp_path / "out" / "demo.json").exists()
+
+
+def test_build_leavings_are_not_fed_to_the_judge(tmp_path):
+ """The first real container run left __pycache__/*.pyc beside solution.py. Those bytes went
+ into the text the judge grades — noise it pays for and can be misled by."""
+ solution = tmp_path / "demo-h0__xy" / "verifier" / "solution"
+ (solution / "__pycache__").mkdir(parents=True)
+ (solution / "solution.py").write_text("def add(a, b): return a + b")
+ (solution / "__pycache__" / "solution.cpython-312.pyc").write_bytes(b"\x00\x01\xfe\xff")
+ answers = H.collect_answers(tmp_path)
+ assert "def add" in answers["demo-h0"][0]
+ assert "pycache" not in answers["demo-h0"][0]
+
+
+def _job_with_stats(tmp_path, **stats):
+ job = tmp_path / "jobs" / "arm"
+ job.mkdir(parents=True)
+ (job / "result.json").write_text(json.dumps({"stats": stats}))
+ return job
+
+
+def test_an_arm_whose_trials_all_crashed_is_refused_not_scored_as_zeros(tmp_path, monkeypatch):
+ """Observed live: harbor exited 0 with n_errored_trials=4, the solution dirs were empty, and
+ score() read four legitimate 0.0s — producing lift +0.750 against a working skill arm. An arm
+ that did not run is missing, not bad."""
+ class Done:
+ returncode, stderr, stdout = 0, "", ""
+
+ monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: Done())
+ monkeypatch.setattr(H, "_refuse_broken_job", H._refuse_broken_job)
+ job = _job_with_stats(tmp_path, n_completed_trials=4, n_errored_trials=4, n_cancelled_trials=4)
+ with pytest.raises(RuntimeError, match="refusing to score"):
+ H._refuse_broken_job(job, "terminus-2", "control")
+
+
+def test_a_clean_arm_is_accepted(tmp_path):
+ job = _job_with_stats(tmp_path, n_completed_trials=4, n_errored_trials=0, n_cancelled_trials=0)
+ H._refuse_broken_job(job, "terminus-2", "skill")
+
+
+def test_an_arm_with_no_result_file_is_refused(tmp_path):
+ """No result.json means harbor never got far enough to report; scoring it would invent data."""
+ with pytest.raises(RuntimeError, match="no result.json"):
+ H._refuse_broken_job(tmp_path / "missing", "codex", "skill")
+
+
+def test_an_arm_that_delivered_nothing_at_all_is_refused_not_scored_as_zeros(monkeypatch):
+ """A trial can complete while its agent never ran: the verifier always reports success, so an
+ agent that died on its first API call still counts completed with an empty workspace. Observed
+ live with aider (temperature rejected by claude-sonnet-5) — it would have scored a clean 0.000
+ and read as 'aider is terrible at this skill' rather than 'aider never ran'."""
+ monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.9})
+ with pytest.raises(RuntimeError, match="empty workspace"):
+ H.score({}, "demo", HOLDOUT)
+
+
+def test_a_partly_empty_arm_still_scores_its_failures_as_zero(monkeypatch):
+ """One task delivering nothing is a real failure of that task, not a broken combination."""
+ monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.9})
+ assert H.score({"demo-h0": "code"}, "demo", HOLDOUT) == [0.9, 0.0]
+
+
+def _trial(job, task, exception_type=None):
+ """Harbor's real trial result shape, not an invented one.
+
+ The first version of this helper wrote a top-level `exception_type`, which Harbor never emits —
+ so the guard read nothing, passed its tests, and was a no-op against real output."""
+ d = job / f"{task}__xy"
+ d.mkdir(parents=True, exist_ok=True)
+ record = {"task_name": f"ingot/{task}", "exception_info": None}
+ if exception_type:
+ record["exception_info"] = {"exception_type": exception_type, "exception_message": ""}
+ (d / "result.json").write_text(json.dumps(record))
+ return d
+
+
+def test_broken_tasks_names_only_the_trials_that_failed(tmp_path):
+ job = tmp_path / "arm"
+ _trial(job, "demo-h0")
+ _trial(job, "demo-h1", exception_type="CancelledError")
+ assert H.broken_tasks(job) == {"demo-h1"}
+
+
+def test_one_transient_trial_failure_does_not_discard_the_whole_arm(tmp_path):
+ """Failing the arm on any broken trial threw away three good trials and the paid-for opposite
+ arm. Observed live: the first grid row died on 1 errored trial of 4."""
+ job = tmp_path / "arm"
+ job.mkdir()
+ (job / "result.json").write_text(json.dumps(
+ {"stats": {"n_completed_trials": 4, "n_errored_trials": 1, "n_cancelled_trials": 0}}))
+ H._refuse_broken_job(job, "terminus-2", "skill") # must not raise
+
+
+def test_an_arm_is_still_refused_when_every_trial_broke(tmp_path):
+ job = tmp_path / "arm"
+ job.mkdir()
+ (job / "result.json").write_text(json.dumps(
+ {"stats": {"n_completed_trials": 4, "n_errored_trials": 4, "n_cancelled_trials": 4}}))
+ with pytest.raises(RuntimeError, match="every one of its"):
+ H._refuse_broken_job(job, "terminus-2", "control")
+
+
+def test_a_task_dropped_from_one_arm_is_dropped_from_both(monkeypatch):
+ """Scoring the arms over different task sets means their difference is not lift."""
+ monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.6})
+ scores = H.score({"demo-h0": "x", "demo-h1": "y"}, "demo", HOLDOUT, skip={"demo-h1"})
+ assert scores == [0.6], "the dropped task must not appear in the scored list"
+
+
+def test_scoring_refuses_when_every_task_was_dropped(monkeypatch):
+ monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.6})
+ with pytest.raises(RuntimeError, match="nothing comparable"):
+ H.score({"demo-h0": "x"}, "demo", HOLDOUT, skip={"demo-h0", "demo-h1"})
+
+
+def _attempt(job, task, suffix, *, errored=False, answer="ok"):
+ """One trial directory as Harbor lays it out, with or without a recorded exception."""
+ trial = job / f"{task}__{suffix}"
+ (trial / "verifier" / "solution").mkdir(parents=True)
+ (trial / "verifier" / "solution" / "answer.py").write_text(answer)
+ record = {"task_name": f"ingot/{task}"}
+ if errored:
+ record["exception_info"] = {"exception_type": "AgentSetupTimeoutError"}
+ (trial / "result.json").write_text(json.dumps(record))
+ return trial
+
+
+def test_one_crashed_attempt_does_not_discard_the_task(tmp_path):
+ """With repeats, dropping a task because one attempt broke throws away the attempts that did
+ run — which are the entire reason for paying for repeats."""
+ job = tmp_path / "control"
+ job.mkdir()
+ _attempt(job, "demo-h0", "aa", errored=True)
+ _attempt(job, "demo-h0", "bb")
+ _attempt(job, "demo-h0", "cc")
+ assert H.broken_tasks(job) == set()
+ assert H.broken_trials(job) == {"demo-h0__aa"}
+
+
+def test_a_task_whose_every_attempt_crashed_is_still_dropped(tmp_path):
+ job = tmp_path / "control"
+ job.mkdir()
+ _attempt(job, "demo-h1", "aa", errored=True)
+ _attempt(job, "demo-h1", "bb", errored=True)
+ assert H.broken_tasks(job) == {"demo-h1"}
+
+
+def test_a_crashed_attempts_empty_workspace_is_not_averaged_in_as_a_zero(tmp_path):
+ """The crashed trial's directory exists and is empty through no fault of the agent. Averaged in
+ it would pull a 3-attempt task's mean down by a third and read as the skill performing worse."""
+ job = tmp_path / "control"
+ job.mkdir()
+ _attempt(job, "demo-h0", "aa", errored=True, answer="")
+ _attempt(job, "demo-h0", "bb", answer="real work")
+ answers = H.collect_answers(job, H.broken_trials(job))
+ assert len(answers["demo-h0"]) == 1
+ assert "real work" in answers["demo-h0"][0]
+
+
+def test_runs_at_different_attempt_counts_do_not_collide(tmp_path, monkeypatch):
+ """Harbor refuses a job directory whose config changed, so re-running a skill at a new -k failed
+ all 13 combinations before a container started. The earlier run's trials are also the evidence a
+ rescore reads, so overwriting them is worse than the collision."""
+ monkeypatch.setattr(H, "load_tasks", lambda skill: ([], HOLDOUT, {}))
+ monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "staged")
+ monkeypatch.setattr(H.shutil, "which", lambda binary: "/usr/bin/harbor")
+ monkeypatch.setattr(H, "build_dataset", lambda *a, **k: tmp_path / "ds")
+ monkeypatch.setattr(H, "HARBOR_DIR", tmp_path / "out")
+ monkeypatch.setattr(H, "collect_answers", lambda job, skip=None: {})
+ monkeypatch.setattr(H, "broken_tasks", lambda job: set())
+ monkeypatch.setattr(H, "broken_trials", lambda job: set())
+ monkeypatch.setattr(H, "score", lambda answers, skill, holdout, skip=None: [1.0, 1.0])
+ seen = []
+
+ def fake_run_arm(dataset, agent, source, jobs_dir, job_name, *a, **k):
+ seen.append(jobs_dir)
+ return jobs_dir / job_name
+
+ monkeypatch.setattr(H, "run_arm", fake_run_arm)
+ H.run_harbor_eval("demo", ["claude-code"], attempts=1, log=lambda *a: None)
+ H.run_harbor_eval("demo", ["claude-code"], attempts=3, log=lambda *a: None)
+
+ roots = {path.parent.name for path in seen}
+ assert roots == {"demo", "demo-k3"}
+
+
+def test_a_subscription_capable_harness_will_not_quietly_bill_per_token(monkeypatch):
+ """The default is silent and expensive. A whole grid ran on metered keys with both subscription
+ logins sitting unused on the same host, and nothing in the output said so — the per-arm dollar
+ figure Harbor prints is a computed estimate that reads the same either way."""
+ monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-whatever")
+ monkeypatch.delenv("CLAUDE_FORCE_OAUTH", raising=False)
+ monkeypatch.delenv(H.ALLOW_API_BILLING, raising=False)
+ refusals = H.billing_refusals(["claude-code@anthropic/claude-opus-5"])
+ assert len(refusals) == 1
+ assert "CLAUDE_FORCE_OAUTH" in refusals[0] and "claude setup-token" in refusals[0]
+
+
+def test_a_harness_with_no_cli_to_harness_is_not_refused(monkeypatch):
+ """terminus-2, goose, aider, opencode and pi drive a provider API directly. There is no
+ subscription to prefer, so an API key is inherent to running them at all."""
+ monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-whatever")
+ monkeypatch.setenv("OPENAI_API_KEY", "sk-whatever")
+ monkeypatch.delenv(H.ALLOW_API_BILLING, raising=False)
+ assert H.billing_refusals(["terminus-2@anthropic/claude-opus-5", "goose@anthropic/claude-sonnet-5",
+ "aider@openai/gpt-5.5", "pi@openai/gpt-5.5"]) == []
+
+
+def test_the_subscription_flag_clears_the_refusal(monkeypatch):
+ monkeypatch.setenv("OPENAI_API_KEY", "sk-whatever")
+ monkeypatch.setenv("CODEX_FORCE_AUTH_JSON", "1")
+ monkeypatch.delenv(H.ALLOW_API_BILLING, raising=False)
+ assert H.billing_refusals(["codex@openai/gpt-5.5"]) == []
+
+
+def test_metered_billing_can_still_be_opted_into_deliberately(monkeypatch):
+ monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-whatever")
+ monkeypatch.delenv("CLAUDE_FORCE_OAUTH", raising=False)
+ monkeypatch.setenv(H.ALLOW_API_BILLING, "1")
+ assert H.billing_refusals(["claude-code@anthropic/claude-opus-5"]) == []
+
+
+def test_the_grid_refuses_to_start_rather_than_billing_then_reporting(tmp_path, monkeypatch):
+ """Per-row would be too late: a grid is hours long and the bill is run up by then."""
+ monkeypatch.setattr(H, "load_tasks", lambda skill: ([], HOLDOUT, {}))
+ monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "staged")
+ monkeypatch.setattr(H.shutil, "which", lambda binary: "/usr/bin/harbor")
+ monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-whatever")
+ monkeypatch.delenv("CLAUDE_FORCE_OAUTH", raising=False)
+ monkeypatch.delenv(H.ALLOW_API_BILLING, raising=False)
+
+ def must_not_run(*a, **k):
+ raise AssertionError("a container was started before the billing check")
+
+ monkeypatch.setattr(H, "run_arm", must_not_run)
+ monkeypatch.setattr(H, "build_dataset", must_not_run)
+ with pytest.raises(SystemExit, match="refusing to start"):
+ H.run_harbor_eval("demo", ["claude-code@anthropic/claude-opus-5"], log=lambda *a: None)
+
+
+def test_every_run_asks_for_build_and_setup_headroom(tmp_path, monkeypatch):
+ """Seeded tasks each build their own image, where an unseeded dataset shared one cached image
+ built once. `apt-get update && install` on an uncached image overran the 120s compose budget,
+ and it surfaced as a bare RuntimeError with an empty verifier directory — a build failure that
+ looks nothing like one, and which scored as a dropped task."""
+ seen = []
+
+ class Done:
+ returncode, stderr, stdout = 0, "", ""
+
+ monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: seen.append(argv) or Done())
+ monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None)
+ H.run_arm(tmp_path / "ds", "terminus-2", None, tmp_path / "jobs", "control", log=lambda *a: None)
+
+ argv = seen[0]
+ assert "--environment-build-timeout-multiplier" in argv
+ assert float(argv[argv.index("--environment-build-timeout-multiplier") + 1]) > 1
+ assert "--agent-setup-timeout-multiplier" in argv
+
+
+def test_the_image_ships_what_agents_would_otherwise_install_themselves(tmp_path):
+ """terminus-2 installs tmux and asciinema into the container when they are missing, and that
+ apt-get overran the 120s exec budget on a cold cache — a bare RuntimeError from
+ _install_recording_tools, empty verifier directory, task counted broken and dropped. Its
+ installer skips the work when both are present. pytest is here because the seeded READMEs tell
+ the agent to run it and Ubuntu 24.04 refuses pip installs under PEP 668."""
+ dockerfile = (H.build_dataset("demo", HOLDOUT, tmp_path) / "demo-h0"
+ / "environment" / "Dockerfile").read_text()
+ for package in ("tmux", "asciinema", "python3-pytest"):
+ assert package in dockerfile, f"{package} must be baked in, not installed per trial"
+
+
+def _local_target(alias="dell-qwen", **changes):
+ models = {"dell-qwen": "dot-backbone", "spark-deepseek": "deepseek-v4-flash",
+ "orin-abliterated": "ablit35b"}
+ values = {
+ "alias": alias,
+ "display_name": alias,
+ "base_url": f"http://{alias}.test:8000",
+ "served_model": models[alias],
+ "context_length": 32768,
+ "protocols": frozenset({"chat", "responses", "messages"}),
+ "family": "Qwen3.6" if alias == "dell-qwen" else "fixture-family",
+ "parameter_billions": 27.0 if alias == "dell-qwen" else 1.0,
+ "quantization": "fp8-published" if alias == "dell-qwen" else "fixture-quant",
+ "tool_parser": "qwen3_xml" if alias == "dell-qwen" else "fixture-parser",
+ }
+ values.update(changes)
+ return LocalTarget(**values)
+
+
+def _completed_canary(job: Path, task: str, *, solution=True, exception=None, completed=1):
+ (job / "result.json").parent.mkdir(parents=True, exist_ok=True)
+ (job / "result.json").write_text(json.dumps({"stats": {"n_completed_trials": completed,
+ "n_errored_trials": 0,
+ "n_cancelled_trials": 0}}))
+ trial = job / f"{task}__run" / "verifier" / "solution"
+ trial.mkdir(parents=True)
+ if solution:
+ (trial / "answer.txt").write_text("done")
+ record = {"task_name": f"ingot/{task}", "exception_info": None}
+ if exception:
+ record["exception_info"] = {"exception_type": exception}
+ (trial.parent.parent / "result.json").write_text(json.dumps(record))
+
+
+def test_canary_requires_a_completed_exception_free_trial_with_a_solution(tmp_path, monkeypatch):
+ """A successful Harbor process with no agent deliverable must block the full sweep."""
+ target = _local_target()
+ job = tmp_path / "canaries" / "demo" / target.job_slug / "codex"
+ _completed_canary(job, "demo-h0", solution=False)
+ calls = []
+
+ def fake_run_arm(*args, **kwargs):
+ calls.append((args, kwargs))
+ return job
+
+ monkeypatch.setattr(H, "run_arm", fake_run_arm)
+ record = H.run_canary("demo", tmp_path / "dataset", HOLDOUT, "/staged/demo", "codex",
+ target, tmp_path / "canaries", log=lambda *a: None)
+ assert record["error"] == "canary produced no nonempty verifier solution artifact"
+ route = H.gateway_route(target, "codex")
+ assert route is not None
+ args, kwargs = calls[0]
+ assert args[3] == job.parent and args[4] == f"codex--{route.identity}"
+ assert args[1] == "ingot.optimize.harbor_codex_gateway:GatewayCodex"
+ assert kwargs["task_name"] == "demo-h0" and kwargs["attempts"] == 1
+ assert kwargs["model"] == route.model
+ assert kwargs["agent_env"]["OPENAI_BASE_URL"].startswith("http://172.17.")
+ assert kwargs["process_env"]["PYTHONPATH"].split(os.pathsep)[0] == str(
+ Path(H.__file__).resolve().parents[2])
+
+
+def test_canary_exports_the_persisted_job_after_run_arm(tmp_path, monkeypatch):
+ """Removing the canary telemetry caller would leave its retained attempt undiscoverable."""
+ target = _local_target()
+ job = tmp_path / "canaries" / "demo" / target.job_slug / "codex"
+ _completed_canary(job, "demo-h0")
+ source = tmp_path / "staged"
+ skill_file = source / "demo" / "SKILL.md"
+ skill_file.parent.mkdir(parents=True)
+ skill_file.write_text("fixture skill body\n")
+ events = []
+
+ def fake_run_arm(*args, **kwargs):
+ events.append(("run", job))
+ return job
+
+ def fake_export(exported_job, metadata):
+ events.append(("export", exported_job, metadata))
+ return [{"status": "verified"}]
+
+ monkeypatch.setattr(H, "run_arm", fake_run_arm)
+ monkeypatch.setattr(H, "export_job_attempts", fake_export)
+
+ record = H.run_canary("demo", tmp_path / "dataset", HOLDOUT, str(source), "codex",
+ target, tmp_path / "canaries", log=lambda *a: None)
+
+ assert [event[0] for event in events] == ["run", "export"]
+ assert events[1][1] == job
+ assert events[1][2]["combination"] == record["combination"]
+ assert events[1][2]["arm"] == "canary"
+ assert events[1][2]["skill"] == "demo"
+ assert events[1][2]["task_texts"] == {"demo-h0": HOLDOUT[0]["task"],
+ "demo-h1": HOLDOUT[1]["task"]}
+ assert events[1][2]["skill_sha256"] == hashlib.sha256(
+ skill_file.read_bytes()).hexdigest()
+ assert events[1][2]["skill_body"] == "fixture skill body\n"
+ assert record["ok"] is True and "telemetry_error" not in record
+
+
+def test_telemetry_provenance_sanitizes_task_text_before_persisting(tmp_path):
+ source = tmp_path / "staged"
+ skill_file = source / "demo" / "SKILL.md"
+ skill_file.parent.mkdir(parents=True)
+ skill_file.write_text("fixture skill body\n")
+ holdout = [{"task": (
+ "Call https://private-endpoint.invalid/v1 with "
+ "OPENAI_API_KEY=fixture-secret-value"
+ )}]
+
+ provenance = H._telemetry_provenance("demo", holdout, str(source))
+
+ persisted = provenance["task_texts"]["demo-h0"]
+ assert "private-endpoint.invalid" not in persisted
+ assert "fixture-secret-value" not in persisted
+ assert " LocalTarget:
+ context = {"dell-qwen": 163840, "spark-deepseek": 1048576,
+ "orin-abliterated": 65536}[alias]
+ return LocalTarget(
+ alias=alias,
+ display_name=TARGETS[alias]["display_name"],
+ base_url="http://local.test:8011",
+ served_model=TARGETS[alias]["served_model"],
+ context_length=context,
+ protocols=frozenset({"chat", "messages", "responses"}),
+ )
+
+
+def test_gateway_routes_only_the_three_observed_role_rejections():
+ dell = _target("dell-qwen")
+ spark = _target("spark-deepseek")
+ assert G.gateway_route(dell, "claude-code") is not None
+ assert G.gateway_route(spark, "claude-code") is not None
+ assert G.gateway_route(dell, "codex") is not None
+ assert G.gateway_route(spark, "codex") is None
+ orin = _target("orin-abliterated")
+ for harness in ("claude-code", "terminus-2", "goose", "opencode", "openclaw",
+ "mini-swe-agent", "codex", "aider", "pi"):
+ assert G.gateway_route(orin, harness) is None
+ for harness in ("terminus-2", "goose", "opencode", "openclaw", "mini-swe-agent", "aider", "pi"):
+ assert G.gateway_route(dell, harness) is None
+ unknown = LocalTarget("third-target", "Third", "http://local.test:9000", "third", 32768,
+ frozenset({"chat", "messages", "responses"}))
+ assert G.gateway_route(unknown, "claude-code") is None
+
+
+def test_gateway_identity_is_stable_and_does_not_expose_upstream_address():
+ route = G.gateway_route(_target(), "claude-code")
+ assert route is not None
+ assert route.revision == f"{G.DELL_CLAUDE_OUTPUT_CAP_REVISION}-20480"
+ assert route.model.startswith("harbor-compat-")
+ assert "local.test" not in route.identity
+ assert route.identity == G.gateway_route(_target(), "claude-code").identity
+ assert G.gateway_metadata(route)["gateway_agent"] == "claude-code"
+
+
+def test_dell_claude_gateway_caps_message_output_at_one_eighth_of_context():
+ route = G.gateway_route(_target(), "claude-code")
+ assert route is not None
+ assert route.output_limit == 20480
+ request = {
+ "model": route.model,
+ "max_tokens": 32000,
+ "messages": [{"role": "user", "content": "start"}],
+ }
+ got = G.normalize_role_request(request, "anthropic_messages", {route.model: route.output_limit})
+ assert got["max_tokens"] == 20480
+
+
+def test_dell_claude_gateway_uses_custom_openai_max_tokens_not_responses_output_key():
+ route = G.gateway_route(_target(), "claude-code")
+ assert route is not None
+ request = {
+ "model": route.model,
+ "max_output_tokens": 32000,
+ "messages": [{"role": "user", "content": "start"}],
+ }
+ got = G.normalize_role_request(request, "anthropic_messages", {route.model: route.output_limit})
+ assert got["max_tokens"] == 20480
+ assert "max_output_tokens" not in got
+
+
+def test_dell_claude_gateway_identity_and_model_change_with_its_context_cap():
+ full = G.gateway_route(_target(), "claude-code")
+ smaller = G.gateway_route(replace(_target(), context_length=16384), "claude-code")
+ assert full is not None and smaller is not None
+ assert full.output_limit == 20480 and smaller.output_limit == 2048
+ assert full.identity != smaller.identity
+ assert full.model != smaller.model
+
+
+def test_gateway_does_not_change_the_passing_spark_claude_output_budget():
+ route = G.gateway_route(_target("spark-deepseek"), "claude-code")
+ assert route is not None
+ assert route.output_limit is None
+
+
+def test_spark_claude_gateway_strips_the_custom_openai_unsupported_output_key():
+ route = G.gateway_route(_target("spark-deepseek"), "claude-code")
+ assert route is not None
+ got = G.normalize_role_request({
+ "model": route.model,
+ "max_output_tokens": 32000,
+ "messages": [{"role": "user", "content": "start"}],
+ }, "anthropic_messages")
+ assert "max_output_tokens" not in got
+ assert "max_tokens" not in got
+
+
+def test_gateway_codex_uses_a_non_websocket_custom_provider_config():
+ setup = G.codex_gateway_setup_command()
+ assert setup.startswith("set -eu\n")
+ assert 'mkdir -p "$CODEX_HOME"' in setup
+ assert "python3 <<'PY'" in setup
+ assert "node <<" not in setup
+ assert 'model_provider = "harbor_compat"' in setup
+ assert 'wire_api = "responses"' in setup
+ assert "supports_websockets = false" in setup
+ assert 'base_url = "${OPENAI_BASE_URL}"' in setup
+
+
+def test_gateway_codex_builds_truthful_catalog_from_runtime_instructions(tmp_path):
+ route = G.gateway_route(_target(), "codex")
+ assert route is not None
+ bin_dir = tmp_path / "bin"
+ bin_dir.mkdir()
+ codex = bin_dir / "codex"
+ codex.write_text("""#!/bin/sh
+cat <<'JSON'
+{"models":[{"slug":"bundled","display_name":"Bundled","description":null,
+"default_reasoning_level":"medium","supported_reasoning_levels":[{"effort":"medium","description":"Medium"}],
+"shell_type":"shell_command","visibility":"list","supported_in_api":true,"priority":1,
+"availability_nux":null,"upgrade":null,
+"model_messages":{"instructions_template":"runtime-owned instructions","instructions_variables":null,
+"approvals":null,"collaboration_modes":null,"auto_review":null,"permissions":null,"token_budget":null},
+"support_verbosity":true,"default_verbosity":"medium","apply_patch_tool_type":"freeform",
+"truncation_policy":{"mode":"tokens","limit":32000},"supports_parallel_tool_calls":true,
+"context_window":272000,"max_context_window":272000,"experimental_supported_tools":["custom"]}]}
+JSON
+""")
+ codex.chmod(0o755)
+ home = tmp_path / "codex-home"
+ completed = subprocess.run(
+ ["sh", "-c", G.codex_gateway_setup_command()],
+ env={
+ **os.environ,
+ "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
+ "CODEX_HOME": str(home),
+ "OPENAI_BASE_URL": "http://gateway.test/v1",
+ "HARBOR_GATEWAY_CODEX_PROVIDER": "1",
+ "HARBOR_GATEWAY_CODEX_MODEL": route.model,
+ "HARBOR_GATEWAY_CODEX_SERVED_MODEL": "dot-backbone",
+ "HARBOR_GATEWAY_CODEX_CONTEXT": "163840",
+ },
+ capture_output=True,
+ text=True,
+ )
+ assert completed.returncode == 0, completed.stderr
+ catalog = json.loads((home / "model-catalog.json").read_text())
+ assert catalog == {"models": [{
+ **catalog["models"][0],
+ "slug": route.model,
+ "display_name": "dot-backbone",
+ "default_reasoning_level": None,
+ "supported_reasoning_levels": [],
+ "supports_reasoning_summary_parameter": False,
+ "support_verbosity": False,
+ "default_verbosity": None,
+ "apply_patch_tool_type": None,
+ "supports_parallel_tool_calls": False,
+ "context_window": 163840,
+ "max_context_window": 163840,
+ "experimental_supported_tools": [],
+ "model_messages": {
+ "instructions_template": "runtime-owned instructions",
+ "instructions_variables": None,
+ "approvals": None,
+ "collaboration_modes": None,
+ "auto_review": None,
+ "permissions": None,
+ "token_budget": None,
+ },
+ }]}
+ config = (home / "config.toml").read_text()
+ assert f'model_catalog_json = "{home}/model-catalog.json"' in config
+
+
+def test_gateway_codex_env_enables_only_its_custom_provider_setup():
+ route = G.gateway_route(_target(), "codex")
+ assert route is not None
+ assert "codex-http-catalog-v8" in route.revision
+ env = G.gateway_agent_env(_target(), route)
+ assert env["HARBOR_GATEWAY_CODEX_PROVIDER"] == "1"
+ assert env["HARBOR_GATEWAY_CODEX_MODEL"] == route.model
+ assert env["HARBOR_GATEWAY_CODEX_SERVED_MODEL"] == "dot-backbone"
+ assert env["HARBOR_GATEWAY_CODEX_CONTEXT"] == "163840"
+
+
+def test_gateway_process_env_precedes_any_existing_import_path_with_the_project_root():
+ routed = G.gateway_process_env({"PYTHONPATH": "/existing", "OPENAI_API_KEY": "secret"})
+ assert routed["PYTHONPATH"].split(os.pathsep) == [str(Path(G.__file__).resolve().parents[2]), "/existing"]
+ assert routed["OPENAI_API_KEY"] == "secret"
+
+
+def test_gateway_codex_imports_through_the_harbor_runtime_subprocess():
+ runtime = os.environ.get("HARBOR_RUNTIME_PYTHON")
+ if not runtime:
+ pytest.skip("requires the installed Harbor Python runtime")
+ root = Path(__file__).resolve().parents[1]
+ completed = subprocess.run(
+ [runtime, "-c", "from ingot.optimize.harbor_codex_gateway import GatewayCodex; "
+ "assert GatewayCodex.name() == 'codex'"],
+ cwd=root, env={**os.environ, "PYTHONPATH": str(root)}, capture_output=True, text=True,
+ )
+ assert completed.returncode == 0, completed.stderr
+
+
+def test_gateway_codex_writes_provider_setup_before_the_harbor_adapter(monkeypatch):
+ if importlib.util.find_spec("harbor") is None:
+ pytest.skip("requires the installed Harbor Python runtime")
+ from harbor.agents.installed.codex import Codex
+ from ingot.optimize.harbor_codex_gateway import GatewayCodex
+
+ assert not any(flag.kwarg == "reasoning_effort" for flag in GatewayCodex.CLI_FLAGS)
+ calls = []
+
+ class ProbeGatewayCodex(GatewayCodex):
+ async def exec_as_agent(self, environment, command, **kwargs):
+ calls.append(("setup", command, kwargs))
+
+ async def fake_parent_run(self, instruction, environment, context):
+ calls.append(("parent", instruction))
+
+ monkeypatch.setattr(Codex, "run", fake_parent_run)
+ asyncio.run(object.__new__(ProbeGatewayCodex).run("task", object(), object()))
+ assert calls[0][0] == "setup" and "supports_websockets = false" in calls[0][1]
+ assert calls[0][2]["env"]["CODEX_HOME"] == "/tmp/codex-home"
+ assert calls[1] == ("parent", "task")
+
+
+def test_normalizer_moves_anthropic_system_content_to_a_user_message_without_touching_tools():
+ tool_use = {"type": "tool_use", "id": "call_1", "name": "shell", "input": {"cmd": "pwd"}}
+ tool_result = {"type": "tool_result", "tool_use_id": "call_1", "content": "ok"}
+ request = {
+ "system": "follow the repository instructions",
+ "messages": [
+ {"role": "user", "content": "start"},
+ {"role": "assistant", "content": [tool_use]},
+ {"role": "user", "content": [tool_result]},
+ ],
+ }
+ got = G.normalize_role_request(request, "anthropic_messages")
+ assert "system" not in got
+ assert got["messages"][0] == {
+ "role": "user", "content": "[system]\nfollow the repository instructions",
+ }
+ assert got["messages"][2]["content"] == [tool_use]
+ assert got["messages"][3]["content"] == [tool_result]
+
+
+def test_normalizer_moves_responses_instructions_and_developer_roles_without_touching_tools():
+ function_call = {"type": "function_call", "call_id": "call_1", "name": "shell", "arguments": "{}"}
+ function_output = {"type": "function_call_output", "call_id": "call_1", "output": "ok"}
+ request = {
+ "instructions": "follow the repository instructions",
+ "input": [
+ {"role": "developer", "content": [{"type": "input_text", "text": "be concise"}]},
+ {"role": "user", "content": [{"type": "input_text", "text": "start"}]},
+ function_call,
+ function_output,
+ ],
+ }
+ got = G.normalize_role_request(request, "aresponses")
+ assert "instructions" not in got
+ assert got["input"][0] == {
+ "role": "user",
+ "content": [{"type": "input_text", "text": "[instructions]\nfollow the repository instructions"}],
+ }
+ assert got["input"][1]["role"] == "user"
+ assert got["input"][3] == function_call
+ assert got["input"][4] == function_output
+
+
+def test_normalizer_strips_codex_reasoning_from_custom_openai_responses():
+ request = {
+ "input": [{"role": "user", "content": [{"type": "input_text", "text": "start"}]}],
+ "reasoning": {"effort": "high", "summary": "auto"},
+ "reasoning_effort": "high",
+ }
+
+ got = G.normalize_role_request(request, "aresponses")
+
+ assert "reasoning" not in got
+ assert "reasoning_effort" not in got
+ assert got["input"] == request["input"]
+
+
+def test_normalizer_leaves_unrelated_routes_unchanged():
+ request = {"messages": [{"role": "system", "content": "unchanged"}]}
+ assert G.normalize_role_request(request, "completion") == request
+
+
+def test_gateway_session_requires_its_revisioned_models_and_writes_cleanup_receipt(tmp_path, monkeypatch):
+ route = G.gateway_route(_target(), "claude-code")
+ assert route is not None
+
+ class Process:
+ def __init__(self):
+ self.stopped = False
+
+ def poll(self):
+ return 0 if self.stopped else None
+
+ def terminate(self):
+ self.stopped = True
+
+ def wait(self, timeout):
+ return 0
+
+ process = Process()
+ calls = []
+ spawned = {}
+
+ class Response:
+ status = 200
+
+ def __init__(self, payload):
+ self.payload = payload
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *args):
+ return False
+
+ def read(self):
+ return json.dumps(self.payload).encode()
+
+ def fake_urlopen(request, timeout):
+ calls.append(request)
+ if isinstance(request, str):
+ return Response({"status": "healthy"})
+ return Response({"data": [{"id": route.model}]})
+
+ def popen(*args, **kwargs):
+ spawned.update(kwargs)
+ return process
+
+ monkeypatch.setattr(G.urllib.request, "urlopen", fake_urlopen)
+ session = G.GatewaySession([(route, _target())], tmp_path / "gateway",
+ litellm_bin=__file__, popen=popen)
+ session.start()
+ assert spawned["env"]["PYTHONPATH"].split(os.pathsep)[0] == str(
+ Path(G.__file__).resolve().parents[2])
+ config = (tmp_path / "gateway" / "config.json").read_text()
+ assert "local.test" not in config and route.upstream_env in config
+ assert f"custom_openai/{route.served_model}" in config
+ assert any(isinstance(call, str) and call.endswith("/health/liveliness") for call in calls)
+ session.close()
+ receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text())
+ assert receipt["stopped"] is True and receipt["routes"] == [route.identity]
+
+
+def test_gateway_failed_start_still_writes_a_cleanup_receipt(tmp_path):
+ route = G.gateway_route(_target(), "claude-code")
+ assert route is not None
+
+ def fail_start(*args, **kwargs):
+ raise OSError("cannot bind")
+
+ session = G.GatewaySession([(route, _target())], tmp_path / "gateway",
+ litellm_bin=__file__, popen=fail_start)
+ with pytest.raises(OSError, match="cannot bind"):
+ session.start()
+ receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text())
+ assert receipt["reason"] == "failed-start" and receipt["stopped"] is True
+
+
+def test_gateway_missing_dedicated_runtime_fails_before_start_and_writes_receipt(tmp_path):
+ route = G.gateway_route(_target(), "claude-code")
+ assert route is not None
+ session = G.GatewaySession([(route, _target())], tmp_path / "gateway",
+ litellm_bin=str(tmp_path / "missing-litellm"))
+ with pytest.raises(RuntimeError, match="runtime is missing"):
+ session.start()
+ receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text())
+ assert receipt["reason"] == "failed-start"
+
+
+def test_default_gateway_runtime_is_checkout_relative():
+ runtime = Path(G._LITELLM_BIN)
+
+ assert runtime.is_absolute()
+ assert runtime == Path(G.__file__).resolve().parents[2] / ".venv-harbor-gateway/bin/litellm"
+
+
+def test_gateway_fixed_bind_probe_is_bounded_and_closes_a_preexisting_listener(monkeypatch):
+ class Connection:
+ closed = False
+
+ def close(self):
+ self.closed = True
+
+ connection = Connection()
+ calls = []
+
+ def connect(address, timeout):
+ calls.append((address, timeout))
+ return connection
+
+ monkeypatch.setattr(G.socket, "create_connection", connect)
+ assert G._fixed_bind_occupied() is True
+ assert connection.closed is True
+ assert calls == [((G.GATEWAY_HOST, G.GATEWAY_PORT), 0.2)]
+
+
+def test_gateway_rejects_any_preexisting_fixed_bind_before_spawn(tmp_path, monkeypatch):
+ route = G.gateway_route(_target(), "claude-code")
+ assert route is not None
+ monkeypatch.setattr(G, "_fixed_bind_occupied", lambda: True, raising=False)
+
+ def unexpected_spawn(*args, **kwargs):
+ pytest.fail("must not spawn beside a stale gateway")
+
+ session = G.GatewaySession([(route, _target())], tmp_path / "gateway",
+ litellm_bin=__file__, popen=unexpected_spawn)
+ with pytest.raises(RuntimeError, match="bind is already occupied"):
+ session.start()
+ receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text())
+ assert receipt["reason"] == "failed-start" and receipt["stopped"] is True
+
+
+def test_gateway_rejects_a_post_spawn_model_superset_and_cleans_up(tmp_path, monkeypatch):
+ route = G.gateway_route(_target(), "claude-code")
+ assert route is not None
+
+ class Process:
+ stopped = False
+
+ def poll(self):
+ return 0 if self.stopped else None
+
+ def terminate(self):
+ self.stopped = True
+
+ def wait(self, timeout):
+ return 0
+
+ process = Process()
+ monkeypatch.setattr(G, "_fixed_bind_occupied", lambda: False, raising=False)
+
+ class Response:
+ status = 200
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *args):
+ return False
+
+ def read(self):
+ return json.dumps({"data": [{"id": route.model}, {"id": "stale-old-route"}]}).encode()
+
+ def fake_urlopen(request, timeout):
+ return Response()
+
+ monkeypatch.setattr(G.urllib.request, "urlopen", fake_urlopen)
+ session = G.GatewaySession([(route, _target())], tmp_path / "gateway",
+ litellm_bin=__file__, popen=lambda *args, **kwargs: process,
+ health_timeout=0.01)
+ with pytest.raises(RuntimeError, match="did not become healthy"):
+ session.start()
+ assert process.stopped is True
+ receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text())
+ assert receipt["stopped"] is True
+
+
+def test_gateway_requires_its_spawned_process_alive_after_exact_model_health(tmp_path, monkeypatch):
+ route = G.gateway_route(_target(), "claude-code")
+ assert route is not None
+
+ class Process:
+ polls = [None, 0]
+ stopped = False
+
+ def poll(self):
+ return self.polls.pop(0) if self.polls else 0
+
+ def terminate(self):
+ self.stopped = True
+
+ def wait(self, timeout):
+ return 0
+
+ process = Process()
+ monkeypatch.setattr(G, "_fixed_bind_occupied", lambda: False, raising=False)
+
+ class Response:
+ status = 200
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *args):
+ return False
+
+ def read(self):
+ return json.dumps({"data": [{"id": route.model}]}).encode()
+
+ monkeypatch.setattr(G.urllib.request, "urlopen", lambda request, timeout: Response())
+ session = G.GatewaySession([(route, _target())], tmp_path / "gateway",
+ litellm_bin=__file__, popen=lambda *args, **kwargs: process,
+ health_timeout=0.01)
+ with pytest.raises(RuntimeError, match="exited before health check"):
+ session.start()
+ assert process.stopped is False
+ receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text())
+ assert receipt["stopped"] is True
diff --git a/tests/test_harbor_langfuse.py b/tests/test_harbor_langfuse.py
new file mode 100644
index 0000000..bf52f1f
--- /dev/null
+++ b/tests/test_harbor_langfuse.py
@@ -0,0 +1,1290 @@
+"""Harbor attempt export and receipt validation without live telemetry calls."""
+from __future__ import annotations
+
+import hashlib
+import json
+import os
+import shutil
+import threading
+from concurrent.futures import ThreadPoolExecutor
+from pathlib import Path
+
+import pytest
+from langfuse import Langfuse
+
+from ingot.optimize import harbor_langfuse as L
+
+
+FIXTURE = Path(__file__).parent / "fixtures" / "harbor" / "langfuse-trial"
+METADATA = {
+ "combination": "codex@fixture-model--dell-fixture",
+ "harness": "codex",
+ "model": "fixture-model",
+ "target_alias": "dell-fixture",
+ "endpoint_fingerprint": "f" * 64,
+ "protocol": "openai",
+ "task_fingerprint": "t" * 64,
+ "attempts": 1,
+ "arm": "skill",
+}
+SKILL_BODY = "fixture skill body\n"
+PROVENANCE_METADATA = {
+ **METADATA,
+ "skill": "demo",
+ "skill_body": SKILL_BODY,
+ "skill_sha256": hashlib.sha256(SKILL_BODY.encode()).hexdigest(),
+ "task_texts": {"fixture-h0": "Use only deterministic fixture input."},
+}
+
+
+def copy_trial(tmp_path: Path) -> Path:
+ trial = tmp_path / "job" / "fixture-h0__attempt-1"
+ shutil.copytree(FIXTURE, trial)
+ return trial
+
+
+def payload_sha256(payload: dict) -> str:
+ encoded = json.dumps(payload, sort_keys=True, separators=(",", ":"),
+ ensure_ascii=False).encode("utf-8")
+ return hashlib.sha256(encoded).hexdigest()
+
+
+class FakeObservation:
+ def __init__(self) -> None:
+ self.ended = 0
+ self.id = ""
+
+ def end(self) -> None:
+ self.ended += 1
+
+
+class FakeLangfuse:
+ def __init__(self, *, readback: str = "matching") -> None:
+ self.readback = readback
+ self.seeds: list[str] = []
+ self.observations: list[dict] = []
+ self.observation = FakeObservation()
+ self.flushes = 0
+ self.reads: list[str] = []
+
+ def create_trace_id(self, *, seed: str) -> str:
+ self.seeds.append(seed)
+ return Langfuse.create_trace_id(seed=seed)
+
+ def start_observation(self, **kwargs):
+ self.observations.append(kwargs)
+ self.observation.id = f"{len(self.observations):016x}"
+ return self.observation
+
+ def flush(self) -> None:
+ self.flushes += 1
+
+ def read_trace(self, trace_id: str):
+ self.reads.append(trace_id)
+ if self.readback == "absent" or not self.observations:
+ return None
+ metadata = dict(self.observations[-1]["metadata"])
+ if self.readback == "wrong-hash":
+ metadata["payload_sha256"] = "0" * 64
+ if self.readback == "wrong-evidence":
+ metadata["telemetry_evidence"] = {
+ **metadata.get("telemetry_evidence", {}), "model": "wrong-model",
+ }
+ if self.readback == "wrong-id":
+ trace_id = "0" * 32
+ return {"id": trace_id, "observations": [
+ {"id": self.observation.id, "name": "harbor-attempt",
+ "type": "AGENT", "metadata": metadata},
+ ]}
+
+
+class FakeSdkOnly:
+ """Langfuse SDK surface without the test-only read_trace seam."""
+
+ def __init__(self) -> None:
+ self.delegate = FakeLangfuse()
+
+ @property
+ def observations(self):
+ return self.delegate.observations
+
+ def create_trace_id(self, *, seed: str) -> str:
+ return self.delegate.create_trace_id(seed=seed)
+
+ def start_observation(self, **kwargs):
+ return self.delegate.start_observation(**kwargs)
+
+ def flush(self) -> None:
+ self.delegate.flush()
+
+
+def test_payload_captures_attempt_evidence_and_structurally_excludes_sensitive_fields(tmp_path):
+ """Removing a retained evidence field or copying Harbor config/env data must fail this test."""
+ trial = copy_trial(tmp_path)
+
+ payload = L.build_attempt_payload(trial, {
+ **METADATA,
+ "base_url": "https://metadata-endpoint.invalid/v1",
+ "skill_body": "PRIVATE SKILL BODY",
+ "skill_sha256": hashlib.sha256(b"PRIVATE SKILL BODY").hexdigest(),
+ "environment": {"TOKEN": "metadata-secret"},
+ })
+
+ assert payload == {
+ "exporter_revision": "harbor-langfuse-v3",
+ "attempt": {
+ "id": "00000000-0000-0000-0000-000000000001",
+ "trial_name": "fixture-h0__attempt-1",
+ },
+ "task": {
+ "name": "ingot/fixture-h0",
+ "checksum": "fixture-task-checksum",
+ "source": "fixture",
+ "text": "",
+ },
+ "skill": "",
+ "trajectory": {
+ "steps": [
+ {"source": "user", "timestamp": "2000-01-01T00:00:00Z"},
+ {"source": "agent", "timestamp": "2000-01-01T00:00:00Z"},
+ {"source": "agent", "timestamp": "2000-01-01T00:00:01Z"},
+ ],
+ },
+ "verifier_output": {"test-stdout.txt": ""},
+ "solution_artifacts": {
+ "_objective_check.txt": (
+ "Fixture objective check\n\n"
+ "- deterministic output exists\n"
+ "- no external endpoint was contacted\n"
+ "- no credential was used\n\nPASS\n"
+ )
+ },
+ "status": "failed",
+ "exception_category": "AgentSetupTimeoutError",
+ "error_detail": "sanitized fixture failure",
+ "timestamps": {
+ "started_at": "2000-01-01T00:00:00Z",
+ "finished_at": "2000-01-01T00:00:01Z",
+ "error_at": "2000-01-01T00:00:00Z",
+ },
+ "usage": {
+ "input_tokens": 120,
+ "cache_tokens": 40,
+ "output_tokens": 30,
+ "cost_usd": 0.01,
+ },
+ "model": "fixture-model",
+ "metadata": {
+ **METADATA,
+ "skill_sha256": hashlib.sha256(b"PRIVATE SKILL BODY").hexdigest(),
+ },
+ }
+ rendered = json.dumps(payload, sort_keys=True)
+ for forbidden in (
+ "fixture-secret-must-not-export",
+ "fixture-endpoint.invalid",
+ "metadata-endpoint.invalid",
+ "metadata-secret",
+ "PRIVATE SKILL BODY",
+ "/fixtures/skills/private-skill/SKILL.md",
+ '"environment"',
+ ):
+ assert forbidden not in rendered
+
+
+def test_payload_bounds_text_and_omits_binary_solution_artifacts(tmp_path):
+ """An oversized or binary solution must never cross the telemetry boundary."""
+ trial = copy_trial(tmp_path)
+ solution = trial / "verifier" / "solution"
+ (solution / "long.txt").write_text("prefix-" + "x" * 3000 + "-tail")
+ (solution / "output.bin").write_bytes(b"\x00\xff\x00private")
+
+ payload = L.build_attempt_payload(trial, METADATA)
+
+ assert "output.bin" not in payload["solution_artifacts"]
+ assert len(payload["solution_artifacts"]["long.txt"]) <= 2000
+ assert payload["solution_artifacts"]["long.txt"].endswith("-tail")
+
+
+def test_payload_accepts_a_bounded_large_trajectory_and_compacts_it(tmp_path):
+ """Harbor trajectories exceed artifact limits but export only a compact safe projection."""
+ trial = copy_trial(tmp_path)
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["steps"][0]["message"] = "x" * L._MAX_TEXT_BYTES
+ trajectory_path.write_text(json.dumps(trajectory))
+ assert trajectory_path.stat().st_size > L._MAX_TEXT_BYTES
+
+ payload = L.build_attempt_payload(trial, METADATA)
+
+ assert len(payload["trajectory"]["steps"]) == 3
+ assert "message" not in json.dumps(payload["trajectory"])
+
+
+def test_payload_accepts_observed_full_arm_goose_trajectory_and_compacts_it(tmp_path):
+ """Goose emitted a 3,368,055-byte trajectory whose freeform fields must be discarded."""
+ trial = copy_trial(tmp_path)
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["steps"][0]["message"] = "PRIVATE-GOOSE-MARKER" + "x" * (13 * 256 * 1024)
+ trajectory_path.write_text(json.dumps(trajectory))
+ assert trajectory_path.stat().st_size > 3 * 1024 * 1024
+
+ payload = L.build_attempt_payload(trial, METADATA)
+
+ assert len(payload["trajectory"]["steps"]) == 3
+ assert "PRIVATE-GOOSE-MARKER" not in json.dumps(payload["trajectory"])
+
+
+def test_export_discovers_a_valid_large_trajectory_and_compacts_it(tmp_path):
+ trial = copy_trial(tmp_path)
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["steps"][0]["message"] = "PRIVATE-LARGE-MARKER" + "x" * L._MAX_TEXT_BYTES
+ trajectory_path.write_text(json.dumps(trajectory))
+ assert trajectory_path.stat().st_size > L._MAX_TEXT_BYTES
+ client = FakeLangfuse()
+
+ receipts = L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert len(receipts) == 1
+ assert len(client.observations) == 1
+ assert "PRIVATE-LARGE-MARKER" not in json.dumps(client.observations)
+
+
+def test_payload_rejects_a_trajectory_above_its_dedicated_bound(tmp_path):
+ """The larger fixed-file allowance must remain bounded against agent-controlled input."""
+ trial = copy_trial(tmp_path)
+ trajectory_path = trial / "agent" / "trajectory.json"
+ with trajectory_path.open("wb") as handle:
+ handle.seek(L._MAX_TRAJECTORY_BYTES)
+ handle.write(b"tail")
+
+ with pytest.raises(L.TelemetryReceiptError, match="byte budget"):
+ L.build_attempt_payload(trial, METADATA)
+
+
+def test_payload_projects_only_allowlisted_trajectory_fields(tmp_path):
+ """Agent step IDs and arbitrary numeric metrics cannot cross the telemetry boundary."""
+ trial = copy_trial(tmp_path)
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["final_metrics"] = {
+ "total_prompt_tokens": 80,
+ "total_cached_tokens": 25,
+ "total_completion_tokens": 20,
+ "total_cost_usd": 0.005,
+ "untrusted_numeric_field": 123,
+ }
+ trajectory_path.write_text(json.dumps(trajectory))
+
+ projected = L.build_attempt_payload(trial, METADATA)["trajectory"]
+
+ assert all("step_id" not in step for step in projected["steps"])
+ assert projected["final_metrics"] == {
+ "total_prompt_tokens": 80,
+ "total_cached_tokens": 25,
+ "total_completion_tokens": 20,
+ "total_cost_usd": 0.005,
+ }
+
+
+def test_payload_treats_missing_trajectory_steps_as_empty(tmp_path):
+ trial = copy_trial(tmp_path)
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ del trajectory["steps"]
+ trajectory_path.write_text(json.dumps(trajectory))
+
+ assert L.build_attempt_payload(trial, METADATA)["trajectory"]["steps"] == []
+
+
+def test_payload_rejects_non_list_trajectory_steps(tmp_path):
+ trial = copy_trial(tmp_path)
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["steps"] = {"source": "agent"}
+ trajectory_path.write_text(json.dumps(trajectory))
+
+ with pytest.raises(L.TelemetryReceiptError, match="trajectory steps"):
+ L.build_attempt_payload(trial, METADATA)
+
+
+def test_payload_trajectory_projection_drops_freeform_strings(tmp_path):
+ trial = copy_trial(tmp_path)
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["schema_version"] = "PRIVATE-SCHEMA-MARKER"
+ trajectory["session_id"] = "PRIVATE-SESSION-MARKER"
+ trajectory["agent"] = {
+ "name": "PRIVATE-AGENT-MARKER",
+ "version": "PRIVATE-VERSION-MARKER",
+ "model_name": "PRIVATE-MODEL-MARKER",
+ }
+ trajectory["steps"] = [
+ {"source": "user", "timestamp": "2000-01-01T00:00:00Z", "message": "PRIVATE"},
+ {"source": "agent", "timestamp": "2000-01-01T00:00:01Z"},
+ {"source": "tool", "timestamp": "2000-01-01T00:00:02Z"},
+ {"source": "PRIVATE-SOURCE-MARKER", "timestamp": "not-a-timestamp"},
+ ]
+ trajectory_path.write_text(json.dumps(trajectory))
+
+ projected = L.build_attempt_payload(trial, METADATA)["trajectory"]
+
+ assert projected == {"steps": [
+ {"source": "user", "timestamp": "2000-01-01T00:00:00Z"},
+ {"source": "agent", "timestamp": "2000-01-01T00:00:01Z"},
+ {"timestamp": "2000-01-01T00:00:02Z"},
+ {},
+ ]}
+
+
+def test_payload_model_does_not_fall_back_to_freeform_trajectory_agent(tmp_path):
+ trial = copy_trial(tmp_path)
+ result_path = trial / "result.json"
+ result = json.loads(result_path.read_text())
+ result["agent_info"] = {}
+ result_path.write_text(json.dumps(result))
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["agent"]["model_name"] = "PRIVATE-MODEL-MARKER"
+ trajectory_path.write_text(json.dumps(trajectory))
+ metadata = {key: value for key, value in METADATA.items() if key != "model"}
+
+ assert L.build_attempt_payload(trial, metadata)["model"] == ""
+
+
+def test_payload_rejects_solution_symlink_outside_the_verifier_root(tmp_path):
+ """Following an agent-controlled artifact symlink would export a parent-process host file."""
+ trial = copy_trial(tmp_path)
+ outside = tmp_path / "parent-secret.txt"
+ outside.write_text("outside-parent-secret")
+ artifact = trial / "verifier" / "solution" / "outside.txt"
+ artifact.symlink_to(outside)
+
+ with pytest.raises(L.TelemetryReceiptError, match="admission"):
+ L.build_attempt_payload(trial, METADATA)
+
+
+def test_payload_omits_large_sparse_artifact_before_decoding(tmp_path, monkeypatch):
+ """A sparse optional artifact is omitted by size before its contents are allocated."""
+ trial = copy_trial(tmp_path)
+ artifact = trial / "verifier" / "solution" / "large.txt"
+ with artifact.open("wb") as handle:
+ handle.seek(8 * 1024 * 1024)
+ handle.write(b"tail")
+ original_read_bytes = Path.read_bytes
+
+ def reject_full_read(path):
+ if path == artifact:
+ pytest.fail("oversized artifact was read before its byte cap was checked")
+ return original_read_bytes(path)
+
+ monkeypatch.setattr(Path, "read_bytes", reject_full_read)
+
+ payload = L.build_attempt_payload(trial, METADATA)
+
+ assert "large.txt" not in payload["solution_artifacts"]
+ assert payload["artifact_projection"]["omitted_files"] == 1
+
+
+def test_payload_rejects_nul_free_control_byte_binary(tmp_path):
+ """Valid UTF-8 with binary control bytes is not a text artifact."""
+ trial = copy_trial(tmp_path)
+ artifact = trial / "verifier" / "solution" / "control.txt"
+ artifact.write_bytes(b"visible\x01binary\x02content")
+
+ payload = L.build_attempt_payload(trial, METADATA)
+
+ assert "control.txt" not in payload["solution_artifacts"]
+
+
+def test_export_rejects_a_symlinked_trial_before_reading_outside_the_job(tmp_path):
+ """A job entry cannot redefine an outside directory as a persisted attempt."""
+ outside_trial = copy_trial(tmp_path / "outside")
+ job = tmp_path / "job"
+ job.mkdir()
+ linked_trial = job / "fixture-h0__attempt-1"
+ linked_trial.symlink_to(outside_trial, target_is_directory=True)
+ client = FakeLangfuse()
+
+ with pytest.raises(L.TelemetryReceiptError, match="symlink|directory|admission"):
+ L.export_job_attempts(job, METADATA, client=client)
+
+ assert client.observations == []
+ assert not (outside_trial / L.RECEIPT_NAME).exists()
+
+
+def test_export_rejects_a_nonregular_verifier_entry(tmp_path):
+ """Silently skipping an agent-controlled pipe would make exported evidence partial."""
+ trial = copy_trial(tmp_path)
+ os.mkfifo(trial / "verifier" / "solution" / "stream.txt")
+ client = FakeLangfuse()
+
+ with pytest.raises(L.TelemetryReceiptError, match="non-regular"):
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert client.observations == []
+ assert not (trial / L.RECEIPT_NAME).exists()
+
+
+def test_export_rejects_an_intermediate_directory_swap(tmp_path, monkeypatch):
+ """A renamed solution directory cannot redirect a later open outside the attempt."""
+ trial = copy_trial(tmp_path)
+ solution = trial / "verifier" / "solution"
+ artifact = solution / "answer.txt"
+ artifact.write_text("inside fixture text")
+ outside = tmp_path / "outside-solution"
+ outside.mkdir()
+ (outside / "answer.txt").write_text("PARENT-ONLY-SECRET")
+ original_open = os.open
+ swapped = False
+
+ def swap_before_artifact_open(path, flags, *args, **kwargs):
+ nonlocal swapped
+ if not swapped and Path(path).name == "answer.txt":
+ swapped = True
+ held = solution.with_name("held-solution")
+ solution.rename(held)
+ solution.symlink_to(outside, target_is_directory=True)
+ return original_open(path, flags, *args, **kwargs)
+
+ monkeypatch.setattr(os, "open", swap_before_artifact_open)
+ client = FakeLangfuse()
+
+ with pytest.raises(L.TelemetryReceiptError, match="changed|admission"):
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert client.observations == []
+ assert not (trial / L.RECEIPT_NAME).exists()
+
+
+def test_export_projects_many_small_artifacts_and_hidden_housekeeping_deterministically(tmp_path):
+ """Optional telemetry stays bounded without rejecting authoritative Harbor evidence."""
+ trial = copy_trial(tmp_path)
+ solution = trial / "verifier" / "solution"
+ hidden_target = tmp_path / "hidden-backup"
+ hidden_target.mkdir()
+ (hidden_target / "secret.txt").write_text("must-not-export")
+ (solution / ".backups").symlink_to(hidden_target, target_is_directory=True)
+ for index in range(129):
+ (solution / f"part-{index:03}.txt").write_text("x")
+ client = FakeLangfuse()
+
+ receipts = L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert len(receipts) == len(client.observations) == 1
+ payload = client.observations[0]["output"]
+ exported = payload["solution_artifacts"]
+ assert "part-000.txt" in exported
+ assert "part-128.txt" not in exported
+ assert ".backups" not in json.dumps(payload)
+ assert "must-not-export" not in json.dumps(payload)
+ assert payload["artifact_projection"]["omitted_hidden_paths"] == 1
+ assert payload["artifact_projection"]["omitted_files"] >= 1
+ assert payload["artifact_projection"]["exported_files"] <= L._MAX_ARTIFACT_PATHS
+ assert (trial / L.RECEIPT_NAME).is_file()
+
+
+def test_export_projects_aggregate_artifact_bytes_above_the_attempt_budget(tmp_path):
+ """Optional files beyond the byte ceiling are omitted without blocking evidence."""
+ trial = copy_trial(tmp_path)
+ solution = trial / "verifier" / "solution"
+ for index in range(80):
+ (solution / f"chunk-{index:03}.txt").write_text("x" * 2000)
+ client = FakeLangfuse()
+
+ receipts = L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert len(receipts) == len(client.observations) == 1
+ payload = client.observations[0]["output"]
+ assert payload["artifact_projection"]["exported_bytes"] <= L._MAX_ARTIFACT_BYTES
+ assert payload["artifact_projection"]["omitted_files"] > 0
+ assert (trial / L.RECEIPT_NAME).is_file()
+
+
+def test_export_prioritizes_solution_answer_over_verifier_noise(tmp_path):
+ trial = copy_trial(tmp_path)
+ verifier = trial / "verifier"
+ solution = verifier / "solution"
+ (solution / "answer.md").write_text("graded deliverable")
+ for index in range(130):
+ (verifier / f"noise-{index:03}.txt").write_text("noise")
+ client = FakeLangfuse()
+
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ payload = client.observations[0]["output"]
+ assert payload["solution_artifacts"]["answer.md"] == "graded deliverable"
+ assert client.observations[0]["metadata"]["exporter_revision"] == "harbor-langfuse-v3"
+
+
+def test_export_migrates_verified_v2_receipt_without_rerunning_attempt(tmp_path):
+ trial = copy_trial(tmp_path)
+ client = FakeLangfuse()
+ old_digest = "a" * 64
+ old_trace = Langfuse.create_trace_id(seed=f"harbor-langfuse-v2:{old_digest}")
+ evidence = L._telemetry_evidence(L.build_attempt_payload(trial, METADATA))
+ old_receipt = {
+ "status": "verified", "trace_id": old_trace, "payload_sha256": old_digest,
+ "exporter_revision": "harbor-langfuse-v2",
+ }
+ (trial / L.RECEIPT_NAME).write_text(json.dumps(old_receipt))
+ client.observation.id = "old-observation"
+ client.observations.append({"metadata": {
+ "payload_sha256": old_digest, "exporter_revision": "harbor-langfuse-v2",
+ "telemetry_evidence": evidence,
+ }})
+ default_read = client.read_trace
+ new_reads = 0
+
+ def versioned_read(trace_id):
+ nonlocal new_reads
+ if trace_id == old_trace:
+ return {"id": old_trace, "observations": [{
+ "id": "old-observation", "name": "harbor-attempt", "type": "AGENT",
+ "metadata": client.observations[0]["metadata"],
+ }]}
+ if len(client.observations) == 1:
+ new_reads += 1
+ if new_reads == 1:
+ raise TimeoutError("transient v3 read failure")
+ return None
+ return default_read(trace_id)
+
+ client.read_trace = versioned_read
+
+ with pytest.raises(TimeoutError, match="transient v3 read failure"):
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert json.loads((trial / L.RECEIPT_NAME).read_text()) == old_receipt
+ assert not (trial / L.LEGACY_V2_RECEIPT_NAME).exists()
+
+ [receipt] = L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert json.loads((trial / L.LEGACY_V2_RECEIPT_NAME).read_text()) == old_receipt
+ assert receipt["status"] == "verified"
+ assert receipt["exporter_revision"] == "harbor-langfuse-v3"
+ assert json.loads((trial / L.RECEIPT_NAME).read_text()) == receipt
+ assert len(client.observations) == 2
+
+
+def test_export_projects_scan_and_depth_exhaustion(tmp_path, monkeypatch):
+ trial = copy_trial(tmp_path)
+ solution = trial / "verifier" / "solution"
+ nested = solution
+ for index in range(4):
+ nested = nested / f"level-{index}"
+ nested.mkdir()
+ (nested / "deep.txt").write_text("deep")
+ (solution / "first.txt").write_text("first")
+ (solution / "second.txt").write_text("second")
+ monkeypatch.setattr(L, "_MAX_ARTIFACT_DEPTH", 2)
+ monkeypatch.setattr(L, "_MAX_ARTIFACT_SCAN_PATHS", 6)
+ client = FakeLangfuse()
+
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ projection = client.observations[0]["output"]["artifact_projection"]
+ assert projection["omitted_files"] > 0
+ assert len(client.observations) == 1
+
+
+def test_payload_uses_trajectory_metrics_when_result_usage_is_unavailable(tmp_path):
+ """Dropping Harbor's alternate usage shape would erase usage from successful adapters."""
+ trial = copy_trial(tmp_path)
+ result_path = trial / "result.json"
+ result = json.loads(result_path.read_text())
+ result["agent_result"] = {
+ "n_input_tokens": None, "n_cache_tokens": None,
+ "n_output_tokens": None, "cost_usd": None,
+ }
+ result_path.write_text(json.dumps(result))
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["final_metrics"] = {
+ "total_prompt_tokens": 80,
+ "total_cached_tokens": 25,
+ "total_completion_tokens": 20,
+ "total_cost_usd": 0.005,
+ }
+ trajectory_path.write_text(json.dumps(trajectory))
+
+ assert L.build_attempt_payload(trial, METADATA)["usage"] == {
+ "input_tokens": 80,
+ "cache_tokens": 25,
+ "output_tokens": 20,
+ "cost_usd": 0.005,
+ }
+
+
+def test_canonical_payload_and_hash_do_not_depend_on_parent_secret_environment(tmp_path, monkeypatch):
+ """A credential rotation must not change the identity of unchanged persisted evidence."""
+ trial = copy_trial(tmp_path)
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["steps"][0]["message"] = "rotated-value"
+ trajectory_path.write_text(json.dumps(trajectory))
+ (trial / "verifier" / "solution" / "answer.txt").write_text("rotated-value")
+ monkeypatch.delenv("CUSTOM_API_KEY", raising=False)
+ first = L.build_attempt_payload(trial, PROVENANCE_METADATA)
+ first_hash = payload_sha256(first)
+ monkeypatch.setenv("CUSTOM_API_KEY", "rotated-value")
+
+ second = L.build_attempt_payload(trial, PROVENANCE_METADATA)
+
+ assert second == first
+ assert payload_sha256(second) == first_hash
+
+
+def test_known_skill_body_is_omitted_from_trajectory_and_solution_artifacts(tmp_path):
+ """Known instruction content must not be exported as attempt evidence."""
+ trial = copy_trial(tmp_path)
+ skill_body = "PRIVATE SKILL BODY\n"
+ metadata = {
+ **PROVENANCE_METADATA,
+ "skill_body": skill_body,
+ "skill_sha256": hashlib.sha256(skill_body.encode()).hexdigest(),
+ }
+ trajectory_path = trial / "agent" / "trajectory.json"
+ trajectory = json.loads(trajectory_path.read_text())
+ trajectory["steps"][0]["message"] = skill_body
+ trajectory_path.write_text(json.dumps(trajectory))
+ (trial / "verifier" / "solution" / "skill.md").write_text(skill_body)
+
+ payload = L.build_attempt_payload(trial, metadata)
+
+ assert "PRIVATE SKILL BODY" not in json.dumps(payload)
+ assert "message" not in json.dumps(payload["trajectory"])
+ assert "skill.md" not in payload["solution_artifacts"]
+
+
+def test_named_sensitive_artifact_classes_are_structurally_excluded(tmp_path):
+ """Known config, environment, credential, secret, and endpoint dumps never export."""
+ trial = copy_trial(tmp_path)
+ solution = trial / "verifier" / "solution"
+ for name in (
+ "config.json", "configuration.txt", "environment.json", "env.txt",
+ "credential.txt", "credentials.log", "secret.txt", "secrets.yaml",
+ "endpoint.json", "endpoints.txt",
+ ):
+ (solution / name).write_text(f"private marker from {name}")
+
+ payload = L.build_attempt_payload(trial, METADATA)
+
+ rendered = json.dumps(payload["solution_artifacts"])
+ assert "private marker" not in rendered
+
+
+def test_artifact_wrapping_the_verified_known_skill_body_is_excluded(tmp_path):
+ """Adding a heading cannot disguise the exact producer-supplied skill body."""
+ trial = copy_trial(tmp_path)
+ skill_body = "PRIVATE SKILL BODY\n"
+ metadata = {
+ **PROVENANCE_METADATA,
+ "skill_body": skill_body,
+ "skill_sha256": hashlib.sha256(skill_body.encode()).hexdigest(),
+ }
+ wrapped = trial / "verifier" / "solution" / "wrapped.md"
+ wrapped.write_text(f"# Copied instructions\n\n{skill_body}\nFixture answer")
+
+ payload = L.build_attempt_payload(trial, metadata)
+
+ assert "wrapped.md" not in payload["solution_artifacts"]
+ assert "PRIVATE SKILL BODY" not in json.dumps(payload)
+
+
+def _shape_trial(tmp_path: Path, shape: str) -> Path:
+ trial = tmp_path / shape / "fixture-h0__attempt-1"
+ trial.mkdir(parents=True)
+ if shape in {"success", "failure"}:
+ result = {
+ "id": f"fixture-{shape}",
+ "task_name": "ingot/fixture-h0",
+ "trial_name": "fixture-h0__attempt-1",
+ "task_checksum": "fixture-checksum",
+ "source": "fixture",
+ "started_at": "2000-01-01T00:00:00Z",
+ "finished_at": "2000-01-01T00:00:01Z",
+ "exception_info": None,
+ }
+ if shape == "failure":
+ result["exception_info"] = {
+ "exception_type": "RuntimeError",
+ "exception_message": "bounded fixture failure",
+ "exception_traceback": "private traceback must not export",
+ "occurred_at": "2000-01-01T00:00:00.500Z",
+ }
+ (trial / "result.json").write_text(json.dumps(result))
+ elif shape == "trajectory-only":
+ trajectory = trial / "agent" / "trajectory.json"
+ trajectory.parent.mkdir()
+ trajectory.write_text(json.dumps({
+ "schema_version": "ATIF-v1.1",
+ "session_id": "fixture-session",
+ "agent": {"name": "codex", "version": "v1", "model_name": "fixture-model"},
+ "steps": [
+ {"step_id": 1, "timestamp": "2000-01-01T00:00:00Z",
+ "source": "agent", "message": "private free text"},
+ {"step_id": 2, "timestamp": "2000-01-01T00:00:01Z",
+ "source": "agent", "message": "private free text"},
+ ],
+ }))
+ elif shape == "verifier-only":
+ verifier = trial / "verifier" / "test-stdout.txt"
+ verifier.parent.mkdir()
+ verifier.write_text("fixture verifier output")
+ elif shape == "solution-only":
+ solution = trial / "verifier" / "solution" / "answer.md"
+ solution.parent.mkdir(parents=True)
+ solution.write_text("fixture solution output")
+ return trial
+
+
+@pytest.mark.parametrize("shape, expected", [
+ ("success", {
+ "status": "succeeded", "error_detail": None,
+ "timestamps": {"started_at": "2000-01-01T00:00:00Z",
+ "finished_at": "2000-01-01T00:00:01Z"},
+ }),
+ ("failure", {
+ "status": "failed", "error_detail": "bounded fixture failure",
+ "timestamps": {"started_at": "2000-01-01T00:00:00Z",
+ "finished_at": "2000-01-01T00:00:01Z",
+ "error_at": "2000-01-01T00:00:00.500Z"},
+ }),
+ ("trajectory-only", {
+ "status": "incomplete", "error_detail": None,
+ "timestamps": {"started_at": "2000-01-01T00:00:00Z",
+ "finished_at": "2000-01-01T00:00:01Z"},
+ }),
+ ("verifier-only", {"status": "incomplete", "error_detail": None, "timestamps": {}}),
+ ("solution-only", {"status": "incomplete", "error_detail": None, "timestamps": {}}),
+])
+def test_attempt_shapes_have_exact_terminal_status_and_provenance(tmp_path, shape, expected):
+ """Missing terminal result evidence must never be promoted to a successful status."""
+ payload = L.build_attempt_payload(_shape_trial(tmp_path, shape), PROVENANCE_METADATA)
+
+ assert {key: payload[key] for key in ("status", "error_detail", "timestamps")} == expected
+ assert payload["skill"] == "demo"
+ assert payload["task"] == {
+ "name": "ingot/fixture-h0" if shape in {"success", "failure"} else "fixture-h0",
+ "checksum": "fixture-checksum" if shape in {"success", "failure"} else "",
+ "source": "fixture" if shape in {"success", "failure"} else "",
+ "text": "Use only deterministic fixture input.",
+ }
+
+
+def test_export_writes_receipt_only_after_deterministic_trace_readback(tmp_path):
+ """A wrong trace seed, observation contract, or unverified receipt must fail this test."""
+ trial = copy_trial(tmp_path)
+ client = FakeLangfuse()
+
+ receipts = L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ payload = L.build_attempt_payload(trial, METADATA)
+ digest = payload_sha256(payload)
+ trace_id = Langfuse.create_trace_id(seed=f"harbor-langfuse-v3:{digest}")
+ expected = {
+ "status": "verified",
+ "trace_id": trace_id,
+ "payload_sha256": digest,
+ "exporter_revision": "harbor-langfuse-v3",
+ }
+ assert receipts == [expected]
+ assert json.loads((trial / "langfuse-receipt.json").read_text()) == expected
+ assert client.seeds == [f"harbor-langfuse-v3:{digest}"]
+ assert client.flushes == 1 and client.reads == [expected["trace_id"], expected["trace_id"]]
+ assert client.observation.ended == 1
+ observation = client.observations[0]
+ assert observation["trace_context"] == {"trace_id": expected["trace_id"]}
+ assert observation["name"] == "harbor-attempt" and observation["as_type"] == "agent"
+ assert observation["input"] == payload["task"]
+ assert observation["output"]["status"] == "failed"
+ assert observation["metadata"]["payload_sha256"] == digest
+ assert observation["metadata"]["telemetry_evidence"] == {
+ "revision": "harbor-agent-evidence-v1",
+ "model": "fixture-model",
+ "status": "failed",
+ "tokens": {"input_tokens": 120, "cache_tokens": 40, "output_tokens": 30},
+ }
+ assert "model" not in observation
+ assert "usage_details" not in observation
+ assert "cost_details" not in observation
+ assert observation["level"] == "ERROR" and observation["status_message"] == "failed"
+
+
+def test_real_langfuse_agent_sink_uses_sdk_ids_for_two_payloads_with_existing_provider(tmp_path):
+ """Langfuse 4.14 owns span IDs while bounded agent metadata survives its real agent path."""
+ from opentelemetry.sdk.trace import TracerProvider
+ from opentelemetry.sdk.trace.export import SimpleSpanProcessor
+ from opentelemetry.sdk.trace.export.in_memory_span_exporter import InMemorySpanExporter
+
+ first = copy_trial(tmp_path)
+ second = first.parent / "fixture-h1__attempt-2"
+ shutil.copytree(FIXTURE, second)
+ result_path = second / "result.json"
+ result = json.loads(result_path.read_text())
+ result["id"] = "00000000-0000-0000-0000-000000000002"
+ result["task_name"] = "ingot/fixture-h1"
+ result["trial_name"] = second.name
+ result_path.write_text(json.dumps(result))
+ preexisting_sink = InMemorySpanExporter()
+ provider = TracerProvider()
+ provider.add_span_processor(SimpleSpanProcessor(preexisting_sink))
+ sink = InMemorySpanExporter()
+ sdk = Langfuse(
+ public_key="pk-task4-real-sink-two-payloads",
+ secret_key="sk-offline-fixture",
+ tracer_provider=provider,
+ span_exporter=sink,
+ )
+
+ class SinkReadback:
+ def create_trace_id(self, **kwargs):
+ return sdk.create_trace_id(**kwargs)
+
+ def start_observation(self, **kwargs):
+ return sdk.start_observation(**kwargs)
+
+ def flush(self):
+ sdk.flush()
+
+ def read_trace(self, trace_id):
+ observations = []
+ for span in sink.get_finished_spans():
+ if f"{span.context.trace_id:032x}" != trace_id:
+ continue
+ metadata = {}
+ prefix = "langfuse.observation.metadata."
+ for key, value in span.attributes.items():
+ if key.startswith(prefix):
+ try:
+ metadata[key.removeprefix(prefix)] = json.loads(value)
+ except (TypeError, ValueError):
+ metadata[key.removeprefix(prefix)] = value
+ observations.append({
+ "id": f"{span.context.span_id:016x}",
+ "name": span.name,
+ "type": span.attributes["langfuse.observation.type"],
+ "metadata": metadata,
+ })
+ return {"id": trace_id, "observations": observations} if observations else None
+
+ receipts = L.export_job_attempts(first.parent, METADATA, client=SinkReadback())
+
+ spans = sink.get_finished_spans()
+ assert len(receipts) == 2 and all(item["status"] == "verified" for item in receipts)
+ assert len(spans) == 2
+ assert len(preexisting_sink.get_finished_spans()) == 2
+ assert len({span.context.span_id for span in spans}) == 2
+ for span in spans:
+ assert span.attributes["langfuse.observation.type"] == "agent"
+ evidence = json.loads(span.attributes["langfuse.observation.metadata.telemetry_evidence"])
+ assert evidence == {
+ "revision": "harbor-agent-evidence-v1",
+ "model": "fixture-model",
+ "status": "failed",
+ "tokens": {"input_tokens": 120, "cache_tokens": 40, "output_tokens": 30},
+ }
+ assert "langfuse.observation.model.name" not in span.attributes
+ assert "langfuse.observation.usage_details" not in span.attributes
+ assert "langfuse.observation.cost_details" not in span.attributes
+
+
+def test_repeated_export_reuses_the_verified_receipt_without_another_trace(tmp_path):
+ """Dropping receipt reuse would duplicate a deterministic attempt in Langfuse."""
+ trial = copy_trial(tmp_path)
+ first = FakeLangfuse()
+ receipt = L.export_job_attempts(trial.parent, METADATA, client=first)
+ must_not_export = FakeLangfuse(readback="absent")
+
+ assert L.export_job_attempts(trial.parent, METADATA, client=must_not_export) == receipt
+ assert must_not_export.observations == []
+ assert must_not_export.flushes == 0 and must_not_export.reads == []
+
+
+def test_production_readback_uses_parent_basic_auth_and_public_trace_endpoint(tmp_path, monkeypatch):
+ """Wrong auth or endpoint would make an SDK enqueue look verified without public read-back."""
+ trial = copy_trial(tmp_path)
+ client = FakeSdkOnly()
+ requests = []
+ monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-fixture")
+ monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-fixture")
+ monkeypatch.setenv("LANGFUSE_BASE_URL", "https://langfuse.fixture.invalid/")
+
+ class Response:
+ status_code = 200
+
+ def json(self):
+ return {
+ "id": client.delegate.seeds[-1] and Langfuse.create_trace_id(
+ seed=client.delegate.seeds[-1]),
+ "observations": [{"id": client.delegate.observation.id,
+ "name": "harbor-attempt", "type": "AGENT",
+ "metadata": client.observations[-1]["metadata"]}],
+ }
+
+ class MissingResponse:
+ status_code = 404
+
+ def fake_get(url, **kwargs):
+ requests.append((url, kwargs))
+ return Response() if client.observations else MissingResponse()
+
+ monkeypatch.setattr(L.httpx, "get", fake_get)
+
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ trace_id = Langfuse.create_trace_id(seed=client.delegate.seeds[-1])
+ assert requests == [
+ (f"https://langfuse.fixture.invalid/api/public/traces/{trace_id}",
+ {"auth": ("pk-fixture", "sk-fixture"), "timeout": 15}),
+ (f"https://langfuse.fixture.invalid/api/public/traces/{trace_id}",
+ {"auth": ("pk-fixture", "sk-fixture"), "timeout": 15}),
+ ]
+
+
+def test_production_readback_error_fails_before_emitting_an_observation(tmp_path, monkeypatch):
+ """Only a confirmed missing deterministic trace permits a first observation."""
+ trial = copy_trial(tmp_path)
+ client = FakeSdkOnly()
+ monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-fixture")
+ monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-fixture")
+
+ class UnauthorizedResponse:
+ status_code = 401
+
+ monkeypatch.setattr(L.httpx, "get", lambda *args, **kwargs: UnauthorizedResponse())
+
+ with pytest.raises(L.TelemetryReceiptError, match="status 401"):
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert client.observations == []
+
+
+@pytest.mark.parametrize("readback", ["absent", "wrong-id", "wrong-hash", "wrong-evidence"])
+def test_unmatched_public_readback_never_creates_a_verified_receipt(tmp_path, readback):
+ """Accepting absent or mismatched public readback would turn pending state into proof."""
+ trial = copy_trial(tmp_path)
+
+ with pytest.raises(L.TelemetryReceiptError, match="read-back"):
+ L.export_job_attempts(trial.parent, METADATA, client=FakeLangfuse(readback=readback))
+
+ receipt = json.loads((trial / "langfuse-receipt.json").read_text())
+ assert receipt["status"] == "pending"
+
+
+def test_failed_attempt_is_traced_and_telemetry_failure_does_not_mutate_harbor_result(tmp_path):
+ """Failure telemetry must describe persisted evidence, never rewrite or rerun it."""
+ trial = copy_trial(tmp_path)
+ before = (trial / "result.json").read_bytes()
+ client = FakeLangfuse(readback="absent")
+
+ with pytest.raises(L.TelemetryReceiptError):
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert client.observations[0]["output"]["status"] == "failed"
+ assert (trial / "result.json").read_bytes() == before
+ assert json.loads((trial / "result.json").read_text())["verifier_result"]["rewards"] == {"reward": 1.0}
+
+
+@pytest.mark.parametrize("state", [
+ "empty", "withheld", "stale", "malformed", "empty-id", "whitespace-id", "wrong-id",
+])
+def test_receipt_validation_rejects_every_incomplete_telemetry_state(tmp_path, state):
+ """No empty, withheld, stale, or malformed receipt may silently publish a matrix."""
+ trial = copy_trial(tmp_path)
+ receipt = trial / "langfuse-receipt.json"
+ if state == "withheld":
+ receipt.write_text(json.dumps({"status": "withheld"}))
+ elif state == "stale":
+ receipt.write_text(json.dumps({
+ "status": "verified",
+ "trace_id": "1234567890abcdef1234567890abcdef",
+ "payload_sha256": "0" * 64,
+ "exporter_revision": "harbor-langfuse-v1",
+ }))
+ elif state == "malformed":
+ receipt.write_text("{not-json")
+ elif state in {"empty-id", "whitespace-id", "wrong-id"}:
+ payload = L.build_attempt_payload(trial, METADATA)
+ receipt.write_text(json.dumps({
+ "status": "verified",
+ "trace_id": {"empty-id": "", "whitespace-id": " ", "wrong-id": "0" * 32}[state],
+ "payload_sha256": payload_sha256(payload),
+ "exporter_revision": "harbor-langfuse-v1",
+ }))
+
+ with pytest.raises(L.TelemetryReceiptError, match="receipt"):
+ L.validate_job_receipts(trial.parent, METADATA)
+
+
+def test_receipt_validation_rejects_an_exact_provenance_free_receipt(tmp_path):
+ """A matching old payload identity cannot authorize publication without provenance."""
+ trial = copy_trial(tmp_path)
+ payload = L.build_attempt_payload(trial, METADATA)
+ digest = payload_sha256(payload)
+ (trial / L.RECEIPT_NAME).write_text(json.dumps({
+ "status": "verified",
+ "trace_id": Langfuse.create_trace_id(seed=f"{L.EXPORTER_REVISION}:{digest}"),
+ "payload_sha256": digest,
+ "exporter_revision": L.EXPORTER_REVISION,
+ }))
+
+ with pytest.raises(L.TelemetryReceiptError, match="provenance"):
+ L.validate_job_receipts(trial.parent, METADATA)
+
+
+class DelayedReadbackLangfuse(FakeLangfuse):
+ """First post-flush read misses; the next preflight sees the persisted observation."""
+
+ def read_trace(self, trace_id: str):
+ self.reads.append(trace_id)
+ if not self.observations or (len(self.observations) == 1 and len(self.reads) < 3):
+ return None
+ metadata = dict(self.observations[0]["metadata"])
+ return {
+ "id": trace_id,
+ "observations": [{"id": self.observation.id,
+ "name": "harbor-attempt", "type": "AGENT",
+ "metadata": metadata}],
+ }
+
+
+def test_post_flush_readback_polls_before_deferring_finalization(tmp_path, monkeypatch):
+ """A short indexing lag must finalize in one exporter pass without a duplicate trace."""
+ trial = copy_trial(tmp_path)
+ client = DelayedReadbackLangfuse()
+ sleeps = []
+ monkeypatch.setattr("time.sleep", sleeps.append)
+
+ receipts = L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert len(receipts) == 1
+ assert len(client.observations) == 1
+ assert client.flushes == 1
+ assert len(client.reads) == 3
+ assert sleeps == [L._READBACK_POLL_SECONDS]
+
+
+def test_repeated_404_after_unknown_dispatch_never_emits_a_second_observation(tmp_path):
+ """A point-in-time missing trace cannot authorize another dispatch after flush."""
+ trial = copy_trial(tmp_path)
+ client = FakeLangfuse(readback="absent")
+
+ for _attempt in range(2):
+ with pytest.raises(L.TelemetryReceiptError):
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ pending = json.loads((trial / L.RECEIPT_NAME).read_text())
+ assert pending["status"] == "pending"
+ assert pending["observation_id"] == client.observation.id
+ assert len(client.observations) == 1
+ with pytest.raises(L.TelemetryReceiptError):
+ L.validate_job_receipts(trial.parent, METADATA)
+
+
+def test_unknown_pending_receipt_never_adopts_an_uncaptured_observation_id(tmp_path):
+ """A crash before persisting the SDK ID leaves non-authorizing state, even after read-back."""
+ trial = copy_trial(tmp_path)
+ payload = L.build_attempt_payload(trial, METADATA)
+ digest = payload_sha256(payload)
+ evidence = L._telemetry_evidence(payload)
+ (trial / L.RECEIPT_NAME).write_text(json.dumps(L._pending_receipt(digest)))
+
+ class UnknownPendingLangfuse(FakeLangfuse):
+ def read_trace(self, trace_id: str):
+ return {"id": trace_id, "observations": [{
+ "id": "2222222222222222",
+ "name": "harbor-attempt",
+ "type": "AGENT",
+ "metadata": {
+ "payload_sha256": digest,
+ "exporter_revision": L.EXPORTER_REVISION,
+ "telemetry_evidence": evidence,
+ },
+ }]}
+
+ def start_observation(self, **kwargs):
+ pytest.fail("an unknown pending attempt must never emit")
+
+ with pytest.raises(L.TelemetryReceiptError, match="pending"):
+ L.export_job_attempts(trial.parent, METADATA, client=UnknownPendingLangfuse())
+
+ assert json.loads((trial / L.RECEIPT_NAME).read_text()) == L._pending_receipt(digest)
+
+
+def test_concurrent_receiptless_exporters_create_at_most_one_observation(tmp_path):
+ """Atomic pending state serializes contenders that both observed the trace missing."""
+ trial = copy_trial(tmp_path)
+
+ class ConcurrentAbsentLangfuse(FakeLangfuse):
+ def __init__(self):
+ super().__init__(readback="absent")
+ self.preflight = threading.Barrier(2)
+ self.read_count = 0
+ self.read_lock = threading.Lock()
+
+ def read_trace(self, trace_id: str):
+ with self.read_lock:
+ self.read_count += 1
+ count = self.read_count
+ if count <= 2:
+ self.preflight.wait(timeout=5)
+ self.reads.append(trace_id)
+ return None
+
+ client = ConcurrentAbsentLangfuse()
+
+ def export():
+ with pytest.raises(L.TelemetryReceiptError):
+ L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ with ThreadPoolExecutor(max_workers=2) as pool:
+ list(pool.map(lambda _index: export(), range(2)))
+
+ pending = json.loads((trial / L.RECEIPT_NAME).read_text())
+ assert pending["status"] == "pending"
+ assert pending["observation_id"] == client.observation.id
+ assert len(client.observations) == 1
+
+
+def test_duplicate_preexisting_observations_fail_closed_without_another_emission(tmp_path):
+ """A trace with duplicate matching observations is not one verified attempt."""
+ trial = copy_trial(tmp_path)
+ payload = L.build_attempt_payload(trial, METADATA)
+ digest = payload_sha256(payload)
+
+ class DuplicateLangfuse(FakeLangfuse):
+ def read_trace(self, trace_id: str):
+ metadata = {"payload_sha256": digest, "exporter_revision": L.EXPORTER_REVISION}
+ observation = {"id": "1111111111111111",
+ "name": "harbor-attempt", "type": "AGENT", "metadata": metadata}
+ return {"id": trace_id, "observations": [observation, observation]}
+
+ def start_observation(self, **kwargs):
+ pytest.fail("duplicate preexisting trace must not emit")
+
+ with pytest.raises(L.TelemetryReceiptError, match="existing"):
+ L.export_job_attempts(trial.parent, METADATA, client=DuplicateLangfuse())
+
+
+def test_evidence_root_exports_preserved_attempts_without_rerunning_agents(tmp_path):
+ """Changing recursive evidence discovery must not orphan preserved proprietary attempts."""
+ trial = copy_trial(tmp_path)
+ combo = trial.parents[1]
+ (combo / "combo.json").write_text(json.dumps(PROVENANCE_METADATA))
+ (trial / "result.json").unlink()
+ client = FakeLangfuse()
+
+ receipts = L.export_evidence_root(tmp_path, client=client)
+
+ assert len(receipts) == 1
+ assert (trial / "langfuse-receipt.json").is_file()
+ assert len(client.observations) == 1
+ assert client.observations[0]["input"] == {
+ "name": "fixture-h0", "checksum": "", "source": "",
+ "text": "Use only deterministic fixture input.",
+ }
+ assert client.observations[0]["output"]["skill"] == "demo"
+ assert client.observations[0]["output"]["timestamps"] == {
+ "started_at": "2000-01-01T00:00:00Z",
+ "finished_at": "2000-01-01T00:00:01Z",
+ }
+
+
+def test_evidence_root_refuses_missing_explicit_provenance_before_emission(tmp_path):
+ """Preserved evidence cannot infer skill or task text from directory names."""
+ trial = copy_trial(tmp_path)
+ (trial.parents[1] / "combo.json").write_text(json.dumps(METADATA))
+ client = FakeLangfuse()
+
+ with pytest.raises(L.TelemetryReceiptError, match="provenance"):
+ L.export_evidence_root(tmp_path, client=client)
+
+ assert client.observations == []
+ assert not (trial / L.RECEIPT_NAME).exists()
+
+
+def test_export_discovers_a_verifier_only_failed_attempt(tmp_path):
+ """A failed trial that retained only verifier evidence is still one persisted attempt."""
+ trial = tmp_path / "job" / "fixture-h1__attempt-2"
+ solution = trial / "verifier" / "solution"
+ solution.mkdir(parents=True)
+ (solution / "answer.txt").write_text("partial fixture answer")
+ client = FakeLangfuse()
+
+ receipts = L.export_job_attempts(trial.parent, METADATA, client=client)
+
+ assert len(receipts) == 1 and len(client.observations) == 1
+ assert client.observations[0]["input"]["name"] == "fixture-h1"
+ assert (trial / "langfuse-receipt.json").is_file()
+
+
+def test_cli_exports_each_repeated_evidence_root(tmp_path, monkeypatch):
+ """Dropping repeated roots would leave part of the preserved evidence unreceipted."""
+ first, second = tmp_path / "first", tmp_path / "second"
+ first_metadata, second_metadata = tmp_path / "first.json", tmp_path / "second.json"
+ first_metadata.write_text(json.dumps(PROVENANCE_METADATA))
+ second_metadata.write_text(json.dumps(PROVENANCE_METADATA))
+ calls = []
+ monkeypatch.setattr(
+ L, "export_evidence_root",
+ lambda root, metadata=None: calls.append((root, metadata)) or [],
+ )
+
+ assert L.main([
+ "--root", str(first), "--metadata", str(first_metadata),
+ "--root", str(second), "--metadata", str(second_metadata),
+ ]) == 0
+ assert calls == [(first, PROVENANCE_METADATA), (second, PROVENANCE_METADATA)]
+
+
+def test_cli_refuses_a_preserved_root_without_paired_migration_metadata(tmp_path):
+ """A root and its migration document are an explicit one-to-one input."""
+ with pytest.raises(SystemExit):
+ L.main(["--root", str(tmp_path / "preserved")])
+
+
+def test_export_job_attempts_filters_native_arm_identity(tmp_path):
+ from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env
+
+ skill_trial = copy_trial(tmp_path)
+ control_trial = tmp_path / "job" / "fixture-h0__attempt-2"
+ shutil.copytree(FIXTURE, control_trial)
+ common = dict(combination_id="codex@fixture-model--dell-fixture-ffffffffffff",
+ endpoint_fingerprint="ffffffffffff", harness="codex",
+ protocol="responses", gateway_revision="direct")
+ skill = NativeTrialIdentity(**common, arm="skill")
+ control = NativeTrialIdentity(**common, arm="control")
+ (skill_trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(skill)}}))
+ (control_trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(control)}}))
+ client = FakeLangfuse()
+
+ receipts = L.export_job_attempts(skill_trial.parent, METADATA, identity=skill, client=client)
+
+ assert len(receipts) == 1
+ assert (skill_trial / L.RECEIPT_NAME).is_file()
+ assert not (control_trial / L.RECEIPT_NAME).exists()
+
+
+def test_identity_filter_skips_unrelated_rejected_artifacts_before_opening_them(tmp_path):
+ from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env
+
+ selected = copy_trial(tmp_path)
+ other = tmp_path / "job" / "fixture-h0__attempt-2"
+ shutil.copytree(FIXTURE, other)
+ common = dict(combination_id="codex@fixture-model--dell-fixture-ffffffffffff",
+ endpoint_fingerprint="ffffffffffff", harness="codex",
+ protocol="responses", gateway_revision="direct")
+ skill = NativeTrialIdentity(**common, arm="skill")
+ control = NativeTrialIdentity(**common, arm="control")
+ (selected / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(skill)}}))
+ (other / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(control)}}))
+ shutil.rmtree(other / "verifier")
+ (other / "verifier").symlink_to(tmp_path, target_is_directory=True)
+
+ attempts = L._job_attempts(selected.parent, skill)
+
+ assert attempts == [selected]
diff --git a/tests/test_harbor_native.py b/tests/test_harbor_native.py
new file mode 100644
index 0000000..534cb92
--- /dev/null
+++ b/tests/test_harbor_native.py
@@ -0,0 +1,199 @@
+import json
+from pathlib import Path
+
+import pytest
+
+from ingot.optimize.harbor_native import (NATIVE_RUNNER_REVISION, NativeCell, NativeTrialIdentity,
+ compile_agent_config, compile_measurement_job, identity_env,
+ read_trial_identity)
+from ingot.optimize.harbor_targets import LocalTarget
+
+
+def _target() -> LocalTarget:
+ return LocalTarget(
+ alias="dell-qwen",
+ display_name="Qwen/Qwen3.6-27B",
+ base_url="http://127.0.0.1:8011",
+ served_model="dot-backbone",
+ context_length=163_840,
+ protocols=frozenset({"chat", "messages", "responses"}),
+ family="Qwen3.6",
+ parameter_billions=27.0,
+ quantization="fp8-published",
+ tool_parser="qwen3_xml",
+ )
+
+
+def test_native_runner_revision_encodes_exact_trial_memory_limit():
+ assert NATIVE_RUNNER_REVISION == "native-v4-memory2048mb"
+
+
+def _identity(arm: str = "skill") -> NativeTrialIdentity:
+ return NativeTrialIdentity(
+ combination_id="aider@dot-backbone--dell-qwen-deadbeefcafe",
+ endpoint_fingerprint="deadbeefcafe",
+ harness="aider",
+ protocol="chat",
+ gateway_revision="direct",
+ arm=arm,
+ )
+
+
+def _write_lock(path: Path, env: dict[str, str]) -> None:
+ path.mkdir()
+ (path / "lock.json").write_text(json.dumps({"agent": {"env": env}}))
+
+
+def test_trial_identity_round_trips_through_harbor_agent_lock(tmp_path):
+ skill = _identity("skill")
+ control = _identity("control")
+ skill_dir = tmp_path / "skill"
+ control_dir = tmp_path / "control"
+ _write_lock(skill_dir, {"OPENAI_API_KEY": "${OPENAI_API_KEY}", **identity_env(skill)})
+ _write_lock(control_dir, {"OPENAI_API_KEY": "${OPENAI_API_KEY}", **identity_env(control)})
+
+ assert read_trial_identity(skill_dir) == skill
+ assert read_trial_identity(control_dir) == control
+ assert read_trial_identity(skill_dir) != read_trial_identity(control_dir)
+
+
+@pytest.mark.parametrize(
+ "mutate",
+ [
+ lambda env: env.pop("INGOT_ARM"),
+ lambda env: env.__setitem__("INGOT_ARM", "treatment"),
+ lambda env: env.__setitem__("INGOT_HARNESS", "unknown"),
+ lambda env: env.__setitem__("INGOT_PROTOCOL", "responses"),
+ lambda env: env.__setitem__("INGOT_ENDPOINT_FINGERPRINT", "not-a-fingerprint"),
+ lambda env: env.__setitem__("INGOT_EXTRA", "surprise"),
+ lambda env: env.__setitem__("INGOT_COMBINATION_ID", "../../mutable/path"),
+ lambda env: env.__setitem__("INGOT_ARM", 7),
+ ],
+)
+def test_trial_identity_rejects_malformed_or_ambiguous_lock(tmp_path, mutate):
+ env = identity_env(_identity())
+ mutate(env)
+ trial = tmp_path / "trial"
+ _write_lock(trial, env)
+
+ with pytest.raises(ValueError, match="Harbor trial identity"):
+ read_trial_identity(trial)
+
+
+def test_trial_identity_rejects_invalid_lock_shape(tmp_path):
+ trial = tmp_path / "trial"
+ trial.mkdir()
+ (trial / "lock.json").write_text("[]")
+
+ with pytest.raises(ValueError, match="Harbor trial identity"):
+ read_trial_identity(trial)
+
+
+def test_agent_compiler_preserves_arm_and_shared_endpoint_limit(tmp_path):
+ target = _target()
+ source = tmp_path / "skill-source"
+ source.mkdir()
+ skill = compile_agent_config(target, "aider", "skill", source, 8)
+ control = compile_agent_config(target, "aider", "control", source, 8)
+
+ assert skill["name"] == control["name"] == "aider"
+ assert skill["model_name"] == control["model_name"] == "openai/dot-backbone"
+ assert skill["n_concurrent"] == control["n_concurrent"] == 8
+ assert skill["concurrency_group"] == control["concurrency_group"] == f"endpoint:{target.fingerprint}"
+ assert skill["skills"] == [str(source)]
+ assert control["skills"] == []
+ assert skill["env"]["INGOT_ARM"] == "skill"
+ assert control["env"]["INGOT_ARM"] == "control"
+ assert skill["env"]["INGOT_COMBINATION_ID"] == (
+ f"aider@{target.served_model}--{target.job_slug}")
+ assert skill["env"]["INGOT_GATEWAY_REVISION"] == f"direct-{NATIVE_RUNNER_REVISION}"
+ assert skill["env"]["OPENAI_API_KEY"] == control["env"]["OPENAI_API_KEY"] == "local"
+
+
+def test_measurement_compiler_can_resume_only_the_missing_arm(tmp_path):
+ target = _target()
+ cell = NativeCell(target, "aider")
+
+ config = compile_measurement_job(
+ tmp_path / "dataset", [f"task-{index}" for index in range(4)], [cell],
+ tmp_path / "skill", tmp_path / "jobs", attempts=3, global_limit=4,
+ endpoint_limits={target.fingerprint: 4}, arms=("control",))
+
+ assert [agent["env"]["INGOT_ARM"] for agent in config["agents"]] == ["control"]
+
+
+def test_agent_compiler_uses_import_path_for_gateway_codex(tmp_path):
+ target = _target()
+ from ingot.optimize.harbor_gateway import gateway_route
+
+ route = gateway_route(target, "codex")
+ assert route is not None
+ config = compile_agent_config(target, "codex", "skill", tmp_path, 4)
+
+ assert config["import_path"] == "ingot.optimize.harbor_codex_gateway:GatewayCodex"
+ assert "name" not in config
+ assert config["concurrency_group"] == f"endpoint:{target.fingerprint}"
+ assert config["env"]["INGOT_GATEWAY_REVISION"] == f"{route.revision}-{NATIVE_RUNNER_REVISION}"
+
+
+def test_iter_attempt_dirs_filters_unified_job_by_exact_identity(tmp_path):
+ from ingot.optimize.harbor_native import iter_attempt_dirs
+
+ wanted = _identity("skill")
+ other = _identity("control")
+ first = tmp_path / "first"
+ second = tmp_path / "second"
+ _write_lock(first, identity_env(wanted))
+ _write_lock(second, identity_env(other))
+ (first / "result.json").write_text('{"task_name":"ingot/h1"}')
+ (second / "result.json").write_text('{"task_name":"ingot/h1"}')
+
+ assert list(iter_attempt_dirs(tmp_path, identity=wanted)) == [first]
+ assert list(iter_attempt_dirs(tmp_path, identity=other)) == [second]
+
+
+def test_iter_attempt_dirs_does_not_inspect_unselected_malformed_sibling(tmp_path):
+ from ingot.optimize.harbor_native import iter_attempt_dirs
+
+ wanted = _identity("skill")
+ first = tmp_path / "first"
+ malformed = tmp_path / "malformed"
+ _write_lock(first, identity_env(wanted))
+ malformed.mkdir()
+ (malformed / "lock.json").write_text("not-json")
+ (first / "result.json").write_text('{"task_name":"ingot/h1"}')
+ (malformed / "result.json").write_text('{"task_name":"ingot/h1"}')
+
+ assert list(iter_attempt_dirs(tmp_path, identity=wanted)) == [first]
+
+
+def test_iter_attempt_dirs_rejects_symlinked_attempt(tmp_path):
+ from ingot.optimize.harbor_native import iter_attempt_dirs
+
+ outside = tmp_path / "outside"
+ outside.mkdir()
+ (outside / "result.json").write_text('{"task_name":"ingot/h1"}')
+ job = tmp_path / "job"
+ job.mkdir()
+ (job / "linked").symlink_to(outside, target_is_directory=True)
+
+ with pytest.raises(ValueError, match="real attempt directory"):
+ list(iter_attempt_dirs(job))
+
+
+@pytest.mark.parametrize("name", ["lock.json", "result.json"])
+def test_native_attempt_rejects_symlinked_identity_or_result_file(tmp_path, name):
+ from ingot.optimize.harbor_native import iter_attempt_dirs
+
+ job = tmp_path / "job"
+ trial = job / "h1__one"
+ trial.mkdir(parents=True)
+ external = tmp_path / f"external-{name}"
+ external.write_text('{}')
+ (trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(_identity())}}))
+ (trial / "result.json").write_text('{"task_name":"ingot/h1"}')
+ (trial / name).unlink()
+ (trial / name).symlink_to(external)
+
+ with pytest.raises(ValueError, match="regular file"):
+ list(iter_attempt_dirs(job, identity=_identity()))
diff --git a/tests/test_harbor_native_jobs.py b/tests/test_harbor_native_jobs.py
new file mode 100644
index 0000000..94b70de
--- /dev/null
+++ b/tests/test_harbor_native_jobs.py
@@ -0,0 +1,141 @@
+import pytest
+
+from ingot.optimize.harbor_native import (NativeCell, compile_canary_job,
+ compile_measurement_job,
+ select_measurement_cells, write_job_config)
+from ingot.optimize.harbor_targets import LocalTarget
+
+
+def _target(alias: str, model: str, port: int) -> LocalTarget:
+ return LocalTarget(
+ alias=alias,
+ display_name=model,
+ base_url=f"http://127.0.0.1:{port}",
+ served_model=model,
+ context_length=131_072,
+ protocols=frozenset({"chat", "messages", "responses"}),
+ )
+
+
+@pytest.fixture
+def cells():
+ first = _target("dell-qwen", "dot-backbone", 8011)
+ second = _target("spark-deepseek", "deepseek-v4-flash", 8000)
+ return [NativeCell(first, "aider"), NativeCell(second, "goose")]
+
+
+def test_canary_job_compiles_one_skill_agent_per_cell(tmp_path, cells):
+ config = compile_canary_job(
+ tmp_path / "dataset", "build-loop-h1", cells, tmp_path / "skill", tmp_path / "jobs",
+ global_limit=16, endpoint_limits={cell.target.fingerprint: 4 for cell in cells},
+ )
+
+ assert config["jobs_dir"] == str(tmp_path / "jobs")
+ assert config["n_attempts"] == 1
+ assert config["n_concurrent_trials"] == 16
+ assert config["datasets"] == [{"path": str(tmp_path / "dataset"),
+ "task_names": ["build-loop-h1"]}]
+ assert len(config["agents"]) == 2
+ assert [agent["env"]["INGOT_ARM"] for agent in config["agents"]] == ["canary", "canary"]
+ assert all(agent["skills"] == [str(tmp_path / "skill")] for agent in config["agents"])
+
+
+def test_measurement_job_interleaves_matched_arms_and_shares_endpoint_caps(tmp_path, cells):
+ limits = {cells[0].target.fingerprint: 8, cells[1].target.fingerprint: 12}
+ config = compile_measurement_job(
+ tmp_path / "dataset", ["h1", "h2", "h3", "h4"], cells,
+ tmp_path / "skill", tmp_path / "jobs", attempts=3, global_limit=32,
+ endpoint_limits=limits,
+ )
+
+ assert config["n_attempts"] == 3
+ assert config["n_concurrent_trials"] == 32
+ assert config["datasets"] == [{"path": str(tmp_path / "dataset"),
+ "task_names": ["h1", "h2", "h3", "h4"]}]
+ assert [agent["env"]["INGOT_ARM"] for agent in config["agents"]] == [
+ "skill", "control", "skill", "control"]
+ assert [agent["env"]["INGOT_COMBINATION_ID"] for agent in config["agents"]][::2] == [
+ cells[0].combination_id, cells[1].combination_id]
+ for index, cell in enumerate(cells):
+ pair = config["agents"][index * 2:index * 2 + 2]
+ assert {agent["n_concurrent"] for agent in pair} == {limits[cell.target.fingerprint]}
+ assert {agent["concurrency_group"] for agent in pair} == {
+ f"endpoint:{cell.target.fingerprint}"}
+
+
+def test_measurement_job_round_robins_target_major_input(tmp_path):
+ first = _target("dell-qwen", "dot-backbone", 8011)
+ second = _target("spark-deepseek", "deepseek-v4-flash", 8000)
+ cells = [NativeCell(first, "aider"), NativeCell(first, "goose"),
+ NativeCell(second, "aider"), NativeCell(second, "goose")]
+ config = compile_measurement_job(
+ tmp_path / "dataset", ["h1", "h2", "h3", "h4"], cells,
+ tmp_path / "skill", tmp_path / "jobs", attempts=3, global_limit=32,
+ endpoint_limits={first.fingerprint: 8, second.fingerprint: 8},
+ )
+
+ skill_agents = config["agents"][::2]
+ assert [agent["env"]["INGOT_ENDPOINT_FINGERPRINT"] for agent in skill_agents] == [
+ first.fingerprint, second.fingerprint, first.fingerprint, second.fingerprint]
+
+
+@pytest.mark.parametrize("attempts", [0, 1, 2, 4, True])
+def test_measurement_job_requires_full_three_attempt_contract(tmp_path, cells, attempts):
+ with pytest.raises(ValueError, match="three attempts"):
+ compile_measurement_job(
+ tmp_path / "dataset", ["h1", "h2", "h3", "h4"], cells,
+ tmp_path / "skill", tmp_path / "jobs", attempts=attempts, global_limit=16,
+ endpoint_limits={cell.target.fingerprint: 4 for cell in cells},
+ )
+
+
+def test_job_compiler_refuses_missing_or_unknown_endpoint_limit(tmp_path, cells):
+ with pytest.raises(ValueError, match="endpoint concurrency"):
+ compile_canary_job(
+ tmp_path / "dataset", "h1", cells, tmp_path / "skill", tmp_path / "jobs",
+ global_limit=16, endpoint_limits={cells[0].target.fingerprint: 4},
+ )
+
+
+def test_failed_canary_becomes_explicit_unmeasured_cell_without_lift(cells):
+ selected, unmeasured = select_measurement_cells(cells, {
+ cells[0].combination_id: {"ok": True},
+ cells[1].combination_id: {"error": "AgentTimeoutError: expired"},
+ })
+
+ assert selected == [cells[0]]
+ assert unmeasured == {cells[1].combination_id: {
+ "combination": cells[1].combination_id,
+ "harness": cells[1].harness,
+ "target_alias": cells[1].target.alias,
+ "endpoint_fingerprint": cells[1].target.fingerprint,
+ "state": "unmeasured",
+ "error": "AgentTimeoutError: expired",
+ }}
+ assert "lift" not in unmeasured[cells[1].combination_id]
+
+
+def test_job_config_write_is_atomic_and_deterministic(tmp_path, cells):
+ config = compile_canary_job(
+ tmp_path / "dataset", "h1", cells, tmp_path / "skill", tmp_path / "jobs",
+ global_limit=16, endpoint_limits={cell.target.fingerprint: 4 for cell in cells},
+ )
+ output = tmp_path / "config.json"
+
+ write_job_config(output, config)
+ first = output.read_bytes()
+ write_job_config(output, config)
+
+ assert output.read_bytes() == first
+ assert not list(tmp_path.glob(".config.json.*.tmp"))
+
+
+@pytest.mark.parametrize("name", ["http:config.json", "model@host:8011.json"])
+def test_job_config_refuses_url_or_port_derived_filename(tmp_path, cells, name):
+ config = compile_canary_job(
+ tmp_path / "dataset", "h1", cells, tmp_path / "skill", tmp_path / "jobs",
+ global_limit=16, endpoint_limits={cell.target.fingerprint: 4 for cell in cells},
+ )
+
+ with pytest.raises(ValueError, match="safe slug"):
+ write_job_config(tmp_path / name, config)
diff --git a/tests/test_harbor_native_runner.py b/tests/test_harbor_native_runner.py
new file mode 100644
index 0000000..47e0b0b
--- /dev/null
+++ b/tests/test_harbor_native_runner.py
@@ -0,0 +1,658 @@
+import json
+from pathlib import Path
+
+import pytest
+
+from ingot.optimize import harbor_eval as H
+from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env
+from ingot.optimize.harbor_targets import LocalTarget
+
+
+def _identity(arm="skill"):
+ return NativeTrialIdentity(
+ combination_id="aider@dot-backbone--dell-qwen-deadbeefcafe",
+ endpoint_fingerprint="deadbeefcafe", harness="aider", protocol="chat",
+ gateway_revision="direct", arm=arm)
+
+
+def _attempt(job: Path, name: str, identity: NativeTrialIdentity):
+ trial = job / name
+ trial.mkdir(parents=True)
+ (trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(identity)}}))
+ (trial / "result.json").write_text(json.dumps({"task_name": "ingot/h1",
+ "finished_at": "2026-08-13T00:00:00Z"}))
+
+
+def test_watch_native_job_releases_each_identity_once_at_exact_cardinality(tmp_path):
+ job = tmp_path / "job"
+ job.mkdir()
+ skill, control = _identity("skill"), _identity("control")
+ seen = []
+
+ _attempt(job, "h1__s1", skill)
+ expected = {skill: {"h1": 2}, control: {"h1": 1}}
+ assert H.watch_native_job(job, expected, seen.append) == set()
+ _attempt(job, "h1__c1", control)
+ assert H.watch_native_job(job, expected, seen.append) == {control}
+ _attempt(job, "h1__s2", skill)
+ assert H.watch_native_job(job, expected, seen.append,
+ released={control}) == {skill, control}
+ assert seen == [control, skill]
+
+
+def test_watch_native_job_refuses_more_attempts_than_contract(tmp_path):
+ job = tmp_path / "job"
+ job.mkdir()
+ identity = _identity()
+ _attempt(job, "h1__one", identity)
+ _attempt(job, "h1__two", identity)
+
+ with pytest.raises(RuntimeError, match="exceeded expected attempts"):
+ H.watch_native_job(job, {identity: {"h1": 1}}, lambda _identity: None)
+
+
+def test_run_native_job_uses_one_harbor_process_and_progress_callback(tmp_path, monkeypatch):
+ config = tmp_path / "config.json"
+ config.write_text("{}")
+ jobs = tmp_path / "jobs"
+ identity = _identity()
+ calls = []
+
+ class Process:
+ returncode = 0
+ count = 0
+ pid = 4242
+
+ def poll(self):
+ self.count += 1
+ if self.count == 1:
+ _attempt(jobs / "full", "h1__one", identity)
+ return None
+ return 0
+
+ def wait(self):
+ return 0
+
+ monkeypatch.setattr(H.subprocess, "Popen", lambda argv, **kwargs:
+ calls.append((argv, kwargs)) or Process())
+ monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}")
+ monkeypatch.setattr(H.time, "sleep", lambda _seconds: None)
+
+ overlay = tmp_path / "network.compose.yml"
+ overlay.write_text("networks: {}")
+ result = H.run_native_job(
+ config, jobs, "full", {identity: {"h1": 1}},
+ on_ready=lambda item: calls.append(item),
+ process_env={"PATH": "/bin", "HARBOR_EXTRA_DOCKER_COMPOSE": str(overlay)},
+ )
+
+ assert result == jobs / "full"
+ assert len([item for item in calls if isinstance(item, tuple)]) == 1
+ argv, kwargs = calls[0]
+ assert argv[:3] == [H.HARBOR_BIN, "run", "--config"]
+ assert argv[-6:] == ["--override-memory-mb", "2048", "--job-name", "full",
+ "--extra-docker-compose", str(overlay)]
+ assert kwargs["env"]["PATH"] == "/bin"
+ assert "HARBOR_EXTRA_DOCKER_COMPOSE" not in kwargs["env"]
+ assert kwargs["env"]["OPENAI_API_KEY"] == "local"
+ assert kwargs["env"]["ANTHROPIC_API_KEY"] == "local"
+ assert kwargs["env"]["CODEX_API_KEY"] == "local"
+ assert calls[-1] == identity
+
+
+def test_run_native_job_does_not_respawn_harbor_for_released_terminal_evidence(
+ tmp_path, monkeypatch):
+ config = tmp_path / "config.json"
+ config.write_text("{}")
+ jobs = tmp_path / "jobs"
+ job = jobs / "full"
+ identity = _identity()
+ _attempt(job, "h1__one", identity)
+ jobs.mkdir(exist_ok=True)
+ (jobs / "full.released.json").write_text(json.dumps({
+ "released": [identity_env(identity)],
+ }))
+ monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}")
+ monkeypatch.setattr(H.subprocess, "Popen",
+ lambda *_args, **_kwargs: pytest.fail("Harbor was respawned"))
+
+ result = H.run_native_job(
+ config, jobs, "full", {identity: {"h1": 1}},
+ on_ready=lambda _item: pytest.fail("released callback repeated"),
+ process_env={"PATH": "/bin"}, allow_completed_reuse=True,
+ )
+
+ assert result == job
+ assert not (jobs / "full.owner.json").exists()
+
+
+def test_run_native_job_refinalizes_terminal_evidence_without_respawning_harbor(
+ tmp_path, monkeypatch):
+ config = tmp_path / "config.json"
+ config.write_text("{}")
+ jobs = tmp_path / "jobs"
+ job = jobs / "full"
+ identity = _identity()
+ _attempt(job, "h1__one", identity)
+ seen = []
+ monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}")
+ monkeypatch.setattr(H.subprocess, "Popen",
+ lambda *_args, **_kwargs: pytest.fail("Harbor was respawned"))
+
+ result = H.run_native_job(
+ config, jobs, "full", {identity: {"h1": 1}}, on_ready=seen.append,
+ process_env={"PATH": "/bin"}, allow_completed_reuse=True,
+ )
+
+ assert result == job
+ assert seen == [identity]
+ assert json.loads((jobs / "full.released.json").read_text()) == {
+ "released": [identity_env(identity)],
+ }
+ assert not (jobs / "full.owner.json").exists()
+
+
+def test_run_native_job_keeps_terminal_finalization_pending_without_respawning_harbor(
+ tmp_path, monkeypatch):
+ config = tmp_path / "config.json"
+ config.write_text("{}")
+ jobs = tmp_path / "jobs"
+ job = jobs / "full"
+ identity = _identity()
+ _attempt(job, "h1__one", identity)
+ monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}")
+ monkeypatch.setattr(H.subprocess, "Popen",
+ lambda *_args, **_kwargs: pytest.fail("Harbor was respawned"))
+
+ with pytest.raises(RuntimeError, match="native Harbor finalization is pending"):
+ H.run_native_job(
+ config, jobs, "full", {identity: {"h1": 1}}, on_ready=lambda _item: False,
+ process_env={"PATH": "/bin"}, allow_completed_reuse=True,
+ )
+
+ assert json.loads((jobs / "full.released.json").read_text()) == {"released": []}
+ assert not (jobs / "full.owner.json").exists()
+
+
+def test_run_native_job_does_not_reuse_released_evidence_without_catalog_proof(
+ tmp_path, monkeypatch):
+ config = tmp_path / "config.json"
+ config.write_text("{}")
+ jobs = tmp_path / "jobs"
+ job = jobs / "full"
+ identity = _identity()
+ _attempt(job, "h1__one", identity)
+ jobs.mkdir(exist_ok=True)
+ (jobs / "full.released.json").write_text(json.dumps({
+ "released": [identity_env(identity)],
+ }))
+ calls = []
+
+ class Process:
+ def poll(self):
+ return 0
+
+ def wait(self):
+ return 0
+
+ monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}")
+ monkeypatch.setattr(H.subprocess, "Popen",
+ lambda *_args, **_kwargs: calls.append(True) or Process())
+
+ H.run_native_job(config, jobs, "full", {identity: {"h1": 1}},
+ on_ready=lambda _item: None, process_env={"PATH": "/bin"})
+
+ assert calls == [True]
+
+
+def test_watch_native_job_does_not_release_started_trial(tmp_path):
+ job = tmp_path / "job"
+ job.mkdir()
+ identity = _identity()
+ _attempt(job, "h1__one", identity)
+ result = job / "h1__one" / "result.json"
+ result.write_text(json.dumps({"task_name": "ingot/h1", "finished_at": None}))
+
+ assert H.watch_native_job(job, {identity: {"h1": 1}}, lambda _item: None) == set()
+
+
+def _target(alias="dell-qwen", model="dot-backbone", port=8011):
+ return LocalTarget(alias=alias, display_name=model, base_url=f"http://host:{port}",
+ served_model=model, context_length=163840,
+ protocols=frozenset({"chat", "responses", "messages"}))
+
+
+def test_native_sweep_runs_one_canary_queue_and_independent_caps(tmp_path, monkeypatch):
+ first = _target()
+ second = _target("spark-deepseek", "deepseek-v4-flash", 8899)
+ monkeypatch.setattr(H, "HARBOR_DIR", tmp_path / "harbor")
+ monkeypatch.setattr(H, "load_tasks", lambda _skill: ([], [{"task": "x"}] * 4, {}))
+ monkeypatch.setattr(H, "stage_skill", lambda _skill: tmp_path / "source")
+ monkeypatch.setattr(H, "build_dataset", lambda *_args: tmp_path / "dataset")
+ monkeypatch.setattr(H.shutil, "which", lambda _binary: "/bin/harbor")
+ monkeypatch.setattr(H, "discover_target", lambda alias, _url: first if alias == first.alias else second)
+ monkeypatch.setattr(H, "probe_protocol", lambda *_args: None)
+ monkeypatch.setattr(H, "probe_chat_tool_round_trip", lambda *_args: None)
+ monkeypatch.setattr(H, "run_canary", lambda *_args, **_kwargs:
+ pytest.fail("serial canary ran"))
+ calls = []
+
+ def canaries(*args, **kwargs):
+ calls.append(("canary", args, kwargs))
+ return {f"aider@{target.served_model}--{target.job_slug}": {"ok": True}
+ for target in (first, second)}
+
+ monkeypatch.setattr(H, "_run_native_canaries", canaries, raising=False)
+ monkeypatch.setattr(H, "_run_native_full_arms",
+ lambda *args, **kwargs: calls.append(("full", args, kwargs)))
+
+ H.run_local_sweep("demo", [first, second], harnesses=("aider",), native_parallel=True,
+ global_concurrency=32, endpoint_concurrency=6, log=lambda *_args: None)
+
+ assert [item[0] for item in calls] == ["canary", "full"]
+ assert calls[0][2]["global_limit"] == 32
+ assert calls[0][2]["endpoint_limit"] == 6
+ assert calls[1][2]["global_limit"] == 32
+ assert calls[1][2]["endpoint_limit"] == 6
+
+
+def test_native_canary_restart_rehydrates_terminal_identity_without_agent_rerun(tmp_path,
+ monkeypatch):
+ target = _target()
+ identity = H.native_trial_identity(target, "aider", "canary")
+ job = tmp_path / "canaries" / "demo" / "native-canaries"
+ trial = job / "demo-h0__one"
+ solution = trial / "verifier" / "solution"
+ solution.mkdir(parents=True)
+ (solution / "answer.md").write_text("done")
+ (trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(identity)}}))
+ (trial / "result.json").write_text(json.dumps({"task_name": "ingot/demo-h0",
+ "finished_at": "now"}))
+ monkeypatch.setattr(H, "compile_canary_job", lambda *_args, **_kwargs: {})
+ monkeypatch.setattr(H, "write_job_config", lambda *_args: None)
+ monkeypatch.setattr(H, "_telemetry_provenance", lambda *_args: {})
+ exported = []
+ monkeypatch.setattr(H, "export_job_attempts", lambda *args, **kwargs: exported.append(kwargs["identity"]))
+ monkeypatch.setattr(H, "run_native_job", lambda *_args, **_kwargs: job)
+
+ records = H._run_native_canaries(
+ "demo", tmp_path / "dataset", [target], ("aider",), [{"task": "x"}] * 4,
+ str(tmp_path / "source"), tmp_path / "canaries", global_limit=16,
+ endpoint_limit=4)
+
+ assert records[identity.combination_id]["ok"] is True
+ assert exported == [identity]
+
+
+def test_native_canary_telemetry_failure_is_unmeasured(tmp_path, monkeypatch):
+ target = _target()
+ identity = H.native_trial_identity(target, "aider", "canary")
+ monkeypatch.setattr(H, "compile_canary_job", lambda *_args, **_kwargs: {})
+ monkeypatch.setattr(H, "write_job_config", lambda *_args: None)
+ monkeypatch.setattr(H, "_telemetry_provenance", lambda *_args: {})
+ monkeypatch.setattr(H, "export_job_attempts",
+ lambda *_args, **_kwargs: (_ for _ in ()).throw(RuntimeError("telemetry")))
+
+ callback_results = []
+
+ def native(_config, jobs_root, job_name, expected, *, on_ready, **_kwargs):
+ for item in expected:
+ callback_results.append(on_ready(item))
+ return jobs_root / job_name
+
+ monkeypatch.setattr(H, "run_native_job", native)
+ records = H._run_native_canaries(
+ "demo", tmp_path / "dataset", [target], ("aider",), [{"task": "x"}] * 4,
+ str(tmp_path / "source"), tmp_path / "canaries", global_limit=16,
+ endpoint_limit=4)
+ record = records[identity.combination_id]
+ assert callback_results == [False]
+ assert "ok" not in record
+ assert record["error"] == "canary telemetry receipt was not verified"
+
+
+def test_native_completed_restart_publishes_failed_canary_without_callbacks(tmp_path, monkeypatch):
+ from ingot.optimize import harbor_rescore as R
+
+ passed = _target()
+ failed = _target("spark-deepseek", "deepseek-v4-flash", 8899)
+ holdout = [{"task": f"task {index}"} for index in range(4)]
+ source = tmp_path / "source"
+ skill_dir = source / "demo"
+ skill_dir.mkdir(parents=True)
+ (skill_dir / "SKILL.md").write_text("fixture skill\n")
+ scoring = {
+ "judge": "agy/fixture",
+ "scoring_revision": "fixture-v1",
+ "judge_billing_mode": "subscription",
+ "judge_runtime": "fixture",
+ }
+ monkeypatch.setattr(R, "current_scoring_identity", lambda: scoring)
+ monkeypatch.setattr(H, "compile_measurement_job", lambda *_args, **_kwargs: {})
+ monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}")
+ monkeypatch.setattr(H, "export_job_attempts", lambda *_args, **_kwargs:
+ pytest.fail("completed restart repeated telemetry export"))
+ monkeypatch.setattr(H.subprocess, "Popen", lambda *_args, **_kwargs:
+ pytest.fail("completed restart respawned Harbor"))
+ rescored = []
+
+ def fake_rescore(_skill, **kwargs):
+ rescored.append(list(kwargs["combination_paths"]))
+
+ monkeypatch.setattr(R, "rescore", fake_rescore)
+ manifest = {"combinations": {}}
+ jobs_root = tmp_path / "jobs"
+ cell = H.NativeCell(passed, "aider")
+ job_name = H._native_full_job_name(cell)
+ job = jobs_root / job_name
+ output_root = tmp_path / "published"
+ output_root.mkdir()
+ passed_id = f"aider@{passed.served_model}--{passed.job_slug}"
+ failed_id = f"aider@{failed.served_model}--{failed.job_slug}"
+ skill_identity = H.native_trial_identity(passed, "aider", "skill")
+ control_identity = H.native_trial_identity(passed, "aider", "control")
+ provenance = H._telemetry_provenance("demo", holdout, str(source))
+ (output_root / "demo.rescored.json").write_text(json.dumps({"combinations": {
+ passed_id: {
+ "combination": passed_id,
+ "endpoint_fingerprint": skill_identity.endpoint_fingerprint,
+ "skill_sha256": provenance["skill_sha256"],
+ "task_fingerprint": H._task_fingerprint(holdout),
+ "attempts": 3,
+ "harness": "aider",
+ "protocol": skill_identity.protocol,
+ "gateway_revision": skill_identity.gateway_revision,
+ **scoring,
+ "lift": 0.1,
+ },
+ }}))
+ for identity in (skill_identity, control_identity):
+ for task_index in range(4):
+ for attempt in range(3):
+ trial = job / f"{identity.arm}-h{task_index}-a{attempt}"
+ trial.mkdir(parents=True)
+ (trial / "lock.json").write_text(json.dumps({
+ "agent": {"env": identity_env(identity)},
+ }))
+ (trial / "result.json").write_text(json.dumps({
+ "task_name": f"ingot/demo-h{task_index}",
+ "finished_at": "2026-08-14T00:00:00Z",
+ }))
+ (jobs_root / f"{job_name}.released.json").write_text(json.dumps({
+ "released": [identity_env(skill_identity), identity_env(control_identity)],
+ }))
+ (jobs_root / "native-full.pipeline.json").write_text(json.dumps({
+ "agent_identity": {
+ "skill_sha256": provenance["skill_sha256"],
+ "task_fingerprint": H._task_fingerprint(holdout),
+ "attempts": 3,
+ "exporter_revision": H.EXPORTER_REVISION,
+ "cells": [[passed_id, skill_identity.gateway_revision]],
+ },
+ "scoring_identity": scoring,
+ "exported": {passed_id: ["control", "skill"]},
+ "graded": [passed_id],
+ "published": [passed_id],
+ }))
+
+ H._run_native_full_arms(
+ "demo", tmp_path / "dataset", [passed, failed], ("aider",), holdout, str(source),
+ jobs_root, manifest, {
+ passed_id: {"ok": True},
+ failed_id: {"error": "adapter did not route"},
+ },
+ global_limit=16, endpoint_limit=4, publish_root=output_root,
+ allow_completed_reuse=True,
+ )
+
+ assert [[path.name for path in paths] for paths in rescored] == [[failed_id]]
+ metadata = json.loads((jobs_root / failed_id / "combo.json").read_text())
+ assert metadata["canary_error"] == "adapter did not route"
+ assert metadata["skill_sha256"]
+
+
+def test_prior_lift_for_now_failed_cell_waits_for_selected_measurement(tmp_path, monkeypatch):
+ from ingot.optimize import harbor_rescore as R
+
+ failed = _target()
+ passed = _target("spark-deepseek", "deepseek-v4-flash", 8899)
+ holdout = [{"task": f"task {index}"} for index in range(4)]
+ source = tmp_path / "source"
+ skill_dir = source / "demo"
+ skill_dir.mkdir(parents=True)
+ (skill_dir / "SKILL.md").write_text("fixture skill\n")
+ scoring = {
+ "judge": "agy/fixture",
+ "scoring_revision": "fixture-v1",
+ "judge_billing_mode": "subscription",
+ "judge_runtime": "fixture",
+ }
+ monkeypatch.setattr(R, "current_scoring_identity", lambda: scoring)
+ monkeypatch.setattr(H, "compile_measurement_job", lambda *_args, **_kwargs: {})
+ monkeypatch.setattr(H, "write_job_config", lambda *_args, **_kwargs: None)
+ exported = []
+ monkeypatch.setattr(H, "export_job_attempts",
+ lambda *_args, **kwargs: exported.append(kwargs["identity"]))
+ rescored = []
+
+ def fake_rescore(_skill, **kwargs):
+ rescored.append(list(kwargs["combination_paths"]))
+
+ monkeypatch.setattr(R, "rescore", fake_rescore)
+
+ def fake_run_native_job(_config, _jobs_root, _job_name, expected, *, on_ready, **_kwargs):
+ for identity in expected:
+ assert identity.combination_id == passed_id
+ assert on_ready(identity) is True
+
+ monkeypatch.setattr(H, "run_native_job", fake_run_native_job)
+ manifest = {"combinations": {}}
+ jobs_root = tmp_path / "jobs"
+ output_root = tmp_path / "published"
+ output_root.mkdir()
+ failed_id = f"aider@{failed.served_model}--{failed.job_slug}"
+ passed_id = f"aider@{passed.served_model}--{passed.job_slug}"
+ failed_identity = H.native_trial_identity(failed, "aider", "skill")
+ provenance = H._telemetry_provenance("demo", holdout, str(source))
+ (output_root / "demo.rescored.json").write_text(json.dumps({"combinations": {
+ failed_id: {
+ "combination": failed_id,
+ "endpoint_fingerprint": failed_identity.endpoint_fingerprint,
+ "skill_sha256": provenance["skill_sha256"],
+ "task_fingerprint": H._task_fingerprint(holdout),
+ "attempts": 3,
+ "harness": "aider",
+ "protocol": failed_identity.protocol,
+ "gateway_revision": failed_identity.gateway_revision,
+ **scoring,
+ "lift": 0.1,
+ },
+ }}))
+
+ H._run_native_full_arms(
+ "demo", tmp_path / "dataset", [failed, passed], ("aider",), holdout, str(source),
+ jobs_root, manifest, {
+ failed_id: {"error": "adapter did not route"},
+ passed_id: {"ok": True},
+ },
+ global_limit=16, endpoint_limit=4, publish_root=output_root,
+ )
+
+ assert [[path.name for path in paths] for paths in rescored] == [[failed_id, passed_id]]
+ assert {identity.combination_id for identity in exported} == {passed_id}
+
+
+def test_native_full_arms_runs_stable_cells_concurrently_instead_of_one_wide_job(
+ tmp_path, monkeypatch):
+ """A partial run must finish cells, not spread work across the whole matrix first."""
+ import threading
+ from ingot.optimize import harbor_rescore as R
+
+ first = _target()
+ second = _target("spark-deepseek", "deepseek-v4-flash", 8899)
+ holdout = [{"task": f"task {index}"} for index in range(4)]
+ source = tmp_path / "source"
+ skill_dir = source / "demo"
+ skill_dir.mkdir(parents=True)
+ (skill_dir / "SKILL.md").write_text("fixture skill\n")
+ monkeypatch.setattr(R, "current_scoring_identity", lambda: {
+ "judge": "agy/fixture", "scoring_revision": "fixture-v1",
+ "judge_billing_mode": "subscription", "judge_runtime": "fixture",
+ })
+ monkeypatch.setattr(R, "rescore", lambda *_args, **_kwargs: None)
+ monkeypatch.setattr(H, "write_job_config", lambda *_args, **_kwargs: None)
+ monkeypatch.setattr(H, "export_job_attempts", lambda *_args, **_kwargs: None)
+ compiled = []
+
+ def compile_job(_dataset, _tasks, cells, *_args, **_kwargs):
+ compiled.append([cell.combination_id for cell in cells])
+ return {}
+
+ monkeypatch.setattr(H, "compile_measurement_job", compile_job)
+ barrier = threading.Barrier(2)
+ calls = []
+
+ def native(_config, jobs_root, job_name, expected, *, on_ready, **_kwargs):
+ calls.append((job_name, threading.get_ident(), set(expected)))
+ barrier.wait(timeout=2)
+ for identity in expected:
+ assert on_ready(identity) is True
+ return jobs_root / job_name
+
+ monkeypatch.setattr(H, "run_native_job", native)
+ ids = [f"aider@{target.served_model}--{target.job_slug}" for target in (first, second)]
+
+ H._run_native_full_arms(
+ "demo", tmp_path / "dataset", [first, second], ("aider",), holdout, str(source),
+ tmp_path / "jobs", {"combinations": {}}, {key: {"ok": True} for key in ids},
+ global_limit=12, endpoint_limit=6, publish_root=tmp_path / "published",
+ )
+
+ assert compiled == [[ids[0]], [ids[1]]]
+ assert len(calls) == 2
+ assert len({thread_id for _name, thread_id, _expected in calls}) == 2
+ assert all(len(expected) == 2 for _name, _thread_id, expected in calls)
+ assert len({name for name, _thread_id, _expected in calls}) == 2
+
+
+def test_native_full_arms_never_overlaps_jobs_for_one_endpoint(tmp_path, monkeypatch):
+ import threading
+ from ingot.optimize import harbor_rescore as R
+
+ target = _target()
+ holdout = [{"task": f"task {index}"} for index in range(4)]
+ source = tmp_path / "source"
+ skill_dir = source / "demo"
+ skill_dir.mkdir(parents=True)
+ (skill_dir / "SKILL.md").write_text("fixture skill\n")
+ monkeypatch.setattr(R, "current_scoring_identity", lambda: {
+ "judge": "agy/fixture", "scoring_revision": "fixture-v1",
+ "judge_billing_mode": "subscription", "judge_runtime": "fixture",
+ })
+ monkeypatch.setattr(R, "rescore", lambda *_args, **_kwargs: None)
+ monkeypatch.setattr(H, "compile_measurement_job", lambda *_args, **_kwargs: {})
+ monkeypatch.setattr(H, "write_job_config", lambda *_args, **_kwargs: None)
+ monkeypatch.setattr(H, "export_job_attempts", lambda *_args, **_kwargs: None)
+ guard = threading.Lock()
+ second_entered = threading.Event()
+ active = 0
+ peak = 0
+
+ def native(_config, jobs_root, job_name, expected, *, on_ready, **_kwargs):
+ nonlocal active, peak
+ with guard:
+ active += 1
+ peak = max(peak, active)
+ first = active == 1 and peak == 1
+ if active == 2:
+ second_entered.set()
+ try:
+ if first:
+ second_entered.wait(timeout=0.2)
+ for identity in expected:
+ assert on_ready(identity) is True
+ finally:
+ with guard:
+ active -= 1
+ return jobs_root / job_name
+
+ monkeypatch.setattr(H, "run_native_job", native)
+ ids = [f"{harness}@{target.served_model}--{target.job_slug}"
+ for harness in ("aider", "pi")]
+
+ H._run_native_full_arms(
+ "demo", tmp_path / "dataset", [target], ("aider", "pi"), holdout, str(source),
+ tmp_path / "jobs", {"combinations": {}}, {key: {"ok": True} for key in ids},
+ global_limit=12, endpoint_limit=6, publish_root=tmp_path / "published",
+ )
+
+ assert peak == 1
+
+
+def test_native_full_arms_adopts_legacy_identity_and_runs_only_missing_arm(tmp_path, monkeypatch):
+ from ingot.optimize import harbor_rescore as R
+
+ target = _target()
+ holdout = [{"task": f"task {index}"} for index in range(4)]
+ source = tmp_path / "source"
+ skill_dir = source / "demo"
+ skill_dir.mkdir(parents=True)
+ (skill_dir / "SKILL.md").write_text("fixture skill\n")
+ monkeypatch.setattr(R, "current_scoring_identity", lambda: {
+ "judge": "agy/fixture", "scoring_revision": "fixture-v1",
+ "judge_billing_mode": "subscription", "judge_runtime": "fixture",
+ })
+ rescored = []
+ monkeypatch.setattr(R, "rescore",
+ lambda *_args, **kwargs: rescored.append(kwargs["combination_paths"]))
+ compiled_arms = []
+
+ def compile_job(*_args, **kwargs):
+ compiled_arms.append(tuple(kwargs["arms"]))
+ return {}
+
+ monkeypatch.setattr(H, "compile_measurement_job", compile_job)
+ monkeypatch.setattr(H, "write_job_config", lambda *_args, **_kwargs: None)
+ jobs_root = tmp_path / "jobs"
+ cell = H.NativeCell(target, "aider")
+ skill_identity = H.native_trial_identity(target, "aider", "skill")
+ control_identity = H.native_trial_identity(target, "aider", "control")
+ legacy = jobs_root / "native-full"
+ for task_index in range(4):
+ for attempt in range(3):
+ trial = legacy / f"skill-h{task_index}-a{attempt}"
+ trial.mkdir(parents=True)
+ (trial / "lock.json").write_text(json.dumps({
+ "agent": {"env": identity_env(skill_identity)},
+ }))
+ (trial / "result.json").write_text(json.dumps({
+ "task_name": f"ingot/demo-h{task_index}",
+ "finished_at": "2026-08-14T00:00:00Z",
+ }))
+ exports = []
+
+ def export(job, _metadata, *, identity):
+ exports.append((job.name, identity.arm))
+
+ monkeypatch.setattr(H, "export_job_attempts", export)
+
+ def native(_config, _root, _name, expected, *, on_ready, **_kwargs):
+ assert set(expected) == {control_identity}
+ assert on_ready(control_identity) is True
+
+ monkeypatch.setattr(H, "run_native_job", native)
+ combination = cell.combination_id
+
+ H._run_native_full_arms(
+ "demo", tmp_path / "dataset", [target], ("aider",), holdout, str(source),
+ jobs_root, {"combinations": {}}, {combination: {"ok": True}},
+ global_limit=8, endpoint_limit=4, publish_root=tmp_path / "published",
+ )
+
+ assert compiled_arms == [("control",)]
+ assert exports == [("native-full", "skill"), (H._native_full_job_name(cell), "control")]
+ assert len(rescored) == 1
+ metadata = json.loads((jobs_root / combination / "combo.json").read_text())
+ assert metadata["native_jobs"] == {
+ "skill": "native-full", "control": H._native_full_job_name(cell),
+ }
diff --git a/tests/test_harbor_report.py b/tests/test_harbor_report.py
new file mode 100644
index 0000000..2da00d4
--- /dev/null
+++ b/tests/test_harbor_report.py
@@ -0,0 +1,326 @@
+"""Unit tests for reading a harness x model matrix for display.
+
+The rules under test are the ones the first live matrix broke: a combination that did not run must
+not reach a reader as a zero, every lift must carry the `n` behind it, and a matrix whose controls
+sit at the ceiling must report the ceiling rather than a winner."""
+import json
+import importlib
+from pathlib import Path
+
+import pytest
+
+from ingot.optimize import harbor_report as R
+from ingot import paths
+
+FIXTURES = Path(__file__).parent / "fixtures" / "harbor"
+
+
+def test_default_matrix_root_uses_the_mutable_state_directory(monkeypatch, tmp_path):
+ with monkeypatch.context() as isolated:
+ isolated.setenv(paths.HOME, str(tmp_path / "state"))
+ importlib.reload(R)
+ assert R.HARBOR_DIR == tmp_path / "state" / "runs" / "harbor"
+ importlib.reload(R)
+
+
+def _write(tmp_path, name, payload):
+ path = tmp_path / name
+ path.write_text(json.dumps(payload))
+ return path
+
+
+LIVE = {"skill": "demo", "judge": "google/gemini-2.5-flash",
+ "harnesses": {
+ "claude-code@anthropic/claude-opus-5": {
+ "skill_mean": 0.75, "control_mean": 0.5, "lift": 0.25,
+ "harness": "claude-code", "model": "anthropic/claude-opus-5",
+ "tasks_scored": 4, "tasks_dropped": []},
+ "aider@openai/gpt-5.5": {
+ "error": "RuntimeError: every task returned an empty workspace",
+ "harness": "aider", "model": "openai/gpt-5.5"}}}
+
+
+def test_a_combination_that_did_not_run_carries_no_lift(tmp_path):
+ """The whole point. A blank or a 0.0 in the lift column reads as 'measured, no effect', which
+ is the opposite of what happened, and is how a crashed control arm became 'lift +0.750'."""
+ _write(tmp_path, "demo.json", LIVE)
+ rows = {r["combination"]: r for r in R.read_matrix("demo", tmp_path)["rows"]}
+ broken = rows["aider@openai/gpt-5.5"]
+ assert "lift" not in broken and "skill_mean" not in broken and "n" not in broken
+ assert "empty workspace" in broken["error"]
+
+
+def test_every_measured_row_carries_the_n_behind_it(tmp_path):
+ _write(tmp_path, "demo.json", LIVE)
+ measured = [r for r in R.read_matrix("demo", tmp_path)["rows"] if "lift" in r]
+ assert [r["n"] for r in measured] == [4]
+
+
+def test_scale_provenance_is_allowlisted_on_measured_and_unmeasured_rows(tmp_path):
+ provenance = {"target_alias": "qwen35-4b", "family": "Qwen3.5", "parameter_billions": 4.0,
+ "quantization": "fp8-load", "tool_parser": "qwen3_coder"}
+ _write(tmp_path, "demo.rescored.json", {"combinations": {
+ "aider@qwen35-4b--qwen35-4b-deadbeef": {
+ **provenance, "harness": "aider", "model": "qwen35-4b", "lift": 0.2,
+ "skill_mean": 0.6, "control_mean": 0.4, "tasks_scored": 4, "attempts": 1,
+ "private_note": "must not escape"},
+ "pi@qwen35-4b--qwen35-4b-deadbeef": {
+ **provenance, "harness": "pi", "model": "qwen35-4b", "error": "canary failed",
+ "private_note": "must not escape"},
+ }})
+ rows = R.read_matrix("demo", tmp_path)["rows"]
+ assert len(rows) == 2
+ assert all({key: row[key] for key in provenance} == provenance for row in rows)
+ assert {row["model"] for row in rows} == {"Qwen/Qwen3.5-4B"}
+ assert all("private_note" not in row for row in rows)
+ assert "lift" not in next(row for row in rows if row["harness"] == "pi")
+
+
+def test_qwen_name_requires_matching_alias_and_family(tmp_path):
+ _write(tmp_path, "demo.json", {"harnesses": {"a@wire-id": {
+ "harness": "a", "model": "wire-id", "target_alias": "qwen35-4b",
+ "family": "Qwen3.6", "error": "not measured",
+ }}})
+
+ assert R.read_matrix("demo", tmp_path)["rows"][0]["model"] == "wire-id"
+
+
+@pytest.mark.parametrize("value", [True, "4", 0, -1, float("inf")])
+def test_invalid_parameter_counts_do_not_reach_the_size_chart_payload(tmp_path, value):
+ _write(tmp_path, "demo.json", {"harnesses": {"a@m": {
+ "harness": "a", "model": "m", "parameter_billions": value,
+ "lift": 0.1, "skill_mean": 0.5, "control_mean": 0.4,
+ "tasks_scored": 4, "attempts": 1}}})
+
+ row = R.read_matrix("demo", tmp_path)["rows"][0]
+ assert "parameter_billions" not in row
+
+
+def test_qwen_size_fixture_exposes_only_observed_scale_points():
+ matrix = R.read_matrix("qwen-size-matrix", FIXTURES)
+
+ assert matrix["models"] == ["dot-backbone", "qwen35-0.8b", "qwen35-2b", "qwen35-4b", "qwen35-9b"]
+ assert len(matrix["rows"]) == 8
+ assert matrix["measured"] == 7 and matrix["unmeasured"] == 1
+ assert all(row["parameter_billions"] in {0.8, 2.0, 4.0, 9.0, 27.0} for row in matrix["rows"])
+
+
+def test_the_broken_row_is_counted_but_not_averaged(tmp_path):
+ _write(tmp_path, "demo.json", LIVE)
+ out = R.read_matrix("demo", tmp_path)
+ assert (out["measured"], out["unmeasured"]) == (1, 1)
+ assert out["mean_lift"] == 0.25 # not 0.125, which is what averaging the failure would give
+
+
+def test_a_ceilinged_matrix_reports_the_ceiling_instead_of_a_winner(tmp_path):
+ """Controls near the top of the scale leave less headroom than the judge's own run-to-run
+ spread. Naming a best combination off that is reporting noise as a result."""
+ _write(tmp_path, "demo.json", {"harnesses": {
+ "a@m": {"lift": 0.02, "skill_mean": 0.87, "control_mean": 0.85, "tasks_scored": 4},
+ "b@m": {"lift": -0.03, "skill_mean": 0.82, "control_mean": 0.85, "tasks_scored": 4}}})
+ out = R.summarize(R.read_matrix("demo", tmp_path)["rows"])
+ assert out["best"] is None
+ assert "ceiling" in out["warning"] and "too easy" in out["warning"]
+
+
+def test_a_matrix_with_headroom_does_name_the_best_combination(tmp_path):
+ _write(tmp_path, "demo.json", {"harnesses": {
+ "a@m": {"lift": 0.25, "skill_mean": 0.75, "control_mean": 0.50, "tasks_scored": 4,
+ "attempts": 3},
+ "b@m": {"lift": 0.05, "skill_mean": 0.55, "control_mean": 0.50, "tasks_scored": 4,
+ "attempts": 3}}})
+ out = R.summarize(R.read_matrix("demo", tmp_path)["rows"])
+ assert out["warning"] == ""
+ assert out["best"]["combination"] == "a@m" and out["best"]["n"] == 4
+
+
+def test_thin_evidence_is_flagged_even_with_headroom(tmp_path):
+ _write(tmp_path, "demo.json", {"harnesses": {
+ "a@m": {"lift": 0.4, "skill_mean": 0.6, "control_mean": 0.2, "tasks_scored": 1,
+ "attempts": 3}}})
+ out = R.summarize(R.read_matrix("demo", tmp_path)["rows"])
+ assert "fewer than 3" in out["warning"] and out["best"] is None
+
+
+def test_a_single_attempt_matrix_will_not_be_read_as_a_ranking(tmp_path):
+ """Two control-arm runs of an identical configuration moved a task by 0.278 and swapped two
+ harnesses' rank, while re-judging one fixed answer three times was identical. One attempt per
+ task sits under that noise, so the difference between these rows is agent variance."""
+ _write(tmp_path, "demo.json", {"harnesses": {
+ "a@m": {"lift": 0.25, "skill_mean": 0.60, "control_mean": 0.35, "tasks_scored": 4},
+ "b@m": {"lift": 0.05, "skill_mean": 0.40, "control_mean": 0.35, "tasks_scored": 4}}})
+ matrix = R.read_matrix("demo", tmp_path)
+ out = R.summarize(matrix["rows"])
+ assert out["best"] is None
+ assert "one attempt" in out["warning"] and "-k 3" in out["warning"]
+ assert matrix["rows"][0]["attempts"] == 1
+
+
+def test_row_level_exploratory_evidence_disables_ranking_without_hiding_numbers(tmp_path):
+ _write(tmp_path, "demo.json", {"harnesses": {
+ "a@m": {"lift": 0.25, "skill_mean": 0.60, "control_mean": 0.35,
+ "tasks_scored": 4, "attempts": 1, "exploratory": True,
+ "rankable": False}}})
+ matrix = R.read_matrix("demo", tmp_path)
+
+ assert matrix["exploratory"] is True and matrix["rankable"] is False
+ assert matrix["mean_lift"] == 0.25 and matrix["rows"][0]["lift"] == 0.25
+ assert matrix["model_summaries"]["m"]["best_harness"] is None
+ assert "exploratory" in matrix["warning"].lower()
+
+
+def test_the_rescored_matrix_wins_when_it_is_newer(tmp_path):
+ """Rescoring is what removed two fabricated cells from the first grid. Showing the raw file
+ over a newer correction would put them back on the page."""
+ _write(tmp_path, "demo.json", {"harnesses": {
+ "a@m": {"lift": -0.375, "skill_mean": 0.2, "control_mean": 0.575, "tasks_scored": 4}}})
+ path = _write(tmp_path, "demo.rescored.json", {"combinations": {
+ "a@m": {"lift": 0.125, "skill_mean": 0.7, "control_mean": 0.575, "tasks_scored": 2}}})
+ import os
+ newer = (tmp_path / "demo.json").stat().st_mtime + 60
+ os.utime(path, (newer, newer))
+ out = R.read_matrix("demo", tmp_path)
+ assert out["rescored"] is True
+ assert out["rows"][0]["lift"] == 0.125
+
+
+def test_the_rescored_schema_recovers_harness_and_model_from_the_key(tmp_path):
+ """harbor_rescore writes rows without harness/model fields; the combination key still has them,
+ and a matrix that cannot say which model a row used cannot be read as a co-occurrence grid."""
+ _write(tmp_path, "demo.rescored.json", {"combinations": {
+ "terminus-2@anthropic/claude-opus-5": {
+ "lift": 0.1, "skill_mean": 0.6, "control_mean": 0.5, "tasks_scored": 4}}})
+ row = R.read_matrix("demo", tmp_path)["rows"][0]
+ assert (row["harness"], row["model"]) == ("terminus-2", "anthropic/claude-opus-5")
+
+
+def test_a_skill_never_run_is_absent_rather_than_empty(tmp_path):
+ assert R.read_matrix("never-run", tmp_path) is None
+
+
+def test_a_corrupt_matrix_raises_rather_than_reading_as_no_results(tmp_path):
+ """'No combinations helped' and 'the file is broken' must not look the same on the page."""
+ (tmp_path / "demo.json").write_text("{not json")
+ with pytest.raises(ValueError, match="unreadable"):
+ R.read_matrix("demo", tmp_path)
+
+
+def test_available_lists_skills_that_have_a_matrix(tmp_path):
+ _write(tmp_path, "demo.json", LIVE)
+ _write(tmp_path, "other.rescored.json", {"combinations": {}})
+ assert sorted(R.available(tmp_path)) == ["demo", "other"]
+
+
+def _model_matrix_fixture():
+ """Five model identities over nine harness identities, with sparse recorded intersections."""
+ rows = {
+ "claude-code@ceiling-model": {
+ "harness": "claude-code", "model": "ceiling-model",
+ "target_alias": "dell-qwen", "endpoint_fingerprint": "q" * 64,
+ "protocol": "messages", "skill_mean": 0.90, "control_mean": 0.86,
+ "lift": 0.04, "tasks_scored": 4, "attempts": 3,
+ },
+ "codex@ceiling-model": {
+ "harness": "codex", "model": "ceiling-model",
+ "skill_mean": 0.88, "control_mean": 0.86, "lift": 0.02,
+ "tasks_scored": 4, "attempts": 3,
+ },
+ "aider@thin-model": {
+ "harness": "aider", "model": "thin-model",
+ "skill_mean": 0.70, "control_mean": 0.40, "lift": 0.30,
+ "tasks_scored": 1, "attempts": 3,
+ },
+ "goose@single-model": {
+ "harness": "goose", "model": "single-model",
+ "skill_mean": 0.70, "control_mean": 0.40, "lift": 0.30,
+ "tasks_scored": 4,
+ },
+ "opencode@headroom-model": {
+ "harness": "opencode", "model": "headroom-model",
+ "skill_mean": 0.80, "control_mean": 0.40, "lift": 0.40,
+ "tasks_scored": 4, "attempts": 3,
+ },
+ "pi@headroom-model": {
+ "harness": "pi", "model": "headroom-model",
+ "skill_mean": 0.60, "control_mean": 0.40, "lift": 0.20,
+ "tasks_scored": 4, "attempts": 3,
+ },
+ "terminus-2@unmeasured-model": {
+ "harness": "terminus-2", "model": "unmeasured-model",
+ "target_alias": "spark-deepseek", "endpoint_fingerprint": "d" * 64,
+ "protocol": "chat", "error": "canary failed",
+ },
+ "mini-swe-agent@headroom-model": {
+ "harness": "mini-swe-agent", "model": "headroom-model",
+ "error": "full run failed",
+ },
+ "openclaw@single-model": {
+ "harness": "openclaw", "model": "single-model",
+ "skill_mean": 0.65, "control_mean": 0.45, "lift": 0.20,
+ "tasks_scored": 4,
+ },
+ }
+ return {"harnesses": rows}
+
+
+def test_report_preserves_sparse_rows_axes_and_identity_metadata(tmp_path):
+ _write(tmp_path, "demo.json", _model_matrix_fixture())
+
+ out = R.read_matrix("demo", tmp_path)
+
+ assert out["models"] == [
+ "ceiling-model", "headroom-model", "single-model",
+ "thin-model", "unmeasured-model",
+ ]
+ assert out["harnesses"] == [
+ "aider", "claude-code", "codex", "goose", "mini-swe-agent", "openclaw",
+ "opencode", "pi", "terminus-2",
+ ]
+ assert len(out["rows"]) == 9
+ row = next(row for row in out["rows"] if row["combination"] == "claude-code@ceiling-model")
+ assert row["target_alias"] == "dell-qwen"
+ assert row["endpoint_fingerprint"] == "q" * 64
+ assert row["protocol"] == "messages"
+
+
+def test_model_summaries_refuse_ceiling_thin_and_single_attempt_independently(tmp_path):
+ _write(tmp_path, "demo.json", _model_matrix_fixture())
+
+ summaries = R.read_matrix("demo", tmp_path)["model_summaries"]
+
+ assert set(summaries) == {
+ "ceiling-model", "headroom-model", "single-model",
+ "thin-model", "unmeasured-model",
+ }
+ assert summaries["ceiling-model"]["best_harness"] is None
+ assert "ceiling" in summaries["ceiling-model"]["warning"]
+ assert summaries["thin-model"]["best_harness"] is None
+ assert "fewer than 3" in summaries["thin-model"]["warning"]
+ assert summaries["single-model"]["best_harness"] is None
+ assert "one attempt" in summaries["single-model"]["warning"]
+ assert summaries["headroom-model"]["best_harness"] == "opencode"
+ assert summaries["unmeasured-model"] == {
+ "model": "unmeasured-model", "measured": 0, "unmeasured": 1,
+ "mean_lift": None, "control_mean": None,
+ "warning": "No combination produced a measurement.", "best_harness": None,
+ }
+
+
+def test_report_has_no_global_best_and_does_not_synthesize_missing_intersections(tmp_path):
+ _write(tmp_path, "demo.json", _model_matrix_fixture())
+
+ out = R.read_matrix("demo", tmp_path)
+ failed = next(row for row in out["rows"] if row["combination"] == "terminus-2@unmeasured-model")
+
+ assert "best" not in out
+ assert (out["measured"], out["unmeasured"]) == (7, 2)
+ assert "lift" not in failed
+ assert "skill_mean" not in failed and "control_mean" not in failed and "n" not in failed
+ assert failed["target_alias"] == "spark-deepseek"
+ assert failed["endpoint_fingerprint"] == "d" * 64 and failed["protocol"] == "chat"
+ measured_without_identity = next(
+ row for row in out["rows"] if row["combination"] == "codex@ceiling-model"
+ )
+ assert all(field not in measured_without_identity for field in (
+ "target_alias", "endpoint_fingerprint", "protocol"))
+ assert len(out["rows"]) < len(out["models"]) * len(out["harnesses"])
diff --git a/tests/test_harbor_rescore.py b/tests/test_harbor_rescore.py
new file mode 100644
index 0000000..021bf63
--- /dev/null
+++ b/tests/test_harbor_rescore.py
@@ -0,0 +1,1109 @@
+"""Regression tests for Harbor evidence-only rescoring.
+
+These fixtures contain completed Harbor-style trials but replace only the external judge. The
+rescorer still reads the same on-disk result and verifier artifact layout that live runs retain.
+"""
+from __future__ import annotations
+
+import hashlib
+import json
+from pathlib import Path
+
+import pytest
+from langfuse import Langfuse
+
+from ingot.optimize import agy_judge as A
+from ingot.optimize import harbor_langfuse as L
+from ingot.optimize import harbor_report
+from ingot.optimize import harbor_rescore as R
+
+
+FINGERPRINT = "tasks-v1"
+HISTORICAL_JUDGE = "judge-a"
+HISTORICAL_REVISION = "harbor-rubric-v1"
+RUNTIME = "agy 1.1.11"
+SKILL_BODY = "fixture skill body\n"
+MIGRATION_PROVENANCE = {
+ "skill": "demo",
+ "skill_body": SKILL_BODY,
+ "skill_sha256": hashlib.sha256(SKILL_BODY.encode()).hexdigest(),
+ "task_texts": {"demo-h0": "Use deterministic fixture evidence."},
+}
+
+
+def metadata(combination: str, *, fingerprint: str = "endpoint-a", attempts: int = 3) -> dict:
+ return {
+ "combination": combination,
+ "harness": "codex",
+ "model": "qwen3.6-27b",
+ "target_alias": "dell",
+ "endpoint_fingerprint": fingerprint,
+ "protocol": "openai",
+ "task_fingerprint": FINGERPRINT,
+ "attempts": attempts,
+ "judge": HISTORICAL_JUDGE,
+ "scoring_revision": HISTORICAL_REVISION,
+ **MIGRATION_PROVENANCE,
+ }
+
+
+def write_combo(root: Path, name: str, record: dict, *, artifact: str = "answer",
+ recorded_attempts: int = 3, receipt_metadata: dict | None = None) -> Path:
+ """Write a completed two-arm Harbor combination with recorded attempts per arm."""
+ combo = root / name
+ combo.mkdir(parents=True)
+ (combo / "combo.json").write_text(json.dumps(record))
+ for arm in ("skill", "control"):
+ for attempt in range(1, recorded_attempts + 1):
+ solution = combo / arm / f"demo-h0__attempt-{attempt}" / "verifier" / "solution"
+ solution.mkdir(parents=True)
+ (solution / "answer.md").write_text(f"{arm} {artifact}")
+ trial = solution.parent.parent
+ (trial / "result.json").write_text(json.dumps({
+ "task_name": f"ingot/demo-h0__attempt-{attempt}", "exception_info": {},
+ }))
+ payload = L.build_attempt_payload(
+ trial, {**(receipt_metadata or record), "arm": arm})
+ digest = L._payload_sha256(payload)
+ (trial / "langfuse-receipt.json").write_text(json.dumps({
+ "status": "verified",
+ "trace_id": Langfuse.create_trace_id(
+ seed=f"{L.EXPORTER_REVISION}:{digest}"),
+ "payload_sha256": digest,
+ "exporter_revision": L.EXPORTER_REVISION,
+ }))
+ return combo
+
+
+def patch_current_facts(monkeypatch: pytest.MonkeyPatch) -> list[dict]:
+ holdout = [{"task": "demo", "rubric": ""}]
+ monkeypatch.setenv("JUDGE_BACKEND", "agy")
+ monkeypatch.setattr(R, "_task_fingerprint", lambda tasks: FINGERPRINT)
+ monkeypatch.setattr(R, "preflight", lambda: {
+ "identity": R.AGY_IDENTITY,
+ "model": "gemini-3.6-flash-medium",
+ "version": RUNTIME,
+ "billing_mode": "subscription",
+ })
+ return holdout
+
+
+def write_legacy_manifest(path: Path, root: Path, *, attempts: int = 3) -> Path:
+ path.write_text(json.dumps({
+ "root": str(root),
+ "task_fingerprint": FINGERPRINT,
+ "attempts": attempts,
+ }))
+ return path
+
+
+def write_provenance(path: Path) -> Path:
+ path.write_text(json.dumps(MIGRATION_PROVENANCE))
+ return path
+
+
+def patch_scoring(monkeypatch: pytest.MonkeyPatch) -> None:
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+
+ def fake_score(answers, *_args):
+ return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4]
+
+ monkeypatch.setattr(R, "score", fake_score)
+
+
+def test_discovery_uses_complete_combo_identity_across_preserved_and_local_roots(tmp_path):
+ preserved = tmp_path / "jobs" / "demo"
+ local = tmp_path / "jobs" / "demo-k3"
+ old = {"combination": "codex@qwen3.6-27b", "harness": "codex", "model": "qwen3.6-27b"}
+ local_record = metadata("codex@qwen3.6-27b--dell-22222222", fingerprint="2" * 64)
+ old_combo = write_combo(preserved, "codex_qwen3.6-27b", old)
+ local_combo = write_combo(local, "codex@qwen3.6-27b--dell-22222222", local_record)
+
+ assert R.discover_combinations([preserved, local]) == [old_combo, local_combo]
+ assert R._combo_identity(local_combo) == local_record
+
+
+@pytest.mark.parametrize("field, value", [
+ ("task_fingerprint", "other-tasks"),
+ ("attempts", 1),
+])
+def test_compatibility_mismatch_stops_before_scoring_or_publication(tmp_path, monkeypatch, field, value):
+ root = tmp_path / "jobs"
+ first = metadata("codex@qwen--dell-a", fingerprint="a" * 64)
+ second = metadata("codex@qwen--dell-b", fingerprint="b" * 64)
+ second[field] = value
+ write_combo(root, "first", first)
+ write_combo(root, "second", second)
+ out = tmp_path / "demo.rescored.json"
+ out.write_bytes(b"known-good")
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start"))
+
+ with pytest.raises(ValueError, match=field):
+ R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ assert out.read_bytes() == b"known-good"
+
+
+def test_legacy_proprietary_combo_uses_explicit_manifest_with_the_real_matrix_shape(tmp_path, monkeypatch):
+ harbor = tmp_path / "harbor"
+ root = harbor / "jobs" / "demo"
+ legacy = {"combination": "codex@qwen", "harness": "codex", "model": "qwen"}
+ write_combo(root, "codex_qwen", legacy,
+ receipt_metadata={**legacy, **MIGRATION_PROVENANCE})
+ (harbor / "demo.json").write_text(json.dumps({
+ "skill": "demo",
+ "tasks": 1,
+ "pinned_model": None,
+ "judge": HISTORICAL_JUDGE,
+ "harnesses": {"codex@qwen": {"attempts": 3}},
+ }))
+ manifest = write_legacy_manifest(tmp_path / "legacy.json", root)
+ provenance = write_provenance(tmp_path / "provenance.json")
+ monkeypatch.setattr(R, "HARBOR_DIR", harbor)
+ patch_scoring(monkeypatch)
+
+ summary = R.rescore(
+ "demo", jobs_roots=[root], legacy_metadata=manifest,
+ provenance_metadata=[provenance], log=lambda *_args: None)
+
+ row = summary["combinations"]["codex@qwen"]
+ assert row["attempts"] == 3
+ assert row["judge"] == R.AGY_IDENTITY
+ assert row["task_fingerprint"] == FINGERPRINT
+ assert row["scoring_revision"] == "harbor-rubric-v2-agy"
+ assert row["judge_runtime"] == RUNTIME
+ assert row["judge_billing_mode"] == "subscription"
+
+
+def test_legacy_manifest_authorizes_its_exact_nondefault_preserved_root(tmp_path, monkeypatch):
+ harbor = tmp_path / "harbor"
+ root = harbor / "jobs" / "demo-k3"
+ legacy = {"combination": "codex@qwen", "harness": "codex", "model": "qwen"}
+ write_combo(root, "codex_qwen", legacy,
+ receipt_metadata={**legacy, **MIGRATION_PROVENANCE})
+ (harbor / "demo.json").write_text(json.dumps({
+ "skill": "demo",
+ "tasks": 1,
+ "pinned_model": None,
+ "judge": HISTORICAL_JUDGE,
+ "harnesses": {"codex@qwen": {"attempts": 3}},
+ }))
+ manifest = write_legacy_manifest(tmp_path / "legacy.json", root)
+ provenance = write_provenance(tmp_path / "provenance.json")
+ monkeypatch.setattr(R, "HARBOR_DIR", harbor)
+ patch_scoring(monkeypatch)
+
+ summary = R.rescore(
+ "demo", jobs_roots=[root], legacy_metadata=manifest,
+ provenance_metadata=[provenance], log=lambda *_args: None)
+
+ assert summary["combinations"]["codex@qwen"]["attempts"] == 3
+
+
+def test_legacy_combo_without_explicit_manifest_is_refused(tmp_path, monkeypatch):
+ harbor = tmp_path / "harbor"
+ root = harbor / "jobs" / "demo"
+ write_combo(root, "codex_qwen", {"combination": "codex@qwen", "harness": "codex", "model": "qwen"})
+ (harbor / "demo.json").write_text(json.dumps({
+ "skill": "demo", "tasks": 1, "pinned_model": None, "judge": HISTORICAL_JUDGE,
+ "harnesses": {"codex@qwen": {"attempts": 3}},
+ }))
+ monkeypatch.setattr(R, "HARBOR_DIR", harbor)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start"))
+
+ with pytest.raises(ValueError, match="manifest"):
+ R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+
+def test_legacy_combo_refuses_matrix_metadata_that_conflicts_with_its_record(tmp_path, monkeypatch):
+ harbor = tmp_path / "harbor"
+ root = harbor / "jobs" / "demo"
+ legacy = {"combination": "codex@qwen", "harness": "codex", "model": "qwen", "attempts": 1}
+ write_combo(root, "codex_qwen", legacy)
+ (harbor / "demo.json").write_text(json.dumps({
+ "skill": "demo", "tasks": 1, "pinned_model": None,
+ "judge": HISTORICAL_JUDGE,
+ "harnesses": {"codex@qwen": {"attempts": 3}},
+ }))
+ manifest = write_legacy_manifest(tmp_path / "legacy.json", root)
+ monkeypatch.setattr(R, "HARBOR_DIR", harbor)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start"))
+
+ with pytest.raises(ValueError, match="attempts"):
+ R.rescore("demo", jobs_roots=[root], legacy_metadata=manifest, log=lambda *_args: None)
+
+
+def test_legacy_combo_refuses_manifest_for_the_wrong_root(tmp_path, monkeypatch):
+ harbor = tmp_path / "harbor"
+ root = harbor / "jobs" / "demo"
+ legacy = {"combination": "codex@qwen", "harness": "codex", "model": "qwen"}
+ write_combo(root, "codex_qwen", legacy)
+ (harbor / "demo.json").write_text(json.dumps({
+ "skill": "demo", "tasks": 1, "pinned_model": None,
+ "judge": HISTORICAL_JUDGE,
+ "harnesses": {"codex@qwen": {"attempts": 3}},
+ }))
+ manifest = write_legacy_manifest(tmp_path / "legacy.json", harbor / "jobs" / "other")
+ monkeypatch.setattr(R, "HARBOR_DIR", harbor)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start"))
+
+ with pytest.raises(ValueError, match="root"):
+ R.rescore("demo", jobs_roots=[root], legacy_metadata=manifest, log=lambda *_args: None)
+
+
+@pytest.mark.parametrize("field, value", [
+ ("attempts", 0),
+ ("attempts", False),
+ ("task_fingerprint", False),
+])
+def test_compatibility_rejects_invalid_shared_metadata(field, value):
+ record = metadata("codex@qwen--dell")
+ record[field] = value
+
+ with pytest.raises(ValueError, match=field):
+ R.validate_compatibility([record])
+
+
+def test_public_compatibility_validator_accepts_one_argument_and_only_checks_records():
+ record = metadata("codex@qwen--dell")
+
+ R.validate_compatibility([record])
+
+
+@pytest.mark.parametrize("field, value", [
+ ("task_fingerprint", "old-tasks"),
+ ("attempts", 1),
+])
+def test_current_fact_mismatch_stops_before_scoring_or_publication(tmp_path, monkeypatch, field, value):
+ root = tmp_path / "jobs"
+ first = metadata("codex@qwen--dell-a", fingerprint="a" * 64)
+ second = metadata("codex@qwen--dell-b", fingerprint="b" * 64)
+ first[field] = second[field] = value
+ write_combo(root, "first", first)
+ write_combo(root, "second", second)
+ out = tmp_path / "demo.rescored.json"
+ out.write_bytes(b"known-good")
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start"))
+
+ with pytest.raises(ValueError, match=field):
+ R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ assert out.read_bytes() == b"known-good"
+
+
+def test_rescore_refuses_missing_attempt_receipt_before_preflight_judge_or_publication(tmp_path, monkeypatch):
+ """Removing the live receipt gate would silently publish a matrix with withheld telemetry."""
+ root = tmp_path / "jobs"
+ combo = write_combo(root, "one", metadata("codex@qwen--dell", fingerprint="a" * 64))
+ missing = next(combo.glob("skill/*/langfuse-receipt.json"))
+ missing.unlink()
+ out = tmp_path / "demo.rescored.json"
+ out.write_bytes(b"known-good")
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ monkeypatch.setenv("JUDGE_BACKEND", "agy")
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], [{"task": "demo", "rubric": ""}], {}))
+ monkeypatch.setattr(R, "_task_fingerprint", lambda tasks: FINGERPRINT)
+ monkeypatch.setattr(R, "preflight", lambda: pytest.fail("judge preflight must not start"))
+ monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("judge must not start"))
+
+ with pytest.raises(L.TelemetryReceiptError, match="receipt"):
+ R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ assert out.read_bytes() == b"known-good"
+
+
+def test_one_persisted_provenance_document_authorizes_export_and_rescore(tmp_path, monkeypatch):
+ """The same explicit migration document must authorize preserved export and publication."""
+ root = tmp_path / "jobs"
+ record = metadata("codex@qwen--dell", fingerprint="a" * 64)
+ for field in MIGRATION_PROVENANCE:
+ record.pop(field)
+ combo = write_combo(
+ root, "one", record, receipt_metadata={**record, **MIGRATION_PROVENANCE})
+ for receipt in combo.glob("*/*/langfuse-receipt.json"):
+ receipt.unlink()
+ provenance_path = write_provenance(tmp_path / "provenance.json")
+
+ class Observation:
+ def __init__(self, identifier: str) -> None:
+ self.id = identifier
+
+ def end(self) -> None:
+ pass
+
+ class PersistedReadback:
+ def __init__(self) -> None:
+ self.observations = {}
+
+ def create_trace_id(self, *, seed: str) -> str:
+ return Langfuse.create_trace_id(seed=seed)
+
+ def start_observation(self, **kwargs):
+ identifier = f"{len(self.observations) + 1:016x}"
+ trace_id = kwargs["trace_context"]["trace_id"]
+ self.observations[trace_id] = {
+ "id": identifier,
+ "name": kwargs["name"],
+ "type": kwargs["as_type"].upper(),
+ "metadata": kwargs["metadata"],
+ }
+ return Observation(identifier)
+
+ def flush(self) -> None:
+ pass
+
+ def read_trace(self, trace_id: str):
+ observation = self.observations.get(trace_id)
+ return {"id": trace_id, "observations": [observation]} if observation else None
+
+ client = PersistedReadback()
+ provenance = L.load_provenance_metadata(provenance_path)
+ assert len(L.export_evidence_root(root, metadata=provenance, client=client)) == 6
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ summary = R.rescore(
+ "demo", jobs_roots=[root], provenance_metadata=[provenance_path],
+ log=lambda *_args: None)
+
+ assert summary["scored"] == 1
+ assert len(client.observations) == 6
+
+
+def test_rescore_rejects_provenance_free_receipts_before_judge_preflight(tmp_path, monkeypatch):
+ """Even exact v2 receipts cannot authorize publication without explicit provenance."""
+ root = tmp_path / "jobs"
+ record = metadata("codex@qwen--dell", fingerprint="a" * 64)
+ for field in MIGRATION_PROVENANCE:
+ record.pop(field)
+ write_combo(root, "one", record)
+ out = tmp_path / "demo.rescored.json"
+ out.write_bytes(b"known-good")
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ monkeypatch.setenv("JUDGE_BACKEND", "agy")
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], [{"task": "demo", "rubric": ""}], {}))
+ monkeypatch.setattr(R, "_task_fingerprint", lambda tasks: FINGERPRINT)
+ monkeypatch.setattr(R, "preflight", lambda: pytest.fail("judge preflight must not start"))
+
+ with pytest.raises(L.TelemetryReceiptError, match="provenance"):
+ R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ assert out.read_bytes() == b"known-good"
+
+
+def test_rescore_refuses_non_agy_backend_before_preflight_or_scoring(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ write_combo(root, "one", metadata("codex@qwen--dell", fingerprint="a" * 64))
+ out = tmp_path / "demo.rescored.json"
+ out.write_bytes(b"known-good")
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setenv("JUDGE_BACKEND", "openrouter")
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(R, "preflight", lambda: pytest.fail("Agy preflight ran under wrong backend"))
+ monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring ran under wrong backend"))
+
+ with pytest.raises(RuntimeError, match="JUDGE_BACKEND=agy"):
+ R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ assert out.read_bytes() == b"known-good"
+
+
+def test_historical_scorers_do_not_block_one_current_agy_rescore(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ first = metadata("codex@qwen--dell-a", fingerprint="a" * 64)
+ second = metadata("codex@qwen--dell-b", fingerprint="b" * 64)
+ second.update({
+ "judge": "other/historical-judge",
+ "scoring_revision": "harbor-rubric-v0",
+ "judge_runtime": "old runtime",
+ "judge_billing_mode": "metered",
+ "cost_usd": 1.23,
+ })
+ write_combo(root, "first", first)
+ write_combo(root, "second", second)
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ preflights = []
+ monkeypatch.setattr(R, "preflight", lambda: preflights.append(True) or {
+ "identity": R.AGY_IDENTITY,
+ "model": "gemini-3.6-flash-medium",
+ "version": RUNTIME,
+ "billing_mode": "subscription",
+ })
+ score_calls = []
+
+ def fake_score(answers, *_args):
+ score_calls.append(answers)
+ return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4]
+
+ monkeypatch.setattr(R, "score", fake_score)
+
+ summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ assert preflights == [True]
+ assert len(score_calls) == 4
+ assert summary["judge"] == R.AGY_IDENTITY
+ assert summary["scoring_revision"] == "harbor-rubric-v2-agy"
+ assert summary["judge_runtime"] == RUNTIME
+ assert summary["judge_billing_mode"] == "subscription"
+ for row in summary["combinations"].values():
+ assert row["judge"] == R.AGY_IDENTITY
+ assert row["scoring_revision"] == "harbor-rubric-v2-agy"
+ assert row["judge_runtime"] == RUNTIME
+ assert row["judge_billing_mode"] == "subscription"
+ assert "cost_usd" not in row
+ assert "cost_usd" not in summary
+
+
+def test_stale_measurements_cannot_make_an_agy_failure_rankable(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ stale = {
+ "error": "historical failure",
+ "score": 0.9,
+ "scores": [0.9],
+ "skill_mean": 0.9,
+ "control_mean": 0.1,
+ "lift": 0.8,
+ "skill_scores": [0.9],
+ "control_scores": [0.1],
+ "tasks_scored": 99,
+ "tasks_dropped": ["old-task"],
+ "mean_lift": 0.8,
+ "scored": 99,
+ "unscorable": 0,
+ "n": 99,
+ "dropped": ["old-task"],
+ "best": {"combination": "historical"},
+ "measured": 99,
+ "unmeasured": 0,
+ }
+ failed = {**metadata("codex@qwen--dell-failed", fingerprint="f" * 64), **stale}
+ measured = metadata("codex@qwen--dell-ok", fingerprint="o" * 64)
+ write_combo(root, "failed", failed, artifact="fail")
+ write_combo(root, "measured", measured)
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+
+ def score_one(answers, *_args):
+ if "fail" in "\n".join(sum(answers.values(), [])):
+ raise A.AgyJudgeError("current Agy failure")
+ return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4]
+
+ monkeypatch.setattr(R, "score", score_one)
+
+ summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ failed_row = summary["combinations"][failed["combination"]]
+ assert failed_row["error"] == "current Agy failure"
+ assert not (set(stale) - {"error"}) & set(failed_row)
+ assert summary["scored"] == 1
+ assert summary["unscorable"] == 1
+ assert summary["mean_lift"] == pytest.approx(0.4)
+
+
+def test_rescore_rejects_missing_recorded_attempt_before_scoring(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ incomplete = metadata("codex@qwen--dell-incomplete", fingerprint="i" * 64)
+ measured = metadata("codex@qwen--dell-ok", fingerprint="o" * 64)
+ write_combo(root, "incomplete", incomplete, recorded_attempts=2)
+ write_combo(root, "measured", measured)
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ score_calls = []
+
+ def score_one(answers, *_args):
+ score_calls.append(answers)
+ return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4]
+
+ monkeypatch.setattr(R, "score", score_one)
+
+ summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ row = summary["combinations"][incomplete["combination"]]
+ assert "attempt" in row["error"].lower()
+ assert not {"skill_scores", "control_scores", "skill_mean", "control_mean", "lift"} & set(row)
+ assert len(score_calls) == 2
+ assert summary["scored"] == 1
+ assert summary["unscorable"] == 1
+
+
+def test_distinct_endpoint_fingerprints_remain_separate_measured_rows(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ first = metadata("codex@qwen--dell-a", fingerprint="a" * 64)
+ second = metadata("codex@qwen--dell-b", fingerprint="b" * 64)
+ write_combo(root, "first", first)
+ write_combo(root, "second", second)
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ assert set(summary["combinations"]) == {first["combination"], second["combination"]}
+ assert summary["combinations"][first["combination"]]["endpoint_fingerprint"] == "a" * 64
+ assert summary["combinations"][second["combination"]]["endpoint_fingerprint"] == "b" * 64
+ assert summary["scored"] == 2
+
+
+def test_rescore_deduplicates_a_combination_when_the_same_root_is_repeated(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ record = metadata("codex@qwen--dell", fingerprint="d" * 64)
+ write_combo(root, "one", record)
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ summary = R.rescore("demo", jobs_roots=[root, root], log=lambda *_args: None)
+
+ assert list(summary["combinations"]) == [record["combination"]]
+ assert summary["scored"] == 1
+
+
+def test_rescore_selects_only_exact_completed_combination_paths(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ complete = metadata("aider@qwen--dell", fingerprint="a" * 64)
+ active = metadata("codex@qwen--dell", fingerprint="b" * 64)
+ write_combo(root, "complete", complete)
+ active_dir = root / "active"
+ active_dir.mkdir(parents=True)
+ (active_dir / "combo.json").write_text("active evidence must not be read")
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ summary = R.rescore(
+ "demo", jobs_roots=[root], combination_paths=[root / "complete"],
+ log=lambda *_args: None)
+
+ assert list(summary["combinations"]) == [complete["combination"]]
+ assert summary["scored"] == 1
+
+
+def test_rescore_rejects_selected_path_outside_jobs_roots_before_preflight(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ write_combo(root, "complete", metadata("aider@qwen--dell", fingerprint="a" * 64))
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(R, "preflight", lambda: pytest.fail("preflight must not start"))
+
+ outside = tmp_path / "outside"
+ write_combo(outside, "other", metadata("other@qwen--dell", fingerprint="b" * 64))
+
+ with pytest.raises(ValueError, match="immediate child"):
+ R.rescore("demo", jobs_roots=[root], combination_paths=[outside / "other"],
+ log=lambda *_args: None)
+
+
+def test_explicit_empty_selection_never_discovers_active_siblings(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ active = root / "active"
+ active.mkdir(parents=True)
+ (active / "combo.json").write_text("must not be read")
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(R, "preflight", lambda: pytest.fail("preflight must not start"))
+
+ with pytest.raises(ValueError, match="no completed combination paths selected"):
+ R.rescore("demo", jobs_roots=[root], combination_paths=[], log=lambda *_args: None)
+
+
+def test_rescore_checkpoints_each_new_lift_to_explicit_output(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ first = metadata("aider@qwen--dell", fingerprint="a" * 64)
+ second = metadata("codex@qwen--dell", fingerprint="b" * 64)
+ write_combo(root, "first", first)
+ write_combo(root, "second", second)
+ output = tmp_path / "published" / "build-loop.rescored.json"
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+ writes = []
+ real_write = R.atomic_write_json
+
+ def record_write(path, payload):
+ writes.append((path, list(payload["combinations"])))
+ real_write(path, payload)
+
+ monkeypatch.setattr(R, "atomic_write_json", record_write)
+
+ summary = R.rescore("demo", jobs_roots=[root], output=output, log=lambda *_args: None)
+
+ assert writes == [
+ (output, [first["combination"]]),
+ (output, [first["combination"], second["combination"]]),
+ ]
+ assert json.loads(output.read_text()) == summary
+
+
+def test_rescore_requests_bounded_parallel_agy_grading(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ write_combo(root, "one", metadata("aider@qwen--dell", fingerprint="a" * 64))
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ concurrencies = []
+
+ def fake_score(_answers, _skill, _holdout, _skipped, concurrency):
+ concurrencies.append(concurrency)
+ return [0.5]
+
+ monkeypatch.setattr(R, "score", fake_score)
+
+ R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ assert concurrencies == [4, 4]
+
+
+def test_selected_rescore_preserves_prior_compatible_lifts(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ first = metadata("aider@qwen--dell", fingerprint="a" * 64)
+ second = metadata("codex@qwen--dell", fingerprint="b" * 64)
+ first_path = write_combo(root, "first", first)
+ second_path = write_combo(root, "second", second)
+ output = tmp_path / "published" / "build-loop.rescored.json"
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+ original_score = R.score
+ score_calls = []
+
+ def counted_score(*args, **kwargs):
+ score_calls.append(args[0])
+ return original_score(*args, **kwargs)
+
+ monkeypatch.setattr(R, "score", counted_score)
+
+ R.rescore("demo", jobs_roots=[root], combination_paths=[first_path], output=output,
+ log=lambda *_args: None)
+ summary = R.rescore("demo", jobs_roots=[root], combination_paths=[second_path], output=output,
+ log=lambda *_args: None)
+
+ assert list(summary["combinations"]) == [first["combination"], second["combination"]]
+ assert summary["scored"] == 2
+ assert len(score_calls) == 4 # two arms once per cell; the first cell is not paid twice
+
+
+def test_current_canary_failure_replaces_its_prior_lift_without_rejudging(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ failed = metadata("aider@qwen--dell", fingerprint="a" * 64)
+ unaffected = metadata("codex@qwen--dell", fingerprint="b" * 64)
+ failed_path = write_combo(root, "failed", failed)
+ unaffected_path = write_combo(root, "unaffected", unaffected)
+ output = tmp_path / "published" / "demo.rescored.json"
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ prior = R.rescore(
+ "demo", jobs_roots=[root], combination_paths=[failed_path, unaffected_path],
+ output=output, log=lambda *_args: None,
+ )
+ unaffected_row = prior["combinations"][unaffected["combination"]]
+ (failed_path / "combo.json").write_text(json.dumps({
+ **failed,
+ "canary_error": "current canary failed",
+ }))
+ monkeypatch.setattr(R, "score", lambda *_args, **_kwargs:
+ pytest.fail("current canary failure invoked the judge"))
+
+ summary = R.rescore(
+ "demo", jobs_roots=[root], combination_paths=[failed_path],
+ output=output, log=lambda *_args: None,
+ )
+
+ failed_row = summary["combinations"][failed["combination"]]
+ assert failed_row["error"] == "current canary failed"
+ assert not {"lift", "skill_mean", "control_mean", "skill_scores", "control_scores"} & set(
+ failed_row)
+ assert summary["combinations"][unaffected["combination"]] == unaffected_row
+ assert summary["scored"] == 1
+ assert summary["mean_lift"] == unaffected_row["lift"]
+
+
+def test_selected_rescore_preserves_scale_provenance_for_report(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ record = metadata("aider@qwen35-4b--dell", fingerprint="4" * 64)
+ record.update({"family": "Qwen3.5", "parameter_billions": 4.0,
+ "quantization": "fp8-load", "tool_parser": "qwen3_coder",
+ "exploratory": True, "rankable": False})
+ combo = write_combo(root, "qwen35-4b", record)
+ output_root = tmp_path / "published"
+ output = output_root / "demo.rescored.json"
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ R.rescore("demo", jobs_roots=[root], combination_paths=[combo], output=output,
+ log=lambda *_args: None)
+ matrix = harbor_report.read_matrix("demo", output_root)
+ row = matrix["rows"][0]
+
+ assert {key: row[key] for key in (
+ "family", "parameter_billions", "quantization", "tool_parser")
+ } == {"family": "Qwen3.5", "parameter_billions": 4.0,
+ "quantization": "fp8-load", "tool_parser": "qwen3_coder"}
+ assert matrix["rankable"] is False
+ assert matrix["exploratory"] is True
+
+
+def test_prior_row_without_treatment_fields_does_not_reuse_lift():
+ identity = metadata("aider@qwen--dell", fingerprint="a" * 64)
+ identity.update({"family": "Qwen3.5", "parameter_billions": 4.0,
+ "quantization": "fp8-load", "tool_parser": "qwen3_coder"})
+ prior = {"prior": {key: value for key, value in identity.items()
+ if key not in {"family", "parameter_billions", "quantization",
+ "tool_parser"}}}
+
+ assert R._matching_row_key(identity, prior) is None
+
+
+def test_prior_row_with_stale_gateway_revision_does_not_reuse_lift():
+ identity = metadata("codex@qwen--dell", fingerprint="a" * 64)
+ identity.update({"gateway_revision": "v8", "gateway_identity": "route-v8"})
+ prior = {"prior": {**identity, "gateway_revision": "v7", "lift": 0.5}}
+
+ assert R._matching_row_key(identity, prior) is None
+
+
+def test_current_runtime_replaces_stale_prior_row_for_same_logical_cell(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ current = metadata("codex@qwen--dell", fingerprint="a" * 64)
+ current.update({"gateway_revision": f"route-v8-{R.NATIVE_RUNNER_REVISION}",
+ "gateway_identity": f"route-v8-{R.NATIVE_RUNNER_REVISION}"})
+ combo = write_combo(root, "codex-qwen", current)
+ output = tmp_path / "demo.rescored.json"
+ stale = {**current, "gateway_revision": "route-v8-native-v2",
+ "gateway_identity": "route-v8-native-v2", "lift": 0.9,
+ "skill_mean": 1.0, "control_mean": 0.1}
+ output.write_text(json.dumps({
+ "skill": "demo", "task_fingerprint": current["task_fingerprint"],
+ "attempts": current["attempts"], "judge": R.AGY_IDENTITY,
+ "scoring_revision": "harbor-rubric-v2-agy", "judge_billing_mode": "subscription",
+ "judge_runtime": RUNTIME, "combinations": {"stale": stale},
+ }))
+ patch_scoring(monkeypatch)
+ calls = []
+ monkeypatch.setattr(R, "score", lambda *_args: calls.append(1) or [0.6])
+
+ result = R.rescore("demo", jobs_roots=[root], combination_paths=[combo], output=output,
+ log=lambda *_args: None)
+
+ assert len(calls) == 2
+ assert result["scored"] == 1
+ assert len(result["combinations"]) == 1
+ [row] = result["combinations"].values()
+ assert row["gateway_revision"] == f"route-v8-{R.NATIVE_RUNNER_REVISION}"
+ assert row["lift"] == 0.0
+
+
+def test_current_runtime_wins_over_later_stale_root_independent_of_input_order(tmp_path,
+ monkeypatch):
+ current_root, stale_root = tmp_path / "current", tmp_path / "stale"
+ base = metadata("codex@qwen--dell", fingerprint="a" * 64)
+ current = {**base, "gateway_revision": f"route-v8-{R.NATIVE_RUNNER_REVISION}",
+ "gateway_identity": f"route-v8-{R.NATIVE_RUNNER_REVISION}"}
+ stale = {**base, "gateway_revision": "route-v8-native-v2",
+ "gateway_identity": "route-v8-native-v2"}
+ write_combo(current_root, "current", current)
+ write_combo(stale_root, "stale", stale)
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path / "published")
+ patch_scoring(monkeypatch)
+
+ result = R.rescore("demo", jobs_roots=[current_root, stale_root], log=lambda *_args: None)
+
+ assert result["scored"] == 1
+ assert len(result["combinations"]) == 1
+ [row] = result["combinations"].values()
+ assert row["gateway_revision"] == f"route-v8-{R.NATIVE_RUNNER_REVISION}"
+
+
+def test_selected_rescore_refuses_incompatible_prior_output(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ combo = write_combo(root, "one", metadata("aider@qwen--dell", fingerprint="a" * 64))
+ output = tmp_path / "build-loop.rescored.json"
+ output.write_text(json.dumps({
+ "skill": "demo", "task_fingerprint": "other", "attempts": 3,
+ "judge": R.AGY_IDENTITY, "scoring_revision": "harbor-rubric-v2-agy",
+ "judge_billing_mode": "subscription", "judge_runtime": RUNTIME,
+ "combinations": {},
+ }))
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+ monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start"))
+
+ with pytest.raises(ValueError, match="incompatible existing rescore output"):
+ R.rescore("demo", jobs_roots=[root], combination_paths=[combo], output=output,
+ log=lambda *_args: None)
+
+
+def test_selected_rescore_drops_prior_lifts_from_another_skill_revision(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ first = metadata("aider@qwen--dell", fingerprint="a" * 64)
+ second = metadata("codex@qwen--dell", fingerprint="b" * 64)
+ second["skill_body"] = "different skill revision\n"
+ second["skill_sha256"] = hashlib.sha256(second["skill_body"].encode()).hexdigest()
+ first_path = write_combo(root, "first", first)
+ second_path = write_combo(root, "second", second)
+ output = tmp_path / "build-loop.rescored.json"
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ R.rescore("demo", jobs_roots=[root], combination_paths=[first_path], output=output,
+ log=lambda *_args: None)
+ summary = R.rescore("demo", jobs_roots=[root], combination_paths=[second_path], output=output,
+ log=lambda *_args: None)
+
+ assert list(summary["combinations"]) == [second["combination"]]
+
+
+def test_agy_failure_row_has_current_identity_and_no_measurements(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ failed = metadata("codex@qwen--dell-failed", fingerprint="f" * 64)
+ measured = metadata("codex@qwen--dell-ok", fingerprint="o" * 64)
+ write_combo(root, "failed", failed, artifact="fail")
+ write_combo(root, "measured", measured)
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+
+ def score_one(answers, *_args):
+ if "fail" in "\n".join(sum(answers.values(), [])):
+ raise A.AgyJudgeError("judge unavailable")
+ return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4]
+
+ monkeypatch.setattr(R, "score", score_one)
+ summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ failed_row = summary["combinations"][failed["combination"]]
+ assert failed_row["error"] == "judge unavailable"
+ assert failed_row["judge"] == R.AGY_IDENTITY
+ assert failed_row["scoring_revision"] == "harbor-rubric-v2-agy"
+ assert failed_row["judge_runtime"] == RUNTIME
+ assert failed_row["judge_billing_mode"] == "subscription"
+ assert not {"score", "skill_scores", "control_scores", "skill_mean", "control_mean", "lift"} & set(failed_row)
+ row = summary["combinations"][measured["combination"]]
+ assert row["attempts"] == 3
+ assert row["target_alias"] == "dell"
+ assert row["endpoint_fingerprint"] == "o" * 64
+ assert row["protocol"] == "openai"
+
+
+def test_rescore_preserves_failed_canary_as_explicit_unmeasured_row(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ failed = {**metadata("codex@qwen--dell-failed", fingerprint="f" * 64),
+ "canary_error": "ApiRateLimitError: local gateway returned 429"}
+ measured = metadata("aider@qwen--dell-ok", fingerprint="o" * 64)
+ failed_dir = root / "failed"
+ failed_dir.mkdir(parents=True)
+ (failed_dir / "combo.json").write_text(json.dumps(failed))
+ write_combo(root, "measured", measured)
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ row = summary["combinations"][failed["combination"]]
+ assert row["error"] == failed["canary_error"]
+ assert "canary_error" not in row
+ assert not {"skill_mean", "control_mean", "lift", "skill_scores", "control_scores"} & set(row)
+ assert summary["scored"] == 1
+ assert summary["unscorable"] == 1
+
+
+def test_rescore_publishes_trailing_failed_canary_in_final_summary(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ measured = metadata("aider@qwen--dell-ok", fingerprint="o" * 64)
+ failed = {**metadata("codex@qwen--dell-failed", fingerprint="f" * 64),
+ "canary_error": "canary failed"}
+ measured_path = write_combo(root, "measured", measured)
+ failed_path = root / "failed"
+ failed_path.mkdir(parents=True)
+ (failed_path / "combo.json").write_text(json.dumps(failed))
+ output = tmp_path / "demo.rescored.json"
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ summary = R.rescore(
+ "demo", jobs_roots=[root], combination_paths=[measured_path, failed_path],
+ output=output, log=lambda *_args: None)
+
+ assert json.loads(output.read_text()) == summary
+ assert summary["combinations"][failed["combination"]]["error"] == "canary failed"
+
+
+def test_rescore_admits_explicit_partial_arm_measurement_error(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ measured = metadata("aider@qwen--dell-ok", fingerprint="o" * 64)
+ partial = {**metadata("codex@qwen--dell-partial", fingerprint="p" * 64),
+ "measurement_error": "control arm failed"}
+ measured_path = write_combo(root, "measured", measured)
+ partial_path = root / "partial"
+ partial_path.mkdir(parents=True)
+ (partial_path / "combo.json").write_text(json.dumps(partial))
+ (partial_path / "skill").mkdir()
+ output = tmp_path / "demo.rescored.json"
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ patch_scoring(monkeypatch)
+
+ summary = R.rescore(
+ "demo", jobs_roots=[root], combination_paths=[measured_path, partial_path],
+ output=output, log=lambda *_args: None)
+
+ row = summary["combinations"][partial["combination"]]
+ assert row["error"] == "control arm failed"
+ assert "lift" not in row
+
+
+def test_rescore_refuses_to_publish_when_every_combination_is_unscorable(tmp_path, monkeypatch):
+ root = tmp_path / "jobs"
+ write_combo(root, "failed", metadata("codex@qwen--dell", fingerprint="f" * 64))
+ out = tmp_path / "demo.rescored.json"
+ out.write_bytes(b"known-good")
+ monkeypatch.setattr(R, "HARBOR_DIR", tmp_path)
+ holdout = patch_current_facts(monkeypatch)
+ monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {}))
+ monkeypatch.setattr(R, "score", lambda *_args: (_ for _ in ()).throw(RuntimeError("judge down")))
+
+ with pytest.raises(SystemExit, match="nothing was measured"):
+ R.rescore("demo", jobs_roots=[root], log=lambda *_args: None)
+
+ assert out.read_bytes() == b"known-good"
+
+
+def test_atomic_write_preserves_existing_bytes_on_serialization_and_replace_failure(tmp_path, monkeypatch):
+ path = tmp_path / "matrix.json"
+ path.write_bytes(b"known-good")
+
+ with pytest.raises(TypeError):
+ R.atomic_write_json(path, {"bad": {1, 2}})
+ assert path.read_bytes() == b"known-good"
+ assert not list(tmp_path.glob(".matrix.json.*.tmp"))
+
+ def broken_replace(source, destination):
+ assert json.loads(Path(source).read_text()) == {"next": 1}
+ raise OSError("disk failed")
+
+ monkeypatch.setattr(R.os, "replace", broken_replace)
+ with pytest.raises(OSError, match="disk failed"):
+ R.atomic_write_json(path, {"next": 1})
+ assert path.read_bytes() == b"known-good"
+ assert not list(tmp_path.glob(".matrix.json.*.tmp"))
+
+
+def test_atomic_write_replaces_only_a_complete_json_file(tmp_path, monkeypatch):
+ path = tmp_path / "matrix.json"
+ seen = []
+ original_replace = R.os.replace
+
+ def inspect_then_replace(source, destination):
+ seen.append(json.loads(Path(source).read_text()))
+ original_replace(source, destination)
+
+ monkeypatch.setattr(R.os, "replace", inspect_then_replace)
+ R.atomic_write_json(path, {"next": [1, 2]})
+
+ assert seen == [{"next": [1, 2]}]
+ assert json.loads(path.read_text()) == {"next": [1, 2]}
+ assert not list(tmp_path.glob(".matrix.json.*.tmp"))
+
+
+def test_native_combo_resolves_exact_sibling_job_and_arm_identity(tmp_path):
+ from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env
+
+ root = tmp_path / "jobs"
+ combo = root / "aider-cell"
+ job = root / "native-full"
+ combo.mkdir(parents=True)
+ job.mkdir()
+ common = dict(combination_id="aider@dot-backbone--dell-qwen-deadbeefcafe",
+ endpoint_fingerprint="deadbeefcafe", harness="aider", protocol="chat",
+ gateway_revision="direct")
+ identities = {arm: NativeTrialIdentity(**common, arm=arm)
+ for arm in ("skill", "control")}
+ (combo / "combo.json").write_text(json.dumps({
+ "combination": common["combination_id"],
+ "harness": common["harness"],
+ "endpoint_fingerprint": common["endpoint_fingerprint"],
+ "protocol": common["protocol"],
+ "gateway_revision": common["gateway_revision"],
+ "native_job": "native-full",
+ "native_identities": {arm: identity_env(identity)
+ for arm, identity in identities.items()},
+ }))
+
+ assert R._arm_evidence(combo, "skill") == (job, identities["skill"])
+ assert R._arm_evidence(combo, "control") == (job, identities["control"])
+
+
+def test_native_combo_refuses_lock_identity_mismatched_to_combo_metadata(tmp_path):
+ from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env
+
+ combo = tmp_path / "jobs" / "cell"
+ combo.mkdir(parents=True)
+ common = dict(combination_id="aider@dot-backbone--dell-qwen-deadbeefcafe",
+ endpoint_fingerprint="deadbeefcafe", harness="aider", protocol="chat",
+ gateway_revision="direct")
+ (combo / "combo.json").write_text(json.dumps({
+ "combination": "aider@other--dell-qwen-deadbeefcafe",
+ "harness": "aider", "endpoint_fingerprint": "deadbeefcafe",
+ "protocol": "chat", "gateway_revision": "direct", "native_job": "native-full",
+ "native_identities": {
+ arm: identity_env(NativeTrialIdentity(**common, arm=arm))
+ for arm in ("skill", "control")},
+ }))
+
+ with pytest.raises(ValueError, match="does not match combo metadata"):
+ R._arm_evidence(combo, "skill")
+
+
+@pytest.mark.parametrize("native_job", ["../other", "/tmp/other", "http:job"])
+def test_native_combo_refuses_non_sibling_job_reference(tmp_path, native_job):
+ combo = tmp_path / "jobs" / "cell"
+ combo.mkdir(parents=True)
+ (combo / "combo.json").write_text(json.dumps({
+ "native_job": native_job,
+ "native_identities": {},
+ }))
+
+ with pytest.raises(ValueError, match="native job"):
+ R._arm_evidence(combo, "skill")
+
+
+def test_discovery_skips_only_native_job_declared_by_a_combo(tmp_path):
+ root = tmp_path / "jobs"
+ combo = root / "cell"
+ native = root / "native-full"
+ combo.mkdir(parents=True)
+ native.mkdir()
+ (combo / "combo.json").write_text(json.dumps({
+ **metadata("aider@dot-backbone--dell-qwen-deadbeefcafe",
+ fingerprint="deadbeefcafe"),
+ "native_job": "native-full",
+ }))
+
+ assert R.discover_combinations([root]) == [combo]
+
+ (root / "native-full--aider--other").mkdir()
+ assert R.discover_combinations([root]) == [combo]
+
+
+def test_discovery_rejects_a_claimed_combo_with_malformed_identity(tmp_path):
+ root = tmp_path / "jobs"
+ malformed = root / "malformed"
+ malformed.mkdir(parents=True)
+ (malformed / "combo.json").write_text("[]")
+
+ with pytest.raises(ValueError, match="no combo identity"):
+ R.discover_combinations([root])
diff --git a/tests/test_harbor_rescore_exploratory.py b/tests/test_harbor_rescore_exploratory.py
new file mode 100644
index 0000000..b186701
--- /dev/null
+++ b/tests/test_harbor_rescore_exploratory.py
@@ -0,0 +1,17 @@
+import pytest
+
+from ingot.optimize import harbor_rescore as R
+
+
+def test_current_compatibility_accepts_only_explicit_k1_exploration():
+ holdout = [{"task": "one"}]
+ fingerprint = R._task_fingerprint(holdout)
+ R._validate_current_compatibility([
+ {"task_fingerprint": fingerprint, "attempts": 1,
+ "exploratory": True, "rankable": False}
+ ], holdout)
+ with pytest.raises(ValueError, match="measurement contract"):
+ R._validate_current_compatibility([
+ {"task_fingerprint": fingerprint, "attempts": 1,
+ "exploratory": False, "rankable": True}
+ ], holdout)
diff --git a/tests/test_harbor_targets.py b/tests/test_harbor_targets.py
new file mode 100644
index 0000000..2fa7385
--- /dev/null
+++ b/tests/test_harbor_targets.py
@@ -0,0 +1,375 @@
+"""Tests for local Harbor target identity, discovery, routing, and isolation."""
+from __future__ import annotations
+
+import json
+from pathlib import Path
+
+import pytest
+
+from ingot.optimize import harbor_targets as H
+
+
+FIXTURES = Path(__file__).parent / "fixtures" / "harbor"
+TARGETS = {
+ "dell-qwen": ("dot-backbone", 163840, FIXTURES / "qwen-models.json"),
+ "spark-deepseek": ("deepseek-v4-flash", 1048576, FIXTURES / "deepseek-models.json"),
+ "orin-abliterated": ("ablit35b", 65536, FIXTURES / "orin-models.json"),
+}
+HARNESS_PROTOCOLS = {
+ "claude-code": "messages",
+ "terminus-2": "chat",
+ "goose": "chat",
+ "opencode": "chat",
+ "openclaw": "chat",
+ "mini-swe-agent": "chat",
+ "codex": "responses",
+ "aider": "chat",
+ "pi": "chat",
+}
+
+
+class _Response:
+ def __init__(self, payload: dict, status: int | None = None):
+ self.payload = payload
+ self.status = status
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *args):
+ return False
+
+ def read(self):
+ return json.dumps(self.payload).encode()
+
+
+def _target(alias: str = "dell-qwen", **changes) -> H.LocalTarget:
+ model, context, _ = TARGETS[alias]
+ values = {
+ "alias": alias,
+ "display_name": H.TARGETS[alias]["display_name"],
+ "base_url": "http://host:8011",
+ "served_model": model,
+ "context_length": context,
+ "protocols": frozenset({"chat", "responses", "messages"}),
+ "family": H.TARGETS[alias]["family"],
+ "parameter_billions": H.TARGETS[alias]["parameter_billions"],
+ "quantization": H.TARGETS[alias]["quantization"],
+ "tool_parser": H.TARGETS[alias]["tool_parser"],
+ }
+ values.update(changes)
+ return H.LocalTarget(**values)
+
+
+def test_fixtures_preserve_served_ids_and_context_lengths():
+ for alias, (served_model, context_length, path) in TARGETS.items():
+ payload = json.loads(path.read_text())
+ model = payload["data"][0]
+ assert model["id"] == served_model
+ assert model.get("max_model_len", model.get("meta", {}).get("n_ctx")) == context_length
+ assert alias in H.TARGETS
+
+
+def test_parse_target_accepts_only_configured_aliases_and_normalizes_url():
+ target = H.parse_target("dell-qwen=http://host:8011/")
+ assert target.alias == "dell-qwen"
+ assert target.base_url == "http://host:8011"
+ assert target.served_model == "dot-backbone"
+ with pytest.raises(ValueError, match="unknown local target alias"):
+ H.parse_target("unknown=http://host:8011")
+
+
+@pytest.mark.parametrize("alias", list(TARGETS))
+def test_discovery_finds_configured_model_and_context(alias, monkeypatch):
+ served_model, context_length, fixture = TARGETS[alias]
+ payload = json.loads(fixture.read_text())
+
+ def fake_urlopen(request, timeout):
+ assert request.full_url == "http://local.test/v1/models"
+ assert timeout == 3.5
+ return _Response(payload)
+
+ monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen)
+ target = H.discover_target(alias, "http://local.test/", timeout=3.5)
+ assert target.served_model == served_model
+ assert target.context_length == context_length
+ assert target.protocols == frozenset({"chat", "responses", "messages"})
+
+
+def test_ollama_discovery_reads_loaded_runtime_context(monkeypatch):
+ seen = []
+
+ def fake_urlopen(request, timeout):
+ seen.append((request.full_url, json.loads(request.data) if request.data else None, timeout))
+ if request.full_url.endswith("/v1/models"):
+ return _Response({"object": "list", "data": [{"id": "qwen3.5:9b"}]})
+ return _Response({"models": [
+ {"name": "other", "context_length": 8192},
+ {"name": "qwen3.5:9b", "context_length": 262144},
+ ]})
+
+ monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen)
+ target = H.discover_target("orin-qwen35-9b", "http://orin.test:11434", timeout=4)
+
+ assert target.served_model == "qwen3.5:9b"
+ assert target.context_length == 262144
+ assert target.display_name == "Qwen/Qwen3.5-9B (Q4_K_M)"
+ assert seen == [
+ ("http://orin.test:11434/v1/models", None, 4),
+ ("http://orin.test:11434/api/ps", None, 4),
+ ]
+
+
+@pytest.mark.parametrize(("meta", "message"), [
+ ({"n_ctx": True}, "no context length"),
+ ({"n_ctx": "65536"}, "no context length"),
+ ({"n_ctx": 32767}, "below the minimum"),
+ ({}, "no context length"),
+ ([], "no context length"),
+])
+def test_discovery_rejects_invalid_llamacpp_context_metadata(meta, message, monkeypatch):
+ payload = {"object": "list", "data": [{"id": "ablit35b", "meta": meta}]}
+ monkeypatch.setattr(H.urllib.request, "urlopen", lambda *args, **kwargs: _Response(payload))
+ with pytest.raises(ValueError, match=message):
+ H.discover_target("orin-abliterated", "http://local.test")
+
+
+def test_discovery_rejects_short_context_and_wrong_served_model(monkeypatch):
+ payload = {"object": "list", "data": [{"id": "dot-backbone", "max_model_len": 8192}]}
+ monkeypatch.setattr(H.urllib.request, "urlopen", lambda *args, **kwargs: _Response(payload))
+ with pytest.raises(ValueError, match="context length"):
+ H.discover_target("dell-qwen", "http://local.test")
+
+ payload["data"][0] = {"id": "other", "max_model_len": 32768}
+ with pytest.raises(ValueError, match="served model"):
+ H.discover_target("dell-qwen", "http://local.test")
+
+
+def test_fingerprint_changes_with_normalized_url_or_served_model():
+ base = _target()
+ assert base.fingerprint == "a7f8512ae664" # pre-scale-metadata identity stays readable
+ assert base.fingerprint == _target(base_url="http://host:8011/").fingerprint
+ assert base.fingerprint != _target(base_url="http://host:8002").fingerprint
+ assert base.fingerprint != _target(served_model="dot-backbone-other").fingerprint
+ assert base.fingerprint == _target(family="Qwen-next").fingerprint
+ assert base.fingerprint == _target(parameter_billions=99.0).fingerprint
+ assert base.fingerprint == _target(quantization="fp8-load").fingerprint
+ assert base.fingerprint == _target(tool_parser="qwen3_coder").fingerprint
+ assert base.alias in base.job_slug
+ assert base.fingerprint in base.job_slug
+ assert "http" not in base.job_slug and "host" not in base.job_slug
+
+
+def test_qwen_size_targets_have_exact_scale_provenance():
+ assert {
+ alias: {key: config[key] for key in (
+ "served_model", "family", "parameter_billions", "quantization", "tool_parser")}
+ for alias, config in H.TARGETS.items() if alias.startswith("qwen35-")
+ } == {
+ "qwen35-08b": {"served_model": "qwen35-0.8b", "family": "Qwen3.5",
+ "parameter_billions": 0.8, "quantization": "fp8-load",
+ "tool_parser": "qwen3_coder"},
+ "qwen35-2b": {"served_model": "qwen35-2b", "family": "Qwen3.5",
+ "parameter_billions": 2.0, "quantization": "fp8-load",
+ "tool_parser": "qwen3_coder"},
+ "qwen35-4b": {"served_model": "qwen35-4b", "family": "Qwen3.5",
+ "parameter_billions": 4.0, "quantization": "fp8-load",
+ "tool_parser": "qwen3_coder"},
+ "qwen35-9b": {"served_model": "qwen35-9b", "family": "Qwen3.5",
+ "parameter_billions": 9.0, "quantization": "fp8-load",
+ "tool_parser": "qwen3_coder"},
+ }
+ assert {key: H.TARGETS["dell-qwen"][key] for key in (
+ "family", "parameter_billions", "quantization", "tool_parser")
+ } == {"family": "Qwen3.6", "parameter_billions": 27.0,
+ "quantization": "fp8-published", "tool_parser": "qwen3_xml"}
+ assert {key: H.TARGETS["orin-qwen35-9b"][key] for key in (
+ "served_model", "family", "parameter_billions", "quantization", "tool_parser")
+ } == {"served_model": "qwen3.5:9b", "family": "Qwen3.5",
+ "parameter_billions": 9.7, "quantization": "Q4_K_M",
+ "tool_parser": "ollama"}
+
+
+@pytest.mark.parametrize("alias", ["qwen35-08b", "qwen35-2b", "qwen35-4b", "qwen35-9b"])
+def test_qwen_size_aliases_propagate_provenance_through_parse(alias):
+ target = H.parse_target(f"{alias}=http://local.test:8020")
+ config = H.TARGETS[alias]
+ assert {field: getattr(target, field) for field in (
+ "family", "parameter_billions", "quantization", "tool_parser")
+ } == {field: config[field] for field in (
+ "family", "parameter_billions", "quantization", "tool_parser")}
+
+
+@pytest.mark.parametrize("harness, protocol", list(HARNESS_PROTOCOLS.items()))
+def test_harnesses_map_to_the_required_protocol(harness, protocol):
+ assert H.protocol_for(harness) == protocol
+ expected = "dot-backbone" if harness == "claude-code" else "openai/dot-backbone"
+ if harness == "opencode":
+ expected = "local/dot-backbone"
+ assert H.harbor_model(_target(), harness) == expected
+
+
+def test_harbor_kwargs_use_direct_api_base_only_for_terminus():
+ target = _target()
+ assert H.harbor_agent_kwargs(target, "terminus-2") == {"api_base": f"{target.base_url}/v1"}
+ assert H.harbor_agent_kwargs(target, "openclaw") == {"thinking": "off"}
+ assert H.harbor_model(target, "opencode") == "local/dot-backbone"
+ assert H.harbor_agent_kwargs(target, "opencode") == {
+ "opencode_config": {
+ "provider": {
+ "local": {
+ "npm": "@ai-sdk/openai-compatible",
+ "options": {"baseURL": f"{target.base_url}/v1", "apiKey": "local"},
+ "models": {"dot-backbone": {
+ "limit": {"context": 163840, "output": 40960},
+ }},
+ }
+ }
+ }
+ }
+ assert H.harbor_agent_kwargs(target, "claude-code") == {}
+
+
+def test_scrub_provider_env_removes_credentials_and_provider_routing():
+ parent = {
+ "PATH": "/bin",
+ "ANTHROPIC_AUTH_TOKEN": "secret-a",
+ "CLAUDE_CODE_OAUTH_TOKEN": "secret-b",
+ "CLAUDE_FORCE_OAUTH": "1",
+ "ANTHROPIC_BASE_URL": "https://provider.invalid",
+ "OPENAI_API_KEY": "secret-c",
+ "CODEX_API_KEY": "secret-d",
+ "CODEX_FORCE_AUTH_JSON": "1",
+ "OPENAI_BASE_URL": "https://provider.invalid",
+ "LITELLM_API_KEY": "secret-e",
+ "GEMINI_API_KEY": "secret-f",
+ "OPENROUTER_API_KEY": "secret-g",
+ "API_KEY": "secret-h",
+ "MODEL_API_KEY": "secret-i",
+ }
+ scrubbed = H.scrub_provider_env(parent)
+ assert scrubbed == {"PATH": "/bin"}
+
+
+@pytest.mark.parametrize("harness, key, base, model", [
+ ("claude-code", "ANTHROPIC_API_KEY", "ANTHROPIC_BASE_URL", "ANTHROPIC_MODEL"),
+ ("codex", "OPENAI_API_KEY", "OPENAI_BASE_URL", None),
+ ("goose", "OPENAI_API_KEY", "OPENAI_BASE_URL", None),
+])
+def test_local_environment_uses_sentinel_key_and_explicit_routing(
+ harness, key, base, model, monkeypatch
+):
+ monkeypatch.setenv("ANTHROPIC_AUTH_TOKEN", "secret")
+ monkeypatch.setenv("OPENAI_API_KEY", "secret")
+ monkeypatch.setenv("OPENAI_BASE_URL", "https://wrong.invalid")
+ monkeypatch.setenv("UNRELATED_AMBIENT_SECRET", "must-not-reach-harbor")
+ target = _target()
+ env = H.local_agent_env(target, harness)
+ assert env[key] == "local"
+ assert env["ANTHROPIC_API_KEY"] == "local"
+ assert env["OPENAI_API_KEY"] == "local"
+ assert env["CODEX_API_KEY"] == "local"
+ expected_base = target.base_url if harness == "claude-code" else f"{target.base_url}/v1"
+ assert env[base] == expected_base
+ if harness != "claude-code":
+ assert env["OPENAI_API_BASE"] == expected_base
+ assert env["OPENAI_HOST"] == target.base_url
+ if model:
+ assert env[model] == target.served_model
+ assert "ANTHROPIC_AUTH_TOKEN" not in env
+ assert "UNRELATED_AMBIENT_SECRET" not in env
+ assert env.get("CODEX_API_KEY", "local") == "local"
+
+
+def test_probe_protocol_posts_one_token_and_requires_nonempty_object(monkeypatch):
+ seen = {}
+
+ def fake_urlopen(request, timeout):
+ seen.update(url=request.full_url, timeout=timeout, body=json.loads(request.data))
+ return _Response({"id": "response-1", "object": "response"})
+
+ monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen)
+ H.probe_protocol(_target(), "responses", timeout=7.0)
+ assert seen["url"] == "http://host:8011/v1/responses"
+ assert seen["timeout"] == 7.0
+ assert seen["body"]["model"] == "dot-backbone"
+ assert seen["body"]["max_output_tokens"] == 1
+
+ monkeypatch.setattr(H.urllib.request, "urlopen", lambda *args, **kwargs: _Response({}))
+ with pytest.raises(RuntimeError, match="non-empty"):
+ H.probe_protocol(_target(), "chat")
+
+
+def test_qwen_chat_probe_suppresses_thinking_before_the_one_token_cap(monkeypatch):
+ seen = {}
+
+ def fake_urlopen(request, timeout):
+ seen.update(body=json.loads(request.data))
+ return _Response({"choices": [{"message": {"content": "ready"}}]})
+
+ monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen)
+ H.probe_protocol(_target(), "chat")
+
+ assert seen["body"]["messages"] == [{"role": "user", "content": "ping /no_think"}]
+
+
+@pytest.mark.parametrize("status", [199, 300, 302, 399, 400, 500])
+def test_probe_protocol_requires_a_2xx_http_status(status, monkeypatch):
+ monkeypatch.setattr(
+ H.urllib.request,
+ "urlopen",
+ lambda *args, **kwargs: _Response({"id": "response-1"}, status=status),
+ )
+ with pytest.raises(RuntimeError, match=f"HTTP {status}"):
+ H.probe_protocol(_target(), "chat")
+
+
+def test_unsupported_harness_and_protocol_fail_closed():
+ target = _target()
+ with pytest.raises(ValueError, match="unsupported harness"):
+ H.protocol_for("not-a-harness")
+ with pytest.raises(ValueError, match="unsupported protocol"):
+ H.probe_protocol(target, "xml")
+ with pytest.raises(ValueError, match="unsupported harness"):
+ H.harbor_agent_kwargs(target, "not-a-harness")
+
+
+def test_probe_chat_tool_round_trip_disables_thinking_and_returns_tool_result(monkeypatch):
+ requests = []
+
+ def fake_urlopen(request, timeout):
+ body = json.loads(request.data)
+ requests.append((request.full_url, timeout, body))
+ if len(requests) == 1:
+ return _Response({
+ "choices": [{"message": {"role": "assistant", "content": "", "reasoning": "hidden",
+ "tool_calls": [{
+ "id": "call-1", "type": "function",
+ "function": {"name": "ingot_echo", "arguments": '{"value":"cutover-ok"}'},
+ }]}}],
+ })
+ return _Response({"choices": [{"message": {"role": "assistant", "content": "cutover-ok"}}]})
+
+ monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen)
+ H.probe_chat_tool_round_trip(_target(), timeout=9.0)
+
+ assert len(requests) == 2
+ assert all(url == "http://host:8011/v1/chat/completions" for url, _, _ in requests)
+ assert all(timeout == 9.0 for _, timeout, _ in requests)
+ assert all(body["model"] == "dot-backbone" for _, _, body in requests)
+ assert all(body["chat_template_kwargs"] == {"enable_thinking": False}
+ for _, _, body in requests)
+ assert all(body["reasoning_effort"] == "none" for _, _, body in requests)
+ assert all(body["messages"][0]["content"].endswith("/no_think")
+ for _, _, body in requests)
+ assert requests[0][2]["tool_choice"] == "required"
+ assert requests[1][2]["messages"][-2] == {
+ "role": "assistant", "content": "", "tool_calls": [{
+ "id": "call-1", "type": "function",
+ "function": {"name": "ingot_echo", "arguments": '{"value":"cutover-ok"}'},
+ }],
+ }
+ assert requests[1][2]["messages"][-1] == {
+ "role": "tool", "tool_call_id": "call-1", "content": "cutover-ok",
+ }
+ assert requests[1][2]["tools"] == requests[0][2]["tools"]
diff --git a/tests/test_ingot_review.py b/tests/test_ingot_review.py
new file mode 100644
index 0000000..bdd3192
--- /dev/null
+++ b/tests/test_ingot_review.py
@@ -0,0 +1,630 @@
+"""`ingot review` — the deterministic, offline, read-only report.
+
+Named for the CLI, not for `ingot.optimize.review`, which is the model-graded advisory pass and a
+different thing entirely (tests/test_review.py covers that one). Nothing here may reach a model, a
+key, a service, or the network."""
+import json
+import os
+import subprocess
+import sys
+import unicodedata
+from pathlib import Path
+
+import pytest
+
+from ingot import cli, review
+from ingot.parse import parse_raw
+
+
+def _skill(root, name, description="Merge and split PDF files.", body="Do the thing.",
+ **frontmatter):
+ """Built line by line rather than from a dedented block: an interpolated multi-line field
+ defeats `textwrap.dedent`, which silently leaves the delimiters indented and turns every
+ package into a frontmatter-missing one. That cost a green test that was asserting nothing."""
+ directory = root / name
+ directory.mkdir(parents=True, exist_ok=True)
+ fields = {"name": name, "description": description, **frontmatter}
+ lines = ["---"] + [f"{key}: {value}" for key, value in fields.items()] + ["---", "", body, ""]
+ (directory / "SKILL.md").write_text("\n".join(lines), encoding="utf-8")
+ return directory
+
+
+def _codes(section) -> list[str]:
+ return [finding["code"] for finding in section["findings"]]
+
+
+def _all_codes(result) -> list[str]:
+ return [finding["code"]
+ for section in result["sections"].values()
+ for finding in section["findings"]]
+
+
+# --- the raw diagnostic parser -------------------------------------------------------------
+
+def test_raw_parser_reports_absent_frontmatter_instead_of_normalizing_it():
+ """`ingot.mcp_server.registry.parse_skill` turns this into empty metadata on purpose, so the server
+ keeps serving. A diagnostic parser that did the same would have nothing to report."""
+ raw = parse_raw("Just a body, no frontmatter.\n")
+
+ assert raw.frontmatter is None
+ assert [f.code for f in raw.findings] == ["frontmatter-missing"]
+
+
+def test_raw_parser_reports_the_yaml_error_rather_than_swallowing_it():
+ raw = parse_raw("---\nname: pdf\ndescription: [unclosed\n---\n\nbody\n")
+
+ assert raw.frontmatter is None
+ assert [f.code for f in raw.findings] == ["frontmatter-invalid"]
+ assert raw.findings[0].message
+
+
+def test_raw_parser_reports_frontmatter_that_is_not_a_mapping():
+ raw = parse_raw("---\n- one\n- two\n---\n\nbody\n")
+
+ assert raw.frontmatter is None
+ assert [f.code for f in raw.findings] == ["frontmatter-not-a-mapping"]
+
+
+def test_raw_parser_keeps_a_good_document_intact():
+ raw = parse_raw("---\nname: pdf\ndescription: Merge PDFs.\n---\n\nThe body.\n")
+
+ assert raw.findings == []
+ assert raw.frontmatter == {"name": "pdf", "description": "Merge PDFs."}
+ assert raw.body == "The body."
+
+
+# --- structural validity -------------------------------------------------------------------
+
+def test_a_known_good_skill_is_valid(tmp_path):
+ directory = _skill(tmp_path, "pdf")
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is True
+ assert result["errors"] == 0
+
+
+def test_a_missing_skill_md_is_an_error(tmp_path):
+ directory = tmp_path / "pdf"
+ directory.mkdir()
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is False
+ assert "skill-md-missing" in _all_codes(result)
+
+
+def test_invalid_frontmatter_is_an_error(tmp_path):
+ directory = tmp_path / "pdf"
+ directory.mkdir()
+ (directory / "SKILL.md").write_text("---\ndescription: [unclosed\n---\n\nbody\n",
+ encoding="utf-8")
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is False
+ assert "frontmatter-invalid" in _all_codes(result)
+
+
+def test_an_empty_description_is_an_error(tmp_path):
+ """The router keys on description. A skill without one is never loaded, so shipping it is a
+ silent no-op rather than a degraded skill."""
+ directory = _skill(tmp_path, "pdf", description="")
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is False
+ assert "description-empty" in _all_codes(result)
+
+
+def test_an_invalid_slug_is_an_error(tmp_path):
+ directory = tmp_path / "Not_A_Slug"
+ directory.mkdir()
+ (directory / "SKILL.md").write_text(
+ "---\nname: Not_A_Slug\ndescription: Something.\n---\n\nbody\n", encoding="utf-8")
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is False
+ assert "name-invalid" in _all_codes(result)
+
+
+def test_a_name_that_disagrees_with_the_directory_warns_without_failing(tmp_path):
+ directory = tmp_path / "pdf"
+ directory.mkdir()
+ (directory / "SKILL.md").write_text(
+ "---\nname: docx\ndescription: Something.\n---\n\nbody\n", encoding="utf-8")
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is True
+ assert "name-directory-mismatch" in _all_codes(result)
+
+
+def test_a_dangling_file_reference_warns_without_failing(tmp_path):
+ """A broken link makes the skill worse, not unrepresentable, so it must not block admission."""
+ directory = _skill(tmp_path, "pdf", body="See [the guide](./guide.md) for details.")
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is True
+ assert "file-reference-missing" in _all_codes(result)
+
+
+def test_a_resolvable_file_reference_is_not_reported(tmp_path):
+ directory = _skill(tmp_path, "pdf", body="See [the guide](./guide.md) for details.")
+ (directory / "guide.md").write_text("guide\n", encoding="utf-8")
+
+ result = review.review_package(directory)
+
+ assert "file-reference-missing" not in _all_codes(result)
+
+
+def test_path_traversal_in_a_reference_is_an_error(tmp_path):
+ directory = _skill(tmp_path, "pdf", body="See [outside](../../etc/passwd).")
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is False
+ assert "path-traversal" in _all_codes(result)
+
+
+@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows")
+def test_a_symlink_escaping_the_package_is_an_error(tmp_path):
+ directory = _skill(tmp_path, "pdf")
+ outside = tmp_path / "outside.txt"
+ outside.write_text("secret\n", encoding="utf-8")
+ (directory / "link.txt").symlink_to(outside)
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is False
+ assert "symlink-unsupported" in _all_codes(result)
+
+
+@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows")
+def test_a_symlink_inside_the_package_is_an_error_too(tmp_path):
+ """Containment is not the question any more. Admission stages exact bytes, and a link is
+ neither preserved (the vault would commit a path leading out of the library) nor followed (the
+ artifact would quietly become a different shape than the one submitted)."""
+ directory = _skill(tmp_path, "pdf")
+ (directory / "real.txt").write_text("data\n", encoding="utf-8")
+ (directory / "link.txt").symlink_to(directory / "real.txt")
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is False
+ assert "symlink-unsupported" in _all_codes(result)
+
+
+@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows")
+def test_a_directory_symlink_does_not_pull_in_what_it_points_at(tmp_path):
+ """`rglob` follows directory symlinks. A package could otherwise absorb a whole tree from
+ outside itself and the review would never mention the link."""
+ directory = _skill(tmp_path, "pdf")
+ outside = tmp_path / "outside"
+ outside.mkdir()
+ (outside / "secret.md").write_text("secret\n", encoding="utf-8")
+ (directory / "docs").symlink_to(outside, target_is_directory=True)
+
+ result = review.review_package(directory)
+
+ assert "symlink-unsupported" in _all_codes(result)
+ assert result["sections"]["structural"]["file_count"] == 1
+
+
+def test_a_non_portable_path_warns(tmp_path):
+ directory = _skill(tmp_path, "pdf")
+ (directory / "why:not.txt").write_text("data\n", encoding="utf-8")
+
+ result = review.review_package(directory)
+
+ assert "path-not-portable" in _all_codes(result)
+
+
+def test_case_insensitive_path_collision_warns(tmp_path):
+ """Two files that differ only by case survive here and collapse into one on macOS or Windows,
+ which changes the package's content hash depending on who checked it out."""
+ directory = _skill(tmp_path, "pdf")
+ (directory / "Guide.md").write_text("one\n", encoding="utf-8")
+ try:
+ (directory / "guide.md").write_text("two\n", encoding="utf-8")
+ except OSError:
+ pytest.skip("filesystem rejected the pair")
+ if len(list(directory.glob("*uide.md"))) < 2:
+ pytest.skip("case-insensitive filesystem collapsed the pair")
+
+ result = review.review_package(directory)
+
+ assert "path-case-collision" in _all_codes(result)
+
+
+def test_unicode_normalization_collision_warns(tmp_path):
+ """The same filename in NFC and NFD is one file on macOS and two on Linux."""
+ directory = _skill(tmp_path, "pdf")
+ composed = unicodedata.normalize("NFC", "café.md")
+ decomposed = unicodedata.normalize("NFD", "café.md")
+ (directory / composed).write_text("one\n", encoding="utf-8")
+ try:
+ (directory / decomposed).write_text("two\n", encoding="utf-8")
+ except OSError:
+ pytest.skip("filesystem rejected the pair")
+ if len(list(directory.glob("*.md"))) < 3:
+ pytest.skip("filesystem normalized the pair")
+
+ result = review.review_package(directory)
+
+ assert "path-unicode-collision" in _all_codes(result)
+
+
+def test_path_collision_logic_without_a_filesystem():
+ """The two collision cases above cannot be staged on a case-insensitive filesystem, which is
+ every macOS development machine, so they skip exactly where most of this code is written. This
+ covers the same logic directly -- `_path_findings` never touches the disk."""
+ package = Path("/pkg")
+ files = [package / "SKILL.md", package / "Guide.md", package / "guide.md"]
+
+ codes = [finding.code for finding in review._path_findings(package, files)]
+
+ assert "path-case-collision" in codes
+
+
+def test_unicode_collision_logic_without_a_filesystem():
+ package = Path("/pkg")
+ files = [package / unicodedata.normalize("NFC", "café.md"),
+ package / unicodedata.normalize("NFD", "café.md")]
+
+ codes = [finding.code for finding in review._path_findings(package, files)]
+
+ assert "path-unicode-collision" in codes
+
+
+def test_distinct_paths_do_not_collide():
+ package = Path("/pkg")
+ files = [package / "SKILL.md", package / "guide.md", package / "notes.md"]
+
+ assert review._path_findings(package, files) == []
+
+
+def test_the_report_carries_a_stable_content_revision(tmp_path):
+ directory = _skill(tmp_path, "pdf")
+
+ first = review.review_package(directory)
+ second = review.review_package(directory)
+
+ assert first["revision"] == second["revision"]
+ assert len(first["revision"]) == 64
+
+
+# --- supply-chain metadata -----------------------------------------------------------------
+
+def test_absent_source_and_license_metadata_are_reported(tmp_path):
+ directory = _skill(tmp_path, "pdf")
+
+ section = review.review_package(directory)["sections"]["supply_chain"]
+
+ assert "source-metadata-missing" in _codes(section)
+ assert "license-metadata-missing" in _codes(section)
+
+
+def test_declared_source_and_license_are_not_reported_missing(tmp_path):
+ directory = _skill(tmp_path, "pdf", license="Apache-2.0", source="https://example.test/repo")
+
+ section = review.review_package(directory)["sections"]["supply_chain"]
+
+ assert "license-metadata-missing" not in _codes(section)
+ assert "source-metadata-missing" not in _codes(section)
+
+
+def test_an_executable_asset_is_reported(tmp_path):
+ directory = _skill(tmp_path, "pdf")
+ script = directory / "run.sh"
+ script.write_text("#!/bin/sh\necho hi\n", encoding="utf-8")
+ script.chmod(0o755)
+
+ section = review.review_package(directory)["sections"]["supply_chain"]
+
+ assert "executable-asset" in _codes(section)
+
+
+def test_a_remote_reference_is_reported(tmp_path):
+ directory = _skill(tmp_path, "pdf", body="Fetch https://example.test/tool.sh and run it.")
+
+ section = review.review_package(directory)["sections"]["supply_chain"]
+
+ assert "remote-reference" in _codes(section)
+
+
+def test_an_unpinned_mutable_reference_is_reported(tmp_path):
+ directory = _skill(tmp_path, "pdf",
+ body="Read https://github.com/acme/tool/blob/main/README.md first.")
+
+ section = review.review_package(directory)["sections"]["supply_chain"]
+
+ assert "reference-unpinned" in _codes(section)
+
+
+def test_supply_chain_findings_never_fail_the_package(tmp_path):
+ """These are advisory. Nothing here claims to be a security verdict, so nothing here may
+ make an otherwise valid package inadmissible."""
+ directory = _skill(tmp_path, "pdf", body="Fetch https://example.test/tool.sh and run it.")
+ script = directory / "run.sh"
+ script.write_text("#!/bin/sh\n", encoding="utf-8")
+ script.chmod(0o755)
+
+ result = review.review_package(directory)
+
+ assert result["valid"] is True
+
+
+# --- collision -----------------------------------------------------------------------------
+
+def test_collision_is_unmeasured_without_a_library_root(tmp_path):
+ directory = _skill(tmp_path, "pdf")
+
+ section = review.review_package(directory)["sections"]["collision"]
+
+ assert section["status"] == review.UNMEASURED
+
+
+def test_a_library_name_collision_is_reported(tmp_path):
+ library = tmp_path / "library"
+ _skill(library, "pdf")
+ candidate = _skill(tmp_path / "candidate", "pdf")
+
+ section = review.review_package(candidate, library_root=library)["sections"]["collision"]
+
+ assert "name-collision" in _codes(section)
+
+
+def test_no_library_collision_when_the_name_is_free(tmp_path):
+ library = tmp_path / "library"
+ _skill(library, "docx")
+ candidate = _skill(tmp_path / "candidate", "pdf")
+
+ section = review.review_package(candidate, library_root=library)["sections"]["collision"]
+
+ assert section["status"] == review.MEASURED
+ assert _codes(section) == []
+
+
+def test_an_abandoned_staging_directory_is_not_a_collision(tmp_path):
+ """Promotion and rollback stage a skill beside the live one as `...stage`, each
+ carrying a complete SKILL.md. A bare glob would report a crashed run's leftovers as a colliding
+ skill, so this reads the library the same way the server does."""
+ library = tmp_path / "library"
+ _skill(library, "docx")
+ stage = library / ".pdf.abc123.stage"
+ stage.mkdir(parents=True)
+ (stage / "SKILL.md").write_text("---\nname: pdf\ndescription: Staged.\n---\n\nbody\n",
+ encoding="utf-8")
+ candidate = _skill(tmp_path / "candidate", "pdf")
+
+ section = review.review_package(candidate, library_root=library)["sections"]["collision"]
+
+ assert _codes(section) == []
+
+
+def test_many_remote_references_produce_one_finding(tmp_path):
+ """A wall of identical warnings is how a reviewer learns to stop reading them."""
+ body = "\n".join(f"See https://example.test/page-{n}" for n in range(12))
+ directory = _skill(tmp_path, "pdf", body=body)
+
+ section = review.review_package(directory)["sections"]["supply_chain"]
+
+ assert _codes(section).count("remote-reference") == 1
+ assert len(section["remote_references"]) == 12
+
+
+def test_semantic_collision_is_reported_unmeasured_not_guessed(tmp_path):
+ """Description shadowing needs the embedding router, which needs a model. The deterministic
+ command must say so and name the command that measures it, never approximate it."""
+ library = tmp_path / "library"
+ _skill(library, "docx")
+ candidate = _skill(tmp_path / "candidate", "pdf")
+
+ section = review.review_package(candidate, library_root=library)["sections"]["collision"]
+
+ assert section["semantic"]["status"] == review.UNMEASURED
+ assert "routing_health" in section["semantic"]["measure_with"]
+
+
+# --- activation and behavioral evidence ------------------------------------------------------
+
+def test_activation_is_unmeasured_without_a_routing_suite(tmp_path):
+ directory = _skill(tmp_path, "pdf")
+
+ section = review.review_package(directory, evidence_root=tmp_path)["sections"]["activation"]
+
+ assert section["status"] == review.UNMEASURED
+ assert section["routing_cases"] == 0
+
+
+def test_activation_reports_an_existing_suite_without_scoring_it(tmp_path):
+ """A suite existing is a fact this command can establish offline. Whether the router loads the
+ skill at the right time is not, so it stays UNMEASURED with the command that would answer it."""
+ directory = _skill(tmp_path, "pdf")
+ tasks = tmp_path / "ingot" / "optimize" / "tasks"
+ tasks.mkdir(parents=True)
+ (tasks / "pdf.yaml").write_text(
+ "routing:\n - prompt: merge two pdfs\n - prompt: split a pdf\n", encoding="utf-8")
+
+ section = review.review_package(directory, evidence_root=tmp_path)["sections"]["activation"]
+
+ assert section["routing_cases"] == 2
+ assert section["status"] == review.UNMEASURED
+ assert "score" not in section
+
+
+def test_behavioral_evidence_is_unmeasured_when_absent(tmp_path):
+ directory = _skill(tmp_path, "pdf")
+
+ section = review.review_package(directory, evidence_root=tmp_path)["sections"]["behavioral"]
+
+ assert section["status"] == review.UNMEASURED
+
+
+def test_existing_compatibility_evidence_is_surfaced_not_recomputed(tmp_path):
+ """`ingot.optimize.compat` already measures skill vs no-skill lift and writes it to runs/compat.
+ Review reads that file. It must never run a model to answer this."""
+ directory = _skill(tmp_path, "pdf")
+ compat = tmp_path / "runs" / "compat"
+ compat.mkdir(parents=True)
+ (compat / "pdf.json").write_text(json.dumps({
+ "skill": "pdf", "tasks": 12, "judge": "test-judge",
+ "models": {"openrouter/some-model": {"skill": 0.8, "baseline": 0.5, "lift": 0.3}},
+ }), encoding="utf-8")
+
+ section = review.review_package(directory, evidence_root=tmp_path)["sections"]["behavioral"]
+
+ assert section["status"] == review.MEASURED
+ assert section["tasks"] == 12
+ assert section["models"]["openrouter/some-model"]["lift"] == 0.3
+
+
+def test_review_reports_no_composite_score(tmp_path):
+ """A single number invites the reward-hacking this whole product exists to prevent."""
+ directory = _skill(tmp_path, "pdf")
+
+ result = review.review_package(directory)
+
+ assert "score" not in result
+ assert "grade" not in result
+
+
+# --- the command ----------------------------------------------------------------------------
+
+def test_command_exits_zero_for_a_valid_package(tmp_path, capsys):
+ directory = _skill(tmp_path, "pdf")
+
+ code = cli.main(["review", str(directory)])
+
+ assert code == 0
+ assert "pdf" in capsys.readouterr().out
+
+
+def test_command_exits_nonzero_for_an_invalid_package(tmp_path, capsys):
+ directory = _skill(tmp_path, "pdf", description="")
+
+ code = cli.main(["review", str(directory)])
+
+ assert code != 0
+
+
+def test_command_exits_zero_when_only_warnings_are_present(tmp_path, capsys):
+ directory = _skill(tmp_path, "pdf", body="See [the guide](./guide.md).")
+
+ code = cli.main(["review", str(directory)])
+
+ assert code == 0
+
+
+def test_command_json_is_the_versioned_payload(tmp_path, capsys):
+ directory = _skill(tmp_path, "pdf")
+
+ code = cli.main(["review", str(directory), "--json"])
+
+ assert code == 0
+ payload = json.loads(capsys.readouterr().out)
+ assert payload["schema_version"] == review.REVIEW_SCHEMA
+ assert set(payload["sections"]) == {"structural", "supply_chain", "collision",
+ "activation", "behavioral"}
+
+
+def test_command_on_a_missing_path_fails_cleanly(tmp_path, capsys):
+ code = cli.main(["review", str(tmp_path / "nowhere")])
+
+ assert code != 0
+ assert "nowhere" in capsys.readouterr().err
+
+
+def test_reviewing_stays_read_only(tmp_path):
+ """Read-only is a promise, not a description. A command that writes while reporting is a
+ command that can change what it is reporting on."""
+ directory = _skill(tmp_path, "pdf")
+ before = {path: path.stat().st_mtime_ns for path in sorted(tmp_path.rglob("*"))}
+
+ review.review_package(directory)
+
+ after = {path: path.stat().st_mtime_ns for path in sorted(tmp_path.rglob("*"))}
+ assert before == after
+
+
+def test_review_runs_without_the_heavy_stack(tmp_path):
+ """The point of the milestone, enforced: no model, no key, no services, no network."""
+ directory = _skill(tmp_path, "pdf")
+ heavy = ["fastapi", "onnxruntime", "langgraph", "langfuse", "litellm", "fastembed", "ingot.optimize"]
+ program = ("import sys, ingot.cli; "
+ f"ingot.cli.main(['review', {str(directory)!r}, '--json']); "
+ f"print([m for m in {heavy!r} if m in sys.modules], file=sys.stderr)")
+
+ result = subprocess.run([sys.executable, "-c", program], capture_output=True, text=True)
+
+ assert result.returncode == 0, result.stderr
+ assert result.stderr.strip().endswith("[]")
+
+
+# --------------------------------------------------------------------------- artifact fidelity
+
+def test_a_binary_asset_is_reported_but_not_refused(tmp_path):
+ """Admission preserves it byte-for-byte, so it is a note. It is still the fact a reviewer most
+ needs: text can be read before approval and a compiled asset cannot."""
+ package = _skill(tmp_path, "pdf")
+ (package / "assets").mkdir()
+ (package / "assets" / "diagram.png").write_bytes(b"\x89PNG\r\n\x1a\n" + b"\xff\xfe" * 16)
+
+ result = review.review_package(package)
+
+ assert result["valid"] is True
+ codes = [f["code"] for f in result["sections"]["structural"]["findings"]]
+ assert "binary-asset" in codes
+ assert result["sections"]["structural"]["binary_assets"] == ["assets/diagram.png"]
+
+
+def test_the_finding_names_every_asset_a_reviewer_cannot_read(tmp_path):
+ package = _skill(tmp_path, "pdf")
+ for name in ("a.png", "b.pdf", "c.bin"):
+ (package / name).write_bytes(b"\xff\xfe\x00")
+
+ result = review.review_package(package)
+
+ assert result["sections"]["structural"]["binary_assets"] == ["a.png", "b.pdf", "c.bin"]
+ message = [f["message"] for f in result["sections"]["structural"]["findings"]
+ if f["code"] == "binary-asset"][0]
+ for name in ("a.png", "b.pdf", "c.bin"):
+ assert name in message
+
+
+def test_editor_and_vcs_metadata_is_not_reported_as_an_asset(tmp_path):
+ """A finding that fires on `.DS_Store` is one people learn to scroll past."""
+ package = _skill(tmp_path, "pdf")
+ (package / ".DS_Store").write_bytes(b"\x00\x00\x00\x01")
+ (package / ".git").mkdir()
+ (package / ".git" / "index").write_bytes(b"DIRC\xff")
+ (package / "__pycache__").mkdir()
+ (package / "__pycache__" / "x.cpython-312.pyc").write_bytes(b"\xff\x00")
+
+ result = review.review_package(package)
+
+ assert result["sections"]["structural"]["binary_assets"] == []
+ assert result["valid"] is True
+
+
+def test_the_extension_is_not_what_decides(tmp_path):
+ """An extension is a claim about a file; the point is to check the file. A `.md` of raw bytes
+ is unreadable, and a `.bin` of UTF-8 is not."""
+ package = _skill(tmp_path, "pdf")
+ (package / "readable.bin").write_text("plain text\n", encoding="utf-8")
+ (package / "unreadable.md").write_bytes(b"\xff\xfe\x00\x01")
+
+ assert review.review_package(package)["sections"]["structural"]["binary_assets"] \
+ == ["unreadable.md"]
+
+
+def test_a_package_of_only_text_reports_no_assets(tmp_path):
+ package = _skill(tmp_path, "pdf")
+ (package / "references").mkdir()
+ (package / "references" / "notes.md").write_text("# Notes\n")
+ (package / "run.sh").write_text("#!/bin/sh\necho hi\n")
+
+ assert review.review_package(package)["sections"]["structural"]["binary_assets"] == []
diff --git a/tests/test_ingress.py b/tests/test_ingress.py
new file mode 100644
index 0000000..5de50db
--- /dev/null
+++ b/tests/test_ingress.py
@@ -0,0 +1,201 @@
+import json
+
+import pytest
+
+from ingot.mcp_server import registry
+from ingot.optimize import promote as P
+
+
+def _store(tmp_path, monkeypatch):
+ root = tmp_path / "skills"
+ root.mkdir()
+ monkeypatch.setenv("INGOT_LIBRARY", str(root))
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root))
+ return root
+
+
+def _proposal(**overrides):
+ data = {
+ "skill": "copywriting",
+ "description": "Write clear conversion copy.",
+ "body": "# Copywriting\n\nUse evidence and preserve supplied facts.",
+ "files": {"references/frameworks.md": "# Frameworks\n\nPAS"},
+ "frontmatter": {},
+ "summary": "Add the vetted copywriting skill.",
+ "source": "dotfiles-claude@c658aee",
+ "producer": "skill-retrospective",
+ "caller": "improve existing intake",
+ "evidence": ["Package is hash-pinned.", "Catalog discovery passed."],
+ "pressure_scenario": "A rushed draft must preserve locked facts.",
+ "risk": "New instructions may route too broadly.",
+ "verification_status": "passed",
+ "verification_command": "python3 scripts/codex-skill-catalog.py list --repo .",
+ "verification_result": "copywriting discovered",
+ }
+ data.update(overrides)
+ return data
+
+
+def test_new_skill_submission_is_visible_but_inert(tmp_path, monkeypatch):
+ from ingot.optimize import ingress
+
+ root = _store(tmp_path, monkeypatch)
+
+ result = ingress.submit_skill_create(**_proposal())
+
+ pending = P.load_pending("copywriting")
+ assert result == {"status": "quarantined", "skill": "copywriting",
+ "proposal_id": pending["creation"]["proposal_id"],
+ "promotable": True}
+ assert pending["kind"] == "creation"
+ assert pending["champion_components"] == {}
+ assert pending["challenger_components"] == {
+ "description": "Write clear conversion copy.",
+ "body": "# Copywriting\n\nUse evidence and preserve supplied facts.",
+ "frontmatter": '{"description":"Write clear conversion copy.","name":"copywriting"}',
+ "file:references/frameworks.md": "# Frameworks\n\nPAS",
+ }
+ assert pending["gate"]["kind"] == "new_skill_admission"
+ assert not (root / "copywriting").exists()
+ assert json.loads((tmp_path / "ingress-audit.jsonl").read_text())["action"] == "quarantine"
+
+
+def test_creation_refuses_existing_skill_unsafe_files_and_unverified_input(tmp_path, monkeypatch):
+ from ingot.optimize import ingress
+
+ root = _store(tmp_path, monkeypatch)
+ existing = root / "copywriting"
+ existing.mkdir()
+ registry.write_skill_md(existing / "SKILL.md",
+ {"name": "copywriting", "description": "Existing."}, "Body")
+
+ with pytest.raises(ValueError, match="already exists"):
+ ingress.submit_skill_create(**_proposal())
+ (existing / "SKILL.md").unlink()
+ existing.rmdir()
+ with pytest.raises(ValueError, match="escapes skill root"):
+ ingress.submit_skill_create(**_proposal(files={"../outside.md": "no"}))
+ with pytest.raises(ValueError, match="verification_status"):
+ ingress.submit_skill_create(**_proposal(verification_status="unavailable"))
+
+
+def test_creation_promotion_is_reversible_to_absence(tmp_path, monkeypatch):
+ from ingot.optimize import ingress
+
+ root = _store(tmp_path, monkeypatch)
+ ingress.submit_skill_create(**_proposal())
+
+ promoted = P._activate_approved("copywriting", P.load_pending("copywriting"),
+ actor="reviewer")
+ created = root / "copywriting"
+ assert "Added 'copywriting'" in promoted
+ assert created.is_dir()
+ assert "Use evidence" in (created / "SKILL.md").read_text()
+ assert (created / "references/frameworks.md").read_text().endswith("PAS")
+ assert not P.pending_path("copywriting").exists()
+
+ removed = P._activate_rollback("copywriting", P.ABSENT_REVISION, actor="reviewer")
+ assert "to absence" in removed
+ assert not created.exists()
+
+ created_revision = next(item["revision"] for item in P.list_revisions("copywriting")
+ if item["revision"] != P.ABSENT_REVISION)
+ restored = P._activate_rollback("copywriting", created_revision, actor="reviewer")
+ assert "Restored absent skill" in restored
+ assert created.is_dir()
+ assert "Use evidence" in (created / "SKILL.md").read_text()
+
+
+def test_duplicate_creation_is_idempotent(tmp_path, monkeypatch):
+ from ingot.optimize import ingress
+
+ _store(tmp_path, monkeypatch)
+
+ first = ingress.submit_skill_create(**_proposal())
+ second = ingress.submit_skill_create(**_proposal())
+
+ assert first["status"] == "quarantined"
+ assert second == {**first, "status": "duplicate"}
+ evidence = tmp_path / "evidence" / "copywriting" / f"creation-{first['proposal_id']}"
+ assert (evidence / "evidence.json").is_file()
+ assert len((tmp_path / "ingress-audit.jsonl").read_text().splitlines()) == 1
+
+
+def test_creation_binds_normalized_full_frontmatter_and_staged_bytes(tmp_path, monkeypatch):
+ from ingot.optimize import ingress
+
+ root = _store(tmp_path, monkeypatch)
+ router = {"harnesses": ["codex"], "platforms": ["macos"],
+ "required_tools": ["rg"], "activation": "explicit"}
+ ingress.submit_skill_create(**_proposal(
+ description="Write clear\nconversion copy.",
+ frontmatter={"license": "MIT", "metadata": {"skill-router": router}},
+ ))
+ pending = P.load_pending("copywriting")
+ expected = pending["evidence"]["challenger"]["revision"]
+
+ P._activate_approved("copywriting", P.load_pending("copywriting"), actor="reviewer")
+
+ active = registry.load_skills(root)[0]
+ meta, _ = registry.parse_skill((root / "copywriting" / "SKILL.md").read_text(), "copywriting")
+ assert active.revision == expected
+ assert active.description == "Write clear conversion copy."
+ assert active.metadata["harnesses"] == ["codex"]
+ assert active.metadata["platforms"] == ["macos"]
+ assert meta["license"] == "MIT"
+
+
+def test_evidence_changes_are_not_aliased_as_duplicates(tmp_path, monkeypatch):
+ from ingot.optimize import ingress
+
+ _store(tmp_path, monkeypatch)
+ first = ingress.submit_skill_create(**_proposal())
+
+ with pytest.raises(ValueError, match="review slot is occupied"):
+ ingress.submit_skill_create(**_proposal(risk="Different material risk."))
+ assert P.load_pending("copywriting")["creation"]["proposal_id"] == first["proposal_id"]
+
+
+def test_audit_failure_does_not_turn_a_queued_skill_into_a_false_refusal(tmp_path, monkeypatch,
+ caplog):
+ from ingot.optimize import ingress
+
+ _store(tmp_path, monkeypatch)
+ monkeypatch.setattr(ingress, "audit_file", lambda: tmp_path / "missing" / "audit.jsonl")
+ monkeypatch.setattr(ingress, "_audit", lambda record: (_ for _ in ()).throw(OSError("full")))
+
+ result = ingress.submit_skill_create(**_proposal())
+
+ assert result["status"] == "quarantined"
+ assert P.load_pending("copywriting") is not None
+ assert "audit write failed" in caplog.text
+
+
+@pytest.mark.parametrize("path", ["C:/x.md", "CON.md", "AUX/file.md", "bad\\name.md"])
+def test_creation_rejects_nonportable_component_paths(tmp_path, monkeypatch, path):
+ from ingot.optimize import ingress
+
+ _store(tmp_path, monkeypatch)
+ with pytest.raises(ValueError, match="portable POSIX path"):
+ ingress.submit_skill_create(**_proposal(files={path: "content"}))
+
+
+def test_creation_bounds_evidence_envelope(tmp_path, monkeypatch):
+ from ingot.optimize import ingress
+
+ _store(tmp_path, monkeypatch)
+ with pytest.raises(ValueError, match="2 to 12"):
+ ingress.submit_skill_create(**_proposal(evidence=[str(i) for i in range(13)]))
+
+
+def test_creation_rejects_unsafe_router_globs_and_unicode_aliases(tmp_path, monkeypatch):
+ from ingot.optimize import ingress
+
+ _store(tmp_path, monkeypatch)
+ frontmatter = {"metadata": {"skill-router": {
+ "scopes": ["project"], "path_patterns": ["/tmp/*"]}}}
+ with pytest.raises(ValueError, match="relative POSIX globs"):
+ ingress.submit_skill_create(**_proposal(frontmatter=frontmatter))
+ with pytest.raises(ValueError, match="NFC-normalized"):
+ ingress.submit_skill_create(**_proposal(files={"references/e\u0301.md": "content"}))
diff --git a/tests/test_judge.py b/tests/test_judge.py
index 690e2e6..bdd2e1a 100644
--- a/tests/test_judge.py
+++ b/tests/test_judge.py
@@ -1,8 +1,12 @@
"""Unit tests for the LLM judge's pure parsing/aggregation logic (LLM calls mocked)."""
+import json
+
import pytest
-from optimize import judge as J
-from optimize.judge import DIMENSIONS, _extract_json, failed_dimensions
+from ingot.optimize import agy_judge as A
+from ingot.optimize import judge as J
+from ingot.optimize.judge import (DEFAULT_CHECKLIST, DIMENSIONS, _extract_json, _weighted,
+ failed_dimensions)
class _FakeMsg:
@@ -17,6 +21,77 @@ def _mock_single_judge(monkeypatch, content):
monkeypatch.setattr(J, "_get_llm", lambda model: type("L", (), {"invoke": lambda self, m: _FakeMsg(content)})())
+def _mock_judges(monkeypatch, contents):
+ """One scripted response per model, in order, for ensemble tests."""
+ monkeypatch.setattr(J, "MODELS", [f"mock-{i}" for i in range(len(contents))])
+ by_model = {f"mock-{i}": c for i, c in enumerate(contents)}
+ monkeypatch.setattr(J, "_get_llm", lambda model: type(
+ "L", (), {"invoke": lambda self, m, _c=by_model[model]: _FakeMsg(_c)})())
+
+
+def _items(**verdicts) -> str:
+ return json.dumps({"items": {k: {"verdict": v, "note": "n"} for k, v in verdicts.items()},
+ "feedback": "f"})
+
+
+def _all(verdict: str) -> str:
+ return _items(**{i["id"]: verdict for i in DEFAULT_CHECKLIST})
+
+
+def _agy_grade(verdict: str = "pass") -> dict:
+ return {
+ "items": {
+ item["id"]: {"verdict": verdict, "note": ""}
+ for item in DEFAULT_CHECKLIST
+ },
+ "feedback": "agy feedback",
+ }
+
+
+# --- backend selection -----------------------------------------------------------------------
+
+def test_agy_backend_uses_one_subscription_grade_without_openrouter(monkeypatch):
+ monkeypatch.setenv("JUDGE_BACKEND", "agy")
+ monkeypatch.setenv("OPENROUTER_API_KEY", "must-not-be-used")
+ monkeypatch.setattr(J, "MODELS", ["openrouter/one", "openrouter/two"])
+ monkeypatch.setattr(J, "_get_llm", lambda *_args: pytest.fail("OpenRouter fallback ran"))
+ usage = {"input_tokens": 11, "output_tokens": 7, "total_tokens": 18}
+ calls = []
+
+ def invoke(prompt, checklist):
+ calls.append((prompt, checklist))
+ return _agy_grade(), usage
+
+ ledger = []
+ monkeypatch.setattr(A, "invoke", invoke)
+ monkeypatch.setattr(
+ J.usage_ledger,
+ "add",
+ lambda role, observed, **metadata: ledger.append((role, observed, metadata)),
+ )
+
+ result = J.judge("task", answer="answer")
+
+ assert result["score"] == 1.0
+ assert result["feedback"] == "agy feedback"
+ assert len(calls) == 1
+ assert ledger == [("judge", usage, {"billing_mode": "subscription"})]
+
+
+def test_agy_backend_error_propagates_without_openrouter_fallback(monkeypatch):
+ monkeypatch.setenv("JUDGE_BACKEND", "agy")
+ monkeypatch.setenv("OPENROUTER_PROVIDERS", "fallback-provider")
+ monkeypatch.setattr(J, "_get_llm", lambda *_args: pytest.fail("OpenRouter fallback ran"))
+ monkeypatch.setattr(
+ A,
+ "invoke",
+ lambda *_args: (_ for _ in ()).throw(A.AgyJudgeError("agy unavailable")),
+ )
+
+ with pytest.raises(A.AgyJudgeError, match="agy unavailable"):
+ J.judge("task", answer="answer")
+
+
# --- _extract_json: robust to prose / fences / stray braces -----------------------------------
def test_extract_json_plain():
@@ -41,30 +116,151 @@ def test_extract_json_returns_empty_when_no_score_object(text):
assert _extract_json(text) == {}
-# --- judge(): score clamping, unparseable fallback, dimension defaulting -----------------------
+def test_extract_json_finds_the_requested_key():
+ """The judge asks for `items`; the score key it used to look for no longer appears."""
+ assert _extract_json('{"items": {"a": "pass"}}', "items")["items"] == {"a": "pass"}
+
-def test_judge_clamps_score_above_one(monkeypatch):
- _mock_single_judge(monkeypatch, '{"score": 1.7, "feedback": "great", "dimensions": {}}')
- assert J.judge("t", "r", "a")["score"] == 1.0
+# --- the score is derived, never taken from the model -----------------------------------------
+def test_score_is_the_weighted_mean_of_the_checklist(monkeypatch):
+ """DEFAULT_CHECKLIST weights are 3/2/2/1. Failing only `efficiency` (weight 1) must cost
+ exactly 1/8 of the score, not whatever the model felt like reporting."""
+ _mock_single_judge(monkeypatch, _items(correctness="pass", completeness="pass",
+ instruction_following="pass", efficiency="fail"))
+ assert J.judge("t", "r", "a")["score"] == pytest.approx(7 / 8)
-def test_judge_clamps_negative_score(monkeypatch):
- _mock_single_judge(monkeypatch, '{"score": -0.4, "feedback": "bad", "dimensions": {}}')
+
+def test_a_model_supplied_score_is_ignored(monkeypatch):
+ """The old contract let the judge name its own number. A model that still emits one must not
+ be able to override the checklist -- that is the entire point of grading against items."""
+ payload = json.loads(_all("fail"))
+ payload["score"] = 1.0
+ _mock_single_judge(monkeypatch, json.dumps(payload))
assert J.judge("t", "r", "a")["score"] == 0.0
-def test_judge_defaults_missing_dimensions_to_pass(monkeypatch):
- _mock_single_judge(monkeypatch, '{"score": 0.5, "dimensions": {"correctness": "wrong API"}}')
+def test_partial_verdicts_score_half(monkeypatch):
+ _mock_single_judge(monkeypatch, _all("partial"))
+ assert J.judge("t", "r", "a")["score"] == pytest.approx(0.5)
+
+
+def test_score_cannot_leave_zero_to_one(monkeypatch):
+ """Clamping used to be needed because the model picked the number. A weighted mean of values
+ in [0,1] cannot leave the range, so the property holds by construction."""
+ for verdict in ("pass", "partial", "fail"):
+ _mock_single_judge(monkeypatch, _all(verdict))
+ assert 0.0 <= J.judge("t", "r", "a")["score"] <= 1.0
+
+
+# --- items the judge did not answer, or answered badly, are not passes ------------------------
+
+def test_an_ungraded_item_is_not_a_pass(monkeypatch):
+ """Omitting an item must cost its weight. Defaulting it to pass would let a lazy judge score
+ 1.0 by answering one item."""
+ _mock_single_judge(monkeypatch, _items(correctness="pass"))
r = J.judge("t", "r", "a")
- assert failed_dimensions(r["dimensions"]) == ["correctness"] # only the one provided fails
- assert set(r["dimensions"]) == set(DIMENSIONS) # the rest are filled in as pass
+ assert r["score"] == pytest.approx(3 / 8)
+ assert r["checklist"]["efficiency"]["note"] == "not graded"
-def test_judge_unparseable_output_scores_zero(monkeypatch):
+def test_an_unrecognized_verdict_fails_rather_than_passes(monkeypatch):
+ _mock_single_judge(monkeypatch, _items(correctness="excellent", completeness="pass",
+ instruction_following="pass", efficiency="pass"))
+ r = J.judge("t", "r", "a")
+ assert r["score"] == pytest.approx(5 / 8) # correctness carries weight 3 of 8
+
+
+def test_an_unrecognized_verdict_without_a_note_says_it_was_ungraded(monkeypatch):
+ """The model's own note is kept when it wrote one; the fallback only fills a silent item so
+ the reviewer never sees a bare zero with no reason."""
+ _mock_single_judge(monkeypatch, json.dumps({"items": {"correctness": "excellent"},
+ "feedback": "f"}))
+ assert "ungraded" in J.judge("t", "r", "a")["checklist"]["correctness"]["note"]
+
+
+def test_unparseable_output_scores_zero_without_blaming_the_skill(monkeypatch):
_mock_single_judge(monkeypatch, "the model rambled and produced no JSON at all")
r = J.judge("t", "r", "a")
assert r["score"] == 0.0 and "unparseable" in r["feedback"]
- assert failed_dimensions(r["dimensions"]) == [] # a parse failure isn't a skill failure
+ assert failed_dimensions(r["dimensions"]) == [] # a parse failure isn't a skill failure
+
+
+# --- custom checklists (the Tessl-style graded rubric) ----------------------------------------
+
+CUSTOM = [
+ {"id": "cites_source", "dimension": "correctness", "weight": 4,
+ "criterion": "Every factual claim names its source."},
+ {"id": "states_tradeoff", "dimension": "completeness", "weight": 1,
+ "criterion": "It names at least one tradeoff."},
+]
+
+
+def test_a_task_checklist_replaces_the_default(monkeypatch):
+ _mock_single_judge(monkeypatch, _items(cites_source="pass", states_tradeoff="fail"))
+ r = J.judge("t", "r", "a", checklist=CUSTOM)
+ assert r["score"] == pytest.approx(4 / 5)
+ assert set(r["checklist"]) == {"cites_source", "states_tradeoff"}
+
+
+def test_a_malformed_checklist_falls_back_to_the_default(monkeypatch):
+ """Items without an id or criterion cannot be graded; an empty result must not mean 'no checks
+ ran, therefore full marks'."""
+ _mock_single_judge(monkeypatch, _all("pass"))
+ r = J.judge("t", "r", "a", checklist=[{"weight": 9}, {"id": "x"}])
+ assert set(r["checklist"]) == {i["id"] for i in DEFAULT_CHECKLIST}
+
+
+def test_custom_items_map_onto_the_reported_dimensions(monkeypatch):
+ """mine.py and the candidate search consume `dimensions`, so a custom rubric still has to
+ report through them."""
+ _mock_single_judge(monkeypatch, _items(cites_source="fail", states_tradeoff="pass"))
+ r = J.judge("t", "r", "a", checklist=CUSTOM)
+ assert failed_dimensions(r["dimensions"]) == ["correctness"]
+
+
+# --- resolution: the reason for the whole change ----------------------------------------------
+
+def test_the_checklist_resolves_finer_than_a_holistic_ladder():
+ """A judge naming one number emits a coarse ladder (0.9 / 0.95 / 1.0), so a mean can move a
+ whole rung because two tasks crossed a boundary. Independent weighted items give many more
+ reachable values, which is what makes a small real effect distinguishable from noise."""
+ weights = [i["weight"] for i in DEFAULT_CHECKLIST]
+ reachable = set()
+ for bits in range(3 ** len(weights)):
+ values, b = [], bits
+ for _ in weights:
+ values.append([0.0, 0.5, 1.0][b % 3]); b //= 3
+ graded = {i["id"]: {"value": v} for i, v in zip(DEFAULT_CHECKLIST, values)}
+ reachable.add(round(_weighted(graded, DEFAULT_CHECKLIST), 6))
+ assert len(reachable) >= 17 # vs the 3 rungs a holistic judge actually used
+
+
+# --- ensembles average per item, not per answer -----------------------------------------------
+
+def test_ensemble_averages_each_item_before_weighting(monkeypatch):
+ """Averaging item values is smoother than averaging whole-answer scores: two judges splitting
+ on one item move the result by half that item's weight, not by half the answer."""
+ _mock_judges(monkeypatch, [_all("pass"),
+ _items(correctness="pass", completeness="pass",
+ instruction_following="pass", efficiency="fail")])
+ r = J.judge("t", "r", "a")
+ assert r["checklist"]["efficiency"]["value"] == pytest.approx(0.5)
+ assert r["score"] == pytest.approx(1 - 0.5 * (1 / 8))
+
+
+def test_a_minority_failure_does_not_fail_the_dimension(monkeypatch):
+ """One judge of three flagging an item leaves its value at 2/3, above the failure threshold,
+ so the dimension still reads as a pass."""
+ _mock_judges(monkeypatch, [_all("pass"), _all("pass"),
+ _items(correctness="fail", completeness="pass",
+ instruction_following="pass", efficiency="pass")])
+ assert failed_dimensions(J.judge("t", "r", "a")["dimensions"]) == []
+
+
+def test_one_unparseable_judge_does_not_sink_the_ensemble(monkeypatch):
+ _mock_judges(monkeypatch, [_all("pass"), "no json here"])
+ assert J.judge("t", "r", "a")["score"] == pytest.approx(1.0)
# --- failed_dimensions: case / synonyms ------------------------------------------------------
diff --git a/tests/test_lite_mode.py b/tests/test_lite_mode.py
index 5b373be..751783b 100644
--- a/tests/test_lite_mode.py
+++ b/tests/test_lite_mode.py
@@ -1,8 +1,8 @@
"""Lite mode: the Langfuse-free A/B variant runner and the cost ledger / spend cap."""
import pytest
-from optimize import ab as ab_mod
-from optimize import usage as usage_ledger
+from ingot.optimize import ab as ab_mod
+from ingot.optimize import usage as usage_ledger
def _fake_run_task(answers: dict):
@@ -17,7 +17,7 @@ def test_local_variant_matches_run_variant_shape(monkeypatch):
tasks = [{"task": "a", "rubric": "ra"}, {"task": "b", "rubric": "rb"}]
monkeypatch.setattr(ab_mod, "run_task", _fake_run_task({"a": "ans-a", "b": "ans-b"}))
monkeypatch.setattr(ab_mod, "judge",
- lambda t, r, ans, check=None, deliverable=None:
+ lambda t, r, ans, check=None, deliverable=None, checklist=None:
{"score": 0.9 if t == "a" else 0.4, "feedback": "f", "dimensions": {}})
scores, usages, behaviors, answers = ab_mod._run_variant_local(agent=None, tasks=tasks)
assert scores == [0.9, 0.4] # aligned to task order
@@ -31,7 +31,7 @@ def test_local_variant_scores_failed_rollouts_zero(monkeypatch):
monkeypatch.setattr(ab_mod, "run_task",
_fake_run_task({"a": RuntimeError("provider down"), "b": "ans-b"}))
monkeypatch.setattr(ab_mod, "judge",
- lambda t, r, ans, check=None, deliverable=None:
+ lambda t, r, ans, check=None, deliverable=None, checklist=None:
{"score": 1.0, "feedback": "f", "dimensions": {}})
scores, usages, behaviors, answers = ab_mod._run_variant_local(agent=None, tasks=tasks)
assert scores == [0.0, 1.0] # failure defaults to 0, like _run_variant
@@ -56,6 +56,47 @@ def test_estimated_cost_uses_role_model_prices(monkeypatch):
usage_ledger.reset()
+def test_compat_spend_is_priced_per_model_and_reaches_the_cap(monkeypatch):
+ """The compatibility sweep changes the serving model on purpose, so it bills to `compat:`
+ rather than one bucket. Under a single "compat" role no price ever matched, the sweep counted
+ as $0 — it reported $0.04 against $1.42 actually spent — and MAX_RUN_USD, which enforces on the
+ same figure, could not see the most expensive role in the run."""
+ usage_ledger.reset()
+ monkeypatch.delenv("MAX_RUN_USD", raising=False)
+ monkeypatch.delenv("BASE_URL", raising=False)
+ monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False)
+ monkeypatch.setenv("JUDGE_MODEL", "m/judge")
+ monkeypatch.setattr(usage_ledger, "_PRICES",
+ {"m/dear": (0.0, 1e-5), "m/cheap": (0.0, 1e-7), "m/judge": (0.0, 0.0)})
+ usage_ledger.add("compat:m/dear", {"input_tokens": 0, "output_tokens": 100_000})
+ usage_ledger.add("compat:m/cheap", {"input_tokens": 0, "output_tokens": 100_000})
+ usage_ledger.add("judge", {"input_tokens": 500_000, "output_tokens": 0})
+
+ assert usage_ledger.estimated_cost() == pytest.approx(1.0 + 0.01)
+ assert usage_ledger.unpriced_roles() == []
+ assert "NOT in that estimate" not in usage_ledger.format_report()
+ usage_ledger.reset()
+
+
+def test_a_role_with_no_price_is_named_instead_of_counting_as_zero(monkeypatch):
+ """An unpriced role contributes $0, which is right for a local endpoint and dangerously wrong
+ for a slug we simply failed to resolve. Either way the report has to say which roles the
+ number excludes, or the next silently-free role hides the same way this one did."""
+ usage_ledger.reset()
+ monkeypatch.delenv("MAX_RUN_USD", raising=False)
+ monkeypatch.delenv("BASE_URL", raising=False)
+ monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False)
+ monkeypatch.setenv("JUDGE_MODEL", "m/judge")
+ monkeypatch.setattr(usage_ledger, "_PRICES", {"m/judge": (1e-6, 1e-6)})
+ usage_ledger.add("judge", {"input_tokens": 1_000_000, "output_tokens": 0})
+ usage_ledger.add("compat:m/unknown", {"input_tokens": 0, "output_tokens": 9_000_000})
+
+ assert usage_ledger.unpriced_roles() == ["compat:m/unknown"]
+ report = usage_ledger.format_report()
+ assert "NOT in that estimate: compat:m/unknown" in report and "m/unknown" in report
+ usage_ledger.reset()
+
+
def test_cost_is_none_on_local_endpoints(monkeypatch):
usage_ledger.reset()
monkeypatch.setenv("BASE_URL", "http://172.17.0.1:11434/v1")
diff --git a/tests/test_local_traces.py b/tests/test_local_traces.py
new file mode 100644
index 0000000..3cd401e
--- /dev/null
+++ b/tests/test_local_traces.py
@@ -0,0 +1,394 @@
+import json
+import os
+from pathlib import Path
+
+import pytest
+
+from ingot.optimize import local_traces
+
+
+def _write_jsonl(path: Path, records: list[dict]) -> None:
+ path.parent.mkdir(parents=True, exist_ok=True)
+ path.write_text("".join(json.dumps(record) + "\n" for record in records))
+
+
+def test_codex_parser_ignores_injected_context_and_keeps_completed_human_turn(tmp_path):
+ session = tmp_path / "rollout.jsonl"
+ records = [
+ {"timestamp": "2026-07-28T10:00:00Z", "type": "session_meta", "payload": {
+ "id": "codex-session", "cwd": "/work/project", "thread_source": "user",
+ }},
+ {"timestamp": "2026-07-28T10:00:01Z", "type": "response_item", "payload": {
+ "type": "message", "role": "user", "content": [
+ {"type": "input_text", "text": "# AGENTS.md instructions\ninternal"},
+ {"type": "input_text", "text": "internal"},
+ ],
+ }},
+ {"timestamp": "2026-07-28T10:01:00Z", "type": "response_item", "payload": {
+ "type": "message", "role": "user",
+ "content": [{"type": "input_text", "text": "Build the landing page."}],
+ }},
+ {"timestamp": "2026-07-28T10:01:00Z", "type": "response_item", "payload": {
+ "type": "message", "role": "user",
+ "content": [{"type": "input_text",
+ "text": "synthetic status"}],
+ }},
+ {"timestamp": "2026-07-28T10:01:01Z", "type": "event_msg", "payload": {
+ "type": "task_started", "turn_id": "turn-1", "started_at": 1000,
+ }},
+ {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": {
+ "type": "function_call", "name": "exec_command", "call_id": "call-1",
+ "arguments": json.dumps({
+ "cmd": "sed -n '1,220p' /Users/example/.agents/skills/frontend-design/SKILL.md",
+ }),
+ }},
+ {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": {
+ "type": "function_call_output", "call_id": "call-1",
+ "output": json.dumps({"exit_code": 1, "output": "failed"}),
+ }},
+ {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": {
+ "type": "function_call", "name": "mcp__ingot__route_and_load",
+ "call_id": "route-1", "arguments": json.dumps({"task": "Build the landing page."}),
+ }},
+ {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": {
+ "type": "function_call_output", "call_id": "route-1",
+ "output": json.dumps({"match": "web-design-flow", "revision": "83a75cf1",
+ "skill_body": "private"}),
+ }},
+ {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": {
+ "type": "function_call", "name": "route_and_load", "arguments": "{}",
+ }},
+ {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": {
+ "type": "function_call_output",
+ "output": json.dumps({"match": "wrong-skill", "revision": "deadbeef"}),
+ }},
+ {"timestamp": "2026-07-28T10:01:03Z", "type": "event_msg", "payload": {
+ "type": "token_count", "info": {"last_token_usage": {
+ "input_tokens": 120, "output_tokens": 30,
+ "cached_input_tokens": 20, "reasoning_output_tokens": 4,
+ }},
+ }},
+ {"timestamp": "2026-07-28T10:01:04Z", "type": "event_msg", "payload": {
+ "type": "task_complete", "turn_id": "turn-1", "duration_ms": 3000,
+ "completed_at": 4000, "last_agent_message": "Landing page shipped.",
+ }},
+ {"timestamp": "2026-07-28T10:02:00Z", "type": "response_item", "payload": {
+ "type": "message", "role": "user",
+ "content": [{"type": "input_text", "text": "Cancel this request."}],
+ }},
+ {"timestamp": "2026-07-28T10:02:01Z", "type": "event_msg", "payload": {
+ "type": "task_started", "turn_id": "turn-2", "started_at": 5000,
+ }},
+ {"timestamp": "2026-07-28T10:02:02Z", "type": "event_msg", "payload": {
+ "type": "turn_aborted", "turn_id": "turn-2", "reason": "interrupted",
+ }},
+ ]
+ _write_jsonl(session, records)
+
+ traces = local_traces.parse_codex_session(session)
+
+ assert len(traces) == 1
+ assert traces[0]["task"] == "Build the landing page."
+ assert traces[0]["answer"] == "Landing page shipped."
+ assert traces[0]["harness"] == "codex"
+ assert traces[0]["session_id"] == "codex-session"
+ assert traces[0]["turn_id"] == "turn-1"
+ assert traces[0]["cwd"] == "/work/project"
+ assert traces[0]["skills"] == [
+ {"name": "frontend-design", "revision": None},
+ {"name": "web-design-flow", "revision": "83a75cf1"},
+ ]
+ assert traces[0]["tags"] == [
+ "skill:frontend-design", "skill:web-design-flow",
+ "revision=web-design-flow@83a75cf1",
+ ]
+ assert traces[0]["usage"] == {
+ "input_tokens": 120, "output_tokens": 30,
+ "cached_input_tokens": 20, "reasoning_output_tokens": 4,
+ }
+ assert traces[0]["duration_ms"] == 3000
+ assert traces[0]["tool_errors"] == 1
+
+
+def test_claude_parser_uses_final_answer_and_skill_tool_without_tool_payloads(tmp_path):
+ session = tmp_path / "claude.jsonl"
+ records = [
+ {"timestamp": "2026-07-28T11:00:00Z", "type": "user", "sessionId": "claude-session",
+ "cwd": "/work/project", "isSidechain": False, "userType": "external",
+ "promptSource": "typed", "origin": "terminal",
+ "message": {"role": "user", "content": "Review this interface."}},
+ {"timestamp": "2026-07-28T11:00:01Z", "type": "assistant",
+ "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False,
+ "message": {"role": "assistant", "stop_reason": "tool_use",
+ "usage": {"input_tokens": 80, "output_tokens": 10,
+ "cache_read_input_tokens": 5},
+ "content": [{"type": "tool_use", "id": "tool-1", "name": "Skill",
+ "input": {"skill": "saas-interface-review", "args": ""}}]}},
+ {"timestamp": "2026-07-28T11:00:02Z", "type": "user",
+ "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False,
+ "message": {"role": "user", "content": [
+ {"type": "tool_result", "tool_use_id": "tool-1",
+ "content": "private skill instructions that must not enter the trace"},
+ ]}},
+ {"timestamp": "2026-07-28T11:00:02Z", "type": "user",
+ "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False,
+ "isMeta": True, "sourceToolUseID": "tool-1",
+ "message": {"role": "user", "content": [
+ {"type": "text", "text": "Base directory for this skill: /private/path"},
+ ]}},
+ {"timestamp": "2026-07-28T11:00:02Z", "type": "user",
+ "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False,
+ "promptSource": "queued",
+ "message": {"role": "user", "content": [
+ {"type": "text", "text": "synthetic status"},
+ ]}},
+ {"timestamp": "2026-07-28T11:00:03Z", "type": "assistant",
+ "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False,
+ "message": {"role": "assistant", "stop_reason": "end_turn", "usage": {},
+ "content": [{"type": "thinking", "thinking": "private reasoning"}]}},
+ {"timestamp": "2026-07-28T11:00:03Z", "type": "assistant",
+ "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False,
+ "message": {"role": "assistant", "stop_reason": "end_turn",
+ "usage": {"input_tokens": 30, "output_tokens": 20,
+ "cache_read_input_tokens": 2},
+ "content": [{"type": "text", "text": "The hierarchy needs work."}]}},
+ ]
+ _write_jsonl(session, records)
+
+ traces = local_traces.parse_claude_session(session)
+
+ assert len(traces) == 1
+ assert traces[0]["task"] == "Review this interface."
+ assert traces[0]["answer"] == "The hierarchy needs work."
+ assert traces[0]["skills"] == [{"name": "saas-interface-review", "revision": None}]
+ assert traces[0]["tags"] == ["skill:saas-interface-review"]
+ assert traces[0]["usage"] == {
+ "input_tokens": 110, "output_tokens": 30, "cached_input_tokens": 7,
+ }
+ assert "private skill instructions" not in json.dumps(traces[0])
+
+
+def test_claude_parser_pins_ingot_route_revision_from_matching_tool_result(tmp_path):
+ session = tmp_path / "claude-route.jsonl"
+ records = [
+ {"timestamp": "2026-07-28T12:00:00Z", "type": "user", "sessionId": "s1",
+ "cwd": "/work", "isSidechain": False, "userType": "external",
+ "promptSource": "typed", "origin": "terminal",
+ "message": {"role": "user", "content": "Merge the PDFs."}},
+ {"timestamp": "2026-07-28T12:00:01Z", "type": "assistant", "sessionId": "s1",
+ "cwd": "/work", "isSidechain": False, "message": {
+ "role": "assistant", "stop_reason": "tool_use", "usage": {},
+ "content": [{"type": "tool_use", "id": "route-1",
+ "name": "mcp__ingot__route_and_load",
+ "input": {"task": "Merge the PDFs.", "harness": "claude", "cwd": "/work"}}],
+ }},
+ {"timestamp": "2026-07-28T12:00:02Z", "type": "user", "sessionId": "s1",
+ "cwd": "/work", "isSidechain": False, "userType": "external", "message": {
+ "role": "user", "content": [{"type": "tool_result", "tool_use_id": "route-1",
+ "content": json.dumps({"match": "pdf", "related_match": None,
+ "revision": "83a75cf1", "skill_body": "private"})}],
+ }},
+ {"timestamp": "2026-07-28T12:00:03Z", "type": "assistant", "sessionId": "s1",
+ "cwd": "/work", "isSidechain": False, "message": {
+ "role": "assistant", "stop_reason": "end_turn", "usage": {},
+ "content": [{"type": "text", "text": "Merged."}],
+ }},
+ ]
+ _write_jsonl(session, records)
+
+ traces = local_traces.parse_claude_session(session)
+
+ assert traces[0]["skills"] == [{"name": "pdf", "revision": "83a75cf1"}]
+ assert traces[0]["tags"] == ["skill:pdf", "revision=pdf@83a75cf1"]
+ assert "skill_body" not in json.dumps(traces[0])
+
+
+def test_route_identity_bounds_nested_envelopes_and_revision_content():
+ assert local_traces._route_identity({
+ "related_match": "pdf", "revision": "83a75cf1",
+ }) == ("pdf", "83a75cf1")
+ assert local_traces._route_identity({
+ "match": "pdf", "revision": {"secret": "must not become a tag"},
+ }) == ("pdf", None)
+ nested = {"match": "pdf", "revision": "83a75cf1"}
+ for _ in range(local_traces._MAX_ROUTE_DEPTH + 1):
+ nested = {"result": nested}
+ assert local_traces._route_identity(nested) is None
+ assert local_traces._route_identity("x" * (local_traces._MAX_ROUTE_ENVELOPE + 1)) is None
+ assert local_traces._route_identity({
+ "content": [{"type": "text", "text": json.dumps({"match": "not-an-envelope"})}],
+ }) is None
+ assert local_traces._route_identity({"match": "pdf", "score": 0.99}) is None
+ wide = [{"type": "text", "text": "{}"}] * local_traces._MAX_ROUTE_NODES
+ wide.append({"type": "text", "text": json.dumps({
+ "match": "pdf", "revision": "83a75cf1",
+ })})
+ assert local_traces._route_identity(wide) is None
+ assert local_traces._tool_failed({"exit_code": "0"}) is False
+ assert local_traces._tool_failed({"exit_code": "2"}) is True
+ assert local_traces._tool_failed({"error": "documented payload field"}) is False
+
+
+def test_scan_writes_deduplicated_store_and_summary_without_answers(tmp_path):
+ codex = tmp_path / "codex"
+ claude = tmp_path / "claude"
+ output = tmp_path / "runs" / "local_traces.json"
+ codex_record = [
+ {"timestamp": "2026-07-28T10:00:00Z", "type": "session_meta", "payload": {
+ "id": "s", "cwd": "/work", "thread_source": "user",
+ }},
+ {"timestamp": "2026-07-28T10:00:01Z", "type": "response_item", "payload": {
+ "type": "message", "role": "user",
+ "content": [{"type": "input_text", "text": "Task"}],
+ }},
+ {"timestamp": "2026-07-28T10:00:02Z", "type": "event_msg", "payload": {
+ "type": "task_started", "turn_id": "t",
+ }},
+ {"timestamp": "2026-07-28T10:00:03Z", "type": "event_msg", "payload": {
+ "type": "task_complete", "turn_id": "t", "last_agent_message": "Secret answer",
+ }},
+ ]
+ _write_jsonl(codex / "one.jsonl", codex_record)
+ _write_jsonl(codex / "copy.jsonl", codex_record)
+
+ result = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output)
+ saved = json.loads(output.read_text())
+ summary = local_traces.store_summary(output)
+
+ assert result["traces"] == 1
+ assert saved["schema_version"] == "ingot/local-traces/v1"
+ assert len(saved["traces"]) == 1
+ assert summary["configured"] is True
+ assert summary["total"] == 1
+ assert summary["harnesses"] == {"codex": 1}
+ assert "task" not in summary["recent"][0]
+ assert local_traces.store_summary(output, include_tasks=True)["recent"][0]["task"] == "Task"
+ assert "answer" not in summary["recent"][0]
+ assert "Secret answer" not in json.dumps(summary)
+
+
+def test_scan_reuses_unchanged_files_and_reparses_changed_files(tmp_path, monkeypatch):
+ codex = tmp_path / "codex"
+ claude = tmp_path / "claude"
+ output = tmp_path / "local_traces.json"
+ session = codex / "one.jsonl"
+ _write_jsonl(session, [
+ {"timestamp": "2026-07-28T10:00:00Z", "type": "session_meta",
+ "payload": {"id": "s", "cwd": "/work/project", "thread_source": "user"}},
+ {"timestamp": "2026-07-28T10:00:01Z", "type": "response_item",
+ "payload": {"type": "message", "role": "user",
+ "content": [{"type": "input_text", "text": "Task"}]}},
+ {"timestamp": "2026-07-28T10:00:02Z", "type": "event_msg",
+ "payload": {"type": "task_started", "turn_id": "t"}},
+ {"timestamp": "2026-07-28T10:00:03Z", "type": "event_msg",
+ "payload": {"type": "task_complete", "turn_id": "t",
+ "last_agent_message": "Answer"}},
+ ])
+ first = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output)
+ assert first["parsed_files"] == 1 and first["reused_files"] == 0
+
+ monkeypatch.setattr(
+ local_traces, "parse_codex_session",
+ lambda _path: pytest.fail("unchanged transcript must come from the cursor"),
+ )
+ second = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output)
+ assert second["parsed_files"] == 0 and second["reused_files"] == 1
+ assert second["traces"] == 1
+
+ calls = []
+ monkeypatch.setattr(
+ local_traces, "parse_codex_session",
+ lambda path: calls.append(path) or [],
+ )
+ with session.open("a") as handle:
+ handle.write(" \n")
+ os.utime(session, None)
+ third = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output)
+ assert calls == [session]
+ assert third["parsed_files"] == 1 and third["reused_files"] == 0
+
+ calls.clear()
+ fourth = local_traces.scan(
+ codex_dir=codex, claude_dir=claude, output=output, force=True)
+ assert calls == [session]
+ assert fourth["parsed_files"] == 1 and fourth["reused_files"] == 0
+
+
+def test_scan_rebuilds_malformed_prior_cursor_instead_of_crashing(tmp_path):
+ codex = tmp_path / "codex"
+ claude = tmp_path / "claude"
+ output = tmp_path / "local_traces.json"
+ _write_jsonl(codex / "one.jsonl", [])
+ output.write_text(json.dumps({
+ "schema_version": local_traces.SCHEMA,
+ "parser_version": local_traces.PARSER_VERSION,
+ "filters": {"projects": [], "since": "", "until": ""},
+ "source_index": {"codex": {}, "claude": {}},
+ "traces": [{}],
+ }))
+
+ result = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output)
+
+ assert result["parsed_files"] == 1
+ assert json.loads(output.read_text())["traces"] == []
+
+
+def test_scan_and_summary_filter_by_project_harness_and_date(tmp_path):
+ store = tmp_path / "local_traces.json"
+ payload = {
+ "schema_version": local_traces.SCHEMA,
+ "generated_at": 1,
+ "traces": [
+ {"id": "one", "timestamp": "2026-07-27T10:00:00Z", "project": "alpha",
+ "harness": "codex", "task": "one", "answer": "a", "skills": [], "usage": {}},
+ {"id": "two", "timestamp": "2026-07-28T10:00:00Z", "project": "beta",
+ "harness": "claude", "task": "two", "answer": "b", "skills": [], "usage": {}},
+ {"id": "three", "timestamp": "2026-07-29T10:00:00Z", "project": "alpha",
+ "harness": "claude", "task": "three", "answer": "c", "skills": [], "usage": {}},
+ ],
+ }
+ store.write_text(json.dumps(payload))
+
+ summary = local_traces.store_summary(
+ store, project="alpha", harness="claude", since="2026-07-28", until="2026-07-29",
+ include_tasks=True)
+
+ assert summary["total"] == 1
+ assert summary["recent"][0]["task"] == "three"
+ assert summary["available_total"] == 3
+ assert summary["available_projects"] == {"alpha": 2, "beta": 1}
+ assert summary["available_harnesses"] == {"claude": 2, "codex": 1}
+ with pytest.raises(ValueError, match="ISO date"):
+ local_traces.store_summary(store, since="yesterday")
+
+ store.write_text("{broken")
+ assert local_traces.store_summary(store)["status"] == "unreadable"
+
+
+def test_store_summary_filters_turns_by_observed_skill(tmp_path):
+ """The traces page lists which skills were observed but could not narrow to one, so 'where was
+ build-loop actually used' meant reading 4,639 turns by eye."""
+ store = tmp_path / "store.json"
+ store.write_text(json.dumps({
+ "schema_version": local_traces.SCHEMA,
+ "generated_at": "2026-08-07T00:00:00Z",
+ "traces": [
+ {"id": "a", "timestamp": "2026-08-01T10:00:00Z", "project": "p", "harness": "claude",
+ "task": "a", "answer": "a", "usage": {},
+ "skills": [{"name": "build-loop"}, {"name": "sota-check"}]},
+ {"id": "b", "timestamp": "2026-08-02T10:00:00Z", "project": "p", "harness": "codex",
+ "task": "b", "answer": "b", "usage": {}, "skills": [{"name": "sota-check"}]},
+ {"id": "c", "timestamp": "2026-08-03T10:00:00Z", "project": "p", "harness": "claude",
+ "task": "c", "answer": "c", "usage": {}, "skills": []},
+ ],
+ }))
+
+ summary = local_traces.store_summary(store, skill="build-loop")
+ assert summary["total"] == 1
+ assert summary["recent"][0]["id"] == "a"
+ assert summary["filters"]["skill"] == "build-loop"
+
+ # Counted over every turn, not the filtered set: otherwise selecting a skill empties the
+ # picker that selected it and you cannot switch to another.
+ assert summary["available_skills"] == {"sota-check": 2, "build-loop": 1}
+ assert local_traces.store_summary(store)["total"] == 3
diff --git a/tests/test_loop.py b/tests/test_loop.py
index b5e4e53..4e6f529 100644
--- a/tests/test_loop.py
+++ b/tests/test_loop.py
@@ -1,5 +1,5 @@
"""Unit tests for the continuous loop's health-gating (mine + run_ab are mocked)."""
-from optimize import loop as L
+from ingot.optimize import loop as L
def test_loop_skips_healthy_skills(monkeypatch):
@@ -53,7 +53,7 @@ def test_loop_runs_description_pass_when_configured(monkeypatch):
monkeypatch.setattr(L, "run_ab",
lambda skill, **k: order.append("body") or {"improved": False,
"gate": {"promotable": False}})
- import optimize.routing as routing_mod
+ import ingot.optimize.routing as routing_mod
monkeypatch.setattr(routing_mod, "run_routing",
lambda skill, **k: order.append("description") or {"improved": True,
"gate": {"promotable": True}})
diff --git a/tests/test_mine.py b/tests/test_mine.py
index 8d775ba..3a5a137 100644
--- a/tests/test_mine.py
+++ b/tests/test_mine.py
@@ -1,11 +1,11 @@
-"""Unit tests for success/failure mining (optimize.mine), Langfuse HTTP and the judge are mocked."""
+"""Unit tests for success/failure mining (ingot.optimize.mine), Langfuse HTTP and the judge are mocked."""
import json
from urllib.parse import parse_qs, urlparse
import pytest
-from optimize import mine
-from optimize.judge import DIMENSIONS
+from ingot.optimize import mine
+from ingot.optimize.judge import DIMENSIONS
class _Resp:
@@ -36,6 +36,51 @@ def test_fetch_traces_keeps_only_usable_task_output_pairs(monkeypatch):
("do X", "r"), ("a bare string", "")]
+def test_fetch_local_traces_requires_normalized_store_and_honors_newest_limit(tmp_path, monkeypatch):
+ path = tmp_path / "local-traces.json"
+ path.write_text(json.dumps({
+ "schema_version": "ingot/local-traces/v1",
+ "traces": [
+ {"id": "old", "timestamp": "2026-07-27T10:00:00Z", "task": "old task",
+ "answer": "old answer", "tags": ["skill:pdf"]},
+ {"id": "new", "timestamp": "2026-07-28T10:00:00Z", "task": "new task",
+ "answer": "new answer", "tags": ["skill:pdf"]},
+ ],
+ }))
+ monkeypatch.setattr(mine, "LOCAL_TRACE_FILE", path)
+
+ assert mine.fetch_local_traces(1) == [{
+ "task": "new task", "rubric": "", "answer": "new answer", "tags": ["skill:pdf"],
+ }]
+
+ path.write_text(json.dumps({"schema_version": "wrong", "traces": []}))
+ with pytest.raises(SystemExit, match="unsupported local trace store"):
+ mine.fetch_local_traces()
+
+
+def test_mine_local_source_requires_consent_and_never_reads_langfuse(monkeypatch):
+ traces = [{"task": "t", "rubric": "", "answer": "a", "tags": ["pdf"]}]
+ monkeypatch.setattr(mine, "fetch_local_traces", lambda limit: traces)
+ monkeypatch.setattr(mine, "fetch_traces",
+ lambda limit: pytest.fail("local mining must not contact Langfuse"))
+ monkeypatch.setattr(mine, "relevant_traces", lambda items, skill, k=5: items)
+ monkeypatch.setattr(mine, "_cluster_traces", lambda items, embed: [[0]])
+ monkeypatch.setattr(mine, "_normalized_embedder", lambda: object())
+ monkeypatch.setattr(mine, "_judge_trace_clusters",
+ lambda items, clusters, log, cache_path: ([
+ (0, {"score": 1.0, "dimensions": {d: "pass" for d in DIMENSIONS}}, 1),
+ ], 0, 0))
+ monkeypatch.setattr(mine, "_select_candidates", lambda *args, **kwargs: [])
+
+ with pytest.raises(SystemExit, match="allow-external-judge"):
+ mine.mine("pdf", source="local", log=lambda *_args: None)
+
+ result = mine.mine(
+ "pdf", source="local", allow_external_judge=True, log=lambda *_args: None)
+
+ assert result["traces"] == 1
+
+
def test_fetch_traces_parses_langgraph_agent_traces(monkeypatch):
lg = lambda task, answer: {"input": {"messages": [{"role": "user", "content": task}]},
"output": {"messages": [{"type": "ai", "content": answer}]}, "tags": ["demo"]}
@@ -175,8 +220,8 @@ def __init__(self, skills):
def suggest(self, task, k=5, min_score=0.0):
return [{"name": "excel"}] if "spreadsheet" in task else [{"name": "other"}]
- monkeypatch.setattr("mcp_server.router.Router", FakeRouter)
- monkeypatch.setattr("mcp_server.registry.load_skills", lambda: [])
+ monkeypatch.setattr("ingot.mcp_server.router.Router", FakeRouter)
+ monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda: [])
traces = [
{"task": "spreadsheet lookup with discounts", "tags": []}, # ranked -> keep (misrouted traffic)
{"task": "rotate a pdf", "tags": ["excel"]}, # tagged -> keep
@@ -185,6 +230,33 @@ def suggest(self, task, k=5, min_score=0.0):
assert mine.relevant_traces(traces, "excel") == traces[:2]
+def test_relevant_traces_matches_every_harness_tag_spelling(monkeypatch):
+ """Real traffic is tagged by whichever harness produced it, not by us. Claude Code writes
+ `skill:`, namespaced by plugin when the skill came from one; our own agent writes the
+ bare name plus a `revision=@` pin. Matching the bare form alone drops every
+ externally-produced trace, and mining then reports a heavily-used skill as never used."""
+ class FakeRouter:
+ def __init__(self, skills):
+ pass
+
+ def suggest(self, task, k=5, min_score=0.0):
+ return [{"name": "other"}] # never ranks, so the tag check alone decides
+
+ monkeypatch.setattr("ingot.mcp_server.router.Router", FakeRouter)
+ monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda: [])
+ traces = [
+ {"task": "t", "tags": ["pdf"]}, # bare, our agent
+ {"task": "t", "tags": ["claude-code", "skill:pdf"]}, # Claude Code
+ {"task": "t", "tags": ["skill:superpowers:pdf"]}, # plugin-namespaced
+ {"task": "t", "tags": ["demo", "revision=pdf@83a75cf1f9b5ada5"]}, # revision pin
+ {"task": "t", "tags": ["skill:pdf-tools"]}, # neighbour -> drop
+ {"task": "t", "tags": ["revision=excel@abc123"]}, # other skill -> drop
+ {"task": "t", "tags": []}, # untagged -> drop
+ {"task": "t"}, # no tags key -> drop
+ ]
+ assert mine.relevant_traces(traces, "pdf") == traces[:4]
+
+
def test_mine_exits_when_no_traces_are_relevant(monkeypatch):
import pytest
monkeypatch.setattr(mine, "fetch_traces",
diff --git a/tests/test_optimize.py b/tests/test_optimize.py
index 4971d97..a7fec63 100644
--- a/tests/test_optimize.py
+++ b/tests/test_optimize.py
@@ -4,32 +4,37 @@
import pytest
-from optimize import ab as ab_mod
-from optimize import judge as judge_mod
-from optimize.ab import body_retention, promotion_gate, retention_warnings
-from optimize.judge import DIMENSIONS, failed_dimensions
+from ingot.optimize import ab as ab_mod
+from ingot.optimize import judge as judge_mod
+from ingot.optimize.ab import body_retention, promotion_gate, retention_warnings
+from ingot.optimize.judge import DIMENSIONS, failed_dimensions
def test_optimizer_has_no_activation_control():
# the canary module is deleted outright on this branch, the strongest form of "no
# activation control"; the remaining assertions cover the surviving surfaces
- from optimize import promote as promotion
+ from ingot.optimize import promote as promotion
assert "promote_now" not in inspect.signature(ab_mod.run_ab).parameters
assert not hasattr(promotion, "promote")
-def test_ensemble_judge_averages_score_and_majority_votes_dimensions(monkeypatch):
- # two judges: one says 1.0 all-pass, one says 0.0 with a correctness failure -> mean 0.5, and
- # correctness fails only if a MAJORITY flag it (here 1 of 2 → still "pass", harder to game)
- fake = iter([
- {"score": 1.0, "feedback": "great", "dimensions": {d: "pass" for d in DIMENSIONS}},
- {"score": 0.0, "feedback": "wrong", "dimensions": {**{d: "pass" for d in DIMENSIONS}, "correctness": "bad API"}},
- ])
+def test_ensemble_judge_averages_items_and_majority_votes_dimensions(monkeypatch):
+ # Two judges disagreeing about correctness and agreeing on everything else. Averaging happens
+ # per checklist item, so the disagreement costs half of correctness's weight (3 of 8) rather
+ # than half of the whole answer: 1 - 0.5*(3/8) = 0.8125. Under the old holistic contract the
+ # same disagreement scored 0.5, which charged the challenger for three checks both judges
+ # passed. Correctness still reads as "pass" because 1 of 2 is not a majority.
+ def items(correctness):
+ return {"items": {i["id"]: {"value": correctness if i["id"] == "correctness" else 1.0,
+ "note": "bad API" if i["id"] == "correctness" else ""}
+ for i in judge_mod.DEFAULT_CHECKLIST},
+ "feedback": "f", "unparseable": False}
+ fake = iter([items(1.0), items(0.0)])
monkeypatch.setattr(judge_mod, "MODELS", ["m1", "m2"])
- monkeypatch.setattr(judge_mod, "_judge_one", lambda model, prompt: next(fake))
+ monkeypatch.setattr(judge_mod, "_judge_one", lambda model, prompt, checklist: next(fake))
r = judge_mod.judge("t", "rubric", "ans")
- assert abs(r["score"] - 0.5) < 1e-9
+ assert abs(r["score"] - 0.8125) < 1e-9
assert failed_dimensions(r["dimensions"]) == [] # 1/2 is not a majority → not flagged
@@ -125,9 +130,51 @@ def test_load_tasks_reads_explicit_train_holdout(tmp_path, monkeypatch):
assert split == {"kind": "holdout", "leakage": False}
+def test_load_tasks_drafts_from_the_skills_own_root(tmp_path, monkeypatch):
+ """A multi-root library serves most skills from read-only mounts, never from the writable
+ authoring root. Drafting must resolve the skill where the registry indexed it — looking under
+ SKILLS_DIR instead meant every mounted skill died on FileNotFoundError before a task set
+ could be written."""
+ from ingot.mcp_server.registry import Skill
+
+ mounted = tmp_path / "mounted-library" / "gb10-serving"
+ mounted.mkdir(parents=True)
+ monkeypatch.setattr(ab_mod, "TASKS_DIR", tmp_path / "tasks")
+ (tmp_path / "tasks").mkdir()
+ monkeypatch.setattr("ingot.mcp_server.registry.load_skills",
+ lambda *a, **k: [Skill(name="gb10-serving", description="d", body="b",
+ path=str(mounted / "SKILL.md"), root=str(mounted))])
+ monkeypatch.setattr("ingot.mcp_server.registry.read_components",
+ lambda d: {"description": f"desc-from:{d}", "body": "body"})
+
+ seen = {}
+
+ def fake_draft(name, description, body, tasks_dir, **kw):
+ seen.update(name=name, description=description)
+ (tasks_dir / f"{name}.yaml").write_text(
+ "skill: gb10-serving\ntrain:\n - task: a\n rubric: r\nholdout:\n - task: b\n rubric: r\n")
+
+ monkeypatch.setattr("ingot.optimize.draft.draft_and_save", fake_draft)
+ train, holdout, split = ab_mod.load_tasks("gb10-serving")
+
+ assert seen["description"] == f"desc-from:{mounted}" # the mount, not SKILLS_DIR
+ assert [t["task"] for t in train] == ["a"] and split["leakage"] is False
+
+
+def test_load_tasks_names_the_skill_when_it_is_not_indexed(tmp_path, monkeypatch):
+ """An unindexed name means the roots are misconfigured. Say so — the old code raised a bare
+ FileNotFoundError on a path the operator never configured directly."""
+ import pytest
+ monkeypatch.setattr(ab_mod, "TASKS_DIR", tmp_path)
+ monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda *a, **k: [])
+ with pytest.raises(SystemExit) as exc:
+ ab_mod.load_tasks("nonexistent")
+ assert "nonexistent" in str(exc.value) and "SKILL_ROUTER_PATHS" in str(exc.value)
+
+
def test_greedy_pick_spreads_across_failure_modes():
import numpy as np
- from optimize.mine import _greedy_pick
+ from ingot.optimize.mine import _greedy_pick
# two orthogonal "failure modes", two near-identical tasks in each; hardest overall is
# index 0, but the second pick must come from the OTHER mode even though index 1 is harder
vecs = np.array([[1.0, 0.0], [1.0, 0.0], [0.0, 1.0], [0.0, 1.0]], dtype=np.float32)
@@ -137,27 +184,28 @@ def test_greedy_pick_spreads_across_failure_modes():
def test_greedy_pick_skips_aced_and_excluded_tasks():
import numpy as np
- from optimize.mine import _greedy_pick
+ from ingot.optimize.mine import _greedy_pick
vecs = np.eye(3, dtype=np.float32)
# one real candidate, one aced task (difficulty 0), one excluded (train near-duplicate)
assert _greedy_pick([0.7, 0.0, -1.0], vecs, k=3) == [0]
def test_save_pending_archives_a_displaced_cross_pass_challenger(tmp_path, monkeypatch):
- from optimize import promote as promote_mod
- monkeypatch.setattr(promote_mod, "PENDING_DIR", tmp_path)
+ from ingot.optimize import promote as promote_mod
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path))
+ pending = promote_mod.pending_dir()
promote_mod.save_pending("pdf", {"changed_components": ["body"], "created": 111})
promote_mod.save_pending("pdf", {"changed_components": ["body"], "created": 222}) # same pass: overwrite
- assert len(list(tmp_path.glob("pdf*"))) == 1
+ assert len(list(pending.glob("pdf*"))) == 1
promote_mod.save_pending("pdf", {"changed_components": ["description"], "created": 333})
import json
- archived = tmp_path / "pdf.displaced-222.json"
+ archived = pending / "pdf.displaced-222.json"
assert json.loads(archived.read_text())["changed_components"] == ["body"] # preserved
- assert json.loads((tmp_path / "pdf.json").read_text())["changed_components"] == ["description"]
+ assert json.loads((pending / "pdf.json").read_text())["changed_components"] == ["description"]
def test_length_penalty_is_zero_under_target_and_grows_above():
- from optimize.rollout import BODY_TARGET_CHARS, LENGTH_PENALTY, length_penalty
+ from ingot.optimize.rollout import BODY_TARGET_CHARS, LENGTH_PENALTY, length_penalty
assert length_penalty("x" * (BODY_TARGET_CHARS // 2)) == 0.0 # concise -> no penalty
assert length_penalty("x" * BODY_TARGET_CHARS) == 0.0 # exactly at target -> no penalty
assert length_penalty("x" * (BODY_TARGET_CHARS * 2)) > 0.0 # bloated -> penalized
@@ -167,7 +215,7 @@ def test_length_penalty_is_zero_under_target_and_grows_above():
def test_reflection_lm_sends_generic_api_key_to_openrouter(monkeypatch):
import sys
from types import SimpleNamespace
- from optimize import rollout as rollout_mod
+ from ingot.optimize import rollout as rollout_mod
captured = {}
@@ -237,7 +285,7 @@ def test_gate_ignores_parity_with_no_cases():
def test_zdr_provider_pinned_and_in_sync():
- from optimize.judge import ZDR_PROVIDER
+ from ingot.optimize.judge import ZDR_PROVIDER
assert ZDR_PROVIDER == {"provider": {"zdr": True, "data_collection": "deny"}}
from agent.run import ZDR_PROVIDER as agent_zdr
assert agent_zdr == ZDR_PROVIDER # duplicated literal (import-weight reasons) must not drift
@@ -275,7 +323,7 @@ def test_optimize_split_rejects_unknown_component(monkeypatch):
def test_skill_adapter_renders_frozen_components():
- from optimize.rollout import assemble
+ from ingot.optimize.rollout import assemble
frozen, candidate = {"description": "when to use me"}, {"body": "the rules"}
text = assemble({**frozen, **candidate})
assert "when to use me" in text and "the rules" in text
@@ -297,8 +345,8 @@ def test_eval_serve_template_injects_body_and_contract():
def test_rollouts_serve_the_exact_serving_contract(monkeypatch):
- from optimize import SERVE_TEMPLATE
- from optimize import rollout as R
+ from ingot.optimize import SERVE_TEMPLATE
+ from ingot.optimize import rollout as R
captured = {}
class FakeLLM:
@@ -311,7 +359,7 @@ class Msg:
adapter = R.SkillAdapter(frozen={"description": "trigger words"})
adapter._llm = FakeLLM()
monkeypatch.setattr(R, "judge",
- lambda t, r, a, reference="", check=None, deliverable=None:
+ lambda t, r, a, reference="", check=None, deliverable=None, checklist=None:
{"score": 1.0, "feedback": "f", "dimensions": {}})
candidate = {"body": "the rules"}
answer, score, _ = adapter._rollout(adapter.serve(candidate), {"task": "t", "rubric": "r"})
@@ -323,10 +371,10 @@ class Msg:
def test_agent_rollout_mode_routes_through_the_scaffold(monkeypatch):
- from optimize import rollout as R
+ from ingot.optimize import rollout as R
monkeypatch.setattr(R, "SKILLOPT_ROLLOUTS", "agent")
monkeypatch.setattr(R, "judge",
- lambda t, r, a, reference="", check=None, deliverable=None:
+ lambda t, r, a, reference="", check=None, deliverable=None, checklist=None:
{"score": 0.5, "feedback": "f", "dimensions": {}})
seen = {}
@@ -398,7 +446,7 @@ def _script_skill(tmp_path, monkeypatch, scripts=("scripts/helper.py",), holdout
holdout[0]["check"] = {"fixture": "x = 1", "assert": "assert x == 1"}
(tasks / "excel.yaml").write_text(yaml.safe_dump(
{"train": [{"task": "t2", "rubric": "r"}], "holdout": holdout}))
- monkeypatch.setattr(ab_mod, "SKILLS_DIR", skills)
+ monkeypatch.setattr(ab_mod, "resolve_skill_dir", lambda name: skills / name)
monkeypatch.setattr(ab_mod, "TASKS_DIR", tasks)
return d
@@ -441,7 +489,7 @@ def fake_skillopt(seed, tasks, frozen=None, acceptance=None, log=print):
calls.append({"key": key, "frozen": dict(frozen)})
return {key: text + "!"}, 0.5, 0.9
- monkeypatch.setattr("optimize.skillopt_loop.run_skillopt", fake_skillopt)
+ monkeypatch.setattr("ingot.optimize.skillopt_loop.run_skillopt", fake_skillopt)
champion = {"description": "d", "body": "b",
"file:scripts/a.py": "A", "file:scripts/b.py": "B"}
challenger, seed_score, best_score = ab_mod._greedy_search(
@@ -465,7 +513,7 @@ def fake_skillopt(seed, tasks, frozen=None, acceptance=None, log=print):
calls.append({"seed": dict(seed), "frozen": dict(frozen)})
return {"body": "better"}, 0.3, 0.8
- monkeypatch.setattr("optimize.skillopt_loop.run_skillopt", fake_skillopt)
+ monkeypatch.setattr("ingot.optimize.skillopt_loop.run_skillopt", fake_skillopt)
champion = {"description": "d", "body": "b"}
challenger, s0, s1 = ab_mod._greedy_search("excel", champion, ["body"],
[{"task": "t", "rubric": "r"}], log=lambda *_: None)
diff --git a/tests/test_paths.py b/tests/test_paths.py
new file mode 100644
index 0000000..91aa599
--- /dev/null
+++ b/tests/test_paths.py
@@ -0,0 +1,183 @@
+"""Where mutable state lives.
+
+The defect these protect against is not hypothetical: a `pip install ingot` kept its review queue,
+its receipts, and its served library inside `site-packages`, so an upgrade discarded them, a
+read-only install could not start, and two deployments sharing one installation shared one queue."""
+import os
+import subprocess
+import sys
+from pathlib import Path
+
+import pytest
+
+from ingot import paths
+
+
+def _clear(monkeypatch):
+ for name in (paths.HOME, paths.LIBRARY, paths.RUNS, paths.TASKS, paths.VAULT,
+ *paths.LEGACY.values(), "XDG_STATE_HOME"):
+ monkeypatch.delenv(name, raising=False)
+
+
+def test_state_defaults_outside_the_installed_package(monkeypatch):
+ """The whole point. Anything under the package directory is discarded by the next upgrade."""
+ _clear(monkeypatch)
+
+ for path in (paths.home(), paths.library(), paths.runs(), paths.tasks(), paths.vault()):
+ assert not path.is_relative_to(paths.PACKAGE_ROOT), path
+
+
+def test_the_default_is_xdg(monkeypatch, tmp_path):
+ _clear(monkeypatch)
+ monkeypatch.setenv("XDG_STATE_HOME", str(tmp_path))
+
+ assert paths.home() == tmp_path / "ingot"
+ assert paths.runs() == tmp_path / "ingot" / "runs"
+
+
+def test_without_xdg_it_falls_under_the_user_home(monkeypatch):
+ _clear(monkeypatch)
+
+ assert paths.home() == Path.home() / ".local" / "state" / "ingot"
+
+
+def test_one_home_moves_every_store(monkeypatch, tmp_path):
+ _clear(monkeypatch)
+ monkeypatch.setenv(paths.HOME, str(tmp_path))
+
+ assert paths.library() == tmp_path / "library"
+ assert paths.runs() == tmp_path / "runs"
+ assert paths.tasks() == tmp_path / "tasks"
+
+
+@pytest.mark.parametrize("setting,accessor,default", [
+ (paths.LIBRARY, paths.library, "library"),
+ (paths.RUNS, paths.runs, "runs"),
+ (paths.TASKS, paths.tasks, "tasks"),
+])
+def test_each_store_can_be_placed_on_its_own(monkeypatch, tmp_path, setting, accessor, default):
+ """A container mounts each one separately; it does not get to choose a single parent."""
+ _clear(monkeypatch)
+ monkeypatch.setenv(paths.HOME, str(tmp_path / "home"))
+ monkeypatch.setenv(setting, str(tmp_path / "elsewhere"))
+
+ assert accessor() == tmp_path / "elsewhere"
+ assert paths.home() == tmp_path / "home"
+
+
+def test_the_vault_defaults_to_the_library(monkeypatch, tmp_path):
+ """In the local backend the served checkout is the vault. Defaulting them apart would invent a
+ projection step that does not exist."""
+ _clear(monkeypatch)
+ monkeypatch.setenv(paths.LIBRARY, str(tmp_path / "library"))
+
+ assert paths.vault() == tmp_path / "library"
+
+ monkeypatch.setenv(paths.VAULT, str(tmp_path / "vault"))
+ assert paths.vault() == tmp_path / "vault"
+
+
+@pytest.mark.parametrize("setting,legacy", sorted(paths.LEGACY.items()))
+def test_the_pre_existing_names_still_work(monkeypatch, tmp_path, setting, legacy):
+ """An existing deployment must not break on upgrade, and `ingot status` says which name won."""
+ _clear(monkeypatch)
+ monkeypatch.setenv(legacy, str(tmp_path / "old"))
+
+ assert paths._env(setting) == str(tmp_path / "old")
+ sources = " ".join(entry["source"] for entry in paths.resolved())
+ assert legacy in sources and "deprecated" in sources
+
+
+def test_the_explicit_setting_outranks_the_legacy_one(monkeypatch, tmp_path):
+ _clear(monkeypatch)
+ monkeypatch.setenv(paths.LIBRARY, str(tmp_path / "new"))
+ monkeypatch.setenv("SKILLS_DIR", str(tmp_path / "old"))
+
+ assert paths.library() == tmp_path / "new"
+
+
+def test_resolved_reports_where_every_path_came_from(monkeypatch, tmp_path):
+ _clear(monkeypatch)
+ monkeypatch.setenv(paths.HOME, str(tmp_path))
+ monkeypatch.setenv(paths.RUNS, str(tmp_path / "elsewhere"))
+
+ report = {entry["name"]: entry for entry in paths.resolved()}
+
+ assert report["runs"]["source"] == paths.RUNS
+ assert report["library"]["source"] == paths.HOME
+ assert report["runs"]["path"] == str(tmp_path / "elsewhere")
+
+
+def test_an_unwritable_parent_is_reported_rather_than_discovered_on_first_write(monkeypatch,
+ tmp_path):
+ """A path that does not exist yet is fine; one whose parent cannot be written is not, and
+ finding out at the first approval is the stall this reports instead."""
+ locked = tmp_path / "locked"
+ locked.mkdir()
+ locked.chmod(0o500)
+ real_access = paths.os.access
+ monkeypatch.setattr(paths.os, "access", lambda path, mode: False
+ if Path(path) == locked else real_access(path, mode))
+ _clear(monkeypatch)
+ monkeypatch.setenv(paths.HOME, str(locked / "ingot"))
+ try:
+ report = {entry["name"]: entry for entry in paths.resolved()}
+ finally:
+ locked.chmod(0o700)
+
+ assert report["runs"]["exists"] is False
+ assert report["runs"]["writable"] is False
+
+
+@pytest.mark.skipif(os.getuid() == 0, reason="root writes a mode-500 directory regardless")
+def test_a_writable_missing_path_is_not_reported_as_a_problem(monkeypatch, tmp_path):
+ _clear(monkeypatch)
+ monkeypatch.setenv(paths.HOME, str(tmp_path / "not-created-yet"))
+
+ report = {entry["name"]: entry for entry in paths.resolved()}
+
+ assert report["runs"]["exists"] is False
+ assert report["runs"]["writable"] is True
+
+
+def test_legacy_state_is_reported_never_migrated(monkeypatch, tmp_path):
+ """Moving someone's review queue on their behalf is a change to controlled state made by a
+ process nobody asked to make it."""
+ monkeypatch.setattr(paths, "PACKAGE_ROOT", tmp_path)
+ assert paths.legacy_state() == []
+
+ (tmp_path / "runs" / "pending").mkdir(parents=True)
+ (tmp_path / "runs" / "pending" / "pdf.json").write_text("{}")
+
+ assert paths.legacy_state() == [str(tmp_path / "runs")]
+
+
+def test_an_empty_package_directory_is_not_legacy_state(monkeypatch, tmp_path):
+ """A checkout ships `skills/.gitkeep`. Reporting that as leftover state would train people to
+ ignore the warning."""
+ monkeypatch.setattr(paths, "PACKAGE_ROOT", tmp_path)
+ (tmp_path / "skills").mkdir()
+ (tmp_path / "skills" / ".gitkeep").touch()
+
+ assert paths.legacy_state() == []
+
+
+def test_an_installed_ingot_keeps_no_state_inside_the_package(tmp_path):
+ """In situ, because this is exactly the failure a unit test cannot see: the process must be
+ started somewhere other than the checkout, or the checkout's own directories answer for it."""
+ script = ("import json, os, sys\n"
+ "os.environ['XDG_STATE_HOME'] = sys.argv[1]\n"
+ "from ingot import paths\n"
+ "print(json.dumps([entry['path'] for entry in paths.resolved()]))\n")
+ environment = {key: value for key, value in os.environ.items()
+ if not key.startswith(("INGOT_", "SKILLS_DIR", "VAULT_DIR", "XDG_"))}
+ environment["PYTHONPATH"] = str(Path(__file__).resolve().parent.parent)
+
+ result = subprocess.run([sys.executable, "-c", script, str(tmp_path)], cwd=tmp_path,
+ capture_output=True, text=True, env=environment)
+
+ assert result.returncode == 0, result.stderr
+ import json
+ for path in json.loads(result.stdout):
+ assert path.startswith(str(tmp_path)), path
+ assert not Path(path).is_relative_to(paths.PACKAGE_ROOT), path
diff --git a/tests/test_promote.py b/tests/test_promote.py
index ef7eca3..abe0a81 100644
--- a/tests/test_promote.py
+++ b/tests/test_promote.py
@@ -4,8 +4,9 @@
import pytest
-from mcp_server.registry import load_skills, optimizable_components, skill_revision
-from optimize import promote as P
+from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision
+from ingot.optimize import promote as P
+from ingot.optimize import publication as Q
def _library(tmp_path, monkeypatch):
@@ -13,9 +14,9 @@ def _library(tmp_path, monkeypatch):
skill = root / "pdf"
skill.mkdir(parents=True)
(skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n")
+ monkeypatch.setenv("INGOT_LIBRARY", str(root))
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path))
monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root))
- monkeypatch.setattr(P, "PENDING_DIR", tmp_path / "pending")
- monkeypatch.setattr(P, "REVISIONS_DIR", tmp_path / "revisions")
return skill
@@ -36,6 +37,10 @@ def _pending(skill_dir, *, promotable=True):
}
+def _complete(skill="pdf"):
+ return P._activate_approved(skill, P.load_pending(skill))
+
+
def test_promote_refuses_blocked_evidence(tmp_path, monkeypatch):
skill = _library(tmp_path, monkeypatch)
P.save_pending("pdf", _pending(skill, promotable=False))
@@ -63,34 +68,189 @@ def test_promote_refuses_bundled_file_drift(tmp_path, monkeypatch):
P.approve_pending("pdf")
+def test_approval_queues_publication_without_activating(tmp_path, monkeypatch):
+ skill = _library(tmp_path, monkeypatch)
+ pending = _pending(skill)
+ P.save_pending("pdf", pending)
+
+ result = P.approve_pending("pdf", actor="admin")
+
+ assert result == "Approved 'pdf'; publishing to vault."
+ assert "old body" in (skill / "SKILL.md").read_text()
+ assert P.load_pending("pdf") == pending
+ publication = Q.publication_for_skill("pdf")
+ assert publication["state"] == "approved_publishing"
+ assert publication["actor"] == "admin"
+
+
def test_promote_snapshots_previous_revision_and_swaps_challenger(tmp_path, monkeypatch):
skill = _library(tmp_path, monkeypatch)
pending = _pending(skill)
old_revision = pending["evidence"]["champion"]["revision"]
P.save_pending("pdf", pending)
- result = P.approve_pending("pdf")
+ result = _complete()
assert "new body" in (skill / "SKILL.md").read_text()
- assert "old body" in (P.REVISIONS_DIR / "pdf" / old_revision / "SKILL.md").read_text()
+ assert "old body" in (P.revisions_dir() / "pdf" / old_revision / "SKILL.md").read_text()
assert not P.pending_path("pdf").exists()
assert old_revision in result
assert load_skills(skill.parent)[0].revision == pending["evidence"]["challenger"]["revision"]
audit = json.loads((tmp_path / "approval-audit.jsonl").read_text())
assert audit["action"] == "approve" and audit["skill"] == "pdf"
- result = P.rollback("pdf", old_revision)
+ result = P._activate_rollback("pdf", old_revision)
assert "old body" in (skill / "SKILL.md").read_text()
assert "Rolled back" in result
records = [json.loads(line) for line in (tmp_path / "approval-audit.jsonl").read_text().splitlines()]
assert [record["action"] for record in records] == ["approve", "rollback"]
+def test_rollback_queues_exact_snapshot_without_changing_served_skill(tmp_path, monkeypatch):
+ skill = _library(tmp_path, monkeypatch)
+ old_revision = load_skills(skill.parent)[0].revision
+ P._snapshot(skill, "pdf", old_revision)
+ (skill / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\ncurrent body\n"
+ )
+ current_revision = load_skills(skill.parent)[0].revision
+
+ result = P.rollback("pdf", old_revision, actor="admin")
+
+ assert result == f"Approved rollback of 'pdf' to {old_revision}; publishing to vault."
+ assert "current body" in (skill / "SKILL.md").read_text()
+ record = Q.publication_for_skill("pdf")
+ assert record["action"] == "rollback"
+ assert record["expected_champion"] == current_revision
+ assert record["candidate_revision"] == old_revision
+
+
+def test_absence_rollback_queues_without_removing_served_skill(tmp_path, monkeypatch):
+ skill = _library(tmp_path, monkeypatch)
+ P._snapshot_absence("pdf")
+ current_revision = load_skills(skill.parent)[0].revision
+
+ P.rollback("pdf", P.ABSENT_REVISION, actor="admin")
+
+ assert skill.is_dir()
+ record = Q.publication_for_skill("pdf")
+ assert record["expected_champion"] == current_revision
+ assert record["candidate_revision"] == P.ABSENT_REVISION
+ assert record["components"] == {}
+
+
+def test_promote_external_skill_through_writable_authoring_root(tmp_path, monkeypatch):
+ local = tmp_path / "local"
+ external = tmp_path / "external"
+ local.mkdir()
+ skill = external / "pdf"
+ skill.mkdir(parents=True)
+ skill_md = skill / "SKILL.md"
+ skill_md.write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n")
+ monkeypatch.setenv("INGOT_LIBRARY", str(local))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external))
+ pending = _pending(skill)
+ old_revision = pending["evidence"]["champion"]["revision"]
+ P.save_pending("pdf", pending)
+
+ skill_md.chmod(0o444)
+ external.chmod(0o555)
+ try:
+ with pytest.warns(UserWarning, match="duplicate skill 'pdf'"):
+ result = _complete()
+ finally:
+ skill_md.chmod(0o644)
+ external.chmod(0o755)
+
+ promoted = local / "pdf"
+ assert "Promoted 'pdf'" in result
+ assert "old body" in (skill / "SKILL.md").read_text()
+ assert "new body" in (promoted / "SKILL.md").read_text()
+ with pytest.warns(UserWarning, match="duplicate skill 'pdf'"):
+ assert load_skills()[0].root == str(promoted.resolve())
+ assert "old body" in (
+ P.revisions_dir() / "pdf" / old_revision / "SKILL.md").read_text()
+
+
+def test_external_promotion_refuses_target_created_after_precheck(tmp_path, monkeypatch):
+ local = tmp_path / "local"
+ external = tmp_path / "external"
+ local.mkdir()
+ source = external / "pdf"
+ source.mkdir(parents=True)
+ (source / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n")
+ target = local / "pdf"
+ monkeypatch.setenv("INGOT_LIBRARY", str(local))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external))
+ P.save_pending("pdf", _pending(source))
+ copytree = P.shutil.copytree
+
+ def race(source_path, destination, *args, **kwargs):
+ result = copytree(source_path, destination, *args, **kwargs)
+ if Path(destination).name.endswith(".stage"):
+ target.mkdir()
+ return result
+
+ monkeypatch.setattr(P.shutil, "copytree", race)
+
+ with pytest.raises(ValueError, match="activation target appeared during promotion"):
+ _complete()
+
+ assert target.is_dir() and list(target.iterdir()) == []
+ assert "old body" in (source / "SKILL.md").read_text()
+ assert P.pending_path("pdf").exists()
+
+
+def test_promote_refuses_unserved_authoring_root_collision(tmp_path, monkeypatch):
+ local = tmp_path / "local"
+ external = tmp_path / "external"
+ collision = local / "pdf"
+ collision.mkdir(parents=True)
+ (collision / "notes.txt").write_text("operator-owned")
+ skill = external / "pdf"
+ skill.mkdir(parents=True)
+ (skill / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n")
+ monkeypatch.setenv("INGOT_LIBRARY", str(local))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external))
+ P.save_pending("pdf", _pending(skill))
+
+ with pytest.raises(ValueError, match="writable activation target already exists"):
+ _complete()
+
+ assert (collision / "notes.txt").read_text() == "operator-owned"
+ assert "old body" in (skill / "SKILL.md").read_text()
+ assert P.pending_path("pdf").exists()
+
+
+def test_promote_refuses_dangling_authoring_root_symlink(tmp_path, monkeypatch):
+ local = tmp_path / "local"
+ external = tmp_path / "external"
+ local.mkdir()
+ collision = local / "pdf"
+ collision.symlink_to(local / "missing", target_is_directory=True)
+ skill = external / "pdf"
+ skill.mkdir(parents=True)
+ (skill / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n")
+ monkeypatch.setenv("INGOT_LIBRARY", str(local))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external))
+ P.save_pending("pdf", _pending(skill))
+
+ with pytest.raises(ValueError, match="writable activation target already exists"):
+ _complete()
+
+ assert collision.is_symlink()
+ assert "old body" in (skill / "SKILL.md").read_text()
+ assert P.pending_path("pdf").exists()
+
+
def test_approval_succeeds_when_audit_write_fails(tmp_path, monkeypatch, caplog):
skill = _library(tmp_path, monkeypatch)
pending = _pending(skill)
P.save_pending("pdf", pending)
monkeypatch.setattr(P, "_audit", lambda *args: (_ for _ in ()).throw(OSError("disk full")))
- result = P.approve_pending("pdf")
+ result = _complete()
assert "Promoted 'pdf'" in result
assert "new body" in (skill / "SKILL.md").read_text()
@@ -103,10 +263,10 @@ def test_rollback_succeeds_when_audit_write_fails(tmp_path, monkeypatch, caplog)
pending = _pending(skill)
old_revision = pending["evidence"]["champion"]["revision"]
P.save_pending("pdf", pending)
- P.approve_pending("pdf")
+ _complete()
monkeypatch.setattr(P, "_audit", lambda *args: (_ for _ in ()).throw(OSError("disk full")))
- result = P.rollback("pdf", old_revision)
+ result = P._activate_rollback("pdf", old_revision)
assert "Rolled back" in result
assert "old body" in (skill / "SKILL.md").read_text()
@@ -122,7 +282,7 @@ def fail_write(*args, **kwargs):
monkeypatch.setattr(P, "write_components", fail_write)
with pytest.raises(RuntimeError, match="stage failed"):
- P.approve_pending("pdf")
+ _complete()
assert "old body" in (skill / "SKILL.md").read_text()
@@ -133,7 +293,7 @@ def test_write_components_rejects_symlink_escape(tmp_path):
(skill / "SKILL.md").write_text("---\nname: skill\ndescription: d\n---\nbody\n")
(skill / "scripts").symlink_to(outside, target_is_directory=True)
with pytest.raises(ValueError, match="escapes skill root"):
- from mcp_server.registry import write_components
+ from ingot.mcp_server.registry import write_components
write_components(skill, {"description": "d", "body": "b", "file:scripts/pwn.py": "bad"})
@@ -150,7 +310,7 @@ def test_promotion_sweeps_a_leftover_staging_directory(tmp_path, monkeypatch):
assert load_skills(skill.parent)[0].body == "old body" # never shadowed the live skill
P.save_pending("pdf", _pending(skill))
- P.approve_pending("pdf")
+ _complete()
assert not stale.exists()
assert list(skill.parent.glob(".pdf.*")) == []
@@ -203,7 +363,7 @@ def test_failed_rollback_copy_leaves_no_partial_staging_directory(tmp_path, monk
skill = _library(tmp_path, monkeypatch)
P.save_pending("pdf", _pending(skill))
old_revision = P.load_pending("pdf")["evidence"]["champion"]["revision"]
- P.approve_pending("pdf")
+ _complete()
def fail_copy(src, dst, **kwargs):
Path(dst).mkdir(parents=True, exist_ok=True) # a partially copied tree
@@ -211,7 +371,7 @@ def fail_copy(src, dst, **kwargs):
monkeypatch.setattr(P.shutil, "copytree", fail_copy)
with pytest.raises(RuntimeError, match="copy failed"):
- P.rollback("pdf", old_revision)
+ P._activate_rollback("pdf", old_revision)
assert list(skill.parent.glob(".pdf.*")) == []
assert "new body" in (skill / "SKILL.md").read_text() # the live skill is untouched
@@ -227,9 +387,9 @@ def fail_copy(src, dst, **kwargs):
monkeypatch.setattr(P.shutil, "copytree", fail_copy)
with pytest.raises(RuntimeError, match="snapshot failed"):
- P.approve_pending("pdf")
+ _complete()
- assert list((P.REVISIONS_DIR / "pdf").glob("*")) == []
+ assert list((P.revisions_dir() / "pdf").glob("*")) == []
assert "old body" in (skill / "SKILL.md").read_text()
@@ -247,7 +407,7 @@ def _promote_body(skill, body):
"evidence": {"champion": {"revision": current.revision},
"challenger": {"revision": skill_revision(skill, challenger)}, "gate": gate},
})
- P.approve_pending("pdf")
+ _complete()
return current.revision
@@ -260,7 +420,7 @@ def test_rollback_then_promote_orders_the_restored_revision_first(tmp_path, monk
assert [r["revision"] for r in P.list_revisions("pdf")] == [second, first]
- P.rollback("pdf", first) # snapshot C (third body), live is back on A
+ P._activate_rollback("pdf", first) # snapshot C (third body), live is back on A
third = [r["revision"] for r in P.list_revisions("pdf")][0]
assert third not in (first, second)
@@ -275,7 +435,7 @@ def test_rollback_then_promote_orders_the_restored_revision_first(tmp_path, monk
def test_list_revisions_falls_back_to_mtime_without_an_index(tmp_path, monkeypatch):
"""Snapshots taken before the index existed still list, ordered below stamped ones."""
_library(tmp_path, monkeypatch)
- legacy = P.REVISIONS_DIR / "pdf" / ("a" * 8)
+ legacy = P.revisions_dir() / "pdf" / ("a" * 8)
legacy.mkdir(parents=True)
assert [r["revision"] for r in P.list_revisions("pdf")] == ["a" * 8]
assert P.list_revisions("pdf")[0]["created"] > 0
@@ -288,7 +448,7 @@ def _write_snapshot_index(text: str) -> None:
def _snapshot_dir(name: str) -> None:
- (P.REVISIONS_DIR / "pdf" / name).mkdir(parents=True, exist_ok=True)
+ (P.revisions_dir() / "pdf" / name).mkdir(parents=True, exist_ok=True)
@pytest.mark.parametrize("index", [
@@ -338,7 +498,7 @@ def test_promotion_survives_a_stamp_failure(tmp_path, monkeypatch, caplog, stamp
monkeypatch.setattr(P, "_stamp_snapshot",
lambda *a: (_ for _ in ()).throw(stamp_error))
- assert "Promoted 'pdf'" in P.approve_pending("pdf")
+ assert "Promoted 'pdf'" in _complete()
assert "new body" in (skill / "SKILL.md").read_text() # the directory swap still happened
assert "snapshot index write failed" in caplog.text
assert [r["revision"] for r in P.list_revisions("pdf")] # mtime fallback still lists it
@@ -348,7 +508,7 @@ def test_snapshot_index_is_not_restored_into_the_live_skill(tmp_path, monkeypatc
skill = _library(tmp_path, monkeypatch)
first = _promote_body(skill, "second body")
- P.rollback("pdf", first)
+ P._activate_rollback("pdf", first)
assert P.snapshot_index_path("pdf").exists()
assert not (skill / ".snapshots.json").exists()
diff --git a/tests/test_publication.py b/tests/test_publication.py
new file mode 100644
index 0000000..c90c4d3
--- /dev/null
+++ b/tests/test_publication.py
@@ -0,0 +1,122 @@
+import json
+import stat
+from pathlib import Path
+
+import pytest
+
+from ingot.optimize import publication as Q
+
+
+@pytest.fixture(autouse=True)
+def publication_store(tmp_path, monkeypatch):
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path))
+
+
+def _pending(*, candidate="b" * 64, body="new body", action="promote"):
+ return {
+ "skill": "copywriting",
+ "kind": "retrospective",
+ "champion_components": {"description": "Write copy.", "body": "old body"},
+ "challenger_components": (
+ {} if candidate == "absent" else {"description": "Write copy.", "body": body}
+ ),
+ "evidence": {
+ "champion": {"revision": "a" * 64},
+ "challenger": {"revision": candidate},
+ "gate": {"promotable": True, "blocked": []},
+ },
+ "retrospective": {"proposal_id": "retro-123"},
+ }
+
+
+def test_queue_is_inert_and_captures_exact_approval(tmp_path, monkeypatch):
+ library = tmp_path / "skills"
+ monkeypatch.setenv("INGOT_LIBRARY", str(library))
+ pending = _pending()
+
+ receipt = Q.queue_publication("copywriting", pending, "admin", "promote")
+
+ assert receipt.state == "approved_publishing"
+ assert not (library / "copywriting").exists()
+ stored = json.loads(receipt.path.read_text())
+ assert stored["proposal_id"] == "retro-123"
+ assert stored["actor"] == "admin"
+ assert stored["action"] == "promote"
+ assert stored["expected_champion"] == "a" * 64
+ assert stored["candidate_revision"] == "b" * 64
+ assert stored["components"] == pending["challenger_components"]
+ assert stored["attempts"] == 0 and stored["last_error"] == ""
+ assert stat.S_IMODE(receipt.path.stat().st_mode) == 0o600
+
+
+def test_exact_retry_is_idempotent():
+ first = Q.queue_publication("copywriting", _pending(), "admin", "promote")
+ second = Q.queue_publication("copywriting", _pending(), "admin", "promote")
+
+ assert second.id == first.id
+ assert list(Q.publications_dir().glob("*.json")) == [first.path]
+
+
+def test_different_candidate_cannot_occupy_same_skill_lane():
+ Q.queue_publication("copywriting", _pending(), "admin", "promote")
+
+ with pytest.raises(ValueError, match="publication is already in progress"):
+ Q.queue_publication(
+ "copywriting", _pending(candidate="c" * 64, body="other body"),
+ "admin", "promote",
+ )
+
+
+def test_final_record_is_absent_when_atomic_publication_fails(monkeypatch):
+ def fail_link(source, destination):
+ raise OSError("disk failure")
+
+ monkeypatch.setattr(Q.os, "link", fail_link)
+ with pytest.raises(OSError, match="disk failure"):
+ Q.queue_publication("copywriting", _pending(), "admin", "promote")
+
+ assert list(Q.publications_dir().glob("*.json")) == []
+
+
+def test_absence_rollback_can_queue_without_skill_components():
+ receipt = Q.queue_publication(
+ "copywriting", _pending(candidate="absent"), "admin", "rollback"
+ )
+
+ stored = json.loads(receipt.path.read_text())
+ assert stored["action"] == "rollback"
+ assert stored["candidate_revision"] == "absent"
+ assert stored["components"] == {}
+
+
+def test_the_newest_receipt_wins_within_one_second(tmp_path, monkeypatch):
+ """Two receipts for one skill are routinely queued in the same second — approve, then roll
+ back. Whole-second timestamps would leave the review surface showing whichever id sorted
+ higher."""
+ monkeypatch.setenv("INGOT_LIBRARY", str(tmp_path / "skills"))
+ first = Q.queue_publication("copywriting", _pending(), "admin", "promote")
+ Q.update_publication(first.id, state="active")
+ rollback = _pending(candidate="absent")
+ second = Q.queue_publication("copywriting", rollback, "admin", "rollback")
+
+ latest = Q.publication_for_skill("copywriting")
+
+ assert latest["id"] == second.id and latest["id"] != first.id
+ assert latest["created"] == json.loads(first.path.read_text())["created"] # the same second
+
+
+@pytest.mark.parametrize("malformed", [[], "", 0])
+def test_rollback_refuses_components_of_the_wrong_shape(malformed):
+ """Absence is expressed by an empty object, not by any falsy value a malformed record holds."""
+ pending = _pending(candidate="absent")
+ pending["challenger_components"] = malformed
+
+ with pytest.raises(ValueError, match="components must be an object"):
+ Q.queue_publication("copywriting", pending, "admin", "rollback")
+
+
+def test_promotion_still_requires_description_and_body():
+ with pytest.raises(ValueError, match="challenger components are required"):
+ Q.queue_publication(
+ "copywriting", _pending(candidate="absent"), "admin", "promote"
+ )
diff --git a/tests/test_publisher.py b/tests/test_publisher.py
new file mode 100644
index 0000000..f677009
--- /dev/null
+++ b/tests/test_publisher.py
@@ -0,0 +1,548 @@
+import json
+import os
+import subprocess
+from pathlib import Path
+
+import pytest
+
+from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision
+from ingot.optimize import promote as P
+from ingot.optimize import publication as Q
+from ingot.optimize import publisher as W
+
+
+def test_publisher_unit_uses_the_portable_managed_configuration():
+ root = Path(__file__).resolve().parents[1]
+ unit = (root / "ops/systemd/ingot-publisher.service").read_text()
+
+ assert "EnvironmentFile=%h/.config/ingot/publisher.env" in unit
+ assert "ExecStart=/usr/bin/python3 -m ingot.optimize.publisher --watch" in unit
+ assert "Slancha" not in unit
+
+
+def _git(path, *args, check=True):
+ return subprocess.run(["git", "-C", str(path), *args], check=check,
+ capture_output=True, text=True).stdout.strip()
+
+
+FORGE_REPOSITORY = "example/skills"
+
+
+def _open(vault, remote, **kwargs):
+ return W.VaultRepo.open(vault, remote="origin",
+ expected_remotes={str(Path(remote).resolve())}, **kwargs)
+
+
+def _publisher(vault, remote, github=None):
+ """A forge-backend publisher over a bare repository standing in for GitHub.
+
+ `expected_remotes` is the configured repository's resolved URL. A test vault's origin is a
+ filesystem path rather than a github.com URL, which is exactly the case the hardcoded pair of
+ literals could not express."""
+ backend = W.ForgeBackend(github if github is not None else FakeGitHub(),
+ repository=FORGE_REPOSITORY,
+ expected_remotes={str(Path(remote).resolve())})
+ return W.Publisher(vault, backend=backend)
+
+
+def _repositories(tmp_path, monkeypatch):
+ remote = tmp_path / "remote.git"
+ vault = tmp_path / "vault"
+ subprocess.run(["git", "init", "--bare", str(remote)], check=True, capture_output=True)
+ subprocess.run(["git", "init", "-b", "main", str(vault)], check=True, capture_output=True)
+ _git(vault, "config", "user.name", "Ingot Test")
+ _git(vault, "config", "user.email", "ingot@test.invalid")
+ skill = vault / "pdf"
+ skill.mkdir()
+ (skill / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n")
+ (vault / "registry.json").write_text(json.dumps({
+ "pdf": {"disposition": "keep", "reason": "test"}
+ }) + "\n")
+ scripts = vault / "scripts"
+ scripts.mkdir()
+ (scripts / "validate.py").write_text("print('valid')\n")
+ _git(vault, "add", ".")
+ _git(vault, "commit", "-m", "Initial vault")
+ _git(vault, "remote", "add", "origin", str(remote))
+ _git(vault, "push", "-u", "origin", "main")
+ subprocess.run(["git", "--git-dir", str(remote), "symbolic-ref", "HEAD", "refs/heads/main"],
+ check=True)
+ monkeypatch.setenv("INGOT_LIBRARY", str(vault))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(vault))
+ return remote, vault, skill
+
+
+def _queue(skill):
+ champion = optimizable_components(skill)
+ challenger = {**champion, "body": "new body"}
+ current = load_skills(skill.parent)[0]
+ pending = {
+ "skill": "pdf", "champion_components": champion,
+ "challenger_components": challenger, "gate": {"promotable": True, "blocked": []},
+ "evidence": {"champion": {"revision": current.revision},
+ "challenger": {"revision": skill_revision(skill, challenger)}},
+ }
+ P.save_pending("pdf", pending)
+ return Q.queue_publication("pdf", pending, "admin", "promote")
+
+
+class FakeGitHub:
+ def __init__(self):
+ self.merged = None
+
+ def create_or_find(self, branch, publication_id):
+ return 17
+
+ def enable_auto_merge(self, pr):
+ return None
+
+ def merged_commit(self, pr):
+ return self.merged
+
+
+def _admin(remote, tmp_path):
+ """A second clone standing in for whoever merges the vault pull request."""
+ admin = tmp_path / "admin"
+ if not admin.exists():
+ subprocess.run(["git", "clone", str(remote), str(admin)], check=True, capture_output=True)
+ _git(admin, "config", "user.name", "Vault Admin")
+ _git(admin, "config", "user.email", "admin@test.invalid")
+ return admin
+
+
+def _merge(remote, tmp_path, branch):
+ admin = _admin(remote, tmp_path)
+ _git(admin, "fetch", "origin")
+ _git(admin, "merge", "--ff-only", f"origin/{branch}")
+ _git(admin, "push", "origin", "main")
+ return _git(admin, "rev-parse", "HEAD")
+
+
+def _publish(vault, publication_id, remote, tmp_path):
+ """Drive one queued publication all the way through a merged vault commit."""
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+ assert publisher.process(publication_id) == "awaiting_merge"
+ github.merged = _merge(remote, tmp_path, Q.load_publication(publication_id)["branch"])
+ assert publisher.process(publication_id) == "active"
+
+
+def test_unmerged_publication_never_changes_served_skill(tmp_path, monkeypatch):
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ assert "old body" in (skill / "SKILL.md").read_text()
+ assert P.pending_path("pdf").exists()
+
+
+def test_merged_exact_revision_activates_and_consumes_pending(tmp_path, monkeypatch):
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ record = Q.load_publication(receipt.id)
+ github.merged = _merge(remote, tmp_path, record["branch"])
+
+ assert publisher.process(receipt.id) == "active"
+ assert "new body" in (skill / "SKILL.md").read_text()
+ assert not P.pending_path("pdf").exists()
+ assert skill_revision(skill) == record["candidate_revision"]
+ assert (P.revisions_dir() / "pdf" / record["expected_champion"]).is_dir()
+
+
+def test_push_failure_preserves_champion_and_pending(tmp_path, monkeypatch):
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ run = W._run
+
+ def reject_push(command, *, cwd):
+ if command[:2] == ["git", "push"]:
+ raise RuntimeError("git push: rejected by test remote")
+ return run(command, cwd=cwd)
+
+ monkeypatch.setattr(W, "_run", reject_push)
+
+ with pytest.raises(RuntimeError, match="git push"):
+ _publisher(vault, remote).process(receipt.id)
+
+ assert "old body" in (skill / "SKILL.md").read_text()
+ assert P.pending_path("pdf").exists()
+ assert Q.load_publication(receipt.id)["last_error"]
+
+
+def test_retry_reuses_the_recorded_branch_after_push_failure(tmp_path, monkeypatch):
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ run = W._run
+ failures = 1
+
+ def fail_once(command, *, cwd):
+ nonlocal failures
+ if command[:2] == ["git", "push"] and failures:
+ failures -= 1
+ raise RuntimeError("git push: transient failure")
+ return run(command, cwd=cwd)
+
+ monkeypatch.setattr(W, "_run", fail_once)
+ publisher = _publisher(vault, remote)
+ with pytest.raises(RuntimeError, match="transient failure"):
+ publisher.process(receipt.id)
+
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ record = Q.load_publication(receipt.id)
+ assert record["branch"] == f"ingot/{receipt.id}"
+
+
+def test_merged_commit_may_be_an_ancestor_of_a_newer_origin_main(tmp_path, monkeypatch):
+ """An unrelated vault commit landing after the approved merge must not wedge the publication."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"])
+ admin = _admin(remote, tmp_path)
+ (admin / "NOTES.md").write_text("an unrelated vault edit\n")
+ _git(admin, "add", "NOTES.md")
+ _git(admin, "commit", "-m", "Unrelated vault commit")
+ _git(admin, "push", "origin", "main")
+
+ assert publisher.process(receipt.id) == "active"
+ assert "new body" in (skill / "SKILL.md").read_text()
+ assert (vault / "NOTES.md").exists()
+
+
+def test_crash_after_fast_forward_finalizes_on_retry(tmp_path, monkeypatch):
+ """The receipt is written after the fast-forward, so a crash between them leaves the approved
+ revision already served. The retry must finalize rather than refuse the departed champion."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"])
+ _git(vault, "fetch", "origin", "main")
+ _git(vault, "merge", "--ff-only", "origin/main") # the fast-forward that survived
+
+ assert publisher.process(receipt.id) == "active"
+ assert "new body" in (skill / "SKILL.md").read_text()
+ assert not P.pending_path("pdf").exists()
+
+
+def test_candidate_mismatch_never_consumes_pending_or_activates(tmp_path, monkeypatch):
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"])
+ pending = P.load_pending("pdf")
+ pending["evidence"]["challenger"]["revision"] = "f" * 64
+ P.save_pending("pdf", pending)
+
+ with pytest.raises(RuntimeError, match="pending review no longer matches"):
+ publisher.process(receipt.id)
+
+ assert P.pending_path("pdf").exists()
+ assert Q.load_publication(receipt.id)["state"] == "awaiting_merge"
+ assert "old body" in (skill / "SKILL.md").read_text() # nothing was fast-forwarded either
+ assert not (P.revisions_dir() / "pdf").exists() # and nothing was snapshotted
+
+
+def test_a_closed_vault_pull_request_is_refused_rather_than_reopened(tmp_path, monkeypatch):
+ """Closing the vault pull request rejects the publication. Opening a second one for the same
+ branch would overrule the person who closed it."""
+ remote, vault, _ = _repositories(tmp_path, monkeypatch)
+ monkeypatch.setattr(W, "_run", lambda command, *, cwd: json.dumps([
+ {"number": 17, "state": "CLOSED"}]) if command[0] == "gh" else "")
+
+ with pytest.raises(ValueError, match="closed without merging"):
+ W.GitHub(vault, FORGE_REPOSITORY).create_or_find("ingot/abc123", "abc123")
+
+
+def test_publication_replaces_a_stale_removal_entry_in_the_registry(tmp_path, monkeypatch):
+ """A skill left marked for removal would land in the vault and then be dropped by the
+ projection: the entry has to follow what the publication actually did."""
+ remote, vault, _ = _repositories(tmp_path, monkeypatch)
+ (vault / "registry.json").write_text(json.dumps({
+ "pdf": {"disposition": "remove", "reason": "removed earlier"}}) + "\n")
+
+ W.Publisher._register(vault, "pdf", present=True)
+
+ assert json.loads((vault / "registry.json").read_text())["pdf"]["disposition"] == "keep"
+
+
+def test_a_leftover_worktree_never_carries_unrelated_edits_into_the_pull_request(tmp_path,
+ monkeypatch):
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ workspace = Q.publications_dir() / "worktrees" / receipt.id
+ workspace.mkdir(parents=True)
+ (workspace / "STRAY.md").write_text("left behind by a killed run\n")
+
+ publisher = _publisher(vault, remote)
+ assert publisher.process(receipt.id) == "awaiting_merge"
+
+ branch = Q.load_publication(receipt.id)["branch"]
+ listed = _git(vault, "ls-tree", "-r", "--name-only", f"refs/heads/{branch}")
+ assert "STRAY.md" not in listed
+ assert "pdf/SKILL.md" in listed
+
+
+def test_rollback_stays_inert_until_the_vault_merge_restores_it(tmp_path, monkeypatch):
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ champion = load_skills(vault)[0].revision
+ _publish(vault, _queue(skill).id, remote, tmp_path)
+ assert "new body" in (skill / "SKILL.md").read_text()
+
+ result = P.rollback("pdf", champion, actor="admin")
+ assert result.endswith("publishing to vault.")
+ receipt = Q.publication_for_skill("pdf")
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+
+ assert publisher.process(receipt["id"]) == "awaiting_merge"
+ assert "new body" in (skill / "SKILL.md").read_text() # inert until the merge lands
+
+ github.merged = _merge(remote, tmp_path, Q.load_publication(receipt["id"])["branch"])
+ assert publisher.process(receipt["id"]) == "active"
+ assert "old body" in (skill / "SKILL.md").read_text()
+ assert skill_revision(skill) == champion
+
+
+def test_absence_rollback_removes_the_skill_only_after_merge(tmp_path, monkeypatch):
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ P._snapshot_absence("pdf")
+
+ P.rollback("pdf", P.ABSENT_REVISION, actor="admin")
+ receipt = Q.publication_for_skill("pdf")
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+
+ assert publisher.process(receipt["id"]) == "awaiting_merge"
+ assert skill.is_dir()
+
+ github.merged = _merge(remote, tmp_path, Q.load_publication(receipt["id"])["branch"])
+ assert publisher.process(receipt["id"]) == "active"
+ assert not skill.exists()
+ assert "pdf" not in json.loads((vault / "registry.json").read_text())
+
+
+def test_absence_rollback_retries_after_the_branch_already_removed_the_skill(tmp_path, monkeypatch):
+ """The second attempt starts from `origin/`, where the skill is already gone. Staging
+ it by name is a fatal `git add` there, which wedged a real publication in the vault."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ P._snapshot_absence("pdf")
+ P.rollback("pdf", P.ABSENT_REVISION, actor="admin")
+ receipt = Q.publication_for_skill("pdf")
+
+ class FailsOnce(FakeGitHub):
+ calls = 0
+
+ def create_or_find(self, branch, publication_id):
+ FailsOnce.calls += 1
+ if FailsOnce.calls == 1:
+ raise RuntimeError("gh pr: transient failure after the push")
+ return 17
+
+ publisher = _publisher(vault, remote, FailsOnce())
+ with pytest.raises(RuntimeError, match="transient failure"):
+ publisher.process(receipt["id"])
+
+ assert publisher.process(receipt["id"]) == "awaiting_merge"
+ branch = Q.load_publication(receipt["id"])["branch"]
+ assert "pdf/SKILL.md" not in _git(vault, "ls-tree", "-r", "--name-only", f"refs/heads/{branch}")
+
+
+def test_a_vault_without_auto_merge_waits_for_a_human_instead_of_wedging(tmp_path, monkeypatch):
+ """`laulpogan/skills` has auto-merge disabled. Treating that as fatal stranded an approval that
+ was already sitting in an open pull request."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+
+ class NoAutoMerge(FakeGitHub):
+ def enable_auto_merge(self, pr):
+ raise RuntimeError("gh pr: GraphQL: Auto merge is not allowed for this repository")
+
+ assert _publisher(vault, remote, NoAutoMerge()).process(receipt.id) == "awaiting_merge"
+
+ record = Q.load_publication(receipt.id)
+ assert record["pr"] == 17 and record["auto_merge"] is False
+ assert "waiting on a human merge" in record["note"]
+ assert record["last_error"] == ""
+ assert "old body" in (skill / "SKILL.md").read_text()
+
+
+def test_rollback_restores_a_file_the_displaced_revision_added(tmp_path, monkeypatch):
+ """Components describe text the optimizer may rewrite, not the whole skill. Restoring the
+ snapshot tree is what makes a rollback exact when a revision added a file."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ champion = load_skills(vault)[0].revision
+ P._snapshot(skill, "pdf", champion)
+ (skill / "extra.md").write_text("added by a later revision\n")
+ _git(vault, "add", "pdf")
+ _git(vault, "commit", "-m", "Add a bundled file")
+ _git(vault, "push", "origin", "main")
+
+ P.rollback("pdf", champion, actor="admin")
+ receipt = Q.publication_for_skill("pdf")
+ _publish(vault, receipt["id"], remote, tmp_path)
+
+ assert not (skill / "extra.md").exists()
+ assert skill_revision(skill) == champion
+
+
+@pytest.mark.skipif(os.getuid() == 0, reason="root reads a mode-000 directory regardless")
+def test_an_unreadable_receipt_store_is_reported_not_read_as_empty(tmp_path):
+ """Path.glob swallows PermissionError, so a queue the publisher cannot list looks exactly like
+ one with nothing in it: approvals pile up in the console and the publisher says nothing. This
+ is the deployment failure where the console writes receipts as one user and the publisher runs
+ as another."""
+ store = tmp_path / "publications"
+ store.mkdir()
+ (store / "abc.json").write_text("{}")
+ store.chmod(0o000)
+ try:
+ assert list(store.glob("*.json")) == [] # indistinguishable from empty
+ blocked = W.unreadable_queue(store)
+ finally:
+ store.chmod(0o700)
+
+ assert blocked and "cannot read the receipt store" in blocked
+ assert W.unreadable_queue(tmp_path / "never-created") is None
+ store.chmod(0o700)
+ assert W.unreadable_queue(store) is None
+
+
+def test_activation_retires_the_publication_branch(tmp_path, monkeypatch):
+ """One branch per publication, kept forever, grows the vault's branch list without bound."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ branch = Q.load_publication(receipt.id)["branch"]
+ github.merged = _merge(remote, tmp_path, branch)
+
+ assert publisher.process(receipt.id) == "active"
+
+ assert branch not in _git(vault, "branch", "--list", branch)
+ assert not _git(vault, "ls-remote", "--heads", "origin", f"refs/heads/{branch}")
+ assert "new body" in (skill / "SKILL.md").read_text()
+
+
+def test_a_branch_that_cannot_be_retired_leaves_the_publication_active(tmp_path, monkeypatch):
+ """Cleanup runs after the receipt is durable, so a failure there must not turn a completed
+ activation back into a retry."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"])
+ run = W._run
+
+ def refuse_delete(command, *, cwd):
+ if "--delete" in command or command[:2] == ["git", "branch"]:
+ raise RuntimeError("git push: the remote refused the deletion")
+ return run(command, cwd=cwd)
+
+ monkeypatch.setattr(W, "_run", refuse_delete)
+
+ assert publisher.process(receipt.id) == "active"
+ assert Q.load_publication(receipt.id)["state"] == "active"
+
+
+def test_an_unrelated_vault_commit_does_not_block_the_next_publication(tmp_path, monkeypatch):
+ """The vault has other writers. The served library is a mirror of vault main, so falling
+ behind is normal — and refusing to publish until someone pulls by hand blocked the lane in
+ production the first time anybody else committed."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ admin = _admin(remote, tmp_path)
+ (admin / "UNRELATED.md").write_text("someone else's vault commit\n")
+ _git(admin, "add", "UNRELATED.md")
+ _git(admin, "commit", "-m", "An unrelated vault commit")
+ _git(admin, "push", "origin", "main")
+
+ with pytest.raises(ValueError, match="HEAD must equal origin/main"):
+ _open(vault, remote) # finalize still refuses to sync silently
+
+ assert _publisher(vault, remote).process(receipt.id) == "awaiting_merge"
+ assert (vault / "UNRELATED.md").exists() # the mirror caught up
+ assert "old body" in (skill / "SKILL.md").read_text()
+
+
+def test_a_diverged_vault_is_never_synced(tmp_path, monkeypatch):
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ (vault / "local.txt").write_text("a commit only this checkout has")
+ _git(vault, "add", "local.txt")
+ _git(vault, "commit", "-m", "Diverge")
+
+ with pytest.raises(RuntimeError, match="diverged"):
+ _publisher(vault, remote).process(receipt.id)
+
+
+@pytest.mark.parametrize("fault", ["dirty", "detached", "wrong_remote", "diverged"])
+def test_repository_guard_fails_before_publication_writes(tmp_path, monkeypatch, fault):
+ remote, vault, _ = _repositories(tmp_path, monkeypatch)
+ if fault == "dirty":
+ (vault / "dirty.txt").write_text("dirty")
+ elif fault == "detached":
+ _git(vault, "checkout", "--detach")
+ elif fault == "wrong_remote":
+ _git(vault, "remote", "set-url", "origin", str(tmp_path / "wrong.git"))
+ else:
+ (vault / "local.txt").write_text("local")
+ _git(vault, "add", "local.txt")
+ _git(vault, "commit", "-m", "Diverge")
+
+ with pytest.raises(ValueError):
+ _open(vault, remote)
+
+
+def test_polling_an_unmerged_publication_does_not_touch_the_vault(tmp_path, monkeypatch):
+ """`watch` polls every few seconds. Fetching and re-validating the checkout on each poll puts a
+ network round trip -- and a failure mode -- in front of a question `gh` already answers, and it
+ made an unrelated dirty working tree fail a receipt that was simply still waiting."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ publisher = _publisher(vault, remote)
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ (vault / "someone-is-editing.txt").write_text("an unrelated local edit")
+
+ run = W._run
+ touched = []
+
+ def record(command, *, cwd):
+ touched.append(command)
+ return run(command, cwd=cwd)
+
+ monkeypatch.setattr(W, "_run", record)
+
+ assert publisher.process(receipt.id) == "awaiting_merge"
+
+ assert not [command for command in touched if command[:2] == ["git", "fetch"]]
+ assert Q.load_publication(receipt.id)["last_error"] == ""
+
+
+def test_a_merge_that_just_landed_is_not_refused_as_missing(tmp_path, monkeypatch):
+ """The ancestry check runs against whatever `origin/main` this checkout last saw. Without a
+ fetch of its own it would refuse the one outcome the lane is waiting for."""
+ remote, vault, skill = _repositories(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ github = FakeGitHub()
+ publisher = _publisher(vault, remote, github)
+ assert publisher.process(receipt.id) == "awaiting_merge"
+ github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"])
+ assert not W.VaultRepo(vault, remote="origin").contains(github.merged) # not fetched yet
+
+ assert publisher.process(receipt.id) == "active"
+ assert "new body" in (skill / "SKILL.md").read_text()
diff --git a/tests/test_publisher_local.py b/tests/test_publisher_local.py
new file mode 100644
index 0000000..a7347cc
--- /dev/null
+++ b/tests/test_publisher_local.py
@@ -0,0 +1,570 @@
+"""The local publication backend: a Git vault on this machine, no network at any point.
+
+Every test here runs offline by construction — the vault has no remote to reach — and several of
+them assert that explicitly, because "air-gappable" is a claim the code has to keep rather than a
+property of how the test happened to be written.
+"""
+import json
+import subprocess
+from pathlib import Path
+
+import pytest
+
+from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision
+from ingot.optimize import promote as P
+from ingot.optimize import publication as Q
+from ingot.optimize import publisher as W
+from ingot import delivery as D
+
+
+def _git(path, *args):
+ return subprocess.run(["git", "-C", str(path), *args], check=True,
+ capture_output=True, text=True).stdout.strip()
+
+
+def _vault(tmp_path, monkeypatch):
+ """A vault with no remote at all. Nothing here has ever heard of GitHub."""
+ vault = tmp_path / "vault"
+ subprocess.run(["git", "init", "-b", "main", str(vault)], check=True, capture_output=True)
+ _git(vault, "config", "user.name", "Vault Owner")
+ _git(vault, "config", "user.email", "owner@test.invalid")
+ skill = vault / "pdf"
+ skill.mkdir()
+ (skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n")
+ (vault / "registry.json").write_text(
+ json.dumps({"pdf": {"disposition": "keep", "reason": "test"}}) + "\n")
+ (vault / "scripts").mkdir()
+ (vault / "scripts" / "validate.py").write_text("print('valid')\n")
+ _git(vault, "add", ".")
+ _git(vault, "commit", "-m", "Initial vault")
+ monkeypatch.setenv("INGOT_LIBRARY", str(vault))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(vault))
+ return vault, skill
+
+
+def _queue(skill, body="new body"):
+ champion = optimizable_components(skill)
+ challenger = {**champion, "body": body}
+ current = load_skills(skill.parent)[0]
+ pending = {
+ "skill": "pdf", "champion_components": champion,
+ "challenger_components": challenger, "gate": {"promotable": True, "blocked": []},
+ "evidence": {"champion": {"revision": current.revision},
+ "challenger": {"revision": skill_revision(skill, challenger)}},
+ }
+ P.save_pending("pdf", pending)
+ return Q.queue_publication("pdf", pending, "admin", "promote")
+
+
+def _publisher(vault):
+ return W.Publisher(vault, backend=W.LocalBackend())
+
+
+def _forbid_network(monkeypatch):
+ """Fail loudly on anything that would leave the machine."""
+ run = W._run
+
+ def offline(command, *, cwd):
+ if command[0] == "gh" or command[:2] == ["git", "push"] or "ls-remote" in command:
+ raise AssertionError(f"the local backend reached the network: {' '.join(command)}")
+ return run(command, cwd=cwd)
+
+ monkeypatch.setattr(W, "_run", offline)
+
+
+# --------------------------------------------------------------------------- the happy path
+
+def test_a_vault_with_no_remote_publishes_and_activates_in_one_pass(tmp_path, monkeypatch):
+ """There is no `awaiting_merge`: nothing external has to agree before the bytes are served."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ receipt = _queue(skill)
+ record = Q.load_publication(receipt.id)
+
+ assert _publisher(vault).process(receipt.id) == "active"
+
+ assert "new body" in (skill / "SKILL.md").read_text()
+ assert skill_revision(skill) == record["candidate_revision"]
+ assert not P.pending_path("pdf").exists()
+ assert (P.revisions_dir() / "pdf" / record["expected_champion"]).is_dir()
+ stored = Q.load_publication(receipt.id)
+ assert stored["state"] == "active"
+ assert stored["merged_commit"] == _git(vault, "rev-parse", "HEAD")
+
+
+def test_approval_alone_leaves_the_served_checkout_byte_identical(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ before = (skill / "SKILL.md").read_bytes()
+ head = _git(vault, "rev-parse", "HEAD")
+
+ _queue(skill)
+
+ assert (skill / "SKILL.md").read_bytes() == before
+ assert _git(vault, "rev-parse", "HEAD") == head
+
+
+def test_the_vault_commit_is_authored_by_the_publisher_not_the_host(tmp_path, monkeypatch):
+ """The commit records who published, not whoever happens to own the shell."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+
+ assert _publisher(vault).process(receipt.id) == "active"
+
+ assert _git(vault, "log", "-1", "--format=%an <%ae>") == "Ingot Publisher "
+ assert _git(vault, "log", "-1", "--format=%s") == f"Publish pdf via Ingot {receipt.id}"
+
+
+def test_activation_retires_the_publication_branch(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+
+ assert _publisher(vault).process(receipt.id) == "active"
+
+ assert _git(vault, "branch", "--list", f"ingot/{receipt.id}") == ""
+
+
+def test_republishing_a_revision_the_vault_already_serves_is_a_no_op(tmp_path, monkeypatch):
+ """An empty diff must converge on the existing tip. `git commit` with nothing staged exits
+ non-zero, so treating this as an error would fail a receipt that has nothing left to do."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ publisher = _publisher(vault)
+ assert publisher.process(receipt.id) == "active"
+ head = _git(vault, "rev-parse", "HEAD")
+
+ Q.update_publication(receipt.id, state="approved_publishing")
+ assert publisher.process(receipt.id) == "active"
+
+ assert _git(vault, "rev-parse", "HEAD") == head
+
+
+def test_an_absence_rollback_removes_the_skill_and_its_registry_entry(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ P._snapshot_absence("pdf")
+ P.rollback("pdf", P.ABSENT_REVISION, actor="admin")
+ receipt = Q.publication_for_skill("pdf")
+
+ assert _publisher(vault).process(receipt["id"]) == "active"
+
+ assert not skill.exists()
+ assert "pdf" not in json.loads((vault / "registry.json").read_text())
+
+
+def test_a_rollback_restores_the_stored_snapshot(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ champion = load_skills(vault)[0].revision
+ assert _publisher(vault).process(_queue(skill).id) == "active"
+ assert "new body" in (skill / "SKILL.md").read_text()
+
+ P.rollback("pdf", champion, actor="admin")
+ receipt = Q.publication_for_skill("pdf")
+
+ assert _publisher(vault).process(receipt["id"]) == "active"
+
+ assert "old body" in (skill / "SKILL.md").read_text()
+ assert skill_revision(skill) == champion
+
+
+# --------------------------------------------------------------------------- recovery
+
+def test_a_leftover_worktree_is_destroyed_rather_than_reused(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ workspace = Q.publications_dir() / "worktrees" / receipt.id
+ workspace.mkdir(parents=True)
+ (workspace / "STRAY.md").write_text("left behind by a killed run\n")
+
+ assert _publisher(vault).process(receipt.id) == "active"
+
+ assert not (vault / "STRAY.md").exists()
+ assert "new body" in (skill / "SKILL.md").read_text()
+
+
+class _AdvanceFails(W.LocalBackend):
+ """A crash after the branch is committed and before the served checkout moves."""
+
+ def advance(self, repo, record, commit):
+ raise RuntimeError("killed between the commit and the fast-forward")
+
+
+class _AdvanceThenCrash(W.LocalBackend):
+ """A crash after the served checkout moves and before the receipt is written."""
+
+ def advance(self, repo, record, commit):
+ super().advance(repo, record, commit)
+ raise RuntimeError("killed after the fast-forward")
+
+
+def test_a_crash_before_the_fast_forward_resumes_from_the_committed_branch(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ with pytest.raises(RuntimeError, match="killed between"):
+ W.Publisher(vault, backend=_AdvanceFails()).process(receipt.id)
+
+ record = Q.load_publication(receipt.id)
+ assert record["state"] == "publishing"
+ assert "old body" in (skill / "SKILL.md").read_text()
+ assert _git(vault, "rev-parse", f"refs/heads/ingot/{receipt.id}") == record["branch_commit"]
+
+ assert _publisher(vault).process(receipt.id) == "active"
+ assert "new body" in (skill / "SKILL.md").read_text()
+ assert not P.pending_path("pdf").exists()
+
+
+def test_a_crash_after_the_fast_forward_finalizes_instead_of_resnapshotting(tmp_path, monkeypatch):
+ """The served bytes already equal the candidate, so the champion this would snapshot is gone.
+ Re-snapshotting would refuse a publication that has in fact already activated."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ with pytest.raises(RuntimeError, match="killed after"):
+ W.Publisher(vault, backend=_AdvanceThenCrash()).process(receipt.id)
+
+ assert "new body" in (skill / "SKILL.md").read_text() # the fast-forward survived
+ assert Q.load_publication(receipt.id)["state"] == "publishing"
+ assert P.pending_path("pdf").exists() # but nothing was finalized
+
+ assert _publisher(vault).process(receipt.id) == "active"
+ assert not P.pending_path("pdf").exists()
+
+
+def test_an_active_receipt_is_never_reprocessed(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ publisher = _publisher(vault)
+ assert publisher.process(receipt.id) == "active"
+ head = _git(vault, "rev-parse", "HEAD")
+
+ assert publisher.process(receipt.id) == "active"
+ assert _git(vault, "rev-parse", "HEAD") == head
+
+
+# --------------------------------------------------------------------------- divergence
+
+def test_advance_refuses_a_branch_that_no_longer_fast_forwards(tmp_path, monkeypatch):
+ """Never a rebase, a merge commit, or a reset. The vault has other legitimate writers, and a
+ publisher that forces past one of them is no longer the only writer of what it serves."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ with pytest.raises(RuntimeError, match="killed between"):
+ W.Publisher(vault, backend=_AdvanceFails()).process(receipt.id)
+ branch = Q.load_publication(receipt.id)["branch"]
+ (vault / "NOTES.md").write_text("someone committed to the vault directly\n")
+ _git(vault, "add", "NOTES.md")
+ _git(vault, "commit", "-m", "A direct vault commit")
+ head = _git(vault, "rev-parse", "HEAD")
+
+ with pytest.raises(RuntimeError, match="git merge"):
+ W.LocalBackend().advance(W.VaultRepo(vault), {"id": receipt.id}, branch)
+
+ assert _git(vault, "rev-parse", "HEAD") == head # nothing moved
+ assert "old body" in (skill / "SKILL.md").read_text()
+
+
+def test_a_publication_branch_left_behind_by_a_direct_vault_commit_is_recut(tmp_path, monkeypatch):
+ """The receipt retries rather than wedging forever: the stale branch is abandoned, the
+ publication is re-materialized against the vault as it now stands, and the unrelated commit
+ survives."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ with pytest.raises(RuntimeError, match="killed between"):
+ W.Publisher(vault, backend=_AdvanceFails()).process(receipt.id)
+ (vault / "NOTES.md").write_text("someone committed to the vault directly\n")
+ _git(vault, "add", "NOTES.md")
+ _git(vault, "commit", "-m", "A direct vault commit")
+
+ assert _publisher(vault).process(receipt.id) == "active"
+
+ assert "new body" in (skill / "SKILL.md").read_text()
+ assert (vault / "NOTES.md").exists()
+
+
+def test_a_champion_changed_out_of_band_is_refused_not_overwritten(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ receipt = _queue(skill)
+ (skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\nedited\n")
+ _git(vault, "add", "pdf")
+ _git(vault, "commit", "-m", "An out-of-band edit to the champion")
+
+ with pytest.raises(RuntimeError, match="champion does not match"):
+ _publisher(vault).process(receipt.id)
+
+ assert "edited" in (skill / "SKILL.md").read_text()
+ assert P.pending_path("pdf").exists()
+
+
+# --------------------------------------------------------------------------- configuration
+
+def test_the_backend_defaults_to_local_and_is_never_inferred_from_a_remote(tmp_path):
+ """A vault that gains an `origin` must not silently start opening pull requests."""
+ config = W.load_config({"INGOT_VAULT_PATH": str(tmp_path)})
+ assert config.backend == "local"
+ assert isinstance(config.build().backend, W.LocalBackend)
+
+
+def test_a_missing_vault_path_is_an_error_not_the_demo_directory(tmp_path):
+ with pytest.raises(W.ConfigurationError, match="no vault configured"):
+ W.load_config({})
+
+
+def test_an_unknown_backend_is_refused_by_name(tmp_path):
+ with pytest.raises(W.ConfigurationError, match="unknown publication backend"):
+ W.load_config({"INGOT_VAULT_PATH": str(tmp_path), "INGOT_PUBLISH_BACKEND": "gitlab"})
+
+
+def test_the_forge_backend_requires_a_repository(tmp_path):
+ with pytest.raises(W.ConfigurationError, match="INGOT_FORGE_REPOSITORY"):
+ W.load_config({"INGOT_VAULT_PATH": str(tmp_path), "INGOT_PUBLISH_BACKEND": "forge"})
+
+
+def test_forge_settings_under_the_local_backend_warn_rather_than_look_active(tmp_path):
+ config = W.load_config({"INGOT_VAULT_PATH": str(tmp_path),
+ "INGOT_FORGE_REPOSITORY": "someone/skills"})
+ assert config.backend == "local"
+ assert any("inert" in warning for warning in config.warnings)
+
+
+def test_an_explicit_argument_outranks_the_environment(tmp_path):
+ config = W.load_config({"INGOT_VAULT_PATH": str(tmp_path / "env"),
+ "INGOT_PUBLISH_BACKEND": "forge",
+ "INGOT_FORGE_REPOSITORY": "someone/skills"},
+ backend="local", vault=tmp_path / "explicit")
+ assert config.backend == "local"
+ assert config.vault_dir == tmp_path / "explicit"
+
+
+def test_a_vault_without_a_validator_refuses_to_start(tmp_path, monkeypatch):
+ """Every publication runs it before committing, so a missing one is a configuration error and
+ not a silent skip."""
+ vault, _ = _vault(tmp_path, monkeypatch)
+ (vault / "scripts" / "validate.py").unlink()
+ _git(vault, "commit", "-am", "Remove the validator")
+
+ with pytest.raises(W.ConfigurationError, match="no validator"):
+ W.validate(W.load_config({"INGOT_VAULT_PATH": str(vault)}))
+
+
+@pytest.mark.parametrize("fault", ["dirty", "detached", "missing"])
+def test_startup_refuses_a_vault_the_publisher_must_not_build_on(tmp_path, monkeypatch, fault):
+ vault, _ = _vault(tmp_path, monkeypatch)
+ if fault == "dirty":
+ (vault / "dirty.txt").write_text("dirty")
+ elif fault == "detached":
+ _git(vault, "checkout", "--detach")
+ else:
+ vault = tmp_path / "nowhere"
+
+ with pytest.raises(W.ConfigurationError):
+ W.validate(W.load_config({"INGOT_VAULT_PATH": str(vault)}))
+
+
+def test_a_valid_local_vault_starts(tmp_path, monkeypatch):
+ vault, _ = _vault(tmp_path, monkeypatch)
+ publisher = W.validate(W.load_config({"INGOT_VAULT_PATH": str(vault)}))
+ assert publisher.vault_dir == Path(vault).resolve()
+ assert publisher.backend.name == "local"
+
+
+# ------------------------------------------------------------------ artifact fidelity, end to end
+
+PNG = b"\x89PNG\r\n\x1a\n" + bytes(range(256)) + b"\xff\xfe\xfd"
+
+
+def _ingested(tmp_path, monkeypatch, vault):
+ """A real `ingot add` of a package carrying bytes no text component could hold."""
+ from ingot import admission
+
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs"))
+ package = tmp_path / "src" / "csv-tidy"
+ (package / "assets").mkdir(parents=True)
+ (package / "SKILL.md").write_text(
+ "---\nname: csv-tidy\ndescription: Tidy CSV files.\n---\n\nUse this to tidy CSVs.\n",
+ encoding="utf-8")
+ (package / "assets" / "logo.png").write_bytes(PNG)
+ (package / "run.sh").write_text("#!/bin/sh\necho tidy\n", encoding="utf-8")
+ (package / "run.sh").chmod(0o755)
+ admission.add_package(package, actor="operator")
+ return package
+
+
+def test_an_ingested_binary_asset_reaches_the_vault_byte_for_byte(tmp_path, monkeypatch):
+ """The end of the chain the whole change exists for. Admission used to reduce a package to
+ decoded text, so this file was reviewed as part of the candidate, approved as part of the
+ candidate, and then was not in the vault -- with no error anywhere along the way."""
+ vault, _ = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ _ingested(tmp_path, monkeypatch, vault)
+ receipt = Q.queue_publication("csv-tidy", P.load_pending("csv-tidy"), "admin", "promote")
+
+ assert _publisher(vault).process(receipt.id) == "active"
+
+ published = vault / "csv-tidy" / "assets" / "logo.png"
+ assert published.read_bytes() == PNG
+ assert _git(vault, "status", "--porcelain") == ""
+ assert skill_revision(vault / "csv-tidy") == Q.load_publication(receipt.id)["candidate_revision"]
+
+
+def test_the_published_executable_bit_survives_the_vault_commit(tmp_path, monkeypatch):
+ vault, _ = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ _ingested(tmp_path, monkeypatch, vault)
+ receipt = Q.queue_publication("csv-tidy", P.load_pending("csv-tidy"), "admin", "promote")
+
+ _publisher(vault).process(receipt.id)
+
+ entry = _git(vault, "ls-files", "-s", "csv-tidy/run.sh")
+ assert entry.split()[0] == "100755"
+
+
+def test_a_staged_asset_altered_after_approval_stops_the_publication(tmp_path, monkeypatch):
+ """The receipt is the authority for what gets served. If the staged bytes moved between the
+ approval and the publication, the publisher must refuse rather than publish what it finds."""
+ from ingot.optimize import tree
+
+ vault, _ = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ _ingested(tmp_path, monkeypatch, vault)
+ pending = P.load_pending("csv-tidy")
+ receipt = Q.queue_publication("csv-tidy", pending, "admin", "promote")
+ staged = tree.staged_dir(pending["tree"]["digest"])
+ (staged / "assets" / "logo.png").write_bytes(b"substituted")
+
+ with pytest.raises(RuntimeError, match="does not match the receipt: assets/logo.png"):
+ _publisher(vault).process(receipt.id)
+
+ assert not (vault / "csv-tidy").exists()
+ assert _git(vault, "rev-parse", "--abbrev-ref", "HEAD") == "main"
+ assert "does not match the receipt" in Q.load_publication(receipt.id)["last_error"]
+
+
+# --------------------------------------------------------------------------- delivery targets
+
+def _delivering(vault, tmp_path, name="claude"):
+ """A publisher that serves the managed vault and one native filesystem root beside it."""
+ native = tmp_path / name
+ targets = D.parse_targets(f"{name}=filesystem:{native}", vault=vault)
+ return W.Publisher(vault, backend=W.LocalBackend(), targets=targets), native
+
+
+def test_an_approved_revision_reaches_every_target(tmp_path, monkeypatch):
+ """The whole point: the vault and a native skill root end up holding the same approved bytes,
+ from one approval, through one publisher."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ publisher, native = _delivering(vault, tmp_path)
+ receipt = _queue(skill)
+
+ assert publisher.process(receipt.id) == "active"
+
+ assert skill_revision(native / "pdf") == skill_revision(skill)
+ assert (native / "pdf" / "SKILL.md").read_bytes() == (skill / "SKILL.md").read_bytes()
+
+
+def test_each_target_is_recorded_on_the_receipt_separately(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ publisher, native = _delivering(vault, tmp_path)
+ publisher.process(_queue(skill).id)
+
+ delivered = Q.publication_for_skill("pdf")["delivery"]
+ assert delivered["vault"]["kind"] == D.MANAGED_MCP
+ assert delivered["claude"] == {"kind": D.FILESYSTEM, "root": str(native),
+ "state": "delivered", "revision": skill_revision(skill),
+ "at": delivered["claude"]["at"]}
+
+
+def test_only_the_altered_target_reports_drift(tmp_path, monkeypatch):
+ """Independent status. Editing one target must not make the other one look wrong."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ publisher, native = _delivering(vault, tmp_path)
+ publisher.process(_queue(skill).id)
+ released = skill_revision(skill)
+
+ (native / "pdf" / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\nedited\n")
+
+ observed = {target.name: D.observed(target, "pdf") for target in publisher.targets}
+ assert observed["vault"] == released
+ assert observed["claude"] != released
+
+
+def test_a_rollback_returns_every_target_to_the_prior_revision(tmp_path, monkeypatch):
+ """Rollback travels the ordinary publication queue, so delivery happens on the way through
+ rather than needing a second mechanism that could disagree with the first."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ publisher, native = _delivering(vault, tmp_path)
+ champion = load_skills(vault)[0].revision
+ publisher.process(_queue(skill).id)
+ assert skill_revision(native / "pdf") != champion
+
+ P.rollback("pdf", champion, actor="admin")
+ assert publisher.process(Q.publication_for_skill("pdf")["id"]) == "active"
+
+ assert skill_revision(skill) == champion
+ assert skill_revision(native / "pdf") == champion
+
+
+def test_a_rollback_to_absence_removes_the_skill_from_every_target(tmp_path, monkeypatch):
+ vault, skill = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ publisher, native = _delivering(vault, tmp_path)
+ publisher.process(_queue(skill).id)
+ assert (native / "pdf").is_dir()
+
+ P._snapshot_absence("pdf")
+ P.rollback("pdf", P.ABSENT_REVISION, actor="admin")
+ assert publisher.process(Q.publication_for_skill("pdf")["id"]) == "active"
+
+ assert not skill.exists()
+ assert not (native / "pdf").exists()
+
+
+def _refuse_to_install(target, skill, source, revision):
+ if target.kind == D.FILESYSTEM:
+ raise OSError("the target is unavailable")
+ return False
+
+
+def test_a_failed_delivery_does_not_leave_an_active_release(tmp_path, monkeypatch):
+ """The release is finished when every target holds it, not when the vault does. Marking it
+ active on a partial delivery would report a change as live in places it never reached."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ publisher, native = _delivering(vault, tmp_path)
+ receipt = _queue(skill)
+ monkeypatch.setattr(D, "install", _refuse_to_install)
+
+ with pytest.raises(RuntimeError, match="the target is unavailable"):
+ publisher.process(receipt.id)
+
+ record = Q.load_publication(receipt.id)
+ assert record["state"] != "active"
+ assert "the target is unavailable" in record["last_error"]
+ assert record["delivery"]["claude"]["state"] == "failed"
+ assert not (native / "pdf").exists()
+
+
+def test_a_retried_publication_finishes_the_delivery_it_could_not_complete(tmp_path, monkeypatch):
+ """The vault has already advanced by then, so the retry must deliver rather than decide the
+ release is finished because the vault looks right."""
+ vault, skill = _vault(tmp_path, monkeypatch)
+ _forbid_network(monkeypatch)
+ publisher, native = _delivering(vault, tmp_path)
+ receipt = _queue(skill)
+ real_install, refused = D.install, []
+
+ def fail_the_first_delivery(target, name, source, revision):
+ if target.kind == D.FILESYSTEM and not refused:
+ refused.append(target.name)
+ raise OSError("the target is unavailable")
+ return real_install(target, name, source, revision)
+
+ monkeypatch.setattr(D, "install", fail_the_first_delivery)
+ with pytest.raises(RuntimeError):
+ publisher.process(receipt.id)
+ assert skill_revision(skill) == Q.load_publication(receipt.id)["candidate_revision"]
+
+ assert publisher.process(receipt.id) == "active"
+ assert skill_revision(native / "pdf") == skill_revision(skill)
+ assert Q.load_publication(receipt.id)["delivery"]["claude"]["state"] == "delivered"
diff --git a/tests/test_records.py b/tests/test_records.py
new file mode 100644
index 0000000..9aa1fd9
--- /dev/null
+++ b/tests/test_records.py
@@ -0,0 +1,251 @@
+"""Candidate manifests and release receipts.
+
+Two versioned records with one job each: a candidate names exactly what is being proposed and where
+it came from; a release names exactly what was published and proves it happened. Both are consumed
+by later PRs -- the candidate by `ingot add`, the receipt by the publisher -- so the shape is fixed
+here and tested here."""
+import json
+
+import pytest
+
+from ingot import records
+
+REVISION = "a" * 64
+OTHER_REVISION = "b" * 64
+
+
+def _review(errors=(), warnings=()):
+ return {"schema_version": "ingot/review/v1",
+ "valid": not errors,
+ "errors": list(errors),
+ "warnings": list(warnings)}
+
+
+def _candidate(**overrides):
+ fields = {"kind": "creation",
+ "skill": "pdf",
+ "source_type": "file",
+ "locator": "./packages/pdf",
+ "resolved_revision": REVISION,
+ "candidate_revision": REVISION,
+ "review": _review(),
+ "created_at": 1_770_000_000}
+ fields.update(overrides)
+ return records.candidate_manifest(**fields)
+
+
+def _receipt(**overrides):
+ fields = {"skill": "pdf",
+ "action": "promote",
+ "proposal_id": "p-1",
+ "publication_id": "pub-1",
+ "expected_champion": OTHER_REVISION,
+ "candidate_revision": REVISION,
+ "evidence_digests": [REVISION],
+ "actor": "operator",
+ "publisher": "local",
+ "target": "managed-library",
+ "published_at": 1_770_000_100,
+ "result": "published"}
+ fields.update(overrides)
+ return records.release_receipt(**fields)
+
+
+# --- digests --------------------------------------------------------------------------------
+
+def test_digest_ignores_key_order():
+ assert records.digest({"a": 1, "b": 2}) == records.digest({"b": 2, "a": 1})
+
+
+def test_digest_changes_with_content():
+ assert records.digest({"a": 1}) != records.digest({"a": 2})
+
+
+def test_digest_is_a_sha256_hex():
+ assert len(records.digest({"a": 1})) == 64
+
+
+# --- candidate manifests --------------------------------------------------------------------
+
+def test_candidate_manifest_carries_its_schema_version():
+ assert _candidate()["schema_version"] == records.CANDIDATE_SCHEMA
+
+
+def test_candidate_manifest_records_the_source_it_resolved():
+ manifest = _candidate()
+
+ assert manifest["source"] == {"type": "file",
+ "locator": "./packages/pdf",
+ "resolved_revision": REVISION}
+
+
+def test_candidate_manifest_references_the_review_by_digest():
+ """Included *and* digested: the report travels with the proposal, and the digest is what binds
+ it, so a report edited after the fact stops matching."""
+ review = _review(warnings=["file-reference-missing"])
+ manifest = _candidate(review=review)
+
+ assert manifest["review"]["digest"] == records.digest(review)
+ assert manifest["review"]["warnings"] == ["file-reference-missing"]
+
+
+def test_candidate_identity_ignores_the_timestamp():
+ """Deterministic apart from timestamps and actor metadata: proposing the same bytes twice must
+ produce the same identity, or idempotent submission is impossible."""
+ first = _candidate(created_at=1_770_000_000)
+ second = _candidate(created_at=1_999_999_999)
+
+ assert records.candidate_identity(first) == records.candidate_identity(second)
+
+
+def test_candidate_identity_ignores_the_local_path():
+ """A local path is operator context. The same package submitted from two checkouts is the same
+ candidate."""
+ first = _candidate(locator="/home/a/pdf")
+ second = _candidate(locator="/home/b/pdf")
+
+ assert records.candidate_identity(first) == records.candidate_identity(second)
+
+
+def test_file_candidate_identity_keeps_the_bound_review_report():
+ """Break caught: changing identity for already-quarantined file candidates during upgrade."""
+ first = _candidate(review={**_review(), "report_digest": REVISION})
+ second = _candidate(review={**_review(), "report_digest": OTHER_REVISION})
+
+ assert records.candidate_identity(first) != records.candidate_identity(second)
+
+
+def test_candidate_identity_changes_with_the_candidate_revision():
+ assert records.candidate_identity(_candidate()) != \
+ records.candidate_identity(_candidate(candidate_revision=OTHER_REVISION))
+
+
+def test_candidate_identity_changes_with_the_skill():
+ assert records.candidate_identity(_candidate()) != \
+ records.candidate_identity(_candidate(skill="docx"))
+
+
+def test_candidate_identity_changes_with_the_review_outcome():
+ """Evidence is revision-bound, and a review is evidence. The same bytes reviewed clean and
+ reviewed with errors are not interchangeable proposals."""
+ assert records.candidate_identity(_candidate()) != \
+ records.candidate_identity(_candidate(review=_review(errors=["description-empty"])))
+
+
+def test_a_well_formed_candidate_validates():
+ assert records.validate_candidate(_candidate()) == []
+
+
+def test_a_candidate_with_the_wrong_schema_is_rejected():
+ manifest = _candidate()
+ manifest["schema_version"] = "ingot/candidate/v99"
+
+ assert any("schema" in problem for problem in records.validate_candidate(manifest))
+
+
+def test_a_candidate_missing_a_field_is_rejected():
+ manifest = _candidate()
+ del manifest["candidate_revision"]
+
+ assert any("candidate_revision" in problem for problem in records.validate_candidate(manifest))
+
+
+def test_a_candidate_with_a_semantic_version_source_is_rejected():
+ """The resolved source revision must be content-based. A tag can be moved; a digest cannot."""
+ problems = records.validate_candidate(_candidate(resolved_revision="v1.2.3"))
+
+ assert any("resolved_revision" in problem for problem in problems)
+
+
+def test_a_candidate_with_an_unknown_kind_is_rejected():
+ assert any("kind" in problem for problem in records.validate_candidate(_candidate(kind="mutate")))
+
+
+def test_a_candidate_with_an_invalid_skill_slug_is_rejected():
+ assert any("skill" in problem for problem in records.validate_candidate(_candidate(skill="Not_A_Slug")))
+
+
+# --- release receipts -----------------------------------------------------------------------
+
+def test_release_receipt_carries_its_schema_version():
+ assert _receipt()["schema_version"] == records.RELEASE_SCHEMA
+
+
+def test_a_well_formed_receipt_validates():
+ assert records.validate_release(_receipt()) == []
+
+
+def test_absence_is_a_valid_revision_on_both_sides():
+ """A creation displaces nothing and a rollback can restore nothing. Absence is a revision, and
+ the existing publisher already treats it as one."""
+ assert records.validate_release(_receipt(expected_champion=records.ABSENT_REVISION)) == []
+ assert records.validate_release(_receipt(action="rollback",
+ candidate_revision=records.ABSENT_REVISION)) == []
+
+
+def test_a_rollback_produces_a_receipt():
+ assert records.validate_release(_receipt(action="rollback")) == []
+
+
+def test_an_unknown_action_is_rejected():
+ assert any("action" in problem for problem in records.validate_release(_receipt(action="delete")))
+
+
+def test_a_failed_receipt_stays_inspectable():
+ """A failure that erased its own reason would leave an operator with a stalled lane and no
+ way to learn why."""
+ receipt = records.release_receipt(
+ skill="pdf", action="promote", proposal_id="p-1", publication_id="pub-1",
+ expected_champion=OTHER_REVISION, candidate_revision=REVISION, evidence_digests=[],
+ actor="operator", publisher="local", target="managed-library",
+ published_at=1_770_000_100, result="failed", error="vault champion did not match")
+
+ assert records.validate_release(receipt) == []
+ assert receipt["result"] == "failed"
+ assert receipt["error"] == "vault champion did not match"
+
+
+def test_a_published_receipt_may_not_carry_an_error():
+ receipt = _receipt()
+ receipt["error"] = "something went wrong"
+
+ assert any("error" in problem for problem in records.validate_release(receipt))
+
+
+def test_a_failed_receipt_must_say_why():
+ receipt = _receipt(result="failed")
+
+ assert any("error" in problem for problem in records.validate_release(receipt))
+
+
+def test_an_unknown_result_is_rejected():
+ assert any("result" in problem for problem in records.validate_release(_receipt(result="queued")))
+
+
+def test_a_receipt_is_not_signed_and_claims_nothing_about_tampering():
+ """No signature field, on purpose. A local record a machine administrator can rewrite must not
+ carry anything that looks like proof it was not."""
+ receipt = _receipt()
+
+ assert "signature" not in receipt
+ assert "attestation" not in receipt
+
+
+# --- round trips ----------------------------------------------------------------------------
+
+def test_both_records_survive_a_json_round_trip():
+ for record in (_candidate(), _receipt()):
+ assert json.loads(json.dumps(record)) == record
+
+
+def test_records_import_without_the_heavy_stack():
+ import subprocess
+ import sys
+
+ heavy = ["fastapi", "onnxruntime", "langgraph", "langfuse", "litellm", "fastembed"]
+ program = f"import sys, ingot.records; print([m for m in {heavy!r} if m in sys.modules])"
+
+ result = subprocess.run([sys.executable, "-c", program], capture_output=True, text=True)
+
+ assert result.returncode == 0, result.stderr
+ assert result.stdout.strip() == "[]"
diff --git a/tests/test_registry.py b/tests/test_registry.py
index 94f4fba..3d17bab 100644
--- a/tests/test_registry.py
+++ b/tests/test_registry.py
@@ -1,10 +1,12 @@
-"""Unit tests for skill discovery / frontmatter parsing edge cases (mcp_server.registry)."""
+"""Unit tests for skill discovery / frontmatter parsing edge cases (ingot.mcp_server.registry)."""
import os
+from pathlib import Path
import pytest
-from mcp_server.registry import (
- configured_roots, load_skills, optimizable_components, parse_skill, write_skill_md,
+from ingot.mcp_server.registry import (
+ configured_roots, load_skills, optimizable_components, parse_skill, writable_skill_dir,
+ write_skill_md,
)
@@ -73,7 +75,7 @@ def test_configured_roots_reads_platform_path_separator(tmp_path, monkeypatch):
a, b, local = tmp_path / "a", tmp_path / "b", tmp_path / "local"
a.mkdir(); b.mkdir(); local.mkdir()
monkeypatch.setenv("SKILL_ROUTER_PATHS", os.pathsep.join([str(a), str(b), str(a)]))
- monkeypatch.setattr("mcp_server.registry.SKILLS_DIR", local)
+ monkeypatch.setenv("INGOT_LIBRARY", str(local))
assert configured_roots() == [local.resolve(), a.resolve(), b.resolve()]
@@ -81,7 +83,7 @@ def test_explicit_roots_override_environment(tmp_path, monkeypatch):
env_root, explicit, local = tmp_path / "env", tmp_path / "explicit", tmp_path / "local"
env_root.mkdir(); explicit.mkdir(); local.mkdir()
monkeypatch.setenv("SKILL_ROUTER_PATHS", str(env_root))
- monkeypatch.setattr("mcp_server.registry.SKILLS_DIR", local)
+ monkeypatch.setenv("INGOT_LIBRARY", str(local))
assert configured_roots([explicit]) == [local.resolve(), explicit.resolve()]
@@ -89,10 +91,17 @@ def test_environment_roots_keep_local_authoring_root(tmp_path, monkeypatch):
external, local = tmp_path / "external", tmp_path / "local"
external.mkdir(); local.mkdir()
monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external))
- monkeypatch.setattr("mcp_server.registry.SKILLS_DIR", local)
+ monkeypatch.setenv("INGOT_LIBRARY", str(local))
assert configured_roots() == [local.resolve(), external.resolve()]
+def test_writable_skill_dir_expands_user_root(tmp_path, monkeypatch):
+ monkeypatch.setenv("HOME", str(tmp_path))
+ monkeypatch.setenv("INGOT_LIBRARY", str(Path("~/skills")))
+
+ assert writable_skill_dir("sample") == tmp_path / "skills" / "sample"
+
+
def test_load_skills_uses_declared_root_precedence_with_warning(tmp_path):
a, b = tmp_path / "a", tmp_path / "b"
_skill(a, body="first"); _skill(b, dirname="other", name="sample", body="second")
@@ -192,7 +201,7 @@ def test_leftover_staging_directory_is_not_published_as_its_own_skill(tmp_path):
@pytest.mark.parametrize("suffix", ["stage", "previous", "rollback"])
def test_skill_sources_skips_every_staging_suffix(tmp_path, suffix):
- from mcp_server.registry import skill_sources
+ from ingot.mcp_server.registry import skill_sources
_live_skill(tmp_path)
_hidden_stage(tmp_path, "pdf", "abandoned body", suffix=suffix)
assert [p.parent.name for p in skill_sources(tmp_path)] == ["pdf"]
diff --git a/tests/test_retrospective.py b/tests/test_retrospective.py
new file mode 100644
index 0000000..6b2b4e2
--- /dev/null
+++ b/tests/test_retrospective.py
@@ -0,0 +1,165 @@
+import json
+
+import pytest
+
+from ingot.mcp_server.registry import load_skills
+from ingot.optimize import promote as P
+from ingot.optimize import retrospective as R
+
+
+def _library(tmp_path, monkeypatch):
+ root = tmp_path / "skills"
+ skill = root / "pdf"
+ skill.mkdir(parents=True)
+ (skill / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n")
+ monkeypatch.setenv("INGOT_LIBRARY", str(root))
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs"))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root))
+ return root, skill
+
+
+def _proposal(root, **overrides):
+ current = load_skills(root)[0]
+ data = {
+ "skill": "pdf",
+ "champion_revision": current.revision,
+ "challenger_body": "new body with a reusable guard",
+ "challenger_description": "",
+ "summary": "Preserve the recovery check after repeated omissions.",
+ "trigger": "Use when the same recovery step fails twice.",
+ "minimal_content": "Require the recovery check before completion.",
+ "producer": "skill-retrospective",
+ "caller": "build-loop after repeat evidence",
+ "evidence": ["Run A skipped the check.", "Run B repeated the same omission."],
+ "pressure_scenario": "A rushed repair reaches completion without checking recovery.",
+ "risk": "The extra gate may slow low-risk repairs.",
+ "verification_status": "passed",
+ "verification_command": "pytest tests/scenarios.md -k recovery",
+ "verification_result": "pressure scenario passed",
+ }
+ data.update(overrides)
+ return data
+
+
+def test_submit_quarantines_a_revision_bound_retrospective(tmp_path, monkeypatch):
+ root, _ = _library(tmp_path, monkeypatch)
+
+ result = R.submit_skill_update(**_proposal(root))
+
+ pending = P.load_pending("pdf")
+ assert result == {
+ "status": "quarantined", "skill": "pdf",
+ "proposal_id": pending["retrospective"]["proposal_id"], "promotable": True}
+ assert pending["kind"] == "retrospective"
+ assert pending["challenger_components"]["description"] == "Merge PDFs."
+ assert pending["challenger_components"]["body"] == "new body with a reusable guard"
+ assert pending["changed_components"] == ["body"]
+ assert pending["gate"] == {
+ "promotable": True,
+ "blocked": [],
+ "warnings": ["Retrospective evidence only; no held-out A/B comparison was run."],
+ "kind": "retrospective_admission",
+ "admission": {"pressure_verification": "passed", "evidence_items": 2},
+ }
+ assert pending["evidence"]["champion"]["revision"] == _proposal(root)["champion_revision"]
+ assert pending["evidence"]["challenger"]["revision"]
+ assert pending["retrospective"]["evidence"] == [
+ "Run A skipped the check.", "Run B repeated the same omission."]
+ paths = pending["evidence_paths"]
+ assert (tmp_path / paths["json"]).exists()
+ markdown = (tmp_path / paths["markdown"]).read_text()
+ assert "Retrospective proposal: pdf" in markdown
+ assert "pressure scenario passed" in markdown
+ audit = json.loads((tmp_path / "runs" / "retrospective-audit.jsonl").read_text())
+ assert audit["action"] == "quarantine" and audit["proposal_id"] == result["proposal_id"]
+
+
+def test_metadata_retry_is_idempotent_and_different_pending_is_preserved(tmp_path, monkeypatch):
+ root, _ = _library(tmp_path, monkeypatch)
+ proposal = _proposal(root)
+ now = [100]
+ monkeypatch.setattr(R.time, "time", lambda: now[0])
+ first = R.submit_skill_update(**proposal)
+ before = P.pending_path("pdf").read_text()
+ now[0] = 101
+
+ duplicate = R.submit_skill_update(**{**proposal, "caller": "same retry from a new session"})
+
+ assert duplicate == {**first, "status": "duplicate"}
+ assert P.pending_path("pdf").read_text() == before
+
+ with pytest.raises(ValueError, match="review slot is occupied"):
+ R.submit_skill_update(**{**proposal, "challenger_body": "different candidate"})
+ assert P.pending_path("pdf").read_text() == before
+
+
+def test_concurrent_writer_cannot_be_displaced_between_check_and_publish(tmp_path, monkeypatch):
+ root, _ = _library(tmp_path, monkeypatch)
+ write_evidence = R._write_evidence
+
+ def collide(skill, proposal, gate):
+ paths = write_evidence(skill, proposal, gate)
+ P.save_pending("pdf", {"skill": "pdf", "kind": "quality", "created": 7})
+ return paths
+
+ monkeypatch.setattr(R, "_write_evidence", collide)
+
+ with pytest.raises(ValueError, match="review slot is occupied"):
+ R.submit_skill_update(**_proposal(root))
+
+ assert P.load_pending("pdf") == {"skill": "pdf", "kind": "quality", "created": 7}
+ assert not list(R.evidence_dir().rglob("retrospective-*"))
+
+
+@pytest.mark.parametrize(("change", "message"), [
+ ({"champion_revision": "stale"}, "champion revision"),
+ ({"skill": "missing"}, "no indexed skill"),
+ ({"evidence": []}, "evidence"),
+ ({"evidence": ["only one occurrence"]}, "at least two"),
+ ({"evidence": ["same occurrence", "same occurrence"]}, "must be distinct"),
+ ({"verification_status": "maybe"}, "verification_status must be passed"),
+ ({"verification_status": "failed"}, "verification_status must be passed"),
+ ({"verification_status": "unavailable"}, "verification_status must be passed"),
+ ({"challenger_body": "x" * 200_001}, "challenger_body"),
+])
+def test_invalid_proposals_fail_before_mutation(tmp_path, monkeypatch, change, message):
+ root, _ = _library(tmp_path, monkeypatch)
+ with pytest.raises(ValueError, match=message):
+ R.submit_skill_update(**_proposal(root, **change))
+ assert not P.pending_dir().exists()
+ assert not R.evidence_dir().exists()
+
+
+def test_atomic_publication_failure_cleans_evidence(tmp_path, monkeypatch):
+ root, _ = _library(tmp_path, monkeypatch)
+
+ def unsupported_link(_source, _destination):
+ raise OSError("hard links unavailable")
+
+ monkeypatch.setattr(R.os, "link", unsupported_link)
+ with pytest.raises(RuntimeError, match="cannot atomically publish"):
+ R.submit_skill_update(**_proposal(root))
+
+ assert not P.pending_path("pdf").exists()
+ assert not list(R.evidence_dir().rglob("retrospective-*"))
+ assert not R.audit_file().exists()
+
+
+def test_mcp_producer_reaches_existing_approval_and_rollback_path(tmp_path, monkeypatch):
+ root, skill = _library(tmp_path, monkeypatch)
+ from ingot.mcp_server import server
+ server.STATE.reload([root])
+
+ result = server.propose_skill_update(**_proposal(root))
+ old_revision = _proposal(root)["champion_revision"]
+ promoted = P._activate_approved("pdf", P.load_pending("pdf"), actor="retrospective-test")
+
+ assert result["status"] == "quarantined"
+ assert "Promoted 'pdf'" in promoted
+ assert "new body with a reusable guard" in (skill / "SKILL.md").read_text()
+ assert "old body" in (
+ P.revisions_dir() / "pdf" / old_revision / "SKILL.md").read_text()
+
+ P._activate_rollback("pdf", old_revision, actor="retrospective-test")
+ assert "old body" in (skill / "SKILL.md").read_text()
diff --git a/tests/test_review.py b/tests/test_review.py
new file mode 100644
index 0000000..aad105c
--- /dev/null
+++ b/tests/test_review.py
@@ -0,0 +1,184 @@
+"""Unit tests for the standalone per-skill review (LLM + judge mocked)."""
+import json
+
+import pytest
+
+from ingot.optimize import review as R
+
+
+def _result(task, checklist, spec):
+ return {"task": task, "score": 0.0, "answer": "a", "feedback": "f",
+ "checklist": checklist, "spec": spec}
+
+
+SPEC = {
+ "cites_source": {"id": "cites_source", "criterion": "Names its source.", "weight": 5,
+ "dimension": "correctness"},
+ "is_terse": {"id": "is_terse", "criterion": "No padding.", "weight": 1,
+ "dimension": "efficiency"},
+}
+
+
+def test_findings_rank_by_cost_not_by_raw_score():
+ """A heavy check scraping a partial outranks a trivial check failing outright. Sorting on the
+ verdict value alone puts the weight-1 failure first and buries the thing worth fixing."""
+ results = [_result("t", {"cites_source": {"value": 0.5, "note": "no source"},
+ "is_terse": {"value": 0.0, "note": "padded"}}, SPEC)]
+ ranked = R.findings(results)
+ assert [f["check"] for f in ranked] == ["cites_source", "is_terse"]
+ assert ranked[0]["cost"] == pytest.approx(2.5) and ranked[1]["cost"] == pytest.approx(1.0)
+
+
+def test_findings_omit_clean_passes():
+ results = [_result("t", {"cites_source": {"value": 1.0, "note": ""},
+ "is_terse": {"value": 0.0, "note": "padded"}}, SPEC)]
+ assert [f["check"] for f in R.findings(results)] == ["is_terse"]
+
+
+def test_findings_default_weight_when_a_task_declared_no_spec():
+ """A task with no checklist is graded on the default four, whose specs are not in `spec`.
+ Those failures still have to appear, at weight 1, rather than vanish."""
+ results = [_result("t", {"correctness": {"value": 0.0, "note": "wrong"}}, {})]
+ found = R.findings(results)
+ assert len(found) == 1 and found[0]["weight"] == 1 and found[0]["cost"] == pytest.approx(1.0)
+
+
+def test_by_dimension_concentrates_losses():
+ results = [_result("t", {"cites_source": {"value": 0.0, "note": "n"},
+ "is_terse": {"value": 0.5, "note": "n"}}, SPEC)]
+ assert R.by_dimension(R.findings(results)) == {"correctness": 5.0, "efficiency": 0.5}
+
+
+def test_by_dimension_drops_dimensions_with_no_losses():
+ results = [_result("t", {"is_terse": {"value": 0.0, "note": "n"}}, SPEC)]
+ assert R.by_dimension(R.findings(results)) == {"efficiency": 1.0}
+
+
+def test_run_review_scores_grades_and_writes_a_report(tmp_path, monkeypatch):
+ from ingot.mcp_server.registry import skill_revision
+ tasks = [{"task": "t1", "rubric": "r", "checklist": [SPEC["cites_source"]]},
+ {"task": "t2", "rubric": "r", "checklist": [SPEC["is_terse"]]}]
+ monkeypatch.setattr(R, "resolve_skill_dir", lambda name: tmp_path)
+ monkeypatch.setattr(R, "load_tasks", lambda skill, **_: (tasks[:1], tasks[1:], {}))
+ monkeypatch.setattr(R, "optimizable_components", lambda d: {"description": "d", "body": "B"})
+ monkeypatch.setattr(R, "assemble", lambda c: c["body"])
+ monkeypatch.setattr(R, "_llm", lambda model: "llm")
+ monkeypatch.setattr(R, "invoke_retry",
+ lambda llm, msgs: type("M", (), {"content": "ans", "usage_metadata": None})())
+ monkeypatch.setattr(R, "judge", lambda task, rubric, answer, **kw: {
+ "score": 0.5, "feedback": "f",
+ "checklist": {c["id"]: {"value": 0.5, "note": "half"} for c in kw["checklist"]}})
+ monkeypatch.setattr(R, "REVIEW_DIR", tmp_path / "reviews")
+ expected_revision = skill_revision(tmp_path)
+
+ out = R.run_review("sk", log=lambda *a: None)
+ assert out["score"] == pytest.approx(0.5)
+ assert out["tasks"] == 2 and out["failed_checks"] == 2
+ written = json.loads((tmp_path / "reviews" / "sk.json").read_text())
+ assert written["skill"] == "sk"
+ assert written["revision"] == expected_revision
+ assert [f["check"] for f in written["findings"]] == ["cites_source", "is_terse"] # by cost
+
+
+def test_run_review_grades_train_and_holdout_together(tmp_path, monkeypatch):
+ """A review is not measuring generalization, so holding half the tasks back would only make it
+ a noisier read on the same skill."""
+ seen = []
+ monkeypatch.setattr(R, "resolve_skill_dir", lambda name: tmp_path)
+ monkeypatch.setattr(R, "load_tasks",
+ lambda skill, **_: ([{"task": "train", "rubric": "r"}],
+ [{"task": "holdout", "rubric": "r"}], {}))
+ monkeypatch.setattr(R, "optimizable_components", lambda d: {"description": "d", "body": "B"})
+ monkeypatch.setattr(R, "assemble", lambda c: c["body"])
+ monkeypatch.setattr(R, "_llm", lambda model: "llm")
+ monkeypatch.setattr(R, "invoke_retry", lambda llm, msgs: (
+ seen.append(msgs[1][1]) or type("M", (), {"content": "a", "usage_metadata": None})()))
+ monkeypatch.setattr(R, "judge", lambda *a, **k: {"score": 1.0, "feedback": "",
+ "checklist": {}})
+ monkeypatch.setattr(R, "REVIEW_DIR", tmp_path / "reviews")
+ R.run_review("sk", log=lambda *a: None)
+ assert sorted(seen) == ["holdout", "train"]
+
+
+def test_run_review_hashes_and_grades_one_immutable_snapshot(tmp_path, monkeypatch):
+ from ingot.mcp_server.registry import skill_revision, write_skill_md
+ skill = tmp_path / "skills" / "sk"
+ skill.mkdir(parents=True)
+ write_skill_md(skill / "SKILL.md", {"name": "sk", "description": "old description"},
+ "old body")
+ expected_revision = skill_revision(skill)
+ seen_systems = []
+
+ monkeypatch.setattr(R, "resolve_skill_dir", lambda name: skill)
+
+ def mutate_then_load(name, **_):
+ write_skill_md(skill / "SKILL.md", {"name": "sk", "description": "new description"},
+ "new body")
+ return ([{"task": "t", "rubric": "r"}], [], {})
+
+ monkeypatch.setattr(R, "load_tasks", mutate_then_load)
+ monkeypatch.setattr(R, "_llm", lambda model: "llm")
+ monkeypatch.setattr(R, "invoke_retry", lambda llm, msgs: (
+ seen_systems.append(msgs[0][1]) or
+ type("M", (), {"content": "a", "usage_metadata": None})()))
+ monkeypatch.setattr(R, "judge", lambda *a, **k: {
+ "score": 1.0, "feedback": "", "checklist": {}})
+ monkeypatch.setattr(R, "REVIEW_DIR", tmp_path / "reviews")
+
+ result = R.run_review("sk", log=lambda *a: None)
+
+ assert result["revision"] == expected_revision
+ assert seen_systems and "old body" in seen_systems[0] and "new body" not in seen_systems[0]
+
+
+def test_run_review_drafts_missing_evals_from_the_reviewed_snapshot(tmp_path, monkeypatch):
+ from pathlib import Path
+ from ingot.mcp_server import registry
+ from ingot.mcp_server.registry import write_skill_md
+ from ingot.optimize import ab, draft
+ root = tmp_path / "skills"
+ skill = root / "sk"
+ skill.mkdir(parents=True)
+ write_skill_md(skill / "SKILL.md", {"name": "sk", "description": "old description"},
+ "old body")
+ tasks = tmp_path / "tasks"
+ captured = {}
+ read_components = registry.read_components
+
+ def mutate_before_live_read(path):
+ if Path(path).resolve() == skill.resolve():
+ write_skill_md(skill / "SKILL.md",
+ {"name": "sk", "description": "new description"}, "new body")
+ return read_components(path)
+
+ def draft_eval(name, description, body, out_dir, log=print):
+ captured.update(description=description, body=body)
+ out_dir.mkdir(parents=True, exist_ok=True)
+ (out_dir / f"{name}.yaml").write_text(
+ "train:\n- task: train\n rubric: r\nholdout:\n- task: holdout\n rubric: r\n")
+
+ monkeypatch.setenv("INGOT_LIBRARY", str(root))
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root))
+ monkeypatch.setattr(ab, "TASKS_DIR", tasks)
+ monkeypatch.setattr(registry, "read_components", mutate_before_live_read)
+ monkeypatch.setattr(draft, "draft_and_save", draft_eval)
+ monkeypatch.setattr(R, "resolve_skill_dir", lambda name: skill)
+ monkeypatch.setattr(R, "_llm", lambda model: "llm")
+ monkeypatch.setattr(R, "invoke_retry", lambda *a, **k: type(
+ "M", (), {"content": "a", "usage_metadata": None})())
+ monkeypatch.setattr(R, "judge", lambda *a, **k: {
+ "score": 1.0, "feedback": "", "checklist": {}})
+ monkeypatch.setattr(R, "REVIEW_DIR", tmp_path / "reviews")
+
+ R.run_review("sk", log=lambda *a: None)
+
+ assert captured == {"description": "old description", "body": "old body"}
+
+
+def test_run_review_refuses_a_skill_with_no_tasks(tmp_path, monkeypatch):
+ from ingot.mcp_server.registry import write_skill_md
+ write_skill_md(tmp_path / "SKILL.md", {"name": "sk", "description": "d"}, "body")
+ monkeypatch.setattr(R, "resolve_skill_dir", lambda name: tmp_path)
+ monkeypatch.setattr(R, "load_tasks", lambda skill, **_: ([], [], {}))
+ with pytest.raises(SystemExit, match="no eval tasks"):
+ R.run_review("sk", log=lambda *a: None)
diff --git a/tests/test_router.py b/tests/test_router.py
index 00c19e6..e0078ac 100644
--- a/tests/test_router.py
+++ b/tests/test_router.py
@@ -3,8 +3,8 @@
collision check all depend on."""
import pytest
-from mcp_server.registry import Skill
-from mcp_server.router import Router
+from ingot.mcp_server.registry import Skill
+from ingot.mcp_server.router import Router
SKILLS = [
Skill("pdf", "Merge, split, and extract text from PDF files and documents.", "body", "p"),
@@ -43,7 +43,7 @@ def test_nearest_empty_router_is_safe():
def test_router_reuses_description_vectors_across_refreshes(monkeypatch):
import numpy as np
- import mcp_server.router as router_mod
+ import ingot.mcp_server.router as router_mod
calls = []
class FakeEmbedding:
@@ -96,6 +96,8 @@ def test_route_returns_clean_no_match_below_threshold():
assert result["skill_body"] == "" and result["skill_root"] is None
assert "threshold" in result["reason"]
assert result["alternatives"][0]["name"] == "pdf"
+ assert result["matched_on"] in {"description", "content"}
+ assert result["score"] == max(result["score_components"].values())
def test_route_novel_flag_signals_weak_strong_escalation():
@@ -127,6 +129,7 @@ def test_route_novel_flag_signals_weak_strong_escalation():
({"required_tools": ["browser"]}, {"available_tools": ["bash"]}),
({"required_mcps": ["github"]}, {"available_mcps": []}),
({"scopes": ["project"], "path_patterns": ["*/wanted/*"]}, {"cwd": "/tmp/other/project"}),
+ ({"scopes": ["project"], "path_patterns": ["/tmp/*"]}, {}),
])
def test_route_filters_incompatible_skills_before_ranking(skill_metadata, context):
router = Router([_skill("blocked", "Merge PDF documents.", **skill_metadata)])
@@ -166,3 +169,131 @@ def test_conflicting_skills_do_not_both_appear_in_ranked_result():
result = Router([one, two]).route("same routing text", "codex", "/tmp", min_score=0.0)
ranked = [result["match"], *[item["name"] for item in result["alternatives"]]]
assert not ({"one", "two"} <= set(ranked))
+
+
+class _BodyAwareEmbedding:
+ """Deterministic vectors: descriptions/billing point east, Kubernetes content/query north."""
+
+ def __init__(self):
+ self.document_calls = []
+
+ @staticmethod
+ def _vector(text):
+ import numpy as np
+ if "CrashLoopBackOff" in text or "kubernetes pod" in text.lower():
+ return np.array([0.0, 1.0], dtype=np.float32)
+ return np.array([1.0, 0.0], dtype=np.float32)
+
+ def embed(self, texts):
+ values = list(texts)
+ self.document_calls.append(values)
+ return iter(self._vector(text) for text in values)
+
+ def embed_query(self, texts):
+ return iter(self._vector(text) for text in texts)
+
+
+def test_body_aware_route_breaks_an_ambiguous_description_tie(monkeypatch):
+ import ingot.mcp_server.router as router_mod
+ embedder = _BodyAwareEmbedding()
+ monkeypatch.setattr(router_mod, "build_embedding", lambda: embedder)
+ router_mod.Router._vector_cache.clear()
+ router = router_mod.Router([
+ _skill("billing-runbook", "Operate a production service."),
+ Skill(**{**_skill("kubernetes-runbook", "Operate a production service.").__dict__,
+ "body": "Diagnose a kubernetes pod in CrashLoopBackOff."}),
+ ])
+
+ result = router.route("diagnose a kubernetes pod", "codex", "/tmp", min_score=0.0)
+
+ assert result["match"] == "kubernetes-runbook"
+ assert result["matched_on"] == "content"
+ assert result["score_components"]["content"] > result["score_components"]["description"]
+ assert all("skill_body" not in item for item in result["alternatives"])
+
+
+def test_body_aware_route_filters_incompatible_content_before_ranking(monkeypatch):
+ import ingot.mcp_server.router as router_mod
+ monkeypatch.setattr(router_mod, "build_embedding", _BodyAwareEmbedding)
+ router_mod.Router._vector_cache.clear()
+ blocked = Skill(**{
+ **_skill("kubernetes-runbook", "Operate a production service.",
+ required_tools=["kubectl"]).__dict__,
+ "body": "Diagnose a kubernetes pod in CrashLoopBackOff.",
+ })
+ router = router_mod.Router([
+ _skill("billing-runbook", "Operate a production service."),
+ blocked,
+ ])
+
+ result = router.route("diagnose a kubernetes pod", "codex", "/tmp",
+ available_tools=[], min_score=0.0)
+
+ assert result["match"] == "billing-runbook"
+
+
+def test_variant_content_is_ranked_for_the_requested_harness(monkeypatch):
+ import ingot.mcp_server.router as router_mod
+ monkeypatch.setattr(router_mod, "build_embedding", _BodyAwareEmbedding)
+ router_mod.Router._vector_cache.clear()
+ alpha = Skill(**{
+ **_skill("alpha", "Operate a production service.").__dict__,
+ "body": "Investigate invoice charges.",
+ "variants": {"codex": "Diagnose a kubernetes pod in CrashLoopBackOff."},
+ })
+ beta = Skill(**{
+ **_skill("beta", "Operate a production service.").__dict__,
+ "body": "Diagnose a kubernetes pod in CrashLoopBackOff.",
+ "variants": {"codex": "Investigate invoice charges."},
+ })
+ router = router_mod.Router([alpha, beta])
+
+ assert router.route("diagnose a kubernetes pod", "codex", "/tmp",
+ min_score=0.0)["match"] == "alpha"
+ assert router.route("diagnose a kubernetes pod", "claude", "/tmp",
+ min_score=0.0)["match"] == "beta"
+
+
+def test_body_change_reuses_description_vector_and_reembeds_content(monkeypatch):
+ import ingot.mcp_server.router as router_mod
+ embedder = _BodyAwareEmbedding()
+ monkeypatch.setattr(router_mod, "build_embedding", lambda: embedder)
+ router_mod.Router._vector_cache.clear()
+ first = _skill("runbook", "Operate a production service.")
+ second = Skill(**{**first.__dict__, "body": "Diagnose a CrashLoopBackOff."})
+
+ router_mod.Router([first]).route("diagnose a kubernetes pod", "codex", "/tmp",
+ min_score=0.0)
+ router_mod.Router([second]).route("diagnose a kubernetes pod", "codex", "/tmp",
+ min_score=0.0)
+
+ assert embedder.document_calls[0] == [first.description]
+ assert len(embedder.document_calls) == 3
+ assert "Instructions:" in embedder.document_calls[1][0]
+ assert "CrashLoopBackOff" in embedder.document_calls[2][0]
+
+
+def test_vector_cache_evicts_stale_body_revisions(monkeypatch):
+ import ingot.mcp_server.router as router_mod
+ embedder = _BodyAwareEmbedding()
+ monkeypatch.setattr(router_mod, "build_embedding", lambda: embedder)
+ monkeypatch.setattr(router_mod.Router, "_vector_cache_limit", 2, raising=False)
+ router_mod.Router._vector_cache.clear()
+
+ for revision in range(4):
+ skill = Skill(**{
+ **_skill("runbook", "Operate a production service.").__dict__,
+ "body": f"revision {revision} Diagnose a CrashLoopBackOff.",
+ })
+ router_mod.Router([skill]).route(
+ "diagnose a kubernetes pod", "codex", "/tmp", min_score=0.0
+ )
+
+ assert len(router_mod.Router._vector_cache) <= 2
+
+
+@pytest.mark.parametrize("value", ["0", "4001", "invalid"])
+def test_body_projection_bound_fails_closed(monkeypatch, value):
+ monkeypatch.setenv("ROUTER_BODY_CHARS", value)
+ with pytest.raises(ValueError, match="integer from 1 to 4000"):
+ Router([])
diff --git a/tests/test_routing.py b/tests/test_routing.py
index 8fd7f4b..d5abe64 100644
--- a/tests/test_routing.py
+++ b/tests/test_routing.py
@@ -2,7 +2,7 @@
no-regression/improvement/collision gate. No embeddings, no LLM, the router is injected."""
import pytest
-from optimize import routing as R
+from ingot.optimize import routing as R
class _ScriptedRouter:
@@ -115,11 +115,11 @@ def test_run_routing_auto_drafts_missing_cases(monkeypatch, tmp_path):
skill = tmp_path / "skills" / "sk"
skill.mkdir(parents=True)
(skill / "SKILL.md").write_text("---\nname: sk\ndescription: d.\n---\nbody\n")
- monkeypatch.setattr(R, "SKILLS_DIR", tmp_path / "skills")
+ monkeypatch.setattr(R, "resolve_skill_dir", lambda name: tmp_path / "skills" / name)
monkeypatch.setattr(R, "TASKS_DIR", tmp_path / "tasks", raising=False)
(tmp_path / "tasks").mkdir()
- from optimize import draft as D
+ from ingot.optimize import draft as D
def sentinel(*a, **k):
raise RuntimeError("drafter invoked")
monkeypatch.setattr(D, "draft_and_append_routing", sentinel)
@@ -132,26 +132,25 @@ def test_run_routing_writes_an_evidence_bundle_and_records_relative_paths(monkey
the same portable bundle rather than a claim that one exists."""
import json
- from optimize import promote as P
+ from ingot.optimize import promote as P
root = tmp_path / "skills"
skill = root / "sk"
skill.mkdir(parents=True)
(skill / "SKILL.md").write_text("---\nname: sk\ndescription: old trigger.\n---\nbody\n")
monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root))
- monkeypatch.setattr(R, "SKILLS_DIR", root)
+ monkeypatch.setattr(R, "resolve_skill_dir", lambda name: root / name)
tasks = tmp_path / "tasks"
tasks.mkdir()
(tasks / "sk.yaml").write_text(
"routing:\n - task: use sk please\n expected: sk\n - task: unrelated\n expected: null\n")
monkeypatch.setattr(R, "TASKS_DIR", tasks, raising=False)
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs"))
evidence_root = tmp_path / "runs" / "evidence"
- monkeypatch.setattr(R, "EVIDENCE_DIR", evidence_root)
- monkeypatch.setattr(P, "PENDING_DIR", tmp_path / "pending")
monkeypatch.setattr(R, "optimize_description",
lambda skill, seed, cases, budget: ("new trigger.", 0.5, 0.9))
monkeypatch.setattr(R, "_description_shadows", lambda skill, desc: ("", 0.0))
- from optimize import ab as A
+ from ingot.optimize import ab as A
monkeypatch.setattr(A, "_routing_metrics", lambda skill, champ, chall: {
"champion": {"top1": 0.5, "recall_at_3": 0.5, "no_route_precision": 1.0},
"challenger": {"top1": 1.0, "recall_at_3": 1.0, "no_route_precision": 1.0},
diff --git a/tests/test_routing_eval.py b/tests/test_routing_eval.py
index f3ed568..43b2a97 100644
--- a/tests/test_routing_eval.py
+++ b/tests/test_routing_eval.py
@@ -1,6 +1,6 @@
-from mcp_server.routing_eval import evaluate_cases, evaluate_parity, load_cases
-from mcp_server.registry import load_skills
-from mcp_server.router import Router
+from ingot.mcp_server.routing_eval import evaluate_cases, evaluate_parity, load_cases
+from ingot.mcp_server.registry import load_skills
+from ingot.mcp_server.router import Router
from pathlib import Path
@@ -54,9 +54,13 @@ def route(self, task, harness, **context):
def test_committed_suite_covers_filter_and_parity_contract():
root = Path(__file__).resolve().parent.parent
cases = load_cases(root / "evals" / "routing.yaml")
- result = evaluate_cases(Router(load_skills(root / "evals" / "fixtures" / "skills")), cases)
- parity = evaluate_parity(Router(load_skills(root / "evals" / "fixtures" / "skills")), cases)
+ router = Router(load_skills(root / "evals" / "fixtures" / "skills"))
+ result = evaluate_cases(router, cases)
+ parity = evaluate_parity(router, cases)
assert len(cases) >= 10
assert result["failures"] == []
assert result["recall_at_3"] == 1.0 and result["no_route_precision"] == 1.0
assert parity["rate"] == 1.0 and parity["total"] >= 2
+ body_case = next(case for case in cases if case["expected"] == "kubernetes-runbook")
+ routed = router.route(body_case["task"], body_case["harness"], body_case.get("cwd", "."))
+ assert routed["match"] == "kubernetes-runbook" and routed["matched_on"] == "content"
diff --git a/tests/test_routing_health.py b/tests/test_routing_health.py
index f55d174..f2534bf 100644
--- a/tests/test_routing_health.py
+++ b/tests/test_routing_health.py
@@ -4,7 +4,7 @@
import yaml
-from optimize import routing_health as H
+from ingot.optimize import routing_health as H
class _ScriptedRouter:
diff --git a/tests/test_run_task.py b/tests/test_run_task.py
index 6102b8f..0b4e5be 100644
--- a/tests/test_run_task.py
+++ b/tests/test_run_task.py
@@ -156,7 +156,7 @@ def test_serving_contract_requires_inline_deliverables():
# the scaffold habit of writing code to its scratch FS and describing it must be countered in
# BOTH serving contracts, symmetrically, production agent and A/B eval agent
from agent.run import INSTRUCTIONS
- from optimize.ab import EVAL_INSTRUCTIONS
+ from ingot.optimize.ab import EVAL_INSTRUCTIONS
for contract in (INSTRUCTIONS, EVAL_INSTRUCTIONS):
assert "final answer must contain the complete deliverable" in contract
assert "cannot" in contract and "workspace" in contract
@@ -240,7 +240,7 @@ async def serve(task, routed, tools):
monkeypatch.setattr(run_mod, "_connect", connect)
monkeypatch.setattr(run_mod, "_serve", serve)
monkeypatch.setattr(run_mod, "_print_route", lambda routed: None)
- monkeypatch.setattr("optimize.openrouter_key_missing", lambda: False)
+ monkeypatch.setattr("ingot.optimize.openrouter_key_missing", lambda: False)
asyncio.run(run_mod.main("write a skill"))
diff --git a/tests/test_security.py b/tests/test_security.py
index 1b790c9..ecb1125 100644
--- a/tests/test_security.py
+++ b/tests/test_security.py
@@ -4,10 +4,10 @@
import pytest
-from mcp_server.registry import (
+from ingot.mcp_server.registry import (
SLUG_RE, parse_skill, read_components, write_components, write_skill_md,
)
-from optimize.promote import check_slug
+from ingot.optimize.promote import check_slug
ROOT = Path(__file__).resolve().parents[1]
diff --git a/tests/test_server.py b/tests/test_server.py
index f5d93b1..ad1e5cd 100644
--- a/tests/test_server.py
+++ b/tests/test_server.py
@@ -1,6 +1,6 @@
import asyncio
-from mcp_server.server import STATE, get_skill, mcp, route_and_load
+from ingot.mcp_server.server import STATE, get_skill, mcp, route_and_load
def test_get_skill_header_carries_revision(tmp_path):
@@ -19,6 +19,8 @@ def test_route_and_load_is_additive_to_existing_mcp_tools():
tools = asyncio.run(mcp.list_tools())
assert {tool.name for tool in tools} == {
"list_skills", "suggest_skills", "get_skill", "reload_skills", "route_and_load",
+ "propose_skill_update",
+ "propose_skill_create",
}
@@ -29,7 +31,7 @@ def test_route_refreshes_after_external_skill_promotion(tmp_path, monkeypatch):
md = skill / "SKILL.md"
md.write_text("---\nname: pdf\ndescription: Merge PDF files.\n---\nbody one\n")
STATE.reload([root])
- monkeypatch.setattr("mcp_server.server.MIN_SCORE", 0.0)
+ monkeypatch.setattr("ingot.mcp_server.server.MIN_SCORE", 0.0)
first = route_and_load("merge PDF", "codex", str(tmp_path))
md.write_text("---\nname: pdf\ndescription: Merge PDF files.\n---\nbody two\n")
second = route_and_load("merge PDF", "codex", str(tmp_path))
@@ -45,18 +47,18 @@ def test_route_and_load_novel_flag_uses_server_thresholds(tmp_path, monkeypatch)
(skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDF files.\n---\nbody\n")
STATE.reload([root])
# match -> weak model serves the skill
- monkeypatch.setattr("mcp_server.server.MIN_SCORE", 0.0)
+ monkeypatch.setattr("ingot.mcp_server.server.MIN_SCORE", 0.0)
assert route_and_load("merge PDF", "codex", str(tmp_path))["novel"] is False
# no match but within the related band -> compose/extend, still not novel
- monkeypatch.setattr("mcp_server.server.MIN_SCORE", 0.99)
- monkeypatch.setattr("mcp_server.server.RELATED_SCORE", 0.0)
+ monkeypatch.setattr("ingot.mcp_server.server.MIN_SCORE", 0.99)
+ monkeypatch.setattr("ingot.mcp_server.server.RELATED_SCORE", 0.0)
related = route_and_load("merge PDF", "codex", str(tmp_path))
assert related["match"] is None and related["related_match"] == "pdf"
assert related["novel"] is False and related["skill_body"] == "body"
assert related["skill_root"] == str(skill)
assert related["revision"]
# nothing even related -> the harness should escalate to its strong model
- monkeypatch.setattr("mcp_server.server.RELATED_SCORE", 0.99)
+ monkeypatch.setattr("ingot.mcp_server.server.RELATED_SCORE", 0.99)
novel = route_and_load("merge PDF", "codex", str(tmp_path))
assert novel["match"] is None and novel["novel"] is True
assert novel["related_match"] is None and novel["skill_body"] == ""
@@ -70,7 +72,7 @@ def test_route_refreshes_revision_after_bundled_file_change(tmp_path, monkeypatc
reference = skill / "reference.md"
reference.write_text("version one")
STATE.reload([root])
- monkeypatch.setattr("mcp_server.server.MIN_SCORE", 0.0)
+ monkeypatch.setattr("ingot.mcp_server.server.MIN_SCORE", 0.0)
first = route_and_load("merge PDF", "codex", str(tmp_path))
reference.write_text("version two")
second = route_and_load("merge PDF", "codex", str(tmp_path))
diff --git a/tests/test_setup_scripts.py b/tests/test_setup_scripts.py
index dedfb1f..a3ac32e 100644
--- a/tests/test_setup_scripts.py
+++ b/tests/test_setup_scripts.py
@@ -46,6 +46,7 @@ def test_codex_setup_is_idempotent_and_writes_private_config(tmp_path):
_executable(fake_bin / "codex", '''
echo "codex $*" >> "$TEST_STATE/calls"
if [ "$1" = "--version" ]; then echo "codex-cli 0.144.5"; exit 0; fi
+if [ "$1 $2" = "plugin --help" ]; then exit 0; fi
if [ "$1 $2 $3" = "mcp get ingot" ]; then
test -f "$TEST_STATE/mcp" && echo "url: http://localhost:8000/mcp"
test -f "$TEST_STATE/mcp"
@@ -79,19 +80,47 @@ def test_codex_setup_is_idempotent_and_writes_private_config(tmp_path):
assert stat.S_IMODE(config.stat().st_mode) == 0o600
-def test_codex_setup_rejects_old_codex_before_writing_config(tmp_path):
+def test_codex_setup_rejects_a_codex_without_the_plugin_subcommand(tmp_path):
+ """The floor is a capability, not a number: the script installs through `codex plugin`, so it
+ probes for that. A build too old to carry it is rejected before any credential is written."""
env, fake_bin = _environment(tmp_path)
_executable(fake_bin / "node", 'echo 22\n')
- _executable(fake_bin / "codex", 'echo "codex-cli 0.127.9"\n')
+ _executable(fake_bin / "codex", '''
+if [ "$1" = "--version" ]; then echo "codex-cli 0.127.9"; exit 0; fi
+if [ "$1" = "plugin" ]; then echo "unrecognized subcommand 'plugin'" >&2; exit 2; fi
+exit 0
+''')
result = subprocess.run([str(ROOT / "scripts" / "codex_setup.sh")], cwd=ROOT, env=env,
text=True, capture_output=True)
assert result.returncode != 0
- assert "Codex 0.128 or newer" in result.stderr
+ assert "no 'plugin' subcommand" in result.stderr
assert not (Path(env["HOME"]) / ".codex" / "langfuse.json").exists()
+def test_codex_setup_accepts_a_local_build_that_stamps_no_version(tmp_path):
+ """A locally built codex reports `codex-cli 0.0.0`, which sorts below every release while
+ carrying the plugin subcommand. A version comparison rejected exactly the build that works —
+ this is the regression the capability probe exists to prevent."""
+ env, fake_bin = _environment(tmp_path)
+ _executable(fake_bin / "node", 'echo 22\n')
+ _executable(fake_bin / "codex", '''
+if [ "$1" = "--version" ]; then echo "codex-cli 0.0.0-wire-persona"; exit 0; fi
+if [ "$1 $2" = "plugin --help" ]; then exit 0; fi
+if [ "$1 $2 $3" = "mcp get ingot" ]; then exit 1; fi
+exit 0
+''')
+ env.update({"LANGFUSE_BASE_URL": "https://langfuse.example",
+ "LANGFUSE_PUBLIC_KEY": "pk-test", "LANGFUSE_SECRET_KEY": "sk-test"})
+
+ result = subprocess.run([str(ROOT / "scripts" / "codex_setup.sh")], cwd=ROOT, env=env,
+ text=True, capture_output=True)
+
+ assert result.returncode == 0, result.stderr
+ assert (Path(env["HOME"]) / ".codex" / "langfuse.json").exists()
+
+
def test_remote_setup_requires_explicit_langfuse_credentials(tmp_path):
env, _ = _environment(tmp_path)
env["LANGFUSE_BASE_URL"] = "https://langfuse.example"
@@ -194,7 +223,7 @@ def test_live_smokes_require_completed_mcp_call_and_mining_parser():
assert "ingot/route_and_load (completed)" in codex
assert "mcp__ingot__route_and_load" in claude and "tool_result" in claude
- assert "from optimize.mine import fetch_traces" in codex
- assert "from optimize.mine import fetch_traces" in claude
+ assert "from ingot.optimize.mine import fetch_traces" in codex
+ assert "from ingot.optimize.mine import fetch_traces" in claude
assert "--add-host host.docker.internal:host-gateway" in codex
assert "--add-host host.docker.internal:host-gateway" in claude
diff --git a/tests/test_skillopt_bridge.py b/tests/test_skillopt_bridge.py
index 6e01608..aed3416 100644
--- a/tests/test_skillopt_bridge.py
+++ b/tests/test_skillopt_bridge.py
@@ -3,7 +3,7 @@
driven by a stub reflection LM."""
import json
-from optimize import skillopt_bridge as sk
+from ingot.optimize import skillopt_bridge as sk
def _lm(reply: str):
diff --git a/tests/test_skillopt_loop.py b/tests/test_skillopt_loop.py
index 3036f4d..86c7f1f 100644
--- a/tests/test_skillopt_loop.py
+++ b/tests/test_skillopt_loop.py
@@ -5,8 +5,8 @@
import pytest
-from optimize import rollout as R
-from optimize import skillopt_loop as S
+from ingot.optimize import rollout as R
+from ingot.optimize import skillopt_loop as S
def _fake_lm(reply_for):
diff --git a/tests/test_status.py b/tests/test_status.py
new file mode 100644
index 0000000..7f33e5c
--- /dev/null
+++ b/tests/test_status.py
@@ -0,0 +1,296 @@
+"""`ingot status`: the four states, decided by observation rather than by a configuration flag.
+
+MANAGED, PENDING, DRIFTED, UNMANAGED are answers about what is actually served compared with what
+the last successful release receipt says should be served. A status command that read its verdict
+back out of the configuration would agree with the claim instead of testing it."""
+import pytest
+
+from ingot import cli, status
+from ingot.optimize import promote as P
+from ingot.optimize import publication as Q
+
+
+def _library(tmp_path, monkeypatch):
+ root = tmp_path / "skills"
+ root.mkdir()
+ monkeypatch.setenv("INGOT_LIBRARY", str(root))
+ return root
+
+
+def _skill(root, name, body="a body"):
+ directory = root / name
+ directory.mkdir(exist_ok=True)
+ (directory / "SKILL.md").write_text(
+ f"---\nname: {name}\ndescription: Does the {name} thing.\n---\n{body}\n")
+ from ingot.mcp_server.registry import skill_revision
+ return skill_revision(directory)
+
+
+def _release(name, revision, *, champion="absent", state="active"):
+ """A receipt for one skill, driven through the real queue rather than hand-authored."""
+ receipt = Q.queue_publication(name, {
+ "skill": name, "kind": "creation",
+ "challenger_components": {"description": f"Does the {name} thing.", "body": "a body"},
+ "evidence": {"champion": {"revision": champion}, "challenger": {"revision": revision}},
+ }, "admin", "promote")
+ if state != "approved_publishing":
+ Q.update_publication(receipt.id, state=state)
+ return receipt
+
+
+def test_a_skill_serving_its_released_revision_is_managed(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ _release("pdf", _skill(root, "pdf"))
+
+ result = status.library_status(root)
+
+ assert result["mode"] == status.MANAGED
+ assert result["skills"] == [{"skill": "pdf", "state": status.MANAGED,
+ "revision": result["skills"][0]["revision"],
+ "released": result["skills"][0]["revision"],
+ "publication": result["skills"][0]["publication"]}]
+
+
+def test_an_out_of_band_edit_reports_drifted(tmp_path, monkeypatch):
+ """A read-only mount does not stop the machine owner from editing the host directory. This is
+ the detection that replaces claiming it does."""
+ root = _library(tmp_path, monkeypatch)
+ _release("pdf", _skill(root, "pdf"))
+ _skill(root, "pdf", body="edited by hand, out of band")
+
+ result = status.library_status(root)
+
+ assert result["mode"] == status.DRIFTED
+ assert result["skills"][0]["state"] == status.DRIFTED
+ assert result["skills"][0]["revision"] != result["skills"][0]["released"]
+ assert "outside the publisher" in status.render(result)
+
+
+def test_restoring_the_released_bytes_returns_to_managed(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ released = _skill(root, "pdf")
+ _release("pdf", released)
+ _skill(root, "pdf", body="edited by hand, out of band")
+ assert status.library_status(root)["mode"] == status.DRIFTED
+
+ _skill(root, "pdf")
+
+ assert status.library_status(root)["mode"] == status.MANAGED
+
+
+def test_a_skill_with_no_release_receipt_is_unmanaged_not_drifted(tmp_path, monkeypatch):
+ """Fetched, copied, or committed by hand. Real and common, and not drift: there is no release
+ for it to have drifted from, and calling it drift would make the alarm meaningless."""
+ root = _library(tmp_path, monkeypatch)
+ _skill(root, "pdf")
+
+ result = status.library_status(root)
+
+ assert result["mode"] == status.UNMANAGED
+ assert result["skills"][0]["state"] == status.UNMANAGED
+ assert result["skills"][0]["released"] is None
+
+
+def test_a_released_skill_with_a_newer_publication_in_flight_is_pending(tmp_path, monkeypatch):
+ """Serving the last release while the next one travels is the normal state, not drift."""
+ root = _library(tmp_path, monkeypatch)
+ released = _skill(root, "pdf")
+ _release("pdf", released)
+ in_flight = _release("pdf", "c" * 64, champion=released, state="publishing")
+
+ result = status.library_status(root)
+
+ assert Q.load_publication(in_flight.id)["state"] == "publishing"
+ assert result["skills"][0]["released"] == released
+ assert result["skills"][0]["state"] == status.PENDING
+ assert result["mode"] == status.PENDING
+
+
+def test_a_skill_in_flight_with_no_release_yet_is_pending(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ _release("pdf", _skill(root, "pdf"), state="publishing")
+
+ assert status.library_status(root)["mode"] == status.PENDING
+
+
+def test_a_creation_in_flight_is_pending_even_though_nothing_serves_it_yet(tmp_path, monkeypatch):
+ """Caught in situ: a new skill is served by nothing and released by nothing, so a status built
+ from those two sets alone called an empty library fully MANAGED mid-publication."""
+ root = _library(tmp_path, monkeypatch)
+ _release("csv-tidy", "e" * 64, state="approved_publishing")
+
+ result = status.library_status(root)
+
+ assert result["mode"] == status.PENDING
+ assert [entry["skill"] for entry in result["skills"]] == ["csv-tidy"]
+ assert result["skills"][0]["revision"] == status.ABSENT
+
+
+def test_a_quarantined_proposal_is_pending(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ _release("pdf", _skill(root, "pdf"))
+ P.save_pending("pdf", {"skill": "pdf", "gate": {"promotable": True},
+ "evidence": {"challenger": {"revision": "b" * 64}}})
+
+ assert status.library_status(root)["mode"] == status.PENDING
+
+
+def test_drift_outranks_a_publication_in_flight(tmp_path, monkeypatch):
+ """The alarm must not be masked by an unrelated change travelling to the vault."""
+ root = _library(tmp_path, monkeypatch)
+ released = _skill(root, "pdf")
+ _release("pdf", released)
+ _release("tailwind", _skill(root, "tailwind"))
+ _release("tailwind", "d" * 64, champion=released, state="publishing")
+ _skill(root, "pdf", body="edited by hand, out of band")
+
+ assert status.library_status(root)["mode"] == status.DRIFTED
+
+
+def test_an_empty_library_is_managed(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+
+ assert status.library_status(root)["mode"] == status.MANAGED
+
+
+def test_development_mode_is_unmanaged_whatever_the_receipts_say(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ _release("pdf", _skill(root, "pdf"))
+ monkeypatch.setenv("INGOT_MODE", "dev")
+
+ result = status.library_status(root)
+
+ assert result["mode"] == status.UNMANAGED
+ assert result["development_mode"] is True
+ assert "do not apply" in status.render(result)
+
+
+def test_a_writable_library_is_reported_without_deciding_the_verdict(tmp_path, monkeypatch):
+ """The administrator who owns the vault can always write it. A status command that answered
+ UNMANAGED from their shell would hide the drift they most need to see."""
+ root = _library(tmp_path, monkeypatch)
+ _release("pdf", _skill(root, "pdf"))
+
+ result = status.library_status(root)
+
+ assert result["mode"] == status.MANAGED
+ assert result["writable_roots"] == [str(root.resolve())]
+ assert "without an approval" in status.render(result)
+
+
+def test_status_exits_zero_only_when_everything_is_as_approved(tmp_path, monkeypatch, capsys):
+ root = _library(tmp_path, monkeypatch)
+ _release("pdf", _skill(root, "pdf"))
+ assert cli.main(["status", "--root", str(root)]) == 0
+
+ _skill(root, "pdf", body="edited by hand, out of band")
+
+ assert cli.main(["status", "--root", str(root)]) == 1
+ assert status.DRIFTED in capsys.readouterr().out
+
+
+def test_the_json_payload_is_versioned(tmp_path, monkeypatch, capsys):
+ import json
+ root = _library(tmp_path, monkeypatch)
+ _skill(root, "pdf")
+
+ cli.main(["status", "--root", str(root), "--json"])
+
+ assert json.loads(capsys.readouterr().out)["schema_version"] == status.STATUS_SCHEMA
+
+
+# --------------------------------------------------------------------------- delivery targets
+
+def _delivery(tmp_path, monkeypatch, vault):
+ """One native filesystem target beside the managed vault, configured the way an operator would."""
+ native = tmp_path / "claude"
+ native.mkdir()
+ monkeypatch.setenv("INGOT_VAULT_PATH", str(vault))
+ monkeypatch.setenv("INGOT_DELIVERY_TARGETS", f"claude=filesystem:{native}")
+ return native
+
+
+def _copy(source, destination):
+ import shutil
+ shutil.copytree(source, destination, dirs_exist_ok=True)
+
+
+def test_a_target_holding_the_released_revision_is_managed(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ native = _delivery(tmp_path, monkeypatch, root)
+ _release("pdf", _skill(root, "pdf"))
+ _copy(root / "pdf", native / "pdf")
+
+ targets = status.target_states()
+
+ assert [(entry["name"], entry["state"]) for entry in targets] == [
+ ("vault", status.MANAGED), ("claude", status.MANAGED)]
+
+
+def test_only_the_target_that_was_altered_reports_drift(tmp_path, monkeypatch):
+ """The acceptance case. Editing a native skill root must not implicate the vault, and the vault
+ still holding the release must not hide the edit."""
+ root = _library(tmp_path, monkeypatch)
+ native = _delivery(tmp_path, monkeypatch, root)
+ _release("pdf", _skill(root, "pdf"))
+ _copy(root / "pdf", native / "pdf")
+
+ (native / "pdf" / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Does the pdf thing.\n---\nedited out of band\n")
+
+ states = {entry["name"]: entry["state"] for entry in status.target_states()}
+ assert states == {"vault": status.MANAGED, "claude": status.DRIFTED}
+
+
+def test_a_released_skill_deleted_from_a_target_is_drift_not_silence(tmp_path, monkeypatch):
+ root = _library(tmp_path, monkeypatch)
+ native = _delivery(tmp_path, monkeypatch, root)
+ _release("pdf", _skill(root, "pdf"))
+
+ states = {entry["name"]: entry["state"] for entry in status.target_states()}
+ assert states["claude"] == status.DRIFTED
+
+
+def test_a_target_is_judged_only_on_the_skills_ingot_released_there(tmp_path, monkeypatch):
+ """A native skill root is shared. Skills the operator put there themselves are not Ingot's to
+ grade, and reporting them would make every real deployment permanently UNMANAGED."""
+ root = _library(tmp_path, monkeypatch)
+ native = _delivery(tmp_path, monkeypatch, root)
+ _release("pdf", _skill(root, "pdf"))
+ _copy(root / "pdf", native / "pdf")
+ _skill(native, "somebody-elses-skill")
+
+ claude = [entry for entry in status.target_states() if entry["name"] == "claude"][0]
+
+ assert claude["state"] == status.MANAGED
+ assert [skill["skill"] for skill in claude["skills"]] == ["pdf"]
+
+
+def test_a_drifted_target_is_not_hidden_by_a_clean_vault(tmp_path, monkeypatch, capsys):
+ """`ingot status` answering MANAGED while a native skill root serves the wrong bytes is exactly
+ the lie this command exists to prevent."""
+ root = _library(tmp_path, monkeypatch)
+ native = _delivery(tmp_path, monkeypatch, root)
+ _release("pdf", _skill(root, "pdf"))
+ _copy(root / "pdf", native / "pdf")
+ (native / "pdf" / "SKILL.md").write_text(
+ "---\nname: pdf\ndescription: Does the pdf thing.\n---\nedited out of band\n")
+
+ assert cli.main(["status", "--root", str(root)]) == 1
+
+ output = capsys.readouterr().out
+ assert output.startswith(status.DRIFTED)
+ assert "claude" in output and str(native) in output
+
+
+def test_an_unusable_delivery_configuration_is_reported_rather_than_raised(tmp_path, monkeypatch,
+ capsys):
+ """Status is the command an operator runs when something is wrong. It has to survive a bad
+ environment variable and say what is wrong with it."""
+ root = _library(tmp_path, monkeypatch)
+ monkeypatch.setenv("INGOT_VAULT_PATH", str(root))
+ monkeypatch.setenv("INGOT_DELIVERY_TARGETS", "claude=carrier-pigeon:/tmp/x")
+
+ assert cli.main(["status", "--root", str(root)]) == 1
+
+ assert "unknown delivery kind" in capsys.readouterr().out
diff --git a/tests/test_tree.py b/tests/test_tree.py
new file mode 100644
index 0000000..6c5296a
--- /dev/null
+++ b/tests/test_tree.py
@@ -0,0 +1,281 @@
+"""The candidate tree: the exact bytes an admitted package publishes.
+
+Every test here exists because the alternative was silent. A package used to be reduced to decoded
+text on the way in, so a file the dictionary could not hold was reviewed as part of the candidate
+and then was not in it -- no error, no warning, just a revision naming a package that no longer
+existed. These check that the bytes survive, that the receipt binds them, and that anything which
+moves them afterwards is refused rather than published."""
+import hashlib
+import os
+import stat
+
+import pytest
+
+from ingot.mcp_server.registry import skill_revision
+from ingot.optimize import tree
+
+PNG = b"\x89PNG\r\n\x1a\n" + bytes(range(256))
+
+
+@pytest.fixture(autouse=True)
+def _runs(tmp_path, monkeypatch):
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs"))
+
+
+def _package(root, name="pdf", description="Merge and split PDF files.", body="Combine PDFs."):
+ directory = root / name
+ directory.mkdir(parents=True, exist_ok=True)
+ (directory / "SKILL.md").write_text(
+ f"---\nname: {name}\ndescription: {description}\n---\n\n{body}\n", encoding="utf-8")
+ return directory
+
+
+def _components(description="Merge and split PDF files.", body="Combine PDFs."):
+ import json
+ return {"description": description, "body": body,
+ "frontmatter": json.dumps({"name": "pdf", "description": description})}
+
+
+# --- describing a package ---------------------------------------------------------------------
+
+def test_every_regular_file_is_described_by_its_raw_bytes(tmp_path):
+ package = _package(tmp_path / "src")
+ (package / "assets").mkdir()
+ (package / "assets" / "logo.png").write_bytes(PNG)
+
+ manifest = tree.build(package)
+
+ entry = next(item for item in manifest["files"] if item["path"] == "assets/logo.png")
+ assert entry == {"path": "assets/logo.png", "mode": 0o644, "size": len(PNG),
+ "sha256": hashlib.sha256(PNG).hexdigest()}
+
+
+def test_the_hash_is_of_bytes_not_of_decoded_text(tmp_path):
+ """A file that is not valid UTF-8 has no decoded form, and one that is would hash differently
+ after a round trip through `errors='ignore'` -- which is how the bytes went missing."""
+ package = _package(tmp_path / "src")
+ raw = b"caf\xe9 latin-1, not utf-8\n"
+ (package / "notes.txt").write_bytes(raw)
+
+ entry = next(item for item in tree.build(package)["files"] if item["path"] == "notes.txt")
+
+ assert entry["sha256"] == hashlib.sha256(raw).hexdigest()
+
+
+def test_modes_are_clamped_to_the_two_a_git_checkout_reproduces(tmp_path):
+ package = _package(tmp_path / "src")
+ (package / "run.sh").write_text("#!/bin/sh\n")
+ (package / "run.sh").chmod(0o764)
+ (package / "notes.md").write_text("# Notes\n")
+ (package / "notes.md").chmod(0o600)
+
+ modes = {item["path"]: item["mode"] for item in tree.build(package)["files"]}
+
+ assert modes["run.sh"] == 0o755
+ assert modes["notes.md"] == 0o644
+
+
+@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows")
+def test_a_symlink_is_refused_by_name(tmp_path):
+ package = _package(tmp_path / "src")
+ (package / "real.md").write_text("# Real\n")
+ (package / "link.md").symlink_to(package / "real.md")
+
+ with pytest.raises(ValueError, match="symlinks are not admissible: link.md"):
+ tree.build(package)
+
+
+@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows")
+def test_a_directory_symlink_is_refused_before_it_is_descended(tmp_path):
+ package = _package(tmp_path / "src")
+ outside = tmp_path / "outside"
+ outside.mkdir()
+ (outside / "secret.md").write_text("secret\n")
+ (package / "docs").symlink_to(outside, target_is_directory=True)
+
+ with pytest.raises(ValueError, match="symlinks are not admissible: docs"):
+ tree.build(package)
+
+
+def test_an_unportable_path_is_refused(tmp_path):
+ package = _package(tmp_path / "src")
+ (package / "a:b.md").write_text("# Colon\n")
+
+ with pytest.raises(ValueError, match="portable POSIX path"):
+ tree.build(package)
+
+
+def test_the_byte_budget_is_enforced(tmp_path, monkeypatch):
+ monkeypatch.setattr(tree, "MAX_TREE_BYTES", 128)
+ package = _package(tmp_path / "src")
+ (package / "big.bin").write_bytes(b"\x00" * 512)
+
+ with pytest.raises(ValueError, match="at most 128 bytes"):
+ tree.build(package)
+
+
+def test_the_file_budget_is_enforced(tmp_path, monkeypatch):
+ monkeypatch.setattr(tree, "MAX_FILES", 3)
+ package = _package(tmp_path / "src")
+ for index in range(5):
+ (package / f"note-{index}.md").write_text("# Note\n")
+
+ with pytest.raises(ValueError, match="at most 3 files"):
+ tree.build(package)
+
+
+# --- binding the manifest ---------------------------------------------------------------------
+
+def test_the_digest_covers_the_file_list(tmp_path):
+ """The digest is what binds a receipt to a staged tree. A receipt whose file list was edited
+ without its digest would publish a tree nobody approved."""
+ package = _package(tmp_path / "src")
+ manifest = tree.build(package)
+ manifest["files"][0]["sha256"] = "0" * 64
+
+ with pytest.raises(ValueError, match="digest does not cover its file list"):
+ tree.verify_manifest(manifest)
+
+
+def test_a_manifest_entry_that_escapes_the_skill_root_is_refused(tmp_path):
+ package = _package(tmp_path / "src")
+ manifest = tree.build(package)
+ manifest["files"][0]["path"] = "../escape.md"
+ manifest["digest"] = tree._digest(manifest["files"])
+
+ with pytest.raises(ValueError, match="escapes skill root"):
+ tree.verify_manifest(manifest)
+
+
+def test_a_mode_the_vault_cannot_serve_is_refused(tmp_path):
+ package = _package(tmp_path / "src")
+ manifest = tree.build(package)
+ manifest["files"][0]["mode"] = 0o777
+ manifest["digest"] = tree._digest(manifest["files"])
+
+ with pytest.raises(ValueError, match="unsupported mode"):
+ tree.verify_manifest(manifest)
+
+
+# --- staging ----------------------------------------------------------------------------------
+
+def test_staging_copies_the_exact_bytes(tmp_path):
+ package = _package(tmp_path / "src")
+ (package / "logo.png").write_bytes(PNG)
+ manifest = tree.build(package)
+
+ staged = tree.stage(package, manifest)
+
+ assert (staged / "logo.png").read_bytes() == PNG
+ assert staged == tree.staged_dir(manifest["digest"])
+
+
+def test_staging_identical_bytes_twice_converges_on_one_directory(tmp_path):
+ """Named by digest, so a resubmission of the same package is not a second copy or a race."""
+ first = _package(tmp_path / "one")
+ second = _package(tmp_path / "two")
+
+ one = tree.stage(first, tree.build(first))
+ two = tree.stage(second, tree.build(second))
+
+ assert one == two
+ assert len(list(tree.candidates_dir().iterdir())) == 1
+
+
+def test_a_failed_staging_leaves_nothing_behind(tmp_path):
+ package = _package(tmp_path / "src")
+ manifest = tree.build(package)
+ (package / "SKILL.md").write_text("changed after the manifest was built\n")
+
+ with pytest.raises(ValueError, match="changed while it was being staged"):
+ tree.stage(package, manifest)
+
+ assert list(tree.candidates_dir().iterdir()) == []
+
+
+# --- materializing ------------------------------------------------------------------------------
+
+def test_materializing_reproduces_the_bytes_and_the_mode(tmp_path):
+ package = _package(tmp_path / "src")
+ (package / "logo.png").write_bytes(PNG)
+ (package / "run.sh").write_text("#!/bin/sh\necho hi\n")
+ (package / "run.sh").chmod(0o755)
+ manifest = tree.build(package)
+ tree.stage(package, manifest)
+
+ destination = tmp_path / "out"
+ tree.materialize(manifest, destination)
+
+ assert (destination / "logo.png").read_bytes() == PNG
+ assert stat.S_IMODE((destination / "run.sh").stat().st_mode) == 0o755
+
+
+def test_a_staged_file_altered_after_approval_is_refused(tmp_path):
+ """The receipt is the authority. If the staged bytes have moved since it was written, the
+ publisher must refuse rather than publish whatever it finds."""
+ package = _package(tmp_path / "src")
+ (package / "logo.png").write_bytes(PNG)
+ manifest = tree.build(package)
+ staged = tree.stage(package, manifest)
+ (staged / "logo.png").write_bytes(b"something else")
+
+ with pytest.raises(ValueError, match="does not match the receipt: logo.png"):
+ tree.materialize(manifest, tmp_path / "out")
+
+
+def test_a_missing_staged_tree_is_refused_not_skipped(tmp_path):
+ package = _package(tmp_path / "src")
+ manifest = tree.build(package)
+
+ with pytest.raises(ValueError, match="staged candidate tree .* is missing"):
+ tree.materialize(manifest, tmp_path / "out")
+
+
+# --- what the library ends up serving -----------------------------------------------------------
+
+def test_skill_md_is_normalized_and_everything_else_is_preserved(tmp_path):
+ """The one documented exception. SKILL.md's frontmatter is the routing interface, so it is
+ re-emitted through a safe YAML dump; every other file is the operator's bytes."""
+ package = _package(tmp_path / "src")
+ (package / "logo.png").write_bytes(PNG)
+ manifest = tree.build(package)
+ tree.stage(package, manifest)
+
+ destination = tmp_path / "out" / "pdf"
+ tree.materialize_creation(manifest, _components(description="Merge and split."),
+ "pdf", destination)
+
+ assert (destination / "logo.png").read_bytes() == PNG
+ assert "description: Merge and split." in (destination / "SKILL.md").read_text()
+
+
+def test_the_approved_revision_is_the_revision_of_the_materialized_tree(tmp_path):
+ """Computed by materializing it. A revision derived some other way would be a second
+ description of the same bytes, and the two would eventually disagree."""
+ package = _package(tmp_path / "src")
+ (package / "logo.png").write_bytes(PNG)
+ manifest = tree.build(package)
+ tree.stage(package, manifest)
+ components = _components()
+
+ revision = tree.revision("pdf", manifest, components)
+
+ destination = tmp_path / "out" / "pdf"
+ tree.materialize_creation(manifest, components, "pdf", destination)
+ assert revision == skill_revision(destination)
+
+
+def test_one_changed_asset_byte_changes_the_revision(tmp_path):
+ """While assets were dropped, two packages differing only in an image hashed identically."""
+ first = _package(tmp_path / "one")
+ (first / "logo.png").write_bytes(PNG)
+ second = _package(tmp_path / "two")
+ (second / "logo.png").write_bytes(PNG[:-1] + b"\x00")
+
+ revisions = set()
+ for package in (first, second):
+ manifest = tree.build(package)
+ tree.stage(package, manifest)
+ revisions.add(tree.revision("pdf", manifest, _components()))
+
+ assert len(revisions) == 2
diff --git a/tests/test_ui.py b/tests/test_ui.py
index 3652975..559f6ae 100644
--- a/tests/test_ui.py
+++ b/tests/test_ui.py
@@ -1,12 +1,17 @@
"""Change-control UI guards: key preflight, slug validation, same-origin check, the pending
lifecycle, and the history/rollback surface."""
+import json
import threading
from html.parser import HTMLParser
import pytest
from fastapi.testclient import TestClient
-from optimize import promote as P
+from ingot.mcp_server import registry
+from ingot.optimize import harbor_report
+from ingot.optimize import promote as P
+from ingot.optimize import publication as Q
+from ui import app as A
from ui.app import app
@@ -45,8 +50,6 @@ def client(tmp_path, monkeypatch):
monkeypatch.setattr(auth, "AUTH_FILE", tmp_path / "no-auth.json") # auth off unless a test opts in
monkeypatch.delenv("AUTH_USER", raising=False)
monkeypatch.delenv("AUTH_PASSWORD", raising=False)
- monkeypatch.setattr(P, "PENDING_DIR", tmp_path / "pending")
- monkeypatch.setattr(P, "REVISIONS_DIR", tmp_path / "revisions")
monkeypatch.setenv("OPENROUTER_API_KEY", "sk-or-test")
return TestClient(app)
@@ -70,6 +73,323 @@ def test_optimize_without_task_set_is_404(client):
assert r.status_code == 404
+def _eval_skill(tmp_path, monkeypatch, name="mounted-skill"):
+ import ui.app as U
+ from ingot.mcp_server.registry import write_skill_md
+
+ root = tmp_path / "read-only-library"
+ skill = root / name
+ skill.mkdir(parents=True)
+ write_skill_md(skill / "SKILL.md", {"name": name, "description": "Mounted description."},
+ "Mounted body.")
+ tasks = tmp_path / "tasks"
+ tasks.mkdir()
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root))
+ monkeypatch.setattr(U, "TASKS_DIR", tasks)
+ U.RUNS.clear()
+ return tasks
+
+
+def _run_threads_inline(monkeypatch):
+ import ui.app as U
+
+ class InlineThread:
+ def __init__(self, target, args=(), **_kwargs):
+ self.target, self.args = target, args
+
+ def start(self):
+ self.target(*self.args)
+
+ monkeypatch.setattr(U.threading, "Thread", InlineThread)
+
+
+def test_create_eval_set_reads_an_indexed_skill_and_persists_the_draft(
+ client, tmp_path, monkeypatch):
+ tasks = _eval_skill(tmp_path, monkeypatch)
+ _run_threads_inline(monkeypatch)
+
+ def draft(name, description, body, tasks_dir, log):
+ assert (name, description, body) == (
+ "mounted-skill", "Mounted description.", "Mounted body.")
+ assert tasks_dir.parent == tasks
+ (tasks_dir / f"{name}.yaml").write_text(
+ "skill: mounted-skill\ntrain:\n- task: train\nholdout:\n- task: holdout\n")
+ log("[draft] wrote eval set")
+ return tasks_dir / f"{name}.yaml"
+
+ monkeypatch.setattr("ingot.optimize.draft.draft_and_save", draft)
+ response = client.post("/api/evals/mounted-skill")
+
+ assert response.status_code == 200
+ assert response.json() == {"started": "mounted-skill"}
+ assert (tasks / "mounted-skill.yaml").exists()
+ assert client.get("/api/runs").json()["mounted-skill"] == {
+ "status": "done", "action": "eval", "log": ["[draft] wrote eval set"]}
+
+
+def test_create_eval_set_refuses_to_overwrite_an_existing_set(client, tmp_path, monkeypatch):
+ tasks = _eval_skill(tmp_path, monkeypatch)
+ (tasks / "mounted-skill.yaml").write_text("keep: me\n")
+
+ response = client.post("/api/evals/mounted-skill")
+
+ assert response.status_code == 409
+ assert "already has" in response.json()["detail"]
+ assert (tasks / "mounted-skill.yaml").read_text() == "keep: me\n"
+
+
+def test_create_eval_set_preserves_a_file_created_while_drafting(
+ client, tmp_path, monkeypatch):
+ tasks = _eval_skill(tmp_path, monkeypatch)
+ _run_threads_inline(monkeypatch)
+
+ def racing_draft(name, _description, _body, tasks_dir, log):
+ staged = tasks_dir / f"{name}.yaml"
+ staged.write_text("draft: mine\n")
+ (tasks / f"{name}.yaml").write_text("draft: theirs\n")
+ return staged
+
+ monkeypatch.setattr("ingot.optimize.draft.draft_and_save", racing_draft)
+
+ assert client.post("/api/evals/mounted-skill").status_code == 200
+ run = client.get("/api/runs").json()["mounted-skill"]
+ assert run["status"] == "error"
+ assert run["log"] == ["ERROR: 'mounted-skill' already has an eval task set"]
+ assert (tasks / "mounted-skill.yaml").read_text() == "draft: theirs\n"
+
+
+def test_create_eval_set_shares_the_paid_run_lock(client, tmp_path, monkeypatch):
+ _eval_skill(tmp_path, monkeypatch)
+ import ui.app as U
+ U.RUNS["other-skill"] = {"status": "running", "action": "optimize", "log": []}
+
+ response = client.post("/api/evals/mounted-skill")
+
+ assert response.status_code == 409
+ assert "already in progress" in response.json()["detail"]
+
+
+def test_create_eval_set_surfaces_background_errors(client, tmp_path, monkeypatch):
+ tasks = _eval_skill(tmp_path, monkeypatch)
+ _run_threads_inline(monkeypatch)
+
+ def partial_draft(name, _description, _body, tasks_dir, log):
+ (tasks_dir / f"{name}.yaml").write_text("partial: true\n")
+ raise RuntimeError("teacher down")
+
+ monkeypatch.setattr("ingot.optimize.draft.draft_and_save", partial_draft)
+
+ assert client.post("/api/evals/mounted-skill").status_code == 200
+ run = client.get("/api/runs").json()["mounted-skill"]
+ assert run["status"] == "error"
+ assert run["action"] == "eval"
+ assert run["log"] == ["ERROR: teacher down"]
+ assert not (tasks / "mounted-skill.yaml").exists()
+
+
+def test_create_eval_set_requires_an_indexed_skill_and_provider(client, tmp_path, monkeypatch):
+ import ui.app as U
+ U.RUNS.clear()
+ assert client.post("/api/evals/not-indexed").status_code == 404
+
+ _eval_skill(tmp_path, monkeypatch)
+ monkeypatch.delenv("OPENROUTER_API_KEY")
+ response = client.post("/api/evals/mounted-skill")
+ assert response.status_code == 400
+ assert "API_KEY" in response.json()["detail"]
+
+
+def test_eval_set_api_exposes_the_measured_inputs(client, tmp_path, monkeypatch):
+ tasks = _eval_skill(tmp_path, monkeypatch)
+ (tasks / "mounted-skill.yaml").write_text(
+ "skill: mounted-skill\n"
+ "train:\n"
+ "- task: Write the train answer.\n"
+ " rubric: Include the train result.\n"
+ " checklist:\n"
+ " - id: train_result\n"
+ " criterion: Includes the train result.\n"
+ " weight: 3\n"
+ " dimension: correctness\n"
+ "holdout:\n"
+ "- task: Write the held-out answer.\n"
+ " rubric: Include the held-out result.\n"
+ "routing:\n"
+ "- task: Use the mounted skill.\n"
+ " expected: mounted-skill\n"
+ "acceptance:\n"
+ "- id: no_placeholder\n"
+ " forbid: TODO\n"
+ " description: No placeholder output.\n")
+
+ response = client.get("/api/evals/mounted-skill")
+
+ assert response.status_code == 200
+ assert response.json() == {
+ "skill": "mounted-skill",
+ "train": [{
+ "task": "Write the train answer.",
+ "rubric": "Include the train result.",
+ "checklist": [{
+ "id": "train_result",
+ "criterion": "Includes the train result.",
+ "weight": 3,
+ "dimension": "correctness",
+ }],
+ }],
+ "holdout": [{
+ "task": "Write the held-out answer.",
+ "rubric": "Include the held-out result.",
+ }],
+ "routing": [{
+ "task": "Use the mounted skill.",
+ "expected": "mounted-skill",
+ }],
+ "acceptance": [{
+ "id": "no_placeholder",
+ "forbid": "TODO",
+ "description": "No placeholder output.",
+ }],
+ "leakage": False,
+ "counts": {"train": 1, "holdout": 1, "routing": 1, "acceptance": 1, "checks": 1},
+ }
+
+
+def test_eval_set_api_marks_legacy_flat_tasks_as_leaky(client, tmp_path, monkeypatch):
+ tasks = _eval_skill(tmp_path, monkeypatch)
+ (tasks / "mounted-skill.yaml").write_text(
+ "tasks:\n- task: Same task trains and gates.\n"
+ " checklist:\n - criterion: Produce the requested result.\n")
+
+ response = client.get("/api/evals/mounted-skill")
+
+ assert response.status_code == 200
+ assert response.json()["leakage"] is True
+ assert response.json()["train"] == response.json()["holdout"]
+ assert response.json()["counts"]["checks"] == 1
+
+
+def test_eval_set_api_reports_missing_and_malformed_files(client, tmp_path, monkeypatch):
+ tasks = _eval_skill(tmp_path, monkeypatch)
+ assert client.get("/api/evals/mounted-skill").status_code == 404
+ (tasks / "mounted-skill.yaml").write_text("train: [")
+ response = client.get("/api/evals/mounted-skill")
+ assert response.status_code == 503
+ assert "eval set is unreadable" in response.json()["detail"]
+
+
+def test_review_api_runs_existing_reviewer_and_serves_persisted_result(
+ client, tmp_path, monkeypatch):
+ tasks = _eval_skill(tmp_path, monkeypatch)
+ (tasks / "mounted-skill.yaml").write_text(
+ "train:\n- task: train\nholdout:\n- task: holdout\n")
+ _run_threads_inline(monkeypatch)
+ import ingot.optimize.review as review_module
+ review_dir = tmp_path / "reviews"
+ monkeypatch.setattr(review_module, "REVIEW_DIR", review_dir)
+
+ def review(skill, log):
+ assert skill == "mounted-skill"
+ result = {
+ "skill": skill, "model": "test-model", "revision": "abc123",
+ "score": 0.75, "tasks": 2, "checks": 4, "failed_checks": 1,
+ "by_dimension": {"correctness": 1.0},
+ "findings": [{"task": "holdout", "check": "answer", "criterion": "Answer it.",
+ "weight": 2, "dimension": "correctness", "value": 0.5,
+ "note": "Missing detail.", "cost": 1.0}],
+ "per_task": [{"task": "train", "score": 1.0}, {"task": "holdout", "score": 0.5}],
+ }
+ review_dir.mkdir()
+ (review_dir / f"{skill}.json").write_text(__import__("json").dumps(result))
+ log("[review] complete")
+ return result
+
+ monkeypatch.setattr(review_module, "run_review", review)
+
+ response = client.post("/api/reviews/mounted-skill")
+
+ assert response.status_code == 200
+ assert response.json() == {"started": "mounted-skill"}
+ assert client.get("/api/runs").json()["mounted-skill"] == {
+ "status": "done", "action": "review", "log": ["[review] complete"]}
+ saved = client.get("/api/reviews/mounted-skill")
+ assert saved.status_code == 200
+ assert saved.json()["score"] == 0.75
+ assert saved.json()["failed_checks"] == 1
+ assert saved.json()["created"] > 0
+
+ persisted = __import__("json").loads(
+ (review_dir / "mounted-skill.json").read_text())
+ persisted["created"] = 123
+ (review_dir / "mounted-skill.json").write_text(__import__("json").dumps(persisted))
+ assert client.get("/api/reviews/mounted-skill").json()["created"] == 123
+
+
+@pytest.mark.parametrize(("recorded", "message"), [
+ (None, "did not record a skill revision"),
+ ("old-revision", "active skill changed since this review ran"),
+])
+def test_review_api_marks_unbound_or_changed_results_stale(
+ client, tmp_path, monkeypatch, recorded, message):
+ _eval_skill(tmp_path, monkeypatch)
+ import ingot.optimize.review as review_module
+ review_dir = tmp_path / "reviews"
+ review_dir.mkdir()
+ monkeypatch.setattr(review_module, "REVIEW_DIR", review_dir)
+ (review_dir / "mounted-skill.json").write_text(json.dumps({
+ "skill": "mounted-skill", "revision": recorded, "score": 1.0,
+ }))
+
+ result = client.get("/api/reviews/mounted-skill")
+
+ assert result.status_code == 200
+ assert message in result.json()["stale"]
+
+
+def test_review_api_accepts_the_revision_recorded_by_the_review_path(
+ client, tmp_path, monkeypatch):
+ from ingot.mcp_server.registry import skill_revision
+ _eval_skill(tmp_path, monkeypatch)
+ import ingot.optimize.review as review_module
+ review_dir = tmp_path / "reviews"
+ review_dir.mkdir()
+ monkeypatch.setattr(review_module, "REVIEW_DIR", review_dir)
+ skill = tmp_path / "read-only-library" / "mounted-skill"
+ (review_dir / "mounted-skill.json").write_text(json.dumps({
+ "skill": "mounted-skill", "revision": skill_revision(skill), "score": 1.0,
+ }))
+
+ result = client.get("/api/reviews/mounted-skill")
+
+ assert result.status_code == 200
+ assert result.json()["stale"] is None
+
+
+def test_review_api_requires_evals_and_shares_the_paid_run_lock(
+ client, tmp_path, monkeypatch):
+ tasks = _eval_skill(tmp_path, monkeypatch)
+ assert client.post("/api/reviews/mounted-skill").status_code == 404
+ (tasks / "mounted-skill.yaml").write_text("train:\n- task: train\n")
+ import ui.app as U
+ U.RUNS["other-skill"] = {"status": "running", "action": "optimize", "log": []}
+ response = client.post("/api/reviews/mounted-skill")
+ assert response.status_code == 409
+ assert "already in progress" in response.json()["detail"]
+
+
+def test_review_result_reports_missing_and_malformed_files(client, tmp_path, monkeypatch):
+ _eval_skill(tmp_path, monkeypatch)
+ import ingot.optimize.review as review_module
+ review_dir = tmp_path / "reviews"
+ review_dir.mkdir()
+ monkeypatch.setattr(review_module, "REVIEW_DIR", review_dir)
+ assert client.get("/api/reviews/mounted-skill").status_code == 404
+ (review_dir / "mounted-skill.json").write_text("{")
+ response = client.get("/api/reviews/mounted-skill")
+ assert response.status_code == 503
+ assert "review result is unreadable" in response.json()["detail"]
+
+
def test_cross_origin_post_refused(client):
r = client.post("/api/optimize/pdf", headers={"origin": "http://evil.example"})
assert r.status_code == 403
@@ -85,6 +405,45 @@ def test_pending_unknown_skill_is_404(client):
assert client.get("/api/pending/pdf").status_code == 404
+def test_pending_creation_appears_as_to_be_added_and_can_be_reviewed(client, tmp_path,
+ monkeypatch):
+ import ui.app as U
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(tmp_path / "empty-library"))
+ components = {"description": "Write conversion copy.", "body": "Body"}
+ revision = registry.skill_revision(registry.writable_skill_dir("copywriting"), components)
+ P.save_pending("copywriting", {
+ "skill": "copywriting", "kind": "creation", "created": 7,
+ "champion_components": {},
+ "challenger_components": components,
+ "changed_components": ["description", "body"],
+ "gate": {"promotable": True, "blocked": [], "kind": "new_skill_admission"},
+ "evidence": {"challenger": {"revision": revision}},
+ "creation": {"summary": "Add vetted copywriting skill."},
+ })
+ U._SKILLS_CACHE = None
+ U._SKILLS_CACHE_KEY = None
+
+ skills = client.get("/api/skills").json()
+ assert skills == [{"name": "copywriting", "description": "Write conversion copy.",
+ "has_tasks": False, "pending": True, "revision": "",
+ "uses": 0, "publishing": False, "provenance": "proposed",
+ "status": None, "active": False}]
+ pending = client.get("/api/pending/copywriting")
+ assert pending.status_code == 200
+ assert pending.json()["kind"] == "creation"
+ assert pending.json()["stale"] is None
+
+ html = client.get("/").text
+ assert "to be added" in html
+ assert "Review addition" in html
+ assert "function creationEvidence(" in html
+ assert 'p.creation ? "Approve & add"' in html
+ assert '["proposed", "To be added"' in html
+ assert 'p.creation ? "Confirm addition, "' in html
+ assert "Submission requirements met; evidence is operator-supplied" in html
+ assert "const activeSkills = skills.filter(skill => skill.active !== false)" in html
+
+
def test_promote_without_pending_is_404(client):
assert client.post("/api/promote/pdf").status_code == 404
@@ -194,6 +553,15 @@ def test_compose_mcp_service_mounts_the_runs_directory():
assert "./runs:/app/runs" in compose["services"]["mcp"]["volumes"]
+def test_compose_mcp_and_ui_share_the_pending_queue_uid():
+ """MCP creation proposals and UI reviews share mode-0600 files in runs/pending."""
+ import yaml
+ from pathlib import Path
+ compose = yaml.safe_load((Path(__file__).resolve().parents[1] / "docker-compose.yml").read_text())
+
+ assert compose["services"]["mcp"]["user"] == compose["services"]["ui"]["user"]
+
+
def test_skills_list_empty_library(client):
r = client.get("/api/skills")
assert r.status_code == 200
@@ -252,7 +620,7 @@ def test_skill_version_explorer_reads_active_pending_and_snapshot(client, tmp_pa
"body": "active body"},
"challenger_components": pending})
- snapshot = P.REVISIONS_DIR / "pdf" / "abc123"
+ snapshot = P.revisions_dir() / "pdf" / "abc123"
snapshot.mkdir(parents=True)
(snapshot / "SKILL.md").write_text(
"---\nname: pdf\ndescription: Snapshot description.\n---\nsnapshot body\n")
@@ -334,6 +702,75 @@ def test_skill_list_ships_search_filters_version_explorer_and_live_updates(clien
assert "renderSkills(skills, runs || runInventory, hist)" in html
+def test_skill_families_filter_the_skill_list_and_category_atlas(client):
+ html = client.get("/").text
+ layout = _Layout(html)
+
+ assert "skill-family-filter" in layout.ancestors
+ assert 'aria-label="Filter atlas by skill family"' in html
+ assert "All families" in html
+ assert "function selectFamily(" in html
+ assert "function familyMatches(" in html
+ assert "familyMatches(s.name)" in html
+ assert "d.clusters.map((cluster, index) => ({cluster, index}))" in html
+ assert '$("#cluster-chips").innerHTML = "";' in html
+ assert "clustersAttempted" in html
+ assert "select.disabled = !clusterData?.clusters?.length" in html
+ assert '$("#nav-clusters").textContent = "0";' in html
+ assert '$("#skill-family-filter").onchange' in html
+
+
+def test_trace_inventory_api_never_returns_answers(client, tmp_path, monkeypatch):
+ import ui.app as ui_app
+
+ store = tmp_path / "local-traces.json"
+ store.write_text(json.dumps({
+ "schema_version": "ingot/local-traces/v1",
+ "generated_at": 1785254400,
+ "traces": [{
+ "id": "trace-1", "timestamp": "2026-07-28T10:00:00Z", "harness": "claude",
+ "task": "Review the interface", "answer": "private answer",
+ "skills": [{"name": "saas-interface-review", "revision": None}],
+ "tags": ["skill:saas-interface-review"], "usage": {"input_tokens": 10,
+ "output_tokens": 4},
+ }],
+ }))
+ monkeypatch.setattr(ui_app, "LOCAL_TRACE_FILE", store)
+
+ response = client.get("/api/traces")
+
+ assert response.status_code == 200
+ payload = response.json()
+ assert payload["total"] == 1
+ assert "task" not in payload["recent"][0]
+ assert "answer" not in payload["recent"][0]
+ assert "private answer" not in response.text
+
+ preview = client.get("/api/traces?include_tasks=true")
+ assert preview.json()["recent"][0]["task"] == "Review the interface"
+ assert "private answer" not in preview.text
+
+ filtered = client.get("/api/traces?project=other&since=2026-07-28")
+ assert filtered.status_code == 200
+ assert filtered.json()["total"] == 0
+ assert client.get("/api/traces?since=not-a-date").status_code == 400
+
+
+def test_trace_inventory_is_a_routed_console_view(client):
+ html = client.get("/").text
+ layout = _Layout(html)
+
+ assert "traces-section" in layout.ancestors
+ assert "trace-summary" in layout.ancestors
+ assert "trace-list" in layout.ancestors
+ assert 'data-route="traces"' in html
+ assert "j(traceUrl())" in html
+ assert 'id="trace-project"' in html
+ assert 'id="trace-since"' in html
+ assert 'id="trace-task-previews"' in html
+ assert 'notation: "compact"' in html
+
+
def test_comparison_panel_orders_tokens_and_tables_numbered_task_scores(client):
html = client.get("/").text
compare = html[html.index("function buildCompare(p)"):html.index("function openCompare()")]
@@ -346,13 +783,14 @@ def test_comparison_panel_orders_tokens_and_tables_numbered_task_scores(client):
assert 'class="cmp-pertask"' not in compare
-def test_api_skills_rows_carry_a_load_count(client, monkeypatch):
+def test_api_skills_rows_carry_a_load_count(client, monkeypatch, tmp_path):
"""Every active skill row exposes `uses` so the UI can render the load-counter chip."""
import ui.app as ui_app
- from mcp_server import usage_counts
+ from ingot.mcp_server import usage_counts
class _Skill:
name, description, revision = "pdf", "merge PDFs", "rev1"
+ root = str(tmp_path / "library" / "pdf") # provenance classifies from the skill's own root
monkeypatch.setattr(ui_app, "load_skills", lambda: [_Skill()])
monkeypatch.setattr(usage_counts, "load_counts", lambda: {"pdf": 7})
active = client.get("/api/skills").json()
@@ -368,6 +806,36 @@ def test_pending_without_search_scores_still_renders(client):
assert client.get("/api/pending/pdf").json()["inner_loop"] is None
+def test_pending_exposes_retrospective_evidence(client):
+ P.save_pending("pdf", {
+ "skill": "pdf", "kind": "retrospective",
+ "champion_components": {"description": "d", "body": "a"},
+ "challenger_components": {"description": "d", "body": "b"},
+ "changed_components": ["body"],
+ "retrospective": {
+ "summary": "Repeated omission.", "trigger": "Two matching failures.",
+ "minimal_content": "Add the missing guard.", "producer": "skill-retrospective",
+ "caller": "build-loop", "evidence": ["run one", "run two"],
+ "pressure_scenario": "A rushed repair.", "risk": "May slow simple work.",
+ "verification": {"status": "passed", "command": "pytest", "result": "passed"},
+ },
+ })
+
+ payload = client.get("/api/pending/pdf").json()
+ assert payload["kind"] == "retrospective"
+ assert payload["retrospective"]["producer"] == "skill-retrospective"
+ assert payload["retrospective"]["verification"]["status"] == "passed"
+
+
+def test_index_renders_retrospective_evidence_on_both_decision_surfaces(client):
+ html = client.get("/").text
+ assert "function retrospectiveEvidence(" in html
+ assert "p.retrospective" in html
+ assert "Pressure scenario" in html
+ assert "Verification" in html
+ assert "border: 1px solid transparent; overflow: hidden;" in html
+
+
def test_promote_passes_through_result(client, monkeypatch):
import ui.app as ui_app
P.save_pending("pdf", {"skill": "pdf", "gate": {"promotable": True, "blocked": []},
@@ -377,6 +845,170 @@ def test_promote_passes_through_result(client, monkeypatch):
assert r.status_code == 200 and r.json() == {"result": "promoted 'pdf'"}
+def test_pending_exposes_approved_publication_and_disables_repeat_approval(client):
+ pending = {
+ "skill": "pdf", "gate": {"promotable": True, "blocked": []},
+ "champion_components": {"description": "Merge PDFs.", "body": "old"},
+ "challenger_components": {"description": "Merge PDFs.", "body": "new"},
+ "evidence": {"champion": {"revision": "a" * 64},
+ "challenger": {"revision": "b" * 64}},
+ }
+ P.save_pending("pdf", pending)
+ Q.queue_publication("pdf", pending, "admin", "promote")
+
+ payload = client.get("/api/pending/pdf").json()
+ assert payload["publication"]["state"] == "approved_publishing"
+ html = client.get("/").text
+ assert "Approved · publishing to vault" in html
+ assert "p.publication" in html
+ assert '$("#approve").disabled' in html
+
+
+def test_pending_does_not_attach_a_receipt_from_an_older_proposal(client):
+ old = {
+ "skill": "pdf", "kind": "retrospective",
+ "champion_components": {"description": "Merge PDFs.", "body": "old"},
+ "challenger_components": {"description": "Merge PDFs.", "body": "first change"},
+ "evidence": {"champion": {"revision": "a" * 64},
+ "challenger": {"revision": "b" * 64}},
+ "retrospective": {"proposal_id": "old-proposal"},
+ }
+ receipt = Q.queue_publication("pdf", old, "admin", "promote")
+ Q.update_publication(receipt.id, state="active")
+ P.save_pending("pdf", {
+ **old,
+ "challenger_components": {"description": "Merge PDFs.", "body": "second change"},
+ "evidence": {"champion": {"revision": "b" * 64},
+ "challenger": {"revision": "c" * 64}},
+ "retrospective": {"proposal_id": "new-proposal"},
+ })
+
+ payload = client.get("/api/pending/pdf").json()
+
+ assert payload["publication"] is None
+
+
+def _queued(skill: str, **changes):
+ pending = {
+ "skill": skill, "gate": {"promotable": True, "blocked": []},
+ "champion_components": {"description": "Merge PDFs.", "body": "old"},
+ "challenger_components": {"description": "Merge PDFs.", "body": "new"},
+ "evidence": {"champion": {"revision": "a" * 64},
+ "challenger": {"revision": "b" * 64}},
+ }
+ receipt = Q.queue_publication(skill, pending, "admin", "promote")
+ return Q.update_publication(receipt.id, **changes) if changes else receipt
+
+
+def _proposed(monkeypatch, tmp_path, skill="copywriting"):
+ """A creation pending, which is what the board surfaces when no library is indexed."""
+ import ui.app as U
+ monkeypatch.setenv("SKILL_ROUTER_PATHS", str(tmp_path / "empty-library"))
+ P.save_pending(skill, {
+ "skill": skill, "kind": "creation",
+ "champion_components": {"description": "", "body": ""},
+ "challenger_components": {"description": "Write conversion copy.", "body": "Body"},
+ "gate": {"promotable": True, "blocked": [], "kind": "new_skill_admission"},
+ "creation": {"summary": "Add vetted copywriting skill."},
+ })
+ U._SKILLS_CACHE = None
+ U._SKILLS_CACHE_KEY = None
+
+
+def test_skills_listing_separates_publishing_from_awaiting_review(client, tmp_path, monkeypatch):
+ """Approval does not free the review slot, so an approved change stays `pending`. Without a
+ second flag the board counts it as still awaiting a decision the reviewer already made — which
+ is what made an approved creation sit in `To be added` looking untouched."""
+ _proposed(monkeypatch, tmp_path)
+ before = {s["name"]: s for s in client.get("/api/skills").json()}["copywriting"]
+ _queued("copywriting", state="awaiting_merge", pr=9)
+
+ after = {s["name"]: s for s in client.get("/api/skills").json()}["copywriting"]
+
+ assert (before["pending"], before["publishing"]) == (True, False)
+ assert (after["pending"], after["publishing"]) == (True, True)
+
+
+def test_a_finished_publication_stops_marking_its_skill_as_publishing(client, tmp_path, monkeypatch):
+ """Only the newest receipt counts, or an earlier attempt would pin the skill to `publishing`."""
+ _proposed(monkeypatch, tmp_path)
+ _queued("copywriting", state="active", pr=9)
+
+ listed = {s["name"]: s for s in client.get("/api/skills").json()}["copywriting"]
+
+ assert listed["publishing"] is False
+
+
+def test_publications_lane_is_empty_before_anything_is_approved(client):
+ payload = client.get("/api/publications").json()
+
+ assert payload["publications"] == []
+ assert payload["unreadable"] is None
+ assert "awaiting_merge" in payload["live_states"]
+
+
+def test_publications_lane_survives_the_pending_record_it_came_from(client, monkeypatch):
+ """The whole point of the lane: `publication_for_skill` needs a pending record, and approval
+ consumes it, so once a change is approved the console could no longer see it travelling."""
+ monkeypatch.setenv("INGOT_FORGE_REPOSITORY", "someone/skills")
+ _queued("pdf", state="awaiting_merge", pr=9)
+
+ payload = client.get("/api/publications").json()
+
+ assert [r["skill"] for r in payload["publications"]] == ["pdf"]
+ assert payload["publications"][0]["state"] == "awaiting_merge"
+ assert payload["publications"][0]["pr_url"] == "https://github.com/someone/skills/pull/9"
+
+
+def test_publications_lane_omits_a_pull_request_url_when_there_is_no_pull_request(client):
+ _queued("pdf")
+
+ assert client.get("/api/publications").json()["publications"][0]["pr_url"] is None
+
+
+def test_publications_lane_links_no_pull_request_under_the_local_backend(client, monkeypatch):
+ """Only the forge backend has a pull request, and only it knows the repository. Linking to a
+ hardcoded one produced a 404 for every deployment that was not the author's."""
+ monkeypatch.delenv("INGOT_FORGE_REPOSITORY", raising=False)
+ _queued("pdf", state="awaiting_merge", pr=9)
+
+ assert client.get("/api/publications").json()["publications"][0]["pr_url"] is None
+
+
+def test_publications_lane_reports_a_receipt_store_it_cannot_read(client, monkeypatch):
+ """`Path.glob` swallows `PermissionError`, so an unreadable store returns an empty list —
+ indistinguishable from a quiet lane, which is the reading a stalled publisher most invites."""
+ _queued("pdf", state="awaiting_merge", pr=9)
+ monkeypatch.setattr(A.os, "access", lambda *a, **k: False)
+
+ payload = client.get("/api/publications").json()
+
+ assert payload["publications"] == []
+ assert "cannot read the receipt store" in payload["unreadable"]
+
+
+def test_pending_names_the_pull_request_a_stalled_publication_waits_on(client):
+ """A vault that cannot auto-merge leaves the receipt waiting on a person. Without the pull
+ request number on the card, the reviewer has no way to learn they are what it waits for."""
+ pending = {
+ "skill": "pdf", "gate": {"promotable": True, "blocked": []},
+ "champion_components": {"description": "Merge PDFs.", "body": "old"},
+ "challenger_components": {"description": "Merge PDFs.", "body": "new"},
+ "evidence": {"champion": {"revision": "a" * 64},
+ "challenger": {"revision": "b" * 64}},
+ }
+ P.save_pending("pdf", pending)
+ receipt = Q.queue_publication("pdf", pending, "admin", "promote")
+ Q.update_publication(receipt.id, state="awaiting_merge", pr=4, auto_merge=False,
+ note="auto-merge unavailable, waiting on a human merge: denied")
+
+ payload = client.get("/api/pending/pdf").json()
+
+ assert payload["publication"]["pr"] == 4
+ assert "waiting on a human merge" in payload["publication"]["note"]
+ assert "merge vault PR #" in client.get("/").text
+
+
def test_cross_origin_promote_and_reject_refused(client):
for endpoint in ("/api/promote/pdf", "/api/reject/pdf"):
assert client.post(endpoint, headers={"origin": "http://evil.example"}).status_code == 403
@@ -410,7 +1042,7 @@ def test_pending_routing_pass_renders_without_ab(client):
def test_optimize_surfaces_pin_conflicts_as_400(client, monkeypatch):
- import optimize
+ import ingot.optimize as optimize
def conflict():
raise SystemExit("error: provider pin conflicts detected before spending any tokens:\n MODEL=x: nope")
monkeypatch.setattr(optimize, "preflight_provider_pins", conflict)
@@ -420,11 +1052,11 @@ def conflict():
def test_skills_api_reports_eval_status_for_all_skills(client, tmp_path, monkeypatch):
# the UI's evals chip keys off has_tasks, every skill must carry it, task set or not
- from mcp_server import registry
- from mcp_server.registry import write_skill_md
+ from ingot.mcp_server import registry
+ from ingot.mcp_server.registry import write_skill_md
import ui.app as U
for name in ("with-evals", "without-evals"):
- d = registry.SKILLS_DIR / name # hermetic per-test root (conftest)
+ d = registry.library_dir() / name # hermetic per-test root (conftest)
d.mkdir(parents=True)
write_skill_md(d / "SKILL.md", {"name": name, "description": "d"}, "b")
tasks = tmp_path / "tasks"
@@ -435,12 +1067,50 @@ def test_skills_api_reports_eval_status_for_all_skills(client, tmp_path, monkeyp
assert flags == {"with-evals": True, "without-evals": False}
-def test_index_ships_eval_chips_and_disabled_candidate_run(client):
+def test_index_ships_eval_creation_for_skills_without_tasks(client):
+ html = client.get("/").text
+ assert "no evals" in html
+ assert "has_tasks" in html
+ assert "Create eval set" in html
+ assert "createEvalSet" in html and "/api/evals/" in html
+ assert "auto-drafts" not in html
+ assert "Optimize with SkillOpt" in html
+
+
+def test_index_surfaces_incomplete_eval_coverage_not_only_zero_coverage(client):
html = client.get("/").text
- assert "no evals" in html # chip for skills without an eval task set
- assert "has_tasks" in html # rendering keys off the API flag
- assert "auto-drafts" in html # the disabled generate button explains how to get evals
- assert "Optimize with SkillOpt" in html # optimization is a first-class, human-gated workflow
+ assert "EVAL COVERAGE" in html
+ assert "without an eval task set" in html
+ assert "withTasks.length < activeSkills.length" in html
+
+
+def test_index_labels_eval_drafting_separately_from_optimization(client):
+ html = client.get("/").text
+ assert 'run?.action === "eval"' in html
+ assert 'const runLabel = drafting ? "Eval draft" : reviewing ? "Current-skill review" : "SkillOpt"' in html
+ assert "${runLabel} ${esc(run.status" in html
+ assert "Drafting eight train/holdout tasks" in html
+
+
+def test_index_exposes_eval_inputs_and_current_skill_review(client):
+ html = client.get("/").text
+ layout = _Layout(html)
+ for element_id in ("skill-evals", "skill-eval-summary", "skill-eval-groups",
+ "skill-review", "skill-review-run", "skill-eval-msg"):
+ assert "skill-overlay" in layout.ancestors[element_id]
+ assert "/api/evals/" in html
+ assert "/api/reviews/" in html
+ assert "Run review" in html
+ assert "Train tasks" in html and "Held-out tasks" in html
+ assert "Routing cases" in html and "Acceptance rules" in html
+ assert "Failed checks" in html
+
+
+def test_index_labels_review_runs_separately_from_drafting_and_optimization(client):
+ html = client.get("/").text
+ assert 'run?.action === "review"' in html
+ assert '"Current-skill review"' in html
+ assert "Review can take a minute" in html
def test_index_leads_with_review_before_candidate_generation(client):
@@ -449,7 +1119,7 @@ def test_index_leads_with_review_before_candidate_generation(client):
assert html.index('id="review-section"') < html.index('id="history-section"')
assert html.index('id="history-section"') < html.index('id="skills"')
assert 'id="run-section"' not in html
- assert "Evidence-gated change control" in html
+ assert "Release control for agent skills" in html
assert "change control" in html and "skill optimizer" not in html
@@ -481,15 +1151,16 @@ def test_carn_viewer_is_gone(client):
def _promoted_skill(tmp_path, monkeypatch):
"""An active skill with one approved promotion behind it, so a snapshot exists to restore."""
- from mcp_server.registry import optimizable_components, skill_revision
+ from ingot.mcp_server.registry import optimizable_components, skill_revision
root = tmp_path / "skills"
skill = root / "pdf"
skill.mkdir(parents=True)
(skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\napproved body\n")
+ monkeypatch.setenv("INGOT_LIBRARY", str(root))
monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root))
champion = optimizable_components(skill)
challenger = {**champion, "body": "promoted body"}
- from mcp_server.registry import load_skills
+ from ingot.mcp_server.registry import load_skills
current = load_skills(root)[0]
gate = {"promotable": True, "blocked": []}
P.save_pending("pdf", {
@@ -499,7 +1170,7 @@ def _promoted_skill(tmp_path, monkeypatch):
"challenger": {"revision": skill_revision(skill, challenger)},
"gate": gate},
})
- P.approve_pending("pdf")
+ P._activate_approved("pdf", P.load_pending("pdf"))
return skill, current.revision
@@ -526,7 +1197,7 @@ def test_history_does_not_rescan_the_skill_library(client, tmp_path, monkeypatch
The counter patches the registry's own library scan, which `load_skills` looks up at call time:
counting `ui.app.load_skills` would have missed a rescan reached through any other module's
import of it, and passed whether or not history scanned anything."""
- from mcp_server import registry
+ from ingot.mcp_server import registry
_promoted_skill(tmp_path, monkeypatch)
real_sources = registry.skill_sources
scans = []
@@ -583,16 +1254,20 @@ def fail(name):
assert [r["action"] for r in history["audit"]["records"]] == ["approve"]
-def test_rollback_restores_a_snapshot_and_records_it(client, tmp_path, monkeypatch):
+def test_rollback_queues_a_snapshot_for_vault_publication(client, tmp_path, monkeypatch):
+ """History rollback takes the same Git lane as approval: it reports publication, and the
+ served skill only changes once the vault merge lands."""
skill, replaced = _promoted_skill(tmp_path, monkeypatch)
assert "promoted body" in (skill / "SKILL.md").read_text()
r = client.post(f"/api/rollback/pdf/{replaced}")
- assert r.status_code == 200 and "Rolled back" in r.json()["result"]
- assert "approved body" in (skill / "SKILL.md").read_text()
+ assert r.status_code == 200 and "publishing to vault" in r.json()["result"]
+ assert "promoted body" in (skill / "SKILL.md").read_text()
+ record = Q.publication_for_skill("pdf")
+ assert record["action"] == "rollback" and record["candidate_revision"] == replaced
trail = client.get("/api/history").json()["audit"]["records"]
- assert [a["action"] for a in trail] == ["rollback", "approve"]
+ assert [a["action"] for a in trail] == ["approve"]
def test_rollback_rejects_unknown_revision_and_bad_names(client, tmp_path, monkeypatch):
@@ -603,7 +1278,7 @@ def test_rollback_rejects_unknown_revision_and_bad_names(client, tmp_path, monke
def test_rollback_refuses_a_traversing_revision_at_the_application(client, tmp_path, monkeypatch):
"""A `..` segment must be refused by revision validation, not merely missed by the router:
- the same string reaching optimize.promote directly has to be rejected there too."""
+ the same string reaching ingot.optimize.promote directly has to be rejected there too."""
_promoted_skill(tmp_path, monkeypatch)
r = client.post("/api/rollback/pdf/%2E%2E", follow_redirects=False)
@@ -683,7 +1358,7 @@ def slow_approve(skill, actor="?"):
def _stale_pending(tmp_path, monkeypatch):
"""A review slot whose champion has since been edited on disk."""
- from mcp_server.registry import load_skills, optimizable_components, skill_revision
+ from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision
root = tmp_path / "skills"
skill = root / "pdf"
skill.mkdir(parents=True)
@@ -722,7 +1397,7 @@ def test_promote_with_stale_evidence_is_409(client, tmp_path, monkeypatch):
def test_pending_is_not_stale_for_a_fresh_change(client, tmp_path, monkeypatch):
- from mcp_server.registry import load_skills, optimizable_components, skill_revision
+ from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision
root = tmp_path / "skills"
skill = root / "pdf"
skill.mkdir(parents=True)
@@ -749,8 +1424,8 @@ def _evidence_bundle(monkeypatch, tmp_path, recorded=None, body="# Behavioral Sk
bundle = evidence_root / "pdf" / "1700000000"
bundle.mkdir(parents=True)
(bundle / "EVIDENCE.md").write_text(body)
- monkeypatch.setattr(U, "REPO_ROOT", tmp_path.resolve())
- monkeypatch.setattr(U, "EVIDENCE_DIR", evidence_root)
+ monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs"))
+ monkeypatch.setattr(U, "STATE_ROOT", tmp_path.resolve())
P.save_pending("pdf", {
"skill": "pdf", "champion_components": {}, "challenger_components": {},
"evidence_paths": {"markdown": recorded or "runs/evidence/pdf/1700000000/EVIDENCE.md"},
@@ -861,12 +1536,18 @@ def test_index_follows_the_queue_when_the_reviewed_card_is_gone(client):
html = client.get("/").text
assert "if (skills) syncReviewCard(skills);" in html
assert "!currentPending || !quarantined.includes(currentPending)" in html
- assert "showPending(quarantined[0], {keepMessage: true});" in html
+ assert "showPending((undecided[0] ?? quarantined[0]), {keepMessage: true});" in html
+ # An approved change keeps its pending record until the vault commit lands, so following the
+ # queue must skip it rather than reopening a card whose decision is already made.
+ assert "skills.filter(s => s.pending && !s.publishing).map(s => s.name)" in html
assert "if (!quarantined.length) { showNoPending(); return; }" in html
# a card opened by hand still clears the previous result
assert "if (!keepMessage) say(\"#pending-msg\", \"\", false);" in html
assert 'onclick="showPending(\'${esc(s.name)}\', {scroll: true})"' in html
- assert 'if (scroll) {' in html and '$("#review-section").scrollIntoView' in html
+ # Review is its own route now, so reaching it is a hash change rather than a scroll within one
+ # long page. The card still has to name the skill whose button was clicked: a bare route change
+ # would land on whichever change the queue happens to list first.
+ assert 'if (scroll) {' in html and 'location.hash = "#/review"' in html
def test_index_renders_the_board_when_history_is_unavailable(client):
@@ -900,7 +1581,7 @@ def test_history_payload_is_byte_stable_between_polls(client, tmp_path, monkeypa
def test_history_orders_rollback_targets_newest_snapshot_first(client, tmp_path, monkeypatch):
"""The picker lists most-recently-snapshotted first, so option 0 is the change you just made."""
- from mcp_server.registry import load_skills, optimizable_components, skill_revision
+ from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision
skill, first = _promoted_skill(tmp_path, monkeypatch)
champion = optimizable_components(skill)
@@ -913,7 +1594,193 @@ def test_history_orders_rollback_targets_newest_snapshot_first(client, tmp_path,
"challenger": {"revision": skill_revision(skill, challenger)}, "gate": gate},
})
second = load_skills(skill.parent)[0].revision
- P.approve_pending("pdf")
+ P._activate_approved("pdf", P.load_pending("pdf"))
listed = [r["revision"] for r in client.get("/api/history").json()["revisions"]["pdf"]]
assert listed == [second, first]
+
+
+def test_publications_reports_a_quarantined_change_it_cannot_read(client):
+ """list_pending skips an unreadable record so one corrupt file cannot break review, which also
+ means an unreadable proposal is indistinguishable from no proposal and the board reports CLEAR
+ over it. Observed live: the MCP container wrote records as root 0600 while the UI ran as uid
+ 1000, and an approved-and-waiting skill sat invisible for hours."""
+ P.pending_dir().mkdir(parents=True, exist_ok=True)
+ (P.pending_dir() / "measurement-integrity.json").write_bytes(b"\xff\xfe not json")
+
+ payload = client.get("/api/publications").json()
+
+ assert payload["pending_blocked"], "an unreadable quarantined change must be surfaced"
+ assert "measurement-integrity.json" in payload["pending_blocked"]
+
+
+def test_publications_stays_quiet_when_the_review_queue_is_readable(client):
+ P.pending_dir().mkdir(parents=True, exist_ok=True)
+ assert client.get("/api/publications").json()["pending_blocked"] is None
+
+
+def test_a_non_utf8_pending_file_does_not_take_down_the_review_page(client):
+ """list_pending caught OSError and JSONDecodeError but not UnicodeDecodeError, so a binary file
+ in the queue raised straight through and the whole review surface 500ed."""
+ P.pending_dir().mkdir(parents=True, exist_ok=True)
+ (P.pending_dir() / "pdf.json").write_bytes(b"\xff\xfe\x00binary")
+ assert client.get("/api/skills").status_code == 200
+
+
+def test_harbor_matrix_is_served_for_a_skill_that_has_been_run(client, monkeypatch, tmp_path):
+ """The console surface for the harness x model grid."""
+ root = tmp_path / "harbor"
+ root.mkdir()
+ (root / "pdf.json").write_text(json.dumps({"judge": "google/gemini-2.5-flash", "harnesses": {
+ "claude-code@anthropic/claude-opus-5": {
+ "skill_mean": 0.75, "control_mean": 0.5, "lift": 0.25, "tasks_scored": 4,
+ "tasks_dropped": [], "endpoint_url": "https://private.invalid/v1"},
+ "aider@openai/gpt-5.5": {"error": "RuntimeError: every task returned an empty workspace"}}}))
+ monkeypatch.setattr(harbor_report, "HARBOR_DIR", root)
+
+ payload = client.get("/api/harbor/pdf").json()
+
+ assert payload["measured"] == 1 and payload["unmeasured"] == 1
+ rows = {r["combination"]: r for r in payload["rows"]}
+ # The rule the whole surface exists for: a combination that did not run reaches the browser
+ # with no lift key at all, so no renderer can put a number in its measurement column.
+ assert "lift" not in rows["aider@openai/gpt-5.5"]
+ assert rows["claude-code@anthropic/claude-opus-5"]["n"] == 4
+ # A renderer may show the recorded alias/protocol, never an endpoint URL supplied by a run.
+ assert "endpoint_url" not in rows["claude-code@anthropic/claude-opus-5"]
+ assert client.get("/api/harbor").json()["skills"] == ["pdf"]
+
+
+def test_harbor_page_pivots_sparse_evidence_by_harness_and_model(client):
+ """The console gives every axis intersection its own evidence state, never an implied zero."""
+ html = client.get("/").text
+
+ # renderHarbor owns the pivot from API axes, rather than relying on a pre-filled rectangular
+ # payload. The API deliberately sends only rows that were attempted.
+ assert "function matrixCell(row)" in html
+ assert "data.harnesses" in html and "data.models" in html
+ assert "byHarness" in html and "byModel" in html
+ # These loops are the rectangular matrix contract. Axis/map names alone would let a renderer
+ # emit one header or skip sparse intersections, which silently changes absence into no cell.
+ assert "${models.map(model => {" in html
+ assert "const body = harnesses.map(harness =>" in html
+ assert "models.map(model => matrixCell(byHarness.get(harness)?.get(model))).join(\"\")" in html
+ assert "never run" in html
+ assert "not measured" in html
+ assert "toFixed(3)" in html
+ assert "skill mean" in html and "control mean" in html
+ assert "attempts" in html and "dropped" in html
+ assert "target alias" in html and "protocol" in html and "error" in html
+ # Measured and failed cells are buttons, so keyboard and pointer activation share one route;
+ # a blank intersection is evidence of absence, not a neutral interactive result.
+ assert 'class="mx-plate mx-measured' in html
+ assert 'class="mx-plate mx-error"' in html
+ assert "mx-blank" in html
+ assert 'aria-label="never run"' in html
+ assert "mx-details" in html and 'aria-live="polite"' in html
+
+
+def test_harbor_matrix_css_contract_preserves_readable_sparse_columns(client):
+ html = client.get("/").text
+
+ assert ".mx-wrap" in html and "overflow-x: auto" in html
+ assert ".mx-sticky" in html and "position: sticky" in html
+ assert ".mx-corner" in html
+ assert ".mx td, .mx th" in html and "min-width:" in html
+ # Per-model warnings wrap inside a fixed evidence column; otherwise one ceiling warning can
+ # stretch a sparse matrix across several screens.
+ assert "table-layout: fixed" in html and "--mx-width" in html
+ assert ".mx-plate:focus-visible" in html
+ assert ".mx-blank" in html and ".mx-error" in html and ".mx-measured" in html
+
+
+def test_harbor_page_plots_only_observed_model_scale_rows(client):
+ html = client.get("/").text
+
+ assert "function sizeLiftChart(rows)" in html
+ assert "Math.log10" in html and "Number.isFinite(size)" in html and "size > 0" in html
+ assert "sizeLiftChart(rows)" in html
+ assert "data.legacy" not in html
+ assert "data-size-point" in html and "generationShape" in html
+ assert "Model size vs lift" in html and "Parameters (billions, log scale)" in html
+ assert "Object.is(rounded, -0) ? 0 : rounded" in html
+ assert "Failed and never-run cells remain in the evidence ledger below." in html
+ assert 'tabindex="0"' in html and "event.key === \"Enter\"" in html
+ assert "parameter_billions" in html and "quantization" in html and "tool parser" in html
+ assert 'includes("Qwen3.5") ? "circle"' in html
+ assert 'includes("Qwen3.6") ? "square" : "diamond"' in html
+ assert "diamond = other model families" in html
+
+
+def test_harbor_size_chart_keeps_endpoint_swarms_inside_the_plot(client):
+ html = client.get("/").text
+
+ assert "const swarmInset = Math.max(...sizes.map(size =>" in html
+ assert "left + swarmInset" in html
+ assert "plotRight - swarmInset" in html
+
+
+def test_secondary_routes_put_their_own_job_first(client):
+ html = client.get("/").text
+
+ assert 'document.querySelector(".board-head").hidden = base !== "review"' in html
+ assert ".board-head[hidden]" in html
+
+
+def test_every_route_uses_the_evidence_cockpit_visual_system(client):
+ """The full-console redesign must alter the shell, not only a chart inside the old page."""
+ html = client.get("/").text
+
+ assert 'class="topbar cockpit-command"' in html
+ assert 'class="shell cockpit-shell"' in html
+ assert 'class="sidebar cockpit-rail"' in html
+ assert 'class="main cockpit-workspace"' in html
+ assert "/* ---- evidence cockpit visual system ---- */" in html
+ assert ".cockpit-rail .navlink.active::before" in html
+ assert ".cockpit-workspace > .view:not([hidden])" in html
+ assert ".cockpit-workspace .review-card" in html
+ assert ".cockpit-workspace .trace-list" in html
+ assert ".cockpit-workspace .mx-wrap" in html
+ assert ".cockpit-shell { grid-template-columns: 1fr; }" in html
+ assert ".cockpit-rail #nav-folders { display: contents; }" in html
+
+
+def test_harbor_size_chart_has_a_persistent_readable_interaction_layer(client):
+ html = client.get("/").text
+
+ assert "mx-chart-stats" in html
+ assert "mx-chart-legend" in html
+ assert 'id="mx-chart-detail"' in html
+ assert "pointOffset" in html and "* 18" in html
+ assert "Measured evidence only" in html
+ assert 'row.n == null ? "not recorded"' in html
+ assert 'aria-live="polite"' in html
+
+
+def test_harbor_poll_discovers_and_rerenders_progressive_results(client):
+ html = client.get("/").text
+
+ assert 'if (currentRoute() === "harnesses") await loadHarbor();' in html
+ assert 'await j("/api/harbor")' in html
+ assert "harborSkills = available" in html
+ assert "const selected = harborSkill" in html
+
+
+def test_a_skill_never_run_across_harnesses_is_a_404_not_an_empty_matrix(client, monkeypatch, tmp_path):
+ """An empty grid on the page would read as 'no combination helps'. It has not been measured."""
+ monkeypatch.setattr(harbor_report, "HARBOR_DIR", tmp_path / "harbor")
+ response = client.get("/api/harbor/pdf")
+ assert response.status_code == 404
+ assert "harbor_eval" in response.json()["detail"]
+
+
+def test_an_unreadable_matrix_is_an_error_not_a_silent_absence(client, monkeypatch, tmp_path):
+ root = tmp_path / "harbor"
+ root.mkdir()
+ (root / "pdf.json").write_text("{ not json")
+ monkeypatch.setattr(harbor_report, "HARBOR_DIR", root)
+ assert client.get("/api/harbor/pdf").status_code == 503
+
+
+def test_harbor_rejects_a_bad_skill_name(client):
+ assert client.get("/api/harbor/..%2Fetc").status_code in (400, 404)
diff --git a/tests/test_usage.py b/tests/test_usage.py
index 7480444..0cf6af1 100644
--- a/tests/test_usage.py
+++ b/tests/test_usage.py
@@ -1,7 +1,7 @@
-"""Unit tests for the per-run token ledger (optimize.usage), including thread-safety."""
+"""Unit tests for the per-run token ledger (ingot.optimize.usage), including thread-safety."""
import threading
-from optimize import usage
+from ingot.optimize import usage
def test_add_accumulates_per_role_and_totals():
@@ -51,3 +51,26 @@ def test_format_report_is_readable():
usage.add("judge", {"input_tokens": 1234, "output_tokens": 56})
out = usage.format_report()
assert "judge" in out and "TOTAL" in out and "1,234" in out
+
+
+def test_subscription_usage_is_counted_without_cost_or_spend_cap(monkeypatch):
+ usage.reset()
+ monkeypatch.delenv("BASE_URL", raising=False)
+ monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False)
+ monkeypatch.setenv("JUDGE_MODEL", "m/metered-judge")
+ monkeypatch.setenv("MAX_RUN_USD", "0.01")
+ monkeypatch.setattr(usage, "_PRICES", {"m/metered-judge": (1.0, 1.0)})
+
+ usage.add(
+ "judge",
+ {"input_tokens": 11, "output_tokens": 7},
+ billing_mode="subscription",
+ )
+
+ assert usage.report()["judge"] == {"input": 11, "output": 7, "calls": 1}
+ assert usage.estimated_cost() is None
+ assert usage.unpriced_roles() == []
+ report = usage.format_report()
+ assert "judge" in report and "subscription" in report
+ assert "$" not in report
+ usage.reset()
diff --git a/tests/test_usage_counts.py b/tests/test_usage_counts.py
index e6535a6..ff5d822 100644
--- a/tests/test_usage_counts.py
+++ b/tests/test_usage_counts.py
@@ -1,5 +1,5 @@
"""Unit tests for the per-skill load counter (no server needed)."""
-from mcp_server import usage_counts
+from ingot.mcp_server import usage_counts
def test_record_use_increments_and_persists(tmp_path, monkeypatch):
diff --git a/tests/test_vault.py b/tests/test_vault.py
new file mode 100644
index 0000000..a5b7297
--- /dev/null
+++ b/tests/test_vault.py
@@ -0,0 +1,109 @@
+"""`ingot vault init`: the one bootstrap command a local deployment needs.
+
+`ingot status`, the other half of the managed-deployment surface, is in test_status.py."""
+import json
+import subprocess
+import sys
+from pathlib import Path
+
+import pytest
+
+from ingot import cli, vault
+
+
+def _git(path, *args):
+ return subprocess.run(["git", "-C", str(path), *args], check=True,
+ capture_output=True, text=True).stdout.strip()
+
+
+# --------------------------------------------------------------------------- vault init
+
+def test_init_creates_a_committed_vault_on_main(tmp_path):
+ result = vault.init_vault(tmp_path / "vault")
+
+ created = tmp_path / "vault"
+ assert result["status"] == "created"
+ assert result["branch"] == "main"
+ assert (created / "registry.json").read_text() == "{}\n"
+ assert (created / "scripts" / "validate.py").is_file()
+ assert _git(created, "status", "--porcelain") == ""
+ assert _git(created, "rev-parse", "HEAD") == result["head"]
+
+
+def test_init_is_idempotent(tmp_path):
+ first = vault.init_vault(tmp_path / "vault")
+ second = vault.init_vault(tmp_path / "vault")
+
+ assert second["status"] == "unchanged"
+ assert second["head"] == first["head"]
+ assert second["added"] == []
+
+
+def test_init_completes_an_existing_repository_without_rewriting_it(tmp_path):
+ """Adopting a Git repository someone already keeps skills in must add what is missing and
+ touch nothing else."""
+ existing = tmp_path / "vault"
+ existing.mkdir()
+ subprocess.run(["git", "init", "-b", "main", str(existing)], check=True, capture_output=True)
+ _git(existing, "config", "user.email", "owner@test.invalid")
+ _git(existing, "config", "user.name", "Owner")
+ (existing / "registry.json").write_text('{"pdf": {"disposition": "keep"}}\n')
+ _git(existing, "add", ".")
+ _git(existing, "commit", "-m", "Existing vault")
+
+ result = vault.init_vault(existing)
+
+ assert result["status"] == "updated"
+ assert "scripts/validate.py" in result["added"]
+ assert json.loads((existing / "registry.json").read_text()) == {"pdf": {"disposition": "keep"}}
+
+
+def test_init_refuses_a_non_empty_directory_that_is_not_a_repository(tmp_path):
+ """Adopting a directory of loose skills would make the first commit look like a publication
+ nobody approved."""
+ loose = tmp_path / "skills"
+ (loose / "pdf").mkdir(parents=True)
+ (loose / "pdf" / "SKILL.md").write_text("---\nname: pdf\ndescription: x\n---\nbody\n")
+
+ with pytest.raises(ValueError, match="not empty and is not a Git repository"):
+ vault.init_vault(loose)
+
+
+def test_the_default_validator_refuses_a_tree_the_server_could_not_serve(tmp_path):
+ created = tmp_path / "vault"
+ vault.init_vault(created)
+ (created / "pdf").mkdir()
+ (created / "pdf" / "SKILL.md").write_text("---\nname: other\ndescription: Merge.\n---\nbody\n")
+
+ result = subprocess.run([sys.executable, "scripts/validate.py"], cwd=created,
+ capture_output=True, text=True)
+
+ assert result.returncode == 1
+ assert "!= directory 'pdf'" in result.stderr
+
+
+def test_the_default_validator_accepts_a_well_formed_tree(tmp_path):
+ created = tmp_path / "vault"
+ vault.init_vault(created)
+ (created / "pdf").mkdir()
+ (created / "pdf" / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge.\n---\nbody\n")
+
+ assert subprocess.run([sys.executable, "scripts/validate.py"], cwd=created,
+ capture_output=True).returncode == 0
+
+
+# --------------------------------------------------------------------------- the command line
+
+def test_vault_init_from_the_command_line(tmp_path, capsys):
+ assert cli.main(["vault", "init", str(tmp_path / "vault"), "--json"]) == 0
+
+ payload = json.loads(capsys.readouterr().out)
+ assert payload["schema_version"] == vault.VAULT_SCHEMA
+ assert Path(payload["path"]) == (tmp_path / "vault").resolve()
+
+
+def test_vault_init_reports_a_refusal_without_a_traceback(tmp_path, capsys):
+ (tmp_path / "loose.txt").write_text("not a vault")
+
+ assert cli.main(["vault", "init", str(tmp_path)]) == 1
+ assert "ingot vault init:" in capsys.readouterr().err
diff --git a/ui/app.py b/ui/app.py
index ad3746e..c797a04 100644
--- a/ui/app.py
+++ b/ui/app.py
@@ -2,31 +2,45 @@
Reviewers see the evidence and the promotion decision first. SkillOpt optimization is a core
workflow that only ever writes to the pending queue. Promotion and
-rollback both go through `optimize.promote`, which snapshots the displaced revision and swaps
+rollback both go through `ingot.optimize.promote`, which snapshots the displaced revision and swaps
directories atomically.
"""
import difflib
+import json
import logging
import os
+import tempfile
import threading
+from collections.abc import Callable
from contextlib import contextmanager
from pathlib import Path
from urllib.parse import urlparse
+import yaml
from fastapi import Depends, FastAPI, HTTPException, Request
from fastapi.responses import FileResponse, RedirectResponse
from pydantic import BaseModel, Field
-from mcp_server.registry import SLUG_RE, load_skills, read_components, skill_revision
-from optimize.ab import TASKS_DIR, run_ab
+from ingot.mcp_server.registry import SLUG_RE, load_skills, read_components, skill_revision
+from ingot.optimize import harbor_report, resolve_skill_dir
+from ingot.optimize.ab import TASKS_DIR, run_ab
+from ingot.optimize.local_traces import LOCAL_TRACE_FILE, store_summary
from ui.auth import (auth_mode, current_actor, require_auth, require_role, using_default_password)
-from optimize.promote import (_audit_best_effort, approve_pending, list_revisions,
- list_snapshotted_skills, load_pending, load_snapshot_components,
- pending_path, read_audit, rollback, stale_evidence_reason)
+from ingot.optimize.promote import (ABSENT_REVISION, _audit_best_effort, approve_pending, list_pending, list_revisions,
+ list_snapshotted_skills, load_pending, load_snapshot_components,
+ pending_path, read_audit, reject_pending, rollback,
+ stale_evidence_reason)
+from ingot.optimize.publication import (publication_for_skill, publishing_skills, recent_publications)
+from ingot import paths
logger = logging.getLogger(__name__)
-REPO_ROOT = Path(__file__).resolve().parent.parent
-EVIDENCE_DIR = (REPO_ROOT / "runs" / "evidence").resolve()
+# Recorded evidence locations are written relative to the state root, so this is what they
+# resolve against -- not the code's directory, which no longer holds state.
+STATE_ROOT = paths.runs().parent
+
+
+def _evidence_dir() -> Path:
+ return (paths.runs() / "evidence").resolve()
# Bundles written inside a container recorded their container-absolute path before evidence
# locations became repo-relative. Both forms name the same file from the host checkout.
CONTAINER_ROOT = Path("/app")
@@ -82,7 +96,8 @@ def auth_me(actor: str = Depends(current_actor)):
return {"authenticated": password, "email": "", "name": actor if password else "",
"role": "admin" if password else ""}
-RUNS: dict[str, dict] = {} # skill -> {"status": running|done|error, "log": [lines]}
+RUNS: dict[str, dict] = {} # skill -> {"status": running|done|error, "action": eval|review|optimize, "log": [lines]}
+RUN_LOCK = threading.Lock()
@app.get("/")
@@ -98,13 +113,78 @@ def config():
return {"langfuse_url": os.environ.get("LANGFUSE_PUBLIC_URL", "http://localhost:3100")}
+@app.get("/api/traces")
+def traces(project: str = "", harness: str = "", skill: str = "", since: str = "",
+ until: str = "", include_tasks: bool = False):
+ """Safe console projection of locally normalized coding-agent turns.
+
+ Answers stay in the on-disk store used by the explicit mining command. The browser receives
+ task text and attribution metadata, never answer text, reasoning, or tool payloads.
+ """
+ try:
+ return store_summary(LOCAL_TRACE_FILE, project=project, harness=harness, skill=skill,
+ since=since, until=until, include_tasks=include_tasks)
+ except ValueError as error:
+ raise HTTPException(400, str(error)) from error
+
+
+@app.get("/api/harbor")
+def harbor_matrices():
+ """Skills that have a cross-harness matrix on disk."""
+ return {"skills": harbor_report.available()}
+
+
+@app.get("/api/harbor/{skill}")
+def harbor_matrix(skill: str):
+ """The harness x model matrix for one skill: does this skill help this combination, and by how
+ much, judged in a sandbox against the same held-out tasks with and without the skill.
+
+ 404 when the skill has never been run, so "not measured yet" cannot be read as "no effect".
+ """
+ _check(skill)
+ try:
+ matrix = harbor_report.read_matrix(skill)
+ except ValueError as error:
+ raise HTTPException(503, str(error)) from error
+ if matrix is None:
+ raise HTTPException(404, f"no cross-harness run for '{skill}'; run "
+ f"`python -m ingot.optimize.harbor_eval {skill} --agent `")
+ return matrix
+
+
+@app.get("/api/clusters")
+def clusters():
+ """The category buckets, as last computed. Clustering loads the embedding model, which is too
+ heavy for a request on the poll path, so it is a command that writes the file and this only
+ serves it. 404 names the command rather than implying the feature is missing."""
+ from ingot.optimize.cluster import CLUSTER_PATH
+ if not CLUSTER_PATH.exists():
+ raise HTTPException(404, "no clusters computed yet; run "
+ "`docker compose run --rm --entrypoint python optimize "
+ "-m ingot.optimize.cluster`")
+ try:
+ data = json.loads(CLUSTER_PATH.read_text())
+ except (OSError, ValueError) as e: # a half-written or hand-edited file is not a 500
+ raise HTTPException(503, f"clusters file is unreadable ({e}); re-run ingot.optimize.cluster")
+ indexed = {s.name for s in _cached_load_skills()}
+ # Skills come and go between clustering runs. Say so rather than drawing a stale map as fact.
+ for cluster in data.get("clusters", []):
+ for member in cluster.get("members", []):
+ member["missing"] = member["name"] not in indexed
+ data["stale_members"] = sum(m.get("missing", False)
+ for c in data.get("clusters", []) for m in c.get("members", []))
+ data["unclustered"] = len(indexed) - sum(len(c.get("members", []))
+ for c in data.get("clusters", []))
+ return data
+
+
_SKILLS_CACHE = None
_SKILLS_CACHE_KEY = None
def _get_skills_cache_key():
"""Fast cache key based on mtime of skill directories and SKILL.md files."""
- from mcp_server.registry import configured_roots, skill_sources
+ from ingot.mcp_server.registry import configured_roots, skill_sources
roots = configured_roots()
mtimes = [id(load_skills)]
for r in roots:
@@ -131,16 +211,37 @@ def _cached_load_skills():
@app.get("/api/skills")
def skills():
tasksets = {p.stem for p in TASKS_DIR.glob("*.yaml")}
- from mcp_server.usage_counts import load_counts
+ from ingot.mcp_server.usage_counts import load_counts
+ from ingot.mcp_server import provenance
counts = load_counts()
- return [
+ # One parsed ledger per root, not per skill: a merged library re-reads the same VENDORED.md
+ # once for every skill it serves otherwise.
+ ledgers: dict = {}
+ # Approval does not free the review slot: the pending record is held until the vault commit
+ # lands, so an approved change keeps reading as one still awaiting a decision.
+ publishing = publishing_skills()
+ active = [
{"name": s.name, "description": s.description, "has_tasks": s.name in tasksets,
"pending": load_pending(s.name) is not None, "revision": s.revision,
+ "publishing": s.name in publishing,
"uses": counts.get(s.name, 0),
+ "provenance": provenance.classify(s.name, s.root, ledgers=ledgers),
"status": RUNS.get(s.name, {}).get("status")}
for s in _cached_load_skills()
if SLUG_RE.fullmatch(s.name) # a non-slug name (hostile frontmatter) can't be optimized anyway
]
+ active_names = {item["name"] for item in active}
+ creations = []
+ for pending in list_pending():
+ if pending.get("kind") != "creation" or pending.get("skill") in active_names:
+ continue
+ components = pending.get("challenger_components") or {}
+ creations.append({"name": pending["skill"],
+ "description": str(components.get("description", "")),
+ "has_tasks": False, "pending": True, "revision": "", "uses": 0,
+ "publishing": pending["skill"] in publishing,
+ "provenance": "proposed", "status": None, "active": False})
+ return active + creations
def _active_skill(skill: str):
@@ -181,7 +282,7 @@ def skill_versions(skill: str):
"revision": _pending_revision(active, pending),
"created": pending.get("created")})
versions.extend({"key": item["revision"], "kind": "snapshot", **item}
- for item in list_revisions(skill))
+ for item in list_revisions(skill) if item["revision"] != ABSENT_REVISION)
return {"skill": skill, "description": active.description, "versions": versions}
@@ -215,29 +316,126 @@ def optimize(skill: str):
result is a quarantined pending record for review."""
_check(skill)
_preflight_optimize(skill)
- state = RUNS[skill] = {"status": "running", "log": []}
+ state, log = _start_run(skill, "optimize")
+
+ threading.Thread(target=_run_optimization, args=(skill, state, log), daemon=True).start()
+ return {"started": skill}
+
+
+def _preflight_optimize(skill: str) -> None:
+ _preflight_provider()
+ if not (TASKS_DIR / f"{skill}.yaml").exists():
+ raise HTTPException(404, f"no eval task set for '{skill}'")
+
+
+def _start_run(skill: str, action: str) -> tuple[dict, Callable[..., None]]:
+ # Drafting and optimization share a process-global token ledger and OpenRouter budget. Claim
+ # the slot under a lock: FastAPI runs sync handlers concurrently, so a check before assignment
+ # lets two paid requests both pass.
+ with RUN_LOCK:
+ if any(s.get("status") == "running" for s in RUNS.values()):
+ raise HTTPException(409, "a paid model run is already in progress")
+ state = RUNS[skill] = {"status": "running", "action": action, "log": []}
def log(*args):
state["log"].append(" ".join(str(a) for a in args))
if len(state["log"]) > 1000:
state["log"] = state["log"][-1000:]
- threading.Thread(target=_run_optimization, args=(skill, state, log), daemon=True).start()
- return {"started": skill}
+ return state, log
-def _preflight_optimize(skill: str) -> None:
+@app.post("/api/evals/{skill}",
+ dependencies=[Depends(same_origin), Depends(require_role("proposer"))])
+def create_eval_set(skill: str):
+ """Draft the missing train/holdout task set without starting an optimization."""
+ _active_skill(_check(skill))
_preflight_provider()
- if not (TASKS_DIR / f"{skill}.yaml").exists():
+ if (TASKS_DIR / f"{skill}.yaml").exists():
+ raise HTTPException(409, f"'{skill}' already has an eval task set")
+ state, log = _start_run(skill, "eval")
+ threading.Thread(target=_run_eval_draft, args=(skill, state, log), daemon=True).start()
+ return {"started": skill}
+
+
+def _eval_payload(skill: str) -> dict:
+ path = TASKS_DIR / f"{skill}.yaml"
+ if not path.exists():
raise HTTPException(404, f"no eval task set for '{skill}'")
- # one SkillOpt run at a time: the token ledger is process-global, and concurrent runs
- # would also contend for the same OpenRouter budget
- if any(s.get("status") == "running" for s in RUNS.values()):
- raise HTTPException(409, "a SkillOpt run is already in progress")
+ try:
+ data = yaml.safe_load(path.read_text()) or {}
+ if not isinstance(data, dict):
+ raise ValueError("top level must be a mapping")
+ train = data.get("train") or data.get("tasks") or []
+ explicit_holdout = bool(data.get("holdout"))
+ holdout = data.get("holdout") or train
+ routing = data.get("routing") or []
+ acceptance = data.get("acceptance") or []
+ if not all(isinstance(section, list)
+ for section in (train, holdout, routing, acceptance)):
+ raise ValueError("train, holdout, routing, and acceptance must be lists")
+ except (OSError, ValueError, yaml.YAMLError) as e:
+ raise HTTPException(
+ 503, f"eval set is unreadable for '{skill}' ({e}); repair its task file") from e
+ tasks = list(train) + (list(holdout) if explicit_holdout else [])
+ return {
+ "skill": skill, "train": train, "holdout": holdout, "routing": routing,
+ "acceptance": acceptance, "leakage": not explicit_holdout,
+ "counts": {
+ "train": len(train), "holdout": len(holdout), "routing": len(routing),
+ "acceptance": len(acceptance),
+ "checks": sum(len(task.get("checklist") or [])
+ for task in tasks if isinstance(task, dict)),
+ },
+ }
+
+
+@app.get("/api/evals/{skill}")
+def eval_set(skill: str):
+ """The exact task-set inputs the optimizer reads, exposed so its scores stay traceable."""
+ _active_skill(_check(skill))
+ return _eval_payload(skill)
+
+
+@app.get("/api/reviews/{skill}")
+def review_result(skill: str):
+ """The newest persisted current-skill review. Reviews measure; they never propose or activate."""
+ active = _active_skill(_check(skill))
+ from ingot.optimize.review import REVIEW_DIR
+ path = REVIEW_DIR / f"{skill}.json"
+ if not path.exists():
+ raise HTTPException(404, f"no review result for '{skill}'")
+ try:
+ data = json.loads(path.read_text())
+ if not isinstance(data, dict):
+ raise ValueError("top level must be an object")
+ data.setdefault("created", int(path.stat().st_mtime))
+ recorded = data.get("revision")
+ if not isinstance(recorded, str) or not recorded:
+ data["stale"] = "review did not record a skill revision; run it again"
+ elif recorded != active.revision:
+ data["stale"] = "active skill changed since this review ran; run it again"
+ else:
+ data["stale"] = None
+ return data
+ except (OSError, ValueError) as e:
+ raise HTTPException(503, f"review result is unreadable ({e}); run the review again") from e
+
+
+@app.post("/api/reviews/{skill}",
+ dependencies=[Depends(same_origin), Depends(require_role("proposer"))])
+def start_review(skill: str):
+ """Score the active skill against its evals without creating a pending revision."""
+ _active_skill(_check(skill))
+ _preflight_provider()
+ _eval_payload(skill) # refuse missing or malformed measurement inputs before spending
+ state, log = _start_run(skill, "review")
+ threading.Thread(target=_run_review, args=(skill, state, log), daemon=True).start()
+ return {"started": skill}
def _preflight_provider() -> None:
- from optimize import openrouter_key_missing, preflight_provider_pins
+ from ingot.optimize import openrouter_key_missing, preflight_provider_pins
if openrouter_key_missing():
raise HTTPException(400, "API_KEY is not set, copy .env.example to .env, "
"add your key (https://openrouter.ai/keys), and restart the stack "
@@ -257,12 +455,47 @@ def _run_optimization(skill: str, state: dict, log) -> None:
state["status"] = "error"
+def _run_eval_draft(skill: str, state: dict, log) -> None:
+ try:
+ from ingot.optimize import usage as usage_ledger
+ from ingot.optimize.draft import draft_and_save
+
+ usage_ledger.reset()
+ components = read_components(resolve_skill_dir(skill))
+ # The teacher may fail after opening its output. Stage beside the mounted task directory,
+ # then hard-link the complete file into place so failure cannot create a bogus eval set
+ # and a concurrent writer cannot be overwritten.
+ with tempfile.TemporaryDirectory(prefix=f".draft-{skill}-", dir=TASKS_DIR) as staging:
+ staged = draft_and_save(skill, components["description"], components["body"],
+ Path(staging), log=log)
+ try:
+ os.link(Path(staged), TASKS_DIR / f"{skill}.yaml")
+ except FileExistsError:
+ raise RuntimeError(f"'{skill}' already has an eval task set") from None
+ state["status"] = "done"
+ except BaseException as e: # surface provider, parse, and SystemExit failures in the card
+ log(f"ERROR: {e}")
+ state["status"] = "error"
+
+
+def _run_review(skill: str, state: dict, log) -> None:
+ try:
+ from ingot.optimize.review import run_review
+ run_review(skill, log=log)
+ state["status"] = "done"
+ except BaseException as e: # provider, judge, and spend-cap failures belong in the skill log
+ log(f"ERROR: {e}")
+ state["status"] = "error"
+
+
@app.get("/api/runs")
def runs():
- return {skill: {"status": s["status"], "log": s["log"][-30:]} for skill, s in RUNS.items()}
+ return {skill: {"status": s["status"], "action": s.get("action", "optimize"),
+ "log": s["log"][-30:]} for skill, s in RUNS.items()}
-_COMPONENT_LABEL = {"description": "SKILL.md (description)", "body": "SKILL.md (body)"}
+_COMPONENT_LABEL = {"description": "SKILL.md (description)", "body": "SKILL.md (body)",
+ "frontmatter": "SKILL.md (frontmatter)"}
def _label(component: str) -> str:
@@ -292,6 +525,22 @@ def _review_risk(champion: dict, challenger: dict) -> dict:
"high_risk": changed_pct >= 50 or size_delta_pct <= -50}
+def _publication_matches_pending(publication: dict, pending: dict) -> bool:
+ """Whether a skill-level receipt belongs to this exact quarantined proposal."""
+ proposal_id = next((str((pending.get(kind) or {}).get("proposal_id") or "")
+ for kind in ("retrospective", "creation")
+ if (pending.get(kind) or {}).get("proposal_id")), "")
+ revision = str(((pending.get("evidence") or {}).get("challenger") or {}).get("revision") or "")
+ identities = []
+ if proposal_id:
+ identities.append(str(publication.get("proposal_id") or "") == proposal_id)
+ if revision:
+ identities.append(str(publication.get("candidate_revision") or "") == revision)
+ if not identities:
+ identities.append(publication.get("components") == pending.get("challenger_components"))
+ return all(identities)
+
+
@app.get("/api/pending/{skill}")
def pending(skill: str):
p = load_pending(_check(skill))
@@ -304,13 +553,24 @@ def pending(skill: str):
for comp in changed_components:
label = _label(comp)
blocks.append("\n".join(difflib.unified_diff(
- champ[comp].splitlines(), chall.get(comp, "").splitlines(),
+ str(champ.get(comp, "")).splitlines(), str(chall.get(comp, "")).splitlines(),
fromfile=f"{label} (champion)", tofile=f"{label} (challenger)", lineterm="")))
comparison = [{"component": _label(component), "before": str(champ.get(component, "")),
"after": str(chall.get(component, ""))}
for component in changed_components]
+ publication = publication_for_skill(skill)
+ if publication and not _publication_matches_pending(publication, p):
+ publication = None
+ # `pr` and `note` are what make a stalled publication legible: a receipt sitting at
+ # awaiting_merge because the vault does not allow auto-merge is waiting on a person, and the
+ # reviewer has no other way to learn which pull request to go and merge.
+ publication_view = ({key: publication.get(key) for key in
+ ("id", "state", "actor", "action", "attempts", "last_error", "pr", "note")}
+ if publication else None)
return {"skill": skill, "kind": p.get("kind", "quality"), "inner_loop": _inner_loop(p),
"ab": p.get("ab"), "routing": p.get("routing"), "dataset": p.get("dataset"),
+ "retrospective": p.get("retrospective"), "creation": p.get("creation"),
+ "publication": publication_view,
"evidence": p.get("evidence_paths"), "stale": _stale_reason(skill, p),
"model": p.get("model"), "judge": p.get("judge"),
"gate": p.get("gate", {"promotable": True, "blocked": []}),
@@ -368,6 +628,57 @@ def approve(skill: str, actor: str = Depends(current_actor)):
raise HTTPException(409, str(e))
+PUBLICATION_FIELDS = ("id", "skill", "action", "state", "actor", "attempts",
+ "pr", "note", "last_error", "created")
+LIVE_STATES = ("approved_publishing", "publishing", "awaiting_merge")
+
+
+def _pending_blocked() -> str | None:
+ """A human-readable warning when a pending record exists but cannot be used, else None."""
+ from ingot.optimize.promote import pending_dir, unreadable_pending
+ queue = pending_dir()
+ if queue.is_dir() and not os.access(queue, os.R_OK | os.X_OK):
+ return (f"cannot read the review queue at {queue} as uid {os.getuid()}; "
+ f"quarantined changes cannot be listed")
+ blocked = unreadable_pending()
+ if not blocked:
+ return None
+ return (f"{len(blocked)} quarantined change(s) in {queue} cannot be read as uid "
+ f"{os.getuid()} and are NOT shown below: {', '.join(blocked)}")
+
+
+@app.get("/api/publications")
+def publications():
+ """The publication lane, read only.
+
+ A receipt outlives the pending record it came from, so this is the only surface that can show
+ an approved change while it is still travelling to the vault. `unreadable` is not cosmetic:
+ `Path.glob` swallows `PermissionError`, so a store this process cannot list looks exactly like
+ an empty one, and an empty lane is the reading a stalled publisher most wants you to make."""
+ # Read the receipt store through the module attribute, not a name bound at import: the path is
+ # configurable, `recent_publications` looks it up per call, and a frozen copy here inspected one
+ # directory while listing another.
+ from ingot.optimize.publication import publications_dir
+
+ # Only the forge backend has a pull request to link to, and only it knows the repository.
+ forge = os.environ.get("INGOT_FORGE_REPOSITORY") or ""
+ store = publications_dir()
+ unreadable = store.is_dir() and not os.access(store, os.R_OK | os.X_OK)
+ records = [] if unreadable else recent_publications()
+ return {"publications": [dict({key: record.get(key) for key in PUBLICATION_FIELDS},
+ pr_url=(f"https://github.com/{forge}/pull/{record['pr']}"
+ if forge and record.get("pr") else None))
+ for record in records],
+ "live_states": list(LIVE_STATES),
+ "unreadable": (f"cannot read the receipt store at {store} as uid "
+ f"{os.getuid()}; publications cannot be listed") if unreadable else None,
+ # The same misreading, one directory over. `list_pending` skips a record it cannot read
+ # so one corrupt file cannot break review, which also means an unreadable proposal is
+ # indistinguishable from no proposal and the board reports CLEAR over it. Observed live:
+ # the MCP container writes records as root 0600 while this process is uid 1000.
+ "pending_blocked": _pending_blocked()}
+
+
@app.get("/api/evidence/{skill}")
def evidence(skill: str):
"""The recorded evidence bundle for a pending change, read only.
@@ -376,7 +687,8 @@ def evidence(skill: str):
runs/evidence. Nothing a request carries selects a file."""
path = _evidence_file(_recorded_location(_check(skill)))
markdown = _read_evidence(path)
- return {"skill": skill, "path": path.relative_to(EVIDENCE_DIR).as_posix(), "markdown": markdown}
+ return {"skill": skill, "path": path.relative_to(_evidence_dir()).as_posix(),
+ "markdown": markdown}
def _recorded_location(skill: str) -> str:
@@ -406,9 +718,9 @@ def _host_path(recorded: str) -> Path:
refuse."""
path = Path(recorded)
if not path.is_absolute():
- return REPO_ROOT / path
+ return STATE_ROOT / path
try:
- return REPO_ROOT / path.relative_to(CONTAINER_ROOT)
+ return STATE_ROOT / path.relative_to(CONTAINER_ROOT)
except ValueError:
return path
@@ -418,7 +730,7 @@ def _evidence_file(recorded: str) -> Path:
Resolution happens before the containment check, so neither `..` nor a symlink out of the
evidence tree can reach another part of the filesystem."""
resolved = _host_path(recorded).resolve()
- if not resolved.is_relative_to(EVIDENCE_DIR):
+ if not resolved.is_relative_to(_evidence_dir()):
raise HTTPException(400, "recorded evidence path is outside runs/evidence")
return resolved
@@ -443,14 +755,12 @@ def reject(skill: str, payload: RejectRequest | None = None,
# later would let a second reject pass the check, then re-delete and double-audit after the
# first released, returning 200 instead of 404 (mirrors approve/rollback holding the lock).
with change_control(skill):
- pending = load_pending(skill)
- if pending is None:
- raise HTTPException(404, f"no pending change for '{skill}'")
- revision = _challenger_revision(pending)
- pending_path(skill).unlink(missing_ok=True)
- reason = " ".join((payload.reason if payload else "").split())
- _audit_best_effort("reject", skill, revision, actor, reason=reason)
- return {"result": f"rejected the pending change for '{skill}'"}
+ try:
+ result = reject_pending(skill, actor=actor,
+ reason=payload.reason if payload else "")
+ except ValueError as exc:
+ raise HTTPException(404, str(exc))
+ return {"result": result}
@app.get("/api/history")
diff --git a/ui/auth.py b/ui/auth.py
index 7ecc567..c56a34e 100644
--- a/ui/auth.py
+++ b/ui/auth.py
@@ -25,8 +25,9 @@
from pathlib import Path
from fastapi import HTTPException, Request
+from ingot import paths
-AUTH_FILE = Path(os.environ.get("AUTH_FILE") or Path(__file__).resolve().parent.parent / "runs" / "auth.json")
+AUTH_FILE = Path(os.environ.get("AUTH_FILE") or paths.runs() / "auth.json")
_ITERATIONS = 200_000
_ANON = "local-operator" # actor when auth is disabled, matches the pre-auth default
# The compose default (docker-compose.yml sets AUTH_PASSWORD=${AUTH_PASSWORD:-ingot}); we warn while
diff --git a/ui/static/index.html b/ui/static/index.html
index 4a69970..0aa5901 100644
--- a/ui/static/index.html
+++ b/ui/static/index.html
@@ -19,7 +19,7 @@
labels) as well as fills/dots, so in light mode it is Slancha's AA-safe blue-text #0F63C9
(≥4.5:1 on canvas/white/wash); --rust-2 is the darker hover. */
--paper: #F7F8FA; --paper-2: #FFFFFF; --paper-3: #EEF2F7;
- --ink: #15171C; --ink-2: #667085; --ink-3: #98A2B3;
+ --ink: #15171C; --ink-2: #515B6E; --ink-3: #667085;
--line: #D9DEE7; --line-2: #C4CDDA;
--rust: #0F63C9; --rust-2: #0B4F9E; --rust-bg: #E8F1FD;
--pass: #0C6E57; --pass-bg: #E1F7EF;
@@ -34,7 +34,7 @@
@media (prefers-color-scheme: dark) {
:root {
--paper: #0E1420; --paper-2: #161D2B; --paper-3: #1D2534;
- --ink: #E8EDF4; --ink-2: #9AA6B8; --ink-3: #69748A;
+ --ink: #E8EDF4; --ink-2: #9AA6B8; --ink-3: #8A94A6;
--line: #253044; --line-2: #313D52;
--rust: #4D9BFF; --rust-2: #7AB4FF; --rust-bg: #16273F;
--pass: #46C39F; --pass-bg: #10251F;
@@ -46,7 +46,7 @@
}
:root[data-theme="light"] { color-scheme: light;
--paper: #F7F8FA; --paper-2: #FFFFFF; --paper-3: #EEF2F7;
- --ink: #15171C; --ink-2: #667085; --ink-3: #98A2B3;
+ --ink: #15171C; --ink-2: #515B6E; --ink-3: #667085;
--line: #D9DEE7; --line-2: #C4CDDA;
--rust: #0F63C9; --rust-2: #0B4F9E; --rust-bg: #E8F1FD;
--pass: #0C6E57; --pass-bg: #E1F7EF; --fail: #BE3B34; --fail-bg: #FFF0EF;
@@ -55,7 +55,7 @@
--shadow: 0 1px 2px rgb(16 24 40 / 4%), 0 14px 40px rgb(35 52 78 / 9%); }
:root[data-theme="dark"] { color-scheme: dark;
--paper: #0E1420; --paper-2: #161D2B; --paper-3: #1D2534;
- --ink: #E8EDF4; --ink-2: #9AA6B8; --ink-3: #69748A;
+ --ink: #E8EDF4; --ink-2: #9AA6B8; --ink-3: #8A94A6;
--line: #253044; --line-2: #313D52;
--rust: #4D9BFF; --rust-2: #7AB4FF; --rust-bg: #16273F;
--pass: #46C39F; --pass-bg: #10251F; --fail: #E0796D; --fail-bg: #2D1A17;
@@ -92,7 +92,269 @@
.pill.run { color: var(--warn); background: var(--warn-bg); }
.pill.act { color: var(--rust); background: var(--rust-bg); }
- .page { max-width: 74rem; margin: 0 auto; padding: 0 1.4rem 4rem; }
+ /* ---- shell: sidebar + one routed pane ----
+ The library is 102 skills in four roots. As one scrolling column that was 8,600px of page,
+ which is a document, not a console: nothing was reachable without scrolling past everything
+ else. The sidebar carries the routes and the counts, and only one view is mounted at a time. */
+ .shell { display: grid; grid-template-columns: 15.5rem minmax(0, 1fr); max-width: 96rem;
+ margin: 0 auto; align-items: start; }
+ .sidebar { position: sticky; top: 3.05rem; height: calc(100vh - 3.05rem); overflow-y: auto;
+ padding: 1.5rem 1rem 2rem 1.5rem; border-right: 1px solid var(--line); }
+ .main { min-width: 0; padding: 1.5rem 1.5rem 4rem 1.7rem; }
+
+ .navgroup { margin-bottom: 1.5rem; }
+ .navgroup h3 { font: 600 .62rem/1 var(--mono); letter-spacing: .11em; text-transform: uppercase;
+ color: var(--ink-3); margin: 0 0 .5rem .55rem; }
+ .navlink { display: flex; align-items: center; gap: .5rem; padding: .38rem .55rem; border-radius: 8px;
+ color: var(--ink-2); text-decoration: none; font-size: .86rem; transition: background .12s, color .12s; }
+ .navlink:hover { background: var(--paper-3); color: var(--ink); }
+ .navlink.active { background: var(--rust-bg); color: var(--rust); font-weight: 600; }
+ .navlink .ico { width: 1rem; flex: none; text-align: center; font-size: .8rem; opacity: .8; }
+ .navlink .txt { flex: 1; min-width: 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
+ .navcount { font-family: var(--mono); font-size: .68rem; color: var(--ink-3);
+ font-variant-numeric: tabular-nums; }
+ .navlink.active .navcount { color: var(--rust); }
+ /* Specificity, not order: `.navlink.active .navcount` is 0-3-0 and would otherwise repaint this
+ badge in --rust on a --rust fill — an invisible count, precisely while something needs review. */
+ .navcount.act, .navlink.active .navcount.act { background: var(--rust); color: var(--paper);
+ border-radius: 999px; padding: .04rem .4rem; font-weight: 700; }
+ .navlink.sub { padding-left: 1.7rem; font-size: .82rem; }
+ .navfoot { border-top: 1px solid var(--line); padding-top: .8rem; margin-top: .3rem;
+ font: .7rem/1.7 var(--mono); color: var(--ink-3); }
+ .navfoot b { color: var(--ink-2); font-weight: 600; }
+
+ /* ---- folder + card grid ---- */
+ .crumbs { display: flex; align-items: center; gap: .4rem; font: .72rem/1 var(--mono);
+ letter-spacing: .05em; text-transform: uppercase; color: var(--ink-3); margin-bottom: .5rem; }
+ .crumbs a { color: var(--rust); text-decoration: none; }
+ .crumbs a:hover { text-decoration: underline; }
+ .grid { display: grid; grid-template-columns: repeat(auto-fill, minmax(17rem, 1fr)); gap: .8rem;
+ margin-top: 1rem; }
+ .fold { display: block; text-align: left; width: 100%; cursor: pointer; font: inherit;
+ background: var(--paper-2); border: 1px solid var(--line); border-radius: 12px;
+ padding: .95rem 1.05rem; text-decoration: none; color: inherit; box-shadow: var(--shadow-sm);
+ transition: border-color .12s, transform .12s; }
+ .fold:hover { border-color: var(--rust); transform: translateY(-1px); }
+ .fold .fname { display: flex; align-items: center; gap: .45rem; font-family: var(--mono);
+ font-size: .92rem; color: var(--rust); }
+ .fold .fcount { margin-left: auto; font-size: .72rem; color: var(--ink-3); }
+ .fold .fnote { font-size: .78rem; color: var(--ink-2); margin-top: .35rem; }
+ .fold .fbar { display: flex; gap: .3rem; margin-top: .6rem; font: .66rem/1 var(--mono);
+ color: var(--ink-3); }
+
+ .scard { background: var(--paper-2); border: 1px solid var(--line); border-radius: 12px;
+ display: flex; flex-direction: column; box-shadow: var(--shadow-sm);
+ transition: border-color .12s; }
+ .scard:hover { border-color: var(--line-2); }
+ .scard.pend { border-color: var(--rust); background: var(--rust-bg); }
+ .scard .open { flex: 1; text-align: left; background: none; border: 0; cursor: pointer;
+ font: inherit; color: inherit; padding: .85rem .95rem .6rem; border-radius: 12px 12px 0 0; }
+ .scard .sname { font-family: var(--mono); font-size: .86rem; color: var(--ink);
+ overflow-wrap: anywhere; }
+ .scard .chips { display: flex; flex-wrap: wrap; gap: .3rem; margin-top: .45rem; }
+ .scard .sdesc { font-size: .78rem; color: var(--ink-2); margin: .5rem 0 0; line-height: 1.5;
+ display: -webkit-box; -webkit-line-clamp: 3; -webkit-box-orient: vertical; overflow: hidden; }
+ .scard .foot { display: flex; flex-wrap: wrap; gap: .4rem; padding: 0 .95rem .8rem; }
+ .scard .foot .btn { font-size: .68rem; padding: .3rem .6rem; }
+
+ /* ---- categories ---- */
+ .cluster-meta { font: .72rem/1.8 var(--mono); color: var(--ink-3); margin-bottom: .9rem;
+ font-variant-numeric: tabular-nums; }
+ .cluster-meta b { color: var(--ink-2); }
+ /* The scatter places a hundred points from a projection that bunches most of them centrally, so
+ picking a bucket by hunting a dot is unreliable. These are the actual control; the plot shows
+ how the library sits around whatever is picked. */
+ .cluster-chips { display: flex; flex-wrap: wrap; gap: .4rem; margin-bottom: .9rem; }
+ .cchip { display: inline-flex; align-items: center; gap: .4rem; cursor: pointer; font: inherit;
+ background: var(--paper-2); border: 1px solid var(--line); border-radius: 999px;
+ padding: .25rem .7rem; font-size: .76rem; color: var(--ink-2); transition: border-color .12s; }
+ .cchip:hover { border-color: var(--line-2); color: var(--ink); }
+ .cchip[aria-selected="true"] { border-color: currentColor; color: var(--ink);
+ background: var(--paper-3); font-weight: 600; }
+ .cchip .swatch { width: .55rem; height: .55rem; border-radius: 50%; flex: none; }
+ .cchip .n { font-family: var(--mono); font-size: .68rem; color: var(--ink-3);
+ font-variant-numeric: tabular-nums; }
+ .cluster-split { display: grid; grid-template-columns: minmax(0, 1fr) 19rem; gap: 1rem;
+ align-items: start; }
+ .cluster-plot { background: var(--paper-2); border: 1px solid var(--line); border-radius: 14px;
+ padding: .6rem; box-shadow: var(--shadow-sm); }
+ .cluster-plot svg { display: block; width: 100%; height: auto; }
+ .cbub { cursor: pointer; transition: opacity .12s; }
+ .cbub:hover { opacity: .85; }
+ .clabel { font: 600 10px var(--mono); fill: var(--ink-3); pointer-events: none;
+ paint-order: stroke; stroke: var(--paper-2); stroke-width: 3px; }
+ .clabel.on { fill: var(--ink); font-size: 11.5px; }
+ .cdot { pointer-events: none; }
+ .cluster-side { background: var(--paper-2); border: 1px solid var(--line); border-radius: 14px;
+ padding: 1rem 1.1rem; box-shadow: var(--shadow-sm); }
+ .cluster-side h3 { font: 600 .95rem/1.3 var(--sans); margin: 0 0 .2rem; }
+ .cluster-side .cterms { font: .72rem/1.7 var(--mono); color: var(--ink-3); margin-bottom: .7rem; }
+ .cluster-side ul { list-style: none; margin: 0; padding: 0; max-height: 22rem; overflow-y: auto; }
+ .cluster-side li { border-top: 1px solid var(--line); }
+ .cluster-side li button { width: 100%; text-align: left; background: none; border: 0;
+ cursor: pointer; font: .78rem/1.5 var(--mono); color: var(--ink-2); padding: .35rem .1rem; }
+ .cluster-side li button:hover { color: var(--rust); }
+ .cluster-side li button .gone { color: var(--warn); font-size: .68rem; }
+ @media (max-width: 1040px) { .cluster-split { grid-template-columns: 1fr; } }
+
+ /* ---- local traces ---- */
+ .trace-note { font: .74rem/1.7 var(--mono); color: var(--ink-3); max-width: 62rem; }
+ .trace-note b { color: var(--ink-2); }
+ .trace-controls { display: grid; grid-template-columns: minmax(10rem, 1.2fr) minmax(9rem, .8fr)
+ minmax(9rem, .7fr) minmax(9rem, .7fr) auto; gap: .7rem; margin-top: 1rem;
+ align-items: end; }
+ .trace-preview-toggle { display: flex; align-items: center; gap: .45rem; min-height: 2.35rem;
+ padding: 0 .15rem; white-space: nowrap; font: .67rem/1.3 var(--mono); color: var(--ink-2); }
+ .trace-preview-toggle input { width: 1rem; height: 1rem; margin: 0; }
+ .trace-stats { display: grid; grid-template-columns: repeat(auto-fit, minmax(10rem, 1fr));
+ border: 1px solid var(--line); border-radius: 12px; background: var(--paper-2);
+ margin-top: 1rem; box-shadow: var(--shadow-sm); }
+ .trace-stat { padding: .9rem 1rem; border-right: 1px solid var(--line); }
+ .trace-stat:last-child { border-right: 0; }
+ .trace-stat .n { display: block; font: 600 1.45rem/1 var(--mono); color: var(--ink); }
+ .trace-stat .l { display: block; margin-top: .35rem; font: .63rem/1.4 var(--mono);
+ color: var(--ink-3); letter-spacing: .08em; text-transform: uppercase; }
+ /* Harness x model matrix. Each model owns a column: the fixed harness ledger lets an evaluator
+ compare like with like while the evidence plates keep absence distinct from a zero effect. */
+ .mx-controls { display: grid; grid-template-columns: minmax(12rem, .5fr) auto; gap: .7rem;
+ margin-top: 1rem; align-items: end; }
+ .mx-warn { margin-top: 1rem; padding: .85rem 1rem; border-radius: 12px; font: .75rem/1.65 var(--mono);
+ border: 1px solid var(--warn); background: var(--warn-bg); color: var(--warn); }
+ .mx-warn b { display: block; margin-bottom: .25rem; letter-spacing: .06em; text-transform: uppercase;
+ font-size: .64rem; }
+ .mx-wrap { margin-top: 1rem; border: 1px solid var(--line); border-radius: 12px;
+ background: var(--paper-2); box-shadow: var(--shadow-sm); overflow-x: auto; position: relative; }
+ .mx { width: var(--mx-width, 100%); min-width: 100%; border-collapse: separate; border-spacing: 0;
+ table-layout: fixed; font: .74rem/1.5 var(--mono); }
+ .mx td, .mx th { min-width: 12.5rem; text-align: left; padding: .7rem .9rem;
+ border-bottom: 1px solid var(--line); color: var(--ink-2); vertical-align: middle; }
+ .mx th { font-size: .63rem; letter-spacing: .08em; text-transform: uppercase; color: var(--ink-3);
+ white-space: normal; background: var(--paper-2); }
+ .mx-sticky { position: sticky; left: 0; z-index: 2; min-width: 13.5rem !important;
+ background: var(--paper-2) !important; box-shadow: 1px 0 0 var(--line); }
+ .mx-corner { z-index: 3; color: var(--ink-3) !important; }
+ .mx-harness { color: var(--ink) !important; white-space: nowrap; }
+ .mx-model { display: block; color: var(--ink-2); overflow-wrap: anywhere; }
+ .mx-count { display: block; margin-top: .35rem; font-size: .58rem; line-height: 1.45;
+ letter-spacing: 0; text-transform: none; color: var(--ink-3); }
+ .mx-column-warn { display: block; margin-top: .3rem; font-size: .58rem; line-height: 1.45;
+ letter-spacing: 0; text-transform: none; color: var(--warn); }
+ .mx-chart { margin-top: 1rem; padding: 1rem; border: 1px solid var(--line); border-radius: 12px;
+ background: var(--paper-2); box-shadow: var(--shadow-sm); }
+ .mx-chart-head { display: flex; flex-wrap: wrap; align-items: start; justify-content: space-between;
+ gap: .7rem 1.25rem; margin-bottom: .8rem; }
+ .mx-chart-head h3 { margin: 0; font: 600 1.08rem/1.3 var(--sans); color: var(--ink); }
+ .mx-chart-kicker { display: block; margin-top: .18rem; font: .66rem/1.55 var(--mono);
+ color: var(--ink-3); }
+ .mx-chart-stats { display: flex; flex-wrap: wrap; gap: .45rem; }
+ .mx-chart-stat { min-width: 6.4rem; padding: .45rem .6rem; border: 1px solid var(--line);
+ border-radius: 8px; background: var(--paper); }
+ .mx-chart-stat b { display: block; color: var(--ink); font: 600 .92rem/1 var(--mono);
+ font-variant-numeric: tabular-nums; }
+ .mx-chart-stat span { display: block; margin-top: .24rem; color: var(--ink-3);
+ font: .58rem/1.3 var(--mono); letter-spacing: .06em; text-transform: uppercase; }
+ .mx-chart-legend { display: flex; flex-wrap: wrap; gap: .4rem .9rem; margin: 0 0 .25rem;
+ padding: .6rem 0; border-top: 1px solid var(--line); border-bottom: 1px solid var(--line); }
+ .mx-chart-legend span { display: inline-flex; align-items: center; gap: .38rem;
+ color: var(--ink-2); font: .65rem/1.4 var(--mono); }
+ .mx-chart-legend i { width: .55rem; height: .55rem; border-radius: 50%; background: currentColor;
+ box-shadow: 0 0 0 2px var(--paper-2), 0 0 0 3px currentColor; }
+ .mx-chart-legend b { color: var(--ink-3); font-weight: 400; }
+ .mx-chart svg { display: block; width: 100%; height: auto; min-height: 15rem; }
+ .mx-grid { stroke: var(--line); stroke-width: 1; }
+ .mx-model-rail { stroke: var(--line-2); stroke-width: 1; stroke-dasharray: 2 4; }
+ .mx-zero { stroke: var(--ink-2); stroke-width: 1.5; }
+ .mx-zone-positive { fill: color-mix(in srgb, var(--pass) 5%, transparent); }
+ .mx-zone-negative { fill: color-mix(in srgb, var(--fail) 4%, transparent); }
+ .mx-axis, .mx-tick, .mx-legend { fill: var(--ink-3); font: 10px var(--mono); }
+ .mx-axis-title { fill: var(--ink-2); font: 11px var(--mono); }
+ .mx-size-mark { cursor: pointer; outline: none; }
+ .mx-size-mark .visible { stroke-width: 2; transition: transform .12s, stroke-width .12s; }
+ .mx-size-mark .hit { fill: transparent; stroke: none; }
+ .mx-size-mark:hover .visible, .mx-size-mark:focus-visible .visible,
+ .mx-size-mark.selected .visible { stroke: var(--ink); stroke-width: 3; transform: scale(1.18); }
+ .mx-mark-code { fill: var(--paper-2); font: 700 6px var(--mono); pointer-events: none;
+ text-anchor: middle; dominant-baseline: central; }
+ .mx-chart-detail { min-height: 3.2rem; display: flex; flex-wrap: wrap; align-items: center;
+ gap: .4rem 1rem; padding: .7rem .8rem; border-radius: 8px; background: var(--paper-3);
+ color: var(--ink-2); font: .68rem/1.5 var(--mono); }
+ .mx-chart-detail strong { color: var(--ink); font-size: .76rem; }
+ .mx-chart-detail .lift { color: var(--pass); font-size: .9rem; font-weight: 700; }
+ .mx-chart-detail .lift.down { color: var(--fail); }
+ .mx-chart-detail .lift.flat { color: var(--ink-3); }
+ .mx-chart-note { margin-top: .45rem; font: .62rem/1.55 var(--mono); color: var(--ink-3); }
+ .mx-chart-empty { margin-top: 1rem; padding: .85rem 1rem; border: 1px dashed var(--line-2);
+ border-radius: 12px; font: .72rem/1.6 var(--mono); color: var(--ink-3); }
+ .mx-cell { padding: .45rem .55rem !important; }
+ /* The plate carries a result, not a miniature dashboard: lift stays primary until provenance is
+ requested. A plain blank cell has no border or control because no run is evidence of absence. */
+ .mx-plate { display: block; width: 100%; padding: .48rem .55rem; border: 1px solid transparent;
+ border-radius: 7px; background: transparent; font: 600 .88rem/1 var(--mono); text-align: left;
+ cursor: pointer; }
+ .mx-plate:focus-visible { outline: 2px solid var(--rust); outline-offset: 2px; }
+ .mx-plate:hover { border-color: var(--line-2); background: var(--paper-3); }
+ .mx-measured { color: var(--ink); }
+ .mx-measured.up { color: var(--pass); }
+ .mx-measured.down { color: var(--fail); }
+ .mx-measured.flat { color: var(--ink-3); }
+ .mx-error { color: var(--fail); border-color: color-mix(in srgb, var(--fail) 22%, transparent); }
+ .mx-blank { color: var(--ink-3); text-align: center; font: .9rem/1 var(--mono); }
+ .mx-details { margin-top: .7rem; min-height: 3rem; padding: .75rem .9rem; border-left: 2px solid var(--line-2);
+ background: var(--paper-2); color: var(--ink-2); font: .7rem/1.65 var(--mono); }
+ .mx-details p { margin: 0; }
+ .mx-provenance { display: grid; grid-template-columns: repeat(auto-fit, minmax(8.5rem, 1fr));
+ gap: .45rem 1rem; margin-top: .55rem; }
+ .mx-provenance span { color: var(--ink-3); }
+ .mx-provenance b { display: block; color: var(--ink-2); font-weight: 500; }
+ @media (max-width: 640px) {
+ .mx td, .mx th { min-width: 10.75rem; }
+ .mx-sticky { min-width: 9.75rem !important; }
+ .mx-chart { padding: .8rem; }
+ .mx-chart-stats { width: 100%; }
+ .mx-chart-stat { flex: 1 1 5.5rem; min-width: 0; }
+ .mx-chart svg { min-width: 38rem; }
+ .mx-chart-plot { overflow-x: auto; }
+ }
+ .mx-foot { margin-top: .7rem; font: .7rem/1.7 var(--mono); color: var(--ink-3); }
+ .trace-split { display: grid; grid-template-columns: minmax(0, 1fr) 18rem; gap: 1rem;
+ align-items: start; margin-top: 1rem; }
+ .trace-list, .trace-skills { border: 1px solid var(--line); border-radius: 12px;
+ background: var(--paper-2); box-shadow: var(--shadow-sm); overflow: hidden; }
+ .trace-row { padding: .75rem .9rem; border-top: 1px solid var(--line); }
+ .trace-row:first-child { border-top: 0; }
+ .trace-task { font-size: .8rem; line-height: 1.5; color: var(--ink); overflow-wrap: anywhere; }
+ .trace-meta { display: flex; flex-wrap: wrap; gap: .35rem .7rem; margin-top: .35rem;
+ font: .66rem/1.4 var(--mono); color: var(--ink-3); }
+ .trace-skills { padding: .85rem 1rem; }
+ .trace-skills h3 { margin: 0 0 .55rem; font-size: .9rem; }
+ .trace-skills ol { margin: 0; padding: 0; list-style: none; }
+ .trace-skills li { display: flex; gap: .6rem; justify-content: space-between;
+ border-top: 1px solid var(--line); padding: .35rem 0; font: .7rem/1.4 var(--mono);
+ color: var(--ink-2); }
+ .trace-skills li span:last-child { color: var(--ink-3); }
+ @media (max-width: 900px) {
+ .trace-split { grid-template-columns: 1fr; }
+ .trace-controls { grid-template-columns: 1fr 1fr; }
+ .trace-stat { border-right: 0; border-bottom: 1px solid var(--line); }
+ .trace-stat:last-child { border-bottom: 0; }
+ }
+ @media (max-width: 560px) { .trace-controls { grid-template-columns: 1fr; } }
+
+ @media (max-width: 900px) {
+ .shell { grid-template-columns: 1fr; }
+ .sidebar { position: static; height: auto; border-right: 0; border-bottom: 1px solid var(--line);
+ padding: 1rem 1.4rem; }
+ .navgroup { margin-bottom: .9rem; }
+ .main { padding: 1.2rem 1.4rem 3rem; }
+ }
+
+ /* Fixed steps, not viewport-fluid: an operator reads this at one desk on one monitor, and a
+ heading that resizes with every pixel of window width just wobbles. One step down where the
+ column actually gets narrow. */
+ @media (max-width: 640px) {
+ .board-title { font-size: 1.9rem; }
+ .head { font-size: 1.55rem; }
+ }
/* ---- board head ---- */
.board-head { padding: 2rem 0 .3rem; }
@@ -101,10 +363,7 @@
.board-head .eyebrow .n { color: var(--rust); }
.board-head .eyebrow .side { color: var(--ink-3); }
.board-title { font-family: var(--serif); font-weight: 700; letter-spacing: -.02em;
- font-size: clamp(1.9rem, 4vw, 2.7rem); margin: .5rem 0 0; text-wrap: balance; }
- .board-sub { color: var(--ink-2); font-size: .96rem; margin: .6rem 0 0; max-width: 48rem; text-wrap: pretty; }
- .board-sub a { color: var(--rust); text-decoration: none; white-space: nowrap; }
- .board-sub a:hover { text-decoration: underline; }
+ font-size: 2.4rem; margin: .5rem 0 0; text-wrap: balance; }
/* ---- KPI strip ---- */
.kpis { display: grid; grid-template-columns: repeat(auto-fit, minmax(160px, 1fr));
@@ -133,11 +392,14 @@
.alert.warn::before { background: var(--warn); }
.alert.ok::before { background: var(--pass); }
.alert.info::before { background: var(--ink-3); }
+ .board-head[hidden] { display: none; }
.alert .atag { font-family: var(--mono); font-size: .74rem; font-weight: 700; letter-spacing: .02em;
white-space: nowrap; }
.alert.act .atag { color: var(--rust); } .alert.warn .atag { color: var(--warn); }
.alert.ok .atag { color: var(--pass); }
.alert .atxt { font-size: .84rem; color: var(--ink-2); }
+ .alert .atxt code { font-family: var(--mono); font-size: .92em; background: var(--paper-3);
+ color: var(--ink); border-radius: 5px; padding: .06rem .3rem; }
/* ---- section scaffold ---- */
.band { padding-top: 2.6rem; }
@@ -147,8 +409,7 @@
.band-head .n { color: var(--rust); }
.band-head .side { color: var(--ink-3); }
h2.head { font-family: var(--serif); font-weight: 700; letter-spacing: -.015em; line-height: 1.05;
- font-size: clamp(1.5rem, 3.5vw, 2.1rem); margin: .8rem 0 0; text-wrap: balance; }
- .sub { font-size: .9rem; color: var(--ink-2); margin: .5rem 0 0; max-width: 48rem; text-wrap: pretty; }
+ font-size: 1.9rem; margin: .8rem 0 0; text-wrap: balance; }
.chip { font-family: var(--mono); font-size: .66rem; letter-spacing: .05em; padding: .12rem .5rem;
border-radius: 6px; white-space: nowrap; text-transform: uppercase; font-weight: 600; }
@@ -195,9 +456,23 @@
.cmp-tbl .win { color: var(--pass); font-weight: 700; }
.cmp-tbl .lose { color: var(--fail); font-weight: 700; }
.cmp-actions { display: flex; align-items: center; gap: .7rem; margin-top: 1.1rem; flex-wrap: wrap; }
+ /* The promotion dialog is the last screen before an irreversible change, and its evidence grows
+ with the held-out set. Scroll the evidence under a pinned decision row rather than scrolling
+ the whole modal, which had already pushed Approve off the bottom edge at eight tasks. */
+ #cmp-overlay .cmp-modal { display: flex; flex-direction: column; overflow: hidden; }
+ #cmp-body { flex: 1 1 auto; overflow: auto; min-height: 0; }
+ #cmp-body > .margin { margin-bottom: 1.1rem; }
+ #cmp-overlay .cmp-actions { flex: none; margin-top: 0; padding-top: 1rem;
+ border-top: 1px solid var(--line); }
/* Read-only skill and revision explorer */
.skill-modal { width: min(900px, 96vw); }
+ .skill-tabs { display: flex; gap: .35rem; margin: -.2rem 0 1rem; border-bottom: 1px solid var(--line); }
+ .skill-tab { border: 0; border-bottom: 2px solid transparent; background: none; color: var(--ink-3);
+ cursor: pointer; padding: .45rem .65rem .6rem; font: 600 .68rem/1 var(--mono);
+ letter-spacing: .06em; text-transform: uppercase; }
+ .skill-tab[aria-selected="true"] { color: var(--rust); border-bottom-color: var(--rust); }
+ .skill-panel[hidden] { display: none; }
.version-toolbar { display: grid; grid-template-columns: minmax(14rem, 1fr) minmax(12rem, 1fr);
gap: .8rem; margin-bottom: 1rem; }
.version-field { display: flex; flex-direction: column; gap: .35rem; }
@@ -210,6 +485,34 @@
.version-pre { font-family: var(--mono); font-size: .74rem; line-height: 1.55; margin: 0;
background: var(--paper); border: 1px solid var(--line); border-radius: 12px; padding: .85rem 1rem;
overflow: auto; min-height: 16rem; max-height: 56vh; white-space: pre; color: var(--ink); }
+ .eval-head { display: flex; align-items: flex-start; justify-content: space-between; gap: .8rem;
+ flex-wrap: wrap; margin-bottom: .8rem; }
+ .eval-summary { display: flex; flex-wrap: wrap; gap: .4rem; }
+ .eval-groups { display: grid; gap: .65rem; }
+ .eval-group { border: 1px solid var(--line); border-radius: 11px; background: var(--paper); }
+ .eval-group > summary { cursor: pointer; padding: .7rem .8rem; color: var(--ink);
+ font: 600 .72rem/1.4 var(--mono); letter-spacing: .03em; }
+ .eval-group > summary .count { color: var(--ink-3); font-weight: 400; margin-left: .35rem; }
+ .eval-items { border-top: 1px solid var(--line); }
+ .eval-item { padding: .78rem .85rem; }
+ .eval-item + .eval-item { border-top: 1px solid var(--line); }
+ .eval-task { margin: 0; color: var(--ink); font-size: .82rem; line-height: 1.55; white-space: pre-wrap; }
+ .eval-meta { margin: .45rem 0 0; color: var(--ink-2); font: .71rem/1.55 var(--mono);
+ white-space: pre-wrap; }
+ .eval-checks { list-style: none; margin: .55rem 0 0; padding: 0; display: grid; gap: .3rem; }
+ .eval-checks li { color: var(--ink-2); font: .7rem/1.5 var(--mono); padding-left: .8rem;
+ border-left: 2px solid var(--line-2); }
+ .review-result { margin-top: 1rem; border-top: 1px solid var(--line); padding-top: 1rem; }
+ .review-result:empty { display: none; }
+ .review-result h3 { margin: 0 0 .65rem; font: 600 .76rem/1.3 var(--mono);
+ letter-spacing: .05em; text-transform: uppercase; }
+ .review-score { display: flex; align-items: baseline; gap: .6rem; margin-bottom: .75rem; }
+ .review-score strong { font: 600 1.7rem/1 var(--mono); color: var(--ink); }
+ .review-score span { color: var(--ink-3); font: .7rem/1.5 var(--mono); }
+ .finding { padding: .65rem .75rem; border-left: 3px solid var(--warn); background: var(--warn-bg);
+ border-radius: 0 9px 9px 0; margin-top: .45rem; }
+ .finding b { font: 600 .72rem/1.4 var(--mono); color: var(--ink); }
+ .finding p { margin: .25rem 0 0; color: var(--ink-2); font-size: .76rem; line-height: 1.5; }
@media (max-width: 620px) { .version-toolbar { grid-template-columns: 1fr; } }
/* ---- skills list ---- */
@@ -221,6 +524,13 @@
background: var(--paper-2); color: var(--ink); }
.filter-count { font: .68rem/1.4 var(--mono); color: var(--ink-3); align-self: center; }
.skilllist { margin-top: 1.1rem; border-top: 1px solid var(--line); }
+ /* Provenance folders: which of these skills did we write, and which came from elsewhere. */
+ .skill-folder + .skill-folder { margin-top: 1.4rem; }
+ .skill-folder-head { display: flex; flex-wrap: wrap; align-items: baseline; gap: .45rem .7rem;
+ padding: .55rem .45rem; border-bottom: 1px solid var(--line-2); background: var(--paper-2);
+ position: sticky; top: 0; z-index: 1; }
+ .skill-folder-title { font-weight: 600; color: var(--ink); letter-spacing: .01em; }
+ .skill-folder-note { color: var(--ink-3); font-size: .82rem; flex: 1 1 14rem; min-width: 0; }
.srow { display: flex; flex-wrap: wrap; align-items: baseline; gap: .5rem 1rem;
padding: .9rem .45rem; border-bottom: 1px solid var(--line); transition: background .15s ease-out; }
.srow.pend { background: var(--rust-bg); }
@@ -292,6 +602,14 @@
font-variant-numeric: tabular-nums; }
.metaline b { color: var(--ink-2); }
.metaline a { color: var(--rust); text-decoration: none; } .metaline a:hover { text-decoration: underline; }
+ .retro-evidence { margin-top: .9rem; border: 1px solid var(--line); border-radius: 12px;
+ background: var(--paper); padding: .8rem .9rem; }
+ .retro-evidence h4 { margin: 0 0 .55rem; font: 600 .68rem/1.3 var(--mono);
+ letter-spacing: .06em; text-transform: uppercase; color: var(--ink-2); }
+ .retro-evidence p { margin: .35rem 0; color: var(--ink-2); font-size: .8rem; line-height: 1.5; }
+ .retro-evidence ul { margin: .45rem 0 .65rem; padding-left: 1.15rem; color: var(--ink-2);
+ font-size: .76rem; line-height: 1.55; }
+ .retro-evidence code { font: .72rem/1.45 var(--mono); overflow-wrap: anywhere; }
.tokline { font-family: var(--mono); font-size: .78rem; color: var(--ink-2); margin-top: .5rem;
font-variant-numeric: tabular-nums; }
.tokline b { color: var(--ink); } .tokline .reg { color: var(--fail); font-weight: 600; }
@@ -303,6 +621,13 @@
.gate.ok { color: var(--pass); background: var(--pass-bg); }
.gate.block { color: var(--fail); background: var(--fail-bg); font-weight: 600; }
.gate.warn { color: var(--warn); background: var(--warn-bg); }
+ /* margin-vs-resolution sits directly under the two scores, because that pair of big numbers is
+ what a reviewer reads first and it is the part that overstates a thin result */
+ .margin { font-family: var(--mono); font-size: .74rem; line-height: 1.65; color: var(--ink-2);
+ margin-top: .7rem; font-variant-numeric: tabular-nums; text-wrap: pretty; }
+ .margin b { color: var(--ink); }
+ .margin.thin { color: var(--warn); border-left: 2px solid var(--warn); padding-left: .6rem; }
+ .margin.thin b { color: var(--warn); }
.metaline b.win { color: var(--pass); } .metaline b.lose { color: var(--fail); }
.review-actions { display: flex; align-items: center; gap: .7rem; margin-top: 1.2rem; flex-wrap: wrap; }
.review-actions .msg { font-family: var(--mono); font-size: .74rem; color: var(--pass); }
@@ -318,6 +643,20 @@
.risk-warning { grid-column: 1 / -1; font: 600 .74rem/1.5 var(--mono); color: var(--warn); }
.difflabel { font-family: var(--mono); font-size: .66rem; letter-spacing: .09em; text-transform: uppercase;
color: var(--rust); margin: 1.4rem 0 .4rem; }
+ .lane { margin: 0 0 1.6rem; }
+ .lane-row { display: grid; grid-template-columns: 1fr auto auto; gap: .6rem 1rem; align-items: baseline;
+ padding: .55rem .8rem; border: 1px solid var(--line); border-radius: 12px; background: var(--paper-2);
+ margin-bottom: .4rem; }
+ .lane-row.live { border-color: var(--warn); }
+ .lane-name { font: 600 .8rem/1.4 var(--mono); color: var(--ink-1); }
+ .lane-name .act { color: var(--ink-3); font-weight: 400; }
+ .lane-state { font: .64rem/1.3 var(--mono); letter-spacing: .06em; text-transform: uppercase;
+ color: var(--ink-3); }
+ .lane-row.live .lane-state { color: var(--warn); }
+ .lane-row.done .lane-state { color: var(--pass); }
+ .lane-row.stuck .lane-state { color: var(--fail); }
+ .lane-note { grid-column: 1 / -1; font: .7rem/1.5 var(--mono); color: var(--ink-3); }
+ .lane-empty { font: .72rem/1.5 var(--mono); color: var(--ink-3); }
pre.diff { font-family: var(--mono); font-size: .72rem; line-height: 1.55; margin: 0;
background: var(--paper-2); border: 1px solid var(--line); border-radius: 14px; padding: .8rem .95rem;
overflow: auto; max-height: 28rem; white-space: pre; color: var(--ink-2); }
@@ -350,8 +689,14 @@
.footer .k { color: var(--rust); }
.footer a { color: var(--ink-3); text-decoration: none; } .footer a:hover { color: var(--rust); }
- .js .reveal { opacity: 0; transform: translateY(12px); }
- .js .reveal.in { opacity: 1; transform: none; transition: opacity .6s ease-out, transform .6s cubic-bezier(.16,1,.3,1); }
+ /* No entrance on the sections themselves. This is a console an operator opens to decide
+ something, not a page that introduces itself, and the four bands are the four objects the
+ product has rather than a narrative to reveal. The scroll-reveal it replaces also gated
+ visibility on a transition: a transition does not run in a background tab or a headless
+ renderer, so the Skills section — every record the product holds — rendered at opacity 0
+ with its 8,600px of content still in the layout. Content is visible by default now, and
+ the only entrances left are the ones that carry state (the stagger over a list that just
+ arrived, the beam over a change that needs a human). */
/* staggered entrance for alert lines + skill rows, first paint only (structure: Magic UI animated-list) */
.js .list-in { animation: list-in .5s cubic-bezier(.16,1,.3,1) both; animation-delay: calc(var(--i, 0) * 70ms); }
@@ -359,12 +704,13 @@
/* review card: the decision surface. While a challenger waits, a slow terracotta beam
circles the border (structure: Magic UI border-beam, gradient square on an offset-path
- rect, masked to the 1px ring). Falls back to the plain border where unsupported. */
+ rect, masked to the 1px ring). Its layer is clipped because the offset square otherwise
+ widens the document at the 640px breakpoint. Falls back to the plain border where unsupported. */
.review-card { position: relative; border: 1px solid var(--line); border-radius: 16px;
padding: 1.15rem 1.25rem 1.25rem; margin-top: 1.1rem; background: var(--paper-2);
box-shadow: var(--shadow); }
.beam { display: none; pointer-events: none; position: absolute; inset: 0; border-radius: inherit;
- border: 1px solid transparent;
+ border: 1px solid transparent; overflow: hidden;
mask-image: linear-gradient(transparent, transparent), linear-gradient(#000, #000);
mask-clip: padding-box, border-box; mask-composite: intersect; }
.beam::before { content: ""; position: absolute; aspect-ratio: 1; width: 90px;
@@ -381,15 +727,157 @@
outline: 2px solid var(--rust); outline-offset: 2px; border-radius: 2px; }
@media (prefers-reduced-motion: reduce) {
html { scroll-behavior: auto; } * { transition: none !important; animation: none !important; }
- .js .reveal { opacity: 1; transform: none; }
.beam { display: none !important; }
}
+
+ /* ---- evidence cockpit visual system ---- */
+ body {
+ background:
+ radial-gradient(circle at 72% -18%, color-mix(in srgb, var(--rust) 11%, transparent), transparent 34rem),
+ var(--paper);
+ }
+ .cockpit-command {
+ min-height: 4rem; padding: .75rem 1.5rem; color: #E8EEF8;
+ background: color-mix(in srgb, #08111F 94%, transparent);
+ border-bottom-color: #1E2C42; box-shadow: 0 12px 32px rgb(3 8 18 / 24%);
+ backdrop-filter: saturate(1.3) blur(16px);
+ }
+ .cockpit-command .brand { align-items: center; gap: .7rem; }
+ .cockpit-command .brand .mark { width: 23px; height: 23px; color: #69A7FF; }
+ .cockpit-command .brand .wm { color: #FFFFFF; font-size: 1.2rem; font-weight: 700; }
+ .cockpit-command .brand .tag { color: #8190A8; }
+ .cockpit-command .lnk { color: #9AA8BC; }
+ .cockpit-command .pill { border: 1px solid #2B405F; background: #101D2F; }
+
+ .cockpit-shell { grid-template-columns: 17rem minmax(0, 1fr); max-width: none; min-height: calc(100vh - 4rem); }
+ .cockpit-rail {
+ top: 4rem; height: calc(100vh - 4rem); padding: 1.7rem 1rem 2rem;
+ color: #B8C4D6; background: linear-gradient(180deg, #0B1422 0%, #0A111C 100%);
+ border-right-color: #1C2A3D; box-shadow: inset -1px 0 rgb(255 255 255 / 2%);
+ }
+ .cockpit-rail .navgroup { margin-bottom: 1.75rem; }
+ .cockpit-rail .navgroup h3 { margin: 0 0 .65rem .75rem; color: #66758C; font-size: .59rem; }
+ .cockpit-rail .navlink {
+ position: relative; min-height: 2.55rem; margin: .16rem 0; padding: .58rem .72rem;
+ border: 1px solid transparent; border-radius: 10px; color: #9EACC0;
+ }
+ .cockpit-rail .navlink:hover { color: #F5F8FC; background: #111F32; border-color: #20314A; }
+ .cockpit-rail .navlink.active {
+ color: #FFFFFF; background: linear-gradient(90deg, #17345A, #122843);
+ border-color: #27507E; box-shadow: 0 8px 20px rgb(0 0 0 / 18%);
+ }
+ .cockpit-rail .navlink.active::before {
+ content: ""; position: absolute; left: -.35rem; top: .55rem; bottom: .55rem; width: 3px;
+ border-radius: 999px; background: #65A8FF; box-shadow: 0 0 14px #65A8FF;
+ }
+ .cockpit-rail .navlink .ico { color: #6F829E; }
+ .cockpit-rail .navlink.active .ico { color: #78B2FF; }
+ .cockpit-rail .navcount { color: #6F7F96; }
+ .cockpit-rail .navlink.active .navcount { color: #C9DFFF; }
+ .cockpit-rail .navfoot { margin: 1.4rem .55rem 0; padding-top: 1rem; border-color: #213047; color: #66758C; }
+ .cockpit-rail .navfoot b { color: #AEBBD0; }
+
+ .cockpit-workspace { width: 100%; max-width: 104rem; padding: 2rem clamp(1.5rem, 3vw, 3.5rem) 4rem; }
+ .cockpit-workspace > .view:not([hidden]) {
+ padding: clamp(1.25rem, 2.2vw, 2rem);
+ border: 1px solid var(--line); border-radius: 20px; background: color-mix(in srgb, var(--paper-2) 96%, transparent);
+ box-shadow: 0 1px 1px rgb(16 24 40 / 3%), 0 22px 60px rgb(22 38 67 / 9%);
+ }
+ .cockpit-workspace > .board-head {
+ margin: 0 0 1rem; padding: clamp(1.4rem, 2.2vw, 2rem); border: 1px solid var(--line);
+ border-radius: 20px; overflow: hidden;
+ background: linear-gradient(135deg, var(--paper-2), color-mix(in srgb, var(--rust-bg) 58%, var(--paper-2)));
+ box-shadow: 0 1px 1px rgb(16 24 40 / 3%), 0 22px 60px rgb(22 38 67 / 8%);
+ }
+ .cockpit-workspace .band { padding-top: 0; }
+ .cockpit-workspace .board-title, .cockpit-workspace h2.head {
+ font-family: var(--sans); font-weight: 720; letter-spacing: -.035em;
+ }
+ .cockpit-workspace h2.head { margin: 0 0 .3rem; font-size: clamp(1.65rem, 2.3vw, 2.25rem); }
+ .cockpit-workspace .crumbs { margin-bottom: .75rem; }
+ .cockpit-workspace .trace-note { max-width: 72rem; font-family: var(--sans); font-size: .84rem; }
+
+ .cockpit-workspace .kpis { gap: .7rem; margin-top: 1.5rem; border: 0; background: transparent; box-shadow: none; }
+ .cockpit-workspace .kpi {
+ min-height: 7rem; padding: 1rem 1.1rem; border: 1px solid var(--line) !important;
+ border-radius: 14px; background: color-mix(in srgb, var(--paper-2) 92%, transparent);
+ }
+ .cockpit-workspace .kpi.act { background: linear-gradient(145deg, var(--rust-bg), var(--paper-2)); }
+ .cockpit-workspace .attention { margin-top: .75rem; padding: .2rem .8rem; border: 1px solid var(--line);
+ border-radius: 14px; background: color-mix(in srgb, var(--paper-2) 80%, transparent); }
+
+ .cockpit-workspace .review-card, .cockpit-workspace .trace-list,
+ .cockpit-workspace .trace-skills, .cockpit-workspace .cluster-plot,
+ .cockpit-workspace .cluster-side, .cockpit-workspace .mx-wrap,
+ .cockpit-workspace .mx-chart, .cockpit-workspace .empty,
+ .cockpit-workspace .skill-run-log {
+ border-radius: 16px; border-color: var(--line-2); background: var(--paper-2);
+ box-shadow: 0 1px 2px rgb(16 24 40 / 4%), 0 12px 32px rgb(23 38 65 / 7%);
+ }
+ .cockpit-workspace .review-card { padding: 1.5rem; }
+ .cockpit-workspace .trace-stats { gap: .6rem; border: 0; background: transparent; box-shadow: none; }
+ .cockpit-workspace .trace-stat { border: 1px solid var(--line) !important; border-radius: 12px;
+ background: var(--paper-2); }
+ .cockpit-workspace .trace-row, .cockpit-workspace .srow, .cockpit-workspace .hrow {
+ padding: .9rem 1rem; transition: background .12s, border-color .12s;
+ }
+ .cockpit-workspace .trace-row:hover, .cockpit-workspace .srow:hover,
+ .cockpit-workspace .hrow:hover { background: var(--paper-3); }
+ .cockpit-workspace .skilllist, .cockpit-workspace .hlist { overflow: hidden; border: 1px solid var(--line);
+ border-radius: 16px; background: var(--paper-2); }
+ .cockpit-workspace .skill-folder-head { padding: .8rem 1rem; background: var(--paper-3); }
+
+ .cockpit-workspace .version-field label { font-size: .61rem; color: var(--ink-3); }
+ .cockpit-workspace input, .cockpit-workspace select, .cockpit-workspace textarea {
+ min-height: 2.7rem; border-radius: 10px !important; background: var(--paper-2) !important;
+ box-shadow: inset 0 1px 2px rgb(16 24 40 / 4%);
+ }
+ .cockpit-workspace .btn { min-height: 2.55rem; padding: .65rem 1rem; border-radius: 10px; }
+ .cockpit-workspace .btn.primary { box-shadow: 0 7px 18px color-mix(in srgb, var(--rust) 25%, transparent); }
+
+ .cockpit-workspace .mx-wrap { border-radius: 16px; }
+ .cockpit-workspace .mx th { background: var(--paper-3); }
+ .cockpit-workspace .mx td { height: 4.5rem; }
+ .cockpit-workspace .mx tr:hover td { background: color-mix(in srgb, var(--rust-bg) 35%, var(--paper-2)); }
+ .cockpit-workspace .mx-sticky { background: var(--paper-2) !important; }
+ .cockpit-workspace .mx tr:hover .mx-sticky { background: color-mix(in srgb, var(--rust-bg) 35%, var(--paper-2)) !important; }
+ .cockpit-workspace .mx-chart { padding: 1.25rem; }
+ .cockpit-workspace .mx-chart-detail { border: 1px solid var(--line); }
+ .cockpit-workspace .lane-row { padding: .75rem 1rem; border-radius: 12px; }
+
+ .cmp-overlay { backdrop-filter: blur(8px); }
+ .cmp-modal { border-radius: 20px; box-shadow: 0 32px 90px rgb(0 0 0 / 35%); }
+
+ @media (prefers-color-scheme: dark) {
+ .cockpit-workspace > .view:not([hidden]), .cockpit-workspace > .board-head {
+ box-shadow: 0 1px 0 rgb(255 255 255 / 2%) inset, 0 26px 80px rgb(0 0 0 / 28%);
+ }
+ }
+ @media (max-width: 900px) {
+ .cockpit-shell { grid-template-columns: 1fr; }
+ .cockpit-rail { position: static; height: auto; padding: .75rem 1rem; border-right: 0;
+ border-bottom: 1px solid #203049; }
+ .cockpit-rail .navgroup { display: flex; gap: .35rem; margin: 0 0 .35rem; overflow-x: auto; }
+ .cockpit-rail .navgroup h3, .cockpit-rail .navfoot { display: none; }
+ .cockpit-rail #nav-folders { display: contents; }
+ .cockpit-rail .navlink { flex: 0 0 auto; min-height: 2.35rem; padding: .48rem .72rem; }
+ .cockpit-rail .navlink.active::before { left: .7rem; right: .7rem; top: auto; bottom: -.2rem;
+ width: auto; height: 3px; }
+ .cockpit-workspace { padding: 1rem; }
+ }
+ @media (max-width: 640px) {
+ .cockpit-command { min-height: 3.5rem; padding: .65rem 1rem; }
+ .cockpit-command .brand .tag { display: none; }
+ .cockpit-workspace > .view:not([hidden]), .cockpit-workspace > .board-head { padding: 1rem;
+ border-radius: 15px; }
+ .cockpit-workspace .review-card, .cockpit-workspace .mx-chart { padding: .9rem; }
+ }
-