diff --git a/.env.example b/.env.example index 2bccaba..f1264dc 100644 --- a/.env.example +++ b/.env.example @@ -28,10 +28,20 @@ API_KEY= # MODEL_BASE_URL=http://172.17.0.1:8000/v1 # serving model only (agent, A/B, rollouts) # BASE_URL=http://172.17.0.1:11434/v1 # or everything: teacher + judge too -# Routing model + thresholds. The defaults are calibrated TOGETHER on a drafted routing eval; +# Routing model + thresholds. The defaults are calibrated TOGETHER on a drafted routing eval. +# CPU stays the safe default. NVIDIA hosts can opt into the larger Apache-2.0 Qwen3-Embedding-8B +# GPU server with: +# docker compose -f docker-compose.yml -f compose.gpu-embeddings.yml \ +# --profile gpu-embeddings up -d embed-gpu mcp +# The overlay sets these four values; set them directly only for an external compatible server: +# EMBED_BACKEND=remote +# EMBED_BASE_URL=http://embed-gpu:8080/v1 +# EMBED_REMOTE_MODEL=Qwen/Qwen3-Embedding-8B-GGUF +# EMBED_TIMEOUT_SECONDS=120 # if you override EMBED_MODEL, recalibrate the three scores with it (cosine distributions differ # per model. For the previous default BAAI/bge-small-en-v1.5 use 0.65 / 0.45 / 0.93): # EMBED_MODEL=onnx-community/Qwen3-Embedding-0.6B-ONNX # or any fastembed model name +# ROUTER_BODY_CHARS=1000 # approved body prefix embedded beside name + description; max 4000 # MIN_SCORE=0.53 # at/above -> routable match; below -> related band or novel # RELATED_SCORE=0.37 # floor of the "related" (compose/extend) band; below it a task # # is novel (weak/strong escalation) @@ -57,6 +67,51 @@ API_KEY= # MINE_MAX_JUDGE_CALLS=24 # new representative judge calls per run; <=0 removes the cap # MINE_CLUSTER_THRESHOLD=0.90 # task cosine at/above this shares a representative verdict +# Vault publisher handoff. Approval writes runs/publications at mode 0700, and the publisher +# (ops/systemd/ingot-publisher.service) reads those receipts from the host as an ordinary user. +# When both run on one machine, set these to that user so the two agree on who owns the receipts. +# Leaving them unset keeps the UI container as root, which is right for every deployment with no +# host publisher. Get them wrong and the failure is silent in the console: approvals queue, the +# publisher lists an empty directory, and nothing publishes. The publisher logs it at each poll. +# INGOT_UID=1000 +# INGOT_GID=1000 + +# Where mutable state lives: the served library, the review queue, publication receipts, evidence, +# snapshots, and eval task sets. Defaults to $XDG_STATE_HOME/ingot (else ~/.local/state/ingot), so +# a pip-installed Ingot keeps nothing inside site-packages where an upgrade would discard it. +# Compose sets each path explicitly to what it mounted. `ingot status` prints every resolved path, +# where it came from, and whether it is writable. +# INGOT_HOME moves all of them at once; the specific settings override it one at a time. +# INGOT_HOME=~/.local/state/ingot +# INGOT_LIBRARY=/srv/ingot/library # SKILLS_DIR is the deprecated name for this +# INGOT_RUNS=/srv/ingot/runs +# INGOT_TASKS=/srv/ingot/tasks + +# Published host ports. Both stay on loopback; only the host side moves. Set them when this box +# already runs something on 8000 or 8080, including a second Ingot stack. `0` asks the kernel for +# a free port, which is what scripts/managed_smoke.sh does: it reaches every container through +# `docker compose exec` and needs no host port at all. +# INGOT_MCP_PORT=8000 +# INGOT_UI_PORT=8080 + +# Publication backend. `local` (the default) publishes into ./vault, a Git repository on this +# machine: no network, no GitHub account, no `gh`. `forge` makes a merged pull request the +# publication authority instead, which anchors activation off-box at the cost of the air gap and +# requires compose.forge.yaml plus a ./vault that is a clone of the repository below. Setting the +# forge variables without selecting the backend is inert, and the publisher says so at startup. +# INGOT_PUBLISH_BACKEND=local +# INGOT_FORGE_REPOSITORY=owner/repo +# INGOT_FORGE_REMOTE=origin +# INGOT_FORGE_BRANCH=main + +# Delivery targets: where an approved revision is installed once the vault carries it. The vault is +# the managed-MCP library, so agents using Ingot's MCP server are already served; this is for an +# agent that reads a native skill directory on disk instead. Comma-separated `name=kind:path`; +# `filesystem` is the only kind you configure (the vault target is always present). The publisher +# creates each root at startup and refuses to start if it cannot. Names are yours -- Ingot knows +# nothing about what reads the directory. +# INGOT_DELIVERY_TARGETS=claude=filesystem:~/.claude/skills,codex=filesystem:~/.codex/skills + # Change-control UI login. Three modes; AUTH_MODE picks one (compose default: password). # See docs/sso.md. To share the UI beyond this machine, use the TLS front door: # `docker compose --profile lan up -d proxy` (docs/security.md "Network exposure"). diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 20efcc9..3a1ccb1 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -13,6 +13,39 @@ jobs: - name: Build image run: docker build -t ingot-mcp . - name: Run tests - run: docker run --rm -v "$PWD:/app" -w /app ingot-mcp python -m pytest tests -q + # git comes from the image now: the publisher container drives real repositories, so it is + # a runtime dependency rather than something only the tests need. + run: > + docker run --rm -v "$PWD:/app" -w /app ingot-mcp + python -m pytest tests -q - name: Smoke test Compose and Langfuse TLS run: ./scripts/compose_smoke.sh + + # The control-plane claim is that no non-publisher service can change what is served. Everything + # else that checks it -- tests/test_compose_managed.py, `docker compose config` -- reads the + # configuration. This job is the only one that watches the kernel refuse the write, so it is the + # one that has to pass before that claim is repeated anywhere. Make it a required check. + managed: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - name: One writer, and it is the publisher + run: ./scripts/managed_smoke.sh + - name: A removed read-only mount must fail the smoke test + # A check that cannot fail proves nothing. This deliberately breaks the invariant and + # requires the script to notice, so a later edit that quietly drops a `:ro` cannot leave a + # green job behind it. + run: | + python3 - <<'PY' + import pathlib + path = pathlib.Path("docker-compose.yml") + text = path.read_text() + broken = text.replace("./vault:/app/skills:ro", "./vault:/app/skills", 1) + assert broken != text, "no read-only served mount left to break" + path.write_text(broken) + PY + if MANAGED_SMOKE_PROJECT=ingot-managed-negative ./scripts/managed_smoke.sh; then + echo "the smoke test passed with a writable served mount; it is checking nothing" + exit 1 + fi + git checkout -- docker-compose.yml diff --git a/.gitignore b/.gitignore index ee200b8..2133f17 100644 --- a/.gitignore +++ b/.gitignore @@ -18,10 +18,51 @@ __pycache__/ .pytest_cache/ .hf_cache/ .venv/ +/.venv-harbor-gateway/ +.worktrees/ + +# packaging build artifacts (`pip install -e .`) +*.egg-info/ +build/ +dist/ # fetched skills are optional & not redistributed here (scripts/fetch_skills.sh) skills/* !skills/.gitkeep +# the managed stack's skill vault: its own Git repository, created by `ingot vault init` +vault/ + # eval task sets are runtime artifacts (auto-drafted or user-authored), not shipped opinions optimize/tasks/*.yaml +# ...except a hand-authored one, which is the measuring instrument rather than its output: the +# seeded working trees and weighted checklists are the experiment's design, and a matrix produced +# by a task set that no longer exists in the tree cannot be reproduced or argued with. +!optimize/tasks/build-loop.yaml +!optimize/tasks/adversarial-council-review.yaml +!optimize/tasks/assumption-audit.yaml +!optimize/tasks/auditing-economic-claims.yaml +!optimize/tasks/auditing-system-claims.yaml +!optimize/tasks/decomposing-skill-libraries.yaml +!optimize/tasks/forward-intro.yaml +!optimize/tasks/isolated-integration-fixtures.yaml +!optimize/tasks/live-caller-gate.yaml +!optimize/tasks/measurement-integrity.yaml +!optimize/tasks/memory-defrag.yaml +!optimize/tasks/memory-notes.yaml +!optimize/tasks/memory-reflect.yaml +!optimize/tasks/operating-accountability-loop.yaml +!optimize/tasks/oss-ready.yaml +!optimize/tasks/product-marketing.yaml +!optimize/tasks/prose-style-hemingway.yaml +!optimize/tasks/copywriting.yaml +!optimize/tasks/routing-economic-evidence.yaml +!optimize/tasks/skill-security.yaml +!optimize/tasks/skill-retrospective.yaml +!optimize/tasks/agentic-action-safety.yaml +!optimize/tasks/turning-buyer-notes-into-decisions.yaml +!optimize/tasks/unattended-overnight-ops.yaml +!optimize/tasks/writing-clearly-and-concisely.yaml +!optimize/tasks/linting-implementation-plans.yaml +!optimize/tasks/op-credentials.yaml +!optimize/tasks/slancha-cred.yaml diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 73d89b7..11b6e97 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -1,10 +1,10 @@ # Architecture -Ingot is a local-first change-control system for agent instructions. A skill folder is the unit of -change; every version of it is content-addressed, every proposed change is quarantined until a human -approves it, and every promotion is atomic and reversible. Routing exists to serve the approved -revision to an agent. Ingot is not a multi-tenant service, and the default Compose deployment -exposes only loopback ports. +Ingot is a local-first, air-gappable control plane for a team's skill library. A skill folder is the +unit of change; every version of it is content-addressed, every proposed change is quarantined until +a human approves it, and every promotion is atomic and reversible. Routing exists to serve the +approved revision to an agent. Ingot is not a multi-tenant service, and the default Compose +deployment exposes only loopback ports. ## The change-control pipeline @@ -36,11 +36,14 @@ exposes only loopback ports. 1. The bundled agent sends task and execution context to `route_and_load` over MCP. 2. The router refreshes the skill registry when files change, filters by harness, platform, scope, - tools, MCPs, activation, and trust, then ranks compatible descriptions. Description embeddings - are cached across refreshes. + tools, MCPs, activation, and trust, then ranks compatible skills by the stronger cosine score + from the description or a bounded document containing name, description, and approved + harness-specific body. Both representations share a 4,096-vector least-recently-used cache + across refreshes; description-only vectors remain authoritative for collision detection. 3. One response is authoritative for the direct `match` or explicit `related_match`, loaded body, - revision, root, body-free alternatives, and `novel` escalation signal. A related match is loaded - for compose-or-extend use. The agent uses the weak model unless `novel` is true. + revision, root, body-free alternatives, component scores, and `novel` escalation signal. A + related match is loaded for compose-or-extend use. The agent uses the weak model unless `novel` + is true. 4. The run is recorded to Langfuse (the default evals backend, or a Langfuse-compatible endpoint `LANGFUSE_*` points at); mining reads it back and has no local fallback. Hosted model calls use the configured OpenAI-compatible endpoint. OpenRouter calls always request ZDR providers. @@ -48,7 +51,7 @@ exposes only loopback ports. ## SkillOpt optimization SkillOpt optimization is a core product capability. It proposes changes but never activates them. -Runs start in the background (`optimize.loop`) or on demand from the UI, and every result enters the +Runs start in the background (`ingot.optimize.loop`) or on demand from the UI, and every result enters the same human review path. Mining reads every usable Langfuse trace by default, with `--limit N` available only as an explicit @@ -102,7 +105,7 @@ text components (`OPTIMIZE_COMPONENTS=body,file:`), diffed for review and ## Stores and ownership -`skills/` contains active skills. `optimize/tasks/` contains eval sets. `runs/pending/` contains one +`skills/` contains active skills. `ingot/optimize/tasks/` contains eval sets. `runs/pending/` contains one active review slot per skill, with displaced candidates archived. `runs/revisions/` contains rollback snapshots, plus a `.snapshots.json` index per skill that records when each revision was last snapshotted; it sits beside the snapshot directories, never inside one, so a rollback restores @@ -163,13 +166,13 @@ Promotion stages changes and restores the prior directory if the swap fails. Eve the displaced revision. Restore it from the UI's History section, or with: ```bash -docker compose run --rm --entrypoint python optimize -m optimize.promote rollback SKILL REVISION +docker compose run --rm --entrypoint python optimize -m ingot.optimize.promote rollback SKILL REVISION ``` -The `optimize` service's entrypoint is `python -m optimize.ab`, so the entrypoint override is what -makes the arguments reach `optimize.promote`. +The `optimize` service's entrypoint is `python -m ingot.optimize.ab`, so the entrypoint override is what +makes the arguments reach `ingot.optimize.promote`. -Operators should back up `skills/`, `runs/`, and `optimize/tasks/`. Container databases require +Operators should back up `skills/`, `runs/`, and `ingot/optimize/tasks/`. Container databases require normal volume backup procedures. ## Trust boundaries diff --git a/Dockerfile b/Dockerfile index 6f0977e..0f40294 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,6 +2,13 @@ FROM python:3.12-slim-bookworm WORKDIR /app +# The publisher's vault is a Git repository and every publication is a worktree, a commit and a +# fast-forward. slim does not ship git, so without this the one service that owns the served +# library cannot start. +RUN apt-get update \ + && apt-get install -y --no-install-recommends git \ + && rm -rf /var/lib/apt/lists/* + COPY requirements.txt . RUN pip install --no-cache-dir pip==26.1.2 \ && pip install --no-cache-dir -r requirements.txt @@ -32,12 +39,19 @@ ARG FALLBACK_EMBED_MODEL=BAAI/bge-small-en-v1.5 RUN python -c "from fastembed import TextEmbedding; TextEmbedding('${FALLBACK_EMBED_MODEL}')" ENV BAKED_FALLBACK_EMBED_MODEL=${FALLBACK_EMBED_MODEL} -COPY mcp_server ./mcp_server +# `mcp_server` and `optimize` live under `ingot/` and arrive with the COPY below. COPY agent ./agent -COPY optimize ./optimize COPY ui ./ui COPY skills ./skills +# The `ingot` console script. `--no-deps` keeps the installed set exactly the pinned +# requirements.txt above instead of re-resolving it from pyproject.toml, and `-e` points the script +# at the /app copies the services already run with `python -m`, so there is only ever one copy of +# the code in the image. +COPY pyproject.toml README.md ./ +COPY ingot ./ingot +RUN pip install --no-cache-dir --no-deps -e . + ENV PYTHONUNBUFFERED=1 -CMD ["python", "-m", "mcp_server.server"] +CMD ["python", "-m", "ingot.mcp_server.server"] diff --git a/PRODUCTION_SETUP.md b/PRODUCTION_SETUP.md index 2de7273..d442d22 100644 --- a/PRODUCTION_SETUP.md +++ b/PRODUCTION_SETUP.md @@ -209,7 +209,7 @@ Codex, with the exact rule in [Make skill loading part of the agent instructions ## Operations Back up all named Langfuse datastore volumes and the repository's `skills/`, `runs/`, and -`optimize/tasks/` directories. Pin image versions, review upgrades before applying them, and test +`ingot/optimize/tasks/` directories. Pin image versions, review upgrades before applying them, and test restore procedures. Monitor container health and disk usage: ```bash diff --git a/README.md b/README.md index 53bea85..e327160 100644 --- a/README.md +++ b/README.md @@ -4,14 +4,18 @@ Ingot, the mascot, handing skills out to AI agents

-**Evidence-gated change control for agent instructions.** +**Open-source release control for agent skills.** Quarantine, prove, approve, publish, +and roll back exact skill revisions. [![CI](https://github.com/SlanchaAI/ingot/actions/workflows/ci.yml/badge.svg)](https://github.com/SlanchaAI/ingot/actions/workflows/ci.yml) [![License: Apache 2.0](https://img.shields.io/github/license/SlanchaAI/ingot)](LICENSE) [![Python 3.12](https://img.shields.io/badge/Python-3.12-3776AB?logo=python&logoColor=white)](Dockerfile) [![Docker](https://img.shields.io/badge/Docker-Compose-2496ED?logo=docker&logoColor=white)](docker-compose.yml) -An agent's [skills](https://github.com/anthropics/skills) are instructions it follows. **Ingot** is a -local-first library and MCP server for individual developers who serve versioned skills, evaluate -optimizer-generated challengers, and require human promotion before those challengers replace live -instructions. +Every tool installs a skill. Ingot quarantines it. + +An agent's [skills](https://github.com/anthropics/skills) are instructions it follows, and they +arrive from anywhere: a marketplace, a teammate, an optimizer. **Ingot** is a local-first, +air-gappable control plane for a team's skill library. It serves an exact revision of each skill +over MCP, holds every proposed change in quarantine, and requires a human approval backed by +evidence before one reaches what agents load. **[SkillOpt integration](https://github.com/microsoft/SkillOpt)** learns from real agent traces, trains bounded instruction edits, compares them on held-out tasks, and produces an evidence-backed @@ -35,10 +39,19 @@ proposal. SkillOpt can propose a change but cannot activate one. installation fails. Restore any snapshot from the UI or CLI. - **Decisions produce a local audit trail.** Approvals, rejections, and rollbacks attempt to append metadata-only records after the transition. Audit-write failures are logged and do not roll back - the decision; the local trail is not tamper-proof. + the decision. Publication history is Git-backed, revision-bound, and externally anchorable in + forge mode; anyone with a shell on the machine can still rewrite the local trail. + +**One writer.** In the tracked stack the served library is a Git vault mounted read-only into every +service except the publisher, and the publisher only acts on an approved receipt. Approval does not +change what is served: it queues a receipt the publisher then commits and activates. `ingot status` +answers whether that actually holds for a given deployment, by asking the filesystem rather than +reading a claim back out of the configuration. -MCP serves the current contents of `skills/`. Direct edits, fetched skills, and copied or restored -folders bypass the proposal workflow and become active. Review them as trusted code before use. +The writable stack still exists, as `compose.dev.yaml`, and it reports `UNMANAGED`. Anything running +as that user can change what is served without an approval, so the guarantees above do not apply to +it. Skills that arrive by direct edit, `cp`, or a restored folder are trusted code either way — +review them as such. ## A recorded gated change @@ -69,7 +82,13 @@ through evidence review, deliberate promotion, and rollback. - **Local development by default.** Compose binds the public ports to localhost and password-gates the UI. The MCP endpoint has no built-in authentication, and Ingot is not a hardened multi-tenant service. Follow the production guide before sharing it beyond one trusted machine. -- **Easy.** A skill is a folder with a `SKILL.md`. Drop one in and it is live on the next request. +- **Offline by default.** The default publication backend is `local`: the vault is a Git repository + on this machine and publishing needs no network, no GitHub account, and no `gh`. Set + `INGOT_PUBLISH_BACKEND=forge` (see `compose.forge.yaml`) to make a merged pull request the + publication authority instead, which anchors activation off-box at the cost of the air gap. +- **Easy.** A skill is a folder with a `SKILL.md`. `ingot add file:./that-folder` quarantines a + local package. `ingot add github:OWNER/REPO --skill path/to/skill` fetches a public repository at + an exact commit and quarantines that package. Neither command publishes it. ## Quickstart @@ -79,20 +98,41 @@ Prerequisites: - Free localhost ports `8000`, `8080`, and `3100`. - An OpenRouter API key, or a reachable OpenAI-compatible Ollama or vLLM endpoint. -`scripts/fetch_skills.sh` copies unpinned third-party skills into the live `skills/` directory. For -this PDF demo, fetch only Anthropic's document skills. Review their instructions and per-skill -licenses in [Skill sources](docs/skill-sources.md) before running the fetch; add other sources after -the first run. +`scripts/fetch_skills.sh` clones unpinned third-party skills and quarantines each one for review. It +serves nothing: the library stays byte-identical until you approve a package. For this PDF demo, +fetch only Anthropic's document skills. Review their instructions and per-skill licenses in +[Skill sources](docs/skill-sources.md) before running the fetch; add other sources after the first +run. ```bash git clone https://github.com/SlanchaAI/ingot.git && cd ingot cp .env.example .env # set API_KEY, or point BASE_URL at Ollama or vLLM -scripts/fetch_skills.sh anthropics # fetch the document skills used by this demo -docker compose up -d --build # router (:8000), UI (:8080), Langfuse (:3100) +pip install -e . # the `ingot` command (Python 3.12+, PyYAML only) +ingot vault init vault # the Git vault the publisher owns +scripts/fetch_skills.sh anthropics # quarantine the document skills used by this demo +docker compose up -d --build # router (:8000), UI (:8080), publisher, Langfuse (:3100) docker compose ps +open http://localhost:8080 # approve `pdf`; the publisher commits and activates it docker compose run --rm agent "How do I merge several PDFs into one and add page numbers?" ``` +`ingot vault init` is idempotent and the publisher runs it on every start, so skipping it only +means the vault appears when the stack does. + +The whole loop also runs from a terminal, and none of it writes a served byte: + +```bash +ingot pending # what is waiting on a decision +ingot approve pdf # queue a publication receipt; the publisher activates it +ingot history pdf # snapshots, receipts, and the decision trail +ingot rollback pdf +ingot status # MANAGED, PENDING, DRIFTED, or UNMANAGED +``` + +`ingot status` compares what is served against what the last release receipt says should be served, +so an out-of-band edit to the library reports `DRIFTED` and the command exits non-zero. See +[Managed deployment](docs/managed-deployment.md). + A successful run names the route and the exact skill revision loaded before the answer. This is an excerpt from the recorded tutorial run; scores, hashes, token counts, and model output vary: @@ -115,9 +155,12 @@ The change-control UI at `localhost:8080` asks for a login; the compose default set `AUTH_MODE=open` explicitly. See [Privacy & security](docs/security.md#network-exposure). `docker compose up` brings up a self-hosted Langfuse (traces + experiment UI) alongside the router -and UI; trace mining reads from it and has no local fallback, so it fails loudly if no -Langfuse-compatible backend is reachable. To send traces to your own Langfuse without starting the -bundled containers, set `LANGFUSE_*` and use `docker-compose.external-langfuse.yml` as documented in +and UI. Langfuse remains the default mining source and fails loudly when unreachable. Historical +Claude Code and Codex transcripts can be normalized locally as a separate, explicit source; +external judging stays disabled until a mining run supplies an affirmative flag. See +[Local coding-agent transcripts](docs/mcp-integration.md#local-coding-agent-transcripts). To send +traces to your own Langfuse without starting the bundled containers, set `LANGFUSE_*` and use +`docker-compose.external-langfuse.yml` as documented in [Configuration](docs/configuration.md#using-your-own-langfuse-project). Backend, model, and gate settings live in [Configuration](docs/configuration.md). @@ -132,10 +175,10 @@ shows `SKILL.md` plus any bundled resources for the selected version. Browsing d revision being served. Each skill row shows the total number of active, pending, and snapshotted versions available in that explorer. -Find a skill with an eval set and click **Optimize with SkillOpt**. The UI immediately explains -that optimization can take a few minutes, disables the button, and opens the live generation log -directly beneath that skill. The log updates automatically, so progress stays attached to the -change you started instead of appearing in a page-level activity panel. +For a skill without measured tasks, click **Create eval set**. The teacher drafts a separate +train/holdout set and the live log stays attached to that skill. When the draft finishes, the card +enables **Optimize with SkillOpt**. Optimization can take a few minutes; its attached log updates +automatically and the resulting challenger remains quarantined. ![SkillOpt optimization progress shown directly beneath the tailwind skill](docs/ui-home.webp) @@ -219,7 +262,8 @@ Ingot does three things around your skill library: human promotion. Promotion is snapshotted and recoverable. - **Improve.** SkillOpt integration mines real traces for failing skills, trains bounded instruction edits with its reflective optimizer, and A/Bs the result on held-out tasks, leaving a reviewable - proposal that only a human can activate. + proposal that only a human can activate. Agents using `skill-retrospective` can also submit a + verified, revision-bound update through the MCP; it lands in the same inert review queue. The component map is in [docs/how-it-works.md](docs/how-it-works.md); deeper design in [ARCHITECTURE.md](ARCHITECTURE.md). @@ -232,6 +276,7 @@ The component map is in [docs/how-it-works.md](docs/how-it-works.md); deeper des | [How it works](docs/how-it-works.md) | Component map (MCP server, agent, optimizer, UI) | | [Configuration](docs/configuration.md) | Env reference, SkillOpt optimization, cross-model compatibility, eval task sets, Langfuse | | [The evidence gate](docs/evidence-gate.md) | The anti reward-hacking checks a reviewer relies on | +| [Managed deployment](docs/managed-deployment.md) | One writer, publication backends, recovery, and what the audit trail does not guarantee | | [Privacy & security](docs/security.md) | Zero-data-retention, network exposure, threat model | | [Sign in with Google (SSO)](docs/sso.md) | Domain-restricted login and roles for a shared deployment | | [Bring your own agent](docs/mcp-integration.md) | Use the MCP server from your own harness; tracing | diff --git a/agent/run.py b/agent/run.py index 0b34cc2..1c138b6 100644 --- a/agent/run.py +++ b/agent/run.py @@ -20,8 +20,8 @@ # Endpoint + ZDR handling is shared with the optimizer (single source of truth): OpenRouter # endpoints get the hardcoded zero-data-retention provider preference; MODEL_BASE_URL points this # serving role at a local vLLM/Ollama server instead (README: Privacy). -from optimize import (ZDR_PROVIDER, agent_model, api_key, client_kwargs, model_api_key, # noqa: E402,F401 - model_base_url, skillopt_model, teacher_base_url) +from ingot.optimize import (ZDR_PROVIDER, agent_model, api_key, client_kwargs, model_api_key, # noqa: E402,F401 + model_base_url, skillopt_model, teacher_base_url) MODEL = agent_model() @@ -276,7 +276,7 @@ async def main(task: str): routed = await _route(task, connected_tools, serving_tools) _print_route(routed) - from optimize import openrouter_key_missing + from ingot.optimize import openrouter_key_missing if openrouter_key_missing(): print("\n[agent] OPENROUTER_API_KEY not set, showing router proposals only.") print(" Set OPENROUTER_API_KEY in .env to run the deep agent (or point") diff --git a/compose.dev.yaml b/compose.dev.yaml new file mode 100644 index 0000000..cf798c1 --- /dev/null +++ b/compose.dev.yaml @@ -0,0 +1,52 @@ +# Development mode: an explicitly UNMANAGED stack. +# +# docker compose -f docker-compose.yml -f compose.dev.yaml up +# +# Every service gets the served library read-write again and the publisher is switched off. That is +# convenient and it is not the product: nothing here prevents a service, a script, or a shell from +# changing what is served without an approval, so quarantine and publication guarantees DO NOT +# APPLY to a stack started this way. It must never be the configuration used to substantiate the +# control-plane claim. `ingot status` reports UNMANAGED, and the `unmanaged` service below runs it +# once at startup so the reason is in the log rather than in a document nobody reads. +# +# The managed default is `docker compose up` with no -f at all. +services: + unmanaged: + build: . + command: ["ingot", "status"] + environment: + INGOT_MODE: dev # `ingot status` reports UNMANAGED for the deployment, not per skill + volumes: + - ./skills:/app/skills + + publisher: + deploy: + replicas: 0 # there is no single writer in development mode + + mcp: + environment: + INGOT_MODE: dev + volumes: + - ./skills:/app/skills + + optimize: + volumes: + - ./skills:/app/skills + + optimize-mine: + volumes: + - ./skills:/app/skills + + optimize-compat: + volumes: + - ./skills:/app/skills + + optimize-loop: + volumes: + - ./skills:/app/skills + + ui: + environment: + INGOT_MODE: dev + volumes: + - ./skills:/app/skills diff --git a/compose.forge.yaml b/compose.forge.yaml new file mode 100644 index 0000000..ed44140 --- /dev/null +++ b/compose.forge.yaml @@ -0,0 +1,26 @@ +# The opt-in GitHub publication lane. +# +# INGOT_FORGE_REPOSITORY=owner/repo docker compose -f docker-compose.yml -f compose.forge.yaml up +# +# Publication authority becomes a merged pull request in the configured repository rather than the +# local commit, so a receipt sits at `awaiting_merge` until the merge lands and the activation is +# anchored somewhere a local administrator cannot quietly rewrite. It also means the deployment is +# no longer air-gapped: the publisher needs the network, an authenticated `gh`, and a vault whose +# remote is the configured repository. +# +# The publisher refuses to start unless `gh` is present, authenticated, and the repository +# resolves — loudly, at startup, rather than on the first approval. +# +# ./vault must already be a clone of INGOT_FORGE_REPOSITORY. `ingot vault init` creates a local +# vault with no remote and cannot make one for you; clone it yourself first. +services: + publisher: + environment: + INGOT_PUBLISH_BACKEND: forge + INGOT_FORGE_REPOSITORY: ${INGOT_FORGE_REPOSITORY:?set INGOT_FORGE_REPOSITORY=owner/repo} + INGOT_FORGE_REMOTE: ${INGOT_FORGE_REMOTE:-origin} + INGOT_FORGE_BRANCH: ${INGOT_FORGE_BRANCH:-main} + volumes: + # `gh` reuses the host's credentials rather than taking a token of its own. Read-only: the + # publisher authenticates with them and must never rewrite them. + - ${GH_CONFIG_DIR:-${HOME}/.config/gh}:/root/.config/gh:ro diff --git a/compose.gpu-embeddings.yml b/compose.gpu-embeddings.yml new file mode 100644 index 0000000..c4a9266 --- /dev/null +++ b/compose.gpu-embeddings.yml @@ -0,0 +1,67 @@ +# Opt-in larger router model for NVIDIA hosts. Keep docker-compose.yml as the CPU-safe default: +# docker compose -f docker-compose.yml -f compose.gpu-embeddings.yml \ +# --profile gpu-embeddings up -d embed-gpu mcp +# +# Qwen3-Embedding-8B is the highest-ranked permissively licensed text model supported by the +# serving stack as checked on 2026-07-28. Q4_K_M keeps its weights inside the spare GPU capacity +# on the production multi-tenant host; routing quality still needs the project eval before this +# overlay becomes the production default. +services: + embed-gpu: + image: ghcr.io/ggml-org/llama.cpp:server-cuda13-b9445@sha256:f92150249e1913ef96e744b5d78f6291f0e4399a7925ffc7b1d0680d82506551 + command: + - --hf-repo + - Qwen/Qwen3-Embedding-8B-GGUF:Q4_K_M + - --embedding + - --pooling + - last + - --ctx-size + - "2048" + - --batch-size + - "2048" + - --ubatch-size + - "2048" + # Routing batches are short-lived and MCP serializes initialization; extra slots multiply + # context buffers on the shared production GPU without improving this caller. + - --parallel + - "1" + - --cache-ram + - "0" + - --n-gpu-layers + - "99" + - --host + - 0.0.0.0 + - --port + - "8080" + - --no-webui + deploy: + resources: + reservations: + devices: + # Fully-qualified CDI avoids Compose v5 probing absent AMD/Intel vendors for `gpus: + # all`. NVIDIA Container Toolkit 1.18+ keeps this device spec current on the host. + - driver: cdi + device_ids: ["nvidia.com/gpu=all"] + capabilities: [gpu] + healthcheck: + test: ["CMD", "curl", "-f", "http://localhost:8080/health"] + interval: 5s + timeout: 3s + retries: 360 + start_period: 30s + volumes: + - embed_models:/root/.cache/huggingface + profiles: ["gpu-embeddings"] + + mcp: + depends_on: + embed-gpu: + condition: service_healthy + environment: + EMBED_BACKEND: remote + EMBED_BASE_URL: http://embed-gpu:8080/v1 + EMBED_REMOTE_MODEL: Qwen/Qwen3-Embedding-8B-GGUF + EMBED_TIMEOUT_SECONDS: "120" + +volumes: + embed_models: diff --git a/docker-compose.yml b/docker-compose.yml index ec610cd..5035f9f 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -32,16 +32,51 @@ services: - ./ops/keycloak:/opt/keycloak/data/import:ro profiles: ["sso"] + # The one writer of the served library. Every other service mounts ./vault read-only, so an + # approved change reaches what is served through this process or not at all. The default backend + # is `local`: the vault is a Git repository on this machine and no network is involved. See + # compose.forge.yaml for the opt-in GitHub lane, and compose.dev.yaml for the writable stack. + # + # `ingot vault init` runs on every start and is idempotent, so a first `docker compose up` on an + # empty checkout is not a failure. Approval writes runs/publications at mode 0700, so this and + # the ui must run as the same user; both default to root and both follow INGOT_UID/INGOT_GID. + publisher: + build: . + user: "${INGOT_UID:-0}:${INGOT_GID:-0}" + command: ["sh", "-c", "ingot vault init /app/vault && exec python -m ingot.optimize.publisher --watch"] + environment: + INGOT_PUBLISH_BACKEND: ${INGOT_PUBLISH_BACKEND:-local} + # State is configuration, not a directory next to the code. Every service names what it + # mounted; a default would put the review queue and the receipts inside the image. + INGOT_VAULT_PATH: /app/vault + INGOT_LIBRARY: /app/vault + INGOT_RUNS: /app/runs + # INGOT_DELIVERY_TARGETS: claude=filesystem:/app/targets/claude + volumes: + - ./vault:/app/vault # the only writable mount of the served library in this stack + - ./runs:/app/runs # the approval receipts it consumes + # Native delivery: mount an agent's skill directory here and name it in + # INGOT_DELIVERY_TARGETS, and the publisher installs each approved revision into it as well + # as the vault (docs/managed-deployment.md, "Delivery targets"). Writable on purpose -- this + # is the destination, and the publisher is still the only thing that writes it. + # - ~/.claude/skills:/app/targets/claude + mcp: build: . - command: python -m mcp_server.server + # Creation proposals are mode 0600 in the same runs/ queue the UI reads. Keep both services on + # one configured host identity or MCP-authored proposals become invisible to review. + user: "${INGOT_UID:-0}:${INGOT_GID:-0}" + command: python -m ingot.mcp_server.server environment: HOST: 0.0.0.0 # bind the container interface (needed for the port publish); # host access stays localhost-only via the 127.0.0.1 mapping + INGOT_LIBRARY: /app/skills + INGOT_RUNS: /app/runs + INGOT_TASKS: /app/ingot/optimize/tasks ports: - - "127.0.0.1:8000:8000" # MCP streamable-HTTP, localhost only (see README: Network exposure) + - "127.0.0.1:${INGOT_MCP_PORT:-8000}:8000" # MCP streamable-HTTP, localhost only (see README: Network exposure) volumes: - - ./skills:/app/skills # promoted skills must reach the running server (hot reload) + - ./vault:/app/skills:ro # the served library, read-only: only the publisher may change it - ./runs:/app/runs # usage counts (runs/skill_usage.json) must reach the host runs/ the ui reads agent: @@ -52,6 +87,9 @@ services: - path: .env # OPENROUTER_API_KEY (see .env.example) required: false # a missing .env/key gets a friendly in-app message instead environment: + INGOT_LIBRARY: /app/skills + INGOT_RUNS: /app/runs + INGOT_TASKS: /app/ingot/optimize/tasks MCP_URL: http://mcp:8000/mcp LANGFUSE_BASE_URL: ${LANGFUSE_BASE_URL:-http://langfuse-web:3000} LANGFUSE_PUBLIC_KEY: ${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo} @@ -68,19 +106,22 @@ services: # Writes a quarantined pending record; it can never activate a skill. optimize: build: . - entrypoint: ["python", "-m", "optimize.ab"] + entrypoint: ["python", "-m", "ingot.optimize.ab"] env_file: - path: .env required: false environment: + INGOT_LIBRARY: /app/skills + INGOT_RUNS: /app/runs + INGOT_TASKS: /app/ingot/optimize/tasks MCP_URL: http://mcp:8000/mcp LANGFUSE_BASE_URL: ${LANGFUSE_BASE_URL:-http://langfuse-web:3000} LANGFUSE_PUBLIC_KEY: ${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo} LANGFUSE_SECRET_KEY: ${LANGFUSE_SECRET_KEY:-sk-lf-local-demo} volumes: - - ./skills:/app/skills + - ./vault:/app/skills:ro # read-only: only the publisher may change what is served - ./runs:/app/runs - - ./optimize/tasks:/app/optimize/tasks # auto-drafted eval sets must survive the container + - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # auto-drafted eval sets must survive the container - /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers (README: execution-grounded judging) depends_on: mcp: @@ -92,18 +133,21 @@ services: # Success/failure mining over real Langfuse traces: `docker compose run --rm optimize-mine pdf` optimize-mine: build: . - entrypoint: ["python", "-m", "optimize.mine"] + entrypoint: ["python", "-m", "ingot.optimize.mine"] env_file: - path: .env required: false environment: + INGOT_LIBRARY: /app/skills + INGOT_RUNS: /app/runs + INGOT_TASKS: /app/ingot/optimize/tasks LANGFUSE_BASE_URL: ${LANGFUSE_BASE_URL:-http://langfuse-web:3000} LANGFUSE_PUBLIC_KEY: ${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo} LANGFUSE_SECRET_KEY: ${LANGFUSE_SECRET_KEY:-sk-lf-local-demo} volumes: - - ./skills:/app/skills + - ./vault:/app/skills:ro # read-only: only the publisher may change what is served - ./runs:/app/runs # usage counts + mined-candidate output - - ./optimize/tasks:/app/optimize/tasks # train sets feed the mined-candidate leakage guard + - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # train sets feed the mined-candidate leakage guard - /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers (README: execution-grounded judging) depends_on: langfuse-web: @@ -115,14 +159,18 @@ services: # Set COMPAT_MODELS=modelA,modelB,... in .env. Langfuse-free (local rollout + judge). optimize-compat: build: . - entrypoint: ["python", "-m", "optimize.compat"] + entrypoint: ["python", "-m", "ingot.optimize.compat"] env_file: - path: .env required: false + environment: + INGOT_LIBRARY: /app/skills + INGOT_RUNS: /app/runs + INGOT_TASKS: /app/ingot/optimize/tasks volumes: - - ./skills:/app/skills + - ./vault:/app/skills:ro # read-only: only the publisher may change what is served - ./runs:/app/runs # writes the compatibility matrix to runs/compat/ - - ./optimize/tasks:/app/optimize/tasks # held-out task sets (auto-drafted if missing) + - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # held-out task sets (auto-drafted if missing) - /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers profiles: ["optimize"] @@ -130,19 +178,22 @@ services: # for review: `docker compose run --rm optimize-loop` optimize-loop: build: . - entrypoint: ["python", "-m", "optimize.loop"] + entrypoint: ["python", "-m", "ingot.optimize.loop"] env_file: - path: .env required: false environment: + INGOT_LIBRARY: /app/skills + INGOT_RUNS: /app/runs + INGOT_TASKS: /app/ingot/optimize/tasks MCP_URL: http://mcp:8000/mcp LANGFUSE_BASE_URL: ${LANGFUSE_BASE_URL:-http://langfuse-web:3000} LANGFUSE_PUBLIC_KEY: ${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo} LANGFUSE_SECRET_KEY: ${LANGFUSE_SECRET_KEY:-sk-lf-local-demo} volumes: - - ./skills:/app/skills + - ./vault:/app/skills:ro # read-only: only the publisher may change what is served - ./runs:/app/runs - - ./optimize/tasks:/app/optimize/tasks # auto-drafted eval sets must survive the container + - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # auto-drafted eval sets must survive the container - /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers (README: execution-grounded judging) depends_on: mcp: @@ -154,13 +205,21 @@ services: # Change-control UI (review evidence, promote, roll back): http://localhost:8080 ui: build: . + # Approval writes runs/publications at mode 0700, and the vault publisher reads those receipts + # from the host as an ordinary user. Where both run on one machine, set INGOT_UID/INGOT_GID in + # .env to that user, or the publisher silently sees an empty queue and approvals never publish. + # The default keeps the container as root, which is every deployment that has no host publisher. + user: "${INGOT_UID:-0}:${INGOT_GID:-0}" command: python -m uvicorn ui.app:app --host 0.0.0.0 --port 8080 ports: - - "127.0.0.1:8080:8080" # localhost only; LAN access goes through the TLS proxy (--profile lan) + - "127.0.0.1:${INGOT_UI_PORT:-8080}:8080" # localhost only; LAN access goes through the TLS proxy (--profile lan) env_file: - path: .env required: false environment: + INGOT_LIBRARY: /app/skills + INGOT_RUNS: /app/runs + INGOT_TASKS: /app/ingot/optimize/tasks MCP_URL: http://mcp:8000/mcp # Change-control UI login. AUTH_MODE defaults to password (HTTP Basic) with the documented # default credentials admin/ingot, CHANGE AUTH_PASSWORD in .env before exposing the UI beyond @@ -184,9 +243,9 @@ services: LANGFUSE_SECRET_KEY: ${LANGFUSE_SECRET_KEY:-sk-lf-local-demo} LANGFUSE_PUBLIC_URL: ${LANGFUSE_PUBLIC_URL:-http://localhost:3100} # where the browser reaches Langfuse volumes: - - ./skills:/app/skills + - ./vault:/app/skills:ro # read-only: only the publisher may change what is served - ./runs:/app/runs - - ./optimize/tasks:/app/optimize/tasks # auto-drafted eval sets must survive the container + - ./ingot/optimize/tasks:/app/ingot/optimize/tasks # auto-drafted eval sets must survive the container - /var/run/docker.sock:/var/run/docker.sock # EXEC_SANDBOX=docker launches locked-down sandbox containers (README: execution-grounded judging) depends_on: mcp: diff --git a/docs/configuration.md b/docs/configuration.md index 25518f4..0c8e8fa 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -16,6 +16,11 @@ Set in `.env` (never committed): | `RELATED_SCORE` | `0.37` | floor of the `related` band; below it a task is novel (weak/strong escalation). Calibrated to `EMBED_MODEL` (0.45 for bge-small) | | `EMBED_MODEL` | `onnx-community/Qwen3-Embedding-0.6B-ONNX` | router embedding model (q4 ONNX, ~15 ms/query on CPU; +7 top-1 over the former bge-small default on a 297-query eval). The default Qwen files are baked into the image and loaded cache-only. Ingot does not use BGE-M3. Any fastembed name also works, but an unbaked override may download at first use and requires recalibrating the three score thresholds. Keep in sync with the Dockerfile's build arg | | `EMBED_ONNX_FILE` | `onnx/model_q4.onnx` | which ONNX weight file to load inside the `EMBED_MODEL` repo; only relevant for ONNX exports that ship multiple quantizations | +| `EMBED_BACKEND` | `local` | `local` uses the in-process CPU backend; `remote` uses an OpenAI-compatible embedding server and fails if it cannot be reached | +| `EMBED_BASE_URL` | unset | required for `remote`, including `/v1` (for example `http://embed-gpu:8080/v1`) | +| `EMBED_REMOTE_MODEL` | unset | required for `remote`; model identifier sent to the server, separate from the local `EMBED_MODEL` so GPU mode cannot accidentally request the CPU ONNX export | +| `EMBED_TIMEOUT_SECONDS` | `60` | remote embedding request timeout | +| `ROUTER_BODY_CHARS` | `1000` | approved body-prefix characters in each body-aware routing document. Name and description are always included; harness variants are embedded separately. Must be 1–4000. The hard ceiling prevents the decoder-style CPU ONNX attention tensor from exceeding 4 GB | | `BODY_TARGET_CHARS` | `6000` | length penalty starts past this body size | | `LENGTH_PENALTY` | `0.10` | max score subtracted for a very long body | | `LOOP_HEALTH_THRESHOLD` | `0.7` | the background loop proposes a change for skills whose mined mean score is below this | @@ -52,6 +57,12 @@ OIDC/SSO variables (`OIDC_ISSUER`, `OIDC_CLIENT_ID`, `OIDC_CLIENT_SECRET`, `OIDC Evidence-gate knobs (`PROMOTE_MIN_MARGIN`, `PROMOTE_MIN_SAMPLES`, `COLLISION_SCORE`, `JUDGE_MODELS`) are covered in [The evidence gate](evidence-gate.md). +Routing responses report `matched_on` (`description` or `content`) and both +cosine values under `score_components`. The aggregate is their maximum, so +existing description scores cannot decrease. Body evidence can create new +matches, however, so run a representative held-out routing suite before +changing thresholds or enabling a new library at scale. + ### SkillOpt optimization The body pass trains the skill body with **[SkillOpt](https://github.com/microsoft/SkillOpt)**'s @@ -99,12 +110,12 @@ the local rollout + judge, so it needs no Langfuse. ### Writing eval task sets Task sets are runtime artifacts, not shipped opinions; the repo commits none. They live in -`optimize/tasks/.yaml` (gitignored). Create one by hand or let `SKILLOPT_MODEL` auto-draft one +`ingot/optimize/tasks/.yaml` (gitignored). Create one by hand or let `SKILLOPT_MODEL` auto-draft one on the first CLI optimize run. To author one manually: -1. Create `optimize/tasks/.yaml`, where `` exactly matches the directory name under +1. Create `ingot/optimize/tasks/.yaml`, where `` exactly matches the directory name under `skills/`. 2. Add separate `train:` and `holdout:` lists. The candidate search sees only `train`; the evidence gate sees only `holdout`. A flat `tasks:` list is treated as train/holdout leakage and cannot diff --git a/docs/how-it-works.md b/docs/how-it-works.md index 8323459..41eb6f6 100644 --- a/docs/how-it-works.md +++ b/docs/how-it-works.md @@ -4,7 +4,7 @@ Ingot, the mascot, handing skills out to AI agents

-- **`mcp_server/`**: [FastMCP](https://github.com/jlowin/fastmcp) v3 server (HTTP transport), five tools: +- **`mcp_server/`**: [FastMCP](https://github.com/jlowin/fastmcp) v3 server (HTTP transport), six tools: - `suggest_skills(task, k)`: routable matches by embedding similarity (Qwen3-Embedding-0.6B q4 on CPU ONNX Runtime, no GPU; any fastembed model via `EMBED_MODEL`); near-misses come back flagged `related`; empty = truly novel @@ -14,6 +14,8 @@ - `route_and_load(task, harness, cwd, available_tools, available_mcps)`: one-round-trip selection and loading for direct or related compatible routes (see [Bring your own agent](mcp-integration.md#bring-your-own-agent-mcp-only)) + - `propose_skill_update(...)`: revision-bound `skill-retrospective` update submission; creates + one inert pending challenger and never activates or displaces instructions - **`agent/run.py`**: [deepagents](https://github.com/langchain-ai/deepagents) LangGraph agent wired to those tools, traced to Langfuse. Serves routed tasks on the weak `AGENT_MODEL` and escalates truly novel tasks to `STRONG_MODEL`. @@ -21,12 +23,15 @@ loads. Its folder's content hash is its revision. - **`optimize/promote.py`**: the change-control core, and the only module that writes under `skills/`: the pending queue, the evidence check, revision snapshots, the atomic promotion and - rollback swaps, and the approval-audit append. + rollback swaps, and the approval-audit append. When the serving revision comes from a read-only + mounted library, approval snapshots that source and atomically installs the challenger in the + first-precedence writable `skills/` root; the source mount remains unchanged. - **`optimize/`**: the SkillOpt integration and evaluation pipeline: trace mining (`mine.py`), multi-dimensional LLM judge (`judge.py`), the SkillOpt candidate search (`skillopt_loop.py` + `skillopt_bridge.py`) and its rollout/teacher plumbing (`rollout.py`), - held-out A/B (`ab.py`), the portable evidence bundle (`evidence.py`), the routing pass + held-out A/B (`ab.py`), the portable evidence bundle (`evidence.py`), retrospective proposal + ingestion (`retrospective.py`), the routing pass (`routing.py`), the background loop (`loop.py`), the library-wide routing health check (`routing_health.py`, embedding-only, cron/CI-friendly, read-only), token ledger (`usage.py`). None of these can activate anything: most write pending records; `routing_health.py` writes diff --git a/docs/managed-deployment.md b/docs/managed-deployment.md new file mode 100644 index 0000000..3dc73fe --- /dev/null +++ b/docs/managed-deployment.md @@ -0,0 +1,350 @@ +# Managed deployment + +The claim is that only an approved change reaches what is served. This page is how that is +enforced, how to check it on your own machine, and exactly what it does and does not protect +against. + +## One writer + +The served skill library is a Git repository — the vault. Every service mounts it read-only except +the publisher, which is the only process allowed to change it and only acts on an approved receipt. + +```text +ingot add file:./pkg quarantine the library is byte-identical +ingot add github:OWNER/REPO --skill path/to/pkg + quarantine the library is byte-identical +approve receipt written the library is byte-identical +publisher commit, activate the library now serves the approved revision +``` + +The whole loop from a terminal, none of which writes a served byte: + +```bash +ingot add file:./pkg # quarantine a package for review +ingot add github:OWNER/REPO --skill path/to/pkg +ingot review ./pkg # what is wrong with it, offline +ingot pending # what is waiting on a decision +ingot approve csv-tidy # queue a publication receipt +ingot history csv-tidy # snapshots, receipts, and the decision trail +ingot rollback csv-tidy +ingot status # is this deployment still what was approved? +``` + +`approve`, `reject`, and `rollback` call the same services the console calls. There is no second +approval path and no command that activates a skill directly. + +Approval is a human gate that writes a receipt. It does not touch the library. The publisher reads +the receipt, materializes exactly the components the receipt names, runs the vault's validator, +commits, snapshots the revision it is about to displace, fast-forwards the served checkout, and +re-verifies that what is served is the revision the receipt named. Any mismatch at any step fails +the receipt and changes nothing. + +`docker compose up` with no `-f` is the managed stack. That is deliberate: a default that launched +a writable stack while this page described controlled activation would make the claim untrue for +almost every reader. + +## Is it actually managed here? + +```bash +ingot status +ingot status --json +``` + +Four answers, decided per skill by comparing what is served against what the last successful +release receipt says should be served: + +| | Meaning | +|---|---| +| `MANAGED` | Every served skill is exactly the revision its release receipt names. | +| `PENDING` | A proposal or publication is in flight. Nothing has drifted. | +| `DRIFTED` | Served bytes differ from the last successful release. Something changed them outside the publisher. | +| `UNMANAGED` | Some served bytes have no release receipt behind them, or the deployment is in development mode. | + +The deployment reports the worst of them, and exits non-zero for anything but `MANAGED`, so a check +can assert it. This is an observation, not a configuration flag: a flag would have agreed with the +claim rather than tested it, which is the failure this command exists to catch. + +A skill with no release receipt is `UNMANAGED`, not `DRIFTED`. Fetched, copied, and hand-committed +skills are real and common; there is no release for them to have drifted from, and calling that +drift would make the alarm mean nothing. + +Whether the library is writable by the calling process is reported alongside the verdict but does +not decide it. The administrator who owns the vault can always write it, and a status command that +answered `UNMANAGED` from their shell would hide the drift they most need to see. + +## Drift, and what to do about it + +A read-only mount does not stop the machine owner from editing the host directory. Rather than +claim it does, Ingot detects it: + +```text +DRIFTED (uid 1000) + + Served bytes differ from the last successful release. Something changed them outside the publisher. + DRIFTED csv-tidy served 35e6c7a05313 != released e660639d81da +``` + +Two ways back to `MANAGED`: + +- **Restore the released bytes.** In the local backend the served checkout is the vault, so the + edit is an uncommitted change: `git -C vault checkout -- `. This is an explicit + administrator action on the vault, not something Ingot does behind the publisher's back. +- **Keep the change and get it approved.** Copy the edited directory somewhere else, restore the + vault, and submit the copy: `ingot add file:./that-copy`. It goes through review and approval + like any other proposal. + +Note the ordering. Until the vault checkout is clean the publisher refuses to run at all — a dirty +vault is exactly the state it must not build on — so a drifted deployment cannot publish its way +out. That is a deliberate refusal, not a deadlock: restoring first is one command. + +**Deliberately not built:** a single `ingot reconcile` that quarantines the drifted bytes for you. +It needs an answer to a question this design has not settled — a reconcile proposal's champion is +the last release while the disk holds the drifted bytes, so the existing freshness check refuses +it, and every way past that either fabricates evidence or opens a second approval path. The two +steps above do the same work with no new mutation path. + +## Development mode + +```bash +docker compose -f docker-compose.yml -f compose.dev.yaml up +``` + +Every service gets the library read-write again and the publisher is switched off. It is convenient +for working on Ingot itself. **Quarantine and publication guarantees do not apply to a stack +started this way**, and it must not be the configuration used to substantiate the control-plane +claim. The stack runs `ingot status` at startup so the reason is in the log. + +## Where state lives + +Nothing mutable is kept beside the code. The served library, the review queue, publication +receipts, evidence bundles, snapshots, and eval task sets all resolve through one setting: + +```text +INGOT_HOME # $XDG_STATE_HOME/ingot, else ~/.local/state/ingot +├── library/ # INGOT_LIBRARY (SKILLS_DIR is the deprecated name) +├── runs/ # INGOT_RUNS pending, publications, evidence, revisions +├── tasks/ # INGOT_TASKS +└── vault/ # INGOT_VAULT_PATH, defaults to the library +``` + +`INGOT_HOME` moves all of them; the specific settings override it one at a time, which is what the +compose stack does — every service names the paths it mounted rather than relying on a default. + +```bash +ingot status # prints every resolved path, where it came from, and whether it is writable +``` + +This is not cosmetic. A `pip install ingot` used to keep its review queue and its receipts inside +`site-packages`, which meant an upgrade discarded them, a read-only or system Python could not +start, and two deployments sharing one installation shared one queue. If `ingot status` finds state +left there by an earlier version it says so and does nothing else: moving a review queue on your +behalf is a change to controlled state made by a process nobody asked to make it. + +## Publication backends + +Selected explicitly with `INGOT_PUBLISH_BACKEND`, never inferred. A vault that later gains a remote +does not start opening pull requests on its own. + +### `local` (default) + +The vault is a Git repository on this machine. No network, no GitHub account, no `gh`, no remote +origin required. `ingot vault init ` creates one; the managed compose runs it on every start +and it is idempotent. + +States: `approved_publishing → publishing → active`. There is no `awaiting_merge`, because there is +nothing external to wait for — the human gate is the approval. + +The vault may have other legitimate writers; a person committing to it directly is fine and the +publisher fast-forwards onto their work. What the publisher will never do is rebase, merge, or +force. If a publication branch can no longer fast-forward, the receipt fails with an inspectable +error and nothing moves; the next attempt re-cuts the branch from the vault as it now stands and +re-checks the champion, so an unrelated commit resolves itself and a conflicting one is refused. + +### `forge` (opt-in) + +```bash +INGOT_FORGE_REPOSITORY=owner/repo \ + docker compose -f docker-compose.yml -f compose.forge.yaml up +``` + +Publication authority becomes a merged pull request. States: +`approved_publishing → publishing → awaiting_merge → active`. The vault must already be a clone of +the configured repository. The publisher verifies `gh` is present, authenticated, and that the +repository resolves — at startup, loudly, rather than on the first approval. + +This anchors activation somewhere a local administrator cannot quietly rewrite. It also ends the +air gap. + +| | `local` | `forge` | +|---|---|---| +| Network | none | required | +| Authority | the approval | a merged pull request | +| Activation record | local Git history | Git history, mirrored off-box | +| Air-gappable | yes | no | + +## Delivery targets + +The vault is the managed-MCP library: agents that load skills through Ingot's MCP server read the +same checkout the publisher commits into. An agent that reads a native skill directory on disk +reads nothing at all. A delivery target is that second destination. + +Configure them with `INGOT_DELIVERY_TARGETS`, a comma-separated list of `name=kind:path`: + +```sh +INGOT_DELIVERY_TARGETS=claude=filesystem:~/.claude/skills,codex=filesystem:~/.codex/skills +``` + +Two kinds: + +| Kind | What it is | +|---|---| +| `managed-mcp` | the vault itself, always present, always named `vault` unless you name it | +| `filesystem` | a directory the publisher installs approved revisions into | + +Ingot knows nothing about Codex or Claude beyond those names being yours to choose. A filesystem +target is a directory; what reads it is not Ingot's business. + +**What delivery does not change.** Publication stays receipt-driven and human-approved, and the +publisher stays the only supported writer. Delivery runs *after* the vault serves the approved +revision and *before* the receipt is marked `active`, so a target that cannot be written leaves a +release that retries rather than one that reports itself finished in places it never reached. Each +target's outcome is recorded on the receipt separately, under `delivery`. + +**The managed target is a deliberate no-op.** It has a name, a status, and a line on every receipt, +but the publication commit and the fast-forward are the only things that write the vault. A second +writer there is the one thing this control plane exists to prevent. + +**Installing is atomic.** The approved revision is staged beside the destination and swapped in with +same-filesystem renames. A failure between the two renames puts the displaced directory back, so an +agent never loads a skill folder that is neither the old revision nor the new one. Whatever the +target held is snapshotted first — keyed by the revision of the bytes actually there, so a target +someone edited by hand is recoverable too. + +**Rollback needs nothing extra.** It travels the ordinary publication queue, so every target returns +to the prior approved revision on the way through. + +**Drift is per target.** `ingot status` reports each one separately, and a drifted target counts +toward the overall verdict — a status that answered MANAGED while a native skill root served the +wrong bytes would be the lie the command exists to prevent. A target is graded only on the skills +Ingot released there: a native skill root is shared with whatever its owner put in it, and those are +not Ingot's to judge. + +`route_and_load` is unchanged and stays the managed-MCP delivery contract. A native agent activates +from its own skill directory in whatever way it already does; Ingot's router is not mandatory. + +## Recovery + +`process()` is re-entrant, and a kill at any point leaves a state the next pass resolves: + +| Killed | On restart | +|---|---| +| before the worktree is cut | re-prepared from scratch | +| worktree cut, before the commit | the stale worktree is destroyed and recut | +| after the commit, before activation | the branch is reused, re-authorized, activated | +| after activation, before the receipt | the receipt is finalized; **nothing is re-snapshotted** | +| after the receipt | no work; the stored state is returned | + +The fourth row is the one that matters. The served bytes already equal the candidate, so the +champion a second snapshot would capture is gone; re-snapshotting would refuse a publication that +has in fact already activated. + +## Artifact fidelity + +A revision names the exact package. `ingot add` stages every regular file byte-for-byte into a +**candidate tree** under `runs/candidates//`, and the receipt carries a manifest recording +each file's relative path, mode, size, and SHA-256 of its raw bytes. Publication copies that staged +tree into the vault worktree and verifies every hash on the way, so a file whose bytes moved between +the approval and the publication stops the publication instead of being served. + +Hashes are of bytes, never of decoded text: a file that is not valid UTF-8 has no decoded form, and +one that is would hash differently after a round trip — which is exactly how files used to go +missing. + +Two behaviours are deliberate, and both are visible rather than silent: + +- **SKILL.md is normalized, not preserved.** Its frontmatter is the routing interface, so the name + is forced to the skill's identity, the description is collapsed to one line, and the file is + re-emitted through a safe YAML dump. The manifest still records the source file's real hash and + size, so the normalization is auditable. The approved revision is computed by performing exactly + this materialization, so it is the revision the library serves. +- **Symlinks are refused.** `ingot review` reports `symlink-unsupported` and `ingot add` stops. + Preserving a link puts a path into the vault that leads a reader back out of the library; + flattening it into its target silently changes the artifact's shape. Neither is a decision + admission should make on an operator's behalf. + +For `github:`, acquisition resolves the public repository's `HEAD` to a commit before cloning and +refuses if the cloned commit differs. It inspects the selected Git tree before fetching its blobs, +rejects gitlinks and symlinks, then reads each blob without a checkout. Repository attributes and +checkout filters cannot rewrite or execute while those bytes enter quarantine. The candidate +records the repository, requested ref, commit, subdirectory, and tree digest. + +Assets a reviewer cannot read are reported rather than refused: + +```console +$ ingot review ./csv-tidy +structural + warning binary-asset: 1 file(s) are not text and cannot be read before approval; they will be + published byte-for-byte: assets/logo.png +``` + +Decodability decides, not the file extension — an extension is a claim about a file, and the point +is to check the file. Editor and VCS metadata (`.git/`, `__pycache__/`, `.DS_Store`) is not skill +content and is not reported; a finding that fires on `.DS_Store` is one people learn to scroll past. + +Modes are clamped to `0644` or `0755`, the two a Git checkout reproduces. A package is capped at +256 files and 20 MB; both refusals name the limit. + +Staged trees are named by their digest, so resubmitting the same package reuses one directory +rather than making a second copy. Nothing removes them afterwards: a rejected proposal leaves its +tree in `runs/candidates/`, bounded by the per-package cap, and deleting them is a housekeeping +decision rather than something publication should make on its own. + +## What the audit trail actually guarantees + +Publication history is Git-backed, revision-bound, and externally anchorable in `forge` mode. +Stated plainly, because a control plane that overstates this is worse than one that has none: + +- **Revision-bound.** Every revision is a digest of the exact package — the parsed SKILL.md plus the + raw bytes of every other file — so the receipt names specific bytes and a moved tag cannot stand + in for them. +- **Git-backed.** Each publication is a commit with the receipt id in its message, so what was + served when is reconstructable from history. +- **Detects normal inconsistency.** A champion that changed under a publication, a materialization + that does not match the approved revision, a staged candidate file whose bytes moved since + approval, a served checkout that does not match after activation, and a pending review that no + longer matches its receipt are all refused. +- **Externally anchorable** in `forge` mode, where the activation record exists somewhere the local + machine does not control. + +Git is not by itself proof that history did not change. A local repository can be rewritten by +anyone with a shell in it; a GitHub repository can be force-pushed or administratively altered, and +`forge` mode is an external anchor rather than an immutable transparency log. Someone with root can +rewrite the vault history, the receipts, and the audit log together. A signed log or a real +transparency log is the answer to that threat, and it waits for a concrete threat model rather than +being guessed at now. The records deliberately carry no signature field: a local record an +administrator can rewrite must not carry anything shaped like proof that they did not. + +## Proving it on your own machine + +```bash +scripts/managed_smoke.sh +``` + +Starts the managed stack and checks what Docker actually enforces: `mcp` and `ui` must fail to +write the served library, the publisher must succeed on the vault, what `mcp` serves must be the +commit the publisher's vault is at, and `ingot status` inside the stack must report `MANAGED`. + +`tests/test_compose_managed.py` checks the same invariant against the tracked YAML on every test +run, which catches a regression in the configuration but cannot prove the containers behave. + +CI runs `managed_smoke.sh` on a Linux runner as a required check, and then runs it again with one +`:ro` deliberately removed and requires it to fail. A check that cannot fail proves nothing. + +## Running the publisher on the host instead + +`ops/systemd/ingot-publisher.service` runs it as a user unit. That sidesteps the uid mismatch +between a container writing receipts at mode 0700 and a host process reading them, and in `forge` +mode it reuses the host's already authenticated `git` and `gh` so no credential has to live in a +container. Copy `ops/systemd/publisher.env.example` to `~/.config/ingot/publisher.env` first. + +Whichever you run — the compose service or the unit — exactly one must. diff --git a/docs/mcp-integration.md b/docs/mcp-integration.md index 19f7c93..fe35ba2 100644 --- a/docs/mcp-integration.md +++ b/docs/mcp-integration.md @@ -136,6 +136,11 @@ returned skill_body while completing the request. If it returns novel, continue Do not merely list or suggest the skill: load it and apply it before doing the task. ``` +When a loaded `skill-retrospective` produces a verified update for an existing skill, agents may +call `ingot.propose_skill_update`. The tool only files a revision-bound challenger in Ingot's +review queue. A `quarantined` response does not change the served revision: do not reload skills or +claim the update is active. Human approval in the console remains a separate action. + Use the same rule in organization-managed agent instructions if repositories should not carry local agent files. After enrollment, verify behavior with a harmless request and confirm both the `route_and_load` tool call and final answer appear in Langfuse. A successful `--doctor` result proves @@ -206,13 +211,46 @@ unchanged. Two caveats: mining re-judges traffic with `JUDGE_MODEL` (on your API candidate rollouts still execute on the bundled scaffold, so set `AGENT_MODEL` to your production serving model. +### Local coding-agent transcripts + +Existing Claude Code and Codex JSONL transcripts can feed the same miner without first uploading +them to Langfuse: + +```bash +python -m ingot.optimize.local_traces +python -m ingot.optimize.mine --source local --allow-external-judge +``` + +The scan writes `runs/local_traces.json`. It keeps completed human turns, final answers, observed +skill names, exact revisions returned by `ingot.route_and_load`, timing, token counts, and tool +error counts. It excludes reasoning, attachments, tool arguments and results, injected agent +instructions, hook output, compaction records, aborted turns, and subagent threads. A historical +skill use without a served revision stays unpinned; the scanner never substitutes the current +revision. + +Scanning is local and makes no model call. Repeated scans reuse unchanged transcript files. Bound +the snapshot at import time with repeatable `--project `, `--since YYYY-MM-DD`, and +`--until YYYY-MM-DD`; use `--force` after a same-size transcript rewrite whose mtime was preserved. +The console also filters the imported snapshot by project, agent, and date. + +Local mining fails closed unless `--allow-external-judge` is present. That flag is the explicit +paid/data-egress boundary: selected task/answer pairs go to `JUDGE_MODEL` under the normal usage +cap. Langfuse and local snapshots are separate sources rather than an implicitly merged corpus. +The console's Traces view reads a safe projection of the normalized store and never returns answer +text. Task previews are also hidden until the reviewer enables **Show task previews**; previews +are capped at 280 characters. + +When the console runs on another host, transfer the normalized file through the deployment's +existing trusted channel into that checkout's `runs/local_traces.json`. Do not mount or copy raw +home-directory transcripts into the UI container. + ## Using your own evals platform -Langfuse is the **default and required** evals backend: it comes up with `docker compose up`, and -trace mining has no local fallback (`optimize-mine` fails loudly if no Langfuse-compatible endpoint -is reachable, rather than returning an empty result that would read as "nothing failing"). You have -three options: +Langfuse is the **default online** evals backend: it comes up with `docker compose up`, and the +default trace source fails loudly if no Langfuse-compatible endpoint is reachable, rather than +returning an empty result that would read as "nothing failing". The local transcript source above +is an explicit historical backfill path, not an online backend. You have three online options: 1. **Bundled Langfuse** (default): self-hosted in the compose stack, nothing to configure. Secure its demo credentials before exposing it: [Securing the Langfuse deployment](security.md#securing-the-langfuse-deployment). diff --git a/docs/security.md b/docs/security.md index 1101944..4866921 100644 --- a/docs/security.md +++ b/docs/security.md @@ -52,6 +52,13 @@ Write paths, and what guards each: - **Generated rewrites** land in `runs/pending/` and cannot activate themselves. They also require evidence whose champion and challenger revisions still match the skill on disk before UI approval. +- **Retrospective MCP submissions** may create one bounded pending update for an existing skill. + They require passed pressure verification, bind to the exact loaded champion revision, refuse an + occupied review slot, treat candidate text and verification commands as inert data, and cannot + approve, reject, or reload anything. Its gate explicitly identifies retrospective evidence and + warns that no held-out A/B quality comparison ran. Exact retries are idempotent. MCP remains + unauthenticated, so this reversible proposal action is available only inside the same trusted + network boundary as the read tools. - **Approval and rollback** are the only application paths that write under `skills/`. Both go through `optimize/promote.py`, both snapshot what they displace, and both append an audit record on a best-effort basis (a failed append is logged and does not undo the committed change). @@ -70,7 +77,8 @@ change-control UI is password-gated by Compose, using the local demo login `admi supports OIDC for shared deployments. Loopback binding remains the first protection layer: - `docker-compose.yml` publishes every port on loopback only (`127.0.0.1:8000` MCP, - `127.0.0.1:8080` UI, `127.0.0.1:3100` Langfuse). + `127.0.0.1:8080` UI, `127.0.0.1:3100` Langfuse). `INGOT_MCP_PORT` and `INGOT_UI_PORT` move the + host port when the box already serves one of them; the loopback binding is not theirs to change. - Run outside Docker, the MCP server also binds `127.0.0.1` by default. To expose MCP, use a private interface override as shown in [Production setup](../PRODUCTION_SETUP.md) diff --git a/docs/tutorial.md b/docs/tutorial.md index b1b491d..0bdfea6 100644 --- a/docs/tutorial.md +++ b/docs/tutorial.md @@ -154,7 +154,7 @@ At this point you know what is wrong and could fix the body by hand. SkillOpt in other half of Ingot's value: it trains a bounded instruction revision from real failures and attaches measured evidence without activating the result. -Write an eval task set for the skill (`optimize/tasks/tailwind.yaml`) with train and holdout tasks +Write an eval task set for the skill (`ingot/optimize/tasks/tailwind.yaml`) with train and holdout tasks whose rubrics carry the v4 ground truth (the teacher can also auto-draft one on a skill's first CLI run). Then run it headless, which is how it is meant to run: @@ -266,8 +266,8 @@ That snapshot is the undo. It appears in the UI's **History** section, and resto click, or one command: ```bash -# --entrypoint python replaces the service's own `python -m optimize.ab` entrypoint -docker compose run --rm --entrypoint python optimize -m optimize.promote rollback tailwind +# --entrypoint python replaces the service's own `python -m ingot.optimize.ab` entrypoint +docker compose run --rm --entrypoint python optimize -m ingot.optimize.promote rollback tailwind ``` Rollback snapshots the revision it displaces too, so the round trip is symmetric, and it writes its @@ -285,7 +285,7 @@ displaced candidate is archived beside the slot (the run tells you where) rather The body is fixed, but step 3's third request still misroutes: the routing key is the `description`, so routing gets its own pass with its own metric, run against the `routing:` cases -in `optimize/tasks/tailwind.yaml`: realistic positive phrasings plus `expected: null` negatives. +in `ingot/optimize/tasks/tailwind.yaml`: realistic positive phrasings plus `expected: null` negatives. The cases that matter are the real misses, so put your mined traffic in the suite (we added the node_modules request verbatim, plus a "classes disappear in the production build" variant): @@ -364,7 +364,7 @@ against the real router plus a description-collision scan, embedding-only, no LL exits non-zero on problems, so it slots into cron or CI: ```bash -docker compose run --rm --entrypoint "python -m optimize.routing_health" optimize +docker compose run --rm --entrypoint "python -m ingot.optimize.routing_health" optimize # [health] tailwind: top1 1.000 · recall@3 1.000 · no-route precision 0.333 (7 cases) # [health] ✓ routing healthy: every suite passes and no descriptions collide. ``` diff --git a/evals/fixtures/skills/billing-runbook/SKILL.md b/evals/fixtures/skills/billing-runbook/SKILL.md new file mode 100644 index 0000000..8d49e51 --- /dev/null +++ b/evals/fixtures/skills/billing-runbook/SKILL.md @@ -0,0 +1,6 @@ +--- +name: billing-runbook +description: Operate a production service. +--- +Investigate invoice charges, subscription renewals, payment failures, credits, +refunds, and billing-account ownership. diff --git a/evals/fixtures/skills/kubernetes-runbook/SKILL.md b/evals/fixtures/skills/kubernetes-runbook/SKILL.md new file mode 100644 index 0000000..07d5ed2 --- /dev/null +++ b/evals/fixtures/skills/kubernetes-runbook/SKILL.md @@ -0,0 +1,6 @@ +--- +name: kubernetes-runbook +description: Operate a production service. +--- +Diagnose a Kubernetes pod stuck in CrashLoopBackOff. Inspect pod events, +container logs, probes, resource limits, and recent deployment changes. diff --git a/evals/routing.yaml b/evals/routing.yaml index 2ce0b65..5ce17b4 100644 --- a/evals/routing.yaml +++ b/evals/routing.yaml @@ -56,6 +56,10 @@ cases: expected: null harness: claude min_score: 0.99 + - task: Diagnose a Kubernetes pod stuck in CrashLoopBackOff. + expected: kubernetes-runbook + harness: codex + parity: true - task: Thanks, that answers my question. expected: null harness: codex diff --git a/ingot/__init__.py b/ingot/__init__.py new file mode 100644 index 0000000..6ab7110 --- /dev/null +++ b/ingot/__init__.py @@ -0,0 +1,6 @@ +"""Ingot's command surface. + +Deliberately empty of imports. `ingot.cli` must stay runnable with nothing installed beyond the +skill loader's own dependency, so anything that reaches for the server, the optimizer, or a model +belongs in the subcommand that needs it, imported inside the function.""" +__version__ = "0.2.0" diff --git a/ingot/acquire.py b/ingot/acquire.py new file mode 100644 index 0000000..a4e3b2c --- /dev/null +++ b/ingot/acquire.py @@ -0,0 +1,121 @@ +"""Fetch remote package bytes without admitting, reviewing, or executing them.""" +from __future__ import annotations + +import os +import re +import subprocess +from pathlib import Path + +from ingot.optimize.tree import MAX_FILES, MAX_TREE_BYTES, portable_path + +_REPOSITORY = re.compile( + r"^[A-Za-z0-9](?:[A-Za-z0-9-]{0,38})/[A-Za-z0-9](?:[A-Za-z0-9._-]{0,99})$") +_COMMIT = re.compile(r"^[0-9a-f]{40,64}$") + + +def _remote_url(repository: str) -> str: + return f"https://github.com/{repository}.git" + + +def _git(*args: str) -> bytes: + environment = {**os.environ, "GIT_TERMINAL_PROMPT": "0"} + try: + result = subprocess.run(["git", *args], capture_output=True, env=environment, + timeout=120) + except (OSError, subprocess.TimeoutExpired) as error: + raise ValueError(f"Git acquisition failed: {error}") from error + if result.returncode: + detail = result.stderr.decode("utf-8", errors="replace").strip() + raise ValueError(f"Git acquisition failed: {detail or 'git exited non-zero'}") + return result.stdout + + +def _resolved_commit(remote: str, ref: str) -> str: + output = _git("ls-remote", remote, ref) + rows = [line.split(b"\t", 1) for line in output.splitlines()] + matches = [sha.decode("ascii") for sha, name in rows + if name.decode("utf-8", errors="replace") == ref] + if len(matches) != 1 or not _COMMIT.fullmatch(matches[0]): + raise ValueError(f"Git acquisition failed: ref {ref!r} did not resolve to one commit") + return matches[0] + + +def _bounded_tree(repository: Path, subdirectory: str) -> list[tuple[str, str, str, int]]: + """Return safe entries after enforcing bounds, before asking Git for blob contents.""" + output = _git("-C", str(repository), "ls-tree", "-r", "-l", "-z", "HEAD", "--", + subdirectory) + entries, total = [], 0 + prefix = f"{subdirectory}/" + for record in output.split(b"\0"): + if not record: + continue + try: + metadata, raw_path = record.split(b"\t", 1) + mode, kind, object_id, raw_size = metadata.split() + path = raw_path.decode("utf-8") + except (UnicodeDecodeError, ValueError) as error: + raise ValueError("Git acquisition failed: the selected tree has an invalid entry") \ + from error + if not path.startswith(prefix): + raise ValueError(f"Git acquisition failed: {path!r} escapes the selected package") + relative = path[len(prefix):] + portable_path(relative, allow_skill_md=True) + if mode == b"120000": + raise ValueError(f"symlinks are not admissible: {relative}") + if kind != b"blob" or raw_size == b"-": + raise ValueError(f"not a regular file: {relative}") + size = int(raw_size) + entries.append((relative, mode.decode("ascii"), object_id.decode("ascii"), size)) + total += size + + if not entries: + raise ValueError(f"Git acquisition failed: {subdirectory!r} is not a package directory") + if len(entries) > MAX_FILES: + raise ValueError(f"a package may hold at most {MAX_FILES} files; this one holds " + f"{len(entries)}") + if total > MAX_TREE_BYTES: + raise ValueError(f"a package may hold at most {MAX_TREE_BYTES} bytes; this one holds {total}") + return entries + + +def _materialize(repository: Path, package: Path, + entries: list[tuple[str, str, str, int]]) -> None: + """Write raw Git blobs, bypassing checkout hooks and attribute-selected filters.""" + for relative, mode, object_id, expected_size in entries: + content = _git("-C", str(repository), "cat-file", "blob", object_id) + if len(content) != expected_size: + raise ValueError(f"Git acquisition failed: {relative} changed while it was fetched") + target = package / relative + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(content) + target.chmod(0o755 if mode == "100755" else 0o644) + + +def github(repository: str, *, ref: str, subdirectory: str, + destination: Path) -> tuple[Path, dict]: + """Fetch one public GitHub repository subdirectory at an exact resolved commit.""" + if not isinstance(repository, str) or not _REPOSITORY.fullmatch(repository) or \ + repository.casefold().endswith(".git"): + raise ValueError(f"GitHub repository must be OWNER/REPO, found {repository!r}") + if ref != "HEAD": + raise ValueError("GitHub acquisition currently supports the remote HEAD ref") + selected = portable_path(subdirectory).as_posix() + remote = _remote_url(repository) + commit = _resolved_commit(remote, ref) + + destination = Path(destination) + clone = destination / "repository" + destination.mkdir(parents=True, exist_ok=True) + _git("-c", "core.hooksPath=/dev/null", "-c", "core.symlinks=false", "clone", + "--depth", "1", "--filter=blob:none", "--no-checkout", "--no-tags", + "--single-branch", remote, str(clone)) + checked_out = _git("-C", str(clone), "rev-parse", "HEAD").decode("ascii").strip() + if checked_out != commit: + raise ValueError( + f"Git acquisition failed: ref moved from {commit} to {checked_out} during acquisition") + + entries = _bounded_tree(clone, selected) + package = destination / "package" + _materialize(clone, package, entries) + return package, {"repository": repository, "ref": ref, "commit": commit, + "subdirectory": selected} diff --git a/ingot/admission.py b/ingot/admission.py new file mode 100644 index 0000000..1330a07 --- /dev/null +++ b/ingot/admission.py @@ -0,0 +1,133 @@ +"""One complete local ingest path: a directory on disk becomes a quarantined proposal. + +The whole product claim lives in this file's one guarantee -- **`ingot add` never activates +anything**. It runs the deterministic review, computes the exact revision the library would serve, +records where the package came from, and takes the review slot. The served library is not touched. + +This is an adapter, not a second admission service. Path validation, component assembly, content +hashing, evidence writing, pending-record creation, and the atomic slot claim all belong to +`ingot.optimize.ingress` and are called, not reimplemented. What is new here is only the part that is +genuinely new: turning a directory into the fields that service already takes, and binding a +provenance manifest to the result. + +`optimize` is imported inside the function rather than at module scope. `ingot list` and +`ingot review` promise to run in a bare virtualenv, and a module-level import here would put the +optimizer on their import path.""" +from __future__ import annotations + +import time +from pathlib import Path + +from . import records +from .parse import ERROR, WARNING, parse_raw +from .review import REVIEW_SCHEMA, review_package + +_SUPPORTED_SCHEMES = ("file", "github") + + +class AdmissionRefused(Exception): + """The package cannot be represented as a candidate. Nothing was written.""" + + +def parse_locator(locator: str) -> tuple[str, Path | str]: + """A file path or GitHub repository. Unknown schemes are refused by name.""" + if locator.startswith("file:"): + return "file", Path(locator[len("file:"):]).expanduser().resolve() + if locator.startswith("github:"): + return "github", locator[len("github:"):] + + head, separator, _ = locator.partition(":") + if separator and head.isalpha() and len(head) > 1: + raise ValueError( + f"unsupported source scheme {head!r}; this version supports " + f"{', '.join(f'{s}:' for s in _SUPPORTED_SCHEMES)} and bare paths") + return "file", Path(locator).expanduser().resolve() + + +def _codes(result: dict, level: str) -> list[str]: + return [finding["code"] + for section in result["sections"].values() + for finding in section["findings"] + if finding["level"] == level] + + +def add_package(package: Path, *, actor: str, producer: str = "ingot-cli", + source_type: str = "file", locator: str | None = None, + provenance: dict | None = None) -> dict: + """Review, quarantine, and report. Leaves the served library byte-identical.""" + from ingot.mcp_server import registry + from ingot.mcp_server.registry import read_components, skill_revision + from ingot.optimize import ingress, tree + + package = Path(package).expanduser().resolve() + if not package.is_dir(): + raise AdmissionRefused(f"{package} is not a directory") + source_locator = locator or str(package) + + # `registry.library_dir()` resolved per call, never bound at import: a frozen copy would + # check collisions + # against a different library than the one this process actually serves. + report = review_package(package, library_root=registry.library_dir()) + errors, warnings = _codes(report, ERROR), _codes(report, WARNING) + if not report["valid"]: + raise AdmissionRefused( + f"{package.name} is not admissible: {', '.join(errors)}") + + raw = parse_raw((package / "SKILL.md").read_text(encoding="utf-8", errors="replace")) + frontmatter = raw.frontmatter or {} + skill = str(frontmatter.get("name") or package.name) + + # No `file:` components. The package's files travel as a staged tree of exact bytes; carrying + # decoded copies of the text ones beside it would be a second description of the same files, + # and the two would eventually disagree about which is authoritative. + read = read_components(package) + components, metadata = ingress.build_components( + skill, read["description"], read["body"], {}, frontmatter) + try: + candidate_tree = tree.build(package) + # Staged before the revision is computed, because the revision *is* the result of + # materializing the staged tree -- deriving it any other way would be a second description + # of the same bytes, and the two would eventually disagree. Staging is named by the tree + # digest and so is idempotent: a submission refused further down leaves nothing behind but + # a directory the next identical one reuses. + tree.stage(package, candidate_tree) + except ValueError as error: + raise AdmissionRefused(f"{package.name} is not admissible: {error}") from error + + manifest = records.candidate_manifest( + kind="creation", + skill=skill, + source_type=source_type, + locator=source_locator, + # What the source resolved to, and what the library will serve. Equal for a package that is + # already canonical, and deliberately separate fields because they are not always equal -- + # admission collapses whitespace in a description, and then the two diverge. + resolved_revision=skill_revision(package), + candidate_revision=tree.revision(skill, candidate_tree, components), + review={"schema_version": REVIEW_SCHEMA, + "valid": report["valid"], + "errors": errors, + "warnings": warnings, + "report_digest": records.digest(report)}, + created_at=int(time.time()), + provenance=({**(provenance or {}), "content_digest": candidate_tree["digest"]} + if provenance is not None else None)) + + problems = records.validate_candidate(manifest) + if problems: + raise AdmissionRefused("the candidate manifest is malformed: " + "; ".join(problems)) + + outcome = ingress.submit_package_ingest( + skill=skill, + components=components, + candidate_tree=candidate_tree, + metadata=metadata, + revision=manifest["candidate_revision"], + source=(source_locator if source_locator.startswith(f"{source_type}:") + else f"{source_type}:{source_locator}"), + candidate=manifest, + identity=records.candidate_identity(manifest), + review_summary=warnings, + producer=producer, + caller=actor) + return {**outcome, "candidate": manifest} diff --git a/ingot/cli.py b/ingot/cli.py new file mode 100644 index 0000000..4338edd --- /dev/null +++ b/ingot/cli.py @@ -0,0 +1,360 @@ +"""The `ingot` command line. + +Nothing here writes a served byte. `add` quarantines, `approve` and `rollback` queue a publication +receipt, `reject` discards a quarantined change: the publisher is the only writer of the served +library, and every mutating verb calls the same service the console calls rather than a second +approval path of its own. + +Every import stays inside the function that needs it. Importing this module must not pull in +FastAPI, ONNX, LangGraph, Langfuse, or the optimizer, because the first thing a developer runs has +to work in a bare virtualenv with no services, no model, and no key.""" +from __future__ import annotations + +import argparse +import getpass +import json +import os +import sys +from pathlib import Path + +LIST_SCHEMA = "ingot/list/v1" + + +def _default_actor() -> str: + """Who a proposal is attributed to. Best effort, and never blank: an unattributed proposal in + the review queue is one nobody can ask about.""" + return os.environ.get("INGOT_ACTOR") or getpass.getuser() + + +def list_library(root: Path | None = None) -> dict: + """The skills a server would serve, with the roots they came from. + + `root` is passed to the loader rather than replacing its configuration: `configured_roots` + always puts the local authoring root first, even ahead of an explicit root, so the answer can + legitimately include skills from somewhere the caller did not name. Reporting `roots` is what + keeps that honest -- a caller who sees an unexpected skill can see which library it came from.""" + from ingot.mcp_server.registry import configured_roots, load_skills + + explicit = [root] if root is not None else None + return { + "schema_version": LIST_SCHEMA, + "roots": [str(path) for path in configured_roots(explicit)], + "skills": [{"name": skill.name, + "description": skill.description, + "revision": skill.revision, + "root": skill.root} + for skill in load_skills(roots=explicit)], + } + + +def _render(result: dict) -> str: + roots = ", ".join(result["roots"]) + skills = result["skills"] + if not skills: + return f"No skills in {roots}" + width = max(len(skill["name"]) for skill in skills) + lines = [f"{len(skills)} skill{'s' if len(skills) != 1 else ''} in {roots}", ""] + lines += [f" {skill['name']:<{width}} {skill['revision'][:8]} {skill['description']}" + for skill in skills] + return "\n".join(lines) + + +def _list(args: argparse.Namespace) -> int: + result = list_library(args.root) + print(json.dumps(result, indent=2) if args.json else _render(result)) + return 0 + + +def _review(args: argparse.Namespace) -> int: + """Exit non-zero only for deterministic validity errors. Warnings are advice, and a command + that fails on advice teaches people to stop reading it.""" + from . import review as review_module + + package = args.path + if not package.is_dir(): + print(f"ingot review: {package} is not a directory", file=sys.stderr) + return 2 + + result = review_module.review_package(package, library_root=args.root) + print(json.dumps(result, indent=2) if args.json else review_module.render(result)) + return 0 if result["valid"] else 1 + + +def _add(args: argparse.Namespace) -> int: + """Quarantine a package. Never activates anything, so the only failures are refusals.""" + from . import admission + + try: + kind, resolved = admission.parse_locator(args.locator) + if kind == "file": + if args.skill: + raise admission.AdmissionRefused("--skill is only valid for github: sources") + result = admission.add_package(resolved, actor=args.actor) + else: + if not args.skill: + raise admission.AdmissionRefused("--skill is required for github: sources") + import tempfile + from pathlib import Path + from . import acquire + + with tempfile.TemporaryDirectory() as temporary: + package, provenance = acquire.github( + resolved, ref="HEAD", subdirectory=args.skill, + destination=Path(temporary)) + result = admission.add_package( + package, actor=args.actor, source_type="github", locator=args.locator, + provenance=provenance) + except (admission.AdmissionRefused, ValueError) as refusal: + print(f"ingot add: {refusal}", file=sys.stderr) + return 1 + + if args.json: + print(json.dumps(result, indent=2)) + return 0 + + verb = "already quarantined" if result["status"] == "duplicate" else "quarantined" + review_hint = (f" ingot review {result['candidate']['source']['locator']}\n" + if result["candidate"]["source"]["type"] == "file" else "") + print(f"{verb} '{result['skill']}' as proposal {result['proposal_id']}\n" + f" revision {result['candidate']['candidate_revision'][:16]}\n" + f" source {result['candidate']['source']['locator']}\n" + f"\nThe served library is unchanged. Review and approve it in the console, or:\n" + f"{review_hint}" + f" ingot approve {result['skill']}") + return 0 + + +def _vault_init(args: argparse.Namespace) -> int: + from . import vault + + try: + result = vault.init_vault(args.path) + except ValueError as refusal: + print(f"ingot vault init: {refusal}", file=sys.stderr) + return 1 + if args.json: + print(json.dumps(result, indent=2)) + return 0 + print(f"{result['status']} vault at {result['path']}\n" + f" branch {result['branch']}\n" + f" head {result['head'][:12]}") + if result["added"]: + print(f" added {', '.join(result['added'])}") + return 0 + + +def _status(args: argparse.Namespace) -> int: + """Exit non-zero when the served library is writable, so a managed deployment can assert it.""" + from . import status as status_module + + result = status_module.library_status(args.root) + print(json.dumps(result, indent=2) if args.json else status_module.render(result)) + return 0 if result["mode"] == status_module.MANAGED else 1 + + +def _when(seconds: object) -> str: + """Unix seconds as a local timestamp, or blank. A record written before the field existed must + print as an empty column rather than a traceback.""" + if not isinstance(seconds, (int, float)) or isinstance(seconds, bool) or seconds <= 0: + return "" + import datetime + return datetime.datetime.fromtimestamp(seconds).strftime("%Y-%m-%d %H:%M") + + +def _pending(args: argparse.Namespace) -> int: + from . import decisions + + result = decisions.pending_view() + if args.json: + print(json.dumps(result, indent=2)) + return 0 + if result["unreadable"]: + print(f"WARNING {len(result['unreadable'])} quarantined change(s) cannot be read and are " + f"not listed: {', '.join(result['unreadable'])}", file=sys.stderr) + if not result["pending"] and not result["publishing"]: + print("Nothing waiting.") + return 0 + for entry in result["pending"]: + verdict = "ready" if entry["promotable"] else "BLOCKED" + print(f" {verdict:<8} {entry['skill']:<20} {entry['kind']:<12} " + f"{entry['revision'][:12]}") + for reason in entry["blocked"]: + print(f" {reason}") + if entry["publication"]: + print(f" publication {entry['publication']['status']}") + for entry in result["publishing"]: + print(f" {entry['status']:<8} {entry['skill']:<20} {entry['action']:<12} " + f"{entry['revision'][:12]}") + if entry["error"]: + print(f" {entry['error']}") + return 0 + + +def _decide(args: argparse.Namespace) -> int: + """Approve, reject, or roll back. Every one queues or discards; none writes a served byte.""" + from . import decisions + + try: + if args.command == "approve": + result = decisions.approve(args.skill, actor=args.actor) + elif args.command == "reject": + result = decisions.reject(args.skill, actor=args.actor, reason=args.reason) + else: + result = decisions.rollback(args.skill, args.revision, actor=args.actor) + except ValueError as refusal: + print(f"ingot {args.command}: {refusal}", file=sys.stderr) + return 1 + + if args.json: + print(json.dumps(result, indent=2)) + return 0 + print(result["result"]) + receipt = result["publication"] + if receipt: + print(f" publication {receipt['id']} {receipt['status']}") + if receipt["error"]: + print(f" {receipt['error']}") + if receipt["status"] != "published": + print(" The served library is unchanged until the publisher activates it.") + return 0 + + +def _history(args: argparse.Namespace) -> int: + from . import decisions + + try: + result = decisions.history_view(args.skill) + except ValueError as refusal: + print(f"ingot history: {refusal}", file=sys.stderr) + return 1 + if args.json: + print(json.dumps(result, indent=2)) + return 0 + print(f"{result['skill']}\n") + print(" snapshots (rollback targets)") + for revision in result["revisions"] or []: + print(f" {revision['revision'][:16]} {_when(revision.get('created'))}") + if not result["revisions"]: + print(" none") + print("\n publications") + for receipt in result["publications"] or []: + print(f" {receipt['status']:<10} {receipt['action']:<9} " + f"{(receipt['revision'] or '')[:16]} {receipt['id']}") + if not result["publications"]: + print(" none") + print("\n decisions") + for record in result["audit"] or []: + print(f" {record.get('action', ''):<10} {record.get('actor', ''):<16} " + f"{(record.get('revision') or '')[:16]} {_when(record.get('ts'))}" + + (f" {record['reason']}" if record.get("reason") else "")) + if not result["audit"]: + print(" none") + return 0 + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + prog="ingot", + description="Release control for agent skills.") + sub = parser.add_subparsers(dest="command", required=True) + + listing = sub.add_parser("list", help="list the skills in a library") + listing.add_argument("--root", type=Path, default=None, + help="library root to read instead of the configured one") + listing.add_argument("--json", action="store_true", help="emit the versioned JSON payload") + listing.set_defaults(handler=_list) + + reviewing = sub.add_parser( + "review", help="report what is wrong with a skill package, offline", + description="Deterministic, model-free, network-free, read-only review of one skill " + "package. Reports six sections and no composite score; questions it cannot " + "answer offline are reported UNMEASURED with the command that answers them.") + reviewing.add_argument("path", type=Path, help="the skill package directory to review") + reviewing.add_argument("--root", type=Path, default=None, + help="library root to check for collisions against") + reviewing.add_argument("--json", action="store_true", help="emit the versioned JSON payload") + reviewing.set_defaults(handler=_review) + + adding = sub.add_parser( + "add", help="quarantine a skill package for review", + description="Review a package and place it in quarantine. The served library is left " + "byte-identical; a human must approve the proposal before anything is served.") + adding.add_argument("locator", help="file:./path/to/skill, a bare path, or github:OWNER/REPO") + adding.add_argument("--skill", help="package subdirectory inside a github: repository") + adding.add_argument("--actor", default=_default_actor(), + help="who is submitting this (defaults to the current user)") + adding.add_argument("--json", action="store_true", help="emit the versioned JSON payload") + adding.set_defaults(handler=_add) + + queue = sub.add_parser( + "pending", help="list quarantined changes waiting on a decision", + description="Everything waiting on a person, plus anything already travelling to the " + "vault. Read-only.") + queue.add_argument("--json", action="store_true", help="emit the versioned JSON payload") + queue.set_defaults(handler=_pending) + + approving = sub.add_parser( + "approve", help="approve a quarantined change for publication", + description="Queues a publication receipt. It does not activate anything: the publisher " + "commits the approved revision and only then does the library serve it.") + approving.add_argument("skill") + + rejecting = sub.add_parser( + "reject", help="discard a quarantined change", + description="Deletes the pending record and records the decision in the approval trail.") + rejecting.add_argument("skill") + rejecting.add_argument("--reason", default="", help="why, for the approval trail") + + reverting = sub.add_parser( + "rollback", help="queue a stored snapshot for publication", + description="Takes the same lane as an approval: the snapshot is published through the " + "publisher rather than copied over the served library.") + reverting.add_argument("skill") + reverting.add_argument("revision", help="a revision from `ingot history SKILL`") + + for decision in (approving, rejecting, reverting): + decision.add_argument("--actor", default=_default_actor(), + help="who is deciding (defaults to the current user)") + decision.add_argument("--json", action="store_true", + help="emit the versioned JSON payload") + decision.set_defaults(handler=_decide) + + past = sub.add_parser( + "history", help="what a skill has been, and what was decided about it", + description="Rollback targets, publication receipts, and the approval trail for one " + "skill. Read-only.") + past.add_argument("skill") + past.add_argument("--json", action="store_true", help="emit the versioned JSON payload") + past.set_defaults(handler=_history) + + vault = sub.add_parser( + "vault", help="the Git vault the publisher owns", + description="The local publication backend publishes into a Git repository on this " + "machine. This creates one, and is idempotent against an existing vault.") + vault_sub = vault.add_subparsers(dest="vault_command", required=True) + initialize = vault_sub.add_parser("init", help="create or complete a local vault") + initialize.add_argument("path", type=Path, nargs="?", default=Path("vault"), + help="where the vault lives (default: ./vault)") + initialize.add_argument("--json", action="store_true", help="emit the versioned JSON payload") + initialize.set_defaults(handler=_vault_init) + + reporting = sub.add_parser( + "status", help="report whether this deployment's guarantees hold", + description="MANAGED when the served library is read-only to this process, so only the " + "publisher can change what is served. UNMANAGED otherwise, and the exit code " + "says so: 0 for MANAGED, 1 for UNMANAGED.") + reporting.add_argument("--root", type=Path, default=None, + help="library root to inspect instead of the configured one") + reporting.add_argument("--json", action="store_true", help="emit the versioned JSON payload") + reporting.set_defaults(handler=_status) + + return parser + + +def main(argv: list[str] | None = None) -> int: + args = build_parser().parse_args(argv) + return args.handler(args) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/ingot/decisions.py b/ingot/decisions.py new file mode 100644 index 0000000..b3ddda9 --- /dev/null +++ b/ingot/decisions.py @@ -0,0 +1,107 @@ +"""Read models and decision verbs for the command line. + +Every mutating verb here calls the same service the console calls. There is no second approval +path, no direct activation helper, and nothing that writes a served byte: approval and rollback +queue a publication receipt, and the publisher is what acts on it.""" +from __future__ import annotations + +PENDING_SCHEMA = "ingot/pending/v1" +HISTORY_SCHEMA = "ingot/history/v1" +DECISION_SCHEMA = "ingot/decision/v1" + +APPROVED = "approved" # a receipt exists; the publisher has not picked it up yet +PUBLISHING = "publishing" # in flight, including waiting on a forge merge +PUBLISHED = "published" # the served library carries it +FAILED = "failed" # the last attempt errored; the receipt is retryable + + +def release_status(record: dict | None) -> dict | None: + """One receipt, in the four words a person making a decision needs. + + `failed` reads off `last_error` rather than the state, because a failed attempt leaves the + receipt in whatever state it was working through. A receipt that reports only its state hides + the one thing an operator has to act on.""" + if not record: + return None + state = record.get("state") + if state == "active": + status = PUBLISHED + elif record.get("last_error"): + status = FAILED + elif state in {"publishing", "awaiting_merge"}: + status = PUBLISHING + else: + status = APPROVED + return {"id": record.get("id"), "status": status, "state": state, + "action": record.get("action"), "attempts": record.get("attempts", 0), + "error": record.get("last_error") or "", "revision": record.get("candidate_revision")} + + +def pending_view() -> dict: + """Everything waiting on a person, with whatever is already travelling to the vault.""" + from ingot.optimize.promote import challenger_revision, list_pending, unreadable_pending + from ingot.optimize.publication import publication_for_skill, recent_publications + + entries = [] + for record in sorted(list_pending(), key=lambda item: item.get("skill", "")): + skill = record.get("skill", "") + gate = record.get("gate") or {} + entries.append({ + "skill": skill, + "kind": record.get("kind", "quality"), + "promotable": gate.get("promotable") is True, + "blocked": list(gate.get("blocked") or []), + "revision": challenger_revision(record), + "publication": release_status(publication_for_skill(skill)), + }) + # A receipt outlives the pending record it came from, so an approved change is invisible to the + # queue above for exactly the window in which it is travelling and someone might be waiting. + queued = {entry["skill"] for entry in entries} + travelling = [release_status(record) | {"skill": record.get("skill")} + for record in recent_publications() + if record.get("skill") not in queued and record.get("state") != "active"] + return {"schema_version": PENDING_SCHEMA, "pending": entries, "publishing": travelling, + "unreadable": unreadable_pending()} + + +def history_view(skill: str, limit: int = 50) -> dict: + """What this skill has been, what it is travelling toward, and what was decided about it.""" + from ingot.optimize.promote import check_slug, list_revisions, read_audit + from ingot.optimize.publication import recent_publications + + skill = check_slug(skill) + audit = [record for record in read_audit(limit=10_000)["records"] + if record.get("skill") == skill][:limit] + return { + "schema_version": HISTORY_SCHEMA, + "skill": skill, + "revisions": list_revisions(skill), + "publications": [release_status(record) for record in recent_publications(10_000) + if record.get("skill") == skill][:limit], + "audit": audit, + } + + +def _decision(skill: str, message: str) -> dict: + from ingot.optimize.publication import publication_for_skill + + return {"schema_version": DECISION_SCHEMA, "skill": skill, "result": message, + "publication": release_status(publication_for_skill(skill))} + + +def approve(skill: str, actor: str) -> dict: + from ingot.optimize.promote import approve_pending + + return _decision(skill, approve_pending(skill, actor=actor)) + + +def reject(skill: str, actor: str, reason: str = "") -> dict: + from ingot.optimize.promote import reject_pending + + return _decision(skill, reject_pending(skill, actor=actor, reason=reason)) + + +def rollback(skill: str, revision: str, actor: str) -> dict: + from ingot.optimize.promote import rollback as rollback_pending + + return _decision(skill, rollback_pending(skill, revision, actor=actor)) diff --git a/ingot/delivery.py b/ingot/delivery.py new file mode 100644 index 0000000..75dc708 --- /dev/null +++ b/ingot/delivery.py @@ -0,0 +1,219 @@ +"""Where an approved revision is installed once the vault already carries it. + +The vault is the publication authority and the managed-MCP library at the same time: the publisher +commits into it, fast-forwards it, and every other service mounts it read-only. That covers agents +that load skills through Ingot's MCP server. It does not cover an agent that reads a native skill +directory on disk, which is most of them. + +A delivery target closes that gap without opening a second way to approve anything. Targets are +configured on the publisher, never named in a receipt's identity, so what is approved does not +depend on where a particular deployment happens to install it. Delivery runs *after* the vault +serves the approved revision and *before* the receipt is marked active, which is what makes a +failed delivery a retryable release rather than a finished one. + +Two kinds: + +- `managed-mcp` is the vault itself. Delivering to it is a deliberate no-op -- the publication + commit and the fast-forward already put the bytes there, and a second writer in the vault is the + one thing this control plane exists to prevent. It appears in the target list so it has a name, + a status, and a per-target line on the receipt like any other destination. +- `filesystem` copies the skill directory out of the vault into a configured root. + +The source of every delivery is the vault checkout, already verified to be at the approved +revision. Nothing here re-derives the bytes from components or from a staged tree: a second +materialization is a second thing that can disagree with the first. +""" +from __future__ import annotations + +import os +import shutil +import uuid +from dataclasses import dataclass +from pathlib import Path + +from ingot.mcp_server.registry import SLUG_RE, skill_revision +from ingot.optimize import promote + +DELIVERY_SCHEMA = "ingot/delivery/v1" +TARGETS = "INGOT_DELIVERY_TARGETS" + +MANAGED_MCP = "managed-mcp" +FILESYSTEM = "filesystem" +KINDS = (MANAGED_MCP, FILESYSTEM) + +VAULT_TARGET = "vault" + + +@dataclass(frozen=True) +class Target: + name: str + kind: str + root: Path + + def path(self, skill: str) -> Path: + return self.root / skill + + +def _parse_one(entry: str, *, vault: Path) -> Target: + name, separator, remainder = entry.partition("=") + kind, kind_separator, raw_path = remainder.partition(":") + if not separator or not kind_separator: + raise ValueError(f"invalid delivery target {entry!r}: expected name=kind:path") + name, kind, raw_path = name.strip(), kind.strip(), raw_path.strip() + if not SLUG_RE.fullmatch(name): + raise ValueError(f"invalid delivery target name {name!r}") + if kind not in KINDS: + raise ValueError(f"unknown delivery kind {kind!r}; expected one of {', '.join(KINDS)}") + path = Path(raw_path).expanduser() + if not path.is_absolute(): + # A relative root resolves against whatever directory the publisher happened to start in, + # which for a systemd unit is not a directory anyone chose. + raise ValueError(f"delivery target {name!r} must be an absolute path, not {raw_path!r}") + return Target(name, kind, path) + + +def _contained(path: Path, root: Path) -> bool: + return path == root or root in path.parents + + +def parse_targets(spec: str, *, vault: Path) -> tuple[Target, ...]: + """Read the configured target list, refusing anything that cannot work. + + Every check here is one the publisher must make before it starts. A delivery target that fails + on the first approval strands a change that has already been approved, which is the stalled-lane + failure the backend configuration validation exists to prevent.""" + vault = vault.expanduser() + targets: list[Target] = [] + seen: dict[str, Target] = {} + roots: dict[Path, str] = {} + for entry in (piece.strip() for piece in spec.split(",")): + if not entry: + continue + target = _parse_one(entry, vault=vault) + if target.name in seen: + raise ValueError(f"duplicate delivery target {target.name!r}") + if target.kind == MANAGED_MCP and target.root != vault: + raise ValueError(f"the managed-mcp target must be the vault ({vault}), " + f"not {target.root}") + if target.kind == FILESYSTEM and _contained(target.root, vault): + raise ValueError(f"delivery target {target.name!r} is inside the vault; it would write " + f"the checkout the publisher just committed") + if target.root in roots: + raise ValueError(f"delivery targets {roots[target.root]!r} and {target.name!r} name the " + f"same directory") + seen[target.name] = target + roots[target.root] = target.name + targets.append(target) + # The vault is not optional. It is what the MCP server serves and what publication authority is + # measured against, so a configuration that omits it gets it anyway rather than silently + # switching off managed delivery. + if not any(target.kind == MANAGED_MCP for target in targets): + if VAULT_TARGET in seen: + raise ValueError(f"delivery target {VAULT_TARGET!r} is reserved for the managed-mcp " + f"vault target") + targets.insert(0, Target(VAULT_TARGET, MANAGED_MCP, vault)) + return tuple(targets) + + +def load_targets(env: dict | None = None, *, vault: Path) -> tuple[Target, ...]: + env = os.environ if env is None else env + return parse_targets(env.get(TARGETS) or "", vault=vault) + + +def observed(target: Target, skill: str) -> str: + """The revision the target actually holds, asked of the filesystem. + + Absence is a revision: a skill a target does not carry is a real, checkable state, and the + publisher delivers it deliberately when a rollback restores one.""" + path = target.path(skill) + if not path.is_dir(): + return promote.ABSENT_REVISION + return skill_revision(path) + + +def _snapshot_displaced(target: Target, skill: str) -> None: + """Preserve whatever the target held, under the revision of the bytes actually there. + + Keyed by observation rather than by the receipt's champion: a target someone edited by hand + holds bytes no release describes, and those are the ones worth keeping.""" + path = target.path(skill) + if not path.is_dir(): + return + promote._snapshot(path, skill, skill_revision(path)) + + +def _refuse_symlinks(source: Path) -> None: + """A delivered tree carries no symlinks, for the same reason an admitted one carries none. + + `copytree` without `symlinks=True` copies what a link points at, so a link to somewhere outside + the skill would land in a native agent's skill root as an ordinary file holding those bytes. + Copying the link instead is no better: it puts a path in an agent's library leading out of it. + Admission already refuses symlinks, so reaching this means the vault acquired one some other + way, and delivering it is not this code's decision to make.""" + stack = [source] + while stack: + for entry in stack.pop().iterdir(): + if entry.is_symlink(): + raise ValueError( + f"symlink-unsupported: {entry.relative_to(source)} is a symbolic link") + if entry.is_dir(): + stack.append(entry) + + +def _swap(staged: Path, destination: Path) -> None: + """Replace one directory with another using same-filesystem renames only. + + POSIX has no atomic directory swap, so this is two renames with the displaced directory kept + until the second one succeeds. The window between them is the only moment the destination is + absent, and a failure inside it puts the original back rather than leaving a half-installed + skill an agent could load.""" + if not destination.exists(): + os.replace(staged, destination) + return + displaced = destination.with_name(f".{destination.name}.{uuid.uuid4().hex}.old") + os.replace(destination, displaced) + try: + os.replace(staged, destination) + except BaseException: + os.replace(displaced, destination) + raise + shutil.rmtree(displaced, ignore_errors=True) + + +def _remove(destination: Path) -> None: + if not destination.exists(): + return + displaced = destination.with_name(f".{destination.name}.{uuid.uuid4().hex}.old") + os.replace(destination, displaced) + shutil.rmtree(displaced, ignore_errors=True) + + +def install(target: Target, skill: str, source: Path | None, revision: str) -> bool: + """Install one approved revision at one target. True when the target was written. + + `source` is the vault's copy of the skill, already verified to be the approved revision, or + None to deliver an absence. `revision` is what the receipt names, and it is the staged copy -- + the bytes that will actually be installed -- that is checked against it, so a source altered + between the vault check and this call cannot install itself.""" + promote.check_slug(skill) + if target.kind == MANAGED_MCP: + return False + destination = target.path(skill) + if observed(target, skill) == revision: + return False + _snapshot_displaced(target, skill) + if source is None or revision == promote.ABSENT_REVISION: + _remove(destination) + return True + _refuse_symlinks(source) + target.root.mkdir(parents=True, exist_ok=True) + staged = target.root / f".{skill}.{uuid.uuid4().hex}.tmp" + try: + shutil.copytree(source, staged, symlinks=False) + if skill_revision(staged) != revision: + raise ValueError(f"delivered '{skill}' does not match the approved revision") + _swap(staged, destination) + except BaseException: + shutil.rmtree(staged, ignore_errors=True) + raise + return True diff --git a/mcp_server/__init__.py b/ingot/mcp_server/__init__.py similarity index 100% rename from mcp_server/__init__.py rename to ingot/mcp_server/__init__.py diff --git a/mcp_server/embedding.py b/ingot/mcp_server/embedding.py similarity index 66% rename from mcp_server/embedding.py rename to ingot/mcp_server/embedding.py index c20565b..faeebeb 100644 --- a/mcp_server/embedding.py +++ b/ingot/mcp_server/embedding.py @@ -1,4 +1,4 @@ -"""Embedding backends for the router, both CPU-only ONNX (no GPU, no torch). +"""Embedding backends for the router: local CPU ONNX or a remote GPU embedding server. Default: the Qwen3-Embedding-0.6B q4 export, chosen on a 297-query drafted routing eval. It is an asymmetric retrieval model: queries carry the official instruction prefix, documents (skill @@ -19,12 +19,16 @@ COLLISION_SCORE together with the model.""" from __future__ import annotations +import json import os +from urllib.error import HTTPError, URLError +from urllib.request import Request, urlopen import numpy as np EMBED_MODEL = os.environ.get("EMBED_MODEL", "onnx-community/Qwen3-Embedding-0.6B-ONNX") EMBED_ONNX_FILE = os.environ.get("EMBED_ONNX_FILE", "onnx/model_q4.onnx") +EMBED_BACKEND = os.environ.get("EMBED_BACKEND", "local") # The official Qwen3-Embedding retrieval instruction: applied to queries only, never documents. QUERY_PREFIX = ("Instruct: Given a web search query, retrieve relevant passages that answer " "the query\nQuery: ") @@ -119,5 +123,69 @@ def embed(self, texts): embed_query = embed +class RemoteQwenEmbedding: + """OpenAI-compatible Qwen embedding server. + + The server stays outside the MCP process so the CPU image remains small and portable. GPU + selection is explicit: requesting this backend without a reachable server fails instead of + silently changing model quality and invalidating calibrated routing thresholds. + """ + + def __init__(self, model: str, base_url: str, timeout: float = 60): + self._model = model + self._url = base_url.rstrip("/") + "/embeddings" + self._timeout = timeout + self.identity = f"remote:{model}@{base_url.rstrip('/')}" + + def _run(self, texts: list[str]) -> list[np.ndarray]: + if not texts: + return [] + body = json.dumps({ + "model": self._model, + "input": texts, + "encoding_format": "float", + }).encode() + request = Request( + self._url, + data=body, + headers={"Content-Type": "application/json", "Authorization": "Bearer no-key"}, + method="POST", + ) + try: + with urlopen(request, timeout=self._timeout) as response: + payload = json.loads(response.read()) + except (HTTPError, URLError, TimeoutError, json.JSONDecodeError) as exc: + raise RuntimeError(f"embedding server unavailable at {self._url}: {exc}") from exc + try: + ordered = sorted(payload["data"], key=lambda item: item["index"]) + vectors = [np.asarray(item["embedding"], dtype=np.float32) for item in ordered] + except (KeyError, TypeError, ValueError) as exc: + raise RuntimeError("embedding server returned an invalid response") from exc + if len(vectors) != len(texts): + raise RuntimeError("embedding server returned the wrong vector count") + return vectors + + def embed(self, texts) -> list[np.ndarray]: + return self._run(list(texts)) + + def embed_query(self, texts) -> list[np.ndarray]: + return self._run([QUERY_PREFIX + text for text in texts]) + + def build_embedding(model: str = EMBED_MODEL): + backend = os.environ.get("EMBED_BACKEND", EMBED_BACKEND).lower() + if backend == "remote": + base_url = os.environ.get("EMBED_BASE_URL", "").strip() + if not base_url: + raise ValueError("EMBED_BASE_URL is required when EMBED_BACKEND=remote") + remote_model = os.environ.get("EMBED_REMOTE_MODEL", "").strip() + if not remote_model: + raise ValueError("EMBED_REMOTE_MODEL is required when EMBED_BACKEND=remote") + try: + timeout = float(os.environ.get("EMBED_TIMEOUT_SECONDS", "60")) + except ValueError as exc: + raise ValueError("EMBED_TIMEOUT_SECONDS must be a number") from exc + return RemoteQwenEmbedding(remote_model, base_url, timeout) + if backend != "local": + raise ValueError("EMBED_BACKEND must be local or remote") return QwenOnnxEmbedding(model) if is_qwen_onnx(model) else FastembedEmbedding(model) diff --git a/ingot/mcp_server/provenance.py b/ingot/mcp_server/provenance.py new file mode 100644 index 0000000..6777462 --- /dev/null +++ b/ingot/mcp_server/provenance.py @@ -0,0 +1,140 @@ +"""Where each served skill came from. + +A merged library hides its own history. Once several roots are mounted together the UI shows +one flat list, so a skill written here looks exactly like one pulled from a third-party repo. +That matters for two questions an operator actually asks: what have I contributed, and whose +licence governs this text. + +Four provenances: + +- ``authored`` — written in this library. +- ``vendored`` — copied into this library from an upstream repo. The library records these in + a ``VENDORED.md`` ledger beside the skills; a vendored skill lives in the same root as an + authored one, so the root alone cannot tell them apart. +- ``fetched`` — pulled by ``scripts/fetch_skills.sh`` into the local ``skills/`` directory. +- ``external`` — served from any other configured root. + +The ledger is the only source of truth for ``vendored``. A skill copied in but never recorded +reads as ``authored``, which overstates authorship — the fix is to record it, not to guess here. +""" + +from __future__ import annotations + +import os +import re +from pathlib import Path +from ingot import paths + +AUTHORED = "authored" +VENDORED = "vendored" +FETCHED = "fetched" +EXTERNAL = "external" + +LEDGER_NAME = "VENDORED.md" + +# `## threejs-{animation,fundamentals}` and `## a, b` both name several skills in one heading. +_BRACE = re.compile(r"^(?P[^{]*)\{(?P[^}]*)\}(?P.*)$") + + +def _split_top_level(heading: str) -> list[str]: + """Split on commas that separate entries, not on commas inside a brace group. + + `threejs-{a,b}, other` is two entries, not three: the first two commas belong to the + brace group. Splitting the raw string first would tear `threejs-{a` off `b}`. + """ + parts, depth, current = [], 0, [] + for ch in heading: + if ch == "{": + depth += 1 + elif ch == "}": + depth = max(0, depth - 1) + if ch == "," and depth == 0: + parts.append("".join(current)) + current = [] + else: + current.append(ch) + parts.append("".join(current)) + return [p.strip() for p in parts if p.strip()] + + +def _expand(heading: str) -> list[str]: + """Skill names named by one ledger heading. + + Handles both shapes the ledger uses: brace expansion (`threejs-{a,b}` -> `threejs-a`, + `threejs-b`) and a plain comma-separated list. Anything else is one name. + """ + names: list[str] = [] + for part in _split_top_level(heading): + match = _BRACE.match(part) + if match: + stem, tail = match.group("stem").strip(), match.group("tail").strip() + names.extend(f"{stem}{opt.strip()}{tail}" + for opt in match.group("options").split(",") if opt.strip()) + else: + names.append(part) + return names + + +def vendored_names(root: Path) -> set[str]: + """Skill names the root's ``VENDORED.md`` declares as copied in from upstream. + + Returns an empty set when the root publishes no ledger, which is the common case for a + fetched or external root. + """ + ledger = root / LEDGER_NAME + try: + text = ledger.read_text(encoding="utf-8", errors="ignore") + except (OSError, ValueError): + return set() + names: set[str] = set() + for line in text.splitlines(): + if not line.startswith("## "): + continue + heading = line[3:].strip() + # Prose sections ("Local divergences from upstream") are not skill names. A real skill + # slug has no spaces once the comma/brace forms are expanded. + for name in _expand(heading): + if name and " " not in name: + names.add(name) + return names + + +def local_root() -> Path: + """The repository's own ``skills/`` directory, where fetch_skills.sh writes.""" + return paths.library() + + +def classify(name: str, skill_dir: str | Path, *, + ledgers: dict[Path, set[str]] | None = None) -> str: + """Provenance of one skill. + + ``skill_dir`` is ``Skill.root``, which is the skill's OWN directory + (``/srv/skills/dotfiles/game-dev``), not the library root. The ledger lives one level up, + beside its sibling skills, so the library root is the parent. + + ``ledgers`` caches each library root's parsed ledger, so a whole inventory costs one read + per root rather than one per skill. + """ + library = Path(skill_dir).parent + if ledgers is None: + ledgers = {} + if library not in ledgers: + ledgers[library] = vendored_names(library) + if name in ledgers[library]: + return VENDORED + try: + if library.resolve() == local_root().resolve(): + return FETCHED + except OSError: + pass + return AUTHORED if (library / LEDGER_NAME).exists() else EXTERNAL + + +def label(provenance: str) -> str: + """Human-facing group name.""" + return { + AUTHORED: "Authored here", + VENDORED: "Vendored in", + FETCHED: "Fetched", + EXTERNAL: "External root", + }.get(provenance, provenance) diff --git a/mcp_server/registry.py b/ingot/mcp_server/registry.py similarity index 74% rename from mcp_server/registry.py rename to ingot/mcp_server/registry.py index 4ea59bc..8d05b17 100644 --- a/mcp_server/registry.py +++ b/ingot/mcp_server/registry.py @@ -2,17 +2,25 @@ key; the markdown body is what an agent loads. No compilation, no DB, a skill is just its SKILL.md.""" from __future__ import annotations import hashlib +import json import os import re import uuid import warnings from dataclasses import dataclass, field from pathlib import Path +from pathlib import PurePosixPath from typing import Iterable import yaml +from ingot import paths + +def library_dir() -> Path: + """The local authoring root. A function, not a constant: it is configuration, and a value + frozen at import cannot follow a process that is told where its state lives.""" + return paths.library() + -SKILLS_DIR = Path(__file__).resolve().parent.parent / "skills" _FRONTMATTER = re.compile(r"^---\s*\n(.*?)\n---\s*\n(.*)$", re.DOTALL) # One slug rule for every layer (promotion, UI), a name one layer accepts @@ -85,7 +93,7 @@ def configured_roots(explicit: Iterable[str | Path] | None = None) -> list[Path] values = list(explicit) if explicit is not None else [ p for p in os.environ.get("SKILL_ROUTER_PATHS", "").split(os.pathsep) if p ] - values = [SKILLS_DIR, *values] if values else [SKILLS_DIR] + values = [library_dir(), *values] if values else [library_dir()] roots: list[Path] = [] seen: set[Path] = set() for value in values: @@ -110,6 +118,11 @@ def _router_metadata(meta: dict) -> dict: for key in list_fields: if not isinstance(result[key], list) or not all(isinstance(item, str) for item in result[key]): raise ValueError(f"metadata.skill-router.{key} must be a list of strings") + for pattern in result["path_patterns"]: + path = PurePosixPath(pattern) + if (not pattern or path.is_absolute() or ".." in path.parts or "\\" in pattern or + any(ord(char) < 32 for char in pattern)): + raise ValueError("metadata.skill-router.path_patterns must be relative POSIX globs") try: result["priority"] = int(result["priority"]) except (TypeError, ValueError): @@ -144,15 +157,32 @@ def skill_revision(skill_root: Path, components: dict[str, str] | None = None) - raise ValueError(f"component escapes skill root: {relative}") _contained_file(skill_root, skill_root / relative) files.setdefault(relative.as_posix(), skill_root / relative) + # A quarantined creation has no directory or SKILL.md yet. Include its logical SKILL.md in + # the same digest shape an activated skill will use, so approval can bind the exact candidate + # before any filesystem mutation occurs. + if components is not None and "SKILL.md" not in files: + files["SKILL.md"] = skill_root / "SKILL.md" digest = hashlib.sha256() for relative, path in sorted(files.items()): digest.update(relative.encode()) digest.update(b"\0") if relative == "SKILL.md": - meta, body = parse_skill(path.read_text(encoding="utf-8", errors="ignore"), skill_root.name) + if path.exists(): + meta, body = parse_skill( + path.read_text(encoding="utf-8", errors="ignore"), skill_root.name) + else: + meta, body = {"name": skill_root.name, "description": ""}, "" if components is not None: - meta["description"] = components["description"] + if "frontmatter" in components: + try: + supplied = json.loads(components["frontmatter"]) + except (TypeError, ValueError) as exc: + raise ValueError("frontmatter component is not valid JSON") from exc + meta = normalized_frontmatter(skill_root.name, + components["description"], supplied) + else: + meta["description"] = components["description"] body = components["body"] digest.update(yaml.safe_dump(meta, sort_keys=True, allow_unicode=True).encode()) digest.update(b"\0") @@ -202,6 +232,31 @@ def load_skills(skills_dir: Path | None = None, *, roots: Iterable[str | Path] | return skills +def resolve_skill_dir(name: str) -> Path: + """The directory `name` actually lives in, across every configured root. + + `library_dir() / name` is only correct for the one *writable* authoring root. A merged library + serves most of its skills from read-only mounts, so that path finds them by luck or not at + all — and every optimize entry point used to build it by hand, which meant the optimizer + silently refused to touch anything it did not itself author.""" + for item in load_skills(): + if item.name == name: + return Path(item.root) + raise LookupError(f"no indexed skill named '{name}'; check SKILL_ROUTER_PATHS") + + +def writable_skill_dir(name: str) -> Path: + """The activation destination in the first-precedence writable authoring root. + + This is not a lookup: callers must still use ``resolve_skill_dir`` to read the serving + revision. Promotion uses this destination only after resolving and validating that revision, + because merged libraries may supply it from a read-only mount. + """ + if not SLUG_RE.fullmatch(name): + raise ValueError(f"invalid skill name: {name!r}") + return library_dir().expanduser().resolve() / name + + # --- writing / full-skill components (used by candidate generation and promotion) --- _TEXT_SUFFIXES = {".md", ".txt", ".py", ".sh", ".js", ".ts", ".json", ".yaml", ".yml", ".toml", ".cfg"} @@ -217,6 +272,27 @@ def write_skill_md(path: Path, meta: dict, body: str) -> None: path.write_text(f"---\n{dumped}---\n\n{body.strip()}\n", encoding="utf-8") +def normalized_frontmatter(name: str, description: str, frontmatter: dict | None = None) -> dict: + """Canonical, router-valid metadata for a proposed skill before revision hashing. + + Creation cannot hash caller bytes and normalize them only during activation: that would let + the reviewed revision differ from what routing serves. A safe YAML round trip also rejects + object types that could not be persisted as portable frontmatter. + """ + if frontmatter is not None and not isinstance(frontmatter, dict): + raise ValueError("frontmatter must be an object") + try: + meta = yaml.safe_load(yaml.safe_dump(frontmatter or {}, allow_unicode=True)) or {} + except yaml.YAMLError as exc: + raise ValueError("frontmatter must contain YAML-safe values") from exc + if not isinstance(meta, dict): + raise ValueError("frontmatter must be an object") + meta["name"] = name + meta["description"] = " ".join(str(description).split()) + _router_metadata(meta) + return meta + + def read_components(skill_dir: Path) -> dict[str, str]: """Every optimizable text component of a skill: its routing `description`, its SKILL.md `body`, and each bundled text file as `file:`. This is the unit a candidate rewrite works on.""" diff --git a/ingot/mcp_server/router.py b/ingot/mcp_server/router.py new file mode 100644 index 0000000..030d5a4 --- /dev/null +++ b/ingot/mcp_server/router.py @@ -0,0 +1,298 @@ +"""Embedding router: cache description and bounded approved-content vectors, then suggest the +top-k skills for a task by cosine similarity. The default remains CPU-only; the tracked GPU +Compose overlay selects the larger remote model. + +Model is `EMBED_MODEL` (default Qwen3-Embedding-0.6B q4, ~15 ms/query on CPU; queries get the +retrieval instruction prefix, descriptions don't). Any fastembed model name also works (e.g. the +previous default `BAAI/bge-small-en-v1.5`, ~4 ms/query), but recalibrate MIN_SCORE / +RELATED_SCORE / COLLISION_SCORE with the model (mcp_server/embedding.py).""" +from __future__ import annotations +import os +import sys +import threading +from collections import OrderedDict +from dataclasses import dataclass +from pathlib import Path + +import numpy as np + +from .embedding import EMBED_MODEL as _MODEL, build_embedding +from .registry import Skill + + +@dataclass(frozen=True) +class _RankedSkill: + skill: Skill + score: float + description_score: float + content_score: float + + @property + def matched_on(self) -> str: + return "description" if self.description_score >= self.content_score else "content" + + def explanation(self) -> dict: + return { + "score_components": { + "description": round(self.description_score, 3), + "content": round(self.content_score, 3), + }, + "matched_on": self.matched_on, + } + + +class Router: + _vector_cache: OrderedDict[tuple[str, str, str, str], np.ndarray] = OrderedDict() + _vector_cache_limit = 4096 + _cache_lock = threading.Lock() + + def __init__(self, skills: list[Skill]): + self.skills = skills + try: + self._body_chars = int(os.environ.get("ROUTER_BODY_CHARS", "1000")) + except ValueError as exc: + raise ValueError("ROUTER_BODY_CHARS must be an integer from 1 to 4000") from exc + if not 1 <= self._body_chars <= 4000: + raise ValueError("ROUTER_BODY_CHARS must be an integer from 1 to 4000") + if not skills: # empty library, don't normalize an empty matrix + self._embed = None + self._mat = np.zeros((0, 0), dtype=np.float32) + return + self._embed = build_embedding() + backend = type(self._embed) + self._embedding_identity = getattr( + self._embed, "identity", f"{backend.__module__}.{backend.__qualname__}") + self._mat = self._matrix("description", [skill.description for skill in skills]) + + def _matrix(self, representation: str, texts: list[str]) -> np.ndarray: + keys = [(_MODEL, self._embedding_identity, representation, text) for text in texts] + resolved = {} + with self._cache_lock: + missing = [] + for key in dict.fromkeys(keys): + vector = self._vector_cache.get(key) + if vector is None: + missing.append(key) + continue + self._vector_cache.move_to_end(key) + resolved[key] = vector + if missing: + vectors = list(self._embed.embed([text for _, _, _, text in missing])) + if len(vectors) != len(missing): + raise RuntimeError("embedding backend returned the wrong vector count") + generated = { + key: np.asarray(vector, dtype=np.float32) + for key, vector in zip(missing, vectors) + } + resolved.update(generated) + with self._cache_lock: + for key, vector in generated.items(): + self._vector_cache[key] = vector + self._vector_cache.move_to_end(key) + while len(self._vector_cache) > self._vector_cache_limit: + self._vector_cache.popitem(last=False) + mat = np.array([resolved[key] for key in keys], dtype=np.float32) + return mat / (np.linalg.norm(mat, axis=1, keepdims=True) + 1e-8) + + def _content_text(self, skill: Skill, harness: str) -> str: + return ( + f"Skill: {skill.name}\n" + f"Description: {skill.description}\n" + f"Instructions:\n{skill.body_for(harness)[:self._body_chars]}" + ) + + def _ranked(self, task: str, harness: str, skills: list[Skill]) -> list[_RankedSkill]: + if not skills: + return [] + query = np.array(next(iter(self._embed.embed_query([task]))), dtype=np.float32) + query = query / (np.linalg.norm(query) + 1e-8) + index_by_name = {skill.name: index for index, skill in enumerate(self.skills)} + content = self._matrix( + f"content:{harness or 'default'}", + [self._content_text(skill, harness) for skill in skills], + ) + ranked = [] + for content_index, skill in enumerate(skills): + description_score = float(self._mat[index_by_name[skill.name]] @ query) + content_score = float(content[content_index] @ query) + ranked.append(_RankedSkill( + skill=skill, + score=max(description_score, content_score), + description_score=description_score, + content_score=content_score, + )) + return sorted( + ranked, + key=lambda item: ( + -item.score, -int(item.skill.metadata.get("priority", 50)), item.skill.name + ), + ) + + def nearest(self, text: str) -> tuple[str, float]: + """The most similar existing skill to `text` and its cosine score, used to reject a new + skill whose description near-duplicates (shadows) an existing one's routing.""" + if not self.skills: + return "", 0.0 + q = np.array(next(iter(self._embed.embed([text]))), dtype=np.float32) + q = q / (np.linalg.norm(q) + 1e-8) + scores = self._mat @ q + i = int(np.argmax(scores)) + return self.skills[i].name, float(scores[i]) + + def suggest(self, task: str, k: int = 5, min_score: float = 0.0) -> list[dict]: + if not self.skills: + return [] + return [ + { + "name": item.skill.name, + "description": item.skill.description, + "score": round(item.score, 3), + **item.explanation(), + } + for item in self._ranked(task, "", self.skills)[:k] if item.score >= min_score + ] + + @staticmethod + def _platform(value: str | None) -> str: + value = (value or sys.platform).lower() + if value.startswith("darwin") or value == "macos": + return "macos" + if value.startswith("win"): + return "windows" + return "linux" if value.startswith("linux") else value + + @staticmethod + def _compatible(skill: Skill, harness: str, cwd: str, available_tools: set[str], + available_mcps: set[str], platform: str) -> bool: + meta = skill.metadata or {} + if harness not in meta.get("harnesses", ["claude", "codex"]): + return False + if platform not in meta.get("platforms", ["macos", "linux", "windows"]): + return False + if meta.get("activation", "automatic") != "automatic" or meta.get("trust") == "blocked": + return False + if not set(meta.get("required_tools", [])).issubset(available_tools): + return False + if not set(meta.get("required_mcps", [])).issubset(available_mcps): + return False + scopes = meta.get("scopes", ["global"]) + if "global" not in scopes: + patterns = meta.get("path_patterns", []) + if "project" not in scopes or not patterns: + return False + project = Path(cwd).expanduser().resolve() + try: + matched = any(any(project.glob(pattern)) for pattern in patterns) + except (NotImplementedError, OSError, ValueError): + return False + if not project.is_dir() or not matched: + return False + return True + + def _eligible_ranking(self, task: str, harness: str, cwd: str, + available_tools: set[str], available_mcps: set[str], + platform: str) -> list[_RankedSkill]: + eligible = [skill for skill in self.skills if self._compatible( + skill, harness, cwd, available_tools, available_mcps, platform + )] + return self._ranked(task, harness, eligible) + + @staticmethod + def _without_conflicts(ranked: list[_RankedSkill]) -> list[_RankedSkill]: + selected = [] + for candidate in ranked: + skill = candidate.skill + if any(skill.name in set(existing.skill.metadata.get("conflicts", [])) or + existing.skill.name in set(skill.metadata.get("conflicts", [])) + for existing in selected): + continue + selected.append(candidate) + return selected + + @staticmethod + def _alternatives(ranked: list[_RankedSkill]) -> list[dict]: + return [ + { + "name": item.skill.name, + "score": round(item.score, 3), + "reason": f"compatible alternative; {item.matched_on} cosine {item.score:.3f}", + **item.explanation(), + } + for item in ranked[1:3] + ] + + @staticmethod + def _novel_response(score: float = 0.0, reason: str = "no compatible skill candidates", + alternatives: list[dict] | None = None, + candidate: _RankedSkill | None = None) -> dict: + explanation = ( + candidate.explanation() + if candidate is not None + else { + "matched_on": None, + "score_components": {"description": 0.0, "content": 0.0}, + } + ) + return { + "match": None, "related_match": None, "score": round(score, 3), + "reason": reason, "skill_body": "", "skill_root": None, "revision": None, + "alternatives": alternatives or [], "novel": True, **explanation, + } + + @staticmethod + def _related_response(item: _RankedSkill, harness: str, min_score: float, + alternatives: list[dict]) -> dict: + skill, score = item.skill, item.score + return { + "match": None, "related_match": skill.name, "score": round(score, 3), + "reason": (f"best compatible score {score:.3f} below direct threshold " + f"{min_score:.3f}; matched on {item.matched_on}; " + "loaded for compose or extend"), + "skill_body": skill.body_for(harness), + "skill_root": skill.root or str(os.path.dirname(skill.path)), + "revision": skill.revision or None, "alternatives": alternatives, "novel": False, + **item.explanation(), + } + + @staticmethod + def _direct_response(item: _RankedSkill, harness: str, + alternatives: list[dict]) -> dict: + skill, score = item.skill, item.score + return { + "match": skill.name, "related_match": None, "score": round(score, 3), + "reason": f"compatible {harness} skill; {item.matched_on} cosine {score:.3f}", + "skill_body": skill.body_for(harness), + "skill_root": skill.root or str(os.path.dirname(skill.path)), + "revision": skill.revision or None, "alternatives": alternatives, "novel": False, + **item.explanation(), + } + + def route(self, task: str, harness: str, cwd: str, available_tools=(), available_mcps=(), + platform: str | None = None, min_score: float = 0.53, + related_score: float = 0.37) -> dict: + """Filter compatible skills, rank them locally, and return at most one instruction body. + `novel` is the escalation signal for the calling harness: True when nothing compatible is + even related (best score below `related_score`), the case where a weak/strong setup should + serve with the strong model, then queue a candidate for human review.""" + harness = harness.lower() + ranked = self._eligible_ranking( + task, harness, cwd, set(available_tools), set(available_mcps), self._platform(platform) + ) + if not ranked: + return self._novel_response() + ranked = self._without_conflicts(ranked) + top = ranked[0] + score = top.score + alternatives = self._alternatives(ranked) + if score < min_score: + if score < related_score: + related = [{"name": top.skill.name, "score": round(score, 3), + "reason": (f"best compatible candidate; {top.matched_on} " + f"cosine {score:.3f}"), + **top.explanation()}, + *alternatives] + reason = (f"best compatible score {score:.3f} below related threshold " + f"{related_score:.3f}") + return self._novel_response(score, reason, related[:3], top) + return self._related_response(top, harness, min_score, alternatives) + return self._direct_response(top, harness, alternatives) diff --git a/mcp_server/routing_eval.py b/ingot/mcp_server/routing_eval.py similarity index 100% rename from mcp_server/routing_eval.py rename to ingot/mcp_server/routing_eval.py diff --git a/mcp_server/server.py b/ingot/mcp_server/server.py similarity index 67% rename from mcp_server/server.py rename to ingot/mcp_server/server.py index 5d90356..ab60ae9 100644 --- a/mcp_server/server.py +++ b/ingot/mcp_server/server.py @@ -13,7 +13,7 @@ MIN_SCORE = float(os.environ.get("MIN_SCORE", "0.53")) RELATED_SCORE = float(os.environ.get("RELATED_SCORE", "0.37")) PORT = int(os.environ.get("PORT", "8000")) -# Loopback by default: the tools are unauthenticated, so a bare `python -m mcp_server.server` must +# Loopback by default: the tools are unauthenticated, so a bare `python -m ingot.mcp_server.server` must # not listen on the network. The compose mcp service sets HOST=0.0.0.0 (required for Docker port # publishing); host access stays localhost-only via the 127.0.0.1 port mapping. HOST = os.environ.get("HOST", "127.0.0.1") @@ -118,6 +118,81 @@ def route_and_load(task: str, harness: str, cwd: str, available_tools: list[str] return result +@mcp.tool() +def propose_skill_update( + skill: str, + champion_revision: str, + challenger_body: str, + summary: str, + trigger: str, + minimal_content: str, + producer: str, + caller: str, + evidence: list[str], + pressure_scenario: str, + risk: str, + verification_status: str, + verification_command: str, + verification_result: str, + challenger_description: str = "", +) -> dict: + """Quarantine an evidence-backed update proposed by skill-retrospective. + + This tool never activates, rejects, or replaces instructions. It accepts one full candidate + for an existing skill after pressure verification passes, binds it to the exact loaded champion + revision, and refuses an occupied review slot. Human approval in the Ingot console remains + required. + """ + from ingot.optimize.retrospective import submit_skill_update + return submit_skill_update( + skill=skill, + champion_revision=champion_revision, + challenger_body=challenger_body, + challenger_description=challenger_description, + summary=summary, + trigger=trigger, + minimal_content=minimal_content, + producer=producer, + caller=caller, + evidence=evidence, + pressure_scenario=pressure_scenario, + risk=risk, + verification_status=verification_status, + verification_command=verification_command, + verification_result=verification_result, + ) + + +@mcp.tool() +def propose_skill_create( + skill: str, + description: str, + body: str, + files: dict[str, str], + frontmatter: dict, + summary: str, + source: str, + producer: str, + caller: str, + evidence: list[str], + pressure_scenario: str, + risk: str, + verification_status: str, + verification_command: str, + verification_result: str, +) -> dict: + """Quarantine a vetted new skill package as “to be added”; never activate it.""" + from ingot.optimize.ingress import submit_skill_create + return submit_skill_create( + skill=skill, description=description, body=body, files=files, frontmatter=frontmatter, + summary=summary, + source=source, producer=producer, caller=caller, evidence=evidence, + pressure_scenario=pressure_scenario, risk=risk, + verification_status=verification_status, verification_command=verification_command, + verification_result=verification_result, + ) + + if __name__ == "__main__": print(f"[ingot] {len(STATE.skills)} skills loaded; serving MCP on :{PORT}/mcp", flush=True) mcp.run(transport="http", host=HOST, port=PORT, path="/mcp", diff --git a/mcp_server/usage_counts.py b/ingot/mcp_server/usage_counts.py similarity index 90% rename from mcp_server/usage_counts.py rename to ingot/mcp_server/usage_counts.py index 2eac885..0b55e30 100644 --- a/mcp_server/usage_counts.py +++ b/ingot/mcp_server/usage_counts.py @@ -6,10 +6,10 @@ import os import threading from pathlib import Path +from ingot import paths _LOCK = threading.Lock() -_PATH = Path(os.environ.get("SKILL_USAGE_FILE") or - Path(__file__).resolve().parent.parent / "runs" / "skill_usage.json") +_PATH = Path(os.environ.get("SKILL_USAGE_FILE") or paths.runs() / "skill_usage.json") def load_counts() -> dict[str, int]: diff --git a/optimize/__init__.py b/ingot/optimize/__init__.py similarity index 89% rename from optimize/__init__.py rename to ingot/optimize/__init__.py index 35cc94a..b73e3fa 100644 --- a/optimize/__init__.py +++ b/ingot/optimize/__init__.py @@ -193,3 +193,30 @@ def preflight_provider_pins() -> list[str]: # Loaded skill {body}""" + + +def resolve_skill_dir(name: str): + """Where a skill lives, as a clean CLI exit rather than a traceback when it is not indexed. + + Every optimize entry point needs this and none of them should hand-build `library_dir() / name`: + only the authoring root is writable, so that path misses every skill served from a read-only + mount.""" + from ingot.mcp_server.registry import resolve_skill_dir as _resolve + try: + return _resolve(name) + except LookupError as e: + raise SystemExit(str(e)) from e + + +def configured_models(setting: str, fallback: str = "") -> list[str]: + """Parse one comma-separated model setting and refuse accidental duplicate votes/spend.""" + configured = os.environ.get(setting) + raw = configured if configured and any(part.strip() for part in configured.split(",")) \ + else fallback + models = [model.strip() for model in raw.split(",") if model.strip()] + seen: set[str] = set() + for model in models: + if model in seen: + raise SystemExit(f"{setting} contains duplicate model '{model}'; list each model once") + seen.add(model) + return models diff --git a/optimize/ab.py b/ingot/optimize/ab.py similarity index 94% rename from optimize/ab.py rename to ingot/optimize/ab.py index e2bfb23..bb1700b 100644 --- a/optimize/ab.py +++ b/ingot/optimize/ab.py @@ -1,12 +1,12 @@ """Generate a candidate change for one skill and produce the evidence a reviewer needs. -The candidate search (optimize.skillopt_loop, driven per-component by `_greedy_search`) turns the +The candidate search (ingot.optimize.skillopt_loop, driven per-component by `_greedy_search`) turns the skill's components into a challenger on the train tasks. Champion and challenger then run through the full agent with a local `route_and_load` on the held-out tasks; each variant is a Langfuse dataset run (side-by-side in the UI) when the stack is up. The result is a quarantined record in runs/pending/.json plus a portable evidence bundle in runs/evidence/. Nothing here activates anything: promotion is a human action in the UI. -Usage: python -m optimize.ab [--description | --scripts] [--skip-search] +Usage: python -m ingot.optimize.ab [--description | --scripts] [--skip-search] """ import argparse import asyncio @@ -22,16 +22,17 @@ from langchain_core.tools import tool from agent.run import build_agent, langfuse_config, run_task -from mcp_server.registry import SKILLS_DIR, load_skills, optimizable_components, skill_revision +from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision -from . import agent_model, langfuse_available +from . import agent_model, langfuse_available, resolve_skill_dir from . import usage as usage_ledger from .acceptance import classify as acceptance_classify, load_criteria as load_acceptance from .judge import MODELS as JUDGE_MODELS, judge from .promote import save_pending from .evidence import build_evidence, recorded_path, write_evidence +from ingot import paths -TASKS_DIR = Path(__file__).resolve().parent / "tasks" +TASKS_DIR = paths.tasks() # --- promotion gate (anti reward-hacking / overfitting) --- PROMOTE_MIN_MARGIN = float(os.environ.get("PROMOTE_MIN_MARGIN", "0.15")) # mean holdout lift required @@ -57,7 +58,7 @@ OPTIMIZE_COMPONENTS = [c.strip() for c in os.environ.get("OPTIMIZE_COMPONENTS", "body").split(",") if c.strip()] -EVAL_CACHE_DIR = Path(__file__).resolve().parent.parent / "runs" / "eval-cache" +EVAL_CACHE_DIR = paths.runs() / "eval-cache" def _champion_cache_key(revision: str, holdout: list[dict], components: list[str]) -> str: @@ -112,8 +113,8 @@ def retention_warnings(champion: dict, challenger: dict, changed: list[str], sam def _description_shadows(skill: str, new_description: str) -> tuple[str, float]: """Nearest OTHER skill to a rewritten description (route-shadow check). ("",0.0) if none.""" - from mcp_server.registry import load_skills - from mcp_server.router import Router + from ingot.mcp_server.registry import load_skills + from ingot.mcp_server.router import Router others = [s for s in load_skills() if s.name != skill] if not others: return "", 0.0 @@ -184,17 +185,20 @@ def promotion_gate(skill: str, champ_scores: list[float], chall_scores: list[flo from . import SERVE_TEMPLATE as EVAL_SERVE_TEMPLATE # noqa: E402 -def load_tasks(skill: str, log=print) -> tuple[list[dict], list[dict], dict]: +def load_tasks(skill: str, log=print, + draft_components: dict[str, str] | None = None) -> tuple[list[dict], list[dict], dict]: """Return train, holdout, and split metadata. The candidate search sees train; the gate is judged on holdout, a leakage-clean split so a challenger has to *generalize*, not memorize. A flat `tasks:` list is marked leaky and cannot produce a promotable gate. If no task set exists, the teacher drafts one; that draft must include a real holdout before promotion.""" p = TASKS_DIR / f"{skill}.yaml" if not p.exists(): - from mcp_server.registry import SKILLS_DIR as _SD, read_components - comps = read_components(_SD / skill) + if draft_components is None: + from ingot.mcp_server.registry import read_components + draft_components = read_components(resolve_skill_dir(skill)) from .draft import draft_and_save - draft_and_save(skill, comps["description"], comps["body"], TASKS_DIR, log=log) + draft_and_save(skill, draft_components["description"], draft_components["body"], + TASKS_DIR, log=log) data = yaml.safe_load(p.read_text()) train = data.get("train") or data.get("tasks") or [] explicit_holdout = bool(data.get("holdout")) @@ -205,7 +209,7 @@ def load_tasks(skill: str, log=print) -> tuple[list[dict], list[dict], dict]: def _variant_tools(skill: str, body: str, description: str): """One read-only route tool backed by the variant description and body.""" - from mcp_server.router import Router + from ingot.mcp_server.router import Router skills = load_skills() variants = [replace(item, description=description, body=body) if item.name == skill else item for item in skills] @@ -221,7 +225,7 @@ async def route_and_load(task: str, harness: str, cwd: str, available_tools: lis def _routing_failures(skill: str, challenger: dict, tasks: list[dict]) -> list[str]: - from mcp_server.router import Router + from ingot.mcp_server.router import Router skills = load_skills() variants = [replace(item, description=challenger["description"], body=challenger["body"]) if item.name == skill else item for item in skills] @@ -235,8 +239,8 @@ def _routing_metrics(skill: str, champion: dict, challenger: dict) -> dict | Non cases = data.get("routing") or [] if not cases: return None - from mcp_server.router import Router - from mcp_server.routing_eval import evaluate_cases, evaluate_parity + from ingot.mcp_server.router import Router + from ingot.mcp_server.routing_eval import evaluate_cases, evaluate_parity skills = load_skills() def variant(components): @@ -273,7 +277,7 @@ async def task_fn(*, item, **kwargs): def judge_evaluator(*, input, output, **kwargs): j = judge(input["task"], input["rubric"], str(output), check=input.get("check"), - deliverable=input.get("deliverable")) + deliverable=input.get("deliverable"), checklist=input.get("checklist")) scores_by_task[input["task"]] = j["score"] return Evaluation(name="judge_score", value=j["score"], comment=j["feedback"]) @@ -309,7 +313,8 @@ async def rollout_all(): with ThreadPoolExecutor(max_workers=max(1, len(ok))) as pool: judgments = list(pool.map( lambda x: judge(x[1]["task"], x[1]["rubric"], str(x[2][0]), - check=x[1].get("check"), deliverable=x[1].get("deliverable")), ok)) + check=x[1].get("check"), deliverable=x[1].get("deliverable"), + checklist=x[1].get("checklist")), ok)) for (i, _, _), j in zip(ok, judgments): scores[i] = j["score"] zero = {"input_tokens": 0, "output_tokens": 0} @@ -436,16 +441,14 @@ def run_ab(skill: str, skip_search: bool = False, challenger_file: str | None = usage_ledger.reset() components = components if components is not None else OPTIMIZE_COMPONENTS train, holdout, split = load_tasks(skill) - skill_dir = SKILLS_DIR / skill - if not (skill_dir / "SKILL.md").exists(): - raise SystemExit(f"No skill named '{skill}' in skills/.") + skill_dir = resolve_skill_dir(skill) # description + body always; bundled file components join only when `components` names # them (they then also render into rollouts and the A/B serving). Everything else stays # untouched on disk. champion = optimizable_components(skill_dir) file_components = [c for c in components if c.startswith("file:")] if file_components: - from mcp_server.registry import read_components + from ingot.mcp_server.registry import read_components everything = read_components(skill_dir) champion.update({k: everything[k] for k in file_components if k in everything}) @@ -464,7 +467,7 @@ def run_ab(skill: str, skip_search: bool = False, challenger_file: str | None = return {"skill": skill, "improved": False} log(f"[opt] components changed: {changed}") # checkpoint the candidate so an A/B failure doesn't cost the whole search - ckpt = Path(__file__).resolve().parent.parent / "runs" / f"challenger-{skill}.json" + ckpt = paths.runs() / f"challenger-{skill}.json" ckpt.parent.mkdir(parents=True, exist_ok=True) ckpt.write_text(json.dumps({"components": challenger, "seed_score": seed_score, "best_score": best_score}, indent=2)) @@ -539,7 +542,7 @@ def run_ab(skill: str, skip_search: bool = False, challenger_file: str | None = summary["optimization_usage"] = usage_ledger.report() evidence = build_evidence(summary, champion_skill.revision, skill_revision(Path(champion_skill.root), challenger)) - evidence_root = Path(__file__).resolve().parent.parent / "runs" / "evidence" / skill / str(ts) + evidence_root = paths.runs() / "evidence" / skill / str(ts) evidence_json, evidence_markdown = write_evidence(evidence, evidence_root) summary["evidence"] = evidence summary["evidence_paths"] = {"json": recorded_path(evidence_json), @@ -575,10 +578,8 @@ def script_pass_components(skill: str) -> list[str]: bundles no scripts, or when no holdout task carries an execution-grounded `check:` (the LLM judge alone cannot tell a broken script from a working one), so a scripts pass without checks would produce evidence worth nothing.""" - from mcp_server.registry import read_components - skill_dir = SKILLS_DIR / skill - if not (skill_dir / "SKILL.md").exists(): - raise SystemExit(f"No skill named '{skill}' in skills/.") + from ingot.mcp_server.registry import read_components + skill_dir = resolve_skill_dir(skill) scripts = sorted(k for k in read_components(skill_dir) if k.startswith("file:scripts/")) if not scripts: raise SystemExit(f"'{skill}' bundles no scripts/ files, nothing for the scripts pass " @@ -590,14 +591,14 @@ def script_pass_components(skill: str) -> list[str]: f"'{skill}' has no execution-grounded holdout checks. The scripts pass needs at " f"least one holdout task with a check: {{fixture, assert}} entry so a broken script " f"fails objectively instead of being waved through by the judge. Add one to " - f"optimize/tasks/{skill}.yaml first.") + f"ingot/optimize/tasks/{skill}.yaml first.") return scripts def build_parser() -> argparse.ArgumentParser: """The candidate-generation CLI. Kept out of `__main__` so its rejections are testable: a flag for a pass that does not exist has to fail loudly, not be quietly accepted or ignored.""" - ap = argparse.ArgumentParser(prog="python -m optimize.ab") + ap = argparse.ArgumentParser(prog="python -m ingot.optimize.ab") ap.add_argument("skill") passes = ap.add_mutually_exclusive_group() passes.add_argument("--body", action="store_true", diff --git a/optimize/acceptance.py b/ingot/optimize/acceptance.py similarity index 100% rename from optimize/acceptance.py rename to ingot/optimize/acceptance.py diff --git a/ingot/optimize/agy_judge.py b/ingot/optimize/agy_judge.py new file mode 100644 index 0000000..4d61ef1 --- /dev/null +++ b/ingot/optimize/agy_judge.py @@ -0,0 +1,256 @@ +"""Fail-closed subprocess adapter for the subscription-backed Agy judge.""" +from __future__ import annotations + +import json +import os +import subprocess +import threading +import time +from pathlib import Path +from typing import Mapping, Sequence + + +AGY_MODEL = "gemini-3.6-flash-medium" +AGY_IDENTITY = f"agy/{AGY_MODEL}" + +_VERDICTS = ("pass", "partial", "fail") +_PROVIDER_PREFIXES = ( + "OPENROUTER_", + "OPENAI_", + "ANTHROPIC_", + "GEMINI_", + "GOOGLE_", + "VERTEX_", +) +_PROVIDER_KEYS = frozenset({"BASE_URL", "API_KEY", "MODEL_API_KEY"}) +_LAUNCH_INTERVAL_SECONDS = 3.2 +_RESOURCE_EXHAUSTED_RETRIES = 3 +_launch_lock = threading.Lock() +_next_launch_at = 0.0 + + +class AgyJudgeError(RuntimeError): + """Agy did not produce one trustworthy checklist grade.""" + + +def agy_process_env(parent: Mapping[str, str]) -> dict[str, str]: + """Copy the parent environment without provider credentials or routing controls.""" + child = { + key: value + for key, value in parent.items() + if key not in _PROVIDER_KEYS and not key.startswith(_PROVIDER_PREFIXES) + } + child["AGY_CLI_DISABLE_AUTO_UPDATE"] = "true" + return child + + +def _checklist_ids(checklist: Sequence[Mapping[str, object]]) -> list[str]: + ids = [item.get("id") for item in checklist] + if not ids or any(not isinstance(item_id, str) or not item_id for item_id in ids): + raise AgyJudgeError("Agy checklist requires non-empty string IDs") + if len(set(ids)) != len(ids): + raise AgyJudgeError("Agy checklist IDs must be unique") + return ids + + +def judge_schema(checklist: Sequence[Mapping[str, object]]) -> dict: + """Build the strict Agy output schema for the exact checklist IDs.""" + ids = _checklist_ids(checklist) + item_schema = { + "type": "object", + "additionalProperties": False, + "required": ["verdict", "note"], + "properties": { + "verdict": {"enum": list(_VERDICTS)}, + "note": {"type": "string"}, + }, + } + return { + "type": "object", + "additionalProperties": False, + "required": ["items", "feedback"], + "properties": { + "items": { + "type": "object", + "additionalProperties": False, + "required": ids, + "properties": {item_id: item_schema.copy() for item_id in ids}, + }, + "feedback": {"type": "string"}, + }, + } + + +def _validate_grade(raw: object, checklist: Sequence[Mapping[str, object]]) -> dict: + ids = _checklist_ids(checklist) + if not isinstance(raw, dict): + raise AgyJudgeError("Agy result has no structured output") + if set(raw) != {"items", "feedback"} or not isinstance(raw.get("feedback"), str): + raise AgyJudgeError("Agy structured output has an invalid top-level shape") + items = raw.get("items") + if not isinstance(items, dict) or set(items) != set(ids): + raise AgyJudgeError("Agy structured output does not match the checklist IDs") + for item_id in ids: + item = items[item_id] + if not isinstance(item, dict) or set(item) != {"verdict", "note"}: + raise AgyJudgeError(f"Agy checklist item {item_id!r} has an invalid shape") + if item["verdict"] not in _VERDICTS: + raise AgyJudgeError(f"Agy checklist item {item_id!r} has an unknown verdict") + if not isinstance(item["note"], str): + raise AgyJudgeError(f"Agy checklist item {item_id!r} has an invalid note") + return raw + + +def parse_stream( + stdout: str, + checklist: Sequence[Mapping[str, object]], +) -> tuple[dict, dict]: + """Return the sole successful terminal grade and usage from an Agy JSONL stream.""" + terminal = [] + for line_number, line in enumerate(stdout.splitlines(), start=1): + if not line.strip(): + continue + try: + event = json.loads(line) + except json.JSONDecodeError as exc: + raise AgyJudgeError(f"Agy emitted malformed JSONL on line {line_number}") from exc + if not isinstance(event, dict): + raise AgyJudgeError(f"Agy emitted a non-object event on line {line_number}") + if event.get("event") == "result": + terminal.append(event) + if len(terminal) != 1: + raise AgyJudgeError(f"Agy emitted {len(terminal)} terminal results; expected one") + + result = terminal[0].get("result") + if not isinstance(result, dict): + raise AgyJudgeError("Agy terminal result has an invalid shape") + if result.get("status") != "SUCCESS": + raise AgyJudgeError("Agy terminal result was not successful") + grade = _validate_grade(result.get("structured_output"), checklist) + usage = result.get("usage") + if not isinstance(usage, dict): + raise AgyJudgeError("Agy terminal result has no usage object") + return grade, usage + + +def _runtime_paths() -> tuple[Path, Path]: + agy_value = os.environ.get("AGY_BIN", "").strip() + workspace_value = os.environ.get("AGY_JUDGE_WORKSPACE", "").strip() + if not agy_value or not workspace_value: + raise AgyJudgeError("AGY_BIN and AGY_JUDGE_WORKSPACE must be set") + agy_bin = Path(agy_value) + workspace = Path(workspace_value) + if not agy_bin.is_absolute() or not workspace.is_absolute(): + raise AgyJudgeError("Agy runtime paths must be absolute") + if not agy_bin.is_file() or not os.access(agy_bin, os.X_OK): + raise AgyJudgeError("AGY_BIN is not an executable file") + if not workspace.is_dir(): + raise AgyJudgeError("AGY_JUDGE_WORKSPACE is not a directory") + return agy_bin, workspace + + +def invoke( + prompt: str, + checklist: Sequence[Mapping[str, object]], + *, + timeout: float = 120.0, +) -> tuple[dict, dict]: + """Invoke Agy once and return no grade unless its complete contract validates.""" + agy_bin, workspace = _runtime_paths() + schema = judge_schema(checklist) + argv = [ + str(agy_bin), + "--model", AGY_MODEL, + "--print", prompt, + "--output-format", "stream-json", + "--json-schema", json.dumps(schema), + "--sandbox", + "--mode", "plan", + "--disable-slash-commands", + "--print-timeout", f"{int(timeout)}s", + ] + for attempt in range(_RESOURCE_EXHAUSTED_RETRIES + 1): + _wait_for_launch_slot() + try: + done = subprocess.run( + argv, + cwd=workspace, + env=agy_process_env(os.environ), + text=True, + capture_output=True, + timeout=timeout + 10, + check=False, + ) + except subprocess.TimeoutExpired as exc: + raise AgyJudgeError("Agy judge timed out") from exc + except OSError as exc: + raise AgyJudgeError("Agy judge process could not start") from exc + if done.returncode == 0: + return parse_stream(done.stdout, checklist) + if not _resource_exhausted(done.stdout) or attempt == _RESOURCE_EXHAUSTED_RETRIES: + raise AgyJudgeError(f"Agy judge exited with status {done.returncode}") + raise AssertionError("unreachable") + + +def _wait_for_launch_slot() -> None: + """Keep the subscription eligibility gate below its observed burst limit.""" + global _next_launch_at + with _launch_lock: + now = time.monotonic() + delay = max(0.0, _next_launch_at - now) + if delay: + time.sleep(delay) + now = time.monotonic() + _next_launch_at = now + _LAUNCH_INTERVAL_SECONDS + + +def _resource_exhausted(stdout: str) -> bool: + for line in stdout.splitlines(): + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + result = event.get("result") if isinstance(event, dict) else None + if not isinstance(result, dict) or result.get("status") != "ERROR": + continue + error = str(result.get("error", "")) + if "RESOURCE_EXHAUSTED" in error and "429" in error: + return True + return False + + +def _run_preflight(argv: list[str], workspace: Path) -> str: + try: + done = subprocess.run( + argv, + cwd=workspace, + env=agy_process_env(os.environ), + text=True, + capture_output=True, + timeout=20.0, + check=False, + ) + except subprocess.TimeoutExpired as exc: + raise AgyJudgeError("Agy preflight timed out") from exc + except OSError as exc: + raise AgyJudgeError("Agy preflight process could not start") from exc + if done.returncode != 0: + raise AgyJudgeError(f"Agy preflight exited with status {done.returncode}") + if not done.stdout.strip(): + raise AgyJudgeError("Agy preflight returned no output") + return done.stdout.strip() + + +def preflight() -> dict[str, object]: + """Verify the explicit runtime and return its fixed judge provenance.""" + agy_bin, workspace = _runtime_paths() + version = _run_preflight([str(agy_bin), "--version"], workspace) + models = _run_preflight([str(agy_bin), "models"], workspace) + if AGY_MODEL not in models.split(): + raise AgyJudgeError(f"Agy model {AGY_MODEL!r} is unavailable") + return { + "identity": AGY_IDENTITY, + "model": AGY_MODEL, + "version": version, + "billing_mode": "subscription", + } diff --git a/ingot/optimize/cluster.py b/ingot/optimize/cluster.py new file mode 100644 index 0000000..57b458a --- /dev/null +++ b/ingot/optimize/cluster.py @@ -0,0 +1,205 @@ +"""Group the library into category buckets from the embeddings the router already uses. + +A merged library is a flat list of names. Provenance folders say where a skill came from, which is +a fact about its origin and not about what it does — "authored here" holds testing skills next to +finance skills. This clusters on meaning instead, so the console can show what the library is +actually about and where it is thin. + +Not UMAP + HDBSCAN, which is what the same view uses over 100k prompts elsewhere — but the reduction +those bring is not optional. Clustering the raw 1024-d embeddings scored silhouette +0.05 and put +half the library in one bucket, because at that width every pair of a hundred points sits at +roughly the same distance. Over the top 10 principal components the same run scores +0.28. So: +cosine k-means on an SVD projection, numpy only, with the reduction doing the job UMAP does there. + +Clustering is not cheap enough to run inside a request (it loads the embedding model), so this is a +command that writes runs/clusters.json and the UI serves the file. + + python -m ingot.optimize.cluster +""" +import json +import os +import re +from collections import Counter +from pathlib import Path + +import numpy as np +from ingot import paths + +CLUSTER_PATH = paths.runs() / "clusters.json" +# Enough buckets to separate concerns, few enough that each one still means something. Bounded +# because both ends degenerate: k=2 says nothing, k=n gives every skill its own bucket. +K_MIN, K_MAX = 4, 12 +# Components to cluster over. Measured on the 102-skill library: raw 1024d scores silhouette +# +0.05, 10d +0.28, 20d +0.19, 40d +0.13. Keep enough signal to separate topics, few enough +# dimensions that distances still mean something. +CLUSTER_DIMS = 10 +_WORD = re.compile(r"[a-z][a-z0-9+-]{2,}") +# Words that describe every skill in a library of instructions and so distinguish none of them. +_STOP = { + "the", "and", "for", "when", "with", "this", "that", "you", "your", "use", "uses", "used", + "using", "user", "from", "into", "not", "any", "are", "its", "has", "have", "was", "will", + "can", "should", "must", "need", "needs", "want", "wants", "asks", "ask", "them", "they", + "what", "why", "how", "who", "which", "than", "then", "there", "here", "over", "under", + "before", "after", "each", "every", "some", "all", "one", "two", "new", "own", "out", + "run", "runs", "get", "gets", "set", "sets", "make", "makes", "does", "done", "via", + "skill", "skills", "task", "tasks", "work", "works", "write", "writes", "write-up", + # Trigger-phrase vocabulary. A routing description is written to be matched ("Use whenever the + # user asks…"), so this register appears in most of them and named a bucket "whenever · api". + "whenever", "asking", "request", "requests", "mentions", "triggers", "trigger", "wants", + "needed", "including", "instead", "rather", "across", "against", "within", "about", +} + + +def skill_texts(skills) -> list[str]: + """What gets embedded. The name carries real signal in this library (`aws-cdk`, + `writing-clearly`), so it leads, and the description is the routing trigger the router itself + matches on — clustering the same text keeps the buckets consistent with routing behaviour.""" + return [f"{s.name.replace('-', ' ')}. {s.description}" for s in skills] + + +def _normalise(matrix: np.ndarray) -> np.ndarray: + norms = np.linalg.norm(matrix, axis=1, keepdims=True) + return matrix / np.maximum(norms, 1e-12) + + +def kmeans(vectors: np.ndarray, k: int, seed: int = 0, iters: int = 60): + """k-means++ init, Lloyd iterations, on unit vectors — so squared euclidean ranks the same as + cosine. Seeded: the same library must produce the same buckets twice, or the view reshuffles + under a poll and nobody can trust what they are looking at.""" + rng = np.random.default_rng(seed) + n = len(vectors) + centres = [vectors[rng.integers(n)]] + for _ in range(k - 1): # k-means++: sample far from what is already chosen + d2 = np.min(((vectors[:, None, :] - np.array(centres)[None]) ** 2).sum(-1), axis=1) + total = d2.sum() + centres.append(vectors[rng.choice(n, p=d2 / total) if total > 0 else rng.integers(n)]) + centres = np.array(centres) + labels = np.zeros(n, dtype=int) + for _ in range(iters): + labels = np.argmin(((vectors[:, None, :] - centres[None]) ** 2).sum(-1), axis=1) + moved = False + for j in range(k): + members = vectors[labels == j] + if not len(members): # an emptied centre is re-seeded, never left to + members = vectors[rng.integers(n)][None] # collapse the run to k-1 buckets + centre = _normalise(members.mean(0, keepdims=True))[0] + if not np.allclose(centre, centres[j]): + centres[j], moved = centre, True + if not moved: + break + return labels, centres + + +def silhouette(vectors: np.ndarray, labels: np.ndarray) -> float: + """Mean silhouette, used only to pick k. A bucket count chosen by hand is a guess that ages + badly as the library grows.""" + unique = np.unique(labels) + if len(unique) < 2: + return -1.0 + dist = np.sqrt(np.maximum(((vectors[:, None, :] - vectors[None]) ** 2).sum(-1), 0)) + scores = [] + for i in range(len(vectors)): + same = labels == labels[i] + same[i] = False + if not same.any(): + continue # a singleton has no cohesion to measure + a = dist[i][same].mean() + b = min(dist[i][labels == other].mean() for other in unique if other != labels[i]) + scores.append((b - a) / max(a, b)) + return float(np.mean(scores)) if scores else -1.0 + + +def project(vectors: np.ndarray, dims: int) -> np.ndarray: + """Top `dims` principal components.""" + centred = vectors - vectors.mean(0, keepdims=True) + _, _, vt = np.linalg.svd(centred, full_matrices=False) + return centred @ vt[:dims].T + + +def project_2d(vectors: np.ndarray) -> np.ndarray: + return project(vectors, 2) + + +def top_terms(texts: list[str], members: list[int], k: int = 6) -> list[str]: + """Terms frequent inside the bucket and rare outside it. Raw frequency returns the words every + skill uses, which names nothing.""" + inside, outside = Counter(), Counter() + member_set = set(members) + for i, text in enumerate(texts): + words = {w for w in _WORD.findall(text.lower()) if w not in _STOP} + (inside if i in member_set else outside).update(words) + n_in, n_out = max(len(members), 1), max(len(texts) - len(members), 1) + scored = {w: (c / n_in) - (outside[w] / n_out) for w, c in inside.items() if c > 1 or n_in < 3} + return [w for w, _ in sorted(scored.items(), key=lambda kv: -kv[1])[:k]] + + +def label_for(terms: list[str]) -> str: + return " · ".join(terms[:3]) if terms else "unlabelled" + + +def build(log=print) -> dict: + from ingot.mcp_server.embedding import EMBED_MODEL, build_embedding + from ingot.mcp_server.registry import load_skills + + skills = load_skills() + if len(skills) < K_MIN: + raise SystemExit(f"only {len(skills)} skills indexed; clustering needs at least {K_MIN}. " + f"Check SKILL_ROUTER_PATHS.") + texts = skill_texts(skills) + log(f"[cluster] embedding {len(skills)} skills with {EMBED_MODEL}…") + vectors = _normalise(np.array([np.asarray(v, dtype=float) + for v in build_embedding().embed(texts)])) + + # Cluster in the reduced space, not the raw one. At 1024 dimensions over a hundred points every + # pair sits at a similar distance (measured on this library: cosine p25 0.42, median 0.51, + # p75 0.60) and k-means has nothing to bite on — it scored silhouette +0.05 and put half the + # library in one bucket. The same run over the top 10 components scores +0.28. This is what UMAP + # is for in the version of this view that runs over 100k prompts; at this size the projection + # the layout already needs is enough, taken a few more components deep. + reduced = _normalise(project(vectors, min(CLUSTER_DIMS, len(skills) - 1))) + + upper = min(K_MAX, len(skills) // 2) + best = None + for k in range(K_MIN, max(K_MIN, upper) + 1): + labels, centres = kmeans(reduced, k) + score = silhouette(reduced, labels) + log(f"[cluster] k={k:<3} silhouette {score:+.3f}") + if best is None or score > best[0]: + best = (score, k, labels, centres) + score, k, labels, centres = best + log(f"[cluster] chose k={k} (silhouette {score:+.3f})") + + xy = project_2d(vectors) + clusters = [] + for j in range(k): + members = [i for i in range(len(skills)) if labels[i] == j] + if not members: + continue + terms = top_terms(texts, members) + clusters.append({ + "id": j, "label": label_for(terms), "size": len(members), "top_terms": terms, + "centroid_2d": [float(xy[members, 0].mean()), float(xy[members, 1].mean())], + "members": sorted(({"name": skills[i].name, + "xy": [float(xy[i, 0]), float(xy[i, 1])], + "provenance": (skills[i].metadata or {}).get("provenance", "")} + for i in members), key=lambda m: m["name"]), + }) + clusters.sort(key=lambda c: -c["size"]) + return {"version": 1, "n_skills": len(skills), "k": k, "silhouette": round(score, 4), + "embedder": EMBED_MODEL, "clusterer": f"cosine k-means (k={k}, seeded)", + "reducer": f"pca-{CLUSTER_DIMS}d cluster / pca-2d layout", "clusters": clusters} + + +def build_and_save(log=print) -> Path: + data = build(log=log) + CLUSTER_PATH.parent.mkdir(parents=True, exist_ok=True) + CLUSTER_PATH.write_text(json.dumps(data, indent=2)) + log(f"[cluster] {data['k']} buckets over {data['n_skills']} skills → {CLUSTER_PATH}") + for c in data["clusters"]: + log(f" {c['size']:>3} {c['label']}") + return CLUSTER_PATH + + +if __name__ == "__main__": + os.environ.setdefault("TOKENIZERS_PARALLELISM", "false") + build_and_save() diff --git a/ingot/optimize/compat.py b/ingot/optimize/compat.py new file mode 100644 index 0000000..76fabdc --- /dev/null +++ b/ingot/optimize/compat.py @@ -0,0 +1,155 @@ +"""Cross-model skill compatibility, how well a skill's body transfers across serving models. + +A skill body is tuned for one serving model (`AGENT_MODEL`); SkillOpt's own result is that good +skills transfer, but not always. For each model in `COMPAT_MODELS`, this runs the skill's held-out +tasks through the one serving contract twice, once with the skill body, once with an empty body +(the no-skill baseline), judges both with the FIXED judge, and reports per-model **lift** +(skill mean − baseline mean). Positive lift = the body helps that model; ~0 = the model already +knows this and the body is dead weight there. + +Langfuse-free: it reuses the direct rollout + judge (the same path the inner loop uses), so it needs +no trace backend or experiment logging. Only the *serving* model varies, the judge stays fixed so +scores are comparable across models. + +Usage: python -m ingot.optimize.compat +Config: COMPAT_MODELS=qwen/qwen3-32b,openai/gpt-5.5,anthropic/claude-sonnet-... (default: AGENT_MODEL) +""" +import hashlib +import json +import os +import statistics +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +from langchain_openai import ChatOpenAI + +from ingot.mcp_server.registry import optimizable_components + +from . import (SERVE_TEMPLATE, agent_model, api_key, client_kwargs, configured_models, + model_api_key, model_base_url, resolve_skill_dir, teacher_base_url) +from . import usage as usage_ledger +from .ab import load_tasks +from .judge import invoke_retry, judge +from .rollout import assemble +from ingot import paths + +_MAX_WORKERS = 8 +COMPAT_DIR = paths.runs() / "compat" +BASELINE_CACHE_DIR = paths.runs() / "compat-baseline" +# The no-skill baseline: the identical serving contract with no skill body, so `lift` isolates the +# body's contribution rather than the difference between two different prompts. +NO_SKILL_BODY = "(no skill loaded, answer the task from your own knowledge)" + + +def compat_models() -> list[str]: + """Models to sweep: COMPAT_MODELS (comma-separated), else just the configured AGENT_MODEL.""" + models = configured_models("COMPAT_MODELS") + return models or [agent_model()] + + +def _llm(model: str): + # Route each row to the endpoint that actually serves it. The model this box serves + # (AGENT_MODEL) comes from MODEL_BASE_URL — a local vLLM, and therefore a free row in the grid; + # every other slug goes to the hosted endpoint. Sending the whole sweep to one endpoint meant + # either the local row was impossible, or MODEL_BASE_URL had to be overridden by hand for the + # run, which loses the free reference row exactly when you want to compare against it. + # reasoning is left at the provider default, some models reject the flag. + if model == agent_model(): + base, key = model_base_url(), model_api_key() + else: + base, key = teacher_base_url(), api_key() + return ChatOpenAI(model=model, temperature=0, **client_kwargs(base, key=key)) + + +def _baseline_cache_key(model: str, holdout: list[dict]) -> str: + """The no-skill baseline is a pure function of (serving model, held-out tasks, judge): the skill + body is precisely what it leaves out, so editing the skill cannot change it. Uncached, every + re-sweep paid again for byte-identical work — half the cost of every run after the first.""" + payload = json.dumps({"v": 1, "model": model, "holdout": holdout, + "judge": os.environ.get("JUDGE_MODELS") or os.environ.get("JUDGE_MODEL", "")}, + sort_keys=True) + return hashlib.sha256(payload.encode()).hexdigest()[:16] + + +def _score(llm, system: str, task: dict, role: str = "compat") -> float: + msg = invoke_retry(llm, [("system", system), ("user", task["task"])]) + usage_ledger.add(role, getattr(msg, "usage_metadata", None)) + return judge(task["task"], task["rubric"], msg.content, + check=task.get("check"), deliverable=task.get("deliverable"), + checklist=task.get("checklist"))["score"] + + +def _run_arm(llm, system: str, tasks: list[dict], role: str = "compat") -> list[float]: + with ThreadPoolExecutor(max_workers=min(_MAX_WORKERS, len(tasks))) as pool: + return list(pool.map(lambda t: _score(llm, system, t, role), tasks)) + + +def _sweep_model(model: str, skill_system: str, base_system: str, holdout: list[dict], + log=print) -> dict: + """One row of the matrix: this model with the skill body, and without it.""" + llm = _llm(model) + # Bill each arm to its own model: one "compat" bucket cannot be priced, because the whole + # point of the sweep is that the serving model changes underneath it. + role = f"compat:{model}" + skill_scores = _run_arm(llm, skill_system, holdout, role) + cache_path = BASELINE_CACHE_DIR / f"{_baseline_cache_key(model, holdout)}.json" + if cache_path.exists(): + base_scores = json.loads(cache_path.read_text()) + log(f"[compat] {model:<34} baseline reused from cache (no spend)") + else: + base_scores = _run_arm(llm, base_system, holdout, role) + BASELINE_CACHE_DIR.mkdir(parents=True, exist_ok=True) + cache_path.write_text(json.dumps(base_scores)) + s_mean, b_mean = statistics.mean(skill_scores), statistics.mean(base_scores) + verdict = "helps" if s_mean - b_mean > 0.05 else "no lift" if s_mean - b_mean >= -0.05 else "HURTS" + log(f"[compat] {model:<34} skill {s_mean:.3f} baseline {b_mean:.3f} " + f"lift {s_mean - b_mean:+.3f} ({verdict})") + return {"skill_mean": s_mean, "baseline_mean": b_mean, "lift": s_mean - b_mean, + "skill_scores": skill_scores, "baseline_scores": base_scores} + + +def run_compat(skill: str, log=print) -> dict: + """Sweep COMPAT_MODELS over the skill's held-out tasks (skill vs no-skill) and write the matrix + to runs/compat/.json. Returns the summary.""" + usage_ledger.reset() + skill_dir = resolve_skill_dir(skill) + _, holdout, _ = load_tasks(skill) + if not holdout: + raise SystemExit(f"'{skill}' has no held-out eval tasks to run.") + skill_system = SERVE_TEMPLATE.format(body=assemble(optimizable_components(skill_dir))) + base_system = SERVE_TEMPLATE.format(body=NO_SKILL_BODY) + models = compat_models() + log(f"[compat] '{skill}': {len(holdout)} held-out tasks × {len(models)} model(s); " + f"judge fixed, serving model varies") + + models_out = {} + for model in models: + # One model the endpoint cannot serve must not discard the rows already paid for. A slug + # with no ZDR-qualified endpoint 404s on the first call, and before this the whole sweep + # died there — losing every earlier model's scores and writing no matrix at all. + try: + models_out[model] = _sweep_model(model, skill_system, base_system, holdout, log) + except Exception as error: # noqa: BLE001 - any provider failure is one unusable row + models_out[model] = {"error": f"{type(error).__name__}: {error}"[:400]} + log(f"[compat] {model:<34} UNAVAILABLE ({type(error).__name__}), skipped") + if not any("error" not in row for row in models_out.values()): + raise SystemExit(f"[compat] no model in COMPAT_MODELS could be reached for '{skill}'; " + f"nothing was measured.") + + summary = {"skill": skill, "tasks": len(holdout), + "judge": os.environ.get("JUDGE_MODELS") or os.environ.get("JUDGE_MODEL", ""), + "models": models_out, "usage": usage_ledger.report()} + COMPAT_DIR.mkdir(parents=True, exist_ok=True) + path = COMPAT_DIR / f"{skill}.json" + path.write_text(json.dumps(summary, indent=2)) + log(f"[compat] matrix written to {path}") + log(usage_ledger.format_report()) + return summary + + +if __name__ == "__main__": + import sys + + from . import require_openrouter_key + require_openrouter_key() + run_compat(sys.argv[1] if len(sys.argv) > 1 else "tailwind") diff --git a/optimize/draft.py b/ingot/optimize/draft.py similarity index 62% rename from optimize/draft.py rename to ingot/optimize/draft.py index 59c68c2..5528baa 100644 --- a/optimize/draft.py +++ b/ingot/optimize/draft.py @@ -2,7 +2,7 @@ immediately optimizable. The authoring model (SKILLOPT_MODEL) reads the skill's description + body and writes train/holdout tasks with judge rubrics, split by *operation* so the holdout tests generalization, not recall. Kept out of the MCP server (no LLM in its hot serving path); the -optimizer calls it on demand when `optimize/tasks/.yaml` is missing.""" +optimizer calls it on demand when `ingot/optimize/tasks/.yaml` is missing.""" import json import os import re @@ -27,21 +27,64 @@ LLM judge (what a correct answer must contain). Phrase tasks so the answer is the deliverable itself (e.g. "Write Python code that…"), not a request to go find files. -Return ONLY JSON: {{"tasks": [{{"task": "...", "rubric": "..."}}, ...]}} with exactly {n} items, -each covering a different operation/capability.""" +Each task also carries a CHECKLIST of {items} independent checks that decide its score. Write checks +a grader can answer without re-reading the whole answer, and that a good and a bad answer would +genuinely split on: + +- Each check tests ONE observable property. "Handles the empty input case" is a check; "is high + quality" is not. +- Make them specific to THIS task, not generic writing advice. Prefer things the skill body says + matter. +- id: short snake_case, unique within the task. weight: 1 (minor) to 5 (the point of the task). +- dimension: one of correctness, completeness, instruction_following, efficiency. + +Return ONLY JSON with exactly {n} items, each covering a different operation/capability: +{{"tasks": [{{"task": "...", "rubric": "...", + "checklist": [{{"id": "...", "criterion": "...", "weight": 3, + "dimension": "correctness"}}, ...]}}, ...]}}""" def _llm(): return ChatOpenAI(model=MODEL, temperature=0.4, **client_kwargs(teacher_base_url())) -def draft_tasks(name: str, description: str, body: str, n: int = 8) -> dict: - """Draft n tasks and split them evenly into train/holdout (disjoint operations).""" - msg = _llm().invoke(_PROMPT.format(name=name, description=description, body=body[:6000], n=n)) +_ID_RE = re.compile(r"^[a-z][a-z0-9_]{1,39}$") + + +def _clean_checklist(raw) -> list[dict]: + """Keep only checks a grader can actually apply, and drop the rest rather than shipping a + rubric with unusable items in it. An empty result is fine: judge() falls back to its default + checklist, which still grades four dimensions independently.""" + from .judge import DIMENSIONS + out, seen = [], set() + for item in raw if isinstance(raw, list) else []: + if not isinstance(item, dict): + continue + item_id, criterion = str(item.get("id", "")).strip().lower(), str(item.get("criterion", "")).strip() + if not _ID_RE.match(item_id) or item_id in seen or len(criterion) < 8: + continue + try: + weight = min(5, max(1, int(item.get("weight", 1)))) + except (TypeError, ValueError): + weight = 1 + dimension = str(item.get("dimension", "")).strip().lower() + seen.add(item_id) + out.append({"id": item_id, "criterion": criterion, "weight": weight, + "dimension": dimension if dimension in DIMENSIONS else "correctness"}) + return out + + +def draft_tasks(name: str, description: str, body: str, n: int = 8, items: int = 6) -> dict: + """Draft n tasks, each with its own weighted checklist, split evenly into train/holdout + (disjoint operations).""" + msg = _llm().invoke(_PROMPT.format(name=name, description=description, body=body[:6000], + n=n, items=items)) usage_ledger.add("draft", getattr(msg, "usage_metadata", None)) m = re.search(r"\{.*\}", msg.content, re.DOTALL) tasks = (json.loads(m.group(0)) if m else {}).get("tasks", []) - tasks = [{"task": str(t["task"]), "rubric": str(t.get("rubric", ""))} for t in tasks if t.get("task")] + tasks = [{"task": str(t["task"]), "rubric": str(t.get("rubric", "")), + "checklist": _clean_checklist(t.get("checklist"))} + for t in tasks if t.get("task")] if len(tasks) < 4: raise SystemExit(f"draft_tasks: teacher returned only {len(tasks)} usable tasks for '{name}'.") half = len(tasks) // 2 @@ -55,7 +98,13 @@ def draft_and_save(name: str, description: str, body: str, tasks_dir, n: int = 8 data = draft_tasks(name, description, body, n=n) path = Path(tasks_dir) / f"{name}.yaml" path.write_text(yaml.safe_dump(data, sort_keys=False, allow_unicode=True, width=100000)) - log(f"[draft] wrote {len(data['train'])} train + {len(data['holdout'])} holdout tasks → {path}") + checks = sum(len(t["checklist"]) for t in data["train"] + data["holdout"]) + bare = [t for t in data["train"] + data["holdout"] if not t["checklist"]] + log(f"[draft] wrote {len(data['train'])} train + {len(data['holdout'])} holdout tasks, " + f"{checks} graded checks → {path}") + if bare: # silently falling back to the default checklist would read as a richer set than it is + log(f"[draft] {len(bare)} task(s) got no usable checklist and will grade on the default " + f"four dimensions; edit {path} to add checks that matter for them.") return path @@ -89,7 +138,7 @@ def draft_routing_cases(name: str, description: str, body: str, if len(pos) < 2 or len(neg) < 1: raise SystemExit(f"draft_routing_cases: teacher returned {len(pos)} positive / {len(neg)} " f"negative cases for '{name}', need at least 2/1. Re-run or hand-write " - f"a routing: block in optimize/tasks/{name}.yaml.") + f"a routing: block in ingot/optimize/tasks/{name}.yaml.") cases = [] for i, task in enumerate(pos): case = {"task": task, "expected": name, "harness": "claude" if i == 1 else "codex"} diff --git a/optimize/evidence.py b/ingot/optimize/evidence.py similarity index 92% rename from optimize/evidence.py rename to ingot/optimize/evidence.py index 5070c8c..009d3fe 100644 --- a/optimize/evidence.py +++ b/ingot/optimize/evidence.py @@ -4,21 +4,28 @@ import json from dataclasses import dataclass from pathlib import Path +from ingot import paths SCHEMA = "skill-router/evidence/v1" ROUTING_SCHEMA = "skill-router/evidence/routing/v1" -_REPO_ROOT = Path(__file__).resolve().parent.parent +_PACKAGE_ROOT = Path(__file__).resolve().parent.parent def recorded_path(path: Path) -> str: - """How an evidence location is written into a pending record: relative to the repo root. - A bundle written inside a container is then still resolvable from the host checkout, and the - review surface has a path it can contain to runs/evidence.""" - try: - return path.resolve().relative_to(_REPO_ROOT).as_posix() - except ValueError: - return str(path) + """How an evidence location is written into a pending record: relative to the state root. + + A bundle written inside a container is then still resolvable from the host, and the review + surface has a path it can contain to runs/evidence. The package root is tried second so a + record written before state moved out of the code directory still reads as `runs/evidence/...` + rather than an absolute path from someone else's machine.""" + resolved = path.resolve() + for anchor in (paths.runs().parent, _PACKAGE_ROOT): + try: + return resolved.relative_to(anchor).as_posix() + except ValueError: + continue + return str(resolved) def first_divergence(champion: list[dict], challenger: list[dict]) -> dict | None: diff --git a/optimize/execcheck.py b/ingot/optimize/execcheck.py similarity index 99% rename from optimize/execcheck.py rename to ingot/optimize/execcheck.py index 271bee1..e1ad339 100644 --- a/optimize/execcheck.py +++ b/ingot/optimize/execcheck.py @@ -92,7 +92,7 @@ def _sandbox(spec: dict, timeout: int) -> dict | None: "--env", "PYTHONPATH=/app"] if SANDBOX_RUNTIME: cmd += ["--runtime", SANDBOX_RUNTIME] - cmd += [SANDBOX_IMAGE, "python", "-m", "optimize.sandbox_driver"] + cmd += [SANDBOX_IMAGE, "python", "-m", "ingot.optimize.sandbox_driver"] try: run = subprocess.run(cmd, input=json.dumps({**spec, "timeout": timeout}), capture_output=True, text=True, timeout=timeout * 3 + 30) diff --git a/ingot/optimize/harbor-shared-network.compose.yml b/ingot/optimize/harbor-shared-network.compose.yml new file mode 100644 index 0000000..ac1ae0f --- /dev/null +++ b/ingot/optimize/harbor-shared-network.compose.yml @@ -0,0 +1,4 @@ +networks: + default: + external: true + name: ingot-harbor-trials diff --git a/ingot/optimize/harbor_catalog.py b/ingot/optimize/harbor_catalog.py new file mode 100644 index 0000000..54e26d4 --- /dev/null +++ b/ingot/optimize/harbor_catalog.py @@ -0,0 +1,386 @@ +"""Durable skill-catalog caller for :func:`ingot.optimize.harbor_eval.run_local_sweep`. + +One controller owns this filesystem queue. Each item invokes the restart-safe native Harbor +one-skill path; Harbor trial, telemetry, grade, and publication receipts remain the source of truth. +The catalog files schedule those calls and never duplicate their evidence state. +""" +from __future__ import annotations + +import argparse +import fcntl +import hashlib +import json +import os +import shutil +import subprocess +import time +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Iterable, Mapping, Sequence + +import yaml + +from ingot.mcp_server.registry import load_skills, skill_revision +from ingot.optimize import resolve_skill_dir +from ingot.optimize.ab import TASKS_DIR +from ingot.optimize.harbor_eval import HARBOR_DIR, SCORING_REVISION, _task_fingerprint, run_local_sweep +from ingot.optimize.harbor_native import NATIVE_RUNNER_REVISION, native_trial_identity +from ingot.optimize.harbor_targets import HARNESS_PROTOCOLS, LocalTarget, discover_target, parse_target + + +_SCHEMA = 1 +CATALOG_OWNER = HARBOR_DIR / "catalog.controller.lock" + + +def _atomic_json(path: Path, payload: dict) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + temporary = path.with_name(f".{path.name}.{os.getpid()}.tmp") + try: + with temporary.open("w", encoding="utf-8") as handle: + json.dump(payload, handle, sort_keys=True, separators=(",", ":")) + handle.write("\n") + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary, path) + finally: + temporary.unlink(missing_ok=True) + + +@dataclass(frozen=True) +class CatalogIntent: + skill: str + skill_sha256: str + task_fingerprint: str + target_specs: tuple[str, ...] + target_fingerprints: tuple[str, ...] + harnesses: tuple[str, ...] + runtime_revisions: tuple[tuple[str, str], ...] + publish_root: str = str(HARBOR_DIR) + global_concurrency: int = 16 + endpoint_concurrency: int = 2 + priority: int = 100 + + def __post_init__(self) -> None: + if not self.skill or len(self.skill_sha256) != 64 or len(self.task_fingerprint) != 64: + raise ValueError("catalog intent requires skill and full SHA-256 identities") + if not self.target_specs or len(self.target_specs) != len(self.target_fingerprints): + raise ValueError("catalog intent target specs and fingerprints must align") + if not self.harnesses or any(item not in HARNESS_PROTOCOLS for item in self.harnesses): + raise ValueError("catalog intent contains no harnesses or an unknown harness") + if len(set(self.target_specs)) != len(self.target_specs) or len(set(self.harnesses)) != len(self.harnesses): + raise ValueError("catalog intent contains duplicate targets or harnesses") + if (self.global_concurrency < 1 or self.endpoint_concurrency < 1 + or self.endpoint_concurrency > self.global_concurrency): + raise ValueError("catalog concurrency requires 1 <= endpoint <= global") + if not self.publish_root or not Path(self.publish_root).is_absolute(): + raise ValueError("catalog publish root must be an absolute path") + + def identity_payload(self) -> dict: + payload = asdict(self) + payload.pop("priority") + payload["target_specs"] = list(self.target_specs) + payload["target_fingerprints"] = list(self.target_fingerprints) + payload["harnesses"] = list(self.harnesses) + payload["runtime_revisions"] = [list(item) for item in self.runtime_revisions] + return payload + + @property + def digest(self) -> str: + encoded = json.dumps(self.identity_payload(), sort_keys=True, + separators=(",", ":")).encode() + return hashlib.sha256(encoded).hexdigest() + + +def _heldout(skill: str) -> list[dict] | None: + path = TASKS_DIR / f"{skill}.yaml" + if not path.is_file(): + return None + data = yaml.safe_load(path.read_text()) + if not isinstance(data, dict): + return None + holdout = data.get("holdout") or data.get("train") or data.get("tasks") or [] + return holdout if isinstance(holdout, list) and holdout else None + + +def _skill_sha(skill: str) -> str: + return hashlib.sha256((resolve_skill_dir(skill) / "SKILL.md").read_bytes()).hexdigest() + + +def _intent_for_skill(skill: str, target_specs: Sequence[str], harnesses: Sequence[str], + *, priority: int = 100, global_concurrency: int = 16, + endpoint_concurrency: int = 2, + publish_root: Path | str = HARBOR_DIR) -> CatalogIntent | None: + provisional = tuple(parse_target(spec) for spec in target_specs) + targets = tuple(discover_target(target.alias, target.base_url) for target in provisional) + holdout = _heldout(skill) + if not holdout: + return None + source = resolve_skill_dir(skill) + revisions = [("harbor", "0.20.0"), ("runner", NATIVE_RUNNER_REVISION), + ("scoring", SCORING_REVISION), ("skill-tree", skill_revision(source))] + revisions.extend( + (f"route:{target.fingerprint}:{harness}", + f"{native_trial_identity(target, harness, 'skill').protocol}/" + f"{native_trial_identity(target, harness, 'skill').gateway_revision}/" + f"context={target.context_length}") + for target in targets for harness in harnesses + ) + return CatalogIntent( + skill=skill, + skill_sha256=_skill_sha(skill), + task_fingerprint=_task_fingerprint(holdout), + target_specs=tuple(target_specs), + target_fingerprints=tuple(target.fingerprint for target in targets), + harnesses=tuple(harnesses), + runtime_revisions=tuple(revisions), + publish_root=str(Path(publish_root)), + global_concurrency=global_concurrency, + endpoint_concurrency=endpoint_concurrency, + priority=priority, + ) + + +def _intent_document(intent: CatalogIntent) -> dict: + return {"schema": _SCHEMA, "digest": intent.digest, "identity": intent.identity_payload()} + + +def _state_path(root: Path, digest: str) -> Path: + return root / "state" / f"{digest}.json" + + +def _read_json(path: Path) -> dict: + value = json.loads(path.read_text()) + if not isinstance(value, dict): + raise RuntimeError(f"catalog record is not an object: {path}") + return value + + +def enqueue_catalog(root: Path | str, intents: Iterable[CatalogIntent]) -> list[Path]: + """Persist measurement intents; scheduling priority never changes content identity.""" + root = Path(root) + intent_dir = root / "intents" + known = [] + if intent_dir.is_dir(): + known = [_read_json(path) for path in intent_dir.glob("*.json")] + written: list[Path] = [] + for supplied in intents: + changed = any(item.get("identity", {}).get("skill") == supplied.skill + and item.get("identity", {}).get("skill_sha256") != supplied.skill_sha256 + for item in known) + path = intent_dir / f"{supplied.digest}.json" + state_path = _state_path(root, supplied.digest) + if not path.exists(): + _atomic_json(path, _intent_document(supplied)) + known.append(_intent_document(supplied)) + if state_path.exists(): + state = _read_json(state_path) + if state.get("status") != "complete": + if state.get("status") == "superseded": + state["status"] = "pending" + state.pop("error", None) + state.pop("finished_at", None) + state["priority"] = max(int(state.get("priority", 0)), + 300 if changed else 200) + _atomic_json(state_path, state) + else: + _atomic_json(state_path, {"schema": _SCHEMA, "intent_digest": supplied.digest, + "status": "pending", + "priority": max(supplied.priority, 300 if changed else 100)}) + written.append(path) + return written + + +def _process_start_token(pid: int) -> str | None: + try: + return Path(f"/proc/{pid}/stat").read_text().split()[21] + except (OSError, IndexError): + done = subprocess.run(["ps", "-o", "lstart=", "-p", str(pid)], capture_output=True, + text=True) + return done.stdout.strip() or None + + +def _claim_controller(path: Path): + """Hold one kernel lock across all catalog roots; no stale-receipt unlink race exists.""" + path.parent.mkdir(parents=True, exist_ok=True) + handle = path.open("a+", encoding="utf-8") + try: + fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError as error: + handle.seek(0) + try: + current = json.load(handle) + except (ValueError, OSError): + current = {} + handle.close() + raise RuntimeError(f"catalog already has live controller PID {current.get('pid', 'unknown')}") from error + handle.seek(0) + handle.truncate() + json.dump({"schema": _SCHEMA, "pid": os.getpid(), + "start_token": _process_start_token(os.getpid())}, handle, sort_keys=True) + handle.flush() + os.fsync(handle.fileno()) + return handle + + +def _load_intent(path: Path, priority: int) -> CatalogIntent: + document = _read_json(path) + identity = document.get("identity") + if not isinstance(identity, dict) or document.get("digest") != path.stem: + raise RuntimeError(f"catalog intent identity is invalid: {path}") + revisions = tuple(tuple(item) for item in identity["runtime_revisions"]) + intent = CatalogIntent(**{**identity, "target_specs": tuple(identity["target_specs"]), + "target_fingerprints": tuple(identity["target_fingerprints"]), + "harnesses": tuple(identity["harnesses"]), + "runtime_revisions": revisions, "priority": priority}) + if intent.digest != path.stem or document.get("digest") != intent.digest: + raise RuntimeError(f"catalog intent digest is invalid: {path}") + return intent + + +def _prepare_execution(root: Path, intent: CatalogIntent) -> tuple[Path, Path]: + execution_root = root / "runs" / intent.digest + source = execution_root / "staged" + staged_skill = source / intent.skill + revisions = dict(intent.runtime_revisions) + expected_tree = revisions.get("skill-tree") + if not expected_tree: + raise RuntimeError("catalog intent lacks full skill-tree revision") + if not staged_skill.exists(): + execution_root.mkdir(parents=True, exist_ok=True) + temporary = execution_root / f".staged.{os.getpid()}.tmp" + shutil.rmtree(temporary, ignore_errors=True) + try: + copied = temporary / intent.skill + shutil.copytree(resolve_skill_dir(intent.skill), copied) + if skill_revision(copied) != expected_tree: + raise RuntimeError(f"catalog skill tree changed while staging: {intent.skill}") + os.replace(temporary, source) + finally: + shutil.rmtree(temporary, ignore_errors=True) + if skill_revision(staged_skill) != expected_tree: + raise RuntimeError(f"catalog staged skill tree identity changed: {intent.skill}") + return source, execution_root + + +def _refuse_live_harbor(execution_root: Path) -> None: + """Do not launch probes/canaries over a surviving Harbor child for this exact intent.""" + listing = subprocess.run(["ps", "-axo", "pid=,args="], capture_output=True, text=True, + check=True).stdout + marker = str(execution_root) + live = [] + for line in listing.splitlines(): + pid, separator, command = line.strip().partition(" ") + if separator and pid.isdigit() and marker in command and "harbor run" in command: + live.append(pid) + if live: + raise RuntimeError(f"catalog intent already has live Harbor child PID {','.join(live)}") + + +def run_catalog(root: Path | str, *, max_skills: int | None = None, + stop_file: Path | str | None = None, controller_owner=None, + process_env: Mapping[str, str] | None = None) -> None: + """Run queued skills serially; native Harbor owns all within-skill parallelism.""" + root = Path(root) + root.mkdir(parents=True, exist_ok=True) + stop = Path(stop_file) if stop_file is not None else root / "STOP" + owner = controller_owner or _claim_controller(CATALOG_OWNER) + owns_controller = controller_owner is None + attempted = 0 + try: + candidates = [] + for state_path in (root / "state").glob("*.json") if (root / "state").is_dir() else (): + state = _read_json(state_path) + if state_path.stem != state.get("intent_digest"): + raise RuntimeError(f"catalog state digest is invalid: {state_path}") + if state.get("status") in {"pending", "running", "failed"}: + candidates.append((int(state.get("priority", 0)), state_path, state)) + for _, state_path, state in sorted(candidates, key=lambda item: (-item[0], item[1].name)): + if stop.exists() or (max_skills is not None and attempted >= max_skills): + break + digest = state_path.stem + intent = _load_intent(root / "intents" / f"{digest}.json", + int(state.get("priority", 0))) + current = _intent_for_skill(intent.skill, intent.target_specs, intent.harnesses, + priority=300, + global_concurrency=intent.global_concurrency, + endpoint_concurrency=intent.endpoint_concurrency, + publish_root=intent.publish_root) + if current is None: + state.update(status="failed", finished_at=time.time(), error="MissingTasks") + _atomic_json(state_path, state) + attempted += 1 + continue + if current.digest != intent.digest: + enqueue_catalog(root, [current]) + state.update(status="superseded", finished_at=time.time(), error="IdentityChanged") + _atomic_json(state_path, state) + continue + source, execution_root = _prepare_execution(root, intent) + _refuse_live_harbor(execution_root) + state.update(status="running", started_at=state.get("started_at") or time.time(), + run_root=str(execution_root)) + _atomic_json(state_path, state) + attempted += 1 + try: + targets: list[LocalTarget] = [parse_target(spec) for spec in intent.target_specs] + manifest = run_local_sweep( + intent.skill, targets, harnesses=intent.harnesses, attempts=3, + native_parallel=True, skill_source=str(source), evidence_root=execution_root, + expected_task_fingerprint=intent.task_fingerprint, + expected_runtime_revisions=dict(intent.runtime_revisions), + global_concurrency=intent.global_concurrency, + endpoint_concurrency=intent.endpoint_concurrency, + publish_root=Path(intent.publish_root), content_addressed_resume=True, + process_env=process_env) + if manifest.get("aborted"): + raise RuntimeError("sweep aborted") + if manifest.get("telemetry_pending"): + raise RuntimeError("sweep telemetry is pending verification") + except Exception as error: # noqa: BLE001 - record one failed item, continue catalog + state.update(status="failed", finished_at=time.time(), error=type(error).__name__) + else: + state.update(status="complete", finished_at=time.time(), + utilization=manifest.get("utilization"), + combinations=len(manifest.get("combinations", {}))) + _atomic_json(state_path, state) + finally: + if owns_controller: + fcntl.flock(owner.fileno(), fcntl.LOCK_UN) + owner.close() + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description="Queue restart-safe native Harbor skill sweeps") + parser.add_argument("--root", type=Path, required=True) + selection = parser.add_mutually_exclusive_group(required=True) + selection.add_argument("--skill", action="append") + selection.add_argument("--all", action="store_true") + parser.add_argument("--target", action="append", required=True, + help="repeat ALIAS=BASE_URL") + parser.add_argument("--harness", action="append", choices=tuple(HARNESS_PROTOCOLS), + default=[]) + parser.add_argument("--max-skills", type=int) + parser.add_argument("--global-concurrency", type=int, default=16) + parser.add_argument("--endpoint-concurrency", type=int, default=2) + parser.add_argument("--publish-root", type=Path, default=HARBOR_DIR, + help="absolute directory consumed by the UI for final matrices") + parser.add_argument("--enqueue-only", action="store_true") + return parser + + +def main(argv: Sequence[str] | None = None) -> int: + args = _parser().parse_args(argv) + skills = [item.name for item in load_skills()] if args.all else args.skill + harnesses = tuple(args.harness or HARNESS_PROTOCOLS) + intents = [intent for skill in skills if (intent := _intent_for_skill( + skill, args.target, harnesses, global_concurrency=args.global_concurrency, + endpoint_concurrency=args.endpoint_concurrency, + publish_root=args.publish_root)) is not None] + enqueue_catalog(args.root, intents) + if not args.enqueue_only: + run_catalog(args.root, max_skills=args.max_skills) + return 0 + + +if __name__ == "__main__": # pragma: no cover - exercised through the installed module CLI + raise SystemExit(main()) diff --git a/ingot/optimize/harbor_codex_gateway.py b/ingot/optimize/harbor_codex_gateway.py new file mode 100644 index 0000000..547d957 --- /dev/null +++ b/ingot/optimize/harbor_codex_gateway.py @@ -0,0 +1,22 @@ +"""Harbor Codex adapter variant for the local LiteLLM compatibility gateway.""" +from __future__ import annotations + +from harbor.agents.installed.codex import Codex + +from .harbor_gateway import codex_gateway_setup_command + + +class GatewayCodex(Codex): + """Use HTTP Responses without the native-only reasoning parameter.""" + # Harbor's parent Codex adapter defaults this flag to "high". LiteLLM custom_openai + # correctly rejects it for the local Qwen endpoint, while its remaining CLI flags still carry + # the normal tool-call behavior. + CLI_FLAGS = [flag for flag in Codex.CLI_FLAGS if flag.kwarg != "reasoning_effort"] + + async def run(self, instruction, environment, context): # type: ignore[no-untyped-def] + await self.exec_as_agent( + environment, + command=codex_gateway_setup_command(), + env={"CODEX_HOME": self._REMOTE_CODEX_HOME.as_posix()}, + ) + await super().run(instruction, environment, context) diff --git a/ingot/optimize/harbor_eval.py b/ingot/optimize/harbor_eval.py new file mode 100644 index 0000000..813e695 --- /dev/null +++ b/ingot/optimize/harbor_eval.py @@ -0,0 +1,1567 @@ +"""Sandboxed cross-harness skill evaluation, built on Harbor. + +`compat.py` answers "does this skill body help this *model*", using one bare completion per task. +This answers "does it help this *harness*" — the real CLI agent, with its own system prompt, tool +loop and configured model, running unrestricted inside a fresh container that Harbor provisions, +injects the agent into, and tears down. + +Harbor (https://github.com/harbor-framework/harbor, Apache-2.0) owns the parts that are not ours: +the per-task container, ~30 CLI agent adapters (claude-code, codex, pi, goose, gemini-cli, aider, +opencode, cursor-cli, openhands, …), concurrency, and trajectory capture. It also ships Terminus-2, +a neutral harness that gives any model the same shell loop — which is the only way to vary the model +without also varying the harness, since `claude` serves only Anthropic models and `codex` only +OpenAI. + +What is ours is the experiment: a skill body is a *treatment*. Every task runs twice, once with the +body injected into the harness's system prompt and once without it, and lift is the difference. The +same fixed Ingot judge grades both arms, so a harness cannot flatter itself. + +The judge runs OUTSIDE the sandbox, over artifacts the verifier copies into `/logs/verifier/`. +That keeps the judge prompt in one place instead of duplicated into every task image, and keeps the +judge's API key out of a container that is running an agent in yolo mode. + +Usage: python -m ingot.optimize.harbor_eval --agent claude-code [--agent codex] [--model M] +""" +from __future__ import annotations + +import argparse +from concurrent.futures import ThreadPoolExecutor +import contextlib +import hashlib +import json +import os +import shlex +import shutil +import stat +import subprocess +import tempfile +import threading +import time +from pathlib import Path +from typing import Any, Mapping, Sequence + +from . import resolve_skill_dir +from .ab import load_tasks +from .harbor_targets import (LocalTarget, discover_target, harbor_agent_kwargs, harbor_model, + local_agent_env, parse_target, probe_chat_tool_round_trip, + probe_protocol, protocol_for, + scrub_provider_env) +from .harbor_gateway import (GatewaySession, gateway_agent_env, gateway_agent_name, gateway_process_env, + gateway_metadata, gateway_route) +from .harbor_langfuse import EXPORTER_REVISION, export_job_attempts +from .harbor_native import (NativeCell, NativeTrialIdentity, compile_canary_job, + compile_measurement_job, + identity_env, identity_from_env, iter_attempt_dirs, native_trial_identity, + NATIVE_TRIAL_MEMORY_MB, + select_measurement_cells, write_job_config) +from .harbor_redaction import _redact_harbor_receipt_output, _redact_persisted +from .judge import judge +from ingot import paths + +HARBOR_DIR = paths.runs() / "harbor" +BUILD_DIR = HARBOR_DIR / "datasets" +HARBOR_BIN = os.environ.get("HARBOR_BIN", "harbor") +LOCAL_HARNESSES = ( + "claude-code", "terminus-2", "goose", "opencode", "openclaw", + "mini-swe-agent", "codex", "aider", "pi", +) + +# The agent works in a container, not a chat window, so the deliverable is a file it produced. This +# is the whole reason to run in a sandbox rather than judge a completion: the harness has to +# actually do the work. +SOLUTION_DIR = "/app/solution" +REPO_DIR = "/app/repo" +_INSTRUCTION_SUFFIX = f""" + +--- + +Write your complete deliverable into `{SOLUTION_DIR}/` (create the directory if it does not exist). +Anything outside that directory is discarded and will not be graded. +""" + +# A process skill has nothing to bite on in an empty container. "Verify in the execution context", +# "feed the guard the input it must reject" and "wire it to the real caller" are all unanswerable +# without existing code to read, run, and change — which is why the first task set could not +# separate any arm from any other. A task may seed a working tree; the agent edits it in place. +_SEEDED_SUFFIX = f""" + +--- + +An existing project is checked out at `{REPO_DIR}/`. Work in it directly. + +When you are done, copy every file you changed or created, plus your evidence, into +`{SOLUTION_DIR}/` (create it if needed), preserving the paths they have in the project. +Only `{SOLUTION_DIR}/` is graded. +""" + +# tmux and asciinema are here because terminus-2 installs them into the container itself when they +# are missing, and that apt-get overran the 120s exec budget on a cold cache — surfacing as a bare +# "RuntimeError: Command timed out after 120 seconds" from _install_recording_tools, with an empty +# verifier directory, counted as a broken task and dropped. Its installer skips the work entirely +# when both are already present. +# +# It is also a fairness fix. Harnesses differ in how much they install before they can start, and a +# harness whose setup is heavier was losing whole tasks for it. That is a measurement of apt, not of +# the harness. pytest is here for the same reason: the seeded tasks' own READMEs tell the agent to +# run it, and Ubuntu 24.04 refuses `pip install` under PEP 668, so every agent would otherwise spend +# its budget discovering that. +_DOCKERFILE = """FROM ubuntu:24.04 +RUN apt-get update && apt-get install -y --no-install-recommends \\ + python3 python3-pip python3-pytest git curl ca-certificates tmux asciinema \\ + && rm -rf /var/lib/apt/lists/* +WORKDIR /app +""" + +_SEEDED_DOCKERFILE = _DOCKERFILE + f"""COPY seed/ {REPO_DIR}/ +RUN find {REPO_DIR} -name '*.sh' -exec chmod +x {{}} + +""" + +# The verifier does not grade. It copies what the agent produced somewhere Harbor persists, and +# always returns 1: a real reward here would be a second, unfixed grader competing with the Ingot +# judge, and the two would disagree. +_TEST_SH = f"""#!/bin/bash +mkdir -p /logs/verifier/solution +cp -r {SOLUTION_DIR}/. /logs/verifier/solution/ 2>/dev/null || true +echo 1 > /logs/verifier/reward.txt +""" + +# An agent's own evidence log is a claim about what it ran, and a claim is exactly what a skill that +# rewards writing evidence logs teaches it to produce. `verify` runs the project's real check after +# the agent is gone and captures the result, so the judge has one outcome signal from outside the +# answer being graded. It is captured evidence, not the reward: the reward stays fixed at 1 so the +# Ingot judge remains the only grader. +_VERIFY_SH = """#!/bin/bash +mkdir -p /logs/verifier/solution +cp -r {solution}/. /logs/verifier/solution/ 2>/dev/null || true +out=/logs/verifier/solution/_objective_check.txt +{{ + echo "Ran by the harness after the agent finished, in {repo}, not by the agent:" + printf ' $ %s\\n' {quoted} + echo "---" + cd {repo} 2>/dev/null && timeout 120 bash -lc {quoted} + echo "--- exit=$?" +}} > "$out" 2>&1 +echo 1 > /logs/verifier/reward.txt +""" + + +def _task_name(skill: str, index: int) -> str: + return f"{skill}-h{index}" + + +def _write_seed(files: dict | None, seed_dir: Path) -> bool: + """Write a task's seeded working tree under the image build context. True if anything was written. + + Paths are confined to `seed_dir`: a task file is authored data, but `../..` in a key would write + outside the dataset and silently corrupt this checkout rather than the container's.""" + if not isinstance(files, dict) or not files: + return False + seed_dir.mkdir(parents=True, exist_ok=True) + root = seed_dir.resolve() + for relative, content in files.items(): + target = (seed_dir / str(relative)).resolve() + if not target.is_relative_to(root): + raise ValueError(f"seeded file path escapes the task directory: {relative!r}") + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(str(content), encoding="utf-8") + return True + + +def build_dataset(skill: str, holdout: list[dict], out_dir: Path) -> Path: + """Write a Harbor dataset with one task per held-out task. Returns the dataset directory. + + Rebuilt from scratch each time: a stale task left behind from an earlier holdout would be run + and scored as though it were part of this skill's current eval set.""" + dataset = out_dir / skill + if dataset.exists(): + shutil.rmtree(dataset) + (dataset).mkdir(parents=True) + + entries = [] + for index, task in enumerate(holdout): + name = _task_name(skill, index) + root = dataset / name + (root / "environment").mkdir(parents=True) + (root / "tests").mkdir(parents=True) + seeded = _write_seed(task.get("files"), root / "environment" / "seed") + (root / "instruction.md").write_text( + task["task"] + (_SEEDED_SUFFIX if seeded else _INSTRUCTION_SUFFIX), encoding="utf-8") + (root / "environment" / "Dockerfile").write_text( + _SEEDED_DOCKERFILE if seeded else _DOCKERFILE, encoding="utf-8") + test_sh = root / "tests" / "test.sh" + verify = str(task.get("verify") or "").strip() + test_sh.write_text( + _VERIFY_SH.format(solution=SOLUTION_DIR, repo=REPO_DIR, quoted=shlex.quote(verify)) + if seeded and verify else _TEST_SH, encoding="utf-8") + test_sh.chmod(0o755) + (root / "task.toml").write_text( + 'schema_version = "1.3"\n' + "artifacts = []\n\n" + "[task]\n" + f'name = "ingot/{name}"\n' + f'description = "held-out eval task {index} for skill {skill}"\n' + "authors = []\n" + "keywords = []\n\n" + "[metadata]\n\n" + "[verifier]\n" + "timeout_sec = 300.0\n" + "collect = []\n\n" + "[verifier.env]\n\n" + "[agent]\n" + "timeout_sec = 900.0\n\n" + "[environment]\n" + # The agent needs the network to reach its own model provider. Containment here is the + # container, not the network: that is what makes yolo mode acceptable. + 'network_mode = "public"\n' + "build_timeout_sec = 600.0\n" + 'os = "linux"\n' + "mcp_servers = []\n\n" + "[environment.env]\n\n" + "[solution.env]\n", + encoding="utf-8") + entries.append(f'[[tasks]]\nname = "ingot/{name}"\n') + + (dataset / "dataset.toml").write_text( + "[dataset]\n" + f'name = "ingot/{skill}"\n' + f'description = "Ingot held-out eval tasks for skill {skill}"\n' + "authors = []\n" + "keywords = []\n\n" + "\n".join(entries), encoding="utf-8") + return dataset + + +def stage_skill(skill: str) -> Path: + """A directory holding exactly one `/SKILL.md`, for Harbor's Agent Skills loader. + + Not the skill's parent directory: the vault holds every other skill beside it, and handing the + loader that whole tree would put 70-odd unrelated skills in front of the agent. The two arms + have to differ by exactly one skill or lift measures the library, not the skill.""" + staged = BUILD_DIR / "staged" / skill + if staged.exists(): + shutil.rmtree(staged) + staged.mkdir(parents=True) + shutil.copytree(resolve_skill_dir(skill), staged / skill) + return staged + + +# Harnesses that are a CLI with its own subscription login, and the flag that makes Harbor use it. +# Harbor's adapters default to the API key and only take the subscription when told: claude-code +# keeps ANTHROPIC_API_KEY unless CLAUDE_FORCE_OAUTH is set (and prefers the key when both are +# present), codex keeps OPENAI_API_KEY unless CODEX_FORCE_AUTH_JSON is. Every other harness here is +# a generic model-caller with no CLI to harness, so an API key is inherent to running it at all. +SUBSCRIPTION_HARNESSES = { + "claude-code": ("CLAUDE_FORCE_OAUTH", "ANTHROPIC_API_KEY", "claude setup-token"), + "codex": ("CODEX_FORCE_AUTH_JSON", "OPENAI_API_KEY", "codex login"), +} +ALLOW_API_BILLING = "HARBOR_ALLOW_API_BILLING" + +# Headroom for the image build and the agent install, both of which a seeded dataset makes heavier. +# Overridable because the right value depends on the host's network and how cold its build cache is. +BUILD_TIMEOUT_MULTIPLIER = float(os.environ.get("HARBOR_BUILD_TIMEOUT_MULTIPLIER", "4")) +SETUP_TIMEOUT_MULTIPLIER = float(os.environ.get("HARBOR_SETUP_TIMEOUT_MULTIPLIER", "3")) + + +def billing_refusals(agents: list[str]) -> list[str]: + """Harnesses about to bill per token when a subscription login was available. + + Fail-closed on purpose. The default is silent and expensive: a whole grid ran on metered API + keys with both subscription credentials sitting unused on the same host, and nothing in the + output said so — the per-arm dollar figure Harbor prints is a computed estimate and reads the + same either way. Set HARBOR_ALLOW_API_BILLING=1 to opt in deliberately.""" + if os.environ.get(ALLOW_API_BILLING, "").strip().lower() in ("1", "true", "yes"): + return [] + refusals = [] + for entry in agents: + harness = entry.partition("@")[0] + pair = SUBSCRIPTION_HARNESSES.get(harness) + if not pair: + continue + flag, key, how = pair + if os.environ.get(flag, "").strip() or not os.environ.get(key, "").strip(): + continue + refusals.append(f"{entry}: would run on {key} (metered) rather than its subscription. " + f"Set {flag}=1 after `{how}`.") + return refusals + + +def _without_langfuse_env(values: Mapping[str, str] | None) -> dict[str, str]: + return {key: value for key, value in (values or {}).items() if not key.startswith("LANGFUSE_")} + + +def _write_json_atomic(path: Path, payload: Mapping[str, object]) -> None: + """Avoid exposing a partially written receipt or canary manifest to an interrupted reader.""" + path.parent.mkdir(parents=True, exist_ok=True) + with tempfile.NamedTemporaryFile("w", encoding="utf-8", dir=path.parent, + prefix=f".{path.name}.", suffix=".tmp", delete=False) as handle: + json.dump(payload, handle, indent=2) + handle.flush() + os.fsync(handle.fileno()) + temporary = Path(handle.name) + os.replace(temporary, path) + + +def _write_harbor_invocation_receipt(jobs_dir: Path, job_name: str, done: subprocess.CompletedProcess, + parent: Mapping[str, str]) -> None: + """Persist only non-sensitive Harbor boundary facts, including zero exits before a trial.""" + job = jobs_dir / job_name + # Harbor normally creates this directory. An early CLI failure has no job state, so create + # only this expected receipt location rather than fabricating a trial or measurement. + _write_json_atomic(job / "harbor-invocation.json", { + "returncode": done.returncode, + # Arbitrary Harbor output can contain a provider header, argv, or raw endpoint in forms a + # redactor cannot soundly enumerate. Counts prove the process boundary without persisting + # any of that material. + "stdout_bytes": len(str(done.stdout or "").encode("utf-8")), + "stderr_bytes": len(str(done.stderr or "").encode("utf-8")), + "stdout_excerpt": _redact_harbor_receipt_output(done.stdout, parent), + "stderr_excerpt": _redact_harbor_receipt_output(done.stderr, parent), + }) + + +def run_arm(dataset: Path, agent: str, skill_source: str | None, jobs_dir: Path, job_name: str, + model: str | None = None, concurrency: int = 2, attempts: int = 1, *, + agent_env: Mapping[str, str] | None = None, + agent_kwargs: Mapping[str, str] | None = None, + task_name: str | None = None, + process_env: Mapping[str, str] | None = None, + log=print) -> Path: + """Run every task in the dataset through one harness, with or without the skill. Returns job dir. + + The treatment is Harbor's own `--skill`, which implements the Agent Skills spec: the skill + directory is mounted into the environment and the harness discovers `SKILL.md` itself. That is + how a skill actually reaches an agent in production, and it applies identically to every + adapter — pasting the body into a system prompt for one harness and a recipe for another would + make the comparison measure the injection channel as much as the skill. + + `skill_source` is a local path or a git source (`org/name[@ref]`), so the benchmark can be + pointed at exactly the bytes the canonical vault publishes. + """ + argv = [HARBOR_BIN, "run", "--path", str(dataset), "--agent", agent, + "--n-concurrent", str(concurrency), "--jobs-dir", str(jobs_dir), + "--job-name", job_name] + if attempts > 1: + argv += ["--n-attempts", str(attempts)] + # Seeded tasks each build their own image, because their seed differs. Before seeding, every + # task in a dataset shared one identical Dockerfile and so one cached image built once; now + # there are as many builds as tasks, and `apt-get update && install` on an uncached image + # overran the 120s compose budget. Observed as `RuntimeError: Command timed out after 120 + # seconds` with an empty verifier directory and `docker inspect returned 1` in the trial log — + # a build failure that looks nothing like one. + argv += ["--environment-build-timeout-multiplier", str(BUILD_TIMEOUT_MULTIPLIER), + "--agent-setup-timeout-multiplier", str(SETUP_TIMEOUT_MULTIPLIER)] + if skill_source: + argv += ["--skill", skill_source] + if model: + argv += ["--model", model] + # Harbor forwards these repeated options to the adapter. Sort keys so an identical local + # target produces identical command evidence regardless of mapping insertion order. + agent_env = _without_langfuse_env(agent_env) + for key in sorted(agent_env): + argv += ["--ae", f"{key}={agent_env[key]}"] + for key in sorted(agent_kwargs or {}): + value = agent_kwargs[key] + rendered = value if isinstance(value, str) else json.dumps(value, separators=(",", ":")) + argv += ["--ak", f"{key}={rendered}"] + if task_name: + # Harbor filters local datasets by the task directory basename, not dataset.toml's + # namespaced task label. The latter looks right but matches no local task. + argv += ["--include-task-name", task_name] + log(f"[harbor] {agent:<14} running {dataset.name} ({job_name} arm)") + run_kwargs = {"capture_output": True, "text": True} + if process_env is None: + # Legacy/nonlocal runs need inherited provider auth, but Langfuse remains parent-only. + run_kwargs["env"] = _without_langfuse_env(os.environ) + else: + # Some Harbor adapters read their routing settings before they construct the + # container command. Preserve only the explicit, local adapter settings here; + # inherited provider credentials remain stripped at this process boundary. + run_kwargs["env"] = _without_langfuse_env(scrub_provider_env(process_env)) + run_kwargs["env"].update(agent_env) + done = subprocess.run(argv, **run_kwargs) + _write_harbor_invocation_receipt(jobs_dir, job_name, done, + process_env if process_env is not None else os.environ) + if done.returncode != 0: + raise RuntimeError(f"harbor run failed for {agent}: {done.stderr.strip()[-600:]}") + job = jobs_dir / job_name + _refuse_broken_job(job, agent, job_name) + return job + + +def watch_native_job(job: Path, expected: Mapping[NativeTrialIdentity, Mapping[str, int]], on_ready, + *, released: set[NativeTrialIdentity] | None = None + ) -> set[NativeTrialIdentity]: + """Release exact identities only after their complete terminal attempt set is persisted.""" + released = set(released or ()) + if not job.is_dir(): + return released + for identity, required in expected.items(): + observed = {task: 0 for task in required} + for attempt in iter_attempt_dirs(job, identity=identity): + try: + record = json.loads((attempt / "result.json").read_text()) + except (OSError, ValueError): + continue + if not isinstance(record, dict): + continue + if not record.get("finished_at"): + continue + task = str(record.get("task_name") or "").split("/")[-1].split("__")[0] + if task not in observed: + raise RuntimeError(f"{identity.combination_id} {identity.arm} wrote unexpected task {task}") + observed[task] += 1 + if any(observed[task] > count for task, count in required.items()): + raise RuntimeError(f"{identity.combination_id} {identity.arm} exceeded expected attempts") + if observed == dict(required) and identity not in released: + if on_ready(identity) is not False: + released.add(identity) + return released + + +def _process_start_token(pid: int) -> str | None: + """Bind an owner receipt to one process lifetime, not a reusable PID.""" + proc_stat = Path(f"/proc/{pid}/stat") + try: + return proc_stat.read_text().split()[21] + except (OSError, IndexError): + done = subprocess.run(["ps", "-o", "lstart=", "-p", str(pid)], capture_output=True, + text=True) + token = done.stdout.strip() + return token or None + + +def _claim_native_owner(path: Path, config: Path) -> None: + owner = {"pid": os.getpid(), "start_token": _process_start_token(os.getpid()), + "config": str(config)} + while True: + try: + descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + except FileExistsError: + try: + existing = json.loads(path.read_text()) + pid = existing.get("pid") + token = existing.get("start_token") + except (OSError, ValueError, AttributeError): + raise RuntimeError("native Harbor owner receipt is unreadable") + if (isinstance(pid, int) and isinstance(token, str) + and _process_start_token(pid) == token): + raise RuntimeError(f"native Harbor job already has live owner PID {pid}") + path.unlink() + continue + with os.fdopen(descriptor, "w") as handle: + json.dump(owner, handle) + handle.flush() + os.fsync(handle.fileno()) + return + + +def run_native_job(config: Path, jobs_dir: Path, job_name: str, + expected: Mapping[NativeTrialIdentity, Mapping[str, int]], *, on_ready, + process_env: Mapping[str, str], poll_seconds: float = 0.5, + allow_completed_reuse: bool = False) -> Path: + """Run one Harbor config and publish complete identity slices while siblings continue.""" + jobs_dir.mkdir(parents=True, exist_ok=True) + job = jobs_dir / job_name + log_path = jobs_dir / f"{job_name}.harbor.log" + owner_path = jobs_dir / f"{job_name}.owner.json" + state_path = jobs_dir / f"{job_name}.released.json" + _claim_native_owner(owner_path, config) + argv = [HARBOR_BIN, "run", "--config", str(config), + "--override-memory-mb", str(NATIVE_TRIAL_MEMORY_MB), "--job-name", job_name] + env = _without_langfuse_env(scrub_provider_env(process_env)) + extra_compose = env.pop("HARBOR_EXTRA_DOCKER_COMPOSE", None) + if extra_compose: + overlay = Path(extra_compose) + if not overlay.is_file(): + raise RuntimeError("Harbor Docker Compose overlay is missing") + argv.extend(["--extra-docker-compose", str(overlay)]) + # Aider checks provider presence in the Harbor parent before building its container command. + # These are local sentinels; endpoint URLs remain isolated in each agent configuration. + env.update({"OPENAI_API_KEY": "local", "ANTHROPIC_API_KEY": "local", + "CODEX_API_KEY": "local"}) + process = None + try: + released: set[NativeTrialIdentity] = set() + if allow_completed_reuse: + if state_path.is_file(): + state = json.loads(state_path.read_text()) + released = {identity_from_env(item) for item in state.get("released", [])} + # Harbor 0.20 redacts credential-shaped agent env values in persisted TrialConfigs. A + # second `harbor run` then compares those placeholders with the resolved plan and + # rejects an otherwise identical completed job. Released state is not enough on its + # own: require the complete terminal artifact set before skipping the subprocess. + terminal = watch_native_job(job, expected, lambda _identity: True) + if released == terminal == set(expected): + return job + if terminal == set(expected): + released = watch_native_job(job, expected, on_ready, released=released) + _write_json_atomic(state_path, {"released": [identity_env(item) + for item in sorted(released, key=repr)]}) + if released == terminal: + return job + raise RuntimeError("native Harbor finalization is pending") + with log_path.open("a", encoding="utf-8") as output: + process = subprocess.Popen(argv, stdout=output, stderr=subprocess.STDOUT, env=env, + text=True) + try: + if not allow_completed_reuse and state_path.is_file(): + state = json.loads(state_path.read_text()) + released = {identity_from_env(item) for item in state.get("released", [])} + while process.poll() is None: + before = set(released) + released = watch_native_job(job, expected, on_ready, released=released) + if released != before: + _write_json_atomic(state_path, {"released": [identity_env(item) + for item in sorted(released, key=repr)]}) + time.sleep(poll_seconds) + returncode = process.wait() + released = watch_native_job(job, expected, on_ready, released=released) + _write_json_atomic(state_path, {"released": [identity_env(item) + for item in sorted(released, key=repr)]}) + finally: + if process.poll() is None: + process.terminate() + process.wait() + finally: + owner_path.unlink(missing_ok=True) + if returncode != 0: + raise RuntimeError(f"native Harbor job exited {returncode}; see {log_path}") + missing = set(expected) - released + if missing: + raise RuntimeError(f"native Harbor job ended before {len(missing)} identity slice(s) completed") + return job + + +def _refuse_broken_job(job: Path, agent: str, arm: str) -> None: + """Fail an arm only when nothing in it ran. + + `harbor run` exits 0 even when trials error, and a crashed trial leaves an empty solution + directory that `score` reads as a legitimate 0.0. Observed live: a control arm whose four trials + were all killed during `docker compose up` scored 0.000 against a skill arm's 0.750 and reported + `lift +0.750` — fabricated, and exactly the failure `compat.py` already guards against. + + Failing the whole arm on *any* broken trial is the opposite mistake: a single transient + container failure then discards three good trials and the paid-for opposite arm. Individual + broken tasks are dropped instead, by `broken_tasks`, from both arms at once.""" + result = job / "result.json" + if not result.is_file(): + raise RuntimeError(f"{agent} {arm} arm wrote no result.json at {job}") + stats = (json.loads(result.read_text()) or {}).get("stats") or {} + ran = (stats.get("n_completed_trials", 0) or 0) + broken = (stats.get("n_errored_trials", 0) or 0) + (stats.get("n_cancelled_trials", 0) or 0) + if ran and broken >= ran: + raise RuntimeError(f"{agent} {arm} arm had every one of its {ran} trial(s) error or " + f"cancel; refusing to score it") + + +def _trial_outcomes(job: Path, identity: NativeTrialIdentity | None = None) -> list[tuple[str, str, bool]]: + """(trial directory name, task name, ok) for every trial in this arm.""" + out = [] + results = ([attempt / "result.json" for attempt in iter_attempt_dirs(job, identity=identity)] + if identity is not None else job.glob("*/result.json")) + for result in results: + try: + record = json.loads(result.read_text()) or {} + except (OSError, ValueError): + continue + # Harbor nests this as exception_info.exception_type, and names the task in `task_name` + # (as "ingot/"). Reading a top-level `exception_type` finds nothing, which made this + # guard a silent no-op: opencode's skill arm lost two tasks to AgentSetupTimeoutError and + # they were scored as two 0.0s, turning an install timeout into "lift -0.375". + failure = ((record.get("exception_info") or {}).get("exception_type") or "").strip() + name = str(record.get("task_name") or result.parent.name).split("/")[-1] + out.append((result.parent.name, name.split("__")[0], not failure)) + return out + + +def broken_trials(job: Path, identity: NativeTrialIdentity | None = None) -> set[str]: + """Trial directory names that errored or were cancelled. + + With more than one attempt per task these have to be excluded individually. A crashed attempt + leaves an empty solution directory, and an empty directory is scored as a real zero — so one + flaky attempt out of three would pull the task's mean down by a third and read as the skill + performing worse.""" + return {trial for trial, _, ok in _trial_outcomes(job, identity) if not ok} + + +def broken_tasks(job: Path, identity: NativeTrialIdentity | None = None) -> set[str]: + """Task names with no surviving attempt in this arm. + + A task that crashed in one arm has to be dropped from *both*, or the arms are scored on + different task sets and the difference between them stops being lift. But with several attempts + per task, dropping the task because one attempt broke discards the attempts that did run — and + they are the whole reason for paying for repeats.""" + outcomes = _trial_outcomes(job, identity) + survivors = {task for _, task, ok in outcomes if ok} + return {task for _, task, _ in outcomes} - survivors + + +def collect_answers(job_dir: Path, skip_trials: set[str] | None = None, *, + identity: NativeTrialIdentity | None = None) -> dict[str, list[str]]: + """The text each task's agent left in the solution directory, keyed by task name. + + `skip_trials` drops individual crashed attempts, whose workspaces are empty through no fault of + the agent and would otherwise be averaged in as zeros. + + A list per task, not a string: with `--n-attempts` above 1 a task has several trials, and + keying a single answer by task name silently kept only whichever was read last — throwing away + exactly the repeated measurements that were paid for to average the agent's own variance out. + + A task that produced nothing maps to "" rather than being dropped: an empty workspace is a real + result (the harness ran and delivered nothing), and silently omitting it would raise the arm's + mean by removing its own failures.""" + skip_trials = skip_trials or set() + answers: dict[str, list[str]] = {} + solutions = ([attempt / "verifier" / "solution" + for attempt in iter_attempt_dirs(job_dir, identity=identity)] + if identity is not None else sorted(job_dir.rglob("verifier/solution"))) + for solution in solutions: + if identity is not None: + verifier = solution.parent + try: + verifier_info = verifier.lstat() + solution_info = solution.lstat() + except OSError: + continue + if (not stat.S_ISDIR(verifier_info.st_mode) + or not stat.S_ISDIR(solution_info.st_mode)): + raise ValueError("native Harbor solution directory is not a real directory") + elif not solution.is_dir(): + continue + if solution.parent.parent.name in skip_trials: + continue + name = _trial_task_name(solution) + parts = [] + for path in sorted(p for p in solution.rglob("*") if p.is_file()): + if identity is not None: + relative = path.relative_to(solution) + current = solution + for part in relative.parts: + current = current / part + if stat.S_ISLNK(current.lstat().st_mode): + raise ValueError("native Harbor solution contains a symlink") + if not _is_deliverable(path.relative_to(solution)): + continue + try: + text = path.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError): + continue # a binary the agent happened to leave behind is not the deliverable + parts.append(f"--- {path.relative_to(solution)} ---\n{text}") + answers.setdefault(name, []).append("\n\n".join(parts)[:60000]) + return answers + + +# Build leavings, not deliverables. The first real container run wrote __pycache__/*.pyc beside +# solution.py, and those bytes went into the text handed to the judge — noise the judge pays for +# and can be misled by. +_IGNORED_DIRS = {"__pycache__", ".git", "node_modules", ".venv", ".pytest_cache", ".mypy_cache"} + + +def _is_deliverable(relative: Path) -> bool: + return not set(relative.parts) & _IGNORED_DIRS + + +def _trial_task_name(solution: Path) -> str: + """The task name for a `/verifier/solution` directory. + + Harbor names the trial `__` (observed: `probe-h0__suyygRM`) so repeated + attempts at one task cannot collide. The suffix has to come off, or no trial ever matches the + task it came from and every score silently reads as a zero.""" + return solution.parent.parent.name.split("__")[0] + + +def score(answers: dict[str, list[str]], skill: str, holdout: list[dict], + skip: set[str] | None = None, concurrency: int = 1) -> list[float]: + """Judge each held-out task's collected artifacts with the fixed Ingot judge. + + A task's score is the mean over its attempts. Measured directly on this eval: re-judging one + fixed answer three times returned an identical 0.278 every time, while re-running the same + agent on the same task under the same model moved the score from 0.278 to 0.556. The variance + is the agent's, not the judge's, so the remedy is repeated attempts rather than a better grader. + + An arm that delivered nothing for *every* task is refused rather than scored. A trial can + "complete" while its agent never worked: the verifier always reports success, so an agent that + died on its first API call still counts as a completed trial with an empty workspace. Observed + live: aider v0.86.2 sends `temperature`, claude-sonnet-5 rejects it as deprecated, and the arm + came back completed-and-empty — which would have scored a clean 0.000 and read as "aider is + terrible at this skill" rather than "aider never ran". Some tasks empty is a real failure and + still scores zero; all tasks empty is a broken combination.""" + skip = skip or set() + kept = [i for i in range(len(holdout)) if _task_name(skill, i) not in skip] + if not kept: + raise RuntimeError("every task crashed in one arm or the other; nothing comparable is left") + if not any(any(answers.get(_task_name(skill, i)) or []) for i in kept): + raise RuntimeError(f"every task returned an empty workspace; the harness produced no " + f"deliverable at all, refusing to score it as zeros") + if not isinstance(concurrency, int) or isinstance(concurrency, bool) or concurrency < 1: + raise ValueError("score concurrency must be a positive integer") + graded: dict[int, list[float]] = {index: [] for index in kept} + jobs = [] + for index in kept: + task = holdout[index] + attempts = answers.get(_task_name(skill, index)) or [""] + for answer in attempts: + if not answer: + graded[index].append(0.0) # ran, produced nothing: a real zero + continue + jobs.append((index, task, answer)) + + def grade(item) -> tuple[int, float]: + index, task, answer = item + # The task's own checklist, not the judge's generic four. Dropping it here is what made + # the first build-loop matrix unreadable: controls piled up at 0.849. + value = judge(task["task"], task["rubric"], answer, + check=task.get("check"), deliverable=task.get("deliverable"), + checklist=task.get("checklist"))["score"] + return index, value + + if concurrency == 1 or len(jobs) < 2: + results = map(grade, jobs) + for index, value in results: + graded[index].append(value) + else: + with ThreadPoolExecutor(max_workers=min(concurrency, len(jobs))) as pool: + for index, value in pool.map(grade, jobs): + graded[index].append(value) + return [sum(graded[index]) / len(graded[index]) for index in kept] + + +# Changes to how persisted Harbor artifacts are interpreted must change this identifier. Rescore +# uses it to refuse evidence made under different scoring semantics rather than mixing the rows. +SCORING_REVISION = "harbor-rubric-v2-agy" + + +def _task_fingerprint(holdout: list[dict]) -> str: + """Stable identity of the exact held-out task set, independent of dict insertion order.""" + canonical = json.dumps(holdout, sort_keys=True, separators=(",", ":"), ensure_ascii=False) + return hashlib.sha256(canonical.encode()).hexdigest() + + +def _combination_id(harness: str, target: LocalTarget) -> str: + """Stable local evidence identity, including the endpoint fingerprint through its job slug.""" + return f"{harness}@{target.served_model}--{target.job_slug}" + + +def _combination_job_slug(harness: str, target: LocalTarget) -> str: + """Keep the evidence identity exact without putting Docker-unsafe model IDs in bind paths.""" + identity = _combination_id(harness, target) + return identity if ":" not in identity else f"{harness}@{target.job_slug}" + + +def _native_full_job_name(cell: NativeCell) -> str: + """Stable one-cell Harbor boundary; changing the rest of a matrix never changes its config.""" + return f"native-full--{cell.harness}--{cell.target.job_slug}" + + +def _round_robin_endpoints(cells: Sequence[NativeCell]) -> list[NativeCell]: + """Keep the next bounded cell jobs on different physical endpoint identities.""" + buckets: dict[str, list[NativeCell]] = {} + for cell in cells: + buckets.setdefault(cell.target.fingerprint, []).append(cell) + ordered = [] + while any(buckets.values()): + for bucket in buckets.values(): + if bucket: + ordered.append(bucket.pop(0)) + return ordered + + +def _canary_artifact(job: Path, task_name: str, + identity: NativeTrialIdentity | None = None) -> str | None: + """Return a diagnostic when the one-task canary did not yield usable Harbor evidence.""" + if identity is not None: + attempts = list(iter_attempt_dirs(job, identity=identity)) + if len(attempts) != 1: + return "canary wrote no completed trial result" + result = attempts[0] / "result.json" + try: + record = json.loads(result.read_text()) or {} + except (OSError, ValueError): + return "canary trial result was unreadable" + recorded_task = str(record.get("task_name") or "").split("/")[-1].split("__")[0] + exception = ((record.get("exception_info") or {}).get("exception_type") or "").strip() + if recorded_task != task_name or exception: + return "canary trial did not complete without an exception" + exception_path = result.parent / "exception.txt" + if exception_path.is_file() and exception_path.read_bytes().strip(): + return "canary trial wrote exception evidence" + answers = collect_answers(job, identity=identity) + if not any(answer.strip() for values in answers.values() for answer in values): + return "canary produced no nonempty verifier solution artifact" + return None + try: + summary = json.loads((job / "result.json").read_text()) or {} + completed = ((summary.get("stats") or {}).get("n_completed_trials", 0) or 0) + except (OSError, ValueError): + return "canary wrote no completed trial result" + if not isinstance(completed, int) or isinstance(completed, bool) or completed < 1: + return "canary wrote no completed trial" + trials = list(job.glob("*/result.json")) + if not trials: + return "canary wrote no completed trial result" + for result in trials: + try: + record = json.loads(result.read_text()) or {} + except (OSError, ValueError): + return "canary trial result was unreadable" + recorded_task = str(record.get("task_name") or "").split("/")[-1].split("__")[0] + exception = ((record.get("exception_info") or {}).get("exception_type") or "").strip() + if recorded_task != task_name or exception: + return "canary trial did not complete without an exception" + exception_path = result.parent / "exception.txt" + if exception_path.is_file() and exception_path.read_bytes().strip(): + return "canary trial wrote exception evidence" + solution = result.parent / "verifier" / "solution" + if not solution.is_dir() or not any(path.is_file() and path.stat().st_size > 0 + for path in solution.rglob("*")): + return "canary produced no nonempty verifier solution artifact" + return None + return "canary wrote no matching held-out trial" + + +def run_canary(skill: str, dataset: Path, holdout: list[dict], source: str, harness: str, + target: LocalTarget, canary_root: Path, *, exploratory: bool = False, log=print) -> dict: + """Run the first held-out task once and retain the diagnostic evidence for this seam.""" + task_name = _task_name(skill, 0) + jobs_dir = canary_root / skill / target.job_slug + route = gateway_route(target, harness) + # Preserve failed native/gateway evidence. A translation revision changes the gateway model + # but Harbor's fixed harness job name otherwise reopens the prior one-task job. + job_name = f"{harness}--{route.identity}" if route else harness + model = route.model if route else harbor_model(target, harness) + record = {"combination": _combination_id(harness, target), "harness": harness, + "model": model, "target_alias": target.alias, + "endpoint_fingerprint": target.fingerprint, "protocol": protocol_for(harness), + "job": str(jobs_dir / job_name), "family": target.family, + "parameter_billions": target.parameter_billions, + "quantization": target.quantization, "tool_parser": target.tool_parser, + "exploratory": exploratory, "rankable": not exploratory} + if route: + record.update(gateway_metadata(route)) + try: + job = run_arm( + dataset, gateway_agent_name(route) if route else harness, source, jobs_dir, job_name, model=model, concurrency=1, + attempts=1, + agent_env=_without_langfuse_env( + gateway_agent_env(target, route) if route else local_agent_env(target, harness)), + agent_kwargs={} if route else harbor_agent_kwargs(target, harness), task_name=task_name, + process_env=gateway_process_env(os.environ) if route and route.harness == "codex" else os.environ, log=log, + ) + except Exception as error: # noqa: BLE001 - retain failed seam evidence and stop before full arms + record["error"] = f"{type(error).__name__}: {error}"[:400] + return record + try: + telemetry_metadata = {key: value for key, value in record.items() if key != "job"} + telemetry_metadata.update(_telemetry_provenance(skill, holdout, source)) + _write_json_atomic(job / "combo.json", telemetry_metadata) + export_job_attempts(job, {**telemetry_metadata, "arm": "canary"}) + except Exception as error: # noqa: BLE001 - measurement survives telemetry repair work + record["telemetry_error"] = _redact_harbor_receipt_output( + f"{type(error).__name__}: {error}", os.environ) + try: + if diagnostic := _canary_artifact(job, task_name): + record["error"] = diagnostic + else: + record["ok"] = True + except Exception as error: # noqa: BLE001 - retain failed seam evidence and stop before full arms + record["error"] = f"{type(error).__name__}: {error}"[:400] + return record + + +def _run_native_canaries(skill: str, dataset: Path, targets: Sequence[LocalTarget], + harnesses: Sequence[str], holdout: list[dict], source: str, + canary_root: Path, *, global_limit: int, endpoint_limit: int, + allow_completed_reuse: bool = False, + process_env: Mapping[str, str] = os.environ, log=print) -> dict: + """Run every skill-specific model×harness canary through one bounded Harbor job.""" + cells = [NativeCell(target, harness) for target in targets for harness in harnesses] + jobs_root = canary_root / skill + job_name = "native-canaries" + config = compile_canary_job( + dataset, _task_name(skill, 0), cells, Path(source), jobs_root, + global_limit=global_limit, + endpoint_limits={cell.target.fingerprint: endpoint_limit for cell in cells}) + config_path = jobs_root / "native-canaries.config.json" + write_job_config(config_path, config) + job = jobs_root / job_name + records = {} + provenance = _telemetry_provenance(skill, holdout, source) + expected = {} + for cell in cells: + identity = native_trial_identity(cell.target, cell.harness, "canary") + expected[identity] = {_task_name(skill, 0): 1} + route = gateway_route(cell.target, cell.harness) + record = {**_combo_metadata(holdout, cell.harness, cell.target, 1), + "model": route.model if route else harbor_model(cell.target, cell.harness), + "job": str(job), "exploratory": False, "rankable": True} + if route: + record.update(gateway_metadata(route)) + record["gateway_revision"] = identity.gateway_revision + records[cell.combination_id] = record + + def on_ready(identity: NativeTrialIdentity) -> bool: + record = records[identity.combination_id] + metadata = {key: value for key, value in record.items() if key != "job"} + metadata.update(provenance) + try: + export_job_attempts(job, {**metadata, "arm": "canary"}, identity=identity) + except Exception as error: # noqa: BLE001 - retain canary evidence + record["telemetry_error"] = _redact_harbor_receipt_output( + f"{type(error).__name__}: {error}", os.environ) + record["error"] = "canary telemetry receipt was not verified" + return False + record.pop("telemetry_error", None) + record.pop("error", None) + diagnostic = _canary_artifact(job, _task_name(skill, 0), identity) + if diagnostic: + record["error"] = diagnostic + else: + record["ok"] = True + return True + + process_env = gateway_process_env(process_env) if any( + cell.harness == "codex" and gateway_route(cell.target, cell.harness) for cell in cells + ) else process_env + run_native_job(config_path, jobs_root, job_name, expected, on_ready=on_ready, + process_env=process_env, allow_completed_reuse=allow_completed_reuse) + # A prior controller may have persisted terminal/released trials before it returned the + # manifest. Re-finalize those exact identities from disk; exporter receipts are idempotent. + for identity in expected: + record = records[identity.combination_id] + if "ok" not in record and "error" not in record: + on_ready(identity) + return records + + +def _telemetry_provenance(skill: str, holdout: list[dict], source: str) -> dict: + skill_file = Path(source) / skill / "SKILL.md" + if not skill_file.is_file(): + skill_file = Path(source) / "SKILL.md" + skill_bytes = skill_file.read_bytes() + skill_body = skill_bytes.decode("utf-8") + return { + "skill": skill, + "skill_body": skill_body, + "skill_sha256": hashlib.sha256(skill_bytes).hexdigest(), + "task_texts": {_task_name(skill, index): _redact_harbor_receipt_output( + str(task.get("task") or ""), {}) + for index, task in enumerate(holdout)}, + } + + +def _combo_metadata(holdout: list[dict], harness: str, target: LocalTarget, + attempts: int, exploratory: bool = False) -> dict: + metadata = { + "combination": _combination_id(harness, target), + "harness": harness, + "model": target.served_model, + "target_alias": target.alias, + "endpoint_fingerprint": target.fingerprint, + "protocol": protocol_for(harness), + "task_fingerprint": _task_fingerprint(holdout), + "attempts": attempts, + "family": target.family, + "parameter_billions": target.parameter_billions, + "quantization": target.quantization, + "tool_parser": target.tool_parser, + "exploratory": exploratory, + "rankable": not exploratory, + } + if route := gateway_route(target, harness): + metadata.update(gateway_metadata(route)) + return metadata + + +def _run_local_full_arms(skill: str, dataset: Path, targets: Sequence[LocalTarget], + harnesses: Sequence[str], holdout: list[dict], source: str, + attempts: int, concurrency: int, + jobs_root: Path, manifest: dict, canaries: dict | None = None, + exploratory: bool = False, log=print) -> None: + """Run full arms for passed canaries and persist failed seams as unmeasured rows.""" + for target in targets: + for harness in harnesses: + route = gateway_route(target, harness) + metadata = _combo_metadata(holdout, harness, target, attempts, exploratory) + key = _combination_id(harness, target) + jobs_dir = jobs_root / _combination_job_slug(harness, target) + canary = (canaries or {}).get(key, {}) + if "error" in canary: + error = _redact_harbor_receipt_output(str(canary["error"]), os.environ)[:400] + jobs_dir.mkdir(parents=True, exist_ok=True) + _write_json_atomic(jobs_dir / "combo.json", {**metadata, "canary_error": error}) + manifest["combinations"][key] = {**metadata, "error": error} + log(f"[harbor] {key:<54} UNMEASURED: {error[:300]}") + continue + try: + jobs_dir.mkdir(parents=True, exist_ok=True) + jobs = {} + routing = { + "model": route.model if route else harbor_model(target, harness), + "concurrency": concurrency, "attempts": attempts, + "agent_env": _without_langfuse_env( + gateway_agent_env(target, route) if route else local_agent_env(target, harness)), + "agent_kwargs": {} if route else harbor_agent_kwargs(target, harness), + "process_env": (gateway_process_env(os.environ) + if route and route.harness == "codex" else os.environ), "log": log, + } + telemetry_errors = {} + telemetry_ready: bool | None = None + for arm in ("skill", "control"): + jobs[arm] = run_arm(dataset, gateway_agent_name(route) if route else harness, + source if arm == "skill" else None, + jobs_dir, arm, **routing) + if telemetry_ready is None: + try: + metadata.update(_telemetry_provenance(skill, holdout, source)) + _write_json_atomic(jobs_dir / "combo.json", metadata) + telemetry_ready = True + except Exception as error: # noqa: BLE001 - retain paid-for arms + telemetry_ready = False + telemetry_errors["provenance"] = _redact_harbor_receipt_output( + f"{type(error).__name__}: {error}", os.environ) + if telemetry_ready: + try: + export_job_attempts(jobs[arm], {**metadata, "arm": arm}) + except Exception as error: # noqa: BLE001 - publication gates later + telemetry_errors[arm] = _redact_harbor_receipt_output( + f"{type(error).__name__}: {error}", os.environ) + skipped = broken_tasks(jobs["skill"]) | broken_tasks(jobs["control"]) + manifest["combinations"][key] = { + **metadata, "raw_evidence": True, + "skill_job": str(jobs["skill"]), "control_job": str(jobs["control"]), + "tasks_dropped": sorted(skipped), + } + if telemetry_errors: + manifest["combinations"][key]["telemetry_errors"] = telemetry_errors + except Exception as error: # noqa: BLE001 - other combinations remain useful evidence + manifest["combinations"][key] = { + **metadata, "error": f"{type(error).__name__}: {error}"[:400], + } + log(f"[harbor] {key:<54} UNAVAILABLE: {str(error)[:300]}") + + +def _run_native_full_arms(skill: str, dataset: Path, targets: Sequence[LocalTarget], + harnesses: Sequence[str], holdout: list[dict], source: str, + jobs_root: Path, manifest: dict, + canaries: Mapping[str, Mapping[str, object]], *, + global_limit: int, endpoint_limit: int, + publish_root: Path = HARBOR_DIR, + allow_completed_reuse: bool = False, + process_env: Mapping[str, str] = os.environ, log=print) -> None: + """Run approved cells in bounded independent Harbor jobs and publish each complete pair.""" + from .harbor_rescore import current_scoring_identity, rescore + + cells = [NativeCell(target, harness) for target in targets for harness in harnesses] + selected, unmeasured = select_measurement_cells(cells, canaries) + manifest["combinations"].update(unmeasured) + provenance = _telemetry_provenance(skill, holdout, source) + unmeasured_paths = [] + for cell in cells: + failed = unmeasured.get(cell.combination_id) + if failed is None: + continue + combo = jobs_root / _combination_job_slug(cell.harness, cell.target) + _write_json_atomic(combo / "combo.json", { + **_combo_metadata(holdout, cell.harness, cell.target, 3), + **provenance, + "canary_error": failed["error"], + }) + unmeasured_paths.append(combo) + if not selected: + return + scoring = current_scoring_identity() + combos = {} + agent_identity = { + "skill_sha256": provenance["skill_sha256"], + "task_fingerprint": _task_fingerprint(holdout), + "attempts": 3, + "exporter_revision": EXPORTER_REVISION, + "cells": sorted((cell.combination_id, + native_trial_identity(cell.target, cell.harness, "skill").gateway_revision) + for cell in selected), + } + pipeline_path = jobs_root / "native-full.pipeline.json" + pipeline = {"agent_identity": agent_identity, "scoring_identity": scoring, + "exported": {}, "graded": [], "published": []} + if pipeline_path.is_file(): + saved = json.loads(pipeline_path.read_text()) + if isinstance(saved, dict): + if saved.get("agent_identity") == agent_identity: + pipeline["exported"] = saved.get("exported", {}) + if saved.get("scoring_identity") == scoring: + pipeline["graded"] = saved.get("graded", []) + pipeline["published"] = saved.get("published", []) + ready = {key: set(value) for key, value in pipeline["exported"].items()} + failed_this_run: set[tuple[str, str]] = set() + unmeasured_pending = bool(unmeasured_paths) + prepared = [] + for cell in _round_robin_endpoints(selected): + combo = jobs_root / _combination_job_slug(cell.harness, cell.target) + job_name = _native_full_job_name(cell) + identities = {} + expected = {} + for arm in ("skill", "control"): + identity = native_trial_identity(cell.target, cell.harness, arm) + identities[arm] = identity + expected[identity] = {_task_name(skill, index): 3 for index in range(len(holdout))} + prepared.append((cell, combo, job_name, identities, expected)) + + legacy_job_name = "native-full" + legacy_expected = {identity: required for _cell, _combo, _job, _identities, expected in prepared + for identity, required in expected.items()} + legacy_terminal = watch_native_job( + jobs_root / legacy_job_name, legacy_expected, lambda _identity: True) + adopted: list[NativeTrialIdentity] = [] + for cell, combo, job_name, identities, expected in prepared: + try: + prior = json.loads((combo / "combo.json").read_text()) + except (OSError, ValueError): + prior = {} + prior = prior if isinstance(prior, dict) else {} + metadata = {**prior, **_combo_metadata(holdout, cell.harness, cell.target, 3), + **provenance, "native_identities": { + arm: identity_env(identity) for arm, identity in identities.items()}} + metadata["gateway_revision"] = identities["skill"].gateway_revision + native_jobs = dict(metadata.get("native_jobs") or {}) + source_jobs = {} + missing = {} + for arm, identity in identities.items(): + if identity in legacy_terminal: + source_jobs[arm] = legacy_job_name + native_jobs[arm] = legacy_job_name + adopted.append(identity) + else: + source_jobs[arm] = job_name + missing[identity] = expected[identity] + if native_jobs: + metadata["native_jobs"] = native_jobs + _write_json_atomic(combo / "combo.json", metadata) + config_path = None + if missing: + config = compile_measurement_job( + dataset, [_task_name(skill, index) for index in range(len(holdout))], [cell], + Path(source), jobs_root, attempts=3, global_limit=endpoint_limit, + endpoint_limits={cell.target.fingerprint: endpoint_limit}, + arms=tuple(identity.arm for identity in missing)) + config_path = jobs_root / f"{job_name}.config.json" + write_job_config(config_path, config) + combos[cell.combination_id] = ( + combo, metadata, identities, job_name, config_path, missing, source_jobs) + + def has_recovered_lift(identity: NativeTrialIdentity, rows: Mapping[str, Any]) -> bool: + return any( + isinstance(row, dict) and row.get("combination") == identity.combination_id + and row.get("endpoint_fingerprint") == identity.endpoint_fingerprint + and row.get("skill_sha256") == agent_identity["skill_sha256"] + and row.get("task_fingerprint") == agent_identity["task_fingerprint"] + and row.get("attempts") == agent_identity["attempts"] + and row.get("harness") == identity.harness + and row.get("protocol") == identity.protocol + and row.get("gateway_revision", "direct") == identity.gateway_revision + and all(row.get(key) == value for key, value in scoring.items()) + and isinstance(row.get("lift"), (int, float)) + and not isinstance(row.get("lift"), bool) + for row in rows.values() + ) + + matrix_output = publish_root / f"{skill}.rescored.json" + if unmeasured_pending and matrix_output.is_file(): + try: + existing = json.loads(matrix_output.read_text()) + except (OSError, ValueError): + existing = {} + rows = existing.get("combinations", {}) if isinstance(existing, dict) else {} + if any(has_recovered_lift(identities["skill"], rows) + for _combo, _metadata, identities, _job, _config, _expected, _sources + in combos.values()): + rescore(skill, jobs_roots=[jobs_root], combination_paths=unmeasured_paths, + output=matrix_output, scoring_identity=scoring, log=log) + unmeasured_pending = False + + state_lock = threading.Lock() + rescore_lock = threading.Lock() + combo_locks = {key: threading.Lock() for key in combos} + + def on_ready(identity: NativeTrialIdentity) -> None: + nonlocal unmeasured_pending + with combo_locks[identity.combination_id]: + combo, metadata, _identities, job_name, _config, _expected, source_jobs = combos[ + identity.combination_id] + native_jobs = metadata.setdefault("native_jobs", {}) + native_jobs[identity.arm] = source_jobs[identity.arm] + if set(native_jobs) >= {"skill", "control"}: + metadata.pop("native_job", None) + _write_json_atomic(combo / "combo.json", metadata) + with state_lock: + arms = ready.setdefault(identity.combination_id, set()) + if identity.arm not in arms: + stage = (identity.combination_id, f"export:{identity.arm}") + with state_lock: + if stage in failed_this_run: + return False + try: + export_job_attempts(jobs_root / source_jobs[identity.arm], + {**metadata, "arm": identity.arm}, + identity=identity) + except Exception as error: # noqa: BLE001 - retry on a later controller run + with state_lock: + failed_this_run.add(stage) + manifest["combinations"].setdefault(identity.combination_id, {}).update( + telemetry_error=f"{type(error).__name__}: {error}"[:400]) + return False + with state_lock: + arms.add(identity.arm) + pipeline["exported"][identity.combination_id] = sorted(arms) + _write_json_atomic(pipeline_path, pipeline) + if arms == {"skill", "control"}: + with state_lock: + needs_grade = identity.combination_id not in pipeline["graded"] + if needs_grade: + stage = (identity.combination_id, "grade") + with state_lock: + if stage in failed_this_run: + return False + try: + with rescore_lock: + existing = (json.loads(matrix_output.read_text()) + if matrix_output.is_file() else {}) + rows = (existing.get("combinations", {}) + if isinstance(existing, dict) else {}) + recovered = has_recovered_lift(identity, rows) + with state_lock: + include_unmeasured = unmeasured_pending + if not recovered or include_unmeasured: + selected_paths = ([*unmeasured_paths, combo] + if include_unmeasured else [combo]) + rescore(skill, jobs_roots=[jobs_root], + combination_paths=selected_paths, + output=matrix_output, scoring_identity=scoring, log=log) + with state_lock: + unmeasured_pending = False + except Exception as error: # noqa: BLE001 - retry later without rerunning agents + with state_lock: + failed_this_run.add(stage) + manifest["combinations"].setdefault(identity.combination_id, {}).update( + scoring_error=f"{type(error).__name__}: {error}"[:400]) + return False + with state_lock: + pipeline["graded"].append(identity.combination_id) + _write_json_atomic(pipeline_path, pipeline) + with state_lock: + needs_publish = identity.combination_id not in pipeline["published"] + if needs_publish: + try: + with state_lock: + manifest["combinations"][identity.combination_id] = { + **metadata, "raw_evidence": True, + "native_jobs": {arm: str(jobs_root / source_job) + for arm, source_job in source_jobs.items()}} + _write_json_atomic(jobs_root / "progress.json", manifest) + except Exception as error: # noqa: BLE001 - grade receipt prevents repeated billing + with state_lock: + manifest["combinations"].setdefault(identity.combination_id, {}).update( + publication_error=f"{type(error).__name__}: {error}"[:400]) + return False + with state_lock: + pipeline["published"].append(identity.combination_id) + _write_json_atomic(pipeline_path, pipeline) + return True + + process_env = gateway_process_env(process_env) if any( + cell.harness == "codex" and gateway_route(cell.target, cell.harness) for cell in selected + ) else process_env + endpoint_locks = {cell.target.fingerprint: threading.Lock() for cell in selected} + + for identity in adopted: + if on_ready(identity) is False: + raise RuntimeError("legacy native Harbor finalization is pending") + + def run_cell(item) -> None: + _combo, _metadata, identities, job_name, config_path, expected, _sources = item + if not expected: + return + assert config_path is not None + fingerprint = identities["skill"].endpoint_fingerprint + with endpoint_locks[fingerprint]: + run_native_job(config_path, jobs_root, job_name, expected, on_ready=on_ready, + process_env=process_env, allow_completed_reuse=allow_completed_reuse) + + # Each one-cell Harbor job can consume at most endpoint_limit slots. Bound the number of live + # jobs so their aggregate cannot exceed the caller's global limit. Harbor then schedules only + # that cell's 24 trials, producing publishable evidence before later cells finish. + workers = max(1, min(len(combos), global_limit // endpoint_limit)) + with ThreadPoolExecutor(max_workers=workers) as pool: + list(pool.map(run_cell, combos.values())) + + +def run_local_sweep(skill: str, targets: list[LocalTarget], *, + harnesses: Sequence[str] = LOCAL_HARNESSES, concurrency: int = 2, + attempts: int = 3, skill_source: str | None = None, canary_only: bool = False, + exploratory: bool = False, native_parallel: bool = False, + evidence_root: Path | None = None, + expected_task_fingerprint: str | None = None, + expected_runtime_revisions: Mapping[str, str] | None = None, + global_concurrency: int | None = None, + endpoint_concurrency: int | None = None, + publish_root: Path = HARBOR_DIR, + content_addressed_resume: bool = False, + process_env: Mapping[str, str] | None = None, log=print) -> dict: + """Evaluate every local target/harness pair whose routing canary passes. + + This deliberately returns raw-evidence manifest only. `harbor_rescore` owns publication of + the visible matrix, so a half-finished local sweep cannot replace a known-good matrix. + """ + if attempts != 3 and not (exploratory and attempts == 1): + raise ValueError("local Harbor sweeps require 3 attempts, or 1 with exploratory=True") + if not targets: + raise ValueError("local Harbor sweep needs at least one target") + harnesses = tuple(harnesses) + for harness in harnesses: + protocol_for(harness) + _, holdout, _ = load_tasks(skill) + if not holdout: + raise SystemExit(f"'{skill}' has no held-out eval tasks to run.") + task_fingerprint = _task_fingerprint(holdout) + if expected_task_fingerprint is not None and task_fingerprint != expected_task_fingerprint: + raise RuntimeError(f"{skill} held-out tasks changed after catalog enqueue") + if not shutil.which(HARBOR_BIN): + raise SystemExit(f"'{HARBOR_BIN}' is not on PATH; install with `uv tool install harbor`.") + source = skill_source or str(stage_skill(skill)) + dataset = build_dataset(skill, holdout, BUILD_DIR) + manifest = {"skill": skill, "tasks": len(holdout), "attempts": attempts, + "canary_only": canary_only, "canaries": {}, "combinations": {}, + "aborted": False, "exploratory": exploratory, "rankable": not exploratory} + run_root = evidence_root or HARBOR_DIR + canary_root = run_root / ("canaries-k1" if exploratory else "canaries") + manifest_path = canary_root / skill / "manifest.json" + global_limit = global_concurrency if global_concurrency is not None else max(16, concurrency) + endpoint_limit = endpoint_concurrency if endpoint_concurrency is not None else max(1, concurrency) + native_process_env = process_env if process_env is not None else os.environ + if (not isinstance(global_limit, int) or isinstance(global_limit, bool) or global_limit < 1 + or not isinstance(endpoint_limit, int) or isinstance(endpoint_limit, bool) + or endpoint_limit < 1 or endpoint_limit > global_limit): + raise ValueError("native concurrency limits must be positive and endpoint <= global") + + def finish() -> dict: + _write_json_atomic(manifest_path, _redact_persisted(manifest, os.environ)) + return manifest + + # Re-discover even targets supplied through the Python API. CLI callers already do this while + # parsing `--target`, but the sweep is also a public orchestration interface and must not trust + # a hand-constructed LocalTarget to have passed the `/v1/models` identity/context preflight. + # No container trial or full job directory exists yet. + try: + targets = [discover_target(target.alias, target.base_url) for target in targets] + if expected_runtime_revisions is not None: + for target in targets: + for harness in harnesses: + identity = native_trial_identity(target, harness, "skill") + key = f"route:{target.fingerprint}:{harness}" + observed = f"{identity.protocol}/{identity.gateway_revision}/context={target.context_length}" + if expected_runtime_revisions.get(key) != observed: + raise RuntimeError(f"{key} changed after catalog enqueue") + # Probe every required adapter protocol for every target before spending one container + # trial. A local endpoint is part of the treatment identity; provider fallback is forbidden. + required_protocols = sorted({protocol_for(harness) for harness in harnesses}) + for target in targets: + for protocol in required_protocols: + probe_protocol(target, protocol) + if "chat" in required_protocols: + probe_chat_tool_round_trip(target) + except Exception as error: # noqa: BLE001 - endpoint preflight is diagnostic, not a measurement + manifest["aborted"] = True + manifest["preflight_error"] = f"{type(error).__name__}: {error}"[:400] + return finish() + + routes = [(route, target) for target in targets for harness in harnesses + if (route := gateway_route(target, harness)) is not None] + try: + gateway_context = (GatewaySession(routes, canary_root / "gateway" / skill) + if routes else contextlib.nullcontext()) + with gateway_context: + if native_parallel and attempts == 3 and not exploratory: + manifest["canaries"].update(_run_native_canaries( + skill, dataset, targets, harnesses, holdout, source, canary_root, + global_limit=global_limit, endpoint_limit=endpoint_limit, + allow_completed_reuse=content_addressed_resume, + process_env=native_process_env, log=log)) + if any("telemetry_error" in record for record in manifest["canaries"].values()): + manifest["telemetry_pending"] = True + else: + for target in targets: + for harness in harnesses: + key = _combination_id(harness, target) + manifest["canaries"][key] = run_canary( + skill, dataset, holdout, source, harness, target, canary_root, + exploratory=exploratory, log=log) + if canary_only: + return finish() + jobs_root = run_root / "jobs" / f"{skill}-k{attempts}" + if native_parallel and attempts == 3 and not exploratory: + _run_native_full_arms( + skill, dataset, targets, harnesses, holdout, source, jobs_root, manifest, + manifest["canaries"], global_limit=global_limit, + endpoint_limit=endpoint_limit, publish_root=publish_root, + allow_completed_reuse=content_addressed_resume, + process_env=native_process_env, log=log) + else: + _run_local_full_arms(skill, dataset, targets, harnesses, holdout, source, + attempts, concurrency, jobs_root, manifest, + canaries=manifest["canaries"], exploratory=exploratory, log=log) + except Exception as error: # noqa: BLE001 - fail before Harbor if the fixed gateway is stale/unreachable + manifest["aborted"] = True + manifest["gateway_error"] = f"{type(error).__name__}: {error}"[:400] + return finish() + return finish() + + +def run_harbor_eval(skill: str, agents: list[str], model: str | None = None, + concurrency: int = 2, skill_source: str | None = None, attempts: int = 1, + log=print) -> dict: + """Skill-vs-control lift for one skill across several harnesses. Writes runs/harbor/.json. + + `skill_source` overrides where the skill is read from — a git source pins the benchmark to the + bytes the canonical vault publishes rather than whatever this checkout happens to hold. + + `attempts` runs each task that many times per arm and averages. One attempt per task is not + enough to see an effect this size: two control-arm runs of an identical configuration moved a + task's score by 0.278 and swapped the ranking of two harnesses, which is larger than any lift + the first grid reported.""" + _, holdout, _ = load_tasks(skill) + if not holdout: + raise SystemExit(f"'{skill}' has no held-out eval tasks to run.") + source = skill_source or str(stage_skill(skill)) + if not shutil.which(HARBOR_BIN): + raise SystemExit(f"'{HARBOR_BIN}' is not on PATH; install with `uv tool install harbor`.") + # Before anything is spent, not per-row after: a grid is hours long and the bill is already run + # up by the time a row would report it. + if refusals := billing_refusals(agents): + raise SystemExit("[harbor] refusing to start; these would bill per token:\n " + + "\n ".join(refusals) + + f"\nOr set {ALLOW_API_BILLING}=1 to accept metered billing.") + dataset = build_dataset(skill, holdout, BUILD_DIR) + # Runs at different attempt counts keep separate roots. Harbor refuses a job directory whose + # config has changed ("cannot be resumed with a different config"), so re-running an existing + # skill at a new -k failed every combination before a single container started; and the earlier + # run's trials are the raw evidence a rescore reads, so overwriting them is worse than the + # collision. Both runs now sit side by side. + jobs_root = HARBOR_DIR / "jobs" / (skill if attempts == 1 else f"{skill}-k{attempts}") + log(f"[harbor] '{skill}': {len(holdout)} held-out tasks × {len(agents)} harness(es), " + f"two arms each, {attempts} attempt(s) per task → {jobs_root}") + + rows: dict[str, dict] = {} + for entry in agents: + # "agent" or "agent@model": the question is which *combination* serves a skill best, so a + # row is one harness paired with one model, not a harness alone. + agent, _, combo_model = entry.partition("@") + row_model = combo_model or model + # One broken harness must not discard the arms already paid for. + try: + jobs_dir = jobs_root / entry.replace("/", "_") + # The directory name has had its slashes flattened, so "openai/gpt-5.5" and a model + # genuinely named "openai_gpt-5.5" are indistinguishable once written. Rescoring reads + # only these directories, and a co-occurrence grid that cannot say which model a row + # used is not a co-occurrence grid. Record the pair beside the arms. + jobs_dir.mkdir(parents=True, exist_ok=True) + (jobs_dir / "combo.json").write_text(json.dumps( + {"combination": entry, "harness": agent, "model": row_model or "harness default"})) + jobs, answers = {}, {} + for arm in ("skill", "control"): + jobs[arm] = run_arm(dataset, agent, source if arm == "skill" else None, + jobs_dir, arm, row_model, concurrency, attempts, log) + answers[arm] = collect_answers(jobs[arm], broken_trials(jobs[arm])) + # A task that crashed in either arm is dropped from both, so the two means are always + # over the same tasks. Otherwise the difference between them is not lift. + skipped = broken_tasks(jobs["skill"]) | broken_tasks(jobs["control"]) + arms = {arm: score(answers[arm], skill, holdout, skipped) for arm in jobs} + s_mean = sum(arms["skill"]) / len(arms["skill"]) + c_mean = sum(arms["control"]) / len(arms["control"]) + verdict = ("helps" if s_mean - c_mean > 0.05 + else "no lift" if s_mean - c_mean >= -0.05 else "HURTS") + note = f" [{len(skipped)} task(s) dropped]" if skipped else "" + log(f"[harbor] {entry:<38} skill {s_mean:.3f} control {c_mean:.3f} " + f"lift {s_mean - c_mean:+.3f} ({verdict}){note}") + rows[entry] = {"skill_mean": s_mean, "control_mean": c_mean, "lift": s_mean - c_mean, + "skill_scores": arms["skill"], "control_scores": arms["control"], + "harness": agent, "model": row_model or "harness default", + # Not cosmetic: a lift over 2 tasks and one over 4 are different claims, + # and a reader comparing rows has to be able to see which is which. + "tasks_scored": len(arms["skill"]), "tasks_dropped": sorted(skipped), + "attempts": attempts} + except Exception as error: # noqa: BLE001 - any harness failure is one unusable row + rows[entry] = {"error": f"{type(error).__name__}: {error}"[:400], "harness": agent, + "model": row_model or "harness default"} + # The message, not just the type: when every row fails there is no matrix to read the + # detail out of, and a log saying only "RuntimeError" cannot be diagnosed at all. + log(f"[harbor] {entry:<38} UNAVAILABLE: {str(error)[:300]}") + if not any("error" not in row for row in rows.values()): + raise SystemExit(f"[harbor] no harness could be run for '{skill}'; nothing was measured.") + + summary = {"skill": skill, "tasks": len(holdout), "pinned_model": model, + "judge": os.environ.get("JUDGE_MODELS") or os.environ.get("JUDGE_MODEL", ""), + "harnesses": rows} + HARBOR_DIR.mkdir(parents=True, exist_ok=True) + path = HARBOR_DIR / f"{skill}.json" + path.write_text(json.dumps(summary, indent=2)) + log(f"[harbor] matrix written to {path}") + return summary + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description="Sandboxed cross-harness skill evaluation.") + parser.add_argument("skill") + parser.add_argument("--agent", action="append", default=None, + help="harness to run (repeatable); default claude-code") + parser.add_argument("--model", default=None, help="pin the model where the harness allows it") + parser.add_argument("--target", action="append", default=None, metavar="ALIAS=URL", + help="local allowlisted endpoint (repeatable); enables local model sweep") + parser.add_argument("-n", "--concurrent", type=int, default=2) + parser.add_argument("--global-concurrency", type=int, + help="native Harbor global trial cap (default max(16, --concurrent))") + parser.add_argument("--endpoint-concurrency", type=int, + help="native Harbor per-endpoint cap (default --concurrent)") + parser.add_argument("-k", "--attempts", type=int, default=None, + help="attempts per task per arm, averaged; 1 is below this eval's noise") + parser.add_argument("--canary-only", action="store_true", + help="run endpoint preflights and one-task routing canaries without full arms") + parser.add_argument("--exploratory", action="store_true", + help="allow a one-attempt local sweep that is explicitly not rankable") + return parser + + +def main(argv: Sequence[str] | None = None) -> int: + parser = build_parser() + args = parser.parse_args(argv) + if args.canary_only and not args.target: + parser.error("--canary-only requires --target") + if args.exploratory and not args.target: + parser.error("--exploratory requires --target") + if args.exploratory and args.attempts != 1: + parser.error("--exploratory requires --attempts 1") + if args.target: + if args.model is not None: + parser.error("--target cannot be combined with --model; targets pin their served model") + attempts = 3 if args.attempts is None else args.attempts + if attempts == 1 and not args.exploratory: + parser.error("--attempts 1 requires --exploratory") + if attempts not in (1, 3): + parser.error("--target requires --attempts 3, or 1 with --exploratory") + targets = [] + for spec in args.target: + provisional = parse_target(spec) + # parse_target enforces the allowlist and canonical URL before discovery makes a request. + targets.append(discover_target(provisional.alias, provisional.base_url)) + manifest = run_local_sweep(args.skill, targets, + harnesses=args.agent or list(LOCAL_HARNESSES), + concurrency=args.concurrent, attempts=attempts, + global_concurrency=args.global_concurrency, + endpoint_concurrency=args.endpoint_concurrency, + canary_only=args.canary_only, + exploratory=args.exploratory, native_parallel=True, log=print) + return 1 if manifest.get("aborted") else 0 + run_harbor_eval(args.skill, args.agent or ["claude-code"], model=args.model, + concurrency=args.concurrent, attempts=args.attempts or 1, log=print) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/ingot/optimize/harbor_gateway.py b/ingot/optimize/harbor_gateway.py new file mode 100644 index 0000000..c0d82fd --- /dev/null +++ b/ingot/optimize/harbor_gateway.py @@ -0,0 +1,380 @@ +"""Narrow, runner-owned LiteLLM compatibility gateway for observed role rejections. + +The gateway is deliberately not a general provider proxy. It exists only while Harbor evaluates +the three local harness/target combinations whose native endpoints rejected system/developer +roles. Its model list has no fallback deployment or provider credential. +""" +from __future__ import annotations + +import hashlib +import json +import os +import socket +import subprocess +import time +import urllib.error +import urllib.request +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Callable, Sequence + +from .harbor_targets import LocalTarget, scrub_provider_env + + +# Docker's bridge gateway is reachable from Harbor's per-trial bridge networks but not published +# outside the Dell host. A fixed address/port makes it possible to health-check before Harbor. +GATEWAY_HOST = "172.17.0.1" +GATEWAY_PORT = 4865 +GATEWAY_URL = f"http://{GATEWAY_HOST}:{GATEWAY_PORT}" +GATEWAY_REVISION = "litellm-1.93-role-user-v1" +DELL_CLAUDE_OUTPUT_CAP_REVISION = "litellm-1.93-role-user-output-v4" +DELL_CODEX_HTTP_REVISION = "litellm-1.93-role-user-codex-http-catalog-v8" +# This is an ignored Dell-only environment, created from requirements-harbor-gateway.txt. Harbor +# and Ingot's primary environment deliberately remain untouched. +_LITELLM_BIN = str(Path(__file__).resolve().parents[2] / ".venv-harbor-gateway/bin/litellm") +_CLAUDE_GATEWAY_TARGETS = frozenset({"dell-qwen", "spark-deepseek"}) + + +@dataclass(frozen=True) +class GatewayRoute: + harness: str + target_alias: str + model: str + served_model: str + upstream_env: str + output_limit: int | None = None + revision: str = GATEWAY_REVISION + + @property + def identity(self) -> str: + payload = json.dumps( + {"harness": self.harness, "target": self.target_alias, "model": self.model, + "served_model": self.served_model, "output_limit": self.output_limit, + "revision": self.revision}, sort_keys=True, separators=(",", ":") + ).encode() + return hashlib.sha256(payload).hexdigest()[:12] + + +def gateway_route(target: LocalTarget, harness: str) -> GatewayRoute | None: + """Return a route only for the three observed local role-protocol failures.""" + if ((harness == "claude-code" and target.alias in _CLAUDE_GATEWAY_TARGETS) + or (harness == "codex" and target.alias == "dell-qwen")): + # Put the translation revision in the gateway's served model name. A surviving process + # from an old run then fails the model-list health check instead of silently reusing + # evidence made under a different role conversion. + output_limit = None + revision = GATEWAY_REVISION + if harness == "claude-code" and target.alias == "dell-qwen": + # A 1/4 cap left a 24,577-token trajectory one token beyond this target's 32,768 + # window. Reserve seven eighths for the accumulated prompt and tool history instead. + output_limit = target.context_length // 8 + revision = f"{DELL_CLAUDE_OUTPUT_CAP_REVISION}-{output_limit}" + elif harness == "codex": + revision = DELL_CODEX_HTTP_REVISION + revision_hash = hashlib.sha256(revision.encode()).hexdigest()[:8] + slug = f"harbor-compat-{target.alias}-{harness}-{revision_hash}" + return GatewayRoute(harness, target.alias, slug, target.served_model, + f"HARBOR_GATEWAY_UPSTREAM_{target.alias.upper().replace('-', '_')}", + output_limit=output_limit, revision=revision) + return None + + +def _system_user(content: Any, label: str) -> dict[str, Any]: + if isinstance(content, str): + content = f"[{label}]\n{content}" + return {"role": "user", "content": content} + + +def normalize_role_request(data: dict[str, Any], call_type: str, + output_limits: dict[str, int] | None = None) -> dict[str, Any]: + """Convert only roles rejected by the local endpoints, retaining every tool object verbatim.""" + data = dict(data) + if call_type == "anthropic_messages": + messages = list(data.get("messages") or []) + system = data.pop("system", None) + if system not in (None, "", []): + messages.insert(0, _system_user(system, "system")) + data["messages"] = [ + _system_user(message.get("content"), message.get("role")) + if isinstance(message, dict) and message.get("role") in {"system", "developer"} + else message + for message in messages + ] + elif call_type == "aresponses": + items = list(data.get("input") or []) + instructions = data.pop("instructions", None) + if instructions not in (None, "", []): + items.insert(0, _system_user( + [{"type": "input_text", "text": f"[instructions]\n{instructions}"}], "instructions")) + data["input"] = [ + {**item, "role": "user"} + if isinstance(item, dict) and item.get("role") in {"system", "developer"} + else item + for item in items + ] + # Codex attaches its model-default reasoning object even when Harbor omits the CLI flag. + # LiteLLM's Responses bridge derives the custom backend's rejected reasoning_effort from it. + data.pop("reasoning", None) + data.pop("reasoning_effort", None) + # LiteLLM 1.93's custom_openai Chat Completions route never accepts this Responses-style + # key. Every Claude Messages request uses that route; only Dell gets a max_tokens cap. + if call_type == "anthropic_messages": + data.pop("max_output_tokens", None) + output_limit = (output_limits or {}).get(str(data.get("model"))) + if output_limit is not None: + keys = ("max_tokens",) if call_type == "anthropic_messages" else ("max_tokens", "max_output_tokens") + for key in keys: + value = data.get(key) + if value is None or isinstance(value, int) and value > output_limit: + data[key] = output_limit + return data + + +def gateway_agent_env(target: LocalTarget, route: GatewayRoute) -> dict[str, str]: + """Explicit Harbor child settings for a gateway route, never a provider key.""" + env = {"ANTHROPIC_API_KEY": "local", "OPENAI_API_KEY": "local", "CODEX_API_KEY": "local"} + if route.harness == "claude-code": + env.update({"ANTHROPIC_BASE_URL": GATEWAY_URL, "ANTHROPIC_MODEL": route.model}) + elif route.harness == "codex": + env.update({"OPENAI_BASE_URL": f"{GATEWAY_URL}/v1", "OPENAI_API_BASE": f"{GATEWAY_URL}/v1", + "OPENAI_HOST": GATEWAY_URL, "HARBOR_GATEWAY_CODEX_PROVIDER": "1", + "HARBOR_GATEWAY_CODEX_MODEL": route.model, + "HARBOR_GATEWAY_CODEX_SERVED_MODEL": route.served_model, + "HARBOR_GATEWAY_CODEX_CONTEXT": str(target.context_length)}) + else: # defensive: callers must not route an unrelated harness through this service + raise ValueError(f"unsupported compatibility gateway harness: {route.harness}") + return env + + +def gateway_process_env(parent: Mapping[str, str]) -> dict[str, str]: + """Expose this checkout only to Harbor when it must import the custom Codex adapter.""" + root = str(Path(__file__).resolve().parents[2]) + inherited = parent.get("PYTHONPATH", "") + return {**parent, "PYTHONPATH": os.pathsep.join(item for item in (root, inherited) if item)} + + +def gateway_metadata(route: GatewayRoute) -> dict[str, str]: + return {"gateway_revision": route.revision, "gateway_identity": route.identity, + "gateway_agent": gateway_agent_name(route)} + + +def codex_gateway_setup_command() -> str: + """Write the HTTP provider and a truthful local catalog using Codex's own instructions.""" + return '''set -eu +test "${HARBOR_GATEWAY_CODEX_PROVIDER:-}" = 1 +test -n "${HARBOR_GATEWAY_CODEX_MODEL:-}" +test -n "${HARBOR_GATEWAY_CODEX_SERVED_MODEL:-}" +case "${HARBOR_GATEWAY_CODEX_CONTEXT:-}" in *[!0-9]*|'') exit 1;; esac +mkdir -p "$CODEX_HOME" +codex debug models --bundled >"$CODEX_HOME/bundled-models.json" +python3 <<'PY' +import json +import os +from pathlib import Path + +home = Path(os.environ["CODEX_HOME"]) +bundled = json.loads((home / "bundled-models.json").read_text()) +models = bundled.get("models") +first = models[0] if isinstance(models, list) and models else None +messages = first.get("model_messages") if isinstance(first, dict) else None +if not isinstance(messages, dict) or not messages.get("instructions_template"): + raise ValueError("Codex bundled catalog has no reusable instruction template") +context = int(os.environ["HARBOR_GATEWAY_CODEX_CONTEXT"]) +if context < 32768 or context > 9_007_199_254_740_991: + raise ValueError("invalid Codex local-model context") +model = dict(first) +model.update({ + "slug": os.environ["HARBOR_GATEWAY_CODEX_MODEL"], + "display_name": os.environ["HARBOR_GATEWAY_CODEX_SERVED_MODEL"], + "description": "Local Qwen model through the Ingot compatibility gateway", + "default_reasoning_level": None, + "supported_reasoning_levels": [], + "shell_type": "default", + "visibility": "none", + "supported_in_api": True, + "priority": 99, + "additional_speed_tiers": [], + "service_tiers": [], + "default_service_tier": None, + "availability_nux": None, + "upgrade": None, + "include_skills_usage_instructions": False, + "include_plugin_usage_instructions": False, + "include_apps_usage_instructions": False, + "supports_reasoning_summary_parameter": False, + "default_reasoning_summary": "none", + "support_verbosity": False, + "default_verbosity": None, + "apply_patch_tool_type": None, + "web_search_tool_type": "text", + "truncation_policy": {"mode": "bytes", "limit": 10000}, + "supports_parallel_tool_calls": False, + "supports_image_detail_original": False, + "context_window": context, + "max_context_window": context, + "auto_compact_token_limit": None, + "comp_hash": os.environ["HARBOR_GATEWAY_CODEX_MODEL"], + "effective_context_window_percent": 95, + "experimental_supported_tools": [], + "input_modalities": ["text"], + "supports_search_tool": False, + "use_responses_lite": False, + "auto_review_model_override": None, + "model_specialty": None, + "tool_mode": None, + "multi_agent_version": None, +}) +(home / "model-catalog.json").write_text(json.dumps({"models": [model]}, separators=(",", ":"))) +PY +cat >>"$CODEX_HOME/config.toml" < str: + return "ingot.optimize.harbor_codex_gateway:GatewayCodex" if route.harness == "codex" else route.harness + + +def _callback_source(routes: Sequence[GatewayRoute]) -> str: + output_limits = {route.model: route.output_limit for route in routes if route.output_limit is not None} + return '''from litellm.integrations.custom_logger import CustomLogger +from ingot.optimize.harbor_gateway import normalize_role_request + +OUTPUT_LIMITS = ''' + repr(output_limits) + ''' + +class GatewayRoleNormalizer(CustomLogger): + async def async_pre_call_hook(self, user_api_key_dict, cache, data, call_type): + return normalize_role_request(data, call_type, OUTPUT_LIMITS) + +gateway_role_normalizer = GatewayRoleNormalizer() +''' + + +def _gateway_config(routes: Sequence[GatewayRoute]) -> dict[str, Any]: + return { + "model_list": [ + {"model_name": route.model, + "litellm_params": {"model": f"custom_openai/{route.served_model}", + "api_base": f"os.environ/{route.upstream_env}", "api_key": "local"}} + for route in routes + ], + "router_settings": {"num_retries": 0, "allowed_fails": 0}, + "litellm_settings": {"callbacks": ["role_normalizer.gateway_role_normalizer"]}, + "general_settings": {"master_key": "local"}, + } + + +def _fixed_bind_occupied() -> bool: + """Bound the stale-listener probe; any prior listener invalidates this gateway session.""" + try: + connection = socket.create_connection((GATEWAY_HOST, GATEWAY_PORT), timeout=0.2) + except OSError: + return False + connection.close() + return True + + +class GatewaySession: + """One LiteLLM process per canary invocation; writes an untracked cleanup receipt on exit.""" + def __init__(self, routes: Sequence[tuple[GatewayRoute, LocalTarget]], runtime_dir: Path, *, + litellm_bin: str = _LITELLM_BIN, + popen: Callable[..., subprocess.Popen] = subprocess.Popen, + health_timeout: float = 20.0): + self.routes = tuple(routes) + self.runtime_dir = runtime_dir + self.litellm_bin = litellm_bin + self.popen = popen + self.health_timeout = health_timeout + self.process: subprocess.Popen | None = None + + def _write_files(self) -> Path: + self.runtime_dir.mkdir(parents=True, exist_ok=True) + (self.runtime_dir / "role_normalizer.py").write_text( + _callback_source([route for route, _ in self.routes]), encoding="utf-8") + config = self.runtime_dir / "config.json" + config.write_text(json.dumps(_gateway_config([route for route, _ in self.routes]), indent=2), encoding="utf-8") + return config + + def _healthy_models(self) -> bool: + request = urllib.request.Request( + f"{GATEWAY_URL}/v1/models", headers={"Authorization": "Bearer local"}, method="GET") + try: + with urllib.request.urlopen(request, timeout=1.0) as response: + if not 200 <= int(getattr(response, "status", 200)) < 300: + return False + payload = json.loads(response.read()) + except (urllib.error.URLError, TimeoutError, OSError, ValueError): + return False + expected = {route.model for route, _ in self.routes} + actual = {str(item.get("id")) for item in payload.get("data", []) + if isinstance(item, dict)} + return expected == actual + + def start(self) -> None: + if self.process is not None: + return + try: + if not self.routes: + raise ValueError("compatibility gateway needs at least one route") + if not Path(self.litellm_bin).is_file(): + raise RuntimeError( + "compatibility gateway runtime is missing; create .venv-harbor-gateway from " + "ingot/optimize/requirements-harbor-gateway.txt") + if _fixed_bind_occupied(): + raise RuntimeError("compatibility gateway bind is already occupied") + config = self._write_files() + env = gateway_process_env(scrub_provider_env(os.environ)) + for route, target in self.routes: + env[route.upstream_env] = f"{target.base_url}/v1" + self.process = self.popen( + [self.litellm_bin, "--config", str(config), "--host", GATEWAY_HOST, + "--port", str(GATEWAY_PORT)], + cwd=Path.cwd(), env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + ) + deadline = time.monotonic() + self.health_timeout + while time.monotonic() < deadline: + if self.process.poll() is not None: + raise RuntimeError("compatibility gateway exited before health check") + try: + with urllib.request.urlopen(f"{GATEWAY_URL}/health/liveliness", timeout=1.0) as response: + if (200 <= int(getattr(response, "status", 200)) < 300 + and self._healthy_models() and self.process.poll() is None): + return + except (urllib.error.URLError, TimeoutError, OSError): + time.sleep(0.2) + raise RuntimeError("compatibility gateway did not become healthy") + except Exception: + self.close("failed-start") + raise + + def close(self, reason: str = "completed") -> None: + if self.process is not None and self.process.poll() is None: + self.process.terminate() + try: + self.process.wait(timeout=10) + except subprocess.TimeoutExpired: + self.process.kill() + self.process.wait(timeout=5) + self.runtime_dir.mkdir(parents=True, exist_ok=True) + (self.runtime_dir / "cleanup.json").write_text(json.dumps({ + "gateway_revisions": sorted({route.revision for route, _ in self.routes}), + "routes": [route.identity for route, _ in self.routes], + "reason": reason, + "stopped": True, + }, indent=2), encoding="utf-8") + + def __enter__(self) -> "GatewaySession": + self.start() + return self + + def __exit__(self, *unused: object) -> None: + self.close() diff --git a/ingot/optimize/harbor_langfuse.py b/ingot/optimize/harbor_langfuse.py new file mode 100644 index 0000000..9c09ac8 --- /dev/null +++ b/ingot/optimize/harbor_langfuse.py @@ -0,0 +1,944 @@ +"""Export persisted Harbor attempts to Langfuse with public read-back receipts.""" +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import re +import stat +import tempfile +import time +from collections.abc import Mapping, Sequence +from pathlib import Path + +import httpx + +from .harbor_redaction import _redact_harbor_receipt_output, _redact_persisted +from .harbor_native import NativeTrialIdentity, read_trial_identity + + +EXPORTER_REVISION = "harbor-langfuse-v3" +RECEIPT_NAME = "langfuse-receipt.json" +LEGACY_V2_RECEIPT_NAME = "langfuse-receipt.v2.json" +_METADATA_FIELDS = ( + "combination", "harness", "model", "target_alias", "endpoint_fingerprint", "protocol", + "task_fingerprint", "attempts", "gateway_revision", "gateway_identity", "gateway_agent", + "arm", "canary", "skill", "skill_sha256", +) +_MAX_TEXT_BYTES = 64 * 1024 +# A retained Goose full-arm trajectory reached 3,368,055 bytes. Four MiB leaves measured headroom; +# the outbound projection still retains only allowlisted timestamps, sources, and numeric totals. +_MAX_TRAJECTORY_BYTES = 4 * 1024 * 1024 +_MAX_ARTIFACT_PATHS = 128 +_MAX_ARTIFACT_SCAN_PATHS = 4096 +_MAX_ARTIFACT_DEPTH = 16 +_MAX_ARTIFACT_BYTES = 128 * 1024 +_READBACK_ATTEMPTS = 5 +_READBACK_POLL_SECONDS = 1.0 +_TEXT_SUFFIXES = { + ".c", ".cc", ".cpp", ".css", ".csv", ".h", ".html", ".ini", ".java", ".js", + ".json", ".jsx", ".log", ".md", ".py", ".rb", ".rs", ".rst", ".sh", ".sql", + ".toml", ".ts", ".tsx", ".txt", ".xml", ".yaml", ".yml", +} +_TEXT_NAMES = {"Dockerfile", "Makefile"} +_STRUCTURAL_EXCLUSIONS = {".env", "config", "credentials", "env", "environment", "secrets"} +_SENSITIVE_ARTIFACT_STEMS = { + "config", "configuration", "credential", "credentials", "endpoint", "endpoints", + "env", "environment", "secret", "secrets", +} + + +class TelemetryReceiptError(RuntimeError): + """Persisted Harbor evidence lacks a matching, publicly readable trace receipt.""" + + +_DIRECTORY_FLAGS = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0) + + +def _same_file(first: os.stat_result, second: os.stat_result) -> bool: + return (first.st_dev, first.st_ino, stat.S_IFMT(first.st_mode)) == ( + second.st_dev, second.st_ino, stat.S_IFMT(second.st_mode)) + + +class _AttemptReader: + """Read only the fixed Harbor attempt layout through job-relative descriptors.""" + + def __init__(self, trial: Path) -> None: + self.trial = trial + self._job_fd = -1 + self._trial_fd = -1 + self._artifact_paths = 0 + self._artifact_files = 0 + self._artifact_bytes = 0 + self._artifact_omitted_files = 0 + self._artifact_omitted_hidden = 0 + self._artifact_scan_truncated = False + + def __enter__(self) -> "_AttemptReader": + try: + job_info = self.trial.parent.lstat() + if stat.S_ISLNK(job_info.st_mode) or not stat.S_ISDIR(job_info.st_mode): + raise TelemetryReceiptError(f"{self.trial.parent} is not a real job directory") + self._job_fd = os.open(self.trial.parent, _DIRECTORY_FLAGS) + if not _same_file(job_info, os.fstat(self._job_fd)): + raise TelemetryReceiptError(f"{self.trial.parent} changed during admission") + trial_info = os.stat(self.trial.name, dir_fd=self._job_fd, follow_symlinks=False) + if stat.S_ISLNK(trial_info.st_mode) or not stat.S_ISDIR(trial_info.st_mode): + raise TelemetryReceiptError(f"{self.trial} is a symlink or non-directory attempt") + self._trial_fd = os.open(self.trial.name, _DIRECTORY_FLAGS, dir_fd=self._job_fd) + if not _same_file(trial_info, os.fstat(self._trial_fd)): + raise TelemetryReceiptError(f"{self.trial} changed during admission") + return self + except (OSError, TelemetryReceiptError): + self.close() + raise + + def __exit__(self, *_args) -> None: + self.close() + + def close(self) -> None: + for descriptor in (self._trial_fd, self._job_fd): + if descriptor >= 0: + os.close(descriptor) + self._trial_fd = self._job_fd = -1 + + def _open_dir(self, parent_fd: int, name: str, *, missing_ok: bool = False) -> tuple[int, os.stat_result] | None: + try: + before = os.stat(name, dir_fd=parent_fd, follow_symlinks=False) + except FileNotFoundError: + if missing_ok: + return None + raise + if stat.S_ISLNK(before.st_mode) or not stat.S_ISDIR(before.st_mode): + raise TelemetryReceiptError(f"attempt directory {name!r} failed admission") + try: + descriptor = os.open(name, _DIRECTORY_FLAGS, dir_fd=parent_fd) + except OSError as error: + raise TelemetryReceiptError(f"attempt directory {name!r} failed admission") from error + if not _same_file(before, os.fstat(descriptor)): + os.close(descriptor) + raise TelemetryReceiptError(f"attempt directory {name!r} changed during admission") + return descriptor, before + + @staticmethod + def _verify_entry(parent_fd: int, name: str, opened: os.stat_result) -> None: + try: + current = os.stat(name, dir_fd=parent_fd, follow_symlinks=False) + except OSError as error: + raise TelemetryReceiptError(f"attempt entry {name!r} changed during admission") from error + if not _same_file(opened, current): + raise TelemetryReceiptError(f"attempt entry {name!r} changed during admission") + + def _read_at(self, parent_fd: int, name: str, *, artifact: bool = False, + limit: int = _MAX_TEXT_BYTES) -> bytes | None: + try: + before = os.stat(name, dir_fd=parent_fd, follow_symlinks=False) + except FileNotFoundError: + return None + if stat.S_ISLNK(before.st_mode) or not stat.S_ISREG(before.st_mode): + raise TelemetryReceiptError(f"attempt entry {name!r} is non-regular") + if before.st_size > limit: + raise TelemetryReceiptError(f"attempt artifact byte budget exceeded by {name!r}") + try: + descriptor = os.open( + name, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0), dir_fd=parent_fd) + except OSError as error: + raise TelemetryReceiptError(f"attempt entry {name!r} failed admission") from error + try: + opened = os.fstat(descriptor) + if not stat.S_ISREG(opened.st_mode) or not _same_file(before, opened): + raise TelemetryReceiptError(f"attempt entry {name!r} changed during admission") + chunks = [] + remaining = limit + 1 + while remaining: + chunk = os.read(descriptor, remaining) + if not chunk: + break + chunks.append(chunk) + remaining -= len(chunk) + data = b"".join(chunks) + finally: + os.close(descriptor) + self._verify_entry(parent_fd, name, before) + if len(data) > limit: + raise TelemetryReceiptError(f"attempt artifact byte budget exceeded by {name!r}") + return data + + def read(self, *parts: str, limit: int = _MAX_TEXT_BYTES) -> bytes | None: + descriptor = os.dup(self._trial_fd) + opened_dirs: list[tuple[int, str, os.stat_result]] = [] + try: + for part in parts[:-1]: + opened = self._open_dir(descriptor, part, missing_ok=True) + if opened is None: + return None + child_fd, child_info = opened + opened_dirs.append((descriptor, part, child_info)) + descriptor = child_fd + return self._read_at(descriptor, parts[-1], limit=limit) + finally: + os.close(descriptor) + for parent_fd, name, opened in reversed(opened_dirs): + self._verify_entry(parent_fd, name, opened) + os.close(parent_fd) + + def read_json(self, *parts: str, limit: int = _MAX_TEXT_BYTES) -> dict: + data = self.read(*parts, limit=limit) + if data is None: + return {} + try: + value = json.loads(data.decode("utf-8")) + except (UnicodeDecodeError, ValueError): + return {} + return value if isinstance(value, dict) else {} + + def artifacts(self, *parts: str, skip_solution: bool = False, + skill_body: bytes | None = None, + first: Sequence[str] = ()) -> dict[str, str]: + descriptor = os.dup(self._trial_fd) + opened_dirs: list[tuple[int, str, os.stat_result]] = [] + try: + for part in parts: + opened = self._open_dir(descriptor, part, missing_ok=True) + if opened is None: + return {} + child_fd, child_info = opened + opened_dirs.append((descriptor, part, child_info)) + descriptor = child_fd + found: list[tuple[str, str]] = [] + for name in first: + try: + info = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + except FileNotFoundError: + continue + self._project_artifact(descriptor, name, Path(name), info, found, skill_body) + self._walk_artifacts(descriptor, (), found, skip_solution, skill_body, depth=0, + excluded=frozenset(first)) + return dict(sorted(found)) + finally: + os.close(descriptor) + for parent_fd, name, opened in reversed(opened_dirs): + self._verify_entry(parent_fd, name, opened) + os.close(parent_fd) + + def artifact_projection(self) -> dict[str, int | bool]: + """Describe a bounded omission without turning optional telemetry into failed evidence.""" + return { + "exported_files": self._artifact_files, + "exported_bytes": self._artifact_bytes, + "omitted_files": self._artifact_omitted_files, + "omitted_hidden_paths": self._artifact_omitted_hidden, + "scan_truncated": self._artifact_scan_truncated, + } + + def _project_artifact(self, descriptor: int, name: str, path: Path, + info: os.stat_result, found: list[tuple[str, str]], + skill_body: bytes | None) -> None: + if stat.S_ISLNK(info.st_mode): + raise TelemetryReceiptError(f"attempt artifact {name!r} failed admission") + if not stat.S_ISREG(info.st_mode): + raise TelemetryReceiptError(f"attempt artifact {name!r} is non-regular") + if name == RECEIPT_NAME: + return + lowered = {part.lower() for part in path.parts} + if (lowered & _STRUCTURAL_EXCLUSIONS + or path.stem.lower() in _SENSITIVE_ARTIFACT_STEMS): + return + if path.suffix.lower() not in _TEXT_SUFFIXES and name not in _TEXT_NAMES: + return + if (info.st_size > _MAX_TEXT_BYTES + or self._artifact_files >= _MAX_ARTIFACT_PATHS + or info.st_size > _MAX_ARTIFACT_BYTES - self._artifact_bytes): + self._artifact_omitted_files += 1 + return + data = self._read_at(descriptor, name) + assert data is not None + if skill_body and skill_body in data: + self._artifact_omitted_files += 1 + return + if any(byte < 32 and byte not in (9, 10, 13) or byte == 127 for byte in data): + self._artifact_omitted_files += 1 + return + try: + value = data.decode("utf-8") + except UnicodeDecodeError: + self._artifact_omitted_files += 1 + return + found.append((path.as_posix(), _redact_harbor_receipt_output(value, {}))) + self._artifact_files += 1 + self._artifact_bytes += len(data) + + def _walk_artifacts(self, descriptor: int, relative: tuple[str, ...], + found: list[tuple[str, str]], skip_solution: bool, + skill_body: bytes | None, *, depth: int, + excluded: frozenset[str] = frozenset()) -> None: + if self._artifact_scan_truncated: + return + if depth > _MAX_ARTIFACT_DEPTH: + self._artifact_omitted_files += 1 + return + names = [] + with os.scandir(descriptor) as entries: + for entry in entries: + if self._artifact_paths >= _MAX_ARTIFACT_SCAN_PATHS: + self._artifact_scan_truncated = True + break + self._artifact_paths += 1 + # Harbor and harnesses may retain caches/backups beside the submitted solution. + # They are never evidence. Skip hidden entries at every depth before stat or + # traversal, so even a hidden symlink cannot redirect the exporter. + if entry.name.startswith("."): + self._artifact_omitted_hidden += 1 + continue + names.append(entry.name) + for name in sorted(names, key=lambda item: (item != "answer.md", item)): + if self._artifact_scan_truncated: + return + if depth == 0 and name in excluded: + continue + path_parts = (*relative, name) + info = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + if stat.S_ISLNK(info.st_mode): + raise TelemetryReceiptError(f"attempt artifact {name!r} failed admission") + if stat.S_ISDIR(info.st_mode): + if skip_solution and path_parts == ("solution",): + continue + opened = self._open_dir(descriptor, name) + assert opened is not None + child_fd, child_info = opened + try: + self._walk_artifacts(child_fd, path_parts, found, skip_solution, skill_body, + depth=depth + 1) + finally: + os.close(child_fd) + self._verify_entry(descriptor, name, child_info) + continue + self._project_artifact(descriptor, name, Path(*path_parts), info, found, skill_body) + + +def _path_json(path: Path, limit: int = _MAX_TEXT_BYTES) -> dict: + try: + info = path.lstat() + if stat.S_ISLNK(info.st_mode) or not stat.S_ISREG(info.st_mode) or info.st_size > limit: + return {} + descriptor = os.open(path, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0)) + try: + data = os.read(descriptor, limit + 1) + finally: + os.close(descriptor) + value = json.loads(data.decode("utf-8")) + except (OSError, UnicodeDecodeError, ValueError): + return {} + return value if isinstance(value, dict) else {} + + +def load_provenance_metadata(path: Path) -> dict: + """Load one explicit bounded migration document used at export and publication.""" + metadata = _path_json(path) + if not metadata: + raise TelemetryReceiptError(f"{path} has no readable provenance metadata") + return metadata + + +def _clean_string(value: object) -> str: + # Canonical payload identity cannot depend on which credentials happen to be active now. + return _redact_harbor_receipt_output(value, {}) + + +def _metadata(metadata: Mapping[str, object]) -> dict: + clean = {} + for key in _METADATA_FIELDS: + value = metadata.get(key) + if isinstance(value, str): + clean[key] = _clean_string(value) + elif isinstance(value, (int, float, bool)) or value is None: + if key in metadata: + clean[key] = value + return clean + + +def _trajectory(reader: _AttemptReader) -> dict: + record = reader.read_json( + "agent", "trajectory.json", limit=_MAX_TRAJECTORY_BYTES) + if not record: + return {} + result = {"steps": []} + raw_steps = record.get("steps", []) + if not isinstance(raw_steps, list): + raise TelemetryReceiptError("trajectory steps must be a list") + for raw in raw_steps: + if not isinstance(raw, Mapping): + continue + step = {} + if raw.get("source") in ("user", "agent"): + step["source"] = raw["source"] + timestamp = raw.get("timestamp") + if isinstance(timestamp, str) and re.fullmatch( + r"\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d(?:\.\d+)?Z", timestamp): + step["timestamp"] = timestamp + result["steps"].append(step) + if isinstance(record.get("final_metrics"), Mapping): + final_metrics = record["final_metrics"] + result["final_metrics"] = {} + for key in ( + "total_prompt_tokens", "total_cached_tokens", + "total_completion_tokens", "total_cost_usd", + ): + value = final_metrics.get(key) + if isinstance(value, (int, float)) and not isinstance(value, bool): + result["final_metrics"][key] = value + return result + + +def _exception_evidence(reader: _AttemptReader, + result: Mapping[str, object]) -> tuple[str | None, str | None]: + info = result.get("exception_info") + if isinstance(info, Mapping) and info.get("exception_type"): + detail = _clean_string(info.get("exception_message")) or None + return _clean_string(info["exception_type"]), detail + data = reader.read("exception.txt") + if data: + try: + exception = _redact_harbor_receipt_output(data.decode("utf-8"), {}) + except UnicodeDecodeError: + return None, None + matches = re.findall(r"(?:^|\n)([A-Za-z_][A-Za-z0-9_.]*)(?=:\s)", exception) + detail = exception.rsplit("\n", 1)[-1] + return (matches[-1] if matches else "Exception"), detail + return None, None + + +def _task_slug(task_name: object) -> str: + return str(task_name or "").split("/")[-1].split("__", 1)[0] + + +def _task_text(metadata: Mapping[str, object], task_name: object) -> str: + texts = metadata.get("task_texts") + if not isinstance(texts, Mapping): + return "" + return _clean_string(texts.get(_task_slug(task_name))) + + +def _timestamps(result: Mapping[str, object], trajectory: Mapping[str, object]) -> dict[str, str]: + timestamps = {} + for source, destination in (("started_at", "started_at"), ("finished_at", "finished_at")): + value = result.get(source) + if isinstance(value, str) and re.fullmatch(r"\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d(?:\.\d+)?Z", value): + timestamps[destination] = value + info = result.get("exception_info") + occurred_at = info.get("occurred_at") if isinstance(info, Mapping) else None + if (isinstance(occurred_at, str) + and re.fullmatch(r"\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d(?:\.\d+)?Z", occurred_at)): + timestamps["error_at"] = occurred_at + step_times = [step.get("timestamp") for step in trajectory.get("steps") or [] + if isinstance(step, Mapping) and isinstance(step.get("timestamp"), str)] + if step_times: + timestamps.setdefault("started_at", step_times[0]) + timestamps.setdefault("finished_at", step_times[-1]) + return timestamps + + +def _known_skill_body(metadata: Mapping[str, object]) -> bytes | None: + body = metadata.get("skill_body") + digest = metadata.get("skill_sha256") + if body in (None, "") and digest in (None, ""): + return None + if not isinstance(body, str) or not isinstance(digest, str): + raise TelemetryReceiptError("skill provenance is incomplete") + encoded = body.encode("utf-8") + if hashlib.sha256(encoded).hexdigest() != digest: + raise TelemetryReceiptError("skill provenance SHA-256 does not match its body") + return encoded + + +def _usage(result: Mapping[str, object], trajectory: Mapping[str, object]) -> dict: + agent_result = result.get("agent_result") + agent_result = agent_result if isinstance(agent_result, Mapping) else {} + fields = { + "input_tokens": agent_result.get("n_input_tokens"), + "cache_tokens": agent_result.get("n_cache_tokens"), + "output_tokens": agent_result.get("n_output_tokens"), + "cost_usd": agent_result.get("cost_usd"), + } + final = trajectory.get("final_metrics") + if isinstance(final, Mapping): + fields = { + "input_tokens": fields["input_tokens"] or final.get("total_prompt_tokens"), + "cache_tokens": fields["cache_tokens"] or final.get("total_cached_tokens"), + "output_tokens": fields["output_tokens"] or final.get("total_completion_tokens"), + "cost_usd": fields["cost_usd"] or final.get("total_cost_usd"), + } + return {key: value for key, value in fields.items() + if isinstance(value, (int, float)) and not isinstance(value, bool)} + + +def build_attempt_payload(trial: Path, metadata: Mapping[str, object]) -> dict: + """Return the bounded, normalized telemetry payload for one persisted Harbor trial.""" + with _AttemptReader(trial) as reader: + result = reader.read_json("result.json") + trajectory = _trajectory(reader) + exception_category, error_detail = _exception_evidence(reader, result) + skill_body = _known_skill_body(metadata) + solution_artifacts = reader.artifacts( + "verifier", "solution", skill_body=skill_body, first=("answer.md",)) + verifier_output = reader.artifacts( + "verifier", skip_solution=True, skill_body=skill_body) + artifact_projection = reader.artifact_projection() + agent_info = result.get("agent_info") + agent_info = agent_info if isinstance(agent_info, Mapping) else {} + model_info = agent_info.get("model_info") + model_info = model_info if isinstance(model_info, Mapping) else {} + model = metadata.get("model") or model_info.get("name") + task_name = result.get("task_name") or trial.name.split("__", 1)[0] + terminal_success = (bool(result) + and isinstance(result.get("finished_at"), str) + and bool(result["finished_at"].strip()) + and result.get("exception_info") is None) + payload = { + "exporter_revision": EXPORTER_REVISION, + "attempt": { + "id": _clean_string(result.get("id") or trial.name), + "trial_name": _clean_string(result.get("trial_name") or trial.name), + }, + "task": { + "name": _clean_string(task_name), + "checksum": _clean_string(result.get("task_checksum")), + "source": _clean_string(result.get("source")), + "text": _task_text(metadata, task_name), + }, + "skill": _clean_string(metadata.get("skill")), + "trajectory": trajectory, + "verifier_output": verifier_output, + "solution_artifacts": solution_artifacts, + "status": "failed" if exception_category else "succeeded" if terminal_success else "incomplete", + "exception_category": exception_category, + "error_detail": error_detail, + "timestamps": _timestamps(result, trajectory), + "usage": _usage(result, trajectory), + "model": _clean_string(model), + "metadata": _metadata(metadata), + } + if (artifact_projection["omitted_files"] or artifact_projection["omitted_hidden_paths"] + or artifact_projection["scan_truncated"]): + payload["artifact_projection"] = artifact_projection + return payload + + +def _payload_sha256(payload: Mapping[str, object]) -> str: + encoded = json.dumps(payload, sort_keys=True, separators=(",", ":"), + ensure_ascii=False).encode("utf-8") + return hashlib.sha256(encoded).hexdigest() + + +def _atomic_receipt(path: Path, receipt: Mapping[str, object]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + descriptor, name = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=path.parent) + temporary = Path(name) + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + json.dump(receipt, handle, indent=2) + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary, path) + except BaseException: + temporary.unlink(missing_ok=True) + raise + + +def _expected_trace_id(digest: str) -> str: + from langfuse import Langfuse + return Langfuse.create_trace_id(seed=f"{EXPORTER_REVISION}:{digest}") + + +def _expected_receipt(digest: str) -> dict: + return { + "status": "verified", + "trace_id": _expected_trace_id(digest), + "payload_sha256": digest, + "exporter_revision": EXPORTER_REVISION, + } + + +def _pending_receipt(digest: str, observation_id: str | None = None) -> dict: + receipt = { + "status": "pending", + "trace_id": _expected_trace_id(digest), + "payload_sha256": digest, + "exporter_revision": EXPORTER_REVISION, + } + if observation_id: + receipt["observation_id"] = observation_id + return receipt + + +def _read_receipt(trial: Path) -> dict: + with _AttemptReader(trial) as reader: + return reader.read_json(RECEIPT_NAME) + + +def _verified_receipt(trial: Path, digest: str) -> dict | None: + receipt = _read_receipt(trial) + if receipt == _expected_receipt(digest): + return receipt + return None + + +def _create_pending_receipt(path: Path, receipt: Mapping[str, object]) -> bool: + encoded = json.dumps(receipt, separators=(",", ":")).encode("utf-8") + try: + descriptor = os.open( + path, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0), + 0o600, + ) + except FileExistsError: + return False + try: + os.write(descriptor, encoded) + os.fsync(descriptor) + finally: + os.close(descriptor) + return True + + +def _read_back(trace_id: str, client) -> dict | None: + if hasattr(client, "read_trace"): + value = client.read_trace(trace_id) + return value if isinstance(value, dict) else None + public_key = os.environ.get("LANGFUSE_PUBLIC_KEY", "").strip() + secret_key = os.environ.get("LANGFUSE_SECRET_KEY", "").strip() + base_url = os.environ.get("LANGFUSE_BASE_URL", "http://langfuse-web:3000").rstrip("/") + if not public_key or not secret_key: + raise TelemetryReceiptError("Langfuse public read-back credentials are missing") + response = httpx.get(f"{base_url}/api/public/traces/{trace_id}", + auth=(public_key, secret_key), timeout=15) + if response.status_code == 404: + return None + if response.status_code != 200: + raise TelemetryReceiptError( + f"Langfuse public read-back returned status {response.status_code}") + value = response.json() + return value if isinstance(value, dict) else None + + +def _telemetry_evidence(payload: Mapping[str, object]) -> dict: + usage = payload.get("usage") if isinstance(payload.get("usage"), Mapping) else {} + return { + "revision": "harbor-agent-evidence-v1", + "model": payload.get("model") or "", + "status": payload.get("status") or "incomplete", + "tokens": {key: usage[key] for key in ( + "input_tokens", "cache_tokens", "output_tokens") if key in usage}, + } + + +def _matching_observation_id(value: dict | None, trace_id: str, digest: str, + evidence: Mapping[str, object], + observation_id: str | None = None, *, + exporter_revision: str = EXPORTER_REVISION) -> str | None: + if not value or value.get("id") != trace_id: + return None + observations = value.get("observations") + if not isinstance(observations, list) or len(observations) != 1: + return None + observation = observations[0] + if not isinstance(observation, Mapping): + return None + metadata = observation.get("metadata") + actual_id = observation.get("id") + if (not isinstance(actual_id, str) or not actual_id.strip() + or (observation_id is not None and actual_id != observation_id) + or observation.get("name") != "harbor-attempt" + or str(observation.get("type") or "").lower() != "agent" + or not isinstance(metadata, Mapping) + or metadata.get("payload_sha256") != digest + or metadata.get("exporter_revision") != exporter_revision + or metadata.get("telemetry_evidence") != evidence): + return None + return actual_id + + +def _wait_for_matching_observation(trace_id: str, digest: str, + evidence: Mapping[str, object], + observation_id: str, client) -> bool: + for attempt in range(_READBACK_ATTEMPTS): + if _matching_observation_id( + _read_back(trace_id, client), trace_id, digest, evidence, observation_id): + return True + if attempt + 1 < _READBACK_ATTEMPTS: + time.sleep(_READBACK_POLL_SECONDS) + return False + + +def _client(): + from langfuse import Langfuse + return Langfuse() + + +def _resume_pending(trial: Path, current: Mapping[str, object], pending: dict, + trace_id: str, digest: str, evidence: Mapping[str, object], client) -> dict: + if current == pending: + raise TelemetryReceiptError( + f"Langfuse trace {trace_id} has unknown pending observation state") + observation_id = current.get("observation_id") + if not (current.get("status") == "pending" + and current.get("trace_id") == trace_id + and current.get("payload_sha256") == digest + and current.get("exporter_revision") == EXPORTER_REVISION + and isinstance(observation_id, str) + and bool(observation_id.strip())): + raise TelemetryReceiptError(f"{trial} has a stale or malformed Langfuse receipt") + if _wait_for_matching_observation(trace_id, digest, evidence, observation_id, client): + receipt = _expected_receipt(digest) + _atomic_receipt(trial / RECEIPT_NAME, receipt) + return receipt + raise TelemetryReceiptError( + f"Langfuse trace {trace_id} remains pending public read-back") + + +def _migrate_verified_v2_receipt(trial: Path, current: Mapping[str, object], pending: dict, + evidence: Mapping[str, object], client) -> bool: + """Preserve one exact verified v2 receipt before authorizing telemetry-only v3 export.""" + from langfuse import Langfuse + + digest = current.get("payload_sha256") + trace_id = current.get("trace_id") + if not (current.get("status") == "verified" + and current.get("exporter_revision") == "harbor-langfuse-v2" + and isinstance(digest, str) and re.fullmatch(r"[0-9a-f]{64}", digest) + and isinstance(trace_id, str) + and trace_id == Langfuse.create_trace_id(seed=f"harbor-langfuse-v2:{digest}")): + return False + if not _matching_observation_id( + _read_back(trace_id, client), trace_id, digest, evidence, + exporter_revision="harbor-langfuse-v2"): + raise TelemetryReceiptError(f"{trial} has no matching verified v2 Langfuse trace") + archive = trial / LEGACY_V2_RECEIPT_NAME + if not _create_pending_receipt(archive, current): + if _path_json(archive) != dict(current): + raise TelemetryReceiptError(f"{trial} has a conflicting archived v2 receipt") + _atomic_receipt(trial / RECEIPT_NAME, pending) + return True + + +def _export_attempt(trial: Path, metadata: Mapping[str, object], client=None) -> dict: + payload = build_attempt_payload(trial, metadata) + digest = _payload_sha256(payload) + evidence = _telemetry_evidence(payload) + if receipt := _verified_receipt(trial, digest): + return receipt + client = client or _client() + trace_id = client.create_trace_id(seed=f"{EXPORTER_REVISION}:{digest}") + if trace_id != _expected_trace_id(digest): + raise TelemetryReceiptError("Langfuse returned a non-deterministic trace ID") + pending = _pending_receipt(digest) + current = _read_receipt(trial) + migrated = False + existing_checked = False + existing = None + if current: + if current.get("exporter_revision") == "harbor-langfuse-v2": + # Keep the verified v2 receipt active until the side-effect-free v3 collision/read + # check succeeds. A transient read error then leaves an exact retryable state. + existing = _read_back(trace_id, client) + existing_checked = True + migrated = _migrate_verified_v2_receipt(trial, current, pending, evidence, client) + if not migrated: + return _resume_pending(trial, current, pending, trace_id, digest, evidence, client) + if not existing_checked: + existing = _read_back(trace_id, client) + if existing is not None: + if not _matching_observation_id(existing, trace_id, digest, evidence): + raise TelemetryReceiptError(f"existing deterministic trace {trace_id} is not one expected attempt") + receipt = _expected_receipt(digest) + _atomic_receipt(trial / RECEIPT_NAME, receipt) + return receipt + if not migrated and not _create_pending_receipt(trial / RECEIPT_NAME, pending): + current = _read_receipt(trial) + if current == _expected_receipt(digest): + return current + return _resume_pending(trial, current, pending, trace_id, digest, evidence, client) + outbound = _redact_persisted(payload, os.environ) + # ISO timestamps are validated structurally before this final free-text scrub. The diagnostic + # host pattern also matches clock fragments, so retain the already-validated canonical values. + outbound["timestamps"] = payload["timestamps"] + for outbound_step, canonical_step in zip( + outbound["trajectory"].get("steps") or [], payload["trajectory"].get("steps") or []): + if "timestamp" in canonical_step: + outbound_step["timestamp"] = canonical_step["timestamp"] + observation_metadata = { + **outbound["metadata"], + "attempt": outbound["attempt"], + "payload_sha256": digest, + "exporter_revision": EXPORTER_REVISION, + "telemetry_evidence": evidence, + } + observation_output = {key: outbound[key] for key in ( + "skill", "status", "exception_category", "error_detail", "timestamps", + "trajectory", "verifier_output", "solution_artifacts")} + if "artifact_projection" in outbound: + observation_output["artifact_projection"] = outbound["artifact_projection"] + kwargs = { + "trace_context": {"trace_id": trace_id}, + "name": "harbor-attempt", + "as_type": "agent", + "input": outbound["task"], + "output": observation_output, + "metadata": observation_metadata, + "level": "ERROR" if outbound["status"] == "failed" else "DEFAULT", + "status_message": outbound["status"], + } + observation = client.start_observation(**kwargs) + observation_id = getattr(observation, "id", None) + if not isinstance(observation_id, str) or not observation_id.strip(): + raise TelemetryReceiptError("Langfuse returned no observation ID") + _atomic_receipt(trial / RECEIPT_NAME, _pending_receipt(digest, observation_id)) + observation.end() + client.flush() + if not _wait_for_matching_observation(trace_id, digest, evidence, observation_id, client): + raise TelemetryReceiptError(f"Langfuse trace {trace_id} failed public read-back verification") + receipt = _expected_receipt(digest) + _atomic_receipt(trial / RECEIPT_NAME, receipt) + return receipt + + +def _is_attempt(path: Path) -> bool: + with _AttemptReader(path) as reader: + if (reader.read( + "agent", "trajectory.json", limit=_MAX_TRAJECTORY_BYTES) is not None + or reader.read("exception.txt") is not None): + return True + opened = reader._open_dir(reader._trial_fd, "verifier", missing_ok=True) + if opened is not None: + descriptor, _info = opened + os.close(descriptor) + return True + return isinstance(reader.read_json("result.json").get("task_name"), str) + + +def _job_attempts(job: Path, identity: NativeTrialIdentity | None = None) -> list[Path]: + try: + job_info = job.lstat() + except OSError: + return [] + if stat.S_ISLNK(job_info.st_mode) or not stat.S_ISDIR(job_info.st_mode): + raise TelemetryReceiptError(f"{job} is not a real job directory") + attempts = [] + for path in sorted(job.iterdir()): + info = path.lstat() + if stat.S_ISLNK(info.st_mode): + raise TelemetryReceiptError(f"{path} is a symlinked attempt entry") + if not stat.S_ISDIR(info.st_mode): + continue + if identity is not None: + try: + if read_trial_identity(path) != identity: + continue + except ValueError: + continue + if _is_attempt(path): + attempts.append(path) + return attempts + + +def export_job_attempts(job: Path, metadata: Mapping[str, object], *, + identity: NativeTrialIdentity | None = None, client=None) -> list[dict]: + """Export every persisted attempt directly beneath one Harbor job directory.""" + attempts = _job_attempts(job, identity) + if not attempts: + raise TelemetryReceiptError(f"{job} has no persisted Harbor attempts") + return [_export_attempt(trial, metadata, client) for trial in attempts] + + +def _evidence_attempts(root: Path) -> list[Path]: + try: + root_info = root.lstat() + except OSError: + return [] + if stat.S_ISLNK(root_info.st_mode) or not stat.S_ISDIR(root_info.st_mode): + raise TelemetryReceiptError(f"{root} is not a real evidence root") + candidates = {path.parent.parent for path in root.rglob("agent/trajectory.json")} + candidates.update(path.parent for path in root.rglob("exception.txt")) + candidates.update(path.parent for path in root.rglob("verifier") if path.is_dir()) + for path in root.rglob("result.json"): + if isinstance(_path_json(path).get("task_name"), str): + candidates.add(path.parent) + return sorted(path for path in candidates if _is_attempt(path)) + + +def _trial_metadata(trial: Path, root: Path, + migration: Mapping[str, object] | None = None) -> dict: + metadata = {} + for ancestor in trial.parents: + combo = ancestor / "combo.json" + if combo.is_file(): + metadata.update(_path_json(combo)) + break + if ancestor == root: + break + if migration: + metadata.update(migration) + metadata["arm"] = "canary" if "canaries" in trial.parts else trial.parent.name + return metadata + + +def _validate_preserved_provenance(trial: Path, metadata: Mapping[str, object]) -> None: + skill = metadata.get("skill") + task_texts = metadata.get("task_texts") + if not isinstance(skill, str) or not skill.strip() or not isinstance(task_texts, Mapping): + raise TelemetryReceiptError(f"{trial} lacks explicit preserved provenance") + body = _known_skill_body(metadata) + with _AttemptReader(trial) as reader: + task_name = _task_slug(reader.read_json("result.json").get("task_name") or trial.name) + task_text = task_texts.get(task_name) + if body is None or not isinstance(task_text, str) or not task_text.strip(): + raise TelemetryReceiptError(f"{trial} lacks matching preserved provenance") + + +def export_evidence_root(root: Path, *, metadata: Mapping[str, object] | None = None, + client=None) -> list[dict]: + """Export attempts recursively from preserved Harbor evidence without invoking an agent.""" + attempts = _evidence_attempts(root) + if not attempts: + raise TelemetryReceiptError(f"{root} has no persisted Harbor attempts") + exports = [] + for trial in attempts: + trial_metadata = _trial_metadata(trial, root, metadata) + _validate_preserved_provenance(trial, trial_metadata) + exports.append(_export_attempt(trial, trial_metadata, client)) + return exports + + +def validate_job_receipts(job: Path, metadata: Mapping[str, object], *, + identity: NativeTrialIdentity | None = None) -> None: + """Fail closed unless every consumed attempt has a current verified receipt.""" + attempts = _job_attempts(job, identity) + if not attempts: + raise TelemetryReceiptError(f"{job} has no attempt receipt evidence") + for trial in attempts: + _validate_preserved_provenance(trial, metadata) + digest = _payload_sha256(build_attempt_payload(trial, metadata)) + with _AttemptReader(trial) as reader: + receipt = reader.read_json(RECEIPT_NAME) + if receipt != _expected_receipt(digest): + raise TelemetryReceiptError(f"{trial} has a stale or withheld Langfuse receipt") + + +def main(argv: Sequence[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--root", action="append", required=True, type=Path, + help="preserved Harbor evidence root (repeatable)") + parser.add_argument("--metadata", action="append", required=True, type=Path, + help="bounded provenance migration JSON paired with each --root") + args = parser.parse_args(argv) + if len(args.root) != len(args.metadata): + parser.error("each --root requires one paired --metadata document") + for root, metadata_path in zip(args.root, args.metadata): + metadata = load_provenance_metadata(metadata_path) + export_evidence_root(root, metadata=metadata) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/ingot/optimize/harbor_native.py b/ingot/optimize/harbor_native.py new file mode 100644 index 0000000..e2e2e61 --- /dev/null +++ b/ingot/optimize/harbor_native.py @@ -0,0 +1,306 @@ +"""Compile and recover exact Ingot identities in native Harbor jobs.""" +from __future__ import annotations + +import json +import os +import re +import stat +import tempfile +from dataclasses import dataclass +from pathlib import Path +from typing import Iterator, Mapping, Sequence + +from .harbor_gateway import (gateway_agent_env, gateway_agent_name, gateway_route) +from .harbor_targets import (HARNESS_PROTOCOLS, LocalTarget, harbor_agent_kwargs, + harbor_model, local_agent_env, protocol_for) + +NATIVE_TRIAL_MEMORY_MB = 2_048 +NATIVE_RUNNER_REVISION = f"native-v4-memory{NATIVE_TRIAL_MEMORY_MB}mb" + +_IDENTITY_KEYS = { + "INGOT_COMBINATION_ID": "combination_id", + "INGOT_ENDPOINT_FINGERPRINT": "endpoint_fingerprint", + "INGOT_HARNESS": "harness", + "INGOT_PROTOCOL": "protocol", + "INGOT_GATEWAY_REVISION": "gateway_revision", + "INGOT_ARM": "arm", +} +_FINGERPRINT = re.compile(r"^[0-9a-f]{12}$") +_REVISION = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*$") + + +@dataclass(frozen=True) +class NativeTrialIdentity: + combination_id: str + endpoint_fingerprint: str + harness: str + protocol: str + gateway_revision: str + arm: str + + def __post_init__(self) -> None: + values = tuple(self.__dict__.values()) + if not all(isinstance(value, str) and value for value in values): + raise ValueError("Harbor trial identity fields must be nonempty strings") + if any(token in self.combination_id for token in ("/", "\\", "\x00", "..")): + raise ValueError("Harbor trial identity combination is not path-safe") + if not _FINGERPRINT.fullmatch(self.endpoint_fingerprint): + raise ValueError("Harbor trial identity endpoint fingerprint is invalid") + if self.harness not in HARNESS_PROTOCOLS: + raise ValueError("Harbor trial identity harness is unsupported") + if self.protocol != protocol_for(self.harness): + raise ValueError("Harbor trial identity protocol does not match its harness") + if not _REVISION.fullmatch(self.gateway_revision): + raise ValueError("Harbor trial identity gateway revision is invalid") + if self.arm not in {"canary", "skill", "control"}: + raise ValueError("Harbor trial identity arm is invalid") + if not self.combination_id.startswith(f"{self.harness}@"): + raise ValueError("Harbor trial identity combination does not match its harness") + if self.endpoint_fingerprint not in self.combination_id: + raise ValueError("Harbor trial identity combination does not match its endpoint") + + +@dataclass(frozen=True) +class NativeCell: + target: LocalTarget + harness: str + + @property + def combination_id(self) -> str: + return native_trial_identity(self.target, self.harness, "skill").combination_id + + +def identity_env(identity: NativeTrialIdentity) -> dict[str, str]: + """Return the exact non-secret fields Harbor persists on ``agent.env``.""" + return {key: getattr(identity, field) for key, field in _IDENTITY_KEYS.items()} + + +def identity_from_env(env: Mapping[str, object]) -> NativeTrialIdentity: + """Parse exactly the six persisted Ingot identity fields from an agent environment.""" + ingot_keys = {key for key in env if isinstance(key, str) and key.startswith("INGOT_")} + if ingot_keys != set(_IDENTITY_KEYS): + raise ValueError("Harbor trial identity fields are incomplete or ambiguous") + values = {field: env[key] for key, field in _IDENTITY_KEYS.items()} + try: + return NativeTrialIdentity(**values) + except (TypeError, ValueError) as exc: + raise ValueError(f"Harbor trial identity is invalid: {exc}") from exc + + +def read_trial_identity(trial_dir: Path) -> NativeTrialIdentity: + """Recover identity from Harbor's resolved per-trial lock and fail closed.""" + lock_path = Path(trial_dir) / "lock.json" + try: + lock_info = lock_path.lstat() + if not stat.S_ISREG(lock_info.st_mode): + raise ValueError("Harbor trial identity lock is not a regular file") + payload = json.loads(lock_path.read_text()) + env = payload["agent"]["env"] + except (OSError, ValueError, TypeError, KeyError) as exc: + raise ValueError("Harbor trial identity lock is unreadable") from exc + if not isinstance(env, dict): + raise ValueError("Harbor trial identity environment is invalid") + return identity_from_env(env) + + +def iter_attempt_dirs(job_dir: Path, *, identity: NativeTrialIdentity | None = None + ) -> Iterator[Path]: + """Yield direct Harbor attempt directories, optionally by exact persisted identity.""" + job_dir = Path(job_dir) + try: + job_info = job_dir.lstat() + except OSError as exc: + raise ValueError("native Harbor job is not a real directory") from exc + if stat.S_ISLNK(job_info.st_mode) or not stat.S_ISDIR(job_info.st_mode): + raise ValueError("native Harbor job is not a real directory") + for attempt in sorted(job_dir.iterdir()): + info = attempt.lstat() + if stat.S_ISLNK(info.st_mode): + raise ValueError("native Harbor entry is not a real attempt directory") + if not stat.S_ISDIR(info.st_mode): + continue + result = attempt / "result.json" + try: + result_info = result.lstat() + except OSError: + continue + if not stat.S_ISREG(result_info.st_mode): + raise ValueError("native Harbor result is not a regular file") + if identity is None: + yield attempt + continue + try: + lock_info = (attempt / "lock.json").lstat() + except OSError: + continue + if not stat.S_ISREG(lock_info.st_mode): + raise ValueError("native Harbor lock is not a regular file") + try: + observed = read_trial_identity(attempt) + except ValueError: + continue + if observed == identity: + yield attempt + + +def native_trial_identity(target: LocalTarget, harness: str, arm: str) -> NativeTrialIdentity: + """Derive the only valid persisted identity for one target, harness, and arm.""" + route = gateway_route(target, harness) + return NativeTrialIdentity( + combination_id=f"{harness}@{target.served_model}--{target.job_slug}", + endpoint_fingerprint=target.fingerprint, + harness=harness, + protocol=protocol_for(harness), + gateway_revision=f"{route.revision if route else 'direct'}-{NATIVE_RUNNER_REVISION}", + arm=arm, + ) + + +def compile_agent_config(target: LocalTarget, harness: str, arm: str, skill_source: Path, + endpoint_limit: int) -> dict: + """Compile one Harbor agent entry whose resolved lock retains exact cell identity.""" + if not isinstance(endpoint_limit, int) or isinstance(endpoint_limit, bool) or endpoint_limit < 1: + raise ValueError("endpoint concurrency must be a positive integer") + + identity = native_trial_identity(target, harness, arm) + route = gateway_route(target, harness) + agent = gateway_agent_name(route) if route else harness + env = gateway_agent_env(target, route) if route else local_agent_env(target, harness) + config = { + "model_name": route.model if route else harbor_model(target, harness), + "n_concurrent": endpoint_limit, + "concurrency_group": f"endpoint:{target.fingerprint}", + "skills": [str(skill_source)] if arm in {"skill", "canary"} else [], + "kwargs": {} if route else harbor_agent_kwargs(target, harness), + "env": {**env, **identity_env(identity)}, + } + if ":" in agent: + config["import_path"] = agent + else: + config["name"] = agent + return config + + +def _job_base(dataset: Path, task_names: Sequence[str], jobs_dir: Path, attempts: int, + global_limit: int, agents: list[dict]) -> dict: + if (not isinstance(global_limit, int) or isinstance(global_limit, bool) + or global_limit < 1): + raise ValueError("global concurrency must be a positive integer") + names = list(task_names) + if not names or not all(isinstance(name, str) and name for name in names): + raise ValueError("task names must be nonempty strings") + if len(names) != len(set(names)): + raise ValueError("task names must be unique") + return { + "jobs_dir": str(jobs_dir), + "n_attempts": attempts, + "n_concurrent_trials": global_limit, + "agents": agents, + "datasets": [{"path": str(dataset), "task_names": names}], + } + + +def _limits(cells: Sequence[NativeCell], endpoint_limits: Mapping[str, int]) -> None: + required = {cell.target.fingerprint for cell in cells} + if set(endpoint_limits) != required: + raise ValueError("endpoint concurrency must cover exactly the requested targets") + for value in endpoint_limits.values(): + if not isinstance(value, int) or isinstance(value, bool) or value < 1: + raise ValueError("endpoint concurrency must be a positive integer") + identities = [cell.combination_id for cell in cells] + if len(identities) != len(set(identities)): + raise ValueError("native Harbor cells must have unique combination identities") + + +def compile_canary_job(dataset: Path, task_name: str, cells: Sequence[NativeCell], + skill_source: Path, jobs_dir: Path, *, global_limit: int, + endpoint_limits: Mapping[str, int]) -> dict: + """Compile one skill-bearing attempt for every requested model/harness cell.""" + cells = list(cells) + if not cells: + raise ValueError("native Harbor job requires at least one cell") + _limits(cells, endpoint_limits) + agents = [compile_agent_config(cell.target, cell.harness, "canary", skill_source, + endpoint_limits[cell.target.fingerprint]) + for cell in cells] + return _job_base(dataset, [task_name], jobs_dir, 1, global_limit, agents) + + +def compile_measurement_job(dataset: Path, task_names: Sequence[str], cells: Sequence[NativeCell], + skill_source: Path, jobs_dir: Path, *, attempts: int, + global_limit: int, + endpoint_limits: Mapping[str, int], + arms: Sequence[str] = ("skill", "control")) -> dict: + """Compile the requested exact arms for every canary-approved cell.""" + if not isinstance(attempts, int) or isinstance(attempts, bool) or attempts != 3: + raise ValueError("full measurement requires exactly three attempts") + names = list(task_names) + if len(names) != 4: + raise ValueError("full measurement requires exactly four tasks") + cells = list(cells) + if not cells: + raise ValueError("native Harbor job requires at least one cell") + arms = tuple(arms) + if not arms or len(arms) != len(set(arms)) or not set(arms) <= {"skill", "control"}: + raise ValueError("measurement arms must be a unique nonempty skill/control subset") + _limits(cells, endpoint_limits) + agents = [] + buckets: dict[str, list[NativeCell]] = {} + for cell in cells: + buckets.setdefault(cell.target.fingerprint, []).append(cell) + ordered = [] + while any(buckets.values()): + for fingerprint in buckets: + if buckets[fingerprint]: + ordered.append(buckets[fingerprint].pop(0)) + for cell in ordered: + limit = endpoint_limits[cell.target.fingerprint] + for arm in arms: + agents.append(compile_agent_config(cell.target, cell.harness, arm, skill_source, limit)) + return _job_base(dataset, names, jobs_dir, attempts, global_limit, agents) + + +def select_measurement_cells(cells: Sequence[NativeCell], canaries: Mapping[str, Mapping[str, object]] + ) -> tuple[list[NativeCell], dict[str, dict[str, object]]]: + """Partition exact passed canaries from explicit unmeasured cell records.""" + selected = [] + unmeasured = {} + for cell in cells: + record = canaries.get(cell.combination_id) + if isinstance(record, Mapping) and record.get("ok") is True and "error" not in record: + selected.append(cell) + continue + error = str((record or {}).get("error") or "canary did not pass")[:400] + unmeasured[cell.combination_id] = { + "combination": cell.combination_id, + "harness": cell.harness, + "target_alias": cell.target.alias, + "endpoint_fingerprint": cell.target.fingerprint, + "state": "unmeasured", + "error": error, + } + return selected, unmeasured + + +def write_job_config(path: Path, config: Mapping[str, object]) -> None: + """Write one deterministic Harbor job document without exposing a partial file.""" + path = Path(path) + if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]*", path.name): + raise ValueError("Harbor config filename must be a safe slug") + path.parent.mkdir(parents=True, exist_ok=True) + with tempfile.NamedTemporaryFile("w", encoding="utf-8", dir=path.parent, + prefix=f".{path.name}.", suffix=".tmp", + delete=False) as handle: + temporary = Path(handle.name) + try: + json.dump(config, handle, indent=2, sort_keys=True) + handle.write("\n") + handle.flush() + os.fsync(handle.fileno()) + except BaseException: + temporary.unlink(missing_ok=True) + raise + try: + os.replace(temporary, path) + finally: + temporary.unlink(missing_ok=True) diff --git a/ingot/optimize/harbor_redaction.py b/ingot/optimize/harbor_redaction.py new file mode 100644 index 0000000..1e2719d --- /dev/null +++ b/ingot/optimize/harbor_redaction.py @@ -0,0 +1,41 @@ +"""Shared redaction for persisted Harbor diagnostics and exported text.""" +from __future__ import annotations + +import re +from collections.abc import Mapping + + +_HARBOR_RECEIPT_EXCERPT_LIMIT = 2000 + + +def _receipt_secret_values(parent: Mapping[str, str]) -> tuple[str, ...]: + """Known ambient secrets, longest first so a shorter value cannot leave a suffix behind.""" + return tuple(sorted({value for key, value in parent.items() if value and + (key.endswith(("API_KEY", "TOKEN", "SECRET", "PASSWORD")) or key == "API_KEY")}, + key=len, reverse=True)) + + +def _redact_harbor_receipt_output(value: object, parent: Mapping[str, str]) -> str: + """Return a bounded Harbor diagnostic excerpt without endpoint or credential material.""" + text = str(value or "") + for secret in _receipt_secret_values(parent): + text = text.replace(secret, "") + text = re.sub(r"https?://[^\s\"']+", "", text, flags=re.IGNORECASE) + text = re.sub(r"(?im)\b(authorization\s*:\s*)[^\r\n]+", r"\1", text) + text = re.sub(r"(?i)\b(x-api-key\s*:\s*)[^\s,;\"']+", r"\1", text) + text = re.sub(r"\b(?:sk|pk)[_-][A-Za-z0-9_-]+\b", "", text) + text = re.sub(r"(?i)([\"']?[A-Z][A-Z0-9_]*(?:API_KEY|TOKEN|SECRET|PASSWORD)[\"']?\s*[:=]\s*[\"']?)[^\s,;\"'}]+", + r"\1", text) + text = re.sub(r"\b(?:localhost|[A-Za-z0-9.-]+):[0-9]{2,5}\b", "", text) + return text[-_HARBOR_RECEIPT_EXCERPT_LIMIT:] + + +def _redact_persisted(value: object, parent: Mapping[str, str]) -> object: + """Copy diagnostics for disk; runtime callers keep their original exception objects.""" + if isinstance(value, str): + return _redact_harbor_receipt_output(value, parent) + if isinstance(value, Mapping): + return {key: _redact_persisted(item, parent) for key, item in value.items()} + if isinstance(value, list): + return [_redact_persisted(item, parent) for item in value] + return value diff --git a/ingot/optimize/harbor_report.py b/ingot/optimize/harbor_report.py new file mode 100644 index 0000000..fd8a665 --- /dev/null +++ b/ingot/optimize/harbor_report.py @@ -0,0 +1,195 @@ +"""Read a harness x model matrix off disk for display. + +`harbor_eval` writes `runs/harbor/.json`; `harbor_rescore` writes `.rescored.json` +from the same job directories when the scoring rules change. The two carry the same rows under +different keys (`harnesses` and `combinations`), and the rescored file is the corrected one wherever +it exists — the first grid shipped two fabricated cells that only rescoring removed. + +Stdlib only, and no judging: this reads results, it does not produce them. + +The reporting rules here are the ones the matrix violated the first time it was run: + +- a combination that failed carries no `lift` at all, so nothing downstream can render it as a zero + in a column of measurements; +- every row carries the `n` it was measured over, because a lift over two tasks and one over four + are different claims; +- and if the control arms sit near the top of the scale, the instrument could not discriminate and + the ranking means nothing, so the ceiling is reported instead of the winner. +""" +from __future__ import annotations + +import json +import math +from pathlib import Path + +from ingot import paths + +from .harbor_targets import TARGETS + +HARBOR_DIR = paths.runs() / "harbor" + +# Above this, the controls are close enough to the top of the scale that the remaining headroom is +# smaller than the judge's own noise (~0.10 observed across re-runs of an unchanged arm), so a +# treatment cannot separate from its control whatever the skill does. Reporting a "best combination" +# off a matrix in this state is reporting the noise. +CEILING = 0.75 + + +def matrix_path(skill: str, root: Path | None = None) -> Path | None: + """The newest of the rescored and raw matrices for `skill`, or None if neither exists.""" + base = root or HARBOR_DIR + found = [p for p in (base / f"{skill}.rescored.json", base / f"{skill}.json") if p.is_file()] + return max(found, key=lambda p: p.stat().st_mtime) if found else None + + +def _split(combo: str) -> tuple[str, str]: + harness, _, model = combo.partition("@") + return harness, model or "harness default" + + +def read_matrix(skill: str, root: Path | None = None) -> dict | None: + """Normalized rows for one skill, or None when the skill has never been run. + + Raises ValueError on an unreadable or malformed file rather than returning an empty matrix: a + page saying "no combinations" when the file is actually corrupt reads as "nothing helps".""" + path = matrix_path(skill, root) + if path is None: + return None + try: + raw = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError) as error: + raise ValueError(f"{path.name} is unreadable ({error})") from error + if not isinstance(raw, dict): + raise ValueError(f"{path.name} does not contain a matrix") + + source = raw.get("combinations") if isinstance(raw.get("combinations"), dict) else raw.get("harnesses") + rows = [] + for combo, record in sorted((source or {}).items()): + if not isinstance(record, dict): + continue + harness, model = _split(combo) + target = TARGETS.get(str(record.get("target_alias") or ""), {}) + family = str(record.get("family") or "") + display_model = (target.get("display_name") + if family.startswith("Qwen") and target.get("family") == family else None) + row = {"combination": combo, + "harness": record.get("harness") or harness, + # Keep wire IDs in provenance and job identity, but present the canonical model + # name whenever a pinned target alias provides one. + "model": display_model or record.get("model") or model} + # Endpoint identity is evidence, not presentation decoration. Legacy matrices do not + # have it, so only copy fields that the row actually recorded; a fabricated null identity + # would falsely suggest that an endpoint was checked. + for field in ("target_alias", "endpoint_fingerprint", "protocol", "family", + "quantization", "tool_parser", "exploratory", + "rankable"): + if field in record: + row[field] = record[field] + size = record.get("parameter_billions") + if isinstance(size, (int, float)) and not isinstance(size, bool) and math.isfinite(size) \ + and size > 0: + row["parameter_billions"] = size + if "lift" not in record: + # No lift, no means, no scores. A row shaped like a measurement is how "the container + # died" became "lift +0.750" the first time this ran. + row["error"] = str(record.get("error") or "not measured") + rows.append(row) + continue + row.update({"lift": record["lift"], + "skill_mean": record.get("skill_mean"), + "control_mean": record.get("control_mean"), + "n": record.get("tasks_scored"), + "attempts": record.get("attempts") or 1, + "dropped": record.get("tasks_dropped") or []}) + rows.append(row) + + summary = summarize(rows) + # A global winner is not meaningful once columns have independent evidence fitness. Keep the + # reusable `summarize` result intact for callers that need it, but do not expose its global + # `best` to the UI alongside per-model winners. + summary.pop("best", None) + models = sorted({row["model"] for row in rows}) + harnesses = sorted({row["harness"] for row in rows}) + exploratory = raw.get("exploratory") is True or any( + row.get("exploratory") is True for row in rows) + rankable = raw.get("rankable") is not False and all( + row.get("rankable") is not False for row in rows) + return {"skill": skill, "source": path.name, "rescored": path.name.endswith(".rescored.json"), + "generated": int(path.stat().st_mtime), "judge": raw.get("judge") or "", + "exploratory": exploratory, "rankable": rankable, + "rows": rows, "models": models, "harnesses": harnesses, + "model_summaries": summarize_models(rows), **summary} + + +def summarize(rows: list[dict]) -> dict: + """Headline numbers, plus whether the matrix is fit to be read as a ranking at all.""" + measured = [r for r in rows if "lift" in r] + if not measured: + return {"measured": 0, "unmeasured": len(rows), "mean_lift": None, + "control_mean": None, "best": None, + "warning": "No combination produced a measurement." if rows else ""} + + controls = [r["control_mean"] for r in measured if isinstance(r.get("control_mean"), (int, float))] + control_mean = sum(controls) / len(controls) if controls else None + best = max(measured, key=lambda r: r["lift"]) + + warning = "" + if any(r.get("rankable") is False for r in measured): + warning = ("Exploratory evidence is shown for inspection but is not readable as a " + "ranking. Re-run with the full measurement contract before naming a winner.") + elif control_mean is not None and control_mean >= CEILING: + # Deliberately not "the best combination is X". At this control level the ranking is noise, + # and naming a winner off it is the failure mode this whole module exists to prevent. + warning = (f"Controls average {control_mean:.3f} of 1.00, at or above the {CEILING:.2f} " + f"ceiling. There is less headroom left than the judge's own run-to-run spread, " + f"so these combinations cannot be ranked against each other — the held-out tasks " + f"are too easy, whatever the lift column says.") + elif all((r.get("n") or 0) < 3 for r in measured): + warning = ("Every combination was measured over fewer than 3 tasks; these differences are " + "not distinguishable from noise.") + elif all((r.get("attempts") or 1) < 2 for r in measured): + # Measured, not assumed: two control-arm runs of an identical configuration moved a task's + # score by 0.278 and swapped two harnesses' rank, while re-judging one fixed answer three + # times was identical. A single attempt per task sits under that, so a ranking built from + # these rows is a ranking of the agents' own run-to-run variance. + warning = ("Every combination was run with one attempt per task. Repeat runs of an " + "identical configuration have moved a task's score by 0.28 and swapped the rank " + "of two harnesses, so single-attempt differences are agent variance rather than " + "an effect. Re-run with -k 3 or more before reading this as a ranking.") + + return {"measured": len(measured), "unmeasured": len(rows) - len(measured), + "mean_lift": sum(r["lift"] for r in measured) / len(measured), + "control_mean": control_mean, + "best": None if warning else {"combination": best["combination"], "lift": best["lift"], + "n": best.get("n")}, + "warning": warning} + + +def summarize_models(rows: list[dict]) -> dict[str, dict]: + """Summarize each recorded model column without filling absent intersections.""" + summaries = {} + for model in sorted({row["model"] for row in rows}): + summary = summarize([row for row in rows if row["model"] == model]) + best = summary.pop("best") + summaries[model] = { + "model": model, + "measured": summary["measured"], + "unmeasured": summary["unmeasured"], + "mean_lift": summary["mean_lift"], + "control_mean": summary["control_mean"], + "warning": summary["warning"], + "best_harness": None if best is None else next( + row["harness"] for row in rows + if row["model"] == model and row.get("combination") == best["combination"] + ), + } + return summaries + + +def available(root: Path | None = None) -> list[str]: + """Skills that have a matrix on disk, newest first.""" + base = root or HARBOR_DIR + if not base.is_dir(): + return [] + skills = {p.name.split(".")[0] for p in base.glob("*.json")} + return sorted((s for s in skills if s), key=lambda s: -(matrix_path(s, base).stat().st_mtime)) diff --git a/ingot/optimize/harbor_rescore.py b/ingot/optimize/harbor_rescore.py new file mode 100644 index 0000000..a15840b --- /dev/null +++ b/ingot/optimize/harbor_rescore.py @@ -0,0 +1,524 @@ +"""Recompute compatible Harbor matrices from job directories already on disk. + +The agents have already run; rescoring only re-reads their persisted verifier artifacts. A matrix +may combine the preserved proprietary jobs and endpoint-qualified local jobs only when the evidence +states that they used the same task set and retry count. Every answer is then judged under the +current scoring provenance. + +Usage: python -m ingot.optimize.harbor_rescore [--jobs runs/harbor/jobs/] +""" +from __future__ import annotations + +import json +import os +import tempfile +from pathlib import Path +from typing import Sequence + +from .ab import load_tasks +from .agy_judge import AGY_IDENTITY, preflight +from .harbor_eval import (HARBOR_DIR, _task_fingerprint, _task_name, + broken_tasks, broken_trials, collect_answers, score) +from .harbor_langfuse import load_provenance_metadata, validate_job_receipts +from .harbor_native import NATIVE_RUNNER_REVISION, NativeTrialIdentity, identity_from_env + + +_EVIDENCE_FIELDS = ("task_fingerprint", "attempts") +_CURRENT_SCORING = { + "judge": "agy/gemini-3.6-flash-medium", + "scoring_revision": "harbor-rubric-v2-agy", + "judge_billing_mode": "subscription", +} +_HISTORICAL_SCORING_FIELDS = { + "judge", "scoring_revision", "judge_runtime", "judge_billing_mode", "billing_mode", + "cost_usd", +} +_MEASUREMENT_FIELDS = { + "error", "score", "scores", "skill_mean", "control_mean", "lift", "skill_scores", + "control_scores", "tasks_scored", "tasks_dropped", "mean_lift", "scored", "unscorable", + "n", "dropped", "best", "measured", "unmeasured", + "canary_error", +} +_AGY_SCORE_CONCURRENCY = 4 + + +def _arm_evidence(combo: Path, arm: str) -> tuple[Path, NativeTrialIdentity | None]: + """Resolve legacy arm directories or one exact arm view inside a native sibling job.""" + metadata = _combo_identity(combo) + native_jobs = metadata.get("native_jobs") + if native_jobs is not None and not isinstance(native_jobs, dict): + raise ValueError("native jobs must map arms to sibling slugs") + native_job = native_jobs.get(arm) if isinstance(native_jobs, dict) else metadata.get("native_job") + if native_job is None: + return combo / arm, None + if (not isinstance(native_job, str) + or not native_job + or not all(character.isalnum() or character in "._-" for character in native_job)): + raise ValueError("native job must be a safe sibling slug") + identities = metadata.get("native_identities") + if not isinstance(identities, dict) or arm not in identities: + raise ValueError(f"native job is missing {arm} identity") + env = identities[arm] + if not isinstance(env, dict): + raise ValueError(f"native job {arm} identity is invalid") + identity = identity_from_env(env) + if identity.arm != arm: + raise ValueError(f"native job {arm} identity has the wrong arm") + expected = { + "combination_id": metadata.get("combination"), + "harness": metadata.get("harness"), + "endpoint_fingerprint": metadata.get("endpoint_fingerprint"), + "protocol": metadata.get("protocol"), + "gateway_revision": metadata.get("gateway_revision", "direct"), + } + for field, value in expected.items(): + if getattr(identity, field) != value: + raise ValueError(f"native job {arm} identity does not match combo metadata") + return combo.parent / native_job, identity + + +def _combo_identity(combo: Path) -> dict: + """Return every identity field the live run recorded beside this combination.""" + try: + record = json.loads((combo / "combo.json").read_text(encoding="utf-8")) + except (OSError, ValueError) as error: + raise ValueError(f"{combo} has no readable combo.json identity") from error + if not isinstance(record, dict) or not record: + raise ValueError(f"{combo} has no combo identity") + return dict(record) + + +def _identity_key(identity: dict) -> str: + """Deduplicate exact recorded identities, never lossy job-directory names.""" + try: + return json.dumps(identity, sort_keys=True, separators=(",", ":"), ensure_ascii=False) + except (TypeError, ValueError) as error: + raise ValueError("combo identity is not JSON data") from error + + +def discover_combinations(roots: Sequence[Path]) -> list[Path]: + """Return unique combinations across roots, keyed by complete combo metadata.""" + found: list[Path] = [] + identities: set[str] = set() + for root in roots: + if not root.is_dir(): + raise SystemExit(f"no job directories at {root}") + combos = sorted(path for path in root.iterdir() + if path.is_dir() and (path / "combo.json").is_file()) + for combo in combos: + record = _combo_identity(combo) + key = _identity_key(record) + if key not in identities: + found.append(combo) + identities.add(key) + return found + + +def select_combinations(roots: Sequence[Path], paths: Sequence[Path]) -> list[Path]: + """Select exact real directories without inspecting sibling combinations.""" + allowed = {root.resolve() for root in roots} + found: list[Path] = [] + identities: set[str] = set() + for path in paths: + if path.is_symlink() or not path.is_dir() or path.parent.resolve() not in allowed: + raise ValueError(f"selected combination must be a real immediate child of a jobs root: {path}") + resolved = path.resolve() + key = _identity_key(_combo_identity(resolved)) + if key not in identities: + found.append(resolved) + identities.add(key) + if not found: + raise ValueError("no completed combination paths selected") + return found + + +def validate_compatibility(records: Sequence[dict]) -> None: + """Refuse raw evidence whose task set or retry count differs.""" + if not records: + raise ValueError("no combination evidence found") + for field in _EVIDENCE_FIELDS: + values = [record.get(field) for record in records] + if field == "attempts": + invalid = any(not isinstance(value, int) or isinstance(value, bool) or value < 1 + for value in values) + else: + invalid = any(not isinstance(value, str) or not value.strip() for value in values) + if invalid: + raise ValueError(f"combination metadata is missing or invalid {field}") + if len(set(values)) != 1: + raise ValueError(f"incompatible combination metadata: {field}") + + +def _validate_current_compatibility(records: Sequence[dict], holdout: list[dict]) -> None: + """Require otherwise-compatible evidence to describe this exact raw task run.""" + if records[0]["task_fingerprint"] != _task_fingerprint(holdout): + raise ValueError("combination metadata does not match current task_fingerprint") + attempts = records[0]["attempts"] + exploratory = all(record.get("exploratory") is True and record.get("rankable") is False + for record in records) + if attempts != 3 and not (attempts == 1 and exploratory): + raise ValueError("combination metadata attempts do not match current measurement contract") + + +def _load_legacy_manifest(path: Path) -> dict: + """Read the user-supplied historical metadata; never fill missing history from today.""" + try: + manifest = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError) as error: + raise ValueError(f"legacy metadata manifest is unreadable: {path}") from error + if not isinstance(manifest, dict) or not isinstance(manifest.get("root"), str): + raise ValueError("legacy metadata manifest needs a root") + missing = [field for field in _EVIDENCE_FIELDS if field not in manifest] + if missing: + raise ValueError(f"legacy metadata manifest is missing {', '.join(missing)}") + return manifest + + +def _legacy_metadata(skill: str, root: Path, combo: Path, identity: dict, + manifest: dict | None) -> dict: + """Attach only explicit history to an old proprietary combination after cross-checking it.""" + if manifest is None: + raise ValueError(f"{combo} needs an explicit legacy metadata manifest") + if Path(manifest["root"]).resolve() != root.resolve(): + raise ValueError(f"{combo} legacy metadata manifest names a different root") + try: + matrix = json.loads((HARBOR_DIR / f"{skill}.json").read_text(encoding="utf-8")) + except (OSError, ValueError) as error: + raise ValueError(f"{combo} lacks explicit matrix metadata") from error + if not isinstance(matrix, dict): + raise ValueError(f"{combo} lacks explicit matrix metadata") + + key = identity.get("combination") or combo.name + rows = matrix.get("harnesses") or {} + row = rows.get(key) if isinstance(rows, dict) else None + if matrix.get("skill") != skill or not isinstance(row, dict): + raise ValueError(f"{combo} lacks explicit matrix metadata") + metadata = {field: manifest[field] for field in _EVIDENCE_FIELDS} + matrix_checks = {"attempts": row.get("attempts")} + conflicting_matrix = [field for field, value in matrix_checks.items() + if value not in (None, "") and value != metadata[field]] + if conflicting_matrix: + raise ValueError(f"{combo} conflicts with legacy matrix metadata: {', '.join(conflicting_matrix)}") + conflicting = [field for field in _EVIDENCE_FIELDS + if identity.get(field) not in (None, "") + and identity[field] != metadata[field]] + if conflicting: + raise ValueError(f"{combo} conflicts with legacy metadata manifest: {', '.join(conflicting)}") + return {**identity, **metadata} + + +def _evidence_metadata(skill: str, root: Path, combo: Path, manifest: dict | None) -> dict: + identity = _combo_identity(combo) + if all(identity.get(field) not in (None, "") for field in _EVIDENCE_FIELDS): + return identity + return _legacy_metadata(skill, root, combo, identity, manifest) + + +def _current_scoring(runtime: dict) -> dict: + """Scoring provenance for this one rescore, without inventing metered cost.""" + return {**_CURRENT_SCORING, "judge_runtime": str(runtime["version"])} + + +def current_scoring_identity() -> dict: + """Resolve the exact current subscription scorer identity once for a controller run.""" + return _current_scoring(preflight()) + + +def _output_identity(identity: dict, scoring: dict) -> dict: + """Replace any historical scoring receipt while preserving raw evidence identity.""" + raw = {key: value for key, value in identity.items() + if key not in _HISTORICAL_SCORING_FIELDS | _MEASUREMENT_FIELDS} + return {**raw, **scoring} + + +def _validate_attempt_artifacts(arm: Path, skill: str, holdout: list[dict], attempts: int, + identity: NativeTrialIdentity | None = None) -> None: + """Require one readable Harbor trial record for every expected task attempt.""" + observed = {_task_name(skill, index): 0 for index in range(len(holdout))} + from .harbor_native import iter_attempt_dirs + results = ([attempt / "result.json" for attempt in iter_attempt_dirs(arm, identity=identity)] + if identity is not None else sorted(arm.glob("*/result.json"))) + for result in results: + try: + record = json.loads(result.read_text(encoding="utf-8")) + except (OSError, ValueError) as error: + raise RuntimeError(f"{arm.name} arm has an unreadable attempt record") from error + task_name = record.get("task_name") if isinstance(record, dict) else None + if not isinstance(task_name, str) or not task_name.strip(): + raise RuntimeError(f"{arm.name} arm has an attempt record without a task name") + task = task_name.split("/")[-1].split("__")[0] + if task not in observed: + raise RuntimeError(f"{arm.name} arm has an unexpected attempt for {task}") + observed[task] += 1 + mismatched = [f"{task}={count}" for task, count in observed.items() if count != attempts] + if mismatched: + raise RuntimeError(f"{arm.name} arm needs exactly {attempts} attempt records per task; " + f"found {', '.join(mismatched)}") + + +def _row_key(identity: dict, rows: dict[str, dict]) -> str: + """Keep a malformed repeated combination label from overwriting distinct endpoint evidence.""" + key = str(identity.get("combination") or "") + if not key: + raise ValueError("combo identity is missing combination") + if key not in rows: + return key + fingerprint = str(identity.get("endpoint_fingerprint") or "unknown") + qualified = f"{key}--{fingerprint}" + if qualified not in rows: + return qualified + suffix = 2 + while f"{qualified}-{suffix}" in rows: + suffix += 1 + return f"{qualified}-{suffix}" + + +def atomic_write_json(path: Path, payload: dict) -> None: + """Publish JSON with one replacement; every earlier failure leaves ``path`` untouched.""" + encoded = json.dumps(payload, indent=2).encode("utf-8") + path.parent.mkdir(parents=True, exist_ok=True) + descriptor, temporary_name = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=path.parent) + temporary = Path(temporary_name) + try: + with os.fdopen(descriptor, "wb") as handle: + handle.write(encoded) + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary, path) + except BaseException: + temporary.unlink(missing_ok=True) + raise + + +def _summary(skill: str, rows: dict[str, dict], shared: dict, scoring: dict) -> dict | None: + scored = {name: row for name, row in rows.items() if "lift" in row} + if not scored: + return None + return { + "skill": skill, + "combinations": rows, + "exploratory": any(row.get("exploratory") is True for row in rows.values()), + "rankable": all(row.get("rankable") is not False for row in rows.values()), + "mean_lift": sum(row["lift"] for row in scored.values()) / len(scored), + "scored": len(scored), + "unscorable": len(rows) - len(scored), + **shared, + **scoring, + } + + +def _prior_rows(path: Path, skill: str, shared: dict, scoring: dict, + skill_sha256: str) -> dict[str, dict]: + if not path.is_file(): + return {} + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError) as error: + raise ValueError(f"incompatible existing rescore output: {path}") from error + expected = {"skill": skill, **shared, **scoring} + if (not isinstance(payload, dict) + or any(payload.get(key) != value for key, value in expected.items()) + or not isinstance(payload.get("combinations"), dict)): + raise ValueError(f"incompatible existing rescore output: {path}") + rows = payload["combinations"] + if any(not isinstance(key, str) or not isinstance(row, dict) for key, row in rows.items()): + raise ValueError(f"incompatible existing rescore output: {path}") + return {key: row for key, row in rows.items() if row.get("skill_sha256") == skill_sha256} + + +def _matching_row_key(identity: dict, rows: dict[str, dict]) -> str | None: + fields = ("combination", "harness", "model", "target_alias", "endpoint_fingerprint", + "protocol", "task_fingerprint", "skill_sha256", "attempts", + "gateway_revision", "gateway_identity", "family", "parameter_billions", + "quantization", "tool_parser") + for key, row in rows.items(): + if all(row.get(field) == identity.get(field) for field in fields): + return key + return None + + +def _same_logical_cell(left: dict, right: dict) -> bool: + fields = ("combination", "harness", "target_alias", "endpoint_fingerprint", "protocol", + "task_fingerprint", "skill_sha256", "attempts") + return all(left.get(field) == right.get(field) for field in fields) + + +def _prefer_current_native_evidence(evidence: list[tuple[Path, dict]]) -> list[tuple[Path, dict]]: + current = [identity for _, identity in evidence + if str(identity.get("gateway_revision") or "").endswith( + f"-{NATIVE_RUNNER_REVISION}")] + if not current: + return evidence + return [ + item for item in evidence if ( + str(item[1].get("gateway_revision") or "").endswith(f"-{NATIVE_RUNNER_REVISION}") + or not any(_same_logical_cell(item[1], identity) for identity in current) + ) + ] + + +def rescore(skill: str, jobs_roots: Sequence[Path] | None = None, + legacy_metadata: Path | None = None, + provenance_metadata: Sequence[Path] | None = None, + combination_paths: Sequence[Path] | None = None, + output: Path | None = None, scoring_identity: dict | None = None, log=print) -> dict: + """Recompute compatible combinations and atomically publish their matrix when measured.""" + if os.environ.get("JUDGE_BACKEND", "").strip().lower() != "agy": + raise RuntimeError("Harbor rescore requires JUDGE_BACKEND=agy; refusing scorer fallback") + selected_mode = combination_paths is not None + roots = list(jobs_roots) if jobs_roots is not None else [HARBOR_DIR / "jobs" / skill] + provenance_paths = list(provenance_metadata or []) + if provenance_paths and len(provenance_paths) != len(roots): + raise ValueError("each jobs root requires one paired provenance metadata document") + provenance_by_root = { + root.resolve(): load_provenance_metadata(path) + for root, path in zip(roots, provenance_paths) + } + _, holdout, _ = load_tasks(skill) + manifest = _load_legacy_manifest(legacy_metadata) if legacy_metadata else None + discovered = (select_combinations(roots, combination_paths or []) + if selected_mode else discover_combinations(roots)) + evidence = _prefer_current_native_evidence([ + (combo, _evidence_metadata(skill, combo.parent, combo, manifest)) + for combo in discovered + ]) + records = [identity for _, identity in evidence] + validate_compatibility(records) + _validate_current_compatibility(records, holdout) + skill_sha256 = "" + if selected_mode: + skill_revisions = {identity.get("skill_sha256") for identity in records} + if (len(skill_revisions) != 1 or not isinstance(next(iter(skill_revisions)), str) + or not next(iter(skill_revisions)).strip()): + raise ValueError("selected combinations need one non-empty skill_sha256") + skill_sha256 = next(iter(skill_revisions)) + for combo, identity in evidence: + if identity.get("canary_error") or identity.get("measurement_error"): + continue + for arm in ("skill", "control"): + job, arm_identity = _arm_evidence(combo, arm) + if not job.is_dir(): + raise RuntimeError(f"selected combination is incomplete (missing {arm} arm): {combo}") + _validate_attempt_artifacts(job, skill, holdout, identity["attempts"], arm_identity) + for combo, identity in evidence: + if identity.get("canary_error") or identity.get("measurement_error"): + continue + receipt_metadata = { + **_combo_identity(combo), + **provenance_by_root.get(combo.parent.resolve(), {}), + } + for arm in ("skill", "control"): + job, arm_identity = _arm_evidence(combo, arm) + if job.is_dir(): + validate_job_receipts(job, {**receipt_metadata, "arm": arm}, + identity=arm_identity) + scoring = dict(scoring_identity or current_scoring_identity()) + shared = {field: evidence[0][1][field] for field in _EVIDENCE_FIELDS} + out = output or HARBOR_DIR / f"{skill}.rescored.json" + + rows = (_prior_rows(out, skill, shared, scoring, skill_sha256) if selected_mode else {}) + last_checkpoint = None + for combo, identity in evidence: + matching = _matching_row_key(identity, rows) + for prior_key, prior_row in list(rows.items()): + if prior_key != matching and _same_logical_cell(identity, prior_row): + del rows[prior_key] + key = matching or _row_key(identity, rows) + # Selected progressive runs carry earlier paths so pre-measurement failures can enter the + # first published matrix. Compatible prior lifts are durable grade receipts, not an order + # to pay Agy again or replace a stochastic score. + output_identity = _output_identity(identity, scoring) + if identity.get("canary_error") or identity.get("measurement_error"): + error = str(identity.get("canary_error") or identity["measurement_error"])[:300] + rows[key] = {**output_identity, "error": error} + log(f"[rescore] {key:<44} {error}") + continue + if "lift" in rows.get(key, {}): + continue + skill_arm, skill_identity = _arm_evidence(combo, "skill") + control_arm, control_identity = _arm_evidence(combo, "control") + if not (skill_arm.is_dir() and control_arm.is_dir()): + error = str(identity.get("canary_error") or identity.get("measurement_error") + or "incomplete (missing an arm)")[:300] + if "lift" not in rows.get(key, {}): + rows[key] = {**output_identity, "error": error} + log(f"[rescore] {key:<44} {error}") + continue + try: + _validate_attempt_artifacts(skill_arm, skill, holdout, identity["attempts"], + skill_identity) + _validate_attempt_artifacts(control_arm, skill, holdout, identity["attempts"], + control_identity) + skipped = (broken_tasks(skill_arm, skill_identity) + | broken_tasks(control_arm, control_identity)) + arms = { + "skill": score(collect_answers( + skill_arm, broken_trials(skill_arm, skill_identity), identity=skill_identity), + skill, holdout, skipped, _AGY_SCORE_CONCURRENCY), + "control": score(collect_answers( + control_arm, broken_trials(control_arm, control_identity), identity=control_identity), + skill, holdout, skipped, _AGY_SCORE_CONCURRENCY), + } + except RuntimeError as error: + if "lift" not in rows.get(key, {}): + rows[key] = {**output_identity, "error": str(error)[:300]} + log(f"[rescore] {key:<44} UNSCORABLE: {str(error)[:90]}") + continue + skill_mean = sum(arms["skill"]) / len(arms["skill"]) + control_mean = sum(arms["control"]) / len(arms["control"]) + lift = skill_mean - control_mean + verdict = "helps" if lift > 0.05 else "no lift" if lift >= -0.05 else "HURTS" + note = f" [{len(skipped)} dropped]" if skipped else "" + log(f"[rescore] {key:<44} skill {skill_mean:.3f} control {control_mean:.3f} " + f"lift {lift:+.3f} ({verdict}) n={len(arms['skill'])}{note}") + rows[key] = { + **output_identity, + "skill_mean": skill_mean, + "control_mean": control_mean, + "lift": lift, + "skill_scores": arms["skill"], + "control_scores": arms["control"], + "tasks_scored": len(arms["skill"]), + "tasks_dropped": sorted(skipped), + } + checkpoint = _summary(skill, rows, shared, scoring) + if checkpoint is not None: + atomic_write_json(out, checkpoint) + last_checkpoint = checkpoint + + summary = _summary(skill, rows, shared, scoring) + if summary is None: + log("[rescore] nothing scorable; no matrix published") + raise SystemExit(f"[rescore] no combination for '{skill}' was measured; nothing was measured.") + if summary != last_checkpoint: + atomic_write_json(out, summary) + log(f"[rescore] {summary['scored']} scorable of {len(rows)}; " + f"mean lift {summary['mean_lift']:+.4f}") + log(f"[rescore] written to {out}") + return summary + + +if __name__ == "__main__": + import argparse + + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("skill") + parser.add_argument("--jobs", action="append", default=None, metavar="PATH", + help="job root (repeatable; default runs/harbor/jobs/)") + parser.add_argument("--legacy-metadata", type=Path, default=None, metavar="PATH", + help="authoritative metadata manifest for preserved proprietary jobs") + parser.add_argument("--provenance-metadata", action="append", type=Path, default=None, + metavar="PATH", + help="provenance migration JSON paired with each repeated --jobs root") + parser.add_argument("--combination-path", action="append", type=Path, default=None, + metavar="PATH", help="exact completed combination directory (repeatable)") + parser.add_argument("--output", type=Path, default=None, metavar="PATH", + help="atomic matrix destination (default runs/harbor/.rescored.json)") + args = parser.parse_args() + rescore( + args.skill, [Path(root) for root in args.jobs] if args.jobs else None, + legacy_metadata=args.legacy_metadata, + provenance_metadata=args.provenance_metadata, + combination_paths=args.combination_path, + output=args.output, + ) diff --git a/ingot/optimize/harbor_targets.py b/ingot/optimize/harbor_targets.py new file mode 100644 index 0000000..3c8f857 --- /dev/null +++ b/ingot/optimize/harbor_targets.py @@ -0,0 +1,425 @@ +"""Identity and safe routing helpers for Harbor's local model targets. + +Endpoint addresses are runtime inputs. They are retained only on the immutable target used to make +the request; the stable job identity uses a fingerprint of the normalized address and served model. +""" +from __future__ import annotations + +import hashlib +import json +import os +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass +from typing import Mapping + + +MIN_CONTEXT_LENGTH = 32_768 +PROTOCOLS = frozenset({"chat", "responses", "messages"}) +PROTOCOL_PATHS = { + "chat": "/v1/chat/completions", + "responses": "/v1/responses", + "messages": "/v1/messages", +} + +# This is the complete local matrix allowlist. Keep it in one mapping so routing cannot silently +# fall back to a provider's default protocol when a new Harbor adapter is added. +HARNESS_PROTOCOLS = { + "claude-code": "messages", + "terminus-2": "chat", + "goose": "chat", + "opencode": "chat", + "openclaw": "chat", + "mini-swe-agent": "chat", + "codex": "responses", + "aider": "chat", + "pi": "chat", +} + +# Alias configuration pins served identity and display/scale provenance. Context length alone comes +# from discovery; parse_target uses the minimum accepted value until that live check completes. +TARGETS = { + "dell-qwen": {"display_name": "Qwen/Qwen3.6-27B", "served_model": "dot-backbone", + "family": "Qwen3.6", "parameter_billions": 27.0, + "quantization": "fp8-published", "tool_parser": "qwen3_xml"}, + # Alias names avoid punctuation; the server's exact model ID retains the decimal point. + "qwen35-08b": {"display_name": "Qwen/Qwen3.5-0.8B", "served_model": "qwen35-0.8b", + "family": "Qwen3.5", "parameter_billions": 0.8, + "quantization": "fp8-load", "tool_parser": "qwen3_coder"}, + "qwen35-2b": {"display_name": "Qwen/Qwen3.5-2B", "served_model": "qwen35-2b", + "family": "Qwen3.5", "parameter_billions": 2.0, + "quantization": "fp8-load", "tool_parser": "qwen3_coder"}, + "qwen35-4b": {"display_name": "Qwen/Qwen3.5-4B", "served_model": "qwen35-4b", + "family": "Qwen3.5", "parameter_billions": 4.0, + "quantization": "fp8-load", "tool_parser": "qwen3_coder"}, + "qwen35-9b": {"display_name": "Qwen/Qwen3.5-9B", "served_model": "qwen35-9b", + "family": "Qwen3.5", "parameter_billions": 9.0, + "quantization": "fp8-load", "tool_parser": "qwen3_coder"}, + "orin-qwen35-9b": {"display_name": "Qwen/Qwen3.5-9B (Q4_K_M)", + "served_model": "qwen3.5:9b", "family": "Qwen3.5", + "parameter_billions": 9.7, "quantization": "Q4_K_M", + "tool_parser": "ollama", "context_api": "ollama-ps"}, + "spark-deepseek": {"display_name": "deepseek-ai/DeepSeek-V4-Flash-0731 (NVFP4)", "served_model": "deepseek-v4-flash", + "family": "DeepSeek V4 Flash", "parameter_billions": None, + "quantization": "nvfp4", "tool_parser": "deepseek_v3"}, + "orin-abliterated": {"display_name": "ablit35b (34.66B, Q4_K_M)", "served_model": "ablit35b", + "family": "Abliterated 35B", "parameter_billions": 35.0, + "quantization": "gguf", "tool_parser": "llama.cpp"}, +} + +_PROVIDER_PREFIXES = ( + "ANTHROPIC_", + "CLAUDE_", + "OPENAI_", + "OPENROUTER_", + "CODEX_", + "LITELLM_", + "GEMINI_", + "GOOSE_", + "AIDER_", + "OPENCODE_", + "OPENCLAW_", + "MINI_SWE_AGENT_", + "PI_", +) +_EXACT_SECRET_KEYS = frozenset({"API_KEY", "MODEL_API_KEY"}) + + +def _canonical_url(base_url: str) -> str: + value = str(base_url).strip() + if value.endswith("/"): + value = value[:-1] + parsed = urllib.parse.urlsplit(value) + if parsed.scheme not in {"http", "https"} or not parsed.netloc: + raise ValueError("base URL must be an http(s) URL") + if parsed.username or parsed.password: + raise ValueError("base URL must not contain credentials") + if parsed.query or parsed.fragment: + raise ValueError("base URL must not contain a query or fragment") + return value + + +def _alias_config(alias: str) -> dict[str, str]: + try: + return TARGETS[alias] + except KeyError as exc: + raise ValueError(f"unknown local target alias: {alias}") from exc + + +@dataclass(frozen=True) +class LocalTarget: + alias: str + display_name: str + base_url: str + served_model: str + context_length: int + protocols: frozenset[str] + family: str = "" + parameter_billions: float | None = None + quantization: str = "" + tool_parser: str = "" + + @property + def fingerprint(self) -> str: + payload = json.dumps( + {"base_url": _canonical_url(self.base_url), "served_model": self.served_model}, + sort_keys=True, + separators=(",", ":"), + ).encode() + return hashlib.sha256(payload).hexdigest()[:12] + + @property + def job_slug(self) -> str: + return f"{self.alias}-{self.fingerprint}" + + +def parse_target(spec: str) -> LocalTarget: + """Parse ``alias=base_url`` into a provisional configured target.""" + alias, separator, base_url = str(spec).partition("=") + if not separator or not alias.strip() or not base_url.strip(): + raise ValueError("target must be specified as ALIAS=BASE_URL") + config = _alias_config(alias.strip()) + return LocalTarget( + alias=alias.strip(), + display_name=config["display_name"], + base_url=_canonical_url(base_url), + served_model=config["served_model"], + context_length=MIN_CONTEXT_LENGTH, + protocols=PROTOCOLS, + family=config["family"], + parameter_billions=config["parameter_billions"], + quantization=config["quantization"], + tool_parser=config["tool_parser"], + ) + + +def _ollama_runtime_context(base_url: str, served_model: str, timeout: float) -> object: + """Read the loaded Ollama runtime context when its OpenAI model list omits that field.""" + request = urllib.request.Request(f"{base_url}/api/ps", method="GET") + try: + with urllib.request.urlopen(request, timeout=timeout) as response: + status = getattr(response, "status", None) + if status is not None and not 200 <= int(status) < 300: + raise RuntimeError(f"Ollama process list returned HTTP {status}") + payload = json.loads(response.read()) + except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, ValueError) as exc: + raise RuntimeError(f"Ollama process list failed: {exc}") from exc + models = payload.get("models") if isinstance(payload, dict) else None + if not isinstance(models, list): + return None + loaded = next((item for item in models + if isinstance(item, dict) and item.get("name") == served_model), None) + return loaded.get("context_length") if loaded else None + + +def discover_target(alias: str, base_url: str, *, timeout: float = 10.0) -> LocalTarget: + """Fetch ``/v1/models`` and return a target after identity/context validation.""" + config = _alias_config(alias) + canonical = _canonical_url(base_url) + request = urllib.request.Request(f"{canonical}/v1/models", method="GET") + try: + with urllib.request.urlopen(request, timeout=timeout) as response: + status = getattr(response, "status", None) + if status is not None and int(status) >= 400: + raise RuntimeError(f"model discovery returned HTTP {status}") + payload = json.loads(response.read()) + except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, ValueError) as exc: + if isinstance(exc, ValueError) and str(exc).startswith("model discovery"): + raise + raise RuntimeError(f"model discovery failed for {alias}: {exc}") from exc + if not isinstance(payload, dict) or not isinstance(payload.get("data"), list): + raise RuntimeError("model discovery returned an invalid response object") + model = next((entry for entry in payload["data"] + if isinstance(entry, dict) and entry.get("id") == config["served_model"]), None) + if model is None: + ids = [entry.get("id") for entry in payload["data"] if isinstance(entry, dict)] + raise ValueError(f"served model {config['served_model']!r} not found; received {ids!r}") + context = model.get("max_model_len", model.get("context_length")) + if context is None and isinstance(model.get("meta"), dict): + context = model["meta"].get("n_ctx") + if context is None and config.get("context_api") == "ollama-ps": + context = _ollama_runtime_context(canonical, config["served_model"], timeout) + if not isinstance(context, int) or isinstance(context, bool): + raise ValueError(f"served model {config['served_model']!r} has no context length") + if context < MIN_CONTEXT_LENGTH: + raise ValueError(f"served model {config['served_model']!r} context length {context} is below " + f"the minimum {MIN_CONTEXT_LENGTH}") + return LocalTarget( + alias=alias, + display_name=config["display_name"], + base_url=canonical, + served_model=config["served_model"], + context_length=context, + protocols=PROTOCOLS, + family=config["family"], + parameter_billions=config["parameter_billions"], + quantization=config["quantization"], + tool_parser=config["tool_parser"], + ) + + +def protocol_for(harness: str) -> str: + try: + return HARNESS_PROTOCOLS[harness] + except KeyError as exc: + raise ValueError(f"unsupported harness for local target: {harness}") from exc + + +def harbor_model(target: LocalTarget, harness: str) -> str: + protocol_for(harness) + # Harbor's local CLI adapters use provider/model identifiers; Claude Code consumes the + # Anthropic-compatible model ID directly. + if harness == "claude-code": + return target.served_model + if harness == "opencode": + return f"local/{target.served_model}" + return f"openai/{target.served_model}" + + +def harbor_agent_kwargs(target: LocalTarget, harness: str) -> dict[str, str]: + """Return only adapter kwargs that cannot be supplied through the child environment.""" + protocol_for(harness) + if harness == "terminus-2": + return {"api_base": f"{_canonical_url(target.base_url)}/v1"} + if harness == "openclaw": + # Harbor defaults this adapter to "high", which both local models reject. + return {"thinking": "off"} + if harness == "opencode": + model = target.served_model + # OpenCode otherwise requests 32k output tokens. Reserve three quarters of the + # discovered context for its prompt, tool calls, and accumulated transcript. + output_limit = target.context_length // 4 + return {"opencode_config": {"provider": {"local": { + "npm": "@ai-sdk/openai-compatible", + "options": {"baseURL": f"{_canonical_url(target.base_url)}/v1", "apiKey": "local"}, + "models": {model: {"limit": {"context": target.context_length, + "output": output_limit}}}, + }}}} + return {} + + +def scrub_provider_env(parent: Mapping[str, str]) -> dict[str, str]: + """Copy a parent environment without provider credentials or routing controls.""" + return {key: value for key, value in parent.items() + if not key.startswith(_PROVIDER_PREFIXES) and key not in _EXACT_SECRET_KEYS} + + +def local_agent_env(target: LocalTarget, harness: str) -> dict[str, str]: + """Build a complete child environment with a literal local credential sentinel.""" + protocol = protocol_for(harness) + # This mapping is repeated as Harbor --ae values. Do not copy ambient process + # variables into it: only local sentinels and routing settings belong here. + env: dict[str, str] = {} + base_url = _canonical_url(target.base_url) + openai_base_url = f"{base_url}/v1" + # Keep every provider key at the non-secret sentinel. Adapters choose their own protocol + # below, but a generic adapter must never discover a real inherited key and fall back to it. + env.update({ + "ANTHROPIC_API_KEY": "local", + "OPENAI_API_KEY": "local", + "CODEX_API_KEY": "local", + }) + if protocol == "messages": + env.update({ + "ANTHROPIC_BASE_URL": base_url, + "ANTHROPIC_MODEL": target.served_model, + }) + elif protocol == "responses": + env.update({ + "OPENAI_BASE_URL": openai_base_url, + "OPENAI_API_BASE": openai_base_url, + "OPENAI_HOST": base_url, + }) + else: + env.update({ + "OPENAI_BASE_URL": openai_base_url, + "OPENAI_API_BASE": openai_base_url, + # Goose reads OPENAI_HOST rather than the OpenAI SDK spelling. + "OPENAI_HOST": base_url, + }) + return env + + +def probe_protocol(target: LocalTarget, protocol: str, *, timeout: float = 20.0) -> None: + """Send a one-token request and require a successful, non-empty JSON object response.""" + if protocol not in PROTOCOL_PATHS: + raise ValueError(f"unsupported protocol: {protocol}") + if protocol not in target.protocols: + raise ValueError(f"target does not support protocol: {protocol}") + body: dict[str, object] + headers = {"Content-Type": "application/json", "Authorization": "Bearer local"} + if protocol == "messages": + body = {"model": target.served_model, "max_tokens": 1, + "messages": [{"role": "user", "content": "ping"}]} + headers["anthropic-version"] = "2023-06-01" + headers["x-api-key"] = "local" + elif protocol == "responses": + body = {"model": target.served_model, "input": "ping", "max_output_tokens": 1} + else: + body = {"model": target.served_model, "messages": [{"role": "user", "content": "ping"}], + "max_tokens": 1} + if target.family.startswith("Qwen"): + body["messages"] = [{"role": "user", "content": "ping /no_think"}] + request = urllib.request.Request( + f"{_canonical_url(target.base_url)}{PROTOCOL_PATHS[protocol]}", + data=json.dumps(body).encode(), + headers=headers, + method="POST", + ) + try: + with urllib.request.urlopen(request, timeout=timeout) as response: + status = getattr(response, "status", None) + if status is not None and not 200 <= int(status) < 300: + raise RuntimeError(f"{protocol} probe returned HTTP {status}") + payload = json.loads(response.read()) + except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, ValueError) as exc: + if isinstance(exc, ValueError) and str(exc).startswith(protocol + " probe returned"): + raise + raise RuntimeError(f"{protocol} probe failed: {exc}") from exc + if not isinstance(payload, dict) or not payload: + raise RuntimeError(f"{protocol} probe returned an empty response object; expected non-empty") + + +def probe_chat_tool_round_trip(target: LocalTarget, *, timeout: float = 20.0) -> None: + """Require one parsed function call and a successful tool-result continuation.""" + url = f"{_canonical_url(target.base_url)}{PROTOCOL_PATHS['chat']}" + headers = {"Content-Type": "application/json", "Authorization": "Bearer local"} + tool = { + "type": "function", + "function": { + "name": "ingot_echo", + "description": "Return the supplied string unchanged.", + "parameters": { + "type": "object", + "properties": {"value": {"type": "string"}}, + "required": ["value"], + }, + }, + } + + def post(body: dict[str, object]) -> dict: + request = urllib.request.Request( + url, data=json.dumps(body).encode(), headers=headers, method="POST") + try: + with urllib.request.urlopen(request, timeout=timeout) as response: + status = getattr(response, "status", None) + if status is not None and not 200 <= int(status) < 300: + raise RuntimeError(f"chat tool probe returned HTTP {status}") + payload = json.loads(response.read()) + except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError, OSError, + ValueError) as exc: + raise RuntimeError(f"chat tool probe failed: {exc}") from exc + if not isinstance(payload, dict): + raise RuntimeError("chat tool probe returned an invalid response object") + return payload + + common = { + "model": target.served_model, + "chat_template_kwargs": {"enable_thinking": False}, + "reasoning_effort": "none", + "temperature": 0, + } + first = post({ + **common, + "messages": [ + {"role": "system", "content": "Call ingot_echo with value cutover-ok. /no_think"}, + {"role": "user", "content": "Use the tool now."}, + ], + "tools": [tool], + "tool_choice": "required", + "max_tokens": 128, + }) + try: + assistant = first["choices"][0]["message"] + call = assistant["tool_calls"][0] + arguments = json.loads(call["function"]["arguments"]) + if call["function"]["name"] != "ingot_echo" or arguments != {"value": "cutover-ok"}: + raise ValueError + call_id = call["id"] + except (KeyError, IndexError, TypeError, ValueError, json.JSONDecodeError) as exc: + raise RuntimeError("chat tool probe did not return the required parsed tool call") from exc + + assistant_turn = { + "role": "assistant", + "content": assistant.get("content") or "", + "tool_calls": assistant["tool_calls"], + } + second = post({ + **common, + "messages": [ + {"role": "system", "content": + "After the tool result, reply exactly cutover-ok. /no_think"}, + {"role": "user", "content": "Use the tool now."}, + assistant_turn, + {"role": "tool", "tool_call_id": call_id, "content": "cutover-ok"}, + ], + "tools": [tool], + "max_tokens": 64, + }) + try: + content = second["choices"][0]["message"]["content"] + except (KeyError, IndexError, TypeError) as exc: + raise RuntimeError("chat tool probe returned an invalid continuation") from exc + if str(content).strip() != "cutover-ok": + raise RuntimeError("chat tool probe did not consume the tool result") diff --git a/ingot/optimize/ingress.py b/ingot/optimize/ingress.py new file mode 100644 index 0000000..44934ca --- /dev/null +++ b/ingot/optimize/ingress.py @@ -0,0 +1,309 @@ +"""Quarantine a vetted new skill for human review without adding it to the served library.""" +from __future__ import annotations + +import hashlib +import json +import logging +import os +import shutil +import threading +import time +import uuid +from pathlib import Path + +from ingot.mcp_server.registry import (SLUG_RE, load_skills, normalized_frontmatter, skill_revision, + writable_skill_dir) +from ingot.optimize import promote, tree +from ingot.optimize.evidence import recorded_path +from ingot import paths + + + +def evidence_dir() -> Path: + return paths.runs() / "evidence" + + +def audit_file() -> Path: + return paths.runs() / "ingress-audit.jsonl" + +MAX_FILES = 32 +MAX_TOTAL_CHARS = 1_000_000 +MAX_FIELD_CHARS = 4_000 +MAX_EVIDENCE_ITEMS = 12 +TEXT_SUFFIXES = {".md", ".txt", ".py", ".sh", ".js", ".ts", ".json", ".yaml", ".yml", + ".toml", ".cfg"} +logger = logging.getLogger(__name__) +_SUBMIT_LOCK = threading.Lock() + + +def _text(name: str, value: object, *, limit: int = MAX_FIELD_CHARS) -> str: + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{name} is required") + value = value.strip() + if len(value) > limit: + raise ValueError(f"{name} exceeds {limit} characters") + return value + + +def _files(value: object) -> dict[str, str]: + if not isinstance(value, dict) or len(value) > MAX_FILES: + raise ValueError(f"files must be an object with at most {MAX_FILES} entries") + result = {} + portable_paths = set() + for raw_path, content in value.items(): + if not isinstance(raw_path, str) or not isinstance(content, str): + raise ValueError("file paths and contents must be strings") + path = tree.portable_path(raw_path) + folded = path.as_posix().casefold() + if folded in portable_paths: + raise ValueError(f"component path collides case-insensitively: {raw_path}") + portable_paths.add(folded) + if path.suffix.lower() not in TEXT_SUFFIXES: + raise ValueError(f"unsupported skill file type: {raw_path}") + result[f"file:{path.as_posix()}"] = content + if sum(len(item) for item in result.values()) > MAX_TOTAL_CHARS: + raise ValueError(f"files exceed {MAX_TOTAL_CHARS} total characters") + return result + + +def _proposal_id(identity: dict) -> str: + canonical = json.dumps(identity, sort_keys=True, separators=(",", ":"), ensure_ascii=False) + return hashlib.sha256(canonical.encode()).hexdigest()[:24] + + +def _write_evidence(skill: str, proposal: dict, gate: dict) -> dict[str, str]: + """The bundle a reviewer reads. + + Narrative sections are optional. An agent proposing a skill it authored can state a pressure + scenario and a verification it ran; an operator ingesting a third-party package cannot, and + inventing one on their behalf would put a fabricated claim in front of the person whose job is + to check claims. Absent sections are omitted rather than filled in.""" + root = evidence_dir() / skill / f"creation-{proposal['proposal_id']}" + root.mkdir(parents=True, exist_ok=True) + bundle = {"schema_version": "ingot/skill-create/v1", "skill": skill, + "created": proposal["created"], "proposal": proposal, "gate": gate} + verification = proposal.get("verification") or {} + markdown = "\n".join([ + f"# New skill proposal: {skill}", "", f"**Source:** `{proposal['source']}`", "", + proposal["summary"], "", "## Evidence", "", + *[f"- {item}" for item in proposal["evidence"]], "", + *(["## Pressure scenario", "", proposal["pressure_scenario"], ""] + if proposal.get("pressure_scenario") else []), + *(["## Verification", "", + f"- Command: `{verification.get('command', '')}`", + f"- Result: {verification.get('result', '')}", ""] if verification else []), + "Human approval is required before this skill enters the served library.", "", + ]) + paths = ((root / "evidence.json", json.dumps(bundle, indent=2) + "\n"), + (root / "EVIDENCE.md", markdown)) + for path, content in paths: + temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp") + temporary.write_text(content, encoding="utf-8") + temporary.replace(path) + return {"json": recorded_path(paths[0][0]), "markdown": recorded_path(paths[1][0])} + + +def _publish(skill: str, record: dict) -> bool: + promote.pending_dir().mkdir(parents=True, exist_ok=True) + destination = promote.pending_path(skill) + temporary = destination.with_name(f".{destination.name}.{uuid.uuid4().hex}.tmp") + payload = (json.dumps(record, indent=2) + "\n").encode() + fd = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + try: + view = memoryview(payload) + while view: + written = os.write(fd, view) + if written <= 0: + raise OSError("pending proposal write made no progress") + view = view[written:] + os.fsync(fd) + finally: + os.close(fd) + try: + os.link(temporary, destination) + except FileExistsError as exc: + existing = promote.load_pending(skill) + existing_id = (existing.get("creation") or {}).get("proposal_id") if existing else None + if existing_id == record["creation"]["proposal_id"]: + return False + raise ValueError(f"review slot is occupied for '{skill}'") from exc + finally: + temporary.unlink(missing_ok=True) + return True + + +def _audit(record: dict) -> None: + audit_file().parent.mkdir(parents=True, exist_ok=True) + fd = os.open(audit_file(), os.O_APPEND | os.O_CREAT | os.O_WRONLY, 0o600) + try: + view = memoryview((json.dumps(record, separators=(",", ":")) + "\n").encode()) + while view: + written = os.write(fd, view) + if written <= 0: + raise OSError("ingress audit write made no progress") + view = view[written:] + os.fsync(fd) + finally: + os.close(fd) + + +def _accept_slot(skill: str) -> Path: + """The skill must be new. Returns the directory it would occupy. + + Shared by every admission front door, so a package cannot enter through one path what another + would refuse.""" + if not isinstance(skill, str) or len(skill) > 80 or not SLUG_RE.fullmatch(skill): + raise ValueError(f"invalid skill name: {skill!r}") + target = writable_skill_dir(skill) + if target.exists() or target.is_symlink() or any(item.name == skill for item in load_skills()): + raise ValueError(f"skill '{skill}' already exists; propose an update instead") + return target + + +def build_components(skill: str, description: str, body: str, files: dict[str, str], + frontmatter: dict) -> tuple[dict[str, str], dict]: + """The component map a candidate is made of, plus its normalized frontmatter. + + One implementation for every source: path validation, portability, and the field limits are + properties of what Ingot will serve, not of who proposed it. Public because an ingest adapter + has to know the candidate revision -- which is a property of these components, not of the + directory they were read from -- before it can build a candidate manifest.""" + description = " ".join(_text("description", description, limit=2_000).split()) + body = _text("body", body, limit=200_000) + metadata = normalized_frontmatter(skill, description, frontmatter) + canonical_frontmatter = json.dumps(metadata, sort_keys=True, separators=(",", ":"), + ensure_ascii=False) + if len(canonical_frontmatter) > 20_000: + raise ValueError("frontmatter exceeds 20000 characters") + return ({"description": description, "body": body, + "frontmatter": canonical_frontmatter, **_files(files)}, metadata) + + +def _quarantine(skill: str, record: dict, proposal: dict, gate: dict) -> dict: + """Claim the review slot, write the evidence bundle, publish the pending record, audit. + + The only path that creates a pending proposal. `_publish` links the record into place, which is + atomic across processes, so two submitters racing for one slot cannot both win -- the in-process + lock below narrows the window but is not what makes it safe.""" + with _SUBMIT_LOCK: + proposal_id = proposal["proposal_id"] + existing = promote.load_pending(skill) + if existing: + existing_id = (existing.get("creation") or {}).get("proposal_id") + if existing_id == proposal_id: + return {"status": "duplicate", "skill": skill, "proposal_id": proposal_id, + "promotable": True} + raise ValueError(f"review slot is occupied for '{skill}'") + record["evidence_paths"] = _write_evidence(skill, proposal, gate) + evidence_root = evidence_dir() / skill / f"creation-{proposal_id}" + try: + published = _publish(skill, record) + except Exception: + shutil.rmtree(evidence_root, ignore_errors=True) + raise + if not published: + return {"status": "duplicate", "skill": skill, "proposal_id": proposal_id, + "promotable": True} + try: + _audit({"schema_version": 1, "ts": int(time.time()), "action": "quarantine", + "skill": skill, "proposal_id": proposal_id, + "challenger_revision": proposal["revision"], + "producer": proposal["producer"]}) + except Exception: + logger.warning("Quarantined creation proposal %s, but its audit write failed", + proposal_id, exc_info=True) + return {"status": "quarantined", "skill": skill, "proposal_id": proposal_id, + "promotable": True} + + +def submit_package_ingest(*, skill: str, components: dict[str, str], metadata: dict, + revision: str, source: str, candidate: dict, identity: str, + review_summary: list[str], producer: str, caller: str, + candidate_tree: dict) -> dict: + """Quarantine a package an operator ingested from somewhere else. + + Distinct from `submit_skill_create` in what it is allowed to claim, not in what it does. An + agent proposing a skill it wrote can attest to a pressure scenario and a verification run; an + operator pointing at a third-party directory can attest only to where it came from and what the + deterministic review found. Both produce the same record, take the same slot, and are equally + inert until a human approves them. + + Also distinct in what it carries. This path has real bytes on disk, so the candidate is a + staged tree covering every file in the package; `submit_skill_create` receives strings over + MCP and has no bytes to preserve.""" + _accept_slot(skill) + tree.verify_manifest(candidate_tree) + + proposal = { + "skill": skill, + "revision": revision, + "source": _text("source", source), + "summary": f"Ingested package '{skill}' from {candidate['source']['type']}", + # Real, machine-produced findings. Nothing here is a claim a person made. + "evidence": review_summary or ["Deterministic review found no findings."], + "created": candidate["created_at"], + "producer": _text("producer", producer), + "caller": _text("caller", caller), + "candidate": candidate, + "frontmatter": metadata, + # Identity is the candidate's, so the same bytes ingested twice are one proposal even + # though the manifest's timestamp differs. + "proposal_id": _proposal_id({"candidate_identity": identity}), + } + gate = {"promotable": True, "blocked": [], + "warnings": ["Ingested package; no behavioural evidence and no held-out A/B exists.", + *review_summary], + "kind": "package_ingest"} + record = {"skill": skill, "kind": "creation", "created": proposal["created"], + "changed_components": list(components), "champion_components": {}, + "challenger_components": components, "tree": candidate_tree, "gate": gate, + "evidence": {"schema_version": "ingot/skill-create/v1", + "challenger": {"revision": revision}, "gate": gate}, + "creation": proposal} + return _quarantine(skill, record, proposal, gate) + + +def submit_skill_create(*, skill: str, description: str, body: str, files: dict[str, str], + frontmatter: dict, + summary: str, source: str, producer: str, caller: str, + evidence: list[str], pressure_scenario: str, risk: str, + verification_status: str, verification_command: str, + verification_result: str) -> dict: + """Validate and quarantine one new skill package; never add it to the active registry.""" + target = _accept_slot(skill) + components, metadata = build_components(skill, description, body, files, frontmatter) + description = components["description"] + if not isinstance(verification_status, str) or verification_status.strip().lower() != "passed": + raise ValueError("verification_status must be passed before proposing") + if not isinstance(evidence, list) or not 2 <= len(evidence) <= MAX_EVIDENCE_ITEMS: + raise ValueError(f"evidence must contain 2 to {MAX_EVIDENCE_ITEMS} concrete items") + evidence = [_text("evidence item", item) for item in evidence] + if len(set(evidence)) != len(evidence): + raise ValueError("evidence items must be distinct") + + revision = skill_revision(target, components) + proposal = {"skill": skill, "revision": revision, "source": _text("source", source), + "summary": _text("summary", summary), "evidence": evidence, + "created": int(time.time()), + "producer": _text("producer", producer), "caller": _text("caller", caller), + "pressure_scenario": _text("pressure_scenario", pressure_scenario), + "risk": _text("risk", risk), + "verification": {"status": "passed", + "command": _text("verification_command", verification_command), + "result": _text("verification_result", verification_result)}, + "frontmatter": metadata} + identity = {key: value for key, value in proposal.items() + if key not in {"created", "producer", "caller"}} + proposal_id = _proposal_id(identity) + proposal["proposal_id"] = proposal_id + gate = {"promotable": True, "blocked": [], + "warnings": ["New skill admission only; no active champion or held-out A/B exists."], + "kind": "new_skill_admission"} + record = {"skill": skill, "kind": "creation", "created": proposal["created"], + "changed_components": list(components), "champion_components": {}, + "challenger_components": components, "gate": gate, + "evidence": {"schema_version": "ingot/skill-create/v1", + "challenger": {"revision": revision}, "gate": gate}, + "creation": proposal} + + return _quarantine(skill, record, proposal, gate) diff --git a/ingot/optimize/judge.py b/ingot/optimize/judge.py new file mode 100644 index 0000000..d0b0a58 --- /dev/null +++ b/ingot/optimize/judge.py @@ -0,0 +1,244 @@ +"""LLM judge: scores an answer 0..1 and, following the SkillForge paper's multi-dimensional Failure +Analyzer (Liu et al., "SkillForge", arXiv:2604.08618), classifies each failure across fixed +dimensions so the search gets *categorized* feedback, not one opaque score. The dimension labels +also drive success/failure mining (optimize/mine.py) and the candidate search's diagnosis. + +Judges against a task `rubric` when given one; with no rubric it grades reference-free (used when +mining real traces). If a task supplies a `reference` answer, consistency-against-reference is added +to the prompt (the paper's Consistency-Rate signal, lower variance than a rubric alone).""" +import json +import os +import re +import time + +from langchain_openai import ChatOpenAI + +from . import agy_judge +from . import configured_models +from . import usage as usage_ledger + +# Reward-hacking guard: the judge must NOT be the same model as SKILLOPT_MODEL. +# If the author and the grader share blind spots, the search learns to please the judge instead of +# improving the skill. Default judge is a model distinct from both the reflection LM (GLM) and the +# student (Qwen). +# JUDGE_MODELS (comma-separated) runs an ensemble and averages, harder still to game. Repeating one +# ID is not an ensemble: it silently gives one grader multiple votes and pays for every duplicate. +MODELS = configured_models( + "JUDGE_MODELS", os.environ.get("JUDGE_MODEL", "google/gemini-2.5-flash")) +from . import ZDR_PROVIDER, client_kwargs, skillopt_model, teacher_base_url # noqa: E402 + +if skillopt_model() in MODELS: + print(f"[judge] WARNING: judge model {MODELS} includes the teacher model, which invites " + f"reward-hacking (author == grader). Set JUDGE_MODEL to a different model.", flush=True) + +# Failure dimensions (the general-purpose analogue of the paper's Knowledge/Tool/Clarification/Style). +DIMENSIONS = ["correctness", "completeness", "instruction_following", "efficiency"] + +# The score is the weighted mean of independent checks, never a number the model picks holistically. +# A single 0..1 judgement is one noisy measurement: in practice models emit it on a coarse ladder +# (0.90 / 0.95 / 1.00), so a mean can move by a whole rung because two tasks crossed a boundary, and +# re-running an unchanged arm moves it by that much. k independent checks average toward the truth +# roughly as sqrt(k). This is also what a graded checklist buys commercially -- Tessl grades ~38 +# weighted items per run against our 4. +# +# A task may declare its own `checklist` (see optimize/draft.py); with none, every task is still +# graded on these four rather than on one holistic guess. +DEFAULT_CHECKLIST = [ + {"id": "correctness", "dimension": "correctness", "weight": 3, + "criterion": "The core logic, API usage, and factual claims are right."}, + {"id": "completeness", "dimension": "completeness", "weight": 2, + "criterion": "It covers the whole request, including any edge cases the task or rubric names."}, + {"id": "instruction_following", "dimension": "instruction_following", "weight": 2, + "criterion": "It did what was asked, in the form asked for (e.g. complete runnable code, " + "not a description of code)."}, + {"id": "efficiency", "dimension": "efficiency", "weight": 1, + "criterion": "It is concise, with no padded, repeated, or irrelevant output."}, +] + +# pass / partial / fail rather than a free float: a three-way verdict per item is a judgement a +# model makes reliably, and the resolution comes from having many of them, not from pretending a +# single one is precise to two decimals. +VERDICT_VALUES = {"pass": 1.0, "partial": 0.5, "fail": 0.0} + +_PROMPT = """You are grading an AI assistant's answer to a task against a checklist. + +TASK: {task} +{rubric_block}{reference_block} +ASSISTANT'S ANSWER: +{answer} +{code_block} +Grade EVERY checklist item independently. Judge each item only on what it asks about, and do not +let a good or bad impression of the answer overall carry across items. Treat any OBJECTIVE CODE +CHECK above as ground truth: do not pass a code item whose code is broken or absent. + +CHECKLIST: +{checklist_block} +For each item give a verdict of "pass", "partial", or "fail", plus a note of at most 12 words. +The note is required when the verdict is not "pass" and must say what is wrong, not restate the +criterion. Then write one short paragraph of concrete, actionable feedback on the answer overall. + +Respond with ONLY a JSON object: +{{"items": {{{item_shape}}}, "feedback": ""}}""" + +_llms: dict[str, ChatOpenAI] = {} + + +def _get_llm(model: str): + if model not in _llms: # built once per model, reuses the HTTP pool across many judge calls + _llms[model] = ChatOpenAI(model=model, temperature=0, **client_kwargs(teacher_base_url())) + return _llms[model] + + +# OpenRouter phrasings that mean "your model/provider configuration can never work", retrying +# only burns time, so explain and stop instead. +_PERMANENT = ("no allowed providers", "no providers are available", "not a valid model", + "no endpoints found", "is not available") + + +def _config_error(exc: Exception) -> str | None: + text = str(exc).lower() + if any(marker in text for marker in _PERMANENT): + pins = os.environ.get("OPENROUTER_PROVIDERS", "") + hint = (f" You have OPENROUTER_PROVIDERS={pins}, the pinned provider may not serve this " + f"model, or may not be ZDR-qualified for it; unset the pin or change the model." + if pins else + " No ZDR-qualified endpoint may exist for this model; try another model.") + return f"OpenRouter cannot route this request: {exc}.{hint}" + return None + + +def invoke_retry(llm, messages, tries: int = 3): + """Retry transient provider failures (corrupted responses, 5xx) with a short backoff. + Permanent configuration errors (model/provider mismatch) fail immediately with an explanation + instead of retrying.""" + for i in range(tries): + try: + return llm.invoke(messages) + except Exception as exc: + explained = _config_error(exc) + if explained: + raise SystemExit(explained) from exc + if i == tries - 1: + raise + time.sleep(5 * (i + 1)) + + +def _extract_json(text: str, key: str = "score") -> dict: + """First valid JSON object carrying `key`, robust to prose/braces around the JSON.""" + dec = json.JSONDecoder() + for m in re.finditer(r"\{", text): + try: + obj, _ = dec.raw_decode(text[m.start():]) + except json.JSONDecodeError: + continue + if isinstance(obj, dict) and key in obj: + return obj + return {} + + +def _verdict(raw) -> tuple[float, str]: + """(value, note) from one item's grade. Accepts the strict {verdict, note} shape and the bare + string a model sometimes emits instead. An unrecognized verdict reads as a fail with the raw + text as its note: silently scoring it 1.0 would let a malformed grade inflate the result.""" + note = "" + if isinstance(raw, dict): + note = str(raw.get("note", "")).strip() + raw = raw.get("verdict", "") + word = str(raw).strip().lower() + if word in VERDICT_VALUES: + return VERDICT_VALUES[word], note + return 0.0, note or f"ungraded ({str(raw)[:40]})" + + +def _judge_one(model: str, prompt: str, checklist: list[dict]) -> dict: + if os.environ.get("JUDGE_BACKEND", "").strip().lower() == "agy": + out, usage = agy_judge.invoke(prompt, checklist) + usage_ledger.add("judge", usage, billing_mode="subscription") + raw = json.dumps(out) + else: + msg = invoke_retry(_get_llm(model), prompt) + usage_ledger.add("judge", getattr(msg, "usage_metadata", None)) + raw = msg.content + out = _extract_json(raw, "items") + items = out.get("items") + if not isinstance(items, dict) or not items: + return {"items": {}, "feedback": f"Judge output unparseable: {raw[:200]}", + "unparseable": True} + graded = {} + for item in checklist: + # a missing item is not a pass; the judge was asked for it and did not answer + value, note = _verdict(items.get(item["id"], "")) if item["id"] in items else (0.0, "not graded") + graded[item["id"]] = {"value": value, "note": note} + return {"items": graded, "feedback": str(out.get("feedback", "")), "unparseable": False} + + +def _weighted(graded: dict, checklist: list[dict]) -> float: + total = sum(float(i.get("weight", 1)) for i in checklist) or 1.0 + return sum(float(i.get("weight", 1)) * graded[i["id"]]["value"] + for i in checklist if i["id"] in graded) / total + + +def judge(task: str, rubric: str = "", answer: str = "", reference: str = "", + check: dict | None = None, deliverable: str | None = None, + checklist: list[dict] | None = None) -> dict: + """Return {score, feedback, dimensions, checklist}. + + `score` is the weighted mean of the checklist verdicts, not a number the judge chose. `checklist` + carries the per-item verdicts so a reviewer can see which checks moved rather than only that the + mean did. `dimensions` keeps its old prose shape for the candidate search and trace mining. + + With multiple JUDGE_MODELS this is an ensemble: each item's value is averaged across judges + before weighting, which is smoother than averaging whole-answer scores, and a dimension counts + as failed only when a majority of judges failed an item mapped to it. + + `deliverable` (task yaml) declares the expected answer kind; non-code values skip the static + Python check, see execcheck.judge_note.""" + checklist = [i for i in (checklist or DEFAULT_CHECKLIST) if i.get("id") and i.get("criterion")] + if not checklist: + checklist = DEFAULT_CHECKLIST + rubric_block = f"GRADING RUBRIC: {rubric}\n" if rubric else "" + reference_block = f"KNOWN-GOOD REFERENCE ANSWER (judge consistency against it): {reference}\n" if reference else "" + from . import execcheck # objective code-validity signal to ground the judge + code_note = execcheck.judge_note(answer, task, rubric, check_spec=check, deliverable=deliverable) + code_block = f"\n{code_note}\n" if code_note else "" + checklist_block = "\n".join(f"- {i['id']} (weight {i.get('weight', 1)}): {i['criterion']}" + for i in checklist) + item_shape = ", ".join(f'"{i["id"]}": {{"verdict": "pass|partial|fail", "note": "..."}}' + for i in checklist) + prompt = _PROMPT.format(task=task, answer=answer, rubric_block=rubric_block, + reference_block=reference_block, code_block=code_block, + checklist_block=checklist_block, item_shape=item_shape) + models = ([agy_judge.AGY_IDENTITY] + if os.environ.get("JUDGE_BACKEND", "").strip().lower() == "agy" else MODELS) + results = [_judge_one(m, prompt, checklist) for m in models] + + usable = [r for r in results if not r["unparseable"]] + if not usable: # a parse failure is not a skill failure, but it must not read as a clean pass + return {"score": 0.0, "feedback": results[0]["feedback"], + "dimensions": {d: "pass" for d in DIMENSIONS}, + "checklist": {i["id"]: {"value": 0.0, "note": "judge output unparseable"} + for i in checklist}} + + merged = {i["id"]: {"value": sum(r["items"][i["id"]]["value"] for r in usable) / len(usable), + "note": next((r["items"][i["id"]]["note"] for r in usable + if r["items"][i["id"]]["value"] < 1.0 + and r["items"][i["id"]]["note"]), "")} + for i in checklist} + score = _weighted(merged, checklist) + + # A dimension fails when the items mapped to it did not clean-pass across a majority of judges. + dims = {d: "pass" for d in DIMENSIONS} + for item in checklist: + d = item.get("dimension") + entry = merged[item["id"]] + if d in dims and entry["value"] < 0.5 and dims[d] == "pass": + dims[d] = entry["note"] or f"failed check '{item['id']}'" + feedback = " | ".join(f"[{m.split('/')[-1]}] {r['feedback']}" + for m, r in zip(models, results) if r["feedback"]) \ + if len(results) > 1 else usable[0]["feedback"] + return {"score": score, "feedback": feedback, "dimensions": dims, "checklist": merged} + + +def failed_dimensions(dimensions: dict) -> list[str]: + """Dimension names the judge did NOT mark as a clean pass.""" + return [d for d, v in dimensions.items() if str(v).strip().lower() not in ("pass", "ok", "", "n/a")] diff --git a/ingot/optimize/local_traces.py b/ingot/optimize/local_traces.py new file mode 100644 index 0000000..1193519 --- /dev/null +++ b/ingot/optimize/local_traces.py @@ -0,0 +1,739 @@ +"""Normalize local Claude Code and Codex transcripts into trace roots Ingot can mine. + +Only the human task, final answer, observed skill identity, timing, token counts, and error count +cross this boundary. Reasoning, tool arguments, tool results, attachments, and injected agent +context stay in the source transcript. Historical revisions are recorded only when an Ingot +`route_and_load` result supplied one; the current skill hash is not evidence of what ran before. + +Usage: + python -m ingot.optimize.local_traces +""" +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import re +import tempfile +import time +from collections import Counter +from datetime import date +from functools import lru_cache +from pathlib import Path +from typing import Iterable +from ingot import paths + +SCHEMA = "ingot/local-traces/v1" +# Cursor reuse is valid only while parser semantics are unchanged. Bump this when accepted record +# shapes, attribution, or usage/error extraction changes; source mtimes cannot invalidate code. +PARSER_VERSION = 3 +RUNS_DIR = paths.runs() +LOCAL_TRACE_FILE = Path(os.environ.get( + "LOCAL_TRACE_FILE", RUNS_DIR / "local_traces.json")) +CODEX_DIR = Path(os.environ.get("CODEX_SESSIONS_DIR", Path.home() / ".codex" / "sessions")) +CLAUDE_DIR = Path(os.environ.get("CLAUDE_PROJECTS_DIR", Path.home() / ".claude" / "projects")) + +_INJECTED_CODEX_PREFIXES = ( + "# AGENTS.md instructions", + "", + "", +) +_SYNTHETIC_CLAUDE_PREFIXES = ( + "", + "", + "", + "", + "", + "", + "", + "", +) +_SKILL_PATH = re.compile(r"(?:^|[\s\"'=:(])[^\s\"']*/skills/([^/\s\"']+)/SKILL\.md") +_SKILL_NAME = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:-]{0,127}$") +_REVISION = re.compile(r"^[0-9a-f]{8,64}$", re.IGNORECASE) +_MAX_ROUTE_ENVELOPE = 1_000_000 +_MAX_ROUTE_DEPTH = 6 +_MAX_ROUTE_NODES = 64 +_ROUTE_TOOL_NAMES = {"route_and_load", "mcp__ingot__route_and_load"} +_ROUTE_MARKERS = {"revision", "skill_body", "related_match"} +_USAGE_KEYS = ( + "input_tokens", "output_tokens", "cached_input_tokens", + "cache_write_input_tokens", "reasoning_output_tokens", +) + + +def _records(path: Path) -> Iterable[dict]: + try: + with path.open(errors="replace") as handle: + for line in handle: + try: + record = json.loads(line) + except (TypeError, ValueError): + continue + if isinstance(record, dict): + yield record + except OSError: + return + + +def _content_text(content, block_types: set[str]) -> str: + if isinstance(content, str): + return content.strip() + if not isinstance(content, list): + return "" + parts = [block.get("text", "") for block in content + if isinstance(block, dict) and block.get("type") in block_types + and isinstance(block.get("text"), str)] + return "\n".join(part.strip() for part in parts if part.strip()).strip() + + +def _codex_task(payload: dict) -> str: + content = payload.get("content") + if not isinstance(content, list): + return "" + parts = [] + for block in content: + if not isinstance(block, dict) or block.get("type") != "input_text": + continue + text = block.get("text") + if not isinstance(text, str) or not text.strip(): + continue + stripped = text.lstrip() + if stripped.startswith(_INJECTED_CODEX_PREFIXES + _SYNTHETIC_CLAUDE_PREFIXES): + continue + parts.append(text.strip()) + return "\n".join(parts).strip() + + +def _valid_skill(value) -> str | None: + value = value.strip() if isinstance(value, str) else "" + return value if _SKILL_NAME.fullmatch(value) else None + + +def _skills_from_command(command: str) -> list[str]: + return list(dict.fromkeys( + name for match in _SKILL_PATH.finditer(command) + if (name := _valid_skill(match.group(1))) + )) + + +def _json_value(value): + if not isinstance(value, str): + return value + try: + return json.loads(value) + except ValueError: + return value + + +def _route_identity(value, depth: int = 0, + budget: list[int] | None = None) -> tuple[str, str | None] | None: + """Extract only route identity from nested MCP result shapes; discard the served body.""" + budget = [_MAX_ROUTE_NODES] if budget is None else budget + if depth > _MAX_ROUTE_DEPTH or budget[0] <= 0: + return None + budget[0] -= 1 + if isinstance(value, str) and len(value) > _MAX_ROUTE_ENVELOPE: + return None + value = _json_value(value) + if isinstance(value, dict): + selected = _valid_skill(value.get("match") or value.get("related_match")) + if selected and _ROUTE_MARKERS.intersection(value): + revision = value.get("revision") + revision = (revision if isinstance(revision, str) + and _REVISION.fullmatch(revision) else None) + return selected, revision + for key in ("result", "structuredContent", "content", "output"): + if key in value and (identity := _route_identity( + value[key], depth + 1, budget)): + return identity + elif isinstance(value, list): + for item in value: + if isinstance(item, dict) and item.get("type") == "text": + item = item.get("text") + if identity := _route_identity(item, depth + 1, budget): + return identity + return None + + +def _merge_skill(skills: dict[str, str | None], name: str, revision: str | None = None) -> None: + if name not in skills or revision: + skills[name] = revision + + +def _tags(skills: dict[str, str | None]) -> list[str]: + out = [] + for name, revision in skills.items(): + out.append(f"skill:{name}") + if revision: + out.append(f"revision={name}@{revision}") + return out + + +def _add_usage(total: dict[str, int], usage) -> None: + if not isinstance(usage, dict): + return + values = { + "input_tokens": usage.get("input_tokens", 0), + "output_tokens": usage.get("output_tokens", 0), + "cached_input_tokens": ( + usage.get("cached_input_tokens", 0) + usage.get("cache_read_input_tokens", 0)), + "cache_write_input_tokens": ( + usage.get("cache_write_input_tokens", 0) + + usage.get("cache_creation_input_tokens", 0)), + "reasoning_output_tokens": usage.get("reasoning_output_tokens", 0), + } + for key, value in values.items(): + try: + amount = int(value or 0) + except (TypeError, ValueError): + continue + if amount: + total[key] = total.get(key, 0) + amount + + +def _tool_failed(value) -> bool: + """Count only structured failure signals; free-form output text is not an error contract.""" + value = _json_value(value) + if not isinstance(value, dict): + return False + exit_code = value.get("exit_code") + try: + failed_exit = exit_code is not None and int(exit_code) != 0 + except (TypeError, ValueError): + failed_exit = False + return failed_exit or value.get("success") is False or value.get("is_error") is True + + +def _trace(*, harness: str, session_id: str, turn_id: str, timestamp: str, cwd: str, + task: str, answer: str, skills: dict[str, str | None], usage: dict[str, int], + duration_ms: int | None = None, tool_errors: int = 0) -> dict: + identity = json.dumps([harness, session_id, turn_id, task, answer], + ensure_ascii=False, separators=(",", ":")) + result = { + "id": hashlib.sha256(identity.encode()).hexdigest(), + "timestamp": timestamp, + "harness": harness, + "session_id": session_id, + "turn_id": turn_id, + "cwd": cwd, + "project": Path(cwd).name if cwd else "", + "task": task, + "rubric": "", + "answer": answer, + "skills": [{"name": name, "revision": revision} for name, revision in skills.items()], + "tags": _tags(skills), + "usage": {key: usage[key] for key in _USAGE_KEYS if usage.get(key)}, + "tool_errors": tool_errors, + } + if duration_ms is not None: + result["duration_ms"] = duration_ms + return result + + +def parse_codex_session(path: Path) -> list[dict]: + session_id = path.stem + cwd = "" + thread_source = "" + pending_task = "" + pending_timestamp = "" + active = None + traces = [] + + for record in _records(path): + kind, payload = record.get("type"), record.get("payload") + payload = payload if isinstance(payload, dict) else {} + if kind == "session_meta": + session_id = str(payload.get("id") or payload.get("session_id") or session_id) + cwd = str(payload.get("cwd") or cwd) + source = payload.get("thread_source") + if not source and isinstance(payload.get("source"), dict): + source = "subagent" if payload["source"].get("subagent") else "" + thread_source = str(source or "") + continue + if thread_source and thread_source != "user": + continue + if (kind == "response_item" and payload.get("type") == "message" + and payload.get("role") == "user"): + task = _codex_task(payload) + if task: + pending_task = task + pending_timestamp = str(record.get("timestamp") or "") + continue + if kind == "event_msg" and payload.get("type") == "task_started": + active = ({ + "task": pending_task, "timestamp": pending_timestamp, + "turn_id": str(payload.get("turn_id") or ""), + "skills": {}, "usage": {}, "route_calls": set(), "tool_errors": 0, + } if pending_task else None) + pending_task = pending_timestamp = "" + continue + if active is None: + continue + if kind == "response_item" and payload.get("type") in { + "function_call", "custom_tool_call"}: + name = str(payload.get("name") or "") + arguments = _json_value(payload.get("arguments")) + if isinstance(arguments, dict): + if isinstance(arguments.get("cmd"), str): + for skill in _skills_from_command(arguments["cmd"]): + _merge_skill(active["skills"], skill) + if name in _ROUTE_TOOL_NAMES: + call_id = str(payload.get("call_id") or "") + if call_id: + active["route_calls"].add(call_id) + continue + if kind == "response_item" and payload.get("type") in { + "function_call_output", "custom_tool_call_output"}: + call_id = str(payload.get("call_id") or "") + output = payload.get("output") + if (call_id and call_id in active["route_calls"] + and (identity := _route_identity(output))): + _merge_skill(active["skills"], *identity) + active["tool_errors"] += int(_tool_failed(output)) + continue + if kind == "event_msg" and payload.get("type") == "token_count": + info = payload.get("info") + if isinstance(info, dict): + _add_usage(active["usage"], info.get("last_token_usage")) + continue + if kind == "event_msg" and payload.get("type") == "turn_aborted": + active = None + continue + if kind == "event_msg" and payload.get("type") == "task_complete": + answer = payload.get("last_agent_message") + if isinstance(answer, str) and answer.strip(): + duration = payload.get("duration_ms") + try: + duration = int(duration) if duration is not None else None + except (TypeError, ValueError): + duration = None + traces.append(_trace( + harness="codex", session_id=session_id, turn_id=active["turn_id"], + timestamp=active["timestamp"], cwd=cwd, task=active["task"], + answer=answer.strip(), skills=active["skills"], usage=active["usage"], + duration_ms=duration, tool_errors=active["tool_errors"], + )) + active = None + return traces + + +def _claude_human_task(record: dict) -> str: + if record.get("type") != "user" or record.get("isSidechain") is True: + return "" + # Claude serializes skill-hook output, compaction summaries, local command echoes, and some + # system injections as role=user. Provenance fields, not message wording, distinguish those + # records from typed/queued/SDK requests; accepting every user role turns tool output into + # fabricated tasks and severs the skill call from its real turn. + if (record.get("isMeta") or record.get("isCompactSummary") + or record.get("isVisibleInTranscriptOnly") or record.get("sourceToolUseID")): + return "" + source = record.get("promptSource") + if source == "system": + return "" + if not record.get("origin") and source not in { + "typed", "queued", "suggestion_accepted", "sdk"}: + return "" + message = record.get("message") + if not isinstance(message, dict): + return "" + content = message.get("content") + if isinstance(content, list) and any( + isinstance(block, dict) and block.get("type") == "tool_result" for block in content): + return "" + task = _content_text(content, {"text"}) + return "" if task.lstrip().startswith(_SYNTHETIC_CLAUDE_PREFIXES) else task + + +def parse_claude_session(path: Path) -> list[dict]: + traces = [] + active = None + sequence = 0 + + for record in _records(path): + task = _claude_human_task(record) + if task: + active = { + "task": task, + "timestamp": str(record.get("timestamp") or ""), + "session_id": str(record.get("sessionId") or path.stem), + "turn_id": str(record.get("uuid") or f"turn-{sequence}"), + "cwd": str(record.get("cwd") or ""), + "skills": {}, "usage": {}, "route_calls": set(), "tool_errors": 0, + } + sequence += 1 + continue + if active is None or record.get("isSidechain") is True: + continue + message = record.get("message") + if not isinstance(message, dict): + continue + content = message.get("content") + if record.get("type") == "assistant": + _add_usage(active["usage"], message.get("usage")) + for block in content if isinstance(content, list) else (): + if not isinstance(block, dict) or block.get("type") != "tool_use": + continue + name, inputs = str(block.get("name") or ""), block.get("input") + inputs = inputs if isinstance(inputs, dict) else {} + if name == "Skill" and (skill := _valid_skill( + inputs.get("skill") or inputs.get("name"))): + _merge_skill(active["skills"], skill) + if name in _ROUTE_TOOL_NAMES: + call_id = str(block.get("id") or "") + if call_id: + active["route_calls"].add(call_id) + if message.get("stop_reason") == "end_turn": + answer = _content_text(content, {"text"}) + if answer: + traces.append(_trace( + harness="claude", session_id=active["session_id"], + turn_id=active["turn_id"], timestamp=active["timestamp"], + cwd=active["cwd"], task=active["task"], answer=answer, + skills=active["skills"], usage=active["usage"], + tool_errors=active["tool_errors"], + )) + active = None + elif record.get("type") == "user" and isinstance(content, list): + for block in content: + if not isinstance(block, dict) or block.get("type") != "tool_result": + continue + if block.get("is_error"): + active["tool_errors"] += 1 + call_id = str(block.get("tool_use_id") or "") + if call_id and call_id in active["route_calls"]: + if identity := _route_identity(block.get("content")): + _merge_skill(active["skills"], *identity) + return traces + + +def _jsonl_files(root: Path) -> list[Path]: + if not root.exists(): + return [] + return sorted(path for path in root.rglob("*.jsonl") if path.is_file()) + + +def _date_value(value: str) -> str: + value = value.strip() + if not value: + return "" + try: + return date.fromisoformat(value).isoformat() + except ValueError as error: + raise ValueError(f"expected an ISO date (YYYY-MM-DD), got {value!r}") from error + + +def _filters(projects: Iterable[str] = (), since: str = "", until: str = "") -> dict: + result = { + "projects": sorted({str(project).strip() for project in projects if str(project).strip()}), + "since": _date_value(since), + "until": _date_value(until), + } + if result["since"] and result["until"] and result["since"] > result["until"]: + raise ValueError("--since must be on or before --until") + return result + + +def _selected(trace: dict, filters: dict) -> bool: + projects = filters["projects"] + if projects and trace.get("project") not in projects: + return False + day = str(trace.get("timestamp") or "")[:10] + if (filters["since"] or filters["until"]) and not re.fullmatch(r"\d{4}-\d{2}-\d{2}", day): + return False + if filters["since"] and day < filters["since"]: + return False + if filters["until"] and day > filters["until"]: + return False + return True + + +def _source_key(path: Path) -> str: + return hashlib.sha256(str(path.resolve()).encode()).hexdigest() + + +def _prior_store(output: Path, filters: dict) -> dict: + try: + payload = json.loads(output.read_text()) + except (OSError, ValueError): + return {} + if (payload.get("schema_version") != SCHEMA + or payload.get("parser_version") != PARSER_VERSION + or payload.get("filters", _filters()) != filters + or not isinstance(payload.get("traces"), list) + or not isinstance(payload.get("source_index"), dict)): + return {} + for trace in payload["traces"]: + if (not isinstance(trace, dict) or not isinstance(trace.get("id"), str) + or not isinstance(trace.get("_sources"), list) + or not trace["_sources"] + or not all(isinstance(source, str) for source in trace["_sources"])): + return {} + return payload + + +def scan(*, codex_dir: Path = CODEX_DIR, claude_dir: Path = CLAUDE_DIR, + output: Path = LOCAL_TRACE_FILE, projects: Iterable[str] = (), + since: str = "", until: str = "", force: bool = False) -> dict: + selected_filters = _filters(projects, since, until) + files = {"codex": _jsonl_files(codex_dir), "claude": _jsonl_files(claude_dir)} + prior = {} if force else _prior_store(output, selected_filters) + prior_index = prior.get("source_index", {}) + by_source: dict[str, list[dict]] = {} + for trace in prior.get("traces", []): + if not isinstance(trace, dict): + continue + for source in trace.get("_sources", []): + if isinstance(source, str): + by_source.setdefault(source, []).append(trace) + + source_index = {"codex": {}, "claude": {}} + collected = [] + reused = parsed = 0 + parsers = {"codex": parse_codex_session, "claude": parse_claude_session} + for harness, paths in files.items(): + old = prior_index.get(harness, {}) if isinstance(prior_index, dict) else {} + old = old if isinstance(old, dict) else {} + for path in paths: + try: + stat = path.stat() + except OSError: + continue + key = _source_key(path) + # This cheap cursor fits append-only agent logs. --force is the escape hatch for a + # same-size rewrite whose mtime was deliberately preserved. + fingerprint = {"size": stat.st_size, "mtime_ns": stat.st_mtime_ns} + source_index[harness][key] = fingerprint + if old.get(key) == fingerprint: + file_traces = by_source.get(key, []) + fresh = False + reused += 1 + else: + file_traces = parsers[harness](path) + fresh = True + parsed += 1 + for trace in file_traces: + if (not isinstance(trace, dict) or not isinstance(trace.get("id"), str) + or not _selected(trace, selected_filters)): + continue + copied = dict(trace) + # Rebuild provenance from files that still exist. Carrying the old list forward + # would keep a removed duplicate as a live source forever. + copied["_sources"] = [key] + copied["_fresh"] = fresh + collected.append(copied) + + deduped = {} + for trace in collected: + existing = deduped.get(trace["id"]) + if existing is None: + deduped[trace["id"]] = trace + else: + sources = sorted( + set(existing.get("_sources", ())) | set(trace.get("_sources", ()))) + existing_source = min(existing.get("_sources", ("",))) + candidate_source = min(trace.get("_sources", ("",))) + if ((trace["_fresh"] and not existing["_fresh"]) + or (trace["_fresh"] == existing["_fresh"] + and candidate_source < existing_source)): + deduped[trace["id"]] = trace + existing = trace + existing["_sources"] = sources + for trace in deduped.values(): + trace.pop("_fresh", None) + ordered = sorted(deduped.values(), key=lambda trace: ( + trace.get("timestamp", ""), trace["id"])) + payload = { + "schema_version": SCHEMA, + "parser_version": PARSER_VERSION, + "generated_at": int(time.time()), + "filters": selected_filters, + "sources": {name: {"files": len(paths)} for name, paths in files.items()}, + "source_index": source_index, + "traces": ordered, + } + output.parent.mkdir(parents=True, exist_ok=True) + staged_path = None + try: + with tempfile.NamedTemporaryFile( + mode="w", dir=output.parent, prefix=f".{output.name}.", delete=False) as staged: + staged_path = Path(staged.name) + json.dump(payload, staged, ensure_ascii=False, separators=(",", ":")) + staged.write("\n") + staged.flush() + os.fsync(staged.fileno()) + os.replace(staged_path, output) + staged_path = None + directory = os.open(output.parent, os.O_RDONLY) + try: + os.fsync(directory) + finally: + os.close(directory) + finally: + if staged_path is not None: + staged_path.unlink(missing_ok=True) + return {"output": str(output), "traces": len(ordered), + "files": sum(len(paths) for paths in files.values()), + "parsed_files": parsed, "reused_files": reused} + + +def _empty_summary(status: str = "missing") -> dict: + return { + "configured": False, "status": status, "generated_at": None, + "total": 0, "harnesses": {}, + "available_total": 0, "available_projects": {}, "available_harnesses": {}, + "available_skills": {}, + "filters": {"projects": [], "harness": "", "skill": "", "since": "", "until": "", + "include_tasks": False}, + "skill_uses": {}, "revision_pinned": 0, "unattributed": 0, + "usage": {}, "tool_errors": 0, "recent": [], + } + + +@lru_cache(maxsize=1) +def _read_store(path: str, mtime_ns: int, size: int) -> dict: + del mtime_ns, size # cache-key material; the file path is the only read target + return json.loads(Path(path).read_text()) + + +def _observed_skills(trace: dict) -> set[str]: + """Validated skill names attributed to one turn.""" + entries = trace.get("skills") if isinstance(trace.get("skills"), list) else [] + return {name for skill in entries if isinstance(skill, dict) + and (name := _valid_skill(skill.get("name")))} + + +def store_summary(path: Path = LOCAL_TRACE_FILE, recent: int = 20, *, project: str = "", + harness: str = "", skill: str = "", since: str = "", until: str = "", + include_tasks: bool = False) -> dict: + try: + stat = path.stat() + except OSError: + return _empty_summary() + try: + payload = _read_store(str(path), stat.st_mtime_ns, stat.st_size) + except (OSError, ValueError): + return _empty_summary("unreadable") + if payload.get("schema_version") != SCHEMA or not isinstance(payload.get("traces"), list): + return _empty_summary("unreadable") + all_traces = [trace for trace in payload["traces"] if isinstance(trace, dict)] + available_projects = Counter( + str(trace.get("project")) for trace in all_traces if trace.get("project")) + available_harnesses = Counter( + str(trace.get("harness")) for trace in all_traces if trace.get("harness")) + # Counted over every turn, not the filtered set: the picker has to keep offering a skill after + # you select it, and offering only skills that survive the current filter would empty itself. + available_skills = Counter( + name for trace in all_traces for name in _observed_skills(trace)) + selected_filters = _filters([project] if project else (), since, until) + traces = [trace for trace in all_traces + if _selected(trace, selected_filters) + and (not harness or trace.get("harness") == harness) + and (not skill or skill in _observed_skills(trace))] + harnesses = Counter() + skill_uses = Counter() + usage = Counter() + pinned = tool_errors = unattributed = 0 + for trace in traces: + if not isinstance(trace, dict): + continue + harnesses[str(trace.get("harness") or "unknown")] += 1 + skills = trace.get("skills") if isinstance(trace.get("skills"), list) else [] + if not skills: + unattributed += 1 + for observed in skills: # not `skill`: that name is the filter parameter + if not isinstance(observed, dict) or not _valid_skill(observed.get("name")): + continue + skill_uses[observed["name"]] += 1 + pinned += int(bool(observed.get("revision"))) + for key in _USAGE_KEYS: + try: + usage[key] += int((trace.get("usage") or {}).get(key, 0)) + except (AttributeError, TypeError, ValueError): + pass + try: + tool_errors += int(trace.get("tool_errors") or 0) + except (TypeError, ValueError): + pass + newest = sorted( + (trace for trace in traces if isinstance(trace, dict)), + key=lambda trace: (trace.get("timestamp", ""), trace.get("id", "")), reverse=True, + )[:recent] + previews = [] + for trace in newest: + preview_skills = [] + for observed in trace.get("skills", []) if isinstance(trace.get("skills"), list) else (): + if not isinstance(observed, dict) or not (name := _valid_skill(observed.get("name"))): + continue + revision = observed.get("revision") + revision = (revision if isinstance(revision, str) + and _REVISION.fullmatch(revision) else None) + preview_skills.append({"name": name, "revision": revision}) + preview_usage = {} + stored_usage = trace.get("usage") if isinstance(trace.get("usage"), dict) else {} + for key in _USAGE_KEYS: + try: + amount = int(stored_usage.get(key, 0)) + except (TypeError, ValueError): + continue + if amount > 0: + preview_usage[key] = amount + try: + preview_errors = max(0, int(trace.get("tool_errors") or 0)) + except (TypeError, ValueError): + preview_errors = 0 + preview = { + "id": str(trace.get("id") or ""), + "timestamp": str(trace.get("timestamp") or ""), + "harness": str(trace.get("harness") or ""), + "project": str(trace.get("project") or ""), + "skills": preview_skills, + "usage": preview_usage, + "tool_errors": preview_errors, + } + if include_tasks: + task = str(trace.get("task") or "") + preview["task"] = task[:280] + ("…" if len(task) > 280 else "") + previews.append(preview) + return { + "configured": True, "status": "ready", "generated_at": payload.get("generated_at"), + "available_total": len(all_traces), + "available_projects": dict(available_projects.most_common()), + "available_harnesses": dict(available_harnesses.most_common()), + "available_skills": dict(available_skills.most_common()), + "filters": {**selected_filters, "harness": harness, "skill": skill, + "include_tasks": include_tasks}, + "total": len(traces), "harnesses": dict(harnesses), + "skill_uses": dict(skill_uses.most_common()), "revision_pinned": pinned, + "unattributed": unattributed, "usage": dict(usage), + "tool_errors": tool_errors, "recent": previews, + } + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--codex-dir", type=Path, default=CODEX_DIR) + parser.add_argument("--claude-dir", type=Path, default=CLAUDE_DIR) + parser.add_argument("--output", type=Path, default=LOCAL_TRACE_FILE) + parser.add_argument("--project", action="append", default=[], + help="keep only this cwd basename; repeat for more than one project") + parser.add_argument("--since", default="", help="keep turns on or after YYYY-MM-DD") + parser.add_argument("--until", default="", help="keep turns on or before YYYY-MM-DD") + parser.add_argument("--force", action="store_true", + help="reparse every transcript instead of reusing the cursor") + args = parser.parse_args() + try: + result = scan(codex_dir=args.codex_dir, claude_dir=args.claude_dir, output=args.output, + projects=args.project, since=args.since, until=args.until, force=args.force) + except ValueError as error: + parser.error(str(error)) + print(f"[traces] normalized {result['traces']} completed turn(s) from " + f"{result['files']} transcript file(s) " + f"({result['parsed_files']} parsed, {result['reused_files']} unchanged) " + f"→ {result['output']}") + + +if __name__ == "__main__": + main() diff --git a/optimize/loop.py b/ingot/optimize/loop.py similarity index 98% rename from optimize/loop.py rename to ingot/optimize/loop.py index 3ab8740..d688061 100644 --- a/optimize/loop.py +++ b/ingot/optimize/loop.py @@ -4,7 +4,7 @@ unattended, off the review path. Every surviving candidate lands quarantined in the review queue and still requires human approval. -Usage: python -m optimize.loop [skill ...] (default: every skill with an eval task set) +Usage: python -m ingot.optimize.loop [skill ...] (default: every skill with an eval task set) """ import argparse import os diff --git a/optimize/mine.py b/ingot/optimize/mine.py similarity index 83% rename from optimize/mine.py rename to ingot/optimize/mine.py index e9436c5..eed4e40 100644 --- a/optimize/mine.py +++ b/ingot/optimize/mine.py @@ -9,7 +9,7 @@ the signal that drives targeted optimization (the paper: Liu et al., "SkillForge: Forging Domain-Specific, Self-Evolving Agent Skills", arXiv:2604.08618). -Usage: python -m optimize.mine [--limit N] +Usage: python -m ingot.optimize.mine [--limit N] """ import argparse import base64 @@ -23,11 +23,13 @@ from urllib.parse import urlencode, urlparse from .judge import DIMENSIONS, MODELS, _PROMPT, failed_dimensions, judge +from .local_traces import LOCAL_TRACE_FILE, SCHEMA as LOCAL_TRACE_SCHEMA +from ingot import paths LF_URL = os.environ.get("LANGFUSE_BASE_URL", "http://langfuse-web:3000") LF_PK = os.environ.get("LANGFUSE_PUBLIC_KEY", "pk-lf-local-demo") LF_SK = os.environ.get("LANGFUSE_SECRET_KEY", "sk-lf-local-demo") -RUNS_DIR = Path(__file__).resolve().parent.parent / "runs" +RUNS_DIR = paths.runs() JUDGE_CACHE_FILE = RUNS_DIR / "mine-cache" / "judgments.json" TRACE_PAGE_SIZE = 100 CLUSTER_THRESHOLD = float(os.environ.get("MINE_CLUSTER_THRESHOLD", "0.90")) @@ -97,6 +99,38 @@ def fetch_traces(limit: int = 0) -> list[dict]: return out +def fetch_local_traces(limit: int = 0) -> list[dict]: + """Read normalized local coding-agent turns. Transcript parsing is a separate, local command + so the paid miner never gains permission to crawl a home directory as a side effect.""" + if not LOCAL_TRACE_FILE.exists(): + raise SystemExit( + f"No local trace store at {LOCAL_TRACE_FILE}; run `python -m ingot.optimize.local_traces` " + "on the machine that owns the Codex and Claude transcripts.") + try: + payload = json.loads(LOCAL_TRACE_FILE.read_text()) + except (OSError, ValueError) as e: + raise SystemExit(f"Local trace store is unreadable ({e}); scan it again.") from e + if (payload.get("schema_version") != LOCAL_TRACE_SCHEMA + or not isinstance(payload.get("traces"), list)): + raise SystemExit( + f"unsupported local trace store; expected {LOCAL_TRACE_SCHEMA}, scan it again") + ordered = sorted( + (trace for trace in payload["traces"] if isinstance(trace, dict)), + key=lambda trace: (trace.get("timestamp", ""), trace.get("id", "")), reverse=True, + ) + out = [] + for trace in ordered: + parsed = _task_answer(trace.get("task"), trace.get("answer")) + if not parsed: + continue + task, rubric, answer = parsed + out.append({"task": task, "rubric": rubric, "answer": answer, + "tags": trace.get("tags", [])}) + if limit and len(out) >= limit: + break + return out + + def _task_answer(inp, ans): """(task, rubric, answer) from supported trace roots: explicit eval dictionaries, LangGraph state, and the root shapes emitted by the Claude Code and Codex Langfuse connectors.""" @@ -121,16 +155,34 @@ def _task_answer(inp, ans): return None +def _tagged_with(tags, skill: str) -> bool: + """Does any tag name `skill`? Harnesses spell it three ways, and only the first is bare: + our own agent writes `pdf` plus a revision pin `revision=pdf@`, while Claude Code + writes `skill:`, namespaced by plugin when the skill came from one + (`skill:superpowers:systematic-debugging`). Matching the bare form alone drops every trace an + external harness produced — which is nearly all real traffic — and mining then reports + 'no traces relevant to X' as though the skill were never used.""" + for tag in tags or []: + tag = str(tag) + if tag == skill: + return True + if tag.startswith("revision=") and tag[len("revision="):].split("@", 1)[0] == skill: + return True + if tag.startswith("skill:") and tag[len("skill:"):].rsplit(":", 1)[-1] == skill: + return True + return False + + def relevant_traces(traces: list[dict], skill: str, k: int = 5) -> list[dict]: """Traces attributable to `skill`: tagged with it (external harnesses tag the routed skill), or ranking it in the embedding top-k for the task text. The rank check is what catches traffic the skill *should* have served but didn't route, under-triggering, the common routing failure, which a tag filter alone would attribute to the wrong skill.""" - from mcp_server.registry import load_skills - from mcp_server.router import Router + from ingot.mcp_server.registry import load_skills + from ingot.mcp_server.router import Router router = Router(load_skills()) return [t for t in traces - if skill in t.get("tags", []) + if _tagged_with(t.get("tags"), skill) or any(s["name"] == skill for s in router.suggest(t["task"], k=k, min_score=0.0))] @@ -182,7 +234,7 @@ def _normalized_embedder(): """Return an embedding function that produces row-normalized float vectors. Task-to-task similarity (diversity + train-dup checks), so both sides use the document embedding.""" import numpy as np - from mcp_server.embedding import build_embedding + from ingot.mcp_server.embedding import build_embedding embedder = build_embedding() @@ -360,10 +412,23 @@ def _select_candidates(traces: list[dict], scores: list[float], skill: str, log= return _rank_candidates(traces, scores, indices, context) -def mine(skill: str, limit: int = 0, log=print) -> dict: +def _source_traces(source: str, limit: int) -> list[dict]: + if source == "langfuse": + return fetch_traces(limit) + if source == "local": + return fetch_local_traces(limit) + raise SystemExit(f"unknown trace source {source!r}; choose langfuse or local") + + +def mine(skill: str, limit: int = 0, log=print, source: str = "langfuse", + allow_external_judge: bool = False) -> dict: + if source == "local" and not allow_external_judge: + raise SystemExit( + "Local transcript content stays local by default. Re-run with " + "`--allow-external-judge` to send selected task/answer pairs to JUDGE_MODEL.") scope = f"the newest {limit}" if limit else "all" - log(f"[mine] pulling {scope} traces from Langfuse for '{skill}'…") - traces = fetch_traces(limit) + log(f"[mine] pulling {scope} {source} traces for '{skill}'…") + traces = _source_traces(source, limit) if not traces: raise SystemExit("No usable traces found; run the agent or a candidate pass first to " "generate some.") @@ -438,7 +503,14 @@ def mine(skill: str, limit: int = 0, log=print) -> dict: ap.add_argument("skill") ap.add_argument("--limit", type=int, default=0, help="newest traces to inspect; 0 (default) paginates through all uses") + ap.add_argument("--source", choices=("langfuse", "local"), default="langfuse", + help="trace store to mine; local requires an explicit transcript scan") + ap.add_argument("--allow-external-judge", action="store_true", + help="allow local task/answer pairs to be sent to JUDGE_MODEL") args = ap.parse_args() + if args.source == "local" and not args.allow_external_judge: + ap.error("--source local requires --allow-external-judge") from . import require_openrouter_key require_openrouter_key() - mine(args.skill, limit=args.limit) + mine(args.skill, limit=args.limit, source=args.source, + allow_external_judge=args.allow_external_judge) diff --git a/optimize/promote.py b/ingot/optimize/promote.py similarity index 57% rename from optimize/promote.py rename to ingot/optimize/promote.py index 7b9b930..a81a0c8 100644 --- a/optimize/promote.py +++ b/ingot/optimize/promote.py @@ -1,20 +1,35 @@ """Gate-enforced, revisioned, atomic promotion and rollback for quarantined skill changes.""" from __future__ import annotations +import ctypes +import errno import json import logging import os import shutil +import sys import time import uuid from pathlib import Path -from mcp_server.registry import (SLUG_RE, load_skills, read_components, skill_revision, - write_components) +from ingot.mcp_server.registry import (SLUG_RE, load_skills, read_components, skill_revision, + writable_skill_dir, write_components) +from ingot import paths -RUNS_DIR = Path(__file__).resolve().parent.parent / "runs" -PENDING_DIR = RUNS_DIR / "pending" -REVISIONS_DIR = RUNS_DIR / "revisions" + + +def pending_dir() -> Path: + """The one review slot per skill. Resolved per call: it is configuration, and a value frozen at + import cannot follow a process that is told where its state lives.""" + return paths.runs() / "pending" + + +def revisions_dir() -> Path: + """Stored snapshots, the rollback targets.""" + return paths.runs() / "revisions" + +ABSENT_REVISION = "absent" +ABSENT_MARKER = ".ingot-absent" logger = logging.getLogger(__name__) @@ -25,11 +40,11 @@ def check_slug(skill: str) -> str: def pending_path(skill: str) -> Path: - return PENDING_DIR / f"{check_slug(skill)}.json" + return pending_dir() / f"{check_slug(skill)}.json" def save_pending(skill: str, data: dict) -> Path: - PENDING_DIR.mkdir(parents=True, exist_ok=True) + pending_dir().mkdir(parents=True, exist_ok=True) path = pending_path(skill) _archive_displaced(skill, path, data) temporary = path.with_suffix(f".{uuid.uuid4().hex}.tmp") @@ -47,7 +62,7 @@ def _archive_displaced(skill: str, path: Path, data: dict) -> None: existing = json.loads(path.read_text()) if sorted(existing.get("changed_components", [])) == sorted(data.get("changed_components", [])): return - archived = PENDING_DIR / f"{skill}.displaced-{existing.get('created', uuid.uuid4().hex)}.json" + archived = pending_dir() / f"{skill}.displaced-{existing.get('created', uuid.uuid4().hex)}.json" shutil.copy(path, archived) print(f"[pending] one review slot per skill: the pending " f"{existing.get('changed_components')} challenger was displaced by this " @@ -60,17 +75,42 @@ def load_pending(skill: str) -> dict | None: return json.loads(path.read_text()) if path.exists() else None +def unreadable_pending() -> list[str]: + """Pending files this process can see but cannot use. + + `list_pending` skips them so one corrupt record cannot break review — which also means a + proposal this process cannot read is indistinguishable from no proposal at all, and the console + reports CLEAR over it. Observed live: the MCP container writes records as root 0600 while the UI + runs as uid 1000, so an approved-and-waiting skill was invisible for hours.""" + if not pending_dir().exists(): + return [] + blocked = [] + for path in sorted(pending_dir().glob("*.json")): + if not SLUG_RE.fullmatch(path.stem): + continue + try: + record = json.loads(path.read_text()) + except (OSError, ValueError): # ValueError covers JSONDecodeError and UnicodeDecodeError + blocked.append(path.name) + continue + if not (isinstance(record, dict) and record.get("skill") == path.stem): + blocked.append(path.name) + return blocked + + def list_pending() -> list[dict]: """Return valid pending records without letting a malformed queue file break the review UI.""" - if not PENDING_DIR.exists(): + if not pending_dir().exists(): return [] records = [] - for path in sorted(PENDING_DIR.glob("*.json")): + for path in sorted(pending_dir().glob("*.json")): if not SLUG_RE.fullmatch(path.stem): continue try: record = json.loads(path.read_text()) - except (OSError, json.JSONDecodeError): + except (OSError, ValueError): + # ValueError, not json.JSONDecodeError: a non-UTF-8 file raises UnicodeDecodeError, + # which escaped this handler and took the whole review page down with it. continue if isinstance(record, dict) and record.get("skill") == path.stem: records.append(record) @@ -81,7 +121,7 @@ def snapshot_index_path(skill: str) -> Path: """When each snapshot was last taken. It lives beside the snapshot directories, never inside one, so a rollback copies back the skill and nothing else. The leading dot also keeps it out of the slug-matched snapshot listing.""" - return REVISIONS_DIR / check_slug(skill) / ".snapshots.json" + return revisions_dir() / check_slug(skill) / ".snapshots.json" def _read_snapshot_index(skill: str) -> dict: @@ -145,7 +185,7 @@ def list_revisions(skill: str) -> list[dict]: entry is `{"revision": , "created": }`. Snapshots taken before the index existed, and snapshots whose index entry is unusable, fall back to directory mtime and sort below stamped ones.""" - root = REVISIONS_DIR / check_slug(skill) + root = revisions_dir() / check_slug(skill) if not root.is_dir(): return [] index = _read_snapshot_index(skill) @@ -171,9 +211,9 @@ def load_snapshot_components(skill: str, revision: str) -> dict[str, str]: def list_snapshotted_skills() -> list[str]: """Skills with at least one snapshot. Reading the snapshot store directly keeps the history view off the skill-library hash scan that the skills listing already pays for.""" - if not REVISIONS_DIR.is_dir(): + if not revisions_dir().is_dir(): return [] - return sorted(path.name for path in REVISIONS_DIR.iterdir() + return sorted(path.name for path in revisions_dir().iterdir() if path.is_dir() and SLUG_RE.fullmatch(path.name)) @@ -213,6 +253,25 @@ def stale_evidence_reason(skill: str, pending: dict) -> str | None: evidence = pending.get("evidence") if not isinstance(evidence, dict) or not evidence: return "evidence is required for promotion" + if pending.get("kind") == "creation": + if any(item.name == skill for item in load_skills()): + return "a skill with this name appeared after the proposal; review it as an update" + expected = evidence.get("challenger", {}).get("revision") + components = pending.get("challenger_components", {}) + manifest = pending.get("tree") + # An ingested package's revision is what materializing its staged tree produces, not what + # its decoded text components produce -- the tree carries files no component describes. + # Recomputing from the components alone would report every package with a second file as + # permanently stale: quarantined, reviewable, and impossible to approve. + if manifest is not None: + from ingot.optimize import tree as candidate_tree + try: + actual = candidate_tree.revision(skill, manifest, components) + except (ValueError, OSError) as exc: + return f"the staged candidate tree is no longer usable: {exc}" + else: + actual = skill_revision(writable_skill_dir(skill), components) + return None if expected == actual else "challenger revision does not match the recorded evidence" try: current = _current_skill(skill) except ValueError as exc: @@ -224,7 +283,7 @@ def _snapshot(skill_dir: Path, skill: str, revision: str) -> Path: """Preserve a revision as a rollback target. Re-snapshotting a revision that is already stored is a no-op on disk but still restamps it: that is what a rollback followed by a promotion does, and the restored revision is then the most recent thing a promotion displaced.""" - destination = REVISIONS_DIR / skill / revision + destination = revisions_dir() / skill / revision if not destination.exists(): destination.parent.mkdir(parents=True, exist_ok=True) temporary = destination.with_name(f".{revision}.{uuid.uuid4().hex}.tmp") @@ -240,7 +299,7 @@ def _snapshot(skill_dir: Path, skill: str, revision: str) -> Path: def audit_path() -> Path: """The approval trail lives beside the review queue, so a relocated queue moves both.""" - return PENDING_DIR.parent / "approval-audit.jsonl" + return pending_dir().parent / "approval-audit.jsonl" def read_audit(limit: int = 50) -> dict: @@ -344,6 +403,38 @@ def _sweep_staging(skill_dir: Path) -> None: logger.warning("Could not remove the stale staging directory %s", stale, exc_info=True) +def _rename_no_replace(source: Path, target: Path) -> None: + """Atomically publish a new skill directory without replacing a target that won a race. + + Python's POSIX rename replaces an empty directory, so an existence check followed by rename is + not a guard. Linux and macOS expose the needed no-replace operation under different names; fail + closed on other POSIX platforms rather than quietly restore the race. + """ + if sys.platform.startswith("linux"): + libc = ctypes.CDLL(None, use_errno=True) + rename = getattr(libc, "renameat2", None) + if rename is None: + raise OSError(errno.ENOTSUP, "atomic no-replace rename is unavailable") + rename.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, + ctypes.c_uint] + rename.restype = ctypes.c_int + result = rename(-100, os.fsencode(source), -100, os.fsencode(target), 1) + elif sys.platform == "darwin": + libc = ctypes.CDLL(None, use_errno=True) + rename = libc.renamex_np + rename.argtypes = [ctypes.c_char_p, ctypes.c_char_p, ctypes.c_uint] + rename.restype = ctypes.c_int + result = rename(os.fsencode(source), os.fsencode(target), 0x00000004) + elif os.name == "nt": + source.rename(target) # Windows rename already refuses an existing destination. + return + else: + raise OSError(errno.ENOTSUP, "atomic no-replace rename is unavailable") + if result: + error = ctypes.get_errno() + raise OSError(error, os.strerror(error), target) + + def _activate_rewrite(skill: str, pending: dict) -> str: components = pending["challenger_components"] evidence = pending.get("evidence") @@ -352,22 +443,35 @@ def _activate_rewrite(skill: str, pending: dict) -> str: current = _current_skill(skill) _validate_evidence(current, components, evidence) - skill_dir = Path(current.root) - _snapshot(skill_dir, skill, current.revision) + source_dir = Path(current.root) + skill_dir = writable_skill_dir(skill) + if source_dir != skill_dir and (skill_dir.exists() or skill_dir.is_symlink()): + raise ValueError( + f"writable activation target already exists but is not the serving skill: {skill_dir}") + _snapshot(source_dir, skill, current.revision) _sweep_staging(skill_dir) stage = skill_dir.with_name(f".{skill_dir.name}.{uuid.uuid4().hex}.stage") previous = skill_dir.with_name(f".{skill_dir.name}.{uuid.uuid4().hex}.previous") try: - shutil.copytree(skill_dir, stage, symlinks=True) + shutil.copytree(source_dir, stage, symlinks=True) write_components(stage, components) - skill_dir.rename(previous) - try: - stage.rename(skill_dir) - except BaseException: - previous.rename(skill_dir) - raise - shutil.rmtree(previous, ignore_errors=True) + if source_dir != skill_dir: + try: + _rename_no_replace(stage, skill_dir) + except OSError as exc: + if exc.errno in (errno.EEXIST, errno.ENOTEMPTY): + raise ValueError( + f"writable activation target appeared during promotion: {skill_dir}") from exc + raise + else: + skill_dir.rename(previous) + try: + stage.rename(skill_dir) + except BaseException: + previous.rename(skill_dir) + raise + shutil.rmtree(previous, ignore_errors=True) except BaseException: shutil.rmtree(stage, ignore_errors=True) raise @@ -376,17 +480,96 @@ def _activate_rewrite(skill: str, pending: dict) -> str: return f"Promoted '{skill}' from revision {current.revision}; previous revision snapshotted." +def _snapshot_absence(skill: str) -> None: + destination = revisions_dir() / skill / ABSENT_REVISION + destination.mkdir(parents=True, exist_ok=True) + (destination / ABSENT_MARKER).touch(exist_ok=True) + _stamp_snapshot_best_effort(skill, ABSENT_REVISION) + + +def _activate_creation(skill: str, pending: dict) -> str: + """Atomically add a reviewed skill while preserving absence as its rollback target.""" + if any(item.name == skill for item in load_skills()): + raise ValueError(f"skill '{skill}' already exists; review it as an update") + problem = stale_evidence_reason(skill, pending) + if problem: + raise ValueError(problem) + components = pending["challenger_components"] + skill_dir = writable_skill_dir(skill) + if skill_dir.exists() or skill_dir.is_symlink(): + raise ValueError(f"writable activation target already exists: {skill_dir}") + skill_dir.parent.mkdir(parents=True, exist_ok=True) + _snapshot_absence(skill) + _sweep_staging(skill_dir) + stage = skill_dir.with_name(f".{skill_dir.name}.{uuid.uuid4().hex}.stage") + try: + stage.mkdir() + from ingot.mcp_server.registry import write_skill_md + try: + frontmatter = json.loads(components.get("frontmatter", "{}")) + except (TypeError, ValueError) as exc: + raise ValueError("creation frontmatter is not valid JSON") from exc + write_skill_md(stage / "SKILL.md", frontmatter, components["body"]) + write_components(stage, components) + expected = pending.get("evidence", {}).get("challenger", {}).get("revision") + actual = skill_revision(stage) + if actual != expected: + raise ValueError("staged creation revision does not match the recorded evidence") + _rename_no_replace(stage, skill_dir) + except BaseException: + shutil.rmtree(stage, ignore_errors=True) + raise + pending_path(skill).unlink(missing_ok=True) + return f"Added '{skill}' to the served library; prior state was absence." + + +def _activate_approved(skill: str, pending: dict, actor: str = "local-operator") -> str: + """Complete a publisher-verified approval. The UI approval path must never call this.""" + skill = check_slug(skill) + _require_promotable(pending) + result = (_activate_creation(skill, pending) if pending.get("kind") == "creation" + else _activate_rewrite(skill, pending)) + revision = _current_skill(skill).revision + _audit_best_effort("approve", skill, revision, actor) + return result + + def approve_pending(skill: str, actor: str = "local-operator") -> str: - """Activate one tested rewrite after an explicit approval action.""" + """Approve one tested challenger for vault publication without activating it.""" skill = check_slug(skill) pending = load_pending(skill) if not pending: raise ValueError(f"no pending challenger for '{skill}'") _require_promotable(pending) - result = _activate_rewrite(skill, pending) - revision = _current_skill(skill).revision - _audit_best_effort("approve", skill, revision, actor) - return result + problem = stale_evidence_reason(skill, pending) + if problem: + raise ValueError(problem) + from ingot.optimize.publication import queue_publication + queue_publication(skill, pending, actor, "promote") + return f"Approved '{skill}'; publishing to vault." + + +def challenger_revision(pending: dict) -> str: + """The challenger's revision from a pending record's evidence, or '' when none is recorded.""" + revision = ((pending.get("evidence") or {}).get("challenger") or {}).get("revision") + return revision if isinstance(revision, str) else "" + + +def reject_pending(skill: str, actor: str = "local-operator", reason: str = "") -> str: + """Discard one quarantined change and record why. + + Shared by the console and the command line so a rejection means the same thing and lands in the + same trail whichever one a reviewer used. Callers that run concurrently -- the console's request + threads -- serialize around this themselves; loading, deleting and auditing must happen under + one lock or a second rejection re-deletes and double-audits after the first releases.""" + skill = check_slug(skill) + pending = load_pending(skill) + if pending is None: + raise ValueError(f"no pending change for '{skill}'") + revision = challenger_revision(pending) + pending_path(skill).unlink(missing_ok=True) + _audit_best_effort("reject", skill, revision, actor, reason=" ".join(reason.split())) + return f"rejected the pending change for '{skill}'" def _rollback_source(skill: str, revision: str) -> Path: @@ -394,7 +577,7 @@ def _rollback_source(skill: str, revision: str) -> Path: skill = check_slug(skill) if not SLUG_RE.fullmatch(revision): raise ValueError(f"invalid revision: {revision!r}") - source = REVISIONS_DIR / skill / revision + source = revisions_dir() / skill / revision if not source.is_dir(): raise ValueError(f"no snapshot for '{skill}' at revision {revision}") return source @@ -430,12 +613,70 @@ def _swap_rollback(skill_dir: Path, stage: Path) -> None: def rollback(skill: str, revision: str, actor: str = "local-operator") -> str: - """Atomically restore a snapshot while preserving the displaced current revision.""" + """Queue a stored snapshot for vault publication without changing any served byte. + + Rollback takes the same Git lane as approval: the served library is a read-only checkout of the + canonical vault, so restoring a revision has to travel through a merged vault commit rather + than through the filesystem beneath it.""" + _rollback_source(skill, revision) + try: + expected = _current_skill(skill).revision + except ValueError: + if revision == ABSENT_REVISION: + raise ValueError(f"skill '{skill}' is already absent") from None + expected = ABSENT_REVISION + if expected == revision: + raise ValueError(f"'{skill}' already serves revision {revision}") + from ingot.optimize.publication import queue_publication + queue_publication(skill, { + "skill": skill, + "kind": "rollback", + "challenger_components": {}, + "evidence": {"champion": {"revision": expected}, + "challenger": {"revision": revision}}, + }, actor, "rollback") + return f"Approved rollback of '{skill}' to {revision}; publishing to vault." + + +def _activate_rollback(skill: str, revision: str, actor: str = "local-operator") -> str: + """Complete a publisher-verified rollback against a writable library. The UI must never call + this: with the canonical vault mounted read-only, the merged vault commit is what restores a + revision. It has no production caller today (see the session log).""" source = _rollback_source(skill, revision) - current = _current_skill(skill) + try: + current = _current_skill(skill) + except ValueError: + if (source / ABSENT_MARKER).is_file(): + raise ValueError(f"skill '{skill}' is already absent") + skill_dir = writable_skill_dir(skill) + if skill_dir.exists() or skill_dir.is_symlink(): + raise ValueError(f"inactive skill target already exists: {skill_dir}") + skill_dir.parent.mkdir(parents=True, exist_ok=True) + _sweep_staging(skill_dir) + stage = _stage_rollback(source, skill_dir) + try: + _rename_no_replace(stage, skill_dir) + except BaseException: + shutil.rmtree(stage, ignore_errors=True) + raise + restored = _current_skill(skill) + _audit_best_effort("rollback", skill, restored.revision, actor) + return f"Restored absent skill '{skill}' at revision {restored.revision}." skill_dir = Path(current.root) _snapshot(skill_dir, skill, current.revision) _sweep_staging(skill_dir) + if (source / ABSENT_MARKER).is_file(): + if skill_dir != writable_skill_dir(skill): + raise ValueError(f"cannot remove mounted skill '{skill}' through an absence rollback") + previous = skill_dir.with_name(f".{skill_dir.name}.{uuid.uuid4().hex}.previous") + skill_dir.rename(previous) + try: + shutil.rmtree(previous) + except BaseException: + previous.rename(skill_dir) + raise + _audit_best_effort("rollback", skill, ABSENT_REVISION, actor) + return f"Rolled back '{skill}' from {current.revision} to absence." stage = _stage_rollback(source, skill_dir) _swap_rollback(skill_dir, stage) restored = _current_skill(skill) diff --git a/ingot/optimize/publication.py b/ingot/optimize/publication.py new file mode 100644 index 0000000..3e9d268 --- /dev/null +++ b/ingot/optimize/publication.py @@ -0,0 +1,303 @@ +"""Durable, inert approvals waiting for publication into the canonical skill vault.""" +from __future__ import annotations + +import fcntl +import hashlib +import json +import os +import time +import uuid +from dataclasses import dataclass +from pathlib import Path, PurePosixPath + +from ingot.mcp_server.registry import SLUG_RE +from ingot.optimize import tree +from ingot import paths + + + + +def publications_dir() -> Path: + """Approved receipts waiting on the publisher. Resolved per call, never bound at import.""" + return paths.runs() / "publications" + +ACTIVE_STATES = {"approved_publishing", "publishing", "awaiting_merge", "merged"} +ACTIONS = {"promote", "rollback"} + + +@dataclass(frozen=True) +class PublicationReceipt: + id: str + state: str + path: Path + + +def _canonical(value: object) -> str: + return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False) + + +def _publication_id(identity: dict) -> str: + return hashlib.sha256(_canonical(identity).encode()).hexdigest()[:24] + + +def _proposal_id(pending: dict) -> str: + for key in ("creation", "retrospective"): + value = (pending.get(key) or {}).get("proposal_id") + if isinstance(value, str) and value: + return value + value = pending.get("proposal_id") + return value if isinstance(value, str) else "" + + +def _components(pending: dict, action: str) -> dict[str, str]: + """A promotion carries the exact approved bytes; a rollback republishes a stored snapshot. + + The snapshot is the authority for a rollback, so carrying components beside it would create a + second description of the same target that could disagree with it.""" + raw = pending.get("challenger_components") + if action == "rollback": + if raw is not None and not isinstance(raw, dict): + raise ValueError("challenger components must be an object") + if raw: + raise ValueError("a rollback republishes a stored snapshot and carries no components") + return {} + if not isinstance(raw, dict) or not raw: + raise ValueError("challenger components are required for publication") + result = {} + for key, value in raw.items(): + if not isinstance(key, str) or not isinstance(value, str): + raise ValueError("component names and contents must be strings") + if key.startswith("file:"): + path = PurePosixPath(key[5:]) + if path.is_absolute() or ".." in path.parts or path.as_posix() in {".", "SKILL.md"}: + raise ValueError(f"component escapes skill root: {path}") + elif key not in {"description", "body", "frontmatter"}: + raise ValueError(f"unsupported component: {key}") + result[key] = value + if "description" not in result or "body" not in result: + raise ValueError("description and body components are required") + return result + + +def _tree(pending: dict, action: str) -> dict | None: + """The staged bytes an ingested package publishes, if it carries any. + + Absent for an optimizer promotion, which rewrites text in a skill the vault already holds, and + for a rollback, which restores a stored snapshot. Present for an ingested package, where it is + the authority for every file: the receipt binds the tree's digest, and the publisher refuses + any staged file whose hash has moved since approval.""" + raw = pending.get("tree") + if raw is None: + return None + if action == "rollback": + raise ValueError("a rollback republishes a stored snapshot and carries no candidate tree") + return tree.verify_manifest(raw) + + +def _read(path: Path) -> dict: + value = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(value, dict): + raise ValueError(f"invalid publication record: {path}") + return value + + +def load_publication(publication_id: str) -> dict | None: + if not SLUG_RE.fullmatch(publication_id): + raise ValueError(f"invalid publication id: {publication_id!r}") + path = publications_dir() / f"{publication_id}.json" + return _read(path) if path.is_file() else None + + +def update_publication(publication_id: str, **changes) -> dict: + """Replace one receipt durably; the publication id and candidate identity cannot change.""" + path = publications_dir() / f"{publication_id}.json" + record = _read(path) + immutable = {"id", "skill", "action", "expected_champion", "candidate_revision", "components", + "tree"} + if immutable.intersection(changes): + raise ValueError("publication identity is immutable") + record.update(changes) + temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp") + fd = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + try: + os.fchmod(fd, 0o600) + _write_all(fd, (_canonical(record) + "\n").encode()) + os.fsync(fd) + finally: + os.close(fd) + try: + temporary.replace(path) + directory_fd = os.open(path.parent, os.O_RDONLY) + try: + os.fsync(directory_fd) + finally: + os.close(directory_fd) + finally: + temporary.unlink(missing_ok=True) + return record + + +def _all_records() -> list[dict]: + """Every readable receipt, newest first. A receipt this process cannot parse is skipped + rather than fatal: one corrupt file must not blank the whole lane.""" + if not publications_dir().is_dir(): + return [] + records = [] + for path in publications_dir().glob("*.json"): + try: + records.append(_read(path)) + except (OSError, ValueError, json.JSONDecodeError): + continue + records.sort(key=_ordering, reverse=True) + return records + + +def publication_for_skill(skill: str) -> dict | None: + for record in _all_records(): + if record.get("skill") == skill: + return record + return None + + +LIVE_STATES = ("approved_publishing", "publishing", "awaiting_merge") + + +def publishing_skills() -> set[str]: + """Skills whose newest receipt is still travelling to the vault. + + One pass over the store, not one per skill: the board asks this for every skill it lists. + Only the newest receipt counts, or a skill would look like it were publishing forever on the + strength of some earlier attempt.""" + newest: dict[str, dict] = {} + for record in _all_records(): + newest.setdefault(record.get("skill", ""), record) + return {skill for skill, record in newest.items() if record.get("state") in LIVE_STATES} + + +def latest_releases() -> dict[str, dict]: + """The newest successful release per skill: what the deployment is supposed to be serving. + + One pass over the store, like `publishing_skills`, because status asks this for every skill it + lists. Only an `active` receipt counts — a receipt that failed on its way to the vault never + described served bytes and must not be mistaken for a release.""" + releases: dict[str, dict] = {} + for record in _all_records(): + if record.get("state") == "active": + releases.setdefault(record.get("skill", ""), record) + return releases + + +def recent_publications(limit: int = 12) -> list[dict]: + """The publication lane as a whole, newest first. + + `publication_for_skill` only answers for a skill that still has a pending record, so once a + change is approved the console loses sight of it — which is exactly the window where it is + waiting on a vault merge and a person needs to see it.""" + return _all_records()[:max(0, limit)] + + +def _ordering(record: dict) -> tuple: + """Newest first. `created` is whole seconds, so a rollback queued in the same second as the + approval it displaces would otherwise be ordered by the hash in its id — and the review surface + would render whichever receipt happened to sort higher.""" + created = record.get("created_ns") + if not isinstance(created, int) or isinstance(created, bool): + created = int(record.get("created", 0) or 0) * 1_000_000_000 + return (created, record.get("id", "")) + + +def _write_all(fd: int, payload: bytes) -> None: + view = memoryview(payload) + while view: + written = os.write(fd, view) + if written <= 0: + raise OSError("publication write made no progress") + view = view[written:] + + +def _publish(path: Path, record: dict) -> None: + temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp") + fd = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + try: + os.fchmod(fd, 0o600) + _write_all(fd, (_canonical(record) + "\n").encode()) + os.fsync(fd) + finally: + os.close(fd) + try: + os.link(temporary, path) + directory_fd = os.open(path.parent, os.O_RDONLY) + try: + os.fsync(directory_fd) + finally: + os.close(directory_fd) + finally: + temporary.unlink(missing_ok=True) + + +def queue_publication(skill: str, pending: dict, actor: str, action: str) -> PublicationReceipt: + """Record one approved candidate without changing any served skill or pending review slot.""" + if not SLUG_RE.fullmatch(skill): + raise ValueError(f"invalid skill name: {skill!r}") + if not isinstance(actor, str) or not actor.strip(): + raise ValueError("approval actor is required") + if action not in ACTIONS: + raise ValueError(f"invalid publication action: {action!r}") + evidence = pending.get("evidence") or {} + expected = ((evidence.get("champion") or {}).get("revision") + if pending.get("kind") != "creation" else "absent") + candidate = (evidence.get("challenger") or {}).get("revision") + if not isinstance(expected, str) or not expected: + raise ValueError("expected champion revision is required") + if not isinstance(candidate, str) or not candidate: + raise ValueError("candidate revision is required") + components = _components(pending, action) + candidate_tree = _tree(pending, action) + proposal_id = _proposal_id(pending) + identity = {"skill": skill, "action": action, "expected_champion": expected, + "candidate_revision": candidate, "components": components} + if candidate_tree: + identity["tree"] = candidate_tree["digest"] + publication_id = _publication_id(identity) + path = publications_dir() / f"{publication_id}.json" + record = { + "schema_version": "ingot/publication/v1", + "id": publication_id, + "skill": skill, + "state": "approved_publishing", + "proposal_id": proposal_id, + "actor": actor.strip(), + "action": action, + "kind": pending.get("kind", "quality"), + "expected_champion": expected, + "candidate_revision": candidate, + "components": components, + "created": int(time.time()), + "created_ns": time.time_ns(), + "attempts": 0, + "last_error": "", + } + if candidate_tree: + record["tree"] = candidate_tree + + publications_dir().mkdir(parents=True, exist_ok=True, mode=0o700) + os.chmod(publications_dir(), 0o700) + lock_path = publications_dir() / ".queue.lock" + lock_fd = os.open(lock_path, os.O_CREAT | os.O_RDWR, 0o600) + try: + fcntl.flock(lock_fd, fcntl.LOCK_EX) + if path.is_file(): + existing = _read(path) + return PublicationReceipt(publication_id, existing["state"], path) + occupied = publication_for_skill(skill) + if occupied and occupied.get("state") in ACTIVE_STATES: + raise ValueError(f"publication is already in progress for '{skill}'") + try: + _publish(path, record) + except FileExistsError: + existing = _read(path) + return PublicationReceipt(publication_id, existing["state"], path) + finally: + fcntl.flock(lock_fd, fcntl.LOCK_UN) + os.close(lock_fd) + return PublicationReceipt(publication_id, record["state"], path) diff --git a/ingot/optimize/publisher.py b/ingot/optimize/publisher.py new file mode 100644 index 0000000..e03a37f --- /dev/null +++ b/ingot/optimize/publisher.py @@ -0,0 +1,713 @@ +"""Publish approved skill bytes through Git, then finalize only the authorized vault revision. + +Two backends sit behind one publication contract. `local` is the default: the vault is a Git +repository on this machine, nothing leaves it, and the approval that queued the receipt is the only +gate. `forge` is opt-in and preserves the GitHub lane — push, pull request, merge — for deployments +that want the vault mirrored and the activation anchored somewhere a local administrator cannot +rewrite. The backend is always selected explicitly; a vault that happens to have an `origin` must +never start opening pull requests on its own. +""" +from __future__ import annotations + +import argparse +import json +import os +import shutil +import subprocess +import sys +import time +from dataclasses import dataclass, field +from pathlib import Path + +from ingot import delivery +from ingot.mcp_server.registry import skill_revision, write_components, write_skill_md +from ingot.optimize import promote +from ingot.optimize import publication +from ingot.optimize import tree + + +BACKENDS = ("local", "forge") +DEFAULT_BACKEND = "local" +DEFAULT_REMOTE = "origin" +DEFAULT_BRANCH = "main" +PUBLISHER_IDENTITY = ("-c", "user.name=Ingot Publisher", "-c", "user.email=ingot@local.invalid") + + +class ConfigurationError(ValueError): + """A publisher that must refuse to start rather than queue approvals it can never publish.""" + + +def _run(command: list[str], *, cwd: Path) -> str: + result = subprocess.run(command, cwd=cwd, capture_output=True, text=True) + if result.returncode: + label = " ".join(command[:2]) + detail = result.stderr.strip().splitlines()[-1] if result.stderr.strip() else "command failed" + raise RuntimeError(f"{label}: {detail}") + return result.stdout.strip() + + +def _forge_urls(repository: str) -> set[str]: + """Every spelling a checkout of one GitHub repository legitimately uses for its remote.""" + return {f"https://github.com/{repository}.git", f"https://github.com/{repository}", + f"git@github.com:{repository}.git", f"git@github.com:{repository}", + f"ssh://git@github.com/{repository}.git", f"ssh://git@github.com/{repository}"} + + +class VaultRepo: + """The vault checkout, which is also the served checkout. + + `remote` is None in local mode, where this repository is the whole history there is: no origin + to verify, nothing to fetch, and `main` is the only authority.""" + + def __init__(self, path: Path, *, remote: str | None = None, branch: str = DEFAULT_BRANCH): + self.path = path + self.remote = remote + self.branch = branch + + @property + def base(self) -> str: + """The ref publications are cut from and fast-forwarded onto.""" + return f"{self.remote}/{self.branch}" if self.remote else self.branch + + def git(self, *args: str) -> str: + return _run(["git", *args], cwd=self.path) + + def _is_ancestor(self, commit: str, of: str) -> bool: + result = subprocess.run(["git", "merge-base", "--is-ancestor", commit, of], + cwd=self.path, capture_output=True) + return result.returncode == 0 + + def contains(self, commit: str) -> bool: + """Whether the base ref already carries this commit, directly or beneath a later one.""" + return self._is_ancestor(commit, self.base) + + def start_point(self, branch: str) -> str: + """Where this publication's worktree is cut from. + + An existing publication branch is reused so a retry is idempotent — the absence rollback + depends on it, because its second attempt must start from a tree where the skill is already + gone. In local mode a branch that no longer descends from `main` (someone committed to the + vault directly in between) can never fast-forward, so it is abandoned and recut from `main` + rather than left to wedge the receipt on every retry. The publication is re-materialized + against the new champion and refused if it no longer matches.""" + if self.remote: + found = self.git("ls-remote", "--heads", self.remote, f"refs/heads/{branch}") + return f"{self.remote}/{branch}" if found else self.base + exists = subprocess.run(["git", "rev-parse", "--verify", "--quiet", f"refs/heads/{branch}"], + cwd=self.path, capture_output=True).returncode == 0 + return branch if exists and self._is_ancestor(self.base, branch) else self.base + + @classmethod + def open(cls, path: Path, *, remote: str | None = None, expected_remotes: set[str] | None = None, + branch: str = DEFAULT_BRANCH, allow_behind: bool = False, + sync: bool = False) -> "VaultRepo": + """Open the served vault checkout, refusing anything the publisher must not build on. + + `sync` fast-forwards a checkout that is merely behind its remote. The vault has other + writers — a person committing directly, another machine — and the served library is by + definition a mirror of vault `main`, so falling behind is normal and must not block + publishing. Only `_prepare` may sync: in `_finalize` the fast-forward is the activation step + itself, and syncing early would discard the champion bytes before they are snapshotted as + the rollback target. In local mode there is no remote, so both are no-ops.""" + resolved = path.expanduser().resolve(strict=True) + repo = cls(resolved, remote=remote, branch=branch) + if remote is not None and expected_remotes is not None: + if repo.git("remote", "get-url", remote) not in expected_remotes: + raise ValueError(f"vault {remote} is not the configured vault repository") + try: + head = repo.git("symbolic-ref", "--short", "HEAD") + except RuntimeError as exc: + raise ValueError("vault checkout is detached") from exc + if head != branch: + raise ValueError(f"vault checkout must remain on {branch}") + if repo.git("status", "--porcelain"): + raise ValueError("vault checkout is dirty") + if remote is None: + return repo + repo.git("fetch", remote, branch) + head_commit = repo.git("rev-parse", "HEAD") + upstream = repo.git("rev-parse", repo.base) + if head_commit != upstream: + if not repo._is_ancestor(head_commit, repo.base): + raise ValueError(f"vault {branch} diverged from {repo.base}") + if sync: + repo.git("merge", "--ff-only", repo.base) + elif not allow_behind: + raise ValueError(f"vault HEAD must equal {repo.base}") + return repo + + +class LocalBackend: + """The default. The vault is a local Git repository and no network is involved at any point. + + There is no `awaiting_merge` state: the commit is authorized the moment it exists, because the + human gate is the approval that queued the receipt, not a second review of the same bytes.""" + + name = "local" + remote = None + expected_remotes = None + + def __init__(self, *, branch: str = DEFAULT_BRANCH): + self.branch = branch + + def authorize(self, repo: VaultRepo, record: dict, branch: str, + workspace: Path | None = None) -> str | None: + return repo.git("rev-parse", f"refs/heads/{branch}") + + def advance(self, repo: VaultRepo, record: dict, commit: str) -> None: + """Fast-forward only. A vault whose `main` moved under the publication is a refusal, never + a rebase, a merge commit, or a reset: the operator resolves it and the receipt retries.""" + repo.git("merge", "--ff-only", commit) + + def retire(self, repo: VaultRepo, record: dict) -> None: + _retire(repo, record, (("branch", "-D", record.get("branch", "")),)) + + +class GitHub: + """`gh` against the vault checkout, whose authenticated host credentials this process reuses.""" + + def __init__(self, vault_dir: Path, repository: str, *, base: str = DEFAULT_BRANCH): + self.vault_dir = vault_dir + self.repository = repository + self.base = base + + def _json(self, args: list[str]): + output = _run(["gh", *args], cwd=self.vault_dir) + return json.loads(output) if output else None + + def create_or_find(self, branch: str, publication_id: str) -> int: + found = self._json(["pr", "list", "--repo", self.repository, "--head", branch, + "--state", "all", "--json", "number,state"]) or [] + if found: + # Someone closing the vault pull request has rejected the publication. Opening a second + # one for the same branch would overrule that, so this refuses and leaves the reason on + # the receipt instead. + if found[0].get("state") == "CLOSED": + raise ValueError(f"the vault pull request for {branch} was closed without merging") + return int(found[0]["number"]) + output = _run(["gh", "pr", "create", "--repo", self.repository, "--base", self.base, + "--head", branch, "--title", f"Publish Ingot approval {publication_id}", + "--body", f"Evidence-gated publication `{publication_id}`."], + cwd=self.vault_dir) + return int(output.rstrip("/").split("/")[-1]) + + def enable_auto_merge(self, pr: int) -> None: + _run(["gh", "pr", "merge", str(pr), "--repo", self.repository, "--auto", "--squash"], + cwd=self.vault_dir) + + def merged_commit(self, pr: int) -> str | None: + data = self._json(["pr", "view", str(pr), "--repo", self.repository, + "--json", "state,mergeCommit"]) + if not data or data.get("state") != "MERGED": + return None + return (data.get("mergeCommit") or {}).get("oid") + + +class ForgeBackend: + """Opt-in. Publication authority is a merged pull request in a configured Git forge.""" + + name = "forge" + + def __init__(self, github=None, *, repository: str, remote: str = DEFAULT_REMOTE, + branch: str = DEFAULT_BRANCH, expected_remotes: set[str] | None = None): + if not repository: + raise ConfigurationError("the forge backend requires a repository " + "(INGOT_FORGE_REPOSITORY)") + self.repository = repository + self.remote = remote + self.branch = branch + self.expected_remotes = expected_remotes or _forge_urls(repository) + self.github = github + + def _forge(self, repo: VaultRepo) -> GitHub: + if self.github is None: + self.github = GitHub(repo.path, self.repository, base=self.branch) + return self.github + + def authorize(self, repo: VaultRepo, record: dict, branch: str, + workspace: Path | None = None) -> str | None: + """With a workspace, submit the branch for authority; without one, ask whether it landed. + + A vault that does not allow auto-merge is not a failure: the pull request is open and a + human merging it is the gate this whole lane exists for. Wedging the receipt here would + strand an approval that is already waiting on GitHub, and nothing downstream acts until the + merge is visible, so the guarantee is unchanged either way.""" + github = self._forge(repo) + if workspace is not None: + _run(["git", "push", self.remote, f"HEAD:refs/heads/{branch}"], cwd=workspace) + pr = github.create_or_find(branch, record["id"]) + publication.update_publication(record["id"], branch=branch, pr=pr) + auto_merge, note = True, "" + try: + github.enable_auto_merge(pr) + except RuntimeError as exc: + auto_merge, note = False, f"auto-merge unavailable, waiting on a human merge: {exc}" + publication.update_publication(record["id"], pr=pr, auto_merge=auto_merge, note=note) + return None + merged = github.merged_commit(int(record["pr"])) + if not merged: + return None + # Fetch here rather than relying on the caller having opened the vault: a merge that landed + # a second ago is not an ancestor of a stale `origin/main`, and refusing it would fail the + # receipt for the one outcome this lane is waiting for. + repo.git("fetch", self.remote, self.branch) + if not repo.contains(merged): + raise ValueError(f"the approved merge commit is not part of {repo.base}") + return merged + + def advance(self, repo: VaultRepo, record: dict, commit: str) -> None: + # The authorized commit is already on the base ref, verified above, so the served checkout + # advances by fast-forwarding onto it rather than onto the branch. + repo.git("merge", "--ff-only", repo.base) + + def retire(self, repo: VaultRepo, record: dict) -> None: + branch = record.get("branch", "") + _retire(repo, record, (("push", self.remote, "--delete", branch), ("branch", "-D", branch))) + + +def _retire(repo: VaultRepo, record: dict, commands) -> None: + """Drop the publication branch once its commit is served. + + Every publication opens one, so leaving them behind grows the vault's branch list without + bound. This runs after the receipt is durable and never raises: the publication is already + active, and an uncleaned branch must not turn a completed activation into a retry.""" + if not record.get("branch"): + return + for command in commands: + try: + repo.git(*command) + except RuntimeError as exc: + print(f"[publisher] {record['id']}: could not remove {record['branch']}: {exc}", + flush=True) + + +class Publisher: + def __init__(self, vault_dir: Path, *, backend=None, targets=None): + self.vault_dir = vault_dir.expanduser().resolve() + self.backend = backend if backend is not None else LocalBackend() + # An unconfigured publisher delivers to the vault and nowhere else, which is what every + # deployment before delivery targets existed already did. + self.targets = (tuple(targets) if targets is not None + else delivery.parse_targets("", vault=self.vault_dir)) + + def _repo(self) -> VaultRepo: + """The vault checkout, unvalidated. For questions that do not build on it.""" + return VaultRepo(self.vault_dir, remote=self.backend.remote, branch=self.backend.branch) + + def _open(self, **kwargs) -> VaultRepo: + return VaultRepo.open(self.vault_dir, remote=self.backend.remote, + expected_remotes=self.backend.expected_remotes, + branch=self.backend.branch, **kwargs) + + @staticmethod + def _branch(record: dict) -> str: + return f"ingot/{record['id']}" + + def _workspace(self, record: dict) -> Path: + return publication.publications_dir() / "worktrees" / record["id"] + + @staticmethod + def _register(root: Path, skill: str, *, present: bool) -> None: + """Keep the vault registry in step with what the publication adds or removes. + + A stale entry left by an earlier removal would otherwise survive: the skill would land in + the vault still marked for removal, and the projection would drop what was just approved. A + curated `keep` entry is left exactly as the operator wrote it.""" + path = root / "registry.json" + registry = json.loads(path.read_text(encoding="utf-8")) + if present: + entry = registry.get(skill) + if not isinstance(entry, dict) or entry.get("disposition") != "keep": + registry[skill] = {"disposition": "keep", "reason": "Approved through Ingot."} + else: + registry.pop(skill, None) + path.write_text(json.dumps(registry, indent=2, sort_keys=True) + "\n") + + def _materialize_rollback(self, root: Path, record: dict) -> None: + """Restore a stored snapshot wholesale. + + Writing the recorded components back would only restore text the optimizer can rewrite: a + file the displaced revision added, or any non-text file, would survive the rollback and the + revision check below would then refuse it. The snapshot is a complete copy of the skill, so + replacing the directory with it is both simpler and exact.""" + skill = root / record["skill"] + if skill.is_dir(): + shutil.rmtree(skill) + if record["candidate_revision"] == promote.ABSENT_REVISION: + self._register(root, record["skill"], present=False) + return + source = promote._rollback_source(record["skill"], record["candidate_revision"]) + shutil.copytree(source, skill, symlinks=True) + self._register(root, record["skill"], present=True) + if skill_revision(skill) != record["candidate_revision"]: + raise ValueError("restored vault package does not match the approved revision") + + def _materialize(self, root: Path, record: dict) -> None: + if record["action"] == "rollback": + return self._materialize_rollback(root, record) + skill = root / record["skill"] + components = record["components"] + candidate_tree = record.get("tree") + if record["kind"] == "creation": + if skill.is_dir() and skill_revision(skill) == record["candidate_revision"]: + return + if skill.exists() or skill.is_symlink(): + raise ValueError(f"skill '{record['skill']}' already exists in the vault") + if candidate_tree: + # Every file the operator ingested, byte for byte, verified against the receipt on + # the way in. `write_components` is deliberately not called afterwards: it would + # rewrite each text file from a decoded copy, and a decoded copy is exactly what + # this exists to stop the vault from serving. + tree.materialize_creation(candidate_tree, components, record["skill"], skill) + self._register(root, record["skill"], present=True) + if skill_revision(skill) != record["candidate_revision"]: + raise ValueError( + "materialized vault package does not match the approved revision") + return + skill.mkdir(parents=True) + try: + metadata = json.loads(components.get("frontmatter", "{}")) + except (TypeError, ValueError) as exc: + raise ValueError("creation frontmatter is not valid JSON") from exc + metadata["name"] = record["skill"] + metadata["description"] = components["description"] + write_skill_md(skill / "SKILL.md", metadata, components["body"]) + self._register(root, record["skill"], present=True) + else: + if not skill.is_dir(): + raise ValueError(f"skill '{record['skill']}' is absent from the vault") + current = skill_revision(skill) + if current == record["candidate_revision"]: + return + if current != record["expected_champion"]: + raise ValueError("vault champion does not match the approved revision") + write_components(skill, components) + if skill_revision(skill) != record["candidate_revision"]: + raise ValueError("materialized vault package does not match the approved revision") + + def _prepare(self, record: dict) -> str: + repo = self._open(sync=True) + branch = record.get("branch") or self._branch(record) + publication.update_publication(record["id"], state="publishing", branch=branch, + attempts=record.get("attempts", 0) + 1, last_error="") + workspace = self._workspace(record) + workspace.parent.mkdir(parents=True, exist_ok=True) + start = repo.start_point(branch) + # A run killed mid-publication leaves its worktree behind. Reusing it would let whatever it + # holds — a half-written materialization, an unrelated edit — ride into the publication, + # so every attempt starts from a worktree cut fresh from the recorded start point. + if workspace.exists(): + try: + repo.git("worktree", "remove", "--force", str(workspace)) + except RuntimeError: + shutil.rmtree(workspace, ignore_errors=True) + repo.git("worktree", "prune") + repo.git("worktree", "add", "-B", branch, str(workspace), start) + self._materialize(workspace, record) + _run([sys.executable, "scripts/validate.py"], cwd=workspace) + # Stage the whole worktree rather than naming the skill: a retry of an absence rollback + # starts from the publication branch, where the skill directory is already gone, and naming + # a pathspec that matches nothing is a fatal `git add`. The worktree is cut fresh from the + # start ref and `_materialize` is its only writer, so there is nothing else in it to stage. + _run(["git", "add", "-A", "."], cwd=workspace) + # An empty diff is a no-op, not an error: re-publishing a revision the vault already + # carries must converge on the existing branch tip. + if _run(["git", "status", "--porcelain"], cwd=workspace): + _run(["git", *PUBLISHER_IDENTITY, "commit", "-m", + f"Publish {record['skill']} via Ingot {record['id']}"], cwd=workspace) + authorized = self.backend.authorize(repo, record, branch, workspace) + repo.git("worktree", "remove", "--force", str(workspace)) + branch_commit = repo.git("rev-parse", f"refs/heads/{branch}") + if authorized is None: + publication.update_publication(record["id"], state="awaiting_merge", branch=branch, + branch_commit=branch_commit, last_error="") + return "awaiting_merge" + # The local backend has no external gate to wait on, so one pass runs both halves. A crash + # in between leaves the receipt at `publishing` with the branch committed, and the next + # pass re-enters here, finds the branch, re-authorizes, and finalizes. + publication.update_publication(record["id"], branch=branch, branch_commit=branch_commit, + last_error="") + return self._finalize(publication.load_publication(record["id"])) + + def _serves(self, record: dict, revision: str) -> bool: + """Whether the served checkout is exactly at one recorded revision. Absence is a revision: + a creation's champion and an absence rollback's candidate are both 'no directory at all'.""" + skill = self.vault_dir / record["skill"] + if revision == promote.ABSENT_REVISION: + return not skill.exists() + return skill.is_dir() and skill_revision(skill) == revision + + def _deliver(self, record: dict) -> None: + """Install the approved revision at every configured target. + + Runs after the vault serves the candidate and before the receipt is marked active, so a + target that cannot be written leaves a release that retries rather than one that claims to + be finished. Each target's outcome is recorded separately: a deployment with two targets + where one failed must be able to say which one. + + The source is the vault's own copy, already checked against the receipt by `_serves`. A + failure raises, and `process` records it on the receipt as `last_error`.""" + skill, revision = record["skill"], record["candidate_revision"] + source = None if revision == promote.ABSENT_REVISION else self.vault_dir / skill + if source is not None and not source.is_dir(): + raise ValueError(f"the vault does not hold '{skill}' to deliver") + delivered = dict(record.get("delivery") or {}) + for target in self.targets: + entry = {"kind": target.kind, "root": str(target.root), "at": int(time.time())} + try: + delivery.install(target, skill, source, revision) + except Exception as exc: + delivered[target.name] = {**entry, "state": "failed", "error": str(exc), + "revision": delivery.observed(target, skill)} + publication.update_publication(record["id"], delivery=delivered) + raise + delivered[target.name] = {**entry, "state": "delivered", "revision": revision} + publication.update_publication(record["id"], delivery=delivered) + + def _finalize(self, record: dict) -> str: + branch = record.get("branch") or self._branch(record) + # Ask whether authority exists before validating the checkout. A receipt waiting on a forge + # merge is polled every few seconds and opening the vault fetches, so validating first put + # a network round trip -- and a failure mode -- in front of a question that needs neither. + authorized = self.backend.authorize(self._repo(), record, branch) + if not authorized: + return "awaiting_merge" + repo = self._open(allow_behind=True) + # Check the review slot before anything is snapshotted or fast-forwarded. Validating it + # afterwards would leave a receipt that no longer matches its review having already moved + # the served bytes, which is the one thing this lane exists to prevent. Only an approval + # owns a review slot: a rollback publishes a stored snapshot and must leave an unrelated + # pending challenger for that skill reviewable. + pending = promote.load_pending(record["skill"]) if record["action"] == "promote" else None + if pending is not None: + recorded = ((pending.get("evidence") or {}).get("challenger") or {}).get("revision") + if recorded != record["candidate_revision"]: + raise ValueError("pending review no longer matches the publication") + # A crash between the fast-forward and this receipt leaves the approved revision already + # served. Re-snapshotting and re-merging then would refuse a champion that is legitimately + # gone, so a checkout already at the candidate skips straight to finalizing the receipt. + if not self._serves(record, record["candidate_revision"]): + if not self._serves(record, record["expected_champion"]): + raise ValueError("served champion changed before vault activation") + if record["expected_champion"] == promote.ABSENT_REVISION: + promote._snapshot_absence(record["skill"]) + else: + promote._snapshot(self.vault_dir / record["skill"], record["skill"], + record["expected_champion"]) + self.backend.advance(repo, record, authorized) + if not self._serves(record, record["candidate_revision"]): + raise ValueError("the fast-forwarded vault does not serve the approved revision") + self._deliver(record) + if pending is not None: + promote.pending_path(record["skill"]).unlink() + promote._audit_best_effort(record["action"] if record["action"] == "rollback" else "approve", + record["skill"], record["candidate_revision"], record["actor"]) + publication.update_publication(record["id"], state="active", merged_commit=authorized, + activated=int(time.time()), last_error="") + self.backend.retire(repo, record) + return "active" + + def process(self, publication_id: str) -> str: + record = publication.load_publication(publication_id) + if not record: + raise ValueError(f"unknown publication: {publication_id}") + try: + # The queue validates both of these when it writes a receipt. Re-checking them here + # keeps a receipt that was edited or corrupted on disk from steering a filesystem path + # or reaching an activation branch it was never approved for. + promote.check_slug(record["skill"]) + if record.get("action") not in publication.ACTIONS: + raise ValueError(f"invalid publication action: {record.get('action')!r}") + if record["state"] in {"approved_publishing", "publishing"}: + return self._prepare(record) + if record["state"] == "awaiting_merge": + return self._finalize(record) + return record["state"] + except Exception as exc: + current = publication.load_publication(publication_id) or record + publication.update_publication( + publication_id, attempts=current.get("attempts", 0) + 1, last_error=str(exc)) + if isinstance(exc, RuntimeError): + raise + raise RuntimeError(f"publication {publication_id}: {exc}") from exc + + +@dataclass(frozen=True) +class PublisherConfig: + """What the publisher was told to be, before anything checks whether it can be.""" + + backend: str + vault_dir: Path + forge_repository: str | None = None + forge_remote: str = DEFAULT_REMOTE + forge_branch: str = DEFAULT_BRANCH + delivery_targets: tuple = field(default=()) + warnings: tuple[str, ...] = field(default=()) + + def build(self) -> Publisher: + targets = self.delivery_targets or None + if self.backend == "forge": + return Publisher(self.vault_dir, targets=targets, backend=ForgeBackend( + repository=self.forge_repository, remote=self.forge_remote, + branch=self.forge_branch)) + return Publisher(self.vault_dir, targets=targets, backend=LocalBackend()) + + +FORGE_KEYS = ("INGOT_FORGE_REPOSITORY", "INGOT_FORGE_REMOTE", "INGOT_FORGE_BRANCH") + + +def load_config(env: dict | None = None, *, backend: str | None = None, + vault: Path | None = None) -> PublisherConfig: + """Explicit argument, then environment. There is no fallback to a writable default path. + + A publisher that guesses its vault would happily own the demo directory, which is the one + outcome managed mode exists to prevent.""" + env = os.environ if env is None else env + chosen = backend or env.get("INGOT_PUBLISH_BACKEND") or DEFAULT_BACKEND + if chosen not in BACKENDS: + raise ConfigurationError(f"unknown publication backend {chosen!r}; " + f"expected one of {', '.join(BACKENDS)}") + warnings = [] + path = vault or env.get("INGOT_VAULT_PATH") or env.get("VAULT_DIR") + if not path: + raise ConfigurationError("no vault configured; set INGOT_VAULT_PATH to the checkout this " + "publisher owns, or pass --vault") + if not vault and not env.get("INGOT_VAULT_PATH") and env.get("VAULT_DIR"): + warnings.append("VAULT_DIR is the pre-backend name for INGOT_VAULT_PATH; prefer the latter") + repository = env.get("INGOT_FORGE_REPOSITORY") + if chosen == "local": + # A half-finished forge configuration under the local backend looks active and is not. Say + # so rather than letting someone believe their pull requests are being opened. + stray = [key for key in FORGE_KEYS if env.get(key)] + if stray: + warnings.append(f"{', '.join(stray)} set but the backend is local; forge settings are " + f"inert until INGOT_PUBLISH_BACKEND=forge") + elif not repository: + raise ConfigurationError("the forge backend requires INGOT_FORGE_REPOSITORY=owner/repo") + vault_dir = Path(path).expanduser() + try: + targets = delivery.parse_targets(env.get(delivery.TARGETS) or "", vault=vault_dir) + except ValueError as exc: + raise ConfigurationError(f"{delivery.TARGETS}: {exc}") from exc + return PublisherConfig( + backend=chosen, vault_dir=Path(path), forge_repository=repository, + forge_remote=env.get("INGOT_FORGE_REMOTE") or DEFAULT_REMOTE, + forge_branch=env.get("INGOT_FORGE_BRANCH") or DEFAULT_BRANCH, + delivery_targets=targets, warnings=tuple(warnings)) + + +def validate(config: PublisherConfig) -> Publisher: + """Refuse to start rather than fail on the first approval. + + An approval queued against a publisher that cannot publish is the stalled-lane failure this + deployment has already hit once: the console reports the change accepted, the receipt sits at + `approved_publishing` forever, and nothing says why.""" + publisher = config.build() + if config.backend == "forge": + if not shutil.which("gh"): + raise ConfigurationError("the forge backend needs the GitHub CLI (`gh`) on PATH") + try: + _run(["gh", "auth", "status"], cwd=Path.cwd()) + except RuntimeError as exc: + raise ConfigurationError(f"`gh` is not authenticated: {exc}") from exc + try: + _run(["gh", "repo", "view", config.forge_repository, "--json", "name"], cwd=Path.cwd()) + except RuntimeError as exc: + raise ConfigurationError( + f"the configured forge repository {config.forge_repository!r} does not " + f"resolve: {exc}") from exc + try: + publisher._open(allow_behind=True) + except (ValueError, OSError) as exc: + raise ConfigurationError(f"vault at {config.vault_dir}: {exc}") from exc + for target in config.delivery_targets: + if target.kind != delivery.FILESYSTEM: + continue + # Created here rather than on the first approval. A delivery root that cannot be made or + # written strands a change that a human has already approved, and this is the one place + # that can say so before anybody approves anything. + try: + target.root.mkdir(parents=True, exist_ok=True) + except OSError as exc: + raise ConfigurationError( + f"delivery target {target.name!r}: cannot create {target.root}: {exc}") from exc + if not os.access(target.root, os.W_OK | os.X_OK): + raise ConfigurationError(f"delivery target {target.name!r}: {target.root} is not " + f"writable by uid {os.getuid()}") + validator = config.vault_dir / "scripts" / "validate.py" + if not validator.is_file(): + raise ConfigurationError( + f"the vault has no validator at {validator}; every publication runs it before " + f"committing, so a missing one is a configuration error and not a silent skip " + f"(`ingot vault init {config.vault_dir}` writes a default)") + return publisher + + +def unreadable_queue(directory: Path) -> str | None: + """Why the receipt store cannot be read, or None. + + `Path.glob` swallows `PermissionError`, so a queue this process cannot list is indistinguishable + from an empty one: approvals pile up in the console while the publisher reports nothing at all. + That is the exact shape of the deployment failure on the host — receipts written by a container + running as root, read by a publisher running as an ordinary user.""" + if not directory.exists(): + return None + if not os.access(directory, os.R_OK | os.X_OK): + return (f"cannot read the receipt store at {directory} as uid {os.getuid()}; approvals will " + f"queue and never publish (the console writes it at mode 0700, so both must run as " + f"the same user)") + return None + + +def watch(publisher: Publisher, interval: float = 5.0) -> None: + reported = None + while True: + blocked = unreadable_queue(publication.publications_dir()) + if blocked != reported: # report each transition once, not every poll + print(f"[publisher] {blocked}" if blocked else "[publisher] receipt store readable again", + flush=True) + reported = blocked + if not blocked: + for path in sorted(publication.publications_dir().glob("*.json")): + try: + publisher.process(path.stem) + except Exception as exc: + print(f"[publisher] {path.stem}: {exc}", flush=True) + time.sleep(interval) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + prog="python -m ingot.optimize.publisher", + description="The one writer of the served skill library.") + parser.add_argument("--watch", action="store_true") + parser.add_argument("--backend", choices=BACKENDS, default=None, + help="publication backend (default: $INGOT_PUBLISH_BACKEND, else local)") + parser.add_argument("--vault", type=Path, default=None, + help="the vault checkout this publisher owns " + "(default: $INGOT_VAULT_PATH)") + parser.add_argument("publication_id", nargs="?") + args = parser.parse_args(argv) + try: + config = load_config(backend=args.backend, vault=args.vault) + for warning in config.warnings: + print(f"[publisher] warning: {warning}", flush=True) + publisher = validate(config) + except ConfigurationError as exc: + print(f"[publisher] refusing to start: {exc}", file=sys.stderr, flush=True) + return 2 + print(f"[publisher] backend={config.backend} vault={config.vault_dir}", flush=True) + for target in config.delivery_targets: + print(f"[publisher] delivering to {target.name} ({target.kind}) at {target.root}", + flush=True) + if args.watch: + watch(publisher) + return 0 + if not args.publication_id: + parser.error("publication_id is required without --watch") + print(publisher.process(args.publication_id)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/ingot/optimize/requirements-harbor-gateway.txt b/ingot/optimize/requirements-harbor-gateway.txt new file mode 100644 index 0000000..4a37abd --- /dev/null +++ b/ingot/optimize/requirements-harbor-gateway.txt @@ -0,0 +1,3 @@ +# Dedicated, ignored Dell-only runtime for Harbor's local role compatibility gateway. +# Keep the proxy extra explicit: LiteLLM's base package does not install its proxy server. +litellm[proxy]==1.93.0 diff --git a/ingot/optimize/retrospective.py b/ingot/optimize/retrospective.py new file mode 100644 index 0000000..222efa2 --- /dev/null +++ b/ingot/optimize/retrospective.py @@ -0,0 +1,306 @@ +"""Turn a skill-retrospective finding into an inert, revision-bound Ingot challenger.""" +from __future__ import annotations + +import hashlib +import json +import logging +import os +import shutil +import threading +import time +import uuid +from pathlib import Path + +from ingot.mcp_server.registry import (load_skills, optimizable_components, resolve_skill_dir, + skill_revision) +from ingot.optimize import promote +from ingot.optimize.evidence import recorded_path +from ingot import paths + +SCHEMA = "ingot/retrospective-proposal/v1" + + +def evidence_dir() -> Path: + return paths.runs() / "evidence" + + +def audit_file() -> Path: + return paths.runs() / "retrospective-audit.jsonl" + +MAX_BODY_CHARS = 200_000 +MAX_FIELD_CHARS = 4_000 +MAX_EVIDENCE_ITEMS = 12 + +logger = logging.getLogger(__name__) +_SUBMIT_LOCK = threading.Lock() + + +def _text(name: str, value: object, *, required: bool = True, + limit: int = MAX_FIELD_CHARS) -> str: + if not isinstance(value, str): + raise ValueError(f"{name} must be a string") + value = value.strip() + if required and not value: + raise ValueError(f"{name} is required") + if len(value) > limit: + raise ValueError(f"{name} exceeds {limit} characters") + return value + + +def _evidence_items(items: object) -> list[str]: + if not isinstance(items, list) or len(items) < 2: + raise ValueError("evidence must contain at least two concrete repeat items") + if len(items) > MAX_EVIDENCE_ITEMS: + raise ValueError(f"evidence exceeds {MAX_EVIDENCE_ITEMS} items") + evidence = [_text(f"evidence[{index}]", item) for index, item in enumerate(items)] + if len(set(evidence)) != len(evidence): + raise ValueError("evidence items must be distinct") + return evidence + + +def _current_skill(skill: str): + try: + skill_dir = resolve_skill_dir(skill) + except LookupError as exc: + raise ValueError(f"no indexed skill named '{skill}'") from exc + current = next((item for item in load_skills() if item.name == skill), None) + if current is None: + raise ValueError(f"no indexed skill named '{skill}'") + return current, skill_dir + + +def _proposal_id(payload: dict) -> str: + canonical = json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=False) + return hashlib.sha256(canonical.encode()).hexdigest()[:24] + + +def _gate(evidence: list[str]) -> dict: + """Record the admission basis without presenting it as held-out quality evidence.""" + return { + "promotable": True, + "blocked": [], + "warnings": ["Retrospective evidence only; no held-out A/B comparison was run."], + "kind": "retrospective_admission", + "admission": {"pressure_verification": "passed", "evidence_items": len(evidence)}, + } + + +def _render_markdown(skill: str, proposal: dict, gate: dict) -> str: + verification = proposal["verification"] + lines = [ + f"# Retrospective proposal: {skill}", + "", + f"**Schema:** `{SCHEMA}`", + f"**Proposal:** `{proposal['proposal_id']}`", + f"**Gate:** {'PASS' if gate['promotable'] else 'BLOCKED'}", + f"**Producer:** {proposal['producer']}", + f"**Live caller:** {proposal['caller']}", + "", + "## Finding", + "", + proposal["summary"], + "", + f"- Trigger: {proposal['trigger']}", + f"- Minimal reusable content: {proposal['minimal_content']}", + f"- Risk: {proposal['risk']}", + "", + "## Evidence", + "", + *[f"- {item}" for item in proposal["evidence"]], + "", + "## Pressure scenario", + "", + proposal["pressure_scenario"], + "", + "## Verification", + "", + f"- Status: **{verification['status'].upper()}**", + f"- Command: `{verification['command']}`", + f"- Result: {verification['result']}", + "", + "Submission only quarantines this challenger. Human approval remains required.", + "", + ] + return "\n".join(lines) + + +def _write_evidence(skill: str, proposal: dict, gate: dict) -> dict[str, str]: + root = evidence_dir() / skill / f"retrospective-{proposal['proposal_id']}" + root.mkdir(parents=True, exist_ok=True) + bundle = { + "schema_version": SCHEMA, + "skill": skill, + "created": proposal["created"], + "proposal": proposal, + "gate": gate, + } + targets = ( + (root / "evidence.json", json.dumps(bundle, indent=2, ensure_ascii=False) + "\n"), + (root / "EVIDENCE.md", _render_markdown(skill, proposal, gate)), + ) + for path, content in targets: + temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp") + temporary.write_text(content, encoding="utf-8") + temporary.replace(path) + return {"json": recorded_path(targets[0][0]), "markdown": recorded_path(targets[1][0])} + + +def _publish_pending(skill: str, record: dict) -> None: + """Create the review slot without replacing a writer that wins the race. + + `save_pending` intentionally archives and replaces another pass. Retrospective submissions have + weaker authority: an agent may propose, but it may not displace something a human can review. + A hard link gives that policy an atomic create-if-absent operation on the queue filesystem. + """ + promote.pending_dir().mkdir(parents=True, exist_ok=True) + destination = promote.pending_path(skill) + temporary = destination.with_name(f".{destination.name}.{uuid.uuid4().hex}.tmp") + with temporary.open("w", encoding="utf-8") as output: + output.write(json.dumps(record, indent=2, ensure_ascii=False) + "\n") + output.flush() + os.fsync(output.fileno()) + try: + os.link(temporary, destination) + except FileExistsError: + raise ValueError(f"review slot is occupied for '{skill}'; resolve it before proposing") + finally: + temporary.unlink(missing_ok=True) + + +def _audit(record: dict) -> None: + audit_file().parent.mkdir(parents=True, exist_ok=True) + fd = os.open(audit_file(), os.O_APPEND | os.O_CREAT | os.O_WRONLY, 0o600) + try: + os.fchmod(fd, 0o600) + data = (json.dumps(record, separators=(",", ":")) + "\n").encode() + written = 0 + while written < len(data): + written += os.write(fd, data[written:]) + os.fsync(fd) + finally: + os.close(fd) + + +def submit_skill_update( + *, + skill: str, + champion_revision: str, + challenger_body: str, + challenger_description: str = "", + summary: str, + trigger: str, + minimal_content: str, + producer: str, + caller: str, + evidence: list[str], + pressure_scenario: str, + risk: str, + verification_status: str, + verification_command: str, + verification_result: str, +) -> dict: + """Validate and quarantine one retrospective update; never activate or replace another slot.""" + skill = _text("skill", skill, limit=80) + champion_revision = _text("champion_revision", champion_revision, limit=128) + challenger_body = _text("challenger_body", challenger_body, limit=MAX_BODY_CHARS) + challenger_description = _text( + "challenger_description", challenger_description, required=False, limit=2_000) + verification_status = _text( + "verification_status", verification_status, limit=20).lower() + if verification_status != "passed": + raise ValueError("verification_status must be passed before proposing") + + current, skill_dir = _current_skill(skill) + if current.revision != champion_revision: + raise ValueError("champion revision changed; reload the skill before proposing") + + champion = optimizable_components(skill_dir) + challenger = dict(champion) + challenger["body"] = challenger_body + if challenger_description: + challenger["description"] = challenger_description + changed = sorted(key for key in challenger if challenger.get(key) != champion.get(key)) + if not changed: + raise ValueError("retrospective proposal does not change the skill") + + created = int(time.time()) + proposal = { + "schema_version": SCHEMA, + "created": created, + "summary": _text("summary", summary), + "trigger": _text("trigger", trigger), + "minimal_content": _text("minimal_content", minimal_content), + "producer": _text("producer", producer, limit=200), + "caller": _text("caller", caller, limit=500), + "evidence": _evidence_items(evidence), + "pressure_scenario": _text("pressure_scenario", pressure_scenario), + "risk": _text("risk", risk), + "verification": { + "status": verification_status, + "command": _text("verification_command", verification_command), + "result": _text("verification_result", verification_result), + }, + } + # Submission time is evidence metadata, not proposal identity. Excluding it makes an exact + # retry idempotent even when a transport retries after the clock advances. + identity = { + **{key: value for key, value in proposal.items() + if key not in {"created", "producer", "caller"}}, + "skill": skill, + "champion_revision": champion_revision, + "challenger_revision": skill_revision(skill_dir, challenger), + } + proposal["proposal_id"] = _proposal_id(identity) + gate = _gate(proposal["evidence"]) + record = { + "skill": skill, + "kind": "retrospective", + "created": created, + "changed_components": changed, + "champion_components": champion, + "challenger_components": challenger, + "gate": gate, + "evidence": { + "schema_version": SCHEMA, + "champion": {"revision": champion_revision}, + "challenger": {"revision": identity["challenger_revision"]}, + "gate": gate, + }, + "retrospective": proposal, + } + + with _SUBMIT_LOCK: + existing = promote.load_pending(skill) + if existing: + existing_id = (existing.get("retrospective") or {}).get("proposal_id") + if existing_id == proposal["proposal_id"]: + return {"status": "duplicate", "skill": skill, + "proposal_id": proposal["proposal_id"], "promotable": gate["promotable"]} + raise ValueError(f"review slot is occupied for '{skill}'; resolve it before proposing") + record["evidence_paths"] = _write_evidence(skill, proposal, gate) + evidence_root = evidence_dir() / skill / f"retrospective-{proposal['proposal_id']}" + try: + _publish_pending(skill, record) + except ValueError: + current = promote.load_pending(skill) + current_id = ((current or {}).get("retrospective") or {}).get("proposal_id") + if current_id == proposal["proposal_id"]: + return {"status": "duplicate", "skill": skill, + "proposal_id": proposal["proposal_id"], "promotable": gate["promotable"]} + shutil.rmtree(evidence_root, ignore_errors=True) + raise + except OSError as exc: + shutil.rmtree(evidence_root, ignore_errors=True) + raise RuntimeError( + f"cannot atomically publish the review slot for '{skill}'") from exc + try: + _audit({"schema_version": 1, "ts": int(time.time()), "action": "quarantine", + "skill": skill, "proposal_id": proposal["proposal_id"], + "challenger_revision": identity["challenger_revision"], + "producer": proposal["producer"]}) + except Exception: + logger.warning("Quarantined retrospective proposal %s, but its audit write failed", + proposal["proposal_id"], exc_info=True) + + return {"status": "quarantined", "skill": skill, + "proposal_id": proposal["proposal_id"], "promotable": gate["promotable"]} diff --git a/ingot/optimize/review.py b/ingot/optimize/review.py new file mode 100644 index 0000000..7dec89b --- /dev/null +++ b/ingot/optimize/review.py @@ -0,0 +1,139 @@ +"""Score one skill as it stands and report which checks it fails, without changing anything. + +The optimize loop answers "is this candidate better than the champion". That is the wrong question +when you have just written a skill and want to know whether it is any good, and it is unreachable +anyway until an eval set exists. This runs the skill against its own tasks, grades every answer +against that task's checklist, and ranks the failures by how much they cost -- so the output is a +list of things to fix, not a number. + +It never writes a pending record, never promotes, and never touches the skill directory. Drafting an +eval set is the one side effect, and only when the skill has none (see optimize/draft.py). + + python -m ingot.optimize.review +""" +import argparse +import json +import os +import shutil +import tempfile +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +from ingot.mcp_server.registry import optimizable_components, skill_revision + +from . import SERVE_TEMPLATE, agent_model, resolve_skill_dir +from . import usage as usage_ledger +from .ab import load_tasks +from .compat import _llm +from .judge import DIMENSIONS, invoke_retry, judge +from .rollout import assemble +from ingot import paths + +REVIEW_DIR = paths.runs() / "reviews" +_MAX_WORKERS = 8 + + +def _review_one(llm, system: str, task: dict) -> dict: + """Answer one task with the skill loaded, then grade that answer against the task's checklist.""" + msg = invoke_retry(llm, [("system", system), ("user", task["task"])]) + usage_ledger.add("review", getattr(msg, "usage_metadata", None)) + verdict = judge(task["task"], task.get("rubric", ""), msg.content, + check=task.get("check"), deliverable=task.get("deliverable"), + checklist=task.get("checklist")) + return {"task": task["task"], "score": verdict["score"], "answer": msg.content, + "feedback": verdict["feedback"], "checklist": verdict["checklist"], + "spec": {i["id"]: i for i in (task.get("checklist") or [])}} + + +def findings(results: list[dict]) -> list[dict]: + """Every check that did not clean-pass, worst first. + + Ranked by weight x shortfall rather than by score: a weight-5 check scraping a partial matters + more than a weight-1 check failing outright, and sorting by the raw value buries it.""" + out = [] + for i, result in enumerate(results): + for check_id, graded in result["checklist"].items(): + if graded["value"] >= 1.0: + continue + spec = result["spec"].get(check_id, {}) + weight = float(spec.get("weight", 1)) + out.append({"task_index": i, "task": result["task"], "check": check_id, + "criterion": spec.get("criterion", ""), "weight": weight, + "dimension": spec.get("dimension", ""), "value": graded["value"], + "note": graded["note"], "cost": weight * (1.0 - graded["value"])}) + return sorted(out, key=lambda f: -f["cost"]) + + +def by_dimension(found: list[dict]) -> dict: + """Where the losses concentrate. A skill failing everything on completeness needs a different + edit from one failing scattered correctness checks.""" + totals = {d: 0.0 for d in DIMENSIONS} + for f in found: + if f["dimension"] in totals: + totals[f["dimension"]] += f["cost"] + return {d: round(v, 3) for d, v in sorted(totals.items(), key=lambda kv: -kv[1]) if v} + + +def run_review(skill: str, log=print) -> dict: + usage_ledger.reset() + skill_dir = resolve_skill_dir(skill) + # A review lasts about a minute. Copy once so a concurrent promotion or trusted filesystem edit + # cannot make the revision describe one read while the prompt grades a later read. + with tempfile.TemporaryDirectory(prefix=f"ingot-review-{skill}-") as temporary: + snapshot = Path(temporary) / skill_dir.name + shutil.copytree(skill_dir, snapshot, symlinks=True) + revision = skill_revision(snapshot) + components = optimizable_components(snapshot) + train, holdout, _ = load_tasks( + skill, draft_components=components) # draft against the same revision being graded + tasks = list(train) + list(holdout) # nothing is being generalized to; grade on all + if not tasks: + raise SystemExit(f"'{skill}' has no eval tasks to run.") + model = agent_model() + system = SERVE_TEMPLATE.format(body=assemble(components)) + graded = sum(len(t.get("checklist") or []) for t in tasks) + log(f"[review] '{skill}': {len(tasks)} tasks, {graded or len(tasks) * 4} checks, on {model}") + + llm = _llm(model) + with ThreadPoolExecutor(max_workers=min(_MAX_WORKERS, len(tasks))) as pool: + results = list(pool.map(lambda t: _review_one(llm, system, t), tasks)) + + found = findings(results) + score = sum(r["score"] for r in results) / len(results) + summary = { + "skill": skill, "model": model, "revision": revision, + "score": round(score, 4), "tasks": len(tasks), + "checks": sum(len(r["checklist"]) for r in results), + "failed_checks": len(found), + "by_dimension": by_dimension(found), + "findings": [{k: v for k, v in f.items() if k != "task_index"} for f in found], + "per_task": [{"task": r["task"], "score": round(r["score"], 4)} for r in results], + } + REVIEW_DIR.mkdir(parents=True, exist_ok=True) + path = REVIEW_DIR / f"{skill}.json" + path.write_text(json.dumps(summary, indent=2)) + + log(f"\n[review] {skill}: {score:.3f} over {len(tasks)} tasks " + f"({summary['checks'] - len(found)}/{summary['checks']} checks clean)") + if summary["by_dimension"]: + log("[review] losses by dimension: " + + ", ".join(f"{d} {v}" for d, v in summary["by_dimension"].items())) + for f in found[:10]: + log(f" - {f['check']} (weight {f['weight']:g}, {'partial' if f['value'] else 'fail'})" + f" — {f['note'] or f['criterion']}") + if len(found) > 10: + log(f" … {len(found) - 10} more in {path}") + log(f"[review] wrote {path}") + log(usage_ledger.format_report()) + return summary + + +if __name__ == "__main__": + ap = argparse.ArgumentParser(description="Score a skill against its own eval tasks and report " + "which checks it fails. Changes nothing.") + ap.add_argument("skill") + args = ap.parse_args() + if os.environ.get("REVIEW_REQUIRE_KEY", "1") != "0": + from . import require_openrouter_key + require_openrouter_key() + run_review(args.skill) diff --git a/optimize/rollout.py b/ingot/optimize/rollout.py similarity index 91% rename from optimize/rollout.py rename to ingot/optimize/rollout.py index 8f3f949..dbb211c 100644 --- a/optimize/rollout.py +++ b/ingot/optimize/rollout.py @@ -4,7 +4,7 @@ judge's textual feedback is what the candidate search reads. This module owns the rollout and the teacher (reflection) client only. The candidate search lives in -`optimize.skillopt_loop`; the held-out A/B that produces promotion evidence lives in `optimize.ab`. +`ingot.optimize.skillopt_loop`; the held-out A/B that produces promotion evidence lives in `ingot.optimize.ab`. """ import os @@ -83,7 +83,8 @@ def _rollout(self, system, ex): usage_ledger.add("rollout", getattr(msg, "usage_metadata", None)) answer = msg.content j = judge(ex["task"], ex["rubric"], answer, reference=ex.get("reference", ""), - check=ex.get("check"), deliverable=ex.get("deliverable")) + check=ex.get("check"), deliverable=ex.get("deliverable"), + checklist=ex.get("checklist")) return answer, j["score"], {"task": ex["task"], "output": answer, "feedback": j["feedback"], "dimensions": j["dimensions"]} @@ -112,6 +113,11 @@ def make_reflection_lm(): local OPENROUTER_BASE_URL (vLLM/Ollama) is honored through litellm's generic openai provider. Shared by SkillOpt's body and description passes.""" import litellm + # We pass a provider-prefixed model, but litellm resolves the provider again after the call + # for cost tracking, from the response's bare slug — which it cannot map, so it prints a red + # "Provider List: ..." banner around calls that succeeded. Nine of them in a 38-line run made + # a healthy loop read as broken. This suppresses the hint only; real errors still raise. + litellm.suppress_debug_info = True litellm.success_callback = [_track_reflection] base = teacher_base_url() diff --git a/optimize/routing.py b/ingot/optimize/routing.py similarity index 96% rename from optimize/routing.py rename to ingot/optimize/routing.py index 0874e19..f608b22 100644 --- a/optimize/routing.py +++ b/ingot/optimize/routing.py @@ -12,17 +12,18 @@ import yaml -from mcp_server.registry import SKILLS_DIR, load_skills, optimizable_components, skill_revision -from mcp_server.router import Router +from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision +from ingot.mcp_server.router import Router +from . import resolve_skill_dir from . import usage as usage_ledger from .ab import COLLISION_SCORE, TASKS_DIR, _description_shadows from .evidence import RoutingRun, build_routing_evidence, recorded_path, write_evidence from .promote import save_pending from .rollout import make_reflection_lm from . import skillopt_bridge as sk +from ingot import paths -EVIDENCE_DIR = Path(__file__).resolve().parent.parent / "runs" / "evidence" _DIAGNOSIS = ("The `description` is a routing trigger matched by embedding similarity against the " "user's task. Adjust trigger phrases so expected tasks match and unrelated ones " @@ -161,9 +162,7 @@ def routing_gate(skill: str, metrics: dict, challenger: dict) -> tuple[bool, lis def run_routing(skill: str, budget: int = 60, log=print) -> dict: usage_ledger.reset() - skill_dir = SKILLS_DIR / skill - if not (skill_dir / "SKILL.md").exists(): - raise SystemExit(f"No skill named '{skill}' in skills/.") + skill_dir = resolve_skill_dir(skill) tasks_path = TASKS_DIR / f"{skill}.yaml" cases = (yaml.safe_load(tasks_path.read_text()) or {}).get("routing") if tasks_path.exists() else None champion = optimizable_components(skill_dir) @@ -207,7 +206,7 @@ def run_routing(skill: str, budget: int = 60, log=print) -> dict: skill=skill, created=created, dataset=dataset, metrics=metrics, champion_revision=current.revision, challenger_revision=challenger_revision, inner_loop=inner_loop, gate=gate)) - evidence_json, evidence_markdown = write_evidence(evidence, EVIDENCE_DIR / skill / str(created)) + evidence_json, evidence_markdown = write_evidence(evidence, paths.runs() / "evidence" / skill / str(created)) log(f"[ci] routing evidence: {evidence_json} and {evidence_markdown}") pending = { diff --git a/optimize/routing_health.py b/ingot/optimize/routing_health.py similarity index 89% rename from optimize/routing_health.py rename to ingot/optimize/routing_health.py index 4ec21a3..956ed3c 100644 --- a/optimize/routing_health.py +++ b/ingot/optimize/routing_health.py @@ -6,18 +6,18 @@ plus a pairwise description-collision scan at the same COLLISION_SCORE cutoff the promotion gate uses. Read-only; it proposes nothing and never touches the review queue. -Usage: python -m optimize.routing_health [skill ...] (default: every skill with routing cases) +Usage: python -m ingot.optimize.routing_health [skill ...] (default: every skill with routing cases) Exit status is non-zero when any suite case fails or any collision is found, so it can run -unattended from cron or CI: docker compose run --rm optimize python -m optimize.routing_health +unattended from cron or CI: docker compose run --rm optimize python -m ingot.optimize.routing_health """ import argparse from pathlib import Path import yaml -from mcp_server.registry import load_skills -from mcp_server.router import Router -from mcp_server.routing_eval import evaluate_cases, evaluate_parity +from ingot.mcp_server.registry import load_skills +from ingot.mcp_server.router import Router +from ingot.mcp_server.routing_eval import evaluate_cases, evaluate_parity from .ab import COLLISION_SCORE, TASKS_DIR @@ -82,14 +82,14 @@ def run_health(skills: list[str] | None = None, log=print) -> list[str]: for p in problems: log(f"[health] - {p}") log("[health] fix: refine the colliding/regressed description by hand, or run the " - "routing pass (python -m optimize.ab --description) and review the result.") + "routing pass (python -m ingot.optimize.ab --description) and review the result.") else: log("\n[health] ✓ routing healthy: every suite passes and no descriptions collide.") return problems if __name__ == "__main__": - ap = argparse.ArgumentParser(prog="python -m optimize.routing_health") + ap = argparse.ArgumentParser(prog="python -m ingot.optimize.routing_health") ap.add_argument("skills", nargs="*", help="skills to check (default: every skill with an eval task set)") raise SystemExit(1 if run_health(ap.parse_args().skills or None) else 0) diff --git a/optimize/sandbox_driver.py b/ingot/optimize/sandbox_driver.py similarity index 100% rename from optimize/sandbox_driver.py rename to ingot/optimize/sandbox_driver.py diff --git a/optimize/skillopt_bridge.py b/ingot/optimize/skillopt_bridge.py similarity index 100% rename from optimize/skillopt_bridge.py rename to ingot/optimize/skillopt_bridge.py diff --git a/optimize/skillopt_loop.py b/ingot/optimize/skillopt_loop.py similarity index 98% rename from optimize/skillopt_loop.py rename to ingot/optimize/skillopt_loop.py index c5e8f96..fde2b98 100644 --- a/optimize/skillopt_loop.py +++ b/ingot/optimize/skillopt_loop.py @@ -2,7 +2,7 @@ adapted to skill_router's rollout + judge. Treats the skill `body` as trainable state and improves it with SkillOpt's four disciplines, -each driven through `optimize.skillopt_bridge` (the single dependency seam): +each driven through `ingot.optimize.skillopt_bridge` (the single dependency seam): 1. a step buffer of prior failures + *rejected* edits, fed back into each reflection so the optimizer stops re-proposing what the gate already threw out; @@ -14,7 +14,7 @@ Drop-in for the inner loop: `run_skillopt(seed, tasks, frozen, ...) -> (best_components, seed_score, best_score)`, scores being the penalized mean judge over the train set (comparable to what the outer -A/B logs). The leakage-clean held-out promotion gate in optimize.ab is unchanged and remains the +A/B logs). The leakage-clean held-out promotion gate in ingot.optimize.ab is unchanged and remains the promotion authority; this only replaces how the challenger body is produced. """ import os diff --git a/optimize/skillopt_prompts/README.md b/ingot/optimize/skillopt_prompts/README.md similarity index 100% rename from optimize/skillopt_prompts/README.md rename to ingot/optimize/skillopt_prompts/README.md diff --git a/optimize/skillopt_prompts/analyst_error.md b/ingot/optimize/skillopt_prompts/analyst_error.md similarity index 100% rename from optimize/skillopt_prompts/analyst_error.md rename to ingot/optimize/skillopt_prompts/analyst_error.md diff --git a/optimize/skillopt_prompts/lr_autonomous.md b/ingot/optimize/skillopt_prompts/lr_autonomous.md similarity index 100% rename from optimize/skillopt_prompts/lr_autonomous.md rename to ingot/optimize/skillopt_prompts/lr_autonomous.md diff --git a/optimize/skillopt_prompts/ranking.md b/ingot/optimize/skillopt_prompts/ranking.md similarity index 100% rename from optimize/skillopt_prompts/ranking.md rename to ingot/optimize/skillopt_prompts/ranking.md diff --git a/optimize/skillopt_prompts/slow_update.md b/ingot/optimize/skillopt_prompts/slow_update.md similarity index 100% rename from optimize/skillopt_prompts/slow_update.md rename to ingot/optimize/skillopt_prompts/slow_update.md diff --git a/optimize/tasks/.gitkeep b/ingot/optimize/tasks/.gitkeep similarity index 100% rename from optimize/tasks/.gitkeep rename to ingot/optimize/tasks/.gitkeep diff --git a/ingot/optimize/tree.py b/ingot/optimize/tree.py new file mode 100644 index 0000000..67b76b0 --- /dev/null +++ b/ingot/optimize/tree.py @@ -0,0 +1,272 @@ +"""The exact bytes of an admitted package, staged for publication. + +Admission used to reduce a package to a dictionary of decoded text components. Anything that +dictionary could not hold -- an image, a PDF, a wheel, a file whose bytes are not valid UTF-8 -- +was dropped on the way in, and the revision the reviewer approved then named a package that was +missing them. A content-addressed release controller has exactly one promise to keep: a revision +names the exact package. Dropping a file breaks it silently, which is the worst way to break it. + +So a candidate is a *tree*, not a dictionary. Every regular file is copied byte-for-byte into a +staging directory named by the tree's digest, and the manifest records each file's relative path, +mode, size, and SHA-256 of its raw bytes. Publication copies that staged tree into the vault +worktree and verifies every hash on the way. The reviewer, the receipt, and the vault are then +looking at the same bytes. + +Two deliberate exceptions, both visible rather than silent: + +- **SKILL.md is normalized, not preserved.** Its frontmatter is the routing interface: the name is + forced to the skill's identity, the description is collapsed to one line, and the whole thing is + re-emitted through a safe YAML dump so a stray `---` in a model's output cannot corrupt it. The + manifest still records the source file's real hash and size, so the normalization is auditable, + but what gets served is the normalized file. `revision()` below hashes the result of doing + exactly that, so the approved revision is the served revision. +- **Symlinks are refused.** Preserving one means the vault commits a link that a reader follows out + of the library; flattening one into its target silently changes the artifact's shape. Neither is + a decision admission should make on an operator's behalf, so a package containing a symlink is + refused by name until there is a reason to build one of them. + +Modes are clamped to 0o644 or 0o755 because those are the only two a Git checkout reproduces, and +the vault is a Git repository. Recording the raw mode would describe bytes the vault cannot serve. +""" +from __future__ import annotations + +import hashlib +import json +import os +import shutil +import tempfile +import unicodedata +import uuid +from pathlib import Path, PurePosixPath + +from ingot import paths + +TREE_SCHEMA = "ingot/candidate-tree/v1" + +# Bounds on what admission will stage. The byte limit does the real work; the file count stops a +# package of a million empty files from turning one approval into a filesystem problem. Both sit +# above the advisory thresholds `ingot review` warns at, so a package it merely warns about is +# still admissible. +MAX_FILES = 256 +MAX_TREE_BYTES = 20_000_000 + +_CHUNK = 1 << 20 +_WINDOWS_RESERVED = {"con", "prn", "aux", "nul", *(f"com{i}" for i in range(1, 10)), + *(f"lpt{i}" for i in range(1, 10))} + + +def candidates_dir() -> Path: + """Staged candidate trees. Resolved per call, never bound at import.""" + return paths.runs() / "candidates" + + +def staged_dir(digest: str) -> Path: + if not isinstance(digest, str) or len(digest) != 64 or not all( + char in "0123456789abcdef" for char in digest): + raise ValueError(f"invalid candidate tree digest: {digest!r}") + return candidates_dir() / digest + + +def portable_path(raw: str, *, allow_skill_md: bool = False) -> PurePosixPath: + """One relative path every filesystem in the deployment agrees on. + + Shared with the text-component path in `ingot.optimize.ingress` so a package cannot enter through one + door what the other would refuse.""" + if not isinstance(raw, str) or "\\" in raw or any(ord(char) < 32 for char in raw): + raise ValueError(f"component is not a portable POSIX path: {raw}") + path = PurePosixPath(raw) + parts = path.parts + forbidden = {"."} if allow_skill_md else {".", "SKILL.md"} + if path.is_absolute() or ".." in parts or path.as_posix() in forbidden: + raise ValueError(f"component escapes skill root: {raw}") + if any(":" in part or part.endswith((".", " ")) or + part.split(".", 1)[0].casefold() in _WINDOWS_RESERVED for part in parts): + raise ValueError(f"component is not a portable POSIX path: {raw}") + if any(unicodedata.normalize("NFC", part) != part for part in parts): + raise ValueError(f"component path must be NFC-normalized: {raw}") + return path + + +def _hash(path: Path) -> tuple[str, int]: + """SHA-256 of the raw bytes, and the byte count. Never a decoded string: a file that is not + valid UTF-8 has no decoded form, and one that is would hash differently after a round trip.""" + digest, size = hashlib.sha256(), 0 + with path.open("rb") as handle: + while chunk := handle.read(_CHUNK): + digest.update(chunk) + size += len(chunk) + return digest.hexdigest(), size + + +def _mode(raw: int) -> int: + return 0o755 if raw & 0o111 else 0o644 + + +def _digest(files: list[dict]) -> str: + canonical = json.dumps({"schema_version": TREE_SCHEMA, "files": files}, + sort_keys=True, separators=(",", ":"), ensure_ascii=False) + return hashlib.sha256(canonical.encode()).hexdigest() + + +def _walk(root: Path, current: Path) -> list[Path]: + """Every regular file under `current`, refusing anything that is not one. + + An explicit descent rather than `rglob`, which follows directory symlinks: a package could + otherwise pull in a whole tree from outside itself and this would never see the link.""" + found = [] + for item in sorted(current.iterdir()): + relative = item.relative_to(root).as_posix() + if item.is_symlink(): + raise ValueError(f"symlinks are not admissible: {relative}") + if item.is_dir(): + found.extend(_walk(root, item)) + elif item.is_file(): + found.append(item) + else: + raise ValueError(f"not a regular file: {relative}") + return found + + +def build(package: Path) -> dict: + """Describe every regular file in `package` exactly as it will be served.""" + package = Path(package).resolve() + files, total, folded = [], 0, set() + for path in _walk(package, package): + raw = path.relative_to(package).as_posix() + relative = portable_path(raw, allow_skill_md=True).as_posix() + if relative.casefold() in folded: + raise ValueError(f"component path collides case-insensitively: {raw}") + folded.add(relative.casefold()) + checksum, size = _hash(path) + total += size + files.append({"path": relative, "mode": _mode(path.stat().st_mode), + "size": size, "sha256": checksum}) + + if not files: + raise ValueError("the package holds no files") + if len(files) > MAX_FILES: + raise ValueError(f"a package may hold at most {MAX_FILES} files; this one holds " + f"{len(files)}") + if total > MAX_TREE_BYTES: + raise ValueError(f"a package may hold at most {MAX_TREE_BYTES} bytes; this one holds " + f"{total}") + files.sort(key=lambda entry: entry["path"]) + return {"schema_version": TREE_SCHEMA, "files": files, "size": total, + "digest": _digest(files)} + + +def verify_manifest(manifest: object) -> dict: + """The manifest is well formed and its digest covers the file list it arrived with. + + Recomputed rather than trusted, because the digest is what binds a receipt to a staged tree: a + receipt whose file list was edited without its digest would otherwise publish a tree nobody + approved.""" + if not isinstance(manifest, dict) or manifest.get("schema_version") != TREE_SCHEMA: + raise ValueError("unsupported candidate tree manifest") + files = manifest.get("files") + if not isinstance(files, list) or not files: + raise ValueError("a candidate tree manifest lists at least one file") + for entry in files: + if not isinstance(entry, dict): + raise ValueError("candidate tree entries must be objects") + if set(entry) != {"path", "mode", "size", "sha256"}: + raise ValueError(f"unexpected candidate tree entry: {sorted(entry)}") + portable_path(entry["path"], allow_skill_md=True) + if entry["mode"] not in (0o644, 0o755): + raise ValueError(f"unsupported mode for {entry['path']}: {entry['mode']}") + if not isinstance(entry["size"], int) or isinstance(entry["size"], bool) \ + or entry["size"] < 0: + raise ValueError(f"invalid size for {entry['path']}") + if _digest(files) != manifest.get("digest"): + raise ValueError("candidate tree digest does not cover its file list") + return manifest + + +def stage(package: Path, manifest: dict) -> Path: + """Copy the described bytes somewhere the publisher can reach them. + + The source directory belongs to whoever ran `ingot add`; it can be edited or deleted between + the approval and the publication, and a publisher that read from it would publish whatever it + found. Staging is named by the tree digest, so ingesting identical bytes twice converges on one + directory instead of racing.""" + package = Path(package).resolve() + destination = staged_dir(manifest["digest"]) + if destination.is_dir(): + return destination + + candidates_dir().mkdir(parents=True, exist_ok=True, mode=0o700) + temporary = candidates_dir() / f".{manifest['digest']}.{uuid.uuid4().hex}.tmp" + try: + for entry in manifest["files"]: + target = temporary / entry["path"] + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(package / entry["path"], target) + os.chmod(target, entry["mode"]) + checksum, size = _hash(target) + if (checksum, size) != (entry["sha256"], entry["size"]): + raise ValueError(f"{entry['path']} changed while it was being staged") + try: + os.rename(temporary, destination) + except OSError: + # Another submitter staged the identical tree first. Same digest, same bytes. + if not destination.is_dir(): + raise + shutil.rmtree(temporary, ignore_errors=True) + except BaseException: + shutil.rmtree(temporary, ignore_errors=True) + raise + return destination + + +def materialize(manifest: dict, destination: Path) -> None: + """Write the staged tree into `destination`, checking every file against the manifest.""" + verify_manifest(manifest) + source = staged_dir(manifest["digest"]) + if not source.is_dir(): + raise ValueError(f"the staged candidate tree {manifest['digest'][:16]} is missing") + destination = Path(destination) + destination.mkdir(parents=True, exist_ok=True) + for entry in manifest["files"]: + relative = portable_path(entry["path"], allow_skill_md=True) + staged = source / relative + checksum, size = _hash(staged) + if (checksum, size) != (entry["sha256"], entry["size"]): + raise ValueError(f"staged candidate file does not match the receipt: {entry['path']}") + target = destination / relative + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(staged, target) + os.chmod(target, entry["mode"]) + + +def materialize_creation(manifest: dict, components: dict, skill: str, destination: Path) -> None: + """The whole tree, then the normalized SKILL.md over the top. + + The one implementation of what an admitted package becomes. `revision()` runs it into a + temporary directory to compute the revision a reviewer approves, and the publisher runs it into + the vault worktree to produce the revision the library serves. They cannot disagree, because + there is nothing here for them to disagree about.""" + from ingot.mcp_server.registry import write_skill_md + + materialize(manifest, destination) + try: + metadata = json.loads(components.get("frontmatter") or "{}") + except (TypeError, ValueError) as exc: + raise ValueError("creation frontmatter is not valid JSON") from exc + if not isinstance(metadata, dict): + raise ValueError("creation frontmatter is not an object") + metadata["name"] = skill + metadata["description"] = components["description"] + write_skill_md(Path(destination) / "SKILL.md", metadata, components["body"]) + + +def revision(skill: str, manifest: dict, components: dict) -> str: + """The revision the library will serve once this tree is published. + + Computed by materializing it, because a revision derived some other way would be a second + description of the same bytes and the two would eventually disagree.""" + from ingot.mcp_server.registry import skill_revision + + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) / skill + materialize_creation(manifest, components, skill, root) + return skill_revision(root) diff --git a/ingot/optimize/usage.py b/ingot/optimize/usage.py new file mode 100644 index 0000000..1205edf --- /dev/null +++ b/ingot/optimize/usage.py @@ -0,0 +1,162 @@ +"""Token ledger for an optimize run: every LLM call is attributed to a role +(rollout / judge / reflection / agent_ab) so the run can report what it actually cost , +including a best-effort USD estimate from OpenRouter list prices, and an optional hard +spend cap (MAX_RUN_USD) that aborts a run before it exceeds the budget.""" +import os +import threading +from collections import defaultdict + +COUNTS: dict[str, dict[str, int]] = defaultdict(lambda: {"input": 0, "output": 0, "calls": 0}) +_SUBSCRIPTION_COUNTS: dict[str, dict[str, int]] = defaultdict( + lambda: {"input": 0, "output": 0, "calls": 0} +) +_LOCK = threading.RLock() # the search fans rollout+judge across a thread pool; add() re-enters for the cap +_PRICES: dict[str, tuple[float, float]] | None = None + + +def reset(): + """Start a fresh ledger, the UI process runs many optimizations; counts must not leak across runs.""" + with _LOCK: + COUNTS.clear() + _SUBSCRIPTION_COUNTS.clear() + + +def add(role: str, usage: dict | None, *, billing_mode: str = "metered"): + """usage: langchain usage_metadata ({'input_tokens','output_tokens'}) or equivalent dict.""" + if not usage: + return + if billing_mode not in {"metered", "subscription"}: + raise ValueError(f"unknown billing mode: {billing_mode}") + with _LOCK: + c = COUNTS[role] + c["input"] += int(usage.get("input_tokens", 0)) + c["output"] += int(usage.get("output_tokens", 0)) + c["calls"] += 1 + if billing_mode == "subscription": + subscription = _SUBSCRIPTION_COUNTS[role] + subscription["input"] += int(usage.get("input_tokens", 0)) + subscription["output"] += int(usage.get("output_tokens", 0)) + subscription["calls"] += 1 + if billing_mode == "metered": + _enforce_cap() + + +def _enforce_cap(): + cap = float(os.environ.get("MAX_RUN_USD", "0") or 0) + if not cap: + return + cost = estimated_cost() + if cost is not None and cost > cap: + raise SystemExit(f"MAX_RUN_USD exceeded: estimated ${cost:.2f} > cap ${cap:.2f}, " + f"aborting before spending more.\n{format_report()}") + + +def _openrouter_prices() -> dict[str, tuple[float, float]]: + """model id -> (prompt, completion) USD per token from OpenRouter's public models API; + {} on any failure (cost reporting is best-effort, never a gate on offline work).""" + import json + import urllib.request + try: + with urllib.request.urlopen("https://openrouter.ai/api/v1/models", timeout=10) as r: + data = json.loads(r.read())["data"] + return {m["id"]: (float(m["pricing"]["prompt"]), float(m["pricing"]["completion"])) + for m in data if m.get("pricing")} + except Exception: + return {} + + +def _role_models() -> dict[str, str]: + """Which model each ledger role runs on (first judge only, for ensemble setups).""" + from . import agent_model, skillopt_model + teacher = skillopt_model() + judge = (os.environ.get("JUDGE_MODELS") or + os.environ.get("JUDGE_MODEL", "google/gemini-2.5-flash")).split(",")[0].strip() + return {"rollout": agent_model(), "agent_ab": agent_model(), "review": agent_model(), + "judge": judge, "reflection": teacher, "draft": teacher} + + +def _model_for(role: str) -> str: + """The model a ledger role ran on. A role may name its own model after a colon + (`compat:anthropic/claude-sonnet-4.5`): the compatibility sweep varies the *serving* model by + design, so unlike the fixed roles it cannot be mapped to one model up front.""" + name, _, explicit = role.partition(":") + return explicit or _role_models().get(name, "") + + +def _metered_counts() -> dict[str, dict[str, int]]: + """Per-role usage remaining after subscription-backed calls are removed.""" + with _LOCK: + return { + role: {key: count - _SUBSCRIPTION_COUNTS[role][key] for key, count in counts.items()} + for role, counts in COUNTS.items() + } + + +def unpriced_roles() -> list[str]: + """Roles that spent tokens but contribute nothing to the estimate, because their model carries + no OpenRouter list price — a local endpoint (genuinely free) or a slug we could not resolve + (not free at all). Surfaced rather than swallowed: an unpriced role silently counts as $0, and + that is how the entire compatibility sweep once vanished from both the cost line and the + MAX_RUN_USD cap, reporting $0.04 against $1.42 actually spent.""" + if _PRICES is None: + return [] + return sorted(role for role, c in _metered_counts().items() + if (c["input"] or c["output"]) and not _PRICES.get(_model_for(role))) + + +def estimated_cost() -> float | None: + """Best-effort USD estimate for the current ledger, from OpenRouter list prices. None when + the endpoint isn't OpenRouter or pricing is unavailable (local endpoints cost nothing). + Roles whose model has no list price contribute nothing — ask unpriced_roles() which those are + before trusting this as a total.""" + global _PRICES + from . import is_openrouter, teacher_base_url + if not is_openrouter(teacher_base_url()): + return None + if _PRICES is None: + _PRICES = _openrouter_prices() + if not _PRICES: + return None + metered = _metered_counts() + if not any(c["input"] or c["output"] for c in metered.values()): + return None + return sum(c["input"] * p[0] + c["output"] * p[1] + for role, c in metered.items() + if (p := _PRICES.get(_model_for(role)))) + + +def report() -> dict: + out = {role: dict(c) for role, c in COUNTS.items()} + out["total"] = { + "input": sum(c["input"] for c in COUNTS.values()), + "output": sum(c["output"] for c in COUNTS.values()), + "calls": sum(c["calls"] for c in COUNTS.values()), + } + return out + + +def format_report() -> str: + r = report() + with _LOCK: + subscriptions = {role: dict(c) for role, c in _SUBSCRIPTION_COUNTS.items()} + lines = [] + for role, c in r.items(): + if role == "total": + continue + subscription = subscriptions.get(role) + suffix = (f" ({subscription['calls']} subscription calls, " + f"{subscription['input']:,} in, {subscription['output']:,} out)" + if subscription and subscription["calls"] else "") + lines.append(f" {role:<12} {c['calls']:>4} calls {c['input']:>9,} in " + f"{c['output']:>8,} out{suffix}") + t = r["total"] + lines.append(f" {'TOTAL':<12} {t['calls']:>4} calls {t['input']:>9,} in {t['output']:>8,} out") + cost = estimated_cost() + if cost is not None: + lines.append(f" estimated cost: ${cost:.2f} (OpenRouter list prices)") + if (blind := unpriced_roles()): + lines.append(f" NOT in that estimate: {', '.join(blind)} — no list price for " + f"{', '.join(sorted({_model_for(r) or '?' for r in blind}))}. " + f"Free if that is a local endpoint; otherwise the figure above is low, " + f"and so is any MAX_RUN_USD cap resting on it.") + return "\n".join(lines) diff --git a/ingot/parse.py b/ingot/parse.py new file mode 100644 index 0000000..2ff58c9 --- /dev/null +++ b/ingot/parse.py @@ -0,0 +1,67 @@ +"""A SKILL.md parser that reports what is wrong instead of repairing it. + +`ingot.mcp_server.registry.parse_skill` is deliberately tolerant: absent frontmatter, unparseable YAML, +and a frontmatter that is not a mapping all normalize to empty metadata so the server keeps +serving. That is correct for serving and useless for diagnosis -- silent normalization is not a +diagnostic API. This parser answers the same question in the opposite direction: it keeps the +malformed input and returns findings. + +The two must agree on the shape of a well-formed document, so the frontmatter pattern is imported +from the serving parser rather than restated here.""" +from __future__ import annotations + +from dataclasses import dataclass, field + +import yaml + +from ingot.mcp_server.registry import _FRONTMATTER + +ERROR = "error" +WARNING = "warning" +INFO = "info" + + +@dataclass(frozen=True) +class Finding: + code: str + level: str + message: str + path: str | None = None + + def as_dict(self) -> dict: + found = {"code": self.code, "level": self.level, "message": self.message} + if self.path is not None: + found["path"] = self.path + return found + + +@dataclass +class RawSkill: + """`frontmatter` is None when the document has none that can be read as a mapping. That is the + distinction the serving parser erases, and every caller here depends on it.""" + frontmatter: dict | None + body: str + findings: list[Finding] = field(default_factory=list) + + +def parse_raw(text: str) -> RawSkill: + match = _FRONTMATTER.match(text) + if not match: + return RawSkill(None, text.strip(), [Finding( + "frontmatter-missing", ERROR, + "no YAML frontmatter: a SKILL.md must open with a '---' delimited block")]) + + try: + loaded = yaml.safe_load(match.group(1)) + except yaml.YAMLError as error: + detail = str(error).replace("\n", " ").strip() + return RawSkill(None, match.group(2).strip(), [Finding( + "frontmatter-invalid", ERROR, f"frontmatter is not valid YAML: {detail}")]) + + if not isinstance(loaded, dict): + kind = type(loaded).__name__ if loaded is not None else "nothing" + return RawSkill(None, match.group(2).strip(), [Finding( + "frontmatter-not-a-mapping", ERROR, + f"frontmatter must be a mapping of fields, found {kind}")]) + + return RawSkill(loaded, match.group(2).strip(), []) diff --git a/ingot/paths.py b/ingot/paths.py new file mode 100644 index 0000000..473a1b7 --- /dev/null +++ b/ingot/paths.py @@ -0,0 +1,128 @@ +"""Where mutable state lives. The one place that answers, so no module derives it from `__file__`. + +State that was package-relative worked for exactly two deployments — a container and a checkout — +and broke everywhere else: a read-only or system Python, an upgrade that replaces the package +directory, a recreated environment, two deployments sharing one installation, and any backup that +expects application code and controlled state to be separable. A release controller that keeps its +own review queue inside `site-packages` cannot claim to control anything. + +Every function reads the environment when called. Modules that still expose a module-level constant +derive it from here, so the default is right for an installation; the resolution rule lives here. +""" +from __future__ import annotations + +import os +from pathlib import Path + +HOME = "INGOT_HOME" +LIBRARY = "INGOT_LIBRARY" +RUNS = "INGOT_RUNS" +TASKS = "INGOT_TASKS" +VAULT = "INGOT_VAULT_PATH" +# Names that predate INGOT_HOME. Honoured so an existing deployment keeps working, and reported by +# `ingot status` so it is visible rather than load-bearing and forgotten. +LEGACY = {LIBRARY: "SKILLS_DIR", VAULT: "VAULT_DIR"} + +PACKAGE_ROOT = Path(__file__).resolve().parent.parent + + +def _env(name: str, env: dict | None = None) -> str: + env = os.environ if env is None else env + value = env.get(name) or env.get(LEGACY.get(name, ""), "") + return value.strip() + + +def home(env: dict | None = None) -> Path: + """The root of everything mutable. + + XDG on Unix by default, so a plain `pip install ingot` puts state somewhere a package upgrade + cannot take with it. Containers and managed deployments override it, or override each path + below individually.""" + env = os.environ if env is None else env + explicit = _env(HOME, env) + if explicit: + return Path(explicit).expanduser() + state = (env.get("XDG_STATE_HOME") or "").strip() + if state: + return Path(state).expanduser() / "ingot" + return Path.home() / ".local" / "state" / "ingot" + + +def _under(name: str, default: str, env: dict | None = None) -> Path: + explicit = _env(name, env) + return Path(explicit).expanduser() if explicit else home(env) / default + + +def library(env: dict | None = None) -> Path: + """The served skill library.""" + return _under(LIBRARY, "library", env) + + +def runs(env: dict | None = None) -> Path: + """Review queue, publication receipts, evidence, snapshots, and audit trails.""" + return _under(RUNS, "runs", env) + + +def tasks(env: dict | None = None) -> Path: + """Held-out evaluation task sets.""" + return _under(TASKS, "tasks", env) + + +def vault(env: dict | None = None) -> Path: + """The Git repository the publisher owns. + + Defaults to the library, because in the local backend the served checkout *is* the vault. A + deployment that wants them separate says so.""" + explicit = _env(VAULT, env) + return Path(explicit).expanduser() if explicit else library(env) + + +def _source(name: str, env: dict | None = None) -> str: + """Which setting decided a path, for a report that has to be checkable rather than believed.""" + env = os.environ if env is None else env + if env.get(name): + return name + legacy = LEGACY.get(name) + if legacy and env.get(legacy): + return f"{legacy} (deprecated; prefer {name})" + if env.get(HOME): + return HOME + if name == VAULT: + return _source(LIBRARY, env) + return "XDG_STATE_HOME" if env.get("XDG_STATE_HOME") else "default" + + +def resolved(env: dict | None = None) -> list[dict]: + """Every path this process would use, where it came from, and whether it can be written.""" + entries = [("home", home(env), HOME), ("library", library(env), LIBRARY), + ("runs", runs(env), RUNS), ("tasks", tasks(env), TASKS), + ("vault", vault(env), VAULT)] + return [{"name": name, "path": str(path), "source": _source(setting, env), + "exists": path.exists(), + # An absent path is not a problem: it is created on first write. An absent path whose + # parent cannot be written is, and reporting only `exists` would hide it. + "writable": os.access(path if path.exists() else _nearest(path), os.W_OK | os.X_OK)} + for name, path, setting in entries] + + +def _nearest(path: Path) -> Path: + for candidate in path.parents: + if candidate.exists(): + return candidate + return path + + +def legacy_state() -> list[str]: + """Package-relative state directories left by a version that kept them next to the code. + + Reported, never migrated. Moving a review queue or a receipt store on someone's behalf is a + change to controlled state made by a process that was not asked to make it.""" + found = [] + for name in ("runs", "skills"): + directory = PACKAGE_ROOT / name + if not directory.is_dir(): + continue + contents = [item for item in directory.iterdir() if item.name != ".gitkeep"] + if contents: + found.append(str(directory)) + return found diff --git a/ingot/records.py b/ingot/records.py new file mode 100644 index 0000000..e3fbabd --- /dev/null +++ b/ingot/records.py @@ -0,0 +1,224 @@ +"""The two versioned records the control plane is built on. + +A **candidate manifest** names exactly what is being proposed, where it came from, and what the +deterministic review said about it. A **release receipt** names exactly what was published, against +which champion, on whose authority. + +Three properties matter more than the field list: + +- **A revision names exact bytes.** Every revision here is a content digest or the absence marker. + A semantic version is not accepted anywhere: a tag can be moved and a digest cannot. +- **Candidate identity is deterministic.** Two submissions of the same bytes, from different + checkouts at different times, are the same candidate. That is what makes an identical resubmission + idempotent instead of a duplicate proposal. +- **A receipt is not a proof.** There is no signature field and there will not be one until there is + a threat model. A local record that a machine administrator can rewrite must not carry anything + shaped like evidence that they did not. + +Consumers: `ingot add` writes candidate manifests; the publisher writes release receipts once +publication succeeds -- never when approval merely queues it. +""" +from __future__ import annotations + +import hashlib +import json +import re + +from ingot.mcp_server.registry import SLUG_RE + +CANDIDATE_SCHEMA = "ingot/candidate/v1" +RELEASE_SCHEMA = "ingot/release/v1" + +# Absence is a revision. A creation displaces nothing; a rollback can restore nothing. The existing +# publisher already treats it this way (optimize/promote.py ABSENT_REVISION), and the two spellings +# must not drift apart. +ABSENT_REVISION = "absent" + +SOURCE_TYPES = frozenset({"file", "github", "optimizer", "mcp"}) +CANDIDATE_KINDS = frozenset({"creation", "update"}) +ACTIONS = frozenset({"promote", "rollback"}) +RESULTS = frozenset({"published", "failed"}) + +_DIGEST = re.compile(r"^[0-9a-f]{64}$") + +# Kept for file candidates created before GitHub acquisition. Their proposal IDs already include +# the complete bound review record except circumstance metadata, and changing that basis would +# turn an identical resubmission into a slot conflict during upgrade. +_NOT_IDENTITY = ("created_at", "actor") + +def digest(payload: object) -> str: + """SHA-256 over a canonical JSON encoding. Key order must not change the answer.""" + canonical = json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=False) + return hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _is_revision(value: object, *, allow_absent: bool = True) -> bool: + if allow_absent and value == ABSENT_REVISION: + return True + return isinstance(value, str) and bool(_DIGEST.fullmatch(value)) + + +def candidate_manifest(*, kind: str, skill: str, source_type: str, locator: str, + resolved_revision: str, candidate_revision: str, review: dict, + created_at: int, provenance: dict | None = None) -> dict: + """Build a candidate manifest. Validation is separate -- callers validate what they build.""" + return { + "schema_version": CANDIDATE_SCHEMA, + "kind": kind, + "skill": skill, + "source": {**(provenance or {}), + "type": source_type, + "locator": locator, + "resolved_revision": resolved_revision}, + "candidate_revision": candidate_revision, + # `digest` binds this summary; `report_digest`, when the caller supplies one, binds the full + # report the summary was drawn from. Both travel, so neither the summary nor the report it + # came from can be edited afterwards without the manifest noticing. + "review": {"schema_version": review.get("schema_version"), + "digest": digest(review), + "valid": review.get("valid"), + "errors": list(review.get("errors") or []), + "warnings": list(review.get("warnings") or []), + **({"report_digest": review["report_digest"]} + if review.get("report_digest") else {})}, + "created_at": created_at, + } + + +def candidate_identity(manifest: dict) -> str: + """The digest that decides whether two submissions are the same candidate. + + Drops timestamps, actor metadata, and the local path the package happened to be read from. + Keeps the skill, the kind, the source type, both revisions, and the review outcome -- a review + is evidence, and evidence is revision-bound, so the same bytes reviewed clean and reviewed with + errors are not interchangeable proposals.""" + source = manifest.get("source") or {} + review = manifest.get("review") or {} + identity_review = ({key: review.get(key) for key in + ("schema_version", "valid", "errors", "warnings")} + if source.get("type") == "github" else + {key: value for key, value in review.items() if key not in _NOT_IDENTITY}) + return digest({ + "schema_version": manifest.get("schema_version"), + "kind": manifest.get("kind"), + "skill": manifest.get("skill"), + "source_type": source.get("type"), + "resolved_revision": source.get("resolved_revision"), + "candidate_revision": manifest.get("candidate_revision"), + # A GitHub report includes its temporary clone path. That path is provenance rather than + # outcome, so GitHub identity keeps the deterministic result while its manifest keeps both + # digests. File candidates retain the original formula above for upgrade compatibility. + "review": identity_review, + }) + + +def validate_candidate(manifest: dict) -> list[str]: + """Every problem with a manifest, or an empty list. Never raises: a caller reporting to a person + wants all of them at once, not the first.""" + problems = [] + if manifest.get("schema_version") != CANDIDATE_SCHEMA: + problems.append(f"schema_version must be {CANDIDATE_SCHEMA}, " + f"found {manifest.get('schema_version')!r}") + if manifest.get("kind") not in CANDIDATE_KINDS: + problems.append(f"kind must be one of {sorted(CANDIDATE_KINDS)}, " + f"found {manifest.get('kind')!r}") + + skill = manifest.get("skill") + if not isinstance(skill, str) or not SLUG_RE.fullmatch(skill): + problems.append(f"skill must be a valid slug, found {skill!r}") + + source = manifest.get("source") + if not isinstance(source, dict): + problems.append("source is missing") + else: + if source.get("type") not in SOURCE_TYPES: + problems.append(f"source.type must be one of {sorted(SOURCE_TYPES)}, " + f"found {source.get('type')!r}") + if not isinstance(source.get("locator"), str) or not source.get("locator"): + problems.append("source.locator is missing") + if not _is_revision(source.get("resolved_revision"), allow_absent=False): + problems.append("source.resolved_revision must be a content digest, not a version: " + f"found {source.get('resolved_revision')!r}") + + if not _is_revision(manifest.get("candidate_revision"), allow_absent=False): + problems.append("candidate_revision must be a content digest, found " + f"{manifest.get('candidate_revision')!r}") + + review = manifest.get("review") + if not isinstance(review, dict): + problems.append("review is missing") + elif not _is_revision(review.get("digest"), allow_absent=False): + problems.append("review.digest must be a content digest") + + if not isinstance(manifest.get("created_at"), int): + problems.append("created_at must be an integer timestamp") + return problems + + +def release_receipt(*, skill: str, action: str, proposal_id: str, publication_id: str, + expected_champion: str, candidate_revision: str, evidence_digests: list[str], + actor: str, publisher: str, target: str, published_at: int, + result: str, error: str | None = None) -> dict: + """Build a release receipt. + + Emitted only once publication has actually succeeded or failed. Approval that merely queues a + publication is a different state and does not produce one of these -- a receipt that appeared at + approval time would claim a skill was released while the served bytes were unchanged.""" + receipt = { + "schema_version": RELEASE_SCHEMA, + "skill": skill, + "action": action, + "proposal_id": proposal_id, + "publication_id": publication_id, + "expected_champion": expected_champion, + "candidate_revision": candidate_revision, + "evidence_digests": list(evidence_digests), + "actor": actor, + "publisher": publisher, + "target": target, + "published_at": published_at, + "result": result, + } + if error is not None: + receipt["error"] = error + return receipt + + +def validate_release(receipt: dict) -> list[str]: + problems = [] + if receipt.get("schema_version") != RELEASE_SCHEMA: + problems.append(f"schema_version must be {RELEASE_SCHEMA}, " + f"found {receipt.get('schema_version')!r}") + + skill = receipt.get("skill") + if not isinstance(skill, str) or not SLUG_RE.fullmatch(skill): + problems.append(f"skill must be a valid slug, found {skill!r}") + if receipt.get("action") not in ACTIONS: + problems.append(f"action must be one of {sorted(ACTIONS)}, found {receipt.get('action')!r}") + if receipt.get("result") not in RESULTS: + problems.append(f"result must be one of {sorted(RESULTS)}, found {receipt.get('result')!r}") + + for field in ("expected_champion", "candidate_revision"): + if not _is_revision(receipt.get(field)): + problems.append(f"{field} must be a content digest or {ABSENT_REVISION!r}, " + f"found {receipt.get(field)!r}") + + for field in ("proposal_id", "publication_id", "actor", "publisher", "target"): + if not isinstance(receipt.get(field), str) or not receipt.get(field): + problems.append(f"{field} is missing") + + if not isinstance(receipt.get("evidence_digests"), list): + problems.append("evidence_digests must be a list") + elif not all(_is_revision(item, allow_absent=False) for item in receipt["evidence_digests"]): + problems.append("every entry in evidence_digests must be a content digest") + + if not isinstance(receipt.get("published_at"), int): + problems.append("published_at must be an integer timestamp") + + # A failure that erased its reason leaves an operator with a stalled lane and nothing to read; + # a success carrying an error is a receipt that disagrees with itself. + if receipt.get("result") == "failed" and not receipt.get("error"): + problems.append("a failed receipt must carry an error explaining why") + if receipt.get("result") == "published" and receipt.get("error"): + problems.append("a published receipt must not carry an error") + return problems diff --git a/ingot/review.py b/ingot/review.py new file mode 100644 index 0000000..e99d12b --- /dev/null +++ b/ingot/review.py @@ -0,0 +1,418 @@ +"""Deterministic, offline, read-only review of one skill package. + +Six sections, never one number. A composite score invites exactly the reward-hacking the evidence +gate exists to prevent, and it hides which of six unrelated questions actually failed. + +What this command will not do: + +- run a model, read a key, reach the network, or start a service; +- convert similarity to another skill into an "activation score" -- that measures collision, not + whether the router loads this skill at the right moment; +- pattern-match for markers and call the result prompt-injection safety; +- write anything, anywhere. + +Where a question needs evidence this command cannot produce, it reports UNMEASURED and names the +command that can. An honest gap beats a confident guess.""" +from __future__ import annotations + +import codecs +import json +import re +import unicodedata +from pathlib import Path + +import yaml + +from ingot.mcp_server.registry import SLUG_RE, skill_revision, skill_sources + +from .parse import ERROR, INFO, WARNING, Finding, parse_raw + +REVIEW_SCHEMA = "ingot/review/v1" + +MEASURED = "measured" +UNMEASURED = "unmeasured" + +# Thresholds are advisory and deliberately loose: they exist to surface a package that will surprise +# someone, not to impose a house style. +LARGE_BODY_BYTES = 40_000 +LARGE_PACKAGE_BYTES = 10_000_000 +MANY_FILES = 200 + +# A relative link in the body that points at something in the package. Skips URLs and anchors. +_LOCAL_LINK = re.compile(r"\[[^\]]*\]\(\s*(?!\w+:|#)([^)\s]+)") +_URL = re.compile(r"https?://[^\s<>\"')\]]+") +# Mutable by construction: a branch ref rather than a commit or tag. +_MUTABLE_REF = re.compile(r"https?://[^\s]*/(?:blob|tree|raw)/(?:main|master|HEAD)/") +_WINDOWS_RESERVED = set('<>:"|?*') + + +def _section(status: str = MEASURED, **extra) -> dict: + return {"status": status, "findings": [], **extra} + + +def _add(section: dict, *findings: Finding) -> None: + section["findings"].extend(finding.as_dict() for finding in findings) + + +def _package_files(package: Path) -> tuple[list[Path], list[Finding]]: + """Every regular file in the package, plus a finding for any symlink. + + Symlinks are refused rather than resolved. Admission stages a package as exact bytes, and there + are only two things it could do with a link: preserve it, which puts a path into the vault that + a reader follows back out of the library, or flatten it into a copy of its target, which + silently changes the artifact's shape. Neither is a call to make on an operator's behalf, so a + package containing one is refused by name. This walks explicitly instead of using `rglob`, + which follows directory symlinks and would pull in a subtree without ever reporting the link.""" + files, findings = [], [] + stack = [package] + while stack: + for path in sorted(stack.pop().iterdir()): + if path.is_symlink(): + findings.append(Finding( + "symlink-unsupported", ERROR, + "symlinks are not admissible: admission stages exact bytes, and a link is " + "neither preserved nor followed", + path.relative_to(package).as_posix())) + elif path.is_dir(): + stack.append(path) + elif path.is_file(): + files.append(path) + return sorted(files), findings + + +# Metadata no packaging format treats as skill content, and nothing a reviewer needs flagged. +IGNORED_PARTS = {".git", "__pycache__", ".pytest_cache"} +IGNORED_NAMES = {".DS_Store", "Thumbs.db"} + + +def binary_assets(package: Path, files: list[Path]) -> list[str]: + """Files whose bytes are not text, so a reviewer knows what they are approving unread. + + Admission preserves these byte-for-byte, which is why this is a note rather than a refusal. It + is still the fact a reviewer most needs in front of them: a skill's text can be read before it + is approved, and a compiled binary or an image cannot. Decodability, not the file extension, + decides -- an extension is a claim about a file, and the point here is to check the file.""" + found = [] + for path in files: + relative = path.relative_to(package) + if relative.name in IGNORED_NAMES or IGNORED_PARTS.intersection(relative.parts): + continue + # Decoded in chunks rather than read whole: this runs before any size limit applies, and a + # review command that a large file can exhaust memory on is one nobody runs on the packages + # that most need reviewing. + decoder = codecs.getincrementaldecoder("utf-8")() + try: + with path.open("rb") as handle: + while chunk := handle.read(1 << 20): + decoder.decode(chunk) + decoder.decode(b"", final=True) + except (UnicodeDecodeError, OSError): + found.append(relative.as_posix()) + return found + + +def _structural(package: Path, files: list[Path], escapes: list[Finding]) -> tuple[dict, dict | None]: + section = _section() + _add(section, *escapes) + + skill_md = package / "SKILL.md" + if not skill_md.is_file(): + _add(section, Finding("skill-md-missing", ERROR, + "no SKILL.md: a skill package is a directory containing one")) + return section, None + + raw = parse_raw(skill_md.read_text(encoding="utf-8", errors="replace")) + _add(section, *raw.findings) + frontmatter = raw.frontmatter or {} + + name = str(frontmatter.get("name") or "").strip() + if raw.frontmatter is not None: + if not name: + _add(section, Finding("name-missing", ERROR, "frontmatter declares no name")) + elif not SLUG_RE.fullmatch(name): + _add(section, Finding( + "name-invalid", ERROR, + f"name {name!r} is not a valid slug (lowercase letters, digits and hyphens, " + f"starting with a letter or digit)")) + elif name != package.name: + _add(section, Finding( + "name-directory-mismatch", WARNING, + f"frontmatter name {name!r} differs from the directory name {package.name!r}; " + f"the frontmatter name is the identity that will be served")) + + if not str(frontmatter.get("description") or "").strip(): + _add(section, Finding( + "description-empty", ERROR, + "no description: the router keys on it, so a skill without one is never loaded")) + + body_bytes = len(raw.body.encode("utf-8")) + if not raw.body.strip(): + _add(section, Finding("body-empty", WARNING, "the body is empty")) + elif body_bytes > LARGE_BODY_BYTES: + _add(section, Finding( + "body-large", WARNING, + f"body is {body_bytes} bytes; every load pays for it in the agent's context")) + + total_bytes = sum(path.stat().st_size for path in files) + if len(files) > MANY_FILES: + _add(section, Finding("file-count-high", WARNING, + f"{len(files)} files in the package")) + if total_bytes > LARGE_PACKAGE_BYTES: + _add(section, Finding("package-large", WARNING, + f"package is {total_bytes} bytes")) + + _add(section, *_path_findings(package, files)) + _add(section, *_reference_findings(package, raw.body)) + binaries = binary_assets(package, files) + if binaries: + _add(section, Finding( + "binary-asset", WARNING, + f"{len(binaries)} file(s) are not text and cannot be read before approval; they will " + f"be published byte-for-byte: {', '.join(binaries[:10])}" + + (f", and {len(binaries) - 10} more" if len(binaries) > 10 else ""))) + + section.update( + name=name or package.name, + description=str(frontmatter.get("description") or "").strip(), + body_bytes=body_bytes, + file_count=len(files), + total_bytes=total_bytes, + file_types=sorted({path.suffix.lower() or "(none)" for path in files}), + binary_assets=binaries, + ) + return section, frontmatter if raw.frontmatter is not None else None + + +def _path_findings(package: Path, files: list[Path]) -> list[Finding]: + """Portability of the paths themselves: characters, case, and Unicode form. + + A pair of files that differ only by case or only by Unicode normalization is two files on Linux + and one on macOS or Windows. That makes the package's content revision depend on who checked it + out, which is the one property a content-addressed system cannot tolerate.""" + findings, by_fold, by_norm = [], {}, {} + for path in files: + relative = path.relative_to(package).as_posix() + if _WINDOWS_RESERVED.intersection(relative) or "\\" in relative: + findings.append(Finding( + "path-not-portable", WARNING, + "path contains characters that are not portable across filesystems", relative)) + by_fold.setdefault(relative.casefold(), []).append(relative) + by_norm.setdefault(unicodedata.normalize("NFC", relative), []).append(relative) + + for group in by_fold.values(): + if len(group) > 1: + findings.append(Finding( + "path-case-collision", WARNING, + f"paths differ only by case and collapse on a case-insensitive filesystem: " + f"{', '.join(sorted(group))}")) + for group in by_norm.values(): + if len(group) > 1 and len({unicodedata.normalize("NFC", p) for p in group}) == 1 \ + and len(set(group)) > 1: + findings.append(Finding( + "path-unicode-collision", WARNING, + f"paths differ only by Unicode normalization form: {', '.join(sorted(group))}")) + return findings + + +def _reference_findings(package: Path, body: str) -> list[Finding]: + """Local links in the body: do they escape the package, and do they resolve?""" + findings = [] + for target in _LOCAL_LINK.findall(body): + cleaned = target.split("#", 1)[0].strip() + if not cleaned: + continue + candidate = Path(cleaned) + if candidate.is_absolute() or ".." in candidate.parts: + findings.append(Finding( + "path-traversal", ERROR, + "reference points outside the package", cleaned)) + continue + if not (package / candidate).exists(): + findings.append(Finding( + "file-reference-missing", WARNING, + "referenced file is not in the package", cleaned)) + return findings + + +def _supply_chain(package: Path, files: list[Path], frontmatter: dict | None) -> dict: + """What the package declares about where it came from, and what it reaches for. + + Everything here is advisory and nothing here fails a package. These are the facts a reviewer + needs in front of them; none of them is a security verdict, and this command does not pretend + to detect prompt injection -- that is a semantic property no deterministic scan establishes.""" + section = _section() + declared = frontmatter or {} + + if not declared.get("source"): + _add(section, Finding("source-metadata-missing", WARNING, + "no source declared: the package does not record where it came from")) + if not declared.get("license"): + _add(section, Finding("license-metadata-missing", WARNING, + "no license declared")) + + executables = [path.relative_to(package).as_posix() for path in files + if path.stat().st_mode & 0o111 and not path.is_symlink()] + for relative in executables: + _add(section, Finding("executable-asset", WARNING, + "asset is executable", relative)) + + urls, mutable = set(), set() + for path in files: + try: + text = path.read_text(encoding="utf-8") + except (UnicodeDecodeError, OSError): + continue + urls.update(_URL.findall(text)) + mutable.update(_MUTABLE_REF.findall(text)) + + # One aggregate finding, not one per URL: a documentation-heavy skill has dozens of links, and + # a wall of identical warnings is how a reviewer learns to stop reading them. The full list + # stays in `remote_references` for anything consuming the JSON. + if urls: + _add(section, Finding( + "remote-reference", WARNING, + f"package references {len(urls)} remote URL(s): {', '.join(sorted(urls)[:3])}" + + (" …" if len(urls) > 3 else ""))) + if mutable: + _add(section, Finding( + "reference-unpinned", WARNING, + "reference points at a moving branch rather than a commit or tag; what it returns " + "today is not what it will return later")) + + section.update(source=declared.get("source"), license=declared.get("license"), + executables=executables, remote_references=sorted(urls)) + return section + + +def _collision(package: Path, name: str, library_root: Path | None) -> dict: + """Deterministic collision only. + + Description shadowing -- the thing that actually steals traffic -- is a cosine comparison that + needs the embedding router, which needs a model. Approximating it here would be worse than not + answering, so it reports UNMEASURED and names `ingot.optimize.routing_health`, which already does the + real library-wide scan.""" + semantic = {"status": UNMEASURED, + "reason": "description shadowing needs the embedding router", + "measure_with": "python -m ingot.optimize.routing_health"} + + if library_root is None: + return _section(UNMEASURED, reason="no library root supplied", semantic=semantic) + + # `skill_sources`, not a bare glob: it skips the dot-prefixed staging directories promotion and + # rollback leave beside a live skill, each of which carries a complete SKILL.md. Globbing + # directly would report an abandoned `.pdf..stage` as a colliding skill. + section = _section(semantic=semantic, library_root=str(library_root)) + existing = {path.parent.name for path in skill_sources(library_root) + if path.parent.resolve() != package.resolve()} + if name in existing: + _add(section, Finding( + "name-collision", WARNING, + f"the library already has a skill named {name!r}; one would shadow the other")) + for other in sorted(existing): + if other != name and other.casefold() == name.casefold(): + _add(section, Finding( + "name-case-collision", WARNING, + f"the library has {other!r}, which differs from {name!r} only by case")) + return section + + +def _activation(name: str, evidence_root: Path) -> dict: + """Whether a routing suite exists is a fact available offline. Whether the router loads this + skill at the right time is not, so it is never reported as a score.""" + suite = evidence_root / "ingot" / "optimize" / "tasks" / f"{name}.yaml" + cases = 0 + if suite.is_file(): + try: + loaded = yaml.safe_load(suite.read_text(encoding="utf-8")) or {} + except yaml.YAMLError: + loaded = {} + routing = loaded.get("routing") if isinstance(loaded, dict) else None + cases = len(routing) if isinstance(routing, list) else 0 + + return _section( + UNMEASURED, + routing_cases=cases, + reason=("a routing suite exists but scoring it needs the embedding router" + if cases else "no routing suite for this skill"), + measure_with=f"python -m ingot.optimize.routing_health {name}", + ) + + +def _behavioral(name: str, evidence_root: Path) -> dict: + """Surface what `ingot.optimize.compat` already measured. Never recompute it: that costs a model, a + key, and money, and this command promises none of the three.""" + path = evidence_root / "runs" / "compat" / f"{name}.json" + if not path.is_file(): + return _section(UNMEASURED, + reason="no compatibility evidence for this skill", + measure_with=f"python -m ingot.optimize.compat {name}") + try: + summary = json.loads(path.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError) as error: + return _section(UNMEASURED, reason=f"compatibility evidence is unreadable: {error}") + + return _section(MEASURED, source=str(path), tasks=summary.get("tasks"), + judge=summary.get("judge"), models=summary.get("models") or {}) + + +def review_package(package: Path, *, library_root: Path | None = None, + evidence_root: Path | None = None) -> dict: + """Review one skill package. Reads only; writes nothing, anywhere.""" + package = Path(package) + evidence_root = Path(evidence_root) if evidence_root is not None \ + else Path(__file__).resolve().parent.parent + + files, escapes = _package_files(package) + structural, frontmatter = _structural(package, files, escapes) + name = structural.get("name", package.name) + + try: + revision = skill_revision(package) + except (ValueError, OSError) as error: + revision = "" + _add(structural, Finding("revision-unavailable", ERROR, + f"cannot compute a content revision: {error}")) + + sections = { + "structural": structural, + "supply_chain": _supply_chain(package, files, frontmatter), + "collision": _collision(package, name, library_root), + "activation": _activation(name, evidence_root), + "behavioral": _behavioral(name, evidence_root), + } + + findings = [finding for section in sections.values() for finding in section["findings"]] + errors = sum(1 for finding in findings if finding["level"] == ERROR) + return { + "schema_version": REVIEW_SCHEMA, + "package": str(package), + "skill": name, + "revision": revision, + "valid": errors == 0, + "errors": errors, + "warnings": sum(1 for finding in findings if finding["level"] == WARNING), + "sections": sections, + } + + +def render(result: dict) -> str: + """One block per section, so a reader sees which question failed rather than a number.""" + verdict = "VALID" if result["valid"] else "INVALID" + lines = [f"{result['skill']} {verdict} {result['errors']} error(s), " + f"{result['warnings']} warning(s)", + f" package {result['package']}", + f" revision {result['revision'][:16] or '(unavailable)'}"] + + for title, section in result["sections"].items(): + status = "" if section["status"] == MEASURED else f" [{section['status'].upper()}]" + lines.append(f"\n{title.replace('_', ' ')}{status}") + if section["status"] == UNMEASURED and section.get("reason"): + lines.append(f" {section['reason']}") + if section.get("measure_with"): + lines.append(f" measure with: {section['measure_with']}") + for finding in section["findings"]: + where = f" ({finding['path']})" if finding.get("path") else "" + lines.append(f" {finding['level']:<7} {finding['code']}: {finding['message']}{where}") + if not section["findings"] and section["status"] == MEASURED: + lines.append(" no findings") + return "\n".join(lines) diff --git a/ingot/status.py b/ingot/status.py new file mode 100644 index 0000000..89b88db --- /dev/null +++ b/ingot/status.py @@ -0,0 +1,221 @@ +"""Whether this deployment's guarantees actually hold, reported as an observation. + +The verdict is not read back out of configuration. It compares what is served against what the +last successful release receipt says should be served, per skill, and reports the worst answer. +A configuration flag would have agreed with the claim rather than checked it, which is the failure +this command exists to catch.""" +from __future__ import annotations + +import os +from pathlib import Path + +STATUS_SCHEMA = "ingot/status/v1" + +MANAGED = "MANAGED" # served bytes are the last successful release +PENDING = "PENDING" # a proposal or publication is in flight +DRIFTED = "DRIFTED" # served bytes differ from the last successful release +UNMANAGED = "UNMANAGED" # no release receipt covers these bytes, or development mode + +# Worst first. Drift is the alarm; an unmanaged skill is one the guarantee never covered; a +# publication in flight is expected and transient. +SEVERITY = (DRIFTED, UNMANAGED, PENDING, MANAGED) +ABSENT = "absent" + + +def _writable(path: Path) -> bool: + """Whether this process could change what is served. + + `os.access` rather than a permission-bit reading: it accounts for the read-only mount managed + mode relies on, which no mode bit describes. Reported as a fact rather than folded into the + verdict — the administrator who owns the vault can always write it, and a status command that + answered UNMANAGED from their shell would hide the drift they most need to see.""" + return path.is_dir() and os.access(path, os.W_OK | os.X_OK) + + +def _worst(states) -> str: + for state in SEVERITY: + if state in states: + return state + return MANAGED + + +def skill_states(root: Path | None = None) -> list[dict]: + """One verdict per skill, plus every skill a release receipt names but nothing serves.""" + from ingot.mcp_server.registry import load_skills + from ingot.optimize.promote import list_pending + from ingot.optimize.publication import latest_releases, publishing_skills + + explicit = [root] if root is not None else None + served = {skill.name: skill.revision for skill in load_skills(roots=explicit)} + releases = latest_releases() + in_flight = publishing_skills() | {record.get("skill") for record in list_pending()} + + states = [] + # In-flight names join the union: a creation that is quarantined or travelling is served by + # nothing and released by nothing, so a status built from those two sets alone would report an + # empty library as fully MANAGED while a publication was in progress. + for name in sorted(set(served) | set(releases) | {name for name in in_flight if name}): + current = served.get(name, ABSENT) + release = releases.get(name) + released = release.get("candidate_revision") if release else None + if released is None: + # Nothing Ingot published is responsible for these bytes: they were fetched, copied, + # or committed to the vault by hand. Real, common, and not drift — there is no release + # to have drifted from. + state = PENDING if name in in_flight else UNMANAGED + elif current != released: + state = DRIFTED + elif name in in_flight: + state = PENDING + else: + state = MANAGED + states.append({"skill": name, "state": state, "revision": current, + "released": released, + "publication": release.get("id") if release else None}) + return states + + +def target_states(env: dict | None = None) -> list[dict]: + """One verdict per delivery target, decided the same way as the library's: by looking. + + A target is graded only on the skills Ingot released there or has in flight for it. A native + skill root is shared with whatever its owner put in it, and those skills are not Ingot's to + judge -- grading them would report every real deployment as permanently UNMANAGED and bury the + one line that matters. A released skill that has been deleted from a target is still drift: + absence is a revision, and it is not the released one.""" + from ingot import delivery + from ingot.optimize.promote import list_pending + from ingot.optimize.publication import latest_releases, publishing_skills + + from . import paths + + env = os.environ if env is None else env + targets = delivery.load_targets(env, vault=paths.vault(env)) + releases = latest_releases() + in_flight = {name for name in + publishing_skills() | {record.get("skill") for record in list_pending()} if name} + + reported = [] + for target in targets: + skills = [] + for name in sorted(set(releases) | in_flight): + released = (releases.get(name) or {}).get("candidate_revision") + current = delivery.observed(target, name) + if released is None: + state = PENDING if name in in_flight else UNMANAGED + elif current != released: + state = DRIFTED + elif name in in_flight: + state = PENDING + else: + state = MANAGED + skills.append({"skill": name, "state": state, "revision": current, + "released": released}) + reported.append({"name": target.name, "kind": target.kind, "root": str(target.root), + "state": _worst({entry["state"] for entry in skills}), "skills": skills}) + return reported + + +def library_status(root: Path | None = None, env: dict | None = None) -> dict: + from ingot.mcp_server.registry import configured_roots + + from . import paths + + env = os.environ if env is None else env + # An explicit root means exactly that root. `configured_roots` always prepends the local + # authoring library even ahead of one, which is right for serving and wrong here: asked whether + # a particular library is managed, this must not answer about a different one. + roots = ([Path(root).expanduser().resolve()] if root is not None + else [Path(path) for path in configured_roots()]) + development = (env.get("INGOT_MODE") or "").strip().lower() in {"dev", "unmanaged"} + states = skill_states(root) + # A misconfigured target list must not take the command down. Status is what an operator runs + # when something is already wrong, so a broken variable is a finding to report, not a crash. + try: + targets, delivery_error = target_states(env), None + except (ValueError, OSError) as exc: + targets, delivery_error = [], str(exc) + verdicts = {entry["state"] for entry in states} + verdicts |= {entry["state"] for entry in targets} + if delivery_error: + verdicts.add(DRIFTED) + mode = UNMANAGED if development else _worst(verdicts) + return { + "schema_version": STATUS_SCHEMA, + "mode": mode, + "development_mode": development, + "uid": os.getuid(), + "roots": [str(path) for path in roots], + "writable_roots": [str(path) for path in roots if _writable(path)], + "skills": states, + "targets": targets, + "delivery_error": delivery_error, + "publish_backend": env.get("INGOT_PUBLISH_BACKEND") or "local", + "vault_path": str(paths.vault(env)), + "forge_repository": env.get("INGOT_FORGE_REPOSITORY"), + "paths": paths.resolved(env), + # Never migrated, only reported: moving a review queue or a receipt store on someone's + # behalf is a change to controlled state made by a process nobody asked to make it. + "legacy_state": paths.legacy_state(), + } + + +_EXPLANATION = { + MANAGED: "Every served skill is exactly the revision its last release receipt names.", + PENDING: "A proposal or publication is in flight. Nothing has drifted.", + DRIFTED: "Served bytes differ from the last successful release. Something changed them " + "outside the publisher.", + UNMANAGED: "Some served bytes have no release receipt behind them, so the quarantine and " + "publication guarantees do not describe them.", +} + + +def render(result: dict) -> str: + lines = [f"{result['mode']} (uid {result['uid']})", ""] + if result["development_mode"]: + lines.append(" Development mode (INGOT_MODE). The served library is writable by every") + lines.append(" service and control-plane guarantees do not apply to this deployment.") + else: + lines.append(f" {_EXPLANATION[result['mode']]}") + for entry in result["skills"]: + if entry["state"] == MANAGED: + continue + detail = (f"served {entry['revision'][:12]} != released {entry['released'][:12]}" + if entry["state"] == DRIFTED else + "in flight" if entry["state"] == PENDING else "no release receipt") + lines.append(f" {entry['state']:<10} {entry['skill']:<20} {detail}") + if result["writable_roots"] and not result["development_mode"]: + lines += ["", " Writable by this process, so nothing here stops this user from changing", + " what is served without an approval:"] + lines += [f" {path}" for path in result["writable_roots"]] + if result.get("delivery_error"): + lines += ["", " The delivery target configuration cannot be read, so nothing here knows", + f" what this deployment installs or where: {result['delivery_error']}"] + for target in result.get("targets") or []: + drifted = [entry for entry in target["skills"] if entry["state"] != MANAGED] + if target["state"] == MANAGED and not drifted: + continue + lines += ["", f" {target['state']:<10} target {target['name']} ({target['kind']}) " + f"{target['root']}"] + for entry in drifted: + detail = (f"holds {entry['revision'][:12]} != released {entry['released'][:12]}" + if entry["state"] == DRIFTED else + "in flight" if entry["state"] == PENDING else "no release receipt") + lines.append(f" {entry['state']:<10} {entry['skill']:<20} {detail}") + lines += ["", f" backend {result['publish_backend']}"] + if result["forge_repository"]: + lines.append(f" forge {result['forge_repository']}") + lines.append(f" roots {', '.join(result['roots'])}") + for target in result.get("targets") or []: + lines.append(f" deliver {target['name']:<12} {target['kind']:<12} {target['root']} " + f"{target['state']}") + lines += ["", " state"] + width = max(len(entry["name"]) for entry in result["paths"]) + for entry in result["paths"]: + note = "" if entry["writable"] else " NOT WRITABLE" + lines.append(f" {entry['name']:<{width}} {entry['path']} [{entry['source']}]{note}") + if result["legacy_state"]: + lines += ["", " State left beside the code by an earlier version. Nothing here reads it;", + " move what you want to keep under the paths above, then delete it:"] + lines += [f" {path}" for path in result["legacy_state"]] + return "\n".join(lines) diff --git a/ingot/vault.py b/ingot/vault.py new file mode 100644 index 0000000..79b3067 --- /dev/null +++ b/ingot/vault.py @@ -0,0 +1,138 @@ +"""Create the local Git vault the publisher owns. + +The local backend needs a Git repository before it can publish anything, and a deployment whose +very first start fails because the directory is empty is a deployment nobody gets past. This is the +one bootstrap command, and it is idempotent so the managed compose can call it on every start.""" +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + +VAULT_SCHEMA = "ingot/vault/v1" + +# Written into the vault, not imported from it: the publisher runs this with the vault as the +# working directory and nothing but the interpreter guaranteed to be there. It refuses the tree +# rather than the change, so a vault a server could not load can never be committed. +VALIDATOR = '''#!/usr/bin/env python3 +"""Refuse a vault tree the skill server could not serve. + +Every publication runs this against a fresh worktree before the commit is made.""" +import json +import re +import sys +from pathlib import Path + +import yaml + +FRONTMATTER = re.compile(r"\\A---\\r?\\n(.*?)\\r?\\n---\\r?\\n?", re.DOTALL) +SLUG = re.compile(r"[a-z0-9]+(?:-[a-z0-9]+)*\\Z") + +root = Path(__file__).resolve().parent.parent +problems = [] + +registry = root / "registry.json" +if registry.is_file(): + try: + if not isinstance(json.loads(registry.read_text(encoding="utf-8")), dict): + problems.append("registry.json is not a JSON object") + except ValueError as exc: + problems.append(f"registry.json is not valid JSON: {exc}") + +for skill_md in sorted(root.glob("*/SKILL.md")): + name = skill_md.parent.name + where = f"{name}/SKILL.md" + if not SLUG.fullmatch(name): + problems.append(f"{name}: directory name is not a slug") + text = skill_md.read_text(encoding="utf-8") + match = FRONTMATTER.match(text) + if not match: + problems.append(f"{where}: no YAML frontmatter") + continue + try: + metadata = yaml.safe_load(match.group(1)) + except yaml.YAMLError as exc: + problems.append(f"{where}: frontmatter is not valid YAML: {exc}") + continue + if not isinstance(metadata, dict): + problems.append(f"{where}: frontmatter is not a mapping") + continue + if metadata.get("name") != name: + problems.append(f"{where}: frontmatter name {metadata.get('name')!r} != directory {name!r}") + if not str(metadata.get("description", "")).strip(): + problems.append(f"{where}: description is empty") + +for problem in problems: + print(f"invalid: {problem}", file=sys.stderr) +sys.exit(1 if problems else 0) +''' + + +def _git(path: Path, *args: str) -> str: + result = subprocess.run(["git", "-C", str(path), *args], capture_output=True, text=True) + if result.returncode: + detail = result.stderr.strip().splitlines()[-1] if result.stderr.strip() else "failed" + verb = next((arg for arg in args if not arg.startswith("-") and "=" not in arg), args[0]) + raise ValueError(f"git {verb}: {detail}") + return result.stdout.strip() + + +def _is_repository(path: Path) -> bool: + result = subprocess.run(["git", "-C", str(path), "rev-parse", "--git-dir"], + capture_output=True, text=True) + return result.returncode == 0 + + +def init_vault(path: Path, *, branch: str = "main") -> dict: + """Create or complete a vault. Idempotent against one that is already valid. + + A non-empty directory that is not a Git repository is refused rather than adopted: pointing the + publisher at a directory of loose skills would make its first commit look like a publication + nobody approved.""" + path = Path(path).expanduser() + existed = path.exists() + if existed and not path.is_dir(): + raise ValueError(f"{path} is not a directory") + if existed and not _is_repository(path) and any(path.iterdir()): + raise ValueError(f"{path} is not empty and is not a Git repository; " + f"initialize an empty directory or point at an existing vault") + path.mkdir(parents=True, exist_ok=True) + path = path.resolve() + + created = not _is_repository(path) + if created: + _git(path, "init", "-b", branch) + + written = [] + for relative, contents in ((Path("registry.json"), "{}\n"), + (Path("scripts") / "validate.py", VALIDATOR)): + target = path / relative + if target.exists(): + continue + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(contents, encoding="utf-8") + written.append(relative.as_posix()) + + head = None + if written or created: + _git(path, "add", "-A", ".") + if _git(path, "status", "--porcelain"): + _git(path, "-c", "user.name=Ingot Publisher", "-c", "user.email=ingot@local.invalid", + "commit", "-m", "Initialize the Ingot skill vault") + try: + head = _git(path, "rev-parse", "HEAD") + except ValueError: + # An adopted repository with no commits at all: give it one so a publication branch has a + # base to be cut from. + _git(path, "-c", "user.name=Ingot Publisher", "-c", "user.email=ingot@local.invalid", + "commit", "--allow-empty", "-m", "Initialize the Ingot skill vault") + head = _git(path, "rev-parse", "HEAD") + + return { + "schema_version": VAULT_SCHEMA, + "path": str(path), + "status": "created" if created else ("updated" if written else "unchanged"), + "branch": _git(path, "symbolic-ref", "--short", "HEAD"), + "head": head, + "added": written, + } diff --git a/mcp_server/router.py b/mcp_server/router.py deleted file mode 100644 index b9be636..0000000 --- a/mcp_server/router.py +++ /dev/null @@ -1,192 +0,0 @@ -"""Tier-1 embedding router: embed every skill's description once, then suggest the top-k skills for a -task by cosine similarity. CPU-only ONNX, no GPU, so the demo is `docker compose up`. - -Model is `EMBED_MODEL` (default Qwen3-Embedding-0.6B q4, ~15 ms/query on CPU; queries get the -retrieval instruction prefix, descriptions don't). Any fastembed model name also works (e.g. the -previous default `BAAI/bge-small-en-v1.5`, ~4 ms/query), but recalibrate MIN_SCORE / -RELATED_SCORE / COLLISION_SCORE with the model (mcp_server/embedding.py).""" -from __future__ import annotations -import os -import sys -import threading -from pathlib import Path - -import numpy as np - -from .embedding import EMBED_MODEL as _MODEL, build_embedding -from .registry import Skill - - -class Router: - _vector_cache: dict[tuple[str, str], np.ndarray] = {} - _cache_lock = threading.Lock() - - def __init__(self, skills: list[Skill]): - self.skills = skills - self._embed = build_embedding() - if not skills: # empty library, don't normalize an empty matrix - self._mat = np.zeros((0, 0), dtype=np.float32) - return - keys = [(_MODEL, skill.description) for skill in skills] - with self._cache_lock: - missing = list(dict.fromkeys(key for key in keys if key not in self._vector_cache)) - if missing: - vectors = self._embed.embed([description for _, description in missing]) - with self._cache_lock: - for key, vector in zip(missing, vectors): - self._vector_cache[key] = np.asarray(vector, dtype=np.float32) - with self._cache_lock: - mat = np.array([self._vector_cache[key] for key in keys], dtype=np.float32) - self._mat = mat / (np.linalg.norm(mat, axis=1, keepdims=True) + 1e-8) - - def nearest(self, text: str) -> tuple[str, float]: - """The most similar existing skill to `text` and its cosine score, used to reject a new - skill whose description near-duplicates (shadows) an existing one's routing.""" - if not self.skills: - return "", 0.0 - q = np.array(next(iter(self._embed.embed([text]))), dtype=np.float32) - q = q / (np.linalg.norm(q) + 1e-8) - scores = self._mat @ q - i = int(np.argmax(scores)) - return self.skills[i].name, float(scores[i]) - - def suggest(self, task: str, k: int = 5, min_score: float = 0.0) -> list[dict]: - if not self.skills: - return [] - q = np.array(next(iter(self._embed.embed_query([task]))), dtype=np.float32) - q = q / (np.linalg.norm(q) + 1e-8) - scores = self._mat @ q - top = np.argsort(-scores)[:k] - return [ - {"name": self.skills[i].name, "description": self.skills[i].description, - "score": round(float(scores[i]), 3)} - for i in top if scores[i] >= min_score - ] - - @staticmethod - def _platform(value: str | None) -> str: - value = (value or sys.platform).lower() - if value.startswith("darwin") or value == "macos": - return "macos" - if value.startswith("win"): - return "windows" - return "linux" if value.startswith("linux") else value - - @staticmethod - def _compatible(skill: Skill, harness: str, cwd: str, available_tools: set[str], - available_mcps: set[str], platform: str) -> bool: - meta = skill.metadata or {} - if harness not in meta.get("harnesses", ["claude", "codex"]): - return False - if platform not in meta.get("platforms", ["macos", "linux", "windows"]): - return False - if meta.get("activation", "automatic") != "automatic" or meta.get("trust") == "blocked": - return False - if not set(meta.get("required_tools", [])).issubset(available_tools): - return False - if not set(meta.get("required_mcps", [])).issubset(available_mcps): - return False - scopes = meta.get("scopes", ["global"]) - if "global" not in scopes: - patterns = meta.get("path_patterns", []) - if "project" not in scopes or not patterns: - return False - project = Path(cwd).expanduser().resolve() - if not project.is_dir() or not any(any(project.glob(pattern)) for pattern in patterns): - return False - return True - - def _eligible_ranking(self, task: str, harness: str, cwd: str, - available_tools: set[str], available_mcps: set[str], - platform: str) -> list[tuple[Skill, float]]: - eligible = [skill for skill in self.skills if self._compatible( - skill, harness, cwd, available_tools, available_mcps, platform - )] - if not eligible: - return [] - query = np.array(next(iter(self._embed.embed_query([task]))), dtype=np.float32) - query = query / (np.linalg.norm(query) + 1e-8) - index_by_name = {skill.name: index for index, skill in enumerate(self.skills)} - return sorted( - ((skill, float(self._mat[index_by_name[skill.name]] @ query)) for skill in eligible), - key=lambda pair: (-pair[1], -int(pair[0].metadata.get("priority", 50)), pair[0].name), - ) - - @staticmethod - def _without_conflicts(ranked: list[tuple[Skill, float]]) -> list[tuple[Skill, float]]: - selected = [] - for candidate in ranked: - skill = candidate[0] - if any(skill.name in set(existing.metadata.get("conflicts", [])) or - existing.name in set(skill.metadata.get("conflicts", [])) - for existing, _ in selected): - continue - selected.append(candidate) - return selected - - @staticmethod - def _alternatives(ranked: list[tuple[Skill, float]]) -> list[dict]: - return [ - {"name": skill.name, "score": round(score, 3), - "reason": f"compatible alternative; cosine {score:.3f}"} - for skill, score in ranked[1:3] - ] - - @staticmethod - def _novel_response(score: float = 0.0, reason: str = "no compatible skill candidates", - alternatives: list[dict] | None = None) -> dict: - return { - "match": None, "related_match": None, "score": round(score, 3), - "reason": reason, "skill_body": "", "skill_root": None, "revision": None, - "alternatives": alternatives or [], "novel": True, - } - - @staticmethod - def _related_response(skill: Skill, score: float, harness: str, min_score: float, - alternatives: list[dict]) -> dict: - return { - "match": None, "related_match": skill.name, "score": round(score, 3), - "reason": (f"best compatible score {score:.3f} below direct threshold " - f"{min_score:.3f}; loaded for compose or extend"), - "skill_body": skill.body_for(harness), - "skill_root": skill.root or str(os.path.dirname(skill.path)), - "revision": skill.revision or None, "alternatives": alternatives, "novel": False, - } - - @staticmethod - def _direct_response(skill: Skill, score: float, harness: str, - alternatives: list[dict]) -> dict: - return { - "match": skill.name, "related_match": None, "score": round(score, 3), - "reason": f"compatible {harness} skill; cosine {score:.3f}", - "skill_body": skill.body_for(harness), - "skill_root": skill.root or str(os.path.dirname(skill.path)), - "revision": skill.revision or None, "alternatives": alternatives, "novel": False, - } - - def route(self, task: str, harness: str, cwd: str, available_tools=(), available_mcps=(), - platform: str | None = None, min_score: float = 0.53, - related_score: float = 0.37) -> dict: - """Filter compatible skills, rank them locally, and return at most one instruction body. - `novel` is the escalation signal for the calling harness: True when nothing compatible is - even related (best score below `related_score`), the case where a weak/strong setup should - serve with the strong model, then queue a candidate for human review.""" - harness = harness.lower() - ranked = self._eligible_ranking( - task, harness, cwd, set(available_tools), set(available_mcps), self._platform(platform) - ) - if not ranked: - return self._novel_response() - ranked = self._without_conflicts(ranked) - top, score = ranked[0] - alternatives = self._alternatives(ranked) - if score < min_score: - if score < related_score: - related = [{"name": top.name, "score": round(score, 3), - "reason": f"best compatible candidate; cosine {score:.3f}"}, - *alternatives] - reason = (f"best compatible score {score:.3f} below related threshold " - f"{related_score:.3f}") - return self._novel_response(score, reason, related[:3]) - return self._related_response(top, score, harness, min_score, alternatives) - return self._direct_response(top, score, harness, alternatives) diff --git a/ops/systemd/ingot-publisher.service b/ops/systemd/ingot-publisher.service new file mode 100644 index 0000000..03b23ca --- /dev/null +++ b/ops/systemd/ingot-publisher.service @@ -0,0 +1,33 @@ +[Unit] +# The one writer of the served skill library. It publishes approved receipts into the vault and +# activates only the revision the receipt names. +# +# Running it on the host rather than in Docker sidesteps the uid mismatch the compose file warns +# about — the receipts it reads are written at mode 0700 by the console — and, in `forge` mode, it +# reuses the host's already authenticated git and gh so no GitHub credential has to live in a +# container or in .env. The compose stack ships an equivalent `publisher` service for the +# all-in-one case. Whichever you run, exactly one must run. +# +# Install (as the user who owns the vault): +# mkdir -p ~/.config/ingot && cp ops/systemd/publisher.env.example ~/.config/ingot/publisher.env +# $EDITOR ~/.config/ingot/publisher.env # at minimum INGOT_VAULT_PATH +# systemctl --user enable --now ingot-publisher +Description=Ingot skill vault publisher +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +# Where this checkout lives. Override with a drop-in rather than editing this unit: +# systemctl --user edit ingot-publisher +WorkingDirectory=%h/Source/ingot +Environment=PYTHONUNBUFFERED=1 +# Backend, vault path, and any forge settings. The publisher refuses to start without a vault +# rather than falling back to a writable default, so this file is required. +EnvironmentFile=%h/.config/ingot/publisher.env +ExecStart=/usr/bin/python3 -m ingot.optimize.publisher --watch +Restart=always +RestartSec=10 + +[Install] +WantedBy=default.target diff --git a/ops/systemd/publisher.env.example b/ops/systemd/publisher.env.example new file mode 100644 index 0000000..f9a7a4f --- /dev/null +++ b/ops/systemd/publisher.env.example @@ -0,0 +1,16 @@ +# Copy to ~/.config/ingot/publisher.env. Read by ops/systemd/ingot-publisher.service. + +# local (default): the vault is a Git repository on this machine and nothing leaves it. +# forge: publication authority is a merged pull request; needs the network and an authenticated gh. +INGOT_PUBLISH_BACKEND=local + +# The vault checkout this publisher owns, and the library the server serves. Required: the +# publisher refuses to start rather than adopt a writable default. `ingot vault init ` +# creates one. Absolute: systemd does not expand `~` or its own specifiers inside an +# EnvironmentFile. +INGOT_VAULT_PATH=/home/you/ingot-vault + +# forge only. Setting these under the local backend is inert and the publisher says so at startup. +# INGOT_FORGE_REPOSITORY=owner/repo +# INGOT_FORGE_REMOTE=origin +# INGOT_FORGE_BRANCH=main diff --git a/optimize/compat.py b/optimize/compat.py deleted file mode 100644 index 36d39bc..0000000 --- a/optimize/compat.py +++ /dev/null @@ -1,108 +0,0 @@ -"""Cross-model skill compatibility, how well a skill's body transfers across serving models. - -A skill body is tuned for one serving model (`AGENT_MODEL`); SkillOpt's own result is that good -skills transfer, but not always. For each model in `COMPAT_MODELS`, this runs the skill's held-out -tasks through the one serving contract twice, once with the skill body, once with an empty body -(the no-skill baseline), judges both with the FIXED judge, and reports per-model **lift** -(skill mean − baseline mean). Positive lift = the body helps that model; ~0 = the model already -knows this and the body is dead weight there. - -Langfuse-free: it reuses the direct rollout + judge (the same path the inner loop uses), so it needs -no trace backend or experiment logging. Only the *serving* model varies, the judge stays fixed so -scores are comparable across models. - -Usage: python -m optimize.compat -Config: COMPAT_MODELS=qwen/qwen3-32b,openai/gpt-5.5,anthropic/claude-sonnet-... (default: AGENT_MODEL) -""" -import json -import os -import statistics -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -from langchain_openai import ChatOpenAI - -from mcp_server.registry import SKILLS_DIR, optimizable_components - -from . import SERVE_TEMPLATE, agent_model, client_kwargs, model_api_key, model_base_url -from . import usage as usage_ledger -from .ab import load_tasks -from .judge import invoke_retry, judge -from .rollout import assemble - -_MAX_WORKERS = 8 -COMPAT_DIR = Path(__file__).resolve().parent.parent / "runs" / "compat" -# The no-skill baseline: the identical serving contract with no skill body, so `lift` isolates the -# body's contribution rather than the difference between two different prompts. -NO_SKILL_BODY = "(no skill loaded, answer the task from your own knowledge)" - - -def compat_models() -> list[str]: - """Models to sweep: COMPAT_MODELS (comma-separated), else just the configured AGENT_MODEL.""" - models = [m.strip() for m in os.environ.get("COMPAT_MODELS", "").split(",") if m.strip()] - return models or [agent_model()] - - -def _llm(model: str): - # OpenRouter (default) selects the model by slug over one endpoint, with ZDR routing applied; - # a local MODEL_BASE_URL serves a single model, so a multi-model sweep only makes sense on a - # multi-model endpoint. reasoning is left at the provider default, some models reject the flag. - return ChatOpenAI(model=model, temperature=0, **client_kwargs(model_base_url(), key=model_api_key())) - - -def _score(llm, system: str, task: dict) -> float: - msg = invoke_retry(llm, [("system", system), ("user", task["task"])]) - usage_ledger.add("compat", getattr(msg, "usage_metadata", None)) - return judge(task["task"], task["rubric"], msg.content, - check=task.get("check"), deliverable=task.get("deliverable"))["score"] - - -def _run_arm(llm, system: str, tasks: list[dict]) -> list[float]: - with ThreadPoolExecutor(max_workers=min(_MAX_WORKERS, len(tasks))) as pool: - return list(pool.map(lambda t: _score(llm, system, t), tasks)) - - -def run_compat(skill: str, log=print) -> dict: - """Sweep COMPAT_MODELS over the skill's held-out tasks (skill vs no-skill) and write the matrix - to runs/compat/.json. Returns the summary.""" - usage_ledger.reset() - if not (SKILLS_DIR / skill / "SKILL.md").exists(): - raise SystemExit(f"No skill named '{skill}' in skills/.") - _, holdout, _ = load_tasks(skill) - if not holdout: - raise SystemExit(f"'{skill}' has no held-out eval tasks to run.") - skill_system = SERVE_TEMPLATE.format(body=assemble(optimizable_components(SKILLS_DIR / skill))) - base_system = SERVE_TEMPLATE.format(body=NO_SKILL_BODY) - models = compat_models() - log(f"[compat] '{skill}': {len(holdout)} held-out tasks × {len(models)} model(s); " - f"judge fixed, serving model varies") - - models_out = {} - for model in models: - llm = _llm(model) - skill_scores = _run_arm(llm, skill_system, holdout) - base_scores = _run_arm(llm, base_system, holdout) - s_mean, b_mean = statistics.mean(skill_scores), statistics.mean(base_scores) - models_out[model] = {"skill_mean": s_mean, "baseline_mean": b_mean, "lift": s_mean - b_mean, - "skill_scores": skill_scores, "baseline_scores": base_scores} - verdict = "helps" if s_mean - b_mean > 0.05 else "no lift" if s_mean - b_mean >= -0.05 else "HURTS" - log(f"[compat] {model:<34} skill {s_mean:.3f} baseline {b_mean:.3f} " - f"lift {s_mean - b_mean:+.3f} ({verdict})") - - summary = {"skill": skill, "tasks": len(holdout), - "judge": os.environ.get("JUDGE_MODELS") or os.environ.get("JUDGE_MODEL", ""), - "models": models_out, "usage": usage_ledger.report()} - COMPAT_DIR.mkdir(parents=True, exist_ok=True) - path = COMPAT_DIR / f"{skill}.json" - path.write_text(json.dumps(summary, indent=2)) - log(f"[compat] matrix written to {path}") - log(usage_ledger.format_report()) - return summary - - -if __name__ == "__main__": - import sys - - from . import require_openrouter_key - require_openrouter_key() - run_compat(sys.argv[1] if len(sys.argv) > 1 else "tailwind") diff --git a/optimize/judge.py b/optimize/judge.py deleted file mode 100644 index 2c08b97..0000000 --- a/optimize/judge.py +++ /dev/null @@ -1,148 +0,0 @@ -"""LLM judge: scores an answer 0..1 and, following the SkillForge paper's multi-dimensional Failure -Analyzer (Liu et al., "SkillForge", arXiv:2604.08618), classifies each failure across fixed -dimensions so the search gets *categorized* feedback, not one opaque score. The dimension labels -also drive success/failure mining (optimize/mine.py) and the candidate search's diagnosis. - -Judges against a task `rubric` when given one; with no rubric it grades reference-free (used when -mining real traces). If a task supplies a `reference` answer, consistency-against-reference is added -to the prompt (the paper's Consistency-Rate signal, lower variance than a rubric alone).""" -import json -import os -import re -import time - -from langchain_openai import ChatOpenAI - -from . import usage as usage_ledger - -# Reward-hacking guard: the judge must NOT be the same model as SKILLOPT_MODEL. -# If the author and the grader share blind spots, the search learns to please the judge instead of -# improving the skill. Default judge is a model distinct from both the reflection LM (GLM) and the -# student (Qwen). -# JUDGE_MODELS (comma-separated) runs an ensemble and averages, harder still to game. -MODELS = [m.strip() for m in os.environ.get( - "JUDGE_MODELS", os.environ.get("JUDGE_MODEL", "google/gemini-2.5-flash")).split(",") if m.strip()] -from . import ZDR_PROVIDER, client_kwargs, skillopt_model, teacher_base_url # noqa: E402 - -if skillopt_model() in MODELS: - print(f"[judge] WARNING: judge model {MODELS} includes the teacher model, which invites " - f"reward-hacking (author == grader). Set JUDGE_MODEL to a different model.", flush=True) - -# Failure dimensions (the general-purpose analogue of the paper's Knowledge/Tool/Clarification/Style). -DIMENSIONS = ["correctness", "completeness", "instruction_following", "efficiency"] - -_PROMPT = """You are grading an AI assistant's answer to a task. - -TASK: {task} -{rubric_block}{reference_block} -ASSISTANT'S ANSWER: -{answer} -{code_block} -Score the answer from 0.0 to 1.0, and write one short paragraph of concrete, actionable feedback. -Treat any OBJECTIVE CODE CHECK above as ground truth, do not rate broken or absent code highly. -Then classify each failure dimension as "pass" or a short (<=12 word) note on what's wrong: -- correctness: is the core logic / API usage right? -- completeness: does it cover the whole request, including edge cases named above? -- instruction_following: did it do what was asked (e.g. output complete runnable code, not a description)? -- efficiency: is it concise, without wasted or padded output? - -Respond with ONLY a JSON object: -{{"score": , "feedback": "", "dimensions": {{"correctness": "...", "completeness": "...", "instruction_following": "...", "efficiency": "..."}}}}""" - -_llms: dict[str, ChatOpenAI] = {} - - -def _get_llm(model: str): - if model not in _llms: # built once per model, reuses the HTTP pool across many judge calls - _llms[model] = ChatOpenAI(model=model, temperature=0, **client_kwargs(teacher_base_url())) - return _llms[model] - - -# OpenRouter phrasings that mean "your model/provider configuration can never work", retrying -# only burns time, so explain and stop instead. -_PERMANENT = ("no allowed providers", "no providers are available", "not a valid model", - "no endpoints found", "is not available") - - -def _config_error(exc: Exception) -> str | None: - text = str(exc).lower() - if any(marker in text for marker in _PERMANENT): - pins = os.environ.get("OPENROUTER_PROVIDERS", "") - hint = (f" You have OPENROUTER_PROVIDERS={pins}, the pinned provider may not serve this " - f"model, or may not be ZDR-qualified for it; unset the pin or change the model." - if pins else - " No ZDR-qualified endpoint may exist for this model; try another model.") - return f"OpenRouter cannot route this request: {exc}.{hint}" - return None - - -def invoke_retry(llm, messages, tries: int = 3): - """Retry transient provider failures (corrupted responses, 5xx) with a short backoff. - Permanent configuration errors (model/provider mismatch) fail immediately with an explanation - instead of retrying.""" - for i in range(tries): - try: - return llm.invoke(messages) - except Exception as exc: - explained = _config_error(exc) - if explained: - raise SystemExit(explained) from exc - if i == tries - 1: - raise - time.sleep(5 * (i + 1)) - - -def _extract_json(text: str) -> dict: - """First valid JSON object with a 'score' key, robust to prose/braces around the JSON.""" - dec = json.JSONDecoder() - for m in re.finditer(r"\{", text): - try: - obj, _ = dec.raw_decode(text[m.start():]) - except json.JSONDecodeError: - continue - if isinstance(obj, dict) and "score" in obj: - return obj - return {} - - -def _judge_one(model: str, prompt: str) -> dict: - msg = invoke_retry(_get_llm(model), prompt) - usage_ledger.add("judge", getattr(msg, "usage_metadata", None)) - out = _extract_json(msg.content) - try: - dims = out.get("dimensions") or {} - return {"score": max(0.0, min(1.0, float(out["score"]))), "feedback": str(out.get("feedback", "")), - "dimensions": {d: str(dims.get(d, "pass")) for d in DIMENSIONS}} - except (KeyError, TypeError, ValueError): - return {"score": 0.0, "feedback": f"Judge output unparseable: {msg.content[:200]}", - "dimensions": {d: "pass" for d in DIMENSIONS}} - - -def judge(task: str, rubric: str = "", answer: str = "", reference: str = "", - check: dict | None = None, deliverable: str | None = None) -> dict: - """Return {score, feedback, dimensions}. With multiple JUDGE_MODELS this is an ensemble: score is - the mean, and a dimension counts as failed if a majority of judges flag it (harder to game). - `deliverable` (task yaml) declares the expected answer kind; non-code values skip the static - Python check, see execcheck.judge_note.""" - rubric_block = f"GRADING RUBRIC: {rubric}\n" if rubric else "" - reference_block = f"KNOWN-GOOD REFERENCE ANSWER (judge consistency against it): {reference}\n" if reference else "" - from . import execcheck # objective code-validity signal to ground the judge - code_note = execcheck.judge_note(answer, task, rubric, check_spec=check, deliverable=deliverable) - code_block = f"\n{code_note}\n" if code_note else "" - prompt = _PROMPT.format(task=task, answer=answer, rubric_block=rubric_block, - reference_block=reference_block, code_block=code_block) - results = [_judge_one(m, prompt) for m in MODELS] - if len(results) == 1: - return results[0] - score = sum(r["score"] for r in results) / len(results) - dims = {} - for d in DIMENSIONS: - notes = [r["dimensions"][d] for r in results if d in failed_dimensions(r["dimensions"])] - dims[d] = notes[0] if len(notes) * 2 > len(results) else "pass" # fail only on majority - feedback = " | ".join(f"[{m.split('/')[-1]}] {r['feedback']}" for m, r in zip(MODELS, results)) - return {"score": score, "feedback": feedback, "dimensions": dims} - - -def failed_dimensions(dimensions: dict) -> list[str]: - """Dimension names the judge did NOT mark as a clean pass.""" - return [d for d, v in dimensions.items() if str(v).strip().lower() not in ("pass", "ok", "", "n/a")] diff --git a/optimize/usage.py b/optimize/usage.py deleted file mode 100644 index 21beb58..0000000 --- a/optimize/usage.py +++ /dev/null @@ -1,103 +0,0 @@ -"""Token ledger for an optimize run: every LLM call is attributed to a role -(rollout / judge / reflection / agent_ab) so the run can report what it actually cost , -including a best-effort USD estimate from OpenRouter list prices, and an optional hard -spend cap (MAX_RUN_USD) that aborts a run before it exceeds the budget.""" -import os -import threading -from collections import defaultdict - -COUNTS: dict[str, dict[str, int]] = defaultdict(lambda: {"input": 0, "output": 0, "calls": 0}) -_LOCK = threading.RLock() # the search fans rollout+judge across a thread pool; add() re-enters for the cap -_PRICES: dict[str, tuple[float, float]] | None = None - - -def reset(): - """Start a fresh ledger, the UI process runs many optimizations; counts must not leak across runs.""" - with _LOCK: - COUNTS.clear() - - -def add(role: str, usage: dict | None): - """usage: langchain usage_metadata ({'input_tokens','output_tokens'}) or equivalent dict.""" - if not usage: - return - with _LOCK: - c = COUNTS[role] - c["input"] += int(usage.get("input_tokens", 0)) - c["output"] += int(usage.get("output_tokens", 0)) - c["calls"] += 1 - _enforce_cap() - - -def _enforce_cap(): - cap = float(os.environ.get("MAX_RUN_USD", "0") or 0) - if not cap: - return - cost = estimated_cost() - if cost is not None and cost > cap: - raise SystemExit(f"MAX_RUN_USD exceeded: estimated ${cost:.2f} > cap ${cap:.2f}, " - f"aborting before spending more.\n{format_report()}") - - -def _openrouter_prices() -> dict[str, tuple[float, float]]: - """model id -> (prompt, completion) USD per token from OpenRouter's public models API; - {} on any failure (cost reporting is best-effort, never a gate on offline work).""" - import json - import urllib.request - try: - with urllib.request.urlopen("https://openrouter.ai/api/v1/models", timeout=10) as r: - data = json.loads(r.read())["data"] - return {m["id"]: (float(m["pricing"]["prompt"]), float(m["pricing"]["completion"])) - for m in data if m.get("pricing")} - except Exception: - return {} - - -def _role_models() -> dict[str, str]: - """Which model each ledger role runs on (first judge only, for ensemble setups).""" - from . import agent_model, skillopt_model - teacher = skillopt_model() - judge = (os.environ.get("JUDGE_MODELS") or - os.environ.get("JUDGE_MODEL", "google/gemini-2.5-flash")).split(",")[0].strip() - return {"rollout": agent_model(), "agent_ab": agent_model(), - "judge": judge, "reflection": teacher} - - -def estimated_cost() -> float | None: - """Best-effort USD estimate for the current ledger, from OpenRouter list prices. None when - the endpoint isn't OpenRouter or pricing is unavailable (local endpoints cost nothing).""" - global _PRICES - from . import is_openrouter, teacher_base_url - if not is_openrouter(teacher_base_url()): - return None - if _PRICES is None: - _PRICES = _openrouter_prices() - if not _PRICES: - return None - models = _role_models() - with _LOCK: - return sum(c["input"] * p[0] + c["output"] * p[1] - for role, c in COUNTS.items() - if (p := _PRICES.get(models.get(role, "")))) - - -def report() -> dict: - out = {role: dict(c) for role, c in COUNTS.items()} - out["total"] = { - "input": sum(c["input"] for c in COUNTS.values()), - "output": sum(c["output"] for c in COUNTS.values()), - "calls": sum(c["calls"] for c in COUNTS.values()), - } - return out - - -def format_report() -> str: - r = report() - lines = [f" {role:<12} {c['calls']:>4} calls {c['input']:>9,} in {c['output']:>8,} out" - for role, c in r.items() if role != "total"] - t = r["total"] - lines.append(f" {'TOTAL':<12} {t['calls']:>4} calls {t['input']:>9,} in {t['output']:>8,} out") - cost = estimated_cost() - if cost is not None: - lines.append(f" estimated cost: ${cost:.2f} (OpenRouter list prices)") - return "\n".join(lines) diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..bfbd524 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,40 @@ +[build-system] +requires = ["setuptools>=77"] +build-backend = "setuptools.build_meta" + +[project] +name = "ingot" +version = "0.2.0" +description = "Release control for AI agent skills" +readme = "README.md" +requires-python = ">=3.12" +license = "Apache-2.0" + +# Deliberately minimal. `ingot list` reads SKILL.md folders through ingot.mcp_server.registry, whose only +# non-stdlib import is PyYAML, and the deterministic commands that follow must keep that property: +# a developer inspecting a skill should not install FastAPI, ONNX, LangGraph, Langfuse, or the +# optimizer to do it. The server, UI, and optimizer keep installing from requirements.txt, which +# stays the pinned set CI exercises -- this is a packaging shim over that, not a replacement for it. +dependencies = [ + "pyyaml==6.0.3", +] + +[project.scripts] +ingot = "ingot.cli:main" + +[tool.setuptools] +# One installed name. `ingot add` reaches the quarantine through `ingot.optimize.ingress`, so the +# optimizer package is installed too -- without it the command works under pytest, where the repo +# root is the working directory, and fails for anyone running the installed script from anywhere +# else. +# +# `mcp_server` and `optimize` were top-level packages until they moved under `ingot`. Both names are +# far too generic to claim on PyPI, and a compatibility shim would have been self-defeating: a shim +# named `optimize` still claims `optimize`. Nothing outside this repository imported them, so they +# were moved rather than aliased. +# +# `ui` and `agent` stay repo-root modules run with `python -m` from /app, as the Dockerfile and +# Compose already do. They are not installed, so they claim no name; moving them under `ingot` would +# make them installed and drag FastAPI and the agent scaffold into a CLI whose whole point is that +# `ingot list` works in a bare virtualenv. +packages = ["ingot", "ingot.mcp_server", "ingot.optimize"] diff --git a/scripts/claude_langfuse_smoke.sh b/scripts/claude_langfuse_smoke.sh index 2b772bf..85a13e3 100755 --- a/scripts/claude_langfuse_smoke.sh +++ b/scripts/claude_langfuse_smoke.sh @@ -3,6 +3,7 @@ set -eu LF_URL=${LANGFUSE_BASE_URL:-http://localhost:3100} LF_DOCKER_URL=${SMOKE_LANGFUSE_DOCKER_URL:-http://host.docker.internal:3100} +MCP_URL=${INGOT_MCP_URL:-http://localhost:8000/mcp} LF_PK=${LANGFUSE_PUBLIC_KEY:-pk-lf-local-demo} LF_SK=${LANGFUSE_SECRET_KEY:-sk-lf-local-demo} MARKER="ingot-claude-smoke-$(date +%s)" @@ -12,7 +13,7 @@ trap 'rm -rf "$WORKDIR"' EXIT INT TERM command -v claude >/dev/null 2>&1 || { echo "error: claude is required" >&2; exit 1; } curl -sSf "$LF_URL/api/public/health" >/dev/null -curl -sS -o /dev/null http://localhost:8000/mcp || [ "$?" -eq 52 ] +curl -sS -o /dev/null "$MCP_URL" || [ "$?" -eq 52 ] (cd "$WORKDIR" && CC_LANGFUSE_DEBUG=1 claude -p --output-format stream-json --verbose \ --permission-mode bypassPermissions --allowedTools=mcp__ingot__route_and_load \ @@ -65,7 +66,7 @@ docker run --rm --add-host host.docker.internal:host-gateway -v "$PWD:/app" -w / -e LANGFUSE_SECRET_KEY="$LF_SK" \ -e SMOKE_MARKER="$MARKER" ingot-mcp python -c ' import os -from optimize.mine import fetch_traces +from ingot.optimize.mine import fetch_traces marker = os.environ["SMOKE_MARKER"] for trace in fetch_traces(50): diff --git a/scripts/codex_langfuse_smoke.sh b/scripts/codex_langfuse_smoke.sh index 1e643a2..97aaf41 100755 --- a/scripts/codex_langfuse_smoke.sh +++ b/scripts/codex_langfuse_smoke.sh @@ -56,7 +56,7 @@ docker run --rm --add-host host.docker.internal:host-gateway -v "$PWD:/app" -w / -e LANGFUSE_SECRET_KEY="$LF_SK" \ -e SMOKE_MARKER="$MARKER" ingot-mcp python -c ' import os -from optimize.mine import fetch_traces +from ingot.optimize.mine import fetch_traces marker = os.environ["SMOKE_MARKER"] for trace in fetch_traces(50): diff --git a/scripts/codex_setup.sh b/scripts/codex_setup.sh index aff0a93..8d5dc11 100755 --- a/scripts/codex_setup.sh +++ b/scripts/codex_setup.sh @@ -36,9 +36,13 @@ require_command node require_command python3 CODEX_VERSION=$(codex --version | awk '{print $2}') -CODEX_MINOR=$(printf '%s' "$CODEX_VERSION" | awk -F. '{print $2}') -if [ "${CODEX_VERSION%%.*}" = "0" ] && [ "${CODEX_MINOR:-0}" -lt 128 ]; then - echo "error: Codex 0.128 or newer is required by the Langfuse plugin" >&2 +# The 0.128 floor stands for one thing: the `codex plugin` subcommand this script +# installs through. Probe for that directly. A locally built codex stamps no +# version — `codex-cli 0.0.0` parses as older than every release while carrying +# the capability, so a version comparison rejects a build that works. +if ! codex plugin --help >/dev/null 2>&1; then + echo "error: this codex build has no 'plugin' subcommand, which the Langfuse plugin needs" >&2 + echo " (released builds carry it from 0.128; found version $CODEX_VERSION)" >&2 exit 1 fi NODE_MAJOR=$(node -p 'process.versions.node.split(".")[0]') @@ -54,6 +58,12 @@ MARKETPLACE_OK=0 codex plugin marketplace list --json 2>/dev/null | grep -F 'codex-observability-plugin' >/dev/null && MARKETPLACE_OK=1 PLUGIN_OK=0 codex plugin list --json 2>/dev/null | grep -F 'tracing@codex-observability-plugin' >/dev/null && PLUGIN_OK=1 +# Codex runs a hook only after a human has reviewed it once and Codex has persisted +# the approval. Until then the Stop hook is skipped in silence: the plugin reports +# installed and enabled, and no trace is ever written. +TRUST_OK=0 +grep -q 'tracing@codex-observability-plugin:hooks/hooks.json:stop' \ + "${CODEX_HOME:-$HOME/.codex}/config.toml" 2>/dev/null && TRUST_OK=1 if [ "$MODE" = "doctor" ]; then echo "Codex version: $CODEX_VERSION" @@ -62,13 +72,29 @@ if [ "$MODE" = "doctor" ]; then [ "$MARKETPLACE_OK" = "1" ] && echo "Langfuse marketplace: installed" || echo "Langfuse marketplace: missing" [ "$PLUGIN_OK" = "1" ] && echo "Langfuse plugin: installed" || echo "Langfuse plugin: missing" [ -f "$LANGFUSE_CONFIG" ] && echo "Langfuse config: present at $LANGFUSE_CONFIG" || echo "Langfuse config: missing" - if command -v curl >/dev/null 2>&1 && curl -sSf "$LF_URL/api/public/health" >/dev/null 2>&1; then + [ "$TRUST_OK" = "1" ] && echo "Stop hook: trusted" || echo "Stop hook: not yet trusted — run codex interactively once and approve it" + # Probe with Node rather than curl. Node is what uploads the traces, and it reads a + # CA store of its own: against a private CA, curl can report a healthy endpoint that + # Node cannot reach. Whichever curl happens to sit first on PATH answers for a third + # trust store, so it speaks for nothing here. + if node -e ' +const url = process.argv[1] + "/api/public/health"; +const lib = url.startsWith("https:") ? require("https") : require("http"); +const req = lib.get(url, { timeout: 10000 }, (res) => { + res.resume(); + process.exit(res.statusCode === 200 ? 0 : 1); +}); +req.on("timeout", () => { req.destroy(); process.exit(1); }); +req.on("error", () => process.exit(1)); +' "$LF_URL" 2>/dev/null; then echo "Langfuse endpoint: healthy at $LF_URL" else - echo "Langfuse endpoint: unreachable at $LF_URL" + echo "Langfuse endpoint: unreachable from Node at $LF_URL" + echo " (if curl reaches it, Node is missing the CA — set NODE_EXTRA_CA_CERTS to the root)" fi trap - 0 - [ "$MCP_OK" = "1" ] && [ "$MARKETPLACE_OK" = "1" ] && [ "$PLUGIN_OK" = "1" ] && [ -f "$LANGFUSE_CONFIG" ] + [ "$MCP_OK" = "1" ] && [ "$MARKETPLACE_OK" = "1" ] && [ "$PLUGIN_OK" = "1" ] \ + && [ -f "$LANGFUSE_CONFIG" ] && [ "$TRUST_OK" = "1" ] exit fi diff --git a/scripts/fetch_skills.sh b/scripts/fetch_skills.sh index c4100b7..7a7bd14 100755 --- a/scripts/fetch_skills.sh +++ b/scripts/fetch_skills.sh @@ -1,7 +1,13 @@ #!/usr/bin/env bash -# Fetch example Agent Skills (SKILL.md format) into ./skills/. Every source is OPTIONAL and nothing -# is redistributed in this repo, each source is cloned from upstream, its skills copied in, and the -# clone deleted, so skills stay under their own upstream licenses. +# Fetch example Agent Skills (SKILL.md format) and QUARANTINE them for review. Every source is +# OPTIONAL and nothing is redistributed in this repo, each source is cloned from upstream, its +# skills submitted for review, and the clone deleted, so skills stay under their own upstream +# licenses. +# +# Nothing here is served. This used to copy third-party directories straight into ./skills, where +# the server picked them up on the next restart — an install, with no review and no record of what +# changed. Every package now goes through `ingot add`, which reviews it, records where it came +# from, and leaves the served library byte-identical until a human approves it. # # Usage: # scripts/fetch_skills.sh all # every source below @@ -15,24 +21,29 @@ # trailofbits trailofbits/skills security analysis CC-BY-SA-4.0 set -euo pipefail -SKILLS="$(cd "$(dirname "$0")/.." && pwd)/skills" -mkdir -p "$SKILLS" +if ! command -v ingot >/dev/null 2>&1; then + echo "fetch_skills.sh needs the \`ingot\` command (pip install -e .)." >&2 + echo "It quarantines each package for review instead of copying it into the served library." >&2 + exit 1 +fi -# copy up to $cap skill dirs (0 = no cap) from a freshly-cloned repo, skipping any that already exist +# quarantine up to $cap skill dirs (0 = no cap) from a freshly-cloned repo fetch() { # repo cap license - local repo="$1" cap="$2" license="$3" tmp added=0 + local repo="$1" cap="$2" license="$3" tmp added=0 refused=0 tmp="$(mktemp -d)" echo "[fetch] cloning $repo …" git clone --depth 1 -q "https://github.com/$repo" "$tmp/repo" while IFS= read -r skill_md; do - local dir name; dir="$(dirname "$skill_md")"; name="$(basename "$dir")" - [ -e "$SKILLS/$name" ] && continue # never clobber + local dir; dir="$(dirname "$skill_md")" [ "$cap" -ne 0 ] && [ "$added" -ge "$cap" ] && break # respect the cap - cp -r "$dir" "$SKILLS/$name" # whole dir: SKILL.md + bundled files - added=$((added + 1)) + if ingot add "file:$dir" >/dev/null; then # whole dir: SKILL.md + bundled files + added=$((added + 1)) + else + refused=$((refused + 1)) # already present, or refused on review + fi done < <(find "$tmp/repo" -name SKILL.md | sort) rm -rf "$tmp" # remove the clone - echo "[fetch] $repo: added $added skills (license: $license)" + echo "[fetch] $repo: quarantined $added, refused or skipped $refused (license: $license)" } # source lookup as a case statement (not `declare -A`): macOS ships bash 3.2, which has no @@ -59,5 +70,6 @@ for t in "${targets[@]}"; do fetch $spec done -echo "[fetch] $(find "$SKILLS" -name SKILL.md | wc -l | tr -d ' ') skills now in $SKILLS" -echo "[fetch] restart the server to pick them up: docker compose restart mcp" +echo "[fetch] nothing is served yet. Review what arrived and approve what you want:" +echo "[fetch] ingot list # unchanged until an approval publishes" +echo "[fetch] open http://localhost:8080 # the change-control console" diff --git a/scripts/managed_smoke.sh b/scripts/managed_smoke.sh new file mode 100755 index 0000000..fc2f2e4 --- /dev/null +++ b/scripts/managed_smoke.sh @@ -0,0 +1,82 @@ +#!/bin/sh +# Prove the managed stack's one-writer invariant against real containers. +# +# scripts/managed_smoke.sh +# +# The static checks (tests/test_compose_managed.py, `docker compose config`) verify what the +# tracked configuration declares. This verifies what Docker actually enforces, which is the only +# thing that makes the claim true: every non-publisher service must fail to write the served +# library, and the publisher must succeed. +# +# Run on dellpromax 2026-08-19 (Docker 29.7.2), passing, with the negative control failing as it +# must. The Mac it was written on has no daemon; run it anywhere that does before repeating the +# control-plane claim. +set -eu + +PROJECT=${MANAGED_SMOKE_PROJECT:-ingot-managed-smoke} +COMPOSE="docker compose -p $PROJECT" + +# Every check below reaches its container through `docker compose exec`, so this stack needs no +# host port. Asking for one anyway makes the run fail on any machine that already serves something +# on 8000 or 8080 -- a shared Docker host, or a box already running Ingot. `0` takes whatever the +# kernel has free. +INGOT_MCP_PORT=${INGOT_MCP_PORT:-0} +INGOT_UI_PORT=${INGOT_UI_PORT:-0} +export INGOT_MCP_PORT INGOT_UI_PORT + +cleanup() { + if [ "${MANAGED_SMOKE_KEEP:-0}" != "1" ]; then + $COMPOSE down -v --remove-orphans + fi +} +trap cleanup EXIT INT TERM + +# `ui` pulls in `mcp`; the publisher owns the vault. Langfuse is not part of this invariant. +$COMPOSE up -d --build publisher mcp ui + +# The publisher initializes the vault on start; wait for the repository rather than racing it. +elapsed=0 +until $COMPOSE exec -T publisher git -C /app/vault rev-parse HEAD >/dev/null 2>&1; do + elapsed=$((elapsed + 2)) + [ "$elapsed" -ge 60 ] && { echo "error: the publisher never initialized /app/vault" >&2; exit 1; } + sleep 2 +done + +failed=0 + +for service in mcp ui; do + if $COMPOSE exec -T "$service" sh -c 'touch /app/skills/.smoke-write' 2>/dev/null; then + echo "FAIL: $service can write the served library; it is not read-only" >&2 + failed=1 + else + echo "ok: $service cannot write /app/skills" + fi +done + +if $COMPOSE exec -T publisher sh -c 'touch /app/vault/.smoke-write && rm /app/vault/.smoke-write'; then + echo "ok: the publisher can write /app/vault" +else + echo "FAIL: the publisher cannot write its own vault" >&2 + failed=1 +fi + +# The served library the other services read is the vault the publisher writes. A stack where they +# are different directories serves stale bytes and reports no drift, because nothing compares them. +vault_head=$($COMPOSE exec -T publisher git -C /app/vault rev-parse HEAD) +served_head=$($COMPOSE exec -T mcp sh -c 'git -C /app/skills rev-parse HEAD' 2>/dev/null || echo "") +if [ "$vault_head" != "$served_head" ]; then + echo "FAIL: mcp serves $served_head but the publisher's vault is at $vault_head" >&2 + failed=1 +else + echo "ok: mcp serves the publisher's vault at $vault_head" +fi + +if $COMPOSE exec -T mcp ingot status; then + echo "ok: ingot status reports MANAGED inside the stack" +else + echo "FAIL: ingot status did not report MANAGED inside the stack" >&2 + failed=1 +fi + +[ "$failed" -eq 0 ] || { echo "managed smoke FAILED" >&2; exit 1; } +echo "managed smoke passed: one writer, and it is the publisher" diff --git a/tests/conftest.py b/tests/conftest.py index 2bce5cd..5abe890 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -11,8 +11,23 @@ @pytest.fixture(autouse=True) -def _isolated_local_skills_root(tmp_path_factory, monkeypatch): - monkeypatch.setattr("mcp_server.registry.SKILLS_DIR", tmp_path_factory.mktemp("local-skills")) +def _isolated_state(tmp_path_factory, monkeypatch): + """No test may touch the developer's real state directory, or the checkout's. + + State resolves through `ingot.paths` at call time and defaults to an XDG directory, so a test + that reaches a default instead of a fixture would write a real review queue and real receipts + into `~/.local/state/ingot` and pass while doing it. One per-test INGOT_HOME contains every one + of them; specific overrides are cleared so an environment variable the developer happens to + have exported cannot reach in either. + + This also replaces the old SKILLS_DIR patch. `configured_roots` always puts the local authoring + root first, even ahead of an explicit root, so a test that loads skills would otherwise see + whatever `scripts/fetch_skills.sh` left in the checkout (first caught on a machine with 72 + fetched skills: 9 failures and a multi-minute embedding stall).""" + from ingot import paths + monkeypatch.setenv(paths.HOME, str(tmp_path_factory.mktemp("state"))) + for name in (paths.LIBRARY, paths.RUNS, paths.TASKS, paths.VAULT, *paths.LEGACY.values()): + monkeypatch.delenv(name, raising=False) monkeypatch.delenv("SKILL_ROUTER_PATHS", raising=False) diff --git a/tests/fixtures/agy/judge-stream.jsonl b/tests/fixtures/agy/judge-stream.jsonl new file mode 100644 index 0000000..60ce466 --- /dev/null +++ b/tests/fixtures/agy/judge-stream.jsonl @@ -0,0 +1,9 @@ +{"event":"init","conversation_id":"","init":{"model":"gemini-3.6-flash-medium","cwd":"","tools":["ask_permission","ask_question","browser_click_element","browser_drag_pixel_to_pixel","browser_get_dom","browser_get_network_request","browser_input","browser_list_network_requests","browser_mouse_down","browser_mouse_up","browser_move_mouse","browser_press_key","browser_refresh_page","browser_resize_window","browser_scroll","browser_scroll_dom","browser_select_option","browser_subagent","call_mcp_tool","capture_browser_console_logs","capture_browser_screenshot","click_browser_pixel","command_status","define_subagent","delete_knowledge","execute_browser_javascript","find_by_name","finish","generate_image","grep_search","invoke_subagent","list_browser_pages","list_permissions","list_resources","manage_inbox","manage_subagents","manage_task","multi_replace_file_content","notebook_edit","notebook_execution","open_browser_url","read_browser_page","read_resource","read_url_content","replace_file_content","run_command","schedule","search_web","sed_file","send_command_input","send_message","view_file","wait","wait_5_seconds","write_to_file"],"permission_mode":"request-review","json_schema":{"type":"object","additionalProperties":false,"required":["items","feedback"],"properties":{"items":{"type":"object","additionalProperties":false,"required":["probe"],"properties":{"probe":{"type":"object","additionalProperties":false,"required":["verdict","note"],"properties":{"verdict":{"enum":["pass","partial","fail"]},"note":{"type":"string"}}}}},"feedback":{"type":"string"}}}}} +{"event":"step_update","step_update":{"conversation_id":"","step_index":0,"state":"DONE","step_type":"user_input"}} +{"event":"step_update","step_update":{"conversation_id":"","step_index":1,"state":"DONE","step_type":"unknown","duration_seconds":0.000501423}} +{"event":"step_update","step_update":{"conversation_id":"","step_index":2,"state":"DONE","step_type":"agent_response","text_delta":"{\"feedback\":\"ok\",\"items\":[{\"probe\":{\"note\":\"\",\"verdict\":\"pass\"}}]}\n","duration_seconds":2.94894296,"usage":{"input_tokens":18403,"output_tokens":1374,"thinking_tokens":1342,"cache_read_tokens":0,"total_tokens":19777}}} +{"event":"step_update","step_update":{"conversation_id":"","step_index":3,"state":"DONE","step_type":"error_message"}} +{"event":"step_update","step_update":{"conversation_id":"","step_index":4,"state":"DONE","step_type":"agent_response","text_delta":"{\"feedback\":\"ok\",\"items\":{\"probe\":{\"note\":\"\",\"verdict\":\"pass\"}}}\n","duration_seconds":1.035306864,"usage":{"input_tokens":3680,"output_tokens":464,"thinking_tokens":434,"cache_read_tokens":16287,"total_tokens":4144}}} +{"event":"step_update","step_update":{"conversation_id":"","step_index":5,"state":"DONE","step_type":"finish","duration_seconds":0.055575545}} +{"event":"step_update","step_update":{"conversation_id":"","step_index":6,"state":"DONE","step_type":"checkpoint","duration_seconds":0.565171204,"usage":{"input_tokens":142,"output_tokens":4,"thinking_tokens":0,"cache_read_tokens":0,"total_tokens":146}}} +{"event":"result","result":{"conversation_id":"","status":"SUCCESS","response":"{\"feedback\":\"ok\",\"items\":[{\"probe\":{\"note\":\"\",\"verdict\":\"pass\"}}]}\n{\"feedback\":\"ok\",\"items\":{\"probe\":{\"note\":\"\",\"verdict\":\"pass\"}}}\n","duration_seconds":4.58409492,"num_turns":1,"structured_output":{"feedback":"ok","items":{"probe":{"note":"","verdict":"pass"}}},"json_schema":{"type":"object","additionalProperties":false,"required":["items","feedback"],"properties":{"items":{"type":"object","additionalProperties":false,"required":["probe"],"properties":{"probe":{"type":"object","additionalProperties":false,"required":["verdict","note"],"properties":{"verdict":{"enum":["pass","partial","fail"]},"note":{"type":"string"}}}}},"feedback":{"type":"string"}}},"usage":{"input_tokens":22225,"output_tokens":1842,"thinking_tokens":1776,"cache_read_tokens":16287,"total_tokens":24067}}} diff --git a/tests/fixtures/harbor/deepseek-models.json b/tests/fixtures/harbor/deepseek-models.json new file mode 100644 index 0000000..b14f102 --- /dev/null +++ b/tests/fixtures/harbor/deepseek-models.json @@ -0,0 +1,28 @@ +{ + "object": "list", + "data": [ + { + "id": "deepseek-v4-flash", + "object": "model", + "owned_by": "vllm", + "root": "deepseek-ai/DeepSeek-V4-Flash-0731", + "parent": null, + "max_model_len": 1048576, + "permission": [ + { + "id": "modelperm-a8f876e158347e2a", + "object": "model_permission", + "allow_create_engine": false, + "allow_sampling": true, + "allow_logprobs": true, + "allow_search_indices": false, + "allow_view": true, + "allow_fine_tuning": false, + "organization": "*", + "group": null, + "is_blocking": false + } + ] + } + ] +} diff --git a/tests/fixtures/harbor/langfuse-trial/agent/trajectory.json b/tests/fixtures/harbor/langfuse-trial/agent/trajectory.json new file mode 100644 index 0000000..73ef6b1 --- /dev/null +++ b/tests/fixtures/harbor/langfuse-trial/agent/trajectory.json @@ -0,0 +1,33 @@ +{ + "schema_version": "ATIF-v1.1", + "session_id": "fixture-session", + "agent": { + "name": "codex", + "version": "fixture-version", + "model_name": "fixture-model", + "extra": { + "originator": "fixture", + "cwd": "/fixtures/workspace" + } + }, + "steps": [ + { + "step_id": 1, + "timestamp": "2000-01-01T00:00:00Z", + "source": "user", + "message": "Create the deterministic fixture deliverable." + }, + { + "step_id": 2, + "timestamp": "2000-01-01T00:00:00Z", + "source": "agent", + "message": "I will inspect the fixture workspace." + }, + { + "step_id": 3, + "timestamp": "2000-01-01T00:00:01Z", + "source": "agent", + "message": "The fixture deliverable is ready." + } + ] +} diff --git a/tests/fixtures/harbor/langfuse-trial/exception.txt b/tests/fixtures/harbor/langfuse-trial/exception.txt new file mode 100644 index 0000000..4c98819 --- /dev/null +++ b/tests/fixtures/harbor/langfuse-trial/exception.txt @@ -0,0 +1,4 @@ +Traceback (most recent call last): + File "/fixtures/harbor/runner.py", line 1, in run + raise AgentSetupTimeoutError("sanitized fixture failure") +AgentSetupTimeoutError: sanitized fixture failure diff --git a/tests/fixtures/harbor/langfuse-trial/result.json b/tests/fixtures/harbor/langfuse-trial/result.json new file mode 100644 index 0000000..3a1e3f6 --- /dev/null +++ b/tests/fixtures/harbor/langfuse-trial/result.json @@ -0,0 +1,124 @@ +{ + "id": "00000000-0000-0000-0000-000000000001", + "task_name": "ingot/fixture-h0", + "trial_name": "fixture-h0__attempt-1", + "trial_uri": "file:///fixtures/harbor/langfuse-trial", + "task_id": { + "path": "/fixtures/tasks/fixture-h0" + }, + "source": "fixture", + "task_checksum": "fixture-task-checksum", + "config": { + "task": { + "path": "/fixtures/tasks/fixture-h0", + "git_url": null, + "git_commit_id": null, + "name": null, + "ref": null, + "overwrite": false, + "download_dir": null, + "source": "fixture" + }, + "trial_name": "fixture-h0__attempt-1", + "trials_dir": "/fixtures/trials", + "install_only": false, + "timeout_multiplier": 1.0, + "agent_timeout_multiplier": null, + "verifier_timeout_multiplier": null, + "agent_setup_timeout_multiplier": 3.0, + "environment_build_timeout_multiplier": 4.0, + "agent": { + "name": "codex", + "import_path": null, + "model_name": "fixture-model", + "n_concurrent": null, + "concurrency_group": null, + "skills": [ + "/fixtures/skills/private-skill/SKILL.md" + ], + "override_timeout_sec": null, + "override_setup_timeout_sec": null, + "max_timeout_sec": null, + "resume_trajectory": false, + "load_trajectory": null, + "extra_allowed_hosts": [], + "kwargs": {}, + "env": { + "OPENAI_API_KEY": "fixture-secret-must-not-export", + "OPENAI_BASE_URL": "https://fixture-endpoint.invalid/v1" + }, + "mcp_servers": [] + }, + "environment": { + "type": "docker", + "import_path": null, + "force_build": false, + "delete": true, + "cpu_enforcement_policy": "none", + "memory_enforcement_policy": "none", + "override_cpus": null, + "override_memory_mb": null, + "override_storage_mb": null, + "override_gpus": null, + "override_tpu": null, + "mounts": null, + "extra_docker_compose": [], + "kwargs": {}, + "extra_allowed_hosts": [] + }, + "verifier": { + "override_timeout_sec": null, + "max_timeout_sec": null, + "disable": false + }, + "artifacts": [], + "extra_instruction_paths": [], + "job_id": "fixture-job" + }, + "agent_info": { + "name": "codex", + "version": "fixture-version", + "model_info": { + "name": "fixture-model", + "provider": null + } + }, + "agent_result": { + "n_input_tokens": 120, + "n_cache_tokens": 40, + "n_output_tokens": 30, + "cost_usd": 0.01, + "rollout_details": null, + "metadata": null + }, + "verifier_result": { + "rewards": { + "reward": 1.0 + } + }, + "exception_info": { + "exception_type": "AgentSetupTimeoutError", + "exception_message": "sanitized fixture failure", + "exception_traceback": "sanitized fixture traceback", + "occurred_at": "2000-01-01T00:00:00Z" + }, + "started_at": "2000-01-01T00:00:00Z", + "finished_at": "2000-01-01T00:00:01Z", + "environment_setup": { + "started_at": "2000-01-01T00:00:00Z", + "finished_at": "2000-01-01T00:00:00Z" + }, + "agent_setup": { + "started_at": "2000-01-01T00:00:00Z", + "finished_at": "2000-01-01T00:00:01Z" + }, + "agent_execution": { + "started_at": null, + "finished_at": null + }, + "verifier": { + "started_at": "2000-01-01T00:00:01Z", + "finished_at": "2000-01-01T00:00:01Z" + }, + "step_results": null +} diff --git a/tests/fixtures/harbor/langfuse-trial/verifier/solution/_objective_check.txt b/tests/fixtures/harbor/langfuse-trial/verifier/solution/_objective_check.txt new file mode 100644 index 0000000..71f7e19 --- /dev/null +++ b/tests/fixtures/harbor/langfuse-trial/verifier/solution/_objective_check.txt @@ -0,0 +1,7 @@ +Fixture objective check + +- deterministic output exists +- no external endpoint was contacted +- no credential was used + +PASS diff --git a/tests/fixtures/harbor/langfuse-trial/verifier/test-stdout.txt b/tests/fixtures/harbor/langfuse-trial/verifier/test-stdout.txt new file mode 100644 index 0000000..e69de29 diff --git a/tests/fixtures/harbor/orin-models.json b/tests/fixtures/harbor/orin-models.json new file mode 100644 index 0000000..7f96b26 --- /dev/null +++ b/tests/fixtures/harbor/orin-models.json @@ -0,0 +1,17 @@ +{ + "object": "list", + "data": [ + { + "id": "ablit35b", + "aliases": ["ablit35b"], + "object": "model", + "owned_by": "llamacpp", + "meta": { + "n_ctx": 65536, + "n_ctx_train": 262144, + "n_params": 34660610688, + "ftype": "Q4_K - Medium" + } + } + ] +} diff --git a/tests/fixtures/harbor/qwen-models.json b/tests/fixtures/harbor/qwen-models.json new file mode 100644 index 0000000..0ff0fef --- /dev/null +++ b/tests/fixtures/harbor/qwen-models.json @@ -0,0 +1,28 @@ +{ + "object": "list", + "data": [ + { + "id": "dot-backbone", + "object": "model", + "owned_by": "vllm", + "root": "Qwen/Qwen3.6-27B-FP8", + "parent": null, + "max_model_len": 163840, + "permission": [ + { + "id": "modelperm-bf3337a2a0df3e11", + "object": "model_permission", + "allow_create_engine": false, + "allow_sampling": true, + "allow_logprobs": true, + "allow_search_indices": false, + "allow_view": true, + "allow_fine_tuning": false, + "organization": "*", + "group": null, + "is_blocking": false + } + ] + } + ] +} diff --git a/tests/fixtures/harbor/qwen-size-matrix.json b/tests/fixtures/harbor/qwen-size-matrix.json new file mode 100644 index 0000000..2120f8f --- /dev/null +++ b/tests/fixtures/harbor/qwen-size-matrix.json @@ -0,0 +1,15 @@ +{ + "judge": "agy/gemini-3.6-flash-medium", + "exploratory": true, + "rankable": false, + "harnesses": { + "aider@qwen35-0.8b": {"harness":"aider","model":"qwen35-0.8b","family":"Qwen3.5","parameter_billions":0.8,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":-0.04,"skill_mean":0.42,"control_mean":0.46,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false}, + "pi@qwen35-0.8b": {"harness":"pi","model":"qwen35-0.8b","family":"Qwen3.5","parameter_billions":0.8,"quantization":"fp8-load","tool_parser":"qwen3_coder","error":"canary failed","exploratory":true,"rankable":false}, + "terminus-2@qwen35-2b": {"harness":"terminus-2","model":"qwen35-2b","family":"Qwen3.5","parameter_billions":2.0,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":0.03,"skill_mean":0.51,"control_mean":0.48,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false}, + "aider@qwen35-4b": {"harness":"aider","model":"qwen35-4b","family":"Qwen3.5","parameter_billions":4.0,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":0.08,"skill_mean":0.57,"control_mean":0.49,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false}, + "pi@qwen35-4b": {"harness":"pi","model":"qwen35-4b","family":"Qwen3.5","parameter_billions":4.0,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":0.05,"skill_mean":0.55,"control_mean":0.50,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false}, + "terminus-2@qwen35-9b": {"harness":"terminus-2","model":"qwen35-9b","family":"Qwen3.5","parameter_billions":9.0,"quantization":"fp8-load","tool_parser":"qwen3_coder","lift":0.12,"skill_mean":0.63,"control_mean":0.51,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false}, + "aider@dot-backbone": {"harness":"aider","model":"dot-backbone","family":"Qwen3.6","parameter_billions":27.0,"quantization":"fp8-published","tool_parser":"qwen3_xml","lift":0.16,"skill_mean":0.68,"control_mean":0.52,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false}, + "pi@dot-backbone": {"harness":"pi","model":"dot-backbone","family":"Qwen3.6","parameter_billions":27.0,"quantization":"fp8-published","tool_parser":"qwen3_xml","lift":0.14,"skill_mean":0.66,"control_mean":0.52,"tasks_scored":4,"attempts":1,"exploratory":true,"rankable":false} + } +} diff --git a/tests/test_acceptance.py b/tests/test_acceptance.py index 85d0843..4ff49ae 100644 --- a/tests/test_acceptance.py +++ b/tests/test_acceptance.py @@ -3,7 +3,7 @@ import pytest -from optimize.acceptance import classify, evaluate, load_criteria +from ingot.optimize.acceptance import classify, evaluate, load_criteria def test_load_criteria_parses_forbid_and_skips_empty(tmp_path): diff --git a/tests/test_acquire.py b/tests/test_acquire.py new file mode 100644 index 0000000..f1b3eaa --- /dev/null +++ b/tests/test_acquire.py @@ -0,0 +1,181 @@ +"""Offline Git fixtures for public GitHub acquisition. + +The tests replace only the OWNER/REPO-to-URL mapping. Resolution, clone, tree inspection, and raw +blob reads all run through real Git against a local repository. +""" +import subprocess +from pathlib import Path + +import pytest + +from ingot import acquire + + +def _git(path: Path, *args: str) -> str: + result = subprocess.run(["git", "-C", str(path), *args], check=True, + capture_output=True, text=True) + return result.stdout.strip() + + +def _remote(tmp_path: Path, *, asset: bytes = b"first bytes") -> tuple[Path, str]: + remote = tmp_path / "remote" + subprocess.run(["git", "init", "-b", "main", str(remote)], check=True, + capture_output=True) + _git(remote, "config", "user.email", "fixture@example.test") + _git(remote, "config", "user.name", "Fixture") + skill = remote / "skills" / "pdf" + skill.mkdir(parents=True) + (skill / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\n\nUse this skill.\n", + encoding="utf-8") + (skill / "asset.bin").write_bytes(asset) + (remote / "outside.txt").write_text("not part of the package", encoding="utf-8") + _git(remote, "add", ".") + _git(remote, "commit", "-m", "Create fixture") + return remote, _git(remote, "rev-parse", "HEAD") + + +def test_github_acquisition_resolves_commit_and_checks_out_only_the_selected_package( + tmp_path, monkeypatch): + """Break caught: recording an unverified ref, or admitting the repository root.""" + remote, commit = _remote(tmp_path) + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + + package, provenance = acquire.github( + "acme/skills", ref="HEAD", subdirectory="skills/pdf", destination=tmp_path / "clone") + + assert package == tmp_path / "clone" / "package" + assert (package / "asset.bin").read_bytes() == b"first bytes" + assert not (package / "outside.txt").exists() + assert provenance == { + "repository": "acme/skills", + "ref": "HEAD", + "commit": commit, + "subdirectory": "skills/pdf", + } + + +@pytest.mark.parametrize("subdirectory", ["../pdf", "/skills/pdf", "skills\\pdf", "skills/\npdf"]) +def test_github_acquisition_refuses_a_subdirectory_that_can_escape_or_change_spelling( + tmp_path, subdirectory): + """Break caught: joining an unchecked remote-controlled path onto the clone root.""" + with pytest.raises(ValueError, match="path|root|portable"): + acquire.github("acme/skills", ref="HEAD", subdirectory=subdirectory, + destination=tmp_path / "clone") + + +@pytest.mark.parametrize("repository", ["acme", "acme/skills/extra", "../skills", "acme\\skills"]) +def test_github_acquisition_refuses_a_non_owner_repository_name(tmp_path, repository): + """Break caught: accepting a URL or option where the public OWNER/REPO locator is required.""" + with pytest.raises(ValueError, match="OWNER/REPO"): + acquire.github(repository, ref="HEAD", subdirectory="skills/pdf", + destination=tmp_path / "clone") + + +def test_github_acquisition_refuses_a_repository_with_a_git_suffix(tmp_path, monkeypatch): + """Break caught: accepting repo.git and then fetching the unintended repo.git.git URL.""" + monkeypatch.setattr( + acquire, "_resolved_commit", + lambda remote, ref: pytest.fail("repository validation must precede resolution")) + + with pytest.raises(ValueError, match="OWNER/REPO"): + acquire.github("acme/skills.git", ref="HEAD", subdirectory="skills/pdf", + destination=tmp_path / "clone") + + +def test_github_acquisition_refuses_a_symlink_recorded_by_git(tmp_path, monkeypatch): + """Break caught: core.symlinks=false flattening a Git symlink before tree.build can see it.""" + remote, _ = _remote(tmp_path) + (remote / "skills" / "pdf" / "link.md").symlink_to("SKILL.md") + _git(remote, "add", ".") + _git(remote, "commit", "-m", "Add a symlink") + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + + with pytest.raises(ValueError, match="symlink"): + acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf", + destination=tmp_path / "clone") + + +def test_github_acquisition_refuses_a_submodule_recorded_by_git(tmp_path, monkeypatch): + """Break caught: treating a gitlink as package bytes or recursively acquiring another repo.""" + child = tmp_path / "child" + subprocess.run(["git", "init", "-b", "main", str(child)], check=True, + capture_output=True) + _git(child, "config", "user.email", "fixture@example.test") + _git(child, "config", "user.name", "Fixture") + (child / "payload.txt").write_text("submodule bytes", encoding="utf-8") + _git(child, "add", ".") + _git(child, "commit", "-m", "Create child") + remote, _ = _remote(tmp_path) + subprocess.run( + ["git", "-c", "protocol.file.allow=always", "-C", str(remote), "submodule", "add", + child.as_uri(), "skills/pdf/vendor"], check=True, capture_output=True) + _git(remote, "commit", "-am", "Add a submodule") + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + + with pytest.raises(ValueError, match="regular file"): + acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf", + destination=tmp_path / "clone") + + +def test_github_acquisition_refuses_an_oversized_tree_before_checkout(tmp_path, monkeypatch): + """Break caught: downloading selected blobs before enforcing the admission byte budget.""" + remote, _ = _remote(tmp_path, asset=b"0123456789") + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + monkeypatch.setattr(acquire, "MAX_TREE_BYTES", 8) + + with pytest.raises(ValueError, match="at most 8 bytes"): + acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf", + destination=tmp_path / "clone") + + assert not (tmp_path / "clone" / "package" / "asset.bin").exists() + + +def test_github_acquisition_refuses_too_many_files_before_checkout(tmp_path, monkeypatch): + """Break caught: enforcing only byte size lets an empty-file tree exhaust the filesystem.""" + remote, _ = _remote(tmp_path) + (remote / "skills" / "pdf" / "extra.txt").touch() + _git(remote, "add", ".") + _git(remote, "commit", "-m", "Add another file") + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + monkeypatch.setattr(acquire, "MAX_FILES", 2) + + with pytest.raises(ValueError, match="at most 2 files"): + acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf", + destination=tmp_path / "clone") + + assert not (tmp_path / "clone" / "package").exists() + + +def test_github_acquisition_refuses_when_the_ref_moves_before_clone(tmp_path, monkeypatch): + """Break caught: recording the ls-remote commit while admitting a later checkout.""" + remote, old_commit = _remote(tmp_path) + (remote / "skills" / "pdf" / "asset.bin").write_bytes(b"new bytes") + _git(remote, "add", ".") + _git(remote, "commit", "-m", "Move HEAD") + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + monkeypatch.setattr(acquire, "_resolved_commit", lambda remote, ref: old_commit) + + with pytest.raises(ValueError, match="ref moved"): + acquire.github("acme/skills", ref="HEAD", subdirectory="skills/pdf", + destination=tmp_path / "clone") + + +def test_github_acquisition_does_not_run_checkout_filters(tmp_path, monkeypatch): + """Break caught: checkout executing a configured smudge command selected by .gitattributes.""" + remote, _ = _remote(tmp_path) + (remote / "skills" / "pdf" / ".gitattributes").write_text( + "*.bin filter=fixture\n", encoding="utf-8") + _git(remote, "add", ".") + _git(remote, "commit", "-m", "Select a checkout filter") + marker = tmp_path / "filter-ran" + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + monkeypatch.setenv("GIT_CONFIG_COUNT", "1") + monkeypatch.setenv("GIT_CONFIG_KEY_0", "filter.fixture.smudge") + monkeypatch.setenv("GIT_CONFIG_VALUE_0", f"touch {marker}; cat") + + package, _ = acquire.github( + "acme/skills", ref="HEAD", subdirectory="skills/pdf", destination=tmp_path / "clone") + + assert (package / "asset.bin").read_bytes() == b"first bytes" + assert not marker.exists() diff --git a/tests/test_admission.py b/tests/test_admission.py new file mode 100644 index 0000000..12e9084 --- /dev/null +++ b/tests/test_admission.py @@ -0,0 +1,619 @@ +"""`ingot add file:` — one complete local ingest path. + +The invariant under test is the product's whole claim: a package can be submitted, reviewed, +quarantined and made visible to a reviewer without a single byte of the served library changing. + +Real filesystem throughout. The exclusivity test uses real processes, because `os.link` is a +cross-process guarantee and a thread-only test would not exercise it.""" +import hashlib +import json +import multiprocessing +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +from ingot import admission, cli, records +from ingot.mcp_server import registry +from ingot.mcp_server.registry import skill_revision +from ingot.optimize import ingress, tree +from ingot.optimize import promote as P + + +def _store(tmp_path, monkeypatch): + """The served library plus every store admission writes to, all under tmp_path.""" + root = tmp_path / "skills" + root.mkdir() + monkeypatch.setenv("INGOT_LIBRARY", str(root)) + monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs")) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root)) + return root + + +def _package(root, name="pdf", description="Merge and split PDF files.", + body="Use this to combine PDFs.", files=None): + directory = root / name + directory.mkdir(parents=True, exist_ok=True) + (directory / "SKILL.md").write_text( + f"---\nname: {name}\ndescription: {description}\n---\n\n{body}\n", encoding="utf-8") + for relative, content in (files or {}).items(): + target = directory / relative + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(content, encoding="utf-8") + return directory + + +def _tree_hash(root): + """Every path and its bytes under a root. The library must be identical after admission, and + 'identical' means content, not just the directory listing.""" + entries = {} + for path in sorted(root.rglob("*")): + entries[str(path.relative_to(root))] = path.read_bytes() if path.is_file() else None + return entries + + +def _github_remote(root, *, body="Use this skill."): + """A real Git remote; only URL construction is replaced in GitHub command tests.""" + subprocess.run(["git", "init", "-b", "main", str(root)], check=True, + capture_output=True) + subprocess.run(["git", "-C", str(root), "config", "user.email", "fixture@example.test"], + check=True) + subprocess.run(["git", "-C", str(root), "config", "user.name", "Fixture"], check=True) + _package(root / "packages", "pdf", body=body, files={"assets/data.bin": "bytes"}) + subprocess.run(["git", "-C", str(root), "add", "."], check=True) + subprocess.run(["git", "-C", str(root), "commit", "-m", "Create fixture"], check=True, + capture_output=True) + return subprocess.run(["git", "-C", str(root), "rev-parse", "HEAD"], check=True, + capture_output=True, text=True).stdout.strip() + + +# --- locators ------------------------------------------------------------------------------- + +def test_a_file_locator_resolves_to_its_directory(tmp_path): + package = _package(tmp_path / "src") + + kind, resolved = admission.parse_locator(f"file:{package}") + + assert kind == "file" + assert resolved == package + + +def test_a_bare_path_is_accepted_as_a_file_locator(tmp_path): + package = _package(tmp_path / "src") + + assert admission.parse_locator(str(package)) == ("file", package) + + +def test_a_github_locator_keeps_the_repository_name_unresolved(): + assert admission.parse_locator("github:acme/skills") == ("github", "acme/skills") + + +def test_an_unsupported_scheme_is_refused(): + with pytest.raises(ValueError, match="tessl"): + admission.parse_locator("tessl:acme/skills") + + +# --- the ingest path ------------------------------------------------------------------------ + +def test_admission_leaves_the_served_library_byte_identical(tmp_path, monkeypatch): + """The claim, tested directly.""" + library = _store(tmp_path, monkeypatch) + _package(library, "docx", description="Edit Word documents.") + package = _package(tmp_path / "src", "pdf") + before = _tree_hash(library) + + admission.add_package(package, actor="operator") + + assert _tree_hash(library) == before + + +def test_admission_writes_a_pending_proposal(tmp_path, monkeypatch): + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf") + + result = admission.add_package(package, actor="operator") + + pending = P.load_pending("pdf") + assert pending is not None + assert pending["kind"] == "creation" + assert result["proposal_id"] == pending["creation"]["proposal_id"] + + +def test_the_pending_revision_matches_a_canonical_package_on_disk(tmp_path, monkeypatch): + """For a package that is already canonical the two revisions agree, bundled files included.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", files={"references/flags.md": "# Flags\n"}) + + admission.add_package(package, actor="operator") + + pending = P.load_pending("pdf") + assert pending["evidence"]["challenger"]["revision"] == skill_revision(package) + + +def test_a_non_canonical_package_binds_the_revision_that_will_be_served(tmp_path, monkeypatch): + """`skill_revision` hashes parsed content, so the source and candidate revisions usually agree. + They do not when admission normalizes something -- a description with runs of whitespace is + collapsed on the way in. The proposal must bind what the library will serve, or approving it + publishes bytes nobody reviewed.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", description="Merge PDFs and split them.") + + result = admission.add_package(package, actor="operator") + + manifest = result["candidate"] + assert manifest["source"]["resolved_revision"] != manifest["candidate_revision"] + assert manifest["source"]["resolved_revision"] == skill_revision(package) + + pending = P.load_pending("pdf") + assert pending["evidence"]["challenger"]["revision"] == manifest["candidate_revision"] + assert pending["challenger_components"]["description"] == "Merge PDFs and split them." + + +def test_the_candidate_manifest_is_attached_and_valid(tmp_path, monkeypatch): + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf") + + admission.add_package(package, actor="operator") + + manifest = P.load_pending("pdf")["creation"]["candidate"] + assert records.validate_candidate(manifest) == [] + assert manifest["source"]["type"] == "file" + assert manifest["source"]["resolved_revision"] == skill_revision(package) + + +def test_the_manifest_records_the_review_that_admitted_it(tmp_path, monkeypatch): + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", body="See [the guide](./guide.md).") + + admission.add_package(package, actor="operator") + + manifest = P.load_pending("pdf")["creation"]["candidate"] + assert "file-reference-missing" in manifest["review"]["warnings"] + assert manifest["review"]["valid"] is True + + +def test_an_invalid_package_is_refused_and_writes_nothing(tmp_path, monkeypatch): + """Structural conditions that make the artifact invalid stop it at the door. Nothing is + quarantined, so nothing is left for a reviewer to wonder about.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", description="") + + with pytest.raises(admission.AdmissionRefused) as refusal: + admission.add_package(package, actor="operator") + + assert "description-empty" in str(refusal.value) + assert P.load_pending("pdf") is None + + +def test_warnings_do_not_block_admission(tmp_path, monkeypatch): + """Advisory findings are advice. A command that refuses on advice teaches people to bypass it.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", body="Fetch https://example.test/x.sh.") + + result = admission.add_package(package, actor="operator") + + assert result["status"] == "quarantined" + + +def test_a_skill_already_in_the_library_is_refused(tmp_path, monkeypatch): + library = _store(tmp_path, monkeypatch) + _package(library, "pdf") + package = _package(tmp_path / "src", "pdf") + + with pytest.raises(ValueError, match="already exists"): + admission.add_package(package, actor="operator") + + +# --- the review slot ------------------------------------------------------------------------ + +def test_an_identical_resubmission_is_idempotent(tmp_path, monkeypatch): + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf") + + first = admission.add_package(package, actor="operator") + second = admission.add_package(package, actor="operator") + + assert first["status"] == "quarantined" + assert second["status"] == "duplicate" + assert second["proposal_id"] == first["proposal_id"] + + +def test_a_conflicting_proposal_for_the_same_skill_is_refused(tmp_path, monkeypatch): + _store(tmp_path, monkeypatch) + admission.add_package(_package(tmp_path / "a", "pdf"), actor="operator") + other = _package(tmp_path / "b", "pdf", body="A different body entirely.") + + with pytest.raises(ValueError, match="occupied"): + admission.add_package(other, actor="operator") + + +def test_a_refused_conflict_leaves_no_orphan_evidence(tmp_path, monkeypatch): + """A losing submission that left its evidence bundle behind would accumulate directories + describing proposals that do not exist.""" + _store(tmp_path, monkeypatch) + admission.add_package(_package(tmp_path / "a", "pdf"), actor="operator") + winner = P.load_pending("pdf")["creation"]["proposal_id"] + other = _package(tmp_path / "b", "pdf", body="A different body entirely.") + + with pytest.raises(ValueError): + admission.add_package(other, actor="operator") + + bundles = sorted(p.name for p in (tmp_path / "runs" / "evidence" / "pdf").iterdir()) + assert bundles == [f"creation-{winner}"] + + +def test_the_evidence_bundle_is_written(tmp_path, monkeypatch): + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf") + + admission.add_package(package, actor="operator") + + proposal_id = P.load_pending("pdf")["creation"]["proposal_id"] + bundle = tmp_path / "runs" / "evidence" / "pdf" / f"creation-{proposal_id}" / "evidence.json" + assert json.loads(bundle.read_text())["skill"] == "pdf" + + +# --- cross-process exclusivity ---------------------------------------------------------------- + +def _submit_in_child(package_dir, library, runs, queue): + """Module scope and explicit configuration on purpose: under macOS `spawn` a nested function is + unpicklable and a parent's monkeypatch never reaches the child, so the child is told where its + state lives through the environment -- the same mechanism a real deployment uses. Setting it + before the import is what makes it stick, and getting it wrong would point both children at the + developer's own state directory while the test still passed.""" + import pathlib + + os.environ["INGOT_LIBRARY"] = str(library) + os.environ["INGOT_RUNS"] = str(runs) + os.environ["SKILL_ROUTER_PATHS"] = str(library) + + from ingot import admission as child_admission + try: + queue.put(("ok", child_admission.add_package(pathlib.Path(package_dir), + actor="operator")["status"])) + except Exception as error: # noqa: BLE001 - the refusal is the result under test + queue.put(("refused", str(error))) + + +def test_two_processes_cannot_both_claim_the_review_slot(tmp_path, monkeypatch): + """`_publish` creates the pending record with `os.link`, which is atomic across processes. + A threading lock alone would not survive the publisher and the console running separately.""" + _store(tmp_path, monkeypatch) + first = _package(tmp_path / "a", "pdf", body="First body.") + second = _package(tmp_path / "b", "pdf", body="Second body, different bytes.") + + context = multiprocessing.get_context("spawn") + queue = context.Queue() + stores = (tmp_path / "skills", tmp_path / "runs") + workers = [context.Process(target=_submit_in_child, + args=(str(package), *[str(s) for s in stores], queue)) + for package in (first, second)] + for worker in workers: + worker.start() + for worker in workers: + worker.join(timeout=120) + + outcomes = [queue.get(timeout=10) for _ in workers] + assert sorted(kind for kind, _ in outcomes) == ["ok", "refused"] + assert P.load_pending("pdf") is not None + + +# --- the command ---------------------------------------------------------------------------- + +def test_command_quarantines_and_names_the_next_action(tmp_path, monkeypatch, capsys): + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf") + + code = cli.main(["add", f"file:{package}"]) + + output = capsys.readouterr().out + assert code == 0 + assert P.load_pending("pdf") is not None + assert "pdf" in output + assert "approve" in output.lower() or "review" in output.lower() + + +def test_command_reports_a_refusal_without_a_traceback(tmp_path, monkeypatch, capsys): + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", description="") + + code = cli.main(["add", f"file:{package}"]) + + assert code != 0 + assert "description-empty" in capsys.readouterr().err + + +def test_command_json_reports_the_proposal(tmp_path, monkeypatch, capsys): + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf") + + code = cli.main(["add", f"file:{package}", "--json"]) + + payload = json.loads(capsys.readouterr().out) + assert code == 0 + assert payload["status"] == "quarantined" + assert records.validate_candidate(payload["candidate"]) == [] + + +def test_file_command_refuses_a_github_skill_subdirectory(tmp_path, monkeypatch, capsys): + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf") + + code = cli.main(["add", f"file:{package}", "--skill", "skills/pdf"]) + + assert code == 1 + assert "--skill is only valid" in capsys.readouterr().err + + +def test_github_command_requires_a_skill_subdirectory(capsys): + code = cli.main(["add", "github:acme/skills"]) + + assert code == 1 + assert "--skill is required" in capsys.readouterr().err + + +def test_github_command_records_exact_provenance_and_leaves_active_targets_unchanged( + tmp_path, monkeypatch, capsys): + """Break caught: Git metadata lost at the shared admission seam, or acquisition delivering.""" + from ingot import acquire + + library = _store(tmp_path, monkeypatch) + native = tmp_path / "native" + _package(library, "docx", description="Edit Word documents.") + _package(native, "existing", description="Existing native skill.") + monkeypatch.setenv("INGOT_DELIVERY_TARGETS", f"native=filesystem:{native}") + before = (_tree_hash(library), _tree_hash(native)) + remote = tmp_path / "remote" + commit = _github_remote(remote) + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + + code = cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"]) + + payload = json.loads(capsys.readouterr().out) + source = payload["candidate"]["source"] + assert code == 0 + assert source["type"] == "github" + assert source["repository"] == "acme/skills" + assert source["ref"] == "HEAD" + assert source["commit"] == commit + assert source["subdirectory"] == "packages/pdf" + assert source["content_digest"] == P.load_pending("pdf")["tree"]["digest"] + assert P.load_pending("pdf")["creation"]["source"] == "github:acme/skills" + assert records.validate_candidate(payload["candidate"]) == [] + assert (_tree_hash(library), _tree_hash(native)) == before + + +def test_moved_github_head_with_identical_bytes_is_idempotent(tmp_path, monkeypatch, capsys): + """Break caught: commit or temporary clone paths leaking into content identity.""" + from ingot import acquire + + _store(tmp_path, monkeypatch) + remote = tmp_path / "remote" + _github_remote(remote) + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + + assert cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"]) == 0 + first = json.loads(capsys.readouterr().out) + subprocess.run(["git", "-C", str(remote), "commit", "--allow-empty", "-m", "Move HEAD"], + check=True, capture_output=True) + assert cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"]) == 0 + second = json.loads(capsys.readouterr().out) + + assert second["status"] == "duplicate" + assert second["proposal_id"] == first["proposal_id"] + + +def test_changed_github_bytes_produce_a_different_candidate_revision(tmp_path, monkeypatch, capsys): + """Break caught: commit provenance changing while the candidate still names old package bytes.""" + from ingot import acquire + + _store(tmp_path, monkeypatch) + remote = tmp_path / "remote" + _github_remote(remote) + monkeypatch.setattr(acquire, "_remote_url", lambda repository: remote.as_uri()) + assert cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"]) == 0 + first = json.loads(capsys.readouterr().out)["candidate"]["candidate_revision"] + P.pending_path("pdf").unlink() + + skill_md = remote / "packages" / "pdf" / "SKILL.md" + skill_md.write_text(skill_md.read_text(encoding="utf-8") + "Changed upstream.\n", + encoding="utf-8") + subprocess.run(["git", "-C", str(remote), "add", "."], check=True) + subprocess.run(["git", "-C", str(remote), "commit", "-m", "Change bytes"], check=True, + capture_output=True) + assert cli.main(["add", "github:acme/skills", "--skill", "packages/pdf", "--json"]) == 0 + second = json.loads(capsys.readouterr().out)["candidate"]["candidate_revision"] + + assert second != first + + +def test_the_installed_script_works_outside_the_repository(tmp_path): + """Run from elsewhere, with the repo root off `sys.path`. + + Every other test in this file runs with the repository as the working directory, which puts + `optimize` on the import path whether or not it was ever installed. That masked a real break: + the console script resolved `ingot` and `mcp_server` from site-packages and then failed on + `import optimize` for anyone who ran it from their own directory. Only a test that leaves the + repository can see it.""" + script = Path(sys.executable).parent / "ingot" + if not script.exists(): + pytest.skip("console script not installed; run `pip install -e .`") + + library, package = tmp_path / "lib", tmp_path / "src" / "pdf" + library.mkdir() + package.mkdir(parents=True) + (package / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\n\nBody.\n", encoding="utf-8") + + environment = {**os.environ, "SKILL_ROUTER_PATHS": str(library), + "INGOT_ACTOR": "integration-test"} + environment.pop("PYTHONPATH", None) + result = subprocess.run([str(script), "review", str(package), "--json"], + cwd=tmp_path, capture_output=True, text=True, env=environment) + + assert result.returncode == 0, result.stderr + assert json.loads(result.stdout)["skill"] == "pdf" + + +def test_the_installed_script_can_reach_the_quarantine_from_outside_the_repository(tmp_path): + """The same escape, for the verb that needs `optimize`. `ingot add` is the whole point of this + PR, so a version of it that only runs inside a checkout is not shipped.""" + script = Path(sys.executable).parent / "ingot" + if not script.exists(): + pytest.skip("console script not installed; run `pip install -e .`") + + package = tmp_path / "src" / "pdf" + package.mkdir(parents=True) + (package / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\n\nBody.\n", encoding="utf-8") + + environment = {**os.environ, "INGOT_ACTOR": "integration-test"} + environment.pop("PYTHONPATH", None) + result = subprocess.run([str(script), "add", "--help"], + cwd=tmp_path, capture_output=True, text=True, env=environment) + + assert result.returncode == 0, result.stderr + # Importing the admission path is what actually proves `optimize` resolves from site-packages. + probe = subprocess.run([sys.executable, "-c", "import ingot.admission; from ingot.optimize import " + "ingress; print(ingress.submit_package_ingest.__name__)"], + cwd=tmp_path, capture_output=True, text=True, env=environment) + assert probe.returncode == 0, probe.stderr + assert probe.stdout.strip() == "submit_package_ingest" + + +def test_listing_the_cli_stays_light_after_admission_exists(): + """`ingot list` and `ingot review` must still import nothing heavy. Admission pulls in + `optimize`, so it has to stay behind a function-local import rather than riding along at module + import time.""" + heavy = ["fastapi", "onnxruntime", "langgraph", "langfuse", "litellm", "fastembed", "ingot.optimize"] + program = f"import sys, ingot.cli; print([m for m in {heavy!r} if m in sys.modules])" + + result = subprocess.run([sys.executable, "-c", program], capture_output=True, text=True) + + assert result.returncode == 0, result.stderr + assert result.stdout.strip() == "[]" + + +# --- artifact fidelity ---------------------------------------------------------------------- + +PNG = b"\x89PNG\r\n\x1a\n" + bytes(range(256)) + b"\xff\xfe\xfd" + + +def test_a_binary_asset_survives_admission_byte_for_byte(tmp_path, monkeypatch): + """The defect this exists to stop: admission used to reduce a package to decoded text, so an + image was reviewed as part of the candidate and then was not in it.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf") + (package / "assets").mkdir() + (package / "assets" / "logo.png").write_bytes(PNG) + + admission.add_package(package, actor="operator") + + entry = _entry(P.load_pending("pdf"), "assets/logo.png") + assert entry["size"] == len(PNG) + assert entry["sha256"] == hashlib.sha256(PNG).hexdigest() + staged = tree.staged_dir(P.load_pending("pdf")["tree"]["digest"]) + assert (staged / "assets" / "logo.png").read_bytes() == PNG + + +def test_the_approved_revision_changes_when_an_asset_changes(tmp_path, monkeypatch): + """The whole point of binding a revision. While assets were dropped, two packages differing + only in an image hashed identically, so approving one approved the other.""" + _store(tmp_path, monkeypatch) + first = _package(tmp_path / "one", "pdf") + (first / "logo.png").write_bytes(PNG) + second = _package(tmp_path / "two", "pdf") + (second / "logo.png").write_bytes(PNG[:-1] + b"\x00") + + one = admission.add_package(first, actor="operator")["candidate"]["candidate_revision"] + P.pending_path("pdf").unlink() + two = admission.add_package(second, actor="operator")["candidate"]["candidate_revision"] + + assert one != two + + +def test_the_executable_bit_is_recorded_and_nothing_else_is(tmp_path, monkeypatch): + """Git stores two modes. Recording the raw one would describe a file the vault cannot serve.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", files={"run.sh": "#!/bin/sh\necho hi\n", + "notes.md": "# Notes\n"}) + (package / "run.sh").chmod(0o777) + + admission.add_package(package, actor="operator") + + pending = P.load_pending("pdf") + assert _entry(pending, "run.sh")["mode"] == 0o755 + assert _entry(pending, "notes.md")["mode"] == 0o644 + + +def test_a_symlink_is_refused_and_the_source_is_left_alone(tmp_path, monkeypatch): + """Admission stages exact bytes. Preserving a link puts a path into the vault that leads out of + the library; flattening it changes the artifact's shape. Neither is admission's call.""" + root = _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"}) + (package / "link.md").symlink_to(package / "notes.md") + before = sorted(item.name for item in package.iterdir()) + + with pytest.raises(admission.AdmissionRefused, match="symlink-unsupported"): + admission.add_package(package, actor="operator") + + assert sorted(item.name for item in package.iterdir()) == before + assert P.load_pending("pdf") is None + assert not (root / "pdf").exists() + + +def test_a_binary_asset_does_not_arrive_as_a_decoded_component(tmp_path, monkeypatch): + """Two descriptions of one file is one description too many: the receipt would carry a decoded + copy beside the exact bytes, and the publisher would have to pick.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"}) + + admission.add_package(package, actor="operator") + + components = P.load_pending("pdf")["challenger_components"] + assert sorted(components) == ["body", "description", "frontmatter"] + + +def test_an_ingested_package_is_approvable_the_moment_it_is_quarantined(tmp_path, monkeypatch): + """The freshness check has to recompute the revision the way admission computed it. Deriving it + from the decoded components instead makes every package with a second file permanently stale: + quarantined, reviewable, and impossible to approve.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"}) + admission.add_package(package, actor="operator") + + assert P.stale_evidence_reason("pdf", P.load_pending("pdf")) is None + + +def test_an_ingested_package_whose_staged_bytes_moved_is_stale(tmp_path, monkeypatch): + """The other direction. A freshness check that cannot go stale is not checking anything.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"}) + admission.add_package(package, actor="operator") + pending = P.load_pending("pdf") + staged = tree.staged_dir(pending["tree"]["digest"]) + (staged / "notes.md").write_text("# Substituted\n", encoding="utf-8") + + assert P.stale_evidence_reason("pdf", pending) is not None + + +def test_an_ingested_package_survives_the_whole_cli_lane(tmp_path, monkeypatch, capsys): + """add, then approve, on the real verbs. The unit tests queue publications directly, so this is + the only place the approval path itself is exercised on an ingested package.""" + _store(tmp_path, monkeypatch) + package = _package(tmp_path / "src", "pdf", files={"notes.md": "# Notes\n"}) + assert cli.main(["add", f"file:{package}"]) == 0 + capsys.readouterr() + + assert cli.main(["approve", "pdf", "--actor", "operator"]) == 0 + + from ingot.optimize import publication as Q + receipt = Q.publication_for_skill("pdf") + assert receipt["state"] == "approved_publishing" + assert receipt["candidate_revision"] == P.load_pending("pdf")["evidence"]["challenger"]["revision"] + + +def _entry(pending, path): + return next(item for item in pending["tree"]["files"] if item["path"] == path) diff --git a/tests/test_agy_judge.py b/tests/test_agy_judge.py new file mode 100644 index 0000000..82969ae --- /dev/null +++ b/tests/test_agy_judge.py @@ -0,0 +1,408 @@ +"""Contract tests for the subscription-backed Agy judge process boundary.""" +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + +import pytest + +from ingot.optimize import agy_judge as A + + +FIXTURE = Path(__file__).parent / "fixtures" / "agy" / "judge-stream.jsonl" +CHECKLIST = [{"id": "probe", "criterion": "Return the requested token.", "weight": 1}] +TWO_ITEM_CHECKLIST = [ + *CHECKLIST, + {"id": "second", "criterion": "Return the second requested token.", "weight": 1}, +] +VALID_GRADE = { + "items": {"probe": {"verdict": "pass", "note": ""}}, + "feedback": "ok", +} +VALID_USAGE = { + "input_tokens": 1, + "output_tokens": 1, + "thinking_tokens": 0, + "cache_read_tokens": 0, + "total_tokens": 2, +} + + +def _terminal(**changes: object) -> dict: + result = { + "status": "SUCCESS", + "structured_output": VALID_GRADE, + "usage": VALID_USAGE, + } + result.update(changes) + return {"event": "result", "result": result} + + +def _jsonl(*events: dict) -> str: + return "\n".join(json.dumps(event) for event in events) + + +def _runtime(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> tuple[Path, Path]: + agy_bin = tmp_path / "agy" + agy_bin.write_text("fixture executable") + agy_bin.chmod(0o700) + workspace = tmp_path / "workspace" + workspace.mkdir() + monkeypatch.setenv("AGY_BIN", str(agy_bin)) + monkeypatch.setenv("AGY_JUDGE_WORKSPACE", str(workspace)) + return agy_bin, workspace + + +def test_captured_stream_supplies_the_fixed_judge_contract(): + grade, usage = A.parse_stream(FIXTURE.read_text(), CHECKLIST) + assert A.AGY_MODEL == "gemini-3.6-flash-medium" + assert A.AGY_IDENTITY == "agy/gemini-3.6-flash-medium" + assert grade["items"]["probe"]["verdict"] == "pass" + assert usage["total_tokens"] == 24067 + + +def test_schema_requires_exact_checklist_items_and_feedback(): + assert A.judge_schema(CHECKLIST) == { + "type": "object", + "additionalProperties": False, + "required": ["items", "feedback"], + "properties": { + "items": { + "type": "object", + "additionalProperties": False, + "required": ["probe"], + "properties": { + "probe": { + "type": "object", + "additionalProperties": False, + "required": ["verdict", "note"], + "properties": { + "verdict": {"enum": ["pass", "partial", "fail"]}, + "note": {"type": "string"}, + }, + } + }, + }, + "feedback": {"type": "string"}, + }, + } + + +def test_child_environment_scrubs_provider_routing_and_disables_updates(): + parent = { + "PATH": "/bin", + "BASE_URL": "https://provider.invalid", + "API_KEY": "secret-a", + "MODEL_API_KEY": "secret-b", + "OPENROUTER_API_KEY": "secret-c", + "OPENAI_API_KEY": "secret-d", + "ANTHROPIC_API_KEY": "secret-e", + "GEMINI_API_KEY": "secret-f", + "GOOGLE_API_KEY": "secret-g", + "VERTEX_API_KEY": "secret-h", + "AGY_CLI_DISABLE_AUTO_UPDATE": "false", + } + child = A.agy_process_env(parent) + assert child == {"PATH": "/bin", "AGY_CLI_DISABLE_AUTO_UPDATE": "true"} + assert "OPENROUTER_API_KEY" not in child + + +@pytest.mark.parametrize( + "case,stdout", + [ + ("zero terminal results", _jsonl({"event": "init"})), + ("two terminal results", _jsonl(_terminal(), _terminal())), + ("non-success status", _jsonl(_terminal(status="FAILED"))), + ( + "missing structured output", + _jsonl(_terminal(structured_output=None, response=json.dumps(VALID_GRADE))), + ), + ( + "missing checklist item", + _jsonl(_terminal(structured_output={"items": {}, "feedback": "bad"})), + ), + ( + "extra checklist item", + _jsonl(_terminal(structured_output={ + "items": { + **VALID_GRADE["items"], + "other": {"verdict": "fail", "note": "not requested"}, + }, + "feedback": "bad", + })), + ), + ( + "unknown verdict", + _jsonl(_terminal(structured_output={ + "items": {"probe": {"verdict": "excellent", "note": ""}}, + "feedback": "bad", + })), + ), + ( + "invalid note", + _jsonl(_terminal(structured_output={ + "items": {"probe": {"verdict": "pass", "note": ["not", "text"]}}, + "feedback": "bad", + })), + ), + ("non-object event", "[]"), + ("non-object result", _jsonl({"event": "result", "result": []})), + ( + "missing usage", + _jsonl({ + "event": "result", + "result": {"status": "SUCCESS", "structured_output": VALID_GRADE}, + }), + ), + ("non-object usage", _jsonl(_terminal(usage=[]))), + ( + "invalid feedback", + _jsonl(_terminal(structured_output={ + "items": VALID_GRADE["items"], + "feedback": ["not", "text"], + })), + ), + ( + "extra grade key", + _jsonl(_terminal(structured_output={**VALID_GRADE, "score": 1.0})), + ), + ( + "extra item key", + _jsonl(_terminal(structured_output={ + "items": {"probe": {"verdict": "pass", "note": "", "score": 1.0}}, + "feedback": "bad", + })), + ), + ("malformed jsonl", '{"event":"result"'), + ], +) +def test_invalid_streams_raise_without_manufacturing_a_grade(case: str, stdout: str): + with pytest.raises(A.AgyJudgeError, match="."): + A.parse_stream(stdout, CHECKLIST) + + +def test_stream_parser_accepts_every_item_in_a_two_id_checklist(): + grade = { + "items": { + "probe": {"verdict": "pass", "note": ""}, + "second": {"verdict": "partial", "note": "one mismatch"}, + }, + "feedback": "Fix the second item.", + } + parsed, usage = A.parse_stream( + _jsonl(_terminal(structured_output=grade, usage=VALID_USAGE)), + TWO_ITEM_CHECKLIST, + ) + assert parsed == grade + assert usage == VALID_USAGE + + +def test_invoke_uses_only_the_explicit_sandboxed_process_boundary(monkeypatch, tmp_path): + agy_bin, workspace = _runtime(monkeypatch, tmp_path) + monkeypatch.setenv("OPENROUTER_API_KEY", "must-not-leak") + seen = {} + + def fake_run(argv, **kwargs): + seen.update(argv=argv, **kwargs) + return subprocess.CompletedProcess(argv, 0, stdout=FIXTURE.read_text(), stderr="ignored") + + monkeypatch.setattr(A.subprocess, "run", fake_run) + grade, usage = A.invoke("grade this", CHECKLIST, timeout=17.9) + + assert grade["items"]["probe"]["verdict"] == "pass" + assert usage["total_tokens"] == 24067 + assert seen["argv"] == [ + str(agy_bin), + "--model", A.AGY_MODEL, + "--print", "grade this", + "--output-format", "stream-json", + "--json-schema", json.dumps(A.judge_schema(CHECKLIST)), + "--sandbox", + "--mode", "plan", + "--disable-slash-commands", + "--print-timeout", "17s", + ] + assert seen["cwd"] == workspace + assert seen["text"] is True + assert seen["capture_output"] is True + assert seen["timeout"] == pytest.approx(27.9) + assert seen["check"] is False + assert "OPENROUTER_API_KEY" not in seen["env"] + assert seen["env"]["AGY_CLI_DISABLE_AUTO_UPDATE"] == "true" + + +@pytest.mark.parametrize("failure", ["timeout", "nonzero"]) +def test_process_failures_raise_without_using_stdout_or_stderr_as_a_grade( + failure, monkeypatch, tmp_path +): + _runtime(monkeypatch, tmp_path) + + def fake_run(argv, **kwargs): + if failure == "timeout": + raise subprocess.TimeoutExpired(argv, kwargs["timeout"], output=FIXTURE.read_text()) + return subprocess.CompletedProcess( + argv, + 9, + stdout=FIXTURE.read_text(), + stderr=json.dumps(VALID_GRADE), + ) + + monkeypatch.setattr(A.subprocess, "run", fake_run) + with pytest.raises(A.AgyJudgeError, match="."): + A.invoke("grade this", CHECKLIST) + + +def test_resource_exhaustion_is_rate_limited_and_retried(monkeypatch, tmp_path): + _runtime(monkeypatch, tmp_path) + calls = [] + sleeps = [] + clock = iter([0.0, 0.0, 3.2]) + + def fake_run(argv, **kwargs): + calls.append(argv) + if len(calls) == 1: + return subprocess.CompletedProcess( + argv, 1, + stdout=_jsonl({ + "event": "result", + "result": { + "status": "ERROR", + "error": "Eligibility check failed: RESOURCE_EXHAUSTED (code 429)", + }, + }), + stderr="", + ) + return subprocess.CompletedProcess(argv, 0, stdout=FIXTURE.read_text(), stderr="") + + monkeypatch.setattr(A.subprocess, "run", fake_run) + monkeypatch.setattr(A.time, "monotonic", lambda: next(clock)) + monkeypatch.setattr(A.time, "sleep", sleeps.append) + A._next_launch_at = 0.0 + + grade, _ = A.invoke("grade this", CHECKLIST) + + assert grade["items"]["probe"]["verdict"] == "pass" + assert len(calls) == 2 + assert sleeps == [pytest.approx(3.2)] + + +def test_non_quota_process_failure_is_not_retried(monkeypatch, tmp_path): + _runtime(monkeypatch, tmp_path) + calls = [] + + def fake_run(argv, **kwargs): + calls.append(argv) + return subprocess.CompletedProcess(argv, 1, stdout="not a quota result", stderr="") + + monkeypatch.setattr(A.subprocess, "run", fake_run) + monkeypatch.setattr(A.time, "monotonic", lambda: 0.0) + A._next_launch_at = 0.0 + + with pytest.raises(A.AgyJudgeError, match="status 1"): + A.invoke("grade this", CHECKLIST) + assert len(calls) == 1 + + +def test_preflight_records_version_and_requires_the_fixed_model(monkeypatch, tmp_path): + agy_bin, workspace = _runtime(monkeypatch, tmp_path) + calls = [] + + def fake_run(argv, **kwargs): + calls.append((argv, kwargs)) + stdout = "agy 1.1.11\n" if argv[-1] == "--version" else f"{A.AGY_MODEL}\n" + return subprocess.CompletedProcess(argv, 0, stdout=stdout, stderr="") + + monkeypatch.setattr(A.subprocess, "run", fake_run) + assert A.preflight() == { + "identity": A.AGY_IDENTITY, + "model": A.AGY_MODEL, + "version": "agy 1.1.11", + "billing_mode": "subscription", + } + assert [call[0] for call in calls] == [ + [str(agy_bin), "--version"], + [str(agy_bin), "models"], + ] + assert all(call[1]["cwd"] == workspace for call in calls) + assert all("OPENROUTER_API_KEY" not in call[1]["env"] for call in calls) + + +@pytest.mark.parametrize( + "invalid_runtime", + ["missing", "relative", "non_executable", "bad_workspace"], +) +def test_preflight_rejects_invalid_runtime_before_starting_a_process( + invalid_runtime, monkeypatch, tmp_path +): + agy_bin = tmp_path / "agy" + agy_bin.write_text("fixture executable") + agy_bin.chmod(0o700) + workspace = tmp_path / "workspace" + workspace.mkdir() + monkeypatch.setenv("AGY_BIN", str(agy_bin)) + monkeypatch.setenv("AGY_JUDGE_WORKSPACE", str(workspace)) + + if invalid_runtime == "missing": + monkeypatch.delenv("AGY_BIN") + elif invalid_runtime == "relative": + monkeypatch.setenv("AGY_BIN", "agy") + elif invalid_runtime == "non_executable": + agy_bin.chmod(0o600) + else: + bad_workspace = tmp_path / "workspace-file" + bad_workspace.write_text("not a directory") + monkeypatch.setenv("AGY_JUDGE_WORKSPACE", str(bad_workspace)) + + calls = [] + monkeypatch.setattr(A.subprocess, "run", lambda *args, **kwargs: calls.append(args)) + with pytest.raises(A.AgyJudgeError): + A.preflight() + assert calls == [] + + +@pytest.mark.parametrize( + "failed_command,failure,expected_calls", + [ + ("version", "timeout", 1), + ("version", "empty", 1), + ("models", "nonzero", 2), + ], +) +def test_preflight_stops_at_a_failed_command( + failed_command, failure, expected_calls, monkeypatch, tmp_path +): + _runtime(monkeypatch, tmp_path) + calls = [] + + def fake_run(argv, **kwargs): + calls.append(argv) + command = "version" if argv[-1] == "--version" else "models" + stdout = "agy 1.1.11\n" if command == "version" else f"{A.AGY_MODEL}\n" + if command != failed_command: + return subprocess.CompletedProcess(argv, 0, stdout=stdout, stderr="") + if failure == "timeout": + raise subprocess.TimeoutExpired(argv, kwargs["timeout"], output=stdout) + if failure == "nonzero": + return subprocess.CompletedProcess(argv, 7, stdout=stdout, stderr="grade-like text") + return subprocess.CompletedProcess(argv, 0, stdout="", stderr="") + + monkeypatch.setattr(A.subprocess, "run", fake_run) + with pytest.raises(A.AgyJudgeError): + A.preflight() + assert len(calls) == expected_calls + + +def test_preflight_requires_the_exact_model_token(monkeypatch, tmp_path): + _runtime(monkeypatch, tmp_path) + calls = [] + + def fake_run(argv, **kwargs): + calls.append(argv) + stdout = "agy 1.1.11\n" if argv[-1] == "--version" else f"{A.AGY_MODEL}-preview\n" + return subprocess.CompletedProcess(argv, 0, stdout=stdout, stderr="") + + monkeypatch.setattr(A.subprocess, "run", fake_run) + with pytest.raises(A.AgyJudgeError): + A.preflight() + assert len(calls) == 2 diff --git a/tests/test_auth.py b/tests/test_auth.py index 82af674..f701ef7 100644 --- a/tests/test_auth.py +++ b/tests/test_auth.py @@ -5,7 +5,7 @@ from fastapi.testclient import TestClient -from optimize import promote as P +from ingot.optimize import promote as P from ui import auth from ui.app import app @@ -107,7 +107,6 @@ def _promotable(skill): def test_promote_endpoint_threads_the_authenticated_actor(tmp_path, monkeypatch): import ui.app as ui_app - monkeypatch.setattr(P, "PENDING_DIR", tmp_path / "pending") f = tmp_path / "auth.json" f.write_text(json.dumps({"alice": auth.hash_password("pw")})) monkeypatch.setattr(auth, "AUTH_FILE", f) diff --git a/tests/test_cli.py b/tests/test_cli.py new file mode 100644 index 0000000..462f74f --- /dev/null +++ b/tests/test_cli.py @@ -0,0 +1,150 @@ +"""The `ingot` console entry point. + +The first product experience must not require the full stack, so these tests care about two things +the rest of the suite cannot see: that the CLI runs as an installed console script, and that +importing it does not drag in FastAPI, ONNX, LangGraph, Langfuse, or the optimizer. A CLI that only +works from a repo checkout with every server dependency present is not the lightweight install this +milestone exists to deliver.""" +import json +import shutil +import subprocess +import sys +import textwrap +from pathlib import Path + +import pytest + +from ingot import cli + + +def _skill(root, name, description, body="Do the thing."): + directory = root / name + directory.mkdir(parents=True) + (directory / "SKILL.md").write_text( + textwrap.dedent(f"""\ + --- + name: {name} + description: {description} + --- + + {body} + """), + encoding="utf-8") + return directory + + +def test_list_reports_a_skill_in_an_explicit_root(tmp_path): + _skill(tmp_path, "pdf", "Merge and split PDF files.") + + result = cli.list_library(tmp_path) + + assert [s["name"] for s in result["skills"]] == ["pdf"] + assert result["skills"][0]["description"] == "Merge and split PDF files." + assert len(result["skills"][0]["revision"]) == 64 + + +def test_list_reports_the_roots_it_actually_read(tmp_path): + """`configured_roots` always prepends the local authoring root, even ahead of an explicit one, + so a caller who passes `--root` can still be shown skills from somewhere else. Naming every root + in the payload is what makes that visible instead of baffling.""" + _skill(tmp_path, "pdf", "Merge and split PDF files.") + + result = cli.list_library(tmp_path) + + assert str(tmp_path.resolve()) in result["roots"] + + +def test_list_of_an_empty_root_is_not_an_error(tmp_path): + result = cli.list_library(tmp_path) + + assert result["skills"] == [] + assert result["schema_version"] == cli.LIST_SCHEMA + + +def test_list_skips_a_skill_with_no_description(tmp_path): + """The router keys on description; a skill without one is unroutable and the loader drops it. + The CLI must report the same library the server would serve, not a more generous one.""" + _skill(tmp_path, "pdf", "Merge and split PDF files.") + empty = tmp_path / "blank" + empty.mkdir() + (empty / "SKILL.md").write_text("---\nname: blank\n---\n\nbody\n", encoding="utf-8") + + result = cli.list_library(tmp_path) + + assert [s["name"] for s in result["skills"]] == ["pdf"] + + +def test_main_list_json_emits_the_versioned_payload(tmp_path, capsys): + _skill(tmp_path, "pdf", "Merge and split PDF files.") + + code = cli.main(["list", "--root", str(tmp_path), "--json"]) + + assert code == 0 + payload = json.loads(capsys.readouterr().out) + assert payload["schema_version"] == cli.LIST_SCHEMA + assert [s["name"] for s in payload["skills"]] == ["pdf"] + + +def test_main_list_human_output_names_the_skill(tmp_path, capsys): + _skill(tmp_path, "pdf", "Merge and split PDF files.") + + code = cli.main(["list", "--root", str(tmp_path)]) + + assert code == 0 + assert "pdf" in capsys.readouterr().out + + +def test_main_with_no_command_fails_rather_than_doing_something(capsys): + with pytest.raises(SystemExit) as exit_info: + cli.main([]) + + assert exit_info.value.code != 0 + + +def _console_script() -> str | None: + """The `ingot` script installed beside the interpreter running these tests. + + Not `shutil.which`: a virtualenv that has not been activated is not on PATH, so searching PATH + silently skips the one test that proves the entry point exists. Fall back to PATH for a system + install.""" + beside = Path(sys.executable).parent / "ingot" + return str(beside) if beside.exists() else shutil.which("ingot") + + +@pytest.mark.skipif(_console_script() is None, + reason="console script not installed; run `pip install -e .`") +def test_installed_console_script_runs(tmp_path): + """In situ: the real entry point, not an in-process call. `--help` is the one command that must + work before anything else is configured.""" + result = subprocess.run([_console_script(), "--help"], capture_output=True, text=True) + + assert result.returncode == 0 + assert "list" in result.stdout + + +def test_installed_console_script_lists_a_real_library(tmp_path): + """The acceptance criterion for this PR, run the way a user runs it: the installed script, a + real directory, no services and no Docker.""" + script = _console_script() + if script is None: + pytest.skip("console script not installed; run `pip install -e .`") + _skill(tmp_path, "pdf", "Merge and split PDF files.") + + result = subprocess.run([script, "list", "--root", str(tmp_path), "--json"], + capture_output=True, text=True) + + assert result.returncode == 0, result.stderr + assert "pdf" in [skill["name"] for skill in json.loads(result.stdout)["skills"]] + + +def test_importing_the_cli_stays_light(): + """A subprocess, because the rest of the suite has already imported the heavy stack in-process. + This is the whole point of the milestone: `ingot list` must not need the server's dependencies.""" + heavy = ["fastapi", "onnxruntime", "langgraph", "langfuse", "litellm", "fastembed", "ingot.optimize"] + program = ("import sys, ingot.cli; " + f"print([m for m in {heavy!r} if m in sys.modules])") + + result = subprocess.run([sys.executable, "-c", program], capture_output=True, text=True) + + assert result.returncode == 0, result.stderr + assert result.stdout.strip() == "[]" diff --git a/tests/test_cluster.py b/tests/test_cluster.py new file mode 100644 index 0000000..e25f8b6 --- /dev/null +++ b/tests/test_cluster.py @@ -0,0 +1,132 @@ +"""Unit tests for skill clustering (embeddings mocked; the numpy maths is exercised for real).""" +import numpy as np +import pytest + +from ingot.optimize import cluster as C + + +def _blobs(seed=0): + """Three well-separated groups on the unit sphere — clustering that cannot recover these is + not going to recover anything subtler.""" + rng = np.random.default_rng(seed) + centres = np.array([[1.0, 0, 0], [0, 1.0, 0], [0, 0, 1.0]]) + pts = np.vstack([c + rng.normal(0, 0.05, (8, 3)) for c in centres]) + return C._normalise(pts) + + +def test_normalise_gives_unit_vectors(): + v = C._normalise(np.array([[3.0, 4.0], [0.0, 2.0]])) + assert np.allclose(np.linalg.norm(v, axis=1), 1.0) + + +def test_normalise_survives_a_zero_vector(): + """A zero row would divide by zero and poison every downstream distance with nan.""" + assert np.isfinite(C._normalise(np.zeros((1, 4)))).all() + + +def test_kmeans_recovers_separated_groups(): + labels, _ = C.kmeans(_blobs(), 3) + groups = [set(np.where(labels == j)[0]) for j in range(3)] + assert sorted(len(g) for g in groups) == [8, 8, 8] + + +def test_kmeans_is_deterministic_for_one_library(): + """The view polls. Buckets that reshuffle between identical runs are not something a reader + can build a mental model of.""" + v = _blobs() + assert np.array_equal(C.kmeans(v, 3)[0], C.kmeans(v, 3)[0]) + + +def test_kmeans_never_returns_fewer_buckets_than_asked(): + """An emptied centre must be re-seeded. Left alone it collapses the run to k-1 and the caller + silently gets a different k than the silhouette was computed for.""" + v = C._normalise(np.array([[1.0, 0.0], [1.0, 0.001], [1.0, 0.002], [1.0, 0.003]])) + labels, centres = C.kmeans(v, 3) + assert len(centres) == 3 and np.isfinite(centres).all() + + +def test_silhouette_prefers_the_true_group_count(): + v = _blobs() + scores = {k: C.silhouette(v, C.kmeans(v, k)[0]) for k in (2, 3, 5)} + assert scores[3] == max(scores.values()) + + +def test_silhouette_is_undefined_for_one_bucket(): + assert C.silhouette(_blobs(), np.zeros(24, dtype=int)) == -1.0 + + +def test_project_2d_returns_two_dimensions_and_is_centred(): + xy = C.project_2d(_blobs()) + assert xy.shape == (24, 2) + assert np.allclose(xy.mean(0), 0, atol=1e-9) + + +def test_project_keeps_the_requested_number_of_components(): + assert C.project(_blobs(), 3).shape == (24, 3) + + +def test_reducing_before_clustering_beats_clustering_raw(): + """The reason CLUSTER_DIMS exists. In high dimensions over few points every pair sits at a + similar distance and k-means has nothing to bite on — measured on the real 102-skill library, + raw 1024d scored +0.05 against +0.28 over 10 components. Reproduced here with padded noise + dimensions so the property is checked, not just remembered.""" + rng = np.random.default_rng(3) + signal = _blobs(seed=2) + noise = rng.normal(0, 0.6, (len(signal), 300)) + wide = C._normalise(np.hstack([signal, noise])) + raw = max(C.silhouette(wide, C.kmeans(wide, k)[0]) for k in (3, 4, 5)) + reduced = C._normalise(C.project(wide, 10)) + cut = max(C.silhouette(reduced, C.kmeans(reduced, k)[0]) for k in (3, 4, 5)) + assert cut > raw + + +def test_top_terms_favours_what_is_distinctive_not_what_is_common(): + """Every skill says "use"; only one group says "invoice". Raw frequency returns the first.""" + texts = ["use invoice billing ledger", "use invoice payment ledger", + "use pytest fixture mock", "use pytest assert mock"] + assert "invoice" in C.top_terms(texts, [0, 1], k=3) + assert "use" not in C.top_terms(texts, [0, 1], k=3) + + +def test_top_terms_drops_stopwords_and_short_tokens(): + texts = ["the and for a bb kubernetes", "the and for a bb postgres"] + assert C.top_terms(texts, [0], k=5) == ["kubernetes"] + + +def test_label_for_names_a_bucket_from_its_terms(): + assert C.label_for(["aws", "lambda", "deploy", "iam"]) == "aws · lambda · deploy" + assert C.label_for([]) == "unlabelled" + + +def test_skill_texts_lead_with_the_name_as_words(): + """Hyphenated names carry the topic in this library, and an embedder splits them better as + words than as one token.""" + s = type("S", (), {"name": "aws-cdk", "description": "Infra as code."})() + assert C.skill_texts([s]) == ["aws cdk. Infra as code."] + + +def test_build_refuses_a_library_too_small_to_cluster(monkeypatch): + monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda *a, **k: []) + with pytest.raises(SystemExit, match="SKILL_ROUTER_PATHS"): + C.build(log=lambda *a: None) + + +def test_build_shapes_the_payload_the_ui_reads(monkeypatch, tmp_path): + names = ["billing-runbook", "invoice-audit", "pytest-fixtures", "mock-patterns", + "k8s-deploy", "helm-charts", "prose-edit", "headline-cuts"] + skills = [type("S", (), {"name": n, "description": n.replace("-", " "), "metadata": {}})() + for n in names] + monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda *a, **k: skills) + rng = np.random.default_rng(1) + vecs = {n: rng.normal(size=6) + (i // 2) * 3 for i, n in enumerate(names)} + monkeypatch.setattr("ingot.mcp_server.embedding.build_embedding", + lambda *a, **k: type("E", (), {"embed": lambda self, t: [ + vecs[n] for n in names]})()) + monkeypatch.setattr(C, "K_MAX", 4) + data = C.build(log=lambda *a: None) + assert data["n_skills"] == 8 and data["clusters"] + assert sum(c["size"] for c in data["clusters"]) == 8 # every skill lands somewhere + assert data["clusters"] == sorted(data["clusters"], key=lambda c: -c["size"]) + member = data["clusters"][0]["members"][0] + assert set(member) == {"name", "xy", "provenance"} and len(member["xy"]) == 2 + assert {m["name"] for c in data["clusters"] for m in c["members"]} == set(names) diff --git a/tests/test_compat.py b/tests/test_compat.py index 5eb6516..7a828ff 100644 --- a/tests/test_compat.py +++ b/tests/test_compat.py @@ -4,7 +4,7 @@ import pytest -from optimize import compat +from ingot.optimize import compat def test_compat_models_parses_env_else_defaults(monkeypatch): @@ -15,10 +15,28 @@ def test_compat_models_parses_env_else_defaults(monkeypatch): assert compat.compat_models() == ["solo/model"] +def test_the_tasks_own_checklist_reaches_the_judge(monkeypatch): + """Without it every task grades on judge()'s generic four, which any capable model passes on an + easy task — so the sweep measures the tasks' difficulty rather than the skill.""" + checklist = [{"id": "cites_success_criterion", "criterion": "cites a criterion", "weight": 3, + "dimension": "correctness"}] + seen = {} + + def fake_judge(task, rubric, answer, **kwargs): + seen.update(kwargs) + return {"score": 1.0} + + monkeypatch.setattr(compat, "judge", fake_judge) + monkeypatch.setattr(compat, "invoke_retry", + lambda llm, messages: type("M", (), {"content": "answer"})()) + compat._score(object(), "system", {"task": "t", "rubric": "r", "checklist": checklist}) + assert seen["checklist"] == checklist + + def test_run_compat_sweeps_models_and_computes_lift(tmp_path, monkeypatch): (tmp_path / "tailwind").mkdir() (tmp_path / "tailwind" / "SKILL.md").write_text("x") - monkeypatch.setattr(compat, "SKILLS_DIR", tmp_path) + monkeypatch.setattr(compat, "resolve_skill_dir", lambda name: tmp_path / name) monkeypatch.setattr(compat, "COMPAT_DIR", tmp_path / "out") monkeypatch.setenv("COMPAT_MODELS", "m1,m2") monkeypatch.setattr(compat, "load_tasks", @@ -26,7 +44,8 @@ def test_run_compat_sweeps_models_and_computes_lift(tmp_path, monkeypatch): monkeypatch.setattr(compat, "optimizable_components", lambda d: {"description": "d", "body": "THEBODY"}) monkeypatch.setattr(compat, "_llm", lambda model: model) # pass the model name through as the "llm" # the skill arm serves THEBODY (score 0.9); the no-skill baseline does not (0.2) - monkeypatch.setattr(compat, "_score", lambda llm, system, task: 0.9 if "THEBODY" in system else 0.2) + monkeypatch.setattr(compat, "_score", + lambda llm, system, task, role="compat": 0.9 if "THEBODY" in system else 0.2) out = compat.run_compat("tailwind", log=lambda *a: None) assert set(out["models"]) == {"m1", "m2"} @@ -39,7 +58,100 @@ def test_run_compat_sweeps_models_and_computes_lift(tmp_path, monkeypatch): assert written["skill"] == "tailwind" and set(written["models"]) == {"m1", "m2"} +def test_baseline_arm_is_cached_across_runs(tmp_path, monkeypatch): + """The no-skill baseline does not depend on the skill body, so a re-sweep after editing a skill + must not pay for it twice — that was half the cost of every run after the first.""" + (tmp_path / "tailwind").mkdir() + (tmp_path / "tailwind" / "SKILL.md").write_text("x") + monkeypatch.setattr(compat, "resolve_skill_dir", lambda name: tmp_path / name) + monkeypatch.setattr(compat, "COMPAT_DIR", tmp_path / "out") + monkeypatch.setattr(compat, "BASELINE_CACHE_DIR", tmp_path / "cache") + monkeypatch.setenv("COMPAT_MODELS", "m1") + holdout = [{"task": "t", "rubric": "r"}] + monkeypatch.setattr(compat, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(compat, "_llm", lambda model: model) + + served = [] + + def fake_score(llm, system, task, role="compat"): + served.append("skill" if "THEBODY" in system else "baseline") + return 0.9 if "THEBODY" in system else 0.2 + + monkeypatch.setattr(compat, "optimizable_components", lambda d: {"description": "d", "body": "THEBODY"}) + monkeypatch.setattr(compat, "_score", fake_score) + + first = compat.run_compat("tailwind", log=lambda *a: None) + assert served == ["skill", "baseline"] + + served.clear() + second = compat.run_compat("tailwind", log=lambda *a: None) + assert served == ["skill"], "the baseline arm was re-run instead of reused" + assert second["models"]["m1"]["baseline_mean"] == first["models"]["m1"]["baseline_mean"] + + +def test_compat_serves_the_local_model_from_the_local_endpoint(monkeypatch): + """AGENT_MODEL is served by this box's own endpoint — a free row in the grid. Every other slug + is hosted. Routing the whole sweep through one endpoint is what forced a by-hand override.""" + monkeypatch.setenv("AGENT_MODEL", "dot-backbone") + monkeypatch.setenv("MODEL_BASE_URL", "http://local:8011/v1") + monkeypatch.setenv("BASE_URL", "https://openrouter.ai/api/v1") + monkeypatch.setenv("MODEL_API_KEY", "local-key") + monkeypatch.setenv("API_KEY", "hosted-key") + + assert str(compat._llm("dot-backbone").openai_api_base) == "http://local:8011/v1" + assert str(compat._llm("anthropic/claude-sonnet-4.5").openai_api_base) == "https://openrouter.ai/api/v1" + + +def _stub_sweep(tmp_path, monkeypatch, models: str, score): + (tmp_path / "tailwind").mkdir(exist_ok=True) + (tmp_path / "tailwind" / "SKILL.md").write_text("x") + monkeypatch.setattr(compat, "resolve_skill_dir", lambda name: tmp_path / name) + monkeypatch.setattr(compat, "COMPAT_DIR", tmp_path / "out") + monkeypatch.setattr(compat, "BASELINE_CACHE_DIR", tmp_path / "cache") + monkeypatch.setenv("COMPAT_MODELS", models) + monkeypatch.setattr(compat, "load_tasks", + lambda skill: ([], [{"task": "t", "rubric": "r"}], {})) + monkeypatch.setattr(compat, "optimizable_components", + lambda d: {"description": "d", "body": "THEBODY"}) + monkeypatch.setattr(compat, "_llm", lambda model: model) + monkeypatch.setattr(compat, "_score", score) + + +def test_an_unreachable_model_does_not_discard_the_rows_already_paid_for(tmp_path, monkeypatch): + """A slug with no ZDR-qualified endpoint 404s on its first call. Before this the exception + escaped the sweep, so an earlier model's scores were computed, billed, and then thrown away + with no matrix written at all.""" + def score(llm, system, task, role="compat"): + if llm == "gone/model": + raise RuntimeError("Error code: 404 - No endpoints found for gone/model.") + return 0.9 if "THEBODY" in system else 0.2 + + _stub_sweep(tmp_path, monkeypatch, "ok/model,gone/model", score) + out = compat.run_compat("tailwind", log=lambda *a: None) + + assert out["models"]["ok/model"]["lift"] == pytest.approx(0.7) + assert "404" in out["models"]["gone/model"]["error"] + # An unavailable row must not read as a measured zero: that is the reading a reader most + # wants to make, and it is the opposite of the truth. + assert "lift" not in out["models"]["gone/model"] + written = json.loads((tmp_path / "out" / "tailwind.json").read_text()) + assert set(written["models"]) == {"ok/model", "gone/model"} + + +def test_a_sweep_where_no_model_could_be_reached_fails_loudly(tmp_path, monkeypatch): + """Writing an all-error matrix would leave a file that looks like a completed measurement.""" + def score(llm, system, task, role="compat"): + raise RuntimeError("Error code: 401 - no key") + + _stub_sweep(tmp_path, monkeypatch, "a/model,b/model", score) + with pytest.raises(SystemExit, match="nothing was measured"): + compat.run_compat("tailwind", log=lambda *a: None) + assert not (tmp_path / "out" / "tailwind.json").exists() + + def test_run_compat_rejects_unknown_skill(tmp_path, monkeypatch): - monkeypatch.setattr(compat, "SKILLS_DIR", tmp_path) - with pytest.raises(SystemExit, match="No skill named"): + """An unresolvable name is a roots misconfiguration, and the message has to say so — the old + text pointed at skills/, a directory the operator may never have configured.""" + monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda *a, **k: []) + with pytest.raises(SystemExit, match="SKILL_ROUTER_PATHS"): compat.run_compat("nope", log=lambda *a: None) diff --git a/tests/test_compose_managed.py b/tests/test_compose_managed.py new file mode 100644 index 0000000..c4f6846 --- /dev/null +++ b/tests/test_compose_managed.py @@ -0,0 +1,103 @@ +"""The tracked deployment must declare the invariant it claims. + +`docker compose up` with no `-f` is the configuration a reader gets, and it must not launch a +writable stack while the README describes controlled activation. These are static reads of the +tracked YAML: they cannot prove the containers behave, which needs a daemon and the compose smoke +script, but they do catch the failure that actually happened — an invariant that lived only in one +machine's untracked overlay.""" +import yaml +import pytest +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +SERVED = "/app/skills" + + +def _compose(name): + return yaml.safe_load((ROOT / name).read_text(encoding="utf-8")) + + +def _mounts(service, target): + for mount in service.get("volumes") or []: + if isinstance(mount, str) and mount.split(":")[1:2] == [target]: + yield mount + + +@pytest.fixture(scope="module") +def managed(): + return _compose("docker-compose.yml") + + +def test_the_default_stack_has_exactly_one_writer_of_the_served_library(managed): + writers = [name for name, service in managed["services"].items() + if any(not mount.endswith(":ro") for mount in _mounts(service, SERVED))] + + assert writers == [], f"these services can change what is served without an approval: {writers}" + + +def test_every_service_that_serves_the_library_mounts_the_vault(managed): + """A service reading a different directory than the publisher writes serves stale bytes and + reports no drift, because nothing is comparing the two.""" + sources = {mount.split(":")[0] + for service in managed["services"].values() + for mount in _mounts(service, SERVED)} + + assert sources == {"./vault"} + + +def test_the_publisher_is_in_the_default_stack_and_owns_the_vault(managed): + publisher = managed["services"]["publisher"] + + assert "profiles" not in publisher, "the one writer must not be opt-in" + assert publisher["environment"]["INGOT_VAULT_PATH"] == "/app/vault" + writable = [mount for mount in publisher["volumes"] + if mount.startswith("./vault:") and not mount.endswith(":ro")] + assert writable, "the publisher must be able to write the vault" + + +def test_the_publisher_and_the_console_run_as_the_same_user(managed): + """Approval writes receipts at mode 0700. A publisher running as a different user sees an + empty queue and approvals never publish — the exact stall this deployment already hit.""" + assert managed["services"]["publisher"]["user"] == managed["services"]["ui"]["user"] + + +def test_every_service_that_touches_state_names_where_it_lives(managed): + """State no longer defaults to a directory beside the code. A service that mounts a state + volume without naming the paths would write its review queue and receipts inside the image, + where the next `docker compose build` discards them.""" + for name, service in managed["services"].items(): + mounts = [mount for mount in (service.get("volumes") or []) if isinstance(mount, str) + and mount.split(":")[1:2] and mount.split(":")[1].startswith(("/app/skills", + "/app/runs", + "/app/vault"))] + if not mounts: + continue + environment = service.get("environment") or {} + assert "INGOT_RUNS" in environment, f"{name} mounts state without naming INGOT_RUNS" + assert environment.get("INGOT_LIBRARY"), f"{name} mounts state without naming INGOT_LIBRARY" + + +def test_the_default_backend_is_local(managed): + backend = managed["services"]["publisher"]["environment"]["INGOT_PUBLISH_BACKEND"] + + assert backend.startswith("${INGOT_PUBLISH_BACKEND:-local}") + + +def test_the_development_override_is_explicit_about_being_unmanaged(): + dev = _compose("compose.dev.yaml") + + assert dev["services"]["publisher"]["deploy"]["replicas"] == 0 + writable = {name for name, service in dev["services"].items() + if any(not mount.endswith(":ro") for mount in _mounts(service, SERVED))} + assert writable, "the development stack is the writable one; that is its whole purpose" + assert dev["services"]["unmanaged"]["command"] == ["ingot", "status"] + assert dev["services"]["unmanaged"]["environment"]["INGOT_MODE"] == "dev" + assert dev["services"]["ui"]["environment"]["INGOT_MODE"] == "dev" + + +def test_the_forge_override_never_infers_the_repository(): + forge = _compose("compose.forge.yaml") + environment = forge["services"]["publisher"]["environment"] + + assert environment["INGOT_PUBLISH_BACKEND"] == "forge" + assert environment["INGOT_FORGE_REPOSITORY"].startswith("${INGOT_FORGE_REPOSITORY:?") diff --git a/tests/test_decisions.py b/tests/test_decisions.py new file mode 100644 index 0000000..42aaccb --- /dev/null +++ b/tests/test_decisions.py @@ -0,0 +1,271 @@ +"""`ingot pending`, `approve`, `reject`, `history`, `rollback`. + +The command line wraps the services the console already calls. What these tests protect is that it +stays a wrapper: no second approval path, no direct activation helper, and no verb that changes a +served byte. Approval and rollback queue a receipt; the publisher is what acts on it.""" +import json + +import pytest + +from ingot import cli, decisions +from ingot.mcp_server.registry import optimizable_components, skill_revision +from ingot.optimize import promote as P +from ingot.optimize import publication as Q + + +def _library(tmp_path, monkeypatch): + root = tmp_path / "skills" + root.mkdir() + monkeypatch.setenv("INGOT_LIBRARY", str(root)) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root)) + return root + + +def _skill(root, name="pdf", body="old body"): + directory = root / name + directory.mkdir(exist_ok=True) + (directory / "SKILL.md").write_text( + f"---\nname: {name}\ndescription: Merge PDFs.\n---\n{body}\n") + return directory + + +def _quarantine(directory, *, promotable=True, blocked=()): + skill = directory.name + champion = optimizable_components(directory) + challenger = {**champion, "body": "new body"} + pending = { + "skill": skill, + "champion_components": champion, + "challenger_components": challenger, + "gate": {"promotable": promotable, "blocked": list(blocked)}, + "evidence": {"champion": {"revision": skill_revision(directory)}, + "challenger": {"revision": skill_revision(directory, challenger)}}, + } + P.save_pending(skill, pending) + return pending + + +# --------------------------------------------------------------------------- the four words + +@pytest.mark.parametrize("record,expected", [ + ({"state": "approved_publishing"}, decisions.APPROVED), + ({"state": "publishing"}, decisions.PUBLISHING), + ({"state": "awaiting_merge"}, decisions.PUBLISHING), + ({"state": "active"}, decisions.PUBLISHED), + ({"state": "publishing", "last_error": "git push: rejected"}, decisions.FAILED), + ({"state": "approved_publishing", "last_error": "vault is dirty"}, decisions.FAILED), +]) +def test_a_receipt_reports_the_word_an_operator_has_to_act_on(record, expected): + """`failed` reads off `last_error`, not the state: a failed attempt leaves the receipt in + whatever state it was working through, so a status taken from the state alone hides it.""" + assert decisions.release_status(record)["status"] == expected + + +def test_an_active_receipt_that_once_failed_reads_as_published(): + """A receipt that recovered is published. `last_error` is cleared on success, and reporting a + stale error over a completed activation would send someone chasing a resolved failure.""" + assert decisions.release_status( + {"state": "active", "last_error": ""})["status"] == decisions.PUBLISHED + + +def test_no_receipt_is_not_a_status(): + assert decisions.release_status(None) is None + + +# --------------------------------------------------------------------------- pending + +def test_pending_lists_what_is_waiting_with_its_gate_verdict(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + _quarantine(_skill(root, "pdf")) + _quarantine(_skill(root, "tailwind"), promotable=False, blocked=["holdout regression"]) + + result = decisions.pending_view() + + assert [entry["skill"] for entry in result["pending"]] == ["pdf", "tailwind"] + assert result["pending"][0]["promotable"] is True + assert result["pending"][1]["promotable"] is False + assert result["pending"][1]["blocked"] == ["holdout regression"] + + +def test_pending_still_shows_a_change_after_approval_consumes_its_record(tmp_path, monkeypatch): + """Approval consumes the pending record at publication, so the window in which someone is most + likely to ask where their change went is exactly the window the queue cannot see.""" + root = _library(tmp_path, monkeypatch) + _quarantine(_skill(root)) + decisions.approve("pdf", actor="admin") + P.pending_path("pdf").unlink() + + result = decisions.pending_view() + + assert result["pending"] == [] + assert [entry["skill"] for entry in result["publishing"]] == ["pdf"] + assert result["publishing"][0]["status"] == decisions.APPROVED + + +def test_pending_reports_an_empty_queue_rather_than_nothing(tmp_path, monkeypatch, capsys): + _library(tmp_path, monkeypatch) + + assert cli.main(["pending"]) == 0 + assert "Nothing waiting." in capsys.readouterr().out + + +# --------------------------------------------------------------------------- decisions + +def test_approve_queues_a_receipt_and_changes_no_served_byte(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + directory = _skill(root) + _quarantine(directory) + before = (directory / "SKILL.md").read_bytes() + + result = decisions.approve("pdf", actor="admin") + + assert result["publication"]["status"] == decisions.APPROVED + assert result["publication"]["action"] == "promote" + assert (directory / "SKILL.md").read_bytes() == before + assert Q.load_publication(result["publication"]["id"])["state"] == "approved_publishing" + + +def test_approve_refuses_a_change_the_evidence_gate_blocked(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + _quarantine(_skill(root), promotable=False, blocked=["holdout regression"]) + + with pytest.raises(ValueError, match="evidence gate blocked"): + decisions.approve("pdf", actor="admin") + + assert Q.publication_for_skill("pdf") is None + + +def test_approve_refuses_a_second_publication_for_the_same_skill(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + _quarantine(_skill(root)) + P._snapshot_absence("pdf") + decisions.approve("pdf", actor="admin") + + with pytest.raises(ValueError, match="already in progress"): + decisions.rollback("pdf", P.ABSENT_REVISION, actor="admin") + + +def test_reject_discards_the_change_and_records_the_reason(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + directory = _skill(root) + _quarantine(directory) + before = (directory / "SKILL.md").read_bytes() + + result = decisions.reject("pdf", actor="admin", reason=" the body loses the API note ") + + assert not P.pending_path("pdf").exists() + assert (directory / "SKILL.md").read_bytes() == before + assert result["publication"] is None + record = P.read_audit()["records"][0] + assert record["action"] == "reject" and record["actor"] == "admin" + assert record["reason"] == "the body loses the API note" + + +def test_reject_without_a_pending_change_is_a_refusal_not_a_crash(tmp_path, monkeypatch): + _library(tmp_path, monkeypatch) + + with pytest.raises(ValueError, match="no pending change"): + decisions.reject("pdf", actor="admin") + + +def test_rollback_queues_the_snapshot_without_restoring_it(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + directory = _skill(root) + champion = skill_revision(directory) + P._snapshot(directory, "pdf", champion) + _skill(root, body="a later revision") + + result = decisions.rollback("pdf", champion, actor="admin") + + assert result["publication"]["action"] == "rollback" + assert result["publication"]["revision"] == champion + assert "a later revision" in (directory / "SKILL.md").read_text() + + +def test_rollback_to_a_revision_with_no_snapshot_is_refused(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + _skill(root) + + with pytest.raises(ValueError, match="no snapshot"): + decisions.rollback("pdf", "f" * 64, actor="admin") + + +# --------------------------------------------------------------------------- history + +def test_history_gathers_snapshots_receipts_and_decisions_for_one_skill(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + directory = _skill(root) + champion = skill_revision(directory) + P._snapshot(directory, "pdf", champion) + _quarantine(directory) + decisions.approve("pdf", actor="admin") + _quarantine(_skill(root, "tailwind")) + decisions.reject("tailwind", actor="admin", reason="not this one") + + result = decisions.history_view("pdf") + + assert [entry["revision"] for entry in result["revisions"]] == [champion] + assert [entry["status"] for entry in result["publications"]] == [decisions.APPROVED] + assert [record["skill"] for record in result["audit"]] == [] # approval audits on activation + + +def test_history_never_reports_another_skills_decisions(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + _quarantine(_skill(root, "tailwind")) + decisions.reject("tailwind", actor="admin", reason="not this one") + + assert decisions.history_view("pdf")["audit"] == [] + assert decisions.history_view("tailwind")["audit"][0]["reason"] == "not this one" + + +def test_history_refuses_an_invalid_skill_name(tmp_path, monkeypatch): + _library(tmp_path, monkeypatch) + + with pytest.raises(ValueError): + decisions.history_view("../etc") + + +# --------------------------------------------------------------------------- the command line + +def test_the_command_line_reports_the_receipt_and_that_nothing_is_served_yet(tmp_path, monkeypatch, + capsys): + root = _library(tmp_path, monkeypatch) + _quarantine(_skill(root)) + + assert cli.main(["approve", "pdf", "--actor", "admin"]) == 0 + + out = capsys.readouterr().out + assert "Approved 'pdf'" in out + assert "approved" in out + assert "The served library is unchanged" in out + + +def test_a_refusal_exits_one_without_a_traceback(tmp_path, monkeypatch, capsys): + _library(tmp_path, monkeypatch) + + assert cli.main(["approve", "pdf"]) == 1 + assert "ingot approve: no pending challenger" in capsys.readouterr().err + + +def test_every_decision_payload_is_versioned(tmp_path, monkeypatch, capsys): + root = _library(tmp_path, monkeypatch) + _quarantine(_skill(root)) + + assert cli.main(["approve", "pdf", "--actor", "admin", "--json"]) == 0 + + assert json.loads(capsys.readouterr().out)["schema_version"] == decisions.DECISION_SCHEMA + + +def test_an_explicit_actor_outranks_the_environment(tmp_path, monkeypatch): + """A decision nobody can be asked about is not much of an audit trail, so the actor is never + blank: an explicit flag, then INGOT_ACTOR, then whoever is running the command.""" + root = _library(tmp_path, monkeypatch) + _quarantine(_skill(root, "pdf")) + _quarantine(_skill(root, "tailwind")) + monkeypatch.setenv("INGOT_ACTOR", "from-the-environment") + + assert cli.main(["reject", "pdf", "--actor", "reviewer", "--reason", "no"]) == 0 + assert cli.main(["reject", "tailwind", "--reason", "no"]) == 0 + + actors = {record["skill"]: record["actor"] for record in P.read_audit()["records"]} + assert actors == {"pdf": "reviewer", "tailwind": "from-the-environment"} diff --git a/tests/test_delivery.py b/tests/test_delivery.py new file mode 100644 index 0000000..a7e7822 --- /dev/null +++ b/tests/test_delivery.py @@ -0,0 +1,290 @@ +"""Delivery targets: the approved revision reaching more than one place, without a second writer. + +Everything here guards the same boundary. A delivery target is somewhere the publisher installs an +approved revision *after* the vault already carries it -- a native agent's skill root, say. It is +not a second authority: it never decides what is approved, it cannot activate anything, and a +target the publisher fails to write must not leave the release looking finished. The tests below +are the difference between that and a `cp` in a cron job.""" +import json +import os +import stat + +import pytest + +from ingot import delivery +from ingot.mcp_server.registry import skill_revision +from ingot.optimize import promote + +PNG = b"\x89PNG\r\n\x1a\n" + bytes(range(256)) + + +@pytest.fixture +def vault(tmp_path): + root = tmp_path / "vault" + root.mkdir() + return root + + +def _package(root, name="pdf", body="Combine PDFs.", asset=PNG): + """A skill directory with a binary asset and an executable, the two things a text-only copy + would quietly damage.""" + directory = root / name + directory.mkdir(parents=True, exist_ok=True) + (directory / "SKILL.md").write_text( + f"---\nname: {name}\ndescription: Merge and split PDF files.\n---\n\n{body}\n", + encoding="utf-8") + if asset is not None: + (directory / "assets").mkdir(exist_ok=True) + (directory / "assets" / "logo.png").write_bytes(asset) + runner = directory / "run.sh" + runner.write_text("#!/bin/sh\necho hi\n", encoding="utf-8") + runner.chmod(0o755) + return directory + + +# --- configuration ------------------------------------------------------------------------------ + +def test_an_unconfigured_deployment_delivers_to_the_vault_and_nowhere_else(vault): + targets = delivery.load_targets({}, vault=vault) + assert [(target.name, target.kind, target.root) for target in targets] == [ + ("vault", delivery.MANAGED_MCP, vault)] + + +def test_a_filesystem_target_joins_the_vault_rather_than_replacing_it(tmp_path, vault): + native = tmp_path / "claude" / "skills" + targets = delivery.load_targets( + {delivery.TARGETS: f"vault=managed-mcp:{vault},claude=filesystem:{native}"}, vault=vault) + assert [target.name for target in targets] == ["vault", "claude"] + assert targets[1].kind == delivery.FILESYSTEM + assert targets[1].root == native + + +def test_the_vault_is_delivered_to_even_when_the_configuration_forgets_it(tmp_path, vault): + """Dropping the managed target from the list must not stop serving MCP. The vault is where + publication authority lives; a delivery list is not the place to switch it off.""" + native = tmp_path / "native" + targets = delivery.load_targets({delivery.TARGETS: f"claude=filesystem:{native}"}, vault=vault) + assert [(target.name, target.kind) for target in targets] == [ + ("vault", delivery.MANAGED_MCP), ("claude", delivery.FILESYSTEM)] + + +@pytest.mark.parametrize("spec, reason", [ + ("claude=carrier-pigeon:/tmp/x", "unknown delivery kind"), + ("=filesystem:/tmp/x", "invalid delivery target name"), + ("Claude Skills=filesystem:/tmp/x", "invalid delivery target name"), + ("claude=filesystem:relative/path", "must be an absolute path"), + ("claude=filesystem:/tmp/x,claude=filesystem:/tmp/y", "duplicate delivery target"), + ("a=filesystem:/tmp/x,b=filesystem:/tmp/x", "same directory"), + ("claude=filesystem", "expected name=kind:path"), +]) +def test_an_unusable_target_is_refused_at_configuration_time(spec, reason, vault): + """The publisher validates its configuration before it will start. A delivery target that + cannot work has to fail there, not on the first approval that tries to use it.""" + with pytest.raises(ValueError, match=reason): + delivery.load_targets({delivery.TARGETS: spec}, vault=vault) + + +def test_a_managed_target_that_is_not_the_vault_is_refused(tmp_path, vault): + with pytest.raises(ValueError, match="managed-mcp target must be the vault"): + delivery.load_targets({delivery.TARGETS: f"x=managed-mcp:{tmp_path / 'elsewhere'}"}, + vault=vault) + + +def test_a_filesystem_target_inside_the_vault_is_refused(vault): + """It would write into the checkout the publisher just committed, so the vault would drift from + its own release the moment delivery finished.""" + with pytest.raises(ValueError, match="inside the vault"): + delivery.load_targets({delivery.TARGETS: f"x=filesystem:{vault / 'nested'}"}, vault=vault) + + +# --- installing --------------------------------------------------------------------------------- + +def test_the_delivered_target_is_byte_identical_to_the_source(tmp_path, vault): + source = _package(vault) + native = tmp_path / "native" + target = delivery.Target("claude", delivery.FILESYSTEM, native) + + delivery.install(target, "pdf", source, skill_revision(source)) + + delivered = native / "pdf" + assert (delivered / "assets" / "logo.png").read_bytes() == PNG + assert (delivered / "SKILL.md").read_bytes() == (source / "SKILL.md").read_bytes() + assert skill_revision(delivered) == skill_revision(source) + + +def test_the_executable_bit_survives_delivery(tmp_path, vault): + source = _package(vault) + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + + delivery.install(target, "pdf", source, skill_revision(source)) + + mode = (tmp_path / "native" / "pdf" / "run.sh").stat().st_mode + assert mode & stat.S_IXUSR + + +def test_delivering_a_revision_the_target_already_holds_changes_nothing(tmp_path, vault): + source = _package(vault) + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + revision = skill_revision(source) + delivery.install(target, "pdf", source, revision) + before = (tmp_path / "native" / "pdf").stat().st_mtime_ns + + delivery.install(target, "pdf", source, revision) + + assert (tmp_path / "native" / "pdf").stat().st_mtime_ns == before + + +def test_the_displaced_target_is_snapshotted_before_it_is_replaced(tmp_path, vault): + """Whatever was there is recoverable, including bytes that never came from a release: a target + someone edited by hand is exactly the case where the previous content matters most.""" + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + displaced = _package(tmp_path / "old", body="The bytes that were there.") + delivery.install(target, "pdf", displaced, skill_revision(displaced)) + displaced_revision = skill_revision(tmp_path / "native" / "pdf") + + replacement = _package(vault, body="The approved bytes.") + delivery.install(target, "pdf", replacement, skill_revision(replacement)) + + stored = promote.revisions_dir() / "pdf" / displaced_revision + assert (stored / "SKILL.md").read_text(encoding="utf-8").endswith("The bytes that were there.\n") + + +def test_a_delivery_that_cannot_finish_leaves_the_previous_revision_in_place(tmp_path, vault, + monkeypatch): + """The half-written target is the failure this exists to prevent: an agent loading a skill + directory that is neither the old revision nor the new one.""" + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + original = _package(tmp_path / "old", body="The bytes that were there.") + delivery.install(target, "pdf", original, skill_revision(original)) + original_revision = skill_revision(tmp_path / "native" / "pdf") + + replacement = _package(vault, body="The approved bytes.") + monkeypatch.setattr(delivery.os, "replace", _explode) + with pytest.raises(OSError): + delivery.install(target, "pdf", replacement, skill_revision(replacement)) + + assert skill_revision(tmp_path / "native" / "pdf") == original_revision + assert [entry for entry in (tmp_path / "native").iterdir() if entry.name != "pdf"] == [] + + +def _explode(*args, **kwargs): + raise OSError("no space left on device") + + +def test_a_delivery_that_fails_after_displacing_the_target_puts_it_back(tmp_path, vault, + monkeypatch): + """The dangerous window. Replacing a directory takes two renames, and a failure between them + leaves the destination missing unless the displaced copy is moved back.""" + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + original = _package(tmp_path / "old", body="The bytes that were there.") + delivery.install(target, "pdf", original, skill_revision(original)) + original_revision = skill_revision(tmp_path / "native" / "pdf") + + real_replace = os.replace + restored = [] + + def fail_installing_the_staged_copy(source, destination, *args, **kwargs): + # Precisely the second rename of the swap: the displaced original is already out of the + # way and the staged copy is going in. Counting calls would catch the snapshot index write + # instead, which is not the window being tested. + if str(source).endswith(".tmp"): + raise OSError("no space left on device") + if str(source).endswith(".old"): + restored.append(destination) + return real_replace(source, destination, *args, **kwargs) + + replacement = _package(vault, body="The approved bytes.") + monkeypatch.setattr(delivery.os, "replace", fail_installing_the_staged_copy) + with pytest.raises(OSError): + delivery.install(target, "pdf", replacement, skill_revision(replacement)) + + assert restored == [tmp_path / "native" / "pdf"], "the displaced directory was never moved back" + assert skill_revision(tmp_path / "native" / "pdf") == original_revision + assert sorted(entry.name for entry in (tmp_path / "native").iterdir()) == ["pdf"] + + +def test_delivering_bytes_that_do_not_match_the_receipt_is_refused(tmp_path, vault): + """The check is against the revision the receipt names, not against the source directory, so a + source that was tampered with between approval and delivery cannot install itself.""" + source = _package(vault) + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + + with pytest.raises(ValueError, match="does not match the approved revision"): + delivery.install(target, "pdf", source, "0" * 16) + + assert not (tmp_path / "native" / "pdf").exists() + + +def test_a_symlink_in_the_source_is_refused_rather_than_followed(tmp_path, vault): + """Following it would copy whatever it points at -- possibly from outside the vault entirely -- + into an agent's skill root as an ordinary file.""" + source = _package(vault) + # A link pointing *out* of the skill is already refused upstream: `skill_revision` will not + # hash one. A contained link is the case that reaches here, and copying it without `symlinks` + # silently turns it into a second regular file holding the same bytes. + (source / "readme.md").symlink_to(source / "SKILL.md") + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + + with pytest.raises(ValueError, match="symlink-unsupported"): + delivery.install(target, "pdf", source, skill_revision(source)) + + assert not (tmp_path / "native" / "pdf").exists() + + +def test_delivering_an_absence_removes_the_skill_and_snapshots_it(tmp_path, vault): + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + source = _package(vault) + delivery.install(target, "pdf", source, skill_revision(source)) + revision = skill_revision(source) + + delivery.install(target, "pdf", None, promote.ABSENT_REVISION) + + assert not (tmp_path / "native" / "pdf").exists() + assert (promote.revisions_dir() / "pdf" / revision).is_dir() + + +def test_removing_a_skill_the_target_never_held_is_not_an_error(tmp_path, vault): + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + delivery.install(target, "pdf", None, promote.ABSENT_REVISION) + assert not (tmp_path / "native" / "pdf").exists() + + +def test_delivery_never_writes_a_managed_target(tmp_path, vault): + """The vault is written by the publication commit and the fast-forward, and by nothing else. + A second writer there is the invariant this whole control plane exists to hold.""" + _package(vault, body="What the vault holds.") + # Deliberately a *different* revision. Asking the managed target to install bytes it does not + # hold is the only way to tell the refusal apart from delivery being a no-op by luck. + elsewhere = _package(tmp_path / "other", body="Something else entirely.") + target = delivery.Target("vault", delivery.MANAGED_MCP, vault) + before = skill_revision(vault / "pdf") + + assert delivery.install(target, "pdf", elsewhere, skill_revision(elsewhere)) is False + + assert skill_revision(vault / "pdf") == before + assert sorted(path.name for path in vault.iterdir()) == ["pdf"] + + +# --- observing ---------------------------------------------------------------------------------- + +def test_a_target_reports_the_revision_it_actually_holds(tmp_path, vault): + source = _package(vault) + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + delivery.install(target, "pdf", source, skill_revision(source)) + + assert delivery.observed(target, "pdf") == skill_revision(source) + + +def test_a_target_edited_out_of_band_reports_the_new_bytes_not_the_old_ones(tmp_path, vault): + source = _package(vault) + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + delivery.install(target, "pdf", source, skill_revision(source)) + + (tmp_path / "native" / "pdf" / "assets" / "logo.png").write_bytes(PNG + b"tampered") + + assert delivery.observed(target, "pdf") != skill_revision(source) + + +def test_a_missing_skill_reads_as_absent_rather_than_as_an_error(tmp_path, vault): + target = delivery.Target("claude", delivery.FILESYSTEM, tmp_path / "native") + assert delivery.observed(target, "pdf") == promote.ABSENT_REVISION diff --git a/tests/test_distribution.py b/tests/test_distribution.py new file mode 100644 index 0000000..2352498 --- /dev/null +++ b/tests/test_distribution.py @@ -0,0 +1,69 @@ +"""What the repository ships. + +A release controller must not arrive with an unexplained pending change already in its queue, and +must not ship a served skill nobody approved. These read what git tracks rather than what happens +to be in one working tree, because that is what a clone gets.""" +import subprocess +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parent.parent + + +def _tracked(pathspec: str) -> list[str]: + result = subprocess.run(["git", "-C", str(ROOT), "ls-files", "--", pathspec], + capture_output=True, text=True) + if result.returncode: + pytest.skip("not a git checkout") + return [line for line in result.stdout.splitlines() if line] + + +def test_a_clean_checkout_begins_with_no_live_pending_proposal(): + """A tracked pending record would arrive as a real quarantined change in every clone: it would + show in `ingot pending`, be approvable, and publish something nobody proposed.""" + assert _tracked("runs") == [] + + +def test_a_clean_checkout_serves_no_skill_it_did_not_publish(): + """The demo library ships empty. A tracked skill would be served with no release receipt behind + it, which is precisely the UNMANAGED state `ingot status` exists to report.""" + assert _tracked("skills") == ["skills/.gitkeep"] + + +def test_the_vault_is_never_tracked(): + """`ingot vault init` makes it a Git repository of its own; a tracked copy would be a second, + stale answer to what is published.""" + assert _tracked("vault") == [] + + +@pytest.mark.parametrize("pathspec", ["runs", "skills/*/", "vault"]) +def test_state_paths_are_ignored_so_a_live_deployment_cannot_commit_itself(pathspec): + """A checkout used as a deployment writes into these. Without ignore rules the first `git add` + would commit a review queue and a set of receipts into the product.""" + probe = {"runs": "runs/pending/probe.json", + "skills/*/": "skills/probe/SKILL.md", + "vault": "vault/registry.json"}[pathspec] + result = subprocess.run(["git", "-c", f"safe.directory={ROOT}", "-C", str(ROOT), + "check-ignore", "-q", probe], + capture_output=True) + + assert result.returncode == 0, f"{probe} is not ignored" + + +def test_the_package_claims_one_installed_name(): + """`mcp_server` and `optimize` are far too generic to own on PyPI. They now live under `ingot`, + and a shim would have been self-defeating: a shim named `optimize` still claims `optimize`.""" + import tomllib + + configured = tomllib.loads((ROOT / "pyproject.toml").read_text()) + packages = configured["tool"]["setuptools"]["packages"] + + assert [name for name in packages if not name.startswith("ingot")] == [] + + +@pytest.mark.parametrize("name", ["mcp_server", "optimize"]) +def test_no_generic_package_reappears_at_the_repository_root(name): + """A stale `build/` directory from before the move will happily reinstall the old top-level + packages, so this checks what git tracks rather than what happens to be on disk.""" + assert _tracked(name) == [] diff --git a/tests/test_draft.py b/tests/test_draft.py index 83d7caa..d3c4581 100644 --- a/tests/test_draft.py +++ b/tests/test_draft.py @@ -1,9 +1,9 @@ -"""Unit tests for auto-drafted eval task sets (optimize.draft), LLM mocked.""" +"""Unit tests for auto-drafted eval task sets (ingot.optimize.draft), LLM mocked.""" import json import pytest -from optimize import draft as D +from ingot.optimize import draft as D class _FakeMsg: @@ -41,6 +41,54 @@ def test_draft_raises_if_too_few_usable_tasks(monkeypatch): D.draft_tasks("pdf", "d", "b", n=8) +def test_draft_carries_a_weighted_checklist_per_task(monkeypatch): + """The checklist is what gives a task more than one measurement, so a drafted set has to carry + it -- otherwise only hand-written tasks ever get graded finely.""" + check = {"id": "handles_empty", "criterion": "Handles an empty input file without raising.", + "weight": 4, "dimension": "completeness"} + _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r", "checklist": [check]} for i in range(4)]) + out = D.draft_tasks("pdf", "d", "b", n=4) + assert out["train"][0]["checklist"] == [check] + + +@pytest.mark.parametrize("bad, why", [ + ({"id": "Has Spaces", "criterion": "a valid criterion here"}, "id is not snake_case"), + ({"id": "ok_id", "criterion": "short"}, "criterion too short to grade"), + ("not a dict", "not an object"), +]) +def test_draft_drops_ungradeable_checks(monkeypatch, bad, why): + good = {"id": "keeps_this", "criterion": "A criterion long enough to grade.", "weight": 2, + "dimension": "correctness"} + _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r", "checklist": [good, bad]} + for i in range(4)]) + out = D.draft_tasks("pdf", "d", "b", n=4) + assert [c["id"] for c in out["train"][0]["checklist"]] == ["keeps_this"], why + + +def test_draft_deduplicates_check_ids(monkeypatch): + """Two checks with one id would collapse in the judge's JSON response, silently dropping a + check while its weight still counted against the total.""" + dup = [{"id": "same", "criterion": "The first criterion text."}, + {"id": "same", "criterion": "A different criterion, same id."}] + _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r", "checklist": dup} for i in range(4)]) + assert len(D.draft_tasks("pdf", "d", "b", n=4)["train"][0]["checklist"]) == 1 + + +def test_draft_clamps_weights_and_defaults_unknown_dimensions(monkeypatch): + wild = {"id": "wild", "criterion": "A criterion long enough to grade.", "weight": 99, + "dimension": "vibes"} + _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r", "checklist": [wild]} for i in range(4)]) + got = D.draft_tasks("pdf", "d", "b", n=4)["train"][0]["checklist"][0] + assert got["weight"] == 5 and got["dimension"] == "correctness" + + +def test_draft_tolerates_a_task_with_no_checklist(monkeypatch): + """judge() falls back to its default checklist, so a missing one degrades to four dimensions + rather than failing the draft.""" + _mock_llm(monkeypatch, [{"task": f"t{i}", "rubric": "r"} for i in range(4)]) + assert D.draft_tasks("pdf", "d", "b", n=4)["train"][0]["checklist"] == [] + + def _mock_routing_llm(monkeypatch, positive, negative): payload = json.dumps({"positive": positive, "negative": negative}) monkeypatch.setattr(D, "_llm", lambda: type("L", (), {"invoke": lambda self, p: _FakeMsg(payload)})()) diff --git a/tests/test_embedding.py b/tests/test_embedding.py index 9c04c00..93ec32f 100644 --- a/tests/test_embedding.py +++ b/tests/test_embedding.py @@ -1,11 +1,13 @@ """Embedding backends: model selection, and the Qwen ONNX pooling/prefix contract with a stubbed session (no model download, no onnxruntime inference).""" +import json import sys from types import SimpleNamespace +from urllib.error import URLError import numpy as np -from mcp_server import embedding as E +from ingot.mcp_server import embedding as E def test_backend_selection_by_model_name(): @@ -144,3 +146,70 @@ def test_qwen_tolerates_empty_text(): def test_fastembed_backend_has_no_query_prefix(): # symmetric embedding is the pre-Qwen contract fastembed overrides rely on assert E.FastembedEmbedding.embed_query is E.FastembedEmbedding.embed + + +class _Response: + def __init__(self, payload): + self._payload = json.dumps(payload).encode() + + def __enter__(self): + return self + + def __exit__(self, *_args): + pass + + def read(self): + return self._payload + + +def test_remote_qwen_backend_batches_and_prefixes_only_queries(monkeypatch): + requests = [] + + def open_request(request, timeout): + requests.append((request, timeout)) + body = json.loads(request.data) + return _Response({ + "data": [ + {"index": index, "embedding": [float(index), 1.0]} + for index, _ in reversed(list(enumerate(body["input"]))) + ] + }) + + monkeypatch.setattr(E, "urlopen", open_request) + backend = E.RemoteQwenEmbedding( + "Qwen/Qwen3-Embedding-8B-GGUF", "http://embed-gpu:8080/v1", timeout=7) + + documents = backend.embed(["first", "second"]) + queries = backend.embed_query(["route this"]) + + assert [vector.tolist() for vector in documents] == [[0.0, 1.0], [1.0, 1.0]] + assert queries[0].tolist() == [0.0, 1.0] + assert requests[0][1] == 7 + assert requests[0][0].full_url == "http://embed-gpu:8080/v1/embeddings" + assert json.loads(requests[0][0].data)["input"] == ["first", "second"] + assert json.loads(requests[1][0].data)["input"] == [E.QUERY_PREFIX + "route this"] + assert backend.identity == "remote:Qwen/Qwen3-Embedding-8B-GGUF@http://embed-gpu:8080/v1" + + +def test_remote_backend_reports_unavailable_server(monkeypatch): + monkeypatch.setattr(E, "urlopen", lambda *_args, **_kwargs: (_ for _ in ()).throw( + URLError("connection refused"))) + backend = E.RemoteQwenEmbedding("model", "http://embed-gpu:8080/v1") + + with np.testing.assert_raises_regex(RuntimeError, "embedding server unavailable"): + backend.embed(["hello"]) + + +def test_build_embedding_requires_remote_url(monkeypatch): + monkeypatch.setenv("EMBED_BACKEND", "remote") + monkeypatch.delenv("EMBED_BASE_URL", raising=False) + with np.testing.assert_raises_regex(ValueError, "EMBED_BASE_URL"): + E.build_embedding() + + +def test_build_embedding_requires_distinct_remote_model(monkeypatch): + monkeypatch.setenv("EMBED_BACKEND", "remote") + monkeypatch.setenv("EMBED_BASE_URL", "http://embed-gpu:8080/v1") + monkeypatch.delenv("EMBED_REMOTE_MODEL", raising=False) + with np.testing.assert_raises_regex(ValueError, "EMBED_REMOTE_MODEL"): + E.build_embedding() diff --git a/tests/test_env_check.py b/tests/test_env_check.py index b4e0f90..8765a26 100644 --- a/tests/test_env_check.py +++ b/tests/test_env_check.py @@ -6,7 +6,7 @@ import pytest -from optimize import require_openrouter_key +from ingot.optimize import require_openrouter_key ROOT = Path(__file__).resolve().parent.parent @@ -35,7 +35,7 @@ def test_set_key_passes(monkeypatch): def test_fully_local_setup_needs_no_key(monkeypatch): - from optimize import openrouter_key_missing, require_openrouter_key + from ingot.optimize import openrouter_key_missing, require_openrouter_key monkeypatch.delenv("OPENROUTER_API_KEY", raising=False) monkeypatch.setenv("OPENROUTER_BASE_URL", "http://localhost:11434/v1") # e.g. Ollama monkeypatch.delenv("MODEL_BASE_URL", raising=False) @@ -45,7 +45,7 @@ def test_fully_local_setup_needs_no_key(monkeypatch): def test_local_model_but_openrouter_teacher_still_needs_key(monkeypatch): import pytest - from optimize import require_openrouter_key + from ingot.optimize import require_openrouter_key monkeypatch.delenv("OPENROUTER_API_KEY", raising=False) monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False) # teacher on OpenRouter monkeypatch.setenv("MODEL_BASE_URL", "http://localhost:8000/v1") # agent on local vLLM @@ -54,7 +54,7 @@ def test_local_model_but_openrouter_teacher_still_needs_key(monkeypatch): def test_client_kwargs_openrouter_gets_zdr_local_does_not(monkeypatch): - from optimize import ZDR_PROVIDER, client_kwargs + from ingot.optimize import ZDR_PROVIDER, client_kwargs monkeypatch.setenv("OPENROUTER_API_KEY", "sk-or-test") kw = client_kwargs("https://openrouter.ai/api/v1") assert kw == {"base_url": "https://openrouter.ai/api/v1", "api_key": "sk-or-test", @@ -66,7 +66,7 @@ def test_client_kwargs_openrouter_gets_zdr_local_does_not(monkeypatch): def test_model_base_url_overrides_only_the_serving_role(monkeypatch): - from optimize import model_base_url, teacher_base_url + from ingot.optimize import model_base_url, teacher_base_url monkeypatch.delenv("MODEL_BASE_URL", raising=False) monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False) assert model_base_url() == teacher_base_url() == "https://openrouter.ai/api/v1" @@ -76,7 +76,7 @@ def test_model_base_url_overrides_only_the_serving_role(monkeypatch): def test_provider_priority_composes_with_zdr(monkeypatch): - from optimize import ZDR_PROVIDER, client_kwargs, openrouter_extra_body + from ingot.optimize import ZDR_PROVIDER, client_kwargs, openrouter_extra_body monkeypatch.delenv("OPENROUTER_PROVIDERS", raising=False) assert openrouter_extra_body() == ZDR_PROVIDER # default: ZDR only, no pin monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq, deepinfra") @@ -106,7 +106,7 @@ def __exit__(self, *a): def test_provider_conflict_loose_name_matching(monkeypatch): import urllib.request - from optimize import provider_conflict + from ingot.optimize import provider_conflict monkeypatch.setattr(urllib.request, "urlopen", lambda url, timeout=10: _FakeEndpoints(["DeepInfra", "Io Net"])) assert provider_conflict("qwen/x", ["deep-infra"]) is None # display-name vs slug @@ -118,7 +118,7 @@ def test_provider_conflict_loose_name_matching(monkeypatch): def test_provider_conflict_unknown_model_and_network_failure(monkeypatch): import urllib.request - from optimize import provider_conflict + from ingot.optimize import provider_conflict monkeypatch.setattr(urllib.request, "urlopen", lambda url, timeout=10: _FakeEndpoints([])) assert "no endpoints on OpenRouter" in provider_conflict("qwen/typo-27b", ["groq"]) def boom(url, timeout=10): @@ -129,7 +129,7 @@ def boom(url, timeout=10): def test_preflight_no_pins_makes_no_network_calls(monkeypatch): import urllib.request - from optimize import preflight_provider_pins + from ingot.optimize import preflight_provider_pins monkeypatch.delenv("OPENROUTER_PROVIDERS", raising=False) def forbidden(url, timeout=10): raise AssertionError("network call without pins") @@ -138,7 +138,7 @@ def forbidden(url, timeout=10): def test_preflight_warns_on_every_uncovered_role(monkeypatch): - import optimize + import ingot.optimize as optimize monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq") monkeypatch.delenv("MODEL_BASE_URL", raising=False) monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False) @@ -150,7 +150,7 @@ def test_preflight_warns_on_every_uncovered_role(monkeypatch): def test_preflight_reports_agent_model_alias_value(monkeypatch): # the pin check must validate the model the agent will actually use, whichever alias set it - import optimize + import ingot.optimize as optimize monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq") for var in ("MODEL_BASE_URL", "BASE_URL", "OPENROUTER_BASE_URL", "MODEL"): monkeypatch.delenv(var, raising=False) @@ -162,7 +162,7 @@ def test_preflight_reports_agent_model_alias_value(monkeypatch): def test_agent_model_resolution(monkeypatch): # AGENT_MODEL wins; MODEL is the legacy alias; then the literal default - from optimize import agent_model + from ingot.optimize import agent_model monkeypatch.delenv("AGENT_MODEL", raising=False) monkeypatch.delenv("MODEL", raising=False) assert agent_model() == "qwen/qwen3-32b" @@ -173,7 +173,7 @@ def test_agent_model_resolution(monkeypatch): def test_skillopt_model_resolution(monkeypatch): - from optimize import skillopt_model + from ingot.optimize import skillopt_model monkeypatch.delenv("SKILLOPT_MODEL", raising=False) assert skillopt_model() == "z-ai/glm-5.2" monkeypatch.setenv("SKILLOPT_MODEL", "author/model") @@ -186,9 +186,9 @@ def test_skillopt_model_reaches_every_authoring_role(): env = {**os.environ, "SKILLOPT_MODEL": "author/model", "JUDGE_MODEL": "different/judge"} code = """ -import optimize.draft as draft -import optimize.rollout as rollout -from optimize.usage import _role_models +import ingot.optimize.draft as draft +import ingot.optimize.rollout as rollout +from ingot.optimize.usage import _role_models assert draft.MODEL == 'author/model' assert rollout.SKILLOPT_MODEL == 'author/model' assert _role_models()['reflection'] == 'author/model' @@ -201,14 +201,37 @@ def test_judge_warns_when_skillopt_model_is_the_grader(): env = {**os.environ, "SKILLOPT_MODEL": "same/model", "JUDGE_MODEL": "same/model"} env.pop("JUDGE_MODELS", None) - result = subprocess.run([sys.executable, "-c", "import optimize.judge"], env=env, + result = subprocess.run([sys.executable, "-c", "import ingot.optimize.judge"], env=env, cwd=ROOT, text=True, capture_output=True, check=True) assert "author == grader" in result.stdout +def test_duplicate_judge_models_fail_closed_before_spend(): + env = {**os.environ, "JUDGE_MODELS": "alpha/model,alpha/model"} + result = subprocess.run([sys.executable, "-c", "import ingot.optimize.judge"], env=env, + cwd=ROOT, text=True, capture_output=True) + assert result.returncode != 0 + assert "JUDGE_MODELS contains duplicate model 'alpha/model'" in result.stderr + + +def test_blank_judge_ensemble_falls_back_to_single_judge(): + env = {**os.environ, "JUDGE_MODELS": " , ", "JUDGE_MODEL": "backup/model"} + result = subprocess.run( + [sys.executable, "-c", "import ingot.optimize.judge as j; print(j.MODELS)"], env=env, + cwd=ROOT, text=True, capture_output=True, check=True) + assert result.stdout.strip() == "['backup/model']" + + +def test_duplicate_compat_models_fail_closed_before_spend(monkeypatch): + from ingot.optimize.compat import compat_models + monkeypatch.setenv("COMPAT_MODELS", "alpha/model,alpha/model") + with pytest.raises(SystemExit, match="COMPAT_MODELS contains duplicate model 'alpha/model'"): + compat_models() + + def test_preflight_checks_strong_model_only_when_explicitly_set(monkeypatch): - import optimize + import ingot.optimize as optimize monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq") monkeypatch.delenv("MODEL_BASE_URL", raising=False) monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False) @@ -222,7 +245,7 @@ def test_preflight_checks_strong_model_only_when_explicitly_set(monkeypatch): def test_preflight_skips_roles_on_local_endpoints(monkeypatch): - import optimize + import ingot.optimize as optimize monkeypatch.setenv("OPENROUTER_PROVIDERS", "groq") monkeypatch.setenv("OPENROUTER_BASE_URL", "http://localhost:11434/v1") # fully local monkeypatch.setattr(optimize, "provider_conflict", @@ -233,7 +256,7 @@ def test_preflight_skips_roles_on_local_endpoints(monkeypatch): def test_invoke_retry_fails_fast_on_permanent_config_errors(monkeypatch): import pytest - from optimize import judge as judge_mod + from ingot.optimize import judge as judge_mod monkeypatch.setattr(judge_mod.time, "sleep", lambda s: (_ for _ in ()).throw(AssertionError("slept"))) class Doomed: @@ -252,7 +275,7 @@ def invoke(self, messages): def test_invoke_retry_still_retries_transient_errors(monkeypatch): - from optimize import judge as judge_mod + from ingot.optimize import judge as judge_mod monkeypatch.setattr(judge_mod.time, "sleep", lambda s: None) class Flaky: @@ -267,7 +290,7 @@ def invoke(self, messages): def test_generic_base_url_and_api_key_win_with_legacy_fallback(monkeypatch): - from optimize import api_key, model_api_key, teacher_base_url + from ingot.optimize import api_key, model_api_key, teacher_base_url monkeypatch.delenv("BASE_URL", raising=False) monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False) monkeypatch.setenv("OPENROUTER_API_KEY", "sk-or-legacy") @@ -284,7 +307,7 @@ def test_generic_base_url_and_api_key_win_with_legacy_fallback(monkeypatch): def test_hosted_https_endpoint_requires_a_key(monkeypatch): - from optimize import openrouter_key_missing + from ingot.optimize import openrouter_key_missing for var in ("API_KEY", "OPENROUTER_API_KEY", "MODEL_API_KEY", "MODEL_BASE_URL"): monkeypatch.delenv(var, raising=False) monkeypatch.setenv("BASE_URL", "https://openrouter.ai/api/v1") @@ -294,7 +317,7 @@ def test_hosted_https_endpoint_requires_a_key(monkeypatch): def test_local_endpoint_gets_clean_openai_request(monkeypatch): - from optimize import client_kwargs + from ingot.optimize import client_kwargs kw = client_kwargs("http://ollama:11434/v1") assert kw == {"base_url": "http://ollama:11434/v1", "api_key": "local", "extra_body": {}} # no OpenRouter provider prefs diff --git a/tests/test_evidence.py b/tests/test_evidence.py index 37965f8..368c085 100644 --- a/tests/test_evidence.py +++ b/tests/test_evidence.py @@ -1,6 +1,6 @@ import json -from optimize.evidence import build_evidence, first_divergence, render_markdown, write_evidence +from ingot.optimize.evidence import build_evidence, first_divergence, render_markdown, write_evidence SUMMARY = { @@ -81,7 +81,7 @@ def test_markdown_surfaces_gate_warnings(): def _routing_evidence(gate=None, parity=True): - from optimize.evidence import RoutingRun, build_routing_evidence + from ingot.optimize.evidence import RoutingRun, build_routing_evidence metrics = dict(ROUTING_METRICS) if not parity: metrics["parity"] = {"rate": 0.0, "total": 0} @@ -93,7 +93,7 @@ def _routing_evidence(gate=None, parity=True): def test_routing_evidence_carries_revisions_and_router_metrics(): - from optimize.evidence import ROUTING_SCHEMA + from ingot.optimize.evidence import ROUTING_SCHEMA evidence = _routing_evidence() assert evidence["schema_version"] == ROUTING_SCHEMA assert evidence["champion"]["revision"] == "champ-rev" @@ -103,7 +103,7 @@ def test_routing_evidence_carries_revisions_and_router_metrics(): def test_write_evidence_renders_the_routing_report(tmp_path): - from optimize.evidence import render_routing_markdown + from ingot.optimize.evidence import render_routing_markdown evidence = _routing_evidence() json_path, md_path = write_evidence(evidence, tmp_path) assert json.loads(json_path.read_text()) == evidence @@ -117,7 +117,7 @@ def test_write_evidence_renders_the_routing_report(tmp_path): def test_routing_report_marks_a_blocked_gate_and_unexercised_parity(): - from optimize.evidence import render_routing_markdown + from ingot.optimize.evidence import render_routing_markdown blocked = {"promotable": False, "blocked": ["routing top1 regressed"], "warnings": []} text = render_routing_markdown(_routing_evidence(gate=blocked, parity=False)) assert "BLOCKED" in text @@ -125,11 +125,27 @@ def test_routing_report_marks_a_blocked_gate_and_unexercised_parity(): assert "Cross-harness parity: not exercised" in text -def test_recorded_path_is_repo_relative_and_leaves_outside_paths_alone(tmp_path): +def test_recorded_path_is_state_relative_and_leaves_outside_paths_alone(tmp_path, monkeypatch): + """A bundle written inside a container has to be resolvable from the host, so the location is + recorded relative to the state root rather than absolutely.""" from pathlib import Path - from optimize.evidence import _REPO_ROOT, recorded_path - inside = _REPO_ROOT / "runs" / "evidence" / "pdf" / "1" / "EVIDENCE.md" + from ingot import paths + from ingot.optimize.evidence import recorded_path + monkeypatch.setenv("INGOT_HOME", str(tmp_path)) + inside = paths.runs() / "evidence" / "pdf" / "1" / "EVIDENCE.md" assert recorded_path(inside) == "runs/evidence/pdf/1/EVIDENCE.md" outside = Path("/somewhere/else/EVIDENCE.md") assert recorded_path(outside) == "/somewhere/else/EVIDENCE.md" + + +def test_a_bundle_recorded_before_state_moved_out_of_the_package_still_reads_as_relative( + tmp_path, monkeypatch): + """Records written when `runs/` lived beside the code carry `runs/evidence/...`. Falling back + to an absolute path from whoever's machine wrote it would make them unresolvable.""" + from ingot.optimize.evidence import _PACKAGE_ROOT, recorded_path + monkeypatch.setenv("INGOT_HOME", str(tmp_path)) + + legacy = _PACKAGE_ROOT / "runs" / "evidence" / "pdf" / "1" / "EVIDENCE.md" + + assert recorded_path(legacy) == "runs/evidence/pdf/1/EVIDENCE.md" diff --git a/tests/test_execcheck.py b/tests/test_execcheck.py index e0aa13b..fbdb9f3 100644 --- a/tests/test_execcheck.py +++ b/tests/test_execcheck.py @@ -1,9 +1,9 @@ -"""Unit tests for execution-based code validation (optimize.execcheck), static path (no EXEC_SANDBOX).""" +"""Unit tests for execution-based code validation (ingot.optimize.execcheck), static path (no EXEC_SANDBOX).""" import os import subprocess import sys -from optimize import execcheck as E +from ingot.optimize import execcheck as E def test_expects_code_gates_on_task_shape(): @@ -79,7 +79,7 @@ def test_exec_sandbox_env_modes(): # fresh-import subprocesses exercise the real env surface: the sandbox is the DEFAULT, # "1" is the legacy bare opt-in, anything else turns execution off entirely base = {k: v for k, v in os.environ.items() if k != "EXEC_SANDBOX"} - probe = "from optimize.execcheck import EXEC_MODE, EXEC_SANDBOX, check; " + probe = "from ingot.optimize.execcheck import EXEC_MODE, EXEC_SANDBOX, check; " default = subprocess.run([sys.executable, "-c", probe + "print(EXEC_MODE)"], capture_output=True, text=True, env=base) assert default.stdout.strip() == "docker" @@ -209,12 +209,13 @@ def test_judge_note_execution_verdicts(monkeypatch): def test_judge_threads_check_spec_into_the_prompt(monkeypatch): monkeypatch.setattr(E, "EXEC_MODE", "1") monkeypatch.setattr(E, "EXEC_SANDBOX", True) # legacy bare path - from optimize import judge as judge_mod + from ingot.optimize import judge as judge_mod seen = {} monkeypatch.setattr(judge_mod, "MODELS", ["m"]) - def capture(model, prompt): + def capture(model, prompt, checklist): seen["prompt"] = prompt - return {"score": 1.0, "feedback": "f", "dimensions": {d: "pass" for d in judge_mod.DIMENSIONS}} + return {"items": {i["id"]: {"value": 1.0, "note": ""} for i in checklist}, + "feedback": "f", "unparseable": False} monkeypatch.setattr(judge_mod, "_judge_one", capture) judge_mod.judge("t", "r", ANSWER_OK, check=CHECK) assert "EXECUTION CHECK, PASSED" in seen["prompt"] @@ -250,7 +251,7 @@ def test_docker_sandbox_container_is_actually_locked_down(monkeypatch): assert " ".join(flag) in joined, f"missing {flag} in {cmd}" assert "--pids-limit" in cmd and "--memory" in cmd assert "-v" not in cmd and "--volume" not in cmd # nothing mounted in - assert cmd[-4:] == [E.SANDBOX_IMAGE, "python", "-m", "optimize.sandbox_driver"] + assert cmd[-4:] == [E.SANDBOX_IMAGE, "python", "-m", "ingot.optimize.sandbox_driver"] def test_docker_sandbox_runtime_flag_enables_gvisor(monkeypatch): @@ -315,12 +316,12 @@ def test_sandbox_driver_end_to_end_without_docker(tmp_path): import json as _json spec = {"fixture": CHECK["fixture"], "code": 'text = open("input.txt").read()\n' 'open("output.txt", "w").write(text.upper())', "assertion": CHECK["assert"]} - run = subprocess.run([sys.executable, "-m", "optimize.sandbox_driver"], + run = subprocess.run([sys.executable, "-m", "ingot.optimize.sandbox_driver"], input=_json.dumps(spec), capture_output=True, text=True, cwd=str(tmp_path), env={**os.environ, "PYTHONPATH": os.getcwd()}) assert _json.loads(run.stdout) == {"ok": True} bad = {**spec, "code": spec["code"].replace(".upper()", ".lower()")} - run = subprocess.run([sys.executable, "-m", "optimize.sandbox_driver"], + run = subprocess.run([sys.executable, "-m", "ingot.optimize.sandbox_driver"], input=_json.dumps(bad), capture_output=True, text=True, cwd=str(tmp_path), env={**os.environ, "PYTHONPATH": os.getcwd()}) verdict = _json.loads(run.stdout) diff --git a/tests/test_harbor_catalog.py b/tests/test_harbor_catalog.py new file mode 100644 index 0000000..f1b4725 --- /dev/null +++ b/tests/test_harbor_catalog.py @@ -0,0 +1,361 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +import ingot.optimize.harbor_catalog as HC +from ingot.optimize.harbor_catalog import CatalogIntent, enqueue_catalog, run_catalog + +_REAL_PREPARE_EXECUTION = HC._prepare_execution +_REAL_INTENT_FOR_SKILL = HC._intent_for_skill + + +def _intent(skill: str = "demo", sha: str = "a" * 64, *, priority: int = 100, + publish_root: str = "/srv/ingot/runs/harbor") -> CatalogIntent: + return CatalogIntent( + skill=skill, + skill_sha256=sha, + task_fingerprint="b" * 64, + target_specs=("dell-qwen=http://127.0.0.1:8011",), + target_fingerprints=("434045372e7c",), + harnesses=("aider", "pi"), + runtime_revisions=(("harbor", "0.20.0"), ("runner", "native-v1")), + publish_root=publish_root, + priority=priority, + ) + + +def _read(path: Path) -> dict: + return json.loads(path.read_text()) + + +@pytest.fixture(autouse=True) +def _catalog_runtime(tmp_path, monkeypatch): + monkeypatch.setattr(HC, "CATALOG_OWNER", tmp_path / "global-controller.lock") + monkeypatch.setattr(HC, "_prepare_execution", + lambda root, intent: (root / "source", root / "runs" / intent.digest)) + monkeypatch.setattr(HC, "_intent_for_skill", + lambda skill, targets, harnesses, priority=100, **_kwargs: + _intent(skill, priority=priority)) + monkeypatch.setattr(HC, "_refuse_live_harbor", lambda _root: None) + + +def test_enqueue_uses_content_identity_and_leaves_compatible_completion_untouched(tmp_path): + intent = _intent() + [path] = enqueue_catalog(tmp_path, [intent]) + assert path.name == f"{intent.digest}.json" + assert _read(path)["identity"] == intent.identity_payload() + state = tmp_path / "state" / path.name + state.write_text(json.dumps({"schema": 1, "status": "complete", "priority": 100, + "intent_digest": intent.digest, "marker": "keep"})) + + [again] = enqueue_catalog(tmp_path, [intent]) + + assert again == path + assert _read(state)["marker"] == "keep" + + +def test_enqueue_prioritizes_changed_then_incomplete_current_revision(tmp_path): + old = _intent(sha="a" * 64) + [old_path] = enqueue_catalog(tmp_path, [old]) + old_state = tmp_path / "state" / old_path.name + old_state.write_text(json.dumps({"schema": 1, "status": "complete", "priority": 100, + "intent_digest": old.digest})) + + changed = _intent(sha="c" * 64) + [changed_path] = enqueue_catalog(tmp_path, [changed]) + assert _read(tmp_path / "state" / changed_path.name)["priority"] == 300 + + current = _intent(skill="other") + [current_path] = enqueue_catalog(tmp_path, [current]) + current_state = tmp_path / "state" / current_path.name + state = _read(current_state) + state["status"] = "running" + current_state.write_text(json.dumps(state)) + enqueue_catalog(tmp_path, [current]) + assert _read(current_state)["priority"] == 200 + + +def test_enqueue_reactivates_a_superseded_identity_that_becomes_current_again(tmp_path): + intent = _intent() + [intent_path] = enqueue_catalog(tmp_path, [intent]) + state_path = tmp_path / "state" / intent_path.name + state = _read(state_path) + state.update(status="superseded", error="IdentityChanged", finished_at=123.0) + state_path.write_text(json.dumps(state)) + + enqueue_catalog(tmp_path, [intent]) + + assert _read(state_path) == { + "schema": 1, + "intent_digest": intent.digest, + "status": "pending", + "priority": 200, + } + + +def test_stop_file_prevents_runner_and_preserves_pending(tmp_path): + [path] = enqueue_catalog(tmp_path, [_intent()]) + stop = tmp_path / "STOP" + stop.touch() + + monkeypatch = pytest.MonkeyPatch() + monkeypatch.setattr(HC, "run_local_sweep", + lambda *_args, **_kwargs: pytest.fail("runner called")) + run_catalog(tmp_path, stop_file=stop) + monkeypatch.undo() + + assert _read(tmp_path / "state" / path.name)["status"] == "pending" + + +def test_catalog_resumes_running_intent_and_advances_without_recreating_completion(tmp_path, + monkeypatch): + first, second = _intent("first", priority=200), _intent("second", priority=100) + first_path, second_path = enqueue_catalog(tmp_path, [first, second]) + first_state = tmp_path / "state" / first_path.name + state = _read(first_state) + state.update(status="running", run_root="same-root") + first_state.write_text(json.dumps(state)) + calls = [] + + def runner(skill, targets, **kwargs): + calls.append((skill, tuple(target.alias for target in targets), kwargs["native_parallel"], + kwargs["evidence_root"], kwargs["publish_root"], + kwargs["content_addressed_resume"])) + return {"skill": skill, "aborted": False, "combinations": {"one": {"lift": 0.1}}} + + monkeypatch.setattr(HC, "run_local_sweep", runner) + run_catalog(tmp_path, max_skills=2) + assert [item[0] for item in calls] == ["first", "second"] + assert all(item[2] is True for item in calls) + assert _read(first_state)["run_root"].endswith(first.digest) + assert calls[0][3] == Path(_read(first_state)["run_root"]) + assert calls[0][4] == Path("/srv/ingot/runs/harbor") + assert calls[0][5] is True + assert _read(first_state)["status"] == "complete" + + run_catalog(tmp_path, max_skills=2) + assert len(calls) == 2 + assert _read(tmp_path / "state" / second_path.name)["status"] == "complete" + + +def test_catalog_forwards_native_process_environment_to_sweep(tmp_path, monkeypatch): + enqueue_catalog(tmp_path, [_intent()]) + captured = [] + + def runner(_skill, _targets, **kwargs): + captured.append(kwargs["process_env"]) + return {"aborted": False} + + monkeypatch.setattr(HC, "run_local_sweep", runner) + process_env = {"PATH": "/bin", "HARBOR_EXTRA_DOCKER_COMPOSE": "/tmp/network.yml"} + + run_catalog(tmp_path, max_skills=1, process_env=process_env) + + assert captured == [process_env] + + +def test_live_controller_refuses_before_runner(tmp_path, monkeypatch): + enqueue_catalog(tmp_path, [_intent()]) + owner = HC._claim_controller(HC.CATALOG_OWNER) + monkeypatch.setattr(HC, "run_local_sweep", + lambda *_args, **_kwargs: pytest.fail("runner called")) + + try: + with pytest.raises(RuntimeError, match="live controller"): + run_catalog(tmp_path) + finally: + owner.close() + + +def test_stale_controller_receipt_is_reused_safely(tmp_path, monkeypatch): + enqueue_catalog(tmp_path, [_intent()]) + HC.CATALOG_OWNER.write_text(json.dumps({"pid": 123, "start_token": "stale"})) + monkeypatch.setattr(HC, "run_local_sweep", lambda *_args, **_kwargs: {"aborted": False}) + + run_catalog(tmp_path, max_skills=1) + assert _read(next((tmp_path / "state").glob("*.json")))["status"] == "complete" + + +def test_failed_skill_is_recorded_and_does_not_abort_sibling(tmp_path, monkeypatch): + enqueue_catalog(tmp_path, [_intent("bad", priority=200), _intent("good", priority=100)]) + + def runner(skill, *_args, **_kwargs): + if skill == "bad": + raise RuntimeError("boom") + return {"aborted": False} + + monkeypatch.setattr(HC, "run_local_sweep", runner) + run_catalog(tmp_path, max_skills=2) + states = [_read(path) for path in (tmp_path / "state").glob("*.json")] + assert {state["status"] for state in states} == {"complete", "failed"} + assert next(state for state in states if state["status"] == "failed")["error"] == "RuntimeError" + + monkeypatch.setattr(HC, "run_local_sweep", lambda *_args, **_kwargs: {"aborted": False}) + run_catalog(tmp_path, max_skills=1) + assert {state["status"] for state in + (_read(path) for path in (tmp_path / "state").glob("*.json"))} == {"complete"} + + +def test_identity_change_enqueues_current_and_runs_no_model(tmp_path, monkeypatch): + old = _intent(sha="a" * 64) + enqueue_catalog(tmp_path, [old]) + current = _intent(sha="c" * 64) + monkeypatch.setattr(HC, "_intent_for_skill", lambda *_args, **_kwargs: current) + monkeypatch.setattr(HC, "run_local_sweep", + lambda *_args, **_kwargs: pytest.fail("runner called")) + + run_catalog(tmp_path, max_skills=1) + + states = [_read(path) for path in (tmp_path / "state").glob("*.json")] + assert {state["status"] for state in states} == {"superseded", "pending"} + assert next(state for state in states if state["status"] == "pending")["priority"] == 300 + + +def test_tampered_intent_or_state_digest_refuses_before_runner(tmp_path, monkeypatch): + [intent_path] = enqueue_catalog(tmp_path, [_intent()]) + document = _read(intent_path) + document["identity"]["skill"] = "tampered" + intent_path.write_text(json.dumps(document)) + monkeypatch.setattr(HC, "run_local_sweep", + lambda *_args, **_kwargs: pytest.fail("runner called")) + with pytest.raises(RuntimeError, match="intent digest"): + run_catalog(tmp_path) + + intent_path.write_text(json.dumps(HC._intent_document(_intent()))) + state_path = tmp_path / "state" / intent_path.name + state = _read(state_path) + state["intent_digest"] = "0" * 64 + state_path.write_text(json.dumps(state)) + with pytest.raises(RuntimeError, match="state digest"): + run_catalog(tmp_path) + + +def test_live_harbor_child_refuses_before_runner(tmp_path, monkeypatch): + intent = _intent() + enqueue_catalog(tmp_path, [intent]) + monkeypatch.setattr(HC, "_refuse_live_harbor", + lambda _root: (_ for _ in ()).throw(RuntimeError("live Harbor child"))) + monkeypatch.setattr(HC, "run_local_sweep", + lambda *_args, **_kwargs: pytest.fail("runner called")) + + with pytest.raises(RuntimeError, match="live Harbor child"): + run_catalog(tmp_path) + + +def test_low_utilization_is_reported_not_used_to_overlap_skills(tmp_path, monkeypatch): + enqueue_catalog(tmp_path, [_intent("first"), _intent("second")]) + active = 0 + peak = 0 + + def runner(*_args, **_kwargs): + nonlocal active, peak + active += 1 + peak = max(peak, active) + active -= 1 + return {"aborted": False, "utilization": 0.1} + + monkeypatch.setattr(HC, "run_local_sweep", runner) + run_catalog(tmp_path, max_skills=2) + assert peak == 1 + assert all("utilization" in _read(path) for path in (tmp_path / "state").glob("*.json")) + + +def test_telemetry_pending_state_retries_on_next_controller_without_marking_complete(tmp_path, + monkeypatch): + enqueue_catalog(tmp_path, [_intent()]) + calls = 0 + + def runner(*_args, **_kwargs): + nonlocal calls + calls += 1 + return {"aborted": False, "telemetry_pending": calls == 1} + + monkeypatch.setattr(HC, "run_local_sweep", runner) + run_catalog(tmp_path, max_skills=1) + state_path = next((tmp_path / "state").glob("*.json")) + assert _read(state_path)["status"] == "failed" + run_catalog(tmp_path, max_skills=1) + assert _read(state_path)["status"] == "complete" + assert calls == 2 + + +def test_all_skips_skills_without_tasks_and_enqueues_eligible_siblings(tmp_path, monkeypatch): + class Skill: + def __init__(self, name): + self.name = name + + monkeypatch.setattr(HC, "load_skills", lambda: [Skill("missing"), Skill("eligible")]) + monkeypatch.setattr(HC, "_intent_for_skill", + lambda skill, *_args, **_kwargs: None if skill == "missing" else _intent(skill)) + + assert HC.main(["--root", str(tmp_path), "--all", "--target", + "dell-qwen=http://127.0.0.1:8011", "--enqueue-only"]) == 0 + documents = [_read(path) for path in (tmp_path / "intents").glob("*.json")] + assert [item["identity"]["skill"] for item in documents] == ["eligible"] + + +def test_intent_route_revision_uses_discovered_target_context(tmp_path, monkeypatch): + monkeypatch.setattr(HC, "_intent_for_skill", _REAL_INTENT_FOR_SKILL) + skill = tmp_path / "skills" / "demo" + skill.mkdir(parents=True) + (skill / "SKILL.md").write_text("---\nname: demo\ndescription: x\n---\nbody") + monkeypatch.setattr(HC, "resolve_skill_dir", lambda _name: skill) + monkeypatch.setattr(HC, "_heldout", lambda _name: [{"task": "x"}] * 4) + live = HC.parse_target("dell-qwen=http://host:8011") + live = HC.LocalTarget(**{**live.__dict__, "context_length": 163840}) + monkeypatch.setattr(HC, "discover_target", lambda *_args: live) + + intent = HC._intent_for_skill("demo", ("dell-qwen=http://host:8011",), + ("claude-code",)) + + revisions = dict(intent.runtime_revisions) + assert any(key.startswith(f"route:{live.fingerprint}:claude-code") + and "20480" in value and "context=163840" in value + for key, value in revisions.items()) + + +def test_execution_stages_full_tree_atomically_and_rejects_partial_reuse(tmp_path, monkeypatch): + monkeypatch.setattr(HC, "_prepare_execution", _REAL_PREPARE_EXECUTION) + source = tmp_path / "library" / "demo" + source.mkdir(parents=True) + (source / "SKILL.md").write_text("---\nname: demo\ndescription: x\n---\nbody") + (source / "reference.md").write_text("evidence") + revision = HC.skill_revision(source) + intent = _intent() + intent = CatalogIntent(**{**intent.__dict__, + "runtime_revisions": (("skill-tree", revision),)}) + monkeypatch.setattr(HC, "resolve_skill_dir", lambda _skill: source) + + staged, execution = HC._prepare_execution(tmp_path, intent) + assert (staged / "demo" / "reference.md").read_text() == "evidence" + assert not list(execution.glob(".staged.*.tmp")) + + (staged / "demo" / "reference.md").unlink() + with pytest.raises(RuntimeError, match="skill tree identity"): + HC._prepare_execution(tmp_path, intent) + + +def test_execution_rejects_source_tree_change_during_copy(tmp_path, monkeypatch): + monkeypatch.setattr(HC, "_prepare_execution", _REAL_PREPARE_EXECUTION) + source = tmp_path / "library" / "demo" + source.mkdir(parents=True) + (source / "SKILL.md").write_text("---\nname: demo\ndescription: x\n---\nbody") + (source / "reference.md").write_text("before") + revision = HC.skill_revision(source) + intent = CatalogIntent(**{**_intent().__dict__, + "runtime_revisions": (("skill-tree", revision),)}) + monkeypatch.setattr(HC, "resolve_skill_dir", lambda _skill: source) + original = HC.shutil.copytree + + def changed_copy(source_path, destination): + result = original(source_path, destination) + (destination / "reference.md").write_text("after") + return result + + monkeypatch.setattr(HC.shutil, "copytree", changed_copy) + with pytest.raises(RuntimeError, match="changed while staging"): + HC._prepare_execution(tmp_path, intent) + assert not (tmp_path / "runs" / intent.digest / "staged").exists() diff --git a/tests/test_harbor_eval.py b/tests/test_harbor_eval.py new file mode 100644 index 0000000..86e72e2 --- /dev/null +++ b/tests/test_harbor_eval.py @@ -0,0 +1,1538 @@ +"""Unit tests for the sandboxed cross-harness skill eval (no Docker, no network, no harness). + +The Harbor subprocess and the judge are stubbed; what is exercised here is the part that is ours: +the generated dataset, the treatment/control arms, the trial-name parsing that reads Harbor's +output layout, and the scoring of what a harness actually produced.""" +import hashlib +import json +import os +import subprocess +import threading +import time +from pathlib import Path + +import pytest + +from ingot.optimize import agy_judge as A +from ingot.optimize import harbor_eval as H +from ingot.optimize.harbor_targets import LocalTarget + + +HOLDOUT = [{"task": "Write add(a, b).", "rubric": "defines add"}, + {"task": "Write sub(a, b).", "rubric": "defines sub"}] + + +@pytest.fixture(autouse=True) +def _avoid_starting_a_gateway_process_in_harbor_unit_tests(monkeypatch): + """Gateway lifecycle has its own unit tests; these tests only assert orchestration order.""" + class Session: + def __init__(self, *args, **kwargs): + pass + + def __enter__(self): + return self + + def __exit__(self, *args): + return None + + monkeypatch.setattr(H, "GatewaySession", Session) + + +def test_dataset_has_one_task_per_holdout_task(tmp_path): + dataset = H.build_dataset("demo", HOLDOUT, tmp_path) + names = sorted(p.name for p in dataset.iterdir() if p.is_dir()) + assert names == ["demo-h0", "demo-h1"] + for index, name in enumerate(names): + root = dataset / name + assert (root / "environment" / "Dockerfile").is_file() + assert (root / "tests" / "test.sh").is_file() + instruction = (root / "instruction.md").read_text() + assert HOLDOUT[index]["task"] in instruction + # Without this the agent answers in chat and the workspace stays empty, which the collector + # would then score as a zero for every task. + assert H.SOLUTION_DIR in instruction + assert 'name = "ingot/demo"' in (dataset / "dataset.toml").read_text() + + +SEEDED = [{"task": "Fix the backup.", "rubric": "fixes it", + "files": {"backup.sh": "#!/bin/bash\necho hi\n", "tools/check.py": "print(1)\n"}, + "verify": 'cd /tmp && python3 -c "print(\'ok\')"'}] + + +def test_a_seeded_task_ships_its_working_tree_into_the_image(tmp_path): + """A process skill has nothing to act on in an empty container. Verifying in the execution + context, feeding a guard its reject input and wiring a real caller all need existing code.""" + root = H.build_dataset("demo", SEEDED, tmp_path) / "demo-h0" + seed = root / "environment" / "seed" + assert (seed / "backup.sh").read_text() == "#!/bin/bash\necho hi\n" + assert (seed / "tools" / "check.py").read_text() == "print(1)\n" + dockerfile = (root / "environment" / "Dockerfile").read_text() + assert f"COPY seed/ {H.REPO_DIR}/" in dockerfile + # The agent is told where the tree is; without this it starts from a blank directory and the + # seeded defect is never seen. + assert H.REPO_DIR in (root / "instruction.md").read_text() + + +def test_an_unseeded_task_keeps_the_blank_environment(tmp_path): + """Seeding is opt-in per task: a task with no files must build the image it always built.""" + root = H.build_dataset("demo", HOLDOUT, tmp_path) / "demo-h0" + assert not (root / "environment" / "seed").exists() + assert "COPY seed/" not in (root / "environment" / "Dockerfile").read_text() + assert "_objective_check" not in (root / "tests" / "test.sh").read_text() + + +def test_the_objective_check_runs_after_the_agent_and_survives_quoting(tmp_path): + """The agent's own evidence log is a claim about what it ran, and a skill that rewards writing + evidence logs is exactly what teaches it to produce one. This is the referent from outside.""" + root = H.build_dataset("demo", SEEDED, tmp_path) / "demo-h0" + test_sh = (root / "tests" / "test.sh").read_text() + assert "_objective_check.txt" in test_sh + # The command embeds both quote kinds. Interpolating it raw produced unrunnable shell. + assert subprocess.run(["bash", "-n", str(root / "tests" / "test.sh")], + capture_output=True).returncode == 0 + # It must stay captured evidence, never a second grader: the Ingot judge owns the score. + assert "echo 1 > /logs/verifier/reward.txt" in test_sh + + +def test_a_seeded_path_cannot_escape_the_task_directory(tmp_path): + """Task files are authored data, but a traversing key would write into this checkout rather + than the container's.""" + escaping = [{"task": "t", "rubric": "r", "files": {"../../pwned": "x"}}] + with pytest.raises(ValueError, match="escapes"): + H.build_dataset("demo", escaping, tmp_path) + assert not (tmp_path.parent / "pwned").exists() + + +def test_rebuilding_a_dataset_drops_tasks_from_a_shorter_holdout(tmp_path): + """A holdout that shrinks must not leave last run's extra task behind to be run and scored.""" + H.build_dataset("demo", HOLDOUT, tmp_path) + dataset = H.build_dataset("demo", HOLDOUT[:1], tmp_path) + assert sorted(p.name for p in dataset.iterdir() if p.is_dir()) == ["demo-h0"] + + +def test_the_skill_arm_passes_a_skill_and_the_control_arm_does_not(tmp_path, monkeypatch): + """Lift is the difference between these two commands. If the control also carried --skill there + would be no control at all, and every lift would come out at zero.""" + seen = [] + + class Done: + returncode, stderr, stdout = 0, "", "" + + monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: seen.append(argv) or Done()) + monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None) # covered by its own tests + H.run_arm(tmp_path / "ds", "claude-code", "/skills/demo", tmp_path / "jobs", "skill", + log=lambda *a: None) + H.run_arm(tmp_path / "ds", "claude-code", None, tmp_path / "jobs", "control", + log=lambda *a: None) + + assert "--skill" in seen[0] and seen[0][seen[0].index("--skill") + 1] == "/skills/demo" + assert "--skill" not in seen[1] + assert "--path" in seen[0], "a local dataset must be passed with --path, not --dataset" + + +def test_run_arm_forwards_agent_env_agent_kwargs_task_name_without_provider_leak( + tmp_path, monkeypatch +): + """Local adapter settings cross the Harbor boundary without exposing the parent credentials.""" + seen = [] + logs = [] + + class Done: + returncode, stderr, stdout = 0, "", "" + + def fake_run(argv, **kwargs): + seen.append((argv, kwargs)) + return Done() + + monkeypatch.setattr(H.subprocess, "run", fake_run) + monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None) + secret = "sk-parent-secret" + child_env = {"PATH": "/usr/bin", "OPENAI_API_KEY": secret, + "ANTHROPIC_API_KEY": "sk-anthropic-parent", "SAFE": "yes", + "SAFE_SECRET": secret, + "LANGFUSE_PUBLIC_KEY": "pk-parent", + "LANGFUSE_SECRET_KEY": "sk-parent", + "LANGFUSE_BASE_URL": "https://langfuse-parent.invalid", + "LANGFUSE_ENCRYPTION_KEY": "encryption-parent", + "LANGFUSE_INIT_USER_PASSWORD": "password-parent", + "LANGFUSE_PUBLIC_URL": "https://public-parent.invalid"} + H.run_arm( + tmp_path / "ds", "terminus-2", None, tmp_path / "jobs", "local", + model="deepseek-v4-flash", concurrency=3, attempts=2, + agent_env={"Z_KEY": "z-value", "A_KEY": "a-value", "OPENAI_API_KEY": "local", + "LANGFUSE_PUBLIC_KEY": "pk-agent", + "LANGFUSE_SECRET_KEY": "sk-agent", + "LANGFUSE_BASE_URL": "https://langfuse-agent.invalid", + "LANGFUSE_ENCRYPTION_KEY": "encryption-agent", + "LANGFUSE_INIT_USER_PASSWORD": "password-agent", + "LANGFUSE_PUBLIC_URL": "https://public-agent.invalid"}, + agent_kwargs={"zeta": "z-value", "alpha": "a-value"}, + task_name="demo-h0", process_env=child_env, log=logs.append, + ) + + argv, kwargs = seen[0] + assert [argv[i + 1] for i, value in enumerate(argv) if value == "--ae"] == [ + "A_KEY=a-value", "OPENAI_API_KEY=local", "Z_KEY=z-value" + ] + assert [argv[i + 1] for i, value in enumerate(argv) if value == "--ak"] == [ + "alpha=a-value", "zeta=z-value" + ] + assert argv[argv.index("--include-task-name") + 1] == "demo-h0" + assert kwargs["env"] is not child_env + assert kwargs["env"]["PATH"] == "/usr/bin" and kwargs["env"]["SAFE"] == "yes" + assert kwargs["env"]["OPENAI_API_KEY"] == "local" + assert "ANTHROPIC_API_KEY" not in kwargs["env"] + assert not any(key.startswith("LANGFUSE_") for key in kwargs["env"]) + assert not any("LANGFUSE_" in value for value in argv) + assert kwargs["capture_output"] is True and kwargs["text"] is True + assert "OPENAI_API_KEY=local" in argv + assert secret not in " ".join(argv) + assert secret not in "\n".join(logs) + assert secret not in str(tmp_path / "jobs" / "local") + + +def test_run_arm_serializes_nested_agent_kwargs_as_json(tmp_path, monkeypatch): + seen = [] + + class Done: + returncode, stderr, stdout = 0, "", "" + + monkeypatch.setattr(H.subprocess, "run", lambda argv, **kwargs: seen.append(argv) or Done()) + monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None) + H.run_arm(tmp_path / "ds", "opencode", None, tmp_path / "jobs", "local", + agent_kwargs={"opencode_config": {"provider": {"local": {"npm": "x"}}}}) + value = seen[0][seen[0].index("--ak") + 1] + assert value == 'opencode_config={"provider":{"local":{"npm":"x"}}}' + + +def test_run_arm_without_local_overrides_keeps_legacy_command_and_process_call( + tmp_path, monkeypatch +): + """Existing proprietary callers keep the old command and inherited process environment.""" + seen = [] + + class Done: + returncode, stderr, stdout = 0, "", "" + + monkeypatch.setenv("LANGFUSE_ENCRYPTION_KEY", "parent-encryption") + monkeypatch.setenv("LANGFUSE_PUBLIC_URL", "https://public.invalid") + monkeypatch.setenv("OPENAI_API_KEY", "provider-key-must-remain") + monkeypatch.setattr(H.subprocess, "run", lambda argv, **kwargs: seen.append((argv, kwargs)) or Done()) + monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None) + H.run_arm(tmp_path / "ds", "claude-code", None, tmp_path / "jobs", "control", + log=lambda *a: None) + argv, kwargs = seen[0] + assert argv == [ + H.HARBOR_BIN, "run", "--path", str(tmp_path / "ds"), "--agent", "claude-code", + "--n-concurrent", "2", "--jobs-dir", str(tmp_path / "jobs"), "--job-name", "control", + "--environment-build-timeout-multiplier", str(H.BUILD_TIMEOUT_MULTIPLIER), + "--agent-setup-timeout-multiplier", str(H.SETUP_TIMEOUT_MULTIPLIER), + ] + assert kwargs["capture_output"] is True and kwargs["text"] is True + assert not any(key.startswith("LANGFUSE_") for key in kwargs["env"]) + assert kwargs["env"]["OPENAI_API_KEY"] == "provider-key-must-remain" + + +def test_a_failing_harbor_run_is_raised_not_silently_scored(tmp_path, monkeypatch): + class Done: + returncode, stderr, stdout = 1, "boom", "outer stdout" + + monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: Done()) + with pytest.raises(RuntimeError, match="boom"): + H.run_arm(tmp_path / "ds", "codex", None, tmp_path / "jobs", "control", log=lambda *a: None) + assert json.loads((tmp_path / "jobs" / "control" / "harbor-invocation.json").read_text()) == { + "returncode": 1, "stdout_bytes": 12, "stderr_bytes": 4, + "stdout_excerpt": "outer stdout", "stderr_excerpt": "boom", + } + + +def test_run_arm_keeps_outer_harbor_exit_evidence_when_a_pending_job_is_refused(tmp_path, monkeypatch): + """A Harbor zero exit can still leave a pending job; preserve the only outer diagnostic.""" + class Done: + returncode = 0 + stdout = "Authorization: Bearer bearer-value http://target.invalid:8001/sk-path" + stderr = ('x-api-key: header-value Authorization: Basic basic-value ' + 'sk-live-value pk_live_value known-secret-value ' + '"OPENAI_API_KEY": "json-value" LITELLM_API_KEY: colon-value ' + 'MODEL_API_KEY space-value internal.example:8001') + + parent = {"OPENAI_API_KEY": "known-secret-value", "CUSTOM_API_KEY": "mapping-secret"} + monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: Done()) + + def refuse_pending(*args): + raise RuntimeError("canary wrote no completed trial") + + monkeypatch.setattr(H, "_refuse_broken_job", refuse_pending) + job = tmp_path / "jobs" / "codex" + with pytest.raises(RuntimeError, match="canary wrote no completed trial"): + H.run_arm(tmp_path / "ds", "codex", None, tmp_path / "jobs", "codex", + process_env=parent, log=lambda *a: None) + receipt_path = job / "harbor-invocation.json" + receipt = json.loads(receipt_path.read_text()) + assert receipt["returncode"] == 0 + assert receipt["stdout_bytes"] == len(Done.stdout.encode()) + assert receipt["stderr_bytes"] == len(Done.stderr.encode()) + persisted = receipt_path.read_text() + for raw in ("target.invalid", "bearer-value", "header-value", "basic-value", "sk-live-value", + "pk_live_value", "known-secret-value", "mapping-secret", "json-value", + "colon-value", "space-value", "internal.example:8001"): + assert raw not in persisted + assert receipt["stdout_excerpt"] and receipt["stderr_excerpt"] + + +def test_trial_name_drops_the_run_suffix_harbor_appends(tmp_path): + """Observed layout: jobs//__/verifier/solution. Keeping the suffix means no + trial ever matches its task and every score reads as a zero.""" + solution = tmp_path / "demo-h1__abc123" / "verifier" / "solution" + solution.mkdir(parents=True) + assert H._trial_task_name(solution) == "demo-h1" + + +def test_collect_reads_what_the_agent_left_in_the_solution_directory(tmp_path): + solution = tmp_path / "demo-h0__xy" / "verifier" / "solution" + (solution / "pkg").mkdir(parents=True) + (solution / "answer.py").write_text("def add(a, b): return a + b") + (solution / "pkg" / "notes.md").write_text("reasoning") + answers = H.collect_answers(tmp_path) + assert set(answers) == {"demo-h0"} + assert "def add" in answers["demo-h0"][0] and "reasoning" in answers["demo-h0"][0] + + +def test_repeated_attempts_at_one_task_are_all_kept(tmp_path): + """Keying a single answer by task name silently kept only whichever trial was read last, so + --n-attempts above 1 paid for repeated measurements and then threw all but one away. Repetition + is the whole remedy for this eval's noise: re-judging a fixed answer three times returned an + identical score, while re-running the agent on the same task moved it by 0.278.""" + for suffix, body in (("aa", "first attempt"), ("bb", "second attempt")): + solution = tmp_path / f"demo-h0__{suffix}" / "verifier" / "solution" + solution.mkdir(parents=True) + (solution / "answer.py").write_text(body) + answers = H.collect_answers(tmp_path) + assert len(answers["demo-h0"]) == 2 + assert {"first attempt", "second attempt"} == {a.split("\n", 1)[1] for a in answers["demo-h0"]} + + +def test_a_tasks_score_is_the_mean_over_its_attempts(monkeypatch): + scores = iter([1.0, 0.0]) + monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": next(scores)}) + got = H.score({"demo-h0": ["one", "two"]}, "demo", HOLDOUT[:1]) + assert got == [0.5] + + +def test_arm_scoring_runs_independent_grades_with_bounded_concurrency(monkeypatch): + active = 0 + peak = 0 + lock = threading.Lock() + + def fake_judge(*_args, **_kwargs): + nonlocal active, peak + with lock: + active += 1 + peak = max(peak, active) + time.sleep(0.02) + with lock: + active -= 1 + return {"score": 0.5} + + monkeypatch.setattr(H, "judge", fake_judge) + answers = {f"demo-h{index}": ["one", "two", "three"] for index in range(2)} + + scores = H.score(answers, "demo", HOLDOUT, concurrency=4) + + assert scores == [0.5, 0.5] + assert 1 < peak <= 4 + + +def test_a_harness_that_produced_nothing_scores_zero_rather_than_being_dropped(monkeypatch): + """Dropping the empty task would raise the arm's mean by removing its own failure. A harness + that ran and delivered nothing has a real score, and it is zero.""" + monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.8}) + scores = H.score({"demo-h0": ["some code"]}, "demo", HOLDOUT) + assert scores == [0.8, 0.0] + + +def test_the_tasks_own_checklist_reaches_the_judge(monkeypatch): + """A task's checklist is what gives its score any resolution. + + `judge()` grades on four generic items unless a task supplies its own, and a frontier model + passes all four on any easy task. Dropping the checklist here is what flattened the first + build-loop matrix: every control landed near 0.85 and no arm could separate from any other.""" + checklist = [{"id": "declares_stakes_tier", "criterion": "names a tier", "weight": 3, + "dimension": "instruction_following"}] + holdout = [{"task": "Write add(a, b).", "rubric": "defines add", "checklist": checklist}] + seen = {} + + def fake_judge(task, rubric, answer, **kwargs): + seen.update(kwargs) + return {"score": 1.0} + + monkeypatch.setattr(H, "judge", fake_judge) + H.score({"demo-h0": ["def add(a, b): return a + b"]}, "demo", holdout) + assert seen["checklist"] == checklist + + +def test_agy_failure_propagates_out_of_arm_scoring(monkeypatch): + monkeypatch.setenv("JUDGE_BACKEND", "agy") + monkeypatch.setattr( + A, + "invoke", + lambda *_args: (_ for _ in ()).throw(A.AgyJudgeError("agy stopped")), + ) + + with pytest.raises(A.AgyJudgeError, match="agy stopped"): + H.score({"demo-h0": ["answer"]}, "demo", HOLDOUT[:1]) + + +def test_matrix_records_a_broken_harness_without_discarding_the_others(tmp_path, monkeypatch): + """One missing harness must not throw away the arms already paid for, and its row must carry no + lift: a blank zero would read as 'measured, no effect'.""" + monkeypatch.setattr(H, "load_tasks", lambda skill: ([], HOLDOUT, {})) + monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "source") + monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "staged") + monkeypatch.setattr(H.shutil, "which", lambda binary: "/usr/bin/harbor") + monkeypatch.setattr(H, "build_dataset", lambda *a, **k: tmp_path / "ds") + monkeypatch.setattr(H, "HARBOR_DIR", tmp_path / "out") + monkeypatch.setattr(H, "collect_answers", lambda job, skip=None: {}) + + def fake_run_arm(dataset, agent, source, jobs_dir, job_name, *a, **k): + if agent == "broken": + raise RuntimeError("no such agent") + return jobs_dir / job_name + + monkeypatch.setattr(H, "run_arm", fake_run_arm) + monkeypatch.setattr(H, "broken_tasks", lambda job: set()) + monkeypatch.setattr(H, "score", lambda answers, skill, holdout, skip=None: ( + [1.0, 1.0] if "skill" in str(answers) else [0.5, 0.5])) + + out = H.run_harbor_eval("demo", ["claude-code", "broken"], log=lambda *a: None) + assert "lift" in out["harnesses"]["claude-code"] + assert "no such agent" in out["harnesses"]["broken"]["error"] + assert "lift" not in out["harnesses"]["broken"] + assert json.loads((tmp_path / "out" / "demo.json").read_text())["skill"] == "demo" + + +def test_a_run_where_no_harness_worked_fails_loudly(tmp_path, monkeypatch): + monkeypatch.setattr(H, "load_tasks", lambda skill: ([], HOLDOUT, {})) + monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "staged") + monkeypatch.setattr(H.shutil, "which", lambda binary: "/usr/bin/harbor") + monkeypatch.setattr(H, "build_dataset", lambda *a, **k: tmp_path / "ds") + monkeypatch.setattr(H, "HARBOR_DIR", tmp_path / "out") + monkeypatch.setattr(H, "run_arm", lambda *a, **k: (_ for _ in ()).throw(RuntimeError("down"))) + with pytest.raises(SystemExit, match="nothing was measured"): + H.run_harbor_eval("demo", ["a", "b"], log=lambda *a: None) + assert not (tmp_path / "out" / "demo.json").exists() + + +def test_build_leavings_are_not_fed_to_the_judge(tmp_path): + """The first real container run left __pycache__/*.pyc beside solution.py. Those bytes went + into the text the judge grades — noise it pays for and can be misled by.""" + solution = tmp_path / "demo-h0__xy" / "verifier" / "solution" + (solution / "__pycache__").mkdir(parents=True) + (solution / "solution.py").write_text("def add(a, b): return a + b") + (solution / "__pycache__" / "solution.cpython-312.pyc").write_bytes(b"\x00\x01\xfe\xff") + answers = H.collect_answers(tmp_path) + assert "def add" in answers["demo-h0"][0] + assert "pycache" not in answers["demo-h0"][0] + + +def _job_with_stats(tmp_path, **stats): + job = tmp_path / "jobs" / "arm" + job.mkdir(parents=True) + (job / "result.json").write_text(json.dumps({"stats": stats})) + return job + + +def test_an_arm_whose_trials_all_crashed_is_refused_not_scored_as_zeros(tmp_path, monkeypatch): + """Observed live: harbor exited 0 with n_errored_trials=4, the solution dirs were empty, and + score() read four legitimate 0.0s — producing lift +0.750 against a working skill arm. An arm + that did not run is missing, not bad.""" + class Done: + returncode, stderr, stdout = 0, "", "" + + monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: Done()) + monkeypatch.setattr(H, "_refuse_broken_job", H._refuse_broken_job) + job = _job_with_stats(tmp_path, n_completed_trials=4, n_errored_trials=4, n_cancelled_trials=4) + with pytest.raises(RuntimeError, match="refusing to score"): + H._refuse_broken_job(job, "terminus-2", "control") + + +def test_a_clean_arm_is_accepted(tmp_path): + job = _job_with_stats(tmp_path, n_completed_trials=4, n_errored_trials=0, n_cancelled_trials=0) + H._refuse_broken_job(job, "terminus-2", "skill") + + +def test_an_arm_with_no_result_file_is_refused(tmp_path): + """No result.json means harbor never got far enough to report; scoring it would invent data.""" + with pytest.raises(RuntimeError, match="no result.json"): + H._refuse_broken_job(tmp_path / "missing", "codex", "skill") + + +def test_an_arm_that_delivered_nothing_at_all_is_refused_not_scored_as_zeros(monkeypatch): + """A trial can complete while its agent never ran: the verifier always reports success, so an + agent that died on its first API call still counts completed with an empty workspace. Observed + live with aider (temperature rejected by claude-sonnet-5) — it would have scored a clean 0.000 + and read as 'aider is terrible at this skill' rather than 'aider never ran'.""" + monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.9}) + with pytest.raises(RuntimeError, match="empty workspace"): + H.score({}, "demo", HOLDOUT) + + +def test_a_partly_empty_arm_still_scores_its_failures_as_zero(monkeypatch): + """One task delivering nothing is a real failure of that task, not a broken combination.""" + monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.9}) + assert H.score({"demo-h0": "code"}, "demo", HOLDOUT) == [0.9, 0.0] + + +def _trial(job, task, exception_type=None): + """Harbor's real trial result shape, not an invented one. + + The first version of this helper wrote a top-level `exception_type`, which Harbor never emits — + so the guard read nothing, passed its tests, and was a no-op against real output.""" + d = job / f"{task}__xy" + d.mkdir(parents=True, exist_ok=True) + record = {"task_name": f"ingot/{task}", "exception_info": None} + if exception_type: + record["exception_info"] = {"exception_type": exception_type, "exception_message": ""} + (d / "result.json").write_text(json.dumps(record)) + return d + + +def test_broken_tasks_names_only_the_trials_that_failed(tmp_path): + job = tmp_path / "arm" + _trial(job, "demo-h0") + _trial(job, "demo-h1", exception_type="CancelledError") + assert H.broken_tasks(job) == {"demo-h1"} + + +def test_one_transient_trial_failure_does_not_discard_the_whole_arm(tmp_path): + """Failing the arm on any broken trial threw away three good trials and the paid-for opposite + arm. Observed live: the first grid row died on 1 errored trial of 4.""" + job = tmp_path / "arm" + job.mkdir() + (job / "result.json").write_text(json.dumps( + {"stats": {"n_completed_trials": 4, "n_errored_trials": 1, "n_cancelled_trials": 0}})) + H._refuse_broken_job(job, "terminus-2", "skill") # must not raise + + +def test_an_arm_is_still_refused_when_every_trial_broke(tmp_path): + job = tmp_path / "arm" + job.mkdir() + (job / "result.json").write_text(json.dumps( + {"stats": {"n_completed_trials": 4, "n_errored_trials": 4, "n_cancelled_trials": 4}})) + with pytest.raises(RuntimeError, match="every one of its"): + H._refuse_broken_job(job, "terminus-2", "control") + + +def test_a_task_dropped_from_one_arm_is_dropped_from_both(monkeypatch): + """Scoring the arms over different task sets means their difference is not lift.""" + monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.6}) + scores = H.score({"demo-h0": "x", "demo-h1": "y"}, "demo", HOLDOUT, skip={"demo-h1"}) + assert scores == [0.6], "the dropped task must not appear in the scored list" + + +def test_scoring_refuses_when_every_task_was_dropped(monkeypatch): + monkeypatch.setattr(H, "judge", lambda *a, **k: {"score": 0.6}) + with pytest.raises(RuntimeError, match="nothing comparable"): + H.score({"demo-h0": "x"}, "demo", HOLDOUT, skip={"demo-h0", "demo-h1"}) + + +def _attempt(job, task, suffix, *, errored=False, answer="ok"): + """One trial directory as Harbor lays it out, with or without a recorded exception.""" + trial = job / f"{task}__{suffix}" + (trial / "verifier" / "solution").mkdir(parents=True) + (trial / "verifier" / "solution" / "answer.py").write_text(answer) + record = {"task_name": f"ingot/{task}"} + if errored: + record["exception_info"] = {"exception_type": "AgentSetupTimeoutError"} + (trial / "result.json").write_text(json.dumps(record)) + return trial + + +def test_one_crashed_attempt_does_not_discard_the_task(tmp_path): + """With repeats, dropping a task because one attempt broke throws away the attempts that did + run — which are the entire reason for paying for repeats.""" + job = tmp_path / "control" + job.mkdir() + _attempt(job, "demo-h0", "aa", errored=True) + _attempt(job, "demo-h0", "bb") + _attempt(job, "demo-h0", "cc") + assert H.broken_tasks(job) == set() + assert H.broken_trials(job) == {"demo-h0__aa"} + + +def test_a_task_whose_every_attempt_crashed_is_still_dropped(tmp_path): + job = tmp_path / "control" + job.mkdir() + _attempt(job, "demo-h1", "aa", errored=True) + _attempt(job, "demo-h1", "bb", errored=True) + assert H.broken_tasks(job) == {"demo-h1"} + + +def test_a_crashed_attempts_empty_workspace_is_not_averaged_in_as_a_zero(tmp_path): + """The crashed trial's directory exists and is empty through no fault of the agent. Averaged in + it would pull a 3-attempt task's mean down by a third and read as the skill performing worse.""" + job = tmp_path / "control" + job.mkdir() + _attempt(job, "demo-h0", "aa", errored=True, answer="") + _attempt(job, "demo-h0", "bb", answer="real work") + answers = H.collect_answers(job, H.broken_trials(job)) + assert len(answers["demo-h0"]) == 1 + assert "real work" in answers["demo-h0"][0] + + +def test_runs_at_different_attempt_counts_do_not_collide(tmp_path, monkeypatch): + """Harbor refuses a job directory whose config changed, so re-running a skill at a new -k failed + all 13 combinations before a container started. The earlier run's trials are also the evidence a + rescore reads, so overwriting them is worse than the collision.""" + monkeypatch.setattr(H, "load_tasks", lambda skill: ([], HOLDOUT, {})) + monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "staged") + monkeypatch.setattr(H.shutil, "which", lambda binary: "/usr/bin/harbor") + monkeypatch.setattr(H, "build_dataset", lambda *a, **k: tmp_path / "ds") + monkeypatch.setattr(H, "HARBOR_DIR", tmp_path / "out") + monkeypatch.setattr(H, "collect_answers", lambda job, skip=None: {}) + monkeypatch.setattr(H, "broken_tasks", lambda job: set()) + monkeypatch.setattr(H, "broken_trials", lambda job: set()) + monkeypatch.setattr(H, "score", lambda answers, skill, holdout, skip=None: [1.0, 1.0]) + seen = [] + + def fake_run_arm(dataset, agent, source, jobs_dir, job_name, *a, **k): + seen.append(jobs_dir) + return jobs_dir / job_name + + monkeypatch.setattr(H, "run_arm", fake_run_arm) + H.run_harbor_eval("demo", ["claude-code"], attempts=1, log=lambda *a: None) + H.run_harbor_eval("demo", ["claude-code"], attempts=3, log=lambda *a: None) + + roots = {path.parent.name for path in seen} + assert roots == {"demo", "demo-k3"} + + +def test_a_subscription_capable_harness_will_not_quietly_bill_per_token(monkeypatch): + """The default is silent and expensive. A whole grid ran on metered keys with both subscription + logins sitting unused on the same host, and nothing in the output said so — the per-arm dollar + figure Harbor prints is a computed estimate that reads the same either way.""" + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-whatever") + monkeypatch.delenv("CLAUDE_FORCE_OAUTH", raising=False) + monkeypatch.delenv(H.ALLOW_API_BILLING, raising=False) + refusals = H.billing_refusals(["claude-code@anthropic/claude-opus-5"]) + assert len(refusals) == 1 + assert "CLAUDE_FORCE_OAUTH" in refusals[0] and "claude setup-token" in refusals[0] + + +def test_a_harness_with_no_cli_to_harness_is_not_refused(monkeypatch): + """terminus-2, goose, aider, opencode and pi drive a provider API directly. There is no + subscription to prefer, so an API key is inherent to running them at all.""" + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-whatever") + monkeypatch.setenv("OPENAI_API_KEY", "sk-whatever") + monkeypatch.delenv(H.ALLOW_API_BILLING, raising=False) + assert H.billing_refusals(["terminus-2@anthropic/claude-opus-5", "goose@anthropic/claude-sonnet-5", + "aider@openai/gpt-5.5", "pi@openai/gpt-5.5"]) == [] + + +def test_the_subscription_flag_clears_the_refusal(monkeypatch): + monkeypatch.setenv("OPENAI_API_KEY", "sk-whatever") + monkeypatch.setenv("CODEX_FORCE_AUTH_JSON", "1") + monkeypatch.delenv(H.ALLOW_API_BILLING, raising=False) + assert H.billing_refusals(["codex@openai/gpt-5.5"]) == [] + + +def test_metered_billing_can_still_be_opted_into_deliberately(monkeypatch): + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-whatever") + monkeypatch.delenv("CLAUDE_FORCE_OAUTH", raising=False) + monkeypatch.setenv(H.ALLOW_API_BILLING, "1") + assert H.billing_refusals(["claude-code@anthropic/claude-opus-5"]) == [] + + +def test_the_grid_refuses_to_start_rather_than_billing_then_reporting(tmp_path, monkeypatch): + """Per-row would be too late: a grid is hours long and the bill is run up by then.""" + monkeypatch.setattr(H, "load_tasks", lambda skill: ([], HOLDOUT, {})) + monkeypatch.setattr(H, "stage_skill", lambda skill: tmp_path / "staged") + monkeypatch.setattr(H.shutil, "which", lambda binary: "/usr/bin/harbor") + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-whatever") + monkeypatch.delenv("CLAUDE_FORCE_OAUTH", raising=False) + monkeypatch.delenv(H.ALLOW_API_BILLING, raising=False) + + def must_not_run(*a, **k): + raise AssertionError("a container was started before the billing check") + + monkeypatch.setattr(H, "run_arm", must_not_run) + monkeypatch.setattr(H, "build_dataset", must_not_run) + with pytest.raises(SystemExit, match="refusing to start"): + H.run_harbor_eval("demo", ["claude-code@anthropic/claude-opus-5"], log=lambda *a: None) + + +def test_every_run_asks_for_build_and_setup_headroom(tmp_path, monkeypatch): + """Seeded tasks each build their own image, where an unseeded dataset shared one cached image + built once. `apt-get update && install` on an uncached image overran the 120s compose budget, + and it surfaced as a bare RuntimeError with an empty verifier directory — a build failure that + looks nothing like one, and which scored as a dropped task.""" + seen = [] + + class Done: + returncode, stderr, stdout = 0, "", "" + + monkeypatch.setattr(H.subprocess, "run", lambda argv, **kw: seen.append(argv) or Done()) + monkeypatch.setattr(H, "_refuse_broken_job", lambda *a: None) + H.run_arm(tmp_path / "ds", "terminus-2", None, tmp_path / "jobs", "control", log=lambda *a: None) + + argv = seen[0] + assert "--environment-build-timeout-multiplier" in argv + assert float(argv[argv.index("--environment-build-timeout-multiplier") + 1]) > 1 + assert "--agent-setup-timeout-multiplier" in argv + + +def test_the_image_ships_what_agents_would_otherwise_install_themselves(tmp_path): + """terminus-2 installs tmux and asciinema into the container when they are missing, and that + apt-get overran the 120s exec budget on a cold cache — a bare RuntimeError from + _install_recording_tools, empty verifier directory, task counted broken and dropped. Its + installer skips the work when both are present. pytest is here because the seeded READMEs tell + the agent to run it and Ubuntu 24.04 refuses pip installs under PEP 668.""" + dockerfile = (H.build_dataset("demo", HOLDOUT, tmp_path) / "demo-h0" + / "environment" / "Dockerfile").read_text() + for package in ("tmux", "asciinema", "python3-pytest"): + assert package in dockerfile, f"{package} must be baked in, not installed per trial" + + +def _local_target(alias="dell-qwen", **changes): + models = {"dell-qwen": "dot-backbone", "spark-deepseek": "deepseek-v4-flash", + "orin-abliterated": "ablit35b"} + values = { + "alias": alias, + "display_name": alias, + "base_url": f"http://{alias}.test:8000", + "served_model": models[alias], + "context_length": 32768, + "protocols": frozenset({"chat", "responses", "messages"}), + "family": "Qwen3.6" if alias == "dell-qwen" else "fixture-family", + "parameter_billions": 27.0 if alias == "dell-qwen" else 1.0, + "quantization": "fp8-published" if alias == "dell-qwen" else "fixture-quant", + "tool_parser": "qwen3_xml" if alias == "dell-qwen" else "fixture-parser", + } + values.update(changes) + return LocalTarget(**values) + + +def _completed_canary(job: Path, task: str, *, solution=True, exception=None, completed=1): + (job / "result.json").parent.mkdir(parents=True, exist_ok=True) + (job / "result.json").write_text(json.dumps({"stats": {"n_completed_trials": completed, + "n_errored_trials": 0, + "n_cancelled_trials": 0}})) + trial = job / f"{task}__run" / "verifier" / "solution" + trial.mkdir(parents=True) + if solution: + (trial / "answer.txt").write_text("done") + record = {"task_name": f"ingot/{task}", "exception_info": None} + if exception: + record["exception_info"] = {"exception_type": exception} + (trial.parent.parent / "result.json").write_text(json.dumps(record)) + + +def test_canary_requires_a_completed_exception_free_trial_with_a_solution(tmp_path, monkeypatch): + """A successful Harbor process with no agent deliverable must block the full sweep.""" + target = _local_target() + job = tmp_path / "canaries" / "demo" / target.job_slug / "codex" + _completed_canary(job, "demo-h0", solution=False) + calls = [] + + def fake_run_arm(*args, **kwargs): + calls.append((args, kwargs)) + return job + + monkeypatch.setattr(H, "run_arm", fake_run_arm) + record = H.run_canary("demo", tmp_path / "dataset", HOLDOUT, "/staged/demo", "codex", + target, tmp_path / "canaries", log=lambda *a: None) + assert record["error"] == "canary produced no nonempty verifier solution artifact" + route = H.gateway_route(target, "codex") + assert route is not None + args, kwargs = calls[0] + assert args[3] == job.parent and args[4] == f"codex--{route.identity}" + assert args[1] == "ingot.optimize.harbor_codex_gateway:GatewayCodex" + assert kwargs["task_name"] == "demo-h0" and kwargs["attempts"] == 1 + assert kwargs["model"] == route.model + assert kwargs["agent_env"]["OPENAI_BASE_URL"].startswith("http://172.17.") + assert kwargs["process_env"]["PYTHONPATH"].split(os.pathsep)[0] == str( + Path(H.__file__).resolve().parents[2]) + + +def test_canary_exports_the_persisted_job_after_run_arm(tmp_path, monkeypatch): + """Removing the canary telemetry caller would leave its retained attempt undiscoverable.""" + target = _local_target() + job = tmp_path / "canaries" / "demo" / target.job_slug / "codex" + _completed_canary(job, "demo-h0") + source = tmp_path / "staged" + skill_file = source / "demo" / "SKILL.md" + skill_file.parent.mkdir(parents=True) + skill_file.write_text("fixture skill body\n") + events = [] + + def fake_run_arm(*args, **kwargs): + events.append(("run", job)) + return job + + def fake_export(exported_job, metadata): + events.append(("export", exported_job, metadata)) + return [{"status": "verified"}] + + monkeypatch.setattr(H, "run_arm", fake_run_arm) + monkeypatch.setattr(H, "export_job_attempts", fake_export) + + record = H.run_canary("demo", tmp_path / "dataset", HOLDOUT, str(source), "codex", + target, tmp_path / "canaries", log=lambda *a: None) + + assert [event[0] for event in events] == ["run", "export"] + assert events[1][1] == job + assert events[1][2]["combination"] == record["combination"] + assert events[1][2]["arm"] == "canary" + assert events[1][2]["skill"] == "demo" + assert events[1][2]["task_texts"] == {"demo-h0": HOLDOUT[0]["task"], + "demo-h1": HOLDOUT[1]["task"]} + assert events[1][2]["skill_sha256"] == hashlib.sha256( + skill_file.read_bytes()).hexdigest() + assert events[1][2]["skill_body"] == "fixture skill body\n" + assert record["ok"] is True and "telemetry_error" not in record + + +def test_telemetry_provenance_sanitizes_task_text_before_persisting(tmp_path): + source = tmp_path / "staged" + skill_file = source / "demo" / "SKILL.md" + skill_file.parent.mkdir(parents=True) + skill_file.write_text("fixture skill body\n") + holdout = [{"task": ( + "Call https://private-endpoint.invalid/v1 with " + "OPENAI_API_KEY=fixture-secret-value" + )}] + + provenance = H._telemetry_provenance("demo", holdout, str(source)) + + persisted = provenance["task_texts"]["demo-h0"] + assert "private-endpoint.invalid" not in persisted + assert "fixture-secret-value" not in persisted + assert " LocalTarget: + context = {"dell-qwen": 163840, "spark-deepseek": 1048576, + "orin-abliterated": 65536}[alias] + return LocalTarget( + alias=alias, + display_name=TARGETS[alias]["display_name"], + base_url="http://local.test:8011", + served_model=TARGETS[alias]["served_model"], + context_length=context, + protocols=frozenset({"chat", "messages", "responses"}), + ) + + +def test_gateway_routes_only_the_three_observed_role_rejections(): + dell = _target("dell-qwen") + spark = _target("spark-deepseek") + assert G.gateway_route(dell, "claude-code") is not None + assert G.gateway_route(spark, "claude-code") is not None + assert G.gateway_route(dell, "codex") is not None + assert G.gateway_route(spark, "codex") is None + orin = _target("orin-abliterated") + for harness in ("claude-code", "terminus-2", "goose", "opencode", "openclaw", + "mini-swe-agent", "codex", "aider", "pi"): + assert G.gateway_route(orin, harness) is None + for harness in ("terminus-2", "goose", "opencode", "openclaw", "mini-swe-agent", "aider", "pi"): + assert G.gateway_route(dell, harness) is None + unknown = LocalTarget("third-target", "Third", "http://local.test:9000", "third", 32768, + frozenset({"chat", "messages", "responses"})) + assert G.gateway_route(unknown, "claude-code") is None + + +def test_gateway_identity_is_stable_and_does_not_expose_upstream_address(): + route = G.gateway_route(_target(), "claude-code") + assert route is not None + assert route.revision == f"{G.DELL_CLAUDE_OUTPUT_CAP_REVISION}-20480" + assert route.model.startswith("harbor-compat-") + assert "local.test" not in route.identity + assert route.identity == G.gateway_route(_target(), "claude-code").identity + assert G.gateway_metadata(route)["gateway_agent"] == "claude-code" + + +def test_dell_claude_gateway_caps_message_output_at_one_eighth_of_context(): + route = G.gateway_route(_target(), "claude-code") + assert route is not None + assert route.output_limit == 20480 + request = { + "model": route.model, + "max_tokens": 32000, + "messages": [{"role": "user", "content": "start"}], + } + got = G.normalize_role_request(request, "anthropic_messages", {route.model: route.output_limit}) + assert got["max_tokens"] == 20480 + + +def test_dell_claude_gateway_uses_custom_openai_max_tokens_not_responses_output_key(): + route = G.gateway_route(_target(), "claude-code") + assert route is not None + request = { + "model": route.model, + "max_output_tokens": 32000, + "messages": [{"role": "user", "content": "start"}], + } + got = G.normalize_role_request(request, "anthropic_messages", {route.model: route.output_limit}) + assert got["max_tokens"] == 20480 + assert "max_output_tokens" not in got + + +def test_dell_claude_gateway_identity_and_model_change_with_its_context_cap(): + full = G.gateway_route(_target(), "claude-code") + smaller = G.gateway_route(replace(_target(), context_length=16384), "claude-code") + assert full is not None and smaller is not None + assert full.output_limit == 20480 and smaller.output_limit == 2048 + assert full.identity != smaller.identity + assert full.model != smaller.model + + +def test_gateway_does_not_change_the_passing_spark_claude_output_budget(): + route = G.gateway_route(_target("spark-deepseek"), "claude-code") + assert route is not None + assert route.output_limit is None + + +def test_spark_claude_gateway_strips_the_custom_openai_unsupported_output_key(): + route = G.gateway_route(_target("spark-deepseek"), "claude-code") + assert route is not None + got = G.normalize_role_request({ + "model": route.model, + "max_output_tokens": 32000, + "messages": [{"role": "user", "content": "start"}], + }, "anthropic_messages") + assert "max_output_tokens" not in got + assert "max_tokens" not in got + + +def test_gateway_codex_uses_a_non_websocket_custom_provider_config(): + setup = G.codex_gateway_setup_command() + assert setup.startswith("set -eu\n") + assert 'mkdir -p "$CODEX_HOME"' in setup + assert "python3 <<'PY'" in setup + assert "node <<" not in setup + assert 'model_provider = "harbor_compat"' in setup + assert 'wire_api = "responses"' in setup + assert "supports_websockets = false" in setup + assert 'base_url = "${OPENAI_BASE_URL}"' in setup + + +def test_gateway_codex_builds_truthful_catalog_from_runtime_instructions(tmp_path): + route = G.gateway_route(_target(), "codex") + assert route is not None + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + codex = bin_dir / "codex" + codex.write_text("""#!/bin/sh +cat <<'JSON' +{"models":[{"slug":"bundled","display_name":"Bundled","description":null, +"default_reasoning_level":"medium","supported_reasoning_levels":[{"effort":"medium","description":"Medium"}], +"shell_type":"shell_command","visibility":"list","supported_in_api":true,"priority":1, +"availability_nux":null,"upgrade":null, +"model_messages":{"instructions_template":"runtime-owned instructions","instructions_variables":null, +"approvals":null,"collaboration_modes":null,"auto_review":null,"permissions":null,"token_budget":null}, +"support_verbosity":true,"default_verbosity":"medium","apply_patch_tool_type":"freeform", +"truncation_policy":{"mode":"tokens","limit":32000},"supports_parallel_tool_calls":true, +"context_window":272000,"max_context_window":272000,"experimental_supported_tools":["custom"]}]} +JSON +""") + codex.chmod(0o755) + home = tmp_path / "codex-home" + completed = subprocess.run( + ["sh", "-c", G.codex_gateway_setup_command()], + env={ + **os.environ, + "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}", + "CODEX_HOME": str(home), + "OPENAI_BASE_URL": "http://gateway.test/v1", + "HARBOR_GATEWAY_CODEX_PROVIDER": "1", + "HARBOR_GATEWAY_CODEX_MODEL": route.model, + "HARBOR_GATEWAY_CODEX_SERVED_MODEL": "dot-backbone", + "HARBOR_GATEWAY_CODEX_CONTEXT": "163840", + }, + capture_output=True, + text=True, + ) + assert completed.returncode == 0, completed.stderr + catalog = json.loads((home / "model-catalog.json").read_text()) + assert catalog == {"models": [{ + **catalog["models"][0], + "slug": route.model, + "display_name": "dot-backbone", + "default_reasoning_level": None, + "supported_reasoning_levels": [], + "supports_reasoning_summary_parameter": False, + "support_verbosity": False, + "default_verbosity": None, + "apply_patch_tool_type": None, + "supports_parallel_tool_calls": False, + "context_window": 163840, + "max_context_window": 163840, + "experimental_supported_tools": [], + "model_messages": { + "instructions_template": "runtime-owned instructions", + "instructions_variables": None, + "approvals": None, + "collaboration_modes": None, + "auto_review": None, + "permissions": None, + "token_budget": None, + }, + }]} + config = (home / "config.toml").read_text() + assert f'model_catalog_json = "{home}/model-catalog.json"' in config + + +def test_gateway_codex_env_enables_only_its_custom_provider_setup(): + route = G.gateway_route(_target(), "codex") + assert route is not None + assert "codex-http-catalog-v8" in route.revision + env = G.gateway_agent_env(_target(), route) + assert env["HARBOR_GATEWAY_CODEX_PROVIDER"] == "1" + assert env["HARBOR_GATEWAY_CODEX_MODEL"] == route.model + assert env["HARBOR_GATEWAY_CODEX_SERVED_MODEL"] == "dot-backbone" + assert env["HARBOR_GATEWAY_CODEX_CONTEXT"] == "163840" + + +def test_gateway_process_env_precedes_any_existing_import_path_with_the_project_root(): + routed = G.gateway_process_env({"PYTHONPATH": "/existing", "OPENAI_API_KEY": "secret"}) + assert routed["PYTHONPATH"].split(os.pathsep) == [str(Path(G.__file__).resolve().parents[2]), "/existing"] + assert routed["OPENAI_API_KEY"] == "secret" + + +def test_gateway_codex_imports_through_the_harbor_runtime_subprocess(): + runtime = os.environ.get("HARBOR_RUNTIME_PYTHON") + if not runtime: + pytest.skip("requires the installed Harbor Python runtime") + root = Path(__file__).resolve().parents[1] + completed = subprocess.run( + [runtime, "-c", "from ingot.optimize.harbor_codex_gateway import GatewayCodex; " + "assert GatewayCodex.name() == 'codex'"], + cwd=root, env={**os.environ, "PYTHONPATH": str(root)}, capture_output=True, text=True, + ) + assert completed.returncode == 0, completed.stderr + + +def test_gateway_codex_writes_provider_setup_before_the_harbor_adapter(monkeypatch): + if importlib.util.find_spec("harbor") is None: + pytest.skip("requires the installed Harbor Python runtime") + from harbor.agents.installed.codex import Codex + from ingot.optimize.harbor_codex_gateway import GatewayCodex + + assert not any(flag.kwarg == "reasoning_effort" for flag in GatewayCodex.CLI_FLAGS) + calls = [] + + class ProbeGatewayCodex(GatewayCodex): + async def exec_as_agent(self, environment, command, **kwargs): + calls.append(("setup", command, kwargs)) + + async def fake_parent_run(self, instruction, environment, context): + calls.append(("parent", instruction)) + + monkeypatch.setattr(Codex, "run", fake_parent_run) + asyncio.run(object.__new__(ProbeGatewayCodex).run("task", object(), object())) + assert calls[0][0] == "setup" and "supports_websockets = false" in calls[0][1] + assert calls[0][2]["env"]["CODEX_HOME"] == "/tmp/codex-home" + assert calls[1] == ("parent", "task") + + +def test_normalizer_moves_anthropic_system_content_to_a_user_message_without_touching_tools(): + tool_use = {"type": "tool_use", "id": "call_1", "name": "shell", "input": {"cmd": "pwd"}} + tool_result = {"type": "tool_result", "tool_use_id": "call_1", "content": "ok"} + request = { + "system": "follow the repository instructions", + "messages": [ + {"role": "user", "content": "start"}, + {"role": "assistant", "content": [tool_use]}, + {"role": "user", "content": [tool_result]}, + ], + } + got = G.normalize_role_request(request, "anthropic_messages") + assert "system" not in got + assert got["messages"][0] == { + "role": "user", "content": "[system]\nfollow the repository instructions", + } + assert got["messages"][2]["content"] == [tool_use] + assert got["messages"][3]["content"] == [tool_result] + + +def test_normalizer_moves_responses_instructions_and_developer_roles_without_touching_tools(): + function_call = {"type": "function_call", "call_id": "call_1", "name": "shell", "arguments": "{}"} + function_output = {"type": "function_call_output", "call_id": "call_1", "output": "ok"} + request = { + "instructions": "follow the repository instructions", + "input": [ + {"role": "developer", "content": [{"type": "input_text", "text": "be concise"}]}, + {"role": "user", "content": [{"type": "input_text", "text": "start"}]}, + function_call, + function_output, + ], + } + got = G.normalize_role_request(request, "aresponses") + assert "instructions" not in got + assert got["input"][0] == { + "role": "user", + "content": [{"type": "input_text", "text": "[instructions]\nfollow the repository instructions"}], + } + assert got["input"][1]["role"] == "user" + assert got["input"][3] == function_call + assert got["input"][4] == function_output + + +def test_normalizer_strips_codex_reasoning_from_custom_openai_responses(): + request = { + "input": [{"role": "user", "content": [{"type": "input_text", "text": "start"}]}], + "reasoning": {"effort": "high", "summary": "auto"}, + "reasoning_effort": "high", + } + + got = G.normalize_role_request(request, "aresponses") + + assert "reasoning" not in got + assert "reasoning_effort" not in got + assert got["input"] == request["input"] + + +def test_normalizer_leaves_unrelated_routes_unchanged(): + request = {"messages": [{"role": "system", "content": "unchanged"}]} + assert G.normalize_role_request(request, "completion") == request + + +def test_gateway_session_requires_its_revisioned_models_and_writes_cleanup_receipt(tmp_path, monkeypatch): + route = G.gateway_route(_target(), "claude-code") + assert route is not None + + class Process: + def __init__(self): + self.stopped = False + + def poll(self): + return 0 if self.stopped else None + + def terminate(self): + self.stopped = True + + def wait(self, timeout): + return 0 + + process = Process() + calls = [] + spawned = {} + + class Response: + status = 200 + + def __init__(self, payload): + self.payload = payload + + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + def read(self): + return json.dumps(self.payload).encode() + + def fake_urlopen(request, timeout): + calls.append(request) + if isinstance(request, str): + return Response({"status": "healthy"}) + return Response({"data": [{"id": route.model}]}) + + def popen(*args, **kwargs): + spawned.update(kwargs) + return process + + monkeypatch.setattr(G.urllib.request, "urlopen", fake_urlopen) + session = G.GatewaySession([(route, _target())], tmp_path / "gateway", + litellm_bin=__file__, popen=popen) + session.start() + assert spawned["env"]["PYTHONPATH"].split(os.pathsep)[0] == str( + Path(G.__file__).resolve().parents[2]) + config = (tmp_path / "gateway" / "config.json").read_text() + assert "local.test" not in config and route.upstream_env in config + assert f"custom_openai/{route.served_model}" in config + assert any(isinstance(call, str) and call.endswith("/health/liveliness") for call in calls) + session.close() + receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text()) + assert receipt["stopped"] is True and receipt["routes"] == [route.identity] + + +def test_gateway_failed_start_still_writes_a_cleanup_receipt(tmp_path): + route = G.gateway_route(_target(), "claude-code") + assert route is not None + + def fail_start(*args, **kwargs): + raise OSError("cannot bind") + + session = G.GatewaySession([(route, _target())], tmp_path / "gateway", + litellm_bin=__file__, popen=fail_start) + with pytest.raises(OSError, match="cannot bind"): + session.start() + receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text()) + assert receipt["reason"] == "failed-start" and receipt["stopped"] is True + + +def test_gateway_missing_dedicated_runtime_fails_before_start_and_writes_receipt(tmp_path): + route = G.gateway_route(_target(), "claude-code") + assert route is not None + session = G.GatewaySession([(route, _target())], tmp_path / "gateway", + litellm_bin=str(tmp_path / "missing-litellm")) + with pytest.raises(RuntimeError, match="runtime is missing"): + session.start() + receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text()) + assert receipt["reason"] == "failed-start" + + +def test_default_gateway_runtime_is_checkout_relative(): + runtime = Path(G._LITELLM_BIN) + + assert runtime.is_absolute() + assert runtime == Path(G.__file__).resolve().parents[2] / ".venv-harbor-gateway/bin/litellm" + + +def test_gateway_fixed_bind_probe_is_bounded_and_closes_a_preexisting_listener(monkeypatch): + class Connection: + closed = False + + def close(self): + self.closed = True + + connection = Connection() + calls = [] + + def connect(address, timeout): + calls.append((address, timeout)) + return connection + + monkeypatch.setattr(G.socket, "create_connection", connect) + assert G._fixed_bind_occupied() is True + assert connection.closed is True + assert calls == [((G.GATEWAY_HOST, G.GATEWAY_PORT), 0.2)] + + +def test_gateway_rejects_any_preexisting_fixed_bind_before_spawn(tmp_path, monkeypatch): + route = G.gateway_route(_target(), "claude-code") + assert route is not None + monkeypatch.setattr(G, "_fixed_bind_occupied", lambda: True, raising=False) + + def unexpected_spawn(*args, **kwargs): + pytest.fail("must not spawn beside a stale gateway") + + session = G.GatewaySession([(route, _target())], tmp_path / "gateway", + litellm_bin=__file__, popen=unexpected_spawn) + with pytest.raises(RuntimeError, match="bind is already occupied"): + session.start() + receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text()) + assert receipt["reason"] == "failed-start" and receipt["stopped"] is True + + +def test_gateway_rejects_a_post_spawn_model_superset_and_cleans_up(tmp_path, monkeypatch): + route = G.gateway_route(_target(), "claude-code") + assert route is not None + + class Process: + stopped = False + + def poll(self): + return 0 if self.stopped else None + + def terminate(self): + self.stopped = True + + def wait(self, timeout): + return 0 + + process = Process() + monkeypatch.setattr(G, "_fixed_bind_occupied", lambda: False, raising=False) + + class Response: + status = 200 + + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + def read(self): + return json.dumps({"data": [{"id": route.model}, {"id": "stale-old-route"}]}).encode() + + def fake_urlopen(request, timeout): + return Response() + + monkeypatch.setattr(G.urllib.request, "urlopen", fake_urlopen) + session = G.GatewaySession([(route, _target())], tmp_path / "gateway", + litellm_bin=__file__, popen=lambda *args, **kwargs: process, + health_timeout=0.01) + with pytest.raises(RuntimeError, match="did not become healthy"): + session.start() + assert process.stopped is True + receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text()) + assert receipt["stopped"] is True + + +def test_gateway_requires_its_spawned_process_alive_after_exact_model_health(tmp_path, monkeypatch): + route = G.gateway_route(_target(), "claude-code") + assert route is not None + + class Process: + polls = [None, 0] + stopped = False + + def poll(self): + return self.polls.pop(0) if self.polls else 0 + + def terminate(self): + self.stopped = True + + def wait(self, timeout): + return 0 + + process = Process() + monkeypatch.setattr(G, "_fixed_bind_occupied", lambda: False, raising=False) + + class Response: + status = 200 + + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + def read(self): + return json.dumps({"data": [{"id": route.model}]}).encode() + + monkeypatch.setattr(G.urllib.request, "urlopen", lambda request, timeout: Response()) + session = G.GatewaySession([(route, _target())], tmp_path / "gateway", + litellm_bin=__file__, popen=lambda *args, **kwargs: process, + health_timeout=0.01) + with pytest.raises(RuntimeError, match="exited before health check"): + session.start() + assert process.stopped is False + receipt = json.loads((tmp_path / "gateway" / "cleanup.json").read_text()) + assert receipt["stopped"] is True diff --git a/tests/test_harbor_langfuse.py b/tests/test_harbor_langfuse.py new file mode 100644 index 0000000..bf52f1f --- /dev/null +++ b/tests/test_harbor_langfuse.py @@ -0,0 +1,1290 @@ +"""Harbor attempt export and receipt validation without live telemetry calls.""" +from __future__ import annotations + +import hashlib +import json +import os +import shutil +import threading +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +import pytest +from langfuse import Langfuse + +from ingot.optimize import harbor_langfuse as L + + +FIXTURE = Path(__file__).parent / "fixtures" / "harbor" / "langfuse-trial" +METADATA = { + "combination": "codex@fixture-model--dell-fixture", + "harness": "codex", + "model": "fixture-model", + "target_alias": "dell-fixture", + "endpoint_fingerprint": "f" * 64, + "protocol": "openai", + "task_fingerprint": "t" * 64, + "attempts": 1, + "arm": "skill", +} +SKILL_BODY = "fixture skill body\n" +PROVENANCE_METADATA = { + **METADATA, + "skill": "demo", + "skill_body": SKILL_BODY, + "skill_sha256": hashlib.sha256(SKILL_BODY.encode()).hexdigest(), + "task_texts": {"fixture-h0": "Use only deterministic fixture input."}, +} + + +def copy_trial(tmp_path: Path) -> Path: + trial = tmp_path / "job" / "fixture-h0__attempt-1" + shutil.copytree(FIXTURE, trial) + return trial + + +def payload_sha256(payload: dict) -> str: + encoded = json.dumps(payload, sort_keys=True, separators=(",", ":"), + ensure_ascii=False).encode("utf-8") + return hashlib.sha256(encoded).hexdigest() + + +class FakeObservation: + def __init__(self) -> None: + self.ended = 0 + self.id = "" + + def end(self) -> None: + self.ended += 1 + + +class FakeLangfuse: + def __init__(self, *, readback: str = "matching") -> None: + self.readback = readback + self.seeds: list[str] = [] + self.observations: list[dict] = [] + self.observation = FakeObservation() + self.flushes = 0 + self.reads: list[str] = [] + + def create_trace_id(self, *, seed: str) -> str: + self.seeds.append(seed) + return Langfuse.create_trace_id(seed=seed) + + def start_observation(self, **kwargs): + self.observations.append(kwargs) + self.observation.id = f"{len(self.observations):016x}" + return self.observation + + def flush(self) -> None: + self.flushes += 1 + + def read_trace(self, trace_id: str): + self.reads.append(trace_id) + if self.readback == "absent" or not self.observations: + return None + metadata = dict(self.observations[-1]["metadata"]) + if self.readback == "wrong-hash": + metadata["payload_sha256"] = "0" * 64 + if self.readback == "wrong-evidence": + metadata["telemetry_evidence"] = { + **metadata.get("telemetry_evidence", {}), "model": "wrong-model", + } + if self.readback == "wrong-id": + trace_id = "0" * 32 + return {"id": trace_id, "observations": [ + {"id": self.observation.id, "name": "harbor-attempt", + "type": "AGENT", "metadata": metadata}, + ]} + + +class FakeSdkOnly: + """Langfuse SDK surface without the test-only read_trace seam.""" + + def __init__(self) -> None: + self.delegate = FakeLangfuse() + + @property + def observations(self): + return self.delegate.observations + + def create_trace_id(self, *, seed: str) -> str: + return self.delegate.create_trace_id(seed=seed) + + def start_observation(self, **kwargs): + return self.delegate.start_observation(**kwargs) + + def flush(self) -> None: + self.delegate.flush() + + +def test_payload_captures_attempt_evidence_and_structurally_excludes_sensitive_fields(tmp_path): + """Removing a retained evidence field or copying Harbor config/env data must fail this test.""" + trial = copy_trial(tmp_path) + + payload = L.build_attempt_payload(trial, { + **METADATA, + "base_url": "https://metadata-endpoint.invalid/v1", + "skill_body": "PRIVATE SKILL BODY", + "skill_sha256": hashlib.sha256(b"PRIVATE SKILL BODY").hexdigest(), + "environment": {"TOKEN": "metadata-secret"}, + }) + + assert payload == { + "exporter_revision": "harbor-langfuse-v3", + "attempt": { + "id": "00000000-0000-0000-0000-000000000001", + "trial_name": "fixture-h0__attempt-1", + }, + "task": { + "name": "ingot/fixture-h0", + "checksum": "fixture-task-checksum", + "source": "fixture", + "text": "", + }, + "skill": "", + "trajectory": { + "steps": [ + {"source": "user", "timestamp": "2000-01-01T00:00:00Z"}, + {"source": "agent", "timestamp": "2000-01-01T00:00:00Z"}, + {"source": "agent", "timestamp": "2000-01-01T00:00:01Z"}, + ], + }, + "verifier_output": {"test-stdout.txt": ""}, + "solution_artifacts": { + "_objective_check.txt": ( + "Fixture objective check\n\n" + "- deterministic output exists\n" + "- no external endpoint was contacted\n" + "- no credential was used\n\nPASS\n" + ) + }, + "status": "failed", + "exception_category": "AgentSetupTimeoutError", + "error_detail": "sanitized fixture failure", + "timestamps": { + "started_at": "2000-01-01T00:00:00Z", + "finished_at": "2000-01-01T00:00:01Z", + "error_at": "2000-01-01T00:00:00Z", + }, + "usage": { + "input_tokens": 120, + "cache_tokens": 40, + "output_tokens": 30, + "cost_usd": 0.01, + }, + "model": "fixture-model", + "metadata": { + **METADATA, + "skill_sha256": hashlib.sha256(b"PRIVATE SKILL BODY").hexdigest(), + }, + } + rendered = json.dumps(payload, sort_keys=True) + for forbidden in ( + "fixture-secret-must-not-export", + "fixture-endpoint.invalid", + "metadata-endpoint.invalid", + "metadata-secret", + "PRIVATE SKILL BODY", + "/fixtures/skills/private-skill/SKILL.md", + '"environment"', + ): + assert forbidden not in rendered + + +def test_payload_bounds_text_and_omits_binary_solution_artifacts(tmp_path): + """An oversized or binary solution must never cross the telemetry boundary.""" + trial = copy_trial(tmp_path) + solution = trial / "verifier" / "solution" + (solution / "long.txt").write_text("prefix-" + "x" * 3000 + "-tail") + (solution / "output.bin").write_bytes(b"\x00\xff\x00private") + + payload = L.build_attempt_payload(trial, METADATA) + + assert "output.bin" not in payload["solution_artifacts"] + assert len(payload["solution_artifacts"]["long.txt"]) <= 2000 + assert payload["solution_artifacts"]["long.txt"].endswith("-tail") + + +def test_payload_accepts_a_bounded_large_trajectory_and_compacts_it(tmp_path): + """Harbor trajectories exceed artifact limits but export only a compact safe projection.""" + trial = copy_trial(tmp_path) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["steps"][0]["message"] = "x" * L._MAX_TEXT_BYTES + trajectory_path.write_text(json.dumps(trajectory)) + assert trajectory_path.stat().st_size > L._MAX_TEXT_BYTES + + payload = L.build_attempt_payload(trial, METADATA) + + assert len(payload["trajectory"]["steps"]) == 3 + assert "message" not in json.dumps(payload["trajectory"]) + + +def test_payload_accepts_observed_full_arm_goose_trajectory_and_compacts_it(tmp_path): + """Goose emitted a 3,368,055-byte trajectory whose freeform fields must be discarded.""" + trial = copy_trial(tmp_path) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["steps"][0]["message"] = "PRIVATE-GOOSE-MARKER" + "x" * (13 * 256 * 1024) + trajectory_path.write_text(json.dumps(trajectory)) + assert trajectory_path.stat().st_size > 3 * 1024 * 1024 + + payload = L.build_attempt_payload(trial, METADATA) + + assert len(payload["trajectory"]["steps"]) == 3 + assert "PRIVATE-GOOSE-MARKER" not in json.dumps(payload["trajectory"]) + + +def test_export_discovers_a_valid_large_trajectory_and_compacts_it(tmp_path): + trial = copy_trial(tmp_path) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["steps"][0]["message"] = "PRIVATE-LARGE-MARKER" + "x" * L._MAX_TEXT_BYTES + trajectory_path.write_text(json.dumps(trajectory)) + assert trajectory_path.stat().st_size > L._MAX_TEXT_BYTES + client = FakeLangfuse() + + receipts = L.export_job_attempts(trial.parent, METADATA, client=client) + + assert len(receipts) == 1 + assert len(client.observations) == 1 + assert "PRIVATE-LARGE-MARKER" not in json.dumps(client.observations) + + +def test_payload_rejects_a_trajectory_above_its_dedicated_bound(tmp_path): + """The larger fixed-file allowance must remain bounded against agent-controlled input.""" + trial = copy_trial(tmp_path) + trajectory_path = trial / "agent" / "trajectory.json" + with trajectory_path.open("wb") as handle: + handle.seek(L._MAX_TRAJECTORY_BYTES) + handle.write(b"tail") + + with pytest.raises(L.TelemetryReceiptError, match="byte budget"): + L.build_attempt_payload(trial, METADATA) + + +def test_payload_projects_only_allowlisted_trajectory_fields(tmp_path): + """Agent step IDs and arbitrary numeric metrics cannot cross the telemetry boundary.""" + trial = copy_trial(tmp_path) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["final_metrics"] = { + "total_prompt_tokens": 80, + "total_cached_tokens": 25, + "total_completion_tokens": 20, + "total_cost_usd": 0.005, + "untrusted_numeric_field": 123, + } + trajectory_path.write_text(json.dumps(trajectory)) + + projected = L.build_attempt_payload(trial, METADATA)["trajectory"] + + assert all("step_id" not in step for step in projected["steps"]) + assert projected["final_metrics"] == { + "total_prompt_tokens": 80, + "total_cached_tokens": 25, + "total_completion_tokens": 20, + "total_cost_usd": 0.005, + } + + +def test_payload_treats_missing_trajectory_steps_as_empty(tmp_path): + trial = copy_trial(tmp_path) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + del trajectory["steps"] + trajectory_path.write_text(json.dumps(trajectory)) + + assert L.build_attempt_payload(trial, METADATA)["trajectory"]["steps"] == [] + + +def test_payload_rejects_non_list_trajectory_steps(tmp_path): + trial = copy_trial(tmp_path) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["steps"] = {"source": "agent"} + trajectory_path.write_text(json.dumps(trajectory)) + + with pytest.raises(L.TelemetryReceiptError, match="trajectory steps"): + L.build_attempt_payload(trial, METADATA) + + +def test_payload_trajectory_projection_drops_freeform_strings(tmp_path): + trial = copy_trial(tmp_path) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["schema_version"] = "PRIVATE-SCHEMA-MARKER" + trajectory["session_id"] = "PRIVATE-SESSION-MARKER" + trajectory["agent"] = { + "name": "PRIVATE-AGENT-MARKER", + "version": "PRIVATE-VERSION-MARKER", + "model_name": "PRIVATE-MODEL-MARKER", + } + trajectory["steps"] = [ + {"source": "user", "timestamp": "2000-01-01T00:00:00Z", "message": "PRIVATE"}, + {"source": "agent", "timestamp": "2000-01-01T00:00:01Z"}, + {"source": "tool", "timestamp": "2000-01-01T00:00:02Z"}, + {"source": "PRIVATE-SOURCE-MARKER", "timestamp": "not-a-timestamp"}, + ] + trajectory_path.write_text(json.dumps(trajectory)) + + projected = L.build_attempt_payload(trial, METADATA)["trajectory"] + + assert projected == {"steps": [ + {"source": "user", "timestamp": "2000-01-01T00:00:00Z"}, + {"source": "agent", "timestamp": "2000-01-01T00:00:01Z"}, + {"timestamp": "2000-01-01T00:00:02Z"}, + {}, + ]} + + +def test_payload_model_does_not_fall_back_to_freeform_trajectory_agent(tmp_path): + trial = copy_trial(tmp_path) + result_path = trial / "result.json" + result = json.loads(result_path.read_text()) + result["agent_info"] = {} + result_path.write_text(json.dumps(result)) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["agent"]["model_name"] = "PRIVATE-MODEL-MARKER" + trajectory_path.write_text(json.dumps(trajectory)) + metadata = {key: value for key, value in METADATA.items() if key != "model"} + + assert L.build_attempt_payload(trial, metadata)["model"] == "" + + +def test_payload_rejects_solution_symlink_outside_the_verifier_root(tmp_path): + """Following an agent-controlled artifact symlink would export a parent-process host file.""" + trial = copy_trial(tmp_path) + outside = tmp_path / "parent-secret.txt" + outside.write_text("outside-parent-secret") + artifact = trial / "verifier" / "solution" / "outside.txt" + artifact.symlink_to(outside) + + with pytest.raises(L.TelemetryReceiptError, match="admission"): + L.build_attempt_payload(trial, METADATA) + + +def test_payload_omits_large_sparse_artifact_before_decoding(tmp_path, monkeypatch): + """A sparse optional artifact is omitted by size before its contents are allocated.""" + trial = copy_trial(tmp_path) + artifact = trial / "verifier" / "solution" / "large.txt" + with artifact.open("wb") as handle: + handle.seek(8 * 1024 * 1024) + handle.write(b"tail") + original_read_bytes = Path.read_bytes + + def reject_full_read(path): + if path == artifact: + pytest.fail("oversized artifact was read before its byte cap was checked") + return original_read_bytes(path) + + monkeypatch.setattr(Path, "read_bytes", reject_full_read) + + payload = L.build_attempt_payload(trial, METADATA) + + assert "large.txt" not in payload["solution_artifacts"] + assert payload["artifact_projection"]["omitted_files"] == 1 + + +def test_payload_rejects_nul_free_control_byte_binary(tmp_path): + """Valid UTF-8 with binary control bytes is not a text artifact.""" + trial = copy_trial(tmp_path) + artifact = trial / "verifier" / "solution" / "control.txt" + artifact.write_bytes(b"visible\x01binary\x02content") + + payload = L.build_attempt_payload(trial, METADATA) + + assert "control.txt" not in payload["solution_artifacts"] + + +def test_export_rejects_a_symlinked_trial_before_reading_outside_the_job(tmp_path): + """A job entry cannot redefine an outside directory as a persisted attempt.""" + outside_trial = copy_trial(tmp_path / "outside") + job = tmp_path / "job" + job.mkdir() + linked_trial = job / "fixture-h0__attempt-1" + linked_trial.symlink_to(outside_trial, target_is_directory=True) + client = FakeLangfuse() + + with pytest.raises(L.TelemetryReceiptError, match="symlink|directory|admission"): + L.export_job_attempts(job, METADATA, client=client) + + assert client.observations == [] + assert not (outside_trial / L.RECEIPT_NAME).exists() + + +def test_export_rejects_a_nonregular_verifier_entry(tmp_path): + """Silently skipping an agent-controlled pipe would make exported evidence partial.""" + trial = copy_trial(tmp_path) + os.mkfifo(trial / "verifier" / "solution" / "stream.txt") + client = FakeLangfuse() + + with pytest.raises(L.TelemetryReceiptError, match="non-regular"): + L.export_job_attempts(trial.parent, METADATA, client=client) + + assert client.observations == [] + assert not (trial / L.RECEIPT_NAME).exists() + + +def test_export_rejects_an_intermediate_directory_swap(tmp_path, monkeypatch): + """A renamed solution directory cannot redirect a later open outside the attempt.""" + trial = copy_trial(tmp_path) + solution = trial / "verifier" / "solution" + artifact = solution / "answer.txt" + artifact.write_text("inside fixture text") + outside = tmp_path / "outside-solution" + outside.mkdir() + (outside / "answer.txt").write_text("PARENT-ONLY-SECRET") + original_open = os.open + swapped = False + + def swap_before_artifact_open(path, flags, *args, **kwargs): + nonlocal swapped + if not swapped and Path(path).name == "answer.txt": + swapped = True + held = solution.with_name("held-solution") + solution.rename(held) + solution.symlink_to(outside, target_is_directory=True) + return original_open(path, flags, *args, **kwargs) + + monkeypatch.setattr(os, "open", swap_before_artifact_open) + client = FakeLangfuse() + + with pytest.raises(L.TelemetryReceiptError, match="changed|admission"): + L.export_job_attempts(trial.parent, METADATA, client=client) + + assert client.observations == [] + assert not (trial / L.RECEIPT_NAME).exists() + + +def test_export_projects_many_small_artifacts_and_hidden_housekeeping_deterministically(tmp_path): + """Optional telemetry stays bounded without rejecting authoritative Harbor evidence.""" + trial = copy_trial(tmp_path) + solution = trial / "verifier" / "solution" + hidden_target = tmp_path / "hidden-backup" + hidden_target.mkdir() + (hidden_target / "secret.txt").write_text("must-not-export") + (solution / ".backups").symlink_to(hidden_target, target_is_directory=True) + for index in range(129): + (solution / f"part-{index:03}.txt").write_text("x") + client = FakeLangfuse() + + receipts = L.export_job_attempts(trial.parent, METADATA, client=client) + + assert len(receipts) == len(client.observations) == 1 + payload = client.observations[0]["output"] + exported = payload["solution_artifacts"] + assert "part-000.txt" in exported + assert "part-128.txt" not in exported + assert ".backups" not in json.dumps(payload) + assert "must-not-export" not in json.dumps(payload) + assert payload["artifact_projection"]["omitted_hidden_paths"] == 1 + assert payload["artifact_projection"]["omitted_files"] >= 1 + assert payload["artifact_projection"]["exported_files"] <= L._MAX_ARTIFACT_PATHS + assert (trial / L.RECEIPT_NAME).is_file() + + +def test_export_projects_aggregate_artifact_bytes_above_the_attempt_budget(tmp_path): + """Optional files beyond the byte ceiling are omitted without blocking evidence.""" + trial = copy_trial(tmp_path) + solution = trial / "verifier" / "solution" + for index in range(80): + (solution / f"chunk-{index:03}.txt").write_text("x" * 2000) + client = FakeLangfuse() + + receipts = L.export_job_attempts(trial.parent, METADATA, client=client) + + assert len(receipts) == len(client.observations) == 1 + payload = client.observations[0]["output"] + assert payload["artifact_projection"]["exported_bytes"] <= L._MAX_ARTIFACT_BYTES + assert payload["artifact_projection"]["omitted_files"] > 0 + assert (trial / L.RECEIPT_NAME).is_file() + + +def test_export_prioritizes_solution_answer_over_verifier_noise(tmp_path): + trial = copy_trial(tmp_path) + verifier = trial / "verifier" + solution = verifier / "solution" + (solution / "answer.md").write_text("graded deliverable") + for index in range(130): + (verifier / f"noise-{index:03}.txt").write_text("noise") + client = FakeLangfuse() + + L.export_job_attempts(trial.parent, METADATA, client=client) + + payload = client.observations[0]["output"] + assert payload["solution_artifacts"]["answer.md"] == "graded deliverable" + assert client.observations[0]["metadata"]["exporter_revision"] == "harbor-langfuse-v3" + + +def test_export_migrates_verified_v2_receipt_without_rerunning_attempt(tmp_path): + trial = copy_trial(tmp_path) + client = FakeLangfuse() + old_digest = "a" * 64 + old_trace = Langfuse.create_trace_id(seed=f"harbor-langfuse-v2:{old_digest}") + evidence = L._telemetry_evidence(L.build_attempt_payload(trial, METADATA)) + old_receipt = { + "status": "verified", "trace_id": old_trace, "payload_sha256": old_digest, + "exporter_revision": "harbor-langfuse-v2", + } + (trial / L.RECEIPT_NAME).write_text(json.dumps(old_receipt)) + client.observation.id = "old-observation" + client.observations.append({"metadata": { + "payload_sha256": old_digest, "exporter_revision": "harbor-langfuse-v2", + "telemetry_evidence": evidence, + }}) + default_read = client.read_trace + new_reads = 0 + + def versioned_read(trace_id): + nonlocal new_reads + if trace_id == old_trace: + return {"id": old_trace, "observations": [{ + "id": "old-observation", "name": "harbor-attempt", "type": "AGENT", + "metadata": client.observations[0]["metadata"], + }]} + if len(client.observations) == 1: + new_reads += 1 + if new_reads == 1: + raise TimeoutError("transient v3 read failure") + return None + return default_read(trace_id) + + client.read_trace = versioned_read + + with pytest.raises(TimeoutError, match="transient v3 read failure"): + L.export_job_attempts(trial.parent, METADATA, client=client) + + assert json.loads((trial / L.RECEIPT_NAME).read_text()) == old_receipt + assert not (trial / L.LEGACY_V2_RECEIPT_NAME).exists() + + [receipt] = L.export_job_attempts(trial.parent, METADATA, client=client) + + assert json.loads((trial / L.LEGACY_V2_RECEIPT_NAME).read_text()) == old_receipt + assert receipt["status"] == "verified" + assert receipt["exporter_revision"] == "harbor-langfuse-v3" + assert json.loads((trial / L.RECEIPT_NAME).read_text()) == receipt + assert len(client.observations) == 2 + + +def test_export_projects_scan_and_depth_exhaustion(tmp_path, monkeypatch): + trial = copy_trial(tmp_path) + solution = trial / "verifier" / "solution" + nested = solution + for index in range(4): + nested = nested / f"level-{index}" + nested.mkdir() + (nested / "deep.txt").write_text("deep") + (solution / "first.txt").write_text("first") + (solution / "second.txt").write_text("second") + monkeypatch.setattr(L, "_MAX_ARTIFACT_DEPTH", 2) + monkeypatch.setattr(L, "_MAX_ARTIFACT_SCAN_PATHS", 6) + client = FakeLangfuse() + + L.export_job_attempts(trial.parent, METADATA, client=client) + + projection = client.observations[0]["output"]["artifact_projection"] + assert projection["omitted_files"] > 0 + assert len(client.observations) == 1 + + +def test_payload_uses_trajectory_metrics_when_result_usage_is_unavailable(tmp_path): + """Dropping Harbor's alternate usage shape would erase usage from successful adapters.""" + trial = copy_trial(tmp_path) + result_path = trial / "result.json" + result = json.loads(result_path.read_text()) + result["agent_result"] = { + "n_input_tokens": None, "n_cache_tokens": None, + "n_output_tokens": None, "cost_usd": None, + } + result_path.write_text(json.dumps(result)) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["final_metrics"] = { + "total_prompt_tokens": 80, + "total_cached_tokens": 25, + "total_completion_tokens": 20, + "total_cost_usd": 0.005, + } + trajectory_path.write_text(json.dumps(trajectory)) + + assert L.build_attempt_payload(trial, METADATA)["usage"] == { + "input_tokens": 80, + "cache_tokens": 25, + "output_tokens": 20, + "cost_usd": 0.005, + } + + +def test_canonical_payload_and_hash_do_not_depend_on_parent_secret_environment(tmp_path, monkeypatch): + """A credential rotation must not change the identity of unchanged persisted evidence.""" + trial = copy_trial(tmp_path) + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["steps"][0]["message"] = "rotated-value" + trajectory_path.write_text(json.dumps(trajectory)) + (trial / "verifier" / "solution" / "answer.txt").write_text("rotated-value") + monkeypatch.delenv("CUSTOM_API_KEY", raising=False) + first = L.build_attempt_payload(trial, PROVENANCE_METADATA) + first_hash = payload_sha256(first) + monkeypatch.setenv("CUSTOM_API_KEY", "rotated-value") + + second = L.build_attempt_payload(trial, PROVENANCE_METADATA) + + assert second == first + assert payload_sha256(second) == first_hash + + +def test_known_skill_body_is_omitted_from_trajectory_and_solution_artifacts(tmp_path): + """Known instruction content must not be exported as attempt evidence.""" + trial = copy_trial(tmp_path) + skill_body = "PRIVATE SKILL BODY\n" + metadata = { + **PROVENANCE_METADATA, + "skill_body": skill_body, + "skill_sha256": hashlib.sha256(skill_body.encode()).hexdigest(), + } + trajectory_path = trial / "agent" / "trajectory.json" + trajectory = json.loads(trajectory_path.read_text()) + trajectory["steps"][0]["message"] = skill_body + trajectory_path.write_text(json.dumps(trajectory)) + (trial / "verifier" / "solution" / "skill.md").write_text(skill_body) + + payload = L.build_attempt_payload(trial, metadata) + + assert "PRIVATE SKILL BODY" not in json.dumps(payload) + assert "message" not in json.dumps(payload["trajectory"]) + assert "skill.md" not in payload["solution_artifacts"] + + +def test_named_sensitive_artifact_classes_are_structurally_excluded(tmp_path): + """Known config, environment, credential, secret, and endpoint dumps never export.""" + trial = copy_trial(tmp_path) + solution = trial / "verifier" / "solution" + for name in ( + "config.json", "configuration.txt", "environment.json", "env.txt", + "credential.txt", "credentials.log", "secret.txt", "secrets.yaml", + "endpoint.json", "endpoints.txt", + ): + (solution / name).write_text(f"private marker from {name}") + + payload = L.build_attempt_payload(trial, METADATA) + + rendered = json.dumps(payload["solution_artifacts"]) + assert "private marker" not in rendered + + +def test_artifact_wrapping_the_verified_known_skill_body_is_excluded(tmp_path): + """Adding a heading cannot disguise the exact producer-supplied skill body.""" + trial = copy_trial(tmp_path) + skill_body = "PRIVATE SKILL BODY\n" + metadata = { + **PROVENANCE_METADATA, + "skill_body": skill_body, + "skill_sha256": hashlib.sha256(skill_body.encode()).hexdigest(), + } + wrapped = trial / "verifier" / "solution" / "wrapped.md" + wrapped.write_text(f"# Copied instructions\n\n{skill_body}\nFixture answer") + + payload = L.build_attempt_payload(trial, metadata) + + assert "wrapped.md" not in payload["solution_artifacts"] + assert "PRIVATE SKILL BODY" not in json.dumps(payload) + + +def _shape_trial(tmp_path: Path, shape: str) -> Path: + trial = tmp_path / shape / "fixture-h0__attempt-1" + trial.mkdir(parents=True) + if shape in {"success", "failure"}: + result = { + "id": f"fixture-{shape}", + "task_name": "ingot/fixture-h0", + "trial_name": "fixture-h0__attempt-1", + "task_checksum": "fixture-checksum", + "source": "fixture", + "started_at": "2000-01-01T00:00:00Z", + "finished_at": "2000-01-01T00:00:01Z", + "exception_info": None, + } + if shape == "failure": + result["exception_info"] = { + "exception_type": "RuntimeError", + "exception_message": "bounded fixture failure", + "exception_traceback": "private traceback must not export", + "occurred_at": "2000-01-01T00:00:00.500Z", + } + (trial / "result.json").write_text(json.dumps(result)) + elif shape == "trajectory-only": + trajectory = trial / "agent" / "trajectory.json" + trajectory.parent.mkdir() + trajectory.write_text(json.dumps({ + "schema_version": "ATIF-v1.1", + "session_id": "fixture-session", + "agent": {"name": "codex", "version": "v1", "model_name": "fixture-model"}, + "steps": [ + {"step_id": 1, "timestamp": "2000-01-01T00:00:00Z", + "source": "agent", "message": "private free text"}, + {"step_id": 2, "timestamp": "2000-01-01T00:00:01Z", + "source": "agent", "message": "private free text"}, + ], + })) + elif shape == "verifier-only": + verifier = trial / "verifier" / "test-stdout.txt" + verifier.parent.mkdir() + verifier.write_text("fixture verifier output") + elif shape == "solution-only": + solution = trial / "verifier" / "solution" / "answer.md" + solution.parent.mkdir(parents=True) + solution.write_text("fixture solution output") + return trial + + +@pytest.mark.parametrize("shape, expected", [ + ("success", { + "status": "succeeded", "error_detail": None, + "timestamps": {"started_at": "2000-01-01T00:00:00Z", + "finished_at": "2000-01-01T00:00:01Z"}, + }), + ("failure", { + "status": "failed", "error_detail": "bounded fixture failure", + "timestamps": {"started_at": "2000-01-01T00:00:00Z", + "finished_at": "2000-01-01T00:00:01Z", + "error_at": "2000-01-01T00:00:00.500Z"}, + }), + ("trajectory-only", { + "status": "incomplete", "error_detail": None, + "timestamps": {"started_at": "2000-01-01T00:00:00Z", + "finished_at": "2000-01-01T00:00:01Z"}, + }), + ("verifier-only", {"status": "incomplete", "error_detail": None, "timestamps": {}}), + ("solution-only", {"status": "incomplete", "error_detail": None, "timestamps": {}}), +]) +def test_attempt_shapes_have_exact_terminal_status_and_provenance(tmp_path, shape, expected): + """Missing terminal result evidence must never be promoted to a successful status.""" + payload = L.build_attempt_payload(_shape_trial(tmp_path, shape), PROVENANCE_METADATA) + + assert {key: payload[key] for key in ("status", "error_detail", "timestamps")} == expected + assert payload["skill"] == "demo" + assert payload["task"] == { + "name": "ingot/fixture-h0" if shape in {"success", "failure"} else "fixture-h0", + "checksum": "fixture-checksum" if shape in {"success", "failure"} else "", + "source": "fixture" if shape in {"success", "failure"} else "", + "text": "Use only deterministic fixture input.", + } + + +def test_export_writes_receipt_only_after_deterministic_trace_readback(tmp_path): + """A wrong trace seed, observation contract, or unverified receipt must fail this test.""" + trial = copy_trial(tmp_path) + client = FakeLangfuse() + + receipts = L.export_job_attempts(trial.parent, METADATA, client=client) + + payload = L.build_attempt_payload(trial, METADATA) + digest = payload_sha256(payload) + trace_id = Langfuse.create_trace_id(seed=f"harbor-langfuse-v3:{digest}") + expected = { + "status": "verified", + "trace_id": trace_id, + "payload_sha256": digest, + "exporter_revision": "harbor-langfuse-v3", + } + assert receipts == [expected] + assert json.loads((trial / "langfuse-receipt.json").read_text()) == expected + assert client.seeds == [f"harbor-langfuse-v3:{digest}"] + assert client.flushes == 1 and client.reads == [expected["trace_id"], expected["trace_id"]] + assert client.observation.ended == 1 + observation = client.observations[0] + assert observation["trace_context"] == {"trace_id": expected["trace_id"]} + assert observation["name"] == "harbor-attempt" and observation["as_type"] == "agent" + assert observation["input"] == payload["task"] + assert observation["output"]["status"] == "failed" + assert observation["metadata"]["payload_sha256"] == digest + assert observation["metadata"]["telemetry_evidence"] == { + "revision": "harbor-agent-evidence-v1", + "model": "fixture-model", + "status": "failed", + "tokens": {"input_tokens": 120, "cache_tokens": 40, "output_tokens": 30}, + } + assert "model" not in observation + assert "usage_details" not in observation + assert "cost_details" not in observation + assert observation["level"] == "ERROR" and observation["status_message"] == "failed" + + +def test_real_langfuse_agent_sink_uses_sdk_ids_for_two_payloads_with_existing_provider(tmp_path): + """Langfuse 4.14 owns span IDs while bounded agent metadata survives its real agent path.""" + from opentelemetry.sdk.trace import TracerProvider + from opentelemetry.sdk.trace.export import SimpleSpanProcessor + from opentelemetry.sdk.trace.export.in_memory_span_exporter import InMemorySpanExporter + + first = copy_trial(tmp_path) + second = first.parent / "fixture-h1__attempt-2" + shutil.copytree(FIXTURE, second) + result_path = second / "result.json" + result = json.loads(result_path.read_text()) + result["id"] = "00000000-0000-0000-0000-000000000002" + result["task_name"] = "ingot/fixture-h1" + result["trial_name"] = second.name + result_path.write_text(json.dumps(result)) + preexisting_sink = InMemorySpanExporter() + provider = TracerProvider() + provider.add_span_processor(SimpleSpanProcessor(preexisting_sink)) + sink = InMemorySpanExporter() + sdk = Langfuse( + public_key="pk-task4-real-sink-two-payloads", + secret_key="sk-offline-fixture", + tracer_provider=provider, + span_exporter=sink, + ) + + class SinkReadback: + def create_trace_id(self, **kwargs): + return sdk.create_trace_id(**kwargs) + + def start_observation(self, **kwargs): + return sdk.start_observation(**kwargs) + + def flush(self): + sdk.flush() + + def read_trace(self, trace_id): + observations = [] + for span in sink.get_finished_spans(): + if f"{span.context.trace_id:032x}" != trace_id: + continue + metadata = {} + prefix = "langfuse.observation.metadata." + for key, value in span.attributes.items(): + if key.startswith(prefix): + try: + metadata[key.removeprefix(prefix)] = json.loads(value) + except (TypeError, ValueError): + metadata[key.removeprefix(prefix)] = value + observations.append({ + "id": f"{span.context.span_id:016x}", + "name": span.name, + "type": span.attributes["langfuse.observation.type"], + "metadata": metadata, + }) + return {"id": trace_id, "observations": observations} if observations else None + + receipts = L.export_job_attempts(first.parent, METADATA, client=SinkReadback()) + + spans = sink.get_finished_spans() + assert len(receipts) == 2 and all(item["status"] == "verified" for item in receipts) + assert len(spans) == 2 + assert len(preexisting_sink.get_finished_spans()) == 2 + assert len({span.context.span_id for span in spans}) == 2 + for span in spans: + assert span.attributes["langfuse.observation.type"] == "agent" + evidence = json.loads(span.attributes["langfuse.observation.metadata.telemetry_evidence"]) + assert evidence == { + "revision": "harbor-agent-evidence-v1", + "model": "fixture-model", + "status": "failed", + "tokens": {"input_tokens": 120, "cache_tokens": 40, "output_tokens": 30}, + } + assert "langfuse.observation.model.name" not in span.attributes + assert "langfuse.observation.usage_details" not in span.attributes + assert "langfuse.observation.cost_details" not in span.attributes + + +def test_repeated_export_reuses_the_verified_receipt_without_another_trace(tmp_path): + """Dropping receipt reuse would duplicate a deterministic attempt in Langfuse.""" + trial = copy_trial(tmp_path) + first = FakeLangfuse() + receipt = L.export_job_attempts(trial.parent, METADATA, client=first) + must_not_export = FakeLangfuse(readback="absent") + + assert L.export_job_attempts(trial.parent, METADATA, client=must_not_export) == receipt + assert must_not_export.observations == [] + assert must_not_export.flushes == 0 and must_not_export.reads == [] + + +def test_production_readback_uses_parent_basic_auth_and_public_trace_endpoint(tmp_path, monkeypatch): + """Wrong auth or endpoint would make an SDK enqueue look verified without public read-back.""" + trial = copy_trial(tmp_path) + client = FakeSdkOnly() + requests = [] + monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-fixture") + monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-fixture") + monkeypatch.setenv("LANGFUSE_BASE_URL", "https://langfuse.fixture.invalid/") + + class Response: + status_code = 200 + + def json(self): + return { + "id": client.delegate.seeds[-1] and Langfuse.create_trace_id( + seed=client.delegate.seeds[-1]), + "observations": [{"id": client.delegate.observation.id, + "name": "harbor-attempt", "type": "AGENT", + "metadata": client.observations[-1]["metadata"]}], + } + + class MissingResponse: + status_code = 404 + + def fake_get(url, **kwargs): + requests.append((url, kwargs)) + return Response() if client.observations else MissingResponse() + + monkeypatch.setattr(L.httpx, "get", fake_get) + + L.export_job_attempts(trial.parent, METADATA, client=client) + + trace_id = Langfuse.create_trace_id(seed=client.delegate.seeds[-1]) + assert requests == [ + (f"https://langfuse.fixture.invalid/api/public/traces/{trace_id}", + {"auth": ("pk-fixture", "sk-fixture"), "timeout": 15}), + (f"https://langfuse.fixture.invalid/api/public/traces/{trace_id}", + {"auth": ("pk-fixture", "sk-fixture"), "timeout": 15}), + ] + + +def test_production_readback_error_fails_before_emitting_an_observation(tmp_path, monkeypatch): + """Only a confirmed missing deterministic trace permits a first observation.""" + trial = copy_trial(tmp_path) + client = FakeSdkOnly() + monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-fixture") + monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-fixture") + + class UnauthorizedResponse: + status_code = 401 + + monkeypatch.setattr(L.httpx, "get", lambda *args, **kwargs: UnauthorizedResponse()) + + with pytest.raises(L.TelemetryReceiptError, match="status 401"): + L.export_job_attempts(trial.parent, METADATA, client=client) + + assert client.observations == [] + + +@pytest.mark.parametrize("readback", ["absent", "wrong-id", "wrong-hash", "wrong-evidence"]) +def test_unmatched_public_readback_never_creates_a_verified_receipt(tmp_path, readback): + """Accepting absent or mismatched public readback would turn pending state into proof.""" + trial = copy_trial(tmp_path) + + with pytest.raises(L.TelemetryReceiptError, match="read-back"): + L.export_job_attempts(trial.parent, METADATA, client=FakeLangfuse(readback=readback)) + + receipt = json.loads((trial / "langfuse-receipt.json").read_text()) + assert receipt["status"] == "pending" + + +def test_failed_attempt_is_traced_and_telemetry_failure_does_not_mutate_harbor_result(tmp_path): + """Failure telemetry must describe persisted evidence, never rewrite or rerun it.""" + trial = copy_trial(tmp_path) + before = (trial / "result.json").read_bytes() + client = FakeLangfuse(readback="absent") + + with pytest.raises(L.TelemetryReceiptError): + L.export_job_attempts(trial.parent, METADATA, client=client) + + assert client.observations[0]["output"]["status"] == "failed" + assert (trial / "result.json").read_bytes() == before + assert json.loads((trial / "result.json").read_text())["verifier_result"]["rewards"] == {"reward": 1.0} + + +@pytest.mark.parametrize("state", [ + "empty", "withheld", "stale", "malformed", "empty-id", "whitespace-id", "wrong-id", +]) +def test_receipt_validation_rejects_every_incomplete_telemetry_state(tmp_path, state): + """No empty, withheld, stale, or malformed receipt may silently publish a matrix.""" + trial = copy_trial(tmp_path) + receipt = trial / "langfuse-receipt.json" + if state == "withheld": + receipt.write_text(json.dumps({"status": "withheld"})) + elif state == "stale": + receipt.write_text(json.dumps({ + "status": "verified", + "trace_id": "1234567890abcdef1234567890abcdef", + "payload_sha256": "0" * 64, + "exporter_revision": "harbor-langfuse-v1", + })) + elif state == "malformed": + receipt.write_text("{not-json") + elif state in {"empty-id", "whitespace-id", "wrong-id"}: + payload = L.build_attempt_payload(trial, METADATA) + receipt.write_text(json.dumps({ + "status": "verified", + "trace_id": {"empty-id": "", "whitespace-id": " ", "wrong-id": "0" * 32}[state], + "payload_sha256": payload_sha256(payload), + "exporter_revision": "harbor-langfuse-v1", + })) + + with pytest.raises(L.TelemetryReceiptError, match="receipt"): + L.validate_job_receipts(trial.parent, METADATA) + + +def test_receipt_validation_rejects_an_exact_provenance_free_receipt(tmp_path): + """A matching old payload identity cannot authorize publication without provenance.""" + trial = copy_trial(tmp_path) + payload = L.build_attempt_payload(trial, METADATA) + digest = payload_sha256(payload) + (trial / L.RECEIPT_NAME).write_text(json.dumps({ + "status": "verified", + "trace_id": Langfuse.create_trace_id(seed=f"{L.EXPORTER_REVISION}:{digest}"), + "payload_sha256": digest, + "exporter_revision": L.EXPORTER_REVISION, + })) + + with pytest.raises(L.TelemetryReceiptError, match="provenance"): + L.validate_job_receipts(trial.parent, METADATA) + + +class DelayedReadbackLangfuse(FakeLangfuse): + """First post-flush read misses; the next preflight sees the persisted observation.""" + + def read_trace(self, trace_id: str): + self.reads.append(trace_id) + if not self.observations or (len(self.observations) == 1 and len(self.reads) < 3): + return None + metadata = dict(self.observations[0]["metadata"]) + return { + "id": trace_id, + "observations": [{"id": self.observation.id, + "name": "harbor-attempt", "type": "AGENT", + "metadata": metadata}], + } + + +def test_post_flush_readback_polls_before_deferring_finalization(tmp_path, monkeypatch): + """A short indexing lag must finalize in one exporter pass without a duplicate trace.""" + trial = copy_trial(tmp_path) + client = DelayedReadbackLangfuse() + sleeps = [] + monkeypatch.setattr("time.sleep", sleeps.append) + + receipts = L.export_job_attempts(trial.parent, METADATA, client=client) + + assert len(receipts) == 1 + assert len(client.observations) == 1 + assert client.flushes == 1 + assert len(client.reads) == 3 + assert sleeps == [L._READBACK_POLL_SECONDS] + + +def test_repeated_404_after_unknown_dispatch_never_emits_a_second_observation(tmp_path): + """A point-in-time missing trace cannot authorize another dispatch after flush.""" + trial = copy_trial(tmp_path) + client = FakeLangfuse(readback="absent") + + for _attempt in range(2): + with pytest.raises(L.TelemetryReceiptError): + L.export_job_attempts(trial.parent, METADATA, client=client) + + pending = json.loads((trial / L.RECEIPT_NAME).read_text()) + assert pending["status"] == "pending" + assert pending["observation_id"] == client.observation.id + assert len(client.observations) == 1 + with pytest.raises(L.TelemetryReceiptError): + L.validate_job_receipts(trial.parent, METADATA) + + +def test_unknown_pending_receipt_never_adopts_an_uncaptured_observation_id(tmp_path): + """A crash before persisting the SDK ID leaves non-authorizing state, even after read-back.""" + trial = copy_trial(tmp_path) + payload = L.build_attempt_payload(trial, METADATA) + digest = payload_sha256(payload) + evidence = L._telemetry_evidence(payload) + (trial / L.RECEIPT_NAME).write_text(json.dumps(L._pending_receipt(digest))) + + class UnknownPendingLangfuse(FakeLangfuse): + def read_trace(self, trace_id: str): + return {"id": trace_id, "observations": [{ + "id": "2222222222222222", + "name": "harbor-attempt", + "type": "AGENT", + "metadata": { + "payload_sha256": digest, + "exporter_revision": L.EXPORTER_REVISION, + "telemetry_evidence": evidence, + }, + }]} + + def start_observation(self, **kwargs): + pytest.fail("an unknown pending attempt must never emit") + + with pytest.raises(L.TelemetryReceiptError, match="pending"): + L.export_job_attempts(trial.parent, METADATA, client=UnknownPendingLangfuse()) + + assert json.loads((trial / L.RECEIPT_NAME).read_text()) == L._pending_receipt(digest) + + +def test_concurrent_receiptless_exporters_create_at_most_one_observation(tmp_path): + """Atomic pending state serializes contenders that both observed the trace missing.""" + trial = copy_trial(tmp_path) + + class ConcurrentAbsentLangfuse(FakeLangfuse): + def __init__(self): + super().__init__(readback="absent") + self.preflight = threading.Barrier(2) + self.read_count = 0 + self.read_lock = threading.Lock() + + def read_trace(self, trace_id: str): + with self.read_lock: + self.read_count += 1 + count = self.read_count + if count <= 2: + self.preflight.wait(timeout=5) + self.reads.append(trace_id) + return None + + client = ConcurrentAbsentLangfuse() + + def export(): + with pytest.raises(L.TelemetryReceiptError): + L.export_job_attempts(trial.parent, METADATA, client=client) + + with ThreadPoolExecutor(max_workers=2) as pool: + list(pool.map(lambda _index: export(), range(2))) + + pending = json.loads((trial / L.RECEIPT_NAME).read_text()) + assert pending["status"] == "pending" + assert pending["observation_id"] == client.observation.id + assert len(client.observations) == 1 + + +def test_duplicate_preexisting_observations_fail_closed_without_another_emission(tmp_path): + """A trace with duplicate matching observations is not one verified attempt.""" + trial = copy_trial(tmp_path) + payload = L.build_attempt_payload(trial, METADATA) + digest = payload_sha256(payload) + + class DuplicateLangfuse(FakeLangfuse): + def read_trace(self, trace_id: str): + metadata = {"payload_sha256": digest, "exporter_revision": L.EXPORTER_REVISION} + observation = {"id": "1111111111111111", + "name": "harbor-attempt", "type": "AGENT", "metadata": metadata} + return {"id": trace_id, "observations": [observation, observation]} + + def start_observation(self, **kwargs): + pytest.fail("duplicate preexisting trace must not emit") + + with pytest.raises(L.TelemetryReceiptError, match="existing"): + L.export_job_attempts(trial.parent, METADATA, client=DuplicateLangfuse()) + + +def test_evidence_root_exports_preserved_attempts_without_rerunning_agents(tmp_path): + """Changing recursive evidence discovery must not orphan preserved proprietary attempts.""" + trial = copy_trial(tmp_path) + combo = trial.parents[1] + (combo / "combo.json").write_text(json.dumps(PROVENANCE_METADATA)) + (trial / "result.json").unlink() + client = FakeLangfuse() + + receipts = L.export_evidence_root(tmp_path, client=client) + + assert len(receipts) == 1 + assert (trial / "langfuse-receipt.json").is_file() + assert len(client.observations) == 1 + assert client.observations[0]["input"] == { + "name": "fixture-h0", "checksum": "", "source": "", + "text": "Use only deterministic fixture input.", + } + assert client.observations[0]["output"]["skill"] == "demo" + assert client.observations[0]["output"]["timestamps"] == { + "started_at": "2000-01-01T00:00:00Z", + "finished_at": "2000-01-01T00:00:01Z", + } + + +def test_evidence_root_refuses_missing_explicit_provenance_before_emission(tmp_path): + """Preserved evidence cannot infer skill or task text from directory names.""" + trial = copy_trial(tmp_path) + (trial.parents[1] / "combo.json").write_text(json.dumps(METADATA)) + client = FakeLangfuse() + + with pytest.raises(L.TelemetryReceiptError, match="provenance"): + L.export_evidence_root(tmp_path, client=client) + + assert client.observations == [] + assert not (trial / L.RECEIPT_NAME).exists() + + +def test_export_discovers_a_verifier_only_failed_attempt(tmp_path): + """A failed trial that retained only verifier evidence is still one persisted attempt.""" + trial = tmp_path / "job" / "fixture-h1__attempt-2" + solution = trial / "verifier" / "solution" + solution.mkdir(parents=True) + (solution / "answer.txt").write_text("partial fixture answer") + client = FakeLangfuse() + + receipts = L.export_job_attempts(trial.parent, METADATA, client=client) + + assert len(receipts) == 1 and len(client.observations) == 1 + assert client.observations[0]["input"]["name"] == "fixture-h1" + assert (trial / "langfuse-receipt.json").is_file() + + +def test_cli_exports_each_repeated_evidence_root(tmp_path, monkeypatch): + """Dropping repeated roots would leave part of the preserved evidence unreceipted.""" + first, second = tmp_path / "first", tmp_path / "second" + first_metadata, second_metadata = tmp_path / "first.json", tmp_path / "second.json" + first_metadata.write_text(json.dumps(PROVENANCE_METADATA)) + second_metadata.write_text(json.dumps(PROVENANCE_METADATA)) + calls = [] + monkeypatch.setattr( + L, "export_evidence_root", + lambda root, metadata=None: calls.append((root, metadata)) or [], + ) + + assert L.main([ + "--root", str(first), "--metadata", str(first_metadata), + "--root", str(second), "--metadata", str(second_metadata), + ]) == 0 + assert calls == [(first, PROVENANCE_METADATA), (second, PROVENANCE_METADATA)] + + +def test_cli_refuses_a_preserved_root_without_paired_migration_metadata(tmp_path): + """A root and its migration document are an explicit one-to-one input.""" + with pytest.raises(SystemExit): + L.main(["--root", str(tmp_path / "preserved")]) + + +def test_export_job_attempts_filters_native_arm_identity(tmp_path): + from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env + + skill_trial = copy_trial(tmp_path) + control_trial = tmp_path / "job" / "fixture-h0__attempt-2" + shutil.copytree(FIXTURE, control_trial) + common = dict(combination_id="codex@fixture-model--dell-fixture-ffffffffffff", + endpoint_fingerprint="ffffffffffff", harness="codex", + protocol="responses", gateway_revision="direct") + skill = NativeTrialIdentity(**common, arm="skill") + control = NativeTrialIdentity(**common, arm="control") + (skill_trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(skill)}})) + (control_trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(control)}})) + client = FakeLangfuse() + + receipts = L.export_job_attempts(skill_trial.parent, METADATA, identity=skill, client=client) + + assert len(receipts) == 1 + assert (skill_trial / L.RECEIPT_NAME).is_file() + assert not (control_trial / L.RECEIPT_NAME).exists() + + +def test_identity_filter_skips_unrelated_rejected_artifacts_before_opening_them(tmp_path): + from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env + + selected = copy_trial(tmp_path) + other = tmp_path / "job" / "fixture-h0__attempt-2" + shutil.copytree(FIXTURE, other) + common = dict(combination_id="codex@fixture-model--dell-fixture-ffffffffffff", + endpoint_fingerprint="ffffffffffff", harness="codex", + protocol="responses", gateway_revision="direct") + skill = NativeTrialIdentity(**common, arm="skill") + control = NativeTrialIdentity(**common, arm="control") + (selected / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(skill)}})) + (other / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(control)}})) + shutil.rmtree(other / "verifier") + (other / "verifier").symlink_to(tmp_path, target_is_directory=True) + + attempts = L._job_attempts(selected.parent, skill) + + assert attempts == [selected] diff --git a/tests/test_harbor_native.py b/tests/test_harbor_native.py new file mode 100644 index 0000000..534cb92 --- /dev/null +++ b/tests/test_harbor_native.py @@ -0,0 +1,199 @@ +import json +from pathlib import Path + +import pytest + +from ingot.optimize.harbor_native import (NATIVE_RUNNER_REVISION, NativeCell, NativeTrialIdentity, + compile_agent_config, compile_measurement_job, identity_env, + read_trial_identity) +from ingot.optimize.harbor_targets import LocalTarget + + +def _target() -> LocalTarget: + return LocalTarget( + alias="dell-qwen", + display_name="Qwen/Qwen3.6-27B", + base_url="http://127.0.0.1:8011", + served_model="dot-backbone", + context_length=163_840, + protocols=frozenset({"chat", "messages", "responses"}), + family="Qwen3.6", + parameter_billions=27.0, + quantization="fp8-published", + tool_parser="qwen3_xml", + ) + + +def test_native_runner_revision_encodes_exact_trial_memory_limit(): + assert NATIVE_RUNNER_REVISION == "native-v4-memory2048mb" + + +def _identity(arm: str = "skill") -> NativeTrialIdentity: + return NativeTrialIdentity( + combination_id="aider@dot-backbone--dell-qwen-deadbeefcafe", + endpoint_fingerprint="deadbeefcafe", + harness="aider", + protocol="chat", + gateway_revision="direct", + arm=arm, + ) + + +def _write_lock(path: Path, env: dict[str, str]) -> None: + path.mkdir() + (path / "lock.json").write_text(json.dumps({"agent": {"env": env}})) + + +def test_trial_identity_round_trips_through_harbor_agent_lock(tmp_path): + skill = _identity("skill") + control = _identity("control") + skill_dir = tmp_path / "skill" + control_dir = tmp_path / "control" + _write_lock(skill_dir, {"OPENAI_API_KEY": "${OPENAI_API_KEY}", **identity_env(skill)}) + _write_lock(control_dir, {"OPENAI_API_KEY": "${OPENAI_API_KEY}", **identity_env(control)}) + + assert read_trial_identity(skill_dir) == skill + assert read_trial_identity(control_dir) == control + assert read_trial_identity(skill_dir) != read_trial_identity(control_dir) + + +@pytest.mark.parametrize( + "mutate", + [ + lambda env: env.pop("INGOT_ARM"), + lambda env: env.__setitem__("INGOT_ARM", "treatment"), + lambda env: env.__setitem__("INGOT_HARNESS", "unknown"), + lambda env: env.__setitem__("INGOT_PROTOCOL", "responses"), + lambda env: env.__setitem__("INGOT_ENDPOINT_FINGERPRINT", "not-a-fingerprint"), + lambda env: env.__setitem__("INGOT_EXTRA", "surprise"), + lambda env: env.__setitem__("INGOT_COMBINATION_ID", "../../mutable/path"), + lambda env: env.__setitem__("INGOT_ARM", 7), + ], +) +def test_trial_identity_rejects_malformed_or_ambiguous_lock(tmp_path, mutate): + env = identity_env(_identity()) + mutate(env) + trial = tmp_path / "trial" + _write_lock(trial, env) + + with pytest.raises(ValueError, match="Harbor trial identity"): + read_trial_identity(trial) + + +def test_trial_identity_rejects_invalid_lock_shape(tmp_path): + trial = tmp_path / "trial" + trial.mkdir() + (trial / "lock.json").write_text("[]") + + with pytest.raises(ValueError, match="Harbor trial identity"): + read_trial_identity(trial) + + +def test_agent_compiler_preserves_arm_and_shared_endpoint_limit(tmp_path): + target = _target() + source = tmp_path / "skill-source" + source.mkdir() + skill = compile_agent_config(target, "aider", "skill", source, 8) + control = compile_agent_config(target, "aider", "control", source, 8) + + assert skill["name"] == control["name"] == "aider" + assert skill["model_name"] == control["model_name"] == "openai/dot-backbone" + assert skill["n_concurrent"] == control["n_concurrent"] == 8 + assert skill["concurrency_group"] == control["concurrency_group"] == f"endpoint:{target.fingerprint}" + assert skill["skills"] == [str(source)] + assert control["skills"] == [] + assert skill["env"]["INGOT_ARM"] == "skill" + assert control["env"]["INGOT_ARM"] == "control" + assert skill["env"]["INGOT_COMBINATION_ID"] == ( + f"aider@{target.served_model}--{target.job_slug}") + assert skill["env"]["INGOT_GATEWAY_REVISION"] == f"direct-{NATIVE_RUNNER_REVISION}" + assert skill["env"]["OPENAI_API_KEY"] == control["env"]["OPENAI_API_KEY"] == "local" + + +def test_measurement_compiler_can_resume_only_the_missing_arm(tmp_path): + target = _target() + cell = NativeCell(target, "aider") + + config = compile_measurement_job( + tmp_path / "dataset", [f"task-{index}" for index in range(4)], [cell], + tmp_path / "skill", tmp_path / "jobs", attempts=3, global_limit=4, + endpoint_limits={target.fingerprint: 4}, arms=("control",)) + + assert [agent["env"]["INGOT_ARM"] for agent in config["agents"]] == ["control"] + + +def test_agent_compiler_uses_import_path_for_gateway_codex(tmp_path): + target = _target() + from ingot.optimize.harbor_gateway import gateway_route + + route = gateway_route(target, "codex") + assert route is not None + config = compile_agent_config(target, "codex", "skill", tmp_path, 4) + + assert config["import_path"] == "ingot.optimize.harbor_codex_gateway:GatewayCodex" + assert "name" not in config + assert config["concurrency_group"] == f"endpoint:{target.fingerprint}" + assert config["env"]["INGOT_GATEWAY_REVISION"] == f"{route.revision}-{NATIVE_RUNNER_REVISION}" + + +def test_iter_attempt_dirs_filters_unified_job_by_exact_identity(tmp_path): + from ingot.optimize.harbor_native import iter_attempt_dirs + + wanted = _identity("skill") + other = _identity("control") + first = tmp_path / "first" + second = tmp_path / "second" + _write_lock(first, identity_env(wanted)) + _write_lock(second, identity_env(other)) + (first / "result.json").write_text('{"task_name":"ingot/h1"}') + (second / "result.json").write_text('{"task_name":"ingot/h1"}') + + assert list(iter_attempt_dirs(tmp_path, identity=wanted)) == [first] + assert list(iter_attempt_dirs(tmp_path, identity=other)) == [second] + + +def test_iter_attempt_dirs_does_not_inspect_unselected_malformed_sibling(tmp_path): + from ingot.optimize.harbor_native import iter_attempt_dirs + + wanted = _identity("skill") + first = tmp_path / "first" + malformed = tmp_path / "malformed" + _write_lock(first, identity_env(wanted)) + malformed.mkdir() + (malformed / "lock.json").write_text("not-json") + (first / "result.json").write_text('{"task_name":"ingot/h1"}') + (malformed / "result.json").write_text('{"task_name":"ingot/h1"}') + + assert list(iter_attempt_dirs(tmp_path, identity=wanted)) == [first] + + +def test_iter_attempt_dirs_rejects_symlinked_attempt(tmp_path): + from ingot.optimize.harbor_native import iter_attempt_dirs + + outside = tmp_path / "outside" + outside.mkdir() + (outside / "result.json").write_text('{"task_name":"ingot/h1"}') + job = tmp_path / "job" + job.mkdir() + (job / "linked").symlink_to(outside, target_is_directory=True) + + with pytest.raises(ValueError, match="real attempt directory"): + list(iter_attempt_dirs(job)) + + +@pytest.mark.parametrize("name", ["lock.json", "result.json"]) +def test_native_attempt_rejects_symlinked_identity_or_result_file(tmp_path, name): + from ingot.optimize.harbor_native import iter_attempt_dirs + + job = tmp_path / "job" + trial = job / "h1__one" + trial.mkdir(parents=True) + external = tmp_path / f"external-{name}" + external.write_text('{}') + (trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(_identity())}})) + (trial / "result.json").write_text('{"task_name":"ingot/h1"}') + (trial / name).unlink() + (trial / name).symlink_to(external) + + with pytest.raises(ValueError, match="regular file"): + list(iter_attempt_dirs(job, identity=_identity())) diff --git a/tests/test_harbor_native_jobs.py b/tests/test_harbor_native_jobs.py new file mode 100644 index 0000000..94b70de --- /dev/null +++ b/tests/test_harbor_native_jobs.py @@ -0,0 +1,141 @@ +import pytest + +from ingot.optimize.harbor_native import (NativeCell, compile_canary_job, + compile_measurement_job, + select_measurement_cells, write_job_config) +from ingot.optimize.harbor_targets import LocalTarget + + +def _target(alias: str, model: str, port: int) -> LocalTarget: + return LocalTarget( + alias=alias, + display_name=model, + base_url=f"http://127.0.0.1:{port}", + served_model=model, + context_length=131_072, + protocols=frozenset({"chat", "messages", "responses"}), + ) + + +@pytest.fixture +def cells(): + first = _target("dell-qwen", "dot-backbone", 8011) + second = _target("spark-deepseek", "deepseek-v4-flash", 8000) + return [NativeCell(first, "aider"), NativeCell(second, "goose")] + + +def test_canary_job_compiles_one_skill_agent_per_cell(tmp_path, cells): + config = compile_canary_job( + tmp_path / "dataset", "build-loop-h1", cells, tmp_path / "skill", tmp_path / "jobs", + global_limit=16, endpoint_limits={cell.target.fingerprint: 4 for cell in cells}, + ) + + assert config["jobs_dir"] == str(tmp_path / "jobs") + assert config["n_attempts"] == 1 + assert config["n_concurrent_trials"] == 16 + assert config["datasets"] == [{"path": str(tmp_path / "dataset"), + "task_names": ["build-loop-h1"]}] + assert len(config["agents"]) == 2 + assert [agent["env"]["INGOT_ARM"] for agent in config["agents"]] == ["canary", "canary"] + assert all(agent["skills"] == [str(tmp_path / "skill")] for agent in config["agents"]) + + +def test_measurement_job_interleaves_matched_arms_and_shares_endpoint_caps(tmp_path, cells): + limits = {cells[0].target.fingerprint: 8, cells[1].target.fingerprint: 12} + config = compile_measurement_job( + tmp_path / "dataset", ["h1", "h2", "h3", "h4"], cells, + tmp_path / "skill", tmp_path / "jobs", attempts=3, global_limit=32, + endpoint_limits=limits, + ) + + assert config["n_attempts"] == 3 + assert config["n_concurrent_trials"] == 32 + assert config["datasets"] == [{"path": str(tmp_path / "dataset"), + "task_names": ["h1", "h2", "h3", "h4"]}] + assert [agent["env"]["INGOT_ARM"] for agent in config["agents"]] == [ + "skill", "control", "skill", "control"] + assert [agent["env"]["INGOT_COMBINATION_ID"] for agent in config["agents"]][::2] == [ + cells[0].combination_id, cells[1].combination_id] + for index, cell in enumerate(cells): + pair = config["agents"][index * 2:index * 2 + 2] + assert {agent["n_concurrent"] for agent in pair} == {limits[cell.target.fingerprint]} + assert {agent["concurrency_group"] for agent in pair} == { + f"endpoint:{cell.target.fingerprint}"} + + +def test_measurement_job_round_robins_target_major_input(tmp_path): + first = _target("dell-qwen", "dot-backbone", 8011) + second = _target("spark-deepseek", "deepseek-v4-flash", 8000) + cells = [NativeCell(first, "aider"), NativeCell(first, "goose"), + NativeCell(second, "aider"), NativeCell(second, "goose")] + config = compile_measurement_job( + tmp_path / "dataset", ["h1", "h2", "h3", "h4"], cells, + tmp_path / "skill", tmp_path / "jobs", attempts=3, global_limit=32, + endpoint_limits={first.fingerprint: 8, second.fingerprint: 8}, + ) + + skill_agents = config["agents"][::2] + assert [agent["env"]["INGOT_ENDPOINT_FINGERPRINT"] for agent in skill_agents] == [ + first.fingerprint, second.fingerprint, first.fingerprint, second.fingerprint] + + +@pytest.mark.parametrize("attempts", [0, 1, 2, 4, True]) +def test_measurement_job_requires_full_three_attempt_contract(tmp_path, cells, attempts): + with pytest.raises(ValueError, match="three attempts"): + compile_measurement_job( + tmp_path / "dataset", ["h1", "h2", "h3", "h4"], cells, + tmp_path / "skill", tmp_path / "jobs", attempts=attempts, global_limit=16, + endpoint_limits={cell.target.fingerprint: 4 for cell in cells}, + ) + + +def test_job_compiler_refuses_missing_or_unknown_endpoint_limit(tmp_path, cells): + with pytest.raises(ValueError, match="endpoint concurrency"): + compile_canary_job( + tmp_path / "dataset", "h1", cells, tmp_path / "skill", tmp_path / "jobs", + global_limit=16, endpoint_limits={cells[0].target.fingerprint: 4}, + ) + + +def test_failed_canary_becomes_explicit_unmeasured_cell_without_lift(cells): + selected, unmeasured = select_measurement_cells(cells, { + cells[0].combination_id: {"ok": True}, + cells[1].combination_id: {"error": "AgentTimeoutError: expired"}, + }) + + assert selected == [cells[0]] + assert unmeasured == {cells[1].combination_id: { + "combination": cells[1].combination_id, + "harness": cells[1].harness, + "target_alias": cells[1].target.alias, + "endpoint_fingerprint": cells[1].target.fingerprint, + "state": "unmeasured", + "error": "AgentTimeoutError: expired", + }} + assert "lift" not in unmeasured[cells[1].combination_id] + + +def test_job_config_write_is_atomic_and_deterministic(tmp_path, cells): + config = compile_canary_job( + tmp_path / "dataset", "h1", cells, tmp_path / "skill", tmp_path / "jobs", + global_limit=16, endpoint_limits={cell.target.fingerprint: 4 for cell in cells}, + ) + output = tmp_path / "config.json" + + write_job_config(output, config) + first = output.read_bytes() + write_job_config(output, config) + + assert output.read_bytes() == first + assert not list(tmp_path.glob(".config.json.*.tmp")) + + +@pytest.mark.parametrize("name", ["http:config.json", "model@host:8011.json"]) +def test_job_config_refuses_url_or_port_derived_filename(tmp_path, cells, name): + config = compile_canary_job( + tmp_path / "dataset", "h1", cells, tmp_path / "skill", tmp_path / "jobs", + global_limit=16, endpoint_limits={cell.target.fingerprint: 4 for cell in cells}, + ) + + with pytest.raises(ValueError, match="safe slug"): + write_job_config(tmp_path / name, config) diff --git a/tests/test_harbor_native_runner.py b/tests/test_harbor_native_runner.py new file mode 100644 index 0000000..47e0b0b --- /dev/null +++ b/tests/test_harbor_native_runner.py @@ -0,0 +1,658 @@ +import json +from pathlib import Path + +import pytest + +from ingot.optimize import harbor_eval as H +from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env +from ingot.optimize.harbor_targets import LocalTarget + + +def _identity(arm="skill"): + return NativeTrialIdentity( + combination_id="aider@dot-backbone--dell-qwen-deadbeefcafe", + endpoint_fingerprint="deadbeefcafe", harness="aider", protocol="chat", + gateway_revision="direct", arm=arm) + + +def _attempt(job: Path, name: str, identity: NativeTrialIdentity): + trial = job / name + trial.mkdir(parents=True) + (trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(identity)}})) + (trial / "result.json").write_text(json.dumps({"task_name": "ingot/h1", + "finished_at": "2026-08-13T00:00:00Z"})) + + +def test_watch_native_job_releases_each_identity_once_at_exact_cardinality(tmp_path): + job = tmp_path / "job" + job.mkdir() + skill, control = _identity("skill"), _identity("control") + seen = [] + + _attempt(job, "h1__s1", skill) + expected = {skill: {"h1": 2}, control: {"h1": 1}} + assert H.watch_native_job(job, expected, seen.append) == set() + _attempt(job, "h1__c1", control) + assert H.watch_native_job(job, expected, seen.append) == {control} + _attempt(job, "h1__s2", skill) + assert H.watch_native_job(job, expected, seen.append, + released={control}) == {skill, control} + assert seen == [control, skill] + + +def test_watch_native_job_refuses_more_attempts_than_contract(tmp_path): + job = tmp_path / "job" + job.mkdir() + identity = _identity() + _attempt(job, "h1__one", identity) + _attempt(job, "h1__two", identity) + + with pytest.raises(RuntimeError, match="exceeded expected attempts"): + H.watch_native_job(job, {identity: {"h1": 1}}, lambda _identity: None) + + +def test_run_native_job_uses_one_harbor_process_and_progress_callback(tmp_path, monkeypatch): + config = tmp_path / "config.json" + config.write_text("{}") + jobs = tmp_path / "jobs" + identity = _identity() + calls = [] + + class Process: + returncode = 0 + count = 0 + pid = 4242 + + def poll(self): + self.count += 1 + if self.count == 1: + _attempt(jobs / "full", "h1__one", identity) + return None + return 0 + + def wait(self): + return 0 + + monkeypatch.setattr(H.subprocess, "Popen", lambda argv, **kwargs: + calls.append((argv, kwargs)) or Process()) + monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}") + monkeypatch.setattr(H.time, "sleep", lambda _seconds: None) + + overlay = tmp_path / "network.compose.yml" + overlay.write_text("networks: {}") + result = H.run_native_job( + config, jobs, "full", {identity: {"h1": 1}}, + on_ready=lambda item: calls.append(item), + process_env={"PATH": "/bin", "HARBOR_EXTRA_DOCKER_COMPOSE": str(overlay)}, + ) + + assert result == jobs / "full" + assert len([item for item in calls if isinstance(item, tuple)]) == 1 + argv, kwargs = calls[0] + assert argv[:3] == [H.HARBOR_BIN, "run", "--config"] + assert argv[-6:] == ["--override-memory-mb", "2048", "--job-name", "full", + "--extra-docker-compose", str(overlay)] + assert kwargs["env"]["PATH"] == "/bin" + assert "HARBOR_EXTRA_DOCKER_COMPOSE" not in kwargs["env"] + assert kwargs["env"]["OPENAI_API_KEY"] == "local" + assert kwargs["env"]["ANTHROPIC_API_KEY"] == "local" + assert kwargs["env"]["CODEX_API_KEY"] == "local" + assert calls[-1] == identity + + +def test_run_native_job_does_not_respawn_harbor_for_released_terminal_evidence( + tmp_path, monkeypatch): + config = tmp_path / "config.json" + config.write_text("{}") + jobs = tmp_path / "jobs" + job = jobs / "full" + identity = _identity() + _attempt(job, "h1__one", identity) + jobs.mkdir(exist_ok=True) + (jobs / "full.released.json").write_text(json.dumps({ + "released": [identity_env(identity)], + })) + monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}") + monkeypatch.setattr(H.subprocess, "Popen", + lambda *_args, **_kwargs: pytest.fail("Harbor was respawned")) + + result = H.run_native_job( + config, jobs, "full", {identity: {"h1": 1}}, + on_ready=lambda _item: pytest.fail("released callback repeated"), + process_env={"PATH": "/bin"}, allow_completed_reuse=True, + ) + + assert result == job + assert not (jobs / "full.owner.json").exists() + + +def test_run_native_job_refinalizes_terminal_evidence_without_respawning_harbor( + tmp_path, monkeypatch): + config = tmp_path / "config.json" + config.write_text("{}") + jobs = tmp_path / "jobs" + job = jobs / "full" + identity = _identity() + _attempt(job, "h1__one", identity) + seen = [] + monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}") + monkeypatch.setattr(H.subprocess, "Popen", + lambda *_args, **_kwargs: pytest.fail("Harbor was respawned")) + + result = H.run_native_job( + config, jobs, "full", {identity: {"h1": 1}}, on_ready=seen.append, + process_env={"PATH": "/bin"}, allow_completed_reuse=True, + ) + + assert result == job + assert seen == [identity] + assert json.loads((jobs / "full.released.json").read_text()) == { + "released": [identity_env(identity)], + } + assert not (jobs / "full.owner.json").exists() + + +def test_run_native_job_keeps_terminal_finalization_pending_without_respawning_harbor( + tmp_path, monkeypatch): + config = tmp_path / "config.json" + config.write_text("{}") + jobs = tmp_path / "jobs" + job = jobs / "full" + identity = _identity() + _attempt(job, "h1__one", identity) + monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}") + monkeypatch.setattr(H.subprocess, "Popen", + lambda *_args, **_kwargs: pytest.fail("Harbor was respawned")) + + with pytest.raises(RuntimeError, match="native Harbor finalization is pending"): + H.run_native_job( + config, jobs, "full", {identity: {"h1": 1}}, on_ready=lambda _item: False, + process_env={"PATH": "/bin"}, allow_completed_reuse=True, + ) + + assert json.loads((jobs / "full.released.json").read_text()) == {"released": []} + assert not (jobs / "full.owner.json").exists() + + +def test_run_native_job_does_not_reuse_released_evidence_without_catalog_proof( + tmp_path, monkeypatch): + config = tmp_path / "config.json" + config.write_text("{}") + jobs = tmp_path / "jobs" + job = jobs / "full" + identity = _identity() + _attempt(job, "h1__one", identity) + jobs.mkdir(exist_ok=True) + (jobs / "full.released.json").write_text(json.dumps({ + "released": [identity_env(identity)], + })) + calls = [] + + class Process: + def poll(self): + return 0 + + def wait(self): + return 0 + + monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}") + monkeypatch.setattr(H.subprocess, "Popen", + lambda *_args, **_kwargs: calls.append(True) or Process()) + + H.run_native_job(config, jobs, "full", {identity: {"h1": 1}}, + on_ready=lambda _item: None, process_env={"PATH": "/bin"}) + + assert calls == [True] + + +def test_watch_native_job_does_not_release_started_trial(tmp_path): + job = tmp_path / "job" + job.mkdir() + identity = _identity() + _attempt(job, "h1__one", identity) + result = job / "h1__one" / "result.json" + result.write_text(json.dumps({"task_name": "ingot/h1", "finished_at": None})) + + assert H.watch_native_job(job, {identity: {"h1": 1}}, lambda _item: None) == set() + + +def _target(alias="dell-qwen", model="dot-backbone", port=8011): + return LocalTarget(alias=alias, display_name=model, base_url=f"http://host:{port}", + served_model=model, context_length=163840, + protocols=frozenset({"chat", "responses", "messages"})) + + +def test_native_sweep_runs_one_canary_queue_and_independent_caps(tmp_path, monkeypatch): + first = _target() + second = _target("spark-deepseek", "deepseek-v4-flash", 8899) + monkeypatch.setattr(H, "HARBOR_DIR", tmp_path / "harbor") + monkeypatch.setattr(H, "load_tasks", lambda _skill: ([], [{"task": "x"}] * 4, {})) + monkeypatch.setattr(H, "stage_skill", lambda _skill: tmp_path / "source") + monkeypatch.setattr(H, "build_dataset", lambda *_args: tmp_path / "dataset") + monkeypatch.setattr(H.shutil, "which", lambda _binary: "/bin/harbor") + monkeypatch.setattr(H, "discover_target", lambda alias, _url: first if alias == first.alias else second) + monkeypatch.setattr(H, "probe_protocol", lambda *_args: None) + monkeypatch.setattr(H, "probe_chat_tool_round_trip", lambda *_args: None) + monkeypatch.setattr(H, "run_canary", lambda *_args, **_kwargs: + pytest.fail("serial canary ran")) + calls = [] + + def canaries(*args, **kwargs): + calls.append(("canary", args, kwargs)) + return {f"aider@{target.served_model}--{target.job_slug}": {"ok": True} + for target in (first, second)} + + monkeypatch.setattr(H, "_run_native_canaries", canaries, raising=False) + monkeypatch.setattr(H, "_run_native_full_arms", + lambda *args, **kwargs: calls.append(("full", args, kwargs))) + + H.run_local_sweep("demo", [first, second], harnesses=("aider",), native_parallel=True, + global_concurrency=32, endpoint_concurrency=6, log=lambda *_args: None) + + assert [item[0] for item in calls] == ["canary", "full"] + assert calls[0][2]["global_limit"] == 32 + assert calls[0][2]["endpoint_limit"] == 6 + assert calls[1][2]["global_limit"] == 32 + assert calls[1][2]["endpoint_limit"] == 6 + + +def test_native_canary_restart_rehydrates_terminal_identity_without_agent_rerun(tmp_path, + monkeypatch): + target = _target() + identity = H.native_trial_identity(target, "aider", "canary") + job = tmp_path / "canaries" / "demo" / "native-canaries" + trial = job / "demo-h0__one" + solution = trial / "verifier" / "solution" + solution.mkdir(parents=True) + (solution / "answer.md").write_text("done") + (trial / "lock.json").write_text(json.dumps({"agent": {"env": identity_env(identity)}})) + (trial / "result.json").write_text(json.dumps({"task_name": "ingot/demo-h0", + "finished_at": "now"})) + monkeypatch.setattr(H, "compile_canary_job", lambda *_args, **_kwargs: {}) + monkeypatch.setattr(H, "write_job_config", lambda *_args: None) + monkeypatch.setattr(H, "_telemetry_provenance", lambda *_args: {}) + exported = [] + monkeypatch.setattr(H, "export_job_attempts", lambda *args, **kwargs: exported.append(kwargs["identity"])) + monkeypatch.setattr(H, "run_native_job", lambda *_args, **_kwargs: job) + + records = H._run_native_canaries( + "demo", tmp_path / "dataset", [target], ("aider",), [{"task": "x"}] * 4, + str(tmp_path / "source"), tmp_path / "canaries", global_limit=16, + endpoint_limit=4) + + assert records[identity.combination_id]["ok"] is True + assert exported == [identity] + + +def test_native_canary_telemetry_failure_is_unmeasured(tmp_path, monkeypatch): + target = _target() + identity = H.native_trial_identity(target, "aider", "canary") + monkeypatch.setattr(H, "compile_canary_job", lambda *_args, **_kwargs: {}) + monkeypatch.setattr(H, "write_job_config", lambda *_args: None) + monkeypatch.setattr(H, "_telemetry_provenance", lambda *_args: {}) + monkeypatch.setattr(H, "export_job_attempts", + lambda *_args, **_kwargs: (_ for _ in ()).throw(RuntimeError("telemetry"))) + + callback_results = [] + + def native(_config, jobs_root, job_name, expected, *, on_ready, **_kwargs): + for item in expected: + callback_results.append(on_ready(item)) + return jobs_root / job_name + + monkeypatch.setattr(H, "run_native_job", native) + records = H._run_native_canaries( + "demo", tmp_path / "dataset", [target], ("aider",), [{"task": "x"}] * 4, + str(tmp_path / "source"), tmp_path / "canaries", global_limit=16, + endpoint_limit=4) + record = records[identity.combination_id] + assert callback_results == [False] + assert "ok" not in record + assert record["error"] == "canary telemetry receipt was not verified" + + +def test_native_completed_restart_publishes_failed_canary_without_callbacks(tmp_path, monkeypatch): + from ingot.optimize import harbor_rescore as R + + passed = _target() + failed = _target("spark-deepseek", "deepseek-v4-flash", 8899) + holdout = [{"task": f"task {index}"} for index in range(4)] + source = tmp_path / "source" + skill_dir = source / "demo" + skill_dir.mkdir(parents=True) + (skill_dir / "SKILL.md").write_text("fixture skill\n") + scoring = { + "judge": "agy/fixture", + "scoring_revision": "fixture-v1", + "judge_billing_mode": "subscription", + "judge_runtime": "fixture", + } + monkeypatch.setattr(R, "current_scoring_identity", lambda: scoring) + monkeypatch.setattr(H, "compile_measurement_job", lambda *_args, **_kwargs: {}) + monkeypatch.setattr(H, "_process_start_token", lambda pid: f"start-{pid}") + monkeypatch.setattr(H, "export_job_attempts", lambda *_args, **_kwargs: + pytest.fail("completed restart repeated telemetry export")) + monkeypatch.setattr(H.subprocess, "Popen", lambda *_args, **_kwargs: + pytest.fail("completed restart respawned Harbor")) + rescored = [] + + def fake_rescore(_skill, **kwargs): + rescored.append(list(kwargs["combination_paths"])) + + monkeypatch.setattr(R, "rescore", fake_rescore) + manifest = {"combinations": {}} + jobs_root = tmp_path / "jobs" + cell = H.NativeCell(passed, "aider") + job_name = H._native_full_job_name(cell) + job = jobs_root / job_name + output_root = tmp_path / "published" + output_root.mkdir() + passed_id = f"aider@{passed.served_model}--{passed.job_slug}" + failed_id = f"aider@{failed.served_model}--{failed.job_slug}" + skill_identity = H.native_trial_identity(passed, "aider", "skill") + control_identity = H.native_trial_identity(passed, "aider", "control") + provenance = H._telemetry_provenance("demo", holdout, str(source)) + (output_root / "demo.rescored.json").write_text(json.dumps({"combinations": { + passed_id: { + "combination": passed_id, + "endpoint_fingerprint": skill_identity.endpoint_fingerprint, + "skill_sha256": provenance["skill_sha256"], + "task_fingerprint": H._task_fingerprint(holdout), + "attempts": 3, + "harness": "aider", + "protocol": skill_identity.protocol, + "gateway_revision": skill_identity.gateway_revision, + **scoring, + "lift": 0.1, + }, + }})) + for identity in (skill_identity, control_identity): + for task_index in range(4): + for attempt in range(3): + trial = job / f"{identity.arm}-h{task_index}-a{attempt}" + trial.mkdir(parents=True) + (trial / "lock.json").write_text(json.dumps({ + "agent": {"env": identity_env(identity)}, + })) + (trial / "result.json").write_text(json.dumps({ + "task_name": f"ingot/demo-h{task_index}", + "finished_at": "2026-08-14T00:00:00Z", + })) + (jobs_root / f"{job_name}.released.json").write_text(json.dumps({ + "released": [identity_env(skill_identity), identity_env(control_identity)], + })) + (jobs_root / "native-full.pipeline.json").write_text(json.dumps({ + "agent_identity": { + "skill_sha256": provenance["skill_sha256"], + "task_fingerprint": H._task_fingerprint(holdout), + "attempts": 3, + "exporter_revision": H.EXPORTER_REVISION, + "cells": [[passed_id, skill_identity.gateway_revision]], + }, + "scoring_identity": scoring, + "exported": {passed_id: ["control", "skill"]}, + "graded": [passed_id], + "published": [passed_id], + })) + + H._run_native_full_arms( + "demo", tmp_path / "dataset", [passed, failed], ("aider",), holdout, str(source), + jobs_root, manifest, { + passed_id: {"ok": True}, + failed_id: {"error": "adapter did not route"}, + }, + global_limit=16, endpoint_limit=4, publish_root=output_root, + allow_completed_reuse=True, + ) + + assert [[path.name for path in paths] for paths in rescored] == [[failed_id]] + metadata = json.loads((jobs_root / failed_id / "combo.json").read_text()) + assert metadata["canary_error"] == "adapter did not route" + assert metadata["skill_sha256"] + + +def test_prior_lift_for_now_failed_cell_waits_for_selected_measurement(tmp_path, monkeypatch): + from ingot.optimize import harbor_rescore as R + + failed = _target() + passed = _target("spark-deepseek", "deepseek-v4-flash", 8899) + holdout = [{"task": f"task {index}"} for index in range(4)] + source = tmp_path / "source" + skill_dir = source / "demo" + skill_dir.mkdir(parents=True) + (skill_dir / "SKILL.md").write_text("fixture skill\n") + scoring = { + "judge": "agy/fixture", + "scoring_revision": "fixture-v1", + "judge_billing_mode": "subscription", + "judge_runtime": "fixture", + } + monkeypatch.setattr(R, "current_scoring_identity", lambda: scoring) + monkeypatch.setattr(H, "compile_measurement_job", lambda *_args, **_kwargs: {}) + monkeypatch.setattr(H, "write_job_config", lambda *_args, **_kwargs: None) + exported = [] + monkeypatch.setattr(H, "export_job_attempts", + lambda *_args, **kwargs: exported.append(kwargs["identity"])) + rescored = [] + + def fake_rescore(_skill, **kwargs): + rescored.append(list(kwargs["combination_paths"])) + + monkeypatch.setattr(R, "rescore", fake_rescore) + + def fake_run_native_job(_config, _jobs_root, _job_name, expected, *, on_ready, **_kwargs): + for identity in expected: + assert identity.combination_id == passed_id + assert on_ready(identity) is True + + monkeypatch.setattr(H, "run_native_job", fake_run_native_job) + manifest = {"combinations": {}} + jobs_root = tmp_path / "jobs" + output_root = tmp_path / "published" + output_root.mkdir() + failed_id = f"aider@{failed.served_model}--{failed.job_slug}" + passed_id = f"aider@{passed.served_model}--{passed.job_slug}" + failed_identity = H.native_trial_identity(failed, "aider", "skill") + provenance = H._telemetry_provenance("demo", holdout, str(source)) + (output_root / "demo.rescored.json").write_text(json.dumps({"combinations": { + failed_id: { + "combination": failed_id, + "endpoint_fingerprint": failed_identity.endpoint_fingerprint, + "skill_sha256": provenance["skill_sha256"], + "task_fingerprint": H._task_fingerprint(holdout), + "attempts": 3, + "harness": "aider", + "protocol": failed_identity.protocol, + "gateway_revision": failed_identity.gateway_revision, + **scoring, + "lift": 0.1, + }, + }})) + + H._run_native_full_arms( + "demo", tmp_path / "dataset", [failed, passed], ("aider",), holdout, str(source), + jobs_root, manifest, { + failed_id: {"error": "adapter did not route"}, + passed_id: {"ok": True}, + }, + global_limit=16, endpoint_limit=4, publish_root=output_root, + ) + + assert [[path.name for path in paths] for paths in rescored] == [[failed_id, passed_id]] + assert {identity.combination_id for identity in exported} == {passed_id} + + +def test_native_full_arms_runs_stable_cells_concurrently_instead_of_one_wide_job( + tmp_path, monkeypatch): + """A partial run must finish cells, not spread work across the whole matrix first.""" + import threading + from ingot.optimize import harbor_rescore as R + + first = _target() + second = _target("spark-deepseek", "deepseek-v4-flash", 8899) + holdout = [{"task": f"task {index}"} for index in range(4)] + source = tmp_path / "source" + skill_dir = source / "demo" + skill_dir.mkdir(parents=True) + (skill_dir / "SKILL.md").write_text("fixture skill\n") + monkeypatch.setattr(R, "current_scoring_identity", lambda: { + "judge": "agy/fixture", "scoring_revision": "fixture-v1", + "judge_billing_mode": "subscription", "judge_runtime": "fixture", + }) + monkeypatch.setattr(R, "rescore", lambda *_args, **_kwargs: None) + monkeypatch.setattr(H, "write_job_config", lambda *_args, **_kwargs: None) + monkeypatch.setattr(H, "export_job_attempts", lambda *_args, **_kwargs: None) + compiled = [] + + def compile_job(_dataset, _tasks, cells, *_args, **_kwargs): + compiled.append([cell.combination_id for cell in cells]) + return {} + + monkeypatch.setattr(H, "compile_measurement_job", compile_job) + barrier = threading.Barrier(2) + calls = [] + + def native(_config, jobs_root, job_name, expected, *, on_ready, **_kwargs): + calls.append((job_name, threading.get_ident(), set(expected))) + barrier.wait(timeout=2) + for identity in expected: + assert on_ready(identity) is True + return jobs_root / job_name + + monkeypatch.setattr(H, "run_native_job", native) + ids = [f"aider@{target.served_model}--{target.job_slug}" for target in (first, second)] + + H._run_native_full_arms( + "demo", tmp_path / "dataset", [first, second], ("aider",), holdout, str(source), + tmp_path / "jobs", {"combinations": {}}, {key: {"ok": True} for key in ids}, + global_limit=12, endpoint_limit=6, publish_root=tmp_path / "published", + ) + + assert compiled == [[ids[0]], [ids[1]]] + assert len(calls) == 2 + assert len({thread_id for _name, thread_id, _expected in calls}) == 2 + assert all(len(expected) == 2 for _name, _thread_id, expected in calls) + assert len({name for name, _thread_id, _expected in calls}) == 2 + + +def test_native_full_arms_never_overlaps_jobs_for_one_endpoint(tmp_path, monkeypatch): + import threading + from ingot.optimize import harbor_rescore as R + + target = _target() + holdout = [{"task": f"task {index}"} for index in range(4)] + source = tmp_path / "source" + skill_dir = source / "demo" + skill_dir.mkdir(parents=True) + (skill_dir / "SKILL.md").write_text("fixture skill\n") + monkeypatch.setattr(R, "current_scoring_identity", lambda: { + "judge": "agy/fixture", "scoring_revision": "fixture-v1", + "judge_billing_mode": "subscription", "judge_runtime": "fixture", + }) + monkeypatch.setattr(R, "rescore", lambda *_args, **_kwargs: None) + monkeypatch.setattr(H, "compile_measurement_job", lambda *_args, **_kwargs: {}) + monkeypatch.setattr(H, "write_job_config", lambda *_args, **_kwargs: None) + monkeypatch.setattr(H, "export_job_attempts", lambda *_args, **_kwargs: None) + guard = threading.Lock() + second_entered = threading.Event() + active = 0 + peak = 0 + + def native(_config, jobs_root, job_name, expected, *, on_ready, **_kwargs): + nonlocal active, peak + with guard: + active += 1 + peak = max(peak, active) + first = active == 1 and peak == 1 + if active == 2: + second_entered.set() + try: + if first: + second_entered.wait(timeout=0.2) + for identity in expected: + assert on_ready(identity) is True + finally: + with guard: + active -= 1 + return jobs_root / job_name + + monkeypatch.setattr(H, "run_native_job", native) + ids = [f"{harness}@{target.served_model}--{target.job_slug}" + for harness in ("aider", "pi")] + + H._run_native_full_arms( + "demo", tmp_path / "dataset", [target], ("aider", "pi"), holdout, str(source), + tmp_path / "jobs", {"combinations": {}}, {key: {"ok": True} for key in ids}, + global_limit=12, endpoint_limit=6, publish_root=tmp_path / "published", + ) + + assert peak == 1 + + +def test_native_full_arms_adopts_legacy_identity_and_runs_only_missing_arm(tmp_path, monkeypatch): + from ingot.optimize import harbor_rescore as R + + target = _target() + holdout = [{"task": f"task {index}"} for index in range(4)] + source = tmp_path / "source" + skill_dir = source / "demo" + skill_dir.mkdir(parents=True) + (skill_dir / "SKILL.md").write_text("fixture skill\n") + monkeypatch.setattr(R, "current_scoring_identity", lambda: { + "judge": "agy/fixture", "scoring_revision": "fixture-v1", + "judge_billing_mode": "subscription", "judge_runtime": "fixture", + }) + rescored = [] + monkeypatch.setattr(R, "rescore", + lambda *_args, **kwargs: rescored.append(kwargs["combination_paths"])) + compiled_arms = [] + + def compile_job(*_args, **kwargs): + compiled_arms.append(tuple(kwargs["arms"])) + return {} + + monkeypatch.setattr(H, "compile_measurement_job", compile_job) + monkeypatch.setattr(H, "write_job_config", lambda *_args, **_kwargs: None) + jobs_root = tmp_path / "jobs" + cell = H.NativeCell(target, "aider") + skill_identity = H.native_trial_identity(target, "aider", "skill") + control_identity = H.native_trial_identity(target, "aider", "control") + legacy = jobs_root / "native-full" + for task_index in range(4): + for attempt in range(3): + trial = legacy / f"skill-h{task_index}-a{attempt}" + trial.mkdir(parents=True) + (trial / "lock.json").write_text(json.dumps({ + "agent": {"env": identity_env(skill_identity)}, + })) + (trial / "result.json").write_text(json.dumps({ + "task_name": f"ingot/demo-h{task_index}", + "finished_at": "2026-08-14T00:00:00Z", + })) + exports = [] + + def export(job, _metadata, *, identity): + exports.append((job.name, identity.arm)) + + monkeypatch.setattr(H, "export_job_attempts", export) + + def native(_config, _root, _name, expected, *, on_ready, **_kwargs): + assert set(expected) == {control_identity} + assert on_ready(control_identity) is True + + monkeypatch.setattr(H, "run_native_job", native) + combination = cell.combination_id + + H._run_native_full_arms( + "demo", tmp_path / "dataset", [target], ("aider",), holdout, str(source), + jobs_root, {"combinations": {}}, {combination: {"ok": True}}, + global_limit=8, endpoint_limit=4, publish_root=tmp_path / "published", + ) + + assert compiled_arms == [("control",)] + assert exports == [("native-full", "skill"), (H._native_full_job_name(cell), "control")] + assert len(rescored) == 1 + metadata = json.loads((jobs_root / combination / "combo.json").read_text()) + assert metadata["native_jobs"] == { + "skill": "native-full", "control": H._native_full_job_name(cell), + } diff --git a/tests/test_harbor_report.py b/tests/test_harbor_report.py new file mode 100644 index 0000000..2da00d4 --- /dev/null +++ b/tests/test_harbor_report.py @@ -0,0 +1,326 @@ +"""Unit tests for reading a harness x model matrix for display. + +The rules under test are the ones the first live matrix broke: a combination that did not run must +not reach a reader as a zero, every lift must carry the `n` behind it, and a matrix whose controls +sit at the ceiling must report the ceiling rather than a winner.""" +import json +import importlib +from pathlib import Path + +import pytest + +from ingot.optimize import harbor_report as R +from ingot import paths + +FIXTURES = Path(__file__).parent / "fixtures" / "harbor" + + +def test_default_matrix_root_uses_the_mutable_state_directory(monkeypatch, tmp_path): + with monkeypatch.context() as isolated: + isolated.setenv(paths.HOME, str(tmp_path / "state")) + importlib.reload(R) + assert R.HARBOR_DIR == tmp_path / "state" / "runs" / "harbor" + importlib.reload(R) + + +def _write(tmp_path, name, payload): + path = tmp_path / name + path.write_text(json.dumps(payload)) + return path + + +LIVE = {"skill": "demo", "judge": "google/gemini-2.5-flash", + "harnesses": { + "claude-code@anthropic/claude-opus-5": { + "skill_mean": 0.75, "control_mean": 0.5, "lift": 0.25, + "harness": "claude-code", "model": "anthropic/claude-opus-5", + "tasks_scored": 4, "tasks_dropped": []}, + "aider@openai/gpt-5.5": { + "error": "RuntimeError: every task returned an empty workspace", + "harness": "aider", "model": "openai/gpt-5.5"}}} + + +def test_a_combination_that_did_not_run_carries_no_lift(tmp_path): + """The whole point. A blank or a 0.0 in the lift column reads as 'measured, no effect', which + is the opposite of what happened, and is how a crashed control arm became 'lift +0.750'.""" + _write(tmp_path, "demo.json", LIVE) + rows = {r["combination"]: r for r in R.read_matrix("demo", tmp_path)["rows"]} + broken = rows["aider@openai/gpt-5.5"] + assert "lift" not in broken and "skill_mean" not in broken and "n" not in broken + assert "empty workspace" in broken["error"] + + +def test_every_measured_row_carries_the_n_behind_it(tmp_path): + _write(tmp_path, "demo.json", LIVE) + measured = [r for r in R.read_matrix("demo", tmp_path)["rows"] if "lift" in r] + assert [r["n"] for r in measured] == [4] + + +def test_scale_provenance_is_allowlisted_on_measured_and_unmeasured_rows(tmp_path): + provenance = {"target_alias": "qwen35-4b", "family": "Qwen3.5", "parameter_billions": 4.0, + "quantization": "fp8-load", "tool_parser": "qwen3_coder"} + _write(tmp_path, "demo.rescored.json", {"combinations": { + "aider@qwen35-4b--qwen35-4b-deadbeef": { + **provenance, "harness": "aider", "model": "qwen35-4b", "lift": 0.2, + "skill_mean": 0.6, "control_mean": 0.4, "tasks_scored": 4, "attempts": 1, + "private_note": "must not escape"}, + "pi@qwen35-4b--qwen35-4b-deadbeef": { + **provenance, "harness": "pi", "model": "qwen35-4b", "error": "canary failed", + "private_note": "must not escape"}, + }}) + rows = R.read_matrix("demo", tmp_path)["rows"] + assert len(rows) == 2 + assert all({key: row[key] for key in provenance} == provenance for row in rows) + assert {row["model"] for row in rows} == {"Qwen/Qwen3.5-4B"} + assert all("private_note" not in row for row in rows) + assert "lift" not in next(row for row in rows if row["harness"] == "pi") + + +def test_qwen_name_requires_matching_alias_and_family(tmp_path): + _write(tmp_path, "demo.json", {"harnesses": {"a@wire-id": { + "harness": "a", "model": "wire-id", "target_alias": "qwen35-4b", + "family": "Qwen3.6", "error": "not measured", + }}}) + + assert R.read_matrix("demo", tmp_path)["rows"][0]["model"] == "wire-id" + + +@pytest.mark.parametrize("value", [True, "4", 0, -1, float("inf")]) +def test_invalid_parameter_counts_do_not_reach_the_size_chart_payload(tmp_path, value): + _write(tmp_path, "demo.json", {"harnesses": {"a@m": { + "harness": "a", "model": "m", "parameter_billions": value, + "lift": 0.1, "skill_mean": 0.5, "control_mean": 0.4, + "tasks_scored": 4, "attempts": 1}}}) + + row = R.read_matrix("demo", tmp_path)["rows"][0] + assert "parameter_billions" not in row + + +def test_qwen_size_fixture_exposes_only_observed_scale_points(): + matrix = R.read_matrix("qwen-size-matrix", FIXTURES) + + assert matrix["models"] == ["dot-backbone", "qwen35-0.8b", "qwen35-2b", "qwen35-4b", "qwen35-9b"] + assert len(matrix["rows"]) == 8 + assert matrix["measured"] == 7 and matrix["unmeasured"] == 1 + assert all(row["parameter_billions"] in {0.8, 2.0, 4.0, 9.0, 27.0} for row in matrix["rows"]) + + +def test_the_broken_row_is_counted_but_not_averaged(tmp_path): + _write(tmp_path, "demo.json", LIVE) + out = R.read_matrix("demo", tmp_path) + assert (out["measured"], out["unmeasured"]) == (1, 1) + assert out["mean_lift"] == 0.25 # not 0.125, which is what averaging the failure would give + + +def test_a_ceilinged_matrix_reports_the_ceiling_instead_of_a_winner(tmp_path): + """Controls near the top of the scale leave less headroom than the judge's own run-to-run + spread. Naming a best combination off that is reporting noise as a result.""" + _write(tmp_path, "demo.json", {"harnesses": { + "a@m": {"lift": 0.02, "skill_mean": 0.87, "control_mean": 0.85, "tasks_scored": 4}, + "b@m": {"lift": -0.03, "skill_mean": 0.82, "control_mean": 0.85, "tasks_scored": 4}}}) + out = R.summarize(R.read_matrix("demo", tmp_path)["rows"]) + assert out["best"] is None + assert "ceiling" in out["warning"] and "too easy" in out["warning"] + + +def test_a_matrix_with_headroom_does_name_the_best_combination(tmp_path): + _write(tmp_path, "demo.json", {"harnesses": { + "a@m": {"lift": 0.25, "skill_mean": 0.75, "control_mean": 0.50, "tasks_scored": 4, + "attempts": 3}, + "b@m": {"lift": 0.05, "skill_mean": 0.55, "control_mean": 0.50, "tasks_scored": 4, + "attempts": 3}}}) + out = R.summarize(R.read_matrix("demo", tmp_path)["rows"]) + assert out["warning"] == "" + assert out["best"]["combination"] == "a@m" and out["best"]["n"] == 4 + + +def test_thin_evidence_is_flagged_even_with_headroom(tmp_path): + _write(tmp_path, "demo.json", {"harnesses": { + "a@m": {"lift": 0.4, "skill_mean": 0.6, "control_mean": 0.2, "tasks_scored": 1, + "attempts": 3}}}) + out = R.summarize(R.read_matrix("demo", tmp_path)["rows"]) + assert "fewer than 3" in out["warning"] and out["best"] is None + + +def test_a_single_attempt_matrix_will_not_be_read_as_a_ranking(tmp_path): + """Two control-arm runs of an identical configuration moved a task by 0.278 and swapped two + harnesses' rank, while re-judging one fixed answer three times was identical. One attempt per + task sits under that noise, so the difference between these rows is agent variance.""" + _write(tmp_path, "demo.json", {"harnesses": { + "a@m": {"lift": 0.25, "skill_mean": 0.60, "control_mean": 0.35, "tasks_scored": 4}, + "b@m": {"lift": 0.05, "skill_mean": 0.40, "control_mean": 0.35, "tasks_scored": 4}}}) + matrix = R.read_matrix("demo", tmp_path) + out = R.summarize(matrix["rows"]) + assert out["best"] is None + assert "one attempt" in out["warning"] and "-k 3" in out["warning"] + assert matrix["rows"][0]["attempts"] == 1 + + +def test_row_level_exploratory_evidence_disables_ranking_without_hiding_numbers(tmp_path): + _write(tmp_path, "demo.json", {"harnesses": { + "a@m": {"lift": 0.25, "skill_mean": 0.60, "control_mean": 0.35, + "tasks_scored": 4, "attempts": 1, "exploratory": True, + "rankable": False}}}) + matrix = R.read_matrix("demo", tmp_path) + + assert matrix["exploratory"] is True and matrix["rankable"] is False + assert matrix["mean_lift"] == 0.25 and matrix["rows"][0]["lift"] == 0.25 + assert matrix["model_summaries"]["m"]["best_harness"] is None + assert "exploratory" in matrix["warning"].lower() + + +def test_the_rescored_matrix_wins_when_it_is_newer(tmp_path): + """Rescoring is what removed two fabricated cells from the first grid. Showing the raw file + over a newer correction would put them back on the page.""" + _write(tmp_path, "demo.json", {"harnesses": { + "a@m": {"lift": -0.375, "skill_mean": 0.2, "control_mean": 0.575, "tasks_scored": 4}}}) + path = _write(tmp_path, "demo.rescored.json", {"combinations": { + "a@m": {"lift": 0.125, "skill_mean": 0.7, "control_mean": 0.575, "tasks_scored": 2}}}) + import os + newer = (tmp_path / "demo.json").stat().st_mtime + 60 + os.utime(path, (newer, newer)) + out = R.read_matrix("demo", tmp_path) + assert out["rescored"] is True + assert out["rows"][0]["lift"] == 0.125 + + +def test_the_rescored_schema_recovers_harness_and_model_from_the_key(tmp_path): + """harbor_rescore writes rows without harness/model fields; the combination key still has them, + and a matrix that cannot say which model a row used cannot be read as a co-occurrence grid.""" + _write(tmp_path, "demo.rescored.json", {"combinations": { + "terminus-2@anthropic/claude-opus-5": { + "lift": 0.1, "skill_mean": 0.6, "control_mean": 0.5, "tasks_scored": 4}}}) + row = R.read_matrix("demo", tmp_path)["rows"][0] + assert (row["harness"], row["model"]) == ("terminus-2", "anthropic/claude-opus-5") + + +def test_a_skill_never_run_is_absent_rather_than_empty(tmp_path): + assert R.read_matrix("never-run", tmp_path) is None + + +def test_a_corrupt_matrix_raises_rather_than_reading_as_no_results(tmp_path): + """'No combinations helped' and 'the file is broken' must not look the same on the page.""" + (tmp_path / "demo.json").write_text("{not json") + with pytest.raises(ValueError, match="unreadable"): + R.read_matrix("demo", tmp_path) + + +def test_available_lists_skills_that_have_a_matrix(tmp_path): + _write(tmp_path, "demo.json", LIVE) + _write(tmp_path, "other.rescored.json", {"combinations": {}}) + assert sorted(R.available(tmp_path)) == ["demo", "other"] + + +def _model_matrix_fixture(): + """Five model identities over nine harness identities, with sparse recorded intersections.""" + rows = { + "claude-code@ceiling-model": { + "harness": "claude-code", "model": "ceiling-model", + "target_alias": "dell-qwen", "endpoint_fingerprint": "q" * 64, + "protocol": "messages", "skill_mean": 0.90, "control_mean": 0.86, + "lift": 0.04, "tasks_scored": 4, "attempts": 3, + }, + "codex@ceiling-model": { + "harness": "codex", "model": "ceiling-model", + "skill_mean": 0.88, "control_mean": 0.86, "lift": 0.02, + "tasks_scored": 4, "attempts": 3, + }, + "aider@thin-model": { + "harness": "aider", "model": "thin-model", + "skill_mean": 0.70, "control_mean": 0.40, "lift": 0.30, + "tasks_scored": 1, "attempts": 3, + }, + "goose@single-model": { + "harness": "goose", "model": "single-model", + "skill_mean": 0.70, "control_mean": 0.40, "lift": 0.30, + "tasks_scored": 4, + }, + "opencode@headroom-model": { + "harness": "opencode", "model": "headroom-model", + "skill_mean": 0.80, "control_mean": 0.40, "lift": 0.40, + "tasks_scored": 4, "attempts": 3, + }, + "pi@headroom-model": { + "harness": "pi", "model": "headroom-model", + "skill_mean": 0.60, "control_mean": 0.40, "lift": 0.20, + "tasks_scored": 4, "attempts": 3, + }, + "terminus-2@unmeasured-model": { + "harness": "terminus-2", "model": "unmeasured-model", + "target_alias": "spark-deepseek", "endpoint_fingerprint": "d" * 64, + "protocol": "chat", "error": "canary failed", + }, + "mini-swe-agent@headroom-model": { + "harness": "mini-swe-agent", "model": "headroom-model", + "error": "full run failed", + }, + "openclaw@single-model": { + "harness": "openclaw", "model": "single-model", + "skill_mean": 0.65, "control_mean": 0.45, "lift": 0.20, + "tasks_scored": 4, + }, + } + return {"harnesses": rows} + + +def test_report_preserves_sparse_rows_axes_and_identity_metadata(tmp_path): + _write(tmp_path, "demo.json", _model_matrix_fixture()) + + out = R.read_matrix("demo", tmp_path) + + assert out["models"] == [ + "ceiling-model", "headroom-model", "single-model", + "thin-model", "unmeasured-model", + ] + assert out["harnesses"] == [ + "aider", "claude-code", "codex", "goose", "mini-swe-agent", "openclaw", + "opencode", "pi", "terminus-2", + ] + assert len(out["rows"]) == 9 + row = next(row for row in out["rows"] if row["combination"] == "claude-code@ceiling-model") + assert row["target_alias"] == "dell-qwen" + assert row["endpoint_fingerprint"] == "q" * 64 + assert row["protocol"] == "messages" + + +def test_model_summaries_refuse_ceiling_thin_and_single_attempt_independently(tmp_path): + _write(tmp_path, "demo.json", _model_matrix_fixture()) + + summaries = R.read_matrix("demo", tmp_path)["model_summaries"] + + assert set(summaries) == { + "ceiling-model", "headroom-model", "single-model", + "thin-model", "unmeasured-model", + } + assert summaries["ceiling-model"]["best_harness"] is None + assert "ceiling" in summaries["ceiling-model"]["warning"] + assert summaries["thin-model"]["best_harness"] is None + assert "fewer than 3" in summaries["thin-model"]["warning"] + assert summaries["single-model"]["best_harness"] is None + assert "one attempt" in summaries["single-model"]["warning"] + assert summaries["headroom-model"]["best_harness"] == "opencode" + assert summaries["unmeasured-model"] == { + "model": "unmeasured-model", "measured": 0, "unmeasured": 1, + "mean_lift": None, "control_mean": None, + "warning": "No combination produced a measurement.", "best_harness": None, + } + + +def test_report_has_no_global_best_and_does_not_synthesize_missing_intersections(tmp_path): + _write(tmp_path, "demo.json", _model_matrix_fixture()) + + out = R.read_matrix("demo", tmp_path) + failed = next(row for row in out["rows"] if row["combination"] == "terminus-2@unmeasured-model") + + assert "best" not in out + assert (out["measured"], out["unmeasured"]) == (7, 2) + assert "lift" not in failed + assert "skill_mean" not in failed and "control_mean" not in failed and "n" not in failed + assert failed["target_alias"] == "spark-deepseek" + assert failed["endpoint_fingerprint"] == "d" * 64 and failed["protocol"] == "chat" + measured_without_identity = next( + row for row in out["rows"] if row["combination"] == "codex@ceiling-model" + ) + assert all(field not in measured_without_identity for field in ( + "target_alias", "endpoint_fingerprint", "protocol")) + assert len(out["rows"]) < len(out["models"]) * len(out["harnesses"]) diff --git a/tests/test_harbor_rescore.py b/tests/test_harbor_rescore.py new file mode 100644 index 0000000..021bf63 --- /dev/null +++ b/tests/test_harbor_rescore.py @@ -0,0 +1,1109 @@ +"""Regression tests for Harbor evidence-only rescoring. + +These fixtures contain completed Harbor-style trials but replace only the external judge. The +rescorer still reads the same on-disk result and verifier artifact layout that live runs retain. +""" +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +import pytest +from langfuse import Langfuse + +from ingot.optimize import agy_judge as A +from ingot.optimize import harbor_langfuse as L +from ingot.optimize import harbor_report +from ingot.optimize import harbor_rescore as R + + +FINGERPRINT = "tasks-v1" +HISTORICAL_JUDGE = "judge-a" +HISTORICAL_REVISION = "harbor-rubric-v1" +RUNTIME = "agy 1.1.11" +SKILL_BODY = "fixture skill body\n" +MIGRATION_PROVENANCE = { + "skill": "demo", + "skill_body": SKILL_BODY, + "skill_sha256": hashlib.sha256(SKILL_BODY.encode()).hexdigest(), + "task_texts": {"demo-h0": "Use deterministic fixture evidence."}, +} + + +def metadata(combination: str, *, fingerprint: str = "endpoint-a", attempts: int = 3) -> dict: + return { + "combination": combination, + "harness": "codex", + "model": "qwen3.6-27b", + "target_alias": "dell", + "endpoint_fingerprint": fingerprint, + "protocol": "openai", + "task_fingerprint": FINGERPRINT, + "attempts": attempts, + "judge": HISTORICAL_JUDGE, + "scoring_revision": HISTORICAL_REVISION, + **MIGRATION_PROVENANCE, + } + + +def write_combo(root: Path, name: str, record: dict, *, artifact: str = "answer", + recorded_attempts: int = 3, receipt_metadata: dict | None = None) -> Path: + """Write a completed two-arm Harbor combination with recorded attempts per arm.""" + combo = root / name + combo.mkdir(parents=True) + (combo / "combo.json").write_text(json.dumps(record)) + for arm in ("skill", "control"): + for attempt in range(1, recorded_attempts + 1): + solution = combo / arm / f"demo-h0__attempt-{attempt}" / "verifier" / "solution" + solution.mkdir(parents=True) + (solution / "answer.md").write_text(f"{arm} {artifact}") + trial = solution.parent.parent + (trial / "result.json").write_text(json.dumps({ + "task_name": f"ingot/demo-h0__attempt-{attempt}", "exception_info": {}, + })) + payload = L.build_attempt_payload( + trial, {**(receipt_metadata or record), "arm": arm}) + digest = L._payload_sha256(payload) + (trial / "langfuse-receipt.json").write_text(json.dumps({ + "status": "verified", + "trace_id": Langfuse.create_trace_id( + seed=f"{L.EXPORTER_REVISION}:{digest}"), + "payload_sha256": digest, + "exporter_revision": L.EXPORTER_REVISION, + })) + return combo + + +def patch_current_facts(monkeypatch: pytest.MonkeyPatch) -> list[dict]: + holdout = [{"task": "demo", "rubric": ""}] + monkeypatch.setenv("JUDGE_BACKEND", "agy") + monkeypatch.setattr(R, "_task_fingerprint", lambda tasks: FINGERPRINT) + monkeypatch.setattr(R, "preflight", lambda: { + "identity": R.AGY_IDENTITY, + "model": "gemini-3.6-flash-medium", + "version": RUNTIME, + "billing_mode": "subscription", + }) + return holdout + + +def write_legacy_manifest(path: Path, root: Path, *, attempts: int = 3) -> Path: + path.write_text(json.dumps({ + "root": str(root), + "task_fingerprint": FINGERPRINT, + "attempts": attempts, + })) + return path + + +def write_provenance(path: Path) -> Path: + path.write_text(json.dumps(MIGRATION_PROVENANCE)) + return path + + +def patch_scoring(monkeypatch: pytest.MonkeyPatch) -> None: + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + + def fake_score(answers, *_args): + return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4] + + monkeypatch.setattr(R, "score", fake_score) + + +def test_discovery_uses_complete_combo_identity_across_preserved_and_local_roots(tmp_path): + preserved = tmp_path / "jobs" / "demo" + local = tmp_path / "jobs" / "demo-k3" + old = {"combination": "codex@qwen3.6-27b", "harness": "codex", "model": "qwen3.6-27b"} + local_record = metadata("codex@qwen3.6-27b--dell-22222222", fingerprint="2" * 64) + old_combo = write_combo(preserved, "codex_qwen3.6-27b", old) + local_combo = write_combo(local, "codex@qwen3.6-27b--dell-22222222", local_record) + + assert R.discover_combinations([preserved, local]) == [old_combo, local_combo] + assert R._combo_identity(local_combo) == local_record + + +@pytest.mark.parametrize("field, value", [ + ("task_fingerprint", "other-tasks"), + ("attempts", 1), +]) +def test_compatibility_mismatch_stops_before_scoring_or_publication(tmp_path, monkeypatch, field, value): + root = tmp_path / "jobs" + first = metadata("codex@qwen--dell-a", fingerprint="a" * 64) + second = metadata("codex@qwen--dell-b", fingerprint="b" * 64) + second[field] = value + write_combo(root, "first", first) + write_combo(root, "second", second) + out = tmp_path / "demo.rescored.json" + out.write_bytes(b"known-good") + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start")) + + with pytest.raises(ValueError, match=field): + R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + assert out.read_bytes() == b"known-good" + + +def test_legacy_proprietary_combo_uses_explicit_manifest_with_the_real_matrix_shape(tmp_path, monkeypatch): + harbor = tmp_path / "harbor" + root = harbor / "jobs" / "demo" + legacy = {"combination": "codex@qwen", "harness": "codex", "model": "qwen"} + write_combo(root, "codex_qwen", legacy, + receipt_metadata={**legacy, **MIGRATION_PROVENANCE}) + (harbor / "demo.json").write_text(json.dumps({ + "skill": "demo", + "tasks": 1, + "pinned_model": None, + "judge": HISTORICAL_JUDGE, + "harnesses": {"codex@qwen": {"attempts": 3}}, + })) + manifest = write_legacy_manifest(tmp_path / "legacy.json", root) + provenance = write_provenance(tmp_path / "provenance.json") + monkeypatch.setattr(R, "HARBOR_DIR", harbor) + patch_scoring(monkeypatch) + + summary = R.rescore( + "demo", jobs_roots=[root], legacy_metadata=manifest, + provenance_metadata=[provenance], log=lambda *_args: None) + + row = summary["combinations"]["codex@qwen"] + assert row["attempts"] == 3 + assert row["judge"] == R.AGY_IDENTITY + assert row["task_fingerprint"] == FINGERPRINT + assert row["scoring_revision"] == "harbor-rubric-v2-agy" + assert row["judge_runtime"] == RUNTIME + assert row["judge_billing_mode"] == "subscription" + + +def test_legacy_manifest_authorizes_its_exact_nondefault_preserved_root(tmp_path, monkeypatch): + harbor = tmp_path / "harbor" + root = harbor / "jobs" / "demo-k3" + legacy = {"combination": "codex@qwen", "harness": "codex", "model": "qwen"} + write_combo(root, "codex_qwen", legacy, + receipt_metadata={**legacy, **MIGRATION_PROVENANCE}) + (harbor / "demo.json").write_text(json.dumps({ + "skill": "demo", + "tasks": 1, + "pinned_model": None, + "judge": HISTORICAL_JUDGE, + "harnesses": {"codex@qwen": {"attempts": 3}}, + })) + manifest = write_legacy_manifest(tmp_path / "legacy.json", root) + provenance = write_provenance(tmp_path / "provenance.json") + monkeypatch.setattr(R, "HARBOR_DIR", harbor) + patch_scoring(monkeypatch) + + summary = R.rescore( + "demo", jobs_roots=[root], legacy_metadata=manifest, + provenance_metadata=[provenance], log=lambda *_args: None) + + assert summary["combinations"]["codex@qwen"]["attempts"] == 3 + + +def test_legacy_combo_without_explicit_manifest_is_refused(tmp_path, monkeypatch): + harbor = tmp_path / "harbor" + root = harbor / "jobs" / "demo" + write_combo(root, "codex_qwen", {"combination": "codex@qwen", "harness": "codex", "model": "qwen"}) + (harbor / "demo.json").write_text(json.dumps({ + "skill": "demo", "tasks": 1, "pinned_model": None, "judge": HISTORICAL_JUDGE, + "harnesses": {"codex@qwen": {"attempts": 3}}, + })) + monkeypatch.setattr(R, "HARBOR_DIR", harbor) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start")) + + with pytest.raises(ValueError, match="manifest"): + R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + +def test_legacy_combo_refuses_matrix_metadata_that_conflicts_with_its_record(tmp_path, monkeypatch): + harbor = tmp_path / "harbor" + root = harbor / "jobs" / "demo" + legacy = {"combination": "codex@qwen", "harness": "codex", "model": "qwen", "attempts": 1} + write_combo(root, "codex_qwen", legacy) + (harbor / "demo.json").write_text(json.dumps({ + "skill": "demo", "tasks": 1, "pinned_model": None, + "judge": HISTORICAL_JUDGE, + "harnesses": {"codex@qwen": {"attempts": 3}}, + })) + manifest = write_legacy_manifest(tmp_path / "legacy.json", root) + monkeypatch.setattr(R, "HARBOR_DIR", harbor) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start")) + + with pytest.raises(ValueError, match="attempts"): + R.rescore("demo", jobs_roots=[root], legacy_metadata=manifest, log=lambda *_args: None) + + +def test_legacy_combo_refuses_manifest_for_the_wrong_root(tmp_path, monkeypatch): + harbor = tmp_path / "harbor" + root = harbor / "jobs" / "demo" + legacy = {"combination": "codex@qwen", "harness": "codex", "model": "qwen"} + write_combo(root, "codex_qwen", legacy) + (harbor / "demo.json").write_text(json.dumps({ + "skill": "demo", "tasks": 1, "pinned_model": None, + "judge": HISTORICAL_JUDGE, + "harnesses": {"codex@qwen": {"attempts": 3}}, + })) + manifest = write_legacy_manifest(tmp_path / "legacy.json", harbor / "jobs" / "other") + monkeypatch.setattr(R, "HARBOR_DIR", harbor) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start")) + + with pytest.raises(ValueError, match="root"): + R.rescore("demo", jobs_roots=[root], legacy_metadata=manifest, log=lambda *_args: None) + + +@pytest.mark.parametrize("field, value", [ + ("attempts", 0), + ("attempts", False), + ("task_fingerprint", False), +]) +def test_compatibility_rejects_invalid_shared_metadata(field, value): + record = metadata("codex@qwen--dell") + record[field] = value + + with pytest.raises(ValueError, match=field): + R.validate_compatibility([record]) + + +def test_public_compatibility_validator_accepts_one_argument_and_only_checks_records(): + record = metadata("codex@qwen--dell") + + R.validate_compatibility([record]) + + +@pytest.mark.parametrize("field, value", [ + ("task_fingerprint", "old-tasks"), + ("attempts", 1), +]) +def test_current_fact_mismatch_stops_before_scoring_or_publication(tmp_path, monkeypatch, field, value): + root = tmp_path / "jobs" + first = metadata("codex@qwen--dell-a", fingerprint="a" * 64) + second = metadata("codex@qwen--dell-b", fingerprint="b" * 64) + first[field] = second[field] = value + write_combo(root, "first", first) + write_combo(root, "second", second) + out = tmp_path / "demo.rescored.json" + out.write_bytes(b"known-good") + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start")) + + with pytest.raises(ValueError, match=field): + R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + assert out.read_bytes() == b"known-good" + + +def test_rescore_refuses_missing_attempt_receipt_before_preflight_judge_or_publication(tmp_path, monkeypatch): + """Removing the live receipt gate would silently publish a matrix with withheld telemetry.""" + root = tmp_path / "jobs" + combo = write_combo(root, "one", metadata("codex@qwen--dell", fingerprint="a" * 64)) + missing = next(combo.glob("skill/*/langfuse-receipt.json")) + missing.unlink() + out = tmp_path / "demo.rescored.json" + out.write_bytes(b"known-good") + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + monkeypatch.setenv("JUDGE_BACKEND", "agy") + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], [{"task": "demo", "rubric": ""}], {})) + monkeypatch.setattr(R, "_task_fingerprint", lambda tasks: FINGERPRINT) + monkeypatch.setattr(R, "preflight", lambda: pytest.fail("judge preflight must not start")) + monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("judge must not start")) + + with pytest.raises(L.TelemetryReceiptError, match="receipt"): + R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + assert out.read_bytes() == b"known-good" + + +def test_one_persisted_provenance_document_authorizes_export_and_rescore(tmp_path, monkeypatch): + """The same explicit migration document must authorize preserved export and publication.""" + root = tmp_path / "jobs" + record = metadata("codex@qwen--dell", fingerprint="a" * 64) + for field in MIGRATION_PROVENANCE: + record.pop(field) + combo = write_combo( + root, "one", record, receipt_metadata={**record, **MIGRATION_PROVENANCE}) + for receipt in combo.glob("*/*/langfuse-receipt.json"): + receipt.unlink() + provenance_path = write_provenance(tmp_path / "provenance.json") + + class Observation: + def __init__(self, identifier: str) -> None: + self.id = identifier + + def end(self) -> None: + pass + + class PersistedReadback: + def __init__(self) -> None: + self.observations = {} + + def create_trace_id(self, *, seed: str) -> str: + return Langfuse.create_trace_id(seed=seed) + + def start_observation(self, **kwargs): + identifier = f"{len(self.observations) + 1:016x}" + trace_id = kwargs["trace_context"]["trace_id"] + self.observations[trace_id] = { + "id": identifier, + "name": kwargs["name"], + "type": kwargs["as_type"].upper(), + "metadata": kwargs["metadata"], + } + return Observation(identifier) + + def flush(self) -> None: + pass + + def read_trace(self, trace_id: str): + observation = self.observations.get(trace_id) + return {"id": trace_id, "observations": [observation]} if observation else None + + client = PersistedReadback() + provenance = L.load_provenance_metadata(provenance_path) + assert len(L.export_evidence_root(root, metadata=provenance, client=client)) == 6 + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + summary = R.rescore( + "demo", jobs_roots=[root], provenance_metadata=[provenance_path], + log=lambda *_args: None) + + assert summary["scored"] == 1 + assert len(client.observations) == 6 + + +def test_rescore_rejects_provenance_free_receipts_before_judge_preflight(tmp_path, monkeypatch): + """Even exact v2 receipts cannot authorize publication without explicit provenance.""" + root = tmp_path / "jobs" + record = metadata("codex@qwen--dell", fingerprint="a" * 64) + for field in MIGRATION_PROVENANCE: + record.pop(field) + write_combo(root, "one", record) + out = tmp_path / "demo.rescored.json" + out.write_bytes(b"known-good") + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + monkeypatch.setenv("JUDGE_BACKEND", "agy") + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], [{"task": "demo", "rubric": ""}], {})) + monkeypatch.setattr(R, "_task_fingerprint", lambda tasks: FINGERPRINT) + monkeypatch.setattr(R, "preflight", lambda: pytest.fail("judge preflight must not start")) + + with pytest.raises(L.TelemetryReceiptError, match="provenance"): + R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + assert out.read_bytes() == b"known-good" + + +def test_rescore_refuses_non_agy_backend_before_preflight_or_scoring(tmp_path, monkeypatch): + root = tmp_path / "jobs" + write_combo(root, "one", metadata("codex@qwen--dell", fingerprint="a" * 64)) + out = tmp_path / "demo.rescored.json" + out.write_bytes(b"known-good") + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setenv("JUDGE_BACKEND", "openrouter") + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(R, "preflight", lambda: pytest.fail("Agy preflight ran under wrong backend")) + monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring ran under wrong backend")) + + with pytest.raises(RuntimeError, match="JUDGE_BACKEND=agy"): + R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + assert out.read_bytes() == b"known-good" + + +def test_historical_scorers_do_not_block_one_current_agy_rescore(tmp_path, monkeypatch): + root = tmp_path / "jobs" + first = metadata("codex@qwen--dell-a", fingerprint="a" * 64) + second = metadata("codex@qwen--dell-b", fingerprint="b" * 64) + second.update({ + "judge": "other/historical-judge", + "scoring_revision": "harbor-rubric-v0", + "judge_runtime": "old runtime", + "judge_billing_mode": "metered", + "cost_usd": 1.23, + }) + write_combo(root, "first", first) + write_combo(root, "second", second) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + preflights = [] + monkeypatch.setattr(R, "preflight", lambda: preflights.append(True) or { + "identity": R.AGY_IDENTITY, + "model": "gemini-3.6-flash-medium", + "version": RUNTIME, + "billing_mode": "subscription", + }) + score_calls = [] + + def fake_score(answers, *_args): + score_calls.append(answers) + return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4] + + monkeypatch.setattr(R, "score", fake_score) + + summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + assert preflights == [True] + assert len(score_calls) == 4 + assert summary["judge"] == R.AGY_IDENTITY + assert summary["scoring_revision"] == "harbor-rubric-v2-agy" + assert summary["judge_runtime"] == RUNTIME + assert summary["judge_billing_mode"] == "subscription" + for row in summary["combinations"].values(): + assert row["judge"] == R.AGY_IDENTITY + assert row["scoring_revision"] == "harbor-rubric-v2-agy" + assert row["judge_runtime"] == RUNTIME + assert row["judge_billing_mode"] == "subscription" + assert "cost_usd" not in row + assert "cost_usd" not in summary + + +def test_stale_measurements_cannot_make_an_agy_failure_rankable(tmp_path, monkeypatch): + root = tmp_path / "jobs" + stale = { + "error": "historical failure", + "score": 0.9, + "scores": [0.9], + "skill_mean": 0.9, + "control_mean": 0.1, + "lift": 0.8, + "skill_scores": [0.9], + "control_scores": [0.1], + "tasks_scored": 99, + "tasks_dropped": ["old-task"], + "mean_lift": 0.8, + "scored": 99, + "unscorable": 0, + "n": 99, + "dropped": ["old-task"], + "best": {"combination": "historical"}, + "measured": 99, + "unmeasured": 0, + } + failed = {**metadata("codex@qwen--dell-failed", fingerprint="f" * 64), **stale} + measured = metadata("codex@qwen--dell-ok", fingerprint="o" * 64) + write_combo(root, "failed", failed, artifact="fail") + write_combo(root, "measured", measured) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + + def score_one(answers, *_args): + if "fail" in "\n".join(sum(answers.values(), [])): + raise A.AgyJudgeError("current Agy failure") + return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4] + + monkeypatch.setattr(R, "score", score_one) + + summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + failed_row = summary["combinations"][failed["combination"]] + assert failed_row["error"] == "current Agy failure" + assert not (set(stale) - {"error"}) & set(failed_row) + assert summary["scored"] == 1 + assert summary["unscorable"] == 1 + assert summary["mean_lift"] == pytest.approx(0.4) + + +def test_rescore_rejects_missing_recorded_attempt_before_scoring(tmp_path, monkeypatch): + root = tmp_path / "jobs" + incomplete = metadata("codex@qwen--dell-incomplete", fingerprint="i" * 64) + measured = metadata("codex@qwen--dell-ok", fingerprint="o" * 64) + write_combo(root, "incomplete", incomplete, recorded_attempts=2) + write_combo(root, "measured", measured) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + score_calls = [] + + def score_one(answers, *_args): + score_calls.append(answers) + return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4] + + monkeypatch.setattr(R, "score", score_one) + + summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + row = summary["combinations"][incomplete["combination"]] + assert "attempt" in row["error"].lower() + assert not {"skill_scores", "control_scores", "skill_mean", "control_mean", "lift"} & set(row) + assert len(score_calls) == 2 + assert summary["scored"] == 1 + assert summary["unscorable"] == 1 + + +def test_distinct_endpoint_fingerprints_remain_separate_measured_rows(tmp_path, monkeypatch): + root = tmp_path / "jobs" + first = metadata("codex@qwen--dell-a", fingerprint="a" * 64) + second = metadata("codex@qwen--dell-b", fingerprint="b" * 64) + write_combo(root, "first", first) + write_combo(root, "second", second) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + assert set(summary["combinations"]) == {first["combination"], second["combination"]} + assert summary["combinations"][first["combination"]]["endpoint_fingerprint"] == "a" * 64 + assert summary["combinations"][second["combination"]]["endpoint_fingerprint"] == "b" * 64 + assert summary["scored"] == 2 + + +def test_rescore_deduplicates_a_combination_when_the_same_root_is_repeated(tmp_path, monkeypatch): + root = tmp_path / "jobs" + record = metadata("codex@qwen--dell", fingerprint="d" * 64) + write_combo(root, "one", record) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + summary = R.rescore("demo", jobs_roots=[root, root], log=lambda *_args: None) + + assert list(summary["combinations"]) == [record["combination"]] + assert summary["scored"] == 1 + + +def test_rescore_selects_only_exact_completed_combination_paths(tmp_path, monkeypatch): + root = tmp_path / "jobs" + complete = metadata("aider@qwen--dell", fingerprint="a" * 64) + active = metadata("codex@qwen--dell", fingerprint="b" * 64) + write_combo(root, "complete", complete) + active_dir = root / "active" + active_dir.mkdir(parents=True) + (active_dir / "combo.json").write_text("active evidence must not be read") + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + summary = R.rescore( + "demo", jobs_roots=[root], combination_paths=[root / "complete"], + log=lambda *_args: None) + + assert list(summary["combinations"]) == [complete["combination"]] + assert summary["scored"] == 1 + + +def test_rescore_rejects_selected_path_outside_jobs_roots_before_preflight(tmp_path, monkeypatch): + root = tmp_path / "jobs" + write_combo(root, "complete", metadata("aider@qwen--dell", fingerprint="a" * 64)) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(R, "preflight", lambda: pytest.fail("preflight must not start")) + + outside = tmp_path / "outside" + write_combo(outside, "other", metadata("other@qwen--dell", fingerprint="b" * 64)) + + with pytest.raises(ValueError, match="immediate child"): + R.rescore("demo", jobs_roots=[root], combination_paths=[outside / "other"], + log=lambda *_args: None) + + +def test_explicit_empty_selection_never_discovers_active_siblings(tmp_path, monkeypatch): + root = tmp_path / "jobs" + active = root / "active" + active.mkdir(parents=True) + (active / "combo.json").write_text("must not be read") + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(R, "preflight", lambda: pytest.fail("preflight must not start")) + + with pytest.raises(ValueError, match="no completed combination paths selected"): + R.rescore("demo", jobs_roots=[root], combination_paths=[], log=lambda *_args: None) + + +def test_rescore_checkpoints_each_new_lift_to_explicit_output(tmp_path, monkeypatch): + root = tmp_path / "jobs" + first = metadata("aider@qwen--dell", fingerprint="a" * 64) + second = metadata("codex@qwen--dell", fingerprint="b" * 64) + write_combo(root, "first", first) + write_combo(root, "second", second) + output = tmp_path / "published" / "build-loop.rescored.json" + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + writes = [] + real_write = R.atomic_write_json + + def record_write(path, payload): + writes.append((path, list(payload["combinations"]))) + real_write(path, payload) + + monkeypatch.setattr(R, "atomic_write_json", record_write) + + summary = R.rescore("demo", jobs_roots=[root], output=output, log=lambda *_args: None) + + assert writes == [ + (output, [first["combination"]]), + (output, [first["combination"], second["combination"]]), + ] + assert json.loads(output.read_text()) == summary + + +def test_rescore_requests_bounded_parallel_agy_grading(tmp_path, monkeypatch): + root = tmp_path / "jobs" + write_combo(root, "one", metadata("aider@qwen--dell", fingerprint="a" * 64)) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + concurrencies = [] + + def fake_score(_answers, _skill, _holdout, _skipped, concurrency): + concurrencies.append(concurrency) + return [0.5] + + monkeypatch.setattr(R, "score", fake_score) + + R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + assert concurrencies == [4, 4] + + +def test_selected_rescore_preserves_prior_compatible_lifts(tmp_path, monkeypatch): + root = tmp_path / "jobs" + first = metadata("aider@qwen--dell", fingerprint="a" * 64) + second = metadata("codex@qwen--dell", fingerprint="b" * 64) + first_path = write_combo(root, "first", first) + second_path = write_combo(root, "second", second) + output = tmp_path / "published" / "build-loop.rescored.json" + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + original_score = R.score + score_calls = [] + + def counted_score(*args, **kwargs): + score_calls.append(args[0]) + return original_score(*args, **kwargs) + + monkeypatch.setattr(R, "score", counted_score) + + R.rescore("demo", jobs_roots=[root], combination_paths=[first_path], output=output, + log=lambda *_args: None) + summary = R.rescore("demo", jobs_roots=[root], combination_paths=[second_path], output=output, + log=lambda *_args: None) + + assert list(summary["combinations"]) == [first["combination"], second["combination"]] + assert summary["scored"] == 2 + assert len(score_calls) == 4 # two arms once per cell; the first cell is not paid twice + + +def test_current_canary_failure_replaces_its_prior_lift_without_rejudging(tmp_path, monkeypatch): + root = tmp_path / "jobs" + failed = metadata("aider@qwen--dell", fingerprint="a" * 64) + unaffected = metadata("codex@qwen--dell", fingerprint="b" * 64) + failed_path = write_combo(root, "failed", failed) + unaffected_path = write_combo(root, "unaffected", unaffected) + output = tmp_path / "published" / "demo.rescored.json" + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + prior = R.rescore( + "demo", jobs_roots=[root], combination_paths=[failed_path, unaffected_path], + output=output, log=lambda *_args: None, + ) + unaffected_row = prior["combinations"][unaffected["combination"]] + (failed_path / "combo.json").write_text(json.dumps({ + **failed, + "canary_error": "current canary failed", + })) + monkeypatch.setattr(R, "score", lambda *_args, **_kwargs: + pytest.fail("current canary failure invoked the judge")) + + summary = R.rescore( + "demo", jobs_roots=[root], combination_paths=[failed_path], + output=output, log=lambda *_args: None, + ) + + failed_row = summary["combinations"][failed["combination"]] + assert failed_row["error"] == "current canary failed" + assert not {"lift", "skill_mean", "control_mean", "skill_scores", "control_scores"} & set( + failed_row) + assert summary["combinations"][unaffected["combination"]] == unaffected_row + assert summary["scored"] == 1 + assert summary["mean_lift"] == unaffected_row["lift"] + + +def test_selected_rescore_preserves_scale_provenance_for_report(tmp_path, monkeypatch): + root = tmp_path / "jobs" + record = metadata("aider@qwen35-4b--dell", fingerprint="4" * 64) + record.update({"family": "Qwen3.5", "parameter_billions": 4.0, + "quantization": "fp8-load", "tool_parser": "qwen3_coder", + "exploratory": True, "rankable": False}) + combo = write_combo(root, "qwen35-4b", record) + output_root = tmp_path / "published" + output = output_root / "demo.rescored.json" + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + R.rescore("demo", jobs_roots=[root], combination_paths=[combo], output=output, + log=lambda *_args: None) + matrix = harbor_report.read_matrix("demo", output_root) + row = matrix["rows"][0] + + assert {key: row[key] for key in ( + "family", "parameter_billions", "quantization", "tool_parser") + } == {"family": "Qwen3.5", "parameter_billions": 4.0, + "quantization": "fp8-load", "tool_parser": "qwen3_coder"} + assert matrix["rankable"] is False + assert matrix["exploratory"] is True + + +def test_prior_row_without_treatment_fields_does_not_reuse_lift(): + identity = metadata("aider@qwen--dell", fingerprint="a" * 64) + identity.update({"family": "Qwen3.5", "parameter_billions": 4.0, + "quantization": "fp8-load", "tool_parser": "qwen3_coder"}) + prior = {"prior": {key: value for key, value in identity.items() + if key not in {"family", "parameter_billions", "quantization", + "tool_parser"}}} + + assert R._matching_row_key(identity, prior) is None + + +def test_prior_row_with_stale_gateway_revision_does_not_reuse_lift(): + identity = metadata("codex@qwen--dell", fingerprint="a" * 64) + identity.update({"gateway_revision": "v8", "gateway_identity": "route-v8"}) + prior = {"prior": {**identity, "gateway_revision": "v7", "lift": 0.5}} + + assert R._matching_row_key(identity, prior) is None + + +def test_current_runtime_replaces_stale_prior_row_for_same_logical_cell(tmp_path, monkeypatch): + root = tmp_path / "jobs" + current = metadata("codex@qwen--dell", fingerprint="a" * 64) + current.update({"gateway_revision": f"route-v8-{R.NATIVE_RUNNER_REVISION}", + "gateway_identity": f"route-v8-{R.NATIVE_RUNNER_REVISION}"}) + combo = write_combo(root, "codex-qwen", current) + output = tmp_path / "demo.rescored.json" + stale = {**current, "gateway_revision": "route-v8-native-v2", + "gateway_identity": "route-v8-native-v2", "lift": 0.9, + "skill_mean": 1.0, "control_mean": 0.1} + output.write_text(json.dumps({ + "skill": "demo", "task_fingerprint": current["task_fingerprint"], + "attempts": current["attempts"], "judge": R.AGY_IDENTITY, + "scoring_revision": "harbor-rubric-v2-agy", "judge_billing_mode": "subscription", + "judge_runtime": RUNTIME, "combinations": {"stale": stale}, + })) + patch_scoring(monkeypatch) + calls = [] + monkeypatch.setattr(R, "score", lambda *_args: calls.append(1) or [0.6]) + + result = R.rescore("demo", jobs_roots=[root], combination_paths=[combo], output=output, + log=lambda *_args: None) + + assert len(calls) == 2 + assert result["scored"] == 1 + assert len(result["combinations"]) == 1 + [row] = result["combinations"].values() + assert row["gateway_revision"] == f"route-v8-{R.NATIVE_RUNNER_REVISION}" + assert row["lift"] == 0.0 + + +def test_current_runtime_wins_over_later_stale_root_independent_of_input_order(tmp_path, + monkeypatch): + current_root, stale_root = tmp_path / "current", tmp_path / "stale" + base = metadata("codex@qwen--dell", fingerprint="a" * 64) + current = {**base, "gateway_revision": f"route-v8-{R.NATIVE_RUNNER_REVISION}", + "gateway_identity": f"route-v8-{R.NATIVE_RUNNER_REVISION}"} + stale = {**base, "gateway_revision": "route-v8-native-v2", + "gateway_identity": "route-v8-native-v2"} + write_combo(current_root, "current", current) + write_combo(stale_root, "stale", stale) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path / "published") + patch_scoring(monkeypatch) + + result = R.rescore("demo", jobs_roots=[current_root, stale_root], log=lambda *_args: None) + + assert result["scored"] == 1 + assert len(result["combinations"]) == 1 + [row] = result["combinations"].values() + assert row["gateway_revision"] == f"route-v8-{R.NATIVE_RUNNER_REVISION}" + + +def test_selected_rescore_refuses_incompatible_prior_output(tmp_path, monkeypatch): + root = tmp_path / "jobs" + combo = write_combo(root, "one", metadata("aider@qwen--dell", fingerprint="a" * 64)) + output = tmp_path / "build-loop.rescored.json" + output.write_text(json.dumps({ + "skill": "demo", "task_fingerprint": "other", "attempts": 3, + "judge": R.AGY_IDENTITY, "scoring_revision": "harbor-rubric-v2-agy", + "judge_billing_mode": "subscription", "judge_runtime": RUNTIME, + "combinations": {}, + })) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + monkeypatch.setattr(R, "score", lambda *_args: pytest.fail("scoring must not start")) + + with pytest.raises(ValueError, match="incompatible existing rescore output"): + R.rescore("demo", jobs_roots=[root], combination_paths=[combo], output=output, + log=lambda *_args: None) + + +def test_selected_rescore_drops_prior_lifts_from_another_skill_revision(tmp_path, monkeypatch): + root = tmp_path / "jobs" + first = metadata("aider@qwen--dell", fingerprint="a" * 64) + second = metadata("codex@qwen--dell", fingerprint="b" * 64) + second["skill_body"] = "different skill revision\n" + second["skill_sha256"] = hashlib.sha256(second["skill_body"].encode()).hexdigest() + first_path = write_combo(root, "first", first) + second_path = write_combo(root, "second", second) + output = tmp_path / "build-loop.rescored.json" + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + R.rescore("demo", jobs_roots=[root], combination_paths=[first_path], output=output, + log=lambda *_args: None) + summary = R.rescore("demo", jobs_roots=[root], combination_paths=[second_path], output=output, + log=lambda *_args: None) + + assert list(summary["combinations"]) == [second["combination"]] + + +def test_agy_failure_row_has_current_identity_and_no_measurements(tmp_path, monkeypatch): + root = tmp_path / "jobs" + failed = metadata("codex@qwen--dell-failed", fingerprint="f" * 64) + measured = metadata("codex@qwen--dell-ok", fingerprint="o" * 64) + write_combo(root, "failed", failed, artifact="fail") + write_combo(root, "measured", measured) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + + def score_one(answers, *_args): + if "fail" in "\n".join(sum(answers.values(), [])): + raise A.AgyJudgeError("judge unavailable") + return [0.8 if "skill answer" in "\n".join(sum(answers.values(), [])) else 0.4] + + monkeypatch.setattr(R, "score", score_one) + summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + failed_row = summary["combinations"][failed["combination"]] + assert failed_row["error"] == "judge unavailable" + assert failed_row["judge"] == R.AGY_IDENTITY + assert failed_row["scoring_revision"] == "harbor-rubric-v2-agy" + assert failed_row["judge_runtime"] == RUNTIME + assert failed_row["judge_billing_mode"] == "subscription" + assert not {"score", "skill_scores", "control_scores", "skill_mean", "control_mean", "lift"} & set(failed_row) + row = summary["combinations"][measured["combination"]] + assert row["attempts"] == 3 + assert row["target_alias"] == "dell" + assert row["endpoint_fingerprint"] == "o" * 64 + assert row["protocol"] == "openai" + + +def test_rescore_preserves_failed_canary_as_explicit_unmeasured_row(tmp_path, monkeypatch): + root = tmp_path / "jobs" + failed = {**metadata("codex@qwen--dell-failed", fingerprint="f" * 64), + "canary_error": "ApiRateLimitError: local gateway returned 429"} + measured = metadata("aider@qwen--dell-ok", fingerprint="o" * 64) + failed_dir = root / "failed" + failed_dir.mkdir(parents=True) + (failed_dir / "combo.json").write_text(json.dumps(failed)) + write_combo(root, "measured", measured) + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + summary = R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + row = summary["combinations"][failed["combination"]] + assert row["error"] == failed["canary_error"] + assert "canary_error" not in row + assert not {"skill_mean", "control_mean", "lift", "skill_scores", "control_scores"} & set(row) + assert summary["scored"] == 1 + assert summary["unscorable"] == 1 + + +def test_rescore_publishes_trailing_failed_canary_in_final_summary(tmp_path, monkeypatch): + root = tmp_path / "jobs" + measured = metadata("aider@qwen--dell-ok", fingerprint="o" * 64) + failed = {**metadata("codex@qwen--dell-failed", fingerprint="f" * 64), + "canary_error": "canary failed"} + measured_path = write_combo(root, "measured", measured) + failed_path = root / "failed" + failed_path.mkdir(parents=True) + (failed_path / "combo.json").write_text(json.dumps(failed)) + output = tmp_path / "demo.rescored.json" + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + summary = R.rescore( + "demo", jobs_roots=[root], combination_paths=[measured_path, failed_path], + output=output, log=lambda *_args: None) + + assert json.loads(output.read_text()) == summary + assert summary["combinations"][failed["combination"]]["error"] == "canary failed" + + +def test_rescore_admits_explicit_partial_arm_measurement_error(tmp_path, monkeypatch): + root = tmp_path / "jobs" + measured = metadata("aider@qwen--dell-ok", fingerprint="o" * 64) + partial = {**metadata("codex@qwen--dell-partial", fingerprint="p" * 64), + "measurement_error": "control arm failed"} + measured_path = write_combo(root, "measured", measured) + partial_path = root / "partial" + partial_path.mkdir(parents=True) + (partial_path / "combo.json").write_text(json.dumps(partial)) + (partial_path / "skill").mkdir() + output = tmp_path / "demo.rescored.json" + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + patch_scoring(monkeypatch) + + summary = R.rescore( + "demo", jobs_roots=[root], combination_paths=[measured_path, partial_path], + output=output, log=lambda *_args: None) + + row = summary["combinations"][partial["combination"]] + assert row["error"] == "control arm failed" + assert "lift" not in row + + +def test_rescore_refuses_to_publish_when_every_combination_is_unscorable(tmp_path, monkeypatch): + root = tmp_path / "jobs" + write_combo(root, "failed", metadata("codex@qwen--dell", fingerprint="f" * 64)) + out = tmp_path / "demo.rescored.json" + out.write_bytes(b"known-good") + monkeypatch.setattr(R, "HARBOR_DIR", tmp_path) + holdout = patch_current_facts(monkeypatch) + monkeypatch.setattr(R, "load_tasks", lambda skill: ([], holdout, {})) + monkeypatch.setattr(R, "score", lambda *_args: (_ for _ in ()).throw(RuntimeError("judge down"))) + + with pytest.raises(SystemExit, match="nothing was measured"): + R.rescore("demo", jobs_roots=[root], log=lambda *_args: None) + + assert out.read_bytes() == b"known-good" + + +def test_atomic_write_preserves_existing_bytes_on_serialization_and_replace_failure(tmp_path, monkeypatch): + path = tmp_path / "matrix.json" + path.write_bytes(b"known-good") + + with pytest.raises(TypeError): + R.atomic_write_json(path, {"bad": {1, 2}}) + assert path.read_bytes() == b"known-good" + assert not list(tmp_path.glob(".matrix.json.*.tmp")) + + def broken_replace(source, destination): + assert json.loads(Path(source).read_text()) == {"next": 1} + raise OSError("disk failed") + + monkeypatch.setattr(R.os, "replace", broken_replace) + with pytest.raises(OSError, match="disk failed"): + R.atomic_write_json(path, {"next": 1}) + assert path.read_bytes() == b"known-good" + assert not list(tmp_path.glob(".matrix.json.*.tmp")) + + +def test_atomic_write_replaces_only_a_complete_json_file(tmp_path, monkeypatch): + path = tmp_path / "matrix.json" + seen = [] + original_replace = R.os.replace + + def inspect_then_replace(source, destination): + seen.append(json.loads(Path(source).read_text())) + original_replace(source, destination) + + monkeypatch.setattr(R.os, "replace", inspect_then_replace) + R.atomic_write_json(path, {"next": [1, 2]}) + + assert seen == [{"next": [1, 2]}] + assert json.loads(path.read_text()) == {"next": [1, 2]} + assert not list(tmp_path.glob(".matrix.json.*.tmp")) + + +def test_native_combo_resolves_exact_sibling_job_and_arm_identity(tmp_path): + from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env + + root = tmp_path / "jobs" + combo = root / "aider-cell" + job = root / "native-full" + combo.mkdir(parents=True) + job.mkdir() + common = dict(combination_id="aider@dot-backbone--dell-qwen-deadbeefcafe", + endpoint_fingerprint="deadbeefcafe", harness="aider", protocol="chat", + gateway_revision="direct") + identities = {arm: NativeTrialIdentity(**common, arm=arm) + for arm in ("skill", "control")} + (combo / "combo.json").write_text(json.dumps({ + "combination": common["combination_id"], + "harness": common["harness"], + "endpoint_fingerprint": common["endpoint_fingerprint"], + "protocol": common["protocol"], + "gateway_revision": common["gateway_revision"], + "native_job": "native-full", + "native_identities": {arm: identity_env(identity) + for arm, identity in identities.items()}, + })) + + assert R._arm_evidence(combo, "skill") == (job, identities["skill"]) + assert R._arm_evidence(combo, "control") == (job, identities["control"]) + + +def test_native_combo_refuses_lock_identity_mismatched_to_combo_metadata(tmp_path): + from ingot.optimize.harbor_native import NativeTrialIdentity, identity_env + + combo = tmp_path / "jobs" / "cell" + combo.mkdir(parents=True) + common = dict(combination_id="aider@dot-backbone--dell-qwen-deadbeefcafe", + endpoint_fingerprint="deadbeefcafe", harness="aider", protocol="chat", + gateway_revision="direct") + (combo / "combo.json").write_text(json.dumps({ + "combination": "aider@other--dell-qwen-deadbeefcafe", + "harness": "aider", "endpoint_fingerprint": "deadbeefcafe", + "protocol": "chat", "gateway_revision": "direct", "native_job": "native-full", + "native_identities": { + arm: identity_env(NativeTrialIdentity(**common, arm=arm)) + for arm in ("skill", "control")}, + })) + + with pytest.raises(ValueError, match="does not match combo metadata"): + R._arm_evidence(combo, "skill") + + +@pytest.mark.parametrize("native_job", ["../other", "/tmp/other", "http:job"]) +def test_native_combo_refuses_non_sibling_job_reference(tmp_path, native_job): + combo = tmp_path / "jobs" / "cell" + combo.mkdir(parents=True) + (combo / "combo.json").write_text(json.dumps({ + "native_job": native_job, + "native_identities": {}, + })) + + with pytest.raises(ValueError, match="native job"): + R._arm_evidence(combo, "skill") + + +def test_discovery_skips_only_native_job_declared_by_a_combo(tmp_path): + root = tmp_path / "jobs" + combo = root / "cell" + native = root / "native-full" + combo.mkdir(parents=True) + native.mkdir() + (combo / "combo.json").write_text(json.dumps({ + **metadata("aider@dot-backbone--dell-qwen-deadbeefcafe", + fingerprint="deadbeefcafe"), + "native_job": "native-full", + })) + + assert R.discover_combinations([root]) == [combo] + + (root / "native-full--aider--other").mkdir() + assert R.discover_combinations([root]) == [combo] + + +def test_discovery_rejects_a_claimed_combo_with_malformed_identity(tmp_path): + root = tmp_path / "jobs" + malformed = root / "malformed" + malformed.mkdir(parents=True) + (malformed / "combo.json").write_text("[]") + + with pytest.raises(ValueError, match="no combo identity"): + R.discover_combinations([root]) diff --git a/tests/test_harbor_rescore_exploratory.py b/tests/test_harbor_rescore_exploratory.py new file mode 100644 index 0000000..b186701 --- /dev/null +++ b/tests/test_harbor_rescore_exploratory.py @@ -0,0 +1,17 @@ +import pytest + +from ingot.optimize import harbor_rescore as R + + +def test_current_compatibility_accepts_only_explicit_k1_exploration(): + holdout = [{"task": "one"}] + fingerprint = R._task_fingerprint(holdout) + R._validate_current_compatibility([ + {"task_fingerprint": fingerprint, "attempts": 1, + "exploratory": True, "rankable": False} + ], holdout) + with pytest.raises(ValueError, match="measurement contract"): + R._validate_current_compatibility([ + {"task_fingerprint": fingerprint, "attempts": 1, + "exploratory": False, "rankable": True} + ], holdout) diff --git a/tests/test_harbor_targets.py b/tests/test_harbor_targets.py new file mode 100644 index 0000000..2fa7385 --- /dev/null +++ b/tests/test_harbor_targets.py @@ -0,0 +1,375 @@ +"""Tests for local Harbor target identity, discovery, routing, and isolation.""" +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from ingot.optimize import harbor_targets as H + + +FIXTURES = Path(__file__).parent / "fixtures" / "harbor" +TARGETS = { + "dell-qwen": ("dot-backbone", 163840, FIXTURES / "qwen-models.json"), + "spark-deepseek": ("deepseek-v4-flash", 1048576, FIXTURES / "deepseek-models.json"), + "orin-abliterated": ("ablit35b", 65536, FIXTURES / "orin-models.json"), +} +HARNESS_PROTOCOLS = { + "claude-code": "messages", + "terminus-2": "chat", + "goose": "chat", + "opencode": "chat", + "openclaw": "chat", + "mini-swe-agent": "chat", + "codex": "responses", + "aider": "chat", + "pi": "chat", +} + + +class _Response: + def __init__(self, payload: dict, status: int | None = None): + self.payload = payload + self.status = status + + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + def read(self): + return json.dumps(self.payload).encode() + + +def _target(alias: str = "dell-qwen", **changes) -> H.LocalTarget: + model, context, _ = TARGETS[alias] + values = { + "alias": alias, + "display_name": H.TARGETS[alias]["display_name"], + "base_url": "http://host:8011", + "served_model": model, + "context_length": context, + "protocols": frozenset({"chat", "responses", "messages"}), + "family": H.TARGETS[alias]["family"], + "parameter_billions": H.TARGETS[alias]["parameter_billions"], + "quantization": H.TARGETS[alias]["quantization"], + "tool_parser": H.TARGETS[alias]["tool_parser"], + } + values.update(changes) + return H.LocalTarget(**values) + + +def test_fixtures_preserve_served_ids_and_context_lengths(): + for alias, (served_model, context_length, path) in TARGETS.items(): + payload = json.loads(path.read_text()) + model = payload["data"][0] + assert model["id"] == served_model + assert model.get("max_model_len", model.get("meta", {}).get("n_ctx")) == context_length + assert alias in H.TARGETS + + +def test_parse_target_accepts_only_configured_aliases_and_normalizes_url(): + target = H.parse_target("dell-qwen=http://host:8011/") + assert target.alias == "dell-qwen" + assert target.base_url == "http://host:8011" + assert target.served_model == "dot-backbone" + with pytest.raises(ValueError, match="unknown local target alias"): + H.parse_target("unknown=http://host:8011") + + +@pytest.mark.parametrize("alias", list(TARGETS)) +def test_discovery_finds_configured_model_and_context(alias, monkeypatch): + served_model, context_length, fixture = TARGETS[alias] + payload = json.loads(fixture.read_text()) + + def fake_urlopen(request, timeout): + assert request.full_url == "http://local.test/v1/models" + assert timeout == 3.5 + return _Response(payload) + + monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen) + target = H.discover_target(alias, "http://local.test/", timeout=3.5) + assert target.served_model == served_model + assert target.context_length == context_length + assert target.protocols == frozenset({"chat", "responses", "messages"}) + + +def test_ollama_discovery_reads_loaded_runtime_context(monkeypatch): + seen = [] + + def fake_urlopen(request, timeout): + seen.append((request.full_url, json.loads(request.data) if request.data else None, timeout)) + if request.full_url.endswith("/v1/models"): + return _Response({"object": "list", "data": [{"id": "qwen3.5:9b"}]}) + return _Response({"models": [ + {"name": "other", "context_length": 8192}, + {"name": "qwen3.5:9b", "context_length": 262144}, + ]}) + + monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen) + target = H.discover_target("orin-qwen35-9b", "http://orin.test:11434", timeout=4) + + assert target.served_model == "qwen3.5:9b" + assert target.context_length == 262144 + assert target.display_name == "Qwen/Qwen3.5-9B (Q4_K_M)" + assert seen == [ + ("http://orin.test:11434/v1/models", None, 4), + ("http://orin.test:11434/api/ps", None, 4), + ] + + +@pytest.mark.parametrize(("meta", "message"), [ + ({"n_ctx": True}, "no context length"), + ({"n_ctx": "65536"}, "no context length"), + ({"n_ctx": 32767}, "below the minimum"), + ({}, "no context length"), + ([], "no context length"), +]) +def test_discovery_rejects_invalid_llamacpp_context_metadata(meta, message, monkeypatch): + payload = {"object": "list", "data": [{"id": "ablit35b", "meta": meta}]} + monkeypatch.setattr(H.urllib.request, "urlopen", lambda *args, **kwargs: _Response(payload)) + with pytest.raises(ValueError, match=message): + H.discover_target("orin-abliterated", "http://local.test") + + +def test_discovery_rejects_short_context_and_wrong_served_model(monkeypatch): + payload = {"object": "list", "data": [{"id": "dot-backbone", "max_model_len": 8192}]} + monkeypatch.setattr(H.urllib.request, "urlopen", lambda *args, **kwargs: _Response(payload)) + with pytest.raises(ValueError, match="context length"): + H.discover_target("dell-qwen", "http://local.test") + + payload["data"][0] = {"id": "other", "max_model_len": 32768} + with pytest.raises(ValueError, match="served model"): + H.discover_target("dell-qwen", "http://local.test") + + +def test_fingerprint_changes_with_normalized_url_or_served_model(): + base = _target() + assert base.fingerprint == "a7f8512ae664" # pre-scale-metadata identity stays readable + assert base.fingerprint == _target(base_url="http://host:8011/").fingerprint + assert base.fingerprint != _target(base_url="http://host:8002").fingerprint + assert base.fingerprint != _target(served_model="dot-backbone-other").fingerprint + assert base.fingerprint == _target(family="Qwen-next").fingerprint + assert base.fingerprint == _target(parameter_billions=99.0).fingerprint + assert base.fingerprint == _target(quantization="fp8-load").fingerprint + assert base.fingerprint == _target(tool_parser="qwen3_coder").fingerprint + assert base.alias in base.job_slug + assert base.fingerprint in base.job_slug + assert "http" not in base.job_slug and "host" not in base.job_slug + + +def test_qwen_size_targets_have_exact_scale_provenance(): + assert { + alias: {key: config[key] for key in ( + "served_model", "family", "parameter_billions", "quantization", "tool_parser")} + for alias, config in H.TARGETS.items() if alias.startswith("qwen35-") + } == { + "qwen35-08b": {"served_model": "qwen35-0.8b", "family": "Qwen3.5", + "parameter_billions": 0.8, "quantization": "fp8-load", + "tool_parser": "qwen3_coder"}, + "qwen35-2b": {"served_model": "qwen35-2b", "family": "Qwen3.5", + "parameter_billions": 2.0, "quantization": "fp8-load", + "tool_parser": "qwen3_coder"}, + "qwen35-4b": {"served_model": "qwen35-4b", "family": "Qwen3.5", + "parameter_billions": 4.0, "quantization": "fp8-load", + "tool_parser": "qwen3_coder"}, + "qwen35-9b": {"served_model": "qwen35-9b", "family": "Qwen3.5", + "parameter_billions": 9.0, "quantization": "fp8-load", + "tool_parser": "qwen3_coder"}, + } + assert {key: H.TARGETS["dell-qwen"][key] for key in ( + "family", "parameter_billions", "quantization", "tool_parser") + } == {"family": "Qwen3.6", "parameter_billions": 27.0, + "quantization": "fp8-published", "tool_parser": "qwen3_xml"} + assert {key: H.TARGETS["orin-qwen35-9b"][key] for key in ( + "served_model", "family", "parameter_billions", "quantization", "tool_parser") + } == {"served_model": "qwen3.5:9b", "family": "Qwen3.5", + "parameter_billions": 9.7, "quantization": "Q4_K_M", + "tool_parser": "ollama"} + + +@pytest.mark.parametrize("alias", ["qwen35-08b", "qwen35-2b", "qwen35-4b", "qwen35-9b"]) +def test_qwen_size_aliases_propagate_provenance_through_parse(alias): + target = H.parse_target(f"{alias}=http://local.test:8020") + config = H.TARGETS[alias] + assert {field: getattr(target, field) for field in ( + "family", "parameter_billions", "quantization", "tool_parser") + } == {field: config[field] for field in ( + "family", "parameter_billions", "quantization", "tool_parser")} + + +@pytest.mark.parametrize("harness, protocol", list(HARNESS_PROTOCOLS.items())) +def test_harnesses_map_to_the_required_protocol(harness, protocol): + assert H.protocol_for(harness) == protocol + expected = "dot-backbone" if harness == "claude-code" else "openai/dot-backbone" + if harness == "opencode": + expected = "local/dot-backbone" + assert H.harbor_model(_target(), harness) == expected + + +def test_harbor_kwargs_use_direct_api_base_only_for_terminus(): + target = _target() + assert H.harbor_agent_kwargs(target, "terminus-2") == {"api_base": f"{target.base_url}/v1"} + assert H.harbor_agent_kwargs(target, "openclaw") == {"thinking": "off"} + assert H.harbor_model(target, "opencode") == "local/dot-backbone" + assert H.harbor_agent_kwargs(target, "opencode") == { + "opencode_config": { + "provider": { + "local": { + "npm": "@ai-sdk/openai-compatible", + "options": {"baseURL": f"{target.base_url}/v1", "apiKey": "local"}, + "models": {"dot-backbone": { + "limit": {"context": 163840, "output": 40960}, + }}, + } + } + } + } + assert H.harbor_agent_kwargs(target, "claude-code") == {} + + +def test_scrub_provider_env_removes_credentials_and_provider_routing(): + parent = { + "PATH": "/bin", + "ANTHROPIC_AUTH_TOKEN": "secret-a", + "CLAUDE_CODE_OAUTH_TOKEN": "secret-b", + "CLAUDE_FORCE_OAUTH": "1", + "ANTHROPIC_BASE_URL": "https://provider.invalid", + "OPENAI_API_KEY": "secret-c", + "CODEX_API_KEY": "secret-d", + "CODEX_FORCE_AUTH_JSON": "1", + "OPENAI_BASE_URL": "https://provider.invalid", + "LITELLM_API_KEY": "secret-e", + "GEMINI_API_KEY": "secret-f", + "OPENROUTER_API_KEY": "secret-g", + "API_KEY": "secret-h", + "MODEL_API_KEY": "secret-i", + } + scrubbed = H.scrub_provider_env(parent) + assert scrubbed == {"PATH": "/bin"} + + +@pytest.mark.parametrize("harness, key, base, model", [ + ("claude-code", "ANTHROPIC_API_KEY", "ANTHROPIC_BASE_URL", "ANTHROPIC_MODEL"), + ("codex", "OPENAI_API_KEY", "OPENAI_BASE_URL", None), + ("goose", "OPENAI_API_KEY", "OPENAI_BASE_URL", None), +]) +def test_local_environment_uses_sentinel_key_and_explicit_routing( + harness, key, base, model, monkeypatch +): + monkeypatch.setenv("ANTHROPIC_AUTH_TOKEN", "secret") + monkeypatch.setenv("OPENAI_API_KEY", "secret") + monkeypatch.setenv("OPENAI_BASE_URL", "https://wrong.invalid") + monkeypatch.setenv("UNRELATED_AMBIENT_SECRET", "must-not-reach-harbor") + target = _target() + env = H.local_agent_env(target, harness) + assert env[key] == "local" + assert env["ANTHROPIC_API_KEY"] == "local" + assert env["OPENAI_API_KEY"] == "local" + assert env["CODEX_API_KEY"] == "local" + expected_base = target.base_url if harness == "claude-code" else f"{target.base_url}/v1" + assert env[base] == expected_base + if harness != "claude-code": + assert env["OPENAI_API_BASE"] == expected_base + assert env["OPENAI_HOST"] == target.base_url + if model: + assert env[model] == target.served_model + assert "ANTHROPIC_AUTH_TOKEN" not in env + assert "UNRELATED_AMBIENT_SECRET" not in env + assert env.get("CODEX_API_KEY", "local") == "local" + + +def test_probe_protocol_posts_one_token_and_requires_nonempty_object(monkeypatch): + seen = {} + + def fake_urlopen(request, timeout): + seen.update(url=request.full_url, timeout=timeout, body=json.loads(request.data)) + return _Response({"id": "response-1", "object": "response"}) + + monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen) + H.probe_protocol(_target(), "responses", timeout=7.0) + assert seen["url"] == "http://host:8011/v1/responses" + assert seen["timeout"] == 7.0 + assert seen["body"]["model"] == "dot-backbone" + assert seen["body"]["max_output_tokens"] == 1 + + monkeypatch.setattr(H.urllib.request, "urlopen", lambda *args, **kwargs: _Response({})) + with pytest.raises(RuntimeError, match="non-empty"): + H.probe_protocol(_target(), "chat") + + +def test_qwen_chat_probe_suppresses_thinking_before_the_one_token_cap(monkeypatch): + seen = {} + + def fake_urlopen(request, timeout): + seen.update(body=json.loads(request.data)) + return _Response({"choices": [{"message": {"content": "ready"}}]}) + + monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen) + H.probe_protocol(_target(), "chat") + + assert seen["body"]["messages"] == [{"role": "user", "content": "ping /no_think"}] + + +@pytest.mark.parametrize("status", [199, 300, 302, 399, 400, 500]) +def test_probe_protocol_requires_a_2xx_http_status(status, monkeypatch): + monkeypatch.setattr( + H.urllib.request, + "urlopen", + lambda *args, **kwargs: _Response({"id": "response-1"}, status=status), + ) + with pytest.raises(RuntimeError, match=f"HTTP {status}"): + H.probe_protocol(_target(), "chat") + + +def test_unsupported_harness_and_protocol_fail_closed(): + target = _target() + with pytest.raises(ValueError, match="unsupported harness"): + H.protocol_for("not-a-harness") + with pytest.raises(ValueError, match="unsupported protocol"): + H.probe_protocol(target, "xml") + with pytest.raises(ValueError, match="unsupported harness"): + H.harbor_agent_kwargs(target, "not-a-harness") + + +def test_probe_chat_tool_round_trip_disables_thinking_and_returns_tool_result(monkeypatch): + requests = [] + + def fake_urlopen(request, timeout): + body = json.loads(request.data) + requests.append((request.full_url, timeout, body)) + if len(requests) == 1: + return _Response({ + "choices": [{"message": {"role": "assistant", "content": "", "reasoning": "hidden", + "tool_calls": [{ + "id": "call-1", "type": "function", + "function": {"name": "ingot_echo", "arguments": '{"value":"cutover-ok"}'}, + }]}}], + }) + return _Response({"choices": [{"message": {"role": "assistant", "content": "cutover-ok"}}]}) + + monkeypatch.setattr(H.urllib.request, "urlopen", fake_urlopen) + H.probe_chat_tool_round_trip(_target(), timeout=9.0) + + assert len(requests) == 2 + assert all(url == "http://host:8011/v1/chat/completions" for url, _, _ in requests) + assert all(timeout == 9.0 for _, timeout, _ in requests) + assert all(body["model"] == "dot-backbone" for _, _, body in requests) + assert all(body["chat_template_kwargs"] == {"enable_thinking": False} + for _, _, body in requests) + assert all(body["reasoning_effort"] == "none" for _, _, body in requests) + assert all(body["messages"][0]["content"].endswith("/no_think") + for _, _, body in requests) + assert requests[0][2]["tool_choice"] == "required" + assert requests[1][2]["messages"][-2] == { + "role": "assistant", "content": "", "tool_calls": [{ + "id": "call-1", "type": "function", + "function": {"name": "ingot_echo", "arguments": '{"value":"cutover-ok"}'}, + }], + } + assert requests[1][2]["messages"][-1] == { + "role": "tool", "tool_call_id": "call-1", "content": "cutover-ok", + } + assert requests[1][2]["tools"] == requests[0][2]["tools"] diff --git a/tests/test_ingot_review.py b/tests/test_ingot_review.py new file mode 100644 index 0000000..bdd3192 --- /dev/null +++ b/tests/test_ingot_review.py @@ -0,0 +1,630 @@ +"""`ingot review` — the deterministic, offline, read-only report. + +Named for the CLI, not for `ingot.optimize.review`, which is the model-graded advisory pass and a +different thing entirely (tests/test_review.py covers that one). Nothing here may reach a model, a +key, a service, or the network.""" +import json +import os +import subprocess +import sys +import unicodedata +from pathlib import Path + +import pytest + +from ingot import cli, review +from ingot.parse import parse_raw + + +def _skill(root, name, description="Merge and split PDF files.", body="Do the thing.", + **frontmatter): + """Built line by line rather than from a dedented block: an interpolated multi-line field + defeats `textwrap.dedent`, which silently leaves the delimiters indented and turns every + package into a frontmatter-missing one. That cost a green test that was asserting nothing.""" + directory = root / name + directory.mkdir(parents=True, exist_ok=True) + fields = {"name": name, "description": description, **frontmatter} + lines = ["---"] + [f"{key}: {value}" for key, value in fields.items()] + ["---", "", body, ""] + (directory / "SKILL.md").write_text("\n".join(lines), encoding="utf-8") + return directory + + +def _codes(section) -> list[str]: + return [finding["code"] for finding in section["findings"]] + + +def _all_codes(result) -> list[str]: + return [finding["code"] + for section in result["sections"].values() + for finding in section["findings"]] + + +# --- the raw diagnostic parser ------------------------------------------------------------- + +def test_raw_parser_reports_absent_frontmatter_instead_of_normalizing_it(): + """`ingot.mcp_server.registry.parse_skill` turns this into empty metadata on purpose, so the server + keeps serving. A diagnostic parser that did the same would have nothing to report.""" + raw = parse_raw("Just a body, no frontmatter.\n") + + assert raw.frontmatter is None + assert [f.code for f in raw.findings] == ["frontmatter-missing"] + + +def test_raw_parser_reports_the_yaml_error_rather_than_swallowing_it(): + raw = parse_raw("---\nname: pdf\ndescription: [unclosed\n---\n\nbody\n") + + assert raw.frontmatter is None + assert [f.code for f in raw.findings] == ["frontmatter-invalid"] + assert raw.findings[0].message + + +def test_raw_parser_reports_frontmatter_that_is_not_a_mapping(): + raw = parse_raw("---\n- one\n- two\n---\n\nbody\n") + + assert raw.frontmatter is None + assert [f.code for f in raw.findings] == ["frontmatter-not-a-mapping"] + + +def test_raw_parser_keeps_a_good_document_intact(): + raw = parse_raw("---\nname: pdf\ndescription: Merge PDFs.\n---\n\nThe body.\n") + + assert raw.findings == [] + assert raw.frontmatter == {"name": "pdf", "description": "Merge PDFs."} + assert raw.body == "The body." + + +# --- structural validity ------------------------------------------------------------------- + +def test_a_known_good_skill_is_valid(tmp_path): + directory = _skill(tmp_path, "pdf") + + result = review.review_package(directory) + + assert result["valid"] is True + assert result["errors"] == 0 + + +def test_a_missing_skill_md_is_an_error(tmp_path): + directory = tmp_path / "pdf" + directory.mkdir() + + result = review.review_package(directory) + + assert result["valid"] is False + assert "skill-md-missing" in _all_codes(result) + + +def test_invalid_frontmatter_is_an_error(tmp_path): + directory = tmp_path / "pdf" + directory.mkdir() + (directory / "SKILL.md").write_text("---\ndescription: [unclosed\n---\n\nbody\n", + encoding="utf-8") + + result = review.review_package(directory) + + assert result["valid"] is False + assert "frontmatter-invalid" in _all_codes(result) + + +def test_an_empty_description_is_an_error(tmp_path): + """The router keys on description. A skill without one is never loaded, so shipping it is a + silent no-op rather than a degraded skill.""" + directory = _skill(tmp_path, "pdf", description="") + + result = review.review_package(directory) + + assert result["valid"] is False + assert "description-empty" in _all_codes(result) + + +def test_an_invalid_slug_is_an_error(tmp_path): + directory = tmp_path / "Not_A_Slug" + directory.mkdir() + (directory / "SKILL.md").write_text( + "---\nname: Not_A_Slug\ndescription: Something.\n---\n\nbody\n", encoding="utf-8") + + result = review.review_package(directory) + + assert result["valid"] is False + assert "name-invalid" in _all_codes(result) + + +def test_a_name_that_disagrees_with_the_directory_warns_without_failing(tmp_path): + directory = tmp_path / "pdf" + directory.mkdir() + (directory / "SKILL.md").write_text( + "---\nname: docx\ndescription: Something.\n---\n\nbody\n", encoding="utf-8") + + result = review.review_package(directory) + + assert result["valid"] is True + assert "name-directory-mismatch" in _all_codes(result) + + +def test_a_dangling_file_reference_warns_without_failing(tmp_path): + """A broken link makes the skill worse, not unrepresentable, so it must not block admission.""" + directory = _skill(tmp_path, "pdf", body="See [the guide](./guide.md) for details.") + + result = review.review_package(directory) + + assert result["valid"] is True + assert "file-reference-missing" in _all_codes(result) + + +def test_a_resolvable_file_reference_is_not_reported(tmp_path): + directory = _skill(tmp_path, "pdf", body="See [the guide](./guide.md) for details.") + (directory / "guide.md").write_text("guide\n", encoding="utf-8") + + result = review.review_package(directory) + + assert "file-reference-missing" not in _all_codes(result) + + +def test_path_traversal_in_a_reference_is_an_error(tmp_path): + directory = _skill(tmp_path, "pdf", body="See [outside](../../etc/passwd).") + + result = review.review_package(directory) + + assert result["valid"] is False + assert "path-traversal" in _all_codes(result) + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows") +def test_a_symlink_escaping_the_package_is_an_error(tmp_path): + directory = _skill(tmp_path, "pdf") + outside = tmp_path / "outside.txt" + outside.write_text("secret\n", encoding="utf-8") + (directory / "link.txt").symlink_to(outside) + + result = review.review_package(directory) + + assert result["valid"] is False + assert "symlink-unsupported" in _all_codes(result) + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows") +def test_a_symlink_inside_the_package_is_an_error_too(tmp_path): + """Containment is not the question any more. Admission stages exact bytes, and a link is + neither preserved (the vault would commit a path leading out of the library) nor followed (the + artifact would quietly become a different shape than the one submitted).""" + directory = _skill(tmp_path, "pdf") + (directory / "real.txt").write_text("data\n", encoding="utf-8") + (directory / "link.txt").symlink_to(directory / "real.txt") + + result = review.review_package(directory) + + assert result["valid"] is False + assert "symlink-unsupported" in _all_codes(result) + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows") +def test_a_directory_symlink_does_not_pull_in_what_it_points_at(tmp_path): + """`rglob` follows directory symlinks. A package could otherwise absorb a whole tree from + outside itself and the review would never mention the link.""" + directory = _skill(tmp_path, "pdf") + outside = tmp_path / "outside" + outside.mkdir() + (outside / "secret.md").write_text("secret\n", encoding="utf-8") + (directory / "docs").symlink_to(outside, target_is_directory=True) + + result = review.review_package(directory) + + assert "symlink-unsupported" in _all_codes(result) + assert result["sections"]["structural"]["file_count"] == 1 + + +def test_a_non_portable_path_warns(tmp_path): + directory = _skill(tmp_path, "pdf") + (directory / "why:not.txt").write_text("data\n", encoding="utf-8") + + result = review.review_package(directory) + + assert "path-not-portable" in _all_codes(result) + + +def test_case_insensitive_path_collision_warns(tmp_path): + """Two files that differ only by case survive here and collapse into one on macOS or Windows, + which changes the package's content hash depending on who checked it out.""" + directory = _skill(tmp_path, "pdf") + (directory / "Guide.md").write_text("one\n", encoding="utf-8") + try: + (directory / "guide.md").write_text("two\n", encoding="utf-8") + except OSError: + pytest.skip("filesystem rejected the pair") + if len(list(directory.glob("*uide.md"))) < 2: + pytest.skip("case-insensitive filesystem collapsed the pair") + + result = review.review_package(directory) + + assert "path-case-collision" in _all_codes(result) + + +def test_unicode_normalization_collision_warns(tmp_path): + """The same filename in NFC and NFD is one file on macOS and two on Linux.""" + directory = _skill(tmp_path, "pdf") + composed = unicodedata.normalize("NFC", "café.md") + decomposed = unicodedata.normalize("NFD", "café.md") + (directory / composed).write_text("one\n", encoding="utf-8") + try: + (directory / decomposed).write_text("two\n", encoding="utf-8") + except OSError: + pytest.skip("filesystem rejected the pair") + if len(list(directory.glob("*.md"))) < 3: + pytest.skip("filesystem normalized the pair") + + result = review.review_package(directory) + + assert "path-unicode-collision" in _all_codes(result) + + +def test_path_collision_logic_without_a_filesystem(): + """The two collision cases above cannot be staged on a case-insensitive filesystem, which is + every macOS development machine, so they skip exactly where most of this code is written. This + covers the same logic directly -- `_path_findings` never touches the disk.""" + package = Path("/pkg") + files = [package / "SKILL.md", package / "Guide.md", package / "guide.md"] + + codes = [finding.code for finding in review._path_findings(package, files)] + + assert "path-case-collision" in codes + + +def test_unicode_collision_logic_without_a_filesystem(): + package = Path("/pkg") + files = [package / unicodedata.normalize("NFC", "café.md"), + package / unicodedata.normalize("NFD", "café.md")] + + codes = [finding.code for finding in review._path_findings(package, files)] + + assert "path-unicode-collision" in codes + + +def test_distinct_paths_do_not_collide(): + package = Path("/pkg") + files = [package / "SKILL.md", package / "guide.md", package / "notes.md"] + + assert review._path_findings(package, files) == [] + + +def test_the_report_carries_a_stable_content_revision(tmp_path): + directory = _skill(tmp_path, "pdf") + + first = review.review_package(directory) + second = review.review_package(directory) + + assert first["revision"] == second["revision"] + assert len(first["revision"]) == 64 + + +# --- supply-chain metadata ----------------------------------------------------------------- + +def test_absent_source_and_license_metadata_are_reported(tmp_path): + directory = _skill(tmp_path, "pdf") + + section = review.review_package(directory)["sections"]["supply_chain"] + + assert "source-metadata-missing" in _codes(section) + assert "license-metadata-missing" in _codes(section) + + +def test_declared_source_and_license_are_not_reported_missing(tmp_path): + directory = _skill(tmp_path, "pdf", license="Apache-2.0", source="https://example.test/repo") + + section = review.review_package(directory)["sections"]["supply_chain"] + + assert "license-metadata-missing" not in _codes(section) + assert "source-metadata-missing" not in _codes(section) + + +def test_an_executable_asset_is_reported(tmp_path): + directory = _skill(tmp_path, "pdf") + script = directory / "run.sh" + script.write_text("#!/bin/sh\necho hi\n", encoding="utf-8") + script.chmod(0o755) + + section = review.review_package(directory)["sections"]["supply_chain"] + + assert "executable-asset" in _codes(section) + + +def test_a_remote_reference_is_reported(tmp_path): + directory = _skill(tmp_path, "pdf", body="Fetch https://example.test/tool.sh and run it.") + + section = review.review_package(directory)["sections"]["supply_chain"] + + assert "remote-reference" in _codes(section) + + +def test_an_unpinned_mutable_reference_is_reported(tmp_path): + directory = _skill(tmp_path, "pdf", + body="Read https://github.com/acme/tool/blob/main/README.md first.") + + section = review.review_package(directory)["sections"]["supply_chain"] + + assert "reference-unpinned" in _codes(section) + + +def test_supply_chain_findings_never_fail_the_package(tmp_path): + """These are advisory. Nothing here claims to be a security verdict, so nothing here may + make an otherwise valid package inadmissible.""" + directory = _skill(tmp_path, "pdf", body="Fetch https://example.test/tool.sh and run it.") + script = directory / "run.sh" + script.write_text("#!/bin/sh\n", encoding="utf-8") + script.chmod(0o755) + + result = review.review_package(directory) + + assert result["valid"] is True + + +# --- collision ----------------------------------------------------------------------------- + +def test_collision_is_unmeasured_without_a_library_root(tmp_path): + directory = _skill(tmp_path, "pdf") + + section = review.review_package(directory)["sections"]["collision"] + + assert section["status"] == review.UNMEASURED + + +def test_a_library_name_collision_is_reported(tmp_path): + library = tmp_path / "library" + _skill(library, "pdf") + candidate = _skill(tmp_path / "candidate", "pdf") + + section = review.review_package(candidate, library_root=library)["sections"]["collision"] + + assert "name-collision" in _codes(section) + + +def test_no_library_collision_when_the_name_is_free(tmp_path): + library = tmp_path / "library" + _skill(library, "docx") + candidate = _skill(tmp_path / "candidate", "pdf") + + section = review.review_package(candidate, library_root=library)["sections"]["collision"] + + assert section["status"] == review.MEASURED + assert _codes(section) == [] + + +def test_an_abandoned_staging_directory_is_not_a_collision(tmp_path): + """Promotion and rollback stage a skill beside the live one as `...stage`, each + carrying a complete SKILL.md. A bare glob would report a crashed run's leftovers as a colliding + skill, so this reads the library the same way the server does.""" + library = tmp_path / "library" + _skill(library, "docx") + stage = library / ".pdf.abc123.stage" + stage.mkdir(parents=True) + (stage / "SKILL.md").write_text("---\nname: pdf\ndescription: Staged.\n---\n\nbody\n", + encoding="utf-8") + candidate = _skill(tmp_path / "candidate", "pdf") + + section = review.review_package(candidate, library_root=library)["sections"]["collision"] + + assert _codes(section) == [] + + +def test_many_remote_references_produce_one_finding(tmp_path): + """A wall of identical warnings is how a reviewer learns to stop reading them.""" + body = "\n".join(f"See https://example.test/page-{n}" for n in range(12)) + directory = _skill(tmp_path, "pdf", body=body) + + section = review.review_package(directory)["sections"]["supply_chain"] + + assert _codes(section).count("remote-reference") == 1 + assert len(section["remote_references"]) == 12 + + +def test_semantic_collision_is_reported_unmeasured_not_guessed(tmp_path): + """Description shadowing needs the embedding router, which needs a model. The deterministic + command must say so and name the command that measures it, never approximate it.""" + library = tmp_path / "library" + _skill(library, "docx") + candidate = _skill(tmp_path / "candidate", "pdf") + + section = review.review_package(candidate, library_root=library)["sections"]["collision"] + + assert section["semantic"]["status"] == review.UNMEASURED + assert "routing_health" in section["semantic"]["measure_with"] + + +# --- activation and behavioral evidence ------------------------------------------------------ + +def test_activation_is_unmeasured_without_a_routing_suite(tmp_path): + directory = _skill(tmp_path, "pdf") + + section = review.review_package(directory, evidence_root=tmp_path)["sections"]["activation"] + + assert section["status"] == review.UNMEASURED + assert section["routing_cases"] == 0 + + +def test_activation_reports_an_existing_suite_without_scoring_it(tmp_path): + """A suite existing is a fact this command can establish offline. Whether the router loads the + skill at the right time is not, so it stays UNMEASURED with the command that would answer it.""" + directory = _skill(tmp_path, "pdf") + tasks = tmp_path / "ingot" / "optimize" / "tasks" + tasks.mkdir(parents=True) + (tasks / "pdf.yaml").write_text( + "routing:\n - prompt: merge two pdfs\n - prompt: split a pdf\n", encoding="utf-8") + + section = review.review_package(directory, evidence_root=tmp_path)["sections"]["activation"] + + assert section["routing_cases"] == 2 + assert section["status"] == review.UNMEASURED + assert "score" not in section + + +def test_behavioral_evidence_is_unmeasured_when_absent(tmp_path): + directory = _skill(tmp_path, "pdf") + + section = review.review_package(directory, evidence_root=tmp_path)["sections"]["behavioral"] + + assert section["status"] == review.UNMEASURED + + +def test_existing_compatibility_evidence_is_surfaced_not_recomputed(tmp_path): + """`ingot.optimize.compat` already measures skill vs no-skill lift and writes it to runs/compat. + Review reads that file. It must never run a model to answer this.""" + directory = _skill(tmp_path, "pdf") + compat = tmp_path / "runs" / "compat" + compat.mkdir(parents=True) + (compat / "pdf.json").write_text(json.dumps({ + "skill": "pdf", "tasks": 12, "judge": "test-judge", + "models": {"openrouter/some-model": {"skill": 0.8, "baseline": 0.5, "lift": 0.3}}, + }), encoding="utf-8") + + section = review.review_package(directory, evidence_root=tmp_path)["sections"]["behavioral"] + + assert section["status"] == review.MEASURED + assert section["tasks"] == 12 + assert section["models"]["openrouter/some-model"]["lift"] == 0.3 + + +def test_review_reports_no_composite_score(tmp_path): + """A single number invites the reward-hacking this whole product exists to prevent.""" + directory = _skill(tmp_path, "pdf") + + result = review.review_package(directory) + + assert "score" not in result + assert "grade" not in result + + +# --- the command ---------------------------------------------------------------------------- + +def test_command_exits_zero_for_a_valid_package(tmp_path, capsys): + directory = _skill(tmp_path, "pdf") + + code = cli.main(["review", str(directory)]) + + assert code == 0 + assert "pdf" in capsys.readouterr().out + + +def test_command_exits_nonzero_for_an_invalid_package(tmp_path, capsys): + directory = _skill(tmp_path, "pdf", description="") + + code = cli.main(["review", str(directory)]) + + assert code != 0 + + +def test_command_exits_zero_when_only_warnings_are_present(tmp_path, capsys): + directory = _skill(tmp_path, "pdf", body="See [the guide](./guide.md).") + + code = cli.main(["review", str(directory)]) + + assert code == 0 + + +def test_command_json_is_the_versioned_payload(tmp_path, capsys): + directory = _skill(tmp_path, "pdf") + + code = cli.main(["review", str(directory), "--json"]) + + assert code == 0 + payload = json.loads(capsys.readouterr().out) + assert payload["schema_version"] == review.REVIEW_SCHEMA + assert set(payload["sections"]) == {"structural", "supply_chain", "collision", + "activation", "behavioral"} + + +def test_command_on_a_missing_path_fails_cleanly(tmp_path, capsys): + code = cli.main(["review", str(tmp_path / "nowhere")]) + + assert code != 0 + assert "nowhere" in capsys.readouterr().err + + +def test_reviewing_stays_read_only(tmp_path): + """Read-only is a promise, not a description. A command that writes while reporting is a + command that can change what it is reporting on.""" + directory = _skill(tmp_path, "pdf") + before = {path: path.stat().st_mtime_ns for path in sorted(tmp_path.rglob("*"))} + + review.review_package(directory) + + after = {path: path.stat().st_mtime_ns for path in sorted(tmp_path.rglob("*"))} + assert before == after + + +def test_review_runs_without_the_heavy_stack(tmp_path): + """The point of the milestone, enforced: no model, no key, no services, no network.""" + directory = _skill(tmp_path, "pdf") + heavy = ["fastapi", "onnxruntime", "langgraph", "langfuse", "litellm", "fastembed", "ingot.optimize"] + program = ("import sys, ingot.cli; " + f"ingot.cli.main(['review', {str(directory)!r}, '--json']); " + f"print([m for m in {heavy!r} if m in sys.modules], file=sys.stderr)") + + result = subprocess.run([sys.executable, "-c", program], capture_output=True, text=True) + + assert result.returncode == 0, result.stderr + assert result.stderr.strip().endswith("[]") + + +# --------------------------------------------------------------------------- artifact fidelity + +def test_a_binary_asset_is_reported_but_not_refused(tmp_path): + """Admission preserves it byte-for-byte, so it is a note. It is still the fact a reviewer most + needs: text can be read before approval and a compiled asset cannot.""" + package = _skill(tmp_path, "pdf") + (package / "assets").mkdir() + (package / "assets" / "diagram.png").write_bytes(b"\x89PNG\r\n\x1a\n" + b"\xff\xfe" * 16) + + result = review.review_package(package) + + assert result["valid"] is True + codes = [f["code"] for f in result["sections"]["structural"]["findings"]] + assert "binary-asset" in codes + assert result["sections"]["structural"]["binary_assets"] == ["assets/diagram.png"] + + +def test_the_finding_names_every_asset_a_reviewer_cannot_read(tmp_path): + package = _skill(tmp_path, "pdf") + for name in ("a.png", "b.pdf", "c.bin"): + (package / name).write_bytes(b"\xff\xfe\x00") + + result = review.review_package(package) + + assert result["sections"]["structural"]["binary_assets"] == ["a.png", "b.pdf", "c.bin"] + message = [f["message"] for f in result["sections"]["structural"]["findings"] + if f["code"] == "binary-asset"][0] + for name in ("a.png", "b.pdf", "c.bin"): + assert name in message + + +def test_editor_and_vcs_metadata_is_not_reported_as_an_asset(tmp_path): + """A finding that fires on `.DS_Store` is one people learn to scroll past.""" + package = _skill(tmp_path, "pdf") + (package / ".DS_Store").write_bytes(b"\x00\x00\x00\x01") + (package / ".git").mkdir() + (package / ".git" / "index").write_bytes(b"DIRC\xff") + (package / "__pycache__").mkdir() + (package / "__pycache__" / "x.cpython-312.pyc").write_bytes(b"\xff\x00") + + result = review.review_package(package) + + assert result["sections"]["structural"]["binary_assets"] == [] + assert result["valid"] is True + + +def test_the_extension_is_not_what_decides(tmp_path): + """An extension is a claim about a file; the point is to check the file. A `.md` of raw bytes + is unreadable, and a `.bin` of UTF-8 is not.""" + package = _skill(tmp_path, "pdf") + (package / "readable.bin").write_text("plain text\n", encoding="utf-8") + (package / "unreadable.md").write_bytes(b"\xff\xfe\x00\x01") + + assert review.review_package(package)["sections"]["structural"]["binary_assets"] \ + == ["unreadable.md"] + + +def test_a_package_of_only_text_reports_no_assets(tmp_path): + package = _skill(tmp_path, "pdf") + (package / "references").mkdir() + (package / "references" / "notes.md").write_text("# Notes\n") + (package / "run.sh").write_text("#!/bin/sh\necho hi\n") + + assert review.review_package(package)["sections"]["structural"]["binary_assets"] == [] diff --git a/tests/test_ingress.py b/tests/test_ingress.py new file mode 100644 index 0000000..5de50db --- /dev/null +++ b/tests/test_ingress.py @@ -0,0 +1,201 @@ +import json + +import pytest + +from ingot.mcp_server import registry +from ingot.optimize import promote as P + + +def _store(tmp_path, monkeypatch): + root = tmp_path / "skills" + root.mkdir() + monkeypatch.setenv("INGOT_LIBRARY", str(root)) + monkeypatch.setenv("INGOT_RUNS", str(tmp_path)) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root)) + return root + + +def _proposal(**overrides): + data = { + "skill": "copywriting", + "description": "Write clear conversion copy.", + "body": "# Copywriting\n\nUse evidence and preserve supplied facts.", + "files": {"references/frameworks.md": "# Frameworks\n\nPAS"}, + "frontmatter": {}, + "summary": "Add the vetted copywriting skill.", + "source": "dotfiles-claude@c658aee", + "producer": "skill-retrospective", + "caller": "improve existing intake", + "evidence": ["Package is hash-pinned.", "Catalog discovery passed."], + "pressure_scenario": "A rushed draft must preserve locked facts.", + "risk": "New instructions may route too broadly.", + "verification_status": "passed", + "verification_command": "python3 scripts/codex-skill-catalog.py list --repo .", + "verification_result": "copywriting discovered", + } + data.update(overrides) + return data + + +def test_new_skill_submission_is_visible_but_inert(tmp_path, monkeypatch): + from ingot.optimize import ingress + + root = _store(tmp_path, monkeypatch) + + result = ingress.submit_skill_create(**_proposal()) + + pending = P.load_pending("copywriting") + assert result == {"status": "quarantined", "skill": "copywriting", + "proposal_id": pending["creation"]["proposal_id"], + "promotable": True} + assert pending["kind"] == "creation" + assert pending["champion_components"] == {} + assert pending["challenger_components"] == { + "description": "Write clear conversion copy.", + "body": "# Copywriting\n\nUse evidence and preserve supplied facts.", + "frontmatter": '{"description":"Write clear conversion copy.","name":"copywriting"}', + "file:references/frameworks.md": "# Frameworks\n\nPAS", + } + assert pending["gate"]["kind"] == "new_skill_admission" + assert not (root / "copywriting").exists() + assert json.loads((tmp_path / "ingress-audit.jsonl").read_text())["action"] == "quarantine" + + +def test_creation_refuses_existing_skill_unsafe_files_and_unverified_input(tmp_path, monkeypatch): + from ingot.optimize import ingress + + root = _store(tmp_path, monkeypatch) + existing = root / "copywriting" + existing.mkdir() + registry.write_skill_md(existing / "SKILL.md", + {"name": "copywriting", "description": "Existing."}, "Body") + + with pytest.raises(ValueError, match="already exists"): + ingress.submit_skill_create(**_proposal()) + (existing / "SKILL.md").unlink() + existing.rmdir() + with pytest.raises(ValueError, match="escapes skill root"): + ingress.submit_skill_create(**_proposal(files={"../outside.md": "no"})) + with pytest.raises(ValueError, match="verification_status"): + ingress.submit_skill_create(**_proposal(verification_status="unavailable")) + + +def test_creation_promotion_is_reversible_to_absence(tmp_path, monkeypatch): + from ingot.optimize import ingress + + root = _store(tmp_path, monkeypatch) + ingress.submit_skill_create(**_proposal()) + + promoted = P._activate_approved("copywriting", P.load_pending("copywriting"), + actor="reviewer") + created = root / "copywriting" + assert "Added 'copywriting'" in promoted + assert created.is_dir() + assert "Use evidence" in (created / "SKILL.md").read_text() + assert (created / "references/frameworks.md").read_text().endswith("PAS") + assert not P.pending_path("copywriting").exists() + + removed = P._activate_rollback("copywriting", P.ABSENT_REVISION, actor="reviewer") + assert "to absence" in removed + assert not created.exists() + + created_revision = next(item["revision"] for item in P.list_revisions("copywriting") + if item["revision"] != P.ABSENT_REVISION) + restored = P._activate_rollback("copywriting", created_revision, actor="reviewer") + assert "Restored absent skill" in restored + assert created.is_dir() + assert "Use evidence" in (created / "SKILL.md").read_text() + + +def test_duplicate_creation_is_idempotent(tmp_path, monkeypatch): + from ingot.optimize import ingress + + _store(tmp_path, monkeypatch) + + first = ingress.submit_skill_create(**_proposal()) + second = ingress.submit_skill_create(**_proposal()) + + assert first["status"] == "quarantined" + assert second == {**first, "status": "duplicate"} + evidence = tmp_path / "evidence" / "copywriting" / f"creation-{first['proposal_id']}" + assert (evidence / "evidence.json").is_file() + assert len((tmp_path / "ingress-audit.jsonl").read_text().splitlines()) == 1 + + +def test_creation_binds_normalized_full_frontmatter_and_staged_bytes(tmp_path, monkeypatch): + from ingot.optimize import ingress + + root = _store(tmp_path, monkeypatch) + router = {"harnesses": ["codex"], "platforms": ["macos"], + "required_tools": ["rg"], "activation": "explicit"} + ingress.submit_skill_create(**_proposal( + description="Write clear\nconversion copy.", + frontmatter={"license": "MIT", "metadata": {"skill-router": router}}, + )) + pending = P.load_pending("copywriting") + expected = pending["evidence"]["challenger"]["revision"] + + P._activate_approved("copywriting", P.load_pending("copywriting"), actor="reviewer") + + active = registry.load_skills(root)[0] + meta, _ = registry.parse_skill((root / "copywriting" / "SKILL.md").read_text(), "copywriting") + assert active.revision == expected + assert active.description == "Write clear conversion copy." + assert active.metadata["harnesses"] == ["codex"] + assert active.metadata["platforms"] == ["macos"] + assert meta["license"] == "MIT" + + +def test_evidence_changes_are_not_aliased_as_duplicates(tmp_path, monkeypatch): + from ingot.optimize import ingress + + _store(tmp_path, monkeypatch) + first = ingress.submit_skill_create(**_proposal()) + + with pytest.raises(ValueError, match="review slot is occupied"): + ingress.submit_skill_create(**_proposal(risk="Different material risk.")) + assert P.load_pending("copywriting")["creation"]["proposal_id"] == first["proposal_id"] + + +def test_audit_failure_does_not_turn_a_queued_skill_into_a_false_refusal(tmp_path, monkeypatch, + caplog): + from ingot.optimize import ingress + + _store(tmp_path, monkeypatch) + monkeypatch.setattr(ingress, "audit_file", lambda: tmp_path / "missing" / "audit.jsonl") + monkeypatch.setattr(ingress, "_audit", lambda record: (_ for _ in ()).throw(OSError("full"))) + + result = ingress.submit_skill_create(**_proposal()) + + assert result["status"] == "quarantined" + assert P.load_pending("copywriting") is not None + assert "audit write failed" in caplog.text + + +@pytest.mark.parametrize("path", ["C:/x.md", "CON.md", "AUX/file.md", "bad\\name.md"]) +def test_creation_rejects_nonportable_component_paths(tmp_path, monkeypatch, path): + from ingot.optimize import ingress + + _store(tmp_path, monkeypatch) + with pytest.raises(ValueError, match="portable POSIX path"): + ingress.submit_skill_create(**_proposal(files={path: "content"})) + + +def test_creation_bounds_evidence_envelope(tmp_path, monkeypatch): + from ingot.optimize import ingress + + _store(tmp_path, monkeypatch) + with pytest.raises(ValueError, match="2 to 12"): + ingress.submit_skill_create(**_proposal(evidence=[str(i) for i in range(13)])) + + +def test_creation_rejects_unsafe_router_globs_and_unicode_aliases(tmp_path, monkeypatch): + from ingot.optimize import ingress + + _store(tmp_path, monkeypatch) + frontmatter = {"metadata": {"skill-router": { + "scopes": ["project"], "path_patterns": ["/tmp/*"]}}} + with pytest.raises(ValueError, match="relative POSIX globs"): + ingress.submit_skill_create(**_proposal(frontmatter=frontmatter)) + with pytest.raises(ValueError, match="NFC-normalized"): + ingress.submit_skill_create(**_proposal(files={"references/e\u0301.md": "content"})) diff --git a/tests/test_judge.py b/tests/test_judge.py index 690e2e6..bdd2e1a 100644 --- a/tests/test_judge.py +++ b/tests/test_judge.py @@ -1,8 +1,12 @@ """Unit tests for the LLM judge's pure parsing/aggregation logic (LLM calls mocked).""" +import json + import pytest -from optimize import judge as J -from optimize.judge import DIMENSIONS, _extract_json, failed_dimensions +from ingot.optimize import agy_judge as A +from ingot.optimize import judge as J +from ingot.optimize.judge import (DEFAULT_CHECKLIST, DIMENSIONS, _extract_json, _weighted, + failed_dimensions) class _FakeMsg: @@ -17,6 +21,77 @@ def _mock_single_judge(monkeypatch, content): monkeypatch.setattr(J, "_get_llm", lambda model: type("L", (), {"invoke": lambda self, m: _FakeMsg(content)})()) +def _mock_judges(monkeypatch, contents): + """One scripted response per model, in order, for ensemble tests.""" + monkeypatch.setattr(J, "MODELS", [f"mock-{i}" for i in range(len(contents))]) + by_model = {f"mock-{i}": c for i, c in enumerate(contents)} + monkeypatch.setattr(J, "_get_llm", lambda model: type( + "L", (), {"invoke": lambda self, m, _c=by_model[model]: _FakeMsg(_c)})()) + + +def _items(**verdicts) -> str: + return json.dumps({"items": {k: {"verdict": v, "note": "n"} for k, v in verdicts.items()}, + "feedback": "f"}) + + +def _all(verdict: str) -> str: + return _items(**{i["id"]: verdict for i in DEFAULT_CHECKLIST}) + + +def _agy_grade(verdict: str = "pass") -> dict: + return { + "items": { + item["id"]: {"verdict": verdict, "note": ""} + for item in DEFAULT_CHECKLIST + }, + "feedback": "agy feedback", + } + + +# --- backend selection ----------------------------------------------------------------------- + +def test_agy_backend_uses_one_subscription_grade_without_openrouter(monkeypatch): + monkeypatch.setenv("JUDGE_BACKEND", "agy") + monkeypatch.setenv("OPENROUTER_API_KEY", "must-not-be-used") + monkeypatch.setattr(J, "MODELS", ["openrouter/one", "openrouter/two"]) + monkeypatch.setattr(J, "_get_llm", lambda *_args: pytest.fail("OpenRouter fallback ran")) + usage = {"input_tokens": 11, "output_tokens": 7, "total_tokens": 18} + calls = [] + + def invoke(prompt, checklist): + calls.append((prompt, checklist)) + return _agy_grade(), usage + + ledger = [] + monkeypatch.setattr(A, "invoke", invoke) + monkeypatch.setattr( + J.usage_ledger, + "add", + lambda role, observed, **metadata: ledger.append((role, observed, metadata)), + ) + + result = J.judge("task", answer="answer") + + assert result["score"] == 1.0 + assert result["feedback"] == "agy feedback" + assert len(calls) == 1 + assert ledger == [("judge", usage, {"billing_mode": "subscription"})] + + +def test_agy_backend_error_propagates_without_openrouter_fallback(monkeypatch): + monkeypatch.setenv("JUDGE_BACKEND", "agy") + monkeypatch.setenv("OPENROUTER_PROVIDERS", "fallback-provider") + monkeypatch.setattr(J, "_get_llm", lambda *_args: pytest.fail("OpenRouter fallback ran")) + monkeypatch.setattr( + A, + "invoke", + lambda *_args: (_ for _ in ()).throw(A.AgyJudgeError("agy unavailable")), + ) + + with pytest.raises(A.AgyJudgeError, match="agy unavailable"): + J.judge("task", answer="answer") + + # --- _extract_json: robust to prose / fences / stray braces ----------------------------------- def test_extract_json_plain(): @@ -41,30 +116,151 @@ def test_extract_json_returns_empty_when_no_score_object(text): assert _extract_json(text) == {} -# --- judge(): score clamping, unparseable fallback, dimension defaulting ----------------------- +def test_extract_json_finds_the_requested_key(): + """The judge asks for `items`; the score key it used to look for no longer appears.""" + assert _extract_json('{"items": {"a": "pass"}}', "items")["items"] == {"a": "pass"} + -def test_judge_clamps_score_above_one(monkeypatch): - _mock_single_judge(monkeypatch, '{"score": 1.7, "feedback": "great", "dimensions": {}}') - assert J.judge("t", "r", "a")["score"] == 1.0 +# --- the score is derived, never taken from the model ----------------------------------------- +def test_score_is_the_weighted_mean_of_the_checklist(monkeypatch): + """DEFAULT_CHECKLIST weights are 3/2/2/1. Failing only `efficiency` (weight 1) must cost + exactly 1/8 of the score, not whatever the model felt like reporting.""" + _mock_single_judge(monkeypatch, _items(correctness="pass", completeness="pass", + instruction_following="pass", efficiency="fail")) + assert J.judge("t", "r", "a")["score"] == pytest.approx(7 / 8) -def test_judge_clamps_negative_score(monkeypatch): - _mock_single_judge(monkeypatch, '{"score": -0.4, "feedback": "bad", "dimensions": {}}') + +def test_a_model_supplied_score_is_ignored(monkeypatch): + """The old contract let the judge name its own number. A model that still emits one must not + be able to override the checklist -- that is the entire point of grading against items.""" + payload = json.loads(_all("fail")) + payload["score"] = 1.0 + _mock_single_judge(monkeypatch, json.dumps(payload)) assert J.judge("t", "r", "a")["score"] == 0.0 -def test_judge_defaults_missing_dimensions_to_pass(monkeypatch): - _mock_single_judge(monkeypatch, '{"score": 0.5, "dimensions": {"correctness": "wrong API"}}') +def test_partial_verdicts_score_half(monkeypatch): + _mock_single_judge(monkeypatch, _all("partial")) + assert J.judge("t", "r", "a")["score"] == pytest.approx(0.5) + + +def test_score_cannot_leave_zero_to_one(monkeypatch): + """Clamping used to be needed because the model picked the number. A weighted mean of values + in [0,1] cannot leave the range, so the property holds by construction.""" + for verdict in ("pass", "partial", "fail"): + _mock_single_judge(monkeypatch, _all(verdict)) + assert 0.0 <= J.judge("t", "r", "a")["score"] <= 1.0 + + +# --- items the judge did not answer, or answered badly, are not passes ------------------------ + +def test_an_ungraded_item_is_not_a_pass(monkeypatch): + """Omitting an item must cost its weight. Defaulting it to pass would let a lazy judge score + 1.0 by answering one item.""" + _mock_single_judge(monkeypatch, _items(correctness="pass")) r = J.judge("t", "r", "a") - assert failed_dimensions(r["dimensions"]) == ["correctness"] # only the one provided fails - assert set(r["dimensions"]) == set(DIMENSIONS) # the rest are filled in as pass + assert r["score"] == pytest.approx(3 / 8) + assert r["checklist"]["efficiency"]["note"] == "not graded" -def test_judge_unparseable_output_scores_zero(monkeypatch): +def test_an_unrecognized_verdict_fails_rather_than_passes(monkeypatch): + _mock_single_judge(monkeypatch, _items(correctness="excellent", completeness="pass", + instruction_following="pass", efficiency="pass")) + r = J.judge("t", "r", "a") + assert r["score"] == pytest.approx(5 / 8) # correctness carries weight 3 of 8 + + +def test_an_unrecognized_verdict_without_a_note_says_it_was_ungraded(monkeypatch): + """The model's own note is kept when it wrote one; the fallback only fills a silent item so + the reviewer never sees a bare zero with no reason.""" + _mock_single_judge(monkeypatch, json.dumps({"items": {"correctness": "excellent"}, + "feedback": "f"})) + assert "ungraded" in J.judge("t", "r", "a")["checklist"]["correctness"]["note"] + + +def test_unparseable_output_scores_zero_without_blaming_the_skill(monkeypatch): _mock_single_judge(monkeypatch, "the model rambled and produced no JSON at all") r = J.judge("t", "r", "a") assert r["score"] == 0.0 and "unparseable" in r["feedback"] - assert failed_dimensions(r["dimensions"]) == [] # a parse failure isn't a skill failure + assert failed_dimensions(r["dimensions"]) == [] # a parse failure isn't a skill failure + + +# --- custom checklists (the Tessl-style graded rubric) ---------------------------------------- + +CUSTOM = [ + {"id": "cites_source", "dimension": "correctness", "weight": 4, + "criterion": "Every factual claim names its source."}, + {"id": "states_tradeoff", "dimension": "completeness", "weight": 1, + "criterion": "It names at least one tradeoff."}, +] + + +def test_a_task_checklist_replaces_the_default(monkeypatch): + _mock_single_judge(monkeypatch, _items(cites_source="pass", states_tradeoff="fail")) + r = J.judge("t", "r", "a", checklist=CUSTOM) + assert r["score"] == pytest.approx(4 / 5) + assert set(r["checklist"]) == {"cites_source", "states_tradeoff"} + + +def test_a_malformed_checklist_falls_back_to_the_default(monkeypatch): + """Items without an id or criterion cannot be graded; an empty result must not mean 'no checks + ran, therefore full marks'.""" + _mock_single_judge(monkeypatch, _all("pass")) + r = J.judge("t", "r", "a", checklist=[{"weight": 9}, {"id": "x"}]) + assert set(r["checklist"]) == {i["id"] for i in DEFAULT_CHECKLIST} + + +def test_custom_items_map_onto_the_reported_dimensions(monkeypatch): + """mine.py and the candidate search consume `dimensions`, so a custom rubric still has to + report through them.""" + _mock_single_judge(monkeypatch, _items(cites_source="fail", states_tradeoff="pass")) + r = J.judge("t", "r", "a", checklist=CUSTOM) + assert failed_dimensions(r["dimensions"]) == ["correctness"] + + +# --- resolution: the reason for the whole change ---------------------------------------------- + +def test_the_checklist_resolves_finer_than_a_holistic_ladder(): + """A judge naming one number emits a coarse ladder (0.9 / 0.95 / 1.0), so a mean can move a + whole rung because two tasks crossed a boundary. Independent weighted items give many more + reachable values, which is what makes a small real effect distinguishable from noise.""" + weights = [i["weight"] for i in DEFAULT_CHECKLIST] + reachable = set() + for bits in range(3 ** len(weights)): + values, b = [], bits + for _ in weights: + values.append([0.0, 0.5, 1.0][b % 3]); b //= 3 + graded = {i["id"]: {"value": v} for i, v in zip(DEFAULT_CHECKLIST, values)} + reachable.add(round(_weighted(graded, DEFAULT_CHECKLIST), 6)) + assert len(reachable) >= 17 # vs the 3 rungs a holistic judge actually used + + +# --- ensembles average per item, not per answer ----------------------------------------------- + +def test_ensemble_averages_each_item_before_weighting(monkeypatch): + """Averaging item values is smoother than averaging whole-answer scores: two judges splitting + on one item move the result by half that item's weight, not by half the answer.""" + _mock_judges(monkeypatch, [_all("pass"), + _items(correctness="pass", completeness="pass", + instruction_following="pass", efficiency="fail")]) + r = J.judge("t", "r", "a") + assert r["checklist"]["efficiency"]["value"] == pytest.approx(0.5) + assert r["score"] == pytest.approx(1 - 0.5 * (1 / 8)) + + +def test_a_minority_failure_does_not_fail_the_dimension(monkeypatch): + """One judge of three flagging an item leaves its value at 2/3, above the failure threshold, + so the dimension still reads as a pass.""" + _mock_judges(monkeypatch, [_all("pass"), _all("pass"), + _items(correctness="fail", completeness="pass", + instruction_following="pass", efficiency="pass")]) + assert failed_dimensions(J.judge("t", "r", "a")["dimensions"]) == [] + + +def test_one_unparseable_judge_does_not_sink_the_ensemble(monkeypatch): + _mock_judges(monkeypatch, [_all("pass"), "no json here"]) + assert J.judge("t", "r", "a")["score"] == pytest.approx(1.0) # --- failed_dimensions: case / synonyms ------------------------------------------------------ diff --git a/tests/test_lite_mode.py b/tests/test_lite_mode.py index 5b373be..751783b 100644 --- a/tests/test_lite_mode.py +++ b/tests/test_lite_mode.py @@ -1,8 +1,8 @@ """Lite mode: the Langfuse-free A/B variant runner and the cost ledger / spend cap.""" import pytest -from optimize import ab as ab_mod -from optimize import usage as usage_ledger +from ingot.optimize import ab as ab_mod +from ingot.optimize import usage as usage_ledger def _fake_run_task(answers: dict): @@ -17,7 +17,7 @@ def test_local_variant_matches_run_variant_shape(monkeypatch): tasks = [{"task": "a", "rubric": "ra"}, {"task": "b", "rubric": "rb"}] monkeypatch.setattr(ab_mod, "run_task", _fake_run_task({"a": "ans-a", "b": "ans-b"})) monkeypatch.setattr(ab_mod, "judge", - lambda t, r, ans, check=None, deliverable=None: + lambda t, r, ans, check=None, deliverable=None, checklist=None: {"score": 0.9 if t == "a" else 0.4, "feedback": "f", "dimensions": {}}) scores, usages, behaviors, answers = ab_mod._run_variant_local(agent=None, tasks=tasks) assert scores == [0.9, 0.4] # aligned to task order @@ -31,7 +31,7 @@ def test_local_variant_scores_failed_rollouts_zero(monkeypatch): monkeypatch.setattr(ab_mod, "run_task", _fake_run_task({"a": RuntimeError("provider down"), "b": "ans-b"})) monkeypatch.setattr(ab_mod, "judge", - lambda t, r, ans, check=None, deliverable=None: + lambda t, r, ans, check=None, deliverable=None, checklist=None: {"score": 1.0, "feedback": "f", "dimensions": {}}) scores, usages, behaviors, answers = ab_mod._run_variant_local(agent=None, tasks=tasks) assert scores == [0.0, 1.0] # failure defaults to 0, like _run_variant @@ -56,6 +56,47 @@ def test_estimated_cost_uses_role_model_prices(monkeypatch): usage_ledger.reset() +def test_compat_spend_is_priced_per_model_and_reaches_the_cap(monkeypatch): + """The compatibility sweep changes the serving model on purpose, so it bills to `compat:` + rather than one bucket. Under a single "compat" role no price ever matched, the sweep counted + as $0 — it reported $0.04 against $1.42 actually spent — and MAX_RUN_USD, which enforces on the + same figure, could not see the most expensive role in the run.""" + usage_ledger.reset() + monkeypatch.delenv("MAX_RUN_USD", raising=False) + monkeypatch.delenv("BASE_URL", raising=False) + monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False) + monkeypatch.setenv("JUDGE_MODEL", "m/judge") + monkeypatch.setattr(usage_ledger, "_PRICES", + {"m/dear": (0.0, 1e-5), "m/cheap": (0.0, 1e-7), "m/judge": (0.0, 0.0)}) + usage_ledger.add("compat:m/dear", {"input_tokens": 0, "output_tokens": 100_000}) + usage_ledger.add("compat:m/cheap", {"input_tokens": 0, "output_tokens": 100_000}) + usage_ledger.add("judge", {"input_tokens": 500_000, "output_tokens": 0}) + + assert usage_ledger.estimated_cost() == pytest.approx(1.0 + 0.01) + assert usage_ledger.unpriced_roles() == [] + assert "NOT in that estimate" not in usage_ledger.format_report() + usage_ledger.reset() + + +def test_a_role_with_no_price_is_named_instead_of_counting_as_zero(monkeypatch): + """An unpriced role contributes $0, which is right for a local endpoint and dangerously wrong + for a slug we simply failed to resolve. Either way the report has to say which roles the + number excludes, or the next silently-free role hides the same way this one did.""" + usage_ledger.reset() + monkeypatch.delenv("MAX_RUN_USD", raising=False) + monkeypatch.delenv("BASE_URL", raising=False) + monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False) + monkeypatch.setenv("JUDGE_MODEL", "m/judge") + monkeypatch.setattr(usage_ledger, "_PRICES", {"m/judge": (1e-6, 1e-6)}) + usage_ledger.add("judge", {"input_tokens": 1_000_000, "output_tokens": 0}) + usage_ledger.add("compat:m/unknown", {"input_tokens": 0, "output_tokens": 9_000_000}) + + assert usage_ledger.unpriced_roles() == ["compat:m/unknown"] + report = usage_ledger.format_report() + assert "NOT in that estimate: compat:m/unknown" in report and "m/unknown" in report + usage_ledger.reset() + + def test_cost_is_none_on_local_endpoints(monkeypatch): usage_ledger.reset() monkeypatch.setenv("BASE_URL", "http://172.17.0.1:11434/v1") diff --git a/tests/test_local_traces.py b/tests/test_local_traces.py new file mode 100644 index 0000000..3cd401e --- /dev/null +++ b/tests/test_local_traces.py @@ -0,0 +1,394 @@ +import json +import os +from pathlib import Path + +import pytest + +from ingot.optimize import local_traces + + +def _write_jsonl(path: Path, records: list[dict]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("".join(json.dumps(record) + "\n" for record in records)) + + +def test_codex_parser_ignores_injected_context_and_keeps_completed_human_turn(tmp_path): + session = tmp_path / "rollout.jsonl" + records = [ + {"timestamp": "2026-07-28T10:00:00Z", "type": "session_meta", "payload": { + "id": "codex-session", "cwd": "/work/project", "thread_source": "user", + }}, + {"timestamp": "2026-07-28T10:00:01Z", "type": "response_item", "payload": { + "type": "message", "role": "user", "content": [ + {"type": "input_text", "text": "# AGENTS.md instructions\ninternal"}, + {"type": "input_text", "text": "internal"}, + ], + }}, + {"timestamp": "2026-07-28T10:01:00Z", "type": "response_item", "payload": { + "type": "message", "role": "user", + "content": [{"type": "input_text", "text": "Build the landing page."}], + }}, + {"timestamp": "2026-07-28T10:01:00Z", "type": "response_item", "payload": { + "type": "message", "role": "user", + "content": [{"type": "input_text", + "text": "synthetic status"}], + }}, + {"timestamp": "2026-07-28T10:01:01Z", "type": "event_msg", "payload": { + "type": "task_started", "turn_id": "turn-1", "started_at": 1000, + }}, + {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": { + "type": "function_call", "name": "exec_command", "call_id": "call-1", + "arguments": json.dumps({ + "cmd": "sed -n '1,220p' /Users/example/.agents/skills/frontend-design/SKILL.md", + }), + }}, + {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": { + "type": "function_call_output", "call_id": "call-1", + "output": json.dumps({"exit_code": 1, "output": "failed"}), + }}, + {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": { + "type": "function_call", "name": "mcp__ingot__route_and_load", + "call_id": "route-1", "arguments": json.dumps({"task": "Build the landing page."}), + }}, + {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": { + "type": "function_call_output", "call_id": "route-1", + "output": json.dumps({"match": "web-design-flow", "revision": "83a75cf1", + "skill_body": "private"}), + }}, + {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": { + "type": "function_call", "name": "route_and_load", "arguments": "{}", + }}, + {"timestamp": "2026-07-28T10:01:02Z", "type": "response_item", "payload": { + "type": "function_call_output", + "output": json.dumps({"match": "wrong-skill", "revision": "deadbeef"}), + }}, + {"timestamp": "2026-07-28T10:01:03Z", "type": "event_msg", "payload": { + "type": "token_count", "info": {"last_token_usage": { + "input_tokens": 120, "output_tokens": 30, + "cached_input_tokens": 20, "reasoning_output_tokens": 4, + }}, + }}, + {"timestamp": "2026-07-28T10:01:04Z", "type": "event_msg", "payload": { + "type": "task_complete", "turn_id": "turn-1", "duration_ms": 3000, + "completed_at": 4000, "last_agent_message": "Landing page shipped.", + }}, + {"timestamp": "2026-07-28T10:02:00Z", "type": "response_item", "payload": { + "type": "message", "role": "user", + "content": [{"type": "input_text", "text": "Cancel this request."}], + }}, + {"timestamp": "2026-07-28T10:02:01Z", "type": "event_msg", "payload": { + "type": "task_started", "turn_id": "turn-2", "started_at": 5000, + }}, + {"timestamp": "2026-07-28T10:02:02Z", "type": "event_msg", "payload": { + "type": "turn_aborted", "turn_id": "turn-2", "reason": "interrupted", + }}, + ] + _write_jsonl(session, records) + + traces = local_traces.parse_codex_session(session) + + assert len(traces) == 1 + assert traces[0]["task"] == "Build the landing page." + assert traces[0]["answer"] == "Landing page shipped." + assert traces[0]["harness"] == "codex" + assert traces[0]["session_id"] == "codex-session" + assert traces[0]["turn_id"] == "turn-1" + assert traces[0]["cwd"] == "/work/project" + assert traces[0]["skills"] == [ + {"name": "frontend-design", "revision": None}, + {"name": "web-design-flow", "revision": "83a75cf1"}, + ] + assert traces[0]["tags"] == [ + "skill:frontend-design", "skill:web-design-flow", + "revision=web-design-flow@83a75cf1", + ] + assert traces[0]["usage"] == { + "input_tokens": 120, "output_tokens": 30, + "cached_input_tokens": 20, "reasoning_output_tokens": 4, + } + assert traces[0]["duration_ms"] == 3000 + assert traces[0]["tool_errors"] == 1 + + +def test_claude_parser_uses_final_answer_and_skill_tool_without_tool_payloads(tmp_path): + session = tmp_path / "claude.jsonl" + records = [ + {"timestamp": "2026-07-28T11:00:00Z", "type": "user", "sessionId": "claude-session", + "cwd": "/work/project", "isSidechain": False, "userType": "external", + "promptSource": "typed", "origin": "terminal", + "message": {"role": "user", "content": "Review this interface."}}, + {"timestamp": "2026-07-28T11:00:01Z", "type": "assistant", + "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False, + "message": {"role": "assistant", "stop_reason": "tool_use", + "usage": {"input_tokens": 80, "output_tokens": 10, + "cache_read_input_tokens": 5}, + "content": [{"type": "tool_use", "id": "tool-1", "name": "Skill", + "input": {"skill": "saas-interface-review", "args": ""}}]}}, + {"timestamp": "2026-07-28T11:00:02Z", "type": "user", + "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False, + "message": {"role": "user", "content": [ + {"type": "tool_result", "tool_use_id": "tool-1", + "content": "private skill instructions that must not enter the trace"}, + ]}}, + {"timestamp": "2026-07-28T11:00:02Z", "type": "user", + "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False, + "isMeta": True, "sourceToolUseID": "tool-1", + "message": {"role": "user", "content": [ + {"type": "text", "text": "Base directory for this skill: /private/path"}, + ]}}, + {"timestamp": "2026-07-28T11:00:02Z", "type": "user", + "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False, + "promptSource": "queued", + "message": {"role": "user", "content": [ + {"type": "text", "text": "synthetic status"}, + ]}}, + {"timestamp": "2026-07-28T11:00:03Z", "type": "assistant", + "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False, + "message": {"role": "assistant", "stop_reason": "end_turn", "usage": {}, + "content": [{"type": "thinking", "thinking": "private reasoning"}]}}, + {"timestamp": "2026-07-28T11:00:03Z", "type": "assistant", + "sessionId": "claude-session", "cwd": "/work/project", "isSidechain": False, + "message": {"role": "assistant", "stop_reason": "end_turn", + "usage": {"input_tokens": 30, "output_tokens": 20, + "cache_read_input_tokens": 2}, + "content": [{"type": "text", "text": "The hierarchy needs work."}]}}, + ] + _write_jsonl(session, records) + + traces = local_traces.parse_claude_session(session) + + assert len(traces) == 1 + assert traces[0]["task"] == "Review this interface." + assert traces[0]["answer"] == "The hierarchy needs work." + assert traces[0]["skills"] == [{"name": "saas-interface-review", "revision": None}] + assert traces[0]["tags"] == ["skill:saas-interface-review"] + assert traces[0]["usage"] == { + "input_tokens": 110, "output_tokens": 30, "cached_input_tokens": 7, + } + assert "private skill instructions" not in json.dumps(traces[0]) + + +def test_claude_parser_pins_ingot_route_revision_from_matching_tool_result(tmp_path): + session = tmp_path / "claude-route.jsonl" + records = [ + {"timestamp": "2026-07-28T12:00:00Z", "type": "user", "sessionId": "s1", + "cwd": "/work", "isSidechain": False, "userType": "external", + "promptSource": "typed", "origin": "terminal", + "message": {"role": "user", "content": "Merge the PDFs."}}, + {"timestamp": "2026-07-28T12:00:01Z", "type": "assistant", "sessionId": "s1", + "cwd": "/work", "isSidechain": False, "message": { + "role": "assistant", "stop_reason": "tool_use", "usage": {}, + "content": [{"type": "tool_use", "id": "route-1", + "name": "mcp__ingot__route_and_load", + "input": {"task": "Merge the PDFs.", "harness": "claude", "cwd": "/work"}}], + }}, + {"timestamp": "2026-07-28T12:00:02Z", "type": "user", "sessionId": "s1", + "cwd": "/work", "isSidechain": False, "userType": "external", "message": { + "role": "user", "content": [{"type": "tool_result", "tool_use_id": "route-1", + "content": json.dumps({"match": "pdf", "related_match": None, + "revision": "83a75cf1", "skill_body": "private"})}], + }}, + {"timestamp": "2026-07-28T12:00:03Z", "type": "assistant", "sessionId": "s1", + "cwd": "/work", "isSidechain": False, "message": { + "role": "assistant", "stop_reason": "end_turn", "usage": {}, + "content": [{"type": "text", "text": "Merged."}], + }}, + ] + _write_jsonl(session, records) + + traces = local_traces.parse_claude_session(session) + + assert traces[0]["skills"] == [{"name": "pdf", "revision": "83a75cf1"}] + assert traces[0]["tags"] == ["skill:pdf", "revision=pdf@83a75cf1"] + assert "skill_body" not in json.dumps(traces[0]) + + +def test_route_identity_bounds_nested_envelopes_and_revision_content(): + assert local_traces._route_identity({ + "related_match": "pdf", "revision": "83a75cf1", + }) == ("pdf", "83a75cf1") + assert local_traces._route_identity({ + "match": "pdf", "revision": {"secret": "must not become a tag"}, + }) == ("pdf", None) + nested = {"match": "pdf", "revision": "83a75cf1"} + for _ in range(local_traces._MAX_ROUTE_DEPTH + 1): + nested = {"result": nested} + assert local_traces._route_identity(nested) is None + assert local_traces._route_identity("x" * (local_traces._MAX_ROUTE_ENVELOPE + 1)) is None + assert local_traces._route_identity({ + "content": [{"type": "text", "text": json.dumps({"match": "not-an-envelope"})}], + }) is None + assert local_traces._route_identity({"match": "pdf", "score": 0.99}) is None + wide = [{"type": "text", "text": "{}"}] * local_traces._MAX_ROUTE_NODES + wide.append({"type": "text", "text": json.dumps({ + "match": "pdf", "revision": "83a75cf1", + })}) + assert local_traces._route_identity(wide) is None + assert local_traces._tool_failed({"exit_code": "0"}) is False + assert local_traces._tool_failed({"exit_code": "2"}) is True + assert local_traces._tool_failed({"error": "documented payload field"}) is False + + +def test_scan_writes_deduplicated_store_and_summary_without_answers(tmp_path): + codex = tmp_path / "codex" + claude = tmp_path / "claude" + output = tmp_path / "runs" / "local_traces.json" + codex_record = [ + {"timestamp": "2026-07-28T10:00:00Z", "type": "session_meta", "payload": { + "id": "s", "cwd": "/work", "thread_source": "user", + }}, + {"timestamp": "2026-07-28T10:00:01Z", "type": "response_item", "payload": { + "type": "message", "role": "user", + "content": [{"type": "input_text", "text": "Task"}], + }}, + {"timestamp": "2026-07-28T10:00:02Z", "type": "event_msg", "payload": { + "type": "task_started", "turn_id": "t", + }}, + {"timestamp": "2026-07-28T10:00:03Z", "type": "event_msg", "payload": { + "type": "task_complete", "turn_id": "t", "last_agent_message": "Secret answer", + }}, + ] + _write_jsonl(codex / "one.jsonl", codex_record) + _write_jsonl(codex / "copy.jsonl", codex_record) + + result = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output) + saved = json.loads(output.read_text()) + summary = local_traces.store_summary(output) + + assert result["traces"] == 1 + assert saved["schema_version"] == "ingot/local-traces/v1" + assert len(saved["traces"]) == 1 + assert summary["configured"] is True + assert summary["total"] == 1 + assert summary["harnesses"] == {"codex": 1} + assert "task" not in summary["recent"][0] + assert local_traces.store_summary(output, include_tasks=True)["recent"][0]["task"] == "Task" + assert "answer" not in summary["recent"][0] + assert "Secret answer" not in json.dumps(summary) + + +def test_scan_reuses_unchanged_files_and_reparses_changed_files(tmp_path, monkeypatch): + codex = tmp_path / "codex" + claude = tmp_path / "claude" + output = tmp_path / "local_traces.json" + session = codex / "one.jsonl" + _write_jsonl(session, [ + {"timestamp": "2026-07-28T10:00:00Z", "type": "session_meta", + "payload": {"id": "s", "cwd": "/work/project", "thread_source": "user"}}, + {"timestamp": "2026-07-28T10:00:01Z", "type": "response_item", + "payload": {"type": "message", "role": "user", + "content": [{"type": "input_text", "text": "Task"}]}}, + {"timestamp": "2026-07-28T10:00:02Z", "type": "event_msg", + "payload": {"type": "task_started", "turn_id": "t"}}, + {"timestamp": "2026-07-28T10:00:03Z", "type": "event_msg", + "payload": {"type": "task_complete", "turn_id": "t", + "last_agent_message": "Answer"}}, + ]) + first = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output) + assert first["parsed_files"] == 1 and first["reused_files"] == 0 + + monkeypatch.setattr( + local_traces, "parse_codex_session", + lambda _path: pytest.fail("unchanged transcript must come from the cursor"), + ) + second = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output) + assert second["parsed_files"] == 0 and second["reused_files"] == 1 + assert second["traces"] == 1 + + calls = [] + monkeypatch.setattr( + local_traces, "parse_codex_session", + lambda path: calls.append(path) or [], + ) + with session.open("a") as handle: + handle.write(" \n") + os.utime(session, None) + third = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output) + assert calls == [session] + assert third["parsed_files"] == 1 and third["reused_files"] == 0 + + calls.clear() + fourth = local_traces.scan( + codex_dir=codex, claude_dir=claude, output=output, force=True) + assert calls == [session] + assert fourth["parsed_files"] == 1 and fourth["reused_files"] == 0 + + +def test_scan_rebuilds_malformed_prior_cursor_instead_of_crashing(tmp_path): + codex = tmp_path / "codex" + claude = tmp_path / "claude" + output = tmp_path / "local_traces.json" + _write_jsonl(codex / "one.jsonl", []) + output.write_text(json.dumps({ + "schema_version": local_traces.SCHEMA, + "parser_version": local_traces.PARSER_VERSION, + "filters": {"projects": [], "since": "", "until": ""}, + "source_index": {"codex": {}, "claude": {}}, + "traces": [{}], + })) + + result = local_traces.scan(codex_dir=codex, claude_dir=claude, output=output) + + assert result["parsed_files"] == 1 + assert json.loads(output.read_text())["traces"] == [] + + +def test_scan_and_summary_filter_by_project_harness_and_date(tmp_path): + store = tmp_path / "local_traces.json" + payload = { + "schema_version": local_traces.SCHEMA, + "generated_at": 1, + "traces": [ + {"id": "one", "timestamp": "2026-07-27T10:00:00Z", "project": "alpha", + "harness": "codex", "task": "one", "answer": "a", "skills": [], "usage": {}}, + {"id": "two", "timestamp": "2026-07-28T10:00:00Z", "project": "beta", + "harness": "claude", "task": "two", "answer": "b", "skills": [], "usage": {}}, + {"id": "three", "timestamp": "2026-07-29T10:00:00Z", "project": "alpha", + "harness": "claude", "task": "three", "answer": "c", "skills": [], "usage": {}}, + ], + } + store.write_text(json.dumps(payload)) + + summary = local_traces.store_summary( + store, project="alpha", harness="claude", since="2026-07-28", until="2026-07-29", + include_tasks=True) + + assert summary["total"] == 1 + assert summary["recent"][0]["task"] == "three" + assert summary["available_total"] == 3 + assert summary["available_projects"] == {"alpha": 2, "beta": 1} + assert summary["available_harnesses"] == {"claude": 2, "codex": 1} + with pytest.raises(ValueError, match="ISO date"): + local_traces.store_summary(store, since="yesterday") + + store.write_text("{broken") + assert local_traces.store_summary(store)["status"] == "unreadable" + + +def test_store_summary_filters_turns_by_observed_skill(tmp_path): + """The traces page lists which skills were observed but could not narrow to one, so 'where was + build-loop actually used' meant reading 4,639 turns by eye.""" + store = tmp_path / "store.json" + store.write_text(json.dumps({ + "schema_version": local_traces.SCHEMA, + "generated_at": "2026-08-07T00:00:00Z", + "traces": [ + {"id": "a", "timestamp": "2026-08-01T10:00:00Z", "project": "p", "harness": "claude", + "task": "a", "answer": "a", "usage": {}, + "skills": [{"name": "build-loop"}, {"name": "sota-check"}]}, + {"id": "b", "timestamp": "2026-08-02T10:00:00Z", "project": "p", "harness": "codex", + "task": "b", "answer": "b", "usage": {}, "skills": [{"name": "sota-check"}]}, + {"id": "c", "timestamp": "2026-08-03T10:00:00Z", "project": "p", "harness": "claude", + "task": "c", "answer": "c", "usage": {}, "skills": []}, + ], + })) + + summary = local_traces.store_summary(store, skill="build-loop") + assert summary["total"] == 1 + assert summary["recent"][0]["id"] == "a" + assert summary["filters"]["skill"] == "build-loop" + + # Counted over every turn, not the filtered set: otherwise selecting a skill empties the + # picker that selected it and you cannot switch to another. + assert summary["available_skills"] == {"sota-check": 2, "build-loop": 1} + assert local_traces.store_summary(store)["total"] == 3 diff --git a/tests/test_loop.py b/tests/test_loop.py index b5e4e53..4e6f529 100644 --- a/tests/test_loop.py +++ b/tests/test_loop.py @@ -1,5 +1,5 @@ """Unit tests for the continuous loop's health-gating (mine + run_ab are mocked).""" -from optimize import loop as L +from ingot.optimize import loop as L def test_loop_skips_healthy_skills(monkeypatch): @@ -53,7 +53,7 @@ def test_loop_runs_description_pass_when_configured(monkeypatch): monkeypatch.setattr(L, "run_ab", lambda skill, **k: order.append("body") or {"improved": False, "gate": {"promotable": False}}) - import optimize.routing as routing_mod + import ingot.optimize.routing as routing_mod monkeypatch.setattr(routing_mod, "run_routing", lambda skill, **k: order.append("description") or {"improved": True, "gate": {"promotable": True}}) diff --git a/tests/test_mine.py b/tests/test_mine.py index 8d775ba..3a5a137 100644 --- a/tests/test_mine.py +++ b/tests/test_mine.py @@ -1,11 +1,11 @@ -"""Unit tests for success/failure mining (optimize.mine), Langfuse HTTP and the judge are mocked.""" +"""Unit tests for success/failure mining (ingot.optimize.mine), Langfuse HTTP and the judge are mocked.""" import json from urllib.parse import parse_qs, urlparse import pytest -from optimize import mine -from optimize.judge import DIMENSIONS +from ingot.optimize import mine +from ingot.optimize.judge import DIMENSIONS class _Resp: @@ -36,6 +36,51 @@ def test_fetch_traces_keeps_only_usable_task_output_pairs(monkeypatch): ("do X", "r"), ("a bare string", "")] +def test_fetch_local_traces_requires_normalized_store_and_honors_newest_limit(tmp_path, monkeypatch): + path = tmp_path / "local-traces.json" + path.write_text(json.dumps({ + "schema_version": "ingot/local-traces/v1", + "traces": [ + {"id": "old", "timestamp": "2026-07-27T10:00:00Z", "task": "old task", + "answer": "old answer", "tags": ["skill:pdf"]}, + {"id": "new", "timestamp": "2026-07-28T10:00:00Z", "task": "new task", + "answer": "new answer", "tags": ["skill:pdf"]}, + ], + })) + monkeypatch.setattr(mine, "LOCAL_TRACE_FILE", path) + + assert mine.fetch_local_traces(1) == [{ + "task": "new task", "rubric": "", "answer": "new answer", "tags": ["skill:pdf"], + }] + + path.write_text(json.dumps({"schema_version": "wrong", "traces": []})) + with pytest.raises(SystemExit, match="unsupported local trace store"): + mine.fetch_local_traces() + + +def test_mine_local_source_requires_consent_and_never_reads_langfuse(monkeypatch): + traces = [{"task": "t", "rubric": "", "answer": "a", "tags": ["pdf"]}] + monkeypatch.setattr(mine, "fetch_local_traces", lambda limit: traces) + monkeypatch.setattr(mine, "fetch_traces", + lambda limit: pytest.fail("local mining must not contact Langfuse")) + monkeypatch.setattr(mine, "relevant_traces", lambda items, skill, k=5: items) + monkeypatch.setattr(mine, "_cluster_traces", lambda items, embed: [[0]]) + monkeypatch.setattr(mine, "_normalized_embedder", lambda: object()) + monkeypatch.setattr(mine, "_judge_trace_clusters", + lambda items, clusters, log, cache_path: ([ + (0, {"score": 1.0, "dimensions": {d: "pass" for d in DIMENSIONS}}, 1), + ], 0, 0)) + monkeypatch.setattr(mine, "_select_candidates", lambda *args, **kwargs: []) + + with pytest.raises(SystemExit, match="allow-external-judge"): + mine.mine("pdf", source="local", log=lambda *_args: None) + + result = mine.mine( + "pdf", source="local", allow_external_judge=True, log=lambda *_args: None) + + assert result["traces"] == 1 + + def test_fetch_traces_parses_langgraph_agent_traces(monkeypatch): lg = lambda task, answer: {"input": {"messages": [{"role": "user", "content": task}]}, "output": {"messages": [{"type": "ai", "content": answer}]}, "tags": ["demo"]} @@ -175,8 +220,8 @@ def __init__(self, skills): def suggest(self, task, k=5, min_score=0.0): return [{"name": "excel"}] if "spreadsheet" in task else [{"name": "other"}] - monkeypatch.setattr("mcp_server.router.Router", FakeRouter) - monkeypatch.setattr("mcp_server.registry.load_skills", lambda: []) + monkeypatch.setattr("ingot.mcp_server.router.Router", FakeRouter) + monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda: []) traces = [ {"task": "spreadsheet lookup with discounts", "tags": []}, # ranked -> keep (misrouted traffic) {"task": "rotate a pdf", "tags": ["excel"]}, # tagged -> keep @@ -185,6 +230,33 @@ def suggest(self, task, k=5, min_score=0.0): assert mine.relevant_traces(traces, "excel") == traces[:2] +def test_relevant_traces_matches_every_harness_tag_spelling(monkeypatch): + """Real traffic is tagged by whichever harness produced it, not by us. Claude Code writes + `skill:`, namespaced by plugin when the skill came from one; our own agent writes the + bare name plus a `revision=@` pin. Matching the bare form alone drops every + externally-produced trace, and mining then reports a heavily-used skill as never used.""" + class FakeRouter: + def __init__(self, skills): + pass + + def suggest(self, task, k=5, min_score=0.0): + return [{"name": "other"}] # never ranks, so the tag check alone decides + + monkeypatch.setattr("ingot.mcp_server.router.Router", FakeRouter) + monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda: []) + traces = [ + {"task": "t", "tags": ["pdf"]}, # bare, our agent + {"task": "t", "tags": ["claude-code", "skill:pdf"]}, # Claude Code + {"task": "t", "tags": ["skill:superpowers:pdf"]}, # plugin-namespaced + {"task": "t", "tags": ["demo", "revision=pdf@83a75cf1f9b5ada5"]}, # revision pin + {"task": "t", "tags": ["skill:pdf-tools"]}, # neighbour -> drop + {"task": "t", "tags": ["revision=excel@abc123"]}, # other skill -> drop + {"task": "t", "tags": []}, # untagged -> drop + {"task": "t"}, # no tags key -> drop + ] + assert mine.relevant_traces(traces, "pdf") == traces[:4] + + def test_mine_exits_when_no_traces_are_relevant(monkeypatch): import pytest monkeypatch.setattr(mine, "fetch_traces", diff --git a/tests/test_optimize.py b/tests/test_optimize.py index 4971d97..a7fec63 100644 --- a/tests/test_optimize.py +++ b/tests/test_optimize.py @@ -4,32 +4,37 @@ import pytest -from optimize import ab as ab_mod -from optimize import judge as judge_mod -from optimize.ab import body_retention, promotion_gate, retention_warnings -from optimize.judge import DIMENSIONS, failed_dimensions +from ingot.optimize import ab as ab_mod +from ingot.optimize import judge as judge_mod +from ingot.optimize.ab import body_retention, promotion_gate, retention_warnings +from ingot.optimize.judge import DIMENSIONS, failed_dimensions def test_optimizer_has_no_activation_control(): # the canary module is deleted outright on this branch, the strongest form of "no # activation control"; the remaining assertions cover the surviving surfaces - from optimize import promote as promotion + from ingot.optimize import promote as promotion assert "promote_now" not in inspect.signature(ab_mod.run_ab).parameters assert not hasattr(promotion, "promote") -def test_ensemble_judge_averages_score_and_majority_votes_dimensions(monkeypatch): - # two judges: one says 1.0 all-pass, one says 0.0 with a correctness failure -> mean 0.5, and - # correctness fails only if a MAJORITY flag it (here 1 of 2 → still "pass", harder to game) - fake = iter([ - {"score": 1.0, "feedback": "great", "dimensions": {d: "pass" for d in DIMENSIONS}}, - {"score": 0.0, "feedback": "wrong", "dimensions": {**{d: "pass" for d in DIMENSIONS}, "correctness": "bad API"}}, - ]) +def test_ensemble_judge_averages_items_and_majority_votes_dimensions(monkeypatch): + # Two judges disagreeing about correctness and agreeing on everything else. Averaging happens + # per checklist item, so the disagreement costs half of correctness's weight (3 of 8) rather + # than half of the whole answer: 1 - 0.5*(3/8) = 0.8125. Under the old holistic contract the + # same disagreement scored 0.5, which charged the challenger for three checks both judges + # passed. Correctness still reads as "pass" because 1 of 2 is not a majority. + def items(correctness): + return {"items": {i["id"]: {"value": correctness if i["id"] == "correctness" else 1.0, + "note": "bad API" if i["id"] == "correctness" else ""} + for i in judge_mod.DEFAULT_CHECKLIST}, + "feedback": "f", "unparseable": False} + fake = iter([items(1.0), items(0.0)]) monkeypatch.setattr(judge_mod, "MODELS", ["m1", "m2"]) - monkeypatch.setattr(judge_mod, "_judge_one", lambda model, prompt: next(fake)) + monkeypatch.setattr(judge_mod, "_judge_one", lambda model, prompt, checklist: next(fake)) r = judge_mod.judge("t", "rubric", "ans") - assert abs(r["score"] - 0.5) < 1e-9 + assert abs(r["score"] - 0.8125) < 1e-9 assert failed_dimensions(r["dimensions"]) == [] # 1/2 is not a majority → not flagged @@ -125,9 +130,51 @@ def test_load_tasks_reads_explicit_train_holdout(tmp_path, monkeypatch): assert split == {"kind": "holdout", "leakage": False} +def test_load_tasks_drafts_from_the_skills_own_root(tmp_path, monkeypatch): + """A multi-root library serves most skills from read-only mounts, never from the writable + authoring root. Drafting must resolve the skill where the registry indexed it — looking under + SKILLS_DIR instead meant every mounted skill died on FileNotFoundError before a task set + could be written.""" + from ingot.mcp_server.registry import Skill + + mounted = tmp_path / "mounted-library" / "gb10-serving" + mounted.mkdir(parents=True) + monkeypatch.setattr(ab_mod, "TASKS_DIR", tmp_path / "tasks") + (tmp_path / "tasks").mkdir() + monkeypatch.setattr("ingot.mcp_server.registry.load_skills", + lambda *a, **k: [Skill(name="gb10-serving", description="d", body="b", + path=str(mounted / "SKILL.md"), root=str(mounted))]) + monkeypatch.setattr("ingot.mcp_server.registry.read_components", + lambda d: {"description": f"desc-from:{d}", "body": "body"}) + + seen = {} + + def fake_draft(name, description, body, tasks_dir, **kw): + seen.update(name=name, description=description) + (tasks_dir / f"{name}.yaml").write_text( + "skill: gb10-serving\ntrain:\n - task: a\n rubric: r\nholdout:\n - task: b\n rubric: r\n") + + monkeypatch.setattr("ingot.optimize.draft.draft_and_save", fake_draft) + train, holdout, split = ab_mod.load_tasks("gb10-serving") + + assert seen["description"] == f"desc-from:{mounted}" # the mount, not SKILLS_DIR + assert [t["task"] for t in train] == ["a"] and split["leakage"] is False + + +def test_load_tasks_names_the_skill_when_it_is_not_indexed(tmp_path, monkeypatch): + """An unindexed name means the roots are misconfigured. Say so — the old code raised a bare + FileNotFoundError on a path the operator never configured directly.""" + import pytest + monkeypatch.setattr(ab_mod, "TASKS_DIR", tmp_path) + monkeypatch.setattr("ingot.mcp_server.registry.load_skills", lambda *a, **k: []) + with pytest.raises(SystemExit) as exc: + ab_mod.load_tasks("nonexistent") + assert "nonexistent" in str(exc.value) and "SKILL_ROUTER_PATHS" in str(exc.value) + + def test_greedy_pick_spreads_across_failure_modes(): import numpy as np - from optimize.mine import _greedy_pick + from ingot.optimize.mine import _greedy_pick # two orthogonal "failure modes", two near-identical tasks in each; hardest overall is # index 0, but the second pick must come from the OTHER mode even though index 1 is harder vecs = np.array([[1.0, 0.0], [1.0, 0.0], [0.0, 1.0], [0.0, 1.0]], dtype=np.float32) @@ -137,27 +184,28 @@ def test_greedy_pick_spreads_across_failure_modes(): def test_greedy_pick_skips_aced_and_excluded_tasks(): import numpy as np - from optimize.mine import _greedy_pick + from ingot.optimize.mine import _greedy_pick vecs = np.eye(3, dtype=np.float32) # one real candidate, one aced task (difficulty 0), one excluded (train near-duplicate) assert _greedy_pick([0.7, 0.0, -1.0], vecs, k=3) == [0] def test_save_pending_archives_a_displaced_cross_pass_challenger(tmp_path, monkeypatch): - from optimize import promote as promote_mod - monkeypatch.setattr(promote_mod, "PENDING_DIR", tmp_path) + from ingot.optimize import promote as promote_mod + monkeypatch.setenv("INGOT_RUNS", str(tmp_path)) + pending = promote_mod.pending_dir() promote_mod.save_pending("pdf", {"changed_components": ["body"], "created": 111}) promote_mod.save_pending("pdf", {"changed_components": ["body"], "created": 222}) # same pass: overwrite - assert len(list(tmp_path.glob("pdf*"))) == 1 + assert len(list(pending.glob("pdf*"))) == 1 promote_mod.save_pending("pdf", {"changed_components": ["description"], "created": 333}) import json - archived = tmp_path / "pdf.displaced-222.json" + archived = pending / "pdf.displaced-222.json" assert json.loads(archived.read_text())["changed_components"] == ["body"] # preserved - assert json.loads((tmp_path / "pdf.json").read_text())["changed_components"] == ["description"] + assert json.loads((pending / "pdf.json").read_text())["changed_components"] == ["description"] def test_length_penalty_is_zero_under_target_and_grows_above(): - from optimize.rollout import BODY_TARGET_CHARS, LENGTH_PENALTY, length_penalty + from ingot.optimize.rollout import BODY_TARGET_CHARS, LENGTH_PENALTY, length_penalty assert length_penalty("x" * (BODY_TARGET_CHARS // 2)) == 0.0 # concise -> no penalty assert length_penalty("x" * BODY_TARGET_CHARS) == 0.0 # exactly at target -> no penalty assert length_penalty("x" * (BODY_TARGET_CHARS * 2)) > 0.0 # bloated -> penalized @@ -167,7 +215,7 @@ def test_length_penalty_is_zero_under_target_and_grows_above(): def test_reflection_lm_sends_generic_api_key_to_openrouter(monkeypatch): import sys from types import SimpleNamespace - from optimize import rollout as rollout_mod + from ingot.optimize import rollout as rollout_mod captured = {} @@ -237,7 +285,7 @@ def test_gate_ignores_parity_with_no_cases(): def test_zdr_provider_pinned_and_in_sync(): - from optimize.judge import ZDR_PROVIDER + from ingot.optimize.judge import ZDR_PROVIDER assert ZDR_PROVIDER == {"provider": {"zdr": True, "data_collection": "deny"}} from agent.run import ZDR_PROVIDER as agent_zdr assert agent_zdr == ZDR_PROVIDER # duplicated literal (import-weight reasons) must not drift @@ -275,7 +323,7 @@ def test_optimize_split_rejects_unknown_component(monkeypatch): def test_skill_adapter_renders_frozen_components(): - from optimize.rollout import assemble + from ingot.optimize.rollout import assemble frozen, candidate = {"description": "when to use me"}, {"body": "the rules"} text = assemble({**frozen, **candidate}) assert "when to use me" in text and "the rules" in text @@ -297,8 +345,8 @@ def test_eval_serve_template_injects_body_and_contract(): def test_rollouts_serve_the_exact_serving_contract(monkeypatch): - from optimize import SERVE_TEMPLATE - from optimize import rollout as R + from ingot.optimize import SERVE_TEMPLATE + from ingot.optimize import rollout as R captured = {} class FakeLLM: @@ -311,7 +359,7 @@ class Msg: adapter = R.SkillAdapter(frozen={"description": "trigger words"}) adapter._llm = FakeLLM() monkeypatch.setattr(R, "judge", - lambda t, r, a, reference="", check=None, deliverable=None: + lambda t, r, a, reference="", check=None, deliverable=None, checklist=None: {"score": 1.0, "feedback": "f", "dimensions": {}}) candidate = {"body": "the rules"} answer, score, _ = adapter._rollout(adapter.serve(candidate), {"task": "t", "rubric": "r"}) @@ -323,10 +371,10 @@ class Msg: def test_agent_rollout_mode_routes_through_the_scaffold(monkeypatch): - from optimize import rollout as R + from ingot.optimize import rollout as R monkeypatch.setattr(R, "SKILLOPT_ROLLOUTS", "agent") monkeypatch.setattr(R, "judge", - lambda t, r, a, reference="", check=None, deliverable=None: + lambda t, r, a, reference="", check=None, deliverable=None, checklist=None: {"score": 0.5, "feedback": "f", "dimensions": {}}) seen = {} @@ -398,7 +446,7 @@ def _script_skill(tmp_path, monkeypatch, scripts=("scripts/helper.py",), holdout holdout[0]["check"] = {"fixture": "x = 1", "assert": "assert x == 1"} (tasks / "excel.yaml").write_text(yaml.safe_dump( {"train": [{"task": "t2", "rubric": "r"}], "holdout": holdout})) - monkeypatch.setattr(ab_mod, "SKILLS_DIR", skills) + monkeypatch.setattr(ab_mod, "resolve_skill_dir", lambda name: skills / name) monkeypatch.setattr(ab_mod, "TASKS_DIR", tasks) return d @@ -441,7 +489,7 @@ def fake_skillopt(seed, tasks, frozen=None, acceptance=None, log=print): calls.append({"key": key, "frozen": dict(frozen)}) return {key: text + "!"}, 0.5, 0.9 - monkeypatch.setattr("optimize.skillopt_loop.run_skillopt", fake_skillopt) + monkeypatch.setattr("ingot.optimize.skillopt_loop.run_skillopt", fake_skillopt) champion = {"description": "d", "body": "b", "file:scripts/a.py": "A", "file:scripts/b.py": "B"} challenger, seed_score, best_score = ab_mod._greedy_search( @@ -465,7 +513,7 @@ def fake_skillopt(seed, tasks, frozen=None, acceptance=None, log=print): calls.append({"seed": dict(seed), "frozen": dict(frozen)}) return {"body": "better"}, 0.3, 0.8 - monkeypatch.setattr("optimize.skillopt_loop.run_skillopt", fake_skillopt) + monkeypatch.setattr("ingot.optimize.skillopt_loop.run_skillopt", fake_skillopt) champion = {"description": "d", "body": "b"} challenger, s0, s1 = ab_mod._greedy_search("excel", champion, ["body"], [{"task": "t", "rubric": "r"}], log=lambda *_: None) diff --git a/tests/test_paths.py b/tests/test_paths.py new file mode 100644 index 0000000..91aa599 --- /dev/null +++ b/tests/test_paths.py @@ -0,0 +1,183 @@ +"""Where mutable state lives. + +The defect these protect against is not hypothetical: a `pip install ingot` kept its review queue, +its receipts, and its served library inside `site-packages`, so an upgrade discarded them, a +read-only install could not start, and two deployments sharing one installation shared one queue.""" +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +from ingot import paths + + +def _clear(monkeypatch): + for name in (paths.HOME, paths.LIBRARY, paths.RUNS, paths.TASKS, paths.VAULT, + *paths.LEGACY.values(), "XDG_STATE_HOME"): + monkeypatch.delenv(name, raising=False) + + +def test_state_defaults_outside_the_installed_package(monkeypatch): + """The whole point. Anything under the package directory is discarded by the next upgrade.""" + _clear(monkeypatch) + + for path in (paths.home(), paths.library(), paths.runs(), paths.tasks(), paths.vault()): + assert not path.is_relative_to(paths.PACKAGE_ROOT), path + + +def test_the_default_is_xdg(monkeypatch, tmp_path): + _clear(monkeypatch) + monkeypatch.setenv("XDG_STATE_HOME", str(tmp_path)) + + assert paths.home() == tmp_path / "ingot" + assert paths.runs() == tmp_path / "ingot" / "runs" + + +def test_without_xdg_it_falls_under_the_user_home(monkeypatch): + _clear(monkeypatch) + + assert paths.home() == Path.home() / ".local" / "state" / "ingot" + + +def test_one_home_moves_every_store(monkeypatch, tmp_path): + _clear(monkeypatch) + monkeypatch.setenv(paths.HOME, str(tmp_path)) + + assert paths.library() == tmp_path / "library" + assert paths.runs() == tmp_path / "runs" + assert paths.tasks() == tmp_path / "tasks" + + +@pytest.mark.parametrize("setting,accessor,default", [ + (paths.LIBRARY, paths.library, "library"), + (paths.RUNS, paths.runs, "runs"), + (paths.TASKS, paths.tasks, "tasks"), +]) +def test_each_store_can_be_placed_on_its_own(monkeypatch, tmp_path, setting, accessor, default): + """A container mounts each one separately; it does not get to choose a single parent.""" + _clear(monkeypatch) + monkeypatch.setenv(paths.HOME, str(tmp_path / "home")) + monkeypatch.setenv(setting, str(tmp_path / "elsewhere")) + + assert accessor() == tmp_path / "elsewhere" + assert paths.home() == tmp_path / "home" + + +def test_the_vault_defaults_to_the_library(monkeypatch, tmp_path): + """In the local backend the served checkout is the vault. Defaulting them apart would invent a + projection step that does not exist.""" + _clear(monkeypatch) + monkeypatch.setenv(paths.LIBRARY, str(tmp_path / "library")) + + assert paths.vault() == tmp_path / "library" + + monkeypatch.setenv(paths.VAULT, str(tmp_path / "vault")) + assert paths.vault() == tmp_path / "vault" + + +@pytest.mark.parametrize("setting,legacy", sorted(paths.LEGACY.items())) +def test_the_pre_existing_names_still_work(monkeypatch, tmp_path, setting, legacy): + """An existing deployment must not break on upgrade, and `ingot status` says which name won.""" + _clear(monkeypatch) + monkeypatch.setenv(legacy, str(tmp_path / "old")) + + assert paths._env(setting) == str(tmp_path / "old") + sources = " ".join(entry["source"] for entry in paths.resolved()) + assert legacy in sources and "deprecated" in sources + + +def test_the_explicit_setting_outranks_the_legacy_one(monkeypatch, tmp_path): + _clear(monkeypatch) + monkeypatch.setenv(paths.LIBRARY, str(tmp_path / "new")) + monkeypatch.setenv("SKILLS_DIR", str(tmp_path / "old")) + + assert paths.library() == tmp_path / "new" + + +def test_resolved_reports_where_every_path_came_from(monkeypatch, tmp_path): + _clear(monkeypatch) + monkeypatch.setenv(paths.HOME, str(tmp_path)) + monkeypatch.setenv(paths.RUNS, str(tmp_path / "elsewhere")) + + report = {entry["name"]: entry for entry in paths.resolved()} + + assert report["runs"]["source"] == paths.RUNS + assert report["library"]["source"] == paths.HOME + assert report["runs"]["path"] == str(tmp_path / "elsewhere") + + +def test_an_unwritable_parent_is_reported_rather_than_discovered_on_first_write(monkeypatch, + tmp_path): + """A path that does not exist yet is fine; one whose parent cannot be written is not, and + finding out at the first approval is the stall this reports instead.""" + locked = tmp_path / "locked" + locked.mkdir() + locked.chmod(0o500) + real_access = paths.os.access + monkeypatch.setattr(paths.os, "access", lambda path, mode: False + if Path(path) == locked else real_access(path, mode)) + _clear(monkeypatch) + monkeypatch.setenv(paths.HOME, str(locked / "ingot")) + try: + report = {entry["name"]: entry for entry in paths.resolved()} + finally: + locked.chmod(0o700) + + assert report["runs"]["exists"] is False + assert report["runs"]["writable"] is False + + +@pytest.mark.skipif(os.getuid() == 0, reason="root writes a mode-500 directory regardless") +def test_a_writable_missing_path_is_not_reported_as_a_problem(monkeypatch, tmp_path): + _clear(monkeypatch) + monkeypatch.setenv(paths.HOME, str(tmp_path / "not-created-yet")) + + report = {entry["name"]: entry for entry in paths.resolved()} + + assert report["runs"]["exists"] is False + assert report["runs"]["writable"] is True + + +def test_legacy_state_is_reported_never_migrated(monkeypatch, tmp_path): + """Moving someone's review queue on their behalf is a change to controlled state made by a + process nobody asked to make it.""" + monkeypatch.setattr(paths, "PACKAGE_ROOT", tmp_path) + assert paths.legacy_state() == [] + + (tmp_path / "runs" / "pending").mkdir(parents=True) + (tmp_path / "runs" / "pending" / "pdf.json").write_text("{}") + + assert paths.legacy_state() == [str(tmp_path / "runs")] + + +def test_an_empty_package_directory_is_not_legacy_state(monkeypatch, tmp_path): + """A checkout ships `skills/.gitkeep`. Reporting that as leftover state would train people to + ignore the warning.""" + monkeypatch.setattr(paths, "PACKAGE_ROOT", tmp_path) + (tmp_path / "skills").mkdir() + (tmp_path / "skills" / ".gitkeep").touch() + + assert paths.legacy_state() == [] + + +def test_an_installed_ingot_keeps_no_state_inside_the_package(tmp_path): + """In situ, because this is exactly the failure a unit test cannot see: the process must be + started somewhere other than the checkout, or the checkout's own directories answer for it.""" + script = ("import json, os, sys\n" + "os.environ['XDG_STATE_HOME'] = sys.argv[1]\n" + "from ingot import paths\n" + "print(json.dumps([entry['path'] for entry in paths.resolved()]))\n") + environment = {key: value for key, value in os.environ.items() + if not key.startswith(("INGOT_", "SKILLS_DIR", "VAULT_DIR", "XDG_"))} + environment["PYTHONPATH"] = str(Path(__file__).resolve().parent.parent) + + result = subprocess.run([sys.executable, "-c", script, str(tmp_path)], cwd=tmp_path, + capture_output=True, text=True, env=environment) + + assert result.returncode == 0, result.stderr + import json + for path in json.loads(result.stdout): + assert path.startswith(str(tmp_path)), path + assert not Path(path).is_relative_to(paths.PACKAGE_ROOT), path diff --git a/tests/test_promote.py b/tests/test_promote.py index ef7eca3..abe0a81 100644 --- a/tests/test_promote.py +++ b/tests/test_promote.py @@ -4,8 +4,9 @@ import pytest -from mcp_server.registry import load_skills, optimizable_components, skill_revision -from optimize import promote as P +from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision +from ingot.optimize import promote as P +from ingot.optimize import publication as Q def _library(tmp_path, monkeypatch): @@ -13,9 +14,9 @@ def _library(tmp_path, monkeypatch): skill = root / "pdf" skill.mkdir(parents=True) (skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n") + monkeypatch.setenv("INGOT_LIBRARY", str(root)) + monkeypatch.setenv("INGOT_RUNS", str(tmp_path)) monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root)) - monkeypatch.setattr(P, "PENDING_DIR", tmp_path / "pending") - monkeypatch.setattr(P, "REVISIONS_DIR", tmp_path / "revisions") return skill @@ -36,6 +37,10 @@ def _pending(skill_dir, *, promotable=True): } +def _complete(skill="pdf"): + return P._activate_approved(skill, P.load_pending(skill)) + + def test_promote_refuses_blocked_evidence(tmp_path, monkeypatch): skill = _library(tmp_path, monkeypatch) P.save_pending("pdf", _pending(skill, promotable=False)) @@ -63,34 +68,189 @@ def test_promote_refuses_bundled_file_drift(tmp_path, monkeypatch): P.approve_pending("pdf") +def test_approval_queues_publication_without_activating(tmp_path, monkeypatch): + skill = _library(tmp_path, monkeypatch) + pending = _pending(skill) + P.save_pending("pdf", pending) + + result = P.approve_pending("pdf", actor="admin") + + assert result == "Approved 'pdf'; publishing to vault." + assert "old body" in (skill / "SKILL.md").read_text() + assert P.load_pending("pdf") == pending + publication = Q.publication_for_skill("pdf") + assert publication["state"] == "approved_publishing" + assert publication["actor"] == "admin" + + def test_promote_snapshots_previous_revision_and_swaps_challenger(tmp_path, monkeypatch): skill = _library(tmp_path, monkeypatch) pending = _pending(skill) old_revision = pending["evidence"]["champion"]["revision"] P.save_pending("pdf", pending) - result = P.approve_pending("pdf") + result = _complete() assert "new body" in (skill / "SKILL.md").read_text() - assert "old body" in (P.REVISIONS_DIR / "pdf" / old_revision / "SKILL.md").read_text() + assert "old body" in (P.revisions_dir() / "pdf" / old_revision / "SKILL.md").read_text() assert not P.pending_path("pdf").exists() assert old_revision in result assert load_skills(skill.parent)[0].revision == pending["evidence"]["challenger"]["revision"] audit = json.loads((tmp_path / "approval-audit.jsonl").read_text()) assert audit["action"] == "approve" and audit["skill"] == "pdf" - result = P.rollback("pdf", old_revision) + result = P._activate_rollback("pdf", old_revision) assert "old body" in (skill / "SKILL.md").read_text() assert "Rolled back" in result records = [json.loads(line) for line in (tmp_path / "approval-audit.jsonl").read_text().splitlines()] assert [record["action"] for record in records] == ["approve", "rollback"] +def test_rollback_queues_exact_snapshot_without_changing_served_skill(tmp_path, monkeypatch): + skill = _library(tmp_path, monkeypatch) + old_revision = load_skills(skill.parent)[0].revision + P._snapshot(skill, "pdf", old_revision) + (skill / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\ncurrent body\n" + ) + current_revision = load_skills(skill.parent)[0].revision + + result = P.rollback("pdf", old_revision, actor="admin") + + assert result == f"Approved rollback of 'pdf' to {old_revision}; publishing to vault." + assert "current body" in (skill / "SKILL.md").read_text() + record = Q.publication_for_skill("pdf") + assert record["action"] == "rollback" + assert record["expected_champion"] == current_revision + assert record["candidate_revision"] == old_revision + + +def test_absence_rollback_queues_without_removing_served_skill(tmp_path, monkeypatch): + skill = _library(tmp_path, monkeypatch) + P._snapshot_absence("pdf") + current_revision = load_skills(skill.parent)[0].revision + + P.rollback("pdf", P.ABSENT_REVISION, actor="admin") + + assert skill.is_dir() + record = Q.publication_for_skill("pdf") + assert record["expected_champion"] == current_revision + assert record["candidate_revision"] == P.ABSENT_REVISION + assert record["components"] == {} + + +def test_promote_external_skill_through_writable_authoring_root(tmp_path, monkeypatch): + local = tmp_path / "local" + external = tmp_path / "external" + local.mkdir() + skill = external / "pdf" + skill.mkdir(parents=True) + skill_md = skill / "SKILL.md" + skill_md.write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n") + monkeypatch.setenv("INGOT_LIBRARY", str(local)) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external)) + pending = _pending(skill) + old_revision = pending["evidence"]["champion"]["revision"] + P.save_pending("pdf", pending) + + skill_md.chmod(0o444) + external.chmod(0o555) + try: + with pytest.warns(UserWarning, match="duplicate skill 'pdf'"): + result = _complete() + finally: + skill_md.chmod(0o644) + external.chmod(0o755) + + promoted = local / "pdf" + assert "Promoted 'pdf'" in result + assert "old body" in (skill / "SKILL.md").read_text() + assert "new body" in (promoted / "SKILL.md").read_text() + with pytest.warns(UserWarning, match="duplicate skill 'pdf'"): + assert load_skills()[0].root == str(promoted.resolve()) + assert "old body" in ( + P.revisions_dir() / "pdf" / old_revision / "SKILL.md").read_text() + + +def test_external_promotion_refuses_target_created_after_precheck(tmp_path, monkeypatch): + local = tmp_path / "local" + external = tmp_path / "external" + local.mkdir() + source = external / "pdf" + source.mkdir(parents=True) + (source / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n") + target = local / "pdf" + monkeypatch.setenv("INGOT_LIBRARY", str(local)) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external)) + P.save_pending("pdf", _pending(source)) + copytree = P.shutil.copytree + + def race(source_path, destination, *args, **kwargs): + result = copytree(source_path, destination, *args, **kwargs) + if Path(destination).name.endswith(".stage"): + target.mkdir() + return result + + monkeypatch.setattr(P.shutil, "copytree", race) + + with pytest.raises(ValueError, match="activation target appeared during promotion"): + _complete() + + assert target.is_dir() and list(target.iterdir()) == [] + assert "old body" in (source / "SKILL.md").read_text() + assert P.pending_path("pdf").exists() + + +def test_promote_refuses_unserved_authoring_root_collision(tmp_path, monkeypatch): + local = tmp_path / "local" + external = tmp_path / "external" + collision = local / "pdf" + collision.mkdir(parents=True) + (collision / "notes.txt").write_text("operator-owned") + skill = external / "pdf" + skill.mkdir(parents=True) + (skill / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n") + monkeypatch.setenv("INGOT_LIBRARY", str(local)) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external)) + P.save_pending("pdf", _pending(skill)) + + with pytest.raises(ValueError, match="writable activation target already exists"): + _complete() + + assert (collision / "notes.txt").read_text() == "operator-owned" + assert "old body" in (skill / "SKILL.md").read_text() + assert P.pending_path("pdf").exists() + + +def test_promote_refuses_dangling_authoring_root_symlink(tmp_path, monkeypatch): + local = tmp_path / "local" + external = tmp_path / "external" + local.mkdir() + collision = local / "pdf" + collision.symlink_to(local / "missing", target_is_directory=True) + skill = external / "pdf" + skill.mkdir(parents=True) + (skill / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n") + monkeypatch.setenv("INGOT_LIBRARY", str(local)) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external)) + P.save_pending("pdf", _pending(skill)) + + with pytest.raises(ValueError, match="writable activation target already exists"): + _complete() + + assert collision.is_symlink() + assert "old body" in (skill / "SKILL.md").read_text() + assert P.pending_path("pdf").exists() + + def test_approval_succeeds_when_audit_write_fails(tmp_path, monkeypatch, caplog): skill = _library(tmp_path, monkeypatch) pending = _pending(skill) P.save_pending("pdf", pending) monkeypatch.setattr(P, "_audit", lambda *args: (_ for _ in ()).throw(OSError("disk full"))) - result = P.approve_pending("pdf") + result = _complete() assert "Promoted 'pdf'" in result assert "new body" in (skill / "SKILL.md").read_text() @@ -103,10 +263,10 @@ def test_rollback_succeeds_when_audit_write_fails(tmp_path, monkeypatch, caplog) pending = _pending(skill) old_revision = pending["evidence"]["champion"]["revision"] P.save_pending("pdf", pending) - P.approve_pending("pdf") + _complete() monkeypatch.setattr(P, "_audit", lambda *args: (_ for _ in ()).throw(OSError("disk full"))) - result = P.rollback("pdf", old_revision) + result = P._activate_rollback("pdf", old_revision) assert "Rolled back" in result assert "old body" in (skill / "SKILL.md").read_text() @@ -122,7 +282,7 @@ def fail_write(*args, **kwargs): monkeypatch.setattr(P, "write_components", fail_write) with pytest.raises(RuntimeError, match="stage failed"): - P.approve_pending("pdf") + _complete() assert "old body" in (skill / "SKILL.md").read_text() @@ -133,7 +293,7 @@ def test_write_components_rejects_symlink_escape(tmp_path): (skill / "SKILL.md").write_text("---\nname: skill\ndescription: d\n---\nbody\n") (skill / "scripts").symlink_to(outside, target_is_directory=True) with pytest.raises(ValueError, match="escapes skill root"): - from mcp_server.registry import write_components + from ingot.mcp_server.registry import write_components write_components(skill, {"description": "d", "body": "b", "file:scripts/pwn.py": "bad"}) @@ -150,7 +310,7 @@ def test_promotion_sweeps_a_leftover_staging_directory(tmp_path, monkeypatch): assert load_skills(skill.parent)[0].body == "old body" # never shadowed the live skill P.save_pending("pdf", _pending(skill)) - P.approve_pending("pdf") + _complete() assert not stale.exists() assert list(skill.parent.glob(".pdf.*")) == [] @@ -203,7 +363,7 @@ def test_failed_rollback_copy_leaves_no_partial_staging_directory(tmp_path, monk skill = _library(tmp_path, monkeypatch) P.save_pending("pdf", _pending(skill)) old_revision = P.load_pending("pdf")["evidence"]["champion"]["revision"] - P.approve_pending("pdf") + _complete() def fail_copy(src, dst, **kwargs): Path(dst).mkdir(parents=True, exist_ok=True) # a partially copied tree @@ -211,7 +371,7 @@ def fail_copy(src, dst, **kwargs): monkeypatch.setattr(P.shutil, "copytree", fail_copy) with pytest.raises(RuntimeError, match="copy failed"): - P.rollback("pdf", old_revision) + P._activate_rollback("pdf", old_revision) assert list(skill.parent.glob(".pdf.*")) == [] assert "new body" in (skill / "SKILL.md").read_text() # the live skill is untouched @@ -227,9 +387,9 @@ def fail_copy(src, dst, **kwargs): monkeypatch.setattr(P.shutil, "copytree", fail_copy) with pytest.raises(RuntimeError, match="snapshot failed"): - P.approve_pending("pdf") + _complete() - assert list((P.REVISIONS_DIR / "pdf").glob("*")) == [] + assert list((P.revisions_dir() / "pdf").glob("*")) == [] assert "old body" in (skill / "SKILL.md").read_text() @@ -247,7 +407,7 @@ def _promote_body(skill, body): "evidence": {"champion": {"revision": current.revision}, "challenger": {"revision": skill_revision(skill, challenger)}, "gate": gate}, }) - P.approve_pending("pdf") + _complete() return current.revision @@ -260,7 +420,7 @@ def test_rollback_then_promote_orders_the_restored_revision_first(tmp_path, monk assert [r["revision"] for r in P.list_revisions("pdf")] == [second, first] - P.rollback("pdf", first) # snapshot C (third body), live is back on A + P._activate_rollback("pdf", first) # snapshot C (third body), live is back on A third = [r["revision"] for r in P.list_revisions("pdf")][0] assert third not in (first, second) @@ -275,7 +435,7 @@ def test_rollback_then_promote_orders_the_restored_revision_first(tmp_path, monk def test_list_revisions_falls_back_to_mtime_without_an_index(tmp_path, monkeypatch): """Snapshots taken before the index existed still list, ordered below stamped ones.""" _library(tmp_path, monkeypatch) - legacy = P.REVISIONS_DIR / "pdf" / ("a" * 8) + legacy = P.revisions_dir() / "pdf" / ("a" * 8) legacy.mkdir(parents=True) assert [r["revision"] for r in P.list_revisions("pdf")] == ["a" * 8] assert P.list_revisions("pdf")[0]["created"] > 0 @@ -288,7 +448,7 @@ def _write_snapshot_index(text: str) -> None: def _snapshot_dir(name: str) -> None: - (P.REVISIONS_DIR / "pdf" / name).mkdir(parents=True, exist_ok=True) + (P.revisions_dir() / "pdf" / name).mkdir(parents=True, exist_ok=True) @pytest.mark.parametrize("index", [ @@ -338,7 +498,7 @@ def test_promotion_survives_a_stamp_failure(tmp_path, monkeypatch, caplog, stamp monkeypatch.setattr(P, "_stamp_snapshot", lambda *a: (_ for _ in ()).throw(stamp_error)) - assert "Promoted 'pdf'" in P.approve_pending("pdf") + assert "Promoted 'pdf'" in _complete() assert "new body" in (skill / "SKILL.md").read_text() # the directory swap still happened assert "snapshot index write failed" in caplog.text assert [r["revision"] for r in P.list_revisions("pdf")] # mtime fallback still lists it @@ -348,7 +508,7 @@ def test_snapshot_index_is_not_restored_into_the_live_skill(tmp_path, monkeypatc skill = _library(tmp_path, monkeypatch) first = _promote_body(skill, "second body") - P.rollback("pdf", first) + P._activate_rollback("pdf", first) assert P.snapshot_index_path("pdf").exists() assert not (skill / ".snapshots.json").exists() diff --git a/tests/test_publication.py b/tests/test_publication.py new file mode 100644 index 0000000..c90c4d3 --- /dev/null +++ b/tests/test_publication.py @@ -0,0 +1,122 @@ +import json +import stat +from pathlib import Path + +import pytest + +from ingot.optimize import publication as Q + + +@pytest.fixture(autouse=True) +def publication_store(tmp_path, monkeypatch): + monkeypatch.setenv("INGOT_RUNS", str(tmp_path)) + + +def _pending(*, candidate="b" * 64, body="new body", action="promote"): + return { + "skill": "copywriting", + "kind": "retrospective", + "champion_components": {"description": "Write copy.", "body": "old body"}, + "challenger_components": ( + {} if candidate == "absent" else {"description": "Write copy.", "body": body} + ), + "evidence": { + "champion": {"revision": "a" * 64}, + "challenger": {"revision": candidate}, + "gate": {"promotable": True, "blocked": []}, + }, + "retrospective": {"proposal_id": "retro-123"}, + } + + +def test_queue_is_inert_and_captures_exact_approval(tmp_path, monkeypatch): + library = tmp_path / "skills" + monkeypatch.setenv("INGOT_LIBRARY", str(library)) + pending = _pending() + + receipt = Q.queue_publication("copywriting", pending, "admin", "promote") + + assert receipt.state == "approved_publishing" + assert not (library / "copywriting").exists() + stored = json.loads(receipt.path.read_text()) + assert stored["proposal_id"] == "retro-123" + assert stored["actor"] == "admin" + assert stored["action"] == "promote" + assert stored["expected_champion"] == "a" * 64 + assert stored["candidate_revision"] == "b" * 64 + assert stored["components"] == pending["challenger_components"] + assert stored["attempts"] == 0 and stored["last_error"] == "" + assert stat.S_IMODE(receipt.path.stat().st_mode) == 0o600 + + +def test_exact_retry_is_idempotent(): + first = Q.queue_publication("copywriting", _pending(), "admin", "promote") + second = Q.queue_publication("copywriting", _pending(), "admin", "promote") + + assert second.id == first.id + assert list(Q.publications_dir().glob("*.json")) == [first.path] + + +def test_different_candidate_cannot_occupy_same_skill_lane(): + Q.queue_publication("copywriting", _pending(), "admin", "promote") + + with pytest.raises(ValueError, match="publication is already in progress"): + Q.queue_publication( + "copywriting", _pending(candidate="c" * 64, body="other body"), + "admin", "promote", + ) + + +def test_final_record_is_absent_when_atomic_publication_fails(monkeypatch): + def fail_link(source, destination): + raise OSError("disk failure") + + monkeypatch.setattr(Q.os, "link", fail_link) + with pytest.raises(OSError, match="disk failure"): + Q.queue_publication("copywriting", _pending(), "admin", "promote") + + assert list(Q.publications_dir().glob("*.json")) == [] + + +def test_absence_rollback_can_queue_without_skill_components(): + receipt = Q.queue_publication( + "copywriting", _pending(candidate="absent"), "admin", "rollback" + ) + + stored = json.loads(receipt.path.read_text()) + assert stored["action"] == "rollback" + assert stored["candidate_revision"] == "absent" + assert stored["components"] == {} + + +def test_the_newest_receipt_wins_within_one_second(tmp_path, monkeypatch): + """Two receipts for one skill are routinely queued in the same second — approve, then roll + back. Whole-second timestamps would leave the review surface showing whichever id sorted + higher.""" + monkeypatch.setenv("INGOT_LIBRARY", str(tmp_path / "skills")) + first = Q.queue_publication("copywriting", _pending(), "admin", "promote") + Q.update_publication(first.id, state="active") + rollback = _pending(candidate="absent") + second = Q.queue_publication("copywriting", rollback, "admin", "rollback") + + latest = Q.publication_for_skill("copywriting") + + assert latest["id"] == second.id and latest["id"] != first.id + assert latest["created"] == json.loads(first.path.read_text())["created"] # the same second + + +@pytest.mark.parametrize("malformed", [[], "", 0]) +def test_rollback_refuses_components_of_the_wrong_shape(malformed): + """Absence is expressed by an empty object, not by any falsy value a malformed record holds.""" + pending = _pending(candidate="absent") + pending["challenger_components"] = malformed + + with pytest.raises(ValueError, match="components must be an object"): + Q.queue_publication("copywriting", pending, "admin", "rollback") + + +def test_promotion_still_requires_description_and_body(): + with pytest.raises(ValueError, match="challenger components are required"): + Q.queue_publication( + "copywriting", _pending(candidate="absent"), "admin", "promote" + ) diff --git a/tests/test_publisher.py b/tests/test_publisher.py new file mode 100644 index 0000000..f677009 --- /dev/null +++ b/tests/test_publisher.py @@ -0,0 +1,548 @@ +import json +import os +import subprocess +from pathlib import Path + +import pytest + +from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision +from ingot.optimize import promote as P +from ingot.optimize import publication as Q +from ingot.optimize import publisher as W + + +def test_publisher_unit_uses_the_portable_managed_configuration(): + root = Path(__file__).resolve().parents[1] + unit = (root / "ops/systemd/ingot-publisher.service").read_text() + + assert "EnvironmentFile=%h/.config/ingot/publisher.env" in unit + assert "ExecStart=/usr/bin/python3 -m ingot.optimize.publisher --watch" in unit + assert "Slancha" not in unit + + +def _git(path, *args, check=True): + return subprocess.run(["git", "-C", str(path), *args], check=check, + capture_output=True, text=True).stdout.strip() + + +FORGE_REPOSITORY = "example/skills" + + +def _open(vault, remote, **kwargs): + return W.VaultRepo.open(vault, remote="origin", + expected_remotes={str(Path(remote).resolve())}, **kwargs) + + +def _publisher(vault, remote, github=None): + """A forge-backend publisher over a bare repository standing in for GitHub. + + `expected_remotes` is the configured repository's resolved URL. A test vault's origin is a + filesystem path rather than a github.com URL, which is exactly the case the hardcoded pair of + literals could not express.""" + backend = W.ForgeBackend(github if github is not None else FakeGitHub(), + repository=FORGE_REPOSITORY, + expected_remotes={str(Path(remote).resolve())}) + return W.Publisher(vault, backend=backend) + + +def _repositories(tmp_path, monkeypatch): + remote = tmp_path / "remote.git" + vault = tmp_path / "vault" + subprocess.run(["git", "init", "--bare", str(remote)], check=True, capture_output=True) + subprocess.run(["git", "init", "-b", "main", str(vault)], check=True, capture_output=True) + _git(vault, "config", "user.name", "Ingot Test") + _git(vault, "config", "user.email", "ingot@test.invalid") + skill = vault / "pdf" + skill.mkdir() + (skill / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n") + (vault / "registry.json").write_text(json.dumps({ + "pdf": {"disposition": "keep", "reason": "test"} + }) + "\n") + scripts = vault / "scripts" + scripts.mkdir() + (scripts / "validate.py").write_text("print('valid')\n") + _git(vault, "add", ".") + _git(vault, "commit", "-m", "Initial vault") + _git(vault, "remote", "add", "origin", str(remote)) + _git(vault, "push", "-u", "origin", "main") + subprocess.run(["git", "--git-dir", str(remote), "symbolic-ref", "HEAD", "refs/heads/main"], + check=True) + monkeypatch.setenv("INGOT_LIBRARY", str(vault)) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(vault)) + return remote, vault, skill + + +def _queue(skill): + champion = optimizable_components(skill) + challenger = {**champion, "body": "new body"} + current = load_skills(skill.parent)[0] + pending = { + "skill": "pdf", "champion_components": champion, + "challenger_components": challenger, "gate": {"promotable": True, "blocked": []}, + "evidence": {"champion": {"revision": current.revision}, + "challenger": {"revision": skill_revision(skill, challenger)}}, + } + P.save_pending("pdf", pending) + return Q.queue_publication("pdf", pending, "admin", "promote") + + +class FakeGitHub: + def __init__(self): + self.merged = None + + def create_or_find(self, branch, publication_id): + return 17 + + def enable_auto_merge(self, pr): + return None + + def merged_commit(self, pr): + return self.merged + + +def _admin(remote, tmp_path): + """A second clone standing in for whoever merges the vault pull request.""" + admin = tmp_path / "admin" + if not admin.exists(): + subprocess.run(["git", "clone", str(remote), str(admin)], check=True, capture_output=True) + _git(admin, "config", "user.name", "Vault Admin") + _git(admin, "config", "user.email", "admin@test.invalid") + return admin + + +def _merge(remote, tmp_path, branch): + admin = _admin(remote, tmp_path) + _git(admin, "fetch", "origin") + _git(admin, "merge", "--ff-only", f"origin/{branch}") + _git(admin, "push", "origin", "main") + return _git(admin, "rev-parse", "HEAD") + + +def _publish(vault, publication_id, remote, tmp_path): + """Drive one queued publication all the way through a merged vault commit.""" + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + assert publisher.process(publication_id) == "awaiting_merge" + github.merged = _merge(remote, tmp_path, Q.load_publication(publication_id)["branch"]) + assert publisher.process(publication_id) == "active" + + +def test_unmerged_publication_never_changes_served_skill(tmp_path, monkeypatch): + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + + assert publisher.process(receipt.id) == "awaiting_merge" + assert publisher.process(receipt.id) == "awaiting_merge" + assert "old body" in (skill / "SKILL.md").read_text() + assert P.pending_path("pdf").exists() + + +def test_merged_exact_revision_activates_and_consumes_pending(tmp_path, monkeypatch): + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + assert publisher.process(receipt.id) == "awaiting_merge" + record = Q.load_publication(receipt.id) + github.merged = _merge(remote, tmp_path, record["branch"]) + + assert publisher.process(receipt.id) == "active" + assert "new body" in (skill / "SKILL.md").read_text() + assert not P.pending_path("pdf").exists() + assert skill_revision(skill) == record["candidate_revision"] + assert (P.revisions_dir() / "pdf" / record["expected_champion"]).is_dir() + + +def test_push_failure_preserves_champion_and_pending(tmp_path, monkeypatch): + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + run = W._run + + def reject_push(command, *, cwd): + if command[:2] == ["git", "push"]: + raise RuntimeError("git push: rejected by test remote") + return run(command, cwd=cwd) + + monkeypatch.setattr(W, "_run", reject_push) + + with pytest.raises(RuntimeError, match="git push"): + _publisher(vault, remote).process(receipt.id) + + assert "old body" in (skill / "SKILL.md").read_text() + assert P.pending_path("pdf").exists() + assert Q.load_publication(receipt.id)["last_error"] + + +def test_retry_reuses_the_recorded_branch_after_push_failure(tmp_path, monkeypatch): + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + run = W._run + failures = 1 + + def fail_once(command, *, cwd): + nonlocal failures + if command[:2] == ["git", "push"] and failures: + failures -= 1 + raise RuntimeError("git push: transient failure") + return run(command, cwd=cwd) + + monkeypatch.setattr(W, "_run", fail_once) + publisher = _publisher(vault, remote) + with pytest.raises(RuntimeError, match="transient failure"): + publisher.process(receipt.id) + + assert publisher.process(receipt.id) == "awaiting_merge" + record = Q.load_publication(receipt.id) + assert record["branch"] == f"ingot/{receipt.id}" + + +def test_merged_commit_may_be_an_ancestor_of_a_newer_origin_main(tmp_path, monkeypatch): + """An unrelated vault commit landing after the approved merge must not wedge the publication.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + assert publisher.process(receipt.id) == "awaiting_merge" + github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"]) + admin = _admin(remote, tmp_path) + (admin / "NOTES.md").write_text("an unrelated vault edit\n") + _git(admin, "add", "NOTES.md") + _git(admin, "commit", "-m", "Unrelated vault commit") + _git(admin, "push", "origin", "main") + + assert publisher.process(receipt.id) == "active" + assert "new body" in (skill / "SKILL.md").read_text() + assert (vault / "NOTES.md").exists() + + +def test_crash_after_fast_forward_finalizes_on_retry(tmp_path, monkeypatch): + """The receipt is written after the fast-forward, so a crash between them leaves the approved + revision already served. The retry must finalize rather than refuse the departed champion.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + assert publisher.process(receipt.id) == "awaiting_merge" + github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"]) + _git(vault, "fetch", "origin", "main") + _git(vault, "merge", "--ff-only", "origin/main") # the fast-forward that survived + + assert publisher.process(receipt.id) == "active" + assert "new body" in (skill / "SKILL.md").read_text() + assert not P.pending_path("pdf").exists() + + +def test_candidate_mismatch_never_consumes_pending_or_activates(tmp_path, monkeypatch): + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + assert publisher.process(receipt.id) == "awaiting_merge" + github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"]) + pending = P.load_pending("pdf") + pending["evidence"]["challenger"]["revision"] = "f" * 64 + P.save_pending("pdf", pending) + + with pytest.raises(RuntimeError, match="pending review no longer matches"): + publisher.process(receipt.id) + + assert P.pending_path("pdf").exists() + assert Q.load_publication(receipt.id)["state"] == "awaiting_merge" + assert "old body" in (skill / "SKILL.md").read_text() # nothing was fast-forwarded either + assert not (P.revisions_dir() / "pdf").exists() # and nothing was snapshotted + + +def test_a_closed_vault_pull_request_is_refused_rather_than_reopened(tmp_path, monkeypatch): + """Closing the vault pull request rejects the publication. Opening a second one for the same + branch would overrule the person who closed it.""" + remote, vault, _ = _repositories(tmp_path, monkeypatch) + monkeypatch.setattr(W, "_run", lambda command, *, cwd: json.dumps([ + {"number": 17, "state": "CLOSED"}]) if command[0] == "gh" else "") + + with pytest.raises(ValueError, match="closed without merging"): + W.GitHub(vault, FORGE_REPOSITORY).create_or_find("ingot/abc123", "abc123") + + +def test_publication_replaces_a_stale_removal_entry_in_the_registry(tmp_path, monkeypatch): + """A skill left marked for removal would land in the vault and then be dropped by the + projection: the entry has to follow what the publication actually did.""" + remote, vault, _ = _repositories(tmp_path, monkeypatch) + (vault / "registry.json").write_text(json.dumps({ + "pdf": {"disposition": "remove", "reason": "removed earlier"}}) + "\n") + + W.Publisher._register(vault, "pdf", present=True) + + assert json.loads((vault / "registry.json").read_text())["pdf"]["disposition"] == "keep" + + +def test_a_leftover_worktree_never_carries_unrelated_edits_into_the_pull_request(tmp_path, + monkeypatch): + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + workspace = Q.publications_dir() / "worktrees" / receipt.id + workspace.mkdir(parents=True) + (workspace / "STRAY.md").write_text("left behind by a killed run\n") + + publisher = _publisher(vault, remote) + assert publisher.process(receipt.id) == "awaiting_merge" + + branch = Q.load_publication(receipt.id)["branch"] + listed = _git(vault, "ls-tree", "-r", "--name-only", f"refs/heads/{branch}") + assert "STRAY.md" not in listed + assert "pdf/SKILL.md" in listed + + +def test_rollback_stays_inert_until_the_vault_merge_restores_it(tmp_path, monkeypatch): + remote, vault, skill = _repositories(tmp_path, monkeypatch) + champion = load_skills(vault)[0].revision + _publish(vault, _queue(skill).id, remote, tmp_path) + assert "new body" in (skill / "SKILL.md").read_text() + + result = P.rollback("pdf", champion, actor="admin") + assert result.endswith("publishing to vault.") + receipt = Q.publication_for_skill("pdf") + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + + assert publisher.process(receipt["id"]) == "awaiting_merge" + assert "new body" in (skill / "SKILL.md").read_text() # inert until the merge lands + + github.merged = _merge(remote, tmp_path, Q.load_publication(receipt["id"])["branch"]) + assert publisher.process(receipt["id"]) == "active" + assert "old body" in (skill / "SKILL.md").read_text() + assert skill_revision(skill) == champion + + +def test_absence_rollback_removes_the_skill_only_after_merge(tmp_path, monkeypatch): + remote, vault, skill = _repositories(tmp_path, monkeypatch) + P._snapshot_absence("pdf") + + P.rollback("pdf", P.ABSENT_REVISION, actor="admin") + receipt = Q.publication_for_skill("pdf") + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + + assert publisher.process(receipt["id"]) == "awaiting_merge" + assert skill.is_dir() + + github.merged = _merge(remote, tmp_path, Q.load_publication(receipt["id"])["branch"]) + assert publisher.process(receipt["id"]) == "active" + assert not skill.exists() + assert "pdf" not in json.loads((vault / "registry.json").read_text()) + + +def test_absence_rollback_retries_after_the_branch_already_removed_the_skill(tmp_path, monkeypatch): + """The second attempt starts from `origin/`, where the skill is already gone. Staging + it by name is a fatal `git add` there, which wedged a real publication in the vault.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + P._snapshot_absence("pdf") + P.rollback("pdf", P.ABSENT_REVISION, actor="admin") + receipt = Q.publication_for_skill("pdf") + + class FailsOnce(FakeGitHub): + calls = 0 + + def create_or_find(self, branch, publication_id): + FailsOnce.calls += 1 + if FailsOnce.calls == 1: + raise RuntimeError("gh pr: transient failure after the push") + return 17 + + publisher = _publisher(vault, remote, FailsOnce()) + with pytest.raises(RuntimeError, match="transient failure"): + publisher.process(receipt["id"]) + + assert publisher.process(receipt["id"]) == "awaiting_merge" + branch = Q.load_publication(receipt["id"])["branch"] + assert "pdf/SKILL.md" not in _git(vault, "ls-tree", "-r", "--name-only", f"refs/heads/{branch}") + + +def test_a_vault_without_auto_merge_waits_for_a_human_instead_of_wedging(tmp_path, monkeypatch): + """`laulpogan/skills` has auto-merge disabled. Treating that as fatal stranded an approval that + was already sitting in an open pull request.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + + class NoAutoMerge(FakeGitHub): + def enable_auto_merge(self, pr): + raise RuntimeError("gh pr: GraphQL: Auto merge is not allowed for this repository") + + assert _publisher(vault, remote, NoAutoMerge()).process(receipt.id) == "awaiting_merge" + + record = Q.load_publication(receipt.id) + assert record["pr"] == 17 and record["auto_merge"] is False + assert "waiting on a human merge" in record["note"] + assert record["last_error"] == "" + assert "old body" in (skill / "SKILL.md").read_text() + + +def test_rollback_restores_a_file_the_displaced_revision_added(tmp_path, monkeypatch): + """Components describe text the optimizer may rewrite, not the whole skill. Restoring the + snapshot tree is what makes a rollback exact when a revision added a file.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + champion = load_skills(vault)[0].revision + P._snapshot(skill, "pdf", champion) + (skill / "extra.md").write_text("added by a later revision\n") + _git(vault, "add", "pdf") + _git(vault, "commit", "-m", "Add a bundled file") + _git(vault, "push", "origin", "main") + + P.rollback("pdf", champion, actor="admin") + receipt = Q.publication_for_skill("pdf") + _publish(vault, receipt["id"], remote, tmp_path) + + assert not (skill / "extra.md").exists() + assert skill_revision(skill) == champion + + +@pytest.mark.skipif(os.getuid() == 0, reason="root reads a mode-000 directory regardless") +def test_an_unreadable_receipt_store_is_reported_not_read_as_empty(tmp_path): + """Path.glob swallows PermissionError, so a queue the publisher cannot list looks exactly like + one with nothing in it: approvals pile up in the console and the publisher says nothing. This + is the deployment failure where the console writes receipts as one user and the publisher runs + as another.""" + store = tmp_path / "publications" + store.mkdir() + (store / "abc.json").write_text("{}") + store.chmod(0o000) + try: + assert list(store.glob("*.json")) == [] # indistinguishable from empty + blocked = W.unreadable_queue(store) + finally: + store.chmod(0o700) + + assert blocked and "cannot read the receipt store" in blocked + assert W.unreadable_queue(tmp_path / "never-created") is None + store.chmod(0o700) + assert W.unreadable_queue(store) is None + + +def test_activation_retires_the_publication_branch(tmp_path, monkeypatch): + """One branch per publication, kept forever, grows the vault's branch list without bound.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + assert publisher.process(receipt.id) == "awaiting_merge" + branch = Q.load_publication(receipt.id)["branch"] + github.merged = _merge(remote, tmp_path, branch) + + assert publisher.process(receipt.id) == "active" + + assert branch not in _git(vault, "branch", "--list", branch) + assert not _git(vault, "ls-remote", "--heads", "origin", f"refs/heads/{branch}") + assert "new body" in (skill / "SKILL.md").read_text() + + +def test_a_branch_that_cannot_be_retired_leaves_the_publication_active(tmp_path, monkeypatch): + """Cleanup runs after the receipt is durable, so a failure there must not turn a completed + activation back into a retry.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + assert publisher.process(receipt.id) == "awaiting_merge" + github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"]) + run = W._run + + def refuse_delete(command, *, cwd): + if "--delete" in command or command[:2] == ["git", "branch"]: + raise RuntimeError("git push: the remote refused the deletion") + return run(command, cwd=cwd) + + monkeypatch.setattr(W, "_run", refuse_delete) + + assert publisher.process(receipt.id) == "active" + assert Q.load_publication(receipt.id)["state"] == "active" + + +def test_an_unrelated_vault_commit_does_not_block_the_next_publication(tmp_path, monkeypatch): + """The vault has other writers. The served library is a mirror of vault main, so falling + behind is normal — and refusing to publish until someone pulls by hand blocked the lane in + production the first time anybody else committed.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + admin = _admin(remote, tmp_path) + (admin / "UNRELATED.md").write_text("someone else's vault commit\n") + _git(admin, "add", "UNRELATED.md") + _git(admin, "commit", "-m", "An unrelated vault commit") + _git(admin, "push", "origin", "main") + + with pytest.raises(ValueError, match="HEAD must equal origin/main"): + _open(vault, remote) # finalize still refuses to sync silently + + assert _publisher(vault, remote).process(receipt.id) == "awaiting_merge" + assert (vault / "UNRELATED.md").exists() # the mirror caught up + assert "old body" in (skill / "SKILL.md").read_text() + + +def test_a_diverged_vault_is_never_synced(tmp_path, monkeypatch): + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + (vault / "local.txt").write_text("a commit only this checkout has") + _git(vault, "add", "local.txt") + _git(vault, "commit", "-m", "Diverge") + + with pytest.raises(RuntimeError, match="diverged"): + _publisher(vault, remote).process(receipt.id) + + +@pytest.mark.parametrize("fault", ["dirty", "detached", "wrong_remote", "diverged"]) +def test_repository_guard_fails_before_publication_writes(tmp_path, monkeypatch, fault): + remote, vault, _ = _repositories(tmp_path, monkeypatch) + if fault == "dirty": + (vault / "dirty.txt").write_text("dirty") + elif fault == "detached": + _git(vault, "checkout", "--detach") + elif fault == "wrong_remote": + _git(vault, "remote", "set-url", "origin", str(tmp_path / "wrong.git")) + else: + (vault / "local.txt").write_text("local") + _git(vault, "add", "local.txt") + _git(vault, "commit", "-m", "Diverge") + + with pytest.raises(ValueError): + _open(vault, remote) + + +def test_polling_an_unmerged_publication_does_not_touch_the_vault(tmp_path, monkeypatch): + """`watch` polls every few seconds. Fetching and re-validating the checkout on each poll puts a + network round trip -- and a failure mode -- in front of a question `gh` already answers, and it + made an unrelated dirty working tree fail a receipt that was simply still waiting.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + publisher = _publisher(vault, remote) + assert publisher.process(receipt.id) == "awaiting_merge" + (vault / "someone-is-editing.txt").write_text("an unrelated local edit") + + run = W._run + touched = [] + + def record(command, *, cwd): + touched.append(command) + return run(command, cwd=cwd) + + monkeypatch.setattr(W, "_run", record) + + assert publisher.process(receipt.id) == "awaiting_merge" + + assert not [command for command in touched if command[:2] == ["git", "fetch"]] + assert Q.load_publication(receipt.id)["last_error"] == "" + + +def test_a_merge_that_just_landed_is_not_refused_as_missing(tmp_path, monkeypatch): + """The ancestry check runs against whatever `origin/main` this checkout last saw. Without a + fetch of its own it would refuse the one outcome the lane is waiting for.""" + remote, vault, skill = _repositories(tmp_path, monkeypatch) + receipt = _queue(skill) + github = FakeGitHub() + publisher = _publisher(vault, remote, github) + assert publisher.process(receipt.id) == "awaiting_merge" + github.merged = _merge(remote, tmp_path, Q.load_publication(receipt.id)["branch"]) + assert not W.VaultRepo(vault, remote="origin").contains(github.merged) # not fetched yet + + assert publisher.process(receipt.id) == "active" + assert "new body" in (skill / "SKILL.md").read_text() diff --git a/tests/test_publisher_local.py b/tests/test_publisher_local.py new file mode 100644 index 0000000..a7347cc --- /dev/null +++ b/tests/test_publisher_local.py @@ -0,0 +1,570 @@ +"""The local publication backend: a Git vault on this machine, no network at any point. + +Every test here runs offline by construction — the vault has no remote to reach — and several of +them assert that explicitly, because "air-gappable" is a claim the code has to keep rather than a +property of how the test happened to be written. +""" +import json +import subprocess +from pathlib import Path + +import pytest + +from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision +from ingot.optimize import promote as P +from ingot.optimize import publication as Q +from ingot.optimize import publisher as W +from ingot import delivery as D + + +def _git(path, *args): + return subprocess.run(["git", "-C", str(path), *args], check=True, + capture_output=True, text=True).stdout.strip() + + +def _vault(tmp_path, monkeypatch): + """A vault with no remote at all. Nothing here has ever heard of GitHub.""" + vault = tmp_path / "vault" + subprocess.run(["git", "init", "-b", "main", str(vault)], check=True, capture_output=True) + _git(vault, "config", "user.name", "Vault Owner") + _git(vault, "config", "user.email", "owner@test.invalid") + skill = vault / "pdf" + skill.mkdir() + (skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n") + (vault / "registry.json").write_text( + json.dumps({"pdf": {"disposition": "keep", "reason": "test"}}) + "\n") + (vault / "scripts").mkdir() + (vault / "scripts" / "validate.py").write_text("print('valid')\n") + _git(vault, "add", ".") + _git(vault, "commit", "-m", "Initial vault") + monkeypatch.setenv("INGOT_LIBRARY", str(vault)) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(vault)) + return vault, skill + + +def _queue(skill, body="new body"): + champion = optimizable_components(skill) + challenger = {**champion, "body": body} + current = load_skills(skill.parent)[0] + pending = { + "skill": "pdf", "champion_components": champion, + "challenger_components": challenger, "gate": {"promotable": True, "blocked": []}, + "evidence": {"champion": {"revision": current.revision}, + "challenger": {"revision": skill_revision(skill, challenger)}}, + } + P.save_pending("pdf", pending) + return Q.queue_publication("pdf", pending, "admin", "promote") + + +def _publisher(vault): + return W.Publisher(vault, backend=W.LocalBackend()) + + +def _forbid_network(monkeypatch): + """Fail loudly on anything that would leave the machine.""" + run = W._run + + def offline(command, *, cwd): + if command[0] == "gh" or command[:2] == ["git", "push"] or "ls-remote" in command: + raise AssertionError(f"the local backend reached the network: {' '.join(command)}") + return run(command, cwd=cwd) + + monkeypatch.setattr(W, "_run", offline) + + +# --------------------------------------------------------------------------- the happy path + +def test_a_vault_with_no_remote_publishes_and_activates_in_one_pass(tmp_path, monkeypatch): + """There is no `awaiting_merge`: nothing external has to agree before the bytes are served.""" + vault, skill = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + receipt = _queue(skill) + record = Q.load_publication(receipt.id) + + assert _publisher(vault).process(receipt.id) == "active" + + assert "new body" in (skill / "SKILL.md").read_text() + assert skill_revision(skill) == record["candidate_revision"] + assert not P.pending_path("pdf").exists() + assert (P.revisions_dir() / "pdf" / record["expected_champion"]).is_dir() + stored = Q.load_publication(receipt.id) + assert stored["state"] == "active" + assert stored["merged_commit"] == _git(vault, "rev-parse", "HEAD") + + +def test_approval_alone_leaves_the_served_checkout_byte_identical(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + before = (skill / "SKILL.md").read_bytes() + head = _git(vault, "rev-parse", "HEAD") + + _queue(skill) + + assert (skill / "SKILL.md").read_bytes() == before + assert _git(vault, "rev-parse", "HEAD") == head + + +def test_the_vault_commit_is_authored_by_the_publisher_not_the_host(tmp_path, monkeypatch): + """The commit records who published, not whoever happens to own the shell.""" + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + + assert _publisher(vault).process(receipt.id) == "active" + + assert _git(vault, "log", "-1", "--format=%an <%ae>") == "Ingot Publisher " + assert _git(vault, "log", "-1", "--format=%s") == f"Publish pdf via Ingot {receipt.id}" + + +def test_activation_retires_the_publication_branch(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + + assert _publisher(vault).process(receipt.id) == "active" + + assert _git(vault, "branch", "--list", f"ingot/{receipt.id}") == "" + + +def test_republishing_a_revision_the_vault_already_serves_is_a_no_op(tmp_path, monkeypatch): + """An empty diff must converge on the existing tip. `git commit` with nothing staged exits + non-zero, so treating this as an error would fail a receipt that has nothing left to do.""" + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + publisher = _publisher(vault) + assert publisher.process(receipt.id) == "active" + head = _git(vault, "rev-parse", "HEAD") + + Q.update_publication(receipt.id, state="approved_publishing") + assert publisher.process(receipt.id) == "active" + + assert _git(vault, "rev-parse", "HEAD") == head + + +def test_an_absence_rollback_removes_the_skill_and_its_registry_entry(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + P._snapshot_absence("pdf") + P.rollback("pdf", P.ABSENT_REVISION, actor="admin") + receipt = Q.publication_for_skill("pdf") + + assert _publisher(vault).process(receipt["id"]) == "active" + + assert not skill.exists() + assert "pdf" not in json.loads((vault / "registry.json").read_text()) + + +def test_a_rollback_restores_the_stored_snapshot(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + champion = load_skills(vault)[0].revision + assert _publisher(vault).process(_queue(skill).id) == "active" + assert "new body" in (skill / "SKILL.md").read_text() + + P.rollback("pdf", champion, actor="admin") + receipt = Q.publication_for_skill("pdf") + + assert _publisher(vault).process(receipt["id"]) == "active" + + assert "old body" in (skill / "SKILL.md").read_text() + assert skill_revision(skill) == champion + + +# --------------------------------------------------------------------------- recovery + +def test_a_leftover_worktree_is_destroyed_rather_than_reused(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + workspace = Q.publications_dir() / "worktrees" / receipt.id + workspace.mkdir(parents=True) + (workspace / "STRAY.md").write_text("left behind by a killed run\n") + + assert _publisher(vault).process(receipt.id) == "active" + + assert not (vault / "STRAY.md").exists() + assert "new body" in (skill / "SKILL.md").read_text() + + +class _AdvanceFails(W.LocalBackend): + """A crash after the branch is committed and before the served checkout moves.""" + + def advance(self, repo, record, commit): + raise RuntimeError("killed between the commit and the fast-forward") + + +class _AdvanceThenCrash(W.LocalBackend): + """A crash after the served checkout moves and before the receipt is written.""" + + def advance(self, repo, record, commit): + super().advance(repo, record, commit) + raise RuntimeError("killed after the fast-forward") + + +def test_a_crash_before_the_fast_forward_resumes_from_the_committed_branch(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + with pytest.raises(RuntimeError, match="killed between"): + W.Publisher(vault, backend=_AdvanceFails()).process(receipt.id) + + record = Q.load_publication(receipt.id) + assert record["state"] == "publishing" + assert "old body" in (skill / "SKILL.md").read_text() + assert _git(vault, "rev-parse", f"refs/heads/ingot/{receipt.id}") == record["branch_commit"] + + assert _publisher(vault).process(receipt.id) == "active" + assert "new body" in (skill / "SKILL.md").read_text() + assert not P.pending_path("pdf").exists() + + +def test_a_crash_after_the_fast_forward_finalizes_instead_of_resnapshotting(tmp_path, monkeypatch): + """The served bytes already equal the candidate, so the champion this would snapshot is gone. + Re-snapshotting would refuse a publication that has in fact already activated.""" + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + with pytest.raises(RuntimeError, match="killed after"): + W.Publisher(vault, backend=_AdvanceThenCrash()).process(receipt.id) + + assert "new body" in (skill / "SKILL.md").read_text() # the fast-forward survived + assert Q.load_publication(receipt.id)["state"] == "publishing" + assert P.pending_path("pdf").exists() # but nothing was finalized + + assert _publisher(vault).process(receipt.id) == "active" + assert not P.pending_path("pdf").exists() + + +def test_an_active_receipt_is_never_reprocessed(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + publisher = _publisher(vault) + assert publisher.process(receipt.id) == "active" + head = _git(vault, "rev-parse", "HEAD") + + assert publisher.process(receipt.id) == "active" + assert _git(vault, "rev-parse", "HEAD") == head + + +# --------------------------------------------------------------------------- divergence + +def test_advance_refuses_a_branch_that_no_longer_fast_forwards(tmp_path, monkeypatch): + """Never a rebase, a merge commit, or a reset. The vault has other legitimate writers, and a + publisher that forces past one of them is no longer the only writer of what it serves.""" + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + with pytest.raises(RuntimeError, match="killed between"): + W.Publisher(vault, backend=_AdvanceFails()).process(receipt.id) + branch = Q.load_publication(receipt.id)["branch"] + (vault / "NOTES.md").write_text("someone committed to the vault directly\n") + _git(vault, "add", "NOTES.md") + _git(vault, "commit", "-m", "A direct vault commit") + head = _git(vault, "rev-parse", "HEAD") + + with pytest.raises(RuntimeError, match="git merge"): + W.LocalBackend().advance(W.VaultRepo(vault), {"id": receipt.id}, branch) + + assert _git(vault, "rev-parse", "HEAD") == head # nothing moved + assert "old body" in (skill / "SKILL.md").read_text() + + +def test_a_publication_branch_left_behind_by_a_direct_vault_commit_is_recut(tmp_path, monkeypatch): + """The receipt retries rather than wedging forever: the stale branch is abandoned, the + publication is re-materialized against the vault as it now stands, and the unrelated commit + survives.""" + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + with pytest.raises(RuntimeError, match="killed between"): + W.Publisher(vault, backend=_AdvanceFails()).process(receipt.id) + (vault / "NOTES.md").write_text("someone committed to the vault directly\n") + _git(vault, "add", "NOTES.md") + _git(vault, "commit", "-m", "A direct vault commit") + + assert _publisher(vault).process(receipt.id) == "active" + + assert "new body" in (skill / "SKILL.md").read_text() + assert (vault / "NOTES.md").exists() + + +def test_a_champion_changed_out_of_band_is_refused_not_overwritten(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + receipt = _queue(skill) + (skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\nedited\n") + _git(vault, "add", "pdf") + _git(vault, "commit", "-m", "An out-of-band edit to the champion") + + with pytest.raises(RuntimeError, match="champion does not match"): + _publisher(vault).process(receipt.id) + + assert "edited" in (skill / "SKILL.md").read_text() + assert P.pending_path("pdf").exists() + + +# --------------------------------------------------------------------------- configuration + +def test_the_backend_defaults_to_local_and_is_never_inferred_from_a_remote(tmp_path): + """A vault that gains an `origin` must not silently start opening pull requests.""" + config = W.load_config({"INGOT_VAULT_PATH": str(tmp_path)}) + assert config.backend == "local" + assert isinstance(config.build().backend, W.LocalBackend) + + +def test_a_missing_vault_path_is_an_error_not_the_demo_directory(tmp_path): + with pytest.raises(W.ConfigurationError, match="no vault configured"): + W.load_config({}) + + +def test_an_unknown_backend_is_refused_by_name(tmp_path): + with pytest.raises(W.ConfigurationError, match="unknown publication backend"): + W.load_config({"INGOT_VAULT_PATH": str(tmp_path), "INGOT_PUBLISH_BACKEND": "gitlab"}) + + +def test_the_forge_backend_requires_a_repository(tmp_path): + with pytest.raises(W.ConfigurationError, match="INGOT_FORGE_REPOSITORY"): + W.load_config({"INGOT_VAULT_PATH": str(tmp_path), "INGOT_PUBLISH_BACKEND": "forge"}) + + +def test_forge_settings_under_the_local_backend_warn_rather_than_look_active(tmp_path): + config = W.load_config({"INGOT_VAULT_PATH": str(tmp_path), + "INGOT_FORGE_REPOSITORY": "someone/skills"}) + assert config.backend == "local" + assert any("inert" in warning for warning in config.warnings) + + +def test_an_explicit_argument_outranks_the_environment(tmp_path): + config = W.load_config({"INGOT_VAULT_PATH": str(tmp_path / "env"), + "INGOT_PUBLISH_BACKEND": "forge", + "INGOT_FORGE_REPOSITORY": "someone/skills"}, + backend="local", vault=tmp_path / "explicit") + assert config.backend == "local" + assert config.vault_dir == tmp_path / "explicit" + + +def test_a_vault_without_a_validator_refuses_to_start(tmp_path, monkeypatch): + """Every publication runs it before committing, so a missing one is a configuration error and + not a silent skip.""" + vault, _ = _vault(tmp_path, monkeypatch) + (vault / "scripts" / "validate.py").unlink() + _git(vault, "commit", "-am", "Remove the validator") + + with pytest.raises(W.ConfigurationError, match="no validator"): + W.validate(W.load_config({"INGOT_VAULT_PATH": str(vault)})) + + +@pytest.mark.parametrize("fault", ["dirty", "detached", "missing"]) +def test_startup_refuses_a_vault_the_publisher_must_not_build_on(tmp_path, monkeypatch, fault): + vault, _ = _vault(tmp_path, monkeypatch) + if fault == "dirty": + (vault / "dirty.txt").write_text("dirty") + elif fault == "detached": + _git(vault, "checkout", "--detach") + else: + vault = tmp_path / "nowhere" + + with pytest.raises(W.ConfigurationError): + W.validate(W.load_config({"INGOT_VAULT_PATH": str(vault)})) + + +def test_a_valid_local_vault_starts(tmp_path, monkeypatch): + vault, _ = _vault(tmp_path, monkeypatch) + publisher = W.validate(W.load_config({"INGOT_VAULT_PATH": str(vault)})) + assert publisher.vault_dir == Path(vault).resolve() + assert publisher.backend.name == "local" + + +# ------------------------------------------------------------------ artifact fidelity, end to end + +PNG = b"\x89PNG\r\n\x1a\n" + bytes(range(256)) + b"\xff\xfe\xfd" + + +def _ingested(tmp_path, monkeypatch, vault): + """A real `ingot add` of a package carrying bytes no text component could hold.""" + from ingot import admission + + monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs")) + package = tmp_path / "src" / "csv-tidy" + (package / "assets").mkdir(parents=True) + (package / "SKILL.md").write_text( + "---\nname: csv-tidy\ndescription: Tidy CSV files.\n---\n\nUse this to tidy CSVs.\n", + encoding="utf-8") + (package / "assets" / "logo.png").write_bytes(PNG) + (package / "run.sh").write_text("#!/bin/sh\necho tidy\n", encoding="utf-8") + (package / "run.sh").chmod(0o755) + admission.add_package(package, actor="operator") + return package + + +def test_an_ingested_binary_asset_reaches_the_vault_byte_for_byte(tmp_path, monkeypatch): + """The end of the chain the whole change exists for. Admission used to reduce a package to + decoded text, so this file was reviewed as part of the candidate, approved as part of the + candidate, and then was not in the vault -- with no error anywhere along the way.""" + vault, _ = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + _ingested(tmp_path, monkeypatch, vault) + receipt = Q.queue_publication("csv-tidy", P.load_pending("csv-tidy"), "admin", "promote") + + assert _publisher(vault).process(receipt.id) == "active" + + published = vault / "csv-tidy" / "assets" / "logo.png" + assert published.read_bytes() == PNG + assert _git(vault, "status", "--porcelain") == "" + assert skill_revision(vault / "csv-tidy") == Q.load_publication(receipt.id)["candidate_revision"] + + +def test_the_published_executable_bit_survives_the_vault_commit(tmp_path, monkeypatch): + vault, _ = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + _ingested(tmp_path, monkeypatch, vault) + receipt = Q.queue_publication("csv-tidy", P.load_pending("csv-tidy"), "admin", "promote") + + _publisher(vault).process(receipt.id) + + entry = _git(vault, "ls-files", "-s", "csv-tidy/run.sh") + assert entry.split()[0] == "100755" + + +def test_a_staged_asset_altered_after_approval_stops_the_publication(tmp_path, monkeypatch): + """The receipt is the authority for what gets served. If the staged bytes moved between the + approval and the publication, the publisher must refuse rather than publish what it finds.""" + from ingot.optimize import tree + + vault, _ = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + _ingested(tmp_path, monkeypatch, vault) + pending = P.load_pending("csv-tidy") + receipt = Q.queue_publication("csv-tidy", pending, "admin", "promote") + staged = tree.staged_dir(pending["tree"]["digest"]) + (staged / "assets" / "logo.png").write_bytes(b"substituted") + + with pytest.raises(RuntimeError, match="does not match the receipt: assets/logo.png"): + _publisher(vault).process(receipt.id) + + assert not (vault / "csv-tidy").exists() + assert _git(vault, "rev-parse", "--abbrev-ref", "HEAD") == "main" + assert "does not match the receipt" in Q.load_publication(receipt.id)["last_error"] + + +# --------------------------------------------------------------------------- delivery targets + +def _delivering(vault, tmp_path, name="claude"): + """A publisher that serves the managed vault and one native filesystem root beside it.""" + native = tmp_path / name + targets = D.parse_targets(f"{name}=filesystem:{native}", vault=vault) + return W.Publisher(vault, backend=W.LocalBackend(), targets=targets), native + + +def test_an_approved_revision_reaches_every_target(tmp_path, monkeypatch): + """The whole point: the vault and a native skill root end up holding the same approved bytes, + from one approval, through one publisher.""" + vault, skill = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + publisher, native = _delivering(vault, tmp_path) + receipt = _queue(skill) + + assert publisher.process(receipt.id) == "active" + + assert skill_revision(native / "pdf") == skill_revision(skill) + assert (native / "pdf" / "SKILL.md").read_bytes() == (skill / "SKILL.md").read_bytes() + + +def test_each_target_is_recorded_on_the_receipt_separately(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + publisher, native = _delivering(vault, tmp_path) + publisher.process(_queue(skill).id) + + delivered = Q.publication_for_skill("pdf")["delivery"] + assert delivered["vault"]["kind"] == D.MANAGED_MCP + assert delivered["claude"] == {"kind": D.FILESYSTEM, "root": str(native), + "state": "delivered", "revision": skill_revision(skill), + "at": delivered["claude"]["at"]} + + +def test_only_the_altered_target_reports_drift(tmp_path, monkeypatch): + """Independent status. Editing one target must not make the other one look wrong.""" + vault, skill = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + publisher, native = _delivering(vault, tmp_path) + publisher.process(_queue(skill).id) + released = skill_revision(skill) + + (native / "pdf" / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\nedited\n") + + observed = {target.name: D.observed(target, "pdf") for target in publisher.targets} + assert observed["vault"] == released + assert observed["claude"] != released + + +def test_a_rollback_returns_every_target_to_the_prior_revision(tmp_path, monkeypatch): + """Rollback travels the ordinary publication queue, so delivery happens on the way through + rather than needing a second mechanism that could disagree with the first.""" + vault, skill = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + publisher, native = _delivering(vault, tmp_path) + champion = load_skills(vault)[0].revision + publisher.process(_queue(skill).id) + assert skill_revision(native / "pdf") != champion + + P.rollback("pdf", champion, actor="admin") + assert publisher.process(Q.publication_for_skill("pdf")["id"]) == "active" + + assert skill_revision(skill) == champion + assert skill_revision(native / "pdf") == champion + + +def test_a_rollback_to_absence_removes_the_skill_from_every_target(tmp_path, monkeypatch): + vault, skill = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + publisher, native = _delivering(vault, tmp_path) + publisher.process(_queue(skill).id) + assert (native / "pdf").is_dir() + + P._snapshot_absence("pdf") + P.rollback("pdf", P.ABSENT_REVISION, actor="admin") + assert publisher.process(Q.publication_for_skill("pdf")["id"]) == "active" + + assert not skill.exists() + assert not (native / "pdf").exists() + + +def _refuse_to_install(target, skill, source, revision): + if target.kind == D.FILESYSTEM: + raise OSError("the target is unavailable") + return False + + +def test_a_failed_delivery_does_not_leave_an_active_release(tmp_path, monkeypatch): + """The release is finished when every target holds it, not when the vault does. Marking it + active on a partial delivery would report a change as live in places it never reached.""" + vault, skill = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + publisher, native = _delivering(vault, tmp_path) + receipt = _queue(skill) + monkeypatch.setattr(D, "install", _refuse_to_install) + + with pytest.raises(RuntimeError, match="the target is unavailable"): + publisher.process(receipt.id) + + record = Q.load_publication(receipt.id) + assert record["state"] != "active" + assert "the target is unavailable" in record["last_error"] + assert record["delivery"]["claude"]["state"] == "failed" + assert not (native / "pdf").exists() + + +def test_a_retried_publication_finishes_the_delivery_it_could_not_complete(tmp_path, monkeypatch): + """The vault has already advanced by then, so the retry must deliver rather than decide the + release is finished because the vault looks right.""" + vault, skill = _vault(tmp_path, monkeypatch) + _forbid_network(monkeypatch) + publisher, native = _delivering(vault, tmp_path) + receipt = _queue(skill) + real_install, refused = D.install, [] + + def fail_the_first_delivery(target, name, source, revision): + if target.kind == D.FILESYSTEM and not refused: + refused.append(target.name) + raise OSError("the target is unavailable") + return real_install(target, name, source, revision) + + monkeypatch.setattr(D, "install", fail_the_first_delivery) + with pytest.raises(RuntimeError): + publisher.process(receipt.id) + assert skill_revision(skill) == Q.load_publication(receipt.id)["candidate_revision"] + + assert publisher.process(receipt.id) == "active" + assert skill_revision(native / "pdf") == skill_revision(skill) + assert Q.load_publication(receipt.id)["delivery"]["claude"]["state"] == "delivered" diff --git a/tests/test_records.py b/tests/test_records.py new file mode 100644 index 0000000..9aa1fd9 --- /dev/null +++ b/tests/test_records.py @@ -0,0 +1,251 @@ +"""Candidate manifests and release receipts. + +Two versioned records with one job each: a candidate names exactly what is being proposed and where +it came from; a release names exactly what was published and proves it happened. Both are consumed +by later PRs -- the candidate by `ingot add`, the receipt by the publisher -- so the shape is fixed +here and tested here.""" +import json + +import pytest + +from ingot import records + +REVISION = "a" * 64 +OTHER_REVISION = "b" * 64 + + +def _review(errors=(), warnings=()): + return {"schema_version": "ingot/review/v1", + "valid": not errors, + "errors": list(errors), + "warnings": list(warnings)} + + +def _candidate(**overrides): + fields = {"kind": "creation", + "skill": "pdf", + "source_type": "file", + "locator": "./packages/pdf", + "resolved_revision": REVISION, + "candidate_revision": REVISION, + "review": _review(), + "created_at": 1_770_000_000} + fields.update(overrides) + return records.candidate_manifest(**fields) + + +def _receipt(**overrides): + fields = {"skill": "pdf", + "action": "promote", + "proposal_id": "p-1", + "publication_id": "pub-1", + "expected_champion": OTHER_REVISION, + "candidate_revision": REVISION, + "evidence_digests": [REVISION], + "actor": "operator", + "publisher": "local", + "target": "managed-library", + "published_at": 1_770_000_100, + "result": "published"} + fields.update(overrides) + return records.release_receipt(**fields) + + +# --- digests -------------------------------------------------------------------------------- + +def test_digest_ignores_key_order(): + assert records.digest({"a": 1, "b": 2}) == records.digest({"b": 2, "a": 1}) + + +def test_digest_changes_with_content(): + assert records.digest({"a": 1}) != records.digest({"a": 2}) + + +def test_digest_is_a_sha256_hex(): + assert len(records.digest({"a": 1})) == 64 + + +# --- candidate manifests -------------------------------------------------------------------- + +def test_candidate_manifest_carries_its_schema_version(): + assert _candidate()["schema_version"] == records.CANDIDATE_SCHEMA + + +def test_candidate_manifest_records_the_source_it_resolved(): + manifest = _candidate() + + assert manifest["source"] == {"type": "file", + "locator": "./packages/pdf", + "resolved_revision": REVISION} + + +def test_candidate_manifest_references_the_review_by_digest(): + """Included *and* digested: the report travels with the proposal, and the digest is what binds + it, so a report edited after the fact stops matching.""" + review = _review(warnings=["file-reference-missing"]) + manifest = _candidate(review=review) + + assert manifest["review"]["digest"] == records.digest(review) + assert manifest["review"]["warnings"] == ["file-reference-missing"] + + +def test_candidate_identity_ignores_the_timestamp(): + """Deterministic apart from timestamps and actor metadata: proposing the same bytes twice must + produce the same identity, or idempotent submission is impossible.""" + first = _candidate(created_at=1_770_000_000) + second = _candidate(created_at=1_999_999_999) + + assert records.candidate_identity(first) == records.candidate_identity(second) + + +def test_candidate_identity_ignores_the_local_path(): + """A local path is operator context. The same package submitted from two checkouts is the same + candidate.""" + first = _candidate(locator="/home/a/pdf") + second = _candidate(locator="/home/b/pdf") + + assert records.candidate_identity(first) == records.candidate_identity(second) + + +def test_file_candidate_identity_keeps_the_bound_review_report(): + """Break caught: changing identity for already-quarantined file candidates during upgrade.""" + first = _candidate(review={**_review(), "report_digest": REVISION}) + second = _candidate(review={**_review(), "report_digest": OTHER_REVISION}) + + assert records.candidate_identity(first) != records.candidate_identity(second) + + +def test_candidate_identity_changes_with_the_candidate_revision(): + assert records.candidate_identity(_candidate()) != \ + records.candidate_identity(_candidate(candidate_revision=OTHER_REVISION)) + + +def test_candidate_identity_changes_with_the_skill(): + assert records.candidate_identity(_candidate()) != \ + records.candidate_identity(_candidate(skill="docx")) + + +def test_candidate_identity_changes_with_the_review_outcome(): + """Evidence is revision-bound, and a review is evidence. The same bytes reviewed clean and + reviewed with errors are not interchangeable proposals.""" + assert records.candidate_identity(_candidate()) != \ + records.candidate_identity(_candidate(review=_review(errors=["description-empty"]))) + + +def test_a_well_formed_candidate_validates(): + assert records.validate_candidate(_candidate()) == [] + + +def test_a_candidate_with_the_wrong_schema_is_rejected(): + manifest = _candidate() + manifest["schema_version"] = "ingot/candidate/v99" + + assert any("schema" in problem for problem in records.validate_candidate(manifest)) + + +def test_a_candidate_missing_a_field_is_rejected(): + manifest = _candidate() + del manifest["candidate_revision"] + + assert any("candidate_revision" in problem for problem in records.validate_candidate(manifest)) + + +def test_a_candidate_with_a_semantic_version_source_is_rejected(): + """The resolved source revision must be content-based. A tag can be moved; a digest cannot.""" + problems = records.validate_candidate(_candidate(resolved_revision="v1.2.3")) + + assert any("resolved_revision" in problem for problem in problems) + + +def test_a_candidate_with_an_unknown_kind_is_rejected(): + assert any("kind" in problem for problem in records.validate_candidate(_candidate(kind="mutate"))) + + +def test_a_candidate_with_an_invalid_skill_slug_is_rejected(): + assert any("skill" in problem for problem in records.validate_candidate(_candidate(skill="Not_A_Slug"))) + + +# --- release receipts ----------------------------------------------------------------------- + +def test_release_receipt_carries_its_schema_version(): + assert _receipt()["schema_version"] == records.RELEASE_SCHEMA + + +def test_a_well_formed_receipt_validates(): + assert records.validate_release(_receipt()) == [] + + +def test_absence_is_a_valid_revision_on_both_sides(): + """A creation displaces nothing and a rollback can restore nothing. Absence is a revision, and + the existing publisher already treats it as one.""" + assert records.validate_release(_receipt(expected_champion=records.ABSENT_REVISION)) == [] + assert records.validate_release(_receipt(action="rollback", + candidate_revision=records.ABSENT_REVISION)) == [] + + +def test_a_rollback_produces_a_receipt(): + assert records.validate_release(_receipt(action="rollback")) == [] + + +def test_an_unknown_action_is_rejected(): + assert any("action" in problem for problem in records.validate_release(_receipt(action="delete"))) + + +def test_a_failed_receipt_stays_inspectable(): + """A failure that erased its own reason would leave an operator with a stalled lane and no + way to learn why.""" + receipt = records.release_receipt( + skill="pdf", action="promote", proposal_id="p-1", publication_id="pub-1", + expected_champion=OTHER_REVISION, candidate_revision=REVISION, evidence_digests=[], + actor="operator", publisher="local", target="managed-library", + published_at=1_770_000_100, result="failed", error="vault champion did not match") + + assert records.validate_release(receipt) == [] + assert receipt["result"] == "failed" + assert receipt["error"] == "vault champion did not match" + + +def test_a_published_receipt_may_not_carry_an_error(): + receipt = _receipt() + receipt["error"] = "something went wrong" + + assert any("error" in problem for problem in records.validate_release(receipt)) + + +def test_a_failed_receipt_must_say_why(): + receipt = _receipt(result="failed") + + assert any("error" in problem for problem in records.validate_release(receipt)) + + +def test_an_unknown_result_is_rejected(): + assert any("result" in problem for problem in records.validate_release(_receipt(result="queued"))) + + +def test_a_receipt_is_not_signed_and_claims_nothing_about_tampering(): + """No signature field, on purpose. A local record a machine administrator can rewrite must not + carry anything that looks like proof it was not.""" + receipt = _receipt() + + assert "signature" not in receipt + assert "attestation" not in receipt + + +# --- round trips ---------------------------------------------------------------------------- + +def test_both_records_survive_a_json_round_trip(): + for record in (_candidate(), _receipt()): + assert json.loads(json.dumps(record)) == record + + +def test_records_import_without_the_heavy_stack(): + import subprocess + import sys + + heavy = ["fastapi", "onnxruntime", "langgraph", "langfuse", "litellm", "fastembed"] + program = f"import sys, ingot.records; print([m for m in {heavy!r} if m in sys.modules])" + + result = subprocess.run([sys.executable, "-c", program], capture_output=True, text=True) + + assert result.returncode == 0, result.stderr + assert result.stdout.strip() == "[]" diff --git a/tests/test_registry.py b/tests/test_registry.py index 94f4fba..3d17bab 100644 --- a/tests/test_registry.py +++ b/tests/test_registry.py @@ -1,10 +1,12 @@ -"""Unit tests for skill discovery / frontmatter parsing edge cases (mcp_server.registry).""" +"""Unit tests for skill discovery / frontmatter parsing edge cases (ingot.mcp_server.registry).""" import os +from pathlib import Path import pytest -from mcp_server.registry import ( - configured_roots, load_skills, optimizable_components, parse_skill, write_skill_md, +from ingot.mcp_server.registry import ( + configured_roots, load_skills, optimizable_components, parse_skill, writable_skill_dir, + write_skill_md, ) @@ -73,7 +75,7 @@ def test_configured_roots_reads_platform_path_separator(tmp_path, monkeypatch): a, b, local = tmp_path / "a", tmp_path / "b", tmp_path / "local" a.mkdir(); b.mkdir(); local.mkdir() monkeypatch.setenv("SKILL_ROUTER_PATHS", os.pathsep.join([str(a), str(b), str(a)])) - monkeypatch.setattr("mcp_server.registry.SKILLS_DIR", local) + monkeypatch.setenv("INGOT_LIBRARY", str(local)) assert configured_roots() == [local.resolve(), a.resolve(), b.resolve()] @@ -81,7 +83,7 @@ def test_explicit_roots_override_environment(tmp_path, monkeypatch): env_root, explicit, local = tmp_path / "env", tmp_path / "explicit", tmp_path / "local" env_root.mkdir(); explicit.mkdir(); local.mkdir() monkeypatch.setenv("SKILL_ROUTER_PATHS", str(env_root)) - monkeypatch.setattr("mcp_server.registry.SKILLS_DIR", local) + monkeypatch.setenv("INGOT_LIBRARY", str(local)) assert configured_roots([explicit]) == [local.resolve(), explicit.resolve()] @@ -89,10 +91,17 @@ def test_environment_roots_keep_local_authoring_root(tmp_path, monkeypatch): external, local = tmp_path / "external", tmp_path / "local" external.mkdir(); local.mkdir() monkeypatch.setenv("SKILL_ROUTER_PATHS", str(external)) - monkeypatch.setattr("mcp_server.registry.SKILLS_DIR", local) + monkeypatch.setenv("INGOT_LIBRARY", str(local)) assert configured_roots() == [local.resolve(), external.resolve()] +def test_writable_skill_dir_expands_user_root(tmp_path, monkeypatch): + monkeypatch.setenv("HOME", str(tmp_path)) + monkeypatch.setenv("INGOT_LIBRARY", str(Path("~/skills"))) + + assert writable_skill_dir("sample") == tmp_path / "skills" / "sample" + + def test_load_skills_uses_declared_root_precedence_with_warning(tmp_path): a, b = tmp_path / "a", tmp_path / "b" _skill(a, body="first"); _skill(b, dirname="other", name="sample", body="second") @@ -192,7 +201,7 @@ def test_leftover_staging_directory_is_not_published_as_its_own_skill(tmp_path): @pytest.mark.parametrize("suffix", ["stage", "previous", "rollback"]) def test_skill_sources_skips_every_staging_suffix(tmp_path, suffix): - from mcp_server.registry import skill_sources + from ingot.mcp_server.registry import skill_sources _live_skill(tmp_path) _hidden_stage(tmp_path, "pdf", "abandoned body", suffix=suffix) assert [p.parent.name for p in skill_sources(tmp_path)] == ["pdf"] diff --git a/tests/test_retrospective.py b/tests/test_retrospective.py new file mode 100644 index 0000000..6b2b4e2 --- /dev/null +++ b/tests/test_retrospective.py @@ -0,0 +1,165 @@ +import json + +import pytest + +from ingot.mcp_server.registry import load_skills +from ingot.optimize import promote as P +from ingot.optimize import retrospective as R + + +def _library(tmp_path, monkeypatch): + root = tmp_path / "skills" + skill = root / "pdf" + skill.mkdir(parents=True) + (skill / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Merge PDFs.\n---\nold body\n") + monkeypatch.setenv("INGOT_LIBRARY", str(root)) + monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs")) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root)) + return root, skill + + +def _proposal(root, **overrides): + current = load_skills(root)[0] + data = { + "skill": "pdf", + "champion_revision": current.revision, + "challenger_body": "new body with a reusable guard", + "challenger_description": "", + "summary": "Preserve the recovery check after repeated omissions.", + "trigger": "Use when the same recovery step fails twice.", + "minimal_content": "Require the recovery check before completion.", + "producer": "skill-retrospective", + "caller": "build-loop after repeat evidence", + "evidence": ["Run A skipped the check.", "Run B repeated the same omission."], + "pressure_scenario": "A rushed repair reaches completion without checking recovery.", + "risk": "The extra gate may slow low-risk repairs.", + "verification_status": "passed", + "verification_command": "pytest tests/scenarios.md -k recovery", + "verification_result": "pressure scenario passed", + } + data.update(overrides) + return data + + +def test_submit_quarantines_a_revision_bound_retrospective(tmp_path, monkeypatch): + root, _ = _library(tmp_path, monkeypatch) + + result = R.submit_skill_update(**_proposal(root)) + + pending = P.load_pending("pdf") + assert result == { + "status": "quarantined", "skill": "pdf", + "proposal_id": pending["retrospective"]["proposal_id"], "promotable": True} + assert pending["kind"] == "retrospective" + assert pending["challenger_components"]["description"] == "Merge PDFs." + assert pending["challenger_components"]["body"] == "new body with a reusable guard" + assert pending["changed_components"] == ["body"] + assert pending["gate"] == { + "promotable": True, + "blocked": [], + "warnings": ["Retrospective evidence only; no held-out A/B comparison was run."], + "kind": "retrospective_admission", + "admission": {"pressure_verification": "passed", "evidence_items": 2}, + } + assert pending["evidence"]["champion"]["revision"] == _proposal(root)["champion_revision"] + assert pending["evidence"]["challenger"]["revision"] + assert pending["retrospective"]["evidence"] == [ + "Run A skipped the check.", "Run B repeated the same omission."] + paths = pending["evidence_paths"] + assert (tmp_path / paths["json"]).exists() + markdown = (tmp_path / paths["markdown"]).read_text() + assert "Retrospective proposal: pdf" in markdown + assert "pressure scenario passed" in markdown + audit = json.loads((tmp_path / "runs" / "retrospective-audit.jsonl").read_text()) + assert audit["action"] == "quarantine" and audit["proposal_id"] == result["proposal_id"] + + +def test_metadata_retry_is_idempotent_and_different_pending_is_preserved(tmp_path, monkeypatch): + root, _ = _library(tmp_path, monkeypatch) + proposal = _proposal(root) + now = [100] + monkeypatch.setattr(R.time, "time", lambda: now[0]) + first = R.submit_skill_update(**proposal) + before = P.pending_path("pdf").read_text() + now[0] = 101 + + duplicate = R.submit_skill_update(**{**proposal, "caller": "same retry from a new session"}) + + assert duplicate == {**first, "status": "duplicate"} + assert P.pending_path("pdf").read_text() == before + + with pytest.raises(ValueError, match="review slot is occupied"): + R.submit_skill_update(**{**proposal, "challenger_body": "different candidate"}) + assert P.pending_path("pdf").read_text() == before + + +def test_concurrent_writer_cannot_be_displaced_between_check_and_publish(tmp_path, monkeypatch): + root, _ = _library(tmp_path, monkeypatch) + write_evidence = R._write_evidence + + def collide(skill, proposal, gate): + paths = write_evidence(skill, proposal, gate) + P.save_pending("pdf", {"skill": "pdf", "kind": "quality", "created": 7}) + return paths + + monkeypatch.setattr(R, "_write_evidence", collide) + + with pytest.raises(ValueError, match="review slot is occupied"): + R.submit_skill_update(**_proposal(root)) + + assert P.load_pending("pdf") == {"skill": "pdf", "kind": "quality", "created": 7} + assert not list(R.evidence_dir().rglob("retrospective-*")) + + +@pytest.mark.parametrize(("change", "message"), [ + ({"champion_revision": "stale"}, "champion revision"), + ({"skill": "missing"}, "no indexed skill"), + ({"evidence": []}, "evidence"), + ({"evidence": ["only one occurrence"]}, "at least two"), + ({"evidence": ["same occurrence", "same occurrence"]}, "must be distinct"), + ({"verification_status": "maybe"}, "verification_status must be passed"), + ({"verification_status": "failed"}, "verification_status must be passed"), + ({"verification_status": "unavailable"}, "verification_status must be passed"), + ({"challenger_body": "x" * 200_001}, "challenger_body"), +]) +def test_invalid_proposals_fail_before_mutation(tmp_path, monkeypatch, change, message): + root, _ = _library(tmp_path, monkeypatch) + with pytest.raises(ValueError, match=message): + R.submit_skill_update(**_proposal(root, **change)) + assert not P.pending_dir().exists() + assert not R.evidence_dir().exists() + + +def test_atomic_publication_failure_cleans_evidence(tmp_path, monkeypatch): + root, _ = _library(tmp_path, monkeypatch) + + def unsupported_link(_source, _destination): + raise OSError("hard links unavailable") + + monkeypatch.setattr(R.os, "link", unsupported_link) + with pytest.raises(RuntimeError, match="cannot atomically publish"): + R.submit_skill_update(**_proposal(root)) + + assert not P.pending_path("pdf").exists() + assert not list(R.evidence_dir().rglob("retrospective-*")) + assert not R.audit_file().exists() + + +def test_mcp_producer_reaches_existing_approval_and_rollback_path(tmp_path, monkeypatch): + root, skill = _library(tmp_path, monkeypatch) + from ingot.mcp_server import server + server.STATE.reload([root]) + + result = server.propose_skill_update(**_proposal(root)) + old_revision = _proposal(root)["champion_revision"] + promoted = P._activate_approved("pdf", P.load_pending("pdf"), actor="retrospective-test") + + assert result["status"] == "quarantined" + assert "Promoted 'pdf'" in promoted + assert "new body with a reusable guard" in (skill / "SKILL.md").read_text() + assert "old body" in ( + P.revisions_dir() / "pdf" / old_revision / "SKILL.md").read_text() + + P._activate_rollback("pdf", old_revision, actor="retrospective-test") + assert "old body" in (skill / "SKILL.md").read_text() diff --git a/tests/test_review.py b/tests/test_review.py new file mode 100644 index 0000000..aad105c --- /dev/null +++ b/tests/test_review.py @@ -0,0 +1,184 @@ +"""Unit tests for the standalone per-skill review (LLM + judge mocked).""" +import json + +import pytest + +from ingot.optimize import review as R + + +def _result(task, checklist, spec): + return {"task": task, "score": 0.0, "answer": "a", "feedback": "f", + "checklist": checklist, "spec": spec} + + +SPEC = { + "cites_source": {"id": "cites_source", "criterion": "Names its source.", "weight": 5, + "dimension": "correctness"}, + "is_terse": {"id": "is_terse", "criterion": "No padding.", "weight": 1, + "dimension": "efficiency"}, +} + + +def test_findings_rank_by_cost_not_by_raw_score(): + """A heavy check scraping a partial outranks a trivial check failing outright. Sorting on the + verdict value alone puts the weight-1 failure first and buries the thing worth fixing.""" + results = [_result("t", {"cites_source": {"value": 0.5, "note": "no source"}, + "is_terse": {"value": 0.0, "note": "padded"}}, SPEC)] + ranked = R.findings(results) + assert [f["check"] for f in ranked] == ["cites_source", "is_terse"] + assert ranked[0]["cost"] == pytest.approx(2.5) and ranked[1]["cost"] == pytest.approx(1.0) + + +def test_findings_omit_clean_passes(): + results = [_result("t", {"cites_source": {"value": 1.0, "note": ""}, + "is_terse": {"value": 0.0, "note": "padded"}}, SPEC)] + assert [f["check"] for f in R.findings(results)] == ["is_terse"] + + +def test_findings_default_weight_when_a_task_declared_no_spec(): + """A task with no checklist is graded on the default four, whose specs are not in `spec`. + Those failures still have to appear, at weight 1, rather than vanish.""" + results = [_result("t", {"correctness": {"value": 0.0, "note": "wrong"}}, {})] + found = R.findings(results) + assert len(found) == 1 and found[0]["weight"] == 1 and found[0]["cost"] == pytest.approx(1.0) + + +def test_by_dimension_concentrates_losses(): + results = [_result("t", {"cites_source": {"value": 0.0, "note": "n"}, + "is_terse": {"value": 0.5, "note": "n"}}, SPEC)] + assert R.by_dimension(R.findings(results)) == {"correctness": 5.0, "efficiency": 0.5} + + +def test_by_dimension_drops_dimensions_with_no_losses(): + results = [_result("t", {"is_terse": {"value": 0.0, "note": "n"}}, SPEC)] + assert R.by_dimension(R.findings(results)) == {"efficiency": 1.0} + + +def test_run_review_scores_grades_and_writes_a_report(tmp_path, monkeypatch): + from ingot.mcp_server.registry import skill_revision + tasks = [{"task": "t1", "rubric": "r", "checklist": [SPEC["cites_source"]]}, + {"task": "t2", "rubric": "r", "checklist": [SPEC["is_terse"]]}] + monkeypatch.setattr(R, "resolve_skill_dir", lambda name: tmp_path) + monkeypatch.setattr(R, "load_tasks", lambda skill, **_: (tasks[:1], tasks[1:], {})) + monkeypatch.setattr(R, "optimizable_components", lambda d: {"description": "d", "body": "B"}) + monkeypatch.setattr(R, "assemble", lambda c: c["body"]) + monkeypatch.setattr(R, "_llm", lambda model: "llm") + monkeypatch.setattr(R, "invoke_retry", + lambda llm, msgs: type("M", (), {"content": "ans", "usage_metadata": None})()) + monkeypatch.setattr(R, "judge", lambda task, rubric, answer, **kw: { + "score": 0.5, "feedback": "f", + "checklist": {c["id"]: {"value": 0.5, "note": "half"} for c in kw["checklist"]}}) + monkeypatch.setattr(R, "REVIEW_DIR", tmp_path / "reviews") + expected_revision = skill_revision(tmp_path) + + out = R.run_review("sk", log=lambda *a: None) + assert out["score"] == pytest.approx(0.5) + assert out["tasks"] == 2 and out["failed_checks"] == 2 + written = json.loads((tmp_path / "reviews" / "sk.json").read_text()) + assert written["skill"] == "sk" + assert written["revision"] == expected_revision + assert [f["check"] for f in written["findings"]] == ["cites_source", "is_terse"] # by cost + + +def test_run_review_grades_train_and_holdout_together(tmp_path, monkeypatch): + """A review is not measuring generalization, so holding half the tasks back would only make it + a noisier read on the same skill.""" + seen = [] + monkeypatch.setattr(R, "resolve_skill_dir", lambda name: tmp_path) + monkeypatch.setattr(R, "load_tasks", + lambda skill, **_: ([{"task": "train", "rubric": "r"}], + [{"task": "holdout", "rubric": "r"}], {})) + monkeypatch.setattr(R, "optimizable_components", lambda d: {"description": "d", "body": "B"}) + monkeypatch.setattr(R, "assemble", lambda c: c["body"]) + monkeypatch.setattr(R, "_llm", lambda model: "llm") + monkeypatch.setattr(R, "invoke_retry", lambda llm, msgs: ( + seen.append(msgs[1][1]) or type("M", (), {"content": "a", "usage_metadata": None})())) + monkeypatch.setattr(R, "judge", lambda *a, **k: {"score": 1.0, "feedback": "", + "checklist": {}}) + monkeypatch.setattr(R, "REVIEW_DIR", tmp_path / "reviews") + R.run_review("sk", log=lambda *a: None) + assert sorted(seen) == ["holdout", "train"] + + +def test_run_review_hashes_and_grades_one_immutable_snapshot(tmp_path, monkeypatch): + from ingot.mcp_server.registry import skill_revision, write_skill_md + skill = tmp_path / "skills" / "sk" + skill.mkdir(parents=True) + write_skill_md(skill / "SKILL.md", {"name": "sk", "description": "old description"}, + "old body") + expected_revision = skill_revision(skill) + seen_systems = [] + + monkeypatch.setattr(R, "resolve_skill_dir", lambda name: skill) + + def mutate_then_load(name, **_): + write_skill_md(skill / "SKILL.md", {"name": "sk", "description": "new description"}, + "new body") + return ([{"task": "t", "rubric": "r"}], [], {}) + + monkeypatch.setattr(R, "load_tasks", mutate_then_load) + monkeypatch.setattr(R, "_llm", lambda model: "llm") + monkeypatch.setattr(R, "invoke_retry", lambda llm, msgs: ( + seen_systems.append(msgs[0][1]) or + type("M", (), {"content": "a", "usage_metadata": None})())) + monkeypatch.setattr(R, "judge", lambda *a, **k: { + "score": 1.0, "feedback": "", "checklist": {}}) + monkeypatch.setattr(R, "REVIEW_DIR", tmp_path / "reviews") + + result = R.run_review("sk", log=lambda *a: None) + + assert result["revision"] == expected_revision + assert seen_systems and "old body" in seen_systems[0] and "new body" not in seen_systems[0] + + +def test_run_review_drafts_missing_evals_from_the_reviewed_snapshot(tmp_path, monkeypatch): + from pathlib import Path + from ingot.mcp_server import registry + from ingot.mcp_server.registry import write_skill_md + from ingot.optimize import ab, draft + root = tmp_path / "skills" + skill = root / "sk" + skill.mkdir(parents=True) + write_skill_md(skill / "SKILL.md", {"name": "sk", "description": "old description"}, + "old body") + tasks = tmp_path / "tasks" + captured = {} + read_components = registry.read_components + + def mutate_before_live_read(path): + if Path(path).resolve() == skill.resolve(): + write_skill_md(skill / "SKILL.md", + {"name": "sk", "description": "new description"}, "new body") + return read_components(path) + + def draft_eval(name, description, body, out_dir, log=print): + captured.update(description=description, body=body) + out_dir.mkdir(parents=True, exist_ok=True) + (out_dir / f"{name}.yaml").write_text( + "train:\n- task: train\n rubric: r\nholdout:\n- task: holdout\n rubric: r\n") + + monkeypatch.setenv("INGOT_LIBRARY", str(root)) + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root)) + monkeypatch.setattr(ab, "TASKS_DIR", tasks) + monkeypatch.setattr(registry, "read_components", mutate_before_live_read) + monkeypatch.setattr(draft, "draft_and_save", draft_eval) + monkeypatch.setattr(R, "resolve_skill_dir", lambda name: skill) + monkeypatch.setattr(R, "_llm", lambda model: "llm") + monkeypatch.setattr(R, "invoke_retry", lambda *a, **k: type( + "M", (), {"content": "a", "usage_metadata": None})()) + monkeypatch.setattr(R, "judge", lambda *a, **k: { + "score": 1.0, "feedback": "", "checklist": {}}) + monkeypatch.setattr(R, "REVIEW_DIR", tmp_path / "reviews") + + R.run_review("sk", log=lambda *a: None) + + assert captured == {"description": "old description", "body": "old body"} + + +def test_run_review_refuses_a_skill_with_no_tasks(tmp_path, monkeypatch): + from ingot.mcp_server.registry import write_skill_md + write_skill_md(tmp_path / "SKILL.md", {"name": "sk", "description": "d"}, "body") + monkeypatch.setattr(R, "resolve_skill_dir", lambda name: tmp_path) + monkeypatch.setattr(R, "load_tasks", lambda skill, **_: ([], [], {})) + with pytest.raises(SystemExit, match="no eval tasks"): + R.run_review("sk", log=lambda *a: None) diff --git a/tests/test_router.py b/tests/test_router.py index 00c19e6..e0078ac 100644 --- a/tests/test_router.py +++ b/tests/test_router.py @@ -3,8 +3,8 @@ collision check all depend on.""" import pytest -from mcp_server.registry import Skill -from mcp_server.router import Router +from ingot.mcp_server.registry import Skill +from ingot.mcp_server.router import Router SKILLS = [ Skill("pdf", "Merge, split, and extract text from PDF files and documents.", "body", "p"), @@ -43,7 +43,7 @@ def test_nearest_empty_router_is_safe(): def test_router_reuses_description_vectors_across_refreshes(monkeypatch): import numpy as np - import mcp_server.router as router_mod + import ingot.mcp_server.router as router_mod calls = [] class FakeEmbedding: @@ -96,6 +96,8 @@ def test_route_returns_clean_no_match_below_threshold(): assert result["skill_body"] == "" and result["skill_root"] is None assert "threshold" in result["reason"] assert result["alternatives"][0]["name"] == "pdf" + assert result["matched_on"] in {"description", "content"} + assert result["score"] == max(result["score_components"].values()) def test_route_novel_flag_signals_weak_strong_escalation(): @@ -127,6 +129,7 @@ def test_route_novel_flag_signals_weak_strong_escalation(): ({"required_tools": ["browser"]}, {"available_tools": ["bash"]}), ({"required_mcps": ["github"]}, {"available_mcps": []}), ({"scopes": ["project"], "path_patterns": ["*/wanted/*"]}, {"cwd": "/tmp/other/project"}), + ({"scopes": ["project"], "path_patterns": ["/tmp/*"]}, {}), ]) def test_route_filters_incompatible_skills_before_ranking(skill_metadata, context): router = Router([_skill("blocked", "Merge PDF documents.", **skill_metadata)]) @@ -166,3 +169,131 @@ def test_conflicting_skills_do_not_both_appear_in_ranked_result(): result = Router([one, two]).route("same routing text", "codex", "/tmp", min_score=0.0) ranked = [result["match"], *[item["name"] for item in result["alternatives"]]] assert not ({"one", "two"} <= set(ranked)) + + +class _BodyAwareEmbedding: + """Deterministic vectors: descriptions/billing point east, Kubernetes content/query north.""" + + def __init__(self): + self.document_calls = [] + + @staticmethod + def _vector(text): + import numpy as np + if "CrashLoopBackOff" in text or "kubernetes pod" in text.lower(): + return np.array([0.0, 1.0], dtype=np.float32) + return np.array([1.0, 0.0], dtype=np.float32) + + def embed(self, texts): + values = list(texts) + self.document_calls.append(values) + return iter(self._vector(text) for text in values) + + def embed_query(self, texts): + return iter(self._vector(text) for text in texts) + + +def test_body_aware_route_breaks_an_ambiguous_description_tie(monkeypatch): + import ingot.mcp_server.router as router_mod + embedder = _BodyAwareEmbedding() + monkeypatch.setattr(router_mod, "build_embedding", lambda: embedder) + router_mod.Router._vector_cache.clear() + router = router_mod.Router([ + _skill("billing-runbook", "Operate a production service."), + Skill(**{**_skill("kubernetes-runbook", "Operate a production service.").__dict__, + "body": "Diagnose a kubernetes pod in CrashLoopBackOff."}), + ]) + + result = router.route("diagnose a kubernetes pod", "codex", "/tmp", min_score=0.0) + + assert result["match"] == "kubernetes-runbook" + assert result["matched_on"] == "content" + assert result["score_components"]["content"] > result["score_components"]["description"] + assert all("skill_body" not in item for item in result["alternatives"]) + + +def test_body_aware_route_filters_incompatible_content_before_ranking(monkeypatch): + import ingot.mcp_server.router as router_mod + monkeypatch.setattr(router_mod, "build_embedding", _BodyAwareEmbedding) + router_mod.Router._vector_cache.clear() + blocked = Skill(**{ + **_skill("kubernetes-runbook", "Operate a production service.", + required_tools=["kubectl"]).__dict__, + "body": "Diagnose a kubernetes pod in CrashLoopBackOff.", + }) + router = router_mod.Router([ + _skill("billing-runbook", "Operate a production service."), + blocked, + ]) + + result = router.route("diagnose a kubernetes pod", "codex", "/tmp", + available_tools=[], min_score=0.0) + + assert result["match"] == "billing-runbook" + + +def test_variant_content_is_ranked_for_the_requested_harness(monkeypatch): + import ingot.mcp_server.router as router_mod + monkeypatch.setattr(router_mod, "build_embedding", _BodyAwareEmbedding) + router_mod.Router._vector_cache.clear() + alpha = Skill(**{ + **_skill("alpha", "Operate a production service.").__dict__, + "body": "Investigate invoice charges.", + "variants": {"codex": "Diagnose a kubernetes pod in CrashLoopBackOff."}, + }) + beta = Skill(**{ + **_skill("beta", "Operate a production service.").__dict__, + "body": "Diagnose a kubernetes pod in CrashLoopBackOff.", + "variants": {"codex": "Investigate invoice charges."}, + }) + router = router_mod.Router([alpha, beta]) + + assert router.route("diagnose a kubernetes pod", "codex", "/tmp", + min_score=0.0)["match"] == "alpha" + assert router.route("diagnose a kubernetes pod", "claude", "/tmp", + min_score=0.0)["match"] == "beta" + + +def test_body_change_reuses_description_vector_and_reembeds_content(monkeypatch): + import ingot.mcp_server.router as router_mod + embedder = _BodyAwareEmbedding() + monkeypatch.setattr(router_mod, "build_embedding", lambda: embedder) + router_mod.Router._vector_cache.clear() + first = _skill("runbook", "Operate a production service.") + second = Skill(**{**first.__dict__, "body": "Diagnose a CrashLoopBackOff."}) + + router_mod.Router([first]).route("diagnose a kubernetes pod", "codex", "/tmp", + min_score=0.0) + router_mod.Router([second]).route("diagnose a kubernetes pod", "codex", "/tmp", + min_score=0.0) + + assert embedder.document_calls[0] == [first.description] + assert len(embedder.document_calls) == 3 + assert "Instructions:" in embedder.document_calls[1][0] + assert "CrashLoopBackOff" in embedder.document_calls[2][0] + + +def test_vector_cache_evicts_stale_body_revisions(monkeypatch): + import ingot.mcp_server.router as router_mod + embedder = _BodyAwareEmbedding() + monkeypatch.setattr(router_mod, "build_embedding", lambda: embedder) + monkeypatch.setattr(router_mod.Router, "_vector_cache_limit", 2, raising=False) + router_mod.Router._vector_cache.clear() + + for revision in range(4): + skill = Skill(**{ + **_skill("runbook", "Operate a production service.").__dict__, + "body": f"revision {revision} Diagnose a CrashLoopBackOff.", + }) + router_mod.Router([skill]).route( + "diagnose a kubernetes pod", "codex", "/tmp", min_score=0.0 + ) + + assert len(router_mod.Router._vector_cache) <= 2 + + +@pytest.mark.parametrize("value", ["0", "4001", "invalid"]) +def test_body_projection_bound_fails_closed(monkeypatch, value): + monkeypatch.setenv("ROUTER_BODY_CHARS", value) + with pytest.raises(ValueError, match="integer from 1 to 4000"): + Router([]) diff --git a/tests/test_routing.py b/tests/test_routing.py index 8fd7f4b..d5abe64 100644 --- a/tests/test_routing.py +++ b/tests/test_routing.py @@ -2,7 +2,7 @@ no-regression/improvement/collision gate. No embeddings, no LLM, the router is injected.""" import pytest -from optimize import routing as R +from ingot.optimize import routing as R class _ScriptedRouter: @@ -115,11 +115,11 @@ def test_run_routing_auto_drafts_missing_cases(monkeypatch, tmp_path): skill = tmp_path / "skills" / "sk" skill.mkdir(parents=True) (skill / "SKILL.md").write_text("---\nname: sk\ndescription: d.\n---\nbody\n") - monkeypatch.setattr(R, "SKILLS_DIR", tmp_path / "skills") + monkeypatch.setattr(R, "resolve_skill_dir", lambda name: tmp_path / "skills" / name) monkeypatch.setattr(R, "TASKS_DIR", tmp_path / "tasks", raising=False) (tmp_path / "tasks").mkdir() - from optimize import draft as D + from ingot.optimize import draft as D def sentinel(*a, **k): raise RuntimeError("drafter invoked") monkeypatch.setattr(D, "draft_and_append_routing", sentinel) @@ -132,26 +132,25 @@ def test_run_routing_writes_an_evidence_bundle_and_records_relative_paths(monkey the same portable bundle rather than a claim that one exists.""" import json - from optimize import promote as P + from ingot.optimize import promote as P root = tmp_path / "skills" skill = root / "sk" skill.mkdir(parents=True) (skill / "SKILL.md").write_text("---\nname: sk\ndescription: old trigger.\n---\nbody\n") monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root)) - monkeypatch.setattr(R, "SKILLS_DIR", root) + monkeypatch.setattr(R, "resolve_skill_dir", lambda name: root / name) tasks = tmp_path / "tasks" tasks.mkdir() (tasks / "sk.yaml").write_text( "routing:\n - task: use sk please\n expected: sk\n - task: unrelated\n expected: null\n") monkeypatch.setattr(R, "TASKS_DIR", tasks, raising=False) + monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs")) evidence_root = tmp_path / "runs" / "evidence" - monkeypatch.setattr(R, "EVIDENCE_DIR", evidence_root) - monkeypatch.setattr(P, "PENDING_DIR", tmp_path / "pending") monkeypatch.setattr(R, "optimize_description", lambda skill, seed, cases, budget: ("new trigger.", 0.5, 0.9)) monkeypatch.setattr(R, "_description_shadows", lambda skill, desc: ("", 0.0)) - from optimize import ab as A + from ingot.optimize import ab as A monkeypatch.setattr(A, "_routing_metrics", lambda skill, champ, chall: { "champion": {"top1": 0.5, "recall_at_3": 0.5, "no_route_precision": 1.0}, "challenger": {"top1": 1.0, "recall_at_3": 1.0, "no_route_precision": 1.0}, diff --git a/tests/test_routing_eval.py b/tests/test_routing_eval.py index f3ed568..43b2a97 100644 --- a/tests/test_routing_eval.py +++ b/tests/test_routing_eval.py @@ -1,6 +1,6 @@ -from mcp_server.routing_eval import evaluate_cases, evaluate_parity, load_cases -from mcp_server.registry import load_skills -from mcp_server.router import Router +from ingot.mcp_server.routing_eval import evaluate_cases, evaluate_parity, load_cases +from ingot.mcp_server.registry import load_skills +from ingot.mcp_server.router import Router from pathlib import Path @@ -54,9 +54,13 @@ def route(self, task, harness, **context): def test_committed_suite_covers_filter_and_parity_contract(): root = Path(__file__).resolve().parent.parent cases = load_cases(root / "evals" / "routing.yaml") - result = evaluate_cases(Router(load_skills(root / "evals" / "fixtures" / "skills")), cases) - parity = evaluate_parity(Router(load_skills(root / "evals" / "fixtures" / "skills")), cases) + router = Router(load_skills(root / "evals" / "fixtures" / "skills")) + result = evaluate_cases(router, cases) + parity = evaluate_parity(router, cases) assert len(cases) >= 10 assert result["failures"] == [] assert result["recall_at_3"] == 1.0 and result["no_route_precision"] == 1.0 assert parity["rate"] == 1.0 and parity["total"] >= 2 + body_case = next(case for case in cases if case["expected"] == "kubernetes-runbook") + routed = router.route(body_case["task"], body_case["harness"], body_case.get("cwd", ".")) + assert routed["match"] == "kubernetes-runbook" and routed["matched_on"] == "content" diff --git a/tests/test_routing_health.py b/tests/test_routing_health.py index f55d174..f2534bf 100644 --- a/tests/test_routing_health.py +++ b/tests/test_routing_health.py @@ -4,7 +4,7 @@ import yaml -from optimize import routing_health as H +from ingot.optimize import routing_health as H class _ScriptedRouter: diff --git a/tests/test_run_task.py b/tests/test_run_task.py index 6102b8f..0b4e5be 100644 --- a/tests/test_run_task.py +++ b/tests/test_run_task.py @@ -156,7 +156,7 @@ def test_serving_contract_requires_inline_deliverables(): # the scaffold habit of writing code to its scratch FS and describing it must be countered in # BOTH serving contracts, symmetrically, production agent and A/B eval agent from agent.run import INSTRUCTIONS - from optimize.ab import EVAL_INSTRUCTIONS + from ingot.optimize.ab import EVAL_INSTRUCTIONS for contract in (INSTRUCTIONS, EVAL_INSTRUCTIONS): assert "final answer must contain the complete deliverable" in contract assert "cannot" in contract and "workspace" in contract @@ -240,7 +240,7 @@ async def serve(task, routed, tools): monkeypatch.setattr(run_mod, "_connect", connect) monkeypatch.setattr(run_mod, "_serve", serve) monkeypatch.setattr(run_mod, "_print_route", lambda routed: None) - monkeypatch.setattr("optimize.openrouter_key_missing", lambda: False) + monkeypatch.setattr("ingot.optimize.openrouter_key_missing", lambda: False) asyncio.run(run_mod.main("write a skill")) diff --git a/tests/test_security.py b/tests/test_security.py index 1b790c9..ecb1125 100644 --- a/tests/test_security.py +++ b/tests/test_security.py @@ -4,10 +4,10 @@ import pytest -from mcp_server.registry import ( +from ingot.mcp_server.registry import ( SLUG_RE, parse_skill, read_components, write_components, write_skill_md, ) -from optimize.promote import check_slug +from ingot.optimize.promote import check_slug ROOT = Path(__file__).resolve().parents[1] diff --git a/tests/test_server.py b/tests/test_server.py index f5d93b1..ad1e5cd 100644 --- a/tests/test_server.py +++ b/tests/test_server.py @@ -1,6 +1,6 @@ import asyncio -from mcp_server.server import STATE, get_skill, mcp, route_and_load +from ingot.mcp_server.server import STATE, get_skill, mcp, route_and_load def test_get_skill_header_carries_revision(tmp_path): @@ -19,6 +19,8 @@ def test_route_and_load_is_additive_to_existing_mcp_tools(): tools = asyncio.run(mcp.list_tools()) assert {tool.name for tool in tools} == { "list_skills", "suggest_skills", "get_skill", "reload_skills", "route_and_load", + "propose_skill_update", + "propose_skill_create", } @@ -29,7 +31,7 @@ def test_route_refreshes_after_external_skill_promotion(tmp_path, monkeypatch): md = skill / "SKILL.md" md.write_text("---\nname: pdf\ndescription: Merge PDF files.\n---\nbody one\n") STATE.reload([root]) - monkeypatch.setattr("mcp_server.server.MIN_SCORE", 0.0) + monkeypatch.setattr("ingot.mcp_server.server.MIN_SCORE", 0.0) first = route_and_load("merge PDF", "codex", str(tmp_path)) md.write_text("---\nname: pdf\ndescription: Merge PDF files.\n---\nbody two\n") second = route_and_load("merge PDF", "codex", str(tmp_path)) @@ -45,18 +47,18 @@ def test_route_and_load_novel_flag_uses_server_thresholds(tmp_path, monkeypatch) (skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDF files.\n---\nbody\n") STATE.reload([root]) # match -> weak model serves the skill - monkeypatch.setattr("mcp_server.server.MIN_SCORE", 0.0) + monkeypatch.setattr("ingot.mcp_server.server.MIN_SCORE", 0.0) assert route_and_load("merge PDF", "codex", str(tmp_path))["novel"] is False # no match but within the related band -> compose/extend, still not novel - monkeypatch.setattr("mcp_server.server.MIN_SCORE", 0.99) - monkeypatch.setattr("mcp_server.server.RELATED_SCORE", 0.0) + monkeypatch.setattr("ingot.mcp_server.server.MIN_SCORE", 0.99) + monkeypatch.setattr("ingot.mcp_server.server.RELATED_SCORE", 0.0) related = route_and_load("merge PDF", "codex", str(tmp_path)) assert related["match"] is None and related["related_match"] == "pdf" assert related["novel"] is False and related["skill_body"] == "body" assert related["skill_root"] == str(skill) assert related["revision"] # nothing even related -> the harness should escalate to its strong model - monkeypatch.setattr("mcp_server.server.RELATED_SCORE", 0.99) + monkeypatch.setattr("ingot.mcp_server.server.RELATED_SCORE", 0.99) novel = route_and_load("merge PDF", "codex", str(tmp_path)) assert novel["match"] is None and novel["novel"] is True assert novel["related_match"] is None and novel["skill_body"] == "" @@ -70,7 +72,7 @@ def test_route_refreshes_revision_after_bundled_file_change(tmp_path, monkeypatc reference = skill / "reference.md" reference.write_text("version one") STATE.reload([root]) - monkeypatch.setattr("mcp_server.server.MIN_SCORE", 0.0) + monkeypatch.setattr("ingot.mcp_server.server.MIN_SCORE", 0.0) first = route_and_load("merge PDF", "codex", str(tmp_path)) reference.write_text("version two") second = route_and_load("merge PDF", "codex", str(tmp_path)) diff --git a/tests/test_setup_scripts.py b/tests/test_setup_scripts.py index dedfb1f..a3ac32e 100644 --- a/tests/test_setup_scripts.py +++ b/tests/test_setup_scripts.py @@ -46,6 +46,7 @@ def test_codex_setup_is_idempotent_and_writes_private_config(tmp_path): _executable(fake_bin / "codex", ''' echo "codex $*" >> "$TEST_STATE/calls" if [ "$1" = "--version" ]; then echo "codex-cli 0.144.5"; exit 0; fi +if [ "$1 $2" = "plugin --help" ]; then exit 0; fi if [ "$1 $2 $3" = "mcp get ingot" ]; then test -f "$TEST_STATE/mcp" && echo "url: http://localhost:8000/mcp" test -f "$TEST_STATE/mcp" @@ -79,19 +80,47 @@ def test_codex_setup_is_idempotent_and_writes_private_config(tmp_path): assert stat.S_IMODE(config.stat().st_mode) == 0o600 -def test_codex_setup_rejects_old_codex_before_writing_config(tmp_path): +def test_codex_setup_rejects_a_codex_without_the_plugin_subcommand(tmp_path): + """The floor is a capability, not a number: the script installs through `codex plugin`, so it + probes for that. A build too old to carry it is rejected before any credential is written.""" env, fake_bin = _environment(tmp_path) _executable(fake_bin / "node", 'echo 22\n') - _executable(fake_bin / "codex", 'echo "codex-cli 0.127.9"\n') + _executable(fake_bin / "codex", ''' +if [ "$1" = "--version" ]; then echo "codex-cli 0.127.9"; exit 0; fi +if [ "$1" = "plugin" ]; then echo "unrecognized subcommand 'plugin'" >&2; exit 2; fi +exit 0 +''') result = subprocess.run([str(ROOT / "scripts" / "codex_setup.sh")], cwd=ROOT, env=env, text=True, capture_output=True) assert result.returncode != 0 - assert "Codex 0.128 or newer" in result.stderr + assert "no 'plugin' subcommand" in result.stderr assert not (Path(env["HOME"]) / ".codex" / "langfuse.json").exists() +def test_codex_setup_accepts_a_local_build_that_stamps_no_version(tmp_path): + """A locally built codex reports `codex-cli 0.0.0`, which sorts below every release while + carrying the plugin subcommand. A version comparison rejected exactly the build that works — + this is the regression the capability probe exists to prevent.""" + env, fake_bin = _environment(tmp_path) + _executable(fake_bin / "node", 'echo 22\n') + _executable(fake_bin / "codex", ''' +if [ "$1" = "--version" ]; then echo "codex-cli 0.0.0-wire-persona"; exit 0; fi +if [ "$1 $2" = "plugin --help" ]; then exit 0; fi +if [ "$1 $2 $3" = "mcp get ingot" ]; then exit 1; fi +exit 0 +''') + env.update({"LANGFUSE_BASE_URL": "https://langfuse.example", + "LANGFUSE_PUBLIC_KEY": "pk-test", "LANGFUSE_SECRET_KEY": "sk-test"}) + + result = subprocess.run([str(ROOT / "scripts" / "codex_setup.sh")], cwd=ROOT, env=env, + text=True, capture_output=True) + + assert result.returncode == 0, result.stderr + assert (Path(env["HOME"]) / ".codex" / "langfuse.json").exists() + + def test_remote_setup_requires_explicit_langfuse_credentials(tmp_path): env, _ = _environment(tmp_path) env["LANGFUSE_BASE_URL"] = "https://langfuse.example" @@ -194,7 +223,7 @@ def test_live_smokes_require_completed_mcp_call_and_mining_parser(): assert "ingot/route_and_load (completed)" in codex assert "mcp__ingot__route_and_load" in claude and "tool_result" in claude - assert "from optimize.mine import fetch_traces" in codex - assert "from optimize.mine import fetch_traces" in claude + assert "from ingot.optimize.mine import fetch_traces" in codex + assert "from ingot.optimize.mine import fetch_traces" in claude assert "--add-host host.docker.internal:host-gateway" in codex assert "--add-host host.docker.internal:host-gateway" in claude diff --git a/tests/test_skillopt_bridge.py b/tests/test_skillopt_bridge.py index 6e01608..aed3416 100644 --- a/tests/test_skillopt_bridge.py +++ b/tests/test_skillopt_bridge.py @@ -3,7 +3,7 @@ driven by a stub reflection LM.""" import json -from optimize import skillopt_bridge as sk +from ingot.optimize import skillopt_bridge as sk def _lm(reply: str): diff --git a/tests/test_skillopt_loop.py b/tests/test_skillopt_loop.py index 3036f4d..86c7f1f 100644 --- a/tests/test_skillopt_loop.py +++ b/tests/test_skillopt_loop.py @@ -5,8 +5,8 @@ import pytest -from optimize import rollout as R -from optimize import skillopt_loop as S +from ingot.optimize import rollout as R +from ingot.optimize import skillopt_loop as S def _fake_lm(reply_for): diff --git a/tests/test_status.py b/tests/test_status.py new file mode 100644 index 0000000..7f33e5c --- /dev/null +++ b/tests/test_status.py @@ -0,0 +1,296 @@ +"""`ingot status`: the four states, decided by observation rather than by a configuration flag. + +MANAGED, PENDING, DRIFTED, UNMANAGED are answers about what is actually served compared with what +the last successful release receipt says should be served. A status command that read its verdict +back out of the configuration would agree with the claim instead of testing it.""" +import pytest + +from ingot import cli, status +from ingot.optimize import promote as P +from ingot.optimize import publication as Q + + +def _library(tmp_path, monkeypatch): + root = tmp_path / "skills" + root.mkdir() + monkeypatch.setenv("INGOT_LIBRARY", str(root)) + return root + + +def _skill(root, name, body="a body"): + directory = root / name + directory.mkdir(exist_ok=True) + (directory / "SKILL.md").write_text( + f"---\nname: {name}\ndescription: Does the {name} thing.\n---\n{body}\n") + from ingot.mcp_server.registry import skill_revision + return skill_revision(directory) + + +def _release(name, revision, *, champion="absent", state="active"): + """A receipt for one skill, driven through the real queue rather than hand-authored.""" + receipt = Q.queue_publication(name, { + "skill": name, "kind": "creation", + "challenger_components": {"description": f"Does the {name} thing.", "body": "a body"}, + "evidence": {"champion": {"revision": champion}, "challenger": {"revision": revision}}, + }, "admin", "promote") + if state != "approved_publishing": + Q.update_publication(receipt.id, state=state) + return receipt + + +def test_a_skill_serving_its_released_revision_is_managed(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + _release("pdf", _skill(root, "pdf")) + + result = status.library_status(root) + + assert result["mode"] == status.MANAGED + assert result["skills"] == [{"skill": "pdf", "state": status.MANAGED, + "revision": result["skills"][0]["revision"], + "released": result["skills"][0]["revision"], + "publication": result["skills"][0]["publication"]}] + + +def test_an_out_of_band_edit_reports_drifted(tmp_path, monkeypatch): + """A read-only mount does not stop the machine owner from editing the host directory. This is + the detection that replaces claiming it does.""" + root = _library(tmp_path, monkeypatch) + _release("pdf", _skill(root, "pdf")) + _skill(root, "pdf", body="edited by hand, out of band") + + result = status.library_status(root) + + assert result["mode"] == status.DRIFTED + assert result["skills"][0]["state"] == status.DRIFTED + assert result["skills"][0]["revision"] != result["skills"][0]["released"] + assert "outside the publisher" in status.render(result) + + +def test_restoring_the_released_bytes_returns_to_managed(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + released = _skill(root, "pdf") + _release("pdf", released) + _skill(root, "pdf", body="edited by hand, out of band") + assert status.library_status(root)["mode"] == status.DRIFTED + + _skill(root, "pdf") + + assert status.library_status(root)["mode"] == status.MANAGED + + +def test_a_skill_with_no_release_receipt_is_unmanaged_not_drifted(tmp_path, monkeypatch): + """Fetched, copied, or committed by hand. Real and common, and not drift: there is no release + for it to have drifted from, and calling it drift would make the alarm meaningless.""" + root = _library(tmp_path, monkeypatch) + _skill(root, "pdf") + + result = status.library_status(root) + + assert result["mode"] == status.UNMANAGED + assert result["skills"][0]["state"] == status.UNMANAGED + assert result["skills"][0]["released"] is None + + +def test_a_released_skill_with_a_newer_publication_in_flight_is_pending(tmp_path, monkeypatch): + """Serving the last release while the next one travels is the normal state, not drift.""" + root = _library(tmp_path, monkeypatch) + released = _skill(root, "pdf") + _release("pdf", released) + in_flight = _release("pdf", "c" * 64, champion=released, state="publishing") + + result = status.library_status(root) + + assert Q.load_publication(in_flight.id)["state"] == "publishing" + assert result["skills"][0]["released"] == released + assert result["skills"][0]["state"] == status.PENDING + assert result["mode"] == status.PENDING + + +def test_a_skill_in_flight_with_no_release_yet_is_pending(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + _release("pdf", _skill(root, "pdf"), state="publishing") + + assert status.library_status(root)["mode"] == status.PENDING + + +def test_a_creation_in_flight_is_pending_even_though_nothing_serves_it_yet(tmp_path, monkeypatch): + """Caught in situ: a new skill is served by nothing and released by nothing, so a status built + from those two sets alone called an empty library fully MANAGED mid-publication.""" + root = _library(tmp_path, monkeypatch) + _release("csv-tidy", "e" * 64, state="approved_publishing") + + result = status.library_status(root) + + assert result["mode"] == status.PENDING + assert [entry["skill"] for entry in result["skills"]] == ["csv-tidy"] + assert result["skills"][0]["revision"] == status.ABSENT + + +def test_a_quarantined_proposal_is_pending(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + _release("pdf", _skill(root, "pdf")) + P.save_pending("pdf", {"skill": "pdf", "gate": {"promotable": True}, + "evidence": {"challenger": {"revision": "b" * 64}}}) + + assert status.library_status(root)["mode"] == status.PENDING + + +def test_drift_outranks_a_publication_in_flight(tmp_path, monkeypatch): + """The alarm must not be masked by an unrelated change travelling to the vault.""" + root = _library(tmp_path, monkeypatch) + released = _skill(root, "pdf") + _release("pdf", released) + _release("tailwind", _skill(root, "tailwind")) + _release("tailwind", "d" * 64, champion=released, state="publishing") + _skill(root, "pdf", body="edited by hand, out of band") + + assert status.library_status(root)["mode"] == status.DRIFTED + + +def test_an_empty_library_is_managed(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + + assert status.library_status(root)["mode"] == status.MANAGED + + +def test_development_mode_is_unmanaged_whatever_the_receipts_say(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + _release("pdf", _skill(root, "pdf")) + monkeypatch.setenv("INGOT_MODE", "dev") + + result = status.library_status(root) + + assert result["mode"] == status.UNMANAGED + assert result["development_mode"] is True + assert "do not apply" in status.render(result) + + +def test_a_writable_library_is_reported_without_deciding_the_verdict(tmp_path, monkeypatch): + """The administrator who owns the vault can always write it. A status command that answered + UNMANAGED from their shell would hide the drift they most need to see.""" + root = _library(tmp_path, monkeypatch) + _release("pdf", _skill(root, "pdf")) + + result = status.library_status(root) + + assert result["mode"] == status.MANAGED + assert result["writable_roots"] == [str(root.resolve())] + assert "without an approval" in status.render(result) + + +def test_status_exits_zero_only_when_everything_is_as_approved(tmp_path, monkeypatch, capsys): + root = _library(tmp_path, monkeypatch) + _release("pdf", _skill(root, "pdf")) + assert cli.main(["status", "--root", str(root)]) == 0 + + _skill(root, "pdf", body="edited by hand, out of band") + + assert cli.main(["status", "--root", str(root)]) == 1 + assert status.DRIFTED in capsys.readouterr().out + + +def test_the_json_payload_is_versioned(tmp_path, monkeypatch, capsys): + import json + root = _library(tmp_path, monkeypatch) + _skill(root, "pdf") + + cli.main(["status", "--root", str(root), "--json"]) + + assert json.loads(capsys.readouterr().out)["schema_version"] == status.STATUS_SCHEMA + + +# --------------------------------------------------------------------------- delivery targets + +def _delivery(tmp_path, monkeypatch, vault): + """One native filesystem target beside the managed vault, configured the way an operator would.""" + native = tmp_path / "claude" + native.mkdir() + monkeypatch.setenv("INGOT_VAULT_PATH", str(vault)) + monkeypatch.setenv("INGOT_DELIVERY_TARGETS", f"claude=filesystem:{native}") + return native + + +def _copy(source, destination): + import shutil + shutil.copytree(source, destination, dirs_exist_ok=True) + + +def test_a_target_holding_the_released_revision_is_managed(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + native = _delivery(tmp_path, monkeypatch, root) + _release("pdf", _skill(root, "pdf")) + _copy(root / "pdf", native / "pdf") + + targets = status.target_states() + + assert [(entry["name"], entry["state"]) for entry in targets] == [ + ("vault", status.MANAGED), ("claude", status.MANAGED)] + + +def test_only_the_target_that_was_altered_reports_drift(tmp_path, monkeypatch): + """The acceptance case. Editing a native skill root must not implicate the vault, and the vault + still holding the release must not hide the edit.""" + root = _library(tmp_path, monkeypatch) + native = _delivery(tmp_path, monkeypatch, root) + _release("pdf", _skill(root, "pdf")) + _copy(root / "pdf", native / "pdf") + + (native / "pdf" / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Does the pdf thing.\n---\nedited out of band\n") + + states = {entry["name"]: entry["state"] for entry in status.target_states()} + assert states == {"vault": status.MANAGED, "claude": status.DRIFTED} + + +def test_a_released_skill_deleted_from_a_target_is_drift_not_silence(tmp_path, monkeypatch): + root = _library(tmp_path, monkeypatch) + native = _delivery(tmp_path, monkeypatch, root) + _release("pdf", _skill(root, "pdf")) + + states = {entry["name"]: entry["state"] for entry in status.target_states()} + assert states["claude"] == status.DRIFTED + + +def test_a_target_is_judged_only_on_the_skills_ingot_released_there(tmp_path, monkeypatch): + """A native skill root is shared. Skills the operator put there themselves are not Ingot's to + grade, and reporting them would make every real deployment permanently UNMANAGED.""" + root = _library(tmp_path, monkeypatch) + native = _delivery(tmp_path, monkeypatch, root) + _release("pdf", _skill(root, "pdf")) + _copy(root / "pdf", native / "pdf") + _skill(native, "somebody-elses-skill") + + claude = [entry for entry in status.target_states() if entry["name"] == "claude"][0] + + assert claude["state"] == status.MANAGED + assert [skill["skill"] for skill in claude["skills"]] == ["pdf"] + + +def test_a_drifted_target_is_not_hidden_by_a_clean_vault(tmp_path, monkeypatch, capsys): + """`ingot status` answering MANAGED while a native skill root serves the wrong bytes is exactly + the lie this command exists to prevent.""" + root = _library(tmp_path, monkeypatch) + native = _delivery(tmp_path, monkeypatch, root) + _release("pdf", _skill(root, "pdf")) + _copy(root / "pdf", native / "pdf") + (native / "pdf" / "SKILL.md").write_text( + "---\nname: pdf\ndescription: Does the pdf thing.\n---\nedited out of band\n") + + assert cli.main(["status", "--root", str(root)]) == 1 + + output = capsys.readouterr().out + assert output.startswith(status.DRIFTED) + assert "claude" in output and str(native) in output + + +def test_an_unusable_delivery_configuration_is_reported_rather_than_raised(tmp_path, monkeypatch, + capsys): + """Status is the command an operator runs when something is wrong. It has to survive a bad + environment variable and say what is wrong with it.""" + root = _library(tmp_path, monkeypatch) + monkeypatch.setenv("INGOT_VAULT_PATH", str(root)) + monkeypatch.setenv("INGOT_DELIVERY_TARGETS", "claude=carrier-pigeon:/tmp/x") + + assert cli.main(["status", "--root", str(root)]) == 1 + + assert "unknown delivery kind" in capsys.readouterr().out diff --git a/tests/test_tree.py b/tests/test_tree.py new file mode 100644 index 0000000..6c5296a --- /dev/null +++ b/tests/test_tree.py @@ -0,0 +1,281 @@ +"""The candidate tree: the exact bytes an admitted package publishes. + +Every test here exists because the alternative was silent. A package used to be reduced to decoded +text on the way in, so a file the dictionary could not hold was reviewed as part of the candidate +and then was not in it -- no error, no warning, just a revision naming a package that no longer +existed. These check that the bytes survive, that the receipt binds them, and that anything which +moves them afterwards is refused rather than published.""" +import hashlib +import os +import stat + +import pytest + +from ingot.mcp_server.registry import skill_revision +from ingot.optimize import tree + +PNG = b"\x89PNG\r\n\x1a\n" + bytes(range(256)) + + +@pytest.fixture(autouse=True) +def _runs(tmp_path, monkeypatch): + monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs")) + + +def _package(root, name="pdf", description="Merge and split PDF files.", body="Combine PDFs."): + directory = root / name + directory.mkdir(parents=True, exist_ok=True) + (directory / "SKILL.md").write_text( + f"---\nname: {name}\ndescription: {description}\n---\n\n{body}\n", encoding="utf-8") + return directory + + +def _components(description="Merge and split PDF files.", body="Combine PDFs."): + import json + return {"description": description, "body": body, + "frontmatter": json.dumps({"name": "pdf", "description": description})} + + +# --- describing a package --------------------------------------------------------------------- + +def test_every_regular_file_is_described_by_its_raw_bytes(tmp_path): + package = _package(tmp_path / "src") + (package / "assets").mkdir() + (package / "assets" / "logo.png").write_bytes(PNG) + + manifest = tree.build(package) + + entry = next(item for item in manifest["files"] if item["path"] == "assets/logo.png") + assert entry == {"path": "assets/logo.png", "mode": 0o644, "size": len(PNG), + "sha256": hashlib.sha256(PNG).hexdigest()} + + +def test_the_hash_is_of_bytes_not_of_decoded_text(tmp_path): + """A file that is not valid UTF-8 has no decoded form, and one that is would hash differently + after a round trip through `errors='ignore'` -- which is how the bytes went missing.""" + package = _package(tmp_path / "src") + raw = b"caf\xe9 latin-1, not utf-8\n" + (package / "notes.txt").write_bytes(raw) + + entry = next(item for item in tree.build(package)["files"] if item["path"] == "notes.txt") + + assert entry["sha256"] == hashlib.sha256(raw).hexdigest() + + +def test_modes_are_clamped_to_the_two_a_git_checkout_reproduces(tmp_path): + package = _package(tmp_path / "src") + (package / "run.sh").write_text("#!/bin/sh\n") + (package / "run.sh").chmod(0o764) + (package / "notes.md").write_text("# Notes\n") + (package / "notes.md").chmod(0o600) + + modes = {item["path"]: item["mode"] for item in tree.build(package)["files"]} + + assert modes["run.sh"] == 0o755 + assert modes["notes.md"] == 0o644 + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows") +def test_a_symlink_is_refused_by_name(tmp_path): + package = _package(tmp_path / "src") + (package / "real.md").write_text("# Real\n") + (package / "link.md").symlink_to(package / "real.md") + + with pytest.raises(ValueError, match="symlinks are not admissible: link.md"): + tree.build(package) + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation needs privilege on Windows") +def test_a_directory_symlink_is_refused_before_it_is_descended(tmp_path): + package = _package(tmp_path / "src") + outside = tmp_path / "outside" + outside.mkdir() + (outside / "secret.md").write_text("secret\n") + (package / "docs").symlink_to(outside, target_is_directory=True) + + with pytest.raises(ValueError, match="symlinks are not admissible: docs"): + tree.build(package) + + +def test_an_unportable_path_is_refused(tmp_path): + package = _package(tmp_path / "src") + (package / "a:b.md").write_text("# Colon\n") + + with pytest.raises(ValueError, match="portable POSIX path"): + tree.build(package) + + +def test_the_byte_budget_is_enforced(tmp_path, monkeypatch): + monkeypatch.setattr(tree, "MAX_TREE_BYTES", 128) + package = _package(tmp_path / "src") + (package / "big.bin").write_bytes(b"\x00" * 512) + + with pytest.raises(ValueError, match="at most 128 bytes"): + tree.build(package) + + +def test_the_file_budget_is_enforced(tmp_path, monkeypatch): + monkeypatch.setattr(tree, "MAX_FILES", 3) + package = _package(tmp_path / "src") + for index in range(5): + (package / f"note-{index}.md").write_text("# Note\n") + + with pytest.raises(ValueError, match="at most 3 files"): + tree.build(package) + + +# --- binding the manifest --------------------------------------------------------------------- + +def test_the_digest_covers_the_file_list(tmp_path): + """The digest is what binds a receipt to a staged tree. A receipt whose file list was edited + without its digest would publish a tree nobody approved.""" + package = _package(tmp_path / "src") + manifest = tree.build(package) + manifest["files"][0]["sha256"] = "0" * 64 + + with pytest.raises(ValueError, match="digest does not cover its file list"): + tree.verify_manifest(manifest) + + +def test_a_manifest_entry_that_escapes_the_skill_root_is_refused(tmp_path): + package = _package(tmp_path / "src") + manifest = tree.build(package) + manifest["files"][0]["path"] = "../escape.md" + manifest["digest"] = tree._digest(manifest["files"]) + + with pytest.raises(ValueError, match="escapes skill root"): + tree.verify_manifest(manifest) + + +def test_a_mode_the_vault_cannot_serve_is_refused(tmp_path): + package = _package(tmp_path / "src") + manifest = tree.build(package) + manifest["files"][0]["mode"] = 0o777 + manifest["digest"] = tree._digest(manifest["files"]) + + with pytest.raises(ValueError, match="unsupported mode"): + tree.verify_manifest(manifest) + + +# --- staging ---------------------------------------------------------------------------------- + +def test_staging_copies_the_exact_bytes(tmp_path): + package = _package(tmp_path / "src") + (package / "logo.png").write_bytes(PNG) + manifest = tree.build(package) + + staged = tree.stage(package, manifest) + + assert (staged / "logo.png").read_bytes() == PNG + assert staged == tree.staged_dir(manifest["digest"]) + + +def test_staging_identical_bytes_twice_converges_on_one_directory(tmp_path): + """Named by digest, so a resubmission of the same package is not a second copy or a race.""" + first = _package(tmp_path / "one") + second = _package(tmp_path / "two") + + one = tree.stage(first, tree.build(first)) + two = tree.stage(second, tree.build(second)) + + assert one == two + assert len(list(tree.candidates_dir().iterdir())) == 1 + + +def test_a_failed_staging_leaves_nothing_behind(tmp_path): + package = _package(tmp_path / "src") + manifest = tree.build(package) + (package / "SKILL.md").write_text("changed after the manifest was built\n") + + with pytest.raises(ValueError, match="changed while it was being staged"): + tree.stage(package, manifest) + + assert list(tree.candidates_dir().iterdir()) == [] + + +# --- materializing ------------------------------------------------------------------------------ + +def test_materializing_reproduces_the_bytes_and_the_mode(tmp_path): + package = _package(tmp_path / "src") + (package / "logo.png").write_bytes(PNG) + (package / "run.sh").write_text("#!/bin/sh\necho hi\n") + (package / "run.sh").chmod(0o755) + manifest = tree.build(package) + tree.stage(package, manifest) + + destination = tmp_path / "out" + tree.materialize(manifest, destination) + + assert (destination / "logo.png").read_bytes() == PNG + assert stat.S_IMODE((destination / "run.sh").stat().st_mode) == 0o755 + + +def test_a_staged_file_altered_after_approval_is_refused(tmp_path): + """The receipt is the authority. If the staged bytes have moved since it was written, the + publisher must refuse rather than publish whatever it finds.""" + package = _package(tmp_path / "src") + (package / "logo.png").write_bytes(PNG) + manifest = tree.build(package) + staged = tree.stage(package, manifest) + (staged / "logo.png").write_bytes(b"something else") + + with pytest.raises(ValueError, match="does not match the receipt: logo.png"): + tree.materialize(manifest, tmp_path / "out") + + +def test_a_missing_staged_tree_is_refused_not_skipped(tmp_path): + package = _package(tmp_path / "src") + manifest = tree.build(package) + + with pytest.raises(ValueError, match="staged candidate tree .* is missing"): + tree.materialize(manifest, tmp_path / "out") + + +# --- what the library ends up serving ----------------------------------------------------------- + +def test_skill_md_is_normalized_and_everything_else_is_preserved(tmp_path): + """The one documented exception. SKILL.md's frontmatter is the routing interface, so it is + re-emitted through a safe YAML dump; every other file is the operator's bytes.""" + package = _package(tmp_path / "src") + (package / "logo.png").write_bytes(PNG) + manifest = tree.build(package) + tree.stage(package, manifest) + + destination = tmp_path / "out" / "pdf" + tree.materialize_creation(manifest, _components(description="Merge and split."), + "pdf", destination) + + assert (destination / "logo.png").read_bytes() == PNG + assert "description: Merge and split." in (destination / "SKILL.md").read_text() + + +def test_the_approved_revision_is_the_revision_of_the_materialized_tree(tmp_path): + """Computed by materializing it. A revision derived some other way would be a second + description of the same bytes, and the two would eventually disagree.""" + package = _package(tmp_path / "src") + (package / "logo.png").write_bytes(PNG) + manifest = tree.build(package) + tree.stage(package, manifest) + components = _components() + + revision = tree.revision("pdf", manifest, components) + + destination = tmp_path / "out" / "pdf" + tree.materialize_creation(manifest, components, "pdf", destination) + assert revision == skill_revision(destination) + + +def test_one_changed_asset_byte_changes_the_revision(tmp_path): + """While assets were dropped, two packages differing only in an image hashed identically.""" + first = _package(tmp_path / "one") + (first / "logo.png").write_bytes(PNG) + second = _package(tmp_path / "two") + (second / "logo.png").write_bytes(PNG[:-1] + b"\x00") + + revisions = set() + for package in (first, second): + manifest = tree.build(package) + tree.stage(package, manifest) + revisions.add(tree.revision("pdf", manifest, _components())) + + assert len(revisions) == 2 diff --git a/tests/test_ui.py b/tests/test_ui.py index 3652975..559f6ae 100644 --- a/tests/test_ui.py +++ b/tests/test_ui.py @@ -1,12 +1,17 @@ """Change-control UI guards: key preflight, slug validation, same-origin check, the pending lifecycle, and the history/rollback surface.""" +import json import threading from html.parser import HTMLParser import pytest from fastapi.testclient import TestClient -from optimize import promote as P +from ingot.mcp_server import registry +from ingot.optimize import harbor_report +from ingot.optimize import promote as P +from ingot.optimize import publication as Q +from ui import app as A from ui.app import app @@ -45,8 +50,6 @@ def client(tmp_path, monkeypatch): monkeypatch.setattr(auth, "AUTH_FILE", tmp_path / "no-auth.json") # auth off unless a test opts in monkeypatch.delenv("AUTH_USER", raising=False) monkeypatch.delenv("AUTH_PASSWORD", raising=False) - monkeypatch.setattr(P, "PENDING_DIR", tmp_path / "pending") - monkeypatch.setattr(P, "REVISIONS_DIR", tmp_path / "revisions") monkeypatch.setenv("OPENROUTER_API_KEY", "sk-or-test") return TestClient(app) @@ -70,6 +73,323 @@ def test_optimize_without_task_set_is_404(client): assert r.status_code == 404 +def _eval_skill(tmp_path, monkeypatch, name="mounted-skill"): + import ui.app as U + from ingot.mcp_server.registry import write_skill_md + + root = tmp_path / "read-only-library" + skill = root / name + skill.mkdir(parents=True) + write_skill_md(skill / "SKILL.md", {"name": name, "description": "Mounted description."}, + "Mounted body.") + tasks = tmp_path / "tasks" + tasks.mkdir() + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root)) + monkeypatch.setattr(U, "TASKS_DIR", tasks) + U.RUNS.clear() + return tasks + + +def _run_threads_inline(monkeypatch): + import ui.app as U + + class InlineThread: + def __init__(self, target, args=(), **_kwargs): + self.target, self.args = target, args + + def start(self): + self.target(*self.args) + + monkeypatch.setattr(U.threading, "Thread", InlineThread) + + +def test_create_eval_set_reads_an_indexed_skill_and_persists_the_draft( + client, tmp_path, monkeypatch): + tasks = _eval_skill(tmp_path, monkeypatch) + _run_threads_inline(monkeypatch) + + def draft(name, description, body, tasks_dir, log): + assert (name, description, body) == ( + "mounted-skill", "Mounted description.", "Mounted body.") + assert tasks_dir.parent == tasks + (tasks_dir / f"{name}.yaml").write_text( + "skill: mounted-skill\ntrain:\n- task: train\nholdout:\n- task: holdout\n") + log("[draft] wrote eval set") + return tasks_dir / f"{name}.yaml" + + monkeypatch.setattr("ingot.optimize.draft.draft_and_save", draft) + response = client.post("/api/evals/mounted-skill") + + assert response.status_code == 200 + assert response.json() == {"started": "mounted-skill"} + assert (tasks / "mounted-skill.yaml").exists() + assert client.get("/api/runs").json()["mounted-skill"] == { + "status": "done", "action": "eval", "log": ["[draft] wrote eval set"]} + + +def test_create_eval_set_refuses_to_overwrite_an_existing_set(client, tmp_path, monkeypatch): + tasks = _eval_skill(tmp_path, monkeypatch) + (tasks / "mounted-skill.yaml").write_text("keep: me\n") + + response = client.post("/api/evals/mounted-skill") + + assert response.status_code == 409 + assert "already has" in response.json()["detail"] + assert (tasks / "mounted-skill.yaml").read_text() == "keep: me\n" + + +def test_create_eval_set_preserves_a_file_created_while_drafting( + client, tmp_path, monkeypatch): + tasks = _eval_skill(tmp_path, monkeypatch) + _run_threads_inline(monkeypatch) + + def racing_draft(name, _description, _body, tasks_dir, log): + staged = tasks_dir / f"{name}.yaml" + staged.write_text("draft: mine\n") + (tasks / f"{name}.yaml").write_text("draft: theirs\n") + return staged + + monkeypatch.setattr("ingot.optimize.draft.draft_and_save", racing_draft) + + assert client.post("/api/evals/mounted-skill").status_code == 200 + run = client.get("/api/runs").json()["mounted-skill"] + assert run["status"] == "error" + assert run["log"] == ["ERROR: 'mounted-skill' already has an eval task set"] + assert (tasks / "mounted-skill.yaml").read_text() == "draft: theirs\n" + + +def test_create_eval_set_shares_the_paid_run_lock(client, tmp_path, monkeypatch): + _eval_skill(tmp_path, monkeypatch) + import ui.app as U + U.RUNS["other-skill"] = {"status": "running", "action": "optimize", "log": []} + + response = client.post("/api/evals/mounted-skill") + + assert response.status_code == 409 + assert "already in progress" in response.json()["detail"] + + +def test_create_eval_set_surfaces_background_errors(client, tmp_path, monkeypatch): + tasks = _eval_skill(tmp_path, monkeypatch) + _run_threads_inline(monkeypatch) + + def partial_draft(name, _description, _body, tasks_dir, log): + (tasks_dir / f"{name}.yaml").write_text("partial: true\n") + raise RuntimeError("teacher down") + + monkeypatch.setattr("ingot.optimize.draft.draft_and_save", partial_draft) + + assert client.post("/api/evals/mounted-skill").status_code == 200 + run = client.get("/api/runs").json()["mounted-skill"] + assert run["status"] == "error" + assert run["action"] == "eval" + assert run["log"] == ["ERROR: teacher down"] + assert not (tasks / "mounted-skill.yaml").exists() + + +def test_create_eval_set_requires_an_indexed_skill_and_provider(client, tmp_path, monkeypatch): + import ui.app as U + U.RUNS.clear() + assert client.post("/api/evals/not-indexed").status_code == 404 + + _eval_skill(tmp_path, monkeypatch) + monkeypatch.delenv("OPENROUTER_API_KEY") + response = client.post("/api/evals/mounted-skill") + assert response.status_code == 400 + assert "API_KEY" in response.json()["detail"] + + +def test_eval_set_api_exposes_the_measured_inputs(client, tmp_path, monkeypatch): + tasks = _eval_skill(tmp_path, monkeypatch) + (tasks / "mounted-skill.yaml").write_text( + "skill: mounted-skill\n" + "train:\n" + "- task: Write the train answer.\n" + " rubric: Include the train result.\n" + " checklist:\n" + " - id: train_result\n" + " criterion: Includes the train result.\n" + " weight: 3\n" + " dimension: correctness\n" + "holdout:\n" + "- task: Write the held-out answer.\n" + " rubric: Include the held-out result.\n" + "routing:\n" + "- task: Use the mounted skill.\n" + " expected: mounted-skill\n" + "acceptance:\n" + "- id: no_placeholder\n" + " forbid: TODO\n" + " description: No placeholder output.\n") + + response = client.get("/api/evals/mounted-skill") + + assert response.status_code == 200 + assert response.json() == { + "skill": "mounted-skill", + "train": [{ + "task": "Write the train answer.", + "rubric": "Include the train result.", + "checklist": [{ + "id": "train_result", + "criterion": "Includes the train result.", + "weight": 3, + "dimension": "correctness", + }], + }], + "holdout": [{ + "task": "Write the held-out answer.", + "rubric": "Include the held-out result.", + }], + "routing": [{ + "task": "Use the mounted skill.", + "expected": "mounted-skill", + }], + "acceptance": [{ + "id": "no_placeholder", + "forbid": "TODO", + "description": "No placeholder output.", + }], + "leakage": False, + "counts": {"train": 1, "holdout": 1, "routing": 1, "acceptance": 1, "checks": 1}, + } + + +def test_eval_set_api_marks_legacy_flat_tasks_as_leaky(client, tmp_path, monkeypatch): + tasks = _eval_skill(tmp_path, monkeypatch) + (tasks / "mounted-skill.yaml").write_text( + "tasks:\n- task: Same task trains and gates.\n" + " checklist:\n - criterion: Produce the requested result.\n") + + response = client.get("/api/evals/mounted-skill") + + assert response.status_code == 200 + assert response.json()["leakage"] is True + assert response.json()["train"] == response.json()["holdout"] + assert response.json()["counts"]["checks"] == 1 + + +def test_eval_set_api_reports_missing_and_malformed_files(client, tmp_path, monkeypatch): + tasks = _eval_skill(tmp_path, monkeypatch) + assert client.get("/api/evals/mounted-skill").status_code == 404 + (tasks / "mounted-skill.yaml").write_text("train: [") + response = client.get("/api/evals/mounted-skill") + assert response.status_code == 503 + assert "eval set is unreadable" in response.json()["detail"] + + +def test_review_api_runs_existing_reviewer_and_serves_persisted_result( + client, tmp_path, monkeypatch): + tasks = _eval_skill(tmp_path, monkeypatch) + (tasks / "mounted-skill.yaml").write_text( + "train:\n- task: train\nholdout:\n- task: holdout\n") + _run_threads_inline(monkeypatch) + import ingot.optimize.review as review_module + review_dir = tmp_path / "reviews" + monkeypatch.setattr(review_module, "REVIEW_DIR", review_dir) + + def review(skill, log): + assert skill == "mounted-skill" + result = { + "skill": skill, "model": "test-model", "revision": "abc123", + "score": 0.75, "tasks": 2, "checks": 4, "failed_checks": 1, + "by_dimension": {"correctness": 1.0}, + "findings": [{"task": "holdout", "check": "answer", "criterion": "Answer it.", + "weight": 2, "dimension": "correctness", "value": 0.5, + "note": "Missing detail.", "cost": 1.0}], + "per_task": [{"task": "train", "score": 1.0}, {"task": "holdout", "score": 0.5}], + } + review_dir.mkdir() + (review_dir / f"{skill}.json").write_text(__import__("json").dumps(result)) + log("[review] complete") + return result + + monkeypatch.setattr(review_module, "run_review", review) + + response = client.post("/api/reviews/mounted-skill") + + assert response.status_code == 200 + assert response.json() == {"started": "mounted-skill"} + assert client.get("/api/runs").json()["mounted-skill"] == { + "status": "done", "action": "review", "log": ["[review] complete"]} + saved = client.get("/api/reviews/mounted-skill") + assert saved.status_code == 200 + assert saved.json()["score"] == 0.75 + assert saved.json()["failed_checks"] == 1 + assert saved.json()["created"] > 0 + + persisted = __import__("json").loads( + (review_dir / "mounted-skill.json").read_text()) + persisted["created"] = 123 + (review_dir / "mounted-skill.json").write_text(__import__("json").dumps(persisted)) + assert client.get("/api/reviews/mounted-skill").json()["created"] == 123 + + +@pytest.mark.parametrize(("recorded", "message"), [ + (None, "did not record a skill revision"), + ("old-revision", "active skill changed since this review ran"), +]) +def test_review_api_marks_unbound_or_changed_results_stale( + client, tmp_path, monkeypatch, recorded, message): + _eval_skill(tmp_path, monkeypatch) + import ingot.optimize.review as review_module + review_dir = tmp_path / "reviews" + review_dir.mkdir() + monkeypatch.setattr(review_module, "REVIEW_DIR", review_dir) + (review_dir / "mounted-skill.json").write_text(json.dumps({ + "skill": "mounted-skill", "revision": recorded, "score": 1.0, + })) + + result = client.get("/api/reviews/mounted-skill") + + assert result.status_code == 200 + assert message in result.json()["stale"] + + +def test_review_api_accepts_the_revision_recorded_by_the_review_path( + client, tmp_path, monkeypatch): + from ingot.mcp_server.registry import skill_revision + _eval_skill(tmp_path, monkeypatch) + import ingot.optimize.review as review_module + review_dir = tmp_path / "reviews" + review_dir.mkdir() + monkeypatch.setattr(review_module, "REVIEW_DIR", review_dir) + skill = tmp_path / "read-only-library" / "mounted-skill" + (review_dir / "mounted-skill.json").write_text(json.dumps({ + "skill": "mounted-skill", "revision": skill_revision(skill), "score": 1.0, + })) + + result = client.get("/api/reviews/mounted-skill") + + assert result.status_code == 200 + assert result.json()["stale"] is None + + +def test_review_api_requires_evals_and_shares_the_paid_run_lock( + client, tmp_path, monkeypatch): + tasks = _eval_skill(tmp_path, monkeypatch) + assert client.post("/api/reviews/mounted-skill").status_code == 404 + (tasks / "mounted-skill.yaml").write_text("train:\n- task: train\n") + import ui.app as U + U.RUNS["other-skill"] = {"status": "running", "action": "optimize", "log": []} + response = client.post("/api/reviews/mounted-skill") + assert response.status_code == 409 + assert "already in progress" in response.json()["detail"] + + +def test_review_result_reports_missing_and_malformed_files(client, tmp_path, monkeypatch): + _eval_skill(tmp_path, monkeypatch) + import ingot.optimize.review as review_module + review_dir = tmp_path / "reviews" + review_dir.mkdir() + monkeypatch.setattr(review_module, "REVIEW_DIR", review_dir) + assert client.get("/api/reviews/mounted-skill").status_code == 404 + (review_dir / "mounted-skill.json").write_text("{") + response = client.get("/api/reviews/mounted-skill") + assert response.status_code == 503 + assert "review result is unreadable" in response.json()["detail"] + + def test_cross_origin_post_refused(client): r = client.post("/api/optimize/pdf", headers={"origin": "http://evil.example"}) assert r.status_code == 403 @@ -85,6 +405,45 @@ def test_pending_unknown_skill_is_404(client): assert client.get("/api/pending/pdf").status_code == 404 +def test_pending_creation_appears_as_to_be_added_and_can_be_reviewed(client, tmp_path, + monkeypatch): + import ui.app as U + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(tmp_path / "empty-library")) + components = {"description": "Write conversion copy.", "body": "Body"} + revision = registry.skill_revision(registry.writable_skill_dir("copywriting"), components) + P.save_pending("copywriting", { + "skill": "copywriting", "kind": "creation", "created": 7, + "champion_components": {}, + "challenger_components": components, + "changed_components": ["description", "body"], + "gate": {"promotable": True, "blocked": [], "kind": "new_skill_admission"}, + "evidence": {"challenger": {"revision": revision}}, + "creation": {"summary": "Add vetted copywriting skill."}, + }) + U._SKILLS_CACHE = None + U._SKILLS_CACHE_KEY = None + + skills = client.get("/api/skills").json() + assert skills == [{"name": "copywriting", "description": "Write conversion copy.", + "has_tasks": False, "pending": True, "revision": "", + "uses": 0, "publishing": False, "provenance": "proposed", + "status": None, "active": False}] + pending = client.get("/api/pending/copywriting") + assert pending.status_code == 200 + assert pending.json()["kind"] == "creation" + assert pending.json()["stale"] is None + + html = client.get("/").text + assert "to be added" in html + assert "Review addition" in html + assert "function creationEvidence(" in html + assert 'p.creation ? "Approve & add"' in html + assert '["proposed", "To be added"' in html + assert 'p.creation ? "Confirm addition, "' in html + assert "Submission requirements met; evidence is operator-supplied" in html + assert "const activeSkills = skills.filter(skill => skill.active !== false)" in html + + def test_promote_without_pending_is_404(client): assert client.post("/api/promote/pdf").status_code == 404 @@ -194,6 +553,15 @@ def test_compose_mcp_service_mounts_the_runs_directory(): assert "./runs:/app/runs" in compose["services"]["mcp"]["volumes"] +def test_compose_mcp_and_ui_share_the_pending_queue_uid(): + """MCP creation proposals and UI reviews share mode-0600 files in runs/pending.""" + import yaml + from pathlib import Path + compose = yaml.safe_load((Path(__file__).resolve().parents[1] / "docker-compose.yml").read_text()) + + assert compose["services"]["mcp"]["user"] == compose["services"]["ui"]["user"] + + def test_skills_list_empty_library(client): r = client.get("/api/skills") assert r.status_code == 200 @@ -252,7 +620,7 @@ def test_skill_version_explorer_reads_active_pending_and_snapshot(client, tmp_pa "body": "active body"}, "challenger_components": pending}) - snapshot = P.REVISIONS_DIR / "pdf" / "abc123" + snapshot = P.revisions_dir() / "pdf" / "abc123" snapshot.mkdir(parents=True) (snapshot / "SKILL.md").write_text( "---\nname: pdf\ndescription: Snapshot description.\n---\nsnapshot body\n") @@ -334,6 +702,75 @@ def test_skill_list_ships_search_filters_version_explorer_and_live_updates(clien assert "renderSkills(skills, runs || runInventory, hist)" in html +def test_skill_families_filter_the_skill_list_and_category_atlas(client): + html = client.get("/").text + layout = _Layout(html) + + assert "skill-family-filter" in layout.ancestors + assert 'aria-label="Filter atlas by skill family"' in html + assert "All families" in html + assert "function selectFamily(" in html + assert "function familyMatches(" in html + assert "familyMatches(s.name)" in html + assert "d.clusters.map((cluster, index) => ({cluster, index}))" in html + assert '$("#cluster-chips").innerHTML = "";' in html + assert "clustersAttempted" in html + assert "select.disabled = !clusterData?.clusters?.length" in html + assert '$("#nav-clusters").textContent = "0";' in html + assert '$("#skill-family-filter").onchange' in html + + +def test_trace_inventory_api_never_returns_answers(client, tmp_path, monkeypatch): + import ui.app as ui_app + + store = tmp_path / "local-traces.json" + store.write_text(json.dumps({ + "schema_version": "ingot/local-traces/v1", + "generated_at": 1785254400, + "traces": [{ + "id": "trace-1", "timestamp": "2026-07-28T10:00:00Z", "harness": "claude", + "task": "Review the interface", "answer": "private answer", + "skills": [{"name": "saas-interface-review", "revision": None}], + "tags": ["skill:saas-interface-review"], "usage": {"input_tokens": 10, + "output_tokens": 4}, + }], + })) + monkeypatch.setattr(ui_app, "LOCAL_TRACE_FILE", store) + + response = client.get("/api/traces") + + assert response.status_code == 200 + payload = response.json() + assert payload["total"] == 1 + assert "task" not in payload["recent"][0] + assert "answer" not in payload["recent"][0] + assert "private answer" not in response.text + + preview = client.get("/api/traces?include_tasks=true") + assert preview.json()["recent"][0]["task"] == "Review the interface" + assert "private answer" not in preview.text + + filtered = client.get("/api/traces?project=other&since=2026-07-28") + assert filtered.status_code == 200 + assert filtered.json()["total"] == 0 + assert client.get("/api/traces?since=not-a-date").status_code == 400 + + +def test_trace_inventory_is_a_routed_console_view(client): + html = client.get("/").text + layout = _Layout(html) + + assert "traces-section" in layout.ancestors + assert "trace-summary" in layout.ancestors + assert "trace-list" in layout.ancestors + assert 'data-route="traces"' in html + assert "j(traceUrl())" in html + assert 'id="trace-project"' in html + assert 'id="trace-since"' in html + assert 'id="trace-task-previews"' in html + assert 'notation: "compact"' in html + + def test_comparison_panel_orders_tokens_and_tables_numbered_task_scores(client): html = client.get("/").text compare = html[html.index("function buildCompare(p)"):html.index("function openCompare()")] @@ -346,13 +783,14 @@ def test_comparison_panel_orders_tokens_and_tables_numbered_task_scores(client): assert 'class="cmp-pertask"' not in compare -def test_api_skills_rows_carry_a_load_count(client, monkeypatch): +def test_api_skills_rows_carry_a_load_count(client, monkeypatch, tmp_path): """Every active skill row exposes `uses` so the UI can render the load-counter chip.""" import ui.app as ui_app - from mcp_server import usage_counts + from ingot.mcp_server import usage_counts class _Skill: name, description, revision = "pdf", "merge PDFs", "rev1" + root = str(tmp_path / "library" / "pdf") # provenance classifies from the skill's own root monkeypatch.setattr(ui_app, "load_skills", lambda: [_Skill()]) monkeypatch.setattr(usage_counts, "load_counts", lambda: {"pdf": 7}) active = client.get("/api/skills").json() @@ -368,6 +806,36 @@ def test_pending_without_search_scores_still_renders(client): assert client.get("/api/pending/pdf").json()["inner_loop"] is None +def test_pending_exposes_retrospective_evidence(client): + P.save_pending("pdf", { + "skill": "pdf", "kind": "retrospective", + "champion_components": {"description": "d", "body": "a"}, + "challenger_components": {"description": "d", "body": "b"}, + "changed_components": ["body"], + "retrospective": { + "summary": "Repeated omission.", "trigger": "Two matching failures.", + "minimal_content": "Add the missing guard.", "producer": "skill-retrospective", + "caller": "build-loop", "evidence": ["run one", "run two"], + "pressure_scenario": "A rushed repair.", "risk": "May slow simple work.", + "verification": {"status": "passed", "command": "pytest", "result": "passed"}, + }, + }) + + payload = client.get("/api/pending/pdf").json() + assert payload["kind"] == "retrospective" + assert payload["retrospective"]["producer"] == "skill-retrospective" + assert payload["retrospective"]["verification"]["status"] == "passed" + + +def test_index_renders_retrospective_evidence_on_both_decision_surfaces(client): + html = client.get("/").text + assert "function retrospectiveEvidence(" in html + assert "p.retrospective" in html + assert "Pressure scenario" in html + assert "Verification" in html + assert "border: 1px solid transparent; overflow: hidden;" in html + + def test_promote_passes_through_result(client, monkeypatch): import ui.app as ui_app P.save_pending("pdf", {"skill": "pdf", "gate": {"promotable": True, "blocked": []}, @@ -377,6 +845,170 @@ def test_promote_passes_through_result(client, monkeypatch): assert r.status_code == 200 and r.json() == {"result": "promoted 'pdf'"} +def test_pending_exposes_approved_publication_and_disables_repeat_approval(client): + pending = { + "skill": "pdf", "gate": {"promotable": True, "blocked": []}, + "champion_components": {"description": "Merge PDFs.", "body": "old"}, + "challenger_components": {"description": "Merge PDFs.", "body": "new"}, + "evidence": {"champion": {"revision": "a" * 64}, + "challenger": {"revision": "b" * 64}}, + } + P.save_pending("pdf", pending) + Q.queue_publication("pdf", pending, "admin", "promote") + + payload = client.get("/api/pending/pdf").json() + assert payload["publication"]["state"] == "approved_publishing" + html = client.get("/").text + assert "Approved · publishing to vault" in html + assert "p.publication" in html + assert '$("#approve").disabled' in html + + +def test_pending_does_not_attach_a_receipt_from_an_older_proposal(client): + old = { + "skill": "pdf", "kind": "retrospective", + "champion_components": {"description": "Merge PDFs.", "body": "old"}, + "challenger_components": {"description": "Merge PDFs.", "body": "first change"}, + "evidence": {"champion": {"revision": "a" * 64}, + "challenger": {"revision": "b" * 64}}, + "retrospective": {"proposal_id": "old-proposal"}, + } + receipt = Q.queue_publication("pdf", old, "admin", "promote") + Q.update_publication(receipt.id, state="active") + P.save_pending("pdf", { + **old, + "challenger_components": {"description": "Merge PDFs.", "body": "second change"}, + "evidence": {"champion": {"revision": "b" * 64}, + "challenger": {"revision": "c" * 64}}, + "retrospective": {"proposal_id": "new-proposal"}, + }) + + payload = client.get("/api/pending/pdf").json() + + assert payload["publication"] is None + + +def _queued(skill: str, **changes): + pending = { + "skill": skill, "gate": {"promotable": True, "blocked": []}, + "champion_components": {"description": "Merge PDFs.", "body": "old"}, + "challenger_components": {"description": "Merge PDFs.", "body": "new"}, + "evidence": {"champion": {"revision": "a" * 64}, + "challenger": {"revision": "b" * 64}}, + } + receipt = Q.queue_publication(skill, pending, "admin", "promote") + return Q.update_publication(receipt.id, **changes) if changes else receipt + + +def _proposed(monkeypatch, tmp_path, skill="copywriting"): + """A creation pending, which is what the board surfaces when no library is indexed.""" + import ui.app as U + monkeypatch.setenv("SKILL_ROUTER_PATHS", str(tmp_path / "empty-library")) + P.save_pending(skill, { + "skill": skill, "kind": "creation", + "champion_components": {"description": "", "body": ""}, + "challenger_components": {"description": "Write conversion copy.", "body": "Body"}, + "gate": {"promotable": True, "blocked": [], "kind": "new_skill_admission"}, + "creation": {"summary": "Add vetted copywriting skill."}, + }) + U._SKILLS_CACHE = None + U._SKILLS_CACHE_KEY = None + + +def test_skills_listing_separates_publishing_from_awaiting_review(client, tmp_path, monkeypatch): + """Approval does not free the review slot, so an approved change stays `pending`. Without a + second flag the board counts it as still awaiting a decision the reviewer already made — which + is what made an approved creation sit in `To be added` looking untouched.""" + _proposed(monkeypatch, tmp_path) + before = {s["name"]: s for s in client.get("/api/skills").json()}["copywriting"] + _queued("copywriting", state="awaiting_merge", pr=9) + + after = {s["name"]: s for s in client.get("/api/skills").json()}["copywriting"] + + assert (before["pending"], before["publishing"]) == (True, False) + assert (after["pending"], after["publishing"]) == (True, True) + + +def test_a_finished_publication_stops_marking_its_skill_as_publishing(client, tmp_path, monkeypatch): + """Only the newest receipt counts, or an earlier attempt would pin the skill to `publishing`.""" + _proposed(monkeypatch, tmp_path) + _queued("copywriting", state="active", pr=9) + + listed = {s["name"]: s for s in client.get("/api/skills").json()}["copywriting"] + + assert listed["publishing"] is False + + +def test_publications_lane_is_empty_before_anything_is_approved(client): + payload = client.get("/api/publications").json() + + assert payload["publications"] == [] + assert payload["unreadable"] is None + assert "awaiting_merge" in payload["live_states"] + + +def test_publications_lane_survives_the_pending_record_it_came_from(client, monkeypatch): + """The whole point of the lane: `publication_for_skill` needs a pending record, and approval + consumes it, so once a change is approved the console could no longer see it travelling.""" + monkeypatch.setenv("INGOT_FORGE_REPOSITORY", "someone/skills") + _queued("pdf", state="awaiting_merge", pr=9) + + payload = client.get("/api/publications").json() + + assert [r["skill"] for r in payload["publications"]] == ["pdf"] + assert payload["publications"][0]["state"] == "awaiting_merge" + assert payload["publications"][0]["pr_url"] == "https://github.com/someone/skills/pull/9" + + +def test_publications_lane_omits_a_pull_request_url_when_there_is_no_pull_request(client): + _queued("pdf") + + assert client.get("/api/publications").json()["publications"][0]["pr_url"] is None + + +def test_publications_lane_links_no_pull_request_under_the_local_backend(client, monkeypatch): + """Only the forge backend has a pull request, and only it knows the repository. Linking to a + hardcoded one produced a 404 for every deployment that was not the author's.""" + monkeypatch.delenv("INGOT_FORGE_REPOSITORY", raising=False) + _queued("pdf", state="awaiting_merge", pr=9) + + assert client.get("/api/publications").json()["publications"][0]["pr_url"] is None + + +def test_publications_lane_reports_a_receipt_store_it_cannot_read(client, monkeypatch): + """`Path.glob` swallows `PermissionError`, so an unreadable store returns an empty list — + indistinguishable from a quiet lane, which is the reading a stalled publisher most invites.""" + _queued("pdf", state="awaiting_merge", pr=9) + monkeypatch.setattr(A.os, "access", lambda *a, **k: False) + + payload = client.get("/api/publications").json() + + assert payload["publications"] == [] + assert "cannot read the receipt store" in payload["unreadable"] + + +def test_pending_names_the_pull_request_a_stalled_publication_waits_on(client): + """A vault that cannot auto-merge leaves the receipt waiting on a person. Without the pull + request number on the card, the reviewer has no way to learn they are what it waits for.""" + pending = { + "skill": "pdf", "gate": {"promotable": True, "blocked": []}, + "champion_components": {"description": "Merge PDFs.", "body": "old"}, + "challenger_components": {"description": "Merge PDFs.", "body": "new"}, + "evidence": {"champion": {"revision": "a" * 64}, + "challenger": {"revision": "b" * 64}}, + } + P.save_pending("pdf", pending) + receipt = Q.queue_publication("pdf", pending, "admin", "promote") + Q.update_publication(receipt.id, state="awaiting_merge", pr=4, auto_merge=False, + note="auto-merge unavailable, waiting on a human merge: denied") + + payload = client.get("/api/pending/pdf").json() + + assert payload["publication"]["pr"] == 4 + assert "waiting on a human merge" in payload["publication"]["note"] + assert "merge vault PR #" in client.get("/").text + + def test_cross_origin_promote_and_reject_refused(client): for endpoint in ("/api/promote/pdf", "/api/reject/pdf"): assert client.post(endpoint, headers={"origin": "http://evil.example"}).status_code == 403 @@ -410,7 +1042,7 @@ def test_pending_routing_pass_renders_without_ab(client): def test_optimize_surfaces_pin_conflicts_as_400(client, monkeypatch): - import optimize + import ingot.optimize as optimize def conflict(): raise SystemExit("error: provider pin conflicts detected before spending any tokens:\n MODEL=x: nope") monkeypatch.setattr(optimize, "preflight_provider_pins", conflict) @@ -420,11 +1052,11 @@ def conflict(): def test_skills_api_reports_eval_status_for_all_skills(client, tmp_path, monkeypatch): # the UI's evals chip keys off has_tasks, every skill must carry it, task set or not - from mcp_server import registry - from mcp_server.registry import write_skill_md + from ingot.mcp_server import registry + from ingot.mcp_server.registry import write_skill_md import ui.app as U for name in ("with-evals", "without-evals"): - d = registry.SKILLS_DIR / name # hermetic per-test root (conftest) + d = registry.library_dir() / name # hermetic per-test root (conftest) d.mkdir(parents=True) write_skill_md(d / "SKILL.md", {"name": name, "description": "d"}, "b") tasks = tmp_path / "tasks" @@ -435,12 +1067,50 @@ def test_skills_api_reports_eval_status_for_all_skills(client, tmp_path, monkeyp assert flags == {"with-evals": True, "without-evals": False} -def test_index_ships_eval_chips_and_disabled_candidate_run(client): +def test_index_ships_eval_creation_for_skills_without_tasks(client): + html = client.get("/").text + assert "no evals" in html + assert "has_tasks" in html + assert "Create eval set" in html + assert "createEvalSet" in html and "/api/evals/" in html + assert "auto-drafts" not in html + assert "Optimize with SkillOpt" in html + + +def test_index_surfaces_incomplete_eval_coverage_not_only_zero_coverage(client): html = client.get("/").text - assert "no evals" in html # chip for skills without an eval task set - assert "has_tasks" in html # rendering keys off the API flag - assert "auto-drafts" in html # the disabled generate button explains how to get evals - assert "Optimize with SkillOpt" in html # optimization is a first-class, human-gated workflow + assert "EVAL COVERAGE" in html + assert "without an eval task set" in html + assert "withTasks.length < activeSkills.length" in html + + +def test_index_labels_eval_drafting_separately_from_optimization(client): + html = client.get("/").text + assert 'run?.action === "eval"' in html + assert 'const runLabel = drafting ? "Eval draft" : reviewing ? "Current-skill review" : "SkillOpt"' in html + assert "${runLabel} ${esc(run.status" in html + assert "Drafting eight train/holdout tasks" in html + + +def test_index_exposes_eval_inputs_and_current_skill_review(client): + html = client.get("/").text + layout = _Layout(html) + for element_id in ("skill-evals", "skill-eval-summary", "skill-eval-groups", + "skill-review", "skill-review-run", "skill-eval-msg"): + assert "skill-overlay" in layout.ancestors[element_id] + assert "/api/evals/" in html + assert "/api/reviews/" in html + assert "Run review" in html + assert "Train tasks" in html and "Held-out tasks" in html + assert "Routing cases" in html and "Acceptance rules" in html + assert "Failed checks" in html + + +def test_index_labels_review_runs_separately_from_drafting_and_optimization(client): + html = client.get("/").text + assert 'run?.action === "review"' in html + assert '"Current-skill review"' in html + assert "Review can take a minute" in html def test_index_leads_with_review_before_candidate_generation(client): @@ -449,7 +1119,7 @@ def test_index_leads_with_review_before_candidate_generation(client): assert html.index('id="review-section"') < html.index('id="history-section"') assert html.index('id="history-section"') < html.index('id="skills"') assert 'id="run-section"' not in html - assert "Evidence-gated change control" in html + assert "Release control for agent skills" in html assert "change control" in html and "skill optimizer" not in html @@ -481,15 +1151,16 @@ def test_carn_viewer_is_gone(client): def _promoted_skill(tmp_path, monkeypatch): """An active skill with one approved promotion behind it, so a snapshot exists to restore.""" - from mcp_server.registry import optimizable_components, skill_revision + from ingot.mcp_server.registry import optimizable_components, skill_revision root = tmp_path / "skills" skill = root / "pdf" skill.mkdir(parents=True) (skill / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge PDFs.\n---\napproved body\n") + monkeypatch.setenv("INGOT_LIBRARY", str(root)) monkeypatch.setenv("SKILL_ROUTER_PATHS", str(root)) champion = optimizable_components(skill) challenger = {**champion, "body": "promoted body"} - from mcp_server.registry import load_skills + from ingot.mcp_server.registry import load_skills current = load_skills(root)[0] gate = {"promotable": True, "blocked": []} P.save_pending("pdf", { @@ -499,7 +1170,7 @@ def _promoted_skill(tmp_path, monkeypatch): "challenger": {"revision": skill_revision(skill, challenger)}, "gate": gate}, }) - P.approve_pending("pdf") + P._activate_approved("pdf", P.load_pending("pdf")) return skill, current.revision @@ -526,7 +1197,7 @@ def test_history_does_not_rescan_the_skill_library(client, tmp_path, monkeypatch The counter patches the registry's own library scan, which `load_skills` looks up at call time: counting `ui.app.load_skills` would have missed a rescan reached through any other module's import of it, and passed whether or not history scanned anything.""" - from mcp_server import registry + from ingot.mcp_server import registry _promoted_skill(tmp_path, monkeypatch) real_sources = registry.skill_sources scans = [] @@ -583,16 +1254,20 @@ def fail(name): assert [r["action"] for r in history["audit"]["records"]] == ["approve"] -def test_rollback_restores_a_snapshot_and_records_it(client, tmp_path, monkeypatch): +def test_rollback_queues_a_snapshot_for_vault_publication(client, tmp_path, monkeypatch): + """History rollback takes the same Git lane as approval: it reports publication, and the + served skill only changes once the vault merge lands.""" skill, replaced = _promoted_skill(tmp_path, monkeypatch) assert "promoted body" in (skill / "SKILL.md").read_text() r = client.post(f"/api/rollback/pdf/{replaced}") - assert r.status_code == 200 and "Rolled back" in r.json()["result"] - assert "approved body" in (skill / "SKILL.md").read_text() + assert r.status_code == 200 and "publishing to vault" in r.json()["result"] + assert "promoted body" in (skill / "SKILL.md").read_text() + record = Q.publication_for_skill("pdf") + assert record["action"] == "rollback" and record["candidate_revision"] == replaced trail = client.get("/api/history").json()["audit"]["records"] - assert [a["action"] for a in trail] == ["rollback", "approve"] + assert [a["action"] for a in trail] == ["approve"] def test_rollback_rejects_unknown_revision_and_bad_names(client, tmp_path, monkeypatch): @@ -603,7 +1278,7 @@ def test_rollback_rejects_unknown_revision_and_bad_names(client, tmp_path, monke def test_rollback_refuses_a_traversing_revision_at_the_application(client, tmp_path, monkeypatch): """A `..` segment must be refused by revision validation, not merely missed by the router: - the same string reaching optimize.promote directly has to be rejected there too.""" + the same string reaching ingot.optimize.promote directly has to be rejected there too.""" _promoted_skill(tmp_path, monkeypatch) r = client.post("/api/rollback/pdf/%2E%2E", follow_redirects=False) @@ -683,7 +1358,7 @@ def slow_approve(skill, actor="?"): def _stale_pending(tmp_path, monkeypatch): """A review slot whose champion has since been edited on disk.""" - from mcp_server.registry import load_skills, optimizable_components, skill_revision + from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision root = tmp_path / "skills" skill = root / "pdf" skill.mkdir(parents=True) @@ -722,7 +1397,7 @@ def test_promote_with_stale_evidence_is_409(client, tmp_path, monkeypatch): def test_pending_is_not_stale_for_a_fresh_change(client, tmp_path, monkeypatch): - from mcp_server.registry import load_skills, optimizable_components, skill_revision + from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision root = tmp_path / "skills" skill = root / "pdf" skill.mkdir(parents=True) @@ -749,8 +1424,8 @@ def _evidence_bundle(monkeypatch, tmp_path, recorded=None, body="# Behavioral Sk bundle = evidence_root / "pdf" / "1700000000" bundle.mkdir(parents=True) (bundle / "EVIDENCE.md").write_text(body) - monkeypatch.setattr(U, "REPO_ROOT", tmp_path.resolve()) - monkeypatch.setattr(U, "EVIDENCE_DIR", evidence_root) + monkeypatch.setenv("INGOT_RUNS", str(tmp_path / "runs")) + monkeypatch.setattr(U, "STATE_ROOT", tmp_path.resolve()) P.save_pending("pdf", { "skill": "pdf", "champion_components": {}, "challenger_components": {}, "evidence_paths": {"markdown": recorded or "runs/evidence/pdf/1700000000/EVIDENCE.md"}, @@ -861,12 +1536,18 @@ def test_index_follows_the_queue_when_the_reviewed_card_is_gone(client): html = client.get("/").text assert "if (skills) syncReviewCard(skills);" in html assert "!currentPending || !quarantined.includes(currentPending)" in html - assert "showPending(quarantined[0], {keepMessage: true});" in html + assert "showPending((undecided[0] ?? quarantined[0]), {keepMessage: true});" in html + # An approved change keeps its pending record until the vault commit lands, so following the + # queue must skip it rather than reopening a card whose decision is already made. + assert "skills.filter(s => s.pending && !s.publishing).map(s => s.name)" in html assert "if (!quarantined.length) { showNoPending(); return; }" in html # a card opened by hand still clears the previous result assert "if (!keepMessage) say(\"#pending-msg\", \"\", false);" in html assert 'onclick="showPending(\'${esc(s.name)}\', {scroll: true})"' in html - assert 'if (scroll) {' in html and '$("#review-section").scrollIntoView' in html + # Review is its own route now, so reaching it is a hash change rather than a scroll within one + # long page. The card still has to name the skill whose button was clicked: a bare route change + # would land on whichever change the queue happens to list first. + assert 'if (scroll) {' in html and 'location.hash = "#/review"' in html def test_index_renders_the_board_when_history_is_unavailable(client): @@ -900,7 +1581,7 @@ def test_history_payload_is_byte_stable_between_polls(client, tmp_path, monkeypa def test_history_orders_rollback_targets_newest_snapshot_first(client, tmp_path, monkeypatch): """The picker lists most-recently-snapshotted first, so option 0 is the change you just made.""" - from mcp_server.registry import load_skills, optimizable_components, skill_revision + from ingot.mcp_server.registry import load_skills, optimizable_components, skill_revision skill, first = _promoted_skill(tmp_path, monkeypatch) champion = optimizable_components(skill) @@ -913,7 +1594,193 @@ def test_history_orders_rollback_targets_newest_snapshot_first(client, tmp_path, "challenger": {"revision": skill_revision(skill, challenger)}, "gate": gate}, }) second = load_skills(skill.parent)[0].revision - P.approve_pending("pdf") + P._activate_approved("pdf", P.load_pending("pdf")) listed = [r["revision"] for r in client.get("/api/history").json()["revisions"]["pdf"]] assert listed == [second, first] + + +def test_publications_reports_a_quarantined_change_it_cannot_read(client): + """list_pending skips an unreadable record so one corrupt file cannot break review, which also + means an unreadable proposal is indistinguishable from no proposal and the board reports CLEAR + over it. Observed live: the MCP container wrote records as root 0600 while the UI ran as uid + 1000, and an approved-and-waiting skill sat invisible for hours.""" + P.pending_dir().mkdir(parents=True, exist_ok=True) + (P.pending_dir() / "measurement-integrity.json").write_bytes(b"\xff\xfe not json") + + payload = client.get("/api/publications").json() + + assert payload["pending_blocked"], "an unreadable quarantined change must be surfaced" + assert "measurement-integrity.json" in payload["pending_blocked"] + + +def test_publications_stays_quiet_when_the_review_queue_is_readable(client): + P.pending_dir().mkdir(parents=True, exist_ok=True) + assert client.get("/api/publications").json()["pending_blocked"] is None + + +def test_a_non_utf8_pending_file_does_not_take_down_the_review_page(client): + """list_pending caught OSError and JSONDecodeError but not UnicodeDecodeError, so a binary file + in the queue raised straight through and the whole review surface 500ed.""" + P.pending_dir().mkdir(parents=True, exist_ok=True) + (P.pending_dir() / "pdf.json").write_bytes(b"\xff\xfe\x00binary") + assert client.get("/api/skills").status_code == 200 + + +def test_harbor_matrix_is_served_for_a_skill_that_has_been_run(client, monkeypatch, tmp_path): + """The console surface for the harness x model grid.""" + root = tmp_path / "harbor" + root.mkdir() + (root / "pdf.json").write_text(json.dumps({"judge": "google/gemini-2.5-flash", "harnesses": { + "claude-code@anthropic/claude-opus-5": { + "skill_mean": 0.75, "control_mean": 0.5, "lift": 0.25, "tasks_scored": 4, + "tasks_dropped": [], "endpoint_url": "https://private.invalid/v1"}, + "aider@openai/gpt-5.5": {"error": "RuntimeError: every task returned an empty workspace"}}})) + monkeypatch.setattr(harbor_report, "HARBOR_DIR", root) + + payload = client.get("/api/harbor/pdf").json() + + assert payload["measured"] == 1 and payload["unmeasured"] == 1 + rows = {r["combination"]: r for r in payload["rows"]} + # The rule the whole surface exists for: a combination that did not run reaches the browser + # with no lift key at all, so no renderer can put a number in its measurement column. + assert "lift" not in rows["aider@openai/gpt-5.5"] + assert rows["claude-code@anthropic/claude-opus-5"]["n"] == 4 + # A renderer may show the recorded alias/protocol, never an endpoint URL supplied by a run. + assert "endpoint_url" not in rows["claude-code@anthropic/claude-opus-5"] + assert client.get("/api/harbor").json()["skills"] == ["pdf"] + + +def test_harbor_page_pivots_sparse_evidence_by_harness_and_model(client): + """The console gives every axis intersection its own evidence state, never an implied zero.""" + html = client.get("/").text + + # renderHarbor owns the pivot from API axes, rather than relying on a pre-filled rectangular + # payload. The API deliberately sends only rows that were attempted. + assert "function matrixCell(row)" in html + assert "data.harnesses" in html and "data.models" in html + assert "byHarness" in html and "byModel" in html + # These loops are the rectangular matrix contract. Axis/map names alone would let a renderer + # emit one header or skip sparse intersections, which silently changes absence into no cell. + assert "${models.map(model => {" in html + assert "const body = harnesses.map(harness =>" in html + assert "models.map(model => matrixCell(byHarness.get(harness)?.get(model))).join(\"\")" in html + assert "never run" in html + assert "not measured" in html + assert "toFixed(3)" in html + assert "skill mean" in html and "control mean" in html + assert "attempts" in html and "dropped" in html + assert "target alias" in html and "protocol" in html and "error" in html + # Measured and failed cells are buttons, so keyboard and pointer activation share one route; + # a blank intersection is evidence of absence, not a neutral interactive result. + assert 'class="mx-plate mx-measured' in html + assert 'class="mx-plate mx-error"' in html + assert "mx-blank" in html + assert 'aria-label="never run"' in html + assert "mx-details" in html and 'aria-live="polite"' in html + + +def test_harbor_matrix_css_contract_preserves_readable_sparse_columns(client): + html = client.get("/").text + + assert ".mx-wrap" in html and "overflow-x: auto" in html + assert ".mx-sticky" in html and "position: sticky" in html + assert ".mx-corner" in html + assert ".mx td, .mx th" in html and "min-width:" in html + # Per-model warnings wrap inside a fixed evidence column; otherwise one ceiling warning can + # stretch a sparse matrix across several screens. + assert "table-layout: fixed" in html and "--mx-width" in html + assert ".mx-plate:focus-visible" in html + assert ".mx-blank" in html and ".mx-error" in html and ".mx-measured" in html + + +def test_harbor_page_plots_only_observed_model_scale_rows(client): + html = client.get("/").text + + assert "function sizeLiftChart(rows)" in html + assert "Math.log10" in html and "Number.isFinite(size)" in html and "size > 0" in html + assert "sizeLiftChart(rows)" in html + assert "data.legacy" not in html + assert "data-size-point" in html and "generationShape" in html + assert "Model size vs lift" in html and "Parameters (billions, log scale)" in html + assert "Object.is(rounded, -0) ? 0 : rounded" in html + assert "Failed and never-run cells remain in the evidence ledger below." in html + assert 'tabindex="0"' in html and "event.key === \"Enter\"" in html + assert "parameter_billions" in html and "quantization" in html and "tool parser" in html + assert 'includes("Qwen3.5") ? "circle"' in html + assert 'includes("Qwen3.6") ? "square" : "diamond"' in html + assert "diamond = other model families" in html + + +def test_harbor_size_chart_keeps_endpoint_swarms_inside_the_plot(client): + html = client.get("/").text + + assert "const swarmInset = Math.max(...sizes.map(size =>" in html + assert "left + swarmInset" in html + assert "plotRight - swarmInset" in html + + +def test_secondary_routes_put_their_own_job_first(client): + html = client.get("/").text + + assert 'document.querySelector(".board-head").hidden = base !== "review"' in html + assert ".board-head[hidden]" in html + + +def test_every_route_uses_the_evidence_cockpit_visual_system(client): + """The full-console redesign must alter the shell, not only a chart inside the old page.""" + html = client.get("/").text + + assert 'class="topbar cockpit-command"' in html + assert 'class="shell cockpit-shell"' in html + assert 'class="sidebar cockpit-rail"' in html + assert 'class="main cockpit-workspace"' in html + assert "/* ---- evidence cockpit visual system ---- */" in html + assert ".cockpit-rail .navlink.active::before" in html + assert ".cockpit-workspace > .view:not([hidden])" in html + assert ".cockpit-workspace .review-card" in html + assert ".cockpit-workspace .trace-list" in html + assert ".cockpit-workspace .mx-wrap" in html + assert ".cockpit-shell { grid-template-columns: 1fr; }" in html + assert ".cockpit-rail #nav-folders { display: contents; }" in html + + +def test_harbor_size_chart_has_a_persistent_readable_interaction_layer(client): + html = client.get("/").text + + assert "mx-chart-stats" in html + assert "mx-chart-legend" in html + assert 'id="mx-chart-detail"' in html + assert "pointOffset" in html and "* 18" in html + assert "Measured evidence only" in html + assert 'row.n == null ? "not recorded"' in html + assert 'aria-live="polite"' in html + + +def test_harbor_poll_discovers_and_rerenders_progressive_results(client): + html = client.get("/").text + + assert 'if (currentRoute() === "harnesses") await loadHarbor();' in html + assert 'await j("/api/harbor")' in html + assert "harborSkills = available" in html + assert "const selected = harborSkill" in html + + +def test_a_skill_never_run_across_harnesses_is_a_404_not_an_empty_matrix(client, monkeypatch, tmp_path): + """An empty grid on the page would read as 'no combination helps'. It has not been measured.""" + monkeypatch.setattr(harbor_report, "HARBOR_DIR", tmp_path / "harbor") + response = client.get("/api/harbor/pdf") + assert response.status_code == 404 + assert "harbor_eval" in response.json()["detail"] + + +def test_an_unreadable_matrix_is_an_error_not_a_silent_absence(client, monkeypatch, tmp_path): + root = tmp_path / "harbor" + root.mkdir() + (root / "pdf.json").write_text("{ not json") + monkeypatch.setattr(harbor_report, "HARBOR_DIR", root) + assert client.get("/api/harbor/pdf").status_code == 503 + + +def test_harbor_rejects_a_bad_skill_name(client): + assert client.get("/api/harbor/..%2Fetc").status_code in (400, 404) diff --git a/tests/test_usage.py b/tests/test_usage.py index 7480444..0cf6af1 100644 --- a/tests/test_usage.py +++ b/tests/test_usage.py @@ -1,7 +1,7 @@ -"""Unit tests for the per-run token ledger (optimize.usage), including thread-safety.""" +"""Unit tests for the per-run token ledger (ingot.optimize.usage), including thread-safety.""" import threading -from optimize import usage +from ingot.optimize import usage def test_add_accumulates_per_role_and_totals(): @@ -51,3 +51,26 @@ def test_format_report_is_readable(): usage.add("judge", {"input_tokens": 1234, "output_tokens": 56}) out = usage.format_report() assert "judge" in out and "TOTAL" in out and "1,234" in out + + +def test_subscription_usage_is_counted_without_cost_or_spend_cap(monkeypatch): + usage.reset() + monkeypatch.delenv("BASE_URL", raising=False) + monkeypatch.delenv("OPENROUTER_BASE_URL", raising=False) + monkeypatch.setenv("JUDGE_MODEL", "m/metered-judge") + monkeypatch.setenv("MAX_RUN_USD", "0.01") + monkeypatch.setattr(usage, "_PRICES", {"m/metered-judge": (1.0, 1.0)}) + + usage.add( + "judge", + {"input_tokens": 11, "output_tokens": 7}, + billing_mode="subscription", + ) + + assert usage.report()["judge"] == {"input": 11, "output": 7, "calls": 1} + assert usage.estimated_cost() is None + assert usage.unpriced_roles() == [] + report = usage.format_report() + assert "judge" in report and "subscription" in report + assert "$" not in report + usage.reset() diff --git a/tests/test_usage_counts.py b/tests/test_usage_counts.py index e6535a6..ff5d822 100644 --- a/tests/test_usage_counts.py +++ b/tests/test_usage_counts.py @@ -1,5 +1,5 @@ """Unit tests for the per-skill load counter (no server needed).""" -from mcp_server import usage_counts +from ingot.mcp_server import usage_counts def test_record_use_increments_and_persists(tmp_path, monkeypatch): diff --git a/tests/test_vault.py b/tests/test_vault.py new file mode 100644 index 0000000..a5b7297 --- /dev/null +++ b/tests/test_vault.py @@ -0,0 +1,109 @@ +"""`ingot vault init`: the one bootstrap command a local deployment needs. + +`ingot status`, the other half of the managed-deployment surface, is in test_status.py.""" +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +from ingot import cli, vault + + +def _git(path, *args): + return subprocess.run(["git", "-C", str(path), *args], check=True, + capture_output=True, text=True).stdout.strip() + + +# --------------------------------------------------------------------------- vault init + +def test_init_creates_a_committed_vault_on_main(tmp_path): + result = vault.init_vault(tmp_path / "vault") + + created = tmp_path / "vault" + assert result["status"] == "created" + assert result["branch"] == "main" + assert (created / "registry.json").read_text() == "{}\n" + assert (created / "scripts" / "validate.py").is_file() + assert _git(created, "status", "--porcelain") == "" + assert _git(created, "rev-parse", "HEAD") == result["head"] + + +def test_init_is_idempotent(tmp_path): + first = vault.init_vault(tmp_path / "vault") + second = vault.init_vault(tmp_path / "vault") + + assert second["status"] == "unchanged" + assert second["head"] == first["head"] + assert second["added"] == [] + + +def test_init_completes_an_existing_repository_without_rewriting_it(tmp_path): + """Adopting a Git repository someone already keeps skills in must add what is missing and + touch nothing else.""" + existing = tmp_path / "vault" + existing.mkdir() + subprocess.run(["git", "init", "-b", "main", str(existing)], check=True, capture_output=True) + _git(existing, "config", "user.email", "owner@test.invalid") + _git(existing, "config", "user.name", "Owner") + (existing / "registry.json").write_text('{"pdf": {"disposition": "keep"}}\n') + _git(existing, "add", ".") + _git(existing, "commit", "-m", "Existing vault") + + result = vault.init_vault(existing) + + assert result["status"] == "updated" + assert "scripts/validate.py" in result["added"] + assert json.loads((existing / "registry.json").read_text()) == {"pdf": {"disposition": "keep"}} + + +def test_init_refuses_a_non_empty_directory_that_is_not_a_repository(tmp_path): + """Adopting a directory of loose skills would make the first commit look like a publication + nobody approved.""" + loose = tmp_path / "skills" + (loose / "pdf").mkdir(parents=True) + (loose / "pdf" / "SKILL.md").write_text("---\nname: pdf\ndescription: x\n---\nbody\n") + + with pytest.raises(ValueError, match="not empty and is not a Git repository"): + vault.init_vault(loose) + + +def test_the_default_validator_refuses_a_tree_the_server_could_not_serve(tmp_path): + created = tmp_path / "vault" + vault.init_vault(created) + (created / "pdf").mkdir() + (created / "pdf" / "SKILL.md").write_text("---\nname: other\ndescription: Merge.\n---\nbody\n") + + result = subprocess.run([sys.executable, "scripts/validate.py"], cwd=created, + capture_output=True, text=True) + + assert result.returncode == 1 + assert "!= directory 'pdf'" in result.stderr + + +def test_the_default_validator_accepts_a_well_formed_tree(tmp_path): + created = tmp_path / "vault" + vault.init_vault(created) + (created / "pdf").mkdir() + (created / "pdf" / "SKILL.md").write_text("---\nname: pdf\ndescription: Merge.\n---\nbody\n") + + assert subprocess.run([sys.executable, "scripts/validate.py"], cwd=created, + capture_output=True).returncode == 0 + + +# --------------------------------------------------------------------------- the command line + +def test_vault_init_from_the_command_line(tmp_path, capsys): + assert cli.main(["vault", "init", str(tmp_path / "vault"), "--json"]) == 0 + + payload = json.loads(capsys.readouterr().out) + assert payload["schema_version"] == vault.VAULT_SCHEMA + assert Path(payload["path"]) == (tmp_path / "vault").resolve() + + +def test_vault_init_reports_a_refusal_without_a_traceback(tmp_path, capsys): + (tmp_path / "loose.txt").write_text("not a vault") + + assert cli.main(["vault", "init", str(tmp_path)]) == 1 + assert "ingot vault init:" in capsys.readouterr().err diff --git a/ui/app.py b/ui/app.py index ad3746e..c797a04 100644 --- a/ui/app.py +++ b/ui/app.py @@ -2,31 +2,45 @@ Reviewers see the evidence and the promotion decision first. SkillOpt optimization is a core workflow that only ever writes to the pending queue. Promotion and -rollback both go through `optimize.promote`, which snapshots the displaced revision and swaps +rollback both go through `ingot.optimize.promote`, which snapshots the displaced revision and swaps directories atomically. """ import difflib +import json import logging import os +import tempfile import threading +from collections.abc import Callable from contextlib import contextmanager from pathlib import Path from urllib.parse import urlparse +import yaml from fastapi import Depends, FastAPI, HTTPException, Request from fastapi.responses import FileResponse, RedirectResponse from pydantic import BaseModel, Field -from mcp_server.registry import SLUG_RE, load_skills, read_components, skill_revision -from optimize.ab import TASKS_DIR, run_ab +from ingot.mcp_server.registry import SLUG_RE, load_skills, read_components, skill_revision +from ingot.optimize import harbor_report, resolve_skill_dir +from ingot.optimize.ab import TASKS_DIR, run_ab +from ingot.optimize.local_traces import LOCAL_TRACE_FILE, store_summary from ui.auth import (auth_mode, current_actor, require_auth, require_role, using_default_password) -from optimize.promote import (_audit_best_effort, approve_pending, list_revisions, - list_snapshotted_skills, load_pending, load_snapshot_components, - pending_path, read_audit, rollback, stale_evidence_reason) +from ingot.optimize.promote import (ABSENT_REVISION, _audit_best_effort, approve_pending, list_pending, list_revisions, + list_snapshotted_skills, load_pending, load_snapshot_components, + pending_path, read_audit, reject_pending, rollback, + stale_evidence_reason) +from ingot.optimize.publication import (publication_for_skill, publishing_skills, recent_publications) +from ingot import paths logger = logging.getLogger(__name__) -REPO_ROOT = Path(__file__).resolve().parent.parent -EVIDENCE_DIR = (REPO_ROOT / "runs" / "evidence").resolve() +# Recorded evidence locations are written relative to the state root, so this is what they +# resolve against -- not the code's directory, which no longer holds state. +STATE_ROOT = paths.runs().parent + + +def _evidence_dir() -> Path: + return (paths.runs() / "evidence").resolve() # Bundles written inside a container recorded their container-absolute path before evidence # locations became repo-relative. Both forms name the same file from the host checkout. CONTAINER_ROOT = Path("/app") @@ -82,7 +96,8 @@ def auth_me(actor: str = Depends(current_actor)): return {"authenticated": password, "email": "", "name": actor if password else "", "role": "admin" if password else ""} -RUNS: dict[str, dict] = {} # skill -> {"status": running|done|error, "log": [lines]} +RUNS: dict[str, dict] = {} # skill -> {"status": running|done|error, "action": eval|review|optimize, "log": [lines]} +RUN_LOCK = threading.Lock() @app.get("/") @@ -98,13 +113,78 @@ def config(): return {"langfuse_url": os.environ.get("LANGFUSE_PUBLIC_URL", "http://localhost:3100")} +@app.get("/api/traces") +def traces(project: str = "", harness: str = "", skill: str = "", since: str = "", + until: str = "", include_tasks: bool = False): + """Safe console projection of locally normalized coding-agent turns. + + Answers stay in the on-disk store used by the explicit mining command. The browser receives + task text and attribution metadata, never answer text, reasoning, or tool payloads. + """ + try: + return store_summary(LOCAL_TRACE_FILE, project=project, harness=harness, skill=skill, + since=since, until=until, include_tasks=include_tasks) + except ValueError as error: + raise HTTPException(400, str(error)) from error + + +@app.get("/api/harbor") +def harbor_matrices(): + """Skills that have a cross-harness matrix on disk.""" + return {"skills": harbor_report.available()} + + +@app.get("/api/harbor/{skill}") +def harbor_matrix(skill: str): + """The harness x model matrix for one skill: does this skill help this combination, and by how + much, judged in a sandbox against the same held-out tasks with and without the skill. + + 404 when the skill has never been run, so "not measured yet" cannot be read as "no effect". + """ + _check(skill) + try: + matrix = harbor_report.read_matrix(skill) + except ValueError as error: + raise HTTPException(503, str(error)) from error + if matrix is None: + raise HTTPException(404, f"no cross-harness run for '{skill}'; run " + f"`python -m ingot.optimize.harbor_eval {skill} --agent `") + return matrix + + +@app.get("/api/clusters") +def clusters(): + """The category buckets, as last computed. Clustering loads the embedding model, which is too + heavy for a request on the poll path, so it is a command that writes the file and this only + serves it. 404 names the command rather than implying the feature is missing.""" + from ingot.optimize.cluster import CLUSTER_PATH + if not CLUSTER_PATH.exists(): + raise HTTPException(404, "no clusters computed yet; run " + "`docker compose run --rm --entrypoint python optimize " + "-m ingot.optimize.cluster`") + try: + data = json.loads(CLUSTER_PATH.read_text()) + except (OSError, ValueError) as e: # a half-written or hand-edited file is not a 500 + raise HTTPException(503, f"clusters file is unreadable ({e}); re-run ingot.optimize.cluster") + indexed = {s.name for s in _cached_load_skills()} + # Skills come and go between clustering runs. Say so rather than drawing a stale map as fact. + for cluster in data.get("clusters", []): + for member in cluster.get("members", []): + member["missing"] = member["name"] not in indexed + data["stale_members"] = sum(m.get("missing", False) + for c in data.get("clusters", []) for m in c.get("members", [])) + data["unclustered"] = len(indexed) - sum(len(c.get("members", [])) + for c in data.get("clusters", [])) + return data + + _SKILLS_CACHE = None _SKILLS_CACHE_KEY = None def _get_skills_cache_key(): """Fast cache key based on mtime of skill directories and SKILL.md files.""" - from mcp_server.registry import configured_roots, skill_sources + from ingot.mcp_server.registry import configured_roots, skill_sources roots = configured_roots() mtimes = [id(load_skills)] for r in roots: @@ -131,16 +211,37 @@ def _cached_load_skills(): @app.get("/api/skills") def skills(): tasksets = {p.stem for p in TASKS_DIR.glob("*.yaml")} - from mcp_server.usage_counts import load_counts + from ingot.mcp_server.usage_counts import load_counts + from ingot.mcp_server import provenance counts = load_counts() - return [ + # One parsed ledger per root, not per skill: a merged library re-reads the same VENDORED.md + # once for every skill it serves otherwise. + ledgers: dict = {} + # Approval does not free the review slot: the pending record is held until the vault commit + # lands, so an approved change keeps reading as one still awaiting a decision. + publishing = publishing_skills() + active = [ {"name": s.name, "description": s.description, "has_tasks": s.name in tasksets, "pending": load_pending(s.name) is not None, "revision": s.revision, + "publishing": s.name in publishing, "uses": counts.get(s.name, 0), + "provenance": provenance.classify(s.name, s.root, ledgers=ledgers), "status": RUNS.get(s.name, {}).get("status")} for s in _cached_load_skills() if SLUG_RE.fullmatch(s.name) # a non-slug name (hostile frontmatter) can't be optimized anyway ] + active_names = {item["name"] for item in active} + creations = [] + for pending in list_pending(): + if pending.get("kind") != "creation" or pending.get("skill") in active_names: + continue + components = pending.get("challenger_components") or {} + creations.append({"name": pending["skill"], + "description": str(components.get("description", "")), + "has_tasks": False, "pending": True, "revision": "", "uses": 0, + "publishing": pending["skill"] in publishing, + "provenance": "proposed", "status": None, "active": False}) + return active + creations def _active_skill(skill: str): @@ -181,7 +282,7 @@ def skill_versions(skill: str): "revision": _pending_revision(active, pending), "created": pending.get("created")}) versions.extend({"key": item["revision"], "kind": "snapshot", **item} - for item in list_revisions(skill)) + for item in list_revisions(skill) if item["revision"] != ABSENT_REVISION) return {"skill": skill, "description": active.description, "versions": versions} @@ -215,29 +316,126 @@ def optimize(skill: str): result is a quarantined pending record for review.""" _check(skill) _preflight_optimize(skill) - state = RUNS[skill] = {"status": "running", "log": []} + state, log = _start_run(skill, "optimize") + + threading.Thread(target=_run_optimization, args=(skill, state, log), daemon=True).start() + return {"started": skill} + + +def _preflight_optimize(skill: str) -> None: + _preflight_provider() + if not (TASKS_DIR / f"{skill}.yaml").exists(): + raise HTTPException(404, f"no eval task set for '{skill}'") + + +def _start_run(skill: str, action: str) -> tuple[dict, Callable[..., None]]: + # Drafting and optimization share a process-global token ledger and OpenRouter budget. Claim + # the slot under a lock: FastAPI runs sync handlers concurrently, so a check before assignment + # lets two paid requests both pass. + with RUN_LOCK: + if any(s.get("status") == "running" for s in RUNS.values()): + raise HTTPException(409, "a paid model run is already in progress") + state = RUNS[skill] = {"status": "running", "action": action, "log": []} def log(*args): state["log"].append(" ".join(str(a) for a in args)) if len(state["log"]) > 1000: state["log"] = state["log"][-1000:] - threading.Thread(target=_run_optimization, args=(skill, state, log), daemon=True).start() - return {"started": skill} + return state, log -def _preflight_optimize(skill: str) -> None: +@app.post("/api/evals/{skill}", + dependencies=[Depends(same_origin), Depends(require_role("proposer"))]) +def create_eval_set(skill: str): + """Draft the missing train/holdout task set without starting an optimization.""" + _active_skill(_check(skill)) _preflight_provider() - if not (TASKS_DIR / f"{skill}.yaml").exists(): + if (TASKS_DIR / f"{skill}.yaml").exists(): + raise HTTPException(409, f"'{skill}' already has an eval task set") + state, log = _start_run(skill, "eval") + threading.Thread(target=_run_eval_draft, args=(skill, state, log), daemon=True).start() + return {"started": skill} + + +def _eval_payload(skill: str) -> dict: + path = TASKS_DIR / f"{skill}.yaml" + if not path.exists(): raise HTTPException(404, f"no eval task set for '{skill}'") - # one SkillOpt run at a time: the token ledger is process-global, and concurrent runs - # would also contend for the same OpenRouter budget - if any(s.get("status") == "running" for s in RUNS.values()): - raise HTTPException(409, "a SkillOpt run is already in progress") + try: + data = yaml.safe_load(path.read_text()) or {} + if not isinstance(data, dict): + raise ValueError("top level must be a mapping") + train = data.get("train") or data.get("tasks") or [] + explicit_holdout = bool(data.get("holdout")) + holdout = data.get("holdout") or train + routing = data.get("routing") or [] + acceptance = data.get("acceptance") or [] + if not all(isinstance(section, list) + for section in (train, holdout, routing, acceptance)): + raise ValueError("train, holdout, routing, and acceptance must be lists") + except (OSError, ValueError, yaml.YAMLError) as e: + raise HTTPException( + 503, f"eval set is unreadable for '{skill}' ({e}); repair its task file") from e + tasks = list(train) + (list(holdout) if explicit_holdout else []) + return { + "skill": skill, "train": train, "holdout": holdout, "routing": routing, + "acceptance": acceptance, "leakage": not explicit_holdout, + "counts": { + "train": len(train), "holdout": len(holdout), "routing": len(routing), + "acceptance": len(acceptance), + "checks": sum(len(task.get("checklist") or []) + for task in tasks if isinstance(task, dict)), + }, + } + + +@app.get("/api/evals/{skill}") +def eval_set(skill: str): + """The exact task-set inputs the optimizer reads, exposed so its scores stay traceable.""" + _active_skill(_check(skill)) + return _eval_payload(skill) + + +@app.get("/api/reviews/{skill}") +def review_result(skill: str): + """The newest persisted current-skill review. Reviews measure; they never propose or activate.""" + active = _active_skill(_check(skill)) + from ingot.optimize.review import REVIEW_DIR + path = REVIEW_DIR / f"{skill}.json" + if not path.exists(): + raise HTTPException(404, f"no review result for '{skill}'") + try: + data = json.loads(path.read_text()) + if not isinstance(data, dict): + raise ValueError("top level must be an object") + data.setdefault("created", int(path.stat().st_mtime)) + recorded = data.get("revision") + if not isinstance(recorded, str) or not recorded: + data["stale"] = "review did not record a skill revision; run it again" + elif recorded != active.revision: + data["stale"] = "active skill changed since this review ran; run it again" + else: + data["stale"] = None + return data + except (OSError, ValueError) as e: + raise HTTPException(503, f"review result is unreadable ({e}); run the review again") from e + + +@app.post("/api/reviews/{skill}", + dependencies=[Depends(same_origin), Depends(require_role("proposer"))]) +def start_review(skill: str): + """Score the active skill against its evals without creating a pending revision.""" + _active_skill(_check(skill)) + _preflight_provider() + _eval_payload(skill) # refuse missing or malformed measurement inputs before spending + state, log = _start_run(skill, "review") + threading.Thread(target=_run_review, args=(skill, state, log), daemon=True).start() + return {"started": skill} def _preflight_provider() -> None: - from optimize import openrouter_key_missing, preflight_provider_pins + from ingot.optimize import openrouter_key_missing, preflight_provider_pins if openrouter_key_missing(): raise HTTPException(400, "API_KEY is not set, copy .env.example to .env, " "add your key (https://openrouter.ai/keys), and restart the stack " @@ -257,12 +455,47 @@ def _run_optimization(skill: str, state: dict, log) -> None: state["status"] = "error" +def _run_eval_draft(skill: str, state: dict, log) -> None: + try: + from ingot.optimize import usage as usage_ledger + from ingot.optimize.draft import draft_and_save + + usage_ledger.reset() + components = read_components(resolve_skill_dir(skill)) + # The teacher may fail after opening its output. Stage beside the mounted task directory, + # then hard-link the complete file into place so failure cannot create a bogus eval set + # and a concurrent writer cannot be overwritten. + with tempfile.TemporaryDirectory(prefix=f".draft-{skill}-", dir=TASKS_DIR) as staging: + staged = draft_and_save(skill, components["description"], components["body"], + Path(staging), log=log) + try: + os.link(Path(staged), TASKS_DIR / f"{skill}.yaml") + except FileExistsError: + raise RuntimeError(f"'{skill}' already has an eval task set") from None + state["status"] = "done" + except BaseException as e: # surface provider, parse, and SystemExit failures in the card + log(f"ERROR: {e}") + state["status"] = "error" + + +def _run_review(skill: str, state: dict, log) -> None: + try: + from ingot.optimize.review import run_review + run_review(skill, log=log) + state["status"] = "done" + except BaseException as e: # provider, judge, and spend-cap failures belong in the skill log + log(f"ERROR: {e}") + state["status"] = "error" + + @app.get("/api/runs") def runs(): - return {skill: {"status": s["status"], "log": s["log"][-30:]} for skill, s in RUNS.items()} + return {skill: {"status": s["status"], "action": s.get("action", "optimize"), + "log": s["log"][-30:]} for skill, s in RUNS.items()} -_COMPONENT_LABEL = {"description": "SKILL.md (description)", "body": "SKILL.md (body)"} +_COMPONENT_LABEL = {"description": "SKILL.md (description)", "body": "SKILL.md (body)", + "frontmatter": "SKILL.md (frontmatter)"} def _label(component: str) -> str: @@ -292,6 +525,22 @@ def _review_risk(champion: dict, challenger: dict) -> dict: "high_risk": changed_pct >= 50 or size_delta_pct <= -50} +def _publication_matches_pending(publication: dict, pending: dict) -> bool: + """Whether a skill-level receipt belongs to this exact quarantined proposal.""" + proposal_id = next((str((pending.get(kind) or {}).get("proposal_id") or "") + for kind in ("retrospective", "creation") + if (pending.get(kind) or {}).get("proposal_id")), "") + revision = str(((pending.get("evidence") or {}).get("challenger") or {}).get("revision") or "") + identities = [] + if proposal_id: + identities.append(str(publication.get("proposal_id") or "") == proposal_id) + if revision: + identities.append(str(publication.get("candidate_revision") or "") == revision) + if not identities: + identities.append(publication.get("components") == pending.get("challenger_components")) + return all(identities) + + @app.get("/api/pending/{skill}") def pending(skill: str): p = load_pending(_check(skill)) @@ -304,13 +553,24 @@ def pending(skill: str): for comp in changed_components: label = _label(comp) blocks.append("\n".join(difflib.unified_diff( - champ[comp].splitlines(), chall.get(comp, "").splitlines(), + str(champ.get(comp, "")).splitlines(), str(chall.get(comp, "")).splitlines(), fromfile=f"{label} (champion)", tofile=f"{label} (challenger)", lineterm=""))) comparison = [{"component": _label(component), "before": str(champ.get(component, "")), "after": str(chall.get(component, ""))} for component in changed_components] + publication = publication_for_skill(skill) + if publication and not _publication_matches_pending(publication, p): + publication = None + # `pr` and `note` are what make a stalled publication legible: a receipt sitting at + # awaiting_merge because the vault does not allow auto-merge is waiting on a person, and the + # reviewer has no other way to learn which pull request to go and merge. + publication_view = ({key: publication.get(key) for key in + ("id", "state", "actor", "action", "attempts", "last_error", "pr", "note")} + if publication else None) return {"skill": skill, "kind": p.get("kind", "quality"), "inner_loop": _inner_loop(p), "ab": p.get("ab"), "routing": p.get("routing"), "dataset": p.get("dataset"), + "retrospective": p.get("retrospective"), "creation": p.get("creation"), + "publication": publication_view, "evidence": p.get("evidence_paths"), "stale": _stale_reason(skill, p), "model": p.get("model"), "judge": p.get("judge"), "gate": p.get("gate", {"promotable": True, "blocked": []}), @@ -368,6 +628,57 @@ def approve(skill: str, actor: str = Depends(current_actor)): raise HTTPException(409, str(e)) +PUBLICATION_FIELDS = ("id", "skill", "action", "state", "actor", "attempts", + "pr", "note", "last_error", "created") +LIVE_STATES = ("approved_publishing", "publishing", "awaiting_merge") + + +def _pending_blocked() -> str | None: + """A human-readable warning when a pending record exists but cannot be used, else None.""" + from ingot.optimize.promote import pending_dir, unreadable_pending + queue = pending_dir() + if queue.is_dir() and not os.access(queue, os.R_OK | os.X_OK): + return (f"cannot read the review queue at {queue} as uid {os.getuid()}; " + f"quarantined changes cannot be listed") + blocked = unreadable_pending() + if not blocked: + return None + return (f"{len(blocked)} quarantined change(s) in {queue} cannot be read as uid " + f"{os.getuid()} and are NOT shown below: {', '.join(blocked)}") + + +@app.get("/api/publications") +def publications(): + """The publication lane, read only. + + A receipt outlives the pending record it came from, so this is the only surface that can show + an approved change while it is still travelling to the vault. `unreadable` is not cosmetic: + `Path.glob` swallows `PermissionError`, so a store this process cannot list looks exactly like + an empty one, and an empty lane is the reading a stalled publisher most wants you to make.""" + # Read the receipt store through the module attribute, not a name bound at import: the path is + # configurable, `recent_publications` looks it up per call, and a frozen copy here inspected one + # directory while listing another. + from ingot.optimize.publication import publications_dir + + # Only the forge backend has a pull request to link to, and only it knows the repository. + forge = os.environ.get("INGOT_FORGE_REPOSITORY") or "" + store = publications_dir() + unreadable = store.is_dir() and not os.access(store, os.R_OK | os.X_OK) + records = [] if unreadable else recent_publications() + return {"publications": [dict({key: record.get(key) for key in PUBLICATION_FIELDS}, + pr_url=(f"https://github.com/{forge}/pull/{record['pr']}" + if forge and record.get("pr") else None)) + for record in records], + "live_states": list(LIVE_STATES), + "unreadable": (f"cannot read the receipt store at {store} as uid " + f"{os.getuid()}; publications cannot be listed") if unreadable else None, + # The same misreading, one directory over. `list_pending` skips a record it cannot read + # so one corrupt file cannot break review, which also means an unreadable proposal is + # indistinguishable from no proposal and the board reports CLEAR over it. Observed live: + # the MCP container writes records as root 0600 while this process is uid 1000. + "pending_blocked": _pending_blocked()} + + @app.get("/api/evidence/{skill}") def evidence(skill: str): """The recorded evidence bundle for a pending change, read only. @@ -376,7 +687,8 @@ def evidence(skill: str): runs/evidence. Nothing a request carries selects a file.""" path = _evidence_file(_recorded_location(_check(skill))) markdown = _read_evidence(path) - return {"skill": skill, "path": path.relative_to(EVIDENCE_DIR).as_posix(), "markdown": markdown} + return {"skill": skill, "path": path.relative_to(_evidence_dir()).as_posix(), + "markdown": markdown} def _recorded_location(skill: str) -> str: @@ -406,9 +718,9 @@ def _host_path(recorded: str) -> Path: refuse.""" path = Path(recorded) if not path.is_absolute(): - return REPO_ROOT / path + return STATE_ROOT / path try: - return REPO_ROOT / path.relative_to(CONTAINER_ROOT) + return STATE_ROOT / path.relative_to(CONTAINER_ROOT) except ValueError: return path @@ -418,7 +730,7 @@ def _evidence_file(recorded: str) -> Path: Resolution happens before the containment check, so neither `..` nor a symlink out of the evidence tree can reach another part of the filesystem.""" resolved = _host_path(recorded).resolve() - if not resolved.is_relative_to(EVIDENCE_DIR): + if not resolved.is_relative_to(_evidence_dir()): raise HTTPException(400, "recorded evidence path is outside runs/evidence") return resolved @@ -443,14 +755,12 @@ def reject(skill: str, payload: RejectRequest | None = None, # later would let a second reject pass the check, then re-delete and double-audit after the # first released, returning 200 instead of 404 (mirrors approve/rollback holding the lock). with change_control(skill): - pending = load_pending(skill) - if pending is None: - raise HTTPException(404, f"no pending change for '{skill}'") - revision = _challenger_revision(pending) - pending_path(skill).unlink(missing_ok=True) - reason = " ".join((payload.reason if payload else "").split()) - _audit_best_effort("reject", skill, revision, actor, reason=reason) - return {"result": f"rejected the pending change for '{skill}'"} + try: + result = reject_pending(skill, actor=actor, + reason=payload.reason if payload else "") + except ValueError as exc: + raise HTTPException(404, str(exc)) + return {"result": result} @app.get("/api/history") diff --git a/ui/auth.py b/ui/auth.py index 7ecc567..c56a34e 100644 --- a/ui/auth.py +++ b/ui/auth.py @@ -25,8 +25,9 @@ from pathlib import Path from fastapi import HTTPException, Request +from ingot import paths -AUTH_FILE = Path(os.environ.get("AUTH_FILE") or Path(__file__).resolve().parent.parent / "runs" / "auth.json") +AUTH_FILE = Path(os.environ.get("AUTH_FILE") or paths.runs() / "auth.json") _ITERATIONS = 200_000 _ANON = "local-operator" # actor when auth is disabled, matches the pre-auth default # The compose default (docker-compose.yml sets AUTH_PASSWORD=${AUTH_PASSWORD:-ingot}); we warn while diff --git a/ui/static/index.html b/ui/static/index.html index 4a69970..0aa5901 100644 --- a/ui/static/index.html +++ b/ui/static/index.html @@ -19,7 +19,7 @@ labels) as well as fills/dots, so in light mode it is Slancha's AA-safe blue-text #0F63C9 (≥4.5:1 on canvas/white/wash); --rust-2 is the darker hover. */ --paper: #F7F8FA; --paper-2: #FFFFFF; --paper-3: #EEF2F7; - --ink: #15171C; --ink-2: #667085; --ink-3: #98A2B3; + --ink: #15171C; --ink-2: #515B6E; --ink-3: #667085; --line: #D9DEE7; --line-2: #C4CDDA; --rust: #0F63C9; --rust-2: #0B4F9E; --rust-bg: #E8F1FD; --pass: #0C6E57; --pass-bg: #E1F7EF; @@ -34,7 +34,7 @@ @media (prefers-color-scheme: dark) { :root { --paper: #0E1420; --paper-2: #161D2B; --paper-3: #1D2534; - --ink: #E8EDF4; --ink-2: #9AA6B8; --ink-3: #69748A; + --ink: #E8EDF4; --ink-2: #9AA6B8; --ink-3: #8A94A6; --line: #253044; --line-2: #313D52; --rust: #4D9BFF; --rust-2: #7AB4FF; --rust-bg: #16273F; --pass: #46C39F; --pass-bg: #10251F; @@ -46,7 +46,7 @@ } :root[data-theme="light"] { color-scheme: light; --paper: #F7F8FA; --paper-2: #FFFFFF; --paper-3: #EEF2F7; - --ink: #15171C; --ink-2: #667085; --ink-3: #98A2B3; + --ink: #15171C; --ink-2: #515B6E; --ink-3: #667085; --line: #D9DEE7; --line-2: #C4CDDA; --rust: #0F63C9; --rust-2: #0B4F9E; --rust-bg: #E8F1FD; --pass: #0C6E57; --pass-bg: #E1F7EF; --fail: #BE3B34; --fail-bg: #FFF0EF; @@ -55,7 +55,7 @@ --shadow: 0 1px 2px rgb(16 24 40 / 4%), 0 14px 40px rgb(35 52 78 / 9%); } :root[data-theme="dark"] { color-scheme: dark; --paper: #0E1420; --paper-2: #161D2B; --paper-3: #1D2534; - --ink: #E8EDF4; --ink-2: #9AA6B8; --ink-3: #69748A; + --ink: #E8EDF4; --ink-2: #9AA6B8; --ink-3: #8A94A6; --line: #253044; --line-2: #313D52; --rust: #4D9BFF; --rust-2: #7AB4FF; --rust-bg: #16273F; --pass: #46C39F; --pass-bg: #10251F; --fail: #E0796D; --fail-bg: #2D1A17; @@ -92,7 +92,269 @@ .pill.run { color: var(--warn); background: var(--warn-bg); } .pill.act { color: var(--rust); background: var(--rust-bg); } - .page { max-width: 74rem; margin: 0 auto; padding: 0 1.4rem 4rem; } + /* ---- shell: sidebar + one routed pane ---- + The library is 102 skills in four roots. As one scrolling column that was 8,600px of page, + which is a document, not a console: nothing was reachable without scrolling past everything + else. The sidebar carries the routes and the counts, and only one view is mounted at a time. */ + .shell { display: grid; grid-template-columns: 15.5rem minmax(0, 1fr); max-width: 96rem; + margin: 0 auto; align-items: start; } + .sidebar { position: sticky; top: 3.05rem; height: calc(100vh - 3.05rem); overflow-y: auto; + padding: 1.5rem 1rem 2rem 1.5rem; border-right: 1px solid var(--line); } + .main { min-width: 0; padding: 1.5rem 1.5rem 4rem 1.7rem; } + + .navgroup { margin-bottom: 1.5rem; } + .navgroup h3 { font: 600 .62rem/1 var(--mono); letter-spacing: .11em; text-transform: uppercase; + color: var(--ink-3); margin: 0 0 .5rem .55rem; } + .navlink { display: flex; align-items: center; gap: .5rem; padding: .38rem .55rem; border-radius: 8px; + color: var(--ink-2); text-decoration: none; font-size: .86rem; transition: background .12s, color .12s; } + .navlink:hover { background: var(--paper-3); color: var(--ink); } + .navlink.active { background: var(--rust-bg); color: var(--rust); font-weight: 600; } + .navlink .ico { width: 1rem; flex: none; text-align: center; font-size: .8rem; opacity: .8; } + .navlink .txt { flex: 1; min-width: 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } + .navcount { font-family: var(--mono); font-size: .68rem; color: var(--ink-3); + font-variant-numeric: tabular-nums; } + .navlink.active .navcount { color: var(--rust); } + /* Specificity, not order: `.navlink.active .navcount` is 0-3-0 and would otherwise repaint this + badge in --rust on a --rust fill — an invisible count, precisely while something needs review. */ + .navcount.act, .navlink.active .navcount.act { background: var(--rust); color: var(--paper); + border-radius: 999px; padding: .04rem .4rem; font-weight: 700; } + .navlink.sub { padding-left: 1.7rem; font-size: .82rem; } + .navfoot { border-top: 1px solid var(--line); padding-top: .8rem; margin-top: .3rem; + font: .7rem/1.7 var(--mono); color: var(--ink-3); } + .navfoot b { color: var(--ink-2); font-weight: 600; } + + /* ---- folder + card grid ---- */ + .crumbs { display: flex; align-items: center; gap: .4rem; font: .72rem/1 var(--mono); + letter-spacing: .05em; text-transform: uppercase; color: var(--ink-3); margin-bottom: .5rem; } + .crumbs a { color: var(--rust); text-decoration: none; } + .crumbs a:hover { text-decoration: underline; } + .grid { display: grid; grid-template-columns: repeat(auto-fill, minmax(17rem, 1fr)); gap: .8rem; + margin-top: 1rem; } + .fold { display: block; text-align: left; width: 100%; cursor: pointer; font: inherit; + background: var(--paper-2); border: 1px solid var(--line); border-radius: 12px; + padding: .95rem 1.05rem; text-decoration: none; color: inherit; box-shadow: var(--shadow-sm); + transition: border-color .12s, transform .12s; } + .fold:hover { border-color: var(--rust); transform: translateY(-1px); } + .fold .fname { display: flex; align-items: center; gap: .45rem; font-family: var(--mono); + font-size: .92rem; color: var(--rust); } + .fold .fcount { margin-left: auto; font-size: .72rem; color: var(--ink-3); } + .fold .fnote { font-size: .78rem; color: var(--ink-2); margin-top: .35rem; } + .fold .fbar { display: flex; gap: .3rem; margin-top: .6rem; font: .66rem/1 var(--mono); + color: var(--ink-3); } + + .scard { background: var(--paper-2); border: 1px solid var(--line); border-radius: 12px; + display: flex; flex-direction: column; box-shadow: var(--shadow-sm); + transition: border-color .12s; } + .scard:hover { border-color: var(--line-2); } + .scard.pend { border-color: var(--rust); background: var(--rust-bg); } + .scard .open { flex: 1; text-align: left; background: none; border: 0; cursor: pointer; + font: inherit; color: inherit; padding: .85rem .95rem .6rem; border-radius: 12px 12px 0 0; } + .scard .sname { font-family: var(--mono); font-size: .86rem; color: var(--ink); + overflow-wrap: anywhere; } + .scard .chips { display: flex; flex-wrap: wrap; gap: .3rem; margin-top: .45rem; } + .scard .sdesc { font-size: .78rem; color: var(--ink-2); margin: .5rem 0 0; line-height: 1.5; + display: -webkit-box; -webkit-line-clamp: 3; -webkit-box-orient: vertical; overflow: hidden; } + .scard .foot { display: flex; flex-wrap: wrap; gap: .4rem; padding: 0 .95rem .8rem; } + .scard .foot .btn { font-size: .68rem; padding: .3rem .6rem; } + + /* ---- categories ---- */ + .cluster-meta { font: .72rem/1.8 var(--mono); color: var(--ink-3); margin-bottom: .9rem; + font-variant-numeric: tabular-nums; } + .cluster-meta b { color: var(--ink-2); } + /* The scatter places a hundred points from a projection that bunches most of them centrally, so + picking a bucket by hunting a dot is unreliable. These are the actual control; the plot shows + how the library sits around whatever is picked. */ + .cluster-chips { display: flex; flex-wrap: wrap; gap: .4rem; margin-bottom: .9rem; } + .cchip { display: inline-flex; align-items: center; gap: .4rem; cursor: pointer; font: inherit; + background: var(--paper-2); border: 1px solid var(--line); border-radius: 999px; + padding: .25rem .7rem; font-size: .76rem; color: var(--ink-2); transition: border-color .12s; } + .cchip:hover { border-color: var(--line-2); color: var(--ink); } + .cchip[aria-selected="true"] { border-color: currentColor; color: var(--ink); + background: var(--paper-3); font-weight: 600; } + .cchip .swatch { width: .55rem; height: .55rem; border-radius: 50%; flex: none; } + .cchip .n { font-family: var(--mono); font-size: .68rem; color: var(--ink-3); + font-variant-numeric: tabular-nums; } + .cluster-split { display: grid; grid-template-columns: minmax(0, 1fr) 19rem; gap: 1rem; + align-items: start; } + .cluster-plot { background: var(--paper-2); border: 1px solid var(--line); border-radius: 14px; + padding: .6rem; box-shadow: var(--shadow-sm); } + .cluster-plot svg { display: block; width: 100%; height: auto; } + .cbub { cursor: pointer; transition: opacity .12s; } + .cbub:hover { opacity: .85; } + .clabel { font: 600 10px var(--mono); fill: var(--ink-3); pointer-events: none; + paint-order: stroke; stroke: var(--paper-2); stroke-width: 3px; } + .clabel.on { fill: var(--ink); font-size: 11.5px; } + .cdot { pointer-events: none; } + .cluster-side { background: var(--paper-2); border: 1px solid var(--line); border-radius: 14px; + padding: 1rem 1.1rem; box-shadow: var(--shadow-sm); } + .cluster-side h3 { font: 600 .95rem/1.3 var(--sans); margin: 0 0 .2rem; } + .cluster-side .cterms { font: .72rem/1.7 var(--mono); color: var(--ink-3); margin-bottom: .7rem; } + .cluster-side ul { list-style: none; margin: 0; padding: 0; max-height: 22rem; overflow-y: auto; } + .cluster-side li { border-top: 1px solid var(--line); } + .cluster-side li button { width: 100%; text-align: left; background: none; border: 0; + cursor: pointer; font: .78rem/1.5 var(--mono); color: var(--ink-2); padding: .35rem .1rem; } + .cluster-side li button:hover { color: var(--rust); } + .cluster-side li button .gone { color: var(--warn); font-size: .68rem; } + @media (max-width: 1040px) { .cluster-split { grid-template-columns: 1fr; } } + + /* ---- local traces ---- */ + .trace-note { font: .74rem/1.7 var(--mono); color: var(--ink-3); max-width: 62rem; } + .trace-note b { color: var(--ink-2); } + .trace-controls { display: grid; grid-template-columns: minmax(10rem, 1.2fr) minmax(9rem, .8fr) + minmax(9rem, .7fr) minmax(9rem, .7fr) auto; gap: .7rem; margin-top: 1rem; + align-items: end; } + .trace-preview-toggle { display: flex; align-items: center; gap: .45rem; min-height: 2.35rem; + padding: 0 .15rem; white-space: nowrap; font: .67rem/1.3 var(--mono); color: var(--ink-2); } + .trace-preview-toggle input { width: 1rem; height: 1rem; margin: 0; } + .trace-stats { display: grid; grid-template-columns: repeat(auto-fit, minmax(10rem, 1fr)); + border: 1px solid var(--line); border-radius: 12px; background: var(--paper-2); + margin-top: 1rem; box-shadow: var(--shadow-sm); } + .trace-stat { padding: .9rem 1rem; border-right: 1px solid var(--line); } + .trace-stat:last-child { border-right: 0; } + .trace-stat .n { display: block; font: 600 1.45rem/1 var(--mono); color: var(--ink); } + .trace-stat .l { display: block; margin-top: .35rem; font: .63rem/1.4 var(--mono); + color: var(--ink-3); letter-spacing: .08em; text-transform: uppercase; } + /* Harness x model matrix. Each model owns a column: the fixed harness ledger lets an evaluator + compare like with like while the evidence plates keep absence distinct from a zero effect. */ + .mx-controls { display: grid; grid-template-columns: minmax(12rem, .5fr) auto; gap: .7rem; + margin-top: 1rem; align-items: end; } + .mx-warn { margin-top: 1rem; padding: .85rem 1rem; border-radius: 12px; font: .75rem/1.65 var(--mono); + border: 1px solid var(--warn); background: var(--warn-bg); color: var(--warn); } + .mx-warn b { display: block; margin-bottom: .25rem; letter-spacing: .06em; text-transform: uppercase; + font-size: .64rem; } + .mx-wrap { margin-top: 1rem; border: 1px solid var(--line); border-radius: 12px; + background: var(--paper-2); box-shadow: var(--shadow-sm); overflow-x: auto; position: relative; } + .mx { width: var(--mx-width, 100%); min-width: 100%; border-collapse: separate; border-spacing: 0; + table-layout: fixed; font: .74rem/1.5 var(--mono); } + .mx td, .mx th { min-width: 12.5rem; text-align: left; padding: .7rem .9rem; + border-bottom: 1px solid var(--line); color: var(--ink-2); vertical-align: middle; } + .mx th { font-size: .63rem; letter-spacing: .08em; text-transform: uppercase; color: var(--ink-3); + white-space: normal; background: var(--paper-2); } + .mx-sticky { position: sticky; left: 0; z-index: 2; min-width: 13.5rem !important; + background: var(--paper-2) !important; box-shadow: 1px 0 0 var(--line); } + .mx-corner { z-index: 3; color: var(--ink-3) !important; } + .mx-harness { color: var(--ink) !important; white-space: nowrap; } + .mx-model { display: block; color: var(--ink-2); overflow-wrap: anywhere; } + .mx-count { display: block; margin-top: .35rem; font-size: .58rem; line-height: 1.45; + letter-spacing: 0; text-transform: none; color: var(--ink-3); } + .mx-column-warn { display: block; margin-top: .3rem; font-size: .58rem; line-height: 1.45; + letter-spacing: 0; text-transform: none; color: var(--warn); } + .mx-chart { margin-top: 1rem; padding: 1rem; border: 1px solid var(--line); border-radius: 12px; + background: var(--paper-2); box-shadow: var(--shadow-sm); } + .mx-chart-head { display: flex; flex-wrap: wrap; align-items: start; justify-content: space-between; + gap: .7rem 1.25rem; margin-bottom: .8rem; } + .mx-chart-head h3 { margin: 0; font: 600 1.08rem/1.3 var(--sans); color: var(--ink); } + .mx-chart-kicker { display: block; margin-top: .18rem; font: .66rem/1.55 var(--mono); + color: var(--ink-3); } + .mx-chart-stats { display: flex; flex-wrap: wrap; gap: .45rem; } + .mx-chart-stat { min-width: 6.4rem; padding: .45rem .6rem; border: 1px solid var(--line); + border-radius: 8px; background: var(--paper); } + .mx-chart-stat b { display: block; color: var(--ink); font: 600 .92rem/1 var(--mono); + font-variant-numeric: tabular-nums; } + .mx-chart-stat span { display: block; margin-top: .24rem; color: var(--ink-3); + font: .58rem/1.3 var(--mono); letter-spacing: .06em; text-transform: uppercase; } + .mx-chart-legend { display: flex; flex-wrap: wrap; gap: .4rem .9rem; margin: 0 0 .25rem; + padding: .6rem 0; border-top: 1px solid var(--line); border-bottom: 1px solid var(--line); } + .mx-chart-legend span { display: inline-flex; align-items: center; gap: .38rem; + color: var(--ink-2); font: .65rem/1.4 var(--mono); } + .mx-chart-legend i { width: .55rem; height: .55rem; border-radius: 50%; background: currentColor; + box-shadow: 0 0 0 2px var(--paper-2), 0 0 0 3px currentColor; } + .mx-chart-legend b { color: var(--ink-3); font-weight: 400; } + .mx-chart svg { display: block; width: 100%; height: auto; min-height: 15rem; } + .mx-grid { stroke: var(--line); stroke-width: 1; } + .mx-model-rail { stroke: var(--line-2); stroke-width: 1; stroke-dasharray: 2 4; } + .mx-zero { stroke: var(--ink-2); stroke-width: 1.5; } + .mx-zone-positive { fill: color-mix(in srgb, var(--pass) 5%, transparent); } + .mx-zone-negative { fill: color-mix(in srgb, var(--fail) 4%, transparent); } + .mx-axis, .mx-tick, .mx-legend { fill: var(--ink-3); font: 10px var(--mono); } + .mx-axis-title { fill: var(--ink-2); font: 11px var(--mono); } + .mx-size-mark { cursor: pointer; outline: none; } + .mx-size-mark .visible { stroke-width: 2; transition: transform .12s, stroke-width .12s; } + .mx-size-mark .hit { fill: transparent; stroke: none; } + .mx-size-mark:hover .visible, .mx-size-mark:focus-visible .visible, + .mx-size-mark.selected .visible { stroke: var(--ink); stroke-width: 3; transform: scale(1.18); } + .mx-mark-code { fill: var(--paper-2); font: 700 6px var(--mono); pointer-events: none; + text-anchor: middle; dominant-baseline: central; } + .mx-chart-detail { min-height: 3.2rem; display: flex; flex-wrap: wrap; align-items: center; + gap: .4rem 1rem; padding: .7rem .8rem; border-radius: 8px; background: var(--paper-3); + color: var(--ink-2); font: .68rem/1.5 var(--mono); } + .mx-chart-detail strong { color: var(--ink); font-size: .76rem; } + .mx-chart-detail .lift { color: var(--pass); font-size: .9rem; font-weight: 700; } + .mx-chart-detail .lift.down { color: var(--fail); } + .mx-chart-detail .lift.flat { color: var(--ink-3); } + .mx-chart-note { margin-top: .45rem; font: .62rem/1.55 var(--mono); color: var(--ink-3); } + .mx-chart-empty { margin-top: 1rem; padding: .85rem 1rem; border: 1px dashed var(--line-2); + border-radius: 12px; font: .72rem/1.6 var(--mono); color: var(--ink-3); } + .mx-cell { padding: .45rem .55rem !important; } + /* The plate carries a result, not a miniature dashboard: lift stays primary until provenance is + requested. A plain blank cell has no border or control because no run is evidence of absence. */ + .mx-plate { display: block; width: 100%; padding: .48rem .55rem; border: 1px solid transparent; + border-radius: 7px; background: transparent; font: 600 .88rem/1 var(--mono); text-align: left; + cursor: pointer; } + .mx-plate:focus-visible { outline: 2px solid var(--rust); outline-offset: 2px; } + .mx-plate:hover { border-color: var(--line-2); background: var(--paper-3); } + .mx-measured { color: var(--ink); } + .mx-measured.up { color: var(--pass); } + .mx-measured.down { color: var(--fail); } + .mx-measured.flat { color: var(--ink-3); } + .mx-error { color: var(--fail); border-color: color-mix(in srgb, var(--fail) 22%, transparent); } + .mx-blank { color: var(--ink-3); text-align: center; font: .9rem/1 var(--mono); } + .mx-details { margin-top: .7rem; min-height: 3rem; padding: .75rem .9rem; border-left: 2px solid var(--line-2); + background: var(--paper-2); color: var(--ink-2); font: .7rem/1.65 var(--mono); } + .mx-details p { margin: 0; } + .mx-provenance { display: grid; grid-template-columns: repeat(auto-fit, minmax(8.5rem, 1fr)); + gap: .45rem 1rem; margin-top: .55rem; } + .mx-provenance span { color: var(--ink-3); } + .mx-provenance b { display: block; color: var(--ink-2); font-weight: 500; } + @media (max-width: 640px) { + .mx td, .mx th { min-width: 10.75rem; } + .mx-sticky { min-width: 9.75rem !important; } + .mx-chart { padding: .8rem; } + .mx-chart-stats { width: 100%; } + .mx-chart-stat { flex: 1 1 5.5rem; min-width: 0; } + .mx-chart svg { min-width: 38rem; } + .mx-chart-plot { overflow-x: auto; } + } + .mx-foot { margin-top: .7rem; font: .7rem/1.7 var(--mono); color: var(--ink-3); } + .trace-split { display: grid; grid-template-columns: minmax(0, 1fr) 18rem; gap: 1rem; + align-items: start; margin-top: 1rem; } + .trace-list, .trace-skills { border: 1px solid var(--line); border-radius: 12px; + background: var(--paper-2); box-shadow: var(--shadow-sm); overflow: hidden; } + .trace-row { padding: .75rem .9rem; border-top: 1px solid var(--line); } + .trace-row:first-child { border-top: 0; } + .trace-task { font-size: .8rem; line-height: 1.5; color: var(--ink); overflow-wrap: anywhere; } + .trace-meta { display: flex; flex-wrap: wrap; gap: .35rem .7rem; margin-top: .35rem; + font: .66rem/1.4 var(--mono); color: var(--ink-3); } + .trace-skills { padding: .85rem 1rem; } + .trace-skills h3 { margin: 0 0 .55rem; font-size: .9rem; } + .trace-skills ol { margin: 0; padding: 0; list-style: none; } + .trace-skills li { display: flex; gap: .6rem; justify-content: space-between; + border-top: 1px solid var(--line); padding: .35rem 0; font: .7rem/1.4 var(--mono); + color: var(--ink-2); } + .trace-skills li span:last-child { color: var(--ink-3); } + @media (max-width: 900px) { + .trace-split { grid-template-columns: 1fr; } + .trace-controls { grid-template-columns: 1fr 1fr; } + .trace-stat { border-right: 0; border-bottom: 1px solid var(--line); } + .trace-stat:last-child { border-bottom: 0; } + } + @media (max-width: 560px) { .trace-controls { grid-template-columns: 1fr; } } + + @media (max-width: 900px) { + .shell { grid-template-columns: 1fr; } + .sidebar { position: static; height: auto; border-right: 0; border-bottom: 1px solid var(--line); + padding: 1rem 1.4rem; } + .navgroup { margin-bottom: .9rem; } + .main { padding: 1.2rem 1.4rem 3rem; } + } + + /* Fixed steps, not viewport-fluid: an operator reads this at one desk on one monitor, and a + heading that resizes with every pixel of window width just wobbles. One step down where the + column actually gets narrow. */ + @media (max-width: 640px) { + .board-title { font-size: 1.9rem; } + .head { font-size: 1.55rem; } + } /* ---- board head ---- */ .board-head { padding: 2rem 0 .3rem; } @@ -101,10 +363,7 @@ .board-head .eyebrow .n { color: var(--rust); } .board-head .eyebrow .side { color: var(--ink-3); } .board-title { font-family: var(--serif); font-weight: 700; letter-spacing: -.02em; - font-size: clamp(1.9rem, 4vw, 2.7rem); margin: .5rem 0 0; text-wrap: balance; } - .board-sub { color: var(--ink-2); font-size: .96rem; margin: .6rem 0 0; max-width: 48rem; text-wrap: pretty; } - .board-sub a { color: var(--rust); text-decoration: none; white-space: nowrap; } - .board-sub a:hover { text-decoration: underline; } + font-size: 2.4rem; margin: .5rem 0 0; text-wrap: balance; } /* ---- KPI strip ---- */ .kpis { display: grid; grid-template-columns: repeat(auto-fit, minmax(160px, 1fr)); @@ -133,11 +392,14 @@ .alert.warn::before { background: var(--warn); } .alert.ok::before { background: var(--pass); } .alert.info::before { background: var(--ink-3); } + .board-head[hidden] { display: none; } .alert .atag { font-family: var(--mono); font-size: .74rem; font-weight: 700; letter-spacing: .02em; white-space: nowrap; } .alert.act .atag { color: var(--rust); } .alert.warn .atag { color: var(--warn); } .alert.ok .atag { color: var(--pass); } .alert .atxt { font-size: .84rem; color: var(--ink-2); } + .alert .atxt code { font-family: var(--mono); font-size: .92em; background: var(--paper-3); + color: var(--ink); border-radius: 5px; padding: .06rem .3rem; } /* ---- section scaffold ---- */ .band { padding-top: 2.6rem; } @@ -147,8 +409,7 @@ .band-head .n { color: var(--rust); } .band-head .side { color: var(--ink-3); } h2.head { font-family: var(--serif); font-weight: 700; letter-spacing: -.015em; line-height: 1.05; - font-size: clamp(1.5rem, 3.5vw, 2.1rem); margin: .8rem 0 0; text-wrap: balance; } - .sub { font-size: .9rem; color: var(--ink-2); margin: .5rem 0 0; max-width: 48rem; text-wrap: pretty; } + font-size: 1.9rem; margin: .8rem 0 0; text-wrap: balance; } .chip { font-family: var(--mono); font-size: .66rem; letter-spacing: .05em; padding: .12rem .5rem; border-radius: 6px; white-space: nowrap; text-transform: uppercase; font-weight: 600; } @@ -195,9 +456,23 @@ .cmp-tbl .win { color: var(--pass); font-weight: 700; } .cmp-tbl .lose { color: var(--fail); font-weight: 700; } .cmp-actions { display: flex; align-items: center; gap: .7rem; margin-top: 1.1rem; flex-wrap: wrap; } + /* The promotion dialog is the last screen before an irreversible change, and its evidence grows + with the held-out set. Scroll the evidence under a pinned decision row rather than scrolling + the whole modal, which had already pushed Approve off the bottom edge at eight tasks. */ + #cmp-overlay .cmp-modal { display: flex; flex-direction: column; overflow: hidden; } + #cmp-body { flex: 1 1 auto; overflow: auto; min-height: 0; } + #cmp-body > .margin { margin-bottom: 1.1rem; } + #cmp-overlay .cmp-actions { flex: none; margin-top: 0; padding-top: 1rem; + border-top: 1px solid var(--line); } /* Read-only skill and revision explorer */ .skill-modal { width: min(900px, 96vw); } + .skill-tabs { display: flex; gap: .35rem; margin: -.2rem 0 1rem; border-bottom: 1px solid var(--line); } + .skill-tab { border: 0; border-bottom: 2px solid transparent; background: none; color: var(--ink-3); + cursor: pointer; padding: .45rem .65rem .6rem; font: 600 .68rem/1 var(--mono); + letter-spacing: .06em; text-transform: uppercase; } + .skill-tab[aria-selected="true"] { color: var(--rust); border-bottom-color: var(--rust); } + .skill-panel[hidden] { display: none; } .version-toolbar { display: grid; grid-template-columns: minmax(14rem, 1fr) minmax(12rem, 1fr); gap: .8rem; margin-bottom: 1rem; } .version-field { display: flex; flex-direction: column; gap: .35rem; } @@ -210,6 +485,34 @@ .version-pre { font-family: var(--mono); font-size: .74rem; line-height: 1.55; margin: 0; background: var(--paper); border: 1px solid var(--line); border-radius: 12px; padding: .85rem 1rem; overflow: auto; min-height: 16rem; max-height: 56vh; white-space: pre; color: var(--ink); } + .eval-head { display: flex; align-items: flex-start; justify-content: space-between; gap: .8rem; + flex-wrap: wrap; margin-bottom: .8rem; } + .eval-summary { display: flex; flex-wrap: wrap; gap: .4rem; } + .eval-groups { display: grid; gap: .65rem; } + .eval-group { border: 1px solid var(--line); border-radius: 11px; background: var(--paper); } + .eval-group > summary { cursor: pointer; padding: .7rem .8rem; color: var(--ink); + font: 600 .72rem/1.4 var(--mono); letter-spacing: .03em; } + .eval-group > summary .count { color: var(--ink-3); font-weight: 400; margin-left: .35rem; } + .eval-items { border-top: 1px solid var(--line); } + .eval-item { padding: .78rem .85rem; } + .eval-item + .eval-item { border-top: 1px solid var(--line); } + .eval-task { margin: 0; color: var(--ink); font-size: .82rem; line-height: 1.55; white-space: pre-wrap; } + .eval-meta { margin: .45rem 0 0; color: var(--ink-2); font: .71rem/1.55 var(--mono); + white-space: pre-wrap; } + .eval-checks { list-style: none; margin: .55rem 0 0; padding: 0; display: grid; gap: .3rem; } + .eval-checks li { color: var(--ink-2); font: .7rem/1.5 var(--mono); padding-left: .8rem; + border-left: 2px solid var(--line-2); } + .review-result { margin-top: 1rem; border-top: 1px solid var(--line); padding-top: 1rem; } + .review-result:empty { display: none; } + .review-result h3 { margin: 0 0 .65rem; font: 600 .76rem/1.3 var(--mono); + letter-spacing: .05em; text-transform: uppercase; } + .review-score { display: flex; align-items: baseline; gap: .6rem; margin-bottom: .75rem; } + .review-score strong { font: 600 1.7rem/1 var(--mono); color: var(--ink); } + .review-score span { color: var(--ink-3); font: .7rem/1.5 var(--mono); } + .finding { padding: .65rem .75rem; border-left: 3px solid var(--warn); background: var(--warn-bg); + border-radius: 0 9px 9px 0; margin-top: .45rem; } + .finding b { font: 600 .72rem/1.4 var(--mono); color: var(--ink); } + .finding p { margin: .25rem 0 0; color: var(--ink-2); font-size: .76rem; line-height: 1.5; } @media (max-width: 620px) { .version-toolbar { grid-template-columns: 1fr; } } /* ---- skills list ---- */ @@ -221,6 +524,13 @@ background: var(--paper-2); color: var(--ink); } .filter-count { font: .68rem/1.4 var(--mono); color: var(--ink-3); align-self: center; } .skilllist { margin-top: 1.1rem; border-top: 1px solid var(--line); } + /* Provenance folders: which of these skills did we write, and which came from elsewhere. */ + .skill-folder + .skill-folder { margin-top: 1.4rem; } + .skill-folder-head { display: flex; flex-wrap: wrap; align-items: baseline; gap: .45rem .7rem; + padding: .55rem .45rem; border-bottom: 1px solid var(--line-2); background: var(--paper-2); + position: sticky; top: 0; z-index: 1; } + .skill-folder-title { font-weight: 600; color: var(--ink); letter-spacing: .01em; } + .skill-folder-note { color: var(--ink-3); font-size: .82rem; flex: 1 1 14rem; min-width: 0; } .srow { display: flex; flex-wrap: wrap; align-items: baseline; gap: .5rem 1rem; padding: .9rem .45rem; border-bottom: 1px solid var(--line); transition: background .15s ease-out; } .srow.pend { background: var(--rust-bg); } @@ -292,6 +602,14 @@ font-variant-numeric: tabular-nums; } .metaline b { color: var(--ink-2); } .metaline a { color: var(--rust); text-decoration: none; } .metaline a:hover { text-decoration: underline; } + .retro-evidence { margin-top: .9rem; border: 1px solid var(--line); border-radius: 12px; + background: var(--paper); padding: .8rem .9rem; } + .retro-evidence h4 { margin: 0 0 .55rem; font: 600 .68rem/1.3 var(--mono); + letter-spacing: .06em; text-transform: uppercase; color: var(--ink-2); } + .retro-evidence p { margin: .35rem 0; color: var(--ink-2); font-size: .8rem; line-height: 1.5; } + .retro-evidence ul { margin: .45rem 0 .65rem; padding-left: 1.15rem; color: var(--ink-2); + font-size: .76rem; line-height: 1.55; } + .retro-evidence code { font: .72rem/1.45 var(--mono); overflow-wrap: anywhere; } .tokline { font-family: var(--mono); font-size: .78rem; color: var(--ink-2); margin-top: .5rem; font-variant-numeric: tabular-nums; } .tokline b { color: var(--ink); } .tokline .reg { color: var(--fail); font-weight: 600; } @@ -303,6 +621,13 @@ .gate.ok { color: var(--pass); background: var(--pass-bg); } .gate.block { color: var(--fail); background: var(--fail-bg); font-weight: 600; } .gate.warn { color: var(--warn); background: var(--warn-bg); } + /* margin-vs-resolution sits directly under the two scores, because that pair of big numbers is + what a reviewer reads first and it is the part that overstates a thin result */ + .margin { font-family: var(--mono); font-size: .74rem; line-height: 1.65; color: var(--ink-2); + margin-top: .7rem; font-variant-numeric: tabular-nums; text-wrap: pretty; } + .margin b { color: var(--ink); } + .margin.thin { color: var(--warn); border-left: 2px solid var(--warn); padding-left: .6rem; } + .margin.thin b { color: var(--warn); } .metaline b.win { color: var(--pass); } .metaline b.lose { color: var(--fail); } .review-actions { display: flex; align-items: center; gap: .7rem; margin-top: 1.2rem; flex-wrap: wrap; } .review-actions .msg { font-family: var(--mono); font-size: .74rem; color: var(--pass); } @@ -318,6 +643,20 @@ .risk-warning { grid-column: 1 / -1; font: 600 .74rem/1.5 var(--mono); color: var(--warn); } .difflabel { font-family: var(--mono); font-size: .66rem; letter-spacing: .09em; text-transform: uppercase; color: var(--rust); margin: 1.4rem 0 .4rem; } + .lane { margin: 0 0 1.6rem; } + .lane-row { display: grid; grid-template-columns: 1fr auto auto; gap: .6rem 1rem; align-items: baseline; + padding: .55rem .8rem; border: 1px solid var(--line); border-radius: 12px; background: var(--paper-2); + margin-bottom: .4rem; } + .lane-row.live { border-color: var(--warn); } + .lane-name { font: 600 .8rem/1.4 var(--mono); color: var(--ink-1); } + .lane-name .act { color: var(--ink-3); font-weight: 400; } + .lane-state { font: .64rem/1.3 var(--mono); letter-spacing: .06em; text-transform: uppercase; + color: var(--ink-3); } + .lane-row.live .lane-state { color: var(--warn); } + .lane-row.done .lane-state { color: var(--pass); } + .lane-row.stuck .lane-state { color: var(--fail); } + .lane-note { grid-column: 1 / -1; font: .7rem/1.5 var(--mono); color: var(--ink-3); } + .lane-empty { font: .72rem/1.5 var(--mono); color: var(--ink-3); } pre.diff { font-family: var(--mono); font-size: .72rem; line-height: 1.55; margin: 0; background: var(--paper-2); border: 1px solid var(--line); border-radius: 14px; padding: .8rem .95rem; overflow: auto; max-height: 28rem; white-space: pre; color: var(--ink-2); } @@ -350,8 +689,14 @@ .footer .k { color: var(--rust); } .footer a { color: var(--ink-3); text-decoration: none; } .footer a:hover { color: var(--rust); } - .js .reveal { opacity: 0; transform: translateY(12px); } - .js .reveal.in { opacity: 1; transform: none; transition: opacity .6s ease-out, transform .6s cubic-bezier(.16,1,.3,1); } + /* No entrance on the sections themselves. This is a console an operator opens to decide + something, not a page that introduces itself, and the four bands are the four objects the + product has rather than a narrative to reveal. The scroll-reveal it replaces also gated + visibility on a transition: a transition does not run in a background tab or a headless + renderer, so the Skills section — every record the product holds — rendered at opacity 0 + with its 8,600px of content still in the layout. Content is visible by default now, and + the only entrances left are the ones that carry state (the stagger over a list that just + arrived, the beam over a change that needs a human). */ /* staggered entrance for alert lines + skill rows, first paint only (structure: Magic UI animated-list) */ .js .list-in { animation: list-in .5s cubic-bezier(.16,1,.3,1) both; animation-delay: calc(var(--i, 0) * 70ms); } @@ -359,12 +704,13 @@ /* review card: the decision surface. While a challenger waits, a slow terracotta beam circles the border (structure: Magic UI border-beam, gradient square on an offset-path - rect, masked to the 1px ring). Falls back to the plain border where unsupported. */ + rect, masked to the 1px ring). Its layer is clipped because the offset square otherwise + widens the document at the 640px breakpoint. Falls back to the plain border where unsupported. */ .review-card { position: relative; border: 1px solid var(--line); border-radius: 16px; padding: 1.15rem 1.25rem 1.25rem; margin-top: 1.1rem; background: var(--paper-2); box-shadow: var(--shadow); } .beam { display: none; pointer-events: none; position: absolute; inset: 0; border-radius: inherit; - border: 1px solid transparent; + border: 1px solid transparent; overflow: hidden; mask-image: linear-gradient(transparent, transparent), linear-gradient(#000, #000); mask-clip: padding-box, border-box; mask-composite: intersect; } .beam::before { content: ""; position: absolute; aspect-ratio: 1; width: 90px; @@ -381,15 +727,157 @@ outline: 2px solid var(--rust); outline-offset: 2px; border-radius: 2px; } @media (prefers-reduced-motion: reduce) { html { scroll-behavior: auto; } * { transition: none !important; animation: none !important; } - .js .reveal { opacity: 1; transform: none; } .beam { display: none !important; } } + + /* ---- evidence cockpit visual system ---- */ + body { + background: + radial-gradient(circle at 72% -18%, color-mix(in srgb, var(--rust) 11%, transparent), transparent 34rem), + var(--paper); + } + .cockpit-command { + min-height: 4rem; padding: .75rem 1.5rem; color: #E8EEF8; + background: color-mix(in srgb, #08111F 94%, transparent); + border-bottom-color: #1E2C42; box-shadow: 0 12px 32px rgb(3 8 18 / 24%); + backdrop-filter: saturate(1.3) blur(16px); + } + .cockpit-command .brand { align-items: center; gap: .7rem; } + .cockpit-command .brand .mark { width: 23px; height: 23px; color: #69A7FF; } + .cockpit-command .brand .wm { color: #FFFFFF; font-size: 1.2rem; font-weight: 700; } + .cockpit-command .brand .tag { color: #8190A8; } + .cockpit-command .lnk { color: #9AA8BC; } + .cockpit-command .pill { border: 1px solid #2B405F; background: #101D2F; } + + .cockpit-shell { grid-template-columns: 17rem minmax(0, 1fr); max-width: none; min-height: calc(100vh - 4rem); } + .cockpit-rail { + top: 4rem; height: calc(100vh - 4rem); padding: 1.7rem 1rem 2rem; + color: #B8C4D6; background: linear-gradient(180deg, #0B1422 0%, #0A111C 100%); + border-right-color: #1C2A3D; box-shadow: inset -1px 0 rgb(255 255 255 / 2%); + } + .cockpit-rail .navgroup { margin-bottom: 1.75rem; } + .cockpit-rail .navgroup h3 { margin: 0 0 .65rem .75rem; color: #66758C; font-size: .59rem; } + .cockpit-rail .navlink { + position: relative; min-height: 2.55rem; margin: .16rem 0; padding: .58rem .72rem; + border: 1px solid transparent; border-radius: 10px; color: #9EACC0; + } + .cockpit-rail .navlink:hover { color: #F5F8FC; background: #111F32; border-color: #20314A; } + .cockpit-rail .navlink.active { + color: #FFFFFF; background: linear-gradient(90deg, #17345A, #122843); + border-color: #27507E; box-shadow: 0 8px 20px rgb(0 0 0 / 18%); + } + .cockpit-rail .navlink.active::before { + content: ""; position: absolute; left: -.35rem; top: .55rem; bottom: .55rem; width: 3px; + border-radius: 999px; background: #65A8FF; box-shadow: 0 0 14px #65A8FF; + } + .cockpit-rail .navlink .ico { color: #6F829E; } + .cockpit-rail .navlink.active .ico { color: #78B2FF; } + .cockpit-rail .navcount { color: #6F7F96; } + .cockpit-rail .navlink.active .navcount { color: #C9DFFF; } + .cockpit-rail .navfoot { margin: 1.4rem .55rem 0; padding-top: 1rem; border-color: #213047; color: #66758C; } + .cockpit-rail .navfoot b { color: #AEBBD0; } + + .cockpit-workspace { width: 100%; max-width: 104rem; padding: 2rem clamp(1.5rem, 3vw, 3.5rem) 4rem; } + .cockpit-workspace > .view:not([hidden]) { + padding: clamp(1.25rem, 2.2vw, 2rem); + border: 1px solid var(--line); border-radius: 20px; background: color-mix(in srgb, var(--paper-2) 96%, transparent); + box-shadow: 0 1px 1px rgb(16 24 40 / 3%), 0 22px 60px rgb(22 38 67 / 9%); + } + .cockpit-workspace > .board-head { + margin: 0 0 1rem; padding: clamp(1.4rem, 2.2vw, 2rem); border: 1px solid var(--line); + border-radius: 20px; overflow: hidden; + background: linear-gradient(135deg, var(--paper-2), color-mix(in srgb, var(--rust-bg) 58%, var(--paper-2))); + box-shadow: 0 1px 1px rgb(16 24 40 / 3%), 0 22px 60px rgb(22 38 67 / 8%); + } + .cockpit-workspace .band { padding-top: 0; } + .cockpit-workspace .board-title, .cockpit-workspace h2.head { + font-family: var(--sans); font-weight: 720; letter-spacing: -.035em; + } + .cockpit-workspace h2.head { margin: 0 0 .3rem; font-size: clamp(1.65rem, 2.3vw, 2.25rem); } + .cockpit-workspace .crumbs { margin-bottom: .75rem; } + .cockpit-workspace .trace-note { max-width: 72rem; font-family: var(--sans); font-size: .84rem; } + + .cockpit-workspace .kpis { gap: .7rem; margin-top: 1.5rem; border: 0; background: transparent; box-shadow: none; } + .cockpit-workspace .kpi { + min-height: 7rem; padding: 1rem 1.1rem; border: 1px solid var(--line) !important; + border-radius: 14px; background: color-mix(in srgb, var(--paper-2) 92%, transparent); + } + .cockpit-workspace .kpi.act { background: linear-gradient(145deg, var(--rust-bg), var(--paper-2)); } + .cockpit-workspace .attention { margin-top: .75rem; padding: .2rem .8rem; border: 1px solid var(--line); + border-radius: 14px; background: color-mix(in srgb, var(--paper-2) 80%, transparent); } + + .cockpit-workspace .review-card, .cockpit-workspace .trace-list, + .cockpit-workspace .trace-skills, .cockpit-workspace .cluster-plot, + .cockpit-workspace .cluster-side, .cockpit-workspace .mx-wrap, + .cockpit-workspace .mx-chart, .cockpit-workspace .empty, + .cockpit-workspace .skill-run-log { + border-radius: 16px; border-color: var(--line-2); background: var(--paper-2); + box-shadow: 0 1px 2px rgb(16 24 40 / 4%), 0 12px 32px rgb(23 38 65 / 7%); + } + .cockpit-workspace .review-card { padding: 1.5rem; } + .cockpit-workspace .trace-stats { gap: .6rem; border: 0; background: transparent; box-shadow: none; } + .cockpit-workspace .trace-stat { border: 1px solid var(--line) !important; border-radius: 12px; + background: var(--paper-2); } + .cockpit-workspace .trace-row, .cockpit-workspace .srow, .cockpit-workspace .hrow { + padding: .9rem 1rem; transition: background .12s, border-color .12s; + } + .cockpit-workspace .trace-row:hover, .cockpit-workspace .srow:hover, + .cockpit-workspace .hrow:hover { background: var(--paper-3); } + .cockpit-workspace .skilllist, .cockpit-workspace .hlist { overflow: hidden; border: 1px solid var(--line); + border-radius: 16px; background: var(--paper-2); } + .cockpit-workspace .skill-folder-head { padding: .8rem 1rem; background: var(--paper-3); } + + .cockpit-workspace .version-field label { font-size: .61rem; color: var(--ink-3); } + .cockpit-workspace input, .cockpit-workspace select, .cockpit-workspace textarea { + min-height: 2.7rem; border-radius: 10px !important; background: var(--paper-2) !important; + box-shadow: inset 0 1px 2px rgb(16 24 40 / 4%); + } + .cockpit-workspace .btn { min-height: 2.55rem; padding: .65rem 1rem; border-radius: 10px; } + .cockpit-workspace .btn.primary { box-shadow: 0 7px 18px color-mix(in srgb, var(--rust) 25%, transparent); } + + .cockpit-workspace .mx-wrap { border-radius: 16px; } + .cockpit-workspace .mx th { background: var(--paper-3); } + .cockpit-workspace .mx td { height: 4.5rem; } + .cockpit-workspace .mx tr:hover td { background: color-mix(in srgb, var(--rust-bg) 35%, var(--paper-2)); } + .cockpit-workspace .mx-sticky { background: var(--paper-2) !important; } + .cockpit-workspace .mx tr:hover .mx-sticky { background: color-mix(in srgb, var(--rust-bg) 35%, var(--paper-2)) !important; } + .cockpit-workspace .mx-chart { padding: 1.25rem; } + .cockpit-workspace .mx-chart-detail { border: 1px solid var(--line); } + .cockpit-workspace .lane-row { padding: .75rem 1rem; border-radius: 12px; } + + .cmp-overlay { backdrop-filter: blur(8px); } + .cmp-modal { border-radius: 20px; box-shadow: 0 32px 90px rgb(0 0 0 / 35%); } + + @media (prefers-color-scheme: dark) { + .cockpit-workspace > .view:not([hidden]), .cockpit-workspace > .board-head { + box-shadow: 0 1px 0 rgb(255 255 255 / 2%) inset, 0 26px 80px rgb(0 0 0 / 28%); + } + } + @media (max-width: 900px) { + .cockpit-shell { grid-template-columns: 1fr; } + .cockpit-rail { position: static; height: auto; padding: .75rem 1rem; border-right: 0; + border-bottom: 1px solid #203049; } + .cockpit-rail .navgroup { display: flex; gap: .35rem; margin: 0 0 .35rem; overflow-x: auto; } + .cockpit-rail .navgroup h3, .cockpit-rail .navfoot { display: none; } + .cockpit-rail #nav-folders { display: contents; } + .cockpit-rail .navlink { flex: 0 0 auto; min-height: 2.35rem; padding: .48rem .72rem; } + .cockpit-rail .navlink.active::before { left: .7rem; right: .7rem; top: auto; bottom: -.2rem; + width: auto; height: 3px; } + .cockpit-workspace { padding: 1rem; } + } + @media (max-width: 640px) { + .cockpit-command { min-height: 3.5rem; padding: .65rem 1rem; } + .cockpit-command .brand .tag { display: none; } + .cockpit-workspace > .view:not([hidden]), .cockpit-workspace > .board-head { padding: 1rem; + border-radius: 15px; } + .cockpit-workspace .review-card, .cockpit-workspace .mx-chart { padding: .9rem; } + } -