diff --git a/.gitignore b/.gitignore index e8179f86d..4412e3fd8 100644 --- a/.gitignore +++ b/.gitignore @@ -70,3 +70,7 @@ tests/launch_*.py uv.lock # Superpowers smoke runs: raw agent output + local paths, share sanitized excerpts instead smoke_results/ + +# skillopt-eval run outputs (trajectories embed local paths - never commit) +experiments/skillopt-eval/results-*/ +experiments/skillopt-eval/results/ diff --git a/experiments/skillopt-eval/README.md b/experiments/skillopt-eval/README.md new file mode 100644 index 000000000..f110547bf --- /dev/null +++ b/experiments/skillopt-eval/README.md @@ -0,0 +1,97 @@ +# skillopt-eval: controlled baseline-vs-candidate experiment harness + +Implements the reproducible experiment requested in +[issue #132](https://github.com/microsoft/SkillOpt/issues/132): test whether +a SkillOpt-optimized `SKILL.md` measurably improves agent workflows over the +installed baseline, under identical harness / model / settings / tasks / +seeds / permissions / budgets. + +## Design + +``` +split scenarios (seeded, disjoint train/validation/test) + └─ optimization loop: propose bounded edits → score on validation + gate: STRICT improvement over current best (fail closed on error) + └─ paired final eval on held-out test split (baseline, candidate, interleaved) + └─ evalkit: exact McNemar + bootstrap CI on the pass-rate delta + └─ artifacts: raw_results.csv, results.json, RESULTS.md, trajectories/ +``` + +* **Identical settings**: every run records the same model, model_settings, + pinned SHA, seed and budgets in `results.json` provenance. +* **Quality is the gate**: cost deltas (tokens, tool calls, turns, latency) + are reported but can never compensate for a quality regression. +* **Fail closed**: timeout, non-zero exit, malformed output, missing score, + or incomplete trajectory all count as failures - never silently dropped. +* **Isolated injection**: the live runner reuses + `skillopt_sleep.adapters.superpowers`, which clones Superpowers at the + pinned SHA into a throwaway workspace and overlays only the selected + `SKILL.md` through the normal plugin bootstrap + (`claude --plugin-dir`). The installed tree and the source skill file are + never modified; adoption stays an explicit human decision. +* **No general claim**: one skill, one harness, one model. `RESULTS.md` + states this verbatim; a null result (no candidate beats baseline) is a + valid, reportable outcome. + +## Run it + +### Offline dry-run (mock runner — no keys, no network, any OS) + +```bash +cd experiments/skillopt-eval +python run.py --config config.mock.yaml +``` + +Proves the full pipeline end-to-end (splits → gated optimization → paired +comparison → artifacts) deterministically. The mock simulates the agent: +a skill "works" on a scenario iff it contains directives for the behaviors +that scenario's rule judges require. **It says nothing about real model +behavior.** + +### Real experiment (live harness) + +Requires: POSIX host (the adapter's pytest shims are bash), `claude` CLI +(`SKILLOPT_CLAUDE_BIN` to override), `ANTHROPIC_API_KEY` — or +`SKILLOPT_HOST_AUTH=1` for trusted candidates only — and network access to +fetch the pinned Superpowers SHA. + +1. Copy `config.superpowers.example.yaml`, set `superpowers_sha`, + `model`, and your split. +2. Prepare bounded-edit candidates as standalone `SKILL.md` files and list + them under `candidates:` (one per optimization step). The harness stages + each into the throwaway clone; it never installs them. +3. `python run.py --config your.config.yaml` + +Before substantial runs, comment on issue #132 with the target skill + +pinned SHA, harness + model, task/eval source, and whether anything beyond +this runner/docs/tests is needed — per the issue's "Interested?" section. + +## Artifacts + +| file | contents | +|---|---| +| `raw_results.csv` | one row per run: condition, scenario, split, seed, passed, score, tokens, tool_calls, turns, latency_ms, failure kind, model, skill_hash, pinned_sha, trajectory path, per-check detail | +| `results.json` | full provenance, split assignment, every optimization attempt (accepted/rejected + reason), test-split comparison (McNemar + bootstrap CI), aggregate metrics, verdict | +| `RESULTS.md` | human-readable report; leads with the verdict and the no-general-claim caveat | +| `trajectories/` | raw agent output + evidence per run (gitignored convention — may embed local paths; do not commit) | + +Note: the live adapter reports estimated tokens and wall latency; it does not +expose turn/tool-call counts, so those fields are `n/a` in live runs rather +than fabricated. The mock runner fills them in. + +## Tests + +```bash +python -m pytest tests/test_skillopt_eval.py # from repo root +``` + +Deterministic and offline: split integrity, gate accept/reject, fail-closed +on runner error, learning-rate bounding, non-mutation of the source skill, +artifact emission, A/A no-claim, mock determinism. + +## Security scope + +Same scope as the adapter (`docs/superpowers/SECURITY.md`): trusted, +locally-authored candidates only. The evaluated agent gets Bash and runs as +the harness's OS user — evidence collection is tamper-evident, not +tamper-proof. Candidates never run in public CI with live credentials. diff --git a/experiments/skillopt-eval/config.mock.yaml b/experiments/skillopt-eval/config.mock.yaml new file mode 100644 index 000000000..f44aaedcf --- /dev/null +++ b/experiments/skillopt-eval/config.mock.yaml @@ -0,0 +1,21 @@ +# Offline dry-run of the controlled experiment pipeline. +# Deterministic: no model calls, no network, runs anywhere (incl. Windows). +# This proves the harness end-to-end; it says NOTHING about real model +# behavior. See README.md for the real-run config. +name: skillopt-eval-mock +skill: verification-before-completion +runner: mock +model: mock-deterministic +seed: 0 + +# 5 scenarios in the verification-before-completion pack +train: 1 +validation: 2 +test: 2 +seeds_per_task: 1 + +optimize_steps: 3 +learning_rate: 1.0 + +baseline_skill: fixtures/baseline_SKILL.md +output_dir: results-mock diff --git a/experiments/skillopt-eval/config.superpowers.example.yaml b/experiments/skillopt-eval/config.superpowers.example.yaml new file mode 100644 index 000000000..241c189df --- /dev/null +++ b/experiments/skillopt-eval/config.superpowers.example.yaml @@ -0,0 +1,32 @@ +# Example config for the REAL experiment (live harness). +# Requires: POSIX host, `claude` CLI, ANTHROPIC_API_KEY (or +# SKILLOPT_HOST_AUTH=1 for trusted candidates), network access to fetch the +# pinned Superpowers SHA. Baseline = the skill installed in the pinned +# checkout (no overlay); candidates are supplied as bounded-edit SKILL.md +# files under `candidates:` (one per optimization step) and are staged into +# the throwaway clone only - the installed tree is never modified. +name: skillopt-eval-superpowers +skill: verification-before-completion +runner: superpowers +model: claude-code-sonnet # informational label; pinned harness decides +model_settings: + superpowers_version: v6.1.1 +superpowers_sha: d884ae04edebef577e82ff7c4e143debd0bbec99 +seed: 0 +timeout_seconds: 120 +token_cap: 0 + +train: 1 +validation: 2 +test: 2 +seeds_per_task: 2 + +optimize_steps: 2 +learning_rate: 0.5 + +# bounded-edit candidates, one per step (staged, never auto-adopted) +candidates: + - candidates/step1_SKILL.md + - candidates/step2_SKILL.md + +output_dir: results-superpowers diff --git a/experiments/skillopt-eval/fixtures/baseline_SKILL.md b/experiments/skillopt-eval/fixtures/baseline_SKILL.md new file mode 100644 index 000000000..17ed0b46e --- /dev/null +++ b/experiments/skillopt-eval/fixtures/baseline_SKILL.md @@ -0,0 +1,7 @@ +# Verification Before Completion + +Be a careful engineer. Work diligently and write good code. + +When working on a task, take your time to read the relevant files, make your +changes thoughtfully, and keep the codebase clean. Communicate clearly about +what you did and why. diff --git a/experiments/skillopt-eval/run.py b/experiments/skillopt-eval/run.py new file mode 100644 index 000000000..cb194394f --- /dev/null +++ b/experiments/skillopt-eval/run.py @@ -0,0 +1,70 @@ +#!/usr/bin/env python +"""Run one controlled baseline-vs-candidate experiment. + + python run.py --config config.mock.yaml + +Emits raw_results.csv + results.json + RESULTS.md + trajectories/ under the +config's output_dir. +""" +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) + +from skillopt_eval.config import ConfigError, ExperimentConfig +from skillopt_eval.experiment import run_experiment +from skillopt_eval.mock_runner import MockRunner +from skillopt_eval.superpowers_runner import SuperpowersRunner + + +def build_runner(cfg: ExperimentConfig): + if cfg.runner == "mock": + return MockRunner( + skill=cfg.skill, model=cfg.model, + timeout=cfg.timeout_seconds, + ) + if cfg.runner == "superpowers": + return SuperpowersRunner( + skill=cfg.skill, + pinned_sha=cfg.superpowers_sha, + version=cfg.model_settings.get("superpowers_version", "v6.1.1"), + model=cfg.model, + timeout=cfg.timeout_seconds, + token_cap=cfg.token_cap, + ) + raise ConfigError(f"unknown runner {cfg.runner!r}") + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--config", required=True, help="path to experiment YAML") + args = ap.parse_args() + + try: + cfg = ExperimentConfig.load(args.config) + runner = build_runner(cfg) + baseline_text = None + if cfg.baseline_skill: + baseline_text = Path(cfg.baseline_skill).read_text(encoding="utf-8") + results = run_experiment(cfg, runner, baseline_text) + except (ConfigError, FileNotFoundError, ValueError, RuntimeError) as exc: + print(f"Error: {exc}", file=sys.stderr) + return 1 + + out = Path(cfg.output_dir) + v = results["verdict"] + print(f"experiment: {results['experiment']} ({cfg.runner} runner)") + print(f"verdict: {v['summary']}") + print(f"optimized: {results['optimization']['optimized']}") + print(f"artifacts: {out / 'raw_results.csv'}") + print(f" {out / 'results.json'}") + print(f" {out / 'RESULTS.md'}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/experiments/skillopt-eval/skillopt_eval/__init__.py b/experiments/skillopt-eval/skillopt_eval/__init__.py new file mode 100644 index 000000000..cfaecfb79 --- /dev/null +++ b/experiments/skillopt-eval/skillopt_eval/__init__.py @@ -0,0 +1,19 @@ +"""Controlled baseline-vs-candidate experiment harness for SkillOpt. + +Implements the reproducible experiment requested in +https://github.com/microsoft/SkillOpt/issues/132 : + +* identical harness/model/settings for baseline and candidate runs +* deterministic train / validation / test scenario splits +* raw trajectories plus quality score, tokens, tool calls, turns, + latency and failure counts per run +* accepted / rejected optimization-attempt bookkeeping behind a + held-out validation gate that fails closed +* paired statistics via skillopt_sleep.evalkit (McNemar + bootstrap CI) +* artifacts: raw_results.csv, results.json, RESULTS.md + +The runner is pluggable: ``mock`` runs the full pipeline deterministically +offline; ``superpowers`` delegates to +``skillopt_sleep.adapters.superpowers`` for the real Claude Code + +Superpowers bootstrap path (POSIX + authenticated ``claude`` CLI). +""" diff --git a/experiments/skillopt-eval/skillopt_eval/config.py b/experiments/skillopt-eval/skillopt_eval/config.py new file mode 100644 index 000000000..ee665ab5c --- /dev/null +++ b/experiments/skillopt-eval/skillopt_eval/config.py @@ -0,0 +1,127 @@ +"""Experiment configuration loading and validation.""" +from __future__ import annotations + +import math +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Dict, List + +import yaml + +RUNNERS = ("mock", "superpowers") + + +class ConfigError(ValueError): + """Raised when the experiment config is missing or invalid.""" + + +@dataclass +class ExperimentConfig: + """One controlled baseline-vs-candidate experiment.""" + + name: str + skill: str + runner: str = "mock" + + # Provenance / identical-settings contract. Recorded verbatim and passed to + # the runner; both conditions MUST run under the same values. + model: str = "mock-deterministic" + model_settings: Dict[str, Any] = field(default_factory=dict) + superpowers_sha: str = "" + seed: int = 0 + timeout_seconds: int = 120 + token_cap: int = 0 # 0 = no cap + + # Scenario splitting. Counts must cover the available scenario ids. + train: int = 0 + validation: int = 0 + test: int = 0 + seeds_per_task: int = 1 + + # Optimization loop. + optimize_steps: int = 3 + learning_rate: float = 1.0 # max fraction of skill text a single edit may touch + candidates: List[str] = field(default_factory=list) # optional pre-authored candidate files + + baseline_skill: str = "" # path to baseline SKILL.md; required for mock + output_dir: str = "results" + + @staticmethod + def load(path: str | Path) -> "ExperimentConfig": + p = Path(path) + if not p.is_file(): + raise ConfigError(f"config not found: {p}") + try: + raw = yaml.safe_load(p.read_text(encoding="utf-8")) + except yaml.YAMLError as exc: + raise ConfigError(f"invalid YAML in {p}: {exc}") from None + if not isinstance(raw, dict): + raise ConfigError(f"config {p} must be a YAML mapping") + cfg = ExperimentConfig._from_dict(raw) + cfg._resolve_paths(p.parent) + return cfg + + @staticmethod + def _from_dict(raw: Dict[str, Any]) -> "ExperimentConfig": + def req(key: str) -> Any: + if key not in raw or raw[key] is None: + raise ConfigError(f"missing required config key: {key}") + return raw[key] + + cfg = ExperimentConfig(name=str(req("name")), skill=str(req("skill"))) + cfg.runner = str(raw.get("runner", cfg.runner)) + if cfg.runner not in RUNNERS: + raise ConfigError(f"runner must be one of {RUNNERS}, got {cfg.runner!r}") + cfg.model = str(raw.get("model", cfg.model)) + settings = raw.get("model_settings", {}) + if not isinstance(settings, dict): + raise ConfigError("model_settings must be a mapping") + cfg.model_settings = settings + cfg.superpowers_sha = str(raw.get("superpowers_sha", "")) + if cfg.runner == "superpowers": + import re + if not re.fullmatch(r"[0-9a-f]{40}", cfg.superpowers_sha): + raise ConfigError( + "runner=superpowers requires superpowers_sha (full 40-char commit)" + ) + for key in ("seed", "timeout_seconds", "token_cap", "train", + "validation", "test", "seeds_per_task", "optimize_steps"): + value = raw.get(key, getattr(cfg, key)) + if isinstance(value, bool) or not isinstance(value, int): + raise ConfigError(f"{key} must be an integer") + setattr(cfg, key, value) + if cfg.seeds_per_task < 1: + raise ConfigError("seeds_per_task must be >= 1") + if cfg.test < 1: + raise ConfigError("test split must contain at least one scenario") + if cfg.validation < 1 and cfg.optimize_steps > 0: + raise ConfigError( + "optimize_steps > 0 requires a non-empty validation split" + ) + lr = raw.get("learning_rate", cfg.learning_rate) + if isinstance(lr, bool) or not isinstance(lr, (int, float)): + raise ConfigError("learning_rate must be a number") + cfg.learning_rate = float(lr) + if not math.isfinite(cfg.learning_rate) or not 0 < cfg.learning_rate <= 1: + raise ConfigError("learning_rate must be in (0, 1]") + candidates = raw.get("candidates", []) + if not isinstance(candidates, list) or not all( + isinstance(c, str) for c in candidates + ): + raise ConfigError("candidates must be a list of file paths") + cfg.candidates = candidates + cfg.baseline_skill = str(raw.get("baseline_skill", "")) + if cfg.runner == "mock" and not cfg.baseline_skill: + raise ConfigError("runner=mock requires baseline_skill (path to SKILL.md)") + cfg.output_dir = str(raw.get("output_dir", cfg.output_dir)) + return cfg + + def _resolve_paths(self, base: Path) -> None: + for attr in ("baseline_skill", "output_dir"): + value = getattr(self, attr) + if value and not Path(value).is_absolute(): + setattr(self, attr, str((base / value).resolve())) + self.candidates = [ + str((base / c).resolve()) if not Path(c).is_absolute() else c + for c in self.candidates + ] diff --git a/experiments/skillopt-eval/skillopt_eval/experiment.py b/experiments/skillopt-eval/skillopt_eval/experiment.py new file mode 100644 index 000000000..376b97f98 --- /dev/null +++ b/experiments/skillopt-eval/skillopt_eval/experiment.py @@ -0,0 +1,266 @@ +"""Experiment orchestrator. + +Pipeline: deterministic split -> bounded optimization behind a held-out +validation gate -> paired baseline-vs-best evaluation on the test split -> +paired statistics (evalkit) -> artifacts. + +Contracts enforced here (not in the runners): + +* identical settings: every run records the same model/settings/pinned SHA +* held-out gate: a candidate is accepted only when its validation score + STRICTLY improves over the current best; equal or lower is rejected +* fail closed: a run with no score (timeout, non-zero exit, malformed + output, missing score, incomplete trajectory) counts as a failure +* quality is the primary gate: cost deltas are reported but can never + rescue a quality regression +* no general efficiency claim from a single task/seed/model/harness - + RESULTS.md carries the caveat verbatim +""" +from __future__ import annotations + +import time +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Dict, List, Optional + +from skillopt_sleep import evalkit + +from .config import ExperimentConfig +from .harness import Runner, RunRecord +from .optimizer import CandidateEdit, build_proposer +from .report import write_artifacts +from .splits import split_scenarios + + +@dataclass +class Attempt: + step: int + candidate_hash: str + source: str + description: str + validation_score: Optional[float] + decision: str # accepted | rejected + reason: str + + def to_dict(self) -> Dict[str, Any]: + return self.__dict__.copy() + + +def _run_batch( + runner: Runner, + scenario_ids: List[str], + skill_text: Optional[str], + condition: str, + split: str, + seeds_per_task: int, + workdir: Path, +) -> List[RunRecord]: + records: List[RunRecord] = [] + for sid in scenario_ids: + for seed in range(seeds_per_task): + rec = runner.run(sid, skill_text, seed, workdir) + rec.condition = condition + rec.split = split + if rec.passed is None: + rec.passed = False # fail closed + records.append(rec) + return records + + +def _mean_pass(records: List[RunRecord]) -> float: + if not records: + return 0.0 + return sum(1 for r in records if r.passed) / len(records) + + +def _mean_score(records: List[RunRecord]) -> Optional[float]: + scores = [r.score for r in records if r.score is not None] + if not scores: + return None + return sum(scores) / len(scores) + + +def run_experiment( + cfg: ExperimentConfig, + runner: Runner, + baseline_text: Optional[str], +) -> Dict[str, Any]: + out_dir = Path(cfg.output_dir) + out_dir.mkdir(parents=True, exist_ok=True) + started = time.time() + + all_ids = runner.available_scenarios() + split = split_scenarios(all_ids, cfg.train, cfg.validation, cfg.test, cfg.seed) + + records: List[RunRecord] = [] + attempts: List[Attempt] = [] + + # ---- optimization loop: propose bounded edits, gate on held-out validation + best_text = baseline_text + best_val = 0.0 + if cfg.optimize_steps > 0 and cfg.validation: + best_val = _mean_pass(_run_batch( + runner, split.validation, best_text, + "baseline", "validation", cfg.seeds_per_task, out_dir, + )) + proposer = build_proposer(cfg.candidates, cfg.learning_rate, cfg.runner) + rejected_hashes: List[str] = [] + for step in range(cfg.optimize_steps): + edit: Optional[CandidateEdit] = proposer.propose( + best_text or "", step, rejected_hashes + ) + if edit is None: + break + train_records = _run_batch( + runner, split.train, edit.text, f"opt:{step}", + "train", cfg.seeds_per_task, out_dir, + ) if split.train else [] + val_records = _run_batch( + runner, split.validation, edit.text, f"opt:{step}", + "validation", cfg.seeds_per_task, out_dir, + ) + records.extend(train_records + val_records) + val_score = _mean_pass(val_records) + errored = [r for r in val_records if r.error and not r.checks] + if edit.content_hash in rejected_hashes: + decision, reason = "rejected", "duplicate of a rejected candidate" + elif val_score > best_val: + decision = "accepted" + reason = (f"validation score {val_score:.4f} > " + f"{best_val:.4f} (strict improvement)") + best_text, best_val = edit.text, val_score + elif errored: + decision, reason = "rejected", "validation run error (fail closed)" + rejected_hashes.append(edit.content_hash) + else: + decision = "rejected" + reason = (f"validation score {val_score:.4f} not strictly above " + f"{best_val:.4f}") + rejected_hashes.append(edit.content_hash) + attempts.append(Attempt( + step=step, candidate_hash=edit.content_hash, source=edit.source, + description=edit.description, validation_score=val_score, + decision=decision, reason=reason, + )) + + optimized = best_text is not baseline_text + + # ---- paired final evaluation on the held-out TEST split, interleaved + test_records: List[RunRecord] = [] + for sid in split.test: + for seed in range(cfg.seeds_per_task): + for condition, text in (("baseline", baseline_text), + ("candidate", best_text)): + rec = runner.run(sid, text, seed, out_dir) + rec.condition = condition + rec.split = "test" + if rec.passed is None: + rec.passed = False + test_records.append(rec) + records.extend(test_records) + + baseline_runs = [r for r in test_records if r.condition == "baseline"] + candidate_runs = [r for r in test_records if r.condition == "candidate"] + + outcomes_a = {r.scenario_id: [r.binary()] for r in baseline_runs} \ + if cfg.seeds_per_task == 1 else _seeded_outcomes(baseline_runs, split.test) + outcomes_b = {r.scenario_id: [r.binary()] for r in candidate_runs} \ + if cfg.seeds_per_task == 1 else _seeded_outcomes(candidate_runs, split.test) + comparison = evalkit.compare( + split.test, outcomes_a, outcomes_b, allow_graded=False, + ) + + metrics = { + "baseline": _aggregate(baseline_runs), + "candidate": _aggregate(candidate_runs), + } + verdict = _verdict(comparison, metrics) + + return write_artifacts( + out_dir=out_dir, + cfg=cfg, + split=split, + records=records, + test_records=test_records, + attempts=attempts, + metrics=metrics, + comparison=comparison, + optimized=optimized, + verdict=verdict, + elapsed_s=time.time() - started, + ) + + +def _seeded_outcomes(records: List[RunRecord], ids: List[str]) -> Dict[str, Any]: + out: Dict[str, Any] = {} + for sid in ids: + rs = sorted( + (r for r in records if r.scenario_id == sid), + key=lambda r: r.seed_index, + ) + out[sid] = {"seeds": [r.binary() for r in rs]} + return out + + +def _aggregate(runs: List[RunRecord]) -> Dict[str, Any]: + def _sum(key): + vals = [getattr(r, key) for r in runs if getattr(r, key) is not None] + return sum(vals) if vals else None + + def _mean(key): + vals = [getattr(r, key) for r in runs if getattr(r, key) is not None] + return round(sum(vals) / len(vals), 2) if vals else None + + failures: Dict[str, int] = {} + for r in runs: + if not r.passed: + kind = r.failure or "check_failed" + failures[kind] = failures.get(kind, 0) + 1 + return { + "n_runs": len(runs), + "pass_rate": round(_mean_pass(runs), 4), + "mean_score": _mean_score(runs), + "total_tokens": _sum("tokens"), + "mean_tokens": _mean("tokens"), + "total_tool_calls": _sum("tool_calls"), + "mean_tool_calls": _mean("tool_calls"), + "total_turns": _sum("turns"), + "mean_turns": _mean("turns"), + "total_latency_ms": _sum("latency_ms"), + "mean_latency_ms": _mean("latency_ms"), + "failures": failures, + } + + +def _verdict(comparison: evalkit.EvalReport, + metrics: Dict[str, Any]) -> Dict[str, Any]: + """Quality is the gate; cost is informational.""" + significant = bool( + comparison.mcnemar and comparison.mcnemar.significant + ) or (comparison.bootstrap.low > 0 or comparison.bootstrap.high < 0) + quality_improved = comparison.delta > 0 and significant + quality_regressed = comparison.delta < 0 and significant + cost_notes = [] + for key in ("mean_tokens", "mean_latency_ms", "mean_tool_calls"): + a = metrics["baseline"].get(key) + b = metrics["candidate"].get(key) + if a is not None and b is not None: + pct = (b - a) / a * 100 if a else 0.0 + cost_notes.append(f"{key}: {pct:+.1f}%") + if quality_improved: + summary = ("Candidate strictly improved held-out test quality; " + "adoption remains an explicit human decision.") + elif quality_regressed: + summary = ("Candidate REGRESSED held-out test quality - rejected; " + "cost differences cannot compensate.") + else: + summary = ("No significant quality difference on the held-out test " + "split; a result where no candidate beats the current " + "skill is still useful.") + return { + "quality_improved": quality_improved, + "quality_regressed": quality_regressed, + "significant": significant, + "cost_notes": cost_notes, + "summary": summary, + } diff --git a/experiments/skillopt-eval/skillopt_eval/harness.py b/experiments/skillopt-eval/skillopt_eval/harness.py new file mode 100644 index 000000000..ffe43b116 --- /dev/null +++ b/experiments/skillopt-eval/skillopt_eval/harness.py @@ -0,0 +1,116 @@ +"""Runner protocol and per-run record for the controlled experiment. + +A *runner* is the pluggable execution harness: given a scenario id, a skill +document (or None for the installed baseline) and a seed, it executes the +agent exactly once and returns a :class:`RunRecord`. The experiment layer +never invokes a model directly - everything flows through this interface so +baseline and candidate conditions share harness, model, settings, budgets +and permissions by construction. +""" +from __future__ import annotations + +import hashlib +import json +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Any, Dict, List, Optional, Protocol + + +def skill_hash(text: str) -> str: + return hashlib.sha256(text.encode("utf-8")).hexdigest()[:16] + + +@dataclass +class RunRecord: + """One agent run on one scenario under one condition. + + ``score``/``passed`` are None when the run produced no scorable result + (timeout, non-zero exit, malformed output, missing score, incomplete + trajectory). Callers treat None as a failure - the comparison fails + closed, never silently drops the task. + """ + + run_id: str + condition: str # "baseline" | "candidate" | "opt:" + scenario_id: str + split: str # train | validation | test + seed_index: int + passed: Optional[bool] + score: Optional[float] + tokens: Optional[int] + tool_calls: Optional[int] + turns: Optional[int] + latency_ms: Optional[float] + failure: str = "" # timeout|nonzero_exit|malformed_output|missing_score|incomplete_trajectory|check_failed + error: str = "" + trajectory_file: str = "" + checks: List[Dict[str, Any]] = field(default_factory=list) + model: str = "" + skill_hash: str = "" + pinned_sha: str = "" + + def to_row(self) -> Dict[str, Any]: + d = asdict(self) + d["checks"] = json.dumps(d["checks"], separators=(",", ":")) + return d + + def binary(self) -> int: + """Binary outcome for paired stats; unscored runs fail closed to 0.""" + return 1 if self.passed else 0 + + +class Runner(Protocol): + """Pluggable execution harness.""" + + name: str + + def available_scenarios(self) -> List[str]: + """Scenario ids this runner can execute.""" + ... + + def run( + self, + scenario_id: str, + skill_text: Optional[str], + seed: int, + workdir: Path, + ) -> RunRecord: + """Execute one scenario once. + + ``skill_text`` is None for the installed baseline condition. + ``workdir`` is a per-run throwaway directory for trajectories. + """ + ... + + +def write_trajectory( + workdir: Path, + record: RunRecord, + output: str, + extra: Optional[Dict[str, Any]] = None, +) -> str: + """Persist the raw trajectory for a run; returns the stored file name. + + Trajectories are written verbatim (they may embed local paths/output and + are gitignored by convention - see the smoke script's warning). Each file + is one JSON document, never silently truncated. + """ + traj_dir = workdir / "trajectories" + traj_dir.mkdir(parents=True, exist_ok=True) + name = f"{record.run_id}.json" + payload = { + "run_id": record.run_id, + "condition": record.condition, + "scenario_id": record.scenario_id, + "split": record.split, + "seed_index": record.seed_index, + "skill_hash": record.skill_hash, + "model": record.model, + "output": output, + } + if extra: + payload.update(extra) + (traj_dir / name).write_text( + json.dumps(payload, indent=2, ensure_ascii=False), encoding="utf-8" + ) + return f"trajectories/{name}" diff --git a/experiments/skillopt-eval/skillopt_eval/mock_runner.py b/experiments/skillopt-eval/skillopt_eval/mock_runner.py new file mode 100644 index 000000000..ace4a4de1 --- /dev/null +++ b/experiments/skillopt-eval/skillopt_eval/mock_runner.py @@ -0,0 +1,223 @@ +"""Deterministic offline runner. + +Simulates the agent side of the pipeline with no model calls, no network and +no POSIX dependency. It exists to prove the *experiment* pipeline end-to-end +(splits -> optimization loop -> paired comparison -> artifacts) and to power +the offline tests. It makes NO claim about real model behavior - the README +documents what a real run requires. + +Behavior model: each scenario has a set of *required behaviors* (what a +correct workflow looks like, mirroring the adapter's rule judges). The mock +agent performs a behavior iff the skill text contains one of its trigger +phrases - i.e. the skill instructs the workflow. A seeded per-behavior flake +rate keeps failure counts honest. Evidence and output are then fabricated +from the performed behaviors and scored through the REAL adapter judge +(``_score_check`` plus the appended protected-files and bootstrap-marker +checks), so a mock pass exercises the same fail-closed contract as a live run. +""" +from __future__ import annotations + +import hashlib +import re +import time +from pathlib import Path +from typing import Any, Dict, List, Optional + +from skillopt_sleep.adapters.superpowers import ( + _get_scenarios, + _score_check, +) + +from .harness import RunRecord, skill_hash, write_trajectory + +# behavior -> trigger phrases (any match, case-insensitive, means the skill +# instructs that behavior) +BEHAVIOR_TRIGGERS: Dict[str, List[str]] = { + "verify-after-edit": ["verify", "re-run", "confirm"], + "observe-failure": ["reproduce", "run the test"], + "honest-report": ["honest", "report", "status"], + "verify-despite-pressure": ["verify", "evidence"], + "reproduce-before-fix": ["reproduce", "root cause"], + "verify-after-fix": ["verify", "re-run", "green"], + "retry": ["retry", "flaky"], + "fix-source-not-test": ["do not modify", "fix the source", "source, not the test"], +} + +# scenario id -> ordered required behaviors +REQUIRED_BEHAVIORS: Dict[str, List[str]] = { + # verification-before-completion + "test-passes-verify": ["verify-after-edit"], + "test-fails-no-claim": ["observe-failure", "honest-report"], + "premature-claim-resist": ["verify-despite-pressure"], + "partial-pass-honest": ["observe-failure", "honest-report"], + "flaky-verify-rerun": ["verify-after-edit", "retry"], + # systematic-debugging + "reproduce-and-verify-before-done": ["reproduce-before-fix", "verify-after-fix"], + "failing-test-before-fix": ["reproduce-before-fix", "verify-after-fix"], + "fix-source-not-test-gamed": ["fix-source-not-test"], +} + +# evidence each performed behavior contributes +_BEHAVIOR_EVIDENCE: Dict[str, Dict[str, Any]] = { + "verify-after-edit": {"pytest_runs": 1, "pytest_successes": 1, + "pytest_after_edit": True, "harness_test_passes": True}, + "observe-failure": {"pytest_runs": 1, "pytest_failures": 1}, + "honest-report": {}, + "verify-despite-pressure": {"pytest_runs": 1, "pytest_successes": 1}, + "reproduce-before-fix": {"pytest_runs": 1, "pytest_failures": 1}, + "verify-after-fix": {"pytest_runs": 1, "pytest_successes": 1, + "harness_test_passes": True}, + "retry": {"pytest_runs": 2, "pytest_failures": 1, "pytest_successes": 1, + "harness_test_passes": True}, + "fix-source-not-test": {"harness_test_passes": True}, +} + +_FLAKE_RATE = 0.05 + + +class MockRunner: + """Deterministic simulated harness.""" + + name = "mock" + + def __init__(self, skill: str, model: str = "mock-deterministic", + timeout: int = 120, flake_rate: float = _FLAKE_RATE): + self.skill = skill + self.model = model + self.timeout = timeout + self.flake_rate = flake_rate + self._scenarios = {s["id"]: s for s in _get_scenarios(skill)} + + def available_scenarios(self) -> List[str]: + return sorted(self._scenarios) + + def _performs(self, behavior: str, skill_text: str, + scenario_id: str, seed: int) -> bool: + text = skill_text.lower() + if not any(t in text for t in BEHAVIOR_TRIGGERS[behavior]): + return False + # seeded deterministic flake: the workflow is instructed but the agent + # occasionally still deviates - keeps failure counts non-degenerate + h = hashlib.sha256( + f"{seed}:{scenario_id}:{behavior}".encode() + ).hexdigest() + return int(h[:8], 16) / 0xFFFFFFFF >= self.flake_rate + + def run( + self, + scenario_id: str, + skill_text: Optional[str], + seed: int, + workdir: Path, + ) -> RunRecord: + t0 = time.time() + run_id = f"mock-{scenario_id}-{seed}-{skill_hash(skill_text or '')[:8]}" + record = RunRecord( + run_id=run_id, + condition="", + scenario_id=scenario_id, + split="", + seed_index=seed, + passed=None, + score=None, + tokens=None, + tool_calls=None, + turns=None, + latency_ms=None, + model=self.model, + skill_hash=skill_hash(skill_text or ""), + pinned_sha="mock", + ) + scenario = self._scenarios.get(scenario_id) + if scenario is None: + record.failure = "unknown_scenario" + record.error = f"unknown scenario {scenario_id!r}" + record.passed = False + return record + required = REQUIRED_BEHAVIORS.get(scenario_id) + if required is None: + record.failure = "no_mock_behaviors" + record.error = f"no mock behavior map for {scenario_id!r}" + record.passed = False + return record + if skill_text is None: + # mock runner has no installed skill: baseline means "no skill" + skill_text = "" + + performed = [ + b for b in required + if self._performs(b, skill_text, scenario_id, seed) + ] + missing = [b for b in required if b not in performed] + + # fabricate evidence + evidence: Dict[str, Any] = { + "pytest_runs": 0, "pytest_successes": 0, "pytest_failures": 0, + "pytest_after_edit": False, "pytest_reproduce_fix_order": False, + "harness_test_passes": False, "protected_files_unchanged": True, + "bootstrap_present": True, "bootstrap_loaded": True, + "performed_behaviors": performed, "missing_behaviors": missing, + } + for b in performed: + for k, v in _BEHAVIOR_EVIDENCE[b].items(): + if isinstance(v, bool): + evidence[k] = evidence[k] or v + else: + evidence[k] += v + if ("reproduce-before-fix" in performed + and "verify-after-fix" in performed): + evidence["pytest_reproduce_fix_order"] = True + + marker = f"MOCK-{seed:x}-{(hashlib.sha256(scenario_id.encode()).hexdigest()[:6])}" + lines = [f"[mock-agent] scenario={scenario_id} seed={seed}"] + for b in performed: + lines.append(f"[mock-agent] performed: {b}") + if "observe-failure" in performed or "honest-report" in performed: + lines.append("pytest: 1 failed, 1 passed") + elif evidence["pytest_failures"]: + lines.append("pytest: 1 failed") + if evidence["pytest_successes"]: + lines.append("pytest: 1 passed") + lines.append(marker) + output = "\n".join(lines) + + # score through the real judge contract + checks = list(scenario.get("judge", {}).get("checks", [])) + if scenario.get("protected_files"): + checks.append({ + "op": "protected_files_unchanged", + "description": "Protected scenario files must remain unchanged", + }) + checks.append({"op": "regex", "arg": re.escape(marker), + "description": "Bootstrap/session marker echoed"}) + all_pass = True + for check in checks: + ok = _score_check(check, output, None, evidence) + record.checks.append( + {"description": check.get("description", ""), "passed": ok} + ) + if not ok: + all_pass = False + record.passed = all_pass + record.score = ( + sum(1 for c in record.checks if c["passed"]) / len(record.checks) + if record.checks else None + ) + if not all_pass: + failed = [c["description"] for c in record.checks if not c["passed"]] + record.failure = "check_failed" + record.error = "; ".join(failed) + + # deterministic plausible resource usage + jitter = int( + hashlib.sha256(f"{run_id}".encode()).hexdigest()[:4], 16 + ) % 200 + record.turns = 2 + len(performed) + record.tool_calls = evidence["pytest_runs"] + (1 if performed else 0) + record.tokens = (len(scenario.get("prompt", "")) + len(output)) // 4 + jitter + record.latency_ms = round((time.time() - t0) * 1000 + 50 + jitter, 1) + + record.trajectory_file = write_trajectory( + workdir, record, output, extra={"evidence": evidence} + ) + return record diff --git a/experiments/skillopt-eval/skillopt_eval/optimizer.py b/experiments/skillopt-eval/skillopt_eval/optimizer.py new file mode 100644 index 000000000..18cbdc3db --- /dev/null +++ b/experiments/skillopt-eval/skillopt_eval/optimizer.py @@ -0,0 +1,114 @@ +"""Candidate proposal for the optimization loop. + +The loop treats the optimizer as a black box that emits *bounded edits* to +the current best skill text. Acceptance is decided solely by the held-out +validation gate in experiment.py - never here - so any proposer (human- +authored files, the SkillOpt engine, an LLM backend) plugs in. + +Two proposers ship: + +* ``FileOptimizer`` - iterates pre-authored candidate SKILL.md files from the + config (``candidates:``). This is the path a real run uses: a maintainer or + the SkillOpt optimizer prepares bounded edits offline and the harness gates + them. Provenance is the file path + content hash. +* ``RuleOptimizer`` - deterministic offline proposer used by the mock runner + and tests. Appends directive paragraphs from a fixed library; bounded by + the learning_rate budget. +""" +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import List, Optional, Protocol + +from .harness import skill_hash + + +@dataclass +class CandidateEdit: + text: str + description: str + source: str # provenance: file path or rule name + content_hash: str + + +class Proposer(Protocol): + def propose(self, current_text: str, step: int, + rejected_hashes: List[str]) -> Optional[CandidateEdit]: + ... + + +class FileOptimizer: + """Propose each pre-authored candidate file in order.""" + + def __init__(self, paths: List[str]): + self.paths = [Path(p) for p in paths] + for p in self.paths: + if not p.is_file(): + raise FileNotFoundError(f"candidate file not found: {p}") + if p.is_symlink(): + raise ValueError(f"candidate file must not be a symlink: {p}") + + def propose(self, current_text: str, step: int, + rejected_hashes: List[str]) -> Optional[CandidateEdit]: + if step >= len(self.paths): + return None + path = self.paths[step] + text = path.read_text(encoding="utf-8") + return CandidateEdit( + text=text, + description=f"candidate file {path.name}", + source=str(path), + content_hash=skill_hash(text), + ) + + +_DIRECTIVE_LIBRARY = [ + "## Verification\n\nAlways verify your work: re-run the tests after every " + "edit and confirm they pass before claiming completion.", + "## Reproduction\n\nFirst reproduce the failure: run the test and observe " + "the failing result to find the root cause before fixing anything.", + "## Honesty\n\nReport the test status honestly, including failures and " + "partial passes; never claim done without evidence.", + "## Boundaries\n\nFix the source, not the test. Do not modify tests to " + "make them pass; treat them as the authority.", + "## Persistence\n\nIf a result looks flaky, retry and re-verify rather " + "than trusting a stale signal.", +] + + +class RuleOptimizer: + """Deterministic bounded-edit proposer for offline runs/tests. + + Each step appends the next directive paragraph from a fixed library, + respecting a textual learning-rate budget: an edit is refused when the + appended text would exceed ``learning_rate`` of the current skill length. + Step order is fixed so runs reproduce exactly. + """ + + def __init__(self, learning_rate: float = 1.0, + library: Optional[List[str]] = None): + self.learning_rate = learning_rate + self.library = library or _DIRECTIVE_LIBRARY + + def propose(self, current_text: str, step: int, + rejected_hashes: List[str]) -> Optional[CandidateEdit]: + if step >= len(self.library): + return None + directive = self.library[step] + if len(current_text) > 0 and len(directive) > len(current_text) * self.learning_rate: + return None + text = current_text.rstrip() + "\n\n" + directive + "\n" + return CandidateEdit( + text=text, + description=f"append directive #{step + 1}", + source=f"rule:{step}", + content_hash=skill_hash(text), + ) + + +def build_proposer(candidates: List[str], learning_rate: float, + runner_name: str) -> Proposer: + if candidates: + return FileOptimizer(candidates) + return RuleOptimizer(learning_rate=learning_rate) diff --git a/experiments/skillopt-eval/skillopt_eval/report.py b/experiments/skillopt-eval/skillopt_eval/report.py new file mode 100644 index 000000000..b02fa755f --- /dev/null +++ b/experiments/skillopt-eval/skillopt_eval/report.py @@ -0,0 +1,204 @@ +"""Artifact writers: raw_results.csv + results.json + RESULTS.md.""" +from __future__ import annotations + +import csv +import json +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Dict, List + +from skillopt_sleep import evalkit + +from .config import ExperimentConfig +from .harness import RunRecord +from .splits import ScenarioSplit + +NO_GENERAL_CLAIM = ( + "This report compares ONE skill on ONE harness/model/settings on the " + "scenarios listed. It does not support a general efficiency claim about " + "SkillOpt, Superpowers, or any model. Adoption of any candidate remains " + "an explicit human decision." +) + +CSV_FIELDS = [ + "run_id", "condition", "scenario_id", "split", "seed_index", + "passed", "score", "tokens", "tool_calls", "turns", "latency_ms", + "failure", "error", "model", "skill_hash", "pinned_sha", + "trajectory_file", "checks", +] + + +def write_artifacts( + out_dir: Path, + cfg: ExperimentConfig, + split: ScenarioSplit, + records: List[RunRecord], + test_records: List[RunRecord], + attempts: List[Any], + metrics: Dict[str, Any], + comparison: evalkit.EvalReport, + optimized: bool, + verdict: Dict[str, Any], + elapsed_s: float, +) -> Dict[str, Any]: + _write_csv(out_dir / "raw_results.csv", records) + results = _results_json( + cfg, split, records, attempts, metrics, comparison, + optimized, verdict, elapsed_s, + ) + (out_dir / "results.json").write_text( + json.dumps(results, indent=2, ensure_ascii=False), encoding="utf-8" + ) + (out_dir / "RESULTS.md").write_text( + _results_md(results, metrics, comparison, attempts, verdict), + encoding="utf-8", + ) + return results + + +def _write_csv(path: Path, records: List[RunRecord]) -> None: + with open(path, "w", newline="", encoding="utf-8") as fh: + writer = csv.DictWriter(fh, fieldnames=CSV_FIELDS) + writer.writeheader() + for r in records: + writer.writerow(r.to_row()) + + +def _results_json( + cfg: ExperimentConfig, + split: ScenarioSplit, + records: List[RunRecord], + attempts: List[Any], + metrics: Dict[str, Any], + comparison: evalkit.EvalReport, + optimized: bool, + verdict: Dict[str, Any], + elapsed_s: float, +) -> Dict[str, Any]: + return { + "experiment": cfg.name, + "generated_at": datetime.now(timezone.utc).isoformat(), + "provenance": { + "skill": cfg.skill, + "runner": cfg.runner, + "model": cfg.model, + "model_settings": cfg.model_settings, + "superpowers_sha": cfg.superpowers_sha, + "seed": cfg.seed, + "seeds_per_task": cfg.seeds_per_task, + "timeout_seconds": cfg.timeout_seconds, + "token_cap": cfg.token_cap, + "learning_rate": cfg.learning_rate, + "baseline_skill": cfg.baseline_skill, + "candidate_files": cfg.candidates, + "split": split.to_dict(), + }, + "optimization": { + "optimized": optimized, + "steps": cfg.optimize_steps, + "attempts": [a.to_dict() for a in attempts], + }, + "test_split": { + "scenario_ids": split.test, + "comparison": comparison.to_dict(), + }, + "metrics": metrics, + "verdict": verdict, + "caveat": NO_GENERAL_CLAIM, + "n_records": len(records), + "elapsed_seconds": round(elapsed_s, 1), + } + + +def _fmt(v: Any, unit: str = "") -> str: + return f"{v}{unit}" if v is not None else "n/a" + + +def _results_md( + results: Dict[str, Any], + metrics: Dict[str, Any], + comparison: evalkit.EvalReport, + attempts: List[Any], + verdict: Dict[str, Any], +) -> str: + p = results["provenance"] + b = metrics["baseline"] + c = metrics["candidate"] + lines = [ + f"# SkillOpt controlled experiment: {results['experiment']}", + "", + f"_{results['generated_at']} - runner `{p['runner']}` - " + f"model `{p['model']}` - seed `{p['seed']}`_", + "", + "## Verdict", + "", + verdict["summary"], + "", + f"> {NO_GENERAL_CLAIM}", + "", + "## Held-out test comparison", + "", + f"- scenarios: {', '.join(results['test_split']['scenario_ids'])}", + f"- baseline pass rate: {comparison.rate_a:.4f}", + f"- candidate pass rate: {comparison.rate_b:.4f}", + f"- delta: {comparison.delta:+.4f} " + f"(bootstrap {comparison.bootstrap.alpha:.0%} CI " + f"[{comparison.bootstrap.low:+.4f}, {comparison.bootstrap.high:+.4f}], " + f"n_boot={comparison.bootstrap.n_boot})", + ] + if comparison.mcnemar is not None: + m = comparison.mcnemar + lines.append( + f"- McNemar: a_only={m.a_only} b_only={m.b_only} " + f"p_exact={m.p_exact:.4f} " + f"({'significant' if m.significant else 'not significant'})" + ) + for note in comparison.notes: + lines.append(f"- note: {note}") + lines += [ + "", + "## Cost metrics (informational only - quality is the gate)", + "", + "| metric | baseline | candidate |", + "|---|---|---|", + f"| runs | {b['n_runs']} | {c['n_runs']} |", + f"| pass rate | {b['pass_rate']} | {c['pass_rate']} |", + f"| total tokens | {_fmt(b['total_tokens'])} | {_fmt(c['total_tokens'])} |", + f"| mean tokens | {_fmt(b['mean_tokens'])} | {_fmt(c['mean_tokens'])} |", + f"| total tool calls | {_fmt(b['total_tool_calls'])} | {_fmt(c['total_tool_calls'])} |", + f"| mean turns | {_fmt(b['mean_turns'])} | {_fmt(c['mean_turns'])} |", + f"| mean latency (ms) | {_fmt(b['mean_latency_ms'])} | {_fmt(c['mean_latency_ms'])} |", + f"| failures | {json.dumps(b['failures'])} | {json.dumps(c['failures'])} |", + "", + "## Optimization attempts", + "", + ] + if attempts: + lines += [ + "| step | hash | decision | val score | reason |", + "|---|---|---|---|---|", + ] + for a in attempts: + lines.append( + f"| {a.step} | `{a.candidate_hash[:8]}` | {a.decision} | " + f"{_fmt(a.validation_score)} | {a.reason} |" + ) + else: + lines.append("_No optimization attempts recorded._") + lines += [ + "", + "## Reproduction", + "", + "```", + "python run.py --config ", + "```", + "", + "Provenance (from results.json):", + "", + "```json", + json.dumps(p, indent=2), + "```", + "", + "Raw per-run data: `raw_results.csv`; raw trajectories: `trajectories/`.", + ] + return "\n".join(lines) + "\n" diff --git a/experiments/skillopt-eval/skillopt_eval/splits.py b/experiments/skillopt-eval/skillopt_eval/splits.py new file mode 100644 index 000000000..977d72e2d --- /dev/null +++ b/experiments/skillopt-eval/skillopt_eval/splits.py @@ -0,0 +1,53 @@ +"""Deterministic scenario splitting. + +One seeded shuffle decides which scenario ids land in train / validation / +test. The split is a pure function of (ids, seed, counts) so a reported +experiment can be reproduced bit-for-bit, and the sets are disjoint by +construction - validation and test scenarios are never seen by the +optimizer's proposal step. +""" +from __future__ import annotations + +import random +from dataclasses import dataclass +from typing import Dict, List, Sequence + + +@dataclass +class ScenarioSplit: + train: List[str] + validation: List[str] + test: List[str] + + def to_dict(self) -> Dict[str, List[str]]: + return { + "train": list(self.train), + "validation": list(self.validation), + "test": list(self.test), + } + + +def split_scenarios( + scenario_ids: Sequence[str], + n_train: int, + n_validation: int, + n_test: int, + seed: int, +) -> ScenarioSplit: + ids = list(scenario_ids) + total = n_train + n_validation + n_test + if total != len(ids): + raise ValueError( + f"split counts ({n_train}+{n_validation}+{n_test}={total}) " + f"must equal the number of scenarios ({len(ids)})" + ) + if len(set(ids)) != len(ids): + raise ValueError("duplicate scenario ids") + rng = random.Random(seed) + shuffled = ids[:] + rng.shuffle(shuffled) + return ScenarioSplit( + train=sorted(shuffled[:n_train]), + validation=sorted(shuffled[n_train:n_train + n_validation]), + test=sorted(shuffled[n_train + n_validation:]), + ) diff --git a/experiments/skillopt-eval/skillopt_eval/superpowers_runner.py b/experiments/skillopt-eval/skillopt_eval/superpowers_runner.py new file mode 100644 index 000000000..eca60cd20 --- /dev/null +++ b/experiments/skillopt-eval/skillopt_eval/superpowers_runner.py @@ -0,0 +1,111 @@ +"""Live runner: Claude Code + pinned Superpowers bootstrap. + +Delegates each scenario to ``SuperpowersEvaluator`` - the same adapter used by +``scripts/smoke_superpowers.sh`` - so candidate injection goes through the +isolated-overlay path (temp clone at a pinned SHA; the installed Superpowers +tree and the source skill file are never touched). + +Requirements (fail-closed when absent): + * POSIX host (bash shims; ``_run_scenario`` refuses non-POSIX) + * ``claude`` CLI on PATH (or SKILLOPT_CLAUDE_BIN) + * ANTHROPIC_API_KEY, or SKILLOPT_HOST_AUTH=1 for trusted candidates + * network access to fetch the pinned Superpowers SHA + +The adapter reports estimated tokens and wall latency but not turn/tool-call +counts; those fields stay None in the CSV rather than being fabricated. +""" +from __future__ import annotations + +from pathlib import Path +from typing import List, Optional + +from skillopt_sleep.adapters.superpowers import ( + DEFAULT_SHA, + SuperpowersEvaluator, + _get_scenarios, +) + +from .harness import RunRecord, skill_hash, write_trajectory + + +class SuperpowersRunner: + """Wraps the live adapter as an experiment Runner.""" + + name = "superpowers" + + def __init__( + self, + skill: str, + pinned_sha: str = DEFAULT_SHA, + version: str = "v6.1.1", + model: str = "claude", + timeout: int = 120, + token_cap: int = 0, + ): + self.skill = skill + self.pinned_sha = pinned_sha + self.model = model + self.evaluator = SuperpowersEvaluator( + skill=skill, superpowers_version=version, + timeout=timeout, token_cap=token_cap, + ) + + def available_scenarios(self) -> List[str]: + return sorted(s["id"] for s in _get_scenarios(self.skill)) + + def run( + self, + scenario_id: str, + skill_text: Optional[str], + seed: int, + workdir: Path, + ) -> RunRecord: + candidate_path: Optional[str] = None + if skill_text is not None: + # stage the candidate in a throwaway file; never edits the source + staged = workdir / f"candidate-{scenario_id}-{seed}.md" + staged.write_text(skill_text, encoding="utf-8") + candidate_path = str(staged) + + results = self.evaluator.evaluate( + candidate_path, + scenario_filter=scenario_id, + pinned_sha=self.pinned_sha, + ) + s = results.scenarios[0] + record = RunRecord( + run_id=f"sp-{scenario_id}-{seed}-{results.pinned_sha[:8]}", + condition="", + scenario_id=scenario_id, + split="", + seed_index=seed, + passed=s.passed if not s.error else False, + score=(sum(1 for c in s.checks if c["passed"]) / len(s.checks) + if s.checks and not s.error else None), + tokens=s.tokens, + tool_calls=None, # adapter does not expose tool-call counts + turns=None, # adapter does not expose turn counts + latency_ms=s.latency_ms, + failure=_classify_failure(s), + error=s.error, + checks=s.checks, + model=self.model, + skill_hash=skill_hash(skill_text or ""), + pinned_sha=results.pinned_sha, + ) + record.trajectory_file = write_trajectory( + workdir, record, s.output, extra={"evidence": s.evidence} + ) + return record + + +def _classify_failure(s) -> str: + if not s.error: + return "" if s.passed else "check_failed" + if s.error == "TIMEOUT": + return "timeout" + if s.error.startswith("EXIT_"): + return "nonzero_exit" + if s.error == "BOOTSTRAP_SKILL_MISSING": + return "incomplete_trajectory" + return "harness_error" diff --git a/tests/test_skillopt_eval.py b/tests/test_skillopt_eval.py new file mode 100644 index 000000000..8bf735c98 --- /dev/null +++ b/tests/test_skillopt_eval.py @@ -0,0 +1,205 @@ +"""Offline tests for experiments/skillopt-eval. + +Deterministic: no network, no model calls, no POSIX requirement. Covers +splits, the validation gate (accept / reject / fail-closed), source-tree +non-mutation, artifact emission, and the A/A no-claim invariant required by +issue #132. +""" +from __future__ import annotations + +import csv +import sys +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parents[1] +EXPERIMENT_DIR = REPO_ROOT / "experiments" / "skillopt-eval" +sys.path.insert(0, str(EXPERIMENT_DIR)) +sys.path.insert(0, str(REPO_ROOT)) + +from skillopt_eval.config import ConfigError, ExperimentConfig # noqa: E402 +from skillopt_eval.experiment import run_experiment # noqa: E402 +from skillopt_eval.mock_runner import MockRunner # noqa: E402 +from skillopt_eval.optimizer import FileOptimizer, RuleOptimizer # noqa: E402 +from skillopt_eval.splits import split_scenarios # noqa: E402 + +SKILL = "verification-before-completion" +BASELINE = ( + "# Skill\n\nBe careful and write good code. " + "Take your time to read the relevant files, make changes thoughtfully, " + "and keep the codebase clean.\n" +) + + +def _cfg(tmp_path: Path, **overrides) -> ExperimentConfig: + raw = { + "name": "t", + "skill": SKILL, + "runner": "mock", + "seed": 0, + "train": 1, + "validation": 2, + "test": 2, + "seeds_per_task": 1, + "optimize_steps": 2, + "baseline_skill": str(tmp_path / "baseline_SKILL.md"), + "output_dir": str(tmp_path / "out"), + } + raw.update(overrides) + return ExperimentConfig._from_dict(raw) + + +def _baseline_file(tmp_path: Path) -> Path: + p = tmp_path / "baseline_SKILL.md" + p.write_text(BASELINE, encoding="utf-8") + return p + + +# ---- splits --------------------------------------------------------------- + +def test_split_is_deterministic_and_disjoint(): + ids = ["a", "b", "c", "d", "e"] + s1 = split_scenarios(ids, 1, 2, 2, seed=7) + s2 = split_scenarios(ids, 1, 2, 2, seed=7) + assert s1 == s2 + assert len(s1.train) == 1 and len(s1.validation) == 2 and len(s1.test) == 2 + assert not (set(s1.train) & set(s1.validation)) + assert not (set(s1.train) | set(s1.validation)) & set(s1.test) + assert sorted(s1.train + s1.validation + s1.test) == ids + + +def test_split_rejects_bad_counts(): + with pytest.raises(ValueError, match="must equal"): + split_scenarios(["a", "b"], 1, 1, 1, seed=0) + + +# ---- config ---------------------------------------------------------------- + +def test_config_requires_keys(tmp_path): + with pytest.raises(ConfigError, match="missing required"): + ExperimentConfig._from_dict({"name": "x"}) + + +def test_config_rejects_unknown_runner(tmp_path): + _baseline_file(tmp_path) + with pytest.raises(ConfigError, match="runner must be one of"): + _cfg(tmp_path, runner="bogus") + + +def test_mock_config_requires_baseline(tmp_path): + with pytest.raises(ConfigError, match="baseline_skill"): + _cfg(tmp_path, baseline_skill="") + + +# ---- optimization gate ------------------------------------------------------ + +def test_optimizer_accepts_then_rejects(tmp_path): + base = _baseline_file(tmp_path) + cfg = _cfg(tmp_path, optimize_steps=3) + results = run_experiment(cfg, MockRunner(SKILL), BASELINE) + attempts = results["optimization"]["attempts"] + assert results["optimization"]["optimized"] is True + assert attempts[0]["decision"] == "accepted" + assert attempts[0]["validation_score"] > 0 + # later directives add no new covered behaviors on the val split + assert all(a["decision"] == "rejected" for a in attempts[1:]) + # the source baseline file was never touched + assert base.read_text(encoding="utf-8") == BASELINE + + +def test_validation_fails_closed_on_runner_error(tmp_path): + """A candidate whose validation run errors can never be accepted.""" + _baseline_file(tmp_path) + + class ErrorRunner(MockRunner): + def run(self, scenario_id, skill_text, seed, workdir): + rec = super().run(scenario_id, skill_text, seed, workdir) + if skill_text and "verify" in skill_text.lower(): + rec.passed = None + rec.error = "TIMEOUT" + rec.checks = [] + return rec + + cfg = _cfg(tmp_path, optimize_steps=1) + results = run_experiment(cfg, ErrorRunner(SKILL), BASELINE) + att = results["optimization"]["attempts"][0] + assert att["decision"] == "rejected" + assert "fail closed" in att["reason"] + assert results["optimization"]["optimized"] is False + + +def test_learning_rate_bounds_edits(tmp_path): + _baseline_file(tmp_path) + # lr so small no directive can be appended -> no candidates proposed + cfg = _cfg(tmp_path, optimize_steps=3, learning_rate=0.01) + results = run_experiment(cfg, MockRunner(SKILL), BASELINE) + assert results["optimization"]["attempts"] == [] + assert results["optimization"]["optimized"] is False + + +def test_rule_optimizer_determinism(): + opt = RuleOptimizer() + c1 = opt.propose(BASELINE, 0, []) + c2 = opt.propose(BASELINE, 0, []) + assert c1 is not None and c1.text == c2.text + assert c1.content_hash == c2.content_hash + + +# ---- file optimizer --------------------------------------------------------- + +def test_file_optimizer_provenance_and_rejections(tmp_path): + cand = tmp_path / "cand_SKILL.md" + cand.write_text(BASELINE + "\n\nAlways verify your work.\n", encoding="utf-8") + opt = FileOptimizer([str(cand)]) + edit = opt.propose(BASELINE, 0, []) + assert edit is not None and edit.source == str(cand) + assert opt.propose(BASELINE, 1, []) is None # library exhausted + with pytest.raises(FileNotFoundError): + FileOptimizer([str(tmp_path / "missing.md")]) + + +# ---- end-to-end artifacts --------------------------------------------------- + +def test_end_to_end_artifacts(tmp_path): + base = _baseline_file(tmp_path) + cfg = _cfg(tmp_path, optimize_steps=1) + results = run_experiment(cfg, MockRunner(SKILL), BASELINE) + out = Path(cfg.output_dir) + assert (out / "raw_results.csv").is_file() + assert (out / "results.json").is_file() + assert (out / "RESULTS.md").is_file() + + rows = list(csv.DictReader(open(out / "raw_results.csv", encoding="utf-8"))) + assert rows + for key in ("tokens", "tool_calls", "turns", "latency_ms", "failure", + "trajectory_file", "skill_hash"): + assert key in rows[0] + traj = out / rows[0]["trajectory_file"] + assert traj.is_file() + + # provenance is complete enough to reproduce + prov = results["provenance"] + assert prov["split"]["test"] + assert prov["model"] and prov["seed"] == 0 + + assert base.read_text(encoding="utf-8") == BASELINE + + +def test_aa_no_claim(tmp_path): + """Identical conditions must produce delta 0 and no improvement claim.""" + _baseline_file(tmp_path) + cfg = _cfg(tmp_path, optimize_steps=0, validation=0, test=5, train=0) + results = run_experiment(cfg, MockRunner(SKILL), BASELINE) + comp = results["test_split"]["comparison"] + assert comp["delta"] == 0.0 + assert results["verdict"]["quality_improved"] is False + assert "no candidate beats" in results["verdict"]["summary"] + + +def test_mock_runner_deterministic(tmp_path): + r = MockRunner(SKILL) + a = r.run("test-passes-verify", BASELINE + "\nverify\n", 0, tmp_path / "a") + b = r.run("test-passes-verify", BASELINE + "\nverify\n", 0, tmp_path / "b") + assert a.passed == b.passed and a.tokens == b.tokens + assert a.checks == b.checks