From 7e45e10f29c6732de59a4e36674774cfe7a44108 Mon Sep 17 00:00:00 2001 From: Toni Nowak Date: Sun, 27 Sep 2026 14:16:12 +0200 Subject: [PATCH] fix(pilot): observe configured triage acknowledgment timing --- PILOT.md | 4 ++ TASKS.md | 4 ++ scripts/pilot_contract_triage_pair.py | 4 +- tests/test_pilot_contract_case_profile.py | 84 +++++++++++++++++++++++ 4 files changed, 95 insertions(+), 1 deletion(-) diff --git a/PILOT.md b/PILOT.md index fd816be..3b86a86 100644 --- a/PILOT.md +++ b/PILOT.md @@ -1727,3 +1727,7 @@ The source audit found that the diagnostic action `confirm_behavior_contract` co Per-arm task correctness and validated completion time are independent of optional advisory delivery. Changed tests, protected fixture files, an incorrect repair, unavailable validation, or an incomplete agent run prevent acceptance. The original failed focused result remains recorded even after a successful repair. Existing invoice trial data are unchanged. This integration is an offline measurement capability, not delivery or efficacy evidence. Provider access remains disabled for these profiles; no receipt may label local fallback or abstention as accepted remote ranking. A subsequent remote trial requires verified credential isolation and typed ranking receipts before preregistration. + +## VCR391 — Profile advice acknowledgment timing + +The profile event parser now observes acknowledgment after both legacy and configured valid triage results. A real synthetic event stream verifies acknowledgment before the next tool and distinguishes missing or late acknowledgment. This fixes delivery measurement only; it does not establish native delivery or efficacy. diff --git a/TASKS.md b/TASKS.md index 1e5eb64..1525bc3 100644 --- a/TASKS.md +++ b/TASKS.md @@ -1131,3 +1131,7 @@ The runner now writes the bounded redacted measurement to its private output dir - Added a bounded JSON case-profile interface to the existing runner: fixture/task paths, focused command, evidence and failure markers, reviewed triage enums, and a supervisor-owned SHA-256-pinned oracle outside the agent fixture. The default invoice protocol remains available without a profile. - Both repair arms receive the same task and independent oracle. Acceptance requires the initial observed failure, a successful post-edit focused test/full suite, an allowed source-only repair, immutable tests and other fixture files, and a passing oracle. Validated completion includes agent and independent validation time. - Optional advice delivery is scored separately from task correctness; abstaining from triage cannot turn a failed repair into success. Current profiles run with provider credentials and network access disabled: remote triage comparisons need a separately verified transport integration. No new model trial or historical rescore occurred. + +## VCR391 — Profile advice acknowledgment timing + +The profile event parser now observes acknowledgment after both legacy and configured valid triage results. A real synthetic event stream verifies acknowledgment before the next tool and distinguishes missing or late acknowledgment. This fixes delivery measurement only; it does not establish native delivery or efficacy. diff --git a/scripts/pilot_contract_triage_pair.py b/scripts/pilot_contract_triage_pair.py index 90e3a9f..5dccdaf 100644 --- a/scripts/pilot_contract_triage_pair.py +++ b/scripts/pilot_contract_triage_pair.py @@ -682,7 +682,9 @@ def _event_receipts( triage_output_status = "valid_local_step" else: triage_output_status = "valid_configured_choice" - if triage_output_status == "valid_local_step": + if triage_output_status in { + "valid_local_step", "valid_configured_choice", + }: triage_result_index = index else: triage_output_status = "out_of_order_or_unverified" diff --git a/tests/test_pilot_contract_case_profile.py b/tests/test_pilot_contract_case_profile.py index fe3cb26..ae4a850 100644 --- a/tests/test_pilot_contract_case_profile.py +++ b/tests/test_pilot_contract_case_profile.py @@ -194,6 +194,90 @@ def fake_arm(**kwargs): ) return receipt, output + def test_profile_configured_choice_ack_is_observed_and_missing_or_late_fails(self): + import shlex + + def command_event(kind, event_id, argv, **extra): + item = { + "id": event_id, + "type": "command_execution", + "command": shlex.join(argv), + **extra, + } + return json.dumps({"type": kind, "item": item}) + + def assistant_event(event_id, text): + return json.dumps({ + "type": "item.completed", + "item": {"id": event_id, "type": "agent_message", "text": text}, + }) + + triage_output = json.dumps({ + "observed_exit_status": 1, + "test_failed": True, + "status": "no-remote-choice", + "steps": [{"id": self.profile.triage_accepted_ids[0]}], + "executed": False, + "decision_usage": None, + "hypothesis_ranking_status": "complete", + "hypothesis_order": list(self.profile.triage_hypotheses), + }) + evidence_argv = ["cat", *self.profile.evidence_files.values()] + evidence_output = " ".join( + marker + for markers in self.profile.evidence_markers.values() + for marker in markers + ) + + def parse(ack_position): + lines = [ + assistant_event("workflow-ack", runner.WORKFLOW_ACK_LINE), + command_event("item.started", "focused", self.profile.focused_command), + command_event( + "item.completed", "focused", self.profile.focused_command, + exit_code=1, aggregated_output="AssertionError: expected value mismatch", + ), + command_event("item.started", "evidence", evidence_argv), + command_event( + "item.completed", "evidence", evidence_argv, + exit_code=0, aggregated_output=evidence_output, + ), + ] + triage_argv = runner._triage_argv(1, self.profile) + lines.extend([ + command_event("item.started", "triage", triage_argv), + command_event( + "item.completed", "triage", triage_argv, + exit_code=0, aggregated_output=triage_output, + ), + ]) + if ack_position == "before_next_tool": + lines.append(assistant_event( + "triage-ack", runner.PROFILE_TRIAGE_RESULT_ACK_LINE, + )) + full_argv = shlex.split(runner.REQUIRED_COMMAND) + lines.append(command_event("item.started", "full", full_argv)) + if ack_position == "after_next_tool": + lines.append(assistant_event( + "triage-ack", runner.PROFILE_TRIAGE_RESULT_ACK_LINE, + )) + return runner._event_receipts( + lines, [float(index + 1) for index in range(len(lines))], 0.0, + advice_policy="nonbinding", profile=self.profile, + )[0] + + accepted = parse("before_next_tool") + self.assertEqual(accepted["triage_output_status"], "valid_configured_choice") + self.assertEqual(accepted["triage_result_acknowledgment"], "before_next_tool") + + missing = parse("missing") + self.assertEqual(missing["triage_output_status"], "valid_configured_choice") + self.assertEqual(missing["triage_result_acknowledgment"], "not_applicable") + + late = parse("after_next_tool") + self.assertEqual(late["triage_output_status"], "valid_configured_choice") + self.assertEqual(late["triage_result_acknowledgment"], "after_next_tool") + def test_valid_profile_parses_allowlisted_paths_and_enums(self): self.assertEqual(self.profile.case_id, "synthetic-repair") self.assertEqual(self.profile.outcome_mode, "repair")