From 7a9733bf4d8ecc4d0054793f3b506869e7de2c34 Mon Sep 17 00:00:00 2001 From: Toni Nowak Date: Sun, 27 Sep 2026 16:01:35 +0200 Subject: [PATCH] feat: add caller-verified diagnostic check costs --- PILOT.md | 7 + README.md | 2 + TASKS.md | 7 + src/jevcompass/cli.py | 19 ++ src/jevcompass/triage.py | 53 +++++- tests/test_triage_diagnostic_cost_cli.py | 48 +++++ tests/test_triage_diagnostic_costs.py | 227 +++++++++++++++++++++++ 7 files changed, 360 insertions(+), 3 deletions(-) create mode 100644 tests/test_triage_diagnostic_cost_cli.py create mode 100644 tests/test_triage_diagnostic_costs.py diff --git a/PILOT.md b/PILOT.md index 26965fe..7d87761 100644 --- a/PILOT.md +++ b/PILOT.md @@ -1837,3 +1837,10 @@ Two new symmetry/compatibility tests, nine profile tests and fifteen triage test Simple Bash and POSIX-shell command wrappers are normalized lexically without executing their contents. Compound commands, substitutions, redirections, environment assignments, newline separators and malformed wrappers cannot receive individual test gate credit. Cross-layer aggregate diagnostics remain separate from successful mandatory-suite observations and support both string and argv-list wrappers. This is prospective instrumentation hardening, not a retrospective explanation or rescore of VCR402. Its original Bash wrapper form was already supported. Twenty-five cross-layer tests and the full 839-test suite pass (40.827 s). Graph refreshed (3439 nodes, 6944 edges, 239 communities). Hosted CI remains required before merge. No new coding efficacy or native-host delivery result is claimed. + + +## VCR405 — Caller-verified diagnostic cost metadata + +The triage API and CLI accept optional fixed-enum relative diagnostic costs for supplied candidate IDs. Only locally verified low/medium/high/unknown tokens may reach Jev. Costs guide next-check ordering and are explicitly not causal likelihood evidence. Default requests, local resolution, original failing exits, non-execution and mandatory validation remain unchanged. CLI rejects unknown values, duplicate costs and costs for unsupplied candidates before backend use. + +Six API tests, including changed-cost cache invalidation, three new CLI integration tests, ten existing triage CLI tests and the full 848-test suite pass (40.532 s). Graph refreshed (3465 nodes, 7016 edges, 235 communities). Hosted CI remains required before merge. README documents source-only scope; the published version is unchanged. No real-trial speed or quality benefit is claimed. diff --git a/README.md b/README.md index d9eca95..a23d004 100644 --- a/README.md +++ b/README.md @@ -99,6 +99,8 @@ For assertion failures, the source candidate also accepts verified `--assertion- For timeouts, the source CLI also accepts repeatable `--timeout-observation` enum facts verified locally. An unsatisfiable wait condition selects the existing wait-condition check locally; observed contention **together with** progress and a satisfiable wait selects the resource-contention check. Conflicting facts abstain. Partial facts can inform an already eligible ambiguous choice; they do not create an API call by themselves. These are next-check suggestions, not confirmed causes, and the original failing exit remains unchanged. Example: `jevcompass triage --exit-code 1 --kind timeout --hypothesis timeout_contention --hypothesis timeout_nonterminating --timeout-observation wait_condition_unsatisfiable --json`. +For unresolved diagnostics, optionally add caller-verified relative check costs, for example `--diagnostic-cost timeout_contention=high --diagnostic-cost timeout_nonterminating=low`. Allowed costs are `low`, `medium`, `high`, and `unknown`, keyed only to supplied hypotheses. Omit costs you cannot establish locally. Costs inform which check to try next; they do not establish causal likelihood, override local resolutions, execute checks, or remove mandatory validation. This is a source-only feature with no measured speed benefit yet. + Triage JSON now includes a fixed `decision_reason` and each step's `selection_source`: a remotely preferred next action, locally resolved guidance, or an unranked local fallback. `hypothesis_ranking_status` remains `not_established`: choosing the next diagnostic action does not establish which cause is most likely. These fields describe the source checkout; the published package remains unchanged. For an optional complete hypothesis order, add `--rank-hypotheses` when 2–4 locally plausible causes remain. Jev receives up to six pairwise `choice` questions in the same request as the separate next-step question. The CLI exposes `hypothesis_order` only when every pair has an allowed choice at the confidence threshold and the comparisons form an acyclic complete order. Otherwise the ranking status is `incomplete` or `not_established`, with no order inferred from caller order; the diagnostic step is validated independently. Local resolutions and unambiguous cases still skip remote calls. Opt-in ranking bypasses the choice-only typed cache. This feature is source-only and does not claim improved diagnosis quality. diff --git a/TASKS.md b/TASKS.md index cefb3bf..8f7403c 100644 --- a/TASKS.md +++ b/TASKS.md @@ -1241,3 +1241,10 @@ Two new symmetry/compatibility tests, nine profile tests and fifteen triage test Simple Bash and POSIX-shell command wrappers are normalized lexically without executing their contents. Compound commands, substitutions, redirections, environment assignments, newline separators and malformed wrappers cannot receive individual test gate credit. Cross-layer aggregate diagnostics remain separate from successful mandatory-suite observations and support both string and argv-list wrappers. This is prospective instrumentation hardening, not a retrospective explanation or rescore of VCR402. Its original Bash wrapper form was already supported. Twenty-five cross-layer tests and the full 839-test suite pass (40.827 s). Graph refreshed (3439 nodes, 6944 edges, 239 communities). Hosted CI remains required before merge. No new coding efficacy or native-host delivery result is claimed. + + +## VCR405 — Caller-verified diagnostic cost metadata + +The triage API and CLI accept optional fixed-enum relative diagnostic costs for supplied candidate IDs. Only locally verified low/medium/high/unknown tokens may reach Jev. Costs guide next-check ordering and are explicitly not causal likelihood evidence. Default requests, local resolution, original failing exits, non-execution and mandatory validation remain unchanged. CLI rejects unknown values, duplicate costs and costs for unsupplied candidates before backend use. + +Six API tests, including changed-cost cache invalidation, three new CLI integration tests, ten existing triage CLI tests and the full 848-test suite pass (40.532 s). Graph refreshed (3465 nodes, 7016 edges, 235 communities). Hosted CI remains required before merge. README documents source-only scope; the published version is unchanged. No real-trial speed or quality benefit is claimed. diff --git a/src/jevcompass/cli.py b/src/jevcompass/cli.py index 46b3c54..5f46d28 100644 --- a/src/jevcompass/cli.py +++ b/src/jevcompass/cli.py @@ -353,6 +353,10 @@ def main(argv: list[str] | None = None) -> int: triage_parser.add_argument("--timeout-observation", action="append", choices=tuple(item.value for item in TimeoutObservation), help="Allowlisted local timeout observation; may be repeated") + triage_parser.add_argument("--diagnostic-cost", action="append", + choices=tuple(f"{item.value}={cost}" for item in HypothesisId + for cost in ("low", "medium", "high", "unknown")), + help="Caller-verified relative diagnostic check cost, HYPOTHESIS=COST; not causal likelihood") triage_parser.add_argument("--rank-hypotheses", action="store_true", help="Opt in to bounded pairwise hypothesis ordering when 2-4 causes remain plausible") triage_parser.add_argument("--json", action="store_true", help="Print machine-readable result") @@ -425,6 +429,20 @@ def main(argv: list[str] | None = None) -> int: return _recommend(args.category, args.domain, args.role) if args.command == "triage": from .triage import TriageDecisionReason, triage_failure + triage_options = {} + if args.diagnostic_cost: + from .triage import DiagnosticCost + costs = {} + supplied = set(args.hypothesis) + for value in args.diagnostic_cost: + identifier, cost = value.split("=", 1) + if identifier not in supplied: + triage_parser.error("diagnostic cost requires a supplied hypothesis") + hypothesis = HypothesisId(identifier) + if hypothesis in costs: + triage_parser.error("duplicate diagnostic cost hypothesis") + costs[hypothesis] = DiagnosticCost(cost) + triage_options["diagnostic_costs"] = costs result = triage_failure( tuple(FailureKind(item) for item in args.kind), tuple(HypothesisId(item) for item in args.hypothesis), @@ -439,6 +457,7 @@ def main(argv: list[str] | None = None) -> int: TimeoutObservation(item) for item in (args.timeout_observation or ()) ), rank_hypotheses=args.rank_hypotheses, + **triage_options, ) try: reason_value = TriageDecisionReason(result.decision_reason).value diff --git a/src/jevcompass/triage.py b/src/jevcompass/triage.py index 085509a..1f77c25 100644 --- a/src/jevcompass/triage.py +++ b/src/jevcompass/triage.py @@ -87,6 +87,15 @@ class TimeoutObservation(str, Enum): WAIT_CONDITION_UNSATISFIABLE = "wait_condition_unsatisfiable" +class DiagnosticCost(str, Enum): + """Caller-verified relative effort for a diagnostic check.""" + + LOW = "low" + MEDIUM = "medium" + HIGH = "high" + UNKNOWN = "unknown" + + @dataclass(frozen=True) class DiagnosticStep: """A locally authored next diagnostic action.""" @@ -393,6 +402,7 @@ def triage_failure( assertion_observations: Sequence[AssertionObservation] | None = None, timeout_observations: Sequence[TimeoutObservation] | None = None, rank_hypotheses: bool = False, + diagnostic_costs: Mapping[HypothesisId, DiagnosticCost] | None = None, ) -> TriageResult: """Return up to two diagnostic steps from allowlisted failure metadata. @@ -403,7 +413,10 @@ def triage_failure( confirmation step; contradictory contract observations abstain. A legacy fixture conflict may be ranked remotely using only its enum token. Raw names, paths, output, prompts, commands, and free-form descriptions are never accepted. - The optional decision reason is a fixed safe enum and retains no exception text. + Optional diagnostic costs are caller-verified relative effort enums keyed by + supplied hypothesis IDs. They can guide the order of diagnostic checks, but + are never evidence of causal likelihood. The optional decision reason is a + fixed safe enum and retains no exception text. """ if isinstance(observed_exit_status, bool) or not isinstance(observed_exit_status, int): raise TypeError("observed-exit-status-must-be-int") @@ -417,6 +430,21 @@ def triage_failure( raise TypeError("hypotheses-must-be-enums") if not isinstance(rank_hypotheses, bool): raise TypeError("rank-hypotheses-must-be-bool") + if diagnostic_costs is None: + cost_facts: dict[HypothesisId, DiagnosticCost] = {} + else: + if not isinstance(diagnostic_costs, Mapping): + raise TypeError("diagnostic-costs-must-be-mapping") + provided_hypotheses = set(hypotheses) + cost_facts = {} + for hypothesis, cost in diagnostic_costs.items(): + if not isinstance(hypothesis, HypothesisId): + raise TypeError("diagnostic-cost-keys-must-be-hypothesis-enums") + if hypothesis not in provided_hypotheses: + raise ValueError("diagnostic-cost-hypothesis-not-provided") + if not isinstance(cost, DiagnosticCost): + raise TypeError("diagnostic-cost-values-must-be-enums") + cost_facts[hypothesis] = cost observations = _normalize_import_observations(import_observations) assertion_facts = _normalize_assertion_observations(assertion_observations) timeout_facts = _normalize_timeout_observations(timeout_observations) @@ -514,17 +542,36 @@ def triage_failure( "failure_kinds": [kind.value for kind in FailureKind if kind in allowed_kinds], "hypotheses": causal_ids, } + cost_tokens = { + entry.step.id.value: cost_facts[entry.step.id].value + for entry in plausible if entry.step.id in cost_facts + } + if cost_tokens: + state["diagnostic_costs"] = cost_tokens if import_evidence_applies: state["import_observations"] = [item.value for item in observations] if assertion_evidence_applies: state["assertion_observations"] = [item.value for item in assertion_facts] if timeout_evidence_applies: state["timeout_observations"] = [item.value for item in timeout_facts] + diagnostic_criteria = { + entry.step.id.value: entry.descriptor for entry in plausible + } + for hypothesis_id, cost in cost_tokens.items(): + diagnostic_criteria[hypothesis_id] += ( + f"; caller-verified relative diagnostic cost: {cost}" + ) + diagnostic_instructions = "Choose the most useful next diagnostic step." + if cost_tokens: + diagnostic_instructions = ( + "Choose the most useful next diagnostic step, considering relative " + "diagnostic cost when ordering checks. Cost is not evidence of causal likelihood." + ) questions = { "diagnostic": { "type": "choice", - "instructions": "Choose the most useful next diagnostic step.", - "criteria": {entry.step.id.value: entry.descriptor for entry in plausible}, + "instructions": diagnostic_instructions, + "criteria": diagnostic_criteria, } } rank_questions: dict[str, dict[str, Any]] = {} diff --git a/tests/test_triage_diagnostic_cost_cli.py b/tests/test_triage_diagnostic_cost_cli.py new file mode 100644 index 0000000..2bb70da --- /dev/null +++ b/tests/test_triage_diagnostic_cost_cli.py @@ -0,0 +1,48 @@ +"""Allowlisted diagnostic costs reach triage without changing causal semantics.""" +import contextlib +import io +import json +import unittest +from unittest import mock + +from jevcompass.cli import main + + +class DiagnosticCostCliTests(unittest.TestCase): + def argv(self): + return ["triage", "--exit-code", "1", "--kind", "timeout", + "--hypothesis", "timeout_contention", "--hypothesis", "timeout_nonterminating", + "--json"] + + def test_costs_reach_safe_metadata_and_preserve_failed_test(self): + output = io.StringIO() + with mock.patch("jevcompass.triage.DecisionsClient") as client, contextlib.redirect_stdout(output): + client.return_value.decide.return_value = { + "diagnostic": {"type": "choice", "choice": "timeout_nonterminating", "confidence": 0.9}} + self.assertEqual(main(self.argv() + ["--diagnostic-cost", "timeout_contention=high", + "--diagnostic-cost", "timeout_nonterminating=low"]), 0) + state = client.return_value.decide.call_args.args[0] + self.assertEqual(state["diagnostic_costs"], {"timeout_contention": "high", "timeout_nonterminating": "low"}) + result = json.loads(output.getvalue()) + self.assertTrue(result["test_failed"]) + self.assertEqual(result["observed_exit_status"], 1) + self.assertEqual(result["hypothesis_ranking_status"], "not_established") + self.assertFalse(result["executed"]) + + def test_bad_costs_are_rejected_before_backend(self): + for options in ( + ["--diagnostic-cost", "timeout_contention=PRIVATE_TEXT"], + ["--diagnostic-cost", "import_path_changed=low"], + ["--diagnostic-cost", "timeout_contention=low", "--diagnostic-cost", "timeout_contention=high"], + ): + with self.subTest(options=options), mock.patch("jevcompass.triage.DecisionsClient") as client: + with contextlib.redirect_stderr(io.StringIO()), self.assertRaises(SystemExit) as raised: + main(self.argv() + options) + self.assertEqual(raised.exception.code, 2) + client.assert_not_called() + + def test_locally_resolved_wait_does_not_call_backend_with_costs(self): + with mock.patch("jevcompass.triage.DecisionsClient") as client, contextlib.redirect_stdout(io.StringIO()): + self.assertEqual(main(self.argv() + ["--timeout-observation", "wait_condition_unsatisfiable", + "--diagnostic-cost", "timeout_nonterminating=low"]), 0) + client.assert_not_called() diff --git a/tests/test_triage_diagnostic_costs.py b/tests/test_triage_diagnostic_costs.py new file mode 100644 index 0000000..7434d7b --- /dev/null +++ b/tests/test_triage_diagnostic_costs.py @@ -0,0 +1,227 @@ +from __future__ import annotations + +from types import MappingProxyType +import json +import os +import tempfile +import unittest +from unittest.mock import patch + +from jevcompass.decisions import DecisionsClient +from jevcompass.triage import ( + NO_REMOTE_CHOICE, + REMOTE_CHOICE, + DiagnosticCost, + FailureKind, + HypothesisId, + ImportObservation, + TriageDecisionReason, + triage_failure, +) + + +HYPOTHESES = ( + HypothesisId.IMPORT_MODULE_MISSING, + HypothesisId.IMPORT_PATH_CHANGED, +) + + +class FakeClient: + def __init__(self, answers): + self.answers = answers + self.calls = [] + + def decide(self, state, questions): + self.calls.append((state, questions)) + return self.answers + + +def valid_diagnostic(choice: HypothesisId): + return { + "diagnostic": { + "type": "choice", + "choice": choice.value, + "confidence": 0.9, + } + } + + +class DiagnosticCostTests(unittest.TestCase): + def test_omitted_costs_preserve_the_default_request_exactly(self): + default_client = FakeClient(valid_diagnostic(HYPOTHESES[0])) + none_client = FakeClient(valid_diagnostic(HYPOTHESES[0])) + empty_client = FakeClient(valid_diagnostic(HYPOTHESES[0])) + triage_failure((FailureKind.IMPORT,), HYPOTHESES, 1, default_client) + triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, none_client, + diagnostic_costs=None, + ) + triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, empty_client, + diagnostic_costs={}, + ) + default_request = default_client.calls[0] + self.assertEqual(default_request, none_client.calls[0]) + self.assertEqual(default_request, empty_client.calls[0]) + self.assertNotIn("diagnostic_costs", default_request[0]) + + def test_cost_tokens_are_safe_and_change_only_diagnostic_request_metadata(self): + costs = MappingProxyType({ + HYPOTHESES[0]: DiagnosticCost.LOW, + HYPOTHESES[1]: DiagnosticCost.HIGH, + }) + baseline_client = FakeClient(valid_diagnostic(HYPOTHESES[0])) + baseline = triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, baseline_client, + rank_hypotheses=True, + ) + client = FakeClient(valid_diagnostic(HYPOTHESES[0])) + result = triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, client, + rank_hypotheses=True, diagnostic_costs=costs, + ) + + state, questions = client.calls[0] + baseline_state, baseline_questions = baseline_client.calls[0] + self.assertEqual(result.status, REMOTE_CHOICE) + self.assertEqual(result.hypothesis_ranking_status, "incomplete") + self.assertEqual(result.hypothesis_order, ()) + self.assertEqual( + state["diagnostic_costs"], + { + HYPOTHESES[0].value: "low", + HYPOTHESES[1].value: "high", + }, + ) + self.assertIn("relative diagnostic cost: low", questions["diagnostic"]["criteria"][HYPOTHESES[0].value]) + self.assertIn("relative diagnostic cost: high", questions["diagnostic"]["criteria"][HYPOTHESES[1].value]) + self.assertIn("not evidence of causal likelihood", questions["diagnostic"]["instructions"]) + self.assertEqual( + questions["hypothesis_pair_0_1"], + baseline_questions["hypothesis_pair_0_1"], + ) + self.assertNotIn("diagnostic_costs", baseline_state) + self.assertNotIn("cost", questions["hypothesis_pair_0_1"]["instructions"].lower()) + self.assertNotIn("diagnostic_costs", questions["hypothesis_pair_0_1"]) + + reverse_client = FakeClient(valid_diagnostic(HYPOTHESES[0])) + triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, reverse_client, + diagnostic_costs={ + HYPOTHESES[0]: DiagnosticCost.HIGH, + HYPOTHESES[1]: DiagnosticCost.LOW, + }, + ) + reverse_state, _ = reverse_client.calls[0] + self.assertEqual(reverse_state["diagnostic_costs"], { + HYPOTHESES[0].value: "high", + HYPOTHESES[1].value: "low", + }) + + def test_changed_costs_miss_real_typed_cache_and_same_cost_hits(self): + calls = [] + + def transport(*_args): + calls.append(1) + return json.dumps({ + "answers": valid_diagnostic(HYPOTHESES[0]), + }).encode() + + decision_client = DecisionsClient(api_key="synthetic", transport=transport) + with tempfile.TemporaryDirectory() as cache_dir: + with patch.dict(os.environ, { + "JEVCOMPASS_TYPED_DECISION_CACHE": "1", + "JEVCOMPASS_TYPED_CACHE_DIR": cache_dir, + }), patch( + "jevcompass.triage.DecisionsClient", + return_value=decision_client, + ): + low_first = triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, + diagnostic_costs={ + HYPOTHESES[0]: DiagnosticCost.LOW, + HYPOTHESES[1]: DiagnosticCost.HIGH, + }, + ) + reversed_costs = triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, + diagnostic_costs={ + HYPOTHESES[0]: DiagnosticCost.HIGH, + HYPOTHESES[1]: DiagnosticCost.LOW, + }, + ) + repeated_reversed_costs = triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, + diagnostic_costs={ + HYPOTHESES[0]: DiagnosticCost.HIGH, + HYPOTHESES[1]: DiagnosticCost.LOW, + }, + ) + + self.assertEqual(low_first.status, REMOTE_CHOICE) + self.assertFalse(low_first.cache_hit) + self.assertEqual(reversed_costs.status, REMOTE_CHOICE) + self.assertFalse(reversed_costs.cache_hit) + self.assertTrue(repeated_reversed_costs.cache_hit) + self.assertEqual(len(calls), 2) + + def test_invalid_cost_metadata_is_rejected_before_network(self): + cases = ( + (["import_module_missing", "low"], TypeError), + ({"import_module_missing": DiagnosticCost.LOW}, TypeError), + ({HYPOTHESES[0]: "low"}, TypeError), + ({HypothesisId.COLLECTION_SYNTAX: DiagnosticCost.LOW}, ValueError), + ) + for costs, error in cases: + client = FakeClient(valid_diagnostic(HYPOTHESES[0])) + with self.subTest(costs=costs), self.assertRaises(error): + triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, client, + diagnostic_costs=costs, + ) + self.assertEqual(client.calls, []) + + def test_local_resolution_still_returns_before_remote_with_costs(self): + client = FakeClient(valid_diagnostic(HYPOTHESES[0])) + result = triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, client, + import_observations=( + ImportObservation.PACKAGE_PRESENT, + ImportObservation.TARGET_MODULE_ABSENT, + ImportObservation.REPLACEMENT_MODULE_PRESENT, + ), + diagnostic_costs={ + HYPOTHESES[0]: DiagnosticCost.LOW, + HYPOTHESES[1]: DiagnosticCost.HIGH, + }, + ) + self.assertEqual(result.status, NO_REMOTE_CHOICE) + self.assertEqual( + [step.id for step in result.steps], + [HypothesisId.IMPORT_PATH_CHANGED], + ) + self.assertEqual(client.calls, []) + + def test_costs_do_not_reorder_fallback_after_malformed_provider_response(self): + client = FakeClient({ + "diagnostic": { + "type": "choice", + "choice": "unknown-hypothesis", + "confidence": 0.99, + }, + }) + result = triage_failure( + (FailureKind.IMPORT,), HYPOTHESES, 1, client, + diagnostic_costs={ + HYPOTHESES[0]: DiagnosticCost.HIGH, + HYPOTHESES[1]: DiagnosticCost.LOW, + }, + ) + self.assertEqual(result.status, NO_REMOTE_CHOICE) + self.assertEqual(result.decision_reason, TriageDecisionReason.INVALID_RESPONSE) + self.assertEqual([step.id for step in result.steps], list(HYPOTHESES)) + self.assertEqual(len(client.calls), 1) + + +if __name__ == "__main__": + unittest.main()