From b8f090ddb1404d39c4bb5a0eaaed800ebf7ada7c Mon Sep 17 00:00:00 2001 From: Toni Nowak Date: Sun, 27 Sep 2026 15:22:08 +0200 Subject: [PATCH] Add optional initial-failure triage trial stage --- PILOT.md | 6 + TASKS.md | 6 + scripts/pilot_contract_triage_pair.py | 152 ++++++++++--- tests/test_pilot_triage_stage.py | 316 ++++++++++++++++++++++++++ 4 files changed, 451 insertions(+), 29 deletions(-) create mode 100644 tests/test_pilot_triage_stage.py diff --git a/PILOT.md b/PILOT.md index 5922f77..765a756 100644 --- a/PILOT.md +++ b/PILOT.md @@ -1798,3 +1798,9 @@ The current contract resolves the apparent legacy conflict on ordinary inspectio The shared event reducer now ignores assistant item.started/item.updated draft lifecycle events when measuring message delivery. A completed assistant message remains eligible; a tool executed before completion still makes subsequent acknowledgment late. Two synthetic lifecycle tests and 63 existing collector tests pass. This prevents an unfinished message from poisoning or prematurely satisfying a delivery gate. The missing VCR396 acknowledgment cannot be attributed to this issue without its raw events. Historical outcomes remain unchanged. This correction concerns prospective measurement, not advisor efficacy or native Desktop delivery. + +## VCR400 — Optional triage before exhaustive evidence review + +The prospective runner adds an explicit initial-failure triage stage for nonbinding validated repair profiles. After a confirmed focused failure and one verified local discriminator, the agent may request enum-only diagnostic advice if competing hypotheses remain. Configured observation flags are omitted from the early command and bridge contract because they have not necessarily been verified. The default evidence-reviewed stage remains strict. + +Complete configured evidence review and every mandatory repair check remain required before acceptance. Early advice eligibility, final evidence completeness, diagnostic selection and causal ranking are measured separately; missing failure, unknown identifiers and late acknowledgment cannot pass delivery gates. Local resolution must skip remote advice. Eight targeted stage tests and the full local suite of 831 tests pass (37.595 seconds). Hosted CI remains the merge gate. Historical receipts are unchanged. diff --git a/TASKS.md b/TASKS.md index 8b10566..98d37ea 100644 --- a/TASKS.md +++ b/TASKS.md @@ -1202,3 +1202,9 @@ The current contract resolves the apparent legacy conflict on ordinary inspectio The shared event reducer now ignores assistant item.started/item.updated draft lifecycle events when measuring message delivery. A completed assistant message remains eligible; a tool executed before completion still makes subsequent acknowledgment late. Two synthetic lifecycle tests and 63 existing collector tests pass. This prevents an unfinished message from poisoning or prematurely satisfying a delivery gate. The missing VCR396 acknowledgment cannot be attributed to this issue without its raw events. Historical outcomes remain unchanged. This correction concerns prospective measurement, not advisor efficacy or native Desktop delivery. + +## VCR400 — Optional triage before exhaustive evidence review + +The prospective runner adds an explicit initial-failure triage stage for nonbinding validated repair profiles. After a confirmed focused failure and one verified local discriminator, the agent may request enum-only diagnostic advice if competing hypotheses remain. Configured observation flags are omitted from the early command and bridge contract because they have not necessarily been verified. The default evidence-reviewed stage remains strict. + +Complete configured evidence review and every mandatory repair check remain required before acceptance. Early advice eligibility, final evidence completeness, diagnostic selection and causal ranking are measured separately; missing failure, unknown identifiers and late acknowledgment cannot pass delivery gates. Local resolution must skip remote advice. Eight targeted stage tests and the full local suite of 831 tests pass (37.595 seconds). Hosted CI remains the merge gate. Historical receipts are unchanged. diff --git a/scripts/pilot_contract_triage_pair.py b/scripts/pilot_contract_triage_pair.py index fd9b991..7af8d39 100644 --- a/scripts/pilot_contract_triage_pair.py +++ b/scripts/pilot_contract_triage_pair.py @@ -61,6 +61,7 @@ ) TRIAGE_IDS = ("confirm_behavior_contract",) ADVICE_POLICIES = ("legacy-required-step", "nonbinding") +TRIAGE_STAGES = ("evidence-reviewed", "initial-failure") WORKFLOW_ACK_LINE = "JevCompass triage workflow instructions received." TRIAGE_RESULT_ACK_LINE = ( "JevCompass local triage result received: confirm_behavior_contract." @@ -348,7 +349,9 @@ def load_case_profile(path: Path) -> CaseProfile: ) -def _profile_bridge_spec(profile: CaseProfile) -> profile_bridge.ProfileTriageSpec: +def _profile_bridge_spec( + profile: CaseProfile, *, include_observations: bool = True, +) -> profile_bridge.ProfileTriageSpec: hypotheses = list(profile.triage_hypotheses) for identifier in profile.triage_accepted_ids: if identifier not in hypotheses: @@ -356,21 +359,45 @@ def _profile_bridge_spec(profile: CaseProfile) -> profile_bridge.ProfileTriageSp return profile_bridge.ProfileTriageSpec( failure_kinds=profile.triage_kinds, hypotheses=tuple(hypotheses), - allowed_observations=profile.triage_observations, + allowed_observations=( + profile.triage_observations if include_observations else {} + ), rank_hypotheses=profile.rank_hypotheses, ) -def _case_treatment_prompt(profile: CaseProfile, advice_policy: str) -> str: +def _case_treatment_prompt( + profile: CaseProfile, advice_policy: str, + triage_stage: str = "evidence-reviewed", +) -> str: + include_observations = triage_stage != "initial-failure" evidence_paths = ", ".join(profile.evidence_files.values()) - triage_command = shlex.join(_triage_argv(1, profile)) - guidance = ( - f"\n\nAfter the initial focused test, inspect these configured evidence files: {evidence_paths}. " - "Use only the locally observed exit code and allowlisted enums in any JevCompass triage command; " - "never pass file contents, paths, diagnostics, or source text to triage. " - "Triage is optional: skip it when local evidence resolves the choice. " - f"If you invoke triage, use this exact command: {triage_command}. " - ) + triage_command = shlex.join(_triage_argv( + 1, profile, include_observations=include_observations, + )) + if triage_stage == "initial-failure": + guidance = ( + "\n\nAfter the initial focused test fails, inspect its result and one " + "relevant configured evidence item as a cheap local discriminator. " + "Only if the failure is confirmed and at least two configured causal " + "hypotheses still fit, you may request the exact enum-only triage " + "command now, before reviewing the remaining evidence. The request " + "is optional; skip it when local evidence resolves the choice. Use " + "only the observed exit code and configured enums. Do not include " + "observation flags that have not been locally verified, or pass " + "file contents, paths, diagnostics, or source text. Continue to " + "inspect all configured evidence before completing the repair: " + f"{evidence_paths}. " + f"If you invoke triage, use this exact command: {triage_command}. " + ) + else: + guidance = ( + f"\n\nAfter the initial focused test, inspect these configured evidence files: {evidence_paths}. " + "Use only the locally observed exit code and allowlisted enums in any JevCompass triage command; " + "never pass file contents, paths, diagnostics, or source text to triage. " + "Triage is optional: skip it when local evidence resolves the choice. " + f"If you invoke triage, use this exact command: {triage_command}. " + ) if profile.outcome_mode == "repair": guidance += ( f"Complete the requested behavior change in {profile.source_file}. " @@ -419,7 +446,10 @@ def _full(item: dict[str, Any], profile: CaseProfile | None = None) -> bool: return common._command_argv(item) == shlex.split(REQUIRED_COMMAND) -def _triage_argv(exit_code: int, profile: CaseProfile | None = None) -> list[str]: +def _triage_argv( + exit_code: int, profile: CaseProfile | None = None, *, + include_observations: bool = True, +) -> list[str]: if isinstance(exit_code, bool) or not isinstance(exit_code, int) or exit_code == 0: raise ValueError("triage requires an observed nonzero focused exit") if profile is None: @@ -436,22 +466,28 @@ def _triage_argv(exit_code: int, profile: CaseProfile | None = None) -> list[str suffix.extend(("--hypothesis", hypothesis)) if profile.rank_hypotheses: suffix.append("--rank-hypotheses") - observation_flags = { - "import": "--import-observation", - "assertion": "--assertion-observation", - "timeout": "--timeout-observation", - } - for category, values in profile.triage_observations.items(): - for value in values: - suffix.extend((observation_flags[category], value)) + if include_observations: + observation_flags = { + "import": "--import-observation", + "assertion": "--assertion-observation", + "timeout": "--timeout-observation", + } + for category, values in profile.triage_observations.items(): + for value in values: + suffix.extend((observation_flags[category], value)) suffix.append("--json") return [*TRIAGE_PREFIX, str(exit_code), *suffix] -def _is_triage(item: dict[str, Any], exit_code: int | None, - profile: CaseProfile | None = None) -> bool: +def _is_triage( + item: dict[str, Any], exit_code: int | None, + profile: CaseProfile | None = None, *, + include_observations: bool = True, +) -> bool: return exit_code is not None and exit_code != 0 and ( - common._command_argv(item) == _triage_argv(exit_code, profile) + common._command_argv(item) == _triage_argv( + exit_code, profile, include_observations=include_observations, + ) ) @@ -582,6 +618,7 @@ def _event_receipts( profile: CaseProfile | None = None, accepted_triage_statuses: tuple[str, ...] | None = None, profile_bridge_receipt: dict[str, Any] | None = None, + triage_stage: str = "evidence-reviewed", ) -> tuple[dict[str, Any], str | None]: lines = list(lines) pending: dict[str, tuple[str, str | None]] = {} @@ -594,6 +631,9 @@ def _event_receipts( useful_failure_observed = False triage_seen = False triage_after_evidence = False + triage_after_local_discriminator = False + triage_before_evidence_complete = False + triage_stage_eligible_before_request = False triage_invalid = False triage_exit: int | None = None triage_choice = {"status": "unscored", "candidate_ids": [], @@ -683,14 +723,33 @@ def _event_receipts( pending[event_id] = ("git-diff-check", None) if argv and "jevcompass" in argv: triage_seen = True - exact = _is_triage(item, focused_exits[0] if focused_exits else None, profile) + exact = _is_triage( + item, focused_exits[0] if focused_exits else None, profile, + include_observations=triage_stage != "initial-failure", + ) required_evidence = frozenset(profile.evidence_files) if profile else _REQUIRED_EVIDENCE evidence_before_triage = required_evidence.issubset(evidence) + stage_eligible = ( + evidence_before_triage if triage_stage == "evidence-reviewed" + else bool(evidence) + ) accepted = ( exact and bool(focused_exits) and focused_exits[0] == 1 - and useful_failure_observed and evidence_before_triage + and useful_failure_observed and stage_eligible + ) + triage_stage_eligible_before_request = ( + triage_stage_eligible_before_request or accepted + ) + triage_after_evidence = triage_after_evidence or ( + accepted and evidence_before_triage + ) + triage_after_local_discriminator = ( + triage_after_local_discriminator or (accepted and bool(evidence)) + ) + triage_before_evidence_complete = ( + triage_before_evidence_complete + or (accepted and not evidence_before_triage) ) - triage_after_evidence = triage_after_evidence or accepted triage_invalid = triage_invalid or not accepted pending[event_id] = ("triage-accepted" if accepted else "triage-invalid", None) if not accepted: @@ -770,8 +829,12 @@ def _event_receipts( "policy_check_timing_status": "unscored", "evidence_categories_verified": sorted(evidence), "evidence_complete_before_triage": evidence_before_triage and triage_after_evidence, + "triage_stage": triage_stage, + "triage_stage_eligible_before_request": triage_stage_eligible_before_request, "triage_invocation_observed": triage_seen, "triage_after_evidence": triage_after_evidence, + "triage_after_local_discriminator": triage_after_local_discriminator, + "triage_before_evidence_complete": triage_before_evidence_complete, "triage_invalid_invocation_observed": triage_invalid, "triage_exit_code": triage_exit, "triage_output_status": triage_output_status, @@ -879,6 +942,7 @@ def _run_arm( allow_network: bool = False, max_tokens: int | None = None, retain_private_events: bool = False, + triage_stage: str = "evidence-reviewed", ) -> tuple[dict[str, Any], str | None]: (home / ".codex").mkdir(mode=0o700, parents=True, exist_ok=True) measurement_path = measurement_path or (home / ".codex" / "agent-measurement.json") @@ -1040,6 +1104,7 @@ def observe(event: dict[str, Any]) -> None: lines, times, started, advice_policy=advice_policy, profile=profile, accepted_triage_statuses=accepted_triage_statuses, profile_bridge_receipt=bridge_summary, + triage_stage=triage_stage, ) except Exception: failed = _empty_arm("event_parser_error") @@ -1230,6 +1295,7 @@ def run_pair( accepted_remote_statuses: tuple[str, ...] = (), max_tokens: int | None = None, retain_private_events: bool = False, + triage_stage: str = "evidence-reviewed", ) -> dict[str, Any]: """Run both local-only CLI arms and save private blind artifacts/receipts.""" if case_profile is not None: @@ -1265,6 +1331,20 @@ def run_pair( raise ValueError("max_tokens is outside the supported event-token budget") if type(retain_private_events) is not bool: raise ValueError("retain_private_events must be boolean") + if not isinstance(triage_stage, str) or triage_stage not in TRIAGE_STAGES: + raise ValueError("invalid triage stage") + if triage_stage == "initial-failure": + if (case_profile is None or case_profile.outcome_mode != "repair" + or advice_policy != "nonbinding"): + raise ValueError( + "initial-failure triage requires a nonbinding repair case profile" + ) + causal_hypotheses = { + item for item in case_profile.triage_hypotheses + if item != "confirm_behavior_contract" + } + if len(causal_hypotheses) < 2: + raise ValueError("initial-failure triage requires multiple causal hypotheses") if not _verify_codex_version(codex): raise ValueError("Codex CLI 0.157.0 is required") @@ -1338,7 +1418,12 @@ def run_pair( ) return receipt - bridge_spec = _profile_bridge_spec(case_profile) if remote_profile_triage else None + bridge_spec = ( + _profile_bridge_spec( + case_profile, include_observations=triage_stage != "initial-failure", + ) + if remote_profile_triage else None + ) arms: dict[str, dict[str, Any]] = {} answers: dict[str, str | None] = {} for true_arm in order: @@ -1348,7 +1433,7 @@ def run_pair( prompt = ( NONBINDING_TREATMENT_PROMPT if case_profile is None and advice_policy == "nonbinding" else TREATMENT_PROMPT if case_profile is None - else _case_treatment_prompt(case_profile, advice_policy) + else _case_treatment_prompt(case_profile, advice_policy, triage_stage) ) arms[label], answers[label] = _run_arm( codex=codex, model=model, reasoning_effort=reasoning_effort, @@ -1366,6 +1451,8 @@ def run_pair( allow_network=remote_profile_triage, max_tokens=max_tokens, **({"retain_private_events": True} if retain_private_events else {}), + **({"triage_stage": triage_stage} + if true_arm == "treatment" and triage_stage != "evidence-reviewed" else {}), ) if repair_profile: arms[label].setdefault("agent_git_diff_check_invocation_observed", False) @@ -1500,7 +1587,9 @@ def run_pair( required_evidence ) valid_triage = ( - treatment.get("triage_after_evidence") is True + (treatment.get("triage_after_evidence") is True + if triage_stage == "evidence-reviewed" + else treatment.get("triage_after_local_discriminator") is True) and treatment.get("triage_invalid_invocation_observed") is False and treatment.get("triage_exit_code") == 0 and treatment.get("triage_output_status") in { @@ -1573,6 +1662,7 @@ def run_pair( "accepted_remote_statuses": list(accepted_remote_statuses), "network_access_enabled_for_both_arms": remote_profile_triage, "observed_token_budget": max_tokens, + "triage_stage": triage_stage, "task_outcome_acceptance_status": ( "passed" if case_profile and case_profile.outcome_mode == "repair" and repair_task_correctness else "failed" if case_profile and case_profile.outcome_mode == "repair" @@ -1648,6 +1738,9 @@ def main(argv: list[str] | None = None) -> int: help="observed completed-turn token cap; not a provider-side limit") parser.add_argument("--retain-private-events", action="store_true", help="archive bounded raw CLI JSONL locally in each private arm directory") + parser.add_argument("--triage-stage", choices=TRIAGE_STAGES, + default="evidence-reviewed", + help="profile triage timing; initial-failure is opt-in and repair-only") parser.add_argument("--output-dir", type=Path, required=True, help="new private directory for blind receipts and artifacts") args = parser.parse_args(argv) @@ -1668,6 +1761,7 @@ def main(argv: list[str] | None = None) -> int: accepted_remote_statuses=tuple(args.accept_profile_triage_status), max_tokens=args.max_tokens, retain_private_events=args.retain_private_events, + triage_stage=args.triage_stage, ) except (OSError, ValueError, RuntimeError): print(json.dumps({"status": "failed", "failure": "runner_setup_failed"})) diff --git a/tests/test_pilot_triage_stage.py b/tests/test_pilot_triage_stage.py new file mode 100644 index 0000000..de81aec --- /dev/null +++ b/tests/test_pilot_triage_stage.py @@ -0,0 +1,316 @@ +from __future__ import annotations + +import hashlib +import json +from pathlib import Path +import shlex +import sys +import tempfile +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "scripts")) + +import pilot_contract_triage_pair as runner + + +def command_event(kind, event_id, argv, **extra): + return json.dumps({ + "type": kind, + "item": { + "id": event_id, + "type": "command_execution", + "command": shlex.join(argv), + **extra, + }, + }) + + +def assistant_event(text): + return json.dumps({ + "type": "item.completed", + "item": {"id": "assistant", "type": "agent_message", "text": text}, + }) + + +class InitialFailureTriageStageTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + root = Path(self.temp.name) + fixture = root / "fixture" + fixture.mkdir() + oracle = root / "oracle.py" + oracle.write_text("pass\n", encoding="utf-8") + self.profile = runner.CaseProfile( + case_id="stage-repair", + fixture_source=fixture, + task_prompt="Repair the behavior in src/service.py.", + source_file="src/service.py", + focused_test_file="tests/test_service.py", + focused_command=("python", "-m", "unittest", "tests.test_service", "-q"), + evidence_files={ + "contract": "contract.md", + "implementation": "src/service.py", + "test": "tests/test_service.py", + }, + evidence_markers={ + "contract": ("contract marker",), + "implementation": ("implementation marker",), + "test": ("test marker",), + }, + failure_markers=("AssertionError: mismatch",), + triage_kinds=("assertion",), + triage_hypotheses=( + "assertion_behavior_regression", + "assertion_expectation_drift", + ), + triage_accepted_ids=("confirm_behavior_contract",), + triage_accepted_statuses=("no-remote-choice",), + triage_observations={"assertion": ("contract_underspecified",)}, + rank_hypotheses=False, + outcome_mode="repair", + oracle_script=oracle, + oracle_sha256=hashlib.sha256(oracle.read_bytes()).hexdigest(), + ) + + def _triage_output(self): + return json.dumps({ + "observed_exit_status": 1, + "test_failed": True, + "status": "no-remote-choice", + "steps": [{"id": "confirm_behavior_contract"}], + "executed": False, + "decision_usage": None, + }) + + def _stream(self, *, stage="initial-failure", failure=True, + bad_command=False, bad_discriminator=False, unknown_output=False, + late_ack=False): + profile = self.profile + lines = [assistant_event(runner.WORKFLOW_ACK_LINE)] + focused = list(profile.focused_command) + if failure: + lines.extend([ + command_event("item.started", "focused", focused), + command_event( + "item.completed", "focused", focused, exit_code=1, + aggregated_output="AssertionError: mismatch", + ), + ]) + else: + lines.append(command_event( + "item.started", "unfailed-tool", ["python", "-c", "print('probe')"], + )) + contract = ["cat", "contract.md"] + lines.extend([ + command_event("item.started", "contract", contract), + command_event( + "item.completed", "contract", contract, exit_code=0, + aggregated_output=("unhelpful local output" if bad_discriminator else "contract marker"), + ), + ]) + argv = runner._triage_argv( + 1, profile, include_observations=stage != "initial-failure", + ) + if bad_command: + argv = [*argv, "--hypothesis", "unknown_hypothesis"] + lines.extend([ + command_event("item.started", "triage", argv), + command_event( + "item.completed", "triage", argv, exit_code=0, + aggregated_output=( + json.dumps({ + "observed_exit_status": 1, + "test_failed": True, + "status": "no-remote-choice", + "steps": [{"id": "unknown_step_id"}], + "executed": False, + }) if unknown_output else self._triage_output() + ), + ), + ]) + if not late_ack: + lines.append(assistant_event(runner.PROFILE_TRIAGE_RESULT_ACK_LINE)) + for kind in ("implementation", "test"): + argv = ["cat", profile.evidence_files[kind]] + lines.extend([ + command_event("item.started", kind, argv), + command_event( + "item.completed", kind, argv, exit_code=0, + aggregated_output=profile.evidence_markers[kind][0], + ), + ]) + if late_ack: + lines.append(assistant_event(runner.PROFILE_TRIAGE_RESULT_ACK_LINE)) + return lines + + def _score(self, **kwargs): + lines = self._stream(**kwargs) + return runner._event_receipts( + lines, [float(index + 1) for index in range(len(lines))], 0.0, + advice_policy="nonbinding", profile=self.profile, + triage_stage=kwargs.get("stage", "initial-failure"), + )[0] + + def test_initial_failure_stage_allows_advice_after_one_verified_discriminator(self): + result = self._score() + self.assertEqual(result["triage_output_status"], "valid_configured_choice") + self.assertTrue(result["triage_stage_eligible_before_request"]) + self.assertTrue(result["triage_after_local_discriminator"]) + self.assertTrue(result["triage_before_evidence_complete"]) + self.assertFalse(result["triage_after_evidence"]) + self.assertFalse(result["evidence_complete_before_triage"]) + self.assertEqual( + result["evidence_categories_verified"], + ["contract", "implementation", "test"], + ) + self.assertEqual(result["triage_result_acknowledgment"], "before_next_tool") + + def test_default_evidence_reviewed_stage_rejects_same_early_request(self): + result = self._score(stage="evidence-reviewed") + self.assertEqual(result["triage_output_status"], "out_of_order_or_unverified") + self.assertTrue(result["triage_invalid_invocation_observed"]) + self.assertFalse(result["triage_stage_eligible_before_request"]) + self.assertFalse(result["triage_after_evidence"]) + + def test_no_observed_failure_or_unknown_hypothesis_never_scores_valid(self): + no_failure = self._score(failure=False) + self.assertEqual(no_failure["initial_focused_exit"], None) + self.assertEqual(no_failure["triage_output_status"], "out_of_order_or_unverified") + self.assertTrue(no_failure["triage_invalid_invocation_observed"]) + self.assertFalse(no_failure["triage_stage_eligible_before_request"]) + + unknown = self._score(bad_command=True) + self.assertEqual(unknown["triage_output_status"], "out_of_order_or_unverified") + self.assertTrue(unknown["triage_invalid_invocation_observed"]) + self.assertFalse(unknown["triage_after_local_discriminator"]) + + def test_missing_local_discriminator_and_unknown_result_id_are_rejected(self): + no_discriminator = self._score(bad_discriminator=True) + self.assertEqual( + no_discriminator["triage_output_status"], + "out_of_order_or_unverified", + ) + self.assertFalse(no_discriminator["triage_stage_eligible_before_request"]) + self.assertTrue(no_discriminator["triage_invalid_invocation_observed"]) + + unknown_result = self._score(unknown_output=True) + self.assertEqual(unknown_result["triage_output_status"], "invalid_output") + self.assertTrue(unknown_result["triage_stage_eligible_before_request"]) + self.assertTrue(unknown_result["triage_after_local_discriminator"]) + + def test_late_result_ack_does_not_satisfy_delivery_ack_gate(self): + result = self._score(late_ack=True) + self.assertEqual(result["triage_output_status"], "valid_configured_choice") + self.assertEqual(result["triage_result_acknowledgment"], "after_next_tool") + self.assertNotEqual(result["triage_result_acknowledgment"], "before_next_tool") + + def test_prompt_and_bridge_omit_unverified_observations_but_keep_repair_gates(self): + prompt = runner._case_treatment_prompt( + self.profile, "nonbinding", "initial-failure", + ) + self.assertIn("one relevant configured evidence item as a cheap local discriminator", prompt) + self.assertIn("at least two configured causal hypotheses still fit", prompt) + self.assertIn("before reviewing the remaining evidence", prompt) + self.assertIn("inspect all configured evidence before completing the repair", prompt) + self.assertIn("git diff --check", prompt) + self.assertIn("focused test again", prompt) + self.assertIn("full test suite", prompt) + self.assertIn("separate immutable oracle", prompt) + initial_argv = runner._triage_argv( + 1, self.profile, include_observations=False, + ) + self.assertNotIn("--assertion-observation", initial_argv) + self.assertIn("--assertion-observation", runner._triage_argv(1, self.profile)) + spec = runner._profile_bridge_spec(self.profile, include_observations=False) + self.assertEqual(spec.allowed_observations, {}) + + def test_initial_failure_stage_is_restricted_to_nonbinding_repair_profiles(self): + with self.assertRaisesRegex(ValueError, "nonbinding repair case profile"): + runner.run_pair( + codex="unused", model="test-model", reasoning_effort="low", + output_dir=Path(self.temp.name) / "not-allowed", + case_profile=self.profile, advice_policy="legacy-required-step", + triage_stage="initial-failure", + ) + triage_profile = __import__("dataclasses").replace( + self.profile, outcome_mode="contract_triage", + ) + with self.assertRaisesRegex(ValueError, "nonbinding repair case profile"): + runner.run_pair( + codex="unused", model="test-model", reasoning_effort="low", + output_dir=Path(self.temp.name) / "not-repair", + case_profile=triage_profile, advice_policy="nonbinding", + triage_stage="initial-failure", + ) + + def test_run_pair_routes_stage_to_treatment_without_bypassing_repair_failure(self): + fixture = self.profile.fixture_source + (fixture / "src").mkdir() + (fixture / "tests").mkdir() + (fixture / "src/service.py").write_text("VALUE = 0\\n", encoding="utf-8") + (fixture / "tests/test_service.py").write_text("def test_service(): pass\\n", encoding="utf-8") + (fixture / "contract.md").write_text("contract marker\\n", encoding="utf-8") + calls = [] + + def fake_arm(**kwargs): + calls.append(kwargs) + return ({ + "cli_status": "completed", + "completion_ms": 10.0, + "initial_focused_exit": 1, + "focused_exit_codes": [1, 1], + "first_useful_failure_observed": True, + "full_suite_invocation_observed": True, + "full_suite_exit": 1, + "evidence_categories_verified": sorted(self.profile.evidence_files), + "triage_invocation_observed": kwargs["treatment"], + "triage_after_evidence": False, + "triage_after_local_discriminator": kwargs["treatment"], + "triage_stage_eligible_before_request": kwargs["treatment"], + "triage_invalid_invocation_observed": False, + "triage_exit_code": 0 if kwargs["treatment"] else None, + "triage_output_status": ( + "valid_configured_choice" if kwargs["treatment"] else "not_invoked" + ), + "workflow_acknowledgment": "before_first_tool", + "triage_result_acknowledgment": "before_next_tool", + "task_outcome_status": "incomplete", + "agent_git_diff_check_invocation_observed": True, + "agent_git_diff_check_exit_codes": [0], + "agent_git_diff_check_exit_code": 0, + "agent_git_diff_check_passed": True, + }, None) + + with ( + mock.patch.object(runner, "_verify_codex_version", return_value=True), + mock.patch.object(runner.core, "_copy_auth", return_value=True), + mock.patch.object(runner, "_run_arm", side_effect=fake_arm), + mock.patch.object(runner, "_run_independent_oracle", return_value={ + "status": "passed", "exit_code": 0, "elapsed_ms": 1.0, + "oracle_sha256": self.profile.oracle_sha256, "oracle_unchanged": True, + }), + ): + receipt = runner.run_pair( + codex="unused", model="test-model", reasoning_effort="low", + output_dir=Path(self.temp.name) / "pair-run", + case_profile=self.profile, + advice_policy="nonbinding", + triage_stage="initial-failure", + ) + + self.assertEqual(len(calls), 2, receipt) + treatment_call = next(call for call in calls if call["treatment"]) + baseline_call = next(call for call in calls if not call["treatment"]) + self.assertEqual(treatment_call["triage_stage"], "initial-failure") + self.assertNotIn("triage_stage", baseline_call) + self.assertEqual(receipt["triage_stage"], "initial-failure") + self.assertEqual(receipt["protocol_delivery_status"], "advice_delivered") + self.assertEqual(receipt["task_outcome_acceptance_status"], "failed") + self.assertEqual(receipt["status"], "incomplete") + + +if __name__ == "__main__": + unittest.main()