From 14c5bc37b25b27c96ac8946e68307a07d728a64d Mon Sep 17 00:00:00 2001 From: Toni Nowak Date: Wed, 30 Sep 2026 23:26:11 +0200 Subject: [PATCH 1/2] feat: wire diagnostic costs into profile triage pairs --- PILOT.md | 7 + TASKS.md | 7 + scripts/pilot_contract_triage_pair.py | 45 ++- scripts/pilot_profile_triage_bridge.py | 56 ++- ...test_pilot_mandatory_command_visibility.py | 2 + tests/test_pilot_profile_diagnostic_costs.py | 348 ++++++++++++++++++ tests/test_pilot_profile_rank_dimensions.py | 1 + tests/test_pilot_profile_triage_bridge.py | 3 +- 8 files changed, 455 insertions(+), 14 deletions(-) create mode 100644 tests/test_pilot_profile_diagnostic_costs.py diff --git a/PILOT.md b/PILOT.md index 5f80b9d..9028805 100644 --- a/PILOT.md +++ b/PILOT.md @@ -1849,3 +1849,10 @@ Six API tests, including changed-cost cache invalidation, three new CLI integrat ## VCR407 — Coding-skill guidance for diagnostic costs The bundled English coding workflow explains optional caller-verified relative check costs, omission of unverified estimates, direct execution of decisive local checks and separation from causal ranking. It preserves required test execution and local fallback. This documents the merged API/CLI without claiming measured benefit or changing the published release. Nine skill-pack tests and three coding-workflow tests pass. Hosted CI remains required before merge. + + +## VCR408 — Diagnostic costs wired into profile triage trials + +Future paired trials may declare caller-verified relative diagnostic costs in a case profile for supplied hypothesis IDs only. Both arms receive the same cost disclosure; the treatment arm is told costs may guide next-check ordering only, are never causal-likelihood evidence, and never waive required checks. Costs reach the remote decision only as fixed low/medium/high/unknown enum tokens through the validated bridge spec, the exact triage command, the intercepted shim request, and the production triage call, keeping transport bridge-counted. Cost-bearing requests are rejected by the bridge and shim unless they match the supervisor-validated profile. Profiles without costs keep identical prompts, commands, requests, and decision state; the pre-existing argv-order difference between the supervisor command and the bridge shim for rank-plus-observation profiles is unchanged. + +This is prospective trial instrumentation; no pair was run. Nine new wiring tests plus updated doubles pass, and the full 857-test suite passes locally with `git diff --check` clean. Hosted CI remains required before merge. No efficacy, delivery, or release claim is made. diff --git a/TASKS.md b/TASKS.md index 66574de..ec408a2 100644 --- a/TASKS.md +++ b/TASKS.md @@ -1253,3 +1253,10 @@ Six API tests, including changed-cost cache invalidation, three new CLI integrat ## VCR407 — Coding-skill guidance for diagnostic costs The bundled English coding workflow explains optional caller-verified relative check costs, omission of unverified estimates, direct execution of decisive local checks and separation from causal ranking. It preserves required test execution and local fallback. This documents the merged API/CLI without claiming measured benefit or changing the published release. Nine skill-pack tests and three coding-workflow tests pass. Hosted CI remains required before merge. + + +## VCR408 — Diagnostic costs wired into profile triage trials + +Case profiles may declare caller-verified relative diagnostic costs for supplied hypothesis IDs. Both arms receive the same cost disclosure and the treatment arm is told costs may guide next-check ordering only, are never causal-likelihood evidence, and never waive required checks. The validated plain-string costs flow through the bridge spec, the exact triage command, the intercepted shim request, and the supervisor bridge into the production triage call, so cost-bearing requests stay bridge-counted instead of silently falling back to the real CLI. Profiles without costs keep identical prompts, commands, requests, and decision state; a pre-existing argv-order difference between the supervisor command and the bridge shim for rank-plus-observation profiles is unchanged. + +Nine new wiring tests plus updated doubles cover loader accept/reject, equal-arm disclosure, supervisor argv construction, spec validation, bridge request matching and forwarding, and shim interception without fallback. The full 857-test suite passes locally (43.6 s) with `git diff --check` clean. Hosted CI remains required before merge. No trial was run and no speed or quality benefit is claimed. diff --git a/scripts/pilot_contract_triage_pair.py b/scripts/pilot_contract_triage_pair.py index c635979..6de9296 100644 --- a/scripts/pilot_contract_triage_pair.py +++ b/scripts/pilot_contract_triage_pair.py @@ -144,6 +144,7 @@ class CaseProfile: outcome_mode: str oracle_script: Path oracle_sha256: str + triage_diagnostic_costs: dict[Any, Any] | None = None def _reject_duplicate_json_keys(pairs: list[tuple[str, Any]]) -> dict[str, Any]: @@ -274,12 +275,12 @@ def load_case_profile(path: Path) -> CaseProfile: for value in failure_markers)): raise ValueError("invalid case-profile failure signature") triage = raw["triage"] - if not isinstance(triage, dict) or set(triage) != { - "kinds", "hypotheses", "accepted_ids", "accepted_statuses", "observations", - "rank_hypotheses", - }: + if not isinstance(triage, dict) or set(triage) not in ( + {"kinds", "hypotheses", "accepted_ids", "accepted_statuses", "observations", "rank_hypotheses"}, + {"kinds", "hypotheses", "accepted_ids", "accepted_statuses", "observations", "rank_hypotheses", "diagnostic_costs"}, + ): raise ValueError("invalid case-profile triage") - from jevcompass.triage import FailureKind, HypothesisId + from jevcompass.triage import DiagnosticCost, FailureKind, HypothesisId kinds, hypotheses, accepted = (triage[key] for key in ("kinds", "hypotheses", "accepted_ids")) statuses = triage["accepted_statuses"] observations = triage["observations"] @@ -319,6 +320,16 @@ def load_case_profile(path: Path) -> CaseProfile: rank_hypotheses = triage["rank_hypotheses"] if type(rank_hypotheses) is not bool: raise ValueError("invalid case-profile ranking opt-in") + raw_costs = triage.get("diagnostic_costs", {}) + if (not isinstance(raw_costs, dict) or len(raw_costs) > len(hypotheses) + or any(not isinstance(key, str) or key not in hypotheses for key in raw_costs)): + raise ValueError("invalid case-profile diagnostic costs") + try: + diagnostic_costs = { + HypothesisId(key): DiagnosticCost(value) for key, value in raw_costs.items() + } + except (TypeError, ValueError) as exc: + raise ValueError("invalid case-profile diagnostic costs") from exc allow_remote = False outcome = raw["outcome_mode"] if outcome not in {"contract_triage", "repair"}: @@ -345,7 +356,7 @@ def load_case_profile(path: Path) -> CaseProfile: case_id, fixture, prompt, source_file, test_file, focused_command, safe_evidence, safe_markers, tuple(failure_markers), tuple(kinds), tuple(hypotheses), tuple(accepted), tuple(statuses), safe_observations, - rank_hypotheses, outcome, oracle_script, expected_hash, + rank_hypotheses, outcome, oracle_script, expected_hash, diagnostic_costs, ) def _case_base_prompt(profile: CaseProfile | None) -> str: @@ -353,6 +364,17 @@ def _case_base_prompt(profile: CaseProfile | None) -> str: if profile is None: return BASE_PROMPT prompt = profile.task_prompt + costs = profile.triage_diagnostic_costs or {} + if costs: + declarations = ", ".join( + f"{hypothesis.value}={cost.value}" + for hypothesis, cost in costs.items() + ) + prompt += ( + "\n\nCaller-declared relative diagnostic check effort (shared with both arms): " + f"{declarations}. These declarations describe next-check effort only; " + "they are not evidence of causal likelihood." + ) if profile.outcome_mode != "repair": return prompt @@ -385,6 +407,10 @@ def _profile_bridge_spec( profile.triage_observations if include_observations else {} ), rank_hypotheses=profile.rank_hypotheses, + diagnostic_costs={ + hypothesis.value: cost.value + for hypothesis, cost in (profile.triage_diagnostic_costs or {}).items() + }, ) @@ -421,6 +447,11 @@ def _case_treatment_prompt( "Triage is optional: skip it when local evidence resolves the choice. " f"If you invoke triage, use this exact command: {triage_command}. " ) + if profile.triage_diagnostic_costs: + guidance += ( + "Caller-declared relative check effort may guide diagnostic next-step ordering only; " + "it is not evidence of causal likelihood and must not be used to skip required checks. " + ) if profile.outcome_mode == "repair": guidance += ( f"Complete the requested behavior change in {profile.source_file}. " @@ -487,6 +518,8 @@ def _triage_argv( candidates.append("confirm_behavior_contract") for hypothesis in candidates: suffix.extend(("--hypothesis", hypothesis)) + for hypothesis, cost in (profile.triage_diagnostic_costs or {}).items(): + suffix.extend(("--diagnostic-cost", f"{hypothesis.value}={cost.value}")) if profile.rank_hypotheses: suffix.append("--rank-hypotheses") if include_observations: diff --git a/scripts/pilot_profile_triage_bridge.py b/scripts/pilot_profile_triage_bridge.py index 0d1b714..1315d21 100644 --- a/scripts/pilot_profile_triage_bridge.py +++ b/scripts/pilot_profile_triage_bridge.py @@ -9,7 +9,7 @@ """ from __future__ import annotations -from dataclasses import dataclass +from dataclasses import dataclass, field import json import math import os @@ -63,9 +63,10 @@ class ProfileTriageSpec: hypotheses: tuple[str, ...] allowed_observations: Mapping[str, tuple[str, ...]] rank_hypotheses: bool = False + diagnostic_costs: Mapping[str, str] = field(default_factory=dict) def __post_init__(self) -> None: - from jevcompass.triage import FailureKind, HypothesisId + from jevcompass.triage import DiagnosticCost, FailureKind, HypothesisId if (not self.failure_kinds or not self.hypotheses or any(not isinstance(value, str) for value in self.failure_kinds) @@ -80,6 +81,15 @@ def __post_init__(self) -> None: raise ValueError("invalid profile triage enum contract") if type(self.rank_hypotheses) is not bool: raise ValueError("rank_hypotheses must be boolean") + # Plain strings only: these values are embedded verbatim in the shim. + if not isinstance(self.diagnostic_costs, Mapping): + raise ValueError("invalid profile diagnostic costs") + allowed_costs = {item.value for item in DiagnosticCost} + for hypothesis, cost in self.diagnostic_costs.items(): + if (type(hypothesis) is not str or type(cost) is not str + or hypothesis not in self.hypotheses + or cost not in allowed_costs): + raise ValueError("invalid profile diagnostic costs") if not isinstance(self.allowed_observations, Mapping): raise ValueError("invalid profile observation allowlist") for group, values in self.allowed_observations.items(): @@ -322,7 +332,7 @@ def _read_request(self, handler: BaseHTTPRequestHandler) -> Any: def _validated_request(self, request: Any, observed_exit: int) -> dict[str, Any]: if not isinstance(request, dict) or set(request) != { "observed_exit_status", "failure_kinds", "hypotheses", "observations", - "rank_hypotheses", + "rank_hypotheses", "diagnostic_costs", }: raise ValueError("invalid enum-only request") if (type(request["observed_exit_status"]) is not int @@ -343,12 +353,21 @@ def _validated_request(self, request: Any, observed_exit: int) -> dict[str, Any] or len(set(values)) != len(values)): raise ValueError("invalid observation enum") observations[group] = tuple(values) - return {"observations": observations} + raw_costs = request["diagnostic_costs"] + if (not isinstance(raw_costs, dict) + or any(type(key) is not str or type(value) is not str + for key, value in raw_costs.items()) + or raw_costs != dict(self.spec.diagnostic_costs)): + raise ValueError("invalid diagnostic cost request") + return { + "observations": observations, + "diagnostic_costs": dict(raw_costs), + } def _decide(self, args: dict[str, Any], observed_exit: int): from jevcompass.decisions import DecisionsClient from jevcompass.triage import ( - CATALOG, AssertionObservation, FailureKind, HypothesisId, + CATALOG, AssertionObservation, DiagnosticCost, FailureKind, HypothesisId, ImportObservation, TriageDecisionReason, TimeoutObservation, TriageResult, triage_failure, ) @@ -367,13 +386,20 @@ def factory(): enum_groups[group][1]: tuple(enum_groups[group][0](value) for value in values) for group, values in args["observations"].items() } + diagnostic_costs = { + HypothesisId(key): DiagnosticCost(value) + for key, value in args["diagnostic_costs"].items() + } + triage_options = dict(triage_observations) + if diagnostic_costs: + triage_options["diagnostic_costs"] = diagnostic_costs result = triage_failure( tuple(FailureKind(value) for value in self.spec.failure_kinds), tuple(HypothesisId(value) for value in self.spec.hypotheses), observed_exit, client=client, rank_hypotheses=self.spec.rank_hypotheses, - **triage_observations, + **triage_options, ) if not isinstance(result, TriageResult) or result.observed_exit_status != observed_exit: raise ValueError("invalid production triage result") @@ -510,6 +536,8 @@ def command_for(spec: ProfileTriageSpec, exit_status: int, args.extend(("--kind", kind)) for hypothesis in spec.hypotheses: args.extend(("--hypothesis", hypothesis)) + for hypothesis, cost in spec.diagnostic_costs.items(): + args.extend(("--diagnostic-cost", f"{hypothesis}={cost}")) for group in ("import", "assertion", "timeout"): values = tuple(observed.get(group, ())) allowed = set(spec.allowed_observations.get(group, ())) @@ -534,6 +562,7 @@ def write_python_shim(directory: Path, bridge: ProfileTriageBridge, bindir.mkdir(mode=0o700, parents=True, exist_ok=True) shim = bindir / "python" groups = {key: list(value) for key, value in bridge.spec.allowed_observations.items()} + costs = dict(bridge.spec.diagnostic_costs) source = f'''#!/usr/bin/env python3 import json, os, sys from urllib.request import Request, urlopen @@ -543,6 +572,7 @@ def write_python_shim(directory: Path, bridge: ProfileTriageBridge, HYPOTHESES = {list(bridge.spec.hypotheses)!r} OBSERVATIONS = {groups!r} RANK = {bridge.spec.rank_hypotheses!r} +COSTS = {costs!r} FLAGS = {dict(_OBSERVATION_FLAGS)!r} ARGS = sys.argv[1:] def fallback(): @@ -564,6 +594,18 @@ def pairs(flag, values): if tail[:len(expected)] != expected: fallback() tail = tail[len(expected):] + supplied_costs = {{}} + while tail and tail[0] == "--diagnostic-cost": + if len(tail) < 2: + fallback() + name, separator, level = tail[1].partition("=") + if (not separator or name not in COSTS or level != COSTS[name] + or name in supplied_costs): + fallback() + supplied_costs[name] = level + tail = tail[2:] + if supplied_costs != COSTS: + fallback() observations = {{group: [] for group in OBSERVATIONS}} while tail and tail[0] != "--json" and tail[0] != "--rank-hypotheses": if len(tail) < 2: @@ -585,7 +627,7 @@ def pairs(flag, values): body = json.dumps({{ "observed_exit_status": code, "failure_kinds": KINDS, "hypotheses": HYPOTHESES, "observations": observations, - "rank_hypotheses": RANK, + "rank_hypotheses": RANK, "diagnostic_costs": supplied_costs, }}, separators=(",", ":")).encode() req = Request(URL, data=body, headers={{"Content-Type": "application/json"}}, method="POST") with urlopen(req, timeout=2.0) as response: diff --git a/tests/test_pilot_mandatory_command_visibility.py b/tests/test_pilot_mandatory_command_visibility.py index 0c659c7..54ebf30 100644 --- a/tests/test_pilot_mandatory_command_visibility.py +++ b/tests/test_pilot_mandatory_command_visibility.py @@ -37,6 +37,7 @@ def setUp(self): triage_accepted_ids=(), rank_hypotheses=False, triage_observations={}, + triage_diagnostic_costs={}, ) def test_exact_standalone_commands_are_shared_before_treatment_extras(self): @@ -64,6 +65,7 @@ def test_legacy_and_nonrepair_prompt_behavior_is_unchanged(self): contract_profile = SimpleNamespace( task_prompt="Inspect and preserve the original failure.", outcome_mode="contract_triage", + triage_diagnostic_costs={}, ) self.assertEqual(runner._case_base_prompt(contract_profile), contract_profile.task_prompt) diff --git a/tests/test_pilot_profile_diagnostic_costs.py b/tests/test_pilot_profile_diagnostic_costs.py new file mode 100644 index 0000000..a630b68 --- /dev/null +++ b/tests/test_pilot_profile_diagnostic_costs.py @@ -0,0 +1,348 @@ +"""Offline tests for diagnostic-cost wiring in profile triage trials.""" +from __future__ import annotations + +import hashlib +import importlib.util +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "src")) +sys.path.insert(0, str(ROOT / "scripts")) + +from jevcompass.decisions import DecisionsClient +from jevcompass.triage import DiagnosticCost, HypothesisId +from pilot_profile_triage_bridge import ( + ProfileTriageBridge, ProfileTriageSpec, command_for, write_python_shim, +) + +SCRIPT = ROOT / "scripts" / "pilot_contract_triage_pair.py" +MODULE_SPEC = importlib.util.spec_from_file_location( + "pilot_contract_triage_pair_diagnostic_cost_tests", SCRIPT, +) +runner = importlib.util.module_from_spec(MODULE_SPEC) +assert MODULE_SPEC and MODULE_SPEC.loader +sys.modules[MODULE_SPEC.name] = runner +MODULE_SPEC.loader.exec_module(runner) + +COSTS = { + "assertion_behavior_regression": "low", + "assertion_expectation_drift": "high", +} + + +def reply_transport(answers=None, usage=None, *, seen=None): + def transport(url, body, api_key, timeout): + if seen is not None: + seen.append(json.loads(body)) + return json.dumps({"answers": answers or {}, "usage": usage}).encode() + return transport + + +def post(url, payload): + data = json.dumps(payload, separators=(",", ":")).encode() + request = Request(url, data=data, headers={"Content-Type": "application/json"}, + method="POST") + try: + with urlopen(request, timeout=2) as response: + return response.status, response.read() + except HTTPError as error: + try: + return error.code, error.read() + finally: + error.close() + + +class ProfileDiagnosticCostTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory(prefix=".pilot-cost-test-", dir=ROOT) + self.addCleanup(self.tmp.cleanup) + self.root = Path(self.tmp.name) + self.fixture = self.root / "fixture" + (self.fixture / "src").mkdir(parents=True) + (self.fixture / "tests").mkdir() + (self.fixture / "src" / "service.py").write_text("VALUE = 0\n", encoding="utf-8") + (self.fixture / "tests" / "test_service.py").write_text( + "def test_value():\n assert True\n", encoding="utf-8" + ) + (self.fixture / "TASK.md").write_text( + "Repair service.py so the frozen expected value is returned.\n", encoding="utf-8" + ) + (self.fixture / "contract.md").write_text( + "The expected value is one.\n", encoding="utf-8" + ) + self.oracle = self.root / "supervisor_oracle.py" + self.oracle.write_text( + "import argparse\nfrom pathlib import Path\n" + "p = argparse.ArgumentParser()\n" + "p.add_argument('--fixture-dir', required=True)\n" + "root = Path(p.parse_args().fixture_dir)\n" + "raise SystemExit(0 if (root / 'src/service.py').read_text() == 'VALUE = 1\\n' else 9)\n", + encoding="utf-8", + ) + self.profile_data = { + "schema_version": 1, + "case_id": "synthetic-costs", + "fixture_source": self.fixture.relative_to(ROOT).as_posix(), + "task_prompt_file": "TASK.md", + "source_file": "src/service.py", + "focused_test_file": "tests/test_service.py", + "focused_command": ["python", "-m", "unittest", "tests.test_service", "-q"], + "evidence_files": {"contract": "contract.md"}, + "evidence_markers": {"contract": ["expected value", "one"]}, + "failure_markers": ["AssertionError"], + "triage": { + "kinds": ["assertion"], + "hypotheses": [ + "assertion_behavior_regression", + "assertion_expectation_drift", + ], + "accepted_ids": ["assertion_behavior_regression"], + "accepted_statuses": ["no-remote-choice"], + "observations": {"assertion": ["contract_underspecified"]}, + "rank_hypotheses": False, + "diagnostic_costs": dict(COSTS), + }, + "outcome_mode": "repair", + "oracle_script": self.oracle.relative_to(ROOT).as_posix(), + "oracle_sha256": hashlib.sha256(self.oracle.read_bytes()).hexdigest(), + } + self.profile_path = self.root / "profile.json" + self.write_profile() + + def write_profile(self, value=None): + self.profile_path.write_text( + json.dumps(self.profile_data if value is None else value), + encoding="utf-8", + ) + + def load(self): + return runner.load_case_profile(self.profile_path) + + def test_loader_parses_valid_costs_as_enums(self): + self.assertEqual(self.load().triage_diagnostic_costs, { + HypothesisId("assertion_behavior_regression"): DiagnosticCost("low"), + HypothesisId("assertion_expectation_drift"): DiagnosticCost("high"), + }) + + def test_loader_defaults_missing_costs_to_empty(self): + del self.profile_data["triage"]["diagnostic_costs"] + self.write_profile() + self.assertEqual(self.load().triage_diagnostic_costs, {}) + + def test_loader_rejects_invalid_costs(self): + invalid = { + "unknown-hypothesis": {"not_a_real_hypothesis": "low"}, + "unsupplied-hypothesis": {"confirm_behavior_contract": "low"}, + "bad-level": {"assertion_behavior_regression": "extreme"}, + "non-string-level": {"assertion_behavior_regression": 3}, + "non-object": ["assertion_behavior_regression=low"], + "too-many-entries": dict(COSTS, confirm_behavior_contract="low"), + } + for name, costs in invalid.items(): + with self.subTest(name): + self.profile_data["triage"]["diagnostic_costs"] = costs + self.write_profile() + with self.assertRaises(ValueError): + self.load() + + def test_supervisor_argv_carries_cost_flags_only_when_declared(self): + expected = [ + "python", "-m", "jevcompass", "triage", "--exit-code", "1", + "--kind", "assertion", + "--hypothesis", "assertion_behavior_regression", + "--hypothesis", "assertion_expectation_drift", + "--diagnostic-cost", "assertion_behavior_regression=low", + "--diagnostic-cost", "assertion_expectation_drift=high", + "--assertion-observation", "contract_underspecified", + "--json", + ] + self.assertEqual(runner._triage_argv(1, self.load()), expected) + del self.profile_data["triage"]["diagnostic_costs"] + self.write_profile() + self.assertNotIn("--diagnostic-cost", runner._triage_argv(1, self.load())) + + def test_prompts_share_cost_disclosure_and_limit_its_use(self): + base = runner._case_base_prompt(self.load()) + self.assertIn( + "Caller-declared relative diagnostic check effort (shared with both arms): " + "assertion_behavior_regression=low, assertion_expectation_drift=high.", + base, + ) + self.assertIn("not evidence of causal likelihood", base) + treatment = runner._case_treatment_prompt(self.load(), "legacy-required-step") + self.assertTrue(treatment.startswith(base)) + self.assertIn("may guide diagnostic next-step ordering only", treatment) + self.assertIn("must not be used to skip required checks", treatment) + + del self.profile_data["triage"]["diagnostic_costs"] + self.write_profile() + self.assertNotIn("diagnostic check effort", + runner._case_base_prompt(self.load())) + self.assertNotIn("diagnostic-cost", + runner._case_treatment_prompt(self.load(), "legacy-required-step")) + + def test_bridge_spec_and_command_match_supervisor_argv(self): + profile = self.load() + spec = runner._profile_bridge_spec(profile) + self.assertEqual(spec.diagnostic_costs, COSTS) + self.assertTrue(all(type(key) is str and type(value) is str + for key, value in spec.diagnostic_costs.items())) + observed = {group: list(values) + for group, values in profile.triage_observations.items()} + command = command_for(spec, 1, observed) + self.assertEqual(command, tuple(runner._triage_argv(1, profile))) + self.assertIn(("--diagnostic-cost", "assertion_behavior_regression=low"), + tuple(zip(command, command[1:]))) + + def test_spec_rejects_mismatched_or_non_string_costs(self): + valid = { + "failure_kinds": ("assertion",), + "hypotheses": ("assertion_behavior_regression", + "assertion_expectation_drift"), + "allowed_observations": {}, + } + self.assertEqual(ProfileTriageSpec(**valid).diagnostic_costs, {}) + with self.assertRaises(ValueError): + ProfileTriageSpec(**valid, + diagnostic_costs={"assertion_behavior_regression": "extreme"}) + with self.assertRaises(ValueError): + ProfileTriageSpec(**valid, + diagnostic_costs={"confirm_behavior_contract": "low"}) + with self.assertRaises(ValueError): + ProfileTriageSpec(**valid, diagnostic_costs={ + HypothesisId("assertion_behavior_regression"): DiagnosticCost("low"), + }) + with self.assertRaises(ValueError): + ProfileTriageSpec(**valid, + diagnostic_costs=["assertion_behavior_regression=low"]) + + def test_bridge_requires_matching_costs_and_forwards_them(self): + spec = ProfileTriageSpec( + failure_kinds=("assertion",), + hypotheses=("assertion_behavior_regression", + "assertion_expectation_drift"), + allowed_observations={}, + diagnostic_costs=dict(COSTS), + ) + seen = [] + factory = lambda: DecisionsClient( + api_key="synthetic", + transport=reply_transport(seen=seen), + ) + base_request = { + "observed_exit_status": 1, + "failure_kinds": ["assertion"], + "hypotheses": ["assertion_behavior_regression", + "assertion_expectation_drift"], + "observations": {}, + "rank_hypotheses": False, + "diagnostic_costs": dict(COSTS), + } + with tempfile.TemporaryDirectory() as temporary: + private = Path(temporary) / "private" + private.mkdir(mode=0o700) + + def fresh_bridge(name): + bridge = ProfileTriageBridge( + spec, receipt_path=private / name, client_factory=factory, + ) + bridge.__enter__() + self.assertTrue(bridge.observe_focused_failure(1, failure_confirmed=True)) + return bridge + + matching = fresh_bridge("matching.json") + status, _ = post(matching.url, base_request) + self.assertEqual(status, 200) + matching.__exit__(None, None, None) + self.assertEqual(len(seen), 1) + self.assertEqual(seen[0]["state"]["diagnostic_costs"], COSTS) + + wrong = fresh_bridge("wrong.json") + status, _ = post(wrong.url, dict( + base_request, diagnostic_costs={ + "assertion_behavior_regression": "medium", + "assertion_expectation_drift": "high", + }, + )) + self.assertEqual(status, 400) + wrong.__exit__(None, None, None) + + legacy = fresh_bridge("legacy.json") + status, _ = post(legacy.url, { + key: value for key, value in base_request.items() + if key != "diagnostic_costs" + }) + self.assertEqual(status, 400) + legacy.__exit__(None, None, None) + self.assertEqual(len(seen), 1) + + def test_shim_intercepts_cost_command_without_fallback(self): + spec = ProfileTriageSpec( + failure_kinds=("assertion",), + hypotheses=("assertion_behavior_regression", + "assertion_expectation_drift"), + allowed_observations={}, + diagnostic_costs=dict(COSTS), + ) + seen = [] + factory = lambda: DecisionsClient( + api_key="SUPERVISOR_ONLY_SECRET", + transport=reply_transport(seen=seen), + ) + with tempfile.TemporaryDirectory() as temporary: + private = Path(temporary) / "private" + private.mkdir(mode=0o700) + bridge = ProfileTriageBridge( + spec, receipt_path=private / "decision.json", client_factory=factory, + ) + with bridge: + self.assertTrue(bridge.observe_focused_failure(1, failure_confirmed=True)) + bin_dir = write_python_shim(private, bridge) + child_env = { + "PATH": str(bin_dir) + os.pathsep + os.environ.get("PATH", ""), + "HOME": str(private), "PYTHONPATH": str(ROOT / "src"), + } + command = command_for(spec, 1) + child = subprocess.run( + [str(bin_dir / "python"), *command[1:]], + cwd=ROOT, env=child_env, stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, + timeout=4, check=False, + ) + self.assertEqual(child.returncode, 0, child.stderr) + response = json.loads(child.stdout) + self.assertEqual(response["status"], "no-remote-choice") + self.assertEqual(bridge.receipt()["request_count"], 1) + self.assertEqual(len(seen), 1) + self.assertEqual(seen[0]["state"]["diagnostic_costs"], COSTS) + + altered = list(command) + altered[altered.index("--diagnostic-cost", + altered.index("--diagnostic-cost") + 1) + 1] = "medium" + fallback_child = subprocess.run( + [str(bin_dir / "python"), *altered[1:]], + cwd=ROOT, env=child_env, stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, + timeout=4, check=False, + ) + # The altered command must fall back to the real interpreter + # instead of being forwarded to the one-shot bridge. + self.assertEqual(bridge.receipt()["request_count"], 1) + self.assertEqual(bridge.receipt()["request_count"], 1) + self.assertNotIn("SUPERVISOR_ONLY_SECRET", + child.stdout + child.stderr + fallback_child.stdout + + fallback_child.stderr) + receipt = json.loads((private / "decision.json").read_text()) + self.assertEqual(receipt["provider_transport_call_count"], 1) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_pilot_profile_rank_dimensions.py b/tests/test_pilot_profile_rank_dimensions.py index 9f3a60d..bba0eb1 100644 --- a/tests/test_pilot_profile_rank_dimensions.py +++ b/tests/test_pilot_profile_rank_dimensions.py @@ -50,6 +50,7 @@ def test_diagnostic_candidate_requested_without_changing_causal_profile(self): triage_hypotheses=("assertion_behavior_regression", "assertion_expectation_drift"), triage_accepted_ids=("assertion_behavior_regression", "confirm_behavior_contract"), rank_hypotheses=True, triage_observations={}, + triage_diagnostic_costs={}, ) argv = runner._triage_argv(1, profile) self.assertIn("confirm_behavior_contract", argv) diff --git a/tests/test_pilot_profile_triage_bridge.py b/tests/test_pilot_profile_triage_bridge.py index 72456a0..b86f07e 100644 --- a/tests/test_pilot_profile_triage_bridge.py +++ b/tests/test_pilot_profile_triage_bridge.py @@ -75,6 +75,7 @@ def request_for(**overrides): "hypotheses": ["timeout_contention", "timeout_nonterminating"], "observations": {"timeout": []}, "rank_hypotheses": True, + "diagnostic_costs": {}, } value.update(overrides) return value @@ -188,7 +189,7 @@ def test_local_abstention_is_preserved_without_provider_call(self): status, body = post(bridge.url, { "observed_exit_status": 1, "failure_kinds": ["timeout"], "hypotheses": ["timeout_nonterminating"], "observations": {}, - "rank_hypotheses": True, + "rank_hypotheses": True, "diagnostic_costs": {}, }) response, receipt = json.loads(body), json.loads(path.read_text()) self.assertEqual(status, 200) From c65eef6fd92a7dd1a18208f792e2e6f35ca3a919 Mon Sep 17 00:00:00 2001 From: Toni Nowak Date: Wed, 30 Sep 2026 23:35:21 +0200 Subject: [PATCH 2/2] fix: order triage observation flags before rank to match the bridge shim --- PILOT.md | 4 +- TASKS.md | 4 +- scripts/pilot_contract_triage_pair.py | 5 ++- tests/test_pilot_profile_diagnostic_costs.py | 42 ++++++++++++++++++++ 4 files changed, 49 insertions(+), 6 deletions(-) diff --git a/PILOT.md b/PILOT.md index 9028805..6b2f437 100644 --- a/PILOT.md +++ b/PILOT.md @@ -1853,6 +1853,6 @@ The bundled English coding workflow explains optional caller-verified relative c ## VCR408 — Diagnostic costs wired into profile triage trials -Future paired trials may declare caller-verified relative diagnostic costs in a case profile for supplied hypothesis IDs only. Both arms receive the same cost disclosure; the treatment arm is told costs may guide next-check ordering only, are never causal-likelihood evidence, and never waive required checks. Costs reach the remote decision only as fixed low/medium/high/unknown enum tokens through the validated bridge spec, the exact triage command, the intercepted shim request, and the production triage call, keeping transport bridge-counted. Cost-bearing requests are rejected by the bridge and shim unless they match the supervisor-validated profile. Profiles without costs keep identical prompts, commands, requests, and decision state; the pre-existing argv-order difference between the supervisor command and the bridge shim for rank-plus-observation profiles is unchanged. +Future paired trials may declare caller-verified relative diagnostic costs in a case profile for supplied hypothesis IDs only. Both arms receive the same cost disclosure; the treatment arm is told costs may guide next-check ordering only, are never causal-likelihood evidence, and never waive required checks. Costs reach the remote decision only as fixed low/medium/high/unknown enum tokens through the validated bridge spec, the exact triage command, the intercepted shim request, and the production triage call, keeping transport bridge-counted. Cost-bearing requests are rejected by the bridge and shim unless they match the supervisor-validated profile. Profiles without costs keep identical prompts, commands, requests, and decision state. The supervisor triage command now orders observation flags before `--rank-hypotheses` to match the bridge command and shim, closing a silent-fallback path for exact rank-plus-observation requests in future trials; no historical trial is rescored or rerun. -This is prospective trial instrumentation; no pair was run. Nine new wiring tests plus updated doubles pass, and the full 857-test suite passes locally with `git diff --check` clean. Hosted CI remains required before merge. No efficacy, delivery, or release claim is made. +This is prospective trial instrumentation; no pair was run. Ten new wiring tests plus updated doubles pass, including parity and interception for rank-plus-observation commands, and the full 858-test suite passes locally with `git diff --check` clean. Hosted CI remains required before merge. No efficacy, delivery, or release claim is made. diff --git a/TASKS.md b/TASKS.md index ec408a2..0e9cf73 100644 --- a/TASKS.md +++ b/TASKS.md @@ -1257,6 +1257,6 @@ The bundled English coding workflow explains optional caller-verified relative c ## VCR408 — Diagnostic costs wired into profile triage trials -Case profiles may declare caller-verified relative diagnostic costs for supplied hypothesis IDs. Both arms receive the same cost disclosure and the treatment arm is told costs may guide next-check ordering only, are never causal-likelihood evidence, and never waive required checks. The validated plain-string costs flow through the bridge spec, the exact triage command, the intercepted shim request, and the supervisor bridge into the production triage call, so cost-bearing requests stay bridge-counted instead of silently falling back to the real CLI. Profiles without costs keep identical prompts, commands, requests, and decision state; a pre-existing argv-order difference between the supervisor command and the bridge shim for rank-plus-observation profiles is unchanged. +Case profiles may declare caller-verified relative diagnostic costs for supplied hypothesis IDs. Both arms receive the same cost disclosure and the treatment arm is told costs may guide next-check ordering only, are never causal-likelihood evidence, and never waive required checks. The validated plain-string costs flow through the bridge spec, the exact triage command, the intercepted shim request, and the supervisor bridge into the production triage call, so cost-bearing requests stay bridge-counted instead of silently falling back to the real CLI. Profiles without costs keep identical prompts, commands, requests, and decision state. The supervisor command now places observation flags before `--rank-hypotheses`, matching the bridge command and generated shim, so exact rank-plus-observation triage requests are intercepted instead of silently falling back to the real interpreter. -Nine new wiring tests plus updated doubles cover loader accept/reject, equal-arm disclosure, supervisor argv construction, spec validation, bridge request matching and forwarding, and shim interception without fallback. The full 857-test suite passes locally (43.6 s) with `git diff --check` clean. Hosted CI remains required before merge. No trial was run and no speed or quality benefit is claimed. +Ten new wiring tests plus updated doubles cover loader accept/reject, equal-arm disclosure, supervisor argv construction, spec validation, bridge request matching and forwarding, shim interception without fallback, and rank-plus-observation command order. The full 858-test suite passes locally (43.1 s) with `git diff --check` clean. Hosted CI remains required before merge. No trial was run and no speed or quality benefit is claimed. diff --git a/scripts/pilot_contract_triage_pair.py b/scripts/pilot_contract_triage_pair.py index 6de9296..b3f7e01 100644 --- a/scripts/pilot_contract_triage_pair.py +++ b/scripts/pilot_contract_triage_pair.py @@ -520,8 +520,7 @@ def _triage_argv( suffix.extend(("--hypothesis", hypothesis)) for hypothesis, cost in (profile.triage_diagnostic_costs or {}).items(): suffix.extend(("--diagnostic-cost", f"{hypothesis.value}={cost.value}")) - if profile.rank_hypotheses: - suffix.append("--rank-hypotheses") + # Observation flags precede --rank-hypotheses to match command_for() and the shim. if include_observations: observation_flags = { "import": "--import-observation", @@ -531,6 +530,8 @@ def _triage_argv( for category, values in profile.triage_observations.items(): for value in values: suffix.extend((observation_flags[category], value)) + if profile.rank_hypotheses: + suffix.append("--rank-hypotheses") suffix.append("--json") return [*TRIAGE_PREFIX, str(exit_code), *suffix] diff --git a/tests/test_pilot_profile_diagnostic_costs.py b/tests/test_pilot_profile_diagnostic_costs.py index a630b68..f469a7a 100644 --- a/tests/test_pilot_profile_diagnostic_costs.py +++ b/tests/test_pilot_profile_diagnostic_costs.py @@ -343,6 +343,48 @@ def test_shim_intercepts_cost_command_without_fallback(self): receipt = json.loads((private / "decision.json").read_text()) self.assertEqual(receipt["provider_transport_call_count"], 1) + def test_rank_observation_command_order_is_intercepted_not_fallback(self): + self.profile_data["triage"]["rank_hypotheses"] = True + self.write_profile() + profile = self.load() + spec = runner._profile_bridge_spec(profile) + observed = {group: list(values) + for group, values in profile.triage_observations.items()} + command = command_for(spec, 1, observed) + self.assertEqual(command, tuple(runner._triage_argv(1, profile))) + argv = runner._triage_argv(1, profile) + self.assertLess(argv.index("--assertion-observation"), + argv.index("--rank-hypotheses")) + self.assertLess(argv.index("--rank-hypotheses"), argv.index("--json")) + + factory = lambda: DecisionsClient( + api_key="SUPERVISOR_ONLY_SECRET", + transport=reply_transport(), + ) + with tempfile.TemporaryDirectory() as temporary: + private = Path(temporary) / "private" + private.mkdir(mode=0o700) + bridge = ProfileTriageBridge( + spec, receipt_path=private / "decision.json", client_factory=factory, + ) + with bridge: + self.assertTrue(bridge.observe_focused_failure(1, failure_confirmed=True)) + bin_dir = write_python_shim(private, bridge) + child_env = { + "PATH": str(bin_dir) + os.pathsep + os.environ.get("PATH", ""), + "HOME": str(private), "PYTHONPATH": str(ROOT / "src"), + } + child = subprocess.run( + [str(bin_dir / "python"), *command[1:]], + cwd=ROOT, env=child_env, stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, + timeout=4, check=False, + ) + self.assertEqual(child.returncode, 0, child.stderr) + self.assertEqual(bridge.receipt()["request_count"], 1) + self.assertNotIn("SUPERVISOR_ONLY_SECRET", + child.stdout + child.stderr) + if __name__ == "__main__": unittest.main()