diff --git a/PILOT.md b/PILOT.md index 990be21..26965fe 100644 --- a/PILOT.md +++ b/PILOT.md @@ -1830,3 +1830,10 @@ The retained private events show test invocations that differ from the frozen re Future repair-profile trials expose exact standalone validation commands to both baseline and treatment before treatment-specific advice instructions. The initial focused check precedes edits; post-edit focused, full-suite and diff checks remain mandatory. Optional initial-failure triage stays optional. Default and non-repair prompts remain unchanged. Historical trials are not rescored or rerun. Two new symmetry/compatibility tests, nine profile tests and fifteen triage tests pass. Graph refreshed (3439 nodes, 6944 edges, 239 communities). The concurrent shell-parser patch still has a separate diagnostic test failure and is excluded from this commit. Hosted CI on this isolated change is required before merge. No coding benefit or native-host delivery is claimed. + + +## VCR403 — Prospective shell-command observation hardening + +Simple Bash and POSIX-shell command wrappers are normalized lexically without executing their contents. Compound commands, substitutions, redirections, environment assignments, newline separators and malformed wrappers cannot receive individual test gate credit. Cross-layer aggregate diagnostics remain separate from successful mandatory-suite observations and support both string and argv-list wrappers. + +This is prospective instrumentation hardening, not a retrospective explanation or rescore of VCR402. Its original Bash wrapper form was already supported. Twenty-five cross-layer tests and the full 839-test suite pass (40.827 s). Graph refreshed (3439 nodes, 6944 edges, 239 communities). Hosted CI remains required before merge. No new coding efficacy or native-host delivery result is claimed. diff --git a/TASKS.md b/TASKS.md index 0cd9d9a..cefb3bf 100644 --- a/TASKS.md +++ b/TASKS.md @@ -1234,3 +1234,10 @@ The retained private events show test invocations that differ from the frozen re Future repair-profile trials expose exact standalone validation commands to both baseline and treatment before treatment-specific advice instructions. The initial focused check precedes edits; post-edit focused, full-suite and diff checks remain mandatory. Optional initial-failure triage stays optional. Default and non-repair prompts remain unchanged. Historical trials are not rescored or rerun. Two new symmetry/compatibility tests, nine profile tests and fifteen triage tests pass. Graph refreshed (3439 nodes, 6944 edges, 239 communities). The concurrent shell-parser patch still has a separate diagnostic test failure and is excluded from this commit. Hosted CI on this isolated change is required before merge. No coding benefit or native-host delivery is claimed. + + +## VCR403 — Prospective shell-command observation hardening + +Simple Bash and POSIX-shell command wrappers are normalized lexically without executing their contents. Compound commands, substitutions, redirections, environment assignments, newline separators and malformed wrappers cannot receive individual test gate credit. Cross-layer aggregate diagnostics remain separate from successful mandatory-suite observations and support both string and argv-list wrappers. + +This is prospective instrumentation hardening, not a retrospective explanation or rescore of VCR402. Its original Bash wrapper form was already supported. Twenty-five cross-layer tests and the full 839-test suite pass (40.827 s). Graph refreshed (3439 nodes, 6944 edges, 239 communities). Hosted CI remains required before merge. No new coding efficacy or native-host delivery result is claimed. diff --git a/scripts/pilot_cross_layer_test_order_pair.py b/scripts/pilot_cross_layer_test_order_pair.py index 8ed79bb..391a8a9 100644 --- a/scripts/pilot_cross_layer_test_order_pair.py +++ b/scripts/pilot_cross_layer_test_order_pair.py @@ -260,6 +260,14 @@ def _required_component_status(item: dict[str, Any], invocation_kind: str) -> st return "unparseable" raw = item.get("command") if isinstance(raw, list) and all(isinstance(part, str) for part in raw): + shell_names = {"bash", "/bin/bash", "/usr/bin/bash", + "sh", "/bin/sh", "/usr/bin/sh"} + if (len(raw) == 3 and raw[0] in shell_names + and raw[1] in {"-c", "-lc"}): + wrapped_command = " ".join(shlex.quote(part) for part in raw) + return _required_component_status( + {"command": wrapped_command}, invocation_kind + ) if any(part in {"&&", "||", ";", "|", "&"} for part in raw): # An argv list is not sufficient evidence that these tokens are shell syntax. return "unparseable" @@ -339,7 +347,7 @@ def _required_component_status(item: dict[str, Any], invocation_kind: str) -> st def _test_invocation_kind(item: dict[str, Any]) -> str | None: - """Classify test-runner invocation metadata without retaining raw command text.""" + """Classify test-runner metadata without retaining raw command text.""" raw = item.get("command") if isinstance(raw, str): raw_text = raw @@ -352,7 +360,37 @@ def _test_invocation_kind(item: dict[str, Any]) -> str | None: argv = engine._command_argv(item) if argv is None: - return "unknown" if re.search(r"(?i)\b(?:pytest|unittest)\b", raw_text) else None + # This path classifies diagnostic metadata only; it does not map a + # compound process exit to any individual shell command. + try: + outer_argv = ( + shlex.split(raw_text) if isinstance(raw, str) + else raw if isinstance(raw, list) else [] + ) + except ValueError: + outer_argv = [] + shell_names = {"bash", "/bin/bash", "/usr/bin/bash", + "sh", "/bin/sh", "/usr/bin/sh"} + if (len(outer_argv) == 3 and outer_argv[0] in shell_names + and outer_argv[1] in {"-c", "-lc"}): + script = outer_argv[2] + elif isinstance(raw, str): + script = raw_text + else: + script = "" + if any(char in script for char in {"$", chr(96), "<", ">", "(", ")"}): + return "unknown" if re.search( + r"(?i)\b(?:pytest|unittest)\b", raw_text + ) else None + try: + lexer = shlex.shlex(script, posix=True, punctuation_chars=";&|") + lexer.whitespace_split = True + lexer.commenters = "" + argv = list(lexer) + except ValueError: + return "unknown" if re.search( + r"(?i)\b(?:pytest|unittest)\b", raw_text + ) else None if sum(len(token) for token in argv) > 8192: return "unknown" if re.search(r"(?i)\b(?:pytest|unittest)\b", raw_text) else None diff --git a/scripts/pilot_test_order_pair.py b/scripts/pilot_test_order_pair.py index 3cc3174..90f674b 100644 --- a/scripts/pilot_test_order_pair.py +++ b/scripts/pilot_test_order_pair.py @@ -85,18 +85,42 @@ def _command_argv(item: dict[str, Any]) -> list[str] | None: if isinstance(raw, list) and all(isinstance(part, str) for part in raw): argv = raw elif isinstance(raw, str): + # CLI JSON may represent embedded shell newlines as backslash-n/r. + if any(marker in raw for marker in ("\n", "\r", r"\n", r"\r")): + return None try: argv = shlex.split(raw) except ValueError: return None else: return None - if (len(argv) == 3 and argv[0] in {"bash", "/bin/bash", "/usr/bin/bash"} - and argv[1] == "-lc"): + + shell_names = {"bash", "/bin/bash", "/usr/bin/bash", + "sh", "/bin/sh", "/usr/bin/sh"} + shell_family = {"bash", "sh", "zsh", "dash", "fish", "ksh"} + shell_name = Path(argv[0]).name if argv else "" + if shell_name in shell_family: + # Only unwrap the command-only forms observed in CLI command events. + # Parsing is lexical: never invoke a shell or evaluate its input. + if argv[0] not in shell_names or len(argv) != 3 or argv[1] not in {"-c", "-lc"}: + return None + script = argv[2] + if any(char in script for char in "$`\n\r"): + return None try: - argv = shlex.split(argv[2]) + lexer = shlex.shlex(script, posix=True, punctuation_chars="();<>|&") + lexer.whitespace_split = True + lexer.commenters = "" + script_argv = list(lexer) except ValueError: return None + if (not script_argv + or any(token and all(char in "();<>|&" for char in token) + for token in script_argv) + or any(re.match(r"^[A-Za-z_][A-Za-z0-9_]*=", token) + for token in script_argv)): + return None + return script_argv return argv diff --git a/tests/test_pilot_cross_layer_test_order_pair.py b/tests/test_pilot_cross_layer_test_order_pair.py index 93f635f..eb9c2fa 100644 --- a/tests/test_pilot_cross_layer_test_order_pair.py +++ b/tests/test_pilot_cross_layer_test_order_pair.py @@ -270,6 +270,11 @@ def test_required_suite_component_diagnostic_is_private_and_never_gate_credit(se + runner.REQUIRED_COMMAND + '"', exit_code=0, ), + event( + "item.completed", "required-in-argv-bash-wrapper", + ["/usr/bin/bash", "-lc", "pytest tests/test_private_case && " + runner.REQUIRED_COMMAND], + exit_code=0, + ), event( "item.completed", "combined-without-required", "python -m unittest tests.test_private_case && pytest tests/test_private_case", @@ -291,16 +296,16 @@ def test_required_suite_component_diagnostic_is_private_and_never_gate_credit(se rows, [start + index / 1000 for index in range(len(rows))], start ) - self.assertEqual(result["unmatched_test_invocation_count"], 6) + self.assertEqual(result["unmatched_test_invocation_count"], 7) self.assertEqual( result["unmatched_test_invocation_required_component_status_counts"], { - "exact_required_component_present": 3, + "exact_required_component_present": 4, "no_exact_required_component": 1, "unparseable": 2, }, ) - self.assertEqual(result["unmatched_test_invocation_kind_counts"]["combined"], 5) + self.assertEqual(result["unmatched_test_invocation_kind_counts"]["combined"], 6) self.assertFalse(result["required_suite_invocation_observed"]) self.assertIsNone(result["required_suite_exit"]) self.assertEqual(result["focused_invocation_count"], 0) diff --git a/tests/test_pilot_shell_wrapper_contract.py b/tests/test_pilot_shell_wrapper_contract.py new file mode 100644 index 0000000..f8319fb --- /dev/null +++ b/tests/test_pilot_shell_wrapper_contract.py @@ -0,0 +1,125 @@ +from __future__ import annotations + +import importlib.util +import json +from pathlib import Path +import sys +import unittest + +ROOT = Path(__file__).resolve().parents[1] +SCRIPT = ROOT / "scripts" / "pilot_test_order_pair.py" +sys.path.insert(0, str(ROOT / "scripts")) +SPEC = importlib.util.spec_from_file_location("pilot_test_order_pair_shell_contract", SCRIPT) +runner = importlib.util.module_from_spec(SPEC) +assert SPEC and SPEC.loader +SPEC.loader.exec_module(runner) + + +def event(event_type: str, identifier: str, command: str, **extra) -> str: + return json.dumps({ + "type": event_type, + "item": { + "id": identifier, + "type": "command_execution", + "command": command, + **extra, + }, + }) + + +class ShellWrapperContractTests(unittest.TestCase): + def test_observed_bash_and_sh_wrappers_match_focused_full_and_rank(self): + focused = "/usr/bin/bash -lc " + json.dumps(runner.UNIT_COMMAND) + full = "sh -c " + json.dumps(runner.REQUIRED_COMMAND) + rank_command = ( + "python -m jevcompass tests rank --input test-options.json --json" + ) + rank = "bash -lc " + json.dumps(rank_command) + + self.assertEqual(runner._test_kind({"command": focused}), "unit") + self.assertEqual(runner._test_kind({"command": full}), "required") + self.assertTrue(runner._rank_invocation({"command": rank})) + + def test_wrapped_evidence_is_correlated_to_the_command_event(self): + focused = "/usr/bin/bash -lc " + json.dumps(runner.CONTRACT_COMMAND) + full = "/usr/bin/bash -lc " + json.dumps(runner.REQUIRED_COMMAND) + started = 100.0 + lines = [ + event("item.started", "focused-1", focused), + event("item.completed", "focused-1", focused, exit_code=0), + event("item.started", "full-1", full), + event("item.completed", "full-1", full, exit_code=0), + ] + + receipt = runner._event_receipts( + lines, [started, started + 0.25, started + 0.5, started + 0.75], started + ) + + self.assertEqual( + receipt["focused_test_exits"], + [{"candidate_id": "contract", "exit_code": 0}], + ) + self.assertEqual(receipt["required_suite_exit"], 0) + self.assertTrue(receipt["required_suite_invocation_observed"]) + self.assertEqual(receipt["focused_invocation_count"], 1) + + def test_simple_quoted_arguments_are_preserved_without_execution(self): + script = "python -m unittest -p 'test_unit*.py' -v" + wrapped = "bash -c " + json.dumps(script) + + self.assertEqual( + runner._command_argv({"command": wrapped}), + ["python", "-m", "unittest", "-p", "test_unit*.py", "-v"], + ) + + def test_direct_argv_behavior_is_unchanged(self): + direct = ["python", "-m", "unittest", "discover", "-s", "tests", "-v"] + + self.assertEqual(runner._command_argv({"command": direct}), direct) + + def test_compound_or_unsafe_shell_syntax_is_rejected(self): + scripts = [ + runner.UNIT_COMMAND + " && echo done", + runner.UNIT_COMMAND + " | tee output.txt", + runner.UNIT_COMMAND + " > output.txt", + runner.UNIT_COMMAND + "; echo done", + runner.REQUIRED_COMMAND + "\n echo done", + runner.UNIT_COMMAND + "\n" + runner.CONTRACT_COMMAND, + "python -m unittest $(echo test)", + "python -m unittest $TEST_TARGET", + "python -m unittest `echo test`", + "PYTHONPATH=src " + runner.UNIT_COMMAND, + "python -m unittest -p 'unterminated", + ] + for script in scripts: + with self.subTest(script=script): + command = "bash -lc " + json.dumps(script) + self.assertIsNone(runner._command_argv({"command": command})) + self.assertIsNone(runner._test_kind({"command": command})) + + for command in ( + "bash -lc " + json.dumps(runner.UNIT_COMMAND) + " extra", + "bash --noprofile -lc " + json.dumps(runner.UNIT_COMMAND), + "zsh -c " + json.dumps(runner.UNIT_COMMAND), + ): + with self.subTest(command=command): + self.assertIsNone(runner._command_argv({"command": command})) + + def test_compound_process_exit_is_not_assigned_to_individual_tests(self): + command = "bash -lc " + json.dumps(runner.UNIT_COMMAND + " && echo done") + started = 100.0 + lines = [ + event("item.started", "compound", command), + event("item.completed", "compound", command, exit_code=1), + ] + + receipt = runner._event_receipts(lines, [started, started + 0.2], started) + + self.assertEqual(receipt["focused_test_exits"], []) + self.assertEqual(receipt["required_suite_exit"], None) + self.assertEqual(receipt["focused_invocation_count"], 0) + self.assertFalse(receipt["required_suite_invocation_observed"]) + + +if __name__ == "__main__": + unittest.main()