Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions PILOT.md
Original file line number Diff line number Diff line change
Expand Up @@ -1830,3 +1830,10 @@ The retained private events show test invocations that differ from the frozen re
Future repair-profile trials expose exact standalone validation commands to both baseline and treatment before treatment-specific advice instructions. The initial focused check precedes edits; post-edit focused, full-suite and diff checks remain mandatory. Optional initial-failure triage stays optional. Default and non-repair prompts remain unchanged. Historical trials are not rescored or rerun.

Two new symmetry/compatibility tests, nine profile tests and fifteen triage tests pass. Graph refreshed (3439 nodes, 6944 edges, 239 communities). The concurrent shell-parser patch still has a separate diagnostic test failure and is excluded from this commit. Hosted CI on this isolated change is required before merge. No coding benefit or native-host delivery is claimed.


## VCR403 — Prospective shell-command observation hardening

Simple Bash and POSIX-shell command wrappers are normalized lexically without executing their contents. Compound commands, substitutions, redirections, environment assignments, newline separators and malformed wrappers cannot receive individual test gate credit. Cross-layer aggregate diagnostics remain separate from successful mandatory-suite observations and support both string and argv-list wrappers.

This is prospective instrumentation hardening, not a retrospective explanation or rescore of VCR402. Its original Bash wrapper form was already supported. Twenty-five cross-layer tests and the full 839-test suite pass (40.827 s). Graph refreshed (3439 nodes, 6944 edges, 239 communities). Hosted CI remains required before merge. No new coding efficacy or native-host delivery result is claimed.
7 changes: 7 additions & 0 deletions TASKS.md
Original file line number Diff line number Diff line change
Expand Up @@ -1234,3 +1234,10 @@ The retained private events show test invocations that differ from the frozen re
Future repair-profile trials expose exact standalone validation commands to both baseline and treatment before treatment-specific advice instructions. The initial focused check precedes edits; post-edit focused, full-suite and diff checks remain mandatory. Optional initial-failure triage stays optional. Default and non-repair prompts remain unchanged. Historical trials are not rescored or rerun.

Two new symmetry/compatibility tests, nine profile tests and fifteen triage tests pass. Graph refreshed (3439 nodes, 6944 edges, 239 communities). The concurrent shell-parser patch still has a separate diagnostic test failure and is excluded from this commit. Hosted CI on this isolated change is required before merge. No coding benefit or native-host delivery is claimed.


## VCR403 — Prospective shell-command observation hardening

Simple Bash and POSIX-shell command wrappers are normalized lexically without executing their contents. Compound commands, substitutions, redirections, environment assignments, newline separators and malformed wrappers cannot receive individual test gate credit. Cross-layer aggregate diagnostics remain separate from successful mandatory-suite observations and support both string and argv-list wrappers.

This is prospective instrumentation hardening, not a retrospective explanation or rescore of VCR402. Its original Bash wrapper form was already supported. Twenty-five cross-layer tests and the full 839-test suite pass (40.827 s). Graph refreshed (3439 nodes, 6944 edges, 239 communities). Hosted CI remains required before merge. No new coding efficacy or native-host delivery result is claimed.
42 changes: 40 additions & 2 deletions scripts/pilot_cross_layer_test_order_pair.py
Original file line number Diff line number Diff line change
Expand Up @@ -260,6 +260,14 @@ def _required_component_status(item: dict[str, Any], invocation_kind: str) -> st
return "unparseable"
raw = item.get("command")
if isinstance(raw, list) and all(isinstance(part, str) for part in raw):
shell_names = {"bash", "/bin/bash", "/usr/bin/bash",
"sh", "/bin/sh", "/usr/bin/sh"}
if (len(raw) == 3 and raw[0] in shell_names
and raw[1] in {"-c", "-lc"}):
wrapped_command = " ".join(shlex.quote(part) for part in raw)
return _required_component_status(
{"command": wrapped_command}, invocation_kind
)
if any(part in {"&&", "||", ";", "|", "&"} for part in raw):
# An argv list is not sufficient evidence that these tokens are shell syntax.
return "unparseable"
Expand Down Expand Up @@ -339,7 +347,7 @@ def _required_component_status(item: dict[str, Any], invocation_kind: str) -> st


def _test_invocation_kind(item: dict[str, Any]) -> str | None:
"""Classify test-runner invocation metadata without retaining raw command text."""
"""Classify test-runner metadata without retaining raw command text."""
raw = item.get("command")
if isinstance(raw, str):
raw_text = raw
Expand All @@ -352,7 +360,37 @@ def _test_invocation_kind(item: dict[str, Any]) -> str | None:

argv = engine._command_argv(item)
if argv is None:
return "unknown" if re.search(r"(?i)\b(?:pytest|unittest)\b", raw_text) else None
# This path classifies diagnostic metadata only; it does not map a
# compound process exit to any individual shell command.
try:
outer_argv = (
shlex.split(raw_text) if isinstance(raw, str)
else raw if isinstance(raw, list) else []
)
except ValueError:
outer_argv = []
shell_names = {"bash", "/bin/bash", "/usr/bin/bash",
"sh", "/bin/sh", "/usr/bin/sh"}
if (len(outer_argv) == 3 and outer_argv[0] in shell_names
and outer_argv[1] in {"-c", "-lc"}):
script = outer_argv[2]
elif isinstance(raw, str):
script = raw_text
else:
script = ""
if any(char in script for char in {"$", chr(96), "<", ">", "(", ")"}):
return "unknown" if re.search(
r"(?i)\b(?:pytest|unittest)\b", raw_text
) else None
try:
lexer = shlex.shlex(script, posix=True, punctuation_chars=";&|")
lexer.whitespace_split = True
lexer.commenters = ""
argv = list(lexer)
except ValueError:
return "unknown" if re.search(
r"(?i)\b(?:pytest|unittest)\b", raw_text
) else None
if sum(len(token) for token in argv) > 8192:
return "unknown" if re.search(r"(?i)\b(?:pytest|unittest)\b", raw_text) else None

Expand Down
30 changes: 27 additions & 3 deletions scripts/pilot_test_order_pair.py
Original file line number Diff line number Diff line change
Expand Up @@ -85,18 +85,42 @@ def _command_argv(item: dict[str, Any]) -> list[str] | None:
if isinstance(raw, list) and all(isinstance(part, str) for part in raw):
argv = raw
elif isinstance(raw, str):
# CLI JSON may represent embedded shell newlines as backslash-n/r.
if any(marker in raw for marker in ("\n", "\r", r"\n", r"\r")):
return None
try:
argv = shlex.split(raw)
except ValueError:
return None
else:
return None
if (len(argv) == 3 and argv[0] in {"bash", "/bin/bash", "/usr/bin/bash"}
and argv[1] == "-lc"):

shell_names = {"bash", "/bin/bash", "/usr/bin/bash",
"sh", "/bin/sh", "/usr/bin/sh"}
shell_family = {"bash", "sh", "zsh", "dash", "fish", "ksh"}
shell_name = Path(argv[0]).name if argv else ""
if shell_name in shell_family:
# Only unwrap the command-only forms observed in CLI command events.
# Parsing is lexical: never invoke a shell or evaluate its input.
if argv[0] not in shell_names or len(argv) != 3 or argv[1] not in {"-c", "-lc"}:
return None
script = argv[2]
if any(char in script for char in "$`\n\r"):
return None
try:
argv = shlex.split(argv[2])
lexer = shlex.shlex(script, posix=True, punctuation_chars="();<>|&")
lexer.whitespace_split = True
lexer.commenters = ""
script_argv = list(lexer)
except ValueError:
return None
if (not script_argv
or any(token and all(char in "();<>|&" for char in token)
for token in script_argv)
or any(re.match(r"^[A-Za-z_][A-Za-z0-9_]*=", token)
for token in script_argv)):
return None
return script_argv
return argv


Expand Down
11 changes: 8 additions & 3 deletions tests/test_pilot_cross_layer_test_order_pair.py
Original file line number Diff line number Diff line change
Expand Up @@ -270,6 +270,11 @@ def test_required_suite_component_diagnostic_is_private_and_never_gate_credit(se
+ runner.REQUIRED_COMMAND + '"',
exit_code=0,
),
event(
"item.completed", "required-in-argv-bash-wrapper",
["/usr/bin/bash", "-lc", "pytest tests/test_private_case && " + runner.REQUIRED_COMMAND],
exit_code=0,
),
event(
"item.completed", "combined-without-required",
"python -m unittest tests.test_private_case && pytest tests/test_private_case",
Expand All @@ -291,16 +296,16 @@ def test_required_suite_component_diagnostic_is_private_and_never_gate_credit(se
rows, [start + index / 1000 for index in range(len(rows))], start
)

self.assertEqual(result["unmatched_test_invocation_count"], 6)
self.assertEqual(result["unmatched_test_invocation_count"], 7)
self.assertEqual(
result["unmatched_test_invocation_required_component_status_counts"],
{
"exact_required_component_present": 3,
"exact_required_component_present": 4,
"no_exact_required_component": 1,
"unparseable": 2,
},
)
self.assertEqual(result["unmatched_test_invocation_kind_counts"]["combined"], 5)
self.assertEqual(result["unmatched_test_invocation_kind_counts"]["combined"], 6)
self.assertFalse(result["required_suite_invocation_observed"])
self.assertIsNone(result["required_suite_exit"])
self.assertEqual(result["focused_invocation_count"], 0)
Expand Down
125 changes: 125 additions & 0 deletions tests/test_pilot_shell_wrapper_contract.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,125 @@
from __future__ import annotations

import importlib.util
import json
from pathlib import Path
import sys
import unittest

ROOT = Path(__file__).resolve().parents[1]
SCRIPT = ROOT / "scripts" / "pilot_test_order_pair.py"
sys.path.insert(0, str(ROOT / "scripts"))
SPEC = importlib.util.spec_from_file_location("pilot_test_order_pair_shell_contract", SCRIPT)
runner = importlib.util.module_from_spec(SPEC)
assert SPEC and SPEC.loader
SPEC.loader.exec_module(runner)


def event(event_type: str, identifier: str, command: str, **extra) -> str:
return json.dumps({
"type": event_type,
"item": {
"id": identifier,
"type": "command_execution",
"command": command,
**extra,
},
})


class ShellWrapperContractTests(unittest.TestCase):
def test_observed_bash_and_sh_wrappers_match_focused_full_and_rank(self):
focused = "/usr/bin/bash -lc " + json.dumps(runner.UNIT_COMMAND)
full = "sh -c " + json.dumps(runner.REQUIRED_COMMAND)
rank_command = (
"python -m jevcompass tests rank --input test-options.json --json"
)
rank = "bash -lc " + json.dumps(rank_command)

self.assertEqual(runner._test_kind({"command": focused}), "unit")
self.assertEqual(runner._test_kind({"command": full}), "required")
self.assertTrue(runner._rank_invocation({"command": rank}))

def test_wrapped_evidence_is_correlated_to_the_command_event(self):
focused = "/usr/bin/bash -lc " + json.dumps(runner.CONTRACT_COMMAND)
full = "/usr/bin/bash -lc " + json.dumps(runner.REQUIRED_COMMAND)
started = 100.0
lines = [
event("item.started", "focused-1", focused),
event("item.completed", "focused-1", focused, exit_code=0),
event("item.started", "full-1", full),
event("item.completed", "full-1", full, exit_code=0),
]

receipt = runner._event_receipts(
lines, [started, started + 0.25, started + 0.5, started + 0.75], started
)

self.assertEqual(
receipt["focused_test_exits"],
[{"candidate_id": "contract", "exit_code": 0}],
)
self.assertEqual(receipt["required_suite_exit"], 0)
self.assertTrue(receipt["required_suite_invocation_observed"])
self.assertEqual(receipt["focused_invocation_count"], 1)

def test_simple_quoted_arguments_are_preserved_without_execution(self):
script = "python -m unittest -p 'test_unit*.py' -v"
wrapped = "bash -c " + json.dumps(script)

self.assertEqual(
runner._command_argv({"command": wrapped}),
["python", "-m", "unittest", "-p", "test_unit*.py", "-v"],
)

def test_direct_argv_behavior_is_unchanged(self):
direct = ["python", "-m", "unittest", "discover", "-s", "tests", "-v"]

self.assertEqual(runner._command_argv({"command": direct}), direct)

def test_compound_or_unsafe_shell_syntax_is_rejected(self):
scripts = [
runner.UNIT_COMMAND + " && echo done",
runner.UNIT_COMMAND + " | tee output.txt",
runner.UNIT_COMMAND + " > output.txt",
runner.UNIT_COMMAND + "; echo done",
runner.REQUIRED_COMMAND + "\n echo done",
runner.UNIT_COMMAND + "\n" + runner.CONTRACT_COMMAND,
"python -m unittest $(echo test)",
"python -m unittest $TEST_TARGET",
"python -m unittest `echo test`",
"PYTHONPATH=src " + runner.UNIT_COMMAND,
"python -m unittest -p 'unterminated",
]
for script in scripts:
with self.subTest(script=script):
command = "bash -lc " + json.dumps(script)
self.assertIsNone(runner._command_argv({"command": command}))
self.assertIsNone(runner._test_kind({"command": command}))

for command in (
"bash -lc " + json.dumps(runner.UNIT_COMMAND) + " extra",
"bash --noprofile -lc " + json.dumps(runner.UNIT_COMMAND),
"zsh -c " + json.dumps(runner.UNIT_COMMAND),
):
with self.subTest(command=command):
self.assertIsNone(runner._command_argv({"command": command}))

def test_compound_process_exit_is_not_assigned_to_individual_tests(self):
command = "bash -lc " + json.dumps(runner.UNIT_COMMAND + " && echo done")
started = 100.0
lines = [
event("item.started", "compound", command),
event("item.completed", "compound", command, exit_code=1),
]

receipt = runner._event_receipts(lines, [started, started + 0.2], started)

self.assertEqual(receipt["focused_test_exits"], [])
self.assertEqual(receipt["required_suite_exit"], None)
self.assertEqual(receipt["focused_invocation_count"], 0)
self.assertFalse(receipt["required_suite_invocation_observed"])


if __name__ == "__main__":
unittest.main()