From ae4868b09a58ef48a8328d9ab7da09895e0c8f77 Mon Sep 17 00:00:00 2001 From: Toni Nowak Date: Thu, 1 Oct 2026 01:36:00 +0200 Subject: [PATCH] feat: prepare v0.1.23 release (docs truth pass, ledger audit, changelog) --- CHANGELOG.md | 82 +++++++++++++++++++++++++++++++++++++ PILOT.md | 9 ++++ README.md | 46 +++++++++++---------- RELEASE_NOTES.md | 20 +++++---- ROADMAP.md | 6 ++- TASKS.md | 9 ++++ pyproject.toml | 4 +- src/jevcompass/__init__.py | 2 +- tools/check_distribution.py | 8 ++-- 9 files changed, 148 insertions(+), 38 deletions(-) create mode 100644 CHANGELOG.md diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..d1ec551 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,82 @@ +# Changelog + +JevCompass publishes a focused changelog per release. Evidence boundaries and pilot +limits live in [PILOT.md](PILOT.md); release-task receipts live in [TASKS.md](TASKS.md). +Earlier releases predate this file and are summarized in +[RELEASE_NOTES.md](RELEASE_NOTES.md) and the +[GitHub releases](https://github.com/acidkill/JevCompass/releases) page. + +## 0.1.23 (2026-10-01) + +This release publishes the three explicit, nonblocking coding-decision commands that +have been maturing on `main` since v0.1.22, together with the fifth bundled skill and +the typed-decision cache. Nothing in this release gates commands, runs tools, changes +permissions, or waives required validation, and no speed, quality, or cost benefit is +claimed: the paired-trial evidence to date remains mixed or censored +(see [PILOT.md](PILOT.md), VCR308/VCR312). + +### Added — user-facing + +- **Pretask strategy choice:** `jevcompass strategy choose --kind coding --signal …` + returns up to two reviewed strategies from allowlisted signals only; it accepts no + prompt, path, source, or log text. Verified local contract evidence + (`--contract-evidence consistent|conflicting|absent|partial|unknown`) and explicitly + resolved strategies (`--resolved-strategy`) route locally without a remote call. + Dependency-change inspection advice includes a short local migration checklist. +- **Opt-in strategy hook:** `jevcompass install --strategy-advice` attaches conservative + pretask guidance to `UserPromptSubmit`; `--disable-strategy-advice` removes it. + Ordinary installs are unchanged and the hook has no evaluated coding benefit. +- **Post-change test order:** `jevcompass tests discover` finds Python focused checks + from staged, unstaged, and untracked changes; `jevcompass tests rank` orders explicit + or discovered candidates. It prints an order and never executes tests; the supplied + required gate is returned unchanged. Optional verified metadata covers change + signals, per-candidate coverage and runtime buckets, and coverage targets. Bounded + `decision_reason` values distinguish local resolution, remote choice, and fallbacks. +- **Failure triage:** `jevcompass triage --exit-code … --kind … --hypothesis …` ranks a + first diagnostic step for ambiguous failures using enums only. Verified observation + facts (`--import-observation`, `--assertion-observation`, `--timeout-observation`) + can resolve or constrain the choice locally and skip the remote call. Opt-in + `--rank-hypotheses` adds a bounded pairwise causal ordering. Output includes a fixed + `decision_reason` and per-step `selection_source` provenance. +- **Caller-verified diagnostic costs:** `--diagnostic-cost HYPOTHESIS=low|medium|high|unknown` + passes fixed relative-effort tokens for supplied hypotheses only. Costs may guide + next-check ordering; they are never treated as causal-likelihood evidence and never + waive required checks. +- **Typed-decision cache (opt-in):** `JEVCOMPASS_TYPED_DECISION_CACHE=1` reuses + validated strategy, test-order, and triage decisions for up to 24 hours with explicit + `cache_hit`/provenance labels; disabled by default. +- **Fifth bundled skill:** the opt-in installer now ships + `jevcompass-coding-workflow` alongside `jevcompass-focused-tests`, + `jevcompass-regression-review`, `jevcompass-plan-implementation`, and + `jevcompass-plan-cutover`. Skills remain explicit opt-in and never install silently. +- The `Changelog` project URL now points to this file. + +### Added — repository trial tooling (not in the wheel) + +- Paired-trial runners, supervisor triage bridges, case profiles, and immutable + fixtures for pretask, test-order, and triage comparisons, including equal-arm + mandatory-command exposure, safe shell-wrapper normalization, profile diagnostic + costs wired end to end through the bridge, and observation flags ordered before + `--rank-hypotheses` so intercepted commands stay bridge-counted. + +### Fixed + +- Triage separates causal hypotheses from diagnostic actions in pilot contracts. +- Coding intent is preserved across mandatory validation commands. +- Validation-timing status labels align with measured evidence. +- Safe simple shell wrappers are normalized without granting compound-command gate + credit. + +### Verification + +- Full offline suite: 858 Python tests pass on Linux at this revision; the hosted CI + job runs the same suite on Python 3.11 for the release pull request and again inside + the publish workflow. Distribution contents are checked by + `tools/check_distribution.py` in the publish workflow. +- Registry publication receipts (Trusted Publishing run, artifact digests, fresh + install check) are recorded in [TASKS.md](TASKS.md), [PILOT.md](PILOT.md), + [README.md](README.md), and [ROADMAP.md](ROADMAP.md) after publication. +- Unchanged boundaries: advice stays nonblocking; remote ranking sees only allowlisted + coarse metadata; original failing exits and required validation are preserved; + macOS runtime and fresh Desktop prompt delivery remain unverified; no repeatable + speed or quality benefit is established. diff --git a/PILOT.md b/PILOT.md index 6b2f437..3029776 100644 --- a/PILOT.md +++ b/PILOT.md @@ -1856,3 +1856,12 @@ The bundled English coding workflow explains optional caller-verified relative c Future paired trials may declare caller-verified relative diagnostic costs in a case profile for supplied hypothesis IDs only. Both arms receive the same cost disclosure; the treatment arm is told costs may guide next-check ordering only, are never causal-likelihood evidence, and never waive required checks. Costs reach the remote decision only as fixed low/medium/high/unknown enum tokens through the validated bridge spec, the exact triage command, the intercepted shim request, and the production triage call, keeping transport bridge-counted. Cost-bearing requests are rejected by the bridge and shim unless they match the supervisor-validated profile. Profiles without costs keep identical prompts, commands, requests, and decision state. The supervisor triage command now orders observation flags before `--rank-hypotheses` to match the bridge command and shim, closing a silent-fallback path for exact rank-plus-observation requests in future trials; no historical trial is rescored or rerun. This is prospective trial instrumentation; no pair was run. Ten new wiring tests plus updated doubles pass, including parity and interception for rank-plus-observation commands, and the full 858-test suite passes locally with `git diff --check` clean. Hosted CI remains required before merge. No efficacy, delivery, or release claim is made. + + +## VCR409 — Ledger numbering audit: no missing records + +A full audit accounted for every VCR identifier used by the acceptance and task ledgers. Twenty-seven identifiers — VCR278, VCR282, VCR300, VCR301, VCR306, VCR307, VCR309, VCR313, VCR316, VCR327, VCR328, VCR337, VCR339, VCR340, VCR357, VCR358, VCR361, VCR363, VCR364, VCR368, VCR369, VCR378, VCR385, VCR386, VCR388, VCR390 and VCR406 — appear in neither ledger, in no commit subject across all refs, and in no other tracked file; they are unused identifiers, not lost or withheld pilot records. VCR283 is recorded here only; the workflow-integration gap it announced was later closed by the phase-aware skill and verified-content refresh it described. A companion documentation pass restated the current capability set in README, ROADMAP, RELEASE_NOTES and a new CHANGELOG.md while preserving every unverified-benefit caveat. No frozen result, gate or score changed, and overall acceptance remains not established. + +## VCR410 — v0.1.23 release candidate + +Version 0.1.23 is a packaging release, not an acceptance result. It publishes the explicit strategy, test-order and triage commands, diagnostic-cost metadata, the typed-decision cache and five opt-in skills; per-change details and boundaries are in CHANGELOG.md. Pre-publication evidence at the release revision: 858 offline tests pass on Linux, `git diff --check` is clean, the exact-tree wheel and sdist pass `tools/check_distribution.py`, and the registry was rechecked (latest published 0.1.22; 0.1.23 unused). Hosted CI and the tag-gated publish workflow remain the merge and release gates. This release claims no speed, quality, cost, host-delivery or platform benefit; the frozen cohorts remain censored or mixed and fresh Desktop delivery stays unverified. diff --git a/README.md b/README.md index a23d004..42c366a 100644 --- a/README.md +++ b/README.md @@ -24,13 +24,13 @@ jevcompass doctor Start with `jevcompass doctor` to see which representative tasks have enough locally available candidates for a recommendation. Before hook installation it may exit nonzero because the hooks are not registered yet; the capacity report is still useful. Then try a manual request only when the report shows a useful choice for that task; for example, run `jevcompass recommend --category coding --domain python` when **coding_python** reports `decision candidates` or `local candidates`. A clean profile may report `low signal skip` or `silent`, in which case that request can correctly return no recommendation. -To install from source instead, clone this repository and run `pipx install .` from the checkout. For the verified release, use `pipx install jevcompass==0.1.22`; check [releases](https://github.com/acidkill/JevCompass/releases) for the latest version. Version 0.1.22 puts validated candidate IDs immediately after the advice ID. It passed 276 Python tests, hosted CI, Trusted Publishing and a fresh registry pipx setup. One isolated CLI session on the published package reported `exec_command` and `unittest` before first tool; broader host-level impact is unmeasured. +To install from source instead, clone this repository and run `pipx install .` from the checkout. For the verified release, use `pipx install jevcompass==0.1.23`; check [releases](https://github.com/acidkill/JevCompass/releases) for the latest version. Version 0.1.23 publishes the three explicit coding-decision commands (`strategy choose`, `tests discover`/`rank`, and `triage` with diagnostic costs), the opt-in typed-decision cache, and a fifth bundled skill; per-change details and evidence boundaries are in [CHANGELOG.md](CHANGELOG.md). The release revision passes the full offline suite (858 Python tests on Linux), and the release pull request and publish workflow re-run the suite on Python 3.11 with exact-tag distribution checks. One earlier isolated CLI session on published 0.1.22 reported `exec_command` and `unittest` before first tool; no broader host-level impact is measured and no speed or quality benefit is established. The supplemental CI-guided coding pilot exposed a measurement gap: older receipts cannot distinguish exact, equivalent or chained test commands. A corrected parser separately records the exact CI invocation and the same suite without `-v`; in a new pair JevCompass ran the exact CI command while baseline ran the equivalent suite. The advised arm was slower, so no efficiency gain is established. [Pilot evidence](PILOT.md) has the blind receipts and limits. -When a coding task already states an exact test command, the source advisor skips duplicate test-selection hints. For a Python repository whose CI clearly runs unittest, the source catalog can now offer that runner for coding tasks without a prescribed test command. This relevance change is not a measured speed gain. [Pilot evidence](PILOT.md) records the control pair. +When a coding task already states an exact test command, the advisor skips duplicate test-selection hints. For a Python repository whose CI clearly runs unittest, the catalog can offer that runner for coding tasks without a prescribed test command. This relevance change is not a measured speed gain. [Pilot evidence](PILOT.md) records the control pair. -The published advice wording remains unchanged. An experimental compact source-checkout variant shortened one synthetic context but was slower than no advisor in one quality-tied P01 pair; it is not a recommended efficiency setting. [Pilot evidence](PILOT.md) records the limits. +The published advice wording remains unchanged. An experimental compact variant shortened one synthetic context but was slower than no advisor in one quality-tied P01 pair; it is not a recommended efficiency setting. [Pilot evidence](PILOT.md) records the limits. The recommendation is environment-dependent: it can show local advice, an unranked shortlist, or no recommendation. We are testing whether advice improves completion time or token usage versus the same Codex task without it. A single keyless CLI P08 surrogate pair tied on quality and saw a shorter treatment turn. In a separate simple P01/P03 repeat, both arms completed each task and required command; treatment with local advice was slower on P01, while treatment without advice was faster on P03. These small mixed results do not prove an efficiency gain. [Pilot evidence](PILOT.md) separates observed process time, root-turn token counters, and human-rated first productive action; no billing savings are claimed. The manual command uses explicit category/domain metadata and never takes a task prompt. @@ -59,51 +59,51 @@ jevcompass skills install jevcompass doctor ``` -The source checkout's installer copies four skills: `jevcompass-focused-tests`, `jevcompass-regression-review`, `jevcompass-plan-implementation`, and `jevcompass-plan-cutover`. The published v0.1.22 package contains the original two coding skills; check `jevcompass skills install --dry-run` for the version you installed. The planning skills cover an ordinary implementation breakdown and a migration or cutover with rollback, respectively; they are distinct optional candidates, and Jev can compare them only when both are installed. With the default profile it uses `~/.agents/skills`; with `CODEX_HOME` set it uses that profile's `skills/` directory. It never changes hooks, executes scripts, calls Jev, or overwrites a different skill with the same name. Identical repeats are no-ops. Read the bundled `SKILL.md` files before enabling them, and start a fresh Codex session if they do not appear. Remove the installed skill directories yourself after checking their contents if you no longer want them. +The installer copies five skills: `jevcompass-focused-tests`, `jevcompass-regression-review`, `jevcompass-plan-implementation`, `jevcompass-plan-cutover`, and `jevcompass-coding-workflow`. Published v0.1.23 contains all five; the published v0.1.22 package contained the original two coding skills, so check `jevcompass skills install --dry-run` for the version you installed. The planning skills cover an ordinary implementation breakdown and a migration or cutover with rollback, respectively; they are distinct optional candidates, and Jev can compare them only when both are installed. With the default profile it uses `~/.agents/skills`; with `CODEX_HOME` set it uses that profile's `skills/` directory. It never changes hooks, executes scripts, calls Jev, or overwrites a different skill with the same name. Identical repeats are no-ops. Read the bundled `SKILL.md` files before enabling them, and start a fresh Codex session if they do not appear. Remove the installed skill directories yourself after checking their contents if you no longer want them. Suggestions for these skills are conditional on the task matching their guidance in `SKILL.md`; a suggestion is not a requirement. This remains an experiment with no demonstrated speed or quality benefit. An installed v0.1.22 CLI surrogate for project setup delivered local `exec_command`/`git` advice before the first tool, but tied on blind scaffold quality. A subsequent paired CLI repeat retained a recognized successful unittest in both arms; it does not verify the planned Desktop case or show a performance benefit. See [pilot evidence](PILOT.md). A separate API-contract skill prototype was withheld after a opt-in blinded P07 trial in which the baseline authored the requested contract and the advised arm did not; see [pilot evidence](PILOT.md). It is not part of the installer. In one source-built C05 pair, the agent read the suggested `create-plan` skill, but the blinded scores tied 6/7. This single result does not establish causality or effectiveness. That C05 pair did not test Desktop. Separate named-role Desktop worker smokes confirmed advice delivery and one actual focused-skill read after opt-in installation, without measuring a speed or quality gain. The skills and installer are included since v0.1.16 but are not installed unless you opt in. They are absent from v0.1.15. -## Opt-in coding strategy hook (source checkout) +## Opt-in coding strategy hook -The source checkout can attach conservative pretask strategy guidance to `UserPromptSubmit` when you explicitly install it with `jevcompass install --strategy-advice`. Reinstalling without a flag preserves the setting; use `--disable-strategy-advice` to turn it off. Ordinary installs remain unchanged. The hook considers only coding prompts with narrow, explicit request wording. These signals describe what the prompt asks for, not verified repository or test facts; prompts, paths and source text are not sent to the strategy API. The strategy selector is the only remote-choice call on this route; catalog advice remains local. Cached selections are labeled and do not make a new request. If the prompt is outside scope, local classification is uncertain, or selection fails, the existing advice path remains available. This opt-in delivery has not been evaluated for coding benefit. +JevCompass can attach conservative pretask strategy guidance to `UserPromptSubmit` when you explicitly install it with `jevcompass install --strategy-advice`. Reinstalling without a flag preserves the setting; use `--disable-strategy-advice` to turn it off. Ordinary installs remain unchanged. The hook considers only coding prompts with narrow, explicit request wording. These signals describe what the prompt asks for, not verified repository or test facts; prompts, paths and source text are not sent to the strategy API. The strategy selector is the only remote-choice call on this route; catalog advice remains local. Cached selections are labeled and do not make a new request. If the prompt is outside scope, local classification is uncertain, or selection fails, the existing advice path remains available. This opt-in delivery has not been evaluated for coding benefit. -## Coding strategy before work (source candidate) +## Coding strategy before work -For a substantial coding task with genuinely competing approaches, use only coarse signals you verified locally. Skip this command for small, obvious edits: one blinded simple-task pair tied on observable quality while the explicit local strategy call added 6.81 seconds and more tokens; Jev did not select a strategy in that pair. A richer retry-contract pair received a real Jev choice, but treatment scored lower and took longer; see [PILOT.md](PILOT.md). Keep strategy selection experimental. The source checkout exposes `jevcompass strategy choose --kind coding --signal existing_symbol --signal behavior_change --json`. It returns up to two reviewed strategies and labels whether Jev selected the first (`remote-choice`) or a deterministic local order applied (`no-remote-choice`). It accepts no prompt, path, source, or log text. With one clear strategy, unavailable Jev, or uncertain output, it avoids a remote choice. This explicit command is not in published v0.1.22, and no paired task gain has been measured. +For a substantial coding task with genuinely competing approaches, use only coarse signals you verified locally. Skip this command for small, obvious edits: one blinded simple-task pair tied on observable quality while the explicit local strategy call added 6.81 seconds and more tokens; Jev did not select a strategy in that pair. A richer retry-contract pair received a real Jev choice, but treatment scored lower and took longer; see [PILOT.md](PILOT.md). Keep strategy selection experimental. Version 0.1.23 exposes `jevcompass strategy choose --kind coding --signal existing_symbol --signal behavior_change --json`. It returns up to two reviewed strategies and labels whether Jev selected the first (`remote-choice`) or a deterministic local order applied (`no-remote-choice`). It accepts no prompt, path, source, or log text. With one clear strategy, unavailable Jev, or uncertain output, it avoids a remote choice. This explicit command is published since v0.1.23; no paired task gain has been measured. -If verified local evidence already determines the approach, continue directly rather than calling a selector to confirm it. Integrations that already carry that local decision can pass `resolved_strategy` to the Python API or `--resolved-strategy` to the source CLI. For example, `jevcompass strategy choose --kind coding --signal existing_symbol --signal behavior_change --resolved-strategy define_contract_then_implement --json` returns only that eligible reviewed strategy, without constructing a decision client or making an API request. Unknown or ineligible explicit resolutions abstain locally. This option does not establish that the caller's evidence is correct, grant permissions or waive tests; do not use it to disguise unresolved decisions. +If verified local evidence already determines the approach, continue directly rather than calling a selector to confirm it. Integrations that already carry that local decision can pass `resolved_strategy` to the Python API or `--resolved-strategy` to the CLI. For example, `jevcompass strategy choose --kind coding --signal existing_symbol --signal behavior_change --resolved-strategy define_contract_then_implement --json` returns only that eligible reviewed strategy, without constructing a decision client or making an API request. Unknown or ineligible explicit resolutions abstain locally. This option does not establish that the caller's evidence is correct, grant permissions or waive tests; do not use it to disguise unresolved decisions. -The source API also accepts `contract_evidence`, and the source CLI accepts `--contract-evidence` with `consistent`, `conflicting`, `absent`, `partial`, or `unknown`. Supply this only after inspecting the relevant contract, tests and callers locally. For coding with exactly `existing_symbol` and `behavior_change`, consistent evidence selects inspection locally; conflicting or absent evidence selects contract definition locally. These routes make no API request. Partial evidence can accompany an already eligible unresolved choice; it does not force a remote call. Unknown evidence preserves the existing fallback. An explicit resolved strategy that contradicts a deterministic evidence result abstains. These are conservative routing rules, not proof that the agent has interpreted the contract correctly or that the task will finish faster. +The API also accepts `contract_evidence`, and the CLI accepts `--contract-evidence` with `consistent`, `conflicting`, `absent`, `partial`, or `unknown`. Supply this only after inspecting the relevant contract, tests and callers locally. For coding with exactly `existing_symbol` and `behavior_change`, consistent evidence selects inspection locally; conflicting or absent evidence selects contract definition locally. These routes make no API request. Partial evidence can accompany an already eligible unresolved choice; it does not force a remote call. Unknown evidence preserves the existing fallback. An explicit resolved strategy that contradicts a deterministic evidence result abstains. These are conservative routing rules, not proof that the agent has interpreted the contract correctly or that the task will finish faster. The latest compliant supervisor-prepared CLI comparison confirmed strategy advice before the first observed tool and complete focused/full validation, but treatment was 0.754 seconds slower with identical final code. See VCR-257-H in [PILOT.md](PILOT.md). Initial-prompt injection in that harness does not verify native Desktop hook delivery. With a verified `dependency_change` signal, source inspection advice also includes a short local migration checklist: compare API/signature, preserve the caller contract, and check exceptions and resource ownership. This adds no provider call and is not a measured quality or speed guarantee. -## Post-change test order (source candidate) +## Post-change test order -The source checkout can discover Python focused checks from staged, unstaged, and untracked changes: `jevcompass tests discover --required 'python -m unittest discover -s tests -v' --json` from the Git root. Supply the actual required command from your repository's instructions or CI; JevCompass does not invent that gate. Discovery bounds the scan, rejects symlinks and ambiguous filename mappings, and abstains when it cannot safely associate code and tests. Unit and integration tests for the same changed module become distinct alternatives. For other languages or curated candidates, `jevcompass tests rank` accepts explicit local JSON metadata. It **prints an order; it does not execute tests**. This command is not in the published v0.1.22 package, and no speed or quality improvement has been established. +JevCompass can discover Python focused checks from staged, unstaged, and untracked changes: `jevcompass tests discover --required 'python -m unittest discover -s tests -v' --json` from the Git root. Supply the actual required command from your repository's instructions or CI; JevCompass does not invent that gate. Discovery bounds the scan, rejects symlinks and ambiguous filename mappings, and abstains when it cannot safely associate code and tests. Unit and integration tests for the same changed module become distinct alternatives. For other languages or curated candidates, `jevcompass tests rank` accepts explicit local JSON metadata. It **prints an order; it does not execute tests**. This command is published since v0.1.23; no speed or quality improvement has been established. ```json {"surface":"api","candidates":[{"id":"unit","kind":"unit","command":"python -m unittest tests.test_api_unit","relevance":0.5},{"id":"contract","kind":"contract","command":"python -m unittest tests.test_api_contract","relevance":0.5}],"required":[{"id":"ci","command":"python -m unittest discover -s tests -v"}]} ``` -Save this as `test-order.json`, then run `jevcompass tests rank --input test-order.json --json`. `status: remote-choice` means Jev ranked genuine competing test kinds; `no-remote-choice` uses stable local relevance order when one choice is obvious, Jev is unavailable, or its answer is uncertain. The source JSON and text output also expose a bounded `decision_reason`: `no_choice_needed`, `local_resolution`, `invalid_response`, `unknown_choice`, `insufficient_confidence`, `provider_error`, or `accepted`. Older manually constructed results may leave it unknown. This distinguishes fallback paths without exposing backend prose or changing confidence thresholds. The required list is returned unchanged and must still be run. Commands and caller IDs stay local: only coarse surface, test kinds, generic descriptors and opaque IDs may reach OpenRouter. Do not put secrets in command strings; this local file and CLI output are readable on your machine. Python filename-based discovery is available in the source checkout; support for other layouts and matched outcome trials remain open. +Save this as `test-order.json`, then run `jevcompass tests rank --input test-order.json --json`. `status: remote-choice` means Jev ranked genuine competing test kinds; `no-remote-choice` uses stable local relevance order when one choice is obvious, Jev is unavailable, or its answer is uncertain. The JSON and text output also expose a bounded `decision_reason`: `no_choice_needed`, `local_resolution`, `invalid_response`, `unknown_choice`, `insufficient_confidence`, `provider_error`, or `accepted`. Older manually constructed results may leave it unknown. This distinguishes fallback paths without exposing backend prose or changing confidence thresholds. The required list is returned unchanged and must still be run. Commands and caller IDs stay local: only coarse surface, test kinds, generic descriptors and opaque IDs may reach OpenRouter. Do not put secrets in command strings; this local file and CLI output are readable on your machine. Python filename-based discovery is published since v0.1.23; support for other layouts and matched outcome trials remain open. Optional test-order metadata can describe verified `signals` (`public_contract_changed`, `boundary_mapping_changed`, `internal_logic_changed`) and each candidate's `coverage` (`direct`, `indirect`, `unknown`) and `runtime` (`fast`, `slow`, `unknown`). Use coverage evidence and comparable local runtime measurements; leave uncertain facts `unknown`. Each candidate may also include `coverage_targets`, a list from the same three change-signal values, after you verify which behaviors the test actually asserts. This distinguishes internal logic from a public mapping without sending test names or source. Omitted targets remain unspecified; unknown or malformed targets abstain locally without reaching OpenRouter. Targets do not establish complete coverage or make a remote call mandatory. Discovery does not invent these observations. A direct fast check against only indirect slow alternatives is chosen locally; mixed tradeoffs can remain eligible for Jev. Test commands, caller IDs and raw durations stay local, and the required validation list remains unchanged. These inputs improve the information available for selection; matched productivity gains remain unverified. -## Triage an ambiguous failed test (source candidate) +## Triage an ambiguous failed test -After a real test fails and you have at least two evidence-backed explanations, classify them locally and ask for a first diagnostic step. For example: `jevcompass triage --exit-code 1 --kind import --hypothesis import_module_missing --hypothesis import_path_changed --json`. The command accepts enums only; never pass a log, source snippet, path, or exception text. It prints at most two locally authored steps and the **observed failing exit code**. It does not rerun tests, execute a fix, or turn failure into success. Run a cheap read-only discriminator before requesting a remote ranking. For import failures, `--import-observation package_present --import-observation target_module_absent --import-observation replacement_module_present` accepts only fixed enum facts from a local check; together they rule out the missing-package hypothesis and skip the Jev call. Contradictory observations abstain. Without an ambiguous choice or confident Jev response it uses local order. This source candidate is not in PyPI v0.1.22 and has no repeatable measured task benefit yet. +After a real test fails and you have at least two evidence-backed explanations, classify them locally and ask for a first diagnostic step. For example: `jevcompass triage --exit-code 1 --kind import --hypothesis import_module_missing --hypothesis import_path_changed --json`. The command accepts enums only; never pass a log, source snippet, path, or exception text. It prints at most two locally authored steps and the **observed failing exit code**. It does not rerun tests, execute a fix, or turn failure into success. Run a cheap read-only discriminator before requesting a remote ranking. For import failures, `--import-observation package_present --import-observation target_module_absent --import-observation replacement_module_present` accepts only fixed enum facts from a local check; together they rule out the missing-package hypothesis and skip the Jev call. Contradictory observations abstain. Without an ambiguous choice or confident Jev response it uses local order. This command is published since v0.1.23 and has no repeatable measured task benefit yet. -For assertion failures, the source candidate also accepts verified `--assertion-observation` enum facts. If the expected-behavior contract is underspecified, include `--hypothesis confirm_behavior_contract --assertion-observation contract_underspecified`: the next step is a local authoritative-policy check, not a remote guess or an automatic repair. A `legacy_fixture_conflict` alone can leave competing diagnostics for Jev. Contradictory `contract_confirmed` and `contract_underspecified` observations abstain. Confirm intended semantics before editing an expectation or implementation; defer unsupported changes if no authoritative policy can be established. If your local review already established that prerequisite, perform the policy check directly and skip the extra CLI call: the first matched contract case tied on observable diagnosis quality while the explicit advisor arm took 2.501 seconds longer. See [PILOT.md](PILOT.md) for its scope and limitations. +For assertion failures, the CLI also accepts verified `--assertion-observation` enum facts. If the expected-behavior contract is underspecified, include `--hypothesis confirm_behavior_contract --assertion-observation contract_underspecified`: the next step is a local authoritative-policy check, not a remote guess or an automatic repair. A `legacy_fixture_conflict` alone can leave competing diagnostics for Jev. Contradictory `contract_confirmed` and `contract_underspecified` observations abstain. Confirm intended semantics before editing an expectation or implementation; defer unsupported changes if no authoritative policy can be established. If your local review already established that prerequisite, perform the policy check directly and skip the extra CLI call: the first matched contract case tied on observable diagnosis quality while the explicit advisor arm took 2.501 seconds longer. See [PILOT.md](PILOT.md) for its scope and limitations. -For timeouts, the source CLI also accepts repeatable `--timeout-observation` enum facts verified locally. An unsatisfiable wait condition selects the existing wait-condition check locally; observed contention **together with** progress and a satisfiable wait selects the resource-contention check. Conflicting facts abstain. Partial facts can inform an already eligible ambiguous choice; they do not create an API call by themselves. These are next-check suggestions, not confirmed causes, and the original failing exit remains unchanged. Example: `jevcompass triage --exit-code 1 --kind timeout --hypothesis timeout_contention --hypothesis timeout_nonterminating --timeout-observation wait_condition_unsatisfiable --json`. +For timeouts, the CLI also accepts repeatable `--timeout-observation` enum facts verified locally. An unsatisfiable wait condition selects the existing wait-condition check locally; observed contention **together with** progress and a satisfiable wait selects the resource-contention check. Conflicting facts abstain. Partial facts can inform an already eligible ambiguous choice; they do not create an API call by themselves. These are next-check suggestions, not confirmed causes, and the original failing exit remains unchanged. Example: `jevcompass triage --exit-code 1 --kind timeout --hypothesis timeout_contention --hypothesis timeout_nonterminating --timeout-observation wait_condition_unsatisfiable --json`. -For unresolved diagnostics, optionally add caller-verified relative check costs, for example `--diagnostic-cost timeout_contention=high --diagnostic-cost timeout_nonterminating=low`. Allowed costs are `low`, `medium`, `high`, and `unknown`, keyed only to supplied hypotheses. Omit costs you cannot establish locally. Costs inform which check to try next; they do not establish causal likelihood, override local resolutions, execute checks, or remove mandatory validation. This is a source-only feature with no measured speed benefit yet. +For unresolved diagnostics, optionally add caller-verified relative check costs, for example `--diagnostic-cost timeout_contention=high --diagnostic-cost timeout_nonterminating=low`. Allowed costs are `low`, `medium`, `high`, and `unknown`, keyed only to supplied hypotheses. Omit costs you cannot establish locally. Costs inform which check to try next; they do not establish causal likelihood, override local resolutions, execute checks, or remove mandatory validation. This is published since v0.1.23 with no measured speed benefit yet. -Triage JSON now includes a fixed `decision_reason` and each step's `selection_source`: a remotely preferred next action, locally resolved guidance, or an unranked local fallback. `hypothesis_ranking_status` remains `not_established`: choosing the next diagnostic action does not establish which cause is most likely. These fields describe the source checkout; the published package remains unchanged. +Triage JSON now includes a fixed `decision_reason` and each step's `selection_source`: a remotely preferred next action, locally resolved guidance, or an unranked local fallback. `hypothesis_ranking_status` remains `not_established`: choosing the next diagnostic action does not establish which cause is most likely. These fields are published since v0.1.23. -For an optional complete hypothesis order, add `--rank-hypotheses` when 2–4 locally plausible causes remain. Jev receives up to six pairwise `choice` questions in the same request as the separate next-step question. The CLI exposes `hypothesis_order` only when every pair has an allowed choice at the confidence threshold and the comparisons form an acyclic complete order. Otherwise the ranking status is `incomplete` or `not_established`, with no order inferred from caller order; the diagnostic step is validated independently. Local resolutions and unambiguous cases still skip remote calls. Opt-in ranking bypasses the choice-only typed cache. This feature is source-only and does not claim improved diagnosis quality. +For an optional complete hypothesis order, add `--rank-hypotheses` when 2–4 locally plausible causes remain. Jev receives up to six pairwise `choice` questions in the same request as the separate next-step question. The CLI exposes `hypothesis_order` only when every pair has an allowed choice at the confidence threshold and the comparisons form an acyclic complete order. Otherwise the ranking status is `incomplete` or `not_established`, with no order inferred from caller order; the diagnostic step is validated independently. Local resolutions and unambiguous cases still skip remote calls. Opt-in ranking bypasses the choice-only typed cache. This feature does not claim improved diagnosis quality. ## Add your own installed skill @@ -177,7 +177,7 @@ For setup diagnostics, run `jevcompass doctor`; use `jevcompass --version` to id ## Privacy -Prompt classification and candidate discovery happen locally. When optional Jev ranking is used, JevCompass sends an allowlisted category, domain, role, an optional coarse planning focus (`migration` or `implementation`), criteria, and generic descriptions of reviewed candidates to OpenRouter's Decisions API. It does **not** send the raw prompt, source code, diffs, repository paths, memory contents, or unapproved `SKILL.md` descriptions. A skill added with `--approve-remote-metadata` is an exception you control: its ID and your generic capability, use and avoid descriptions can be sent to OpenRouter when remote ranking is enabled. Preview these fields first; omit names or details you consider private. The source checkout's explicit coding workflows can additionally send fixed strategy/failure/change signals and opaque test candidates with allowlisted kind, coverage and runtime buckets. Raw diagnostics, command strings and measured durations stay local. OpenRouter receives the API key in the HTTPS authorization header. +Prompt classification and candidate discovery happen locally. When optional Jev ranking is used, JevCompass sends an allowlisted category, domain, role, an optional coarse planning focus (`migration` or `implementation`), criteria, and generic descriptions of reviewed candidates to OpenRouter's Decisions API. It does **not** send the raw prompt, source code, diffs, repository paths, memory contents, or unapproved `SKILL.md` descriptions. A skill added with `--approve-remote-metadata` is an exception you control: its ID and your generic capability, use and avoid descriptions can be sent to OpenRouter when remote ranking is enabled. Preview these fields first; omit names or details you consider private. The explicit coding workflows can additionally send fixed strategy/failure/change signals and opaque test candidates with allowlisted kind, coverage and runtime buckets. Raw diagnostics, command strings and measured durations stay local. OpenRouter receives the API key in the HTTPS authorization header. Local metrics contain event/category/outcome/timing and a short advice ID; they do not record the prompt, model response, paths, or memory content. Local cache entries contain selection metadata. As with any external service, review OpenRouter's terms and data handling before enabling remote ranking. @@ -197,6 +197,8 @@ JevCompass does not infer Codex Plan UI mode. Check `/hooks` and start a fresh s Evidence is deliberately limited to the environments tested: +- Version 0.1.23 publishes the three explicit coding-decision commands (`strategy choose`, `tests discover`/`rank`, `triage` with observations, ranking and diagnostic costs), the opt-in typed-decision cache, and five opt-in bundled skills; per-change details and evidence boundaries are in [CHANGELOG.md](CHANGELOG.md). The release revision passed 858 offline Python tests on Linux, and the release pull request and publish workflow re-run the full suite on Python 3.11 with exact-tag distribution checks; registry receipts are recorded in [TASKS.md](TASKS.md) after publication. No repeatable task benefit is claimed. + - The published 0.1.22 package places selected IDs at the top of each advisory. A fresh isolated CLI 0.155.1 session reported `exec_command` and `unittest` with correlated advice ID `0623c35b` before its first tool. The matching local hook took 5.99 ms, exit was 0, and the fictional fixture was unchanged. This single smoke does not measure task benefit. - [PyPI v0.1.21](https://pypi.org/project/jevcompass/0.1.21/) and its [GitHub release](https://github.com/acidkill/JevCompass/releases/tag/v0.1.21) passed 275 Python 3.11 tests, hosted CI, Trusted Publishing, exact-tag distribution checks and a clean pipx registry install with two advisory hooks and doctor PASS. A single isolated synthetic `testing/python` request composed the shell executor and `unittest` locally in 2.96 ms; it is not a host delivery or broad effectiveness measurement. @@ -237,7 +239,7 @@ JevCompass is licensed under the [Apache License 2.0](LICENSE). Copyright 2026 T ### Optional typed-decision cache -**Source-build feature:** this cache and its readable provenance labels are present on `main`, but are not included in the currently published PyPI 0.1.22 artifact. A private wheel built from current source may still report version 0.1.22; that does not make it the published artifact. +**Included since v0.1.23:** this cache and its readable provenance labels ship in the published package. On versions up to 0.1.22 the cache existed only in source builds, and a private wheel built from that source may still report version 0.1.22; that does not make it the published artifact. Set `JEVCOMPASS_TYPED_DECISION_CACHE=1` to reuse validated strategy, test-order and triage choices across CLI processes for up to 24 hours. It is disabled by default. The cache is bounded to 128 records and uses private files under `~/.cache/jevcompass/typed-decisions-v1`; `JEVCOMPASS_TYPED_CACHE_DIR` can select a dedicated private directory. An existing directory with broader permissions is skipped, not modified. diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index ff1f913..cab947b 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -1,40 +1,46 @@ -# Unreleased supplemental P09 coding pilot (no package release) +# JevCompass v0.1.23 + +Publishes the three explicit coding-decision commands (pretask strategy, post-change test order, ambiguous-failure triage) with verified-evidence local routing, bounded decision reasons and selection provenance, opt-in hypothesis ranking, caller-verified diagnostic-cost metadata, the opt-in typed-decision cache, and the fifth bundled skill `jevcompass-coding-workflow`; the opt-in installer now offers five skills. The full change list with evidence boundaries is in [CHANGELOG.md](CHANGELOG.md). The seven "Included in v0.1.23" sections below were written while these changes were unreleased: their statements about the then-published distribution describe v0.1.22 and earlier, and their evidence limits continue to apply. Local verification at the release revision: 858 offline Python tests on Linux with `git diff --check` clean; the release pull request and the publish workflow re-run the full suite on Python 3.11 with exact-tag distribution checks. Registry receipts (Trusted Publishing run, artifact digests, fresh install) are recorded in [TASKS.md](TASKS.md) and [PILOT.md](PILOT.md) after publication. No speed, quality, or cost benefit is claimed; macOS runtime and fresh Desktop prompt delivery remain unverified. + +--- + +# Included in v0.1.23 — supplemental P09 coding pilot The source-only P09 fixture removes README test-runner guidance and adds local unittest CI; it does not change the default 20-case bank. Two blinded source pairs tied on authored quality, but their substring detector cannot establish exact standalone CI invocation, equivalence or chains after raw events were discarded. A controlled probe identified Codex CLI's `bash -lc` wrapper. The corrected source parser recognizes only that simple wrapper or a bare command, and records exact CI and unverbose equivalent suite separately without retaining command text. A fresh pair tied on blind code quality: treatment ran exact CI, baseline the equivalent suite; advised treatment took 55.60 s versus 29.38 s. No speed or cost benefit is established. See [PILOT.md](PILOT.md). --- -# Unreleased coding test-signal relevance (no package release) +# Included in v0.1.23 — coding test-signal relevance The source advisor omits duplicate focused-test and runner suggestions when a coding task gives an exact test command, except when the task expressly asks to choose tests. The unittest candidate now covers Python coding only when local CI unambiguously requires unittest. A randomized source P01 negative control returned no advice (4.49 ms), both arms passed blind quality and unittest; treatment 16.97 s versus baseline 22.24 s does not demonstrate recommendation benefit. Frozen score SHA-256: `a701e203b86a0c4e2b3be69b3815baeac5646ad9ad684c58da758b2c2600ebef`. --- -# Unreleased compact-advice experiment (default unchanged) +# Included in v0.1.23 — compact-advice experiment (default unchanged) An opt-in `JEVCOMPASS_ADVICE_STYLE=compact` option consolidates repeated context instructions for controlled pilots. One randomized P01 pair tied on blind authored quality and successful unittest in both arms; compact treatment advice arrived before first tool but took 21.24 s versus 18.50 s baseline, with 71,592 versus 69,164 input tokens. The frozen blind score SHA-256 is `7fd6e333abaa5e41b1cc50b93edfe970740cbeed665ee89e1bc3a3ae816aae61`. There is no established speed or billing-cost benefit; the published default remains unchanged. --- -# Unreleased CLI efficiency telemetry (no new package release) +# Included in v0.1.23 — CLI efficiency telemetry The pilot runner now records privacy-safe, bounded root-turn `turn.completed` token counters and full Codex process elapsed time in blind v2 receipts; old v1 receipts remain readable. Cached input and reasoning output are subsets of input and output. A fresh installed-v0.1.22 keyless P08 pair tied on authored quality and recognized successful unittest in both arms; advised treatment took 28.18 s versus 35.07 s baseline, with 71,558 versus 84,762 input tokens (uncached input differed by −148). This one pair is not a causal speed or billing-cost result. An initial P01/P03 simple-task pair had incomplete required-check evidence. A revised randomized repeat with equal bundled skills tied on blind authored quality and confirmed the required command passed in all four arms (score SHA `2b34b161...`). P01 local advice reached treatment before its first tool, but treatment took 20.60 s versus 13.79 s baseline and used 70,734 versus 54,925 input tokens. P03 treatment abstained and took 15.57 s versus 18.10 s baseline, with 54,517 versus 68,752 input tokens. Neither pair demonstrates an advice-driven speed or cost benefit. An in-session Desktop prompt exposed advice ID `44d672d6` before the primary agent's next tool, matched to a 74.85 ms local metric; fresh-session delivery remains open. See [PILOT.md](PILOT.md) for scoring and limits. --- -# Unreleased P08 receipt correction (no new package release) +# Included in v0.1.23 — P08 receipt correction P08 blind receipts now retain allowlisted scaffold presence and observed unittest exit while dropping command content. In a fresh installed-v0.1.22 keyless CLI surrogate pair, both arms authored coherent packages and had recognized successful unittest exits; treatment reported local `exec_command`/`git` advice before first tool (`project-setup/local`, 10.46 ms). The blind quality scores were frozen before mapping (SHA-256 `d438a21996ab95d3cf39bf80902fe5915f10f01d165da1868f5b9e22a781b723`). This is a check-preservation observation, not a demonstrated speed/quality improvement or a Desktop P08 result. The published distribution is unchanged. --- -# Unreleased pilot harness evidence (no new package release) +# Included in v0.1.23 — pilot harness evidence The CLI runner now includes a P08 project-scaffold surrogate in an isolated fixture with a three-file blind export and prospective unittest-exit observation. The planned P08 Desktop case remains open. On installed v0.1.22, a keyless randomized CLI pair tied on blind authored-package quality; treatment reported `exec_command`/`git` advice before first tool (`project-setup/local`, 8.11 ms). The original receipt did not assess test completion, and this small pair does not prove speed or quality benefit. The product distribution and hooks are unchanged. See [PILOT.md](PILOT.md). --- -# Unreleased pilot harness clarification (no new package release) +# Included in v0.1.23 — pilot harness clarification P07 now explicitly requests an authored `STATUS_API.md` while retaining its `api-design/python` preflight label. This changes the synthetic case wording, so older P07 pair scores remain historical. An unpublished API-contract skill prototype received local advice in two equal-profile source trials: the first produced no file in either arm, while the revised blind pair produced a valid baseline contract and no treatment file. Its catalog and bundled-skill edits were removed; the shipped package, hook configuration and optional two-skill installer remain as in v0.1.22. No speed, quality, or remote Jev benefit is claimed. See [PILOT.md](PILOT.md). diff --git a/ROADMAP.md b/ROADMAP.md index 581a485..7cd23c0 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -2,7 +2,7 @@ **Read by stage:** [current pilot evidence](PILOT.md) · [20-case design (not execution)](PILOT_CASES.md) · [release/task receipts](TASKS.md) · [quick start](README.md). -Published [v0.1.21](https://github.com/acidkill/JevCompass/releases/tag/v0.1.21) removes the false remote choice between a shell executor and its test command. The Apache-2.0-licensed [GitHub releases](https://github.com/acidkill/JevCompass/releases) contain a reviewed local catalog, two default nonblocking Codex hooks, optional OpenRouter Decisions ranking, explicit recommendations, an installer, and diagnostics. PyPI v0.1.17 added explicit custom-skill curation; published v0.1.18 corrects its privacy disclosure and the four-document release gate is enforced in the existing CI job. Published 0.1.20 suppresses the generic singleton shell for package documentation after the installed 0.1.19 P05 pair tied on blind README quality. Published 0.1.19 locally filters Python test runner choices using unambiguous CI commands after 0.1.18 remotely selected pytest for a unittest-gated repository; its exact-tag Python suite, distribution check and clean registry pipx install passed. Registration requires a preview and approval of generic fields before they may be sent to OpenRouter. The bundled coding skills and agent-spawn advice remain opt-in. Neither macOS runtime nor a repeatable speed or quality gain is verified. The first four synthetic CLI pairs tied on blind quality; later focused evidence was mixed. See [PILOT.md](PILOT.md) for observations and [TASKS.md](TASKS.md) for task/commit tracking. +Published [v0.1.23](https://github.com/acidkill/JevCompass/releases/tag/v0.1.23) adds the three explicit coding-decision commands (pretask strategy, post-change test order, ambiguous-failure triage) with verified-evidence local routing, bounded decision reasons, opt-in hypothesis ranking, caller-verified diagnostic costs, the opt-in typed-decision cache, and a fifth bundled skill; [CHANGELOG.md](CHANGELOG.md) carries the full change list and evidence boundaries. No repeatable speed, quality, or cost benefit is established by this release. Published [v0.1.21](https://github.com/acidkill/JevCompass/releases/tag/v0.1.21) removes the false remote choice between a shell executor and its test command. The Apache-2.0-licensed [GitHub releases](https://github.com/acidkill/JevCompass/releases) contain a reviewed local catalog, two default nonblocking Codex hooks, optional OpenRouter Decisions ranking, explicit recommendations, an installer, and diagnostics. PyPI v0.1.17 added explicit custom-skill curation; published v0.1.18 corrects its privacy disclosure and the four-document release gate is enforced in the existing CI job. Published 0.1.20 suppresses the generic singleton shell for package documentation after the installed 0.1.19 P05 pair tied on blind README quality. Published 0.1.19 locally filters Python test runner choices using unambiguous CI commands after 0.1.18 remotely selected pytest for a unittest-gated repository; its exact-tag Python suite, distribution check and clean registry pipx install passed. Registration requires a preview and approval of generic fields before they may be sent to OpenRouter. The bundled coding skills and agent-spawn advice remain opt-in. Neither macOS runtime nor a repeatable speed or quality gain is verified. The first four synthetic CLI pairs tied on blind quality; later focused evidence was mixed. See [PILOT.md](PILOT.md) for observations and [TASKS.md](TASKS.md) for task/commit tracking. ## Coding decision workflow: next delivery track @@ -14,6 +14,10 @@ This track develops three explicit, nonblocking decisions. The default hooks rem For each stage: define fixtures and scoring before execution, implement typed and bounded Decisions validation with timeout/uncertainty abstention, test privacy and no-blocking behavior, run equivalent randomized Codex Desktop and CLI cases, and record negative as well as positive outcomes in [PILOT.md](PILOT.md). Do not claim speed, cost or quality improvements until repeated paired evidence supports them. Token counts are a proxy unless model and pricing provenance allow a defensible currency estimate. Each implementation stage gets its own task, commit, PR, green hosted CI, and smem checkpoint. +## 0.1.23 candidate and release evidence + +Version 0.1.23 publishes everything the coding-decision track implemented since v0.1.22: `strategy choose` with contract-evidence and resolved-strategy local routing, the opt-in strategy hook, `tests discover`/`rank` with coverage/runtime signals and decision reasons, `triage` with observation enums, opt-in hypothesis ranking and caller-verified diagnostic costs, the opt-in typed-decision cache, and five opt-in bundled skills. Trial tooling (paired runners, supervisor bridges, case profiles) stays repository-only and is not in the wheel. Local release evidence: 858 offline Python tests on Linux with `git diff --check` clean at the release revision; the release pull request re-runs the full suite on hosted CI, and the publish workflow rebuilds from the exact tag with distribution checks before Trusted Publishing. Registry receipts, artifact digests and the fresh-install check are recorded after publication. A ledger-numbering audit (VCR409) confirms the evidence ledgers have no lost records. Efficacy remains open: the frozen benchmark cohorts are censored or mixed, and fresh Desktop prompt delivery plus macOS runtime stay unverified. + ## 0.1.22 candidate and remaining release evidence Published v0.1.22 places selected candidate IDs immediately below the advice ID. Installed-0.1.21 CLI probes prepared a full `exec_command`/`unittest` advisory but sometimes got an agent pre-tool report with no candidate IDs or no advice. One source-built synthetic CLI session reported both IDs before first command (9.24 ms). PR #101 passed hosted CI; tag peels to `9720cd7`, exact-tree artifact audit and Trusted Publishing run 36156580002 passed. Registry wheel package members matched local bytes. Fresh registry pipx 0.1.22 doctor found two hooks; an isolated CLI session reported both IDs and matching trace `0623c35b` before first command (5.99 ms local hook). Different model runs and a tiny sample preclude a causal effectiveness claim. Fresh Desktop prompt delivery remains open. diff --git a/TASKS.md b/TASKS.md index 0e9cf73..eb2229e 100644 --- a/TASKS.md +++ b/TASKS.md @@ -1260,3 +1260,12 @@ The bundled English coding workflow explains optional caller-verified relative c Case profiles may declare caller-verified relative diagnostic costs for supplied hypothesis IDs. Both arms receive the same cost disclosure and the treatment arm is told costs may guide next-check ordering only, are never causal-likelihood evidence, and never waive required checks. The validated plain-string costs flow through the bridge spec, the exact triage command, the intercepted shim request, and the supervisor bridge into the production triage call, so cost-bearing requests stay bridge-counted instead of silently falling back to the real CLI. Profiles without costs keep identical prompts, commands, requests, and decision state. The supervisor command now places observation flags before `--rank-hypotheses`, matching the bridge command and generated shim, so exact rank-plus-observation triage requests are intercepted instead of silently falling back to the real interpreter. Ten new wiring tests plus updated doubles cover loader accept/reject, equal-arm disclosure, supervisor argv construction, spec validation, bridge request matching and forwarding, shim interception without fallback, and rank-plus-observation command order. The full 858-test suite passes locally (43.1 s) with `git diff --check` clean. Hosted CI remains required before merge. No trial was run and no speed or quality benefit is claimed. + + +## VCR409 — Ledger numbering audit and documentation truth pass + +A full audit accounted for every VCR identifier across both evidence ledgers. Twenty-seven identifiers — VCR278, VCR282, VCR300, VCR301, VCR306, VCR307, VCR309, VCR313, VCR316, VCR327, VCR328, VCR337, VCR339, VCR340, VCR357, VCR358, VCR361, VCR363, VCR364, VCR368, VCR369, VCR378, VCR385, VCR386, VCR388, VCR390 and VCR406 — appear in neither TASKS.md nor PILOT.md, in no commit subject across all refs, and in no other tracked file; they are unused identifiers, not lost or withheld records. VCR283 is recorded in PILOT.md only; the workflow-integration gap it announced was later closed by the phase-aware skill and the verified-content refresh it described. The same pass updated README.md, ROADMAP.md and RELEASE_NOTES.md to the current capability set (five bundled skills, the three explicit coding-decision commands, the typed-decision cache) with version-accurate phrasing while preserving every unverified-benefit caveat, retitled the seven formerly unreleased RELEASE_NOTES sections as included in v0.1.23, and added CHANGELOG.md as the canonical per-release changelog with the pyproject Changelog URL repointed. No frozen result, gate or score changed. + +## VCR410 — v0.1.23 release candidate + +Version 0.1.23 publishes the three explicit coding-decision commands with their local-routing and provenance features, caller-verified diagnostic costs, the opt-in typed-decision cache, and the fifth bundled skill; `pyproject.toml` and the package `__version__` are bumped to 0.1.23 and CHANGELOG.md carries the full change list. Pre-publication local evidence at the release revision: the full offline suite passes (858 tests, Linux), `git diff --check` is clean, `uv build` produced the wheel and sdist, and `tools/check_distribution.py` passed after its expected bundled-skill set was corrected to enumerate `bundled_skills/*/SKILL.md` (five skills) instead of a stale two-name list. The PyPI registry was rechecked before the release claim: the latest published version is 0.1.22 and 0.1.23 is unused. Hosted CI on the release pull request is required before merge; the publish workflow rebuilds from the exact tag, reruns the suite and distribution check, and publishes through Trusted Publishing. Registry receipts are appended after publication. No trial was run; no efficacy, delivery or platform claim is made. diff --git a/pyproject.toml b/pyproject.toml index 395e852..28528b5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "jevcompass" -version = "0.1.22" +version = "0.1.23" description = "Privacy-first tool and skill recommendations for Codex Desktop and CLI, powered by Jev" readme = "README.md" license = "Apache-2.0" @@ -25,7 +25,7 @@ classifiers = [ Homepage = "https://github.com/acidkill/JevCompass" Repository = "https://github.com/acidkill/JevCompass" Issues = "https://github.com/acidkill/JevCompass/issues" -Changelog = "https://github.com/acidkill/JevCompass/blob/main/RELEASE_NOTES.md" +Changelog = "https://github.com/acidkill/JevCompass/blob/main/CHANGELOG.md" [project.optional-dependencies] secure-store = ["keyring>=25"] diff --git a/src/jevcompass/__init__.py b/src/jevcompass/__init__.py index f5f73e6..b4d7b74 100644 --- a/src/jevcompass/__init__.py +++ b/src/jevcompass/__init__.py @@ -1,3 +1,3 @@ """Privacy-first tool and skill advice for Codex Desktop and CLI.""" -__version__ = "0.1.22" +__version__ = "0.1.23" diff --git a/tools/check_distribution.py b/tools/check_distribution.py index 6880ecc..08147ab 100644 --- a/tools/check_distribution.py +++ b/tools/check_distribution.py @@ -45,11 +45,9 @@ def check(archive: Path, source_root: Path = PROJECT_ROOT) -> None: for path in package_root.iterdir() if path.is_file() and (path.suffix == ".py" or path.name == "catalog_data.json") } - for name in ("jevcompass-focused-tests", "jevcompass-regression-review"): - path = package_root / "bundled_skills" / name / "SKILL.md" - assert path.is_file(), f"missing bundled skill: {name}" - relative = path.relative_to(package_root).as_posix() - expected_package[f"jevcompass/{relative}"] = path.read_bytes() + for skill_path in sorted((package_root / "bundled_skills").glob("*/SKILL.md")): + relative = skill_path.relative_to(package_root).as_posix() + expected_package[f"jevcompass/{relative}"] = skill_path.read_bytes() package_prefix = "src/jevcompass/" if is_sdist else "jevcompass/" actual_package = {name: data for name, data in files.items() if name.startswith(package_prefix)}