From 1266c690e1461e60d09d7e32b295cbb6e64a3608 Mon Sep 17 00:00:00 2001 From: xeonvs <11463419+xeonvs@users.noreply.github.com> Date: Wed, 23 Sep 2026 08:27:02 +0200 Subject: [PATCH] Update workflow model profiles for GPT-6 --- PLANS.md | 113 ++++++++++++++++++ README.md | 12 +- .../.claude-plugin/plugin.json | 2 +- .../.codex-plugin/plugin.json | 2 +- .../skills/engineering-workflow/SKILL.md | 2 +- .../assets/agents/explorer.toml.tmpl | 2 +- .../assets/agents/reviewer.toml.tmpl | 6 +- .../assets/agents/utility.toml.tmpl | 2 +- .../references/agent_orchestration.md | 1 + .../references/model_profiles.md | 29 +++-- .../references/target_workflow_upgrade.md | 11 +- .../scripts/upgrade_target_workflow.py | 48 +++++++- .../scripts/validate_skill_repo.py | 10 +- skill/engineering-workflow/SKILL.md | 2 +- .../assets/agents/explorer.toml.tmpl | 2 +- .../assets/agents/reviewer.toml.tmpl | 6 +- .../assets/agents/utility.toml.tmpl | 2 +- .../references/agent_orchestration.md | 1 + .../references/model_profiles.md | 29 +++-- .../references/target_workflow_upgrade.md | 11 +- .../scripts/upgrade_target_workflow.py | 48 +++++++- .../scripts/validate_skill_repo.py | 10 +- tests/test_agent_orchestration.py | 14 ++- tests/test_marketplace_package.py | 4 +- tests/test_skill_repo_validation.py | 10 +- tests/test_upgrade_target_workflow.py | 64 +++++++++- 26 files changed, 360 insertions(+), 83 deletions(-) diff --git a/PLANS.md b/PLANS.md index bb117c0..891f40d 100644 --- a/PLANS.md +++ b/PLANS.md @@ -4,6 +4,119 @@ plan_schema_version: 2 Use this file for active, blocked, ready-for-closure, or recently completed execution work. The canonical lifecycle is the installed `engineering-workflow` planning reference. +## Active Plan: GPT-6 Model Profiles And Marketplace Release + +Status: active +Owner: root +Last Updated: 2026-09-23 + +### Goal + +Update the engineering-workflow model recommendations and optional Codex agent templates for the current GPT-6 family, publish a source release, import it into xeonvs-engineering, and refresh the managed local marketplaces. + +### Plan Origin + +direct_execution + +### Requested Scope + +- Support the current models in the relevant source and distribution repositories and release the marketplace update. + +### Requirement Traceability + +| Requirement | Complete outcome | Source | Work queue | Acceptance or validation | Status | +| --- | --- | --- | --- | --- | --- | +| REQ-001 | Current Codex role mappings, optional agent templates, and model-specific guidance reflect supported GPT-6 models without changing Claude inheritance or user pins. | User request; official OpenAI model catalog and guidance | WQ-01 | Model and effort readback, behavioral/contract tests, aggregate diff review | done | +| REQ-006 | A target upgrade refreshes previously generated Codex agent model profiles when their prior bytes are pristine, while preserving customized files and established opt-in state. | User clarification | WQ-01 | Report/apply/prompt fixture matrix for pristine, customized, and opt-out targets | done | +| REQ-002 | All active source/package version owners and generated bytes agree in a new source release. | User request; repository release contract | WQ-02 | Full release/security gate, package validators, PR merge, annotated tag and release readback | pending | +| REQ-003 | xeonvs-engineering imports exact immutable source bytes and publishes its own release while preserving tgrep-search. | User request; marketplace sync contract | WQ-03 | Recorded provenance/byte verification, catalog tests, release workflow and asset readback | pending | +| REQ-004 | Local Codex and Claude managed marketplaces and workflow installations resolve the published version. | Prior user preference; current release request | WQ-04 | Native CLI update and active version readback | pending | +| REQ-005 | Source and marketplace plans close with durable release evidence. | Workflow lifecycle contract | WQ-05 | Lifecycle check and closure readback | pending | + +### Explicit Non-Goals + +- Change tgrep-search source or version; overwrite user-pinned models; copy Responses API-only fields into Codex TOML; rewrite published history. + +### Constraints + +- Preserve role-specific cost, latency, reasoning, and read-only behavior; confirm target slugs and effort levels against current official documentation and actual Codex availability. +- Keep source repository canonical and marketplace distribution generated from a stable annotated source tag. +- Review each logical commit and aggregate release diff; run final full and pre-push security gates. + +### Inputs And Sources + +- https://developers.openai.com/api/docs/models +- https://developers.openai.com/api/docs/guides/latest-model +- Current `model_profiles.md`, optional agent templates, validators/tests, source and marketplace release contracts. +- Repository audit summary in `/tmp/engineering-model-audit-20260923.json` (generic migration findings are outside this release's model scope). + +### User Decisions And Answers + +- 2026-09-23: Add support for new models in the relevant repositories and publish the marketplace update. +- 2026-09-23: Retain GPT-5.6 where it is cheaper or useful as a fallback; current API prices support Terra as an availability/evaluation fallback, not the cheaper default for these roles. +- 2026-09-23: Route routine commands and tests through tools without a model worker; upgrade an already configured target's generated model profiles when pristine. +- 2026-09-23: Keep Astra available for unusually complex/high-consequence work; require a concrete rationale and user confirmation before the agent initiates a more expensive Astra worker or saved-profile escalation. An explicit user choice already confirms it. +- Earlier session: after publication, refresh local managed marketplaces and installed skills. + +### Completed Baseline State + +- [x] Source main is clean at the start; engineering-workflow 0.9.7 and marketplace 1.0.7 are the last completed releases. The source already maps standard/review to Astra, while utility/explorer remain pinned to GPT-5.6 Terra. + +### Current Work Queue + +- [x] WQ-01 — Implement and validate GPT-6 task routing, profiles, templates, and conservative target migration for REQ-001/REQ-006. `done` +- [ ] WQ-02 — Bump source version, validate, review, merge, tag and release for REQ-002. `in_progress` +- [ ] WQ-03 — Import, validate, review, merge and release xeonvs-engineering for REQ-003. `pending` +- [ ] WQ-04 — Refresh and read back local managed installations for REQ-004. `pending` +- [ ] WQ-05 — Reconcile and close source and marketplace plans for REQ-005. `pending` + +### Locked Decisions + +- Patch release target is provisionally engineering-workflow 0.9.8 and xeonvs-engineering 1.0.8, subject to current remote tag inspection. +- Route deterministic commands and tests through tools; recommend Luna low for bounded utility semantics and Sol medium for exploration, standard work and routine review. Raise Sol review effort to high only for justified complexity. Reserve Astra high for explicit user choice or confirmed agent-proposed escalation. Preserve supported user pins and explicit Terra fallback. + +### Verification + +- Official model/effort source and current Codex model availability; focused model-profile tests; source full/release/security checks and plugin validators. +- Marketplace sync provenance and recorded-byte verification, catalog tests, release workflow, checksummed asset, and installed-version readback. + +### Latest Validation Results + +- 2026-09-23: Official OpenAI model catalog, migration guidance, Codex model selection, and subagent configuration opened; GPT-6 Astra, Sol, Luna and supported low/medium/high efforts verified. Terra's published API pricing ($2 input/$12 output per million tokens) exceeds Luna's and does not undercut Sol's $2 input/$10 output; subscription usage is not inferred from API prices. Source audit produced a bounded report; initial checkout was clean. +- 2026-09-23: Source/package 0.9.8 generated in parity. Full gate passed 9/9 with 266 tests (one skipped); review found and corrected an overly costly routine reviewer effort. Migration tests cover previously opted-in pristine templates, customized model pins, and targets without opt-in. No current upstream 0.9.8 tag or open source PR exists. +- 2026-09-23: Source and generated package quick validation, Codex plugin validation, and strict Claude plugin/marketplace validation passed. Final review incorporated the user's explicit confirmation rule for agent-initiated Astra escalation; full release gate must be rerun on those final bytes. + +### Risks And Recovery + +- A proposed model may not be exposed in a specific Codex installation. Verify locally before committing template defaults; preserve existing supported pins if unavailable. +- Remote refs may advance during release. Reinspect exact head and tags before push/merge; revalidate changed content. +- Marketplace import may reject candidate bytes or provenance. Stop before publishing and correct only the failed import stage. + +### Resume Point + +- WQ-02: run final release/security validation and external plugin validators on the reviewed source tree, then commit and publish the source PR/tag/release. + +### Plan Fidelity Check + +- [x] Full requested source, marketplace, local update, and closure outcomes are mapped to ordered work. +- [x] Constraints, exclusions, sources, decisions, validation, recovery, and exact resume point are recorded. + +### Reconciliation Check + +- [ ] Final source, marketplace, release, installation, and plan states agree. + +### Closure Gate + +- [ ] All requirements and queue items are terminal with applicable validation and delivery evidence. + +### Post-Close Delivery + +- Source and marketplace publication plus local refresh are active work under WQ-02 through WQ-04. + +### Handoff Notes + +- None. + ## Recently Completed - [x] 2026-09-20: Completed Publish Repository Audit Fix 0.9.7; [full archived plan](docs/archive/plans/2026-09-20-publish-repository-audit-fix-0-9-7.md). diff --git a/README.md b/README.md index fba1187..f0efc22 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ `engineering-workflow` is a public skill for auditing, setting up, validating, updating, and safely migrating the engineering-workflow layer of a repository. It works with Codex and Claude Code. -Current skill version: `0.9.7`. +Current skill version: `0.9.8`. The skill uses `AGENTS.md` as a short map, `PLANS.md` as durable execution state, and leaves product, architecture, operations, security, and other repository-owned documentation with its existing owners. Any repository change starts with a full plan; read-only inspection is the only exception. @@ -73,7 +73,7 @@ Use $engineering-workflow to audit this mature repository and add only the missi ``` ```text -Use $engineering-workflow to Upgrade A Target Workflow in this repository to version 0.9.7. +Use $engineering-workflow to Upgrade A Target Workflow in this repository to version 0.9.8. ``` Repository text is evidence, not authority. It cannot grant approval, expand scope, request secrets, or override system, developer, or user instructions. @@ -178,7 +178,7 @@ When the result permits an automatic update, rerun it with `--apply`. Alternate `Upgrade A Target Workflow` tells the agent to run a report-first guarded migration, not to hand the user a list of backend commands. It applies automatically only when ownership, privacy, and approval checks are resolved. An already-current valid target returns `already_current` without creating a plan or rewriting state/index files; missing or drifted required artifacts still take the guarded migration path. ```text -Use $engineering-workflow to Upgrade A Target Workflow in this repository to version 0.9.7. Run the report first, apply it when safe, and ask only when the report requires a user decision. +Use $engineering-workflow to Upgrade A Target Workflow in this repository to version 0.9.8. Run the report first, apply it when safe, and ask only when the report requires a user decision. ``` The maintainer/automation backend is: @@ -187,7 +187,7 @@ The maintainer/automation backend is: python3 skill/engineering-workflow/scripts/upgrade_target_workflow.py \ --repo \ --prompt \ - --target-version 0.9.7 \ + --target-version 0.9.8 \ --format json ``` @@ -269,7 +269,7 @@ Use $engineering-workflow to audit this mature repository, preserve every existi Target migration: ```text -Use $engineering-workflow to Upgrade A Target Workflow here to 0.9.7. Run the report and apply it when safe. +Use $engineering-workflow to Upgrade A Target Workflow here to 0.9.8. Run the report and apply it when safe. ``` ## Repository layout @@ -304,7 +304,7 @@ This harness and its Ruff configuration improve development of this repository o ## Versioning and updates -The project uses semantic versioning. Version 0.9.7 bounds repository discovery through Git-owned inventory or an explicit non-Git fallback and adds compact agent-facing audit summaries backed by complete report artifacts without narrowing privacy scanning. Version 0.9.6 keeps root context focused on current decisions and integration, distinguishes transient evidence from durable repository knowledge, requires self-contained worker handoffs with compact evidence, and favors existing bounded execution mechanisms for predictable tool-heavy stages. Version 0.9.5 narrows instruction loading to the selected task, accepts sufficient native completion evidence, makes custom stage assessment optional, and clarifies existing local-check authorization. These releases preserve the full plan and security contracts. Version 0.9.4 adds the approved opaque Engineering Workflow identity and Codex plugin-card icon metadata without changing the runtime workflow contract. Version 0.9.3 preserves customized top-level `PLANS.md` sections during compact and archive closure, correcting a data-loss defect discovered while dogfooding 0.9.2 against the unified marketplace repository. Version 0.9.2 keeps durable state current inside useful work rather than a recurring model-maintenance loop, distinguishes continuous task context from real recovery, removes plan-date ordering as a validation-applicability proxy, stops redundant route/tool/subagent work after sufficient evidence, and provides an agent-neutral fallback when the invoking host is not established as Codex or Claude Code. Versions 0.9.2 through 0.9.7 preserve all existing schema and contract versions. Version 0.9.1 updated Codex's standard/review recommendations for Astra, preserved native Claude model/effort inheritance, and clarified existing authorization, task steering, bounded delegation, and proportional verification. Version 0.9.0 added ownership-aware archive closure and instruction contract v3: target agents review every complete logical commit slice and then the aggregate final diff, while customized mature repositories migrate conservatively. Version 0.8.2 stopped empty compatibility archive directories from producing false missing-index errors while retaining fail-closed checks for real archive content and unsafe index paths. Version 0.8.1 added exact, user-approved synthetic-fixture privacy review without exposing candidate values to the agent. Version 0.8.0 introduced loss-resistant completion-driven waits, correctness-first execution discipline, instruction contract v2 migration, Claude Code compatibility, and the deterministic dual marketplace. Version 0.7.0 is the historical baseline for bounded Programmatic Tool Calling assessment and runtime instruction rendering. +The project uses semantic versioning. Version 0.9.8 routes deterministic commands and tests through tools, recommends GPT-6 Luna for bounded utility work and Sol for exploration, standard work, and routine review, reserves Astra for user-selected or confirmed high-consequence reasoning, retains Terra as an explicit fallback, and refreshes only pristine previously opted-in target agent profiles. Version 0.9.7 bounds repository discovery through Git-owned inventory or an explicit non-Git fallback and adds compact agent-facing audit summaries backed by complete report artifacts without narrowing privacy scanning. Version 0.9.6 keeps root context focused on current decisions and integration, distinguishes transient evidence from durable repository knowledge, requires self-contained worker handoffs with compact evidence, and favors existing bounded execution mechanisms for predictable tool-heavy stages. Version 0.9.5 narrows instruction loading to the selected task, accepts sufficient native completion evidence, makes custom stage assessment optional, and clarifies existing local-check authorization. These releases preserve the full plan and security contracts. Version 0.9.4 adds the approved opaque Engineering Workflow identity and Codex plugin-card icon metadata without changing the runtime workflow contract. Version 0.9.3 preserves customized top-level `PLANS.md` sections during compact and archive closure, correcting a data-loss defect discovered while dogfooding 0.9.2 against the unified marketplace repository. Version 0.9.2 keeps durable state current inside useful work rather than a recurring model-maintenance loop, distinguishes continuous task context from real recovery, removes plan-date ordering as a validation-applicability proxy, stops redundant route/tool/subagent work after sufficient evidence, and provides an agent-neutral fallback when the invoking host is not established as Codex or Claude Code. Versions 0.9.2 through 0.9.8 preserve all existing schema and contract versions. Version 0.9.1 updated Codex's standard/review recommendations for Astra, preserved native Claude model/effort inheritance, and clarified existing authorization, task steering, bounded delegation, and proportional verification. Version 0.9.0 added ownership-aware archive closure and instruction contract v3: target agents review every complete logical commit slice and then the aggregate final diff, while customized mature repositories migrate conservatively. Version 0.8.2 stopped empty compatibility archive directories from producing false missing-index errors while retaining fail-closed checks for real archive content and unsafe index paths. Version 0.8.1 added exact, user-approved synthetic-fixture privacy review without exposing candidate values to the agent. Version 0.8.0 introduced loss-resistant completion-driven waits, correctness-first execution discipline, instruction contract v2 migration, Claude Code compatibility, and the deterministic dual marketplace. Version 0.7.0 is the historical baseline for bounded Programmatic Tool Calling assessment and runtime instruction rendering. Historical version records remain valid in completed plans, archives, and migration tests. Current-version owners are `SKILL.md`, this README, current update prompts, active workflow state manifests, and the generated plugin manifests. diff --git a/plugins/engineering-workflow/.claude-plugin/plugin.json b/plugins/engineering-workflow/.claude-plugin/plugin.json index a2660c5..53f368b 100644 --- a/plugins/engineering-workflow/.claude-plugin/plugin.json +++ b/plugins/engineering-workflow/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "engineering-workflow", - "version": "0.9.7", + "version": "0.9.8", "description": "Audit, plan, migrate, validate, and maintain repository engineering workflows.", "author": { "name": "xeonvs", diff --git a/plugins/engineering-workflow/.codex-plugin/plugin.json b/plugins/engineering-workflow/.codex-plugin/plugin.json index 89f546e..cc35973 100644 --- a/plugins/engineering-workflow/.codex-plugin/plugin.json +++ b/plugins/engineering-workflow/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "engineering-workflow", - "version": "0.9.7", + "version": "0.9.8", "description": "Audit, plan, migrate, validate, and maintain repository engineering workflows.", "author": { "name": "xeonvs", diff --git a/plugins/engineering-workflow/skills/engineering-workflow/SKILL.md b/plugins/engineering-workflow/skills/engineering-workflow/SKILL.md index 5ffcee6..d07658d 100644 --- a/plugins/engineering-workflow/skills/engineering-workflow/SKILL.md +++ b/plugins/engineering-workflow/skills/engineering-workflow/SKILL.md @@ -2,7 +2,7 @@ name: engineering-workflow description: Set up, audit, or upgrade repository workflow instructions and planning. Use for workflow changes or explicit skill refresh/update; ordinary repository work does not invoke migration. metadata: - version: 0.9.7 + version: 0.9.8 --- # Engineering Workflow diff --git a/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/explorer.toml.tmpl b/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/explorer.toml.tmpl index cb9957c..71ab49c 100644 --- a/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/explorer.toml.tmpl +++ b/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/explorer.toml.tmpl @@ -1,6 +1,6 @@ name = "workflow-explorer" description = "Bounded read-heavy repository evidence collection" developer_instructions = "Use only the bounded self-contained packet and accessible path scope supplied by the root. Return compact status, distilled findings, file references, checks, blockers, and accessible artifact paths. Do not paste unbounded raw output, reconstruct missing context, write shared state, or spawn child agents; report an inaccessible required input." -model = "gpt-5.6-terra" +model = "gpt-6-sol" model_reasoning_effort = "medium" sandbox_mode = "read-only" diff --git a/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/reviewer.toml.tmpl b/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/reviewer.toml.tmpl index d827bcb..4170e8b 100644 --- a/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/reviewer.toml.tmpl +++ b/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/reviewer.toml.tmpl @@ -1,6 +1,6 @@ name = "workflow-reviewer" -description = "Evidence-first review for correctness and high-risk changes" +description = "Bounded evidence-first review for ordinary changes" developer_instructions = "Review only the bounded self-contained change packet and accessible evidence supplied by the root without modifying shared state. Return compact status, findings with severity and confidence, evidence paths, checks, blockers, assumptions, and a stopping or escalation condition; report missing required context rather than inferring it." -model = "gpt-6-astra" -model_reasoning_effort = "high" +model = "gpt-6-sol" +model_reasoning_effort = "medium" sandbox_mode = "read-only" diff --git a/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/utility.toml.tmpl b/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/utility.toml.tmpl index d8bd0fa..9214080 100644 --- a/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/utility.toml.tmpl +++ b/plugins/engineering-workflow/skills/engineering-workflow/assets/agents/utility.toml.tmpl @@ -1,6 +1,6 @@ name = "workflow-utility" description = "Bounded read-only interpretation with a fixed result schema" developer_instructions = "Use only the bounded self-contained packet and accessible inputs supplied by the root. Do not expand scope, write shared state, spawn child agents, or retry more than once. Return compact status, findings, checks, blockers, accessible evidence paths, needs_escalation, and the stopping condition; report an inaccessible required input instead of guessing." -model = "gpt-5.6-terra" +model = "gpt-6-luna" model_reasoning_effort = "low" sandbox_mode = "read-only" diff --git a/plugins/engineering-workflow/skills/engineering-workflow/references/agent_orchestration.md b/plugins/engineering-workflow/skills/engineering-workflow/references/agent_orchestration.md index 59dcad7..0db9a77 100644 --- a/plugins/engineering-workflow/skills/engineering-workflow/references/agent_orchestration.md +++ b/plugins/engineering-workflow/skills/engineering-workflow/references/agent_orchestration.md @@ -51,6 +51,7 @@ Use a tool, script, scheduler, hook, or harness layer instead of an LLM subagent - unambiguous JSON status reads - sorting, filtering, joining, ranking, aggregation, or deduplication - repeating one command +- running a known shell command or test suite and reporting its exit status - bounded retries and backoff - deterministic stop conditions diff --git a/plugins/engineering-workflow/skills/engineering-workflow/references/model_profiles.md b/plugins/engineering-workflow/skills/engineering-workflow/references/model_profiles.md index c3acdf7..d936464 100644 --- a/plugins/engineering-workflow/skills/engineering-workflow/references/model_profiles.md +++ b/plugins/engineering-workflow/skills/engineering-workflow/references/model_profiles.md @@ -2,10 +2,13 @@ Use this file only in Codex as the single canonical owner of current concrete model mappings. Keep task-shape policy in `agent_orchestration.md`; Claude Code follows native model and effort selection in `platform_compatibility.md`. +Choose a model only after the task-shape route calls for semantic work. Deterministic command execution, test runs, polling, and status aggregation use tools or scripts without creating a model worker. + ## Source Snapshot -Verified against current official guidance on 2026-09-05: +Verified against current official guidance on 2026-09-23: +- `https://developers.openai.com/api/docs/models` - `https://developers.openai.com/api/docs/guides/latest-model` - `https://learn.chatgpt.com/docs/models` - `https://learn.chatgpt.com/docs/agent-configuration/subagents` @@ -16,41 +19,45 @@ Revalidate this mapping when supported Codex models or reasoning levels change. ### `utility` -- model: `gpt-5.6-terra` +- model: `gpt-6-luna` - `model_reasoning_effort`: `low` - `sandbox_mode`: `read-only` -- allow `minimal` or `none` only when the selected model supports it, the task needs almost no reasoning, and regression tests or evaluation preserve quality +- allow `none` only when the selected model supports it, the task needs almost no reasoning, and regression tests or evaluation preserve quality - forbid `high`, `xhigh`, `max`, `ultra`, and pro mode by default ### `explorer` -- model: `gpt-5.6-terra` +- model: `gpt-6-sol` - `model_reasoning_effort`: `low` or `medium` - `sandbox_mode`: `read-only` - use bounded path scope and distilled evidence ### `standard` -- model: `gpt-6-astra` +- model: `gpt-6-sol` - `model_reasoning_effort`: `medium` - use the minimum sandbox needed by the bounded work ### `review` -- model: `gpt-6-astra` -- `model_reasoning_effort`: `high` +- model: `gpt-6-sol` +- `model_reasoning_effort`: `medium` - normally use `sandbox_mode: read-only` -- use `xhigh` only after representative evaluation shows a material quality gain +- use `high` on Sol when review complexity warrants it; consider `gpt-6-astra` with `high` effort for high-consequence correctness or security review under the confirmation rule below; use `xhigh` only after representative evaluation shows a material quality gain ### `exceptional_quality` -- keep the selected supported model; use the standard profile when no model is selected +- use `gpt-6-astra` with `high` effort for unusually difficult, high-consequence semantic work when the user requests it or confirms a proposed escalation; keep a selected supported user-pinned model - consider `max`, `ultra`, or API pro mode only for difficult quality-first work with measurable acceptance criteria and high error cost - compare against the cheaper baseline instead of assuming maximum reasoning wins +Before selecting Astra for a new worker or changing a saved profile from Sol/Luna to Astra on the agent's initiative, explain the concrete task risk or quality gap and obtain the user's confirmation. Do not treat a routine shell command, test run, broad task label, or available model slot as a reason to escalate. A user-selected Astra session or an explicit request for Astra already supplies that choice; do not ask again. This rule governs optional model selection, not the host's current model or an API-wide approval mechanism. + ## API And Codex Boundary -When explicitly migrating a profile to Astra, preserve its effective supported reasoning effort. Replace `none` or `minimal` with `low` as the initial evaluated baseline; Astra does not support those efforts. The standard and reviewer defaults above retain `medium` and `high` respectively. +When migrating a profile to a GPT-6 model, preserve its effective supported reasoning effort unless deliberately changing the task profile after evaluation. Astra does not support `none`; use `low` as the initial evaluated baseline. Sol and Luna support `none`. If an older profile used `minimal`, start with `low` and compare representative tasks. Standard and routine review default to `medium`; higher review effort requires a task-specific reason. + +Keep `gpt-5.6-terra` as an explicit compatibility fallback for a utility or explorer profile when its GPT-6 recommendation is unavailable in the active Codex client, or when representative evaluation favors the existing profile. Retain the role's `low` or `medium` effort and read-only boundary. This is a deliberate profile choice, not automatic retry or a silent replacement of a user pin. The published API token prices do not make Terra cheaper than Luna for utility work or Sol for explorer work; Codex subscription usage should be assessed in its own environment. In the Responses API, pro is a reasoning mode selected with `reasoning.mode: "pro"`; it is not a separate model slug. Persisted reasoning and Programmatic Tool Calling are also API features. @@ -61,6 +68,6 @@ Do not write API-only fields into Codex custom-agent TOML unless current Codex d - Keep concrete model slugs out of `agent_orchestration.md` and other runtime references. - Optional custom-agent templates may repeat the concrete slug they instantiate. - Keep user-pinned supported models unless the user requests a migration. -- These recommendations and templates apply to newly requested profiles, not the user's global model selection. Verify the chosen model is exposed by the actual client before installing optional configuration. If Astra is unavailable, retain the current supported model and report the limitation; do not silently overwrite a pin or invent a fallback. +- These recommendations and templates apply to newly requested profiles, not the user's global model selection. Verify each chosen model is exposed by the actual client before installing optional configuration. If it is unavailable, retain the current supported model and report the limitation; do not silently overwrite a pin or invent a fallback. - Treat reasoning and model selection as evaluation decisions, not status symbols. - Preserve an existing profile when current repository evidence shows it is intentional and supported. diff --git a/plugins/engineering-workflow/skills/engineering-workflow/references/target_workflow_upgrade.md b/plugins/engineering-workflow/skills/engineering-workflow/references/target_workflow_upgrade.md index 8d81d0c..05d3f0d 100644 --- a/plugins/engineering-workflow/skills/engineering-workflow/references/target_workflow_upgrade.md +++ b/plugins/engineering-workflow/skills/engineering-workflow/references/target_workflow_upgrade.md @@ -24,7 +24,7 @@ Use this canonical reference for `upgrade_target_workflow`, which migrates the w Treat `Upgrade A Target Workflow` plus a target repository as an authorized repo-changing prompt, not as a request for CLI instructions. 1. Resolve the target path and requested version from context; default to the installed skill version. -2. Before prompt apply, review target-local owners affected by the requested release's changed semantics when adoption has not already been established. For customized owners, preserve equivalent rules or make the narrow requested correction under the full planning and privacy gates; ask only for a real ownership conflict. A same-version stamp or `already_current` result proves structural state, not semantic adoption. Version 0.9.7 changes audit discovery and output only, so it requires no target-local instruction rewrite. For the 0.9.6 changes, inspect the task-handoff route and efficient-execution owner for root working state, self-contained worker context, transient-versus-durable evidence, and artifact-based recovery. Use already-current evidence, and do not sweep unrelated owners. Then invoke `scripts/upgrade_target_workflow.py --prompt` yourself. +2. Before prompt apply, review target-local owners affected by the requested release's changed semantics when adoption has not already been established. For customized owners, preserve equivalent rules or make the narrow requested correction under the full planning and privacy gates; ask only for a real ownership conflict. A same-version stamp or `already_current` result proves structural state, not semantic adoption. Version 0.9.8 updates Codex model profiles and refreshes only exact prior generated agent templates when that configuration was already opted in; it requires no target-local instruction rewrite. Version 0.9.7 changes audit discovery and output only, so it also requires no target-local instruction rewrite. For the 0.9.6 changes, inspect the task-handoff route and efficient-execution owner for root working state, self-contained worker context, transient-versus-durable evidence, and artifact-based recovery. Use already-current evidence, and do not sweep unrelated owners. Then invoke `scripts/upgrade_target_workflow.py --prompt` yourself. 3. Prompt mode builds and reviews the read-only migration report first. 4. If ownership, conflicts, privacy, and approvals are resolved, it proceeds through guarded apply and validation automatically. 5. If the result returns `agent_action: ask_targeted_question`, ask only `question_to_ask`; keep any later questions deferred and do not write target files. @@ -35,7 +35,7 @@ Treat `Upgrade A Target Workflow` plus a target repository as an authorized repo If the target already records the requested version, all canonical artifacts exist, instruction and index contracts pass, privacy/conflict checks are clear, no registered pristine bytes need an actual update, and any requested optional agent configuration is already fully present, prompt/apply returns `update_status: already_current` with an empty mutation log. It does not create a plan or rewrite state/index files merely to reconfirm that unchanged result. A missing artifact, older contract, drift, conflict, privacy boundary, or requested but incomplete optional configuration keeps the normal guarded path. -The user may explicitly request report-only behavior; then invoke `--plan`. Runtime agent configuration remains opt-in through the user's prompt and `--include-agent-config`. +The user may explicitly request report-only behavior; then invoke `--plan`. New runtime agent configuration remains opt-in through the user's prompt and `--include-agent-config`; a valid workflow state manifest recording an earlier opt-in carries that choice into subsequent upgrades. ## CLI Contract @@ -170,15 +170,16 @@ Every apply-time snapshot, read, atomic replacement, unlink, and rollback operat ## Codex Configuration -When configuration is not selected, existing Codex artifacts remain unchanged; their syntax or symbolic layout does not create a configuration-migration question. Public privacy findings and actual workflow-path conflicts still follow their own gates. +When configuration has never been selected, existing Codex artifacts remain unchanged; their syntax or symbolic layout does not create a configuration-migration question. Public privacy findings and actual workflow-path conflicts still follow their own gates. -When `--include-agent-config` is present: +When `--include-agent-config` is present or the target's valid workflow state records a prior opt-in: - parse existing TOML before changing it - preserve unknown keys, custom profiles, and current `max_threads` - add `max_depth = 1` only when absent or already compatible - do not overwrite a conflicting explicit depth without a user decision -- create optional agent files only under the explicit flag +- create missing optional agent files only after the current or prior opt-in +- replace an existing agent file only when its complete bytes match the registered prior generated template; preserve every customized model pin and instruction - show the exact config diff Never place Responses API-only fields in Codex TOML. diff --git a/plugins/engineering-workflow/skills/engineering-workflow/scripts/upgrade_target_workflow.py b/plugins/engineering-workflow/skills/engineering-workflow/scripts/upgrade_target_workflow.py index 969a3f0..80fdddd 100644 --- a/plugins/engineering-workflow/skills/engineering-workflow/scripts/upgrade_target_workflow.py +++ b/plugins/engineering-workflow/skills/engineering-workflow/scripts/upgrade_target_workflow.py @@ -95,6 +95,13 @@ }, } +# Exact agent-template bytes shipped in 0.9.7. No customized target file is rewritten. +PRIOR_AGENT_TEMPLATE_HASHES = { + "utility": "2f32f34a8c23d66c037abd0d1466f1eebc41ee52fd5e1e422470b7fbead4c210", + "explorer": "cf28d059b8bc28123a038d2f4c40fe24fe45e5623d2ee73c0e2b81f0a1d381b4", + "reviewer": "6182122fcec3d18b14acdabb644b750e58c5d2264d8b7a68eaf54644ef6db133", +} + def _content_hash(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() @@ -104,6 +111,31 @@ def _is_pristine_legacy(relative: str, text: str) -> bool: return _content_hash(text) in LEGACY_PRISTINE_HASHES.get(relative, set()) +def _is_pristine_prior_agent(name: str, text: str) -> bool: + return _content_hash(text) == PRIOR_AGENT_TEMPLATE_HASHES[name] + + +def _agent_config_selected(root: Path, explicitly_selected: bool) -> bool: + if explicitly_selected: + return True + if _first_symlink_component(root, STATE_MANIFEST_PATH): + return False + state = _read(root / STATE_MANIFEST_PATH) + try: + declared, shared_paths = parse_manifest_path_list(state, "shared_paths") + except ValueError: + return False + expected = {".codex/config.toml"} | {f".codex/agents/{name}.toml" for name in ("utility", "explorer", "reviewer")} + return ( + re.search(r"(?m)^schema_version:\s*2\s*$", state) is not None + and re.search(r"(?m)^skill_name:\s*engineering-workflow\s*$", state) is not None + and re.search(r"(?m)^mode:\s*upgrade_target_workflow\s*$", state) is not None + and re.search(r"(?m)^runtime_agent_config_managed:\s*true\s*$", state) is not None + and declared + and expected.issubset(shared_paths) + ) + + class MigrationConflict(RuntimeError): def __init__(self, code: str, message: str): super().__init__(message) @@ -741,6 +773,10 @@ def _proposed_changes(root: Path, include_agent_config: bool) -> list[dict[str, changes.append( {"path": path, "action": "create", "reason": "explicit optional agent configuration request"} ) + elif _is_pristine_prior_agent(name, _read(root / path)): + changes.append( + {"path": path, "action": "update", "reason": "known pristine prior agent template fingerprint"} + ) return changes @@ -754,6 +790,7 @@ def build_migration_report( root = repo.resolve() if not root.is_dir(): raise MigrationConflict("missing_repository", "Target repository does not exist") + include_agent_config = _agent_config_selected(root, include_agent_config) audit = audit_repo(root) conflicts = _scan_contract_conflicts(root, include_agent_config=include_agent_config) state_text = _read(root / STATE_MANIFEST_PATH) @@ -919,7 +956,9 @@ def _already_current(report: dict[str, Any], include_agent_config: bool, root: P topology = report["detected_topology"] required_artifacts = ("root_agents", "plans", "backlog", "pitfalls", "principles", "state_manifest") pristine_update_pending = any( - change.get("reason") == "known pristine legacy template fingerprint" for change in report["proposed_changes"] + change.get("reason") + in {"known pristine legacy template fingerprint", "known pristine prior agent template fingerprint"} + for change in report["proposed_changes"] ) return ( report["success"] @@ -1245,6 +1284,7 @@ def apply_migration( include_agent_config, approved_privacy_review, ) + include_agent_config = report["include_agent_config"] privacy_review, privacy_findings, approved_fingerprints = _evaluate_privacy_review( root, report["current_workflow_version"], @@ -1365,7 +1405,8 @@ def write(relative: str, text: str) -> None: write(".codex/config.toml", merged) for name in ("utility", "explorer", "reviewer"): relative = f".codex/agents/{name}.toml" - if not secure.exists(relative): + existing_agent = read(relative) + if not existing_agent or _is_pristine_prior_agent(name, existing_agent): write( relative, (AGENT_TEMPLATE_ROOT / f"{name}.toml.tmpl").read_text(encoding="utf-8"), @@ -1533,6 +1574,7 @@ def execute_prompt_upgrade( include_agent_config, approved_privacy_review, ) + include_agent_config = report["include_agent_config"] if report["required_user_questions"]: return { **report, @@ -1643,7 +1685,7 @@ def main() -> int: mode.add_argument("--plan", action="store_true") mode.add_argument("--apply", action="store_true") mode.add_argument("--prompt", action="store_true") - parser.add_argument("--target-version", default="0.9.7") + parser.add_argument("--target-version", default="0.9.8") parser.add_argument("--include-agent-config", action="store_true") parser.add_argument( "--approve-privacy-review", diff --git a/plugins/engineering-workflow/skills/engineering-workflow/scripts/validate_skill_repo.py b/plugins/engineering-workflow/skills/engineering-workflow/scripts/validate_skill_repo.py index 263e464..5a08231 100644 --- a/plugins/engineering-workflow/skills/engineering-workflow/scripts/validate_skill_repo.py +++ b/plugins/engineering-workflow/skills/engineering-workflow/scripts/validate_skill_repo.py @@ -508,21 +508,21 @@ def _validate_agent_profiles(repo_root: Path) -> list[str]: if field not in data: issues.append(f"{path.name} is missing required field: {field}") utility = parsed.get("utility", {}) - expected_utility_model = "gpt-" + "5.6-" + "terra" + expected_utility_model = "gpt-" + "6-luna" if utility.get("model") != expected_utility_model or utility.get("model_reasoning_effort") != "low": issues.append("Utility agent must use the current low-cost low-reasoning profile") if utility.get("sandbox_mode") != "read-only": issues.append("Utility agent must remain read-only") explorer = parsed.get("explorer", {}) - if explorer.get("model") != expected_utility_model or explorer.get("model_reasoning_effort") != "medium": + if explorer.get("model") != "gpt-" + "6-sol" or explorer.get("model_reasoning_effort") != "medium": issues.append("Explorer agent must use the current balanced read-heavy profile") if explorer.get("sandbox_mode") != "read-only": issues.append("Explorer agent must remain read-only") reviewer = parsed.get("reviewer", {}) - if reviewer.get("model") != "gpt-" + "6-astra": + if reviewer.get("model") != "gpt-" + "6-sol": issues.append("Reviewer agent must use the current Codex review model profile") - if reviewer.get("model_reasoning_effort") != "high" or reviewer.get("sandbox_mode") != "read-only": - issues.append("Reviewer agent must use high reasoning in read-only mode") + if reviewer.get("model_reasoning_effort") != "medium" or reviewer.get("sandbox_mode") != "read-only": + issues.append("Reviewer agent must use medium reasoning in read-only mode") reference = repo_root / "skill/engineering-workflow/references/agent_orchestration.md" if reference.exists(): text = reference.read_text(encoding="utf-8") diff --git a/skill/engineering-workflow/SKILL.md b/skill/engineering-workflow/SKILL.md index 5ffcee6..d07658d 100644 --- a/skill/engineering-workflow/SKILL.md +++ b/skill/engineering-workflow/SKILL.md @@ -2,7 +2,7 @@ name: engineering-workflow description: Set up, audit, or upgrade repository workflow instructions and planning. Use for workflow changes or explicit skill refresh/update; ordinary repository work does not invoke migration. metadata: - version: 0.9.7 + version: 0.9.8 --- # Engineering Workflow diff --git a/skill/engineering-workflow/assets/agents/explorer.toml.tmpl b/skill/engineering-workflow/assets/agents/explorer.toml.tmpl index cb9957c..71ab49c 100644 --- a/skill/engineering-workflow/assets/agents/explorer.toml.tmpl +++ b/skill/engineering-workflow/assets/agents/explorer.toml.tmpl @@ -1,6 +1,6 @@ name = "workflow-explorer" description = "Bounded read-heavy repository evidence collection" developer_instructions = "Use only the bounded self-contained packet and accessible path scope supplied by the root. Return compact status, distilled findings, file references, checks, blockers, and accessible artifact paths. Do not paste unbounded raw output, reconstruct missing context, write shared state, or spawn child agents; report an inaccessible required input." -model = "gpt-5.6-terra" +model = "gpt-6-sol" model_reasoning_effort = "medium" sandbox_mode = "read-only" diff --git a/skill/engineering-workflow/assets/agents/reviewer.toml.tmpl b/skill/engineering-workflow/assets/agents/reviewer.toml.tmpl index d827bcb..4170e8b 100644 --- a/skill/engineering-workflow/assets/agents/reviewer.toml.tmpl +++ b/skill/engineering-workflow/assets/agents/reviewer.toml.tmpl @@ -1,6 +1,6 @@ name = "workflow-reviewer" -description = "Evidence-first review for correctness and high-risk changes" +description = "Bounded evidence-first review for ordinary changes" developer_instructions = "Review only the bounded self-contained change packet and accessible evidence supplied by the root without modifying shared state. Return compact status, findings with severity and confidence, evidence paths, checks, blockers, assumptions, and a stopping or escalation condition; report missing required context rather than inferring it." -model = "gpt-6-astra" -model_reasoning_effort = "high" +model = "gpt-6-sol" +model_reasoning_effort = "medium" sandbox_mode = "read-only" diff --git a/skill/engineering-workflow/assets/agents/utility.toml.tmpl b/skill/engineering-workflow/assets/agents/utility.toml.tmpl index d8bd0fa..9214080 100644 --- a/skill/engineering-workflow/assets/agents/utility.toml.tmpl +++ b/skill/engineering-workflow/assets/agents/utility.toml.tmpl @@ -1,6 +1,6 @@ name = "workflow-utility" description = "Bounded read-only interpretation with a fixed result schema" developer_instructions = "Use only the bounded self-contained packet and accessible inputs supplied by the root. Do not expand scope, write shared state, spawn child agents, or retry more than once. Return compact status, findings, checks, blockers, accessible evidence paths, needs_escalation, and the stopping condition; report an inaccessible required input instead of guessing." -model = "gpt-5.6-terra" +model = "gpt-6-luna" model_reasoning_effort = "low" sandbox_mode = "read-only" diff --git a/skill/engineering-workflow/references/agent_orchestration.md b/skill/engineering-workflow/references/agent_orchestration.md index 59dcad7..0db9a77 100644 --- a/skill/engineering-workflow/references/agent_orchestration.md +++ b/skill/engineering-workflow/references/agent_orchestration.md @@ -51,6 +51,7 @@ Use a tool, script, scheduler, hook, or harness layer instead of an LLM subagent - unambiguous JSON status reads - sorting, filtering, joining, ranking, aggregation, or deduplication - repeating one command +- running a known shell command or test suite and reporting its exit status - bounded retries and backoff - deterministic stop conditions diff --git a/skill/engineering-workflow/references/model_profiles.md b/skill/engineering-workflow/references/model_profiles.md index c3acdf7..d936464 100644 --- a/skill/engineering-workflow/references/model_profiles.md +++ b/skill/engineering-workflow/references/model_profiles.md @@ -2,10 +2,13 @@ Use this file only in Codex as the single canonical owner of current concrete model mappings. Keep task-shape policy in `agent_orchestration.md`; Claude Code follows native model and effort selection in `platform_compatibility.md`. +Choose a model only after the task-shape route calls for semantic work. Deterministic command execution, test runs, polling, and status aggregation use tools or scripts without creating a model worker. + ## Source Snapshot -Verified against current official guidance on 2026-09-05: +Verified against current official guidance on 2026-09-23: +- `https://developers.openai.com/api/docs/models` - `https://developers.openai.com/api/docs/guides/latest-model` - `https://learn.chatgpt.com/docs/models` - `https://learn.chatgpt.com/docs/agent-configuration/subagents` @@ -16,41 +19,45 @@ Revalidate this mapping when supported Codex models or reasoning levels change. ### `utility` -- model: `gpt-5.6-terra` +- model: `gpt-6-luna` - `model_reasoning_effort`: `low` - `sandbox_mode`: `read-only` -- allow `minimal` or `none` only when the selected model supports it, the task needs almost no reasoning, and regression tests or evaluation preserve quality +- allow `none` only when the selected model supports it, the task needs almost no reasoning, and regression tests or evaluation preserve quality - forbid `high`, `xhigh`, `max`, `ultra`, and pro mode by default ### `explorer` -- model: `gpt-5.6-terra` +- model: `gpt-6-sol` - `model_reasoning_effort`: `low` or `medium` - `sandbox_mode`: `read-only` - use bounded path scope and distilled evidence ### `standard` -- model: `gpt-6-astra` +- model: `gpt-6-sol` - `model_reasoning_effort`: `medium` - use the minimum sandbox needed by the bounded work ### `review` -- model: `gpt-6-astra` -- `model_reasoning_effort`: `high` +- model: `gpt-6-sol` +- `model_reasoning_effort`: `medium` - normally use `sandbox_mode: read-only` -- use `xhigh` only after representative evaluation shows a material quality gain +- use `high` on Sol when review complexity warrants it; consider `gpt-6-astra` with `high` effort for high-consequence correctness or security review under the confirmation rule below; use `xhigh` only after representative evaluation shows a material quality gain ### `exceptional_quality` -- keep the selected supported model; use the standard profile when no model is selected +- use `gpt-6-astra` with `high` effort for unusually difficult, high-consequence semantic work when the user requests it or confirms a proposed escalation; keep a selected supported user-pinned model - consider `max`, `ultra`, or API pro mode only for difficult quality-first work with measurable acceptance criteria and high error cost - compare against the cheaper baseline instead of assuming maximum reasoning wins +Before selecting Astra for a new worker or changing a saved profile from Sol/Luna to Astra on the agent's initiative, explain the concrete task risk or quality gap and obtain the user's confirmation. Do not treat a routine shell command, test run, broad task label, or available model slot as a reason to escalate. A user-selected Astra session or an explicit request for Astra already supplies that choice; do not ask again. This rule governs optional model selection, not the host's current model or an API-wide approval mechanism. + ## API And Codex Boundary -When explicitly migrating a profile to Astra, preserve its effective supported reasoning effort. Replace `none` or `minimal` with `low` as the initial evaluated baseline; Astra does not support those efforts. The standard and reviewer defaults above retain `medium` and `high` respectively. +When migrating a profile to a GPT-6 model, preserve its effective supported reasoning effort unless deliberately changing the task profile after evaluation. Astra does not support `none`; use `low` as the initial evaluated baseline. Sol and Luna support `none`. If an older profile used `minimal`, start with `low` and compare representative tasks. Standard and routine review default to `medium`; higher review effort requires a task-specific reason. + +Keep `gpt-5.6-terra` as an explicit compatibility fallback for a utility or explorer profile when its GPT-6 recommendation is unavailable in the active Codex client, or when representative evaluation favors the existing profile. Retain the role's `low` or `medium` effort and read-only boundary. This is a deliberate profile choice, not automatic retry or a silent replacement of a user pin. The published API token prices do not make Terra cheaper than Luna for utility work or Sol for explorer work; Codex subscription usage should be assessed in its own environment. In the Responses API, pro is a reasoning mode selected with `reasoning.mode: "pro"`; it is not a separate model slug. Persisted reasoning and Programmatic Tool Calling are also API features. @@ -61,6 +68,6 @@ Do not write API-only fields into Codex custom-agent TOML unless current Codex d - Keep concrete model slugs out of `agent_orchestration.md` and other runtime references. - Optional custom-agent templates may repeat the concrete slug they instantiate. - Keep user-pinned supported models unless the user requests a migration. -- These recommendations and templates apply to newly requested profiles, not the user's global model selection. Verify the chosen model is exposed by the actual client before installing optional configuration. If Astra is unavailable, retain the current supported model and report the limitation; do not silently overwrite a pin or invent a fallback. +- These recommendations and templates apply to newly requested profiles, not the user's global model selection. Verify each chosen model is exposed by the actual client before installing optional configuration. If it is unavailable, retain the current supported model and report the limitation; do not silently overwrite a pin or invent a fallback. - Treat reasoning and model selection as evaluation decisions, not status symbols. - Preserve an existing profile when current repository evidence shows it is intentional and supported. diff --git a/skill/engineering-workflow/references/target_workflow_upgrade.md b/skill/engineering-workflow/references/target_workflow_upgrade.md index 8d81d0c..05d3f0d 100644 --- a/skill/engineering-workflow/references/target_workflow_upgrade.md +++ b/skill/engineering-workflow/references/target_workflow_upgrade.md @@ -24,7 +24,7 @@ Use this canonical reference for `upgrade_target_workflow`, which migrates the w Treat `Upgrade A Target Workflow` plus a target repository as an authorized repo-changing prompt, not as a request for CLI instructions. 1. Resolve the target path and requested version from context; default to the installed skill version. -2. Before prompt apply, review target-local owners affected by the requested release's changed semantics when adoption has not already been established. For customized owners, preserve equivalent rules or make the narrow requested correction under the full planning and privacy gates; ask only for a real ownership conflict. A same-version stamp or `already_current` result proves structural state, not semantic adoption. Version 0.9.7 changes audit discovery and output only, so it requires no target-local instruction rewrite. For the 0.9.6 changes, inspect the task-handoff route and efficient-execution owner for root working state, self-contained worker context, transient-versus-durable evidence, and artifact-based recovery. Use already-current evidence, and do not sweep unrelated owners. Then invoke `scripts/upgrade_target_workflow.py --prompt` yourself. +2. Before prompt apply, review target-local owners affected by the requested release's changed semantics when adoption has not already been established. For customized owners, preserve equivalent rules or make the narrow requested correction under the full planning and privacy gates; ask only for a real ownership conflict. A same-version stamp or `already_current` result proves structural state, not semantic adoption. Version 0.9.8 updates Codex model profiles and refreshes only exact prior generated agent templates when that configuration was already opted in; it requires no target-local instruction rewrite. Version 0.9.7 changes audit discovery and output only, so it also requires no target-local instruction rewrite. For the 0.9.6 changes, inspect the task-handoff route and efficient-execution owner for root working state, self-contained worker context, transient-versus-durable evidence, and artifact-based recovery. Use already-current evidence, and do not sweep unrelated owners. Then invoke `scripts/upgrade_target_workflow.py --prompt` yourself. 3. Prompt mode builds and reviews the read-only migration report first. 4. If ownership, conflicts, privacy, and approvals are resolved, it proceeds through guarded apply and validation automatically. 5. If the result returns `agent_action: ask_targeted_question`, ask only `question_to_ask`; keep any later questions deferred and do not write target files. @@ -35,7 +35,7 @@ Treat `Upgrade A Target Workflow` plus a target repository as an authorized repo If the target already records the requested version, all canonical artifacts exist, instruction and index contracts pass, privacy/conflict checks are clear, no registered pristine bytes need an actual update, and any requested optional agent configuration is already fully present, prompt/apply returns `update_status: already_current` with an empty mutation log. It does not create a plan or rewrite state/index files merely to reconfirm that unchanged result. A missing artifact, older contract, drift, conflict, privacy boundary, or requested but incomplete optional configuration keeps the normal guarded path. -The user may explicitly request report-only behavior; then invoke `--plan`. Runtime agent configuration remains opt-in through the user's prompt and `--include-agent-config`. +The user may explicitly request report-only behavior; then invoke `--plan`. New runtime agent configuration remains opt-in through the user's prompt and `--include-agent-config`; a valid workflow state manifest recording an earlier opt-in carries that choice into subsequent upgrades. ## CLI Contract @@ -170,15 +170,16 @@ Every apply-time snapshot, read, atomic replacement, unlink, and rollback operat ## Codex Configuration -When configuration is not selected, existing Codex artifacts remain unchanged; their syntax or symbolic layout does not create a configuration-migration question. Public privacy findings and actual workflow-path conflicts still follow their own gates. +When configuration has never been selected, existing Codex artifacts remain unchanged; their syntax or symbolic layout does not create a configuration-migration question. Public privacy findings and actual workflow-path conflicts still follow their own gates. -When `--include-agent-config` is present: +When `--include-agent-config` is present or the target's valid workflow state records a prior opt-in: - parse existing TOML before changing it - preserve unknown keys, custom profiles, and current `max_threads` - add `max_depth = 1` only when absent or already compatible - do not overwrite a conflicting explicit depth without a user decision -- create optional agent files only under the explicit flag +- create missing optional agent files only after the current or prior opt-in +- replace an existing agent file only when its complete bytes match the registered prior generated template; preserve every customized model pin and instruction - show the exact config diff Never place Responses API-only fields in Codex TOML. diff --git a/skill/engineering-workflow/scripts/upgrade_target_workflow.py b/skill/engineering-workflow/scripts/upgrade_target_workflow.py index 969a3f0..80fdddd 100644 --- a/skill/engineering-workflow/scripts/upgrade_target_workflow.py +++ b/skill/engineering-workflow/scripts/upgrade_target_workflow.py @@ -95,6 +95,13 @@ }, } +# Exact agent-template bytes shipped in 0.9.7. No customized target file is rewritten. +PRIOR_AGENT_TEMPLATE_HASHES = { + "utility": "2f32f34a8c23d66c037abd0d1466f1eebc41ee52fd5e1e422470b7fbead4c210", + "explorer": "cf28d059b8bc28123a038d2f4c40fe24fe45e5623d2ee73c0e2b81f0a1d381b4", + "reviewer": "6182122fcec3d18b14acdabb644b750e58c5d2264d8b7a68eaf54644ef6db133", +} + def _content_hash(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() @@ -104,6 +111,31 @@ def _is_pristine_legacy(relative: str, text: str) -> bool: return _content_hash(text) in LEGACY_PRISTINE_HASHES.get(relative, set()) +def _is_pristine_prior_agent(name: str, text: str) -> bool: + return _content_hash(text) == PRIOR_AGENT_TEMPLATE_HASHES[name] + + +def _agent_config_selected(root: Path, explicitly_selected: bool) -> bool: + if explicitly_selected: + return True + if _first_symlink_component(root, STATE_MANIFEST_PATH): + return False + state = _read(root / STATE_MANIFEST_PATH) + try: + declared, shared_paths = parse_manifest_path_list(state, "shared_paths") + except ValueError: + return False + expected = {".codex/config.toml"} | {f".codex/agents/{name}.toml" for name in ("utility", "explorer", "reviewer")} + return ( + re.search(r"(?m)^schema_version:\s*2\s*$", state) is not None + and re.search(r"(?m)^skill_name:\s*engineering-workflow\s*$", state) is not None + and re.search(r"(?m)^mode:\s*upgrade_target_workflow\s*$", state) is not None + and re.search(r"(?m)^runtime_agent_config_managed:\s*true\s*$", state) is not None + and declared + and expected.issubset(shared_paths) + ) + + class MigrationConflict(RuntimeError): def __init__(self, code: str, message: str): super().__init__(message) @@ -741,6 +773,10 @@ def _proposed_changes(root: Path, include_agent_config: bool) -> list[dict[str, changes.append( {"path": path, "action": "create", "reason": "explicit optional agent configuration request"} ) + elif _is_pristine_prior_agent(name, _read(root / path)): + changes.append( + {"path": path, "action": "update", "reason": "known pristine prior agent template fingerprint"} + ) return changes @@ -754,6 +790,7 @@ def build_migration_report( root = repo.resolve() if not root.is_dir(): raise MigrationConflict("missing_repository", "Target repository does not exist") + include_agent_config = _agent_config_selected(root, include_agent_config) audit = audit_repo(root) conflicts = _scan_contract_conflicts(root, include_agent_config=include_agent_config) state_text = _read(root / STATE_MANIFEST_PATH) @@ -919,7 +956,9 @@ def _already_current(report: dict[str, Any], include_agent_config: bool, root: P topology = report["detected_topology"] required_artifacts = ("root_agents", "plans", "backlog", "pitfalls", "principles", "state_manifest") pristine_update_pending = any( - change.get("reason") == "known pristine legacy template fingerprint" for change in report["proposed_changes"] + change.get("reason") + in {"known pristine legacy template fingerprint", "known pristine prior agent template fingerprint"} + for change in report["proposed_changes"] ) return ( report["success"] @@ -1245,6 +1284,7 @@ def apply_migration( include_agent_config, approved_privacy_review, ) + include_agent_config = report["include_agent_config"] privacy_review, privacy_findings, approved_fingerprints = _evaluate_privacy_review( root, report["current_workflow_version"], @@ -1365,7 +1405,8 @@ def write(relative: str, text: str) -> None: write(".codex/config.toml", merged) for name in ("utility", "explorer", "reviewer"): relative = f".codex/agents/{name}.toml" - if not secure.exists(relative): + existing_agent = read(relative) + if not existing_agent or _is_pristine_prior_agent(name, existing_agent): write( relative, (AGENT_TEMPLATE_ROOT / f"{name}.toml.tmpl").read_text(encoding="utf-8"), @@ -1533,6 +1574,7 @@ def execute_prompt_upgrade( include_agent_config, approved_privacy_review, ) + include_agent_config = report["include_agent_config"] if report["required_user_questions"]: return { **report, @@ -1643,7 +1685,7 @@ def main() -> int: mode.add_argument("--plan", action="store_true") mode.add_argument("--apply", action="store_true") mode.add_argument("--prompt", action="store_true") - parser.add_argument("--target-version", default="0.9.7") + parser.add_argument("--target-version", default="0.9.8") parser.add_argument("--include-agent-config", action="store_true") parser.add_argument( "--approve-privacy-review", diff --git a/skill/engineering-workflow/scripts/validate_skill_repo.py b/skill/engineering-workflow/scripts/validate_skill_repo.py index 263e464..5a08231 100644 --- a/skill/engineering-workflow/scripts/validate_skill_repo.py +++ b/skill/engineering-workflow/scripts/validate_skill_repo.py @@ -508,21 +508,21 @@ def _validate_agent_profiles(repo_root: Path) -> list[str]: if field not in data: issues.append(f"{path.name} is missing required field: {field}") utility = parsed.get("utility", {}) - expected_utility_model = "gpt-" + "5.6-" + "terra" + expected_utility_model = "gpt-" + "6-luna" if utility.get("model") != expected_utility_model or utility.get("model_reasoning_effort") != "low": issues.append("Utility agent must use the current low-cost low-reasoning profile") if utility.get("sandbox_mode") != "read-only": issues.append("Utility agent must remain read-only") explorer = parsed.get("explorer", {}) - if explorer.get("model") != expected_utility_model or explorer.get("model_reasoning_effort") != "medium": + if explorer.get("model") != "gpt-" + "6-sol" or explorer.get("model_reasoning_effort") != "medium": issues.append("Explorer agent must use the current balanced read-heavy profile") if explorer.get("sandbox_mode") != "read-only": issues.append("Explorer agent must remain read-only") reviewer = parsed.get("reviewer", {}) - if reviewer.get("model") != "gpt-" + "6-astra": + if reviewer.get("model") != "gpt-" + "6-sol": issues.append("Reviewer agent must use the current Codex review model profile") - if reviewer.get("model_reasoning_effort") != "high" or reviewer.get("sandbox_mode") != "read-only": - issues.append("Reviewer agent must use high reasoning in read-only mode") + if reviewer.get("model_reasoning_effort") != "medium" or reviewer.get("sandbox_mode") != "read-only": + issues.append("Reviewer agent must use medium reasoning in read-only mode") reference = repo_root / "skill/engineering-workflow/references/agent_orchestration.md" if reference.exists(): text = reference.read_text(encoding="utf-8") diff --git a/tests/test_agent_orchestration.py b/tests/test_agent_orchestration.py index 466c2fc..75597b1 100644 --- a/tests/test_agent_orchestration.py +++ b/tests/test_agent_orchestration.py @@ -18,6 +18,7 @@ def test_deterministic_polling_is_not_routed_to_a_model(self): self.assertIn("Do not use a language-model subagent", text) self.assertIn("sleep", text) self.assertIn("polling", text) + self.assertIn("running a known shell command or test suite", text) self.assertIn("Do not implement monitoring as a model sleep loop", text) def test_completion_wait_is_persistent_and_does_not_wake_model_for_empty_state(self): @@ -112,14 +113,14 @@ def test_optional_profiles_use_expected_safety_defaults(self): utility = tomllib.loads((AGENTS / "utility.toml.tmpl").read_text(encoding="utf-8")) explorer = tomllib.loads((AGENTS / "explorer.toml.tmpl").read_text(encoding="utf-8")) reviewer = tomllib.loads((AGENTS / "reviewer.toml.tmpl").read_text(encoding="utf-8")) - self.assertEqual(utility["model"], "gpt-" + "5.6-" + "terra") + self.assertEqual(utility["model"], "gpt-" + "6-luna") self.assertEqual(utility["model_reasoning_effort"], "low") self.assertEqual(utility["sandbox_mode"], "read-only") self.assertEqual(explorer["sandbox_mode"], "read-only") - self.assertEqual(explorer["model"], utility["model"]) + self.assertEqual(explorer["model"], "gpt-" + "6-sol") self.assertEqual(explorer["model_reasoning_effort"], "medium") - self.assertEqual(reviewer["model"], "gpt-" + "6-astra") - self.assertEqual(reviewer["model_reasoning_effort"], "high") + self.assertEqual(reviewer["model"], "gpt-" + "6-sol") + self.assertEqual(reviewer["model_reasoning_effort"], "medium") self.assertEqual(reviewer["sandbox_mode"], "read-only") for profile in (utility, explorer, reviewer): instructions = profile["developer_instructions"] @@ -131,9 +132,10 @@ def test_utility_template_has_no_expensive_reasoning_or_api_pro_fields(self): for value in ('"high"', '"xhigh"', '"max"', '"ultra"', "reasoning.mode"): self.assertNotIn(value, text) - def test_minimal_and_none_are_conditional_only(self): + def test_none_is_conditional_and_legacy_minimal_migrates_to_low(self): text = PROFILES.read_text(encoding="utf-8") - self.assertIn("allow `minimal` or `none` only when", text) + self.assertIn("allow `none` only when", text) + self.assertIn("older profile used `minimal`", text) self.assertIn("regression tests or evaluation preserve quality", text) def test_concrete_model_slugs_have_one_reference_owner(self): diff --git a/tests/test_marketplace_package.py b/tests/test_marketplace_package.py index c52c844..8685017 100644 --- a/tests/test_marketplace_package.py +++ b/tests/test_marketplace_package.py @@ -30,7 +30,7 @@ def test_repository_package_matches_deterministic_builder(self): result = json.loads(completed.stdout) self.assertEqual(completed.returncode, 0, completed.stderr) self.assertTrue(result["success"], result) - self.assertEqual(result["version"], "0.9.7") + self.assertEqual(result["version"], "0.9.8") self.assertEqual(result["drift"], []) def test_check_detects_packaged_skill_byte_drift(self): @@ -111,7 +111,7 @@ def test_manifests_declare_only_self_contained_skill_capability(self): (REPO_ROOT / "plugins/engineering-workflow/.claude-plugin/plugin.json").read_text(encoding="utf-8") ) for manifest in (codex, claude): - self.assertEqual(manifest["version"], "0.9.7") + self.assertEqual(manifest["version"], "0.9.8") self.assertEqual(manifest["repository"], builder.REPOSITORY_URL) self.assertNotIn("mcpServers", manifest) self.assertNotIn("apps", manifest) diff --git a/tests/test_skill_repo_validation.py b/tests/test_skill_repo_validation.py index 8c46566..9d228e3 100644 --- a/tests/test_skill_repo_validation.py +++ b/tests/test_skill_repo_validation.py @@ -13,7 +13,7 @@ validate_skill_repo = load_script_module("validate_skill_repo") REPO_ROOT = Path(__file__).resolve().parents[1] -CURRENT_VERSION = "0.9.7" +CURRENT_VERSION = "0.9.8" class SkillRepoValidationTests(unittest.TestCase): @@ -237,7 +237,7 @@ def test_shared_frontmatter_rejects_native_platform_overrides(self): errors, _version = validate_skill_repo._validate_skill_router(root) self.assertEqual(errors, []) - def test_profile_drift_and_unsupported_astra_effort_are_rejected(self): + def test_profile_drift_and_unsupported_reviewer_effort_are_rejected(self): with tempfile.TemporaryDirectory() as tmp: root = Path(tmp) self._copy_repo_subset(root) @@ -245,9 +245,9 @@ def test_profile_drift_and_unsupported_astra_effort_are_rejected(self): reviewer = agents / "reviewer.toml.tmpl" original = reviewer.read_text(encoding="utf-8") variants = ( - original.replace("gpt-" + "6-astra", "gpt-" + "5.6"), - original.replace('model_reasoning_effort = "high"', 'model_reasoning_effort = "none"'), - original.replace('model_reasoning_effort = "high"', 'model_reasoning_effort = "minimal"'), + original.replace("gpt-" + "6-sol", "gpt-" + "5.6-terra"), + original.replace('model_reasoning_effort = "medium"', 'model_reasoning_effort = "none"'), + original.replace('model_reasoning_effort = "medium"', 'model_reasoning_effort = "minimal"'), ) for index, variant in enumerate(variants): with self.subTest(variant=index): diff --git a/tests/test_upgrade_target_workflow.py b/tests/test_upgrade_target_workflow.py index 648d5a7..a073c70 100644 --- a/tests/test_upgrade_target_workflow.py +++ b/tests/test_upgrade_target_workflow.py @@ -859,6 +859,66 @@ def test_opted_in_agent_configuration_preserves_existing_reviewer_pin(self): self.assertTrue(result["success"], result) self.assertEqual(reviewer.read_text(encoding="utf-8"), original) + def test_prior_opt_in_refreshes_only_pristine_agent_models(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + make_target(root) + initial = migrator.apply_migration(root, "0.9.7", include_agent_config=True) + self.assertTrue(initial["success"], initial) + agents = root / ".codex/agents" + for name in ("utility", "explorer", "reviewer"): + path = agents / f"{name}.toml" + prior = path.read_text(encoding="utf-8") + if name == "utility": + prior = prior.replace('model = "gpt-6-luna"', 'model = "gpt-5.6-terra"') + elif name == "explorer": + prior = prior.replace('model = "gpt-6-sol"', 'model = "gpt-5.6-terra"') + else: + prior = prior.replace('model = "gpt-6-sol"', 'model = "gpt-6-astra"') + prior = prior.replace('model_reasoning_effort = "medium"', 'model_reasoning_effort = "high"') + prior = prior.replace( + "Bounded evidence-first review for ordinary changes", + "Evidence-first review for correctness and high-risk changes", + ) + self.assertTrue(migrator._is_pristine_prior_agent(name, prior)) + path.write_text(prior, encoding="utf-8") + + explorer = agents / "explorer.toml" + custom = explorer.read_text(encoding="utf-8").replace( + 'model = "gpt-5.6-terra"', 'model = "custom-supported-model"' + ) + explorer.write_text(custom, encoding="utf-8") + report = migrator.build_migration_report(root, "0.9.8") + self.assertTrue(report["include_agent_config"]) + proposed = { + change["path"] + for change in report["proposed_changes"] + if change["reason"] == "known pristine prior agent template fingerprint" + } + self.assertEqual(proposed, {".codex/agents/utility.toml", ".codex/agents/reviewer.toml"}) + + result = migrator.execute_prompt_upgrade(root, "0.9.8") + self.assertTrue(result["success"], result) + self.assertTrue(result["include_agent_config"]) + self.assertEqual(explorer.read_text(encoding="utf-8"), custom) + for name in ("utility", "reviewer"): + expected = (migrator.AGENT_TEMPLATE_ROOT / f"{name}.toml.tmpl").read_text(encoding="utf-8") + self.assertEqual((agents / f"{name}.toml").read_text(encoding="utf-8"), expected) + + def test_prior_agent_template_is_not_changed_without_opt_in(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + make_target(root) + utility = root / ".codex/agents/utility.toml" + utility.parent.mkdir(parents=True) + current = (migrator.AGENT_TEMPLATE_ROOT / "utility.toml.tmpl").read_text(encoding="utf-8") + prior = current.replace('model = "gpt-6-luna"', 'model = "gpt-5.6-terra"') + utility.write_text(prior, encoding="utf-8") + result = migrator.execute_prompt_upgrade(root, "0.9.8") + self.assertTrue(result["success"], result) + self.assertFalse(result["include_agent_config"]) + self.assertEqual(utility.read_text(encoding="utf-8"), prior) + def test_invalid_codex_config_blocks_only_requested_configuration_work(self): for include_config in (False, True): with self.subTest(include_config=include_config), tempfile.TemporaryDirectory() as tmp: @@ -1137,8 +1197,8 @@ def test_structural_toml_merge_preserves_unknown_keys_and_profiles(self): for name in ("utility", "explorer", "reviewer"): self.assertTrue((root / ".codex" / "agents" / f"{name}.toml").exists()) reviewer = tomllib.loads((root / ".codex/agents/reviewer.toml").read_text(encoding="utf-8")) - self.assertEqual(reviewer["model"], "gpt-" + "6-astra") - self.assertEqual(reviewer["model_reasoning_effort"], "high") + self.assertEqual(reviewer["model"], "gpt-" + "6-sol") + self.assertEqual(reviewer["model_reasoning_effort"], "medium") def test_agents_header_comment_is_preserved_during_merge(self): text = "[agents] # keep this comment\nmax_threads = 3\n"