diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json index 25e7d04..21902cc 100644 --- a/.agents/plugins/marketplace.json +++ b/.agents/plugins/marketplace.json @@ -11,7 +11,7 @@ "plugins": [ { "name": "codex-co-engineer", - "version": "3.4.0", + "version": "3.4.2", "description": "Give Codex a team of external co-engineers without giving up control. Delegating to Co-Engineer starts one bounded run. The honest shape is up to eight isolated external co-engineers, one bounded run, one coordinated wait, one verified decision.", "keywords": [ "codex", diff --git a/.codex/release-gate.toml b/.codex/release-gate.toml index 521b1cd..84652ae 100644 --- a/.codex/release-gate.toml +++ b/.codex/release-gate.toml @@ -21,7 +21,6 @@ kind = "bootstrap" command = ["npm", "--prefix", "tools/acpx-vendor", "ci", "--ignore-scripts", "--no-audit", "--no-fund"] failure_class = "environment_blocked" timeout_seconds = 180 -env = { ACPX_NPM_CACHE = "/tmp/codex-acpx-release-npm-cache", NPM_CONFIG_CACHE = "/tmp/codex-acpx-release-npm-cache" } [[stages]] name = "co-engineer-unit" @@ -74,7 +73,6 @@ kind = "build" command = ["npm", "--prefix", "tools/acpx-vendor", "run", "test:reproducible"] failure_class = "product_test_failed" timeout_seconds = 180 -env = { ACPX_NPM_CACHE = "/tmp/codex-acpx-release-npm-cache", NPM_CONFIG_CACHE = "/tmp/codex-acpx-release-npm-cache" } [[stages]] name = "acpx-publish-provenance" @@ -82,7 +80,6 @@ kind = "security_checks" command = ["npm", "--prefix", "tools/acpx-vendor", "run", "verify:publish-provenance"] failure_class = "product_test_failed" timeout_seconds = 180 -env = { ACPX_NPM_CACHE = "/tmp/codex-acpx-release-npm-cache", NPM_CONFIG_CACHE = "/tmp/codex-acpx-release-npm-cache" } [[stages]] name = "co-engineer-package-inventory" diff --git a/CHANGELOG.md b/CHANGELOG.md index b256398..d3925e8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,100 @@ ## [Unreleased] +## [3.4.2] - 2026-09-08 + +### Fixed + +- Instruct providers to honor the requested answer format and omit routine + narration and unrequested evidence headings. Controller evidence requirements + remain intact and do not act as an answer template. +- Return Grok's final response after a settled tool round on successful tasks. + Preserve progress in task events and retain aggregate output when framing is + ambiguous, incomplete, over limit, or includes inline web search. +- Handle processes disappearing during verification cleanup scans and parse + parenthesized process names correctly, while preserving fail-closed errors. +- Remember explicit native approval for the same repository and selected + providers, with a one-run option and local revocation. Remove the redundant + approval checkbox and allow more time to answer the native form. +- Detect missing installed worker entrypoints before preparing workspaces or + dispatching providers, with actionable reinstall and restart guidance. +- Direct provider workers to their assigned working directory and clarify + controller-owned receipts, reducing unnecessary setup and evidence searches. + +- Adapt Cursor Cloud SDK results before internal validation so documented + timing, model, and usage metadata does not prevent terminal result delivery. +- Use ACPX's one-shot command for DSH, preserving bounded provider errors and + results without a nested flow or replaying a submitted prompt. + +- Deliver the compiled assignment instructions and pin local workspaces to + the recorded commit; reject unsupported provider model overrides instead + of silently launching another model. + +- Align semantic runs with the authoritative provider task lifecycle, including + result delivery, recoverable observation failures, cancellation, and restart. + Reuse event-driven task waits and keep unchanged run receipts stable. + +- Wire semantic launch consent to native MCP forms, with explicit unsupported + host handling and same-run consent retry. Preserve pending/terminal receipts + and cursors; distinguish planned review work from completed evidence. + +- Reject future-dated repository-exposure approvals and make consent-window + fixtures independent of today's date. Remove ambient Git/network and + fixed-sleep dependencies from affected regression tests. + +- Semantic `run_request` launches now satisfy the advertised MCP JSON Schema + without legacy single-task fields; ambiguous dual envelopes are rejected. +- Valid disjoint writer globs use the authoritative overlap validator instead + of a duplicate check that rejected separate directories. +- Repeated status checks no longer repeat terminal cleanup grace periods for + previously reconciled receipts; ownership inspection remains active. +- Preserve structured Cursor permission options through grouped questions and + stop treating shell command punctuation as user input. +- Distinguish explicit Grok login status from ancillary command errors during + readiness checks. +- All-Cursor Cloud runs skip the irrelevant local systemd/cgroup prerequisite; + mixed and local runs still enforce it. +- Assignment access can be derived from its explicit role, eliminating a + redundant launch field while rejecting explicit role/access conflicts. + +### Changed + +- Bundle MIT-licensed Worktree Bootstrap 1.1.0 with recorded source provenance. + Local setup no longer depends on a separately installed private tool; Python + 3.11+ is required. Validate runtime prerequisites before installation. +- Reorganize installation and first-use documentation, include detailed upgrade + notes, and keep Luna/Sol coordination explicitly optional. + +- Default Muse to `meta/muse-spark-1.3-contributor` through OpenRouter with + XHigh reasoning, using only the OpenRouter credential for that route. +- Release checks honor the operator's configured npm cache and temporary + storage instead of forcing a machine-specific cache path. + +- Added evidence-qualified Luna / Terra / Sol / Astra task-selection guidance. + Prefer one owner with bounded workers; an extra coordinator and reasoning + effort changes require task-specific justification. Model choice stays host-owned. +- Made the installed skill path use existing provider choices and semantic + admission directly, with setup and manager tasks outside routine launch. + Clarified continuation, typed approvals, dependent reviews, and verification. + Provider prompts stay focused on requested output, tests and brief evidence; + the controller creates managed worktrees while the worker wrapper owns + verification, hidden writer tokens, lifecycle and machine receipts. + +## [3.4.1] - 2026-09-01 + +Reliability candidate for bounded Co-Engineer implementation lanes. Preserves +the five public tools and 3.4.0 compatibility while adding a server-compiled +`run_request`, atomic pre-prompt admission, truthful dispatch evidence, +restart-safe session recovery, typed repository-exposure consent, +capability-aware attention batching, idempotent cancellation, silence-aware +handoffs, readiness caching, and bounded simple-run responses. See +[`docs/releases/v3.4.1.md`](docs/releases/v3.4.1.md) and the exact baseline +ledger in [`docs/releases/v3.4.1-baseline.md`](docs/releases/v3.4.1-baseline.md). + +The candidate still requires host consent integration, an official exact-SHA +local-only `worktree-bootstrap` capability, the Codex Desktop history-bound +regression, and live provider acceptance before release publication. + ## [3.4.0] - 2026-08-28 3.4.0 is the Co-Engineer experience and efficiency release. It keeps the diff --git a/README.md b/README.md index 47d5a5d..886076d 100644 --- a/README.md +++ b/README.md @@ -1,533 +1,266 @@ # Codex-Co-Engineer -Give Codex a team of external co-engineers without giving up control. +**Give Codex a team. Keep control of the result.** -[![Codex-Co-Engineer CI status](https://github.com/ajhcs/Codex-Co-Engineer/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/ajhcs/Codex-Co-Engineer/actions/workflows/ci.yml) -[![Latest Codex-Co-Engineer release](https://img.shields.io/github/v/release/ajhcs/Codex-Co-Engineer?sort=semver)](https://github.com/ajhcs/Codex-Co-Engineer/releases/latest) -[![Node.js 24 or newer](https://img.shields.io/badge/Node.js-24%2B-339933?logo=nodedotjs&logoColor=white)](https://nodejs.org/) +[![CI](https://github.com/ajhcs/Codex-Co-Engineer/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/ajhcs/Codex-Co-Engineer/actions/workflows/ci.yml) +[![Latest release](https://img.shields.io/github/v/release/ajhcs/Codex-Co-Engineer?sort=semver)](https://github.com/ajhcs/Codex-Co-Engineer/releases/latest) +[![Node.js 24+](https://img.shields.io/badge/Node.js-24%2B-339933?logo=nodedotjs&logoColor=white)](https://nodejs.org/) [![MIT license](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE) -Codex remains chief engineer and reviewer. External co-engineers do -isolated assigned work. External workers may commit. A scoped publisher -may non-force push only the task branch and open a draft PR. Sol High -or Sol XHigh alone may perform a regular merge after deterministic -exact-head/tree, current green CI, verifier, and topology checks. The -user retains version, tag, release, protected-ref, and product-policy -authority. You keep control. The honest shape is up to eight isolated -external co-engineers, one bounded run, one coordinated wait, one -verified decision. +[Install](#install-and-authentication) · [Try it](#your-first-delegation) · [Providers](#provider-choices) · [Release notes](docs/releases/v3.4.2.md) · [Troubleshooting](#troubleshooting) -`Delegating to Co-Engineer` starts that one bounded run. `Chatting with -Co-Engineer` inspects, continues, answers grouped attention, or cancels -the run that already exists. If you ask to chat and nothing is running, -Codex says chatting needs existing work and offers to delegate. It does -not silently submit. +Ask Codex to bring in **Grok, Cursor, or Muse** for implementation, investigation, +or a second opinion. Co-Engineer prepares isolated workspaces, coordinates up to +**eight independent assignments**, and brings their results back for Codex to review. +You decide what ships. -The public co-engineer names are `Using Grok Co-Engineer`, `Using Cursor -Co-Engineer`, and `Using Muse Co-Engineer`. Cursor on this computer and -Cursor Cloud both stay Cursor Co-Engineer in public speech. +> Use Grok Co-Engineer to review this change. Focus on correctness and regressions. -Any extra Co-Engineer panel is optional, feature-detected, and -host-specific. The same conversation works headless in Codex CLI. This -documentation does not claim a Co-Engineer UI on every Codex Desktop -host. - -The stable machine identifier is `codex-co-engineer`. Published package -notes: [docs/releases/v3.4.0.md](docs/releases/v3.4.0.md). - -## Visual demo - -The static Co-Engineer architecture illustration is the authoritative -GitHub-compatible visual. It uses the approved Co-Engineer identity. It -is not a live-run screenshot. There is no autoplay audio. +No hand-written tool payloads. No profile required for an ordinary launch. +Choose a provider, describe the work, and keep talking in the same Codex task. ![Give Codex a team of external co-engineers without giving up control.](docs/assets/co-engineer-3.4.0/final/derived/hero-demo.jpg) -## First 60 seconds - -After [install and authentication](#install-and-authentication), start a -**new** Codex session and speak in ordinary language. You do not write -tool payloads. - -### Your first delegation - -> Delegating to Co-Engineer: review the auth change with Grok Co-Engineer. +*Architecture illustration, not a live screenshot. There is no autoplay audio.* -Codex: +## What makes it useful -> I am delegating this to Co-Engineer. Using Grok Co-Engineer. -> Co-Engineer is running 1 independent assignment. - -That is the only submission. Codex waits once. When the work is -complete, Codex inspects it: - -> Co-Engineer finished, and I verified the candidate. - - +| You want to… | Co-Engineer handles… | +| --- | --- | +| Get another model's perspective | Explicit Grok, Cursor, and Muse assignments using your provider accounts | +| Work on several independent changes | A separate managed Git worktree and branch for each local assignment | +| Keep Codex focused | One submission, coordinated waits, compact results, and details on demand | +| Continue after a disconnect | Durable task identities and receipts; accepted prompts are never blindly replayed | +| Avoid repeated setup decisions | Existing provider choices and optional remembered repository/provider approval | +| Review before integrating | Retained branches, output, and handoffs for Codex to inspect | -![Conceptual illustration of a first Co-Engineer delegation: one explicit Grok assignment, one submission, and one verified candidate.](docs/assets/co-engineer-3.4.0/final/derived/first-delegation.jpg) +**New in 3.4.2:** simpler launches, reusable consent, more reliable provider +completion and cleanup, and concise Grok results. Read the +[detailed release notes](docs/releases/v3.4.2.md) for compatibility and limits. -You still decide whether to keep, change, or discard the result. +## Install and authentication -### Several independent assignments +### 1. Check your host -If you want several independent assignments in one run: +You need **Node.js 24+**, **Git**, **Python 3.11+** for the bundled setup, and a +current **Codex CLI** with plugin support. +Local Grok, Cursor, and Muse also require **Linux**, a working `systemd --user` +manager, `systemd-run` 244+, and unified cgroup v2. +Cursor Cloud runs remotely and does not require that local process boundary. -> Split this into three isolated independent assignments: API -> validation, the operator guide, and a review of both diffs. +Install the CLI and account access for **only the providers you plan to use**. +The provider table below separates these requirements. Co-Engineer does not +install or sign you into Grok or Cursor. -Codex: +### 2. Install the release -> I am delegating this to Co-Engineer. Co-Engineer is running 3 -> independent assignments. +Run these commands from the directory where you keep your projects: - +```bash +git clone --branch v3.4.2 --single-branch https://github.com/ajhcs/Codex-Co-Engineer.git +cd Codex-Co-Engineer +npm --prefix plugins/codex-co-engineer run setup +codex plugin marketplace add "$PWD" +codex plugin add codex-co-engineer@codex-co-engineer +npm --prefix plugins/codex-co-engineer run setup:check +``` -Independent assignments stay isolated. They do not share a writer path. -Codex submits once, waits once, and inspects the whole set together. +Keep this clone: it is the registered local marketplace source. Setup installs +pinned ACPX, Cursor SDK, and DSH dependencies globally and creates key-free DSH +configuration. It preserves existing compatible configuration and reports +incompatible profiles instead of overwriting them. Use a user-writable npm global +prefix on your `PATH`; a Node version manager is one way to provide it. -If you have no saved profile and name no co-engineers, Codex asks once -which co-engineers should take the independent assignments: Grok, -Cursor, or Muse. It does not keep asking and does not invent a default -router. +`setup:check` checks Node/Python prerequisites, installed dependencies, and DSH configuration. Provider login +and the running MCP process's Linux boundary are checked separately by `status`. +The worktree tool is bundled; no separate `worktree-bootstrap` installation is needed. -A longer walkthrough lives in -[docs/co-engineer-quickstart.md](docs/co-engineer-quickstart.md). +### 3. Connect your chosen provider -## How a run works +| Provider | One-time authentication | Runs where? | +| --- | --- | --- | +| **Grok** | Install [Grok Build](https://docs.x.ai/build/cli), then run `grok login` | Local managed worktree | +| **Cursor Local** | Install [Cursor CLI](https://cursor.com/docs/cli/installation), then run `cursor-agent login` | Local managed worktree | +| **Cursor Cloud** | Configure `CURSOR_API_KEY` or an owner-only key file; see [configuration](docs/configuration.md#cursor-cloud) | Cursor's remote environment | +| **Muse** | From this clone, run `plugins/codex-co-engineer/bin/set-model-api-key` to save your OpenRouter key | Local DSH managed worktree | -A run is one submission, one coordinated wait, and one verified -decision. Independent means the assignments do not share a writer path. -The bound is eight. +Muse defaults to **Muse Spark 1.3 Contributor, XHigh, through OpenRouter**. +Provider credentials stay in normal login state, environment variables, or +owner-only key files. Never paste them into task prompts or MCP arguments. -Codex speech during a run uses these utterances: +### 4. Start a new Codex session -- `I am delegating this to Co-Engineer` -- `Co-Engineer is running N independent assignments` (`assignment` when - N is 1) -- `Co-Engineer needs one decision from you` -- `Co-Engineer finished, and I verified the candidate.` +Ask: -The verified-final sentence includes its period. Codex uses it only -after it has inspected a complete candidate. Failure, cancel, and -unresolved outcomes must not use it. +> Show Co-Engineer status, then use Grok Co-Engineer to review the latest change. -Use Codex-Co-Engineer when you want Codex to keep control while isolated -external co-engineers do assigned review or implementation work. Do not -use it as a security sandbox, a credential broker, or a replacement for -the provider's own login and approval flow. +The first repository-sharing form offers **this run only** or **remember access +for this repository and the selected providers**. Remembered access works across +linked worktrees. Adding a provider or changing the repository origin requires a +new decision. [Inspect or revoke remembered access](plugins/codex-co-engineer/README.md#remembered-repository-consent). -The bundled skill is `control-codex-co-engineer-agents`. Normal-user -journeys: [docs/co-engineer-user-journeys.md](docs/co-engineer-user-journeys.md). +
+See the installation overview -## Provider choices + -Provider and model are explicit, chosen by you, or filled from one named -profile. Missing selection is one ask, not a router. Codex never ranks, -predicts cost, or substitutes a different co-engineer. +![Conceptual illustration of a clean Co-Engineer install and provider sign-in.](docs/assets/co-engineer-3.4.0/final/derived/install-auth.jpg) -| You say | Codex says | -| --- | --- | -| Grok Co-Engineer | Using Grok Co-Engineer | -| Cursor Co-Engineer | Using Cursor Co-Engineer | -| Muse Co-Engineer | Using Muse Co-Engineer | +
-- **Using Grok Co-Engineer** runs an explicitly selected Grok assignment. -- **Using Cursor Co-Engineer** covers Cursor on this computer or Cursor - Cloud; Codex keeps the location and immutable starting commit explicit. -- **Using Muse Co-Engineer** runs the explicitly selected Muse profile and - model. It does not silently become another provider. +## Your first delegation -You may name the Cursor place in plain language. Public speech still -stays `Using Cursor Co-Engineer`. Muse is the public name for that -route. Optional Ox Alpha stays a Muse-route model choice, not a separate -public co-engineer name. +> Use Grok Co-Engineer to review the authentication changes. Report actionable findings. -Example: +Codex submits the assignment, Co-Engineer prepares its workspace, and the provider +runs. Codex then reads the result and checks the supporting evidence. A launch +acknowledgement is not a completed review. -> Use Grok Co-Engineer for the API change and Muse Co-Engineer for the -> docs. Keep the review on Cursor Co-Engineer. +
+See a single-assignment walkthrough -Codex: + -> I am delegating this to Co-Engineer. Using Grok Co-Engineer. -> Using Muse Co-Engineer. Using Cursor Co-Engineer. Co-Engineer is -> running 3 independent assignments. +![Conceptual illustration of one explicit Grok assignment, one submission, and a verified candidate.](docs/assets/co-engineer-3.4.0/final/derived/first-delegation.jpg) - +
-![Conceptual illustration of explicit Grok, Cursor, and Muse Co-Engineer choices, with no learned or global router.](docs/assets/co-engineer-3.4.0/final/derived/provider-choices.jpg) - -## Codex authority and safety - -Codex remains the chief engineer and reviewer. External workers may -commit. A scoped publisher may non-force push only the task branch and -open a draft PR. Sol High or Sol XHigh alone may perform a regular merge -after deterministic exact-head/tree, current green CI, verifier, and -topology checks. The user retains version, tag, release, protected-ref, -and product-policy authority. External co-engineers stay isolated. One -bounded run is in flight at a time for this work. Chatting never becomes -a second submission. Failure stays visible. - -Selecting a provider authorizes the assignment prompt and repository -content to be sent to that provider. Private repositories are supported -when the configured provider is authorized to review them. Provider -children inherit the user's normal authenticated environment because -they are trusted peer coding agents. - -Local workers are launched as manager-owned transient `systemd --user` -services with `KillMode=control-group` solely so cancellation reaches -detached descendants and the worker survives the launching client. This -is a lifecycle/cleanup boundary, not a sandbox: providers inherit the -normal environment, network, filesystem, credentials, and shell -capabilities. Local dispatch fails closed when the Linux systemd/cgroup -prerequisite is not available. The check occurs before Codex-Co-Engineer -creates a managed worktree, task receipt, or prompt file. Cursor Cloud -runs in the provider's remote environment and does not depend on the -local process boundary. - -Cursor Local and DSH's official fallback CLIs take the prompt -positionally, so it may be visible to other processes running as the -same Unix user for the duration of that fallback. Grok fallback uses an -owner-only prompt file. - -Managed local work uses one locked `worktree-bootstrap` worktree and -branch per assignment: - -```text -one task → one worktree → one branch → one writer -``` +### Several independent assignments -Direct mutation of a supplied checkout is an explicit 3.2.1 single-task -choice only. Bounded-run submissions do not use direct mode. +> Use Grok for the API validator and Muse for the operator guide. Give them separate workspaces. -If `worktree-bootstrap` fails before returning an authoritative receipt -and path, Codex-Co-Engineer does not guess at or delete an unknown -worktree. Inspect the repository with `git worktree list` and the -`worktree-bootstrap` lock tooling; clean only an exact task/lock that -the tooling identifies. + -Cursor Cloud does not use a local worktree. The supplied repository must -have an origin that Cursor can access. A bounded-run Cloud lane needs an -exact, immutable commit SHA that has already been pushed to that origin. -An exact SHA that is reachable only from a feature branch can still be -invisible to Cursor until that branch is provider-visible through an -open pull request or the default branch. Create the draft PR (or make -the commit reachable from the default branch) before final Cloud -acceptance. If Cursor returns HTTP 400 for an otherwise-valid SHA, treat -it as a provider visibility failure and fix reachability before -retrying; do not blindly replay the work. +Independent assignments stay isolated. Assign work that can proceed independently; +ask for a review of the resulting changes after the implementation is available. -Local implementations return a branch and handoff for Codex to inspect. -External workers may commit. A scoped publisher may non-force push only -the task branch and open a draft PR after confirming that real commits -exist. Sol High or Sol XHigh alone may perform a regular merge after -deterministic exact-head/tree, current green CI, verifier, and topology -checks. +If you have no saved profile and do not name a provider, Codex asks which one to +use. It does not silently choose a different provider or model. -Task prompts, events, logs, runtime identities, local paths, branch -names, and opaque provider IDs are stored under the owner-only state -directory, normally `$XDG_STATE_HOME/codex-co-engineer` or -`~/.local/state/codex-co-engineer`. Task directories are `0700`; files -are `0600`. See [data handling](docs/data-handling.md). +### Continue, answer, or cancel -## Chatting, grouped attention, and the final decision +> Chat with Co-Engineer: show the current result. +> +> Use the stricter validation option. +> +> Cancel that Co-Engineer run. -Chatting with Co-Engineer during a live run is not a second delegation. -It can inspect the current run, continue from its recorded cursor, answer -one grouped decision, or cancel the existing run. It never creates -unlimited persistent chat and never silently starts new work. + -The run is already in its one coordinated wait. More than one assignment -needs a choice. Codex groups those questions into one decision. -Unaffected assignments keep working. +One grouped decision covers every assignment that asked; unaffected assignments +can keep working. That answer is chatting with the existing run, not a new launch. +Chatting requires an existing run. Starting new work remains an explicit delegation. -Codex: + -> Co-Engineer needs one decision from you. +The verified result is an evidence packet: the changes, branch, tests, and review +that support Codex's decision. It does not automatically merge your code. -You: +If a required assignment fails, Codex reports the gap and available recovery steps. +It does not describe an incomplete run as a verified result. -> Use the stricter validator and keep the docs change as written. +
+What a failed or unresolved assignment means - + -One grouped decision covers every assignment that asked. Lanes that did -not ask keep working. Codex resumes from the same cursor after you -answer. +![Conceptual illustration of a required assignment that failed or stayed unresolved, with no verified candidate claimed.](docs/assets/co-engineer-3.4.0/final/derived/failure-unresolved.jpg) -That answer is chatting: answer grouped attention. It is not a debate -loop. Codex continues the same run with the same single wait. +
-After the wait, Codex inspects the candidate. Only then may it say: +## Provider choices -> Co-Engineer finished, and I verified the candidate. +Use your preferred providers for the work at hand. Grok and Cursor keep their +configured provider model; unsupported overrides fail before launch. Muse uses +its configured DSH profile. Provider selection is explicit, and an assignment's +role is an instruction to a trusted coding agent, not a filesystem sandbox. - +
+Grok, Cursor, and Muse at a glance -The verified result is an evidence packet Codex has inspected: the -candidate, its branch, head, and tree, plus the scope, tests, and -reviews that support the decision. Verification is Codex's review of -that packet, not automatic merge. You keep control. + -If a required assignment fails, cannot be answered, or stays unresolved, -Codex reports that honestly. It does not say Co-Engineer finished, and I -verified the candidate. A required gap blocks a complete candidate. -Cancel is chatting with Co-Engineer, not a new delegation. +![Conceptual illustration of explicit Grok, Cursor, and Muse Co-Engineer choices.](docs/assets/co-engineer-3.4.0/final/derived/provider-choices.jpg) - +
-![Conceptual illustration of a required Co-Engineer assignment that failed or stayed unresolved, so no verified candidate is claimed.](docs/assets/co-engineer-3.4.0/final/derived/failure-unresolved.jpg) +For Codex itself, use one task owner and add workers when useful. Luna, Terra, +Sol, and Astra can cover different scopes; a second coordinator is optional. +The [model-role guide](plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/model-roles.md) +separates practical suggestions from official model documentation. Co-Engineer +does not change your Codex model, reasoning effort, or experimental settings. -## Install and authentication +## Upgrade to 3.4.2 -Requires Node.js 24+, Git, and Codex CLI. Local providers also need -Linux `systemd --user`, `systemd-run` 244 or newer, unified cgroup v2, -and `worktree-bootstrap` on `PATH`. +Finish or cancel active runs first. In a **clean existing source clone**: ```bash -git clone https://github.com/ajhcs/Codex-Co-Engineer.git -cd Codex-Co-Engineer -codex plugin marketplace add "$PWD" -codex plugin add codex-co-engineer@codex-co-engineer +git fetch origin tag v3.4.2 +git switch --detach v3.4.2 npm --prefix plugins/codex-co-engineer run setup +codex plugin remove codex-co-engineer@codex-co-engineer +codex plugin add codex-co-engineer@codex-co-engineer npm --prefix plugins/codex-co-engineer run setup:check ``` - +Start a new Codex session, then check Co-Engineer status. Keep your local changes +if the source clone is dirty; use a separate clean release clone instead of +resetting it. If your marketplace uses a different name, use the installed +identity shown by `codex plugin list`. -![Conceptual illustration of a clean Codex-Co-Engineer install and provider sign-in, with credentials and private paths omitted.](docs/assets/co-engineer-3.4.0/final/derived/install-auth.jpg) - -`npm run setup` installs pinned ACPX `0.13.0`, Cursor SDK `1.0.28`, and -the cohesive DSH `0.1.0-rc.7` composition. It does not log you into -Grok, Cursor Local, or Cursor Cloud. `setup:check` verifies those pinned -packages plus `worktree-bootstrap`. Start a **new** Codex session after -the plugin add. - -Sign in once for only the providers you will use: - -```bash -grok login -cursor-agent login -plugins/codex-co-engineer/bin/set-model-api-key -``` - -Cursor Cloud uses `CURSOR_API_KEY`, `CURSOR_API_KEY_FILE`, or the -owner-only `~/.config/cursor-cloud-control/api-key`. DSH uses -`MODEL_API_KEY`, `CODEX_CO_ENGINEER_MODEL_API_KEY_FILE`, or -`~/.config/codex-co-engineer/model-api-key` for Muse. The optional Ox -Alpha route uses `OPENROUTER_API_KEY`, -`CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE`, or -`~/.config/codex-co-engineer/openrouter-api-key`. Never put credentials -in MCP arguments or prompts. - -Host variables, profiles, and package-local setup live in -[docs/configuration.md](docs/configuration.md) and the -[plugin README](plugins/codex-co-engineer/README.md). - -The older `cursor-cloud-control` package remains in this repository as a -compatibility plugin for existing installations. New installations need -only Codex-Co-Engineer. - -## Migrating from 3.2.1 - -The public catalog is still five tools. Bounded runs are additive. If -you omit run fields, exact 3.2.1 single-task behavior remains, including -direct mode on that path only. - -The normal 3.2.1 habit of writing a tool payload is no longer the -visitor path. Say `Delegating to Co-Engineer` and name Grok, Cursor, or -Muse Co-Engineer. Codex submits once and waits once. - -Details: [docs/co-engineer-migration-3.2.1.md](docs/co-engineer-migration-3.2.1.md). +Existing task receipts and provider accounts are retained. Users with a direct +Meta Muse profile must migrate to OpenRouter; see the +[upgrade notes](docs/releases/v3.4.2.md#upgrading). ## Troubleshooting -**How do I check whether Codex-Co-Engineer can dispatch locally?** -Ask Codex: `Show Codex-Co-Engineer status.` Local providers are ready -only when `local_boundary.ready` is true. If it is false, the MCP -process is missing Linux `systemd --user`, `systemd-run` 244+, unified -cgroup v2, or the forwarded user-session locators (`XDG_RUNTIME_DIR`, -`DBUS_SESSION_BUS_ADDRESS`). - -**Setup passed, but local providers are unavailable.** -`setup:check` does not prove the MCP environment. Re-run status from the -actual MCP server process, then confirm the plugin `.mcp.json` allowlist -forwards `HOME`, `PATH`, `XDG_*`, and `DBUS_SESSION_BUS_ADDRESS`. - -**Where is the installed plugin?** -After `codex plugin add codex-co-engineer@codex-co-engineer`, Codex -reports the cached install path. The source package in this repository -is `plugins/codex-co-engineer`. Run `npm run setup` from that source -package (or with `npm --prefix plugins/codex-co-engineer`) rather than -guessing a cache path. - -**A managed worktree appeared without a receipt.** -Do not guess or delete it. Inspect `git worktree list` and -`worktree-bootstrap lock inspect`, then clean only an exact identified -task/lock. - -**Cursor Cloud returned HTTP 400 for a valid SHA.** -Treat it as a provider visibility failure. Make the commit reachable -from an open PR or the default branch, then retry. Do not replay a -prompt that was already dispatched. +| Symptom | Next step | +| --- | --- | +| Plugin tools are missing | Start a new Codex session after installation; check `codex plugin list` | +| Local provider is unavailable | Ask for Co-Engineer status; inspect `local_boundary` and the named missing dependency | +| Setup reports an incompatible Muse profile | Follow the OpenRouter migration in the release notes; keep a backup of your configuration | +| Repeated repository-sharing prompts | Choose remembered access; confirm the provider and repository origin have not changed | +| Installed files disappear or an old version returns | Check for another local marketplace using the same identity; use a distinct marketplace name for development candidates | +| Cursor Cloud cannot see a commit | Push the exact SHA and make the branch visible through an open PR or the default branch | +| No extra panel appears | Continue in the conversation; the CLI workflow is complete without an optional host UI | -**Can I put API keys in the MCP tool arguments?** -No. Use normal provider login or the owner-only key files. Credentials -must not appear in MCP arguments, prompts, receipts, fixtures, or Git. +More detail: [troubleshooting](docs/co-engineer-troubleshooting.md) · +[configuration](docs/configuration.md) · [quickstart](docs/co-engineer-quickstart.md). -**There is no Co-Engineer panel in this host.** -That is expected on hosts that do not expose one. The headless -Delegating/Chatting conversation is complete. Missing UI is not a failed -install. +## Control and data handling -**Chatting did nothing.** -Chatting with Co-Engineer needs an existing run. Codex should offer to -delegate instead of silently submitting. +External workers may commit within their assigned scope. Publication and merge require user authorization and Codex review. -More cases: [docs/co-engineer-troubleshooting.md](docs/co-engineer-troubleshooting.md). +You authorize the repository and providers. Codex reviews the result and integrates +only the work you approve. Provider agents can use their normal coding tools, +shell, network, and authenticated accounts; **Co-Engineer is not a sandbox**. +Local process control exists to keep workers durable and cancel their descendants. -## Advanced Co-Engineer Control/API +Repository content and history accessible from the assigned workspace can reach +the selected provider. Task state is retained locally with owner-only permissions. +Managed worktrees remain available for inspection; cleanup is explicit and tied +to the recorded task. Read [Security](SECURITY.md) and +[data handling](docs/data-handling.md) before delegating private repositories. -This section is for operators and Codex internals. Normal users do not -construct these payloads. +## For integrators and contributors The catalog remains exactly `status`, `delegate`, `task`, `tasks`, and -`cancel`. There is no sixth tool. Bounded runs use additive `run`, -`run_id`, `attention`, `run_reply`, `cleanup`, and -`wait_until: "decision_or_attention"` on those tools; omitting them -keeps exact 3.2.1 single-task behavior. - -| Coordination step | Tool | Additive mode | -| --- | --- | --- | -| one submission | `delegate` | `run` with 1–8 isolated assignments | -| inspect | `status` or `task` | `run_id` | -| one aggregate wait | `task` or `tasks` | `wait_until: "decision_or_attention"` | -| answer grouped attention | `task` | one reply for the grouped decision | -| cancel | `cancel` | `run_id` | - -The repository argument is the literal MCP property `repo`. Always send -it as `"repo": "/absolute/path/to/git-worktree"`; `git_root`, -`repository`, and other aliases are unknown properties and fail schema -validation. Cursor Cloud also requires `repo` for the clean local -checkout. Its pushed immutable commit SHA is a separate, Cursor -Cloud-only `starting_ref` property. - -`delegate` records `expected_duration_ms` or `timeout_ms` and a 20% -deadline margin. `task` can wait with `wait_until: "terminal"` until the -recorded deadline, inspect summary/compact/diagnostics views, extend a -deadline with an explicit reason, and deliver a same-session reply. It -does not push unsolicited stdio callbacks across assistant turns. - -Local 3.2.1 review: - -```json -{ - "task_id": "review-auth-refactor", - "provider": "grok", - "repo": "/absolute/path/to/git-worktree", - "role": "review", - "workspace_mode": "managed", - "prompt": "Review the current branch and report concrete correctness risks.", - "expected_duration_ms": 600000 -} -``` - -```json -{ - "task_id": "review-auth-refactor", - "wait_until": "terminal" -} -``` +`cancel`. New integrations use the small semantic `run_request`; legacy +single-task calls and full run envelopes remain supported. -Inspect the receipt before a scoped publisher non-force pushes the task -branch, opens a draft PR, or Sol High or Sol XHigh merges. - -Cursor Cloud implementation: - -```json -{ - "task_id": "cloud-auth-refactor", - "provider": "cursor-cloud", - "repo": "/absolute/path/to/clean-checkout", - "role": "implement", - "starting_ref": "0123456789abcdef0123456789abcdef01234567", - "prompt": "Implement the requested change, run tests, and commit the result.", - "expected_duration_ms": 3600000, - "create_pr": true -} -``` - -Wait on a bounded run without waking on routine text: - -```json -{ - "run_id": "auth-split", - "wait_until": "decision_or_attention" -} -``` - -Internal provider slots remain `grok`, `cursor-local`, `cursor-cloud`, -and `dsh`. Roles are `review` and `implement`. An accepted prompt is -never replayed through another transport. ACPX does not provide an -authoritative prompt-sent acknowledgement, so a DSH task is marked -`dispatch_uncertain` as soon as ACPX spawns and is never replayed -through CLI. - -`create_pr` is a Cursor Cloud-only option and defaults to `false`. Local -tasks reject it. - -For parallel 3.2.1 tasks, coordinate the result set with one `tasks` -wait-any call instead of polling every task. Use `status` and `task` -compact views for routine decisions; open diagnostics pages only when a -task needs attention or fails. Clients that consume `structuredContent` -can opt into `response_mode: "structured"`. Text-only clients should -omit it. The compact single-task projection is capped at 8,192 UTF-8 -bytes by the MCP server. That is a server payload guarantee, not a -measured or claimed hard limit in the Codex desktop renderer. See the -[efficient dogfood workflow](docs/efficient-dogfood.md). - -Terminal managed tasks retain their worktree and branch for Codex -inspection; they are not silently deleted. Watch with `task` -(`wait_until: "terminal"` plus optional `cursor`), then run the -authoritative handoff from the recorded worktree: - -```bash -worktree-bootstrap handoff TASK --repo /absolute/worktree --format markdown -``` - -Inspect commits, diff, tests, and ownership evidence before pushing or -opening a PR. After merge or deliberate discard: - -```bash -worktree-bootstrap lock inspect TASK --repo /absolute/worktree -worktree-bootstrap lock clean TASK --repo /absolute/worktree \ - --policy dead-local --lock-id LOCK_ID -git worktree remove /absolute/worktree -``` - -Clean only the exact corresponding branch and terminal task-state -directory after its receipt is no longer needed. Direct tasks have no -managed worktree; review their caller checkout explicitly. Cursor Cloud -agents are archived after terminal completion where supported, while -their remote branch/PR remains for Codex review. - -```bash -npm --prefix plugins/codex-co-engineer test -node scripts/validate-release.mjs -node scripts/inspector-preflight.mjs -``` - -The authoritative release gate runs against one exact local candidate -using Node 24. GitHub Actions is a credential-free mirror; live Grok, -Cursor, Cursor Cloud, and DSH acceptance is recorded separately because -CI must not send repository content to model providers. - -This repository does not create the GitHub Release, tag, or remote from -the published notes file. - -## License - -MIT. See [LICENSE](LICENSE). +| Guide | What it covers | +| --- | --- | +| [Run tool API](docs/run-tool-api.md) | Submission, waits, attention, diagnostics, and cancellation | +| [Plugin reference](plugins/codex-co-engineer/README.md) | Installed-package setup, authentication, and API examples | +| [Configuration](docs/configuration.md) | Providers, credentials, profiles, and host variables | +| [Contributing](CONTRIBUTING.md) | Local commands, compatibility, and review expectations | +| [Release process](docs/release.md) | Exact-candidate qualification and publication | + +Co-Engineer keeps coordination compact, but token parity with native subagents +has not been established. The [efficiency guide](docs/efficient-dogfood.md) explains +what to measure. Licensed under [MIT](LICENSE). + +Historical [3.4.0 notes](docs/releases/v3.4.0.md) and historical +3.3.0 notes in [the release archive](docs/releases/v3.3.0.md) remain available. diff --git a/SECURITY.md b/SECURITY.md index 4ad931c..e477280 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -87,16 +87,28 @@ The 3.3.0 authority split and honest threat model are commits are handed off for Codex to inspect and push or turn into a PR. Codex remains the merge authority. +## Remembered repository consent + +Native form acceptance can authorize this run only or remember access for the +repository and selected providers. Remembered grants are owner-only local state, +shared across linked worktrees of the same repository. They do not authorize new +providers, changed origins, or replacement repositories. Earlier one-run approvals +are not automatically converted into remembered grants. + +Use the packaged `bin/consent-grants.mjs` command to inspect or revoke grants; +see the [plugin README](plugins/codex-co-engineer/README.md) for invocation and +command arguments. Revocation affects later admissions, not already authorized +running tasks; cancel those tasks separately when required. + ## Credentials and persistent sessions Provider authentication is normal, persistent user authentication: - Grok and Cursor Local use their CLI-managed login/session state. - Cursor Cloud uses `CURSOR_API_KEY` or its owner-only key file. -- DSH Muse uses `MODEL_API_KEY` or the owner-only model-key file created by - `bin/set-model-api-key`. The optional Ox Alpha route uses the separate - `OPENROUTER_API_KEY` or owner-only OpenRouter key file; selecting one route - never substitutes the other route's credential. +- DSH Muse and the optional Ox Alpha route use `OPENROUTER_API_KEY` or the + owner-only OpenRouter key file created by `bin/set-model-api-key`; their + ACP configuration and model selection remain separate. Credentials are not MCP arguments, prompts, task records, or committed files. The supervisor may inherit the user's normal provider environment because diff --git a/docs/adr/0002-native-run-lifecycle.md b/docs/adr/0002-native-run-lifecycle.md new file mode 100644 index 0000000..801ea53 --- /dev/null +++ b/docs/adr/0002-native-run-lifecycle.md @@ -0,0 +1,44 @@ +# Native run lifecycle design + +Status: implemented for 3.4.2; fresh Desktop acceptance follows installation. + +## Evidence + +Native MCP consent and Cursor dispatch work. A run inspection supplied nested lane identity to a supervisor callback expecting a top-level task ID. The run then finalized a partial handoff while its actual provider task continued and eventually completed. Provider text is not carried through the run result. Routine run waits also poll and persist at 50 ms intervals. Existing supervisor integration tests replace these production callbacks and consequently miss their contracts. + +## Ownership + +The task supervisor owns provider process/session identity, execution state, deadlines, cancellation, and provider output. The run owns immutable assignment bindings, consent/admission barriers, and an aggregate view over those tasks. The public adapter projects that view; it must not invent acceptance or lifecycle outcomes. Codex owns final review and integration. + +Keep the established task runtime and the compatibility full-run path. Correct the semantic run boundary coherently rather than replace the provider implementations or introduce another scheduler. + +## Decisions + +1. Give every task-facing lifecycle operation one explicit identity context. Inspect, reconnect, reply, cancel, and result retrieval must reference the same bound task. Test the production bridge using task-store fixtures, not overridden inspect/reconnect/cancel callbacks. +2. Observation failure is uncertainty, not proof of provider termination. Preserve the ability to inspect and cancel that task; reconcile on a later request without replaying its prompt. Never report a terminal run while an owned optional or required provider task is still active or unconfirmed. +3. Preserve provider results durably and return a sanitized, bounded result on the normal run receipt. Use the existing exact task ID for any expanded diagnostics. Ordinary completion must not require reverse engineering the underlying task store. +4. Reuse task-store event waits. Do not write or increment run revisions for unchanged observations. Status remains immediate; waits wake for the requested progress, decision, terminal state, deadline, or caller cancellation. +5. Separate completed work from accepted or verified work. Required cancellation/failure blocks success. A partial handoff is evidence for review, not proof of successful verification. Presentation uses the same lifecycle meanings as admission, including unresolved observations. +6. Make tool metadata explain the native run workflow first; retain supported legacy inputs. Include useful titles, honest annotations, structured output schemas, and short cross-tool server instructions. Skills describe the workflow; the server owns execution and authorization. + +The production bridge also sends the existing compiled child envelope and pins +managed workspaces to its recorded commit. Unsupported model overrides fail +before launch; configured provider defaults are not represented as model +attestation. No new scheduler, dependency, or permission framework is added. + +## Acceptance + +- A native semantic submission completes through the same run ID and returns provider output. +- Inspect/reconnect/cancel operate on the actual bound task IDs across restart. +- A temporary inspection error can recover to the actual final task state without dispatch duplication. +- Required failure/cancellation cannot become verified success; optional live tasks remain owned and cancellable. +- Stable status means stable cursor and no routine persistence; event waits do not hot-poll. +- Results remain bounded and sanitized; old saved runs and legacy task calls remain usable. +- Tests exercise real production adapters with fake task/process boundaries. The release gate is necessary; a separate opt-in host flow establishes app behavior. + +## Official guidance + +- https://developers.openai.com/plugins/concepts/plugins — smallest useful plugin shape; optional UI; headless operation. +- https://developers.openai.com/plugins/build/mcp-server — focused tools, schemas, annotations, stable IDs, useful structured results, and compatibility. +- https://developers.openai.com/plugins/build/skills — concise workflow guidance; server-owned authorization and execution. +- https://developers.openai.com/codex/mcp — local stdio is supported; server instructions describe cross-tool workflows. Public ChatGPT HTTP deployment requirements do not require replacing this local Codex transport. diff --git a/docs/co-engineer-experience-contract.md b/docs/co-engineer-experience-contract.md index e1faf6b..6b25e75 100644 --- a/docs/co-engineer-experience-contract.md +++ b/docs/co-engineer-experience-contract.md @@ -4,13 +4,11 @@ Give Codex a team of external co-engineers without giving up control. This is the public-language contract later UX slices consume. It freezes terminology, Codex speech, activation, the five-tool catalog, and the -one-submission coordination shape. The Luna Max TaskPort skill guarantees -policy and is the host executor for Desktop task tools. The JS adapter -plans and validates those call shapes; it does not invoke host callbacks. -Co-Engineer MCP still does not invoke those host-only tools and must not -add a sixth tool to simulate the Desktop host. This slice does not change -README, manifest, marketplace, asset, version, changelog, or release -surfaces. +one-submission coordination shape. The run card now distinguishes preparation +from execution: it does not claim `running` until every required lane has +authoritative prompt-dispatch evidence. +The optional Luna TaskPort adapter plans and validates host call shapes; +Co-Engineer MCP does not invoke host-only task tools. Owned files: @@ -37,10 +35,10 @@ Exact sentence: Codex remains chief engineer and reviewer. External co-engineers do isolated assigned work. External workers may commit. A scoped publisher -may non-force push only the task branch and open a draft PR. Sol High -or Sol XHigh alone performs regular merge after deterministic -exact-head, current-green-CI, and topology checks. The user retains -release, tag, version, and protected-ref authority. +may non-force push only the task branch and open a draft PR. Codex remains +the merge authority and may merge only after deterministic exact-head, +current-green-CI, and topology checks and the user's authorization. The +user retains release, tag, version, and protected-ref authority. ## Canonical phrases @@ -63,6 +61,8 @@ Exact Codex speech. Substitute `N` with an integer from 1 through 8. Use `assignment` when `N` is 1 and `assignments` when `N` is 2 through 8. - `I am delegating this to Co-Engineer` +- `Co-Engineer is preparing N assignments` +- `Co-Engineer is preparing 1 independent assignment` - `Co-Engineer is running N independent assignments` - `Co-Engineer is running 1 independent assignment` - `Co-Engineer needs one decision from you` @@ -120,21 +120,23 @@ Every bounded-run journey is one submission, one aggregate progress never wakes that wait. Codex does not poll each assignment and does not add a second submission to continue, inspect, or answer. -## Luna Max project manager - -Luna Max is the default routine project manager when the user authorizes -it and Luna Max is actually available. It is not a sixth public phrase -or sixth tool. The Co-Engineer run/event transport stays available and -performs no model polling. Co-Engineer MCP cannot invoke host-only Codex -task tools. The skill guarantees policy and executes the host tools. The -JS adapter plans and validates those call shapes. The Co-Engineer event -store remains source of truth. A user-authorized -pinned Luna Max task wakes on completed, blocked, failed, question, -timeout, or user_update envelopes. Routine progress does not wake a -model. A distinct `merge_ready` envelope may wake Sol High or Sol XHigh -exactly once, and only when exact head and tree, verifier acceptance, -current green CI, zero failed or hidden checks, and topology facts all -pass. +## Optional legacy Luna/Sol host relay + +Ordinary delegation stays in the current Codex task and uses the user's +selected model. When the user explicitly requests the legacy host relay +and Luna Max is available, Codex may pin one Luna Max project-manager +task for that run. It is not a sixth public phrase or sixth tool. The +Co-Engineer run/event transport stays available and performs no model +polling. Co-Engineer MCP cannot invoke host-only Codex task tools. The +skill guarantees policy and executes the host tools. The JS adapter +plans and validates those call shapes. The Co-Engineer event store +remains source of truth. The explicitly requested pinned Luna Max task +wakes on completed, blocked, failed, question, timeout, or user_update +envelopes. Routine progress does not wake a model. A distinct +`merge_ready` envelope may notify Sol High or Sol XHigh exactly once only +when the user explicitly selected that optional relay target and exact +head and tree, verifier acceptance, current green CI, zero failed or +hidden checks, and topology facts all pass. This skill layer can guarantee that policy, identity binding, monotonic cursor resume, sanitized evidence references, and honest degraded @@ -153,19 +155,20 @@ actual thread id, host id, and cursor the host returns. If those tools or Luna Max are unavailable, Codex continues in the current Codex task and says so. It never silently substitutes Sol or invents another model. -Sol Medium is not a mandatory layer. External workers may commit. A +No Luna or Sol model is a mandatory layer. External workers may commit. A scoped publisher may non-force push only the task-owned unprotected Codex branch and open or update a draft pull request after the user authorizes -publication. Sol High or Sol XHigh alone performs regular merge after -deterministic exact-head, current-green-CI, and topology checks, and -remains an on-demand exception adjudicator. The user retains release, -tag, version, and protected-ref authority. No worker or message can -force-push, merge, rebase, tag, release, delete refs, or override -verification. Normal completion never wakes Sol. Sol escalation -is exactly: a verified `merge_ready` packet, conflicting exact evidence or -reviewer verdicts, security or protected-ref risk, composition -ambiguity, repeated deterministic rejection, a release-authority -decision, or explicit user escalation. +publication. Codex remains the merge authority and may regular-merge only +after deterministic exact-head, current-green-CI, and topology checks and +the user's authorization. +The user retains release, tag, version, and protected-ref authority. No +worker or message can force-push, merge, rebase, tag, release, delete +refs, or override verification. In an explicitly requested relay, normal +completion never wakes Sol. Sol notification is limited to a verified +`merge_ready` packet, conflicting exact evidence or reviewer verdicts, +security or protected-ref risk, composition ambiguity, repeated +deterministic rejection, a release-authority decision, or explicit user +escalation. Notification does not grant merge or release authority. Luna may use bounded native read-only subagents for local analysis with inherited capabilities, depth at most 2, counted against the eight-lane diff --git a/docs/co-engineer-quickstart.md b/docs/co-engineer-quickstart.md index cec575c..c8a5c8b 100644 --- a/docs/co-engineer-quickstart.md +++ b/docs/co-engineer-quickstart.md @@ -19,7 +19,10 @@ You: Codex: > I am delegating this to Co-Engineer. Using Grok Co-Engineer. -> Co-Engineer is running 1 independent assignment. +> Co-Engineer is preparing 1 independent assignment. + +After admission and authoritative prompt dispatch, the run card may change +that phrase to `Co-Engineer is running 1 independent assignment`. Codex waits once. When the work is complete, Codex inspects it: @@ -40,9 +43,12 @@ You: Codex: -> I am delegating this to Co-Engineer. Co-Engineer is running 3 +> I am delegating this to Co-Engineer. Co-Engineer is preparing 3 > independent assignments. +The first card says `preparing` until every required lane has authoritative +prompt-dispatch evidence; only then does it say `running`. + Name co-engineers when you care which route takes which assignment: > Use Grok Co-Engineer for the API change and Muse Co-Engineer for the @@ -52,7 +58,7 @@ Codex: > I am delegating this to Co-Engineer. Using Grok Co-Engineer. > Using Muse Co-Engineer. Using Cursor Co-Engineer. Co-Engineer is -> running 3 independent assignments. +> preparing 3 assignments. ## Ask once when nothing is named @@ -74,8 +80,7 @@ You: Codex: > I am delegating this to Co-Engineer. Using Grok Co-Engineer. -> Using Muse Co-Engineer. Co-Engineer is running 2 independent -> assignments. +> Using Muse Co-Engineer. Co-Engineer is preparing 2 assignments. ## Chat with existing work @@ -98,12 +103,10 @@ Co-Engineer finished, and I verified the candidate. You may cancel: The honest shape is up to eight isolated external co-engineers, one bounded run, one coordinated wait, one verified decision. -Codex remains chief engineer and reviewer. External workers may commit. -A scoped publisher may non-force push only the task branch and open a -draft PR. Sol High or Sol XHigh alone may perform a regular merge after -deterministic exact-head/tree, current green CI, verifier, and topology -checks. The user retains version, tag, release, protected-ref, and -product-policy authority. +Codex remains chief engineer and reviewer. External workers may commit within +their assigned scope. Publication and merge require user authorization and Codex review. +Review exact commit and tree identities, verification results, and current CI +before integration. The user retains version, tag, release, and protected-ref authority. Next: diff --git a/docs/co-engineer-troubleshooting.md b/docs/co-engineer-troubleshooting.md index ddeb5bd..f5e221d 100644 --- a/docs/co-engineer-troubleshooting.md +++ b/docs/co-engineer-troubleshooting.md @@ -40,6 +40,21 @@ pinned ACPX `0.13.0`, Cursor SDK `1.0.28`, and the cohesive DSH `0.1.0-rc.7` composition. It does not log you into Grok, Cursor Local, or Cursor Cloud. +**The plugin cache reverts after Windows Desktop reconnects to a remote host.** +A host-only cache restore is not durable if the Desktop client resyncs an older +copy. Add or update the exact local marketplace and install and test the plugin +on the Desktop computer too, following the +[official local-plugin install guide](https://developers.openai.com/plugins/build/plugins): + +```text +codex plugin marketplace add LOCAL_ROOT +codex plugin add codex-co-engineer@codex-co-engineer +``` + +Treat client-to-host resync as suspected until versions or file hashes confirm +it. Do not add an auto-repair cron, replace the cache with a symlink, or disable +unrelated configuration to mask the problem. + ## Worktrees and cleanup **A managed worktree appeared without a receipt.** @@ -67,6 +82,13 @@ Cursor Cloud does not use a local worktree. A bounded-run Cloud lane needs a provider-accessible origin and an exact already-pushed commit SHA. Individual 3.2.1 Cloud tasks still treat that SHA as optional. +**Cursor Cloud completed, but the answer is incomplete.** +`completed` proves terminal lifecycle state; it does not prove that the answer +satisfies the task. Codex must inspect the actual result before accepting it. +If the SDK transcript and result both contain only the same progress sentence, +record the task outcome as incomplete. Do not infer a hidden final answer from +natural language and do not replay the prompt automatically. + ## Credentials **Can I put API keys in the MCP tool arguments?** @@ -75,9 +97,7 @@ must not appear in MCP arguments, prompts, receipts, fixtures, or Git. - Grok: `grok login` - Cursor Local: `cursor-agent login` -- Muse: `MODEL_API_KEY`, `CODEX_CO_ENGINEER_MODEL_API_KEY_FILE`, or - `~/.config/codex-co-engineer/model-api-key` -- Optional Ox Alpha: `OPENROUTER_API_KEY`, +- Muse and optional Ox Alpha: `OPENROUTER_API_KEY`, `CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE`, or `~/.config/codex-co-engineer/openrouter-api-key` - Cursor Cloud: `CURSOR_API_KEY`, `CURSOR_API_KEY_FILE`, or diff --git a/docs/co-engineer-user-journeys.md b/docs/co-engineer-user-journeys.md index be1bdce..b621ae0 100644 --- a/docs/co-engineer-user-journeys.md +++ b/docs/co-engineer-user-journeys.md @@ -18,7 +18,10 @@ The user wants one isolated assignment. User: Review the auth change with Grok Co-Engineer. Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. -Co-Engineer is running 1 independent assignment. +Co-Engineer is preparing 1 independent assignment. + +After every required lane has authoritative prompt-dispatch evidence, the +run card may say that Co-Engineer is running 1 independent assignment. Codex waits once. When the work is complete, Codex inspects it. @@ -35,9 +38,13 @@ The user wants several independent assignments in one run. User: Split this into three isolated independent assignments: API validation, the operator guide, and a review of both diffs. -Codex: I am delegating this to Co-Engineer. Co-Engineer is running 3 +Codex: I am delegating this to Co-Engineer. Co-Engineer is preparing 3 independent assignments. +Only after all three required lanes have authoritative prompt-dispatch +evidence does the run card say that Co-Engineer is running 3 independent +assignments. + Codex does not start three separate runs and does not poll each assignment. One coordinated wait covers the whole run. Independent means the assignments do not share a writer path. The bound is eight. @@ -75,11 +82,11 @@ Co-Engineer finished, and I verified the candidate. Verification is Codex's review of the candidate, not itself a merge. External workers may commit. A scoped publisher may non-force push only -the task branch and open a draft PR. Sol High or Sol XHigh alone -performs regular merge after exact-head, current-green-CI, and topology -checks. The user retains release, tag, version, and protected-ref -authority. Downstream slices must not treat the sentence as a completed -merge. +the task branch and open a draft PR. Codex remains the merge authority +and may merge only after exact-head, current-green-CI, and topology checks +and the user's authorization. The user retains release, tag, version, +and protected-ref authority. Downstream slices must not treat the +sentence as a completed merge. ## Failure/unresolved @@ -104,7 +111,10 @@ the docs. Keep the review on Cursor Co-Engineer. Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. Using Muse Co-Engineer. Using Cursor Co-Engineer. Co-Engineer is -running 3 independent assignments. +preparing 3 independent assignments. + +The card changes to running only after all three required lanes have +authoritative prompt-dispatch evidence. Codex does not pick co-engineers by cost, speed, or a hidden router. Cursor on this computer and Cursor Cloud both stay Using Cursor @@ -126,20 +136,24 @@ invent a default router, and does not submit before the choice exists. User: Grok for the validator. Muse for the docs. Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. -Using Muse Co-Engineer. Co-Engineer is running 2 independent +Using Muse Co-Engineer. Co-Engineer is preparing 2 independent assignments. +After dispatch evidence is authoritative for both required lanes, the +card may say that Co-Engineer is running 2 independent assignments. + That is still one submission and one coordinated wait. The ask happens before delegation. Afterward, chatting with Co-Engineer can inspect, continue, answer grouped attention, or cancel. -## Project manager +## Optional legacy host relay The public phrases stay Delegating to Co-Engineer and Chatting with -Co-Engineer. Luna Max is the default project manager for that same run -when the user authorizes a pinned task, Luna Max is available, and the -host can create a thread, send a message to that thread, and wait on or -read it. +Co-Engineer. Ordinary delegation stays in the current Codex task and uses +the user's selected model. Luna Max becomes project manager for a run only +when the user explicitly requests the legacy relay, authorizes a pinned +task, Luna Max is available, and the host can create a thread, send a +message to that thread, and wait on or read it. User: Delegating to Co-Engineer: pin Luna Max as the project manager for this isolated review. @@ -153,8 +167,9 @@ when the work is completed, blocked, failed, asking a question, timed out, or carrying a user update. Routine progress does not wake it. Normal completion does not wake Sol. When the work is publication-ready and exact head, tree, verifier, current green CI, and topology facts -pass, Codex may ask Sol High or Sol XHigh once to integrate the draft -pull request. +pass, the optional relay may notify Sol High or Sol XHigh once only when +the user explicitly selected that target. The notification does not +grant merge or release authority. If Luna Max or those host task tools are missing, Codex continues in this conversation and says so. It does not substitute Sol. @@ -166,9 +181,10 @@ Across every journey: - Codex remains the chief engineer and reviewer. - External workers may commit. A scoped publisher may non-force push only the task branch and open a draft pull request. Luna Max does - not merge. Sol High or Sol XHigh alone performs regular merge after - exact-head, current-green-CI, and topology checks. The user retains - release, tag, version, and protected-ref authority. + not merge. Codex remains the merge authority and may merge only after + exact-head, current-green-CI, and topology checks and the user's + authorization. The user retains release, tag, version, and + protected-ref authority. - External co-engineers stay isolated. - One bounded run is in flight at a time for this work. - Chatting never becomes a second submission. diff --git a/docs/configuration.md b/docs/configuration.md index 3d5c665..3c1be3b 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -40,17 +40,16 @@ Visitor first-run speech lives in the | `CODEX_CO_ENGINEER_DSH_ACP_COMMAND` | DSH ACP adapter executable. Defaults to `dsh-acp-demo`. | | `CODEX_CO_ENGINEER_DSH_ACP_CONFIG` | Absolute DSH ACP YAML path. | | `CODEX_CO_ENGINEER_DSH_OX_ACP_CONFIG` | Absolute Ox Alpha DSH ACP YAML path. | -| `CODEX_CO_ENGINEER_MODEL_API_KEY_FILE` | Owner-only Muse/DSH model key file. | -| `CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE` | Owner-only OpenRouter key file for Ox Alpha. | +| `CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE` | Owner-only OpenRouter key file for DSH Muse and Ox Alpha. | | `CURSOR_API_KEY_FILE` | Owner-only Cursor Cloud API key file. | -| `MODEL_API_KEY`, `OPENROUTER_API_KEY`, `XAI_API_KEY`, `CURSOR_API_KEY` | Optional process-level provider credentials. | +| `OPENROUTER_API_KEY`, `XAI_API_KEY`, `CURSOR_API_KEY` | Optional process-level provider credentials. | The default DSH configuration is -`~/.config/codex-co-engineer/dsh-acp.yml`; its model key defaults to -`~/.config/codex-co-engineer/model-api-key`. Setup also creates the +`~/.config/codex-co-engineer/dsh-acp.yml`; its OpenRouter key defaults to +`~/.config/codex-co-engineer/openrouter-api-key`. Setup also creates the optional Ox Alpha configuration at -`~/.config/codex-co-engineer/dsh-acp-ox-alpha.yml`; its OpenRouter key -defaults to `~/.config/codex-co-engineer/openrouter-api-key`. Cursor +`~/.config/codex-co-engineer/dsh-acp-ox-alpha.yml`, using that same +OpenRouter key while keeping a separate model/config route. Cursor Cloud also recognizes the existing owner-only `~/.config/cursor-cloud-control/api-key`. @@ -80,7 +79,8 @@ credentials, or provider shell capabilities. Local dispatch fails closed when this boundary cannot be verified. `npm run setup:check` validates the DSH/ACPX composition and CLI, Cursor -SDK, and `worktree-bootstrap` dependency. It does not install or +SDK, Node.js 24+, Python 3.11+, and the bundled `worktree-bootstrap` executable. +No separate worktree-tool installation is needed. It does not install or authenticate Grok or Cursor Local or validate the Cursor Cloud key. Ask Codex to show Co-Engineer status after setup. Its `local_boundary` object validates the systemd/cgroup prerequisite in the MCP process's @@ -93,12 +93,12 @@ allowlisted environment. Any extra Co-Engineer panel is optional, feature-detected, and host-specific. Headless Codex CLI remains a complete fallback. -## Luna Max project manager +## Optional Luna/Sol host relay -Delegating to Co-Engineer and Chatting with Co-Engineer treat a -user-authorized pinned Luna Max task as the default routine project -manager. That is skill policy. It is not a sixth public skill, a sixth -MCP tool, or a change to the Co-Engineer run/event transport. +Ordinary delegation stays in the current Codex task and uses your selected model. +The following legacy Luna/Sol relay is available only when explicitly requested. +It is not a sixth public skill, a sixth MCP tool, or a change to the +Co-Engineer run/event transport. The skill can guarantee policy. Codex is the host executor for Desktop task tools. The JS adapter plans and validates those call shapes and @@ -110,8 +110,8 @@ The skill can guarantee: - one Co-Engineer submission, one `decision_or_attention` wait - no model polling on the Co-Engineer transport -- Luna Max as the default manager when the user authorizes a pinned - task and Luna Max is actually available +- Luna Max as manager for an explicitly requested relay when a pinned + task is authorized and Luna Max is actually available - wake on completed, blocked, failed, question, timeout, or user_update; never on routine progress - a distinct `merge_ready` envelope that may wake Sol High or Sol @@ -266,9 +266,9 @@ Muse. Codex does not invent a default router. ## Authentication -Authenticate Grok and Cursor Local with their normal CLIs. DSH Muse uses -the owner-only model key, DSH Ox Alpha uses the separate owner-only -OpenRouter key, and Cursor Cloud uses its normal API key. Credentials +Authenticate Grok and Cursor Local with their normal CLIs. DSH Muse and DSH +Ox Alpha use the owner-only OpenRouter key, and Cursor Cloud uses its normal +API key. Credentials must not be placed in MCP arguments, prompts, receipts, fixtures, or Git. Provider login state persists in the provider's normal user configuration between Codex tasks. @@ -346,7 +346,8 @@ properties. Pass `expected_duration_ms` or a backwards-compatible `ceil(expected_duration_ms * 1.20)` unless an explicit `timeout_ms` of at least that margin is supplied. -DSH uses Muse Spark 1.2 Contributor when `dsh_model` is omitted. To +DSH uses `meta/muse-spark-1.3-contributor` with xhigh reasoning when +`dsh_model` is omitted. To select Ox Alpha for one task, keep `provider: "dsh"` and add `dsh_model: "stealth/ox-alpha"`. The field is rejected for other providers and unknown model values fail before workspace creation or @@ -395,7 +396,9 @@ events or emit unsolicited stdio callbacks across assistant turns. See ### Bounded runs (additive) The five-tool catalog does not gain a sixth tool. One run is submitted -through `delegate.run` with 1–8 lanes. `status`, `task`, `tasks`, and +through `delegate.run_request` with 1–8 lanes. The legacy full `delegate.run` +envelope remains accepted for compatibility, but skills do not construct it. +`status`, `task`, `tasks`, and `cancel` accept `run_id` to inspect, wait (`wait_until: "decision_or_attention"`), latch attention, reply exactly once (`run_reply`), cancel named lanes, or request proof-bound `cleanup`. diff --git a/docs/credential-isolation.md b/docs/credential-isolation.md index 1418636..8f0ee53 100644 --- a/docs/credential-isolation.md +++ b/docs/credential-isolation.md @@ -41,7 +41,7 @@ platform's push/credential-helper environment. | --- | --- | --- | | Grok | `XAI_API_KEY` when present, Grok command, operational keys | Muse/Ox/Cursor keys, Git/SSH/hosting, key-file paths | | Cursor Local | Cursor command, operational keys (CLI session under `HOME`) | `CURSOR_API_KEY`, Muse/Ox/Grok keys | -| DSH Muse | `MODEL_API_KEY`, Muse config path, DSH/ACPX commands | `OPENROUTER_API_KEY`, Grok/Cursor keys | +| DSH Muse | `OPENROUTER_API_KEY`, Muse config path, DSH/ACPX commands | `MODEL_API_KEY`, Grok/Cursor keys | | DSH Ox | `OPENROUTER_API_KEY`, Ox config path, DSH/ACPX commands | `MODEL_API_KEY`, Grok/Cursor keys | | Cursor Cloud local SDK | `CURSOR_API_KEY` plus bounded repository/ref/prompt data | Other provider keys, Git/SSH/hosting, key-file paths | | Cursor Cloud remote | Credential-free origin URL, pinned SHA, prompt, optional `create_pr` flag | Local credentials, SSH agent, hosting tokens, key files | @@ -133,5 +133,6 @@ P23 `provider-registry.mjs` remains the only composition authority for the four accepted adapters. P29 does not add a fifth slot, a wrapper factory, or ambient discovery. P28 `git-authority.mjs` remains policy at the authority seam; P29 consults its denied-operation vocabulary without -mutating Git. 3.2.1 Muse/Ox credential routing (one route never substitutes -the other) is preserved. +mutating Git. The Muse and Ox routes use the same OpenRouter credential while +keeping separate model/config selections; neither route accepts the retired +`MODEL_API_KEY` credential. diff --git a/docs/cursor-cloud-driver.md b/docs/cursor-cloud-driver.md index ef21565..8e0bad5 100644 --- a/docs/cursor-cloud-driver.md +++ b/docs/cursor-cloud-driver.md @@ -114,5 +114,7 @@ Coverage lives in `test/r1-cursor-cloud-driver.test.mjs` and This slice does NOT qualify a real Cursor Cloud transport. It claims no durable store, scheduler, registry cutover, supervisor cutover, merge -authority, or automatic PR creation. One Luna Max exact review follows; -real Cursor Cloud transport qualification follows exact acceptance. +authority, or automatic PR creation. Any further review uses the current +Codex task and its user-selected model unless the user explicitly requests +the optional legacy host relay. Real Cursor Cloud transport qualification +follows exact acceptance. diff --git a/docs/data-handling.md b/docs/data-handling.md index b577b7e..dec2b30 100644 --- a/docs/data-handling.md +++ b/docs/data-handling.md @@ -60,9 +60,8 @@ their branch/handoff for Codex to inspect before any push or PR creation. ## Credentials Grok and Cursor Local use their normal persistent CLI login/session state. -Cursor Cloud uses its normal API key. DSH Muse uses its normal owner-only -model-key file, while DSH Ox Alpha uses a separate owner-only OpenRouter key -file. Credentials are never accepted as MCP arguments and are not +Cursor Cloud uses its normal API key. DSH Muse and DSH Ox Alpha use the +owner-only OpenRouter key file. Credentials are never accepted as MCP arguments and are not written to task records. Provider workers inherit the trusted user's normal environment; use a dedicated account or narrower environment if that trust model is not appropriate. diff --git a/docs/dsh-acpx-driver.md b/docs/dsh-acpx-driver.md index 6f45836..d0b320e 100644 --- a/docs/dsh-acpx-driver.md +++ b/docs/dsh-acpx-driver.md @@ -14,10 +14,11 @@ capability schema. - Provider slot: `dsh` exactly. Every other provider fails closed with `provider_slot_mismatch`. -- Models: `muse-spark-1.2-contributor` or `stealth/ox-alpha` exactly. Any other - model fails closed with `dsh_model_denied`. The informational identity map - mirrors the shipped supervisor routing (`dsh-acp.yml` + `MODEL_API_KEY` / - `model-api-key`; `dsh-acp-ox-alpha.yml` + `OPENROUTER_API_KEY` / +- Models: `meta/muse-spark-1.3-contributor` or `stealth/ox-alpha` exactly. Any + other model fails closed with `dsh_model_denied`. The Muse model uses + xhigh reasoning. The informational identity map mirrors the shipped + supervisor routing (`dsh-acp.yml` + `OPENROUTER_API_KEY` / + `openrouter-api-key`; `dsh-acp-ox-alpha.yml` + `OPENROUTER_API_KEY` / `openrouter-api-key`) but resolves nothing by itself. - Workspace: `workspace_mode: "managed"` at construction (required, no hidden default), local managed worktree semantics anchored at the immutable run base @@ -125,11 +126,11 @@ A real-transport qualification harness must supply a production port that: 1. resolves the exact config file and credential per model (env first, then the owner-only file), returning their sha256 digests and never the secret bytes; -2. spawns `acpx flow run ` one-shot with the exact envelope text as - the bounded input payload, returning a stable bounded `session_ref`; -3. projects recorded ACPX flow output (session records, NDJSON traces, exit - status) into the closed evidence/page/cancel receipts above, including - needs-attention detection with a bounded question ref; +2. spawns `acpx exec --file -` with JSON output and the exact envelope on + stdin, returning a stable bounded `session_ref`; +3. correlates JSON-RPC responses and session updates, projecting only bounded + received output into the closed evidence/page/cancel receipts above; + outgoing prompt frames must never become receipt content; 4. performs tree-scoped cancellation and reports `confirmed` only after the process group is observed stopped; 5. keeps every receipt free of provider-authored content beyond the closed diff --git a/docs/final-decision-card.md b/docs/final-decision-card.md index 58ea62a..a76c572 100644 --- a/docs/final-decision-card.md +++ b/docs/final-decision-card.md @@ -71,9 +71,13 @@ the closed vocabulary in manifest order. ## Merge authority The card never merges, pushes, rebases, creates a PR, tags, or releases. -`merge.card_can_merge` is always false. Sol High/XHigh alone may -regular-merge after exact-head, exact-tree, current-green-CI, and topology -CAS checks. Those CAS results are recorded on `merge.cas`. +`merge.card_can_merge` is always false. The compatibility fields +`ready_for_sol_merge` and `sol_regular_merge_permitted` report typed +readiness only; they do not select a model or grant authority. Codex remains +the merge authority and may regular-merge only after exact-head, exact-tree, +current-green-CI, and topology CAS checks and the user's authorization. +Those CAS results are recorded on `merge.cas`. The user retains release, +tag, version, and protected-ref authority. ## Non-goals diff --git a/docs/future-work.md b/docs/future-work.md index b26d5c4..bf7406d 100644 --- a/docs/future-work.md +++ b/docs/future-work.md @@ -27,7 +27,7 @@ managed-worktree/run-base semantics, and no merge/PR authority. Launch confirms only after an authoritative ACP acknowledgement; post-spawn loss is `dispatch_uncertain` and is never replayed. The P20 DSH adapter (`DshApxDriverV1`) drives the contract over an injected bounded ACPX -one-shot transport port for Muse Spark 1.2 Contributor and Ox Alpha, +one-shot transport port for Muse Spark 1.3 Contributor and Ox Alpha, with honest post-spawn uncertainty, unsupported same-session reply, bounded recorded-evidence reconcile/restart/cancel behavior, and fail-closed identity/correlation drift denials. Neither adapter's diff --git a/docs/release.md b/docs/release.md index bcb5c6e..00a7d97 100644 --- a/docs/release.md +++ b/docs/release.md @@ -5,7 +5,7 @@ The authoritative gate runs once against one exact clean local candidate: ```sh release-gate plan --repo "$PWD" release-gate run --repo "$PWD" \ - --receipt /tmp/codex-co-engineer-v3.4.0-release-gate.json + --receipt /tmp/codex-co-engineer-v3.4.2-release-gate.json ``` The package supports Node.js 24 and newer. The release gate is intentionally @@ -24,7 +24,7 @@ replacement for the exact-tree local receipt. After the provider-free gate passes: 1. Run `npm run setup:check` on the target host. - This validates DSH/ACPX, the Cursor SDK, and `worktree-bootstrap` + This validates DSH/ACPX, the Cursor SDK, runtime versions, and bundled `worktree-bootstrap` dependencies. It does not install or authenticate Grok or Cursor Local, validate their CLIs or the Cursor Cloud key, or prove the local process boundary. Use the `status` tool to check provider readiness. @@ -34,8 +34,8 @@ After the provider-free gate passes: `KillMode=control-group` solely for descendant cleanup and to survive the launching client; it is not a sandbox or capability restriction. 3. Verify persistent normal authentication for Grok and Cursor Local, the - owner-only DSH Muse key, the separate owner-only OpenRouter key for Ox - Alpha, and the owner-only Cursor Cloud API key. + OpenRouter key for Muse and Ox Alpha (with their separate model configuration), + and the owner-only Cursor Cloud API key. 4. Run one bounded opt-in acceptance task through Grok, Cursor Local, Cursor Cloud, DSH Muse, and DSH Ox Alpha. 5. For local tasks, verify ACP first; if fallback occurs, prove it happened @@ -64,6 +64,49 @@ After the provider-free gate passes: 4-hour `tool_timeout_sec` as a measured Desktop limit until those probes have been run on the shipping host. +## Local release installation identity + +Use a distinct local marketplace name for an unpublished release candidate when +an open project still contains an older plugin under the public marketplace name. +Keep the plugin name and tested plugin bytes unchanged. Remove the conflicting +installed identity through the supported plugin CLI; preserve any dirty source +checkout instead of changing its version label or replacing its files. + +Codex can refresh installed local plugins when listing project marketplaces. +A project source with the same marketplace/plugin identity can replace the +shared cache even when its version is older than the configured global source. +An existing MCP process then retains paths into the removed version. + +After installation, verify the actual project-scoped `plugin/list` operation for +open development checkouts: the candidate must remain installed and enabled, +its complete file inventory must match the qualified source, and old identities +must remain uninstalled. Repeat the inventory check after the host connection +restarts, then run native provider acceptance. CLI marketplace listing alone +and a successful check immediately after copying files do not prove persistence. + +## Native run acceptance + +Qualify the normal semantic `run_request` path as a complete user workflow. +A successful handshake, accepted consent form, or provider prompt dispatch is +an intermediate result. Acceptance requires the same run ID to deliver the +provider result, accurate terminal state, and retained handoff without asking +the operator to discover a hidden task ID or repair configuration. + +Exercise the production task bridge with synthetic task-store fixtures in CI. +Only provider/process boundaries should be replaced for these contract tests; +replacing inspection, cancellation, and result projection would hide the very +interfaces being qualified. Cover completion, transient observation failure, +restart, cancellation, optional active assignments, unchanged cursors, and +bounded results. Verify that idle waits use event notifications and do not +continually rewrite run state. + +After the exact-candidate gate, repeat one authorized host run through native +consent and normal run completion. A result recovered through a legacy task +fallback is useful diagnostic evidence, but it does not pass this acceptance. +Record the tested commit and whether other provider routes were exercised. +The lifecycle ownership decision is +[ADR 0002](adr/0002-native-run-lifecycle.md). + ## Handoff and cleanup Codex reviews and merges. Managed local worktrees remain until their result is @@ -88,26 +131,56 @@ independent provider review pass. `create_pr` is Cloud-only; local workers return commits and handoff evidence for Codex to decide whether a PR should be opened. Never create an empty PR. -## GitHub Release notes - -The 3.4.0 GitHub Release body is -[releases/v3.4.0.md](releases/v3.4.0.md). Keep historical -[releases/v3.3.0.md](releases/v3.3.0.md), -[releases/v3.2.1.md](releases/v3.2.1.md), -[releases/v3.2.0.md](releases/v3.2.0.md), -[releases/v3.1.0.md](releases/v3.1.0.md) and -[releases/v3.1.1.md](releases/v3.1.1.md) in the repository even when -GitHub Releases is empty. Documentation work must not tag, push, or -publish the GitHub Release. - -Capture the exact-tree gate receipt before any later publication: - -```sh -release-gate run --repo "$PWD" \ - --receipt /tmp/codex-co-engineer-v3.4.0-release-gate.json -``` - -When a maintainer later publishes against an exact reviewed `main` SHA, -use the placeholder form in [releases/v3.4.0.md](releases/v3.4.0.md) -(`EXACT_REVIEWED_MAIN_SHA`). Do not invent a tag or remote mutation from -documentation work. +## Authorized GitHub publication + +The release body is [releases/v3.4.2.md](releases/v3.4.2.md). Preserve all historical +release notes, including [3.4.0](releases/v3.4.0.md). Documentation changes alone +are not publication authorization; an explicit maintainer release instruction is. + +1. Fetch public main and reconcile it into the candidate. Both the published + baseline and the accepted local fixes must be ancestors of the release. +2. Review the diff against public main, including deleted files, historical + feature inventory, installation, packaged documentation, and privacy. +3. Run the exact-candidate local gate. Keep private host evidence outside Git. +4. Push the release branch without force and open a release PR. Verify current + GitHub CI and required review before merging; do not bypass failed checks. +5. Merge the reviewed branch, capture the exact resulting main SHA, and verify + that its tree matches the qualified candidate. If content changed, qualify + the new candidate before tagging. +6. Create `v3.4.2` at that reviewed main SHA and publish the body from + `docs/releases/v3.4.2.md`. Verify the remote tag, release body, source download, + and tag-based installation instructions after publication. + +The public release includes source and documentation. Never attach owner-only +live receipts, prompts, credentials, machine configuration, or private logs. + +## Provider boundary regressions + +Before live acceptance, exercise the installed ACPX CLI against local fake ACP +agents for success and model/API rejection. A nonzero exit must retain a useful +sanitized cause; raw outgoing requests and prompts must not enter public logs. +Test fragmented and oversized output, cancellation, deadlines, and cleanup. + +Feed the Cursor adapter realistic SDK completion objects, including duration, +model, and usage metadata. Keep provider metadata separate from verified Git +and identity facts. A passing mock with a reduced response shape is insufficient. + +Native consent checks cover timeout and same-run recovery. Host presentation +latency requires actual host observations; a protocol test does not establish +when the user saw the form. + +## Orchestrator efficiency acceptance + +Measure cold skill/schema loading separately from warm delegation. Compare the +same one-worker and four-worker tasks against native subagents with equivalent +prompts and required result detail. Count all orchestrator-visible requests, +responses, retries and recovery turns; exclude external worker tokens. Record +actual host usage where available and label tokenizer estimates otherwise. +Payload bytes alone do not establish token cost or native-subagent parity. + +Exercise success, slow acknowledgement, one actionable question, cancellation +and failure. The normal path is one semantic submission and one aggregate wait; +routine progress must not require orchestrator polling. Inspect compact results +for retained errors, review artifacts and retrieval references for omitted detail. +Native comparison and live provider acceptance follow the exact-candidate gate +and a refreshed host connection; automated projection tests do not replace them. diff --git a/docs/releases/v3.4.1-baseline.md b/docs/releases/v3.4.1-baseline.md new file mode 100644 index 0000000..eb9d784 --- /dev/null +++ b/docs/releases/v3.4.1-baseline.md @@ -0,0 +1,61 @@ +# Codex-Co-Engineer 3.4.1 baseline + +3.4.1 work starts from the accepted 3.4.0 qualification candidate, not from +the dirty/stale checkout that happened to be present on the host. + +## Source identity + +| Item | Exact value | +| --- | --- | +| accepted commit | `08c68d5d92e33a61556f15d235f496d6a40c1648` | +| repository tree | `6f93a9c5515d0f25f36fba1b72272fe463a69f3c` | +| Co-Engineer plugin tree | `d06426610e6a59d97deb462a9d95c25e4ed748da` | +| qualification bundle SHA-256 | `01f250446aefb413df609a337005e5cf4673a8fc6a551ca3efe951b1c17cd581` | +| source evidence | [`PROVENANCE.json`](../../DesktopQualification/codex-co-engineer-3.4.0-desktop-qualification-08c68d5/PROVENANCE.json) | + +The authoritative qualification package also contains `SHA256SUMS` and +`QUALIFICATION-OVERLAY.md`. The overlay changed only the qualification +identity files and package-local marketplace metadata; it did not change the +accepted source tree. The 3.4.1 worktree is an isolated branch created from +the exact accepted commit. The dirty `main` checkout is intentionally not a +baseline input. + +The qualification-only version overlay has been converted into ordinary +3.4.1 source metadata in the package manifest, plugin manifest, MCP contract, +and repository marketplace catalog. Existing 3.4.0 asset manifests remain +historical provenance and are not rewritten by this release. + +## Regression fixture ledger + +The ten reported dogfood failures are named here before behavior changes so +each remains independently reviewable: + +| Fixture ID | Failure represented | 3.4.1 coverage | +| --- | --- | --- | +| `premature_running` | lanes were called running before launch evidence | admission phases and experience projection | +| `restart_denied_no_replay` | ACP restart could not safely resume | same-session reconnect without a second prompt | +| `untyped_repo_consent` | natural-language approval crossed exposure boundary | typed, scope-bound `approval_ref` | +| `no_upstream_local_sha` | remote/no-upstream source needed a manual mirror | exact-SHA managed-workspace capability seam | +| `hand_constructed_provenance` | callers had to build derived identities/digests | server `run_request` compiler | +| `slow_readiness` | cold status blocked on repeated provider probes | shared in-flight readiness cache and bounded probes | +| `repeated_receipt_attention` | equivalent safe reads prompted repeatedly | capability-aware attention deduplication | +| `silent_deadline_no_handoff` | timeout ended without retained-work evidence | silence watchdog and bounded terminal handoff | +| `non_idempotent_cancel` | repeated cancel calls failed or changed meaning | stable authoritative cancel receipts | +| `oversized_history` | full receipts/arguments expanded task history | compact simple-run transport and byte caps | + +Focused regression coverage is in the new `r1-*` compiler/admission/ +supervisor/transport/experience tests plus the existing 3.4.0 compatibility +fixtures. The original qualification package remains the reference for the +3.4.0 test and release evidence; a fresh 3.4.1 gate receipt must still be +captured against one clean exact candidate before publication. + +## Baseline guardrails + +- The five public tools remain `status`, `delegate`, `task`, `tasks`, and + `cancel`. +- Existing full `run` envelopes and single-task calls remain compatibility + paths. +- No 3.4.0 receipt migration or destructive state rewrite is performed. +- Codex remains reviewer and merge authority. +- Provider-free tests do not imply live provider, Codex Desktop host, or + Windows/WSL acceptance. diff --git a/docs/releases/v3.4.1.md b/docs/releases/v3.4.1.md new file mode 100644 index 0000000..1126b83 --- /dev/null +++ b/docs/releases/v3.4.1.md @@ -0,0 +1,61 @@ +# Codex-Co-Engineer 3.4.1 + +3.4.1 makes bounded Co-Engineer runs dependable enough for required +implementation lanes while preserving the 3.4.0 five-tool surface, safety +guarantees, run envelopes, single-task calls, and existing receipts. + +## What changed + +- `delegate.run_request` is the small semantic ingress. The server observes + the clean exact Git identity and derives idempotency, manifest, + prompt-envelope, child, lane, task, workspace, provider, and dispatch + identities. The legacy full `run` envelope remains accepted. +- Run admission is split from dispatch. Provider readiness, local process + boundary, repository identity, repository-exposure consent, every managed + workspace, and disjoint writer scopes are verified before the first prompt. + A failed required lane therefore produces zero prompts. +- Run and lane phases expose preparation, session readiness, authoritative + prompt dispatch, degraded/attention states, verification, and terminal + handoff separately. The public experience says `preparing` until all + required lanes have authoritative prompt-dispatch evidence. +- ACP recovery reconnects to the recorded session. Prompt-dispatch + uncertainty is never replayed. Unrecoverable post-prompt work and deadlines + produce bounded partial handoffs with Git/worktree evidence and safe next + actions. +- Repository exposure uses a typed, run-bound approval reference. Natural + language is never treated as consent, and an approval does not authorize + push, deployment, merge, secrets, or production changes. +- Safe run-artifact capabilities deduplicate equivalent attention and route one + grouped reply exactly once. Cancellation is idempotent for terminal runs. +- Readiness probes share an in-flight promise, use short per-provider bounds, + and cache warm results for a short TTL. Simple-run responses are compact and + byte-bounded by default for capable clients. + +## Compatibility and safety + +The public MCP catalog remains exactly `status`, `delegate`, `task`, `tasks`, +and `cancel`. Existing 3.4.0 full envelopes, single-task calls, and receipts +remain readable. No provider receives a second prompt after acknowledged +dispatch, and Codex remains the review and merge authority. + +## Qualification dependencies + +This source candidate is not a substitute for host acceptance. Full release +qualification still requires: + +- a host-integrated native repository-exposure consent card that mints an + opaque `approval_ref`; without it the plugin returns an actionable blocked + state before export, +- an official `worktree-bootstrap` release exposing the exact local-SHA, + `--local-only` managed-workspace capability for remote/no-upstream, + detached, and WSL repositories; 3.4.1 deliberately has no raw `git + worktree` fallback that bypasses locks, +- the Codex Desktop regression for + `read_thread(includeOutputs: false)` proving tool outputs and oversized MCP + arguments remain bounded, and +- live Grok, Cursor Local, Cursor Cloud, Muse/DSH, worker restart, attention, + cancel, silence-threshold, Windows/WSL, and exact-tree release acceptance. + +The release note records these dependencies so plugin-only tests cannot be +mistaken for closure of a host trust boundary. Do not tag, push, publish, or +create a release from this document. diff --git a/docs/releases/v3.4.2.md b/docs/releases/v3.4.2.md new file mode 100644 index 0000000..5687181 --- /dev/null +++ b/docs/releases/v3.4.2.md @@ -0,0 +1,265 @@ +# Codex-Co-Engineer 3.4.2 + +**Less setup. Clearer results. More dependable delegation.** + +Co-Engineer 3.4.2 makes the everyday workflow simpler: tell Codex which external +co-engineer should do the work, let the controller prepare the assignment, and +receive the result in the same task. This release brings the semantic launch +work developed for 3.4.1 into the public release, along with fixes found during +live Grok, Cursor, and Muse testing. + +[Installation](../../README.md#install-and-authentication) · +[Configuration](../configuration.md) · [Troubleshooting](../co-engineer-troubleshooting.md) + +## Highlights + +| Before | With 3.4.2 | +| --- | --- | +| Agents spend turns building launch metadata or looking for setup instructions | A small `run_request` describes the assignment; the server derives protected identities and prepares workspaces | +| Each new run can interrupt the user with another sharing decision | Native consent can remember access for the same repository and selected providers | +| A run appears finished while its underlying provider task is still active | Run inspection, waiting, cancellation, and result retrieval follow the same durable task identity | +| Grok progress text and report labels clutter a short answer | Reliable completed tool rounds return the final response; progress remains in task events | +| SDK or transport errors obscure useful provider output | Provider-specific adapters preserve supported result shapes and bounded diagnostic causes | + +## Simpler launches + +Use ordinary language: + +> Use Grok Co-Engineer to review the latest change. Focus on correctness. + +Codex uses `delegate.run_request` with one to eight independent assignments. +The server derives Git identity, workspace bindings, dispatch identity, and +provenance. A saved profile and a separate coordinator are optional. + +- Readiness, repository identity, consent, managed workspaces, and disjoint + writer scopes are checked before dispatch. +- Omitted assignment access follows the role; explicit conflicts fail early. +- Duration estimates are optional. The default is ten minutes with the existing + 20% deadline margin. +- Provider choices remain explicit. Grok and Cursor use their configured provider + defaults; unsupported model overrides fail before launch. +- Missing worker entrypoints are reported before workspace preparation, with + actionable reinstall/restart guidance. +- Worker instructions identify the assigned working directory and clarify that + the controller owns machine receipts. Routine work no longer asks the provider + to reconstruct Co-Engineer's setup or lifecycle machinery. + +The five-tool catalog remains `status`, `delegate`, `task`, `tasks`, and `cancel`. +Existing single-task calls and full `run` envelopes remain supported. + +## Consent that can be remembered + +The native repository-sharing form offers two scopes: + +- **This run only:** authorize the current request. +- **Repository and selected providers:** remember explicit access for subsequent + runs, including linked worktrees of the same repository. + +Accepting the form is sufficient; there is no redundant approval checkbox. +The form allows more time for a human response. Interrupted decisions can be +requested again on the existing run without inventing an approval. + +Remembered grants are owner-only local state. They do not automatically cover +new providers, changed origins, or replacement repositories. Earlier one-run +approvals are not promoted. Revocation applies to future admissions; cancel +already-running work separately when needed. + +From a repository clone: + +```bash +node plugins/codex-co-engineer/bin/consent-grants.mjs list +node plugins/codex-co-engineer/bin/consent-grants.mjs revoke --repo /absolute/path/to/repository +``` + +A host without native MCP form elicitation returns a capability blocker before +repository export. Consent authorizes repository sharing, not an automatic +push, merge, deployment, or production change. + +## Reliable continuation and completion + +The task supervisor owns provider execution. Runs coordinate those tasks and +project their state rather than creating a competing lifecycle. + +- Inspection, reconnect, reply, cancellation, and result retrieval use the same + bound task identity. +- Temporary observation failure stays uncertainty and can reconcile later. + It is not treated as proof that a worker stopped. +- Slow acknowledgement keeps independent work active. Later authoritative + dispatch evidence updates the existing run. +- Accepted or uncertain prompts are never blindly replayed through another + transport. +- Required cancellation or failure cannot become successful completion. Optional + active tasks remain owned and cancellable. +- Event-driven waits avoid routine hot polling and repeated state writes. + Unchanged observations keep a stable cursor. +- The normal run receipt includes provider output and available branch/handoff + information; users do not need to discover a hidden child task ID. + +## Compact coordination and Grok results + +Shorter skills keep installation details outside routine delegation. Semantic +runs return compact coordination results by default; diagnostics remain +available through `task` with `run_id` and `view: "diagnostics"`. + +Structured-capable clients receive structured results with bounded fallback +text. Text-only clients retain compatible full-text receipt encoding. Output +limits remain explicit, including truncation metadata where applicable. + +Grok's ACP stream carries progress and answer text without a dedicated final +message marker. On a successful `end_turn`, Co-Engineer selects text after a +known, fully settled tool round, following the response-framing approach in +Grok's own structured-output client. The returned result and stored provider +report agree; progress stays in the task event log. + +Selection is deliberately conservative. Missing or duplicate tool identities, +unsettled tools, interleaved text, web search, missing final text, output overflow, +failed turns, and other stop reasons retain the existing aggregate output. +No sentence matching, regular-expression trimming, or extra model call is used +to guess the answer. Other providers retain their existing output behavior. + +This improves the amount of result text an orchestrator must read. It does not +establish token parity with native subagents; that requires a controlled host +comparison including tool/schema loading, waits, and recovery. + +## Provider fixes + +### Grok + +- Readiness distinguishes explicit login status from ancillary command failures. +- Working-directory guidance and controller-owned evidence labels reduce + unnecessary navigation and unrequested response sections. +- Final-response framing produces exact short text and JSON-only output in the + tested normal tool-completion path, with conservative fallback elsewhere. + +### Cursor Local and Cursor Cloud + +- SDK discovery uses a stable working directory and reuses successful discovery + within the server process. Deleted inherited working directories produce typed + discovery behavior instead of an opaque failure. +- Cloud completion accepts the documented SDK result shape, including duration, + model, and usage metadata, while retaining strict internal identity checks. +- Cloud sanitization preserves legitimate answer text that overlaps instructions + while still redacting full prompt echoes, credentials, and diagnostic fragments. +- Structured Cursor permission options survive grouped questions and same-session + replies. Ordinary shell punctuation is not treated as a user question. +- Cloud-only runs do not require the local Linux process boundary; mixed and + local runs still do. + +### Muse through DeepSeek Harness + +- The default is **Muse Spark 1.3 Contributor with XHigh reasoning via OpenRouter**. +- The route uses `OPENROUTER_API_KEY` or its owner-only key file. It does not + receive a Meta credential. +- The supported ACPX one-shot path preserves useful bounded model/API errors. +- The optional Ox Alpha route remains a separate configured Muse/DSH model choice. +- ACPX dispatch uncertainty remains explicit and is never treated as permission + to resend an accepted prompt. + +## Cleanup and packaging + +- Worktree Bootstrap 1.1.0 is bundled under the MIT license with a pinned source + digest and provenance record. Local execution uses the package copy, eliminating + a separately installed private dependency. It requires Python 3.11+ and the + Python standard library only. +- Setup checks runtime prerequisites early and preserves provider credentials. + +- Linux ACP cleanup discovers detached descendants from process records. + Failed process-list commands cannot silently hide active children. +- Verification cleanup tolerates processes disappearing during inspection and + parses parenthesized process names correctly. +- Repeated reconciliation avoids repeatedly paying cleanup grace periods while + preserving the initial grace period and ownership checks. +- Setup, configuration, troubleshooting, API, and release guides are included + in the package. Their relative links are validated against the packed inventory. +- Existing incompatible operator configuration is reported instead of overwritten. +- Development candidates can use a distinct local marketplace identity to avoid + a stale project marketplace replacing their shared installed cache. + +## Upgrading + +### From the public 3.4.0 release or an earlier version + +Finish or cancel active runs. In a clean source clone registered as your local +marketplace: + +```bash +git fetch origin tag v3.4.2 +git switch --detach v3.4.2 +npm --prefix plugins/codex-co-engineer run setup +codex plugin remove codex-co-engineer@codex-co-engineer +codex plugin add codex-co-engineer@codex-co-engineer +npm --prefix plugins/codex-co-engineer run setup:check +``` + +Use your actual installed marketplace identity if it differs. Start a new Codex +session and ask for Co-Engineer status. A dirty development clone should be +preserved; install from a separate clean clone rather than resetting it. + +### From a local 3.4.1 or 3.4.2 candidate + +The public version may match a local candidate's label. Reinstall the plugin to +refresh its bytes, then restart the Codex session. Do not assume the version +string proves that the currently running MCP process contains the new build. +Existing durable state retains its 3.4.1 directory identity for compatibility. +Do not delete task state or provider login files as an upgrade step. + +### Migrating a direct Meta Muse profile + +Back up your DSH configuration, then update the Muse route to: + +| Setting | Value | +| --- | --- | +| Provider | `openrouter` | +| Endpoint | `https://openrouter.ai/api/v1` | +| Model | `meta/muse-spark-1.3-contributor` | +| Reasoning | `xhigh` | +| Credential variable | `OPENROUTER_API_KEY` | + +Save your OpenRouter key through `plugins/codex-co-engineer/bin/set-model-api-key` +or the documented owner-only key-file route. Setup reports incompatible existing +profiles rather than silently replacing them. Review custom YAML against +[configuration](../configuration.md) before restarting. + +## Published 3.4.0 compatibility + +This release retains the published usage ledger, provider-event rules, lane-health +and decision reducers, PR-ready decision card, proof-bound cleanup planner, and +optional Luna/Sol host relay. The relay is opt-in; ordinary delegation does not +require a separate manager or change your Codex model. + +## Requirements and known limits + +- Node.js 24+, Git, Python 3.11+ for bundled setup, and Codex plugin support are + required. The release gate runs + on Node.js 24 for reproducibility. +- Local providers require Linux, a working `systemd --user` manager, + `systemd-run` 244+, unified cgroup v2, and Python 3.11+ for the bundled + worktree tool. Lifecycle + control is not a sandbox. +- Native repository consent requires MCP form elicitation. Extra panels are + optional and host-dependent; the conversation works without them. +- Cursor Cloud needs a provider-accessible origin and a pushed immutable SHA. + A feature branch may need an open PR before the provider can see its commit. +- Grok and Cursor authentication remain provider-managed. Co-Engineer does not + promise that a provider session can never expire. +- Managed worktrees and owner-only task state are retained for review. Cleanup + is explicit; no background garbage collector deletes them. +- The advertised four-hour pending-call budget is not a measured universal + Codex Desktop limit. Windows/macOS native local execution is not qualified + by the Linux acceptance evidence. + +## Validation + +The release process binds checks to an exact candidate rather than only a +version label. The provider-free gate covers unit and compatibility tests, +MCP Inspector, process/environment boundaries, reproducibility, provenance, +and package inventories. GitHub CI provides separate portable evidence. + +Live acceptance during this release cycle covered provider completion, +remembered consent, continuation, cancellation, and cleanup. The final Grok +framing change passed real exact-text and JSON-only assignments without a new +consent prompt; both workers stopped normally. Earlier provider and host checks +remain scoped to their tested candidates, not universal platform guarantees. +Private prompts, credentials, local paths, and machine receipts are not release +assets. See [the release process](https://github.com/ajhcs/Codex-Co-Engineer/blob/main/docs/release.md) +for the qualification procedure. diff --git a/docs/run-tool-api.md b/docs/run-tool-api.md index 01cf628..abac0a3 100644 --- a/docs/run-tool-api.md +++ b/docs/run-tool-api.md @@ -1,9 +1,10 @@ -# Run tool API (R-CUTOVER) +# Run tool API (3.4.1) -R-CUTOVER is the additive five-tool wiring of bounded runs onto the -3.2.1 MCP catalog. It does not add a sixth tool. Submit, status, wait, -attention, reply, cancel, and cleanup are parameters and modes on -`status`, `delegate`, `task`, `tasks`, and `cancel`. +3.4.1 keeps the five-tool MCP catalog. Submit, status, wait, attention, +reply, cancel, and cleanup are parameters and modes on `status`, `delegate`, +`task`, `tasks`, and `cancel`. The preferred bounded-run ingress is the small +server-compiled `run_request`; the 3.4.0 full `run` envelope remains accepted +for compatibility and is not constructed by the skills. Owned files: @@ -28,7 +29,7 @@ structured/wait/diagnostic/reply/cancel behavior and response shapes. | Operation | Tool | Additive parameter or mode | | --- | --- | --- | -| submit | `delegate` | `run` | +| submit | `delegate` | `run_request` (preferred), `run` (legacy compatibility) | | status | `status` or `task` | `run_id` | | wait | `task` or `tasks` | `run_id` plus `wait_until: "decision_or_attention"` | | attention | `task` | `run_id` plus `attention` | @@ -42,6 +43,87 @@ run mode is `decision_or_attention`. Routine progress never wakes. Mixing a run body with 3.2.1 `task_id` / `workspace_mode` / `create_pr` / `reply` fields fails closed. +## Simple run request + +Call `delegate` with only semantic intent: + +```json +{ + "run_request": { + "run_id": "vale-hardening", + "repo": "/absolute/repository/path", + "objective": "Implement and review the hardening plan.", + "assignments": [ + { + "assignment_id": "social-implementation", + "provider": "grok", + "role": "implement", + "prompt": "Implement the social ingestion slice.", + "expected_duration_ms": 900000 + } + ] + } +} +``` + +Assignment `access` is optional: `implement` derives `writer`, while `review` +and `verify` derive `read_only`. An explicit value must agree with the role. +Omitting access and supplying its equivalent explicit value produce the same +normalized request. Multiple writer lanes need explicit disjoint write scopes. + +The server observes the clean exact Git identity, resolves the provider model, +and derives the request idempotency key, manifest/prompt-envelope/lane +digests, child and task identities, and managed-workspace policy. Callers +cannot provide those derived fields. A changed objective, assignment, +provider, SHA, or scope produces a different identity. + +The stdio server requests repository exposure through the host's native MCP +form. The form names the repository/base, run, and selected providers. Native +**Accept** is the approval; there is no second approval checkbox. The required +selector clearly defaults to remembering approval for the canonical Git +repository and exactly those providers, with **This run only** as the one-time +choice. A remembered provider subset can be reused across worktrees that share +the same Git common directory. A new provider, changed origin, unrelated or +recreated repository, rejected form, or malformed owner state cannot reuse it. +No earlier run-only approval is migrated. A model-authored boolean or prose +reply does not approve access. Hosts without form elicitation return an +explicit capability blocker before any workspace or prompt dispatch. + +Remembered grants live in the owner-only Co-Engineer state directory and never +contain credentials. Inspect or revoke them with the installed package command: + +```sh +node /absolute/path/to/plugin/bin/consent-grants.mjs list +node /absolute/path/to/plugin/bin/consent-grants.mjs revoke --repo /absolute/path/to/repository +node /absolute/path/to/plugin/bin/consent-grants.mjs revoke --grant-id GRANT_ID_FROM_LIST +``` + +An npm package installation also provides the shorter +`codex-co-engineer-consent` command. If a process stops during a grant update +and later commands report `consent_grant_store_busy`, first verify that no MCP +server or consent command is running, then remove only +`.consent-grants.lock` from the owner-only Co-Engineer state directory. + +A dismissed or interrupted approval remains inspectable. To request the +native form again for a pending run, call `task` with the same `run_id` and +`run_reply: { "request_consent": true }`. This requests a decision; it is not +approval. Ordinary status and wait calls never reopen the form. The existing +opaque `approval_ref` continuation remains available to embedding hosts with +a trusted verifier; ordinary stdio clients do not construct these references. + +Admission has two barriers. The server validates consent, provider and local +boundary readiness, repository identity, every workspace, and disjoint writer +scope before sending any prompt. Pending consent is shown as attention; workspace admission is +`preparing`. The run can say `running` only when every required lane has +authoritative `prompt_dispatched` evidence. A mid-dispatch failure is +`degraded` with exact dispatched, undispatched, and uncertain lane lists. + +Run and lane phases are explicit and receipts are restartable. Prompt-dispatch +uncertainty is never replayed. Post-prompt unrecoverable work produces a +bounded partial handoff with the retained worktree, starting/current SHA, +clean state, changed files, commits, last provider event, recovery class, and +safe next actions. + ## Bounds and selection One run submission carries 1–8 lanes. Provider/model is explicit on each diff --git a/plugins/codex-co-engineer/.codex-plugin/plugin.json b/plugins/codex-co-engineer/.codex-plugin/plugin.json index e6017b3..d376a5a 100644 --- a/plugins/codex-co-engineer/.codex-plugin/plugin.json +++ b/plugins/codex-co-engineer/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "codex-co-engineer", - "version": "3.4.0", + "version": "3.4.2", "description": "Give Codex a team of external co-engineers without giving up control. Delegating to Co-Engineer starts one bounded run. The honest shape is up to eight isolated external co-engineers, one bounded run, one coordinated wait, one verified decision.", "author": { "name": "Codex-Co-Engineer" diff --git a/plugins/codex-co-engineer/.mcp.json b/plugins/codex-co-engineer/.mcp.json index 5e85add..a200e9a 100644 --- a/plugins/codex-co-engineer/.mcp.json +++ b/plugins/codex-co-engineer/.mcp.json @@ -15,13 +15,11 @@ "HOME", "PATH", "XDG_CONFIG_HOME", - "MODEL_API_KEY", "OPENROUTER_API_KEY", "XAI_API_KEY", "CURSOR_API_KEY", "CURSOR_API_KEY_FILE", "CODEX_CO_ENGINEER_STATE_DIR", - "CODEX_CO_ENGINEER_MODEL_API_KEY_FILE", "CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE", "CODEX_CO_ENGINEER_DSH_ACP_CONFIG", "CODEX_CO_ENGINEER_DSH_OX_ACP_CONFIG", diff --git a/plugins/codex-co-engineer/README.md b/plugins/codex-co-engineer/README.md index ae2b4b1..75679aa 100644 --- a/plugins/codex-co-engineer/README.md +++ b/plugins/codex-co-engineer/README.md @@ -1,100 +1,90 @@ # Codex-Co-Engineer -Give Codex a team of external co-engineers without giving up control. +**Give Codex a team. Keep control of the result.** -Codex remains the chief engineer, reviewer, and merge authority. External -co-engineers do isolated assigned work. The honest shape is up to eight -isolated external co-engineers, one bounded run, one coordinated wait, -one verified decision. +Bring Grok, Cursor, and Muse into the same Codex task for implementation, +investigation, or review. Co-Engineer prepares isolated workspaces for up to +eight independent assignments and returns their results for Codex to inspect. +You decide what ships. -`Delegating to Co-Engineer` starts one bounded run. `Chatting with -Co-Engineer` inspects, continues, answers grouped attention, or cancels -that existing run. Public names are `Using Grok Co-Engineer`, `Using -Cursor Co-Engineer`, and `Using Muse Co-Engineer`. Cursor on this -computer and Cursor Cloud both stay Cursor Co-Engineer in public speech. +[Quickstart](docs/co-engineer-quickstart.md) · [Configuration](docs/configuration.md) · +[Troubleshooting](docs/co-engineer-troubleshooting.md) · [3.4.2 release notes](docs/releases/v3.4.2.md) -The package, plugin, and MCP server identifier is `codex-co-engineer`. -The current release candidate is 3.4.0. Publication is bound to the exact -reviewed tag and GitHub Release rather than inferred from this README. +> Use Grok Co-Engineer to review the latest change. Report actionable findings. -Any extra Co-Engineer panel is optional, feature-detected, and -host-specific. Complete headless fallback: the Delegating/Chatting -conversation in Codex CLI is enough. This package does not claim a -Co-Engineer UI on every Codex Desktop host. - -Visitor install, first-run speech, and safety live in the -[repository README](../../README.md). Concise guides: - -- [Quickstart](../../docs/co-engineer-quickstart.md) -- [Troubleshooting](../../docs/co-engineer-troubleshooting.md) -- [3.2.1 migration](../../docs/co-engineer-migration-3.2.1.md) -- [Configuration](../../docs/configuration.md) - -Normal users speak ordinary language. They do not write tool payloads. +Speak naturally; you do not need tool payloads, a profile, or another manager +for an ordinary launch. A separate host panel is optional. The CLI conversation +is a complete workflow. The stable plugin and MCP identifier is `codex-co-engineer`. ## Install and authentication -Fresh-visitor install (clone, then Codex plugin, then setup) is -documented in the [repository README](../../README.md). The commands -below are the package-local scripts from this directory. - -Requirements: - -- Node.js 24 or newer (the release gate is pinned to Node 24) -- Git and `worktree-bootstrap` -- Linux with a working `systemd --user` manager, `systemd-run` 244 or - newer, and unified cgroup v2 for local providers -- the official Grok Build CLI -- `cursor-agent` -- a Cursor Cloud API key -- a Muse/Meta model API key for default DSH use -- an OpenRouter API key when selecting DSH Ox Alpha +### Requirements -From this package directory—the directory containing `package.json` and -`bin/setup.mjs`: +- Node.js 24+, Git, Python 3.11+ for bundled setup, and a current Codex CLI with plugin support. +- For local providers: Linux, `systemd --user`, `systemd-run` 244+, unified + cgroup v2. +- Install and authenticate only the provider routes you intend to use. -```bash -npm run setup -npm run setup:check -npm test -``` +The worktree tool is bundled; no separate `worktree-bootstrap` installation is needed. -From a repository clone, the same scripts are: +### Install from a release clone ```bash +git clone --branch v3.4.2 --single-branch https://github.com/ajhcs/Codex-Co-Engineer.git +cd Codex-Co-Engineer npm --prefix plugins/codex-co-engineer run setup +codex plugin marketplace add "$PWD" +codex plugin add codex-co-engineer@codex-co-engineer npm --prefix plugins/codex-co-engineer run setup:check ``` -Setup installs pinned ACPX `0.13.0`, Cursor SDK `1.0.28`, and the -cohesive official DSH `0.1.0-rc.7` composition. It writes key-free Muse -and Ox Alpha DSH ACP configurations plus an owner-only session -directory; it does not perform login. -`npm run setup:check` validates the DSH/ACPX composition and CLI, Cursor -SDK, and `worktree-bootstrap` dependencies. It does not install or -authenticate Grok or Cursor Local or validate the Cursor Cloud key. Ask -Codex to show Co-Engineer status after setup: `local_boundary` verifies -the Linux systemd/cgroup prerequisite from the MCP process's actual -environment, and local providers are reported unavailable when it fails. -Release acceptance also launches the MCP through the manifest's exact -environment allowlist before local dispatch is allowed. - -Authenticate providers normally so sessions persist across Codex tasks: +Keep the clone as the registered marketplace source. Setup installs pinned ACPX +0.13.0, Cursor SDK 1.0.28, and the DSH 0.1.0-rc.7 composition globally. Use a +user-writable npm global prefix on your `PATH`; a Node version manager is one +way to provide it. Setup creates key-free DSH profiles and owner-only session +storage. It preserves compatible profiles and reports incompatible ones. + +From an installed package directory containing `package.json`, the equivalent +commands are `npm run setup` and `npm run setup:check`. + +### Connect one or more providers + +| Route | Authentication | +| --- | --- | +| Grok | Install the official Grok Build CLI, then `grok login` | +| Cursor Local | Install Cursor CLI, then `cursor-agent login` | +| Cursor Cloud | `CURSOR_API_KEY`, `CURSOR_API_KEY_FILE`, or its owner-only key file | +| Muse / DSH | From the clone, run `plugins/codex-co-engineer/bin/set-model-api-key` for OpenRouter | + +Muse defaults to Muse Spark 1.3 Contributor with XHigh reasoning through +OpenRouter. The optional Ox Alpha route has a separate DSH profile. +Authentication persists in normal provider login state or owner-only key files; +never include credentials in prompts or tool arguments. See +[configuration](docs/configuration.md#authentication) for locations and variables. + +### Verify and start + +Start a new Codex session and ask: **Show Co-Engineer status.** Then name a +provider and describe its first assignment. Setup checks dependencies and DSH +profiles; the live `status` tool checks provider readiness and the MCP process's +local Linux boundary. Setup does not install or authenticate Grok or Cursor. + +### Upgrade + +Finish or cancel active runs, then update your clean registered source clone: ```bash -grok login -cursor-agent login -bin/set-model-api-key +git fetch origin tag v3.4.2 +git switch --detach v3.4.2 +npm --prefix plugins/codex-co-engineer run setup +codex plugin remove codex-co-engineer@codex-co-engineer +codex plugin add codex-co-engineer@codex-co-engineer +npm --prefix plugins/codex-co-engineer run setup:check ``` -DSH Muse uses `MODEL_API_KEY`, `CODEX_CO_ENGINEER_MODEL_API_KEY_FILE`, -or the default owner-only `~/.config/codex-co-engineer/model-api-key`. -DSH Ox Alpha uses `OPENROUTER_API_KEY`, -`CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE`, or -`~/.config/codex-co-engineer/openrouter-api-key`. Cursor Cloud uses -`CURSOR_API_KEY`, `CURSOR_API_KEY_FILE`, or the existing owner-only -`~/.config/cursor-cloud-control/api-key`. Credentials are never MCP -arguments or task receipts. +Restart the Codex session. Use the identity from `codex plugin list` if your +marketplace name differs. Preserve dirty source clones and existing task state. +Direct Meta Muse profiles need the [OpenRouter migration](docs/releases/v3.4.2.md#upgrading). ## Execution and safety model @@ -161,8 +151,13 @@ are visible to that process. **Where should I run setup?** From this package directory (`plugins/codex-co-engineer` in a clone), or with `npm --prefix plugins/codex-co-engineer run setup` from the -repository root. See the repository README for the copy/paste Codex -plugin install. +repository root. The copy/paste plugin registration from the repository root +is: + +```bash +codex plugin marketplace add "$PWD" +codex plugin add codex-co-engineer@codex-co-engineer +``` **A managed worktree appeared without a receipt.** Do not guess or delete it. Inspect `git worktree list` and @@ -182,7 +177,7 @@ must not appear in MCP arguments, prompts, receipts, fixtures, or Git. Use the headless Delegating/Chatting conversation. Missing UI is not a failed package install. -More cases: [troubleshooting](../../docs/co-engineer-troubleshooting.md). +More cases: [troubleshooting](docs/co-engineer-troubleshooting.md). ## Data handling @@ -193,6 +188,31 @@ repository files. Task receipts contain bounded output, provider/session identifiers, branch and PR information, lifecycle state, and runtime identity; they do not contain credentials or the full prompt. +## Remembered repository consent + +The native repository consent form treats the host's **Accept** action as the +approval. Its required selector defaults, as disclosed in the form, to +remembering the canonical repository and selected providers; **This run only** +keeps the approval one-time. Adding a provider or changing repository identity +asks again. Existing run-only approvals are never promoted automatically. + +```bash +node /absolute/path/to/plugin/bin/consent-grants.mjs list +node /absolute/path/to/plugin/bin/consent-grants.mjs revoke --repo /absolute/path/to/repository +node /absolute/path/to/plugin/bin/consent-grants.mjs revoke --grant-id GRANT_ID_FROM_LIST +``` + +An npm installation also exposes `codex-co-engineer-consent`. A plugin install +does not place package bins on the global `PATH`, so use the explicit script +path shown above. If an interrupted update leaves `.consent-grants.lock`, +verify that no MCP server or consent command is running before removing that +one owner-only lock file from the Co-Engineer state directory. + +The grant file is owner-only state. It contains repository identity, normalized +credential-free origin identity, provider IDs, and grant time, never prompts or +credentials. Revocation is observed by an already-running MCP server on its +next request. + ## Advanced Co-Engineer Control/API This section is for operators and Codex internals. Normal users do not @@ -209,18 +229,24 @@ The MCP server exposes five tools: | `cancel` | Stop one owned local process group, Cursor Cloud run, or run | The catalog is still those five tools. Bounded runs use additive -parameters (`run`, `run_id`, `attention`, `run_reply`, `cleanup`, and +parameters (`run_request`, legacy `run`, `run_id`, `attention`, `run_reply`, `cleanup`, and `wait_until: "decision_or_attention"`) on the same tools. Omit them to keep exact 3.2.1 single-task behavior. Run wait is a bounded `decision_or_attention` wait. See -[the run tool API](../../docs/run-tool-api.md). +[the run tool API](docs/run-tool-api.md). -`delegate` requires a stable `task_id`, a provider, an absolute Git -worktree path in the property named `repo`, a prompt, and -`expected_duration_ms` or a backwards-compatible `timeout_ms`. Providers +For a bounded 3.4.2 run, `delegate` accepts the small semantic +`run_request` body. The server derives the clean Git identity, provider +model, task/workspace/dispatch identities, prompt and manifest digests, and +managed-workspace policy. Do not construct the legacy full `run` envelope or +derived provenance; it remains accepted only for compatibility. Providers are `grok`, `cursor-local`, `cursor-cloud`, and `dsh`; roles are `review` and `implement`. -DSH defaults to `muse-spark-1.2-contributor`. Set + +Legacy single-task `delegate` still requires a stable `task_id`, a provider, +an absolute Git worktree path in the property named `repo`, a prompt, and +`expected_duration_ms` or a backwards-compatible `timeout_ms`. +DSH defaults to `meta/muse-spark-1.3-contributor` with xhigh reasoning. Set `dsh_model: "stealth/ox-alpha"` to select the separate OpenRouter-backed Ox Alpha configuration for that task. @@ -264,10 +290,11 @@ The coordination path keeps the same five tools and the no-argument options. Its task snapshots and live event previews are bounded; when present, `progress.detail_hint` directs the caller to `task` for the target's full live event detail. -- Add `response_mode: "structured"` when the client reads authoritative - `structuredContent`. The text content becomes a bounded fallback. If - the property is omitted, `content[0].text` remains the exact full JSON - serialization of `structuredContent` for legacy clients. +- Capable clients default to structured-first bounded responses. Add + `response_mode: "structured"` explicitly when the client advertises + authoritative `structuredContent`; text-only clients may omit it for the + compatible full JSON text receipt. Simple-run status is capped at 24 KiB + and other simple-run receipts at 72 KiB. Terminal provider results are redacted and bounded, including values returned as nested objects. When evidence is clipped, the receipt @@ -278,7 +305,7 @@ is enforced by the MCP server and is not a measured hard limit of the Codex desktop renderer. For an end-to-end pattern, see the repository's -[efficient dogfood guide](../../docs/efficient-dogfood.md). +[efficient dogfood guide](docs/efficient-dogfood.md). ### Provider matrix @@ -298,8 +325,7 @@ acknowledgement. | Variable | Purpose | | --- | --- | | `CODEX_CO_ENGINEER_STATE_DIR` | Owner-only task-state root. | -| `CODEX_CO_ENGINEER_MODEL_API_KEY_FILE` | Owner-only DSH/Muse key file. | -| `CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE` | Owner-only OpenRouter key file for DSH Ox Alpha. | +| `CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE` | Owner-only OpenRouter key file for DSH Muse and Ox Alpha. | | `CODEX_CO_ENGINEER_DSH_ACP_CONFIG` | Absolute DSH ACP YAML path. | | `CODEX_CO_ENGINEER_DSH_OX_ACP_CONFIG` | Absolute Ox Alpha DSH ACP YAML path. | | `CURSOR_API_KEY_FILE` | Owner-only Cursor Cloud key file. | @@ -336,6 +362,32 @@ their remote branch or PR remains for Codex review. ### Examples +Preferred bounded-run submission (the server compiles the protected +identities and provenance): + +```json +{ + "run_request": { + "run_id": "auth-hardening", + "repo": "/absolute/path/to/git-worktree", + "objective": "Implement and review the auth hardening change.", + "assignments": [ + { + "assignment_id": "auth-implementation", + "provider": "grok", + "role": "implement", + "access": "write", + "prompt": "Implement the auth hardening slice and commit it.", + "expected_duration_ms": 900000 + } + ] + } +} +``` + +The following single-task and Cloud examples are legacy compatibility +examples. New skills and callers should use `run_request` for bounded runs. + Local review: ```json diff --git a/plugins/codex-co-engineer/assets/acpx-runtime.manifest.json b/plugins/codex-co-engineer/assets/acpx-runtime.manifest.json index 1158766..9481ce4 100644 --- a/plugins/codex-co-engineer/assets/acpx-runtime.manifest.json +++ b/plugins/codex-co-engineer/assets/acpx-runtime.manifest.json @@ -1,7 +1,7 @@ { "schema": 1, "bundle": "acpx-runtime.mjs", - "bundle_sha512": "sha512-+Zx12JTts8E/58x3tapyoB1HxGwC3e5ibpMy4Bg9Oi0wqFnQXeN4ZkQzjyD2/uXheuvXe94kK+3GuAjfDJsoOw==", + "bundle_sha512": "sha512-x+vqruKkvLflf6Ef/8NmXfN2X2+3M7o9sVdI/YTWejlphXSGUlLNIpv+BylInQkep8fq9Z9HO79vjny+0zJM6g==", "exports": [ "createAcpRuntime", "createAgentRegistry", @@ -17,7 +17,7 @@ }, "hardening_overlay": { "path": "tools/acpx-vendor/src/hardening-overlay.mjs", - "sha512": "sha512-GE+B14YrleuPhpfkO3QO+c+OVLe3em3IdfeBqne1G6Ps/B3YMHOkxnz6syJx9aydHBz+SaFHn1lJDfSlooc/lQ==", + "sha512": "sha512-gCOwrvbgkEVhuvT28X+mPkaeOpnGOgyrVf3K1RtGOP2Pz7XTRob5jN4dC0tvJjahf+wTVy6AQwx9/bCU+vqM6g==", "application": "append_after_upstream_bundle" }, "bundled_packages": [ diff --git a/plugins/codex-co-engineer/assets/acpx-runtime.mjs b/plugins/codex-co-engineer/assets/acpx-runtime.mjs index f7fdfb6..b3e1413 100644 --- a/plugins/codex-co-engineer/assets/acpx-runtime.mjs +++ b/plugins/codex-co-engineer/assets/acpx-runtime.mjs @@ -268,44 +268,94 @@ AsyncEventQueue = class CoEngineerAsyncEventQueue { }; async function coEngineerRememberAgentDescendants(child) { - if (!child?.pid) return new Set(); + if (!child?.pid) return new Map(); const descendants = child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] - ?? (child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] = new Set()); - for (const pid of await listDescendantPids(child.pid)) descendants.add(pid); + ?? (child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] = new Map()); + if (process.platform === 'linux') { + const processTable = coEngineerReadLinuxProcessTable(); + const root = processTable.get(child.pid); + if (!root) { + try { + process.kill(child.pid, 0); + } catch { + return descendants; + } + throw new Error('Could not inspect the live ACP agent in /proc.'); + } + const children = new Map(); + for (const identity of processTable.values()) { + if (identity.state === 'Z') continue; + const siblings = children.get(identity.parentPid) ?? []; + siblings.push(identity); + children.set(identity.parentPid, siblings); + } + const pending = [child.pid]; + const visited = new Set(pending); + for (let index = 0; index < pending.length; index += 1) { + for (const identity of children.get(pending[index]) ?? []) { + if (visited.has(identity.pid)) continue; + visited.add(identity.pid); + descendants.set(identity.pid, identity.startTime); + pending.push(identity.pid); + } + } + for (const identity of processTable.values()) { + if (identity.pid !== child.pid && identity.processGroupId === child.pid && identity.state !== 'Z') { + descendants.set(identity.pid, identity.startTime); + } + } + return descendants; + } + for (const pid of await listDescendantPids(child.pid)) descendants.set(pid, null); for (const pid of await listProcessGroupPids(child.pid)) { - if (pid !== child.pid) descendants.add(pid); + if (pid !== child.pid) descendants.set(pid, null); } return descendants; } -function coEngineerAgentTreeAlive(child) { - if (!child?.pid) return false; - if (isChildProcessRunning(child)) return true; - return coEngineerHasLivePid(child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] ?? new Set()); -} - -/* - * On Linux, a killed detached child can remain as a zombie until its new - * parent reaps it. `kill(pid, 0)` still succeeds for that zombie, but it has - * no running work or handles left. Treat the process as terminated for - * containment waits so a reaper delay cannot consume the close deadline. - */ -function coEngineerPidIsZombie(pid) { - if (process.platform !== 'linux') return false; +function coEngineerReadLinuxProcessIdentity(pid) { try { const stat = readFileSync(`/proc/${pid}/stat`, 'utf8'); const stateOffset = stat.lastIndexOf(')') + 2; - return stateOffset > 1 && stat[stateOffset] === 'Z'; - } catch { - return false; + if (stateOffset <= 1) throw new Error(`Malformed /proc/${pid}/stat.`); + const fields = stat.slice(stateOffset).trim().split(/\s+/u); + const parentPid = Number(fields[1]); + const processGroupId = Number(fields[2]); + const startTime = fields[19]; + if (!Number.isInteger(parentPid) || !Number.isInteger(processGroupId) || !startTime) { + throw new Error(`Malformed /proc/${pid}/stat.`); + } + return { pid, state: fields[0], parentPid, processGroupId, startTime }; + } catch (error) { + if (error?.code === 'ENOENT' || error?.code === 'ESRCH') return null; + throw error; } } +function coEngineerReadLinuxProcessTable() { + const processes = new Map(); + for (const entry of fs.readdirSync('/proc', { withFileTypes: true })) { + if (!entry.isDirectory() || !/^\d+$/u.test(entry.name)) continue; + const identity = coEngineerReadLinuxProcessIdentity(Number(entry.name)); + if (identity) processes.set(identity.pid, identity); + } + return processes; +} + +function coEngineerAgentTreeAlive(child) { + if (!child?.pid) return false; + if (isChildProcessRunning(child)) return true; + return coEngineerHasLivePid(child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] ?? new Map()); +} + function coEngineerHasLivePid(pids) { - for (const pid of pids) { - if (coEngineerPidIsZombie(pid)) { - pids.delete(pid); - continue; + for (const [pid, startTime] of pids) { + if (process.platform === 'linux') { + const identity = coEngineerReadLinuxProcessIdentity(pid); + if (!identity || identity.state === 'Z' || identity.startTime !== startTime) { + pids.delete(pid); + continue; + } } try { process.kill(pid, 0); @@ -322,11 +372,20 @@ async function coEngineerSignalAgentTree(child, signal) { const descendants = await coEngineerRememberAgentDescendants(child); if (process.platform === 'win32') { await killWindowsProcessTree(child.pid, signal); - for (const pid of descendants) await killWindowsProcessTree(pid, signal); + for (const pid of descendants.keys()) await killWindowsProcessTree(pid, signal); return; } if (isChildProcessRunning(child) && hasLiveProcessGroup(child.pid)) sendSignal(-child.pid, signal); - for (const pid of descendants) sendSignal(pid, signal); + for (const [pid, startTime] of descendants) { + if (process.platform === 'linux') { + const identity = coEngineerReadLinuxProcessIdentity(pid); + if (!identity || identity.state === 'Z' || identity.startTime !== startTime) { + descendants.delete(pid); + continue; + } + } + sendSignal(pid, signal); + } } async function coEngineerWaitForAgentTree(child, waitMs) { @@ -461,7 +520,7 @@ AcpClient.prototype.spawnAgentProcess = async function coEngineerSpawnAgentProce detached: process.platform !== 'win32', windowsVerbatimArguments: spawnCommand.windowsVerbatimArguments, }); - spawnedChild[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] = new Set(); + spawnedChild[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] = new Map(); spawnedChild.once('exit', () => { void coEngineerRememberAgentDescendants(spawnedChild).catch(() => {}); }); diff --git a/plugins/codex-co-engineer/bin/consent-grants.mjs b/plugins/codex-co-engineer/bin/consent-grants.mjs new file mode 100755 index 0000000..7763135 --- /dev/null +++ b/plugins/codex-co-engineer/bin/consent-grants.mjs @@ -0,0 +1,54 @@ +#!/usr/bin/env node + +import { createConsentGrantStore } from '../mcp/v3/consent-grants.mjs'; +import { stateRoot } from '../mcp/v3/task-store.mjs'; + +function usage() { + return [ + 'Usage:', + ' codex-co-engineer-consent list', + ' codex-co-engineer-consent revoke --repo /absolute/path/to/repository', + ' codex-co-engineer-consent revoke --grant-id 64_HEX_CHARACTERS', + ].join('\n'); +} + +function fail(message) { + process.stderr.write(`${message}\n${usage()}\n`); + process.exitCode = 2; +} + +async function main(argv) { + const [command, option, value, ...extra] = argv; + if (extra.length > 0 || !['list', 'revoke'].includes(command)) return fail('Unknown consent command.'); + const store = createConsentGrantStore({ root: stateRoot() }); + if (command === 'list') { + if (option !== undefined) return fail('The list command takes no arguments.'); + const grants = await store.list(); + if (grants.length === 0) { + process.stdout.write('No remembered repository consent grants.\n'); + return; + } + for (const grant of grants) { + process.stdout.write([ + grant.grant_id, + ` repository: ${grant.repository}`, + ` origin: ${grant.origin ?? '(none)'}`, + ` providers: ${grant.providers.join(', ')}`, + ` granted: ${grant.granted_at}`, + ].join('\n') + '\n'); + } + return; + } + if (value === undefined || (option !== '--repo' && option !== '--grant-id')) { + return fail('Revoke requires --repo or --grant-id.'); + } + const revoked = await store.revoke(option === '--repo' + ? { repositoryPath: value } + : { grantId: value }); + process.stdout.write(revoked ? 'Consent grant revoked.\n' : 'No matching consent grant.\n'); +} + +main(process.argv.slice(2)).catch((error) => { + process.stderr.write(`Consent grant command failed: ${error?.code ?? 'unknown_error'}: ${error?.message ?? 'unknown error'}\n`); + process.exitCode = 1; +}); diff --git a/plugins/codex-co-engineer/bin/set-model-api-key b/plugins/codex-co-engineer/bin/set-model-api-key index 01ae764..2216e72 100755 --- a/plugins/codex-co-engineer/bin/set-model-api-key +++ b/plugins/codex-co-engineer/bin/set-model-api-key @@ -2,8 +2,8 @@ set -euo pipefail config_home="${XDG_CONFIG_HOME:-${HOME:?}/.config}" -default_file="${config_home}/codex-co-engineer/model-api-key" -secret_file="${CODEX_CO_ENGINEER_MODEL_API_KEY_FILE:-${default_file}}" +default_file="${config_home}/codex-co-engineer/openrouter-api-key" +secret_file="${CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE:-${default_file}}" secret_dir="$(dirname -- "${secret_file}")" umask 077 @@ -14,17 +14,17 @@ cleanup() { } trap cleanup EXIT -IFS= read -r -s -p "Model API key: " model_api_key +IFS= read -r -s -p "OpenRouter API key: " openrouter_api_key printf '\n' -if [[ ${#model_api_key} -lt 10 || "${model_api_key}" =~ [[:space:]] ]]; then +if [[ ${#openrouter_api_key} -lt 10 || "${openrouter_api_key}" =~ [[:space:]] ]]; then printf 'The key was empty, too short, or contained whitespace. Nothing was changed.\n' >&2 exit 1 fi -printf '%s\n' "${model_api_key}" > "${temporary_file}" +printf '%s\n' "${openrouter_api_key}" > "${temporary_file}" chmod 600 "${temporary_file}" mv -- "${temporary_file}" "${secret_file}" chmod 600 "${secret_file}" trap - EXIT -unset model_api_key +unset openrouter_api_key printf 'Saved the provider key to the configured user-secret path.\n' diff --git a/plugins/codex-co-engineer/bin/setup.mjs b/plugins/codex-co-engineer/bin/setup.mjs index 72d94d3..ec3ab62 100755 --- a/plugins/codex-co-engineer/bin/setup.mjs +++ b/plugins/codex-co-engineer/bin/setup.mjs @@ -7,6 +7,8 @@ import path from 'node:path'; import { fileURLToPath } from 'node:url'; import { promisify } from 'node:util'; +import { BUNDLED_WORKTREE_BOOTSTRAP } from '../mcp/v3/worktree-bootstrap-runtime.mjs'; + const run = promisify(execFile); const PLUGIN = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); const VENDOR = path.join(PLUGIN, 'vendor', 'dsh-acp-demo'); @@ -41,10 +43,11 @@ const commands = Object.freeze({ dsh: env.CODEX_CO_ENGINEER_DSH_COMMAND?.trim() || 'dsh', acpx: env.CODEX_CO_ENGINEER_ACPX_COMMAND?.trim() || 'acpx', dshAcp: env.CODEX_CO_ENGINEER_DSH_ACP_COMMAND?.trim() || 'dsh-acp-demo', - worktreeBootstrap: 'worktree-bootstrap', + worktreeBootstrap: BUNDLED_WORKTREE_BOOTSTRAP, }); const DSH_RC7 = '0.1.0-rc.7'; -const MUSE_MODEL = 'muse-spark-1.2-contributor'; +const WORKTREE_BOOTSTRAP_VERSION = '1.1.0'; +const MUSE_MODEL = 'meta/muse-spark-1.3-contributor'; const OX_MODEL = 'stealth/ox-alpha'; const vendorPackage = JSON.parse(await readFile(path.join(VENDOR, 'package.json'), 'utf8')); @@ -134,12 +137,78 @@ async function validConfig(file, { provider, model, apiKeyEnv }) { } } +function versionAtLeast(value, minimumMajor, minimumMinor = 0) { + const match = String(value ?? '').match(/(\d+)\.(\d+)/u); + if (!match) return false; + const major = Number(match[1]); + const minor = Number(match[2]); + return major > minimumMajor || (major === minimumMajor && minor >= minimumMinor); +} + +async function runtimePrerequisites() { + const results = { + node: { + ok: versionAtLeast(process.versions.node, 24), + output: `Node.js ${process.versions.node}`, + required: '>=24.0.0', + }, + }; + try { + const { stdout, stderr } = await run('python3', ['--version'], { + cwd: tmpdir(), + encoding: 'utf8', + timeout: 10_000, + }); + const output = `${stdout}${stderr}`.trim().slice(0, 500); + results.python = { + ok: versionAtLeast(output, 3, 11), + output, + required: '>=3.11', + }; + } catch (error) { + results.python = { + ok: false, + output: `${error?.stdout ?? ''}${error?.stderr ?? ''}`.trim().slice(0, 500) || 'python3 was not found on PATH.', + required: '>=3.11', + }; + } + try { + const { stdout, stderr } = await run(commands.worktreeBootstrap, ['--version'], { + cwd: tmpdir(), + encoding: 'utf8', + timeout: 10_000, + }); + const output = `${stdout}${stderr}`.trim().slice(0, 500); + results.worktreeBootstrap = { + ok: output === `worktree-bootstrap ${WORKTREE_BOOTSTRAP_VERSION}`, + output, + expected: WORKTREE_BOOTSTRAP_VERSION, + source: 'bundled', + }; + } catch (error) { + results.worktreeBootstrap = { + ok: false, + output: `${error?.stdout ?? ''}${error?.stderr ?? ''}`.trim().slice(0, 500), + source: 'bundled', + }; + } + return results; +} + +async function assertRuntimePrerequisites() { + const results = await runtimePrerequisites(); + const failed = Object.entries(results).filter(([, result]) => !result.ok); + if (failed.length > 0) { + const details = failed.map(([name, result]) => `${name}: ${result.output || 'unavailable'}${result.required ? ` (requires ${result.required})` : ''}`).join('; '); + throw new Error(`Unsupported setup runtime: ${details}. Co-Engineer requires Node.js 24+ and Python 3.11+; reinstall the plugin if its bundled worktree-bootstrap is missing or has the wrong version.`); + } +} + async function check() { - const results = {}; + const results = await runtimePrerequisites(); for (const [name, command, args] of [ ['dsh', commands.dsh, ['--version']], ['acpx', commands.acpx, ['--version']], - ['worktreeBootstrap', commands.worktreeBootstrap, ['--version']], ]) { try { const { stdout, stderr } = await run(command, args, { cwd: tmpdir(), encoding: 'utf8', timeout: 10_000 }); @@ -176,7 +245,7 @@ async function check() { results.packages = { ok: false, output: error?.message ?? String(error) }; } results.config = { - ok: await validConfig(configFile, { provider: 'meta', model: MUSE_MODEL, apiKeyEnv: 'MODEL_API_KEY' }), + ok: await validConfig(configFile, { provider: 'openrouter', model: MUSE_MODEL, apiKeyEnv: 'OPENROUTER_API_KEY' }), path: configFile, model: MUSE_MODEL, }; @@ -196,6 +265,7 @@ async function check() { } async function install() { + await assertRuntimePrerequisites(); const staging = await mkdtemp(path.join(tmpdir(), 'co-engineer-dsh-acp-')); try { await run('npm', ['pack', VENDOR, '--pack-destination', staging], { @@ -229,17 +299,20 @@ async function install() { await chmod(persistenceRoot, 0o700); const museProviders = [ ' providers:', - ' meta:', - ' displayName: Meta Model API', - ' apiKeyEnv: MODEL_API_KEY', + ' openrouter:', + ' displayName: OpenRouter', + ' apiKeyEnv: OPENROUTER_API_KEY', ' api: openai-completions', - ' baseURL: https://api.meta.ai/v1', + ' baseURL: https://openrouter.ai/api/v1', + ' reasoning: xhigh', ' models:', ` - id: ${MUSE_MODEL}`, - ' name: Muse Spark 1.2 Contributor', + ' name: Muse Spark 1.3 Contributor', ' contextWindow: 1048576', ' maxTokens: 131072', ' input: [text, image]', + ' reasoningEfforts:', + ' xhigh: xhigh', ]; const oxProviders = [ ' providers:', @@ -286,10 +359,10 @@ async function install() { if (!await exists(configFile)) { await writeFile( configFile, - configYaml({ provider: 'meta', model: MUSE_MODEL, providers: museProviders }), + configYaml({ provider: 'openrouter', model: MUSE_MODEL, providers: museProviders }), { encoding: 'utf8', mode: 0o600, flag: 'wx' }, ); - } else if (!await validConfig(configFile, { provider: 'meta', model: MUSE_MODEL, apiKeyEnv: 'MODEL_API_KEY' })) { + } else if (!await validConfig(configFile, { provider: 'openrouter', model: MUSE_MODEL, apiKeyEnv: 'OPENROUTER_API_KEY' })) { throw new Error(`Existing DSH ACP config is incompatible or not owner-only: ${configFile}`); } if (!await exists(oxConfigFile)) { diff --git a/plugins/codex-co-engineer/docs/co-engineer-migration-3.2.1.md b/plugins/codex-co-engineer/docs/co-engineer-migration-3.2.1.md new file mode 100644 index 0000000..a2e4d97 --- /dev/null +++ b/plugins/codex-co-engineer/docs/co-engineer-migration-3.2.1.md @@ -0,0 +1,100 @@ +# Migrating from Codex-Co-Engineer 3.2.1 + +Give Codex a team of external co-engineers without giving up control. + +3.2.1 is the last single-task visitor path. Later published notes added +a bounded run on the same five-tool catalog. This onboarding describes +that run in public language. It does not publish a new package version. + +The honest shape is up to eight isolated external co-engineers, one +bounded run, one coordinated wait, one verified decision. + +## What stays the same + +- Identifier: `codex-co-engineer` +- Catalog: `status`, `delegate`, `task`, `tasks`, `cancel` +- There is no sixth tool +- Install, setup, and authentication commands +- Pinned ACPX `0.13.0`, Cursor SDK `1.0.28`, and DSH `0.1.0-rc.7` +- Codex remains chief engineer, reviewer, and merge authority + +Omit additive run fields and exact 3.2.1 single-task behavior remains, +including compact views, wait-any, structured transport, the Muse +default, and the optional Ox Alpha selector. + +## What changes for normal use + +Stop constructing a tool payload as the first-run habit. In a new Codex +session say: + +> Delegating to Co-Engineer: review the auth change with Grok Co-Engineer. + +Codex: + +> I am delegating this to Co-Engineer. Using Grok Co-Engineer. +> Co-Engineer is running 1 independent assignment. + +That one submission replaces a 3.2.1 `delegate` plus a separate +`wait_until: "terminal"` loop in the visitor path. Codex waits once, then +inspects: + +> Co-Engineer finished, and I verified the candidate. + +`Chatting with Co-Engineer` inspects, continues, answers grouped +attention, or cancels that existing run. Chatting is not a second +3.2.1-style submit. + +Public names are `Using Grok Co-Engineer`, `Using Cursor Co-Engineer`, +and `Using Muse Co-Engineer`. Cursor Local and Cursor Cloud both stay +Cursor Co-Engineer in public speech. + +## Compatibility facts + +- Direct mode remains available only for 3.2.1 single-task `delegate`. + Bounded-run submissions never use direct mode. +- Individual 3.2.1 Cursor Cloud tasks still treat `starting_ref` as + optional. Every bounded-run Cloud lane must pin one exact + already-pushed provider-visible SHA. +- `create_pr` remains Cursor Cloud-only and defaults to `false`. Local + tasks still reject it. +- Mixing a run body with 3.2.1 `task_id` / `workspace_mode` / + `create_pr` / `reply` fields fails closed. +- Provider and model are explicit or filled from one named data-only + profile. Missing selection is one ask, not a router. + +## UI + +Any extra Co-Engineer panel is optional, feature-detected, and +host-specific. Complete headless fallback remains: Codex CLI with +Delegating and Chatting is enough. Do not treat a missing Codex Desktop +panel as a 3.2.1 incompatibility. + +## Advanced Co-Engineer Control/API + +JSON belongs only in this labeled material and in +[configuration](configuration.md). A 3.2.1 single-task review still +looks like: + +```json +{ + "task_id": "review-auth-refactor", + "provider": "grok", + "repo": "/absolute/path/to/git-worktree", + "role": "review", + "workspace_mode": "managed", + "prompt": "Review the current branch and report concrete correctness risks.", + "expected_duration_ms": 600000 +} +``` + +The repository argument remains the literal property `repo`. Send +`"repo": "/absolute/path/to/git-worktree"`. Do not rename it. + +A bounded-run wait uses `wait_until: "decision_or_attention"` instead of +polling each assignment. Routine progress never wakes that wait. + +## Next + +- [Quickstart](co-engineer-quickstart.md) +- [Troubleshooting](co-engineer-troubleshooting.md) +- [Published 3.3.0 notes](releases/v3.3.0.md) diff --git a/plugins/codex-co-engineer/docs/co-engineer-quickstart.md b/plugins/codex-co-engineer/docs/co-engineer-quickstart.md new file mode 100644 index 0000000..c8a5c8b --- /dev/null +++ b/plugins/codex-co-engineer/docs/co-engineer-quickstart.md @@ -0,0 +1,115 @@ +# Co-Engineer quickstart + +Give Codex a team of external co-engineers without giving up control. + +This is the 60-second path after +[install and authentication](../README.md#install-and-authentication). +Speak in ordinary language. You do not write tool payloads. + +Start a **new** Codex session after the plugin add. Any extra Co-Engineer +panel is optional, feature-detected, and host-specific. If this host has +no panel, keep talking in Codex CLI. That headless path is complete. + +## Delegate one assignment + +You: + +> Delegating to Co-Engineer: review the auth change with Grok Co-Engineer. + +Codex: + +> I am delegating this to Co-Engineer. Using Grok Co-Engineer. +> Co-Engineer is preparing 1 independent assignment. + +After admission and authoritative prompt dispatch, the run card may change +that phrase to `Co-Engineer is running 1 independent assignment`. + +Codex waits once. When the work is complete, Codex inspects it: + +> Co-Engineer finished, and I verified the candidate. + +You still decide whether to keep, change, or discard the result. That +sentence is Codex's review, not a merge, push, or pull request. + +## Delegate several independent assignments + +Independent means the assignments do not share a writer path. The bound +is eight. This is still one bounded run and one coordinated wait. + +You: + +> Split this into three isolated independent assignments: API +> validation, the operator guide, and a review of both diffs. + +Codex: + +> I am delegating this to Co-Engineer. Co-Engineer is preparing 3 +> independent assignments. + +The first card says `preparing` until every required lane has authoritative +prompt-dispatch evidence; only then does it say `running`. + +Name co-engineers when you care which route takes which assignment: + +> Use Grok Co-Engineer for the API change and Muse Co-Engineer for the +> docs. Keep the review on Cursor Co-Engineer. + +Codex: + +> I am delegating this to Co-Engineer. Using Grok Co-Engineer. +> Using Muse Co-Engineer. Using Cursor Co-Engineer. Co-Engineer is +> preparing 3 assignments. + +## Ask once when nothing is named + +If you want a team and have no saved profile and no named co-engineers: + +You: + +> Give Codex a team of external co-engineers without giving up control. +> Split the validator and the docs. + +Codex asks once which co-engineers should take the independent +assignments: Grok, Cursor, or Muse. It does not keep asking and does not +invent a default router. + +You: + +> Grok for the validator. Muse for the docs. + +Codex: + +> I am delegating this to Co-Engineer. Using Grok Co-Engineer. +> Using Muse Co-Engineer. Co-Engineer is preparing 2 assignments. + +## Chat with existing work + +`Chatting with Co-Engineer` never starts a run. It inspects, continues, +answers grouped attention, or cancels work that already exists. + +If Codex groups questions from more than one assignment: + +> Co-Engineer needs one decision from you. + +Answer once. That is not a second delegation. + +If a required assignment fails or stays unresolved, Codex does not say +Co-Engineer finished, and I verified the candidate. You may cancel: + +> Chatting with Co-Engineer: cancel that run. + +## What success looks like + +The honest shape is up to eight isolated external co-engineers, one +bounded run, one coordinated wait, one verified decision. + +Codex remains chief engineer and reviewer. External workers may commit within +their assigned scope. Publication and merge require user authorization and Codex review. +Review exact commit and tree identities, verification results, and current CI +before integration. The user retains version, tag, release, and protected-ref authority. + +Next: + +- [Troubleshooting](co-engineer-troubleshooting.md) +- [3.2.1 migration](co-engineer-migration-3.2.1.md) +- [Configuration](configuration.md) diff --git a/plugins/codex-co-engineer/docs/co-engineer-troubleshooting.md b/plugins/codex-co-engineer/docs/co-engineer-troubleshooting.md new file mode 100644 index 0000000..f5e221d --- /dev/null +++ b/plugins/codex-co-engineer/docs/co-engineer-troubleshooting.md @@ -0,0 +1,142 @@ +# Co-Engineer troubleshooting + +Give Codex a team of external co-engineers without giving up control. + +Ask Codex in ordinary language first. You do not write tool payloads. +Any extra Co-Engineer panel is optional, feature-detected, and +host-specific. Missing UI is not a failed install; the headless +Delegating/Chatting conversation is complete. + +## Status and local readiness + +**How do I check whether Codex-Co-Engineer can dispatch locally?** +Ask Codex: `Show Codex-Co-Engineer status.` Local providers are ready +only when `local_boundary.ready` is true. If it is false, the MCP +process is missing Linux `systemd --user`, `systemd-run` 244+, unified +cgroup v2, or the forwarded user-session locators (`XDG_RUNTIME_DIR`, +`DBUS_SESSION_BUS_ADDRESS`). + +**Setup passed, but local providers are unavailable.** +`setup:check` does not prove the MCP environment. Re-run status from the +actual MCP server process, then confirm the plugin `.mcp.json` allowlist +forwards `HOME`, `PATH`, `XDG_*`, and `DBUS_SESSION_BUS_ADDRESS`. + +Local dispatch fails closed when that boundary cannot be verified. The +check occurs before Codex-Co-Engineer creates a managed worktree, task +receipt, or prompt file. `systemd --user` with +`KillMode=control-group` is a lifecycle/cleanup boundary, not a sandbox. + +## Install path + +**Where is the installed plugin?** +After `codex plugin add codex-co-engineer@codex-co-engineer`, Codex +reports the cached install path. The source package in this repository +is `plugins/codex-co-engineer`. Run `npm run setup` from that source +package (or with `npm --prefix plugins/codex-co-engineer`) rather than +guessing a cache path. + +Start a **new** Codex session after the plugin add. Setup installs +pinned ACPX `0.13.0`, Cursor SDK `1.0.28`, and the cohesive DSH +`0.1.0-rc.7` composition. It does not log you into Grok, Cursor Local, +or Cursor Cloud. + +**The plugin cache reverts after Windows Desktop reconnects to a remote host.** +A host-only cache restore is not durable if the Desktop client resyncs an older +copy. Add or update the exact local marketplace and install and test the plugin +on the Desktop computer too, following the +[official local-plugin install guide](https://developers.openai.com/plugins/build/plugins): + +```text +codex plugin marketplace add LOCAL_ROOT +codex plugin add codex-co-engineer@codex-co-engineer +``` + +Treat client-to-host resync as suspected until versions or file hashes confirm +it. Do not add an auto-repair cron, replace the cache with a symlink, or disable +unrelated configuration to mask the problem. + +## Worktrees and cleanup + +**A managed worktree appeared without a receipt.** +Do not guess or delete it. Inspect `git worktree list` and +`worktree-bootstrap lock inspect`, then clean only an exact identified +task/lock. + +Managed local work keeps one writer per worktree and branch. If +bootstrap fails before an authoritative receipt and path, the supervisor +cannot safely identify an unknown worktree. + +After Codex has inspected a complete candidate, clean only the exact +corresponding branch, lock, and terminal task-state directory. Direct +3.2.1 tasks have no managed worktree; review their caller checkout +explicitly. + +## Cursor Cloud + +**Cursor Cloud returned HTTP 400 for a valid SHA.** +Treat it as a provider visibility failure. Make the commit reachable +from an open PR or the default branch, then retry. Do not replay a +prompt that was already dispatched. + +Cursor Cloud does not use a local worktree. A bounded-run Cloud lane +needs a provider-accessible origin and an exact already-pushed commit +SHA. Individual 3.2.1 Cloud tasks still treat that SHA as optional. + +**Cursor Cloud completed, but the answer is incomplete.** +`completed` proves terminal lifecycle state; it does not prove that the answer +satisfies the task. Codex must inspect the actual result before accepting it. +If the SDK transcript and result both contain only the same progress sentence, +record the task outcome as incomplete. Do not infer a hidden final answer from +natural language and do not replay the prompt automatically. + +## Credentials + +**Can I put API keys in the MCP tool arguments?** +No. Use normal provider login or the owner-only key files. Credentials +must not appear in MCP arguments, prompts, receipts, fixtures, or Git. + +- Grok: `grok login` +- Cursor Local: `cursor-agent login` +- Muse and optional Ox Alpha: `OPENROUTER_API_KEY`, + `CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE`, or + `~/.config/codex-co-engineer/openrouter-api-key` +- Cursor Cloud: `CURSOR_API_KEY`, `CURSOR_API_KEY_FILE`, or + `~/.config/cursor-cloud-control/api-key` + +## Conversation surprises + +**Chatting did nothing.** +`Chatting with Co-Engineer` needs an existing run. It inspects, +continues, answers grouped attention, or cancels. If no run exists, +Codex should say chatting needs existing work and offer to delegate. It +must not silently submit. + +**Codex asked which co-engineers to use.** +That is the one allowed ask when no profile and no named co-engineers +exist. Answer Grok, Cursor, or Muse. Codex does not invent a default +router. + +**Codex did not say the work was verified.** +A required assignment failed, could not be answered, or stayed +unresolved. A required gap blocks a complete candidate. Codex must not +say `Co-Engineer finished, and I verified the candidate.` Cancel is +chatting, not a new delegation. + +**More than one assignment asked a question.** +Codex groups those questions into one decision: + +> Co-Engineer needs one decision from you. + +Answer once. Unaffected assignments keep working. That is not a second +delegation and not a debate loop. + +**This host has no Co-Engineer UI.** +Use the headless conversation. This documentation does not claim a +Co-Engineer UI on every Codex Desktop host. + +## Related + +- [Quickstart](co-engineer-quickstart.md) +- [3.2.1 migration](co-engineer-migration-3.2.1.md) +- [Configuration](configuration.md) +- [Repository README](../README.md) diff --git a/plugins/codex-co-engineer/docs/configuration.md b/plugins/codex-co-engineer/docs/configuration.md new file mode 100644 index 0000000..3c1be3b --- /dev/null +++ b/plugins/codex-co-engineer/docs/configuration.md @@ -0,0 +1,439 @@ +# Configuration + +Give Codex a team of external co-engineers without giving up control. + +Codex-Co-Engineer has no executable project policy file. The only +project-scoped configuration data is the data-only ProfileV1 catalog +described in [Profiles](#profiles); verification commands never come from +profiles and remain a separate owner-maintained `VerificationPolicyV1`. +The approved-command resolver consumes that owner policy plus a closed +Codex/owner `command_id` selection and returns a frozen ExecutionIntent +receipt; it does not execute the command. The constrained verification +runner consumes only a genuine approved-command receipt plus that trusted +policy, runs the exact owner-approved executable and argv once in a +disposable workspace separate from the candidate, and records bounded +sanitized host-observed evidence. It is not wired into the MCP server. +Provider authentication is normal persistent login/session state or an +owner-only key file. The setup command installs the pinned local +composition and creates the default DSH configuration; it never performs +login on the user's behalf. + +Normal users speak ordinary language (`Delegating to Co-Engineer`, +`Chatting with Co-Engineer`, `Using Grok Co-Engineer`, `Using Cursor +Co-Engineer`, `Using Muse Co-Engineer`). They do not write tool +payloads. JSON in this file belongs under +[Advanced Co-Engineer Control/API](#advanced-co-engineer-controlapi). + +Visitor first-run speech lives in the +[repository README](../README.md) and +[quickstart](co-engineer-quickstart.md). + +## Host environment + +| Variable | Purpose | +| --- | --- | +| `CODEX_CO_ENGINEER_STATE_DIR` | Absolute owner-only task-state root. Defaults to the XDG state directory. | +| `CODEX_CO_ENGINEER_GROK_COMMAND` | Grok CLI executable. Defaults to `grok`. | +| `CODEX_CO_ENGINEER_CURSOR_COMMAND` | Cursor Local executable. Defaults to `cursor-agent`. | +| `CODEX_CO_ENGINEER_DSH_COMMAND` | DSH CLI fallback executable. Defaults to `dsh`. | +| `CODEX_CO_ENGINEER_ACPX_COMMAND` | ACPX executable used for DSH. Defaults to `acpx`. | +| `CODEX_CO_ENGINEER_DSH_ACP_COMMAND` | DSH ACP adapter executable. Defaults to `dsh-acp-demo`. | +| `CODEX_CO_ENGINEER_DSH_ACP_CONFIG` | Absolute DSH ACP YAML path. | +| `CODEX_CO_ENGINEER_DSH_OX_ACP_CONFIG` | Absolute Ox Alpha DSH ACP YAML path. | +| `CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE` | Owner-only OpenRouter key file for DSH Muse and Ox Alpha. | +| `CURSOR_API_KEY_FILE` | Owner-only Cursor Cloud API key file. | +| `OPENROUTER_API_KEY`, `XAI_API_KEY`, `CURSOR_API_KEY` | Optional process-level provider credentials. | + +The default DSH configuration is +`~/.config/codex-co-engineer/dsh-acp.yml`; its OpenRouter key defaults to +`~/.config/codex-co-engineer/openrouter-api-key`. Setup also creates the +optional Ox Alpha configuration at +`~/.config/codex-co-engineer/dsh-acp-ox-alpha.yml`, using that same +OpenRouter key while keeping a separate model/config route. Cursor +Cloud also recognizes the existing owner-only +`~/.config/cursor-cloud-control/api-key`. + +The visitor install is clone-first. From a repository checkout: + +```bash +codex plugin marketplace add "$PWD" +codex plugin add codex-co-engineer@codex-co-engineer +npm --prefix plugins/codex-co-engineer run setup +npm --prefix plugins/codex-co-engineer run setup:check +``` + +The same setup scripts also run from `plugins/codex-co-engineer` as +`npm run setup` and `npm run setup:check`. See the [repository +README](../README.md) for the first-run flow. + +The package supports Node.js 24 and newer. The exact release gate is +pinned to Node.js 24 so release receipts are reproducible. + +Local worker launch additionally requires Linux with a working +`systemd --user` manager, `systemd-run` 244 or newer, and a unified +cgroup v2 hierarchy. The manager-owned transient systemd user service +uses `KillMode=control-group` only to make cancellation reach detached +descendants and let the worker survive the launching client; it is not a +sandbox and does not restrict environment, network, filesystem, +credentials, or provider shell capabilities. Local dispatch fails closed +when this boundary cannot be verified. + +`npm run setup:check` validates the DSH/ACPX composition and CLI, Cursor +SDK, Node.js 24+, Python 3.11+, and the bundled `worktree-bootstrap` executable. +No separate worktree-tool installation is needed. It does not install or +authenticate Grok or Cursor Local or validate the Cursor Cloud key. Ask +Codex to show Co-Engineer status after setup. Its `local_boundary` +object validates the systemd/cgroup prerequisite in the MCP process's +real environment; Grok, Cursor Local, and DSH are forced to +`ready: false` when the boundary is unavailable. Local delegation +repeats the check before creating any workspace or task artifact. The +release gate also tests a server launched with only the MCP manifest's +allowlisted environment. + +Any extra Co-Engineer panel is optional, feature-detected, and +host-specific. Headless Codex CLI remains a complete fallback. + +## Optional Luna/Sol host relay + +Ordinary delegation stays in the current Codex task and uses your selected model. +The following legacy Luna/Sol relay is available only when explicitly requested. +It is not a sixth public skill, a sixth MCP tool, or a change to the +Co-Engineer run/event transport. + +The skill can guarantee policy. Codex is the host executor for Desktop +task tools. The JS adapter plans and validates those call shapes and +does not invoke host callbacks. Co-Engineer MCP still cannot invoke +those host-only tools and must not add a sixth tool to simulate the +Desktop host. + +The skill can guarantee: + +- one Co-Engineer submission, one `decision_or_attention` wait +- no model polling on the Co-Engineer transport +- Luna Max as manager for an explicitly requested relay when a pinned + task is authorized and Luna Max is actually available +- wake on completed, blocked, failed, question, timeout, or + user_update; never on routine progress +- a distinct `merge_ready` envelope that may wake Sol High or Sol + XHigh exactly once, and only when exact head and tree, verifier + acceptance, current green CI, zero failed or hidden checks, and + topology facts all pass +- exact identity binding with a monotonic cursor and a bounded + recent-id window, never an unbounded seen-event list +- grouped attention that retains a routing tuple for every + question_id and routes one structured response covering all + answerable questions exactly once +- bounded sanitized evidence references or artifact_ref spill, never + raw transcripts or secrets +- Luna-native read-only subagents for local analysis only, depth at + most 2, counted against the eight-lane ceiling, never a duplicate + external writer assignment +- Sol Medium is not a mandatory layer +- external workers may commit; a scoped publisher may non-force push + only the task branch and open a draft PR; Sol High or Sol XHigh + alone performs regular merge after deterministic exact-head, + current-green-CI, and topology checks; the user retains release, + tag, version, and protected-ref authority +- no worker or message can force-push, merge, rebase, tag, release, + delete refs, or override verification +- honest continuation in the current Codex task when the manager + cannot be pinned + +The following remain host-dependent Codex Desktop task-management +capabilities. Codex executes them. The JS adapter feature-detects and +validates their live call shapes and does not invent them. Co-Engineer +MCP cannot invoke these host-only tools: + +- `create_thread`, which is usable only with `threadId` and `hostId`. + A result containing only `clientThreadId` is setup-pending; do not + send or wait until the host supplies a real `threadId` and `hostId` +- `send_message_to_thread`, which requires `threadId` plus a real + prompt body +- `wait_threads` with `targets` that MUST include `threadId` and MAY + include `hostId` and `afterCursor`, plus a bounded timeout, or + `read_thread` +- optional `set_thread_archived` or `set_thread_pinned` +- binding the actual thread id, host id, and cursor the host returns +- availability of Luna Max, Sol High, or Sol XHigh on the host + +The host has no cancel primitive. Do not invent `cancel_thread`, +`archive_thread`, or `pin_thread`, and do not claim cancellation +support. + +If those tools or Luna Max are missing, Codex reports the degraded +mode and continues in the current Codex task. It never silently +substitutes Sol or another model. Explicit user model overrides and +Grok, Cursor, or Muse co-engineer selection stay intact. There is no +learned routing or semantic memory. Luna does not merge. External +workers may commit. A scoped publisher may non-force push only the +task branch and open a draft PR. Sol High or Sol XHigh alone performs +regular merge after exact-head, current-green-CI, and topology checks. +The user retains release, tag, version, and protected-ref authority. + +## Profiles + +Profiles are owner-authored, data-only selection records used by the +deterministic run resolver. A profile may name a provider, a model, a +role, an expected duration, and bounded non-executable selection policy. +A profile **MUST NOT** define executables, argv, shell strings, command +templates or catalogs (including anything shaped like +`VerificationPolicyV1`), credentials, tokens, secrets, environment +values, moving refs, direct-mode workspace configuration, +merge/push/create-PR authority, or embedded prompt/result content. + +There are exactly two roots: + +| Scope | Path | +| --- | --- | +| Project | `/.codex/co-engineer-profiles.json` | +| Owner | `/codex-co-engineer/profiles.json` | + +Profile names match `^[a-z0-9][a-z0-9._-]{0,63}$`. Catalog files must be +regular non-symlink files of at most 64 KiB holding at most 64 profiles; +duplicate JSON keys are rejected instead of silently last-wins. +Precedence is fixed and deterministic: when both scopes define the same +name, the project record applies and the owner record is reported as +deterministically shadowed. Every loaded profile carries a stable +SHA-256 provenance digest computed over its validated canonical form +plus its exact name, so identical data yields identical digests +regardless of key order or whitespace. The owner-scope directory and +catalog must be owned by the current user and must not be writable by +group or other users. Project-scope ownership follows the repository's +normal access policy. + +The optional `default: true` flag is ordinary to omit. The resolver +applies it only after explicit execution, an assignment-named profile, +and the run `profile` leave an omitted execution unresolved. A profile +whose name is `default` still has no authority by name. + +### Field validation + +Profiles validate against one bounded run grammar shared with assignment +manifests. The grammar is mirrored locally in the profile module and +guarded against drift by shared test fixtures; profile loading imports +no run-manifest runtime module. + +- `provider` is one of `dsh`, `grok`, `cursor-local`, `cursor-cloud`. + Public speech for those slots is Using Muse Co-Engineer, Using Grok + Co-Engineer, and Using Cursor Co-Engineer (Cursor Local and Cursor + Cloud both display as Cursor Co-Engineer). +- `model` may be named beside any explicit provider from that list and + must match the bounded model identifier grammar + `^[A-Za-z0-9][A-Za-z0-9._/:-]{0,127}$` (at most 128 UTF-8 bytes). The + check is syntax and requested-byte size only: profiles carry no + model-membership, availability, qualification, resolution, or + attestation data, and no advertised-model list is ever enforced + against a requested model. Whether a provider actually offers the + named model is attested at preflight, not at authoring time. The + `PROFILE_DSH_MODELS` constant survives only as deprecated + informational compatibility data and is never consulted by validation. +- `role` is `review`, `implement`, or `verify` (read-only verification). +- `expected_duration_ms` is an integer from 1,000 to 86,400,000. +- `default` is optional. When present it must be the primitive boolean + `true` exactly; absence is ordinary. Lookup stays exact-name and a + profile named `default` has no authority by name. The resolver may + use the single `default: true` catalog record only for a truly omitted + execution that neither the assignment profile nor the run `profile` + resolved. Two `default: true` records fail closed; the resolver never + ranks defaults. +- `policy` is data-only selection policy. Today it may contain exactly + `pre_dispatch_provider_preference`: one to four unique known provider + names in the owner's deterministic pre-dispatch preference order. + +Unknown fields are rejected at every level. Fields naming credentials, +environment values, executables/argv/shell/command catalogs or +templates, merge/push/create-PR or protected-ref authority, moving refs, +direct-mode workspace configuration, or embedded prompt/result content +fail closed with dedicated error codes, as do string values that look +like secret material, environment interpolation, shell syntax, or a +branch/ref name - except the grammar-governed top-level `model` +identifier itself, which is an opaque identifier validated only by the +bounded model grammar above, never parsed as a path, ref, command, or +credential. + +Profiles only name selections. `resolveRunSelectionV1` is the +deterministic resolver: explicit assignment `execution.provider`/`model` +wins, then the assignment-named profile, then the run `profile` for +omitted executions, then the single `default: true` record. Availability +`models` null or absent means membership is undeclared for every +provider, including DSH. Unsupported, unavailable, and `not_supported` +routes stay unresolved and are never substituted. Cursor Cloud lanes +still require one pinned already-pushed `starting_ref`. Provider/model +attestation of the effective pair remains a later preflight concern. + +Missing selection in the visitor path is one ask for Grok, Cursor, or +Muse. Codex does not invent a default router. + +## Authentication + +Authenticate Grok and Cursor Local with their normal CLIs. DSH Muse and DSH +Ox Alpha use the owner-only OpenRouter key, and Cursor Cloud uses its normal +API key. Credentials +must not be placed in MCP arguments, prompts, receipts, fixtures, or +Git. Provider login state persists in the provider's normal user +configuration between Codex tasks. + +## State and retention + +Task records, prompts, events, worker logs, runtime identities, and +session data live under the owner-only state root +`$XDG_STATE_HOME/codex-co-engineer` or +`~/.local/state/codex-co-engineer`. Task directories are `0700`; files +are `0600`. Terminal task state is retained until the operator removes +that exact task directory after handoff and review. Never delete the +whole state root or another task's state as cleanup. + +## Advanced Co-Engineer Control/API + +This section is for operators and Codex internals. Normal users do not +construct these payloads. The five-tool catalog remains `status`, +`delegate`, `task`, `tasks`, and `cancel`. There is no sixth tool. + +### Profile catalog example + +```json +{ + "deep-security-review": { + "schema": "codex-co-engineer.profile.v1", + "provider": "dsh", + "model": "stealth/ox-alpha", + "role": "implement", + "expected_duration_ms": 1200000, + "default": true + } +} +``` + +### Whole-catalog snapshots + +`loadProfileCatalogSnapshot(options)` takes the same arguments as +`loadProfiles` and reads both catalogs exactly once. It returns one +detached, deeply frozen snapshot `{ schema, catalog_digest, roots, +sources, profiles, shadowed }`: `profiles` lists normalized ProfileV1 +records (`name`, `scope`, `source`, `definition`, `digest`) in +deterministic name order, `shadowed` keeps project-precedence losers +visible, and `catalog_digest` binds the whole ordered catalog - each +scope's presence and exact source file plus every record's provenance +digest - so content, precedence, ordering, presence, or origin drift +yields a different digest. Per-record provenance digests stay +content-only and path-independent. The snapshot carries no executable, +environment, default, or route-selection behavior; per-name resolution +reuses `findProfile(snapshot, name)`, so a run resolves its +`run_profile` plus every assignment profile from that one read without +rereading files, and no mutable Map or live object escapes the boundary. + +### Run selection + +`resolveRunSelectionV1` consumes one parsed run manifest, one +availability snapshot, one complete capability snapshot, and an optional +branded or load-shaped profile catalog. It never ranks, scores, falls +back, replays, or substitutes a provider. `SelectionRequestV1` identity +binds both recomputed snapshot digests; `request_id` is `sel-` plus +exactly 32 lowercase hex characters. `resolveSelectionAnswersV1` is +pure: an accepted answer batch re-resolves a cloned manifest, echoes the +outstanding request identity, and leaves the original manifest and +snapshots unchanged. + +### Task inputs + +Repository paths, prompts, roles, deadlines, and workspace/PR intent are +inputs to `delegate`; they are not global policy. The absolute Git +worktree path must be passed in the property named `repo`, for example +`"repo": "/absolute/path/to/git-worktree"`. Do not rename it to +`git_root` or `repository`; the strict MCP schema rejects unknown +properties. Pass `expected_duration_ms` or a backwards-compatible +`timeout_ms` so the recorded deadline is +`ceil(expected_duration_ms * 1.20)` unless an explicit `timeout_ms` of +at least that margin is supplied. + +DSH uses `meta/muse-spark-1.3-contributor` with xhigh reasoning when +`dsh_model` is omitted. To +select Ox Alpha for one task, keep `provider: "dsh"` and add +`dsh_model: "stealth/ox-alpha"`. The field is rejected for other +providers and unknown model values fail before workspace creation or +prompt dispatch: + +```json +{ + "task_id": "ox-review", + "provider": "dsh", + "dsh_model": "stealth/ox-alpha", + "repo": "/absolute/path/to/git-worktree", + "prompt": "Review the current branch.", + "expected_duration_ms": 600000 +} +``` + +The bundled Ox profile follows OpenRouter's model metadata: a +1,048,576-token context, a 131,072-token output ceiling, mandatory +reasoning at `max`, and the provider's native temperature `1` / top-p +`0.95` defaults. DSH ACP supports text and raster-image prompts, so the +profile does not over-advertise the model's separate video input +capability. + +For routine coordination, use `task` with `view: "compact"`, or `status` +and `tasks` with `detail: "compact"`. Compact status/task pages preserve +each full task ID so the returned key can be passed unchanged to `task`, +`cancel`, or a wait-any call. `status` accepts `include_tasks` and +`task_limit`; `tasks` accepts a bounded `limit`, opaque keyset `cursor`, +and provider/state filters. To wait for the first change among 1–8 exact +tasks, pass `task_ids`, optional per-task `cursors`, and one shared +`wait_ms` / `wait_until` to `tasks`. Do not mix wait-any fields with +list pagination or filters. + +`task` accepts `wait_until` (`progress` or `terminal`), optional +`wait_ms` (0-14400000), `cursor`, `view` (`summary`, `diagnostics`, or +`compact`), audited deadline extension fields, and a same-session +`reply` object. Terminal waits are event-driven and do not wake on +routine text. Diagnostics are side-effect-free and redacted. Clients +that actually consume `structuredContent` may opt into +`response_mode: "structured"` on any tool; otherwise omit it to preserve +the default compatible text receipt. The server does not stream raw +events or emit unsolicited stdio callbacks across assistant turns. See +[MCP pending-call budget](mcp-pending-call.md) and the +[efficient dogfood workflow](efficient-dogfood.md). + +### Bounded runs (additive) + +The five-tool catalog does not gain a sixth tool. One run is submitted +through `delegate.run_request` with 1–8 lanes. The legacy full `delegate.run` +envelope remains accepted for compatibility, but skills do not construct it. +`status`, `task`, `tasks`, and +`cancel` accept `run_id` to inspect, wait (`wait_until: +"decision_or_attention"`), latch attention, reply exactly once +(`run_reply`), cancel named lanes, or request proof-bound `cleanup`. +Omitted run fields keep the 3.2.1 shapes above, including `view: +"compact"`, `detail: "compact"`, `task_ids` wait-any, and +`response_mode: "structured"`. Provider/model is explicit or filled from +one named profile; mixed explicit/profile values fail closed when they +conflict. The future-harness template is not a provider; parsing +failures dispatch nothing. `decision_or_attention` waits until an +attention or terminal decision (or `wait_ms`). See +[the run tool API](run-tool-api.md). + +### Local providers + +`workspace_mode: "managed"` is the default. It creates one locked +`worktree-bootstrap` worktree and branch per task. Set +`workspace_mode: "direct"` only when direct mutation of the supplied +checkout is intentional. Direct mode does not create a disposable +worktree. Bounded-run submissions reject direct mode. + +### Cursor Cloud + +Cursor Cloud still requires `repo`, identifying the clean local checkout +with a provider-accessible Git origin. It additionally requires an exact +immutable commit SHA in the separate Cursor Cloud-only `starting_ref` +property for every bounded-run Cloud lane. The SHA must already be +pushed; a local branch name or unpushed work is not an acceptable cloud +starting point. Individual 3.2.1 Cursor Cloud tasks still treat +`starting_ref` as optional. +An exact SHA reachable only from a feature branch can remain invisible +to Cursor until the branch is provider-visible through an open pull +request or the default branch. Create the draft PR (or make the commit +reachable from the default branch) before final Cloud acceptance. +Surface an HTTP 400 for an otherwise-valid SHA as a provider visibility +failure and fix reachability before retrying. +`create_pr` is supported only for Cursor Cloud and defaults to `false`. +Local tasks reject `create_pr`; Codex inspects their handoff and commits +before deciding whether to push or open a PR. diff --git a/plugins/codex-co-engineer/docs/efficient-dogfood.md b/plugins/codex-co-engineer/docs/efficient-dogfood.md new file mode 100644 index 0000000..2608f07 --- /dev/null +++ b/plugins/codex-co-engineer/docs/efficient-dogfood.md @@ -0,0 +1,263 @@ +# Efficient Codex-Co-Engineer dogfood workflow + +This workflow for Codex-Co-Engineer 3.2.0 minimizes coordination calls and +repeated receipt content without weakening Codex's review and merge +authority. The core pattern is: + +```text +independent work -> parallel delegation -> one wait-any -> compact inspection + -> diagnostics only for attention/failure -> Codex review +``` + +The examples use neutral task IDs and paths. Every delegated writer still gets +one task, one managed worktree, and one branch. + +## 1. Check readiness without loading receipts + +Use a readiness-only status call before local delegation: + +```json +{ + "detail": "compact", + "include_tasks": false +} +``` + +When recent coordination state is useful, request only the number of compact +cards needed: + +```json +{ + "detail": "compact", + "task_limit": 6 +} +``` + +`task_limit` accepts 0 through 20. `include_tasks: false` is the readiness-only +path and omits task cards. Full status remains available for compatibility and +deep inspection, but it should not be the routine readiness probe. + +## 2. Delegate independent work in parallel + +Split work at ownership boundaries that can be reviewed and merged as separate +commits or pull requests. Submit each task once; never replay a prompt merely +because a waiter disconnected. + +```json +{ + "task_id": "change-api-validation", + "provider": "grok", + "repo": "/absolute/path/to/git-worktree", + "role": "implement", + "workspace_mode": "managed", + "prompt": "Edit only packages/api and its focused tests. Implement the scoped validation change, test it, and commit once.", + "expected_duration_ms": 1800000 +} +``` + +```json +{ + "task_id": "update-operator-guide", + "provider": "cursor-local", + "repo": "/absolute/path/to/git-worktree", + "role": "implement", + "workspace_mode": "managed", + "prompt": "Edit only docs/operator. Improve navigation and repair broken links without changing implementation or tests. Commit once.", + "expected_duration_ms": 1200000 +} +``` + +These two examples own disjoint paths and do not review each other's changing +state. Parallel delegation is appropriate only for independent ownership. Keep +one writer per managed worktree and branch, and serialize changes that edit the +same contract or generated artifact. + +## 3. Replace polling loops with one wait-any + +Coordinate up to eight exact tasks through the existing `tasks` tool: + +```json +{ + "task_ids": [ + "change-api-validation", + "update-operator-guide" + ], + "wait_until": "terminal", + "wait_ms": 3600000 +} +``` + +The wait shares one deadline across all targets and returns when the first task +reaches the requested condition. A terminal wait also wakes for +`needs_attention`, transport loss, silence, or a recorded deadline as +applicable. A timeout returns compact current snapshots for all targets. It +does not cancel provider work. Each wait-any task snapshot and live event +preview is bounded so an eight-target response cannot reproduce eight full +receipts. When an event preview is present, `progress.detail_hint` tells the +caller to use `task` with that target ID for full live event detail. Treat the +preview as coordination evidence, not the complete provider result. + +Continue with the returned event cursor for each unfinished task so already +delivered progress does not wake the next call: + +```json +{ + "task_ids": [ + "change-api-validation", + "update-operator-guide" + ], + "cursors": { + "change-api-validation": "1842", + "update-operator-guide": "967" + }, + "wait_until": "terminal", + "wait_ms": 3600000 +} +``` + +If `cursors` are omitted from a positive progress wait, the current event-log +tail becomes the baseline and the call waits for newer progress. Do not replace +this event-driven flow with short `task` or `tasks` polling intervals. + +Calling `tasks` with no arguments retains the legacy recent-task list exactly. +Wait-any mode is selected by providing `task_ids` with 1 through 8 unique IDs. +Do not mix its wait-any properties (`task_ids`, `cursors`, `wait_ms`, +`wait_until`, `wake_on_needs_attention`) with list properties (`detail`, +`limit`, `cursor`, `provider`, `state`, `status`). `cursor` is one keyset list +cursor; `cursors` maps task IDs to event cursors. + +Review work that depends on an implementation only after the writer is +terminal. Verify its recorded worktree, branch, and commit, then review that +exact checkout rather than starting the review in parallel with the writer: + +```json +{ + "task_id": "review-api-validation", + "provider": "dsh", + "repo": "/absolute/path/to/recorded-writer-worktree", + "role": "review", + "workspace_mode": "direct", + "prompt": "Review the committed implementation on the recorded branch. Do not modify files; report actionable correctness findings.", + "expected_duration_ms": 1200000 +} +``` + +Use `direct` here intentionally only after the writer has stopped. Codex must +confirm that the review receipt still identifies the expected branch and HEAD. + +## 4. Inspect compact first + +For a routine result or handoff, request the bounded compact projection: + +```json +{ + "task_id": "change-api-validation", + "view": "compact" +} +``` + +The compact task's structured JSON is capped at 8,192 UTF-8 bytes. It includes +coordination state, progress cursor, bounded result and handoff previews, and +the diagnostic summary without returning full task or runtime bodies. This is +an MCP-server payload guarantee. It is not evidence of, or a claim about, a +hard payload limit in the Codex desktop renderer. + +Use `view: "summary"` when the full sanitized receipt is actually required. +Use `view: "diagnostics"` only after an attention or failure signal, or when a +task appears stuck: + +```json +{ + "task_id": "change-api-validation", + "view": "diagnostics", + "cursor": "1842", + "max_bytes": 8192 +} +``` + +Diagnostics are redacted, byte-bounded, side-effect-free pages. Follow their +cursor until the required evidence is covered; do not repeatedly reread page +one. Provider terminal results retain the conclusion-oriented tail. Text and +nested values are bounded and redacted, including credentials split across +stream chunks. A clipped result sets `result_truncated: true` and includes +`result_original_chars` when the original count is knowable. Character counts +are Unicode code points; diagnostic and structured payload caps are UTF-8 +bytes. + +## 5. Page task history intentionally + +For coordination history, use a compact keyset page: + +```json +{ + "detail": "compact", + "limit": 20, + "provider": "grok", + "state": "running" +} +``` + +Use the returned opaque `next_cursor` for the next page without changing +`provider`, `state`/`status`, or `detail`. The cursor is bound to those filters +and orders tasks by creation time and task ID, so insertions or deletions do +not create offset-pagination drift. Page size is 1 through 20. Compact mode +defaults to 20; full mode retains the legacy unbounded default when no paging +arguments are supplied. + +## 6. Opt into structured transport only for a capable client + +All five tools accept the optional presentation property: + +```json +{ + "response_mode": "structured" +} +``` + +In structured mode, `structuredContent` is authoritative and +`content[0].text` is only a bounded, redacted fallback with truthful truncation +metadata. Omit `response_mode` for exact backwards compatibility: +`content[0].text` remains `JSON.stringify(structuredContent)`. Structured mode +therefore saves duplicated text only when the calling client actually consumes +`structuredContent`. Keep the property out of routine examples and never set +it for a text-only client, which would otherwise receive only the fallback. + +## 7. Treat preflight as an identity boundary + +Cursor Cloud requires a clean Git checkout and an exact pushed 40-character +commit SHA: + +```json +{ + "task_id": "cloud-release-review", + "provider": "cursor-cloud", + "repo": "/absolute/path/to/clean-checkout", + "role": "review", + "starting_ref": "0123456789abcdef0123456789abcdef01234567", + "prompt": "Review this exact release candidate and report blockers.", + "expected_duration_ms": 3600000, + "create_pr": false +} +``` + +Before any provider SDK call, the supervisor pins the checkout SHA and a +canonical, credential-free provider origin. The worker verifies that the +checkout is still clean, `HEAD` still equals the pinned SHA, and a derived +origin still has the same repository identity. An explicit provider-repository +override remains explicit and is validated as such. If the checkout or origin +changes, create a new task from the intended exact commit; do not allow a stale +receipt to dispatch a different tree. + +Managed local delegation similarly validates the bootstrap receipt against the +actual worktree: task identity, Git root, branch, starting SHA, and current +`HEAD` must agree before provider launch. This prevents a shape-valid but stale +or forged workspace receipt from selecting another tree. + +## Review and merge remain Codex work + +A terminal provider verdict is evidence, not merge authority. Codex should +inspect the compact handoff, open the relevant diagnostics page when needed, +verify the exact diff and tests in the recorded worktree, and only then push, +open, or merge a pull request. Keep task receipts until that handoff is no +longer needed, then clean only the exact recorded worktree, branch, and task +state. diff --git a/plugins/codex-co-engineer/docs/mcp-pending-call.md b/plugins/codex-co-engineer/docs/mcp-pending-call.md new file mode 100644 index 0000000..cec2ef0 --- /dev/null +++ b/plugins/codex-co-engineer/docs/mcp-pending-call.md @@ -0,0 +1,65 @@ +# MCP pending-call budget + +Codex-Co-Engineer 3.2.0 advertises a 4-hour pending MCP tool-call budget so a +`task(wait_until="terminal")` call can cover a multi-hour delegated job +without once-per-minute model wakeups. This is a plugin setting, not a +measured Codex Desktop hard limit. + +| Setting | Value | Source | +| --- | --- | --- | +| Advertised wait budget | 14,400,000 ms (4 hours) | `MCP_PENDING_CALL_BUDGET_MS` | +| Plugin `tool_timeout_sec` | 14405 | `.mcp.json` (budget/1000 + 5s margin) | +| Previous 3.0.2 pair | 60,000 ms / 65s | Implementation setting, not a product limit | +| Measured Desktop 5 / 30 / 240 min | **unmeasured in this worktree** | Must be run on a real Codex Desktop host | + +The supervisor reports this as `status.mcp_pending_call`. +`measured_desktop_limit_ms` is `null` until an operator records a real-host +result. If a wait hits the advertised budget before the task deadline, +`wait_reason` is `transport_budget` and Codex should reconnect from +`event_cursor`. + +## Deterministic fixture (this worktree) + +Unit tests cover short, medium, and multi-hour **deadline math** +(`ceil(expected_duration_ms * 1.20)`) with injected clocks and compact +waits. They do not sleep for 4 hours. + +`scripts/mcp-pending-call-probe.mjs` is the harness for a real pending-call +measurement. Default duration is 2 seconds. It creates a durable running +task, calls `task(wait_until=terminal, wait_ms=N)`, and reports elapsed +time plus whether the MCP server returned before the requested wait. + +## Real-host acceptance procedure + +Run these on the Codex Desktop host that will ship, with the 3.2.0 plugin +installed and a Linux systemd/cgroup-ready environment. Do not run them in +CI. + +1. Confirm `status.local_boundary.ready` is true. +2. Probe from this repository: + + ```bash + node scripts/mcp-pending-call-probe.mjs --wait-ms 120000 + node scripts/mcp-pending-call-probe.mjs --wait-ms 300000 # 5 minutes + node scripts/mcp-pending-call-probe.mjs --wait-ms 1800000 # 30 minutes + node scripts/mcp-pending-call-probe.mjs --wait-ms 14400000 # 4 hours + ``` + +3. Repeat each interval from Codex Desktop itself: delegate a long-running + fixture (or keep a task in `running`) and call + `task({ wait_until: "terminal", wait_ms })` once. The model must not + resume until the tool call returns. +4. While a wait is pending, verify: + - sending another user message does not kill the delegated worker + - closing/restarting Codex Desktop does not kill the delegated worker + - restarting the MCP server leaves the worker running (systemd user + service) and a later `task` call resumes from `event_cursor` + - `cancel` still stops the owned process group + - `notifications/cancelled` on the wait returns `wait_reason: + disconnected` and leaves the task running +5. Record the longest interval that stayed pending without a host-side + timeout. If that interval is shorter than 4 hours, keep reconnecting + from `event_cursor` at that interval and update + `mcp_pending_call.measured_desktop_limit_ms` in a follow-up. + +This worktree must not claim those Desktop intervals were measured. diff --git a/plugins/codex-co-engineer/docs/releases/v3.3.0.md b/plugins/codex-co-engineer/docs/releases/v3.3.0.md new file mode 100644 index 0000000..7fa2bdd --- /dev/null +++ b/plugins/codex-co-engineer/docs/releases/v3.3.0.md @@ -0,0 +1,95 @@ +# Codex-Co-Engineer 3.3.0 + +Repository-side GitHub Release body for **Codex-Co-Engineer 3.3.0**. + +## Bounded runs on the five-tool catalog + +3.3.0 adds a first-class bounded run: 1–8 independent assignments against one +immutable repository/base identity. The MCP catalog remains `status`, +`delegate`, `task`, `tasks`, and `cancel`. A run is additive parameters on +those tools, not a sixth tool. + +- Submit with `delegate` `run` (1–8 lanes). Direct mode, replay, fallback, + merge, push, and create-PR are rejected on that path. +- Inspect with `status` or `task` plus `run_id`. +- Wait with `wait_until: "decision_or_attention"`; routine progress never + wakes. +- Latch attention and deliver one same-session run reply with `task` + `attention` / `run_reply`. +- Cancel or proof-bound cleanup with `cancel` `run_id`. + +Provider and model are explicit on each assignment or filled from one named +data-only profile. `VerificationPolicyV1` is the only executable command +catalog. Codex remains the only final acceptance and merge authority. Frozen +verified child deltas may be composed into one run-owned, single-parent, +non-authoritative candidate; that candidate is never the integration +authority. + +Every 3.3.0 run Cloud lane must pin one exact already-pushed +provider-visible SHA in `starting_ref`. Individual 3.2.1 Cloud tasks still +treat `starting_ref` as optional. + +The repository argument remains the literal MCP property `repo`: + +```json +{ + "task_id": "review-auth-refactor", + "provider": "grok", + "repo": "/absolute/path/to/git-worktree", + "role": "review", + "workspace_mode": "managed", + "prompt": "Review the current branch and return concrete evidence.", + "expected_duration_ms": 600000 +} +``` + +Omit additive run fields to keep that exact 3.2.1 single-task path. + +Wait on a run without waking on routine text: + +```json +{ + "run_id": "auth-split", + "wait_until": "decision_or_attention" +} +``` + +## Compatibility + +- The MCP tool count remains five. +- Omitting run fields preserves exact 3.2.1 single-task, compact, wait-any, + structured transport, DSH Muse default, and Ox Alpha selector behavior. +- Direct mode remains available only for 3.2.1 single-task `delegate`. +- Run submissions never use direct mode. +- Gate B (context-efficiency) and Gate C (credit economics) stay advisory. + +## Out of scope + +3.3.0 does not add semantic memory, cross-run search, learned routing, +protected-branch integration, or automatic garbage collection. + +## Validation + +```bash +npm --prefix plugins/codex-co-engineer test +node scripts/validate-release.mjs +npm --prefix plugins/codex-co-engineer run setup:check +``` + +This document does not publish live provider transcripts or host +measurements. + +## Maintainer publication + +After the exact candidate is reviewed, replace `EXACT_REVIEWED_MAIN_SHA` with +that immutable commit and verify it before publishing: + +```bash +test "$(git rev-parse HEAD)" = "EXACT_REVIEWED_MAIN_SHA" +gh release create v3.3.0 \ + --target EXACT_REVIEWED_MAIN_SHA \ + --title "Codex-Co-Engineer 3.3.0" \ + --notes-file docs/releases/v3.3.0.md +``` + +This document does not authorize tagging, pushing, or release publication. diff --git a/plugins/codex-co-engineer/docs/releases/v3.4.1.md b/plugins/codex-co-engineer/docs/releases/v3.4.1.md new file mode 100644 index 0000000..1126b83 --- /dev/null +++ b/plugins/codex-co-engineer/docs/releases/v3.4.1.md @@ -0,0 +1,61 @@ +# Codex-Co-Engineer 3.4.1 + +3.4.1 makes bounded Co-Engineer runs dependable enough for required +implementation lanes while preserving the 3.4.0 five-tool surface, safety +guarantees, run envelopes, single-task calls, and existing receipts. + +## What changed + +- `delegate.run_request` is the small semantic ingress. The server observes + the clean exact Git identity and derives idempotency, manifest, + prompt-envelope, child, lane, task, workspace, provider, and dispatch + identities. The legacy full `run` envelope remains accepted. +- Run admission is split from dispatch. Provider readiness, local process + boundary, repository identity, repository-exposure consent, every managed + workspace, and disjoint writer scopes are verified before the first prompt. + A failed required lane therefore produces zero prompts. +- Run and lane phases expose preparation, session readiness, authoritative + prompt dispatch, degraded/attention states, verification, and terminal + handoff separately. The public experience says `preparing` until all + required lanes have authoritative prompt-dispatch evidence. +- ACP recovery reconnects to the recorded session. Prompt-dispatch + uncertainty is never replayed. Unrecoverable post-prompt work and deadlines + produce bounded partial handoffs with Git/worktree evidence and safe next + actions. +- Repository exposure uses a typed, run-bound approval reference. Natural + language is never treated as consent, and an approval does not authorize + push, deployment, merge, secrets, or production changes. +- Safe run-artifact capabilities deduplicate equivalent attention and route one + grouped reply exactly once. Cancellation is idempotent for terminal runs. +- Readiness probes share an in-flight promise, use short per-provider bounds, + and cache warm results for a short TTL. Simple-run responses are compact and + byte-bounded by default for capable clients. + +## Compatibility and safety + +The public MCP catalog remains exactly `status`, `delegate`, `task`, `tasks`, +and `cancel`. Existing 3.4.0 full envelopes, single-task calls, and receipts +remain readable. No provider receives a second prompt after acknowledged +dispatch, and Codex remains the review and merge authority. + +## Qualification dependencies + +This source candidate is not a substitute for host acceptance. Full release +qualification still requires: + +- a host-integrated native repository-exposure consent card that mints an + opaque `approval_ref`; without it the plugin returns an actionable blocked + state before export, +- an official `worktree-bootstrap` release exposing the exact local-SHA, + `--local-only` managed-workspace capability for remote/no-upstream, + detached, and WSL repositories; 3.4.1 deliberately has no raw `git + worktree` fallback that bypasses locks, +- the Codex Desktop regression for + `read_thread(includeOutputs: false)` proving tool outputs and oversized MCP + arguments remain bounded, and +- live Grok, Cursor Local, Cursor Cloud, Muse/DSH, worker restart, attention, + cancel, silence-threshold, Windows/WSL, and exact-tree release acceptance. + +The release note records these dependencies so plugin-only tests cannot be +mistaken for closure of a host trust boundary. Do not tag, push, publish, or +create a release from this document. diff --git a/plugins/codex-co-engineer/docs/releases/v3.4.2.md b/plugins/codex-co-engineer/docs/releases/v3.4.2.md new file mode 100644 index 0000000..5687181 --- /dev/null +++ b/plugins/codex-co-engineer/docs/releases/v3.4.2.md @@ -0,0 +1,265 @@ +# Codex-Co-Engineer 3.4.2 + +**Less setup. Clearer results. More dependable delegation.** + +Co-Engineer 3.4.2 makes the everyday workflow simpler: tell Codex which external +co-engineer should do the work, let the controller prepare the assignment, and +receive the result in the same task. This release brings the semantic launch +work developed for 3.4.1 into the public release, along with fixes found during +live Grok, Cursor, and Muse testing. + +[Installation](../../README.md#install-and-authentication) · +[Configuration](../configuration.md) · [Troubleshooting](../co-engineer-troubleshooting.md) + +## Highlights + +| Before | With 3.4.2 | +| --- | --- | +| Agents spend turns building launch metadata or looking for setup instructions | A small `run_request` describes the assignment; the server derives protected identities and prepares workspaces | +| Each new run can interrupt the user with another sharing decision | Native consent can remember access for the same repository and selected providers | +| A run appears finished while its underlying provider task is still active | Run inspection, waiting, cancellation, and result retrieval follow the same durable task identity | +| Grok progress text and report labels clutter a short answer | Reliable completed tool rounds return the final response; progress remains in task events | +| SDK or transport errors obscure useful provider output | Provider-specific adapters preserve supported result shapes and bounded diagnostic causes | + +## Simpler launches + +Use ordinary language: + +> Use Grok Co-Engineer to review the latest change. Focus on correctness. + +Codex uses `delegate.run_request` with one to eight independent assignments. +The server derives Git identity, workspace bindings, dispatch identity, and +provenance. A saved profile and a separate coordinator are optional. + +- Readiness, repository identity, consent, managed workspaces, and disjoint + writer scopes are checked before dispatch. +- Omitted assignment access follows the role; explicit conflicts fail early. +- Duration estimates are optional. The default is ten minutes with the existing + 20% deadline margin. +- Provider choices remain explicit. Grok and Cursor use their configured provider + defaults; unsupported model overrides fail before launch. +- Missing worker entrypoints are reported before workspace preparation, with + actionable reinstall/restart guidance. +- Worker instructions identify the assigned working directory and clarify that + the controller owns machine receipts. Routine work no longer asks the provider + to reconstruct Co-Engineer's setup or lifecycle machinery. + +The five-tool catalog remains `status`, `delegate`, `task`, `tasks`, and `cancel`. +Existing single-task calls and full `run` envelopes remain supported. + +## Consent that can be remembered + +The native repository-sharing form offers two scopes: + +- **This run only:** authorize the current request. +- **Repository and selected providers:** remember explicit access for subsequent + runs, including linked worktrees of the same repository. + +Accepting the form is sufficient; there is no redundant approval checkbox. +The form allows more time for a human response. Interrupted decisions can be +requested again on the existing run without inventing an approval. + +Remembered grants are owner-only local state. They do not automatically cover +new providers, changed origins, or replacement repositories. Earlier one-run +approvals are not promoted. Revocation applies to future admissions; cancel +already-running work separately when needed. + +From a repository clone: + +```bash +node plugins/codex-co-engineer/bin/consent-grants.mjs list +node plugins/codex-co-engineer/bin/consent-grants.mjs revoke --repo /absolute/path/to/repository +``` + +A host without native MCP form elicitation returns a capability blocker before +repository export. Consent authorizes repository sharing, not an automatic +push, merge, deployment, or production change. + +## Reliable continuation and completion + +The task supervisor owns provider execution. Runs coordinate those tasks and +project their state rather than creating a competing lifecycle. + +- Inspection, reconnect, reply, cancellation, and result retrieval use the same + bound task identity. +- Temporary observation failure stays uncertainty and can reconcile later. + It is not treated as proof that a worker stopped. +- Slow acknowledgement keeps independent work active. Later authoritative + dispatch evidence updates the existing run. +- Accepted or uncertain prompts are never blindly replayed through another + transport. +- Required cancellation or failure cannot become successful completion. Optional + active tasks remain owned and cancellable. +- Event-driven waits avoid routine hot polling and repeated state writes. + Unchanged observations keep a stable cursor. +- The normal run receipt includes provider output and available branch/handoff + information; users do not need to discover a hidden child task ID. + +## Compact coordination and Grok results + +Shorter skills keep installation details outside routine delegation. Semantic +runs return compact coordination results by default; diagnostics remain +available through `task` with `run_id` and `view: "diagnostics"`. + +Structured-capable clients receive structured results with bounded fallback +text. Text-only clients retain compatible full-text receipt encoding. Output +limits remain explicit, including truncation metadata where applicable. + +Grok's ACP stream carries progress and answer text without a dedicated final +message marker. On a successful `end_turn`, Co-Engineer selects text after a +known, fully settled tool round, following the response-framing approach in +Grok's own structured-output client. The returned result and stored provider +report agree; progress stays in the task event log. + +Selection is deliberately conservative. Missing or duplicate tool identities, +unsettled tools, interleaved text, web search, missing final text, output overflow, +failed turns, and other stop reasons retain the existing aggregate output. +No sentence matching, regular-expression trimming, or extra model call is used +to guess the answer. Other providers retain their existing output behavior. + +This improves the amount of result text an orchestrator must read. It does not +establish token parity with native subagents; that requires a controlled host +comparison including tool/schema loading, waits, and recovery. + +## Provider fixes + +### Grok + +- Readiness distinguishes explicit login status from ancillary command failures. +- Working-directory guidance and controller-owned evidence labels reduce + unnecessary navigation and unrequested response sections. +- Final-response framing produces exact short text and JSON-only output in the + tested normal tool-completion path, with conservative fallback elsewhere. + +### Cursor Local and Cursor Cloud + +- SDK discovery uses a stable working directory and reuses successful discovery + within the server process. Deleted inherited working directories produce typed + discovery behavior instead of an opaque failure. +- Cloud completion accepts the documented SDK result shape, including duration, + model, and usage metadata, while retaining strict internal identity checks. +- Cloud sanitization preserves legitimate answer text that overlaps instructions + while still redacting full prompt echoes, credentials, and diagnostic fragments. +- Structured Cursor permission options survive grouped questions and same-session + replies. Ordinary shell punctuation is not treated as a user question. +- Cloud-only runs do not require the local Linux process boundary; mixed and + local runs still do. + +### Muse through DeepSeek Harness + +- The default is **Muse Spark 1.3 Contributor with XHigh reasoning via OpenRouter**. +- The route uses `OPENROUTER_API_KEY` or its owner-only key file. It does not + receive a Meta credential. +- The supported ACPX one-shot path preserves useful bounded model/API errors. +- The optional Ox Alpha route remains a separate configured Muse/DSH model choice. +- ACPX dispatch uncertainty remains explicit and is never treated as permission + to resend an accepted prompt. + +## Cleanup and packaging + +- Worktree Bootstrap 1.1.0 is bundled under the MIT license with a pinned source + digest and provenance record. Local execution uses the package copy, eliminating + a separately installed private dependency. It requires Python 3.11+ and the + Python standard library only. +- Setup checks runtime prerequisites early and preserves provider credentials. + +- Linux ACP cleanup discovers detached descendants from process records. + Failed process-list commands cannot silently hide active children. +- Verification cleanup tolerates processes disappearing during inspection and + parses parenthesized process names correctly. +- Repeated reconciliation avoids repeatedly paying cleanup grace periods while + preserving the initial grace period and ownership checks. +- Setup, configuration, troubleshooting, API, and release guides are included + in the package. Their relative links are validated against the packed inventory. +- Existing incompatible operator configuration is reported instead of overwritten. +- Development candidates can use a distinct local marketplace identity to avoid + a stale project marketplace replacing their shared installed cache. + +## Upgrading + +### From the public 3.4.0 release or an earlier version + +Finish or cancel active runs. In a clean source clone registered as your local +marketplace: + +```bash +git fetch origin tag v3.4.2 +git switch --detach v3.4.2 +npm --prefix plugins/codex-co-engineer run setup +codex plugin remove codex-co-engineer@codex-co-engineer +codex plugin add codex-co-engineer@codex-co-engineer +npm --prefix plugins/codex-co-engineer run setup:check +``` + +Use your actual installed marketplace identity if it differs. Start a new Codex +session and ask for Co-Engineer status. A dirty development clone should be +preserved; install from a separate clean clone rather than resetting it. + +### From a local 3.4.1 or 3.4.2 candidate + +The public version may match a local candidate's label. Reinstall the plugin to +refresh its bytes, then restart the Codex session. Do not assume the version +string proves that the currently running MCP process contains the new build. +Existing durable state retains its 3.4.1 directory identity for compatibility. +Do not delete task state or provider login files as an upgrade step. + +### Migrating a direct Meta Muse profile + +Back up your DSH configuration, then update the Muse route to: + +| Setting | Value | +| --- | --- | +| Provider | `openrouter` | +| Endpoint | `https://openrouter.ai/api/v1` | +| Model | `meta/muse-spark-1.3-contributor` | +| Reasoning | `xhigh` | +| Credential variable | `OPENROUTER_API_KEY` | + +Save your OpenRouter key through `plugins/codex-co-engineer/bin/set-model-api-key` +or the documented owner-only key-file route. Setup reports incompatible existing +profiles rather than silently replacing them. Review custom YAML against +[configuration](../configuration.md) before restarting. + +## Published 3.4.0 compatibility + +This release retains the published usage ledger, provider-event rules, lane-health +and decision reducers, PR-ready decision card, proof-bound cleanup planner, and +optional Luna/Sol host relay. The relay is opt-in; ordinary delegation does not +require a separate manager or change your Codex model. + +## Requirements and known limits + +- Node.js 24+, Git, Python 3.11+ for bundled setup, and Codex plugin support are + required. The release gate runs + on Node.js 24 for reproducibility. +- Local providers require Linux, a working `systemd --user` manager, + `systemd-run` 244+, unified cgroup v2, and Python 3.11+ for the bundled + worktree tool. Lifecycle + control is not a sandbox. +- Native repository consent requires MCP form elicitation. Extra panels are + optional and host-dependent; the conversation works without them. +- Cursor Cloud needs a provider-accessible origin and a pushed immutable SHA. + A feature branch may need an open PR before the provider can see its commit. +- Grok and Cursor authentication remain provider-managed. Co-Engineer does not + promise that a provider session can never expire. +- Managed worktrees and owner-only task state are retained for review. Cleanup + is explicit; no background garbage collector deletes them. +- The advertised four-hour pending-call budget is not a measured universal + Codex Desktop limit. Windows/macOS native local execution is not qualified + by the Linux acceptance evidence. + +## Validation + +The release process binds checks to an exact candidate rather than only a +version label. The provider-free gate covers unit and compatibility tests, +MCP Inspector, process/environment boundaries, reproducibility, provenance, +and package inventories. GitHub CI provides separate portable evidence. + +Live acceptance during this release cycle covered provider completion, +remembered consent, continuation, cancellation, and cleanup. The final Grok +framing change passed real exact-text and JSON-only assignments without a new +consent prompt; both workers stopped normally. Earlier provider and host checks +remain scoped to their tested candidates, not universal platform guarantees. +Private prompts, credentials, local paths, and machine receipts are not release +assets. See [the release process](https://github.com/ajhcs/Codex-Co-Engineer/blob/main/docs/release.md) +for the qualification procedure. diff --git a/plugins/codex-co-engineer/docs/run-tool-api.md b/plugins/codex-co-engineer/docs/run-tool-api.md new file mode 100644 index 0000000..abac0a3 --- /dev/null +++ b/plugins/codex-co-engineer/docs/run-tool-api.md @@ -0,0 +1,197 @@ +# Run tool API (3.4.1) + +3.4.1 keeps the five-tool MCP catalog. Submit, status, wait, attention, +reply, cancel, and cleanup are parameters and modes on `status`, `delegate`, +`task`, `tasks`, and `cancel`. The preferred bounded-run ingress is the small +server-compiled `run_request`; the 3.4.0 full `run` envelope remains accepted +for compatibility and is not constructed by the skills. + +Owned files: + +- `plugins/codex-co-engineer/mcp/v3/run-tool-adapter.mjs` +- `plugins/codex-co-engineer/mcp/v3/server.mjs` +- `plugins/codex-co-engineer/mcp/v3/supervisor.mjs` +- `plugins/codex-co-engineer/test/r1-run-tool-adapter.test.mjs` +- `plugins/codex-co-engineer/test/r1-run-tool-adapter-adversarial.test.mjs` +- `plugins/codex-co-engineer/test/fixtures/r1-run-tool-adapter-fixtures.mjs` +- this document + +## Catalog + +The public catalog remains exactly: + +`status`, `delegate`, `task`, `tasks`, `cancel` + +Omitted additive fields preserve exact 3.2.1 direct/single-task/full-text/ +structured/wait/diagnostic/reply/cancel behavior and response shapes. + +## Frozen mapping + +| Operation | Tool | Additive parameter or mode | +| --- | --- | --- | +| submit | `delegate` | `run_request` (preferred), `run` (legacy compatibility) | +| status | `status` or `task` | `run_id` | +| wait | `task` or `tasks` | `run_id` plus `wait_until: "decision_or_attention"` | +| attention | `task` | `run_id` plus `attention` | +| reply | `task` | `run_id` plus `run_reply` | +| cancel | `cancel` | `run_id` plus optional `assignment_ids` | +| cleanup | `cancel` | `run_id` plus `cleanup: true` | + +`wait_until` remains `progress` and `terminal` for 3.2.1. The additive +run mode is `decision_or_attention`. Routine progress never wakes. + +Mixing a run body with 3.2.1 `task_id` / `workspace_mode` / `create_pr` / +`reply` fields fails closed. + +## Simple run request + +Call `delegate` with only semantic intent: + +```json +{ + "run_request": { + "run_id": "vale-hardening", + "repo": "/absolute/repository/path", + "objective": "Implement and review the hardening plan.", + "assignments": [ + { + "assignment_id": "social-implementation", + "provider": "grok", + "role": "implement", + "prompt": "Implement the social ingestion slice.", + "expected_duration_ms": 900000 + } + ] + } +} +``` + +Assignment `access` is optional: `implement` derives `writer`, while `review` +and `verify` derive `read_only`. An explicit value must agree with the role. +Omitting access and supplying its equivalent explicit value produce the same +normalized request. Multiple writer lanes need explicit disjoint write scopes. + +The server observes the clean exact Git identity, resolves the provider model, +and derives the request idempotency key, manifest/prompt-envelope/lane +digests, child and task identities, and managed-workspace policy. Callers +cannot provide those derived fields. A changed objective, assignment, +provider, SHA, or scope produces a different identity. + +The stdio server requests repository exposure through the host's native MCP +form. The form names the repository/base, run, and selected providers. Native +**Accept** is the approval; there is no second approval checkbox. The required +selector clearly defaults to remembering approval for the canonical Git +repository and exactly those providers, with **This run only** as the one-time +choice. A remembered provider subset can be reused across worktrees that share +the same Git common directory. A new provider, changed origin, unrelated or +recreated repository, rejected form, or malformed owner state cannot reuse it. +No earlier run-only approval is migrated. A model-authored boolean or prose +reply does not approve access. Hosts without form elicitation return an +explicit capability blocker before any workspace or prompt dispatch. + +Remembered grants live in the owner-only Co-Engineer state directory and never +contain credentials. Inspect or revoke them with the installed package command: + +```sh +node /absolute/path/to/plugin/bin/consent-grants.mjs list +node /absolute/path/to/plugin/bin/consent-grants.mjs revoke --repo /absolute/path/to/repository +node /absolute/path/to/plugin/bin/consent-grants.mjs revoke --grant-id GRANT_ID_FROM_LIST +``` + +An npm package installation also provides the shorter +`codex-co-engineer-consent` command. If a process stops during a grant update +and later commands report `consent_grant_store_busy`, first verify that no MCP +server or consent command is running, then remove only +`.consent-grants.lock` from the owner-only Co-Engineer state directory. + +A dismissed or interrupted approval remains inspectable. To request the +native form again for a pending run, call `task` with the same `run_id` and +`run_reply: { "request_consent": true }`. This requests a decision; it is not +approval. Ordinary status and wait calls never reopen the form. The existing +opaque `approval_ref` continuation remains available to embedding hosts with +a trusted verifier; ordinary stdio clients do not construct these references. + +Admission has two barriers. The server validates consent, provider and local +boundary readiness, repository identity, every workspace, and disjoint writer +scope before sending any prompt. Pending consent is shown as attention; workspace admission is +`preparing`. The run can say `running` only when every required lane has +authoritative `prompt_dispatched` evidence. A mid-dispatch failure is +`degraded` with exact dispatched, undispatched, and uncertain lane lists. + +Run and lane phases are explicit and receipts are restartable. Prompt-dispatch +uncertainty is never replayed. Post-prompt unrecoverable work produces a +bounded partial handoff with the retained worktree, starting/current SHA, +clean state, changed files, commits, last provider event, recovery class, and +safe next actions. + +## Bounds and selection + +One run submission carries 1–8 lanes. Provider/model is explicit on each +assignment or filled from one named profile. Explicit fields must agree +with that profile when both are present; omitted fields may be filled +from the profile. A missing, invalid, or conflicting profile fails +closed before dispatch. The adapter never learns, ranks, or globally +routes. The accepted four-slot registry is `grok`, `cursor-local`, +`cursor-cloud`, and `dsh`. P22 future-harness conformance remains +evidence, never a provider slot. + +Direct mode, replay, fallback, merge, push, create-PR, GitHub, and +remote keys fail before any provider dispatch, ref creation, or cleanup. + +## Authority preserved + +| Surface | Owner | Use here | +| --- | --- | --- | +| One-submission runtime, proof-bound cleanup, lifecycle finality | P33 | injected `submitRun` / `inspectRun` / `resumeRun` / `cancelRun` | +| Attention latch and exactly-once reply | P34 | `resumeRun` attention items and injected `attention.reply` | +| Run-owned candidate ref | P35 | projected `refs/codex-co-engineer/runs//candidate`; never composed here | +| Four-slot provider composition | P23 | exact `{provider, model}` lookup; no fallback | +| Terminal false-success projection | R-TRUTH | model-facing lane receipts | +| Remote mutation denial | P29 / P28 | `denyRunToolRemoteMutationV1` | + +Unaffected lanes continue. A required unresolved lane blocks a complete +candidate. Decision results are the verified P33/P34 receipts, not caller +prose. MCP output is model-facing: owner-only raw artifacts are stripped. + +Production default seams are durable P24/P25/P34/P32 authorities under +the supervisor state root (`runs/store`, `runs/journal`, +`runs/attention`, `runs/scheduler`, `runs/artifacts`). Tests may inject +`createInProcessRunSeams` or an explicit `seams` object. Restart +recovery, journal cursors, attention CAS, scheduler-plan identity, and +proof-bound cleanup are not process-local Maps. + +Production `attention.reply` binds P34 to the supervisor proof-bound +same-session mailbox (`submitReply`): the one reply round must match the +latched task/session/question identity and is delivered exactly once. +Failed or unconfirmed cancellation stays unresolved/unsafe, including +after durable restart, and never projects cancelled/safe. Authoritative +artifact bytes and scheduler-plan identity persist before or atomically +with provider dispatch; stale, partial, or mismatched state fails closed +and never duplicates dispatch. One immutable run-level profile/catalog +snapshot is loaded, bound, and persisted at submit; later catalog +mutation cannot change assignment resolution. + +`wait_until: "decision_or_attention"` performs a bounded wait +(`wait_ms` 0 is a snapshot; omit follows the MCP pending-call budget). +It wakes only on attention or terminal lane decisions, keeps the last +lane cursor, and never replays. Model-facing receipts recursively strip +owner-only `raw` / `bytes` / `secret` evidence. Attention items are +validated before any scheduler resume. + +## API + +- `classifyRunToolCall(tool, args)` — pure; `legacy` or `run` +- `createRunToolAdapter({ runtime, attention?, projectLaneTask?, classifyLaneTask?, rememberSubmitContext? })` +- `createDurableRunSeams({ root, delegateTask, inspectTask, cancelTask, settleLocalTaskLifecycle, cleanupLocalTaskLifecycle, clock? })` +- `createInProcessRunSeams({ delegateTask, inspectTask, cancelTask, settleLocalTaskLifecycle, cleanupLocalTaskLifecycle, clock? })` +- `describeRunToolAdapterV1()` +- `denyRunToolRemoteMutationV1(operation)` + +## Testing + +``` +node --no-warnings --test test/r1-run-tool-adapter.test.mjs \ + test/r1-run-tool-adapter-adversarial.test.mjs \ + test/v3-server.test.mjs \ + test/v3-supervisor.test.mjs +``` diff --git a/plugins/codex-co-engineer/mcp/v3/acp-worker.mjs b/plugins/codex-co-engineer/mcp/v3/acp-worker.mjs index 40a4fa7..fbdef63 100644 --- a/plugins/codex-co-engineer/mcp/v3/acp-worker.mjs +++ b/plugins/codex-co-engineer/mcp/v3/acp-worker.mjs @@ -1,9 +1,10 @@ -import { randomUUID } from 'node:crypto'; +import { createHash, randomUUID } from 'node:crypto'; import { execFile, spawn } from 'node:child_process'; import { readFileSync, watch as watchDirectory } from 'node:fs'; import { chmod, mkdir, readFile, readdir, rm, writeFile } from 'node:fs/promises'; import { homedir } from 'node:os'; import path from 'node:path'; +import { StringDecoder } from 'node:string_decoder'; import { fileURLToPath } from 'node:url'; import { promisify } from 'node:util'; @@ -34,12 +35,12 @@ import { import { recordNeedsAttention, replyDecision, waitForReply } from './mailbox.mjs'; import { boundedProviderResult, boundedProviderValue, createProviderResultAccumulator, providerCharCount } from './provider-result.mjs'; import { RunContractV1Error } from './run-manifest.mjs'; +import { BUNDLED_WORKTREE_BOOTSTRAP } from './worktree-bootstrap-runtime.mjs'; import { appendTaskEvent, readPrompt, readRuntimeRecord, readTask, taskPaths, updateTask } from './task-store.mjs'; process.umask(0o077); const RUNTIME_URL = new URL('../../assets/acpx-runtime.mjs', import.meta.url); -const SINGLE_TURN_FLOW = fileURLToPath(new URL('./single-turn.flow.mjs', import.meta.url)); const runFile = promisify(execFile); const PROVIDERS = Object.freeze({ grok: { agent: 'grok-build' }, @@ -54,7 +55,11 @@ const MAX_EVENT_DEPTH = 6; const MAX_EVENT_KEYS = 64; const MAX_EVENT_ITEMS = 64; const MAX_CLI_OUTPUT = 1024 * 1024; -const DEFAULT_DSH_MODEL = 'muse-spark-1.2-contributor'; +const MAX_ACPX_EXEC_FRAME = 256 * 1024; +const MAX_ACPX_EXEC_STDERR = 64 * 1024; +const MAX_ACPX_EXEC_EVENTS = 256; +const MAX_GROK_FINAL_TOOL_IDS = 512; +const DEFAULT_DSH_MODEL = 'meta/muse-spark-1.3-contributor'; const PROCESS_LIST_MAX_BUFFER = 4 * 1024 * 1024; const ACPX_TERMINATION_GRACE_MS = 1_000; @@ -295,7 +300,7 @@ export async function boundedWtbHandoff({ }; try { const { stdout } = await withBound( - () => runFileImpl('worktree-bootstrap', [ + () => runFileImpl(BUNDLED_WORKTREE_BOOTSTRAP, [ 'handoff', taskName, '--repo', @@ -389,28 +394,41 @@ function taskTimeoutMs(task, now = Date.now()) { fail('invalid_timeout', 'Task is missing a recorded deadline.'); } -function isUserFacingPermission(params) { - const title = String(params?.raw?.toolCall?.title ?? params?.raw?.question ?? ''); - const toolCallId = String(params?.raw?.toolCall?.toolCallId ?? params?.raw?.toolCall?.id ?? ''); - if (isAskUserQuestionName(title) || isAskUserQuestionName(toolCallId)) return true; - return /\?|user input|needs? attention|confirm|approval required|fake permission/iu.test(title); +export function isUserFacingPermission(params) { + const explicitQuestion = params?.raw?.question; + if (typeof explicitQuestion === 'string' && explicitQuestion.trim().length > 0) return true; + const questionTool = params?.raw?.toolCall; + if (isAskUserQuestionName(questionTool?.title) || isAskUserQuestionName(questionTool?.toolCallId ?? questionTool?.id)) return true; + const kind = params?.raw?.toolCall?.kind ?? params?.inferredKind; + if (typeof kind === 'string' && kind !== 'other') return false; + const title = String(params?.raw?.toolCall?.title ?? '').trim(); + return /^(?:user input|needs? attention|approval required|fake permission|confirm(?:ation)?(?: required)?)(?:\b|:|\?)/iu.test(title) + || (title.endsWith('?') && !title.includes('$?')); } -function safeQuestionId(value) { - const normalized = String(value ?? 'permission').replace(/[^A-Za-z0-9._-]/gu, '-').replace(/^[^A-Za-z0-9]+/u, 'q'); - return (normalized || 'permission').slice(0, 80); +export function safeQuestionId(value) { + const source = String(value ?? 'permission'); + const normalized = source.replace(/[^A-Za-z0-9._-]/gu, '-').replace(/^[^A-Za-z0-9]+/u, 'q') || 'permission'; + if (source === normalized && normalized.length <= 80) return normalized; + const suffix = createHash('sha256').update(source, 'utf8').digest('hex').slice(0, 16); + return `${normalized.slice(0, 63)}-${suffix}`; } -async function handlePermissionRequest(root, taskId, params, signal) { +export async function handlePermissionRequest(root, taskId, params, signal) { if (!isUserFacingPermission(params)) return undefined; const { task } = await readTask(root, taskId); const sessionId = params.sessionId ?? task.acp_session_id; if (typeof sessionId !== 'string' || sessionId.length === 0) return undefined; const questionId = safeQuestionId(params.raw?.toolCall?.toolCallId ?? randomUUID()); + const explicitQuestion = params.raw?.question; await recordNeedsAttention(root, taskId, { session_id: sessionId, question_id: questionId, - prompt: typeof params.raw?.toolCall?.title === 'string' ? params.raw.toolCall.title : 'Provider requested approval.', + prompt: typeof explicitQuestion === 'string' && explicitQuestion.trim().length > 0 + ? explicitQuestion + : typeof params.raw?.toolCall?.title === 'string' + ? params.raw.toolCall.title + : 'Provider requested approval.', options: Array.isArray(params.raw?.options) ? params.raw.options : null, stage: 'provider_feedback', }); @@ -569,6 +587,82 @@ export function boundedEvent(event, prompt = '') { return plainObject(safe) ? safe : { type: 'status', text: boundedText(safe, prompt, budget) }; } +/** + * Follow Grok's native Messages reducer boundary: a completed client-tool + * round closes the assistant frame and the final frame becomes `result`. + * Source: https://github.com/xai-org/grok-build + * Grok pager: src/headless/reducer/messages/mod.rs + * + * This projection is deliberately more conservative than xAI's formatter: + * ambiguous/interleaved tool streams and every WebSearch fall back to the + * complete provider stream because ACPX does not forward Grok's `_meta.backend`. + */ +export function createGrokFinalResponseReducerV1({ sanitize = (text) => text } = {}) { + let currentOutput = createProviderResultAccumulator({ sanitize }); + let currentComplete = createLocalProviderResultCollectorV1(); + let currentHasNonWhitespace = false; + const pendingToolIds = new Set(); + let sawSettledRound = false; + let reliable = true; + + const resetCurrent = () => { + currentOutput = createProviderResultAccumulator({ sanitize }); + currentComplete = createLocalProviderResultCollectorV1(); + currentHasNonWhitespace = false; + }; + const toolId = (event) => ( + typeof event?.toolCallId === 'string' && event.toolCallId.length > 0 && event.toolCallId.length <= 512 + ? event.toolCallId + : null + ); + + return { + append(event) { + if (event?.type === 'text_delta' && event.stream !== 'thought' && typeof event.text === 'string') { + if (event.text.length > 0 && pendingToolIds.size > 0) reliable = false; + currentOutput.append(event.text); + currentComplete.append(event.text); + currentHasNonWhitespace ||= /\S/u.test(event.text); + return; + } + if (event?.type !== 'tool_call') return; + if (event?.rawInput?.variant === 'WebSearch') reliable = false; + const id = toolId(event); + if (event.tag === 'tool_call') { + if (id === null || pendingToolIds.has(id) || pendingToolIds.size >= MAX_GROK_FINAL_TOOL_IDS) { + reliable = false; + return; + } + pendingToolIds.add(id); + return; + } + if (event.tag !== 'tool_call_update' || !['completed', 'failed'].includes(event.status)) return; + if (id === null || !pendingToolIds.delete(id)) { + reliable = false; + return; + } + if (pendingToolIds.size > 0) { + return; + } + sawSettledRound = true; + resetCurrent(); + }, + finish({ turnResult, fullSnapshot }) { + const eligible = reliable + && sawSettledRound + && pendingToolIds.size === 0 + && currentHasNonWhitespace + && turnResult?.status === 'completed' + && turnResult?.stopReason === 'end_turn' + && fullSnapshot?.overflow !== true; + if (!eligible) return null; + const snapshot = currentComplete.snapshot(); + if (snapshot.overflow === true) return null; + return Object.freeze({ bounded: currentOutput.finish(), snapshot }); + }, + }; +} + export function publicError(error, prompt = '') { const rawCode = typeof error?.code === 'string' ? error.code : 'acp_worker_failed'; const code = /^[A-Za-z0-9._-]{1,128}$/u.test(rawCode) ? rawCode : 'acp_worker_failed'; @@ -984,6 +1078,7 @@ export async function runCliFallback({ root, task, prompt, signal } = {}) { started_at: new Date().toISOString(), }); await appendTaskEvent(root, task.id, { type: 'transport', state: 'prompt_dispatched', transport: 'cli', fallback_from: 'acp' }); + await updateTask(root, task.id, { dispatch_evidence: 'authoritative' }); stopDeadline = startDeadlineWatch(root, task.id, () => { timedOut = true; cancel(); @@ -1041,23 +1136,247 @@ function commandString(argv) { return argv.map((entry) => `'${entry.replaceAll("'", "'\\''")}'`).join(' '); } -function parseFlowResult(stdout) { - const lines = stdout.trim().split(/\r?\n/u).filter(Boolean); - for (let index = lines.length - 1; index >= 0; index -= 1) { - try { - const value = JSON.parse(lines[index]); - if (value?.action === 'flow_run_result') return value; - } catch { - // Ignore non-JSON progress; --json-strict should normally prevent it. - } +function acpRpcIdKey(value) { + if (typeof value === 'string' && value.length > 0 && value.length <= 128) return `string:${value}`; + if (typeof value === 'number' && Number.isSafeInteger(value)) return `number:${value}`; + return null; +} + +const ACP_PROVIDER_ERROR_MESSAGES = Object.freeze({ + provider_billing_required: 'The provider billing configuration is unavailable.', + authentication_required: 'The provider authentication failed.', + provider_rate_limited: 'The provider rate limit was reached.', + provider_failed: 'The provider request failed.', +}); + +function classifyAcpProviderError(error) { + const code = typeof error?.code === 'string' ? error.code : ''; + const message = typeof error?.message === 'string' ? error.message : ''; + const detail = `${code} ${message}`.trim(); + const normalized = detail.toLowerCase(); + let stableCode = 'provider_failed'; + if (/provider_billing_required|provider_billing|billing|payment|credit|quota|insufficient\s+funds|\b402\b/iu.test(normalized)) { + stableCode = 'provider_billing_required'; + } else if (/authentication_required|provider_auth|auth(?:entication|orization)?|credential|api[_ -]?key|not\s+signed\s+in|needs?[_ -]?login|\b40[13]\b|forbidden|unauthori[sz]ed/iu.test(normalized)) { + stableCode = 'authentication_required'; + } else if (/provider_rate_limited|rate[_ -]?limit|too\s+many\s+requests|throttl|\b429\b/iu.test(normalized)) { + stableCode = 'provider_rate_limited'; } - fail('acpx_invalid_result', 'ACPX did not return a flow result.'); + // Provider text is useful for local diagnostics but is not safe to expose + // through a native run receipt. Keep this error's public message fixed; + // callers can retain only the bounded, separately sanitized transport + // detail when they explicitly need it. + return new AcpWorkerError(stableCode, ACP_PROVIDER_ERROR_MESSAGES[stableCode]); +} + +function acpTextValue(value) { + if (typeof value === 'string') return value; + if (!plainObject(value)) return null; + for (const key of ['text', 'delta', 'output', 'answer']) { + if (typeof value[key] === 'string') return value[key]; + } + if (plainObject(value.content)) return acpTextValue(value.content); + if (Array.isArray(value.content)) { + const text = value.content + .filter((entry) => plainObject(entry) && (entry.type === 'text' || entry.type === 'text_delta')) + .map((entry) => acpTextValue(entry)) + .filter((entry) => typeof entry === 'string') + .join(''); + if (text.length > 0) return text; + } + if (plainObject(value.message)) return acpTextValue(value.message); + return null; +} + +function acpStopReason(value) { + if (typeof value !== 'string' || value.length === 0) return null; + if (/^[A-Za-z0-9._-]{1,64}$/u.test(value)) return value; + return null; } -async function runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, signal }) { +/** + * Collect the JSON-RPC transcript produced by `acpx exec --file -`. + * + * ACPX's JSON formatter prints both directions of the protocol. Outbound + * frames are inspected only long enough to correlate their ids; their params + * are deliberately never retained, because the prompt is inside one of + * those params. Only bounded provider text and correlated response metadata + * leave this collector. + */ +function createAcpExecCollector({ prompt, onSession, onAuthoritative, onText } = {}) { + const pending = new Map(); + const output = createProviderResultAccumulator({ sanitize: (text) => sanitizeText(text, prompt) }); + const decoder = new StringDecoder('utf8'); + let lineBuffer = ''; + let sessionId = null; + let promptRequestId = null; + let promptAttempted = false; + let promptResponded = false; + let authoritative = false; + let stopReason = 'end_turn'; + let promptError = null; + let resultError = null; + let transportError = null; + let structuredResult; + let structuredResultSet = false; + let protocolError = null; + let textEventCount = 0; + + const protocolFailure = (message) => { + const error = new AcpWorkerError('acpx_protocol_invalid', message); + protocolError ??= error; + throw error; + }; + const markAuthoritative = () => { + if (authoritative) return; + authoritative = true; + try { onAuthoritative?.(); } catch { /* evidence is persisted by the caller */ } + }; + const appendText = (text) => { + if (typeof text !== 'string' || text.length === 0) return; + output.append(text); + if (textEventCount >= MAX_ACPX_EXEC_EVENTS) return; + textEventCount += 1; + try { onText?.(text); } catch { /* output remains available in the bounded accumulator */ } + }; + const processFrame = (frame) => { + if (!plainObject(frame) || frame.jsonrpc !== '2.0') protocolFailure('ACPX returned an invalid JSON-RPC frame.'); + const idKey = acpRpcIdKey(frame.id); + if (typeof frame.method === 'string') { + if (frame.method === 'initialize' || frame.method === 'session/new' || frame.method === 'session/prompt') { + if (idKey === null) protocolFailure('ACPX returned an uncorrelatable outbound request.'); + if (pending.has(idKey)) protocolFailure('ACPX reused an outstanding JSON-RPC request id.'); + if (!pending.has(idKey) && pending.size >= MAX_ACPX_EXEC_EVENTS) { + protocolFailure('ACPX returned too many pending JSON-RPC requests.'); + } + pending.set(idKey, frame.method); + if (frame.method === 'session/prompt') { + const outboundSessionId = frame.params?.sessionId; + if (typeof outboundSessionId !== 'string' || outboundSessionId.length === 0 || outboundSessionId !== sessionId) { + protocolFailure('ACPX prompt request did not match the established session.'); + } + promptRequestId = idKey; + promptAttempted = true; + } + return; + } + if (frame.method === 'session/update') { + const incomingSessionId = frame.params?.sessionId; + if (typeof incomingSessionId !== 'string' || incomingSessionId.length === 0 || incomingSessionId !== sessionId) return; + if (!promptAttempted) return; + markAuthoritative(); + if (frame.params?.update?.sessionUpdate !== 'agent_message_chunk') return; + const text = acpTextValue(frame.params?.update?.content); + if (text !== null) appendText(text); + return; + } + // Permission requests and formatter metadata are not provider output. + return; + } + + const method = idKey === null ? null : pending.get(idKey); + if (method) pending.delete(idKey); + if (plainObject(frame.error)) { + if (method === 'session/prompt') { + promptResponded = true; + markAuthoritative(); + promptError ??= classifyAcpProviderError(frame.error); + } else if (idKey === null || method) { + transportError ??= classifyAcpProviderError(frame.error); + } + return; + } + if (!method) return; + if (method === 'session/new') { + const candidate = frame.result?.sessionId; + if (typeof candidate === 'string' && candidate.length > 0 && candidate.length <= 256) { + sessionId = candidate; + try { onSession?.(candidate); } catch { /* terminal receipt carries the session id */ } + } + return; + } + if (method === 'session/prompt') { + if (!plainObject(frame.result)) { + resultError ??= new AcpWorkerError('acpx_invalid_result', 'ACPX prompt response was missing a result.'); + return; + } + const reason = acpStopReason(frame.result.stopReason); + if (reason === null) { + resultError ??= new AcpWorkerError('acpx_invalid_result', 'ACPX prompt response had no valid stop reason.'); + return; + } + if (frame.result.sessionId !== undefined && frame.result.sessionId !== sessionId) { + resultError ??= new AcpWorkerError('acpx_invalid_result', 'ACPX prompt response did not match the established session.'); + return; + } + promptResponded = true; + markAuthoritative(); + stopReason = reason; + if (Object.hasOwn(frame.result, 'output') && frame.result.output !== null + && typeof frame.result.output === 'object') { + structuredResult = frame.result.output; + structuredResultSet = true; + } + const text = acpTextValue(frame.result); + if (text !== null) appendText(text); + } + }; + const feed = (chunk) => { + if (protocolError) return; + const buffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(String(chunk ?? '')); + lineBuffer += decoder.write(buffer); + let newline; + while ((newline = lineBuffer.indexOf('\n')) >= 0) { + const line = lineBuffer.slice(0, newline).replace(/\r$/u, ''); + lineBuffer = lineBuffer.slice(newline + 1); + if (line.length === 0) continue; + if (Buffer.byteLength(line, 'utf8') > MAX_ACPX_EXEC_FRAME) { + protocolFailure('ACPX returned an overlarge JSON-RPC frame.'); + } + let frame; + try { frame = JSON.parse(line); } catch { protocolFailure('ACPX returned malformed JSON-RPC output.'); } + processFrame(frame); + if (protocolError) return; + } + if (Buffer.byteLength(lineBuffer, 'utf8') > MAX_ACPX_EXEC_FRAME) { + protocolFailure('ACPX returned an overlarge JSON-RPC frame.'); + } + }; + const finish = () => { + if (protocolError) throw protocolError; + lineBuffer += decoder.end(); + if (lineBuffer.trim().length > 0) { + if (Buffer.byteLength(lineBuffer, 'utf8') > MAX_ACPX_EXEC_FRAME) { + protocolFailure('ACPX returned an overlarge JSON-RPC frame.'); + } + let frame; + try { frame = JSON.parse(lineBuffer); } catch { protocolFailure('ACPX returned malformed JSON-RPC output.'); } + lineBuffer = ''; + processFrame(frame); + } + if (protocolError) throw protocolError; + const streamed = output.finish(); + return { + output: structuredResultSet + ? boundedProviderValue(structuredResult, { sanitize: (text) => sanitizeText(text, prompt) }) + : streamed, + sessionId, + promptRequestId, + promptAttempted, + promptResponded, + authoritative, + stopReason, + promptError, + resultError, + transportError, + }; + }; + return { feed, finish }; +} + +async function runDshExec({ root, task, prompt, cwd, configuration, timeoutMs, signal }) { const taskDirectory = taskPaths(root, task.id).directory; const acpxHome = path.join(taskDirectory, 'acpx-home'); - const inputFile = path.join(taskDirectory, `flow-input-${randomUUID()}.json`); const timeoutSeconds = Math.max(1, Math.ceil(timeoutMs / 1000)); const argv = [ '--agent', commandString(configuration.override), @@ -1066,11 +1385,9 @@ async function runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, s '--format', 'json', '--json-strict', '--timeout', String(timeoutSeconds), - 'flow', 'run', SINGLE_TURN_FLOW, - '--input-file', inputFile, + 'exec', '--file', '-', ]; let child; - let stdout = ''; let stderr = ''; let termination; let timer; @@ -1078,6 +1395,13 @@ async function runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, s let timedOut = false; let cancel; let dispatchUncertain = false; + let evidenceReady = false; + let evidencePersisted = false; + let evidenceObserved = false; + let eventWrites = Promise.resolve(); + let sessionObserved = null; + let parserError = null; + let collector; let closePromise; const closeOnce = () => { closePromise ??= closeRetainedAcpResources({ @@ -1086,6 +1410,7 @@ async function runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, s extraClosers: [ () => { clearTimeout(timer); }, () => { if (cancel) signal?.removeEventListener('abort', cancel); }, + () => { child?.stdin?.destroy(); }, ], }).then((evidence) => { stopDeadline = undefined; @@ -1098,13 +1423,48 @@ async function runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, s if (signal?.aborted) fail('cancelled', 'DSH ACP task was cancelled before startup.'); await mkdir(acpxHome, { recursive: true, mode: 0o700 }); await chmod(acpxHome, 0o700); - await writeFile(inputFile, `${JSON.stringify({ prompt })}\n`, { encoding: 'utf8', mode: 0o600, flag: 'wx' }); await updateTask(root, task.id, { status: 'starting', transport: 'acp', acp_client: 'acpx-cli', started_at: new Date().toISOString() }); + const persistEvidence = () => { + if (!evidenceObserved || !evidenceReady || evidencePersisted) return; + evidencePersisted = true; + eventWrites = eventWrites.then(async () => { + await updateTask(root, task.id, { + prompt_dispatched: true, + dispatch_evidence: 'authoritative', + dispatch_uncertain: false, + }); + await appendTaskEvent(root, task.id, { + type: 'transport', + state: 'prompt_dispatched', + transport: 'acp', + client: 'acpx-cli', + dispatch_evidence: 'authoritative', + }); + }); + }; + collector = createAcpExecCollector({ + prompt, + onSession: (value) => { + sessionObserved = value; + eventWrites = eventWrites.then(async () => { + await updateTask(root, task.id, { acp_session_id: value }); + await appendTaskEvent(root, task.id, { type: 'transport', state: 'session_ready', transport: 'acp' }); + }); + }, + onAuthoritative: () => { + evidenceObserved = true; + persistEvidence(); + }, + onText: (value) => { + const compact = { type: 'text_delta', text: boundedText(value, prompt, { remaining: MAX_EVENT_TEXT }) }; + eventWrites = eventWrites.then(() => appendTaskEvent(root, task.id, { type: 'provider', event: compact })); + }, + }); child = spawn(process.env.CODEX_CO_ENGINEER_ACPX_COMMAND ?? 'acpx', argv, { cwd, env: acpxTaskEnvironment(acpxHome, task), detached: true, - stdio: ['ignore', 'pipe', 'pipe'], + stdio: ['pipe', 'pipe', 'pipe'], }); const spawned = new Promise((resolve, reject) => { child.once('spawn', resolve); @@ -1113,8 +1473,21 @@ async function runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, s const closed = new Promise((resolve) => { child.once('close', (code, childSignal) => resolve({ code, signal: childSignal })); }); - child.stdout.on('data', (chunk) => { stdout = `${stdout}${chunk}`.slice(-1024 * 1024); }); - child.stderr.on('data', (chunk) => { stderr = `${stderr}${chunk}`.slice(-256 * 1024); }); + child.stdout.on('data', (chunk) => { + if (parserError) return; + try { + collector.feed(chunk); + persistEvidence(); + } catch (error) { + parserError ??= error; + cancel?.(); + } + }); + child.stderr.on('data', (chunk) => { + stderr = `${stderr}${chunk.toString('utf8')}`; + if (Buffer.byteLength(stderr, 'utf8') > MAX_ACPX_EXEC_STDERR) stderr = stderr.slice(-MAX_ACPX_EXEC_STDERR); + }); + child.stdin.on('error', () => {}); cancel = () => { termination ??= requestChildTreeTermination(child); }; @@ -1124,15 +1497,17 @@ async function runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, s cancel(); }); await spawned; - // ACPX has spawned, but its JSON flow protocol does not acknowledge that - // the prompt was accepted. Treat all later failures as non-replayable. + // ACPX has spawned, but an outbound transcript frame only proves an + // attempt. Treat all later failures as non-replayable even before the + // correlated inbound session/update or prompt response arrives. dispatchUncertain = true; + const requestId = task.request_id ?? randomUUID(); await updateTask(root, task.id, { status: 'running', dispatch_intent: true, dispatch_uncertain: true, fallback_safe: false, - request_id: randomUUID(), + request_id: requestId, provider_process_group: child.pid, provider_process_start_ticks: processStartTicks(child.pid), }); @@ -1143,53 +1518,64 @@ async function runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, s client: 'acpx-cli', reason: 'ACPX does not provide an authoritative prompt-sent acknowledgement.', }); + evidenceReady = true; + persistEvidence(); + if (parserError) throw parserError; + child.stdin.end(prompt); if (signal?.aborted) fail('cancelled', 'DSH ACP task was cancelled before dispatch acknowledgement.'); if (timedOut) fail('timeout', 'DSH ACP task exceeded its independent deadline.'); const exit = await closed; const treeStopped = await (termination ??= terminateChildTree(child)); signal?.removeEventListener('abort', cancel); + if (parserError) throw parserError; + const collected = collector.finish(); + // finish() can consume a final frame without a trailing newline and thus + // enqueue the last evidence/event writes. Drain only after parsing it. + await eventWrites; if (signal?.aborted) fail('cancelled', 'DSH ACP task was cancelled.'); if (timedOut) fail('timeout', 'DSH ACP task exceeded its independent deadline.'); if (!treeStopped) fail('acpx_cleanup_incomplete', 'ACPX process tree remained after termination.'); + if (collected.transportError) throw collected.transportError; + if (collected.resultError) throw collected.resultError; + if (collected.promptError) throw collected.promptError; if (exit.code !== 0) { - const detail = sanitizeText(stderr.trim() || stdout.trim() || `ACPX exited ${exit.code ?? exit.signal}`, prompt); + const detail = sanitizeText(stderr.trim(), prompt).replace(/[\r\n\t]+/gu, ' ').trim() + || `ACPX exited ${exit.code ?? exit.signal}`; fail('acpx_failed', detail.slice(-MAX_EVENT_TEXT)); } - const flow = parseFlowResult(stdout); - if (flow.status !== 'completed') fail('acpx_failed', `ACPX flow ended in ${flow.status}.`); - const rawOutput = flow.outputs?.delegate; - const outputCandidates = [rawOutput?.text, rawOutput?.result, rawOutput?.output]; - const outputValue = typeof rawOutput === 'string' - ? rawOutput - : outputCandidates.find((candidate) => typeof candidate === 'string' && candidate.length > 0) - ?? outputCandidates.find((candidate) => typeof candidate === 'string') - ?? rawOutput; - const bounded = typeof outputValue === 'string' - ? boundedProviderResult(outputValue, { sanitize: (text) => sanitizeText(text, prompt) }) - : boundedProviderValue(outputValue, { sanitize: (text) => sanitizeText(text, prompt) }); + if (!collected.authoritative || !collected.promptResponded) { + fail('acpx_invalid_result', 'ACPX did not return a correlated prompt response.'); + } + if (collected.stopReason === 'cancelled') fail('cancelled', 'DSH ACP task was cancelled by the provider.'); + const bounded = collected.output; const output = bounded.value; const compact = { type: 'text_delta', text: typeof output === 'string' ? output : 'DSH ACP task completed.' }; - await appendTaskEvent(root, task.id, { type: 'provider', event: compact }); const closeEvidence = await closeOnce(); const terminal = await persistWorkerTerminal(root, task.id, { status: 'completed', error: null, - stop_reason: 'end_turn', + stop_reason: collected.stopReason, last_event: compact, result: output, ...Object.fromEntries(Object.entries(bounded).filter(([key]) => key.startsWith('result_'))), provider_process_group: null, provider_process_start_ticks: null, - acp_session_id: Object.values(flow.sessionBindings ?? {})[0]?.acpSessionId ?? null, + acp_session_id: collected.sessionId ?? sessionObserved, + prompt_dispatched: true, + dispatch_evidence: 'authoritative', + dispatch_uncertain: false, + dispatch_intent: true, }, closeEvidence, { wtb_handoff: 'not_applicable' }); - return attachLocalProviderResultSink(root, terminal, outputValue, false); + return attachLocalProviderResultSink(root, terminal, output, bounded.result_truncated === true); } catch (error) { if (!dispatchUncertain && !authenticationFailure(error)) { const fallback = await fallbackToCliIfSafe({ root, task, prompt, signal, error }); if (fallback) return fallback; } + await eventWrites.catch(() => {}); const current = (await readTask(root, task.id)).task; - const status = signal?.aborted ? 'cancelled' : (error?.code === 'timeout' || timedOut ? 'timeout' : 'failed'); + const status = signal?.aborted || error?.code === 'cancelled' + ? 'cancelled' : (error?.code === 'timeout' || timedOut ? 'timeout' : 'failed'); const failure = publicError(error, prompt); const closeEvidence = await closeOnce(); await persistWorkerTerminal(root, task.id, { @@ -1203,7 +1589,6 @@ async function runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, s } finally { await closeOnce(); await removeAcpxTaskHome(root, task.id, acpxHome); - await rm(inputFile, { force: true }); } } @@ -1226,6 +1611,64 @@ async function makeRuntime({ root, cwd, configuration, timeoutMs, taskId, signal }); } +/** + * Reattach to an already acknowledged persistent ACP session. This path is + * intentionally observation-only: it calls ensureSession with the recorded + * session identity and never starts a new turn, so a worker restart cannot + * replay the acknowledged prompt. + */ +export async function reconnectAcpTask({ root, taskId, signal, runtimeFactory } = {}) { + if (typeof root !== 'string' || !path.isAbsolute(root)) fail('invalid_state_dir', 'root must be absolute.'); + const { task } = await readTask(root, taskId); + if (!['grok', 'cursor-local'].includes(task.provider) + || task.prompt_dispatched !== true + || task.dispatch_evidence !== 'authoritative' + || typeof task.acp_session_id !== 'string' + || task.acp_session_id.length === 0 + || ['cancelled', 'cancelling', 'completed', 'failed', 'timeout'].includes(task.status)) { + return Object.freeze({ reconnected: false, reason: 'session_resume_not_available' }); + } + if (signal?.aborted) return Object.freeze({ reconnected: false, reason: 'cancelled' }); + const cwd = requireAbsoluteDirectory(task.cwd); + const configuration = providerConfiguration(task); + const timeoutMs = taskTimeoutMs(task); + const childEnv = providerChildEnvironment(task); + const runtime = await (typeof runtimeFactory === 'function' + ? runtimeFactory({ root, task, cwd, configuration, timeoutMs, signal, env: childEnv }) + : makeRuntime({ root, cwd, configuration, timeoutMs, taskId, signal, env: childEnv })); + let handle = null; + try { + handle = await runtime.ensureSession({ + sessionKey: task.session_key ?? `${task.provider}:${task.id}`, + agent: configuration.agent, + mode: 'persistent', + cwd, + resumeSessionId: task.acp_session_id, + }); + // getStatus is deliberately used only to prove that the attached session + // is observable. Its provider payload is not copied into the receipt. + if (typeof runtime.getStatus === 'function') await runtime.getStatus({ handle }); + const sessionId = handle.backendSessionId ?? handle.agentSessionId ?? task.acp_session_id; + await updateTask(root, taskId, { + status: 'running', + transport: 'acp', + acp_session_id: sessionId, + runtime_recovery: 'same_session_reconnected', + }); + await appendTaskEvent(root, taskId, { + type: 'transport', + state: 'session_reconnected', + transport: 'acp', + prompt_replayed: false, + }); + return Object.freeze({ reconnected: true, session_id: sessionId, prompt_replayed: false }); + } catch { + return Object.freeze({ reconnected: false, reason: 'session_reconnect_failed' }); + } finally { + try { await runtime.close?.({ handle, discardPersistentState: false }); } catch { /* retain evidence for later reconciliation */ } + } +} + /** * Run one task through ACP. The task prompt is read from the owner-only task * store, so it never appears in argv or the public task record. @@ -1243,7 +1686,7 @@ export async function runAcpTask({ root, taskId, signal } = {}) { if (!Number.isSafeInteger(timeoutMs) || timeoutMs < 1) fail('invalid_timeout', 'timeout_ms must be at least 1000.'); if (task.provider === 'dsh') { - return runDshFlow({ root, task, prompt, cwd, configuration, timeoutMs, signal }); + return runDshExec({ root, task, prompt, cwd, configuration, timeoutMs, signal }); } const childEnv = providerChildEnvironment(task); @@ -1302,19 +1745,25 @@ export async function runAcpTask({ root, taskId, signal } = {}) { // From this point onward the provider may have accepted the prompt. A // supervisor must reconcile this task, never replay it through CLI. await updateTask(root, taskId, { prompt_dispatched: true }); + await appendTaskEvent(root, taskId, { type: 'transport', state: 'prompt_dispatched', transport: 'acp' }); + await updateTask(root, taskId, { dispatch_evidence: 'authoritative' }); const cancel = () => turn.cancel({ reason: 'signal' }).catch(() => {}); controller.signal.addEventListener('abort', cancel, { once: true }); let lastEvent = null; - const observedEvents = []; + let observedUnsupportedQuestion = false; const output = createProviderResultAccumulator({ sanitize: (text) => sanitizeText(text, prompt) }); const complete = createLocalProviderResultCollectorV1(); + const grokFinal = task.provider === 'grok' + ? createGrokFinalResponseReducerV1({ sanitize: (text) => sanitizeText(text, prompt) }) + : null; try { for await (const event of turn.events) { - observedEvents.push(event); + observedUnsupportedQuestion ||= isStructuredAskUserQuestionUnsupported(event); if (event?.type === 'text_delta' && event.stream !== 'thought' && typeof event.text === 'string') { output.append(event.text); complete.append(event.text); } + grokFinal?.append(event); const compact = boundedEvent(event, prompt); await appendTaskEvent(root, taskId, { type: 'provider', event: compact }); lastEvent = compact; @@ -1326,7 +1775,7 @@ export async function runAcpTask({ root, taskId, signal } = {}) { const result = await turn.result; const current = (await readTask(root, taskId)).task; const unsupportedQuestion = isStructuredAskUserQuestionUnsupported(lastEvent) - || eventsHaveUnsupportedAskUserQuestion(observedEvents); + || observedUnsupportedQuestion; const latchedQuestion = liveQuestionId(current.attention); let status = result.status === 'completed' ? 'completed' : result.status; let terminalError; @@ -1336,7 +1785,10 @@ export async function runAcpTask({ root, taskId, signal } = {}) { status = 'failed'; terminalError = questionBridgeUnavailableError(); } - const bounded = output.finish(); + const fullBounded = output.finish(); + const fullSnapshot = complete.snapshot(); + const reduced = grokFinal?.finish({ turnResult: { ...result, status }, fullSnapshot }); + const bounded = reduced?.bounded ?? fullBounded; const closeEvidence = await closeOnce(); const terminal = await persistWorkerTerminal(root, taskId, { status, @@ -1347,9 +1799,10 @@ export async function runAcpTask({ root, taskId, signal } = {}) { ...(unsupportedQuestion && !latchedQuestion ? { question_bridge: 'unavailable' } : {}), ...(terminalError ? { error: terminalError, fallback_safe: false } : {}), }, closeEvidence, { wtb_handoff: 'not_applicable' }); - const snapshot = complete.snapshot(); + const snapshot = reduced?.snapshot ?? fullSnapshot; return attachLocalProviderResultSink( - root, terminal, snapshot.source, false, snapshot.overflow === true, + root, terminal, snapshot.source, false, + fullSnapshot.overflow === true || snapshot.overflow === true, ); } catch (error) { const failure = publicError(error, prompt); @@ -1416,7 +1869,7 @@ export async function runAcpWorkerCli(argv, { await awaitSupervisorRegistration(request.root, request.task_id, controller.signal); await removeStalePromptTransports(request.root, request.task_id); if (env.WORKTREE_BOOTSTRAP_TASK) { - await runFileImpl('worktree-bootstrap', [ + await runFileImpl(BUNDLED_WORKTREE_BOOTSTRAP, [ 'verify', env.WORKTREE_BOOTSTRAP_TASK, '--repo', diff --git a/plugins/codex-co-engineer/mcp/v3/aggregate-run-anchor.mjs b/plugins/codex-co-engineer/mcp/v3/aggregate-run-anchor.mjs index b7eccb0..4aedf21 100644 --- a/plugins/codex-co-engineer/mcp/v3/aggregate-run-anchor.mjs +++ b/plugins/codex-co-engineer/mcp/v3/aggregate-run-anchor.mjs @@ -117,7 +117,10 @@ export const SELECTION_ANSWER_KEYS = capturedFreeze(['assignment_id', 'model', ' export const MAX_AGGREGATE_RUNS = 64; export const MAX_AGGREGATE_RUN_DIRECTORY_ENTRIES = 16; -export const MAX_AGGREGATE_ROOT_ENTRIES = 8; +export const MAX_AGGREGATE_TEMPORARIES = 8; +// The initialized root has marker, claims, and runs entries. Audits can also +// overlap the namespace lock plus the declared bounded owner/temp allowance. +export const MAX_AGGREGATE_ROOT_ENTRIES = MAX_AGGREGATE_TEMPORARIES + 4; export const MAX_AGGREGATE_CLAIMS_DIRECTORY_ENTRIES = MAX_AGGREGATE_RUNS + 4; export const MAX_AGGREGATE_RUNS_DIRECTORY_ENTRIES = MAX_AGGREGATE_RUNS + 4; export const MAX_AGGREGATE_ANCHOR_BYTES = 64 * 1024; @@ -127,7 +130,6 @@ export const MAX_AGGREGATE_MARKER_BYTES = 512; export const MAX_AGGREGATE_STAMP_BYTES = 256; export const MAX_AGGREGATE_LOCK_BYTES = 160; export const MAX_AGGREGATE_RECORD_BYTES = 16 * 1024; -export const MAX_AGGREGATE_TEMPORARIES = 8; export const MAX_AGGREGATE_FILENAME_BYTES = 80; export const MAX_AGGREGATE_DIAGNOSTIC_BYTES = 160; export const AGGREGATE_LOCK_WAIT_MS = 8_000; diff --git a/plugins/codex-co-engineer/mcp/v3/attention-batch.mjs b/plugins/codex-co-engineer/mcp/v3/attention-batch.mjs index d448c24..eab4540 100644 --- a/plugins/codex-co-engineer/mcp/v3/attention-batch.mjs +++ b/plugins/codex-co-engineer/mcp/v3/attention-batch.mjs @@ -83,6 +83,10 @@ export const ATTENTION_BATCH_ITEM_KEYS = capturedFreeze([ 'question_id', 'event_cursor', 'question_digest', 'prompt', 'options', 'reply_capability', 'disposition', 'deadline_at', ]); +export const ATTENTION_BATCH_OPTION_KEYS = capturedFreeze([ + 'optionId', 'kind', 'name', 'label', 'description', +]); +const ATTENTION_BATCH_OPTION_REQUIRED_KEYS = capturedFreeze(['kind']); export const ATTENTION_BATCH_PROVIDERS = capturedFreeze([ 'grok', 'cursor-local', 'cursor-cloud', 'dsh', ]); @@ -425,12 +429,33 @@ function assertOptions(value, path) { for (let index = 0; index < value.length; index += 1) { const optionPath = `${path}[${index}]`; const option = ownDataValue(value, STRING(index), optionPath); - if (typeof option !== 'string' || option.length === 0 - || capturedUtf8ByteLength(option) > MAX_ATTENTION_OPTION_BYTES) { + if (typeof option === 'string') { + if (option.length === 0 || capturedUtf8ByteLength(option) > MAX_ATTENTION_OPTION_BYTES) { + failBatch('invalid_format', optionPath, + `${optionPath} must be a bounded non-empty option string.`); + } + options.push(option); + continue; + } + const fields = closedObject( + option, optionPath, ATTENTION_BATCH_OPTION_KEYS, ATTENTION_BATCH_OPTION_REQUIRED_KEYS, + ); + const normalized = capturedCreate(null); + for (const key of ATTENTION_BATCH_OPTION_KEYS) { + if (!capturedHasOwn(fields, key)) continue; + const text = fields[key]; + if (typeof text !== 'string' || text.length === 0 + || capturedUtf8ByteLength(text) > MAX_ATTENTION_OPTION_BYTES) { + failBatch('invalid_format', `${optionPath}.${key}`, + `${optionPath}.${key} must be bounded non-empty option text.`); + } + normalized[key] = text; + } + if (typeof normalized.kind !== 'string') { failBatch('invalid_format', optionPath, - `${optionPath} must be a bounded non-empty option string.`); + `${optionPath}.kind is required for a structured option.`); } - options.push(option); + options.push(normalized); } return options; } diff --git a/plugins/codex-co-engineer/mcp/v3/compact-task.mjs b/plugins/codex-co-engineer/mcp/v3/compact-task.mjs index 2126741..d3dcaf7 100644 --- a/plugins/codex-co-engineer/mcp/v3/compact-task.mjs +++ b/plugins/codex-co-engineer/mcp/v3/compact-task.mjs @@ -74,6 +74,27 @@ export function utf8Head(value, maxBytes) { return `${buffer.subarray(0, end).toString('utf8')}${ELLIPSIS}`; } +// Shared by durable run storage and public result projection. +export function boundProviderResult(value, maxBytes = 8_192) { + if (value === undefined || value === null) return { value: null, truncated: false }; + let safe; + let encoded; + try { + safe = sanitizePublicReceipt(value); + encoded = JSON.stringify(safe); + } catch { + return { value: null, truncated: false }; + } + if (encoded === undefined) return { value: null, truncated: false }; + if (Buffer.byteLength(encoded, 'utf8') <= maxBytes) return { value: safe, truncated: false }; + let preview = utf8Head(typeof safe === 'string' ? safe : encoded, maxBytes - 2); + // Escaped quotes/control characters also count toward the JSON byte cap. + while (byteLength(preview) > maxBytes) { + preview = utf8Head(preview, Math.floor(Buffer.byteLength(preview, 'utf8') / 2)); + } + return { value: preview, truncated: true }; +} + function boundedString(value, maxBytes, { tail = false } = {}) { if (typeof value !== 'string') return null; return tail ? tailText(value, maxBytes) : utf8Head(value, maxBytes); @@ -265,7 +286,7 @@ function essentialCompactEnvelope(payload) { task_id: boundedString(payload.task_id, COMPACT_ID_BYTES) ?? utf8Head('unknown', COMPACT_ID_BYTES), provider: boundedString(payload.provider, COMPACT_SCALAR_BYTES), ...(payload.provider === 'dsh' ? { - dsh_model: boundedString(payload.dsh_model, COMPACT_SCALAR_BYTES) ?? 'muse-spark-1.2-contributor', + dsh_model: boundedString(payload.dsh_model, COMPACT_SCALAR_BYTES) ?? 'meta/muse-spark-1.3-contributor', } : {}), role: payload.role === 'review' || payload.role === 'implement' ? payload.role : null, status: boundedString(payload.status, COMPACT_SCALAR_BYTES), @@ -312,7 +333,7 @@ function lastResortEnvelope(payload) { task_id: boundedString(payload.task_id, COMPACT_ID_BYTES) ?? utf8Head('unknown', COMPACT_ID_BYTES), provider: payload.provider === 'dsh' ? 'dsh' : null, ...(payload.provider === 'dsh' ? { - dsh_model: boundedString(payload.dsh_model, COMPACT_SCALAR_BYTES) ?? 'muse-spark-1.2-contributor', + dsh_model: boundedString(payload.dsh_model, COMPACT_SCALAR_BYTES) ?? 'meta/muse-spark-1.3-contributor', } : {}), role: payload.role === 'review' || payload.role === 'implement' ? payload.role : null, status: boundedString(payload.status, COMPACT_SCALAR_BYTES), @@ -510,7 +531,7 @@ export function projectCompactTask({ task, progress = null, runtime = null, extr task_id: typeof task.id === 'string' ? task.id : 'unknown', provider: typeof task.provider === 'string' ? task.provider : null, ...(task.provider === 'dsh' ? { - dsh_model: typeof task.dsh_model === 'string' ? task.dsh_model : 'muse-spark-1.2-contributor', + dsh_model: typeof task.dsh_model === 'string' ? task.dsh_model : 'meta/muse-spark-1.3-contributor', } : {}), role: task.role === 'review' || task.role === 'implement' ? task.role : null, status: typeof task.status === 'string' ? task.status : null, diff --git a/plugins/codex-co-engineer/mcp/v3/consent-grants.mjs b/plugins/codex-co-engineer/mcp/v3/consent-grants.mjs new file mode 100644 index 0000000..1a0c34f --- /dev/null +++ b/plugins/codex-co-engineer/mcp/v3/consent-grants.mjs @@ -0,0 +1,397 @@ +import { execFile as nodeExecFile } from 'node:child_process'; +import { createHash, randomUUID } from 'node:crypto'; +import { + chmod, lstat, mkdir, open, readFile, realpath, rename, stat, unlink, writeFile, +} from 'node:fs/promises'; +import path from 'node:path'; +import { promisify } from 'node:util'; + +export const CONSENT_GRANT_SCHEMA = 'codex-co-engineer.consent-grants.v1'; +export const CONSENT_GRANT_VERSION = 1; +export const CONSENT_GRANT_DURATION = 'repository_and_selected_providers'; +export const CONSENT_RUN_DURATION = 'this_run_only'; +export const CONSENT_GRANT_STORE_FILE = 'consent-grants.json'; + +const execFile = promisify(nodeExecFile); +const PROVIDERS = new Set(['grok', 'cursor-local', 'cursor-cloud', 'dsh']); +const MAX_STORE_BYTES = 64 * 1024; +const LOCK_WAIT_ATTEMPTS = 200; +const SHA256 = /^[0-9a-f]{64}$/u; +const CLOSED_GIT_ENV = Object.freeze({ + PATH: '/usr/bin:/bin', + GIT_CONFIG_NOSYSTEM: '1', + GIT_CONFIG_GLOBAL: '/dev/null', + GIT_CONFIG_SYSTEM: '/dev/null', + GIT_TERMINAL_PROMPT: '0', + GIT_OPTIONAL_LOCKS: '0', + GIT_PAGER: 'cat', + LANG: 'C', + LC_ALL: 'C', +}); + +function grantError(code, message) { + return Object.assign(new Error(message), { code }); +} + +function plainObject(value) { + if (value === null || typeof value !== 'object' || Array.isArray(value)) return false; + const prototype = Object.getPrototypeOf(value); + return prototype === Object.prototype || prototype === null; +} + +function exactKeys(value, keys) { + if (!plainObject(value)) return false; + const actual = Object.keys(value).sort(); + const expected = [...keys].sort(); + return actual.length === expected.length && actual.every((key, index) => key === expected[index]); +} + +function canonicalProviders(values) { + if (!Array.isArray(values) || values.length < 1 || values.length > 4) { + throw grantError('consent_grant_invalid', 'Consent grant providers are invalid.'); + } + const providers = []; + for (const value of values) { + if (typeof value !== 'string' || !PROVIDERS.has(value) || providers.includes(value)) { + throw grantError('consent_grant_invalid', 'Consent grant providers are invalid.'); + } + providers.push(value); + } + return providers.sort(); +} + +function normalizeOrigin(value) { + if (typeof value !== 'string' || value.length < 1 || value.length > 4096) return null; + const scp = /^(?:[^@/:]+@)?([^/:]+):(.+)$/u.exec(value); + if (scp && !value.includes('://')) { + const pathname = scp[2].replace(/^\/+|\/+$/gu, ''); + return /^[A-Za-z0-9.-]+$/u.test(scp[1]) && pathname.length > 0 + ? `ssh://${scp[1].toLowerCase()}/${pathname}` + : `sha256:${createHash('sha256').update(value, 'utf8').digest('hex')}`; + } + try { + const parsed = new URL(value); + if (!['https:', 'http:', 'ssh:', 'git:'].includes(parsed.protocol)) { + return `sha256:${createHash('sha256').update(value, 'utf8').digest('hex')}`; + } + parsed.username = ''; + parsed.password = ''; + parsed.search = ''; + parsed.hash = ''; + parsed.hostname = parsed.hostname.toLowerCase(); + if ((parsed.protocol === 'https:' && parsed.port === '443') + || (parsed.protocol === 'http:' && parsed.port === '80') + || (parsed.protocol === 'ssh:' && parsed.port === '22')) parsed.port = ''; + parsed.pathname = parsed.pathname.replace(/\/+$/gu, ''); + return parsed.hostname && parsed.pathname + ? parsed.toString() + : `sha256:${createHash('sha256').update(value, 'utf8').digest('hex')}`; + } catch { + return `sha256:${createHash('sha256').update(value, 'utf8').digest('hex')}`; + } +} + +async function git(repositoryPath, args, { optional = false } = {}) { + try { + const result = await execFile('git', ['-C', repositoryPath, ...args], { + env: CLOSED_GIT_ENV, + encoding: 'utf8', + timeout: 5_000, + maxBuffer: 16 * 1024, + windowsHide: true, + }); + return result.stdout.trim(); + } catch (error) { + if (optional && error?.code === 1 && error?.killed !== true) return null; + throw grantError('consent_repository_identity_failed', + 'The repository identity could not be resolved for consent.'); + } +} + +export async function resolveConsentRepositoryIdentity(repositoryPath) { + if (typeof repositoryPath !== 'string' || !path.isAbsolute(repositoryPath)) { + throw grantError('consent_repository_identity_failed', + 'Consent requires an absolute repository path.'); + } + const commonDirValue = await git(repositoryPath, + ['rev-parse', '--path-format=absolute', '--git-common-dir']); + const commonDirPath = await realpath(commonDirValue).catch(() => null); + if (commonDirPath === null || !path.isAbsolute(commonDirPath)) { + throw grantError('consent_repository_identity_failed', + 'The repository common directory could not be resolved for consent.'); + } + const metadata = await stat(commonDirPath, { bigint: true }).catch(() => null); + if (metadata === null || !metadata.isDirectory()) { + throw grantError('consent_repository_identity_failed', + 'The repository common directory is unavailable for consent.'); + } + const rawOrigin = await git(repositoryPath, ['config', '--local', '--get', 'remote.origin.url'], { + optional: true, + }); + return Object.freeze({ + common_dir_path: commonDirPath, + device: String(metadata.dev), + inode: String(metadata.ino), + birthtime_ns: String(metadata.birthtimeNs), + origin: normalizeOrigin(rawOrigin), + }); +} + +function identityKey(identity) { + return JSON.stringify([ + identity.common_dir_path, identity.device, identity.inode, identity.birthtime_ns, identity.origin, + ]); +} + +function grantId(identity) { + return createHash('sha256').update(identityKey(identity), 'utf8').digest('hex'); +} + +function validateIdentity(value) { + if (!exactKeys(value, ['common_dir_path', 'device', 'inode', 'birthtime_ns', 'origin']) + || typeof value.common_dir_path !== 'string' || !path.isAbsolute(value.common_dir_path) + || typeof value.device !== 'string' || !/^\d+$/u.test(value.device) + || typeof value.inode !== 'string' || !/^\d+$/u.test(value.inode) + || typeof value.birthtime_ns !== 'string' || !/^\d+$/u.test(value.birthtime_ns) + || (value.origin !== null && (typeof value.origin !== 'string' + || (!/^sha256:[0-9a-f]{64}$/u.test(value.origin) + && normalizeOrigin(value.origin) !== value.origin)))) { + throw grantError('consent_grant_store_invalid', 'Consent grant state is malformed.'); + } + return value; +} + +function validateStore(value) { + if (!exactKeys(value, ['schema', 'version', 'grants']) + || value.schema !== CONSENT_GRANT_SCHEMA || value.version !== CONSENT_GRANT_VERSION + || !Array.isArray(value.grants) || value.grants.length > 256) { + throw grantError('consent_grant_store_invalid', 'Consent grant state is malformed.'); + } + const ids = new Set(); + for (const grant of value.grants) { + if (!exactKeys(grant, ['grant_id', 'repository', 'providers', 'granted_at']) + || typeof grant.grant_id !== 'string' || !SHA256.test(grant.grant_id) + || typeof grant.granted_at !== 'string' || !Number.isFinite(Date.parse(grant.granted_at))) { + throw grantError('consent_grant_store_invalid', 'Consent grant state is malformed.'); + } + validateIdentity(grant.repository); + if (grant.grant_id !== grantId(grant.repository) || ids.has(grant.grant_id)) { + throw grantError('consent_grant_store_invalid', 'Consent grant state is malformed.'); + } + ids.add(grant.grant_id); + try { canonicalProviders(grant.providers); } catch { + throw grantError('consent_grant_store_invalid', 'Consent grant state is malformed.'); + } + if (JSON.stringify(grant.providers) !== JSON.stringify(canonicalProviders(grant.providers))) { + throw grantError('consent_grant_store_invalid', 'Consent grant state is malformed.'); + } + } + return value; +} + +function emptyStore() { + return { schema: CONSENT_GRANT_SCHEMA, version: CONSENT_GRANT_VERSION, grants: [] }; +} + +async function assertPrivate(metadata, kind) { + const uid = typeof process.getuid === 'function' ? process.getuid() : null; + if (metadata.isSymbolicLink() || (uid !== null && metadata.uid !== uid) + || (metadata.mode & 0o077) !== 0) { + throw grantError('consent_grant_store_unsafe', `Consent grant ${kind} is not owner-only.`); + } +} + +export function createConsentGrantStore({ root }) { + if (typeof root !== 'string' || !path.isAbsolute(root)) { + throw new TypeError('createConsentGrantStore requires an absolute state root.'); + } + const directory = path.resolve(root); + const file = path.join(directory, CONSENT_GRANT_STORE_FILE); + const lockFile = path.join(directory, '.consent-grants.lock'); + + async function prepareDirectory() { + await mkdir(directory, { recursive: true, mode: 0o700 }); + await assertPrivate(await lstat(directory), 'directory'); + } + + async function acquireLock() { + const nonce = randomUUID(); + for (let attempt = 0; attempt < LOCK_WAIT_ATTEMPTS; attempt += 1) { + try { + const handle = await open(lockFile, 'wx', 0o600); + try { + await handle.writeFile(`${JSON.stringify({ pid: process.pid, nonce })}\n`); + await handle.sync(); + } catch (error) { + await handle.close().catch(() => {}); + await unlink(lockFile).catch(() => {}); + throw error; + } + return { handle, nonce }; + } catch (error) { + if (error?.code !== 'EEXIST') throw error; + if (attempt === LOCK_WAIT_ATTEMPTS - 1) { + throw grantError('consent_grant_store_busy', + 'Consent grant state is busy. If no consent command or MCP server is running, remove the owner-only .consent-grants.lock file from the Co-Engineer state directory.'); + } + await new Promise((resolve) => setTimeout(resolve, 25)); + } + } + throw grantError('consent_grant_store_busy', 'Consent grant state is busy.'); + } + + async function withMutationLock(operation) { + await prepareDirectory(); + const lock = await acquireLock(); + try { + return await operation(); + } finally { + await lock.handle.close().catch(() => {}); + try { + const current = JSON.parse(await readFile(lockFile, 'utf8')); + if (current.nonce === lock.nonce) await unlink(lockFile); + } catch (error) { + if (error?.code !== 'ENOENT') throw error; + } + } + } + + async function load() { + await prepareDirectory(); + let metadata; + try { metadata = await lstat(file); } catch (error) { + if (error?.code === 'ENOENT') return emptyStore(); + throw grantError('consent_grant_store_invalid', 'Consent grant state could not be read.'); + } + await assertPrivate(metadata, 'file'); + if (!metadata.isFile() || metadata.size > MAX_STORE_BYTES) { + throw grantError('consent_grant_store_invalid', 'Consent grant state is malformed.'); + } + let parsed; + try { parsed = JSON.parse(await readFile(file, 'utf8')); } catch { + throw grantError('consent_grant_store_invalid', 'Consent grant state is malformed.'); + } + return validateStore(parsed); + } + + async function save(store) { + validateStore(store); + await prepareDirectory(); + const temporary = `${file}.tmp-${process.pid}-${randomUUID()}`; + const bytes = `${JSON.stringify(store, null, 2)}\n`; + if (Buffer.byteLength(bytes) > MAX_STORE_BYTES) { + throw grantError('consent_grant_store_invalid', 'Consent grant state exceeds its size limit.'); + } + try { + await writeFile(temporary, bytes, { encoding: 'utf8', mode: 0o600, flag: 'wx' }); + await chmod(temporary, 0o600); + const handle = await open(temporary, 'r'); + try { await handle.sync(); } finally { await handle.close(); } + await rename(temporary, file); + const directoryHandle = await open(directory, 'r'); + try { await directoryHandle.sync(); } finally { await directoryHandle.close(); } + } finally { + await unlink(temporary).catch(() => {}); + } + } + + async function lookup({ repositoryPath, repositoryIdentity, providers }) { + const requested = canonicalProviders(providers); + const repository = repositoryIdentity === undefined + ? await resolveConsentRepositoryIdentity(repositoryPath) + : validateIdentity(repositoryIdentity); + const store = await load(); + const grant = store.grants.find((item) => item.grant_id === grantId(repository)); + if (!grant || identityKey(grant.repository) !== identityKey(repository) + || !requested.every((provider) => grant.providers.includes(provider))) return null; + return Object.freeze({ + approved: true, + duration: CONSENT_GRANT_DURATION, + source: 'durable_grant', + grant_id: grant.grant_id, + }); + } + + async function assertIdentityCurrent({ repositoryPath, repositoryIdentity }) { + const expected = validateIdentity(repositoryIdentity); + const current = await resolveConsentRepositoryIdentity(repositoryPath); + if (identityKey(expected) !== identityKey(current)) { + throw grantError('consent_repository_identity_changed', + 'The repository identity changed while consent was open. Request consent again.'); + } + return true; + } + + async function remember({ + repositoryPath, repositoryIdentity, providers, grantedAt = new Date().toISOString(), + }) { + const requested = canonicalProviders(providers); + if (typeof grantedAt !== 'string' || !Number.isFinite(Date.parse(grantedAt))) { + throw grantError('consent_grant_invalid', 'Consent grant time is invalid.'); + } + const repository = await resolveConsentRepositoryIdentity(repositoryPath); + if (repositoryIdentity !== undefined + && identityKey(validateIdentity(repositoryIdentity)) !== identityKey(repository)) { + throw grantError('consent_repository_identity_changed', + 'The repository identity changed while consent was open.'); + } + return withMutationLock(async () => { + const store = await load(); + const id = grantId(repository); + const existing = store.grants.find((item) => item.grant_id === id); + const combined = canonicalProviders([...(existing?.providers ?? []), ...requested] + .filter((provider, index, values) => values.indexOf(provider) === index)); + const grant = { grant_id: id, repository, providers: combined, granted_at: grantedAt }; + store.grants = [...store.grants.filter((item) => item.grant_id !== id), grant] + .sort((left, right) => left.grant_id.localeCompare(right.grant_id)); + await save(store); + return Object.freeze({ ...grant, duration: CONSENT_GRANT_DURATION }); + }); + } + + async function list() { + const store = await load(); + return Object.freeze(store.grants.map((grant) => Object.freeze({ + grant_id: grant.grant_id, + repository: grant.repository.common_dir_path, + origin: grant.repository.origin, + providers: Object.freeze([...grant.providers]), + granted_at: grant.granted_at, + }))); + } + + async function revoke({ repositoryPath, grantId: requestedGrantId }) { + if ((repositoryPath === undefined) === (requestedGrantId === undefined)) { + throw grantError('consent_grant_invalid', 'Revoke requires exactly one repository path or grant id.'); + } + const repository = requestedGrantId === undefined + ? await resolveConsentRepositoryIdentity(repositoryPath) + : null; + if (requestedGrantId !== undefined + && (typeof requestedGrantId !== 'string' || !SHA256.test(requestedGrantId))) { + throw grantError('consent_grant_invalid', 'Consent grant id is invalid.'); + } + return withMutationLock(async () => { + const store = await load(); + const grants = store.grants.filter((item) => requestedGrantId !== undefined + ? item.grant_id !== requestedGrantId + : !(item.repository.common_dir_path === repository.common_dir_path + && item.repository.device === repository.device + && item.repository.inode === repository.inode + && item.repository.birthtime_ns === repository.birthtime_ns)); + if (grants.length === store.grants.length) return false; + store.grants = grants; + await save(store); + return true; + }); + } + + return Object.freeze({ + lookup, + remember, + list, + revoke, + resolveIdentity: resolveConsentRepositoryIdentity, + assertIdentityCurrent, + }); +} diff --git a/plugins/codex-co-engineer/mcp/v3/consent.mjs b/plugins/codex-co-engineer/mcp/v3/consent.mjs new file mode 100644 index 0000000..49aeaae --- /dev/null +++ b/plugins/codex-co-engineer/mcp/v3/consent.mjs @@ -0,0 +1,383 @@ +import { randomUUID } from 'node:crypto'; + +import { MCP_PENDING_CALL_BUDGET_MS } from './contract.mjs'; +import { + CONSENT_GRANT_DURATION, + CONSENT_RUN_DURATION, +} from './consent-grants.mjs'; + +export const NATIVE_CONSENT_METHOD = 'elicitation/create'; +export const NATIVE_CONSENT_MAX_PENDING = 8; +export const NATIVE_CONSENT_TIMEOUT_MS = 600_000; +export const NATIVE_CONSENT_SUPPORTED_PROTOCOLS = Object.freeze([ + '2025-11-25', + '2025-06-18', +]); +export const NATIVE_CONSENT_BLOCKED_CODES = Object.freeze([ + 'consent_host_unavailable', + 'consent_declined', + 'consent_cancelled', + 'consent_timed_out', + 'consent_response_invalid', + 'consent_request_aborted', + 'consent_grant_store_invalid', + 'consent_repository_identity_changed', +]); + +const RUN_ID = /^[a-z][a-z0-9-]{2,63}$/u; +const BASE_SHA = /^[a-f0-9]{40}$/iu; +const MAX_TIMEOUT_MS = MCP_PENDING_CALL_BUDGET_MS; +const ACCEPT_ACTIONS = new Set(['accept', 'decline', 'cancel']); +const PROVIDER = /^[a-z][a-z0-9-]{0,63}$/u; +const REMEMBER_LABEL = 'Remember for this repository and these providers'; +const THIS_RUN_LABEL = 'This run only'; + +const REQUESTED_SCHEMA = Object.freeze({ + type: 'object', + properties: Object.freeze({ + approval_duration: Object.freeze({ + type: 'string', + title: 'Remember this approval?', + description: 'Choose whether to remember access for this repository and exactly these providers. The default remembers it; choose this run only for a one-time approval.', + enum: Object.freeze([REMEMBER_LABEL, THIS_RUN_LABEL]), + default: REMEMBER_LABEL, + }), + }), + required: Object.freeze(['approval_duration']), + additionalProperties: false, +}); + +function isRecord(value) { + if (value === null || typeof value !== 'object' || Array.isArray(value)) return false; + try { + const prototype = Object.getPrototypeOf(value); + return prototype === Object.prototype || prototype === null; + } catch { + return false; + } +} + +function own(value, key) { + return isRecord(value) && Object.hasOwn(value, key); +} + +export function supportsNativeForm(capabilities) { + if (!isRecord(capabilities) || !isRecord(capabilities.elicitation)) return false; + const elicitation = capabilities.elicitation; + if (own(elicitation, 'form')) return isRecord(elicitation.form); + return Object.keys(elicitation).length === 0; +} + +function consentBinding(compiled) { + if (!isRecord(compiled) || typeof compiled.run_id !== 'string' || !RUN_ID.test(compiled.run_id)) { + return null; + } + const git = isRecord(compiled.git) ? compiled.git : {}; + const repositoryPath = typeof git.repository_path === 'string' + ? git.repository_path + : (typeof compiled.repository_path === 'string' ? compiled.repository_path : null); + const baseSha = typeof git.base_sha === 'string' + ? git.base_sha + : (typeof compiled.base_sha === 'string' ? compiled.base_sha : null); + if (typeof repositoryPath !== 'string' || repositoryPath.length === 0 + || !repositoryPath.startsWith('/') + || typeof baseSha !== 'string' || !BASE_SHA.test(baseSha)) { + return null; + } + if (!Array.isArray(compiled.assignments) || compiled.assignments.length < 1 + || compiled.assignments.length > NATIVE_CONSENT_MAX_PENDING) { + return null; + } + const providers = []; + for (const assignment of compiled.assignments) { + if (!isRecord(assignment) || typeof assignment.provider !== 'string' + || !PROVIDER.test(assignment.provider)) return null; + if (!providers.includes(assignment.provider)) providers.push(assignment.provider); + } + return Object.freeze({ + run_id: compiled.run_id, + repository_path: repositoryPath, + base_sha: baseSha, + providers: Object.freeze(providers), + }); +} + +function bindingKey(binding) { + return JSON.stringify([ + binding.run_id, + binding.repository_path, + binding.base_sha, + binding.providers, + ]); +} + +function consentMessage(binding) { + return [ + `Co-Engineer requests full repository access for run ${binding.run_id}.`, + `Repository: ${binding.repository_path}`, + `Base SHA: ${binding.base_sha}`, + `Selected providers: ${binding.providers.join(', ')}`, + 'Scope: full repository and Git history.', + 'Default: remember approval for this repository and the selected providers.', + 'Alternative: approve this run only.', + 'Adding a provider, changing the repository origin, or recreating the repository asks again.', + 'Remote mutations: none.', + ].join('\n'); +} + +function blocked(code) { + const status = [ + 'consent_cancelled', 'consent_timed_out', 'consent_request_aborted', + 'consent_repository_identity_changed', + ].includes(code) + ? 'required' : 'blocked'; + return Object.freeze({ status, code }); +} + +function responseInvalid() { + return blocked('consent_response_invalid'); +} + +function validateResult(result) { + if (!isRecord(result) || typeof result.action !== 'string' + || !ACCEPT_ACTIONS.has(result.action)) { + return responseInvalid(); + } + const hasContent = own(result, 'content'); + if (result.action === 'accept') { + if (!hasContent || !isRecord(result.content) + || Object.keys(result.content).length !== 1 + || !own(result.content, 'approval_duration') + || ![REMEMBER_LABEL, THIS_RUN_LABEL] + .includes(result.content.approval_duration)) { + return responseInvalid(); + } + return Object.freeze({ + approved: true, + duration: result.content.approval_duration === REMEMBER_LABEL + ? CONSENT_GRANT_DURATION + : CONSENT_RUN_DURATION, + source: 'native_form', + }); + } + if (hasContent && result.content !== undefined + && (!isRecord(result.content) || Object.keys(result.content).length > 0)) { + return responseInvalid(); + } + return blocked(result.action === 'decline' ? 'consent_declined' : 'consent_cancelled'); +} + +function validTimeout(value, fallback) { + if (!Number.isSafeInteger(value) || value < 1 || value > MAX_TIMEOUT_MS) return fallback; + return value; +} + +function validPendingLimit(value, fallback) { + if (!Number.isSafeInteger(value) || value < 1 || value > NATIVE_CONSENT_MAX_PENDING) return fallback; + return value; +} + +function nextRequestId(pending) { + let requestId; + do { + requestId = `cce-consent-${randomUUID()}`; + } while (pending.has(requestId)); + return requestId; +} + +function protocolVersion(options, getProtocolVersion) { + if (own(options, 'protocolVersion')) return options.protocolVersion; + return typeof getProtocolVersion === 'function' ? getProtocolVersion() : '2025-11-25'; +} + +function capabilitiesValue(options, getCapabilities) { + if (own(options, 'capabilities')) return options.capabilities; + return typeof getCapabilities === 'function' ? getCapabilities() : null; +} + +export function createNativeConsentTransport(options = {}) { + if (!isRecord(options) || typeof options.send !== 'function') { + throw new TypeError('createNativeConsentTransport requires a send function.'); + } + const send = options.send; + const getCapabilities = options.getCapabilities; + const getProtocolVersion = options.getProtocolVersion; + const defaultTimeout = validTimeout(options.timeoutMs, NATIVE_CONSENT_TIMEOUT_MS); + const maxPending = validPendingLimit(options.maxPending, NATIVE_CONSENT_MAX_PENDING); + const grantStore = options.grantStore ?? null; + const pending = new Map(); + const byRun = new Map(); + let closed = false; + + function settle(entry, value) { + if (pending.get(entry.request_id) !== entry) return false; + pending.delete(entry.request_id); + if (byRun.get(entry.binding.run_id) === entry) byRun.delete(entry.binding.run_id); + clearTimeout(entry.timer); + entry.signal?.removeEventListener('abort', entry.onAbort); + entry.resolve(value); + return true; + } + + async function requestConsent(compiled, requestOptions = {}) { + const binding = consentBinding(compiled); + const capabilities = capabilitiesValue(requestOptions, getCapabilities); + const version = protocolVersion(requestOptions, getProtocolVersion); + if (closed || !supportsNativeForm(capabilities) + || !NATIVE_CONSENT_SUPPORTED_PROTOCOLS.includes(version)) { + return blocked('consent_host_unavailable'); + } + if (binding === null) return responseInvalid(); + const signal = requestOptions?.signal; + if (signal?.aborted) return blocked('consent_request_aborted'); + const existing = byRun.get(binding.run_id); + if (existing) { + return existing.binding_key === bindingKey(binding) + ? existing.promise + : responseInvalid(); + } + let repositoryIdentity = null; + if (grantStore !== null) { + let remembered; + try { + repositoryIdentity = await grantStore.resolveIdentity(binding.repository_path); + remembered = await grantStore.lookup({ + repositoryPath: binding.repository_path, + repositoryIdentity, + providers: binding.providers, + }); + } catch { + return blocked('consent_grant_store_invalid'); + } + if (closed) return blocked('consent_host_unavailable'); + if (signal?.aborted) return blocked('consent_request_aborted'); + const afterLookup = byRun.get(binding.run_id); + if (afterLookup) { + return afterLookup.binding_key === bindingKey(binding) + ? afterLookup.promise + : responseInvalid(); + } + if (remembered?.approved === true + && remembered.duration === CONSENT_GRANT_DURATION + && remembered.source === 'durable_grant') return remembered; + } + if (pending.size >= maxPending) return blocked('consent_host_unavailable'); + + const request_id = nextRequestId(pending); + const params = { + message: consentMessage(binding), + requestedSchema: REQUESTED_SCHEMA, + }; + if (version !== '2025-06-18') params.mode = 'form'; + + let resolvePromise; + const promise = new Promise((resolve) => { resolvePromise = resolve; }); + const entry = { + request_id, + binding, + binding_key: bindingKey(binding), + promise, + resolve: resolvePromise, + timer: null, + signal, + onAbort: null, + repository_identity: repositoryIdentity, + responded: false, + }; + entry.onAbort = () => settle(entry, blocked('consent_request_aborted')); + pending.set(request_id, entry); + byRun.set(binding.run_id, entry); + entry.timer = setTimeout(() => settle(entry, blocked('consent_timed_out')), validTimeout( + requestOptions?.timeoutMs, + defaultTimeout, + )); + signal?.addEventListener('abort', entry.onAbort, { once: true }); + try { + send({ + jsonrpc: '2.0', + id: request_id, + method: NATIVE_CONSENT_METHOD, + params, + }); + } catch { + settle(entry, blocked('consent_host_unavailable')); + } + return promise; + } + + function handleMessage(message) { + if (!isRecord(message) || message.jsonrpc !== '2.0') return false; + const hasResult = own(message, 'result'); + const hasError = own(message, 'error'); + if (!hasResult && !hasError && own(message, 'method')) return false; + // A response is consumed before the server's method router, including an + // unmatched/late response, so it cannot trigger an error-response loop. + if (!hasResult && !hasError) return message.id !== undefined; + const entry = pending.get(message.id); + if (!entry) return true; + if (entry.responded) return true; + entry.responded = true; + if (own(message, 'method') || (hasResult && hasError)) { + settle(entry, responseInvalid()); + } else if (hasError) { + settle(entry, responseInvalid()); + } else { + const result = validateResult(message.result); + if (result.approved === true && grantStore !== null) { + Promise.resolve(grantStore.assertIdentityCurrent({ + repositoryPath: entry.binding.repository_path, + repositoryIdentity: entry.repository_identity, + })).then(async () => { + if (pending.get(entry.request_id) !== entry) return false; + if (result.duration === CONSENT_GRANT_DURATION) { + await grantStore.remember({ + repositoryPath: entry.binding.repository_path, + repositoryIdentity: entry.repository_identity, + providers: entry.binding.providers, + }); + } + return true; + }).then((active) => active && settle(entry, result)).catch((error) => { + const code = error?.code === 'consent_repository_identity_changed' + ? 'consent_repository_identity_changed' + : 'consent_grant_store_invalid'; + settle(entry, blocked(code)); + }); + } else if (result.approved === true && result.duration === CONSENT_GRANT_DURATION) { + settle(entry, blocked('consent_grant_store_invalid')); + } else { + settle(entry, result); + } + } + return true; + } + + function cancelRequest(requestId) { + const entry = pending.get(requestId); + if (!entry) return false; + return settle(entry, blocked('consent_request_aborted')); + } + + function cancelRun(runId) { + const entry = byRun.get(runId); + if (!entry) return false; + return settle(entry, blocked('consent_cancelled')); + } + + function close(reason = 'disconnect') { + closed = true; + const code = 'consent_request_aborted'; + for (const entry of [...pending.values()]) settle(entry, blocked(code)); + } + + return Object.freeze({ + cancelRequest, + cancelRun, + close, + handleMessage, + requestConsent, + get pendingCount() { return pending.size; }, + pendingRequestIds: () => Object.freeze([...pending.keys()]), + supports: () => supportsNativeForm(capabilitiesValue({}, getCapabilities)), + }); +} diff --git a/plugins/codex-co-engineer/mcp/v3/constrained-verification-runner.mjs b/plugins/codex-co-engineer/mcp/v3/constrained-verification-runner.mjs index 48d1e57..800cf23 100644 --- a/plugins/codex-co-engineer/mcp/v3/constrained-verification-runner.mjs +++ b/plugins/codex-co-engineer/mcp/v3/constrained-verification-runner.mjs @@ -656,37 +656,52 @@ function defaultKillProcessGroup(pid, signal) { } } -async function defaultListDescendants(pid) { +export async function listProcDescendants(pid, { + readDir = readdir, + readText = readFile, +} = {}) { if (!NUMBER_IS_SAFE_INTEGER(pid) || pid <= 0) deny('cleanup_uncertain', 'execution'); + if (typeof readDir !== 'function' || typeof readText !== 'function') deny('cleanup_uncertain', 'execution'); let dir; try { - dir = await readdir('/proc'); + dir = await readDir('/proc'); } catch { deny('cleanup_uncertain', 'execution'); } + if (!capturedIsArray(dir)) deny('cleanup_uncertain', 'execution'); const leftover = []; for (let index = 0; index < dir.length; index += 1) { const name = dir[index]; - if (!/^[0-9]+$/u.test(name)) continue; + if (typeof name !== 'string' || !/^[0-9]+$/u.test(name)) continue; const other = Number(name); - if (other === pid || !NUMBER_IS_SAFE_INTEGER(other)) continue; + if (other === pid || !NUMBER_IS_SAFE_INTEGER(other) || other <= 0) continue; let stat; try { - stat = await readFile(`/proc/${other}/stat`, 'utf8'); + stat = await readText(`/proc/${other}/stat`, 'utf8'); } catch (error) { - if (error && error.code === 'ENOENT') continue; + if (error && (error.code === 'ENOENT' || error.code === 'ESRCH')) continue; deny('cleanup_uncertain', 'execution'); } - const close = stat.indexOf(')'); + if (typeof stat !== 'string') deny('cleanup_uncertain', 'execution'); + const close = stat.lastIndexOf(')'); if (close < 0) deny('cleanup_uncertain', 'execution'); - const rest = stat.slice(close + 2).split(' '); + const rest = stat.slice(close + 2).trim().split(/\s+/u); + if (!/^[A-Za-z]$/u.test(rest[0] ?? '')) deny('cleanup_uncertain', 'execution'); const ppid = Number(rest[1]); const pgid = Number(rest[2]); + if (!NUMBER_IS_SAFE_INTEGER(ppid) || ppid < 0 + || !NUMBER_IS_SAFE_INTEGER(pgid) || pgid < 0) { + deny('cleanup_uncertain', 'execution'); + } if (ppid === pid || pgid === pid) ARRAY_PUSH.call(leftover, other); } return freezeList(leftover); } +async function defaultListDescendants(pid) { + return listProcDescendants(pid); +} + async function collectChildOutput(child, timeoutMs) { const stdoutChunks = []; const stderrChunks = []; diff --git a/plugins/codex-co-engineer/mcp/v3/contract.mjs b/plugins/codex-co-engineer/mcp/v3/contract.mjs index 51fbc5e..7593284 100644 --- a/plugins/codex-co-engineer/mcp/v3/contract.mjs +++ b/plugins/codex-co-engineer/mcp/v3/contract.mjs @@ -1,4 +1,4 @@ -export const VERSION = '3.4.0'; +export const VERSION = '3.4.2'; export const DURATION_MARGIN = 1.20; export const MIN_DURATION_MS = 1_000; export const MAX_EXPECTED_DURATION_MS = 86_400_000; diff --git a/plugins/codex-co-engineer/mcp/v3/credential-boundary.mjs b/plugins/codex-co-engineer/mcp/v3/credential-boundary.mjs index 20b044b..200c8c0 100644 --- a/plugins/codex-co-engineer/mcp/v3/credential-boundary.mjs +++ b/plugins/codex-co-engineer/mcp/v3/credential-boundary.mjs @@ -37,7 +37,7 @@ export const REDACTION_FRAGMENT_BYTES = 32; export const REDACTION_FRAGMENT_STRIDE = 16; export const REDACTED = '[REDACTED]'; export const HANDOFF_ENV_KEY = 'CODEX_CO_ENGINEER_CREDENTIAL_HANDOFF'; -export const DEFAULT_DSH_MODEL = 'muse-spark-1.2-contributor'; +export const DEFAULT_DSH_MODEL = 'meta/muse-spark-1.3-contributor'; export const DSH_OX_MODEL = 'stealth/ox-alpha'; export const OPERATIONAL_ENV_KEYS = capturedFreeze([ @@ -95,9 +95,9 @@ export const PROVIDER_COMMAND_KEYS = capturedFreeze([ export const DSH_ROUTE = capturedFreeze({ [DEFAULT_DSH_MODEL]: capturedFreeze({ - credentialEnv: 'MODEL_API_KEY', - credentialFileEnv: 'CODEX_CO_ENGINEER_MODEL_API_KEY_FILE', - credentialFile: 'model-api-key', + credentialEnv: 'OPENROUTER_API_KEY', + credentialFileEnv: 'CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE', + credentialFile: 'openrouter-api-key', configEnv: 'CODEX_CO_ENGINEER_DSH_ACP_CONFIG', }), [DSH_OX_MODEL]: capturedFreeze({ diff --git a/plugins/codex-co-engineer/mcp/v3/cursor-cloud-result-source.mjs b/plugins/codex-co-engineer/mcp/v3/cursor-cloud-result-source.mjs index f86b4c2..23ddbac 100644 --- a/plugins/codex-co-engineer/mcp/v3/cursor-cloud-result-source.mjs +++ b/plugins/codex-co-engineer/mcp/v3/cursor-cloud-result-source.mjs @@ -170,6 +170,17 @@ export const CURSOR_CLOUD_SDK_RESULT_KEYS = capturedFreeze([ ...CURSOR_CLOUD_SDK_PROVIDER_REPORT_KEYS, ]); +// @cursor/sdk's RunResult carries transport metadata that is intentionally +// absent from the strict result-source receipt. Keep this list separate from +// CURSOR_CLOUD_SDK_RESULT_KEYS: the latter is the closed internal projector +// vocabulary, while this list documents the SDK fields consumed and discarded +// at the adapter boundary. +export const CURSOR_CLOUD_SDK_RESULT_METADATA_KEYS = capturedFreeze([ + 'durationMs', + 'model', + 'usage', +]); + export const CURSOR_CLOUD_RESULT_CORRELATION_KEYS = capturedFreeze([ 'recorded', 'observed', @@ -1578,6 +1589,68 @@ export function projectCursorCloudProviderReportV1(result) { return freezeData(projected); } +/** + * Adapt one installed @cursor/sdk RunResult into the narrow internal shape + * consumed by the result-source projector. The SDK has added benign terminal + * metadata (durationMs, model, and usage) that is useful to its callers but + * is not part of the provider-report/Git-evidence receipt contract. Read + * those known fields through own data descriptors so accessors and proxies + * cannot execute, then discard them. Unknown SDK metadata is likewise ignored + * at this boundary; only the explicitly copied fields reach the strict + * projector below. + */ +export function adaptCursorCloudSdkResultV1(result) { + assertNotProxySurface( + result, + 'result', + 'A Cursor Cloud SDK result must be a plain result object.', + ); + assertPlainObject( + result, + 'malformed_result', + 'result', + 'A Cursor Cloud SDK result must be a plain result object.', + ); + + // Consume the official SDK metadata without allowing it to widen the + // internal receipt vocabulary. Nested metadata is intentionally opaque: it + // is neither trusted nor persisted by this adapter. + for (let index = 0; index < CURSOR_CLOUD_SDK_RESULT_METADATA_KEYS.length; index += 1) { + const key = CURSOR_CLOUD_SDK_RESULT_METADATA_KEYS[index]; + if (!hasOwn(result, key)) continue; + const value = optionalOwnDataValue(result, key, `result.${key}`); + if (value === undefined) continue; + if (value !== null && (typeof value === 'object' || typeof value === 'function')) { + assertNotProxy(value, `result.${key}`); + } + } + + const adapted = {}; + for (let index = 0; index < CURSOR_CLOUD_SDK_RESULT_KEYS.length; index += 1) { + const key = CURSOR_CLOUD_SDK_RESULT_KEYS[index]; + if (!hasOwn(result, key)) continue; + const value = optionalOwnDataValue(result, key, `result.${key}`); + if (value !== undefined) adapted[key] = value; + } + // Freeze only the detached top-level shape. Nested SDK values are projected + // and cloned by projectCursorCloudResultSourcesV1; recursively freezing here + // would freeze caller-owned result, error, or Git graphs. + return capturedFreeze(adapted); +} + +function optionalOwnDataValue(value, key, path) { + const descriptor = ownDescriptor(value, key); + if (descriptor === undefined || !descriptor.enumerable) { + fail('non_enumerable_property_denied', path, + `${path} could not be described as an own enumerable data property.`); + } + if (descriptor.get !== undefined || descriptor.set !== undefined) { + fail('accessor_property_denied', path, + `${path} is an accessor property; resolver data must be direct JSON values and getters are never invoked.`); + } + return descriptor.value; +} + function ownedProviderReportInput(result) { const projected = {}; for (let index = 0; index < CURSOR_CLOUD_SDK_PROVIDER_REPORT_KEYS.length; index += 1) { @@ -1807,6 +1880,7 @@ capturedFreeze(openCursorCloudResultArtifactStoreV1); capturedFreeze(projectCursorCloudResultSourcesV1); capturedFreeze(projectCursorCloudProviderReportV1); capturedFreeze(projectCursorCloudGitEvidenceV1); +capturedFreeze(adaptCursorCloudSdkResultV1); capturedFreeze(assertCursorCloudResultCorrelationV1); capturedFreeze(contentFreeCloudResultSourceFailureV1); capturedFreeze(cursorCloudResultSourceIdentityFromTaskV1); @@ -1827,6 +1901,7 @@ capturedFreeze(CURSOR_CLOUD_GIT_EVIDENCE_INPUT_KEYS); capturedFreeze(CURSOR_CLOUD_RESULT_SOURCE_OBSERVED_KEYS); capturedFreeze(CURSOR_CLOUD_SDK_PROVIDER_REPORT_KEYS); capturedFreeze(CURSOR_CLOUD_SDK_RESULT_KEYS); +capturedFreeze(CURSOR_CLOUD_SDK_RESULT_METADATA_KEYS); capturedFreeze(CURSOR_CLOUD_RESULT_CORRELATION_KEYS); capturedFreeze(CURSOR_CLOUD_TASK_IDENTITY_KEYS); capturedFreeze(CURSOR_CLOUD_RECORDED_IDENTITY_KEYS); diff --git a/plugins/codex-co-engineer/mcp/v3/cursor-cloud-worker.mjs b/plugins/codex-co-engineer/mcp/v3/cursor-cloud-worker.mjs index e302974..ac4921e 100644 --- a/plugins/codex-co-engineer/mcp/v3/cursor-cloud-worker.mjs +++ b/plugins/codex-co-engineer/mcp/v3/cursor-cloud-worker.mjs @@ -19,6 +19,7 @@ import { appendTaskEvent, readPrompt, readRuntimeRecord, readTask, taskPaths, up import { boundedProviderResult, boundedProviderValue } from './provider-result.mjs'; import { assertCursorCloudResultCorrelationV1, + adaptCursorCloudSdkResultV1, contentFreeCloudResultSourceFailureV1, cursorCloudResultSourceIdentityFromTaskV1, isCursorCloudResultIdentityMismatchV1, @@ -34,6 +35,7 @@ const CLEANUP_TIMEOUT_MS = 10_000; const DEADLINE_REFRESH_MS = 1_000; const CANCEL_TIMEOUT_MS = 30_000; const PROVIDER_CALL_TIMEOUT_MS = 5_000; +const GLOBAL_DISCOVERY_CWD = path.parse(process.execPath).root; const COMMIT_SHA = /^[0-9a-f]{40}$/iu; const AMBIGUOUS_SEND_CODES = new Set([ 'network_error', @@ -215,6 +217,14 @@ function redactProviderText(value, sensitiveValues = []) { .replace(COMMON_TOKEN_PATTERNS[2], '[redacted]'); } +function redactProviderResultText(value, credentialValues = [], prompt = '') { + let message = String(value ?? ''); + if (typeof prompt === 'string' && prompt.length > 0 && message.includes(prompt)) { + message = message.split(prompt).join('[REDACTED]'); + } + return redactProviderText(message, credentialValues); +} + function redactedString(value, sensitiveValues = []) { return redactProviderText(value, sensitiveValues).slice(0, 4096); } @@ -281,18 +291,48 @@ export async function loadCursorApiKey(env = process.env) { } } -export async function loadCursorSdk() { - const { stdout } = await runFile('npm', ['root', '--global'], { - encoding: 'utf8', - timeout: PROVIDER_CALL_TIMEOUT_MS, - env: projectProviderEnvironment({ operation: 'sdk_probe', source: process.env }), - }); - const module = path.join(stdout.trim(), '@cursor', 'sdk', 'dist', 'esm', 'index.js'); - try { return await import(pathToFileURL(module).href); } catch (error) { +let cachedCursorSdkLoad = null; + +async function discoverCursorSdk({ + execute = runFile, + env = process.env, + importModule = (specifier) => import(specifier), +} = {}) { + let stdout; + try { + ({ stdout } = await execute('npm', ['root', '--global'], { + cwd: GLOBAL_DISCOVERY_CWD, + encoding: 'utf8', + timeout: PROVIDER_CALL_TIMEOUT_MS, + env: projectProviderEnvironment({ operation: 'sdk_probe', source: env }), + })); + } catch (error) { + throw Object.assign( + new Error('Cursor SDK global installation path could not be discovered.', { cause: error }), + { code: 'cursor_sdk_discovery_failed' }, + ); + } + const globalRoot = String(stdout ?? '').trim(); + if (!path.isAbsolute(globalRoot)) { + fail('cursor_sdk_discovery_failed', 'Cursor SDK global installation path could not be discovered.'); + } + const module = path.join(globalRoot, '@cursor', 'sdk', 'dist', 'esm', 'index.js'); + try { return await importModule(pathToFileURL(module).href); } catch (error) { throw Object.assign(new Error('Install @cursor/sdk@1.0.28 with the Co-Engineer setup command.', { cause: error }), { code: 'cursor_sdk_missing' }); } } +export async function loadCursorSdk(options) { + if (options !== undefined) return discoverCursorSdk(options); + if (cachedCursorSdkLoad === null) { + cachedCursorSdkLoad = discoverCursorSdk().catch((error) => { + cachedCursorSdkLoad = null; + throw error; + }); + } + return cachedCursorSdkLoad; +} + async function gitValue(cwd, args) { const { stdout } = await runFile('git', ['-C', cwd, ...args], { cwd, @@ -837,21 +877,26 @@ function assertTerminalResultSources(task, run, agentId, result) { return sources; } -async function persistTerminalRun({ root, taskId, client, key, prompt, agentId, run, result }) { - if (result?.id !== undefined && result.id !== run.id) { +async function persistTerminalRun({ root, taskId, client, key, prompt, agentId, run, adaptedResult }) { + if (adaptedResult.id !== undefined && adaptedResult.id !== run.id) { fail('cursor_run_identity_mismatch', 'Cursor Cloud returned a different run identity at completion.'); } const { task: current } = await readTask(root, taskId); - const sources = assertTerminalResultSources(current, run, agentId, result); - const status = result.status === 'finished' ? 'completed' : result.status === 'cancelled' ? 'cancelled' : 'failed'; + const sources = assertTerminalResultSources(current, run, agentId, adaptedResult); + const status = adaptedResult.status === 'finished' ? 'completed' : adaptedResult.status === 'cancelled' ? 'cancelled' : 'failed'; const providerSecrets = [key, prompt]; - const sanitizedBranches = sanitizeProviderValue(result.git?.branches ?? [], providerSecrets); + const providerCredentials = [key]; + const sanitizedBranches = sanitizeProviderValue(adaptedResult.git?.branches ?? [], providerSecrets); const branches = Array.isArray(sanitizedBranches) ? sanitizedBranches : []; - const bounded = typeof result.result === 'string' - ? boundedProviderResult(result.result, { sanitize: (text) => redactProviderText(text, providerSecrets) }) - : boundedProviderValue(result.result ?? null, { sanitize: (text) => redactProviderText(text, providerSecrets) }); + const bounded = typeof adaptedResult.result === 'string' + ? boundedProviderResult(adaptedResult.result, { + sanitize: (text) => redactProviderResultText(text, providerCredentials, prompt), + }) + : boundedProviderValue(adaptedResult.result ?? null, { + sanitize: (text) => redactProviderResultText(text, providerCredentials, prompt), + }); const providerResult = bounded.value; - const providerError = sanitizeProviderValue(result.error ?? null, providerSecrets); + const providerError = sanitizeProviderValue(adaptedResult.error ?? null, providerSecrets); let archived = false; try { archived = await archiveAgent(client, agentId, key); @@ -863,7 +908,7 @@ async function persistTerminalRun({ root, taskId, client, key, prompt, agentId, } const terminal = await updateTask(root, taskId, { status, - provider_run_id: result.id, + provider_run_id: adaptedResult.id, result: providerResult, ...Object.fromEntries(Object.entries(bounded).filter(([key]) => key.startsWith('result_'))), provider_error: providerError, @@ -872,7 +917,7 @@ async function persistTerminalRun({ root, taskId, client, key, prompt, agentId, provider_agent_archived: archived, finished_at: new Date().toISOString(), }); - await appendTaskEvent(root, taskId, { type: 'terminal', status, run_id: result.id }); + await appendTaskEvent(root, taskId, { type: 'terminal', status, run_id: adaptedResult.id }); return attachCursorCloudResultSource(root, terminal, sources); } @@ -1038,6 +1083,7 @@ export async function runCursorCloudTask({ }); if (running.status !== 'running') fail('transport_lost', 'Cursor Cloud run could not be registered in the task receipt.'); await appendTaskEvent(root, taskId, { type: 'transport', state: 'prompt_dispatched', transport: 'cursor-sdk', agent_id: agentId, run_id: run.id }); + await updateTask(root, taskId, { dispatch_evidence: 'authoritative' }); const stopRemote = () => { if (!stopPromise) stopPromise = stopRemoteRun(client, { ...task, provider_agent_id: agentId }, key, run, waitPromise); return stopPromise; @@ -1059,7 +1105,10 @@ export async function runCursorCloudTask({ watch, refreshMs: deadlineRefreshMs, }); - return persistTerminalRun({ root, taskId, client, key, prompt, agentId, run, result }); + return persistTerminalRun({ + root, taskId, client, key, prompt, agentId, run, + adaptedResult: adaptCursorCloudSdkResultV1(result), + }); } catch (error) { timedOut ||= error?.code === 'timeout'; let current = (await readTask(root, taskId)).task; @@ -1338,7 +1387,8 @@ export async function reconcileCursorCloudTask({ root, taskId, sdk, apiKey, load error: publicError(Object.assign(new Error('Cursor Cloud run completion could not be reconciled.', { cause: error }), { code: 'cursor_reconcile_failed' }), [key, prompt]), }); } - if (result?.id !== undefined && result.id !== run.id) { + const adaptedResult = adaptCursorCloudSdkResultV1(result); + if (adaptedResult.id !== undefined && adaptedResult.id !== run.id) { return updateTask(root, taskId, { status: 'transport_lost', error: { code: 'cursor_run_identity_mismatch', message: 'Cursor Cloud returned a different run identity at reconciliation completion.' }, @@ -1346,7 +1396,7 @@ export async function reconcileCursorCloudTask({ root, taskId, sdk, apiKey, load } return persistTerminalRun({ root, taskId, client, key, prompt, - agentId: task.provider_agent_id, run, result, + agentId: task.provider_agent_id, run, adaptedResult, }); } finally { try { recoveryAgent?.close?.(); } catch {} diff --git a/plugins/codex-co-engineer/mcp/v3/diagnostics.mjs b/plugins/codex-co-engineer/mcp/v3/diagnostics.mjs index 2f3e1b4..e2c5fa0 100644 --- a/plugins/codex-co-engineer/mcp/v3/diagnostics.mjs +++ b/plugins/codex-co-engineer/mcp/v3/diagnostics.mjs @@ -202,7 +202,7 @@ export function diagnosticEnvelope(task, runtime = null, extras = {}) { return Object.freeze({ task_id: task.id, provider: task.provider ?? null, - ...(task.provider === 'dsh' ? { dsh_model: task.dsh_model ?? 'muse-spark-1.2-contributor' } : {}), + ...(task.provider === 'dsh' ? { dsh_model: task.dsh_model ?? 'meta/muse-spark-1.3-contributor' } : {}), session_id: task.acp_session_id ?? task.attention?.session_id ?? null, provider_run_id: task.provider_run_id ?? null, question_id: task.attention?.question_id ?? extras.question_id ?? null, diff --git a/plugins/codex-co-engineer/mcp/v3/dsh-acpx-driver.mjs b/plugins/codex-co-engineer/mcp/v3/dsh-acpx-driver.mjs index d92be67..0a7b3b1 100644 --- a/plugins/codex-co-engineer/mcp/v3/dsh-acpx-driver.mjs +++ b/plugins/codex-co-engineer/mcp/v3/dsh-acpx-driver.mjs @@ -5,7 +5,7 @@ // module owns ONLY the DSH-specific wiring; every request/result shape, // capability posture, transition rule, and denial code is inherited from the // accepted contract (no parallel envelope or capability schema): -// - hard binding: provider `dsh` plus exactly `muse-spark-1.2-contributor` +// - hard binding: provider `dsh` plus exactly `meta/muse-spark-1.3-contributor` // or `stealth/ox-alpha`; every other provider/model pairing fails closed; // - the four lifecycle operations (preflight, launch, reconcile, cancel) // are driven through an INJECTED BOUNDED ACPX ONE-SHOT TRANSPORT PORT. @@ -90,7 +90,7 @@ export const DSH_ACPX_DRIVER_VERSION = 1; export const DSH_PROVIDER = 'dsh'; export const DSH_ALLOWED_MODELS = capturedFreeze([ - 'muse-spark-1.2-contributor', + 'meta/muse-spark-1.3-contributor', 'stealth/ox-alpha', ]); @@ -98,11 +98,11 @@ export const DSH_ALLOWED_MODELS = capturedFreeze([ // The injected port resolves real paths and credentials; this map never // touches the filesystem and confers no authority by itself. export const DSH_MODEL_IDENTITIES = capturedFreeze({ - 'muse-spark-1.2-contributor': capturedFreeze({ + 'meta/muse-spark-1.3-contributor': capturedFreeze({ config_file: 'dsh-acp.yml', - credential_env: 'MODEL_API_KEY', - credential_file_env: 'CODEX_CO_ENGINEER_MODEL_API_KEY_FILE', - credential_file: 'model-api-key', + credential_env: 'OPENROUTER_API_KEY', + credential_file_env: 'CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE', + credential_file: 'openrouter-api-key', }), 'stealth/ox-alpha': capturedFreeze({ config_file: 'dsh-acp-ox-alpha.yml', @@ -250,7 +250,7 @@ function buildDshAcpDeclarationV1() { dispatch_certainty: 'uncertain_after_spawn', exact_model_selection: 'exact_and_attested', merge_authority: 'none_codex_only_integration', - notes: 'DSH ACPX one-shot flow for Muse Spark 1.2 Contributor or Ox Alpha. ' + notes: 'DSH ACPX one-shot flow for Muse Spark 1.3 Contributor or Ox Alpha. ' + 'ACPX gives no authoritative prompt-sent acknowledgement, so launches stay ' + 'uncertain after spawn and are never replayed. Same-session reply is ' + 'unsupported; attention surfaces unresolved instead of starting a ' diff --git a/plugins/codex-co-engineer/mcp/v3/final-decision-card.mjs b/plugins/codex-co-engineer/mcp/v3/final-decision-card.mjs index 151db87..04d98c7 100644 --- a/plugins/codex-co-engineer/mcp/v3/final-decision-card.mjs +++ b/plugins/codex-co-engineer/mcp/v3/final-decision-card.mjs @@ -20,8 +20,10 @@ // infer a run-level external artifact class. // // The card cannot merge, push, rebase, create a PR, tag, or release. -// Sol High/XHigh alone may regular-merge after exact-head, exact-tree, -// current-green-CI, and topology CAS checks. Progressive disclosure keeps +// Codex remains the merge authority and may regular-merge only after +// exact-head, exact-tree, current-green-CI, and topology CAS checks plus +// the user's authorization. The card cannot grant merge or release +// authority. Progressive disclosure keeps // a compact model-facing summary plus artifact references for expensive // diff/log/evidence, with deterministic truncation provenance. // diff --git a/plugins/codex-co-engineer/mcp/v3/mailbox.mjs b/plugins/codex-co-engineer/mcp/v3/mailbox.mjs index cf20e00..708241a 100644 --- a/plugins/codex-co-engineer/mcp/v3/mailbox.mjs +++ b/plugins/codex-co-engineer/mcp/v3/mailbox.mjs @@ -4,10 +4,19 @@ import path from 'node:path'; import { providerCapabilities } from './contract.mjs'; import { appendTaskEvent, readTask, taskPaths, updateTask, waitDelay } from './task-store.mjs'; +import { assertDirectJsonClosure, assertNotProxy, assertPlainObject } from './selection-json.mjs'; const QUESTION_ID = /^[A-Za-z0-9][A-Za-z0-9._-]{0,79}$/u; const SESSION_ID = /^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/u; const RESPONSE_MAX_BYTES = 16 * 1024; +const SAFE_ATTENTION_CAPABILITIES = Object.freeze([ + 'read_run_receipts', 'read_provider_logs', 'read_own_worktree', +]); +const CAPABILITY_RESOURCES = Object.freeze({ + read_run_receipts: 'run_receipt', + read_provider_logs: 'provider_log', + read_own_worktree: 'own_worktree', +}); function fail(code, message) { throw Object.assign(new Error(message), { code }); @@ -27,6 +36,35 @@ function requireSessionId(value) { return value; } +function normalizeOptions(value) { + if (value === undefined || value === null) return null; + if (!Array.isArray(value) || value.length > 8) { + fail('invalid_attention', 'The attention options must be a bounded list.'); + } + return value.map((option, index) => { + if (typeof option === 'string') { + if (option.length > 128) fail('invalid_attention', `The attention option ${index} is too long.`); + return option; + } + try { + assertNotProxy(option, `attention.options[${index}]`); + assertPlainObject(option, 'invalid_attention', `attention.options[${index}]`, 'Attention option'); + assertDirectJsonClosure(option, `attention.options[${index}]`); + } catch { + fail('invalid_attention', 'The attention options must contain direct JSON values.'); + } + const allowed = new Set(['optionId', 'kind', 'name', 'label', 'description']); + for (const key of Object.keys(option)) { + if (!allowed.has(key)) fail('invalid_attention', 'The attention option has an unknown field.'); + if (typeof option[key] !== 'string' || option[key].length === 0 || option[key].length > 128) { + fail('invalid_attention', 'The attention option text is invalid.'); + } + } + if (typeof option.kind !== 'string') fail('invalid_attention', 'The attention option kind is required.'); + return { ...option }; + }); +} + function replyPaths(root, taskId, questionId) { const paths = taskPaths(root, taskId); const directory = path.join(paths.directory, 'replies'); @@ -68,13 +106,32 @@ export async function readAttention(root, taskId) { export async function recordNeedsAttention(root, taskId, attention) { const sessionId = requireSessionId(attention?.session_id); const questionId = requireQuestionId(attention?.question_id); + const capability = attention?.capability; + const resource = attention?.resource; + const action = attention?.action; + if (capability !== undefined + && (!SAFE_ATTENTION_CAPABILITIES.includes(capability) + || resource !== CAPABILITY_RESOURCES[capability] + || action !== 'read')) { + fail('invalid_attention', 'The attention capability is outside the safe typed vocabulary.'); + } + if (resource !== undefined && (typeof resource !== 'string' || resource.length === 0 || resource.length > 128)) { + fail('invalid_attention', 'The attention resource is invalid.'); + } + if (action !== undefined && (typeof action !== 'string' || action.length === 0 || action.length > 64)) { + fail('invalid_attention', 'The attention action is invalid.'); + } + const options = normalizeOptions(attention?.options); const paths = replyPaths(root, taskId, questionId); const record = { session_id: sessionId, question_id: questionId, prompt: typeof attention.prompt === 'string' ? attention.prompt.slice(0, 4_096) : null, - options: Array.isArray(attention.options) ? attention.options.slice(0, 8) : null, + options, stage: typeof attention.stage === 'string' ? attention.stage : 'provider_feedback', + ...(capability !== undefined ? { capability } : {}), + ...(resource !== undefined ? { resource } : {}), + ...(action !== undefined ? { action } : {}), at: new Date().toISOString(), }; const current = await readAttention(root, taskId); @@ -88,6 +145,9 @@ export async function recordNeedsAttention(root, taskId, attention) { session_id: sessionId, question_id: questionId, stage: record.stage, + ...(capability !== undefined ? { capability } : {}), + ...(resource !== undefined ? { resource } : {}), + ...(action !== undefined ? { action } : {}), }, }); await appendTaskEvent(root, taskId, { diff --git a/plugins/codex-co-engineer/mcp/v3/profile.mjs b/plugins/codex-co-engineer/mcp/v3/profile.mjs index e49c9d1..802fabf 100644 --- a/plugins/codex-co-engineer/mcp/v3/profile.mjs +++ b/plugins/codex-co-engineer/mcp/v3/profile.mjs @@ -134,7 +134,7 @@ const MODEL_SCAN_EXEMPT_PATHS = new SetCtor(['profile.model']); * resolver concerns. * @deprecated Informational compatibility data only; not an authorization list. */ -export const PROFILE_DSH_MODELS = objectFreeze(['muse-spark-1.2-contributor', 'stealth/ox-alpha']); +export const PROFILE_DSH_MODELS = objectFreeze(['meta/muse-spark-1.3-contributor', 'stealth/ox-alpha']); export const MIN_PROFILE_EXPECTED_DURATION_MS = MIN_DURATION_MS; export const MAX_PROFILE_EXPECTED_DURATION_MS = MAX_EXPECTED_DURATION_MS; diff --git a/plugins/codex-co-engineer/mcp/v3/prompt-compiler.mjs b/plugins/codex-co-engineer/mcp/v3/prompt-compiler.mjs index 7886261..4d12529 100644 --- a/plugins/codex-co-engineer/mcp/v3/prompt-compiler.mjs +++ b/plugins/codex-co-engineer/mcp/v3/prompt-compiler.mjs @@ -73,6 +73,10 @@ import { parseRunManifestV1 } from './run-policy.mjs'; export const CHILD_ENVELOPE_SCHEMA_ID = 'codex-co-engineer.child-envelope.v1'; export const CHILD_ENVELOPE_VERSION = 1; +const PROVIDER_WORKSPACE = 'work only in the current working directory (assigned worktree); repository_path is source identity, not a navigation target'; +const PROVIDER_GUIDANCE = 'task prompt and local repository instructions are sufficient unless the assignment explicitly requests an external skill; honor an exact requested output and format exactly; required_evidence labels are controller metadata, not worker response sections; omit routine progress narration and repeated identity or report blocks unless the task prompt requests them, while surfacing blockers and necessary questions; do not seek receipt artifacts because the controller owns lifecycle and machine receipts'; +const PREVIOUS_PROVIDER_GUIDANCE = 'task prompt and local repository instructions are sufficient unless the assignment explicitly requests an external skill; return requested results and evidence without seeking receipt artifacts because the controller owns lifecycle and machine receipts'; + // Worst-case rendering bounds. They mirror the P02 validators so the cap can // never reject a manifest those validators accept. JSON string escaping can // expand one parameter value to at most 3x its UTF-8 byte length (one astral @@ -244,6 +248,8 @@ function renderEnvelope(fields) { pushLine(`lane_index: ${fields.lane_index}`); pushLine(`assignment_count: ${fields.assignment_count}`); pushLine(`repository_path: ${fields.repository.path}`); + pushLine(`provider_workspace: ${PROVIDER_WORKSPACE}`); + pushLine(`provider_guidance: ${PROVIDER_GUIDANCE}`); pushLine(`base_sha: ${fields.repository.base_sha}`); pushBlock('objective', fields.objective); pushLine(`assignment_id: ${fields.assignment_id}`); @@ -529,6 +535,24 @@ export function parseChildEnvelopeV1(envelopeText) { } const repositoryPath = expectLine(reader, 'repository_path'); assertRepositoryPath(repositoryPath, 'envelope.repository_path'); + // Older v1 envelopes either carried the previous exact guidance or no + // provider execution guidance. Continue accepting those durable bytes while + // every newly compiled envelope makes response precedence explicit. + const legacyOffset = reader.offset; + const next = splitScaffoldLine(readScaffoldLine(reader)); + if (next.key === 'provider_workspace') { + if (next.value !== PROVIDER_WORKSPACE) { + fail('invalid_format', 'envelope.provider_workspace', + 'provider_workspace must preserve the compiler execution guidance.'); + } + const guidance = expectLine(reader, 'provider_guidance'); + if (guidance !== PROVIDER_GUIDANCE && guidance !== PREVIOUS_PROVIDER_GUIDANCE) { + fail('invalid_format', 'envelope.provider_guidance', + 'provider_guidance must preserve the compiler execution guidance.'); + } + } else { + reader.offset = legacyOffset; + } const baseSha = expectLine(reader, 'base_sha'); assertBaseSha(baseSha, 'envelope.base_sha'); const objectiveBlock = expectBlock(reader, 'objective', 'envelope.objective'); diff --git a/plugins/codex-co-engineer/mcp/v3/readiness-snapshot.mjs b/plugins/codex-co-engineer/mcp/v3/readiness-snapshot.mjs new file mode 100644 index 0000000..675e144 --- /dev/null +++ b/plugins/codex-co-engineer/mcp/v3/readiness-snapshot.mjs @@ -0,0 +1,148 @@ +// Owner-only, content-free readiness snapshot for the 3.4.1 status path. +// This is a cache hint, never dispatch authority: an admission still performs +// its own provider, boundary, and repository checks before sending prompts. + +import { constants as fsConstants } from 'node:fs'; +import { randomUUID } from 'node:crypto'; +import { chmod, mkdir, open, rename, lstat } from 'node:fs/promises'; +import path from 'node:path'; + +import { assertDirectJsonClosure } from './selection-json.mjs'; + +export const READINESS_SNAPSHOT_SCHEMA = 'codex-co-engineer.readiness-snapshot.v1'; +export const READINESS_SNAPSHOT_FILE = 'readiness-3.4.1.json'; +export const READINESS_SNAPSHOT_MAX_BYTES = 64 * 1024; + +const ROOT_FLAGS = fsConstants.O_RDONLY + | (fsConstants.O_DIRECTORY ?? 0) + | (fsConstants.O_NOFOLLOW ?? 0) + | (fsConstants.O_NONBLOCK ?? 0); + +function unsafe(code, message) { + throw Object.assign(new Error(message), { code }); +} + +function normalizeRoot(root) { + if (typeof root !== 'string' || !path.isAbsolute(root) || path.resolve(root) !== root || root.includes('\0')) { + unsafe('readiness_snapshot_path_unsafe', 'The readiness snapshot root must be a normalized absolute path.'); + } + return root; +} + +function assertPrivateDirectory(metadata) { + if (metadata.isSymbolicLink?.() || !metadata.isDirectory?.() + || (metadata.mode & 0o077) !== 0) { + unsafe('readiness_snapshot_root_unsafe', 'The readiness snapshot root is not a private directory.'); + } + if (typeof process.geteuid === 'function' && Number(metadata.uid) !== process.geteuid()) { + unsafe('readiness_snapshot_root_unsafe', 'The readiness snapshot root has the wrong owner.'); + } +} + +async function ensureRoot(root) { + await mkdir(root, { recursive: true, mode: 0o700 }); + await chmod(root, 0o700); + const metadata = await lstat(root); + assertPrivateDirectory(metadata); +} + +function snapshotPath(root) { + return path.join(normalizeRoot(root), READINESS_SNAPSHOT_FILE); +} + +function canonicalSnapshot(value) { + if (!value || typeof value !== 'object' || Array.isArray(value) + || value.schema !== READINESS_SNAPSHOT_SCHEMA + || value.version !== 1 + || typeof value.observed_at !== 'string' + || !Number.isFinite(Date.parse(value.observed_at)) + || !value.readiness || typeof value.readiness !== 'object' + || Array.isArray(value.readiness)) { + unsafe('readiness_snapshot_invalid', 'The readiness snapshot is invalid.'); + } + try { + assertDirectJsonClosure(value, 'readiness_snapshot'); + } catch { + unsafe('readiness_snapshot_invalid', 'The readiness snapshot is not direct JSON.'); + } + const serialized = JSON.stringify(value); + if (Buffer.byteLength(serialized, 'utf8') > READINESS_SNAPSHOT_MAX_BYTES) { + unsafe('readiness_snapshot_too_large', 'The readiness snapshot exceeds its size bound.'); + } + return `${serialized}\n`; +} + +export async function loadReadinessSnapshot(root) { + const target = snapshotPath(root); + try { + await ensureRoot(root); + const rootHandle = await open(root, ROOT_FLAGS); + try { + const metadata = await lstat(target); + if (metadata.isSymbolicLink?.() || !metadata.isFile?.() + || Number(metadata.nlink) !== 1 || (metadata.mode & 0o077) !== 0 + || (typeof process.geteuid === 'function' && Number(metadata.uid) !== process.geteuid())) { + unsafe('readiness_snapshot_unsafe', 'The readiness snapshot is not a private regular file.'); + } + if (Number(metadata.size) > READINESS_SNAPSHOT_MAX_BYTES) { + unsafe('readiness_snapshot_too_large', 'The readiness snapshot exceeds its size bound.'); + } + const handle = await open(target, fsConstants.O_RDONLY + | (fsConstants.O_NOFOLLOW ?? 0) | (fsConstants.O_NONBLOCK ?? 0)); + try { + const bytes = await handle.readFile(); + const after = await handle.stat(); + if (Number(after.ino) !== Number(metadata.ino) || Number(after.size) !== Number(metadata.size)) { + unsafe('readiness_snapshot_changed', 'The readiness snapshot changed while it was read.'); + } + const parsed = JSON.parse(bytes.toString('utf8')); + canonicalSnapshot(parsed); + return { + observed_at: parsed.observed_at, + readiness: parsed.readiness, + probe_duration_ms: parsed.probe_duration_ms ?? null, + }; + } finally { + await handle.close().catch(() => {}); + } + } finally { + await rootHandle.close().catch(() => {}); + } + } catch (error) { + if (error?.code === 'ENOENT') return null; + if (error?.code?.startsWith?.('readiness_snapshot_')) return null; + return null; + } +} + +export async function saveReadinessSnapshot(root, readiness, { observed_at, probe_duration_ms } = {}) { + const normalizedRoot = normalizeRoot(root); + await ensureRoot(normalizedRoot); + const value = { + schema: READINESS_SNAPSHOT_SCHEMA, + version: 1, + observed_at: typeof observed_at === 'string' ? observed_at : new Date().toISOString(), + readiness, + ...(Number.isSafeInteger(probe_duration_ms) && probe_duration_ms >= 0 + ? { probe_duration_ms } : {}), + }; + const text = canonicalSnapshot(value); + const rootHandle = await open(normalizedRoot, ROOT_FLAGS); + const temporary = path.join(normalizedRoot, `.tmp-readiness-${randomUUID()}.json`); + const target = snapshotPath(normalizedRoot); + try { + const handle = await open(temporary, fsConstants.O_WRONLY + | fsConstants.O_CREAT | fsConstants.O_EXCL | (fsConstants.O_NOFOLLOW ?? 0), 0o600); + try { + await handle.writeFile(text, 'utf8'); + await handle.chmod(0o600); + await handle.sync(); + } finally { + await handle.close().catch(() => {}); + } + await rename(temporary, target); + await rootHandle.sync().catch(() => {}); + } finally { + await rootHandle.close().catch(() => {}); + } +} diff --git a/plugins/codex-co-engineer/mcp/v3/response.mjs b/plugins/codex-co-engineer/mcp/v3/response.mjs index e4739e6..1b01c8c 100644 --- a/plugins/codex-co-engineer/mcp/v3/response.mjs +++ b/plugins/codex-co-engineer/mcp/v3/response.mjs @@ -5,6 +5,7 @@ export const TEXT_FALLBACK_SCHEMA = 'co_engineer.mcp_text_fallback.v1'; export const TEXT_FALLBACK_MAX_BYTES = 2_048; export const TEXT_FALLBACK_TASK_PREVIEW = 5; export const RESPONSE_MODE_STRUCTURED = 'structured'; +export const RESPONSE_MODE_LEGACY = 'legacy'; /** UX-04 experience projection: three semantic cards over sanitized run receipts. */ export const EXPERIENCE_SCHEMA = 'codex-co-engineer.experience-projection.v1'; @@ -16,14 +17,18 @@ export const EXPERIENCE_QUESTION_BYTES = 320; export const EXPERIENCE_SCOPE_PATTERN_BYTES = 96; export const EXPERIENCE_MAX_LANES = 8; export const EXPERIENCE_MAX_QUESTIONS = 8; +export const EXPERIENCE_RESULT_META_KEY = 'codex-co-engineer/experience'; export const PUBLIC_MCP_TOOLS = Object.freeze([ 'status', 'delegate', 'task', 'tasks', 'cancel', ]); export const EXPERIENCE_PHRASES = Object.freeze({ delegating: 'I am delegating this to Co-Engineer', + preparing_one: 'Co-Engineer is preparing 1 assignment', + preparing_template: 'Co-Engineer is preparing N assignments', running_one: 'Co-Engineer is running 1 independent assignment', running_template: 'Co-Engineer is running N independent assignments', + reconciling: 'Co-Engineer is reconciling an uncertain assignment', attention: 'Co-Engineer needs one decision from you', verified_final: 'Co-Engineer finished, and I verified the candidate.', }); @@ -58,6 +63,10 @@ export const EXPERIENCE_COORDINATION = Object.freeze({ grouped_reply: 1, }); +// The values above describe the bounded-run contract. A projection also +// carries observed counts so an in-progress or unsuccessful receipt cannot +// look like it already reached the contract's verified-final outcome. + export const EXPERIENCE_DENIED_CONTROLS = Object.freeze({ merge: false, push: false, @@ -80,10 +89,32 @@ const TERMINAL_LANE_STATUSES = Object.freeze([ ]); const ACCEPTED_LANE_STATUSES = Object.freeze(['completed']); const FAILED_LANE_STATUSES = Object.freeze([ - 'failed', 'timeout', 'transport_lost', 'environment_blocked', + 'failed', 'failed_pre_prompt', 'timeout', 'transport_lost', 'environment_blocked', +]); +const UNRESOLVED_LANE_STATUSES = Object.freeze([ + 'unresolved', 'partial_handoff', 'unrecoverable_post_prompt', 'lifecycle_pending', +]); +const RECONCILIATION_LANE_STATUSES = Object.freeze([ + 'partial_handoff', 'unrecoverable_post_prompt', 'lifecycle_pending', +]); +const ACTIVE_LANE_STATUSES = Object.freeze([ + 'accepted', 'starting', 'running', 'cancelling', 'dispatched', + 'prompt_dispatched', 'needs_attention', 'session_ready', +]); +const RECONCILIATION_PHASES = Object.freeze([ + 'degraded', 'unresolved', 'partial_handoff', 'unrecoverable_post_prompt', + 'lifecycle_pending', ]); -const UNRESOLVED_LANE_STATUSES = Object.freeze(['unresolved']); const ATTENTION_OPEN_STATUSES = Object.freeze(['open', 'needs_attention']); +const RUN_LIFECYCLE_PHASES = Object.freeze([ + 'validating', 'awaiting_consent', 'preparing_workspaces', 'dispatching', + 'running', 'needs_attention', 'degraded', 'unresolved', 'verifying', 'completed', + 'failed', 'cancelled', 'partial_handoff', 'unrecoverable_post_prompt', + 'lifecycle_pending', +]); +const RUN_NONTERMINAL_PHASES = Object.freeze([ + 'validating', 'preparing_workspaces', 'dispatching', 'running', 'verifying', +]); const KNOWN_EVIDENCE_KINDS = Object.freeze([ 'acceptance_results', 'artifact_integrity', 'command_reported', 'files_changed', 'git_diff', 'git_identity', 'head_reached', 'head_sha', 'model_attested', @@ -357,7 +388,9 @@ export function sanitizeToolPayload(value) { } export function normalizeResponseMode(responseMode) { - return responseMode === RESPONSE_MODE_STRUCTURED ? RESPONSE_MODE_STRUCTURED : null; + if (responseMode === RESPONSE_MODE_STRUCTURED) return RESPONSE_MODE_STRUCTURED; + if (responseMode === RESPONSE_MODE_LEGACY) return RESPONSE_MODE_LEGACY; + return null; } /** @@ -497,7 +530,9 @@ function laneId(lane) { } function laneStatus(lane) { - return typeof lane?.status === 'string' ? lane.status : null; + return typeof lane?.status === 'string' + ? lane.status + : (typeof lane?.phase === 'string' ? lane.phase : null); } function laneProvider(lane) { @@ -525,6 +560,23 @@ export function runningPhrase(assignmentCount) { return null; } +export function preparingPhrase(assignmentCount) { + if (assignmentCount === 1) return EXPERIENCE_PHRASES.preparing_one; + if (Number.isInteger(assignmentCount) && assignmentCount >= 2 && assignmentCount <= EXPERIENCE_MAX_LANES) { + return `Co-Engineer is preparing ${assignmentCount} assignments`; + } + return null; +} + +function simpleRunHasAuthoritativeRequiredDispatch(receipt, lanes) { + if (receipt?.schema !== 'codex-co-engineer.run-admission.v1') return true; + if (receipt?.authoritative_required_dispatch !== true) return false; + const required = lanes.filter((lane) => lane.required !== false); + return required.length > 0 && required.every((lane) => ( + lane.prompt_dispatched === true && lane.dispatch_confidence === 'authoritative' + )); +} + function attentionItems(receipt) { const direct = receipt?.attention; const fromRecord = Array.isArray(direct?.items) ? direct.items : []; @@ -548,27 +600,85 @@ function isOpenAttention(receipt, lanes, items) { } function isTerminalLane(lane) { + if (typeof lane?.task_final === 'boolean') return lane.task_final; return TERMINAL_LANE_STATUSES.includes(laneStatus(lane)); } +function laneNeedsReconciliation(lane) { + if (lane?.task_final === true) return false; + const status = laneStatus(lane); + if (RECONCILIATION_LANE_STATUSES.includes(status)) return true; + return status === 'unresolved' && lane?.prompt_dispatched === true; +} + +function laneBlocksTerminal(lane) { + if (lane?.task_final === false) return true; + if (lane?.task_final === true) return false; + const status = laneStatus(lane); + if (laneNeedsReconciliation(lane)) return true; + if (ACTIVE_LANE_STATUSES.includes(status)) return true; + return lane?.prompt_dispatched === true && !isTerminalLane(lane); +} + +function receiptNeedsReconciliation(receipt, lanes) { + const phase = typeof receipt?.phase === 'string' + ? receipt.phase + : (typeof receipt?.status === 'string' ? receipt.status : null); + return RECONCILIATION_PHASES.includes(phase) || lanes.some(laneNeedsReconciliation); +} + function isFinalRun(receipt, lanes) { + if (lanes.some(laneBlocksTerminal)) return false; if (receipt?.journal?.terminal === true) return true; if (lanes.length === 0) return false; if (lanes.every(isTerminalLane)) return true; - if (receipt?.complete_candidate_blocked === true && !lanes.some((lane) => { - const status = laneStatus(lane); - return status === 'running' || status === 'needs_attention' || status === 'dispatched' - || status === 'starting' || status === 'accepted'; - })) { + if (receipt?.complete_candidate_blocked === true && !lanes.some(laneBlocksTerminal)) { return true; } return false; } +function explicitRunLifecycleCard(receipt, lanes, items) { + const phase = typeof receipt?.phase === 'string' + ? receipt.phase + : (typeof receipt?.status === 'string' ? receipt.status : null); + if (!RUN_LIFECYCLE_PHASES.includes(phase)) return null; + if (phase === 'awaiting_consent' || phase === 'needs_attention') return 'attention'; + if (phase === 'completed' || phase === 'failed' || phase === 'cancelled') { + return lanes.some(laneBlocksTerminal) ? 'run' : 'final'; + } + if (phase === 'degraded' || phase === 'unresolved') { + if (isOpenAttention(receipt, lanes, items)) return 'attention'; + if (lanes.length === 0 || lanes.some(laneBlocksTerminal)) return 'run'; + return 'final'; + } + if (RECONCILIATION_PHASES.includes(phase) || RUN_NONTERMINAL_PHASES.includes(phase)) return 'run'; + return null; +} + +function consentObject(receipt) { + const consent = receipt?.consent; + if (!consent || typeof consent !== 'object' || Array.isArray(consent)) return null; + const request = consent.request && typeof consent.request === 'object' && !Array.isArray(consent.request) + ? consent.request + : null; + return { consent, request }; +} + +function consentNeedsDecision(receipt) { + const entry = consentObject(receipt); + const status = typeof entry?.consent?.status === 'string' ? entry.consent.status : null; + if (receipt?.phase === 'awaiting_consent') return true; + return (status === 'pending' || status === 'required') && entry?.request !== null; +} + export function classifyExperienceCard(receipt) { if (!receipt || typeof receipt !== 'object' || Array.isArray(receipt)) return 'run'; const lanes = asLanes(receipt); const items = attentionItems(receipt); + const explicit = explicitRunLifecycleCard(receipt, lanes, items); + if (explicit) return explicit; + if (consentNeedsDecision(receipt)) return 'attention'; if (isOpenAttention(receipt, lanes, items) && receipt?.attention?.status !== 'resolved' && receipt?.attention?.status !== 'reply_committed') { return 'attention'; @@ -794,6 +904,58 @@ function projectQuestions(items, lanes) { return questions; } +const CONSENT_STATUSES = Object.freeze([ + 'pending', 'required', 'approved', 'blocked', 'declined', 'cancelled', 'timed_out', +]); + +function publicProviderNames(providers) { + if (!Array.isArray(providers)) return []; + return providers + .filter((provider) => Object.hasOwn(PROVIDER_DISPLAY, provider)) + .map((provider) => providerPhrase(provider)) + .filter(Boolean); +} + +function consentProviders(providers) { + if (!Array.isArray(providers)) return []; + return providers.filter((provider) => Object.hasOwn(PROVIDER_DISPLAY, provider)); +} + +function projectConsent(receipt) { + const entry = consentObject(receipt); + if (!entry && receipt?.phase !== 'awaiting_consent') return null; + const request = entry?.request ?? {}; + const rawStatus = typeof entry?.consent?.status === 'string' ? entry.consent.status : null; + const status = CONSENT_STATUSES.includes(rawStatus) + ? rawStatus + : (receipt?.phase === 'awaiting_consent' ? 'required' : null); + const providers = consentProviders(request.providers); + const repositoryIdentity = digestValue(request.repository_identity); + const requestKind = request.kind === 'repository_exposure_consent' + ? request.kind + : null; + const pending = consentNeedsDecision(receipt); + if (!pending && status !== 'blocked' && status !== 'declined' && status !== 'cancelled') return null; + return { + kind: requestKind, + status, + decision_authority: 'host', + message: pending + ? 'This run needs your approval to share the full repository with the selected co-engineers for this run.' + : 'The host did not approve repository exposure for this run.', + request: { + kind: requestKind, + run_id: typeof request.run_id === 'string' ? request.run_id : null, + repository_identity: repositoryIdentity, + providers, + provider_phrases: publicProviderNames(providers), + scope: request.scope === 'full_repository' ? request.scope : null, + duration: request.duration === 'this_run_only' ? request.duration : null, + remote_mutation: request.remote_mutation === false ? false : null, + }, + }; +} + function projectEvidence(raw) { if (!raw || typeof raw !== 'object' || Array.isArray(raw)) { return { present: false, digest: null, fact_count: 0, claim_count: 0, kinds: [] }; @@ -882,8 +1044,14 @@ function experienceSummaryPhrases(card, receipt, lanes) { if (card === 'run') { phrases.push(EXPERIENCE_PHRASES.delegating); phrases.push(...uniqueProviderPhrases(lanes)); - const running = runningPhrase(count); - if (running) phrases.push(running); + if (receiptNeedsReconciliation(receipt, lanes)) { + phrases.push(EXPERIENCE_PHRASES.reconciling); + } else { + const running = simpleRunHasAuthoritativeRequiredDispatch(receipt, lanes) + ? runningPhrase(count) + : preparingPhrase(count); + if (running) phrases.push(running); + } } else if (card === 'attention') { phrases.push(EXPERIENCE_PHRASES.attention); } else if (card === 'final' && verifiedFinalAllowed(receipt, lanes)) { @@ -922,9 +1090,18 @@ function projectRunCard(receipt, lanes) { } function projectAttentionCard(receipt, lanes, items) { - const questions = projectQuestions(items, lanes); + const awaitingConsent = receipt?.phase === 'awaiting_consent'; + const consent = projectConsent(receipt); + const consentOnly = awaitingConsent || (consent != null && consentNeedsDecision(receipt)); + const questions = consentOnly ? [] : projectQuestions(items, lanes); const affected = []; const unsupportedLanes = []; + if (consentOnly) { + for (const lane of lanes) { + const id = laneId(lane); + if (id) affected.push(id); + } + } for (const question of questions) { if (question.assignment_id && !affected.includes(question.assignment_id)) { affected.push(question.assignment_id); @@ -960,32 +1137,35 @@ function projectAttentionCard(receipt, lanes, items) { const batchId = typeof receipt.attention?.batch_id === 'string' ? receipt.attention.batch_id : null; const revision = Number.isInteger(receipt.attention?.revision) ? receipt.attention.revision : null; return { + ...(consent ? { consent } : {}), questions, affected_lanes: affected, unaffected_lanes: unaffected, - reply: { - structured: true, - rounds: 1, - cursor_resume: true, - event_cursor: cursor, - run_reply: { - batch_id: batchId, - expected_revision: revision, - reply: { - round: 1, + reply: consentOnly + ? null + : { + structured: true, + rounds: 1, + cursor_resume: true, + event_cursor: cursor, + run_reply: { batch_id: batchId, - answers: questions - .filter((question) => question.reply_capability === 'same_session') - .map((question) => ({ - assignment_id: question.assignment_id, - question_id: question.question_id, - session_id: question.session_id, - task_id: question.task_id, - response: null, - })), + expected_revision: revision, + reply: { + round: 1, + batch_id: batchId, + answers: questions + .filter((question) => question.reply_capability === 'same_session') + .map((question) => ({ + assignment_id: question.assignment_id, + question_id: question.question_id, + session_id: question.session_id, + task_id: question.task_id, + response: null, + })), + }, }, }, - }, unsupported: { lanes: unsupportedLanes, unresolved: unsupportedLanes.length > 0, @@ -1001,15 +1181,32 @@ function bucketLanes(lanes, statuses) { .filter(Boolean); } +function laneHasObservedOutcome(lane) { + const status = laneStatus(lane); + if (status === 'planned' || status === 'prepared' || status === 'session_ready') return false; + if (lane.prompt_dispatched === false || status === 'failed_pre_prompt') return false; + return TERMINAL_LANE_STATUSES.includes(status) || lane.prompt_dispatched === true; +} + function projectFinalCard(receipt, lanes) { const accepted = bucketLanes(lanes, ACCEPTED_LANE_STATUSES); const failed = bucketLanes(lanes, FAILED_LANE_STATUSES); const unresolved = bucketLanes(lanes, UNRESOLVED_LANE_STATUSES); - const reviews = lanes.filter((lane) => lane.role === 'review').map(laneId).filter(Boolean); - const tests = lanes.filter((lane) => { + const reviewLanes = lanes.filter((lane) => lane.role === 'review'); + const testLanes = lanes.filter((lane) => { const scope = Array.isArray(lane.write_scope) ? lane.write_scope.join(' ') : ''; return lane.role === 'verify' || /test/iu.test(scope); - }).map(laneId).filter(Boolean); + }); + const reviews = reviewLanes.filter(laneHasObservedOutcome).map(laneId).filter(Boolean); + const plannedReviews = reviewLanes + .filter((lane) => !laneHasObservedOutcome(lane)) + .map(laneId) + .filter(Boolean); + const tests = testLanes.filter(laneHasObservedOutcome).map(laneId).filter(Boolean); + const plannedTests = testLanes + .filter((lane) => !laneHasObservedOutcome(lane)) + .map(laneId) + .filter(Boolean); const candidate = receipt.candidate && typeof receipt.candidate === 'object' ? { ref: typeof receipt.candidate.ref === 'string' ? receipt.candidate.ref : null, @@ -1049,8 +1246,16 @@ function projectFinalCard(receipt, lanes) { scope: lane.scope, role: lane.role, })), - tests: { lanes: tests, present: tests.length > 0 }, - reviews: { lanes: reviews, present: reviews.length > 0 }, + tests: { + lanes: tests, + present: tests.length > 0, + planned_lanes: plannedTests, + }, + reviews: { + lanes: reviews, + present: reviews.length > 0, + planned_lanes: plannedReviews, + }, candidate, evidence, evidence_refs: evidenceRefs, @@ -1085,6 +1290,18 @@ function projectFinalCard(receipt, lanes) { }; } +function projectCoordination(receipt, verified) { + const hasRun = typeof receipt?.run_id === 'string' && receipt.run_id !== ''; + const operation = typeof receipt?.operation === 'string' ? receipt.operation : null; + return { + aggregate_wait: EXPERIENCE_COORDINATION.aggregate_wait, + submissions: operation === 'submit' || hasRun ? 1 : 0, + aggregate_wait_count: operation === 'wait' ? 1 : 0, + verified_final_decisions: verified === true ? 1 : 0, + grouped_reply: operation === 'reply' ? 1 : 0, + }; +} + function boundProjection(projection) { let text = JSON.stringify(projection); if (byteLength(text) <= EXPERIENCE_MAX_BYTES) return projection; @@ -1115,7 +1332,7 @@ function boundProjection(projection) { phrases: (projection.summary?.phrases ?? []).slice(0, 2), }, truncated: true, - coordination: EXPERIENCE_COORDINATION, + coordination: projection.coordination, authority: EXPERIENCE_AUTHORITY, }; } @@ -1127,6 +1344,7 @@ export function projectExperience(receipt) { const items = bindProofBoundQuestions(attentionItems(safe), boundAttention); const card = classifyExperienceCard(safe); const phrases = experienceSummaryPhrases(card, safe, lanes); + const verifiedFinal = card === 'final' && verifiedFinalAllowed(safe, lanes); const projection = { schema: EXPERIENCE_SCHEMA, version: EXPERIENCE_VERSION, @@ -1134,15 +1352,24 @@ export function projectExperience(receipt) { summary: { phrases, delegating: card === 'run' ? EXPERIENCE_PHRASES.delegating : null, - running: card === 'run' ? runningPhrase( - Number.isInteger(safe.assignment_count) ? safe.assignment_count : lanes.length, - ) : null, + running: card === 'run' && !receiptNeedsReconciliation(safe, lanes) + ? (simpleRunHasAuthoritativeRequiredDispatch(safe, lanes) + ? runningPhrase(Number.isInteger(safe.assignment_count) ? safe.assignment_count : lanes.length) + : preparingPhrase(Number.isInteger(safe.assignment_count) ? safe.assignment_count : lanes.length)) + : null, + preparing: card === 'run' && !receiptNeedsReconciliation(safe, lanes) + && !simpleRunHasAuthoritativeRequiredDispatch(safe, lanes) + ? preparingPhrase(Number.isInteger(safe.assignment_count) ? safe.assignment_count : lanes.length) + : null, + reconciling: card === 'run' && receiptNeedsReconciliation(safe, lanes) + ? EXPERIENCE_PHRASES.reconciling + : null, attention: card === 'attention' ? EXPERIENCE_PHRASES.attention : null, - verified_final: card === 'final' && verifiedFinalAllowed(safe, lanes) + verified_final: verifiedFinal ? EXPERIENCE_PHRASES.verified_final : null, }, - coordination: { ...EXPERIENCE_COORDINATION }, + coordination: projectCoordination(safe, verifiedFinal), authority: { ...EXPERIENCE_AUTHORITY }, run_id: typeof safe.run_id === 'string' ? safe.run_id : null, truncated: false, @@ -1249,6 +1476,7 @@ export function resolveExperienceToolMeta(toolName, { export function resolveExperienceResultMeta({ card = null, + experience = null, clientCapabilities = null, resources = experienceUiResourceRegistry(), } = {}) { @@ -1260,7 +1488,11 @@ export function resolveExperienceResultMeta({ for (const uri of candidates) { const resource = typeof resources?.get === 'function' ? resources.get(uri) : null; if (resource && resource.mimeType === MCP_APPS_MIME_TYPE && isMcpAppsResourceUri(uri)) { - return { ui: { resourceUri: uri } }; + const safeExperience = normalizeExperienceResultMeta(experience); + return { + ui: { resourceUri: uri }, + ...(safeExperience ? { [EXPERIENCE_RESULT_META_KEY]: safeExperience } : {}), + }; } } return null; @@ -1313,5 +1545,16 @@ function normalizeToolResultUiMeta(uiMeta) { ? uiMeta.ui.resourceUri : null; if (!isMcpAppsResourceUri(nested)) return null; - return { ui: { resourceUri: nested } }; + const safeExperience = normalizeExperienceResultMeta(uiMeta[EXPERIENCE_RESULT_META_KEY]); + return { + ui: { resourceUri: nested }, + ...(safeExperience ? { [EXPERIENCE_RESULT_META_KEY]: safeExperience } : {}), + }; +} + +function normalizeExperienceResultMeta(experience) { + if (!experience || typeof experience !== 'object' || Array.isArray(experience)) return null; + const safe = stripOwnerOnly(experience); + if (safe?.schema !== EXPERIENCE_SCHEMA || !EXPERIENCE_CARD_STATES.includes(safe?.card)) return null; + return byteLength(JSON.stringify(safe)) <= EXPERIENCE_MAX_BYTES ? safe : null; } diff --git a/plugins/codex-co-engineer/mcp/v3/run-admission-store.mjs b/plugins/codex-co-engineer/mcp/v3/run-admission-store.mjs new file mode 100644 index 0000000..109b30a --- /dev/null +++ b/plugins/codex-co-engineer/mcp/v3/run-admission-store.mjs @@ -0,0 +1,231 @@ +// Private, additive persistence for 3.4.1 simple-run coordination. +// +// The legacy run store remains the authority for 3.4.0 full envelopes. This +// store keeps the compiled request and lifecycle record under a separate +// owner-only directory so a server restart can observe the same run without +// rewriting or migrating existing receipts. + +import { constants as fsConstants } from 'node:fs'; +import { randomUUID } from 'node:crypto'; +import { chmod, mkdir, open, rename, lstat, unlink } from 'node:fs/promises'; +import path from 'node:path'; + +import { assertRunId } from './run-manifest.mjs'; +import { assertDirectJsonClosure } from './selection-json.mjs'; + +export const RUN_ADMISSION_STORE_SCHEMA = 'codex-co-engineer.run-admission-store.v1'; +export const RUN_ADMISSION_STORE_DIRECTORY = 'runs-3.4.1'; +export const RUN_ADMISSION_STORE_MAX_BYTES = 512 * 1024; + +const ROOT_FLAGS = fsConstants.O_RDONLY + | (fsConstants.O_DIRECTORY ?? 0) + | (fsConstants.O_NOFOLLOW ?? 0) + | (fsConstants.O_NONBLOCK ?? 0); +const READ_FLAGS = fsConstants.O_RDONLY + | (fsConstants.O_NOFOLLOW ?? 0) + | (fsConstants.O_NONBLOCK ?? 0); +const WRITE_FLAGS = fsConstants.O_WRONLY + | fsConstants.O_CREAT + | fsConstants.O_EXCL + | (fsConstants.O_NOFOLLOW ?? 0); +const RUN_ID = /^[a-z][a-z0-9-]{2,63}$/u; +const HEX = /^[0-9a-f]{64}$/u; + +function storeError(code, message) { + throw Object.assign(new Error(message), { code }); +} + +function safeRoot(value) { + if (typeof value !== 'string' || !path.isAbsolute(value) || path.resolve(value) !== value || value.includes('\0')) { + storeError('run_store_path_unsafe', 'The run admission store root must be a normalized absolute path.'); + } + return value; +} + +function safeRunId(value) { + try { + assertRunId(value, 'run_id'); + } catch { + storeError('run_store_identity_invalid', 'The run admission store run id is invalid.'); + } + if (!RUN_ID.test(value)) storeError('run_store_identity_invalid', 'The run admission store run id is invalid.'); + return value; +} + +function fileName(runId) { + return `${safeRunId(runId)}.json`; +} + +function ownerUid() { + return typeof process.geteuid === 'function' ? process.geteuid() : undefined; +} + +function assertPrivateDirectory(metadata, code = 'run_store_root_unsafe') { + if (metadata.isSymbolicLink?.() || !metadata.isDirectory?.()) storeError(code, 'The run admission store directory is not a real private directory.'); + const uid = ownerUid(); + if (uid !== undefined && Number(metadata.uid) !== uid) storeError(code, 'The run admission store directory has the wrong owner.'); + if ((metadata.mode & 0o077) !== 0) storeError(code, 'The run admission store directory is not owner-only.'); +} + +function assertPrivateFile(metadata) { + if (metadata.isSymbolicLink?.() || !metadata.isFile?.() || Number(metadata.nlink) !== 1) { + storeError('run_store_record_unsafe', 'The run admission store record is not a private regular file.'); + } + const uid = ownerUid(); + if (uid !== undefined && Number(metadata.uid) !== uid) storeError('run_store_record_unsafe', 'The run admission store record has the wrong owner.'); + if ((metadata.mode & 0o077) !== 0) storeError('run_store_record_unsafe', 'The run admission store record is not owner-only.'); +} + +async function ensureDirectory(directory) { + await mkdir(directory, { recursive: true, mode: 0o700 }); + await chmod(directory, 0o700); + const metadata = await lstat(directory); + assertPrivateDirectory(metadata); + return metadata; +} + +async function openRoot(directory) { + let handle; + try { + handle = await open(directory, ROOT_FLAGS); + } catch (error) { + storeError('run_store_root_unsafe', `The run admission store directory could not be opened (${error?.code ?? 'unknown'}).`); + } + try { + const metadata = await handle.stat(); + assertPrivateDirectory(metadata); + return { handle, metadata }; + } catch (error) { + await handle.close().catch(() => {}); + throw error; + } +} + +function canonicalRecord(record) { + if (!record || typeof record !== 'object' || Array.isArray(record)) { + storeError('run_store_record_invalid', 'The run admission record must be a JSON object.'); + } + let plain; + try { + // The compiled request intentionally shares immutable identity objects + // between its public and child views. Persistence stores a JSON tree, so + // de-alias before applying the strict on-disk closure checks. + plain = JSON.parse(JSON.stringify(record)); + } catch { + storeError('run_store_record_invalid', 'The run admission record is not serializable JSON.'); + } + assertDirectJsonClosure(plain, 'run_record'); + if (plain.schema !== 'codex-co-engineer.run-admission.v1' || typeof plain.run_id !== 'string') { + storeError('run_store_record_invalid', 'The run admission record identity is invalid.'); + } + safeRunId(plain.run_id); + const serialized = JSON.stringify(plain); + if (typeof serialized !== 'string' || Buffer.byteLength(serialized, 'utf8') > RUN_ADMISSION_STORE_MAX_BYTES) { + storeError('run_store_record_too_large', 'The run admission record exceeds its private size bound.'); + } + return `${serialized}\n`; +} + +function parseRecord(text, runId) { + let record; + try { + record = JSON.parse(text); + } catch { + storeError('run_store_record_invalid', 'The run admission record is not valid JSON.'); + } + canonicalRecord(record); + if (record.run_id !== runId) storeError('run_store_identity_mismatch', 'The run admission record belongs to another run.'); + return record; +} + +export function createRunAdmissionStore(root) { + const stateRoot = safeRoot(root); + const directory = path.join(stateRoot, RUN_ADMISSION_STORE_DIRECTORY); + let ready; + const initialize = async () => { + ready ??= ensureDirectory(directory); + await ready; + }; + + async function load(runId) { + const name = fileName(runId); + await initialize(); + const rootHandle = await openRoot(directory); + const target = path.join(directory, name); + try { + let handle; + try { + handle = await open(target, READ_FLAGS); + } catch (error) { + if (error?.code === 'ENOENT') return null; + storeError('run_store_record_unsafe', 'The run admission record could not be opened safely.'); + } + try { + const metadata = await handle.stat(); + assertPrivateFile(metadata); + if (Number(metadata.size) > RUN_ADMISSION_STORE_MAX_BYTES) { + storeError('run_store_record_too_large', 'The run admission record exceeds its private size bound.'); + } + const bytes = await handle.readFile(); + const after = await handle.stat(); + if (Number(after.ino) !== Number(metadata.ino) || Number(after.size) !== Number(metadata.size)) { + storeError('run_store_record_changed', 'The run admission record changed while it was read.'); + } + return parseRecord(bytes.toString('utf8'), runId); + } finally { + await handle.close().catch(() => {}); + } + } finally { + await rootHandle.handle.close().catch(() => {}); + } + } + + async function save(record) { + const runId = safeRunId(record?.run_id); + const text = canonicalRecord(record); + await initialize(); + const rootHandle = await openRoot(directory); + const name = fileName(runId); + const temporaryName = `.tmp-${randomUUID()}.json`; + const temporary = path.join(directory, temporaryName); + const target = path.join(directory, name); + try { + try { + const existing = await lstat(target); + if (existing.isSymbolicLink?.()) storeError('run_store_record_unsafe', 'The run admission record target is a symbolic link.'); + assertPrivateFile(existing); + } catch (error) { + if (error?.code !== 'ENOENT') throw error; + } + const handle = await open(temporary, WRITE_FLAGS, 0o600); + try { + await handle.chmod(0o600); + await handle.writeFile(text, 'utf8'); + await handle.sync(); + } finally { + await handle.close().catch(() => {}); + } + await rename(temporary, target); + await rootHandle.handle.sync().catch(() => {}); + } finally { + await rootHandle.handle.close().catch(() => {}); + // A failed write may leave only our uniquely named temporary behind. + // Do not unlink arbitrary paths or a caller-selected target here. + try { + const leftover = await lstat(temporary); + if (leftover.isFile?.() && !leftover.isSymbolicLink?.()) { + await unlink(temporary); + } + } catch { + // No temporary or an unsafe replacement is left for the next + // fail-closed store inspection. + } + } + } + + async function has(runId) { + return (await load(runId)) !== null; + } + + return Object.freeze({ directory, load, save, has }); +} diff --git a/plugins/codex-co-engineer/mcp/v3/run-admission.mjs b/plugins/codex-co-engineer/mcp/v3/run-admission.mjs new file mode 100644 index 0000000..e89e9a1 --- /dev/null +++ b/plugins/codex-co-engineer/mcp/v3/run-admission.mjs @@ -0,0 +1,1984 @@ +// RunAdmissionV1 — the 3.4.1 two-barrier coordinator for SimpleRunRequestV1. +// +// The legacy RunRuntimeV1 remains available for existing full run envelopes. +// This module owns the additive simple-request lifecycle so admission, +// provider dispatch, recovery, consent, cancellation, and terminal handoffs +// have an explicit state machine without changing the 3.4.0 durable schemas. +// +// Barrier A (admission) completes all validation/readiness/workspace work +// before dispatchPrompt is called for any lane. Barrier B records session and +// prompt evidence lane by lane. Provider dispatch is not transactional: a +// later failure is represented as degraded with exact sent/unsent lanes. + +import { + capturedFreeze, + capturedHasOwn, + capturedIncludes, + capturedIsArray, + capturedOwnKeys, + capturedTest, +} from './grammar.mjs'; +import { + compileRunRequestV1, + RUN_REQUEST_SCHEMA_ID, + RUN_REQUEST_VERSION, +} from './run-request-compiler.mjs'; +import { + MAX_ASSIGNMENTS, + MIN_ASSIGNMENTS, + RunContractV1Error, + assertRunId, + isAssignmentId, +} from './run-manifest.mjs'; +import { + assertDirectJsonClosure, + assertNotProxy, + assertPlainObject, + freezeData, + identityBoundDigest, +} from './selection-json.mjs'; +import { IDENTITY_LABELS } from './identity.mjs'; +import { boundProviderResult } from './compact-task.mjs'; +import { + validateChildIdentityV1, + validateDispatchAttemptV1, + validateGitIdentityV1, + validateProviderRunIdentityV1, + validateRunIdentityV1, + validateWorkspaceIdentityV1, +} from './protected-identity.mjs'; + +export const RUN_ADMISSION_SCHEMA_ID = 'codex-co-engineer.run-admission.v1'; +export const RUN_ADMISSION_VERSION = 1; +export const RUN_PHASES = capturedFreeze([ + 'validating', + 'awaiting_consent', + 'preparing_workspaces', + 'dispatching', + 'running', + 'needs_attention', + 'degraded', + 'verifying', + 'completed', + 'failed', + 'cancelled', +]); +export const RUN_TERMINAL_PHASES = capturedFreeze(['completed', 'failed', 'cancelled']); +export const RUN_ATTENTION_PHASES = capturedFreeze(['needs_attention', 'degraded']); +export const LANE_PHASES = capturedFreeze([ + 'planned', + 'prepared', + 'session_ready', + 'prompt_dispatched', + 'running', + 'needs_attention', + 'completed', + 'partial_handoff', + 'failed_pre_prompt', + 'unrecoverable_post_prompt', + 'cancelled', +]); +export const LANE_TERMINAL_PHASES = capturedFreeze([ + 'completed', + 'partial_handoff', + 'failed_pre_prompt', + 'unrecoverable_post_prompt', + 'cancelled', +]); +export const RUN_ADMISSION_METHODS = capturedFreeze([ + 'submitRunRequest', 'inspectRun', 'resumeRun', 'replyRun', 'cancelRun', 'waitRun', +]); +export const RUN_ADMISSION_DEPENDENCIES = capturedFreeze([ + 'requestConsent', 'verifyConsent', 'providerReady', 'processBoundaryReady', + 'verifyRepository', 'prepareWorkspace', 'cleanupWorkspace', 'createSession', + 'dispatchPrompt', 'inspectLane', 'reconnectLane', 'replyAttention', 'cancelLane', + 'inspectWorkspace', 'buildHandoff', 'verifyRun', 'clock', 'sleep', 'compile', + 'loadRecord', 'persistRecord', 'waitForProgress', +]); +const MAX_PROVIDER_RESULT_BYTES = 8 * 1024; + +export const RUN_ADMISSION_CAPS = capturedFreeze({ + max_lanes: MAX_ASSIGNMENTS, + max_handoff_bytes: 16_384, + max_error_bytes: 160, + max_changed_files: 64, + max_commits: 64, + max_next_actions: 8, + max_provider_result_bytes: MAX_PROVIDER_RESULT_BYTES, +}); + +const TELEMETRY_DEFAULTS = Object.freeze({ + admission_duration_ms: null, + admission_failure_stage: null, + workspace_preparation_duration_ms: null, + dispatch_duration_ms: null, + dispatch_confidence: 'not_dispatched', + time_to_session_ms: null, + time_to_prompt_dispatch_ms: null, + time_to_first_event_ms: null, + last_meaningful_activity_at: null, + silence_duration_ms: null, + time_to_terminal_handoff_ms: null, + recovery_path: null, + handoff_class: null, + attention_count: 0, + attention_deduplicated_count: 0, + cancel_attempts: 0, + cancel_confirmed: null, + response_bytes: null, + response_truncated: false, +}); + +const RUN_ID_PATTERN = /^[a-z][a-z0-9-]{2,63}$/u; +const TASK_ID_PATTERN = /^[A-Za-z0-9][A-Za-z0-9._-]{0,79}$/u; +const CURSOR_PATTERN = /^\d{1,16}$/u; +const MAX_WAIT_MS = 14_400_000; +const OBSERVATION_BACKOFF_MS = 1_000; +const OBSERVATION_RECOVERY_CLASSIFICATION = 'post_prompt_inspection_failed_no_replay'; +const KNOWN_LANE_OBSERVATION_STATUSES = capturedFreeze([ + 'running', 'starting', 'accepted', 'cancelling', 'needs_attention', + 'completed', 'succeeded', 'cancelled', 'transport_lost', + 'environment_blocked', 'failed', 'timeout', 'timed_out', +]); +const SAFE_CONSENT_ERROR_CODES = capturedFreeze([ + 'consent_host_unavailable', + 'consent_declined', + 'consent_cancelled', + 'consent_timed_out', + 'consent_response_invalid', + 'consent_request_aborted', + 'consent_grant_store_invalid', + 'consent_repository_identity_changed', +]); +const RETRYABLE_CONSENT_ERROR_CODES = capturedFreeze([ + 'consent_cancelled', + 'consent_timed_out', + 'consent_request_aborted', + 'consent_repository_identity_changed', +]); +const SAFE_ATTENTION_CAPABILITIES = capturedFreeze([ + 'read_run_receipts', 'read_provider_logs', 'read_own_worktree', +]); +const CAPABILITY_RESOURCES = Object.freeze({ + read_run_receipts: 'run_receipt', + read_provider_logs: 'provider_log', + read_own_worktree: 'own_worktree', +}); + +function admissionError(code, field, message = 'The run admission request is invalid.') { + throw new RunContractV1Error(code, field, message); +} + +function asErrorCode(error, fallback = 'provider_failure') { + return typeof error?.code === 'string' && /^[a-z][a-z0-9_]{1,63}$/u.test(error.code) + ? error.code + : fallback; +} + +const ADMISSION_FAILURE_MESSAGES = Object.freeze({ + runtime_install_incomplete: 'The installed Codex-Co-Engineer runtime is incomplete. Reinstall the plugin, then restart Codex.', +}); + +function errorSummary(error, fallback = 'The run operation failed.') { + const code = asErrorCode(error, fallback); + const message = ADMISSION_FAILURE_MESSAGES[code] ?? code; + return { code, message: message.length > RUN_ADMISSION_CAPS.max_error_bytes + ? message.slice(0, RUN_ADMISSION_CAPS.max_error_bytes) : message }; +} + +function ownObject(value, field) { + assertNotProxy(value, field); + assertPlainObject(value, 'invalid_type', field, field); + assertDirectJsonClosure(value, field); + return value; +} + +function assertKeys(value, allowed, field) { + for (const key of capturedOwnKeys(value)) { + if (typeof key !== 'string' || !allowed.includes(key)) { + admissionError('unknown_key', `${field}.${String(key)}`, 'The run operation field is outside the closed vocabulary.'); + } + } +} + +function requiredString(value, key, field, predicate = null) { + if (!capturedHasOwn(value, key)) admissionError('missing_key', field, 'A required run field is missing.'); + const result = value[key]; + if (typeof result !== 'string' || (predicate && !predicate(result))) { + admissionError('invalid_format', field, 'A run field is not in the required format.'); + } + return result; +} + +function optionalString(value, key, field, predicate = null) { + if (!capturedHasOwn(value, key)) return undefined; + const result = value[key]; + if (typeof result !== 'string' || (predicate && !predicate(result))) { + admissionError('invalid_format', field, 'A run field is not in the required format.'); + } + return result; +} + +function nowIso(clock) { + const value = typeof clock === 'function' ? clock() : new Date().toISOString(); + return typeof value === 'string' ? value : new Date().toISOString(); +} + +function cloneError(error) { + return error ? freezeData({ + code: typeof error.code === 'string' ? error.code : 'provider_failure', + message: typeof error.message === 'string' + ? error.message.slice(0, RUN_ADMISSION_CAPS.max_error_bytes) + : 'The provider operation failed.', + }) : null; +} + +const PROVIDER_FAILURE_MESSAGES = Object.freeze({ + provider_billing_required: 'The provider requires billing setup or available credit.', + authentication_required: 'The provider requires valid authentication.', + provider_rate_limited: 'The provider rate limit was reached. Retry later.', +}); + +function terminalLaneError(response, fallbackCode) { + const code = response.error?.code; + return cloneError(typeof code === 'string' && capturedHasOwn(PROVIDER_FAILURE_MESSAGES, code) + ? { code, message: PROVIDER_FAILURE_MESSAGES[code] } + : { code: fallbackCode }); +} + +function normalizeAttention(value) { + if (value === null || value === undefined) return null; + try { + assertNotProxy(value, 'attention'); + assertPlainObject(value, 'invalid_type', 'attention', 'attention'); + assertDirectJsonClosure(value, 'attention'); + } catch { + return { invalid: true }; + } + const allowed = new Set([ + 'session_id', 'question_id', 'capability', 'resource', 'action', + 'prompt', 'options', 'required', 'event_cursor', 'deadline_at', 'stage', + ]); + for (const key of Object.keys(value)) if (!allowed.has(key)) return { invalid: true }; + for (const key of ['session_id', 'question_id', 'capability', 'resource', 'action', 'event_cursor', 'deadline_at']) { + if (value[key] !== undefined && (typeof value[key] !== 'string' || value[key].length === 0 || value[key].length > 256)) { + return { invalid: true }; + } + } + if (value.capability !== undefined && !SAFE_ATTENTION_CAPABILITIES.includes(value.capability)) return { invalid: true }; + if (value.capability !== undefined + && (value.resource !== CAPABILITY_RESOURCES[value.capability] || value.action !== 'read')) return { invalid: true }; + if (value.prompt !== undefined && (typeof value.prompt !== 'string' || value.prompt.length > 4096)) return { invalid: true }; + if (value.stage !== undefined && (typeof value.stage !== 'string' || value.stage.length === 0 || value.stage.length > 128)) return { invalid: true }; + let options; + if (value.options !== undefined && value.options !== null) { + if (!Array.isArray(value.options) || value.options.length > 8) return { invalid: true }; + options = []; + const allowedOptionKeys = new Set(['optionId', 'kind', 'name', 'label', 'description']); + for (const option of value.options) { + if (typeof option === 'string') { + if (option.length > 128) return { invalid: true }; + options.push(option); + continue; + } + try { + assertNotProxy(option, 'attention.options'); + assertPlainObject(option, 'invalid_type', 'attention.options', 'attention option'); + assertDirectJsonClosure(option, 'attention.options'); + } catch { + return { invalid: true }; + } + for (const key of Object.keys(option)) { + if (!allowedOptionKeys.has(key) + || typeof option[key] !== 'string' + || option[key].length === 0 + || option[key].length > 128) return { invalid: true }; + } + if (typeof option.kind !== 'string') return { invalid: true }; + options.push({ ...option }); + } + } + if (value.required !== undefined && typeof value.required !== 'boolean') return { invalid: true }; + return { + ...value, + ...(options !== undefined ? { options } : {}), + }; +} + +function capabilityQuestionKey(attention) { + if (!attention?.capability || !attention.resource || !attention.action) return null; + return `${attention.capability}\u0000${attention.resource}\u0000${attention.action}`; +} + +function attentionDedupKey(attention) { + return capabilityQuestionKey(attention) + ?? (attention?.question_id ? `question\u0000${attention.question_id}` : null); +} + +function attentionTarget(lane, attention) { + return { + assignment_id: lane.assignment_id, + task_id: lane.task_id, + provider: lane.provider, + session_id: attention.session_id ?? lane.session_id ?? null, + question_id: attention.question_id ?? null, + }; +} + +function isCapabilityCovered(assignment, attention) { + const capability = attention?.capability; + if (!SAFE_ATTENTION_CAPABILITIES.includes(capability)) return false; + return attention.resource === CAPABILITY_RESOURCES[capability] + && attention.action === 'read' + && Array.isArray(assignment?.capabilities) + && assignment.capabilities.includes(capability); +} + +function mergeAttention(record, lane, assignment, attention) { + const key = attentionDedupKey(attention); + const capabilityKey = capabilityQuestionKey(attention); + const target = attentionTarget(lane, attention); + const item = { + assignment_id: lane.assignment_id, + task_id: lane.task_id, + provider: lane.provider, + required: lane.required, + ...attention, + ...(capabilityKey ? { capability_key: capabilityKey } : {}), + ...(key ? { attention_key: key } : {}), + targets: [target], + }; + if (!Array.isArray(record.attention_questions)) record.attention_questions = []; + if (key && record.attention?.status === 'open') { + const existingIndex = record.attention_questions.findIndex((entry) => entry.attention_key === key); + if (existingIndex >= 0) { + const existing = record.attention_questions[existingIndex]; + const existingTargets = Array.isArray(existing.targets) && existing.targets.length > 0 + ? existing.targets + : [attentionTarget(existing, existing)]; + const targetKey = JSON.stringify(target); + const hasTarget = existingTargets.some((entry) => JSON.stringify(entry) === targetKey); + const updated = { + ...existing, + ...(hasTarget ? {} : { targets: [...existingTargets, target] }), + }; + const questions = record.attention_questions.slice(); + questions[existingIndex] = updated; + record.attention_questions = questions; + const items = record.attention_questions.slice(-MAX_ASSIGNMENTS); + return { + ...record.attention, + revision: record.revision, + items, + }; + } + } + record.attention_questions.push(item); + const items = record.attention_questions.slice(-MAX_ASSIGNMENTS); + return { + kind: 'grouped_attention', + status: 'open', + batch_id: `att-${identityBoundDigest(IDENTITY_LABELS.RUN_IDENTITY, { + run_id: record.run_id, + items: items.map((entry) => ({ + assignment_id: entry.assignment_id, + attention_key: entry.attention_key ?? null, + question_id: entry.question_id ?? null, + })), + }).slice(7, 39)}`, + revision: record.revision, + items, + }; +} + +function attentionWasSatisfied(lane, key) { + return key !== null && Array.isArray(lane.attention_satisfied_keys) + && lane.attention_satisfied_keys.includes(key); +} + +function publicConsentRequest(compiled) { + const providers = []; + for (const assignment of compiled.assignments) { + if (!providers.includes(assignment.provider)) providers.push(assignment.provider); + } + const repositoryIdentity = compiled.git_identity?.digest + ?? compiled.run_identity?.digest + ?? identityBoundDigest(IDENTITY_LABELS.RUN_IDENTITY, { + run_id: compiled.run_id, + base_sha: compiled.git?.base_sha ?? null, + }); + return freezeData({ + kind: 'repository_exposure_consent', + run_id: compiled.run_id, + repository_identity: repositoryIdentity, + providers, + scope: 'full_repository', + duration: 'user_selected', + duration_options: ['repository_and_selected_providers', 'this_run_only'], + default_duration: 'repository_and_selected_providers', + remote_mutation: false, + }); +} + +function bindingForConsent(compiled) { + return freezeData({ + run_id: compiled.run_id, + repository_identity: compiled.git_identity?.digest ?? compiled.run_identity?.digest ?? null, + providers: [...new Set(compiled.assignments.map((assignment) => assignment.provider))].sort(), + scope: 'full_repository', + duration: 'this_run_only', + remote_mutation: false, + }); +} + +function validateCompiled(compiled) { + assertNotProxy(compiled, 'compiled_run'); + assertPlainObject(compiled, 'invalid_type', 'compiled_run', 'compiled_run'); + if (compiled.schema !== RUN_REQUEST_SCHEMA_ID || compiled.version !== RUN_REQUEST_VERSION) { + admissionError('durable_state_mismatch', 'compiled_run', 'Compiled run schema or version is invalid.'); + } + const runId = requiredString(compiled, 'run_id', 'compiled_run.run_id', (value) => RUN_ID_PATTERN.test(value)); + let gitIdentity; + let runIdentity; + try { + gitIdentity = validateGitIdentityV1(compiled.git_identity, 'compiled_run.git_identity'); + runIdentity = validateRunIdentityV1(compiled.run_identity, 'compiled_run.run_identity'); + } catch (error) { + if (error instanceof RunContractV1Error) throw error; + admissionError('durable_state_mismatch', 'compiled_run.identity', 'Compiled run identity is invalid.'); + } + if (gitIdentity.repository_path !== compiled.git?.repository_path + || gitIdentity.base_sha !== compiled.git?.base_sha + || runIdentity.run_id !== runId + || runIdentity.git.digest !== gitIdentity.digest + || runIdentity.manifest_digest !== compiled.manifest_digest) { + admissionError('durable_state_mismatch', 'compiled_run.identity', 'Compiled run identity does not match its Git or manifest binding.'); + } + const assignments = compiled.assignments; + if (!capturedIsArray(assignments) || assignments.length < MIN_ASSIGNMENTS || assignments.length > MAX_ASSIGNMENTS) { + admissionError('invalid_format', 'compiled_run.assignments', 'Compiled run assignments are outside the supported bound.'); + } + const ids = new Set(); + for (let index = 0; index < assignments.length; index += 1) { + const assignment = ownObject(assignments[index], `compiled_run.assignments[${index}]`); + const assignmentId = requiredString(assignment, 'assignment_id', `compiled_run.assignments[${index}].assignment_id`, isAssignmentId); + const taskId = requiredString(assignment, 'task_id', `compiled_run.assignments[${index}].task_id`, (value) => TASK_ID_PATTERN.test(value)); + if (ids.has(assignmentId)) admissionError('duplicate_assignment_id', 'compiled_run.assignments', 'Compiled assignment IDs must be unique.'); + ids.add(assignmentId); + if (typeof assignment.provider !== 'string' || typeof assignment.model !== 'string') { + admissionError('invalid_format', `compiled_run.assignments[${index}]`, 'Compiled provider selection is incomplete.'); + } + try { + const child = validateChildIdentityV1(assignment.child_identity, `compiled_run.assignments[${index}].child_identity`); + const dispatch = validateDispatchAttemptV1(assignment.dispatch_identity, `compiled_run.assignments[${index}].dispatch_identity`); + const providerRun = validateProviderRunIdentityV1(assignment.provider_run_identity, `compiled_run.assignments[${index}].provider_run_identity`); + if (child.run_id !== runId || child.assignment_id !== assignmentId + || dispatch.run_id !== runId || dispatch.assignment_id !== assignmentId || dispatch.attempt !== 1 + || providerRun.run_id !== runId || providerRun.assignment_id !== assignmentId + || providerRun.attempt !== 1 || providerRun.provider !== assignment.provider + || providerRun.model !== assignment.model || providerRun.git.digest !== gitIdentity.digest + || providerRun.manifest_digest !== compiled.manifest_digest + || providerRun.prompt_envelope_digest !== assignment.prompt_envelope_digest + || providerRun.resolved_lane_digest !== assignment.lane_digest + || providerRun.capability_snapshot_digest !== assignment.capability_digest) { + admissionError('durable_state_mismatch', `compiled_run.assignments[${index}]`, 'Compiled lane identity does not match its assignment binding.'); + } + } catch (error) { + if (error instanceof RunContractV1Error) throw error; + admissionError('durable_state_mismatch', `compiled_run.assignments[${index}]`, 'Compiled lane identity is invalid.'); + } + void taskId; + } + return { runId, assignments }; +} + +function rejectNestedKey(value, forbiddenKey, field = 'persisted_run') { + if (value === null || typeof value !== 'object') return; + for (const key of capturedOwnKeys(value)) { + if (key === forbiddenKey) { + admissionError('durable_state_mismatch', `${field}.${String(key)}`, + 'Persisted run state contains a forbidden secret-bearing field.'); + } + rejectNestedKey(value[key], forbiddenKey, `${field}.${String(key)}`); + } +} + +function validatePersistedRecord(record, runId) { + ownObject(record, 'persisted_run'); + if (record.schema !== RUN_ADMISSION_SCHEMA_ID + || record.version !== RUN_ADMISSION_VERSION + || record.run_id !== runId) { + admissionError('durable_state_mismatch', 'persisted_run', + 'Persisted run schema or identity is invalid.'); + } + if (!Number.isSafeInteger(record.revision) || record.revision < 0 + || !RUN_PHASES.includes(record.phase) + || !capturedIsArray(record.lanes) + || record.lanes.length < MIN_ASSIGNMENTS + || record.lanes.length > MAX_ASSIGNMENTS) { + admissionError('durable_state_mismatch', 'persisted_run', + 'Persisted run lifecycle state is invalid.'); + } + const { assignments } = validateCompiled(record.compiled); + const assignmentsById = new Map(assignments.map((assignment) => [assignment.assignment_id, assignment])); + const seen = new Set(); + for (let index = 0; index < record.lanes.length; index += 1) { + const lane = ownObject(record.lanes[index], `persisted_run.lanes[${index}]`); + const assignment = assignmentsById.get(lane.assignment_id); + if (!assignment || seen.has(lane.assignment_id) + || lane.task_id !== assignment.task_id + || lane.provider !== assignment.provider + || lane.model !== assignment.model + || lane.role !== assignment.role + || lane.access !== assignment.access + || lane.required !== assignment.required + || !LANE_PHASES.includes(lane.phase)) { + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}]`, + 'Persisted lane identity or phase is invalid.'); + } + seen.add(lane.assignment_id); + if (lane.child_identity?.digest !== assignment.child_identity?.digest + || lane.dispatch_identity?.digest !== assignment.dispatch_identity?.digest + || lane.provider_run_identity?.digest !== assignment.provider_run_identity?.digest) { + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}]`, + 'Persisted lane identities do not match the compiled assignment.'); + } + for (const key of ['prepared', 'session_ready', 'prompt_attempted', 'prompt_dispatched']) { + if (typeof lane[key] !== 'boolean') { + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}].${key}`, + 'Persisted lane evidence is invalid.'); + } + } + if (!['not_sent', 'authoritative', 'uncertain'].includes(lane.dispatch_confidence)) { + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}].dispatch_confidence`, + 'Persisted dispatch confidence is invalid.'); + } + if (lane.prompt_dispatched && (!lane.prompt_attempted || lane.dispatch_confidence !== 'authoritative')) { + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}]`, + 'Persisted dispatch evidence violates the no-replay boundary.'); + } + if (lane.result_truncated !== undefined && typeof lane.result_truncated !== 'boolean') { + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}].result_truncated`, + 'Persisted result truncation state is invalid.'); + } + if (lane.cancel_confirmed !== undefined && lane.cancel_confirmed !== null + && typeof lane.cancel_confirmed !== 'boolean') { + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}].cancel_confirmed`, + 'Persisted cancellation state is invalid.'); + } + if (!capturedIsArray(lane.attention_satisfied_keys)) { + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}].attention_satisfied_keys`, + 'Persisted attention state is invalid.'); + } + if (lane.workspace_identity !== null) { + try { + const workspace = validateWorkspaceIdentityV1(lane.workspace_identity, `persisted_run.lanes[${index}].workspace_identity`); + if (workspace.run_id !== runId || workspace.assignment_id !== lane.assignment_id) { + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}].workspace_identity`, + 'Persisted workspace identity is bound to another run or assignment.'); + } + } catch (error) { + if (error instanceof RunContractV1Error) throw error; + admissionError('durable_state_mismatch', `persisted_run.lanes[${index}].workspace_identity`, + 'Persisted workspace identity is invalid.'); + } + } + } + if (seen.size !== assignments.length) { + admissionError('durable_state_mismatch', 'persisted_run.lanes', + 'Persisted run does not contain every compiled assignment exactly once.'); + } + if (!capturedIsArray(record.attention_questions) + || record.attention_questions.length > MAX_ASSIGNMENTS + || (record.consent_status !== 'pending' + && record.consent_status !== 'required' + && record.consent_status !== 'approved' + && record.consent_status !== 'blocked')) { + admissionError('durable_state_mismatch', 'persisted_run', + 'Persisted consent or attention state is invalid.'); + } + rejectNestedKey(record, 'approval_ref'); + return record; +} + +function ensureTelemetry(record) { + if (record.telemetry !== undefined + && (record.telemetry === null || typeof record.telemetry !== 'object' || Array.isArray(record.telemetry))) { + admissionError('durable_state_mismatch', 'persisted_run.telemetry', 'Persisted telemetry is invalid.'); + } + record.telemetry = { ...TELEMETRY_DEFAULTS, ...(record.telemetry ?? {}) }; + for (const key of ['attention_count', 'attention_deduplicated_count', 'cancel_attempts']) { + if (!Number.isSafeInteger(record.telemetry[key]) || record.telemetry[key] < 0) { + admissionError('durable_state_mismatch', `persisted_run.telemetry.${key}`, 'Persisted telemetry counter is invalid.'); + } + } + if (!['not_dispatched', 'authoritative', 'partially_authoritative', 'uncertain'] + .includes(record.telemetry.dispatch_confidence)) { + admissionError('durable_state_mismatch', 'persisted_run.telemetry.dispatch_confidence', 'Persisted dispatch telemetry is invalid.'); + } + if (typeof record.telemetry.response_truncated !== 'boolean') { + admissionError('durable_state_mismatch', 'persisted_run.telemetry.response_truncated', 'Persisted response telemetry is invalid.'); + } + return record; +} + +function laneStatus(phase) { + if (phase === 'prompt_dispatched' || phase === 'running') return 'running'; + if (phase === 'needs_attention') return 'needs_attention'; + if (phase === 'completed') return 'completed'; + if (phase === 'cancelled') return 'cancelled'; + if (phase === 'failed_pre_prompt') return 'failed_pre_prompt'; + if (phase === 'partial_handoff') return 'partial_handoff'; + if (phase === 'unrecoverable_post_prompt') return 'unrecoverable_post_prompt'; + return phase; +} + +function isTerminalLane(lane) { + if (lane.phase === 'partial_handoff' || lane.phase === 'unrecoverable_post_prompt') { + return ['post_prompt_failure_no_replay', 'post_prompt_environment_blocked_no_replay', + 'timed_out_with_partial_work', 'timed_out_no_changes', + 'terminal_dispatch_uncertain_no_replay'].includes(lane.recovery_classification); + } + return capturedIncludes(LANE_TERMINAL_PHASES, lane.phase); +} + +function isTerminalRun(record) { + return capturedIncludes(RUN_TERMINAL_PHASES, record.phase); +} + +function authoritativeRequiredDispatch(record) { + return record.lanes + .filter((lane) => lane.required !== false) + .every((lane) => lane.prompt_dispatched === true && lane.dispatch_confidence === 'authoritative'); +} + + +function allLanesTerminal(record) { + return record.lanes.length > 0 && record.lanes.every(isTerminalLane); +} + +function laneNeedsObservation(lane) { + return lane.prompt_attempted === true && !isTerminalLane(lane); +} + +function laneHasPendingDispatchEvidence(lane) { + return lane.prompt_attempted === true + && lane.prompt_dispatched !== true + && lane.dispatch_confidence === 'uncertain' + && ['session_ready', 'running'].includes(lane.phase) + && ['dispatch_pending_no_replay', 'post_prompt_session_reconnected'].includes(lane.recovery_classification) + && !isTerminalLane(lane); +} + +function completeCandidateBlocked(record) { + return record.phase !== 'completed' + || record.lanes.some((lane) => !isTerminalLane(lane) + || (lane.required !== false && lane.phase !== 'completed')); +} + +function hasPromptEvidence(record) { + return record.lanes.some((lane) => lane.prompt_attempted === true + || lane.prompt_dispatched === true + || lane.dispatch_confidence === 'uncertain'); +} + +function changedFiles(value) { + if (!capturedIsArray(value)) return []; + return value.filter((entry) => typeof entry === 'string').slice(0, RUN_ADMISSION_CAPS.max_changed_files); +} + +function commitList(value) { + if (!capturedIsArray(value)) return []; + return value.filter((entry) => typeof entry === 'string').slice(0, RUN_ADMISSION_CAPS.max_commits); +} + +function observedResult(response) { + if (response === null || typeof response !== 'object' || capturedIsArray(response)) { + return { present: false, value: null }; + } + if (capturedHasOwn(response, 'result')) return { present: true, value: response.result }; + if (capturedHasOwn(response, 'provider_result')) return { present: true, value: response.provider_result }; + return { present: false, value: null }; +} + +function handoffFallback(record, lane, workspace = null, inspection = null) { + const worktree = workspace?.worktree_path + ?? workspace?.path + ?? inspection?.worktree_path + ?? inspection?.worktree + ?? null; + const retainedWorktree = typeof worktree === 'string' && worktree.length > 0; + return { + schema: 'codex-co-engineer.partial-handoff.v1', + assignment_id: lane.assignment_id, + worktree: retainedWorktree ? worktree : null, + branch: workspace?.branch ?? null, + starting_sha: workspace?.start_sha ?? workspace?.starting_sha ?? record.compiled.git?.base_sha ?? null, + current_head: inspection?.current_head ?? inspection?.head_sha ?? workspace?.current_head ?? null, + clean: typeof inspection?.clean === 'boolean' ? inspection.clean : null, + changed_files: changedFiles(inspection?.changed_files), + commits: commitList(inspection?.commits), + no_commit: !(commitList(inspection?.commits).length > 0), + partial_diff: inspection?.partial_diff === true || changedFiles(inspection?.changed_files).length > 0, + last_acknowledged_provider_event: inspection?.last_acknowledged_provider_event ?? null, + recovery_classification: lane.recovery_classification ?? 'no_recovery_needed', + safe_next_actions: retainedWorktree + ? ['Review the retained worktree and handoff evidence.'] + : ['Review the run receipt and provider outcome.'], + }; +} + +function boundedHandoff(value, fallback) { + const candidate = value && typeof value === 'object' ? { ...fallback, ...value } : fallback; + const retainedWorktree = typeof fallback?.worktree === 'string' && fallback.worktree.length > 0; + if (!retainedWorktree) candidate.worktree = null; + candidate.changed_files = changedFiles(candidate.changed_files); + candidate.commits = commitList(candidate.commits); + candidate.no_commit = candidate.no_commit === true || candidate.commits.length === 0; + candidate.partial_diff = candidate.partial_diff === true; + candidate.safe_next_actions = capturedIsArray(candidate.safe_next_actions) + ? candidate.safe_next_actions.filter((entry) => typeof entry === 'string').slice(0, RUN_ADMISSION_CAPS.max_next_actions) + : fallback.safe_next_actions; + if (!retainedWorktree) { + // The fallback is the only authoritative description of what remains + // when admission stopped before a workspace was retained. + candidate.safe_next_actions = fallback.safe_next_actions; + } + const text = JSON.stringify(candidate); + if (Buffer.byteLength(text, 'utf8') <= RUN_ADMISSION_CAPS.max_handoff_bytes) return freezeData(candidate); + candidate.changed_files = candidate.changed_files.slice(0, 16); + candidate.commits = candidate.commits.slice(0, 16); + candidate.safe_next_actions = candidate.safe_next_actions.slice(0, 3); + return freezeData(candidate); +} + +function laneReceipt(lane) { + return { + assignment_id: lane.assignment_id, + task_id: lane.task_id, + provider: lane.provider, + model: lane.model, + role: lane.role, + access: lane.access, + required: lane.required, + phase: lane.phase, + status: laneStatus(lane.phase), + prepared: lane.prepared === true, + session_ready: lane.session_ready === true, + prompt_attempted: lane.prompt_attempted === true, + prompt_dispatched: lane.prompt_dispatched === true, + dispatch_confidence: lane.dispatch_confidence, + session_id: lane.session_id ?? null, + cursor: lane.cursor ?? null, + last_event: lane.last_event ?? null, + result: lane.result ?? null, + result_truncated: lane.result_truncated === true, + cancel_confirmed: lane.cancel_confirmed ?? null, + task_final: isTerminalLane(lane), + child_identity_digest: lane.child_identity?.digest ?? null, + dispatch_identity_digest: lane.dispatch_identity?.digest ?? null, + provider_run_identity_digest: lane.provider_run_identity?.digest ?? null, + workspace_identity_digest: lane.workspace_identity?.digest ?? null, + workspace_identity: lane.workspace_identity ?? null, + error: lane.error ?? null, + recovery_classification: lane.recovery_classification ?? null, + handoff: lane.handoff ?? null, + }; +} + +function receipt(record, extras = {}) { + const dispatched = record.lanes.filter((lane) => lane.prompt_dispatched === true).map((lane) => lane.assignment_id); + const undispatched = record.lanes.filter((lane) => lane.prompt_attempted !== true).map((lane) => lane.assignment_id); + const uncertain = record.lanes + .filter((lane) => lane.dispatch_confidence === 'uncertain') + .map((lane) => lane.assignment_id); + return freezeData({ + schema: RUN_ADMISSION_SCHEMA_ID, + version: RUN_ADMISSION_VERSION, + run_id: record.run_id, + phase: record.phase, + status: record.phase, + revision: record.revision, + cursor: String(record.revision), + objective: record.compiled.objective, + base_sha: record.compiled.git.base_sha, + git: { + base_sha: record.compiled.git.base_sha, + digest: record.compiled.git_identity.digest, + }, + assignment_count: record.lanes.length, + lanes: record.lanes.map(laneReceipt), + consent: record.consent_request + ? { status: record.consent_status, request: record.consent_request } + : { + status: record.consent_status, + duration: record.consent_binding?.duration ?? 'this_run_only', + source: record.consent_binding?.source ?? null, + }, + admission: record.admission, + dispatched_assignment_ids: dispatched, + undispatched_assignment_ids: undispatched, + dispatch_uncertain_assignment_ids: uncertain, + authoritative_required_dispatch: authoritativeRequiredDispatch(record), + cancel_requested: record.cancel_requested === true, + already_terminal: extras.already_terminal === true, + error: record.error ?? null, + attention: record.attention ?? null, + handoff: record.handoff ?? null, + complete_candidate_blocked: completeCandidateBlocked(record), + // Never hand the mutable internal telemetry object to freezeData: receipts + // are immutable snapshots, while later cancellation/reconciliation still + // needs to update the record's counters. + telemetry: { ...record.telemetry }, + ...extras, + }); +} + +function defaultSleep(milliseconds, signal) { + return new Promise((resolve) => { + if (signal?.aborted) { + resolve(); + return; + } + const timer = setTimeout(resolve, milliseconds); + signal?.addEventListener('abort', () => { + clearTimeout(timer); + resolve(); + }, { once: true }); + }); +} + +function serializeConsentResponse(value) { + if (value === true) return { approved: true }; + if (!value || typeof value !== 'object') return { approved: false }; + return { + approved: value.approved === true || value.status === 'approved', + expires_at: typeof value.expires_at === 'string' ? value.expires_at : null, + approved_at: typeof value.approved_at === 'string' ? value.approved_at : null, + }; +} + +function consentErrorCode(value, fallback = 'consent_response_invalid') { + const candidates = [ + value?.code, + value?.reason, + value?.error?.code, + value?.error?.reason, + ]; + for (const candidate of candidates) { + if (SAFE_CONSENT_ERROR_CODES.includes(candidate)) return candidate; + } + return fallback; +} + +function consentRequestValue(compiled, value) { + if (value === undefined || value === null) return publicConsentRequest(compiled); + try { + assertNotProxy(value, 'consent.request'); + assertPlainObject(value, 'invalid_type', 'consent.request', 'Consent request'); + assertDirectJsonClosure(value, 'consent.request'); + return freezeData(value); + } catch { + return publicConsentRequest(compiled); + } +} + +function isConsentApproved(value) { + return value?.approved === true || value?.status === 'approved'; +} + +function isConsentPending(value) { + return value?.status === 'required' + || value?.status === 'pending' + || value?.status === 'awaiting_consent'; +} + +function validConsentWindow(consent, clock) { + if (consent?.approved !== true + || typeof consent.approved_at !== 'string' + || typeof consent.expires_at !== 'string') return false; + const approvedAt = Date.parse(consent.approved_at); + const expiresAt = Date.parse(consent.expires_at); + const now = Date.parse(nowIso(clock)); + if (!Number.isFinite(approvedAt) || !Number.isFinite(expiresAt)) return false; + const current = Number.isFinite(now) ? now : Date.now(); + return approvedAt <= current && expiresAt > current && expiresAt > approvedAt; +} + +function createDefaultDependencies(overrides) { + const dependencies = { + requestConsent: async (compiled) => ({ status: 'required', request: publicConsentRequest(compiled) }), + verifyConsent: async () => ({ approved: false }), + providerReady: async () => ({ ready: true }), + processBoundaryReady: async () => ({ ready: true }), + verifyRepository: async () => ({ verified: true }), + prepareWorkspace: async () => ({ prepared: true, workspace: null }), + cleanupWorkspace: async () => ({ cleaned: true }), + createSession: async ({ assignment }) => ({ ready: true, session_id: `${assignment.task_id}-session` }), + dispatchPrompt: async () => ({ dispatched: true, confidence: 'authoritative' }), + inspectLane: async () => ({ status: 'running', cursor: '0' }), + reconnectLane: async () => ({ reconnected: true }), + replyAttention: async () => ({ delivered: true }), + cancelLane: async () => ({ confirmed: true, cancelled: true }), + inspectWorkspace: async () => ({}), + buildHandoff: async ({ fallback }) => fallback, + verifyRun: async () => ({ verified: true }), + clock: () => new Date().toISOString(), + sleep: defaultSleep, + waitForProgress: async ({ wait_ms, signal }) => defaultSleep(wait_ms, signal), + compile: compileRunRequestV1, + loadRecord: async () => null, + persistRecord: async () => {}, + }; + for (const key of RUN_ADMISSION_DEPENDENCIES) { + if (capturedHasOwn(overrides ?? {}, key)) { + if (typeof overrides[key] !== 'function') admissionError('injected_dependency_invalid', `dependencies.${key}`); + dependencies[key] = overrides[key]; + } + } + return dependencies; +} + +export function createRunAdmissionRuntime(overrides = {}) { + assertNotProxy(overrides, 'dependencies'); + assertPlainObject(overrides, 'injected_dependency_invalid', 'dependencies', 'Run admission dependencies'); + assertDirectJsonClosure(Object.fromEntries( + capturedOwnKeys(overrides).filter((key) => typeof key === 'string').map((key) => [key, null]), + ), 'dependencies'); + for (const key of capturedOwnKeys(overrides)) { + if (typeof key !== 'string' || !capturedIncludes(RUN_ADMISSION_DEPENDENCIES, key)) { + admissionError('unknown_key', `dependencies.${String(key)}`, 'Run admission dependencies are closed.'); + } + } + const injected = createDefaultDependencies(overrides); + const records = new Map(); + const chains = new Map(); + + async function loadRecord(runId) { + const existing = records.get(runId); + if (existing) return existing; + const loaded = await injected.loadRecord(runId); + if (loaded === null || loaded === undefined) return null; + try { + validatePersistedRecord(loaded, runId); + ensureTelemetry(loaded); + } catch (error) { + if (error instanceof RunContractV1Error) throw error; + admissionError('durable_state_mismatch', 'persisted_run', 'Persisted run state is invalid.'); + } + records.set(runId, loaded); + return loaded; + } + + async function persist(record) { + try { + await injected.persistRecord(record); + } catch { + admissionError('run_persistence_failed', 'run_id', 'The run state could not be durably persisted.'); + } + } + + function enqueue(runId, operation) { + const previous = chains.get(runId) ?? Promise.resolve(); + const current = previous.catch(() => {}).then(operation); + chains.set(runId, current); + current.finally(() => { + if (chains.get(runId) === current) chains.delete(runId); + }).catch(() => {}); + return current; + } + + function bump(record) { + record.revision += 1; + record.updated_at = nowIso(injected.clock); + } + + function makeRecord(compiled) { + return { + schema: RUN_ADMISSION_SCHEMA_ID, + version: RUN_ADMISSION_VERSION, + run_id: compiled.run_id, + compiled, + phase: 'validating', + revision: 0, + created_at: nowIso(injected.clock), + updated_at: nowIso(injected.clock), + consent_status: 'pending', + consent_request: null, + consent_binding: bindingForConsent(compiled), + admission: null, + error: null, + attention: null, + attention_questions: [], + handoff: null, + cancel_requested: false, + telemetry: { ...TELEMETRY_DEFAULTS }, + lanes: compiled.assignments.map((assignment) => ({ + assignment_id: assignment.assignment_id, + task_id: assignment.task_id, + provider: assignment.provider, + model: assignment.model, + role: assignment.role, + access: assignment.access, + required: assignment.required !== false, + child_identity: assignment.child_identity, + dispatch_identity: assignment.dispatch_identity, + provider_run_identity: assignment.provider_run_identity, + phase: 'planned', + prepared: false, + workspace: null, + session_ready: false, + session_id: null, + prompt_attempted: false, + prompt_dispatched: false, + dispatch_confidence: 'not_sent', + cursor: null, + error: null, + recovery_classification: null, + handoff: null, + workspace_identity: null, + last_event: null, + attention_satisfied_keys: [], + result: null, + result_truncated: false, + cancel_confirmed: null, + })), + }; + } + + async function finishLane(record, lane, inspection = null) { + // A handoff may describe partial work while provider termination is still unknown. + if (!capturedIncludes(LANE_TERMINAL_PHASES, lane.phase)) return; + let workspaceInspection = inspection; + if (workspaceInspection === null) { + try { + workspaceInspection = await injected.inspectWorkspace({ + run_id: record.run_id, + assignment_id: lane.assignment_id, + task_id: lane.task_id, + workspace: lane.workspace, + }); + } catch { + workspaceInspection = {}; + } + } + if (lane.error?.code === 'timeout') { + const hasPartialWork = workspaceInspection?.partial_diff === true + || changedFiles(workspaceInspection?.changed_files).length > 0 + || commitList(workspaceInspection?.commits).length > 0; + lane.recovery_classification = hasPartialWork + ? 'timed_out_with_partial_work' + : 'timed_out_no_changes'; + } + const fallback = handoffFallback(record, lane, lane.workspace, workspaceInspection); + try { + lane.handoff = boundedHandoff(await injected.buildHandoff({ + run_id: record.run_id, + assignment: record.compiled.assignments.find((assignment) => assignment.assignment_id === lane.assignment_id), + lane, + workspace: lane.workspace, + inspection: workspaceInspection, + fallback, + }), fallback); + } catch { + lane.handoff = boundedHandoff(null, fallback); + } + record.telemetry.handoff_class = lane.phase; + if (record.telemetry.time_to_terminal_handoff_ms === null) { + const createdAt = Date.parse(record.created_at); + if (Number.isFinite(createdAt)) { + record.telemetry.time_to_terminal_handoff_ms = Math.max(0, Date.now() - createdAt); + } + } + } + + async function failAdmission(record, stage, error, startedAt = null) { + record.phase = 'failed'; + record.error = cloneError(errorSummary(error, `admission_${stage}_failed`)); + record.telemetry.admission_failure_stage = stage; + if (Number.isFinite(startedAt)) record.telemetry.admission_duration_ms = Math.max(0, Date.now() - startedAt); + for (const lane of record.lanes) { + if (lane.phase !== 'completed' && lane.phase !== 'cancelled') { + lane.phase = 'failed_pre_prompt'; + lane.error = record.error; + } + await finishLane(record, lane); + } + bump(record); + await persist(record); + return receipt(record); + } + + async function requestConsent(record, options = {}) { + const signal = options?.signal; + // Establish the pending state before entering a host callback. Native + // form callbacks may wait for a user decision or outlive this MCP call; + // the durable record must therefore expose the exact request first. + record.consent_request = publicConsentRequest(record.compiled); + record.consent_status = 'required'; + record.phase = 'awaiting_consent'; + record.error = null; + bump(record); + await persist(record); + + if (signal?.aborted) { + record.consent_status = 'blocked'; + record.error = cloneError({ code: 'consent_request_aborted' }); + bump(record); + await persist(record); + return false; + } + + let response; + try { + response = await injected.requestConsent(record.compiled, { signal }); + } catch (error) { + const code = signal?.aborted + ? 'consent_request_aborted' + : consentErrorCode(error, 'consent_host_unavailable'); + record.consent_status = 'blocked'; + record.error = cloneError({ code }); + // A transport/host failure leaves the run reopenable. An abort is also + // request scoped, so both retain the pending consent request. + bump(record); + await persist(record); + return false; + } + + // A callback can resolve with approval after its owning MCP request was + // cancelled. Never cross the admission barrier after that cancellation. + if (signal?.aborted) { + record.consent_status = 'blocked'; + record.error = cloneError({ code: 'consent_request_aborted' }); + bump(record); + await persist(record); + return false; + } + + if (isConsentApproved(response)) { + record.consent_status = 'approved'; + record.consent_request = null; + record.consent_binding = freezeData({ + ...record.consent_binding, + duration: response?.duration === 'repository_and_selected_providers' + ? 'repository_and_selected_providers' + : 'this_run_only', + source: response?.source === 'durable_grant' ? 'durable_grant' : 'native_form', + }); + record.error = null; + return true; + } + if (isConsentPending(response)) { + record.consent_request = consentRequestValue(record.compiled, response?.request); + record.consent_status = 'required'; + record.phase = 'awaiting_consent'; + const code = consentErrorCode(response, null); + record.error = code === null ? null : cloneError({ code }); + bump(record); + await persist(record); + return false; + } + + const code = consentErrorCode(response); + record.consent_status = 'blocked'; + record.error = cloneError({ code }); + if (RETRYABLE_CONSENT_ERROR_CODES.includes(code)) { + // Dismissal, timeout, and request-scoped abort leave the same run + // reopenable. Preserve the blocked reason while retaining the pending + // public request for an explicit native retry. + record.phase = 'awaiting_consent'; + bump(record); + await persist(record); + return false; + } + // Explicit decline/host responses are terminal admission failures. Keep + // their safe reason code on the authoritative receipt. + await failAdmission(record, 'consent', { code }); + return false; + } + + async function admit(record) { + const started = Date.now(); + record.phase = 'preparing_workspaces'; + bump(record); + await persist(record); + let readiness; + try { + const providerChecks = await Promise.all(record.lanes.map(async (lane) => { + const result = await injected.providerReady({ + run_id: record.run_id, + assignment: record.compiled.assignments.find((assignment) => assignment.assignment_id === lane.assignment_id), + }); + return result?.ready === true; + })); + // Cursor Cloud owns its remote process boundary. Requiring the local + // systemd/cgroup boundary for an all-Cloud run incorrectly blocks a + // valid dispatch on hosts where only the Cloud provider is available. + // Mixed and local-only runs still fail closed on the local boundary. + const needsLocalBoundary = record.lanes.some((lane) => lane.provider !== 'cursor-cloud'); + const boundary = needsLocalBoundary + ? await injected.processBoundaryReady({ run_id: record.run_id }) + : { ready: true }; + const repository = await injected.verifyRepository({ + run_id: record.run_id, + git: record.compiled.git, + repository_identity: record.compiled.git_identity, + }); + readiness = providerChecks.every(Boolean) && boundary?.ready === true && repository?.verified === true; + if (!readiness) return failAdmission(record, 'readiness', { code: 'admission_not_ready' }, started); + } catch (error) { + return failAdmission(record, 'readiness', error, started); + } + const preparedAt = Date.now(); + const prepared = await Promise.all(record.lanes.map(async (lane) => { + try { + const assignment = record.compiled.assignments.find((entry) => entry.assignment_id === lane.assignment_id); + const result = await injected.prepareWorkspace({ + run_id: record.run_id, + assignment, + git: record.compiled.git, + managed_workspace_policy: record.compiled.managed_workspace_policy, + }); + if (result?.prepared !== true) throw Object.assign(new Error('workspace not ready'), { code: 'workspace_not_ready' }); + return { lane, result }; + } catch (error) { + return { lane, error }; + } + })); + record.telemetry.workspace_preparation_duration_ms = Date.now() - preparedAt; + const failed = prepared.find((entry) => entry.error || entry.result?.prepared !== true); + if (failed) { + for (const entry of prepared) { + if (!entry.error && entry.result?.workspace) { + await injected.cleanupWorkspace({ run_id: record.run_id, assignment_id: entry.lane.assignment_id, workspace: entry.result.workspace }).catch(() => {}); + } + } + return failAdmission(record, 'workspace_preparation', failed.error ?? { code: 'workspace_not_ready' }, started); + } + for (const entry of prepared) { + entry.lane.prepared = true; + entry.lane.workspace = entry.result.workspace ?? null; + entry.lane.phase = 'prepared'; + } + record.admission = freezeData({ + status: 'admitted', + stage: 'all_lanes_prepared', + duration_ms: Date.now() - started, + prompt_count_before_dispatch: 0, + }); + record.telemetry.admission_duration_ms = Date.now() - started; + bump(record); + await persist(record); + return dispatch(record); + } + + async function dispatch(record) { + record.phase = 'dispatching'; + const started = Date.now(); + bump(record); + await persist(record); + for (let index = 0; index < record.lanes.length; index += 1) { + const lane = record.lanes[index]; + if (record.cancel_requested) { + lane.phase = 'cancelled'; + lane.error = cloneError({ code: 'cancel_requested' }); + await finishLane(record, lane); + continue; + } + const assignment = record.compiled.assignments[index]; + try { + const session = await injected.createSession({ + run_id: record.run_id, + assignment, + lane, + workspace: lane.workspace, + git: record.compiled.git, + }); + if (session?.ready !== true) throw Object.assign(new Error('provider session not ready'), { code: 'session_not_ready' }); + lane.session_ready = true; + lane.session_id = typeof session.session_id === 'string' ? session.session_id : null; + lane.phase = 'session_ready'; + if (record.telemetry.time_to_session_ms === null) { + const createdAt = Date.parse(record.created_at); + if (Number.isFinite(createdAt)) record.telemetry.time_to_session_ms = Math.max(0, Date.now() - createdAt); + } + bump(record); + await persist(record); + const result = await injected.dispatchPrompt({ + run_id: record.run_id, + assignment, + lane, + workspace: lane.workspace, + session, + git: record.compiled.git, + managed_workspace_policy: record.compiled.managed_workspace_policy, + prompt: assignment.prompt, + attempt: 1, + }); + const confidence = result?.confidence === 'uncertain' || result?.dispatch_uncertain === true + ? 'uncertain' : 'authoritative'; + if (result?.session_ready === true || typeof result?.session_id === 'string') { + lane.session_ready = true; + if (typeof result.session_id === 'string') lane.session_id = result.session_id; + } + let workspaceIdentityInvalid = false; + if (result?.workspace_identity !== undefined) { + try { + lane.workspace_identity = validateWorkspaceIdentityV1( + result.workspace_identity, + `dispatch.${lane.assignment_id}.workspace_identity`, + ); + } catch (error) { + lane.workspace_identity = null; + workspaceIdentityInvalid = true; + } + } + if (workspaceIdentityInvalid) { + const promptEvidence = result?.dispatched === true || result?.prompt_dispatched === true + || result?.sent === true || confidence === 'uncertain'; + lane.prompt_attempted = promptEvidence; + lane.prompt_dispatched = result?.dispatched === true || result?.prompt_dispatched === true; + lane.dispatch_confidence = confidence; + lane.phase = promptEvidence ? 'unrecoverable_post_prompt' : 'failed_pre_prompt'; + lane.recovery_classification = promptEvidence + ? 'workspace_identity_invalid_no_replay' + : 'workspace_identity_invalid_pre_prompt'; + lane.error = cloneError({ code: 'workspace_identity_invalid' }); + await finishLane(record, lane); + record.phase = promptEvidence ? 'degraded' : (hasPromptEvidence(record) ? 'degraded' : 'failed'); + await persist(record); + break; + } + if (result?.dispatched !== true && result?.prompt_dispatched !== true) { + if (confidence === 'uncertain' || result?.sent === true) { + const pending = result?.dispatch_pending === true && result?.terminal !== true; + lane.prompt_attempted = true; + lane.dispatch_confidence = 'uncertain'; + lane.phase = pending ? 'session_ready' : 'unrecoverable_post_prompt'; + lane.recovery_classification = result?.terminal === true + ? 'terminal_dispatch_uncertain_no_replay' + : pending ? 'dispatch_pending_no_replay' : 'dispatch_uncertain_no_replay'; + lane.error = pending ? null : terminalLaneError(result, 'dispatch_uncertain'); + if (!pending) await finishLane(record, lane); + await persist(record); + if (pending) continue; + record.phase = 'degraded'; + break; + } + lane.phase = 'failed_pre_prompt'; + lane.error = cloneError({ code: 'prompt_dispatch_failed' }); + await finishLane(record, lane); + record.phase = hasPromptEvidence(record) ? 'degraded' : 'failed'; + await persist(record); + break; + } + if (confidence === 'uncertain') { + // The provider may have accepted the prompt before its durable + // acknowledgement became observable. Keep observing this same task; + // never convert the bounded acknowledgement wait into a replay. + const pending = result?.dispatch_pending === true && result?.terminal !== true; + lane.prompt_attempted = true; + lane.dispatch_confidence = 'uncertain'; + lane.phase = pending ? 'session_ready' : 'unrecoverable_post_prompt'; + lane.recovery_classification = result?.terminal === true + ? 'terminal_dispatch_uncertain_no_replay' + : pending ? 'dispatch_pending_no_replay' : 'dispatch_uncertain_no_replay'; + lane.error = pending ? null : terminalLaneError(result, 'dispatch_uncertain'); + if (!pending) await finishLane(record, lane); + await persist(record); + if (pending) continue; + record.phase = 'degraded'; + break; + } + lane.prompt_attempted = true; + lane.prompt_dispatched = true; + lane.dispatch_confidence = confidence; + lane.cursor = typeof result.cursor === 'string' ? result.cursor : '0'; + lane.phase = 'prompt_dispatched'; + lane.last_event = 'prompt_dispatched'; + if (record.telemetry.time_to_prompt_dispatch_ms === null) { + const createdAt = Date.parse(record.created_at); + if (Number.isFinite(createdAt)) record.telemetry.time_to_prompt_dispatch_ms = Math.max(0, Date.now() - createdAt); + } + bump(record); + await persist(record); + } catch (error) { + const uncertain = error?.dispatch_uncertain === true || error?.sent === true || error?.code === 'dispatch_uncertain'; + if (uncertain) { + lane.prompt_attempted = true; + lane.dispatch_confidence = 'uncertain'; + lane.phase = 'unrecoverable_post_prompt'; + lane.recovery_classification = error?.terminal === true + ? 'terminal_dispatch_uncertain_no_replay' + : 'dispatch_uncertain_no_replay'; + lane.error = cloneError({ code: 'dispatch_uncertain' }); + } else { + lane.phase = 'failed_pre_prompt'; + lane.error = cloneError(errorSummary(error, 'prompt_dispatch_failed')); + } + await finishLane(record, lane); + record.phase = hasPromptEvidence(record) ? 'degraded' : 'failed'; + await persist(record); + break; + } + } + for (const lane of record.lanes) { + if (lane.phase === 'prepared' || lane.phase === 'planned') { + lane.phase = 'failed_pre_prompt'; + lane.error = cloneError({ code: 'dispatch_not_attempted' }); + await finishLane(record, lane); + } + } + record.telemetry.dispatch_duration_ms = Date.now() - started; + record.telemetry.dispatch_confidence = record.lanes.some((lane) => lane.dispatch_confidence === 'uncertain') + ? 'uncertain' + : authoritativeRequiredDispatch(record) + ? 'authoritative' + : record.lanes.some((lane) => lane.prompt_dispatched === true) + ? 'partially_authoritative' + : 'not_dispatched'; + const recovery = record.lanes.map((lane) => lane.recovery_classification).find(Boolean); + if (recovery) record.telemetry.recovery_path = recovery; + if (record.phase === 'dispatching') { + record.phase = authoritativeRequiredDispatch(record) + ? 'running' + : record.lanes.some(laneHasPendingDispatchEvidence) + ? 'dispatching' + : (hasPromptEvidence(record) ? 'degraded' : 'failed'); + } + bump(record); + await persist(record); + return receipt(record); + } + + async function reconcile(record) { + const before = JSON.stringify(record); + // Inspection is a read operation and must always return the authoritative + // snapshot, including durable terminal and consent-pending records. An + // undefined result here makes waiters lose their cursor/phase and lets the + // adapter manufacture an empty success receipt. + if (isTerminalRun(record) || record.phase === 'awaiting_consent') return receipt(record); + const active = record.lanes.filter(laneNeedsObservation); + for (const lane of active) { + try { + const assignment = record.compiled.assignments.find((entry) => entry.assignment_id === lane.assignment_id); + const response = await injected.inspectLane({ + run_id: record.run_id, + assignment_id: lane.assignment_id, + task_id: lane.task_id, + assignment, + lane, + }); + const status = response?.status ?? response?.phase; + if (!KNOWN_LANE_OBSERVATION_STATUSES.includes(status)) { + throw Object.assign(new Error('Supervisor returned an invalid lane observation.'), { + code: 'lane_observation_invalid', + }); + } + if (response?.task_id !== undefined && response.task_id !== lane.task_id) { + throw Object.assign(new Error('Supervisor lane observation identity did not match the requested task.'), { + code: 'lane_observation_identity_mismatch', + }); + } + const authoritativeDispatchEvidence = response?.dispatch_evidence === 'authoritative' + && response?.prompt_dispatched === true; + if (authoritativeDispatchEvidence && lane.prompt_dispatched !== true) { + lane.prompt_attempted = true; + lane.prompt_dispatched = true; + lane.dispatch_confidence = 'authoritative'; + if (typeof response.session_id === 'string') lane.session_id = response.session_id; + if (['dispatch_pending_no_replay', 'dispatch_uncertain_no_replay'].includes(lane.recovery_classification)) { + lane.recovery_classification = null; + if (lane.error?.code === 'dispatch_uncertain') lane.error = null; + } + if (lane.phase === 'session_ready' || lane.phase === 'unrecoverable_post_prompt') { + lane.phase = 'prompt_dispatched'; + } + if (record.telemetry.time_to_prompt_dispatch_ms === null) { + const createdAt = Date.parse(record.created_at); + if (Number.isFinite(createdAt)) record.telemetry.time_to_prompt_dispatch_ms = Math.max(0, Date.now() - createdAt); + } + } + const previousLastEvent = lane.last_event; + const recoveringObservation = lane.recovery_classification === OBSERVATION_RECOVERY_CLASSIFICATION; + if (recoveringObservation && !['failed', 'timeout', 'environment_blocked'].includes(status)) { + lane.phase = lane.phase === 'partial_handoff' ? 'running' : lane.phase; + lane.recovery_classification = null; + lane.error = null; + } + const eventChanged = typeof response?.last_event === 'string' + && response.last_event !== previousLastEvent; + if (typeof response?.cursor === 'string' && CURSOR_PATTERN.test(response.cursor)) lane.cursor = response.cursor; + if (typeof response?.last_event === 'string' && eventChanged) { + lane.last_event = response.last_event; + if (record.telemetry.time_to_first_event_ms === null) { + const createdAt = Date.parse(record.created_at); + if (Number.isFinite(createdAt)) record.telemetry.time_to_first_event_ms = Math.max(0, Date.now() - createdAt); + } + record.telemetry.last_meaningful_activity_at = nowIso(injected.clock); + } + if (Number.isSafeInteger(response?.silence_duration_ms) && response.silence_duration_ms >= 0 + && (eventChanged || status === 'needs_attention' + || ['completed', 'succeeded', 'cancelled', 'failed', 'timeout'].includes(status))) { + record.telemetry.silence_duration_ms = response.silence_duration_ms; + } + const result = observedResult(response); + if (result.present) { + const bounded = boundProviderResult(result.value, MAX_PROVIDER_RESULT_BYTES); + lane.result = bounded.value; + lane.result_truncated = bounded.truncated; + record.telemetry.response_truncated = record.telemetry.response_truncated || bounded.truncated; + } + if (status === 'needs_attention') { + const attention = normalizeAttention(response.attention); + if (!attention || attention.invalid === true) { + lane.phase = 'partial_handoff'; + lane.recovery_classification = 'attention_evidence_invalid_no_replay'; + lane.error = cloneError({ code: 'attention_evidence_invalid' }); + await finishLane(record, lane, response.workspace_inspection ?? null); + record.phase = 'degraded'; + continue; + } + const key = capabilityQuestionKey(attention); + const dedupKey = attentionDedupKey(attention); + const targetKey = JSON.stringify(attentionTarget(lane, attention)); + const alreadyOpen = record.attention?.status === 'open' + && record.attention_questions.some((entry) => entry.attention_key === dedupKey + && entry.targets?.some((target) => JSON.stringify(target) === targetKey)); + if (alreadyOpen) { + lane.phase = 'needs_attention'; + continue; + } + if (attentionWasSatisfied(lane, key)) { + lane.phase = 'running'; + continue; + } + record.telemetry.attention_count += 1; + if (dedupKey && record.attention_questions.some((entry) => entry.attention_key === dedupKey)) { + record.telemetry.attention_deduplicated_count += 1; + } + if (isCapabilityCovered(assignment, attention)) { + if (attentionWasSatisfied(lane, key)) { + lane.phase = 'running'; + continue; + } + let delivered = null; + try { + delivered = await injected.replyAttention({ + run_id: record.run_id, + task_id: lane.task_id, + assignment, + lane, + attention, + capability_satisfied: true, + reply: { + session_id: attention.session_id ?? lane.session_id, + question_id: attention.question_id, + response: { + outcome: 'allow_once', + capability: attention.capability, + resource: attention.resource, + action: attention.action, + }, + }, + }); + } catch { + delivered = null; + } + if (delivered?.delivered !== false) { + lane.attention_satisfied_keys = [...new Set([ + ...(lane.attention_satisfied_keys ?? []), key, + ])].slice(-8); + lane.phase = 'running'; + lane.last_event = 'capability_satisfied'; + lane.recovery_classification = 'capability_pre_authorized'; + continue; + } + } + lane.phase = 'needs_attention'; + record.phase = 'needs_attention'; + record.attention = mergeAttention(record, lane, assignment, attention); + } else if (status === 'completed' || status === 'succeeded') { + if (lane.prompt_dispatched === true && lane.dispatch_confidence === 'authoritative') { + lane.phase = 'completed'; + } else { + lane.phase = 'unrecoverable_post_prompt'; + lane.recovery_classification = 'terminal_dispatch_uncertain_no_replay'; + lane.error = cloneError({ code: 'dispatch_uncertain' }); + } + await finishLane(record, lane, response.workspace_inspection ?? null); + } else if (status === 'cancelled') { + lane.phase = 'cancelled'; + lane.cancel_confirmed = true; + lane.error = null; + lane.recovery_classification = null; + await finishLane(record, lane, response.workspace_inspection ?? null); + } else if (status === 'transport_lost') { + let reconnected = null; + try { + reconnected = await injected.reconnectLane({ + run_id: record.run_id, + assignment_id: lane.assignment_id, + task_id: lane.task_id, + assignment, + lane, + response, + }); + } catch { + reconnected = null; + } + if (reconnected?.reconnected === true) { + lane.phase = 'running'; + lane.recovery_classification = 'post_prompt_session_reconnected'; + lane.error = null; + if (typeof reconnected.session_id === 'string') lane.session_id = reconnected.session_id; + if (typeof reconnected.cursor === 'string' && CURSOR_PATTERN.test(reconnected.cursor)) lane.cursor = reconnected.cursor; + lane.last_event = 'session_reconnected'; + record.telemetry.recovery_path = 'post_prompt_session_reconnected'; + } else { + lane.phase = 'running'; + lane.recovery_classification = 'post_prompt_session_reconnect_required'; + lane.error = cloneError({ code: 'transport_lost' }); + record.telemetry.recovery_path = 'post_prompt_session_reconnect_required'; + } + } else if (status === 'environment_blocked') { + if (laneHasPendingDispatchEvidence(lane)) { + lane.phase = 'unrecoverable_post_prompt'; + lane.recovery_classification = 'terminal_dispatch_uncertain_no_replay'; + lane.error = terminalLaneError(response, 'environment_blocked'); + await finishLane(record, lane, response.workspace_inspection ?? null); + continue; + } + lane.phase = lane.prompt_dispatched ? 'partial_handoff' : 'failed_pre_prompt'; + lane.recovery_classification = 'post_prompt_environment_blocked_no_replay'; + lane.error = cloneError({ code: 'environment_blocked' }); + await finishLane(record, lane, response.workspace_inspection ?? null); + } else if (status === 'failed' || status === 'timeout' || status === 'timed_out') { + if (laneHasPendingDispatchEvidence(lane)) { + lane.phase = 'unrecoverable_post_prompt'; + lane.recovery_classification = 'terminal_dispatch_uncertain_no_replay'; + lane.error = terminalLaneError(response, status === 'timed_out' ? 'timeout' : status); + await finishLane(record, lane, response.workspace_inspection ?? null); + continue; + } + lane.phase = lane.prompt_dispatched ? 'partial_handoff' : 'failed_pre_prompt'; + lane.recovery_classification = 'post_prompt_failure_no_replay'; + lane.error = terminalLaneError(response, status === 'timed_out' ? 'timeout' : status); + record.telemetry.recovery_path = 'post_prompt_failure_no_replay'; + await finishLane(record, lane, response.workspace_inspection ?? null); + } else if (lane.phase === 'prompt_dispatched') { + lane.phase = 'running'; + } + } catch (error) { + // A failed read is not proof that the provider stopped. Keep the + // canonical task ID and retry on a later inspect/wait/cancel request. + if (!isTerminalLane(lane) || lane.recovery_classification === OBSERVATION_RECOVERY_CLASSIFICATION) { + if (lane.recovery_classification === OBSERVATION_RECOVERY_CLASSIFICATION) { + // A legacy partial handoff is only terminal because its old + // inspection failed. Make it active again before retrying. + lane.phase = 'running'; + } else { + lane.phase = lane.phase === 'prompt_dispatched' ? 'running' : lane.phase; + } + lane.recovery_classification = OBSERVATION_RECOVERY_CLASSIFICATION; + lane.error = cloneError(errorSummary(error, 'lane_observation_unavailable')); + } + } + } + record.telemetry.dispatch_confidence = record.lanes.some(laneHasPendingDispatchEvidence) + ? 'uncertain' + : authoritativeRequiredDispatch(record) + ? 'authoritative' + : record.lanes.some((lane) => lane.prompt_dispatched === true) + ? 'partially_authoritative' + : 'not_dispatched'; + if (record.cancel_requested && record.lanes.every((lane) => lane.phase === 'cancelled' || isTerminalLane(lane))) { + record.phase = 'cancelled'; + record.telemetry.cancel_confirmed = record.lanes.every((lane) => lane.phase === 'cancelled' || lane.phase === 'completed'); + } else if (record.lanes.some((lane) => lane.phase === 'needs_attention')) { + record.phase = 'needs_attention'; + } else if (record.lanes.some((lane) => lane.phase === 'partial_handoff' || lane.phase === 'unrecoverable_post_prompt' + || (laneNeedsObservation(lane) && lane.error !== null))) { + record.phase = 'degraded'; + } else if (allLanesTerminal(record)) { + record.phase = 'verifying'; + try { + const verification = await injected.verifyRun({ run_id: record.run_id, compiled: record.compiled, lanes: record.lanes }); + record.phase = verification?.verified === true + && authoritativeRequiredDispatch(record) + && record.lanes.filter((lane) => lane.required !== false).every((lane) => lane.phase === 'completed') + ? 'completed' : 'failed'; + if (record.phase === 'failed') record.error = cloneError({ code: 'verification_failed' }); + } catch (error) { + record.phase = 'failed'; + record.error = cloneError(errorSummary(error, 'verification_failed')); + } + } else if (authoritativeRequiredDispatch(record)) { + record.phase = 'running'; + } else if (record.lanes.some(laneHasPendingDispatchEvidence)) { + record.phase = 'dispatching'; + } + const after = JSON.stringify(record); + if (after !== before) { + bump(record); + await persist(record); + } + return receipt(record); + } + + async function submitRunRequest(request, options = {}) { + const compiled = await injected.compile(request, options.compile_options ?? {}); + const { runId } = validateCompiled(compiled); + return enqueue(runId, async () => { + const existing = await loadRecord(runId); + if (existing) { + if (existing.compiled.request_idempotency_key !== compiled.request_idempotency_key) { + admissionError('run_identity_conflict', 'run_request.run_id', 'run_id is already bound to a different semantic request.'); + } + return receipt(existing, { idempotent: true }); + } + const record = makeRecord(compiled); + records.set(runId, record); + bump(record); + await persist(record); + const consented = await requestConsent(record, options); + await persist(record); + if (!consented) return receipt(record); + return admit(record); + }); + } + + async function inspectRun(request) { + const parsed = ownObject(request, 'request'); + assertKeys(parsed, ['run_id'], 'request'); + const runId = requiredString(parsed, 'run_id', 'run_id', (value) => RUN_ID_PATTERN.test(value)); + const record = await loadRecord(runId); + if (!record) admissionError('run_not_found', 'run_id', 'The requested run is not known to this server.'); + // A native consent callback may keep submitRunRequest queued while the + // durable record is already awaiting a decision. Read snapshots directly + // so status/wait remain usable during that host interaction. + if (isTerminalRun(record) || record.phase === 'awaiting_consent') return receipt(record); + return enqueue(runId, async () => reconcile(record)); + } + + async function resumeRun(request) { + const parsed = ownObject(request, 'request'); + assertKeys(parsed, ['run_id'], 'request'); + const runId = requiredString(parsed, 'run_id', 'run_id', (value) => RUN_ID_PATTERN.test(value)); + const record = await loadRecord(runId); + if (!record) admissionError('run_not_found', 'run_id', 'The requested run is not known to this server.'); + return enqueue(runId, async () => { + if (record.cancel_requested || record.phase === 'cancelled') return receipt(record); + if (record.phase === 'awaiting_consent') return receipt(record); + return reconcile(record); + }); + } + + async function replyRun(request, options = {}) { + const parsed = ownObject(request, 'request'); + assertKeys(parsed, ['run_id', 'approval_ref', 'attention_reply', 'request_consent'], 'request'); + const runId = requiredString(parsed, 'run_id', 'run_id', (value) => RUN_ID_PATTERN.test(value)); + const record = await loadRecord(runId); + if (!record) admissionError('run_not_found', 'run_id', 'The requested run is not known to this server.'); + return enqueue(runId, async () => { + const requestConsentAgain = capturedHasOwn(parsed, 'request_consent'); + if (requestConsentAgain && parsed.request_consent !== true) { + admissionError('invalid_format', 'request_consent', 'A consent continuation must be exactly true.'); + } + if (requestConsentAgain && (capturedHasOwn(parsed, 'approval_ref') + || capturedHasOwn(parsed, 'attention_reply'))) { + admissionError('mixed_run_operation', 'request', 'Consent continuation cannot be combined with another reply.'); + } + if (record.cancel_requested || record.phase === 'cancelled') return receipt(record); + const approvalRef = optionalString(parsed, 'approval_ref', 'approval_ref'); + if (record.phase === 'awaiting_consent') { + if (requestConsentAgain) { + const consented = await requestConsent(record, options); + if (!consented) return receipt(record); + return admit(record); + } + if (!approvalRef) { + record.error = cloneError({ code: 'approval_ref_required' }); + bump(record); + await persist(record); + return receipt(record); + } + let verified; + try { + verified = serializeConsentResponse(await injected.verifyConsent({ + approval_ref: approvalRef, + binding: record.consent_binding, + })); + } catch { + verified = { approved: false }; + } + if (!validConsentWindow(verified, injected.clock)) { + record.error = cloneError({ code: 'approval_ref_invalid_or_expired' }); + bump(record); + await persist(record); + return receipt(record); + } + record.consent_status = 'approved'; + record.consent_request = null; + record.consent_binding = freezeData({ + ...record.consent_binding, + approved_at: verified.approved_at, + expires_at: verified.expires_at, + }); + record.error = null; + return admit(record); + } + if (requestConsentAgain) return receipt(record); + if (capturedHasOwn(parsed, 'attention_reply')) { + let delivered = false; + try { + const response = await injected.replyAttention({ + run_id: runId, + reply: parsed.attention_reply, + attention: record.attention, + }); + delivered = response?.delivered !== false; + } catch (error) { + record.error = cloneError(errorSummary(error, 'attention_reply_failed')); + } + if (delivered) { + record.attention = null; + for (const lane of record.lanes) { + if (lane.phase === 'needs_attention') { + lane.phase = 'running'; + lane.last_event = 'attention_reply_dispatched'; + } + } + } + await persist(record); + return reconcile(record); + } + return reconcile(record); + }); + } + + async function cancelRun(request) { + const parsed = ownObject(request, 'request'); + assertKeys(parsed, ['run_id'], 'request'); + const runId = requiredString(parsed, 'run_id', 'run_id', (value) => RUN_ID_PATTERN.test(value)); + const record = await loadRecord(runId); + if (!record) admissionError('run_not_found', 'run_id', 'The requested run is not known to this server.'); + return enqueue(runId, async () => { + if (isTerminalRun(record)) return receipt(record, { already_terminal: true, cancel_requested: record.cancel_requested }); + record.cancel_requested = true; + record.telemetry.cancel_attempts += 1; + await persist(record); + const noPromptHasBeenAttempted = record.lanes.every((lane) => lane.prompt_attempted !== true); + if (record.phase === 'awaiting_consent' || noPromptHasBeenAttempted) { + for (const lane of record.lanes) { + if (isTerminalLane(lane) && lane.recovery_classification !== OBSERVATION_RECOVERY_CLASSIFICATION) continue; + lane.phase = 'cancelled'; + lane.cancel_confirmed = true; + lane.error = cloneError({ code: 'cancel_requested' }); + await finishLane(record, lane); + } + record.telemetry.cancel_confirmed = true; + record.phase = 'cancelled'; + bump(record); + await persist(record); + return receipt(record); + } + for (const lane of record.lanes) { + if (isTerminalLane(lane) && lane.recovery_classification !== OBSERVATION_RECOVERY_CLASSIFICATION) continue; + try { + const result = await injected.cancelLane({ + run_id: runId, + assignment_id: lane.assignment_id, + task_id: lane.task_id, + lane, + }); + if (result?.confirmed === true || result?.cancelled === true) { + lane.phase = 'cancelled'; + lane.cancel_confirmed = true; + lane.error = null; + } else { + lane.phase = lane.prompt_attempted ? 'running' : 'failed_pre_prompt'; + lane.recovery_classification = 'cancel_unconfirmed'; + lane.cancel_confirmed = false; + lane.error = cloneError({ code: 'cancel_unconfirmed' }); + } + } catch (error) { + lane.phase = lane.prompt_attempted ? 'running' : 'failed_pre_prompt'; + lane.recovery_classification = 'cancel_unconfirmed'; + lane.cancel_confirmed = false; + lane.error = cloneError(errorSummary(error, 'cancel_unconfirmed')); + } + await finishLane(record, lane); + } + record.telemetry.cancel_confirmed = record.lanes.every((lane) => lane.phase === 'cancelled' || lane.phase === 'completed'); + record.phase = record.telemetry.cancel_confirmed ? 'cancelled' : 'degraded'; + bump(record); + await persist(record); + return receipt(record); + }); + } + + async function waitRun(request, options = {}) { + const parsed = ownObject(request, 'request'); + assertKeys(parsed, ['run_id', 'wait_until', 'wait_ms', 'cursor'], 'request'); + const runId = requiredString(parsed, 'run_id', 'run_id', (value) => RUN_ID_PATTERN.test(value)); + const waitUntil = parsed.wait_until ?? 'decision_or_attention'; + const waitMs = parsed.wait_ms ?? MAX_WAIT_MS; + if (!['progress', 'terminal', 'decision_or_attention'].includes(waitUntil)) admissionError('invalid_format', 'wait_until'); + if (!Number.isInteger(waitMs) || waitMs < 0 || waitMs > MAX_WAIT_MS) admissionError('invalid_format', 'wait_ms'); + const started = Date.now(); + let current = await inspectRun({ run_id: runId }); + const initialCursor = parsed.cursor ?? current.cursor; + const actionable = (value) => RUN_TERMINAL_PHASES.includes(value.phase) + || (waitUntil !== 'progress' && RUN_ATTENTION_PHASES.includes(value.phase)) + || (waitUntil === 'progress' && value.cursor !== initialCursor); + if (actionable(current) || waitMs === 0) return freezeData({ ...current, wait_until: waitUntil, waited_ms: 0 }); + while (Date.now() - started < waitMs) { + if (options.signal?.aborted) break; + const remaining = Math.max(0, waitMs - (Date.now() - started)); + const taskLanes = current.lanes.filter(laneNeedsObservation); + const taskIds = taskLanes.map((lane) => lane.task_id); + const cursors = Object.fromEntries(taskLanes + .filter((lane) => typeof lane.cursor === 'string' && CURSOR_PATTERN.test(lane.cursor)) + .map((lane) => [lane.task_id, lane.cursor])); + try { + await injected.waitForProgress({ + run_id: runId, + task_ids: taskIds, + cursors, + wait_ms: remaining, + wait_until: waitUntil === 'terminal' ? 'terminal' : 'progress', + signal: options.signal, + }); + } catch { + // A wait provider failure is an observation uncertainty, so use a + // bounded backoff before retrying the authoritative read. + await injected.sleep(Math.min(250, remaining), options.signal); + } + if (options.signal?.aborted) break; + const previousCursor = current.cursor; + current = await inspectRun({ run_id: runId }); + if (actionable(current)) break; + // Task-store waits can wake immediately for a terminal task whose + // cleanup boundary is still unfinal, or for a temporarily missing task. + // If admission made no progress, avoid turning that wake into a hot loop. + const remainingAfterWait = Math.max(0, waitMs - (Date.now() - started)); + if (remainingAfterWait > 0 && current.cursor === previousCursor) { + await injected.sleep(Math.min(OBSERVATION_BACKOFF_MS, remainingAfterWait), options.signal); + } + } + return freezeData({ ...current, wait_until: waitUntil, waited_ms: Date.now() - started }); + } + + return capturedFreeze({ + submitRunRequest, + inspectRun, + resumeRun, + replyRun, + cancelRun, + waitRun, + hasRun: (runId) => records.has(runId), + hasRunAsync: async (runId) => (await loadRecord(runId)) !== null, + records, + }); +} diff --git a/plugins/codex-co-engineer/mcp/v3/run-orchestration.mjs b/plugins/codex-co-engineer/mcp/v3/run-orchestration.mjs index c990961..d84c0b8 100644 --- a/plugins/codex-co-engineer/mcp/v3/run-orchestration.mjs +++ b/plugins/codex-co-engineer/mcp/v3/run-orchestration.mjs @@ -38,7 +38,6 @@ import { CREDENTIAL_BOUNDARY_SCHEMA_ID, CREDENTIAL_FILE_ENV_KEYS, CredentialBoundaryError, - DSH_OX_MODEL, assertNoWorkerPushUrl, collectLaneSecrets, createCredentialHandoff, @@ -275,8 +274,7 @@ function takeEnvAfterPreflight(parsed) { function allowedCredentialKey(provider, dshModel) { if (provider === 'grok') return 'XAI_API_KEY'; if (provider === 'cursor-cloud') return 'CURSOR_API_KEY'; - if (provider === 'dsh' && dshModel === DSH_OX_MODEL) return 'OPENROUTER_API_KEY'; - if (provider === 'dsh') return 'MODEL_API_KEY'; + if (provider === 'dsh') return 'OPENROUTER_API_KEY'; return null; } diff --git a/plugins/codex-co-engineer/mcp/v3/run-request-compiler.mjs b/plugins/codex-co-engineer/mcp/v3/run-request-compiler.mjs new file mode 100644 index 0000000..a40c9dd --- /dev/null +++ b/plugins/codex-co-engineer/mcp/v3/run-request-compiler.mjs @@ -0,0 +1,692 @@ +// SimpleRunRequestV1 compiler. +// +// This is the small public ingress for 3.4.1 run submissions. Callers supply +// only semantic intent; all protected identities, model defaults, task IDs, +// prompt-envelope digests, and telemetry-safe facts are derived here. The +// existing full RunManifestV1 / protected run envelope remains a separate +// compatibility path and is intentionally not rewritten by this module. + +import { execFile as nodeExecFile } from 'node:child_process'; +import { realpath as nodeRealpath } from 'node:fs/promises'; +import path from 'node:path'; +import { promisify } from 'node:util'; + +import { + capturedFreeze, + capturedHasOwn, + capturedIncludes, + capturedIsArray, + capturedOwnKeys, + capturedTest, + isKnownProvider, + isKnownRole, + isModelId, + knownProvidersJoined, + requiredAccessForRole, +} from './grammar.mjs'; +import { + childEnvelopeDigestV1, + IDENTITY_LABELS, + runManifestDigestV1, +} from './identity.mjs'; +import { + buildChildIdentityV1, + buildDispatchAttemptV1, + buildGitIdentityV1, + buildProviderRunIdentityV1, + buildRunIdentityV1, +} from './protected-identity.mjs'; +import { resolveRegistrySelectionV1 } from './provider-registry.mjs'; +import { compileChildEnvelopeV1 } from './prompt-compiler.mjs'; +import { + MAX_ASSIGNMENTS, + MIN_ASSIGNMENTS, + PROMPT_MAX_BYTES, + PROMPT_MIN_BYTES, + RUN_ID_PATTERN, + RunContractV1Error, + assertBaseSha, + assertBoundedText, + assertExpectedDurationMs, + assertRepositoryPath, + assertRunId, + assertWriteScopePatterns, + isAssignmentId, +} from './run-manifest.mjs'; +import { parseRunManifestV1 } from './run-policy.mjs'; +import { + assertDirectJsonClosure, + assertNotProxy, + assertPlainObject, + freezeData, + identityBoundDigest, + ownDataValue, +} from './selection-json.mjs'; +import { GIT_CLOSED_ENV, GIT_EXECUTABLE } from './git-identity.mjs'; + +const execFile = promisify(nodeExecFile); +const REALPATH = nodeRealpath; + +export const RUN_REQUEST_SCHEMA_ID = 'codex-co-engineer.run-request.v1'; +export const RUN_REQUEST_VERSION = 1; +export const RUN_REQUEST_ALLOWED_KEYS = capturedFreeze([ + 'run_id', 'repo', 'objective', 'base_sha', 'assignments', +]); +export const RUN_REQUEST_ASSIGNMENT_ALLOWED_KEYS = capturedFreeze([ + 'assignment_id', 'provider', 'model', 'role', 'access', 'prompt', + 'expected_duration_ms', 'write_scope', 'required', 'capabilities', +]); +export const RUN_REQUEST_DERIVED_KEYS = capturedFreeze([ + 'identity', 'git', 'provenance', 'telemetry', 'request_idempotency_key', + 'task_id', 'workspace', 'dispatch', 'child_id', 'manifest_digest', + 'prompt_envelope_digest', 'capability_snapshot_digest', 'child_identity', + 'dispatch_identity', 'provider_run_identity', 'workspace_identity', + 'child_identities', 'dispatch_identities', 'provider_run_identities', + 'workspace_identities', +]); +export const RUN_REQUEST_CAPABILITIES = capturedFreeze([ + 'read_run_receipts', 'read_provider_logs', 'read_own_worktree', +]); +export const RUN_REQUEST_DEFAULT_CAPABILITIES = RUN_REQUEST_CAPABILITIES; +export const RUN_REQUEST_DEFAULT_EXPECTED_DURATION_MS = 600_000; +export const RUN_REQUEST_DEFAULT_MODELS = capturedFreeze({ + grok: 'grok-4', + 'cursor-local': 'composer-1', + 'cursor-cloud': 'claude-sonnet-4-5', + dsh: 'meta/muse-spark-1.3-contributor', +}); +export const RUN_REQUEST_RUN_POLICY = capturedFreeze({ + max_concurrency: MAX_ASSIGNMENTS, + require_same_base: true, + require_disjoint_writer_scopes: true, + allow_post_dispatch_fallback: false, + allow_merge: false, + allow_create_pr: false, + attention_mode: 'aggregate', + completion_mode: 'all_settled_then_verify', +}); +export const RUN_REQUEST_RETURN_CONTRACT = capturedFreeze({ + mode: 'verified_decision', + include_artifact_refs: true, +}); + +const GIT_TIMEOUT_MS = 5_000; +const GIT_MAX_BUFFER = 16 * 1024; +const TASK_ID_MAX = 80; +const TASK_ID_PREFIX = 'ce-'; +const WRITE_ACCESS_ALIASES = capturedFreeze({ + write: 'writer', + writer: 'writer', + read: 'read_only', + read_only: 'read_only', +}); +const DERIVED_KEY_SET = new Set(RUN_REQUEST_DERIVED_KEYS); +const ASSIGNMENT_KEY_SET = new Set(RUN_REQUEST_ASSIGNMENT_ALLOWED_KEYS); +const REQUEST_KEY_SET = new Set(RUN_REQUEST_ALLOWED_KEYS); + +function compilerError(code, field, message = 'The run request is invalid.') { + throw new RunContractV1Error(code, field, message); +} + +function rejectUnknownKeys(value, allowed, field) { + for (const key of capturedOwnKeys(value)) { + if (typeof key !== 'string') compilerError('symbol_key_denied', field); + if (!allowed.has(key)) { + if (DERIVED_KEY_SET.has(key)) { + compilerError('derived_field_denied', `${field}.${key}`, + 'Derived run identity and provenance fields are server-owned.'); + } + compilerError('unknown_key', `${field}.${key}`, 'The run request field is outside the closed vocabulary.'); + } + } +} + +function readRequired(value, key, field) { + if (!capturedHasOwn(value, key)) compilerError('missing_key', field, 'A required run request field is missing.'); + return ownDataValue(value, key, field); +} + +function readOptional(value, key, field) { + if (!capturedHasOwn(value, key)) return undefined; + return ownDataValue(value, key, field); +} + +function assertArray(value, field, min, max) { + if (!capturedIsArray(value)) compilerError('invalid_type', field, 'The run request collection must be an array.'); + if (value.length < min || value.length > max) { + compilerError('out_of_range', field, 'The run request collection is outside the supported bound.'); + } +} + +function normalizeAccess(value, field) { + if (typeof value !== 'string' || !capturedHasOwn(WRITE_ACCESS_ALIASES, value)) { + compilerError('unknown_access', field, 'access must be write, writer, read, or read_only.'); + } + return WRITE_ACCESS_ALIASES[value]; +} + +function validateCapabilities(value, field) { + if (value === undefined) return [...RUN_REQUEST_DEFAULT_CAPABILITIES]; + assertArray(value, field, 0, RUN_REQUEST_CAPABILITIES.length); + const result = []; + const seen = new Set(); + for (let index = 0; index < value.length; index += 1) { + const capability = value[index]; + if (typeof capability !== 'string' || !capturedIncludes(RUN_REQUEST_CAPABILITIES, capability)) { + compilerError('unknown_capability', `${field}[${index}]`, 'The capability is not safe for Co-Engineer-owned artifacts.'); + } + if (seen.has(capability)) compilerError('duplicate_capability', `${field}[${index}]`, 'Capabilities must be unique.'); + seen.add(capability); + result.push(capability); + } + return result; +} + +function defaultEvidence(role) { + if (role === 'verify') return ['provider_report', 'git_identity', 'acceptance_results']; + if (role === 'review') return ['provider_report', 'git_identity']; + return ['provider_report', 'git_identity', 'git_diff']; +} + +function taskIdFor(runId, assignmentId, requestKey) { + const runPart = runId.slice(0, 24); + const assignmentPart = assignmentId.slice(0, 24); + const digestPart = requestKey.slice('sha256:'.length, 'sha256:'.length + 12); + const assignmentDigest = identityBoundDigest(IDENTITY_LABELS.CHILD_IDENTITY, { + run_id: runId, + assignment_id: assignmentId, + }).slice('sha256:'.length, 'sha256:'.length + 8); + const candidate = `${TASK_ID_PREFIX}${runPart}-${assignmentPart}-${digestPart}-${assignmentDigest}`; + return candidate.length <= TASK_ID_MAX ? candidate : candidate.slice(0, TASK_ID_MAX); +} + +function normalizeRepoInput(value) { + if (typeof value !== 'string') compilerError('invalid_type', 'run_request.repo', 'repo must be an absolute path.'); + try { + assertRepositoryPath(value, 'run_request.repo'); + } catch (error) { + throw error; + } + return value; +} + +function normalizeBaseSha(value) { + if (value === undefined) return undefined; + assertBaseSha(value, 'run_request.base_sha'); + return value; +} + +function normalizeModel(provider, value, field) { + const model = value === undefined ? RUN_REQUEST_DEFAULT_MODELS[provider] : value; + if (typeof model !== 'string' || !isModelId(model)) { + compilerError('invalid_model', field, 'The selected model is not in the provider model grammar.'); + } + try { + resolveRegistrySelectionV1({ provider, model }); + } catch (error) { + if (error instanceof RunContractV1Error) throw error; + compilerError('invalid_model', field, 'The selected model is not accepted by the provider registry.'); + } + return model; +} + +function assertSemanticRequestShape(request) { + assertNotProxy(request, 'run_request'); + assertPlainObject(request, 'invalid_type', 'run_request', 'run_request'); + assertDirectJsonClosure(request, 'run_request'); + rejectUnknownKeys(request, REQUEST_KEY_SET, 'run_request'); +} + +function assertAssignmentShape(value, index) { + const field = `run_request.assignments[${index}]`; + assertNotProxy(value, field); + assertPlainObject(value, 'invalid_type', field, 'assignment'); + assertDirectJsonClosure(value, field); + rejectUnknownKeys(value, ASSIGNMENT_KEY_SET, field); +} + +function normalizeAssignment(value, index, baseSha) { + const field = `run_request.assignments[${index}]`; + assertAssignmentShape(value, index); + const assignmentId = readRequired(value, 'assignment_id', `${field}.assignment_id`); + if (typeof assignmentId !== 'string' || !isAssignmentId(assignmentId)) { + compilerError('invalid_format', `${field}.assignment_id`, 'assignment_id is not valid.'); + } + const provider = readRequired(value, 'provider', `${field}.provider`); + if (typeof provider !== 'string' || !isKnownProvider(provider)) { + compilerError('unknown_provider', `${field}.provider`, `provider must be one of ${knownProvidersJoined()}.`); + } + const role = readRequired(value, 'role', `${field}.role`); + if (typeof role !== 'string' || !isKnownRole(role)) { + compilerError('unknown_role', `${field}.role`, 'role must be implement, review, or verify.'); + } + const requestedAccess = readOptional(value, 'access', `${field}.access`); + const access = requestedAccess === undefined + ? requiredAccessForRole(role) + : normalizeAccess(requestedAccess, `${field}.access`); + if (access !== requiredAccessForRole(role)) { + compilerError('role_access_mismatch', `${field}.access`, 'The selected role and access do not match.'); + } + const prompt = readRequired(value, 'prompt', `${field}.prompt`); + assertBoundedText(prompt, { + min: PROMPT_MIN_BYTES, + max: PROMPT_MAX_BYTES, + path: `${field}.prompt`, + label: `${field}.prompt`, + }); + const requestedDuration = readOptional(value, 'expected_duration_ms', `${field}.expected_duration_ms`); + const expectedDuration = requestedDuration === undefined + ? RUN_REQUEST_DEFAULT_EXPECTED_DURATION_MS + : requestedDuration; + assertExpectedDurationMs(expectedDuration, `${field}.expected_duration_ms`); + const required = readOptional(value, 'required', `${field}.required`); + if (required !== undefined && typeof required !== 'boolean') { + compilerError('invalid_type', `${field}.required`, 'required must be a boolean.'); + } + const model = normalizeModel( + provider, + readOptional(value, 'model', `${field}.model`), + `${field}.model`, + ); + const requestedScope = readOptional(value, 'write_scope', `${field}.write_scope`); + if (access === 'read_only') { + if (requestedScope !== undefined) { + assertArray(requestedScope, `${field}.write_scope`, 0, 0); + } + } else if (requestedScope !== undefined) { + assertWriteScopePatterns(requestedScope, `${field}.write_scope`, { minPatterns: 1 }); + } + const capabilities = validateCapabilities( + readOptional(value, 'capabilities', `${field}.capabilities`), + `${field}.capabilities`, + ); + const startingRef = provider === 'cursor-cloud' ? baseSha : undefined; + return { + assignment_id: assignmentId, + role, + access, + prompt, + expected_duration_ms: expectedDuration, + required: required ?? true, + provider, + model, + ...(requestedScope !== undefined ? { requested_write_scope: [...requestedScope] } : {}), + capabilities, + ...(startingRef !== undefined ? { starting_ref: startingRef } : {}), + }; +} + +function assignWriterScopes(assignments) { + const writers = assignments.filter((assignment) => assignment.access === 'writer'); + if (writers.length > 1 && writers.some((assignment) => !assignment.requested_write_scope)) { + compilerError('write_scope_required', 'run_request.assignments', + 'Every writer in a multi-writer run must declare an explicit disjoint write_scope.'); + } + return assignments.map((assignment) => { + if (assignment.access === 'read_only') return { ...assignment, write_scope: [] }; + return { + ...assignment, + write_scope: assignment.requested_write_scope + ? [...assignment.requested_write_scope] + : ['**'], + }; + }); +} + +function manifestAssignment(assignment) { + return { + assignment_id: assignment.assignment_id, + role: assignment.role, + access: assignment.access, + prompt: assignment.prompt, + execution: { provider: assignment.provider, model: assignment.model }, + write_scope: [...assignment.write_scope], + acceptance: [], + expected_duration_ms: assignment.expected_duration_ms, + required_evidence: defaultEvidence(assignment.role), + ...(assignment.starting_ref !== undefined ? { starting_ref: assignment.starting_ref } : {}), + }; +} + +async function runGit(execute, repository, args) { + try { + const result = await execute(GIT_EXECUTABLE, ['-C', repository, ...args], { + encoding: 'utf8', + env: GIT_CLOSED_ENV, + timeout: GIT_TIMEOUT_MS, + maxBuffer: GIT_MAX_BUFFER, + }); + return String(result?.stdout ?? '').trim(); + } catch { + compilerError('repository_invalid', 'run_request.repo', 'Git repository observation failed.'); + } +} + +async function observeGit(repository, requestedBaseSha, execute = execFile) { + const [root, headSha, treeSha, branch, status, remotes] = await Promise.all([ + runGit(execute, repository, ['rev-parse', '--show-toplevel']), + runGit(execute, repository, ['rev-parse', '--verify', 'HEAD^{commit}']), + runGit(execute, repository, ['rev-parse', '--verify', 'HEAD^{tree}']), + runGit(execute, repository, ['branch', '--show-current']), + runGit(execute, repository, ['status', '--porcelain=v1', '--untracked-files=all']), + runGit(execute, repository, ['remote']), + ]); + if (path.resolve(root) !== repository) { + compilerError('repository_alias_denied', 'run_request.repo', 'repo must identify the Git worktree root exactly.'); + } + if (status.length > 0) { + compilerError('repository_dirty', 'run_request.repo', 'The source Git worktree must be clean before admission.'); + } + const baseSha = requestedBaseSha ?? headSha; + assertBaseSha(baseSha, 'run_request.base_sha'); + const verifiedBase = await runGit(execute, repository, [ + 'rev-parse', '--verify', `${baseSha}^{commit}`, + ]); + if (verifiedBase !== baseSha) { + compilerError('base_sha_mismatch', 'run_request.base_sha', 'The exact base SHA was not observed in the repository.'); + } + return freezeData({ + repository_path: repository, + base_sha: baseSha, + head_sha: headSha, + tree_sha: treeSha, + branch: branch || null, + clean: true, + remote_present: remotes.length > 0, + remote_count: remotes.length > 0 ? remotes.split(/\r?\n/u).filter(Boolean).length : 0, + }); +} + +function buildRequestIdempotencyKey({ request, manifest, git, assignments }) { + return identityBoundDigest(IDENTITY_LABELS.REQUEST_IDEMPOTENCY, { + schema: RUN_REQUEST_SCHEMA_ID, + run_id: request.run_id, + repository: manifest.repository, + objective: manifest.objective, + assignments: assignments.map((assignment) => ({ + assignment_id: assignment.assignment_id, + role: assignment.role, + access: assignment.access, + provider: assignment.provider, + model: assignment.model, + prompt: assignment.prompt, + expected_duration_ms: assignment.expected_duration_ms, + required: assignment.required, + write_scope: assignment.write_scope, + capabilities: assignment.capabilities, + })), + source_head_sha: git.head_sha, + source_tree_sha: git.tree_sha, + }); +} + +function buildCapabilityDigest(capabilities) { + return identityBoundDigest(IDENTITY_LABELS.PROVIDER_CAPABILITY, { + schema: 'codex-co-engineer.assignment-capabilities.v1', + capabilities: [...capabilities].sort(), + }); +} + +function buildLaneDigest({ runId, assignment, manifestDigest, childEnvelopeDigest, capabilityDigest }) { + return identityBoundDigest(IDENTITY_LABELS.RESOLVED_LANE_BINDING, { + schema: 'codex-co-engineer.resolved-lane-binding.v1', + run_id: runId, + assignment_id: assignment.assignment_id, + provider: assignment.provider, + model: assignment.model, + role: assignment.role, + access: assignment.access, + write_scope: assignment.write_scope, + required: assignment.required, + manifest_digest: manifestDigest, + prompt_envelope_digest: childEnvelopeDigest, + capability_snapshot_digest: capabilityDigest, + }); +} + +function makeManifest(request, assignments, baseSha) { + return parseRunManifestV1({ + schema: 'codex-co-engineer.run.v1', + run_id: request.run_id, + repository: { path: request.repo, base_sha: baseSha }, + objective: request.objective, + assignments: assignments.map(manifestAssignment), + policy: { + ...RUN_REQUEST_RUN_POLICY, + max_concurrency: assignments.length, + }, + return_contract: RUN_REQUEST_RETURN_CONTRACT, + }); +} + +function makePublicSummary(compiled) { + return freezeData({ + schema: RUN_REQUEST_SCHEMA_ID, + version: RUN_REQUEST_VERSION, + run_id: compiled.run_id, + repository_path: compiled.git.repository_path, + base_sha: compiled.git.base_sha, + manifest_digest: compiled.manifest_digest, + request_idempotency_key: compiled.request_idempotency_key, + assignment_count: compiled.assignments.length, + assignments: compiled.assignments.map((assignment) => ({ + assignment_id: assignment.assignment_id, + task_id: assignment.task_id, + provider: assignment.provider, + model: assignment.model, + role: assignment.role, + access: assignment.access, + required: assignment.required, + write_scope: [...assignment.write_scope], + capability_digest: assignment.capability_digest, + prompt_envelope_digest: assignment.prompt_envelope_digest, + lane_digest: assignment.lane_digest, + dispatch_digest: assignment.dispatch_identity.digest, + provider_run_digest: assignment.provider_run_identity.digest, + })), + }); +} + +/** + * Compile a server-owned simple run request into a frozen dispatch snapshot. + * `options.observeGit` is an internal test/host seam; the public request never + * supplies observations or derived identities. + */ +export async function compileRunRequestV1(request, options = {}) { + assertSemanticRequestShape(request); + const runId = readRequired(request, 'run_id', 'run_request.run_id'); + assertRunId(runId, 'run_request.run_id'); + if (!capturedTest(RUN_ID_PATTERN, runId)) compilerError('invalid_format', 'run_request.run_id'); + const repo = normalizeRepoInput(readRequired(request, 'repo', 'run_request.repo')); + const objective = readRequired(request, 'objective', 'run_request.objective'); + assertBoundedText(objective, { + min: 1, + max: 4096, + path: 'run_request.objective', + label: 'run_request.objective', + }); + const requestedBaseSha = normalizeBaseSha(readOptional(request, 'base_sha', 'run_request.base_sha')); + const rawAssignments = readRequired(request, 'assignments', 'run_request.assignments'); + assertArray(rawAssignments, 'run_request.assignments', MIN_ASSIGNMENTS, MAX_ASSIGNMENTS); + const observed = typeof options.observeGit === 'function' + ? await options.observeGit(repo, requestedBaseSha) + : await (async () => { + let canonicalRepo = repo; + try { + canonicalRepo = await REALPATH(repo); + } catch { + compilerError('repository_invalid', 'run_request.repo', 'The repository path could not be resolved.'); + } + if (canonicalRepo !== repo) { + compilerError('repository_alias_denied', 'run_request.repo', 'repo must be the canonical Git worktree path.'); + } + return observeGit(canonicalRepo, requestedBaseSha, options.executeGit ?? execFile); + })(); + if (!observed || typeof observed !== 'object') { + compilerError('repository_invalid', 'run_request.repo', 'Git observation did not return a valid identity.'); + } + const git = freezeData({ + repository_path: repo, + base_sha: observed.base_sha, + head_sha: observed.head_sha, + tree_sha: observed.tree_sha, + branch: observed.branch ?? null, + clean: observed.clean === true, + remote_present: observed.remote_present === true, + remote_count: observed.remote_count ?? 0, + }); + assertBaseSha(git.base_sha, 'git.base_sha'); + assertBaseSha(git.head_sha, 'git.head_sha'); + if (git.clean !== true) compilerError('repository_dirty', 'run_request.repo', 'The source Git worktree must be clean before admission.'); + + const normalized = []; + const seenIds = new Set(); + for (let index = 0; index < rawAssignments.length; index += 1) { + const assignment = normalizeAssignment(rawAssignments[index], index, git.base_sha); + if (seenIds.has(assignment.assignment_id)) { + compilerError('duplicate_assignment_id', `run_request.assignments[${index}].assignment_id`, 'Assignment IDs must be unique.'); + } + seenIds.add(assignment.assignment_id); + normalized.push(assignment); + } + const assignments = assignWriterScopes(normalized); + // makeManifest -> parseRunManifestV1 performs the authoritative + // conservative overlap check. Keep one implementation of the glob-prefix + // rule so disjoint scopes such as src/api/** and src/ui/** are not rejected + // by a divergent preflight parser. + const manifest = makeManifest({ run_id: runId, repo, objective }, assignments, git.base_sha); + const manifestDigestDescriptor = runManifestDigestV1(manifest); + const manifestDigest = manifestDigestDescriptor.digest; + const gitIdentity = buildGitIdentityV1({ + repository_path: repo, + base_sha: git.base_sha, + }); + const runIdentity = buildRunIdentityV1({ + run_id: runId, + git: gitIdentity, + manifest_digest: manifestDigest, + }); + const childIdentities = []; + const dispatchIdentities = []; + const providerRunIdentities = []; + const compiledAssignments = []; + const semanticRequest = { run_id: runId, repo, objective, assignments }; + const provisionalRequestKey = buildRequestIdempotencyKey({ + request: semanticRequest, + manifest, + git, + assignments, + }); + for (let index = 0; index < manifest.assignments.length; index += 1) { + const manifestAssignmentSnapshot = manifest.assignments[index]; + const assignment = assignments[index]; + const childIdentity = buildChildIdentityV1({ + run_id: runId, + assignment_id: assignment.assignment_id, + }); + const dispatchIdentity = buildDispatchAttemptV1({ + run_id: runId, + assignment_id: assignment.assignment_id, + attempt: 1, + }); + const envelope = compileChildEnvelopeV1(manifest, assignment.assignment_id); + const envelopeDigest = childEnvelopeDigestV1(envelope).digest; + const capabilityDigest = buildCapabilityDigest(assignment.capabilities); + const laneDigest = buildLaneDigest({ + runId, + assignment, + manifestDigest, + childEnvelopeDigest: envelopeDigest, + capabilityDigest, + }); + const providerRunIdentity = buildProviderRunIdentityV1({ + run_id: runId, + assignment_id: assignment.assignment_id, + attempt: 1, + provider: assignment.provider, + model: assignment.model, + git: gitIdentity, + manifest_digest: manifestDigest, + prompt_envelope_digest: envelopeDigest, + resolved_lane_digest: laneDigest, + capability_snapshot_digest: capabilityDigest, + agent_id: null, + provider_run_id: null, + }); + childIdentities.push(childIdentity); + dispatchIdentities.push(dispatchIdentity); + providerRunIdentities.push(providerRunIdentity); + compiledAssignments.push({ + assignment_id: manifestAssignmentSnapshot.assignment_id, + task_id: taskIdFor(runId, assignment.assignment_id, provisionalRequestKey), + provider: assignment.provider, + model: assignment.model, + role: assignment.role, + access: assignment.access, + required: assignment.required, + prompt: assignment.prompt, + expected_duration_ms: assignment.expected_duration_ms, + write_scope: [...assignment.write_scope], + capabilities: [...assignment.capabilities], + child_identity: childIdentity, + dispatch_identity: dispatchIdentity, + provider_run_identity: providerRunIdentity, + child_envelope: envelope, + prompt_envelope_digest: envelopeDigest, + capability_digest: capabilityDigest, + lane_digest: laneDigest, + ...(assignment.starting_ref !== undefined ? { starting_ref: assignment.starting_ref } : {}), + }); + } + const requestIdempotencyKey = buildRequestIdempotencyKey({ + request: semanticRequest, + manifest, + git, + assignments: compiledAssignments, + }); + // Task IDs are derived from the final request key. Rebuild the bounded + // assignment snapshots so a future change to the identity framing cannot + // leave task IDs tied to a provisional key. + const finalAssignments = compiledAssignments.map((assignment) => ({ + ...assignment, + task_id: taskIdFor(runId, assignment.assignment_id, requestIdempotencyKey), + })); + const result = { + schema: RUN_REQUEST_SCHEMA_ID, + version: RUN_REQUEST_VERSION, + run_id: runId, + objective, + repo, + git, + base_sha_explicit: requestedBaseSha !== undefined, + manifest, + manifest_digest: manifestDigest, + git_identity: gitIdentity, + run_identity: runIdentity, + request_idempotency_key: requestIdempotencyKey, + child_identities: childIdentities, + dispatch_identities: dispatchIdentities, + provider_run_identities: providerRunIdentities, + assignments: finalAssignments, + managed_workspace_policy: freezeData({ + mode: 'managed', + source: 'server_default', + local_sha_allowed: true, + base_sha_explicit: requestedBaseSha !== undefined, + direct_mode: false, + remote_mutation: false, + }), + public_summary: null, + }; + result.public_summary = makePublicSummary(result); + return freezeData(result); +} + +export function isSimpleRunRequest(value) { + return value !== null && typeof value === 'object' && capturedHasOwn(value, 'run_id') + && capturedHasOwn(value, 'repo') && capturedHasOwn(value, 'objective') + && capturedHasOwn(value, 'assignments'); +} + +capturedFreeze(compileRunRequestV1); +capturedFreeze(isSimpleRunRequest); diff --git a/plugins/codex-co-engineer/mcp/v3/run-scheduler.mjs b/plugins/codex-co-engineer/mcp/v3/run-scheduler.mjs index 5fb7618..ffeb85f 100644 --- a/plugins/codex-co-engineer/mcp/v3/run-scheduler.mjs +++ b/plugins/codex-co-engineer/mcp/v3/run-scheduler.mjs @@ -751,15 +751,32 @@ function boundText(value, maxBytes) { } function boundOptions(value) { - if (!ARRAY_IS_ARRAY(value) || IS_PROXY(value)) return null; + if (value === undefined || value === null) return { options: null, invalid: false }; + if (!ARRAY_IS_ARRAY(value) || IS_PROXY(value)) return { options: null, invalid: true }; const options = []; + const allowedKeys = new Set(['optionId', 'kind', 'name', 'label', 'description']); const limit = Math.min(value.length, MAX_SCHEDULER_ATTENTION_OPTIONS); for (let index = 0; index < limit; index += 1) { - const option = boundText(value[index], MAX_SCHEDULER_ATTENTION_OPTION_BYTES); - if (option === null) continue; + const raw = value[index]; + if (typeof raw === 'string') { + const option = boundText(raw, MAX_SCHEDULER_ATTENTION_OPTION_BYTES); + if (option !== null) options.push(option); + continue; + } + if (raw === null || typeof raw !== 'object' || ARRAY_IS_ARRAY(raw) || IS_PROXY(raw)) { + return { options: null, invalid: true }; + } + const option = {}; + for (const key of capturedOwnKeys(raw)) { + if (typeof key !== 'string' || !allowedKeys.has(key)) return { options: null, invalid: true }; + const text = boundText(pickOwn(raw, key), MAX_SCHEDULER_ATTENTION_OPTION_BYTES); + if (text === null || text.length === 0) return { options: null, invalid: true }; + option[key] = text; + } + if (typeof option.kind !== 'string') return { options: null, invalid: true }; options.push(option); } - return options.length === 0 ? null : options; + return { options: options.length === 0 ? null : options, invalid: false }; } function projectAttention(raw, provider) { @@ -773,12 +790,14 @@ function projectAttention(raw, provider) { || typeof questionId !== 'string' || !capturedTest(QUESTION_ID_PATTERN, questionId)) { return { attention: null, invalid: true }; } + const boundedOptions = boundOptions(pickOwn(raw, 'options')); + if (boundedOptions.invalid) return { attention: null, invalid: true }; return { attention: { session_id: sessionId, question_id: questionId, prompt: boundText(pickOwn(raw, 'prompt'), MAX_SCHEDULER_ATTENTION_PROMPT_BYTES), - options: boundOptions(pickOwn(raw, 'options')), + options: boundedOptions.options, reply_capability: expectedReplyCapability(provider), }, invalid: false, diff --git a/plugins/codex-co-engineer/mcp/v3/run-tool-adapter.mjs b/plugins/codex-co-engineer/mcp/v3/run-tool-adapter.mjs index 3367b6f..7065b7e 100644 --- a/plugins/codex-co-engineer/mcp/v3/run-tool-adapter.mjs +++ b/plugins/codex-co-engineer/mcp/v3/run-tool-adapter.mjs @@ -79,12 +79,18 @@ import { assertRunId, isAssignmentId, } from './run-manifest.mjs'; +import { + RUN_REQUEST_ALLOWED_KEYS, + RUN_REQUEST_DERIVED_KEYS, +} from './run-request-compiler.mjs'; +import { RUN_ADMISSION_METHODS } from './run-admission.mjs'; import { RUN_RUNTIME_METHODS, createRunRuntime, } from './run-runtime.mjs'; import { createRunScheduler } from './run-scheduler.mjs'; import { openRunStore } from './run-store.mjs'; +import { boundProviderResult, utf8Head } from './compact-task.mjs'; import { projectExperience } from './response.mjs'; import { assertDirectJsonClosure, @@ -96,6 +102,13 @@ import { ownDataValue, } from './selection-json.mjs'; +const semanticRunExperiences = new WeakMap(); + +export function experienceForRunToolResult(value) { + if (!value || typeof value !== 'object') return null; + return semanticRunExperiences.get(value) ?? value.experience ?? null; +} + export const RUN_TOOL_ADAPTER_SCHEMA_ID = 'codex-co-engineer.run-tool-adapter.v1'; export const RUN_TOOL_ADAPTER_VERSION = 1; export const RUN_TOOL_ADAPTER_RECEIPT_SCHEMA_ID = 'codex-co-engineer.run-tool-receipt.v1'; @@ -113,7 +126,7 @@ export const WAIT_UNTIL_VALUES = capturedFreeze([ ]); export const ADDITIVE_STATUS_KEYS = capturedFreeze(['run_id']); -export const ADDITIVE_DELEGATE_KEYS = capturedFreeze(['run']); +export const ADDITIVE_DELEGATE_KEYS = capturedFreeze(['run', 'run_request']); export const ADDITIVE_TASK_KEYS = capturedFreeze([ 'run_id', 'assignment_id', 'attention', 'run_reply', ]); @@ -141,13 +154,17 @@ export const RUN_ASSIGNMENT_RUNTIME_KEYS = capturedFreeze([ ]); export const ATTENTION_REQUEST_KEYS = capturedFreeze(['expected_revision', 'items']); export const RUN_REPLY_KEYS = capturedFreeze([ - 'batch_id', 'expected_revision', 'reply', + 'approval_ref', 'batch_id', 'expected_revision', 'reply', 'request_consent', ]); export const RUN_TOOL_RECEIPT_KEYS = capturedFreeze([ 'assignment_count', 'attention', 'audience', 'candidate', 'checks', 'cleanup', 'complete_candidate_blocked', 'decision_or_attention', - 'experience', 'lanes', 'mode', 'operation', 'remote_mutated', 'run_id', - 'schema', 'side_effects', 'status', 'tool', 'version', 'wake', + 'dispatch_uncertain_assignment_ids', 'dispatched_assignment_ids', + 'consent', 'cursor', 'error', 'experience', 'handoff', 'lanes', 'mode', 'operation', 'phase', + 'revision', + 'remote_mutated', 'run_id', 'schema', 'side_effects', 'status', 'tool', + 'undispatched_assignment_ids', 'version', 'wait_until', 'waited_ms', 'wake', + 'result', ]); export const RUN_TOOL_ADAPTER_CHECKS = capturedFreeze([ @@ -185,6 +202,8 @@ export const RUN_TOOL_ADAPTER_ALWAYS_FALSE_SIDE_EFFECTS = capturedFreeze([ 'remote_mutated', 'sixth_tool_exposed', ]); +export const SIMPLE_RUN_RECEIPT_STRUCTURED_BYTES_MAX = 72 * 1024; +export const SIMPLE_RUN_STATUS_STRUCTURED_BYTES_MAX = 24 * 1024; export const MAX_ADAPTER_DIAGNOSTIC_BYTES = 160; export const RUN_ID_SCHEMA_PATTERN = RUN_ID_PATTERN.source; @@ -195,7 +214,8 @@ export const OWNER_ONLY_EVIDENCE_KEYS = capturedFreeze([ ]); export const ACTIONABLE_LANE_STATUSES = capturedFreeze([ 'needs_attention', 'completed', 'failed', 'cancelled', 'unresolved', - 'timeout', 'transport_lost', 'environment_blocked', + 'timeout', 'transport_lost', 'environment_blocked', 'failed_pre_prompt', + 'partial_handoff', 'unrecoverable_post_prompt', ]); const IS_PROXY = utilTypes.isProxy; @@ -290,6 +310,7 @@ const CONTENT_FREE = capturedFreeze({ unknown_operation: 'The tool arguments do not map to a frozen run operation.', unknown_provider: 'The provider is not an accepted four-slot registry entry.', unknown_tool: 'The public catalog remains status, delegate, task, tasks, cancel.', + simple_runtime_unavailable: 'The 3.4.2 simple run runtime is unavailable.', }); export const RUN_TOOL_ADAPTER_ERROR_CODES = capturedFreeze([ @@ -323,11 +344,12 @@ export const RUN_TOOL_ADAPTER_ERROR_CODES = capturedFreeze([ 'unknown_operation', 'unknown_provider', 'unknown_tool', + 'simple_runtime_unavailable', ]); const ADAPTER_DEPENDENCY_KEYS = capturedFreeze([ 'attention', 'classifyLaneTask', 'projectLaneTask', 'rememberSubmitContext', - 'runtime', + 'runtime', 'simpleRuntime', ]); const RUNTIME_METHODS = RUN_RUNTIME_METHODS; const ATTENTION_METHODS = capturedFreeze(['get', 'reply']); @@ -338,6 +360,10 @@ const RECONSTRUCT_LANE_STATUSES = capturedFreeze([ ]); const pendingRunCatalogSnapshots = new Map(); const pendingRunExperienceContext = new Map(); +const SIMPLE_RUN_REQUEST_KEYS = capturedFreeze([ + ...RUN_REQUEST_ALLOWED_KEYS, + ...RUN_REQUEST_DERIVED_KEYS, +]); function rememberExperienceContext(runId, patch) { if (typeof runId !== 'string' || runId.length === 0) return; @@ -362,12 +388,34 @@ function emptySideEffects() { return sideEffects; } -function emptyChecks() { +function emptyChecks(observed) { const checks = {}; - for (const name of RUN_TOOL_ADAPTER_CHECKS) checks[name] = true; + for (const name of RUN_TOOL_ADAPTER_CHECKS) checks[name] = null; + if (observed && typeof observed === 'object' && !capturedIsArray(observed)) { + for (const name of RUN_TOOL_ADAPTER_CHECKS) { + if (capturedHasOwn(observed, name) && typeof observed[name] === 'boolean') { + checks[name] = observed[name]; + } + } + } return checks; } +function providerResultFor(value) { + if (value === undefined || value === null || typeof value !== 'object') return undefined; + if (capturedHasOwn(value, 'result')) return value.result; + if (capturedHasOwn(value, 'provider_result')) return value.provider_result; + if (value.task && typeof value.task === 'object') { + if (capturedHasOwn(value.task, 'result')) return value.task.result; + if (capturedHasOwn(value.task, 'provider_result')) return value.task.provider_result; + } + return undefined; +} + +function boundedProviderResult(raw) { + return raw === undefined ? undefined : boundProviderResult(raw).value; +} + function ownKeySet(value, field) { assertNotProxy(value, field); let keys; @@ -807,6 +855,32 @@ async function parseSubmit(args) { if (mixLegacySingleTask('delegate', args)) { failAdapter('mixed_tool_mode', 'delegate', CONTENT_FREE.mixed_tool_mode); } + const simpleRequest = optionalValue(args, 'run_request', 'run_request'); + if (simpleRequest !== undefined) { + if (capturedHasOwn(args, 'run')) { + failAdapter('mixed_run_operation', 'delegate', CONTENT_FREE.mixed_run_operation); + } + const request = quarantineObject(simpleRequest, 'run_request', SIMPLE_RUN_REQUEST_KEYS); + const runId = requireString(request, 'run_id', 'run_request.run_id', (value) => { + try { assertRunId(value, 'run_request.run_id'); return true; } catch { return false; } + }); + const objective = optionalValue(request, 'objective', 'run_request.objective'); + return { + runId, + simpleRequest: request, + context: freezeData({ + run_id: runId, + objective: typeof objective === 'string' ? objective : null, + profile: null, + catalog_digest: null, + prompts: {}, + durations: {}, + repository_path: typeof request.repo === 'string' ? request.repo : null, + base_sha: typeof request.base_sha === 'string' ? request.base_sha : null, + }), + catalogSnapshot: null, + }; + } const run = quarantineObject(optionalValue(args, 'run', 'run') ?? failAdapter('missing_key', 'run', CONTENT_FREE.missing_key), 'run', RUN_SUBMIT_KEYS); for (const key of RUN_SUBMIT_REQUIRED_KEYS) { @@ -886,19 +960,38 @@ function parseAssignmentIds(value, field) { return ids; } -function projectCandidate(runId) { - const ref = expectedCandidateRefV1({ run_id: runId }); +function projectCandidate(runId, runtimeCandidate) { + const expectedRef = expectedCandidateRefV1({ run_id: runId }); + const actual = runtimeCandidate && typeof runtimeCandidate === 'object' + && !capturedIsArray(runtimeCandidate) + ? runtimeCandidate + : null; + const ref = actual && typeof actual.ref === 'string' + && isRunOwnedCandidateRefV1(actual.ref, runId) + ? actual.ref + : expectedRef; + const authoritative = actual?.authority === 'p35'; return freezeData({ ref, - composed: false, - ready_for_codex_review: false, + composed: actual?.composed === true, + ready_for_codex_review: actual?.ready_for_codex_review === true, authority: 'p35', namespace: CANDIDATE_REF_NAMESPACE, - accepted: isRunOwnedCandidateRefV1(ref, runId), + accepted: authoritative && actual?.accepted === true + && isRunOwnedCandidateRefV1(ref, runId), }); } -function projectLane(lane, projectLaneTask, classifyLaneTask) { +function providerResultExceededDefaultBound(raw) { + if (raw === undefined) return false; + try { + return byteLength(raw) > 8_192; + } catch { + return false; + } +} + +function projectLane(lane, projectLaneTask, classifyLaneTask, reportTruncation = false) { if (lane === undefined || lane === null || typeof lane !== 'object') return lane; const copy = { ...lane }; if (copy.task && typeof copy.task === 'object') { @@ -912,6 +1005,16 @@ function projectLane(lane, projectLaneTask, classifyLaneTask) { if (copy.artifacts && typeof copy.artifacts === 'object') { copy.artifacts = sanitizeModelFacing(copy.artifacts); } + const rawResult = providerResultFor(copy); + if (rawResult !== undefined) { + const bounded = boundProviderResult(rawResult); + copy.result = bounded.value; + if (reportTruncation === true + && (bounded.truncated === true || providerResultExceededDefaultBound(rawResult))) { + copy.result_truncated = true; + } + if (capturedHasOwn(copy, 'provider_result')) delete copy.provider_result; + } return sanitizeModelFacing(copy); } @@ -936,11 +1039,400 @@ function attentionRecord(receipt) { return receipt; } -function projectReceipt(tool, operation, runtimeReceipt, projectLaneTask, classifyLaneTask, wakeRequested = false) { +function compactAdmissionHandoff(handoff) { + if (handoff === undefined || handoff === null || typeof handoff !== 'object' || Array.isArray(handoff)) return null; + return { + schema: handoff.schema ?? 'codex-co-engineer.partial-handoff.v1', + assignment_id: handoff.assignment_id ?? null, + worktree: typeof handoff.worktree === 'string' ? handoff.worktree.slice(0, 512) : null, + branch: typeof handoff.branch === 'string' ? handoff.branch.slice(0, 256) : null, + starting_sha: handoff.starting_sha ?? null, + current_head: handoff.current_head ?? null, + clean: typeof handoff.clean === 'boolean' ? handoff.clean : null, + changed_files: Array.isArray(handoff.changed_files) ? handoff.changed_files.slice(0, 16) : [], + commits: Array.isArray(handoff.commits) ? handoff.commits.slice(0, 16) : [], + no_commit: handoff.no_commit === true, + partial_diff: handoff.partial_diff === true, + last_acknowledged_provider_event: handoff.last_acknowledged_provider_event ?? null, + recovery_classification: handoff.recovery_classification ?? null, + safe_next_actions: Array.isArray(handoff.safe_next_actions) ? handoff.safe_next_actions.slice(0, 3) : [], + diagnostics: 'Use the run-scoped diagnostics/provenance view for full handoff evidence.', + }; +} + +function compactAdmissionLane(lane) { + if (lane === undefined || lane === null || typeof lane !== 'object' || Array.isArray(lane)) return lane; + const rawResult = providerResultFor(lane); + const bounded = rawResult === undefined ? null : boundProviderResult(rawResult); + const result = bounded?.value; + return { + assignment_id: lane.assignment_id ?? null, + task_id: lane.task_id ?? null, + provider: lane.provider ?? null, + model: lane.model ?? null, + role: lane.role ?? null, + access: lane.access ?? null, + required: lane.required !== false, + phase: lane.phase ?? lane.status ?? null, + status: lane.status ?? lane.phase ?? null, + prepared: lane.prepared === true, + session_ready: lane.session_ready === true, + prompt_attempted: lane.prompt_attempted === true, + prompt_dispatched: lane.prompt_dispatched === true, + dispatch_confidence: lane.dispatch_confidence ?? null, + session_id: lane.session_id ?? null, + cursor: lane.cursor ?? null, + child_identity_digest: lane.child_identity_digest ?? null, + dispatch_identity_digest: lane.dispatch_identity_digest ?? null, + provider_run_identity_digest: lane.provider_run_identity_digest ?? null, + workspace_identity_digest: lane.workspace_identity_digest + ?? lane.workspace_identity?.digest + ?? null, + error: lane.error ?? null, + recovery_classification: lane.recovery_classification ?? null, + cancel_confirmed: lane.cancel_confirmed ?? null, + task_final: lane.task_final ?? null, + result_truncated: lane.result_truncated === true || bounded?.truncated === true, + handoff: compactAdmissionHandoff(lane.handoff), + ...(rawResult !== undefined ? { result } : {}), + }; +} + +function compactSemanticHandoff(handoff) { + if (handoff === null || typeof handoff !== 'object' || Array.isArray(handoff)) return null; + return { + ...(typeof handoff.worktree === 'string' ? { worktree: handoff.worktree } : {}), + ...(typeof handoff.branch === 'string' ? { branch: handoff.branch } : {}), + ...(typeof handoff.current_head === 'string' ? { head: handoff.current_head } : {}), + ...(typeof handoff.clean === 'boolean' ? { clean: handoff.clean } : {}), + ...(handoff.partial_diff === true ? { partial: true } : {}), + }; +} + +function compactSemanticLane(lane) { + const projected = { + assignment_id: lane.assignment_id ?? null, + task_id: lane.task_id ?? null, + provider: lane.provider ?? null, + ...(typeof lane.role === 'string' ? { role: lane.role } : {}), + status: lane.status ?? lane.phase ?? null, + required: lane.required !== false, + prompt_dispatched: lane.prompt_dispatched === true, + }; + if (lane.dispatch_confidence != null) projected.dispatch_confidence = lane.dispatch_confidence; + if (lane.result != null) projected.result = lane.result; + if (lane.result_truncated === true) projected.result_truncated = true; + if (lane.error != null) projected.error = lane.error; + if (lane.recovery_classification != null) { + projected.recovery_classification = lane.recovery_classification; + } + const handoff = compactSemanticHandoff(lane.handoff); + if (handoff) projected.artifacts = handoff; + return projected; +} + +function cleanupNeedsAttention(receipt, unconfirmed) { + const cleanup = receipt.cleanup; + return unconfirmed === true + || receipt.phase === 'lifecycle_pending' + || cleanup?.proof_bound === false + || (Array.isArray(cleanup?.unresolved) && cleanup.unresolved.length > 0) + || (Number.isSafeInteger(cleanup?.remaining) && cleanup.remaining > 0) + || (receipt.operation === 'cleanup' && cleanup?.cleaned !== true) + || receipt.lanes.some((lane) => lane?.status === 'lifecycle_pending' + || (lane?.task_final === false && ['cancelled', 'unresolved'].includes(lane?.status))); +} + +function compactSha(value) { + return typeof value === 'string' && capturedTest(/^[0-9a-fA-F]{40}$/u, value) + ? value.toLowerCase() + : null; +} + +function compactSemanticCandidate(runtimeReceipt, projectedCandidate) { + const source = runtimeReceipt?.candidate; + const handoff = runtimeReceipt?.handoff; + const hasSource = source && typeof source === 'object' && !Array.isArray(source); + const hasHandoff = handoff && typeof handoff === 'object' && !Array.isArray(handoff); + if (!hasSource && !hasHandoff) return null; + const authoritative = source?.authority === 'p35'; + const head = compactSha(source?.head) ?? compactSha(handoff?.current_head); + const tree = compactSha(source?.tree); + const candidate = { + ...(hasSource && isRunOwnedCandidateRefV1(source.ref, runtimeReceipt.run_id) + ? { ref: projectedCandidate?.ref ?? null } + : {}), + ...(head ? { head } : {}), + ...(tree ? { tree } : {}), + ...(authoritative && projectedCandidate?.ready_for_codex_review === true + ? { ready_for_codex_review: true } + : {}), + ...(projectedCandidate?.accepted === true ? { accepted: true } : {}), + }; + return Object.keys(candidate).length > 0 ? candidate : null; +} + +function compactOverflowReceipt(compact, simpleResponseCap) { + const fallback = { + schema: compact.schema, + version: compact.version, + mode: compact.mode, + tool: compact.tool, + operation: compact.operation, + run_id: compact.run_id, + status: compact.status, + phase: compact.phase, + cursor: compact.cursor, + revision: compact.revision, + assignment_count: compact.assignment_count, + authoritative_required_dispatch: compact.authoritative_required_dispatch === true, + lanes: compact.lanes.map((lane) => ({ + assignment_id: lane.assignment_id, + task_id: lane.task_id, + status: lane.status, + required: lane.required, + prompt_dispatched: lane.prompt_dispatched, + ...(lane.result !== undefined ? { result_omitted: true } : {}), + ...(lane.error?.code ? { error: { code: utf8Head(lane.error.code, 128) } } : {}), + })), + ...(compact.attention ? { attention: { + status: compact.attention.status ?? null, + batch_id: compact.attention.batch_id ?? null, + revision: compact.attention.revision ?? null, + details_omitted: true, + reply_blocked: true, + } } : {}), + ...(compact.consent ? { consent: { + status: compact.consent.status ?? compact.consent.consent?.status ?? null, + details_omitted: true, + decision_blocked: true, + } } : {}), + ...(compact.error?.code ? { error: { code: utf8Head(compact.error.code, 128) } } : {}), + ...(compact.result !== undefined ? { result_omitted: true } : {}), + ...(compact.candidate ? { candidate: compact.candidate } : {}), + ...(compact.blockers ? { blockers: compact.blockers } : {}), + diagnostics: { + view: 'diagnostics', + details_omitted: true, + reason: 'response_size_limit', + instruction: 'Repeat task with this run_id and view="diagnostics" before replying or accepting.', + }, + }; + if (byteLength(fallback) <= simpleResponseCap) return fallback; + return { + schema: compact.schema, + version: compact.version, + mode: compact.mode, + run_id: utf8Head(compact.run_id, 64), + status: 'unresolved', + phase: 'unresolved', + cursor: utf8Head(compact.cursor, 256), + error: { code: 'response_projection_overflow' }, + diagnostics: fallback.diagnostics, + }; +} + +function projectSemanticRunReceipt(receipt, runtimeReceipt, { + unconfirmed, + simpleResponseCap, + topLevelResultTruncated, +}) { + const lanes = receipt.lanes.map(compactSemanticLane); + const terminal = [ + 'completed', 'failed', 'cancelled', 'degraded', 'unresolved', + 'partial_handoff', 'unrecoverable_post_prompt', 'lifecycle_pending', + ].includes(receipt.phase); + const verificationBlocked = terminal && receipt.complete_candidate_blocked === true; + const cleanupBlocked = cleanupNeedsAttention(receipt, unconfirmed); + const attention = receipt.attention; + const attentionRequired = attention?.status === 'open' + || lanes.some((lane) => lane.status === 'needs_attention'); + const candidate = compactSemanticCandidate(runtimeReceipt, receipt.candidate); + const verification = runtimeReceipt?.verification?.authority === 'p35' + ? sanitizeModelFacing(runtimeReceipt.verification) + : null; + const topLevelResult = lanes.some((lane) => lane.result != null) + ? undefined + : receipt.result; + const compact = { + schema: receipt.schema, + version: receipt.version, + mode: receipt.mode, + tool: receipt.tool, + operation: receipt.operation, + run_id: receipt.run_id, + status: receipt.status, + phase: receipt.phase, + cursor: receipt.cursor, + revision: receipt.revision, + assignment_count: receipt.assignment_count, + authoritative_required_dispatch: receipt.authoritative_required_dispatch === true, + lanes, + ...(attentionRequired || attention?.status === 'reply_committed' || attention?.status === 'resolved' + ? { attention } + : {}), + ...(receipt.consent != null ? { consent: receipt.consent } : {}), + ...(receipt.error != null ? { error: receipt.error } : {}), + ...(topLevelResult !== undefined ? { result: topLevelResult } : {}), + ...(topLevelResult !== undefined && topLevelResultTruncated === true + ? { result_truncated: true } + : {}), + ...(candidate ? { candidate } : {}), + ...(verification ? { verification } : {}), + ...(receipt.operation === 'wait' ? { + wait_until: receipt.wait_until, + ...(Number.isSafeInteger(receipt.waited_ms) ? { waited_ms: receipt.waited_ms } : {}), + wake: receipt.wake === true, + } : {}), + ...((verificationBlocked || cleanupBlocked) ? { + blockers: { + ...(verificationBlocked ? { verification: true } : {}), + ...(cleanupBlocked ? { cleanup: true } : {}), + }, + } : {}), + ...(cleanupBlocked || receipt.operation === 'cleanup' + ? { cleanup: receipt.cleanup } + : {}), + diagnostics: { + view: 'diagnostics', + }, + }; + if (byteLength(compact) > simpleResponseCap) { + const resultCount = lanes.filter((lane) => lane.result !== undefined).length + + (topLevelResult !== undefined ? 1 : 0); + const resultFree = { + ...compact, + lanes: compact.lanes.map(({ result: _result, ...lane }) => lane), + ...(topLevelResult !== undefined ? { result: undefined } : {}), + }; + const available = Math.max(128, simpleResponseCap - byteLength(resultFree) - 2_048); + const perResultBytes = Math.max(128, Math.floor(available / Math.max(1, resultCount))); + const topBounded = topLevelResult !== undefined + ? boundProviderResult(topLevelResult, perResultBytes) + : null; + const boundedCompact = { + ...compact, + lanes: lanes.map((lane) => { + if (lane.result === undefined) return lane; + const bounded = boundProviderResult(lane.result, perResultBytes); + return { + ...lane, + result: bounded.value, + result_truncated: lane.result_truncated === true || bounded.truncated === true, + }; + }), + ...(topBounded ? { + result: topBounded.value, + ...(topBounded.truncated === true ? { result_truncated: true } : {}), + } : {}), + }; + if (byteLength(boundedCompact) <= simpleResponseCap) return boundedCompact; + return compactOverflowReceipt(compact, simpleResponseCap); + } + return compact; +} + +function byteLength(value) { + return Buffer.byteLength(JSON.stringify(value), 'utf8'); +} + +function compactProviderResult(result, taskId) { + if (result === undefined) return undefined; + let serialized; + try { + serialized = JSON.stringify(result); + } catch { + serialized = null; + } + return { + truncated: true, + detail_task_id: typeof taskId === 'string' ? taskId : null, + preview: utf8Head(serialized ?? '', 768), + }; +} + +function malformedRuntimeReceipt(runId) { + return { + schema: 'codex-co-engineer.run-admission.v1', + version: 1, + run_id: typeof runId === 'string' ? runId : null, + phase: 'unresolved', + status: 'unresolved', + revision: null, + cursor: null, + assignment_count: 0, + lanes: [], + complete_candidate_blocked: true, + error: { + code: 'durable_state_mismatch', + message: CONTENT_FREE.durable_state_mismatch, + }, + attention: null, + consent: null, + admission: null, + dispatched_assignment_ids: [], + undispatched_assignment_ids: [], + dispatch_uncertain_assignment_ids: [], + authoritative_required_dispatch: false, + already_terminal: false, + telemetry: null, + cleanup: null, + }; +} + +function isRuntimeReceipt(value, expectedRunId) { + if (value === undefined || value === null || typeof value !== 'object' + || ARRAY_IS_ARRAY(value) || IS_PROXY(value)) return false; + if (typeof value.run_id !== 'string' + || (typeof expectedRunId === 'string' && value.run_id !== expectedRunId)) return false; + if (!ARRAY_IS_ARRAY(value.lanes) + || value.lanes.length < MIN_ASSIGNMENTS + || value.lanes.length > MAX_ASSIGNMENTS) return false; + for (const lane of value.lanes) { + if (lane === null || typeof lane !== 'object' + || ARRAY_IS_ARRAY(lane) || IS_PROXY(lane) + || typeof lane.assignment_id !== 'string' + || lane.assignment_id.length === 0 + || typeof lane.task_id !== 'string' + || lane.task_id.length === 0 + || (lane.status !== undefined && typeof lane.status !== 'string') + || (lane.phase !== undefined && typeof lane.phase !== 'string') + || (lane.status === undefined && lane.phase === undefined)) return false; + } + if (value.assignment_count !== undefined + && (!Number.isSafeInteger(value.assignment_count) + || value.assignment_count < 0 + || value.assignment_count !== value.lanes.length)) return false; + if (value.status !== undefined && (typeof value.status !== 'string' || value.status.length === 0)) return false; + if (value.phase !== undefined && (typeof value.phase !== 'string' || value.phase.length === 0)) return false; + if (value.status === undefined && value.phase === undefined) return false; + if (value.revision !== undefined && value.revision !== null + && (!Number.isSafeInteger(value.revision) || value.revision < 0)) return false; + if (value.cursor !== undefined && value.cursor !== null + && typeof value.cursor !== 'string' + && (typeof value.cursor !== 'object' || ARRAY_IS_ARRAY(value.cursor) || IS_PROXY(value.cursor))) return false; + if (value.schema === 'codex-co-engineer.run-admission.v1' + && typeof value.cursor === 'string' + && !/^\d{1,16}$/u.test(value.cursor)) return false; + return true; +} + +function projectReceipt(tool, operation, runtimeReceipt, projectLaneTask, classifyLaneTask, + wakeRequested = false, expectedRunId = null, view = null) { + if (!isRuntimeReceipt(runtimeReceipt, expectedRunId)) { + runtimeReceipt = malformedRuntimeReceipt(expectedRunId); + } const runId = runtimeReceipt?.run_id; + const simpleAdmission = runtimeReceipt?.schema === 'codex-co-engineer.run-admission.v1'; let lanes = ARRAY_IS_ARRAY(runtimeReceipt?.lanes) - ? runtimeReceipt.lanes.map((lane) => projectLane(lane, projectLaneTask, classifyLaneTask)) + ? runtimeReceipt.lanes.map((lane) => projectLane( + lane, projectLaneTask, classifyLaneTask, simpleAdmission, + )) : []; + const simpleResponseCap = operation === 'status' + ? SIMPLE_RUN_STATUS_STRUCTURED_BYTES_MAX + : SIMPLE_RUN_RECEIPT_STRUCTURED_BYTES_MAX; + if (simpleAdmission) lanes = lanes.map(compactAdmissionLane); const unconfirmed = cancellationUnconfirmed(operation, lanes); if (unconfirmed === true) { lanes = lanes.map((lane) => { @@ -952,14 +1444,26 @@ function projectReceipt(tool, operation, runtimeReceipt, projectLaneTask, classi } const blocked = unconfirmed === true || runtimeReceipt?.complete_candidate_blocked === true + || [ + 'awaiting_consent', 'degraded', 'failed', 'cancelled', 'needs_attention', + 'validating', 'preparing_workspaces', 'dispatching', 'verifying', 'unresolved', + 'partial_handoff', 'unrecoverable_post_prompt', 'lifecycle_pending', + ].includes(runtimeReceipt?.phase) || lanes.some((lane) => lane?.required !== false && ( lane.status === 'unresolved' || lane.status === 'failed' || lane.status === 'lifecycle_pending' || lane.status === 'transport_lost' + || lane.status === 'failed_pre_prompt' + || lane.status === 'partial_handoff' + || lane.status === 'unrecoverable_post_prompt' )); const sideEffects = emptySideEffects(); - if (runtimeReceipt?.side_effects?.task_dispatched === true) sideEffects.provider_dispatched = true; + if (runtimeReceipt?.side_effects?.task_dispatched === true + || (ARRAY_IS_ARRAY(runtimeReceipt?.dispatched_assignment_ids) + && runtimeReceipt.dispatched_assignment_ids.length > 0)) { + sideEffects.provider_dispatched = true; + } const cleanupSource = runtimeReceipt?.cleanup ?? freezeData({ cleaned: false, proof_bound: true, removed: 0, remaining: null, unresolved: [], }); @@ -990,7 +1494,11 @@ function projectReceipt(tool, operation, runtimeReceipt, projectLaneTask, classi const wake = wakeRequested === true && actionable === true; const status = unconfirmed === true ? 'unresolved' - : (runtimeReceipt?.status ?? 'inspected'); + : (runtimeReceipt?.status ?? runtimeReceipt?.phase ?? 'inspected'); + const phase = runtimeReceipt?.phase ?? status; + const rawResult = providerResultFor(runtimeReceipt); + const boundedResult = rawResult === undefined ? null : boundProviderResult(rawResult); + const result = boundedResult?.value; const receiptBody = { schema: RUN_TOOL_ADAPTER_RECEIPT_SCHEMA_ID, version: RUN_TOOL_ADAPTER_VERSION, @@ -998,8 +1506,11 @@ function projectReceipt(tool, operation, runtimeReceipt, projectLaneTask, classi tool, operation, status, + phase, run_id: runId, assignment_count: runtimeReceipt?.assignment_count ?? lanes.length, + revision: Number.isSafeInteger(runtimeReceipt?.revision) ? runtimeReceipt.revision : null, + cursor: typeof runtimeReceipt?.cursor === 'string' ? runtimeReceipt.cursor : null, lanes, attention, cleanup: sanitizeModelFacing(cleanup), @@ -1009,9 +1520,27 @@ function projectReceipt(tool, operation, runtimeReceipt, projectLaneTask, classi unresolved_required_blocks: blocked, exactly_once_reply: attention?.status === 'reply_committed' || attention?.status === 'resolved', }), - candidate: runId ? projectCandidate(runId) : null, + candidate: runId ? projectCandidate(runId, runtimeReceipt?.candidate) : null, complete_candidate_blocked: blocked, - checks: emptyChecks(), + consent: sanitizeModelFacing(runtimeReceipt?.consent ?? null), + admission: sanitizeModelFacing(runtimeReceipt?.admission ?? null), + dispatched_assignment_ids: sanitizeModelFacing(runtimeReceipt?.dispatched_assignment_ids ?? []), + undispatched_assignment_ids: sanitizeModelFacing(runtimeReceipt?.undispatched_assignment_ids ?? []), + dispatch_uncertain_assignment_ids: sanitizeModelFacing(runtimeReceipt?.dispatch_uncertain_assignment_ids ?? []), + authoritative_required_dispatch: runtimeReceipt?.authoritative_required_dispatch === true, + already_terminal: runtimeReceipt?.already_terminal === true, + error: sanitizeModelFacing(runtimeReceipt?.error ?? null), + telemetry: sanitizeModelFacing(runtimeReceipt?.telemetry ?? null), + ...(operation === 'wait' ? { + wait_until: capturedIncludes(WAIT_UNTIL_VALUES, runtimeReceipt?.wait_until) + ? runtimeReceipt.wait_until + : ADDITIVE_WAIT_UNTIL, + ...(Number.isSafeInteger(runtimeReceipt?.waited_ms) && runtimeReceipt.waited_ms >= 0 + ? { waited_ms: runtimeReceipt.waited_ms } + : {}), + } : {}), + checks: emptyChecks(runtimeReceipt?.checks), + ...(result !== undefined ? { result } : {}), side_effects: sideEffects, audience: 'model', wake, @@ -1031,11 +1560,93 @@ function projectReceipt(tool, operation, runtimeReceipt, projectLaneTask, classi }, journal: runtimeReceipt?.journal ?? null, evidence: runtimeReceipt?.evidence ?? null, + candidate: runtimeReceipt?.candidate ?? receiptBody.candidate, }); - return freezeData({ + if (simpleAdmission && view !== 'diagnostics') { + const compact = freezeData(projectSemanticRunReceipt(receiptBody, runtimeReceipt, { + unconfirmed, + simpleResponseCap, + topLevelResultTruncated: boundedResult?.truncated === true, + })); + semanticRunExperiences.set(compact, experience); + return compact; + } + if (simpleAdmission && byteLength(receiptBody) > simpleResponseCap) { + receiptBody.attention = receiptBody.attention && { + ...receiptBody.attention, + items: Array.isArray(receiptBody.attention.items) + ? receiptBody.attention.items.map((item) => ({ + ...item, + prompt: typeof item.prompt === 'string' ? item.prompt.slice(0, 512) : item.prompt, + options: Array.isArray(item.options) ? item.options.slice(0, 4) : item.options, + })) + : receiptBody.attention.items, + }; + receiptBody.lanes = receiptBody.lanes.map((lane) => ({ + ...lane, + handoff: lane.handoff ? { ...lane.handoff, changed_files: [], commits: [], safe_next_actions: [] } : null, + })); + } + if (simpleAdmission && byteLength(receiptBody) > simpleResponseCap) { + receiptBody.lanes = receiptBody.lanes.map(({ handoff: _handoff, ...lane }) => lane); + } + let projected = { ...receiptBody, experience, - }); + }; + if (simpleAdmission && byteLength(projected) > simpleResponseCap) { + projected = { + ...projected, + attention: projected.attention && { + ...projected.attention, + items: Array.isArray(projected.attention.items) + ? projected.attention.items.slice(0, 2).map((item) => ({ + assignment_id: item.assignment_id ?? null, + task_id: item.task_id ?? null, + provider: item.provider ?? null, + question_id: item.question_id ?? null, + capability: item.capability ?? null, + resource: item.resource ?? null, + action: item.action ?? null, + prompt: typeof item.prompt === 'string' ? item.prompt.slice(0, 256) : null, + })) + : projected.attention.items, + }, + lanes: projected.lanes.map(({ handoff: _handoff, ...lane }) => lane), + telemetry: projected.telemetry && { + admission_duration_ms: projected.telemetry.admission_duration_ms ?? null, + dispatch_duration_ms: projected.telemetry.dispatch_duration_ms ?? null, + dispatch_confidence: projected.telemetry.dispatch_confidence ?? null, + handoff_class: projected.telemetry.handoff_class ?? null, + }, + }; + if (byteLength(projected) > simpleResponseCap) { + projected = { + ...projected, + lanes: projected.lanes.slice(0, 8).map((lane) => { + const result = compactProviderResult(lane.result, lane.task_id); + return { + assignment_id: lane.assignment_id ?? null, + task_id: lane.task_id ?? null, + provider: lane.provider ?? null, + model: lane.model ?? null, + phase: lane.phase ?? null, + status: lane.status ?? null, + prompt_dispatched: lane.prompt_dispatched === true, + dispatch_confidence: lane.dispatch_confidence ?? null, + task_final: lane.task_final ?? null, + ...(result !== undefined ? { result } : {}), + }; + }), + experience: projectExperience({ + ...projected, + lanes: [], + assignment_count: projected.assignment_count, + }), + }; + } + } + return freezeData(projected); } function assertFunctionMap(value, field, methods) { @@ -1112,6 +1723,10 @@ export function createRunToolAdapter(dependencies) { 'dependencies.runtime', RUNTIME_METHODS, ); + const simpleRuntime = hasOwn(dependencies, 'simpleRuntime') + ? assertFunctionMap(ownDataValue(dependencies, 'simpleRuntime', 'dependencies.simpleRuntime'), + 'dependencies.simpleRuntime', RUN_ADMISSION_METHODS) + : null; const attention = hasOwn(dependencies, 'attention') ? assertFunctionMap(ownDataValue(dependencies, 'attention', 'dependencies.attention'), 'dependencies.attention', ATTENTION_METHODS) @@ -1141,8 +1756,44 @@ export function createRunToolAdapter(dependencies) { const counters = { submit: 0, inspect: 0, resume: 0, cancel: 0, reply: 0, }; + const simpleRunIds = new Set(); + + function isSimpleRun(runId) { + if (simpleRunIds.has(runId)) return true; + if (simpleRuntime && typeof simpleRuntime.hasRun === 'function') { + try { + if (simpleRuntime.hasRun(runId) === true) { + simpleRunIds.add(runId); + return true; + } + } catch { + // A failed lookup must not make a legacy run appear simple. + } + } + return false; + } + + async function resolveSimpleRun(runId) { + if (isSimpleRun(runId)) return true; + if (simpleRuntime && typeof simpleRuntime.hasRunAsync === 'function') { + try { + if (await simpleRuntime.hasRunAsync(runId)) { + simpleRunIds.add(runId); + return true; + } + } catch { + // Fall through to the legacy runtime when the simple store cannot + // prove ownership of this run id. + } + } + return false; + } async function inspectLive(runId, previous) { + if (await resolveSimpleRun(runId)) { + counters.inspect += 1; + return simpleRuntime.inspectRun({ run_id: runId }); + } const cursors = cursorsFromReceipt(previous); if (cursors.length > 0) { counters.resume += 1; @@ -1181,6 +1832,13 @@ export function createRunToolAdapter(dependencies) { assertDirectJsonClosure(args, 'arguments'); denyForbiddenTree(args, 'arguments'); const operation = resolveOperation(tool, args); + const requestedView = tool === 'task' && capturedHasOwn(args, 'view') + ? optionalValue(args, 'view', 'view') + : null; + if (requestedView !== null + && !capturedIncludes(['summary', 'compact', 'diagnostics'], requestedView)) { + failAdapter('invalid_format', 'view', CONTENT_FREE.invalid_format); + } if ((tool === 'task' || tool === 'cancel' || tool === 'delegate' || tool === 'tasks') && mixLegacySingleTask(tool, args) && tool !== 'tasks') { failAdapter('mixed_tool_mode', tool, CONTENT_FREE.mixed_tool_mode); @@ -1191,8 +1849,24 @@ export function createRunToolAdapter(dependencies) { const signal = options && typeof options === 'object' ? options.signal : undefined; let runtimeReceipt; + let requestedRunId = null; if (operation === 'submit') { const parsed = await parseSubmit(args); + requestedRunId = parsed.runId; + if (parsed.simpleRequest !== undefined) { + if (simpleRuntime === null) { + failAdapter('simple_runtime_unavailable', 'run_request', CONTENT_FREE.simple_runtime_unavailable); + } + if (rememberSubmitContext) rememberSubmitContext(parsed.context); + rememberExperienceContext(parsed.runId, { + objective: parsed.context.objective, + base_sha: parsed.context.base_sha, + digest: null, + }); + counters.submit += 1; + runtimeReceipt = await simpleRuntime.submitRunRequest(parsed.simpleRequest, { signal }); + simpleRunIds.add(parsed.runId); + } else { if (parsed.catalogSnapshot !== null && parsed.catalogSnapshot !== undefined) { pendingRunCatalogSnapshots.set(parsed.runId, parsed.catalogSnapshot); } @@ -1204,26 +1878,37 @@ export function createRunToolAdapter(dependencies) { }); counters.submit += 1; runtimeReceipt = await runtime.submitRun(parsed.runtimeRequest); + } } else if (operation === 'status') { const runId = requireRunId(args); + requestedRunId = runId; const assignmentId = optionalValue(args, 'assignment_id', 'assignment_id'); if (assignmentId !== undefined && !isAssignmentId(assignmentId)) { failAdapter('invalid_format', 'assignment_id', CONTENT_FREE.invalid_format); } counters.inspect += 1; - runtimeReceipt = await runtime.inspectRun({ + runtimeReceipt = await ((await resolveSimpleRun(runId)) ? simpleRuntime.inspectRun({ run_id: runId }) : runtime.inspectRun({ run_id: runId, ...(assignmentId !== undefined ? { assignment_id: assignmentId } : {}), - }); + })); } else if (operation === 'wait') { const runId = requireRunId(args); + requestedRunId = runId; const waitUntil = waitUntilValue(args); if (waitUntil !== undefined && !capturedIncludes(WAIT_UNTIL_VALUES, waitUntil)) { failAdapter('invalid_format', 'wait_until', CONTENT_FREE.invalid_format); } - runtimeReceipt = await waitForDecision(runId, args, signal); + runtimeReceipt = (await resolveSimpleRun(runId)) + ? await simpleRuntime.waitRun({ + run_id: runId, + ...(waitUntil !== undefined ? { wait_until: waitUntil } : {}), + ...(capturedHasOwn(args, 'wait_ms') ? { wait_ms: optionalValue(args, 'wait_ms', 'wait_ms') } : {}), + ...(capturedHasOwn(args, 'cursor') ? { cursor: optionalValue(args, 'cursor', 'cursor') } : {}), + }, { signal }) + : await waitForDecision(runId, args, signal); } else if (operation === 'attention') { const runId = requireRunId(args); + requestedRunId = runId; const attentionRequest = quarantineObject( ownDataValue(args, 'attention', 'attention'), 'attention', @@ -1236,21 +1921,61 @@ export function createRunToolAdapter(dependencies) { validateAttentionItemsV1(items, 'attention.items'); rememberExperienceContext(runId, { attention_items: items }); counters.resume += 1; - runtimeReceipt = await runtime.resumeRun({ - run_id: runId, - attention_items: items, - }); + runtimeReceipt = (await resolveSimpleRun(runId)) + ? await simpleRuntime.replyRun({ run_id: runId, attention_reply: { items } }, { signal }) + : await runtime.resumeRun({ run_id: runId, attention_items: items }); } else if (operation === 'reply') { const runId = requireRunId(args); - if (attention === null) { - failAdapter('injected_dependency_invalid', 'attention', CONTENT_FREE.injected_dependency_invalid); - } + requestedRunId = runId; const replyRequest = quarantineObject( ownDataValue(args, 'run_reply', 'run_reply'), 'run_reply', RUN_REPLY_KEYS, ); + const requestConsentAgain = capturedHasOwn(replyRequest, 'request_consent'); + if (requestConsentAgain && ownDataValue(replyRequest, 'request_consent', 'run_reply.request_consent') !== true) { + failAdapter('invalid_format', 'run_reply.request_consent', CONTENT_FREE.invalid_format); + } + if (requestConsentAgain && ( + capturedHasOwn(replyRequest, 'approval_ref') + || capturedHasOwn(replyRequest, 'batch_id') + || capturedHasOwn(replyRequest, 'expected_revision') + || capturedHasOwn(replyRequest, 'reply') + )) { + failAdapter('mixed_run_operation', 'run_reply', CONTENT_FREE.mixed_run_operation); + } counters.reply += 1; + if (await resolveSimpleRun(runId)) { + const simpleReply = { run_id: runId }; + if (capturedHasOwn(replyRequest, 'approval_ref')) { + simpleReply.approval_ref = ownDataValue(replyRequest, 'approval_ref', 'run_reply.approval_ref'); + } + if (requestConsentAgain) { + simpleReply.request_consent = true; + } else if (!capturedHasOwn(replyRequest, 'approval_ref') + || capturedHasOwn(replyRequest, 'batch_id') + || capturedHasOwn(replyRequest, 'expected_revision') + || capturedHasOwn(replyRequest, 'reply')) { + simpleReply.attention_reply = { + ...(capturedHasOwn(replyRequest, 'batch_id') + ? { batch_id: ownDataValue(replyRequest, 'batch_id', 'run_reply.batch_id') } + : {}), + ...(capturedHasOwn(replyRequest, 'expected_revision') + ? { expected_revision: optionalValue(replyRequest, 'expected_revision', 'run_reply.expected_revision') } + : {}), + ...(capturedHasOwn(replyRequest, 'reply') + ? { reply: ownDataValue(replyRequest, 'reply', 'run_reply.reply') } + : {}), + }; + } + runtimeReceipt = await simpleRuntime.replyRun(simpleReply, { signal }); + } else { + if (requestConsentAgain) { + failAdapter('simple_runtime_unavailable', 'run_reply.request_consent', CONTENT_FREE.simple_runtime_unavailable); + } + if (attention === null) { + failAdapter('injected_dependency_invalid', 'attention', CONTENT_FREE.injected_dependency_invalid); + } const attentionReceipt = await attention.reply({ run_id: runId, batch_id: ownDataValue(replyRequest, 'batch_id', 'run_reply.batch_id'), @@ -1263,13 +1988,25 @@ export function createRunToolAdapter(dependencies) { ...runtimeReceipt, attention: attentionRecord(attentionReceipt), }); + } } else if (operation === 'cancel' || operation === 'cleanup') { const runId = requireRunId(args); + requestedRunId = runId; let assignmentIds = parseAssignmentIds( optionalValue(args, 'assignment_ids', 'assignment_ids'), 'assignment_ids', ); if (assignmentIds === undefined) { + if (await resolveSimpleRun(runId)) { + counters.cancel += 1; + runtimeReceipt = await simpleRuntime.cancelRun({ run_id: runId }); + const projected = projectReceipt( + tool, operation, runtimeReceipt, projectLaneTask, classifyLaneTask, + operation === 'wait', + runId, + ); + return projected; + } counters.inspect += 1; const inspected = await runtime.inspectRun({ run_id: runId }); assignmentIds = ARRAY_IS_ARRAY(inspected?.lanes) @@ -1280,11 +2017,11 @@ export function createRunToolAdapter(dependencies) { } } counters.cancel += 1; - runtimeReceipt = await runtime.cancelRun({ + runtimeReceipt = await ((await resolveSimpleRun(runId)) ? simpleRuntime.cancelRun({ run_id: runId }) : runtime.cancelRun({ run_id: runId, assignment_ids: assignmentIds, cleanup: operation === 'cleanup', - }); + })); } else { failAdapter('unknown_operation', 'tool', CONTENT_FREE.unknown_operation); } @@ -1292,6 +2029,8 @@ export function createRunToolAdapter(dependencies) { const projected = projectReceipt( tool, operation, runtimeReceipt, projectLaneTask, classifyLaneTask, operation === 'wait', + requestedRunId, + requestedView, ); return projected; } diff --git a/plugins/codex-co-engineer/mcp/v3/runtime-entrypoints.mjs b/plugins/codex-co-engineer/mcp/v3/runtime-entrypoints.mjs new file mode 100644 index 0000000..379fe1e --- /dev/null +++ b/plugins/codex-co-engineer/mcp/v3/runtime-entrypoints.mjs @@ -0,0 +1,31 @@ +import { constants } from 'node:fs'; +import { access, stat } from 'node:fs/promises'; +import { fileURLToPath } from 'node:url'; + +const ENTRYPOINTS = Object.freeze({ + local: Object.freeze([ + fileURLToPath(new URL('./acp-worker.mjs', import.meta.url)), + fileURLToPath(new URL('./credential-handoff-loader.mjs', import.meta.url)), + ]), + cloud: Object.freeze([ + fileURLToPath(new URL('./cursor-cloud-worker.mjs', import.meta.url)), + ]), +}); + +async function inspectReadableFile(file) { + const metadata = await stat(file); + if (!metadata.isFile()) throw Object.assign(new Error('Runtime entrypoint is not a regular file.'), { code: 'EINVAL' }); + await access(file, constants.R_OK); +} + +export async function assertRuntimeEntrypoints(provider, { inspectFile = inspectReadableFile } = {}) { + const entrypoints = provider === 'cursor-cloud' ? ENTRYPOINTS.cloud : ENTRYPOINTS.local; + try { + await Promise.all(entrypoints.map((entrypoint) => inspectFile(entrypoint))); + } catch { + throw Object.assign( + new Error('The installed Codex-Co-Engineer runtime is incomplete.'), + { code: 'runtime_install_incomplete' }, + ); + } +} diff --git a/plugins/codex-co-engineer/mcp/v3/server.mjs b/plugins/codex-co-engineer/mcp/v3/server.mjs index d6e37b2..9aeda69 100644 --- a/plugins/codex-co-engineer/mcp/v3/server.mjs +++ b/plugins/codex-co-engineer/mcp/v3/server.mjs @@ -22,8 +22,10 @@ import { compactTaskCard, sanitizePublicReceipt } from './diagnostics.mjs'; import { advertiseMcpAppsCapability, buildToolResult, + classifyExperienceCard, listExperienceUiResourcesForClient, normalizeResponseMode, + projectExperience, readExperienceUiResourceForClient, resolveExperienceResultMeta, resolveExperienceToolMeta, @@ -46,7 +48,10 @@ import { } from './supervisor.mjs'; import { classifyRunToolCall, + experienceForRunToolResult, } from './run-tool-adapter.mjs'; +import { createNativeConsentTransport } from './consent.mjs'; +import { createConsentGrantStore } from './consent-grants.mjs'; const PROTOCOLS = new Set(['2025-11-25', '2025-06-18', '2025-03-26']); let negotiated = '2025-11-25'; @@ -83,22 +88,64 @@ function advertisedTools() { const RESPONSE_MODE_PROPERTY = { type: 'string', - enum: ['structured'], - description: 'Optional presentation control stripped before business logic. Omit or leave unset for the 3.1.1-compatible full sanitized receipt in content[0].text (equals JSON.stringify(structuredContent)). Set to "structured" for a bounded text fallback while structuredContent remains the authoritative receipt.', + enum: ['structured', 'legacy'], + description: 'Optional presentation control stripped before business logic. Native runs default to bounded structured-first transport. Set legacy only for a text-only run client that requires the same compact semantic receipt fully encoded in content[0].text. Omitted legacy single-task calls retain their compatible full text.', }; -const RESPONSE_MODE_HINT = ' Optional response_mode="structured" opts into bounded content[0].text with authoritative structuredContent; omit for legacy full-text duplication.'; +const RESPONSE_MODE_HINT = ' Native runs default to bounded structured-first text; text-only run clients may set response_mode="legacy" to encode the same compact semantic receipt fully in text. Use task.run_id with view="diagnostics" for detailed run evidence. Omitted legacy single-task calls retain full compatible text.'; + +const SERVER_INSTRUCTIONS = 'Use delegate.run_request for one bounded run, then task.run_id with the returned cursor for status or waits; use task.run_reply for one same-session decision, tasks.run_id for aggregate waits, and cancel.run_id to cancel. Use task_id for expanded task diagnostics or legacy single-task calls.'; + +const RUN_TOOL_OUTPUT_SCHEMA = { + type: 'object', + properties: { + schema: { type: 'string' }, + version: { type: ['integer', 'string'] }, + mode: { type: 'string', enum: ['run', 'legacy'] }, + operation: { type: 'string' }, + tool: { type: 'string', enum: ['status', 'delegate', 'task', 'tasks', 'cancel'] }, + status: { type: 'string' }, + phase: { type: 'string' }, + run_id: { type: ['string', 'null'] }, + assignment_count: { type: 'integer' }, + revision: { type: ['integer', 'null'] }, + cursor: { type: ['string', 'object', 'null'] }, + wait_until: { type: 'string' }, + waited_ms: { type: 'integer' }, + lanes: { type: 'array', items: { type: 'object' } }, + task: { type: 'object' }, + tasks: { type: 'array', items: { type: 'object' } }, + result: {}, + candidate: { type: ['object', 'null'] }, + complete_candidate_blocked: { type: 'boolean' }, + error: { type: ['object', 'null'] }, + experience: { type: ['object', 'null'] }, + }, + additionalProperties: true, +}; + +const TOOL_METADATA = { + status: { title: 'Inspect a Co-Engineer run', annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: true, openWorldHint: true } }, + delegate: { title: 'Start a Co-Engineer run', annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true } }, + task: { title: 'Inspect or wait for a Co-Engineer run', annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true } }, + tasks: { title: 'Wait for Co-Engineer tasks', annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: true, openWorldHint: true } }, + cancel: { title: 'Cancel a Co-Engineer run', annotations: { readOnlyHint: false, destructiveHint: true, idempotentHint: true, openWorldHint: true } }, +}; const TOOLS = [ { name: 'status', - description: `Show the local Co-Engineer supervisor, provider capabilities, advertised MCP pending-call budget, and recent task state.${RESPONSE_MODE_HINT}`, + title: TOOL_METADATA.status.title, + annotations: TOOL_METADATA.status.annotations, + outputSchema: RUN_TOOL_OUTPUT_SCHEMA, + description: `Inspect one bounded native Co-Engineer run by run_id, including lifecycle state, cursor, attention, and bounded result. Omit run_id to show the compatible 3.2.1 supervisor snapshot, provider capabilities, advertised MCP pending-call budget, and recent task state.${RESPONSE_MODE_HINT}`, inputSchema: { type: 'object', properties: { detail: { type: 'string', enum: ['full', 'compact'], description: 'full returns full receipts (default). compact returns redacted compact cards.' }, task_limit: { type: 'integer', minimum: 0, maximum: 20, description: 'Maximum tasks to return (0-20). Default 20. Ignored when include_tasks is false.' }, include_tasks: { type: 'boolean', description: 'When false, omit recent tasks for readiness-only checks.' }, + refresh: { type: 'boolean', description: 'Force a fresh provider-readiness probe. Without refresh, readiness is shared for a short local TTL and cold probes are bounded.' }, response_mode: RESPONSE_MODE_PROPERTY, run_id: { type: 'string', @@ -111,7 +158,10 @@ const TOOLS = [ }, { name: 'delegate', - description: `Delegate a review or implementation task to Grok, Cursor Local, Cursor Cloud, or DSH. The absolute Git worktree path must be supplied in the property named repo. Provide expected_duration_ms or a backwards-compatible timeout_ms; the recorded deadline is ceil(expected_duration_ms * 1.20) unless timeout_ms is an explicit override of at least that margin. Local tasks use a managed worktree by default; direct mode is explicit.${RESPONSE_MODE_HINT}`, + title: TOOL_METADATA.delegate.title, + annotations: TOOL_METADATA.delegate.annotations, + outputSchema: RUN_TOOL_OUTPUT_SCHEMA, + description: `Start one bounded native Co-Engineer run with run_request for a review or implementation task. A run_request assignment may omit expected_duration_ms to use the 600000 ms default; explicit estimates keep the 20% deadline margin. The compatible single-task path remains available for Grok, Cursor Local, Cursor Cloud, or DSH and still requires expected_duration_ms or a backwards-compatible timeout_ms. The absolute Git worktree path must be supplied in the property named repo. Local tasks use a managed worktree by default; direct mode is explicit.${RESPONSE_MODE_HINT}`, inputSchema: { type: 'object', properties: { @@ -119,8 +169,8 @@ const TOOLS = [ provider: { type: 'string', enum: ['grok', 'cursor-local', 'cursor-cloud', 'dsh'] }, dsh_model: { type: 'string', - enum: ['muse-spark-1.2-contributor', 'stealth/ox-alpha'], - description: 'DSH only. Defaults to Muse Spark 1.2 Contributor; select stealth/ox-alpha for the OpenRouter-backed Ox Alpha route.', + enum: ['meta/muse-spark-1.3-contributor', 'stealth/ox-alpha'], + description: 'DSH only. Defaults to Muse Spark 1.3 Contributor; select stealth/ox-alpha for the OpenRouter-backed Ox Alpha route.', }, repo: { type: 'string', description: 'Required property named repo: absolute path to the Git worktree (for example, /absolute/path/to/git-worktree). Do not rename this property to git_root or repository.' }, prompt: { type: 'string', minLength: 1, maxLength: 262144 }, @@ -149,6 +199,40 @@ const TOOLS = [ provider_repo_url: { type: 'string', minLength: 1, maxLength: 4096, description: 'Optional credential-free provider-visible repository URL override for Cursor Cloud. SSH origins are canonicalized to HTTPS without credentials.' }, provider_repo: { type: 'string', minLength: 1, maxLength: 4096, description: 'Backward-compatible alias for provider_repo_url; Cursor Cloud only.' }, response_mode: RESPONSE_MODE_PROPERTY, + run_request: { + type: 'object', + additionalProperties: false, + required: ['run_id', 'repo', 'objective', 'assignments'], + description: '3.4.2 simple run request. The server derives Git identity, provider models, task IDs, workspaces, dispatch identities, and protected telemetry. Do not supply derived provenance fields.', + properties: { + run_id: { type: 'string', pattern: '^[a-z][a-z0-9-]{2,63}$' }, + repo: { type: 'string', description: 'Canonical absolute Git worktree path.' }, + objective: { type: 'string', minLength: 1, maxLength: 4096 }, + base_sha: { type: 'string', pattern: '^[0-9a-f]{40}$', description: 'Optional exact local base SHA; omitted means the observed clean HEAD.' }, + assignments: { + type: 'array', + minItems: 1, + maxItems: 8, + items: { + type: 'object', + additionalProperties: false, + required: ['assignment_id', 'provider', 'role', 'prompt'], + properties: { + assignment_id: { type: 'string', pattern: '^[a-z][a-z0-9-]{0,63}$' }, + provider: { type: 'string', enum: ['grok', 'cursor-local', 'cursor-cloud', 'dsh'] }, + model: { type: 'string', maxLength: 128, description: 'Optional exact model override; otherwise the closed provider default is derived.' }, + role: { type: 'string', enum: ['implement', 'review', 'verify'] }, + access: { type: 'string', enum: ['write', 'writer', 'read', 'read_only'], description: 'Optional explicit access. Omitted access is derived from role: implement means writer; review and verify mean read_only.' }, + prompt: { type: 'string', minLength: 1, maxLength: 16384 }, + expected_duration_ms: { type: 'integer', minimum: MIN_DURATION_MS, maximum: MAX_EXPECTED_DURATION_MS, default: 600000, description: 'Optional expected duration. Omitted means 600000 ms; the server records a deadline with the existing 20% margin. Explicit estimates retain the same validation and margin.' }, + write_scope: { type: 'array', minItems: 0, maxItems: 16, items: { type: 'string' }, description: 'Optional for writers; required explicitly for each writer when more than one writer lane exists. Read-only lanes must use an empty scope.' }, + required: { type: 'boolean', default: true }, + capabilities: { type: 'array', uniqueItems: true, items: { type: 'string', enum: ['read_run_receipts', 'read_provider_logs', 'read_own_worktree'] } }, + }, + }, + }, + }, + }, run: { type: 'object', additionalProperties: false, @@ -174,8 +258,8 @@ const TOOLS = [ }, allOf: [ { - if: { required: ['run'] }, - then: { required: ['run'] }, + if: { anyOf: [{ required: ['run'] }, { required: ['run_request'] }] }, + then: { oneOf: [{ required: ['run'] }, { required: ['run_request'] }] }, else: { required: ['task_id', 'provider', 'repo', 'prompt'], anyOf: [ @@ -190,7 +274,10 @@ const TOOLS = [ }, { name: 'task', - description: `Inspect one task. view=summary is the default receipt plus diagnostic envelope and event_cursor. view=compact is a bounded coordination payload without full task or runtime bodies. view=diagnostics is a side-effect-free cursor-paged evidence page. wait_until=terminal waits for a terminal or needs-attention state without waking on routine text. Optional reply delivers a same-session answer exactly once. Optional extend_* records an audited deadline extension. Disconnecting this waiter does not stop provider work. Unsolicited stdio callbacks across assistant turns are not available.${RESPONSE_MODE_HINT}`, + title: TOOL_METADATA.task.title, + annotations: TOOL_METADATA.task.annotations, + outputSchema: RUN_TOOL_OUTPUT_SCHEMA, + description: `Inspect or wait on one bounded native run using run_id and its returned cursor. Native run_request calls return the compact coordination receipt by default. Use view=diagnostics for the detailed run receipt; view=compact explicitly selects the normal compact run projection. task_id remains the compatible 3.2.1 path and uses event_cursor for expanded lane progress and diagnostics. wait_until=terminal waits for a terminal or needs-attention state without waking on routine text. Optional reply delivers a same-session answer exactly once. Optional extend_* records an audited deadline extension. Disconnecting this waiter does not stop provider work. Unsolicited stdio callbacks across assistant turns are not available.${RESPONSE_MODE_HINT}`, inputSchema: { type: 'object', properties: { @@ -214,7 +301,7 @@ const TOOLS = [ view: { type: 'string', enum: ['summary', 'diagnostics', 'compact'], - description: 'summary is the default receipt plus diagnostic envelope. compact is a bounded coordination payload without full task or runtime bodies. diagnostics is a bounded, redacted, cursor-paged evidence page and never waits.', + description: 'For native run_request calls, omitted, summary, and compact return the compact coordination receipt; diagnostics returns the detailed sanitized run receipt. The compatible task_id path retains its existing summary, compact, and cursor-paged diagnostics behavior.', }, cursor: { type: 'string', @@ -274,13 +361,29 @@ const TOOLS = [ run_reply: { type: 'object', additionalProperties: false, - required: ['batch_id', 'reply'], - description: 'Exactly-once run attention reply. Do not mix with 3.2.1 task.reply.', + description: 'Exactly-once run attention reply, a host approval reference, or a typed repository-consent continuation. Do not mix alternatives or 3.2.1 task.reply.', properties: { + approval_ref: { type: 'string', minLength: 1, maxLength: 4096, description: 'Opaque host-minted repository-exposure approval reference. It is bound to this run and never emitted in telemetry.' }, batch_id: { type: 'string', minLength: 1, maxLength: 128 }, expected_revision: { type: 'integer', minimum: 0 }, reply: { type: 'object' }, + request_consent: { type: 'boolean', const: true, description: 'Ask the host to reopen the native repository-access form for this run.' }, }, + oneOf: [ + { + required: ['batch_id', 'reply'], + not: { anyOf: [{ required: ['approval_ref'] }, { required: ['request_consent'] }] }, + }, + { + required: ['approval_ref'], + not: { anyOf: [{ required: ['batch_id'] }, { required: ['expected_revision'] }, { required: ['reply'] }, { required: ['request_consent'] }] }, + }, + { + required: ['request_consent'], + properties: { request_consent: { const: true } }, + not: { anyOf: [{ required: ['approval_ref'] }, { required: ['batch_id'] }, { required: ['expected_revision'] }, { required: ['reply'] }] }, + }, + ], }, }, allOf: [ @@ -289,13 +392,17 @@ const TOOLS = [ then: { required: ['run_id'] }, else: { required: ['task_id'] }, }, + { not: { required: ['run_id', 'task_id'] } }, ], additionalProperties: false, }, }, { name: 'tasks', - description: `List recent task receipts with optional compact keyset pagination and filters. With task_ids, wait concurrently for the first of 1-8 exact tasks to reach progress or terminal (including needs_attention), using optional per-task cursors and one bounded wait; a timeout returns compact current snapshots for every target. Wait-any task snapshots and live event previews are individually bounded; call task with a target ID for full event detail. Disconnecting the waiter does not stop providers.${RESPONSE_MODE_HINT}`, + title: TOOL_METADATA.tasks.title, + annotations: TOOL_METADATA.tasks.annotations, + outputSchema: RUN_TOOL_OUTPUT_SCHEMA, + description: `Wait on a bounded native run aggregate with run_id and its returned cursor, or use task_ids for the compatible 3.2.1 wait-any path. List recent task receipts with optional compact keyset pagination and filters when no wait options are supplied. A timeout returns bounded current snapshots; call task with a target ID for full event detail. Disconnecting the waiter does not stop providers.${RESPONSE_MODE_HINT}`, inputSchema: { type: 'object', properties: { @@ -391,7 +498,10 @@ const TOOLS = [ }, { name: 'cancel', - description: `Cancel one owned local process group or Cursor Cloud run.${RESPONSE_MODE_HINT}`, + title: TOOL_METADATA.cancel.title, + annotations: TOOL_METADATA.cancel.annotations, + outputSchema: RUN_TOOL_OUTPUT_SCHEMA, + description: `Cancel one owned bounded native Co-Engineer run with run_id; task_id remains the compatible 3.2.1 path. Cancellation preserves durable evidence and does not claim provider termination until observed.${RESPONSE_MODE_HINT}`, inputSchema: { type: 'object', properties: { @@ -421,6 +531,7 @@ const TOOLS = [ then: { required: ['run_id'] }, else: { required: ['task_id'] }, }, + { not: { required: ['run_id', 'task_id'] } }, ], additionalProperties: false, }, @@ -464,15 +575,25 @@ function projectWaitAnyEntry(entry) { function takePresentationArgs(args = {}) { const { response_mode: responseModeRaw, ...businessArgs } = args; return { - responseMode: normalizeResponseMode(responseModeRaw), + responseMode: responseModeRaw === undefined && clientSupportsStructuredResponses() + ? 'structured' + : normalizeResponseMode(responseModeRaw), args: businessArgs, }; } +function clientSupportsStructuredResponses() { + if (!clientCapabilities || typeof clientCapabilities !== 'object' || Array.isArray(clientCapabilities)) return false; + if (clientCapabilities.structuredContent === true) return true; + if (clientCapabilities.experimental?.structuredContent === true) return true; + return clientCapabilities.experimental?.['codex-co-engineer']?.structured_content === true; +} + function result(value, { responseMode } = {}) { const uiMeta = value?.mode === 'run' ? resolveExperienceResultMeta({ - card: value?.experience?.card ?? null, + card: value?.experience?.card ?? classifyExperienceCard(value), + experience: experienceForRunToolResult(value) ?? projectExperience(value), clientCapabilities, resources: uiResources(), }) @@ -490,15 +611,24 @@ async function callTool(name, args = {}, { signal, responseMode } = {}) { const root = stateRoot(); const classified = classifyRunToolCall(name, args); if (classified.mode === 'run') { - const value = await invokeRunTool(root, name, args, { signal }); + if (name === 'cancel' && typeof args?.run_id === 'string') { + nativeConsent.cancelRun(args.run_id); + } + const value = await invokeRunTool(root, name, args, { + signal, + requestConsent: nativeConsent.requestConsent, + }); if (value?.mode === 'legacy') { // Fall through only when classification and dispatch disagree; omission stays 3.2.1. } else { - return result(value, { responseMode }); + // Native run receipts are structured-first even when older hosts omit + // the optional capability advertisement. The bounded content fallback + // remains valid MCP text and points at the authoritative projection. + return result(value, { responseMode: responseMode ?? 'structured' }); } } if (name === 'status') { - const hasCompact = args && (args.detail !== undefined || args.task_limit !== undefined || args.include_tasks !== undefined); + const hasCompact = args && (args.detail !== undefined || args.task_limit !== undefined || args.include_tasks !== undefined || args.refresh !== undefined); if (!hasCompact) { const value = await supervisorStatus(root); return result({ ...value, tasks: value.tasks.map(publicTask) }, { responseMode }); @@ -575,11 +705,20 @@ function send(value) { process.stdout.write(`${JSON.stringify(value)}\n`); } +const nativeConsent = createNativeConsentTransport({ + send, + getCapabilities: () => clientCapabilities, + getProtocolVersion: () => negotiated, + grantStore: createConsentGrantStore({ root: stateRoot() }), +}); + async function handle(message) { if (!message || message.jsonrpc !== '2.0') return; + if (nativeConsent.handleMessage(message)) return; if (message.method === 'notifications/initialized') return; if (message.method === 'notifications/cancelled') { const requestId = message.params?.requestId ?? message.params?.id; + nativeConsent.cancelRequest(requestId); inflight.get(requestId)?.abort(); return; } @@ -593,6 +732,7 @@ async function handle(message) { result: { protocolVersion: negotiated, capabilities: serverCapabilities(), + instructions: SERVER_INSTRUCTIONS, serverInfo: { name: 'codex-co-engineer', title: 'Codex-Co-Engineer', version: VERSION }, }, }); @@ -648,14 +788,23 @@ async function handle(message) { const controller = new AbortController(); if (message.id !== undefined) inflight.set(message.id, controller); const { responseMode, args } = takePresentationArgs(message.params?.arguments ?? {}); + let effectiveResponseMode = responseMode; + try { + if (effectiveResponseMode == null + && classifyRunToolCall(message.params?.name, args).mode === 'run') { + effectiveResponseMode = 'structured'; + } + } catch { + // callTool performs authoritative validation and returns the typed error. + } let response; try { response = await callTool(message.params?.name, args, { signal: controller.signal, - responseMode, + responseMode: effectiveResponseMode, }); } catch (error) { - response = errorResult(error, { responseMode }); + response = errorResult(error, { responseMode: effectiveResponseMode }); } finally { inflight.delete(message.id); } @@ -678,3 +827,6 @@ input.on('line', (line) => { if (message?.id !== undefined) send({ jsonrpc: '2.0', id: message.id, error: { code: -32603, message: error?.message ?? 'Internal error' } }); }); }); +input.on('close', () => { + nativeConsent.close('disconnect'); +}); diff --git a/plugins/codex-co-engineer/mcp/v3/single-turn.flow.mjs b/plugins/codex-co-engineer/mcp/v3/single-turn.flow.mjs deleted file mode 100644 index dd12573..0000000 --- a/plugins/codex-co-engineer/mcp/v3/single-turn.flow.mjs +++ /dev/null @@ -1,12 +0,0 @@ -import { acp, defineFlow } from 'acpx/flows'; - -export default defineFlow({ - name: 'codex-co-engineer-single-turn', - startAt: 'delegate', - nodes: { - delegate: acp({ - prompt: ({ input }) => input.prompt, - }), - }, - edges: [], -}); diff --git a/plugins/codex-co-engineer/mcp/v3/supervisor.mjs b/plugins/codex-co-engineer/mcp/v3/supervisor.mjs index 281ebb8..0082144 100644 --- a/plugins/codex-co-engineer/mcp/v3/supervisor.mjs +++ b/plugins/codex-co-engineer/mcp/v3/supervisor.mjs @@ -24,7 +24,7 @@ import { COMPACT_VIEW, projectCompactTask, resolveTaskView } from './compact-tas import { deadlineReached, nextDeadlineExtension, resolveTaskDeadline } from './deadline.mjs'; import { compactSummary, compactTaskCard, diagnosticEnvelope, projectCompactStatus, readTaskDiagnostics } from './diagnostics.mjs'; import { completedWithoutLiveQuestionIdentity } from './grok-question-bridge.mjs'; -import { submitReply } from './mailbox.mjs'; +import { readAttention, submitReply } from './mailbox.mjs'; import { appendTaskEvent, clearTaskLaunchReservation, @@ -43,6 +43,7 @@ import { stateRoot, taskPaths, updateTask, + waitForAnyTaskProgress, waitForTaskProgress, writeRuntimeRecord, } from './task-store.mjs'; @@ -53,6 +54,7 @@ import { preflightCursorCloudOrigin, reconcileCursorCloudTask, } from './cursor-cloud-worker.mjs'; +import { reconnectAcpTask } from './acp-worker.mjs'; import { inspectExactProcessBoundary, launchProcessBoundary, @@ -69,20 +71,30 @@ import { deliverSupervisorSameSessionReplyV1, cancelSupervisorSameSessionReplyV1, } from './run-tool-adapter.mjs'; +import { createRunAdmissionRuntime } from './run-admission.mjs'; +import { createRunAdmissionStore } from './run-admission-store.mjs'; +import { + compileRunRequestV1, + RUN_REQUEST_DEFAULT_MODELS, +} from './run-request-compiler.mjs'; +import { loadReadinessSnapshot, saveReadinessSnapshot } from './readiness-snapshot.mjs'; +import { buildGitIdentityV1, buildWorkspaceIdentityV1 } from './protected-identity.mjs'; +import { assertRuntimeEntrypoints } from './runtime-entrypoints.mjs'; +import { BUNDLED_WORKTREE_BOOTSTRAP } from './worktree-bootstrap-runtime.mjs'; const execFile = promisify(nodeExecFile); const WORKER = path.join(path.dirname(fileURLToPath(import.meta.url)), 'acp-worker.mjs'); const CLOUD_WORKER = path.join(path.dirname(fileURLToPath(import.meta.url)), 'cursor-cloud-worker.mjs'); const ACTIVE = new Set(ACTIVE_STATUSES); const PROVIDERS = new Set(['grok', 'cursor-local', 'cursor-cloud', 'dsh']); -const DEFAULT_DSH_MODEL = 'muse-spark-1.2-contributor'; +const DEFAULT_DSH_MODEL = 'meta/muse-spark-1.3-contributor'; const DSH_MODELS = Object.freeze({ [DEFAULT_DSH_MODEL]: Object.freeze({ configEnv: 'CODEX_CO_ENGINEER_DSH_ACP_CONFIG', configFile: 'dsh-acp.yml', - credentialEnv: 'MODEL_API_KEY', - credentialFileEnv: 'CODEX_CO_ENGINEER_MODEL_API_KEY_FILE', - credentialFile: 'model-api-key', + credentialEnv: 'OPENROUTER_API_KEY', + credentialFileEnv: 'CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE', + credentialFile: 'openrouter-api-key', }), 'stealth/ox-alpha': Object.freeze({ configEnv: 'CODEX_CO_ENGINEER_DSH_OX_ACP_CONFIG', @@ -94,6 +106,10 @@ const DSH_MODELS = Object.freeze({ }); const WORKSPACE_MODES = new Set(['managed', 'direct']); const WORKTREE_CREATE_MAX_BUFFER = 16 * 1024 * 1024; +const SIMPLE_DISPATCH_EVIDENCE_TIMEOUT_MS = 5_000; +const SIMPLE_DISPATCH_EVIDENCE_POLL_MS = 25; +const PROVIDER_READINESS_TTL_MS = 30_000; +const PROVIDER_READINESS_PROBE_TIMEOUT_MS = 2_000; const PUBLIC_STARTUP_MESSAGES = Object.freeze({ credential_permissions: 'Provider credential configuration is invalid.', dsh_acp_not_configured: 'DSH ACP configuration is invalid.', @@ -111,6 +127,7 @@ const PUBLIC_STARTUP_MESSAGES = Object.freeze({ workspace_root_mismatch: 'The requested workspace path is not its Git worktree root.', workspace_dirty: 'The source worktree has uncommitted changes; clean it before managed delegation.', worktree_create_failed: 'The managed worktree could not be prepared.', + worktree_bootstrap_exact_sha_unsupported: 'The installed worktree-bootstrap does not support exact local-SHA creation; upgrade worktree-bootstrap before retrying.', worker_boundary_uncertain: 'The worker boundary could not be stopped; reconcile or cancel this task.', worker_boundary_pending: 'The worker process boundary is not yet final.', worker_boundary_missing: 'The worker process boundary receipt is missing.', @@ -134,6 +151,8 @@ const PUBLIC_STARTUP_MESSAGES = Object.freeze({ cgroup_not_empty: 'Owned systemd process boundary still has descendants after exact unit stop.', cancelled: 'The task was cancelled before worker startup.', provider_startup_failed: 'Provider startup could not be prepared.', + model_unattested: 'The configured provider route cannot select the requested model.', + prompt_envelope_missing: 'The compiled child prompt envelope is missing.', task_launch_busy: 'Another worker already owns this task launch.', local_boundary_unavailable: 'The local systemd/cgroup process boundary is unavailable.', systemd_user_manager_unavailable: 'The local systemd user manager is unavailable.', @@ -162,6 +181,7 @@ const PUBLIC_STARTUP_MESSAGES = Object.freeze({ cursor_cloud_start_ref_unavailable: 'Cursor Cloud requires an immutable starting commit.', cursor_cloud_start_ref_invalid: 'Cursor Cloud requires a full 40-character commit starting reference.', invalid_provider_repo: 'provider_repo_url is supported only for Cursor Cloud tasks.', + runtime_install_incomplete: 'The installed Codex-Co-Engineer runtime is incomplete. Reinstall the plugin, then restart Codex.', }); export class SupervisorError extends Error { @@ -203,6 +223,34 @@ function resolveDshModel(value) { return model; } +// The live Grok, Cursor Local, and Cursor Cloud transports currently expose +// only their configured provider defaults. Keep semantic model overrides from +// being presented as effective selections until a provider-specific control +// can select and attest them. DSH is routed through its existing config +// selection below. +function assertProviderModelDispatchable(provider, model) { + if (provider === 'dsh') { + resolveDshModel(model); + return; + } + if (model !== RUN_REQUEST_DEFAULT_MODELS[provider]) { + fail('model_unattested', 'The configured provider route cannot select the requested model.'); + } +} + +function resolveTaskModel(provider, model, dshModel) { + if (model === undefined) return undefined; + if (provider === 'dsh') { + const resolved = resolveDshModel(model); + if (dshModel !== undefined && dshModel !== resolved) { + fail('invalid_dsh_model', 'The DSH model selections do not agree.'); + } + return resolved; + } + assertProviderModelDispatchable(provider, model); + return model; +} + function providerArgv(provider, env = process.env, dshModel) { if (provider === 'grok') return [env.CODEX_CO_ENGINEER_GROK_COMMAND ?? 'grok', 'agent', '--always-approve', 'stdio']; if (provider === 'cursor-local') return [env.CODEX_CO_ENGINEER_CURSOR_COMMAND ?? 'cursor-agent', 'acp']; @@ -300,13 +348,20 @@ function missingWorkspaceError(error) { return /(?:no such file|cannot change to|does not exist)/iu.test(`${error?.message ?? ''} ${error?.stderr ?? ''}`); } -async function validateManagedSource({ repo, execute = execFile }) { +async function validateManagedSource({ repo, baseSha, execute = execFile }) { normalizedAbsolute(repo, 'repo'); let branchOutput; let statusOutput; + let headOutput; try { - ({ stdout: branchOutput } = await execute('git', ['-C', repo, 'branch', '--show-current'], { encoding: 'utf8' })); - ({ stdout: statusOutput } = await execute('git', ['-C', repo, 'status', '--porcelain=v1', '--untracked-files=all'], { encoding: 'utf8' })); + const checks = await Promise.all([ + execute('git', ['-C', repo, 'branch', '--show-current'], { encoding: 'utf8' }), + execute('git', ['-C', repo, 'status', '--porcelain=v1', '--untracked-files=all'], { encoding: 'utf8' }), + ...(baseSha ? [execute('git', ['-C', repo, 'rev-parse', '--verify', `${baseSha}^{commit}`], { encoding: 'utf8' })] : []), + ]); + ({ stdout: branchOutput } = checks[0]); + ({ stdout: statusOutput } = checks[1]); + if (baseSha) ({ stdout: headOutput } = checks[2]); } catch (error) { const code = missingWorkspaceError(error) ? 'workspace_missing' : 'workspace_invalid'; throw new SupervisorError(code, code === 'workspace_missing' @@ -314,16 +369,23 @@ async function validateManagedSource({ repo, execute = execFile }) { : 'The source workspace is not a valid Git worktree.', { cause: error }); } const branch = String(branchOutput ?? '').trim(); - if (!branch) fail('workspace_branch_missing', 'The source workspace must be attached to a branch.'); + if (!branch && !baseSha) fail('workspace_branch_missing', 'The source workspace must be attached to a branch.'); if (managedSourceDirty(statusOutput)) { fail('workspace_dirty', 'The source worktree must be clean before managed delegation.'); } - return { branch }; + if (baseSha) { + if (!/^[0-9a-f]{40}$/iu.test(baseSha)) fail('workspace_start_ref_invalid', 'The exact local base must be a full commit SHA.'); + if (String(headOutput ?? '').trim().toLowerCase() !== baseSha.toLowerCase()) { + fail('workspace_start_ref_invalid', 'The exact local base could not be resolved to the requested commit.'); + } + } + return { branch, base_sha: baseSha ?? null }; } async function validateWorkspaceContract(workspace, taskId, { execute = execFile, checkPath = stat, + expectedStartSha, } = {}) { if (!workspace || typeof workspace !== 'object' || Array.isArray(workspace)) { fail('workspace_missing', 'Managed delegation did not return a workspace.'); @@ -348,6 +410,9 @@ async function validateWorkspaceContract(workspace, taskId, { if (!/^[0-9a-f]{40}$/iu.test(workspace.start_sha)) { fail('workspace_start_ref_invalid', 'Managed delegation returned an invalid starting commit.'); } + if (expectedStartSha !== undefined && workspace.start_sha.toLowerCase() !== String(expectedStartSha).toLowerCase()) { + fail('workspace_start_ref_invalid', 'Managed delegation returned a workspace at the wrong exact starting commit.'); + } let metadata; try { metadata = await checkPath(worktreePath); @@ -396,17 +461,35 @@ async function validateWorkspaceContract(workspace, taskId, { }; } -export async function createWriterWorkspace({ taskId, repo, execute = execFile, checkPath = stat }) { +export async function createWriterWorkspace({ taskId, repo, baseSha, execute = execFile, checkPath = stat }) { requireTaskId(taskId); - const source = await validateManagedSource({ repo, execute }); + const source = await validateManagedSource({ repo, baseSha, execute }); try { - const base = source.branch; + const exactSha = baseSha ?? null; + const base = exactSha ?? source.branch; if (!base) fail('workspace_branch_missing', 'Writer source must be attached to a branch.'); - const { stdout } = await execute('worktree-bootstrap', ['create', taskId, '--repo', repo, '--base', base], { + const argv = ['create', taskId, '--repo', repo, '--base', base]; + // Exact-SHA creation is deliberately delegated to the official + // worktree-bootstrap capability. There is no raw `git worktree` fallback: + // the bootstrap owns locks, branch identity, and handoff semantics. + if (exactSha) argv.push('--local-only'); + let result; + try { + result = await execute(BUNDLED_WORKTREE_BOOTSTRAP, argv, { encoding: 'utf8', maxBuffer: WORKTREE_CREATE_MAX_BUFFER, + }); + } catch (error) { + if (exactSha && /(?:unknown option|unrecognized option|invalid option|local-only)/iu.test(`${error?.message ?? ''} ${error?.stderr ?? ''}`)) { + throw new SupervisorError('worktree_bootstrap_exact_sha_unsupported', 'The installed worktree-bootstrap does not support exact local-SHA creation.', { cause: error }); + } + throw error; + } + return await validateWorkspaceContract(parseWorktreeResult(result.stdout, taskId), taskId, { + execute, + checkPath, + ...(exactSha ? { expectedStartSha: exactSha } : {}), }); - return await validateWorkspaceContract(parseWorktreeResult(stdout, taskId), taskId, { execute, checkPath }); } catch (error) { if (error instanceof SupervisorError) throw error; throw new SupervisorError('worktree_create_failed', error?.stderr?.trim() || error?.message || 'worktree-bootstrap failed.', { cause: error }); @@ -475,7 +558,7 @@ export async function cleanupManagedWorkspace({ workspace, taskId, execute = exe const reference = workspaceReference(workspace, taskId); if (!reference) return { state: 'unavailable', cleaned: false }; try { - const { stdout } = await execute('worktree-bootstrap', [ + const { stdout } = await execute(BUNDLED_WORKTREE_BOOTSTRAP, [ 'lock', 'inspect', reference.task, '--repo', reference.worktree_path, ], { encoding: 'utf8', maxBuffer: 1024 * 1024 }); const lock = parseJsonSuffix(stdout); @@ -494,7 +577,7 @@ export async function cleanupManagedWorkspace({ workspace, taskId, execute = exe if (health.state !== 'abandoned' || typeof lock.lock_id !== 'string' || lock.lock_id.length === 0) { return { state: health.state ?? lock.state ?? 'unknown', cleaned: false }; } - await execute('worktree-bootstrap', [ + await execute(BUNDLED_WORKTREE_BOOTSTRAP, [ 'lock', 'clean', reference.task, '--repo', reference.worktree_path, '--policy', 'dead-local', @@ -597,7 +680,7 @@ export async function launchWorker({ const log = await open(paths.log, 'a', 0o600); const worker = provider === 'cursor-cloud' ? CLOUD_WORKER : WORKER; const workerArgv = [process.execPath, '--no-warnings', worker, '--request', paths.request]; - const command = writer ? 'worktree-bootstrap' : workerArgv.shift(); + const command = writer ? BUNDLED_WORKTREE_BOOTSTRAP : workerArgv.shift(); const args = writer ? ['launch', taskId, '--repo', cwd, '--', ...workerArgv] : workerArgv; @@ -642,7 +725,7 @@ export async function launchWorker({ pid: child.pid, process_group: boundary ? null : child.pid, process_start_ticks: processStartTicks(child.pid), - command: writer ? 'worktree-bootstrap' : process.execPath, + command: writer ? BUNDLED_WORKTREE_BOOTSTRAP : process.execPath, ...(boundary ? { process_boundary: boundary.receipt } : {}), }); await appendTaskEvent(root, taskId, { type: 'worker', state: 'spawned', pid: child.pid }); @@ -658,7 +741,7 @@ export async function launchWorker({ pid: child.pid, process_group: null, process_start_ticks: processStartTicks(child.pid), - command: writer ? 'worktree-bootstrap' : process.execPath, + command: writer ? BUNDLED_WORKTREE_BOOTSTRAP : process.execPath, process_boundary: boundary.receipt, updated_at: new Date().toISOString(), }; @@ -693,6 +776,7 @@ export async function submitTask(input, dependencies = {}) { fail('invalid_dsh_model', 'dsh_model is supported only for DSH tasks.'); } const dshModel = input.provider === 'dsh' ? resolveDshModel(input.dsh_model) : undefined; + const taskModel = resolveTaskModel(input.provider, input.model, dshModel); if (typeof input.prompt !== 'string' || input.prompt.trim().length === 0) fail('invalid_prompt', 'prompt must be non-empty text.'); const role = input.role ?? 'implement'; if (!['review', 'implement'].includes(role)) fail('invalid_role', 'role must be review or implement.'); @@ -724,6 +808,14 @@ export async function submitTask(input, dependencies = {}) { if (error instanceof SupervisorError) throw error; if (error?.code !== 'ENOENT') throw error; } + try { + await (dependencies.preflightRuntime ?? assertRuntimeEntrypoints)(input.provider); + } catch { + throw publicStartupError( + new SupervisorError('runtime_install_incomplete', 'The installed runtime is incomplete.'), + 'runtime_install_incomplete', + ); + } if (input.provider !== 'cursor-cloud') { requireLocalBoundary(await localBoundaryReadiness(dependencies.probeBoundary)); } @@ -739,13 +831,33 @@ export async function submitTask(input, dependencies = {}) { let cloudPreflight = null; let taskCreated = false; try { - workspace = managed - ? await (dependencies.createWorkspace ?? createWriterWorkspace)({ - taskId: id, - repo: input.repo, - ...(dependencies.createWorkspace ? {} : { execute: dependencies.execute, checkPath: dependencies.checkPath }), - }) - : await readerWorkspace(input.repo, dependencies.execute); + const preparedWorkspace = dependencies.preparedWorkspace; + if (preparedWorkspace !== undefined) { + workspace = preparedWorkspace; + if (managed) { + workspace = await validateWorkspaceContract(workspace, id, { + execute: dependencies.execute, + checkPath: dependencies.checkPath, + expectedStartSha: dependencies.baseSha, + }); + } else { + const verified = await readerWorkspace(input.repo, dependencies.execute); + if (workspace?.worktree_path !== verified.worktree_path + || workspace?.start_sha?.toLowerCase() !== verified.start_sha?.toLowerCase()) { + fail('workspace_invalid', 'The prepared provider workspace changed before launch.'); + } + workspace = verified; + } + } else { + workspace = managed + ? await (dependencies.createWorkspace ?? createWriterWorkspace)({ + taskId: id, + repo: input.repo, + ...(dependencies.baseSha !== undefined ? { baseSha: dependencies.baseSha } : {}), + ...(dependencies.createWorkspace ? {} : { execute: dependencies.execute, checkPath: dependencies.checkPath }), + }) + : await readerWorkspace(input.repo, dependencies.execute); + } // The built-in bootstrap already returns a verified contract. Re-verify // only injected workspace factories so normal dispatch does not repeat a // stat plus three Git subprocesses on every managed task. @@ -779,6 +891,15 @@ export async function submitTask(input, dependencies = {}) { status: 'accepted', provider: input.provider, ...(dshModel ? { dsh_model: dshModel } : {}), + ...(taskModel ? { model: taskModel } : {}), + ...(typeof input.run_id === 'string' ? { run_id: input.run_id } : {}), + ...(typeof input.assignment_id === 'string' ? { assignment_id: input.assignment_id } : {}), + ...(typeof input.access === 'string' ? { access: input.access } : {}), + ...(Array.isArray(input.write_scope) ? { write_scope: [...input.write_scope] } : {}), + ...(Array.isArray(input.capabilities) ? { capabilities: [...input.capabilities] } : {}), + ...(typeof input.child_envelope_digest === 'string' + ? { child_envelope_digest: input.child_envelope_digest } + : {}), role, source_repo: input.repo, cwd: workspace.worktree_path, @@ -1142,7 +1263,7 @@ async function inspectManagedLockState(task, runtime, dependencies = {}) { if (!reference) return { lock: 'unknown', code: 'worktree_lock_inspect_failed', cleaned: false }; const execute = dependencies.execute ?? execFile; try { - const { stdout } = await execute('worktree-bootstrap', [ + const { stdout } = await execute(BUNDLED_WORKTREE_BOOTSTRAP, [ 'lock', 'inspect', reference.task, '--repo', reference.worktree_path, ], { encoding: 'utf8', maxBuffer: 1024 * 1024 }); const lock = parseJsonSuffix(stdout); @@ -1166,7 +1287,7 @@ async function cleanManagedLockAfterBoundary(task, runtime, dependencies = {}) { const reference = workspaceReference(task, task.worktree_task ?? task.id); const execute = dependencies.execute ?? execFile; try { - await execute('worktree-bootstrap', [ + await execute(BUNDLED_WORKTREE_BOOTSTRAP, [ 'lock', 'clean', reference.task, '--repo', reference.worktree_path, '--policy', 'dead-local', @@ -1231,9 +1352,18 @@ export async function settleLocalTaskLifecycle(root, task, runtime, dependencies } const sleep = dependencies.sleep ?? wait; - const drainMs = Number.isFinite(dependencies.drainGraceMs) - ? dependencies.drainGraceMs - : PROCESS_BOUNDARY_LIFECYCLE_BOUNDS_MS.natural_boundary_and_lock_drain; + // Grant the natural worker/lock drain once, on the first terminal + // reconciliation. Only supervisor boundary/lock evidence proves that a + // prior reconciliation already waited; worker ACP/handoff cleanup does not. Repeating the grace on every + // readiness/status call made retained unknown receipts add two seconds + // apiece before any fresh inspection. + const reconciled = Object.hasOwn(current.cleanup ?? {}, 'boundary') + && Object.hasOwn(current.cleanup ?? {}, 'lock'); + const drainMs = reconciled + ? 0 + : Number.isFinite(dependencies.drainGraceMs) + ? dependencies.drainGraceMs + : PROCESS_BOUNDARY_LIFECYCLE_BOUNDS_MS.natural_boundary_and_lock_drain; if (drainMs > 0) await sleep(drainMs); let inspection = await inspectRuntimeBoundary(boundRuntime, dependencies); @@ -1745,22 +1875,556 @@ export function projectSupervisorTaskRecords(tasks) { return tasks.map((task) => projectSupervisorTerminalReceipt(task)); } -async function probeCommand(command, args, authenticatedPattern, env) { +async function verifySimpleRunRepository({ git, execute = execFile } = {}) { + const repository = git?.repository_path; + if (typeof repository !== 'string' || !path.isAbsolute(repository) || path.resolve(repository) !== repository) { + return { verified: false, reason: 'repository_invalid' }; + } + try { + const read = (args) => execute('git', ['-C', repository, ...args], { encoding: 'utf8' }); + const [root, head, tree, base, status] = await Promise.all([ + read(['rev-parse', '--show-toplevel']), + read(['rev-parse', '--verify', 'HEAD^{commit}']), + read(['rev-parse', '--verify', 'HEAD^{tree}']), + read(['rev-parse', '--verify', `${git.base_sha}^{commit}`]), + read(['status', '--porcelain=v1', '--untracked-files=all']), + ]); + const rootValue = String(root?.stdout ?? '').trim(); + const headValue = String(head?.stdout ?? '').trim().toLowerCase(); + const treeValue = String(tree?.stdout ?? '').trim().toLowerCase(); + const baseValue = String(base?.stdout ?? '').trim().toLowerCase(); + const clean = String(status?.stdout ?? '').trim() === ''; + const verified = rootValue === repository + && clean + && headValue === String(git.head_sha ?? '').toLowerCase() + && treeValue === String(git.tree_sha ?? '').toLowerCase() + && baseValue === String(git.base_sha ?? '').toLowerCase(); + return { + verified, + ...(verified ? {} : { reason: clean ? 'repository_identity_changed' : 'repository_dirty' }), + }; + } catch { + return { verified: false, reason: 'repository_invalid' }; + } +} + +function simpleChangedFiles(status) { + return String(status ?? '').split(/\r?\n/u) + .filter((line) => line.length > 2) + .map((line) => line.slice(3).trim()) + .filter(Boolean) + .slice(0, 64); +} + +function simpleSilenceTimeout(role) { + return role === 'implement' ? 600_000 : 300_000; +} + +async function inspectSimpleWorkspace({ root, task_id: taskId, workspace, execute = execFile } = {}) { + const cwd = workspace?.worktree_path; + if (typeof cwd !== 'string' || !path.isAbsolute(cwd) || path.resolve(cwd) !== cwd) return {}; + const read = (args) => execute('git', ['-C', cwd, ...args], { encoding: 'utf8' }); try { - const { stdout, stderr } = await execFile(command, args, { - cwd: '/tmp', encoding: 'utf8', timeout: 5_000, maxBuffer: 256 * 1024, + const startSha = typeof workspace.start_sha === 'string' ? workspace.start_sha : null; + const [head, branch, status, commits] = await Promise.all([ + read(['rev-parse', 'HEAD']), + read(['branch', '--show-current']), + read(['status', '--porcelain=v1', '--untracked-files=all']), + startSha ? read(['log', '--format=%H', `${startSha}..HEAD`]).catch(() => ({ stdout: '' })) : Promise.resolve({ stdout: '' }), + ]); + const currentHead = String(head?.stdout ?? '').trim(); + const changed = simpleChangedFiles(status?.stdout); + const commitValues = String(commits?.stdout ?? '').split(/\r?\n/u).filter(Boolean).slice(0, 64); + const task = typeof taskId === 'string' + ? await readTask(root ?? stateRoot(), taskId).then((result) => result.task).catch(() => null) + : null; + return { + current_head: currentHead || null, + branch: String(branch?.stdout ?? '').trim() || (workspace.branch ?? null), + clean: changed.length === 0, + changed_files: changed, + commits: commitValues, + partial_diff: changed.length > 0 || (startSha !== null && currentHead.toLowerCase() !== startSha.toLowerCase()), + last_acknowledged_provider_event: task?.dispatch_evidence === 'authoritative' + ? 'prompt_dispatched' + : (typeof task?.last_event === 'string' ? task.last_event : null), + }; + } catch { + return {}; + } +} + +async function workspaceLockId(workspace, execute) { + if (typeof workspace?.task !== 'string' || typeof workspace?.worktree_path !== 'string') return null; + try { + const result = await execute(BUNDLED_WORKTREE_BOOTSTRAP, [ + 'lock', 'inspect', workspace.task, '--repo', workspace.worktree_path, + ], { encoding: 'utf8', maxBuffer: 64 * 1024 }); + const lock = parseJsonSuffix(result?.stdout); + if (lock?.state !== 'active' || typeof lock.lock_id !== 'string' || lock.lock_id.length < 8) return null; + return lock.lock_id; + } catch { + return null; + } +} + +async function buildSimpleWorkspaceIdentity({ run_id: runId, assignment, git, workspace, execute }) { + try { + const cloud = assignment.provider === 'cursor-cloud'; + const lockId = cloud ? null : await workspaceLockId(workspace, execute); + if (!cloud && lockId === null) return null; + return buildWorkspaceIdentityV1({ + run_id: runId, + assignment_id: assignment.assignment_id, + git, + semantics: cloud ? 'remote_provider_managed' : 'local_managed_worktree', + starting_point: cloud ? 'pinned_pushed_sha' : 'run_base_sha', + worktree_path: cloud ? null : workspace?.worktree_path, + branch: cloud ? null : workspace?.branch, + lock_id: cloud ? null : lockId, + starting_ref: cloud ? (assignment.starting_ref ?? git.base_sha) : null, + }); + } catch { + // Workspace identity is protected telemetry. The launch path has already + // re-verified the worktree contract; an unavailable lock observation must + // not fabricate an identity or replay a prompt. + return null; + } +} + +async function waitForSimpleDispatchEvidence(root, taskId, { + timeoutMs = SIMPLE_DISPATCH_EVIDENCE_TIMEOUT_MS, + sleep = (milliseconds) => new Promise((resolve) => setTimeout(resolve, milliseconds)), + now = Date.now, +} = {}) { + const deadline = now() + Math.max(0, timeoutMs); + let current = null; + while (true) { + current = (await readTask(root, taskId)).task; + const sessionId = current.acp_session_id ?? current.provider_run_id ?? current.provider_agent_id ?? null; + if (current.dispatch_evidence === 'authoritative') { + return { + dispatched: true, + prompt_dispatched: true, + confidence: 'authoritative', + session_ready: true, + session_id: typeof sessionId === 'string' ? sessionId : null, + cursor: '0', + }; + } + if (['failed', 'timeout', 'cancelled', 'transport_lost'].includes(current.status) + && current.dispatch_intent !== true && current.prompt_dispatched !== true) { + return { + dispatched: false, + confidence: 'authoritative', + error: current.error ?? { code: 'provider_start_failed' }, + }; + } + if (['completed', 'succeeded', 'failed', 'timeout', 'cancelled', 'transport_lost'].includes(current.status)) { + return { + dispatched: false, + dispatch_uncertain: true, + terminal: true, + confidence: 'uncertain', + sent: current.dispatch_intent === true || current.prompt_dispatched === true, + session_ready: Boolean(sessionId), + session_id: typeof sessionId === 'string' ? sessionId : null, + ...(typeof current.error?.code === 'string' ? { error: { code: current.error.code } } : {}), + }; + } + if (now() >= deadline) { + const active = ['running', 'starting', 'accepted', 'needs_attention'].includes(current.status); + return { + dispatched: false, + ...(active ? { dispatch_pending: true } : { terminal: true }), + dispatch_uncertain: true, + confidence: 'uncertain', + sent: current.dispatch_intent === true || current.prompt_dispatched === true, + session_ready: Boolean(sessionId), + session_id: typeof sessionId === 'string' ? sessionId : null, + }; + } + await sleep(Math.min(SIMPLE_DISPATCH_EVIDENCE_POLL_MS, Math.max(1, deadline - now()))); + } +} + +function childEnvelopePrompt(assignment) { + const envelopeText = assignment?.child_envelope?.envelope_text; + if (typeof envelopeText !== 'string' || envelopeText.length === 0) { + fail('prompt_envelope_missing', 'The compiled child prompt envelope is missing.'); + } + return envelopeText; +} + +function stalledTaskAttention(task) { + const sessionId = task?.acp_session_id ?? task?.provider_run_id ?? task?.provider_agent_id ?? `local-${task?.id}`; + return { + session_id: String(sessionId).slice(0, 128), + question_id: `stalled-${task.id}`, + stage: 'stalled', + prompt: 'No meaningful provider event was observed before the configured silence threshold.', + required: true, + }; +} + +async function waitForSupervisorRunProgress(root, { task_ids, cursors, wait_ms, wait_until, signal } = {}) { + if (!Array.isArray(task_ids) || task_ids.length === 0) { + const delayMs = Number.isFinite(wait_ms) && wait_ms > 0 ? wait_ms : 0; + await new Promise((resolve) => { + let timer = null; + const finish = () => { + if (timer !== null) clearTimeout(timer); + signal?.removeEventListener('abort', finish); + resolve(); + }; + if (signal?.aborted || delayMs === 0) { + finish(); + return; + } + timer = setTimeout(finish, delayMs); + signal?.addEventListener('abort', finish, { once: true }); + }); + return { wait_reason: signal?.aborted ? 'disconnected' : 'timeout' }; + } + try { + return await waitForAnyTaskProgress(root, { + task_ids, + cursors, + wait_ms, + wait_until: wait_until === 'terminal' ? 'terminal' : 'progress', + signal, + }); + } catch (error) { + return { wait_reason: 'observation_uncertain', error: { code: error?.code ?? 'task_observation_unavailable' } }; + } +} + +async function inspectSupervisorLane(root, taskId) { + // The run bridge needs the authoritative task result and lifecycle overlay. + // Compact task projection intentionally omits legacy `completed` results. + const result = await inspectTask(root, { task_id: taskId, view: 'summary', wait_ms: 0 }); + let task = result.task ?? result; + const waitReason = result.progress?.wait_reason; + if (waitReason === 'silence' && task.status !== 'needs_attention' + && !['completed', 'succeeded', 'failed', 'timeout', 'timed_out', 'cancelled', + 'transport_lost', 'environment_blocked', 'cancelling'].includes(task.status)) { + const attention = stalledTaskAttention(task); + task = await updateTask(root, taskId, { status: 'needs_attention', attention }); + await appendTaskEvent(root, taskId, { + type: 'stalled', + session_id: attention.session_id, + question_id: attention.question_id, + silence_threshold: task.silence_timeout_ms ?? null, + }).catch(() => {}); + } + let attention = null; + if (task.status === 'needs_attention') { + attention = await readAttention(root, taskId).catch(() => null) ?? task.attention ?? null; + } + return { result, task, waitReason, attention }; +} + +function createSupervisorRunAdmissionRuntime(options = {}) { + const root = options.root ?? stateRoot(); + const execute = options.execute ?? execFile; + const env = options.env ?? process.env; + const checkPath = options.checkPath ?? stat; + const admissionStore = options.admissionStore ?? createRunAdmissionStore(root); + const submitTaskFn = options.submitTask ?? submitTask; + const waitForDispatchEvidence = options.waitForDispatchEvidence + ?? ((stateRootValue, taskId) => waitForSimpleDispatchEvidence(stateRootValue, taskId, { + ...(options.dispatchEvidenceTimeoutMs !== undefined + ? { timeoutMs: options.dispatchEvidenceTimeoutMs } : {}), + ...(options.dispatchEvidenceSleep ? { sleep: options.dispatchEvidenceSleep } : {}), + ...(options.dispatchEvidenceNow ? { now: options.dispatchEvidenceNow } : {}), + })); + const simpleDeps = { + compile: options.compile ?? compileRunRequestV1, + ...(options.requestConsent ? { requestConsent: options.requestConsent } : {}), + ...(options.verifyConsent ? { verifyConsent: options.verifyConsent } : {}), + providerReady: options.providerReady ?? (async ({ assignment }) => { + assertProviderModelDispatchable(assignment.provider, assignment.model); + await (options.preflightRuntime ?? assertRuntimeEntrypoints)(assignment.provider); + try { + await workerEnvironment(assignment.provider, env, assignment.provider === 'dsh' ? assignment.model : undefined); + return { ready: true }; + } catch (error) { + return { ready: false, reason: error?.code ?? 'provider_not_ready' }; + } + }), + processBoundaryReady: options.processBoundaryReady ?? (() => localBoundaryReadiness(options.probeBoundary)), + verifyRepository: options.verifyRepository ?? ((request) => verifySimpleRunRepository({ ...request, execute })), + prepareWorkspace: options.prepareWorkspace ?? (async ({ assignment, git }) => { + try { + const repository = git.repository_path; + if (assignment.provider === 'cursor-cloud') { + return { prepared: true, workspace: await readerWorkspace(repository, execute) }; + } + const baseSha = git.base_sha; + const workspace = await createWriterWorkspace({ + taskId: assignment.task_id, + repo: repository, + ...(baseSha ? { baseSha } : {}), + execute, + checkPath, + }); + return { prepared: true, workspace }; + } catch (error) { + return { prepared: false, error }; + } + }), + cleanupWorkspace: options.cleanupWorkspace ?? (({ workspace, assignment_id: assignmentId }) => cleanupManagedWorkspace({ + workspace, + taskId: workspace?.task ?? assignmentId, + execute, + })), + // The current provider workers create/attach the persistent session as + // part of launch. The dispatch barrier still records this reservation, + // then replaces the placeholder with the worker's session identity once + // authoritative prompt evidence is durable. + createSession: options.createSession ?? (async () => ({ ready: true, session_id: null })), + dispatchPrompt: options.dispatchPrompt ?? (async ({ run_id: runId, assignment, workspace, git }) => { + const input = { + task_id: assignment.task_id, + run_id: runId, + assignment_id: assignment.assignment_id, + provider: assignment.provider, + model: assignment.model, + repo: git.repository_path, + prompt: childEnvelopePrompt(assignment), + role: assignment.role === 'verify' ? 'review' : assignment.role, + access: assignment.access, + write_scope: [...assignment.write_scope], + capabilities: [...assignment.capabilities], + child_envelope_digest: assignment.prompt_envelope_digest, + expected_duration_ms: assignment.expected_duration_ms, + silence_timeout_ms: simpleSilenceTimeout(assignment.role), + workspace_mode: assignment.provider === 'cursor-cloud' ? 'direct' : 'managed', + }; + assertProviderModelDispatchable(assignment.provider, assignment.model); + if (assignment.provider === 'dsh') input.dsh_model = assignment.model; + if (assignment.provider === 'cursor-cloud' && assignment.starting_ref) input.starting_ref = assignment.starting_ref; + const submitted = await submitTaskFn(input, { + root, + env, + execute, + checkPath, + preparedWorkspace: workspace, + ...(assignment.provider !== 'cursor-cloud' ? { baseSha: git.base_sha } : {}), + }); + const evidence = await waitForDispatchEvidence(root, submitted.task.id); + const workspaceIdentity = await buildSimpleWorkspaceIdentity({ + run_id: runId, + assignment, + git: buildGitIdentityV1({ + repository_path: git.repository_path, + base_sha: git.base_sha, + }), + workspace, + execute, + }); + return { + ...evidence, + ...(workspaceIdentity ? { workspace_identity: workspaceIdentity } : {}), + }; + }), + replyAttention: options.replyAttention ?? (async ({ task_id: taskId, attention, reply, capability_satisfied: capabilitySatisfied }) => { + const items = Array.isArray(attention?.items) + ? attention.items + : (attention ? [attention] : []); + const answers = Array.isArray(reply?.answers) + ? reply.answers + : (Array.isArray(reply?.reply?.answers) + ? reply.reply.answers + : (reply?.question_id ? [reply] : [])); + let delivered = 0; + const deliveredQuestions = new Set(); + for (const item of items) { + const targets = Array.isArray(item?.targets) && item.targets.length > 0 + ? item.targets + : [item]; + const canonicalAnswer = answers.find((candidate) => ( + (candidate?.task_id === item?.task_id && candidate?.question_id === item?.question_id) + || (candidate?.assignment_id === item?.assignment_id && candidate?.question_id === item?.question_id) + )); + for (const target of targets) { + const answer = answers.find((candidate) => ( + (candidate?.task_id === target?.task_id && candidate?.question_id === target?.question_id) + || (candidate?.assignment_id === target?.assignment_id && candidate?.question_id === target?.question_id) + )) ?? canonicalAnswer; + if (!answer || typeof target?.task_id !== 'string') continue; + const sessionId = answer.session_id ?? target.session_id ?? item.session_id; + const questionId = answer.question_id ?? target.question_id ?? item.question_id; + const response = answer.response + ?? (capabilitySatisfied === true ? { + outcome: 'allow_once', + capability: item.capability, + resource: item.resource, + action: item.action, + } : undefined); + if (typeof sessionId !== 'string' || typeof questionId !== 'string' || response === undefined) continue; + const key = `${target.task_id}\u0000${questionId}`; + if (deliveredQuestions.has(key)) continue; + deliveredQuestions.add(key); + try { + await submitReply(root, target.task_id, { + session_id: sessionId, + question_id: questionId, + response, + }); + delivered += 1; + } catch (error) { + if (error?.code === 'reply_already_recorded') delivered += 1; + } + } + } + const expected = items.reduce((count, item) => count + ( + Array.isArray(item?.targets) && item.targets.length > 0 ? item.targets.length : 1 + ), 0); + return { delivered: expected > 0 && delivered === expected }; + }), + waitForProgress: options.waitForProgress ?? ((request) => waitForSupervisorRunProgress(root, request)), + inspectLane: options.inspectLane ?? (async ({ task_id: taskId }) => { + const inspected = await inspectSupervisorLane(root, taskId); + const { result, task } = inspected; + const status = task?.status; + if (typeof status !== 'string') { + throw Object.assign(new Error('Supervisor task observation is malformed.'), { code: 'task_observation_invalid' }); + } + const cursor = typeof result?.progress?.event_cursor === 'string' + ? result.progress.event_cursor : '0'; + const observedTaskId = task.id ?? task.task_id; + if (observedTaskId !== undefined && observedTaskId !== taskId) { + throw Object.assign(new Error('Supervisor task observation identity did not match the requested task.'), { + code: 'task_observation_identity_mismatch', + }); + } + const observed = { + task_id: observedTaskId ?? taskId, + cursor, + ...(task.dispatch_evidence === 'authoritative' + ? { dispatch_evidence: 'authoritative' } : {}), + ...(task.prompt_dispatched === true ? { prompt_dispatched: true } : {}), + ...(task.dispatch_uncertain === true ? { dispatch_uncertain: true } : {}), + ...((task.acp_session_id ?? task.provider_run_id ?? task.provider_agent_id) !== undefined + ? { session_id: task.acp_session_id ?? task.provider_run_id ?? task.provider_agent_id } : {}), + ...(typeof task.last_event === 'string' ? { last_event: task.last_event } : {}), + ...(task.result !== undefined && task.result !== null + ? { result: task.result } : {}), + }; + if (status === 'completed' || status === 'succeeded') return { ...observed, status: 'completed' }; + if (status === 'needs_attention') return { ...observed, status: 'needs_attention', attention: inspected.attention }; + if (status === 'cancelled') return { ...observed, status: 'cancelled' }; + if (status === 'timeout' || status === 'timed_out') return { ...observed, status: 'timeout' }; + if (status === 'transport_lost') return { ...observed, status: 'transport_lost' }; + if (status === 'environment_blocked') return { ...observed, status: 'environment_blocked' }; + if (status === 'failed') return { ...observed, status: 'failed', error: { code: task.error?.code } }; + if (status === 'running' || status === 'starting' || status === 'accepted' || status === 'cancelling') return { ...observed, status: 'running' }; + throw Object.assign(new Error('Supervisor task observation has an unknown status.'), { code: 'task_observation_invalid' }); + }), + reconnectLane: options.reconnectLane ?? (async ({ task_id: taskId }) => { + const { task } = await readTask(root, taskId); + if (task.provider === 'grok' || task.provider === 'cursor-local') { + return reconnectAcpTask({ root, taskId }); + } + if (task.provider === 'cursor-cloud' && task.provider_agent_id) { + try { + const current = await reconcileCursorCloudTask({ root, taskId }); + return { + reconnected: !['failed', 'timeout', 'cancelled', 'transport_lost'].includes(current?.status), + session_id: task.provider_agent_id, + }; + } catch { + return { reconnected: false, reason: 'provider_run_reconnect_failed' }; + } + } + return { reconnected: false, reason: 'provider_transport_no_resume' }; + }), + cancelLane: options.cancelLane ?? (async ({ task_id: taskId }) => { + const task = await cancelTask(root, taskId); + return { + confirmed: task?.status === 'cancelled', + cancelled: task?.status === 'cancelled', + task_id: task?.id ?? taskId, + status: task?.status ?? null, + }; + }), + inspectWorkspace: options.inspectWorkspace ?? ((request) => inspectSimpleWorkspace({ ...request, execute })), + buildHandoff: options.buildHandoff ?? (async ({ fallback }) => fallback), + verifyRun: options.verifyRun ?? (async ({ lanes }) => ({ + verified: Array.isArray(lanes) && lanes.length > 0 + && lanes.every((lane) => lane.phase === 'completed' || lane.phase === 'cancelled') + && lanes.filter((lane) => lane.required !== false).every((lane) => lane.phase === 'completed'), + })), + loadRecord: options.loadRecord ?? admissionStore.load, + persistRecord: options.persistRecord ?? admissionStore.save, + }; + return createRunAdmissionRuntime(simpleDeps); +} + +const AUTHENTICATION_FAILURE_PATTERN = /not signed in|not authenticated|log ?in required|unauthori[sz]ed/iu; +const GROK_EXPLICIT_LOGGED_IN_PATTERN = /(?:^|\r?\n)\s*you are logged in(?:\s+with [^\r\n]+)?\.?\s*(?=\r?\n|$)/iu; +const GROK_EXPLICIT_LOGGED_OUT_PATTERN = /(?:^|\r?\n)\s*(?:you are\s+)?(?:not logged in|not signed in|not authenticated|log ?in required|login required|authentication required)\b/iu; + +export function classifyGrokReadinessOutput(stdout = '', stderr = '') { + const standardOutput = String(stdout); + const standardError = String(stderr); + const output = `${standardOutput}\n${standardError}`; + if (GROK_EXPLICIT_LOGGED_OUT_PATTERN.test(output) + || AUTHENTICATION_FAILURE_PATTERN.test(standardOutput)) { + return { ready: false, reason: 'needs_login' }; + } + if (GROK_EXPLICIT_LOGGED_IN_PATTERN.test(standardOutput)) return { ready: true }; + if (AUTHENTICATION_FAILURE_PATTERN.test(standardError)) { + return { ready: false, reason: 'needs_login' }; + } + return { ready: true }; +} + +function classifyReadinessProbeFailure(error) { + return { + installed: error?.code !== 'ENOENT', + ready: false, + reason: error?.code === 'ENOENT' ? 'not_installed' : 'probe_failed', + }; +} + +async function probeCommand(command, args, authenticatedPattern, env, timeoutMs = PROVIDER_READINESS_PROBE_TIMEOUT_MS, outputClassifier = null, execute = execFile) { + const started = Date.now(); + try { + const { stdout, stderr } = await execute(command, args, { + cwd: '/tmp', encoding: 'utf8', timeout: timeoutMs, maxBuffer: 256 * 1024, env, }); + if (outputClassifier) { + return { + installed: true, + ...outputClassifier(stdout, stderr), + probe_duration_ms: Date.now() - started, + }; + } const output = `${stdout}${stderr}`; - if (/not signed in|not authenticated|log ?in required|unauthori[sz]ed/iu.test(output)) { - return { installed: true, ready: false, reason: 'needs_login' }; + if (AUTHENTICATION_FAILURE_PATTERN.test(output)) { + return { installed: true, ready: false, reason: 'needs_login', probe_duration_ms: Date.now() - started }; } - return { installed: true, ready: authenticatedPattern ? authenticatedPattern.test(output) : true }; + return { installed: true, ready: authenticatedPattern ? authenticatedPattern.test(output) : true, probe_duration_ms: Date.now() - started }; } catch (error) { - return { installed: error?.code !== 'ENOENT', ready: false, reason: error?.code === 'ENOENT' ? 'not_installed' : 'probe_failed' }; + return { ...classifyReadinessProbeFailure(error), probe_duration_ms: Date.now() - started }; } } +export async function probeGrokReadiness(command, env, options = {}) { + return probeCommand( + command, + ['models'], + undefined, + env, + options.timeoutMs ?? PROVIDER_READINESS_PROBE_TIMEOUT_MS, + classifyGrokReadinessOutput, + options.execute ?? execFile, + ); +} + async function providerReadiness(env = process.env) { const grokEnv = projectProviderEnvironment({ provider: 'grok', source: env, operation: 'readiness' }); const cursorLocalEnv = projectProviderEnvironment({ provider: 'cursor-local', source: env, operation: 'readiness' }); @@ -1771,7 +2435,7 @@ async function providerReadiness(env = process.env) { const acpxCommand = dshProbeEnv.CODEX_CO_ENGINEER_ACPX_COMMAND ?? 'acpx'; const dshAcpCommand = dshProbeEnv.CODEX_CO_ENGINEER_DSH_ACP_COMMAND ?? 'dsh-acp-demo'; const [grok, cursorLocal, dshCli, acpx, dshAcp, dshMuseCredential, dshOxCredential, cursorCloud] = await Promise.all([ - probeCommand(grokCommand, ['models'], undefined, grokEnv), + probeGrokReadiness(grokCommand, grokEnv), probeCommand(cursorCommand, ['status'], /logged in|authenticated|access token/iu, cursorLocalEnv), probeCommand(dshCommand, ['--version'], undefined, dshProbeEnv), probeCommand(acpxCommand, ['--version'], undefined, dshProbeEnv), @@ -1800,6 +2464,57 @@ async function providerReadiness(env = process.env) { }; } +let providerReadinessCache = null; +let providerReadinessCacheExpiresAt = 0; +let providerReadinessInFlight = null; +let providerReadinessCacheRoot = null; + +function cloneReadiness(value) { + return value && typeof value === 'object' ? JSON.parse(JSON.stringify(value)) : value; +} + +async function cachedProviderReadiness(env = process.env, { refresh = false, root = null } = {}) { + const now = Date.now(); + if (providerReadinessCacheRoot !== root) { + providerReadinessCache = null; + providerReadinessCacheExpiresAt = 0; + providerReadinessInFlight = null; + providerReadinessCacheRoot = root; + } + if (!refresh && providerReadinessCache !== null && providerReadinessCacheExpiresAt > now) { + return cloneReadiness(providerReadinessCache); + } + if (!refresh && root) { + const snapshot = await loadReadinessSnapshot(root); + const observedAt = Date.parse(snapshot?.observed_at ?? ''); + if (snapshot && Number.isFinite(observedAt) && observedAt + PROVIDER_READINESS_TTL_MS > now) { + providerReadinessCache = cloneReadiness(snapshot.readiness); + providerReadinessCacheExpiresAt = observedAt + PROVIDER_READINESS_TTL_MS; + return cloneReadiness(providerReadinessCache); + } + } + if (providerReadinessInFlight !== null) { + return cloneReadiness(await providerReadinessInFlight); + } + const started = Date.now(); + const pending = providerReadiness(env); + providerReadinessInFlight = pending; + try { + const value = await pending; + providerReadinessCache = cloneReadiness(value); + providerReadinessCacheExpiresAt = Date.now() + PROVIDER_READINESS_TTL_MS; + if (root) { + await saveReadinessSnapshot(root, value, { + observed_at: new Date().toISOString(), + probe_duration_ms: Date.now() - started, + }).catch(() => {}); + } + return cloneReadiness(value); + } finally { + if (providerReadinessInFlight === pending) providerReadinessInFlight = null; + } +} + export async function cancelTask(root, taskId, dependencies = {}) { const { task } = await readTask(root, taskId); if (!ACTIVE.has(task.status)) { @@ -1971,12 +2686,18 @@ export async function inspectTask(root, args = {}, options = {}) { export async function supervisorStatus(root = stateRoot(), dependencies = {}, options = {}) { // Allow calling as supervisorStatus(root, opts) for backward compat in tests. const hasDepsShape = dependencies && typeof dependencies === 'object' && ('probeBoundary' in dependencies || 'readProviderReadiness' in dependencies); - const looksLikeOpts = dependencies && typeof dependencies === 'object' && ('detail' in dependencies || 'task_limit' in dependencies || 'include_tasks' in dependencies || 'taskLimit' in dependencies || 'includeTasks' in dependencies); + const looksLikeOpts = dependencies && typeof dependencies === 'object' && ('detail' in dependencies || 'task_limit' in dependencies || 'include_tasks' in dependencies || 'taskLimit' in dependencies || 'includeTasks' in dependencies || 'refresh' in dependencies); if (!hasDepsShape && looksLikeOpts) { options = dependencies; dependencies = {}; } - const hasOptions = options && typeof options === 'object' && (options.detail !== undefined || options.task_limit !== undefined || options.taskLimit !== undefined || options.include_tasks !== undefined || options.includeTasks !== undefined); + const hasOptions = options && typeof options === 'object' && (options.detail !== undefined || options.task_limit !== undefined || options.taskLimit !== undefined || options.include_tasks !== undefined || options.includeTasks !== undefined || options.refresh !== undefined); + const readReadiness = dependencies.readProviderReadiness + ? () => dependencies.readProviderReadiness() + : () => cachedProviderReadiness(process.env, { + refresh: options.refresh === true, + root, + }); // Legacy no-arg path: must preserve exact 3.2 shape and reconcile ALL tasks before slicing (active/task values are durable truth). if (!hasOptions) { const tasksAll = await listTasks(root); @@ -1987,7 +2708,7 @@ export async function supervisorStatus(root = stateRoot(), dependencies = {}, op tasksAll[index] = await reconcileInactiveTask(root, task, runtime, dependencies); } const boundary = await localBoundaryReadiness(dependencies.probeBoundary); - const readiness = await (dependencies.readProviderReadiness ?? providerReadiness)(); + const readiness = await readReadiness(); for (const provider of ['grok', 'cursor-local', 'dsh']) { if (!boundary.ready) readiness[provider] = { ...readiness[provider], @@ -2039,7 +2760,7 @@ export async function supervisorStatus(root = stateRoot(), dependencies = {}, op allTasks[index] = await reconcileInactiveTask(root, task, runtime, dependencies); } const boundary = await localBoundaryReadiness(dependencies.probeBoundary); - const readiness = await (dependencies.readProviderReadiness ?? providerReadiness)(); + const readiness = await readReadiness(); for (const provider of ['grok', 'cursor-local', 'dsh']) { if (!boundary.ready) readiness[provider] = { ...readiness[provider], @@ -2098,6 +2819,9 @@ function liveTaskFns(root, contextByRun) { const input = { task_id: plan.task_id, provider: plan.provider, + model: plan.model, + run_id: plan.run_id, + assignment_id: plan.assignment_id, repo: ctx.repository_path, prompt, role, @@ -2113,7 +2837,7 @@ function liveTaskFns(root, contextByRun) { if (plan.provider === 'cursor-cloud' && typeof plan.starting_ref === 'string') { input.starting_ref = plan.starting_ref; } - const result = await submitTask(input, { root }); + const result = await submitTask(input, { root, ...(ctx.base_sha ? { baseSha: ctx.base_sha } : {}) }); return { task_id: result.task.id, status: result.task.status, @@ -2189,8 +2913,44 @@ export async function createSupervisorRunToolAdapter(options = {}) { }), } : seams.attention; + const simpleRuntime = options.simpleRuntime + ?? createSupervisorRunAdmissionRuntime({ + root, + env: options.env, + execute: options.execute, + checkPath: options.checkPath, + probeBoundary: options.probeBoundary, + requestConsent: options.requestConsent, + verifyConsent: options.verifyConsent, + providerReady: options.providerReady, + processBoundaryReady: options.processBoundaryReady, + preflightRuntime: options.preflightRuntime, + verifyRepository: options.verifyRepository, + prepareWorkspace: options.prepareWorkspace, + cleanupWorkspace: options.cleanupWorkspace, + createSession: options.createSession, + dispatchPrompt: options.dispatchPrompt, + replyAttention: options.replyAttention, + inspectLane: options.inspectLane, + reconnectLane: options.reconnectLane, + cancelLane: options.cancelLane, + inspectWorkspace: options.inspectWorkspace, + buildHandoff: options.buildHandoff, + verifyRun: options.verifyRun, + submitTask: options.submitTask, + waitForDispatchEvidence: options.waitForDispatchEvidence, + dispatchEvidenceTimeoutMs: options.dispatchEvidenceTimeoutMs, + dispatchEvidenceSleep: options.dispatchEvidenceSleep, + dispatchEvidenceNow: options.dispatchEvidenceNow, + compile: options.compile, + admissionStore: options.admissionStore, + loadRecord: options.loadRecord, + persistRecord: options.persistRecord, + waitForProgress: options.waitForProgress, + }); return createRunToolAdapter({ runtime: seams.runtime, + simpleRuntime, attention, projectLaneTask: projectSupervisorTerminalReceipt, classifyLaneTask: classifySupervisorTerminalReceipt, diff --git a/plugins/codex-co-engineer/mcp/v3/ui/display-only.js b/plugins/codex-co-engineer/mcp/v3/ui/display-only.js index f65c6f4..243442a 100644 --- a/plugins/codex-co-engineer/mcp/v3/ui/display-only.js +++ b/plugins/codex-co-engineer/mcp/v3/ui/display-only.js @@ -28,8 +28,9 @@ worktree_path: true, agent_argv: true, cli_argv: true, + approval_ref: true, }; - var INLINE_CARDS = { run: true, final: true }; + var INLINE_CARDS = { run: true, attention: true, final: true }; var KNOWN_EVIDENCE_KINDS = { acceptance_results: true, artifact_integrity: true, @@ -58,6 +59,8 @@ var OBJECTIVE_MAX = 512; var QUESTION_MAX = 320; + var CONSENT_MESSAGE = + 'This run needs your approval to share the full repository with the selected co-engineers for this run.'; function escapeHtml(value) { return String(value ?? '').replace(/[&<>"']/g, function (ch) { @@ -421,7 +424,9 @@ var phrase = typeof experience?.summary?.delegating === 'string' ? experience.summary.delegating : (phrases[0] || 'I am delegating this to Co-Engineer'); - var running = typeof experience?.summary?.running === 'string' ? experience.summary.running : ''; + var running = typeof experience?.summary?.reconciling === 'string' + ? experience.summary.reconciling + : (typeof experience?.summary?.running === 'string' ? experience.summary.running : ''); var baseSha = sha40(repository.base_sha); var digest = digestValue(repository.digest); var runningProvider = displayString(runningInfo.provider_phrase || runningInfo.provider, NOT_AVAILABLE); @@ -474,6 +479,60 @@ return value === true ? 'yes' : 'no'; } + function consentRecord(experience) { + var attention = experience && experience.attention && typeof experience.attention === 'object' + ? experience.attention + : null; + var consent = attention && attention.consent && typeof attention.consent === 'object' + ? attention.consent + : null; + var request = consent && consent.request && typeof consent.request === 'object' + ? consent.request + : {}; + if (!consent) return null; + return { consent: consent, request: request }; + } + + function consentProviderLine(providers) { + if (!Array.isArray(providers) || providers.length === 0) return NOT_AVAILABLE; + return providers.slice(0, 8).map(function (provider) { + return displayString(provider, 'Provider not named'); + }).join(', '); + } + + function renderConsentCardHtml(experience) { + var entry = consentRecord(experience) || { consent: {}, request: {} }; + var consent = entry.consent; + var request = entry.request; + var status = displayString(consent.status, 'required'); + var pending = status === 'pending' || status === 'required'; + var message = pending ? CONSENT_MESSAGE : 'The host did not approve repository exposure for this run.'; + return [ + '
', + '', + '
', + ].join(''); + } + function renderFinalCardHtml(experience) { var finalCard = experience && experience.final && typeof experience.final === 'object' ? experience.final : {}; var git = finalCard.git && typeof finalCard.git === 'object' ? finalCard.git : {}; @@ -562,7 +621,7 @@ '', '', '
', - '

Ready for Sol merge

', + '

Ready for integration review

', '

' + escapeHtml(solReady) + '

', '
', '
', @@ -610,6 +669,7 @@ function renderInlineCardHtml(card, experience) { var safe = stripOwnerOnly(experience) || {}; if (card === 'run') return renderRunCardHtml(safe); + if (card === 'attention' && consentRecord(safe)) return renderConsentCardHtml(safe); if (card === 'final') return renderFinalCardHtml(safe); return ''; } @@ -619,6 +679,15 @@ if (data.experience && typeof data.experience === 'object') return stripOwnerOnly(data.experience); var params = data.params && typeof data.params === 'object' ? data.params : null; var result = params && params.result && typeof params.result === 'object' ? params.result : params; + var resultMeta = result && result._meta && typeof result._meta === 'object' + ? result._meta + : (params && params._meta && typeof params._meta === 'object' + ? params._meta + : (data._meta && typeof data._meta === 'object' ? data._meta : null)); + var metaExperience = resultMeta && resultMeta['codex-co-engineer/experience']; + if (metaExperience && typeof metaExperience === 'object') { + return stripOwnerOnly(metaExperience); + } var structured = result && result.structuredContent && typeof result.structuredContent === 'object' ? result.structuredContent : (data.structuredContent && typeof data.structuredContent === 'object' ? data.structuredContent : null); @@ -640,7 +709,9 @@ var phrase = typeof safe.summary?.delegating === 'string' ? safe.summary.delegating : (phrases[0] || 'I am delegating this to Co-Engineer'); - var running = typeof safe.summary?.running === 'string' ? safe.summary.running : ''; + var running = typeof safe.summary?.reconciling === 'string' + ? safe.summary.reconciling + : (typeof safe.summary?.running === 'string' ? safe.summary.running : ''); return { phrase: joinRunPhrases(phrase, running), objective: displayString(run.objective), @@ -715,6 +786,22 @@ })(), }; } + if (card === 'attention' && consentRecord(safe)) { + var consentEntry = consentRecord(safe); + var consent = consentEntry.consent; + var consentRequest = consentEntry.request; + return { + consent_message: consent.status === 'pending' || consent.status === 'required' + ? CONSENT_MESSAGE + : 'The host did not approve repository exposure for this run.', + consent_status: displayString(consent.status, 'required'), + consent_providers: consentProviderLine(consentRequest.provider_phrases || consentRequest.providers), + consent_scope: displayString(consentRequest.scope), + consent_duration: displayString(consentRequest.duration), + consent_remote_mutation: consentRequest.remote_mutation === false ? 'no' : NOT_AVAILABLE, + consent_repository_identity: displayString(consentRequest.repository_identity), + }; + } return {}; } @@ -889,6 +976,7 @@ function paint(experience) { var safe = stripOwnerOnly(experience); if (!safe || safe.card !== card) return false; + if (card === 'attention' && !consentRecord(safe)) return false; painted.push(safe.card); if (typeof opts.applyHtml === 'function') opts.applyHtml(renderInlineCardHtml(card, safe)); if (opts.root && opts.document) paintDom(opts.document, opts.root, card, safe); @@ -990,6 +1078,7 @@ redactDisplay: redactDisplay, renderFinalCardHtml: renderFinalCardHtml, renderInlineCardHtml: renderInlineCardHtml, + renderConsentCardHtml: renderConsentCardHtml, renderRunCardHtml: renderRunCardHtml, stripOwnerOnly: stripOwnerOnly, unwrapExperience: unwrapExperience, diff --git a/plugins/codex-co-engineer/mcp/v3/ui/final.html b/plugins/codex-co-engineer/mcp/v3/ui/final.html index 9104f18..f307707 100644 --- a/plugins/codex-co-engineer/mcp/v3/ui/final.html +++ b/plugins/codex-co-engineer/mcp/v3/ui/final.html @@ -65,7 +65,7 @@

Push / PR

-

Ready for Sol merge

+

Ready for integration review

no

diff --git a/plugins/codex-co-engineer/mcp/v3/worktree-bootstrap-runtime.mjs b/plugins/codex-co-engineer/mcp/v3/worktree-bootstrap-runtime.mjs new file mode 100644 index 0000000..1ccbb2c --- /dev/null +++ b/plugins/codex-co-engineer/mcp/v3/worktree-bootstrap-runtime.mjs @@ -0,0 +1,5 @@ +import { fileURLToPath } from 'node:url'; + +export const BUNDLED_WORKTREE_BOOTSTRAP = fileURLToPath( + new URL('../../vendor/worktree-bootstrap/worktree-bootstrap', import.meta.url), +); diff --git a/plugins/codex-co-engineer/package.json b/plugins/codex-co-engineer/package.json index 5bf7c74..b74b35b 100644 --- a/plugins/codex-co-engineer/package.json +++ b/plugins/codex-co-engineer/package.json @@ -1,6 +1,6 @@ { "name": "codex-co-engineer", - "version": "3.4.0", + "version": "3.4.2", "private": false, "description": "Codex-Co-Engineer: ACP-first delegation to Grok, Cursor, Cursor Cloud, and DeepSeek Harness.", "license": "MIT", @@ -8,10 +8,14 @@ "node": ">=24.0.0" }, "type": "module", + "bin": { + "codex-co-engineer-consent": "./bin/consent-grants.mjs" + }, "files": [ ".codex-plugin", ".mcp.json", "README.md", + "docs", "assets", "bin", "mcp", diff --git a/plugins/codex-co-engineer/skills/chat-with-co-engineer/SKILL.md b/plugins/codex-co-engineer/skills/chat-with-co-engineer/SKILL.md index 58a05dc..298bb2a 100644 --- a/plugins/codex-co-engineer/skills/chat-with-co-engineer/SKILL.md +++ b/plugins/codex-co-engineer/skills/chat-with-co-engineer/SKILL.md @@ -1,22 +1,29 @@ --- name: chat-with-co-engineer -description: Act on an existing Co-Engineer run by inspecting, continuing, answering grouped attention, or cancelling. Use when the user says Chatting with Co-Engineer or asks to check, continue, answer, or cancel running Co-Engineer work. Message a bound Luna Max project-manager thread for compact run events when create_thread, send_message_to_thread, and wait_threads or read_thread exist; continue in this Codex task if they do not. Never start a new run. Do not use for first-time Delegating to Co-Engineer or for raw MCP or control-plane debugging. +description: Inspect, continue, answer grouped questions, or cancel an existing Co-Engineer run. Use for Chatting with Co-Engineer; never start new work. --- # Chatting with Co-Engineer -Never start a run. Codex remains chief engineer and reviewer. External workers may commit. A scoped publisher may non-force push only the task branch and open a draft PR. Sol High or Sol XHigh alone performs regular merge after deterministic exact-head, current-green-CI, and topology checks. The user retains release, tag, version, and protected-ref authority. Act on the existing run in exactly one of these ways: inspect, continue, answer grouped attention, or cancel. +Never start a run. Codex remains reviewer and merge authority. Use the existing +run to inspect, continue, answer grouped attention, or cancel. If no run exists, +offer `$delegate-to-co-engineer`. -If no run exists, say chatting needs existing work and offer `$delegate-to-co-engineer`. Do not silently submit. +Wait through `task` with the run ID, `decision_or_attention`, and the same run +cursor. Routine progress needs no polling. On host timeout, reconnect to the same +run. Answer with `run_reply`; cancel with `cancel.run_id`. A side question does +not create a replacement assignment. -Keep the same run cursor and the same wait. The run stays on its one aggregate `decision_or_attention` wait. Answering grouped attention is one user decision, not a second delegation and not a debate loop. Unaffected assignments keep working. When that grouped decision is required, say `Co-Engineer needs one decision from you`. +For interrupted repository consent, reopen the actual host form on the same run +with `run_reply.request_consent` set to true; this does not grant approval. +Group actionable questions. For example: `Co-Engineer needs one decision from you`. +Unaffected assignments continue. Inspect the result and relevant checks before +claiming verification; report any failure or unresolved work honestly. -If a user-authorized Luna Max project-manager thread exists, Codex is the host executor: bind the actual host threadId and hostId, then call send_message_to_thread with a real prompt body and wait_threads with targets that include threadId. A create_thread result with only clientThreadId is setup_pending; do not send or wait until the host supplies threadId and hostId. Message the bound thread for completed, blocked, failed, question, timeout, or user_update. Routine progress does not wake it. A merge_ready envelope may wake Sol High or Sol XHigh once, and only when exact head and tree, verifier acceptance, current green CI, zero failed or hidden checks, and topology facts all pass. Deduplicate by cursor and a bounded recent-id window. Forward sanitized evidence references only. Grouped attention retains a routing tuple for every question_id and routes one structured response covering all answerable questions exactly once. Normal completion does not wake Sol. Do not invent cancel_thread. If task messaging or Luna Max is unavailable, continue in the current Codex task and say so. I am not substituting Sol. +Read [existing-run details](references/existing-run.md) only for an unfamiliar +reply or diagnostic operation. Raw lifecycle debugging uses +`$control-codex-co-engineer-agents`. Never ask the user to construct tool payloads. -After a complete candidate exists, inspect it, then say `Co-Engineer finished, and I verified the candidate.` That sentence is not itself a merge. External workers may commit. A scoped publisher may non-force push only the task branch and open a draft PR. Luna does not merge. Sol High or Sol XHigh alone may regular-merge after deterministic exact-head, current-green-CI, and topology checks. The user retains release, tag, version, and protected-ref authority. If a required assignment fails or stays unresolved, report the gap honestly and never use the verified-final sentence. Cancel is chatting, not a new delegation, and does not claim a host cancel_thread tool. +For an explicitly requested legacy Luna/Sol relay, see [optional host relay](references/luna-pm-events.md). This is not required for ordinary delegation. -The PR-ready final decision card shows the exact candidate branch, HEAD, and tree bound to current verifier, test, CI, push, and pull-request evidence; whether the worktree is clean and free of an in-progress git operation; whether required lanes are accepted; whether the head was published without a force push; the open draft pull request's repository, host, number, and target branch; and which co-engineer owns each assignment. It is ready for Sol merge only when every typed check passes; otherwise it lists the exact blockers. The card cannot merge. Only Sol High or XHigh may regular-merge after exact HEAD, tree, current green CI, and topology checks. - -Raw MCP, payload, cursor, or control-plane debugging uses `$control-codex-co-engineer-agents`. Never ask the user to construct tool payloads. - -For inspect, continue, grouped-attention, cancel, failure, and no-run procedures, read [references/existing-run.md](references/existing-run.md) only when that case applies. For Luna Max event relay, Sol exception criteria, and inline fallback, read [references/luna-pm-events.md](references/luna-pm-events.md) only when that case applies. +External workers may commit within their assigned scope. Publication and merge require user authorization and Codex review. diff --git a/plugins/codex-co-engineer/skills/chat-with-co-engineer/references/existing-run.md b/plugins/codex-co-engineer/skills/chat-with-co-engineer/references/existing-run.md index 9a89a69..af85154 100644 --- a/plugins/codex-co-engineer/skills/chat-with-co-engineer/references/existing-run.md +++ b/plugins/codex-co-engineer/skills/chat-with-co-engineer/references/existing-run.md @@ -10,7 +10,7 @@ Inspect first. Cancel only that same run if it is stuck. Say that chatting inspe User: Chatting with Co-Engineer: continue and tell me when you have checked the candidate. -If the candidate is complete, inspect it, then say `Co-Engineer finished, and I verified the candidate. You still decide whether to keep, change, or discard it.` That sentence is not itself a merge. External workers may commit. A scoped publisher may non-force push only the task branch and open a draft pull request. The PR-ready card reports exact HEAD and tree bound to current evidence, cleanliness including any in-progress Git operation, accepted required lanes, the open draft pull request's repository and host identity, and either ready for Sol merge or the exact blockers. Only Sol High or Sol XHigh may regular-merge after exact-head, current-green-CI, verifier, and topology checks. The user retains release, tag, version, protected-ref, and product-policy authority. +If the candidate is complete, inspect it, then say `Co-Engineer finished, and I verified the candidate. You still decide whether to keep, change, or discard it.` That sentence is not itself a merge. External workers may commit. A scoped publisher may non-force push only the task branch and open a draft pull request. The PR-ready card reports exact HEAD and tree bound to current evidence, cleanliness including any in-progress Git operation, accepted required lanes, the open draft pull request's repository and host identity, and either the legacy `ready_for_sol_merge` readiness result or the exact blockers. That compatibility field does not select a model or grant authority. Codex remains the merge authority and may merge only after exact-head, current-green-CI, verifier, and topology checks and the user's authorization. The user retains release, tag, version, protected-ref, and product-policy authority. ## Grouped attention diff --git a/plugins/codex-co-engineer/skills/chat-with-co-engineer/references/luna-pm-events.md b/plugins/codex-co-engineer/skills/chat-with-co-engineer/references/luna-pm-events.md index 8cf9ece..8497a28 100644 --- a/plugins/codex-co-engineer/skills/chat-with-co-engineer/references/luna-pm-events.md +++ b/plugins/codex-co-engineer/skills/chat-with-co-engineer/references/luna-pm-events.md @@ -1,7 +1,8 @@ # Luna Max event relay -Read this only when chatting on an existing Co-Engineer run that has, or -should have, a Luna Max project manager. Chatting still never starts a +Read this only when chatting on an existing Co-Engineer run for which the +user explicitly requested and authorized a Luna Max project manager. +Chatting still never starts a second bounded run. Keep the same run cursor and the same `decision_or_attention` wait. Co-Engineer MCP cannot invoke host-only Codex task tools and must not add a sixth tool to simulate the Desktop @@ -37,15 +38,17 @@ question identity, attempt, generation, and correlation or digest. Route one structured response covering all answerable questions exactly once. Do not alias the group to a single question_id. -Sol Medium is not a mandatory layer. Normal completion does not wake -Sol. A distinct `merge_ready` envelope may wake Sol High or Sol XHigh -exactly once for final integration, and only when exact head and tree, -verifier acceptance, current green CI, zero failed or hidden checks, and -topology facts all pass. Sol High or Sol XHigh may also adjudicate -conflicting exact evidence or reviewer verdicts, security or -protected-ref risk, composition ambiguity, repeated deterministic -rejection, a release-authority decision, or explicit user escalation. -Only Luna's parent project-manager task may terminalize work for Sol. +Sol Medium is not a mandatory layer, and neither is any other Luna or Sol +model. Normal completion does not wake Sol. A distinct `merge_ready` +envelope may notify Sol High or Sol +XHigh exactly once only when the user explicitly selected that optional +relay target and exact head and tree, verifier acceptance, current green +CI, zero failed or hidden checks, and topology facts all pass. The +optional relay may also notify that selected target about conflicting +exact evidence or reviewer verdicts, security or protected-ref risk, +composition ambiguity, repeated deterministic rejection, a +release-authority decision, or explicit user escalation. Only Luna's +parent project-manager task may send those relay notifications. Forward bounded sanitized evidence references only. Auto-spill over-bound bodies to artifact references. Do not send raw transcripts or secrets. @@ -53,11 +56,12 @@ Luna native subagents stay local analysis, inherit capabilities, stay at depth at most 2, and must not duplicate a Co-Engineer external writer assignment. External workers may commit. A scoped publisher may non-force push the task-owned unprotected Codex branch and open or update a draft pull -request. Sol High or Sol XHigh alone performs regular merge after -deterministic exact-head, current-green-CI, and topology checks. The -user retains release, tag, version, and protected-ref authority. No -worker or message can force-push, merge, rebase, tag, release, delete -refs, or override verification. Luna does not merge. +request. Codex remains the merge authority and may merge only after +deterministic exact-head, current-green-CI, and topology checks and the +user's authorization. The user retains release, tag, version, and +protected-ref authority. No worker or message can force-push, merge, +rebase, tag, release, delete refs, or override verification. Luna does +not merge, and a relay notification grants no merge or release authority. The shared TaskPort policy is [../../delegate-to-co-engineer/references/luna-pm-relay.mjs](../../delegate-to-co-engineer/references/luna-pm-relay.mjs). diff --git a/plugins/codex-co-engineer/skills/control-codex-co-engineer-agents/SKILL.md b/plugins/codex-co-engineer/skills/control-codex-co-engineer-agents/SKILL.md index 98111eb..9ea640f 100644 --- a/plugins/codex-co-engineer/skills/control-codex-co-engineer-agents/SKILL.md +++ b/plugins/codex-co-engineer/skills/control-codex-co-engineer-agents/SKILL.md @@ -1,16 +1,24 @@ --- name: control-codex-co-engineer-agents -description: Operate Codex-Co-Engineer through the five MCP tools for raw status, payload, cursor, deadline, worktree, and control-plane debugging. Use when the user asks to inspect MCP arguments, event cursors, diagnostics, or lifecycle internals. Do not use for ordinary Delegating to Co-Engineer, Chatting with Co-Engineer, or Using Grok, Cursor, or Muse Co-Engineer. +description: Debug Co-Engineer MCP payloads, cursors, deadlines, worktrees, or lifecycle internals. Use for explicit troubleshooting; ordinary delegation and existing-run management use their dedicated skills. --- # Advanced Co-Engineer Control This is the raw MCP lifecycle skill, not the natural-language Co-Engineer experience. For ordinary outcomes use `$delegate-to-co-engineer`, `$chat-with-co-engineer`, `$use-grok-co-engineer`, `$use-cursor-co-engineer`, or `$use-muse-co-engineer`. Load this skill when the user asks to debug status, payloads, cursors, deadlines, worktrees, or other control-plane internals. +For ordinary launch friction, first inspect the existing receipt and connected +catalog. Do not repair an unselected provider, create a profile, or rewrite +host configuration merely because a delegation skill was invoked. Prefer the +advertised `run_request` path; a catalog mismatch is a version issue, not a +reason to construct protected identities by hand. + Use the five MCP tools for delegation and lifecycle control. 1. Call `status` when provider or supervisor readiness is unknown. Prefer - `detail: "compact", include_tasks: false` for a readiness-only check. Require + `detail: "compact", include_tasks: false` for a readiness-only check; use + `refresh: true` only when an explicitly fresh probe is needed. Warm status + is cache-backed and cold status may return `unknown/probing`. Require `local_boundary.ready: true` before local dispatch; local provider readiness is forced false when the boundary is unavailable. 2. Local dispatch requires Linux, a working `systemd --user` manager, @@ -18,9 +26,47 @@ Use the five MCP tools for delegation and lifecycle control. CLI/worktree dependencies; `status`, dispatch preflight, and release/live acceptance validate the boundary from their actual MCP environment. 3. Choose `grok`, `cursor-local`, `cursor-cloud`, or `dsh`. -4. Call `delegate` with a stable task ID, the absolute Git worktree path in - the property named `repo`, a clear prompt, and `expected_duration_ms` or a - backwards-compatible `timeout_ms`. The argument shape is literal: +4. For a new bounded run, call `delegate` with the small semantic + `run_request` body. The server derives the exact Git identity, manifest and + prompt digests, provider model, child/workspace/dispatch identities, task + IDs, and default managed-workspace policy. The argument shape is: + + ```json + { + "run_request": { + "run_id": "auth-hardening", + "repo": "/absolute/path/to/git-worktree", + "objective": "Implement and review the auth hardening change.", + "assignments": [ + { + "assignment_id": "auth-implementation", + "provider": "grok", + "role": "implement", + "access": "write", + "prompt": "Implement the auth hardening slice and commit it.", + "expected_duration_ms": 900000 + } + ] + } + } + ``` + + Do not supply derived fields, hand-constructed digests, child IDs, or a + manually assembled full `run` envelope. The full 3.4.0 `run` envelope + remains accepted for compatibility, but skills do not construct it. + + Repository exposure uses the host's native MCP form. Only the user's + accepted approval permits admission. If the receipt remains + `awaiting_consent` after dismissal or interruption, continue the same run + with `task` and `run_reply: {"request_consent": true}` to reopen the form. + This requests a decision; it does not grant consent. Status and waits never + reopen the form. Surface `consent_host_unavailable` as a host capability + blocker; do not invent an `approval_ref` or switch launch paths to bypass it. + + Call `delegate` with a stable task ID for a legacy single task, the + absolute Git worktree path in the property named `repo`, a clear prompt, + and `expected_duration_ms` or a backwards-compatible `timeout_ms`. The + argument shape is literal: ```json { @@ -40,7 +86,7 @@ Use the five MCP tools for delegation and lifecycle control. `extend_expected_duration_ms` and `extend_reason` before expiry, and only when the new deadline is strictly later. 5. Use `role: "review"` for analysis and `role: "implement"` for changes. - DSH defaults to Muse Spark 1.2 Contributor. For the optional OpenRouter Ox + DSH defaults to Muse Spark 1.3 Contributor. For the optional OpenRouter Ox Alpha route, keep `provider: "dsh"` and add `dsh_model: "stealth/ox-alpha"`. Never send its API key in the task. 6. For local tasks, use `workspace_mode: "managed"` by default. Use @@ -56,7 +102,12 @@ Use the five MCP tools for delegation and lifecycle control. reachability before retrying. 8. Set `create_pr` only for Cursor Cloud. Local tasks reject it; Codex decides whether local commits justify a PR after inspecting the handoff. -9. Coordinate without polling: +9. Coordinate without polling. For a bounded run use one run-scoped wait: + `task` or `tasks` with `run_id`, `wait_until: "decision_or_attention"`, + and the opaque run `cursor`. The adapter keeps legacy single-task and + wait-any behavior available, but skills should use the run path for a + bounded multi-lane submission. Routine progress does not wake the wait. + For legacy tasks: - For one task, call `task` with `view: "compact"`. For a durable wait, add `wait_until: "terminal"` and the previous `event_cursor`. Optional `wait_ms` caps the call; omission follows the recorded deadline within the @@ -73,10 +124,11 @@ Use the five MCP tools for delegation and lifecycle control. - Inspect `view: "diagnostics"` only for needs-attention, failure, or a task that appears stuck. It is side-effect free and never waits. Deliver a same-session `reply` exactly once only when the capability allows it. -10. Omit `response_mode` by default. Set `response_mode: "structured"` only - when the calling client consumes authoritative `structuredContent`; a - text-only client would receive only the bounded fallback. Omission retains - the exact legacy full JSON text response. +10. Capable clients default to structured-first bounded responses. Set + `response_mode: "structured"` explicitly when the client advertises + structured-content support. Legacy/text-only clients may omit it to retain + the compatible full sanitized JSON text response. Compact status and run + receipts are bounded by the server. 11. Use `cancel` for explicit cancellation or verified orphan recovery. 12. Inspect commits, handoff, and receipts before Codex merges anything. @@ -94,6 +146,10 @@ workers survive the launching client. It is not a sandbox and does not restrict environment, network, filesystem, credentials, or shell capabilities. Local dispatch fails closed if the boundary is unavailable. +After a new run is admitted, the truthful public phase is `preparing` until +every required lane has authoritative `prompt_dispatched` evidence. Only then +may the UI or skill say that Co-Engineer is `running`. + Use normal persistent provider authentication. Never put credentials in MCP arguments or prompts. Configured provider sessions are standing authorization for task-scoped calls; preserve normal approval boundaries for deployments, diff --git a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/SKILL.md b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/SKILL.md index 968f64a..6dc84a3 100644 --- a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/SKILL.md +++ b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/SKILL.md @@ -1,30 +1,30 @@ --- name: delegate-to-co-engineer -description: Start one new bounded Co-Engineer run of one to eight isolated independent assignments. Use when the user says Delegating to Co-Engineer, asks for a team of external co-engineers, or wants parallel independent assignments without naming a single co-engineer. Pin a user-authorized Luna Max task as the default project manager when create_thread, send_message_to_thread, and wait_threads or read_thread exist; continue in this Codex task if they do not. Do not use to inspect, continue, answer grouped attention, or cancel an existing run, and do not use for raw MCP or control-plane debugging. +description: Start a new Co-Engineer run with up to eight independent assignments. Use for external delegation or several named providers; existing-run management and raw MCP debugging use their dedicated skills. --- # Delegating to Co-Engineer -Start exactly one new bounded run. That is the only submission. - -Give Codex a team of external co-engineers without giving up control. Codex remains chief engineer and reviewer. External workers may commit. A scoped publisher may non-force push only the task branch and open a draft PR. Sol High or Sol XHigh alone performs regular merge after deterministic exact-head, current-green-CI, and topology checks. The user retains release, tag, version, and protected-ref authority. The substantiated shape is up to eight isolated external co-engineers, one bounded run, one coordinated wait, one verified decision. - -## Route first - -- Inspecting, continuing, answering grouped attention, or cancelling existing work uses `$chat-with-co-engineer`. Do not submit again. -- Exactly one named Grok, Cursor, or Muse Co-Engineer uses `$use-grok-co-engineer`, `$use-cursor-co-engineer`, or `$use-muse-co-engineer`. -- Several named co-engineers stay in this one run. Do not rank, predict cost, or substitute a different co-engineer. -- No named co-engineer and no named profile means ask once among Grok, Cursor, or Muse. Do not submit before that choice exists. -- Raw MCP, payload, cursor, or control-plane debugging uses `$control-codex-co-engineer-agents`. - -Independent assignments do not share a writer path. The bound is eight. Wait once for `decision_or_attention`. Routine progress does not wake that wait. Keep the same run cursor. The Co-Engineer run/event transport does not poll models. - -When the user authorizes it and Luna Max is available and the host can create_thread, send_message_to_thread, and wait_threads or read_thread, pin one Luna Max task as the default routine project manager. Codex is the host executor: call create_thread, bind only a real threadId plus hostId, then send_message_to_thread and wait_threads. A create_thread result with only clientThreadId is setup_pending; do not send or wait until the host supplies threadId and hostId. The JS adapter only plans and validates those call shapes. Luna Max may use bounded native read-only subagents for local analysis only; they must not duplicate a Co-Engineer writer assignment or widen Git or merge authority. Sol Medium is not required. Sol High or Sol XHigh alone may regular-merge a verified merge_ready packet after expected-head compare-and-swap, current-green-CI, and topology checks, and remains an exception adjudicator. External workers may commit. A scoped publisher may non-force push only the task branch and open a draft PR. The user retains release, tag, version, and protected-ref authority. No worker or message can force-push, merge, rebase, tag, release, delete refs, or override verification. If task messaging or Luna Max is unavailable, continue in the current Codex task and say so. I am not substituting Sol. - -Speak `I am delegating this to Co-Engineer`, then `Co-Engineer is running 1 independent assignment` or `Co-Engineer is running N independent assignments` for N from 2 through 8. When a chosen co-engineer is actually used, also say `Using Grok Co-Engineer`, `Using Cursor Co-Engineer`, or `Using Muse Co-Engineer`. Cursor on this computer and Cursor Cloud both display as Using Cursor Co-Engineer. Never say Using DSH Co-Engineer, Using Ox Co-Engineer, Using Cursor Local Co-Engineer, or Using Cursor Cloud Co-Engineer. +Give Codex a team of external co-engineers without giving up control. +Codex remains chief engineer, reviewer, and merge authority. The supported shape +is up to eight isolated external co-engineers, one bounded run, one coordinated wait, one verified decision. + +Read the short [launch path](references/launch.md) once, then reuse it. +Reuse authorized provider choices; if missing, ask once among Grok, Cursor, or Muse. +Keep several named providers in one run. Existing work uses +`$chat-with-co-engineer`; raw lifecycle debugging uses +`$control-codex-co-engineer-agents`. + +Use natural, concise updates. For example: `I am delegating this to Co-Engineer`. +Describe preparation honestly; claim running only with authoritative dispatch. +After inspecting a complete candidate, you may say +`Co-Engineer finished, and I verified the candidate.` Report failures and gaps +instead when work is incomplete. No fixed narration sequence is required. +Never ask the user to construct tool payloads. -After a complete candidate exists, inspect it, then say `Co-Engineer finished, and I verified the candidate.` That sentence is not itself a merge. External workers may commit. A scoped publisher may non-force push only the task-owned unprotected Codex branch and open or update a draft pull request. Luna does not merge. Sol High or Sol XHigh alone may regular-merge after deterministic exact-head, current-green-CI, and topology checks. The user retains release, tag, version, and protected-ref authority. Failure, cancel, and unresolved work must not use the verified-final sentence. +Read [model guidance](references/model-roles.md) only when choosing models is +part of the task; a separate coordinator is optional. -Never ask the user to construct tool payloads. +For an explicitly requested legacy Luna/Sol relay, see [optional host relay](references/luna-pm.md). This is not required for ordinary delegation. -For one-lane, multi-lane, ask-once, and provider-choice procedures, read [references/runs.md](references/runs.md) only when that case applies. For Luna Max project-manager pinning, host-tool detection, native subagents, and inline fallback, read [references/luna-pm.md](references/luna-pm.md) only when that case applies. +External workers may commit within their assigned scope. Publication and merge require user authorization and Codex review. diff --git a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/agents/openai.yaml b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/agents/openai.yaml index 7106eb9..8c05e8f 100644 --- a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/agents/openai.yaml +++ b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/agents/openai.yaml @@ -1,7 +1,7 @@ interface: display_name: "Delegating to Co-Engineer" short_description: "Start one bounded Co-Engineer run" - default_prompt: "Use $delegate-to-co-engineer to start one bounded run of isolated independent assignments." + default_prompt: "Use $delegate-to-co-engineer to launch the requested work with existing provider choices, one semantic submission, and one coordinated wait." dependencies: tools: - type: "mcp" diff --git a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/launch.md b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/launch.md new file mode 100644 index 0000000..a329852 --- /dev/null +++ b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/launch.md @@ -0,0 +1,35 @@ +# Launch an installed Co-Engineer + +Submit `delegate.run_request` once: a stable `run_id`, absolute Git `repo`, +`objective`, and one to eight `assignments`. Each assignment needs an +`assignment_id`, chosen `provider`, `role`, and `prompt`. Optional +`expected_duration_ms` defaults to ten minutes, with the existing 20% deadline +margin; supply an estimate when the task needs a different duration. +State the requested output, allowed changes, tests and brief relevant evidence +in the prompt. Preserve repository instructions and required verification, but +do not ask the provider to duplicate machine-generated lifecycle or handoff receipts. +Reuse existing provider/model choices. Model overrides are optional. +The controller creates managed worktrees; the worker wrapper verifies them and +owns the writer lock and lifecycle. Do not ask the provider to reconstruct that +harness setup or supply its hidden writer token. The server derives identities +and defaults; do not construct the legacy full `run` envelope. Multiple writers +need disjoint `write_scope` paths. Dependent review starts after its input exists. + +Keep the returned run ID and same run cursor. Wait through `task` with +`wait_until` set to `decision_or_attention`. Preparation and pending acknowledgement +are active work. On timeout or disconnect, reconnect to the same run; never replay +or submit a replacement. Routine progress stays internal. + +Repository exposure uses the host's actual consent form. If interrupted, reopen +it with `task.run_reply.request_consent` on the same run. Never invent approval. +Answer actionable input through the returned reply identity. Unaffected work continues. + +Inspect results, changes and checks before accepting them. Required failures, +uncertainty or unfinished cleanup block a verified result. Retrieve diagnostics +only for a concrete gap; use artifact references for omitted detail. + +Admission checks readiness. Run compact `status` only to resolve an actual +readiness question. Setup, manual worktrees, extra coordinators, configuration +changes and repeated status checks are not launch prerequisites. Report a missing +capability or selected-provider prerequisite directly; do not reconfigure other +providers. Broader repair belongs to an explicit setup/debugging task. diff --git a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/luna-pm.md b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/luna-pm.md index d02b519..092af11 100644 --- a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/luna-pm.md +++ b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/luna-pm.md @@ -1,6 +1,7 @@ # Luna Max project manager -Read this only when starting or continuing a bounded Co-Engineer run. The +Read this only for an explicitly requested legacy Luna/Sol host relay. +Ordinary delegation stays in the current Codex task. The five public skills stay Delegating, Chatting, Grok, Cursor, and Muse. Luna Max is not a sixth public skill. @@ -39,13 +40,13 @@ fails, continue in the current Codex task and say that Luna Max project-manager messaging is unavailable. I am not substituting Sol. Do not invent another model. -## Pin the default project manager +## Pin the explicitly requested relay manager When the user authorizes it and Luna Max is actually available and the -host tools exist, pin one Luna Max task as the default routine project -manager for this run. After `create_thread`, bind a usable `threadId` -and `hostId`. Message that same thread later. Do not start a second -Co-Engineer submission to get a manager. +host tools exist, pin one Luna Max task as the optional relay manager for +this run. After `create_thread`, bind a usable `threadId` and `hostId`. +Message that same thread later. Do not start a second Co-Engineer +submission to get a manager. ## Exact Codex Desktop host-tool sequence @@ -73,10 +74,10 @@ this sequence when the host tools exist: 8. Optionally call `set_thread_archived` with `{ threadId, archived }`. Preserve explicit user model overrides and the named Grok, Cursor, or -Muse co-engineers. Sol Medium is not a mandatory layer and must not -become the default manager. Sol High or Sol XHigh is the merge actor -for a verified `merge_ready` packet and an on-demand exception -adjudicator. +Muse co-engineers. No Luna or Sol model is a mandatory layer or default +manager. If the user explicitly selects Sol High or Sol XHigh as an +optional relay target, it may receive a verified `merge_ready` packet or +an on-demand exception notification. Codex remains the merge authority. Luna may use native read-only analysis subagents with inherited capabilities. Depth from the root is at most 2. Their descendant budget @@ -89,10 +90,11 @@ project-manager task may reply to or terminalize work for Sol. Wake the pinned Luna Max task on completed, blocked, failed, question, timeout, or user_update envelopes. Routine progress does not wake it. A -distinct `merge_ready` envelope may wake Sol High or Sol XHigh exactly -once, and only when exact head and tree, verifier acceptance, current -green CI, zero failed or hidden checks, and topology facts all pass. -Normal lane completion stays with Luna Max. +distinct `merge_ready` envelope may notify Sol High or Sol XHigh exactly +once only when the user explicitly selected that optional relay target +and exact head and tree, verifier acceptance, current green CI, zero +failed or hidden checks, and topology facts all pass. Normal lane +completion stays with Luna Max. Each envelope carries message id, run id, assignment id when applicable, attempt id, from and to task ids, parent or reply id when applicable, @@ -116,7 +118,8 @@ verification facts when merge-ready. ## Sol, publisher, and merge Normal completion never wakes Sol. A verified publication-ready packet -may wake Sol High or Sol XHigh once. Escalate otherwise only for +may notify Sol High or Sol XHigh once only when the user explicitly +selected that optional relay target. Notify it otherwise only for conflicting exact evidence or reviewer verdicts, security or protected-ref risk, composition ambiguity, repeated deterministic rejection, a release-authority decision, or explicit user escalation. @@ -124,13 +127,13 @@ rejection, a release-authority decision, or explicit user escalation. External workers may commit in their managed worktree. A scoped publisher may non-force push only the exact task-owned unprotected Codex branch and open or update its draft pull request after the user -authorized publication. Sol High or Sol XHigh alone performs regular -merge after deterministic exact-head, current-green-CI, and topology -checks. The user retains release, tag, version, and protected-ref -authority. Luna may request readiness or publishing and does not merge. -No worker or message can force-push, merge, rebase, tag, release, -delete refs, or override verification. Identity drift fails closed back -to Luna. +authorized publication. Codex remains the merge authority and may merge +only after deterministic exact-head, current-green-CI, and topology +checks and the user's authorization. The user retains release, tag, +version, and protected-ref authority. Luna may request readiness or +publishing and does not merge. No worker or message can force-push, +merge, rebase, tag, release, delete refs, or override verification. +Identity drift fails closed back to Luna. The executable TaskPort policy is [luna-pm-relay.mjs](luna-pm-relay.mjs). The runtime host adapter is diff --git a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/model-roles.md b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/model-roles.md new file mode 100644 index 0000000..b43f207 --- /dev/null +++ b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/model-roles.md @@ -0,0 +1,60 @@ +# Model selection for Co-Engineer + +Reviewed 2026-09-04. Choose one owner for the task. Add workers for bounded, +independent assignments when their benefit exceeds coordination overhead. +Co-Engineer delegates to external providers; native model selection and +reasoning effort belong to the host. Preserve user choices and stock defaults. + +## Practical task ladder + +These job titles and effort thresholds are workflow heuristics, not official +OpenAI roles, benchmark rankings, or a requirement to climb every rung. + +| Model | Useful starting assignment | +| --- | --- | +| Luna | Clearly specified, localized edits or repeatable work with straightforward checks | +| Terra | A bounded feature requiring reasonable local implementation decisions | +| Sol | Understand a subsystem and implement it with substantial engineering judgment | +| Astra Medium | Own difficult work spanning multiple subsystems | +| Astra High | Investigate substantial ambiguity, architecture, or a difficult failure | +| Astra XHigh / Max | Exceptionally difficult work where additional reasoning justifies its cost and latency | + +Effort is a separate control within a model, not a seniority guarantee across +models. Astra also supports low; medium is not a mandatory minimum. Use only +effort values exposed by the host. Increase effort for reasoning difficulty; +missing evidence, access, or a reproduction needs tools or information first. +Do not exhaust cheaper models before choosing Astra for an evidently hard task. + +## Official basis + +OpenAI describes [Luna](https://developers.openai.com/api/docs/models/gpt-5.6-luna) +as cost-focused, [Terra](https://developers.openai.com/api/docs/models/gpt-5.6-terra) +as balancing intelligence and cost, [Sol](https://developers.openai.com/api/docs/models/gpt-5.6-sol) +as a flagship for complex professional work, and +[Astra](https://developers.openai.com/api/docs/models/gpt-6-astra) as its most +capable model for the hardest end-to-end work. These descriptions support the +broad ladder, not the exact assignment or effort boundaries above. + +[Model selection guidance](https://developers.openai.com/api/docs/guides/model-selection) +prioritizes reaching a quality target, then reducing cost and latency while +preserving it. Establish that target with a capable model and representative +repository tasks. Compare accepted results, missed defects, rework, total +usage, and elapsed time before adopting a cheaper default. Sol High replacing +Astra for most tasks and a fixed Astra → Sol → Luna hierarchy remain unproven. + +## Keep coordination small + +The owner can implement, coordinate, and review. Sol Medium may coordinate +under Astra when a large set of independent assignments warrants it; skip +that extra layer otherwise. Choose Grok, Cursor, or DSH for the requested +provider or useful environment. Retain each worker's original lifecycle tools, +task identity, and evidence. Reconcile active work before replacing it. +[OpenAI orchestration guidance](https://developers.openai.com/api/docs/guides/agents/orchestration) +distinguishes helpers from ownership handoffs and recommends narrow specialist +jobs with short routing descriptions. + +Keep routine launch instructions focused on the assignment, constraints, and +acceptance checks. Do not load this selection guide for an ordinary delegation. +Astra's async tools, steering, and cache-preserving reasoning updates require +host/API support; MCP schema fields cannot enable them. Preserve the existing +wait and reply contracts. See [Astra guidance](https://developers.openai.com/api/docs/guides/latest-model). diff --git a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/runs.md b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/runs.md index d14487e..1bca823 100644 --- a/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/runs.md +++ b/plugins/codex-co-engineer/skills/delegate-to-co-engineer/references/runs.md @@ -8,17 +8,21 @@ User: Review the auth change with Grok Co-Engineer. This request named one co-engineer, so `$use-grok-co-engineer` owns it. If this skill is already loaded because the user said Delegating to Co-Engineer without a name, keep the assignment on the chosen co-engineer. -Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. Co-Engineer is running 1 independent assignment. +Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. Co-Engineer is preparing 1 independent assignment. + +The card changes to running only after authoritative prompt-dispatch evidence exists for the required lane. Wait once. When the work is complete, inspect it, then say `Co-Engineer finished, and I verified the candidate.` The user still decides whether to keep, change, or discard the result. ## Multi-lane -User: Split this into three isolated independent assignments: API validation, the operator guide, and a review of both diffs. +User: Split this into three isolated independent assignments: API validation, the operator guide, and a review of the existing authentication code. + +Keep all three in this one run. A review of the two new diffs must follow their completion; it cannot run independently against a base that does not contain them. Do not start three runs and do not poll each assignment. Independent means the assignments do not share a writer path. Refuse a ninth assignment. -Keep all three in this one run. Do not start three runs and do not poll each assignment. Independent means the assignments do not share a writer path. Refuse a ninth assignment. +Codex: I am delegating this to Co-Engineer. Co-Engineer is preparing 3 independent assignments. -Codex: I am delegating this to Co-Engineer. Co-Engineer is running 3 independent assignments. +The card changes to running only after authoritative prompt-dispatch evidence exists for every required lane. One `decision_or_attention` wait covers the whole run. Keep the same run cursor. When the run completes, inspect the combined result before any integration. @@ -28,17 +32,17 @@ User: Use Grok Co-Engineer for the API change and Muse Co-Engineer for the docs. Honor the named co-engineers in one run. Do not pick by cost, speed, or a hidden router. Cursor on this computer and Cursor Cloud both stay Using Cursor Co-Engineer in public speech. -Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. Using Muse Co-Engineer. Using Cursor Co-Engineer. Co-Engineer is running 3 independent assignments. +Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. Using Muse Co-Engineer. Using Cursor Co-Engineer. Co-Engineer is preparing 3 independent assignments. ## No-profile ask-once User: Give Codex a team of external co-engineers without giving up control. Split the validator and the docs. -Ask once which co-engineers should take the independent assignments: Grok, Cursor, or Muse. Do not keep asking, invent a default router, or submit before the choice exists. +If this task has no existing provider choice, ask once which co-engineers should take the independent assignments: Grok, Cursor, or Muse. Do not keep asking, invent a default router, or submit before the choice exists. After the user answers, for example Grok for the validator and Muse for the docs: -Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. Using Muse Co-Engineer. Co-Engineer is running 2 independent assignments. +Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. Using Muse Co-Engineer. Co-Engineer is preparing 2 independent assignments. That is still one submission and one coordinated wait. The ask happens before delegation. diff --git a/plugins/codex-co-engineer/skills/use-cursor-co-engineer/SKILL.md b/plugins/codex-co-engineer/skills/use-cursor-co-engineer/SKILL.md index 4e8f2e0..664bcb0 100644 --- a/plugins/codex-co-engineer/skills/use-cursor-co-engineer/SKILL.md +++ b/plugins/codex-co-engineer/skills/use-cursor-co-engineer/SKILL.md @@ -1,18 +1,19 @@ --- name: use-cursor-co-engineer -description: Delegate isolated Co-Engineer work specifically to Cursor. Use when the user says Using Cursor Co-Engineer or names Cursor Co-Engineer, including Cursor on this computer or Cursor Cloud. Do not use for Grok or Muse, for Chatting with Co-Engineer, when several co-engineers are named, or for raw MCP or control-plane debugging. +description: Launch new work specifically with Cursor Co-Engineer, locally or in Cloud. Use shared delegation for multiple providers and the chat skill for existing work. --- # Using Cursor Co-Engineer -This is one new bounded run with Cursor as the chosen co-engineer. Codex remains chief engineer and reviewer. External workers may commit. A scoped publisher may non-force push only the task branch and open a draft PR. Sol High or Sol XHigh alone may perform a regular merge after deterministic exact-head/tree, current green CI, verifier, and topology checks. The user retains version, tag, release, protected-ref, and product-policy authority. +Codex remains reviewer and merge authority. Use the short +[launch path](../delegate-to-co-engineer/references/launch.md) with Cursor +selected. Public name: `Using Cursor Co-Engineer`. +Cursor on this computer and Cursor Cloud share the public name; do not expose internal slot names. Reuse the chosen location, or ask once if it is missing. -If the user also named Grok or Muse, use `$delegate-to-co-engineer` and keep every named co-engineer in that one run. Inspecting, continuing, answering, or cancelling existing work uses `$chat-with-co-engineer`. Raw MCP, event-cursor, or control-plane debugging uses `$control-codex-co-engineer-agents`. +Several named providers belong in one `$delegate-to-co-engineer` run. +Existing work uses `$chat-with-co-engineer`; raw lifecycle debugging uses +`$control-codex-co-engineer-agents`. Preserve the same run cursor and use +`decision_or_attention` for its aggregate wait. +Never ask the user to construct tool payloads. -Do not rank, predict cost, or substitute Grok or Muse. Wait once for `decision_or_attention`. Keep the same run cursor. The bound is eight isolated assignments. - -Speak `I am delegating this to Co-Engineer. Using Cursor Co-Engineer.` Then say `Co-Engineer is running 1 independent assignment` or `Co-Engineer is running N independent assignments` for N from 2 through 8. Cursor on this computer and Cursor Cloud both display as Using Cursor Co-Engineer. The user may name the Cursor place in plain language; do not expose internal slot names. Never say Using Cursor Local Co-Engineer or Using Cursor Cloud Co-Engineer. - -After a complete candidate exists, inspect it, then say `Co-Engineer finished, and I verified the candidate.` Never ask the user to construct tool payloads. - -For Cursor-place procedures, read [references/cursor-assignment.md](references/cursor-assignment.md) only when local versus Cloud still needs a public-speech decision. +External workers may commit within their assigned scope. Publication and merge require user authorization and Codex review. diff --git a/plugins/codex-co-engineer/skills/use-cursor-co-engineer/agents/openai.yaml b/plugins/codex-co-engineer/skills/use-cursor-co-engineer/agents/openai.yaml index 666d5fd..16d5d50 100644 --- a/plugins/codex-co-engineer/skills/use-cursor-co-engineer/agents/openai.yaml +++ b/plugins/codex-co-engineer/skills/use-cursor-co-engineer/agents/openai.yaml @@ -1,7 +1,7 @@ interface: display_name: "Using Cursor Co-Engineer" short_description: "Assign isolated work to Cursor Co-Engineer" - default_prompt: "Use $use-cursor-co-engineer to give Cursor Co-Engineer one isolated assignment." + default_prompt: "Use $use-cursor-co-engineer to launch the requested work with existing provider choices, one semantic submission, and one coordinated wait." dependencies: tools: - type: "mcp" diff --git a/plugins/codex-co-engineer/skills/use-cursor-co-engineer/references/cursor-assignment.md b/plugins/codex-co-engineer/skills/use-cursor-co-engineer/references/cursor-assignment.md index 52fac78..2ee065c 100644 --- a/plugins/codex-co-engineer/skills/use-cursor-co-engineer/references/cursor-assignment.md +++ b/plugins/codex-co-engineer/skills/use-cursor-co-engineer/references/cursor-assignment.md @@ -6,7 +6,9 @@ Read this only when the Cursor place or public speech still needs a decision. Th User: Using Cursor Co-Engineer, review the operator guide on this computer. -Codex: I am delegating this to Co-Engineer. Using Cursor Co-Engineer. Co-Engineer is running 1 independent assignment. +Codex: I am delegating this to Co-Engineer. Using Cursor Co-Engineer. Co-Engineer is preparing 1 independent assignment. + +The card changes to running only after authoritative prompt-dispatch evidence exists for the required lane. The user named this computer in plain language. Public speech stays Using Cursor Co-Engineer. Do not say Using Cursor Local Co-Engineer. diff --git a/plugins/codex-co-engineer/skills/use-grok-co-engineer/SKILL.md b/plugins/codex-co-engineer/skills/use-grok-co-engineer/SKILL.md index 44439dc..b801d52 100644 --- a/plugins/codex-co-engineer/skills/use-grok-co-engineer/SKILL.md +++ b/plugins/codex-co-engineer/skills/use-grok-co-engineer/SKILL.md @@ -1,18 +1,18 @@ --- name: use-grok-co-engineer -description: Delegate isolated Co-Engineer work specifically to Grok. Use when the user says Using Grok Co-Engineer or names Grok Co-Engineer for an assignment. Do not use for Cursor or Muse, for Chatting with Co-Engineer, when several co-engineers are named, or for raw MCP or control-plane debugging. +description: Launch new work specifically with Grok Co-Engineer. Use shared delegation for multiple providers and the chat skill for existing work. --- # Using Grok Co-Engineer -This is one new bounded run with Grok as the chosen co-engineer. Codex remains chief engineer and reviewer. External workers may commit. A scoped publisher may non-force push only the task branch and open a draft PR. Sol High or Sol XHigh alone may perform a regular merge after deterministic exact-head/tree, current green CI, verifier, and topology checks. The user retains version, tag, release, protected-ref, and product-policy authority. +Codex remains reviewer and merge authority. Use the short +[launch path](../delegate-to-co-engineer/references/launch.md) with Grok +selected. Public name: `Using Grok Co-Engineer`. -If the user also named Cursor or Muse, use `$delegate-to-co-engineer` and keep every named co-engineer in that one run. Inspecting, continuing, answering, or cancelling existing work uses `$chat-with-co-engineer`. Raw MCP or control-plane debugging uses `$control-codex-co-engineer-agents`. +Several named providers belong in one `$delegate-to-co-engineer` run. +Existing work uses `$chat-with-co-engineer`; raw lifecycle debugging uses +`$control-codex-co-engineer-agents`. Preserve the same run cursor and use +`decision_or_attention` for its aggregate wait. +Never ask the user to construct tool payloads. -Do not rank, predict cost, or substitute Cursor or Muse. Wait once for `decision_or_attention`. Keep the same run cursor. The bound is eight isolated assignments. - -Speak `I am delegating this to Co-Engineer. Using Grok Co-Engineer.` Then say `Co-Engineer is running 1 independent assignment` or `Co-Engineer is running N independent assignments` for N from 2 through 8. Never say Using DSH Co-Engineer or Using Ox Co-Engineer. - -After a complete candidate exists, inspect it, then say `Co-Engineer finished, and I verified the candidate.` Never ask the user to construct tool payloads. - -For one-lane Grok procedures, read [references/grok-assignment.md](references/grok-assignment.md) only when the assignment shape is not already explicit. +External workers may commit within their assigned scope. Publication and merge require user authorization and Codex review. diff --git a/plugins/codex-co-engineer/skills/use-grok-co-engineer/agents/openai.yaml b/plugins/codex-co-engineer/skills/use-grok-co-engineer/agents/openai.yaml index 8166845..b04e41d 100644 --- a/plugins/codex-co-engineer/skills/use-grok-co-engineer/agents/openai.yaml +++ b/plugins/codex-co-engineer/skills/use-grok-co-engineer/agents/openai.yaml @@ -1,7 +1,7 @@ interface: display_name: "Using Grok Co-Engineer" short_description: "Assign isolated work to Grok Co-Engineer" - default_prompt: "Use $use-grok-co-engineer to give Grok Co-Engineer one isolated assignment." + default_prompt: "Use $use-grok-co-engineer to launch the requested work with existing provider choices, one semantic submission, and one coordinated wait." dependencies: tools: - type: "mcp" diff --git a/plugins/codex-co-engineer/skills/use-grok-co-engineer/references/grok-assignment.md b/plugins/codex-co-engineer/skills/use-grok-co-engineer/references/grok-assignment.md index c19ac10..4cb82ce 100644 --- a/plugins/codex-co-engineer/skills/use-grok-co-engineer/references/grok-assignment.md +++ b/plugins/codex-co-engineer/skills/use-grok-co-engineer/references/grok-assignment.md @@ -6,7 +6,9 @@ Read this only when the Grok assignment still needs a concrete lane shape. This User: Using Grok Co-Engineer, implement only the validator tests. -Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. Co-Engineer is running 1 independent assignment. +Codex: I am delegating this to Co-Engineer. Using Grok Co-Engineer. Co-Engineer is preparing 1 independent assignment. + +The card changes to running only after authoritative prompt-dispatch evidence exists for the required lane. That is one submission. Wait once for `decision_or_attention`. Keep the same run cursor. When the work is complete, inspect it, then say `Co-Engineer finished, and I verified the candidate.` diff --git a/plugins/codex-co-engineer/skills/use-muse-co-engineer/SKILL.md b/plugins/codex-co-engineer/skills/use-muse-co-engineer/SKILL.md index a433b2c..c9010ec 100644 --- a/plugins/codex-co-engineer/skills/use-muse-co-engineer/SKILL.md +++ b/plugins/codex-co-engineer/skills/use-muse-co-engineer/SKILL.md @@ -1,18 +1,19 @@ --- name: use-muse-co-engineer -description: Delegate isolated Co-Engineer work specifically to Muse. Use when the user says Using Muse Co-Engineer or names Muse Co-Engineer for an assignment. Do not use DSH or Ox as public names, and do not use for Grok, Cursor, Chatting with Co-Engineer, several named co-engineers, or raw MCP or control-plane debugging. +description: Launch new work specifically with Muse Co-Engineer. Use shared delegation for multiple providers and the chat skill for existing work. --- # Using Muse Co-Engineer -This is one new bounded run with Muse as the chosen co-engineer. Codex remains chief engineer and reviewer. External workers may commit. A scoped publisher may non-force push only the task branch and open a draft PR. Sol High or Sol XHigh alone may perform a regular merge after deterministic exact-head/tree, current green CI, verifier, and topology checks. The user retains version, tag, release, protected-ref, and product-policy authority. +Codex remains reviewer and merge authority. Use the short +[launch path](../delegate-to-co-engineer/references/launch.md) with Muse +selected. Public name: `Using Muse Co-Engineer`. +Never say Using DSH Co-Engineer or Using Ox Co-Engineer. -If the user also named Grok or Cursor, use `$delegate-to-co-engineer` and keep every named co-engineer in that one run. Inspecting, continuing, answering, or cancelling existing work uses `$chat-with-co-engineer`. Raw MCP or control-plane debugging uses `$control-codex-co-engineer-agents`. Internal DSH or Ox slot names are not this skill. +Several named providers belong in one `$delegate-to-co-engineer` run. +Existing work uses `$chat-with-co-engineer`; raw lifecycle debugging uses +`$control-codex-co-engineer-agents`. Preserve the same run cursor and use +`decision_or_attention` for its aggregate wait. +Never ask the user to construct tool payloads. -Do not rank, predict cost, or substitute Grok or Cursor. Wait once for `decision_or_attention`. Keep the same run cursor. The bound is eight isolated assignments. - -Speak `I am delegating this to Co-Engineer. Using Muse Co-Engineer.` Then say `Co-Engineer is running 1 independent assignment` or `Co-Engineer is running N independent assignments` for N from 2 through 8. Muse is the public name. Never say Using DSH Co-Engineer or Using Ox Co-Engineer. - -After a complete candidate exists, inspect it, then say `Co-Engineer finished, and I verified the candidate.` Never ask the user to construct tool payloads. - -For Muse naming procedures, read [references/muse-assignment.md](references/muse-assignment.md) only when public speech still needs a correction. +External workers may commit within their assigned scope. Publication and merge require user authorization and Codex review. diff --git a/plugins/codex-co-engineer/skills/use-muse-co-engineer/agents/openai.yaml b/plugins/codex-co-engineer/skills/use-muse-co-engineer/agents/openai.yaml index 55a5fbb..a3a7c8b 100644 --- a/plugins/codex-co-engineer/skills/use-muse-co-engineer/agents/openai.yaml +++ b/plugins/codex-co-engineer/skills/use-muse-co-engineer/agents/openai.yaml @@ -1,7 +1,7 @@ interface: display_name: "Using Muse Co-Engineer" short_description: "Assign isolated work to Muse Co-Engineer" - default_prompt: "Use $use-muse-co-engineer to give Muse Co-Engineer one isolated assignment." + default_prompt: "Use $use-muse-co-engineer to launch the requested work with existing provider choices, one semantic submission, and one coordinated wait." dependencies: tools: - type: "mcp" diff --git a/plugins/codex-co-engineer/skills/use-muse-co-engineer/references/muse-assignment.md b/plugins/codex-co-engineer/skills/use-muse-co-engineer/references/muse-assignment.md index 45a4d79..19e4023 100644 --- a/plugins/codex-co-engineer/skills/use-muse-co-engineer/references/muse-assignment.md +++ b/plugins/codex-co-engineer/skills/use-muse-co-engineer/references/muse-assignment.md @@ -6,7 +6,9 @@ Read this only when Muse public speech still needs a decision. This skill owns M User: Using Muse Co-Engineer, review the isolated docs change. -Codex: I am delegating this to Co-Engineer. Using Muse Co-Engineer. Co-Engineer is running 1 independent assignment. +Codex: I am delegating this to Co-Engineer. Using Muse Co-Engineer. Co-Engineer is preparing 1 independent assignment. + +The card changes to running only after authoritative prompt-dispatch evidence exists for the required lane. That is one submission. Wait once for `decision_or_attention`. Keep the same run cursor. When the work is complete, inspect it, then say `Co-Engineer finished, and I verified the candidate.` diff --git a/plugins/codex-co-engineer/test/acpx-fake-agent.mjs b/plugins/codex-co-engineer/test/acpx-fake-agent.mjs index 21cd144..639b0e2 100644 --- a/plugins/codex-co-engineer/test/acpx-fake-agent.mjs +++ b/plugins/codex-co-engineer/test/acpx-fake-agent.mjs @@ -6,6 +6,7 @@ import { join } from 'node:path'; import { createInterface } from 'node:readline'; const FIXTURE_MODES = new Set([ + 'framed-final', 'normal', 'raw-partial-frame', 'silent-initialize', @@ -95,6 +96,10 @@ function sessionUpdate(sessionId, text) { }); } +function toolUpdate(sessionId, update) { + send({ jsonrpc: '2.0', method: 'session/update', params: { sessionId, update } }); +} + function toolCallUpdate(sessionId, payload) { send({ jsonrpc: '2.0', @@ -198,6 +203,28 @@ async function handleRequest(message) { pendingPrompts.set(id, { sessionId: params.sessionId, timer: null, hostileTimeout: true }); return; } + if (fixtureMode === 'framed-final') { + sessionUpdate(params.sessionId, 'fake-opening-preamble'); + toolUpdate(params.sessionId, { + sessionUpdate: 'tool_call', + toolCallId: 'fake-read', + title: 'Read package.json', + kind: 'read', + status: 'pending', + rawInput: { variant: 'ReadFile', target_file: 'package.json' }, + }); + toolUpdate(params.sessionId, { + sessionUpdate: 'tool_call_update', + toolCallId: 'fake-read', + title: 'Read package.json', + kind: 'read', + status: 'completed', + rawOutput: { text: '{"version":"3.4.2"}' }, + }); + sessionUpdate(params.sessionId, 'fake-final-answer'); + await finishPrompt(id, params.sessionId); + return; + } sessionUpdate( params.sessionId, text.includes('terminal-verdict') @@ -262,7 +289,10 @@ async function handleRequest(message) { method: 'session/request_permission', params: { sessionId: params.sessionId, - toolCall: { toolCallId: 'fake-permission', title: 'Fake permission' }, + toolCall: { + toolCallId: 'fake-permission', + title: text.includes('title question') ? 'Which environment should I use?' : 'Fake permission', + }, options: [ { optionId: 'allow', kind: 'allow_once', name: 'Allow once' }, { optionId: 'reject', kind: 'reject_once', name: 'Reject once' }, diff --git a/plugins/codex-co-engineer/test/acpx-runtime.test.mjs b/plugins/codex-co-engineer/test/acpx-runtime.test.mjs index 4e6a67a..1e3ca06 100644 --- a/plugins/codex-co-engineer/test/acpx-runtime.test.mjs +++ b/plugins/codex-co-engineer/test/acpx-runtime.test.mjs @@ -111,6 +111,7 @@ test('kills hostile detached ACP descendants during runtime close', async () => const value = await fixture('normal', 3_000); let descendantPid; let closed = false; + const originalPath = process.env.PATH; try { const turn = value.runtime.startTurn({ handle: value.handle, @@ -123,10 +124,14 @@ test('kills hostile detached ACP descendants during runtime close', async () => assert.equal(result.status, 'completed'); descendantPid = Number(await readFile(path.join(value.cwd, '.acpx-fake-descendant.pid'), 'utf8')); assert.ok(processAlive(descendantPid), 'fixture descendant should still be running before close'); + // Linux cleanup must not depend on an external process-list command. + if (process.platform === 'linux') process.env.PATH = path.join(value.root, 'no-process-list-command'); await value.runtime.close({ handle: value.handle, reason: 'test_cleanup' }); closed = true; assert.equal(await waitForProcessExit(descendantPid), true); } finally { + if (originalPath === undefined) delete process.env.PATH; + else process.env.PATH = originalPath; if (!closed) await value.runtime.close({ handle: value.handle, reason: 'test_cleanup' }).catch(() => {}); if (descendantPid && processAlive(descendantPid)) { try { process.kill(descendantPid, 'SIGKILL'); } catch {} diff --git a/plugins/codex-co-engineer/test/branding.test.mjs b/plugins/codex-co-engineer/test/branding.test.mjs index bf3e1ac..9e2be8c 100644 --- a/plugins/codex-co-engineer/test/branding.test.mjs +++ b/plugins/codex-co-engineer/test/branding.test.mjs @@ -14,7 +14,7 @@ test('plugin presents the Co-Engineer brand with usable icon assets', async () = ); assert.equal(manifest.name, 'codex-co-engineer'); - assert.equal(manifest.version, '3.4.0'); + assert.equal(manifest.version, '3.4.2'); assert.equal(manifest.interface.displayName, 'Codex-Co-Engineer'); assert.equal(manifest.interface.developerName, 'Codex-Co-Engineer'); assert.equal( @@ -64,10 +64,14 @@ test('plugin presents the Co-Engineer brand with usable icon assets', async () = const environment = mcp.mcpServers['codex-co-engineer'].env_vars; assert.ok(environment.includes('XDG_RUNTIME_DIR')); assert.ok(environment.includes('DBUS_SESSION_BUS_ADDRESS')); + assert.ok(environment.includes('OPENROUTER_API_KEY')); + assert.ok(environment.includes('CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE')); + assert.equal(environment.includes('MODEL_API_KEY'), false); + assert.equal(environment.includes('CODEX_CO_ENGINEER_MODEL_API_KEY_FILE'), false); const packageJson = JSON.parse(await readFile(path.join(ROOT, 'package.json'), 'utf8')); assert.equal(packageJson.name, 'codex-co-engineer'); - assert.equal(packageJson.version, '3.4.0'); + assert.equal(packageJson.version, '3.4.2'); const skill = await readFile( path.join(ROOT, 'skills', 'control-codex-co-engineer-agents', 'SKILL.md'), @@ -108,7 +112,7 @@ test('visitor README leads with the product shot and copy/paste install', async assert.ok(readme.includes(poster)); assert.equal(readme.includes('docs/assets/co-engineer-3.4.0/final/derived/hero-muted.mp4'), false); assert.equal(readme.includes('docs/assets/co-engineer-3.4.0/final/derived/hero-muted.webm'), false); - assert.match(readme, /static Co-Engineer architecture illustration is the authoritative/u); + assert.match(readme, /Architecture illustration, not a live screenshot/u); assert.match(readme, /There is no autoplay audio/u); assert.doesNotMatch(readme, /Optional silent architecture animation/u); assert.doesNotMatch(readme, / { - const readme = await readFile(path.join(REPO, 'README.md'), 'utf8'); - const quickstart = await readFile(path.join(REPO, 'docs/co-engineer-quickstart.md'), 'utf8'); - for (const [label, text] of [ - ['README', readme], - ['quickstart', quickstart], - ]) { - assert.match(text, /External workers may commit/u, label); - assert.match(text, /scoped publisher may non-force push only the task branch/u, label); - assert.match(text, /Sol High or Sol XHigh/u, label); - assert.match(text, /exact-head\/tree/u, label); - assert.match(text, /current green CI/u, label); - assert.match(text, /verifier/u, label); - assert.match(text, /product-policy/u, label); - assert.doesNotMatch(text, /reviewer, and merge authority/u, label); - assert.doesNotMatch(text, /You remain(?: the)? merge authority/u, label); - assert.doesNotMatch(text, /Codex controls the final merge/u, label); +test('public README and quickstart retain user publication authority', async () => { + for (const relative of ['README.md', 'docs/co-engineer-quickstart.md']) { + const text = await readFile(path.join(REPO, relative), 'utf8'); + assert.match(text, /External workers may commit/u); + assert.match(text, /Publication and merge require user authorization and Codex review/u); } }); @@ -200,16 +193,14 @@ test('README information architecture maps safe final-art slots and keeps explic const readme = await readFile(path.join(REPO, 'README.md'), 'utf8'); const slotContract = await readFile(path.join(REPO, 'docs', 'readme-image-slot-contract.md'), 'utf8'); const headings = [ - '## Visual demo', - '## First 60 seconds', - '## How a run works', - '## Provider choices', - '## Codex authority and safety', - '## Chatting, grouped attention, and the final decision', + '## What makes it useful', '## Install and authentication', - '## Migrating from 3.2.1', + '## Your first delegation', + '## Provider choices', + '## Upgrade to 3.4.2', '## Troubleshooting', - '## Advanced Co-Engineer Control/API', + '## Control and data handling', + '## For integrators and contributors', ]; let previous = -1; for (const heading of headings) { @@ -246,29 +237,34 @@ test('README information architecture maps safe final-art slots and keeps explic assert.doesNotMatch(readme, /placeholder/iu); }); -test('every repository-relative README link resolves from the repository root', async () => { - const readme = await readFile(path.join(REPO, 'README.md'), 'utf8'); - const targets = new Set(); - for (const match of readme.matchAll(/\[[^\]]*\]\(([^)\s]+)\)/gu)) { - const target = match[1]; - if (/^(?:https?:|mailto:|#)/iu.test(target)) continue; - targets.add(decodeURIComponent(target.split('#', 1)[0])); - } - for (const target of targets) { - if (target.startsWith('docs/assets/co-engineer-3.4.0/final/')) { - assert.match(target, /^docs\/assets\/co-engineer-3\.4\.0\/final\/[A-Za-z0-9./_-]+$/u); - try { - await access(path.resolve(REPO, target)); - } catch (error) { - assert.equal(error?.code, 'ENOENT', target); - } - continue; +test('every repository-relative README link resolves from its README location', async () => { + const readmes = [ + path.join(REPO, 'README.md'), + path.join(ROOT, 'README.md'), + ]; + for (const readmePath of readmes) { + const readme = await readFile(readmePath, 'utf8'); + const readmeDirectory = path.dirname(readmePath); + const targets = new Set(); + for (const match of readme.matchAll(/\[[^\]]*\]\(([^)\s]+)\)/gu)) { + const target = match[1]; + if (/^(?:https?:|mailto:)/iu.test(target)) continue; + targets.add(decodeURIComponent(target.split('#', 1)[0])); + } + for (const target of targets) { + const resolved = path.resolve(readmeDirectory, target); + const relative = path.relative(readmeDirectory, resolved); + assert.equal( + relative.startsWith('..') || path.isAbsolute(relative), + false, + `${readmePath}: ${target}`, + ); + await access(resolved); } - await access(path.resolve(REPO, target)); } }); -test('repository marketplace catalogs Codex-Co-Engineer 3.4.0', async () => { +test('repository marketplace catalogs Codex-Co-Engineer 3.4.2', async () => { const marketplace = JSON.parse( await readFile(path.join(REPO, '.agents', 'plugins', 'marketplace.json'), 'utf8'), ); @@ -276,7 +272,7 @@ test('repository marketplace catalogs Codex-Co-Engineer 3.4.0', async () => { assert.equal(marketplace.interface.displayName, 'Codex-Co-Engineer'); assert.equal(marketplace.plugins.length, 1); assert.equal(marketplace.plugins[0].name, 'codex-co-engineer'); - assert.equal(marketplace.plugins[0].version, '3.4.0'); + assert.equal(marketplace.plugins[0].version, '3.4.2'); assert.equal(marketplace.plugins[0].source.path, './plugins/codex-co-engineer'); assert.equal( marketplace.interface.shortDescription, diff --git a/plugins/codex-co-engineer/test/fake-acpx.mjs b/plugins/codex-co-engineer/test/fake-acpx.mjs index 8e5b94d..eb23bd0 100755 --- a/plugins/codex-co-engineer/test/fake-acpx.mjs +++ b/plugins/codex-co-engineer/test/fake-acpx.mjs @@ -1,26 +1,28 @@ #!/usr/bin/env node -import { mkdir, stat, readFile, writeFile } from 'node:fs/promises'; +import { mkdir, readFile, stat, writeFile } from 'node:fs/promises'; import { spawn } from 'node:child_process'; import path from 'node:path'; const argv = process.argv.slice(2); -const inputIndex = argv.indexOf('--input-file'); const agentIndex = argv.indexOf('--agent'); const cwdIndex = argv.indexOf('--cwd'); const timeoutIndex = argv.indexOf('--timeout'); -const flowIndex = argv.indexOf('flow'); +const execIndex = argv.indexOf('exec'); +const fileIndex = argv.indexOf('--file'); if ( - inputIndex < 0 - || agentIndex < 0 + agentIndex < 0 || cwdIndex < 0 || timeoutIndex < 0 - || flowIndex < 0 - || argv[flowIndex + 1] !== 'run' + || execIndex < 0 + || fileIndex < 0 + || argv[fileIndex + 1] !== '-' + || argv[execIndex + 1] !== '--file' || !argv.includes('--approve-all') || !argv.includes('--json-strict') + || argv[argv.indexOf('--format') + 1] !== 'json' ) { - process.stderr.write('missing required ACPX arguments\n'); + process.stderr.write('missing required ACPX exec arguments\n'); process.exit(2); } const agent = argv[agentIndex + 1]; @@ -37,23 +39,24 @@ if ( process.stderr.write('invalid deterministic ACPX invocation\n'); process.exit(5); } -const input = argv[inputIndex + 1]; -const metadata = await stat(input); -if ((metadata.mode & 0o077) !== 0) { - process.stderr.write('flow input is not owner-only\n'); - process.exit(3); -} -const value = JSON.parse(await readFile(input, 'utf8')); -if (typeof value.prompt !== 'string' || value.prompt.length === 0) { - process.stderr.write('missing prompt\n'); + +let prompt = ''; +for await (const chunk of process.stdin) prompt += chunk.toString('utf8'); +if (prompt.length === 0) { + process.stderr.write('missing stdin prompt\n'); process.exit(4); } -if (argv.some((entry) => entry === value.prompt)) { +if (argv.some((entry) => entry === prompt)) { process.stderr.write('prompt must not be passed in ACPX argv\n'); process.exit(6); } const fakeMode = process.env.FAKE_ACPX_MODE ?? 'success'; +await writeFile( + path.join(cwd, '.acpx-fake-observed.json'), + `${JSON.stringify({ argv: process.argv.slice(1), cwd, env_keys: Object.keys(process.env).sort() })}\n`, + { mode: 0o600 }, +); await writeFile( path.join(cwd, '.fake-acpx-env-keys.json'), `${JSON.stringify(Object.keys(process.env).sort())}\n`, @@ -69,21 +72,42 @@ if (!homeMetadata.isDirectory() || (homeMetadata.mode & 0o077) !== 0) { process.stderr.write('task-scoped HOME is not owner-only\n'); process.exit(8); } -await mkdir(path.join(home, '.acpx', 'flows', 'runs', 'fake-run'), { recursive: true, mode: 0o700 }); await mkdir(path.join(home, '.acpx', 'sessions'), { recursive: true, mode: 0o700 }); -await writeFile(path.join(home, '.acpx', 'flows', 'runs', 'fake-run', 'trace.ndjson'), 'private fake flow trace\n', { mode: 0o600 }); await writeFile(path.join(home, '.acpx', 'sessions', 'session.json'), 'private fake session\n', { mode: 0o600 }); if (process.env.FAKE_ACPX_ARTIFACT_MARKER) { await writeFile(process.env.FAKE_ACPX_ARTIFACT_MARKER, 'created\n', { mode: 0o600 }); } -if (fakeMode === 'fail-after-spawn') { - process.stderr.write('fake ACPX failed after spawn\n'); - process.exit(17); +function send(message) { + return new Promise((resolve, reject) => { + process.stdout.write(`${JSON.stringify(message)}\n`, (error) => (error ? reject(error) : resolve())); + }); +} + +async function sendSplit(message) { + const payload = Buffer.from(`${JSON.stringify(message)}\n`, 'utf8'); + const marker = Buffer.from('😀', 'utf8'); + const markerOffset = payload.indexOf(marker); + const split = markerOffset >= 0 ? markerOffset + 1 : Math.floor(payload.length / 2); + await new Promise((resolve, reject) => process.stdout.write(payload.subarray(0, split), (error) => (error ? reject(error) : resolve()))); + await new Promise((resolve) => setTimeout(resolve, 5)); + await new Promise((resolve, reject) => process.stdout.write(payload.subarray(split), (error) => (error ? reject(error) : resolve()))); +} + +async function sendRequest(id, method, params = {}) { + await send({ jsonrpc: '2.0', id, method, params }); +} + +async function sendResponse(id, result) { + await send({ jsonrpc: '2.0', id, result }); +} + +async function sendError(id, code, message) { + await send({ jsonrpc: '2.0', id, error: { code, message } }); } if (fakeMode === 'timeout-tree') { - const descendant = spawn(process.execPath, ['-e', 'process.on("SIGTERM", () => {}); setInterval(() => {}, 1000);'], { + const descendant = spawn(process.execPath, ['-e', 'process.on("SIGTERM",()=>{});setInterval(()=>{},1000)'], { cwd, detached: true, stdio: 'ignore', @@ -93,17 +117,76 @@ if (fakeMode === 'timeout-tree') { await writeFile(process.env.FAKE_ACPX_DESCENDANT_PID_FILE, `${descendant.pid}\n`, { mode: 0o600 }); } process.on('SIGTERM', () => {}); - setInterval(() => {}, 1000); } -const output = fakeMode === 'terminal-verdict' - ? `${'x'.repeat(5000)}\nVERDICT: DSH PASS` - : fakeMode === 'terminal-object' - ? { progress: 'x'.repeat(5000), nested: { final: `${'y'.repeat(5000)}\nVERDICT: DSH OBJECT PASS` } } - : 'DSH_FAKE_OK'; -process.stdout.write(`${JSON.stringify({ - action: 'flow_run_result', - status: 'completed', - outputs: { delegate: output }, - sessionBindings: { delegate: { acpSessionId: 'dsh-fake-session' } }, -})}\n`); +await sendRequest(0, 'initialize', { protocolVersion: 1 }); +if (fakeMode === 'auth-error') { + await sendError(0, -32001, '401: authentication_required (synthetic test)'); + process.exit(1); +} +await sendResponse(0, { protocolVersion: 1, agentCapabilities: {} }); +await sendRequest(1, 'session/new', { cwd }); +if (fakeMode === 'duplicate-id') await sendRequest(1, 'session/new', { cwd }); +await sendResponse(1, { sessionId: 'dsh-fake-session' }); +await sendRequest(2, 'session/prompt', { + sessionId: fakeMode === 'wrong-session' ? 'unexpected-session' : 'dsh-fake-session', + prompt: [{ type: 'text', text: prompt }], +}); + +if (fakeMode === 'fail-after-spawn') { + process.stderr.write('fake ACPX failed after spawn\n'); + process.exitCode = 17; +} else if (fakeMode === 'provider-error') { + await sendError(2, -32603, '402: billing_not_configured (synthetic test)'); + process.exitCode = 1; +} else if (fakeMode === 'invalid-result') { + await send({ jsonrpc: '2.0', id: 2, result: {} }); +} else if (fakeMode === 'unmatched-result') { + await send({ jsonrpc: '2.0', id: 999, result: { stopReason: 'end_turn', content: [{ type: 'text', text: 'SHOULD_NOT_SUCCEED' }] } }); +} else if (fakeMode === 'malformed') { + await new Promise((resolve, reject) => process.stdout.write('not-json\n', (error) => (error ? reject(error) : resolve()))); +} else if (fakeMode === 'wrong-result-session') { + await send({ jsonrpc: '2.0', id: 2, result: { stopReason: 'end_turn', sessionId: 'unexpected-session' } }); +} else if (fakeMode === 'timeout-tree') { + setInterval(() => {}, 1_000); +} else { + const output = fakeMode === 'terminal-verdict' + ? `${'x'.repeat(5000)}\nVERDICT: DSH PASS` + : fakeMode === 'terminal-object' + ? { progress: 'x'.repeat(5000), nested: { final: `${'y'.repeat(5000)}\nVERDICT: DSH OBJECT PASS` } } + : fakeMode === 'utf8' + ? 'UTF8_OK 😀 café' + : fakeMode === 'thought-first' + ? 'THOUGHT_RESULT' + : 'DSH_FAKE_OK'; + if (fakeMode === 'thought-first') { + await send({ + jsonrpc: '2.0', + method: 'session/update', + params: { + sessionId: 'dsh-fake-session', + update: { + sessionUpdate: 'agent_thought_chunk', + content: { type: 'text', text: 'PRIVATE_THOUGHT_SHOULD_NOT_BE_STORED' }, + }, + }, + }); + } + const update = { + jsonrpc: '2.0', + method: 'session/update', + params: { + sessionId: 'dsh-fake-session', + update: { + sessionUpdate: 'agent_message_chunk', + content: { type: 'text', text: typeof output === 'string' ? output : 'DSH_OBJECT_PROGRESS' }, + }, + }, + }; + if (fakeMode === 'utf8') await sendSplit(update); + else await send(update); + await sendResponse(2, { + stopReason: 'end_turn', + ...(fakeMode === 'terminal-object' ? { output } : {}), + }); +} diff --git a/plugins/codex-co-engineer/test/fixtures/r1-dsh-acpx-fixtures.mjs b/plugins/codex-co-engineer/test/fixtures/r1-dsh-acpx-fixtures.mjs index 77fe3fc..422017d 100644 --- a/plugins/codex-co-engineer/test/fixtures/r1-dsh-acpx-fixtures.mjs +++ b/plugins/codex-co-engineer/test/fixtures/r1-dsh-acpx-fixtures.mjs @@ -13,7 +13,7 @@ export const DSH_BASE_SHA = 'c3d4e5f6a7b8c9d0e1f2a3b4c5d6e7f8a9b0c1d2'; export const DSH_REPOSITORY_PATH = '/opt/codex-co-engineer-dsh-driver/smoke'; export const DSH_RUN_ID = 'dsh-acpx-smoke'; -export const MUSE_MODEL = 'muse-spark-1.2-contributor'; +export const MUSE_MODEL = 'meta/muse-spark-1.3-contributor'; export const OX_MODEL = 'stealth/ox-alpha'; // Marker used to prove prompt/question content never reaches driver outputs. diff --git a/plugins/codex-co-engineer/test/fixtures/r1-local-provider-result-sink-fixtures.mjs b/plugins/codex-co-engineer/test/fixtures/r1-local-provider-result-sink-fixtures.mjs index 987649a..5fd0c76 100644 --- a/plugins/codex-co-engineer/test/fixtures/r1-local-provider-result-sink-fixtures.mjs +++ b/plugins/codex-co-engineer/test/fixtures/r1-local-provider-result-sink-fixtures.mjs @@ -20,7 +20,7 @@ export const CHILD_C = 'lane-gamma'; export const GROK_MODEL = 'grok-code'; export const CURSOR_MODEL = 'auto'; -export const DSH_MODEL = 'muse-spark-1.2-contributor'; +export const DSH_MODEL = 'meta/muse-spark-1.3-contributor'; export const SECRET = 'sk-live-secret-1234567890'; export const SPLIT_SECRET = 'sk-split-token-1234567890'; diff --git a/plugins/codex-co-engineer/test/fixtures/r1-provider-model-fixtures.mjs b/plugins/codex-co-engineer/test/fixtures/r1-provider-model-fixtures.mjs index 0cea1c4..87eb1aa 100644 --- a/plugins/codex-co-engineer/test/fixtures/r1-provider-model-fixtures.mjs +++ b/plugins/codex-co-engineer/test/fixtures/r1-provider-model-fixtures.mjs @@ -25,7 +25,7 @@ export const PROVIDER_MODEL_FIXTURES = Object.freeze([ Object.freeze({ provider: 'dsh', models: Object.freeze([ - 'muse-spark-1.2-contributor', + 'meta/muse-spark-1.3-contributor', 'stealth/ox-alpha', 'future-dsh-model', 'unlisted-future/model.9', diff --git a/plugins/codex-co-engineer/test/fixtures/r1-run-orchestration-fixtures.mjs b/plugins/codex-co-engineer/test/fixtures/r1-run-orchestration-fixtures.mjs index 6fbdd0c..3dfeccf 100644 --- a/plugins/codex-co-engineer/test/fixtures/r1-run-orchestration-fixtures.mjs +++ b/plugins/codex-co-engineer/test/fixtures/r1-run-orchestration-fixtures.mjs @@ -68,7 +68,7 @@ export function mixedProviderManifest(overrides = {}) { starting_ref: baseSha, }), writerLane('lane-muse', ['src/muse/**'], { - execution: { provider: 'dsh', model: 'muse-spark-1.2-contributor' }, + execution: { provider: 'dsh', model: 'meta/muse-spark-1.3-contributor' }, }), writerLane('lane-ox', ['src/ox/**'], { execution: { provider: 'dsh', model: 'stealth/ox-alpha' }, diff --git a/plugins/codex-co-engineer/test/fixtures/r1-run-scheduler-fixtures.mjs b/plugins/codex-co-engineer/test/fixtures/r1-run-scheduler-fixtures.mjs index ee9ba66..a205f64 100644 --- a/plugins/codex-co-engineer/test/fixtures/r1-run-scheduler-fixtures.mjs +++ b/plugins/codex-co-engineer/test/fixtures/r1-run-scheduler-fixtures.mjs @@ -103,7 +103,7 @@ export function mixedLaneRequest(overrides = {}) { taskId: TASK_B, writeScope: ['src/beta/**'], provider: 'dsh', - model: 'muse-spark-1.2-contributor', + model: 'meta/muse-spark-1.3-contributor', }), verifierAssignment(), ], diff --git a/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/backend-writer.envelope.txt b/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/backend-writer.envelope.txt index 64996b4..809a7a1 100644 --- a/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/backend-writer.envelope.txt +++ b/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/backend-writer.envelope.txt @@ -4,6 +4,8 @@ run_id: golden-prompt-fixtures lane_index: 0 assignment_count: 3 repository_path: /run-fixtures/repository +provider_workspace: work only in the current working directory (assigned worktree); repository_path is source identity, not a navigation target +provider_guidance: task prompt and local repository instructions are sufficient unless the assignment explicitly requests an external skill; honor an exact requested output and format exactly; required_evidence labels are controller metadata, not worker response sections; omit routine progress narration and repeated identity or report blocks unless the task prompt requests them, while surfacing blockers and necessary questions; do not seek receipt artifacts because the controller owns lifecycle and machine receipts base_sha: a1b2c3d4e5f6a7b8c9d0e1f2a3b4c5d6e7f8a9b0 begin-objective 113 bytes Compile 3.3.0 golden child-envelope fixtures with exact UTF-8 byte boundaries. Ziel: byte-exakte Umschläge. 🦊 diff --git a/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/cloud-verify.envelope.txt b/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/cloud-verify.envelope.txt index 9290cb1..e0b10d6 100644 --- a/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/cloud-verify.envelope.txt +++ b/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/cloud-verify.envelope.txt @@ -4,6 +4,8 @@ run_id: golden-prompt-fixtures lane_index: 2 assignment_count: 3 repository_path: /run-fixtures/repository +provider_workspace: work only in the current working directory (assigned worktree); repository_path is source identity, not a navigation target +provider_guidance: task prompt and local repository instructions are sufficient unless the assignment explicitly requests an external skill; honor an exact requested output and format exactly; required_evidence labels are controller metadata, not worker response sections; omit routine progress narration and repeated identity or report blocks unless the task prompt requests them, while surfacing blockers and necessary questions; do not seek receipt artifacts because the controller owns lifecycle and machine receipts base_sha: a1b2c3d4e5f6a7b8c9d0e1f2a3b4c5d6e7f8a9b0 begin-objective 113 bytes Compile 3.3.0 golden child-envelope fixtures with exact UTF-8 byte boundaries. Ziel: byte-exakte Umschläge. 🦊 diff --git a/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/docs-reviewer.envelope.txt b/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/docs-reviewer.envelope.txt index 9d13ad8..4a7b275 100644 --- a/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/docs-reviewer.envelope.txt +++ b/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/docs-reviewer.envelope.txt @@ -4,6 +4,8 @@ run_id: golden-prompt-fixtures lane_index: 1 assignment_count: 3 repository_path: /run-fixtures/repository +provider_workspace: work only in the current working directory (assigned worktree); repository_path is source identity, not a navigation target +provider_guidance: task prompt and local repository instructions are sufficient unless the assignment explicitly requests an external skill; honor an exact requested output and format exactly; required_evidence labels are controller metadata, not worker response sections; omit routine progress narration and repeated identity or report blocks unless the task prompt requests them, while surfacing blockers and necessary questions; do not seek receipt artifacts because the controller owns lifecycle and machine receipts base_sha: a1b2c3d4e5f6a7b8c9d0e1f2a3b4c5d6e7f8a9b0 begin-objective 113 bytes Compile 3.3.0 golden child-envelope fixtures with exact UTF-8 byte boundaries. Ziel: byte-exakte Umschläge. 🦊 diff --git a/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/identity-digests.json b/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/identity-digests.json index 6f04d86..aca9f25 100644 --- a/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/identity-digests.json +++ b/plugins/codex-co-engineer/test/fixtures/v3-prompt-golden/identity-digests.json @@ -40,24 +40,24 @@ "domain": "codex-co-engineer.identity.v1", "version": 1, "label": "child-envelope.v1", - "input_bytes": 3555, - "digest": "caf1dc18402d1a05e3fd985b23700e4509f32e808f13c759238b3c52839a0e9a" + "input_bytes": 4222, + "digest": "0064faadfb6e6d9734d8adb2c335f3abe16e6d790cf733e789717f93d5ca6724" }, "docs-reviewer": { "algorithm": "sha256", "domain": "codex-co-engineer.identity.v1", "version": 1, "label": "child-envelope.v1", - "input_bytes": 1648, - "digest": "28fb6b34fbfce6a1516c3cf8fa088701b081a7eb8118b2225e0dbb8283d9be5c" + "input_bytes": 2317, + "digest": "8d9718b5f4c74ea7c66ce6b031b1c71cf09d1f2492e6632b504b6a4b45535597" }, "cloud-verify": { "algorithm": "sha256", "domain": "codex-co-engineer.identity.v1", "version": 1, "label": "child-envelope.v1", - "input_bytes": 1689, - "digest": "fbe3cbcec07b90645f0ebb987cc74d2584c455bd8592dface3ec887384ed6a84" + "input_bytes": 2358, + "digest": "ac8cb61f55170c403c55416232e6923d77128d6c8b8a7f9c9c52eaee98bee87d" } } } diff --git a/plugins/codex-co-engineer/test/packaged-docs.test.mjs b/plugins/codex-co-engineer/test/packaged-docs.test.mjs new file mode 100644 index 0000000..319fcd9 --- /dev/null +++ b/plugins/codex-co-engineer/test/packaged-docs.test.mjs @@ -0,0 +1,224 @@ +import assert from 'node:assert/strict'; +import { createHash } from 'node:crypto'; +import { execFile } from 'node:child_process'; +import { access, mkdir, mkdtemp, readdir, readFile, rm, stat } from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import { promisify } from 'node:util'; +import test from 'node:test'; +import { fileURLToPath } from 'node:url'; + +import { PACKAGE_DOCUMENTS } from '../../../scripts/validate-package-docs.mjs'; + +const execFileAsync = promisify(execFile); +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); +const REPO = path.resolve(ROOT, '..', '..'); +const NPM = process.env.npm_execpath || 'npm'; + +function slashPath(value) { + return value.split(path.sep).join('/'); +} + + +async function assertBundledWorktreeBootstrap(root) { + const executable = path.join(root, 'vendor', 'worktree-bootstrap', 'worktree-bootstrap'); + const [metadata, bytes, provenance] = await Promise.all([ + stat(executable), + readFile(executable), + readFile(path.join(root, 'vendor', 'worktree-bootstrap', 'PROVENANCE.json'), 'utf8').then(JSON.parse), + ]); + assert.notEqual(metadata.mode & 0o111, 0, 'packed worktree-bootstrap lost its executable mode'); + assert.equal(createHash('sha256').update(bytes).digest('hex'), provenance.artifact_sha256); + assert.equal(provenance.version, '1.1.0'); + assert.equal(provenance.license, 'MIT'); +} + +async function listFiles(directory, prefix = '') { + const files = []; + for (const entry of await readdir(directory, { withFileTypes: true })) { + const relative = prefix ? path.join(prefix, entry.name) : entry.name; + const target = path.join(directory, entry.name); + if (entry.isDirectory()) { + files.push(...await listFiles(target, relative)); + } else { + assert.equal(entry.isFile(), true, `archive contains a non-file entry: ${relative}`); + files.push(slashPath(relative)); + } + } + return files.sort(); +} + +function decode(value, label) { + try { + return decodeURIComponent(value); + } catch (error) { + assert.fail(`${label} contains an invalid percent escape: ${error.message}`); + } +} + +function stripMarkdown(text) { + return text + .replace(/!\[([^\]]*)\]\([^)]*\)/gu, '$1') + .replace(/\[([^\]]*)\]\([^)]*\)/gu, '$1') + .replace(/[`*_~]/gu, '') + .replace(/<[^>]*>/gu, ''); +} + +function githubHeadingAnchors(markdown) { + const anchors = new Set(); + const counts = new Map(); + for (const match of markdown.matchAll(/^ {0,3}#{1,6}[ \t]+(.+?)[ \t]*#*[ \t]*$/gmu)) { + const heading = stripMarkdown(match[1]) + .normalize('NFKD') + .replace(/[\u0300-\u036f]/gu, '') + .toLowerCase() + .replace(/[^\p{L}\p{N}\s-]/gu, '') + .trim() + .replace(/[\s]+/gu, '-'); + if (!heading) continue; + const count = counts.get(heading) || 0; + counts.set(heading, count + 1); + anchors.add(count === 0 ? heading : `${heading}-${count}`); + } + for (const match of markdown.matchAll(/\bid=["']([^"']+)["']/giu)) { + anchors.add(match[1]); + } + return anchors; +} + +function markdownLinks(markdown) { + const links = []; + const pattern = /!?\[[^\]\n]*\]\(\s*(?:<([^>\n]*)>|([^\s)\n]+))/gu; + for (const match of markdown.matchAll(pattern)) { + links.push(match[1] ?? match[2]); + } + return links; +} + +function isWithin(root, target) { + const relative = path.relative(root, target); + return relative === '' || (!relative.startsWith('..') && !path.isAbsolute(relative)); +} + +async function assertMarkdownLinks(root) { + const documentation = (await listFiles(root)).filter((file) => file.endsWith('.md')); + for (const relative of documentation) { + const markdownPath = path.join(root, relative); + const markdown = await readFile(markdownPath, 'utf8'); + for (const target of markdownLinks(markdown)) { + if (/^https?:\/\//iu.test(target) || /^mailto:/iu.test(target)) continue; + assert.equal(/^[A-Za-z][A-Za-z0-9+.-]*:/u.test(target), false, `${relative}: unsupported URL ${target}`); + assert.equal(target.startsWith('//'), false, `${relative}: protocol-relative URL ${target}`); + + const hash = target.indexOf('#'); + const rawPath = hash === -1 ? target : target.slice(0, hash); + const fragment = hash === -1 ? undefined : target.slice(hash + 1); + const targetPath = decode(rawPath, `${relative}: ${target}`); + assert.equal(targetPath.startsWith('/'), false, `${relative}: absolute path ${target}`); + const resolved = path.resolve(path.dirname(markdownPath), targetPath || path.basename(markdownPath)); + assert.equal(isWithin(root, resolved), true, `${relative}: link escapes package root: ${target}`); + const targetStat = await stat(resolved).catch((error) => { + assert.fail(`${relative}: missing link target ${target}: ${error.message}`); + }); + assert.equal(targetStat.isFile(), true, `${relative}: link target is not a file: ${target}`); + + if (fragment !== undefined) { + const anchor = decode(fragment, `${relative}: ${target}`); + assert.notEqual(anchor, '', `${relative}: empty anchor in ${target}`); + const targetMarkdown = await readFile(resolved, 'utf8'); + assert.equal( + githubHeadingAnchors(targetMarkdown).has(anchor), + true, + `${relative}: invalid anchor in ${target}`, + ); + } + } + } +} + +async function assertPackageDocs(root, { compareSource = false } = {}) { + const allFiles = await listFiles(root); + for (const file of allFiles) { + assert.doesNotMatch( + file, + /(?:^|\/)(?:\.git|node_modules|tests?|receipts?|private-receipts?|machine-specific|desktopqualification)(?:\/|$)/iu, + `archive contains an unapproved private or machine-specific file: ${file}`, + ); + } + const expected = PACKAGE_DOCUMENTS.map(({ packageRelative }) => `docs/${packageRelative}`).sort(); + const actual = allFiles.filter((file) => file === 'README.md' || file.startsWith('docs/')); + assert.deepEqual(actual, ['README.md', ...expected].sort(), 'README/docs inventory is not allowlisted'); + + for (const file of actual.filter((entry) => entry.startsWith('docs/'))) { + assert.doesNotMatch( + file, + /(?:receipt|private|machine|host|desktopqualification|(?:^|\/)\.codex|(?:^|\/)(?:tmp|home|mnt|users)(?:\/|$))/iu, + `unapproved private or machine-specific documentation: ${file}`, + ); + } + + if (compareSource) { + for (const { source, packageRelative } of PACKAGE_DOCUMENTS) { + const sourceBytes = await readFile(path.join(REPO, source)); + const packageBytes = await readFile(path.join(root, 'docs', packageRelative)); + assert.deepEqual(packageBytes, sourceBytes, `packaged documentation drifted: ${packageRelative}`); + } + } +} + +test('packed documentation is self-contained and installs without the source repository', async () => { + const scratch = await mkdtemp(path.join(os.tmpdir(), 'cce-packaged-docs-')); + const packDirectory = path.join(scratch, 'pack'); + const extractDirectory = path.join(scratch, 'extract'); + const installDirectory = path.join(scratch, 'install'); + const npmCache = path.join(scratch, 'npm-cache'); + await Promise.all([ + mkdir(packDirectory), + mkdir(extractDirectory), + mkdir(installDirectory), + mkdir(npmCache), + ]); + const npmEnvironment = { + ...process.env, + npm_config_cache: npmCache, + npm_config_update_notifier: 'false', + }; + + try { + await execFileAsync(NPM, [ + 'pack', '.', + '--pack-destination', packDirectory, + '--ignore-scripts', + '--offline', + '--json', + ], { cwd: ROOT, env: npmEnvironment, maxBuffer: 4 * 1024 * 1024 }); + + const archives = (await readdir(packDirectory)).filter((entry) => entry.endsWith('.tgz')); + assert.equal(archives.length, 1, 'npm pack did not produce exactly one archive'); + const archive = path.join(packDirectory, archives[0]); + + await execFileAsync('tar', ['-xzf', archive, '-C', extractDirectory], { maxBuffer: 4 * 1024 * 1024 }); + const packedRoot = path.join(extractDirectory, 'package'); + await access(path.join(packedRoot, 'README.md')); + await assertBundledWorktreeBootstrap(packedRoot); + await assertPackageDocs(packedRoot, { compareSource: true }); + await assertMarkdownLinks(packedRoot); + + await execFileAsync(NPM, [ + 'install', + '--prefix', installDirectory, + '--no-save', + '--no-package-lock', + '--ignore-scripts', + '--offline', + archive, + ], { cwd: installDirectory, env: npmEnvironment, maxBuffer: 4 * 1024 * 1024 }); + const installedRoot = path.join(installDirectory, 'node_modules', 'codex-co-engineer'); + await access(path.join(installedRoot, 'README.md')); + await assertBundledWorktreeBootstrap(installedRoot); + await assertPackageDocs(installedRoot); + await assertMarkdownLinks(installedRoot); + } finally { + await rm(scratch, { recursive: true, force: true }); + } +}); diff --git a/plugins/codex-co-engineer/test/r1-aggregate-run-anchor.test.mjs b/plugins/codex-co-engineer/test/r1-aggregate-run-anchor.test.mjs index c4fb2c2..228ba13 100644 --- a/plugins/codex-co-engineer/test/r1-aggregate-run-anchor.test.mjs +++ b/plugins/codex-co-engineer/test/r1-aggregate-run-anchor.test.mjs @@ -7,6 +7,7 @@ import test from 'node:test'; import { AGGREGATE_RUN_ANCHOR_SCHEMA_ID, AGGREGATE_STORAGE_ROOT_KIND, + MAX_AGGREGATE_TEMPORARIES, STORAGE_ROOT_SCHEMA_ID, initializeAggregateRunAnchorRoot, openAggregateRunAnchor, @@ -126,6 +127,21 @@ test('initialize publishes an owner-only marker and open rejects unmarked empty } }); +test('root audit tolerates the declared bounded lock-owner temporary allowance', async () => { + const root = await makePrivateRoot(); + try { + await initializeAggregateRunAnchorRoot(root); + await writeFile(path.join(root, 'lock'), '{}\n', { mode: 0o600 }); + for (let index = 0; index < MAX_AGGREGATE_TEMPORARIES; index += 1) { + await writeFile(path.join(root, `.lock-${index.toString(16).padStart(32, '0')}`), '{}\n', { mode: 0o600 }); + } + const reopened = await openAggregateRunAnchor(root); + assert.equal(reopened.root, root); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); + test('submit persists 1- and 8-assignment identities and exact retry is idempotent', async () => { await withAnchor(async (root, store) => { const one = makeSubmitInput({ assignmentCount: 1 }); @@ -404,7 +420,7 @@ test('cross-process 13/13 submit and mutation keep one winner without ENOTEMPTY const mutationCreated = mutationResults.filter((result) => result.ok && result.created); const mutationReplay = mutationResults.filter((result) => result.ok && result.created === false); assert.equal(mutationCreated.length, 1, JSON.stringify(mutationResults)); - assert.equal(mutationCreated.length + mutationReplay.length, 13); + assert.equal(mutationCreated.length + mutationReplay.length, 13, JSON.stringify(mutationResults)); for (const result of mutationResults) { assert.equal(result.ok, true, result.code); assert.notEqual(result.code, 'ENOTEMPTY'); diff --git a/plugins/codex-co-engineer/test/r1-attention-batch-adversarial.test.mjs b/plugins/codex-co-engineer/test/r1-attention-batch-adversarial.test.mjs index 1057107..0e2b839 100644 --- a/plugins/codex-co-engineer/test/r1-attention-batch-adversarial.test.mjs +++ b/plugins/codex-co-engineer/test/r1-attention-batch-adversarial.test.mjs @@ -125,6 +125,37 @@ test('unknown keys, extra item fields, and capability mismatches fail closed', a }); }); +test('structured options reject unknown fields, missing kinds, and hostile containers', async () => { + await withRoot(async ({ handle }) => { + for (const [options, expectedCode] of [ + [[{ optionId: 'allow', kind: 'allow_once', extra: HOSTILE_SECRET }], 'unknown_key'], + [[{ optionId: 'allow' }], 'missing_key'], + ]) { + const item = grokItem({ options }); + const pair = itemsAndSource([item]); + const error = await errorOf(() => handle.latch({ + run_id: RUN_ID, source: pair.source, items: pair.items, expected_revision: 0, + })); + assert.equal(error.code, expectedCode); + assertContentFree(error); + } + + const proxyItem = grokItem(); + const { proxy, counts } = countingProxy({ optionId: 'allow', kind: 'allow_once' }); + proxyItem.options = [proxy]; + const proxyPair = itemsAndSource([proxyItem]); + const proxyError = await errorOf(() => handle.latch({ + run_id: RUN_ID, + source: proxyPair.source, + items: proxyPair.items, + expected_revision: 0, + })); + assert.equal(proxyError.code, 'proxy_denied'); + assert.equal(trapTotal(counts), 0); + assertContentFree(proxyError); + }); +}); + test('duplicate assignment ids, empty batches, and oversized prompts fail closed', async () => { await withRoot(async ({ handle }) => { const grok = grokItem(); diff --git a/plugins/codex-co-engineer/test/r1-attention-batch.test.mjs b/plugins/codex-co-engineer/test/r1-attention-batch.test.mjs index abe0c92..f8746c2 100644 --- a/plugins/codex-co-engineer/test/r1-attention-batch.test.mjs +++ b/plugins/codex-co-engineer/test/r1-attention-batch.test.mjs @@ -12,6 +12,7 @@ import { ATTENTION_BATCH_DISPOSITIONS, ATTENTION_BATCH_FILE_NAME, ATTENTION_BATCH_ITEM_KEYS, + ATTENTION_BATCH_OPTION_KEYS, ATTENTION_BATCH_PROVIDERS, ATTENTION_BATCH_RECORD_KEYS, ATTENTION_BATCH_REPLY_CAPABILITIES, @@ -21,10 +22,17 @@ import { ATTENTION_BATCH_TASK_CURSOR_KEYS, ATTENTION_BATCH_UNRESOLVED_CODES, ATTENTION_BATCH_VERSION, + attentionQuestionDigestV1, describeAttentionBatchV1, openAttentionRoot, } from '../mcp/v3/attention-batch.mjs'; import { canonicalJsonStringify } from '../mcp/v3/identity.mjs'; +import { + consumeReply, + recordNeedsAttention, + replyDecision, + submitReply, +} from '../mcp/v3/mailbox.mjs'; import { RunContractV1Error } from '../mcp/v3/run-manifest.mjs'; import { RUN_JOURNAL_EVENT_KINDS, @@ -32,6 +40,7 @@ import { } from '../mcp/v3/run-reducer.mjs'; import { createRunJournal } from '../mcp/v3/run-journal.mjs'; import { openRunStore } from '../mcp/v3/run-store.mjs'; +import { createTask } from '../mcp/v3/task-store.mjs'; import { makePrivateRoot as makeStoreRoot, makeSubmission, @@ -103,6 +112,9 @@ test('AttentionBatchV1 is the frozen v1 contract with exact record keys', () => 'question_id', 'event_cursor', 'question_digest', 'prompt', 'options', 'reply_capability', 'disposition', 'deadline_at', ]); + assert.deepEqual([...ATTENTION_BATCH_OPTION_KEYS], [ + 'optionId', 'kind', 'name', 'label', 'description', + ]); assert.deepEqual([...ATTENTION_BATCH_PROVIDERS], [ 'grok', 'cursor-local', 'cursor-cloud', 'dsh', ]); @@ -248,6 +260,68 @@ test('one reply round is durable before delivery and resolves exact same-session }); }); +test('structured option id survives restart and reaches the real mailbox reply decision', async () => { + const attentionRoot = await makePrivateRoot('r1-p34-typed-options-'); + const mailboxRoot = await makeStoreRoot('r1-p34-typed-options-mailbox-'); + try { + const structuredOptions = [{ + optionId: 'allow-once-id', + kind: 'allow_once', + name: 'Allow once', + description: 'Approve this request once.', + }]; + const grok = grokItem({ options: structuredOptions }); + await createTask({ + root: mailboxRoot, + prompt: 'ask a typed question', + record: { + id: grok.task_id, + status: 'running', + provider: 'grok', + transport: 'acp', + acp_session_id: grok.session_id, + }, + }); + const recorded = await recordNeedsAttention(mailboxRoot, grok.task_id, { + session_id: grok.session_id, + question_id: grok.question_id, + prompt: grok.prompt, + options: structuredOptions, + }); + grok.options = recorded.attention.options; + grok.question_digest = attentionQuestionDigestV1(grok); + const { items, source } = itemsAndSource([grok]); + const firstHandle = await openAttentionRoot(attentionRoot); + const latched = await firstHandle.latch({ + run_id: RUN_ID, source, items, expected_revision: 0, + }); + + const restartedHandle = await openAttentionRoot(attentionRoot); + const reopened = await restartedHandle.get(RUN_ID); + assert.deepEqual(reopened.record.items[0].options, structuredOptions); + const replied = await restartedHandle.reply({ + run_id: RUN_ID, + batch_id: latched.record.batch_id, + expected_revision: reopened.record.revision, + reply: makeReply(latched.record.batch_id, [grok], 'allow-once-id'), + deliver: async (identity) => { + await submitReply(mailboxRoot, identity.task_id, identity); + return { outcome: 'delivered' }; + }, + }); + assert.equal(replied.record.status, 'resolved'); + const delivered = await consumeReply(mailboxRoot, grok.task_id, grok.question_id); + assert.equal(delivered.response, 'allow-once-id'); + assert.deepEqual(replyDecision(delivered, reopened.record.items[0].options), { + outcome: 'allow_once', + optionId: 'allow-once-id', + }); + } finally { + await rm(attentionRoot, { recursive: true, force: true }); + await rm(mailboxRoot, { recursive: true, force: true }); + } +}); + test('restart retries only the exact latched identities after a durable reply', async () => { await withRoot(async ({ handle }) => { const grok = grokItem(); diff --git a/plugins/codex-co-engineer/test/r1-brand-assets.test.mjs b/plugins/codex-co-engineer/test/r1-brand-assets.test.mjs index 65aaf90..9e89f8d 100644 --- a/plugins/codex-co-engineer/test/r1-brand-assets.test.mjs +++ b/plugins/codex-co-engineer/test/r1-brand-assets.test.mjs @@ -359,7 +359,7 @@ test('plugin defaultPrompt stays within PluginInterface bounds and UX-01 coverag ]); }); -test('marketplace and plugin metadata stay on 3.4.0, UX-01 language, and 3.4.0 assets', async () => { +test('marketplace and plugin metadata stay on 3.4.2, UX-01 language, and historical 3.4.0 assets', async () => { const fixture = JSON.parse(await readFile(CONTRACT_JSON, 'utf8')); const plugin = JSON.parse( await readFile(path.join(PLUGIN, '.codex-plugin', 'plugin.json'), 'utf8'), @@ -370,8 +370,8 @@ test('marketplace and plugin metadata stay on 3.4.0, UX-01 language, and 3.4.0 a const provenanceText = await readFile(PROVENANCE, 'utf8'); const poster = await readFile(path.join(REPO, 'docs/assets/co-engineer-3.4.0/poster.svg'), 'utf8'); - assert.equal(plugin.version, '3.4.0'); - assert.equal(marketplace.plugins[0].version, '3.4.0'); + assert.equal(plugin.version, '3.4.2'); + assert.equal(marketplace.plugins[0].version, '3.4.2'); for (const text of [ JSON.stringify(plugin), JSON.stringify(marketplace), diff --git a/plugins/codex-co-engineer/test/r1-consent-grants.test.mjs b/plugins/codex-co-engineer/test/r1-consent-grants.test.mjs new file mode 100644 index 0000000..e5acc33 --- /dev/null +++ b/plugins/codex-co-engineer/test/r1-consent-grants.test.mjs @@ -0,0 +1,122 @@ +import assert from 'node:assert/strict'; +import { execFileSync } from 'node:child_process'; +import { chmod, mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import test from 'node:test'; + +import { + CONSENT_GRANT_DURATION, + CONSENT_GRANT_STORE_FILE, + createConsentGrantStore, +} from '../mcp/v3/consent-grants.mjs'; + +function git(repo, ...args) { + return execFileSync('git', args, { cwd: repo, stdio: 'pipe', encoding: 'utf8' }).trim(); +} + +async function repository(root, name, origin = `https://example.test/${name}.git`) { + const repo = path.join(root, name); + await mkdir(repo); + git(repo, 'init', '--quiet'); + git(repo, 'config', 'user.name', 'Fixture'); + git(repo, 'config', 'user.email', 'fixture@example.test'); + git(repo, 'config', 'remote.origin.url', origin); + await writeFile(path.join(repo, 'README.md'), `${name}\n`); + git(repo, 'add', 'README.md'); + git(repo, '-c', 'commit.gpgsign=false', 'commit', '--quiet', '-m', 'fixture'); + return repo; +} + +async function fixture(exercise) { + const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-consent-grants-')); + try { await exercise(root); } finally { await rm(root, { recursive: true, force: true }); } +} + +test('remembered grants follow the Git common directory and exact provider subset', async () => { + await fixture(async (root) => { + const repo = await repository(root, 'repo'); + const linked = path.join(root, 'linked'); + git(repo, 'worktree', 'add', '--quiet', '-b', 'linked', linked); + const state = path.join(root, 'state'); + const store = createConsentGrantStore({ root: state }); + await store.remember({ repositoryPath: repo, providers: ['cursor-local', 'grok'] }); + + const linkedGrant = await store.lookup({ repositoryPath: linked, providers: ['grok'] }); + assert.equal(linkedGrant.duration, CONSENT_GRANT_DURATION); + assert.equal(linkedGrant.source, 'durable_grant'); + assert.equal(await store.lookup({ repositoryPath: linked, providers: ['grok', 'dsh'] }), null); + + git(repo, 'config', 'remote.origin.url', 'https://example.test/changed.git'); + assert.equal(await store.lookup({ repositoryPath: repo, providers: ['grok'] }), null); + }); +}); + +test('unrelated and recreated repositories do not inherit remembered consent', async () => { + await fixture(async (root) => { + const repo = await repository(root, 'repo'); + const other = await repository(root, 'other', 'https://example.test/repo.git'); + const state = path.join(root, 'state'); + const store = createConsentGrantStore({ root: state }); + await store.remember({ repositoryPath: repo, providers: ['grok'] }); + assert.equal(await store.lookup({ repositoryPath: other, providers: ['grok'] }), null); + + await rm(repo, { recursive: true, force: true }); + const recreated = await repository(root, 'repo'); + assert.equal(await store.lookup({ repositoryPath: recreated, providers: ['grok'] }), null); + }); +}); + +test('grant persistence, provider union, and revocation are process-independent', async () => { + await fixture(async (root) => { + const repo = await repository(root, 'repo'); + const state = path.join(root, 'state'); + const first = createConsentGrantStore({ root: state }); + await Promise.all([ + first.remember({ repositoryPath: repo, providers: ['grok'] }), + first.remember({ repositoryPath: repo, providers: ['dsh'] }), + ]); + const restarted = createConsentGrantStore({ root: state }); + assert.ok(await restarted.lookup({ repositoryPath: repo, providers: ['grok', 'dsh'] })); + const [grant] = await restarted.list(); + assert.deepEqual(grant.providers, ['dsh', 'grok']); + assert.equal(await restarted.revoke({ grantId: grant.grant_id }), true); + assert.equal(await first.lookup({ repositoryPath: repo, providers: ['grok'] }), null, + 'an already-running process reloads state after CLI-style revocation'); + }); +}); + +test('concurrent revoke and remember cannot resurrect an unrelated grant', async () => { + await fixture(async (root) => { + const repoA = await repository(root, 'repo-a'); + const repoB = await repository(root, 'repo-b'); + const state = path.join(root, 'state'); + const storeA = createConsentGrantStore({ root: state }); + const storeB = createConsentGrantStore({ root: state }); + await storeA.remember({ repositoryPath: repoA, providers: ['grok'] }); + await storeA.remember({ repositoryPath: repoB, providers: ['grok'] }); + const grantA = (await storeA.list()).find((grant) => grant.repository.includes('repo-a')); + await Promise.all([ + storeA.revoke({ grantId: grantA.grant_id }), + storeB.remember({ repositoryPath: repoB, providers: ['dsh'] }), + ]); + assert.equal(await storeA.lookup({ repositoryPath: repoA, providers: ['grok'] }), null); + assert.ok(await storeA.lookup({ repositoryPath: repoB, providers: ['grok', 'dsh'] })); + }); +}); + +test('malformed and unsafe state fails closed without repair', async () => { + await fixture(async (root) => { + const repo = await repository(root, 'repo'); + const state = path.join(root, 'state'); + await mkdir(state, { mode: 0o700 }); + const file = path.join(state, CONSENT_GRANT_STORE_FILE); + await writeFile(file, '{"schema":"wrong"}\n', { mode: 0o600 }); + const store = createConsentGrantStore({ root: state }); + await assert.rejects(store.lookup({ repositoryPath: repo, providers: ['grok'] }), + { code: 'consent_grant_store_invalid' }); + assert.equal(await readFile(file, 'utf8'), '{"schema":"wrong"}\n'); + await chmod(file, 0o644); + await assert.rejects(store.list(), { code: 'consent_grant_store_unsafe' }); + }); +}); diff --git a/plugins/codex-co-engineer/test/r1-constrained-verification-runner-adversarial.test.mjs b/plugins/codex-co-engineer/test/r1-constrained-verification-runner-adversarial.test.mjs index 08b545c..9077a75 100644 --- a/plugins/codex-co-engineer/test/r1-constrained-verification-runner-adversarial.test.mjs +++ b/plugins/codex-co-engineer/test/r1-constrained-verification-runner-adversarial.test.mjs @@ -19,6 +19,7 @@ import { WORKSPACE_NAME_PREFIX, WORKSPACE_PARENT_PREFIX, executeConstrainedVerificationV1, + listProcDescendants, } from '../mcp/v3/constrained-verification-runner.mjs'; import { RunContractV1Error } from '../mcp/v3/run-manifest.mjs'; import { @@ -476,6 +477,28 @@ test('parent-failing /proc enumeration failure is cleanup_uncertain not pass', a } }); +test('proc descendant scan keeps unexpected read failures and malformed identities fail-closed', async () => { + const unreadable = await errorOf(() => listProcDescendants(50, { + readDir: async () => ['51'], + readText: async () => { throw Object.assign(new Error('denied'), { code: 'EACCES' }); }, + })); + assert.equal(unreadable.code, 'cleanup_uncertain'); + assert.equal(unreadable.path, 'execution'); + + for (const malformed of [ + '51 (worker) S not-a-pid 50 0 0 0', + '51 (worker) S 1 not-a-group 0 0 0', + '51 worker S 1 50 0 0 0', + ]) { + const error = await errorOf(() => listProcDescendants(50, { + readDir: async () => ['51'], + readText: async () => malformed, + })); + assert.equal(error.code, 'cleanup_uncertain'); + assert.equal(error.path, 'execution'); + } +}); + test('parent-failing special files and copy races are rejected before execution', async () => { const { root, candidate, head } = await candidatePair(); try { diff --git a/plugins/codex-co-engineer/test/r1-constrained-verification-runner.test.mjs b/plugins/codex-co-engineer/test/r1-constrained-verification-runner.test.mjs index 42ac037..10e4171 100644 --- a/plugins/codex-co-engineer/test/r1-constrained-verification-runner.test.mjs +++ b/plugins/codex-co-engineer/test/r1-constrained-verification-runner.test.mjs @@ -18,6 +18,7 @@ import { UNSHARE_EXECUTABLE, VERIFICATION_EXECUTION_DIGEST_LABEL, executeConstrainedVerificationV1, + listProcDescendants, } from '../mcp/v3/constrained-verification-runner.mjs'; import { SHA_DIFF, @@ -429,6 +430,21 @@ test('the approved command is spawned exactly once per invocation', async () => }); }); +test('proc descendant scan tolerates vanished entries and parses command names containing parentheses', async () => { + const reads = []; + const descendants = await listProcDescendants(50, { + readDir: async () => ['50', '51', '52', 'self'], + readText: async (file) => { + reads.push(file); + if (file === '/proc/51/stat') throw Object.assign(new Error('vanished'), { code: 'ESRCH' }); + return '52 (worker) helper) S 1 50 50 0 0 0 0'; + }, + }); + assert.deepEqual(descendants, [52]); + assert.equal(Object.isFrozen(descendants), true); + assert.deepEqual(reads, ['/proc/51/stat', '/proc/52/stat']); +}); + test('the disposable workspace materializes candidate README bytes for the command', async () => { await withHarness(async ({ request, candidate }) => { const receipt = await executeConstrainedVerificationV1(request); diff --git a/plugins/codex-co-engineer/test/r1-credential-boundary-adversarial.test.mjs b/plugins/codex-co-engineer/test/r1-credential-boundary-adversarial.test.mjs index b489fdb..65b448c 100644 --- a/plugins/codex-co-engineer/test/r1-credential-boundary-adversarial.test.mjs +++ b/plugins/codex-co-engineer/test/r1-credential-boundary-adversarial.test.mjs @@ -98,7 +98,7 @@ test('readiness children spawned through projection cannot observe stripped secr const env = projectProviderEnvironment({ provider: 'dsh', source: HOSTILE_ENV, - dshModel: 'muse-spark-1.2-contributor', + dshModel: 'meta/muse-spark-1.3-contributor', operation: 'readiness_probe', }); assert.equal(env.MODEL_API_KEY, undefined); @@ -128,45 +128,46 @@ test('push URL objects and insteadOf maps are denied', async () => { test('materialize does not copy key-file paths into the child environment', async () => { await withTempDir('cce-p29-mat-', async (root) => { - const keyFile = await writeOwnerFile(path.join(root, 'model-api-key'), 'loaded-muse-secret\n'); + const keyFile = await writeOwnerFile(path.join(root, 'openrouter-api-key'), 'loaded-muse-secret\n'); const env = await materializeProviderEnvironment({ provider: 'dsh', - dshModel: 'muse-spark-1.2-contributor', + dshModel: 'meta/muse-spark-1.3-contributor', operation: 'lane', source: { PATH: '/usr/bin:/bin', HOME: root, - CODEX_CO_ENGINEER_MODEL_API_KEY_FILE: keyFile, + CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE: keyFile, }, }); - assert.equal(env.MODEL_API_KEY, 'loaded-muse-secret'); - assert.equal(env.CODEX_CO_ENGINEER_MODEL_API_KEY_FILE, undefined); + assert.equal(env.OPENROUTER_API_KEY, 'loaded-muse-secret'); + assert.equal(env.MODEL_API_KEY, undefined); + assert.equal(env.CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE, undefined); assert.equal(JSON.stringify(env).includes(keyFile), false); }); }); test('credential-file overrides reject relative non-normalized and double-separator paths before resolve', async () => { await withTempDir('cce-p29-override-', async (root) => { - const keyFile = await writeOwnerFile(path.join(root, 'model-api-key'), 'muse-from-override\n'); + const keyFile = await writeOwnerFile(path.join(root, 'openrouter-api-key'), 'muse-from-override\n'); const loaded = await loadProviderCredential({ provider: 'dsh', - dshModel: 'muse-spark-1.2-contributor', - source: { HOME: root, CODEX_CO_ENGINEER_MODEL_API_KEY_FILE: keyFile }, + dshModel: 'meta/muse-spark-1.3-contributor', + source: { HOME: root, CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE: keyFile }, }); assert.equal(loaded.value, 'muse-from-override'); const overrides = [ 'relative-key', - `${root}/../${path.basename(root)}/model-api-key`, - `${root}/./model-api-key`, - `${root}//model-api-key`, + `${root}/../${path.basename(root)}/openrouter-api-key`, + `${root}/./openrouter-api-key`, + `${root}//openrouter-api-key`, `${keyFile}/`, ]; for (const override of overrides) { const error = await errorOf(() => loadProviderCredential({ provider: 'dsh', - dshModel: 'muse-spark-1.2-contributor', - source: { HOME: root, CODEX_CO_ENGINEER_MODEL_API_KEY_FILE: override }, + dshModel: 'meta/muse-spark-1.3-contributor', + source: { HOME: root, CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE: override }, })); assert.equal(error.code, 'invalid_credential_path', override); assert.equal(error.message.includes(override), false, override); diff --git a/plugins/codex-co-engineer/test/r1-credential-boundary.test.mjs b/plugins/codex-co-engineer/test/r1-credential-boundary.test.mjs index 1e43c21..d1c2331 100644 --- a/plugins/codex-co-engineer/test/r1-credential-boundary.test.mjs +++ b/plugins/codex-co-engineer/test/r1-credential-boundary.test.mjs @@ -103,8 +103,8 @@ test('Muse Ox Grok and Cursor Cloud routes do not share credentials', async () = const local = await materializeProviderEnvironment({ provider: 'cursor-local', source: HOSTILE_ENV, operation: 'lane', }); - assert.equal(muse.MODEL_API_KEY, HOSTILE_ENV.MODEL_API_KEY); - assert.equal(muse.OPENROUTER_API_KEY, undefined); + assert.equal(muse.OPENROUTER_API_KEY, HOSTILE_ENV.OPENROUTER_API_KEY); + assert.equal(muse.MODEL_API_KEY, undefined); assert.equal(ox.OPENROUTER_API_KEY, HOSTILE_ENV.OPENROUTER_API_KEY); assert.equal(ox.MODEL_API_KEY, undefined); assert.equal(grok.XAI_API_KEY, HOSTILE_ENV.XAI_API_KEY); diff --git a/plugins/codex-co-engineer/test/r1-cursor-cloud-result-source-adversarial.test.mjs b/plugins/codex-co-engineer/test/r1-cursor-cloud-result-source-adversarial.test.mjs index 2a2e26e..17081dd 100644 --- a/plugins/codex-co-engineer/test/r1-cursor-cloud-result-source-adversarial.test.mjs +++ b/plugins/codex-co-engineer/test/r1-cursor-cloud-result-source-adversarial.test.mjs @@ -8,6 +8,7 @@ import { CURSOR_CLOUD_RESULT_SOURCE_IDENTITY_MISMATCH_CODES, CURSOR_CLOUD_RESULT_SOURCE_SCHEMA_ID, assertCursorCloudResultCorrelationV1, + adaptCursorCloudSdkResultV1, contentFreeCloudResultSourceFailureV1, cursorCloudResultSourceIdentityFromTaskV1, isCursorCloudResultIdentityMismatchV1, @@ -40,6 +41,7 @@ import { makeStoreRoot, providerOutput, removeRoot, + sdkResultFor, } from './fixtures/r1-cursor-cloud-result-source-fixtures.mjs'; async function expectCode(action, code, expectedPath) { @@ -335,6 +337,41 @@ test('SDK projectors reject accessors, proxies, and extra keys without invoking ); }); +test('SDK adapter rejects unsafe declared fields while ignoring unknown metadata', async () => { + for (const key of ['durationMs', 'model', 'usage', 'requestId', 'error', 'git']) { + const input = sdkResultFor(); + let getterRuns = 0; + Object.defineProperty(input, key, { + enumerable: true, + configurable: true, + get() { + getterRuns += 1; + throw new Error(`must not read ${SECRET}`); + }, + }); + const error = await expectCode( + () => adaptCursorCloudSdkResultV1(input), + 'accessor_property_denied', + `result.${key}`, + ); + assert.equal(getterRuns, 0); + assert.equal(String(error.message).includes(SECRET), false); + } + + const input = sdkResultFor(); + let unknownGetterRuns = 0; + Object.defineProperty(input, 'futureMetadata', { + enumerable: true, + get() { + unknownGetterRuns += 1; + throw new Error(`must not read ${SECRET}`); + }, + }); + const adapted = adaptCursorCloudSdkResultV1(input); + assert.equal(unknownGetterRuns, 0); + assert.equal(Object.hasOwn(adapted, 'futureMetadata'), false); +}); + test('SDK projectors copy caller-owned output and never freeze the input', () => { const output = { text: 'hello', nested: { n: 1 } }; const error = { message: 'bounded' }; diff --git a/plugins/codex-co-engineer/test/r1-cursor-cloud-result-source.test.mjs b/plugins/codex-co-engineer/test/r1-cursor-cloud-result-source.test.mjs index f21fcb4..9096e02 100644 --- a/plugins/codex-co-engineer/test/r1-cursor-cloud-result-source.test.mjs +++ b/plugins/codex-co-engineer/test/r1-cursor-cloud-result-source.test.mjs @@ -18,6 +18,7 @@ import { CURSOR_CLOUD_RESULT_SOURCE_SCHEMA_ID, CURSOR_CLOUD_RESULT_SOURCE_SLOT_KEYS, CURSOR_CLOUD_RESULT_SOURCE_VERSION, + adaptCursorCloudSdkResultV1, cursorCloudGitEvidencePathV1, cursorCloudProviderReportPathV1, materializeCursorCloudResultSourceV1, @@ -226,6 +227,58 @@ test('the SDK projector never copies provider output into Git evidence', () => { assert.equal(projected.provider_report.output.git.head_sha, HOSTILE_SHA); }); +test('the SDK adapter consumes RunResult metadata before the strict receipt projector', async () => { + const sdkResult = sdkResultFor({ + durationMs: 1_234, + model: { id: MODEL, params: [{ name: 'reasoning', value: 'high' }] }, + usage: { + inputTokens: 11, + outputTokens: 7, + cacheReadTokens: 3, + cacheWriteTokens: 2, + totalTokens: 23, + reasoningTokens: 4, + }, + benignMetadata: { trace: 'future-sdk-field' }, + }); + const adapted = adaptCursorCloudSdkResultV1(sdkResult); + assert.equal(Object.isFrozen(adapted), true); + assert.equal(Object.hasOwn(adapted, 'durationMs'), false); + assert.equal(Object.hasOwn(adapted, 'model'), false); + assert.equal(Object.hasOwn(adapted, 'usage'), false); + assert.equal(Object.hasOwn(adapted, 'benignMetadata'), false); + assert.equal(Object.isFrozen(sdkResult), false); + assert.equal(Object.isFrozen(sdkResult.git), false); + + const projected = projectCursorCloudResultSourcesV1(adapted); + assert.equal(projected.observed.provider_run_id, PROVIDER_RUN_ID); + assert.equal(projected.provider_report.status, 'finished'); + assert.equal(projected.git_evidence.branch, BRANCH); + await errorOfAsync( + () => projectCursorCloudResultSourcesV1(sdkResult), + 'unknown_key', + 'result.durationMs', + ); +}); + +test('the SDK adapter normalizes own undefined optional RunResult fields to omission', () => { + const sdkResult = sdkResultFor({ + requestId: undefined, + error: undefined, + git: undefined, + durationMs: undefined, + model: undefined, + usage: undefined, + }); + const adapted = adaptCursorCloudSdkResultV1(sdkResult); + for (const key of ['requestId', 'error', 'git', 'durationMs', 'model', 'usage']) { + assert.equal(Object.hasOwn(adapted, key), false, `expected ${key} to be omitted`); + } + const projected = projectCursorCloudResultSourcesV1(adapted); + assert.equal(projected.observed.provider_run_id, PROVIDER_RUN_ID); + assert.equal(projected.git_evidence, null); +}); + test('provider text cannot be supplied as Git evidence', async () => { await withStore(async (store) => { await errorOfAsync( diff --git a/plugins/codex-co-engineer/test/r1-cursor-local-driver-adversarial.test.mjs b/plugins/codex-co-engineer/test/r1-cursor-local-driver-adversarial.test.mjs index 8ceaee6..369ce23 100644 --- a/plugins/codex-co-engineer/test/r1-cursor-local-driver-adversarial.test.mjs +++ b/plugins/codex-co-engineer/test/r1-cursor-local-driver-adversarial.test.mjs @@ -198,7 +198,7 @@ test('the transport surface accepts only plain concrete methods', () => { test('provenance substitution and omitted-field derivation are refused on every surface', () => { const dshFixture = buildCursorLocalFixtureV1({ - provider: 'dsh', model: 'muse-spark-1.2-contributor', run_id: 'substituted-dsh-run', + provider: 'dsh', model: 'meta/muse-spark-1.3-contributor', run_id: 'substituted-dsh-run', }); const grokFixture = buildCursorLocalFixtureV1({ provider: 'grok', model: 'grok-4', run_id: 'substituted-grok-run', diff --git a/plugins/codex-co-engineer/test/r1-dsh-acpx-driver.test.mjs b/plugins/codex-co-engineer/test/r1-dsh-acpx-driver.test.mjs index 2062fd1..754c9f9 100644 --- a/plugins/codex-co-engineer/test/r1-dsh-acpx-driver.test.mjs +++ b/plugins/codex-co-engineer/test/r1-dsh-acpx-driver.test.mjs @@ -129,9 +129,9 @@ test('describe reports bounded vocabularies and only false non-claims', () => { test('model identity data stays an informational mirror of the supervisor routing', () => { assert.deepEqual({ ...DSH_MODEL_IDENTITIES[MUSE_MODEL] }, { config_file: 'dsh-acp.yml', - credential_env: 'MODEL_API_KEY', - credential_file_env: 'CODEX_CO_ENGINEER_MODEL_API_KEY_FILE', - credential_file: 'model-api-key', + credential_env: 'OPENROUTER_API_KEY', + credential_file_env: 'CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE', + credential_file: 'openrouter-api-key', }); assert.deepEqual({ ...DSH_MODEL_IDENTITIES[OX_MODEL] }, { config_file: 'dsh-acp-ox-alpha.yml', diff --git a/plugins/codex-co-engineer/test/r1-experience-contract.test.mjs b/plugins/codex-co-engineer/test/r1-experience-contract.test.mjs index 58c3e57..afd6599 100644 --- a/plugins/codex-co-engineer/test/r1-experience-contract.test.mjs +++ b/plugins/codex-co-engineer/test/r1-experience-contract.test.mjs @@ -308,16 +308,17 @@ function assertDocsMatchFixture(contractText, journeysText, fixture) { assert.equal(folded(groupedBody).includes(fixture.codex_phrases.attention), true); assert.match(folded(groupedBody), /not a second delegation/u); - assert.match(contractText, /Luna Max project manager/u); + assert.match(contractText, /## Optional legacy Luna\/Sol host relay/u); + assert.match(folded(contractText), /Ordinary delegation stays in the current Codex task and uses the user's selected model/u); assert.match(contractText, /never silently substitutes Sol/u); - assert.match(contractText, /Normal completion never wakes Sol/u); + assert.match(folded(contractText), /normal completion never wakes Sol/iu); assert.match(contractText, /create_thread/u); assert.match(contractText, /send_message_to_thread/u); assert.match(contractText, /wait_threads/u); assert.match(contractText, /read_thread/u); - assert.match(journeysText, /## Project manager/u); + assert.match(journeysText, /## Optional legacy host relay/u); assert.match(journeysText, /does not substitute Sol/u); - assert.match(journeysText, /Luna Max is the default project manager/u); + assert.match(journeysText, /Luna Max becomes project manager for a run only/u); const askOnceBody = sectionBody(journeysText, JOURNEY_HEADINGS['no-profile-ask-once']); assert.match(folded(askOnceBody), /asks once/u); diff --git a/plugins/codex-co-engineer/test/r1-experience-presentation.test.mjs b/plugins/codex-co-engineer/test/r1-experience-presentation.test.mjs new file mode 100644 index 0000000..79983fb --- /dev/null +++ b/plugins/codex-co-engineer/test/r1-experience-presentation.test.mjs @@ -0,0 +1,169 @@ +// Presentation truthfulness: lifecycle precedence, host-owned consent, and +// executed evidence versus planned assignments. + +import assert from 'node:assert/strict'; +import { readFile } from 'node:fs/promises'; +import vm from 'node:vm'; +import path from 'node:path'; +import test from 'node:test'; +import { fileURLToPath } from 'node:url'; + +import { + EXPERIENCE_COORDINATION, + EXPERIENCE_PHRASES, + EXPERIENCE_MAX_BYTES, + projectExperience, +} from '../mcp/v3/response.mjs'; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const DISPLAY_ONLY = path.join(HERE, '..', 'mcp', 'v3', 'ui', 'display-only.js'); +const SHA = 'aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa'; + +function receipt(overrides = {}) { + return { + schema: 'codex-co-engineer.run-admission.v1', + phase: 'running', + status: 'running', + run_id: 'presentation-run', + assignment_count: 2, + complete_candidate_blocked: true, + lanes: [ + { + assignment_id: 'writer', + provider: 'grok', + role: 'implement', + required: true, + phase: 'planned', + status: 'planned', + prompt_dispatched: false, + }, + { + assignment_id: 'review', + provider: 'cursor-local', + role: 'review', + required: false, + phase: 'planned', + status: 'planned', + prompt_dispatched: false, + }, + ], + ...overrides, + }; +} + +test('pending repository consent stays attention and exposes a host-owned request', () => { + const projected = projectExperience(receipt({ + phase: 'awaiting_consent', + status: 'awaiting_consent', + consent: { + status: 'required', + request: { + kind: 'repository_exposure_consent', + run_id: 'presentation-run', + repository_identity: `sha256:${SHA}${SHA.slice(0, 24)}`, + providers: ['grok', 'cursor-local'], + scope: 'full_repository', + duration: 'this_run_only', + remote_mutation: false, + }, + }, + })); + + assert.equal(projected.card, 'attention'); + assert.equal(projected.summary.attention, EXPERIENCE_PHRASES.attention); + assert.equal(projected.summary.verified_final, null); + assert.equal(projected.attention.consent.decision_authority, 'host'); + assert.equal(projected.attention.consent.request.scope, 'full_repository'); + assert.deepEqual(projected.attention.consent.request.providers, ['grok', 'cursor-local']); + assert.match(projected.attention.consent.message, /approval.*full repository/iu); + assert.equal(projected.attention.reply, null); + assert.deepEqual(projected.attention.affected_lanes, ['review', 'writer']); + assert.equal(projected.coordination.verified_final_decisions, 0); + assert.equal(EXPERIENCE_COORDINATION.verified_final_decisions, 1); +}); + +test('explicit failed and cancelled lifecycles outrank planned-lane fallback', () => { + for (const phase of ['failed', 'cancelled']) { + const projected = projectExperience(receipt({ phase, status: phase })); + assert.equal(projected.card, 'final', phase); + assert.equal(projected.summary.verified_final, null, phase); + assert.equal(projected.coordination.verified_final_decisions, 0, phase); + } +}); + +test('planned tests and reviews remain absent until an observed outcome exists', () => { + const planned = projectExperience(receipt({ phase: 'failed', status: 'failed', lanes: [ + { assignment_id: 'planned-review', provider: 'grok', role: 'review', status: 'planned', phase: 'planned' }, + { assignment_id: 'planned-tests', provider: 'grok', role: 'verify', status: 'planned', phase: 'planned' }, + ] })); + assert.equal(planned.final.reviews.present, false); + assert.deepEqual(planned.final.reviews.lanes, []); + assert.deepEqual(planned.final.reviews.planned_lanes, ['planned-review']); + assert.equal(planned.final.tests.present, false); + assert.deepEqual(planned.final.tests.lanes, []); + assert.deepEqual(planned.final.tests.planned_lanes, ['planned-tests']); + + const completed = projectExperience(receipt({ + phase: 'completed', + status: 'completed', + complete_candidate_blocked: false, + journal: { terminal: true, run_outcome: 'completed' }, + lanes: [ + { assignment_id: 'completed-review', provider: 'grok', role: 'review', status: 'completed', phase: 'completed', prompt_dispatched: true }, + { assignment_id: 'completed-tests', provider: 'grok', role: 'verify', status: 'completed', phase: 'completed', prompt_dispatched: true }, + ], + candidate: { + composed: true, + ready_for_codex_review: true, + accepted: true, + authority: 'p35', + }, + })); + assert.equal(completed.final.reviews.present, true); + assert.deepEqual(completed.final.reviews.lanes, ['completed-review']); + assert.equal(completed.final.tests.present, true); + assert.deepEqual(completed.final.tests.lanes, ['completed-tests']); + assert.equal(completed.summary.verified_final, EXPERIENCE_PHRASES.verified_final); + assert.equal(completed.coordination.verified_final_decisions, 1); +}); + +test('display-only consent rendering has no reply or approval controls', async () => { + const source = await readFile(DISPLAY_ONLY, 'utf8'); + const sandbox = { console }; + sandbox.globalThis = sandbox; + vm.createContext(sandbox); + vm.runInContext(source, sandbox, { filename: DISPLAY_ONLY }); + const ui = sandbox.CodexCoEngineerExperienceUi; + const projected = projectExperience(receipt({ + phase: 'awaiting_consent', + consent: { + status: 'required', + request: { + kind: 'repository_exposure_consent', + providers: ['grok'], + scope: 'full_repository', + duration: 'this_run_only', + remote_mutation: false, + repository_identity: `sha256:${SHA}${SHA.slice(0, 24)}`, + }, + }, + })); + const html = ui.renderConsentCardHtml(projected); + assert.match(html, /Host-owned decision/u); + assert.match(html, /full repository/u); + assert.equal(ui.documentContainsActionControls(html), false); + assert.equal(html.includes('tools/call'), false); + const session = ui.createDisplayOnlySession({ card: 'attention' }); + assert.equal(session.paint(projected), true); + assert.equal(session.outbound.length, 0); + assert.ok(Buffer.byteLength(JSON.stringify(projected), 'utf8') <= EXPERIENCE_MAX_BYTES); +}); + +test('cancellation before dispatch does not report a review or test result', () => { + const projected = projectExperience(receipt({ phase: 'cancelled', status: 'cancelled', lanes: [ + { assignment_id: 'review', role: 'review', status: 'cancelled', prompt_dispatched: false }, + { assignment_id: 'tests', role: 'verify', status: 'cancelled', prompt_dispatched: false }, + ] })); + assert.equal(projected.final.reviews.present, false); + assert.equal(projected.final.tests.present, false); +}); diff --git a/plugins/codex-co-engineer/test/r1-experience-response-adversarial.test.mjs b/plugins/codex-co-engineer/test/r1-experience-response-adversarial.test.mjs index 36ed54b..06e5759 100644 --- a/plugins/codex-co-engineer/test/r1-experience-response-adversarial.test.mjs +++ b/plugins/codex-co-engineer/test/r1-experience-response-adversarial.test.mjs @@ -113,7 +113,7 @@ test('unsupported same-session reply marks the lane unresolved and does not inve assignmentId: 'dsh-lane', taskId: 'task-dsh', provider: 'dsh', - model: 'muse-spark-1.2-contributor', + model: 'meta/muse-spark-1.3-contributor', writeScope: ['docs/**'], }), ]; diff --git a/plugins/codex-co-engineer/test/r1-experience-response.test.mjs b/plugins/codex-co-engineer/test/r1-experience-response.test.mjs index 6ec5798..5d7f9f0 100644 --- a/plugins/codex-co-engineer/test/r1-experience-response.test.mjs +++ b/plugins/codex-co-engineer/test/r1-experience-response.test.mjs @@ -32,6 +32,7 @@ import { readExperienceUiResourceForClient, resolveExperienceResultMeta, resolveExperienceToolMeta, + preparingPhrase, runningPhrase, sanitizeToolPayload, } from '../mcp/v3/response.mjs'; @@ -117,6 +118,34 @@ test('inline run card projects objective, repository SHA, lanes, and Codex autho assert.equal(Object.hasOwn(first.run.repository, 'repository_path'), false); }); +test('simple run cards say preparing until required prompt evidence is authoritative', () => { + const receipt = { + schema: 'codex-co-engineer.run-admission.v1', + run_id: 'simple-card', + assignment_count: 2, + authoritative_required_dispatch: false, + lanes: [ + { assignment_id: 'one', provider: 'grok', role: 'implement', required: true, phase: 'prepared', prompt_dispatched: false, dispatch_confidence: 'not_sent' }, + { assignment_id: 'two', provider: 'cursor-local', role: 'review', required: true, phase: 'session_ready', prompt_dispatched: false, dispatch_confidence: 'not_sent' }, + ], + }; + const preparing = projectExperience(receipt); + assert.equal(preparing.summary.running, preparingPhrase(2)); + assert.equal(preparing.summary.phrases.includes('Co-Engineer is running 2 independent assignments'), false); + + const running = projectExperience({ + ...receipt, + authoritative_required_dispatch: true, + lanes: receipt.lanes.map((lane) => ({ + ...lane, + phase: 'prompt_dispatched', + prompt_dispatched: true, + dispatch_confidence: 'authoritative', + })), + }); + assert.equal(running.summary.running, 'Co-Engineer is running 2 independent assignments'); +}); + test('grouped attention card collects questions once, marks lanes, and keeps one structured reply', async () => { const receipt = await loadJson('attention-receipt.json'); const projection = projectExperience(receipt); @@ -170,6 +199,22 @@ test('final card buckets lanes, git identity, and evidence without merge control assert.equal(projection.final.controls.create_pr, false); }); +test('final card reports failed_pre_prompt lanes as failures', async () => { + const receipt = await loadJson('final-receipt.json'); + receipt.lanes.push({ + assignment_id: 'startup-failure', + task_id: 'startup-failure-task', + provider: 'grok', + status: 'failed_pre_prompt', + required: true, + prompt_dispatched: false, + }); + receipt.assignment_count = receipt.lanes.length; + const projection = projectExperience(receipt); + assert.equal(projection.card, 'final'); + assert.deepEqual(projection.final.failed_lanes, ['docs', 'startup-failure']); +}); + test('verified-final sentence is used only for an accepted complete candidate', async () => { const receipt = await loadJson('final-receipt.json'); const lanesOnly = structuredClone(receipt); diff --git a/plugins/codex-co-engineer/test/r1-experience-ui.test.mjs b/plugins/codex-co-engineer/test/r1-experience-ui.test.mjs index e8207e1..2bf002a 100644 --- a/plugins/codex-co-engineer/test/r1-experience-ui.test.mjs +++ b/plugins/codex-co-engineer/test/r1-experience-ui.test.mjs @@ -24,6 +24,7 @@ import { } from '../mcp/v3/experience-ui-resource.mjs'; import { EXPERIENCE_PHRASES, + EXPERIENCE_RESULT_META_KEY, EXPERIENCE_UI_RESOURCE_URIS, MCP_APPS_EXTENSION_ID, MCP_APPS_MIME_TYPE, @@ -201,6 +202,155 @@ test('compatible Apps plus resources clients receive nested metadata for all thr assert.equal(Object.hasOwn(wrapped._meta, 'ui/resourceUri'), false); }); +test('compact semantic protocol payloads render run, attention, consent, and final UI from result metadata', () => { + const capabilities = compatibleCapabilities(); + const resources = experienceUiResourcesForClient(capabilities); + const protocolPayload = (compact, experienceSource = compact) => { + assert.equal(Object.hasOwn(compact, 'experience'), false); + assert.equal(Object.hasOwn(compact, 'checks'), false); + assert.equal(Object.hasOwn(compact, 'telemetry'), false); + const experience = projectExperience(experienceSource); + const uiMeta = resolveExperienceResultMeta({ + card: experience.card, experience, clientCapabilities: capabilities, resources, + }); + const result = buildToolResult(compact, { responseMode: 'structured', uiMeta }); + assert.deepEqual(result.structuredContent, compact); + assert.equal(Object.hasOwn(result.structuredContent, 'experience'), false); + assert.ok(Buffer.byteLength(JSON.stringify(result._meta[EXPERIENCE_RESULT_META_KEY]), 'utf8') <= 8_192); + return { + experience, + message: { + jsonrpc: '2.0', + method: 'ui/notifications/tool-result', + params: { result }, + }, + }; + }; + const lane = (status, provider = 'grok') => ({ + assignment_id: 'validator', + task_id: 'compact-ui-validator', + provider, + status, + required: true, + prompt_dispatched: status !== 'planned', + ...(status !== 'planned' ? { dispatch_confidence: 'authoritative' } : {}), + }); + const compact = (runId, status, extra = {}) => ({ + schema: 'codex-co-engineer.run-admission.v1', + version: 1, + mode: 'run', + operation: 'status', + run_id: runId, + status, + phase: status, + assignment_count: 1, + lanes: [lane(status)], + cursor: '4', + revision: 4, + diagnostics: { view: 'diagnostics' }, + ...extra, + }); + + const run = protocolPayload(compact('compact-ui-run', 'running', { + authoritative_required_dispatch: true, + })); + const runExtracted = ui.unwrapExperience(run.message); + assert.equal(JSON.stringify(runExtracted), JSON.stringify(run.experience)); + assert.match(ui.visiblePlainText(ui.renderInlineCardHtml('run', runExtracted)), /running 1 independent assignment/u); + + const attention = protocolPayload(compact('compact-ui-attention', 'needs_attention', { + operation: 'attention', + attention: { + status: 'open', + batch_id: 'batch-ui-1', + revision: 3, + items: [{ + assignment_id: 'validator', + task_id: 'compact-ui-validator', + question_id: 'strictness', + session_id: 'session-ui', + question: 'Use the stricter validator?', + options: ['stricter', 'compatible'], + event_cursor: '7', + }], + }, + })); + const attentionExtracted = ui.unwrapExperience(attention.message); + assert.equal(JSON.stringify(attentionExtracted), JSON.stringify(attention.experience)); + const attentionText = attentionUi.visiblePlainText(attentionUi.renderAttentionCardHtml(attentionExtracted)); + assert.match(attentionText, /Use the stricter validator\?/u); + assert.match(attentionText, /stricter/u); + assert.equal(attentionUi.bindAttention(attentionExtracted).batch_id, 'batch-ui-1'); + + const consent = protocolPayload(compact('compact-ui-consent', 'awaiting_consent', { + operation: 'submit', + lanes: [lane('planned', 'cursor-cloud')], + consent: { + status: 'required', + request: { + kind: 'repository_exposure_consent', + run_id: 'compact-ui-consent', + repository_identity: `sha256:${'a'.repeat(64)}`, + providers: ['cursor-cloud'], + scope: 'full_repository', + duration: 'this_run_only', + remote_mutation: false, + }, + }, + })); + const consentText = ui.visiblePlainText( + ui.renderInlineCardHtml('attention', ui.unwrapExperience(consent.message)), + ); + assert.match(consentText, /approval to share the full repository/u); + assert.match(consentText, /Using Cursor Co-Engineer/u); + + const finalCompact = compact('compact-ui-final', 'completed', { + authoritative_required_dispatch: true, + complete_candidate_blocked: false, + lanes: [{ + ...lane('completed'), + role: 'review', + artifacts: { + branch: 'codex/compact-ui-final', + head: 'b'.repeat(40), + clean: true, + }, + }], + candidate: { + ref: 'refs/codex-co-engineer/runs/compact-ui-final/candidate', + head: 'b'.repeat(40), + tree: 'c'.repeat(40), + ready_for_codex_review: true, + accepted: true, + }, + }); + const final = protocolPayload(finalCompact, { + ...finalCompact, + lanes: [{ + ...finalCompact.lanes[0], + handoff: { + branch: 'codex/compact-ui-final', + current_head: 'b'.repeat(40), + tree_sha: 'c'.repeat(40), + }, + }], + evidence: { + facts: [{ fact_kind: 'git_identity' }], + claims: [{ claim_kind: 'tests_passed' }], + digest: `sha256:${'d'.repeat(64)}`, + }, + }); + const finalExtracted = ui.unwrapExperience(final.message); + assert.equal(JSON.stringify(finalExtracted), JSON.stringify(final.experience)); + const finalText = ui.visiblePlainText(ui.renderInlineCardHtml('final', finalExtracted)); + assert.match(finalText, /compact-ui-final/u); + assert.match(finalText, /b{40}/u); + assert.match(finalText, /validator/u); + assert.match(finalText, /git_identity/u); + assert.match(finalText, /tests_passed/u); + assert.match(finalText, /Ready for Codex review yes/u); +}); + test('run card HTML shows objective, repository SHA, lanes, and Codex authority', async () => { const receipt = await loadJson(RESPONSE_DIR, 'run-receipt.json'); const projection = projectExperience(receipt); diff --git a/plugins/codex-co-engineer/test/r1-final-art-readme.test.mjs b/plugins/codex-co-engineer/test/r1-final-art-readme.test.mjs index 572d75f..661d447 100644 --- a/plugins/codex-co-engineer/test/r1-final-art-readme.test.mjs +++ b/plugins/codex-co-engineer/test/r1-final-art-readme.test.mjs @@ -328,16 +328,16 @@ test('README does not autoplay audio and does not embed a GitHub video player', assert.doesNotMatch(readme, /Optional silent architecture animation/u); }); -test('package and marketplace stay on 3.4.0 with the five-tool catalog', async () => { +test('package and marketplace stay on 3.4.2 with the five-tool catalog', async () => { const plugin = JSON.parse(await readFile(path.join(ROOT, '.codex-plugin', 'plugin.json'), 'utf8')); const marketplace = JSON.parse( await readFile(path.join(REPO, '.agents', 'plugins', 'marketplace.json'), 'utf8'), ); const packageJson = JSON.parse(await readFile(path.join(ROOT, 'package.json'), 'utf8')); const readme = await readFile(path.join(REPO, 'README.md'), 'utf8'); - assert.equal(plugin.version, '3.4.0'); - assert.equal(marketplace.plugins[0].version, '3.4.0'); - assert.equal(packageJson.version, '3.4.0'); + assert.equal(plugin.version, '3.4.2'); + assert.equal(marketplace.plugins[0].version, '3.4.2'); + assert.equal(packageJson.version, '3.4.2'); assert.match(readme, /The catalog remains exactly `status`, `delegate`, `task`, `tasks`, and\s+`cancel`/u); for (const tool of FIVE_TOOLS) { assert.match(readme, new RegExp(`\`${tool}\``, 'u')); diff --git a/plugins/codex-co-engineer/test/r1-local-provider-result-sink.test.mjs b/plugins/codex-co-engineer/test/r1-local-provider-result-sink.test.mjs index 7db230a..68aa312 100644 --- a/plugins/codex-co-engineer/test/r1-local-provider-result-sink.test.mjs +++ b/plugins/codex-co-engineer/test/r1-local-provider-result-sink.test.mjs @@ -452,6 +452,25 @@ test('Grok ACP with run identity stores the complete result after terminal publi assert.ok(stored.length > String(terminal.result).length); }); +test('Grok ACP sinks only the final reliably framed assistant response', async () => { + const value = await workerFixture({ + provider: 'grok', + id: 'grok-final-frame-sink', + run_id: RUN_ID, + assignment_id: 'lane-final-frame', + model: GROK_MODEL, + prompt: 'review the framed result', + mode: 'framed-final', + }); + const terminal = await runAcpTask({ root: value.root, taskId: value.taskId }); + assert.equal(terminal.result, 'fake-final-answer'); + assert.equal(terminal.provider_result_sink.published, true); + assert.equal(terminal.provider_result_sink.inline_tail.text, 'fake-final-answer'); + const store = await openLocalProviderArtifactStoreV1(value.root); + const stored = await readAllSanitized(store, terminal.provider_result_sink.sanitized_ref); + assert.equal(stored, 'fake-final-answer'); +}); + test('Cursor Local ACP with run identity preserves legacy bounded result beside artifacts', async () => { const value = await workerFixture({ provider: 'cursor-local', diff --git a/plugins/codex-co-engineer/test/r1-native-consent-roundtrip.test.mjs b/plugins/codex-co-engineer/test/r1-native-consent-roundtrip.test.mjs new file mode 100644 index 0000000..0d893dd --- /dev/null +++ b/plugins/codex-co-engineer/test/r1-native-consent-roundtrip.test.mjs @@ -0,0 +1,304 @@ +import assert from 'node:assert/strict'; +import { spawn, execFileSync } from 'node:child_process'; +import { existsSync } from 'node:fs'; +import { mkdtemp, mkdir, writeFile, rm } from 'node:fs/promises'; +import path from 'node:path'; +import os from 'node:os'; +import readline from 'node:readline'; +import { fileURLToPath } from 'node:url'; +import test from 'node:test'; + +const SERVER = fileURLToPath(new URL('../mcp/v3/server.mjs', import.meta.url)); +const CONSENT_CLI = fileURLToPath(new URL('../bin/consent-grants.mjs', import.meta.url)); + +async function withClient(capabilities, formResult, exercise, options = {}) { + const ownsRoot = options.root === undefined; + const root = options.root ?? await mkdtemp(path.join(os.tmpdir(), 'co-engineer-consent-roundtrip-')); + const repo = path.join(root, 'repo'); + const fixtureBin = path.join(root, 'bin'); + await Promise.all([mkdir(repo, { recursive: true }), mkdir(fixtureBin, { recursive: true })]); + // Keep unrelated SDK readiness discovery from running real npm and writing logs after teardown. + await writeFile(path.join(fixtureBin, 'npm'), '#!/bin/sh\nexit 1\n', { mode: 0o755 }); + const git = (...args) => execFileSync('git', args, { cwd: repo, stdio: 'pipe' }); + if (!existsSync(path.join(repo, '.git'))) { + git('init', '--quiet'); + git('config', 'user.name', 'Fixture'); + git('config', 'user.email', 'fixture@example.test'); + await writeFile(path.join(repo, 'README.md'), 'Synthetic consent fixture.\n'); + git('add', 'README.md'); + git('-c', 'commit.gpgsign=false', 'commit', '--quiet', '-m', 'Fixture'); + } + const sha = git('rev-parse', 'HEAD').toString().trim(); + const child = spawn(process.execPath, ['--no-warnings', SERVER, '--stdio'], { + env: { + PATH: `${fixtureBin}${path.delimiter}${process.env.PATH}`, + HOME: root, + XDG_CONFIG_HOME: path.join(root, 'config'), + XDG_STATE_HOME: path.join(root, 'state'), + // An accepted form must stop before real workspaces or provider jobs. + XDG_RUNTIME_DIR: path.join(root, 'unavailable-runtime'), + DBUS_SESSION_BUS_ADDRESS: `unix:path=${path.join(root, 'absent-bus')}`, + CODEX_CO_ENGINEER_STATE_DIR: path.join(root, 'receipts'), + CODEX_CO_ENGINEER_GROK_COMMAND: '/bin/false', + CODEX_CO_ENGINEER_CURSOR_COMMAND: '/bin/false', + CODEX_CO_ENGINEER_DSH_COMMAND: '/bin/false', + CODEX_CO_ENGINEER_ACPX_COMMAND: '/bin/false', + }, + stdio: ['pipe', 'pipe', 'pipe'], + }); + const pending = new Map(); + const forms = []; + let nextId = 0; + let stderr = ''; + child.stderr.on('data', (chunk) => { stderr = `${stderr}${chunk}`.slice(-2048); }); + const lines = readline.createInterface({ input: child.stdout }); + lines.on('line', (line) => { + const message = JSON.parse(line); + if (message.method === 'elicitation/create') { + forms.push(message); + const result = typeof formResult === 'function' ? formResult(forms.length) : formResult; + if (result !== undefined) { + child.stdin.write(`${JSON.stringify({ jsonrpc: '2.0', id: message.id, result })}\n`); + } + return; + } + const item = pending.get(message.id); + if (!item) return; + clearTimeout(item.timer); + pending.delete(message.id); + item.resolve(message); + }); + child.once('exit', () => { + for (const item of pending.values()) { + clearTimeout(item.timer); + item.reject(new Error(`Server exited: ${stderr}`)); + } + pending.clear(); + }); + const request = (method, params) => new Promise((resolve, reject) => { + const id = ++nextId; + const timer = setTimeout(() => { + pending.delete(id); + reject(new Error(`Timed out: ${method}; ${stderr}`)); + }, 15000); + pending.set(id, { resolve, reject, timer }); + child.stdin.write(`${JSON.stringify({ jsonrpc: '2.0', id, method, params })}\n`); + }); + const call = async (name, args) => { + const response = await request('tools/call', { name, arguments: { ...args, response_mode: 'structured' } }); + assert.equal(response.error, undefined); + assert.ok(response.result?.structuredContent, JSON.stringify(response.result)); + return response.result.structuredContent; + }; + try { + const initialized = await request('initialize', { protocolVersion: '2025-11-25', capabilities, clientInfo: { name: 'fixture', version: '1' } }); + await exercise({ root, repo, sha, forms, call, request, initialized: initialized.result }); + } finally { + for (const item of pending.values()) clearTimeout(item.timer); + pending.clear(); + child.stdin.end(); + child.kill('SIGTERM'); + lines.close(); + await new Promise((resolve) => child.exitCode !== null || child.signalCode !== null ? resolve() : child.once('exit', resolve)); + if (ownsRoot) await rm(root, { recursive: true, force: true }); + } +} + +function submission(repo, runId = 'native-consent-roundtrip', providers = ['cursor-local']) { + return { run_request: { + run_id: runId, repo, + objective: 'Exercise native consent without launching a provider.', + assignments: providers.map((provider, index) => ({ + assignment_id: `review-${index}`, provider, role: 'review', + prompt: 'Review the synthetic README.', expected_duration_ms: 60000, + })), + } }; +} + +async function noDispatch(call, receipt) { + assert.equal(receipt.authoritative_required_dispatch, false); + assert.ok(receipt.lanes.every((lane) => !lane.prompt_dispatched)); + const diagnostics = await call('task', { + run_id: receipt.run_id, view: 'diagnostics', wait_ms: 0, + }); + assert.equal(diagnostics.side_effects.provider_dispatched, false); + assert.ok(diagnostics.lanes.every((lane) => !lane.prepared && !lane.prompt_dispatched)); + return diagnostics; +} + +for (const [action, approved] of [['decline', false], ['cancel', false], ['accept', false], ['accept', 'true']]) { + test(`stdio native form ${action}/${approved} never approves repository exposure`, async () => { + const result = action === 'accept' ? { action, content: { approved } } : { action }; + await withClient({ elicitation: { form: {} } }, result, async ({ repo, forms, call }) => { + const receipt = await call('delegate', submission(repo)); + assert.equal(forms.length, 1); + assert.notEqual(receipt.consent.status, 'approved'); + await noDispatch(call, receipt); + const inspected = await call('task', { run_id: receipt.run_id, wait_until: 'decision_or_attention', wait_ms: 0 }); + assert.equal(inspected.run_id, receipt.run_id); + assert.equal(inspected.phase, receipt.phase); + assert.equal(forms.length, 1, 'inspection must not request another form'); + }); + }); +} + +test('stdio native acceptance crosses consent and stops at the isolated readiness barrier', async () => { + await withClient({ elicitation: { form: {} } }, { + action: 'accept', content: { approval_duration: 'This run only' }, + }, async ({ repo, sha, forms, call }) => { + const receipt = await call('delegate', submission(repo)); + assert.equal(forms.length, 1); + const form = forms[0].params; + assert.ok(form.message.includes(repo)); + assert.ok(form.message.includes(sha)); + assert.equal(form.requestedSchema.properties.approval_duration.type, 'string'); + assert.deepEqual(form.requestedSchema.properties.approval_duration.enum, + ['Remember for this repository and these providers', 'This run only']); + assert.ok(form.requestedSchema.required.includes('approval_duration')); + assert.equal(receipt.consent.status, 'approved'); + assert.equal(receipt.consent.duration, 'this_run_only'); + assert.equal(receipt.phase, 'failed'); + const diagnostics = await noDispatch(call, receipt); + assert.equal(diagnostics.telemetry.admission_failure_stage, 'readiness'); + const inspected = await call('status', { run_id: receipt.run_id }); + assert.equal(inspected.run_id, receipt.run_id); + assert.equal(inspected.phase, 'failed'); + assert.equal(forms.length, 1); + }); +}); + +test('stdio host without form capability gives an explicit blocker without elicitation', async () => { + await withClient({}, null, async ({ repo, forms, call }) => { + const receipt = await call('delegate', submission(repo)); + assert.equal(forms.length, 0); + assert.equal(receipt.error.code, 'consent_host_unavailable'); + assert.equal(receipt.blockers.verification, true); + const diagnostics = await noDispatch(call, receipt); + assert.equal(diagnostics.complete_candidate_blocked, true); + }); +}); + +test('a dismissed form can be reopened explicitly on the same run', async () => { + const response = (attempt) => attempt === 1 + ? { action: 'cancel' } + : { action: 'accept', content: { approval_duration: 'This run only' } }; + await withClient({ elicitation: { form: {} } }, response, async ({ repo, forms, call }) => { + const pending = await call('delegate', submission(repo)); + assert.equal(pending.phase, 'awaiting_consent'); + await noDispatch(call, pending); + const resumed = await call('task', { + run_id: pending.run_id, run_reply: { request_consent: true }, + }); + assert.equal(forms.length, 2); + assert.equal(resumed.run_id, pending.run_id); + assert.equal(resumed.consent.status, 'approved'); + const diagnostics = await noDispatch(call, resumed); + assert.equal(diagnostics.telemetry.admission_failure_stage, 'readiness'); + }); +}); + +test('status and cancellation remain usable while a native form is open', async () => { + let formOpened; + const opened = new Promise((resolve) => { formOpened = resolve; }); + await withClient({ elicitation: { form: {} } }, () => { formOpened(); }, async ({ repo, call }) => { + const submissionResult = call('delegate', submission(repo)); + await opened; + const pending = await call('status', { run_id: 'native-consent-roundtrip' }); + assert.equal(pending.phase, 'awaiting_consent'); + await noDispatch(call, pending); + const cancelled = await call('cancel', { run_id: pending.run_id }); + assert.equal(cancelled.phase, 'cancelled'); + await noDispatch(call, cancelled); + await noDispatch(call, await submissionResult); + const terminal = await call('status', { run_id: pending.run_id }); + assert.equal(terminal.phase, 'cancelled'); + }); +}); + +test('stdio remembers across restart, reprompts on expansion and repository change, and observes CLI revoke', async () => { + const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-consent-restart-')); + const acceptRemember = { action: 'accept', content: { + approval_duration: 'Remember for this repository and these providers', + } }; + try { + await withClient({ elicitation: { form: {} } }, acceptRemember, + async ({ repo, forms, call }) => { + const receipt = await call('delegate', submission(repo, 'remember-first')); + assert.equal(forms.length, 1); + assert.equal(receipt.consent.duration, 'repository_and_selected_providers'); + assert.equal(receipt.consent.source, 'native_form'); + await noDispatch(call, receipt); + }, { root }); + + await withClient({ elicitation: { form: {} } }, acceptRemember, + async ({ repo, forms, call }) => { + const reused = await call('delegate', submission(repo, 'remember-restart')); + assert.equal(forms.length, 0); + assert.equal(reused.consent.duration, 'repository_and_selected_providers'); + assert.equal(reused.consent.source, 'durable_grant'); + await noDispatch(call, reused); + + const expanded = await call('delegate', submission(repo, 'remember-expanded', + ['cursor-local', 'dsh'])); + assert.equal(forms.length, 1, 'adding a provider asks again'); + assert.equal(expanded.consent.source, 'native_form'); + await noDispatch(call, expanded); + + const other = path.join(root, 'other-repo'); + await mkdir(other); + const otherGit = (...args) => execFileSync('git', args, { cwd: other, stdio: 'pipe' }); + otherGit('init', '--quiet'); + otherGit('config', 'user.name', 'Fixture'); + otherGit('config', 'user.email', 'fixture@example.test'); + await writeFile(path.join(other, 'README.md'), 'Other repository.\n'); + otherGit('add', 'README.md'); + otherGit('-c', 'commit.gpgsign=false', 'commit', '--quiet', '-m', 'Fixture'); + const different = await call('delegate', submission(other, 'remember-different')); + assert.equal(forms.length, 2, 'a different repository asks again'); + await noDispatch(call, different); + + execFileSync(process.execPath, [CONSENT_CLI, 'revoke', '--repo', repo], { + env: { ...process.env, CODEX_CO_ENGINEER_STATE_DIR: path.join(root, 'receipts') }, + stdio: 'pipe', + }); + const revoked = await call('delegate', submission(repo, 'remember-revoked')); + assert.equal(forms.length, 3, 'the running server reloads a CLI revocation'); + await noDispatch(call, revoked); + }, { root }); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); + +test('native discovery metadata describes the workflow and accepts real receipt field types', async () => { + await withClient({ elicitation: { form: {} } }, { action: 'decline' }, async ({ repo, request, call, initialized }) => { + assert.ok(initialized.instructions.length <= 512); + assert.match(initialized.instructions, /run_request/); + const { result: { tools } } = await request('tools/list', {}); + assert.equal(tools.length, 5); + const byName = new Map(tools.map(tool => [tool.name, tool])); + for (const tool of tools) { + assert.ok(tool.title); + assert.equal(tool.outputSchema.type, 'object'); + assert.equal(typeof tool.annotations.readOnlyHint, 'boolean'); + } + const cases = [ + ['status', { detail: 'compact', include_tasks: false }], + ['delegate', submission(repo)], + ['task', { run_id: 'native-consent-roundtrip', wait_ms: 0 }], + ['cancel', { run_id: 'native-consent-roundtrip' }], + ]; + for (const [name, args] of cases) { + const receipt = await call(name, args); + for (const [field, rule] of Object.entries(byName.get(name).outputSchema.properties)) { + if (!(field in receipt) || !rule.type) continue; + const value = receipt[field]; + const actual = value === null ? 'null' : Array.isArray(value) ? 'array' : typeof value; + const allowed = Array.isArray(rule.type) ? rule.type : [rule.type]; + assert.ok(allowed.includes(actual) || (allowed.includes('integer') && Number.isInteger(value)), + `${name}.${field}: advertised ${allowed}, received ${actual}`); + if (rule.enum) assert.ok(rule.enum.includes(value), `${name}.${field}: unexpected enum value`); + } + } + }); +}); diff --git a/plugins/codex-co-engineer/test/r1-native-consent-transport.test.mjs b/plugins/codex-co-engineer/test/r1-native-consent-transport.test.mjs new file mode 100644 index 0000000..0d8636c --- /dev/null +++ b/plugins/codex-co-engineer/test/r1-native-consent-transport.test.mjs @@ -0,0 +1,338 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; + +import { + NATIVE_CONSENT_MAX_PENDING, + NATIVE_CONSENT_METHOD, + createNativeConsentTransport, + supportsNativeForm, +} from '../mcp/v3/consent.mjs'; + +function compiled(runId = 'consent-run', providers = ['grok', 'cursor-local']) { + return { + run_id: runId, + git: { + repository_path: '/tmp/consent-repository', + base_sha: '0123456789abcdef0123456789abcdef01234567', + }, + assignments: providers.map((provider, index) => ({ + assignment_id: `lane-${index + 1}`, + provider, + })), + }; +} + +function transport(options = {}) { + const sent = []; + const instance = createNativeConsentTransport({ + send: (message) => sent.push(message), + getCapabilities: () => ({ elicitation: { form: {} } }), + getProtocolVersion: () => '2025-11-25', + ...options, + }); + return { instance, sent }; +} + +test('native form capability detection fails closed for missing and URL-only clients', () => { + assert.equal(supportsNativeForm({}), false); + assert.equal(supportsNativeForm({ elicitation: null }), false); + assert.equal(supportsNativeForm({ elicitation: false }), false); + assert.equal(supportsNativeForm({ elicitation: { url: {} } }), false); + assert.equal(supportsNativeForm({ elicitation: { form: null } }), false); + assert.equal(supportsNativeForm({ elicitation: {} }), true); + assert.equal(supportsNativeForm({ elicitation: { form: {} } }), true); +}); + +test('native form sends a bounded request and Accept approves the explicit duration choice', async () => { + const { instance, sent } = transport(); + const pending = instance.requestConsent(compiled()); + assert.equal(sent.length, 1); + assert.equal(sent[0].method, NATIVE_CONSENT_METHOD); + assert.equal(sent[0].params.mode, 'form'); + assert.match(sent[0].params.message, /Repository: \/tmp\/consent-repository/u); + assert.match(sent[0].params.message, /Base SHA: 0123456789abcdef0123456789abcdef01234567/u); + assert.match(sent[0].params.message, /full repository and Git history/u); + assert.match(sent[0].params.message, /remember approval/u); + assert.match(sent[0].params.message, /this run only/u); + assert.match(sent[0].params.message, /Remote mutations: none/u); + assert.deepEqual(sent[0].params.requestedSchema.required, ['approval_duration']); + assert.deepEqual(sent[0].params.requestedSchema.properties.approval_duration.enum, + ['Remember for this repository and these providers', 'This run only']); + assert.equal(sent[0].params.requestedSchema.properties.approval_duration.default, + 'Remember for this repository and these providers'); + + assert.equal(instance.handleMessage({ + jsonrpc: '2.0', + id: sent[0].id, + result: { action: 'accept', content: { approval_duration: 'This run only' } }, + }), true); + assert.deepEqual(await pending, { + approved: true, duration: 'this_run_only', source: 'native_form', + }); + assert.equal(instance.pendingCount, 0); +}); + +test('decline, malformed, timeout, abort, and cancel are fail-closed and late responses are ignored', async () => { + const { instance, sent } = transport({ timeoutMs: 25 }); + const declined = instance.requestConsent(compiled('decline-run')); + instance.handleMessage({ + jsonrpc: '2.0', + id: sent[0].id, + result: { action: 'decline' }, + }); + assert.deepEqual(await declined, { status: 'blocked', code: 'consent_declined' }); + + const reopened = instance.requestConsent(compiled('decline-run')); + const oldId = sent[0].id; + const newId = sent[1].id; + assert.notEqual(oldId, newId); + assert.equal(instance.handleMessage({ + jsonrpc: '2.0', + id: oldId, + result: { action: 'accept', content: { approval_duration: 'This run only' } }, + }), true); + assert.equal(instance.pendingCount, 1); + instance.handleMessage({ + jsonrpc: '2.0', + id: newId, + result: { action: 'accept', content: {} }, + }); + assert.deepEqual(await reopened, { status: 'blocked', code: 'consent_response_invalid' }); + + const malformed = instance.requestConsent(compiled('malformed-run')); + instance.handleMessage({ + jsonrpc: '2.0', + id: sent[2].id, + result: { action: 'accept', content: { approval_duration: 'This run only', extra: 'nope' } }, + }); + assert.deepEqual(await malformed, { status: 'blocked', code: 'consent_response_invalid' }); + + const timed = instance.requestConsent(compiled('timed-run')); + assert.deepEqual(await timed, { status: 'required', code: 'consent_timed_out' }); + + const controller = new AbortController(); + const aborted = instance.requestConsent(compiled('aborted-run'), { signal: controller.signal }); + controller.abort(); + assert.deepEqual(await aborted, { status: 'required', code: 'consent_request_aborted' }); + + const cancelled = instance.requestConsent(compiled('cancelled-run')); + assert.equal(instance.cancelRun('cancelled-run'), true); + assert.deepEqual(await cancelled, { status: 'required', code: 'consent_cancelled' }); + + assert.equal(instance.handleMessage({ + jsonrpc: '2.0', + id: 'unknown-id', + result: { action: 'accept', content: { approved: true } }, + }), true); +}); + +test('unsupported clients never emit a form and pending consent is bounded', async () => { + const unsupported = createNativeConsentTransport({ + send: () => assert.fail('unsupported client must not receive a form'), + getCapabilities: () => ({ elicitation: { url: {} } }), + }); + assert.deepEqual(await unsupported.requestConsent(compiled()), { + status: 'blocked', + code: 'consent_host_unavailable', + }); + + const { instance, sent } = transport({ maxPending: NATIVE_CONSENT_MAX_PENDING }); + const pending = []; + for (let index = 0; index < NATIVE_CONSENT_MAX_PENDING; index += 1) { + pending.push(instance.requestConsent(compiled(`bounded-${index}`))); + } + assert.equal(sent.length, NATIVE_CONSENT_MAX_PENDING); + assert.deepEqual(await instance.requestConsent(compiled('overflow-run')), { + status: 'blocked', + code: 'consent_host_unavailable', + }); + instance.close(); + for (const value of pending) { + assert.deepEqual(await value, { status: 'required', code: 'consent_request_aborted' }); + } +}); + +test('2025-06-18 form requests omit mode while retaining the standard form schema', async () => { + const { instance, sent } = transport({ getProtocolVersion: () => '2025-06-18' }); + const pending = instance.requestConsent(compiled('compat-run')); + assert.equal(Object.hasOwn(sent[0].params, 'mode'), false); + instance.handleMessage({ + jsonrpc: '2.0', + id: sent[0].id, + result: { action: 'cancel' }, + }); + assert.deepEqual(await pending, { status: 'required', code: 'consent_cancelled' }); +}); + + +test('a matching id cannot approve through a request-shaped frame', async () => { + const { instance, sent } = transport(); + const pending = instance.requestConsent(compiled()); + instance.handleMessage({ jsonrpc: '2.0', id: sent[0].id, method: 'tools/call', + result: { action: 'accept', content: { approval_duration: 'This run only' } } }); + assert.deepEqual(await pending, { status: 'blocked', code: 'consent_response_invalid' }); +}); + +test('concurrent responses and cancellation stay bound to their run', async () => { + const { instance, sent } = transport(); + const first = instance.requestConsent(compiled('first-run')); + const second = instance.requestConsent(compiled('second-run')); + instance.cancelRun('first-run'); + instance.handleMessage({ jsonrpc: '2.0', id: sent[0].id, + result: { action: 'accept', content: { approval_duration: 'This run only' } } }); + assert.equal(instance.pendingCount, 1); + instance.handleMessage({ jsonrpc: '2.0', id: sent[1].id, + result: { action: 'accept', content: { approval_duration: 'This run only' } } }); + assert.deepEqual(await first, { status: 'required', code: 'consent_cancelled' }); + assert.deepEqual(await second, { + approved: true, duration: 'this_run_only', source: 'native_form', + }); + assert.equal(instance.pendingCount, 0); +}); + +test('remembered approval skips the form and provider expansion asks again', async () => { + const remembered = { + approved: true, + duration: 'repository_and_selected_providers', + source: 'durable_grant', + grant_id: 'a'.repeat(64), + }; + const grantedProviders = new Set(['grok']); + const { instance, sent } = transport({ + grantStore: { + resolveIdentity: async () => ({ id: 'repo' }), + assertIdentityCurrent: async () => true, + lookup: async ({ providers }) => providers.every((provider) => grantedProviders.has(provider)) + ? remembered : null, + remember: async () => assert.fail('lookup reuse must not write'), + }, + }); + assert.deepEqual(await instance.requestConsent(compiled('reuse-run', ['grok'])), remembered); + assert.equal(sent.length, 0); + const expanded = instance.requestConsent(compiled('expanded-run', ['grok', 'dsh'])); + await new Promise((resolve) => setImmediate(resolve)); + assert.equal(sent.length, 1); + instance.handleMessage({ jsonrpc: '2.0', id: sent[0].id, + result: { action: 'decline' } }); + assert.deepEqual(await expanded, { status: 'blocked', code: 'consent_declined' }); +}); + +test('lookup races recheck abort, close, and same-run pending state', async () => { + let releaseLookup; + const lookup = new Promise((resolve) => { releaseLookup = resolve; }); + const { instance, sent } = transport({ + grantStore: { + resolveIdentity: async () => ({ id: 'repo' }), + assertIdentityCurrent: async () => true, + lookup: async () => lookup, + remember: async () => {}, + }, + }); + const first = instance.requestConsent(compiled('raced-run')); + const second = instance.requestConsent(compiled('raced-run')); + releaseLookup(null); + await new Promise((resolve) => setImmediate(resolve)); + assert.equal(sent.length, 1, 'concurrent same-run lookups open one form'); + instance.handleMessage({ jsonrpc: '2.0', id: sent[0].id, + result: { action: 'accept', content: { approval_duration: 'This run only' } } }); + assert.deepEqual(await first, await second); + + let releaseAborted; + const abortedLookup = new Promise((resolve) => { releaseAborted = resolve; }); + const aborting = transport({ grantStore: { + resolveIdentity: async () => ({ id: 'repo' }), + assertIdentityCurrent: async () => true, + lookup: async () => abortedLookup, + remember: async () => {}, + } }); + const controller = new AbortController(); + const aborted = aborting.instance.requestConsent(compiled('lookup-abort'), { signal: controller.signal }); + controller.abort(); + releaseAborted(rememberedResult()); + assert.deepEqual(await aborted, { status: 'required', code: 'consent_request_aborted' }); + assert.equal(aborting.sent.length, 0); +}); + +function rememberedResult() { + return { + approved: true, + duration: 'repository_and_selected_providers', + source: 'durable_grant', + }; +} + +test('the first host response is consumed before durable persistence settles', async () => { + let finishRemember; + let rememberCalls = 0; + const { instance, sent } = transport({ grantStore: { + resolveIdentity: async () => ({ id: 'captured' }), + assertIdentityCurrent: async () => true, + lookup: async () => null, + remember: async ({ repositoryIdentity }) => { + rememberCalls += 1; + assert.deepEqual(repositoryIdentity, { id: 'captured' }); + return new Promise((resolve) => { finishRemember = resolve; }); + }, + } }); + const pending = instance.requestConsent(compiled('remember-race')); + await new Promise((resolve) => setImmediate(resolve)); + const response = { jsonrpc: '2.0', id: sent[0].id, + result: { action: 'accept', content: { + approval_duration: 'Remember for this repository and these providers', + } } }; + instance.handleMessage(response); + instance.handleMessage({ ...response, result: { action: 'decline' } }); + await new Promise((resolve) => setImmediate(resolve)); + assert.equal(rememberCalls, 1); + finishRemember(); + assert.deepEqual(await pending, { + approved: true, + duration: 'repository_and_selected_providers', + source: 'native_form', + }); +}); + +test('cancellation during identity recheck prevents a durable write', async () => { + let finishIdentityCheck; + let rememberCalls = 0; + const { instance, sent } = transport({ grantStore: { + resolveIdentity: async () => ({ id: 'captured' }), + lookup: async () => null, + assertIdentityCurrent: async () => new Promise((resolve) => { finishIdentityCheck = resolve; }), + remember: async () => { rememberCalls += 1; }, + } }); + const pending = instance.requestConsent(compiled('cancel-before-write')); + await new Promise((resolve) => setImmediate(resolve)); + instance.handleMessage({ jsonrpc: '2.0', id: sent[0].id, + result: { action: 'accept', content: { + approval_duration: 'Remember for this repository and these providers', + } } }); + await new Promise((resolve) => setImmediate(resolve)); + assert.equal(instance.cancelRun('cancel-before-write'), true); + finishIdentityCheck(true); + assert.deepEqual(await pending, { status: 'required', code: 'consent_cancelled' }); + await new Promise((resolve) => setImmediate(resolve)); + assert.equal(rememberCalls, 0); +}); + +for (const approvalDuration of [ + 'Remember for this repository and these providers', 'This run only', +]) { + test(`repository drift rejects ${approvalDuration} acceptance and remains retryable`, async () => { + const { instance, sent } = transport({ grantStore: { + resolveIdentity: async () => ({ id: 'before-form' }), + lookup: async () => null, + assertIdentityCurrent: async () => { + throw Object.assign(new Error('changed'), { code: 'consent_repository_identity_changed' }); + }, + remember: async () => assert.fail('drift must not persist'), + } }); + const pending = instance.requestConsent(compiled(`drift-${approvalDuration === 'This run only' ? 'once' : 'remember'}`)); + await new Promise((resolve) => setImmediate(resolve)); + instance.handleMessage({ jsonrpc: '2.0', id: sent[0].id, + result: { action: 'accept', content: { approval_duration: approvalDuration } } }); + assert.deepEqual(await pending, { + status: 'required', code: 'consent_repository_identity_changed', + }); + }); +} diff --git a/plugins/codex-co-engineer/test/r1-native-run-lifecycle.test.mjs b/plugins/codex-co-engineer/test/r1-native-run-lifecycle.test.mjs new file mode 100644 index 0000000..a60a9c1 --- /dev/null +++ b/plugins/codex-co-engineer/test/r1-native-run-lifecycle.test.mjs @@ -0,0 +1,115 @@ +import assert from 'node:assert/strict'; +import { execFileSync } from 'node:child_process'; +import { mkdtemp, mkdir, writeFile, rename, rm } from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import test from 'node:test'; +import { createSupervisorRunToolAdapter } from '../mcp/v3/supervisor.mjs'; +import { createTask, taskPaths } from '../mcp/v3/task-store.mjs'; + +const REVIEW = 'Fixture review completed: the documented launch flow is clear.'; + +async function fixture(exercise, taskFields = {}) { + const directory = await mkdtemp(path.join(os.tmpdir(), 'cce-native-lifecycle-')); + const repo = path.join(directory, 'repository'); + const root = path.join(directory, 'state'); + await mkdir(repo); + const git = (...args) => execFileSync('git', args, { cwd: repo, stdio: 'pipe' }); + git('init', '--quiet'); + git('config', 'user.name', 'Fixture'); + git('config', 'user.email', 'fixture@example.test'); + await writeFile(path.join(repo, 'README.md'), 'Synthetic repository.\n'); + git('add', 'README.md'); + git('-c', 'commit.gpgsign=false', 'commit', '--quiet', '-m', 'Fixture'); + const base = git('rev-parse', 'HEAD').toString().trim(); + let dispatches = 0; + const options = { + root, + requestConsent: async () => ({ approved: true }), + providerReady: async () => ({ ready: true }), + processBoundaryReady: async () => ({ ready: true }), + prepareWorkspace: async ({ assignment }) => ({ prepared: true, workspace: { + task: assignment.task_id, worktree_path: repo, branch: 'fixture', start_sha: base, + } }), + // Simulate the provider boundary only. Inspection, reconciliation, task-store + // reads, output projection and durable reopening use the production bridge. + dispatchPrompt: async ({ assignment }) => { + dispatches += 1; + await createTask({ root, prompt: 'Synthetic review request.', record: { + id: assignment.task_id, provider: 'cursor-local', role: 'review', + status: 'completed', prompt_dispatched: true, dispatch_evidence: 'authoritative', + stop_reason: 'end_turn', finished_at: new Date().toISOString(), + result: REVIEW, repo, start_sha: base, worktree_path: repo, + ...taskFields, + } }); + return { dispatched: true, confidence: 'authoritative', session_id: 'fixture-session' }; + }, + }; + try { + const adapter = await createSupervisorRunToolAdapter(options); + const submitted = await adapter.dispatch('delegate', { run_request: { + run_id: 'native-lifecycle', repo, objective: 'Review a synthetic repository.', + assignments: [{ assignment_id: 'review', provider: 'cursor-local', role: 'review', + prompt: 'Read the README and report your assessment.', expected_duration_ms: 60000 }], + } }); + await exercise({ adapter, options, submitted, root, dispatches: () => dispatches }); + } finally { + await rm(directory, { recursive: true, force: true }); + } +} + +test('native run returns actual task output through the production bridge and after restart', async () => { + await fixture(async ({ adapter, options, submitted, dispatches }) => { + const result = await adapter.dispatch('task', { + run_id: submitted.run_id, cursor: submitted.cursor, wait_until: 'decision_or_attention', wait_ms: 0, + }); + assert.equal(result.phase, 'completed'); + assert.ok(JSON.stringify(result).includes(REVIEW), 'normal run result must carry the provider answer'); + const reopened = await createSupervisorRunToolAdapter(options); + const recovered = await reopened.dispatch('status', { run_id: submitted.run_id }); + assert.equal(recovered.phase, 'completed'); + assert.equal(recovered.cursor, result.cursor); + assert.ok(JSON.stringify(recovered).includes(REVIEW)); + assert.equal(dispatches(), 1); + }); +}); + +test('an observation failure can recover the same task without another dispatch', async () => { + await fixture(async ({ adapter, submitted, root, dispatches }) => { + const record = taskPaths(root, submitted.lanes[0].task_id).record; + const hidden = `${record}.temporarily-unavailable`; + await rename(record, hidden); + const uncertain = await adapter.dispatch('status', { run_id: submitted.run_id }); + assert.notEqual(uncertain.phase, 'completed'); + assert.equal(uncertain.blockers.verification, true); + const uncertainDiagnostics = await adapter.dispatch('task', { + run_id: submitted.run_id, view: 'diagnostics', wait_ms: 0, + }); + assert.equal(uncertainDiagnostics.complete_candidate_blocked, true); + await rename(hidden, record); + const recovered = await adapter.dispatch('status', { run_id: submitted.run_id }); + assert.equal(recovered.phase, 'completed'); + assert.ok(JSON.stringify(recovered).includes(REVIEW)); + assert.equal(dispatches(), 1); + }); +}); + +for (const status of ['cancelled', 'failed', 'environment_blocked']) { + test(`a required ${status} task cannot become a successful run`, async () => { + await fixture(async ({ adapter, submitted, dispatches }) => { + const result = await adapter.dispatch('status', { run_id: submitted.run_id }); + assert.notEqual(result.phase, 'completed'); + assert.equal(result.blockers.verification, true); + assert.notEqual(result.candidate?.accepted, true); + const expectedLaneStatus = status === 'cancelled' ? 'cancelled' : 'partial_handoff'; + assert.equal(result.lanes[0].status, expectedLaneStatus); + const diagnostics = await adapter.dispatch('task', { + run_id: submitted.run_id, view: 'diagnostics', wait_ms: 0, + }); + assert.equal(diagnostics.complete_candidate_blocked, true); + assert.equal(diagnostics.lanes[0].task_final, true); + assert.equal(diagnostics.experience.card, 'final'); + assert.equal(dispatches(), 1); + }, { status, result: null, stop_reason: status }); + }); +} diff --git a/plugins/codex-co-engineer/test/r1-profile-provider-model-role-contract.test.mjs b/plugins/codex-co-engineer/test/r1-profile-provider-model-role-contract.test.mjs index 09de1bd..3836a5c 100644 --- a/plugins/codex-co-engineer/test/r1-profile-provider-model-role-contract.test.mjs +++ b/plugins/codex-co-engineer/test/r1-profile-provider-model-role-contract.test.mjs @@ -118,7 +118,7 @@ test('profile vocabulary mirrors the shared run grammar without importing it', ( test('PROFILE_DSH_MODELS stays deprecated informational data that never authorizes a model', () => { assert.ok(Object.isFrozen(PROFILE_DSH_MODELS)); - assert.deepEqual([...PROFILE_DSH_MODELS], ['muse-spark-1.2-contributor', 'stealth/ox-alpha']); + assert.deepEqual([...PROFILE_DSH_MODELS], ['meta/muse-spark-1.3-contributor', 'stealth/ox-alpha']); const declaration = profileSource.indexOf('export const PROFILE_DSH_MODELS'); assert.ok(declaration > 0, 'deprecated export must survive'); const docblock = profileSource.lastIndexOf('/**', declaration); diff --git a/plugins/codex-co-engineer/test/r1-profile.test.mjs b/plugins/codex-co-engineer/test/r1-profile.test.mjs index b62e562..683170b 100644 --- a/plugins/codex-co-engineer/test/r1-profile.test.mjs +++ b/plugins/codex-co-engineer/test/r1-profile.test.mjs @@ -397,7 +397,7 @@ test('provider, model, and role fields validate against one bounded run grammar' const providerModels = new Map([ // Opaque grammar-valid identifiers load like any other model: model IDs // are not paths, refs, commands, or credentials. - ['dsh', ['muse-spark-1.2-contributor', 'stealth/ox-alpha', 'a..b', 'sk-abcdefghijklmnop']], + ['dsh', ['meta/muse-spark-1.3-contributor', 'stealth/ox-alpha', 'a..b', 'sk-abcdefghijklmnop']], ['grok', ['grok-4', 'grok-code-fast-1', 'refs/heads/model']], ['cursor-local', ['composer-1', 'cursor_smoke_model', 'cmd:model']], ['cursor-cloud', ['claude-sonnet-4-5', 'origin/model']], diff --git a/plugins/codex-co-engineer/test/r1-provider-registry.test.mjs b/plugins/codex-co-engineer/test/r1-provider-registry.test.mjs index 9ea19e2..6a397db 100644 --- a/plugins/codex-co-engineer/test/r1-provider-registry.test.mjs +++ b/plugins/codex-co-engineer/test/r1-provider-registry.test.mjs @@ -284,7 +284,7 @@ test('resolveRegistrySelectionV1 maps exact pairs deterministically without cons provider: 'dsh', model: 'not-a-dsh-model', }), 'unknown_model', 'closed dsh model list is enforced against the accepted constant'); expectCode(() => resolveRegistrySelectionV1({ - provider: 'dsh', model: 'muse-spark-1.2-contributor ', + provider: 'dsh', model: 'meta/muse-spark-1.3-contributor ', }), 'unknown_model'); expectCode(() => resolveRegistrySelectionV1({ provider: 'dsh' }), 'missing_key'); expectCode(() => resolveRegistrySelectionV1({ model: 'm' }), 'missing_key'); diff --git a/plugins/codex-co-engineer/test/r1-readiness-snapshot.test.mjs b/plugins/codex-co-engineer/test/r1-readiness-snapshot.test.mjs new file mode 100644 index 0000000..b947240 --- /dev/null +++ b/plugins/codex-co-engineer/test/r1-readiness-snapshot.test.mjs @@ -0,0 +1,42 @@ +import assert from 'node:assert/strict'; +import { mkdtemp, readFile, rm } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import path from 'node:path'; +import test from 'node:test'; + +import { + READINESS_SNAPSHOT_FILE, + loadReadinessSnapshot, + saveReadinessSnapshot, +} from '../mcp/v3/readiness-snapshot.mjs'; + +test('readiness snapshots persist bounded content-free setup results', async () => { + const root = await mkdtemp(path.join(tmpdir(), 'co-engineer-readiness-')); + try { + const observedAt = new Date().toISOString(); + const readiness = { + grok: { installed: true, ready: false, reason: 'needs_login', probe_duration_ms: 21 }, + 'cursor-local': { installed: false, ready: false, reason: 'not_installed', probe_duration_ms: 4 }, + dsh: { + installed: true, + ready: false, + reason: 'credentials_missing', + transport: 'acpx', + model_options: { + 'meta/muse-spark-1.3-contributor': { ready: false }, + 'stealth/ox-alpha': { ready: false }, + }, + }, + 'cursor-cloud': { installed: true, ready: true, transport: 'cursor-sdk' }, + }; + await saveReadinessSnapshot(root, readiness, { observed_at: observedAt, probe_duration_ms: 28 }); + const raw = await readFile(path.join(root, READINESS_SNAPSHOT_FILE), 'utf8'); + assert.equal(raw.includes('MODEL_API_KEY'), false); + assert.equal(raw.includes('CURSOR_API_KEY'), false); + assert.equal(raw.includes('/home/'), false); + const loaded = await loadReadinessSnapshot(root); + assert.deepEqual(loaded, { observed_at: observedAt, readiness, probe_duration_ms: 28 }); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); diff --git a/plugins/codex-co-engineer/test/r1-run-admission.test.mjs b/plugins/codex-co-engineer/test/r1-run-admission.test.mjs new file mode 100644 index 0000000..78e4ca7 --- /dev/null +++ b/plugins/codex-co-engineer/test/r1-run-admission.test.mjs @@ -0,0 +1,919 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; + +import { compileRunRequestV1 } from '../mcp/v3/run-request-compiler.mjs'; +import { + createRunAdmissionRuntime, +} from '../mcp/v3/run-admission.mjs'; + +const BASE_SHA = 'a'.repeat(40); +const OBSERVED = Object.freeze({ + base_sha: BASE_SHA, + head_sha: BASE_SHA, + tree_sha: 'b'.repeat(40), + branch: 'main', + clean: true, + remote_present: true, + remote_count: 1, +}); + +function request(overrides = {}) { + const assignments = overrides.assignments ?? [ + { + assignment_id: 'lane-one', + provider: 'grok', + role: 'implement', + access: 'write', + write_scope: ['src/one/**'], + prompt: 'Implement lane one.', + expected_duration_ms: 60_000, + }, + { + assignment_id: 'lane-two', + provider: 'cursor-local', + role: 'review', + access: 'read', + prompt: 'Review lane two.', + expected_duration_ms: 60_000, + }, + ]; + return { + run_id: 'admission-test', + repo: '/tmp/fixture-repo', + objective: 'Exercise the admission state machine.', + ...overrides, + assignments, + }; +} + +function makeCompiled(requestValue) { + return compileRunRequestV1(requestValue, { + observeGit: async (_repo, base) => ({ ...OBSERVED, base_sha: base ?? BASE_SHA }), + }); +} + +function baseDependencies(overrides = {}) { + const calls = { + consent: 0, + prepare: [], + dispatch: [], + cancel: [], + attention: [], + }; + const dependencies = { + compile: makeCompiled, + requestConsent: async () => { + calls.consent += 1; + return { status: 'required' }; + }, + verifyConsent: async () => ({ + approved: true, + approved_at: '2026-09-01T00:00:00.000Z', + expires_at: '2026-09-02T00:00:00.000Z', + }), + clock: () => '2026-09-01T12:00:00.000Z', + providerReady: async () => ({ ready: true }), + processBoundaryReady: async () => ({ ready: true }), + verifyRepository: async () => ({ verified: true }), + prepareWorkspace: async ({ assignment }) => { + calls.prepare.push(assignment.assignment_id); + return { + prepared: true, + workspace: { + worktree_path: `/tmp/${assignment.assignment_id}`, + branch: `ce/${assignment.assignment_id}`, + start_sha: BASE_SHA, + }, + }; + }, + createSession: async ({ assignment }) => ({ ready: true, session_id: `${assignment.assignment_id}-session` }), + dispatchPrompt: async ({ assignment }) => { + calls.dispatch.push(assignment.assignment_id); + return { dispatched: true, confidence: 'authoritative', cursor: '1' }; + }, + inspectLane: async () => ({ status: 'running', cursor: '1' }), + replyAttention: async (value) => { + calls.attention.push(value); + return { delivered: true }; + }, + inspectWorkspace: async () => ({ + current_head: BASE_SHA, + clean: true, + changed_files: [], + commits: [], + partial_diff: false, + last_acknowledged_provider_event: 'prompt_dispatched', + }), + verifyRun: async () => ({ verified: true }), + cancelLane: async ({ assignment_id }) => { + calls.cancel.push(assignment_id); + return { confirmed: true, cancelled: true }; + }, + ...overrides, + }; + return { calls, dependencies }; +} + +test('consent pauses before workspace export and resumes the same run', async () => { + const { calls, dependencies } = baseDependencies(); + const runtime = createRunAdmissionRuntime(dependencies); + const first = await runtime.submitRunRequest(request()); + + assert.equal(first.phase, 'awaiting_consent'); + assert.equal(first.consent.status, 'required'); + assert.equal(calls.prepare.length, 0); + assert.equal(calls.dispatch.length, 0); + + const resumed = await runtime.replyRun({ run_id: 'admission-test', approval_ref: 'opaque-approval-ref' }); + assert.equal(resumed.phase, 'running'); + assert.equal(resumed.authoritative_required_dispatch, true); + assert.deepEqual(calls.dispatch, ['lane-one', 'lane-two']); + assert.equal(resumed.lanes.every((lane) => lane.prompt_dispatched), true); + assert.equal(JSON.stringify(resumed).includes('opaque-approval-ref'), false); +}); + +test('natural-language approval cannot cross the repository exposure boundary', async () => { + const { calls, dependencies } = baseDependencies({ + requestConsent: async () => ({ + status: 'required', + request: { kind: 'repository_exposure_consent', run_id: 'typed-consent' }, + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ run_id: 'typed-consent' })); + + const blocked = await runtime.replyRun({ + run_id: 'typed-consent', + attention_reply: { reply: 'yes, approved for this repository' }, + }); + assert.equal(blocked.phase, 'awaiting_consent'); + assert.equal(blocked.error.code, 'approval_ref_required'); + assert.equal(calls.prepare.length, 0); + assert.equal(calls.dispatch.length, 0); +}); + +test('a future-dated approval cannot cross the repository exposure boundary', async () => { + const { calls, dependencies } = baseDependencies({ + verifyConsent: async () => ({ + approved: true, + approved_at: '2026-09-02T00:00:00.000Z', + expires_at: '2026-09-03T00:00:00.000Z', + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ run_id: 'future-consent' })); + + const blocked = await runtime.replyRun({ + run_id: 'future-consent', + approval_ref: 'future-dated-approval', + }); + assert.equal(blocked.phase, 'awaiting_consent'); + assert.equal(blocked.error.code, 'approval_ref_invalid_or_expired'); + assert.equal(calls.prepare.length, 0); + assert.equal(calls.dispatch.length, 0); +}); + +test('one workspace admission failure dispatches zero prompts', async () => { + const { calls, dependencies } = baseDependencies({ + prepareWorkspace: async ({ assignment }) => { + calls.prepare.push(assignment.assignment_id); + if (assignment.assignment_id === 'lane-two') return { prepared: false }; + return { prepared: true, workspace: { worktree_path: `/tmp/${assignment.assignment_id}` } }; + }, + requestConsent: async () => ({ status: 'approved' }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + const receipt = await runtime.submitRunRequest(request()); + + assert.equal(receipt.phase, 'failed'); + assert.deepEqual(calls.dispatch, []); + assert.equal(receipt.lanes.every((lane) => lane.phase === 'failed_pre_prompt'), true); + assert.equal(receipt.lanes.every((lane) => lane.handoff !== null), true); +}); + +test('an all-Cursor Cloud run does not require the local process boundary', async () => { + let boundaryChecks = 0; + const { calls, dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + processBoundaryReady: async () => { + boundaryChecks += 1; + return { ready: false, reason: 'systemd_user_manager_unavailable' }; + }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const receipt = await runtime.submitRunRequest(request({ + run_id: 'cloud-only-admission', + assignments: [{ + assignment_id: 'cloud-review', + provider: 'cursor-cloud', + role: 'review', + access: 'read', + prompt: 'Review the candidate in Cursor Cloud.', + expected_duration_ms: 60_000, + }], + })); + + assert.equal(boundaryChecks, 0); + assert.equal(receipt.phase, 'running'); + assert.deepEqual(calls.dispatch, ['cloud-review']); +}); + +test('a mixed Cloud and local run still requires the local process boundary', async () => { + let boundaryChecks = 0; + const { calls, dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + processBoundaryReady: async () => { + boundaryChecks += 1; + return { ready: false, reason: 'systemd_user_manager_unavailable' }; + }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const receipt = await runtime.submitRunRequest(request({ + run_id: 'mixed-boundary-admission', + assignments: [ + { + assignment_id: 'cloud-review', provider: 'cursor-cloud', role: 'review', access: 'read', + prompt: 'Review in Cursor Cloud.', expected_duration_ms: 60_000, + }, + { + assignment_id: 'local-review', provider: 'grok', role: 'review', access: 'read', + prompt: 'Review locally.', expected_duration_ms: 60_000, + }, + ], + })); + + assert.equal(boundaryChecks, 1); + assert.equal(receipt.phase, 'failed'); + assert.deepEqual(calls.dispatch, []); +}); + +test('mid-dispatch failure identifies sent and unsent lanes and never says running', async () => { + const { calls, dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + dispatchPrompt: async ({ assignment }) => { + calls.dispatch.push(assignment.assignment_id); + if (assignment.assignment_id === 'lane-two') { + throw Object.assign(new Error('provider failed'), { code: 'provider_exit' }); + } + return { dispatched: true, confidence: 'authoritative', cursor: '1' }; + }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const receipt = await runtime.submitRunRequest(request({ + run_id: 'mid-dispatch', + assignments: [ + ...request().assignments, + { + assignment_id: 'lane-three', + provider: 'dsh', + role: 'review', + access: 'read', + prompt: 'Review lane three.', + expected_duration_ms: 60_000, + }, + ], + })); + + assert.equal(receipt.phase, 'degraded'); + assert.deepEqual(receipt.dispatched_assignment_ids, ['lane-one']); + assert.deepEqual(receipt.undispatched_assignment_ids, ['lane-two', 'lane-three']); + assert.equal(receipt.authoritative_required_dispatch, false); + assert.deepEqual(calls.dispatch, ['lane-one', 'lane-two']); + assert.equal(receipt.lanes[2].phase, 'failed_pre_prompt'); +}); + +test('pending dispatch stays active, launches independent lanes, and is never replayed', async () => { + let dispatchCount = 0; + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + dispatchPrompt: async () => { + dispatchCount += 1; + return { + dispatched: false, + dispatch_pending: true, + dispatch_uncertain: true, + confidence: 'uncertain', + }; + }, + inspectLane: async () => ({ status: 'transport_lost' }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + const first = await runtime.submitRunRequest(request({ run_id: 'uncertain-dispatch' })); + const second = await runtime.resumeRun({ run_id: 'uncertain-dispatch' }); + + assert.equal(first.phase, 'dispatching'); + assert.equal(first.lanes.every((lane) => lane.phase === 'session_ready'), true); + assert.equal(first.lanes.every((lane) => lane.task_final === false), true); + assert.equal(first.lanes.every((lane) => lane.recovery_classification === 'dispatch_pending_no_replay'), true); + assert.equal(second.phase, 'dispatching'); + assert.equal(dispatchCount, 2); +}); + +test('generic and thrown uncertain dispatches stay unresolved without falsely active lanes', async () => { + for (const mode of ['returned', 'thrown']) { + let dispatchCount = 0; + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + dispatchPrompt: async () => { + dispatchCount += 1; + if (mode === 'thrown') { + throw Object.assign(new Error('unknown dispatch outcome'), { code: 'dispatch_uncertain', sent: true }); + } + return { sent: true, dispatched: false, confidence: 'uncertain' }; + }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const receipt = await runtime.submitRunRequest(request({ run_id: `uncertain-${mode}` })); + const reconciled = await runtime.inspectRun({ run_id: receipt.run_id }); + + assert.equal(receipt.phase, 'degraded'); + assert.equal(receipt.lanes[0].phase, 'unrecoverable_post_prompt'); + assert.equal(receipt.lanes[0].recovery_classification, 'dispatch_uncertain_no_replay'); + assert.equal(receipt.lanes[1].phase, 'failed_pre_prompt'); + assert.equal(reconciled.phase, 'degraded'); + assert.equal(reconciled.lanes[0].phase, 'unrecoverable_post_prompt'); + assert.equal(dispatchCount, 1); + } +}); + +test('authoritative task-bound evidence promotes pending dispatch and clears only its uncertainty', async () => { + let completed = false; + let dispatchCount = 0; + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + dispatchPrompt: async () => { + dispatchCount += 1; + return { sent: true, dispatched: false, dispatch_pending: true, confidence: 'uncertain' }; + }, + inspectLane: async ({ task_id }) => completed + ? { + task_id, + status: 'completed', + cursor: '2', + prompt_dispatched: true, + dispatch_evidence: 'authoritative', + session_id: `${task_id}-session`, + } + : { task_id, status: 'running', cursor: '1', dispatch_uncertain: true }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const pending = await runtime.submitRunRequest(request({ run_id: 'late-dispatch-evidence' })); + const stillPending = await runtime.inspectRun({ run_id: pending.run_id }); + completed = true; + const final = await runtime.inspectRun({ run_id: pending.run_id }); + + assert.equal(stillPending.phase, 'dispatching'); + assert.equal(final.phase, 'completed'); + assert.equal(final.lanes.every((lane) => lane.prompt_dispatched), true); + assert.equal(final.lanes.every((lane) => lane.dispatch_confidence === 'authoritative'), true); + assert.equal(final.lanes.every((lane) => lane.error === null), true); + assert.equal(dispatchCount, 2, 'late acknowledgement must observe existing tasks without replay'); +}); + +test('terminal completion without authoritative dispatch evidence remains an honest ambiguity', async () => { + let dispatchCount = 0; + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + dispatchPrompt: async () => { + dispatchCount += 1; + return { sent: true, dispatched: false, dispatch_pending: true, confidence: 'uncertain' }; + }, + inspectLane: async ({ task_id }) => ({ task_id, status: 'completed', cursor: '2' }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + const submitted = await runtime.submitRunRequest(request({ run_id: 'terminal-dispatch-ambiguity' })); + const final = await runtime.inspectRun({ run_id: submitted.run_id }); + + assert.equal(final.phase, 'degraded'); + assert.equal(final.complete_candidate_blocked, true); + assert.equal(final.lanes.every((lane) => lane.task_final), true); + assert.equal(final.lanes.every((lane) => lane.phase === 'unrecoverable_post_prompt'), true); + assert.equal(final.lanes.every((lane) => lane.error.code === 'dispatch_uncertain'), true); + assert.equal(dispatchCount, 2); +}); + +test('cancellation is idempotent and does not resume a cancelled run', async () => { + const { calls, dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ run_id: 'cancel-test' })); + const first = await runtime.cancelRun({ run_id: 'cancel-test' }); + const second = await runtime.cancelRun({ run_id: 'cancel-test' }); + const resumed = await runtime.resumeRun({ run_id: 'cancel-test' }); + + assert.equal(first.phase, 'cancelled'); + assert.equal(first.cancel_requested, true); + assert.equal(second.already_terminal, true); + assert.equal(second.phase, 'cancelled'); + assert.equal(resumed.phase, 'cancelled'); + assert.equal(calls.cancel.length, 2); +}); + +test('cancellation before consent or prompt dispatch is provider-free', async () => { + const { calls, dependencies } = baseDependencies(); + const runtime = createRunAdmissionRuntime(dependencies); + const pending = await runtime.submitRunRequest(request({ run_id: 'cancel-before-dispatch' })); + const cancelled = await runtime.cancelRun({ run_id: 'cancel-before-dispatch' }); + + assert.equal(pending.phase, 'awaiting_consent'); + assert.equal(cancelled.phase, 'cancelled'); + assert.equal(cancelled.cancel_requested, true); + assert.deepEqual(calls.cancel, []); + assert.equal(cancelled.lanes.every((lane) => lane.phase === 'cancelled'), true); + assert.equal(cancelled.lanes.every((lane) => lane.handoff?.worktree === null), true); + assert.deepEqual(cancelled.lanes[0].handoff.safe_next_actions, [ + 'Review the run receipt and provider outcome.', + ]); +}); + +test('pre-authorized receipt access is satisfied without user attention and without replay', async () => { + const { calls, dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + inspectLane: async ({ assignment }) => ({ + status: 'needs_attention', + attention: { + session_id: `${assignment.assignment_id}-session`, + question_id: 'receipt-read', + capability: 'read_run_receipts', + resource: 'run_receipt', + action: 'read', + prompt: 'Read the release receipt.', + }, + cursor: '2', + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ run_id: 'safe-attention' })); + + const first = await runtime.inspectRun({ run_id: 'safe-attention' }); + const second = await runtime.inspectRun({ run_id: 'safe-attention' }); + + assert.equal(first.phase, 'running'); + assert.equal(second.phase, 'running'); + assert.equal(first.attention, null); + assert.equal(second.attention, null); + assert.equal(calls.attention.length, 2, 'each provider question is answered exactly once per lane'); + assert.equal(calls.attention.every((call) => call.capability_satisfied === true), true); + assert.equal(first.lanes.every((lane) => lane.recovery_classification === 'capability_pre_authorized'), true); +}); + +test('new equivalent attention questions are grouped once across lanes', async () => { + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + inspectLane: async ({ assignment }) => ({ + status: 'needs_attention', + attention: { + session_id: `${assignment.assignment_id}-session`, + question_id: 'permission-read', + prompt: 'May the provider read the release receipt?', + options: ['allow_once', 'cancel'], + }, + cursor: '2', + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ + run_id: 'grouped-attention', + assignments: request().assignments, + })); + + const receipt = await runtime.inspectRun({ run_id: 'grouped-attention' }); + assert.equal(receipt.phase, 'needs_attention'); + assert.equal(receipt.attention.kind, 'grouped_attention'); + assert.equal(receipt.attention.items.length, 1); + assert.equal(receipt.attention.items[0].question_id, 'permission-read'); + assert.equal(receipt.attention.items[0].targets.length, 2); + const repeated = await runtime.inspectRun({ run_id: 'grouped-attention' }); + assert.equal(repeated.cursor, receipt.cursor); + assert.deepEqual(repeated.attention, receipt.attention); +}); + +test('structured ACP options and the selected option identity survive aggregate reply delivery', async () => { + const options = [ + { optionId: 'allow-once', name: 'Allow once', kind: 'allow_once' }, + { optionId: 'reject-once', name: 'Reject', kind: 'reject_once' }, + ]; + const { calls, dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + inspectLane: async ({ assignment }) => ({ + status: 'needs_attention', + attention: { + session_id: `${assignment.assignment_id}-session`, + question_id: `permission-${assignment.assignment_id}`, + prompt: 'Allow this operation?', + options, + }, + cursor: '2', + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ run_id: 'structured-attention' })); + const attention = await runtime.inspectRun({ run_id: 'structured-attention' }); + assert.deepEqual(attention.attention.items[0].options, options); + + const item = attention.attention.items[0]; + const selected = { + assignment_id: item.assignment_id, + task_id: item.task_id, + session_id: item.session_id, + question_id: item.question_id, + response: { optionId: 'allow-once' }, + }; + await runtime.replyRun({ + run_id: 'structured-attention', + attention_reply: { reply: { answers: [selected] } }, + }); + assert.deepEqual(calls.attention[0].reply.reply.answers[0], selected); +}); + +test('hostile nested attention evidence fails closed with a partial handoff', async () => { + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + inspectLane: async () => ({ + status: 'needs_attention', + attention: { + session_id: 'lane-one-session', + question_id: 'hostile-question', + options: [{ allow: true }], + }, + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ run_id: 'hostile-attention' })); + const receipt = await runtime.inspectRun({ run_id: 'hostile-attention' }); + + assert.equal(receipt.phase, 'degraded'); + assert.equal(receipt.attention, null); + assert.equal(receipt.lanes[0].phase, 'partial_handoff'); + assert.equal(receipt.lanes[0].error.code, 'attention_evidence_invalid'); + assert.equal(receipt.lanes[0].handoff !== null, true); +}); + +test('deadline reconciliation returns an evidence-bearing partial handoff', async () => { + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + inspectLane: async () => ({ status: 'timeout', cursor: 'deadline-1' }), + inspectWorkspace: async ({ assignment_id }) => ({ + worktree: `/tmp/${assignment_id}`, + branch: `ce/${assignment_id}`, + starting_sha: BASE_SHA, + current_head: 'c'.repeat(40), + clean: false, + changed_files: ['src/partial.ts'], + commits: ['d'.repeat(40)], + partial_diff: true, + last_acknowledged_provider_event: 'file_changed', + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ run_id: 'timeout-handoff' })); + + const receipt = await runtime.inspectRun({ run_id: 'timeout-handoff' }); + const lane = receipt.lanes[0]; + assert.equal(receipt.phase, 'degraded'); + assert.equal(lane.phase, 'partial_handoff'); + assert.equal(lane.handoff.current_head, 'c'.repeat(40)); + assert.deepEqual(lane.handoff.changed_files, ['src/partial.ts']); + assert.equal(lane.handoff.partial_diff, true); + assert.equal(lane.handoff.commits[0], 'd'.repeat(40)); + assert.equal(lane.handoff.recovery_classification, 'timed_out_with_partial_work'); + assert.match(lane.handoff.safe_next_actions.join(' '), /review/i); +}); + + +test('consent-pending inspect and wait preserve the authoritative cursor and revision', async () => { + const { dependencies } = baseDependencies(); + const runtime = createRunAdmissionRuntime(dependencies); + const pending = await runtime.submitRunRequest(request({ run_id: 'pending-receipt' })); + + assert.equal(pending.phase, 'awaiting_consent'); + assert.equal(pending.consent.status, 'required'); + assert.equal(typeof pending.cursor, 'string'); + assert.equal(Number.isSafeInteger(pending.revision), true); + + const inspected = await runtime.inspectRun({ run_id: 'pending-receipt' }); + assert.equal(inspected.phase, 'awaiting_consent'); + assert.equal(inspected.cursor, pending.cursor); + assert.equal(inspected.revision, pending.revision); + + const omitted = await runtime.waitRun({ run_id: 'pending-receipt', wait_ms: 0 }); + assert.equal(omitted.phase, 'awaiting_consent'); + assert.equal(omitted.cursor, pending.cursor); + assert.equal(omitted.revision, pending.revision); + assert.equal(omitted.wait_until, 'decision_or_attention'); + assert.equal(omitted.waited_ms, 0); + + const provided = await runtime.waitRun({ + run_id: 'pending-receipt', + cursor: pending.cursor, + wait_until: 'terminal', + wait_ms: 0, + }); + assert.equal(provided.phase, 'awaiting_consent'); + assert.equal(provided.cursor, pending.cursor); + assert.equal(provided.revision, pending.revision); + assert.equal(provided.wait_until, 'terminal'); + assert.equal(provided.waited_ms, 0); +}); + +test('terminal inspect and wait remain stable across repeated reads', async () => { + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + inspectLane: async () => ({ status: 'completed', cursor: 'terminal-1' }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ run_id: 'terminal-receipt' })); + + const first = await runtime.inspectRun({ run_id: 'terminal-receipt' }); + const second = await runtime.inspectRun({ run_id: 'terminal-receipt' }); + const waited = await runtime.waitRun({ + run_id: 'terminal-receipt', + cursor: first.cursor, + wait_until: 'terminal', + wait_ms: 0, + }); + + assert.equal(first.phase, 'completed'); + assert.equal(second.phase, 'completed'); + assert.equal(second.cursor, first.cursor); + assert.equal(second.revision, first.revision); + assert.equal(waited.phase, 'completed'); + assert.equal(waited.cursor, first.cursor); + assert.equal(waited.revision, first.revision); + assert.equal(waited.wait_until, 'terminal'); + assert.equal(waited.waited_ms, 0); +}); + +test('native consent continuation carries the signal and admits only after callback approval', async () => { + const callbackOptions = []; + const { dependencies } = baseDependencies({ + requestConsent: async (_compiled, options) => { + callbackOptions.push(options); + return callbackOptions.length === 1 + ? { status: 'blocked', code: 'consent_cancelled' } + : { status: 'approved' }; + }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const pending = await runtime.submitRunRequest(request({ run_id: 'consent-continuation' })); + const controller = new AbortController(); + const resumed = await runtime.replyRun( + { run_id: 'consent-continuation', request_consent: true }, + { signal: controller.signal }, + ); + + assert.equal(pending.phase, 'awaiting_consent'); + assert.equal(pending.consent.status, 'blocked'); + assert.equal(pending.error.code, 'consent_cancelled'); + assert.equal(resumed.phase, 'running'); + assert.equal(callbackOptions.length, 2); + assert.equal(callbackOptions[1].signal, controller.signal); +}); + +test('consent approval resolving after abort never admits or dispatches', async () => { + const controller = new AbortController(); + const { calls, dependencies } = baseDependencies({ + requestConsent: async (_compiled, options) => { + assert.equal(options.signal, controller.signal); + controller.abort(); + return { status: 'approved' }; + }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const receipt = await runtime.submitRunRequest( + request({ run_id: 'aborted-consent' }), + { signal: controller.signal }, + ); + + assert.equal(receipt.phase, 'awaiting_consent'); + assert.equal(receipt.consent.status, 'blocked'); + assert.equal(receipt.error.code, 'consent_request_aborted'); + assert.deepEqual(calls.prepare, []); + assert.deepEqual(calls.dispatch, []); +}); + +test('blocked native consent preserves its exact allowlisted reason code', async () => { + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'blocked', code: 'consent_declined' }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + const receipt = await runtime.submitRunRequest(request({ run_id: 'declined-consent' })); + + assert.equal(receipt.phase, 'failed'); + assert.equal(receipt.consent.status, 'blocked'); + assert.equal(receipt.error.code, 'consent_declined'); +}); + + +test('pending consent remains inspectable while the native callback is still open', async () => { + let callbackStarted; + const callbackStartedPromise = new Promise((resolve) => { + callbackStarted = resolve; + }); + let resolveCallback; + const callbackResult = new Promise((resolve) => { + resolveCallback = resolve; + }); + const { dependencies } = baseDependencies({ + requestConsent: async () => { + callbackStarted(); + return callbackResult; + }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const submitting = runtime.submitRunRequest(request({ run_id: 'deferred-consent' })); + await callbackStartedPromise; + + const inspected = await Promise.race([ + runtime.inspectRun({ run_id: 'deferred-consent' }), + new Promise((_, reject) => setTimeout(() => reject(new Error('inspect remained queued')), 100)), + ]); + const waited = await runtime.waitRun({ run_id: 'deferred-consent', wait_ms: 0 }); + assert.equal(inspected.phase, 'awaiting_consent'); + assert.equal(waited.phase, 'awaiting_consent'); + assert.equal(inspected.cursor, waited.cursor); + assert.equal(inspected.revision, waited.revision); + + resolveCallback({ status: 'required' }); + const submitted = await submitting; + assert.equal(submitted.phase, 'awaiting_consent'); +}); + +test('unchanged running observations preserve cursor and avoid persistence', async () => { + let writes = 0; + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ approved: true }), + persistRecord: async () => { writes += 1; }, + inspectLane: async () => ({ status: 'running', cursor: '2', last_event: 'text_delta' }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request()); + const first = await runtime.inspectRun({ run_id: 'admission-test' }); + const baseline = writes; + const second = await runtime.inspectRun({ run_id: 'admission-test' }); + assert.equal(second.cursor, first.cursor); + assert.equal(writes, baseline); +}); + +test('an optional active assignment remains owned and cancellable after required completion', async () => { + const { calls, dependencies } = baseDependencies({ + requestConsent: async () => ({ approved: true }), + inspectLane: async ({ assignment_id }) => ({ status: assignment_id === 'lane-one' ? 'completed' : 'running', cursor: '2' }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + const assignments = request().assignments.map((assignment, index) => ({ ...assignment, required: index === 0 })); + await runtime.submitRunRequest(request({ assignments })); + const current = await runtime.inspectRun({ run_id: 'admission-test' }); + assert.notEqual(current.phase, 'completed'); + assert.equal(current.complete_candidate_blocked, true); + const cancelled = await runtime.cancelRun({ run_id: 'admission-test' }); + assert.equal(cancelled.phase, 'cancelled'); + assert.deepEqual(calls.cancel, ['lane-two']); +}); + +test('observation errors retain cancellation ownership and an unconfirmed cancel can be retried', async () => { + let confirm = false; + const targets = []; + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ approved: true }), + inspectLane: async () => { throw Object.assign(new Error('unavailable'), { code: 'ENOENT' }); }, + cancelLane: async ({ task_id }) => { targets.push(task_id); return { confirmed: confirm }; }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const submitted = await runtime.submitRunRequest(request()); + const uncertain = await runtime.inspectRun({ run_id: submitted.run_id }); + assert.equal(uncertain.phase, 'degraded'); + const first = await runtime.cancelRun({ run_id: submitted.run_id }); + assert.equal(first.phase, 'degraded'); + assert.equal(first.telemetry.cancel_confirmed, false); + confirm = true; + const final = await runtime.cancelRun({ run_id: submitted.run_id }); + assert.equal(final.phase, 'cancelled'); + assert.equal(final.telemetry.cancel_confirmed, true); + assert.deepEqual(targets, [...submitted.lanes, ...submitted.lanes].map(lane => lane.task_id)); +}); + +test('run waits use task events and reread the same assignments on progress', async () => { + let completed = false; + let waits = 0; + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ approved: true }), + inspectLane: async () => ({ status: completed ? 'completed' : 'running', cursor: completed ? '3' : '2' }), + waitForProgress: async ({ task_ids, cursors }) => { + waits += 1; + assert.equal(task_ids.length, 2); + assert.deepEqual(Object.values(cursors), ['2', '2']); + completed = true; + return { wait_reason: 'progress' }; + }, + sleep: async () => { assert.fail('healthy event wait must not poll'); }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request()); + const result = await runtime.waitRun({ run_id: 'admission-test', wait_until: 'terminal', wait_ms: 500 }); + assert.equal(result.phase, 'completed'); + assert.equal(waits, 1); +}); + +test('a terminal task wake without settled run progress uses bounded backoff', async () => { + let waits = 0; + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ approved: true }), + waitForProgress: async () => { waits += 1; return { wait_reason: 'terminal' }; }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request()); + const result = await runtime.waitRun({ run_id: 'admission-test', wait_until: 'terminal', wait_ms: 35 }); + assert.equal(result.phase, 'running'); + assert.ok(waits <= 2, `unexpected hot loop: ${waits} waits`); +}); + +test('a pending dispatch remains cancellable even without authoritative prompt acknowledgement', async () => { + const { calls, dependencies } = baseDependencies({ + requestConsent: async () => ({ approved: true }), + dispatchPrompt: async () => ({ + sent: true, + dispatched: false, + dispatch_pending: true, + confidence: 'uncertain', + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + const submitted = await runtime.submitRunRequest(request()); + assert.equal(submitted.lanes[0].task_final, false); + const cancelled = await runtime.cancelRun({ run_id: submitted.run_id }); + assert.deepEqual(calls.cancel, ['lane-one', 'lane-two']); + assert.equal(cancelled.lanes.every((lane) => lane.phase === 'cancelled'), true); +}); + +test('terminal pending failure preserves an allowlisted provider cause', async () => { + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ approved: true }), + dispatchPrompt: async () => ({ + sent: true, + dispatched: false, + dispatch_pending: true, + confidence: 'uncertain', + }), + inspectLane: async ({ task_id }) => ({ + task_id, + status: 'failed', + error: { code: 'provider_billing_required', message: 'PRIVATE_CREDENTIAL' }, + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + const submitted = await runtime.submitRunRequest(request({ run_id: 'pending-billing-failure' })); + const final = await runtime.inspectRun({ run_id: submitted.run_id }); + + assert.equal(final.phase, 'degraded'); + assert.equal(final.lanes.every((lane) => lane.task_final), true); + assert.equal(final.lanes.every((lane) => lane.error.code === 'provider_billing_required'), true); + assert.doesNotMatch(JSON.stringify(final), /PRIVATE_CREDENTIAL/); +}); + +test('pending dispatch reconnect observes the same task and never replays it', async () => { + let dispatchCount = 0; + let reconnectCount = 0; + const { dependencies } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + dispatchPrompt: async () => { + dispatchCount += 1; + return { sent: true, dispatched: false, dispatch_pending: true, confidence: 'uncertain' }; + }, + inspectLane: async ({ task_id }) => ({ task_id, status: 'transport_lost', cursor: '1' }), + reconnectLane: async ({ task_id }) => { + reconnectCount += 1; + return { reconnected: true, session_id: `${task_id}-session`, cursor: '1' }; + }, + }); + const runtime = createRunAdmissionRuntime(dependencies); + const submitted = await runtime.submitRunRequest(request({ run_id: 'pending-reconnect' })); + const reconnected = await runtime.inspectRun({ run_id: submitted.run_id }); + + assert.equal(reconnected.phase, 'dispatching'); + assert.equal(reconnected.lanes.every((lane) => lane.task_final === false), true); + assert.equal(dispatchCount, 2); + assert.equal(reconnectCount, 2); +}); + +for (const code of ['provider_billing_required', 'authentication_required', 'provider_rate_limited', 'unknown_private_error']) { + test(`terminal lane failure preserves safe category ${code} without provider text or replay`, async () => { + const { dependencies, calls } = baseDependencies({ + requestConsent: async () => ({ status: 'approved' }), + inspectLane: async () => ({ + status: 'failed', + error: { code, message: 'PRIVATE_PROMPT api-key=PRIVATE_CREDENTIAL' }, + }), + }); + const runtime = createRunAdmissionRuntime(dependencies); + await runtime.submitRunRequest(request({ run_id: `failure-${code.replaceAll("_", "-")}` })); + const receipt = await runtime.inspectRun({ run_id: `failure-${code.replaceAll("_", "-")}` }); + assert.equal(receipt.phase, 'degraded'); + assert.equal(receipt.lanes[0].phase, 'partial_handoff'); + assert.equal(receipt.lanes[0].error.code, code === 'unknown_private_error' ? 'failed' : code); + assert.doesNotMatch(JSON.stringify(receipt), /PRIVATE_PROMPT|PRIVATE_CREDENTIAL/); + await runtime.inspectRun({ run_id: `failure-${code.replaceAll("_", "-")}` }); + assert.equal(calls.dispatch.length, 2, 'each original lane is dispatched only once'); + }); +} diff --git a/plugins/codex-co-engineer/test/r1-run-orchestration.test.mjs b/plugins/codex-co-engineer/test/r1-run-orchestration.test.mjs index ddd835e..2d08b65 100644 --- a/plugins/codex-co-engineer/test/r1-run-orchestration.test.mjs +++ b/plugins/codex-co-engineer/test/r1-run-orchestration.test.mjs @@ -126,7 +126,7 @@ test('mixed-provider prepare isolates each closed route and never shares credent assert.equal(byId['lane-local'].provider, 'cursor-local'); assert.equal(byId['lane-cloud'].provider, 'cursor-cloud'); assert.equal(byId['lane-muse'].provider, 'dsh'); - assert.equal(byId['lane-muse'].model, 'muse-spark-1.2-contributor'); + assert.equal(byId['lane-muse'].model, 'meta/muse-spark-1.3-contributor'); assert.equal(byId['lane-ox'].provider, 'dsh'); assert.equal(byId['lane-ox'].model, 'stealth/ox-alpha'); assert.equal(byId['lane-grok'].credential_present, true); @@ -136,7 +136,8 @@ test('mixed-provider prepare isolates each closed route and never shares credent assert.equal(byId['lane-ox'].credential_present, true); assert.ok(byId['lane-grok'].projected_keys.includes('XAI_API_KEY')); assert.equal(byId['lane-grok'].projected_keys.includes('MODEL_API_KEY'), false); - assert.equal(byId['lane-muse'].projected_keys.includes('OPENROUTER_API_KEY'), false); + assert.ok(byId['lane-muse'].projected_keys.includes('OPENROUTER_API_KEY')); + assert.equal(byId['lane-muse'].projected_keys.includes('MODEL_API_KEY'), false); assert.equal(byId['lane-ox'].projected_keys.includes('MODEL_API_KEY'), false); assert.equal(byId['lane-local'].projected_keys.includes('CURSOR_API_KEY'), false); assertAlwaysFalse(receipt); @@ -168,15 +169,15 @@ test('dispatch uses closed env, never puts secrets in argv, and still creates no const grok = byProvider['grok:grok-4']; const local = byProvider['cursor-local:composer-1']; const cloud = byProvider['cursor-cloud:claude-sonnet-4-5']; - const muse = byProvider['dsh:muse-spark-1.2-contributor']; + const muse = byProvider['dsh:meta/muse-spark-1.3-contributor']; const ox = byProvider['dsh:stealth/ox-alpha']; assert.equal(grok.env.XAI_API_KEY, HOSTILE_ENV.XAI_API_KEY); assert.equal(Object.hasOwn(grok.env, 'MODEL_API_KEY'), false); assert.equal(Object.hasOwn(local.env, 'XAI_API_KEY'), false); assert.equal(Object.hasOwn(local.env, 'CURSOR_API_KEY'), false); assert.equal(cloud.env.CURSOR_API_KEY, HOSTILE_ENV.CURSOR_API_KEY); - assert.equal(muse.env.MODEL_API_KEY, HOSTILE_ENV.MODEL_API_KEY); - assert.equal(Object.hasOwn(muse.env, 'OPENROUTER_API_KEY'), false); + assert.equal(muse.env.OPENROUTER_API_KEY, HOSTILE_ENV.OPENROUTER_API_KEY); + assert.equal(Object.hasOwn(muse.env, 'MODEL_API_KEY'), false); assert.equal(ox.env.OPENROUTER_API_KEY, HOSTILE_ENV.OPENROUTER_API_KEY); assert.equal(Object.hasOwn(ox.env, 'MODEL_API_KEY'), false); for (const call of dispatcher.calls) { diff --git a/plugins/codex-co-engineer/test/r1-run-request-compiler.test.mjs b/plugins/codex-co-engineer/test/r1-run-request-compiler.test.mjs new file mode 100644 index 0000000..2c37fef --- /dev/null +++ b/plugins/codex-co-engineer/test/r1-run-request-compiler.test.mjs @@ -0,0 +1,225 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; + +import { + compileRunRequestV1, + RUN_REQUEST_DEFAULT_CAPABILITIES, + RUN_REQUEST_DEFAULT_EXPECTED_DURATION_MS, +} from '../mcp/v3/run-request-compiler.mjs'; + +const BASE_SHA = 'a'.repeat(40); +const HEAD_SHA = 'b'.repeat(40); +const TREE_SHA = 'c'.repeat(40); +const OBSERVED = Object.freeze({ + base_sha: BASE_SHA, + head_sha: HEAD_SHA, + tree_sha: TREE_SHA, + branch: 'main', + clean: true, + remote_present: true, + remote_count: 1, +}); + +function observeGit(_repo, requestedBaseSha) { + return Promise.resolve({ + ...OBSERVED, + base_sha: requestedBaseSha ?? OBSERVED.head_sha, + }); +} + +function request(overrides = {}) { + return { + run_id: 'vale-hardening', + repo: '/tmp/fixture-repo', + objective: 'Implement and review the hardening plan.', + assignments: [{ + assignment_id: 'social-implementation', + provider: 'grok', + role: 'implement', + access: 'write', + prompt: 'Implement the social ingestion slice.', + expected_duration_ms: 900_000, + }], + ...overrides, + }; +} + +test('compiles a small request into server-owned identities and bounded lane data', async () => { + const compiled = await compileRunRequestV1(request(), { observeGit }); + + assert.equal(compiled.schema, 'codex-co-engineer.run-request.v1'); + assert.equal(compiled.git.base_sha, HEAD_SHA); + assert.equal(compiled.git.clean, true); + assert.equal(compiled.manifest.repository.base_sha, HEAD_SHA); + assert.equal(compiled.assignments[0].model, 'grok-4'); + assert.deepEqual(compiled.assignments[0].write_scope, ['**']); + assert.deepEqual(compiled.assignments[0].capabilities, [...RUN_REQUEST_DEFAULT_CAPABILITIES]); + assert.match(compiled.assignments[0].task_id, /^ce-/u); + assert.match(compiled.request_idempotency_key, /^sha256:[0-9a-f]{64}$/u); + assert.match(compiled.run_identity.digest, /^sha256:[0-9a-f]{64}$/u); + assert.match(compiled.assignments[0].dispatch_identity.digest, /^sha256:[0-9a-f]{64}$/u); + assert.match(compiled.assignments[0].provider_run_identity.digest, /^sha256:[0-9a-f]{64}$/u); + assert.equal(compiled.dispatch_identities[0].assignment_id, 'social-implementation'); + assert.equal(compiled.provider_run_identities[0].provider, 'grok'); + assert.match(compiled.manifest_digest, /^[0-9a-f]{64}$/u); + assert.equal(compiled.assignments[0].prompt, 'Implement the social ingestion slice.'); + assert.equal(Object.isFrozen(compiled), true); + assert.equal(Object.isFrozen(compiled.assignments[0]), true); + assert.equal(compiled.public_summary.assignments[0].prompt, undefined); +}); + +test('derives access from role when omitted and still rejects an explicit mismatch', async () => { + const withoutAccess = request({ + assignments: [{ + assignment_id: 'auth-review', + provider: 'grok', + role: 'review', + prompt: 'Review the current auth changes.', + expected_duration_ms: 300_000, + }], + }); + const compiled = await compileRunRequestV1(withoutAccess, { observeGit }); + assert.equal(compiled.assignments[0].access, 'read_only'); + assert.deepEqual(compiled.assignments[0].write_scope, []); + assert.equal(compiled.manifest.assignments[0].access, 'read_only'); + + await assert.rejects( + compileRunRequestV1({ + ...withoutAccess, + assignments: [{ ...withoutAccess.assignments[0], access: 'writer' }], + }, { observeGit }), + (error) => error.code === 'role_access_mismatch', + ); +}); + +test('semantic request identities are stable across object key order', async () => { + const first = await compileRunRequestV1(request(), { observeGit }); + const second = await compileRunRequestV1({ + assignments: [{ + expected_duration_ms: 900_000, + prompt: 'Implement the social ingestion slice.', + access: 'write', + role: 'implement', + provider: 'grok', + assignment_id: 'social-implementation', + }], + objective: 'Implement and review the hardening plan.', + repo: '/tmp/fixture-repo', + run_id: 'vale-hardening', + }, { observeGit }); + + assert.equal(second.request_idempotency_key, first.request_idempotency_key); + assert.equal(second.manifest_digest, first.manifest_digest); + assert.equal(second.run_identity.digest, first.run_identity.digest); + assert.equal(second.assignments[0].task_id, first.assignments[0].task_id); +}); + +test('omitted duration is identical to the explicit semantic default and invalid supplied values fail', async () => { + const assignment = { ...request().assignments[0] }; + delete assignment.expected_duration_ms; + const omitted = await compileRunRequestV1(request({ assignments: [assignment] }), { observeGit }); + const explicit = await compileRunRequestV1(request({ + assignments: [{ ...assignment, expected_duration_ms: RUN_REQUEST_DEFAULT_EXPECTED_DURATION_MS }], + }), { observeGit }); + + assert.equal(omitted.assignments[0].expected_duration_ms, 600_000); + assert.equal(omitted.request_idempotency_key, explicit.request_idempotency_key); + assert.equal(omitted.manifest_digest, explicit.manifest_digest); + assert.equal(omitted.assignments[0].task_id, explicit.assignments[0].task_id); + + await assert.rejects( + compileRunRequestV1(request({ + assignments: [{ ...assignment, expected_duration_ms: 0 }], + }), { observeGit }), + (error) => error.code === 'out_of_range', + ); + await assert.rejects( + compileRunRequestV1(request({ + assignments: [{ ...assignment, expected_duration_ms: null }], + }), { observeGit }), + (error) => error.code === 'invalid_type', + ); +}); + +test('semantic changes alter the derived request identity', async () => { + const baseline = await compileRunRequestV1(request(), { observeGit }); + const changedObjective = await compileRunRequestV1(request({ + objective: 'Implement a different hardening plan.', + }), { observeGit }); + const changedProvider = await compileRunRequestV1(request({ + assignments: [{ + ...request().assignments[0], + provider: 'cursor-local', + }], + }), { observeGit }); + const changedScope = await compileRunRequestV1(request({ + assignments: [{ + ...request().assignments[0], + write_scope: ['src/**'], + }], + }), { observeGit }); + const changedBase = await compileRunRequestV1(request({ base_sha: BASE_SHA }), { observeGit }); + + assert.notEqual(changedObjective.request_idempotency_key, baseline.request_idempotency_key); + assert.notEqual(changedProvider.request_idempotency_key, baseline.request_idempotency_key); + assert.notEqual(changedScope.request_idempotency_key, baseline.request_idempotency_key); + assert.notEqual(changedBase.request_idempotency_key, baseline.request_idempotency_key); +}); + +test('derivation rejects caller-supplied provenance and dirty source state', async () => { + await assert.rejects( + compileRunRequestV1(request({ request_idempotency_key: 'sha256:' + '0'.repeat(64) }), { observeGit }), + (error) => error.code === 'derived_field_denied', + ); + await assert.rejects( + compileRunRequestV1(request(), { + observeGit: async () => ({ ...OBSERVED, clean: false }), + }), + (error) => error.code === 'repository_dirty', + ); +}); + +test('cloud lanes receive an exact server-derived starting ref', async () => { + const compiled = await compileRunRequestV1(request({ + assignments: [{ + assignment_id: 'cloud-review', + provider: 'cursor-cloud', + role: 'review', + access: 'read', + prompt: 'Review the hardening changes.', + expected_duration_ms: 300_000, + }], + }), { observeGit }); + + assert.equal(compiled.assignments[0].starting_ref, HEAD_SHA); + assert.equal(compiled.manifest.assignments[0].starting_ref, HEAD_SHA); + assert.deepEqual(compiled.manifest.assignments[0].write_scope, []); +}); + +test('multiple writers accept disjoint static scope prefixes and reject overlapping ones', async () => { + const writer = (assignmentId, writeScope) => ({ + assignment_id: assignmentId, + provider: 'grok', + role: 'implement', + access: 'write', + prompt: `Implement ${assignmentId}.`, + expected_duration_ms: 60_000, + write_scope: [writeScope], + }); + const disjoint = await compileRunRequestV1(request({ + run_id: 'disjoint-writers', + assignments: [writer('api-writer', 'src/api/**'), writer('ui-writer', 'src/ui/**')], + }), { observeGit }); + assert.deepEqual(disjoint.assignments.map((entry) => entry.write_scope), [ + ['src/api/**'], + ['src/ui/**'], + ]); + + await assert.rejects( + compileRunRequestV1(request({ + run_id: 'overlapping-writers', + assignments: [writer('src-writer', 'src/**'), writer('api-writer', 'src/api/**')], + }), { observeGit }), + (error) => error.code === 'overlapping_writer_scope', + ); +}); diff --git a/plugins/codex-co-engineer/test/r1-run-runtime-adversarial.test.mjs b/plugins/codex-co-engineer/test/r1-run-runtime-adversarial.test.mjs index 15c2b78..0c58e19 100644 --- a/plugins/codex-co-engineer/test/r1-run-runtime-adversarial.test.mjs +++ b/plugins/codex-co-engineer/test/r1-run-runtime-adversarial.test.mjs @@ -3,7 +3,6 @@ // receipts, and denied remote mutation. import assert from 'node:assert/strict'; -import { spawnSync } from 'node:child_process'; import test from 'node:test'; import { types as utilTypes } from 'node:util'; @@ -205,16 +204,37 @@ test('injected objects missing required methods fail at factory time', () => { }), (error) => error.code === 'injected_dependency_invalid'); }); -test('remote git mutation stays denied after a successful submit', async () => { +test('remote mutation requests stay denied after submit without redispatch or state drift', async () => { const harness = createRuntime(); - const receipt = await harness.runtime.submitRun(makeSubmitRequest()); + const request = makeSubmitRequest(); + const receipt = await harness.runtime.submitRun(request); assert.equal(receipt.remote_mutated, false); assert.equal(receipt.side_effects.remote_mutated, false); - const gitPush = spawnSync('git', ['push', '--dry-run'], { - encoding: 'utf8', - timeout: 5000, - }); - assert.notEqual(gitPush.status, 0); + const before = structuredClone(await harness.runStore.getByRunId(request.run_id)); + const dispatchBefore = { + submit: harness.scheduler.calls.submit, + delegate: [...harness.scheduler.calls.delegate], + resume: harness.scheduler.calls.resume, + }; + + const pushError = await errorOf(() => harness.runtime.inspectRun({ + run_id: request.run_id, + push: true, + })); + assert.equal(pushError.code, 'merge_authority_denied'); + assertContentFree(pushError); + + const remoteError = await errorOf(() => harness.runtime.resumeRun({ + run_id: request.run_id, + remote: { operation: 'push', repository: 'forged' }, + })); + assert.equal(remoteError.code, 'remote_mutation_denied'); + assertContentFree(remoteError); + + assert.deepEqual(await harness.runStore.getByRunId(request.run_id), before); + assert.equal(harness.scheduler.calls.submit, dispatchBefore.submit); + assert.deepEqual(harness.scheduler.calls.delegate, dispatchBefore.delegate); + assert.equal(harness.scheduler.calls.resume, dispatchBefore.resume); }); test('forged inspect receipts cannot broaden path or candidate authority', async () => { diff --git a/plugins/codex-co-engineer/test/r1-run-scheduler-adversarial.test.mjs b/plugins/codex-co-engineer/test/r1-run-scheduler-adversarial.test.mjs index df410a9..5bddd07 100644 --- a/plugins/codex-co-engineer/test/r1-run-scheduler-adversarial.test.mjs +++ b/plugins/codex-co-engineer/test/r1-run-scheduler-adversarial.test.mjs @@ -290,6 +290,30 @@ test('oversized attention is dropped and malformed attention is unresolved', asy assert.doesNotMatch(serialized, /sk-live/u); }); +test('structured ACP attention options retain their provider reply identities', async () => { + const options = [ + { optionId: 'allow-once', name: 'Allow once', kind: 'allow_once' }, + { optionId: 'reject-once', name: 'Reject', kind: 'reject_once' }, + ]; + const harness = createScopedStubs({ + attentionByTask: { + [TASK_A]: { + session_id: 'sess-a', + question_id: 'permission-a', + prompt: 'Allow this operation?', + options, + }, + }, + inspectStatusByTask: { [TASK_A]: 'needs_attention' }, + }); + await harness.scheduler.submitAssignments(twoWriterRequest()); + const receipt = await harness.scheduler.resumeAssignments({ run_id: RUN_ID }); + const lane = receipt.lanes.find((entry) => entry.assignment_id === ASSIGNMENT_A); + assert.equal(lane.status, 'needs_attention'); + assert.deepEqual(lane.attention.options, options); + assert.equal(lane.attention.options[0].optionId, 'allow-once'); +}); + test('resume and cancel never accept a second run identity or replay flag', async () => { const harness = createScopedStubs(); await harness.scheduler.submitAssignments(twoWriterRequest()); diff --git a/plugins/codex-co-engineer/test/r1-run-tool-adapter.test.mjs b/plugins/codex-co-engineer/test/r1-run-tool-adapter.test.mjs index 97b7fdf..ea9eb2a 100644 --- a/plugins/codex-co-engineer/test/r1-run-tool-adapter.test.mjs +++ b/plugins/codex-co-engineer/test/r1-run-tool-adapter.test.mjs @@ -25,6 +25,8 @@ import { PUBLIC_MCP_CATALOG, RUN_TOOL_ADAPTER_ALWAYS_FALSE_SIDE_EFFECTS, RUN_TOOL_ADAPTER_SCHEMA_ID, + SIMPLE_RUN_RECEIPT_STRUCTURED_BYTES_MAX, + SIMPLE_RUN_STATUS_STRUCTURED_BYTES_MAX, RUN_TOOL_OPERATIONS, classifyDeniedGitOperationV1, classifyRunToolCall, @@ -32,6 +34,7 @@ import { createRunToolAdapter, denyRunToolRemoteMutationV1, describeRunToolAdapterV1, + experienceForRunToolResult, } from '../mcp/v3/run-tool-adapter.mjs'; import { ASSIGNMENT_ID, @@ -132,6 +135,75 @@ test('submit maps delegate.run onto one 1-8 lane runtime submission', async () = assert.match(receipt.candidate.ref, new RegExp(`^${CANDIDATE_REF_NAMESPACE}`)); }); +test('simple run status and receipts remain within their structured byte caps', async () => { + const legacy = createAdapter(); + const handoff = { + schema: 'codex-co-engineer.partial-handoff.v1', + worktree: '/tmp/worktree', + branch: 'codex/very-long-branch', + starting_sha: 'a'.repeat(40), + current_head: 'b'.repeat(40), + clean: false, + changed_files: Array.from({ length: 64 }, (_, index) => `${'src/'.padEnd(1000, 'x')}${index}`), + commits: Array.from({ length: 64 }, () => 'c'.repeat(40)), + no_commit: false, + partial_diff: true, + last_acknowledged_provider_event: 'file_changed', + recovery_classification: 'timed_out_with_partial_work', + safe_next_actions: Array.from({ length: 8 }, () => 'Review the retained worktree and handoff evidence.'.repeat(100)), + }; + const simpleReceipt = { + schema: 'codex-co-engineer.run-admission.v1', + run_id: 'bounded-receipt', + phase: 'degraded', + status: 'degraded', + assignment_count: 8, + lanes: Array.from({ length: 8 }, (_, index) => ({ + assignment_id: `lane-${index}`, + task_id: `task-${index}`, + provider: 'grok', + model: 'grok-4', + role: 'implement', + access: 'writer', + required: true, + phase: 'partial_handoff', + prompt_dispatched: true, + dispatch_confidence: 'authoritative', + handoff, + })), + attention: null, + telemetry: { noisy: 'z'.repeat(50_000) }, + }; + const simpleRuntime = { + submitRunRequest: async () => simpleReceipt, + inspectRun: async () => simpleReceipt, + resumeRun: async () => simpleReceipt, + replyRun: async () => simpleReceipt, + cancelRun: async () => simpleReceipt, + waitRun: async () => simpleReceipt, + }; + const adapter = createRunToolAdapter({ runtime: legacy.runtime, simpleRuntime }); + const request = { + run_request: { + run_id: 'bounded-receipt', + repo: '/tmp/repo', + objective: 'Bound the receipt.', + assignments: [{ + assignment_id: 'lane-one', + provider: 'grok', + role: 'implement', + access: 'write', + prompt: 'Implement the bounded slice.', + expected_duration_ms: 60_000, + }], + }, + }; + const submitted = await adapter.dispatch('delegate', request); + assert.ok(Buffer.byteLength(JSON.stringify(submitted), 'utf8') <= SIMPLE_RUN_RECEIPT_STRUCTURED_BYTES_MAX); + const status = await adapter.dispatch('status', { run_id: 'bounded-receipt' }); + assert.ok(Buffer.byteLength(JSON.stringify(status), 'utf8') <= SIMPLE_RUN_STATUS_STRUCTURED_BYTES_MAX); +}); + test('eight-lane submit aggregates and a required unresolved lane blocks the candidate', async () => { const assignments = [ makeAssignment({ assignmentId: 'w1', taskId: 'task-w1', writeScope: ['a/**'] }), @@ -296,7 +368,7 @@ test('unsupported same-session providers cancel only the affected lane', async ( assignmentId: 'dsh-lane', taskId: 'task-dsh', provider: 'dsh', - model: 'muse-spark-1.2-contributor', + model: 'meta/muse-spark-1.3-contributor', writeScope: ['docs/**'], }); const writer = makeAssignment(); @@ -382,7 +454,7 @@ test('explicit provider/model plus a conflicting named profile fails closed', as const definition = { schema: PROFILE_SCHEMA, provider: 'dsh', - model: 'muse-spark-1.2-contributor', + model: 'meta/muse-spark-1.3-contributor', }; const name = 'writer-profile'; const catalog = { @@ -591,7 +663,7 @@ test('named profile snapshot is bound once and survives catalog mutation after s [name]: { schema: PROFILE_SCHEMA, provider: 'dsh', - model: 'muse-spark-1.2-contributor', + model: 'meta/muse-spark-1.3-contributor', }, })); const inspected = await adapter.dispatch('status', { run_id: RUN_ID }); @@ -626,3 +698,365 @@ test('durable restart fallback keeps unconfirmed cancel unresolved/unsafe', asyn await rm(first.root, { recursive: true, force: true }); } }); + + +test('simple receipts preserve cursor/revision and wait metadata through the bounded adapter', async () => { + const runId = 'adapter-continuity'; + const makeReceipt = ({ + status = 'running', + phase = status, + revision = 11, + cursor = '11', + wait_until, + waited_ms, + } = {}) => ({ + schema: 'codex-co-engineer.run-admission.v1', + version: 1, + run_id: runId, + phase, + status, + revision, + cursor, + assignment_count: 1, + lanes: [{ + assignment_id: 'adapter-lane', + task_id: 'adapter-task', + provider: 'grok', + model: 'grok-4', + role: 'implement', + access: 'write', + required: true, + phase, + status, + prompt_dispatched: true, + }], + complete_candidate_blocked: false, + attention: null, + consent: null, + admission: null, + dispatched_assignment_ids: ['adapter-lane'], + undispatched_assignment_ids: [], + dispatch_uncertain_assignment_ids: [], + authoritative_required_dispatch: true, + ...(wait_until === undefined ? {} : { wait_until }), + ...(waited_ms === undefined ? {} : { waited_ms }), + }); + const replyCalls = []; + const legacy = createAdapter(); + const simpleRuntime = { + hasRun: (value) => value === runId, + submitRunRequest: async () => makeReceipt(), + inspectRun: async () => makeReceipt(), + resumeRun: async () => makeReceipt(), + replyRun: async (value, options) => { + replyCalls.push({ value, options }); + return makeReceipt(); + }, + cancelRun: async () => makeReceipt({ status: 'cancelled', phase: 'cancelled' }), + waitRun: async (value) => makeReceipt({ + wait_until: value.wait_until, + waited_ms: 17, + }), + }; + const adapter = createRunToolAdapter({ runtime: legacy.runtime, simpleRuntime }); + + const status = await adapter.dispatch('status', { run_id: runId }); + assert.equal(status.revision, 11); + assert.equal(status.cursor, '11'); + assert.equal(Object.hasOwn(status, 'blockers'), false); + + const waited = await adapter.dispatch('task', { + run_id: runId, + cursor: '10', + wait_until: 'terminal', + wait_ms: 0, + }); + assert.equal(waited.revision, 11); + assert.equal(waited.cursor, '11'); + assert.equal(waited.wait_until, 'terminal'); + assert.equal(waited.waited_ms, 17); + + const controller = new AbortController(); + await adapter.dispatch('task', { + run_id: runId, + run_reply: { request_consent: true }, + }, { signal: controller.signal }); + assert.deepEqual(replyCalls[0].value, { + run_id: runId, + request_consent: true, + }); + assert.equal(replyCalls[0].options.signal, controller.signal); + + const invalid = await errorOf(() => adapter.dispatch('task', { + run_id: runId, + run_reply: { request_consent: false }, + })); + assert.equal(invalid.code, 'invalid_format'); + + const mixed = await errorOf(() => adapter.dispatch('task', { + run_id: runId, + run_reply: { request_consent: true, approval_ref: 'ambiguous' }, + })); + assert.equal(mixed.code, 'mixed_run_operation'); +}); + +test('undefined simple runtime receipts fail closed as blocked unresolved output', async () => { + const runId = 'malformed-runtime-receipt'; + const legacy = createAdapter(); + const simpleRuntime = { + hasRun: (value) => value === runId, + submitRunRequest: async () => undefined, + inspectRun: async () => undefined, + resumeRun: async () => undefined, + replyRun: async () => undefined, + cancelRun: async () => undefined, + waitRun: async () => undefined, + }; + const adapter = createRunToolAdapter({ runtime: legacy.runtime, simpleRuntime }); + const submitted = await adapter.dispatch('delegate', { + run_request: { + run_id: runId, + repo: '/tmp/repo', + objective: 'Defend malformed receipts.', + assignments: [{ + assignment_id: 'malformed-lane', + provider: 'grok', + role: 'implement', + access: 'read', + prompt: 'Synthetic only.', + }], + }, + }); + const inspected = await adapter.dispatch('status', { run_id: runId }); + + for (const receipt of [submitted, inspected]) { + assert.equal(receipt.run_id, runId); + assert.equal(receipt.status, 'unresolved'); + assert.equal(receipt.phase, 'unresolved'); + assert.equal(receipt.blockers.verification, true); + assert.equal(receipt.error.code, 'durable_state_mismatch'); + assert.deepEqual(receipt.lanes, []); + assert.deepEqual(receipt.diagnostics, { view: 'diagnostics' }); + } +}); + +test('large multi-assignment results retain bounded useful previews without inventing acceptance', async () => { + const runId = 'bounded-answer-results'; + const lanes = Array.from({ length: 8 }, (_, index) => ({ + assignment_id: `lane-${index}`, task_id: `task-${index}`, provider: 'grok', + phase: 'completed', status: 'completed', required: true, prompt_dispatched: true, + dispatch_confidence: 'authoritative', result: `Answer ${index}: ` + '🙂'.repeat(5000), + })); + const receipt = { schema: 'codex-co-engineer.run-admission.v1', run_id: runId, + phase: 'completed', status: 'completed', lanes, assignment_count: 8, cursor: '9', revision: 9 }; + const simpleRuntime = { hasRun: () => true }; + for (const name of ['submitRunRequest', 'inspectRun', 'resumeRun', 'replyRun', 'cancelRun', 'waitRun']) { + simpleRuntime[name] = async () => receipt; + } + const { runtime } = createAdapter(); + const adapter = createRunToolAdapter({ runtime, simpleRuntime }); + for (const [tool, cap] of [['status', SIMPLE_RUN_STATUS_STRUCTURED_BYTES_MAX], ['task', SIMPLE_RUN_RECEIPT_STRUCTURED_BYTES_MAX]]) { + const result = await adapter.dispatch(tool, { run_id: runId, wait_ms: 0 }); + assert.ok(Buffer.byteLength(JSON.stringify(result)) <= cap); + assert.equal(result.lanes.length, 8); + for (let index = 0; index < 8; index += 1) { + assert.ok(JSON.stringify(result.lanes[index].result).includes(`Answer ${index}`)); + assert.equal(result.lanes[index].result_truncated, true); + } + for (const omitted of ['candidate', 'checks', 'telemetry', 'experience', 'side_effects', 'admission']) { + assert.equal(Object.hasOwn(result, omitted), false, `compact receipt included ${omitted}`); + } + assert.equal(Object.hasOwn(result, 'blockers'), false); + assert.deepEqual(result.diagnostics, { view: 'diagnostics' }); + } +}); + +test('adversarial semantic metadata stays within caps and marks retrievable omissions', async () => { + const runId = 'bounded-adversarial-metadata'; + const huge = '🙂'.repeat(30_000); + const lanes = Array.from({ length: 8 }, (_, index) => ({ + assignment_id: `lane-${index}`, + task_id: `task-${index}`, + provider: 'grok', + role: index === 0 ? 'review' : 'implement', + phase: 'needs_attention', + status: 'needs_attention', + required: true, + prompt_dispatched: true, + dispatch_confidence: 'authoritative', + result: `result-${index}-${huge}`, + error: { code: 'provider_error', message: huge, detail: huge }, + handoff: { worktree: `/workspace/${huge}`, branch: `codex/${huge}` }, + })); + const items = lanes.map((lane, index) => ({ + assignment_id: lane.assignment_id, + task_id: lane.task_id, + question_id: `question-${index}`, + session_id: `session-${index}`, + question: `Choose for lane ${index}: ${huge}`, + options: Array.from({ length: 8 }, (_, option) => `choice-${index}-${option}-${huge}`), + event_cursor: String(index + 1), + })); + const receipt = { + schema: 'codex-co-engineer.run-admission.v1', + version: 1, + run_id: runId, + phase: 'needs_attention', + status: 'needs_attention', + cursor: '9', + revision: 9, + assignment_count: 8, + lanes, + attention: { status: 'open', batch_id: 'batch-adversarial', revision: 9, items }, + cleanup: { cleaned: false, proof_bound: false, unresolved: [huge], remaining: 8 }, + error: { code: 'run_error', message: huge, detail: huge }, + }; + const simpleRuntime = { hasRun: () => true }; + for (const name of ['submitRunRequest', 'inspectRun', 'resumeRun', 'replyRun', 'cancelRun', 'waitRun']) { + simpleRuntime[name] = async () => receipt; + } + const { runtime } = createAdapter(); + const adapter = createRunToolAdapter({ runtime, simpleRuntime }); + for (const [tool, cap] of [ + ['status', SIMPLE_RUN_STATUS_STRUCTURED_BYTES_MAX], + ['task', SIMPLE_RUN_RECEIPT_STRUCTURED_BYTES_MAX], + ]) { + const result = await adapter.dispatch(tool, { run_id: runId }); + assert.ok(Buffer.byteLength(JSON.stringify(result), 'utf8') <= cap, tool); + assert.equal(result.diagnostics.view, 'diagnostics'); + assert.equal(result.diagnostics.details_omitted, true); + assert.equal(result.diagnostics.reason, 'response_size_limit'); + assert.equal(result.attention.details_omitted, true); + assert.equal(result.attention.reply_blocked, true); + assert.equal(Object.hasOwn(result.attention, 'items'), false); + assert.equal(result.lanes.every((lane) => lane.result_omitted === true), true); + assert.equal(result.error.code, 'run_error'); + assert.equal(result.blockers.cleanup, true); + assert.equal(Object.hasOwn(result, 'cleanup'), false); + assert.match(result.diagnostics.instruction, /view="diagnostics"/u); + } + + const topReceipt = { + ...receipt, + phase: 'completed', + status: 'completed', + lanes: lanes.map(({ result: _result, error: _error, ...lane }) => ({ + ...lane, phase: 'completed', status: 'completed', + })), + attention: null, + cleanup: { cleaned: true, proof_bound: true, unresolved: [], remaining: 0 }, + error: null, + result: huge, + }; + const topRuntime = { hasRun: () => true }; + for (const name of ['submitRunRequest', 'inspectRun', 'resumeRun', 'replyRun', 'cancelRun', 'waitRun']) { + topRuntime[name] = async () => topReceipt; + } + const top = await createRunToolAdapter({ runtime, simpleRuntime: topRuntime }) + .dispatch('status', { run_id: runId }); + assert.ok(Buffer.byteLength(JSON.stringify(top), 'utf8') <= SIMPLE_RUN_STATUS_STRUCTURED_BYTES_MAX); + assert.equal(top.result_truncated, true); + + const overflowReceipt = { ...receipt, status: huge, phase: huge }; + const overflowRuntime = { hasRun: () => true }; + for (const name of ['submitRunRequest', 'inspectRun', 'resumeRun', 'replyRun', 'cancelRun', 'waitRun']) { + overflowRuntime[name] = async () => overflowReceipt; + } + const overflow = await createRunToolAdapter({ runtime, simpleRuntime: overflowRuntime }) + .dispatch('status', { run_id: runId }); + assert.ok(Buffer.byteLength(JSON.stringify(overflow), 'utf8') <= SIMPLE_RUN_STATUS_STRUCTURED_BYTES_MAX); + assert.equal(overflow.status, 'unresolved'); + assert.equal(overflow.phase, 'unresolved'); + assert.equal(overflow.error.code, 'response_projection_overflow'); +}); + +test('simple run defaults to compact semantics and exposes detailed diagnostics explicitly', async () => { + const runId = 'compact-with-diagnostics'; + const runtimeReceipt = { + schema: 'codex-co-engineer.run-admission.v1', version: 1, run_id: runId, + phase: 'running', status: 'running', cursor: '7', revision: 7, + assignment_count: 1, authoritative_required_dispatch: true, + lanes: [{ + assignment_id: 'worker', task_id: 'worker-task', provider: 'grok', + status: 'running', phase: 'running', required: true, + prompt_dispatched: true, dispatch_confidence: 'authoritative', + }], + telemetry: { admission_duration_ms: 123 }, + checks: { catalog_five_tools: true }, + }; + const simpleRuntime = { hasRun: () => true }; + for (const name of ['submitRunRequest', 'inspectRun', 'resumeRun', 'replyRun', 'cancelRun', 'waitRun']) { + simpleRuntime[name] = async () => runtimeReceipt; + } + const { runtime } = createAdapter(); + const adapter = createRunToolAdapter({ runtime, simpleRuntime }); + + const compact = await adapter.dispatch('task', { run_id: runId }); + assert.deepEqual(Object.keys(compact), [ + 'schema', 'version', 'mode', 'tool', 'operation', 'run_id', 'status', 'phase', + 'cursor', 'revision', 'assignment_count', 'authoritative_required_dispatch', + 'lanes', 'diagnostics', + ]); + assert.equal(compact.lanes[0].prompt_dispatched, true); + assert.equal(compact.diagnostics.view, 'diagnostics'); + + const detailed = await adapter.dispatch('task', { run_id: runId, view: 'diagnostics' }); + assert.equal(detailed.telemetry.admission_duration_ms, 123); + assert.equal(detailed.checks.catalog_five_tools, true); + assert.equal(detailed.experience.schema, 'codex-co-engineer.experience-projection.v1'); +}); + +test('compact semantic finals retain actual candidate, verification, and top-level-only result', async () => { + const runId = 'compact-review-artifact'; + const runtimeReceipt = { + schema: 'codex-co-engineer.run-admission.v1', version: 1, run_id: runId, + phase: 'completed', status: 'completed', cursor: '8', revision: 8, + assignment_count: 1, authoritative_required_dispatch: true, + complete_candidate_blocked: false, + lanes: [{ + assignment_id: 'worker', task_id: 'worker-task', provider: 'grok', + role: 'review', status: 'completed', phase: 'completed', required: true, + prompt_dispatched: true, dispatch_confidence: 'authoritative', task_final: true, + handoff: { + branch: 'codex/compact-review-artifact', + current_head: 'b'.repeat(40), + tree_sha: 'c'.repeat(40), + }, + }], + result: { summary: 'Integrated candidate is ready.' }, + candidate: { + ref: expectedCandidateRefV1({ run_id: runId }), + head: 'b'.repeat(40), tree: 'c'.repeat(40), + ready_for_codex_review: true, accepted: true, authority: 'p35', + }, + verification: { status: 'passed', authority: 'p35', tests: ['node --test'] }, + evidence: { + facts: [{ fact_kind: 'git_identity' }], + claims: [{ claim_kind: 'tests_passed' }], + digest: `sha256:${'d'.repeat(64)}`, + }, + }; + const simpleRuntime = { hasRun: () => true }; + for (const name of ['submitRunRequest', 'inspectRun', 'resumeRun', 'replyRun', 'cancelRun', 'waitRun']) { + simpleRuntime[name] = async () => runtimeReceipt; + } + const { runtime } = createAdapter(); + const compact = await createRunToolAdapter({ runtime, simpleRuntime }) + .dispatch('task', { run_id: runId }); + + assert.deepEqual(compact.result, { summary: 'Integrated candidate is ready.' }); + assert.equal(compact.candidate.ref, expectedCandidateRefV1({ run_id: runId })); + assert.equal(compact.candidate.head, 'b'.repeat(40)); + assert.equal(compact.candidate.tree, 'c'.repeat(40)); + assert.equal(compact.candidate.ready_for_codex_review, true); + assert.deepEqual(compact.verification, { + status: 'passed', authority: 'p35', tests: ['node --test'], + }); + const uiExperience = experienceForRunToolResult(compact); + assert.equal(uiExperience.final.reviews.present, true); + assert.deepEqual(uiExperience.final.reviews.lanes, ['worker']); + assert.equal(uiExperience.final.git.head, 'b'.repeat(40)); + assert.deepEqual(uiExperience.final.evidence.kinds, ['git_identity', 'tests_passed']); + assert.equal(Object.hasOwn(compact, 'experience'), false); + assert.equal(Object.hasOwn(compact, 'blockers'), false); +}); diff --git a/plugins/codex-co-engineer/test/r1-skill-activation.test.mjs b/plugins/codex-co-engineer/test/r1-skill-activation.test.mjs index 6301c86..6543e4b 100644 --- a/plugins/codex-co-engineer/test/r1-skill-activation.test.mjs +++ b/plugins/codex-co-engineer/test/r1-skill-activation.test.mjs @@ -312,18 +312,7 @@ function assertGoalEntrypoint(skillId, files, catalog) { assert.match(packText, /same run cursor/u, `${skillId} missing same run cursor`); assert.match(files.skillMd, /Codex remains/u, `${skillId} missing Codex authority`); assert.match(files.skillMd, /External workers may commit/u, `${skillId} missing worker commit authority`); - assert.match( - files.skillMd, - /scoped publisher may non-force push only the task branch/u, - `${skillId} missing scoped publisher`, - ); - assert.match(files.skillMd, /Sol High or Sol XHigh/u, `${skillId} missing Sol merge actor`); - assert.match(files.skillMd, /The user retains/u, `${skillId} missing retained user authority`); - assert.doesNotMatch( - files.skillMd, - /reviewer and merge authority/u, - `${skillId} still names Codex as merge authority`, - ); + assert.match(files.skillMd, /Publication and merge require user authorization and Codex review/u); assert.match(files.skillMd, /Never ask the user to construct tool payloads/u, skillId); assert.match(files.skillMd, /\$control-codex-co-engineer-agents/u, skillId); } @@ -596,7 +585,7 @@ test('activation contract rejects jargon entrypoints, sixth tools, and misrouted ); }); -test('Delegating and Chatting teach Luna Max PM without a sixth public skill', async () => { +test('Delegating and Chatting retain optional Luna relay without a sixth public skill', async () => { const bundle = await loadBundle(); const skillDirs = (await readdir(SKILLS)).sort(); assert.deepEqual(skillDirs, [...ALL_SKILLS].sort()); @@ -611,8 +600,10 @@ test('Delegating and Chatting teach Luna Max PM without a sixth public skill', a assert.equal(delegatePack.includes(phrase), true, `delegate pack missing ${phrase}`); assert.equal(chatPack.includes(phrase), true, `chat pack missing ${phrase}`); } - assert.match(bundle.skillFiles['delegate-to-co-engineer'].skillMd, /pin one Luna Max task/u); - assert.match(bundle.skillFiles['chat-with-co-engineer'].skillMd, /Normal completion does not wake Sol/u); + assert.match(bundle.skillFiles['delegate-to-co-engineer'].skillMd, /explicitly requested legacy Luna\/Sol relay/u); + assert.match(delegatePack, /pin one Luna Max task/u); + assert.match(bundle.skillFiles['chat-with-co-engineer'].skillMd, /explicitly requested legacy Luna\/Sol relay/u); + assert.match(chatPack, /Normal completion does not wake\s+Sol/u); assert.match(delegatePack, /I am not substituting Sol/u); assert.match(chatPack, /I am not substituting Sol/u); diff --git a/plugins/codex-co-engineer/test/r1-supervisor-simple-run.test.mjs b/plugins/codex-co-engineer/test/r1-supervisor-simple-run.test.mjs new file mode 100644 index 0000000..e37c0fe --- /dev/null +++ b/plugins/codex-co-engineer/test/r1-supervisor-simple-run.test.mjs @@ -0,0 +1,355 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import { mkdtemp, rm } from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; + +import { createSupervisorRunToolAdapter, submitTask } from '../mcp/v3/supervisor.mjs'; +import { createTask, updateTask } from '../mcp/v3/task-store.mjs'; +import { parseChildEnvelopeV1 } from '../mcp/v3/prompt-compiler.mjs'; +import { + compileRunRequestV1, + RUN_REQUEST_DEFAULT_CAPABILITIES, +} from '../mcp/v3/run-request-compiler.mjs'; + +const BASE_SHA = 'a'.repeat(40); +const OBSERVED = Object.freeze({ + base_sha: BASE_SHA, + head_sha: BASE_SHA, + tree_sha: 'b'.repeat(40), + branch: 'main', + clean: true, + remote_present: true, + remote_count: 1, +}); + +function request() { + return { + run_id: 'simple-supervisor', + repo: '/tmp/fixture-repo', + objective: 'Exercise the supervisor simple-run adapter.', + assignments: [{ + assignment_id: 'implementation', + provider: 'grok', + role: 'implement', + access: 'write', + prompt: 'Implement the bounded slice.', + expected_duration_ms: 60_000, + }], + }; +} + + +test('default simple dispatch sends the compiled envelope with native workspace guidance and pins its workspace base', async () => { + const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-simple-dispatch-contract-')); + const calls = []; + try { + const adapter = await createSupervisorRunToolAdapter({ + root, + inProcess: true, + execute: async () => ({ stdout: '' }), + compile: (value) => compileRunRequestV1(value, { observeGit: async () => OBSERVED }), + requestConsent: async () => ({ approved: true }), + providerReady: async () => ({ ready: true }), + processBoundaryReady: async () => ({ ready: true }), + verifyRepository: async () => ({ verified: true }), + prepareWorkspace: async ({ assignment }) => ({ + prepared: true, + workspace: { + task: assignment.task_id, + status: 'ready', + worktree_path: '/tmp/' + assignment.task_id, + branch: 'codex/' + assignment.assignment_id, + start_sha: BASE_SHA, + }, + }), + submitTask: async (input, dependencies) => { + calls.push({ input, dependencies }); + return { task: { id: input.task_id } }; + }, + waitForDispatchEvidence: async (_stateRoot, taskId) => ({ + dispatched: true, + prompt_dispatched: true, + confidence: 'authoritative', + session_ready: true, + session_id: taskId + '-session', + cursor: '0', + }), + }); + + const submitted = await adapter.dispatch('delegate', { run_request: request() }); + assert.equal(submitted.phase, 'running'); + assert.equal(calls.length, 1); + + const compiled = await compileRunRequestV1(request(), { observeGit: async () => OBSERVED }); + const call = calls[0]; + const envelope = parseChildEnvelopeV1(call.input.prompt); + assert.equal(call.input.run_id, 'simple-supervisor'); + assert.equal(call.input.assignment_id, 'implementation'); + assert.equal(call.input.provider, 'grok'); + assert.equal(call.input.model, 'grok-4'); + assert.equal(call.input.access, 'writer'); + assert.deepEqual(call.input.write_scope, ['**']); + assert.deepEqual(call.input.capabilities, [...RUN_REQUEST_DEFAULT_CAPABILITIES]); + assert.equal(call.input.child_envelope_digest, compiled.assignments[0].prompt_envelope_digest); + assert.equal(call.input.prompt, compiled.assignments[0].child_envelope.envelope_text); + assert.match(call.input.prompt, /^repository_path: \/tmp\/fixture-repo$/mu); + assert.match(call.input.prompt, /^provider_workspace: work only in the current working directory \(assigned worktree\); repository_path is source identity, not a navigation target$/mu); + assert.match(call.input.prompt, /^provider_guidance: .*honor an exact requested output and format exactly; required_evidence labels are controller metadata, not worker response sections; omit routine progress narration and repeated identity or report blocks unless the task prompt requests them, while surfacing blockers and necessary questions; do not seek receipt artifacts because the controller owns lifecycle and machine receipts$/mu); + assert.equal(envelope.prompt, 'Implement the bounded slice.'); + assert.equal(envelope.execution.provider, 'grok'); + assert.equal(envelope.execution.model, 'grok-4'); + assert.equal(envelope.role, 'implement'); + assert.equal(envelope.access, 'writer'); + assert.deepEqual(envelope.write_scope, ['**']); + assert.equal(call.dependencies.baseSha, BASE_SHA); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); + +test('unsupported live model overrides fail before task submission', async () => { + await assert.rejects( + submitTask({ + task_id: 'unsupported-model', + provider: 'grok', + model: 'grok/custom', + repo: '/repo', + prompt: 'must not dispatch', + }), + (error) => error.code === 'model_unattested', + ); +}); + +test('run request reports an actionable incomplete-runtime failure before workspace preparation', async () => { + const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-run-runtime-preflight-')); + const calls = []; + try { + const adapter = await createSupervisorRunToolAdapter({ + root, + inProcess: true, + compile: (value) => compileRunRequestV1(value, { observeGit: async () => OBSERVED }), + requestConsent: async () => ({ approved: true }), + preflightRuntime: async (provider) => { + calls.push(['runtime', provider]); + throw Object.assign(new Error('/deleted/cache/acp-worker.mjs?token=secret'), { + code: 'runtime_install_incomplete', + }); + }, + processBoundaryReady: async () => { calls.push(['boundary']); return { ready: true }; }, + verifyRepository: async () => { calls.push(['repository']); return { verified: true }; }, + prepareWorkspace: async () => { calls.push(['workspace']); throw new Error('must not prepare'); }, + createSession: async () => { calls.push(['session']); throw new Error('must not create session'); }, + dispatchPrompt: async () => { calls.push(['prompt']); throw new Error('must not dispatch'); }, + }); + + const receipt = await adapter.dispatch('delegate', { run_request: request() }); + assert.equal(receipt.phase, 'failed'); + assert.equal(receipt.error.code, 'runtime_install_incomplete'); + assert.equal(receipt.error.message, 'The installed Codex-Co-Engineer runtime is incomplete. Reinstall the plugin, then restart Codex.'); + assert.equal(receipt.lanes[0].error.code, 'runtime_install_incomplete'); + assert.notEqual(receipt.lanes[0].prepared, true); + assert.equal(receipt.lanes[0].prompt_dispatched, false); + assert.deepEqual(calls, [['runtime', 'grok']]); + assert.doesNotMatch(JSON.stringify(receipt), /deleted|cache|token|PRIVATE_PROMPT/iu); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); + +test('supervisor wires run_request through admission while preserving bounded receipts', async () => { + const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-simple-run-')); + const calls = []; + try { + const adapterOptions = { + root, + inProcess: true, + compile: (value) => compileRunRequestV1(value, { observeGit: async () => OBSERVED }), + requestConsent: async () => ({ approved: true }), + providerReady: async () => ({ ready: true }), + processBoundaryReady: async () => ({ ready: true }), + verifyRepository: async () => ({ verified: true }), + prepareWorkspace: async ({ assignment }) => ({ + prepared: true, + workspace: { + task: assignment.task_id, + status: 'ready', + worktree_path: `/tmp/${assignment.task_id}`, + branch: `codex/${assignment.assignment_id}`, + start_sha: BASE_SHA, + }, + }), + createSession: async ({ assignment }) => ({ ready: true, session_id: `${assignment.task_id}-session` }), + dispatchPrompt: async ({ assignment }) => { + calls.push(assignment.assignment_id); + return { + dispatched: true, + confidence: 'authoritative', + session_ready: true, + session_id: `${assignment.task_id}-session`, + cursor: '0', + }; + }, + inspectLane: async () => ({ status: 'running', cursor: '0' }), + cancelLane: async () => ({ confirmed: true, cancelled: true }), + inspectWorkspace: async () => ({ current_head: BASE_SHA, clean: true, changed_files: [], commits: [] }), + }; + const adapter = await createSupervisorRunToolAdapter(adapterOptions); + + const submitted = await adapter.dispatch('delegate', { run_request: request() }); + assert.equal(submitted.mode, 'run'); + assert.equal(submitted.phase, 'running'); + assert.equal(submitted.authoritative_required_dispatch, true); + assert.equal(submitted.lanes[0].prompt_dispatched, true); + assert.deepEqual(calls, ['implementation']); + assert.equal(JSON.stringify(submitted).includes('Implement the bounded slice.'), false); + + const status = await adapter.dispatch('task', { run_id: 'simple-supervisor' }); + assert.equal(status.phase, 'running'); + assert.equal(status.lanes[0].prompt_dispatched, true); + assert.equal(Object.hasOwn(status, 'telemetry'), false); + + const restarted = await createSupervisorRunToolAdapter(adapterOptions); + const recovered = await restarted.dispatch('task', { run_id: 'simple-supervisor' }); + assert.equal(recovered.phase, 'running'); + assert.equal(calls.length, 1, 'a restarted adapter must not dispatch the prompt again'); + + const waited = await adapter.dispatch('tasks', { + run_id: 'simple-supervisor', + wait_until: 'decision_or_attention', + wait_ms: 0, + }); + assert.equal(waited.phase, 'running'); + + const cancelled = await adapter.dispatch('cancel', { run_id: 'simple-supervisor' }); + assert.equal(cancelled.phase, 'cancelled'); + assert.equal(cancelled.lanes[0].status, 'cancelled'); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); + +test('production supervisor observation carries a safe provider failure into the native run receipt', async () => { + const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-provider-error-')); + let dispatches = 0; + try { + const adapter = await createSupervisorRunToolAdapter({ + root, + inProcess: true, + compile: (value) => compileRunRequestV1(value, { observeGit: async () => OBSERVED }), + requestConsent: async () => ({ approved: true }), + providerReady: async () => ({ ready: true }), + processBoundaryReady: async () => ({ ready: true }), + verifyRepository: async () => ({ verified: true }), + prepareWorkspace: async ({ assignment }) => ({ + prepared: true, + workspace: { task: assignment.task_id, worktree_path: root, branch: 'codex/provider-error', start_sha: BASE_SHA }, + }), + createSession: async () => ({ ready: true, session_id: 'provider-error-session' }), + dispatchPrompt: async ({ assignment }) => { + dispatches += 1; + await createTask({ + root, + prompt: 'PRIVATE_PROMPT', + record: { + id: assignment.task_id, provider: 'dsh', status: 'failed', cwd: root, + workspace_kind: 'direct', prompt_dispatched: true, + error: { code: 'provider_billing_required', message: 'PRIVATE_CREDENTIAL' }, + }, + }); + return { dispatched: true, confidence: 'authoritative', cursor: '0' }; + }, + inspectWorkspace: async () => ({ current_head: BASE_SHA, clean: true, changed_files: [], commits: [] }), + }); + await adapter.dispatch('delegate', { run_request: request() }); + const receipt = await adapter.dispatch('task', { run_id: 'simple-supervisor' }); + assert.equal(receipt.phase, 'degraded'); + assert.equal(receipt.lanes[0].error.code, 'provider_billing_required'); + assert.match(receipt.lanes[0].error.message, /billing/); + assert.doesNotMatch(JSON.stringify(receipt), /PRIVATE_PROMPT|PRIVATE_CREDENTIAL/); + await adapter.dispatch('task', { run_id: 'simple-supervisor' }); + assert.equal(dispatches, 1); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); + +test('production dispatch wait keeps a delayed acknowledgement active and reconciles late success without replay', async () => { + const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-delayed-dispatch-')); + let now = 0; + let dispatches = 0; + let taskId; + try { + const adapter = await createSupervisorRunToolAdapter({ + root, + inProcess: true, + execute: async () => ({ stdout: '' }), + compile: (value) => compileRunRequestV1(value, { observeGit: async () => OBSERVED }), + requestConsent: async () => ({ approved: true }), + providerReady: async () => ({ ready: true }), + processBoundaryReady: async () => ({ ready: true }), + verifyRepository: async () => ({ verified: true }), + prepareWorkspace: async ({ assignment }) => ({ + prepared: true, + workspace: { + task: assignment.task_id, + status: 'ready', + worktree_path: root, + branch: 'codex/delayed-dispatch', + start_sha: BASE_SHA, + }, + }), + submitTask: async (input) => { + dispatches += 1; + taskId = input.task_id; + return createTask({ + root, + prompt: 'PRIVATE_PROMPT', + record: { + id: input.task_id, + provider: 'dsh', + status: 'running', + cwd: root, + workspace_kind: 'direct', + dispatch_intent: true, + dispatch_uncertain: true, + prompt_dispatched: false, + provider_run_id: 'delayed-session', + }, + }); + }, + dispatchEvidenceTimeoutMs: 5_000, + dispatchEvidenceNow: () => now, + dispatchEvidenceSleep: async (milliseconds) => { now += milliseconds; }, + inspectWorkspace: async () => ({ + current_head: BASE_SHA, + clean: true, + changed_files: [], + commits: [], + }), + }); + + const pending = await adapter.dispatch('delegate', { run_request: request() }); + assert.equal(now, 5_000, 'the bounded wait elapsed without terminating the lane'); + assert.equal(pending.phase, 'dispatching'); + assert.equal(pending.lanes[0].status, 'session_ready'); + assert.equal(pending.lanes[0].prompt_dispatched, false); + assert.equal(pending.lanes[0].dispatch_confidence, 'uncertain'); + assert.equal(dispatches, 1); + + await updateTask(root, taskId, { + status: 'completed', + prompt_dispatched: true, + dispatch_evidence: 'authoritative', + dispatch_uncertain: false, + result: { summary: 'late success' }, + }); + const completed = await adapter.dispatch('task', { run_id: 'simple-supervisor' }); + assert.equal(completed.phase, 'completed'); + assert.equal(completed.lanes[0].prompt_dispatched, true); + assert.deepEqual(completed.lanes[0].result, { summary: 'late success' }); + assert.equal(dispatches, 1, 'late success must reconcile the original task without replay'); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); diff --git a/plugins/codex-co-engineer/test/r1-terminal-worker-exit-adversarial.test.mjs b/plugins/codex-co-engineer/test/r1-terminal-worker-exit-adversarial.test.mjs index 19ad2ff..81a1231 100644 --- a/plugins/codex-co-engineer/test/r1-terminal-worker-exit-adversarial.test.mjs +++ b/plugins/codex-co-engineer/test/r1-terminal-worker-exit-adversarial.test.mjs @@ -19,6 +19,7 @@ import { runAcpWorkerCli, workerSeamIncident, } from '../mcp/v3/acp-worker.mjs'; +import { BUNDLED_WORKTREE_BOOTSTRAP } from '../mcp/v3/worktree-bootstrap-runtime.mjs'; import { createTask, readTask, updateTask, writeRuntimeRecord } from '../mcp/v3/task-store.mjs'; import { CONTENT_FREE, @@ -59,8 +60,8 @@ function sanitizedWtbEnv() { function stubWtbRunFile(calls) { return async (command, argv = []) => { calls.push({ command, argv: [...argv] }); - assert.equal(command, 'worktree-bootstrap'); - assert.equal(path.basename(command), command); + assert.equal(command, BUNDLED_WORKTREE_BOOTSTRAP); + assert.equal(path.isAbsolute(command), true); return { stdout: '{}' }; }; } @@ -73,7 +74,7 @@ function assertStubbedWtbOnly(calls, { env, cwd }) { if (env.WORKTREE_BOOTSTRAP_TASK) { assert.equal(calls.length, 1); assert.deepEqual(calls[0], { - command: 'worktree-bootstrap', + command: BUNDLED_WORKTREE_BOOTSTRAP, argv: ['verify', env.WORKTREE_BOOTSTRAP_TASK, '--repo', cwd, '--require-writer'], }); } else { diff --git a/plugins/codex-co-engineer/test/setup.test.mjs b/plugins/codex-co-engineer/test/setup.test.mjs index 99f8309..08a317a 100644 --- a/plugins/codex-co-engineer/test/setup.test.mjs +++ b/plugins/codex-co-engineer/test/setup.test.mjs @@ -91,7 +91,7 @@ async function packageTree(root, versions = {}) { } } -async function fixture({ includeWorktree = true, versions, configMode = 0o600, recordInstall = false } = {}) { +async function fixture({ includePython = true, versions, configMode = 0o600, recordInstall = false } = {}) { const root = await mkdtemp(path.join(tmpdir(), 'co-engineer-setup-test-')); const bin = path.join(root, 'bin'); const home = path.join(root, 'home'); @@ -114,9 +114,6 @@ async function fixture({ includeWorktree = true, versions, configMode = 0o600, r ]) { await executable(path.join(bin, name), `#!/bin/sh\nprintf '%s\\n' '${output}'\n`); } - if (includeWorktree) { - await executable(path.join(bin, 'worktree-bootstrap'), '#!/bin/sh\nprintf \'%s\\n\' \'worktree-override\'\n'); - } const installArgsFile = path.join(root, 'npm-install.args'); const setupOutputFile = path.join(root, 'setup-output.txt'); await executable(path.join(bin, 'npm'), `#!/bin/sh @@ -155,9 +152,9 @@ fi "- id: acp-agent", " name: '@deepseek-ai/dsh-acp-demo'", ' config:', - ' provider: meta', - ' model: muse-spark-1.2-contributor', - ' apiKeyEnv: MODEL_API_KEY', + ' provider: openrouter', + ' model: meta/muse-spark-1.3-contributor', + ' apiKeyEnv: OPENROUTER_API_KEY', '', ].join('\n'), { encoding: 'utf8', mode: configMode }); await chmod(configFile, configMode); @@ -193,7 +190,7 @@ fi HOME: home, XDG_CONFIG_HOME: configHome, XDG_STATE_HOME: stateHome, - PATH: bin, + PATH: includePython ? `${bin}:/usr/bin:/bin` : bin, CODEX_CO_ENGINEER_DSH_COMMAND: path.join(bin, 'dsh'), CODEX_CO_ENGINEER_ACPX_COMMAND: path.join(bin, 'acpx'), CODEX_CO_ENGINEER_DSH_ACP_COMMAND: path.join(bin, 'dsh-acp-demo'), @@ -229,7 +226,7 @@ async function runCheck(environment) { return { code: 0, value: JSON.parse(output) }; } -test('setup check honors command and config overrides and verifies worktree-bootstrap', async () => { +test('setup check uses bundled worktree-bootstrap without an ambient PATH command', async () => { const value = await fixture(); try { const result = await runCheck(value.environment); @@ -237,7 +234,12 @@ test('setup check honors command and config overrides and verifies worktree-boot assert.equal(result.value.dsh.output, 'dsh-override'); assert.equal(result.value.acpx.output, 'acpx-override'); assert.equal(result.value.dshAcp.output, path.join(value.bin, 'dsh-acp-demo')); - assert.equal(result.value.worktreeBootstrap.output, 'worktree-override'); + assert.equal(result.value.node.ok, true); + assert.equal(result.value.node.required, '>=24.0.0'); + assert.equal(result.value.python.ok, true); + assert.equal(result.value.python.required, '>=3.11'); + assert.equal(result.value.worktreeBootstrap.output, 'worktree-bootstrap 1.1.0'); + assert.equal(result.value.worktreeBootstrap.source, 'bundled'); assert.equal(result.value.config.path, value.configFile); assert.equal(result.value.config.ok, true); assert.equal(result.value.oxConfig.path, value.oxConfigFile); @@ -251,12 +253,16 @@ test('setup check honors command and config overrides and verifies worktree-boot } }); -test('setup check fails closed when worktree-bootstrap is unavailable', async () => { - const value = await fixture({ includeWorktree: false }); +test('setup check diagnoses missing Python required by bundled worktree-bootstrap', async () => { + const value = await fixture({ includePython: false }); try { const result = await runCheck(value.environment); assert.equal(result.code, 1); + assert.equal(result.value.node.ok, true); + assert.equal(result.value.python.ok, false); + assert.equal(result.value.python.required, '>=3.11'); assert.equal(result.value.worktreeBootstrap.ok, false); + assert.equal(result.value.worktreeBootstrap.source, 'bundled'); assert.equal(result.value.config.ok, true); assert.equal(result.value.oxConfig.ok, true); assert.equal(result.value.packages.ok, true); @@ -335,10 +341,13 @@ test('setup install pins the exact DSH rc.7 composition', async () => { } assert.ok(args.at(-1)?.endsWith('fake.tgz')); const museConfig = await readFile(value.configFile, 'utf8'); - assert.match(museConfig, /provider: meta/u); - assert.match(museConfig, /model: muse-spark-1\.2-contributor/u); - assert.match(museConfig, /apiKeyEnv: MODEL_API_KEY/u); - assert.doesNotMatch(museConfig, /openrouter|OPENROUTER_API_KEY|stealth\/ox-alpha/u); + assert.match(museConfig, /provider: openrouter/u); + assert.match(museConfig, /model: meta\/muse-spark-1\.3-contributor/u); + assert.match(museConfig, /apiKeyEnv: OPENROUTER_API_KEY/u); + assert.match(museConfig, /baseURL: https:\/\/openrouter\.ai\/api\/v1/u); + assert.match(museConfig, /reasoning: xhigh/u); + assert.match(museConfig, /reasoningEfforts:\n\s+xhigh: xhigh/u); + assert.doesNotMatch(museConfig, /api\.meta\.ai|MODEL_API_KEY|stealth\/ox-alpha/u); const oxConfig = await readFile(value.oxConfigFile, 'utf8'); assert.match(oxConfig, /provider: openrouter/u); assert.match(oxConfig, /model: stealth\/ox-alpha/u); @@ -346,7 +355,7 @@ test('setup install pins the exact DSH rc.7 composition', async () => { assert.match(oxConfig, /baseURL: https:\/\/openrouter\.ai\/api\/v1/u); assert.match(oxConfig, /reasoning: max/u); assert.match(oxConfig, /reasoningEfforts:\n\s+low: low\n\s+high: high\n\s+max: max/u); - assert.doesNotMatch(oxConfig, /api\.meta\.ai|MODEL_API_KEY|muse-spark-1\.2-contributor/u); + assert.doesNotMatch(oxConfig, /api\.meta\.ai|MODEL_API_KEY|muse-spark-1\.2-contributor|meta\/muse-spark-1\.3-contributor/u); const setupOutput = child.stdout?.trim() ? child.stdout : await readFile(value.setupOutputFile, 'utf8'); assert.match(setupOutput, /Installed Co-Engineer agent dependencies/u); } finally { diff --git a/plugins/codex-co-engineer/test/v3-acp-worker.test.mjs b/plugins/codex-co-engineer/test/v3-acp-worker.test.mjs index fc5107e..9f41d11 100644 --- a/plugins/codex-co-engineer/test/v3-acp-worker.test.mjs +++ b/plugins/codex-co-engineer/test/v3-acp-worker.test.mjs @@ -5,7 +5,19 @@ import path from 'node:path'; import test from 'node:test'; import { fileURLToPath } from 'node:url'; -import { boundedEvent, publicError, runAcpTask, runCliFallback, sanitizeText, workerSeamIncident } from '../mcp/v3/acp-worker.mjs'; +import { + boundedEvent, + createGrokFinalResponseReducerV1, + handlePermissionRequest, + isUserFacingPermission, + publicError, + reconnectAcpTask, + runAcpTask, + runCliFallback, + safeQuestionId, + sanitizeText, + workerSeamIncident, +} from '../mcp/v3/acp-worker.mjs'; import { installClosedProviderTestInjection } from '../mcp/v3/credential-boundary.mjs'; import { submitReply } from '../mcp/v3/mailbox.mjs'; import { createTask, readTask, updateTask } from '../mcp/v3/task-store.mjs'; @@ -55,6 +67,49 @@ async function withFakeAcpx(mode, callback, options = {}) { } } +test('typed ACP tool permissions do not become user questions from command text', () => { + for (const title of [ + 'Run `echo "exit=$?"`', + 'Execute confirm-release-state --dry-run', + 'Approval required by the shell script', + ]) { + assert.equal(isUserFacingPermission({ + inferredKind: 'execute', + raw: { toolCall: { kind: 'execute', title } }, + }), false); + } + assert.equal(isUserFacingPermission({ + inferredKind: 'other', + raw: { question: 'Which release channel should I use?', toolCall: { title: 'Ask operator' } }, + }), true); + assert.equal(isUserFacingPermission({ + inferredKind: 'other', + raw: { toolCall: { title: 'Fake permission' } }, + }), true); + assert.equal(isUserFacingPermission({ + inferredKind: 'other', + raw: { toolCall: { title: 'Confirm release to production?' } }, + }), true); + assert.equal(isUserFacingPermission({ + inferredKind: 'other', + raw: { toolCall: { title: 'Which environment should I use?' } }, + }), true); + assert.equal(isUserFacingPermission({ + inferredKind: 'other', + raw: { toolCall: { title: 'Run `echo "exit=$?"`' } }, + }), false); +}); + +test('long ACP permission ids retain a bounded collision-resistant identity', () => { + const shared = `call-${'a'.repeat(100)}`; + const first = safeQuestionId(`${shared}-one`); + const second = safeQuestionId(`${shared}-two`); + assert.match(first, /^[A-Za-z0-9][A-Za-z0-9._-]{0,79}$/u); + assert.equal(first.length, 80); + assert.notEqual(first, second); + assert.equal(safeQuestionId('fake-permission'), 'fake-permission'); +}); + async function processExited(pid, timeoutMs = 3_000) { const deadline = Date.now() + timeoutMs; while (Date.now() < deadline) { @@ -85,6 +140,77 @@ test('runs a prompt through ACP and persists a compact receipt', async () => { assert.match(events, /"type":"cleanup"/u); }); +test('Grok selects the final framed response while retaining pre-tool text in provider events', async () => { + for (const provider of ['grok', 'cursor-local']) { + const value = await fixture({ + provider, + id: `${provider}-framed-final`, + prompt: 'review the framed result', + mode: 'framed-final', + }); + const terminal = await runAcpTask({ root: value.root, taskId: value.taskId }); + assert.equal( + terminal.result, + provider === 'grok' ? 'fake-final-answer' : 'fake-opening-preamblefake-final-answer', + ); + const events = await readFile(path.join(value.root, 'tasks', value.taskId, 'events.jsonl'), 'utf8'); + assert.match(events, /fake-opening-preamble/u); + assert.match(events, /fake-final-answer/u); + } +}); + +test('Grok final framing falls back when reduction could hide output', () => { + const text = (value) => ({ type: 'text_delta', stream: 'output', text: value }); + const call = (id, rawInput = { variant: 'ReadFile' }) => ({ + type: 'tool_call', tag: 'tool_call', toolCallId: id, rawInput, + }); + const done = (id) => ({ + type: 'tool_call', tag: 'tool_call_update', toolCallId: id, status: 'completed', + }); + const completed = { status: 'completed', stopReason: 'end_turn' }; + const finish = (events, options = {}) => { + const reducer = createGrokFinalResponseReducerV1(); + for (const event of events) reducer.append(event); + return reducer.finish({ + turnResult: options.turnResult ?? completed, + fullSnapshot: { overflow: options.overflow === true }, + }); + }; + + const selected = finish([text('preamble'), call('read'), done('read'), text('final')]); + assert.equal(selected.bounded.value, 'final'); + assert.equal(selected.snapshot.source.toString('utf8'), 'final'); + + const parallel = finish([ + text('preamble'), call('one'), call('two'), done('one'), done('two'), text('parallel-final'), + ]); + assert.equal(parallel.bounded.value, 'parallel-final'); + const sequential = finish([ + text('preamble'), call('one'), done('one'), text('between'), call('two'), done('two'), + text('final-'), text('chunks'), + ]); + assert.equal(sequential.bounded.value, 'final-chunks'); + + const fallbacks = [ + ['whitespace final', [text('preamble'), call('read'), done('read'), text(' ')], {}], + ['full collector overflow', [text('preamble'), call('read'), done('read'), text('final')], { overflow: true }], + ['text while tool pending', [text('preamble'), call('read'), text('interleaved'), done('read'), text('final')], {}], + ['text after partial settle', [text('preamble'), call('one'), call('two'), done('one'), text('interleaved'), done('two'), text('final')], {}], + ['web search', [text('preamble'), call('search', { variant: 'WebSearch' }), done('search'), text('final')], {}], + ['unmatched terminal', [text('preamble'), done('missing'), text('final')], {}], + ['duplicate pending id', [text('preamble'), call('same'), call('same'), done('same'), text('final')], {}], + ['missing id', [text('preamble'), call(null), text('final')], {}], + ['overlong id', [text('preamble'), call('x'.repeat(513)), text('final')], {}], + ['pending tool', [text('partial'), call('pending')], {}], + ['pending after settled round', [text('preamble'), call('one'), done('one'), text('candidate'), call('pending')], {}], + ['failed turn', [text('preamble'), call('read'), done('read'), text('failure detail')], { turnResult: { status: 'failed', stopReason: 'error' } }], + ['non-end turn', [text('preamble'), call('read'), done('read'), text('partial')], { turnResult: { status: 'completed', stopReason: 'max_tokens' } }], + ]; + for (const [name, events, options] of fallbacks) { + assert.equal(finish(events, options), null, name); + } +}); + for (const provider of ['grok', 'cursor-local']) { test(`${provider} preserves a terminal verdict at the end of long ACP output`, async () => { const value = await fixture({ @@ -111,6 +237,53 @@ test('does not start a fresh ACP worker from transport_lost', async () => { ); }); +test('reconnects an acknowledged ACP session without replaying its prompt', async () => { + const value = await fixture({ id: 'same-session-reconnect' }); + await updateTask(value.root, value.taskId, { + status: 'transport_lost', + transport: 'acp', + prompt_dispatched: true, + dispatch_evidence: 'authoritative', + acp_session_id: 'persisted-acp-session', + }); + let ensureInput; + let startTurnCalled = false; + let closed = false; + const resumed = await reconnectAcpTask({ + root: value.root, + taskId: value.taskId, + runtimeFactory: async () => ({ + ensureSession: async (input) => { + ensureInput = input; + return { + backendSessionId: 'persisted-acp-session', + agentSessionId: 'persisted-agent-session', + }; + }, + getStatus: async () => ({ status: 'running' }), + startTurn: async () => { + startTurnCalled = true; + throw new Error('resume path must not start a turn'); + }, + close: async () => { + closed = true; + }, + }), + }); + + assert.equal(resumed.reconnected, true); + assert.equal(resumed.prompt_replayed, false); + assert.equal(ensureInput.resumeSessionId, 'persisted-acp-session'); + assert.equal(startTurnCalled, false); + assert.equal(closed, true); + const { task } = await readTask(value.root, value.taskId); + assert.equal(task.status, 'running'); + assert.equal(task.prompt_dispatched, true); + const events = await readFile(path.join(value.root, 'tasks', value.taskId, 'events.jsonl'), 'utf8'); + assert.match(events, /session_reconnected/u); + assert.match(events, /prompt_replayed":false/u); +}); + test('recursively bounds and redacts provider events and errors', () => { const prompt = 'private prompt sk-prompt-secret-1234567890'; const event = { @@ -237,7 +410,7 @@ test('provider failure after dispatch is never marked safe to replay', async () assert.equal(workerSeamIncident(task), false); }); -test('DSH scopes ACPX artifacts to the task and removes them after persistence', async () => { +test('DSH uses bounded ACPX exec stdin and scopes artifacts to the task', async () => { const value = await fixture({ provider: 'dsh', id: 'dsh-flow' }); await updateTask(value.root, value.taskId, { error: { code: 'worker_boundary_uncertain', message: 'stale reconciliation marker' }, @@ -250,8 +423,14 @@ test('DSH scopes ACPX artifacts to the task and removes them after persistence', assert.equal(terminal.result, 'DSH_FAKE_OK'); assert.equal(terminal.error, null); assert.equal(terminal.acp_session_id, 'dsh-fake-session'); - assert.equal(terminal.dispatch_uncertain, true); - assert.equal(terminal.prompt_dispatched, undefined); + assert.equal(terminal.dispatch_uncertain, false); + assert.equal(terminal.prompt_dispatched, true); + assert.equal(terminal.dispatch_evidence, 'authoritative'); + const observed = JSON.parse(await readFile(path.join(value.cwd, '.acpx-fake-observed.json'), 'utf8')); + assert.ok(observed.argv.includes('exec')); + assert.ok(observed.argv.includes('--file')); + assert.ok(observed.argv.includes('-')); + assert.equal(observed.argv.includes('review this repository'), false); await access(artifactMarker); const entries = await readdir(path.join(value.root, 'tasks', value.taskId)); assert.equal(entries.some((entry) => entry.startsWith('flow-input-')), false); @@ -277,6 +456,141 @@ test('DSH ACPX preserves bounded nested output values', async () => { assert.equal(terminal.result_original_chars, 10_025); }); +test('DSH exposes a correlated provider billing failure without replay or prompt leakage', async () => { + const cliMarker = path.join((await mkdtemp(path.join(tmpdir(), 'co-engineer-v3-dsh-billing-'))), 'cli-ran'); + const prompt = 'private billing prompt'; + const value = await fixture({ + provider: 'dsh', + id: 'dsh-provider-billing', + prompt, + cliArgv: [process.execPath, '-e', `require('node:fs').writeFileSync(${JSON.stringify(cliMarker)}, 'ran')`], + }); + await assert.rejects( + withFakeAcpx('provider-error', () => runAcpTask({ root: value.root, taskId: value.taskId })), + (error) => error.code === 'provider_billing_required', + ); + const { task } = await readTask(value.root, value.taskId); + assert.equal(task.status, 'failed'); + assert.equal(task.error.code, 'provider_billing_required'); + assert.equal(task.error.message, 'The provider billing configuration is unavailable.'); + assert.equal(task.prompt_dispatched, true); + assert.equal(task.dispatch_evidence, 'authoritative'); + assert.equal(task.dispatch_uncertain, false); + assert.equal(task.fallback_safe, false); + await assert.rejects(access(cliMarker)); + const events = await readFile(path.join(value.root, 'tasks', value.taskId, 'events.jsonl'), 'utf8'); + assert.doesNotMatch(events, new RegExp(prompt, 'u')); +}); + +test('DSH keeps an ACP authentication rejection pre-dispatch and does not fall back', async () => { + const cliMarker = path.join((await mkdtemp(path.join(tmpdir(), 'co-engineer-v3-dsh-auth-'))), 'cli-ran'); + const value = await fixture({ + provider: 'dsh', + id: 'dsh-provider-auth', + cliArgv: [process.execPath, '-e', `require('node:fs').writeFileSync(${JSON.stringify(cliMarker)}, 'ran')`], + }); + await assert.rejects( + withFakeAcpx('auth-error', () => runAcpTask({ root: value.root, taskId: value.taskId })), + (error) => error.code === 'authentication_required', + ); + const { task } = await readTask(value.root, value.taskId); + assert.equal(task.status, 'failed'); + assert.equal(task.error.code, 'authentication_required'); + assert.equal(task.prompt_dispatched, undefined); + assert.equal(task.fallback_safe, false); + await assert.rejects(access(cliMarker)); +}); + +test('DSH rejects an uncorrelated or incomplete ACP prompt result', async () => { + const value = await fixture({ provider: 'dsh', id: 'dsh-invalid-result' }); + await assert.rejects( + withFakeAcpx('invalid-result', () => runAcpTask({ root: value.root, taskId: value.taskId })), + (error) => error.code === 'acpx_invalid_result', + ); + const { task } = await readTask(value.root, value.taskId); + assert.equal(task.status, 'failed'); + assert.equal(task.prompt_dispatched, undefined); + assert.equal(task.dispatch_uncertain, true); + assert.equal(task.fallback_safe, false); +}); + +test('DSH rejects an unmatched terminal response instead of manufacturing success', async () => { + const value = await fixture({ provider: 'dsh', id: 'dsh-unmatched-result' }); + await assert.rejects( + withFakeAcpx('unmatched-result', () => runAcpTask({ root: value.root, taskId: value.taskId })), + (error) => error.code === 'acpx_invalid_result', + ); + const { task } = await readTask(value.root, value.taskId); + assert.equal(task.status, 'failed'); + assert.equal(task.result, undefined); + assert.equal(task.dispatch_uncertain, true); + assert.equal(task.fallback_safe, false); +}); + +test('DSH treats thought updates as dispatch evidence without persisting thought text', async () => { + const value = await fixture({ provider: 'dsh', id: 'dsh-thought-first' }); + const terminal = await withFakeAcpx('thought-first', () => runAcpTask({ root: value.root, taskId: value.taskId })); + assert.equal(terminal.status, 'completed'); + assert.equal(terminal.result, 'THOUGHT_RESULT'); + assert.equal(terminal.dispatch_evidence, 'authoritative'); + const events = await readFile(path.join(value.root, 'tasks', value.taskId, 'events.jsonl'), 'utf8'); + assert.doesNotMatch(events, /PRIVATE_THOUGHT_SHOULD_NOT_BE_STORED/u); +}); + +test('DSH preserves split UTF-8 ACPX output while parsing correlated frames', async () => { + const value = await fixture({ provider: 'dsh', id: 'dsh-utf8-output' }); + const terminal = await withFakeAcpx('utf8', () => runAcpTask({ root: value.root, taskId: value.taskId })); + assert.equal(terminal.status, 'completed'); + assert.equal(terminal.result, 'UTF8_OK 😀 café'); +}); + +test('DSH fails closed on malformed ACPX JSON output', async () => { + const value = await fixture({ provider: 'dsh', id: 'dsh-malformed-output' }); + await assert.rejects( + withFakeAcpx('malformed', () => runAcpTask({ root: value.root, taskId: value.taskId })), + (error) => error.code === 'acpx_protocol_invalid', + ); + const { task } = await readTask(value.root, value.taskId); + assert.equal(task.status, 'failed'); + assert.equal(task.result, undefined); + assert.equal(task.fallback_safe, false); +}); + +test('DSH rejects a prompt request whose session differs from session/new', async () => { + const value = await fixture({ provider: 'dsh', id: 'dsh-wrong-prompt-session' }); + await assert.rejects( + withFakeAcpx('wrong-session', () => runAcpTask({ root: value.root, taskId: value.taskId })), + (error) => error.code === 'acpx_protocol_invalid', + ); + const { task } = await readTask(value.root, value.taskId); + assert.equal(task.status, 'failed'); + assert.equal(task.prompt_dispatched, undefined); + assert.equal(task.dispatch_uncertain, true); +}); + +test('DSH rejects duplicate outstanding ACPX request ids', async () => { + const value = await fixture({ provider: 'dsh', id: 'dsh-duplicate-request-id' }); + await assert.rejects( + withFakeAcpx('duplicate-id', () => runAcpTask({ root: value.root, taskId: value.taskId })), + (error) => error.code === 'acpx_protocol_invalid', + ); + const { task } = await readTask(value.root, value.taskId); + assert.equal(task.status, 'failed'); + assert.equal(task.result, undefined); +}); + +test('DSH rejects a terminal response with a mismatched session id', async () => { + const value = await fixture({ provider: 'dsh', id: 'dsh-wrong-result-session' }); + await assert.rejects( + withFakeAcpx('wrong-result-session', () => runAcpTask({ root: value.root, taskId: value.taskId })), + (error) => error.code === 'acpx_invalid_result', + ); + const { task } = await readTask(value.root, value.taskId); + assert.equal(task.status, 'failed'); + assert.equal(task.result, undefined); + assert.equal(task.dispatch_uncertain, true); +}); + test('DSH does not fall back after ACPX has spawned without an acknowledgement', async () => { const cliMarker = path.join((await mkdtemp(path.join(tmpdir(), 'co-engineer-v3-dsh-cli-'))), 'cli-ran'); const value = await fixture({ @@ -330,7 +644,7 @@ test('DSH Ox Alpha fails closed instead of using a model-blind pre-spawn CLI fal test('DSH Muse still allows pre-spawn CLI fallback when ACPX cannot start', async () => { const value = await fixture({ provider: 'dsh', - dshModel: 'muse-spark-1.2-contributor', + dshModel: 'meta/muse-spark-1.3-contributor', id: 'dsh-muse-pre-spawn-fallback', cliArgv: [process.execPath, '-e', 'process.stdout.write("MUSE_CLI_FALLBACK_OK")'], }); @@ -347,7 +661,7 @@ test('DSH Muse still allows pre-spawn CLI fallback when ACPX cannot start', asyn const { task } = await readTask(value.root, value.taskId); assert.equal(task.status, 'completed'); assert.equal(task.transport, 'cli'); - assert.equal(task.dsh_model, 'muse-spark-1.2-contributor'); + assert.equal(task.dsh_model, 'meta/muse-spark-1.3-contributor'); assert.equal(task.fallback_from, 'acp'); assert.equal(task.prompt_dispatched, true); assert.equal(task.fallback_safe, false); @@ -373,28 +687,89 @@ test('DSH deadline kills a detached ACPX descendant before terminalizing', async assert.equal((await readdir(path.join(value.root, 'tasks', value.taskId))).includes('acpx-home'), false); }); -test('user-facing ACP permission requests persist needs_attention and accept one same-session reply', async () => { - const value = await fixture({ prompt: 'need permission please', id: 'perm-one', timeoutMs: 8_000 }); - const running = runAcpTask({ root: value.root, taskId: value.taskId }); - const deadline = Date.now() + 5_000; - let attention; - while (Date.now() < deadline) { - const current = (await readTask(value.root, value.taskId)).task; - if (current.status === 'needs_attention') { - attention = current; - break; +test('worker permission handling persists question text and resumes through the mailbox reply', async () => { + const value = await fixture({ id: 'perm-question' }); + const controller = new AbortController(); + const question = 'Which environment should I use?'; + const options = [ + { optionId: 'allow', kind: 'allow_once', name: 'Allow once' }, + { optionId: 'reject', kind: 'reject_once', name: 'Reject' }, + ]; + const pending = handlePermissionRequest(value.root, value.taskId, { + sessionId: 'fake-session-question', + inferredKind: 'other', + raw: { + question, + toolCall: { toolCallId: 'question-environment', title: 'Ask operator' }, + options, + }, + }, controller.signal); + try { + const deadline = Date.now() + 1_000; + let attention; + while (Date.now() < deadline) { + const current = (await readTask(value.root, value.taskId)).task; + if (current.status === 'needs_attention') { + attention = current; + break; + } + await new Promise((resolve) => setTimeout(resolve, 10)); } - await new Promise((resolve) => setTimeout(resolve, 25)); + assert.equal(attention?.status, 'needs_attention'); + const stored = JSON.parse(await readFile( + path.join(value.root, 'tasks', value.taskId, 'attention.json'), + 'utf8', + )); + assert.equal(stored.prompt, question); + assert.deepEqual(stored.options, options); + await submitReply(value.root, value.taskId, { + session_id: attention.attention.session_id, + question_id: attention.attention.question_id, + response: { optionId: 'allow' }, + }); + assert.deepEqual(await pending, { outcome: 'allow_once', optionId: 'allow' }); + assert.equal((await readTask(value.root, value.taskId)).task.status, 'running'); + } finally { + controller.abort(); + await pending.catch(() => {}); + } +}); + +test('title-only ACP questions persist and continue the real worker session', async () => { + const controller = new AbortController(); + let running; + try { + const question = 'Which environment should I use?'; + const value = await fixture({ + prompt: 'need permission title question please', + id: 'perm-title-question', + timeoutMs: 8_000, + }); + running = runAcpTask({ root: value.root, taskId: value.taskId, signal: controller.signal }); + const deadline = Date.now() + 5_000; + let attention; + while (Date.now() < deadline) { + const current = (await readTask(value.root, value.taskId)).task; + if (current.status === 'needs_attention') { + attention = current; + break; + } + await new Promise((resolve) => setTimeout(resolve, 25)); + } + assert.equal(attention?.status, 'needs_attention'); + const stored = JSON.parse(await readFile( + path.join(value.root, 'tasks', value.taskId, 'attention.json'), + 'utf8', + )); + assert.equal(stored.prompt, question); + await submitReply(value.root, value.taskId, { + session_id: attention.attention.session_id, + question_id: attention.attention.question_id, + response: { optionId: 'allow' }, + }); + assert.equal((await running).status, 'completed'); + } finally { + controller.abort(); + await running?.catch(() => {}); } - assert.equal(attention?.status, 'needs_attention'); - assert.ok(attention.attention?.session_id); - assert.ok(attention.attention?.question_id); - await submitReply(value.root, value.taskId, { - session_id: attention.attention.session_id, - question_id: attention.attention.question_id, - response: 'allow_once', - }); - const terminal = await running; - assert.equal(terminal.status, 'completed'); - assert.equal((await readTask(value.root, value.taskId)).task.status, 'completed'); }); diff --git a/plugins/codex-co-engineer/test/v3-child-envelope.test.mjs b/plugins/codex-co-engineer/test/v3-child-envelope.test.mjs index c03975e..78be16f 100644 --- a/plugins/codex-co-engineer/test/v3-child-envelope.test.mjs +++ b/plugins/codex-co-engineer/test/v3-child-envelope.test.mjs @@ -324,7 +324,8 @@ test('an envelope embeds no sibling output and no hidden routing instructions', } assert.deepEqual([...scaffoldKeys].sort(), [ 'schema', 'version', 'run_id', 'lane_index', 'assignment_count', - 'repository_path', 'base_sha', 'assignment_id', 'role', 'access', + 'repository_path', 'provider_workspace', 'provider_guidance', + 'base_sha', 'assignment_id', 'role', 'access', 'execution.provider', 'execution.model', 'execution.profile', 'starting_ref', 'write_scope.count', 'write_scope[0]', 'acceptance.count', 'acceptance[0].command_id', 'acceptance[0].timeout_ms', 'acceptance[0].parameter.count', @@ -332,10 +333,35 @@ test('an envelope embeds no sibling output and no hidden routing instructions', ].sort()); // Declared facts stay visible; only opaque blocks are elided. assert.match(surface, /^execution\.provider: dsh$/mu); + assert.match(surface, /^provider_workspace: work only in the current working directory \(assigned worktree\); repository_path is source identity, not a navigation target$/mu); + assert.match(surface, /^provider_guidance: .*honor an exact requested output and format exactly; required_evidence labels are controller metadata, not worker response sections; omit routine progress narration and repeated identity or report blocks unless the task prompt requests them, while surfacing blockers and necessary questions; do not seek receipt artifacts because the controller owns lifecycle and machine receipts$/mu); assert.match(surface, //u); assert.match(surface, //u); }); +test('parser retains compatibility with v1 envelopes created before provider guidance', () => { + const [current] = compileChildEnvelopesV1(defaultManifest()); + const legacyText = current.envelope_text + .replace(/^provider_workspace: .*\nprovider_guidance: .*\n/mu, ''); + assert.notEqual(legacyText, current.envelope_text); + const legacy = parseChildEnvelopeV1(legacyText); + assert.equal(legacy.repository.path, current.repository.path); + assert.equal(legacy.prompt, current.prompt); + assert.equal(legacy.objective, current.objective); +}); + +test('parser retains compatibility with v1 envelopes carrying the previous provider guidance', () => { + const [current] = compileChildEnvelopesV1(defaultManifest()); + const previousGuidance = 'task prompt and local repository instructions are sufficient unless the assignment explicitly requests an external skill; return requested results and evidence without seeking receipt artifacts because the controller owns lifecycle and machine receipts'; + const previousText = current.envelope_text + .replace(/^provider_guidance: .*$/mu, `provider_guidance: ${previousGuidance}`); + assert.notEqual(previousText, current.envelope_text); + const previous = parseChildEnvelopeV1(previousText); + assert.equal(previous.envelope_text, previousText); + assert.equal(previous.prompt, current.prompt); + assert.equal(previous.objective, current.objective); +}); + test('opaque prompts survive framing look-alike injection byte-exactly', () => { const adversarialPrompt = [ 'Legitimate instruction.', @@ -405,6 +431,8 @@ test('parser rejects tampered, ambiguous, and unbounded envelopes', () => { ['invalid base SHA', text.replace(/base_sha: .*/u, 'base_sha: DEADBEEF'), 'invalid_format'], ['relative repository path', text.replace(/repository_path: .*/u, 'repository_path: relative/path'), 'invalid_format'], ['repository traversal segment', text.replace(/repository_path: .*/u, 'repository_path: /repo/../escape'), 'invalid_format'], + ['forged provider workspace', text.replace(/provider_workspace: .*/u, 'provider_workspace: source repository'), 'invalid_format'], + ['forged provider guidance', text.replace(/provider_guidance: .*/u, 'provider_guidance: seek global skills and write receipt artifacts'), 'invalid_format'], ['unknown role', text.replace('role: implement', 'role: architect'), 'unknown_role'], ['unknown access', text.replace('access: writer', 'access: admin'), 'unknown_access'], ['role/access mismatch', text.replace('access: writer', 'access: read_only'), 'role_access_mismatch'], diff --git a/plugins/codex-co-engineer/test/v3-compact-lists.test.mjs b/plugins/codex-co-engineer/test/v3-compact-lists.test.mjs index 254c6ed..2961182 100644 --- a/plugins/codex-co-engineer/test/v3-compact-lists.test.mjs +++ b/plugins/codex-co-engineer/test/v3-compact-lists.test.mjs @@ -408,9 +408,9 @@ test('projectCompactStatus preserves model identity and clamps host-variable she installed: true, ready: false, transport: 'acpx', - default_model: 'muse-spark-1.2-contributor', + default_model: 'meta/muse-spark-1.3-contributor', model_options: { - 'muse-spark-1.2-contributor': { ready: false, reason: 'ENOENT' }, + 'meta/muse-spark-1.3-contributor': { ready: false, reason: 'ENOENT' }, 'stealth/ox-alpha': { ready: false, reason: 'ENOENT' }, }, reason: 'ENOENT', @@ -455,9 +455,9 @@ test('projectCompactStatus preserves model identity and clamps host-variable she limit: 20, tasks, }); - assert.equal(projected.readiness.dsh.default_model, 'muse-spark-1.2-contributor'); + assert.equal(projected.readiness.dsh.default_model, 'meta/muse-spark-1.3-contributor'); assert.deepEqual(Object.keys(projected.readiness.dsh.model_options).sort(), [ - 'muse-spark-1.2-contributor', + 'meta/muse-spark-1.3-contributor', 'stealth/ox-alpha', ].sort()); assert.equal(projected.readiness.dsh.model_options['stealth/ox-alpha'].ready, false); @@ -480,7 +480,7 @@ test('byte targets under worst valid values: readiness <=8192, compact status 20 const readiness = await request({ jsonrpc:'2.0', id:1, method:'tools/call', params:{name:'status', arguments:{detail:'compact', include_tasks:false}}}); const readinessBytes = jsonRpcBytes(readiness.result.structuredContent); assert.ok(readinessBytes <= 8192, `readiness ${readinessBytes} exceeds 8192`); - assert.equal(readiness.result.structuredContent.readiness.dsh.default_model, 'muse-spark-1.2-contributor'); + assert.equal(readiness.result.structuredContent.readiness.dsh.default_model, 'meta/muse-spark-1.3-contributor'); assert.ok(readiness.result.structuredContent.readiness.dsh.model_options['stealth/ox-alpha']); assert.equal(Object.hasOwn(readiness.result.structuredContent.mcp_pending_call, 'notes'), false); for (let i=0;i<20;i++) { diff --git a/plugins/codex-co-engineer/test/v3-cursor-cloud-worker.test.mjs b/plugins/codex-co-engineer/test/v3-cursor-cloud-worker.test.mjs index af43381..125546a 100644 --- a/plugins/codex-co-engineer/test/v3-cursor-cloud-worker.test.mjs +++ b/plugins/codex-co-engineer/test/v3-cursor-cloud-worker.test.mjs @@ -1,12 +1,18 @@ import assert from 'node:assert/strict'; -import { execFile } from 'node:child_process'; -import { mkdtemp, writeFile } from 'node:fs/promises'; +import { execFile, spawn } from 'node:child_process'; +import { once } from 'node:events'; +import { mkdtemp, rm, writeFile } from 'node:fs/promises'; import { tmpdir } from 'node:os'; import path from 'node:path'; import test from 'node:test'; import { promisify } from 'node:util'; -import { cancelCursorCloudTask, reconcileCursorCloudTask, runCursorCloudTask } from '../mcp/v3/cursor-cloud-worker.mjs'; +import { + cancelCursorCloudTask, + loadCursorSdk, + reconcileCursorCloudTask, + runCursorCloudTask, +} from '../mcp/v3/cursor-cloud-worker.mjs'; import { extendTaskDeadline } from '../mcp/v3/supervisor.mjs'; import { createTask, readTask, updateTask } from '../mcp/v3/task-store.mjs'; @@ -23,6 +29,72 @@ async function createCloudTask({ root, prompt, record }) { const run = promisify(execFile); +test('global SDK discovery survives a deleted inherited working directory', async () => { + const inheritedCwd = await mkdtemp(path.join(tmpdir(), 'co-engineer-deleted-cwd-')); + const workerModule = new URL('../mcp/v3/cursor-cloud-worker.mjs', import.meta.url).href; + const script = ` + import { once } from 'node:events'; + import { loadCursorSdk } from ${JSON.stringify(workerModule)}; + process.stdout.write('ready\\n'); + await once(process.stdin, 'data'); + const sdk = await loadCursorSdk({ + execute: async (_command, _args, options) => { + if (options.cwd !== ${JSON.stringify(path.parse(process.execPath).root)}) { + throw Object.assign(new Error('unstable discovery cwd'), { code: 'wrong_cwd' }); + } + return { stdout: ${JSON.stringify(path.join(path.parse(process.execPath).root, 'synthetic-global'))} }; + }, + importModule: async () => ({ loaded: true }), + }); + process.stdout.write(JSON.stringify(sdk)); + `; + const child = spawn(process.execPath, ['--input-type=module', '--eval', script], { + cwd: inheritedCwd, + stdio: ['pipe', 'pipe', 'pipe'], + }); + let stdout = ''; + let stderr = ''; + child.stdout.setEncoding('utf8'); + child.stderr.setEncoding('utf8'); + child.stdout.on('data', (chunk) => { stdout += chunk; }); + child.stderr.on('data', (chunk) => { stderr += chunk; }); + await once(child.stdout, 'data'); + await rm(inheritedCwd, { recursive: true, force: true }); + child.stdin.end('continue'); + const [code] = await once(child, 'close'); + + assert.equal(code, 0, stderr); + assert.match(stdout, /ready\n/u); + assert.match(stdout, /"loaded":true/u); +}); + +test('global SDK discovery reports a typed local failure and preserves credential projection', async () => { + let observed; + await assert.rejects( + loadCursorSdk({ + env: { + PATH: process.env.PATH, + HOME: '/synthetic-home', + CURSOR_API_KEY: 'PRIVATE_CURSOR_KEY', + OPENAI_API_KEY: 'PRIVATE_OPENAI_KEY', + }, + execute: async (command, args, options) => { + observed = { command, args, options }; + throw Object.assign(new Error('npm failed PRIVATE_CURSOR_KEY'), { code: 7 }); + }, + }), + (error) => error.code === 'cursor_sdk_discovery_failed' + && error.message === 'Cursor SDK global installation path could not be discovered.' + && !error.message.includes('PRIVATE_CURSOR_KEY'), + ); + + assert.equal(observed.command, 'npm'); + assert.deepEqual(observed.args, ['root', '--global']); + assert.equal(observed.options.cwd, path.parse(process.execPath).root); + assert.equal(observed.options.env.CURSOR_API_KEY, undefined); + assert.equal(observed.options.env.OPENAI_API_KEY, undefined); +}); + async function commitRepo(repo) { await run('git', ['-C', repo, '-c', 'user.name=Co-Engineer Test', '-c', 'user.email=test@example.invalid', 'commit', '--allow-empty', '-m', 'initial']); } @@ -53,6 +125,17 @@ test('uses stable Cursor agent/run idempotency and records returned PR', async ( observed.send = { prompt, options: options2 }; return { id: 'run-one', wait: async () => ({ id: 'run-one', status: 'finished', result: 'done', + durationMs: 1_234, + model: { id: 'claude-sonnet-4-5', params: [{ name: 'reasoning', value: 'high' }] }, + usage: { + inputTokens: 11, + outputTokens: 7, + cacheReadTokens: 3, + cacheWriteTokens: 2, + totalTokens: 23, + reasoningTokens: 4, + }, + benignMetadata: { trace: 'future-sdk-field' }, git: { branches: [{ repoUrl: 'https://github.com/example/repo.git', branch: 'cursor/work', prUrl: 'https://github.com/example/repo/pull/1' }] }, }) }; }, @@ -726,7 +809,14 @@ test('reconciliation records a truthful archive result', async () => { taskId: 'cloud-reconcile-archive', sdk: { Agent: { getRun: async () => ({ id: 'run-finished', status: 'finished', wait: async () => ({ - id: 'run-finished', status: 'finished', result: 'done', git: { branches: [] }, + id: 'run-finished', status: 'finished', result: 'done', + requestId: undefined, + error: undefined, + git: undefined, + durationMs: 987, + model: { id: 'claude-sonnet-4-5' }, + usage: undefined, + benignMetadata: { trace: 'future-sdk-field' }, }) }), archive: async () => { throw Object.assign(new Error('archive unavailable'), { code: 'network_error' }); }, } }, @@ -985,6 +1075,40 @@ test('accepts one exact provider request identity and leaves ambiguous matches t assert.equal(ambiguousCreated, 0); }); +test('preserves a legitimate answer contained in the prompt while diagnostics stay redacted', async () => { + const root = await mkdtemp(path.join(tmpdir(), 'co-engineer-cursor-answer-overlap-')); + const repo = await createCloudRepo(root); + const prompt = 'Return exactly CLOUD_NATIVE_OK'; + const apiKey = 'cursor-answer-secret-123456'; + await createCloudTask({ root, prompt, record: { + id: 'cloud-answer-overlap', status: 'accepted', provider: 'cursor-cloud', role: 'review', cwd: repo, + } }); + const sdk = { Agent: { + create: async () => ({ + send: async () => ({ id: 'run-answer-overlap', wait: async () => ({ + id: 'run-answer-overlap', status: 'finished', + result: `CLOUD_NATIVE_OK; Authorization: Bearer ${apiKey}`, + error: { + message: `diagnostic echoed ${prompt}; Bearer ${apiKey}`, + detail: 'CLOUD_NATIVE_OK', + }, + git: { branches: [] }, + }) }), + close() {}, + }), + archive: async () => {}, + } }; + + const terminal = await runCursorCloudTask({ root, taskId: 'cloud-answer-overlap', sdk, apiKey }); + const serializedError = JSON.stringify(terminal.provider_error); + assert.equal(terminal.status, 'completed'); + assert.match(terminal.result, /^CLOUD_NATIVE_OK;/u); + assert.doesNotMatch(terminal.result, new RegExp(apiKey, 'u')); + assert.doesNotMatch(serializedError, new RegExp(prompt, 'u')); + assert.doesNotMatch(serializedError, new RegExp(apiKey, 'u')); + assert.equal(terminal.provider_error.detail, '[REDACTED]'); +}); + test('recursively redacts prompt, bearer, and API-key material from normal provider results', async () => { const root = await mkdtemp(path.join(tmpdir(), 'co-engineer-cursor-result-redaction-')); const repo = path.join(root, 'repo'); diff --git a/plugins/codex-co-engineer/test/v3-durable-wait.test.mjs b/plugins/codex-co-engineer/test/v3-durable-wait.test.mjs index 8ebd8b9..de29a33 100644 --- a/plugins/codex-co-engineer/test/v3-durable-wait.test.mjs +++ b/plugins/codex-co-engineer/test/v3-durable-wait.test.mjs @@ -315,24 +315,37 @@ test('watcher failure uses a low-frequency fallback instead of model-driven poll }); const delays = []; const { watch, state } = createMockWatch(); - const pending = waitForTaskProgress(root, 'fallback-one', { + let clock = 0; + const timedOut = await waitForTaskProgress(root, 'fallback-one', { wait_until: 'terminal', - wait_ms: 80, - watch, + wait_ms: 5_000, + now: () => clock, + watch: (...args) => { + const watcher = watch(...args); + const on = watcher.on.bind(watcher); + watcher.on = (event, handler) => { + on(event, handler); + // Fail only after the handler is registered. Fixed sleeps could + // fire before filesystem setup completed under parallel test load. + if (event === 'error') queueMicrotask(() => handler(new Error('watch failed'))); + return watcher; + }; + return watcher; + }, delay: (milliseconds, signal) => { delays.push(milliseconds); + if (state.closed === 2) { + clock = 5_000; + return Promise.resolve('timeout'); + } return waitDelay(milliseconds, signal); }, fallback_ms: 1_000, }); - const firstErrorHandler = await waitForMockErrorHandler(state); - firstErrorHandler(new Error('watch failed')); - const secondErrorHandler = await waitForMockErrorHandler(state, firstErrorHandler); - secondErrorHandler(new Error('watch failed again')); - const timedOut = await pending; assert.equal(timedOut.progress.wait_reason, 'timeout'); - assert.equal(state.closed, state.opened); - assert.ok(state.opened >= 2); + assert.equal(state.closed, 2); + assert.equal(state.opened, 2); + assert.ok(delays.includes(1_000)); assert.equal(delays.filter((value) => value > 0 && value <= 20).length, 0); } finally { await rm(root, { recursive: true, force: true }); diff --git a/plugins/codex-co-engineer/test/v3-identity-digest-authority.test.mjs b/plugins/codex-co-engineer/test/v3-identity-digest-authority.test.mjs index fb8394a..9a82562 100644 --- a/plugins/codex-co-engineer/test/v3-identity-digest-authority.test.mjs +++ b/plugins/codex-co-engineer/test/v3-identity-digest-authority.test.mjs @@ -942,8 +942,9 @@ test('descriptors are deeply frozen, primitive-only, and detached', () => { test('manifest, prompt, envelope, and run-identity goldens keep their exact bytes', () => { const manifest = goldenManifest(); const envelope = compileChildEnvelopeV1(manifest, 'backend-writer'); - // Pinned against the accepted product-foundation identity bytes for this - // fixture; the authority refactor must not move one byte. + // Pinned against the current deterministic execution envelope. The task + // prompt digest remains stable while canonical provider guidance changes + // the exact dispatched envelope bytes and therefore its own digest. assert.deepEqual(runManifestDigestV1(manifest), { algorithm: 'sha256', domain: 'codex-co-engineer.identity.v1', @@ -973,8 +974,8 @@ test('manifest, prompt, envelope, and run-identity goldens keep their exact byte domain: 'codex-co-engineer.identity.v1', version: 1, label: 'child-envelope.v1', - input_bytes: 1519, - digest: '61359b14dcaf7dcf5015ea1242ecdbd49540f125a75942f5bd31527ee4b85f30', + input_bytes: 2188, + digest: 'fa0f7729b3f2f8295843caf6869cc81e222de04c8a390757645c76c158e74ecb', }); const identity = describeRunIdentityV1(manifest); assert.deepEqual(identity.assignment_prompt_digests.map((entry) => entry.digest), [ diff --git a/plugins/codex-co-engineer/test/v3-runtime-entrypoints.test.mjs b/plugins/codex-co-engineer/test/v3-runtime-entrypoints.test.mjs new file mode 100644 index 0000000..0159053 --- /dev/null +++ b/plugins/codex-co-engineer/test/v3-runtime-entrypoints.test.mjs @@ -0,0 +1,31 @@ +import assert from 'node:assert/strict'; +import path from 'node:path'; +import test from 'node:test'; + +import { assertRuntimeEntrypoints } from '../mcp/v3/runtime-entrypoints.mjs'; + +test('runtime preflight checks only the entrypoints needed by each provider path', async () => { + const local = []; + await assertRuntimeEntrypoints('grok', { inspectFile: async (file) => local.push(path.basename(file)) }); + assert.deepEqual(local.sort(), ['acp-worker.mjs', 'credential-handoff-loader.mjs']); + + const cloud = []; + await assertRuntimeEntrypoints('cursor-cloud', { inspectFile: async (file) => cloud.push(path.basename(file)) }); + assert.deepEqual(cloud, ['cursor-cloud-worker.mjs']); +}); + +test('runtime preflight returns one bounded error for a missing installed artifact', async () => { + await assert.rejects( + assertRuntimeEntrypoints('cursor-local', { + inspectFile: async (file) => { + throw Object.assign(new Error(`missing ${file}?token=secret`), { code: 'ENOENT' }); + }, + }), + (error) => { + assert.equal(error.code, 'runtime_install_incomplete'); + assert.equal(error.message, 'The installed Codex-Co-Engineer runtime is incomplete.'); + assert.doesNotMatch(error.message, /token|credential-handoff-loader/iu); + return true; + }, + ); +}); diff --git a/plugins/codex-co-engineer/test/v3-server.test.mjs b/plugins/codex-co-engineer/test/v3-server.test.mjs index f7cde94..11e18d1 100644 --- a/plugins/codex-co-engineer/test/v3-server.test.mjs +++ b/plugins/codex-co-engineer/test/v3-server.test.mjs @@ -97,12 +97,12 @@ test('advertises only the thin public tool surface', async () => { ]); assert.equal(values[0].result.serverInfo.name, 'codex-co-engineer'); assert.equal(values[0].result.serverInfo.title, 'Codex-Co-Engineer'); - assert.equal(values[0].result.serverInfo.version, '3.4.0'); + assert.equal(values[0].result.serverInfo.version, '3.4.2'); assert.deepEqual(values[1].result.tools.map((tool) => tool.name), ['status', 'delegate', 'task', 'tasks', 'cancel']); assert.equal(values[1].result.tools.length, 5); const statusTool = values[1].result.tools.find((tool) => tool.name === 'status'); assert.deepEqual(Object.keys(statusTool.inputSchema.properties), [ - 'detail', 'task_limit', 'include_tasks', 'response_mode', 'run_id', + 'detail', 'task_limit', 'include_tasks', 'refresh', 'response_mode', 'run_id', ]); const taskTool = values[1].result.tools.find((tool) => tool.name === 'task'); assert.deepEqual(Object.keys(taskTool.inputSchema.properties), [ @@ -114,7 +114,7 @@ test('advertises only the thin public tool surface', async () => { assert.equal(taskTool.inputSchema.properties.wait_until.enum[0], 'progress'); assert.equal(taskTool.inputSchema.properties.wait_until.enum[1], 'terminal'); assert.equal(taskTool.inputSchema.properties.wait_until.enum[2], 'decision_or_attention'); - assert.deepEqual(taskTool.inputSchema.properties.response_mode.enum, ['structured']); + assert.deepEqual(taskTool.inputSchema.properties.response_mode.enum, ['structured', 'legacy']); const tasksTool = values[1].result.tools.find((tool) => tool.name === 'tasks'); assert.deepEqual(Object.keys(tasksTool.inputSchema.properties), [ 'detail', 'limit', 'cursor', 'provider', 'state', 'status', 'response_mode', @@ -122,11 +122,17 @@ test('advertises only the thin public tool surface', async () => { 'run_id', ]); for (const tool of values[1].result.tools) { - assert.deepEqual(tool.inputSchema.properties.response_mode.enum, ['structured']); - assert.match(tool.description, /response_mode="structured"/u); + assert.deepEqual(tool.inputSchema.properties.response_mode.enum, ['structured', 'legacy']); + assert.match(tool.description, /response_mode="legacy"/u); } const delegateTool = values[1].result.tools.find((tool) => tool.name === 'delegate'); assert.match(delegateTool.description, /property named repo/u); + assert.deepEqual(delegateTool.inputSchema.allOf[0].if.anyOf, [ + { required: ['run'] }, { required: ['run_request'] }, + ]); + assert.deepEqual(delegateTool.inputSchema.allOf[0].then.oneOf, [ + { required: ['run'] }, { required: ['run_request'] }, + ]); assert.deepEqual(delegateTool.inputSchema.allOf[0].else.required, ['task_id', 'provider', 'repo', 'prompt']); assert.match(delegateTool.inputSchema.properties.repo.description, /Required property named repo/u); assert.match(delegateTool.inputSchema.properties.repo.description, /\/absolute\/path\/to\/git-worktree/u); @@ -134,7 +140,7 @@ test('advertises only the thin public tool surface', async () => { assert.match(delegateTool.inputSchema.properties.starting_ref.description, /Cursor Cloud only/u); assert.match(delegateTool.inputSchema.properties.starting_ref.description, /does not replace the required repo/u); assert.deepEqual(delegateTool.inputSchema.properties.dsh_model.enum, [ - 'muse-spark-1.2-contributor', + 'meta/muse-spark-1.3-contributor', 'stealth/ox-alpha', ]); assert.match(delegateTool.inputSchema.properties.dsh_model.description, /DSH only/u); @@ -149,6 +155,12 @@ test('advertises only the thin public tool surface', async () => { assert.ok(Object.hasOwn(delegateTool.inputSchema.properties, 'run')); assert.equal(delegateTool.inputSchema.properties.run.properties.assignments.minItems, 1); assert.equal(delegateTool.inputSchema.properties.run.properties.assignments.maxItems, 8); + const runRequestAssignment = delegateTool.inputSchema.properties.run_request.properties.assignments.items; + assert.equal(runRequestAssignment.required.includes('role'), true); + assert.equal(runRequestAssignment.required.includes('access'), false); + assert.equal(runRequestAssignment.required.includes('expected_duration_ms'), false); + assert.equal(runRequestAssignment.properties.expected_duration_ms.default, 600000); + assert.match(runRequestAssignment.properties.access.description, /derived from role/u); assert.match(taskTool.description, /event_cursor/u); assert.match(taskTool.description, /Unsolicited stdio callbacks/u); assert.match(taskTool.description, /view=compact/u); @@ -510,7 +522,7 @@ test('live MCP tool results use structured-first text fallback when response_mod const toolsList = await request({ jsonrpc: '2.0', id: 39, method: 'tools/list' }); for (const tool of toolsList.result.tools) { assert.equal(tool.inputSchema.properties.response_mode.enum[0], 'structured'); - assert.match(tool.description, /response_mode="structured"/u); + assert.match(tool.description, /response_mode="legacy"/u); } for (const [index, call] of [ { name: 'status', arguments: { response_mode: 'structured' } }, @@ -550,6 +562,66 @@ test('live MCP tool results use structured-first text fallback when response_mod }); }); +test('structured-capable clients receive bounded transport by default', async () => { + await withServer(async ({ request }) => { + const initialize = await request({ + jsonrpc: '2.0', + id: 58, + method: 'initialize', + params: { + protocolVersion: '2025-11-25', + capabilities: { + structuredContent: true, + extensions: { + 'io.modelcontextprotocol/ui': { + mimeTypes: ['text/html;profile=mcp-app'], + }, + }, + }, + }, + }); + assert.equal(initialize.result.protocolVersion, '2025-11-25'); + const response = await request({ + jsonrpc: '2.0', + id: 59, + method: 'tools/call', + params: { name: 'status', arguments: {} }, + }); + assert.equal(response.result.content[0].type, 'text'); + assert.notEqual(response.result.content[0].text, JSON.stringify(response.result.structuredContent)); + assert.equal(JSON.parse(response.result.content[0].text).authoritative, 'structuredContent'); + }); +}); + +test('native run transport is structured-first when the host omits capability advertisement', async () => { + await withServer(async ({ request }) => { + await request({ + jsonrpc: '2.0', id: 590, method: 'initialize', + params: { protocolVersion: '2025-11-25', capabilities: {} }, + }); + const response = await request({ + jsonrpc: '2.0', id: 591, method: 'tools/call', + params: { name: 'status', arguments: { run_id: 'missing-semantic-run' } }, + }); + assert.equal(response.result.content[0].type, 'text'); + assert.notEqual(response.result.content[0].text, JSON.stringify(response.result.structuredContent)); + assert.equal(JSON.parse(response.result.content[0].text).authoritative, 'structuredContent'); + assert.equal(response.result.structuredContent.error.code, 'runtime_run_unknown'); + + const textOnly = await request({ + jsonrpc: '2.0', id: 592, method: 'tools/call', + params: { + name: 'status', + arguments: { run_id: 'missing-semantic-run', response_mode: 'legacy' }, + }, + }); + assert.equal( + textOnly.result.content[0].text, + JSON.stringify(textOnly.result.structuredContent), + ); + }); +}); + test('tasks list, paged full/compact, and wait-any cannot project a completed PING-timeout as succeeded', async () => { await withServer(async ({ state, request }) => { await createTask({ diff --git a/plugins/codex-co-engineer/test/v3-supervisor.test.mjs b/plugins/codex-co-engineer/test/v3-supervisor.test.mjs index d8899e6..c16a309 100644 --- a/plugins/codex-co-engineer/test/v3-supervisor.test.mjs +++ b/plugins/codex-co-engineer/test/v3-supervisor.test.mjs @@ -9,12 +9,14 @@ import { promisify } from 'node:util'; import { cancelTask, + classifyGrokReadinessOutput, cleanupLocalTaskLifecycle, cleanupManagedWorkspace, createSupervisorRunToolAdapter, createWriterWorkspace, invokeRunTool, launchWorker, + probeGrokReadiness, settleLocalTaskLifecycle, submitTask, supervisorStatus, @@ -31,6 +33,7 @@ import { import { createClock, createLifecycleFns } from './fixtures/r1-run-runtime-fixtures.mjs'; import { appendTaskEvent, createLaunchReservation, createTask, readRuntimeRecord, readTask, updateTask, writeRuntimeRecord } from '../mcp/v3/task-store.mjs'; import { runCursorCloudTask } from '../mcp/v3/cursor-cloud-worker.mjs'; +import { BUNDLED_WORKTREE_BOOTSTRAP } from '../mcp/v3/worktree-bootstrap-runtime.mjs'; const SHA = 'a'.repeat(40); const run = promisify(execFile); @@ -41,6 +44,44 @@ const readyBoundary = async () => ({ boundary: 'systemd-user-service-cgroup', }); +test('Grok readiness accepts explicit login despite ancillary unauthorized settings stderr', () => { + assert.deepEqual(classifyGrokReadinessOutput( + 'You are logged in with grok.com.\n\nDefault model: grok-4.6\n', + 'ERROR Settings fetch failed: 401 Unauthorized\n', + ), { ready: true }); +}); + +test('Grok readiness treats an explicit logout as authoritative over a stale login marker', () => { + assert.deepEqual(classifyGrokReadinessOutput( + 'You are logged in with grok.com.', + 'You are not logged in. Run `grok login` to continue.\n', + ), { ready: false, reason: 'needs_login' }); + assert.deepEqual(classifyGrokReadinessOutput( + '', + 'Models request failed: 401 Unauthorized\n', + ), { ready: false, reason: 'needs_login' }); +}); + +test('Grok readiness keeps execution failures distinct from authentication failures', async () => { + const executeWith = (code) => async () => { + throw Object.assign(new Error('synthetic probe failure'), { code }); + }; + const missing = await probeGrokReadiness('grok', {}, { execute: executeWith('ENOENT') }); + assert.deepEqual({ ...missing, probe_duration_ms: 0 }, { + installed: false, + ready: false, + reason: 'not_installed', + probe_duration_ms: 0, + }); + const timedOut = await probeGrokReadiness('grok', {}, { execute: executeWith('ETIMEDOUT') }); + assert.deepEqual({ ...timedOut, probe_duration_ms: 0 }, { + installed: true, + ready: false, + reason: 'probe_failed', + probe_duration_ms: 0, + }); +}); + test('writer workspace parses noisy pretty JSON and requests a bounded large buffer', async () => { const calls = []; const result = await createWriterWorkspace({ @@ -64,13 +105,65 @@ test('writer workspace parses noisy pretty JSON and requests a bounded large buf }, checkPath: async () => ({ isDirectory: () => true }), }); - assert.equal(calls[0][0], 'worktree-bootstrap'); + assert.equal(calls[0][0], BUNDLED_WORKTREE_BOOTSTRAP); assert.deepEqual(calls[0][1], ['create', 'parallel-one', '--repo', '/repo', '--base', 'feature']); assert.ok(calls[0][2].maxBuffer >= 16 * 1024 * 1024); assert.equal(result.worktree_path, '/worktrees/parallel-one'); assert.equal(result.branch, 'codex/parallel-one'); }); +test('exact local SHA workspace creation does not require an upstream or source branch', async () => { + const calls = []; + const result = await createWriterWorkspace({ + taskId: 'local-sha', + repo: '/repo', + baseSha: SHA, + execute: async (command, args, options) => { + calls.push([command, args, options]); + if (command === 'git') { + if (args.includes('--show-current')) return { stdout: args[1] === '/repo' ? '\n' : 'codex/local-sha\n' }; + if (args.includes('--porcelain=v1')) return { stdout: '\n' }; + if (args.includes('--show-toplevel')) return { stdout: '/worktrees/local-sha\n' }; + if (args.includes('--verify') || args.includes('HEAD')) return { stdout: `${SHA}\n` }; + return { stdout: '\n' }; + } + return { stdout: JSON.stringify({ + task: 'local-sha', + branch: 'codex/local-sha', + start_sha: SHA, + worktree_path: '/worktrees/local-sha', + status: 'ready', + }) }; + }, + checkPath: async () => ({ isDirectory: () => true }), + }); + + assert.deepEqual(calls.find(([command]) => command === BUNDLED_WORKTREE_BOOTSTRAP)?.[1], [ + 'create', 'local-sha', '--repo', '/repo', '--base', SHA, '--local-only', + ]); + assert.equal(result.start_sha, SHA); + assert.equal(result.branch, 'codex/local-sha'); +}); + +test('exact local SHA reports an actionable capability error when bootstrap is too old', async () => { + await assert.rejects( + createWriterWorkspace({ + taskId: 'local-sha-old-bootstrap', + repo: '/repo', + baseSha: SHA, + execute: async (command, args) => { + if (command === 'git') { + if (args.includes('--show-current')) return { stdout: '\n' }; + if (args.includes('--porcelain=v1')) return { stdout: '\n' }; + if (args.includes('--verify')) return { stdout: `${SHA}\n` }; + } + throw Object.assign(new Error('unknown option --local-only'), { stderr: 'unknown option --local-only' }); + }, + }), + (error) => error.code === 'worktree_bootstrap_exact_sha_unsupported', + ); +}); + test('invalid worktree receipt fails before dispatch', async () => { const execute = async (command) => command === 'git' ? { stdout: 'feature\n' } : { stdout: '{}' }; await assert.rejects( @@ -79,6 +172,47 @@ test('invalid worktree receipt fails before dispatch', async () => { ); }); +test('incomplete installed runtime fails before local provisioning or cloud dispatch', async () => { + for (const provider of ['grok', 'cursor-cloud']) { + const root = await mkdtemp(path.join(os.tmpdir(), `co-engineer-runtime-preflight-${provider}-`)); + const calls = []; + try { + await assert.rejects( + submitTask({ + task_id: `runtime-preflight-${provider}`, + provider, + repo: '/repo', + prompt: 'PRIVATE_PROMPT', + expected_duration_ms: 10_000, + }, { + root, + preflightRuntime: async (selectedProvider) => { + calls.push(['runtime', selectedProvider]); + throw Object.assign(new Error('/deleted/cache/credential-handoff-loader.mjs'), { code: 'ENOENT' }); + }, + probeBoundary: async () => { calls.push(['boundary']); return readyBoundary(); }, + createWorkspace: async () => { calls.push(['workspace']); throw new Error('must not provision'); }, + execute: async () => { calls.push(['git']); throw new Error('must not inspect repository'); }, + launch: async () => { calls.push(['launch']); throw new Error('must not launch'); }, + }), + (error) => { + assert.equal(error.code, 'runtime_install_incomplete'); + assert.equal(error.message, 'The installed Codex-Co-Engineer runtime is incomplete. Reinstall the plugin, then restart Codex.'); + assert.doesNotMatch(error.message, /deleted|cache|credential/iu); + return true; + }, + ); + assert.deepEqual(calls, [['runtime', provider]]); + await assert.rejects( + readTask(root, `runtime-preflight-${provider}`), + (error) => error.code === 'ENOENT', + ); + } finally { + await rm(root, { recursive: true, force: true }); + } + } +}); + test('managed delegation rejects a missing or invalid workspace before provider launch', async () => { const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-supervisor-workspace-contract-')); const launches = []; @@ -293,7 +427,7 @@ test('direct local mode uses the caller worktree and does not invoke bootstrap', assert.equal(value.task.branch, 'feature'); assert.equal(value.task.start_sha, SHA); assert.equal(launches[0].writer, false); - assert.equal(calls.some(([command]) => command === 'worktree-bootstrap'), false); + assert.equal(calls.some(([command]) => command === BUNDLED_WORKTREE_BOOTSTRAP), false); } finally { await rm(root, { recursive: true, force: true }); } @@ -388,7 +522,7 @@ test('DSH omission keeps the Muse config and credential as the stored default', const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-supervisor-dsh-muse-')); const repo = path.join(root, 'repo'); const museConfig = path.join(root, 'dsh-acp.yml'); - const museKey = path.join(root, 'model-api-key'); + const museKey = path.join(root, 'openrouter-api-key'); let launched; try { await mkdir(repo); @@ -411,7 +545,7 @@ test('DSH omission keeps the Muse config and credential as the stored default', root, env: { CODEX_CO_ENGINEER_DSH_ACP_CONFIG: museConfig, - CODEX_CO_ENGINEER_MODEL_API_KEY_FILE: museKey, + CODEX_CO_ENGINEER_OPENROUTER_API_KEY_FILE: museKey, }, execute, probeBoundary: readyBoundary, @@ -420,10 +554,10 @@ test('DSH omission keeps the Muse config and credential as the stored default', return { pid: 9003, process_group: 9003, process_start_ticks: '3' }; }, }); - assert.equal(value.task.dsh_model, 'muse-spark-1.2-contributor'); + assert.equal(value.task.dsh_model, 'meta/muse-spark-1.3-contributor'); assert.deepEqual(value.task.agent_argv, ['dsh-acp-demo', '--config', museConfig]); - assert.equal(launched.env.MODEL_API_KEY, 'test-muse-value'); - assert.equal(launched.env.OPENROUTER_API_KEY, undefined); + assert.equal(launched.env.OPENROUTER_API_KEY, 'test-muse-value'); + assert.equal(launched.env.MODEL_API_KEY, undefined); } finally { await rm(root, { recursive: true, force: true }); } @@ -464,9 +598,9 @@ test('status makes boundary health explicit and fails only local providers close installed: true, ready: true, transport: 'acpx', - default_model: 'muse-spark-1.2-contributor', + default_model: 'meta/muse-spark-1.3-contributor', model_options: { - 'muse-spark-1.2-contributor': { ready: true }, + 'meta/muse-spark-1.3-contributor': { ready: true }, 'stealth/ox-alpha': { ready: true }, }, }, @@ -543,7 +677,7 @@ test('managed launch failure marks the task failed and cleans an abandoned write assert.equal(task.error.code, 'worker_failed'); assert.doesNotMatch(task.error.message, /private|secret/iu); assert.deepEqual(calls.at(-1), [ - 'worktree-bootstrap', + BUNDLED_WORKTREE_BOOTSTRAP, ['lock', 'clean', 'launch-fail', '--repo', worktreePath, '--policy', 'dead-local', '--lock-id', 'dead-lock'], ]); } finally { @@ -895,6 +1029,52 @@ test('exports identity-bound local lifecycle settlement without rewriting stored } }); +for (const [label, cleanup, expectedDrain] of [ + ['grants the first grace after worker cleanup', { status: 'pending', acp_close: 'closed', wtb_handoff: 'recorded' }, 123], + ['skips repeated grace after supervisor cleanup', { status: 'unknown', boundary: 'unknown', lock: 'not_applicable' }, 0], +]) { +test(`reconciliation ${label}`, async () => { + const root = await mkdtemp(path.join(os.tmpdir(), 'co-engineer-supervisor-repeat-drain-')); + const boundary = { + version: 1, + boundary: 'systemd-user-service-cgroup', + unit: 'codex-co-engineer-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa.service', + description: 'codex-co-engineer-task:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa', + invocation_id: 'bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb', + control_group: '/user.slice/user-1000.slice/user@1000.service/app.slice/codex-co-engineer-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa.service', + }; + let slept = 0; + try { + await createTask({ + root, + prompt: 'reconcile retained receipt', + record: { + id: 'repeat-drain', + status: 'completed', + provider: 'grok', + cwd: root, + workspace_kind: 'direct', + cleanup, + }, + }); + const task = (await readTask(root, 'repeat-drain')).task; + const settled = await settleLocalTaskLifecycle(root, task, { + process_boundary: boundary, + }, { + drainGraceMs: 123, + sleep: async (milliseconds) => { slept += milliseconds; }, + inspectBoundary: async () => ({ state: 'inactive_empty', empty: true, stop_allowed: false }), + }); + + assert.equal(slept, expectedDrain); + assert.equal(settled.final, true); + assert.equal(settled.boundary, 'inactive_empty'); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); +} + test('default run seams are durable P33/P34 authorities and cancel confirms', async () => { const source = await readFile(new URL('../mcp/v3/supervisor.mjs', import.meta.url), 'utf8'); assert.match(source, /createDurableRunSeams/u); diff --git a/plugins/codex-co-engineer/test/v3-task-store.test.mjs b/plugins/codex-co-engineer/test/v3-task-store.test.mjs index 4fd57b2..8b80695 100644 --- a/plugins/codex-co-engineer/test/v3-task-store.test.mjs +++ b/plugins/codex-co-engineer/test/v3-task-store.test.mjs @@ -694,17 +694,30 @@ test('explicit cursorless progress wait starts from the current event-log tail', } let settled = false; + let clock = 0; + const waiting = Promise.withResolvers(); const { watch, state } = createMockWatch(); const pending = waitForTaskProgress(root, 'wait-cursorless', { wait_until: 'progress', wait_ms: 1_000, watch, + now: () => clock, + delay: (milliseconds, signal) => { + waiting.resolve(); + return waitDelay(milliseconds, signal); + }, }).then((value) => { settled = true; return value; }); - await new Promise((resolve) => setTimeout(resolve, 40)); + // Wait until the implementation has baselined historical events and + // entered its wait, rather than racing filesystem setup against a timer. + await Promise.race([ + waiting.promise, + pending.then(() => assert.fail('historical events must not settle the wait')), + ]); assert.equal(settled, false); + clock = 40; await appendTaskEvent(root, 'wait-cursorless', { type: 'provider', @@ -714,8 +727,7 @@ test('explicit cursorless progress wait starts from the current event-log tail', const woke = await pending; assert.equal(woke.progress.wait_reason, 'progress'); assert.equal(woke.progress.last_event.text, 'fresh'); - assert.ok(woke.progress.waited_ms >= 40); - assert.ok(woke.progress.waited_ms < 400); + assert.equal(woke.progress.waited_ms, 40); assert.equal(state.closed, state.opened); const timedOut = await waitForTaskProgress(root, 'wait-cursorless', { diff --git a/plugins/codex-co-engineer/vendor/worktree-bootstrap/LICENSE b/plugins/codex-co-engineer/vendor/worktree-bootstrap/LICENSE new file mode 100644 index 0000000..161b80b --- /dev/null +++ b/plugins/codex-co-engineer/vendor/worktree-bootstrap/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Plumbob + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/plugins/codex-co-engineer/vendor/worktree-bootstrap/PROVENANCE.json b/plugins/codex-co-engineer/vendor/worktree-bootstrap/PROVENANCE.json new file mode 100644 index 0000000..e76664f --- /dev/null +++ b/plugins/codex-co-engineer/vendor/worktree-bootstrap/PROVENANCE.json @@ -0,0 +1,11 @@ +{ + "name": "worktree-bootstrap", + "version": "1.1.0", + "artifact": "worktree-bootstrap", + "artifact_sha256": "4138a49e42148db9e5c63dfd5386fdac235f19e445453e702f342b1a7002b813", + "runtime": "Python 3.11 or newer (standard library only)", + "license": "MIT", + "license_file": "LICENSE", + "source": "Maintainer-authored Worktree Bootstrap skill, bundled with the copyright holder's permission.", + "note": "The executable is preserved byte-for-byte from the qualified 1.1.0 source." +} diff --git a/plugins/codex-co-engineer/vendor/worktree-bootstrap/worktree-bootstrap b/plugins/codex-co-engineer/vendor/worktree-bootstrap/worktree-bootstrap new file mode 100755 index 0000000..0f729e4 --- /dev/null +++ b/plugins/codex-co-engineer/vendor/worktree-bootstrap/worktree-bootstrap @@ -0,0 +1,1131 @@ +#!/usr/bin/env python3 +"""Create, validate, lock, and report isolated Git writer worktrees.""" + +from __future__ import annotations + +import argparse +import datetime as dt +import fnmatch +import hashlib +import json +import os +import platform +import re +import shlex +import signal +import socket +import subprocess +import sys +import time +import tomllib +import uuid +from pathlib import Path +from typing import Any, Sequence + + +SCHEMA = "worktree-bootstrap/v1" +HANDOFF_SCHEMA = "worktree-bootstrap-handoff/v1" +VERSION = "1.1.0" +CONFIG_PATH = Path(".codex/worktree-bootstrap.toml") +TASK_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]{0,79}$") +EXACT_SHA_RE = re.compile(r"^[0-9a-fA-F]{40}$") + + +class ToolError(RuntimeError): + def __init__(self, message: str, *, exit_code: int = 3, details: Any = None): + super().__init__(message) + self.exit_code = exit_code + self.details = details + + +def utc_now() -> str: + return dt.datetime.now(dt.timezone.utc).isoformat().replace("+00:00", "Z") + + +def eprint(message: str) -> None: + print(message, file=sys.stderr, flush=True) + + +def run( + argv: Sequence[str], + *, + cwd: Path | None = None, + env: dict[str, str] | None = None, + check: bool = True, + text: bool = True, +) -> subprocess.CompletedProcess[Any]: + try: + completed = subprocess.run( + [str(value) for value in argv], + cwd=str(cwd) if cwd else None, + env=env, + check=False, + capture_output=True, + text=text, + ) + except FileNotFoundError as exc: + raise ToolError(f"required command not found: {argv[0]}", exit_code=4) from exc + if check and completed.returncode != 0: + stderr = completed.stderr.strip() if text else completed.stderr.decode(errors="replace").strip() + stdout = completed.stdout.strip() if text else completed.stdout.decode(errors="replace").strip() + detail = stderr or stdout or f"exit {completed.returncode}" + raise ToolError(f"command failed: {shlex.join(argv)}: {detail}", exit_code=4) + return completed + + +def git(repo: Path, *args: str, check: bool = True, text: bool = True) -> subprocess.CompletedProcess[Any]: + return run(["git", "-C", str(repo), *args], check=check, text=text) + + +def sha256_bytes(value: bytes) -> str: + return hashlib.sha256(value).hexdigest() + + +def implementation_sha256() -> str: + return sha256_bytes(Path(__file__).resolve().read_bytes()) + + +def json_bytes(value: Any) -> bytes: + return (json.dumps(value, indent=2, sort_keys=True) + "\n").encode() + + +def atomic_json(path: Path, value: Any, *, exclusive: bool = False) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + payload = json_bytes(value) + if exclusive: + descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(descriptor, "wb") as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + return + temporary = path.with_name(f".{path.name}.{os.getpid()}.{uuid.uuid4().hex}.tmp") + with temporary.open("wb") as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary, path) + + +def read_json(path: Path) -> dict[str, Any]: + try: + value = json.loads(path.read_text()) + except FileNotFoundError as exc: + raise ToolError(f"record not found: {path}") from exc + except json.JSONDecodeError as exc: + raise ToolError(f"invalid JSON record: {path}: {exc}") from exc + if not isinstance(value, dict): + raise ToolError(f"invalid record shape: {path}") + return value + + +def validate_task(task: str) -> str: + if not TASK_RE.fullmatch(task): + raise ToolError("task must match [A-Za-z0-9][A-Za-z0-9._-]{0,79}") + return task + + +def repo_root(candidate: str | Path) -> Path: + path = Path(candidate).expanduser().resolve() + completed = git(path, "rev-parse", "--show-toplevel", check=False) + if completed.returncode != 0: + raise ToolError(f"not a Git working tree: {path}") + return Path(completed.stdout.strip()).resolve(strict=True) + + +def common_git_dir(repo: Path) -> Path: + value = git(repo, "rev-parse", "--git-common-dir").stdout.strip() + path = Path(value) + if not path.is_absolute(): + path = repo / path + return path.resolve(strict=True) + + +def state_root() -> Path: + configured = os.environ.get("WORKTREE_BOOTSTRAP_STATE_DIR") + if configured: + return Path(configured).expanduser().resolve() + xdg = os.environ.get("XDG_STATE_HOME") + base = Path(xdg).expanduser() if xdg else Path.home() / ".local/state" + return (base / "worktree-bootstrap").resolve() + + +def repo_identity(repo: Path) -> tuple[str, Path]: + common = common_git_dir(repo) + digest = sha256_bytes(str(common).encode())[:16] + source_name = common.parent.name if common.name == ".git" else common.stem + name = re.sub(r"[^A-Za-z0-9._-]+", "-", source_name).strip("-") or "repo" + return f"{name}-{digest}", common + + +def task_dir(repo: Path, task: str) -> Path: + identity, _ = repo_identity(repo) + return state_root() / "repositories" / identity / "tasks" / validate_task(task) + + +def manifest_path(repo: Path, task: str) -> Path: + return task_dir(repo, task) / "manifest.json" + + +def lock_path(repo: Path, task: str) -> Path: + return task_dir(repo, task) / "writer.lock.json" + + +def load_config(repo: Path) -> tuple[dict[str, Any], str | None]: + path = repo / CONFIG_PATH + if not path.exists(): + return {}, None + raw = path.read_bytes() + try: + value = tomllib.loads(raw.decode()) + except (UnicodeDecodeError, tomllib.TOMLDecodeError) as exc: + raise ToolError(f"invalid {CONFIG_PATH}: {exc}") from exc + if value.get("version", 1) != 1: + raise ToolError(f"unsupported {CONFIG_PATH} version: {value.get('version')}") + return value, sha256_bytes(raw) + + +def worktree_entries(repo: Path) -> list[dict[str, str]]: + raw = git(repo, "worktree", "list", "--porcelain").stdout + entries: list[dict[str, str]] = [] + current: dict[str, str] = {} + for line in raw.splitlines() + [""]: + if not line: + if current: + entries.append(current) + current = {} + continue + key, _, value = line.partition(" ") + current[key] = value + return entries + + +def detect_base_branch(repo: Path, configured: str | None) -> str: + if configured: + return configured.removeprefix("refs/heads/") + symbolic = git(repo, "symbolic-ref", "--quiet", "--short", "refs/remotes/origin/HEAD", check=False) + if symbolic.returncode == 0 and symbolic.stdout.strip(): + return symbolic.stdout.strip().split("/", 1)[-1] + for candidate in ("main", "master"): + if git(repo, "show-ref", "--verify", "--quiet", f"refs/heads/{candidate}", check=False).returncode == 0: + return candidate + raise ToolError("cannot determine base branch; pass --base or set base_branch in the policy") + + +def locate_base_worktree(repo: Path, base: str) -> Path: + expected = f"refs/heads/{base}" + for entry in worktree_entries(repo): + if entry.get("branch") == expected: + return Path(entry["worktree"]).resolve(strict=True) + raise ToolError(f"base branch {base!r} is not attached to a working tree") + + +def require_clean(path: Path, label: str) -> None: + raw = git(path, "status", "--porcelain=v1", "-z", "--untracked-files=all", text=False).stdout + if raw: + preview = git(path, "status", "--short").stdout.strip() + raise ToolError( + f"{label} is dirty; refusing writer workspace creation\n{preview}" + ) + + +def upstream_for(repo: Path, base: str) -> str | None: + result = git(repo, "for-each-ref", "--format=%(upstream:short)", f"refs/heads/{base}") + value = result.stdout.strip() + return value or None + + +def synchronize_base(base_worktree: Path, base: str, *, fetch: bool) -> tuple[str, str | None, str]: + local_sha = git(base_worktree, "rev-parse", f"refs/heads/{base}^{{commit}}").stdout.strip() + upstream = upstream_for(base_worktree, base) + if upstream is None: + remotes = [ + value + for value in git(base_worktree, "remote").stdout.splitlines() + if value + ] + if remotes: + raise ToolError( + f"base branch {base!r} has configured remotes but no upstream; " + "configure its tracking branch before launching a writer" + ) + return local_sha, None, "local" + if not fetch: + raise ToolError("an upstream exists, so --no-fetch cannot establish a fresh base") + remote, upstream_branch = upstream.split("/", 1) + eprint(f"preflight: fetching {remote} for authoritative base comparison") + fetched = git( + base_worktree, + "fetch", + "--prune", + remote, + upstream_branch, + check=False, + ) + if fetched.returncode != 0: + detail = fetched.stderr.strip() or fetched.stdout.strip() + raise ToolError(f"cannot refresh {upstream}; base freshness is unknown: {detail}", exit_code=4) + upstream_sha = git(base_worktree, "rev-parse", f"{upstream}^{{commit}}").stdout.strip() + local_sha = git(base_worktree, "rev-parse", f"refs/heads/{base}^{{commit}}").stdout.strip() + counts = git(base_worktree, "rev-list", "--left-right", "--count", f"{local_sha}...{upstream_sha}").stdout.split() + ahead, behind = (int(counts[0]), int(counts[1])) + if ahead or behind: + if behind and not ahead: + condition = "stale (behind upstream)" + elif ahead and not behind: + condition = "diverged (local base is ahead of upstream)" + else: + condition = "diverged (both sides contain unique commits)" + raise ToolError( + f"base branch {base!r} is {condition}; local={local_sha} upstream={upstream_sha}" + ) + return local_sha, upstream_sha, "remote-tracking" + + +def resolve_local_exact_base(source: Path, base: str | None) -> str: + if not base: + raise ToolError("--local-only requires --base with a full 40-character commit SHA") + if not EXACT_SHA_RE.fullmatch(base): + raise ToolError("--local-only requires --base to be a full 40-character commit SHA") + resolved = git(source, "rev-parse", "--verify", f"{base}^{{commit}}", check=False) + if resolved.returncode != 0: + raise ToolError(f"exact local base is not an available commit: {base}") + commit = resolved.stdout.strip().lower() + if commit != base.lower(): + raise ToolError(f"exact local base resolved unexpectedly: requested={base} resolved={commit}") + return commit + + +def validate_branch(repo: Path, branch: str) -> None: + if git(repo, "check-ref-format", "--branch", branch, check=False).returncode != 0: + raise ToolError(f"invalid branch name: {branch}") + if git(repo, "show-ref", "--verify", "--quiet", f"refs/heads/{branch}", check=False).returncode == 0: + raise ToolError(f"branch already exists: {branch}") + + +def command_list(value: Any, label: str) -> list[str]: + if not isinstance(value, list) or not value or not all(isinstance(item, str) and item for item in value): + raise ToolError(f"{label} must be a non-empty TOML string array") + return list(value) + + +def bootstrap_plan(repo: Path, config: dict[str, Any]) -> list[dict[str, Any]]: + section = config.get("bootstrap", {}) + declared = section.get("commands") if isinstance(section, dict) else None + plan: list[dict[str, Any]] = [] + if declared is not None: + if not isinstance(declared, list): + raise ToolError("bootstrap.commands must be an array of command arrays") + for index, item in enumerate(declared): + argv = command_list(item, f"bootstrap.commands[{index}]") + plan.append({"argv": argv, "source": "declared"}) + else: + if (repo / "package-lock.json").exists() or (repo / "npm-shrinkwrap.json").exists(): + plan.append({"argv": ["npm", "ci"], "source": "package-lock"}) + elif (repo / "pnpm-lock.yaml").exists(): + plan.append({"argv": ["corepack", "pnpm", "install", "--frozen-lockfile"], "source": "pnpm-lock"}) + elif (repo / "yarn.lock").exists(): + plan.append({"argv": ["corepack", "yarn", "install", "--immutable"], "source": "yarn-lock"}) + elif (repo / "bun.lock").exists() or (repo / "bun.lockb").exists(): + plan.append({"argv": ["bun", "install", "--frozen-lockfile"], "source": "bun-lock"}) + if (repo / "uv.lock").exists() and (repo / "pyproject.toml").exists(): + plan.append({"argv": ["uv", "sync", "--locked"], "source": "uv-lock"}) + elif (repo / "poetry.lock").exists() and (repo / "pyproject.toml").exists(): + plan.append({"argv": ["poetry", "install", "--sync"], "source": "poetry-lock"}) + if (repo / "go.sum").exists() and (repo / "go.mod").exists(): + plan.append({"argv": ["go", "mod", "download"], "source": "go-sum"}) + if (repo / "Cargo.lock").exists() and (repo / "Cargo.toml").exists(): + plan.append({"argv": ["cargo", "fetch", "--locked"], "source": "cargo-lock"}) + for entry in plan: + argv = entry["argv"] + executable = Path(argv[0]).name + joined = " ".join(argv).lower() + if "node_modules" in joined and executable in {"cp", "rsync", "ln", "install"}: + raise ToolError("bootstrap policy may not copy or symlink node_modules") + if executable == "npm" and len(argv) > 1 and argv[1] == "install": + raise ToolError("use npm ci, not npm install, for locked bootstrap") + if executable == "uv" and argv[1:2] == ["sync"] and "--locked" not in argv: + raise ToolError("uv sync bootstrap must include --locked") + return plan + + +def node_module_roots(worktree: Path) -> list[Path]: + roots: list[Path] = [] + for current, directories, _files in os.walk(worktree, followlinks=False): + retained: list[str] = [] + for directory in directories: + candidate = Path(current) / directory + if directory == "node_modules": + roots.append(candidate) + elif directory != ".git": + retained.append(directory) + directories[:] = retained + return roots + + +def guard_node_modules(worktree: Path, *, allow_unmarked: bool, task: str) -> list[str]: + records: list[str] = [] + for root in node_module_roots(worktree): + if root.is_symlink(): + raise ToolError(f"node_modules symlink is forbidden: {root}") + resolved = root.resolve(strict=True) + try: + resolved.relative_to(worktree) + except ValueError as exc: + raise ToolError(f"node_modules resolves outside the worktree: {root} -> {resolved}") from exc + marker = root / ".worktree-bootstrap-owner.json" + if marker.exists(): + owner = read_json(marker) + expected = str(worktree) + if owner.get("worktree_path") != expected or owner.get("task") != task: + raise ToolError( + f"node_modules ownership mismatch (likely copied from another worktree): {root}" + ) + elif not allow_unmarked: + raise ToolError(f"unowned node_modules is forbidden in a writer worktree: {root}") + records.append(str(resolved)) + return records + + +def mark_node_modules(worktree: Path, task: str, start_sha: str) -> list[str]: + roots = guard_node_modules(worktree, allow_unmarked=True, task=task) + for value in roots: + marker = Path(value) / ".worktree-bootstrap-owner.json" + atomic_json(marker, { + "schema": SCHEMA, + "task": task, + "worktree_path": str(worktree), + "start_sha": start_sha, + "recorded_at": utc_now(), + }) + return roots + + +def execute_bootstrap(worktree: Path, plan: list[dict[str, Any]], task: str, start_sha: str) -> list[dict[str, Any]]: + guard_node_modules(worktree, allow_unmarked=False, task=task) + results: list[dict[str, Any]] = [] + for entry in plan: + argv = entry["argv"] + eprint(f"bootstrap: {shlex.join(argv)}") + started = time.monotonic() + try: + process = subprocess.Popen(argv, cwd=worktree) + except FileNotFoundError as exc: + raise ToolError(f"bootstrap command not found: {argv[0]}", exit_code=4) from exc + return_code = process.wait() + result = { + **entry, + "duration_seconds": round(time.monotonic() - started, 3), + "exit_code": return_code, + } + results.append(result) + if return_code != 0: + raise ToolError(f"bootstrap failed ({return_code}): {shlex.join(argv)}", exit_code=4, details=results) + mark_node_modules(worktree, task, start_sha) + return results + + +def default_worktree_path(base_worktree: Path, task: str) -> Path: + return base_worktree.parent / f".{base_worktree.name}-worktrees" / task + + +def create_command(args: argparse.Namespace) -> int: + source = repo_root(args.repo) + config, config_digest = load_config(source) + if args.local_only: + base_worktree = source + require_clean(base_worktree, f"source worktree {base_worktree}") + start_sha = resolve_local_exact_base(base_worktree, args.base) + upstream_sha = None + authority = "local-exact-sha" + base_branch = None + else: + base_branch = detect_base_branch(source, args.base or config.get("base_branch")) + base_worktree = locate_base_worktree(source, base_branch) + require_clean(base_worktree, f"base worktree {base_worktree}") + start_sha, upstream_sha, authority = synchronize_base( + base_worktree, base_branch, fetch=not args.no_fetch + ) + task = validate_task(args.task) + branch = args.branch or f"codex/{task}" + validate_branch(base_worktree, branch) + if manifest_path(base_worktree, task).exists(): + raise ToolError(f"task already has a manifest: {task}") + + configured_root = config.get("worktree_root") + if args.path: + requested_path = Path(args.path).expanduser() + elif configured_root: + root_value = str(configured_root).format(repo=base_worktree.name, task=task) + requested_path = Path(root_value).expanduser() + if not requested_path.is_absolute(): + requested_path = base_worktree / requested_path + if "{task}" not in str(configured_root): + requested_path /= task + else: + requested_path = default_worktree_path(base_worktree, task) + requested_path = requested_path.absolute() + resolved_target = requested_path.resolve(strict=False) + for entry in worktree_entries(base_worktree): + registered = Path(entry["worktree"]).resolve(strict=True) + try: + resolved_target.relative_to(registered) + except ValueError: + pass + else: + raise ToolError( + f"worktree target may not be nested inside existing checkout " + f"{registered}: {resolved_target}" + ) + if requested_path.exists() or requested_path.is_symlink(): + raise ToolError(f"worktree path already exists: {requested_path}") + requested_path.parent.mkdir(parents=True, exist_ok=True) + + eprint(f"create: {branch} at {requested_path} from {start_sha}") + git(base_worktree, "worktree", "add", "-b", branch, str(requested_path), start_sha) + actual_path = requested_path.resolve(strict=True) + actual_root = repo_root(actual_path) + if actual_root != actual_path: + raise ToolError(f"Git reported a different worktree root: {actual_root}") + identity, common = repo_identity(actual_path) + origin = git(actual_path, "remote", "get-url", "origin", check=False).stdout.strip() or None + plan = bootstrap_plan(actual_path, config) + record: dict[str, Any] = { + "schema": SCHEMA, + "tool": { + "version": VERSION, + "implementation_sha256": implementation_sha256(), + }, + "status": "bootstrapping", + "task": task, + "branch": branch, + "base_branch": base_branch, + "start_sha": start_sha, + "upstream_sha": upstream_sha, + "base_authority": authority, + "source_worktree_path": str(base_worktree), + "worktree_path": str(actual_path), + "requested_worktree_path": str(requested_path), + "common_git_dir": str(common), + "repository_identity": identity, + "origin_url": origin, + "policy_path": str(CONFIG_PATH) if config_digest else None, + "policy_sha256": config_digest, + "bootstrap_plan": plan, + "bootstrap_results": [], + "created_at": utc_now(), + "updated_at": utc_now(), + } + path = manifest_path(actual_path, task) + atomic_json(path, record, exclusive=True) + try: + record["bootstrap_results"] = execute_bootstrap(actual_path, plan, task, start_sha) + record["status"] = "ready" + except ToolError as exc: + record["status"] = "bootstrap_failed" + record["failure"] = str(exc) + if exc.details: + record["bootstrap_results"] = exc.details + record["updated_at"] = utc_now() + atomic_json(path, record) + raise + record["updated_at"] = utc_now() + atomic_json(path, record) + print(json.dumps({ + "task": task, + "branch": branch, + "start_sha": start_sha, + "worktree_path": str(actual_path), + "manifest_path": str(path), + "status": "ready", + }, indent=2)) + return 0 + + +def load_manifest(repo: Path, task: str) -> tuple[dict[str, Any], Path]: + path = manifest_path(repo, task) + value = read_json(path) + if value.get("schema") != SCHEMA or value.get("task") != task: + raise ToolError(f"manifest identity mismatch: {path}") + return value, path + + +def proc_start_ticks(pid: int) -> str | None: + try: + fields = Path(f"/proc/{pid}/stat").read_text().split() + except (FileNotFoundError, PermissionError, ProcessLookupError): + return None + return fields[21] if len(fields) > 21 else None + + +def process_exists(pid: int) -> bool: + if pid <= 0: + return False + try: + os.kill(pid, 0) + except ProcessLookupError: + return False + except PermissionError: + return True + return True + + +def boot_id() -> str | None: + try: + return Path("/proc/sys/kernel/random/boot_id").read_text().strip() + except (FileNotFoundError, PermissionError): + return None + + +def lock_health(lock: dict[str, Any]) -> dict[str, Any]: + local_host = socket.gethostname() + result: dict[str, Any] = {"state": "active", "abandoned": False, "reason": None} + if lock.get("hostname") != local_host: + result.update(state="remote_or_unknown", reason="lock belongs to another host") + return result + if lock.get("boot_id") and lock.get("boot_id") != boot_id(): + result.update(state="abandoned", abandoned=True, reason="host rebooted since lock acquisition") + return result + pid = int(lock.get("wrapper_pid", 0)) + observed = proc_start_ticks(pid) + if not process_exists(pid): + result.update(state="abandoned", abandoned=True, reason="wrapper process no longer exists") + elif ( + observed is not None + and lock.get("process_start_ticks") + and observed != str(lock.get("process_start_ticks")) + ): + result.update(state="abandoned", abandoned=True, reason="PID was reused") + return result + + +def acquire_lock(repo: Path, manifest: dict[str, Any], command: list[str]) -> tuple[dict[str, Any], Path]: + path = lock_path(repo, manifest["task"]) + if path.exists(): + existing = read_json(path) + health = lock_health(existing) + raise ToolError( + f"writer lock already exists ({health['state']}): {path}; inspect it and use explicit lock clean policy if abandoned", + exit_code=5, + ) + lock = { + "schema": SCHEMA, + "lock_id": uuid.uuid4().hex, + "token": uuid.uuid4().hex, + "task": manifest["task"], + "branch": manifest["branch"], + "worktree_path": manifest["worktree_path"], + "start_sha": manifest["start_sha"], + "hostname": socket.gethostname(), + "platform": platform.platform(), + "boot_id": boot_id(), + "wrapper_pid": os.getpid(), + "process_start_ticks": proc_start_ticks(os.getpid()), + "command": command, + "acquired_at": utc_now(), + "heartbeat_at": utc_now(), + } + try: + atomic_json(path, lock, exclusive=True) + except FileExistsError as exc: + raise ToolError(f"writer lock raced with another writer: {path}", exit_code=5) from exc + return lock, path + + +def refresh_lock(path: Path, lock: dict[str, Any]) -> None: + current = read_json(path) + if current.get("lock_id") != lock.get("lock_id"): + raise ToolError("writer lock identity changed while command was running", exit_code=5) + lock["heartbeat_at"] = utc_now() + atomic_json(path, lock) + + +def release_lock(path: Path, lock: dict[str, Any]) -> None: + if not path.exists(): + return + current = read_json(path) + if current.get("lock_id") != lock.get("lock_id"): + raise ToolError("refusing to release a replacement writer lock", exit_code=5) + path.unlink() + + +def verify_workspace( + repo: Path, + manifest: dict[str, Any], + *, + require_writer: bool, + writer_token: str | None = None, +) -> dict[str, Any]: + actual_repo = repo_root(repo) + expected_repo = Path(manifest["worktree_path"]).resolve(strict=True) + errors: list[str] = [] + if manifest.get("status") != "ready": + errors.append(f"workspace manifest is not ready: {manifest.get('status')}") + if actual_repo != expected_repo: + errors.append(f"repository path mismatch: expected {expected_repo}, got {actual_repo}") + actual_common = common_git_dir(actual_repo) + expected_common = Path(manifest["common_git_dir"]).resolve(strict=True) + if actual_common != expected_common: + errors.append(f"common Git directory mismatch: expected {expected_common}, got {actual_common}") + branch = git(actual_repo, "symbolic-ref", "--quiet", "--short", "HEAD", check=False).stdout.strip() + if branch != manifest["branch"]: + errors.append(f"branch mismatch: expected {manifest['branch']}, got {branch or 'detached HEAD'}") + merge_base = git(actual_repo, "merge-base", manifest["start_sha"], "HEAD", check=False) + if merge_base.returncode != 0 or merge_base.stdout.strip() != manifest["start_sha"]: + errors.append("recorded starting SHA is no longer an ancestor of HEAD") + try: + modules = guard_node_modules( + actual_repo, + allow_unmarked=False, + task=manifest["task"], + ) + except ToolError as exc: + errors.append(str(exc)) + modules = [] + lock_summary: dict[str, Any] | None = None + if require_writer: + path = lock_path(actual_repo, manifest["task"]) + if not path.exists(): + errors.append("writer verification requested but no writer lock exists") + else: + lock = read_json(path) + health = lock_health(lock) + token = writer_token or os.environ.get("WORKTREE_BOOTSTRAP_WRITER_TOKEN") + if not token or token != lock.get("token"): + errors.append("writer token does not match the active lock") + if health["state"] != "active": + errors.append(f"writer lock is not active: {health['reason']}") + lock_summary = {"lock_id": lock.get("lock_id"), **health} + result = { + "ok": not errors, + "task": manifest["task"], + "repository_path": str(actual_repo), + "branch": branch, + "start_sha": manifest["start_sha"], + "node_modules_roots": modules, + "writer_lock": lock_summary, + "errors": errors, + "verified_at": utc_now(), + } + if errors: + raise ToolError("workspace verification failed: " + "; ".join(errors), details=result) + return result + + +def verify_command(args: argparse.Namespace) -> int: + repo = repo_root(args.repo) + manifest, _ = load_manifest(repo, validate_task(args.task)) + print(json.dumps(verify_workspace(repo, manifest, require_writer=args.require_writer), indent=2)) + return 0 + + +def launch_command(args: argparse.Namespace) -> int: + if not args.command: + raise ToolError("launch requires a command after --", exit_code=2) + command = args.command[1:] if args.command[:1] == ["--"] else args.command + repo = repo_root(args.repo) + task = validate_task(args.task) + manifest, _ = load_manifest(repo, task) + if manifest.get("status") != "ready": + raise ToolError(f"task workspace is not ready: {manifest.get('status')}") + verify_workspace(repo, manifest, require_writer=False) + lock, path = acquire_lock(repo, manifest, command) + environment = os.environ.copy() + environment.update({ + "WORKTREE_BOOTSTRAP_TASK": task, + "WORKTREE_BOOTSTRAP_BRANCH": manifest["branch"], + "WORKTREE_BOOTSTRAP_START_SHA": manifest["start_sha"], + "WORKTREE_BOOTSTRAP_MANIFEST": str(manifest_path(repo, task)), + "WORKTREE_BOOTSTRAP_WRITER_TOKEN": lock["token"], + }) + child: subprocess.Popen[Any] | None = None + exit_code = 1 + try: + verify_workspace( + repo, + manifest, + require_writer=True, + writer_token=lock["token"], + ) + eprint(f"launch: lock {lock['lock_id']} acquired for {shlex.join(command)}") + child = subprocess.Popen(command, cwd=repo, env=environment) + while True: + try: + exit_code = child.wait(timeout=max(1, args.heartbeat_seconds)) + break + except subprocess.TimeoutExpired: + refresh_lock(path, lock) + except FileNotFoundError as exc: + raise ToolError(f"writer command not found: {command[0]}", exit_code=4) from exc + except KeyboardInterrupt: + if child and child.poll() is None: + child.send_signal(signal.SIGINT) + exit_code = child.wait() + finally: + release_lock(path, lock) + eprint(f"launch: lock {lock['lock_id']} released") + return exit_code + + +def lock_inspect_command(args: argparse.Namespace) -> int: + repo = repo_root(args.repo) + path = lock_path(repo, validate_task(args.task)) + if not path.exists(): + print(json.dumps({ + "task": args.task, + "state": "unlocked", + "path": str(path), + "inspected_at": utc_now(), + }, indent=2, sort_keys=True)) + return 0 + lock = read_json(path) + safe = {key: value for key, value in lock.items() if key != "token"} + safe["health"] = lock_health(lock) + safe["path"] = str(path) + print(json.dumps(safe, indent=2, sort_keys=True)) + return 0 + + +def lock_clean_command(args: argparse.Namespace) -> int: + repo = repo_root(args.repo) + task = validate_task(args.task) + path = lock_path(repo, task) + lock = read_json(path) + if lock.get("lock_id") != args.lock_id: + raise ToolError("lock ID changed; inspect the current lock before cleaning", exit_code=5) + health = lock_health(lock) + if args.policy == "dead-local": + if not health["abandoned"]: + raise ToolError(f"dead-local policy does not apply: {health['reason'] or health['state']}", exit_code=5) + elif args.policy == "expired-heartbeat": + config, _ = load_config(repo) + locks = config.get("locks", {}) if isinstance(config.get("locks", {}), dict) else {} + if not locks.get("allow_expired_heartbeat_cleanup", False): + raise ToolError("repository policy does not allow expired-heartbeat cleanup", exit_code=5) + try: + heartbeat = dt.datetime.fromisoformat(str(lock["heartbeat_at"]).replace("Z", "+00:00")) + except (KeyError, ValueError) as exc: + raise ToolError("lock has no valid heartbeat timestamp", exit_code=5) from exc + age = (dt.datetime.now(dt.timezone.utc) - heartbeat).total_seconds() + minimum = int(locks.get("expired_heartbeat_seconds", 86400)) + if age < minimum: + raise ToolError(f"lock heartbeat age {age:.0f}s is below policy threshold {minimum}s", exit_code=5) + path.unlink() + print(json.dumps({ + "cleaned": True, + "task": task, + "lock_id": args.lock_id, + "policy": args.policy, + "prior_health": health, + "cleaned_at": utc_now(), + }, indent=2)) + return 0 + + +def nul_items(value: bytes) -> list[str]: + return [item.decode(errors="surrogateescape") for item in value.split(b"\0") if item] + + +def changed_files(repo: Path, base: str) -> tuple[list[dict[str, Any]], str]: + merge_base = git(repo, "merge-base", base, "HEAD").stdout.strip() + raw = git(repo, "diff", "--name-status", "-z", "--find-renames", merge_base, text=False).stdout + parts = nul_items(raw) + files: list[dict[str, Any]] = [] + index = 0 + while index < len(parts): + status = parts[index] + index += 1 + if status.startswith(("R", "C")): + old_path, path = parts[index], parts[index + 1] + index += 2 + files.append({"path": path, "status": status, "old_path": old_path, "tracked": True}) + else: + path = parts[index] + index += 1 + files.append({"path": path, "status": status, "tracked": True}) + tracked = {item["path"] for item in files} + for path in nul_items(git(repo, "ls-files", "--others", "--exclude-standard", "-z", text=False).stdout): + if path not in tracked: + files.append({"path": path, "status": "?", "tracked": False}) + return files, merge_base + + +def path_matches(pattern: str, path: str) -> bool: + normalized = pattern.strip() + if not normalized or normalized.startswith("#"): + return False + if normalized.startswith("!"): + normalized = normalized[1:] + anchored = normalized.startswith("/") + normalized = normalized.lstrip("/") + if normalized.endswith("/"): + normalized += "**" + if anchored or "/" in normalized: + return fnmatch.fnmatchcase(path, normalized) + return fnmatch.fnmatchcase(Path(path).name, normalized) or fnmatch.fnmatchcase(path, f"**/{normalized}") + + +def ownership_rules(repo: Path, config: dict[str, Any]) -> tuple[list[dict[str, Any]], bool, str | None]: + ownership = config.get("ownership", {}) if isinstance(config.get("ownership", {}), dict) else {} + declared = ownership.get("rules", []) + if declared: + rules = [] + for index, value in enumerate(declared): + if not isinstance(value, dict) or not isinstance(value.get("pattern"), str): + raise ToolError(f"ownership.rules[{index}] is invalid") + owners = value.get("owners", value.get("owner", [])) + if isinstance(owners, str): + owners = [owners] + if not isinstance(owners, list) or not all(isinstance(item, str) for item in owners): + raise ToolError(f"ownership.rules[{index}].owners is invalid") + rules.append({"pattern": value["pattern"], "owners": owners}) + return rules, bool(ownership.get("required", False)), str(CONFIG_PATH) + for candidate in (repo / ".github/CODEOWNERS", repo / "CODEOWNERS", repo / "docs/CODEOWNERS"): + if candidate.exists(): + rules = [] + for line in candidate.read_text(errors="replace").splitlines(): + stripped = line.strip() + if not stripped or stripped.startswith("#"): + continue + parts = stripped.split() + if len(parts) >= 2: + rules.append({"pattern": parts[0], "owners": parts[1:]}) + return rules, bool(ownership.get("required", False)), str(candidate.relative_to(repo)) + return [], bool(ownership.get("required", False)), None + + +def assign_owners(path: str, rules: list[dict[str, Any]]) -> list[str]: + owners: list[str] = [] + for rule in rules: + if path_matches(rule["pattern"], path): + owners = list(rule["owners"]) + return owners + + +def hunk_counts(repo: Path, merge_base: str) -> tuple[dict[str, int], int]: + patch = git(repo, "diff", "--no-color", "--unified=0", merge_base, "--").stdout + current: str | None = None + counts: dict[str, int] = {} + total = 0 + for line in patch.splitlines(): + if line.startswith("+++ b/"): + current = line[6:] + elif line.startswith("@@") and current: + counts[current] = counts.get(current, 0) + 1 + total += 1 + return counts, total + + +def untracked_hunks(repo: Path, item: dict[str, Any]) -> int: + path = repo / item["path"] + if not path.is_file() or path.is_symlink(): + return 0 + try: + raw = path.read_bytes() + except OSError: + return 0 + if b"\0" in raw: + return 0 + return 1 if raw else 0 + + +def handoff_markdown(value: dict[str, Any]) -> str: + lines = [ + f"# Handoff: {value['task']}", + "", + f"- Repository: `{value['repository_path']}`", + f"- Branch: `{value['branch']}`", + f"- Starting SHA: `{value['start_sha']}`", + f"- Merge base: `{value['merge_base']}`", + f"- HEAD: `{value['head_sha']}`", + f"- Dirty: `{str(value['dirty']).lower()}`", + f"- Changed files: `{value['summary']['changed_files']}`", + f"- Changed hunks: `{value['summary']['changed_hunks']}`", + "", + "## Files", + "", + ] + if not value["files"]: + lines.append("No changes detected.") + else: + for item in value["files"]: + if item["owners"]: + owners = ", ".join(item["owners"]) + elif value["ownership"]["required"]: + owners = "UNOWNED (violation)" + else: + owners = "no owner declared (optional)" + lines.append(f"- `{item['status']}` `{item['path']}` — {item['hunks']} hunk(s); {owners}") + if value["summary"]["ownership_violations"]: + lines.extend(["", "## Ownership violations", ""]) + lines.extend(f"- `{path}`" for path in value["summary"]["ownership_violations"]) + lines.extend(["", f"Evidence JSON: `{value['receipt_path']}`", ""]) + return "\n".join(lines) + + +def handoff_command(args: argparse.Namespace) -> int: + repo = repo_root(args.repo) + task = validate_task(args.task) + manifest, _ = load_manifest(repo, task) + verification = verify_workspace(repo, manifest, require_writer=False) + config, config_digest = load_config(repo) + files, merge_base = changed_files(repo, manifest["start_sha"]) + counts, total_hunks = hunk_counts(repo, merge_base) + rules, ownership_required, ownership_source = ownership_rules(repo, config) + violations: list[str] = [] + for item in files: + if not item["tracked"]: + counts[item["path"]] = untracked_hunks(repo, item) + total_hunks += counts[item["path"]] + item["hunks"] = counts.get(item["path"], 0) + item["owners"] = assign_owners(item["path"], rules) + if ownership_required and not item["owners"]: + violations.append(item["path"]) + status = git(repo, "status", "--short", "--branch").stdout.rstrip() + handoff_dir = task_dir(repo, task) / "handoffs" + handoff_dir.mkdir(parents=True, exist_ok=True) + receipt_path = handoff_dir / f"{dt.datetime.now(dt.timezone.utc).strftime('%Y%m%dT%H%M%SZ')}-{uuid.uuid4().hex[:8]}.json" + value: dict[str, Any] = { + "schema": HANDOFF_SCHEMA, + "tool": { + "version": VERSION, + "implementation_sha256": implementation_sha256(), + }, + "task": task, + "repository_path": str(repo), + "common_git_dir": str(common_git_dir(repo)), + "branch": manifest["branch"], + "start_sha": manifest["start_sha"], + "merge_base": merge_base, + "head_sha": git(repo, "rev-parse", "HEAD").stdout.strip(), + "dirty": bool(files), + "git_status": status, + "files": sorted(files, key=lambda item: item["path"]), + "ownership": { + "required": ownership_required, + "source": ownership_source, + "policy_sha256": config_digest, + }, + "summary": { + "changed_files": len(files), + "changed_hunks": total_hunks, + "ownership_violations": sorted(violations), + }, + "workspace_verification": verification, + "generated_at": utc_now(), + "receipt_path": str(receipt_path), + } + atomic_json(receipt_path, value, exclusive=True) + output = json.dumps(value, indent=2, sort_keys=True) + "\n" if args.format == "json" else handoff_markdown(value) + if args.output: + output_path = Path(args.output).expanduser().resolve() + try: + output_path.relative_to(repo) + except ValueError: + pass + else: + raise ToolError( + "handoff output must stay outside the writer worktree so it " + "cannot invalidate its own evidence" + ) + output_path.parent.mkdir(parents=True, exist_ok=True) + output_path.write_text(output) + print(str(output_path)) + else: + print(output, end="" if output.endswith("\n") else "\n") + if violations: + eprint("handoff: ownership policy violations are recorded in the receipt") + return 6 + return 0 + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + prog="worktree-bootstrap", + description="Enforce one task -> one worktree -> one branch -> one writer.", + ) + parser.add_argument("--version", action="version", version=f"worktree-bootstrap {VERSION}") + subparsers = parser.add_subparsers(dest="subcommand", required=True) + + create = subparsers.add_parser("create", help="validate the base, create a worktree, and run locked bootstrap") + create.add_argument("task") + create.add_argument("--repo", default=".") + create.add_argument("--base") + create.add_argument("--branch") + create.add_argument("--path") + authority = create.add_mutually_exclusive_group() + authority.add_argument("--no-fetch", action="store_true", help="allowed only for repositories without an upstream") + authority.add_argument( + "--local-only", + action="store_true", + help="create from a full local commit SHA without branch or upstream discovery", + ) + create.set_defaults(handler=create_command) + + verify = subparsers.add_parser("verify", help="verify repository, branch, start SHA, dependencies, and optional writer lock") + verify.add_argument("task") + verify.add_argument("--repo", default=".") + verify.add_argument("--require-writer", action="store_true") + verify.set_defaults(handler=verify_command) + + launch = subparsers.add_parser( + "launch", + help="acquire the single-writer lock and run a command", + usage=( + "worktree-bootstrap launch TASK [--repo REPO] " + "[--heartbeat-seconds N] -- WRITER [ARG ...]" + ), + ) + launch.add_argument("task") + launch.add_argument("--repo", default=".") + launch.add_argument("--heartbeat-seconds", type=int, default=10) + launch.set_defaults(handler=launch_command) + + lock = subparsers.add_parser("lock", help="inspect or explicitly clean writer locks") + lock_subparsers = lock.add_subparsers(dest="lock_command", required=True) + inspect = lock_subparsers.add_parser("inspect") + inspect.add_argument("task") + inspect.add_argument("--repo", default=".") + inspect.set_defaults(handler=lock_inspect_command) + clean = lock_subparsers.add_parser("clean") + clean.add_argument("task") + clean.add_argument("--repo", default=".") + clean.add_argument("--policy", required=True, choices=["dead-local", "expired-heartbeat"]) + clean.add_argument("--lock-id", required=True) + clean.set_defaults(handler=lock_clean_command) + + handoff = subparsers.add_parser("handoff", help="derive a handoff from Git and ownership evidence") + handoff.add_argument("task") + handoff.add_argument("--repo", default=".") + handoff.add_argument("--format", choices=["json", "markdown"], default="markdown") + handoff.add_argument("--output") + handoff.set_defaults(handler=handoff_command) + return parser + + +def main(argv: Sequence[str] | None = None) -> int: + parser = build_parser() + raw = list(argv) if argv is not None else sys.argv[1:] + writer_command: list[str] | None = None + if raw[:1] == ["launch"] and not any( + value in {"-h", "--help"} for value in raw[1:] + ): + if "--" not in raw: + parser.error("launch requires a writer command after --") + boundary = raw.index("--") + writer_command = raw[boundary + 1 :] + raw = raw[:boundary] + args = parser.parse_args(raw) + if args.subcommand == "launch": + args.command = writer_command or [] + try: + return int(args.handler(args)) + except ToolError as exc: + eprint(f"worktree-bootstrap: {exc}") + if exc.details and os.environ.get("WORKTREE_BOOTSTRAP_DEBUG"): + eprint(json.dumps(exc.details, indent=2, sort_keys=True)) + return exc.exit_code + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/inspector-preflight.mjs b/scripts/inspector-preflight.mjs index 67df617..fc35dbf 100755 --- a/scripts/inspector-preflight.mjs +++ b/scripts/inspector-preflight.mjs @@ -34,7 +34,7 @@ try { const statusEnvelope = inspect('tools/call', 'status'); const status = statusEnvelope.structuredContent ?? JSON.parse(statusEnvelope.content?.[0]?.text ?? '{}'); - assert.equal(status.version, '3.4.0'); + assert.equal(status.version, '3.4.2'); assert.equal(status.healthy, status.local_boundary.ready); assert.equal(status.active, 0); assert.deepEqual(status.tasks, []); diff --git a/scripts/mcp-environment-preflight.mjs b/scripts/mcp-environment-preflight.mjs index 102ba39..f4569ec 100644 --- a/scripts/mcp-environment-preflight.mjs +++ b/scripts/mcp-environment-preflight.mjs @@ -64,7 +64,7 @@ try { })}\n`); }); const status = response.result?.structuredContent; - assert.equal(status?.version, '3.4.0'); + assert.equal(status?.version, '3.4.2'); assert.equal(status?.healthy, true, JSON.stringify(status?.local_boundary)); assert.equal(status?.local_boundary?.ready, true, JSON.stringify(status?.local_boundary)); assert.equal(status?.local_boundary?.boundary, 'systemd-user-service-cgroup'); diff --git a/scripts/process-boundary-preflight.mjs b/scripts/process-boundary-preflight.mjs index 56b2ba7..7055c2d 100644 --- a/scripts/process-boundary-preflight.mjs +++ b/scripts/process-boundary-preflight.mjs @@ -64,7 +64,7 @@ try { }); descendantPid = Number(await waitFor( () => readFile(pidFile, 'utf8'), - (value) => Number.isInteger(Number(value.trim())), + (value) => Number.isSafeInteger(Number(value.trim())) && Number(value.trim()) > 1, )); const observed = await observeExactProcessIdentity(descendantPid); descendantIdentity = freezeExactProcessIdentity(observed); diff --git a/scripts/validate-package-docs.mjs b/scripts/validate-package-docs.mjs new file mode 100644 index 0000000..dc0f480 --- /dev/null +++ b/scripts/validate-package-docs.mjs @@ -0,0 +1,96 @@ +#!/usr/bin/env node + +import { access, mkdir, readdir, readFile, writeFile } from 'node:fs/promises'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); + +// Keep this list deliberately small. The 3.3.0 note is included only because +// the migration guide links to it; it is not an invitation to package the +// repository's internal docs tree. +export const PACKAGE_DOCUMENTS = Object.freeze([ + ['docs/co-engineer-quickstart.md', 'co-engineer-quickstart.md'], + ['docs/co-engineer-troubleshooting.md', 'co-engineer-troubleshooting.md'], + ['docs/co-engineer-migration-3.2.1.md', 'co-engineer-migration-3.2.1.md'], + ['docs/configuration.md', 'configuration.md'], + ['docs/mcp-pending-call.md', 'mcp-pending-call.md'], + ['docs/run-tool-api.md', 'run-tool-api.md'], + ['docs/efficient-dogfood.md', 'efficient-dogfood.md'], + ['docs/releases/v3.3.0.md', 'releases/v3.3.0.md'], + ['docs/releases/v3.4.1.md', 'releases/v3.4.1.md'], + ['docs/releases/v3.4.2.md', 'releases/v3.4.2.md'], +].map(([source, packageRelative]) => Object.freeze({ source, packageRelative }))); + +export const PACKAGE_DOC_ROOT = 'plugins/codex-co-engineer/docs'; + +function sameBytes(left, right) { + return left.length === right.length && left.equals(right); +} + +async function regularFiles(directory, prefix = '') { + const files = []; + for (const entry of await readdir(directory, { withFileTypes: true })) { + const relative = prefix ? path.join(prefix, entry.name) : entry.name; + const target = path.join(directory, entry.name); + if (entry.isDirectory()) { + files.push(...await regularFiles(target, relative)); + } else if (entry.isFile()) { + files.push(relative.split(path.sep).join('/')); + } else { + throw new Error(`Package documentation contains a non-regular entry: ${relative}`); + } + } + return files.sort(); +} + +export async function syncPackageDocs(root = ROOT) { + for (const { source, packageRelative } of PACKAGE_DOCUMENTS) { + const sourcePath = path.join(root, source); + const packagePath = path.join(root, PACKAGE_DOC_ROOT, packageRelative); + await mkdir(path.dirname(packagePath), { recursive: true }); + await writeFile(packagePath, await readFile(sourcePath)); + } +} + +export async function validatePackageDocs(root = ROOT) { + const expected = new Set(PACKAGE_DOCUMENTS.map(({ packageRelative }) => packageRelative)); + const packageRoot = path.join(root, PACKAGE_DOC_ROOT); + await access(packageRoot); + + for (const { source, packageRelative } of PACKAGE_DOCUMENTS) { + const sourcePath = path.join(root, source); + const packagePath = path.join(packageRoot, packageRelative); + const [sourceBytes, packageBytes] = await Promise.all([ + readFile(sourcePath), + readFile(packagePath), + ]); + if (!sameBytes(sourceBytes, packageBytes)) { + throw new Error(`Packaged documentation differs from its source: ${packageRelative}`); + } + } + + const actual = await regularFiles(packageRoot); + const extras = actual.filter((file) => !expected.has(file)); + const missing = [...expected].filter((file) => !actual.includes(file)); + if (extras.length > 0) { + throw new Error(`Unallowlisted package documentation: ${extras.join(', ')}`); + } + if (missing.length > 0) { + throw new Error(`Missing package documentation: ${missing.join(', ')}`); + } +} + +const isMain = process.argv[1] + && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url); + +if (isMain) { + try { + if (process.argv.includes('--sync')) await syncPackageDocs(); + await validatePackageDocs(); + process.stdout.write(`Package documentation validation passed (${PACKAGE_DOCUMENTS.length} files).\n`); + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`); + process.exitCode = 1; + } +} diff --git a/scripts/validate-release.mjs b/scripts/validate-release.mjs index f890964..5c92df9 100755 --- a/scripts/validate-release.mjs +++ b/scripts/validate-release.mjs @@ -11,10 +11,11 @@ import { THREAT_MODEL_RELATIVE, assertR1FirstReleaseContract, } from './r1-first-release-contract.mjs'; +import { validatePackageDocs } from './validate-package-docs.mjs'; const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); const PLUGIN = 'plugins/codex-co-engineer'; -const RELEASE_VERSION = '3.4.0'; +const RELEASE_VERSION = '3.4.2'; function fail(message) { throw new Error(message); } const absolute = (relative) => path.join(ROOT, relative); @@ -22,22 +23,51 @@ const text = (relative) => readFile(absolute(relative), 'utf8'); const json = async (relative) => JSON.parse(await text(relative)); const required = [ + // Preserve published contracts: globbed test runs cannot detect missing suites. + 'plugins/codex-co-engineer/test/r1-decision-reducer-lane-health-adversarial.test.mjs', + 'plugins/codex-co-engineer/test/r1-decision-reducer.test.mjs', + 'plugins/codex-co-engineer/test/r1-final-decision-card-adversarial.test.mjs', + 'plugins/codex-co-engineer/test/r1-final-decision-card.test.mjs', + 'plugins/codex-co-engineer/test/r1-grok-attention-bridge-truthfulness.test.mjs', + 'plugins/codex-co-engineer/test/r1-lane-health.test.mjs', + 'plugins/codex-co-engineer/test/r1-luna-pm-host-adapter-adversarial.test.mjs', + 'plugins/codex-co-engineer/test/r1-luna-pm-host-adapter.test.mjs', + 'plugins/codex-co-engineer/test/r1-luna-pm-relay-adversarial.test.mjs', + 'plugins/codex-co-engineer/test/r1-luna-pm-relay.test.mjs', + 'plugins/codex-co-engineer/test/r1-provider-event-rules-adversarial.test.mjs', + 'plugins/codex-co-engineer/test/r1-provider-event-rules-performance.test.mjs', + 'plugins/codex-co-engineer/test/r1-provider-event-rules.test.mjs', + 'plugins/codex-co-engineer/test/r1-usage-ledger-adversarial.test.mjs', + 'plugins/codex-co-engineer/test/r1-usage-ledger.test.mjs', + 'plugins/codex-co-engineer/test/r1-worktree-cleanup-adapter.test.mjs', + 'plugins/codex-co-engineer/test/r1-worktree-cleanup-planner-adversarial.test.mjs', + 'plugins/codex-co-engineer/test/r1-worktree-cleanup-planner.test.mjs', + 'README.md', 'CHANGELOG.md', 'LICENSE', 'SECURITY.md', 'docs/configuration.md', 'docs/data-handling.md', 'docs/efficient-dogfood.md', 'docs/release.md', 'docs/future-work.md', 'docs/mcp-pending-call.md', 'docs/adr/0001-r1-bounded-run-architecture.md', 'docs/threat-model.md', 'docs/releases/v3.1.0.md', 'docs/releases/v3.1.1.md', 'docs/releases/v3.2.0.md', 'docs/releases/v3.2.1.md', - 'docs/releases/v3.3.0.md', 'docs/releases/v3.4.0.md', + 'docs/releases/v3.3.0.md', 'docs/releases/v3.4.0.md', 'docs/releases/v3.4.1.md', 'docs/releases/v3.4.2.md', 'docs/assets/codex-co-engineer-3.1.0.svg', 'docs/assets/codex-co-engineer-3.1.0.jpg', '.agents/plugins/marketplace.json', 'scripts/mcp-pending-call-probe.mjs', '.codex/release-gate.toml', '.github/workflows/ci.yml', `${PLUGIN}/.codex-plugin/plugin.json`, `${PLUGIN}/.mcp.json`, `${PLUGIN}/package.json`, `${PLUGIN}/README.md`, `${PLUGIN}/bin/setup.mjs`, + `${PLUGIN}/mcp/v3/usage-ledger.mjs`, + `${PLUGIN}/mcp/v3/luna-pm-host-adapter.mjs`, + `${PLUGIN}/mcp/v3/final-decision-card.mjs`, + `${PLUGIN}/mcp/v3/provider-event-rules.mjs`, + `${PLUGIN}/mcp/v3/decision-reducer.mjs`, + `${PLUGIN}/mcp/v3/lane-health.mjs`, + `${PLUGIN}/mcp/v3/grok-question-bridge.mjs`, + `${PLUGIN}/mcp/v3/worktree-cleanup-planner.mjs`, + `${PLUGIN}/mcp/v3/worktree-cleanup-adapter.mjs`, `${PLUGIN}/mcp/v3/server.mjs`, `${PLUGIN}/mcp/v3/supervisor.mjs`, `${PLUGIN}/mcp/v3/task-store.mjs`, `${PLUGIN}/mcp/v3/acp-worker.mjs`, `${PLUGIN}/mcp/v3/contract.mjs`, `${PLUGIN}/mcp/v3/deadline.mjs`, `${PLUGIN}/mcp/v3/diagnostics.mjs`, `${PLUGIN}/mcp/v3/mailbox.mjs`, - `${PLUGIN}/mcp/v3/cursor-cloud-worker.mjs`, `${PLUGIN}/mcp/v3/single-turn.flow.mjs`, + `${PLUGIN}/mcp/v3/cursor-cloud-worker.mjs`, `${PLUGIN}/mcp/v3/process-boundary.mjs`, `${PLUGIN}/mcp/v3/compact-task.mjs`, `${PLUGIN}/mcp/v3/provider-result.mjs`, `${PLUGIN}/mcp/v3/response.mjs`, @@ -45,10 +75,14 @@ const required = [ `${PLUGIN}/assets/acpx-third-party-notices.md`, `${PLUGIN}/vendor/dsh-acp-demo/LICENSE`, `${PLUGIN}/vendor/dsh-acp-demo/PROVENANCE.json`, `${PLUGIN}/vendor/dsh-acp-demo/package.json`, + `${PLUGIN}/vendor/worktree-bootstrap/LICENSE`, `${PLUGIN}/vendor/worktree-bootstrap/PROVENANCE.json`, + `${PLUGIN}/vendor/worktree-bootstrap/worktree-bootstrap`, + `${PLUGIN}/mcp/v3/worktree-bootstrap-runtime.mjs`, `${PLUGIN}/skills/control-codex-co-engineer-agents/SKILL.md`, `${PLUGIN}/skills/control-codex-co-engineer-agents/agents/openai.yaml`, 'scripts/release-prerequisites.mjs', 'scripts/validate-release.mjs', 'scripts/r1-first-release-contract.mjs', - 'scripts/inspector-preflight.mjs', `${PLUGIN}/test/r1-first-release-non-goals.test.mjs`, + 'scripts/inspector-preflight.mjs', 'scripts/validate-package-docs.mjs', + `${PLUGIN}/test/r1-first-release-non-goals.test.mjs`, 'scripts/process-boundary-preflight.mjs', 'scripts/mcp-environment-preflight.mjs', 'tools/acpx-vendor/package.json', 'tools/acpx-vendor/package-lock.json', ]; @@ -99,9 +133,11 @@ if (!serverText.includes('Required property named repo') } if (packageJson.scripts?.test !== 'node --no-warnings --test test/*.test.mjs') fail('Unexpected test script.'); if (JSON.stringify(packageJson.files) !== JSON.stringify([ - '.codex-plugin', '.mcp.json', 'README.md', 'assets', 'bin', 'mcp', 'skills', 'vendor', 'package.json', + '.codex-plugin', '.mcp.json', 'README.md', 'docs', 'assets', 'bin', 'mcp', 'skills', 'vendor', 'package.json', ])) fail('Co-Engineer package roots changed.'); +await validatePackageDocs(ROOT); + const server = mcp.mcpServers?.['codex-co-engineer']; if (server?.command !== 'node' || JSON.stringify(server.args) !== JSON.stringify(['--no-warnings', './mcp/v3/server.mjs', '--stdio']) @@ -129,11 +165,11 @@ if (!serverText.includes('dsh_model') || !serverText.includes('stealth/ox-alpha' fail('3.2.1 public contract must advertise the optional Ox Alpha DSH model selector.'); } if (!serverText.includes('response_mode') - || !serverText.includes("enum: ['structured']") + || !serverText.includes("enum: ['structured', 'legacy']") || !serverText.includes("enum: ['summary', 'diagnostics', 'compact']") || !serverText.includes('task_ids') || !serverText.includes('cursors')) { - fail('3.2 public contract must advertise structured response_mode, compact task view, and wait-any task_ids/cursors.'); + fail('3.2 public contract must advertise structured/legacy response_mode, compact task view, and wait-any task_ids/cursors.'); } const compactTaskText = await text(`${PLUGIN}/mcp/v3/compact-task.mjs`); if (!compactTaskText.includes('WAIT_ANY_RESPONSE_STRUCTURED_BYTES_MAX') @@ -184,6 +220,16 @@ for (const file of ['LICENSE', 'PROVENANCE.json']) { if (!dshPackage.files?.includes(file)) fail(`DSH package omits ${file}.`); } +const worktreeBootstrapProvenance = await json(`${PLUGIN}/vendor/worktree-bootstrap/PROVENANCE.json`); +const worktreeBootstrap = await readFile(absolute(`${PLUGIN}/vendor/worktree-bootstrap/worktree-bootstrap`)); +if (worktreeBootstrapProvenance.version !== '1.1.0' + || worktreeBootstrapProvenance.license !== 'MIT' + || createHash('sha256').update(worktreeBootstrap).digest('hex') !== worktreeBootstrapProvenance.artifact_sha256) { + fail('Bundled worktree-bootstrap provenance/hash mismatch.'); +} +const worktreeBootstrapMode = (await lstat(absolute(`${PLUGIN}/vendor/worktree-bootstrap/worktree-bootstrap`))).mode; +if ((worktreeBootstrapMode & 0o111) === 0) fail('Bundled worktree-bootstrap must be executable.'); + async function releaseFiles(directory = ROOT) { const values = []; for (const entry of await readdir(directory, { withFileTypes: true })) { diff --git a/tools/acpx-vendor/src/hardening-overlay.mjs b/tools/acpx-vendor/src/hardening-overlay.mjs index 584e907..b9755ae 100644 --- a/tools/acpx-vendor/src/hardening-overlay.mjs +++ b/tools/acpx-vendor/src/hardening-overlay.mjs @@ -176,44 +176,94 @@ AsyncEventQueue = class CoEngineerAsyncEventQueue { }; async function coEngineerRememberAgentDescendants(child) { - if (!child?.pid) return new Set(); + if (!child?.pid) return new Map(); const descendants = child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] - ?? (child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] = new Set()); - for (const pid of await listDescendantPids(child.pid)) descendants.add(pid); + ?? (child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] = new Map()); + if (process.platform === 'linux') { + const processTable = coEngineerReadLinuxProcessTable(); + const root = processTable.get(child.pid); + if (!root) { + try { + process.kill(child.pid, 0); + } catch { + return descendants; + } + throw new Error('Could not inspect the live ACP agent in /proc.'); + } + const children = new Map(); + for (const identity of processTable.values()) { + if (identity.state === 'Z') continue; + const siblings = children.get(identity.parentPid) ?? []; + siblings.push(identity); + children.set(identity.parentPid, siblings); + } + const pending = [child.pid]; + const visited = new Set(pending); + for (let index = 0; index < pending.length; index += 1) { + for (const identity of children.get(pending[index]) ?? []) { + if (visited.has(identity.pid)) continue; + visited.add(identity.pid); + descendants.set(identity.pid, identity.startTime); + pending.push(identity.pid); + } + } + for (const identity of processTable.values()) { + if (identity.pid !== child.pid && identity.processGroupId === child.pid && identity.state !== 'Z') { + descendants.set(identity.pid, identity.startTime); + } + } + return descendants; + } + for (const pid of await listDescendantPids(child.pid)) descendants.set(pid, null); for (const pid of await listProcessGroupPids(child.pid)) { - if (pid !== child.pid) descendants.add(pid); + if (pid !== child.pid) descendants.set(pid, null); } return descendants; } -function coEngineerAgentTreeAlive(child) { - if (!child?.pid) return false; - if (isChildProcessRunning(child)) return true; - return coEngineerHasLivePid(child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] ?? new Set()); -} - -/* - * On Linux, a killed detached child can remain as a zombie until its new - * parent reaps it. `kill(pid, 0)` still succeeds for that zombie, but it has - * no running work or handles left. Treat the process as terminated for - * containment waits so a reaper delay cannot consume the close deadline. - */ -function coEngineerPidIsZombie(pid) { - if (process.platform !== 'linux') return false; +function coEngineerReadLinuxProcessIdentity(pid) { try { const stat = readFileSync(`/proc/${pid}/stat`, 'utf8'); const stateOffset = stat.lastIndexOf(')') + 2; - return stateOffset > 1 && stat[stateOffset] === 'Z'; - } catch { - return false; + if (stateOffset <= 1) throw new Error(`Malformed /proc/${pid}/stat.`); + const fields = stat.slice(stateOffset).trim().split(/\s+/u); + const parentPid = Number(fields[1]); + const processGroupId = Number(fields[2]); + const startTime = fields[19]; + if (!Number.isInteger(parentPid) || !Number.isInteger(processGroupId) || !startTime) { + throw new Error(`Malformed /proc/${pid}/stat.`); + } + return { pid, state: fields[0], parentPid, processGroupId, startTime }; + } catch (error) { + if (error?.code === 'ENOENT' || error?.code === 'ESRCH') return null; + throw error; } } +function coEngineerReadLinuxProcessTable() { + const processes = new Map(); + for (const entry of fs.readdirSync('/proc', { withFileTypes: true })) { + if (!entry.isDirectory() || !/^\d+$/u.test(entry.name)) continue; + const identity = coEngineerReadLinuxProcessIdentity(Number(entry.name)); + if (identity) processes.set(identity.pid, identity); + } + return processes; +} + +function coEngineerAgentTreeAlive(child) { + if (!child?.pid) return false; + if (isChildProcessRunning(child)) return true; + return coEngineerHasLivePid(child[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] ?? new Map()); +} + function coEngineerHasLivePid(pids) { - for (const pid of pids) { - if (coEngineerPidIsZombie(pid)) { - pids.delete(pid); - continue; + for (const [pid, startTime] of pids) { + if (process.platform === 'linux') { + const identity = coEngineerReadLinuxProcessIdentity(pid); + if (!identity || identity.state === 'Z' || identity.startTime !== startTime) { + pids.delete(pid); + continue; + } } try { process.kill(pid, 0); @@ -230,11 +280,20 @@ async function coEngineerSignalAgentTree(child, signal) { const descendants = await coEngineerRememberAgentDescendants(child); if (process.platform === 'win32') { await killWindowsProcessTree(child.pid, signal); - for (const pid of descendants) await killWindowsProcessTree(pid, signal); + for (const pid of descendants.keys()) await killWindowsProcessTree(pid, signal); return; } if (isChildProcessRunning(child) && hasLiveProcessGroup(child.pid)) sendSignal(-child.pid, signal); - for (const pid of descendants) sendSignal(pid, signal); + for (const [pid, startTime] of descendants) { + if (process.platform === 'linux') { + const identity = coEngineerReadLinuxProcessIdentity(pid); + if (!identity || identity.state === 'Z' || identity.startTime !== startTime) { + descendants.delete(pid); + continue; + } + } + sendSignal(pid, signal); + } } async function coEngineerWaitForAgentTree(child, waitMs) { @@ -369,7 +428,7 @@ AcpClient.prototype.spawnAgentProcess = async function coEngineerSpawnAgentProce detached: process.platform !== 'win32', windowsVerbatimArguments: spawnCommand.windowsVerbatimArguments, }); - spawnedChild[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] = new Set(); + spawnedChild[CO_ENGINEER_ACPX_AGENT_DESCENDANTS] = new Map(); spawnedChild.once('exit', () => { void coEngineerRememberAgentDescendants(spawnedChild).catch(() => {}); });