diff --git a/.gitleaks.toml b/.gitleaks.toml index ba943440..e1a9fe8d 100644 --- a/.gitleaks.toml +++ b/.gitleaks.toml @@ -117,3 +117,10 @@ paths = ['''(^|/)dist/webui/server\.js$'''] regexTarget = "match" regexes = ['''^[A-Za-z_$][A-Za-z0-9_$]*\.setRsaPrivateKey = [A-Za-z_$][A-Za-z0-9_$]*\.rsa\.setPrivateKey ?$''', '''^[A-Za-z_$][A-Za-z0-9_$]*\.privateKeyToAsn1 = [A-Za-z_$][A-Za-z0-9_$]*\.privateKeyToRSAPrivateKey ?$''', '''^[A-Za-z_$][A-Za-z0-9_$]*\.generateKey = [A-Za-z_$][A-Za-z0-9_$]*\.pbe\.generatePkcs12Key;?$'''] +[[rules.allowlists]] +description = "Leak-prevention tripwire fixture: asserts the facade never serializes this fake key" +condition = "AND" +paths = ['''(^|/)packages/webui/test/lib/engine/model-reads\.test\.js$'''] +regexTarget = "match" +regexes = ['''apiKey: "sk-secret-should-never-leak"'''] + diff --git a/docs/tui-capabilities.md b/docs/tui-capabilities.md index 4b53b058..df506203 100644 --- a/docs/tui-capabilities.md +++ b/docs/tui-capabilities.md @@ -220,7 +220,11 @@ Status legend: ✅ wired · ⚠ partial / path differs · ❌ no path · 🚧 re The webui does not yet expose `/fork`, `/resume`, or a "rewind last turn" action — those acp methods (`fork`, `resume`) are reported by -`MCODE_ACP_CAPABILITIES` but no webui route wraps them. +`MCODE_ACP_CAPABILITIES` but no webui route wraps them. (Since M3-B4 that +table is no longer what `GET /api/protocol/capabilities` returns; the +endpoint serves the engine's declared 14-key capability object instead. +The table is still exported and still pinned by +`test/lib/mcode-rpc.check.mjs`.) ## ACP Skill commands @@ -252,7 +256,15 @@ entries by default. mcodeVersion, mcodeName?, mcodeTitle?, - capabilities: MCODE_ACP_CAPABILITIES, // see packages/webui/server/lib/mcode-rpc.js + // The engine's DECLARED 14-key capability object, served by + // packages/webui/server/engine/capability-reads.js. Before M3-B4 this + // field carried MCODE_ACP_CAPABILITIES (the ACP wire table in + // packages/webui/server/lib/mcode-rpc.js), which is still exported + // there and still a true statement about the ENGINE's ACP surface. + capabilities: <14-key declaration>, + capabilitiesProvider, // which provider's declaration answered + capabilitiesProviderFor, // "transport" | "default" (see API.md) + capabilitiesUnavailable, // the degradation roll-up notes: { set_mode, set_config_option, cancel, activate, fork, load, list, close, new, prompt, diff --git a/docs/webui.md b/docs/webui.md index d86fbb85..473eee90 100644 --- a/docs/webui.md +++ b/docs/webui.md @@ -2411,7 +2411,7 @@ marker), not by tool name. | `POST` | `/api/protocol/load-session` | `routes/protocol.js#handleLoadSession` | `?cwd=`, fallback to current | | `POST` | `/api/protocol/activate-session` | `routes/protocol.js#handleActivateSession` | one acp client tracks one active session | | `GET` | `/api/protocol/list-sessions` | `routes/protocol.js#handleListSessions` | `?cwd=` filtered | -| `GET` | `/api/protocol/capabilities` | `routes/protocol.js#handleCapabilities` | `{mcodeVersion, mcodeName?, mcodeTitle?, capabilities: MCODE_ACP_CAPABILITIES, notes}` | +| `GET` | `/api/protocol/capabilities` | `routes/protocol.js#handleCapabilities` | `{mcodeVersion, mcodeName?, mcodeTitle?, capabilities, capabilitiesProvider, capabilitiesProviderFor, capabilitiesUnavailable, notes}` — `capabilities` is the engine's declared 14-key capability object (it was the ACP wire table `MCODE_ACP_CAPABILITIES` before M3-B4) | ### Legacy dispatcher (`server/router.js`) diff --git a/docs/webui.zh-CN.md b/docs/webui.zh-CN.md index a37229b7..3c18abc5 100644 --- a/docs/webui.zh-CN.md +++ b/docs/webui.zh-CN.md @@ -1797,7 +1797,7 @@ createdAtMs, updatedAtMs}`)下发,按 `toolCallId` 幂等、上限 32 条、 | `POST` | `/api/protocol/load-session` | `routes/protocol.js#handleLoadSession` | `?cwd=`,缺省取当前 | | `POST` | `/api/protocol/activate-session` | `routes/protocol.js#handleActivateSession` | 一个 acp 客户端跟踪一个活动会话 | | `GET` | `/api/protocol/list-sessions` | `routes/protocol.js#handleListSessions` | `?cwd=` 过滤 | -| `GET` | `/api/protocol/capabilities` | `routes/protocol.js#handleCapabilities` | `{mcodeVersion, mcodeName?, mcodeTitle?, capabilities: MCODE_ACP_CAPABILITIES, notes}` | +| `GET` | `/api/protocol/capabilities` | `routes/protocol.js#handleCapabilities` | `{mcodeVersion, mcodeName?, mcodeTitle?, capabilities, capabilitiesProvider, capabilitiesProviderFor, capabilitiesUnavailable, notes}`——`capabilities` 是引擎声明的 14 键能力对象(M3-B4 之前是 ACP wire 表 `MCODE_ACP_CAPABILITIES`) | ### 旧派发器(`server/router.js`) diff --git a/packages/webui/docs/API.md b/packages/webui/docs/API.md index 4ddb7595..b28d3fa6 100644 --- a/packages/webui/docs/API.md +++ b/packages/webui/docs/API.md @@ -2416,10 +2416,25 @@ insensitive, trailing slash-insensitive, `\` and `/` interchangeable). ### `GET /api/protocol/capabilities` -Returns the engine's `agentInfo` (from the `initialize` reply) plus the -capability table webui knows about (`MCODE_ACP_CAPABILITIES` in -`server/lib/mcode-rpc.js`). Used by the webui to decide which UI -controls to enable. +Returns the engine's `agentInfo` (from the `initialize` reply) and the +**engine-capabilities view**: the declared 14-key capability surface of the +active engine provider, the same declaration `GET /api/engine-capabilities` +serves. Used by the webui to decide which UI controls to enable. + +**This field's contract changed in M3 batch B4.** `capabilities` used to +carry `MCODE_ACP_CAPABILITIES`, a hand-maintained flat `{method: boolean}` +table of the ACP JSON-RPC surface (`set_mode`, `set_config_option`, +`cancel`, `activate`, `fork`, `resume`, `delete`, `load`, `close`, `list`, +`new`, `prompt`). Those twelve keys are **gone**: a consumer reading +`capabilities.set_mode` now gets `undefined` and must fail loudly. What +replaced them answers a different question — **"does the engine have this +capability at all"** — with the 14 matrix keys, each +`{level, missing?, reason?}`. The ACP wire table is still exported from +`server/lib/mcode-rpc.js` and is still a true statement about the +engine's ACP surface; it simply no longer travels on this endpoint. + +The declaration appears exactly once, under `capabilities`, and three +sibling keys say where it came from and what to do about its gaps. **Response 200** ```json @@ -2429,18 +2444,45 @@ controls to enable. "mcodeName": "mcode", "mcodeTitle": "mcode", "capabilities": { - "set_mode": true, - "set_config_option": true, - "cancel": true, - "activate": true, - "fork": true, - "resume": true, - "delete": false, - "load": true, - "close": true, - "list": true, - "new": true, - "prompt": true + "sessionCrud": { "level": "full" }, + "streamingSend": { "level": "full" }, + "interrupt": { "level": "full" }, + "toolSkillInvocation": { "level": "full" }, + "turnDiff": { "level": "full" }, + "turnRewindRedo": { "level": "full" }, + "plugins": { "level": "full" }, + "mcp": { "level": "full" }, + "subagents": { + "level": "partial", + "missing": ["getDelegationSnapshot", "stopDelegation"], + "reason": "delegation snapshot/stop live on the TuiRuntimeAdapter access-context, not on the v2 CliService surface (design §1.3 v2)" + }, + "usageStats": { "level": "full" }, + "authCredentials": { "level": "full" }, + "updateCheck": { + "level": "none", + "reason": "interface-absent: no update-check method anywhere in local-runtime-v2 (design §1.3 v2)" + }, + "fileReadWrite": { + "level": "partial", + "missing": ["file-write"], + "reason": "workspace read browsing only; no write API — writes go through in-turn tools (design §1.3 v2)" + }, + "gitOperations": { + "level": "partial", + "missing": ["git-diff", "git-commit", "git-branch"], + "reason": "read-only metadata + review link; change mutation is outside this package (same discipline as v1's read-only Git facade)" + } + }, + "capabilitiesProvider": "local-runtime-v2", + "capabilitiesProviderFor": "transport", + "capabilitiesUnavailable": { + "none": ["updateCheck"], + "partial": [ + { "key": "subagents", "missing": ["getDelegationSnapshot", "stopDelegation"] }, + { "key": "fileReadWrite", "missing": ["file-write"] }, + { "key": "gitOperations", "missing": ["git-diff", "git-commit", "git-branch"] } + ] }, "notes": { "set_mode": "Takes a modeId from the session's availableModes.", @@ -2455,6 +2497,23 @@ controls to enable. `mcodeVersion` is `"unknown"` before a client has attached (no `initialize` reply yet); the endpoint does not invent a version. +`capabilitiesProvider` is the provider whose declaration answered, and +`capabilitiesProviderFor` says HOW it was chosen. A consumer should +branch on the second one: + +- `"transport"` — the active `MCODE_WEBUI_TRANSPORT`'s own registered + provider answered. +- `"default"` — no provider claims that transport yet (arrives with M4), so + the default provider's declaration is standing in. The view is still a + real, reviewed declaration, but it is not necessarily the connected + engine's, and reporting it as such would be a lie. + +`capabilitiesUnavailable` is the degradation summary the capability-driven +UI renders from: a `none` key means hide the entry point, a `partial` key +means hide or disable exactly the listed sub-actions. It is the one field +that is not the declaration itself, and a consumer should not have to +re-derive it from a taxonomy with three levels and two optional fields. + --- ## Authorize decisions diff --git a/packages/webui/docs/API.zh-CN.md b/packages/webui/docs/API.zh-CN.md index 98d51f25..05d0ffc2 100644 --- a/packages/webui/docs/API.zh-CN.md +++ b/packages/webui/docs/API.zh-CN.md @@ -2230,9 +2230,24 @@ code, killEndpoint: "/api/stop" }`。温和版→SIGKILL 的级联 ### `GET /api/protocol/capabilities` -返回引擎的 `agentInfo`(取自 `initialize` 应答)以及 webui -已知的 capability 表(`server/lib/mcode-rpc.js` 里的 -`MCODE_ACP_CAPABILITIES`)。webui 用它来决定启用哪些 UI 控件。 +返回引擎的 `agentInfo`(取自 `initialize` 应答)与 +**engine-capabilities 视图**:当前引擎 provider 声明的 14 键能力面, +也就是 `GET /api/engine-capabilities` 所服务的同一份声明。webui 用它来 +决定启用哪些 UI 控件。 + +**这个字段的契约在 M3 批次 B4 变更过。** `capabilities` 过去承载 +`MCODE_ACP_CAPABILITIES`——一张手工维护的扁平 `{方法: 布尔}` 表,描述 +ACP JSON-RPC 面(`set_mode`、`set_config_option`、`cancel`、`activate`、 +`fork`、`resume`、`delete`、`load`、`close`、`list`、`new`、 +`prompt`)。这 12 个键**已经没有了**:读 `capabilities.set_mode` 的消费方 +现在拿到 `undefined`,会响亮地失败。顶替它们回答的是另一个问题 +——**「引擎到底有没有这项能力」**——用 14 个矩阵键,每项形如 +`{level, missing?, reason?}`。ACP wire 表仍从 +`server/lib/mcode-rpc.js` 导出,且仍是对**引擎** ACP 面的真实陈述; +它只是不再随这个端点返回。 + +声明在响应里只出现一次,就在 `capabilities` 下;三个兄弟键说明它从哪来 +以及该拿它的缺口怎么办。 **响应 200** ```json @@ -2242,18 +2257,45 @@ code, killEndpoint: "/api/stop" }`。温和版→SIGKILL 的级联 "mcodeName": "mcode", "mcodeTitle": "mcode", "capabilities": { - "set_mode": true, - "set_config_option": true, - "cancel": true, - "activate": true, - "fork": true, - "resume": true, - "delete": false, - "load": true, - "close": true, - "list": true, - "new": true, - "prompt": true + "sessionCrud": { "level": "full" }, + "streamingSend": { "level": "full" }, + "interrupt": { "level": "full" }, + "toolSkillInvocation": { "level": "full" }, + "turnDiff": { "level": "full" }, + "turnRewindRedo": { "level": "full" }, + "plugins": { "level": "full" }, + "mcp": { "level": "full" }, + "subagents": { + "level": "partial", + "missing": ["getDelegationSnapshot", "stopDelegation"], + "reason": "delegation snapshot/stop live on the TuiRuntimeAdapter access-context, not on the v2 CliService surface (design §1.3 v2)" + }, + "usageStats": { "level": "full" }, + "authCredentials": { "level": "full" }, + "updateCheck": { + "level": "none", + "reason": "interface-absent: no update-check method anywhere in local-runtime-v2 (design §1.3 v2)" + }, + "fileReadWrite": { + "level": "partial", + "missing": ["file-write"], + "reason": "workspace read browsing only; no write API — writes go through in-turn tools (design §1.3 v2)" + }, + "gitOperations": { + "level": "partial", + "missing": ["git-diff", "git-commit", "git-branch"], + "reason": "read-only metadata + review link; change mutation is outside this package (same discipline as v1's read-only Git facade)" + } + }, + "capabilitiesProvider": "local-runtime-v2", + "capabilitiesProviderFor": "transport", + "capabilitiesUnavailable": { + "none": ["updateCheck"], + "partial": [ + { "key": "subagents", "missing": ["getDelegationSnapshot", "stopDelegation"] }, + { "key": "fileReadWrite", "missing": ["file-write"] }, + { "key": "gitOperations", "missing": ["git-diff", "git-commit", "git-branch"] } + ] }, "notes": { "set_mode": "Takes a modeId from the session's availableModes.", @@ -2268,6 +2310,21 @@ code, killEndpoint: "/api/stop" }`。温和版→SIGKILL 的级联 `mcodeVersion` 在尚无客户端挂接(还没收到 `initialize` 应答) 时为 `"unknown"`;本端点不会臆造一个版本号。 +`capabilitiesProvider` 是应答了的那份声明所属的 provider, +`capabilitiesProviderFor` 说明它是**怎么**被选中的。消费方应当对后者 +分支: + +- `"transport"`——当前 `MCODE_WEBUI_TRANSPORT` 自己的已注册 provider + 应答的。 +- `"default"`——尚无任何 provider 声明该传输(M4 引入),由默认 + provider 的声明顶替。这份视图仍是一份真实且经评审的声明,但它未必 + 是已连接引擎的那份;把它当成后者报出去就是撒谎。 + +`capabilitiesUnavailable` 是能力驱动型 UI 据以渲染的降级摘要:`none` +的键意味着隐藏整个入口,`partial` 的键意味着恰好隐藏或禁用列出的那些 +子动作。它是唯一一个并非声明本身的字段,消费方不该被迫从一个有三级 +两可选字段的分类法里重新推导它。 + --- ## 授权决策 diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index b75babf0..bcd4a159 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -489,8 +489,8 @@ not import it but adopts the same shape. Unknown future statuses render as ### `engine/` (capability declarations + the local-runtime-v2 host) The engine abstraction lives at `server/engine/` (engine-abstraction -batch B1; migration state M1, plus M3 batches B0, B1, B2 and B3). Eleven -files, one job each: +batch B1; migration state M1, plus M3 batches B0, B1, B2, B3, B4 and B5). +Fifteen files, one job each: | File | Owns | | --- | --- | @@ -505,6 +505,10 @@ files, one job each: | `engine/session-tree-reads.js` | The session-tree family's facade call (`readEngineSessionTree`) and the endpoint→capability table `SESSION_TREE_ENDPOINTS` (step M3, batch B2). Gates **hard**: `assertSessionTreeCapability` throws → 501, because the tree is entirely engine data. Forwards to `lib/session-tree.js#getSessionTree`; the assembler is not duplicated | | `engine/session-export.js` | The export family's facade call (`readEngineSessionTranscript`) and the endpoint→capability table `SESSION_EXPORT_ENDPOINTS` (step M3, batch B2). Gates **soft**: `checkSessionExportCapability` reports and never throws, because export's primary source is `sessions.json`, not the engine | | `engine/usage-reads.js` | The usage family's facade calls (`readEngineAccountQuota`, `readEngineSessionUsage`, `readEngineQuotaForecast`), the derived figure `contextUsedTokens`, and the endpoint→capability table `USAGE_READ_ENDPOINTS` (step M3, batch B3). Gates **hard** on the two engine reads and declares **no capability at all** for #19, which touches no engine surface | +| `engine/account-reads.js` | The account family's facade call (`readEngineAccount`) and the endpoint→capability table `ACCOUNT_READ_ENDPOINTS` (step M3, batch B4). Gates **hard** on `authCredentials` · `getAccountStatus` — the same pair and the same provider method as `engine/usage-reads.js`, because #20 and #15/#16 read the same engine projection. Its read is **asynchronous** and it lives under the ordinary `await import()` boot-path rule | +| `engine/model-reads.js` | The model-catalogue family's facade call (`readEngineModelCatalogue`), the whole projection as named pure functions (`projectModelCatalogue`, `deriveModelSelection`, `buildModelCataloguePayload`, `catalogueSourceLabel`, `webuiFullModelId`, `providerOfModelId`, `attachContextWindowOptions`, `configOption`), and the endpoint→capability table `MODEL_READ_ENDPOINTS` (step M3, batch B4). Gates **soft**: `checkModelReadCapability` reports and never throws, because the catalogue's primary sources are files webui owns. Its read is **synchronous**, and it is the one engine module **not** re-exported from `engine/index.js` — see the boot-path note below | +| `engine/capability-reads.js` | The capability-declaration family's facade call (`readEngineCapabilityView`) and the endpoint→capability table `CAPABILITY_READ_ENDPOINTS` (step M3, batch B4). Declares **no capability for #73** — it IS the declaration endpoint, and gating the gate would let a `none` hide the declaration that says so. It is the only endpoint in the migration whose response CONTRACT changed (`capabilities` is now the 14-key declaration, replacing the ACP wire table) | +| `engine/session-writes.js` | The session WRITE family's facade calls (`planEngineSessionDelete`, `commitEngineSessionDelete`, `commitEngineOrphanSessionDelete`, `previewEngineSessionDelete`, `applyEngineSessionRename`, `readOrphanSessionWriteIds`), the pure derivations they are built from (`resolveSessionTarget`, `isMcodeSessionId`, `isOrphanSessionRecord`, `selectOrphanSessionIds`, the two fan-out predicates, the per-client state resets), and the endpoint→capability table `SESSION_WRITE_ENDPOINTS` (step M3, batch B5). Gates **hard** on `sessionCrud` · `deleteSession` for #7 and #6, and declares **no capability at all** for #4. Forwards the 32-table SQL to `lib/mcode-session-delete.js` rather than moving it — see the write-path section below | Routes take the host from the facade and never from `lib/acp-client.js`: `routes/plugins.js` and `routes/turn-diff.js` call @@ -583,13 +587,43 @@ everything it imports statically must stay free of `@mavis/*`, declaration and construction were split). `test/lib/engine/host-facade.test.js` enforces it against the real module graph rather than against source text. `engine/session-reads.js`, `engine/session-tree-reads.js`, -`engine/session-export.js` and `engine/usage-reads.js` all live under the +`engine/session-export.js`, `engine/usage-reads.js`, +`engine/account-reads.js`, `engine/capability-reads.js` and +`engine/session-writes.js` all live under the same rule: their static imports are `engine/capabilities.js` and `engine/index.js` only, and every heavier dependency — `lib/acp-client.js`, `lib/config.js`, `lib/session-tree.js`, -`lib/transcript.js`, `lib/usage.js`, `lib/mavis-usage.js` and -`lib/quota-forecast.js` — is reached through `await import()` inside the -functions. +`lib/transcript.js`, `lib/usage.js`, `lib/mavis-usage.js`, +`lib/quota-forecast.js`, `lib/mcode-rpc.js` — is reached through +`await import()` inside the functions. `engine/session-writes.js` adds +`node:fs` at module scope (a builtin, and `engine/usage-reads.js` +already does the same) and reaches `lib/sessions.js`, +`lib/mcode-session-delete.js`, `lib/state-bus.js` and +`lib/config.js` dynamically — all six of its storage dependencies, which +is what lets it be re-exported from `engine/index.js` at all. + +`engine/model-reads.js` is the one deliberate exception, and it deviates on +**both** sides of the import. Its four sources — `lib/config.js`, +`lib/engine-catalogue.js`, `lib/models.js`, `lib/providers-config.js` — are +static imports, because `routes/model.js` already imported all four +**before** M3-B4 and the server's boot cost is therefore exactly what it +was. They reach `@mavis/shared/local-runtime-paths` (via `lib/config.js`) +and `js-yaml` (via `engine-provider-sync.js`), so the module is deliberately +**not** re-exported from `engine/index.js`: making the shared facade — the +one import site the whole server shares, and the one `routes/plugins.js` +must stay light through — heavier than it has ever been would buy nothing. +`routes/model.js` therefore imports `../engine/model-reads.js` directly, +the same shape `routes/protocol.js` already uses for `engine/session-reads.js`. +`test/lib/engine/host-facade.test.js` is the gate that forced this, and it +is right to. + +The price is a **synchronous** read. Making the four imports dynamic would +let the module re-export from the facade again, at the cost of turning +`handleGetModels` into an async handler — a contract change for any caller +that does not await, and the one thing this batch promises not to do. When +the catalogue read becomes async (M4, with a provider-backed source) the +module can move back behind `await import()` and be re-exported with the +rest. #### Which endpoints read through the facade (step M3, batch B1) @@ -749,6 +783,102 @@ reports `_meta.mcode_unavailable: true` with `_meta.source: "webui"`. That is pre-existing and deliberately preserved — re-enabling it is a behaviour change for a later slice, not a refactor. +#### Which endpoints read through the facade (step M3, batch B4) + +Batch B4 adds three endpoints, and they are the first three whose gate +policies are **all different from each other**: one hard, one soft, one +declared-as-nothing. Three modules, for the reason B2 gave — a shared table +would force one family to inherit another's policy. + +| Endpoint | Capability · sub-item | Enforcement | Value source | +| --- | --- | --- | --- | +| `GET /api/account` | `authCredentials` · `getAccountStatus` | hard — 501 | `lib/mcode-rpc.js#getAccountStatus`, the engine's `mcode/account/status` projection. The response body is built by the facade: `{ok:true, ...data}` on success, `{ok:false, reason}` at HTTP 200 otherwise | +| `GET /api/models` | `authCredentials` · `listModelProviders` | soft — reported | three layered sources: the engine session's `model` config option, the merged providers config (webui `env > cwd > user` over the engine's `custom_provider` tree, via `lib/engine-catalogue.js`), and the builtin cli-bundle extraction | +| `GET /api/protocol/capabilities` | none of the 14 keys | none — the gate is a reported no-op | the registered provider's 14-key declaration, its `summarizeUnavailableCapabilities` roll-up, and the ACP `initialize` `agentInfo` mirror | + +**Why #20 gates hard and #57 does not.** The account card is 100% engine +data: there is no webui-side fallback for "who am I" or for a plan tier, so +a provider that cannot report an account has nothing to return and 501 is the +honest answer. The model catalogue is not: its primary sources are files +webui owns and can read without the engine — `models.json`, +`~/.mcode-webui/providers.json`, and a cli-bundle extraction — plus the +engine's own `config.yaml`. Gating #57 hard would delete a working picker in +response to a declaration about a capability it does not depend on, which is +the same reasoning `engine/session-export.js` records for #11. So +`checkModelReadCapability` reports and returns; the read is unaffected by +what it reports. + +**Why #73 declares nothing.** It is the declaration endpoint. A gate on it +would be circular, and a `none` anywhere in the declaration could hide the +declaration that says so — the same reason B1's `/api/health` and B3's +`/api/usage/forecast` declare no capability. `checkCapabilityReadCapability` +is exported anyway, so the symmetry with the other families is visible and +testable. + +Four properties this batch holds, each with a test behind it: + +1. **#57 is a full snapshot, and the oracle is the pre-refactor code.** + `test/lib/engine/model-reads.test.js` projects one rich fixture — engine + session option, engine `custom_provider` layer, webui config layer, + builtin layer, a builtin that **collides** with a config entry, a + switchable variant model, an effort-list model, a `forced_on` model, two + providers with overlapping upstream model ids, one provider with a key + and one without — and compares the whole response body, field for field + and key for key, against a literal captured from `3362c9be`. The oracle + is not recomputed by the functions under test. The load-bearing part is + what is **absent**: the config layer takes the `minimax_api/MiniMax-M3` + slot wholesale, so that entry appears once, with the operator's label and + `contextLimit`, and **without** the builtin's `thinkingLevels` and + `contextWindowOptions`. +2. **Grouping is by provider, and the dedupe is per provider.** The webui id + is always `/`, even when the upstream model id + already contains `/` (ticket 09-02). `nousresearch/z-ai/glm-5.3` and + `zai-max/z-ai/glm-5.3` are two rows in two groups; the previous + behaviour let one swallow the other. The builtin shell is keyed by + `minimax_api` **regardless of the recorded pick**, which is the + "8 config + 6 misplaced builtins = 14 in `nousresearch`" replay. +3. **The two builtin-tree projections reach two sites, and a miss is a miss.** + `readEngineBuiltinThinking` and `readEngineBuiltinContextWindows` are two + views of `provider.minimax.models`, read once per request and consumed at + the engine-session site (keyed by the wire form's **bare** model id) and at + the builtin shell. A wire form whose model segment does not parse, or a + model absent from the tree, produces a field-free entry — never a + half-annotation. The section that perturbs the tree asserts which entries + move for which record. +4. **#73's contract CHANGED, deliberately, and the declaration appears + once.** `capabilities` used to be `MCODE_ACP_CAPABILITIES`, a + hand-maintained flat `{method: boolean}` table of the ACP JSON-RPC + surface; it is now the engine's **declared** 14-key object, forwarded by + identity. The twelve old accessors are asserted gone, so a consumer + reading `capabilities.set_mode` gets `undefined` and fails loudly + rather than receiving a truthy object field. This is the one + user-authorised endpoint contract change in the migration, and the + first shape of it — an additive `engine` block carrying the view + beside the old table — was rejected in review precisely because it + would have carried the same 14 keys twice in one response. What + survives from that shape is the provenance, hoisted to + `capabilitiesProvider` / `capabilitiesProviderFor`, plus + `capabilitiesUnavailable` for the derived roll-up. The test counts the + declaration's occurrences structurally, so re-introducing a second + carrier is a red bar. `providerFor` is the honest bit: a + capability-detection endpoint must not report a standing-in + declaration as though it were the connected engine's, and under the + default `acp` transport that standing-in is the normal case until M4. + `docs/API.md`, `docs/webui.md` and `docs/tui-capabilities.md` all + record the new shape in both languages. + +**The three "what is active" figures are derived once.** `current` prefers +the engine's `currentValue` and falls back to the recorded pre-session pick; +`currentThinking` prefers the engine's `thinkingEffort` option; and +`currentContextWindow` is the recorded window with the current model's +catalogue `contextLimit` as the fallback. When neither exists the answer is +`null`, never a default model — the old behaviour invented an active model +the engine never confirmed and the composer chip claimed it. + +**`handleGetModels` is still a synchronous handler.** The facade read is +synchronous too, and the test asserts it: the body must be complete when the +handler returns, because that is what the pre-M3 handler guaranteed. + ## 4. The `clientState` payload This is the shape every SSE `state` event contains. The webui mirrors @@ -875,6 +1005,116 @@ same event is safe. The server uses an at-most-once delivery model (SSE drops on disconnect → no retry), which the client handles by fetching `/api/state` on reconnect. +#### Which endpoints write through the facade (step M3, batch B5) + +Batch B5 is the first family in the migration whose endpoints **destroy** +data rather than read it, and that changes what the gate question is +asking. For a read, hard or soft is decided by "is the data the engine's +or webui's". For a write it is decided by **who owns the rows the write +destroys** — and in this family that question does not have the same +answer twice in a row. + +| Endpoint | Capability · sub-item | Enforcement | Value source | +| --- | --- | --- | --- | +| `DELETE /api/sessions/:id` (#7) | `sessionCrud` · `deleteSession` | hard — 501 | the webui session store, the in-memory ACP session cache, the sidebar tree cache, and the engine's own `local_runtime_*` rows via `lib/mcode-session-delete.js` | +| `POST /api/sessions/rename` (#4) | none of the 14 keys | none — the gate is a reported no-op | webui's own session store, and nothing else. The engine's title is not written | +| `POST /api/sessions/cleanup-orphans` (#6) | `sessionCrud` · `deleteSession` | hard — 501 | the same store, plus each selected id delegated to #7, so it reaches the same engine rows | + +**Why #7 and #6 gate hard.** Both destroy rows in the engine's own +`local_runtime_*` tables, and there is no webui-side copy of a transcript +that survives: once those rows are gone, the conversation is gone. A +provider that declares no session deletion genuinely cannot have these +endpoints serve a truthful answer, so 501 is the honest one. #6 +deliberately declares the *same* pair as #7 — the sweep selects webui-side +orphan records, but each selected id goes through #7's real-delete branch, +and a record carrying an `mcodeSessionId` takes the engine's rows with it. +Gating the sweep soft would let a provider that cannot delete engine +sessions reach those tables through a back door, and would also produce a +worse failure than a 501: an authorized destructive sweep that writes its +intent audit event and then fails every single delegated delete. + +**Why #4 declares nothing.** Rename writes `title` / `titleCustom` / +`updatedAt` into webui's own store and touches no engine surface at all. +Its one engine touch is `invalidateSessionTree()` — a cache drop, which is +the read-side consequence of the sidebar projecting titles from the engine, +and that projection is B2's `GET /api/session-tree` with its own gate. +Naming a capability here would be a lie of the kind B3 declined for +`GET /api/usage/forecast`: gating a working endpoint on a declaration +about something it does not depend on. + +This family also deviates from its siblings in one deliberate way: every +row of `SESSION_WRITE_ENDPOINTS` carries the same three keys — +`capability`, `subItem`, `enforcement` — including the row that has no +capability. B3 expressed "no engine surface" as a `null` table entry; +here two of three endpoints *do* cross the seam, and a `null` hole in the +middle of the table reads like "not filled in yet" rather than like a +decision. The gate **descriptor** keeps the six fields every family +returns, plus `enforcement`. + +**The plan/commit split, and why the route did not shrink to nothing.** +#7 is exported as a pair rather than one `deleteSession(options)`: + +1. `planEngineSessionDelete` resolves the id and runs the gate. It + mutates nothing, so it is safe to run *before* the user is asked + anything. +2. `authorize()` and the write-ahead `session.delete.intent` audit happen + **between** the plan and the commit. The intent line has to be durably + recorded before any row is removed, and it records the match kind and + chat length the plan produced. +3. `commitEngineSessionDelete` / `commitEngineOrphanSessionDelete` / + `previewEngineSessionDelete` perform the write and fan-out. + +A facade that owned the whole operation would have had to swallow that +ordering into a callback. The route keeps request parsing, the authorize +modal, the audit ordering and every status code; the facade keeps the +sequencing, the gate and the response bodies. + +**The ordering inside a commit is the feature, and it is asserted as a +sequence.** `test/lib/engine/session-writes.test.js` journals every +mutation and asserts the order, because an end-state assertion cannot see +a resurrected session: + +``` +invalidate-tree → kill-acp-child → drop-cache: → sql: → push: +``` + +The tree cache is dropped *before* the engine write so a concurrent read +cannot repopulate it from the pre-delete database. The ACP child is +stopped *before* the rows are removed, because it holds the session in +memory and rewrites its registry row on its next request — that is the +"deleted session reappears" bug. Only the **one** deleted sid leaves the +cache: invalidating the whole cache empties the sidebar, refills it, and +reads to the user like the delete failed. + +**The 32-table SQL was not moved, and that is recorded rather than +quietly dropped.** The plan for this batch annotated +`lib/mcode-session-delete.js` "delete". It is kept because +`lib/acp-client.js` imports `deleteMcodeSessionFromDb` from it and four +test files bind to that specifier; collecting it means moving those +first. The facade reaches it through `await import()` and issues no SQL +of its own — the same split B3 drew for `lib/mavis-usage.js` and B4 for +`lib/mcode-rpc.js`. A test asserts both halves: the table list is still +32 entries exported from the lib module, and the facade contains no SQL +verb at all. + +**Three things this batch records as known debt instead of deciding:** + +1. The 32-table SQL is still in `lib/mcode-session-delete.js`, for the + consumer reasons above. +2. A rename is a **webui-side label only**. The engine's own title in + `local_runtime_sessions` is untouched while the sidebar tree reads its + titles from the engine, so for an engine-backed session a rename can be + visible in the wrapper list and not in the tree. This is pre-existing + behaviour and the batch did not change it; closing it means deciding + which store is authoritative for a display title, which is a product + call. +3. #7 does not detect "this session is running right now". Deleting an + in-flight session stops the ACP child out from under the turn and then + proceeds. That is the pre-facade behaviour and arguably the correct + one (the user asked), but refusing to delete a running session is a + defensible alternative and the choice is not the batch's to make. A + test pins the semantics that exist so the behaviour is at least stated. + ## 6. Frontend topology ``` diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index 3bce7f75..60242458 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -461,7 +461,7 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 ### `engine/`(能力声明 + local-runtime-v2 host) 引擎抽象层位于 `server/engine/`(engine-abstraction 批次 B1;迁移 -状态 M1,外加 M3 的 B0、B1、B2 与 B3 四批)。十一个文件,各管一件事: +状态 M1,外加 M3 的 B0、B1、B2、B3 与 B4 五批)。十四个文件,各管一件事: | 文件 | 职责 | | --- | --- | @@ -476,6 +476,9 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 | `engine/session-tree-reads.js` | 会话树族的面板调用 `readEngineSessionTree` 与端点→能力对照表 `SESSION_TREE_ENDPOINTS`(迁移步 M3 批次 B2)。**硬门控**:`assertSessionTreeCapability` 抛出 → 501,因为树完全由引擎数据构成。转发到 `lib/session-tree.js#getSessionTree`,树的装配逻辑不复制第二份 | | `engine/session-export.js` | 导出族的面板调用 `readEngineSessionTranscript` 与端点→能力对照表 `SESSION_EXPORT_ENDPOINTS`(迁移步 M3 批次 B2)。**软门控**:`checkSessionExportCapability` 只报告、从不抛出,因为导出的主数据源是 `sessions.json` 而非引擎 | | `engine/usage-reads.js` | 用量族的面板调用(`readEngineAccountQuota`、`readEngineSessionUsage`、`readEngineQuotaForecast`)、派生量 `contextUsedTokens`,与端点→能力对照表 `USAGE_READ_ENDPOINTS`(迁移步 M3 批次 B3)。两个引擎读**硬门控**;#19 **完全不声明能力**,因为它不触达任何引擎面 | +| `engine/account-reads.js` | 账户族的面板调用 `readEngineAccount` 与端点→能力对照表 `ACCOUNT_READ_ENDPOINTS`(迁移步 M3 批次 B4)。**硬门控**,门控在 `authCredentials` · `getAccountStatus`——与 `engine/usage-reads.js` 同一对、同一个 provider 方法,因为 #20 与 #15/#16 读的是同一份引擎投影。它的读是**异步的**,服从普通的 `await import()` 启动路径纪律 | +| `engine/model-reads.js` | 模型目录族的面板调用 `readEngineModelCatalogue`、整套投影的具名纯函数(`projectModelCatalogue`、`deriveModelSelection`、`buildModelCataloguePayload`、`catalogueSourceLabel`、`webuiFullModelId`、`providerOfModelId`、`attachContextWindowOptions`、`configOption`),与端点→能力对照表 `MODEL_READ_ENDPOINTS`(迁移步 M3 批次 B4)。**软门控**:`checkModelReadCapability` 只报告、从不抛出,因为目录的主数据源是 webui 自己拥有的文件。它的读是**同步的**,并且它是唯一一个**没有**从 `engine/index.js` 转发导出的引擎模块——见下面的启动路径说明 | +| `engine/capability-reads.js` | 能力声明族的面板调用 `readEngineCapabilityView` 与端点→能力对照表 `CAPABILITY_READ_ENDPOINTS`(迁移步 M3 批次 B4)。#73 **不声明任何能力**——它本身就是声明端点,给门控上门控会让某个 `none` 把声明它的那份声明藏起来。它是本次迁移中唯一一个响应**契约**发生变更的端点(`capabilities` 现在是 14 键声明,顶替了 ACP wire 表) | 路由从门面取 host,不从 `lib/acp-client.js` 取:`routes/plugins.js` 与 `routes/turn-diff.js` 调 `getEngineCatalogueHost()`。两者都保留 `deps` @@ -540,12 +543,33 @@ handler 层测试因此保持封闭。 门面自身加载 4685ms → 5ms)。`test/lib/engine/host-facade.test.js` 对着真实模块图强制它,而不是对着源码文本。 `engine/session-reads.js`、`engine/session-tree-reads.js`、 -`engine/session-export.js` 与 `engine/usage-reads.js` 全部服从同一条 +`engine/session-export.js`、`engine/usage-reads.js`、 +`engine/account-reads.js` 与 `engine/capability-reads.js` 全部服从同一条 纪律:静态 import 只有 `engine/capabilities.js` 与 `engine/index.js`, 而每个更重的依赖——`lib/acp-client.js`、`lib/config.js`、 `lib/session-tree.js`、`lib/transcript.js`、`lib/usage.js`、 -`lib/mavis-usage.js` 与 `lib/quota-forecast.js`——都在函数体内用 -`await import()` 触达。 +`lib/mavis-usage.js`、`lib/quota-forecast.js`、`lib/mcode-rpc.js`——都在 +函数体内用 `await import()` 触达。 + +`engine/model-reads.js` 是唯一一处刻意例外,而且它在 import 的**两侧** +都刻意偏离。它的四个数据源——`lib/config.js`、 +`lib/engine-catalogue.js`、`lib/models.js`、`lib/providers-config.js`—— +是静态 import,因为 M3-B4 之前 `routes/model.js` 就静态 import 了这四个, +所以 server 的启动成本分文未增。但它们会经 `lib/config.js` 抵达 +`@mavis/shared/local-runtime-paths`、经 `engine-provider-sync.js` 抵达 +`js-yaml`,所以这个模块**刻意没有**从 `engine/index.js` 转发导出:让 +共享门面——整个 server 唯一的共享 import 站点,也是 +`routes/plugins.js` 必须保持轻量的那个——比它历来更重,换不来任何东西。 +因此 `routes/model.js` 直接 import `../engine/model-reads.js`,这与 +`routes/protocol.js` 对 `engine/session-reads.js` 的写法同形。 +`test/lib/engine/host-facade.test.js` 正是逼出这个决定的那道门禁,而它 +是对的。 + +代价是一次**同步**读。把那四个 import 改成动态的,就能让这个模块重新 +被门面前转发,代价是把 `handleGetModels` 变成异步处理器——这对任何不 +await 的调用方都是契约变更,也正是本批承诺不做的那件事。等目录读变成 +异步时(M4,接上 provider 支撑的数据源),这个模块就可以退回 +`await import()` 之后,与其余各族一起被转发导出。 #### 哪些端点走门面读(迁移步 M3 批次 B1) @@ -678,6 +702,86 @@ provider 确实没有树可返回,501 才是诚实答案。 `_meta.source: "webui"`。这是既有行为且被刻意保留——重新启用它是一次行为 变更,属于后续切片,不属于这次收编。 +#### 哪些端点走门面读(迁移步 M3 批次 B4) + +批次 B4 加入 3 个端点,它们是首批**门控策略彼此全都不同**的三个: +一个硬门控、一个软门控、一个声明为「什么都不声明」。因此是三个模块, +理由与 B2 相同——共用一张表会逼其中一族继承另一族的策略。 + +| 端点 | 能力 · 子项 | 强制方式 | 取值来源 | +| --- | --- | --- | --- | +| `GET /api/account` | `authCredentials` · `getAccountStatus` | 硬——501 | `lib/mcode-rpc.js#getAccountStatus`,即引擎的 `mcode/account/status` 投影。响应体由门面组装:成功是 `{ok:true, ...data}`,失败在 HTTP 200 上是 `{ok:false, reason}` | +| `GET /api/models` | `authCredentials` · `listModelProviders` | 软——只报告 | 三个分层来源:引擎会话的 `model` 配置项、合并后的 provider 配置(webui 的 `env > cwd > user` 叠在引擎 `custom_provider` 树之上,经 `lib/engine-catalogue.js`)、以及内建 cli 包抽取 | +| `GET /api/protocol/capabilities` | 14 个键里的任何一个都不适用 | 不门控——门控是「被报告的空操作」 | 已注册 provider 的 14 键声明、它的 `summarizeUnavailableCapabilities` 汇总,以及 ACP `initialize` 的 `agentInfo` 镜像 | + +**为什么 #20 硬门控而 #57 不硬。** 账户卡 100% 由引擎数据构成: +「我是谁」和「什么套餐」都没有 webui 侧的兜底,所以报不出账户的 +provider 确实无物可报,501 才是诚实答案。模型目录不是:它的主数据源是 +webui 自己拥有、不依赖引擎就能读的文件——`models.json`、 +`~/.mcode-webui/providers.json`、cli 包抽取——再加上引擎自己的 +`config.yaml`。对 #57 硬门控,等于用一份它并不依赖的能力声明去删掉一个 +能用的选择器,这与 `engine/session-export.js` 为 #11 记下的理由同源。所以 +`checkModelReadCapability` 只报告然后返回;这次读不受它报告结果的影响。 + +**为什么 #73 什么都不声明。** 它就是声明端点。给它上门控是循环论证,而且 +声明里任何一处 `none` 都能把声明它的那份声明藏起来——这与 B1 的 +`/api/health`、B3 的 `/api/usage/forecast` 不声明能力同源。即便如此 +`checkCapabilityReadCapability` 仍然导出,好让与其他各族的对称关系可见、 +可测。 + +本批持有的四条性质,每条背后都有一个测试: + +1. **#57 是全量快照,且预言机取自收编前的代码。** + `test/lib/engine/model-reads.test.js` 用一套内容丰富的 fixture 做投影 + ——引擎会话配置项、引擎 `custom_provider` 层、webui 配置层、内建层、 + 一个与配置项**撞 id** 的内建模型、一个可切换 variant 模型、一个 + effort 列表模型、一个 `forced_on` 模型、两个上游模型 id 重叠的 + provider、一个有 key 与一个没 key 的 provider——并把整个响应体逐字段、 + 逐键地与一份从 `3362c9be` 抓下来的字面量比对。预言机不是被测函数自己 + 算出来的。承重的是**缺席**的那部分:配置层整体接管了 + `minimax_api/MiniMax-M3` 这个位置,所以该条目只出现一次,带着运维的 + label 与 `contextLimit`,而**没有**内建模型的 `thinkingLevels` 与 + `contextWindowOptions`。 +2. **分组按 provider,去重也按 provider。** webui id 恒为 + `/`,即使上游模型 id 本身已含 `/` + (ticket 09-02)。`nousresearch/z-ai/glm-5.3` 与 + `zai-max/z-ai/glm-5.3` 是两组里的两行;旧行为会让其中一个吞掉另一个。 + 内建外壳**无论当前记录选了什么**都归到 `minimax_api`——这正是 + 「8 个配置 + 6 个错位的内建 = `nousresearch` 里 14 个」那次回放的 + 结论。 +3. **两棵内建树投影会抵达两个站点,而「查不到」就是查不到。** + `readEngineBuiltinThinking` 与 `readEngineBuiltinContextWindows` 是 + `provider.minimax.models` 的两个视图,每次请求读一次,分别在引擎会话 + 站点(按 wire 形式的**裸**模型 id 查)与内建外壳处被消费。wire 形式的 + 模型段解析不出来、或模型不在树里,产出的就是一个无这些字段的条目, + 绝不会是「半吊子标注」。扰动那棵树的那一节断言了:哪条引擎记录会让 + 哪些条目发生变化。 +4. **#73 的契约是「变更」了,且是刻意的,声明只出现一次。** `capabilities` + 过去是 `MCODE_ACP_CAPABILITIES`——一张手工维护的扁平 `{方法: 布尔}` + 表,描述 ACP JSON-RPC 面;现在是引擎**声明的** 14 键对象,按引用 + 转发。那 12 个旧访问器被断言为**已消失**,所以读 + `capabilities.set_mode` 的消费方拿到 `undefined`、响亮地失败,而不是 + 收到一个真值对象字段。这是本次迁移里唯一一处经用户授权的端点契约 + 变更;它的第一个形状——在旧表旁边增一个 `engine` 块承载视图——在评审 + 中被否掉,正因为那会让同一份 14 键声明在一次响应里出现两次。留下来的是 + 出处信息,上提为 `capabilitiesProvider` / + `capabilitiesProviderFor`,外加派生汇总 + `capabilitiesUnavailable`。测试用结构化计数断言声明的出现次数,所以再 + 引入第二个承载者就是一条红条。`providerFor` 是诚实位:能力探测端点 + 绝不能把顶替声明当作已连接引擎的声明报出去,而在默认 `acp` 传输下, + 直到 M4 之前这种顶替都是常态。`docs/API.md`、`docs/webui.md`、 + `docs/tui-capabilities.md` 都以两种语言记录了新形状。 + +**三个「当前生效」的量只派生一次。** `current` 优先取引擎的 +`currentValue`,回落到记录在案的会话前选择;`currentThinking` 优先取引擎的 +`thinkingEffort` 配置项;`currentContextWindow` 是记录在案的窗口,回落 +到当前模型在目录里的 `contextLimit`。两者都没有时答案是 `null`,而不是 +某个默认模型——旧行为会凭空造出一个引擎从未确认的活跃模型,而 composer +的芯片会把它当成正在跑的模型宣称出去。 + +**`handleGetModels` 仍是同步处理器。** 门面的读同样是同步的,测试对此有 +断言:处理器返回时响应体必须已经写完,因为这是 M3 之前处理器给出的保证。 + ## 4. `clientState` 载荷 这是每个 SSE `state` 事件所包含的形状。webui 将其 diff --git a/packages/webui/scripts/check-docs-alignment.mjs b/packages/webui/scripts/check-docs-alignment.mjs index ce22bf06..e53a2d88 100644 --- a/packages/webui/scripts/check-docs-alignment.mjs +++ b/packages/webui/scripts/check-docs-alignment.mjs @@ -509,6 +509,8 @@ const NOT_ON_DISK = new Set([ "server/routes/foo.js", // the illustrative path in §9's recipe "sessions.json", // runtime data under WEBUI_DATA_DIR, not a source file "mcp.json", // user-authored MCP server config, not a source file + "models.json", // operator-authored provider catalogue (cwd layer), not a source file + "config.yaml", // the ENGINE's own config under its data dir, not a source file "index.html", // Next export output (webapp/out/index.html), not a source file ]); diff --git a/packages/webui/server/engine/account-reads.js b/packages/webui/server/engine/account-reads.js new file mode 100644 index 00000000..659bf562 --- /dev/null +++ b/packages/webui/server/engine/account-reads.js @@ -0,0 +1,228 @@ +// webui/server/engine/account-reads.js +// +// Migration step M3, batch B4: the account read (账户读) — +// +// #20 GET /api/account — the account card's identity / plan-tier data +// +// What this file is for. #20 is a small endpoint with a strict privacy +// contract: the payload is the user's own display name, plan tier and +// quota, it is fetched ON DEMAND rather than pushed into the state +// snapshot (the snapshot is broadcast to every SSE subscriber, LAN +// included), and the engine's projection carries no credential. The +// route therefore has to stay a one-liner that writes a body and never +// accumulates identity state — and that is exactly what a facade read +// gives it. After M3-B4 the route asks this file, this file asks the +// provider whether it may, and only then forwards to +// `lib/mcode-rpc.js#getAccountStatus`. +// +// Why the gate is HARD here while #57 and #73 are not. This endpoint is +// 100% engine data: there is no webui-side fallback for "who am I" and +// no webui-side fallback for the plan tier. The empty state the card +// renders when the engine cannot be reached is a RUNTIME outcome +// (HTTP 200 + `{ok:false, reason}`), which this file preserves +// verbatim; a provider that declares no `getAccountStatus` is a +// different, structural outcome, and the only honest answer to it is +// the 501 that `app.js#invokeHandler` derives from +// `EngineCapabilityNotSupportedError`. Same rule, same pair, same +// provider method as B3's `POST /api/usage` / `POST /api/usage-trigger` +// — the usage popover and the account card read the SAME engine +// projection through the SAME `mcode/account/status` extension method, +// so a declaration that removes it must take both down together. Two +// modules, not one: the usage family owns the quota DERIVATIONS +// (`contextUsedTokens`, the least-squares forecast) and the usage +// family has its own gate policy for #19; merging them would force one +// to inherit the other's. +// +// What this file deliberately does NOT do: +// +// - It does not reshape the engine's projection. `r.data` is spread +// into the response verbatim (`{ok:true, ...r.data}`), so a new +// engine field reaches the card without a webui edit, and an +// absent one does not become a `null` this layer invented. +// - It does not invent a reason. The failure body is +// `{ok:false, reason: r.code || "account_unavailable"}` — the +// endpoint's own fallback, kept byte-for-byte. The account card +// (`webapp/components/shell.tsx#SidebarFooter`) renders its +// 本地用户 placeholder on any failure, and it must keep doing so +// for the engine-could-not-be-reached case that has always produced +// it. +// - It does not construct a host. `getAccountStatus` goes through the +// process-singleton ACP client, the same path it has always taken. +// - It does not log the payload. Identity data must not reach a log +// line; the only logging this path can do is whatever +// `mcode-rpc.js#sanitizeError` already does to an error string. +// +// Boot-path weight. `app.js` imports `routes/account.js`, the route +// imports this file, so this file is on the boot path. It statically +// imports nothing heavier than `capabilities.js` and `index.js` (both +// pure declaration modules); `lib/mcode-rpc.js` and `lib/config.js` are +// reached through `await import()` inside the read. That split is the +// M1 lesson — putting the `@mavis/*` tree on the boot path once cost +// 209ms → 2700ms of server start and broke the integration tests' 3s +// window. +// +// Provider selection is M4's job, same as B1, B2 and B3: +// `providerByTransport()` maps a transport to a REGISTERED provider id; +// today only `runtime` has one, so under the default `acp` transport the +// gate reports `gate: "unregistered-transport"` and the read proceeds — +// which is correct, because the pre-M4 behaviour under `acp` is the +// only behaviour this endpoint has ever had. + +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, exactly as in + * `session-reads.js#providerByTransport`, + * `session-tree-reads.js#providerByTransport` and + * `usage-reads.js#providerByTransport`, which this mirrors rather than + * merges: the four families have separate read contracts and a shared + * table would force one of them to inherit another's policy. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. Every + * consumer of the table is a function anyway. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +/** + * The declaration this endpoint needs, and the sub-item it needs from + * that capability. + * + * `authCredentials` / `getAccountStatus` is the honest mapping, and it + * is deliberately the SAME pair `usage-reads.js` uses for #15 / #16: + * both endpoints read the engine's own account projection through the + * `mcode/account/status` extension method, so they depend on the same + * provider method and must be gated by the same declaration. Naming a + * different sub-item here would let a `partial` provider drop + * `getAccountStatus` from the account card while the usage popover + * still claimed to have it. + * + * @type {Readonly>} + */ +export const ACCOUNT_READ_ENDPOINTS = Object.freeze({ + "GET /api/account": { capability: "authCredentials", subItem: "getAccountStatus" }, +}); + +/** + * Resolve the provider that answers the account read on `transport`, or + * `null` when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveAccountReadProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Check the account read against the active provider's declaration. + * Throws `EngineCapabilityNotSupportedError` — which + * `app.js#invokeHandler` turns into 501 — when the declaration says + * the capability (or the exact sub-item) is absent. + * + * @param {string} endpoint A key of ACCOUNT_READ_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null}} + */ +export function assertAccountReadCapability(endpoint, transport) { + const need = ACCOUNT_READ_ENDPOINTS[endpoint]; + if (need === undefined) { + // Caller confusion, not an engine limitation — a plain Error so the + // HTTP layer never answers 501 for a typo in webui's own code. + const err = new Error( + `assertAccountReadCapability: "${endpoint}" is not part of the account family ` + + `(known: ${Object.keys(ACCOUNT_READ_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_account_read_endpoint"; + throw err; + } + const provider = resolveAccountReadProvider(transport); + if (!provider) { + return { + endpoint, + gate: "unregistered-transport", + provider: null, + capability: need.capability, + subItem: need.subItem, + }; + } + assertEngineCapability(provider.capabilities, need.capability, provider.id, need.subItem); + return { + endpoint, + gate: "checked", + provider: provider.id, + capability: need.capability, + subItem: need.subItem, + }; +} + +/** + * Where the account bytes came from. Always `account-status`: the read + * is the engine's `mcode/account/status` extension method, reached + * through the ACP client, under every transport. The value exists so a + * consumer never has to guess whether a webui-side fallback answered — + * there is none, and saying so in a field is cheaper than a reader + * assuming one. + * + * @typedef {"account-status"} AccountReadSource + */ + +/** + * The #20 (`GET /api/account`) read. + * + * `payload` IS the endpoint's response body, built here once so the + * route is a single `res.end(JSON.stringify(payload))` and the body has + * exactly one home: + * + * - success → `{ok:true, ...(r.data || {})}`. The engine's projection + * is spread verbatim, so `identity` / `tokenPlan` / any future field + * arrive exactly as the engine framed them, and an engine that + * answers `{ok:true, data:null}` still produces `{ok:true}` rather + * than a `TypeError` on the spread. + * - failure → `{ok:false, reason: r.code || "account_unavailable"}`. + * Soft by contract: the REQUEST succeeded, so the status stays 200 + * and the card renders its empty state. `r.code` is preferred + * because it is the engine's own machine-readable reason + * (`no_client`, `unauthorized`, …); the string fallback is the + * endpoint's own and predates every code. + * + * @param {object} [options] + * @param {object} [options.cs] The webui client state; only + * `cs.mcodeSessionId` is read, exactly as the route read it. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/account`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. Exists so tests can exercise both + * the `runtime` and the unregistered `acp` branch without mutating + * process env. + * @returns {Promise<{payload: object, source: AccountReadSource, gate: object, transport: string}>} + */ +export async function readEngineAccount(options = {}) { + const endpoint = options.endpoint || "GET /api/account"; + const [rpc, config] = await Promise.all([ + import("../lib/mcode-rpc.js"), + import("../lib/config.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertAccountReadCapability(endpoint, transport); + // `cs && cs.mcodeSessionId` is forwarded EXACTLY as the route used to + // compute it, including the `undefined` a missing ctx produces — + // `getAccountStatus` turns any falsy id into `{}`, and a test that + // pins the forwarded argument must see the same value the route sent. + const r = await rpc.getAccountStatus(options.cs && options.cs.mcodeSessionId); + const payload = r && r.ok + ? { ok: true, ...(r.data || {}) } + : { ok: false, reason: (r && r.code) || "account_unavailable" }; + return { payload, source: "account-status", gate, transport }; +} diff --git a/packages/webui/server/engine/capability-reads.js b/packages/webui/server/engine/capability-reads.js new file mode 100644 index 00000000..3c054ca7 --- /dev/null +++ b/packages/webui/server/engine/capability-reads.js @@ -0,0 +1,234 @@ +// webui/server/engine/capability-reads.js +// +// Migration step M3, batch B4: the capability-declaration read +// (能力声明读) — +// +// #73 GET /api/protocol/capabilities — "what can this engine do?" +// +// What this file is for, and why it is the odd one out in this batch. +// #73 is the endpoint the FRONTEND uses to decide which controls to +// enable, and until M3-B4 it answered from two places webui maintains +// by hand: +// +// - `MCODE_ACP_CAPABILITIES`, a flat `{method: boolean}` table in +// `lib/mcode-rpc.js` describing the ACP JSON-RPC surface; and +// - `getMcodeServerInfo()`, the ACP `initialize` mirror, for the +// engine's name / title / version. +// +// Neither is the engine's DECLARED capability surface. That surface +// already exists — it is the 14-key per-provider declaration in +// `engine/capabilities.js` and the registry in `engine/index.js`, and +// `GET /api/engine-capabilities` already serves it. So webui was +// carrying two parallel answers to "what can the engine do", able to +// disagree, with no test able to notice. After M3-B4 `capabilities` IS +// the engine-capabilities view: the route no longer reaches into +// `lib/mcode-rpc.js` and `lib/acp-client.js` on its own, and there is +// one answer rather than two. +// +// The ACP wire table is REPLACED, not kept alongside — a reviewed, +// user-authorised endpoint contract change, not a refactor side effect. +// `MCODE_ACP_CAPABILITIES` described a different taxonomy (which ACP +// JSON-RPC method exists) and it had drifted into being the endpoint's +// headline field while nothing in the webapp read it. Carrying both +// would have meant the 14-key declaration appeared twice in one +// response, once as the answer and once as a decoration, so the extra +// `engine` key this batch first shipped was removed rather than kept. +// What survives from that first shape is the honest provenance — which +// provider answered, and whether it was standing in — hoisted to +// `capabilitiesProvider` / `capabilitiesProviderFor`. +// +// KNOWN DEBT, recorded rather than acted on: `MCODE_ACP_CAPABILITIES` +// in `lib/mcode-rpc.js` now has no consumer. It is still exported and +// still pinned by `test/lib/mcode-rpc.check.mjs`, and `docs/CAPABILITIES.md` +// cites it as a fact about the ENGINE's ACP surface (which it still +// is), so deleting it is a separate decision about dead code, not a +// side effect of replacing a response field. `test/helpers/_setup.js` +// mirrors the export for the same reason. +// +// Why this endpoint declares NO capability. It is the declaration +// endpoint: gating the gate is circular, and a `none` anywhere in the +// declaration must not be able to hide the declaration that says so. +// The value is `null` for the same reason B1's `/api/health` and B3's +// `/api/usage/forecast` are, and the gate REPORTS the no-op rather +// than passing silently. `checkCapabilityReadCapability` is exported so +// the symmetry with the other families is visible and testable, and so +// a future batch that adds a REAL capability-gated sibling has a +// predicate to build on. +// +// What this file deliberately does NOT do: +// +// - It does not probe. Runtime probing (design §2.3 step 2) is +// deliberately absent for every family in this migration; this +// endpoint is declaration-backed, and a probe result that silently +// overrode the declaration would make the frontend's rendering +// depend on timing. +// - It does not construct a host. +// - It does not convert the engine's `"unknown"` version into an +// error. `/api/health` (#75) already answers that same figure with +// the same fallback through `session-reads.js#readEngineVersion`, +// and two endpoints asking the same protocol question with the same +// answer is correct; two endpoints answering it DIFFERENTLY is +// not, which is why both read `getMcodeServerInfo()` and both keep +// the literal `"unknown"` fallback. +// +// Boot-path weight. `app.js` imports `routes/protocol.js`, the route +// imports this file, so this file is on the boot path. It statically +// imports nothing heavier than `capabilities.js` and `index.js`; +// `lib/acp-client.js` and `lib/config.js` are reached through +// `await import()` inside the read — the M1 lesson, and the reason the +// route's own `await import(...)` lines moved behind this boundary +// rather than being duplicated. (`lib/mcode-rpc.js` was in that list +// while the endpoint still served the ACP wire table; replacing the +// field removed the dependency, not just the field.) + +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * The declaration this endpoint needs: `null`, for the reason in the + * header. The type keeps the `|null` branch so a future gated sibling + * can be added to the same table without changing its shape. + * + * @type {Readonly>} + */ +export const CAPABILITY_READ_ENDPOINTS = Object.freeze({ + "GET /api/protocol/capabilities": null, +}); + +/** + * Resolve the provider whose declaration answers the capability read on + * `transport`. + * + * Unlike every other family this one ALWAYS answers, because an + * empty capability view would be worse than useless for a capability + * DETECTION endpoint: the frontend would learn nothing and could not + * distinguish "no engine" from "this build has no declarations". So + * when no provider claims the transport, the DEFAULT provider's + * declaration is served and the descriptor says so. + * + * @param {string} transport + * @returns {{provider: {id: string, transport: string, capabilities: object}, providerFor: "transport"|"default"}} + */ +export function resolveCapabilityReadProvider(transport) { + const providerId = transport === "runtime" ? DEFAULT_ENGINE_PROVIDER_ID : null; + if (providerId) { + return { provider: getEngineProvider(providerId), providerFor: "transport" }; + } + return { provider: getEngineProvider(), providerFor: "default" }; +} + +/** + * Read the declaration for #73 WITHOUT enforcing it. + * + * Same `gate` vocabulary as `session-export.js#checkSessionExportCapability` + * and `model-reads.js#checkModelReadCapability`, with one difference that + * is forced by the `null` row: the result is always `no-capability-key` + * and never `unregistered-transport`, because the provider this + * endpoint serves is always resolvable (see + * `resolveCapabilityReadProvider`). + * + * @param {string} endpoint A key of CAPABILITY_READ_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null, enforcement: "soft"}} + */ +export function checkCapabilityReadCapability(endpoint, transport) { + const need = CAPABILITY_READ_ENDPOINTS[endpoint]; + if (need === undefined) { + const err = new Error( + `checkCapabilityReadCapability: "${endpoint}" is not part of the capability family ` + + `(known: ${Object.keys(CAPABILITY_READ_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_capability_read_endpoint"; + throw err; + } + const { provider } = resolveCapabilityReadProvider(transport); + return { + endpoint, + gate: "no-capability-key", + provider: provider.id, + capability: null, + subItem: null, + enforcement: "soft", + }; +} + +/** + * How the declaration that answered was chosen. + * + * `providerFor` is the honest bit: `"transport"` means the active + * transport's own provider answered; `"default"` means no provider + * claims that transport yet (M4) and the default provider's + * declaration is standing in. A capability-detection endpoint that + * reported `"default"` as though it were `"transport"` would be + * answering a question about a different engine than the one + * connected — the same lie B1 declined for `/api/health` and B3 + * declined for #19, in the one place where it is most tempting because + * the fallback is silent. + * + * @typedef {{ + * provider: string, + * providerFor: "transport"|"default", + * transport: string, + * }} EngineCapabilityProvenance + */ + +/** + * The #73 (`GET /api/protocol/capabilities`) read. + * + * `declaration` is the provider's 14-key capability object FORWARDED BY + * IDENTITY — not a copy, not a re-projection. A copy would be a second + * thing that can drift from the reviewed declaration, which is the whole + * failure this endpoint had before M3-B4. + * + * `unavailable` is the DERIVED roll-up (`summarizeUnavailableCapabilities`) + * and is the one field here that is not the declaration itself: a + * `none` key means hide the entry point, a `partial` key means hide or + * disable exactly the listed sub-actions (design §4.2). It is kept + * because it is the shape the capability-driven UI renders from, and a + * consumer should not have to re-derive it from a taxonomy that has + * three levels and two optional fields. + * + * `agent` is the ACP `initialize` mirror: `{version, name, title}` with + * the endpoint's own `"unknown"` / `null` fallbacks, applied here so + * the route does not repeat them. + * + * @param {object} [options] + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/protocol/capabilities`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @returns {Promise<{declaration: object, unavailable: {none: string[], partial: Array<{key: string, missing: string[]}>}, provider: string, providerFor: "transport"|"default", engineTransport: string, agent: {version: string, name: string|null, title: string|null}, source: "declaration", gate: object, transport: string}>} + */ +export async function readEngineCapabilityView(options = {}) { + const endpoint = options.endpoint || "GET /api/protocol/capabilities"; + const [acp, config, capabilities] = await Promise.all([ + import("../lib/acp-client.js"), + import("../lib/config.js"), + import("./capabilities.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = checkCapabilityReadCapability(endpoint, transport); + const { provider, providerFor } = resolveCapabilityReadProvider(transport); + // `initialize` answers with `agentInfo: {name, title, version}` (not + // `serverInfo`); the mirror is empty until something attaches. + const agentInfo = acp.getMcodeServerInfo(); + return { + declaration: provider.capabilities, + unavailable: capabilities.summarizeUnavailableCapabilities(provider.capabilities), + provider: provider.id, + providerFor, + // The PROVIDER's wire form, named apart from the ambient + // `transport` the read ran under: under the default `acp` transport + // the declaration served belongs to a `runtime` provider, and + // collapsing the two into one field would say exactly the thing + // `providerFor` exists to prevent. + engineTransport: provider.transport, + agent: { + version: (agentInfo && agentInfo.version) || "unknown", + name: (agentInfo && agentInfo.name) || null, + title: (agentInfo && agentInfo.title) || null, + }, + source: "declaration", + gate, + transport, + }; +} diff --git a/packages/webui/server/engine/index.js b/packages/webui/server/engine/index.js index ed0f6680..589a6ac9 100644 --- a/packages/webui/server/engine/index.js +++ b/packages/webui/server/engine/index.js @@ -35,9 +35,10 @@ // no route's behaviour changed. M3's first batch (B0) done — the // catalogue host itself is now reached through this facade too // (engine/host.js), so the plugins and turn-diff routes no longer name -// lib/acp-client.js. M3 batches B1 (#9 #10 #72 #74 #75) and B3 (#15 #16 -// #17 #19) done. B2 (#8 #11) and the rest of M3, then M4, will route -// their consumers through this facade one endpoint family at a time. +// lib/acp-client.js. M3 batches B1 (#9 #10 #72 #74 #75), B2 (#8 #11), +// B3 (#15 #16 #17 #19), B4 (#20 #57 #73), B5 (#7 #4 #6) and B6 (#3) +// done. The rest of M3, then M4, will route their consumers through +// this facade one endpoint family at a time. import { ENGINE_CAPABILITY_KEYS } from "./capabilities.js"; // Declarations only — importing the provider *host-construction* modules @@ -104,6 +105,68 @@ export { readEngineSessionTranscript, resolveSessionExportProvider, } from "./session-export.js"; +// The account read (step M3, batch B4). Same cycle, same rule, same +// reasoning as session-reads.js: account-reads.js reads NOTHING from +// this module at module scope — its `ACCOUNT_READ_ENDPOINTS` table is a +// literal and every binding it needs (`getEngineProvider`, +// `DEFAULT_ENGINE_PROVIDER_ID`) is read inside a function body. A new +// top-level `const X = SOMETHING_FROM_INDEX` in account-reads.js breaks +// the re-export exactly as it would in session-reads.js. It gates HARD +// on `authCredentials.getAccountStatus` — the same pair and the same +// provider method B3's `POST /api/usage` / `POST /api/usage-trigger` +// use, because both read the engine's account projection; the modules +// stay separate because the usage family owns derivations this one +// does not have. +export { + ACCOUNT_READ_ENDPOINTS, + assertAccountReadCapability, + readEngineAccount, + resolveAccountReadProvider, +} from "./account-reads.js"; +// The model-catalogue read (step M3, batch B4) is deliberately NOT +// re-exported here, and that is the one place this file's shape +// disagrees with its siblings. It gates SOFT (the catalogue's primary +// sources are files webui owns, so a provider that declared no model +// surface would not remove the picker — the `session-export.js` +// reasoning, reused rather than re-argued), and its read is +// SYNCHRONOUS, which is what keeps `handleGetModels` synchronous. Both +// properties come from one decision: the three catalogue sources are +// static imports of this module, because `routes/model.js` already +// imported `lib/models.js`, `lib/providers-config.js`, +// `lib/engine-catalogue.js` and `lib/config.js` before M3-B4. +// +// Those four reach `js-yaml` and `@mavis/shared/local-runtime-paths`, +// and `test/lib/engine/host-facade.test.js` is right to refuse that +// under a facade `app.js` loads: it would make `engine/index.js` — the +// one import site the whole server shares, and the one +// `routes/plugins.js` must stay light through — heavier than it has ever +// been, for no saving. So `routes/model.js` imports +// `../engine/model-reads.js` directly, the same shape +// `routes/protocol.js` already uses for `session-reads.js`. The server's +// own boot cost is unchanged: every module involved was already on it +// through the route. When the catalogue read becomes async (M4, with a +// provider-backed source), the module can move back behind +// `await import()` and be re-exported here with the rest. +// The capability-declaration read (step M3, batch B4). Declares NO +// capability for #73 — it IS the declaration endpoint, and gating the +// gate would let a `none` hide the declaration that says so. It is the +// one endpoint in the migration whose response CONTRACT changed: +// `capabilities` used to carry `MCODE_ACP_CAPABILITIES`, the flat ACP +// wire table, and now carries the provider's DECLARED 14-key object — a +// user-authorised replacement, not an addition. The `engine` key an +// earlier shape of this batch shipped was removed rather than kept, +// because with the declaration already under `capabilities` it would +// have carried the same 14 keys a second time in one response; what +// survives is the provenance (`capabilitiesProvider` / +// `capabilitiesProviderFor`) and the derived `capabilitiesUnavailable`. +// See the module header for the full statement and the debt note on the +// now-unconsumed constant. +export { + CAPABILITY_READ_ENDPOINTS, + checkCapabilityReadCapability, + readEngineCapabilityView, + resolveCapabilityReadProvider, +} from "./capability-reads.js"; // The usage family's gated reads (step M3, batch B3). Same cycle, same // rule, same reasoning as session-reads.js above: usage-reads.js reads // NOTHING from this module at module scope — its `USAGE_READ_ENDPOINTS` @@ -120,6 +183,83 @@ export { readEngineSessionUsage, resolveUsageReadProvider, } from "./usage-reads.js"; +// The session WRITE family (step M3, batch B5): #7 delete, #4 rename, +// #6 cleanup-orphans. Same cycle, same TDZ rule, same reasoning as +// session-reads.js above: session-writes.js reads NOTHING from this +// module at module scope — its `SESSION_WRITE_ENDPOINTS` table is a +// literal and every binding it needs (`getEngineProvider`, +// `DEFAULT_ENGINE_PROVIDER_ID`) is read inside a function body. A new +// top-level `const X = SOMETHING_FROM_INDEX` in session-writes.js breaks +// the re-export exactly as it would in session-reads.js. Its static +// imports are `engine/capabilities.js`, `engine/index.js` and `node:fs` +// (a builtin); all six of its storage dependencies are reached through +// `await import()` inside the functions, so the boot-path rule the other +// families follow holds here too. +// +// Two of its three endpoints gate HARD on `sessionCrud` · `deleteSession` +// — #7 and #6, both because they destroy rows in the engine's own +// `local_runtime_*` tables — and the third, #4, declares NO capability +// because a rename writes webui's own session store and touches no engine +// surface at all. The policy is decided by who owns the rows the write +// destroys, which is a different question from the read families' and +// does not have the same answer twice in a row here. See the module +// header for the full argument and for the known debt this batch records +// rather than settles. +export { + ORPHAN_STALE_MS, + SESSION_WRITE_ENDPOINTS, + applyDeletedSessionToClientState, + applyEngineSessionRename, + applyRenamedSessionToClientState, + assertSessionWriteCapability, + clientMatchesDeletedSession, + clientMatchesRenamedSession, + commitEngineOrphanSessionDelete, + commitEngineSessionDelete, + isMcodeSessionId, + isOrphanSessionRecord, + planEngineSessionDelete, + previewEngineSessionDelete, + readOrphanSessionWriteIds, + resolveSessionTarget, + resolveSessionWriteProvider, + selectOrphanSessionIds, +} from "./session-writes.js"; +// The session SWITCH family (step M3, batch B6): #3 +// POST /api/sessions/switch. Same cycle, same TDZ rule, same reasoning as +// session-writes.js above: session-switch.js reads NOTHING from this module +// at module scope — its `SESSION_SWITCH_ENDPOINTS` table is a literal and +// every binding it needs (`getEngineProvider`, `DEFAULT_ENGINE_PROVIDER_ID`) +// is read inside a function body. A new top-level `const X = +// SOMETHING_FROM_INDEX` in session-switch.js breaks the re-export exactly as +// it would in session-writes.js. Its ONLY static import beyond this module +// is `engine/capabilities.js`; the session store, the ACP client, the +// transcript reader, the usage tables, the workspace gate, the state bus +// and the config are all reached through `await import()` inside the +// data-plane function. +// +// It gates SOFT (`checkSessionSwitchCapability` reports, never throws) for +// the reason `session-export.js` does: the switch's primary data is webui's +// own session record, and both of its engine touches (the title and the +// transcript) have a defined degradation. Gating hard would remove a +// working endpoint in response to a declaration about an enrichment it can +// live without — and would do it on the default `acp` transport first, +// where the enrichment is the only part in question. The 501 machinery +// stays unused by this family, and the suite pins that. +export { + SESSION_SWITCH_ENDPOINTS, + applyEngineSessionSwitch, + applySwitchedSessionToClientState, + chatLooksCumulative, + checkSessionSwitchCapability, + isSwitchableMcodeSessionId, + lookupCachedMcodeTitle, + readEngineSwitchTranscript, + resolveSessionSwitchProvider, + resolveSwitchTarget, + resolveSwitchWorkspace, + selectTranscriptBackfill, +} from "./session-switch.js"; export { LOCAL_RUNTIME_V2_CAPABILITIES } from "./providers/local-runtime-v2.capabilities.js"; export { TUI_RUNTIME_ADAPTER_CAPABILITIES } from "./providers/tui-runtime-adapter.js"; diff --git a/packages/webui/server/engine/model-reads.js b/packages/webui/server/engine/model-reads.js new file mode 100644 index 00000000..c7eb163e --- /dev/null +++ b/packages/webui/server/engine/model-reads.js @@ -0,0 +1,732 @@ +// webui/server/engine/model-reads.js +// +// Migration step M3, batch B4: the model-catalogue read (模型目录读) — +// +// #57 GET /api/models — the composer's provider-grouped model picker +// +// What this file is for. #57 is the LARGEST projection in webui and +// the one a refactor can damage most quietly. It merges three +// independent sources, dedupes them by a key that has changed shape +// twice, annotates each surviving entry with two separate projections +// of the engine's materialised builtin tree, and derives three +// "what is active right now" figures — none of which is compared +// against anything at runtime. A change that drops one annotation, or +// moves one entry into the wrong provider group, or resolves `current` +// to a different id, changes what the user picks and reports nothing. +// So the whole projection lives HERE, once, as named pure functions +// tested on their INPUTS, and the route assembles nothing but JSON. +// +// The three sources, in the order they are consumed (priority order is +// the endpoint's, not this file's invention — see `projectModelCatalogue`): +// +// 1. the engine SESSION's `model` config option — its `options[].value` +// is the engine's encoded wire form, forwarded verbatim so +// `POST /api/set-model` round-trips; +// 2. the PROVIDERS config — webui's `env > cwd > user` layers with the +// engine's `custom_provider` tree as a new bottom layer +// (`lib/engine-catalogue.js` owns the merge); +// 3. the BUILTIN catalogue — extracted from the engine's own cli +// bundle, so the list tracks the engine without a webui release. +// +// Plus two projections of the SAME engine tree (`provider.minimax.models`) +// that annotate entries in both source 1 and source 3: the variant-style +// thinking schema (`readEngineBuiltinThinking`) and the context-window +// options (`readEngineBuiltinContextWindows`, "U6"). Both are read once +// per request and handed to the two annotation sites — the redundancy of +// reading them per entry was a real cost, and the two reads must agree +// because they are two views of one file. +// +// Why this family's gate is SOFT while the account family gates hard. +// The catalogue is NOT engine data in the way an account is. Its +// primary sources are files webui owns and can read without the engine: +// `models.json` / `~/.mcode-webui/providers.json` on the webui side, and +// a cli-bundle extraction on the engine side. A provider that declared +// no model surface would still leave a fully working picker over the +// webui layers plus the builtins. Gating the endpoint hard would REMOVE +// working functionality in response to a declaration about a capability +// the endpoint does not actually depend on — the exact reasoning +// `session-export.js` records for #11, reused here rather than +// re-argued. So `checkModelReadCapability` REPORTS and never throws; the +// read is unaffected by what it reports, and the report is what a later +// batch needs in order to decide whether the engine LAYER may be trusted. +// +// What this file deliberately does NOT do: +// +// - It does not re-implement the engine's own projections. +// `lib/engine-catalogue.js` owns the thinking schema, the +// context-window hygiene rules, the wire-form parser, the +// `custom_provider` read and the layer merge. A second projection +// here would be a second answer to a question with exactly one. +// - It does not write. `handleSetModel` stays in the route for B7/B9; +// this batch only moves the READ. +// - It does not build a second provider host and does not call any +// provider method: every source here is a file read, which is why +// the sub-item below names a READ rather than a method webui calls. +// +// Boot-path weight — the ONE place this batch deviates from the +// sibling families, and the deviation is deliberate on both sides of +// the import. +// +// The four static imports below (`lib/config.js`, +// `lib/engine-catalogue.js`, `lib/models.js`, `lib/providers-config.js`) +// were ALREADY static imports of `routes/model.js` before M3-B4, so the +// server's boot cost is exactly what it was. What they must not do is +// reach `@mavis/*` or `js-yaml` through the SHARED facade — and they +// do reach `@mavis/shared/local-runtime-paths` (via `lib/config.js`) +// and `js-yaml` (via `engine-provider-sync.js`). That is why this module +// is deliberately NOT re-exported from `engine/index.js`, and why +// `routes/model.js` imports it directly: `test/lib/engine/host-facade.test.js` +// guards `engine/index.js` and `routes/plugins.js` against exactly that +// pull, and the guard is right. See `engine/index.js` for the full +// statement and for what has to be true before this can move back. +// +// The price is that the read is SYNCHRONOUS. Making the four imports +// dynamic would let the module re-export from the facade, at the cost of +// turning `handleGetModels` into an async handler — a contract change +// for any caller that does not await, and the exact thing this batch +// promises not to do. The M1 lesson (209ms → 2700ms) was about the +// `@mavis/*` TypeScript host tree, which nothing here touches. +// +// Provider selection is M4's job, same as every other family: +// `providerByTransport()` maps a transport to a REGISTERED provider id; +// today only `runtime` has one, so the default `acp` transport reports +// `gate: "unregistered-transport"` and the read proceeds unchanged. + +import { MCODE_WEBUI_TRANSPORT } from "../lib/config.js"; +import { + mergeEngineAndWebuiProviders, + parseEngineModelWireValue, + readEngineBuiltinContextWindows, + readEngineBuiltinThinking, + readEngineCatalogue, +} from "../lib/engine-catalogue.js"; +import { getBuiltinModelsFromMcode } from "../lib/models.js"; +import { loadProvidersConfig } from "../lib/providers-config.js"; +// `assertEngineCapability` is deliberately NOT imported: this family's +// gate is soft, so it INSPECTS the declaration (`checkModelReadCapability` +// below) and reports what it found rather than delegating the verdict to +// the throwing helper. Importing it here would be a dead import that +// reads as if the soft path could still throw. +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * The provider every builtin entry belongs to. + * + * The builtin shell is keyed by `minimax_api` REGARDLESS of the + * recorded pick. Deriving the group from `currentName.split("/")[0]` + * was the bug ticket 09-02's acceptance replay caught as "8 config + 6 + * misplaced MiniMax builtins = 14 in `nousresearch`": a user who picked + * a BYOK model dragged the engine's own builtins into that provider's + * group. The constant lives here now because the attribution rule and + * the group id are one decision. + */ +const BUILTIN_PROVIDER = "minimax_api"; + +/** The synthetic group id the engine session's own option list renders as. */ +const ENGINE_GROUP_ID = "__engine"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, mirroring + * `session-reads.js#providerByTransport`, `session-tree-reads.js`, + * `session-export.js` and `usage-reads.js`. Kept per-family so each + * family owns its own gate policy; collapse them in M4, not here. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +/** + * The declaration this endpoint's ENGINE LAYER needs, and the sub-item + * it needs from that capability. + * + * `authCredentials` is where the engine's model/provider surface is + * declared — the local-runtime-v2 declaration says so in its own + * comment ("full user model provider CRUD/test/discover, same source as + * service/model-system"), and the 14 matrix keys have no separate + * "models" row. `listModelProviders` names the READ, not a method webui + * calls: the custom_provider tree and the builtin tree are files the + * engine owns, read through `lib/engine-catalogue.js`, not a + * `CliService` method. That distinction is the reason this family's + * gate is soft — a missing declaration here removes ONE of the + * catalogue's three sources, never the endpoint. + * + * @type {Readonly>} + */ +export const MODEL_READ_ENDPOINTS = Object.freeze({ + "GET /api/models": { + capability: "authCredentials", + subItem: "listModelProviders", + enforcement: "soft", + }, +}); + +/** + * Resolve the provider that answers the model read on `transport`, or + * `null` when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveModelReadProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Read the declaration for #57 WITHOUT enforcing it. + * + * The `gate` values are the same vocabulary `session-export.js` uses, + * for the same reason: + * + * - `"checked"` — provider resolved, capability `full`. + * - `"unregistered-transport"` — no provider claims this transport yet. + * - `"capability-absent"` — the provider WAS found and does not + * offer the model surface. The caller's next move is to distrust the + * ENGINE LAYER, not to fail the request. + * - `"partial"` — the provider is `partial` and this + * sub-item is absent. + * + * Deliberately never throws `EngineCapabilityNotSupportedError`. See the + * header for why a hard gate here would remove working functionality. + * A genuinely unknown endpoint key is still a plain Error — caller + * confusion is not a capability question. + * + * @param {string} endpoint A key of MODEL_READ_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null, enforcement: "soft"}} + */ +export function checkModelReadCapability(endpoint, transport) { + const need = MODEL_READ_ENDPOINTS[endpoint]; + if (need === undefined) { + const err = new Error( + `checkModelReadCapability: "${endpoint}" is not part of the model family ` + + `(known: ${Object.keys(MODEL_READ_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_model_read_endpoint"; + throw err; + } + const base = { + endpoint, + provider: null, + capability: need.capability, + subItem: need.subItem, + enforcement: need.enforcement, + }; + const provider = resolveModelReadProvider(transport); + if (!provider) return { ...base, gate: "unregistered-transport" }; + const entry = provider.capabilities ? provider.capabilities[need.capability] : undefined; + const descriptor = { ...base, provider: provider.id }; + if (entry && entry.level === "full") { + return { ...descriptor, gate: "checked" }; + } + if (entry && entry.level === "partial") { + const absent = Array.isArray(entry.missing) && entry.missing.includes(need.subItem); + return { ...descriptor, gate: absent ? "partial" : "checked" }; + } + return { ...descriptor, gate: "capability-absent" }; +} + +// --------------------------------------------------------------------------- +// The projections. Pure functions, exported, and tested on their INPUTS. +// --------------------------------------------------------------------------- + +/** + * Coerce a provider prefix out of a model id. + * + * `minimax_api/MiniMax-M3` → `minimax_api`. Bare `MiniMax-M3` falls back + * to `minimax_api` (the engine's only shipping builtin provider) so a + * user-typed short id still resolves to a known group instead of + * orphaning itself. + * + * Used only for engine session entries (their ids are the engine's wire + * form `m:::u`); webui-side entries carry the + * provider as an explicit `entry.provider` field, and the multi-segment + * model id stays whole (see `webuiFullModelId`). + * + * @param {string} modelId + * @param {string} [fallback] + * @returns {string} + */ +export function providerOfModelId(modelId, fallback = BUILTIN_PROVIDER) { + if (!modelId) return fallback; + const i = modelId.indexOf("/"); + if (i <= 0) return fallback; + return modelId.slice(0, i); +} + +/** + * Build the webui internal id for a catalogue entry: `/`. + * + * The webui id is always two segments where the first is the provider + * key and the second is the engine-side model id verbatim (the engine + * allows `/` inside model ids — see `engine-catalogue.js`; the wire + * form `formatModelKey(, )` uses `/` as the only + * structural separator, so a downstream `/` webui + * form survives the round-trip through `resolveModelId`). + * + * Ticket 09-02 (grouping attribution): the previous implementation + * skipped the prefix when `m.id.includes("/")` and let the bare + * upstream id stand. That pushed the picker into the wrong group (the + * id's first segment was used as a fallback for the provider + * extraction) and let two providers with overlapping upstream ids + * collide on the `seen` dedupe (e.g. `z-ai/glm-5.3` in `nousresearch` + * ate the sibling `zai-max/glm-5.3`). Always prefixing — even when the + * model id already contains `/` — keys every entry by + * `(providerKey, modelId)` and the dedupe is per provider, as the + * ticket requires. + * + * @param {string} providerKey + * @param {string} modelId + * @returns {string} + */ +export function webuiFullModelId(providerKey, modelId) { + return `${providerKey}/${modelId}`; +} + +/** + * Attach the engine's context-window metadata ("U6") onto a catalogue + * entry, mutating `entry`. + * + * `contextWindowOptions` / `contextWindowOptionHints` come from the + * engine's materialised builtin tree (same read as the thinking + * projection — see `lib/engine-catalogue.js`). Only the `minimax_api` + * builtin entries carry them today: the engine's ACP `model` config + * option (the engine-session entries' source) does not advertise the + * metadata, so those entries are annotated through the same builtin + * projection keyed by the wire form's model id. Custom-provider / + * config-layer entries never get the fields — a model without options + * must stay field-free so the composer mounts no control. + * + * `contextLimit` (the CURRENT effective window, from the engine tree's + * `limit.context`) is attached when the entry has none yet — a config + * layer entry keeps its own value; builtin shell entries get the + * engine's current window so the picker can show the active radio + * before the user's first in-webui pick. + * + * @param {object} entry Mutated in place; the caller owns it. + * @param {{options: number[], hints?: object, currentLimit?: number}|null} projection + * @returns {void} + */ +export function attachContextWindowOptions(entry, projection) { + if (!projection) return; + entry.contextWindowOptions = [...projection.options]; + if (projection.hints) { + entry.contextWindowOptionHints = { ...projection.hints }; + } + if (entry.contextLimit === undefined && projection.currentLimit !== undefined) { + entry.contextLimit = projection.currentLimit; + } +} + +/** + * The engine session's `model` config option with this id, or `null` + * before a session exists. + * + * @param {object} cs + * @param {string} id + * @returns {object|null} + */ +export function configOption(cs, id) { + const options = Array.isArray(cs && cs.configOptions) ? cs.configOptions : []; + return options.find((o) => o && o.id === id) || null; +} + +/** + * Build the flat `models` list and the provider-grouped `groups` array. + * + * Pure, and the whole of #57's payload except the three derived "what + * is active" figures. The rules it encodes, in the order the endpoint + * has always applied them: + * + * 1. Engine session entries first, under the synthetic `__engine` + * group, with the engine's wire ids kept verbatim. They carry BOTH + * `name` and `label` because pre-existing callers (the composer + * chip) read `name` while the provider-grouped panel reads `label`. + * 2. Config-layer providers next, each under its own group, with the + * operator's per-model metadata winning wholesale on an id + * collision (the `seen` dedupe). A group is emitted even when its + * model list is empty — an operator who configured a provider with + * no models yet must still see the group to add one. + * 3. Builtins last, appended to the `minimax_api` group (created on + * demand) — and the empty shell is dropped only when there is no + * providers config at all, so a fresh install with a config that + * names no models still has somewhere to attach the builtins once + * the engine reports them. + * + * @param {object} options + * @param {object|null} options.sessionOption The engine `model` config option. + * @param {{providers: Array}|null} options.providers The merged + * providers config, or `null` when every layer was missing. + * @param {string[]} options.builtins Bare builtin model ids. + * @param {Map} options.builtinThinking + * @param {Map} options.builtinContextWindows + * @returns {{list: Array, groups: Array}} + */ +export function projectModelCatalogue(options = {}) { + const sessionOption = options.sessionOption || null; + const providers = options.providers || null; + const builtins = Array.isArray(options.builtins) ? options.builtins : []; + const builtinThinking = options.builtinThinking || new Map(); + const builtinContextWindows = options.builtinContextWindows || new Map(); + const parseWire = options.parseEngineModelWireValue || defaultParseWireStub; + + const list = []; + const groups = []; + const seen = new Set(); + + // 1) Engine session config option — authoritative when present. + if (sessionOption) { + const engineGroup = { id: ENGINE_GROUP_ID, label: "Engine session", models: [] }; + for (const o of Array.isArray(sessionOption.options) ? sessionOption.options : []) { + const id = o && typeof o.value === "string" ? o.value : null; + if (!id) continue; + if (seen.has(id)) continue; + seen.add(id); + const displayName = (o && o.name) || id; + const entry = { + id, + name: displayName, + label: displayName, + provider: providerOfModelId(id), + source: "engine", + }; + // The engine's wire-form `currentValue` is mirrored into + // `cs.model.name` outside the pick window and the composer matches + // the active model by id — annotate the wire-form entries too so + // the thinking and context controls survive a cross-client change. + const wire = parseWire(id); + if (wire && wire.providerId === BUILTIN_PROVIDER) { + const proj = builtinThinking.get(wire.modelId); + if (proj) entry.thinkingLevels = [...proj.levels]; + attachContextWindowOptions(entry, builtinContextWindows.get(wire.modelId)); + } + engineGroup.models.push(entry); + list.push(entry); + } + if (engineGroup.models.length > 0) groups.push(engineGroup); + } + + // 2) Providers config — read every request so editing the file does not + // require a restart. Config wins on id collision with the builtin + // catalogue so providers can override labels and contextLimit. + if (providers) { + for (const p of providers.providers) { + if (!p || typeof p.id !== "string" || !p.id) continue; + const models = []; + for (const m of Array.isArray(p.models) ? p.models : []) { + if (!m || typeof m.id !== "string" || !m.id) continue; + const fullId = webuiFullModelId(p.id, m.id); + if (seen.has(fullId)) continue; + seen.add(fullId); + const entry = { + id: fullId, + label: typeof m.label === "string" && m.label ? m.label : m.id, + provider: p.id, + source: "config", + }; + if (typeof m.contextLimit === "number" && m.contextLimit > 0) { + entry.contextLimit = m.contextLimit; + } + // v2 schema surfaces: each model carries protocol + + // thinkingLevels + modalities so the selector can pick the right + // controls without a second round-trip. `auth` only exposes + // hasKey + type — an apiKey NEVER reaches this response. + if (typeof p.protocol === "string" && p.protocol) { + entry.protocol = p.protocol; + } + if (Array.isArray(m.thinkingLevels) && m.thinkingLevels.length > 0) { + entry.thinkingLevels = [...m.thinkingLevels]; + } + if (Array.isArray(m.modalities) && m.modalities.length > 0) { + entry.modalities = [...m.modalities]; + } + models.push(entry); + list.push(entry); + } + // Auth shape: only `hasKey` and `type`; no apiKey/baseURL. + // Operators see "configured or not" without leaking the secret. + // The merged layer (engine + webui) may carry `hasKey` either via + // `p.auth.apiKey` (webui-side plaintext — masked elsewhere) or via + // `p.auth.hasKey` (engine-side boolean, set by + // `lib/engine-catalogue.js`). Either signal means the provider is + // configurable from the picker. + const groupHasKey = !!((p.auth && p.auth.apiKey) || (p.auth && p.auth.hasKey)); + groups.push({ + id: p.id, + label: typeof p.label === "string" && p.label ? p.label : p.id, + auth: { + hasKey: groupHasKey, + type: p.auth && typeof p.auth.type === "string" ? p.auth.type : "byok", + }, + protocol: typeof p.protocol === "string" ? p.protocol : "openai", + models, + }); + } + } + + // 3) Builtin catalogue. The builtins all belong to the engine's + // `minimax_api` provider (see `lib/models.js#getBuiltinModelsFromMcode` + // — the cli.js extraction regex targets `MiniMax-M*`). + let builtinGroup = groups.find((g) => g.id === BUILTIN_PROVIDER); + if (!builtinGroup) { + builtinGroup = { id: BUILTIN_PROVIDER, label: BUILTIN_PROVIDER, models: [] }; + groups.push(builtinGroup); + } + for (const m of builtins) { + const fullId = webuiFullModelId(BUILTIN_PROVIDER, m); + if (seen.has(fullId)) continue; + seen.add(fullId); + const entry = { + id: fullId, + label: m, + provider: BUILTIN_PROVIDER, + source: "builtin", + }; + // `thinkingLevels` is exactly what the engine's tree supports — + // ["off","on"] for a switchable variant toggle, the engine's effort + // list when the model has one, and ABSENT for a forced_on model with + // nothing user-settable (the composer then mounts no control, by + // design). A config-layer entry with the same id has already taken + // the slot (seen dedupe) — the operator's config wins wholesale, + // unchanged rule. + const proj = builtinThinking.get(m); + if (proj) entry.thinkingLevels = [...proj.levels]; + attachContextWindowOptions(entry, builtinContextWindows.get(m)); + list.push(entry); + builtinGroup.models.push(entry); + } + + // Drop the empty builtin shell — a no-bundle empty group is noise. + // The drop is gated on "no providers config" so a fresh install with + // a config that names no models still has somewhere to attach the + // builtins once the engine reports them. + if (builtinGroup.models.length === 0 && !providers) { + const idx = groups.indexOf(builtinGroup); + if (idx >= 0) groups.splice(idx, 1); + } + + return { list, groups }; +} + +/** + * The three "what is active right now" figures #57 reports. + * + * `current` is the engine's value when one exists, otherwise the + * recorded pre-session choice (`cs.model.name`, written by + * `handleSetModel`). When neither exists the answer is `null` rather + * than a fallback to a default model — the old behaviour invented an + * active model the engine never confirmed, and the chip ended up + * claiming a model the session was not actually running. The chip + * renders a neutral label when `current` is `null` (see + * `composer.tsx#currentModelLabel`). + * + * `currentThinking` prefers the engine's `thinkingEffort` option and + * falls back to `cs.model.thinking` (the pre-session record that + * `applyConfigOptionUpdate` refreshes). The selector reads it to + * highlight the active level and to skip the picker when the active + * model has no `thinkingLevels`. + * + * `currentContextWindow` is the recorded choice (`handleSetModel` + * writes `cs.model.contextWindow`) with the current model's catalogue + * `contextLimit` as fallback. There is no engine-value branch for the + * window, deliberately: the engine's ACP surface has no context + * channel, so the recorded pick is the only source. A recorded value + * the current model no longer advertises is still reported verbatim — + * the stale-pick display rule lives in the composer. + * + * @param {object} options + * @param {object|null} options.sessionOption + * @param {object} options.cs + * @param {Array} options.list The projected flat list. + * @returns {{current: string|null, currentThinking: string|null, currentContextWindow: number|null}} + */ +export function deriveModelSelection(options = {}) { + const sessionOption = options.sessionOption || null; + const cs = options.cs || {}; + const list = Array.isArray(options.list) ? options.list : []; + const current = + (sessionOption && sessionOption.currentValue) || + (cs.model && typeof cs.model.name === "string" && cs.model.name) || + null; + const thinkingEffortOption = Array.isArray(cs.configOptions) + ? cs.configOptions.find((o) => o && o.id === "thinkingEffort") + : null; + const currentThinking = + (thinkingEffortOption && typeof thinkingEffortOption.currentValue === "string" + ? thinkingEffortOption.currentValue + : null) || + (cs.model && typeof cs.model.thinking === "string" && cs.model.thinking) || + null; + const recordedContextWindow = + cs.model && Number.isSafeInteger(cs.model.contextWindow) && cs.model.contextWindow > 0 + ? cs.model.contextWindow + : null; + const currentModelEntry = current ? list.find((m) => m.id === current) : null; + const currentContextWindow = + recordedContextWindow ?? + (currentModelEntry && + Number.isSafeInteger(currentModelEntry.contextLimit) && + currentModelEntry.contextLimit > 0 + ? currentModelEntry.contextLimit + : null); + return { current, currentThinking, currentContextWindow }; +} + +/** + * The endpoint's `source` label — which of the three layers won. + * + * @param {object} options + * @param {object|null} options.sessionOption + * @param {{providers: Array}|null} options.providers + * @returns {"acp-session-config"|"config+mcode-cli-bundle"|"mcode-cli-bundle"} + */ +export function catalogueSourceLabel(options = {}) { + const sessionOption = options.sessionOption || null; + if (sessionOption && Array.isArray(sessionOption.options) && sessionOption.options.length > 0) { + return "acp-session-config"; + } + return options.providers ? "config+mcode-cli-bundle" : "mcode-cli-bundle"; +} + +/** + * Compose the endpoint's response body. The key ORDER is the endpoint's + * and is asserted by the test suite: `ok`, `models`, `groups`, the three + * derived figures, `source`, and the soft-failure `reason` marker that + * is spread LAST and only when the catalogue came out empty. + * + * The marker is backwards compatibility with the older engine-only + * build. With the merge it should be rare (builtin catalogue + + * providers config cover most installs), but a missing cli bundle AND + * an absent config leave the catalogue empty — and a caller that wants + * to know "is this a hard failure or just no engine attached?" still + * gets the same hint. + * + * @param {object} options + * @param {object|null} options.sessionOption + * @param {{providers: Array}|null} options.providers + * @param {string[]} options.builtins + * @param {Map} options.builtinThinking + * @param {Map} options.builtinContextWindows + * @param {object} options.cs + * @param {Function} [options.parseEngineModelWireValue] + * @returns {object} The exact #57 response body. + */ +export function buildModelCataloguePayload(options = {}) { + const { list, groups } = projectModelCatalogue(options); + const selection = deriveModelSelection({ + sessionOption: options.sessionOption, + cs: options.cs, + list, + }); + const source = catalogueSourceLabel(options); + return { + ok: true, + models: list, + groups, + current: selection.current, + currentThinking: selection.currentThinking, + currentContextWindow: selection.currentContextWindow, + source, + ...(list.length === 0 ? { reason: "no_catalogue" } : {}), + }; +} + +/** + * The wire-form parser used when the caller does not inject one. Only + * ever reached from a unit test that calls `projectModelCatalogue` + * without the engine-catalogue module; the read always injects the real + * parser. A stub that returns `null` is the honest "not a wire form" + * answer, which simply skips the builtin annotation — the same path a + * plain id takes. + */ +function defaultParseWireStub() { + return null; +} + +// --------------------------------------------------------------------------- +// The read +// --------------------------------------------------------------------------- + +/** + * Where the catalogue's bytes came from. Always a layered `config`: + * three sources, of which only the `custom_provider` layer is the + * engine's, and the merged shape is webui's v2 `{providers}` view. The + * per-entry `source` field ("engine" | "config" | "builtin") is the + * fine-grained answer; this is the coarse one, kept so the descriptor + * vocabulary matches the other families. + * + * @typedef {"config"} ModelReadSource + */ + +/** + * The #57 (`GET /api/models`) read. + * + * SYNCHRONOUS, deliberately — see the boot-path note in the header. The + * route's handler signature is part of its contract: `app.js#invokeHandler` + * accepts both shapes, but a caller that does not await gets a + * half-written response from an async handler and a complete one from a + * sync handler, and this batch is an absorption, not a scheduling + * change. + * + * Every source is re-read on every call, exactly as before: editing + * `models.json`, `~/.mcode-webui/providers.json` or the engine's + * `config.yaml` must not require a server restart. The `payload` is + * the endpoint's response body verbatim, including the soft-failure + * `reason` marker for an empty catalogue — this facade does not convert + * that into an error, because "no engine attached yet" is a state the + * picker renders, not a failure. + * + * @param {object} [options] + * @param {object} [options.cs] The webui client state; `configOptions`, + * `model.name`, `model.thinking` and `model.contextWindow` are + * read from it, and the first two are echoed into the derived + * figures. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/models`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @returns {{payload: object, source: ModelReadSource, gate: object, transport: string}} + */ +export function readEngineModelCatalogue(options = {}) { + const endpoint = options.endpoint || "GET /api/models"; + const transport = options.transport || MCODE_WEBUI_TRANSPORT; + const gate = checkModelReadCapability(endpoint, transport); + const cs = options.cs || {}; + const sessionOption = configOption(cs, "model"); + // The merged `{providers}` view, or `null` when every layer is + // missing. The engine catalogue read is best-effort: a missing + // `config.yaml` or a YAML parse error yields `[]`, and the merge + // treats an empty engine catalogue as "no engine layer" — matching + // the pre-ticket-06 behaviour for installs without an engine config. + // The `try/catch` is the endpoint's own: a malformed webui layer must + // degrade the catalogue to "webui layers only", never 500 the picker. + let providers = null; + try { + const cfg = loadProvidersConfig(); + const webuiProviders = cfg && Array.isArray(cfg.providers) ? cfg.providers : []; + const merged = mergeEngineAndWebuiProviders(readEngineCatalogue(), webuiProviders); + if (merged.length > 0) providers = { providers: merged }; + } catch { + providers = null; + } + const payload = buildModelCataloguePayload({ + sessionOption, + providers, + builtins: getBuiltinModelsFromMcode(), + builtinThinking: readEngineBuiltinThinking(), + builtinContextWindows: readEngineBuiltinContextWindows(), + cs, + parseEngineModelWireValue, + }); + return { payload, source: "config", gate, transport }; +} diff --git a/packages/webui/server/engine/session-switch.js b/packages/webui/server/engine/session-switch.js new file mode 100644 index 00000000..a04cc467 --- /dev/null +++ b/packages/webui/server/engine/session-switch.js @@ -0,0 +1,992 @@ +// webui/server/engine/session-switch.js +// +// Migration step M3, batch B6: the session SWITCH endpoint (会话切换) — +// +// #3 POST /api/sessions/switch — switch the active session by webui +// uuid or `mvs_…` sid +// +// What this file is for. #3 is the busiest single endpoint in this +// migration and the one whose failure modes are all user-visible at once: +// a wrong answer here loses the conversation on screen, re-roots the file +// tree on the wrong project, or resurrects the "extra untitled entry" +// sidebar confusion. Before M3 all of it lived in the route — resolve, +// overlay creation, title lookup, transcript backfill, workspace +// containment, per-client state mutation, the response body — in one +// ~290-line handler whose middle half (the engine-facing half) reached +// into `lib/acp-client.js` and `lib/transcript.js` directly. Three facts +// about that handler are load-bearing and none of them is visible from +// the route's edge any more: +// +// 1. THE BACKFILL DECISION IS A DATA DECISION, NOT A ROUTE DECISION. +// A stored chat buffer is written when it is empty OR when it looks +// cumulative (a later `●` line is a strict superset of an earlier +// one — the segment-accumulator bug, session-isolation/06). A clean +// stored buffer is kept untouched even though the engine DB is +// DB-authoritative, because transcript-sync overwrites the stored +// chat within ~4s anyway and clobbering a clean buffer on EVERY +// switch is a worse failure than not backfilling. That rule, and +// the predicate that decides it, belong with the reader that backs +// it up — not in a route that would have to know the difference. +// +// 2. THE READ MUST NEVER BREAK THE SWITCH. Every transcript failure +// path — missing db, unloadable better-sqlite3, schema drift, a +// throwing probe — degrades to "keep the stored chat" and the +// switch still answers 200. That is the endpoint's oldest promise +// and it is the reason this family's gate is SOFT (see below): a +// switch that 501s because an enrichment was unavailable has +// turned a degraded read into a dead endpoint. +// +// 3. THE WORKSPACE WRITE IS A CONTAINMENT GATED SIDE EFFECT. The +// target's stored `workspace` is historical input — it may name a +// directory the user has since removed from the allowed roots. The +// switch resolves target-first, NEVER falls back to the workspace +// the user is currently in (that is the reported "file tree still +// shows the previous project" defect), and refuses with a 400 +// rather than writing a path the picker would have rejected. The +// gate is `lib/workspace.js#assertWorkspacePath` — the same one +// `handleWorkspaceChange`, `handleNewSession` and the fs routes use +// — and it runs BEFORE any `cs` mutation, so a refused switch +// leaves the client state exactly as it was. +// +// Why this family's gate is SOFT, when the write family (B5) gates hard +// and the tree family (B2) gates hard. The question behind that choice +// is "if the provider declares this capability absent, can the endpoint +// still serve a truthful answer?" — and for #3 the answer is yes: +// +// - The payload's primary data is webui's OWN store. The record, its +// title, its chat and its workspace all live in `sessions.json`. +// - Both engine touches are enrichments that already have a defined +// degradation: the title falls back to the cache and then to the +// "Mcode session" placeholder, the transcript falls back to the +// stored chat. Neither failure is visible as a failure. +// - Gating hard would REMOVE a working endpoint in response to a +// declaration about a capability it does not depend on, and it would +// do so under exactly the transport where the endpoint has the most +// users. That is B2's `session-export.js` argument, reused rather +// than re-argued: a missing enrichment must not be dressed up as a +// failure (#110 fake-success discipline, applied in the other +// direction). +// +// So `checkSessionSwitchCapability` REPORTS and never throws. The 501 +// machinery in `errors.js` stays unused by this family — a policy +// statement, and the suite pins that it stays unused. +// +// What this file deliberately does NOT do: +// +// - It does not re-implement the transcript. `lib/transcript.js` owns +// the read and `messagesToChatLines` owns the chat-line grammar; this +// file owns the DECISION to read and the decision to keep what came +// back. See KNOWN DEBT 1 for why the 3-candidate probe behind that +// read survives this batch. +// - It does not own the workspace boundary. `assertWorkspacePath` +// stays the single gate every workspace write funnels through. +// - It does not own the session store. `lib/sessions.js` keeps the +// load/save and the overlay rule; this file orders the calls. +// - It does not build a host. There is no host on this path at all. +// +// Boot-path weight. `app.js` imports the routes, the routes import this +// file, so this file is on the boot path. It statically imports nothing +// heavier than `engine/capabilities.js` and `engine/index.js` (both pure +// declaration modules) and nothing else; `lib/sessions.js`, +// `lib/acp-client.js`, `lib/transcript.js`, `lib/mavis-usage.js`, +// `lib/models.js`, `lib/workspace.js`, `lib/state-bus.js` and +// `lib/config.js` are all reached through `await import()` inside the +// data-plane function. That split is the M1 lesson, and it is what lets +// this module be re-exported from `engine/index.js` at all. The pure +// derivations below take their dependencies as arguments for the same +// reason twice over: they stay testable without a module registry, and +// the boot path never sees a workspace or sqlite import. +// +// Provider selection is M4's job, same as B1 through B5: +// `providerByTransport()` maps a transport to a REGISTERED provider id; +// today only `runtime` has one, so under the default `acp` transport the +// gate reports `gate: "unregistered-transport"` and the switch proceeds — +// which is correct, because the pre-M4 behaviour under `acp` is the only +// behaviour this endpoint has ever had. + +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, exactly as in + * `session-reads.js#providerByTransport`, `session-tree-reads.js`, + * `usage-reads.js`, `account-reads.js` and `session-writes.js`, which + * this mirrors rather than merges: six families with separate contracts, + * and a shared table would force this one to inherit another's policy. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. Every + * consumer of the table is a function anyway. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +// --------------------------------------------------------------------------- +// The declaration, and the gate policy that goes with it +// --------------------------------------------------------------------------- + +/** + * The declaration this endpoint's ENRICHMENTS need. + * + * `sessionCrud` / `getSession` is the honest mapping and it is the same + * pair B1 uses for `GET /api/acp-session-title` and B2 uses for the + * export enrichment: reading a session's title and reading its transcript + * are both reading that session. `usageStats` is deliberately NOT + * declared, and the reason is worth stating because the switch does + * touch token usage: `applyMavisUsageToCs` reads webui's OWN mavis usage + * tables, not a provider method, and it is fire-and-forget — its failure + * path has been a `catch` with a debug-only warning since before M3. A + * declaration there would gate a working endpoint on a capability whose + * absence changes nothing the user can see. + * + * @type {Readonly>} + */ +export const SESSION_SWITCH_ENDPOINTS = Object.freeze({ + "POST /api/sessions/switch": Object.freeze({ + capability: "sessionCrud", + subItem: "getSession", + enforcement: "soft", + }), +}); + +/** + * Resolve the provider that answers the switch on `transport`, or `null` + * when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveSessionSwitchProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Read the declaration for this endpoint WITHOUT enforcing it. + * + * Returns a descriptor whose `gate` field says what happened: + * + * - `"checked"` — provider resolved, capability is `full`. + * - `"unregistered-transport"` — no provider claims this transport yet. + * This is the DEFAULT `acp` transport, and the switch proceeding + * here is the pre-M3 behaviour, not a hole in the gate. + * - `"capability-absent"` — the provider WAS found and DOES declare + * the capability as `none`. The caller's next move is to degrade the + * enrichment (placeholder title, stored chat), never to fail the + * request. + * - `"partial"` — provider is `partial` and this sub-item + * is absent; the endpoint still degrades, but says so precisely. + * + * Deliberately never throws `EngineCapabilityNotSupportedError`. A + * genuinely unknown endpoint key is still a plain Error — caller + * confusion is not a capability question, and the HTTP layer must never + * answer 501 for a typo in webui's own code. + * + * @param {string} endpoint A key of SESSION_SWITCH_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null, enforcement: "soft"}} + */ +export function checkSessionSwitchCapability(endpoint, transport) { + const need = SESSION_SWITCH_ENDPOINTS[endpoint]; + if (need === undefined) { + const err = new Error( + `checkSessionSwitchCapability: "${endpoint}" is not part of the session-switch family ` + + `(known: ${Object.keys(SESSION_SWITCH_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_session_switch_endpoint"; + throw err; + } + const base = { + endpoint, + provider: null, + capability: need.capability, + subItem: need.subItem, + enforcement: need.enforcement, + }; + const provider = resolveSessionSwitchProvider(transport); + if (!provider) return { ...base, gate: "unregistered-transport" }; + const entry = provider.capabilities ? provider.capabilities[need.capability] : undefined; + const descriptor = { ...base, provider: provider.id }; + if (entry && entry.level === "full") { + return { ...descriptor, gate: "checked" }; + } + if (entry && entry.level === "partial") { + const absent = Array.isArray(entry.missing) && entry.missing.includes(need.subItem); + return { ...descriptor, gate: absent ? "partial" : "checked" }; + } + // `none`, or no entry at all — the provider was found and does not + // offer this. Report it; the caller degrades the enrichment. + return { ...descriptor, gate: "capability-absent" }; +} + +// --------------------------------------------------------------------------- +// Pure derivations. Exported and tested on their INPUTS. +// --------------------------------------------------------------------------- + +/** + * The engine's own session-id shape, as this endpoint asks it. + * + * The same regex as `session-writes.js#isMcodeSessionId`, spelled again + * rather than imported: the write family's copy is reachable only + * through the write gate's module, and a switch that could not run + * without the delete family's declaration would couple two endpoints + * that have no reason to move together. The rule itself is one line and + * both copies are pinned by both suites, so a drift shows up as a red + * test in whichever family changed, not as a silent behaviour change. + * + * @param {unknown} id + * @returns {boolean} + */ +export function isSwitchableMcodeSessionId(id) { + return typeof id === "string" && /^mvs_[a-f0-9]{32}$/.test(id); +} + +/** + * Resolve a caller-supplied id against the session store. + * + * The order is MCODE SID FIRST, then webui uuid — and that is the OPPOSITE + * of `session-writes.js#resolveSessionTarget`, which is uuid-first. The + * two are not interchangeable and the difference is a product rule, not a + * style choice: the switch's whole point is single base-session identity + * (one conversation, one record — the "extra untitled entry" sidebar + * confusion the overlay rule was written to kill), so a switch addressed + * by `mvs_` must land on the record that IS that engine session even if + * some other record's uuid could be made to match the same string. The + * delete and rename paths are addressed by a user who already has the + * record in front of them and look the uuid up first. + * + * `matchKind` is `null` — never `"unknown"`, never `""` — exactly when the + * id resolved to nothing. The caller writes `matchKind || "new_from_mcode"` + * into the audit payload itself, because that fallback is part of the + * audit contract and is spelled out at its one call site. + * + * @param {Array} records The loaded session store. + * @param {string} id The id from the request. + * @returns {{index: number, matchKind: "webuiId"|"mcodeSessionId"|null, target: object|null}} + */ +export function resolveSwitchTarget(records, id) { + const list = Array.isArray(records) ? records : []; + let index = list.findIndex((s) => s && s.mcodeSessionId === id); + let matchKind = index >= 0 ? "mcodeSessionId" : null; + if (index < 0) { + index = list.findIndex((s) => s && s.id === id); + if (index >= 0) matchKind = "webuiId"; + } + return { + index, + matchKind, + target: index >= 0 ? list[index] : null, + }; +} + +/** + * Detect the cumulative-render pollution pattern in a stored chat buffer + * (session-isolation/06). When the engine emits each segment of an + * `agent_message`, the stream writer emits a new `●` line; a + * non-cumulative buffer has each line containing only its own segment's + * text. A cumulative buffer — the bug — has at least one later `●` line + * whose text is a strict superset of an earlier `●` line (the accumulator + * never reset between segments and every later line re-wrote every prior + * segment's text). + * + * This predicate is O(n^2) in the number of `●` lines, but a single + * session's `chat` is bounded (~400 lines by the transcript cap) so the + * worst case is a few thousand substring checks per switch — cheap + * enough to run on the hot path. + * + * Conservative on both sides: + * - a single-`●`-line buffer is never cumulative; + * - non-`●` lines (system, tool, ▲ thought) are ignored — only `●` + * rows matter, since the cumulative bug only affects message + * segments; + * - ties (equal-length `●` lines) are NOT cumulative — same length, no + * superset relation. + * + * @param {unknown} chat + * @returns {boolean} + */ +export function chatLooksCumulative(chat) { + if (!Array.isArray(chat) || chat.length === 0) return false; + const dots = []; + for (const line of chat) { + if (typeof line !== "string") continue; + // Match the same prefix the streamer writes: `● ` then text. Also + // accept a bare `●` at end-of-line (transcript-sync appends + // stripped-down `●` markers in some paths) without treating it as + // evidence of anything. + if (line.startsWith("● ")) dots.push(line.slice(2)); + } + for (let i = 0; i < dots.length; i += 1) { + for (let j = i + 1; j < dots.length; j += 1) { + const a = dots[i]; + const b = dots[j]; + if (b.length <= a.length) continue; // strict superset ⇒ longer + if (b.includes(a)) return true; + } + } + return false; +} + +/** + * Whether the switch should read the engine transcript for a stored + * buffer, and why — the decision, with no I/O in it. + * + * The rule (session-isolation/06) and its three branches: + * + * - stored chat empty → backfill. Unchanged since the first version of + * this path: a session that has never been rendered must show its + * history, not "No messages yet". + * - stored chat looks cumulative → prefer the engine read and + * re-persist. The original rule only backfilled when the buffer was + * empty, so a polluted buffer persisted via `saveSessions` and won + * forever. `reason` reports which branch fired so the operator log + * distinguishes "first touch" from "repaired pollution". + * - otherwise → keep the stored chat. DB-authoritative: + * transcript-sync overwrites the stored chat from the engine within + * ~4s, so stored-only lines a user typed but never sent will be lost + * regardless, and clobbering a clean buffer on EVERY switch is the + * worse failure. This deliberately does NOT promise draft + * preservation — the composer keeps its own draft in its own state + * (see `composer-draft.test.ts`). + * + * @param {unknown} chat The record's stored `chat` array. + * @returns {{storedHasChat: boolean, storedCumulative: boolean, shouldBackfill: boolean, reason: "empty"|"stored_cumulative"|"stored_shrinks"|null}} + */ +export function selectTranscriptBackfill(chat) { + const storedHasChat = Array.isArray(chat) && chat.length > 0; + const storedCumulative = storedHasChat && chatLooksCumulative(chat); + if (!storedHasChat) { + return { storedHasChat, storedCumulative, shouldBackfill: true, reason: "empty" }; + } + if (storedCumulative) { + return { storedHasChat, storedCumulative, shouldBackfill: true, reason: "stored_cumulative" }; + } + return { storedHasChat, storedCumulative, shouldBackfill: false, reason: "stored_shrinks" }; +} + +/** + * Resolve the title of an `mvs_…` session from the in-memory + * walked-session cache, WITHOUT awaiting anything and WITHOUT touching + * the ACP child. + * + * The cache-first rule is a latency rule with a measured number behind + * it: `getMcodeSessionTitle` boots the ACP child, ~2.17s end-to-end with + * a broken mcode binary, AND used to degrade the title to the "Mcode + * session" placeholder even though the cache already held the real one. + * + * Cross-workspace matching within what the module exposes: the cache + * holds ONE workspace's list, keyed by ws. Both the fresh (30s TTL) and + * the stale (same-ws, TTL-expired) readers are probed, plus the `""` + * key — the unfiltered list, so a cache walked without a workspace still + * answers. A miss returns `null` and the caller falls back to + * `getMcodeSessionTitle`. + * + * The two getters are PARAMETERS rather than imports so this stays a + * pure function over the cache, and so the boot path never reaches + * `lib/acp-client.js` (which carries the ACP client tree). + * + * @param {string} mcodeSessionId + * @param {string} ws The workspace the caller is currently in. + * @param {object} getters `{fresh, stale}` — the two cache readers. + * @returns {string|null} + */ +export function lookupCachedMcodeTitle(mcodeSessionId, ws, getters) { + if (!mcodeSessionId) return null; + const fresh = getters && getters.fresh; + const stale = getters && getters.stale; + if (typeof fresh !== "function" || typeof stale !== "function") return null; + for (const wsKey of [ws || "", ""]) { + for (const getter of [fresh, stale]) { + let sessions = null; + try { + sessions = getter(wsKey); + } catch { + sessions = null; + } + if (!Array.isArray(sessions)) continue; + const hit = sessions.find( + (s) => s && s.sessionId === mcodeSessionId && s.title, + ); + if (hit && hit.title) return hit.title; + } + } + return null; +} + +/** + * Pick the workspace the switched-into session "belongs to" and run it + * through the same containment gate the workspace picker / + * `handleNewSession` / `browseWorkspace` all funnel through. + * + * Source priority (s39 — webui-parity ticket 39: the file tree must + * follow the switched session): + * + * 1. The target session's stored `workspace` field — that IS the + * workspace the user was in when they last had it open, modulo any + * pollution the old code introduced. Real existence + containment + * are checked; an out-of-bounds or stale value surfaces as a 400 + * so the user can either widen the allowed roots or pick a fresh + * workspace, instead of silently landing on the previous project. + * + * 2. `defaultWorkspace` (env `MCODE_WORKSPACE` > mcode TUI cwd.json > + * homedir) when the stored value is empty. Empty is also the value + * seen for (a) records created by the old code that polluted + * freshly-typed mvs sessions with the current `cs.workspace` (the + * data-corruption bug this ticket fixes), and (b) older sessions + * that pre-date the workspace field. `DEFAULT_WORKSPACE` is already + * in the default allowed-roots surface (see + * `getAllowedWorkspaceRoots`), so containment accepts it without env + * setup. + * + * Critical invariants: + * - The switch NEVER keeps `cs.workspace` on the prior project. The + * user-reported symptom was exactly that: "the file tree still shows + * the previous project's files". Falling back to the current + * workspace when the target's is empty is the bug being removed — + * which is why `currentWs` is NOT a parameter of this function even + * though the route still computes it for the log line. + * - The switch NEVER writes a path the containment gate rejected. A + * 400 carrying the gate's actionable error is the only acceptable + * outcome. + * - The switch NEVER overwrites a target session's stored workspace + * with the current one. That was the pollution path; new overlay + * records (mvs_ first-touch) get `workspace: ""` and the + * target-first read lands on the default for them. + * + * Both dependencies are parameters for the same reason as + * `lookupCachedMcodeTitle`: this is a decision over two values, and it + * has to be testable — and boot-path-light — without the workspace + * module and the config module in the graph. + * + * @param {object} target The resolved session record. + * @param {object} deps + * @param {string} deps.defaultWorkspace `DEFAULT_WORKSPACE`. + * @param {(p: string) => {ok: boolean, path?: string, real?: string, error?: string}} deps.assertPath + * @returns {{ok: true, dir: string, real: string|undefined, fallback: boolean}|{ok: false, error: string, attempted: string}} + */ +export function resolveSwitchWorkspace(target, deps) { + const raw = + target && typeof target.workspace === "string" ? target.workspace.trim() : ""; + // Empty / non-string / null → the default workspace. Never the + // current one — that is the user-reported "stays on the old project" + // failure mode this rule removes. + const candidate = raw || (deps && deps.defaultWorkspace) || ""; + const gate = deps.assertPath(candidate); + if (!gate.ok) { + return { ok: false, error: gate.error, attempted: candidate }; + } + return { ok: true, dir: gate.path, real: gate.real, fallback: !raw }; +} + +/** + * The per-client state a switch applies, as a pure field assignment over + * one client's state object. + * + * What it does and does not touch. It sets the identity, the title, the + * chat buffer, zeroes the three cumulative per-session usage counters + * and RE-ROOTS the workspace. It is not responsible for `resetContext` — + * that is a `lib/sessions.js` call with its own mocked parity, and the + * caller runs it right after, so the ordering (`resetContext` sees the + * new identity) stays the caller's to keep. + * + * `lastUsedWorkspace` is deliberately untouched, and that is a product + * rule rather than an omission: last-used is written only by the send + * path (a workspace change / a sent prompt), because switching is + * browsing. Pinning the browsed workspace to the top of the sidebar is + * the user-reported "click any session in C and C auto-sorts first" + * behaviour, and this function is where that is kept true. + * + * @param {object} cs A webui client state. Mutated in place. + * @param {object} opts + * @param {object} opts.target The resolved session record. + * @param {string} opts.workspaceDir The containment-gated directory. + * @returns {object} The same `cs`, for chaining. + */ +export function applySwitchedSessionToClientState(cs, opts) { + const { target, workspaceDir } = opts; + cs.sessionId = target.id; + cs.mcodeSessionId = target.mcodeSessionId || null; + cs.sessionTitle = target.title || "Untitled"; + cs.chat = Array.isArray(target.chat) ? target.chat : []; + cs.usage = { + ...cs.usage, + sessionInput: 0, + sessionOutput: 0, + sessionTotal: 0, + }; + cs.workspace = { + dir: workspaceDir, + branch: null, + tree: null, + }; + return cs; +} + +// --------------------------------------------------------------------------- +// The engine-facing read +// --------------------------------------------------------------------------- + +/** + * Where the transcript read's bytes came from. `engine` when the reader + * answered with lines; `none` when it did not, and the caller keeps the + * stored chat. The value exists so a consumer never has to guess. + * + * @typedef {"engine" | "none"} SessionSwitchTranscriptSource + */ + +/** + * The switch's one engine-facing read: one session's transcript, mapped + * into the webui chat-line grammar, best-effort. + * + * NEVER THROWS. Every failure — unknown endpoint key aside, which is a + * caller bug — lands as `{ok: false, reason}` and the caller keeps the + * stored chat. That containment used to live in a `try/catch` wrapped + * around the whole block in the route; it is a property of the READ + * here, so a future caller of this seam cannot get it wrong. + * + * `lines` / `messageCount` / `truncated` / `probe` are the reader's own + * values forwarded verbatim — this facade invents no reason code and + * never converts a failure into an exception, because the operator log + * that reports `reason` and the log's own vocabulary are one contract. + * + * Async even though the reader is synchronous (better-sqlite3 is sync): + * the route is already async, and a uniform awaitable `readEngine*` + * seam means a provider-backed transcript source that IS async (a network + * engine) needs no signature change at this layer. + * + * @param {object} [options] + * @param {string} [options.mcodeSessionId] The `mvs_…` id to read. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/switch`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @returns {Promise<{mcodeSessionId: string, lines: Array, ok: boolean, reason: string|null, probeTable: string|null, probe: string|null, messageCount: number, truncated: boolean, source: SessionSwitchTranscriptSource, gate: object, transport: string}>} + */ +export async function readEngineSwitchTranscript(options = {}) { + const endpoint = options.endpoint || "POST /api/sessions/switch"; + const [transcript, config] = await Promise.all([ + import("../lib/transcript.js"), + import("../lib/config.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = checkSessionSwitchCapability(endpoint, transport); + const mcodeSessionId = options.mcodeSessionId || ""; + const r = transcript.loadTranscriptChatLines(mcodeSessionId, { + dbPath: config.MCODE_RUNTIME_DB, + }); + return { + mcodeSessionId, + lines: r.ok && Array.isArray(r.lines) ? r.lines : [], + ok: r.ok === true, + reason: r.ok === true ? null : r.reason || "unknown", + probeTable: r.source || null, + probe: r.probe || null, + messageCount: r.messageCount || 0, + truncated: r.truncated === true, + source: r.ok === true ? "engine" : "none", + gate, + transport, + }; +} + +// --------------------------------------------------------------------------- +// Data plane +// --------------------------------------------------------------------------- + +/** + * #3 — the switch. + * + * The order below IS the endpoint's contract, and each step is here + * because moving it would change what the user sees: + * + * 1. LOAD + RESOLVE. `mvs_` sid first, then webui uuid (see + * `resolveSwitchTarget`). + * 2. FIRST TOUCH. An `mvs_` sid with no webui record gets ONE overlay + * record whose id IS the mvs sid (idempotent create), titled from + * the walked-session cache and only then from the engine. An id that + * is neither → `not_found` and the route answers 404. Note that an + * unresolved id that is NOT an mvs sid is a value, not an error: + * the route owns the status code. + * 3. PLACEHOLDER REPAIR. Wrappers created during the broken-title + * window carry "Mcode session" forever; if the walked cache now has + * the real title, repair the stored wrapper. Cache-only, and it runs + * BEFORE the backfill so the repaired title is what the response + * carries. + * 4. TRANSCRIPT BACKFILL, under `selectTranscriptBackfill`'s rule. The + * only step that writes a non-empty buffer, and the only one that + * can fail harmlessly. + * 5. WORKSPACE CONTAINMENT. Runs before ANY `cs` mutation, so a + * refused switch (`workspace_refused`) leaves the client exactly as + * it was — which is why this outcome exists as a third value next to + * `ok` and `not_found` instead of an exception. + * 6. APPLY. Identity, title, chat, usage counters, workspace — then + * `resetContext`, then the fire-and-forget usage sync. + * + * The usage sync is started here and NOT awaited, exactly as the route + * did: it pushes a state frame on its own when it settles, and the + * switch's own response must not wait on a usage table read. + * + * The response body is built HERE and never re-assembled in the route, + * including the `chat` projection: `runChatViewChat` is a pure read of + * the run registry and the client state, and nothing between this call + * and the response mutates either, so computing it one step earlier + * cannot change a byte. The test suite pins the mid-run case (the + * run-mirror contract, session-isolation/02) to keep that true. + * + * @param {object} options + * @param {string} options.id The id from the request; already + * validated non-empty by the route. + * @param {object} options.cs The requesting client's state. Mutated. + * @param {string} [options.cid] Requesting client id. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/switch`. + * @param {string} [options.transport] Transport override. + * @returns {Promise<{outcome: "ok"|"not_found"|"workspace_refused", matchKind: string|null, target: object|null, workspace: object|null, transcript: object|null, audit: object|null, payload: object, statusHint: number, gate: object, transport: string}>} + */ +export async function applyEngineSessionSwitch(options = {}) { + const endpoint = options.endpoint || "POST /api/sessions/switch"; + const [sessions, acp, config, workspaceLib, bus, mavis, models] = await Promise.all([ + import("../lib/sessions.js"), + import("../lib/acp-client.js"), + import("../lib/config.js"), + import("../lib/workspace.js"), + import("../lib/state-bus.js"), + import("../lib/mavis-usage.js"), + import("../lib/models.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = checkSessionSwitchCapability(endpoint, transport); + const { id, cs, cid } = options; + const all = sessions.loadSessions(); + console.log( + `[switch] cid=${cid} incoming id=${id.substring(0, 12)}… isMcodeSid=${isSwitchableMcodeSessionId(id)} allTotal=${all.length}`, + ); + const { matchKind: foundKind, target: found } = resolveSwitchTarget(all, id); + let target = found; + let matchKind = foundKind; + console.log( + `[switch] cid=${cid} match=${matchKind || "NONE"} target.id=${target ? target.id.substring(0, 8) : "null"}… target.mcodeSid=${target && target.mcodeSessionId ? target.mcodeSessionId.substring(0, 12) : "null"}… target.chatLen=${target ? (target.chat ? target.chat.length : 0) : 0} target.title="${target ? (target.title || "").substring(0, 30) : ""}"`, + ); + + if (!target) { + if (!isSwitchableMcodeSessionId(id)) { + console.log( + `[switch] cid=${cid} 404 id=${id} not found and not mcode sid`, + ); + return { + outcome: "not_found", + matchKind: null, + target: null, + workspace: null, + transcript: null, + audit: null, + payload: { ok: false, error: "session not found" }, + statusHint: 404, + gate, + transport, + }; + } + // Cache-first title — the walked session cache usually already holds + // the real title (the sidebar just rendered it). Only a total cache + // miss pays the `getMcodeSessionTitle` cost. + const currentWs = (cs && cs.workspace && cs.workspace.dir) || ""; + let title = lookupCachedMcodeTitle(id, currentWs, { + fresh: acp.getMcodeSessionsCacheSync, + stale: acp.getMcodeSessionsStaleSync, + }); + const titleSource = title ? "cache" : "acp"; + if (!title) { + title = (await acp.getMcodeSessionTitle(id)) || "Mcode session"; + } + // Single base session — overlay record id === mcode session id, + // idempotent create. The old model gave each mvs_ switch a fresh uuid + // wrapper, so the same conversation had two identities, the direct + // cause of the "extra untitled entry" sidebar confusion. + // + // No workspace argument, and none is ever passed: stamping the + // freshly-created overlay with the CURRENT workspace stamped every + // first-touch of an mvs session from project A with project A's + // path, and switching back from project B then either left the file + // tree stuck on B or overwrote the overlay (s39 / webui-parity 63). + // New overlays start with `workspace: ""`; the target-first read + // below lands on the default for them. + const existed = sessions.findOverlayForMcodeSid(all, id); + target = sessions.ensureOverlayForMcodeSid(all, id, { title }); + target.updatedAt = Date.now(); + sessions.saveSessions(all); + // `matchKind` stays `null` here on purpose: the audit payload's + // `matchKind || "new_from_mcode"` fallback is part of the B01 audit + // contract, and "new_from_mcode" is the label operators read when a + // switch invented the wrapper. Labelling it `mcodeSessionId` would + // rewrite history for every first-touch switch. + console.log( + `[switch] cid=${cid} ${existed ? "reused" : "created"} overlay ${target.id.substring(0, 12)}… (id=mcode sid) title="${title}" titleSource=${titleSource}`, + ); + } else if ( + // Placeholder refresh — wrappers created during a broken-title window + // carry "Mcode session" forever. If the walked cache now has the real + // title, repair the stored wrapper. Cache-only (sync, no ACP boot): + // an existing wrapper must never make the hot path slower. + target.title === "Mcode session" && + target.mcodeSessionId && + isSwitchableMcodeSessionId(target.mcodeSessionId) + ) { + const cachedTitle = lookupCachedMcodeTitle( + target.mcodeSessionId, + (cs.workspace && cs.workspace.dir) || "", + { + fresh: acp.getMcodeSessionsCacheSync, + stale: acp.getMcodeSessionsStaleSync, + }, + ); + if (cachedTitle) { + target.title = cachedTitle; + target.updatedAt = Date.now(); + sessions.saveSessions(all); + console.log( + `[switch] cid=${cid} refreshed placeholder title for ${target.id.substring(0, 8)}… → "${cachedTitle}"`, + ); + } + } + + // Transcript backfill — when the resolved target has NO webui chat yet + // but IS a real mvs_ session, load the engine transcript and map it + // into the webui chat-line grammar BEFORE responding, so the response + // `session.chat` and `cs.chat` both carry history. Caps inside the + // reader (last 400 lines / 200KB) keep the SSE state push bounded; a + // 1000+-message session must not balloon it. + // + // FAILURE MUST NOT BREAK SWITCHING: the read is contained in + // `readEngineSwitchTranscript`, and a failure here logs and continues + // with the original chat — the switch itself always succeeds. + let transcript = null; + if (target.mcodeSessionId && isSwitchableMcodeSessionId(target.mcodeSessionId)) { + const decision = selectTranscriptBackfill(target.chat); + if (decision.shouldBackfill) { + try { + const read = await readEngineSwitchTranscript({ + mcodeSessionId: target.mcodeSessionId, + endpoint, + transport, + }); + // `decision` is the BRANCH that fired, not the read's outcome — + // the two answer different questions and the operator log needs + // both ("we re-read because the buffer was polluted" versus "the + // re-read found nothing"). It rides on the read's result because + // that object only exists when a read was actually attempted. + transcript = { ...read, decision: decision.reason }; + if (transcript.ok && transcript.lines.length > 0) { + target.chat = transcript.lines; + target.updatedAt = Date.now(); + sessions.saveSessions(all); // persist the populated wrapper + console.log( + `[switch] cid=${cid} transcript backfill ${target.id.substring(0, 8)}… mcode=${target.mcodeSessionId.substring(0, 12)}… reason=${decision.reason} lines=${transcript.lines.length} msgs=${transcript.messageCount} probe=${transcript.probe}${transcript.truncated ? " (capped)" : ""}`, + ); + } else if (!transcript.ok) { + console.log( + `[switch] cid=${cid} transcript unavailable for ${target.mcodeSessionId.substring(0, 12)}… reason=${transcript.reason || "unknown"}`, + ); + } else if (decision.storedCumulative) { + // Cumulative buffer + the read came back empty — preserve the + // stored chat (which is at least the user's last view) and log + // the discrepancy so a post-mortem can see what happened. + console.log( + `[switch] cid=${cid} stored chat looked cumulative but the transcript read returned no lines; preserving stored chat for ${target.mcodeSessionId.substring(0, 12)}…`, + ); + } + } catch (e) { + // Belt and braces: the read is written not to throw, but a + // module-load failure in the dynamic import would land here, and + // a switch that 500s because a transcript could not be loaded is + // the failure mode this endpoint has never had. + console.warn( + `[switch] cid=${cid} transcript backfill failed for ${target.mcodeSessionId.substring(0, 12)}… (continuing with stored chat):`, + e && e.message ? e.message : e, + ); + } + } + } + + const prevSid = cs.sessionId; + // s39: resolve the target session's workspace and re-point + // `cs.workspace.dir` to it BEFORE any other cs mutation, so the SSE + // state push and the response payload both carry the new workspace in + // lockstep with the session-id switch. The pre-fix behaviour read + // `cs.workspace` without writing it, which left the file tree bound to + // the previous project. + const switchWs = resolveSwitchWorkspace(target, { + defaultWorkspace: config.DEFAULT_WORKSPACE, + assertPath: workspaceLib.assertWorkspacePath, + }); + if (!switchWs.ok) { + console.log( + `[switch] cid=${cid} REFUSED id=${id.substring(0, 12)}… reason=workspace_containment attempted="${switchWs.attempted}"`, + ); + return { + outcome: "workspace_refused", + matchKind, + target, + workspace: switchWs, + transcript, + audit: null, + payload: { + ok: false, + error: switchWs.error, + attempted: switchWs.attempted, + }, + statusHint: 400, + gate, + transport, + }; + } + if (switchWs.fallback) { + console.log( + `[switch] cid=${cid} target ${target.id.substring(0, 8)}… had no workspace — fell back to DEFAULT_WORKSPACE=${switchWs.dir}`, + ); + } + applySwitchedSessionToClientState(cs, { target, workspaceDir: switchWs.dir }); + sessions.resetContext(cs); + // Sync real token usage from the mavis db on switch to a historical + // session. Fire-and-forget, exactly as before: its failure path is a + // debug-only warning and the switch's own response must not wait on a + // usage table read. + if (cs.mcodeSessionId) { + const switchedSid = cs.mcodeSessionId; + mavis + .applyMavisUsageToCs(cs, switchedSid, { getMcodeModelLimit: models.getMcodeModelLimit }) + .then(() => bus.pushStateFor(cid)) + .catch((e) => { + if (process.env.MCODE_USAGE_DEBUG) + console.warn(`[switch.mavis] cid=${cid} error: ${e.message}`); + }); + } + return { + outcome: "ok", + matchKind, + target, + workspace: switchWs, + transcript, + // The audit event. B01: a switch records which session was activated + // and from which prior session, plus (s39) which workspace the switch + // landed on and whether that was the DEFAULT_WORKSPACE fallback — + // both useful when auditing "why did the file tree change" or "why is + // the sidebar sorting by a directory I never opened". The route + // appends it and owns the fail-closed 500, because the write-ahead + // ordering between "know what to switch to" and "tell anyone" is the + // route's to keep. + audit: { + event: "session.switch", + target: cs.sessionId, + cid, + actor: "user", + payload: { + from: prevSid || "", + matchKind: matchKind || "new_from_mcode", + mcodeSessionId: cs.mcodeSessionId || "", + title: cs.sessionTitle, + workspace: switchWs.dir, + workspaceFallback: !!switchWs.fallback, + }, + }, + payload: { + ok: true, + session: { + id: target.id, + mcodeSessionId: cs.mcodeSessionId, + title: cs.sessionTitle, + // s39: surface the new workspace in the response so the client + // (url-restore + session-tree) can update its in-memory state + // without waiting for the SSE state-bus push to land. + workspace: switchWs.dir, + workspaceFallback: !!switchWs.fallback, + // session-isolation/02 (run-mirror): switching back to the + // session that is mid-run must show what it produced so far. + chat: bus.runChatViewChat(cid, cs), + }, + }, + statusHint: 200, + gate, + transport, + }; +} + +// --------------------------------------------------------------------------- +// KNOWN DEBT +// --------------------------------------------------------------------------- +// +// Recorded here rather than fixed, because each item is a decision that +// belongs to a human and not to a refactor: +// +// 1. THE 3-CANDIDATE TRANSCRIPT PROBE IS STILL HERE, and this batch is +// the batch the plan named for retiring it (plan §7: "transcript DB +// 探针 … → getMessages"). It could not be retired without breaking +// this batch's own red line, and the reason is not a matter of taste: +// +// a. THE DEFAULT TRANSPORT HAS NO ENGINE SURFACE. The `acp` +// transport is the default and NO provider is registered for +// it — `providerByTransport()` returns `{runtime: …}` only, +// precisely so this gate reports +// `gate: "unregistered-transport"` and the pre-M3 behaviour +// survives. `cliService.getMessages` is reachable only through +// the v2 catalogue host, which only the `runtime` transport +// boots. Deleting the probe therefore empties the backfill on +// the default transport and on half of the two-transport test +// matrix this batch is gated on. That is red line 1 +// (转录回填) failing, not a refactor completing. +// b. THE TWO READS CAP DIFFERENT THINGS. The probe reads a whole +// session and caps the mapped LINES at 400 / 200KB +// (`messagesToChatLines`). `getMessages` paginates — +// `limit`, `before`, `nextCursor`, `hasMore` — so it caps +// MESSAGES. The two are interchangeable only after proving +// that the tail of a bounded message page yields the same +// 400 lines, which needs a live v2 host to measure. +// c. THE ORDERING IS NOT THE SAME ORDERING. The probe orders +// `created_at_ms ASC, rowid ASC`; `getMessages` orders by +// `MessageQueryService`'s own key. On ties the two disagree, +// and a transcript whose order flips is a transcript the user +// reads wrong. +// d. EXPORT STILL OWNS THE LEGACY CANDIDATES. B2 left +// `GET /api/sessions/:id/export` on the legacy-only probe set +// on purpose — its `mcode_unavailable` shape is byte-pinned by +// existing tests against exactly those three candidates, and +// widening export's set would change its enrichment from +// "unavailable" to "answering", which is a product change, not +// a migration step. +// +// What this batch DID collect is the coupling that made the probe +// look unremovable: `routes/sessions.js` no longer names +// `lib/transcript.js` at all, the read has one seam +// (`readEngineSwitchTranscript`), and the 3-candidate list plus the +// v2 data_json probe are now an implementation detail of the engine +// layer rather than something two routes import directly. The +// remaining work is a SEAM SWAP, not a redesign, and it belongs to +// M4-1 — the batch that registers an ACP provider and therefore +// makes an engine surface reachable under the default transport. +// It should land together with an equivalence test against a live v2 +// host, and with export's probe set widened in the same commit so +// the two endpoints cannot drift apart again. +// +// 2. THE FIRST-TOUCH OVERLAY IS STILL A WEBUI-SIDE WRITE. A bare +// `mvs_…` switch creates a record in `sessions.json` that the +// engine knows nothing about, and the engine's own session list and +// webui's wrapper list are two different questions that happen to +// agree. This is pre-existing behaviour (the alternative — +// registering the session engine-side — is a product decision about +// who owns session identity), and this batch did not change it. +// +// 3. THE USAGE SYNC IS NOT GATED. `applyMavisUsageToCs` reads webui's +// own mavis tables, so it declares no capability, and its failure +// is still swallowed with a debug-only warning. That asymmetry — +// identity and transcript are degraded, usage is dropped silently — +// predates this batch. Naming `usageStats` here would gate a working +// endpoint on a capability whose absence changes nothing visible; +// the real question is whether a silent drop is the right product +// behaviour at all, and that is not this batch's to decide. diff --git a/packages/webui/server/engine/session-writes.js b/packages/webui/server/engine/session-writes.js new file mode 100644 index 00000000..8510328c --- /dev/null +++ b/packages/webui/server/engine/session-writes.js @@ -0,0 +1,936 @@ +// webui/server/engine/session-writes.js +// +// Migration step M3, batch B5: the session WRITE family (会话写族) — the +// three endpoints that change stored state rather than read it: +// +// #7 DELETE /api/sessions/:id — delete a session +// #4 POST /api/sessions/rename — rename a session +// #6 POST /api/sessions/cleanup-orphans — sweep default-named empties +// +// Why a write family needs a facade at all, when a read family is a +// one-liner that forwards. #7 is the only endpoint in the whole migration +// that can DESTROY data the engine owns, and it destroys it three ways +// at once: the engine's own `local_runtime_*` rows, the webui session +// record, and the in-memory caches two readers are assembled from. Three +// facts about that delete are load-bearing and none of them is visible +// at the call site once the route has grown to 270 lines: +// +// 1. THE RESURRECTION GUARD. The long-lived mcode ACP child holds the +// session in memory and rewrites its registry row on the next +// request, so a delete that only removes SQL rows comes BACK. The +// order is the whole mechanism: kill the child → delete the rows → +// drop ONLY the deleted sid from the cache (not the whole cache — +// invalidating everything flashes the sidebar 42 → 16 → 42 and +// reads to the user like the delete failed). A refactor that +// reorders these three steps reintroduces "deleted session +// reappears" without failing any single assertion. +// 2. CACHE INVALIDATION PRECEDES THE ENGINE WRITE. +// `invalidateSessionTree()` runs before the engine delete so the +// next read cannot repopulate a cache from a database this call is +// about to change. Same reason, same asymmetry. +// 3. THE CROSS-TAB FAN-OUT. Every client whose `sessionId` or +// `mcodeSessionId` pointed at the deleted record has its active +// session cleared and its usage counters zeroed, because the next +// interaction in that tab would otherwise silently recreate a webui +// wrapper for the very `mvs_` sid that was just deleted. The +// orphan branch clears only the REQUESTING client, because an +// orphan mcode session has no wrapper for another tab to be +// "inside". That asymmetry is real and load-bearing; flattening it +// would clear tabs that were never on the deleted session. +// +// So the sequencing lives here, named, and tested on its steps; the route +// keeps what is genuinely its own — HTTP parsing, the `authorize()` +// modal, the write-ahead audit ordering, and every status code. +// +// The one thing this file does NOT do is move the SQL. +// `lib/mcode-session-delete.js` keeps the 32-table `local_runtime_*` +// delete (its own header, its own per-table error classification, its +// own `getDb` seam) and this file reaches it through `await import()`. +// That is the same split B3 and B4 drew for their storage access +// (`lib/mavis-usage.js` owns the usage SQL, `lib/mcode-rpc.js` owns the +// account RPC), and it is the only shape that survives a real +// second reader appearing. The plan for this batch annotated +// `mcode-session-delete.js` "delete"; it is KEPT, and the reason is +// recorded as KNOWN DEBT in the module header of +// `lib/mcode-session-delete.js` itself. `lib/acp-client.js` imports +// `deleteMcodeSessionFromDb` from it, and four test files +// (`mcode-session-delete.test.js`, +// `mcode-session-delete-outcomes.test.js`, `sqlite-resolver-c01.test.js`, +// `sessions-switch.check.mjs`) bind to that exact specifier — deleting +// the module would break a live consumer and silently de-mock two +// existing route suites. KNOWN DEBT means "recorded and still +// uncollected", not "safe to remove". +// +// Boot-path weight. `app.js` imports the routes, the routes import this +// file, so this file is on the boot path. It statically imports nothing +// heavier than `capabilities.js` and `index.js` (both pure declaration +// modules); `lib/sessions.js`, `lib/acp-client.js`, +// `lib/mcode-session-delete.js`, `lib/session-tree.js`, `lib/state-bus.js` +// and `lib/config.js` are all reached through `await import()` inside +// the functions. That split is the M1 lesson, and it is what lets this +// module be re-exported from `engine/index.js` at all. +// +// Provider selection is M4's job, same as B1 through B4: +// `providerByTransport()` maps a transport to a REGISTERED provider id; +// today only `runtime` has one, so under the default `acp` transport the +// gate reports `gate: "unregistered-transport"` and the write proceeds — +// which is correct, because the pre-M4 behaviour under `acp` is the +// only behaviour these endpoints have ever had. + +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +// `node:fs` is a builtin, not a project dependency, and `usage-reads.js` +// already reaches for it at module scope for the same reason. It is here +// for exactly two calls: the orphan sweep's "is there a sessions store +// at all" probe and its BOM-tolerant read. +import { existsSync, readFileSync } from "node:fs"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, exactly as in + * `session-reads.js#providerByTransport`, + * `session-tree-reads.js#providerByTransport`, + * `usage-reads.js#providerByTransport` and + * `account-reads.js#providerByTransport`, which this mirrors rather than + * merges: the five families have separate contracts, and a shared table + * would force the write family to inherit a read family's policy. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. Every + * consumer of the table is a function anyway. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +// --------------------------------------------------------------------------- +// The declaration, and the gate policy that goes with it +// --------------------------------------------------------------------------- + +/** + * The declaration each endpoint of this family needs, the sub-item it + * needs from that capability, and HOW that declaration is enforced. + * + * The third field is this family's own addition, and it is not + * decoration — the hard/soft question for a write is decided by WHO + * OWNS THE ROWS THE WRITE DESTROYS, which is a different question from + * the read families' "is the data engine data or webui data", and it + * does not have the same answer twice in a row here: + * + * - #7 DELETE — **hard** on `sessionCrud` · `deleteSession`. The write + * destroys rows in the ENGINE's own `local_runtime_*` tables. There + * is no webui-side copy of a transcript that survives: once those + * rows are gone, the conversation is gone. A provider that declares + * no session deletion genuinely cannot have this endpoint serve a + * truthful answer, and the honest one is the 501 that + * `app.js#invokeHandler` derives from + * `EngineCapabilityNotSupportedError`. This is B4's account-read + * reasoning applied to a write: the data has exactly one owner, and + * it is not us. + * + * - #6 cleanup-orphans — **hard** on the SAME + * `sessionCrud` · `deleteSession` pair, deliberately. The sweep + * selects webui-side orphan RECORDS, but each selected id is fed + * through #7's real-delete branch, and a record carrying an + * `mcodeSessionId` takes the engine's rows down with it. Gating the + * sweep soft would mean a provider that cannot delete engine + * sessions could still reach the engine's tables through a back + * door — the exact shape this batch exists to close. A sweep that + * authorizes, writes its intent audit event and then fails every + * single delegated delete is also the fake-success shape: an + * authorized destructive action that accomplished nothing. + * + * - #4 rename — **no capability at all**, and this row is the one a + * reader will double-take, so here is the whole argument. Rename + * writes `item.title` / `item.titleCustom` / `item.updatedAt` into + * webui's OWN session store and nothing else: not the engine, not + * `local_runtime_sessions`, not any provider method. Its one engine + * touch is `invalidateSessionTree()`, a cache drop — the read-side + * consequence of the sidebar projecting titles from the engine, and + * the projection itself is B2's `GET /api/session-tree`, which + * carries its own gate. Naming a capability here would be a lie of + * the same kind B3 declined for `GET /api/usage/forecast`: a write + * that touches no engine surface must not be gated on an engine + * declaration, because gating it hard would remove a working + * endpoint in response to a statement about something it does not + * depend on. Note what this row also records about the product: a + * rename is a webui-side LABEL, and the engine's own title is not + * touched. That is pre-existing behaviour and this batch does not + * change it — see KNOWN DEBT at the end of this header. + * + * Every row carries all three keys, including the row that has no + * capability. B3 expressed "no engine surface" as a `null` table entry; + * this family has three endpoints of which two DO cross the seam, and a + * `null` hole in the middle of the table is the kind of shape a later + * edit mistakes for "not filled in yet". Uniform rows make the + * enforcement decision reviewable as one diff. + * + * @typedef {{capability: string|null, subItem: string|null, enforcement: "hard"|"soft"|"none"}} SessionWriteDeclaration + * @type {Readonly>} + */ +export const SESSION_WRITE_ENDPOINTS = Object.freeze({ + "DELETE /api/sessions/:id": Object.freeze({ + capability: "sessionCrud", + subItem: "deleteSession", + enforcement: "hard", + }), + "POST /api/sessions/rename": Object.freeze({ + capability: null, + subItem: null, + enforcement: "none", + }), + "POST /api/sessions/cleanup-orphans": Object.freeze({ + capability: "sessionCrud", + subItem: "deleteSession", + enforcement: "hard", + }), +}); + +/** + * Resolve the provider that answers session writes on `transport`, or + * `null` when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveSessionWriteProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Check one endpoint of this family against the active provider's + * declaration. Throws `EngineCapabilityNotSupportedError` — which + * `app.js#invokeHandler` turns into 501 — when the declaration says the + * capability (or the exact sub-item) is absent. + * + * Every row of `SESSION_WRITE_ENDPOINTS` is enforced at the strength its + * `enforcement` field names, and today only `"hard"` rows can throw: + * `"soft"` reports and returns (B2's `session-export.js` policy, for a + * family that has no soft row yet — the field is declared uniform so + * that adding one is a table edit rather than a signature change), and + * `"none"` never consults the provider at all. + * + * @param {string} endpoint A key of SESSION_WRITE_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null, enforcement: string}} + */ +export function assertSessionWriteCapability(endpoint, transport) { + const need = SESSION_WRITE_ENDPOINTS[endpoint]; + if (need === undefined) { + // Caller confusion, not an engine limitation — a plain Error so the + // HTTP layer never answers 501 for a typo in webui's own code. + const err = new Error( + `assertSessionWriteCapability: "${endpoint}" is not part of the session write family ` + + `(known: ${Object.keys(SESSION_WRITE_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_session_write_endpoint"; + throw err; + } + const provider = resolveSessionWriteProvider(transport); + if (need.capability === null) { + return { + endpoint, + gate: "no-capability-key", + provider: provider ? provider.id : null, + capability: null, + subItem: null, + enforcement: need.enforcement, + }; + } + if (!provider) { + return { + endpoint, + gate: "unregistered-transport", + provider: null, + capability: need.capability, + subItem: need.subItem, + enforcement: need.enforcement, + }; + } + assertEngineCapability(provider.capabilities, need.capability, provider.id, need.subItem); + return { + endpoint, + gate: "checked", + provider: provider.id, + capability: need.capability, + subItem: need.subItem, + enforcement: need.enforcement, + }; +} + +// --------------------------------------------------------------------------- +// Pure derivations. Exported and tested on their INPUTS. +// --------------------------------------------------------------------------- + +/** + * The engine's own session-id shape. Four call sites in the pre-facade + * delete path spelled this regex out inline, which is how a fifth call + * site eventually spelled it with a different quantifier. It is the + * predicate that separates "an id the engine minted" (orphan branch: + * delete the engine's rows directly) from "an id webui minted" (wrapper + * branch), so it is named rather than repeated. + * + * @param {unknown} id + * @returns {boolean} + */ +export function isMcodeSessionId(id) { + return typeof id === "string" && /^mvs_[a-f0-9]{32}$/.test(id); +} + +/** + * Resolve a caller-supplied id against the session store: by webui uuid + * first, then by the engine sid a record is bound to. + * + * This is the single-identity rule made explicit, and it was duplicated + * verbatim in `handleRenameSession` and `handleDeleteSession` before + * this batch — the same eleven lines, twice, with the same three + * possible answers. A third write endpoint would have been a third copy, + * and the copy that drifts is the one where a rename resolves a session + * the delete path cannot find, or the reverse. + * + * `matchKind` is `null` — never `"unknown"`, never `""` — exactly when + * the id resolved to nothing. Callers that need a label for the audit + * payload write `matchKind || "unknown"` themselves, because the two + * places that do (#7's `authorize()` context and #7's intent event) + * spell that fallback out and it is part of the audit contract. + * + * @param {Array} records The loaded session store. + * @param {string} id The id from the request. + * @returns {{index: number, matchKind: "webuiId"|"mcodeSessionId"|null, target: object|null}} + */ +export function resolveSessionTarget(records, id) { + const list = Array.isArray(records) ? records : []; + let index = list.findIndex((s) => s && s.id === id); + let matchKind = index >= 0 ? "webuiId" : null; + if (index < 0) { + index = list.findIndex((s) => s && s.mcodeSessionId === id); + if (index >= 0) matchKind = "mcodeSessionId"; + } + return { + index, + matchKind, + target: index >= 0 ? list[index] : null, + }; +} + +/** + * The staleness window the orphan sweep uses — 24h. Matches + * `lib/sessions.js#cleanupEmptyDefaultSessions`, which prunes the same + * class of leftover at startup; the sweep endpoint and the startup pass + * agree on what "leftover" means, and a future edit that moves one of + * them must move both. + */ +export const ORPHAN_STALE_MS = 24 * 60 * 60 * 1000; + +/** + * The "empty AND default-titled AND older than a day" rule behind + * `POST /api/sessions/cleanup-orphans`, as a pure predicate over ONE + * record. Split out of the store read so the rule is testable without a + * file and so the threshold is named rather than inlined at the filter + * site. + * + * `(record.title || "").trim()` is kept exactly as it was, including its + * behaviour on a non-string truthy title (a `TypeError`, which + * propagates out of the sweep as it always has). Tightening it here + * would be a behaviour change dressed as a hardening, and this batch + * promises none. + * + * @param {object} record + * @param {number} now Epoch ms, injected so the rule is pure. + * @param {number} staleMs The staleness threshold. + * @returns {boolean} + */ +export function isOrphanSessionRecord(record, now, staleMs) { + if (!record || !record.id) return false; + const hasChat = Array.isArray(record.chat) && record.chat.length > 0; + if (hasChat) return false; + const title = (record.title || "").trim(); + const isDefault = + title === "New session" || title === "Untitled" || /^对话 \d+$/.test(title); + if (!isDefault) return false; + if (record.updatedAt && now - record.updatedAt < staleMs) return false; + return true; +} + +/** + * The ids `POST /api/sessions/cleanup-orphans` would delete, in store + * order, under `isOrphanSessionRecord`. + * + * @param {Array} records + * @param {object} [options] + * @param {number} [options.now] Epoch ms; defaults to `Date.now()`. + * @param {number} [options.staleMs] Defaults to `ORPHAN_STALE_MS`. + * @returns {string[]} + */ +export function selectOrphanSessionIds(records, options = {}) { + const now = options.now === undefined ? Date.now() : options.now; + const staleMs = options.staleMs === undefined ? ORPHAN_STALE_MS : options.staleMs; + const list = Array.isArray(records) ? records : []; + return list.filter((s) => isOrphanSessionRecord(s, now, staleMs)).map((s) => s.id); +} + +/** + * The per-client state reset a delete fans out, as a PURE field + * assignment over one client's state object. + * + * Note what it does and does not touch. It clears the identity + * (`sessionId`, `mcodeSessionId`), the title and the chat buffer. It is + * not responsible for `resetContext` — that is a `lib/sessions.js` call + * with its own mocked parity in the test helper, and the caller runs it + * right after this so the ordering (`resetContext` sees the cleared + * identity) is the caller's to keep. + * + * `resetUsage` exists because the two delete branches genuinely differ + * here and the difference predates this batch. The wrapper branch zeroes + * the three cumulative session-usage counters, because the tab was + * showing a real session's spend and must stop. The orphan branch does + * NOT, because an orphan mcode session has no webui record and no tab + * can have accumulated webui-side per-session usage against it. Zeroing + * them there would be harmless; unifying the two branches is a product + * decision, not a refactor, so the asymmetry is a parameter with a + * comment rather than a silent difference between two call sites. + * + * @param {object} cs A webui client state. Mutated in place — every + * consumer of this predicate is already mutating `cs` in place. + * @param {object} [options] + * @param {boolean} [options.resetUsage] Default true (the wrapper + * branch). False for the orphan branch; see above. + * @returns {object} The same `cs`, for chaining. + */ +export function applyDeletedSessionToClientState(cs, options = {}) { + const resetUsage = options.resetUsage !== false; + cs.sessionId = null; + cs.mcodeSessionId = null; + cs.sessionTitle = "Untitled"; + cs.chat = []; + if (resetUsage) { + cs.usage = { + ...cs.usage, + sessionInput: 0, + sessionOutput: 0, + sessionTotal: 0, + }; + } + return cs; +} + +/** + * The per-client title fan-out a rename performs, as a pure assignment. + * Extracted for the same reason as the delete reset: the rename path + * runs it once per client whose identity matches, and a test that wants + * to prove "the other tab's title changed too" should be able to point at + * a named predicate instead of re-deriving the match rule. + * + * @param {object} cs + * @param {string} title The new title, already trimmed and validated. + * @returns {object} The same `cs. + */ +export function applyRenamedSessionToClientState(cs, title) { + cs.sessionTitle = title; + return cs; +} + +/** + * Whether a client is inside the record a RENAME is renaming, and so + * needs its title pushed. Matches on the record's webui id OR on the + * engine sid THE RECORD is bound to. + * + * This is deliberately NOT the same predicate as + * `clientMatchesDeletedSession`, even though both were one inline + * condition before this batch. They differ on the second clause, and the + * difference is load-bearing in both directions: + * + * - rename matches `record.mcodeSessionId`, because the record is the + * subject and every tab that adopted that engine session should see + * the new label. + * - delete matches the REQUEST id, because a tab is only "inside" the + * deletion if it is pointing at what the user asked to delete. A + * tab bound to the record's engine sid under a different webui id is + * a different wrapper record and must not be cleared. + * + * Merging them would either resurrect a wrapper in a tab the user just + * cleared, or blank the title of an unrelated tab. They stay two + * predicates, each named for the branch that uses it. + * + * @param {object} cs + * @param {object} record The session record being renamed. + * @returns {boolean} + */ +export function clientMatchesRenamedSession(cs, record) { + if (!cs || !record) return false; + if (cs.sessionId === record.id) return true; + return !!(record.mcodeSessionId && cs.mcodeSessionId === record.mcodeSessionId); +} + +/** + * Whether a client is inside the session a DELETE removed, and so needs + * its active session cleared. Matches on the record's webui id OR on the + * id the request named. See `clientMatchesRenamedSession` for why this + * is not the same predicate. + * + * @param {object} cs + * @param {object} record The deleted record; `null` for the orphan + * branch, where there is no record to match against. + * @param {string} requestId The id the caller asked to delete. + * @returns {boolean} + */ +export function clientMatchesDeletedSession(cs, record, requestId) { + if (!cs) return false; + if (record && cs.sessionId === record.id) return true; + return !!requestId && cs.mcodeSessionId === requestId; +} + +// --------------------------------------------------------------------------- +// Data-plane writes +// --------------------------------------------------------------------------- + +/** + * Load the store and resolve the requested id, without mutating + * anything. This is the half of #7 that has to happen BEFORE + * `authorize()` (the modal is shown for a specific record with a + * specific match kind and chat length) and before the write-ahead intent + * audit (which records the same three facts). + * + * Splitting plan from commit is what keeps the audit chain intact. The + * route must be able to interleave a governance decision and a durable + * event between "know what the user asked to delete" and "delete it", + * and a facade that owned the whole operation would have swallowed that + * ordering into a callback. Nothing here touches the database, the + * store, the caches or any client state. + * + * @param {object} options + * @param {string} options.id The requested id. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/:id`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @returns {Promise<{id: string, records: Array, index: number, matchKind: string|null, target: object|null, isOrphan: boolean, chatLen: number, gate: object, transport: string}>} + */ +export async function planEngineSessionDelete(options = {}) { + const endpoint = options.endpoint || "DELETE /api/sessions/:id"; + const [sessions, config] = await Promise.all([ + import("../lib/sessions.js"), + import("../lib/config.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertSessionWriteCapability(endpoint, transport); + const records = sessions.loadSessions(); + const { index, matchKind, target } = resolveSessionTarget(records, options.id); + return { + id: options.id, + records, + index, + matchKind, + target, + isOrphan: index < 0, + // `target && Array.isArray(target.chat)` rather than + // `Array.isArray(target?.chat)`: a store record that is not an + // object must read as "no chat", and the audit payload's `chatLen` + // has always been 0 for that case. + chatLen: index >= 0 && target && Array.isArray(target.chat) ? target.chat.length : 0, + gate, + transport, + }; +} + +/** + * #7's orphan branch: the id is an `mvs_…` sid with NO webui record, so + * there is no wrapper to remove and the only thing to delete is the + * engine's own rows. + * + * The resurrection guard and the cache drop are ordered deliberately and + * the order is the feature (see this file's header): kill the child + * that would rewrite the registry row, then delete, then drop the ONE + * cache entry — never the whole cache. + * + * `dryRun` suppresses the kill and the cache drop, because a preview + * mutates nothing and a preview that shuts down the user's ACP child is + * a side effect the `?dryRun=true` contract does not include. The COUNT + * still runs, read-only, inside `lib/mcode-session-delete.js`. + * + * The requesting client is reset when — and only when — it was + * currently sitting on that sid. No other tab can be: an orphan has no + * webui record for a tab to be inside. + * + * @param {object} options + * @param {object} options.plan A `planEngineSessionDelete` result. + * @param {object} [options.cs] The requesting client's state. + * @param {string} [options.cid] Requesting client id, for the state push. + * @param {boolean} [options.dryRun] + * @returns {Promise<{mcodeDbDel: object, payload: object, failed: boolean}>} + */ +export async function commitEngineOrphanSessionDelete(options = {}) { + const { plan, cs, cid, dryRun = false } = options; + const [deleter, config, acp, tree, bus, sessions] = await Promise.all([ + import("../lib/mcode-session-delete.js"), + import("../lib/config.js"), + import("../lib/acp-client.js"), + import("../lib/session-tree.js"), + import("../lib/state-bus.js"), + import("../lib/sessions.js"), + ]); + const id = plan.id; + if (!dryRun) { + // The child shutdown is wrapped in try/catch exactly as the + // pre-facade `killMcodeSessionResurrection` wrapped it: a live child + // that refuses to die must not abort the delete that follows. The + // cache drop is not wrapped, because a cache that cannot be dropped + // is the resurrection this branch exists to prevent. + try { + acp.shutdownMcodeAcpSingleton(); + } catch {} + acp.dropMcodeSessionFromCache(id); + } + const mcodeDbDel = deleter.deleteMcodeSessionFromDb(id, { + MCODE_RUNTIME_DB: config.MCODE_RUNTIME_DB, + dryRun, + }); + if (!dryRun) tree.invalidateSessionTree(); + if (!mcodeDbDel.ok) { + return { + mcodeDbDel, + failed: true, + payload: { ok: false, error: "orphan mcode delete failed", mcodeDbDel }, + }; + } + if (cs && cs.mcodeSessionId === id) { + // `resetUsage: false` — see `applyDeletedSessionToClientState`. An + // orphan has no webui record, so no tab accumulated per-session + // usage against it. + applyDeletedSessionToClientState(cs, { resetUsage: false }); + sessions.resetContext(cs); + bus.pushStateFor(cid); + } + return { + mcodeDbDel, + failed: false, + payload: { + ok: true, + deleted: id, + matchKind: "orphan_mcode", + dryRun, + mcodeDbDel, + }, + }; +} + +/** + * #7's `?dryRun=true` preview for a record that DOES have a webui + * wrapper: the readonly per-table count for the linked engine session, + * plus the webui entry that WOULD be removed. Nothing is written, no + * child is killed, no cache is dropped. + * + * A record with no `mcodeSessionId` (a webui-only session that never + * reached the engine) still previews — with an empty log and zero rows, + * the same literal the pre-facade route used to inline. A preview that + * refused to answer for those would be a new failure mode. + * + * @param {object} options + * @param {object} options.plan A `planEngineSessionDelete` result. + * @returns {Promise<{mcodeDbDel: object, payload: object}>} + */ +export async function previewEngineSessionDelete(options = {}) { + const { plan } = options; + const [deleter, config] = await Promise.all([ + import("../lib/mcode-session-delete.js"), + import("../lib/config.js"), + ]); + const mcodeSid = plan.target ? plan.target.mcodeSessionId : undefined; + const mcodeDbDel = mcodeSid + ? deleter.deleteMcodeSessionFromDb(mcodeSid, { + MCODE_RUNTIME_DB: config.MCODE_RUNTIME_DB, + dryRun: true, + }) + : { ok: true, dryRun: true, log: [], totalRows: 0 }; + return { + mcodeDbDel, + payload: { + ok: true, + dryRun: true, + matchKind: plan.matchKind, + mcodeDbDel, + webuiEntryWouldBeDeleted: { + id: plan.target.id, + title: plan.target.title, + mcodeSessionId: mcodeSid, + }, + }, + }; +} + +/** + * #7's real delete of a record that HAS a webui wrapper: splice the + * store, persist it, drop the tree cache, mirror the delete on the + * engine, then fan the cleared state out to every tab that was inside + * the record. + * + * The order is load-bearing in three places, all noted above: the tree + * cache is dropped BEFORE the engine write so a concurrent read cannot + * repopulate it from the pre-delete database; the engine mirror runs + * only when the record carries an `mcodeSessionId` (a webui-only session + * has no engine rows, and calling the deleter with `undefined` would + * report `not_mcode_sid` into the audit payload as if it had failed); + * and the fan-out runs AFTER both, so a tab is never told its session is + * gone while the rows still exist. + * + * `touchedCids` falls back to `[cid]` when no tab matched. That is not a + * no-op: it guarantees the requesting tab always gets a state push, so + * the client cannot be left rendering a session the server has already + * deleted. + * + * @param {object} options + * @param {object} options.plan A `planEngineSessionDelete` result. + * @param {string} [options.cid] Requesting client id. + * @returns {Promise<{deletedItem: object, records: Array, mcodeDbDel: object|null, touchedCids: string[], payload: object}>} + */ +export async function commitEngineSessionDelete(options = {}) { + const { plan, cid } = options; + const [deleter, config, acp, tree, bus, sessions] = await Promise.all([ + import("../lib/mcode-session-delete.js"), + import("../lib/config.js"), + import("../lib/acp-client.js"), + import("../lib/session-tree.js"), + import("../lib/state-bus.js"), + import("../lib/sessions.js"), + ]); + const deletedItem = plan.records[plan.index]; + const records = plan.records; + records.splice(plan.index, 1); + sessions.saveSessions(records); + tree.invalidateSessionTree(); + const mcodeSid = deletedItem.mcodeSessionId; + let mcodeDbDel = null; + if (mcodeSid) { + try { + acp.shutdownMcodeAcpSingleton(); + } catch {} + acp.dropMcodeSessionFromCache(mcodeSid); + mcodeDbDel = deleter.deleteMcodeSessionFromDb(mcodeSid, { + MCODE_RUNTIME_DB: config.MCODE_RUNTIME_DB, + }); + // The pre-facade route logged this from inside the `if (mcodeSid)` + // block, so the engine-mirror line only appears for records that + // actually have one. Kept here, next to the call it describes, so + // the operator log and the code that produced it stay together. + console.log( + `[delete] mcode db delete sid=${mcodeSid.substring(0, 12)}… ok=${mcodeDbDel.ok}` + + (mcodeDbDel.ok + ? ` log=[${(mcodeDbDel.log || []).join(",")}]` + : ` reason=${mcodeDbDel.reason || "-"} error=${mcodeDbDel.error || "-"}`), + ); + } + const touchedCids = []; + for (const [c, ccs] of bus.clients) { + if (!clientMatchesDeletedSession(ccs, deletedItem, plan.id)) continue; + applyDeletedSessionToClientState(ccs); + sessions.resetContext(ccs); + touchedCids.push(c); + } + // Exactly the pre-facade fallback, including the `undefined` it would + // push when the caller supplied no cid: the point is that the + // requesting tab ALWAYS gets a state push, so it cannot be left + // rendering a session the server has already deleted. + if (touchedCids.length === 0) touchedCids.push(cid); + for (const c of touchedCids) bus.pushStateFor(c); + return { + deletedItem, + records, + mcodeDbDel, + touchedCids, + payload: { + ok: true, + deleted: plan.id, + matchKind: plan.matchKind, + dryRun: false, + remaining: records.length, + mcodeDbDel, + }, + }; +} + +/** + * #4 — the rename write. + * + * Everything this endpoint persists lands in webui's own session store. + * The single engine touch is `invalidateSessionTree()`, and it is there + * because the sidebar tree reads titles out of the engine — the same + * reason the pre-facade route had it. The engine's own title is NOT + * written; see the `SESSION_WRITE_ENDPOINTS` row for why that makes the + * capability declaration `null` rather than a guess. + * + * The `not_found` outcome is a value, not an exception: a bare `mvs_…` + * id with no webui record gets an overlay record to carry the title + * (the single-identity rule, same as the switch path), and anything + * else is a 404 because the id is simply wrong. Returning which of the + * three happened is what lets the route write the right status without + * this module knowing what a status is. + * + * Validation of `id` and `title` is NOT done here. It is HTTP request + * validation with three 400 bodies this module would then have to + * reproduce byte for byte, and the route already owns the request. + * + * @param {object} options + * @param {string} options.id The record's webui uuid or `mvs_…` sid. + * @param {string} options.title New title; already trimmed and validated. + * @param {string} [options.cid] Requesting client id. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/rename`. + * @param {string} [options.transport] Transport override. + * @returns {Promise<{outcome: "ok"|"not_found", matchKind: string, from: string, to: string, item: object|null, payload: object|null, gate: object, transport: string}>} + */ +export async function applyEngineSessionRename(options = {}) { + const endpoint = options.endpoint || "POST /api/sessions/rename"; + const [sessions, config, tree, bus] = await Promise.all([ + import("../lib/sessions.js"), + import("../lib/config.js"), + import("../lib/session-tree.js"), + import("../lib/state-bus.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertSessionWriteCapability(endpoint, transport); + const id = options.id; + const title = options.title; + const all = sessions.loadSessions(); + const { index, matchKind: foundKind } = resolveSessionTarget(all, id); + let item; + let matchKind; + if (index < 0) { + if (!isMcodeSessionId(id)) { + return { + outcome: "not_found", + matchKind: null, + from: "", + to: title, + item: null, + payload: { ok: false, error: "session not found" }, + gate, + transport, + }; + } + // A bare mvs_ id with no webui shell gets one created to carry the + // title. No workspace argument, and none was ever passed: stamping + // the caller's current workspace onto someone else's record + // attributes a workspace the session never ran in, and re-roots the + // file tree on every later switch (webui-parity 63, defect F). + item = sessions.ensureOverlayForMcodeSid(all, id); + matchKind = "orphan_mcode"; + } else { + item = all[index]; + matchKind = foundKind; + } + const from = item.title || ""; + item.title = title; + item.titleCustom = true; + item.updatedAt = Date.now(); + sessions.saveSessions(all); + tree.invalidateSessionTree(); + let touchedCids = []; + for (const [c, ccs] of bus.clients) { + if (!clientMatchesRenamedSession(ccs, item)) continue; + applyRenamedSessionToClientState(ccs, title); + touchedCids.push(c); + } + if (touchedCids.length === 0) touchedCids.push(options.cid); + for (const c of touchedCids) bus.pushStateFor(c); + return { + outcome: "ok", + matchKind, + from, + to: title, + item, + payload: { + ok: true, + session: { + id: item.id, + mcodeSessionId: item.mcodeSessionId || null, + title: item.title, + titleCustom: true, + }, + }, + gate, + transport, + }; +} + +/** + * #6 — read the orphan sweep's target list. + * + * The file read stays here rather than in the route because the rule + * and the bytes it reads are one decision: a sweep that read a different + * file than the one whose rule it applies would be a bug waiting for a + * config change. The BOM strip is the store's own on-disk convention + * (written by an editor, not by webui) and is preserved exactly; a + * parse failure answers `[]`, which the pre-facade code did too, and a + * corrupt store must not turn a cleanup request into a 500. + * + * The response shape this backs is the batch's byte-for-byte red line, + * so the payload is built HERE and never re-assembled in the route: + * `{ok, dryRun, count, ids}` — four keys, in that order, for the + * preview; `{ok, dryRun:false, deleted, ids}` for the no-op real path. + * + * @param {object} [options] + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/cleanup-orphans`. + * @param {string} [options.transport] Transport override. + * @returns {Promise<{ids: string[], payload: object, gate: object, transport: string}>} + */ +export async function readOrphanSessionWriteIds(options = {}) { + const endpoint = options.endpoint || "POST /api/sessions/cleanup-orphans"; + const config = await import("../lib/config.js"); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertSessionWriteCapability(endpoint, transport); + const dbPath = config.SESSIONS_DB; + let records = []; + if (existsSync(dbPath)) { + try { + let raw = readFileSync(dbPath, "utf8"); + if (raw.charCodeAt(0) === 0xfeff) raw = raw.slice(1); + const parsed = JSON.parse(raw); + if (Array.isArray(parsed)) records = parsed; + } catch { + records = []; + } + } + const ids = selectOrphanSessionIds(records); + return { + ids, + payload: { ok: true, dryRun: true, count: ids.length, ids }, + gate, + transport, + }; +} + +// --------------------------------------------------------------------------- +// KNOWN DEBT +// --------------------------------------------------------------------------- +// +// Recorded here rather than fixed, because each item is a decision that +// belongs to a human and not to a refactor: +// +// 1. `lib/mcode-session-delete.js` still owns the 32-table SQL. The +// plan for this batch annotated it "delete"; it is kept because +// `lib/acp-client.js` imports from it and four test files bind to +// the specifier. Collecting it means moving those first. +// +// 2. #4 rename writes a WEBUI-side label only. The engine's own title +// in `local_runtime_sessions` is untouched, while the sidebar tree +// reads its titles from the engine. So for an engine-backed +// session a rename can be visible in the wrapper list and not in +// the tree. This is pre-existing behaviour and this batch did not +// change it; closing it means deciding which store is +// authoritative for a display title, which is a product call. +// +// 3. #7 does not detect "this session is running right now". A delete +// of an in-flight session kills the ACP child out from under the +// turn. That is the pre-facade behaviour and it is arguably the +// correct one (the user asked), but "refuse to delete a running +// session" is a defensible alternative and the choice is not this +// batch's to make. diff --git a/packages/webui/server/routes/account.js b/packages/webui/server/routes/account.js index 915c9a18..f9efb57f 100644 --- a/packages/webui/server/routes/account.js +++ b/packages/webui/server/routes/account.js @@ -1,7 +1,21 @@ // webui/server/routes/account.js // GET /api/account — the account card's data. +// +// M3-B4: the read now goes through the engine facade +// (`server/engine/account-reads.js`) instead of naming +// `lib/mcode-rpc.js` directly, so the endpoint is gated on the same +// declared `authCredentials.getAccountStatus` the usage popover +// (#15 / #16) is gated on — the two read the SAME engine projection +// through the SAME `mcode/account/status` method, and a provider that +// drops it must take both down together. +// +// Nothing about the wire changed. The facade builds the response body +// (success spreads the engine's projection verbatim; failure keeps the +// `{ok:false, reason}` soft-fail shape), and the HTTP status stays 200 +// in both cases: the REQUEST succeeded, and the card renders its empty +// state from `ok:false`. -import { getAccountStatus } from "../lib/mcode-rpc.js"; +import { readEngineAccount } from "../engine/account-reads.js"; /** * Fetched on demand rather than pushed in the state snapshot. @@ -12,14 +26,13 @@ import { getAccountStatus } from "../lib/mcode-rpc.js"; * no credential (see acp/extensions.ts), and nothing here logs the response. * * A failure is a soft one, like /api/session-tree: the card renders its empty - * state rather than the route inventing a name or a plan. + * state rather than the route inventing a name or a plan. The capability gate is + * a different question from that one — "may this provider report an account at + * all" versus "could we read the account this time" — and only the first one + * produces a 501, through `app.js#invokeHandler`. */ export async function handleGetAccount(_req, res, ctx) { - const cs = ctx && ctx.cs; - const r = await getAccountStatus(cs && cs.mcodeSessionId); + const { payload } = await readEngineAccount({ cs: ctx && ctx.cs }); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - if (!r.ok) { - return res.end(JSON.stringify({ ok: false, reason: r.code || "account_unavailable" })); - } - return res.end(JSON.stringify({ ok: true, ...(r.data || {}) })); + return res.end(JSON.stringify(payload)); } diff --git a/packages/webui/server/routes/model.js b/packages/webui/server/routes/model.js index 6117e67a..ef1526e0 100644 --- a/packages/webui/server/routes/model.js +++ b/packages/webui/server/routes/model.js @@ -1,5 +1,18 @@ // webui/server/routes/model.js // GET /api/models, POST /api/set-model, POST /api/permissions, POST /api/answer (legacy) +// +// M3-B4: `GET /api/models` now reads the catalogue through the engine +// facade (`server/engine/model-reads.js`) instead of assembling it +// here. The three sources (the engine session's `model` config option, +// the merged providers config with the engine's `custom_provider` +// tree as its bottom layer, the builtin cli-bundle extraction), the +// two builtin-tree annotations (variant-style thinking levels and +// context-window options) and the three derived "what is active" figures +// all moved with it, as named pure functions pinned on their inputs. +// +// The response is byte-identical. This batch only moves the READ: the +// WRITE half (`handleSetModel`) stays here for B7/B9, together with the +// two other handlers below. import { readFileSync } from "node:fs"; import { join } from "node:path"; @@ -11,70 +24,26 @@ import { webuiPermissionToMcode, PERMISSION_MODES, } from "../lib/mcode-rpc.js"; -import { getBuiltinModelsFromMcode } from "../lib/models.js"; -import { loadProvidersConfig } from "../lib/providers-config.js"; -import { - readEngineCatalogue, - readEngineBuiltinThinking, - readEngineBuiltinContextWindows, - parseEngineModelWireValue, - variantChannelFor, - resolveModelId, - mergeEngineAndWebuiProviders, -} from "../lib/engine-catalogue.js"; +import { readEngineModelCatalogue } from "../engine/model-reads.js"; +import { variantChannelFor, resolveModelId } from "../lib/engine-catalogue.js"; import { webuiModeToLabel } from "../lib/interaction/permission-presets.js"; import { readJson } from "../lib/read-json.js"; -/** The engine's `select` config option with this id, or null before a session exists. */ -function configOption(cs, id) { - const options = Array.isArray(cs && cs.configOptions) ? cs.configOptions : []; - return options.find((o) => o && o.id === id) || null; -} - /** - * Attach the engine's context-window metadata (U6) onto a builtin - * `minimax_api` catalogue entry, mutating `entry`. - * - * `contextWindowOptions` / `contextWindowOptionHints` come from the - * engine's materialised builtin tree (same read as the thinking - * projection — see `lib/engine-catalogue.js`). Only the minimax_api - * builtin entries carry them today: the engine's ACP `model` config - * option (the engine-session entries' source) does not advertise the - * metadata, so those entries are annotated through the same builtin - * projection keyed by the wire form's model id. Custom-provider / - * config-layer entries never get the fields — a model without options - * must stay field-free so the composer mounts no control. - * - * `contextLimit` (the CURRENT effective window, from the engine tree's - * `limit.context`) is attached when the entry has none yet — a config - * layer entry keeps its own value; builtin shell entries get the - * engine's current window so the picker can show the active radio - * before the user's first in-webui pick. - */ -function attachContextWindowOptions(entry, projection) { - if (!projection) return; - entry.contextWindowOptions = [...projection.options]; - if (projection.hints) { - entry.contextWindowOptionHints = { ...projection.hints }; - } - if (entry.contextLimit === undefined && projection.currentLimit !== undefined) { - entry.contextLimit = projection.currentLimit; - } -} - -/** - * Read the optional providers-config file. + * Read the optional providers-config file — the v1 single-file reader. * * Path precedence: `MCODE_WEBUI_MODELS_CONFIG` env → `/models.json`. * Shape: `{ providers: [{ id, label, models: [{ id, label?, contextLimit? }] }] }`. - * Re-read on every request: editing the file does not require a server restart. * Missing / unreadable / malformed → null (treated as "no config"). * - * v2 layered resolution lives in `loadProvidersConfig()` (env > cwd > - * user-level with deep merge). The /api/models route now reads - * through that helper, so an env override of `MCODE_WEBUI_MODELS_CONFIG` - * continues to win over the cwd file (matching the v1 contract), and - * a `~/.mcode-webui/providers.json` layer is layered under both. + * KNOWN DEBT, kept deliberately: nothing calls this any more. The v2 + * layered resolution in `loadProvidersConfig()` (env > cwd > user-level + * with deep merge) replaced it when #57 moved into + * `engine/model-reads.js`, and the function was already unreferenced + * before that move. It is retained rather than deleted because it is + * the written record of the v1 contract `loadProvidersConfig`'s own + * header cites; delete it in a batch whose subject is dead code, not as + * a side effect of moving a read. */ function readModelsConfig() { const path = @@ -89,94 +58,21 @@ function readModelsConfig() { } } -/** - * Layered resolver used by /api/models. Returns the merged - * `{ providers }` (v2 shape) or `null` when every layer is missing. - * - * Ticket 06: the engine's `custom_provider` tree is the new bottom - * layer; the webui layers (env > cwd > user, already merged inside - * `loadProvidersConfig`) win on id collision. The merge itself - * lives in `mergeEngineAndWebuiProviders()` — see its file header - * for the precedence rules. The helper here just shapes its - * return into the legacy `{ providers: [...] }` view that - * handleGetModels already understood. - */ -function readProvidersConfigForModels() { - try { - const cfg = loadProvidersConfig(); - const webuiProviders = (cfg && Array.isArray(cfg.providers)) ? cfg.providers : []; - // Engine catalogue read is best-effort: a missing `config.yaml` - // or a YAML parse error yields []. The merge below treats an - // empty engine catalogue as "no engine layer" and returns the - // webui layers verbatim — matching the pre-ticket-06 behaviour - // for installs without an engine config. - const engineProviders = readEngineCatalogue(); - const merged = mergeEngineAndWebuiProviders(engineProviders, webuiProviders); - if (merged.length === 0) return null; - return { providers: merged }; - } catch { - return null; - } -} - -/** - * Coerce a provider prefix out of a model id. - * - * `minimax_api/MiniMax-M3` → `minimax_api`. Bare `MiniMax-M3` falls back to - * `minimax_api` (the engine's only shipping builtin provider) so a user-typed - * short id still resolves to a known group instead of orphaning itself. - * - * Used only for engine session entries (their ids are the engine's wire - * form `m:::u`); webui-side entries now carry the - * provider as an explicit `entry.provider = p.id` field, and the multi-segment - * model id stays whole (see `webuiFullModelId`). - */ -function providerOf(modelId, fallback = "minimax_api") { - if (!modelId) return fallback; - const i = modelId.indexOf("/"); - if (i <= 0) return fallback; - return modelId.slice(0, i); -} - -/** - * Build the webui internal id for a catalogue entry: `/`. - * - * The webui id is always two segments where the first is the provider key - * and the second is the engine-side model id verbatim (the engine allows - * `/` inside model ids — see engine-catalogue.js; the wire form - * `formatModelKey(, ) = /` uses - * `/` as the only structural separator, so a downstream `/` - * webui form survives the round-trip through `resolveModelId`). - * - * Ticket 09-02 (grouping attribution): the previous implementation - * skipped the prefix when `m.id.includes("/")` and let the bare upstream - * id stand. That pushed the picker into the wrong group (the id's first - * segment was used as a fallback for `providerOf`) and let two providers - * with overlapping upstream ids collide on the `seen` dedupe (e.g. - * `z-ai/glm-5.3` in `nousresearch` ate the sibling `zai-max/glm-5.3`). - * Always prefixing — even when the model id already contains `/` — - * keys every entry by `(providerKey, modelId)` and the dedupe is per - * provider, as the ticket requires. - */ -function webuiFullModelId(providerKey, modelId) { - return `${providerKey}/${modelId}`; -} - /** * Translate a webui-recorded model id to the engine's wire form. * * The webui records `cs.model.name` in `/` - * form (see `webuiFullModelId`). The engine's `set_config_option` for - * `configId: "model"` rejects anything that isn't the wire form - * `m:::u` (see + * form (see `engine/model-reads.js#webuiFullModelId`). The engine's + * `set_config_option` for `configId: "model"` rejects anything that + * isn't the wire form `m:::u` (see * packages/tui/src/acp/control-state.ts#modelConfigValue / agent.ts * `parseModelConfigValue`). Without this translation a mid-session * pick of a multi-segment model id (`nousresearch/deepseek/x`) would * 400 from the engine. * - * `resolveModelId` (in `lib/mcode-acp.js`) owns the resolver — it is - * the same code path `applyRecordedModel` uses on session boot, so the - * mid-session push and the boot-time replay share one source of + * `resolveModelId` (in `lib/engine-catalogue.js`) owns the resolver — + * it is the same code path `applyRecordedModel` uses on session boot, so + * the mid-session push and the boot-time replay share one source of * truth. Returns `null` when the engine has no matching option yet * (the engine configOptions list is empty before the first session * event lands); the caller falls back to the recorded id and the @@ -195,321 +91,41 @@ function translateWebuiModelIdToEngineValue(cs, modelId, resolveOpts) { } /** - * GET /api/models — catalogue, with priority-aware merging. + * GET /api/models — the composer model picker, through the engine + * facade. + * + * The endpoint's whole contract is the payload the facade built: * - * Priority order (highest wins for `current`, first wins for each id): - * 1. Engine session's `model` config option. Its `options[].value` is - * the engine's encoded id (e.g. `m:::v:`), - * so it round-trips straight through `POST /api/set-model`. Used - * when a session is active. - * 2. Optional `MCODE_WEBUI_MODELS_CONFIG` / `models.json` providers - * config. Per-provider groups with labels and `contextLimit`s. - * Ticket 06: this layer is the webui-side merge of - * `env > cwd > user-level`, with the engine's - * `custom_provider` tree as a new bottom layer — see - * `lib/engine-catalogue.js` for the merge rules. - * 3. `getBuiltinModelsFromMcode()` — extracted from mcode's own - * cli.js bundle, so the list tracks mcode's TUI without a webui - * release. + * - `models` — the flat list, every entry carrying `id` / `label` / + * `provider` / `source` plus whatever that source contributes + * (`contextLimit`, `protocol`, `thinkingLevels`, `modalities`, + * `contextWindowOptions`). + * - `groups` — the same entries grouped by provider, so the picker can + * render sections instead of a flat list. This is red line five's + * "模型按供应商分组": the group id is the provider key, and the + * builtin shell is always `minimax_api` regardless of the recorded + * pick. + * - `current` / `currentThinking` / `currentContextWindow` — the three + * derived figures, resolved engine-value-first and never invented + * from a default. + * - `source` — which layer won. + * - `reason: "no_catalogue"` — the soft marker, spread last and only + * when the catalogue came out empty. * - * `current` resolution: - * - With an active session config option: `option.currentValue`. - * - Without one: the recorded pre-session choice (`cs.model.name`), - * which `handleSetModel` already writes — so the selector shows - * the user's pick even before the engine attaches. + * `engine/model-reads.js` owns the projection rules and their + * derivations; this route writes the body. The facade re-reads every + * source on every request, so editing `models.json`, + * `~/.mcode-webui/providers.json` or the engine's `config.yaml` still + * takes effect without a restart. * - * Response carries `groups` so the UI can render provider sections, - * alongside the flat `models` array for callers that do not care - * about grouping. + * Still a SYNCHRONOUS handler, exactly as before: the facade's read is + * synchronous too, because every source it needs was already a static + * import of this route (see the boot-path note in the engine module). */ export function handleGetModels(_req, res, ctx) { - const cs = ctx.cs; - const option = configOption(cs, "model"); - const engineOption = option; // keep the alias so reviewers can read priority order - - const list = []; - const groups = []; - const seen = new Set(); - // Ticket 36 — the engine's materialised builtin tree (provider. - // minimax.models) carries the variant-style thinking schema that - // /api/models never projected: switchable models became a two-state - // ["off","on"] toggle, forced_on+effortOptions models expose the - // engine's depth list verbatim, everything else stays metadata-free. - // One read serves both annotation sites below (engine-session - // entries and the builtin shell). - const builtinThinking = readEngineBuiltinThinking(); - // U6 — same tree, context-window projection. One read serves both - // annotation sites below (engine-session entries and the builtin - // shell), exactly like `builtinThinking`. - const builtinContextWindows = readEngineBuiltinContextWindows(); - - // 1) Engine session config option — authoritative when present. We keep - // its encoded ids verbatim so /api/set-model round-trips. Both `name` - // and `label` are set on engine-sourced entries because pre-existing - // callers (the composer chip) read `name`, while the new - // provider-grouped panel reads `label`. - if (engineOption) { - const engineGroupId = "__engine"; - const engineGroup = { - id: engineGroupId, - label: "Engine session", - models: [], - }; - for (const o of Array.isArray(engineOption.options) ? engineOption.options : []) { - const id = o && typeof o.value === "string" ? o.value : null; - if (!id) continue; - if (seen.has(id)) continue; - seen.add(id); - const displayName = (o && o.name) || id; - const entry = { - id, - name: displayName, - label: displayName, - provider: providerOf(id), - source: "engine", - }; - // Ticket 36: applyConfigOptionUpdate mirrors the engine's - // wire-form currentValue into cs.model.name outside the pick - // window, and the composer matches the active model by id — - // annotate the wire-form entries too so the thinking control - // survives a cross-client change. - const wire = parseEngineModelWireValue(id); - if (wire && wire.providerId === "minimax_api") { - const proj = builtinThinking.get(wire.modelId); - if (proj) entry.thinkingLevels = [...proj.levels]; - // U6: annotate the wire-form entries with the engine's - // context-window options too, so the picker's detail area - // survives a cross-client model change (same reasoning as the - // thinkingLevels annotation above). - attachContextWindowOptions(entry, builtinContextWindows.get(wire.modelId)); - } - engineGroup.models.push(entry); - list.push(entry); - } - if (engineGroup.models.length > 0) groups.push(engineGroup); - } - - // 2) Providers config — read every request so editing the file does not - // require a restart. Config wins on id collision with the builtin - // catalogue so providers can override labels and contextLimit. - // - // v2 layered resolution (env > cwd > user-level) is provided by - // `loadProvidersConfig()`; the v1 single-file reader stays as a - // fallback for callers that pass the legacy `models.json` - // through a different code path (none today, but keeping it - // documents the contract). - // - // Ticket 06: the engine's `custom_provider` tree is also a - // catalogue source — readEngineCatalogue() projects it to the - // v2 shape (no key material) and mergeEngineAndWebuiProviders() - // unions it with the webui layers (webui wins on collision). - const config = readProvidersConfigForModels(); - if (config) { - for (const p of config.providers) { - if (!p || typeof p.id !== "string" || !p.id) continue; - const models = []; - for (const m of Array.isArray(p.models) ? p.models : []) { - if (!m || typeof m.id !== "string" || !m.id) continue; - // Ticket 09-02: always prefix the webui id with ``. The - // upstream-style model id (`deepseek/x`, `z-ai/glm-5.3`, - // `openai/gpt-5.6-sol`) is kept verbatim inside the model id - // portion — the engine allows `/` inside model keys, the wire - // form `/` uses `/` only as the structural - // separator, and the `seen` dedupe is per provider (so two - // sibling providers with overlapping upstream ids stay - // distinct instead of one swallowing the other). - const fullId = webuiFullModelId(p.id, m.id); - if (seen.has(fullId)) continue; - seen.add(fullId); - const entry = { - id: fullId, - label: typeof m.label === "string" && m.label ? m.label : m.id, - provider: p.id, - source: "config", - }; - if (typeof m.contextLimit === "number" && m.contextLimit > 0) { - entry.contextLimit = m.contextLimit; - } - // v2 schema surfaces: each model carries protocol + - // thinkingLevels + modalities so the selector can pick the - // right controls without a second round-trip. `auth` only - // exposes hasKey + type — apiKey NEVER reaches this response. - if (typeof p.protocol === "string" && p.protocol) { - entry.protocol = p.protocol; - } - if (Array.isArray(m.thinkingLevels) && m.thinkingLevels.length > 0) { - entry.thinkingLevels = [...m.thinkingLevels]; - } - if (Array.isArray(m.modalities) && m.modalities.length > 0) { - entry.modalities = [...m.modalities]; - } - models.push(entry); - list.push(entry); - } - // Auth shape: only `hasKey` and `type`; no apiKey/baseURL. - // Operators see "configured or not" without leaking the secret. - // Ticket 06: the merged layer (engine + webui) may carry - // `hasKey` either via `p.auth.apiKey` (webui-side plaintext — - // masked elsewhere) or via `p.auth.hasKey` (engine-side - // boolean, set by `lib/engine-catalogue.js`). Either signal - // means the provider is configurable from the picker. - const groupHasKey = !!( - (p.auth && p.auth.apiKey) || - (p.auth && p.auth.hasKey) - ); - groups.push({ - id: p.id, - label: typeof p.label === "string" && p.label ? p.label : p.id, - auth: { - hasKey: groupHasKey, - type: p.auth && typeof p.auth.type === "string" ? p.auth.type : "byok", - }, - protocol: typeof p.protocol === "string" ? p.protocol : "openai", - models, - }); - } - } - - // 3) Builtin catalogue (extracted from mcode's cli.js bundle). The - // builtins all belong to the engine's `minimax_api` provider - // (see `lib/models.js#getBuiltinModelsFromMcode` — the cli.js - // extraction regex targets `MiniMax-M*`). The builtin shell is - // keyed by `minimax_api` regardless of the recorded pick, so a - // pick of `nousresearch/openai/gpt-5.6-sol` doesn't drag the - // MiniMax builtins into the `nousresearch` group. The previous - // behaviour derived the builtin group's id from - // `currentName.split("/")[0]`, which landed the builtins under - // whichever provider the user happened to have picked (the - // ticket 09-02 acceptance replay caught this as "8 config + 6 - // misplaced MiniMax builtins = 14 in `nousresearch`"). - const builtins = getBuiltinModelsFromMcode(); - const BUILTIN_PROVIDER = "minimax_api"; - // The recorded pre-session pick — used below for `current`, NOT for - // builtin-group attribution (the builtin shell is keyed by - // BUILTIN_PROVIDER above). - const currentName = - (cs.model && typeof cs.model.name === "string" && cs.model.name) || ""; - let builtinGroup = groups.find((g) => g.id === BUILTIN_PROVIDER); - if (!builtinGroup) { - builtinGroup = { id: BUILTIN_PROVIDER, label: BUILTIN_PROVIDER, models: [] }; - groups.push(builtinGroup); - } - for (const m of builtins) { - const fullId = `${BUILTIN_PROVIDER}/${m}`; - if (seen.has(fullId)) continue; - seen.add(fullId); - const entry = { - id: fullId, - label: m, - provider: BUILTIN_PROVIDER, - source: "builtin", - }; - // Ticket 36: attach the engine's thinking metadata for this - // builtin. `thinkingLevels` is exactly what the engine's tree - // supports — ["off","on"] for a switchable variant toggle, the - // engine's effort list when the model has one, and ABSENT for a - // forced_on model with nothing user-settable (the composer then - // mounts no control, by design). A config-layer entry with the - // same id has already taken the slot (seen dedupe) — the - // operator's config wins wholesale, unchanged rule. - const proj = builtinThinking.get(m); - if (proj) entry.thinkingLevels = [...proj.levels]; - // U6: the engine's context-window options for this builtin, plus - // its current effective window as the `contextLimit` fallback. - attachContextWindowOptions(entry, builtinContextWindows.get(m)); - list.push(entry); - builtinGroup.models.push(entry); - } - - // Drop the empty builtin shell — a no-bundle empty group is noise. - // The drop is gated on "no providers config" so a fresh install with - // a config that names no models still has somewhere to attach the - // builtins once mcode reports them. - if (builtinGroup && builtinGroup.models.length === 0 && !config) { - const idx = groups.indexOf(builtinGroup); - if (idx >= 0) groups.splice(idx, 1); - } - - // `current` is the engine's value when one exists; otherwise the - // recorded pre-session choice (`cs.model.name`, written by - // `handleSetModel`). When neither exists we report `null` rather than - // falling back to `DEFAULT_MODEL` — the old behaviour invented an - // active model the engine never confirmed, and the chip ended up - // claiming a model the session was not actually running. The chip - // renders a neutral label when `current` is `null` (see composer.tsx - // currentModelLabel). - const current = - (option && option.currentValue) || - currentName || - null; - - // Current thinking-effort level: read the engine's `thinkingEffort` - // option when present; otherwise fall back to `cs.model.thinking`, - // which `handleSetModel` writes (pre-session record) and which the - // engine's `config_option_update` notification refreshes via - // `applyConfigOptionUpdate` (see lib/mcode-acp.js). The selector - // reads this to highlight the active level and to skip the picker - // when the active model has no `thinkingLevels`. - const thinkingEffortOption = - Array.isArray(cs && cs.configOptions) ? cs.configOptions.find((o) => o && o.id === "thinkingEffort") : null; - const currentThinking = - (thinkingEffortOption && typeof thinkingEffortOption.currentValue === "string" - ? thinkingEffortOption.currentValue - : null) || - (cs && cs.model && typeof cs.model.thinking === "string" && cs.model.thinking) || - null; - - // U6 — the recorded context-window choice (`handleSetModel` writes - // `cs.model.contextWindow`). There is no engine config option behind - // it (the engine's ACP surface has no context channel — see the - // handleSetModel header), so unlike `currentThinking` there is no - // engine-value branch: the recorded pick is the only source. A - // recorded value the current model no longer advertises is still - // reported verbatim — the stale-pick display rule lives in the - // composer (same split as the thinking level's stale-suffix guard). - const recordedContextWindow = - cs && cs.model && Number.isSafeInteger(cs.model.contextWindow) && cs.model.contextWindow > 0 - ? cs.model.contextWindow - : null; - // Fallback: the current model's catalogue `contextLimit` (the - // engine's current effective window), so the picker can highlight - // the active radio before the user's first in-webui pick. - const currentModelEntry = current ? list.find((m) => m.id === current) : null; - const currentContextWindow = - recordedContextWindow ?? - (currentModelEntry && - Number.isSafeInteger(currentModelEntry.contextLimit) && - currentModelEntry.contextLimit > 0 - ? currentModelEntry.contextLimit - : null); - - const source = - option && Array.isArray(option.options) && option.options.length > 0 - ? "acp-session-config" - : config - ? "config+mcode-cli-bundle" - : "mcode-cli-bundle"; - + const { payload } = readEngineModelCatalogue({ cs: ctx && ctx.cs }); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - return res.end( - JSON.stringify({ - ok: true, - models: list, - groups, - current, - currentThinking, - currentContextWindow, - source, - // Backwards-compat: surface the same soft-failure marker the older - // engine-only build did when nothing could be sourced. With the - // merge it should be rare (builtin catalogue + providers config - // cover most installs), but a missing mcode bundle AND an absent - // config leaves the catalogue empty — and a caller that wants to - // know "is this a hard failure or just no engine attached?" still - // gets the same hint. - ...(list.length === 0 ? { reason: "no_catalogue" } : {}), - }), - ); + return res.end(JSON.stringify(payload)); } // POST /api/set-model — only updates cs.model; with a session the same value diff --git a/packages/webui/server/routes/protocol.js b/packages/webui/server/routes/protocol.js index 763a1036..10d49047 100644 --- a/packages/webui/server/routes/protocol.js +++ b/packages/webui/server/routes/protocol.js @@ -20,12 +20,19 @@ import { activateSession, mcodePermissionToWebui, } from "../lib/mcode-rpc.js"; -// M3-B1 (engine facade): only #72 (`list-sessions`) is gated in this +// M3-B1 (engine facade): only #72 (`list-sessions`) is gated in that // batch. The other five handlers here still call mcode-rpc directly — -// they belong to B4 (#73 capabilities) and B7/B9 (cancel, load, activate, -// set-mode, set-config-option), each of which lands its own facade call -// with its own regression evidence. +// they belong to B7/B9 (cancel, load, activate, set-mode, +// set-config-option), each of which lands its own facade call with its +// own regression evidence. import { readEngineSessionList } from "../engine/session-reads.js"; +// M3-B4 (engine facade): #73 (`capabilities`) now reads the engine's +// declared capability surface through the facade instead of reaching +// into `lib/mcode-rpc.js` and `lib/acp-client.js` from inside the +// handler. See `engine/capability-reads.js` for why the response gains +// the `engine` view rather than replacing the ACP wire table, and why +// this endpoint declares no capability of its own. +import { readEngineCapabilityView } from "../engine/capability-reads.js"; import { loadSessions, saveSessions, resetContext } from "../lib/sessions.js"; import { pushStateFor } from "../lib/state-bus.js"; import { readJson } from "../lib/read-json.js"; @@ -261,19 +268,44 @@ export async function handleListSessions(req, res, ctx) { // 列出 mcode acp 实际支持的能力 — 供前端 capability detection, // 决定按钮是否 disable / 降级路径 // mcode version 动态从 acp client initialize 响应读 (不再 hardcode) +// +// M3-B4: the handler no longer names `lib/mcode-rpc.js` or +// `lib/acp-client.js` — both moved behind +// `engine/capability-reads.js#readEngineCapabilityView`, which also +// resolves the provider whose DECLARED surface this endpoint now serves. +// +// `capabilities` IS the 14-key engine-capabilities view: a replacement +// for the `MCODE_ACP_CAPABILITIES` ACP wire table this field used to +// carry, approved as an endpoint contract change. The four +// `capabilities*` keys form one group — the declaration, which provider +// answered, how it was chosen, and the derived degradation roll-up — and +// the declaration appears exactly once. +// +// `capabilitiesProviderFor` says whether the declaration came from the +// active transport's provider or from the default provider standing in +// for a transport no provider claims yet (M4), so a consumer never +// mistakes a standing-in declaration for the connected engine's. +// +// `notes` stays here: it is prose about webui's own routes, not an +// engine read, and the facade has no business restating it. // ============================================================ export async function handleCapabilities(_req, res) { - const { MCODE_ACP_CAPABILITIES } = await import("../lib/mcode-rpc.js"); - const { getMcodeServerInfo } = await import("../lib/acp-client.js"); - // initialize answers with `agentInfo: { name, title, version }` (not `serverInfo`). - const agentInfo = getMcodeServerInfo(); - const mcodeVersion = (agentInfo && agentInfo.version) || "unknown"; + const { + declaration, + unavailable, + provider, + providerFor, + agent, + } = await readEngineCapabilityView(); return respond(res, 200, { ok: true, - mcodeVersion, - mcodeName: (agentInfo && agentInfo.name) || null, - mcodeTitle: (agentInfo && agentInfo.title) || null, - capabilities: MCODE_ACP_CAPABILITIES, + mcodeVersion: agent.version, + mcodeName: agent.name, + mcodeTitle: agent.title, + capabilities: declaration, + capabilitiesProvider: provider, + capabilitiesProviderFor: providerFor, + capabilitiesUnavailable: unavailable, notes: { set_mode: "Takes a modeId from the session's availableModes.", set_config_option: diff --git a/packages/webui/server/routes/sessions.js b/packages/webui/server/routes/sessions.js index b986f7e0..3da48c3a 100644 --- a/packages/webui/server/routes/sessions.js +++ b/packages/webui/server/routes/sessions.js @@ -4,41 +4,34 @@ // GET /api/acp-sessions, GET /api/acp-session-title, // GET /api/sessions/search (Lease C05 — cross-workspace fuzzy match) // (v0.5.bx-33: 删 POST /api/sessions/cleanup-orphans — Wzdhehe 不要这个 UI,API 一起删) +// +// What is left in this file after M3 is the HTTP surface of the session +// endpoints: parse the request, pick the status code, run the fail-closed +// audit, push the SSE frame, answer. Every endpoint that crosses the +// engine seam now asks `engine/` instead of this file's own imports — +// #9 #10 #72 #74 #75 (B1), #8 #11 (B2), #15 #16 #17 #19 (B3), +// #20 #57 #73 (B4), #7 #4 #6 (B5), #3 (B6) — and the imports that +// remain below are the ones that are genuinely webui-local: the session +// store, the workspace gate and the audit sink. import { randomUUID } from "node:crypto"; import { loadSessions, saveSessions, resetContext, - ensureOverlayForMcodeSid, - findOverlayForMcodeSid, } from "../lib/sessions.js"; -import { deleteMcodeSessionFromDb } from "../lib/mcode-session-delete.js"; -import { - getMcodeSessionTitle, - getMcodeSessionsCacheSync, - getMcodeSessionsStaleSync, - shutdownMcodeAcpSingleton, - dropMcodeSessionFromCache, -} from "../lib/acp-client.js"; -// Switch-path transcript backfill — load mcode session history from -// the runtime DB so switching to an mvs_ session with no webui wrapper -// shows real chat instead of "No messages yet". -import { loadTranscriptChatLines } from "../lib/transcript.js"; -import { applyMavisUsageToCs } from "../lib/mavis-usage.js"; -import { getMcodeModelLimit } from "../lib/models.js"; -import { - pushStateFor, - clients, - runChatViewChat, -} from "../lib/state-bus.js"; -import { MCODE_RUNTIME_DB, DEFAULT_WORKSPACE } from "../lib/config.js"; -import { invalidateSessionTree } from "../lib/session-tree.js"; +// `pushStateFor` stays a direct import: it is a pure SSE write with no +// I/O and no engine surface, and three of this module's handlers call it +// on their way out. `clients` and `runChatViewChat` left this file in +// M3-B5 and M3-B6 respectively — the delete fan-out enumerates clients +// inside the facade, and the run-mirror projection belongs with the +// switch that produces it. +import { pushStateFor } from "../lib/state-bus.js"; // M3-B1 (engine facade): #9 and #10 read the engine through the declared // capability rather than straight off the ACP client. Both facade -// functions forward to the same acp-client exports this module already -// imported, so the wire shape, the cache and the transport switch are -// unchanged — only the gate in front of them is new. +// functions forward to the same acp-client exports this module used to +// import directly, so the wire shape, the cache and the transport switch +// are unchanged — only the gate in front of them is new. import { readEngineSessionListForWorkspace, readEngineSessionTitle, @@ -46,11 +39,59 @@ import { // M3-B2 (engine facade): #8 asks the facade, which checks the provider's // declaration (sessionCrud.listSessions → 501 when absent) and then // forwards to the same `getSessionTree` this module used to call -// directly. `invalidateSessionTree` stays a direct import: it is a -// synchronous cache drop with no I/O, it is called from the rename and -// delete paths, and routing a one-line invalidation through an async -// facade would make those paths wait on a module load to do nothing. +// directly. `invalidateSessionTree` was a direct import here from B2 +// through B4 on the grounds that it is a synchronous cache drop with no +// I/O and routing a one-line invalidation through an async facade would +// make the caller wait on a module load to do nothing. M3-B5 retired +// that exception: the only three call sites were the rename and delete +// paths, and those moved into `engine/session-writes.js` as part of the +// ordered write sequences they belong to. A cache drop is not a +// standalone concern here — it is step two of a three-step resurrection +// guard, and keeping it addressable from the route was what made it +// possible to call it out of order. import { readEngineSessionTree } from "../engine/session-tree-reads.js"; +// M3-B5 (engine facade): #7 delete, #4 rename and #6 cleanup-orphans are +// the three WRITES of this module, and they ask the engine facade rather +// than driving the store, the caches and the engine's own `local_runtime_*` +// tables from the route. The split is deliberate and is the reason the +// handlers below shrank rather than grew: +// +// - The gate in front of each write is the facade's, not this file's. +// #7 and #6 gate hard on `sessionCrud` · `deleteSession` (the rows +// they destroy are the engine's own); #4 declares no capability at +// all, because a rename writes webui's store and nothing else. +// - The load→resolve→authorize→intent-audit→mutate ORDER is still +// this file's, and had to stay: the write-ahead audit has to land +// between "know what the user asked to delete" and "delete it". So +// the facade exposes a plan/commit pair rather than one +// `deleteSession(options)` that would have swallowed the ordering. +// - The response BODIES are built in the facade, once. #6's dryRun +// shape is a byte-for-byte red line for this batch, so it is pinned +// there by test instead of re-assembled in two places here. +// - `deleteMcodeSessionFromDb` and the 32-table SQL stay in +// `lib/mcode-session-delete.js` and are reached by the facade through +// a dynamic import; see KNOWN DEBT in `engine/session-writes.js`. +import { + applyEngineSessionRename, + commitEngineOrphanSessionDelete, + commitEngineSessionDelete, + isMcodeSessionId, + planEngineSessionDelete, + previewEngineSessionDelete, + readOrphanSessionWriteIds, +} from "../engine/session-writes.js"; +// M3-B6 (engine facade): #3 switch. This is the endpoint that emptied the +// most imports out of this file — the walked-session title cache +// (`lib/acp-client.js`), the transcript read (`lib/transcript.js`), the +// usage sync (`lib/mavis-usage.js` + `lib/models.js`), the switch +// workspace gate (`lib/workspace.js#assertWorkspacePath`, still imported +// for handleNewSession) and `DEFAULT_WORKSPACE` / `MCODE_RUNTIME_DB` +// (`lib/config.js`, now referenced by no route in this file at all) +// all live behind `applyEngineSessionSwitch` now. See that module's +// header for the four load-bearing facts it took over, and KNOWN DEBT 1 +// for why the 3-candidate transcript probe it forwards to survives this +// batch while the route's direct reach for it does not. +import { applyEngineSessionSwitch } from "../engine/session-switch.js"; // The capability-error predicate `handleSessionTree` uses to tell the gate's // 501 apart from a soft-fail. Taken from the facade entry, which re-exports // the same binding `app.js#invokeHandler` matches on, so the two ends of this @@ -68,102 +109,6 @@ import { append as _eventsAppend } from "../lib/events.js"; // workspace write lands on the same boundary. import { assertWorkspacePath } from "../lib/workspace.js"; -// _resolveSwitchWorkspace — pick the workspace the switched-into session -// "belongs to" and run it through the same containment gate that the -// workspace picker / handleNewSession / browseWorkspace all funnel through. -// -// Source priority (s39 — webui-parity ticket 39: file tree must follow the -// switched session): -// -// 1. The target session's stored `workspace` field — that IS the -// workspace the user was in when they last had it open, modulo any -// pollution the old code introduced. Real existence + containment -// are checked; an out-of-bounds or stale value surfaces as a 400 -// so the user can either widen the allowed roots or pick a fresh -// workspace, instead of silently landing on the previous project. -// -// 2. DEFAULT_WORKSPACE (env MCODE_WORKSPACE > mcode TUI cwd.json > homedir) -// when the stored value is empty. Empty is also the value seen for -// (a) records created by the old code that polled freshly-typed mvs -// sessions with the current cs.workspace (the data-corruption bug -// this ticket fixes), and (b) older sessions that pre-date the -// workspace field. DEFAULT_WORKSPACE is already in the default -// allowed-roots surface (see getAllowedWorkspaceRoots), so the -// containment check accepts it without env setup. -// -// Critical invariants: -// - The switch NEVER keeps cs.workspace on the prior project. The -// user-reported symptom was exactly that: "the file tree still -// shows the previous project's files". Falling back to current ws -// when target.workspace is empty is the bug we are removing. -// - The switch NEVER writes cs.workspace.dir to a path the -// containment gate rejected. A 400 with the gate's actionable -// error is the only acceptable outcome. -// - The switch NEVER overwrites a target session's stored workspace -// with the current cs.workspace. That was the ② pollution path — -// re-introducing it would re-break the regression we just fixed. -// New overlay records (mvs_ first-touch) get workspace:"" here; the -// target-first read picks DEFAULT_WORKSPACE for them. -function _resolveSwitchWorkspace(target, currentWs) { - const raw = target && typeof target.workspace === "string" ? target.workspace.trim() : ""; - // Empty / non-string / null → DEFAULT_WORKSPACE. Never the current cs - // workspace — that's the user-reported "stays on the old project" - // failure mode this fix removes. - const candidate = raw || DEFAULT_WORKSPACE; - const gate = assertWorkspacePath(candidate); - if (!gate.ok) { - return { ok: false, error: gate.error, attempted: candidate }; - } - return { ok: true, dir: gate.path, real: gate.real, fallback: !raw }; -} - -/** - * Detect the cumulative-render pollution pattern in a stored chat - * buffer (session-isolation/06). When the engine emits each segment - * of an `agent_message`, streamUpdateLine writes a new `●` line; a - * non-cumulative buffer has each line containing only its own - * segment's text. A cumulative buffer — the bug — has at least one - * later `●` line whose text is a strict superset of an earlier - * `●` line (because the accumulator never reset between segments and - * every later line re-wrote every prior segment's text). This - * predicate is O(n^2) in the number of `●` lines but a single - * session's `chat` is bounded (~400 lines by the transcript cap) so - * the worst case is a few thousand substring checks per switch — - * cheap enough. - * - * Returns true when the buffer is clearly cumulative (an earlier - * `●` line is a strict substring of a later one AND the longer line - * strictly extends the shorter). Conservative on both sides: - * - a single-`●`-line buffer is never cumulative; - * - non-`●` lines (system, tool, ▲ thought) are ignored — only - * `●` rows matter, since the cumulative bug only affects message - * segments; - * - ties (equal-length `●` lines) are NOT cumulative — same - * length, no superset relation. - */ -function chatLooksCumulative(chat) { - if (!Array.isArray(chat) || chat.length === 0) return false; - const dots = []; - for (const line of chat) { - if (typeof line !== "string") continue; - // Match the same prefix the streamer writes: `● ` then text. - // Also accept bare `●` at end-of-line (transcript-sync appends - // stripped-down `●` markers in some paths). - if (line.startsWith("● ")) dots.push(line.slice(2)); - else if (line === "●") continue; - else continue; - } - for (let i = 0; i < dots.length; i += 1) { - for (let j = i + 1; j < dots.length; j += 1) { - const a = dots[i]; - const b = dots[j]; - if (b.length <= a.length) continue; // strict superset ⇒ longer - if (b.includes(a)) return true; - } - } - return false; -} - // _auditFail — shared failure sink for audit writes. events.js#append // THROWS on write failure; a governance action must not complete with // a missing audit trail, so every route-level append is wrapped and @@ -189,58 +134,17 @@ function _auditFail(res, e, what) { return undefined; } -// Prevent "deleted session reappears": the long-lived mcode acp child -// still holds the session in memory and will rewrite the registry row -// on its next request — so we must (1) kill the child, (2) SQL-delete -// the rows, (3) drop ONLY the deleted sid from the in-memory cache (not -// the whole cache — invalidating the whole cache sends an empty -// placeholder to the sidebar which flashes from 42 → 16 → 42 entries, -// looking like the delete failed). -function killMcodeSessionResurrection(mcodeSid) { - try { - shutdownMcodeAcpSingleton(); - } catch {} - dropMcodeSessionFromCache(mcodeSid); -} - - -// Title fast path — resolve an mvs_ session's title from the -// in-memory walked-session cache (the same cache behind -// GET /api/acp-sessions via getMcodeSessionsForWorkspace) BEFORE -// awaiting getMcodeSessionTitle. The fallback boots the ACP child; with -// a missing/broken mcode binary that path measured ~2.17s end-to-end -// AND degraded the title to the "Mcode session" placeholder even -// though the cache already held the real title. Cache getters are sync -// and spawn nothing, so a hit keeps the switch hot path at zero ACP -// cost. -// -// Cross-workspace matching within what the module exposes: the cache -// holds ONE workspace's list, keyed by ws. We probe the client's -// current ws with both the fresh (30s TTL) and stale (same-ws, -// TTL-expired) readers, plus the "" key — getMcodeSessionsForWorkspace("") -// caches the UNFILTERED list, so a cache walked without a workspace -// still answers. A miss returns null and the caller falls back to -// getMcodeSessionTitle. -function _lookupCachedMcodeTitle(mcodeSessionId, ws) { - if (!mcodeSessionId) return null; - const keys = [ws || "", ""]; - for (const wsKey of keys) { - for (const getter of [getMcodeSessionsCacheSync, getMcodeSessionsStaleSync]) { - let sessions = null; - try { - sessions = getter(wsKey); - } catch { - sessions = null; - } - if (!Array.isArray(sessions)) continue; - const hit = sessions.find( - (s) => s && s.sessionId === mcodeSessionId && s.title, - ); - if (hit && hit.title) return hit.title; - } - } - return null; -} +// Prevent "deleted session reappears" — moved to the engine facade in +// M3-B5. The long-lived mcode acp child still holds the session in +// memory and will rewrite the registry row on its next request, so the +// delete has to (1) kill the child, (2) SQL-delete the rows, (3) drop +// ONLY the deleted sid from the in-memory cache (not the whole cache — +// invalidating the whole cache sends an empty placeholder to the sidebar +// which flashes from 42 → 16 → 42 entries, looking like the delete +// failed). That sequence is now +// `engine/session-writes.js`, where it is named and tested step by step +// instead of being a two-line helper a route could call in the wrong +// order. // GET /api/sessions — list // qa (session-workspace-crud): 响应瘦身为 sidebar 元数据 — 与 docs/API.md @@ -333,8 +237,37 @@ export async function handleNewSession(req, res, ctx) { } // POST /api/sessions/switch — switch to session by webui id or mvs_xxx +// +// M3-B6 (engine facade): everything this endpoint does to the engine — +// resolve, first-touch overlay creation, the cache-first title lookup, +// the transcript backfill decision and its read, the workspace +// containment gate, the per-client state mutation and the response body +// — happens in `engine/session-switch.js#applyEngineSessionSwitch`, and +// the response shape is built there once. What stays HERE is what is +// genuinely the route's, and the split is the same one B5 drew for the +// write family: +// +// - HTTP request parsing and the ONE validation body this endpoint +// has. A missing id is a 400 with `{ok:false,error:"id required"}` +// and a bare "application/json" content type, and that body has +// nothing to do with the engine. +// - THE STATUS CODES. The facade returns outcomes (`ok`, +// `not_found`, `workspace_refused`) and never learns what a status +// is; `statusHint` carries the number so the mapping is one table +// here instead of three branches inside the engine layer. +// - THE AUDIT, fail-closed. `_eventsAppend` THROWS on write failure +// and a governance action must not complete with a missing audit +// trail, so the append sits between the facade's work and the +// response, and its failure answers 500 through `_auditFail`. +// - The state push and the two log lines that bracket the response. +// +// The ordering constraint the facade could not own is the reason the +// audit stays put: the switch has ALREADY mutated `cs` by the time this +// append runs (that is pre-existing behaviour — a failed audit leaves +// the client switched and reports 500, which is what the operator sees +// today), and the SSE push must not fire when that append failed. Both +// properties are the route's to keep. export async function handleSwitchSession(req, res, ctx) { - const cs = ctx.cs; const cid = ctx.cid; const payload = await readJson(req); const id = (payload.id || "").trim(); @@ -342,290 +275,31 @@ export async function handleSwitchSession(req, res, ctx) { res.writeHead(400, { "Content-Type": "application/json" }); return res.end(JSON.stringify({ ok: false, error: "id required" })); } - const all = loadSessions(); - console.log( - `[switch] cid=${cid} incoming id=${id.substring(0, 12)}… isMcodeSid=${/^mvs_[a-f0-9]{32}$/.test(id)} allTotal=${all.length}`, - ); - // 优先按 mcode session id 找(v0.5.bv: 1:1 关联) - let target = all.find((s) => s.mcodeSessionId === id); - let matchKind = target ? "mcodeSessionId" : null; - if (!target) { - target = all.find((s) => s.id === id); - if (target) matchKind = "webuiId"; - } - console.log( - `[switch] cid=${cid} match=${matchKind || "NONE"} target.id=${target ? target.id.substring(0, 8) : "null"}… target.mcodeSid=${target && target.mcodeSessionId ? target.mcodeSessionId.substring(0, 12) : "null"}… target.chatLen=${target ? (target.chat ? target.chat.length : 0) : 0} target.title="${target ? (target.title || "").substring(0, 30) : ""}"`, - ); - if (!target) { - const isMcodeSid = /^mvs_[a-f0-9]{32}$/.test(id); - if (isMcodeSid) { - // Cache-first title — the walked session cache usually already - // holds the real title (the sidebar just rendered it). Only a - // total cache miss pays the getMcodeSessionTitle cost, which - // boots the ACP child (~2.17s measured with a broken mcode - // binary) and used to degrade every first switch to the - // "Mcode session" placeholder. - const ws = (cs.workspace && cs.workspace.dir) || ""; - let title = _lookupCachedMcodeTitle(id, ws); - let titleSource = title ? "cache" : "acp"; - if (!title) { - title = (await getMcodeSessionTitle(id)) || "Mcode session"; - } - // Single base session — overlay record id === mcode session id, - // idempotent create. Old model gave each mvs_ switch a fresh - // uuid wrapper → the same conversation had two identities, the - // direct cause of the "extra untitled entry" sidebar confusion. - // Repeated switches now hit the same record. - // - // s39 (webui-parity ticket 39): the workspace argument is GONE. - // The old `workspace: ws` here stamped the freshly-created overlay - // with the CURRENT cs.workspace, so every first-touch of an mvs_ - // session from project A inherited project A's path. Switching - // back to that mvs_ session from project B then either (a) was - // ignored by the read-only switch path, leaving the file tree - // stuck on B, or (b) — under the prior mutation — overwrote the - // overlay's workspace with B's path, polluting every per-project - // grouping. New overlays start with workspace:"" (set inside - // ensureOverlayForMcodeSid when no value is passed); the - // target-first read below then lands on DEFAULT_WORKSPACE for - // first-touch mvs_ switches, with no per-session pollution. - const existed = findOverlayForMcodeSid(all, id); - target = ensureOverlayForMcodeSid(all, id, { title }); - target.updatedAt = Date.now(); - saveSessions(all); - console.log( - `[switch] cid=${cid} ${existed ? "reused" : "created"} overlay ${target.id.substring(0, 12)}… (id=mcode sid) title="${title}" titleSource=${titleSource}`, - ); - } else { - console.log( - `[switch] cid=${cid} 404 id=${id} not found and not mcode sid`, - ); - res.writeHead(404, { "Content-Type": "application/json" }); - return res.end(JSON.stringify({ ok: false, error: "session not found" })); - } - } else if ( - // Placeholder refresh — wrappers created during a broken-title - // window carry "Mcode session" forever. If the walked cache now - // has the real title, repair the stored wrapper. Cache-only (sync, - // no ACP boot): an existing wrapper must never make the hot path - // slower. - target.title === "Mcode session" && - target.mcodeSessionId && - /^mvs_[a-f0-9]{32}$/.test(target.mcodeSessionId) - ) { - const cachedTitle = _lookupCachedMcodeTitle( - target.mcodeSessionId, - (cs.workspace && cs.workspace.dir) || "", - ); - if (cachedTitle) { - target.title = cachedTitle; - target.updatedAt = Date.now(); - saveSessions(all); - console.log( - `[switch] cid=${cid} refreshed placeholder title for ${target.id.substring(0, 8)}… → "${cachedTitle}"`, - ); - } - } - // Transcript backfill — when the resolved target has NO webui chat - // yet but IS a real mvs_ session, load the mcode transcript from - // the runtime DB (read-only) and map it into the webui chat-line - // grammar BEFORE responding, so response session.chat and cs.chat - // carry history. Caps inside (last 400 lines / 200KB) keep the SSE - // state push bounded; a 1000+-message session must not balloon it. - // - // session-isolation/06 (persist hygiene): the original rule only - // backfilled when target.chat was empty, so a polluted buffer - // (the cumulative-render bug from Item 1, before its fix) would - // persist via saveSessions and win forever. The new rule is: - // - if stored chat is empty → backfill (unchanged). - // - if stored chat looks cumulative → prefer DB read and re-persist. - // "cumulative" = at least two `●` lines whose text is a strict - // superset of an earlier `●` line (the engine emits each - // segment's full text per line, so a non-cumulative buffer has - // no such inclusion pair). - // - otherwise → keep stored chat. DB-authoritative: transcript-sync - // overwrites the stored chat from the engine DB on the next tick - // (~4s later), so any stored-only lines a user typed into the - // composer but never sent will be lost. The rule above does not - // promise draft preservation; it promises to NOT clobber a - // clean stored buffer with the DB read on every switch. Draft - // preservation is a separate concern (the composer keeps its - // own draft in its own state, see composer-draft.test.ts). - // FAILURE MUST NOT BREAK SWITCHING: any error logs and continues - // with the original chat — the switch itself always succeeds. - if ( - target.mcodeSessionId && - /^mvs_[a-f0-9]{32}$/.test(target.mcodeSessionId) - ) { - const storedHasChat = Array.isArray(target.chat) && target.chat.length > 0; - const storedCumulative = storedHasChat && chatLooksCumulative(target.chat); - const shouldBackfill = - !storedHasChat || storedCumulative; - if (shouldBackfill) { - try { - const r = loadTranscriptChatLines(target.mcodeSessionId, { - dbPath: MCODE_RUNTIME_DB, - }); - if (r.ok && r.lines.length > 0) { - const dbEmpty = target.chat.length === 0; - const dbShrinks = r.lines.length < target.chat.length; - const reason = dbEmpty - ? "empty" - : storedCumulative - ? "stored_cumulative" - : "stored_shrinks"; - target.chat = r.lines; - target.updatedAt = Date.now(); - saveSessions(all); // persist the populated wrapper (updatedAt bumped) - console.log( - `[switch] cid=${cid} transcript backfill ${target.id.substring(0, 8)}… mcode=${target.mcodeSessionId.substring(0, 12)}… reason=${reason} lines=${r.lines.length} msgs=${r.messageCount} probe=${r.probe}${r.truncated ? " (capped)" : ""}`, - ); - } else if (!r.ok) { - console.log( - `[switch] cid=${cid} transcript unavailable for ${target.mcodeSessionId.substring(0, 12)}… reason=${r.reason || "unknown"}`, - ); - } else if (storedCumulative) { - // Cumulative buffer + DB read came back empty — preserve - // the stored chat (which is at least the user's last view) - // and log the discrepancy so a post-mortem can see what - // happened. - console.log( - `[switch] cid=${cid} stored chat looked cumulative but DB read returned no lines; preserving stored chat for ${target.mcodeSessionId.substring(0, 12)}…`, - ); - } - } catch (e) { - console.warn( - `[switch] cid=${cid} transcript backfill failed for ${target.mcodeSessionId.substring(0, 12)}… (continuing with stored chat):`, - e && e.message ? e.message : e, - ); - } - } - } - const prevSid = cs.sessionId; - // s39 (webui-parity ticket 39): resolve the target session's workspace - // and re-point cs.workspace.dir to it BEFORE any other cs mutation, - // so the SSE state push (pushStateFor at the end) and the response - // session payload both carry the new workspace in lockstep with the - // session-id switch. The pre-fix behaviour read cs.workspace without - // writing it, which left the file tree bound to the previous project; - // this is the user-reported defect the ticket fixes. - // - // Containment gate is mandatory (s39 boundary): session-stored - // workspace is historical input — it may point to a directory the - // user removed from the allowed roots since the session was last - // opened, or to a path that was legal at the time but no longer is. - // assertWorkspacePath runs the same boundary the workspace picker, - // browseWorkspace, and the new-session POST funnel through; refusing - // here keeps that boundary singular. - const currentWs = (cs && cs.workspace && cs.workspace.dir) || ""; - const switchWs = _resolveSwitchWorkspace(target, currentWs); - if (!switchWs.ok) { - console.log( - `[switch] cid=${cid} REFUSED id=${id.substring(0, 12)}… reason=workspace_containment attempted="${switchWs.attempted}"`, - ); - res.writeHead(400, { "Content-Type": "application/json; charset=utf-8" }); - return res.end(JSON.stringify({ - ok: false, - error: switchWs.error, - attempted: switchWs.attempted, - })); + const r = await applyEngineSessionSwitch({ id, cs: ctx.cs, cid }); + if (r.outcome !== "ok") { + const contentType = + r.outcome === "workspace_refused" + ? "application/json; charset=utf-8" + : "application/json"; + res.writeHead(r.statusHint, { "Content-Type": contentType }); + return res.end(JSON.stringify(r.payload)); } - // cs.sessionId / mcodeSessionId / title / chat come first; the - // workspace write is paired with the session-id swap. Last-used-ws - // is intentionally untouched (a switch is browsing, not a workspace - // change — see the comment on handleWorkspaceChange for the same - // reasoning that protects lastUsedWorkspace from the switch path). - cs.sessionId = target.id; - cs.mcodeSessionId = target.mcodeSessionId || null; - cs.sessionTitle = target.title || "Untitled"; - cs.chat = Array.isArray(target.chat) ? target.chat : []; - cs.usage = { - ...cs.usage, - sessionInput: 0, - sessionOutput: 0, - sessionTotal: 0, - }; - cs.workspace = { - dir: switchWs.dir, - branch: null, - tree: null, - }; - if (switchWs.fallback) { - console.log( - `[switch] cid=${cid} target ${target.id.substring(0, 8)}… had no workspace — fell back to DEFAULT_WORKSPACE=${switchWs.dir}`, - ); - } - // Switching session must NOT mutate cs.lastUsedWorkspace — last-used - // is written only by handleSend (workspace change / send prompt); - // switching is browsing; pinning the browsed workspace to the top of - // the sidebar was the user-reported "click any session in C and C - // auto-sorts first" behavior. - resetContext(cs); - // Sync real token usage from mavis db on switch to a historical session - if (cs.mcodeSessionId) { - const switchedSid = cs.mcodeSessionId; - applyMavisUsageToCs(cs, switchedSid, { getMcodeModelLimit }) - .then(() => pushStateFor(cid)) - .catch((e) => { - if (process.env.MCODE_USAGE_DEBUG) - console.warn(`[switch.mavis] cid=${cid} error: ${e.message}`); - }); - } - // B01: session switch — record which session was activated and from - // which prior session. matchKind tells us whether we matched by - // mcodeSessionId or webuiId (useful when debugging "why did this - // resolve to session X"). prevSid is the prior session id (or "" if - // this was the first switch). Fail-closed → 5xx + alert. try { - _eventsAppend("session.switch", { - target: cs.sessionId, - cid, - actor: "user", - payload: { - from: prevSid || "", - matchKind: matchKind || "new_from_mcode", - mcodeSessionId: cs.mcodeSessionId || "", - title: cs.sessionTitle, - // s39 (webui-parity ticket 39): record which workspace the - // switch landed on, plus whether it was a fallback to - // DEFAULT_WORKSPACE. Both pieces are useful when auditing - // "why did the file tree change" or "why is the sidebar - // sorting by a directory I never opened". - workspace: switchWs.dir, - workspaceFallback: !!switchWs.fallback, - }, + _eventsAppend(r.audit.event, { + target: r.audit.target, + cid: r.audit.cid, + actor: r.audit.actor, + payload: r.audit.payload, }); } catch (e) { - return _auditFail(res, e, "session.switch"); + return _auditFail(res, e, r.audit.event); } pushStateFor(cid); console.log( - `[switch] cid=${cid} OK prev.sessionId=${prevSid ? prevSid.substring(0, 8) : "null"}… → new.sessionId=${cs.sessionId.substring(0, 8)}… title="${cs.sessionTitle}" chatLen=${cs.chat.length} workspace=${switchWs.dir}${switchWs.fallback ? " (DEFAULT_WORKSPACE fallback)" : ""}`, + `[switch] cid=${cid} OK prev.sessionId=${(r.audit.payload.from || "").substring(0, 8)}… → new.sessionId=${r.payload.session.id.substring(0, 8)}… title="${r.payload.session.title}" chatLen=${r.payload.session.chat.length} workspace=${r.payload.session.workspace}${r.payload.session.workspaceFallback ? " (DEFAULT_WORKSPACE fallback)" : ""}`, ); res.writeHead(200, { "Content-Type": "application/json" }); - return res.end( - JSON.stringify({ - ok: true, - session: { - id: target.id, - mcodeSessionId: cs.mcodeSessionId, - title: cs.sessionTitle, - // s39 (webui-parity ticket 39): surface the new workspace in - // the response so the client (url-restore + session-tree) can - // update its in-memory state without waiting for the SSE - // state-bus push to land — important for the file-tree panel - // that re-roots under the new workspaceDir on first render. - workspace: switchWs.dir, - workspaceFallback: !!switchWs.fallback, - // session-isolation/02 (run-mirror): switching back to the - // session that is mid-run must show what it produced so far. - // cs.chat holds the record's lines; the live turn's output is - // still in the runChat buffer — re-attach it for the owning - // view (same contract as every state snapshot). - chat: runChatViewChat(cid, cs), - }, - }), - ); + return res.end(JSON.stringify(r.payload)); } // POST /api/sessions/rename — rename a session (CRUD "update"). @@ -638,6 +312,16 @@ export async function handleSwitchSession(req, res, ctx) { // overlay record to carry the title (single-identity rule, same as the switch // path). Audit: session.rename records from → to, fail-closed. Not behind the // authorize() modal — renaming is reversible; only destructive actions prompt. +// +// M3-B5: the write itself — resolve, overlay, title write, store save, tree +// cache drop, cross-tab title fan-out — happens in +// `engine/session-writes.js#applyEngineSessionRename`, and the response body +// is built there. What stays HERE is what is genuinely the route's: the three +// 400 bodies (request validation the facade has no business reproducing), the +// 404 status for the facade's `not_found` outcome, the fail-closed audit, and +// the log line. The facade's gate for this endpoint declares NO capability — +// a rename writes webui's own store and touches no engine surface; see the +// `SESSION_WRITE_ENDPOINTS` row for the full argument. export async function handleRenameSession(req, res, ctx) { const cid = ctx.cid; const payload = await readJson(req); @@ -657,95 +341,64 @@ export async function handleRenameSession(req, res, ctx) { JSON.stringify({ ok: false, error: "title too long (max 200)" }), ); } - const all = loadSessions(); - let idx = all.findIndex((s) => s.id === id); - let matchKind = idx >= 0 ? "webuiId" : null; - if (idx < 0) { - idx = all.findIndex((s) => s.mcodeSessionId === id); - if (idx >= 0) matchKind = "mcodeSessionId"; - } - let item; - if (idx < 0) { - // 纯 mcode 会话(sidebar 的 mvs_ 条目还没有 webui 壳)→ 建壳承接改名。 - // 其余 id 不硬造记录:404,让调用方知道 id 写错了。 - if (/^mvs_[a-f0-9]{32}$/.test(id)) { - // webui-parity 63 (defect F): no workspace argument, for the same - // reason the switch path dropped it (see the s39 note above) — and here - // it was the last remaining writer. Stamping cs.workspace.dir onto - // someone else's record attributes a workspace the session never ran - // in, and cs.workspace.dir is not even necessarily a real one: a - // switch to a session that stores no workspace leaves it holding the - // DEFAULT_WORKSPACE fallback, which then got persisted and re-rooted - // the file tree on every later switch. Unknown stays unknown (""); - // the target-first read picks the fallback at read time instead. - item = ensureOverlayForMcodeSid(all, id); - matchKind = "orphan_mcode"; - } else { - res.writeHead(404, { "Content-Type": "application/json; charset=utf-8" }); - return res.end(JSON.stringify({ ok: false, error: "session not found" })); - } - } else { - item = all[idx]; - } - const from = item.title || ""; - item.title = title; - item.titleCustom = true; - item.updatedAt = Date.now(); - saveSessions(all); - // The sidebar tree reads titles from the runtime db, so drop its cache or the - // renamed title stays hidden for up to CACHE_TTL_MS. - invalidateSessionTree(); - // 所有把该会话当"当前会话"的 client 同步 sessionTitle(多 tab 一致)。 - let touchedCids = []; - for (const [c, ccs] of clients) { - if ( - ccs.sessionId === item.id || - (item.mcodeSessionId && ccs.mcodeSessionId === item.mcodeSessionId) - ) { - ccs.sessionTitle = title; - touchedCids.push(c); - } + const w = await applyEngineSessionRename({ id, title, cid }); + if (w.outcome === "not_found") { + res.writeHead(404, { "Content-Type": "application/json; charset=utf-8" }); + return res.end(JSON.stringify(w.payload)); } - if (touchedCids.length === 0) touchedCids = [cid]; - for (const c of touchedCids) pushStateFor(c); try { _eventsAppend("session.rename", { - target: item.id, + target: w.item.id, cid, actor: "user", payload: { - matchKind, - from, - to: title, - mcodeSessionId: item.mcodeSessionId || "", + matchKind: w.matchKind, + from: w.from, + to: w.to, + mcodeSessionId: w.item.mcodeSessionId || "", }, }); } catch (e) { return _auditFail(res, e, "session.rename"); } console.log( - `[rename] cid=${cid} OK match=${matchKind} id=${item.id.substring(0, 8)}… "${from}" → "${title}"`, + `[rename] cid=${cid} OK match=${w.matchKind} id=${w.item.id.substring(0, 8)}… "${w.from}" → "${w.to}"`, ); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - return res.end( - JSON.stringify({ - ok: true, - session: { - id: item.id, - mcodeSessionId: item.mcodeSessionId || null, - title: item.title, - titleCustom: true, - }, - }), - ); + return res.end(JSON.stringify(w.payload)); } // DELETE /api/sessions/:id — delete a session. // // ?dryRun=true takes the readonly SQL path (counts rows per table, // mutates nothing). Real delete passes authorize() and only then -// touches db / saveSessions / killMcodeSessionResurrection (the gate -// is the only async hop on the real path). +// touches db / saveSessions / the caches (the gate is the only async hop +// on the real path). +// +// M3-B5: this handler is now a PLAN → GOVERN → COMMIT sequence, and that +// shape is the point rather than an accident of the refactor. +// +// planEngineSessionDelete resolves the id and runs the gate. No +// mutation, so it is safe to run BEFORE +// the user is asked anything. +// authorize() + intent audit unchanged, and still strictly between +// the plan and the commit. The write-ahead +// intent line has to be durably recorded +// before any row is removed, and it +// records the match kind and chat length +// the plan produced. +// commit*EngineSessionDelete splices the store, drops the tree cache, +// mirrors the delete into the engine's +// `local_runtime_*` tables and fans the +// cleared state out to every tab. The +// ORDER of those steps inside the facade +// is the resurrection guard; see the +// facade's module header. +// +// Every status code and every response body below is unchanged. The +// bodies are now BUILT in the facade rather than here, which is what lets +// the dryRun shape be pinned byte-for-byte by a unit test instead of by a +// route test that has to stand up the whole request. export async function handleDeleteSession(req, res, ctx) { const cs = ctx.cs; const cid = ctx.cid; @@ -764,26 +417,20 @@ export async function handleDeleteSession(req, res, ctx) { } } catch {} console.log( - `[delete] cid=${cid} incoming id=${id.substring(0, 12)}… isMcodeSid=${/^mvs_[a-f0-9]{32}$/.test(id)} dryRun=${dryRun}`, + `[delete] cid=${cid} incoming id=${id.substring(0, 12)}… isMcodeSid=${isMcodeSessionId(id)} dryRun=${dryRun}`, ); - const all = loadSessions(); - let idx = all.findIndex((s) => s.id === id); - let matchKind = idx >= 0 ? "webuiId" : null; - if (idx < 0) { - idx = all.findIndex((s) => s.mcodeSessionId === id); - if (idx >= 0) matchKind = "mcodeSessionId"; - } + const plan = await planEngineSessionDelete({ id }); // B03: real-delete path must pass per-request authorize() before - // mutating db / saveSessions / killMcodeSessionResurrection. + // mutating db / saveSessions / the caches. // dryRun=true bypasses (preview only — no side effects to gate). if (!dryRun) { const authResult = await authorize("session.delete", { cid, targetSessionId: id, - matchKind: matchKind || (idx < 0 ? "unknown" : "webuiId"), - isMcodeSid: /^mvs_[a-f0-9]{32}$/.test(id), - isOrphan: idx < 0, - chatLen: idx >= 0 && all[idx] && Array.isArray(all[idx].chat) ? all[idx].chat.length : 0, + matchKind: plan.matchKind || (plan.isOrphan ? "unknown" : "webuiId"), + isMcodeSid: isMcodeSessionId(id), + isOrphan: plan.isOrphan, + chatLen: plan.chatLen, }); if (!authResult.approved) { console.log( @@ -809,9 +456,9 @@ export async function handleDeleteSession(req, res, ctx) { cid, actor: "user", payload: { - matchKind: matchKind || "unknown", - isOrphan: idx < 0, - chatLen: idx >= 0 && all[idx] && Array.isArray(all[idx].chat) ? all[idx].chat.length : 0, + matchKind: plan.matchKind || "unknown", + isOrphan: plan.isOrphan, + chatLen: plan.chatLen, decidedBy: authResult.decidedBy, }, }); @@ -822,68 +469,41 @@ export async function handleDeleteSession(req, res, ctx) { // Fallback: id is mvs_xxx but absent from webui session db — // treat it as an orphan mcode session and delete the SQL rows // directly (the webui side has no wrapper to remove). - if (idx < 0) { - if (/^mvs_[a-f0-9]{32}$/.test(id)) { - if (!dryRun) killMcodeSessionResurrection(id); - const mcodeDbDel = deleteMcodeSessionFromDb(id, { MCODE_RUNTIME_DB, dryRun }); - // Same reason as the wrapper-delete path below: this removes rows from - // the db the cached sidebar tree is built from. Skipped on a dry run, - // which mutates nothing. - if (!dryRun) invalidateSessionTree(); + if (plan.isOrphan) { + if (isMcodeSessionId(id)) { + const w = await commitEngineOrphanSessionDelete({ plan, cs, cid, dryRun }); console.log( - `[delete] cid=${cid} ORPHAN mcode session sid=${id.substring(0, 12)}… ok=${mcodeDbDel.ok}` + - (mcodeDbDel.ok - ? ` log=[${(mcodeDbDel.log || []).join(",")}]` - : ` reason=${mcodeDbDel.reason || "-"} error=${mcodeDbDel.error || "-"}`), + `[delete] cid=${cid} ORPHAN mcode session sid=${id.substring(0, 12)}… ok=${w.mcodeDbDel.ok}` + + (w.mcodeDbDel.ok + ? ` log=[${(w.mcodeDbDel.log || []).join(",")}]` + : ` reason=${w.mcodeDbDel.reason || "-"} error=${w.mcodeDbDel.error || "-"}`), ); - if (mcodeDbDel.ok) { - if (cs.mcodeSessionId === id) { - cs.mcodeSessionId = null; - cs.sessionId = null; - cs.sessionTitle = "Untitled"; - cs.chat = []; - resetContext(cs); - pushStateFor(cid); - } - // B01: orphan mcode session deletion (no webui session row). - // Outcome event; the intent line was written before the gate - // fan-out above. Failure → 5xx + alert (rows are already gone; - // the operator must see the audit gap, not a silent success). - try { - _eventsAppend("session.delete", { - target: id, - cid, - actor: "user", - payload: { - matchKind: "orphan_mcode", - dryRun, - rowsAffected: (mcodeDbDel.log || []).length, - }, - }); - } catch (e) { - return _auditFail(res, e, "session.delete(orphan_mcode)"); - } - res.writeHead(200, { - "Content-Type": "application/json; charset=utf-8", - }); - return res.end( - JSON.stringify({ - ok: true, - deleted: id, + if (w.failed) { + res.writeHead(500, { "Content-Type": "application/json" }); + return res.end(JSON.stringify(w.payload)); + } + // B01: orphan mcode session deletion (no webui session row). + // Outcome event; the intent line was written before the gate + // fan-out above. Failure → 5xx + alert (rows are already gone; + // the operator must see the audit gap, not a silent success). + try { + _eventsAppend("session.delete", { + target: id, + cid, + actor: "user", + payload: { matchKind: "orphan_mcode", dryRun, - mcodeDbDel, - }), - ); + rowsAffected: (w.mcodeDbDel.log || []).length, + }, + }); + } catch (e) { + return _auditFail(res, e, "session.delete(orphan_mcode)"); } - res.writeHead(500, { "Content-Type": "application/json" }); - return res.end( - JSON.stringify({ - ok: false, - error: "orphan mcode delete failed", - mcodeDbDel, - }), - ); + res.writeHead(200, { + "Content-Type": "application/json; charset=utf-8", + }); + return res.end(JSON.stringify(w.payload)); } console.log(`[delete] cid=${cid} 404 id=${id.substring(0, 12)}… not found`); res.writeHead(404, { "Content-Type": "application/json" }); @@ -891,12 +511,9 @@ export async function handleDeleteSession(req, res, ctx) { } // dryRun: 不真删 webui session entry,只预览 mcode db 影响 if (dryRun) { - const mcodeSid = all[idx].mcodeSessionId; - const mcodeDbDel = mcodeSid - ? deleteMcodeSessionFromDb(mcodeSid, { MCODE_RUNTIME_DB, dryRun: true }) - : { ok: true, dryRun: true, log: [], totalRows: 0 }; + const w = await previewEngineSessionDelete({ plan }); console.log( - `[delete] cid=${cid} DRYRUN id=${id.substring(0, 12)}… mcodeDbDel=${JSON.stringify(mcodeDbDel)}`, + `[delete] cid=${cid} DRYRUN id=${id.substring(0, 12)}… mcodeDbDel=${JSON.stringify(w.mcodeDbDel)}`, ); // B01: dryRun is itself a state-touching action — the operator // is previewing a delete, so record the preview but never the @@ -910,9 +527,9 @@ export async function handleDeleteSession(req, res, ctx) { cid, actor: "user", payload: { - matchKind, + matchKind: plan.matchKind, dryRun: true, - previewedRows: mcodeDbDel.totalRows || 0, + previewedRows: w.mcodeDbDel.totalRows || 0, }, }); } catch (e) { @@ -921,69 +538,9 @@ export async function handleDeleteSession(req, res, ctx) { res.writeHead(200, { "Content-Type": "application/json; charset=utf-8", }); - return res.end( - JSON.stringify({ - ok: true, - dryRun: true, - matchKind, - mcodeDbDel, - webuiEntryWouldBeDeleted: { - id: all[idx].id, - title: all[idx].title, - mcodeSessionId: mcodeSid, - }, - }), - ); + return res.end(JSON.stringify(w.payload)); } - const deletedItem = all[idx]; - all.splice(idx, 1); - saveSessions(all); - // The sidebar tree is assembled from `local_runtime_sessions` in the runtime - // db, and it is cached for CACHE_TTL_MS (the git probe per directory is the - // expensive part). A delete removes rows from that db, so the cache has to go - // or the row stays in the sidebar — still clickable — for up to 15s. This - // was the one mutation that missed it; rename had been handled, and - // switch/new were never wrong (switch does not change the set, and a new - // webui session has no engine row until its first prompt). - // - // Invalidate before the engine delete below, so the next read cannot repopulate - // from a db this call is about to change. - invalidateSessionTree(); - // Mirror the delete on the mcode side when this record has an mcode sid. - const mcodeSid = deletedItem.mcodeSessionId; - let mcodeDbDel = null; - if (mcodeSid) { - killMcodeSessionResurrection(mcodeSid); - mcodeDbDel = deleteMcodeSessionFromDb(mcodeSid, { MCODE_RUNTIME_DB }); - console.log( - `[delete] cid=${cid} mcode db delete sid=${mcodeSid.substring(0, 12)}… ok=${mcodeDbDel.ok}` + - (mcodeDbDel.ok - ? ` log=[${(mcodeDbDel.log || []).join(",")}]` - : ` reason=${mcodeDbDel.reason || "-"} error=${mcodeDbDel.error || "-"}`), - ); - } - // Clear active session on every client that pointed at this id (or - // its mcode sibling) — otherwise the next interaction in that tab - // silently recreates a webui wrapper for the same mvs sid. - let touchedCids = []; - for (const [c, ccs] of clients) { - if (ccs.sessionId === deletedItem.id || ccs.mcodeSessionId === id) { - ccs.sessionId = null; - ccs.mcodeSessionId = null; - ccs.sessionTitle = "Untitled"; - ccs.chat = []; - ccs.usage = { - ...ccs.usage, - sessionInput: 0, - sessionOutput: 0, - sessionTotal: 0, - }; - resetContext(ccs); - touchedCids.push(c); - } - } - if (touchedCids.length === 0) touchedCids = [cid]; - for (const c of touchedCids) pushStateFor(c); + const w = await commitEngineSessionDelete({ plan, cid }); // B01: real session delete (the dangerous one). Record which webui // session was deleted, what the match kind was, how many cids had // their active session cleared (this is the "fan-out" effect that @@ -998,31 +555,22 @@ export async function handleDeleteSession(req, res, ctx) { cid, actor: "user", payload: { - matchKind, + matchKind: plan.matchKind, dryRun: false, - remaining: all.length, - touchedCids: touchedCids.length, - mcodeRowsAffected: mcodeDbDel && mcodeDbDel.log ? mcodeDbDel.log.length : 0, - title: deletedItem.title, + remaining: w.records.length, + touchedCids: w.touchedCids.length, + mcodeRowsAffected: w.mcodeDbDel && w.mcodeDbDel.log ? w.mcodeDbDel.log.length : 0, + title: w.deletedItem.title, }, }); } catch (e) { return _auditFail(res, e, "session.delete"); } console.log( - `[delete] cid=${cid} OK match=${matchKind} deleted.webuiId=${deletedItem.id.substring(0, 8)}… remaining=${all.length}`, + `[delete] cid=${cid} OK match=${plan.matchKind} deleted.webuiId=${w.deletedItem.id.substring(0, 8)}… remaining=${w.records.length}`, ); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - return res.end( - JSON.stringify({ - ok: true, - deleted: id, - matchKind, - dryRun: false, - remaining: all.length, - mcodeDbDel, - }), - ); + return res.end(JSON.stringify(w.payload)); } // GET /api/session-tree — the sidebar's Project → directory → session → subagent @@ -1279,39 +827,23 @@ export async function handleSearchSessions(req, res, ctx) { // The cleanup targets: default-named webui sessions (New session / // Untitled / 对话 N) whose chat is empty AND whose updatedAt is older // than 24h — same rule as cleanupEmptyDefaultSessions() in lib/sessions.js. -import { existsSync, readFileSync } from "node:fs"; -import { SESSIONS_DB } from "../lib/config.js"; +// +// M3-B5: the SELECTION moved into the facade +// (`engine/session-writes.js#readOrphanSessionWriteIds`), together with +// the store read it applies the rule to and with the two response bodies +// the batch's red line pins byte-for-byte. The rule and the file it reads +// are one decision; splitting them across two modules is how a sweep ends +// up pruning a different store than the one it was written for. +// +// The DELEGATION stays here and is not an oversight. Each selected id is +// routed back through `handleDeleteSession` precisely so that every +// orphan costs the same `session.delete.intent` / `session.delete` audit +// pair, the same authorize() decision and the same cross-tab fan-out that +// a hand-deleted session costs. Re-implementing the delete inside the +// sweep would produce a cheaper path that is not the same path, and the +// audit chain is the thing this endpoint exists to preserve. import { readJson } from "../lib/read-json.js"; -const ORPHAN_STALE_MS = 24 * 60 * 60 * 1000; - -function _findOrphanIds() { - if (!existsSync(SESSIONS_DB)) return []; - let all; - try { - let raw = readFileSync(SESSIONS_DB, "utf8"); - if (raw.charCodeAt(0) === 0xfeff) raw = raw.slice(1); // 剥 BOM - all = JSON.parse(raw); - } catch { - return []; - } - if (!Array.isArray(all) || all.length === 0) return []; - const now = Date.now(); - return all - .filter((s) => { - if (!s || !s.id) return false; - const hasChat = Array.isArray(s.chat) && s.chat.length > 0; - if (hasChat) return false; - const t = (s.title || "").trim(); - const isDefault = - t === "New session" || t === "Untitled" || /^对话 \d+$/.test(t); - if (!isDefault) return false; - if (s.updatedAt && now - s.updatedAt < ORPHAN_STALE_MS) return false; - return true; - }) - .map((s) => s.id); -} - export async function handleCleanupOrphans(req, res, ctx) { const cid = (ctx && ctx.cid) || ""; let dryRun = false; @@ -1322,19 +854,17 @@ export async function handleCleanupOrphans(req, res, ctx) { dryRun = params.get("dryRun") === "true"; } } catch {} - const targetIds = _findOrphanIds(); - // Preview path: no authorize gate (no side effects). + const sweep = await readOrphanSessionWriteIds(); + const targetIds = sweep.ids; + // Preview path: no authorize gate (no side effects). The body is + // `{ok, dryRun, count, ids}` — four keys, in that order — and it is + // built in the facade so that shape has exactly one home. if (dryRun) { console.log( `[cleanup-orphans] cid=${cid} DRYRUN would-delete=${targetIds.length}`, ); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - return res.end(JSON.stringify({ - ok: true, - dryRun: true, - count: targetIds.length, - ids: targetIds, - })); + return res.end(JSON.stringify(sweep.payload)); } // Real path: gate with authorize() before touching any session. if (targetIds.length === 0) { diff --git a/packages/webui/test/helpers/_setup.js b/packages/webui/test/helpers/_setup.js index d30db95f..af48af8c 100644 --- a/packages/webui/test/helpers/_setup.js +++ b/packages/webui/test/helpers/_setup.js @@ -294,6 +294,41 @@ export async function setupMocks(t, overrides = {}) { } }, persistCurrentChat: () => {}, + // M3-B5: lib/sessions.js really exports this one — the + // single-identity rule, an overlay record whose `id` IS the engine + // sid — and routes/sessions.js has imported it since the switch + // path added it, but the mock never grew it. Every consumer so far + // either never called it or owned its own store mock, and a missing + // name only bites at module-instantiation time. M3-B5 moved the + // RENAME path's call into the engine facade, whose orphan-mcode + // branch calls it, so the omission became reachable from this + // shared helper rather than from a test that could stub around it. + // Mirrors the real body, including the placeholder-title repair and + // the unshift, so a test that renames a bare mvs_ id sees the + // record it would see in production. + ensureOverlayForMcodeSid: (all, sid, { title, workspace } = {}) => { + if (!Array.isArray(all) || !sid) return null; + let rec = all.find((s) => s && s.mcodeSessionId === sid) || null; + if (rec) { + if (title && rec.title === "Mcode session") rec.title = title; + return rec; + } + rec = { + id: sid, + mcodeSessionId: sid, + title: title || "Mcode session", + workspace: workspace || "", + createdAt: Date.now(), + updatedAt: Date.now(), + chat: [], + }; + all.unshift(rec); + return rec; + }, + findOverlayForMcodeSid: (all, sid) => { + if (!Array.isArray(all) || !sid) return null; + return all.find((s) => s && s.mcodeSessionId === sid) || null; + }, // session-isolation/02 (run-mirror): the buffer-drain finalize path // (routes/chat.js) writes the turn back to the owning session's // persisted record; mirror the real lookup (by webui id, then by diff --git a/packages/webui/test/lib/engine/account-reads.test.js b/packages/webui/test/lib/engine/account-reads.test.js new file mode 100644 index 00000000..94648f0b --- /dev/null +++ b/packages/webui/test/lib/engine/account-reads.test.js @@ -0,0 +1,449 @@ +// webui/test/lib/engine/account-reads.test.js +// +// M3-B4: the account read's engine facade (#20). +// +// What this file pins, and why the family needs pinning at all when +// the endpoint is five lines long: +// +// 1. THE DECLARATION. #20 and B3's #15 / #16 read the SAME engine +// projection through the SAME `mcode/account/status` method, so +// they must be gated by the SAME `authCredentials.getAccountStatus` +// pair. If the two ever drift, a provider that drops the method +// takes one endpoint down and leaves the other claiming a quota it +// cannot read — section 1 asserts the pair against the usage +// family's own table, not against a copy of it. +// +// 2. THE SOFT-FAILURE BODY. `{ok:false, reason}` at HTTP 200 is the +// account card's documented empty state, and it is produced by the +// ENGINE failing, not by the request failing. A refactor that +// converts it into a thrown error or a 500 turns a card that +// renders 本地用户 into a broken menu. +// +// 3. THE SUCCESS BODY'S SPREAD. `{ok:true, ...r.data}` means the +// engine frames its own projection; a layer that started picking +// fields (`payload.identity`, `payload.tokenPlan`) would silently +// drop every field the engine adds next year, and no test that +// only checks today's fields would notice. +// +// 4. THE GATE IS REAL, AND THE MOCK IS REAL. The registered provider +// declares `authCredentials` `full`, so only this file can prove +// the gate would bite. And node:test's `mock.module` re-evaluates +// only the MOCKED specifier, so a route module already in the +// registry keeps its old live binding — every route test here +// re-imports the route under a fresh `?bust=N`, and section 5 ends +// with the control that proves the mock took: with no mock at all, +// the same request answers from the real rpc layer. +// +// Test style follows test/lib/engine/usage-reads.test.js (B3) and +// test/lib/engine/session-tree-reads.test.js (B2): table-driven, one row +// per case. + +import { test, describe, after } from "node:test"; +import assert from "node:assert/strict"; + +import { setupMocks, absPath, registerRpcMock } from "../../helpers/_setup.js"; + +const { ENGINE_CAPABILITY_KEYS } = await import("../../../server/engine/index.js"); +const { + ACCOUNT_READ_ENDPOINTS, + assertAccountReadCapability, + readEngineAccount, + resolveAccountReadProvider, +} = await import("../../../server/engine/account-reads.js"); +const { EngineCapabilityNotSupportedError, isEngineCapabilityNotSupportedError } = await import( + "../../../server/engine/errors.js" +); +const { USAGE_READ_ENDPOINTS } = await import("../../../server/engine/usage-reads.js"); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +after(() => { + registerRpcMock({ getAccountStatus: async () => ({ ok: false, code: "no_client" }) }); +}); + +// --------------------------------------------------------------------------- +// 1. The endpoint → capability declaration table +// --------------------------------------------------------------------------- + +describe("ACCOUNT_READ_ENDPOINTS — this batch's declaration table", () => { + test("covers exactly the one endpoint of the account family", () => { + assert.deepEqual(Object.keys(ACCOUNT_READ_ENDPOINTS), ["GET /api/account"]); + }); + + test("GET /api/account declares authCredentials.getAccountStatus", () => { + // Table-driven: editing this row is a capability decision and must be + // reviewed as one, so the table IS the assertion. + const row = { capability: "authCredentials", subItem: "getAccountStatus" }; + assert.deepEqual(ACCOUNT_READ_ENDPOINTS["GET /api/account"], row); + assert.ok(ENGINE_CAPABILITY_KEYS.includes(row.capability)); + }); + + test("it is the SAME pair B3's usage endpoints declare, because it is the same engine call", () => { + // The whole point of section 1. #20, #15 and #16 all read the + // engine's account projection through `mcode/account/status`; a + // `partial` provider that drops `getAccountStatus` must be refused + // by all three, in the same way, naming the same method. + for (const endpoint of ["POST /api/usage", "POST /api/usage-trigger"]) { + assert.deepEqual( + ACCOUNT_READ_ENDPOINTS["GET /api/account"], + USAGE_READ_ENDPOINTS[endpoint], + `${endpoint} drifted from the account family`, + ); + } + }); + + test("an endpoint outside this family is caller confusion, not an engine limitation", () => { + assert.throws( + () => assertAccountReadCapability("GET /api/nope", RUNTIME), + (err) => { + assert.ok(!(err instanceof EngineCapabilityNotSupportedError)); + assert.equal(err.code, "unknown_account_read_endpoint"); + assert.match(err.message, /not part of the account family/); + return true; + }, + ); + }); +}); + +// --------------------------------------------------------------------------- +// 2. Provider resolution + the gate +// --------------------------------------------------------------------------- + +describe("resolveAccountReadProvider / assertAccountReadCapability", () => { + // Table-driven. Absent means "no provider claims this transport yet" + // (M4), which is NOT the same answer as "capability unavailable" — + // the default `acp` transport must keep answering, so it must NOT + // throw. + const TRANSPORTS = [ + [RUNTIME, true, "checked", "local-runtime-v2"], + [ACP, false, "unregistered-transport", null], + ["exec", false, "unregistered-transport", null], + ["", false, "unregistered-transport", null], + ]; + for (const [transport, hasProvider, gate, providerId] of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → ${gate}`, () => { + const provider = resolveAccountReadProvider(transport); + assert.equal(!!provider, hasProvider); + const g = assertAccountReadCapability("GET /api/account", transport); + assert.equal(g.gate, gate); + assert.equal(g.provider, providerId); + assert.equal(g.capability, "authCredentials"); + assert.equal(g.subItem, "getAccountStatus"); + assert.equal(g.endpoint, "GET /api/account"); + }); + } + + test("the descriptor has exactly the six fields every family's descriptor has", () => { + // A consumer that reads `gate.provider` under `acp` must get `null`, + // not `undefined` — the key must EXIST. Same key set as B1/B2/B3. + assert.deepEqual(Object.keys(assertAccountReadCapability("GET /api/account", RUNTIME)), [ + "endpoint", + "gate", + "provider", + "capability", + "subItem", + ]); + }); +}); + +// --------------------------------------------------------------------------- +// 3. The payload — the soft-failure body and the verbatim spread +// --------------------------------------------------------------------------- + +describe("readEngineAccount — the payload is the endpoint's, in both shapes", () => { + // Every case in this table is a REAL engine answer shape the endpoint + // has to render. The row is [engine result, expected payload, why]. + const TABLE = [ + [ + { ok: true, data: { identity: { name: "Ada" }, tokenPlan: { tier: "pro" } } }, + { ok: true, identity: { name: "Ada" }, tokenPlan: { tier: "pro" } }, + "the projection is spread verbatim", + ], + [ + { ok: true, data: { identity: { name: "Ada" }, futureEngineField: 7 } }, + { ok: true, identity: { name: "Ada" }, futureEngineField: 7 }, + "a field webui has never heard of still reaches the card", + ], + [ + { ok: true, data: null }, + { ok: true }, + "`data:null` must not throw on the spread", + ], + [ + { ok: true, data: undefined }, + { ok: true }, + "an absent `data` behaves the same as a null one", + ], + [ + { ok: true }, + { ok: true }, + "no `data` key at all", + ], + [ + { ok: false, code: "no_client" }, + { ok: false, reason: "no_client" }, + "the engine's own machine-readable code becomes the reason", + ], + [ + { ok: false, code: "unauthorized" }, + { ok: false, reason: "unauthorized" }, + "any code, verbatim", + ], + [ + { ok: false, error: "boom" }, + { ok: false, reason: "account_unavailable" }, + "no code → the endpoint's own historical fallback string", + ], + [ + { ok: false, code: "" }, + { ok: false, reason: "account_unavailable" }, + "an empty code is falsy and falls back, exactly as `||` did", + ], + [ + null, + { ok: false, reason: "account_unavailable" }, + "a null result must not throw — it is a failure, not a crash", + ], + ]; + for (const [result, expected, why] of TABLE) { + test(`${why}: ${JSON.stringify(result)} → ${JSON.stringify(expected)}`, async (t) => { + await setupMocks(t, { acp: {} }); + // `setupMocks` already registered the `lib/mcode-rpc.js` mock and + // node:test refuses a second registration for the same specifier + // (ERR_INVALID_STATE), so the payload is injected through the + // helper's mutable dispatch-through holder — the mechanism + // `registerRpcMock` exists for. + registerRpcMock({ getAccountStatus: async () => result }); + const read = await readEngineAccount({ cs: { mcodeSessionId: "mvs_1" }, transport: RUNTIME }); + assert.deepEqual(read.payload, expected); + assert.equal(read.source, "account-status"); + assert.equal(read.gate.gate, "checked"); + }); + } + + test("the failure payload has EXACTLY two keys, in order", async (t) => { + // A key-set assertion, not a subset: a facade that helpfully added + // `provider` or `gate` to the failure body would be a frontend + // contract change, and `ok:false` bodies are what the card branches + // on. + await setupMocks(t, { acp: {} }); + registerRpcMock({ getAccountStatus: async () => ({ ok: false, code: "no_client" }) }); + const read = await readEngineAccount({ cs: {}, transport: RUNTIME }); + assert.deepEqual(Object.keys(read.payload), ["ok", "reason"]); + }); + + test("cs.mcodeSessionId is forwarded EXACTLY as the route computed it", async (t) => { + // Table-driven: [ctx-ish cs, expected forwarded argument]. The route + // used to evaluate `ctx && ctx.cs && ctx.cs.mcodeSessionId`, so a + // missing ctx forwarded `undefined` and a cs without a session id + // forwarded `undefined` too — but a cs whose id is `""` forwarded + // `""`. `getAccountStatus` turns any falsy value into `{}`, so the + // difference is invisible on the wire and very visible to a test + // that pins the call. + await setupMocks(t, { acp: {} }); + const seen = []; + registerRpcMock({ + getAccountStatus: async (sessionId) => { + seen.push(sessionId); + return { ok: true, data: {} }; + }, + }); + const CASES = [ + [{ mcodeSessionId: "mvs_1" }, "mvs_1"], + [{ mcodeSessionId: "" }, ""], + [{}, undefined], + [{ mcodeSessionId: null }, null], + [{ mcodeSessionId: 0 }, 0], + ]; + for (const [cs] of CASES) { + await readEngineAccount({ cs, transport: RUNTIME }); + } + // A missing ctx entirely: the facade must not throw on `undefined`. + await readEngineAccount({ transport: RUNTIME }); + seen.push(""); + assert.deepEqual(seen, ["mvs_1", "", undefined, null, 0, undefined, ""]); + }); + + test("the gate runs BEFORE the engine call", async (t) => { + // Order matters: a provider that does not offer `getAccountStatus` + // must cost zero engine calls, so the 501 does not depend on the + // engine answering anything at all. + await setupMocks(t, { acp: {} }); + let called = 0; + registerRpcMock({ + getAccountStatus: async () => { + called += 1; + return { ok: true, data: {} }; + }, + }); + await assert.rejects( + () => readEngineAccount({ cs: {}, endpoint: "GET /api/nope", transport: RUNTIME }), + (err) => { + assert.ok(!isEngineCapabilityNotSupportedError(err)); + assert.equal(err.code, "unknown_account_read_endpoint"); + return true; + }, + ); + assert.equal(called, 0); + }); +}); + +// --------------------------------------------------------------------------- +// 4. The route +// --------------------------------------------------------------------------- + +describe("handleGetAccount — the route asks the facade", () => { + // One fresh route module per test: node:test's `mock.module` + // re-evaluates only the MOCKED specifier, but a route module already + // in the registry keeps its old LIVE BINDING to the facade — without + // the `?bust=N` re-import the second test here would silently + // exercise the first test's mock and pass for the wrong reason. + let bust = 0; + const loadRoute = async () => import(`${absPath("routes/account.js")}?bust=${bust++}`); + + // `mock.module` REPLACES the whole namespace, so a partial mock makes + // the route fail to instantiate on the exports it did not stub + // ("does not provide an export named …"). `readEngineAccount` is the + // route's only facade import, but the helper is kept so the next + // family to copy this file has the shape ready. + const NOT_STUBBED = (name) => async () => { + throw new Error(`B4 test called ${name}, which this case did not stub`); + }; + function mockFacade(t, overrides) { + t.mock.module(absPath("engine/account-reads.js"), { + namedExports: { readEngineAccount: NOT_STUBBED("readEngineAccount"), ...overrides }, + }); + } + + function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; + } + + test("both bodies are written byte-for-byte at HTTP 200", async (t) => { + // Two cases, one mock registration: node:test refuses to mock the + // same specifier twice inside one test, and a mutable holder is the + // honest way to say "the same route, two payloads". + const CASES = [ + { ok: true, identity: { name: "Ada" }, tokenPlan: { tier: "pro" } }, + { ok: false, reason: "no_client" }, + ]; + let current = CASES[0]; + mockFacade(t, { + readEngineAccount: async () => ({ + payload: current, + source: "account-status", + gate: {}, + transport: RUNTIME, + }), + }); + for (const payload of CASES) { + current = payload; + const route = await loadRoute(); + const res = mkRes(); + await route.handleGetAccount(null, res, { cs: { mcodeSessionId: "mvs_1" } }); + assert.equal(res.written[0].status, 200); + assert.equal(res.written[0].headers["Content-Type"], "application/json; charset=utf-8"); + assert.equal(res.written[1].body, JSON.stringify(payload)); + } + }); + + test("the route hands its ctx straight through and does not read cs itself", async (t) => { + await setupMocks(t, { acp: {} }); + const seen = []; + mockFacade(t, { + readEngineAccount: async (o) => { + seen.push(o); + return { payload: { ok: true }, source: "account-status", gate: {}, transport: RUNTIME }; + }, + }); + const route = await loadRoute(); + // A missing ctx is a real call shape (`invokeHandler` always sets + // one, but the route's signature must not assume it) and must not + // throw — `ctx && ctx.cs` is what the pre-facade route evaluated. + for (const ctx of [{ cs: { mcodeSessionId: "mvs_1" } }, { cs: null }, undefined, {}]) { + await route.handleGetAccount(null, mkRes(), ctx); + } + assert.equal(seen.length, 4); + assert.deepEqual(seen[0].cs, { mcodeSessionId: "mvs_1" }); + assert.equal(seen[1].cs, null); + assert.equal(seen[2].cs, undefined); + assert.equal(seen[3].cs, undefined); + // The route must not pass an endpoint key of its own: the facade's + // default IS the endpoint, and a route that spelled it out would be + // a second place to get it wrong. + for (const o of seen) assert.equal(o.endpoint, undefined); + }); + + test("a capability error PROPAGATES so invokeHandler can answer 501", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineAccount: async () => { + throw new EngineCapabilityNotSupportedError({ + capability: "authCredentials", + provider: "fixture-provider", + missing: ["getAccountStatus"], + reason: "test fixture", + }); + }, + }); + const route = await loadRoute(); + await assert.rejects( + () => route.handleGetAccount(null, mkRes(), { cs: {} }), + isEngineCapabilityNotSupportedError, + ); + }); + + // ---- proof the mock actually took ------------------------------------ + + test("PROOF the facade mock took: a marker error escapes the untouched route", async (t) => { + // Without a fresh `?bust=` re-import, `mock.module` would leave the + // route holding the PREVIOUS test's live binding, the marker would + // never be thrown, and this assertion would fail — which is the + // point: it is the only assertion here that cannot pass by accident. + await setupMocks(t, { acp: {} }); + const marker = new Error("B4-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + readEngineAccount: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleGetAccount(null, mkRes(), { cs: {} }); + } catch (err) { + caught = err; + } + assert.ok(caught, "the route swallowed the facade error — either the mock did not take, or the route grew a catch"); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + + test("CONTROL: with no facade mock, the same request reaches the rpc layer", async (t) => { + // The other half of the proof. A `?bust=` re-import under a fresh + // test hook gives a route bound to the REAL facade, so the request + // answers from the rpc layer. The holder is process-global and the + // previous cases left payloads in it, so this one puts back the + // clean-disk default — `no_client`, the answer the account card + // renders its empty state from in production when no engine has + // attached. + await setupMocks(t, { acp: {} }); + registerRpcMock({ getAccountStatus: async () => ({ ok: false, code: "no_client" }) }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleGetAccount(null, res, { cs: { mcodeSessionId: "mvs_1" } }); + assert.equal(res.written[0].status, 200); + assert.deepEqual(JSON.parse(res.written[1].body), { ok: false, reason: "no_client" }); + }); +}); diff --git a/packages/webui/test/lib/engine/capability-reads.test.js b/packages/webui/test/lib/engine/capability-reads.test.js new file mode 100644 index 00000000..41041619 --- /dev/null +++ b/packages/webui/test/lib/engine/capability-reads.test.js @@ -0,0 +1,551 @@ +// webui/test/lib/engine/capability-reads.test.js +// +// M3-B4: the capability-declaration read's engine facade (#73). +// +// This is the one endpoint in the migration whose RESPONSE CONTRACT +// changes, by explicit decision: `capabilities` used to be +// `MCODE_ACP_CAPABILITIES`, a hand-maintained flat `{method: boolean}` +// table of the ACP JSON-RPC surface, and it is now the engine's +// DECLARED 14-key capability object. So the tests here pin the +// replacement, not an absence of change: +// +// 1. THE REPLACEMENT. `capabilities` carries the declaration, forwarded +// by identity, and the twelve old accessors are asserted GONE — a +// consumer that still reads `capabilities.set_mode` must get +// `undefined` and fail loudly rather than silently receive a +// truthy object field. The declaration must appear exactly once in +// the serialised body: the `engine` key an earlier shape of this +// batch shipped was removed precisely because it carried the same +// 14 keys a second time. `mcodeVersion` / `mcodeName` / +// `mcodeTitle` and `notes` are untouched, and `notes` stays last. +// +// 2. `capabilitiesProviderFor`. The response must say whether the +// declaration came from the ACTIVE transport's provider or from the +// default provider standing in for a transport nothing claims yet +// (M4). A capability-detection endpoint that reported a +// standing-in declaration as though it were the connected engine's +// is the same lie B1 declined for `/api/health` — and this is the +// one endpoint where it is most tempting, because the fallback is +// silent and always succeeds. +// +// 3. THE EMPTY-DECLARATION RULE. #73 must never answer an empty +// view. A frontend that gets `{capabilities:{}}` cannot tell "no +// engine" from "this build has no declarations", and the whole +// point of the endpoint is that distinction. +// +// 4. THE GATE IS A NO-OP, AND SAYS SO. #73 is the declaration +// endpoint; gating the gate would let a `none` hide the +// declaration that says so. `checkCapabilityReadCapability` must +// report `no-capability-key` under EVERY transport, including a +// provider that declares nothing at all. +// +// Test style follows test/lib/engine/usage-reads.test.js (B3) and +// test/lib/engine/account-reads.test.js (B4 #20). + +import { test, describe, after } from "node:test"; +import assert from "node:assert/strict"; + +import { setupMocks, absPath, registerAcpMock } from "../../helpers/_setup.js"; + +const { + CAPABILITY_READ_ENDPOINTS, + checkCapabilityReadCapability, + readEngineCapabilityView, + resolveCapabilityReadProvider, +} = await import("../../../server/engine/capability-reads.js"); +const { ENGINE_CAPABILITY_KEYS, LOCAL_RUNTIME_V2_CAPABILITIES } = await import( + "../../../server/engine/index.js" +); +const { summarizeUnavailableCapabilities } = await import("../../../server/engine/capabilities.js"); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +const AGENT_INFO = { name: "mcode", title: "Mcode", version: "0.5.5" }; + +after(() => { + registerAcpMock({ getMcodeServerInfo: () => null }); +}); + +// --------------------------------------------------------------------------- +// 1. The declaration table — the no-op, pinned +// --------------------------------------------------------------------------- + +describe("CAPABILITY_READ_ENDPOINTS — the gate is a reported no-op", () => { + test("covers exactly the one endpoint of the capability family", () => { + assert.deepEqual(Object.keys(CAPABILITY_READ_ENDPOINTS), [ + "GET /api/protocol/capabilities", + ]); + }); + + test("#73 declares NO capability — it IS the declaration endpoint", () => { + // Gating the gate is circular: a `none` anywhere in the declaration + // could hide the declaration that says so. The value is `null` for + // the same reason B1's `/api/health` and B3's `/api/usage/forecast` + // are. + assert.equal(CAPABILITY_READ_ENDPOINTS["GET /api/protocol/capabilities"], null); + }); + + // Table-driven over EVERY transport, not just the two that matter: the + // assertion is that the no-op is unconditional. + const TRANSPORTS = [RUNTIME, ACP, "exec", "", "nonsense"]; + for (const transport of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → no-capability-key`, () => { + const g = checkCapabilityReadCapability("GET /api/protocol/capabilities", transport); + assert.equal(g.gate, "no-capability-key"); + assert.equal(g.capability, null); + assert.equal(g.subItem, null); + assert.equal(g.enforcement, "soft"); + // The provider is still NAMED even though nothing is checked — + // "no capability key" must not degrade into "no provider". + assert.equal(g.provider, "local-runtime-v2"); + }); + } + + test("an endpoint outside this family is caller confusion", () => { + assert.throws( + () => checkCapabilityReadCapability("GET /api/nope", RUNTIME), + (err) => { + assert.equal(err.code, "unknown_capability_read_endpoint"); + assert.match(err.message, /not part of the capability family/); + return true; + }, + ); + }); +}); + +// --------------------------------------------------------------------------- +// 2. Provider resolution — always answers, and says how +// --------------------------------------------------------------------------- + +describe("resolveCapabilityReadProvider — it never returns nothing", () => { + // Table-driven. `[transport, providerFor]` — the whole family differs + // from B1/B2/B3 here: there is no `null` row, because an empty + // capability view is worse than useless for a capability-DETECTION + // endpoint. The `providerFor` field is what keeps the fallback honest. + const TRANSPORTS = [ + [RUNTIME, "transport"], + [ACP, "default"], + ["exec", "default"], + ["", "default"], + ["nonsense", "default"], + ]; + for (const [transport, providerFor] of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → providerFor=${providerFor}`, () => { + const { provider, providerFor: actual } = resolveCapabilityReadProvider(transport); + assert.equal(provider.id, "local-runtime-v2"); + assert.equal(provider.transport, "runtime"); + assert.equal(actual, providerFor); + // The declaration served is the real reviewed object, not a copy + // that could drift from it. + assert.equal(provider.capabilities, LOCAL_RUNTIME_V2_CAPABILITIES); + }); + } +}); + +// --------------------------------------------------------------------------- +// 3. The view +// --------------------------------------------------------------------------- + +describe("readEngineCapabilityView", () => { + test("the read is the engine-capabilities payload /api/engine-capabilities serves", async (t) => { + // Same declaration, same source object. If the two endpoints ever + // answer different declarations there are two truths in webui, and + // this assertion is what stops that. + await setupMocks(t, { acp: { getMcodeServerInfo: () => AGENT_INFO } }); + const read = await readEngineCapabilityView({ transport: RUNTIME }); + // The read's key set, asserted exactly: the ACP wire table is gone + // from this layer, and a `wire` field reappearing here would put a + // second "what can the engine do" answer back in the facade. + assert.deepEqual(Object.keys(read), [ + "declaration", + "unavailable", + "provider", + "providerFor", + "engineTransport", + "agent", + "source", + "gate", + "transport", + ]); + assert.deepEqual(Object.keys(read.declaration), [...ENGINE_CAPABILITY_KEYS]); + assert.equal(read.declaration, LOCAL_RUNTIME_V2_CAPABILITIES); + assert.deepEqual( + read.unavailable, + summarizeUnavailableCapabilities(LOCAL_RUNTIME_V2_CAPABILITIES), + ); + assert.equal(read.provider, "local-runtime-v2"); + assert.equal(read.engineTransport, "runtime"); + assert.equal(read.source, "declaration"); + assert.equal(read.transport, RUNTIME); + }); + + // Table-driven. The `initialize` mirror is empty until something + // attaches, and the endpoint's own fallbacks must survive that — #75 + // answers the same figure with the same fallback, and two endpoints + // answering it differently would be the defect. + const AGENT_CASES = [ + [{ name: "mcode", title: "Mcode", version: "0.5.5" }, { version: "0.5.5", name: "mcode", title: "Mcode" }], + [{ version: "0.5.5" }, { version: "0.5.5", name: null, title: null }], + [{ name: "mcode" }, { version: "unknown", name: "mcode", title: null }], + [null, { version: "unknown", name: null, title: null }], + ]; + for (const [info, expected] of AGENT_CASES) { + test(`agentInfo ${JSON.stringify(info)} → ${JSON.stringify(expected)}`, async (t) => { + await setupMocks(t, { acp: { getMcodeServerInfo: () => info } }); + const read = await readEngineCapabilityView({ transport: RUNTIME }); + assert.deepEqual(read.agent, expected); + assert.deepEqual(Object.keys(read.agent), ["version", "name", "title"]); + }); + } + + // Table-driven. The VIEW's `providerFor` — not just the resolver's — + // is what a consumer branches on, so a facade that resolved the + // provider honestly and then hard-coded the label in the payload would + // defeat the whole point. This table is the assertion that separates + // those two. + const PROVIDER_FOR = [ + [RUNTIME, "transport"], + [ACP, "default"], + ["exec", "default"], + ["nonsense", "default"], + ]; + for (const [transport, expected] of PROVIDER_FOR) { + test(`the view reports providerFor=${expected} on transport ${JSON.stringify(transport)}`, async (t) => { + await setupMocks(t, { acp: {} }); + const read = await readEngineCapabilityView({ transport }); + assert.equal(read.providerFor, expected); + // And the two halves cannot disagree: `providerFor: "transport"` + // with a provider the transport does not own is the lie. + assert.equal(read.providerFor === "transport", transport === RUNTIME); + }); + } + + test("an empty transport override means 'the ambient one', and the view says so", async (t) => { + // `options.transport || config.MCODE_WEBUI_TRANSPORT` treats `""` as + // "not specified" — the same idiom every other read family uses. It + // is also why the table above has no `""` row: the answer would + // depend on the gate's own `MCODE_WEBUI_TRANSPORT`, and a test whose + // expected value depends on the ambient env is a test that is green + // on one transport and red on the other. + await setupMocks(t, { acp: {} }); + const { MCODE_WEBUI_TRANSPORT } = await import(absPath("lib/config.js")); + const read = await readEngineCapabilityView({ transport: "" }); + assert.equal(read.transport, MCODE_WEBUI_TRANSPORT); + assert.equal( + read.providerFor, + MCODE_WEBUI_TRANSPORT === RUNTIME ? "transport" : "default", + ); + }); + + test("the declaration is forwarded by IDENTITY, and there is no ACP wire field", async (t) => { + // A copy would be a second thing that can drift from the reviewed + // declaration, which is the failure this endpoint had before M3-B4. + // Identity pins the forwarding; the absence assertion pins the + // replacement, so re-adding `MCODE_ACP_CAPABILITIES` anywhere in + // this layer is a red bar rather than a silent second answer. + await setupMocks(t, { acp: {} }); + const read = await readEngineCapabilityView({ transport: RUNTIME }); + assert.equal(read.declaration, LOCAL_RUNTIME_V2_CAPABILITIES); + assert.equal("wire" in read, false); + // And the facade must not even REACH for the rpc module any more: + // the field it used to carry is the only reason it did. Asserted on + // the SOURCE, because an unused import is behaviourally inert and no + // behavioural test can tell it apart from a clean module — but it + // would put `lib/mcode-rpc.js` (and its `acp.mjs` / settings chain) + // back on the lazy-import path of a boot-reachable module for + // nothing. A static tripwire is the honest instrument here. + const { readFileSync } = await import("node:fs"); + const { fileURLToPath } = await import("node:url"); + const source = readFileSync( + fileURLToPath(new URL(absPath("engine/capability-reads.js"))), + "utf8", + ); + // Matched on the IMPORT FORM, not the bare file name: this module's + // header deliberately names `lib/mcode-rpc.js` in prose (the debt + // note, the boot-path note), and a tripwire that fired on the prose + // would be a tripwire nobody could satisfy. + assert.equal( + /\bimport\s*\(?\s*["'][^"']*lib\/mcode-rpc\.js/.test(source), + false, + "capability-reads.js must not import lib/mcode-rpc.js — the ACP wire table is no longer part of this read", + ); + // The constant itself is untouched; it is simply unconsumed (see + // the KNOWN DEBT note in the module header). + const rpc = await import(absPath("lib/mcode-rpc.js")); + assert.equal(typeof rpc.MCODE_ACP_CAPABILITIES, "object"); + }); + + test("the gate is evaluated and reported, and never blocks the read", async (t) => { + await setupMocks(t, { acp: {} }); + // Every transport, including one no provider claims. A read that + // gated would throw here; a read that skipped the check entirely + // would have no `gate` field at all. + for (const transport of [RUNTIME, ACP, "exec"]) { + const read = await readEngineCapabilityView({ transport }); + assert.equal(read.gate.gate, "no-capability-key"); + assert.equal(read.gate.endpoint, "GET /api/protocol/capabilities"); + } + }); + + test("an unknown endpoint key is a plain Error, not 501 material", async (t) => { + await setupMocks(t, { acp: {} }); + await assert.rejects( + () => readEngineCapabilityView({ endpoint: "GET /api/nope", transport: RUNTIME }), + (err) => { + assert.equal(err.code, "unknown_capability_read_endpoint"); + return true; + }, + ); + }); +}); + +// --------------------------------------------------------------------------- +// 4. The route — the REPLACEMENT, pinned key by key +// --------------------------------------------------------------------------- + +describe("handleCapabilities — capabilities is the engine-capabilities view", () => { + let bust = 0; + const loadRoute = async () => import(`${absPath("routes/protocol.js")}?bust=${bust++}`); + + // `mock.module` REPLACES the whole namespace; the route binds one + // facade import from this family, but the module it mocks is imported + // by six other handlers in the same file, so the mock must answer for + // everything the route module evaluates at load time. + const NOT_STUBBED = (name) => async () => { + throw new Error(`B4 test called ${name}, which this case did not stub`); + }; + function mockFacade(t, overrides) { + t.mock.module(absPath("engine/capability-reads.js"), { + namedExports: { readEngineCapabilityView: NOT_STUBBED("readEngineCapabilityView"), ...overrides }, + }); + } + + function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; + } + + const DECLARATION = { sessionCrud: { level: "full" } }; + const UNAVAILABLE = { none: [], partial: [] }; + const VIEW = { + declaration: DECLARATION, + unavailable: UNAVAILABLE, + provider: "local-runtime-v2", + providerFor: "transport", + engineTransport: "runtime", + agent: { version: "0.5.5", name: "mcode", title: "Mcode" }, + }; + const stub = () => ({ ...VIEW, source: "declaration", gate: {}, transport: RUNTIME }); + + test("the response key order is the endpoint's, in four `capabilities*` siblings", async (t) => { + // The four `capabilities*` keys form one group — declaration, which + // provider answered, how it was chosen, the derived roll-up — and + // `notes` stays last. A route that nested them under an `engine` + // key, or that ordered them differently, is a contract change the + // key-set assertion catches. + await setupMocks(t, { acp: {} }); + mockFacade(t, { readEngineCapabilityView: async () => stub() }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + assert.equal(res.written[0].status, 200); + const body = JSON.parse(res.written[1].body); + assert.deepEqual(Object.keys(body), [ + "ok", + "mcodeVersion", + "mcodeName", + "mcodeTitle", + "capabilities", + "capabilitiesProvider", + "capabilitiesProviderFor", + "capabilitiesUnavailable", + "notes", + ]); + }); + + test("`capabilities` IS the 14-key declaration, and the ACP wire table is gone", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { readEngineCapabilityView: async () => stub() }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + assert.equal(body.ok, true); + // The declared taxonomy replaced the flat `{method: boolean}` one. + // The old accessors are asserted ABSENT: a consumer that still read + // `capabilities.set_mode` must get `undefined` and fail loudly, not + // silently receive a truthy object field. + for (const gone of ["set_mode", "set_config_option", "cancel", "activate", "fork", "resume", "delete", "load", "close", "list", "new", "prompt"]) { + assert.equal(gone in body.capabilities, false, `capabilities.${gone} must be gone`); + } + // The four group members, each forwarded as the facade gave them. + // `deepEqual`, not identity: the body has been through + // `JSON.parse`, so reference identity is gone by construction — the + // identity assertion that actually matters (the facade forwarding + // the reviewed declaration rather than a copy) lives in section 3, + // one layer below the JSON. + assert.deepEqual(body.capabilities, DECLARATION); + assert.equal(body.capabilitiesProvider, "local-runtime-v2"); + assert.equal(body.capabilitiesProviderFor, "transport"); + assert.deepEqual(body.capabilitiesUnavailable, UNAVAILABLE); + // The `initialize` mirror is untouched by all of this. + assert.equal(body.mcodeVersion, "0.5.5"); + assert.equal(body.mcodeName, "mcode"); + assert.equal(body.mcodeTitle, "Mcode"); + // `notes` is route-owned prose about webui's own routes; the facade + // never restates it, so it is still exactly these five strings. + assert.deepEqual(Object.keys(body.notes), ["set_mode", "set_config_option", "cancel", "activate", "fork"]); + }); + + test("the declaration appears EXACTLY ONCE in the serialised body", async (t) => { + // The reason the `engine` key this batch first shipped was removed: + // with the declaration already under `capabilities`, an `engine` + // block carrying it again would put the same 14 keys in the + // response twice, and a consumer could not tell which one is the + // contract. This counts them structurally, not textually. + await setupMocks(t, { acp: {} }); + mockFacade(t, { readEngineCapabilityView: async () => stub() }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + const asJson = JSON.stringify(DECLARATION); + const carriers = Object.entries(body).filter(([, v]) => JSON.stringify(v) === asJson); + assert.deepEqual(carriers.map(([k]) => k), ["capabilities"]); + // And no nested key repeats it either: one declaration, one home. + assert.equal(JSON.stringify(body).split(asJson).length - 1, 1); + assert.equal("engine" in body, false); + }); + + test("the route adds nothing to the view and leaks none of its bookkeeping", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { readEngineCapabilityView: async () => stub() }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + // A route that re-projected either half would be a second place for + // the taxonomy to be reshaped; `deepEqual` is the strongest + // statement available after `JSON.parse`, and section 3 pins the + // reference identity one layer down. + assert.deepEqual(body.capabilities, VIEW.declaration); + assert.deepEqual(body.capabilitiesUnavailable, VIEW.unavailable); + // The facade's own bookkeeping (`source`, `gate`, the ambient + // `transport`, the provider's `engineTransport`) is diagnostic + // vocabulary, not part of this endpoint's contract. + for (const key of ["source", "gate", "engineTransport"]) { + assert.equal(key in body, false, `${key} leaked into the response`); + } + }); + + // Table-driven: [agent version, expected mcodeVersion]. The route is a + // PASS-THROUGH — including for the empty string, which the facade has + // already turned into `"unknown"` (section 3 pins that), so a route + // that applied its own `|| "unknown"` would double-apply it and a + // route that dropped the fallback entirely would ship an empty + // version. This table is the split made visible: the fallback lives + // in the engine layer, once. + const VERSION_CASES = [ + ["0.5.5", "0.5.5"], + ["unknown", "unknown"], + ["", ""], + ]; + for (const [version, expected] of VERSION_CASES) { + test(`agent.version=${JSON.stringify(version)} → mcodeVersion=${JSON.stringify(expected)}`, async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineCapabilityView: async () => ({ + ...VIEW, + agent: { version, name: null, title: null }, + source: "declaration", + gate: {}, + transport: RUNTIME, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + assert.equal(body.mcodeVersion, expected); + assert.equal(body.mcodeName, null); + assert.equal(body.mcodeTitle, null); + }); + } + + test("a facade error PROPAGATES so invokeHandler can answer 501", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineCapabilityView: async () => { + const err = new Error("fixture capability refusal"); + err.name = "EngineCapabilityNotSupportedError"; + throw err; + }, + }); + const route = await loadRoute(); + await assert.rejects(() => route.handleCapabilities(null, mkRes()), /fixture capability refusal/); + }); + + // ---- proof the mock actually took ------------------------------------ + + test("PROOF the facade mock took: a marker error escapes the untouched route", async (t) => { + await setupMocks(t, { acp: {} }); + const marker = new Error("B4-CAPABILITY-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + readEngineCapabilityView: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleCapabilities(null, mkRes()); + } catch (err) { + caught = err; + } + assert.ok(caught, "the route swallowed the facade error — either the mock did not take, or the route grew a catch"); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + + test("CONTROL: with no facade mock, the real view reaches the response", async (t) => { + // The other half of the proof: a fresh `?bust=` re-import binds the + // route to the REAL facade, so the body carries the actual + // registered declaration rather than the fixture's. + await setupMocks(t, { acp: { getMcodeServerInfo: () => AGENT_INFO } }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + assert.equal(body.capabilitiesProvider, "local-runtime-v2"); + // The real view must SAY whether it is standing in. Under the + // default `acp` transport that is `"default"`; reporting + // `"transport"` there would be the one lie this endpoint cannot + // afford, because the declaration it would attribute to a connected + // engine came from a provider that transport never chose. The + // expectation follows the ambient transport so the control holds on + // both gate legs. + const { MCODE_WEBUI_TRANSPORT } = await import(absPath("lib/config.js")); + assert.equal( + body.capabilitiesProviderFor, + MCODE_WEBUI_TRANSPORT === "runtime" ? "transport" : "default", + ); + // `setupMocks`'s acp holder is process-global and an earlier case + // left the agent mirror in it, so the version here is the real + // `initialize` mirror's, not the fixture's. + assert.equal(body.mcodeVersion, "0.5.5"); + // And the declaration served is the REAL reviewed one, key for key. + assert.deepEqual(Object.keys(body.capabilities), [...ENGINE_CAPABILITY_KEYS]); + assert.deepEqual(body.capabilities, LOCAL_RUNTIME_V2_CAPABILITIES); + assert.equal(body.capabilitiesUnavailable.none.length >= 1, true); + }); +}); diff --git a/packages/webui/test/lib/engine/model-reads.test.js b/packages/webui/test/lib/engine/model-reads.test.js new file mode 100644 index 00000000..0806defc --- /dev/null +++ b/packages/webui/test/lib/engine/model-reads.test.js @@ -0,0 +1,1181 @@ +// webui/test/lib/engine/model-reads.test.js +// +// M3-B4: the model-catalogue read's engine facade (#57). +// +// This is the batch's red line. #57 is the largest projection in webui +// and the one a refactor can damage most quietly: three sources, a +// dedupe key that has changed shape twice, two projections of one +// engine file annotating entries from two different sources, and three +// derived "what is active" figures — none of which is compared against +// anything at runtime. So the four things pinned here are: +// +// 1. THE FULL SNAPSHOT (section 5). One rich fixture — engine session +// option, engine `custom_provider` layer, webui config layer, +// builtin layer, a builtin that COLLIDES with a config entry, a +// switchable variant model, an effort-list model, a forced_on +// model, two providers with overlapping upstream model ids, a +// provider with a key and one without — projected to the exact +// response body the pre-refactor route produced. The expected +// value below was captured from the implementation at 3362c9be +// (B3's rebase tip) and pasted in longhand: it is NOT recomputed +// by the functions under test, because a snapshot whose oracle is +// the implementation proves nothing. The two `minimax_api` models +// that are ABSENT from the builtin half of the `minimax_api` +// group are the load-bearing part: the config layer took those +// slots wholesale, which is ticket 09-02's dedupe rule. +// +// 2. THE PURE PROJECTIONS ON THEIR INPUTS (sections 3–4). Each rule +// the snapshot exercises incidentally is also asserted on a +// minimal input of its own, so a failure names the RULE that broke +// rather than pointing at a 280-line diff. +// +// 3. THE VARIANT / CONTEXT PERTURBATION (section 6). The thinking +// levels and the context-window options are two projections of one +// engine file, and the interesting failure is a cross-wiring: an +// annotation attached to the wrong entry, or the builtin tree read +// twice so the two sites disagree. The test perturbs one engine +// model at a time and records exactly which entries move. +// +// 4. THE GATE IS SOFT, AND THE MOCK IS REAL. The registered provider +// declares `authCredentials` `full`, so only this file can prove +// the soft gate reports what it claims; and node:test's +// `mock.module` re-evaluates only the MOCKED specifier, so every +// route test re-imports the route under a fresh `?bust=N`, and +// section 7 ends with the control that proves the mock took. +// +// Fixture ordering is load-bearing, not stylistic. `lib/config.js` +// resolves `MCODE_WEBUI_DATA_DIR` / `MINIMAX_DATA_DIR` at MODULE LOAD, +// and `engine/model-reads.js` imports it statically — so the fixture +// directories and the env are built at module top level, BEFORE the +// first import that reaches a server module. A `before()` that set the +// env would be too late: the first import would already have frozen the +// real ~/.minimax path, and every case below would read the developer's +// own config instead of the fixture. +// +// Test style follows test/lib/engine/usage-reads.test.js (B3) and +// test/lib/engine/account-reads.test.js (B4 #20): table-driven, one row +// per case. + +import { test, describe, before, after } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; +import { setupMocks, absPath, setBuiltinModelsMock } from "../../helpers/_setup.js"; + +// --------------------------------------------------------------------------- +// Fixture — built BEFORE any server module is imported (see the header). +// +// One root, three children, one registered prefix: the engine's data +// dir (its `config.yaml` — both the `custom_provider` tree and the +// materialised `provider.minimax.models` builtin tree), the webui data +// dir (where the user-level `providers.json` would live), and the env +// layer file. The prefix is registered in +// scripts/test-tmp-leak.check.mjs#KNOWN_PREFIXES; a new prefix without +// that entry fails the test:release-tools gate. +// --------------------------------------------------------------------------- + +const root = mkTmpDir("webui-model-reads-"); +const engineDir = join(root, "engine"); +const webuiDir = join(root, "webui"); +mkdirSync(engineDir, { recursive: true }); +mkdirSync(webuiDir, { recursive: true }); + +// The engine's own config: a materialised builtin tree (one switchable +// variant model, one effort-list model, one forced_on model with +// nothing user-settable) and a `custom_provider` tree with two +// providers whose model ids OVERLAP (`z-ai/glm-5.3` is deliberately the +// kind of id that used to make one provider swallow another's entry). +writeFileSync( + join(engineDir, "config.yaml"), + `provider: + minimax: + models: + MiniMax-M3: + thinking_config: + mode: switchable + default_value: 'true' + variants: + none-thinking: { thinking: { type: disabled } } + thinking: { thinking: { type: adaptive } } + contextWindowOptions: [512000, 1000000] + contextWindowOptionHints: { "1000000": "higher_usage" } + limit: { context: 512000 } + MiniMax-M2.7: + thinking: + effortOptions: [low, medium, high] + contextWindowOptions: [128000, 256000] + limit: { context: 128000 } + MiniMax-M2.5: + thinking_config: + mode: forced_on +custom_provider: + deepseek-cn: + api: openai-completions + kind: custom + options: { apiKey: "sk-secret-should-never-leak" } + models: + deepseek-chat: + name: DeepSeek Chat + thinking: { effortOptions: [low, high] } + modalities: { input: [text, image] } + limit: { context: 64000 } + deepseek-reasoner: {} + nousresearch: + api: openai-responses + kind: custom + options: { apiKey: "" } + models: + z-ai/glm-5.3: {} + openai/gpt-5.6-sol: {} +`, + "utf8", +); + +// The webui's env layer. `minimax_api` is present ON PURPOSE: its +// `MiniMax-M3` entry collides with the builtin of the same name, so the +// `seen` dedupe has to let the config layer win wholesale — which is +// why the builtin half of that group is missing the entry, the +// `thinkingLevels`, and the `contextWindowOptions`. +const modelsConfigPath = join(root, "models.json"); +writeFileSync( + modelsConfigPath, + JSON.stringify({ + providers: [ + { + id: "minimax_api", + label: "MiniMax builtins", + auth: { type: "byok", apiKey: "sk-webui-fixture-key-0001" }, + protocol: "anthropic", + models: [ + { id: "MiniMax-M3", label: "M3 config override", contextLimit: 123456 }, + { id: "MiniMax-Text-01", thinkingLevels: ["off", "on"], modalities: ["text", "image"] }, + ], + }, + { id: "local-ollama", models: [{ id: "qwen3:8b" }] }, + ], + }), + "utf8", +); + +process.env.MINIMAX_DATA_DIR = engineDir; +delete process.env.MAVIS_DATA_DIR; +process.env.MCODE_WEBUI_DATA_DIR = webuiDir; +process.env.MCODE_WEBUI_MODELS_CONFIG = modelsConfigPath; + +/** The builtin list the mocked `getBuiltinModelsFromMcode` answers with. */ +const BUILTINS = ["MiniMax-M3", "MiniMax-M2.7", "MiniMax-M2.5", "MiniMax-M2.7-highspeed"]; + +/** + * A `model` config option shaped like the engine's, in the wire form + * `m::[:v:]` that `control-state.ts` emits for + * the builtin tree. The model segment is the BARE engine-side model key, + * which is what makes the two builtin-tree projections reachable from + * this source at all (see section 6). + */ +const MODEL_OPTION = { + type: "select", + id: "model", + name: "Model", + category: "model", + currentValue: "m:minimax_api:MiniMax-M3:v:thinking", + options: [ + { value: "m:minimax_api:MiniMax-M3:v:thinking", name: "MiniMax-M3" }, + { value: "m:minimax_api:MiniMax-M2.7:u", name: "MiniMax-M2.7" }, + ], +}; + +/** The `cs` the snapshot runs against. */ +const SNAPSHOT_CS = { + model: { name: "minimax_api/MiniMax-M3", thinking: "on", contextWindow: 1000000 }, + configOptions: [MODEL_OPTION, { id: "thinkingEffort", currentValue: "high" }], +}; + +/** The REAL wire-form parser, so the snapshot exercises the real parse. */ +const { parseEngineModelWireValue, readEngineBuiltinThinking, readEngineBuiltinContextWindows } = + await import("../../../server/lib/engine-catalogue.js"); +// `capabilities.js` directly, NOT `engine/index.js`: the facade re-exports +// `model-reads.js`, so importing it at module scope would evaluate the +// module under test — and its STATIC import of `lib/models.js` — BEFORE +// `before()` registers the builtin-catalogue mock, and the snapshot +// would then read whatever `mcode` bundle the host has installed. +const { ENGINE_CAPABILITY_KEYS } = await import("../../../server/engine/capabilities.js"); + +// --- now, and only now, the modules under test --------------------------- +let engine; +let modelRouteBaseline; +before(async (t) => { + // `setupMocks` must precede the SUT import: `engine/model-reads.js` + // imports `lib/models.js` STATICALLY, and the builtin catalogue must + // come from the mock rather than from whatever `mcode` bundle happens + // to be installed on the host. + await setupMocks(t, { acp: {} }); + setBuiltinModelsMock(BUILTINS); + engine = await import(absPath("engine/model-reads.js")); + modelRouteBaseline = await import(absPath("routes/model.js")); +}); + +after(() => { + rmTmpDir(root); + delete process.env.MINIMAX_DATA_DIR; + delete process.env.MAVIS_DATA_DIR; + delete process.env.MCODE_WEBUI_DATA_DIR; + delete process.env.MCODE_WEBUI_MODELS_CONFIG; +}); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +// --------------------------------------------------------------------------- +// 1. The endpoint → capability declaration table +// --------------------------------------------------------------------------- + +describe("MODEL_READ_ENDPOINTS — this batch's declaration table", () => { + test("covers exactly the one endpoint of the model family", () => { + assert.deepEqual(Object.keys(engine.MODEL_READ_ENDPOINTS), ["GET /api/models"]); + }); + + test("GET /api/models declares authCredentials.listModelProviders, enforced SOFT", () => { + // Table-driven: editing this row is a capability decision and must be + // reviewed as one, so the table IS the assertion. + const row = { + capability: "authCredentials", + subItem: "listModelProviders", + enforcement: "soft", + }; + assert.deepEqual(engine.MODEL_READ_ENDPOINTS["GET /api/models"], row); + assert.ok(ENGINE_CAPABILITY_KEYS.includes(row.capability)); + }); + + test("it shares the capability KEY with the account family, and differs in the other two fields", async () => { + // Both families ride `authCredentials` because the 14 matrix keys + // have no separate "models" row — the engine's model/provider + // surface is declared there. What differs is the sub-item and the + // enforcement, and both differences are asserted rather than + // assumed: a models read gated on `getAccountStatus` would let a + // provider that cannot report a plan still be trusted for a + // catalogue, and vice versa. + const { ACCOUNT_READ_ENDPOINTS } = await import("../../../server/engine/account-reads.js"); + assert.equal( + engine.MODEL_READ_ENDPOINTS["GET /api/models"].capability, + ACCOUNT_READ_ENDPOINTS["GET /api/account"].capability, + ); + assert.notEqual( + engine.MODEL_READ_ENDPOINTS["GET /api/models"].subItem, + ACCOUNT_READ_ENDPOINTS["GET /api/account"].subItem, + ); + assert.equal(ACCOUNT_READ_ENDPOINTS["GET /api/account"].enforcement, undefined); + }); +}); + +// --------------------------------------------------------------------------- +// 2. Provider resolution + the SOFT gate +// --------------------------------------------------------------------------- + +describe("resolveModelReadProvider / checkModelReadCapability", () => { + // Table-driven. The gate values are `session-export.js`'s vocabulary, + // reused rather than re-invented. + const TRANSPORTS = [ + [RUNTIME, true, "checked", "local-runtime-v2"], + [ACP, false, "unregistered-transport", null], + ["exec", false, "unregistered-transport", null], + ["", false, "unregistered-transport", null], + ]; + for (const [transport, hasProvider, gate, providerId] of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → ${gate}`, () => { + const provider = engine.resolveModelReadProvider(transport); + assert.equal(!!provider, hasProvider); + const g = engine.checkModelReadCapability("GET /api/models", transport); + assert.equal(g.gate, gate); + assert.equal(g.provider, providerId); + assert.equal(g.capability, "authCredentials"); + assert.equal(g.subItem, "listModelProviders"); + assert.equal(g.enforcement, "soft"); + }); + } + + test("the gate NEVER throws, under any transport or endpoint key", () => { + // The whole reason this family's gate is soft: the catalogue's + // primary sources are files webui owns. A provider that declared no + // model surface would still leave a working picker, so a hard gate + // here would REMOVE working functionality — the #11 reasoning, + // reused. + for (const transport of [RUNTIME, ACP, "exec", "", "nonsense"]) { + assert.doesNotThrow(() => engine.checkModelReadCapability("GET /api/models", transport)); + } + }); + + test("an unknown endpoint key is a plain Error, not 501 material", () => { + assert.throws( + () => engine.checkModelReadCapability("GET /api/nope", RUNTIME), + (err) => { + assert.equal(err.code, "unknown_model_read_endpoint"); + assert.match(err.message, /not part of the model family/); + return true; + }, + ); + }); +}); + +// --------------------------------------------------------------------------- +// 3. The two id helpers +// --------------------------------------------------------------------------- + +describe("providerOfModelId / webuiFullModelId", () => { + // Table-driven: [modelId, fallback, expected]. The bare-id fallback to + // `minimax_api` is what keeps a user-typed short id out of a phantom + // group; the `i <= 0` guard is what keeps a leading `/` from + // producing an empty provider key. + const PROVIDER_CASES = [ + ["minimax_api/MiniMax-M3", "minimax_api", "minimax_api"], + ["nousresearch/deepseek/x", "minimax_api", "nousresearch"], + ["/leading-slash", "minimax_api", "minimax_api"], + ["MiniMax-M3", "minimax_api", "minimax_api"], + ["", "minimax_api", "minimax_api"], + [null, "minimax_api", "minimax_api"], + [undefined, "minimax_api", "minimax_api"], + // An explicit fallback is honoured for a bare id and for an empty + // one — the engine builtin provider is a DEFAULT, not a constant. + ["minimax_api/MiniMax-M3", "fallback-provider", "minimax_api"], + ["", "fallback-provider", "fallback-provider"], + [null, "fallback-provider", "fallback-provider"], + ]; + for (const [modelId, fallback, expected] of PROVIDER_CASES) { + test(`providerOfModelId(${JSON.stringify(modelId)}, ${JSON.stringify(fallback)}) → ${expected}`, () => { + assert.equal(engine.providerOfModelId(modelId, fallback), expected); + }); + } + + // The webui id is ALWAYS two segments, even when the upstream model + // id already contains `/`. That is ticket 09-02: skipping the prefix + // put the picker in the wrong group and let overlapping upstream ids + // collide on the dedupe. + const ID_CASES = [ + ["minimax_api", "MiniMax-M3", "minimax_api/MiniMax-M3"], + ["nousresearch", "z-ai/glm-5.3", "nousresearch/z-ai/glm-5.3"], + ["minimax_api", "MiniMax-M2.7-highspeed", "minimax_api/MiniMax-M2.7-highspeed"], + ]; + for (const [providerKey, modelId, expected] of ID_CASES) { + test(`webuiFullModelId(${providerKey}, ${modelId})`, () => { + assert.equal(engine.webuiFullModelId(providerKey, modelId), expected); + }); + } +}); + +describe("attachContextWindowOptions", () => { + // Table-driven. Each row is a rule with a failure mode: a missing + // projection must leave the entry field-free (so the composer mounts + // no control), an existing `contextLimit` must NOT be overwritten (a + // config layer's value wins), and both the array and the hints object + // must be COPIED so a caller mutating the entry cannot corrupt the + // engine projection for the next entry. + const entry = () => ({ id: "x", label: "x" }); + const CASES = [ + ["no projection at all", null, {}, null], + ["options only", { options: [1, 2] }, {}, { contextWindowOptions: [1, 2] }], + [ + "options + currentLimit, entry has no limit", + { options: [1, 2], currentLimit: 9 }, + {}, + { contextWindowOptions: [1, 2], contextLimit: 9 }, + ], + [ + "options + currentLimit, entry KEEPS its own limit", + { options: [1, 2], currentLimit: 9 }, + { contextLimit: 5 }, + { contextWindowOptions: [1, 2] }, + ], + [ + "hints ride along only when present", + { options: [1, 2], hints: { 2: "higher_usage" } }, + {}, + { contextWindowOptions: [1, 2], contextWindowOptionHints: { 2: "higher_usage" } }, + ], + [ + "currentLimit of 0 is still attached (the projection decided)", + { options: [1], currentLimit: 0 }, + {}, + { contextWindowOptions: [1], contextLimit: 0 }, + ], + ]; + for (const [name, projection, pre, expected] of CASES) { + test(name, () => { + const e = { ...entry(), ...pre }; + engine.attachContextWindowOptions(e, projection); + assert.deepEqual(e, { ...entry(), ...pre, ...expected }); + }); + } + + test("the array and the hints are copies, not aliases of the projection", () => { + const projection = { options: [1, 2], hints: { 1: "higher_usage" } }; + const e = {}; + engine.attachContextWindowOptions(e, projection); + e.contextWindowOptions.push(3); + e.contextWindowOptionHints[1] = "tampered"; + assert.deepEqual(projection.options, [1, 2]); + assert.deepEqual(projection.hints, { 1: "higher_usage" }); + }); +}); + +// --------------------------------------------------------------------------- +// 4. The projections, rule by rule +// --------------------------------------------------------------------------- + +const THINKING_M3 = new Map([["MiniMax-M3", { levels: ["off", "on"] }]]); +const WINDOWS_M3 = new Map([ + ["MiniMax-M3", { options: [512000, 1000000], hints: { 1000000: "higher_usage" }, currentLimit: 512000 }], +]); + +describe("projectModelCatalogue — grouping, dedupe and the empty shell", () => { + // Table-driven: [name, options, expected]. `list` and `groups` are the + // ordered id lists, written out longhand rather than recomputed. + const wire = parseEngineModelWireValue; + const CASES = [ + [ + "no sources at all: the empty builtin shell is dropped, list is empty", + { providers: null, builtins: [], sessionOption: null }, + { list: [], groups: [] }, + ], + [ + "a providers config with no models still emits its (empty) group", + { providers: { providers: [{ id: "p", label: "P", models: [] }] }, builtins: [] }, + { + list: [], + groups: ["p", "minimax_api"], + group: { id: "p", label: "P", auth: { hasKey: false, type: "byok" }, protocol: "openai", models: [] }, + }, + ], + [ + "a providers config with no models KEEPS the empty builtin shell next to it", + { providers: { providers: [{ id: "p", models: [] }] }, builtins: ["MiniMax-M3"] }, + { list: ["minimax_api/MiniMax-M3"], groups: ["p", "minimax_api"] }, + ], + [ + "a builtin is attributed to minimax_api, and only to it", + { providers: null, builtins: ["MiniMax-M3"] }, + { list: ["minimax_api/MiniMax-M3"], groups: ["minimax_api"] }, + ], + [ + "a config entry COLLIDING with a builtin wins wholesale", + { + providers: { providers: [{ id: "minimax_api", models: [{ id: "MiniMax-M3", label: "override" }] }] }, + builtins: ["MiniMax-M3"], + }, + { list: ["minimax_api/MiniMax-M3"], groups: ["minimax_api"], labels: { "minimax_api/MiniMax-M3": "override" } }, + ], + [ + "two providers with the same upstream model id stay distinct", + { + providers: { + providers: [ + { id: "nousresearch", models: [{ id: "z-ai/glm-5.3" }] }, + { id: "zai-max", models: [{ id: "z-ai/glm-5.3" }] }, + ], + }, + builtins: [], + }, + { list: ["nousresearch/z-ai/glm-5.3", "zai-max/z-ai/glm-5.3"], groups: ["nousresearch", "zai-max", "minimax_api"] }, + ], + [ + "an engine option with an empty value list yields NO group", + { sessionOption: { options: [] }, providers: null, builtins: [] }, + { list: [], groups: [] }, + ], + [ + "entries with no usable value are skipped; a duplicate value is deduped", + { + sessionOption: { options: [{ value: "a" }, { value: null }, null, { value: "a" }] }, + providers: null, + builtins: [], + }, + { list: ["a"], groups: ["__engine"] }, + ], + ]; + for (const [name, options, expected] of CASES) { + test(name, () => { + const { list, groups } = engine.projectModelCatalogue({ + ...options, + parseEngineModelWireValue: wire, + }); + assert.deepEqual(list.map((e) => e.id), expected.list, "flat list"); + assert.deepEqual(groups.map((g) => g.id), expected.groups, "group ids"); + if (expected.group) { + assert.deepEqual(groups.find((g) => g.id === expected.group.id), expected.group); + } + if (expected.labels) { + for (const [id, label] of Object.entries(expected.labels)) { + assert.equal(list.find((e) => e.id === id).label, label); + } + } + }); + } + + test("a builtin already present from the config layer is deduped, not appended twice", () => { + // The `seen` set is per `(providerKey, modelId)`, and it is what keeps + // the picker from showing `MiniMax-M3` twice when the operator has + // configured the same builtin id. Removing the check on the builtin + // side would duplicate the row in BOTH the flat list and the group. + const { list, groups } = engine.projectModelCatalogue({ + providers: { providers: [{ id: "minimax_api", models: [{ id: "MiniMax-M3", label: "config" }] }] }, + builtins: ["MiniMax-M3", "MiniMax-M3"], + parseEngineModelWireValue: wire, + }); + assert.deepEqual(list.map((e) => e.id), ["minimax_api/MiniMax-M3"]); + assert.deepEqual(groups.find((g) => g.id === "minimax_api").models.map((e) => e.id), [ + "minimax_api/MiniMax-M3", + ]); + // And the surviving entry is the CONFIG one — the operator's layer + // wins wholesale, it does not merge with the builtin. + assert.equal(list[0].source, "config"); + assert.equal(list[0].label, "config"); + assert.equal("thinkingLevels" in list[0], false); + }); + + test("the engine group id and label are the endpoint's, not the provider's", () => { + const { groups } = engine.projectModelCatalogue({ + sessionOption: MODEL_OPTION, + providers: null, + builtins: [], + parseEngineModelWireValue: wire, + }); + assert.equal(groups.length, 1); + assert.equal(groups[0].id, "__engine"); + assert.equal(groups[0].label, "Engine session"); + // The engine group carries NO auth block — a session option is not + // a provider the operator configured. + assert.equal("auth" in groups[0], false); + assert.equal("protocol" in groups[0], false); + }); + + test("engine-session entries carry BOTH `name` and `label`, and the same value", () => { + // Pre-existing callers (the composer chip) read `name`; the + // provider-grouped panel reads `label`. Dropping either is a + // frontend break that a single-key test would miss. + const { list } = engine.projectModelCatalogue({ + sessionOption: MODEL_OPTION, + providers: null, + builtins: [], + parseEngineModelWireValue: wire, + }); + for (const e of list) { + assert.equal(e.name, e.label); + assert.equal(typeof e.name, "string"); + } + // And an option with no `name` falls back to the value itself. + const { list: l2 } = engine.projectModelCatalogue({ + sessionOption: { options: [{ value: "m:x:y:u" }] }, + providers: null, + builtins: [], + parseEngineModelWireValue: () => null, + }); + assert.equal(l2[0].name, "m:x:y:u"); + assert.equal(l2[0].label, "m:x:y:u"); + }); + + test("the config group reports hasKey from EITHER signal, and never a key", () => { + // The security contract: the group reports whether a key is + // configured, never the key. Both signals mean "configurable from + // the picker" — a webui-side plaintext apiKey and an engine-side + // boolean alike. + const { groups } = engine.projectModelCatalogue({ + providers: { + providers: [ + { id: "webui-key", auth: { apiKey: "sk-secret" }, models: [] }, + { id: "engine-key", auth: { hasKey: true }, models: [] }, + { id: "no-key", auth: { hasKey: false }, models: [] }, + { id: "no-auth", models: [] }, + ], + }, + builtins: [], + parseEngineModelWireValue: wire, + }); + // The empty `minimax_api` shell is also a group and has no `auth`, + // so the comparison is over the four CONFIG groups. + const byId = Object.fromEntries(groups.filter((g) => g.auth).map((g) => [g.id, g.auth])); + assert.deepEqual(byId, { + "webui-key": { hasKey: true, type: "byok" }, + "engine-key": { hasKey: true, type: "byok" }, + "no-key": { hasKey: false, type: "byok" }, + "no-auth": { hasKey: false, type: "byok" }, + }); + // And the secret itself is nowhere in the group. + assert.equal(JSON.stringify(groups).includes("sk-secret"), false); + }); + + test("contextLimit rides along only for a positive number", () => { + const { list } = engine.projectModelCatalogue({ + providers: { + providers: [ + { id: "p", models: [{ id: "a", contextLimit: 1000 }, { id: "b", contextLimit: 0 }, { id: "c", contextLimit: -5 }] }, + ], + }, + builtins: [], + parseEngineModelWireValue: wire, + }); + assert.equal("contextLimit" in list.find((e) => e.id === "p/a"), true); + assert.equal("contextLimit" in list.find((e) => e.id === "p/b"), false); + assert.equal("contextLimit" in list.find((e) => e.id === "p/c"), false); + }); + + test("empty thinkingLevels / modalities are omitted, not sent as []", () => { + // An empty array would make a consumer mount a control with no + // choices; the endpoint has always omitted the key. + const { list } = engine.projectModelCatalogue({ + providers: { + providers: [{ id: "p", models: [{ id: "a", thinkingLevels: [], modalities: [] }, { id: "b", thinkingLevels: ["x"], modalities: ["text"] }] }], + }, + builtins: [], + parseEngineModelWireValue: wire, + }); + assert.deepEqual(Object.keys(list[0]), ["id", "label", "provider", "source"]); + assert.deepEqual(list[1].thinkingLevels, ["x"]); + assert.deepEqual(list[1].modalities, ["text"]); + }); +}); + +describe("deriveModelSelection — the three derived figures", () => { + // Table-driven. Each row is a resolution rule, including the two + // "never invent" rules (a null `current`, a null window) that the + // composer depends on to render a neutral chip. + const entry = (id, contextLimit) => (contextLimit === undefined ? { id } : { id, contextLimit }); + const CASES = [ + [ + "the engine's currentValue wins over the recorded name", + { sessionOption: { currentValue: "wire" }, cs: { model: { name: "recorded" } } }, + { current: "wire", currentThinking: null, currentContextWindow: null }, + ], + [ + "the recorded name is the fallback, and null when there is none", + { sessionOption: null, cs: { model: { name: "recorded" } } }, + { current: "recorded", currentThinking: null, currentContextWindow: null }, + ], + [ + "no engine value and no record → null, never a default model", + { sessionOption: null, cs: {} }, + { current: null, currentThinking: null, currentContextWindow: null }, + ], + [ + "the engine's thinkingEffort wins over the recorded level", + { cs: { configOptions: [{ id: "thinkingEffort", currentValue: "high" }], model: { thinking: "low" } } }, + { currentThinking: "high" }, + ], + [ + "the recorded level is the fallback", + { cs: { model: { thinking: "low" } } }, + { currentThinking: "low" }, + ], + [ + "a non-string recorded level is ignored, not coerced", + { cs: { model: { thinking: 7 } } }, + { currentThinking: null }, + ], + [ + "a non-string engine level falls through to the record", + { cs: { configOptions: [{ id: "thinkingEffort", currentValue: 7 }], model: { thinking: "low" } } }, + { currentThinking: "low" }, + ], + [ + "the recorded window wins over the catalogue limit", + { cs: { model: { name: "m", contextWindow: 1000000 } }, list: [entry("m", 512000)] }, + { currentContextWindow: 1000000 }, + ], + [ + "the current model's catalogue limit is the fallback", + { cs: { model: { name: "m" } }, list: [entry("m", 512000)] }, + { currentContextWindow: 512000 }, + ], + [ + "a recorded window is reported even when the model no longer advertises it", + { cs: { model: { name: "m", contextWindow: 1000000 } }, list: [entry("m")] }, + { currentContextWindow: 1000000 }, + ], + [ + "a non-positive or non-integer recorded window is not a window", + { cs: { model: { name: "m", contextWindow: 0 } }, list: [entry("m", 512000)] }, + { currentContextWindow: 512000 }, + ], + [ + "a model with no limit and no record → null", + { cs: { model: { name: "m" } }, list: [entry("m")] }, + { currentContextWindow: null }, + ], + ]; + for (const [name, options, expected] of CASES) { + test(name, () => { + const got = engine.deriveModelSelection({ list: [], ...options }); + for (const [k, v] of Object.entries(expected)) assert.equal(got[k], v, k); + }); + } + + test("the result has exactly the three figures, in order", () => { + assert.deepEqual( + Object.keys(engine.deriveModelSelection({ cs: {}, list: [] })), + ["current", "currentThinking", "currentContextWindow"], + ); + }); +}); + +describe("catalogueSourceLabel", () => { + // Table-driven. The label is the endpoint's answer to "which layer + // won", and it keys off the OPTION LIST's length, not off the + // option's existence — an engine that advertises the option with no + // choices has not contributed anything. + const CASES = [ + [{ sessionOption: { options: [{ value: "a" }] }, providers: null }, "acp-session-config"], + [{ sessionOption: { options: [] }, providers: null }, "mcode-cli-bundle"], + [{ sessionOption: null, providers: { providers: [] } }, "config+mcode-cli-bundle"], + [{ sessionOption: null, providers: null }, "mcode-cli-bundle"], + [{}, "mcode-cli-bundle"], + ]; + for (const [options, expected] of CASES) { + test(`${JSON.stringify(options).slice(0, 60)} → ${expected}`, () => { + assert.equal(engine.catalogueSourceLabel(options), expected); + }); + } +}); + +// --------------------------------------------------------------------------- +// 5. THE FULL SNAPSHOT — the red line +// --------------------------------------------------------------------------- + +describe("readEngineModelCatalogue — the full projection, end to end", () => { + /** + * The response body the PRE-refactor `routes/model.js#handleGetModels` + * produced for the fixture above, captured from the implementation at + * 3362c9be and pasted in longhand. Not recomputed by the functions + * under test. + * + * Read it as the batch's contract, in this order: + * + * - 2 engine-session entries FIRST, under `__engine`, ids kept in + * the engine's wire form so `POST /api/set-model` round-trips — + * and BOTH annotated from the builtin tree, because the wire + * form's model segment is the bare engine model key. This is the + * second of the two annotation sites, and the only place the + * snapshot shows both of them at once. + * - 2 providers projected from the engine's `custom_provider` tree + * (`deepseek-cn`, `nousresearch`), each with `auth.hasKey` + * answering the engine's own `options.apiKey` (true / false) and + * `protocol` mapped from the engine's `api` (openai / openai). + * - 3 webui config entries, including `minimax_api/MiniMax-M3` with + * the operator's label and `contextLimit` — the entry that TOOK + * the builtin's slot, which is why the builtin `MiniMax-M3` is + * absent below and carries no `thinkingLevels` and no + * `contextWindowOptions`. + * - 3 surviving builtins: the effort-list model with its context + * windows, and the two bare ones. `MiniMax-M2.5` is the + * forced_on model — the engine's tree has nothing user-settable, + * so the entry stays field-free and the composer mounts no + * control. + * - The three derived figures, and the `source` label. + */ +const EXPECTED = { + ok: true, + models: [ + {"id": "m:minimax_api:MiniMax-M3:v:thinking", "name": "MiniMax-M3", "label": "MiniMax-M3", "provider": "minimax_api", "source": "engine", "thinkingLevels": ["off", "on"], "contextWindowOptions": [512000, 1000000], "contextWindowOptionHints": {"1000000": "higher_usage"}, "contextLimit": 512000}, + {"id": "m:minimax_api:MiniMax-M2.7:u", "name": "MiniMax-M2.7", "label": "MiniMax-M2.7", "provider": "minimax_api", "source": "engine", "thinkingLevels": ["low", "medium", "high"], "contextWindowOptions": [128000, 256000], "contextLimit": 128000}, + {"id": "deepseek-cn/deepseek-chat", "label": "DeepSeek Chat", "provider": "deepseek-cn", "source": "config", "contextLimit": 64000, "protocol": "openai", "thinkingLevels": ["low", "high"], "modalities": ["text", "image"]}, + {"id": "deepseek-cn/deepseek-reasoner", "label": "deepseek-reasoner", "provider": "deepseek-cn", "source": "config", "protocol": "openai"}, + {"id": "nousresearch/z-ai/glm-5.3", "label": "z-ai/glm-5.3", "provider": "nousresearch", "source": "config", "protocol": "openai"}, + {"id": "nousresearch/openai/gpt-5.6-sol", "label": "openai/gpt-5.6-sol", "provider": "nousresearch", "source": "config", "protocol": "openai"}, + {"id": "minimax_api/MiniMax-M3", "label": "M3 config override", "provider": "minimax_api", "source": "config", "contextLimit": 123456, "protocol": "anthropic"}, + {"id": "minimax_api/MiniMax-Text-01", "label": "MiniMax-Text-01", "provider": "minimax_api", "source": "config", "protocol": "anthropic", "thinkingLevels": ["off", "on"], "modalities": ["text", "image"]}, + {"id": "local-ollama/qwen3:8b", "label": "qwen3:8b", "provider": "local-ollama", "source": "config", "protocol": "openai"}, + {"id": "minimax_api/MiniMax-M2.7", "label": "MiniMax-M2.7", "provider": "minimax_api", "source": "builtin", "thinkingLevels": ["low", "medium", "high"], "contextWindowOptions": [128000, 256000], "contextLimit": 128000}, + {"id": "minimax_api/MiniMax-M2.5", "label": "MiniMax-M2.5", "provider": "minimax_api", "source": "builtin"}, + {"id": "minimax_api/MiniMax-M2.7-highspeed", "label": "MiniMax-M2.7-highspeed", "provider": "minimax_api", "source": "builtin"}, + ], + groups: [ + { ...{"id": "__engine", "label": "Engine session"}, models: [ + {"id": "m:minimax_api:MiniMax-M3:v:thinking", "name": "MiniMax-M3", "label": "MiniMax-M3", "provider": "minimax_api", "source": "engine", "thinkingLevels": ["off", "on"], "contextWindowOptions": [512000, 1000000], "contextWindowOptionHints": {"1000000": "higher_usage"}, "contextLimit": 512000}, + {"id": "m:minimax_api:MiniMax-M2.7:u", "name": "MiniMax-M2.7", "label": "MiniMax-M2.7", "provider": "minimax_api", "source": "engine", "thinkingLevels": ["low", "medium", "high"], "contextWindowOptions": [128000, 256000], "contextLimit": 128000}, + ] }, + { ...{"id": "deepseek-cn", "label": "deepseek-cn", "auth": {"hasKey": true, "type": "byok"}, "protocol": "openai"}, models: [ + {"id": "deepseek-cn/deepseek-chat", "label": "DeepSeek Chat", "provider": "deepseek-cn", "source": "config", "contextLimit": 64000, "protocol": "openai", "thinkingLevels": ["low", "high"], "modalities": ["text", "image"]}, + {"id": "deepseek-cn/deepseek-reasoner", "label": "deepseek-reasoner", "provider": "deepseek-cn", "source": "config", "protocol": "openai"}, + ] }, + { ...{"id": "nousresearch", "label": "nousresearch", "auth": {"hasKey": false, "type": "byok"}, "protocol": "openai"}, models: [ + {"id": "nousresearch/z-ai/glm-5.3", "label": "z-ai/glm-5.3", "provider": "nousresearch", "source": "config", "protocol": "openai"}, + {"id": "nousresearch/openai/gpt-5.6-sol", "label": "openai/gpt-5.6-sol", "provider": "nousresearch", "source": "config", "protocol": "openai"}, + ] }, + { ...{"id": "minimax_api", "label": "MiniMax builtins", "auth": {"hasKey": true, "type": "byok"}, "protocol": "anthropic"}, models: [ + {"id": "minimax_api/MiniMax-M3", "label": "M3 config override", "provider": "minimax_api", "source": "config", "contextLimit": 123456, "protocol": "anthropic"}, + {"id": "minimax_api/MiniMax-Text-01", "label": "MiniMax-Text-01", "provider": "minimax_api", "source": "config", "protocol": "anthropic", "thinkingLevels": ["off", "on"], "modalities": ["text", "image"]}, + {"id": "minimax_api/MiniMax-M2.7", "label": "MiniMax-M2.7", "provider": "minimax_api", "source": "builtin", "thinkingLevels": ["low", "medium", "high"], "contextWindowOptions": [128000, 256000], "contextLimit": 128000}, + {"id": "minimax_api/MiniMax-M2.5", "label": "MiniMax-M2.5", "provider": "minimax_api", "source": "builtin"}, + {"id": "minimax_api/MiniMax-M2.7-highspeed", "label": "MiniMax-M2.7-highspeed", "provider": "minimax_api", "source": "builtin"}, + ] }, + { ...{"id": "local-ollama", "label": "local-ollama", "auth": {"hasKey": false, "type": "byok"}, "protocol": "openai"}, models: [ + {"id": "local-ollama/qwen3:8b", "label": "qwen3:8b", "provider": "local-ollama", "source": "config", "protocol": "openai"}, + ] }, + ], + current: "m:minimax_api:MiniMax-M3:v:thinking", + currentThinking: "high", + currentContextWindow: 1000000, + source: "acp-session-config", + }; + + test("the payload is the pre-refactor body, field for field and key for key", () => { + const read = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + assert.equal(read.source, "config"); + assert.equal(read.gate.gate, "checked"); + assert.deepEqual(read.payload, EXPECTED); + }); + + test("the payload's key order is the endpoint's", () => { + // A key-set check alone lets a body that carries the right fields + // in a different order pass; JSON key order is what a snapshot + // diff and a careless consumer both depend on. + const read = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + assert.deepEqual(Object.keys(read.payload), [ + "ok", + "models", + "groups", + "current", + "currentThinking", + "currentContextWindow", + "source", + ]); + }); + + test("the projection is DETERMINISTIC — two reads are deep-equal", () => { + // Every source is re-read per call, so a read that leaked state + // between calls (a shared `seen` set, a mutated projection) would + // show up here and nowhere else. + const a = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + const b = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + assert.deepEqual(a.payload, b.payload); + }); + + test("the `minimax_api` group is the builtins' group, and the config entry took the slot", () => { + // The two facts red line five is really about: grouping is BY + // PROVIDER, and the dedupe is per provider, so an operator's + // override of a builtin id does not leave two `MiniMax-M3` rows in + // the picker. + const { payload } = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + const group = payload.groups.find((g) => g.id === "minimax_api"); + const ids = group.models.map((m) => m.id); + assert.deepEqual(ids, [ + "minimax_api/MiniMax-M3", + "minimax_api/MiniMax-Text-01", + "minimax_api/MiniMax-M2.7", + "minimax_api/MiniMax-M2.5", + "minimax_api/MiniMax-M2.7-highspeed", + ]); + assert.equal(new Set(ids).size, ids.length, "no id may appear twice in a group"); + // Every group holds the SAME entry objects as the flat list — a + // second copy would let the picker and the chip disagree. + for (const g of payload.groups) { + for (const m of g.models) { + assert.equal(payload.models.includes(m), true, `${m.id} is not the same object as the flat entry`); + } + } + }); + + test("no apiKey ever reaches the payload", () => { + // The security contract, end to end: the engine stores its key in + // plaintext and the webui stores one too, and neither may travel. + const { payload } = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + const serialised = JSON.stringify(payload); + assert.equal(serialised.includes("sk-secret-should-never-leak"), false); + assert.equal(serialised.includes("sk-webui-fixture-key-0001"), false); + assert.equal(serialised.includes("apiKey"), false); + }); + + test("an empty catalogue answers the soft marker LAST, not an error", () => { + // The endpoint's long-standing hint: with no engine tree, no + // providers config and no builtins, the picker renders "nothing + // attached" and the caller still gets `ok:true` — plus `reason`, + // which is spread AFTER `source` so a consumer reading the body + // positionally sees the same order as on a populated catalogue. + // + // Every source is re-read per call, so pointing the three env vars + // at empty directories for the duration of ONE call is enough; no + // module reload and no test-ordering constraint. + const emptyEngine = mkTmpDir("webui-model-reads-", { parent: root }); + const emptyWebui = mkTmpDir("webui-model-reads-", { parent: root }); + const prev = { + engine: process.env.MINIMAX_DATA_DIR, + webui: process.env.MCODE_WEBUI_DATA_DIR, + config: process.env.MCODE_WEBUI_MODELS_CONFIG, + }; + process.env.MINIMAX_DATA_DIR = emptyEngine; + process.env.MCODE_WEBUI_DATA_DIR = emptyWebui; + process.env.MCODE_WEBUI_MODELS_CONFIG = join(emptyWebui, "absent.json"); + try { + setBuiltinModelsMock([]); + const read = engine.readEngineModelCatalogue({ cs: {}, transport: RUNTIME }); + assert.deepEqual(read.payload, { + ok: true, + models: [], + groups: [], + current: null, + currentThinking: null, + currentContextWindow: null, + source: "mcode-cli-bundle", + reason: "no_catalogue", + }); + assert.deepEqual(Object.keys(read.payload), [ + "ok", + "models", + "groups", + "current", + "currentThinking", + "currentContextWindow", + "source", + "reason", + ]); + // And the soft gate still reports — a soft gate is not a missing + // gate. + assert.equal(read.gate.gate, "checked"); + } finally { + process.env.MINIMAX_DATA_DIR = prev.engine; + process.env.MCODE_WEBUI_DATA_DIR = prev.webui; + process.env.MCODE_WEBUI_MODELS_CONFIG = prev.config; + setBuiltinModelsMock(BUILTINS); + } + }); +}); + +// --------------------------------------------------------------------------- +// 6. Variant / context perturbation — which input moves which annotation +// --------------------------------------------------------------------------- + +describe("the variant and context projections are two views of ONE engine read", () => { + // The engine tree is read twice per request — once for thinking, once + // for context windows — and both are consumed at two sites (the + // engine-session entries and the builtin shell). The failure this + // section exists for is a CROSS-WIRING: one annotation attached to the + // wrong entry, or the two sites disagreeing about the same model. + test("the two real readers agree on the set of models they know", () => { + const thinking = readEngineBuiltinThinking(); + const windows = readEngineBuiltinContextWindows(); + // Same keys, same order — the two readers project the same record. + assert.deepEqual([...thinking.keys()], [...windows.keys()]); + assert.deepEqual([...thinking.keys()], ["MiniMax-M3", "MiniMax-M2.7", "MiniMax-M2.5"]); + }); + + // Table-driven: [model, expected thinkingLevels-or-undefined, + // expected contextWindowOptions-or-undefined, expected contextLimit-or-undefined]. + // Each row is one engine record; the projection must attach EXACTLY + // what that record says, and a record with nothing user-settable + // (the forced_on `MiniMax-M2.5`) must stay field-free. + const TABLE = [ + ["MiniMax-M3", ["off", "on"], [512000, 1000000], 512000], + ["MiniMax-M2.7", ["low", "medium", "high"], [128000, 256000], 128000], + ["MiniMax-M2.5", undefined, undefined, undefined], + ["not-in-the-tree", undefined, undefined, undefined], + ]; + for (const [model, levels, options, limit] of TABLE) { + test(`${model}: levels=${JSON.stringify(levels)} windows=${JSON.stringify(options)}`, () => { + const { list } = engine.projectModelCatalogue({ + providers: null, + builtins: [model], + builtinThinking: readEngineBuiltinThinking(), + builtinContextWindows: readEngineBuiltinContextWindows(), + parseEngineModelWireValue, + }); + const entry = list[0]; + if (levels === undefined) assert.equal("thinkingLevels" in entry, false); + else assert.deepEqual(entry.thinkingLevels, levels); + if (options === undefined) assert.equal("contextWindowOptions" in entry, false); + else assert.deepEqual(entry.contextWindowOptions, options); + if (limit === undefined) assert.equal("contextLimit" in entry, false); + else assert.equal(entry.contextLimit, limit); + }); + } + + test("the same annotations reach the ENGINE-SESSION site, keyed by the wire form's model id", () => { + // The two annotation sites exist because the engine's ACP `model` + // option advertises wire ids, not bare ids. If the lookup used the + // wire VALUE instead of the parsed model id, a cross-client model + // change would silently lose the composer's controls. + // A wire form whose MODEL SEGMENT is the bare builtin id — which is + // what the engine emits for `provider.minimax.models` entries. A + // wire form whose model segment is itself prefixed (or one that + // does not parse at all) misses the builtin tree, and the entry + // stays field-free; that is a miss, not a crash. + const { list } = engine.projectModelCatalogue({ + sessionOption: { options: [{ value: "m:minimax_api:MiniMax-M3:v:thinking", name: "MiniMax-M3" }, { value: "m:minimax_api:MiniMax-M2.7:u", name: "MiniMax-M2.7" }] }, + providers: null, + builtins: [], + builtinThinking: THINKING_M3, + builtinContextWindows: WINDOWS_M3, + parseEngineModelWireValue, + }); + const m3 = list.find((e) => e.id.includes("MiniMax-M3")); + assert.deepEqual(m3.thinkingLevels, ["off", "on"]); + assert.deepEqual(m3.contextWindowOptions, [512000, 1000000]); + assert.deepEqual(m3.contextWindowOptionHints, { 1000000: "higher_usage" }); + // A model that is NOT in the tree gets nothing: the BARE id is + // looked up, and a miss is a miss rather than a partial annotation. + const m27 = list.find((e) => e.id.includes("MiniMax-M2.7")); + assert.equal("thinkingLevels" in m27, false); + assert.equal("contextWindowOptions" in m27, false); + }); + + test("a non-minimax wire form is never annotated from the minimax builtin tree", () => { + // The engine-session annotation is gated on the wire form's + // providerId. A BYOK provider that happens to have a model id + // colliding with a builtin name must not inherit the builtin's + // context windows. + const { list } = engine.projectModelCatalogue({ + sessionOption: { options: [{ value: "m:nousresearch%3Anousresearch%2FMiniMax-M3:u", name: "x" }] }, + providers: null, + builtins: [], + builtinThinking: THINKING_M3, + builtinContextWindows: WINDOWS_M3, + parseEngineModelWireValue, + }); + assert.deepEqual(Object.keys(list[0]), ["id", "name", "label", "provider", "source"]); + }); +}); + +// --------------------------------------------------------------------------- +// 7. The route +// --------------------------------------------------------------------------- + +describe("handleGetModels — the route asks the facade", () => { + // No `setupMocks` in these cases: the file-level `before` hook already + // registered the shared mocks, and node:test's file-level `before` and + // its subtests share ONE MockTracker — a second `setupMocks` here is + // ERR_INVALID_STATE ("already mocked"), not a re-registration. + let bust = 0; + const loadRoute = async () => import(`${absPath("routes/model.js")}?bust=${bust++}`); + + // `mock.module` REPLACES the whole namespace, so a partial mock makes + // the route fail to instantiate on the exports it did not stub. + const NOT_STUBBED = (name) => async () => { + throw new Error(`B4 test called ${name}, which this case did not stub`); + }; + function mockFacade(t, overrides) { + t.mock.module(absPath("engine/model-reads.js"), { + namedExports: { readEngineModelCatalogue: NOT_STUBBED("readEngineModelCatalogue"), ...overrides }, + }); + } + + function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; + } + + test("the handler is still SYNCHRONOUS — the body is complete when it returns", () => { + // The route's signature is part of its contract: an async handler + // would leave `res._body` null for any caller that does not await, + // and the pre-M3 handler was sync. This is the assertion that keeps + // the next reader from "simplifying" the facade to an async one. + const res = mkRes(); + const returned = modelRouteBaseline.handleGetModels(null, res, { cs: SNAPSHOT_CS }); + assert.equal(typeof returned.then, "undefined"); + assert.equal(res.written.length, 2); + assert.equal(res.written[0].status, 200); + }); + + test("the response body is the facade's payload, byte-for-byte", async (t) => { + const payload = { ok: true, models: [], groups: [], current: null, currentThinking: null, currentContextWindow: null, source: "mcode-cli-bundle" }; + mockFacade(t, { readEngineModelCatalogue: () => ({ payload, source: "config", gate: {}, transport: RUNTIME }) }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleGetModels(null, res, { cs: {} }); + assert.equal(res.written[0].headers["Content-Type"], "application/json; charset=utf-8"); + assert.equal(res.written[1].body, JSON.stringify(payload)); + }); + + test("the route hands its ctx through and does not read cs itself", async (t) => { + const seen = []; + mockFacade(t, { + readEngineModelCatalogue: (o) => { + seen.push(o); + return { payload: { ok: true, models: [], groups: [], current: null, currentThinking: null, currentContextWindow: null, source: "mcode-cli-bundle" }, source: "config", gate: {}, transport: RUNTIME }; + }, + }); + const route = await loadRoute(); + for (const ctx of [{ cs: SNAPSHOT_CS }, { cs: null }, undefined, {}]) { + await route.handleGetModels(null, mkRes(), ctx); + } + assert.equal(seen.length, 4); + assert.deepEqual(seen[0].cs, SNAPSHOT_CS); + assert.equal(seen[1].cs, null); + assert.equal(seen[2].cs, undefined); + assert.equal(seen[3].cs, undefined); + for (const o of seen) assert.equal(o.endpoint, undefined); + }); + + test("a gate error PROPAGATES (it is soft, but it must not be swallowed silently)", async (t) => { + // The model's gate never throws today, so this row pins the + // ROUTE's half of the contract: if a future family decision makes + // this gate hard, the route must not grow a catch that turns the + // 501 into a silent empty catalogue — the #110 fake-success + // failure mode, and the one this batch's soft gate exists to avoid + // reaching for. + const marker = new Error("model-gate-refused"); + mockFacade(t, { + readEngineModelCatalogue: () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleGetModels(null, mkRes(), { cs: {} }); + } catch (err) { + caught = err; + } + assert.equal(caught, marker); + }); + + // ---- proof the mock actually took ------------------------------------ + + test("PROOF the facade mock took: a marker error escapes the untouched route", async (t) => { + const marker = new Error("B4-MODEL-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + readEngineModelCatalogue: () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleGetModels(null, mkRes(), { cs: {} }); + } catch (err) { + caught = err; + } + assert.ok(caught, "the route swallowed the facade error — either the mock did not take, or the route grew a catch"); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + + test("CONTROL: with no facade mock, the route answers from the real projection", async (t) => { + // The other half of the proof: a fresh `?bust=` re-import binds the + // route to the REAL facade, so the body is the fixture projection — + // the same one the snapshot above pins, now through the route. + setBuiltinModelsMock(BUILTINS); + const route = await loadRoute(); + const res = mkRes(); + await route.handleGetModels(null, res, { cs: SNAPSHOT_CS }); + const body = JSON.parse(res.written[1].body); + assert.equal(body.ok, true); + assert.equal(body.models.length, 12); + assert.deepEqual(body.groups.map((g) => g.id), [ + "__engine", + "deepseek-cn", + "nousresearch", + "minimax_api", + "local-ollama", + ]); + assert.equal(body.current, MODEL_OPTION.currentValue); + assert.equal(body.currentThinking, "high"); + assert.equal(body.currentContextWindow, 1000000); + assert.equal(body.source, "acp-session-config"); + }); +}); diff --git a/packages/webui/test/lib/engine/session-switch.test.js b/packages/webui/test/lib/engine/session-switch.test.js new file mode 100644 index 00000000..ebfb4bae --- /dev/null +++ b/packages/webui/test/lib/engine/session-switch.test.js @@ -0,0 +1,1392 @@ +// webui/test/lib/engine/session-switch.test.js +// +// M3-B6: the session SWITCH family's engine facade — #3 +// POST /api/sessions/switch. +// +// Sections are ordered by how much user-visible damage a regression in +// each one does, not by which module the function came from: +// +// 1. THE DECLARATION AND ITS SOFT-GATE POLICY. The most consequential +// judgement call in this batch: #3 gates SOFT because the switch's +// primary data is webui's own session record and both of its engine +// touches have a defined degradation. A hard gate would delete a +// working endpoint over an enrichment. Section 1 proves the gate +// reports and never throws — including on the DEFAULT `acp` +// transport, where no provider is registered at all. +// 2. THE FOUR RED LINES. 转录回填 (backfill), cumulative detection, +// workspace containment, single base-session identity. One named +// test per line, plus the negative half of each, because a red line +// that is only asserted in its happy direction is a red line nobody +// is watching. +// 3. THE BYTE-FOR-BYTE WIRE SHAPES, table-driven across all four +// outcomes: status, Content-Type, the exact body string and the key +// ORDER of the success payload. +// 4. THE PURE DERIVATIONS, on their inputs. +// 5. THE ROUTE, with the proof that the facade mock actually took. +// 6. THE TRANSCRIPT SEAM, and what this batch did and did not retire +// about the 3-candidate probe (KNOWN DEBT 1 in the module header). +// +// Two module-mock traps apply here exactly as they did in B3/B4/B5, and +// both are load-bearing rather than incidental: +// +// 1. `t.mock.module` REPLACES the WHOLE NAMESPACE; it does not merge. +// A mock naming only the export under test leaves every other name +// undefined and the consumer fails at INSTANTIATION with +// `SyntaxError: … does not provide an export named …` — a failure +// that reads like a product bug and is not one. Every facade mock +// below goes through `mockAll()`, which fills the un-stubbed names +// with a function that THROWS, so an unexpected call is loud +// instead of returning a plausible payload. +// 2. `mock.module` re-evaluates only the MOCKED specifier. A consumer +// already in the registry keeps its old LIVE BINDING, so a second +// test in the same file would silently reuse the first test's mock +// and pass for the wrong reason. Every route re-import in section 5 +// carries a fresh `?bust=N`, and section 5 ends with marker controls +// that prove it. + +import { test, describe, before, after, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, mkdirSync, readFileSync, realpathSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { Readable } from "node:stream"; + +import { + setupMocks, + absPath, + registerSessionsStore, + getSessionsStore, + registerAcpMock, +} from "../../helpers/_setup.js"; +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; +// Type discrimination goes through the exported predicate, never +// `err.name`. `engine/capabilities.js` is never `mock.module`d by this +// file, so the `instanceof` inside it resolves against the same class +// `checkSessionSwitchCapability` would have thrown from had it thrown at +// all. The string comparison it replaces could not tell a capability +// error from any other error that happened to carry a name. +const { isEngineCapabilityNotSupportedError } = await import( + "../../../server/engine/errors.js" +); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +/** A syntactically valid engine sid — 32 lowercase hex digits. */ +const SID_A = "mvs_aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; +const SID_B = "mvs_bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"; +/** Not an engine sid: too short. Must take the 404 branch. */ +const NOT_A_SID = "webui-does-not-exist"; + +let bust = 0; + +/** A JSON request body the real `lib/read-json.js` can consume. */ +function jsonReq(body) { + return Readable.from([Buffer.from(JSON.stringify(body), "utf8")]); +} + +/** A minimal `ServerResponse` stand-in that records what was written. */ +function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; +} + +/** The v2 probe SQL, read from the ONE declaration (never re-typed). */ +let _v2Sql; +async function v2ProbeSql() { + if (_v2Sql) return _v2Sql; + const mod = await import(absPath("lib/transcript.js")); + _v2Sql = mod.V2_DATA_JSON_PROBES[0].sql; + return _v2Sql; +} + +/** + * Fake better-sqlite3 keyed by SQL string. `prepare()` throws for any SQL + * the fixture does not carry, exactly as a real prepare does on a missing + * column — which is what makes the legacy 3-candidate probes "miss" the + * way they miss against the live v2 schema. + */ +function makeFakeDb({ rowsBySql = {}, constructThrows = false } = {}) { + return class FakeDb { + constructor(path, opts) { + if (constructThrows) throw new Error("fake better-sqlite3: boom"); + this.path = path; + this.opts = opts; + } + prepare(sql) { + const bySid = rowsBySql[sql]; + if (!bySid) throw new Error(`fake db: no such column (${sql.slice(0, 52)}…)`); + return { all: (sid) => (bySid[sid] || []).slice() }; + } + close() {} + }; +} + +// Workspace fixtures. `assertWorkspacePath` is NOT mocked anywhere in +// this file — the containment red line is exactly the real gate's +// behaviour, so the fixtures are real directories under a real +// allowed-roots tree, and every path is realpath'd once at setup so the +// assertions compare against the same form the gate normalises to (Linux +// /tmp vs macOS /private/tmp — see fs-write.test.js, #81). +let WS_ROOT, WS_A, WS_B, WS_DEFAULT, WS_OUTSIDE, DEFAULT_DIR, DB_PATH; +let _eventsDir; +// Mutable sqlite fixture, read at call time by the resolver mock +// registered in `before()`. See `bootFacade` for why it cannot be a +// per-test registration. +let _dbOpts = {}; + +before((t) => { + _eventsDir = mkTmpDir("webui-switch-facade-events-"); + WS_ROOT = mkTmpDir("webui-switch-facade-roots-"); + DB_PATH = mkTmpDir("webui-switch-facade-db-"); + // Pinned BEFORE any SUT import: `lib/config.js` freezes + // MCODE_RUNTIME_DB, DEFAULT_WORKSPACE and the audit path at module load. + process.env.MCODE_WEBUI_EVENTS_PATH = join(_eventsDir, "events.ndjson"); + process.env.MCODE_RUNTIME_DB = join(DB_PATH, "runtime-state.sqlite"); + writeFileSync(process.env.MCODE_RUNTIME_DB, ""); + for (const name of ["projectA", "projectB", "default-workspace"]) { + mkdirSync(join(WS_ROOT, name), { recursive: true }); + } + WS_A = realpathSync(join(WS_ROOT, "projectA")); + WS_B = realpathSync(join(WS_ROOT, "projectB")); + WS_DEFAULT = realpathSync(join(WS_ROOT, "default-workspace")); + // A real directory that is deliberately OUTSIDE the allowed roots, so + // a record pointing at it is refused rather than silently accepted. + WS_OUTSIDE = realpathSync(mkTmpDir("webui-switch-facade-outside-")); + DEFAULT_DIR = WS_DEFAULT; + process.env.MCODE_WORKSPACE = DEFAULT_DIR; + process.env.MCODE_WEBUI_WORKSPACE_ROOTS = WS_ROOT; + // `lib/transcript.js` (the seam's reader) and `lib/session-tree.js` + // both import this module. Registered ONCE, before any SUT import, + // because both of them keep a live binding to it afterwards. + t.mock.module(absPath("lib/sqlite-resolver.js"), { + namedExports: { + getMcodeBetterSqlite3: () => makeFakeDb(_dbOpts), + _getBetterSqlite3Candidates: () => [], + }, + }); +}); + +after(() => { + delete process.env.MCODE_WEBUI_EVENTS_PATH; + delete process.env.MCODE_RUNTIME_DB; + delete process.env.MCODE_WEBUI_WORKSPACE_ROOTS; + delete process.env.MCODE_WORKSPACE; + for (const d of [_eventsDir, WS_ROOT, DB_PATH, WS_OUTSIDE]) { + if (d) rmTmpDir(d); + } +}); + +/** + * Boot the REAL facade over mocked storage. The sqlite fixture is what + * decides whether the transcript read answers, so every data-plane test + * that cares about the backfill passes `db` explicitly. + */ +async function bootFacade(t, { db = {}, store = [], acp = {}, mavis = {} } = {}) { + await setupMocks(t, { mavis: { applyMavisUsageToCs: async () => {}, ...mavis } }); + // The sqlite fixture is read at CALL time by the mock registered in + // `before()`. It cannot be re-registered per test: `lib/transcript.js` + // holds a live binding to `lib/sqlite-resolver.js` after its first + // import, and `mock.module` re-evaluates only the specifier it is given + // — so a second registration here would leave the reader on the FIRST + // test's fake and every later case would silently answer the wrong + // thing. This is mock trap #2, and it is why `before()` owns it. + _dbOpts = db; + registerSessionsStore({ initial: store }); + registerAcpMock({ + getMcodeSessionsCacheSync: () => null, + getMcodeSessionsStaleSync: () => null, + getMcodeSessionTitle: async () => null, + ...acp, + }); + // Imported AFTER the mocks: the facade reaches its storage through + // `await import()` at call time, so the registry mocks are what it + // gets — and importing here (not at file scope) keeps the real module + // the one under test in this section. + return import(absPath("engine/session-switch.js")); +} + +/** A minimal webui client state — only the fields the switch reads. */ +function mkCs(workspaceDir = WS_A) { + return { + sessionId: "webui-previous", + mcodeSessionId: null, + sessionTitle: "Previous", + chat: [], + usage: { sessionInput: 7, sessionOutput: 8, sessionTotal: 15, contextUsed: 3 }, + workspace: { dir: workspaceDir, branch: "main", tree: null }, + }; +} + +/** Three transcript rows that exercise user / thinking+tool / assistant. */ +function transcriptRows(sid) { + return { + [sid]: [ + { + role: "user", + turn_id: "turn-a", + msg_id: "msg-user-1", + data_json: JSON.stringify({ role: "user", msg_content: "调研工具" }), + }, + { + role: "assistant", + turn_id: "turn-a", + msg_id: "msg-assistant-1", + data_json: JSON.stringify({ + role: "assistant", + msg_content: "我先看看", + thinking_content: "先搜索", + tool_calls: [ + { + tool_name: "bash", + tool_call_id: "c1", + tool_call_status: 2, + tool_call_args: '{"command":"ls"}', + tool_call_result_data: '{"content":[{"type":"text","text":"file1"}]}', + }, + ], + }), + }, + { + role: "assistant", + turn_id: "turn-a", + msg_id: "msg-assistant-2", + data_json: JSON.stringify({ role: "assistant", msg_content: "结论" }), + }, + ], + }; +} + +const EXPECTED_LINES = [ + "› 调研工具", + "▲ 先搜索", + "● 我先看看", + '→ bash {"command":"ls"}', + " [completed]", + " file1", + "● 结论", + "§§ turn_msg=msg-assistant-2", +]; + +describe("M3-B6 — session switch family", () => { + // --------------------------------------------------------------------- + // 1. The declaration table and its soft-gate policy + // --------------------------------------------------------------------- + + describe("SESSION_SWITCH_ENDPOINTS — one endpoint, one soft declaration", () => { + test("covers exactly this batch's one endpoint", async () => { + const { SESSION_SWITCH_ENDPOINTS } = await import( + absPath("engine/session-switch.js") + ); + assert.deepEqual(Object.keys(SESSION_SWITCH_ENDPOINTS), [ + "POST /api/sessions/switch", + ]); + }); + + test("the row names the pair the ENRICHMENTS need, enforced softly", async () => { + const { SESSION_SWITCH_ENDPOINTS } = await import( + absPath("engine/session-switch.js") + ); + const { ENGINE_CAPABILITY_KEYS } = await import(absPath("engine/index.js")); + const row = SESSION_SWITCH_ENDPOINTS["POST /api/sessions/switch"]; + assert.deepEqual(Object.keys(row), [ + "capability", + "subItem", + "enforcement", + ]); + assert.deepEqual(row, { + capability: "sessionCrud", + subItem: "getSession", + enforcement: "soft", + }); + assert.ok( + ENGINE_CAPABILITY_KEYS.includes(row.capability), + "the declared capability must be a real registry key, not an invented one", + ); + }); + + test(`the DEFAULT transport (${ACP}) is UNREGISTERED and the gate says so`, async () => { + const { checkSessionSwitchCapability } = await import( + absPath("engine/session-switch.js") + ); + const gate = checkSessionSwitchCapability("POST /api/sessions/switch", ACP); + assert.equal(gate.gate, "unregistered-transport"); + assert.equal(gate.provider, null); + assert.equal(gate.enforcement, "soft"); + }); + + test(`the ${RUNTIME} transport resolves the v2 provider and checks the declaration`, async () => { + const { checkSessionSwitchCapability } = await import( + absPath("engine/session-switch.js") + ); + const gate = checkSessionSwitchCapability("POST /api/sessions/switch", RUNTIME); + assert.equal(gate.gate, "checked"); + assert.equal(gate.provider, "local-runtime-v2"); + }); + + test("NO transport ever produces a capability error — the family declares no throwing gate", async () => { + // A registry-driven assertion cannot cover the "provider declares + // sessionCrud: none" case, because no registered provider does and + // PROVIDERS is frozen. So the policy claim is pinned statically: + // this module must not import `assertEngineCapability` (the only + // thrower) and must not export an `assert*` gate. If a later + // editor adds either, this test is the thing that says no. + const src = readFileSync( + fileURLToPath(absPath("engine/session-switch.js")), + "utf8", + ); + assert.equal( + src.includes("assertEngineCapability("), + false, + "session-switch.js started calling the throwing gate — the 501 policy is a decision, not a refactor", + ); + const mod = await import(absPath("engine/session-switch.js")); + assert.deepEqual( + Object.keys(mod).filter((k) => /^assert/i.test(k)), + [], + "this family must expose no assert* gate; use checkSessionSwitchCapability", + ); + }); + + test("an unknown endpoint key is a plain Error, never a capability error", async () => { + const { checkSessionSwitchCapability } = await import( + absPath("engine/session-switch.js") + ); + let caught = null; + try { + checkSessionSwitchCapability("POST /api/sessions/nope", RUNTIME); + } catch (e) { + caught = e; + } + assert.ok(caught, "an unknown key must throw"); + assert.equal(caught.code, "unknown_session_switch_endpoint"); + assert.equal( + isEngineCapabilityNotSupportedError(caught), + false, + "caller confusion must never be dressed up as an engine limitation", + ); + }); + }); + + // --------------------------------------------------------------------- + // 2. The four red lines + // --------------------------------------------------------------------- + + describe("RED LINE 1 — 转录回填: a switch shows the conversation, it does not show an empty screen", () => { + test("empty stored chat → the engine transcript lands in cs.chat, the response and the persisted record", async (t) => { + const mod = await bootFacade(t, { + db: { rowsBySql: { [await v2ProbeSql()]: transcriptRows(SID_A) } }, + }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ id: SID_A, cs, cid: "cid-1" }); + assert.equal(r.outcome, "ok"); + assert.deepEqual(cs.chat, EXPECTED_LINES, "cs.chat carries the mapped transcript"); + assert.deepEqual( + r.payload.session.chat, + EXPECTED_LINES, + "the response carries the same lines the client state does", + ); + const saved = getSessionsStore()[0]; + assert.deepEqual(saved.chat, EXPECTED_LINES, "and the wrapper was re-persisted"); + assert.equal(r.transcript.ok, true); + assert.equal( + r.transcript.decision, + "empty", + "first touch fires the EMPTY branch of the backfill rule", + ); + assert.equal(r.transcript.reason, null, "and a successful read has no failure reason"); + }); + + test("NEGATIVE half: a CLEAN stored chat is kept even though the engine read would answer", async (t) => { + const mod = await bootFacade(t, { + db: { rowsBySql: { [await v2ProbeSql()]: transcriptRows(SID_A) } }, + store: [ + { + id: "webui-keep", + mcodeSessionId: SID_A, + title: "Keep", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● mine already"], + }, + ], + }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-keep", + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + assert.deepEqual(cs.chat, ["● mine already"], "a clean buffer is never clobbered"); + assert.equal(r.transcript, null, "and the read was not even attempted"); + }); + + test("a read that fails NEVER breaks the switch (missing db / bad driver / schema drift)", async (t) => { + // Three failure shapes, one promise: the switch answers 200 and + // keeps the stored chat. This is the endpoint's oldest contract + // and the reason the family's gate is soft. + for (const [name, db] of [ + ["constructor throws", { constructThrows: true }], + ["every prepare throws (schema drift)", {}], + ]) { + await t.test(name, async (t2) => { + const mod = await bootFacade(t2, { db }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ + id: SID_A, + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "ok", name); + assert.equal(r.payload.ok, true, name); + assert.deepEqual(r.payload.session.chat, [], name); + assert.equal(r.transcript.ok, false, name); + assert.ok(r.transcript.reason, name); + }); + } + }); + }); + + describe("RED LINE 2 — cumulative detection: a polluted buffer is repaired, a clean one is not", () => { + // Table-driven on the predicate, because the predicate is the whole + // red line and a change to it must be reviewed as a rule change. + const CUMULATIVE_TABLE = [ + ["empty buffer", [], false], + ["one dot line", ["● only one"], false], + [ + "non-cumulative segments", + ["● seg one", "● seg two", "● seg three"], + false, + ], + [ + "cumulative: a later line strictly contains an earlier one", + ["● part one", "● part one plus part two"], + true, + ], + [ + "cumulative anywhere in the buffer, not just the first pair", + ["● a", "● b", "● a and b and c"], + true, + ], + ["equal-length dots are NOT a superset", ["● ab", "● ba"], false], + [ + "non-dot lines are ignored entirely", + ["› prompt", "▲ thought", "→ tool {}", "○ system"], + false, + ], + [ + "a non-string entry does not throw the predicate", + ["● prefix", null, 42, "● prefix and more"], + true, + ], + ["a bare dot marker is not evidence", ["●", "● later"], false], + ["a shorter later line is not a superset", ["● long line", "● short"], false], + ]; + for (const [name, chat, expected] of CUMULATIVE_TABLE) { + test(`chatLooksCumulative: ${name} → ${expected}`, async () => { + const { chatLooksCumulative } = await import( + absPath("engine/session-switch.js") + ); + assert.equal(chatLooksCumulative(chat), expected); + }); + } + + test("selectTranscriptBackfill reads the predicate into the three-branch rule", async () => { + const { selectTranscriptBackfill } = await import( + absPath("engine/session-switch.js") + ); + assert.deepEqual(selectTranscriptBackfill([]), { + storedHasChat: false, + storedCumulative: false, + shouldBackfill: true, + reason: "empty", + }); + assert.deepEqual(selectTranscriptBackfill(["● a", "● a and b"]), { + storedHasChat: true, + storedCumulative: true, + shouldBackfill: true, + reason: "stored_cumulative", + }); + assert.deepEqual(selectTranscriptBackfill(["● a", "● b"]), { + storedHasChat: true, + storedCumulative: false, + shouldBackfill: false, + reason: "stored_shrinks", + }); + }); + + test("end to end: a cumulative stored buffer is replaced by the engine read and re-persisted", async (t) => { + const mod = await bootFacade(t, { + db: { rowsBySql: { [await v2ProbeSql()]: transcriptRows(SID_A) } }, + store: [ + { + id: "webui-polluted", + mcodeSessionId: SID_A, + title: "Polluted", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● seg one", "● seg one and seg two"], + }, + ], + }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-polluted", + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + assert.deepEqual(cs.chat, EXPECTED_LINES, "the polluted buffer is gone"); + assert.deepEqual(getSessionsStore()[0].chat, EXPECTED_LINES, "and stays gone"); + assert.equal(r.transcript.ok, true); + }); + + test("a cumulative buffer whose read comes back EMPTY is preserved, not blanked", async (t) => { + const mod = await bootFacade(t, { + db: {}, + store: [ + { + id: "webui-polluted-2", + mcodeSessionId: SID_A, + title: "Polluted", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● seg one", "● seg one and seg two"], + }, + ], + }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-polluted-2", + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + assert.deepEqual( + cs.chat, + ["● seg one", "● seg one and seg two"], + "an empty read must not delete the user's last view", + ); + assert.equal( + r.transcript.decision, + "stored_cumulative", + "and the log says the pollution branch fired, not the empty one", + ); + }); + }); + + describe("RED LINE 3 — workspace containment: the switch writes a gated path or it does not write one", () => { + test("an out-of-bounds stored workspace is REFUSED and the client state is untouched", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-outside", + mcodeSessionId: SID_A, + title: "Outside", + workspace: WS_OUTSIDE, + createdAt: 1, + updatedAt: 1, + chat: [], + }, + ], + }); + const cs = mkCs(WS_B); + const before = JSON.parse(JSON.stringify(cs)); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-outside", + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "workspace_refused"); + assert.equal(r.statusHint, 400); + assert.equal(r.audit, null, "a refused switch writes no audit event"); + assert.equal(r.payload.ok, false); + assert.equal(r.payload.attempted, WS_OUTSIDE, "the 400 names the path it refused"); + assert.ok(typeof r.payload.error === "string" && r.payload.error.length > 0); + assert.deepEqual( + cs, + before, + "a refused switch must leave identity, chat, usage and workspace exactly as they were", + ); + }); + + test("an empty stored workspace falls back to the DEFAULT, never to the caller's current one", async (t) => { + // The user-reported defect: "the file tree still shows the previous + // project". The caller is sitting in projectB; the record has no + // workspace of its own; the answer must be the default, not B. + const mod = await bootFacade(t, { store: [] }); + const cs = mkCs(WS_B); + const r = await mod.applyEngineSessionSwitch({ id: SID_A, cs, cid: "cid-1" }); + assert.equal(r.outcome, "ok"); + assert.equal(r.workspace.fallback, true); + assert.equal(cs.workspace.dir, WS_DEFAULT); + assert.notEqual(cs.workspace.dir, WS_B, "the current workspace must never be the fallback"); + assert.equal(r.payload.session.workspaceFallback, true); + }); + + test("a stored workspace wins over the default and the target record is never rewritten with the caller's", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-A", + mcodeSessionId: SID_A, + title: "Project A session", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● a"], + }, + ], + }); + const cs = mkCs(WS_B); + const r = await mod.applyEngineSessionSwitch({ id: "webui-A", cs, cid: "cid-1" }); + assert.equal(r.outcome, "ok"); + assert.equal(cs.workspace.dir, WS_A, "the file tree follows the switched session"); + assert.equal(r.workspace.fallback, false); + assert.equal(getSessionsStore()[0].workspace, WS_A, "the record keeps its own workspace"); + }); + + test("resolveSwitchWorkspace prefers target-first and reports the refusal shape", async () => { + const { resolveSwitchWorkspace } = await import( + absPath("engine/session-switch.js") + ); + const refuse = (p) => ({ ok: false, error: `outside: ${p}` }); + const accept = (p) => ({ ok: true, path: p, real: p }); + // Target-first. + assert.deepEqual( + resolveSwitchWorkspace({ workspace: " /ws/a " }, { + defaultWorkspace: "/ws/default", + assertPath: accept, + }), + { ok: true, dir: "/ws/a", real: "/ws/a", fallback: false }, + "the stored value is trimmed and used as-is", + ); + // Empty / missing / non-string → the default, flagged as a fallback. + for (const record of [{}, { workspace: "" }, { workspace: " " }, { workspace: 7 }]) { + const got = resolveSwitchWorkspace(record, { + defaultWorkspace: "/ws/default", + assertPath: accept, + }); + assert.equal(got.dir, "/ws/default", JSON.stringify(record)); + assert.equal(got.fallback, true, JSON.stringify(record)); + } + // Refusal carries the attempted path so the 400 can be actionable. + assert.deepEqual( + resolveSwitchWorkspace({ workspace: "/nope" }, { + defaultWorkspace: "/ws/default", + assertPath: refuse, + }), + { ok: false, error: "outside: /nope", attempted: "/nope" }, + ); + }); + }); + + describe("RED LINE 4 — single base session identity: one conversation, one record", () => { + test("first touch creates exactly ONE record whose id IS the engine sid", async (t) => { + const mod = await bootFacade(t, { store: [] }); + const r = await mod.applyEngineSessionSwitch({ + id: SID_A, + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + const store = getSessionsStore(); + assert.equal(store.length, 1, "one conversation must not produce two entries"); + assert.equal(store[0].id, SID_A, "the overlay record's id IS the engine sid"); + assert.equal(store[0].mcodeSessionId, SID_A); + assert.equal(r.payload.session.id, SID_A); + assert.equal(r.matchKind, null, "first touch is not a match against an existing record"); + }); + + test("a second switch to the same sid REUSES the record — no second entry appears", async (t) => { + const mod = await bootFacade(t, { store: [] }); + await mod.applyEngineSessionSwitch({ id: SID_A, cs: mkCs(), cid: "cid-1" }); + await mod.applyEngineSessionSwitch({ id: SID_A, cs: mkCs(WS_B), cid: "cid-1" }); + const store = getSessionsStore(); + assert.equal(store.length, 1, "repeated switches must hit the same record"); + assert.equal(store[0].id, SID_A); + }); + + test("resolveSwitchTarget prefers the engine sid over a webui uuid, the opposite of the write family", async () => { + // The single-identity rule, stated as a resolution order. Two + // discriminating cases, because the order is only OBSERVABLE when + // both passes could match — and a reader who writes one fixture + // will not notice that the other order passes it too. + const { resolveSwitchTarget } = await import(absPath("engine/session-switch.js")); + + // (a) The label case, and the common one: an overlay record's id + // IS its engine sid, so both passes match the same record and only + // `matchKind` tells the two orders apart. It is observable — the + // audit payload carries the label, and `new_from_mcode` vs + // `mcodeSessionId` is the difference between "we just created + // this" and "this already existed". + const overlay = { id: SID_A, mcodeSessionId: SID_A, title: "Overlay" }; + assert.deepEqual( + resolveSwitchTarget([overlay], SID_A), + { index: 0, matchKind: "mcodeSessionId", target: overlay }, + "an overlay addressed by its sid is an mcodeSessionId match, not a webuiId one", + ); + + // (b) The conflict case: two records could answer, and the one + // that IS the engine session wins. + const bySid = { id: "webui-1", mcodeSessionId: SID_A }; + const byUuid = { id: SID_B, mcodeSessionId: null }; + const records = [byUuid, bySid]; + assert.deepEqual(resolveSwitchTarget(records, SID_A), { + index: 1, + matchKind: "mcodeSessionId", + target: bySid, + }); + assert.deepEqual( + resolveSwitchTarget(records, SID_B), + { + index: 0, + // No record is BOUND to SID_B — the one whose UUID is SID_B + // has no mcodeSessionId at all — so the sid pass misses and the + // uuid pass wins. The order is only observable in the case + // where both passes could match. + matchKind: "webuiId", + target: byUuid, + }, + "an id that is a record's uuid but no record's engine sid is a webuiId match", + ); + assert.deepEqual(resolveSwitchTarget(records, "webui-1"), { + index: 1, + matchKind: "webuiId", + target: bySid, + }); + assert.deepEqual(resolveSwitchTarget(records, NOT_A_SID), { + index: -1, + matchKind: null, + target: null, + }); + }); + + test("an id that is neither a record nor an engine sid is `not_found`, never an invented overlay", async (t) => { + const mod = await bootFacade(t, { store: [] }); + const r = await mod.applyEngineSessionSwitch({ + id: NOT_A_SID, + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(r.outcome, "not_found"); + assert.equal(r.statusHint, 404); + assert.deepEqual(r.payload, { ok: false, error: "session not found" }); + assert.equal(getSessionsStore().length, 0, "a wrong id must not create a record"); + }); + }); + + // --------------------------------------------------------------------- + // 3. The byte-for-byte wire shapes + // --------------------------------------------------------------------- + + describe("the response shape is pinned byte-for-byte, in every outcome", () => { + test("success: exact body string and key order", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-A", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● x"], + }, + ], + }); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-A", + cs: mkCs(), + cid: "cid-1", + }); + assert.equal( + JSON.stringify(r.payload), + `{"ok":true,"session":{"id":"webui-A","mcodeSessionId":"${SID_A}","title":"T",` + + `"workspace":"${WS_A}","workspaceFallback":false,"chat":["● x"]}}`, + "the success body's key ORDER is a frontend contract (url-restore reads workspace)", + ); + assert.deepEqual(Object.keys(r.payload), ["ok", "session"]); + assert.deepEqual(Object.keys(r.payload.session), [ + "id", + "mcodeSessionId", + "title", + "workspace", + "workspaceFallback", + "chat", + ]); + }); + + test("not_found / workspace_refused bodies, key order included", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-outside", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_OUTSIDE, + createdAt: 1, + updatedAt: 1, + chat: [], + }, + ], + }); + const notFound = await mod.applyEngineSessionSwitch({ + id: NOT_A_SID, + cs: mkCs(), + cid: "cid-1", + }); + assert.equal( + JSON.stringify(notFound.payload), + '{"ok":false,"error":"session not found"}', + ); + const refused = await mod.applyEngineSessionSwitch({ + id: "webui-outside", + cs: mkCs(), + cid: "cid-1", + }); + assert.deepEqual(Object.keys(refused.payload), ["ok", "error", "attempted"]); + assert.equal(refused.payload.attempted, WS_OUTSIDE); + }); + + test("a record with no mcodeSessionId and no chat still answers the same six-key body", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-local", + title: "Local only", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: [], + }, + ], + }); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-local", + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + assert.deepEqual(Object.keys(r.payload.session), [ + "id", + "mcodeSessionId", + "title", + "workspace", + "workspaceFallback", + "chat", + ]); + assert.equal(r.payload.session.mcodeSessionId, null); + assert.deepEqual(r.payload.session.chat, []); + }); + + test("the audit payload is the B01 contract, and first touch keeps its own label", async (t) => { + const mod = await bootFacade(t, { store: [] }); + const r = await mod.applyEngineSessionSwitch({ + id: SID_A, + cs: mkCs(), + cid: "cid-9", + }); + assert.equal(r.audit.event, "session.switch"); + assert.equal(r.audit.target, SID_A); + assert.equal(r.audit.cid, "cid-9"); + assert.equal(r.audit.actor, "user"); + assert.deepEqual(Object.keys(r.audit.payload), [ + "from", + "matchKind", + "mcodeSessionId", + "title", + "workspace", + "workspaceFallback", + ]); + assert.equal(r.audit.payload.from, "webui-previous", "the prior session is recorded"); + assert.equal( + r.audit.payload.matchKind, + "new_from_mcode", + "a first touch is labelled new_from_mcode, NOT mcodeSessionId", + ); + assert.equal(r.audit.payload.workspace, WS_DEFAULT); + assert.equal(r.audit.payload.workspaceFallback, true); + }); + + test("an existing record reports its real matchKind in the audit", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-A", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: [], + }, + ], + }); + const bySid = await mod.applyEngineSessionSwitch({ + id: SID_A, + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(bySid.matchKind, "mcodeSessionId"); + assert.equal(bySid.audit.payload.matchKind, "mcodeSessionId"); + const byUuid = await mod.applyEngineSessionSwitch({ + id: "webui-A", + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(byUuid.matchKind, "webuiId"); + assert.equal(byUuid.audit.payload.matchKind, "webuiId"); + }); + }); + + // --------------------------------------------------------------------- + // 4. The pure derivations + // --------------------------------------------------------------------- + + describe("the pure derivations, on their inputs", () => { + test("isSwitchableMcodeSessionId is the 32-hex rule and nothing looser", async () => { + const { isSwitchableMcodeSessionId } = await import( + absPath("engine/session-switch.js") + ); + for (const good of [SID_A, SID_B, `mvs_${"a".repeat(32)}`]) { + assert.equal(isSwitchableMcodeSessionId(good), true, good); + } + for (const bad of [ + "mvs_short", + `mvs_${"a".repeat(31)}`, + `mvs_${"a".repeat(33)}`, + `mvs_${"A".repeat(32)}`, + "webui-1", + "", + null, + undefined, + 42, + ]) { + assert.equal(isSwitchableMcodeSessionId(bad), false, String(bad)); + } + }); + + test("lookupCachedMcodeTitle probes the current ws, then the stale reader, then the unfiltered key", async () => { + const { lookupCachedMcodeTitle } = await import( + absPath("engine/session-switch.js") + ); + const fresh = (ws) => + ws === "/ws/a" ? [{ sessionId: SID_A, title: "from fresh" }] : null; + const stale = () => [{ sessionId: SID_A, title: "from stale" }]; + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/a", { fresh, stale }), + "from fresh", + "the fresh reader for the current workspace wins", + ); + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/other", { fresh, stale }), + "from stale", + "a miss falls through to the stale reader", + ); + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/none", { + fresh: () => null, + stale: () => null, + }), + null, + "a total miss is null so the caller can pay for the ACP path", + ); + assert.equal(lookupCachedMcodeTitle("", "/ws/a", { fresh, stale }), null); + // A throwing cache reader is a miss, not a crash: the switch must + // still be able to fall back to the engine title. + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/a", { + fresh: () => { + throw new Error("cache exploded"); + }, + stale: () => null, + }), + null, + ); + }); + + test("lookupCachedMcodeTitle finds a title cached under the unfiltered key", async () => { + const { lookupCachedMcodeTitle } = await import( + absPath("engine/session-switch.js") + ); + // getMcodeSessionsForWorkspace("") caches the UNFILTERED list, so a + // cache walked without a workspace still answers the first touch. + const unfiltered = [{ sessionId: SID_A, title: "Unfiltered title" }]; + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/a", { + fresh: (ws) => (ws === "" ? unfiltered : null), + stale: () => null, + }), + "Unfiltered title", + ); + }); + + test("applySwitchedSessionToClientState sets identity, chat, usage and workspace — and nothing else", async () => { + const { applySwitchedSessionToClientState } = await import( + absPath("engine/session-switch.js") + ); + const cs = { + sessionId: "old", + mcodeSessionId: "old-sid", + sessionTitle: "Old", + chat: ["● stale"], + usage: { sessionInput: 7, sessionOutput: 8, sessionTotal: 15, contextUsed: 3 }, + workspace: { dir: "/ws/old", branch: "main", tree: ["t"] }, + lastUsedWorkspace: "/ws/last-used", + }; + const out = applySwitchedSessionToClientState(cs, { + target: { id: "new", mcodeSessionId: SID_A, title: "New", chat: ["● fresh"] }, + workspaceDir: "/ws/new", + }); + assert.equal(out, cs, "the same object is mutated in place"); + assert.equal(cs.sessionId, "new"); + assert.equal(cs.mcodeSessionId, SID_A); + assert.equal(cs.sessionTitle, "New"); + assert.deepEqual(cs.chat, ["● fresh"]); + assert.deepEqual(cs.usage, { + sessionInput: 0, + sessionOutput: 0, + sessionTotal: 0, + contextUsed: 3, + }, "the three cumulative counters zero, every other key preserved"); + assert.deepEqual(cs.workspace, { dir: "/ws/new", branch: null, tree: null }); + assert.equal( + cs.lastUsedWorkspace, + "/ws/last-used", + "switching is browsing: last-used-workspace must not move", + ); + }); + + test("applySwitchedSessionToClientState normalises the three optional target fields", async () => { + const { applySwitchedSessionToClientState } = await import( + absPath("engine/session-switch.js") + ); + const cs = { usage: {} }; + applySwitchedSessionToClientState(cs, { + target: { id: "u1" }, + workspaceDir: "/ws/x", + }); + assert.equal(cs.mcodeSessionId, null, "a record with no engine sid binds to null"); + assert.equal(cs.sessionTitle, "Untitled", "and an absent title reads as Untitled"); + assert.deepEqual(cs.chat, [], "a non-array chat is an empty chat, never a crash"); + }); + }); + + // --------------------------------------------------------------------- + // 5. The route + // --------------------------------------------------------------------- + + describe("handleSwitchSession — HTTP parsing, status codes, and the fail-closed audit", () => { + /** Every export the REAL facade has, so a partial mock fails loud. */ + const FACADE_EXPORTS = [ + "SESSION_SWITCH_ENDPOINTS", + "applyEngineSessionSwitch", + "applySwitchedSessionToClientState", + "chatLooksCumulative", + "checkSessionSwitchCapability", + "isSwitchableMcodeSessionId", + "lookupCachedMcodeTitle", + "readEngineSwitchTranscript", + "resolveSessionSwitchProvider", + "resolveSwitchTarget", + "resolveSwitchWorkspace", + "selectTranscriptBackfill", + ]; + function mockFacade(t, impls) { + const namedExports = {}; + for (const name of FACADE_EXPORTS) { + namedExports[name] = () => { + throw new Error(`B6 test called engine/session-switch.js#${name}, which this case did not stub`); + }; + } + Object.assign(namedExports, impls); + t.mock.module(absPath("engine/session-switch.js"), { namedExports }); + } + const loadRoute = async () => + import(`${absPath("routes/sessions.js")}?bust=${bust++}`); + + const OK_AUDIT = { + event: "session.switch", + target: "webui-A", + cid: "tab-1", + actor: "user", + payload: { + from: "webui-previous", + matchKind: "webuiId", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_A, + workspaceFallback: false, + }, + }; + const OK_BODY = { + ok: true, + session: { + id: "webui-A", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_A, + workspaceFallback: false, + chat: ["● x"], + }, + }; + + test("a missing id is the route's own 400, in the route's own words", async (t) => { + await setupMocks(t, {}); + mockFacade(t, {}); + const route = await loadRoute(); + // `{id: 42}` is deliberately NOT in this list: `(payload.id || "").trim()` + // throws a TypeError on a number, which is the pre-facade behaviour + // and a 500 rather than a 400. Tightening it would be a behaviour + // change dressed as a hardening, and this batch promises none — + // it is recorded as a question for the request-validation pass + // instead (see KNOWN DEBT, `routes/sessions.js`). + for (const body of [{}, { id: "" }, { id: " " }, { id: null }]) { + const res = mkRes(); + await route.handleSwitchSession(jsonReq(body), res, { cs: mkCs(), cid: "tab-1" }); + assert.equal(res.written[0].status, 400); + // Pre-existing asymmetry, preserved: this body is the ONE shape + // on this endpoint that does not carry the charset. + assert.equal(res.written[0].headers["Content-Type"], "application/json"); + assert.equal(res.written[1].body, '{"ok":false,"error":"id required"}'); + } + }); + + // Table-driven across every outcome the facade can report. The + // status, the Content-Type and the body are all pinned; the two + // Content-Type spellings are the pre-existing asymmetry and must not + // be tidied into one. + const OUTCOMES = [ + [ + "not_found", + 404, + "application/json", + '{"ok":false,"error":"session not found"}', + ], + [ + "workspace_refused", + 400, + "application/json; charset=utf-8", + JSON.stringify({ ok: false, error: "outside: /nope", attempted: "/nope" }), + ], + ]; + for (const [outcome, status, contentType, body] of OUTCOMES) { + test(`${outcome} → ${status} with Content-Type ${contentType}`, async (t) => { + await setupMocks(t, {}); + mockFacade(t, { + applyEngineSessionSwitch: async () => ({ + outcome, + statusHint: status, + payload: JSON.parse(body), + audit: null, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleSwitchSession(jsonReq({ id: "x" }), res, { + cs: mkCs(), + cid: "tab-1", + }); + assert.equal(res.written[0].status, status); + assert.equal(res.written[0].headers["Content-Type"], contentType); + assert.equal(res.written[1].body, body); + assert.equal(res.written.length, 2, "a non-ok outcome writes exactly one response"); + }); + } + + test("the route writes the facade's audit event verbatim, then the state push, then the 200", async (t) => { + await setupMocks(t, {}); + mockFacade(t, { + applyEngineSessionSwitch: async () => ({ + outcome: "ok", + statusHint: 200, + matchKind: "webuiId", + workspace: { ok: true, dir: WS_A, fallback: false }, + transcript: null, + audit: OK_AUDIT, + payload: OK_BODY, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleSwitchSession(jsonReq({ id: "webui-A" }), res, { + cs: mkCs(), + cid: "tab-1", + }); + assert.equal(res.written[0].status, 200); + assert.equal(res.written[0].headers["Content-Type"], "application/json"); + assert.equal(res.written[1].body, JSON.stringify(OK_BODY)); + // The audit really landed: lib/events.js appends one NDJSON line + // per call, and the line carries the event name. + const auditPath = process.env.MCODE_WEBUI_EVENTS_PATH; + assert.ok(existsSync(auditPath), "the switch wrote no audit line at all"); + const raw = readFileSync(auditPath, "utf8"); + const last = raw.trim().split("\n").pop(); + assert.ok(last.includes("session.switch"), `last audit line: ${last}`); + }); + + test("a failed audit is fail-closed: 500, and the audit sink's own body", async (t) => { + await setupMocks(t, {}); + mockFacade(t, { + applyEngineSessionSwitch: async () => ({ + outcome: "ok", + statusHint: 200, + audit: OK_AUDIT, + payload: OK_BODY, + }), + }); + const route = await loadRoute(); + // Point the audit stream at a DIRECTORY: `events.js#append` writes + // atomically and throws EISDIR, which is the failure the fail-closed + // branch exists for. `_eventsPath()` reads the env lazily, so no + // re-import is needed. + const prev = process.env.MCODE_WEBUI_EVENTS_PATH; + const asDir = join(_eventsDir, "events-as-a-directory"); + mkdirSync(asDir, { recursive: true }); + process.env.MCODE_WEBUI_EVENTS_PATH = asDir; + try { + const res = mkRes(); + await route.handleSwitchSession(jsonReq({ id: "webui-A" }), res, { + cs: mkCs(), + cid: "tab-1", + }); + assert.equal(res.written[0].status, 500); + assert.equal( + res.written[0].headers["Content-Type"], + "application/json; charset=utf-8", + ); + assert.equal( + res.written[1].body, + '{"ok":false,"error":"audit write failed","detail":"session.switch"}', + ); + } finally { + process.env.MCODE_WEBUI_EVENTS_PATH = prev; + } + }); + + test("PROOF: a marker error from the facade escapes the route", async (t) => { + // Without a fresh `?bust=` re-import, `mock.module` would leave the + // route holding the PREVIOUS test's live binding, the marker would + // never be thrown, and this assertion would fail — which is the + // point: it is the only assertion in this section that cannot pass + // by accident. + await setupMocks(t, {}); + const marker = new Error("B6-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + applyEngineSessionSwitch: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleSwitchSession(jsonReq({ id: "webui-A" }), mkRes(), { + cs: mkCs(), + cid: "tab-1", + }); + } catch (err) { + caught = err; + } + assert.ok( + caught, + "the route swallowed the facade error — either the mock did not take, or the route grew a catch", + ); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + }); + + // --------------------------------------------------------------------- + // 6. The transcript seam, and what this batch retired + // --------------------------------------------------------------------- + + describe("the transcript seam", () => { + test("readEngineSwitchTranscript never throws — every failure is a value", async (t) => { + const mod = await bootFacade(t, { db: { constructThrows: true } }); + for (const mcodeSessionId of [SID_A, "not-a-sid", ""]) { + const r = await mod.readEngineSwitchTranscript({ mcodeSessionId }); + assert.equal(r.ok, false, mcodeSessionId); + assert.equal(r.source, "none"); + assert.deepEqual(r.lines, []); + assert.ok(r.reason, "a failure always names its reason for the operator log"); + } + }); + + test("readEngineSwitchTranscript reports the gate and the transport it asked under", async (t) => { + const mod = await bootFacade(t, { db: {} }); + const r = await mod.readEngineSwitchTranscript({ mcodeSessionId: SID_A }); + assert.equal(r.gate.endpoint, "POST /api/sessions/switch"); + assert.equal(r.gate.enforcement, "soft"); + assert.equal( + r.gate.gate === "unregistered-transport" || r.gate.gate === "checked", + true, + `unexpected gate ${r.gate.gate}`, + ); + }); + + test("RETIRED: routes/sessions.js no longer names lib/transcript.js at all", async () => { + // The part of the probe debt this batch actually collected. A + // static source assertion is the right instrument here: the claim + // is about an IMPORT GRAPH, and this suite has no render harness + // that could observe it. `export.js`'s comment still names the + // switch path by prose, which is exactly the kind of drift the + // assertion below is here to catch. + const src = readFileSync( + fileURLToPath(absPath("routes/sessions.js")), + "utf8", + ); + assert.equal( + /from\s+"\.\.\/lib\/transcript\.js"/.test(src), + false, + "the route must not import the transcript reader directly any more", + ); + assert.equal( + /loadTranscriptChatLines|readMcodeTranscript/.test(src), + false, + "the route must not call a transcript reader directly any more", + ); + // And the read is reachable exactly once, through the seam. + const facadeSrc = readFileSync( + fileURLToPath(absPath("engine/session-switch.js")), + "utf8", + ); + assert.equal( + /import\("\.\.\/lib\/transcript\.js"\)/.test(facadeSrc), + true, + "the seam is the single owner of the transcript read now", + ); + }); + + test("KEPT, deliberately: the probe set behind the seam is unchanged", async (t) => { + // KNOWN DEBT 1. The 3-candidate legacy probe set is still the + // implementation, because the default `acp` transport has no + // engine surface to replace it with and export's enrichment is + // byte-pinned to those same candidates. This test is the tripwire + // that makes the debt VISIBLE: if a later batch swaps the seam to + // the engine's `getMessages`, the switch's line set changes here + // and the failure names the batch that has to justify it. + const mod = await bootFacade(t, { + db: { rowsBySql: { [await v2ProbeSql()]: transcriptRows(SID_A) } }, + }); + const r = await mod.readEngineSwitchTranscript({ mcodeSessionId: SID_A }); + assert.equal(r.ok, true); + assert.equal(r.probeTable, "local_runtime_message_rows"); + assert.equal(r.probe, "v2-data-json", "the v2 data_json probe is the one that answers"); + assert.equal(r.messageCount, 3); + }); + }); +}); diff --git a/packages/webui/test/lib/engine/session-writes.test.js b/packages/webui/test/lib/engine/session-writes.test.js new file mode 100644 index 00000000..24239521 --- /dev/null +++ b/packages/webui/test/lib/engine/session-writes.test.js @@ -0,0 +1,1780 @@ +// webui/test/lib/engine/session-writes.test.js +// +// M3-B5: the session WRITE family's engine facade — #7 delete, #4 +// rename, #6 cleanup-orphans. +// +// This is the first suite in the migration that tests a family which +// DESTROYS data, so the sections below are ordered by how much damage a +// regression in each one does, not by which module the function came +// from: +// +// 1. THE DECLARATION AND ITS POLICY. The hard/none split in here is +// the batch's most consequential judgement call: #7 and #6 are hard +// because they destroy the engine's own rows, #4 declares no +// capability because it touches no engine surface. Section 2 proves +// the asymmetry is real by driving all three endpoints from ONE +// provider fixture. +// +// 2. THE FIVE DELETE RED LINES. "Deleted sessions must not come back", +// "deleting a session is not deleting files", "a running session +// has defined semantics", "the other tab must lose the entry", and +// "the audit chain stays intact". These are the checks a reviewer +// should read first, so they get their own section with one test +// per line. +// +// 3. THE BYTE-FOR-BYTE PREVIEW SHAPES. #6's dryRun body is a hard red +// line for this batch; #7's is pinned beside it because the same +// edit touched both. +// +// 4. THE PURE DERIVATIONS, on their inputs. +// +// 5. THE ROUTE, with the proof that the facade mock actually took. +// +// Two module-mock traps apply here exactly as they did in B3/B4, and +// both are load-bearing rather than incidental: +// +// 1. `t.mock.module` REPLACES the WHOLE NAMESPACE; it does not merge. +// A mock naming only the export under test leaves every other name +// undefined and the consumer fails at INSTANTIATION with +// `SyntaxError: … does not provide an export named …` — a failure +// that reads like a product bug and is not one. Every mock below +// goes through `mockAll()`, which fills the un-stubbed names with a +// function that THROWS, so an unexpected call is loud instead of +// returning a plausible payload. +// 2. `mock.module` re-evaluates only the MOCKED specifier. A consumer +// already in the registry keeps its old LIVE BINDING, so a second +// test in the same file would silently reuse the first test's mock +// and pass for the wrong reason. Every route re-import carries a +// fresh `?bust=N`, and section 5 ends with the marker control that +// proves it. + +import { test, describe, before, after, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, mkdirSync, writeFileSync, readFileSync } from "node:fs"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { Readable } from "node:stream"; +import { spawnSync } from "node:child_process"; + +import { + setupMocks, + absPath, + registerSessionsStore, + registerAcpMock, + withDecisions, +} from "../../helpers/_setup.js"; +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; + +// --------------------------------------------------------------------------- +// Per-file path isolation (B5 test-hygiene fix) +// --------------------------------------------------------------------------- +// `readOrphanSessionWriteIds` reads `lib/config.js#SESSIONS_DB`, and that +// constant is frozen when config.js is FIRST evaluated — which happens inside +// the first test that pulls `engine/index.js` into the registry, long before +// the preview test below runs. So the pin has to sit at module scope: setting +// it inside the test body would be a no-op dressed up as isolation. +// +// The bug this kills: the preview test asserted `count:0` because the +// developer's `~/.mcode-webui/sessions.json` "does not exist in this +// environment". On any machine that has actually used the app it DOES exist, +// and the assertion was a statement about the developer's home directory +// rather than about the facade — green on a clean CI runner, red on every +// workstation, and unfixable by editing the product. +// +// Four variables, all rooted in one tracked temp directory (SPEC §7's +// isolation trio plus the file under test): +// +// MCODE_WEBUI_SESSIONS_DB — the file the sweep reads; the one that leaked +// MCODE_WEBUI_DATA_DIR — its parent, so every other path config.js +// derives from the data dir lands here too +// MCODE_WEBUI_SETTINGS_PATH — settings.json, which config.js reads at import +// MINIMAX_DATA_DIR — the engine's data dir; without it the +// MCODE_RUNTIME_DB contract still resolves +// against the real ~/.minimax +// +// SESSIONS_DB is deliberately left NON-EXISTENT. The empty sweep is the shape +// this red line pins, and after this change it is guaranteed by construction +// instead of by the absence of a file the test never created. +const ISOLATED_DIR = mkTmpDir("webui-session-writes-b5-"); +process.env.MCODE_WEBUI_SESSIONS_DB = join(ISOLATED_DIR, "sessions.json"); +process.env.MCODE_WEBUI_DATA_DIR = ISOLATED_DIR; +process.env.MCODE_WEBUI_SETTINGS_PATH = join(ISOLATED_DIR, "settings.json"); +process.env.MINIMAX_DATA_DIR = ISOLATED_DIR; + +after(() => { + rmTmpDir(ISOLATED_DIR); + delete process.env.MCODE_WEBUI_SESSIONS_DB; + delete process.env.MCODE_WEBUI_DATA_DIR; + delete process.env.MCODE_WEBUI_SETTINGS_PATH; + delete process.env.MINIMAX_DATA_DIR; +}); + +// Type discrimination goes through the exported predicate, never +// `err.name`. `engine/capabilities.js` is never `mock.module`d by this +// file, so the `instanceof` inside it resolves against the same class the +// gate throws from; the sibling batches (account-reads, session-export) +// assert the same way. The string comparison it replaces could not tell a +// capability error from any other error that happened to carry a name. +const { isEngineCapabilityNotSupportedError } = await import( + "../../../server/engine/errors.js" +); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +// A syntactically valid engine sid — `isMcodeSessionId` requires exactly +// 32 lowercase hex digits, and every fixture below that wants the ORPHAN +// branch has to satisfy the same regex the pre-facade route spelled +// inline four times. +const ORPHAN_SID = "mvs_aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; + +let bust = 0; + +/** A JSON request body the real `lib/read-json.js` can consume. */ +function jsonReq(body) { + return Readable.from([Buffer.from(JSON.stringify(body), "utf8")]); +} + +/** A minimal `ServerResponse` stand-in that records what was written. */ +function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; +} + +/** + * A fresh copy of `routes/sessions.js`. + * + * `mock.module` re-evaluates only the MOCKED specifier, but a route + * module already in the registry keeps its old LIVE BINDING to the + * facade — without the `?bust=N` re-import a second test would silently + * exercise the first test's mock and pass for the wrong reason. That is + * what the PROOF cases below exist to catch. + */ +const loadRoute = async () => import(`${absPath("routes/sessions.js")}?bust=${bust++}`); + +/** + * Register a module mock that satisfies the namespace contract. + * + * @param {object} t The test context. + * @param {string} rel Server-relative specifier, e.g. "lib/foo.js". + * @param {object} impls The exports this test stubs. + * @param {string[]} known Every export name the REAL module has, so + * anything this test does not stub is present-but-throwing rather + * than absent. + */ +function mockAll(t, rel, impls, known) { + const namedExports = {}; + for (const name of known) { + namedExports[name] = (...a) => { + throw new Error(`B5 test called ${rel}#${name}, which this case did not stub`); + }; + } + Object.assign(namedExports, impls); + t.mock.module(absPath(rel), { namedExports }); +} + +/** + * The record-ordering journal the delete tests assert on. Every mutation + * the write path performs appends its name here, so a test can assert + * the SEQUENCE rather than the end state — and a sequence is the only + * thing that distinguishes a correct delete from a resurrecting one. + */ +const journal = []; +function resetJournal() { + journal.length = 0; +} + +describe("M3-B5 — session write family", () => { + // --------------------------------------------------------------------- + // 1. The declaration table and the gate policy it records + // --------------------------------------------------------------------- + + describe("SESSION_WRITE_ENDPOINTS — the three writes, and who owns the rows they destroy", () => { + test("covers exactly this batch's three endpoints", async () => { + const { SESSION_WRITE_ENDPOINTS } = await import( + absPath("engine/session-writes.js") + ); + assert.deepEqual(Object.keys(SESSION_WRITE_ENDPOINTS), [ + "DELETE /api/sessions/:id", + "POST /api/sessions/rename", + "POST /api/sessions/cleanup-orphans", + ]); + }); + + test("every row declares the same three keys, including the no-capability one", async () => { + // The uniformity is the point of this family's table shape: a + // `null` hole for rename would read as "not filled in yet" to the + // next editor rather than as a decision. + const { SESSION_WRITE_ENDPOINTS } = await import( + absPath("engine/session-writes.js") + ); + for (const [endpoint, row] of Object.entries(SESSION_WRITE_ENDPOINTS)) { + assert.deepEqual( + Object.keys(row), + ["capability", "subItem", "enforcement"], + `${endpoint} has a different row shape`, + ); + assert.ok(["hard", "soft", "none"].includes(row.enforcement), endpoint); + } + }); + + // Table-driven: the table IS the assertion, because editing a row is + // a capability decision and has to be reviewed as one. + const TABLE = [ + [ + "DELETE /api/sessions/:id", + { capability: "sessionCrud", subItem: "deleteSession", enforcement: "hard" }, + "the delete destroys rows in the engine's own local_runtime_* tables", + ], + [ + "POST /api/sessions/rename", + { capability: null, subItem: null, enforcement: "none" }, + "a rename writes webui's store and crosses no engine surface", + ], + [ + "POST /api/sessions/cleanup-orphans", + { capability: "sessionCrud", subItem: "deleteSession", enforcement: "hard" }, + "the sweep delegates to #7, so it destroys the same engine rows", + ], + ]; + for (const [endpoint, row, why] of TABLE) { + test(`${endpoint} → ${row.enforcement}${row.capability ? ` on ${row.capability}.${row.subItem}` : ""} (${why})`, async () => { + const { SESSION_WRITE_ENDPOINTS } = await import( + absPath("engine/session-writes.js") + ); + const { ENGINE_CAPABILITY_KEYS } = await import(absPath("engine/index.js")); + assert.deepEqual(SESSION_WRITE_ENDPOINTS[endpoint], row); + if (row.capability) assert.ok(ENGINE_CAPABILITY_KEYS.includes(row.capability)); + }); + } + + test("#6 declares the SAME pair as #7 — the sweep is a delete by another name", async () => { + // If these two ever drift, a provider that cannot delete engine + // sessions could still reach the engine's tables through the + // sweep's back door. The assertion compares against #7's own row, + // not against a copy, so it fails the moment either one moves. + const { SESSION_WRITE_ENDPOINTS } = await import( + absPath("engine/session-writes.js") + ); + assert.deepEqual( + SESSION_WRITE_ENDPOINTS["POST /api/sessions/cleanup-orphans"], + SESSION_WRITE_ENDPOINTS["DELETE /api/sessions/:id"], + ); + }); + + test("an endpoint outside this family is caller confusion, not an engine limitation", async () => { + const { assertSessionWriteCapability } = await import( + absPath("engine/session-writes.js") + ); + assert.throws( + () => assertSessionWriteCapability("DELETE /api/sessions", RUNTIME), + (err) => { + // A caller-typo must NOT answer 501, so the proof is that it is + // not a capability error at all — a positive check on the code + // and message alone would also pass if the error carried both + // by accident. + assert.ok(!isEngineCapabilityNotSupportedError(err)); + assert.equal(err.code, "unknown_session_write_endpoint"); + assert.match(err.message, /not part of the session write family/); + return true; + }, + ); + }); + }); + + describe("resolveSessionWriteProvider / assertSessionWriteCapability", () => { + // Table-driven. Absent means "no provider claims this transport yet" + // (M4), which is NOT the same answer as "capability unavailable" — + // the default `acp` transport must keep deleting sessions, so it + // must NOT throw. + const TRANSPORTS = [ + [RUNTIME, true, "checked", "local-runtime-v2"], + [ACP, false, "unregistered-transport", null], + ["exec", false, "unregistered-transport", null], + ["", false, "unregistered-transport", null], + ]; + for (const [transport, hasProvider, gate, providerId] of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → ${gate}`, async () => { + const { assertSessionWriteCapability, resolveSessionWriteProvider } = + await import(absPath("engine/session-writes.js")); + const provider = resolveSessionWriteProvider(transport); + assert.equal(!!provider, hasProvider); + const g = assertSessionWriteCapability("DELETE /api/sessions/:id", transport); + assert.equal(g.gate, gate); + assert.equal(g.provider, providerId); + assert.equal(g.capability, "sessionCrud"); + assert.equal(g.subItem, "deleteSession"); + assert.equal(g.endpoint, "DELETE /api/sessions/:id"); + assert.equal(g.enforcement, "hard"); + }); + } + + test("rename reports no-capability-key on EVERY transport, provider or not", async () => { + // The single most important assertion about #4: renaming a + // session works on a webui-only store and must not become a 501 + // because of anything a provider declares. Checked across all + // four transports so a future `if (provider)` shortcut cannot + // reintroduce the dependency behind the "it only fires on runtime" + // argument. + const { assertSessionWriteCapability } = await import( + absPath("engine/session-writes.js") + ); + for (const [transport] of TRANSPORTS) { + const g = assertSessionWriteCapability("POST /api/sessions/rename", transport); + assert.equal(g.gate, "no-capability-key", transport); + assert.equal(g.capability, null, transport); + assert.equal(g.subItem, null, transport); + assert.equal(g.enforcement, "none", transport); + } + }); + + test("the descriptor carries the six B1–B4 fields plus `enforcement`", async () => { + // A consumer reading `gate.provider` under `acp` must get `null`, + // not `undefined` — the key must EXIST. The six shared fields are + // asserted by name so the families cannot drift apart, and + // `enforcement` is the write family's own addition. + const { assertSessionWriteCapability } = await import( + absPath("engine/session-writes.js") + ); + assert.deepEqual(Object.keys(assertSessionWriteCapability("DELETE /api/sessions/:id", RUNTIME)), [ + "endpoint", + "gate", + "provider", + "capability", + "subItem", + "enforcement", + ]); + }); + }); + + // --------------------------------------------------------------------- + // 2. The hard / none asymmetry, driven from ONE provider fixture + // --------------------------------------------------------------------- + + // The proof that section 1's policy is enforced by code and not by the + // provider's shape. One fixture provider, three endpoints, three + // different answers — and the two "must throw" rows are what stop + // #7/#6 from silently degrading into a no-op delete on a provider that + // cannot delete. + const CAPABILITY_FIXTURES = [ + ["none", { level: "none", reason: "fixture: interface-absent" }], + [ + "partial missing deleteSession", + { level: "partial", missing: ["deleteSession"], reason: "fixture: no delete surface" }, + ], + [ + "partial keeping deleteSession", + { level: "partial", missing: ["getSession"], reason: "fixture: delete present" }, + ], + ["full", { level: "full" }], + ]; + + for (const [name, sessionCrud] of CAPABILITY_FIXTURES) { + test(`provider sessionCrud=${name}: #7 and #6 THROW, #4 never does`, async (t) => { + await setupMocks(t, { acp: {} }); + // `mock.module` replaces the whole namespace; session-writes.js + // reads two names from engine/index.js and the test re-imports the + // facade under a fresh bust so the mock is the one it sees. + t.mock.module(absPath("engine/index.js"), { + namedExports: { + DEFAULT_ENGINE_PROVIDER_ID: "fixture-provider", + getEngineProvider: () => ({ + id: "fixture-provider", + transport: "runtime", + capabilities: { sessionCrud }, + }), + }, + }); + const mod = await import(`${absPath("engine/session-writes.js")}?caps=${bust++}`); + // The gate throws when the declaration withholds `deleteSession` + // itself: `none` withholds the whole capability, and a `partial` + // withholds the sub-item. A `full`, or a `partial` that still + // carries `deleteSession`, passes — which is the sub-item + // granularity the declaration contract exists to provide. + const throws = + sessionCrud.level === "none" || + (sessionCrud.level === "partial" && sessionCrud.missing.includes("deleteSession")); + for (const endpoint of ["DELETE /api/sessions/:id", "POST /api/sessions/cleanup-orphans"]) { + if (throws) { + assert.throws( + () => mod.assertSessionWriteCapability(endpoint, "runtime"), + (err) => { + assert.ok(isEngineCapabilityNotSupportedError(err)); + assert.equal(err.capability, "sessionCrud"); + assert.equal(err.provider, "fixture-provider"); + return true; + }, + `${endpoint} should have thrown for sessionCrud=${name}`, + ); + } else { + const g = mod.assertSessionWriteCapability(endpoint, "runtime"); + assert.equal(g.gate, "checked", `${endpoint} / ${name}`); + } + } + // Rename, whatever the provider says. This is the assertion that + // fails loudly if someone "helpfully" gives #4 a capability. + const rename = mod.assertSessionWriteCapability("POST /api/sessions/rename", "runtime"); + assert.equal(rename.gate, "no-capability-key", name); + assert.equal(rename.provider, "fixture-provider", "it still reports which provider is live"); + }); + } + + test("the hard gate costs ZERO deletions: it throws before the plan reads the store", async (t) => { + // Ordering matters for a destructive endpoint. A gate that ran after + // the store load would still be correct, but a gate that ran after + // the COMMIT would be theatre — so the proof is that the plan + // rejects without ever resolving a target. + await setupMocks(t, { acp: {} }); + registerSessionsStore({ initial: [{ id: "webui-A", title: "A", chat: [] }] }); + t.mock.module(absPath("engine/index.js"), { + namedExports: { + DEFAULT_ENGINE_PROVIDER_ID: "fixture-provider", + getEngineProvider: () => ({ + id: "fixture-provider", + transport: "runtime", + capabilities: { sessionCrud: { level: "none", reason: "fixture" } }, + }), + }, + }); + const mod = await import(`${absPath("engine/session-writes.js")}?order=${bust++}`); + await assert.rejects( + () => mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }), + (err) => { + assert.ok(isEngineCapabilityNotSupportedError(err)); + return true; + }, + ); + // The store is untouched: `getSessionsStore` still holds the record. + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.equal(getSessionsStore().length, 1); + }); + + // --------------------------------------------------------------------- + // 3. The pure derivations + // --------------------------------------------------------------------- + + describe("pure derivations", () => { + test("isMcodeSessionId accepts only the engine's 32-hex shape", async () => { + const { isMcodeSessionId } = await import(absPath("engine/session-writes.js")); + const TABLE = [ + [ORPHAN_SID, true], + [`mvs_${"a".repeat(32)}`, true], + [`mvs_${"A".repeat(32)}`, false, "uppercase hex is not the engine's shape"], + [`mvs_${"a".repeat(31)}`, false], + [`mvs_${"a".repeat(33)}`, false], + ["mvs_", false], + ["webui-A", false], + ["", false], + [null, false], + [undefined, false], + [42, false, "a non-string must not throw — it is simply not an engine sid"], + ]; + for (const [input, expected, why] of TABLE) { + assert.equal(isMcodeSessionId(input), expected, `${JSON.stringify(input)}: ${why || "shape"}`); + } + }); + + test("resolveSessionTarget answers the same three ways for both writes", async () => { + // Rename and delete used to carry this lookup as two identical + // copies. Table-driven over one store so the shared predicate is + // pinned for both. + const { resolveSessionTarget } = await import(absPath("engine/session-writes.js")); + const records = [ + { id: "webui-A", mcodeSessionId: "mvs_11111111111111111111111111111111" }, + { id: "webui-B" }, + ]; + const TABLE = [ + ["webui-A", 0, "webuiId", "matched by the webui uuid"], + ["mvs_11111111111111111111111111111111", 0, "mcodeSessionId", "matched by the bound engine sid"], + ["webui-B", 1, "webuiId", "a record with no engine sid still matches its own id"], + ["nope", -1, null, "an unknown id resolves to nothing, and matchKind is null — not \"unknown\""], + ]; + for (const [id, index, matchKind, why] of TABLE) { + const r = resolveSessionTarget(records, id); + assert.equal(r.index, index, why); + assert.equal(r.matchKind, matchKind, why); + assert.equal(r.target, index >= 0 ? records[index] : null, why); + } + // A non-array store must not throw: the store is a file on disk and + // a corrupt one answers `[]`, never a TypeError inside a gate. + assert.deepEqual(resolveSessionTarget(null, "x"), { index: -1, matchKind: null, target: null }); + }); + + test("the orphan rule: empty AND default-titled AND older than 24h", async () => { + const { isOrphanSessionRecord, ORPHAN_STALE_MS } = await import( + absPath("engine/session-writes.js") + ); + const NOW = 1_700_000_000_000; + const old = NOW - ORPHAN_STALE_MS - 1; + const TABLE = [ + [{ id: "a", title: "Untitled", chat: [], updatedAt: old }, true, "the canonical leftover"], + [{ id: "b", title: "New session", chat: [], updatedAt: old }, true, "the other default name"], + [{ id: "c", title: "对话 7", chat: [], updatedAt: old }, true, "the numbered default"], + [{ id: "d", title: "Untitled", chat: [], updatedAt: NOW }, false, "too fresh"], + [ + { id: "e", title: "Untitled", chat: [], updatedAt: NOW - ORPHAN_STALE_MS + 1 }, + false, + "one millisecond inside the window is still fresh", + ], + [ + { id: "f", title: "Untitled", chat: [], updatedAt: NOW - ORPHAN_STALE_MS }, + true, + "exactly at the threshold is stale — the rule is `<`, not `<=`", + ], + [{ id: "g", title: "Untitled", chat: ["● hi"], updatedAt: old }, false, "has chat"], + [{ id: "h", title: "Real work", chat: [], updatedAt: old }, false, "not a default title"], + [{ id: "i", title: "Untitled", chat: [], updatedAt: 0 }, true, "updatedAt 0 is falsy, so the age check is skipped — preserved"], + [{ id: "j", title: " Untitled ", chat: [], updatedAt: old }, true, "titles are trimmed before matching"], + [{ id: "k", title: "对话7", chat: [], updatedAt: old }, false, "the numbered form needs the space"], + [{ id: "", title: "Untitled", chat: [], updatedAt: old }, false, "no id"], + [null, false, "a null record"], + [{ title: "Untitled", chat: [], updatedAt: old }, false, "no id"], + [{ id: "m", title: "Untitled", updatedAt: old }, true, "a missing chat counts as empty"], + ]; + for (const [record, expected, why] of TABLE) { + assert.equal(isOrphanSessionRecord(record, NOW, ORPHAN_STALE_MS), expected, why); + } + }); + + test("selectOrphanSessionIds keeps store order and survives a non-array", async () => { + const { selectOrphanSessionIds, ORPHAN_STALE_MS } = await import( + absPath("engine/session-writes.js") + ); + const now = 1_700_000_000_000; + const old = now - ORPHAN_STALE_MS - 1; + assert.deepEqual( + selectOrphanSessionIds( + [ + { id: "keep-me", title: "Real", chat: [], updatedAt: old }, + { id: "b", title: "Untitled", chat: [], updatedAt: old }, + { id: "a", title: "Untitled", chat: [], updatedAt: old }, + ], + { now }, + ), + ["b", "a"], + "store order, not sorted order — the ids are reported in the order they would be deleted", + ); + assert.deepEqual(selectOrphanSessionIds(null, { now }), []); + }); + + test("the two fan-out predicates really are different predicates", async () => { + // The temptation this test exists to kill: one shared + // "is this client in this session" helper. It would be wrong in + // both directions — clearing a tab that was never deleted, and + // blanking the title of a tab bound to a DIFFERENT wrapper record. + const { clientMatchesDeletedSession, clientMatchesRenamedSession } = await import( + absPath("engine/session-writes.js") + ); + const record = { id: "webui-A", mcodeSessionId: "mvs_sid_A" }; + const inRecord = { sessionId: "webui-A", mcodeSessionId: null }; + const byEngineSid = { sessionId: "webui-OTHER", mcodeSessionId: "mvs_sid_A" }; + const byRequestId = { sessionId: "webui-THIRD", mcodeSessionId: "mvs_sid_B" }; + + assert.equal(clientMatchesRenamedSession(inRecord, record), true, "rename: same webui id"); + assert.equal(clientMatchesRenamedSession(byEngineSid, record), true, "rename: same engine sid"); + assert.equal(clientMatchesRenamedSession(byRequestId, record), false, "rename: unrelated tab"); + + assert.equal(clientMatchesDeletedSession(inRecord, record, "webui-A"), true, "delete: same webui id"); + assert.equal(clientMatchesDeletedSession(byEngineSid, record, "webui-A"), false, + "delete: a tab bound to the record's engine sid under ANOTHER wrapper is a different record and must not be cleared"); + assert.equal(clientMatchesDeletedSession(byRequestId, record, "mvs_sid_B"), true, + "delete: matches the id the REQUEST named, which is the orphan branch's only handle"); + }); + + test("the delete reset clears identity, title, chat and the three usage counters", async () => { + const { applyDeletedSessionToClientState } = await import( + absPath("engine/session-writes.js") + ); + const cs = { + sessionId: "webui-A", + mcodeSessionId: "mvs_sid_A", + sessionTitle: "A", + chat: ["● hi", "● there"], + usage: { sessionInput: 10, sessionOutput: 20, sessionTotal: 30, cost: 1.5 }, + somethingElse: "kept", + }; + applyDeletedSessionToClientState(cs); + assert.equal(cs.sessionId, null); + assert.equal(cs.mcodeSessionId, null); + assert.equal(cs.sessionTitle, "Untitled"); + assert.deepEqual(cs.chat, []); + assert.equal(cs.usage.sessionInput, 0); + assert.equal(cs.usage.sessionOutput, 0); + assert.equal(cs.usage.sessionTotal, 0); + assert.equal(cs.usage.cost, 1.5, "unrelated usage fields survive"); + assert.equal(cs.somethingElse, "kept", "the reset touches only what it names"); + }); + + test("the orphan branch does NOT zero usage — the asymmetry is a parameter, not an accident", async () => { + const { applyDeletedSessionToClientState } = await import( + absPath("engine/session-writes.js") + ); + const cs = { + sessionId: null, + mcodeSessionId: ORPHAN_SID, + sessionTitle: "Orphan", + chat: [], + usage: { sessionInput: 10, sessionOutput: 20, sessionTotal: 30 }, + }; + applyDeletedSessionToClientState(cs, { resetUsage: false }); + assert.equal(cs.sessionTitle, "Untitled", "the identity and title are still cleared"); + assert.deepEqual(cs.chat, []); + assert.equal(cs.usage.sessionTotal, 30, "an orphan has no webui record, so no tab accrued usage for it"); + }); + }); + + // --------------------------------------------------------------------- + // 4. The delete red lines + // --------------------------------------------------------------------- + + // The real store / cache / SQL collaborators, journalled. Every + // mutation appends its name in the order it happened, because for a + // delete the ORDER is the feature and an end-state assertion cannot see + // a resurrected session or an out-of-order cache drop. + async function loadWritePath(t, options = {}) { + await setupMocks(t, { acp: {}, sessions: { initial: options.records || [] } }); + registerAcpMock({ + shutdownMcodeAcpSingleton: () => { + journal.push("kill-acp-child"); + }, + dropMcodeSessionFromCache: (sid) => { + journal.push(`drop-cache:${sid}`); + }, + }); + const dbCalls = []; + mockAll( + t, + "lib/mcode-session-delete.js", + { + deleteMcodeSessionFromDb: (sid, o) => { + journal.push(`sql:${sid}:dryRun=${!!o.dryRun}`); + dbCalls.push({ sid, dryRun: !!o.dryRun, db: o.MCODE_RUNTIME_DB }); + return ( + options.dbResult || { + ok: true, + outcome: "deleted", + log: ["local_runtime_sessions:1"], + totalRowsDeleted: 1, + tablesAbsent: 0, + } + ); + }, + }, + ["deleteMcodeSessionFromDb", "MCODE_SESSION_DELETE_TABLES"], + ); + mockAll( + t, + "lib/session-tree.js", + { + invalidateSessionTree: () => { + journal.push("invalidate-tree"); + }, + getSessionTree: () => { + journal.push("getSessionTree"); + return { ok: true, tree: [] }; + }, + }, + ["getSessionTree", "invalidateSessionTree"], + ); + const clients = new Map(); + const pushes = []; + mockAll( + t, + "lib/state-bus.js", + { + clients, + pushStateFor: (c) => { + journal.push(`push:${c}`); + pushes.push(c); + }, + runChatViewChat: () => ({}), + makeClientState: () => ({ usage: {} }), + }, + [ + "clients", + "pushStateFor", + "runChatViewChat", + "makeClientState", + "setState", + "getClient", + "sseByCid", + "pushAlert", + ], + ); + return { + dbCalls, + clients, + pushes, + mod: await import(`${absPath("engine/session-writes.js")}?w=${bust++}`), + }; + } + + beforeEach(() => { + resetJournal(); + }); + + describe("RED LINE — a deleted session does not come back", () => { + test("the write path runs kill → SQL → scoped cache drop, in that order", async (t) => { + // THE ordering assertion. The long-lived mcode ACP child holds the + // session in memory and rewrites its registry row on its next + // request, so a delete that removes the rows but leaves the child + // alive produces a session that reappears on the next read. The + // tree-cache drop is asserted FIRST because it must precede the + // engine write: a concurrent read must not be able to repopulate + // the cache from the pre-delete database. + const { mod, dbCalls } = await loadWritePath(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● hi"] }, + { id: "webui-B", title: "B", chat: [] }, + ], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.deepEqual(journal, [ + "invalidate-tree", + "kill-acp-child", + "drop-cache:mvs_sid_A", + "sql:mvs_sid_A:dryRun=false", + // The requesting tab is pushed even though no tab matched it — + // see the fan-out section. It is part of the delete, not after it. + "push:tab-1", + ]); + assert.equal(dbCalls.length, 1, "exactly one engine delete, for the linked sid"); + assert.equal(dbCalls[0].dryRun, false, "a real delete is never a dry run"); + }); + + test("only the DELETED sid leaves the cache — the other session is untouched", async (t) => { + // The regression this guards is the sidebar flash: invalidating the + // WHOLE cache empties the list, refills it, and reads to the user + // like the delete failed. The per-sid drop is why a 42-entry + // sidebar goes to 41 and stays there. + const { mod, clients } = await loadWritePath(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }, + { id: "webui-B", mcodeSessionId: "mvs_sid_B", title: "B", chat: [] }, + ], + }); + clients.set("tab-1", { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", usage: {} }); + clients.set("tab-2", { sessionId: "webui-B", mcodeSessionId: "mvs_sid_B", usage: {} }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.equal( + journal.filter((j) => j.startsWith("drop-cache:")).length, + 1, + "exactly one cache entry dropped", + ); + assert.ok(!journal.includes("drop-cache:mvs_sid_B"), "the untouched session keeps its cache entry"); + assert.deepEqual( + w.records.map((r) => r.id), + ["webui-B"], + "the sibling record survives in the persisted store", + ); + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.deepEqual( + getSessionsStore().map((r) => r.id), + ["webui-B"], + "and it is gone from the STORE, not just from the returned array", + ); + }); + + test("it is a TRUE delete: re-deleting the same sid still reaches the engine", async (t) => { + // The distinction the batch's red line 3 turns on — a real delete + // versus a frontend fake. After the first delete the webui record + // is gone, so the SECOND delete of the same engine sid resolves as + // an ORPHAN and goes straight to the SQL deleter. If the first + // delete had only hidden the record (or if the store save were + // skipped), this second call would resolve `webuiId` again and the + // engine would never learn the session is gone. + const { mod, dbCalls } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + const first = await mod.planEngineSessionDelete({ id: "mvs_sid_A", transport: RUNTIME }); + assert.equal(first.matchKind, "mcodeSessionId"); + await mod.commitEngineSessionDelete({ plan: first, cid: "tab-1" }); + + resetJournal(); + const second = await mod.planEngineSessionDelete({ id: "mvs_sid_A", transport: RUNTIME }); + assert.equal(second.isOrphan, true, "the record is really gone from the store"); + assert.equal(second.matchKind, null); + const w = await mod.commitEngineOrphanSessionDelete({ plan: second, cs: null, cid: "tab-1" }); + assert.equal(w.failed, false); + assert.equal(dbCalls.length, 2, "the engine was told twice — the delete is not a UI illusion"); + assert.equal(dbCalls[1].sid, "mvs_sid_A"); + }); + }); + + describe("RED LINE — deleting a session is not deleting files", () => { + // The session's ARTEFACTS live on the filesystem under the run + // directory (`~/tmp/run_*/`), and red line 4 makes the side file + // tree part of the contract. A session delete removes rows in a + // SQLite database; it must not remove a single byte of the user's + // output. This test puts a real file there and checks it afterwards. + test("a real run-directory artefact survives the delete", async (t) => { + const dir = mkTmpDir("webui-session-writes-b5-"); + try { + const runDir = join(dir, "run_20260920_120000"); + mkdirSync(runDir, { recursive: true }); + const artefact = join(runDir, "build.log"); + writeFileSync(artefact, "compiled output the user still wants\n"); + const { mod } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● done"] }], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.equal(w.payload.ok, true, "the delete itself succeeded"); + assert.ok(existsSync(artefact), "the artefact file is still on disk"); + assert.equal(readFileSync(artefact, "utf8"), "compiled output the user still wants\n"); + } finally { + rmTmpDir(dir); + } + }); + + test("the same holds for a dryRun preview and for the orphan branch", async (t) => { + const dir = mkTmpDir("webui-session-writes-b5-"); + try { + const artefact = join(dir, "report.md"); + writeFileSync(artefact, "# notes\n"); + const { mod } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + await mod.previewEngineSessionDelete({ plan }); + resetJournal(); + const orphan = await mod.planEngineSessionDelete({ id: ORPHAN_SID, transport: RUNTIME }); + await mod.commitEngineOrphanSessionDelete({ plan: orphan, cs: null, cid: "tab-1" }); + assert.ok(existsSync(artefact), "neither the preview nor the orphan branch touches files"); + } finally { + rmTmpDir(dir); + } + }); + + test("the 32-table delete list is unchanged — the debt is recorded, not silently collected", async () => { + // The plan for this batch annotated `lib/mcode-session-delete.js` + // "delete". It is kept, because `lib/acp-client.js` imports from it + // and four test files bind to the specifier. This test pins the + // consequence: the table list is still exported, still has 32 + // entries, and the facade reaches it rather than duplicating it. + // A future collection that moves the list has to change this + // assertion in the same commit — which is the point. + const { MCODE_SESSION_DELETE_TABLES } = await import( + absPath("lib/mcode-session-delete.js") + ); + assert.equal(MCODE_SESSION_DELETE_TABLES.length, 32, "the 32-table list, still owned by the lib module"); + assert.equal(MCODE_SESSION_DELETE_TABLES[0], "local_runtime_sessions"); + assert.ok(MCODE_SESSION_DELETE_TABLES.includes("local_runtime_token_usage")); + const src = readFileSync(fileURLToPath(absPath("engine/session-writes.js")), "utf8"); + // The facade must NOT have grown its own copy of the list, or its + // own SQL. A second list is precisely how two writers end up + // deleting different sets of rows. The check is on SQL VERBS + // rather than on a table name, because the module's own comments + // legitimately name the tables while explaining what it does not + // do; a `DELETE FROM` or `SELECT` in this file would be the real + // smell. + for (const verb of ["DELETE FROM", "SELECT ", "INSERT ", "UPDATE ", "prepare("]) { + assert.ok( + !src.includes(verb), + `the facade must issue no SQL, found ${JSON.stringify(verb)} — the delete SQL stays in the lib module it forwards to`, + ); + } + // And the forwarding itself is real: the facade reaches that module + // through a dynamic import, not a second static one. + assert.match( + src, + /import\("\.\.\/lib\/mcode-session-delete\.js"\)/, + "the facade forwards to lib/mcode-session-delete.js through a lazy import", + ); + assert.ok( + !/^import .*mcode-session-delete/m.test(src), + "and never through a static one — a static import would put the SQL on the boot path", + ); + }); + }); + + describe("RED LINE — a running session has defined semantics", () => { + test("deleting an in-flight session kills the child that is driving the turn", async (t) => { + // There is no "refuse to delete a running session" guard, and this + // pins the semantics that DO exist rather than leaving it implied: + // the user asked, the child stops, the rows go. Recorded as KNOWN + // DEBT in the facade header — "refuse" is a defensible product + // decision, but it is not this batch's to make, and an unstated + // behaviour is worse than a stated one. + const { mod, clients } = await loadWritePath(t, { + records: [ + { + id: "webui-A", + mcodeSessionId: "mvs_sid_A", + title: "Running", + chat: ["● working"], + }, + ], + }); + const cs = { + sessionId: "webui-A", + mcodeSessionId: "mvs_sid_A", + sessionTitle: "Running", + chat: ["● working"], + running: { active: true, pid: 4242 }, + usage: { sessionTotal: 99 }, + }; + clients.set("tab-1", cs); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.ok(journal.includes("kill-acp-child"), "the child driving the turn is stopped"); + assert.ok( + journal.indexOf("kill-acp-child") < journal.findIndex((j) => j.startsWith("sql:")), + "and it is stopped BEFORE the rows go — otherwise it rewrites them", + ); + assert.equal(w.payload.ok, true, "and the delete proceeds — there is no refusal semantics"); + assert.equal(w.deletedItem.title, "Running"); + assert.equal(cs.mcodeSessionId, null, "the tab is not left pointing at a dead turn"); + }); + + test("the requesting tab's in-flight state is reset by the fan-out, not left dangling", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● x"] }], + }); + const cs = { + sessionId: "webui-A", + mcodeSessionId: "mvs_sid_A", + sessionTitle: "A", + chat: ["● x"], + usage: { sessionInput: 5, sessionOutput: 6, sessionTotal: 11 }, + }; + clients.set("tab-1", cs); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.equal(cs.sessionId, null); + assert.equal(cs.mcodeSessionId, null); + assert.equal(cs.sessionTitle, "Untitled"); + assert.deepEqual(cs.chat, []); + assert.equal(cs.usage.sessionTotal, 0, "the tab stops reporting the deleted session's spend"); + assert.deepEqual(pushes, ["tab-1"]); + assert.equal(w.touchedCids.length, 1); + }); + }); + + describe("RED LINE — the delete fans out to every other tab", () => { + test("a second tab inside the same session is cleared and pushed", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● x"] }], + }); + const tab1 = { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", sessionTitle: "A", chat: ["● x"], usage: { sessionTotal: 3 } }; + const tab2 = { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", sessionTitle: "A", chat: ["● x"], usage: { sessionTotal: 4 } }; + const tab3 = { sessionId: "webui-Z", mcodeSessionId: "mvs_sid_Z", sessionTitle: "Z", chat: ["● z"], usage: { sessionTotal: 5 } }; + clients.set("tab-1", tab1); + clients.set("tab-2", tab2); + clients.set("tab-3", tab3); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.equal(tab1.mcodeSessionId, null); + assert.equal(tab2.mcodeSessionId, null); + assert.equal(tab2.chat.length, 0, "the OTHER tab loses the entry too — this is the cross-tab red line"); + assert.equal(tab3.mcodeSessionId, "mvs_sid_Z", "an unrelated tab is left completely alone"); + assert.equal(tab3.usage.sessionTotal, 5); + assert.deepEqual(pushes.sort(), ["tab-1", "tab-2"], "both affected tabs are pushed"); + assert.equal(w.touchedCids.length, 2); + assert.equal(w.payload.ok, true); + }); + + test("a tab that matched nothing still gets exactly one push, so it cannot render a ghost", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + clients.set("tab-elsewhere", { sessionId: "webui-Z", mcodeSessionId: "mvs_sid_Z", usage: {} }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.deepEqual(pushes, ["tab-1"], "the requesting tab is pushed even though it matched nothing"); + assert.equal(w.touchedCids.length, 1); + }); + + test("the orphan branch clears ONLY the requesting tab — there is no record for others to be inside", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { records: [] }); + const cs = { sessionId: "webui-OTHER", mcodeSessionId: ORPHAN_SID, sessionTitle: "Orphan", chat: ["● x"], usage: { sessionTotal: 7 } }; + clients.set("tab-1", cs); + const plan = await mod.planEngineSessionDelete({ id: ORPHAN_SID, transport: RUNTIME }); + assert.equal(plan.isOrphan, true); + const w = await mod.commitEngineOrphanSessionDelete({ plan, cs, cid: "tab-1" }); + assert.equal(w.failed, false); + assert.equal(cs.mcodeSessionId, null, "the tab that was sitting on the orphan is cleared"); + assert.equal(cs.sessionTitle, "Untitled"); + assert.equal(cs.usage.sessionTotal, 7, "and its usage is NOT zeroed — the orphan branch's documented asymmetry"); + assert.deepEqual(pushes, ["tab-1"], "only that one tab is pushed"); + assert.equal(w.payload.matchKind, "orphan_mcode"); + }); + + test("the orphan branch leaves a tab that was NOT on the orphan alone", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { records: [] }); + const other = { sessionId: "webui-Z", mcodeSessionId: "mvs_sid_Z", sessionTitle: "Z", usage: {} }; + clients.set("tab-2", other); + const plan = await mod.planEngineSessionDelete({ id: ORPHAN_SID, transport: RUNTIME }); + await mod.commitEngineOrphanSessionDelete({ plan, cs: null, cid: "tab-1" }); + assert.equal(other.mcodeSessionId, "mvs_sid_Z"); + assert.deepEqual(pushes, [], "no tab matched, so no tab was disturbed"); + }); + }); + + // --------------------------------------------------------------------- + // 5. The byte-for-byte preview shapes + // --------------------------------------------------------------------- + + describe("preview shapes — #6's dryRun body is a hard red line for this batch", () => { + test("the cleanup-orphans dryRun body is exactly four keys, in order", async (t) => { + // Compared as a STRING, not as a parsed object: key ORDER is part + // of a byte-for-byte contract, and `deepEqual` on objects would not + // notice a reshuffle. + await setupMocks(t, { acp: {} }); + const mod = await import(`${absPath("engine/session-writes.js")}?shape=${bust++}`); + // The store read is against the SESSIONS_DB pinned at module scope, a + // path that intentionally does not exist, so the answer is the empty + // case — which is the shape most likely to be "simplified". The same + // assertion held on a CI runner by accident; here it holds because the + // test owns the path it reads. + const sweep = await mod.readOrphanSessionWriteIds({ transport: RUNTIME }); + assert.equal( + JSON.stringify(sweep.payload), + '{"ok":true,"dryRun":true,"count":0,"ids":[]}', + ); + assert.equal(sweep.gate.gate, "checked"); + assert.equal(sweep.gate.enforcement, "hard"); + }); + + test("a populated sweep answers the same four keys with the selected ids", async (t) => { + await setupMocks(t, { acp: {} }); + const { selectOrphanSessionIds } = await import( + `${absPath("engine/session-writes.js")}?shape=${bust++}` + ); + // The selection is pure, so the populated case is pinned through it + // while the SHAPE stays pinned through the real read above. The + // response is the same object the read would build. + const ids = selectOrphanSessionIds( + [ + { id: "old-1", title: "Untitled", chat: [], updatedAt: 1 }, + { id: "keep", title: "Real", chat: [], updatedAt: 1 }, + { id: "old-2", title: "对话 3", chat: [], updatedAt: 1 }, + ], + { now: Number.MAX_SAFE_INTEGER }, + ); + assert.equal( + JSON.stringify({ ok: true, dryRun: true, count: ids.length, ids }), + '{"ok":true,"dryRun":true,"count":2,"ids":["old-1","old-2"]}', + ); + }); + + test("#7's dryRun body keeps its four keys and the webuiEntryWouldBeDeleted block", async (t) => { + const { mod } = await loadWritePath(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A on tmp", chat: ["● hi"] }, + ], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.previewEngineSessionDelete({ plan }); + assert.equal( + JSON.stringify(w.payload), + JSON.stringify({ + ok: true, + dryRun: true, + matchKind: "webuiId", + mcodeDbDel: { ok: true, outcome: "deleted", log: ["local_runtime_sessions:1"], totalRowsDeleted: 1, tablesAbsent: 0 }, + webuiEntryWouldBeDeleted: { id: "webui-A", title: "A on tmp", mcodeSessionId: "mvs_sid_A" }, + }), + ); + assert.deepEqual(Object.keys(w.payload), [ + "ok", + "dryRun", + "matchKind", + "mcodeDbDel", + "webuiEntryWouldBeDeleted", + ]); + }); + + test("a webui-only session with no engine sid still previews, with an empty log", async (t) => { + // A record that never reached the engine has no rows to count. + // Refusing to preview for those would be a new failure mode, and + // the empty-log literal is the endpoint's own. + const { mod, dbCalls } = await loadWritePath(t, { + records: [{ id: "webui-B", title: "B", chat: [] }], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-B", transport: RUNTIME }); + const w = await mod.previewEngineSessionDelete({ plan }); + assert.deepEqual(w.mcodeDbDel, { ok: true, dryRun: true, log: [], totalRows: 0 }); + assert.equal(w.payload.webuiEntryWouldBeDeleted.mcodeSessionId, undefined); + assert.equal(dbCalls.length, 0, "and the SQL deleter is never asked about a non-sid"); + }); + + test("a dryRun mutates nothing: no kill, no cache drop, no tree invalidation, no store write", async (t) => { + // A preview that shuts down the user's ACP child is a side effect + // the `?dryRun=true` contract does not include, and a preview that + // drops the tree cache is a lie ("nothing changed" while the + // sidebar re-renders). The journal is empty except for the + // read-only SQL count. + const { mod, clients } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + clients.set("tab-1", { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", sessionTitle: "A", chat: ["● x"], usage: {} }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + await mod.previewEngineSessionDelete({ plan }); + assert.deepEqual( + journal, + ["sql:mvs_sid_A:dryRun=true"], + "the ONLY thing a preview does is ask the SQL layer to count", + ); + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.equal(getSessionsStore().length, 1, "the record is still there after a preview"); + assert.equal( + clients.get("tab-1").mcodeSessionId, + "mvs_sid_A", + "and the tab is still inside it", + ); + }); + }); + + // --------------------------------------------------------------------- + // 6. The rename write + // --------------------------------------------------------------------- + + describe("#4 rename — a webui-side label, and nothing else", () => { + test("a rename writes the store, drops the tree cache and pushes every matching tab", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "Old", chat: ["● x"] }, + { id: "webui-B", title: "B", chat: [] }, + ], + }); + const tab1 = { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", sessionTitle: "Old", usage: {} }; + const tab2 = { sessionId: "webui-OTHER", mcodeSessionId: "mvs_sid_A", sessionTitle: "Old", usage: {} }; + const tab3 = { sessionId: "webui-B", mcodeSessionId: null, sessionTitle: "B", usage: {} }; + clients.set("tab-1", tab1); + clients.set("tab-2", tab2); + clients.set("tab-3", tab3); + const w = await mod.applyEngineSessionRename({ id: "webui-A", title: "New", cid: "tab-1" }); + assert.equal(w.outcome, "ok"); + assert.equal(w.matchKind, "webuiId"); + assert.equal(w.from, "Old"); + assert.deepEqual(w.payload, { + ok: true, + session: { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "New", titleCustom: true }, + }); + assert.equal(tab1.sessionTitle, "New"); + assert.equal(tab2.sessionTitle, "New", "a tab bound to the same engine sid sees the new label too"); + assert.equal(tab3.sessionTitle, "B", "an unrelated tab is untouched"); + assert.deepEqual(pushes.sort(), ["tab-1", "tab-2"]); + // The rename touches NO engine surface: no SQL, no kill, no cache + // drop. Only the tree cache, because the sidebar projects titles + // from the engine and would otherwise show a stale one. + assert.deepEqual(journal, ["invalidate-tree", "push:tab-1", "push:tab-2"]); + }); + + test("a bare mvs_ id gets an overlay record; an unknown id is a 404 outcome", async (t) => { + // `setupMocks` owns `lib/sessions.js` in this test context and + // node:test refuses a second registration for the same specifier + // (ERR_INVALID_STATE), so the store comes from the shared helper's + // mutable holder. M3-B5 added `ensureOverlayForMcodeSid` to that + // helper's namespace for exactly this case; the fixture therefore + // observes the real single-identity behaviour instead of a private + // stub that could drift from it. + const { mod, clients } = await loadWritePath(t, { records: [] }); + const w = await mod.applyEngineSessionRename({ id: ORPHAN_SID, title: "Named", cid: "tab-1" }); + assert.equal(w.outcome, "ok"); + assert.equal(w.matchKind, "orphan_mcode"); + assert.equal(w.from, "Mcode session", "the placeholder title the overlay was born with"); + assert.equal(w.to, "Named"); + assert.equal(w.item.id, ORPHAN_SID, "the overlay's webui id IS the engine sid (single identity)"); + assert.equal(w.item.mcodeSessionId, ORPHAN_SID); + assert.equal(w.item.titleCustom, true); + assert.deepEqual(w.payload, { + ok: true, + session: { id: ORPHAN_SID, mcodeSessionId: ORPHAN_SID, title: "Named", titleCustom: true }, + }); + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.deepEqual( + getSessionsStore().map((r) => r.id), + [ORPHAN_SID], + "and the overlay is PERSISTED — a rename that is not saved is a label the next load loses", + ); + // No workspace is stamped onto someone else's record + // (webui-parity 63, defect F): the fixture's store never carried a + // workspace argument and the overlay's is "". + assert.equal(getSessionsStore()[0].workspace, ""); + + // A webui uuid that resolves to nothing is a 404, and it must NOT + // fabricate a record — a wrong id should say so. + const missing = await mod.applyEngineSessionRename({ id: "no-such-id", title: "Named", cid: "tab-1" }); + assert.equal(missing.outcome, "not_found"); + assert.deepEqual(missing.payload, { ok: false, error: "session not found" }); + assert.equal( + getSessionsStore().length, + 1, + "no second overlay was fabricated for the unknown id", + ); + void clients; + }); + + test("rename never throws a capability error, whatever the provider says", async (t) => { + // The end-to-end statement of the `null` declaration row: a + // provider that has NO session CRUD at all cannot stop a rename, + // because a rename does not ask the engine for anything. + await setupMocks(t, { acp: {}, sessions: { initial: [{ id: "webui-A", title: "Old", chat: [] }] } }); + t.mock.module(absPath("engine/index.js"), { + namedExports: { + DEFAULT_ENGINE_PROVIDER_ID: "fixture-provider", + getEngineProvider: () => ({ + id: "fixture-provider", + transport: "runtime", + capabilities: { sessionCrud: { level: "none", reason: "fixture: no CRUD at all" } }, + }), + }, + }); + const mod = await import(`${absPath("engine/session-writes.js")}?ren=${bust++}`); + // planEngineSessionDelete WOULD throw here — that is the point of + // the hard row. Rename must not. + await assert.rejects(() => mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME })); + const w = await mod.applyEngineSessionRename({ id: "webui-A", title: "New", cid: "tab-1" }); + assert.equal(w.outcome, "ok"); + assert.equal(w.gate.gate, "no-capability-key"); + }); + }); + + // --------------------------------------------------------------------- + // 7. The route — and the proof that the facade mock took + // --------------------------------------------------------------------- + + describe("the routes ask the facade and keep their own HTTP contract", () => { + // Every export `routes/sessions.js` binds from the facade. A + // `mock.module` that omits one of these makes the route fail at + // INSTANTIATION with a `SyntaxError` that reads like a product bug; + // the ones a case does not want are filled with throwers. + const FACADE_EXPORTS = [ + "ORPHAN_STALE_MS", + "SESSION_WRITE_ENDPOINTS", + "applyDeletedSessionToClientState", + "applyEngineSessionRename", + "applyRenamedSessionToClientState", + "assertSessionWriteCapability", + "clientMatchesDeletedSession", + "clientMatchesRenamedSession", + "commitEngineOrphanSessionDelete", + "commitEngineSessionDelete", + "isMcodeSessionId", + "isOrphanSessionRecord", + "planEngineSessionDelete", + "previewEngineSessionDelete", + "readOrphanSessionWriteIds", + "resolveSessionTarget", + "resolveSessionWriteProvider", + "selectOrphanSessionIds", + ]; + const NOT_STUBBED = (name) => () => { + throw new Error(`B5 route test called ${name}, which this case did not stub`); + }; + function mockFacade(t, overrides) { + // The route legitimately calls one PURE facade export before it + // calls any writer: `isMcodeSessionId`, for the arrival log line + // and the `authorize()` context. It is filled with the REAL + // implementation rather than a thrower, because it has no side + // effects and a stubbed copy could disagree with the engine module + // the CONTROL test exercises. Every export that WRITES keeps the + // thrower, so an unexpected mutation stays loud. + const namedExports = { + isMcodeSessionId: (id) => typeof id === "string" && /^mvs_[a-f0-9]{32}$/.test(id), + ORPHAN_STALE_MS: 24 * 60 * 60 * 1000, + }; + for (const name of FACADE_EXPORTS) { + if (namedExports[name] === undefined) namedExports[name] = NOT_STUBBED(name); + } + Object.assign(namedExports, overrides); + t.mock.module(absPath("engine/session-writes.js"), { namedExports }); + } + + test("#7 writes the facade's 200 body verbatim, with the charset header", async (t) => { + await setupMocks(t, { acp: {} }); + const payload = { + ok: true, + deleted: "webui-A", + matchKind: "webuiId", + dryRun: false, + remaining: 3, + mcodeDbDel: { ok: true, log: ["local_runtime_sessions:1"] }, + }; + mockFacade(t, { + planEngineSessionDelete: async () => ({ + id: "webui-A", + records: [], + index: 0, + matchKind: "webuiId", + target: { id: "webui-A" }, + isOrphan: false, + chatLen: 0, + gate: { gate: "checked" }, + transport: RUNTIME, + }), + commitEngineSessionDelete: async () => ({ + deletedItem: { id: "webui-A", title: "A" }, + records: [{}, {}, {}], + mcodeDbDel: payload.mcodeDbDel, + touchedCids: ["tab-1"], + payload, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession( + { url: "/api/sessions/webui-A" }, + res, + { cs: {}, cid: "tab-1", pathname: "/api/sessions/webui-A" }, + ), + ); + assert.equal(res.written[0].status, 200); + assert.equal( + res.written[0].headers["Content-Type"], + "application/json; charset=utf-8", + ); + assert.equal(res.written[1].body, JSON.stringify(payload)); + }); + + // Table-driven: the status and the Content-Type per branch. The + // charset is NOT uniform in the pre-facade code — the 400/404/500 + // branches send bare `application/json` while the 200/403 branches + // send the charset form — and a refactor that "tidied" that would be + // a silent contract change, so the exact pair is pinned per branch. + const STATUSES = [ + ["400", { "Content-Type": "application/json" }, "/api/sessions/"], + ["404", { "Content-Type": "application/json" }, "/api/sessions/no-such-webui-id"], + ]; + for (const [status, headers, pathname] of STATUSES) { + test(`#7 answers ${status} with ${JSON.stringify(headers)} — unchanged`, async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + planEngineSessionDelete: async () => ({ + id: "no-such-webui-id", + records: [], + index: -1, + matchKind: null, + target: null, + isOrphan: true, + chatLen: 0, + gate: { gate: "checked" }, + transport: RUNTIME, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession({ url: pathname }, res, { + cs: {}, + cid: "tab-1", + pathname, + }), + ); + assert.equal(res.written[0].status, Number(status)); + assert.deepEqual(res.written[0].headers, headers); + if (status === "404") { + assert.deepEqual(JSON.parse(res.written[1].body), { + ok: false, + error: "session not found", + }); + } + }); + } + + test("#7 still 403s on a declined authorize(), before anything is mutated", async (t) => { + await setupMocks(t, { acp: {} }); + let committed = false; + mockFacade(t, { + planEngineSessionDelete: async () => ({ + id: "webui-A", + records: [{ id: "webui-A", title: "A", chat: ["● hi"] }], + index: 0, + matchKind: "webuiId", + target: { id: "webui-A" }, + isOrphan: false, + chatLen: 1, + gate: { gate: "checked" }, + transport: RUNTIME, + }), + commitEngineSessionDelete: async () => { + committed = true; + return { payload: { ok: true } }; + }, + }); + const route = await loadRoute(); + const res = mkRes(); + await withDecisions( + () => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + { approve: false }, + ); + assert.equal(res.written[0].status, 403); + assert.equal( + JSON.parse(res.written[1].body).error, + "authorize declined", + ); + assert.equal(committed, false, "a declined gate must not reach the commit at all"); + }); + + test("#4 keeps its three 400 bodies and never reaches the facade", async (t) => { + await setupMocks(t, { acp: {} }); + let called = false; + mockFacade(t, { + applyEngineSessionRename: async () => { + called = true; + return { outcome: "ok" }; + }, + }); + const route = await loadRoute(); + // Table-driven over the three validation failures, all of which are + // request validation and therefore stay in the route. + const CASES = [ + [{ title: "New" }, "id required"], + [{ id: "webui-A" }, "title required"], + [{ id: "webui-A", title: " " }, "title required"], + [{ id: "webui-A", title: "x".repeat(201) }, "title too long (max 200)"], + ]; + for (const [body, error] of CASES) { + const res = mkRes(); + await route.handleRenameSession(jsonReq(body), res, { cid: "tab-1" }); + assert.equal(res.written[0].status, 400, JSON.stringify(body).slice(0, 40)); + assert.equal( + res.written[0].headers["Content-Type"], + "application/json; charset=utf-8", + ); + assert.deepEqual(JSON.parse(res.written[1].body), { ok: false, error }); + } + assert.equal(called, false, "validation happens before the facade is consulted"); + }); + + test("#4 answers 404 for the facade's not_found outcome and 200 otherwise", async (t) => { + await setupMocks(t, { acp: {} }); + let current = { + outcome: "not_found", + payload: { ok: false, error: "session not found" }, + matchKind: null, + from: "", + to: "T", + item: null, + }; + mockFacade(t, { applyEngineSessionRename: async () => current }); + const route = await loadRoute(); + for (const [outcome, expectedStatus, body] of [ + ["not_found", 404, { ok: false, error: "session not found" }], + [ + "ok", + 200, + { ok: true, session: { id: "webui-A", mcodeSessionId: null, title: "T", titleCustom: true } }, + ], + ]) { + current = + outcome === "not_found" + ? { outcome, payload: body, matchKind: null, from: "", to: "T", item: null } + : { outcome, payload: body, matchKind: "webuiId", from: "Old", to: "T", item: { id: "webui-A", title: "T" } }; + const res = mkRes(); + await route.handleRenameSession(jsonReq({ id: "webui-A", title: "T" }), res, { + cid: "tab-1", + }); + assert.equal(res.written[0].status, expectedStatus, outcome); + assert.deepEqual(JSON.parse(res.written[1].body), body); + } + }); + + test("#6 writes the facade's preview body byte-for-byte", async (t) => { + await setupMocks(t, { acp: {} }); + const ids = ["old-1", "old-2"]; + mockFacade(t, { + readOrphanSessionWriteIds: async () => ({ + ids, + payload: { ok: true, dryRun: true, count: 2, ids }, + gate: { gate: "checked", enforcement: "hard" }, + transport: RUNTIME, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCleanupOrphans({ url: "/api/sessions/cleanup-orphans?dryRun=true" }, res, { + cid: "tab-1", + }); + assert.equal(res.written[0].status, 200); + assert.equal( + res.written[0].headers["Content-Type"], + "application/json; charset=utf-8", + ); + // The batch's byte-for-byte red line, asserted at the HTTP edge + // and not only inside the facade. + assert.equal( + res.written[1].body, + '{"ok":true,"dryRun":true,"count":2,"ids":["old-1","old-2"]}', + ); + }); + + test("#6's no-op real path keeps its own four-key body", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readOrphanSessionWriteIds: async () => ({ + ids: [], + payload: { ok: true, dryRun: true, count: 0, ids: [] }, + gate: { gate: "checked" }, + transport: RUNTIME, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCleanupOrphans({ url: "/api/sessions/cleanup-orphans" }, res, { + cid: "tab-1", + }); + assert.equal(res.written[0].status, 200); + assert.equal( + res.written[1].body, + '{"ok":true,"dryRun":false,"deleted":0,"ids":[]}', + ); + }); + + // ---- proof the facade mock actually took --------------------------- + + test("PROOF: a marker error escapes the untouched #7 route", async (t) => { + // Without a fresh `?bust=` re-import, `mock.module` would leave the + // route holding the PREVIOUS test's live binding, the marker would + // never be thrown, and this assertion would fail — which is the + // point: it is the only assertion in this section that cannot pass + // by accident. + await setupMocks(t, { acp: {} }); + const marker = new Error("B5-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + planEngineSessionDelete: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await withDecisions(() => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, mkRes(), { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + ); + } catch (err) { + caught = err; + } + assert.ok( + caught, + "the route swallowed the facade error — either the mock did not take, or the route grew a catch", + ); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + + test("PROOF: a marker error escapes the untouched #6 route", async (t) => { + // The same proof for the second facade consumer. A single proof + // would not cover a route that imported a different subset of the + // module. + await setupMocks(t, { acp: {} }); + const marker = new Error("B5-SWEEP-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + readOrphanSessionWriteIds: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleCleanupOrphans({ url: "/api/sessions/cleanup-orphans" }, mkRes(), { + cid: "tab-1", + }); + } catch (err) { + caught = err; + } + assert.ok(caught, "the sweep route swallowed the facade error"); + assert.equal(caught, marker); + }); + + test("CONTROL: with no facade mock, #7 reaches the real write path", async (t) => { + // The other half of the proof. A `?bust=` re-import under a fresh + // test hook gives a route bound to the REAL facade, so the request + // runs the actual plan → commit sequence against the mocked store + // and the real SQL deleter. If this answered from a mock, the two + // PROOF cases above would be proving nothing. + const { dbCalls } = await loadWritePath(t, { + records: [{ id: "webui-A", title: "A", chat: ["● hi"] }], + }); + const route = await loadRoute(); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + ); + assert.equal(res.written[0].status, 200); + const body = JSON.parse(res.written[1].body); + assert.equal(body.ok, true); + assert.equal(body.deleted, "webui-A"); + assert.equal(body.matchKind, "webuiId"); + // A webui-only record has no engine sid, so the deleter is not + // asked — and the response says so rather than inventing a result. + assert.equal(body.mcodeDbDel, null); + assert.equal(dbCalls.length, 0); + }); + }); + + // --------------------------------------------------------------------- + // 8. The audit chain, across the route/facade boundary + // --------------------------------------------------------------------- + + describe("RED LINE — the audit chain is intact across the split", () => { + // The one thing the plan→commit split could have broken: the + // write-ahead intent line has to land BETWEEN the plan and the + // mutation. These run the REAL route against the REAL facade, with + // only the audit sink and the SQL layer journalled, and assert the + // ORDER of the three events. + async function loadAuditedRoute(t, options = {}) { + await setupMocks(t, { acp: {}, sessions: { initial: options.records || [] } }); + const events = []; + mockAll( + t, + "lib/events.js", + { + append: (kind, data) => { + events.push({ kind, data }); + }, + }, + ["append", "read", "readAll", "verifyChain", "EVENTS_PATH"], + ); + mockAll( + t, + "lib/mcode-session-delete.js", + { + deleteMcodeSessionFromDb: (sid, o) => { + events.push({ kind: `sql(${sid},dryRun=${!!o.dryRun})` }); + return { ok: true, outcome: "deleted", log: ["local_runtime_sessions:1"], totalRowsDeleted: 1, tablesAbsent: 0 }; + }, + }, + ["deleteMcodeSessionFromDb", "MCODE_SESSION_DELETE_TABLES"], + ); + mockAll( + t, + "lib/session-tree.js", + { invalidateSessionTree: () => {}, getSessionTree: () => ({ ok: true, tree: [] }) }, + ["getSessionTree", "invalidateSessionTree"], + ); + mockAll( + t, + "lib/state-bus.js", + { + clients: new Map(), + pushStateFor: () => {}, + runChatViewChat: () => ({}), + makeClientState: () => ({ usage: {} }), + }, + ["clients", "pushStateFor", "runChatViewChat", "makeClientState", "setState", "getClient", "sseByCid", "pushAlert"], + ); + return { events, route: await loadRoute() }; + } + + test("a real delete writes intent BEFORE the rows go, and the outcome after", async (t) => { + const { events, route } = await loadAuditedRoute(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● hi", "● there"] }, + ], + }); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + ); + assert.equal(res.written[0].status, 200); + assert.deepEqual( + events.map((e) => e.kind), + ["session.delete.intent", "sql(mvs_sid_A,dryRun=false)", "session.delete"], + "the intent line lands before the mutation, the outcome line after it", + ); + // The intent payload carries exactly the three facts the authorize + // modal showed the user, which is the point of computing them in + // the plan and passing them through unchanged. + assert.equal(events[0].data.payload.matchKind, "webuiId"); + assert.equal(events[0].data.payload.isOrphan, false); + assert.equal(events[0].data.payload.chatLen, 2); + assert.ok(events[0].data.payload.decidedBy, "and the authorizer's decision"); + assert.equal(events[2].data.payload.dryRun, false); + assert.equal(events[2].data.payload.title, "A", "the title is logged — it was user-visible in the sidebar"); + assert.ok("touchedCids" in events[2].data.payload, "the fan-out effect is recorded"); + }); + + test("a declined authorize writes NO intent line and never touches the engine", async (t) => { + const { events, route } = await loadAuditedRoute(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + const res = mkRes(); + await withDecisions( + () => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + { approve: false }, + ); + assert.equal(res.written[0].status, 403); + assert.deepEqual(events, [], "a refused delete is not an audited one — nothing was attempted"); + }); + + test("a dryRun preview is audited as a PREVIEW and never mutates", async (t) => { + const { events, route } = await loadAuditedRoute(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession({ url: "/api/sessions/webui-A?dryRun=true" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + ); + assert.equal(res.written[0].status, 200); + assert.deepEqual( + events.map((e) => e.kind), + ["sql(mvs_sid_A,dryRun=true)", "session.delete"], + ); + assert.equal( + events[1].data.payload.dryRun, + true, + "the dryRun marker is what lets an operator tell a preview from a real delete", + ); + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.equal(getSessionsStore().length, 1, "and the record is still there"); + }); + + test("a rename is audited with from → to and the resolved match kind", async (t) => { + const { events, route } = await loadAuditedRoute(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "Old", chat: [] }], + }); + const res = mkRes(); + await route.handleRenameSession(jsonReq({ id: "mvs_sid_A", title: "New" }), res, { + cid: "tab-1", + }); + assert.equal(res.written[0].status, 200); + assert.deepEqual(events.map((e) => e.kind), ["session.rename"]); + assert.equal(events[0].data.payload.matchKind, "mcodeSessionId", "renamed through the engine sid"); + assert.equal(events[0].data.payload.from, "Old"); + assert.equal(events[0].data.payload.to, "New"); + assert.equal(events[0].data.payload.mcodeSessionId, "mvs_sid_A"); + }); + }); +}); + diff --git a/packages/webui/test/lib/mcode-rpc.check.mjs b/packages/webui/test/lib/mcode-rpc.check.mjs index a777ad7a..c6abeaed 100644 --- a/packages/webui/test/lib/mcode-rpc.check.mjs +++ b/packages/webui/test/lib/mcode-rpc.check.mjs @@ -3,9 +3,17 @@ // // Why this test exists: mcode-rpc.js is the clean wrapper around mcode 0.1.5 // acp JSON-RPC. PERMISSION_MODES + mcodePermissionToWebui are the enum used -// by routes/model.js. MCODE_ACP_CAPABILITIES drives the capability detection -// in routes/protocol.js. Bugs here = wrong permission labels shown to user -// or capability detection thinks mcode supports methods it doesn't. +// by routes/model.js. Bugs here = wrong permission labels shown to user. +// +// MCODE_ACP_CAPABILITIES is KNOWN DEBT as of M3-B4: `GET +// /api/protocol/capabilities` used to serve this table under `capabilities` +// and now serves the engine's DECLARED 14-key capability object instead +// (a user-authorised endpoint contract change — see +// engine/capability-reads.js). The constant is still exported and still +// pinned here, because it remains a true statement about the ENGINE's +// ACP surface and `docs/CAPABILITIES.md` cites it as one. It has no +// webui consumer left; deleting it is a separate dead-code decision, not +// a side effect of the replacement. // // Test strategy: NO setupMocks. We import the REAL mcode-rpc.js so we test // the actual exports. We only test the safe-to-call functions: diff --git a/packages/webui/webapp/app/error.tsx b/packages/webui/webapp/app/error.tsx index d0d1308d..c6985df5 100644 --- a/packages/webui/webapp/app/error.tsx +++ b/packages/webui/webapp/app/error.tsx @@ -176,7 +176,7 @@ export default function RouteError({ error, reset }: ErrorBoundaryProps) { type="button" onClick={onReset} data-testid="route-error-reload" - className="h-8 rounded-[8px] bg-bg_interaction_primary_default px-4 text-sm font-weight_medium text-text_default_inverted_static transition-colors hover:bg-bg_interaction_primary_hover" + className="h-8 rounded-[8px] bg-bg_interaction_primary_default px-4 text-sm font-weight_medium text-text_label_primary_default transition-colors hover:bg-bg_interaction_primary_hover" > {t("webui.errorBoundary.reload")} diff --git a/packages/webui/webapp/components/activity-group.tsx b/packages/webui/webapp/components/activity-group.tsx index 4d826c2c..043bfd7f 100644 --- a/packages/webui/webapp/components/activity-group.tsx +++ b/packages/webui/webapp/components/activity-group.tsx @@ -187,18 +187,33 @@ export function ActivityGroup({ // While a call is in flight upstream names it instead of listing categories // ("已使用 3 次工具|bash"); once the turn settles it lists the per-category // contributions joined with ", " (「查看 2 个文件, 执行 1 条命令」). + // + // The `thinking` category is counted but NOT printed. UAT fix: the group + // header and the turn bar are both inside the same turn, and both used the + // same 「思考 N 次」 wording for the same count, so a thinking-only run read + // 「思考 1 次」 twice (header above the thought, turn bar under the answer). + // The turn bar keeps it — that is where the reference `WebuiTurnProcess` + // puts it (`processSummaryParts` in the desktop `AssistantBody`), and it is + // the one row that survives the group's collapse. A run with nothing but + // thoughts therefore falls back to the qualitative 「思考过程」 label: still + // one label for the fold, and never a second copy of the count. + const printable = summary.contributions.filter( + (entry) => entry.category !== "thinking", + ); const label = summary.activeTool ? t("activity.activeTool") .replace("{{count}}", String(summary.tools)) .replace("{{tool}}", summary.activeTool) - : summary.contributions - .map((entry) => - t((SUMMARY_CATEGORY_KEY[entry.category] ?? "activity.usedTools") as MessageKey).replace( - "{{count}}", - String(entry.count), - ), - ) - .join(", "); + : printable.length > 0 + ? printable + .map((entry) => + t((SUMMARY_CATEGORY_KEY[entry.category] ?? "activity.usedTools") as MessageKey).replace( + "{{count}}", + String(entry.count), + ), + ) + .join(", ") + : t("activity.thoughtProcess"); // The forced-open state (active tool, or streaming thought). While it // holds, a user click on the summary must not collapse the group. diff --git a/packages/webui/webapp/components/add-model-dialog.tsx b/packages/webui/webapp/components/add-model-dialog.tsx index b18338db..930d9a25 100644 --- a/packages/webui/webapp/components/add-model-dialog.tsx +++ b/packages/webui/webapp/components/add-model-dialog.tsx @@ -688,7 +688,7 @@ export function AddModelDialogForm({ disabled={busy || (!skipTest && formTest?.status !== "ok")} aria-busy={busy || undefined} onClick={onCommit} - className="h-9 min-w-20 rounded-lg bg-bg_interaction_primary_default px-5 text-sm font-weight_medium text-text_default_inverted_static shadow-[var(--shadow_default)] transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" + className="h-9 min-w-20 rounded-lg bg-bg_interaction_primary_default px-5 text-sm font-weight_medium text-text_label_primary_default shadow-[var(--shadow_default)] transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" > {busy ? t("providers.saving") : t("providers.dialog.save")} @@ -1527,7 +1527,7 @@ export function FetchedModelsDialogBody({ data-testid="fetched-models-add" disabled={!presetMode || checked.size === 0} onClick={onAdd} - className="h-9 min-w-20 rounded-lg bg-bg_interaction_primary_default px-5 text-sm font-weight_medium text-text_default_inverted_static shadow-[var(--shadow_default)] transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" + className="h-9 min-w-20 rounded-lg bg-bg_interaction_primary_default px-5 text-sm font-weight_medium text-text_label_primary_default shadow-[var(--shadow_default)] transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" > {t("providers.fetched.add")} diff --git a/packages/webui/webapp/components/markdown-html.tsx b/packages/webui/webapp/components/markdown-html.tsx index 011db3d6..ebb37c9c 100644 --- a/packages/webui/webapp/components/markdown-html.tsx +++ b/packages/webui/webapp/components/markdown-html.tsx @@ -93,6 +93,15 @@ export function MarkdownHtml({ html }: { html: string }) { ); } +/** + * Elements whose children React accepts only as elements — never as text. + * + * Mirrors React DOM's own `validateTextNesting` table for the tags + * `lib/markdown.ts` allowlists. `` and `` are absent from + * that allowlist, so they are absent here too. + */ +const TABLE_STRUCTURE_TAGS = new Set(["table", "thead", "tbody", "tfoot", "tr"]); + /** * Walk a sanitised HTML string and convert it to a React tree. * @@ -109,12 +118,19 @@ export function MarkdownHtml({ html }: { html: string }) { * attributes (`class`, `href`, `title`, `align`) so a markdown * document looks the same as before — only the mermaid fences * are upgraded from inert HTML to a live component. + * - drops whitespace-only text under a table-family element (see + * `TABLE_STRUCTURE_TAGS`): React refuses those children, and table + * layout never paints them. * * Returns a single `dangerouslySetInnerHTML` element from inside the * tree on SSR (when `DOMParser` is undefined); the prerender still * produces a non-empty HTML response. + * + * Exported for the render-harness test, which drives this walker over a + * real `marked` table and asserts the React tree it produces — the only + * place the hydration contract is actually checkable without a browser. */ -function htmlToReact(html: string, theme: "light" | "dark"): ReactNode { +export function htmlToReact(html: string, theme: "light" | "dark"): ReactNode { if (typeof DOMParser === "undefined") { return (
{ + const walkChildren = (parent: Element | Document, parentTag: string | null): ReactNode[] => { const out: ReactNode[] = []; for (const child of [...parent.childNodes]) { if (child.nodeType === 3 /* text */) { const text = child.textContent ?? ""; if (text.length === 0) continue; + // UAT fix — `validateTextNesting` (React DOM, dev builds) rejects + // ANY text node under a table-family element, whitespace included, + // and answers with "In HTML, whitespace text nodes cannot be a + // child of . This will cause a hydration error." `marked` + // indents every table line, so the sanitised HTML the walker is + // handed carries those newlines as real text nodes under + //
///. Table layout collapses inter-tag + // whitespace and never paints it, so dropping it here removes the + // console flood and the hydration error without moving a pixel. + if (parentTag !== null && TABLE_STRUCTURE_TAGS.has(parentTag) && /^\s*$/.test(text)) { + continue; + } out.push(text); continue; } @@ -178,7 +206,7 @@ function htmlToReact(html: string, theme: "light" | "dark"): ReactNode { props[attr.name] = attr.value; } } - const children = walkChildren(el); + const children = walkChildren(el, tag); out.push( createElement(tag, props, children.length > 0 ? children : undefined), ); @@ -186,7 +214,7 @@ function htmlToReact(html: string, theme: "light" | "dark"): ReactNode { return out; }; - return walkChildren(doc.body); + return walkChildren(doc.body, null); } /** diff --git a/packages/webui/webapp/components/modals.tsx b/packages/webui/webapp/components/modals.tsx index 016f91cd..abc028d6 100644 --- a/packages/webui/webapp/components/modals.tsx +++ b/packages/webui/webapp/components/modals.tsx @@ -214,7 +214,7 @@ function AskModal({ t }: { t: (key: MessageKey) => string }) { "flex flex-none items-center justify-center text-caption-small-strong text-text_default_secondary", multiSelect ? isPicked - ? "size-5 rounded bg-bg_interaction_primary_default text-text_default_inverted_static" + ? "size-5 rounded bg-bg_interaction_primary_default text-text_label_primary_default" : "size-5 rounded border border-border_default bg-bg_grouped_primary" : "size-5 rounded-full bg-bg_grouped_primary", ].join(" ")} @@ -392,6 +392,20 @@ export function Modal({ ); } +/** + * The primary (filled) button of a confirm modal. + * + * UAT fix — the label colour. The label used + * `text-text_label_primary_default`, which the design system defines as + * "text on an inverted surface" and never re-themes: it stays near-white + * in BOTH `:root` (95%) and `.dark` (80%) — see `styles/tokens.css`. The + * dark theme inverts `--bg_interaction_primary_default` to `--gray_0` + * (white), so the pair composited to white on white and the 「批准」 label + * disappeared — an unlabelled button, not a missing string. The token that + * actually pairs with this background is `--text_label_primary_default` + * (white on light, near-black on dark), the same pairing the upstream + * `.mavis-button.black` rule uses (`styles/official-utilities.css`). + */ function PrimaryButton({ disabled, onClick, @@ -406,7 +420,7 @@ function PrimaryButton({ type="button" disabled={disabled} onClick={onClick} - className="h-8 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_default_inverted_static transition-colors hover:bg-bg_interaction_primary_hover disabled:opacity-50" + className="h-8 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_label_primary_default transition-colors hover:bg-bg_interaction_primary_hover disabled:opacity-50" > {children} diff --git a/packages/webui/webapp/components/panels.tsx b/packages/webui/webapp/components/panels.tsx index c714cb6b..ccc42443 100644 --- a/packages/webui/webapp/components/panels.tsx +++ b/packages/webui/webapp/components/panels.tsx @@ -2979,7 +2979,7 @@ function WorkspaceBrowseTab({ disabled={busy || !listing?.dir} onClick={() => void pick()} data-testid="workspace-picker-confirm" - className="h-8 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_default_inverted_static transition-colors hover:bg-bg_interaction_primary_hover disabled:opacity-50" + className="h-8 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_label_primary_default transition-colors hover:bg-bg_interaction_primary_hover disabled:opacity-50" > {t("workspace.picker.useWorkspace")} diff --git a/packages/webui/webapp/components/provider-management.tsx b/packages/webui/webapp/components/provider-management.tsx index e8b06409..ac6dde24 100644 --- a/packages/webui/webapp/components/provider-management.tsx +++ b/packages/webui/webapp/components/provider-management.tsx @@ -537,7 +537,7 @@ export function ProviderManagementPanel({ disabled={busy || !validation.ok} aria-busy={busy || undefined} onClick={() => void save()} - className="flex h-8 items-center gap-1.5 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_default_inverted_static transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" + className="flex h-8 items-center gap-1.5 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_label_primary_default transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" > {busy ? ( { assert.match(html, /data-testid="thinking-summary-icon"/); assert.match(html, /data-tool-icon-type="thinking"/); assert.match(html, /已完成推理/); - assert.doesNotMatch(html, /思考过程/); + // Scoped to the thinking SUMMARY ROW, which is what this test names: + // the reference's `WebuiThinkingBlock` defaults `showDetailHeading` to + // false, so the body must not open with a 「思考过程」 heading. The + // assertion used to run over the whole group markup, which made it a + // de-facto ban on the string anywhere — including the group header, + // which now labels a thoughts-only run 「思考过程」 (see the UAT block + // below). Reading the whole document here tested more than it claimed. + const summaryStart = html.indexOf('data-testid="thinking-summary"'); + const rowStart = html.lastIndexOf("", summaryStart); + assert.ok(rowStart >= 0 && rowEnd > summaryStart, "the thinking summary row is missing"); + assert.doesNotMatch( + html.slice(rowStart, rowEnd), + /思考过程/, + "the thinking summary row must carry the status copy, not a body heading", + ); }); test("the body renders through the Markdown pipeline, not plain text", () => { @@ -761,6 +776,88 @@ describe("TurnProcessDisclosure — the turn bar (D6, PR3)", () => { }); }); +describe("UAT fix — 「思考 N 次」 is labelled once per turn, not twice", () => { + /** The header row alone: the assertion must not be satisfied (or + * broken) by anything in the folded body. */ + const headerOf = (html: string): string => { + const start = html.indexOf('data-testid="activity-group-header"'); + const open = html.lastIndexOf("", start); + assert.ok(open >= 0 && end > start, "the group header row is missing"); + return html.slice(open, end); + }; + + test("a thoughts-only run labels the group qualitatively, not with the count", () => { + const header = headerOf( + renderGroup([thinkingBlock("let me check")], thoughtsOnlySummary), + ); + // Regression: the header used to read 「思考 1 次」 — byte-identical to + // the turn bar under the same turn's answer. + assert.doesNotMatch( + header, + /思考 1 次/, + "the group header must not repeat the turn bar's thinking count", + ); + assert.match(header, /思考过程/); + }); + + test("a mixed run keeps its tool contributions and drops only the count", () => { + const header = headerOf( + renderGroup( + [thinkingBlock("let me check"), richToolBlock()], + mixedSummary, + ), + ); + assert.doesNotMatch(header, /思考 1 次/); + assert.match(header, /执行 1 条命令/); + }); + + test("the turn bar still carries the count (the one surviving label)", () => { + // The count must not simply be deleted: the turn bar is the row that + // survives the group's collapse, and it is where the reference + // `WebuiTurnProcess` puts it. + const html = renderToStaticMarkup( + createElement(TurnProcessDisclosure, { + stats: { thinking: 1, tools: 0, answerChars: 0 }, + processedDurationMs: 5000, + t, + }), + ); + assert.match(html, /思考 1 次,共执行 5 秒/); + }); + + test("a whole turn states the count exactly once", () => { + // The rendered end-to-end shape the UAT screenshot captured: a + // thoughts-only group, then the turn bar for the same turn. The bar + // carries the summary twice in its markup (the `data-summary-text` + // mirror plus the visible text), so the attribute is stripped first — + // this counts what the reader sees, not what the DOM stores. + const group = renderGroup([thinkingBlock("let me check")], thoughtsOnlySummary); + const bar = renderToStaticMarkup( + createElement(TurnProcessDisclosure, { + stats: { thinking: 1, tools: 0, answerChars: 0 }, + processedDurationMs: 5000, + t, + }), + ); + const visible = (group + bar).replace(/ data-summary-text="[^"]*"/g, ""); + const occurrences = visible.match(/思考 1 次/g) ?? []; + assert.equal( + occurrences.length, + 1, + `「思考 1 次」 must appear once per turn, found ${occurrences.length}`, + ); + }); + + test("the leading icon of a thoughts-only run is unchanged", () => { + // The fix filters the LABEL, not the summary: `iconType` still comes + // from the `thinking` contribution, so the ⓘ glyph does not regress to + // the generic tool icon. + const html = renderGroup([thinkingBlock("let me check")], thoughtsOnlySummary); + assert.match(html, /data-tool-icon-type="thinking"/); + }); +}); + describe("chat.tsx wiring — the streaming derivation stays put", () => { test("the tail-run derivation feeds streaming and startedAtMs into the group", () => { assert.match(chatSource, /const streamingActivityIndex = useMemo/); diff --git a/packages/webui/webapp/test/helpers/dom-shim.ts b/packages/webui/webapp/test/helpers/dom-shim.ts new file mode 100644 index 00000000..bae13eb4 --- /dev/null +++ b/packages/webui/webapp/test/helpers/dom-shim.ts @@ -0,0 +1,165 @@ +// webapp/test/helpers/dom-shim.ts +// +// A `DOMParser` stand-in for the Node test runner, built on the `parse5` +// already in the dependency tree. +// +// Why it exists +// ------------- +// +// `components/markdown-html.tsx` converts the sanitised markdown into a +// React tree by walking a `DOMParser` document. Node has no `DOMParser`, +// and the project deliberately does not pull in jsdom/happy-dom (a +// multi-megabyte dependency for one walker). That left the walker +// untested: `MarkdownHtml` silently took its SSR `dangerouslySetInnerHTML` +// branch in every unit test, so a defect in the walker's output — the +// whitespace text nodes React refuses under `
`, for one — reached +// production with a green gate. +// +// The shim exposes exactly the DOM Level 1 surface the walker touches: +// `parseFromString`, `body`, `childNodes`, `nodeType`, `tagName`, +// `attributes`, `classList.contains`, `textContent`, `previousSibling`. +// It is deliberately not a general DOM: a walker that grows a new DOM +// dependency fails here loudly (undefined method) instead of silently +// testing against a fake that agrees with it. +// +// On parse5: the workspace has no HTML parser of its own and does not +// depend on one. `parse5` arrives through the Next.js tree and is pinned +// in `pnpm-lock.yaml`; it is reached with `createRequire` rather than an +// `import` because `@mavis/webui` does not declare it, and an undeclared +// `import` would break `webapp:typecheck` with TS7016. The coupling is +// test-only and stated here rather than hidden: if the transitive copy ever +// disappears, the shim throws the message below and the affected tests +// fail loudly instead of quietly passing against a stub. + +import { createRequire } from "node:module"; + +interface Parse5 { + parse(html: string): unknown; +} + +const requireFromHere = createRequire(import.meta.url); + +function loadParse5(): Parse5 { + try { + return requireFromHere("parse5") as Parse5; + } catch (cause) { + throw new Error( + "the markdown DOM shim needs `parse5`, which no longer resolves from " + + "packages/webui. Declare it as a devDependency of @mavis/webui, or " + + "replace this shim.", + { cause }, + ); + } +} + +const parse5 = loadParse5(); + +/** parse5 node, narrowed to the fields the shim reads. */ +interface Parse5Node { + nodeName: string; + value?: string; + tagName?: string; + attrs?: { name: string; value: string }[]; + childNodes?: Parse5Node[]; +} + +/** A node in the shimmed tree: DOM Level 1 fields over a parse5 node. */ +export interface ShimNode { + nodeType: number; + nodeName: string; + tagName: string; + textContent: string; + childNodes: ShimNode[]; + parentNode: ShimNode | null; + previousSibling: ShimNode | null; + nextSibling: ShimNode | null; + attributes: { name: string; value: string }[]; + classList: { contains(token: string): boolean }; +} + +/** Minimal `document` the walker consumes (`htmlToReact` reads `.body`). */ +export interface ShimDocument { + body: ShimNode; +} + +const ELEMENT_NODE = 1; +const TEXT_NODE = 3; + +function toShimNode(node: Parse5Node, parent: ShimNode | null): ShimNode { + const isText = node.nodeName === "#text"; + const isElement = node.tagName !== undefined; + + const attributes = (node.attrs ?? []).map((attr) => ({ + name: attr.name, + value: attr.value, + })); + + const shim: ShimNode = { + nodeType: isText ? TEXT_NODE : isElement ? ELEMENT_NODE : 0, + nodeName: node.nodeName, + tagName: node.tagName ?? "", + textContent: isText + ? (node.value ?? "") + : (node.childNodes ?? []).map((child) => child.value ?? "").join(""), + childNodes: [], + parentNode: parent, + previousSibling: null, + nextSibling: null, + attributes, + classList: { + contains(token: string): boolean { + const classAttr = attributes.find((attr) => attr.name === "class"); + return (classAttr?.value ?? "").split(/\s+/).includes(token); + }, + }, + }; + + shim.childNodes = (node.childNodes ?? []).map((child) => { + const childShim = toShimNode(child, shim); + const previous = shim.childNodes[shim.childNodes.length - 1]; + if (previous) previous.nextSibling = childShim; + return childShim; + }); + + return shim; +} + +/** A `DOMParser` whose `parseFromString` returns the shimmed tree. */ +export class ShimDomParser { + parseFromString(html: string, _type: "text/html"): ShimDocument { + // parse5 builds the implied // around the fragment, + // so is a grandchild of the document, not a child. + const bodyNode = findBody(parse5.parse(html) as Parse5Node); + if (!bodyNode) throw new Error("parse5 produced no for the fixture"); + return { body: toShimNode(bodyNode, null) }; + } +} + +function findBody(node: Parse5Node): Parse5Node | undefined { + if (node.nodeName === "body") return node; + for (const child of node.childNodes ?? []) { + const found = findBody(child); + if (found) return found; + } + return undefined; +} + +/** + * Run `fn` with `DOMParser` shimmed in, then restore whatever was there. + * + * The walker resolves the bare identifier `DOMParser`, so the global must + * be installed before `htmlToReact` is called. It is restored on the + * throw path too: a failing assertion must not leave a fake DOM + * installed for the rest of the file. + */ +export function withDomParserShim(fn: () => T): T { + const globals = globalThis as { DOMParser?: unknown }; + const previous = globals.DOMParser; + globals.DOMParser = ShimDomParser; + try { + return fn(); + } finally { + if (previous === undefined) delete globals.DOMParser; + else globals.DOMParser = previous; + } +} diff --git a/packages/webui/webapp/test/markdown-html-render.test.ts b/packages/webui/webapp/test/markdown-html-render.test.ts index 9ba9d7db..1604d22c 100644 --- a/packages/webui/webapp/test/markdown-html-render.test.ts +++ b/packages/webui/webapp/test/markdown-html-render.test.ts @@ -62,7 +62,7 @@ import { test, describe } from "node:test"; import assert from "node:assert/strict"; -import { renderMarkdown } from "../lib/markdown"; +import { parseMarkdown, renderMarkdown } from "../lib/markdown"; import "../lib/mermaid-renderer"; // auto-registers the mermaid language renderer import { _stripMermaidInitForTest, @@ -70,7 +70,8 @@ import { _mermaidConfigKeyForTest, _mermaidFontFamilyForTest, } from "../components/mermaid-block"; -import { findMermaidSourceBefore } from "../components/markdown-html"; +import { findMermaidSourceBefore, htmlToReact } from "../components/markdown-html"; +import { withDomParserShim } from "./helpers/dom-shim"; describe("MarkdownHtml render path — registry-side evidence", () => { test("a mermaid fence produces the placeholder pair the walker expects", () => { @@ -215,6 +216,163 @@ describe("MermaidBlock — sanitiser hooks (test-only exports)", () => { // full mermaid render path is not exercised in the unit harness. // --------------------------------------------------------------------------- +// --------------------------------------------------------------------------- +// UAT fix — the walker's output must satisfy React's `validateTextNesting`. +// +// The defect: `marked` indents every line of a GFM table, so the sanitised +// HTML carries `\n` as real text nodes under
///. +// React DOM (dev build) rejects ANY text child of those elements and logs +// "In HTML, whitespace text nodes cannot be a child of
. This will +// cause a hydration error." once per tag per page load. The walker now +// drops whitespace-only text under the table family; table layout never +// painted it, so no visual output changes. +// +// This is the first test in the file that drives the REAL walker: the +// earlier ones had to assert inputs and exported helpers because Node has +// no DOMParser. `test/helpers/dom-shim.ts` supplies one over parse5, so the +// contract is now checked where it actually lives — on the React tree the +// component mounts. +// --------------------------------------------------------------------------- + +/** The tags React's `validateTextNesting` refuses text children under. */ +const TABLE_STRUCTURE_TAGS = ["table", "thead", "tbody", "tfoot", "tr"] as const; + +type ReactLikeNode = + | string + | ReactLikeNode[] + | { type?: unknown; props?: { children?: ReactLikeNode } }; + +/** Every whitespace-only string anywhere in the tree, with its parent tag. */ +function whitespaceTextUnderTableTags( + node: ReactLikeNode, + parentTag: string | null = null, + found: { parentTag: string; text: string }[] = [], +): { parentTag: string; text: string }[] { + if (typeof node === "string") { + if (parentTag !== null && /^\s+$/.test(node)) found.push({ parentTag, text: node }); + return found; + } + // `htmlToReact` returns the body's children as one array; elements nest + // their own children as an array or a single node. + if (Array.isArray(node)) { + for (const child of node) whitespaceTextUnderTableTags(child, parentTag, found); + return found; + } + const tag = typeof node.type === "string" ? node.type : parentTag; + if (node.props?.children !== undefined) { + whitespaceTextUnderTableTags(node.props.children, tag, found); + } + return found; +} + +/** Every element tag in the tree, in document order. */ +function collectTags(node: ReactLikeNode, tags: string[] = []): string[] { + if (typeof node === "string") return tags; + if (Array.isArray(node)) { + for (const child of node) collectTags(child, tags); + return tags; + } + if (typeof node.type === "string") tags.push(node.type); + if (node.props?.children !== undefined) collectTags(node.props.children, tags); + return tags; +} + +describe("markdown-html — the table React tree has no text children", () => { + const GFM_TABLE = [ + "| 名称 | 说明 |", + "| --- | :---: |", + "| 端口 | 监听端口 |", + "| 路径 | 根路径 |", + ].join("\n"); + + test("a GFM table yields no whitespace text node under any table-family tag", () => { + // Sanitising needs a DOM too, so drive the parser directly: the walker + // is the unit under test, and the raw marked output is what it is fed + // (lib/markdown.ts#renderMarkdown hands it `sanitize(parseMarkdown(...))`, + // and the sanitiser never touches text nodes). + const tree = withDomParserShim(() => htmlToReact(parseMarkdown(GFM_TABLE), "light")); + + const offenders = whitespaceTextUnderTableTags(tree as ReactLikeNode); + assert.deepEqual( + offenders, + [], + `whitespace text nodes reached a table-family element: ${JSON.stringify(offenders)}`, + ); + }); + + test("the mutation guard — marked really does emit that whitespace", () => { + // Without this, the test above would also pass if `marked` stopped + // indenting its tables, i.e. for the wrong reason. + const html = parseMarkdown(GFM_TABLE); + assert.match(html, /
[\s\S]*\n[\s\S]*<\/table>/); + assert.match(html, /[\s\S]*\n[\s\S]*<\/tr>/); + }); + + test("the table keeps every cell — only whitespace was dropped", () => { + const tree = withDomParserShim(() => htmlToReact(parseMarkdown(GFM_TABLE), "light")); + const tags = collectTags(tree as ReactLikeNode); + assert.deepEqual( + tags, + [ + "table", + "thead", + "tr", + "th", "th", + "tbody", + "tr", "td", "td", + "tr", "td", "td", + ], + "the element structure of a GFM table must survive the fix untouched", + ); + }); + + test("cell text is preserved verbatim", () => { + const tree = withDomParserShim(() => htmlToReact(parseMarkdown(GFM_TABLE), "light")); + const rendered = JSON.stringify(tree, (key, value) => + typeof value === "function" ? "[fn]" : value, + ); + for (const cell of ["名称", "说明", "端口", "监听端口", "路径", "根路径"]) { + assert.ok(rendered.includes(cell), `cell ${cell} disappeared from the React tree`); + } + }); + + test("whitespace between BLOCK tags is still rendered (prose is not a table)", () => { + // The drop is scoped to the table family. A paragraph's inter-block + // newlines are renderable whitespace and must survive — dropping them + // everywhere would reflow prose the markdown never asked to reflow. + const tree = withDomParserShim(() => + htmlToReact(parseMarkdown("# Title\n\nbody text\n"), "light"), + ); + const texts = whitespaceTextUnderTableTags(tree as ReactLikeNode); + assert.deepEqual( + texts, + [], + "a

is not a table tag, so this guard is about the block level", + ); + // The `\n` between `` and `

` sits at body level: the walker + // keeps it, and so must the tree. + const topLevel = (tree as ReactLikeNode[]).filter( + (node): node is string => typeof node === "string", + ); + assert.ok( + topLevel.some((text) => /^\s+$/.test(text)), + "inter-block whitespace outside tables must still reach the tree", + ); + }); + + test("text with content under a table tag is still rendered (not over-trimmed)", () => { + // A `

` may legitimately hold leading/trailing spaces around its + // content (" a "). Only WHITESPACE-ONLY nodes may be dropped. + const tree = withDomParserShim(() => + htmlToReact(parseMarkdown("| a |\n| --- |\n| padded |"), "light"), + ); + const rendered = JSON.stringify(tree, (key, value) => + typeof value === "function" ? "[fn]" : value, + ); + assert.ok(rendered.includes("padded"), "cell content was lost"); + }); +}); + describe("markdown-html — findMermaidSourceBefore (blocker 1: copy-source byte-exact)", () => { /** * Hand-built DOM Level 1 element mock. The walker only touches diff --git a/packages/webui/webapp/test/modals-decision-channels.test.ts b/packages/webui/webapp/test/modals-decision-channels.test.ts index 3239d751..3aeda1aa 100644 --- a/packages/webui/webapp/test/modals-decision-channels.test.ts +++ b/packages/webui/webapp/test/modals-decision-channels.test.ts @@ -134,6 +134,151 @@ describe("ticket 70 — the plan prompt offers no decision it cannot deliver", ( }); }); +describe("UAT fix — the authorize button's label is legible in BOTH themes", () => { + // The reported symptom was a blank 「批准」 button. The string was always + // there (pinned above); the LABEL COLOUR was the defect, so a + // string assertion could never have caught it. These tests resolve the + // real design tokens out of styles/tokens.css and assert the pair the + // button renders is legible in each theme. + + const tokensCss = readFileSync(resolve(here, "../styles/tokens.css"), "utf8"); + + /** + * The declarations of every top-level `:root { }` / `.dark { }` block, + * merged. tokens.css is split into many sibling blocks (primitives, then + * one per semantic group) rather than a single one, so a reader that + * stops at the first block would only ever see the colour ramp. + */ + const blockVars = (selector: ":root" | ".dark"): Map => { + const vars = new Map(); + const open = new RegExp(`^${selector} \\{`, "gm"); + let match: RegExpExecArray | null; + while ((match = open.exec(tokensCss)) !== null) { + const body = tokensCss.slice(match.index, tokensCss.indexOf("\n}", match.index)); + for (const line of body.split("\n")) { + const declaration = /^\s*(--[\w-]+):\s*(.+?);\s*$/.exec(line); + if (declaration) vars.set(declaration[1]!, declaration[2]!); + } + } + assert.ok(vars.size > 0, `no ${selector} block found in tokens.css`); + return vars; + }; + + /** Follow `var(--x)` indirections until a literal value is reached. */ + const resolveToken = (vars: Map, name: string, depth = 0): string => { + if (depth > 8) throw new Error(`token cycle at ${name}`); + const value = vars.get(name); + if (value === undefined) throw new Error(`token ${name} is not defined`); + const inner = /^var\((--[\w-]+)\)$/.exec(value); + return inner ? resolveToken(vars, inner[1]!, depth + 1) : value.trim(); + }; + + // `.dark` only carries the semantic overrides; the primitives stay in + // `:root`, so the dark resolution layers the two. + const light = blockVars(":root"); + const dark = new Map([...light, ...blockVars(".dark")]); + + /** + * Composite a text colour over an opaque fill — the colour a pixel of the + * label actually takes. + * + * Needed because the old label is not pure white: the dark theme sets it + * to 80%-white (`#fffc`, the four-digit `#rgba` CSS form). Over an opaque + * white fill that composites to exactly the fill, which is why comparing + * the raw hex strings would have missed the defect while the button was + * plainly unreadable. + */ + const compositeOver = (text: string, fill: string): string => { + /** `#rgb` / `#rgba` / `#rrggbb` / `#rrggbbaa` → [r, g, b, a] with a in 0..1. */ + const channels = (value: string): [number, number, number, number] => { + const digits = value.slice(1).toLowerCase(); + assert.match(digits, /^([0-9a-f]{3,8})$/, `unsupported colour literal: ${value}`); + const wide = digits.length <= 4 + ? [...digits].map((digit) => digit + digit).join("") + : digits; + const byte = (index: number) => Number.parseInt(wide.slice(index, index + 2), 16); + return [byte(0), byte(2), byte(4), wide.length === 8 ? byte(6) / 255 : 1]; + }; + const [tr, tg, tb, alpha] = channels(text); + const [fr, fg, fb] = channels(fill); + const over = (t: number, f: number) => + Math.round(t * alpha + f * (1 - alpha)) + .toString(16) + .padStart(2, "0"); + return `#${over(tr, fr)}${over(tg, fg)}${over(tb, fb)}`; + }; + + test("the defect is reproducible on the token pair the button used to render", () => { + // Regression context, stated as an executable claim: the old pairing + // composited to the fill in the dark theme. If a future token + // regeneration ever themes `--text_default_inverted_static`, this stops + // holding and the note in modals.tsx must be revisited. + const background = resolveToken(dark, "--bg_interaction_primary_default"); + const oldLabel = resolveToken(dark, "--text_default_inverted_static"); + assert.equal( + compositeOver(oldLabel, background), + compositeOver(background, background), + "the dark theme is expected to invert the primary fill to white and leave " + + "the label 80%-white — that pair is what made 「批准」 unreadable", + ); + // Light theme was never affected, and saying so keeps the fix honest + // about what it changes. + const lightFill = resolveToken(light, "--bg_interaction_primary_default"); + const lightLabel = resolveToken(light, "--text_default_inverted_static"); + assert.notEqual( + compositeOver(lightLabel, lightFill), + compositeOver(lightFill, lightFill), + ); + }); + + test("the label token the button now uses contrasts with the fill in BOTH themes", () => { + for (const theme of [ + { name: ":root", vars: light }, + { name: ".dark", vars: dark }, + ]) { + const background = resolveToken(theme.vars, "--bg_interaction_primary_default"); + const label = resolveToken(theme.vars, "--text_label_primary_default"); + assert.notEqual( + compositeOver(label, background), + compositeOver(background, background), + `${theme.name}: the primary button would render its label invisibly ` + + `(${label} on ${background})`, + ); + } + }); + + test("PrimaryButton pairs the primary fill with the matching label token", () => { + // The same pairing the upstream `.mavis-button.black` rule uses + // (styles/official-utilities.css), so this button now matches the + // reference skin in both themes. + const primary = / { + const utilities = readFileSync( + resolve(here, "../styles/official-utilities.css"), + "utf8", + ); + assert.match( + utilities, + /\.mavis-button\.black \{[^}]*background-color:var\(--bg_interaction_primary_default\);color:var\(--text_label_primary_default\)/, + ); + }); +}); + describe("ticket 70 — dictionary parity for the changed keys", () => { const LOCALES = ["en", "zh"] as const; diff --git a/packages/webui/webapp/test/theme-token-pairing.test.ts b/packages/webui/webapp/test/theme-token-pairing.test.ts new file mode 100644 index 00000000..202242cd --- /dev/null +++ b/packages/webui/webapp/test/theme-token-pairing.test.ts @@ -0,0 +1,78 @@ +// webui/webapp/test/theme-token-pairing.test.ts +// +// Guardrail for the "invisible label" class of bug (UAT 2026-10-03, defect 3). +// +// `--text_default_inverted_static` does NOT flip with the theme: it stays a +// near-white in both light and dark palettes. Paired with +// `--bg_interaction_primary_default` — which dark mode DOES invert to pure +// white — the label composites to white-on-white and vanishes. The pairing is +// correct only on top of status colors (bg_status_warning / bg_status_error), +// which stay saturated in dark mode (see toolbar.tsx badge). +// +// This scan keeps the pairing out of primary-interaction surfaces. If a future +// component genuinely needs inverted text over a self-provided non-flipping +// surface, add an explicit allowlist entry here with a comment saying why. + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { createRequire } from "node:module"; +import { promises as fs } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const require = createRequire(import.meta.url); +const here = path.dirname(fileURLToPath(import.meta.url)); +const WEBAPP_DIR = path.resolve(here, ".."); +const SCAN_ROOTS = [ + path.join(WEBAPP_DIR, "components"), + path.join(WEBAPP_DIR, "app"), +]; + +async function listTsxFiles(dir: string): Promise { + const out = []; + const entries = await fs.readdir(dir, { withFileTypes: true }); + for (const entry of entries) { + const full = path.join(dir, entry.name); + if (entry.isDirectory()) out.push(...(await listTsxFiles(full))); + else if (entry.isFile() && entry.name.endsWith(".tsx")) out.push(full); + } + return out.sort(); +} + +test("no primary-interaction surface pairs with the non-flipping inverted text token", async () => { + const offenders = []; + for (const root of SCAN_ROOTS) { + for (const file of await listTsxFiles(root)) { + const source = await fs.readFile(file, "utf8"); + if (!source.includes("text-text_default_inverted_static")) continue; + // An offender is a className string that carries BOTH the primary + // interaction background and the non-flipping inverted token. The + // bg classes and the text token appear in the same string when they + // style the same element — that is the composite that vanishes. + const classNameStrings = + source.match(/"(?:[^"\\]|\\.)*text-text_default_inverted_static(?:[^"\\]|\\.)*"/g) ?? []; + for (const raw of classNameStrings) { + if (!raw.includes("bg-bg_interaction_primary_default")) continue; + offenders.push(`${path.relative(WEBAPP_DIR, file)}: ${raw.slice(0, 100)}`); + } + } + } + assert.deepEqual( + offenders, + [], + "Found primary buttons whose label vanishes in dark theme. Use " + + "text-text_label_primary_default (the token paired with " + + "bg_interaction_primary) instead:\n" + + offenders.join("\n"), + ); +}); + +test("the sanctioned exception survives: the toolbar status badge keeps its inverted token", async () => { + // The toolbar badge sits on bg_status_warning / bg_status_error, which stay + // saturated in dark mode — near-white text is correct there. If this file + // ever drops the pairing, re-evaluate rather than blindly restoring it. + const toolbarPath = path.join(WEBAPP_DIR, "components", "toolbar.tsx"); + const source = await fs.readFile(toolbarPath, "utf8"); + assert.match(source, /text-text_default_inverted_static/); + assert.match(source, /bg-bg_status_(warning|error)/); +}); diff --git a/release/public-source.json b/release/public-source.json index d571fd89..54af5e37 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3445,16 +3445,21 @@ "packages/webui/server/app.js", "packages/webui/server/bootstrap.js", "packages/webui/server/cleanup.js", + "packages/webui/server/engine/account-reads.js", "packages/webui/server/engine/capabilities.js", + "packages/webui/server/engine/capability-reads.js", "packages/webui/server/engine/errors.js", "packages/webui/server/engine/host.js", "packages/webui/server/engine/index.js", + "packages/webui/server/engine/model-reads.js", "packages/webui/server/engine/providers/local-runtime-v2.capabilities.js", "packages/webui/server/engine/providers/local-runtime-v2.js", "packages/webui/server/engine/providers/tui-runtime-adapter.js", "packages/webui/server/engine/session-export.js", "packages/webui/server/engine/session-reads.js", + "packages/webui/server/engine/session-switch.js", "packages/webui/server/engine/session-tree-reads.js", + "packages/webui/server/engine/session-writes.js", "packages/webui/server/engine/usage-reads.js", "packages/webui/server/lib/acp-client.js", "packages/webui/server/lib/agent-team-detect.js", @@ -3591,12 +3596,17 @@ "packages/webui/test/lib/context-percent.test.js", "packages/webui/test/lib/engine-catalogue.test.js", "packages/webui/test/lib/engine-provider-sync.test.js", + "packages/webui/test/lib/engine/account-reads.test.js", "packages/webui/test/lib/engine/capabilities.test.js", + "packages/webui/test/lib/engine/capability-reads.test.js", "packages/webui/test/lib/engine/capability-snapshot.test.js", "packages/webui/test/lib/engine/host-facade.test.js", + "packages/webui/test/lib/engine/model-reads.test.js", "packages/webui/test/lib/engine/session-export.test.js", "packages/webui/test/lib/engine/session-reads.test.js", + "packages/webui/test/lib/engine/session-switch.test.js", "packages/webui/test/lib/engine/session-tree-reads.test.js", + "packages/webui/test/lib/engine/session-writes.test.js", "packages/webui/test/lib/engine/usage-reads.test.js", "packages/webui/test/lib/events-concurrency.test.js", "packages/webui/test/lib/events-hash.test.js", @@ -3933,6 +3943,7 @@ "packages/webui/webapp/test/fs-tree-reveal.test.ts", "packages/webui/webapp/test/git-panel.test.ts", "packages/webui/webapp/test/greeting.test.ts", + "packages/webui/webapp/test/helpers/dom-shim.ts", "packages/webui/webapp/test/i18n-appearance.test.ts", "packages/webui/webapp/test/i18n-browser.test.ts", "packages/webui/webapp/test/i18n-file-open.test.ts", @@ -3966,6 +3977,7 @@ "packages/webui/webapp/test/sse.test.ts", "packages/webui/webapp/test/store-revision.test.ts", "packages/webui/webapp/test/stream-cursor.test.ts", + "packages/webui/webapp/test/theme-token-pairing.test.ts", "packages/webui/webapp/test/theme.test.ts", "packages/webui/webapp/test/thinking-phrase-rotation.test.ts", "packages/webui/webapp/test/tool-paths.test.ts", diff --git a/scripts/test-tmp-leak.check.mjs b/scripts/test-tmp-leak.check.mjs index 171aa12c..5ebd12ac 100644 --- a/scripts/test-tmp-leak.check.mjs +++ b/scripts/test-tmp-leak.check.mjs @@ -287,6 +287,7 @@ const KNOWN_PREFIXES = [ "webui-first-turn-guard-", "webui-lan-gate-test-events-", "webui-model-engine-cat-", + "webui-model-reads-", "webui-model-user-level-", "webui-models-merge-", "webui-origingate-events-", @@ -312,9 +313,14 @@ const KNOWN_PREFIXES = [ "webui-sec-net-settings-", "webui-sessdb-", "webui-session-delete-test-", + "webui-session-writes-b5-", "webui-sessions-search-check-", "webui-sessions-test-events-", "webui-settings-test-events-", + "webui-switch-facade-db-", + "webui-switch-facade-events-", + "webui-switch-facade-outside-", + "webui-switch-facade-roots-", "webui-switch-test-db-", "webui-switch-test-events-", "webui-transcript-test-",