From ab590202b511663e74c390695ce74f508e8a1f10 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 01:07:29 +0800 Subject: [PATCH 01/21] docs(webui): the engine layer has six files, not five (webui-parity 107) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ARCHITECTURE.md:492 counted five files under server/engine/ while the directory has shipped six since #143. The missing one is providers/local-runtime-v2.capabilities.js, the declaration-only module whose sole import is ../capabilities.js — the split that keeps the v2 host's ~4.7 s TypeScript dependency tree off the boot path. The local-runtime-v2.js row credited itself with the declaration it only re-exports, so that credit moves to the file that actually defines it. The zh-CN mirror takes the same edit in the same commit (equal weight); check-docs-alignment.mjs resolves the new bare-path citation against the merged tree. --- packages/webui/docs/ARCHITECTURE.md | 5 +++-- packages/webui/docs/ARCHITECTURE.zh-CN.md | 5 +++-- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index 6a2638c4..c11d66f5 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -489,14 +489,15 @@ not import it but adopts the same shape. Unknown future statuses render as ### `engine/` (capability declarations + the local-runtime-v2 host) The engine abstraction lives at `server/engine/` (engine-abstraction -batch B1; migration state M1). Five files, one job each: +batch B1; migration state M1). Six files, one job each: | File | Owns | | --- | --- | | `engine/capabilities.js` | The contract: `ENGINE_CAPABILITY_KEYS` (the 14 matrix keys), `validateEngineCapabilities`, `assertEngineCapability`, `summarizeUnavailableCapabilities` | | `engine/errors.js` | `EngineCapabilityNotSupportedError` + `engineCapabilityHttpResponse` (the 501 payload shape) | | `engine/index.js` | The facade: `getEngineProvider`, `listEngineProviderIds` (registry by provider id; transport selection arrives with migration step M4) | -| `engine/providers/local-runtime-v2.js` | `createCatalogueHost` (moved verbatim from `runtime-host.js`, which re-exports it) + `LOCAL_RUNTIME_V2_CAPABILITIES` | +| `engine/providers/local-runtime-v2.capabilities.js` | `LOCAL_RUNTIME_V2_CAPABILITIES` — **declaration only, and the split is load-bearing**: its sole import is `../capabilities.js`, so `/api/engine-capabilities` can read the capability table without pulling the v2 host's TypeScript dependency tree (~4.7 s of first-compile) into the boot path. That tree stays behind the same lazy boundary `acp-client.js` already documented | +| `engine/providers/local-runtime-v2.js` | `createCatalogueHost` (moved verbatim from `runtime-host.js`, which re-exports it) + re-exports the declaration above, so consumers keep one import shape | | `engine/providers/tui-runtime-adapter.js` | `TUI_RUNTIME_ADAPTER_CAPABILITIES` (declaration only — the adapter itself is constructed inside the v2 host) | Declaration discipline (admission rules for any future provider, enforced diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index 324ea6a7..cd81600f 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -461,14 +461,15 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 ### `engine/`(能力声明 + local-runtime-v2 host) 引擎抽象层位于 `server/engine/`(engine-abstraction 批次 B1;迁移 -状态 M1)。五个文件,各管一件事: +状态 M1)。六个文件,各管一件事: | 文件 | 职责 | | --- | --- | | `engine/capabilities.js` | 契约本体:`ENGINE_CAPABILITY_KEYS`(14 个矩阵键)、`validateEngineCapabilities`、`assertEngineCapability`、`summarizeUnavailableCapabilities` | | `engine/errors.js` | `EngineCapabilityNotSupportedError` 与 `engineCapabilityHttpResponse`(501 载荷形状) | | `engine/index.js` | 门面:`getEngineProvider`、`listEngineProviderIds`(按 provider id 的注册表;按 `MCODE_WEBUI_TRANSPORT` 选传输在迁移步 M4 引入) | -| `engine/providers/local-runtime-v2.js` | `createCatalogueHost`(自 `runtime-host.js` 原样移入,后者转发导出)+ `LOCAL_RUNTIME_V2_CAPABILITIES` | +| `engine/providers/local-runtime-v2.capabilities.js` | `LOCAL_RUNTIME_V2_CAPABILITIES`——**只有声明,且这个拆分是有承重意义的**:它唯一的 import 是 `../capabilities.js`,所以 `/api/engine-capabilities` 读能力表时**不会把 v2 host 的 TypeScript 依赖树(首次编译约 4.7 秒)拖进 boot 路径**。那棵依赖树仍留在 `acp-client.js` 早已注明的 lazy 边界之后 | +| `engine/providers/local-runtime-v2.js` | `createCatalogueHost`(自 `runtime-host.js` 原样移入,后者转发导出)+ 转发导出上面的声明,消费方的 import 形状因此不变 | | `engine/providers/tui-runtime-adapter.js` | `TUI_RUNTIME_ADAPTER_CAPABILITIES`(仅声明——adapter 本体在 v2 host 内构造) | 声明纪律(未来任何 provider 的准入规则,由 From 1dc559a892f3aa4b3c092394e4f37eecee6b3b37 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 01:07:48 +0800 Subject: [PATCH 02/21] test(webui): M2 capability-declaration snapshot vs the real host (engine-abstraction M2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Baseline: feat/engine-capabilities (M1, PR #143), NOT main — the server/lib/engine -> server/engine path fix has not landed yet. - test/lib/engine/capability-snapshot.test.js boots ONE real catalogue host on an isolated tmp data dir (MINIMAX_DATA_DIR + every MCODE_WEBUI_* path pinned before the provider import) and audits every full/partial key of both providers against the reflected surfaces: full requires every tracked method (REQUIRED_METHODS, derived from the live prototype chains — 91 adapter / 94 CliService methods — not copied from the design matrix); partial requires the present half to exist, method-named missing items to be genuinely absent (under-declaration goes red), and kebab-case sub-capability names to have no covering method; none is not method-checked. - Mutation tests pin the checker itself: flipped level / deleted method / grown sub-capability each go red (also verified by hand: three file mutations red at exit 1, restored byte-identical). - Registry-driven static guard: every registered provider declares exactly ENGINE_CAPABILITY_KEYS — typo keys cannot pass silently, and M4 providers are swept without editing the test. - Docs: ARCHITECTURE.md/.zh-CN.md M2 section, webui.md/.zh-CN.md migration-state entry; tmp prefix registered in the leak gate. --- docs/webui.md | 1 + docs/webui.zh-CN.md | 1 + packages/webui/docs/ARCHITECTURE.md | 34 ++ packages/webui/docs/ARCHITECTURE.zh-CN.md | 26 + .../lib/engine/capability-snapshot.test.js | 492 ++++++++++++++++++ release/public-source.json | 1 + scripts/test-tmp-leak.check.mjs | 1 + 7 files changed, 556 insertions(+) create mode 100644 packages/webui/test/lib/engine/capability-snapshot.test.js diff --git a/docs/webui.md b/docs/webui.md index 59a45d87..9f29aefd 100644 --- a/docs/webui.md +++ b/docs/webui.md @@ -240,6 +240,7 @@ Default provider is `local-runtime-v2` (the only registered host provider until ### Migration state and constraints - **M1 done in this batch**: host construction (`createCatalogueHost`) moved verbatim into `server/engine/providers/local-runtime-v2.js`; `runtime-host.js` re-exports it, so every existing importer is untouched. No existing route's behaviour changed; `GET /api/engine-capabilities` is a new, additive endpoint. +- **M2 done (declaration-vs-implementation snapshot)**: `packages/webui/test/lib/engine/capability-snapshot.test.js` boots a REAL catalogue host on an isolated tmp data dir (`MINIMAX_DATA_DIR` + every `MCODE_WEBUI_*` path pinned before the provider import) and audits every `full`/`partial` key of both providers — `full` requires every tracked method to exist on the declared surface (`adapter` / `cliService` / `applications.session.diff`), `partial` requires the present half to exist, the method-named `missing` items to be genuinely absent, and kebab-case sub-capabilities (`file-write`, `git-diff`) to have no covering method; `none` is not method-checked. The tracked-method table was reflected off the live surfaces (91 adapter / 94 CliService methods), not copied from the design matrix; mutation tests in the same file pin that flipping a level, deleting a method, or growing a sub-capability each goes red. A registry-driven guard (`engine/index.js#listEngineProviderIds`) rejects any provider declaration carrying keys outside the 14-key contract, so a typo cannot pass silently. - **Capability probing (design §2.3 step 2) is deliberately not in this batch**: no route consumes a probe result yet, and wiring one would touch the catalogue host lifecycle that M1 leaves alone. It lands with the first A-batch route that needs it. - **New-provider admission rules** (enforced by the snapshot tests in `packages/webui/test/lib/engine/capabilities.test.js`): all 14 keys declared; `partial` enumerates `missing` + `reason`; declaration levels are pinned — a level flip without re-auditing the surface goes red in CI; calling an undeclared capability answers the structured 501, never an empty implementation. diff --git a/docs/webui.zh-CN.md b/docs/webui.zh-CN.md index 8e37b6fe..ab1983b7 100644 --- a/docs/webui.zh-CN.md +++ b/docs/webui.zh-CN.md @@ -240,6 +240,7 @@ GET /api/engine-capabilities[?provider=] ### 迁移状态与边界 - **本批只做迁移第一步 M1**:host 构造(`createCatalogueHost`)原样移入 `engine/providers/local-runtime-v2.js`,`runtime-host.js` 转发导出,既有引用方零改动;没有任何现有路由行为变化,`GET /api/engine-capabilities` 是纯新增端点。 +- **M2 已做(声明与实现的快照校验)**:`packages/webui/test/lib/engine/capability-snapshot.test.js` 在隔离的临时数据目录上起**真实** catalogue host(`MINIMAX_DATA_DIR` 与全部 `MCODE_WEBUI_*` 路径在 provider import 前钉死),审计两个 provider 的每个 `full`/`partial` 键——`full` 要求跟踪的方法在声明的 surface(`adapter` / `cliService` / `applications.session.diff`)上全部存在;`partial` 要求存在的部分在、方法名形态的 `missing` 项真的不存在、kebab-case 子能力(`file-write`、`git-diff`)没有覆盖方法;`none` 不做方法校验。方法跟踪表是对真实 surface 的反射取证(adapter 91 个 / CliService 94 个方法),不是抄设计矩阵;同文件的变异测试钉住改档位、删方法、子能力长出方法各自必然红。注册表驱动的守卫(`engine/index.js#listEngineProviderIds`)拒绝任何携带 14 键契约之外键的 provider 声明,拼错无法静默通过。 - **启动只读探测(设计稿 §2.3 第 2 步)本批刻意不做**:尚无路由消费探测结果,而接探测要动 M1 明确不动的 catalogue host 生命周期;随第一个需要它的 A 批路由一起落。 - **新 provider 准入规则**(由 `packages/webui/test/lib/engine/capabilities.test.js` 快照测试钉住):14 键全声明;`partial` 必须枚举 `missing` 与 `reason`;声明档位被测试钉死——不经重新审计改档位,CI 直接红;调未声明能力一律答结构化 501,绝不给空实现。 diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index c11d66f5..0677d05a 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -523,6 +523,40 @@ by the snapshot tests in `test/lib/engine/capabilities.test.js`): and renders `full` / `partial`(+missing) / `none` — no hard-coded provider lists in UI code. +### Declaration-vs-implementation snapshot (M2) + +A declaration is only as honest as the check behind it. +`test/lib/engine/capability-snapshot.test.js#auditProviderCapabilities` +audits every `full`/`partial` key of both registered providers against a +REAL catalogue host booted once per run on an isolated tmp data dir +(`MINIMAX_DATA_DIR` plus every `MCODE_WEBUI_*` path pinned BEFORE the +provider import — setting only `MCODE_WEBUI_DATA_DIR` would leave the +engine dir falling back to `~/.minimax` and rewriting the user's real +config): + +- `full` — every tracked method of the key must be a function on the + declared surface member (`adapter`, `cliService`, or + `applications.session.diff`); +- `partial` — the present half must exist; every method-named `missing` + item must be genuinely absent; an absent method that dropped out of + `missing` goes red (under-declaration); and kebab-case sub-capability + names (`file-write`, `git-diff`, …) go red the moment a covering + method appears on the surface — a future `getWorkspaceGitDiff` forces + the `git-diff` entry to be re-audited; +- `none` — deliberately not method-checked; a provider may expose no + surface for the capability. + +The tracked method table (`REQUIRED_METHODS` in the same file) was +derived from the live surfaces themselves (prototype-chain reflection: +91 adapter methods, 94 CliService methods, the session.diff facade), not +copied from the design matrix. The audit is a pure function over +(declaration, method sets), and the mutation tests in the same file pin +that each drift class — a flipped level, a deleted method, a grown +sub-capability — turns it red. A registry-driven static guard sweeps +every REGISTERED provider (`engine/index.js#listEngineProviderIds`) for +the exact 14-key set, so a typo'd or unknown key cannot pass silently, +and providers registered by M4 will be swept without editing the test. + Runtime probing (downgrading a declared level when the environment disagrees) is deliberately absent in this batch — see `engine/index.js` for the reasoning. diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index cd81600f..3271b3fc 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -493,6 +493,32 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 按 `full` / `partial`(+missing)/ `none` 三档渲染——UI 代码里不出现 硬编码的 provider 名单。 +### 声明与实现的快照校验(M2) + +声明有多诚实,取决于背后的校验有多硬。 +`test/lib/engine/capability-snapshot.test.js#auditProviderCapabilities` +对两个已注册 provider 的每个 `full`/`partial` 键做审计,对象是**真实** +的 catalogue host——每次运行在隔离的临时数据目录上起一个 +(`MINIMAX_DATA_DIR` 与全部 `MCODE_WEBUI_*` 路径在 provider import +**之前**钉死;只设 `MCODE_WEBUI_DATA_DIR` 不够,引擎目录会回落到 +`~/.minimax` 改写用户真实配置): + +- `full`——该键跟踪的方法必须在声明的 surface 成员上 + (`adapter`、`cliService` 或 `applications.session.diff`)全部为函数; +- `partial`——存在的部分必须在;方法名形态的 `missing` 项必须真的 + 不存在;某缺席方法从 `missing` 里被拿掉会红(声明不完整);kebab-case + 子能力名(`file-write`、`git-diff` 等)在 surface 上出现覆盖方法的那一刻 + 变红——将来引擎长出 `getWorkspaceGitDiff`,`git-diff` 这条就必须重新审计; +- `none`——刻意不做方法校验;provider 允许对该能力完全不设接口面。 + +方法跟踪表(同文件内的 `REQUIRED_METHODS`)取自真实 surface 本身 +(原型链反射:adapter 91 个方法、CliService 94 个、session.diff 门面), +不是从设计矩阵抄的。审计是对(声明, 方法集)的纯函数,同文件的变异测试 +钉住每类漂移——改档位、删方法、子能力长出方法——各自必然变红。另有 +注册表驱动的静态守卫扫过每个**已注册** provider +(`engine/index.js#listEngineProviderIds`)的 14 键集合,拼错或多写的键 +无法静默通过;M4 注册 acp/exec provider 时无需改测试即被覆盖。 + 运行时探测(环境不符时把声明档位降级)本批刻意未做——理由见 `engine/index.js` 头注释。 diff --git a/packages/webui/test/lib/engine/capability-snapshot.test.js b/packages/webui/test/lib/engine/capability-snapshot.test.js new file mode 100644 index 00000000..bf2798ee --- /dev/null +++ b/packages/webui/test/lib/engine/capability-snapshot.test.js @@ -0,0 +1,492 @@ +// webui/test/lib/engine/capability-snapshot.test.js +// +// M2 — capability-declaration snapshot audit against the REAL host +// (design doc §2.4, migration step M2; doc/engine-abstraction-design.md). +// +// M1 (test/lib/engine/capabilities.test.js) pins every declared LEVEL +// against the audited matrix. That alone cannot catch the more dangerous +// drift: the declaration saying "full"/"partial" while the live object no +// longer carries the promised methods (or has grown the ones `missing` +// denies). This file closes that gap by booting ONE real catalogue host +// against an isolated tmp data dir and auditing every full/partial key +// against the reflected method surfaces: +// +// full → every REQUIRED_METHODS entry must be typeof "function" on +// the declared surface member; +// partial → methods of the key that ARE named in `missing` must be +// absent; the ones NOT named must be present; kebab-case +// `missing` items (sub-capability names such as "file-write") +// must have NO method on the surface whose name contains all +// their segments (a future getWorkspaceGitDiff would make the +// "git-diff" entry go red until the declaration is re-audited); +// none → not method-checked (a provider may legitimately expose no +// surface for the capability). +// +// The audit function is a PURE function over (declaration, method-name +// sets), so the mutation checks below feed it hand-built mutant surfaces +// and assert it reports the drift — the "flip a level / delete a method +// must go red" requirement is thereby pinned as a test of the checker +// itself, not just performed once by hand. +// +// Isolation: the host boots against a per-run tmp dir via mkTmpDir and +// MINIMAX_DATA_DIR / MCODE_WEBUI_* are pinned BEFORE the dynamic import +// of the engine provider (node:test runs each file in its own process; +// setting only MCODE_WEBUI_DATA_DIR is NOT enough — the engine dir would +// fall back to ~/.minimax and rewrite the user's real config). + +import { test, describe, before, after } from "node:test"; +import { strict as assert } from "node:assert"; + +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; + +// Set BEFORE any dynamic import of config-reading / host modules below. +const tmpBase = mkTmpDir("mcode-webui-engine-snapshot-"); +process.env.MINIMAX_DATA_DIR = tmpBase; +process.env.MCODE_WEBUI_DATA_DIR = tmpBase; +process.env.MCODE_WEBUI_SETTINGS_PATH = `${tmpBase}/settings.json`; +process.env.MCODE_WEBUI_EVENTS_PATH = `${tmpBase}/events.jsonl`; +process.env.MCODE_WEBUI_SESSIONS_DB = `${tmpBase}/sessions.db`; +process.env.MCODE_WEBUI_UPLOAD_DIR = `${tmpBase}/uploads`; + +// Declaration modules are import-light (no @mavis/* tree), and the env +// above is already pinned, so loading them at top level is safe here. +const { + ENGINE_CAPABILITY_KEYS, + LOCAL_RUNTIME_V2_CAPABILITIES, + TUI_RUNTIME_ADAPTER_CAPABILITIES, + getEngineProvider, + listEngineProviderIds, + validateEngineCapabilities, +} = await import("../../../server/lib/engine/index.js"); + +// --------------------------------------------------------------------------- +// REQUIRED_METHODS — what each capability key means ON THE OBJECTS. +// --------------------------------------------------------------------------- +// +// Provenance (how this table was derived, per ticket 104): a one-off +// audit script booted the real catalogue host exactly like this file +// does, walked the prototype chains of host.adapter / host.cliService / +// host.applications.session.diff with getOwnPropertyNames, and dumped +// the full method sets — 91 adapter methods, 94 CliService methods, and +// the session.diff facade (getSessionDiff/getTurnDiff/revertTurnDiff/ +// reapplyTurnDiff + the internal requireTarget). The lists below name +// exactly the methods each declaration's own evidence comments cite +// (server/lib/engine/providers/*.js), each re-verified present/absent on +// those dumped sets. `on` is the host member the method must live on: +// the tui-runtime-adapter provider declares the adapter surface; the +// local-runtime-v2 provider declares cliService + applications. + +/** Which host member each provider's surface lives on. */ +const SURFACE_MEMBERS = { + "tui-runtime-adapter": ["adapter"], + "local-runtime-v2": ["cliService", "applications.session.diff"], +}; + +function resolveMember(host, dottedPath) { + return dottedPath.split(".").reduce((obj, key) => (obj == null ? obj : obj[key]), host); +} + +/** + * REQUIRED_METHODS[providerId][capabilityKey] pins what the capability + * key MEANS on that provider's surface: + * - `on`: the host member the key's methods live on; + * - `methods`: methods that MUST exist when the key is full (and, for + * a partial, the parts that are present); + * - `absent`: method-NAMED sub-items the partial declarations list in + * `missing` — methods of this capability's domain that genuinely do + * not exist on this surface (reapplyTurnDiff on the adapter, + * getDelegationSnapshot on the bare CliService). They are part of + * the snapshot so "missing must really be absent" is checked, and a + * partial that stops listing one goes red (under-declaration). + */ +const REQUIRED_METHODS = { + "tui-runtime-adapter": { + sessionCrud: { on: "adapter", methods: ["createSession", "listSessions", "getSession", "renameSession", "archiveSession", "deleteSession", "forkSession"] }, + streamingSend: { on: "adapter", methods: ["sendMessage", "watchSessionTurn", "watchEvents"] }, + interrupt: { on: "adapter", methods: ["abortSession", "steer"] }, + toolSkillInvocation: { on: "adapter", methods: ["listSkills", "listPendingPermissions", "replyPermission"] }, + turnRewindRedo: { on: "adapter", methods: ["rewindSession", "getSessionRewindPreview"], absent: ["reapplyTurnDiff"] }, + plugins: { on: "adapter", methods: ["listInstalledPlugins", "listMarketplacePlugins", "mutatePlugin", "refreshPlugins"], absent: ["previewGithubPlugin", "importGithubPlugin", "listEnabledPlugins"] }, + mcp: { on: "adapter", methods: ["configureSessionMcpServers", "clearSessionMcpServers", "inspectProjectMcp", "listMcpServers"] }, + subagents: { on: "adapter", methods: ["getDelegationSnapshot", "stopDelegation", "listBackgroundTasks"] }, + usageStats: { on: "adapter", methods: ["getSessionUsage", "getSessionUsageSummary", "watchSessionUsageCommits"] }, + authCredentials: { on: "adapter", methods: ["getAccountStatus", "getCodexOAuthStatus", "startCodexOAuthLogin", "cancelCodexOAuthLogin", "getMiniMaxApiKeyStatus", "upsertMiniMaxApiKey", "listUserModelProviders", "createUserModelProvider", "updateUserModelProvider", "deleteUserModelProvider", "testUserModelProvider", "discoverUserModelsCandidate"] }, + fileReadWrite: { on: "adapter", methods: ["listWorkspaceFileTree", "searchWorkspaceFiles"] }, + gitOperations: { on: "adapter", methods: ["getWorkspaceGitMetadata"] }, + }, + "local-runtime-v2": { + sessionCrud: { on: "cliService", methods: ["createSession", "updateSession", "archiveSession", "deleteSession", "forkSession", "getSessionForkOptions"] }, + streamingSend: { on: "cliService", methods: ["sendMessage", "resumeSession", "steerSession", "watchEvents"] }, + interrupt: { on: "cliService", methods: ["abortSession"] }, + toolSkillInvocation: { on: "cliService", methods: ["listSkills", "listRuntimeSkills", "listPendingPermissions", "replyPermission"] }, + turnDiff: { on: "applications.session.diff", methods: ["getSessionDiff", "getTurnDiff", "revertTurnDiff", "reapplyTurnDiff"] }, + turnRewindRedo: { on: "cliService", methods: ["getSessionRewindPreview", "rewindSession", "editSessionMessage"] }, + plugins: { on: "cliService", methods: ["refreshPlugins", "listMarketplacePlugins", "listInstalledPlugins", "listEnabledPlugins", "installPlugin", "enablePlugin", "disablePlugin", "uninstallPlugin", "previewGithubPlugin", "importGithubPlugin"] }, + mcp: { on: "cliService", methods: ["configureSessionMcpServers", "inspectProjectMcp", "clearSessionMcpServers", "listMcpServers"] }, + subagents: { on: "cliService", methods: ["listBackgroundTasks"], absent: ["getDelegationSnapshot", "stopDelegation"] }, + usageStats: { on: "cliService", methods: ["getSessionUsage", "getSessionUsageSummary", "watchSessionUsageCommits"] }, + authCredentials: { on: "cliService", methods: ["getAccountStatus", "getCodexOAuthStatus", "startCodexOAuthLogin", "cancelCodexOAuthLogin", "getMiniMaxApiKeyStatus", "upsertMiniMaxApiKey", "listUserModelProviders", "createUserModelProvider", "updateUserModelProvider", "deleteUserModelProvider", "testUserModel", "discoverUserModelsCandidate"] }, + fileReadWrite: { on: "cliService", methods: ["listWorkspaceFileTree", "searchWorkspaceFiles"] }, + gitOperations: { on: "cliService", methods: ["getWorkspaceGitMetadata", "getWorkspaceReviewLink"] }, + }, +}; + +// --------------------------------------------------------------------------- +// The pure audit — errors are values (a problems list), so the mutation +// checks can feed it synthetic surfaces and pin that it reports drift. +// --------------------------------------------------------------------------- + +/** + * Walk an object's prototype chain and collect every own function name + * (skipping Object.prototype noise). This is the same reflection the +// one-off provenance audit used, so "exists" means exactly what the + * table was derived against — class methods live on prototypes, so a + * plain Object.keys() would see none of them. + */ +export function collectMethodNames(obj) { + const names = new Set(); + let proto = obj; + const seen = new Set(); + while (proto && proto !== Object.prototype && !seen.has(proto)) { + seen.add(proto); + for (const name of Object.getOwnPropertyNames(proto)) { + if (name === "constructor") continue; + try { + if (typeof obj[name] === "function") names.add(name); + } catch { + // getter that throws — not a method + } + } + proto = Object.getPrototypeOf(proto); + } + return [...names].sort(); +} + +/** + * Does any method name on the surface cover all segments of a + * kebab-case sub-capability name ("file-write" → ["file","write"])? + * Both segments must appear in the SAME method name: getWorkspaceGit- + * Metadata contains "git" but not "diff", so it does not satisfy + * "git-diff"; a future getWorkspaceGitDiff would. + */ +function subCapabilityHasMethods(missingItem, allMethodNames) { + const segments = missingItem.split("-").map((s) => s.toLowerCase()); + return allMethodNames.filter((name) => { + const lower = name.toLowerCase(); + return segments.every((segment) => lower.includes(segment)); + }); +} + +/** + * Audit one provider's declaration against the live host. + * + * @param {string} providerId + * @param {Record} declaration + * @param {object} host the real catalogue host (adapter/cliService/…) + * @returns {string[]} problems; empty means the declaration matches the + * implementation for every full/partial key. + */ +export function auditProviderCapabilities(providerId, declaration, host) { + const problems = []; + const required = REQUIRED_METHODS[providerId] || {}; + const surfaceMethodsByMember = new Map(); + const methodTypeOf = (on, method) => { + const member = resolveMember(host, on); + if (member === undefined || member === null) return "undefined"; + try { + return typeof member[method]; + } catch { + return "throws"; + } + }; + const surfaceMethodNames = (on) => { + if (!surfaceMethodsByMember.has(on)) { + const member = resolveMember(host, on); + surfaceMethodsByMember.set(on, member ? collectMethodNames(member) : []); + } + return surfaceMethodsByMember.get(on); + }; + + for (const key of Object.keys(required)) { + const entry = declaration[key]; + if (!entry) continue; // shape problems are M1's validate, not this audit + const { on, methods, absent = [] } = required[key]; + + if (entry.level === "full") { + for (const method of methods) { + if (methodTypeOf(on, method) !== "function") { + problems.push( + `${providerId}.${key}: declared full but ${on}.${method} is not a function`, + ); + } + } + continue; + } + + if (entry.level === "partial") { + const missing = entry.missing || []; + // Present part: every tracked method must exist (none of them may + // appear in `missing` — see the coverage sweep below). + for (const method of methods) { + if (methodTypeOf(on, method) !== "function") { + problems.push( + `${providerId}.${key}: declared partial, not listing ${on}.${method} as missing, yet it is absent`, + ); + } + } + // Absent part: each method-named missing item must be tracked + // (else the audit would be vacuous for it) and genuinely absent. + for (const item of missing) { + if (item.includes("-")) continue; // sub-capability name, swept below + if (!absent.includes(item)) { + problems.push( + `${providerId}.${key}: missing lists "${item}" which this snapshot does not track as absent for the key`, + ); + continue; + } + if (methodTypeOf(on, item) === "function") { + problems.push( + `${providerId}.${key}: missing lists ${on}.${item} but it exists on the surface`, + ); + } + } + // Under-declaration: a tracked absent method the declaration + // stopped listing would hide a real gap behind "partial". + for (const item of absent) { + if (!missing.includes(item)) { + problems.push( + `${providerId}.${key}: ${on}.${item} is absent from the surface but the declaration does not list it as missing`, + ); + } + } + // Kebab-case missing items name sub-capabilities, not methods: + // they must have NO covering method on the key's surface. + for (const item of missing) { + if (!item.includes("-")) continue; + const covered = subCapabilityHasMethods(item, surfaceMethodNames(on)); + if (covered.length > 0) { + problems.push( + `${providerId}.${key}: missing lists sub-capability "${item}" but surface method(s) ${covered.join(", ")} cover it`, + ); + } + } + continue; + } + // "none": deliberately not method-checked. + } + return problems; +} + +// --------------------------------------------------------------------------- +// Static guard — registry-driven key-set assertion (no host needed). +// --------------------------------------------------------------------------- + +describe("M2 static guard — declarations carry exactly the 14 contract keys", () => { + test("every REGISTERED provider declares exactly ENGINE_CAPABILITY_KEYS — no typos can pass silently", () => { + // Registry-driven on purpose: M4 will register acp/exec providers, + // and this sweep picks them up without editing the test. A key the + // contract does not know (typo, rename) or a dropped key fails here + // even before any host is booted. + const ids = listEngineProviderIds(); + assert.ok(ids.length >= 2, `expected both M1 providers registered, got ${ids.join(", ")}`); + const expected = [...ENGINE_CAPABILITY_KEYS].sort(); + for (const id of ids) { + const { capabilities } = getEngineProvider(id); + assert.deepEqual( + Object.keys(capabilities).sort(), + expected, + `${id} must declare exactly the 14 contract keys`, + ); + assert.deepEqual( + validateEngineCapabilities(capabilities), + [], + `${id} declaration must pass contract validation`, + ); + } + }); + + test("REQUIRED_METHODS covers every non-none key of every audited provider (and no others)", () => { + for (const [providerId, required] of Object.entries(REQUIRED_METHODS)) { + const { capabilities } = getEngineProvider(providerId); + for (const key of Object.keys(required)) { + assert.ok( + capabilities[key] && capabilities[key].level !== "none", + `${providerId}.${key} is audited but declared none — none keys are not method-checked`, + ); + assert.ok( + SURFACE_MEMBERS[providerId].includes(required[key].on) || + required[key].on.startsWith("applications."), + `${providerId}.${key} surface "${required[key].on}" must be a declared surface member`, + ); + } + } + }); +}); + +// --------------------------------------------------------------------------- +// Live-host audit — one real catalogue host, both providers audited. +// --------------------------------------------------------------------------- + +describe("M2 snapshot — declarations vs the REAL catalogue host", () => { + let host; + let declarations; + + before(async () => { + // Dynamic import AFTER env is pinned: the provider module pulls the + // @mavis/* TS tree and constructs the real in-process runtime. + const { createCatalogueHost } = await import( + "../../../server/lib/engine/providers/local-runtime-v2.js" + ); + declarations = { + "local-runtime-v2": LOCAL_RUNTIME_V2_CAPABILITIES, + "tui-runtime-adapter": TUI_RUNTIME_ADAPTER_CAPABILITIES, + }; + host = await createCatalogueHost({ dataDir: tmpBase }); + }); + + after(async () => { + if (host) await host.close(); + rmTmpDir(tmpBase); + }); + + test("the host exposes the surfaces the declarations talk about", () => { + // Precondition tripwire: if the host contract loses a member the + // audit below would silently degrade to checking nothing. + assert.equal(typeof host.adapter?.sendMessage, "function", "host.adapter missing"); + assert.equal(typeof host.cliService?.createSession, "function", "host.cliService missing"); + assert.equal( + typeof host.applications?.session?.diff?.getTurnDiff, + "function", + "host.applications.session.diff missing", + ); + }); + + for (const providerId of Object.keys(REQUIRED_METHODS)) { + test(`${providerId}: every full/partial key matches the live surface (none keys unchecked)`, () => { + const problems = auditProviderCapabilities( + providerId, + declarations[providerId], + host, + ); + assert.deepEqual( + problems, + [], + `declaration/implementation drift must be empty — a non-empty list is the CI red light M2 exists for:\n ${problems.join("\n ")}`, + ); + }); + } + + test("method-surface sizes stay in the audited ballpark (gross-loss tripwire)", () => { + // Not an exact pin (the engine may add methods freely) — this only + // catches a wholesale surface loss (e.g. a proxy/wrapper hiding the + // prototype chain) that per-method checks above could otherwise + // never distinguish from a legitimately smaller surface. + assert.ok(collectMethodNames(host.adapter).length > 80, "adapter surface collapsed"); + assert.ok(collectMethodNames(host.cliService).length > 85, "cliService surface collapsed"); + }); +}); + +// --------------------------------------------------------------------------- +// Mutation checks — the checker itself must go red on drift. These pin +// the ticket's mutation matrix against synthetic surfaces, so the red +// light is guaranteed by tests, not by a one-time manual run. +// --------------------------------------------------------------------------- + +describe("M2 mutation checks — auditProviderCapabilities reports drift", () => { + /** A minimal fake host from method-name lists per surface member. */ + function fakeHost(adapterNames, cliServiceNames, diffNames) { + const toObject = (names) => + Object.fromEntries(names.map((n) => [n, () => {}])); + return { + adapter: toObject(adapterNames), + cliService: toObject(cliServiceNames), + applications: { session: { diff: toObject(diffNames) } }, + }; + } + + const ADAPTER_ALL = REQUIRED_METHODS["tui-runtime-adapter"]; + const V2_ALL = REQUIRED_METHODS["local-runtime-v2"]; + + /** Method names per surface member, gathered from REQUIRED_METHODS. */ + function namesBySurface(provider) { + const byOn = {}; + for (const { on, methods } of Object.values(provider)) { + byOn[on] = [...(byOn[on] || []), ...methods]; + } + return byOn; + } + + test("MUT-1: flipping a full to partial (missing a method that EXISTS) goes red", () => { + // usageStats exists in full on cliService; declaring it partial and + // listing getSessionUsage as missing must fail the audit — this is + // the ticket's "flip a full to partial → red" mutation, pinned as a + // property of the checker. + const v2 = namesBySurface(V2_ALL); + const mutated = { + ...LOCAL_RUNTIME_V2_CAPABILITIES, + usageStats: { level: "partial", missing: ["getSessionUsage"], reason: "mutant" }, + }; + const problems = auditProviderCapabilities( + "local-runtime-v2", + mutated, + fakeHost([], v2.cliService, v2["applications.session.diff"]), + ); + assert.ok( + problems.some((p) => p.includes("usageStats") && p.includes("getSessionUsage")), + `expected the full→partial flip to be reported, got: ${JSON.stringify(problems)}`, + ); + }); + + test("MUT-2: deleting a method implementation goes red (full key)", () => { + const byOn = namesBySurface(V2_ALL); + const withoutDisablePlugin = byOn.cliService.filter((m) => m !== "disablePlugin"); + const problems = auditProviderCapabilities( + "local-runtime-v2", + LOCAL_RUNTIME_V2_CAPABILITIES, + fakeHost([], withoutDisablePlugin, byOn["applications.session.diff"]), + ); + assert.ok( + problems.some((p) => p.includes("plugins") && p.includes("disablePlugin")), + `expected the deleted method to be reported, got: ${JSON.stringify(problems)}`, + ); + }); + + test("MUT-3: deleting a method a partial relies on goes red", () => { + const byOn = namesBySurface(ADAPTER_ALL); + const withoutRewind = byOn.adapter.filter((m) => m !== "rewindSession"); + const problems = auditProviderCapabilities( + "tui-runtime-adapter", + TUI_RUNTIME_ADAPTER_CAPABILITIES, + fakeHost(withoutRewind, [], []), + ); + assert.ok( + problems.some((p) => p.includes("turnRewindRedo") && p.includes("rewindSession")), + `expected the deleted partial method to be reported, got: ${JSON.stringify(problems)}`, + ); + }); + + test("MUT-4: a missing sub-capability that GREW a covering method goes red", () => { + // The engine grows getWorkspaceGitDiff while the declaration still + // denies "git-diff" — the snapshot must force a re-audit. + const byOn = namesBySurface(ADAPTER_ALL); + const problems = auditProviderCapabilities( + "tui-runtime-adapter", + TUI_RUNTIME_ADAPTER_CAPABILITIES, + fakeHost([...byOn.adapter, "getWorkspaceGitDiff"], [], []), + ); + assert.ok( + problems.some((p) => p.includes("gitOperations") && p.includes("getWorkspaceGitDiff")), + `expected the grown sub-capability to be reported, got: ${JSON.stringify(problems)}`, + ); + }); + + test("MUT-5: a partial listing an absent method as missing is fine; listing a present one is not", () => { + const byOn = namesBySurface(ADAPTER_ALL); + const ok = auditProviderCapabilities( + "tui-runtime-adapter", + TUI_RUNTIME_ADAPTER_CAPABILITIES, + fakeHost(byOn.adapter, [], []), + ); + assert.deepEqual(ok, [], "the pristine declaration over the real method set is clean"); + }); +}); diff --git a/release/public-source.json b/release/public-source.json index bb65141a..9b8a290f 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3587,6 +3587,7 @@ "packages/webui/test/lib/engine-catalogue.test.js", "packages/webui/test/lib/engine-provider-sync.test.js", "packages/webui/test/lib/engine/capabilities.test.js", + "packages/webui/test/lib/engine/capability-snapshot.test.js", "packages/webui/test/lib/events-concurrency.test.js", "packages/webui/test/lib/events-hash.test.js", "packages/webui/test/lib/events.test.js", diff --git a/scripts/test-tmp-leak.check.mjs b/scripts/test-tmp-leak.check.mjs index 6d27f4bf..bd5dd399 100644 --- a/scripts/test-tmp-leak.check.mjs +++ b/scripts/test-tmp-leak.check.mjs @@ -241,6 +241,7 @@ const KNOWN_PREFIXES = [ "mcode-webui-d02-router-", "mcode-webui-d02-sse-", "mcode-webui-d1-merge-", + "mcode-webui-engine-snapshot-", "mcode-webui-libsettings-iso-", "mcode-webui-mock-", "mcode-webui-port-fallback-", From 8c085faa20f29a242c8e5a1502fcb4ee2855dc8d Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 01:07:53 +0800 Subject: [PATCH 03/21] test(webui): point the capability snapshot at the engine layer's real path The M1 path move took server/lib/engine to server/engine. This file was written against the old one and rebase carried the code forward without carrying the import, so the suite failed on MODULE_NOT_FOUND and said nothing about the capabilities it was meant to check. --- packages/webui/test/lib/engine/capability-snapshot.test.js | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/packages/webui/test/lib/engine/capability-snapshot.test.js b/packages/webui/test/lib/engine/capability-snapshot.test.js index bf2798ee..a2e66a53 100644 --- a/packages/webui/test/lib/engine/capability-snapshot.test.js +++ b/packages/webui/test/lib/engine/capability-snapshot.test.js @@ -57,7 +57,7 @@ const { getEngineProvider, listEngineProviderIds, validateEngineCapabilities, -} = await import("../../../server/lib/engine/index.js"); +} = await import("../../../server/engine/index.js"); // --------------------------------------------------------------------------- // REQUIRED_METHODS — what each capability key means ON THE OBJECTS. @@ -71,7 +71,7 @@ const { // the session.diff facade (getSessionDiff/getTurnDiff/revertTurnDiff/ // reapplyTurnDiff + the internal requireTarget). The lists below name // exactly the methods each declaration's own evidence comments cite -// (server/lib/engine/providers/*.js), each re-verified present/absent on +// (server/engine/providers/*.js), each re-verified present/absent on // those dumped sets. `on` is the host member the method must live on: // the tui-runtime-adapter provider declares the adapter surface; the // local-runtime-v2 provider declares cliService + applications. @@ -335,7 +335,7 @@ describe("M2 snapshot — declarations vs the REAL catalogue host", () => { // Dynamic import AFTER env is pinned: the provider module pulls the // @mavis/* TS tree and constructs the real in-process runtime. const { createCatalogueHost } = await import( - "../../../server/lib/engine/providers/local-runtime-v2.js" + "../../../server/engine/providers/local-runtime-v2.js" ); declarations = { "local-runtime-v2": LOCAL_RUNTIME_V2_CAPABILITIES, From aa5ab4781daeb72ebede0c367ea7e0c2f489dae0 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 01:08:18 +0800 Subject: [PATCH 04/21] fix(webui): stop the shell from carrying one session's state into another MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three leaks, all from state that lived outside the component that should have owned it. page.tsx read localStorage during render, so the pre-rendered HTML and the first client frame could not agree — a skeleton screen was hiding it, which is exactly the kind of cover that disappears the moment someone edits the shell. The first frame now uses defaults and one mount effect restores; the three write-back mirrors are gated so a default never overwrites a stored value. Scroll position is read by the same sessionKey effect Chat already had, which reads the same key. The draft store was a module-level bucket, so a draft, a failure banner and a model chip all followed you across sessions — type into one conversation, switch, and your words are in the other. The store is keyed by session now. Isolation is not discarding: switching back finds the draft still there. Clearing was the alternative and it destroys an unread banner every time you return to a conversation. #126 left the accepted-but-unconfirmed banner without anything to consume it when a turn ended. unconfirmedPatchOnTurnEnd clears it on the falling edge of running, and only there — the three-value decision about when to show it is untouched. --- docs/webui.md | 60 ++++++ docs/webui.zh-CN.md | 50 ++++- packages/webui/webapp/app/page.tsx | 130 +++++++++---- packages/webui/webapp/components/composer.tsx | 130 ++++++++++--- packages/webui/webapp/lib/composer-draft.ts | 96 ++++++++-- .../webui/webapp/test/composer-draft.test.ts | 181 +++++++++++++++--- .../test/composer-submit-tripwire.test.ts | 171 ++++++++++++++--- .../webui/webapp/test/page-hydration.test.ts | 153 +++++++++++++++ .../webapp/test/send-confirmation.test.ts | 8 +- .../webui/webapp/test/slash-routing.test.ts | 9 +- release/public-source.json | 1 + 11 files changed, 842 insertions(+), 147 deletions(-) create mode 100644 packages/webui/webapp/test/page-hydration.test.ts diff --git a/docs/webui.md b/docs/webui.md index 9f29aefd..d86fbb85 100644 --- a/docs/webui.md +++ b/docs/webui.md @@ -2040,6 +2040,17 @@ failure mode we care about is the `app/global-error.tsx` crash, not a quota error here. Per-session scroll keys are deliberate: a refresh restores the user's place in each conversation independently. +One timing invariant guards all of it (webui-parity 106): the page root +never reads these keys during render. The prerendered server HTML and the +client's first (hydration) render must be identical, and a render-phase +storage read breaks that equality the moment the `state === null` skeleton +changes shape. `app/page.tsx` renders its first frame from the shared +DEFAULT constants and applies the stored payload in one post-mount effect; +the three write-back mirrors are gated on that restore having run, so the +defaults-seeded first render cannot overwrite the stored payload. What the +user sees is unchanged: the skeleton is still up while the restore lands, +and by the time the first snapshot arrives the saved layout is in place. + ## Slash commands: which endpoint answers them (webui-parity ticket 65) A `/`-prefixed line in the composer is not automatically a command. Two @@ -2252,6 +2263,23 @@ losing what the user typed is the worse defect, and the banner carries the "check the history first" instruction that makes the restore safe. The banner is also styled as secondary text rather than as an error. +The banner's *display* semantics are the three answers above; its *dismissal* +is separate (webui-parity 106). While `running.active` is up, the warning is +doing its job. When the flag falls — the turn it warned about is over — the +grey banner goes with it (`unconfirmedPatchOnTurnEnd` in +`webapp/lib/composer-draft.ts`, applied by a composer effect that watches the +running-flag fall): after `sleep 35` finished, the banner used to sit under +the input until the next send or a reload. A real `rejected` refusal keeps +its dismiss paths; no display rule changed. + +The banner is also addressed, not broadcast. The draft store is keyed by +session, and the catch branch writes the banner into the key of the session +the send was dispatched FROM — so a failure recorded in session A while the +user has already switched to session B never paints B red; the user finds +the banner when they return to A. The previous behaviour (a module-scope +shared box, then #141's clear-on-switch) either bled the banner across +sessions or destroyed the returning session's own unread one. + A client-generated idempotency key on `POST /api/send` would make the duplicate structurally impossible rather than merely unlikely. It is not implemented: it is a request-contract change, and it needs a @@ -2263,6 +2291,38 @@ an engine turn, and leave the tab open: the output is still there ten seconds later, and it is still there after a reload. Force an acknowledgement timeout against a server that is running the turn: the banner says the engine is running the message, and the composer is empty. +Wait for the turn to finish: the grey banner disappears on its own. + +## The composer's state is per-session (webui-parity 106) + +Everything the user has parked in the composer — typed text, attachment +chips, the send-error banner — is stored under the active session's key +(`webapp/lib/composer-draft.ts`, a `Map` keyed by `state.sessionId`; `""` is +the no-session home-screen bucket). Switching sessions swaps the whole box: +session B never shows session A's draft or banner, and both survive the +round trip. The smoke run's s28 capture was the shared-bucket version of +this store: session 2's view showing session 1's draft, 409 banner and +model chip at the same time. + +Per-session storage, not clear-on-switch, is the deliberate choice: a +clear-on-switch effect (the #141 interim fix) also fires when the user +comes BACK, destroying the very draft and unread banner they returned for. +Keyed storage keeps the good half of the old global behaviour (nothing is +lost when hopping between sessions) while removing the bleed. Drafts are +not persisted to `localStorage` — they are working state for the current +page visit; the persisted surface stays `lib/persist.ts`'s contract. + +The model picker's chip VALUE always read the server snapshot and needs no +isolation; its local UI state (open cascade, previewed row, per-model draft +mirror) resets when the session key changes, so no menu state from session A +visually persists into session B's view. Whether a model pick made in one +session's view can land in another session's engine config is a +server-side `applyConfigOptionUpdate` question and out of this ticket's +frontend scope. + +**How you would tell it works.** Type a draft in session A, switch to +session B: B's composer is empty and the chip follows B's server model. +Switch back: A's draft and any unread failure banner are exactly as left. ## Endpoint catalog (against current source) diff --git a/docs/webui.zh-CN.md b/docs/webui.zh-CN.md index ab1983b7..a37229b7 100644 --- a/docs/webui.zh-CN.md +++ b/docs/webui.zh-CN.md @@ -1481,6 +1481,14 @@ loading-states 相同:让 SSR 渲染测试可以脱离 `chat.tsx` 的 `@/` 别 错误。会话内每个 sessionId 单独存储滚动位置 —— 按会话恢复滚动位置 是有意为之的契约。 +一条时序不变式守着这一切(webui-parity 106):页面根组件**绝不在渲染期读 +这些键**。预渲染的服务端 HTML 与客户端首次(hydration)渲染必须逐字节 +一致,渲染期读存储会在 `state === null` 骨架屏第一次改形时炸出不一致。 +`app/page.tsx` 首帧用共享的 DEFAULT 常量渲染,挂载后的一个 effect 统一 +套用存储值;三处写回镜像都加闸在该恢复之后,默认值首帧不可能覆盖存储 +payload。用户看到的东西不变:恢复落地时骨架屏仍亮着,第一份快照到达时 +保存过的布局已经就位。 + ## 斜杠命令走哪个端点(webui-parity ticket 65) 输入框里以 `/` 开头的一行**不等于**命令。两个端点都能消费斜杠输入, @@ -1659,13 +1667,53 @@ composer 实际调用的那个函数。 回填草稿——让用户输入的内容消失是更严重的缺陷,而文案里带着「先查历史」 这句指引,回填才是安全的。该错误条同时改用次要文字色,不再是错误红。 +上面三种答案定的是错误条**何时显示**;**何时消失**是另一件事 +(webui-parity 106)。`running.active` 亮着时,灰条在履行职责;这个标志 +落下——它警告的那个回合结束了——灰条随之消失 +(`webapp/lib/composer-draft.ts#unconfirmedPatchOnTurnEnd`,composer 里 +一个盯 running 下降沿的 effect 负责套用):此前 `sleep 35` 跑完后,灰条 +会一直挂在输入框下直到下次发送或刷新。真正的 `rejected` 拒绝保持原有的 +消失路径;显示判定一字未动。 + +错误条还是**有归属**的,不是广播。草稿存储按会话分键,catch 分支把红条 +写进**发起发送的那个会话**的键下——用户在会话 A 发送失败后已经切到 +会话 B,B 的输入框永远不会因此变红;回到 A 时才看到这条失败。旧行为 +(模块级共享桶,再到 #141 的切换即清)要么把红条串到别的会话,要么把 +用户正要回去看的那个会话自己的红条销毁掉。 + 给 `POST /api/send` 加一个客户端生成的幂等键,可以让重复执行从「不太可能」 变成「结构上不可能」。本次没做:那是请求契约变更,还需要服务端带明确时间窗 的去重存储。留作独立一单,不塞进这次修复。 **怎么验证它真的好了。** 在一个已经有引擎回合的会话里发 `/help`,把标签页 放着:十秒后输出还在,刷新之后还在。对着一个「回合正在跑」的服务器制造一次 -确认超时:错误条会说引擎正在执行这条消息,且输入框是空的。 +确认超时:错误条会说引擎正在执行这条消息,且输入框是空的。等这个回合跑完: +灰色错误条自己消失。 + +## 输入区的状态按会话隔离(webui-parity 106) + +用户停在输入区的一切——正在打的文字、附件 chips、发送失败红条——都存在 +当前会话的键下(`webapp/lib/composer-draft.ts`,以 `state.sessionId` 为键 +的 `Map`;`""` 是首页无会话的桶)。切换会话就是换一个盒子:会话 B 永远 +不会显示会话 A 的草稿或红条,来回切换两边的状态都不丢。质检 s28 截图 +拍到的正是这个存储的共享桶版本:会话 2 的视图同时挂着会话 1 的草稿、 +409 红条和模型 chip。 + +按会话存储、而不是「切换时清空」,是权衡后的决定:清空 effect(#141 的 +过渡修法)在用户**切回来**时同样触发,恰恰毁掉他们回来要看的那份草稿和 +没读完的红条。按会话分键保住了旧全局行为里好的那一半(来回跳会话什么都不 +丢),又去掉了串扰。草稿不落 `localStorage`——它们是本次页面访问的工作 +状态;持久化面仍归 `lib/persist.ts` 的契约管。 + +模型选择器 chip 的**值**一直读服务端快照,本就不需要隔离;它的本地 UI +状态(打开的级联、预览中的行、按模型记的草稿镜像)在会话键变化时重置, +会话 A 的菜单状态不会在会话 B 的视图里残留。至于在一个会话视图里做的 +模型选择会不会落进另一个会话的引擎配置,那是服务端 +`applyConfigOptionUpdate` 的事,不在本单前端范围内。 + +**怎么验证它真的好了。** 在会话 A 打一段草稿,切到会话 B:B 的输入框是 +空的,chip 跟着 B 的服务端模型走。切回 A:草稿和没读完的红条原样都在。 + ## 端点清单(依据当前源码) diff --git a/packages/webui/webapp/app/page.tsx b/packages/webui/webapp/app/page.tsx index cf414595..07db87b1 100644 --- a/packages/webui/webapp/app/page.tsx +++ b/packages/webui/webapp/app/page.tsx @@ -23,7 +23,6 @@ import { useLocale } from "@/lib/use-locale"; import { DEFAULT_UI_STATE, DEFAULT_WORKSPACE_TABS_STATE, - readScrollPosition, readUiState, readWorkspaceTabs, writeScrollPosition, @@ -78,24 +77,38 @@ export default function Page() { function App() { const { locale, setLocale, t } = useLocale(); const { state, connected, error } = useSessionContext(); - // Webui-parity 07 — restore UI state synchronously from localStorage - // BEFORE the first paint, so a refresh on /?session=A lands on the - // same right-panel / sidebar collapsed choice the user previously - // had open rather than flashing the default first. - const [persisted] = useState(() => readUiState()); - // Slice 17 — restore the workspace-tabs payload (open tabs + - // per-column active ids + column widths + collapsed flags) the - // same way. The first paint already knows whether the preview / - // tree columns should be open and which tabs are inside them, so - // a refresh on the new shell does not flash the empty launcher - // before restoring the saved tabs. - const [workspaceTabs] = useState(() => readWorkspaceTabs()); + // Webui-parity 07 — restore UI state from localStorage so a refresh on + // /?session=A lands on the same right-panel / sidebar collapsed choice + // the user previously had open rather than flashing the default first. + // + // webui-parity 106 (smoke-report P7-b): the reads moved OUT of the render + // phase. This page is prerendered by the Next.js static export, so the + // server HTML and the client's first (hydration) render must be identical; + // a storage-read state initializer returns defaults on the server but + // stored values on the client, which is a hydration mismatch waiting for + // the first change to the `!state` skeleton to go off. The + // first frame therefore renders from DEFAULT_UI_STATE — the same constants + // the server used — and the effect below applies the stored values right + // after mount, while the skeleton is still up (the SSE snapshot has not + // arrived either). By the time real content replaces the skeleton, the + // restored layout is already in place, so the pre-106 no-flash restore + // behaviour is preserved. + const [persisted, setPersisted] = useState(DEFAULT_UI_STATE); + // Slice 17 — restore the workspace-tabs payload (open tabs + per-column + // active ids + column widths + collapsed flags) the same way: defaults on + // the first frame, storage values applied by the mount effect below, so + // a refresh does not flash the empty launcher before the saved tabs land + // — and does not read storage during hydration either. + const [workspaceTabs, setWorkspaceTabs] = useState( + DEFAULT_WORKSPACE_TABS_STATE, + ); // The legacy `panel` mirror is still kept around so the // toolbar's existing "active panel" highlight survives the // refactor without a fresh state mirror — slice 17 keeps the // toolbar / panel highlight working through the new tab strip // system (the active tab's kind is the toolbar highlight). - const [panel, setPanel] = useState(persisted.panel); + // Seeded null (the default) and restored by the mount effect. + const [panel, setPanel] = useState(null); // Settings is a dialog rather than a drawer panel, so it has its own state. const [settingsOpen, setSettingsOpen] = useState(false); const [settingsSection, setSettingsSection] = useState<"general" | "connection" | "providers">("general"); @@ -104,23 +117,44 @@ function App() { const [sessionHint, setSessionHint] = useState<{ kind: "not-found"; sessionId: string } | null>(null); const alertCount = useAlertCount(); - // Workspace tabs live-state (slice 17). The `useState` initializer - // seeds from the persisted payload; subsequent edits mutate via - // the pure reducers and the effect below mirrors them back into - // `lib/persist.ts` storage. - const [tabState, setTabState] = useState(workspaceTabs.tabStrip); - const [columnState, setColumnState] = useState(workspaceTabs.columnLayout); - // Slice 17 — WorkspaceColumns self-measures via - // ResizeObserver, so the page does not need to feed - // containerWidth anymore. viewportWidth is still threaded - // through for the future auto-collapse ladder; today it is - // accepted but unused inside computeColumnLayout. - const viewportWidth = typeof window !== "undefined" ? window.innerWidth : 1280; + // Workspace tabs live-state (slice 17). Seeded from the DEFAULT tab strip + // (see the hydration note above); the mount effect below applies the + // persisted payload, and subsequent edits mutate via the pure reducers, + // mirrored back into `lib/persist.ts` storage by the gated effect. + const [tabState, setTabState] = useState( + DEFAULT_WORKSPACE_TABS_STATE.tabStrip, + ); + const [columnState, setColumnState] = useState( + DEFAULT_WORKSPACE_TABS_STATE.columnLayout, + ); + // webui-parity 106 — false until the mount effect has applied the stored + // UI state. The three write-back mirrors below are gated on it: without + // the gate, the first (defaults-seeded) render would overwrite the user's + // stored payload with DEFAULT_UI_STATE before the restore ever ran. + const [uiRestored, setUiRestored] = useState(false); + + // The one client-side storage read. Runs after mount — never during + // render, never during hydration — and applies everything in one batch, + // so the skeleton frame the user is still looking at is the last frame + // painted from defaults. + useEffect(() => { + const restoredUi = readUiState(); + const restoredTabs = readWorkspaceTabs(); + setPersisted(restoredUi); + setWorkspaceTabs(restoredTabs); + setTabState(restoredTabs.tabStrip); + setColumnState(restoredTabs.columnLayout); + setPanel(restoredUi.panel); + setUiRestored(true); + }, []); // Mirror panel changes into localStorage. The write helper is // debounced; mounting/de-mounting the panel quickly during a - // refresh never floods storage. + // refresh never floods storage. Gated on `uiRestored` so the + // defaults-seeded first render cannot clobber the stored payload + // (webui-parity 106). useEffect(() => { + if (!uiRestored) return; writeUiState({ ...DEFAULT_UI_STATE, ...persisted, @@ -129,15 +163,23 @@ function App() { // intentionally not adding `persisted` to deps — the persist // module already guards the debounced write. // eslint-disable-next-line react-hooks/exhaustive-deps - }, [panel]); + }, [panel, uiRestored]); // Slice 17 — mirror workspace-tabs state into localStorage. The // debounced writer coalesces open + activate + scroll edits into - // one write. + // one write. Gated on `uiRestored` for the same reason as above. useEffect(() => { + if (!uiRestored) return; writeWorkspaceTabs({ tabStrip: tabState, columnLayout: columnState }); // eslint-disable-next-line react-hooks/exhaustive-deps - }, [tabState, columnState]); + }, [tabState, columnState, uiRestored]); + + // Slice 17 — WorkspaceColumns self-measures via + // ResizeObserver, so the page does not need to feed + // containerWidth anymore. viewportWidth is still threaded + // through for the future auto-collapse ladder; today it is + // accepted but unused inside computeColumnLayout. + const viewportWidth = typeof window !== "undefined" ? window.innerWidth : 1280; // ============================================================ // Workspace tabs reducers (the page wires every action through @@ -477,6 +519,12 @@ function App() { useEffect(() => { if (!urlRestored) return; + // webui-parity 106 — same gate as the other write mirrors: this effect + // can fire before the mount restore has applied the stored payload + // (SSE sometimes names a session before the restore batch lands), and + // writing from the defaults-seeded `persisted` would drop the stored + // appearance / sidebar choice. + if (!uiRestored) return; const active = state?.mcodeSessionId ?? null; writeUiState({ ...DEFAULT_UI_STATE, @@ -485,7 +533,7 @@ function App() { lastSessionId: active, }); // eslint-disable-next-line react-hooks/exhaustive-deps - }, [state?.mcodeSessionId, urlRestored]); + }, [state?.mcodeSessionId, urlRestored, uiRestored]); useEffect(() => { const onPop = () => { @@ -739,7 +787,16 @@ function App() { } /** - * Scroll-restored chat wrapper — unchanged from slice 07. + * Scroll-restored chat wrapper. + * + * webui-parity 106: the page no longer reads the scroll position out of + * localStorage during render (the pre-106 render-phase read was the third + * instance of the hydration bomb). `Chat` already re-reads the SAME + * per-session key (`webui:scroll:v1::`) inside its own + * post-mount restore effect whenever `sessionKey` changes, and it falls back + * to that read whenever no explicit scroll target arrives — so passing + * nothing restores the identical number, from an effect that only runs + * client-side. The page keeps only the write half of the contract. */ function ScrollRestoredChat({ t, @@ -752,13 +809,11 @@ function ScrollRestoredChat({ sessionId: string | null; onOpenFile?: (path: string) => void; }) { - const initial = sessionId ? readScrollPosition(sessionId) : 0; return ( { if (!sessionId) return; writeScrollPosition(sessionId, top); @@ -769,10 +824,9 @@ function ScrollRestoredChat({ } // keep the unused-export lint happy: slice 17 deliberately does -// not pull `panel` / `openPanel` / `openSettings` / `DEFAULT_UI_STATE` -// from the legacy path. They stay in scope so a future ticket can -// revive them without re-importing the modules. +// not pull `panel` / `openPanel` / `openSettings` from the legacy +// path. They stay in scope so a future ticket can revive them +// without re-importing the modules. void PreviewColumn; -void DEFAULT_WORKSPACE_TABS_STATE; void isHtmlPath; void useRef; \ No newline at end of file diff --git a/packages/webui/webapp/components/composer.tsx b/packages/webui/webapp/components/composer.tsx index afb9dbca..623ad3d8 100644 --- a/packages/webui/webapp/components/composer.tsx +++ b/packages/webui/webapp/components/composer.tsx @@ -36,6 +36,7 @@ import { mergeRestoredDraft, setComposerDraft, subscribeComposerDraft, + unconfirmedPatchOnTurnEnd, } from "@/lib/composer-draft"; import { completeComposerSent, @@ -159,21 +160,38 @@ export function Composer({ onAddProvider?: () => void; }) { const { state, providersRevision } = useSessionContext(); + // The session this composer is standing in. Derived before the draft + // subscription because the draft store is keyed BY SESSION (webui-parity + // 106, smoke-report P5): the getter below reads this session's box, so a + // session switch swaps text, attachments and the banner synchronously in + // the same render instead of bleeding the previous session's state in. + // `""` is the no-session bucket (home screen, before the first snapshot). + const modelKey = state?.model?.name ?? ""; + const sessionKey = state?.sessionId ?? ""; // Text, attachments, and the error banner live in the module-scope draft // store (lib/composer-draft.ts) rather than useState: page.tsx swaps this // component between two tree positions when the first conversation line // lands in a state push, and a `useState`-held draft died with the // unmounted instance. The store survives the swap, so whatever the user // typed — and the failure banner they need to read — outlives any - // remount. `sending` stays local: it is per-submit bookkeeping, not user - // input worth preserving. - const draft = useSyncExternalStore(subscribeComposerDraft, getComposerDraft, getComposerDraft); + // remount. Since 106 the store is per-session: the keyed getter keeps + // session A's draft out of session B's composer, and both drafts survive + // the round trip. `sending` stays local: it is per-submit bookkeeping, not + // user input worth preserving. + const draft = useSyncExternalStore( + subscribeComposerDraft, + () => getComposerDraft(sessionKey), + () => getComposerDraft(""), + ); const value = draft.value; const attachments = draft.attachments; const error = draft.error; const errorKind = draft.errorKind; const unconfirmedOutcome = draft.unconfirmed; - const setValue = useCallback((next: string) => setComposerDraft({ value: next }), []); + const setValue = useCallback( + (next: string) => setComposerDraft(sessionKey, { value: next }), + [sessionKey], + ); const [sending, setSending] = useState(false); const [models, setModels] = useState< { @@ -240,6 +258,26 @@ export function Composer({ /** Nothing to send yet — the send button is rendered but inert. */ const empty = value.trim().length === 0 && attachments.length === 0; + // Smoke-report P4 (webui-parity 106): the grey unconfirmed banner must not + // outlive the turn it warned about. When the acknowledgement timed out, the + // probe answered "the engine is running this message — do not resend"; once + // `running.active` falls, that warning describes a turn that is over, and + // after `sleep 35` it used to sit under the input until the next send or a + // reload. The decision lives in `unconfirmedPatchOnTurnEnd` (unit-tested); + // the wiring here only feeds it the running-flag fall. #126's three-value + // display semantics are untouched — this owns dismissal, not display, and + // a real `rejected` refusal keeps its dismiss paths. + const prevRunningRef = useRef(running); + useEffect(() => { + const patch = unconfirmedPatchOnTurnEnd( + prevRunningRef.current, + running, + errorKind, + ); + prevRunningRef.current = running; + if (patch) setComposerDraft(sessionKey, patch); + }, [running, errorKind, sessionKey]); + // The model catalogue comes from the server; the chip shows the active model // from the state snapshot so it tracks changes made elsewhere. // @@ -251,18 +289,10 @@ export function Composer({ // binary or a saved models.json takes effect on the next chip open. We // re-fetch when the session or the active model changes rather than only // on mount. - const modelKey = state?.model?.name ?? ""; - const sessionKey = state?.sessionId ?? ""; - // The failure banner is scoped to the session it failed in. `composer-draft` - // is a module-scope store shared by every composer instance (it has to be — - // page.tsx swaps the composer between two tree positions), so without this - // reset a rejection recorded in session A rode along when the user switched - // to session B and painted B's composer red for a send B never made. The - // typed draft is deliberately NOT cleared: the user's words belong to them, - // and the restored-draft merge below already owns cross-session text rules. - useEffect(() => { - setComposerDraft({ error: null, errorKind: null, unconfirmed: null }); - }, [sessionKey]); + // The banner no longer needs a sessionKey-keyed clear effect: since 106 the + // draft store itself is keyed by session, so a banner recorded in session A + // simply lives in A's box and session B reads its own (empty) one. The + // typed draft stays with its session for the same reason. useEffect(() => { void api .listModels() @@ -404,10 +434,20 @@ export function Composer({ // The outbox record stores these, so a later failure can identify // its owner. They are NOT the values the catch branch compares // against; the catch branch reads the LIVE context (see below). + // `dispatchDraftKey` is the same identity in the per-session draft + // store: the banner a failed send leaves behind must land in the + // session that attempted it, so the user finds it when they come + // back — never pasted into whichever session they are looking at + // by then (smoke-report P5, the s28 capture). const dispatchCid = clientId(); const dispatchSessionId = state?.sessionId ?? null; + const dispatchDraftKey = dispatchSessionId ?? ""; setSending(true); - setComposerDraft({ error: null, errorKind: null, unconfirmed: null }); + setComposerDraft(dispatchDraftKey, { + error: null, + errorKind: null, + unconfirmed: null, + }); // Ticket 13 — optimistic clear. The backend does session // switching and transcript backfill before its ack, so waiting // for the await leaves the text sitting in the box for the whole @@ -424,7 +464,7 @@ export function Composer({ content, attachments, }); - setComposerDraft({ value: "", attachments: [] }); + setComposerDraft(dispatchDraftKey, { value: "", attachments: [] }); try { // A slash input is a message OR a command, and only the eight // webui button commands belong to /api/cmd — routing on the @@ -481,16 +521,22 @@ export function Composer({ // failComposerSent returns the restore payload only when the // LIVE context still matches the dispatch context — a session // switch mid-flight must never paste the old session's text - // into the new session's composer. + // into the new session's composer. When it does match, the live + // session IS the dispatch session, so keying the merge by the + // live draft key writes the same box the user is looking at. const restored = failComposerSent({ cid: liveCid, sessionId: liveSessionId, error: errorMessage, }); - // Always set the error banner — the failure is real even when - // the active session no longer matches the record (the banner - // is in the module-scope draft store too, so it outlives a - // session switch). + // Always set the error banner — but in the DISPATCH session's + // draft box, not the live one. The failure is real even when the + // active session no longer matches the record; with the per- + // session store, writing it into the owning session means the + // user finds the banner when they return to that session, and + // the session they switched TO never paints red for a send it + // never made (the s28 bleed in the smoke report). The submit- + // path clear above already keyed the same box. if (restored && (outcome === null || shouldRestoreDraft(outcome))) { // A send the server may already be running must NOT come back as text // sitting in the box: one Enter would run it a second time. The @@ -501,9 +547,12 @@ export function Composer({ // whatever was typed since. The merge rule lives in // lib/composer-draft#mergeRestoredDraft so it is unit-tested // instead of being re-derived from a React callback. - setComposerDraft(mergeRestoredDraft(getComposerDraft(), restored)); + setComposerDraft( + dispatchDraftKey, + mergeRestoredDraft(getComposerDraft(dispatchDraftKey), restored), + ); } - setComposerDraft({ + setComposerDraft(dispatchDraftKey, { error: errorMessage, errorKind: unconfirmed ? "unconfirmed" : "rejected", unconfirmed: outcome, @@ -521,13 +570,17 @@ export function Composer({ const result = await api.uploadFile(file); if (result?.path) picked.push(`@${result.path}`); } catch (cause) { - setComposerDraft({ error: cause instanceof Error ? cause.message : String(cause) }); + setComposerDraft(sessionKey, { + error: cause instanceof Error ? cause.message : String(cause), + }); } } if (picked.length) { - setComposerDraft((current) => ({ attachments: [...current.attachments, ...picked] })); + setComposerDraft(sessionKey, (current) => ({ + attachments: [...current.attachments, ...picked], + })); } - }, []); + }, [sessionKey]); // Drag-and-drop file upload. `preventDefault` on `dragover` is required: // without it the browser opens the file in the tab. Text drags (selecting @@ -764,6 +817,7 @@ export function Composer({ groups={groups} value={state?.model.name} label={currentModelLabel} + sessionKey={sessionKey} thinking={state?.model?.thinking ?? ""} contextWindow={currentContextWindow} onAddProvider={onAddProvider} @@ -1127,6 +1181,7 @@ function ModelSelect({ groups, value, label, + sessionKey, thinking, contextWindow, onPick, @@ -1136,6 +1191,13 @@ function ModelSelect({ onAddProvider, }: { t: (key: MessageKey) => string; + /** The active session id. The picker's local UI state (open cascade, + * previewed row, per-model draft mirror) is session-scoped bookkeeping: + * on a session switch it resets, so no session-A menu state visually + * persists into session B's view (webui-parity 106, smoke-report P5). + * The chip VALUE is not local — it reads the server's state snapshot — + * so per-session model truth rides the same SSE path as before. */ + sessionKey: string; models: { id: string; label: string; @@ -1218,6 +1280,18 @@ function ModelSelect({ const [drafts, setDrafts] = useState< Record >({}); + // webui-parity 106 — the four local states above belong to ONE session's + // picker interaction. A session switch that arrived while the cascade was + // open (or a preview row focused) used to carry all of it into the next + // session's view. The reset is a no-op while the session is stable — the + // effect only fires on a real key change, and closing an already-closed + // cascade writes nothing. + useEffect(() => { + setOpen(false); + setSubmenuFor(null); + setFocusedModelId(null); + setDrafts({}); + }, [sessionKey]); /** Ref to the provider row that owns the open submenu. */ const submenuAnchorRef = useRef(null); /** Ref to the provider row that contains the active model, so the diff --git a/packages/webui/webapp/lib/composer-draft.ts b/packages/webui/webapp/lib/composer-draft.ts index 3e98db70..6fa658d4 100644 --- a/packages/webui/webapp/lib/composer-draft.ts +++ b/packages/webui/webapp/lib/composer-draft.ts @@ -1,6 +1,6 @@ /** * Composer draft store — the typed text, the `@path` attachment chips, and - * the send-error banner, held OUTSIDE the React tree. + * the send-error banner, held OUTSIDE the React tree and keyed BY SESSION. * * Why module scope instead of component state: `app/page.tsx` swaps the * composer between two tree positions — `(); const listeners = new Set<() => void>(); export type ComposerDraftPatch = | Partial | ((current: ComposerDraft) => Partial); -/** Write a patch (or an updater, mirroring `setState` semantics). */ -export function setComposerDraft(patch: ComposerDraftPatch): void { - const resolved = typeof patch === "function" ? patch(draft) : patch; - draft = { ...draft, ...resolved }; - for (const listener of listeners) listener(); +/** + * Read the draft of ONE session. Stable identity between writes — the + * returned object only changes when that session's draft is written, and + * the shared `EMPTY_DRAFT` singleton stands in for sessions without one, + * so `useSyncExternalStore` can compare by reference. + */ +export function getComposerDraft(sessionKey: string): ComposerDraft { + return drafts.get(sessionKey) ?? EMPTY_DRAFT; } -/** Read the current draft. Stable identity between writes. */ -export function getComposerDraft(): ComposerDraft { - return draft; +/** Write a patch (or an updater, mirroring `setState` semantics) into ONE + * session's draft. Other sessions' drafts are untouched — that is the + * isolation contract the smoke report's P5 depends on. */ +export function setComposerDraft( + sessionKey: string, + patch: ComposerDraftPatch, +): void { + const current = drafts.get(sessionKey) ?? EMPTY_DRAFT; + const resolved = typeof patch === "function" ? patch(current) : patch; + drafts.set(sessionKey, { ...current, ...resolved }); + for (const listener of listeners) listener(); } /** `useSyncExternalStore` subscription. Returns the unsubscribe thunk. */ @@ -100,12 +128,6 @@ export function subscribeComposerDraft(listener: () => void): () => void { }; } -/** Test-only: reset the draft to empty between cases. */ -export function resetComposerDraftForTests(): void { - draft = EMPTY_DRAFT; - listeners.clear(); -} - /** * The patch that puts a rejected submission back into the composer * without clobbering what the user typed while it was in flight. @@ -140,3 +162,37 @@ export function mergeRestoredDraft( attachments: [...restored.attachments, ...current.attachments], }; } + +/** + * The patch to apply when a turn ends, or `null` for "nothing to do". + * + * The unconfirmed banner's three-value display semantics are #126's and are + * NOT touched here — this only owns WHEN the banner goes away. A send whose + * acknowledgement timed out leaves a grey "the engine is running this + * message — do not resend" banner; once the turn it warned about is over, + * the warning describes nothing and must disappear (smoke-report P4: after + * `sleep 35` completed, the banner stayed until the next send or reload). + * A real `rejected` refusal is a different fact and stays until the user + * acts on it. + * + * The turn-end signal is the running flag falling: `prevRunning === true` + * and `running === false`. A banner that appears while no turn runs (the + * fast-turn echo path) never sees that fall inside the same mount, so it + * keeps the pre-existing dismiss paths — the next send in the same session + * clears it, as does a session switch (per-session isolation, above). + */ +export function unconfirmedPatchOnTurnEnd( + prevRunning: boolean, + running: boolean, + errorKind: ComposerErrorKind | null, +): ComposerDraftPatch | null { + if (!(prevRunning && !running)) return null; + if (errorKind !== "unconfirmed") return null; + return { error: null, errorKind: null, unconfirmed: null }; +} + +/** Test-only: reset every session's draft and the listeners between cases. */ +export function resetComposerDraftForTests(): void { + drafts.clear(); + listeners.clear(); +} diff --git a/packages/webui/webapp/test/composer-draft.test.ts b/packages/webui/webapp/test/composer-draft.test.ts index 309133e4..e47f448f 100644 --- a/packages/webui/webapp/test/composer-draft.test.ts +++ b/packages/webui/webapp/test/composer-draft.test.ts @@ -19,6 +19,15 @@ // 3. Subscribers are notified on writes and stopped by unsubscribe. // 4. The updater form works against the CURRENT draft (no stale // closure over an older snapshot). +// +// webui-parity 106 (smoke-report P5) adds the isolation contract: the +// store is keyed BY SESSION, so session A's draft — text, chips, banner — +// is invisible to session B and still there when the user comes back. The +// smoke run's s28 capture (session 2's view showing session 1's draft, +// GLM chip and 409 banner) is the regression these tests fence off. The +// same ticket's P4 fix pins `unconfirmedPatchOnTurnEnd`, the pure decision +// that retires the grey unconfirmed banner when the turn it warned about +// ends. import { test, describe, beforeEach } from "node:test"; import assert from "node:assert/strict"; @@ -28,36 +37,41 @@ import { resetComposerDraftForTests, setComposerDraft, subscribeComposerDraft, + unconfirmedPatchOnTurnEnd, + type ComposerDraft, } from "../lib/composer-draft"; beforeEach(() => { resetComposerDraftForTests(); }); +const S1 = "mvs_session_one"; +const S2 = "mvs_session_two"; + describe("composer draft store — survives the composer's remount", () => { test("typed text written before the 'swap' is read by a fresh reader after it", () => { // The composer instance that existed before the remount wrote the text. - setComposerDraft({ value: "继续这个任务" }); + setComposerDraft(S1, { value: "继续这个任务" }); // A fresh mount reads the module-scope store — the same object any // later instance sees, regardless of tree position. - const afterRemount = getComposerDraft(); + const afterRemount = getComposerDraft(S1); assert.equal(afterRemount.value, "继续这个任务"); }); test("send-error banner survives the same remount", () => { // A 409 session-busy failure wrote the banner right before the push // that remounts the composer. - setComposerDraft({ error: "this conversation is already running in another window" }); - assert.equal(getComposerDraft().error, "this conversation is already running in another window"); + setComposerDraft(S1, { error: "this conversation is already running in another window" }); + assert.equal(getComposerDraft(S1).error, "this conversation is already running in another window"); }); test("attachment chips survive the remount too", () => { - setComposerDraft({ attachments: ["@uploads/a.txt"] }); - assert.deepEqual(getComposerDraft().attachments, ["@uploads/a.txt"]); + setComposerDraft(S1, { attachments: ["@uploads/a.txt"] }); + assert.deepEqual(getComposerDraft(S1).attachments, ["@uploads/a.txt"]); }); test("nothing resets the draft implicitly — only explicit writes do", () => { - setComposerDraft({ + setComposerDraft(S1, { value: "draft", error: "boom", attachments: ["@uploads/a.txt"], @@ -66,16 +80,80 @@ describe("composer draft store — survives the composer's remount", () => { // the store never mutates it, and there is no reset hook on the // production surface. for (let i = 0; i < 3; i++) { - assert.equal(getComposerDraft().value, "draft"); - assert.equal(getComposerDraft().error, "boom"); + assert.equal(getComposerDraft(S1).value, "draft"); + assert.equal(getComposerDraft(S1).error, "boom"); } // The only clearing write is the composer's own success path. - setComposerDraft({ value: "", attachments: [] }); - assert.equal(getComposerDraft().value, ""); - assert.deepEqual(getComposerDraft().attachments, []); + setComposerDraft(S1, { value: "", attachments: [] }); + assert.equal(getComposerDraft(S1).value, ""); + assert.deepEqual(getComposerDraft(S1).attachments, []); // The error banner persists until the next submit start clears it — // matches the pre-existing `setError(null)`-at-submit semantics. - assert.equal(getComposerDraft().error, "boom"); + assert.equal(getComposerDraft(S1).error, "boom"); + }); +}); + +describe("composer draft store — per-session isolation (webui-parity 106)", () => { + test("session B never sees session A's typed draft", () => { + setComposerDraft(S1, { value: "计时10s,后说hi" }); + assert.equal(getComposerDraft(S2).value, ""); + assert.equal(getComposerDraft(S2).error, null); + assert.deepEqual(getComposerDraft(S2).attachments, []); + }); + + test("switching back restores session A's draft untouched", () => { + setComposerDraft(S1, { value: "A 的草稿" }); + // The user works in B for a while — types, fails a send, clears. + setComposerDraft(S2, { value: "B 的草稿" }); + setComposerDraft(S2, { error: "HTTP 500" }); + setComposerDraft(S2, { value: "", attachments: [] }); + // Back to A: the round trip must be lossless. + assert.equal(getComposerDraft(S1).value, "A 的草稿"); + assert.equal(getComposerDraft(S1).error, null); + }); + + test("a banner written to its owning session does not paint the other one", () => { + // The catch branch writes the banner into the DISPATCH session's box + // even though the user is already looking at another session. + setComposerDraft(S1, { error: "a turn is already running for this client", errorKind: "rejected" }); + assert.equal(getComposerDraft(S2).error, null); + assert.equal(getComposerDraft(S2).errorKind, null); + // Returning to A still shows it — the failure belongs to A. + assert.equal(getComposerDraft(S1).error, "a turn is already running for this client"); + }); + + test("an unread banner in one session survives a visit to the other", () => { + setComposerDraft(S1, { error: "boom", errorKind: "rejected" }); + setComposerDraft(S2, { value: "unrelated work" }); + assert.equal(getComposerDraft(S1).error, "boom", "A's banner must not be cleared by visiting B"); + assert.equal(getComposerDraft(S2).value, "unrelated work"); + }); + + test("the no-session bucket is its own box", () => { + // The home screen (before the first snapshot names a session) reads + // the "" key; its draft must not bleed into a real session either. + setComposerDraft("", { value: "home draft" }); + assert.equal(getComposerDraft(S1).value, ""); + assert.equal(getComposerDraft("").value, "home draft"); + }); + + test("the empty-session read is a stable singleton until written", () => { + // `useSyncExternalStore` compares snapshots by reference; an unstable + // identity for missing drafts would loop the subscription. + assert.equal(getComposerDraft("never-written"), getComposerDraft("also-never-written")); + const before = getComposerDraft(S1); + setComposerDraft(S2, { value: "b" }); + assert.equal(getComposerDraft(S1), before, "writing B must not change A's snapshot identity"); + }); + + test("updater form patches against the CURRENT session's draft", () => { + setComposerDraft(S1, { attachments: ["@one"] }); + setComposerDraft(S2, { attachments: ["@b-one"] }); + setComposerDraft(S2, (current: ComposerDraft) => ({ + attachments: [...current.attachments, "@b-two"], + })); + assert.deepEqual(getComposerDraft(S2).attachments, ["@b-one", "@b-two"]); + assert.deepEqual(getComposerDraft(S1).attachments, ["@one"], "A untouched by B's updater"); }); }); @@ -83,25 +161,72 @@ describe("composer draft store — subscription", () => { test("listeners are notified on every write", () => { const seen: string[] = []; const unsubscribe = subscribeComposerDraft(() => { - seen.push(getComposerDraft().value); + seen.push(getComposerDraft(S1).value); }); - setComposerDraft({ value: "a" }); - setComposerDraft({ value: "ab" }); + setComposerDraft(S1, { value: "a" }); + setComposerDraft(S1, { value: "ab" }); unsubscribe(); - setComposerDraft({ value: "abc" }); + setComposerDraft(S1, { value: "abc" }); assert.deepEqual(seen, ["a", "ab"]); }); - test("updater form patches against the CURRENT draft", () => { - setComposerDraft({ attachments: ["@one"] }); - setComposerDraft((current) => ({ - attachments: [...current.attachments, "@two"], - })); - assert.deepEqual(getComposerDraft().attachments, ["@one", "@two"]); - // Patches merge — an attachments write must not drop `value`. - setComposerDraft({ value: "text" }); - setComposerDraft((current) => ({ attachments: [...current.attachments, "@three"] })); - assert.equal(getComposerDraft().value, "text"); - assert.deepEqual(getComposerDraft().attachments, ["@one", "@two", "@three"]); + test("a write to ANY session notifies subscribers (the composer re-checks its key)", () => { + // The keyed `useSyncExternalStore` getter re-reads on notification; a + // write that no listener ever hears about could leave a stale box on + // screen after a switch. + let notified = 0; + const unsubscribe = subscribeComposerDraft(() => { + notified += 1; + }); + setComposerDraft(S1, { value: "a" }); + setComposerDraft(S2, { value: "b" }); + unsubscribe(); + assert.equal(notified, 2); + }); + + test("patches merge — an attachments write must not drop `value`", () => { + setComposerDraft(S1, { value: "text" }); + setComposerDraft(S1, { attachments: ["@one"] }); + assert.equal(getComposerDraft(S1).value, "text"); + assert.deepEqual(getComposerDraft(S1).attachments, ["@one"]); + }); +}); + +describe("unconfirmedPatchOnTurnEnd — the grey banner dies with its turn (P4)", () => { + const unconfirmed: Pick = { errorKind: "unconfirmed" }; + + test("running falling (true → false) clears the unconfirmed banner", () => { + // The smoke-report scenario: `sleep 35` timed out at the 30s ack + // deadline, the probe said "accepted", the turn finished — and the + // grey banner stayed on screen. The fall is the retire signal. + assert.deepEqual( + unconfirmedPatchOnTurnEnd(true, false, unconfirmed.errorKind), + { error: null, errorKind: null, unconfirmed: null }, + ); + }); + + test("a turn still running keeps the banner", () => { + assert.equal(unconfirmedPatchOnTurnEnd(true, true, "unconfirmed"), null); + }); + + test("no observed turn (false → false, e.g. banner set after a fast turn) keeps it", () => { + // The fast-turn echo path never shows a running fall inside the same + // mount; the pre-existing dismiss paths (next send, session switch) + // stay responsible for that case. #126's display semantics untouched. + assert.equal(unconfirmedPatchOnTurnEnd(false, false, "unconfirmed"), null); + }); + + test("a turn starting (false → true) never clears anything", () => { + assert.equal(unconfirmedPatchOnTurnEnd(false, true, "unconfirmed"), null); + }); + + test("a real rejected refusal is NOT cleared by the turn ending", () => { + // "消息发送失败" is a different fact — it stays until the user acts on + // it or the next send in that session starts. + assert.equal(unconfirmedPatchOnTurnEnd(true, false, "rejected"), null); + }); + + test("no banner at all → no patch", () => { + assert.equal(unconfirmedPatchOnTurnEnd(true, false, null), null); }); }); diff --git a/packages/webui/webapp/test/composer-submit-tripwire.test.ts b/packages/webui/webapp/test/composer-submit-tripwire.test.ts index 7ca9569b..80c47f12 100644 --- a/packages/webui/webapp/test/composer-submit-tripwire.test.ts +++ b/packages/webui/webapp/test/composer-submit-tripwire.test.ts @@ -82,8 +82,8 @@ describe("composer submit ordering — ticket 13 wiring tripwire", () => { ); const clearDraftIdx = indexOfOrThrow( composerSource, - 'setComposerDraft({ value: "", attachments: [] })', - 'setComposerDraft({ value: "", attachments: [] })', + 'setComposerDraft(dispatchDraftKey, { value: "", attachments: [] })', + 'setComposerDraft(dispatchDraftKey, { value: "", attachments: [] })', ); const awaitSendIdx = indexOfOrThrow( composerSource, @@ -246,38 +246,161 @@ describe("composer submit ordering — ticket 13 wiring tripwire", () => { ); }); }); -describe("switching sessions clears the failure banner", () => { - // The error banner lives in the module-scope composer-draft store, which is - // shared by every composer instance (page.tsx swaps the composer between two - // tree positions, so a useState-held draft would die on the swap). Without a - // per-session reset, a rejection recorded in session A kept painting - // session B's composer red after the switch — a send B never made, with a - // red "消息发送失败: HTTP 500" banner appearing "on switching" (P5/P6 of the - // 100-ticket smoke report). The draft TEXT is deliberately not cleared: the - // restored-draft merge owns cross-session text rules. - test("an effect keyed on sessionKey resets error/errorKind/unconfirmed", () => { - // The effect body and the submit-path reset must carry the same three - // fields — a banner kind added later has to join both, and a revert that - // drops the effect (or re-keys it to something that never changes, like a - // stable ref) fails the dependency-array assertion. +describe("the draft store is keyed by session (webui-parity 106, smoke P5)", () => { + // The pre-106 store was ONE shared bucket: session A's draft, chips and + // failure banner rode into session B's view on a switch (the s28 capture + // in the smoke report). #141 papered over the banner half with a + // sessionKey-keyed clear effect — which also destroyed the banner of the + // session the user was RETURNING to. Since 106 the isolation is + // structural: the composer reads and writes the store THROUGH the active + // session key, and the catch branch writes the banner into the DISPATCH + // session's box. These tripwires pin that wiring; the store-level + // behaviour itself is unit-tested in composer-draft.test.ts. + + test("the composer's draft snapshot is read through the session key", () => { + // A revert to the shared bucket re-appears as `getComposerDraft` being + // called with NO key in the useSyncExternalStore call. assert.match( composerSource, - /useEffect\(\(\) => \{\s*setComposerDraft\(\{\s*error: null,\s*errorKind: null,\s*unconfirmed: null,?\s*\}\);\s*\}, \[sessionKey\]\);/, - "composer must reset the banner fields in an effect keyed on sessionKey — " + - "the module-scope draft store outlives sessions, so the banner must be " + - "scoped to the session it failed in", + /useSyncExternalStore\(\s*subscribeComposerDraft,\s*\(\) => getComposerDraft\(sessionKey\),\s*\(\) => getComposerDraft\(""\),?\s*\)/, + "the draft snapshot must be read through the session key so a switch " + + "swaps boxes synchronously — a key-less getter is the shared-bucket " + + "regression this ticket fixes", + ); + }); + + test("the #141 clear effect is gone — isolation is structural now", () => { + // The clear-on-switch effect destroyed a RETURNING session's own + // unread banner. With per-session boxes it is wrong in every case. + assert.ok( + !/useEffect\(\(\) => \{\s*setComposerDraft\(\{?\s*error: null,\s*errorKind: null,\s*unconfirmed: null,?\s*\}?\);?\s*\}, \[sessionKey\]\);/.test( + composerSource, + ), + "the sessionKey-keyed banner-clear effect must not come back — the " + + "keyed store already isolates banners per session, and clearing on " + + "switch loses the session the user returns to", + ); + }); + + test("every submit-path write carries a session key", () => { + // A key-less setComposerDraft call site would write into whatever + // box... nothing — it is a type error; the tripwire pins the two + // load-bearing literals so a refactor that drops the key from them + // fails here rather than silently changing boxes. + assert.ok( + /setComposerDraft\(dispatchDraftKey, \{\s*error: null,\s*errorKind: null,\s*unconfirmed: null,?\s*\}\)/.test( + composerSource, + ), + "submit must clear the banner in the DISPATCH session's box", + ); + assert.match( + composerSource, + /setComposerDraft\(dispatchDraftKey, \{ value: "", attachments: \[\] \}\)/, + "the optimistic clear must write the dispatch session's box", + ); + }); + + test("the catch branch writes the banner into the dispatch session's box", () => { + // The failure belongs to the session that attempted the send. Writing + // it into the LIVE key would repaint the session the user switched TO + // — the exact bleed the s28 capture shows. + const catchStartIdx = indexOfOrThrow( + composerSource, + "} catch (cause) {", + "} catch (cause) {", + ); + const catchBody = composerSource.slice(catchStartIdx); + assert.match( + catchBody, + /setComposerDraft\(dispatchDraftKey, \{\s*error: errorMessage,/, + "the banner must be keyed by dispatchDraftKey inside the catch branch", + ); + assert.ok( + !/setComposerDraft\(sessionKey, \{[^}]*errorMessage/.test(catchBody), + "the banner must NOT be written into the live session's box — a send " + + "that failed in session A must never paint session B red", ); }); test("the submit path still clears the banner before dispatching", () => { - // The session-switch reset is additive; it must not replace the - // clear-on-submit (a retry in the SAME session also has to clear the old - // rejection before the new attempt is judged). + // A same-session retry also has to clear the old rejection before the + // new attempt is judged. assert.match( composerSource, - /setSending\(true\);\s*setComposerDraft\(\{\s*error: null,\s*errorKind: null,\s*unconfirmed: null,?\s*\}\);/, + /setSending\(true\);\s*setComposerDraft\(dispatchDraftKey, \{\s*error: null,\s*errorKind: null,\s*unconfirmed: null,?\s*\}\);/, "submit must clear the banner right after setSending(true), before the " + "optimistic park — a same-session retry starts clean", ); }); }); + +describe("the unconfirmed banner retires when its turn ends (webui-parity 106, smoke P4)", () => { + // After `sleep 35` finished, the grey "服务器一直没有确认" banner stayed + // under the input until the next send or a reload. The decision + // (running-flag fall + errorKind === "unconfirmed") is unit-tested in + // composer-draft.test.ts; this pins the WIRING: the composer must feed + // the turn-end transition into it and apply the patch it returns. + + test("the composer watches the running flag and applies the turn-end patch", () => { + assert.match( + composerSource, + /const prevRunningRef = useRef\(running\);/, + "the previous running value must be captured per render", + ); + const effectIdx = indexOfOrThrow( + composerSource, + "unconfirmedPatchOnTurnEnd(", + "unconfirmedPatchOnTurnEnd( call", + ); + const wiring = composerSource.slice(effectIdx - 200, effectIdx + 400); + assert.match( + wiring, + /prevRunningRef\.current,\s*running,\s*errorKind,/, + "the decision must receive (previous running, running, errorKind)", + ); + assert.match( + wiring, + /prevRunningRef\.current = running;/, + "the reference must advance after the decision, or one stale value " + + "would clear (or keep) the banner on unrelated re-renders", + ); + assert.match( + wiring, + /if \(patch\) setComposerDraft\(sessionKey, patch\);/, + "a non-null patch must be applied to the ACTIVE session's box", + ); + }); + + test("the turn-end decision is imported from the draft module", () => { + assert.match( + composerSource, + /import\s+\{[^}]*\bunconfirmedPatchOnTurnEnd\b[^}]*\}\s+from\s+["']@\/lib\/composer-draft["']/, + "the decision must be the product function, not an inline re-derivation", + ); + }); +}); + +describe("the model picker's local state resets on a session switch (webui-parity 106, smoke P5)", () => { + // The chip VALUE reads the server snapshot, but the cascade's open flag, + // previewed row and per-model draft mirror are component-local; without a + // reset, session A's open menu / preview state visually persisted into + // session B's view. + + test("ModelSelect receives the session key", () => { + assert.match( + composerSource, + /]*sessionKey=\{sessionKey\}/, + "the composer must pass the session key down to the picker", + ); + }); + + test("the picker resets its local states in an effect keyed on sessionKey", () => { + const resetIdx = indexOfOrThrow( + composerSource, + "setOpen(false);\n setSubmenuFor(null);\n setFocusedModelId(null);\n setDrafts({});", + "ModelSelect's four local-state resets", + ); + const deps = composerSource.slice(resetIdx, resetIdx + 200); + assert.match(deps, /\}, \[sessionKey\]\);/, "the reset must be keyed on sessionKey"); + }); +}); diff --git a/packages/webui/webapp/test/page-hydration.test.ts b/packages/webui/webapp/test/page-hydration.test.ts new file mode 100644 index 00000000..4fd7ca35 --- /dev/null +++ b/packages/webui/webapp/test/page-hydration.test.ts @@ -0,0 +1,153 @@ +// webapp/test/page-hydration.test.ts +// +// Static-source tripwires for the page root's storage-access timing +// (webui-parity 106, smoke-report P7-b) and for the scroll-restore +// contract that must survive it (red line: 刷新后滚动位置还在). +// +// Why a tripwire and not a unit test: this suite has no React render +// harness (plain `node --test` over the lib modules), and the defect is +// not a function's output but WHERE a function is called from — the +// render phase of the prerendered root component. `app/page.tsx` is +// pre-rendered by the Next.js static export, so the server HTML and the +// client's first (hydration) render must be byte-identical. Any +// `localStorage` read that runs during render returns defaults on the +// server and stored values on the client — a hydration mismatch that +// today is masked by the `state === null` skeleton and detonates the +// moment that skeleton changes. The reads therefore live in exactly one +// place: the post-mount restore effect. +// +// The load-bearing survivor of that move is the transcript scroll +// restore: the page dropped its render-phase `readScrollPosition` call +// because `Chat` already re-reads the SAME per-session key inside its own +// post-mount effect (and falls back to it whenever `initialScrollTop` is +// absent). That fallback IS the restore behaviour now, so it is pinned +// here too. + +import { test, describe } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { resolve, dirname } from "node:path"; + +const here = dirname(fileURLToPath(import.meta.url)); +const pageSource = readFileSync(resolve(here, "../app/page.tsx"), "utf8"); +const chatSource = readFileSync(resolve(here, "../components/chat.tsx"), "utf8"); + +describe("page.tsx never reads storage during render (webui-parity 106)", () => { + test("no useState initializer calls a persist reader", () => { + // The pre-106 shape — `useState(() => readUiState())` — ran + // localStorage reads on every (re)render entry, server included. + assert.ok( + !/useState[^;]*\(\)\s*=>\s*read(?:UiState|WorkspaceTabs|ScrollPosition)\(/.test(pageSource), + "page.tsx must not seed state from a storage read — the initializer " + + "runs during render, and render runs on the static-export server " + + "too (the hydration bomb P7-b)", + ); + }); + + test("readScrollPosition is gone from the page entirely", () => { + // The scroll wrapper's render-phase read was the third instance. Chat + // re-reads the same key post-mount, so the page must not re-grow it. + assert.ok( + !pageSource.includes("readScrollPosition"), + "the page must not read scroll positions at all — the restore lives " + + "in Chat's sessionKey effect (same key, client-only timing)", + ); + }); + + test("the one storage read lives in the post-mount restore effect", () => { + const effectIdx = pageSource.indexOf("const restoredUi = readUiState();"); + assert.ok(effectIdx >= 0, "the mount restore must call readUiState()"); + const effect = pageSource.slice( + pageSource.lastIndexOf("useEffect(", effectIdx), + effectIdx + 400, + ); + assert.match(effect, /readWorkspaceTabs\(\)/, "tabs restore rides the same effect"); + assert.match(effect, /setPersisted\(restoredUi\)/); + assert.match(effect, /setTabState\(restoredTabs\.tabStrip\)/); + assert.match(effect, /setColumnState\(restoredTabs\.columnLayout\)/); + assert.match(effect, /setPanel\(restoredUi\.panel\)/); + assert.match(effect, /setUiRestored\(true\)/, "the write-back gate must open in the same batch"); + }); + + test("every storage write-back is gated on uiRestored", () => { + // Without the gate, the defaults-seeded first effects would overwrite + // the stored payload BEFORE the restore ran — the red-line-3 data + // loss (panel / tabs / appearance gone after a refresh). + const gateCount = ( + pageSource.match(/if \(!uiRestored\) return;/g) ?? [] + ).length; + assert.ok( + gateCount >= 3, + `expected the write-gate in the panel, tabs and lastSessionId mirrors, found ${gateCount}`, + ); + assert.match( + pageSource, + /useEffect\(\(\) => \{\s*if \(!uiRestored\) return;\s*writeUiState\(/, + "the panel mirror must be gated", + ); + assert.match( + pageSource, + /useEffect\(\(\) => \{\s*if \(!uiRestored\) return;\s*writeWorkspaceTabs\(/, + "the workspace-tabs mirror must be gated", + ); + const lastSessionIdx = pageSource.indexOf("lastSessionId: active,"); + assert.ok(lastSessionIdx >= 0); + const lastSessionEffect = pageSource.slice( + pageSource.lastIndexOf("useEffect(", lastSessionIdx), + lastSessionIdx, + ); + assert.match( + lastSessionEffect, + /if \(!uiRestored\) return;/, + "the lastSessionId mirror must be gated — it can fire before the " + + "restore batch and would drop the stored appearance fields", + ); + }); + + test("the first frame still renders the state=null skeleton uniformly", () => { + // The skeleton is what makes server and client renders identical on + // the first frame; the restore must not have traded it for a + // different first-paint path. + assert.match(pageSource, /if \(!state\) \{/); + assert.match(pageSource, //); + }); + + test("the default-seeded state declarations still exist", () => { + // Belt and braces: the two boxes must start from the shared DEFAULT + // constants (identical on server and client), not from undefined. + assert.match(pageSource, /useState\(DEFAULT_UI_STATE\)/); + assert.match( + pageSource, + /useState\(\s*DEFAULT_WORKSPACE_TABS_STATE,?\s*\)/, + ); + }); +}); + +describe("Chat owns the scroll restore (red line: 刷新后滚动位置还在)", () => { + test("the restore reads the persisted key inside the sessionKey effect", () => { + // Chat's effect re-reads `webui:scroll:v1::` whenever + // the session changes and whenever `initialScrollTop` is absent — + // which is now ALWAYS, since the page passes no such prop. Break this + // line and a refresh lands every conversation back at the top. + assert.match( + chatSource, + /const saved = readPersistedScroll\(sessionKey\);/, + "Chat must re-read the persisted scroll position per session key", + ); + assert.match( + chatSource, + /const best = explicit !== null && explicit > 0 \? explicit : saved;/, + "the saved value must be the fallback when no explicit prop arrives", + ); + }); + + test("the page still persists scroll positions through onScrollPersist", () => { + assert.match( + pageSource, + /onScrollPersist=\{\(top\) => \{/, + "the write half of the scroll contract stays on the page", + ); + assert.match(pageSource, /writeScrollPosition\(sessionId, top\)/); + }); +}); diff --git a/packages/webui/webapp/test/send-confirmation.test.ts b/packages/webui/webapp/test/send-confirmation.test.ts index 122685a1..a7203ae6 100644 --- a/packages/webui/webapp/test/send-confirmation.test.ts +++ b/packages/webui/webapp/test/send-confirmation.test.ts @@ -312,7 +312,7 @@ describe("the composer is wired to the probe, not to the deadline", () => { describe("the draft store carries the kind, not a string to match on", () => { test("reset gives a clean record", () => { resetComposerDraftForTests(); - assert.deepEqual(getComposerDraft(), { + assert.deepEqual(getComposerDraft("s1"), { value: "", error: null, errorKind: null, @@ -323,8 +323,8 @@ describe("the draft store carries the kind, not a string to match on", () => { test("the kind and the outcome are independent fields", () => { resetComposerDraftForTests(); - setComposerDraft({ error: "", errorKind: "unconfirmed", unconfirmed: "accepted" }); - assert.equal(getComposerDraft().errorKind, "unconfirmed"); - assert.equal(getComposerDraft().unconfirmed, "accepted"); + setComposerDraft("s1", { error: "", errorKind: "unconfirmed", unconfirmed: "accepted" }); + assert.equal(getComposerDraft("s1").errorKind, "unconfirmed"); + assert.equal(getComposerDraft("s1").unconfirmed, "accepted"); }); }); diff --git a/packages/webui/webapp/test/slash-routing.test.ts b/packages/webui/webapp/test/slash-routing.test.ts index 920aa049..2d1c1f2c 100644 --- a/packages/webui/webapp/test/slash-routing.test.ts +++ b/packages/webui/webapp/test/slash-routing.test.ts @@ -296,14 +296,15 @@ describe("rejected submissions come back into the composer", () => { // merged patch into it is the difference between "the text came // back" and "the text came back until the next re-render". resetComposerDraftForTests(); - setComposerDraft({ value: "", attachments: [] }); + setComposerDraft("s1", { value: "", attachments: [] }); setComposerDraft( - mergeRestoredDraft(getComposerDraft(), { + "s1", + mergeRestoredDraft(getComposerDraft("s1"), { content: "/stop", attachments: [], }), ); - assert.equal(getComposerDraft().value, "/stop"); + assert.equal(getComposerDraft("s1").value, "/stop"); resetComposerDraftForTests(); }); @@ -361,7 +362,7 @@ describe("rejected submissions come back into the composer", () => { const guardEnd = composerSource.indexOf("}", guardIdx + guard.length); const body = composerSource.slice(guardIdx, guardEnd); assert.ok( - /setComposerDraft\(\s*mergeRestoredDraft\(getComposerDraft\(\),\s*restored\)\s*\)/.test( + /setComposerDraft\(\s*dispatchDraftKey,\s*mergeRestoredDraft\(getComposerDraft\(dispatchDraftKey\),\s*restored\),?\s*\)/.test( body, ), `the guarded body must write mergeRestoredDraft(…) back through ` + diff --git a/release/public-source.json b/release/public-source.json index 9b8a290f..f149168c 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3938,6 +3938,7 @@ "packages/webui/webapp/test/message-enter-animation.test.ts", "packages/webui/webapp/test/modals-decision-channels.test.ts", "packages/webui/webapp/test/open-file.test.ts", + "packages/webui/webapp/test/page-hydration.test.ts", "packages/webui/webapp/test/plugins-surface.test.ts", "packages/webui/webapp/test/preview-edit.test.ts", "packages/webui/webapp/test/provider-management.test.ts", From af01e0ef4c0d2f7ae441fba1809b006aaa93409c Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 01:15:21 +0800 Subject: [PATCH 05/21] refactor(webui): the plugins and turn-diff routes take the host from the engine facade MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Migration step M3, batch B0 (engine-abstraction). 13 endpoints across routes/plugins.js and routes/turn-diff.js reached the catalogue host by importing lib/acp-client.js#getCatalogueHost directly. They now call getEngineCatalogueHost() from the facade. - engine/host.js: the lazy bridge. Its only import is a dynamic `await import("../lib/acp-client.js")` inside the function body, so engine/index.js gains a function and not a module load. That boundary is the whole point: app.js reaches engine/index.js through routes/engine-capabilities.js, and a static import of acp-client there would put the ACP client tree on every server start — the regression M1 paid for once (209ms -> 2700ms; facade load 4685ms -> 5ms once declaration and construction were split). The value is forwarded verbatim, `null` included, so "host did not boot" stays RUNTIME_UNAVAILABLE and never a second host. - engine/index.js re-exports the getter; the two routes import it from there and no longer name acp-client.js. - No endpoint behaviour changes: same wire shapes, statuses, codes, same `deps.getCliService` / `deps.getDiffApplication` seams, same one process-wide host. Measured on the module graph: routes/plugins.js drops from 13 product files + @mavis/shared to 8 files and zero bare packages; engine/index.js's whole closure is 6 files and 0 bare specifiers. Server start and the facade's own load are unchanged (facade ~1.2ms -> ~3ms, i.e. one more 45-line zero-import file; boot stays in the same 200-300ms band) because lib/state-bus.js already pulls acp-client into app.js's boot graph — closing that edge belongs to the catalogue read/write batches (M3-B1+), not here. Tests: test/lib/engine/host-facade.test.js pins the contract against the real module graph rather than against source text — a resolve hook (module.registerHooks) in a fresh process reports, per parent, which specifiers each entry resolved. It asserts neither route has a direct edge to acp-client/runtime-host/acp.mjs, that loading engine/index.js pulls no host module and no @mavis/* or @minimax/* package, that engine/host.js is in that closure, and the source-shape tripwires (dynamic import only, facade re-export). Mutation-checked: making the facade import statically turns 4 tests red, making plugins.js import directly turns 4 more red. The existing plugins/turn-diff suites pass unchanged under both transports (158 tests x acp and x runtime). Docs: ARCHITECTURE.md + .zh-CN.md — the engine/ file table gains engine/host.js on top of the six files #143 + the doc batch settled, the "one host" rule now names the facade, and the boot-path discipline is stated where the file list lives. docs/webui.md + .zh-CN.md are untouched: no user-visible change. Source inventory regenerated for the two new files (rebase conflict in it was resolved by taking the upstream copy and regenerating, never by hand). --- packages/webui/docs/ARCHITECTURE.md | 28 +- packages/webui/docs/ARCHITECTURE.zh-CN.md | 24 +- packages/webui/server/engine/host.js | 40 +++ packages/webui/server/engine/index.js | 12 +- packages/webui/server/lib/acp-client.js | 24 +- packages/webui/server/routes/plugins.js | 9 +- packages/webui/server/routes/turn-diff.js | 9 +- .../webui/test/lib/engine/host-facade.test.js | 264 ++++++++++++++++++ release/public-source.json | 2 + 9 files changed, 384 insertions(+), 28 deletions(-) create mode 100644 packages/webui/server/engine/host.js create mode 100644 packages/webui/test/lib/engine/host-facade.test.js diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index 0677d05a..30968c00 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -489,17 +489,25 @@ not import it but adopts the same shape. Unknown future statuses render as ### `engine/` (capability declarations + the local-runtime-v2 host) The engine abstraction lives at `server/engine/` (engine-abstraction -batch B1; migration state M1). Six files, one job each: +batch B1; migration state M1, plus M3's first batch B0). Seven files, +one job each: | File | Owns | | --- | --- | | `engine/capabilities.js` | The contract: `ENGINE_CAPABILITY_KEYS` (the 14 matrix keys), `validateEngineCapabilities`, `assertEngineCapability`, `summarizeUnavailableCapabilities` | | `engine/errors.js` | `EngineCapabilityNotSupportedError` + `engineCapabilityHttpResponse` (the 501 payload shape) | -| `engine/index.js` | The facade: `getEngineProvider`, `listEngineProviderIds` (registry by provider id; transport selection arrives with migration step M4) | +| `engine/host.js` | `getEngineCatalogueHost` — the lazy bridge to the one catalogue host. No static import of the host module: the getter body is a dynamic `import()` of `lib/acp-client.js`, so the facade costs a function, not a module load | +| `engine/index.js` | The facade: `getEngineProvider`, `listEngineProviderIds`, `getEngineCatalogueHost` (registry by provider id; transport selection arrives with migration step M4) | | `engine/providers/local-runtime-v2.capabilities.js` | `LOCAL_RUNTIME_V2_CAPABILITIES` — **declaration only, and the split is load-bearing**: its sole import is `../capabilities.js`, so `/api/engine-capabilities` can read the capability table without pulling the v2 host's TypeScript dependency tree (~4.7 s of first-compile) into the boot path. That tree stays behind the same lazy boundary `acp-client.js` already documented | -| `engine/providers/local-runtime-v2.js` | `createCatalogueHost` (moved verbatim from `runtime-host.js`, which re-exports it) + re-exports the declaration above, so consumers keep one import shape | +| `engine/providers/local-runtime-v2.js` | `createCatalogueHost` (moved verbatim from `runtime-host.js`, which re-exports it) + re-exports the declaration above, so consumers keep one import shape. This is the heavy one — `@mavis/local-runtime-v2`, `@mavis/config`, `@minimax/code/runtime-adapter` — and no file `app.js` reaches may import it | | `engine/providers/tui-runtime-adapter.js` | `TUI_RUNTIME_ADAPTER_CAPABILITIES` (declaration only — the adapter itself is constructed inside the v2 host) | +Routes take the host from the facade and never from `lib/acp-client.js`: +`routes/plugins.js` and `routes/turn-diff.js` call +`getEngineCatalogueHost()`. Both keep a `deps`-injected data source +(`deps.getCliService`, `deps.getDiffApplication`) so the handler suites stay +hermetic. + Declaration discipline (admission rules for any future provider, enforced by the snapshot tests in `test/lib/engine/capabilities.test.js`): @@ -516,8 +524,11 @@ by the snapshot tests in `test/lib/engine/capabilities.test.js`): forbidden** — a missing capability must be legible before the call and loud after it (#110 fake-success discipline). 4. One host per provider process-wide: `createCatalogueHost` remains the - single owner of the runtime instance (`acp-client.js#getCatalogueHost` - keeps its "Never build a second host" rule); `close()` stays bounded. + single owner of the runtime instance, and the only way to reach it is the + facade's `getEngineCatalogueHost()` (which forwards to + `acp-client.js#getCatalogueHost` and its "Never build a second host" rule); + `close()` stays bounded. Two `CliService` instances over one dataDir is a + split brain against the plugin / local-disable tables, not a redundancy. 5. Levels drive the UI, never provider names: the frontend reads `GET /api/engine-capabilities` (`routes/engine-capabilities.js#handleEngineCapabilities`) and renders `full` / `partial`(+missing) / `none` — no hard-coded @@ -561,6 +572,13 @@ Runtime probing (downgrading a declared level when the environment disagrees) is deliberately absent in this batch — see `engine/index.js` for the reasoning. +Boot-path discipline: `app.js` reaches `engine/index.js`, so that file and +everything it imports statically must stay free of `@mavis/*`, +`@minimax/*` and the host modules. M1 learned that by paying for it +(209ms → 2700ms at server start; the facade's own load 4685ms → 5ms after +declaration and construction were split). `test/lib/engine/host-facade.test.js` +enforces it against the real module graph rather than against source text. + ## 4. The `clientState` payload This is the shape every SSE `state` event contains. The webui mirrors diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index 3271b3fc..10884175 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -461,17 +461,23 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 ### `engine/`(能力声明 + local-runtime-v2 host) 引擎抽象层位于 `server/engine/`(engine-abstraction 批次 B1;迁移 -状态 M1)。六个文件,各管一件事: +状态 M1,外加 M3 的首批 B0)。七个文件,各管一件事: | 文件 | 职责 | | --- | --- | | `engine/capabilities.js` | 契约本体:`ENGINE_CAPABILITY_KEYS`(14 个矩阵键)、`validateEngineCapabilities`、`assertEngineCapability`、`summarizeUnavailableCapabilities` | | `engine/errors.js` | `EngineCapabilityNotSupportedError` 与 `engineCapabilityHttpResponse`(501 载荷形状) | -| `engine/index.js` | 门面:`getEngineProvider`、`listEngineProviderIds`(按 provider id 的注册表;按 `MCODE_WEBUI_TRANSPORT` 选传输在迁移步 M4 引入) | +| `engine/host.js` | `getEngineCatalogueHost`——通往那唯一 catalogue host 的惰性桥。对 host 模块零静态 import:函数体里是 `lib/acp-client.js` 的动态 `import()`,所以门面付出的是一个函数,不是一次模块加载 | +| `engine/index.js` | 门面:`getEngineProvider`、`listEngineProviderIds`、`getEngineCatalogueHost`(按 provider id 的注册表;按 `MCODE_WEBUI_TRANSPORT` 选传输在迁移步 M4 引入) | | `engine/providers/local-runtime-v2.capabilities.js` | `LOCAL_RUNTIME_V2_CAPABILITIES`——**只有声明,且这个拆分是有承重意义的**:它唯一的 import 是 `../capabilities.js`,所以 `/api/engine-capabilities` 读能力表时**不会把 v2 host 的 TypeScript 依赖树(首次编译约 4.7 秒)拖进 boot 路径**。那棵依赖树仍留在 `acp-client.js` 早已注明的 lazy 边界之后 | -| `engine/providers/local-runtime-v2.js` | `createCatalogueHost`(自 `runtime-host.js` 原样移入,后者转发导出)+ 转发导出上面的声明,消费方的 import 形状因此不变 | +| `engine/providers/local-runtime-v2.js` | `createCatalogueHost`(自 `runtime-host.js` 原样移入,后者转发导出)+ 转发导出上面的声明,消费方的 import 形状因此不变。它是重的那一个——`@mavis/local-runtime-v2`、`@mavis/config`、`@minimax/code/runtime-adapter`——`app.js` 能触达的文件里绝不许 import 它 | | `engine/providers/tui-runtime-adapter.js` | `TUI_RUNTIME_ADAPTER_CAPABILITIES`(仅声明——adapter 本体在 v2 host 内构造) | +路由从门面取 host,不从 `lib/acp-client.js` 取:`routes/plugins.js` 与 +`routes/turn-diff.js` 调 `getEngineCatalogueHost()`。两者都保留 `deps` +注入的数据源(`deps.getCliService`、`deps.getDiffApplication`), +handler 层测试因此保持封闭。 + 声明纪律(未来任何 provider 的准入规则,由 `test/lib/engine/capabilities.test.js` 的快照测试强制): @@ -485,8 +491,10 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 `501 engine_capability_not_supported`。**禁止空实现**——缺能力必须在 调用前可读、调用后响亮(#110 假成功纪律)。 4. 每 provider 进程内单 host:`createCatalogueHost` 仍是运行时实例的 - 唯一所有者(`acp-client.js#getCatalogueHost` 的「绝不建第二个 host」 - 规则不变);`close()` 保持有界。 + 唯一所有者,触达它的唯一入口是门面的 `getEngineCatalogueHost()` + (转发到 `acp-client.js#getCatalogueHost`,其「绝不建第二个 host」 + 规则不变);`close()` 保持有界。同一 dataDir 上两个 `CliService` + 实例是对 plugin / local-disable 表的脑裂,不是冗余。 5. 驱动 UI 的是档位,不是 provider 名单:前端读 `GET /api/engine-capabilities` (`routes/engine-capabilities.js#handleEngineCapabilities`), @@ -522,6 +530,12 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 运行时探测(环境不符时把声明档位降级)本批刻意未做——理由见 `engine/index.js` 头注释。 +启动路径纪律:`app.js` 会触达 `engine/index.js`,因此该文件及其全部 +静态依赖必须不含 `@mavis/*`、`@minimax/*` 与任何 host 模块。M1 是交过 +学费才换来这条(server 启动 209ms → 2700ms;声明与构造拆成两个文件后, +门面自身加载 4685ms → 5ms)。`test/lib/engine/host-facade.test.js` +对着真实模块图强制它,而不是对着源码文本。 + ## 4. `clientState` 载荷 这是每个 SSE `state` 事件所包含的形状。webui 将其 diff --git a/packages/webui/server/engine/host.js b/packages/webui/server/engine/host.js new file mode 100644 index 00000000..3050d1e4 --- /dev/null +++ b/packages/webui/server/engine/host.js @@ -0,0 +1,40 @@ +// webui/server/engine/host.js +// +// The lazy half of the engine facade: the one place route code asks for +// the live catalogue host (migration step M3, batch B0). +// +// Why the getter cannot simply live in `lib/acp-client.js` and be +// imported from `engine/index.js`: `getCatalogueHost()` is already lazy +// *inside* — it `await import("./runtime-host.js")` on first call — but +// the MODULE is not. `lib/acp-client.js` statically imports +// `../../acp.mjs`, the command registry, the settings/config chain and the +// session-delete module. `engine/index.js` is loaded by `app.js` at boot, +// so a static import of `acp-client.js` there would put the ACP client and +// everything behind it on every server start. That is the exact regression +// M1 already paid for once (209ms → 2700ms; index load 4685ms → 5ms after +// the declaration/construction split). This file exists to keep that +// boundary: the only thing `engine/index.js` gains is a function, and the +// function does not touch the module graph until it is called. +// +// Discipline, unchanged by the indirection: one host per process. This +// function FORWARDS to `acp-client.js#getCatalogueHost`, it does not +// construct anything. Two callers must never end up with two CliService +// instances over one dataDir — that is both wasteful and a split brain +// against the plugin / local-disable tables. +// +// The return value is passed through untouched, `null` included: a host +// that failed to boot is an answer (routes answer `RUNTIME_UNAVAILABLE`), +// never a licence to build a second one or to fall back to another path. + +/** + * The process-lifetime catalogue host, booted on first call. + * + * @returns {Promise} The host (the same object + * `lib/acp-client.js#getCatalogueHost` returns), or `null` when the + * runtime failed to boot. Errors thrown by the getter propagate + * unchanged — callers own the failure mapping. + */ +export async function getEngineCatalogueHost() { + const { getCatalogueHost } = await import("../lib/acp-client.js"); + return getCatalogueHost(); +} diff --git a/packages/webui/server/engine/index.js b/packages/webui/server/engine/index.js index 2237a5c8..da9ef63d 100644 --- a/packages/webui/server/engine/index.js +++ b/packages/webui/server/engine/index.js @@ -32,8 +32,11 @@ // // Migration state (design §2.4): M1 done — the host construction moved // into providers/local-runtime-v2.js and runtime-host.js re-exports it; -// no route's behaviour changed. M2–M4 will route new consumers through -// this facade one endpoint family at a time. +// no route's behaviour changed. M3's first batch (B0) done — the +// catalogue host itself is now reached through this facade too +// (engine/host.js), so the plugins and turn-diff routes no longer name +// lib/acp-client.js. The rest of M3, then M4, will route new consumers +// through this facade one endpoint family at a time. import { ENGINE_CAPABILITY_KEYS } from "./capabilities.js"; // Declarations only — importing the provider *host-construction* modules @@ -42,6 +45,9 @@ import { ENGINE_CAPABILITY_KEYS } from "./capabilities.js"; // Host construction stays behind the lazy boundary runtime-host.js // always had; nothing on the boot path may import // providers/local-runtime-v2.js or providers/acp.js-style host modules. +// The same rule applies one level up: engine/host.js reaches +// lib/acp-client.js through a dynamic import, so re-exporting it here +// costs a function, not a module load. import { LOCAL_RUNTIME_V2_CAPABILITIES } from "./providers/local-runtime-v2.capabilities.js"; import { TUI_RUNTIME_ADAPTER_CAPABILITIES } from "./providers/tui-runtime-adapter.js"; @@ -52,6 +58,8 @@ export { engineCapabilityHttpResponse, isEngineCapabilityNotSupportedError, } from "./errors.js"; +// The lazy host getter: a function definition, no host, no @mavis/* import. +export { getEngineCatalogueHost } from "./host.js"; export { LOCAL_RUNTIME_V2_CAPABILITIES } from "./providers/local-runtime-v2.capabilities.js"; export { TUI_RUNTIME_ADAPTER_CAPABILITIES } from "./providers/tui-runtime-adapter.js"; diff --git a/packages/webui/server/lib/acp-client.js b/packages/webui/server/lib/acp-client.js index 68a36d3c..1d48b802 100644 --- a/packages/webui/server/lib/acp-client.js +++ b/packages/webui/server/lib/acp-client.js @@ -36,14 +36,22 @@ let _catalogueHostInitPromise = null; /** * The process-lifetime catalogue host singleton, booted on first call. * - * Exported because `/api/plugins/*` (routes/plugins.js) needs the runtime's - * `cliService` as its only data source, and the host is the single owner of - * that service. Routing plugins through the exported getter is deliberate: - * `transportWantsCatalogue()` below gates *session-list* traffic only — in - * ACP protocol there is no plugin method at all, so gating plugins on the - * transport would leave the panel dead in the default `acp` mode. Callers - * must never construct a second host: two CliService instances on one dataDir - * is both wasteful and a split-brain against the plugin/local-disable tables. + * Exported because `/api/plugins/*` (routes/plugins.js) and + * `/api/turn-diff*` (routes/turn-diff.js) need the runtime's `cliService` + * and `applications.session.diff` as their only data sources, and the host + * is the single owner of both. Since migration step M3's first batch (B0) + * those routes no longer import this module: they call the facade's + * `getEngineCatalogueHost()` (server/engine/host.js), which forwards here + * through a dynamic import, because `app.js` loads the engine facade at + * boot and this module carries the ACP client tree. The reasons below are + * the facade's reasons now, and the facade forwards them unchanged. + * + * Routing plugins through the host is deliberate: `transportWantsCatalogue()` + * below gates *session-list* traffic only — in ACP protocol there is no + * plugin method at all, so gating plugins on the transport would leave the + * panel dead in the default `acp` mode. Callers must never construct a + * second host: two CliService instances on one dataDir is both wasteful and + * a split-brain against the plugin/local-disable tables. * * Resolves to `null` when the runtime fails to boot; callers answer * `RUNTIME_UNAVAILABLE` rather than falling back to another path. diff --git a/packages/webui/server/routes/plugins.js b/packages/webui/server/routes/plugins.js index 1f24b6da..01023a34 100644 --- a/packages/webui/server/routes/plugins.js +++ b/packages/webui/server/routes/plugins.js @@ -19,8 +19,9 @@ // stays in the runtime. // // Data source: the catalogue host's `cliService`, reached through the -// exported `getCatalogueHost()` singleton in `lib/acp-client.js`. The host is -// booted unconditionally on first call, on purpose: +// engine facade's `getEngineCatalogueHost()` (server/engine/host.js), which +// forwards to the `getCatalogueHost()` singleton in `lib/acp-client.js`. +// The host is booted unconditionally on first call, on purpose: // // - `MCODE_WEBUI_TRANSPORT` defaults to `acp`, and `transportWantsCatalogue()` // only gates *session-list* traffic. ACP has no plugin method at all, so @@ -47,7 +48,7 @@ // ("official" | "local") so the webapp never has to import the protocol // package to tell the two apart (`@mavis/webui` does not depend on it). -import { getCatalogueHost } from "../lib/acp-client.js"; +import { getEngineCatalogueHost } from "../engine/index.js"; import { readJson } from "../lib/read-json.js"; /** Page size when the caller sends no `limit`; matches the facade default. */ @@ -78,7 +79,7 @@ function json(res, status, payload) { /** The default data source: the catalogue host singleton's bare cliService. */ async function defaultGetCliService() { - const host = await getCatalogueHost(); + const host = await getEngineCatalogueHost(); return host ? host.cliService : null; } diff --git a/packages/webui/server/routes/turn-diff.js b/packages/webui/server/routes/turn-diff.js index bc2ee628..e5cbbbcc 100644 --- a/packages/webui/server/routes/turn-diff.js +++ b/packages/webui/server/routes/turn-diff.js @@ -8,8 +8,9 @@ // Zero new backend. Every endpoint is a thin projection over // `applications.session.diff` (getTurnDiff / revertTurnDiff / reapplyTurnDiff) // on the catalogue host — the same runtime application `routes/plugins.js` -// reaches through `getCatalogueHost()`. This file owns input validation, the -// wire shape, and the post-mutation refresh; it owns no diff logic. +// reaches through the engine facade's `getEngineCatalogueHost()`. This file +// owns input validation, the wire shape, and the post-mutation refresh; it +// owns no diff logic. // // Two deliberate constraints, both from the real-run verification in // `.tickets/webui-parity/82-coord-premise-verification.md`: @@ -43,7 +44,7 @@ // "Only the latest turn diff can be changed" / content-conflict gate, and the // card shows that message verbatim instead of a generic failure. -import { getCatalogueHost } from "../lib/acp-client.js"; +import { getEngineCatalogueHost } from "../engine/index.js"; import { readJson } from "../lib/read-json.js"; import { invalidateSessionTree } from "../lib/session-tree.js"; import { @@ -96,7 +97,7 @@ function readSelector(source) { /** `applications.session.diff` — and only that. */ async function defaultGetDiffApplication() { - const host = await getCatalogueHost(); + const host = await getEngineCatalogueHost(); const diff = host && host.applications ? host.applications.session?.diff : undefined; return diff ?? null; } diff --git a/packages/webui/test/lib/engine/host-facade.test.js b/packages/webui/test/lib/engine/host-facade.test.js new file mode 100644 index 00000000..48cb1d64 --- /dev/null +++ b/packages/webui/test/lib/engine/host-facade.test.js @@ -0,0 +1,264 @@ +// webui/test/lib/engine/host-facade.test.js +// +// Regression guard for migration step M3, batch B0: the catalogue host is +// reached through the engine facade, and the facade itself stays on the +// right side of the boot-path boundary. +// +// The M1 lesson is why this file exists. Moving the plugins and turn-diff +// endpoints onto the facade looks like a rename, and the tempting way to +// write it is a static `import { getCatalogueHost } from +// "../lib/acp-client.js"` inside `engine/host.js`. That compiles and passes +// every handler test — they inject `deps.getCliService` / +// `deps.getDiffApplication`, so the default getter never runs — while +// putting the ACP client tree behind `engine/index.js`, which `app.js` loads +// at boot. M1 already paid for that mistake once (209ms → 2700ms; the +// facade's own load 4685ms → 5ms after declaration and construction were +// split into two files). +// +// So the assertions come in two kinds, and the second is the load-bearing +// one: +// +// 1. Source shape — the two route files name the facade and never +// `lib/acp-client.js`; `engine/host.js` reaches the singleton through a +// dynamic import; `engine/index.js` re-exports the getter. +// 2. The real module graph — a fresh child process installs a +// `module.registerHooks` resolve hook, imports one entry, and reports +// every specifier the loader was asked to resolve, per parent. That +// yields the entry's direct edges and its transitive closure without +// guessing from the source text. A timing assertion would pass on a +// fast machine and fail on a loaded one; the module graph is a fact. +// +// Scope note, so this file is not mistaken for a global invariant: +// `routes/turn-diff.js` still pulls `lib/acp-client.js` TRANSITIVELY, through +// `lib/state-bus.js` (which app.js loads anyway). B0 removes the two direct +// edges; closing the state-bus one belongs to the batches that route the +// catalogue read/write families (M3-B1+), not here. The turn-diff assertions +// are therefore about its direct edges only, and say so. + +import { test, describe } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { execFileSync } from "node:child_process"; +import { join, relative } from "node:path"; +import { fileURLToPath, pathToFileURL } from "node:url"; + +const packageDir = join(import.meta.dirname, "..", "..", ".."); +const serverDir = join(packageDir, "server"); +const read = (rel) => readFileSync(join(serverDir, rel), "utf8"); + +/** Strip comments — the prose in these files legitimately names the modules under guard. */ +function codeOf(source) { + return source + .replace(/\/\*[\s\S]*?\*\//g, "") + .split("\n") + .map((line) => line.replace(/^\s*\/\/.*$/, "")) + .join("\n"); +} + +/** + * Product files that carry the runtime / ACP tree. Loading any of them from + * the facade, whatever the reason given, is the regression. + */ +const FORBIDDEN_ON_BOOT_PATH = new Set([ + "lib/acp-client.js", + "lib/runtime-host.js", + "acp.mjs", + "providers/local-runtime-v2.js", +]); + +/** Bare specifiers whose package alone is enough to blow the boot budget. */ +const HEAVY_PACKAGE_PREFIXES = ["@mavis/", "@minimax/"]; + +// --- 1. source shape ------------------------------------------------------- + +describe("the host-consuming routes reach the host through the facade", () => { + for (const route of ["routes/plugins.js", "routes/turn-diff.js"]) { + test(`${route} imports the facade, not the host singleton module`, () => { + const code = codeOf(read(route)); + assert.ok( + !/from\s+"\.\.\/lib\/acp-client\.js"/.test(code), + `${route} must not import lib/acp-client.js directly — take the host from ../engine/index.js`, + ); + assert.ok( + /import\s*\{[^}]*getEngineCatalogueHost[^}]*\}\s*from\s+"\.\.\/engine\/index\.js"/.test(code), + `${route} must import getEngineCatalogueHost from ../engine/index.js`, + ); + assert.match(code, /await getEngineCatalogueHost\(\)/); + }); + + test(`${route} never names getCatalogueHost`, () => { + // Belt and braces: a future re-export of the raw getter from the facade + // would satisfy the import assertion above while quietly restoring the + // old name. The call site is what has to move. + assert.ok( + !/\bgetCatalogueHost\b/.test(codeOf(read(route))), + `${route} must not mention getCatalogueHost — the facade getter is getEngineCatalogueHost`, + ); + }); + } + + test("engine/host.js reaches the singleton through a dynamic import only", () => { + const code = codeOf(read("engine/host.js")); + assert.ok( + /await\s+import\(\s*"\.\.\/lib\/acp-client\.js"\s*\)/.test(code), + "the facade must load lib/acp-client.js with await import()", + ); + // A static `import ... from "../lib/acp-client.js"` anywhere in this file + // — even one used for nothing but a type — puts the module back on the + // boot path, because app.js loads engine/index.js. + assert.ok( + !/^\s*import\s[^\n]*"\.\.\/lib\/acp-client\.js"/m.test(code), + "engine/host.js must not statically import lib/acp-client.js", + ); + }); + + test("engine/index.js re-exports the facade getter", () => { + // Unexported, both routes would import `undefined` and throw on the first + // real request — a failure no handler test reaches, since they inject + // their own data source. + assert.match( + read("engine/index.js"), + /export\s*\{[^}]*getEngineCatalogueHost[^}]*\}\s*from\s*"\.\/host\.js"/, + "engine/index.js must re-export getEngineCatalogueHost from ./host.js", + ); + }); +}); + +// --- 2. the real module graph ---------------------------------------------- + +/** + * Import `entryRel` in a fresh Node process and report its module graph: + * the specifiers resolved with the entry as their direct parent, the + * transitive set of product files, and the bare package specifiers. + * + * `module.registerHooks` is in-thread and unflagged on every Node this + * package supports (engines: >=22.19), so this needs no loader file and no + * experimental flag. The child mirrors the server's own source-layout + * bootstrap (`registerWorkspaceSources`, see server/lib/workspace-sources.js) + * and installs the hook AFTER it, so the recorded set is the entry's graph + * and not the harness's. The payload is framed by a sentinel because + * importing the graph legitimately prints lines of its own (lib/config.js + * logs the resolved workspace on load). + */ +function moduleGraphOf(entryRel) { + const entryUrl = pathToFileURL(join(serverDir, entryRel)).href; + const sentinel = "__MODULE_GRAPH__"; + const script = ` + import { registerHooks } from "node:module"; + const { registerWorkspaceSources } = await import(${JSON.stringify( + pathToFileURL(join(serverDir, "lib/workspace-sources.js")).href, + )}); + registerWorkspaceSources(); + const seen = []; + registerHooks({ + resolve(specifier, context, nextResolve) { + const resolved = nextResolve(specifier, context); + seen.push({ specifier, url: resolved.url, parent: context.parentURL }); + return resolved; + }, + }); + const entry = await import(${JSON.stringify(entryUrl)}); + process.stdout.write("\\n${sentinel}" + JSON.stringify({ seen, exports: Object.keys(entry) })); + `; + const stdout = execFileSync(process.execPath, ["--input-type=module", "-e", script], { + cwd: packageDir, + encoding: "utf8", + stdio: ["ignore", "pipe", "pipe"], + }); + const framed = stdout.slice(stdout.lastIndexOf(sentinel) + sentinel.length); + const { seen, exports: exportNames } = JSON.parse(framed); + + const isProductFile = (url) => url.startsWith(pathToFileURL(join(serverDir, "")).href); + const asProductPath = (url) => relative(serverDir, fileURLToPath(url)).split("\\").join("/"); + + return { + exportNames, + // Direct edges of the entry: what this file itself asks the loader for. + directSpecifiers: seen.filter((e) => e.parent === entryUrl).map((e) => e.specifier), + // Everything reachable, transitively. Product files only: a dependency's + // own internals are not what this guard is about. + productFiles: [...new Set(seen.filter((e) => isProductFile(e.url)).map((e) => asProductPath(e.url)))], + bareSpecifiers: [ + ...new Set( + seen + .map((e) => e.specifier) + .filter((s) => !s.startsWith(".") && !s.startsWith("file:") && !s.startsWith("node:")), + ), + ], + }; +} + +describe("the two routes resolve the host module through the facade only", () => { + for (const route of ["routes/plugins.js", "routes/turn-diff.js"]) { + test(`${route} has no direct edge to the acp client module`, () => { + // The resolved graph, not the source text: a barrel re-export that + // pulls acp-client in behind the facade would still show up here. + const graph = moduleGraphOf(route); + assert.ok( + graph.directSpecifiers.includes("../engine/index.js"), + `${route} must resolve ../engine/index.js directly (got: ${graph.directSpecifiers.join(", ")})`, + ); + for (const specifier of graph.directSpecifiers) { + assert.ok( + !/acp-client|runtime-host|acp\.mjs/.test(specifier), + `${route} must not resolve ${specifier} directly`, + ); + } + }); + } +}); + +describe("the facade stays off the heavy side of the boot path", () => { + // engine/index.js is loaded by app.js at boot (via + // routes/engine-capabilities.js), so its closure is boot cost. + test("importing engine/index.js loads neither a host module nor a @mavis package", () => { + const graph = moduleGraphOf("engine/index.js"); + + for (const file of graph.productFiles) { + assert.ok( + !FORBIDDEN_ON_BOOT_PATH.has(file), + `engine/index.js loaded ${file} — the host modules stay behind a dynamic import`, + ); + } + for (const specifier of graph.bareSpecifiers) { + for (const prefix of HEAVY_PACKAGE_PREFIXES) { + assert.ok( + !specifier.startsWith(prefix), + `engine/index.js resolved ${specifier} — @mavis/* and @minimax/* are not boot-path modules`, + ); + } + } + }); + + test("the facade getter reaches the real module — lazily, not by copying it", () => { + // Laziness is a property of the graph (the assertions above); this is the + // other half: the lazy path is wired to the actual singleton rather than + // to a local stand-in. host-facade's closure contains host.js but not + // acp-client.js, so the edge has to be made by the dynamic import inside + // host.js — asserted on the source in the suite above. + const graph = moduleGraphOf("engine/index.js"); + assert.ok( + graph.productFiles.includes("engine/host.js"), + "engine/index.js must re-export from engine/host.js", + ); + assert.ok( + !graph.productFiles.includes("lib/acp-client.js"), + "engine/host.js must not have hoisted the acp client into a static import", + ); + assert.ok( + graph.exportNames.includes("getEngineCatalogueHost") && graph.exportNames.includes("getEngineProvider"), + "the facade must keep exporting getEngineCatalogueHost and getEngineProvider", + ); + }); + + test("the plugins route is fully light — the facade is its only engine import", () => { + // plugins.js has no other lib dependency, so its whole closure is the + // assertion: before B0 it was 13 product files plus @mavis/shared via the + // direct acp-client import, now the facade and the body reader alone. + const graph = moduleGraphOf("routes/plugins.js"); + for (const file of graph.productFiles) { + assert.ok(!FORBIDDEN_ON_BOOT_PATH.has(file), `routes/plugins.js loaded ${file}`); + } + assert.deepEqual(graph.bareSpecifiers, [], "routes/plugins.js must not pull a bare package"); + }); +}); diff --git a/release/public-source.json b/release/public-source.json index f149168c..2cdd8f24 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3447,6 +3447,7 @@ "packages/webui/server/cleanup.js", "packages/webui/server/engine/capabilities.js", "packages/webui/server/engine/errors.js", + "packages/webui/server/engine/host.js", "packages/webui/server/engine/index.js", "packages/webui/server/engine/providers/local-runtime-v2.capabilities.js", "packages/webui/server/engine/providers/local-runtime-v2.js", @@ -3588,6 +3589,7 @@ "packages/webui/test/lib/engine-provider-sync.test.js", "packages/webui/test/lib/engine/capabilities.test.js", "packages/webui/test/lib/engine/capability-snapshot.test.js", + "packages/webui/test/lib/engine/host-facade.test.js", "packages/webui/test/lib/events-concurrency.test.js", "packages/webui/test/lib/events-hash.test.js", "packages/webui/test/lib/events.test.js", From f1842ba5341cff97783ea44bb66c3e33741b3107 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 01:25:42 +0800 Subject: [PATCH 06/21] test(webui): make the run-mirror, first-turn-guard and mavis-usage suites immune to the gate's isolation env MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The webui gate runs with MCODE_WEBUI_DATA_DIR, MCODE_WEBUI_SETTINGS_PATH and MINIMAX_DATA_DIR exported at a scratch directory. Two suites read paths those exports take away from them: - config.js#resolveDataDir reads MINIMAX_DATA_DIR ?? MAVIS_DATA_DIR, so the gate's MINIMAX_DATA_DIR outranked mavis-usage.check.mjs's own MAVIS_DATA_DIR fixture and every DB-backed case resolved null against a scratch dir that holds no runtime-state.sqlite. The suite now exports the name that wins. - config.js resolves SESSIONS_DB as MCODE_WEBUI_SESSIONS_DB || join(WEBUI_DATA_DIR, "sessions.json"). A caller that exports MCODE_WEBUI_SESSIONS_DB redirects the store, while the suite's beforeEach still cleared join(DATA_DIR, "sessions.json") — so each run read the previous run's records and the mid-run switch resolved an id whose workspace belonged to a since-removed tmp dir. Both chat-route suites now pin MCODE_WEBUI_SESSIONS_DB to the same path their cleanup clears. Test-only: no server/ code, no helper under test/helpers/_setup.js, and no assertion weakened or skipped. Verified with the three variables set, with MCODE_WEBUI_SESSIONS_DB additionally set, and bare. --- packages/webui/test/lib/mavis-usage.check.mjs | 17 ++++++++++++++++- .../chat-first-turn-session-guard.check.mjs | 14 +++++++++++++- .../webui/test/routes/chat-run-mirror.check.mjs | 17 ++++++++++++++++- 3 files changed, 45 insertions(+), 3 deletions(-) diff --git a/packages/webui/test/lib/mavis-usage.check.mjs b/packages/webui/test/lib/mavis-usage.check.mjs index 63e6afce..26f4fd5d 100644 --- a/packages/webui/test/lib/mavis-usage.check.mjs +++ b/packages/webui/test/lib/mavis-usage.check.mjs @@ -32,8 +32,23 @@ import { setupMocks, absPath } from "../helpers/_setup.js"; // Point config.js's MAVIS_DATA_DIR at our fixture dir BEFORE mavis-usage.js // is imported. config.js reads process.env.MAVIS_DATA_DIR at module-load // time, so the env var must be set before the dynamic import below. +// +// BOTH names must be set, not just MAVIS_DATA_DIR. config.js#resolveDataDir +// reads `MINIMAX_DATA_DIR ?? MAVIS_DATA_DIR` — the newer name wins — and +// MAVIS_DB_PATH (the fixture sqlite this suite queries) is derived from it. +// A gate command that isolates the runtime data dir exports MINIMAX_DATA_DIR +// pointing at a scratch directory, and that scratch directory has no +// runtime-state.sqlite, so every DB-backed case here resolved null. +// +// The test's own fixture must outrank whatever the outer environment exports +// or the suite is only green when run bare — which is the trap this pins +// shut. The production precedence in config.js is deliberate and shared with +// packages/config, so the fix belongs here, not there: a test that wants a +// fixture owns the variable, and it owns it by exporting the name that wins. const TEST_DIR = dirname(fileURLToPath(import.meta.url)); -process.env.MAVIS_DATA_DIR = resolve(TEST_DIR, "..", "fixtures"); +const _fixtureDataDir = resolve(TEST_DIR, "..", "fixtures"); +process.env.MAVIS_DATA_DIR = _fixtureDataDir; +process.env.MINIMAX_DATA_DIR = _fixtureDataDir; // Fixture session IDs (created by scripts/create-test-db.mjs). // MUST match /mvs_[a-f0-9]{16,}/i — only hex chars allowed (no 'l', 'u' etc). diff --git a/packages/webui/test/routes/chat-first-turn-session-guard.check.mjs b/packages/webui/test/routes/chat-first-turn-session-guard.check.mjs index be430936..581e2ed9 100644 --- a/packages/webui/test/routes/chat-first-turn-session-guard.check.mjs +++ b/packages/webui/test/routes/chat-first-turn-session-guard.check.mjs @@ -42,8 +42,20 @@ import { createTurnDrain } from "../helpers/turn-drain.mjs"; // MCODE_WEBUI_DATA_DIR at import time, and lib/events.js resolves the // audit-log path per append (alerts audit-writes on failed sends). Neither // this check nor the operator's real ~/.mcode-webui may see the other. +// +// SESSIONS_DB is pinned EXPLICITLY, for the same reason as its sibling +// chat-run-mirror.check.mjs: config.js resolves it as +// `MCODE_WEBUI_SESSIONS_DB || join(WEBUI_DATA_DIR, "sessions.json")`, so an +// outer MCODE_WEBUI_SESSIONS_DB outranks the default and would leave the +// `beforeEach` below clearing a file this suite never reads. The assertions +// here happen to tolerate a store carrying records from an earlier run, so +// the hazard is latent rather than red — but a suite that writes to a store +// it does not own is one refactor away from the red sibling, and it still +// pollutes whatever store the caller pointed it at. const _tmpDataDir = mkTmpDir("webui-first-turn-guard-"); +const _sessionsDb = join(_tmpDataDir, "sessions.json"); process.env.MCODE_WEBUI_DATA_DIR = _tmpDataDir; +process.env.MCODE_WEBUI_SESSIONS_DB = _sessionsDb; process.env.MCODE_WEBUI_EVENTS_PATH = join(_tmpDataDir, "events.ndjson"); const SERVER_DIR = resolve(import.meta.dirname, "..", "..", "server"); @@ -292,7 +304,7 @@ beforeEach(() => { sb.resetCoalesceState(); // Fresh redirected sessions store per case. try { - rmSync(join(_tmpDataDir, "sessions.json"), { force: true }); + rmSync(_sessionsDb, { force: true }); } catch {} sessions._resetSessionsCacheForTests(); alerts._resetForTests(); diff --git a/packages/webui/test/routes/chat-run-mirror.check.mjs b/packages/webui/test/routes/chat-run-mirror.check.mjs index aca1f8dd..a6468f8c 100644 --- a/packages/webui/test/routes/chat-run-mirror.check.mjs +++ b/packages/webui/test/routes/chat-run-mirror.check.mjs @@ -52,8 +52,23 @@ import { createTurnDrain } from "../helpers/turn-drain.mjs"; // Isolation FIRST — lib/config.js resolves SESSIONS_DB / UPLOAD_DIR from // MCODE_WEBUI_DATA_DIR at import time. Neither this check nor the // operator's real ~/.mcode-webui may see the other. +// +// SESSIONS_DB is pinned EXPLICITLY, not left to the DATA_DIR default. +// config.js resolves it as `MCODE_WEBUI_SESSIONS_DB || join(WEBUI_DATA_DIR, +// "sessions.json")`, so an outer MCODE_WEBUI_SESSIONS_DB — which an +// isolation-minded gate command sets to keep a spawned server.js off the +// real store — outranks the default and silently redirects the store this +// file's `beforeEach` then fails to clear. The result is not a missing-file +// error but a worse one: every run reads the previous run's records, the +// mid-run switch resolves an id whose workspace belongs to a tmp dir that no +// longer exists (`workspace_containment` refusal), and the buffer and the +// record assertions both diverge. Pinning the variable here makes the store +// this file reads and the store this file cleans the same path, whatever the +// caller exports. const _tmpDataDir = mkTmpDir("webui-run-mirror-"); +const _sessionsDb = join(_tmpDataDir, "sessions.json"); process.env.MCODE_WEBUI_DATA_DIR = _tmpDataDir; +process.env.MCODE_WEBUI_SESSIONS_DB = _sessionsDb; process.env.MCODE_WEBUI_EVENTS_PATH = join(_tmpDataDir, "events.ndjson"); const SERVER_DIR = resolve(import.meta.dirname, "..", "..", "server"); @@ -357,7 +372,7 @@ beforeEach(() => { sb.clients.clear(); sb.resetCoalesceState(); try { - rmSync(join(_tmpDataDir, "sessions.json"), { force: true }); + rmSync(_sessionsDb, { force: true }); } catch {} sessions._resetSessionsCacheForTests(); alerts._resetForTests(); From 726d2865c9c65d18deb36182d75ecbb216bfe88d Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 01:26:48 +0800 Subject: [PATCH 07/21] refactor(webui): the plugins and turn-diff routes take the host from the engine facade MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Migration step M3, batch B0 (engine-abstraction). 13 endpoints across routes/plugins.js and routes/turn-diff.js reached the catalogue host by importing lib/acp-client.js#getCatalogueHost directly. They now call getEngineCatalogueHost() from the facade. - engine/host.js: the lazy bridge. Its only import is a dynamic `await import("../lib/acp-client.js")` inside the function body, so engine/index.js gains a function and not a module load. That boundary is the whole point: app.js reaches engine/index.js through routes/engine-capabilities.js, and a static import of acp-client there would put the ACP client tree on every server start — the regression M1 paid for once (209ms -> 2700ms; facade load 4685ms -> 5ms once declaration and construction were split). The value is forwarded verbatim, `null` included, so "host did not boot" stays RUNTIME_UNAVAILABLE and never a second host. - engine/index.js re-exports the getter; the two routes import it from there and no longer name acp-client.js. - No endpoint behaviour changes: same wire shapes, statuses, codes, same `deps.getCliService` / `deps.getDiffApplication` seams, same one process-wide host. Measured on the module graph: routes/plugins.js drops from 13 product files + @mavis/shared to 8 files and zero bare packages; engine/index.js's whole closure is 6 files and 0 bare specifiers. Server start and the facade's own load are unchanged (facade ~1.2ms -> ~3ms, i.e. one more 45-line zero-import file; boot stays in the same 200-300ms band) because lib/state-bus.js already pulls acp-client into app.js's boot graph — closing that edge belongs to the catalogue read/write batches (M3-B1+), not here. Tests: test/lib/engine/host-facade.test.js pins the contract against the real module graph rather than against source text — a resolve hook (module.registerHooks) in a fresh process reports, per parent, which specifiers each entry resolved. It asserts neither route has a direct edge to acp-client/runtime-host/acp.mjs, that loading engine/index.js pulls no host module and no @mavis/* or @minimax/* package, that engine/host.js is in that closure, and the source-shape tripwires (dynamic import only, facade re-export). Mutation-checked: making the facade import statically turns 4 tests red, making plugins.js import directly turns 4 more red. The existing plugins/turn-diff suites pass unchanged under both transports (158 tests x acp and x runtime). Docs: ARCHITECTURE.md + .zh-CN.md — the engine/ file table gains engine/host.js on top of the six files #143 + the doc batch settled, the "one host" rule now names the facade, and the boot-path discipline is stated where the file list lives. docs/webui.md + .zh-CN.md are untouched: no user-visible change. Source inventory regenerated for the two new files (rebase conflict in it was resolved by taking the upstream copy and regenerating, never by hand). From a4fad9614b4a067adb4b324e5181c0b534c92710 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 01:26:57 +0800 Subject: [PATCH 08/21] test(webui): make the run-mirror, first-turn-guard and mavis-usage suites immune to the gate's isolation env MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The webui gate runs with MCODE_WEBUI_DATA_DIR, MCODE_WEBUI_SETTINGS_PATH and MINIMAX_DATA_DIR exported at a scratch directory. Two suites read paths those exports take away from them: - config.js#resolveDataDir reads MINIMAX_DATA_DIR ?? MAVIS_DATA_DIR, so the gate's MINIMAX_DATA_DIR outranked mavis-usage.check.mjs's own MAVIS_DATA_DIR fixture and every DB-backed case resolved null against a scratch dir that holds no runtime-state.sqlite. The suite now exports the name that wins. - config.js resolves SESSIONS_DB as MCODE_WEBUI_SESSIONS_DB || join(WEBUI_DATA_DIR, "sessions.json"). A caller that exports MCODE_WEBUI_SESSIONS_DB redirects the store, while the suite's beforeEach still cleared join(DATA_DIR, "sessions.json") — so each run read the previous run's records and the mid-run switch resolved an id whose workspace belonged to a since-removed tmp dir. Both chat-route suites now pin MCODE_WEBUI_SESSIONS_DB to the same path their cleanup clears. Test-only: no server/ code, no helper under test/helpers/_setup.js, and no assertion weakened or skipped. Verified with the three variables set, with MCODE_WEBUI_SESSIONS_DB additionally set, and bare. From e7c0ce935b003ad8c07cad704250d107da2db077 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 01:57:19 +0800 Subject: [PATCH 09/21] feat(webui): the five read endpoints ask the engine facade, not the transport (M3-B1) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The directory-read family — #9 acp-sessions, #10 acp-session-title, #72 protocol/list-sessions, #74 state, #75 health — reached the engine through whatever MCODE_WEBUI_TRANSPORT happened to be, so "does the engine support this" had no answer anywhere except the absence of a crash. server/engine/ session-reads.js gives it one: each endpoint declares the capability and the provider method it needs, the facade checks the registered provider's declaration first, and a provider that does not offer the read answers 501 through invokeHandler instead of an empty list. Nothing on the wire moves. The facade forwards to the same acp-client exports the routes already called, so the 30s cache, the cwd normalisation, the 30s-stale sidebar push semantics and the catalogue-sessions projection are the same code; handleHealth becomes async because the version now resolves through the facade, which is why app-hono's legacy-parity helper learned to await it. /api/state's snapshot field list is untouched — snapshotViewFields and mcodeSessionsSnapshotFields are the first-frame render contract and this batch adds and removes nothing. Each read also reports where its bytes came from — catalogue, acp, or acp-fallback when the runtime transport asked for a host that never booted. That is metadata, not wire, and it is the difference between a sidebar that degraded and one that pretends. Two things this batch found rather than assumed: the catalogue host exposes no version accessor, so /api/health keeps answering from the ACP initialize mirror and says so rather than inventing a method; and protocol.js#72's old test drove a mock key nothing read, so "the cwd filter works" had never actually been proven. --- packages/webui/docs/ARCHITECTURE.md | 44 +- packages/webui/docs/ARCHITECTURE.zh-CN.md | 39 +- packages/webui/server/engine/index.js | 22 + packages/webui/server/engine/session-reads.js | 302 +++++++++++ packages/webui/server/routes/health.js | 26 +- packages/webui/server/routes/protocol.js | 19 +- packages/webui/server/routes/sessions.js | 29 +- packages/webui/server/routes/state.js | 28 +- packages/webui/test/helpers/_setup.js | 9 + .../test/lib/engine/session-reads.test.js | 486 ++++++++++++++++++ packages/webui/test/routes/health.check.mjs | 67 ++- packages/webui/test/routes/protocol.check.mjs | 53 +- .../webui/test/routes/session-reads.check.mjs | 464 +++++++++++++++++ packages/webui/test/server/app-hono.test.js | 19 +- release/public-source.json | 3 + 15 files changed, 1547 insertions(+), 63 deletions(-) create mode 100644 packages/webui/server/engine/session-reads.js create mode 100644 packages/webui/test/lib/engine/session-reads.test.js create mode 100644 packages/webui/test/routes/session-reads.check.mjs diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index 30968c00..67749444 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -489,7 +489,7 @@ not import it but adopts the same shape. Unknown future statuses render as ### `engine/` (capability declarations + the local-runtime-v2 host) The engine abstraction lives at `server/engine/` (engine-abstraction -batch B1; migration state M1, plus M3's first batch B0). Seven files, +batch B1; migration state M1, plus M3 batches B0 and B1). Eight files, one job each: | File | Owns | @@ -501,6 +501,7 @@ one job each: | `engine/providers/local-runtime-v2.capabilities.js` | `LOCAL_RUNTIME_V2_CAPABILITIES` — **declaration only, and the split is load-bearing**: its sole import is `../capabilities.js`, so `/api/engine-capabilities` can read the capability table without pulling the v2 host's TypeScript dependency tree (~4.7 s of first-compile) into the boot path. That tree stays behind the same lazy boundary `acp-client.js` already documented | | `engine/providers/local-runtime-v2.js` | `createCatalogueHost` (moved verbatim from `runtime-host.js`, which re-exports it) + re-exports the declaration above, so consumers keep one import shape. This is the heavy one — `@mavis/local-runtime-v2`, `@mavis/config`, `@minimax/code/runtime-adapter` — and no file `app.js` reaches may import it | | `engine/providers/tui-runtime-adapter.js` | `TUI_RUNTIME_ADAPTER_CAPABILITIES` (declaration only — the adapter itself is constructed inside the v2 host) | +| `engine/session-reads.js` | The directory-read family's facade calls (`readEngineSessionList`, `readEngineSessionListForWorkspace`, `readEngineSessionTitle`, `readEngineVersion`) and the endpoint→capability table `SESSION_READ_ENDPOINTS` (step M3, batch B1) | Routes take the host from the facade and never from `lib/acp-client.js`: `routes/plugins.js` and `routes/turn-diff.js` call @@ -578,6 +579,47 @@ everything it imports statically must stay free of `@mavis/*`, (209ms → 2700ms at server start; the facade's own load 4685ms → 5ms after declaration and construction were split). `test/lib/engine/host-facade.test.js` enforces it against the real module graph rather than against source text. +`engine/session-reads.js` lives under the same rule: its static imports are +`engine/capabilities.js` and `engine/index.js` only, and `lib/acp-client.js` + +`lib/config.js` are reached through `await import()` inside the functions. + +#### Which endpoints read through the facade (step M3, batch B1) + +`engine/session-reads.js` covers the five directory-read endpoints. Each +row names the capability it gates on and the provider method it depends +on, so a `partial` declaration that drops exactly that method answers 501 +naming it: + +| Endpoint | Capability · sub-item | Value source | +| --- | --- | --- | +| `GET /api/acp-sessions` | `sessionCrud` · `listSessions` | `acp-client.js#getMcodeSessionsForWorkspace` (30s cache, cwd normalisation) | +| `GET /api/acp-session-title` | `sessionCrud` · `getSession` | `acp-client.js#getMcodeSessionTitle` | +| `GET /api/protocol/list-sessions` | `sessionCrud` · `listSessions` | `acp-client.js#listAllMcodeSessions`; the route keeps its own cwd filter | +| `GET /api/state` | `sessionCrud` · `listSessions` | the `mcodeSessions` mirror only — `snapshotViewFields` / `mcodeSessionsSnapshotFields` are untouched | +| `GET /api/health` | none of the 14 keys | the ACP `initialize` `agentInfo.version` mirror; the catalogue host exposes no version accessor, so the facade reports the source instead of inventing one | + +Three properties this layer holds, each with a test behind it: + +1. **One normalizer.** The runtime path is projected by + `lib/catalogue-sessions.js#projectTuiSessionToAcp`, which mirrors the + ACP adapter's `toAcpSessionInfo` rule for rule — `title` and + `updatedAt` are omitted when absent, never emitted as `null`. The + facade forwards that projection; it does not re-project it. +2. **Where the bytes came from is reported, not assumed.** Every read + answers a `source` of `catalogue`, `acp` or `acp-fallback` (the + transport asked for the catalogue host and got `null`). It is + metadata, not wire — the endpoints' payloads are byte-identical before + and after the facade. +3. **The gate is real.** The registered provider declares `sessionCrud` + `full`, so nothing 501s today; the tests drive a fixture declaration + that lacks `listSessions` and assert the 501 payload. A gate nobody + ever exercises is indistinguishable from no gate. + +The transport→provider table has one entry (`runtime`). Under the default +`acp` transport no provider is registered yet, so the gate reports +`unregistered-transport` and passes through — M4 registers the ACP +provider and the table gains its row. Passing through is not the same as +claiming support, and the two are reported differently on purpose. ## 4. The `clientState` payload diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index 10884175..2f8cf85e 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -461,7 +461,7 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 ### `engine/`(能力声明 + local-runtime-v2 host) 引擎抽象层位于 `server/engine/`(engine-abstraction 批次 B1;迁移 -状态 M1,外加 M3 的首批 B0)。七个文件,各管一件事: +状态 M1,外加 M3 的 B0 与 B1 两批)。八个文件,各管一件事: | 文件 | 职责 | | --- | --- | @@ -472,6 +472,7 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 | `engine/providers/local-runtime-v2.capabilities.js` | `LOCAL_RUNTIME_V2_CAPABILITIES`——**只有声明,且这个拆分是有承重意义的**:它唯一的 import 是 `../capabilities.js`,所以 `/api/engine-capabilities` 读能力表时**不会把 v2 host 的 TypeScript 依赖树(首次编译约 4.7 秒)拖进 boot 路径**。那棵依赖树仍留在 `acp-client.js` 早已注明的 lazy 边界之后 | | `engine/providers/local-runtime-v2.js` | `createCatalogueHost`(自 `runtime-host.js` 原样移入,后者转发导出)+ 转发导出上面的声明,消费方的 import 形状因此不变。它是重的那一个——`@mavis/local-runtime-v2`、`@mavis/config`、`@minimax/code/runtime-adapter`——`app.js` 能触达的文件里绝不许 import 它 | | `engine/providers/tui-runtime-adapter.js` | `TUI_RUNTIME_ADAPTER_CAPABILITIES`(仅声明——adapter 本体在 v2 host 内构造) | +| `engine/session-reads.js` | 目录读族的面板调用(`readEngineSessionList`、`readEngineSessionListForWorkspace`、`readEngineSessionTitle`、`readEngineVersion`)与端点→能力对照表 `SESSION_READ_ENDPOINTS`(迁移步 M3 批次 B1) | 路由从门面取 host,不从 `lib/acp-client.js` 取:`routes/plugins.js` 与 `routes/turn-diff.js` 调 `getEngineCatalogueHost()`。两者都保留 `deps` @@ -535,6 +536,42 @@ handler 层测试因此保持封闭。 学费才换来这条(server 启动 209ms → 2700ms;声明与构造拆成两个文件后, 门面自身加载 4685ms → 5ms)。`test/lib/engine/host-facade.test.js` 对着真实模块图强制它,而不是对着源码文本。 +`engine/session-reads.js` 服从同一条纪律:它的静态 import 只有 +`engine/capabilities.js` 与 `engine/index.js`,`lib/acp-client.js` + `lib/config.js` +都在函数体内用 `await import()` 触达。 + +#### 哪些端点走门面读(迁移步 M3 批次 B1) + +`engine/session-reads.js` 覆盖 5 个目录读端点。每一行写明它门控的 +能力键与它依赖的 provider 方法,因此一份恰好缺该方法的 `partial` +声明会 501 并点名是哪个方法: + +| 端点 | 能力 · 子项 | 取值来源 | +| --- | --- | --- | +| `GET /api/acp-sessions` | `sessionCrud` · `listSessions` | `acp-client.js#getMcodeSessionsForWorkspace`(30s 缓存 + cwd 归一化) | +| `GET /api/acp-session-title` | `sessionCrud` · `getSession` | `acp-client.js#getMcodeSessionTitle` | +| `GET /api/protocol/list-sessions` | `sessionCrud` · `listSessions` | `acp-client.js#listAllMcodeSessions`;cwd 过滤仍留在路由里 | +| `GET /api/state` | `sessionCrud` · `listSessions` | 只作用于 `mcodeSessions` 镜像——`snapshotViewFields` / `mcodeSessionsSnapshotFields` 一字未动 | +| `GET /api/health` | 14 键中无对应键 | ACP `initialize` 的 `agentInfo.version` 镜像;catalogue host 没有版本访问器,面板如实报告来源而不是凭空造一个方法 | + +本层守住三条性质,每条背后都有测试: + +1. **只有一个 normalizer。** runtime 路径由 + `lib/catalogue-sessions.js#projectTuiSessionToAcp` 投影,逐条镜像 + ACP adapter 的 `toAcpSessionInfo` 规则——`title` 与 `updatedAt` + 缺失时**省略该键**,绝不输出 `null`。面板原样转发这份投影,不做 + 二次投影。 +2. **字节来自哪里是报告出来的,不是假设的。** 每次读都回答一个 + `source`:`catalogue` / `acp` / `acp-fallback`(传输要了 catalogue + host 但拿到 `null`)。它是元数据,不上线——端点载荷在接面板前后 + 逐字节相同。 +3. **门控是真的。** 已注册的 provider 声明 `sessionCrud` 为 `full`, + 所以今天没有任何端点会 501;测试用一份缺 `listSessions` 的样本声明 + 驱动出 501 载荷。没人跑过的门控与没有门控无法区分。 + +传输→provider 表目前只有 `runtime` 一条。默认 `acp` 传输下尚无已注册 +provider,于是门控报告 `unregistered-transport` 并放行——M4 注册 ACP +provider 后该表补上对应行。放行不等于声称支持,二者刻意分开报告。 ## 4. `clientState` 载荷 diff --git a/packages/webui/server/engine/index.js b/packages/webui/server/engine/index.js index da9ef63d..c3a8d851 100644 --- a/packages/webui/server/engine/index.js +++ b/packages/webui/server/engine/index.js @@ -60,6 +60,28 @@ export { } from "./errors.js"; // The lazy host getter: a function definition, no host, no @mavis/* import. export { getEngineCatalogueHost } from "./host.js"; +// The directory-read family's gated reads (step M3, batch B1). Re-exported +// here so the facade is the one import site for engine reads, but the +// dependency runs the other way too — session-reads.js consults +// getEngineProvider. That cycle is safe for one concrete reason: +// session-reads.js reads NOTHING from this module while it is being +// evaluated. Its own module-scope constant is a literal table, and every +// binding it needs from here (getEngineProvider, DEFAULT_ENGINE_PROVIDER_ID) +// is read inside a function body, so a cold `import("./engine/index.js")` +// can never hit a temporal dead zone. Keep it that way: a new top-level +// `const X = SOMETHING_FROM_INDEX` in session-reads.js breaks the re-export. +// It also stays off the boot path for the reason host.js does — +// lib/acp-client.js and lib/config.js are reached through dynamic import() +// inside the read functions. +export { + SESSION_READ_ENDPOINTS, + assertSessionReadCapability, + readEngineSessionList, + readEngineSessionListForWorkspace, + readEngineSessionTitle, + readEngineVersion, + resolveSessionReadProvider, +} from "./session-reads.js"; export { LOCAL_RUNTIME_V2_CAPABILITIES } from "./providers/local-runtime-v2.capabilities.js"; export { TUI_RUNTIME_ADAPTER_CAPABILITIES } from "./providers/tui-runtime-adapter.js"; diff --git a/packages/webui/server/engine/session-reads.js b/packages/webui/server/engine/session-reads.js new file mode 100644 index 00000000..b1df87d7 --- /dev/null +++ b/packages/webui/server/engine/session-reads.js @@ -0,0 +1,302 @@ +// webui/server/engine/session-reads.js +// +// Migration step M3, batch B1: the directory-read family (目录读族) — +// the five endpoints that only ever ASK the engine what it knows: +// +// #9 GET /api/acp-sessions — sidebar session list (cwd filtered) +// #10 GET /api/acp-session-title — one session's title +// #72 GET /api/protocol/list-sessions — remote-control session list (all) +// #74 GET /api/state — snapshot, mcodeSessions mirror source +// #75 GET /api/health — engine version, /api/state sibling +// +// What this file is for. Before M3 a route asked the ACP client directly +// and inherited whatever the transport happened to be. After M3 the route +// asks the facade, the facade checks the provider's DECLARATION first, and +// a provider that does not offer the read answers 501 through +// `invokeHandler`'s `EngineCapabilityNotSupportedError` mapping instead of +// quietly returning `[]` (the #110 fake-success failure mode). +// +// What this file deliberately does NOT do: +// +// - It does not re-implement listing. `lib/acp-client.js` already owns +// the transport switch and already normalises the catalogue host's +// `TuiSession` through `lib/catalogue-sessions.js#projectTuiSessionToAcp` +// — the one normalizer whose output is the ACP `session/list` wire +// shape the sidebar tree speaks. A second normalizer here would be a +// second answer to a shape question that must have exactly one. +// - It does not construct a host. `getCatalogueHost()` is the process +// singleton; this module only forwards to it (see `Never build a +// second host`). +// - It does not build a host-shaped error of its own for "the engine +// could not boot": a read family that fails over to the ACP mirror +// reports WHERE the bytes came from (`source`) instead of pretending +// the engine answered. Only endpoints whose whole contract is the +// engine (plugins / turn-diff) answer `RUNTIME_UNAVAILABLE`. +// +// Boot-path weight. `app.js` imports the routes, the routes import this +// file, so this file is on the boot path. It therefore statically imports +// nothing heavier than `capabilities.js` and `index.js` (both pure +// declaration modules); `lib/acp-client.js` and `lib/config.js` are +// reached through `await import()` inside the functions. That split is the +// M1 lesson — putting the `@mavis/*` tree on the boot path once cost +// 209ms → 2700ms of server start and broke the integration tests' 3s +// window. +// +// Provider selection is M4's job. `providerByTransport()` maps a transport +// to a REGISTERED provider id; today only `runtime` has one, so under the +// default `acp` transport there is no declaration to check and the gate +// reports `gate: "unregistered-transport"` instead of inventing one. When +// M4 registers the ACP provider this table gains its entry and the gate +// starts answering for the default transport too. + +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose: a missing provider answer is the + * pre-M4 passthrough, an unsupported capability answer is 501. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its temporal + * dead zone on a cold `import("./engine/index.js")` — the evaluation order + * of a re-export is the importer's, not this module's. Every consumer of + * the table is a function anyway. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +/** + * The declaration each endpoint in this family needs, and the sub-item it + * needs from that capability. `subItem` is the provider method the route + * ultimately depends on, so a `partial` declaration that omits exactly that + * method yields 501 naming the method rather than a generic refusal. + * + * `/api/health` is `null`: reading the engine's own version is not any of + * the 14 matrix keys (`updateCheck` is about checking for a NEW version, + * not reporting the installed one), and inventing a key here would put a + * lie in the capability registry. See `readEngineVersion` for what the + * endpoint does instead. + * + * @type {Readonly>} + */ +export const SESSION_READ_ENDPOINTS = Object.freeze({ + "GET /api/acp-sessions": { capability: "sessionCrud", subItem: "listSessions" }, + "GET /api/acp-session-title": { capability: "sessionCrud", subItem: "getSession" }, + "GET /api/protocol/list-sessions": { capability: "sessionCrud", subItem: "listSessions" }, + "GET /api/state": { capability: "sessionCrud", subItem: "listSessions" }, + "GET /api/health": null, +}); + +/** + * Resolve the provider that answers catalogue reads on `transport`, or + * `null` when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveSessionReadProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Check one endpoint of this family against the active provider's + * declaration. Throws `EngineCapabilityNotSupportedError` — which + * `app.js#invokeHandler` turns into 501 — when the declaration says the + * capability (or the exact sub-item) is absent. + * + * @param {string} endpoint A key of SESSION_READ_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null}} + */ +export function assertSessionReadCapability(endpoint, transport) { + const need = SESSION_READ_ENDPOINTS[endpoint]; + if (need === undefined) { + // Caller confusion, not an engine limitation — a plain Error so the + // HTTP layer never answers 501 for a typo in webui's own code. + const err = new Error( + `assertSessionReadCapability: "${endpoint}" is not part of the session-read family ` + + `(known: ${Object.keys(SESSION_READ_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_session_read_endpoint"; + throw err; + } + const provider = resolveSessionReadProvider(transport); + if (need === null) { + return { + endpoint, + gate: "no-capability-key", + provider: provider ? provider.id : null, + capability: null, + subItem: null, + }; + } + if (!provider) { + return { + endpoint, + gate: "unregistered-transport", + provider: null, + capability: need.capability, + subItem: need.subItem, + }; + } + assertEngineCapability(provider.capabilities, need.capability, provider.id, need.subItem); + return { + endpoint, + gate: "checked", + provider: provider.id, + capability: need.capability, + subItem: need.subItem, + }; +} + +// --------------------------------------------------------------------------- +// Reads +// --------------------------------------------------------------------------- + +/** + * Where a list read's bytes actually came from. `catalogue` means the + * in-process host answered (and went through `catalogue-sessions.js`); + * `acp` means the `mcode acp` subprocess answered; `acp-fallback` means + * the transport ASKED for the catalogue host and the host was null, so the + * ACP mirror answered instead — reported rather than hidden, because a + * sidebar that silently loses its runtime path is exactly the degradation + * this batch exists to make visible. + * + * @typedef {"catalogue" | "acp" | "acp-fallback"} SessionReadSource + */ + +/** + * Lazily resolve the acp-client module and the active transport. Dynamic + * on both counts: `lib/acp-client.js` pulls `acp.mjs` and the settings + * chain, `lib/config.js` reads env — neither may sit on the boot path. + */ +async function readDeps() { + const [acp, config] = await Promise.all([ + import("../lib/acp-client.js"), + import("../lib/config.js"), + ]); + return { acp, transport: config.MCODE_WEBUI_TRANSPORT }; +} + +/** + * Did the catalogue host answer on this transport? Returns `false` when + * the transport never wanted the catalogue, and also when it wanted it but + * the host failed to boot (the `acp-fallback` case). + */ +async function catalogueAnswered(acp, transport) { + if (transport !== "runtime") return false; + const host = await acp.getCatalogueHost(); + return host !== null && host !== undefined; +} + +/** + * Every session the engine knows, across all workspaces — the #72 + * (`/api/protocol/list-sessions`) read. The caller applies its own cwd + * filter, exactly as the endpoint did before, so the filtering rule and + * the response shape stay in one place. + * + * @param {object} [options] + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/protocol/list-sessions`. + * @returns {Promise<{sessions: Array, source: SessionReadSource, gate: object, transport: string}>} + */ +export async function readEngineSessionList(options = {}) { + const endpoint = options.endpoint || "GET /api/protocol/list-sessions"; + const { acp, transport } = await readDeps(); + const gate = assertSessionReadCapability(endpoint, transport); + const answered = await catalogueAnswered(acp, transport); + return { + sessions: await acp.listAllMcodeSessions(), + source: transport !== "runtime" ? "acp" : answered ? "catalogue" : "acp-fallback", + gate, + transport, + }; +} + +/** + * The workspace-filtered session list — the #9 (`/api/acp-sessions`) and + * #74 (`/api/state` `mcodeSessions` mirror) read. Same 30s cache and same + * path normalisation as before, because the call goes to the same + * `getMcodeSessionsForWorkspace`. + * + * @param {object} options + * @param {string} [options.cwd] Workspace to filter by; empty means + * "no filter" and the caller decides. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/acp-sessions`. + * @returns {Promise<{sessions: Array, source: SessionReadSource, gate: object, transport: string}>} + */ +export async function readEngineSessionListForWorkspace(options = {}) { + const endpoint = options.endpoint || "GET /api/acp-sessions"; + const { acp, transport } = await readDeps(); + const gate = assertSessionReadCapability(endpoint, transport); + const answered = await catalogueAnswered(acp, transport); + const sessions = await acp.getMcodeSessionsForWorkspace(options.cwd || ""); + return { + sessions, + source: transport !== "runtime" ? "acp" : answered ? "catalogue" : "acp-fallback", + gate, + transport, + }; +} + +/** + * One session's title — the #10 (`/api/acp-session-title`) read. `null` + * for "no such session" and `null` for "engine has no title", which is the + * contract the endpoint has always had; the bridge does not merge them. + * + * @param {object} options + * @param {string} options.sessionId + * @returns {Promise<{sessionId: string, title: string|null, source: SessionReadSource, gate: object, transport: string}>} + */ +export async function readEngineSessionTitle(options = {}) { + const { acp, transport } = await readDeps(); + const gate = assertSessionReadCapability("GET /api/acp-session-title", transport); + const answered = await catalogueAnswered(acp, transport); + const sessionId = options.sessionId || ""; + return { + sessionId, + title: sessionId ? await acp.getMcodeSessionTitle(sessionId) : null, + source: transport !== "runtime" ? "acp" : answered ? "catalogue" : "acp-fallback", + gate, + transport, + }; +} + +/** + * The engine's installed version — the #75 (`/api/health`) read. + * + * The one honest answer available today comes from the ACP `initialize` + * reply's `agentInfo.version`; the in-process catalogue host exposes no + * version accessor (its surface is `adapter` / `cliService` / `apiHost` / + * `controller` / `application` / `applications`, see + * `providers/local-runtime-v2.js`), so "read it from the v2 host" as the + * batch plan imagined is not implementable without inventing a method. + * Rather than fabricate one, the bridge names the source it used and + * keeps the endpoint's `"unknown"` fallback for "nothing has attached + * yet". The shape of `/api/health` is untouched. + * + * @returns {Promise<{version: string, source: SessionReadSource, transport: string, gate: object}>} + */ +export async function readEngineVersion() { + const { acp, transport } = await readDeps(); + const gate = assertSessionReadCapability("GET /api/health", transport); + const info = acp.getMcodeServerInfo(); + return { + version: (info && info.version) || "unknown", + // The version is a protocol fact, not a catalogue fact: it is answered + // from the ACP `initialize` mirror under every transport, including + // `runtime`, where the mirror is simply empty until something attaches. + source: "acp", + transport, + gate, + }; +} diff --git a/packages/webui/server/routes/health.js b/packages/webui/server/routes/health.js index a782b837..9da6f65e 100644 --- a/packages/webui/server/routes/health.js +++ b/packages/webui/server/routes/health.js @@ -8,22 +8,16 @@ import { DEFAULT_MODEL, DEFAULT_WORKSPACE, } from "../lib/config.js"; -import { getMcodeServerInfo } from "../lib/acp-client.js"; +// M3-B1 (engine facade): `mcodeVersion` is read through the facade so +// the endpoint records WHICH source answered. See +// `readEngineVersion` for why the answer is still the ACP `initialize` +// mirror — the in-process catalogue host exposes no version accessor, and +// inventing one is exactly the "claim a capability that does not exist" +// this batch exists to prevent. +import { readEngineVersion } from "../engine/session-reads.js"; -/** - * The engine's own version, from the `agentInfo` in its ACP `initialize` reply. - * - * This used to be a pinned constant, which meant the endpoint reported whatever - * version webui was written against rather than the one installed. Before a - * client attaches there is no version to report, hence `unknown` — the same - * value `/api/protocol/capabilities` uses for the same fact. - */ -function engineVersion() { - const info = getMcodeServerInfo(); - return (info && info.version) || "unknown"; -} - -export function handleHealth(_req, res) { +export async function handleHealth(_req, res) { + const { version } = await readEngineVersion(); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); return res.end( JSON.stringify({ @@ -32,7 +26,7 @@ export function handleHealth(_req, res) { defaultModel: DEFAULT_MODEL, defaultWorkspace: DEFAULT_WORKSPACE, mcodeCmd: MCODE_CMD, - mcodeVersion: engineVersion(), + mcodeVersion: version, maxConcurrent: MAX_CONCURRENT, }), ); diff --git a/packages/webui/server/routes/protocol.js b/packages/webui/server/routes/protocol.js index 4d0de38c..763a1036 100644 --- a/packages/webui/server/routes/protocol.js +++ b/packages/webui/server/routes/protocol.js @@ -18,9 +18,14 @@ import { cancelSession, loadSession, activateSession, - listSessions, mcodePermissionToWebui, } from "../lib/mcode-rpc.js"; +// M3-B1 (engine facade): only #72 (`list-sessions`) is gated in this +// batch. The other five handlers here still call mcode-rpc directly — +// they belong to B4 (#73 capabilities) and B7/B9 (cancel, load, activate, +// set-mode, set-config-option), each of which lands its own facade call +// with its own regression evidence. +import { readEngineSessionList } from "../engine/session-reads.js"; import { loadSessions, saveSessions, resetContext } from "../lib/sessions.js"; import { pushStateFor } from "../lib/state-bus.js"; import { readJson } from "../lib/read-json.js"; @@ -227,11 +232,21 @@ export async function handleActivateSession(req, res, ctx) { // ============================================================ // GET /api/protocol/list-sessions?cwd=... // 列 mcode session, 供前端 "远控 TUI" UI 用 +// +// M3-B1: the list now comes from the engine facade +// (`server/engine/session-reads.js`) instead of `mcode-rpc.js#listSessions` +// directly, so this endpoint is gated on the same declared +// `sessionCrud.listSessions` as the sidebar's #9 and #72 share. The +// facade forwards to the same `listAllMcodeSessions()` the rpc wrapper +// called, which means the runtime path already went through +// `lib/catalogue-sessions.js`; the cwd filter below and the response +// shape are untouched — `mcode-rpc.js#listSessions` is still exported +// for the write-family callers that arrive with later batches. // ============================================================ export async function handleListSessions(req, res, ctx) { const url = new URL(req.url, "http://localhost"); const cwd = url.searchParams.get("cwd") || ctx?.cs?.workspace?.dir || ""; - const all = await listSessions(); + const { sessions: all } = await readEngineSessionList(); if (!cwd) return respond(res, 200, { ok: true, sessions: all }); // 按 cwd 过滤 (norm 路径对齐) const norm = (p) => diff --git a/packages/webui/server/routes/sessions.js b/packages/webui/server/routes/sessions.js index 8638ffb4..3f40937e 100644 --- a/packages/webui/server/routes/sessions.js +++ b/packages/webui/server/routes/sessions.js @@ -16,7 +16,6 @@ import { import { deleteMcodeSessionFromDb } from "../lib/mcode-session-delete.js"; import { getMcodeSessionTitle, - getMcodeSessionsForWorkspace, getMcodeSessionsCacheSync, getMcodeSessionsStaleSync, shutdownMcodeAcpSingleton, @@ -35,6 +34,15 @@ import { } from "../lib/state-bus.js"; import { MCODE_RUNTIME_DB, DEFAULT_WORKSPACE } from "../lib/config.js"; import { getSessionTree, invalidateSessionTree } from "../lib/session-tree.js"; +// M3-B1 (engine facade): #9 and #10 read the engine through the declared +// capability rather than straight off the ACP client. Both facade +// functions forward to the same acp-client exports this module already +// imported, so the wire shape, the cache and the transport switch are +// unchanged — only the gate in front of them is new. +import { + readEngineSessionListForWorkspace, + readEngineSessionTitle, +} from "../engine/session-reads.js"; import { authorize } from "../lib/authorize.js"; import { pushAlert } from "../lib/alerts.js"; import { append as _eventsAppend } from "../lib/events.js"; @@ -1036,17 +1044,32 @@ export function handleSessionTree(req, res, _ctx) { } // GET /api/acp-sessions?cwd=... — mcode acp session/list +// +// M3-B1: the read goes through the engine facade +// (engine/session-reads.js) so the sidebar's data source is a DECLARED +// capability rather than "whatever the transport happens to be". A +// provider that does not declare `sessionCrud.listSessions` answers 501 +// through app.js#invokeHandler instead of an empty list. The response +// shape is byte-for-byte what it was: the facade forwards to the same +// `getMcodeSessionsForWorkspace` (same 30s cache, same cwd +// normalisation, same `catalogue-sessions.js` projection on the runtime +// path). export async function handleAcpSessions(req, res, ctx) { const cs = ctx.cs; const url = new URL(req.url, "http://localhost"); const cwd = url.searchParams.get("cwd") || (cs.workspace && cs.workspace.dir) || ""; - const sessions = await getMcodeSessionsForWorkspace(cwd); + const { sessions } = await readEngineSessionListForWorkspace({ cwd }); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); return res.end(JSON.stringify({ ok: true, cwd, sessions })); } // GET /api/acp-session-title?sessionId=... +// +// M3-B1: gated on `sessionCrud.getSession` — the engine method the ACP +// `session/list` title lookup corresponds to. `title` stays `null` for +// both "no such session" and "engine has no title": the endpoint has +// always collapsed those two and callers depend on it. export async function handleAcpSessionTitle(req, res, _ctx) { const url = new URL(req.url, "http://localhost"); const sid = url.searchParams.get("sessionId") || ""; @@ -1054,7 +1077,7 @@ export async function handleAcpSessionTitle(req, res, _ctx) { res.writeHead(400, { "Content-Type": "application/json" }); return res.end(JSON.stringify({ ok: false, error: "sessionId required" })); } - const title = await getMcodeSessionTitle(sid); + const { title } = await readEngineSessionTitle({ sessionId: sid }); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); return res.end( JSON.stringify({ ok: true, sessionId: sid, title: title || null }), diff --git a/packages/webui/server/routes/state.js b/packages/webui/server/routes/state.js index aa6cc17f..796625c3 100644 --- a/packages/webui/server/routes/state.js +++ b/packages/webui/server/routes/state.js @@ -14,10 +14,13 @@ import { sessionsListForSnapshot, nextRevisionFor, } from "../lib/state-bus.js"; -import { - getMcodeSessionsForWorkspace, - getCachedMcodeCommands, -} from "../lib/acp-client.js"; +import { getCachedMcodeCommands } from "../lib/acp-client.js"; +// M3-B1 (engine facade): the declared-capability gate in front of the +// mcodeSessions mirror. The SSE first frame below keeps calling +// `mcodeSessionsSnapshotFields` directly — the SSE channel is a P2 +// migration, out of scope for this batch, and it must keep its exact +// pending/stale semantics. +import { readEngineSessionListForWorkspace } from "../engine/session-reads.js"; import { getLanBroadcast } from "../lib/settings.js"; import { applyMavisUsageToCs } from "../lib/mavis-usage.js"; import { getMcodeModelLimit } from "../lib/models.js"; @@ -110,9 +113,20 @@ const SSE_HEADERS = { export async function handleState(req, res, ctx) { const cs = getClient(ctx.cid); - const mcodeSessions = await getMcodeSessionsForWorkspace( - cs.workspace && cs.workspace.dir, - ); + // M3-B1: the mcodeSessions mirror now comes from the engine facade, + // which gates it on the declared `sessionCrud.listSessions` and reports + // (in the return value, not on the wire) whether the in-process host or + // the ACP mirror answered. The VALUE is the same array the endpoint + // built before — `readEngineSessionListForWorkspace` forwards to the + // same `getMcodeSessionsForWorkspace`, cache and cwd normalisation + // included. The snapshot body below is unchanged field for field: + // `snapshotViewFields` / `mcodeSessionsSnapshotFields` are the + // frontend's first-frame contract and this batch adds and removes + // nothing. + const { sessions: mcodeSessions } = await readEngineSessionListForWorkspace({ + cwd: (cs.workspace && cs.workspace.dir) || "", + endpoint: "GET /api/state", + }); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); // v0.5.bx-29: /api/state 也尝试 hydrate mavis db 真值 (best-effort) // SSE 客户端 (EventSource) 也会调这个端点, 所以 hydrate 也能发生在 reconnect 时 diff --git a/packages/webui/test/helpers/_setup.js b/packages/webui/test/helpers/_setup.js index 9ee72588..d30db95f 100644 --- a/packages/webui/test/helpers/_setup.js +++ b/packages/webui/test/helpers/_setup.js @@ -51,6 +51,12 @@ const _acpMock = { getMcodeAcpClient: async () => null, listAllMcodeSessions: async () => [], getMcodeServerInfo: () => null, + // M3-B1: the engine facade (server/engine/session-reads.js) asks whether + // the in-process catalogue host answered before it reports where a read's + // bytes came from. Default null = "the host never booted", i.e. the + // acp-fallback case. Tests that want the catalogue case register + // `getCatalogueHost: async () => ({ adapter: {} })`. + getCatalogueHost: async () => null, invalidateMcodeSessionsCache: () => {}, shutdownMcodeAcpSingleton: () => {}, dropMcodeSessionFromCache: () => {}, // v1.0: 删除路由防复活用 @@ -217,6 +223,9 @@ export async function setupMocks(t, overrides = {}) { getMcodeAcpClient: (...a) => _acpMock.getMcodeAcpClient(...a), listAllMcodeSessions: (...a) => _acpMock.listAllMcodeSessions(...a), getMcodeServerInfo: (...a) => _acpMock.getMcodeServerInfo(...a), + // M3-B1: engine/session-reads.js asks this to report whether a read + // came from the in-process catalogue host or from the ACP mirror. + getCatalogueHost: (...a) => _acpMock.getCatalogueHost(...a), invalidateMcodeSessionsCache: (...a) => _acpMock.invalidateMcodeSessionsCache(...a), shutdownMcodeAcpSingleton: (...a) => diff --git a/packages/webui/test/lib/engine/session-reads.test.js b/packages/webui/test/lib/engine/session-reads.test.js new file mode 100644 index 00000000..b8ca6b9a --- /dev/null +++ b/packages/webui/test/lib/engine/session-reads.test.js @@ -0,0 +1,486 @@ +// webui/test/lib/engine/session-reads.test.js +// +// M3-B1: the directory-read family's engine facade. +// +// Two things are pinned here, and they are the two ways this batch could +// have gone wrong: +// +// 1. The WIRE SHAPE of the five endpoints (#9, #10, #72, #74, #75) is +// a frontend contract. The sidebar tree and the /api/state first +// frame both render from it, so a field added "just in case", a +// `null` quietly turned into `[]`, or a reordered object all look +// harmless in a diff and all break a render. The shape tables below +// are the regression net for that, table-driven per the repo's +// convention so a new case is one row, not one test. +// +// 2. The GATE is real, not decorative. A provider that declares +// `sessionCrud: none` must produce EngineCapabilityNotSupportedError +// → 501 through app.js#invokeHandler, never an empty list. That is +// the whole point of routing reads through a declaration instead of +// through whatever transport happens to be configured, and the +// registered provider declares `full` today, so only this file can +// prove the gate would bite. +// +// Test style follows test/lib/engine/capabilities.test.js (batch B1). +// Where a route handler is exercised it goes through the same +// setupMocks/registerAcpMock infrastructure the other route suites use. + +import { test, describe, before, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { + ENGINE_CAPABILITY_KEYS, + LOCAL_RUNTIME_V2_CAPABILITIES, +} from "../../../server/engine/index.js"; +import { + EngineCapabilityNotSupportedError, + isEngineCapabilityNotSupportedError, + engineCapabilityHttpResponse, +} from "../../../server/engine/errors.js"; +import { + SESSION_READ_ENDPOINTS, + assertSessionReadCapability, + readEngineSessionList, + readEngineSessionListForWorkspace, + readEngineSessionTitle, + readEngineVersion, + resolveSessionReadProvider, +} from "../../../server/engine/session-reads.js"; +import { + assertEngineCapability, + validateEngineCapabilities, +} from "../../../server/engine/capabilities.js"; + +// --------------------------------------------------------------------------- +// The endpoint → capability declaration table +// --------------------------------------------------------------------------- + +describe("SESSION_READ_ENDPOINTS — the batch's declaration table", () => { + // Table-driven: [endpoint, capability, subItem]. Editing a row here is a + // capability decision and must be reviewed as one. + const TABLE = [ + ["GET /api/acp-sessions", "sessionCrud", "listSessions"], + ["GET /api/acp-session-title", "sessionCrud", "getSession"], + ["GET /api/protocol/list-sessions", "sessionCrud", "listSessions"], + ["GET /api/state", "sessionCrud", "listSessions"], + ]; + + test("covers exactly the five endpoints of batch B1", () => { + assert.deepEqual(Object.keys(SESSION_READ_ENDPOINTS).sort(), [ + "GET /api/acp-session-title", + "GET /api/acp-sessions", + "GET /api/health", + "GET /api/protocol/list-sessions", + "GET /api/state", + ]); + }); + + for (const [endpoint, capability, subItem] of TABLE) { + test(`${endpoint} needs ${capability}.${subItem}`, () => { + assert.deepEqual(SESSION_READ_ENDPOINTS[endpoint], { capability, subItem }); + // Every named capability must be one of the 14 matrix keys — the + // table must not grow a private key (that would put an unreviewed + // capability in the registry, which validateEngineCapabilities + // exists to prevent). + assert.ok(ENGINE_CAPABILITY_KEYS.includes(capability)); + }); + } + + test("/api/health declares no capability — the 14 keys have no honest match", () => { + // Reading the engine's *installed* version is not `updateCheck` + // (checking for a NEW version). Declaring one anyway would be the + // "claim a capability that does not exist" failure this batch exists + // to prevent, so the table says `null` and the facade reports + // gate: "no-capability-key" instead. + assert.equal(SESSION_READ_ENDPOINTS["GET /api/health"], null); + }); + + test("the gate is a no-op for a null capability rather than a throw", () => { + const gate = assertSessionReadCapability("GET /api/health", "runtime"); + assert.equal(gate.gate, "no-capability-key"); + assert.equal(gate.capability, null); + assert.equal(gate.subItem, null); + }); + + test("an endpoint outside this family throws a plain Error, not 501 material", () => { + // Caller confusion must never be dressed up as an engine limitation: + // app.js answers 404-ish for a plain Error and 501 for + // EngineCapabilityNotSupportedError. + assert.throws( + () => assertSessionReadCapability("GET /api/fs/read", "runtime"), + (err) => + !(err instanceof EngineCapabilityNotSupportedError) && + err.code === "unknown_session_read_endpoint", + ); + }); +}); + +// --------------------------------------------------------------------------- +// Provider resolution +// --------------------------------------------------------------------------- + +describe("resolveSessionReadProvider", () => { + test("runtime maps to the registered local-runtime-v2 provider", () => { + const provider = resolveSessionReadProvider("runtime"); + assert.equal(provider.id, "local-runtime-v2"); + assert.equal(provider.capabilities, LOCAL_RUNTIME_V2_CAPABILITIES); + }); + + // M4 registers the acp / exec providers. Until then there is no + // declaration to check under those transports, and the gate says so + // instead of borrowing the v2 provider's declaration (which would be + // answering for a provider that is not on the wire). + for (const transport of ["acp", "exec"]) { + test(`${transport} has no registered provider yet → null`, () => { + assert.equal(resolveSessionReadProvider(transport), null); + const gate = assertSessionReadCapability("GET /api/acp-sessions", transport); + assert.equal(gate.gate, "unregistered-transport"); + assert.equal(gate.provider, null); + // Still names what WOULD be needed, so the passthrough is auditable. + assert.equal(gate.capability, "sessionCrud"); + assert.equal(gate.subItem, "listSessions"); + }); + } +}); + +// --------------------------------------------------------------------------- +// The gate actually bites +// --------------------------------------------------------------------------- + +describe("assertSessionReadCapability — a limited provider answers 501", () => { + // A hypothetical future provider (M4's ACP provider is the real case): + // it has the session read surface but no delete/rename. The read family + // must keep working — that is the point of naming the sub-item rather + // than the capability alone. + const PARTIAL_NO_LIST = { + ...Object.fromEntries(ENGINE_CAPABILITY_KEYS.map((k) => [k, { level: "full" }])), + sessionCrud: { + level: "partial", + missing: ["listSessions", "getSession", "deleteSession", "renameSession"], + reason: "test fixture: provider exposes no session read surface", + }, + }; + // And a provider with no session support at all. + const NONE = { + ...Object.fromEntries(ENGINE_CAPABILITY_KEYS.map((k) => [k, { level: "full" }])), + sessionCrud: { level: "none", reason: "test fixture: interface-absent" }, + }; + + // Table-driven over the four gated endpoints: every one of them must + // refuse under a provider that cannot list, and name the method it + // needed. A row that silently passes is a route that would answer `[]`. + const GATED = [ + ["GET /api/acp-sessions", "listSessions"], + ["GET /api/acp-session-title", "getSession"], + ["GET /api/protocol/list-sessions", "listSessions"], + ["GET /api/state", "listSessions"], + ]; + + for (const [endpoint, subItem] of GATED) { + test(`${endpoint} throws EngineCapabilityNotSupportedError naming ${subItem}`, () => { + const need = SESSION_READ_ENDPOINTS[endpoint]; + assert.throws( + () => assertEngineCapability(PARTIAL_NO_LIST, need.capability, "fixture-provider", need.subItem), + (err) => { + assert.ok(isEngineCapabilityNotSupportedError(err)); + assert.equal(err.capability, "sessionCrud"); + assert.equal(err.provider, "fixture-provider"); + assert.deepEqual(err.missing, [subItem]); + // The HTTP mapping is the frontend's contract for degradation. + const { status, payload } = engineCapabilityHttpResponse(err); + assert.equal(status, 501); + assert.equal(payload.code, "engine_capability_not_supported"); + assert.equal(payload.capability, "sessionCrud"); + assert.deepEqual(payload.missing, [subItem]); + return true; + }, + ); + }); + + test(`${endpoint} throws for a provider that declares sessionCrud: none`, () => { + const need = SESSION_READ_ENDPOINTS[endpoint]; + assert.throws( + () => assertEngineCapability(NONE, need.capability, "fixture-provider"), + isEngineCapabilityNotSupportedError, + ); + }); + } + + test("a partial declaration that keeps listSessions lets the read family through", () => { + const PARTIAL_WITH_READ = { + ...PARTIAL_NO_LIST, + sessionCrud: { + level: "partial", + missing: ["deleteSession", "renameSession"], + reason: "test fixture: read surface present, write surface absent", + }, + }; + for (const [endpoint] of GATED) { + const need = SESSION_READ_ENDPOINTS[endpoint]; + assert.doesNotThrow(() => + assertEngineCapability(PARTIAL_WITH_READ, need.capability, "fixture-provider", need.subItem), + ); + } + }); + + test("the fixtures themselves are valid declarations (the gate is the only difference)", () => { + assert.deepEqual(validateEngineCapabilities(PARTIAL_NO_LIST), []); + assert.deepEqual(validateEngineCapabilities(NONE), []); + }); + + test("the registered provider passes every row of the table", () => { + for (const [endpoint, capability, subItem] of [ + ...GATED.map(([e]) => [e, SESSION_READ_ENDPOINTS[e].capability, SESSION_READ_ENDPOINTS[e].subItem]), + ]) { + assert.doesNotThrow(() => + assertEngineCapability(LOCAL_RUNTIME_V2_CAPABILITIES, capability, "local-runtime-v2", subItem), + `${endpoint} must pass against the registered provider today`, + ); + } + }); +}); + +// --------------------------------------------------------------------------- +// The reads, with the acp-client mocked the way routes are tested +// --------------------------------------------------------------------------- + +// `t.mock` only exists on the context a TOP-LEVEL hook receives, so the +// registration lives at file scope like every other suite in the repo +// (see test/routes/sessions.check.mjs). The facade reaches the acp-client +// through `await import()` inside its read functions, so a registration +// made here still lands before the first read. +let acpMock; +before(async (t) => { + const { setupMocks, acpMock: handle } = await import("../../helpers/_setup.js"); + await setupMocks(t, {}); + acpMock = handle; +}); + +beforeEach(() => { + acpMock.listAllMcodeSessions = async () => []; + acpMock.getMcodeSessionsForWorkspace = async () => []; + acpMock.getMcodeSessionTitle = async () => null; + acpMock.getMcodeServerInfo = () => null; + acpMock.getCatalogueHost = async () => null; +}); + +describe("session-reads — the reads themselves", () => { + // ------------------------------------------------------------------------- + // #9 / #72 — the sidebar list shape, field by field + // ------------------------------------------------------------------------- + + // One ACP-wire session entry. `title` and `updatedAt` are OPTIONAL on the + // wire: `catalogue-sessions.js#projectTuiSessionToAcp` omits `title` when + // empty and `updatedAt` when unparseable, exactly like the ACP adapter's + // `toAcpSessionInfo`. The facade forwards that projection untouched, so a + // normalizer that started defaulting either to `null` / `""` would change + // what the sidebar renders for unnamed sessions. + const WIRE_SESSION = { + sessionId: "mvs_aaaa1111222233334444555566667777", + cwd: "/ws/a", + title: "Engine generated title", + updatedAt: "2026-10-03T00:00:00.000Z", + }; + + describe("readEngineSessionList (#72 — all workspaces)", () => { + // Table-driven: [name, engineAnswer, expectedSource, expectedReason]. + // `transport` here is whatever the process was started with — the + // suite is run under both by the batch's gate, so the assertion is on + // the RULE, not on one transport's value. + const CASES = [ + ["one session, all four wire fields", [WIRE_SESSION], null], + ["no sessions at all → [] (never null)", [], null], + ]; + + for (const [name, engineAnswer] of CASES) { + test(name, async () => { + acpMock.listAllMcodeSessions = async () => engineAnswer; + const result = await readEngineSessionList(); + assert.deepEqual(result.sessions, engineAnswer); + assert.ok(Array.isArray(result.sessions), "sessions is always an array"); + assert.equal(result.transport, process.env.MCODE_WEBUI_TRANSPORT || "acp"); + // `source` is metadata, not wire: the route ignores it, the tests + // and the log read it. + assert.ok( + ["catalogue", "acp", "acp-fallback"].includes(result.source), + `unexpected source ${result.source}`, + ); + assert.equal(result.gate.endpoint, "GET /api/protocol/list-sessions"); + }); + } + + test("field set of an entry is exactly the ACP wire projection — no more, no less", async () => { + acpMock.listAllMcodeSessions = async () => [WIRE_SESSION]; + const { sessions } = await readEngineSessionList(); + assert.deepEqual(Object.keys(sessions[0]), ["sessionId", "cwd", "title", "updatedAt"]); + }); + + // null vs [] is the distinction the sidebar actually depends on: an + // absent list must render "no sessions", a null must not crash the + // render that maps over it. + test("an unnamed session keeps `title` ABSENT, not null and not \"\"", async () => { + const untitled = { sessionId: "mvs_bbbb", cwd: "/ws/b" }; + acpMock.listAllMcodeSessions = async () => [untitled]; + const { sessions } = await readEngineSessionList(); + assert.equal("title" in sessions[0], false, "catalogue-sessions.js omits an empty title"); + assert.equal("updatedAt" in sessions[0], false, "…and an unparseable updatedAt"); + assert.deepEqual(Object.keys(sessions[0]), ["sessionId", "cwd"]); + }); + + test("the facade does not filter by cwd — #72's cwd filter is the route's", async () => { + acpMock.listAllMcodeSessions = async () => [WIRE_SESSION]; + const { sessions } = await readEngineSessionList(); + assert.equal(sessions.length, 1, "an unfiltered read returns every workspace's sessions"); + }); + }); + + describe("readEngineSessionListForWorkspace (#9 / #74 — cwd filtered)", () => { + const CASES = [ + ["cwd given, engine answers one session", "/ws/a", [WIRE_SESSION]], + ["cwd empty → no filter, engine answers as-is", "", [WIRE_SESSION]], + ["no sessions → [] (never null)", "/ws/a", []], + ]; + + for (const [name, cwd, engineAnswer] of CASES) { + test(name, async () => { + acpMock.getMcodeSessionsForWorkspace = async () => engineAnswer; + const result = await readEngineSessionListForWorkspace({ cwd }); + assert.deepEqual(result.sessions, engineAnswer); + assert.equal(result.gate.endpoint, "GET /api/acp-sessions"); + }); + } + + test("the endpoint key is honoured, so /api/state is gated under its own name", async () => { + const result = await readEngineSessionListForWorkspace({ + cwd: "/ws/a", + endpoint: "GET /api/state", + }); + assert.equal(result.gate.endpoint, "GET /api/state"); + }); + + test("an undefined cwd is normalised to \"\" before it reaches the client", async () => { + const seen = []; + acpMock.getMcodeSessionsForWorkspace = async (ws) => { + seen.push(ws); + return []; + }; + await readEngineSessionListForWorkspace({}); + assert.deepEqual(seen, [""], "the facade never forwards undefined"); + }); + }); + + // ------------------------------------------------------------------------- + // #10 — the title + // ------------------------------------------------------------------------- + + describe("readEngineSessionTitle (#10)", () => { + // Table-driven on the VALUE. The bridge is a pass-through: it reports + // exactly what the engine answered and does not decide what "no title" + // means. Collapsing `""` / undefined to `null` is `handleAcpSessionTitle`'s + // `title || null`, pinned in test/routes/session-reads.check.mjs — if the + // bridge started normalising too, one of the two layers would own a rule + // the other also owns, and an "improvement" to one would silently change + // the wire. + const CASES = [ + ["a titled session answers the title verbatim", "mvs_1", "My Title", "My Title"], + ["an untitled session passes null through", "mvs_2", null, null], + ["an empty title passes \"\" through (the route collapses it)", "mvs_3", "", ""], + ["an undefined title passes through undefined", "mvs_4", undefined, undefined], + ]; + + for (const [name, sessionId, engineAnswer, expected] of CASES) { + test(name, async () => { + acpMock.getMcodeSessionTitle = async () => engineAnswer; + const result = await readEngineSessionTitle({ sessionId }); + assert.equal(result.sessionId, sessionId); + assert.equal(result.title, expected); + assert.equal(result.gate.endpoint, "GET /api/acp-session-title"); + }); + } + + test("a missing sessionId never reaches the client", async () => { + let called = 0; + acpMock.getMcodeSessionTitle = async () => { + called++; + return "should not happen"; + }; + const result = await readEngineSessionTitle({ sessionId: "" }); + assert.equal(result.title, null); + assert.equal(called, 0, "an empty id is answered locally, not by a lookup"); + }); + }); + + // ------------------------------------------------------------------------- + // #75 — the version + // ------------------------------------------------------------------------- + + describe("readEngineVersion (#75)", () => { + // Table-driven: [name, agentInfo, expectedVersion]. `unknown` is the + // documented sentinel for "nothing has attached yet" and must stay a + // string — /api/protocol/capabilities uses the same value for the same + // fact, and a monitor semver-parsing the field would throw on null. + const CASES = [ + ["an attached client answers its version", { name: "mcode", version: "0.5.7" }, "0.5.7"], + ["a client without a version answers unknown", { name: "mcode" }, "unknown"], + ["no client at all answers unknown", null, "unknown"], + ]; + + for (const [name, info, expected] of CASES) { + test(name, async () => { + acpMock.getMcodeServerInfo = () => info; + const result = await readEngineVersion(); + assert.equal(result.version, expected); + assert.equal(typeof result.version, "string"); + // The version is a protocol fact. The catalogue host exposes no + // version accessor, so the facade reports the mirror it used + // rather than claiming the engine answered. + assert.equal(result.source, "acp"); + assert.equal(result.gate.gate, "no-capability-key"); + }); + } + }); + + // ------------------------------------------------------------------------- + // host === null — the degradation this batch must not hide + // ------------------------------------------------------------------------- + + describe("catalogue host returned null", () => { + test("a null host is reported as a fallback, never as the engine answering", async () => { + acpMock.getCatalogueHost = async () => null; + acpMock.listAllMcodeSessions = async () => [WIRE_SESSION]; + const result = await readEngineSessionList(); + if (result.transport === "runtime") { + // The transport ASKED for the host and did not get one. The read + // still succeeds from the ACP mirror (the sidebar must not break), + // but the source says so — silently claiming "catalogue" here is + // the fake-success shape. + assert.equal(result.source, "acp-fallback"); + } else { + assert.equal(result.source, "acp"); + } + // Either way the sessions still come back: a read family answers. + assert.deepEqual(result.sessions, [WIRE_SESSION]); + }); + + test("a live host is reported as catalogue under the runtime transport", async () => { + acpMock.getCatalogueHost = async () => ({ adapter: {} }); + acpMock.getMcodeSessionsForWorkspace = async () => [WIRE_SESSION]; + const result = await readEngineSessionListForWorkspace({ cwd: "/ws/a" }); + assert.equal( + result.source, + result.transport === "runtime" ? "catalogue" : "acp", + "the source must follow the transport, not a fixed string", + ); + }); + + test("a host that boots but whose list throws still surfaces the throw", async () => { + acpMock.getCatalogueHost = async () => ({ adapter: {} }); + acpMock.listAllMcodeSessions = async () => { + throw new Error("sqlite locked"); + }; + // The facade does not swallow a broken engine into an empty list — + // that is the #110 fake-success shape. acp-client.js owns the + // ACP failover; the facade adds no second, quieter one. + await assert.rejects(() => readEngineSessionList(), /sqlite locked/); + }); + }); +}); diff --git a/packages/webui/test/routes/health.check.mjs b/packages/webui/test/routes/health.check.mjs index 3dae0dd6..f54a8a65 100644 --- a/packages/webui/test/routes/health.check.mjs +++ b/packages/webui/test/routes/health.check.mjs @@ -5,8 +5,15 @@ // and the agent-browser probe hits. If it returns wrong shape, monitoring // breaks and we don't notice the server is broken. // +// M3-B1: handleHealth became `async` when `mcodeVersion` moved behind the +// engine facade (server/engine/session-reads.js#readEngineVersion, which +// resolves its acp-client dependency with a dynamic import to stay off the +// boot path). The response shape is unchanged; the tests below pin the +// field list so the await-vs-sync change cannot smuggle a field edit in. +// // Test strategy: NO setupMocks. handleHealth is a pure function over config -// constants. No webui deps, no fs. +// constants plus the ACP `initialize` mirror, which is null with no client +// attached. No webui deps, no fs. import { test, describe } from "node:test"; import assert from "node:assert/strict"; @@ -33,19 +40,22 @@ function fakeRes() { return res; } +async function callHealth() { + const res = fakeRes(); + await health.handleHealth(null, res); + return res; +} + describe("handleHealth — /api/health", () => { - test("returns 200 + ok:true", () => { - const res = fakeRes(); - health.handleHealth(null, res); + test("returns 200 + ok:true", async () => { + const res = await callHealth(); assert.equal(res._status, 200); const body = JSON.parse(res._body); assert.equal(body.ok, true); }); - test("response includes all expected fields", () => { - const res = fakeRes(); - health.handleHealth(null, res); - const body = JSON.parse(res._body); + test("response includes all expected fields", async () => { + const body = JSON.parse((await callHealth())._body); // Check all documented fields exist with correct types assert.equal(typeof body.port, "number"); assert.equal(typeof body.defaultModel, "string"); @@ -55,24 +65,43 @@ describe("handleHealth — /api/health", () => { assert.equal(typeof body.maxConcurrent, "number"); }); - test("Content-Type is application/json", () => { - const res = fakeRes(); - health.handleHealth(null, res); + // M3-B1 shape snapshot: the field list IS the contract for every monitor + // and probe. This catches both a removed field and a "just one more" field. + test("field list is exactly the seven documented keys, in order", async () => { + const body = JSON.parse((await callHealth())._body); + assert.deepEqual(Object.keys(body), [ + "ok", + "port", + "defaultModel", + "defaultWorkspace", + "mcodeCmd", + "mcodeVersion", + "maxConcurrent", + ]); + }); + + test("Content-Type is application/json", async () => { + const res = await callHealth(); assert.match(res._headers["Content-Type"], /application\/json/); }); - test("port is a valid port number (1-65535)", () => { - const res = fakeRes(); - health.handleHealth(null, res); - const body = JSON.parse(res._body); + test("port is a valid port number (1-65535)", async () => { + const body = JSON.parse((await callHealth())._body); assert.ok(body.port > 0 && body.port < 65536); }); - test("maxConcurrent is a positive integer", () => { - const res = fakeRes(); - health.handleHealth(null, res); - const body = JSON.parse(res._body); + test("maxConcurrent is a positive integer", async () => { + const body = JSON.parse((await callHealth())._body); assert.ok(Number.isInteger(body.maxConcurrent)); assert.ok(body.maxConcurrent > 0); }); + + // M3-B1: no ACP client is attached in this suite, so the engine facade + // must still answer with the documented `"unknown"` sentinel rather than + // `null`/`undefined` — the field is typed `string` in the contract and a + // monitor doing `semver` parsing on it would throw on null. + test('mcodeVersion is the string "unknown" when no client has attached', async () => { + const body = JSON.parse((await callHealth())._body); + assert.equal(body.mcodeVersion, "unknown"); + }); }); diff --git a/packages/webui/test/routes/protocol.check.mjs b/packages/webui/test/routes/protocol.check.mjs index 3b8c28d9..1ad3cd9c 100644 --- a/packages/webui/test/routes/protocol.check.mjs +++ b/packages/webui/test/routes/protocol.check.mjs @@ -17,7 +17,7 @@ // Test strategy: USE setupMocks to mock mcode-rpc.js. We can control the // returned code per test to verify each branch of the status-code mapping. -import { test, describe, before } from "node:test"; +import { test, describe, before, beforeEach } from "node:test"; import assert from "node:assert/strict"; import { Readable } from "node:stream"; import { setupMocks, absPath } from "../helpers/_setup.js"; @@ -246,30 +246,63 @@ describe("handleActivateSession — /api/protocol/activate-session", () => { }); describe("handleListSessions — /api/protocol/list-sessions", () => { - // Note: the real listSessions returns an array (not {sessions: [...]}). - // We need to override the default mock to return []. - before(async () => { + // The handler reads the engine through the facade + // (server/engine/session-reads.js → acp-client.js#listAllMcodeSessions), + // so that is the seam a test has to drive. It used to reach for + // `mcode-rpc.js#listSessions` and the override below landed on + // `registerAcpMock({ listSessions })` — a key nothing read, which made + // both cases assert against a hard-coded empty list. M3-B1 drives the + // real seam so "the filter works" is actually proven. + // + // Note: the engine answer is an array (not `{sessions: [...]}`). + const WIRE = [ + { sessionId: "mvs_x", cwd: "/ws-X", title: "X", updatedAt: "2026-10-03T00:00:00.000Z" }, + { sessionId: "mvs_y", cwd: "/ws-Other" }, + ]; + + beforeEach(async () => { const { registerAcpMock } = await import("../helpers/_setup.js"); - registerAcpMock({ listSessions: async () => [] }); + registerAcpMock({ listAllMcodeSessions: async () => [...WIRE] }); }); - test("returns 200 + sessions array (no cwd filter)", async () => { - const req = { url: "/api/protocol/list-sessions" }; + test("returns 200 + the unfiltered list when neither ?cwd nor cs.workspace.dir is set", async () => { + // `fakeCs()` carries workspace.dir = "/ws-X", which the handler uses as + // the cwd fallback — so the unfiltered branch needs a cs without one. const res = fakeRes(); - await protoRoute.handleListSessions(req, res, { cs: fakeCs(), cid: "cid-1" }); + await protoRoute.handleListSessions( + { url: "/api/protocol/list-sessions" }, + res, + { cs: { workspace: { dir: null } }, cid: "cid-1" }, + ); assert.equal(res._status, 200); const body = JSON.parse(res._body); assert.equal(body.ok, true); - assert.ok(Array.isArray(body.sessions)); + assert.deepEqual(body.sessions, WIRE); + // No cwd to filter by means the endpoint does not echo a cwd key. + assert.equal("cwd" in body, false); }); - test("returns 200 + filtered sessions when cwd query is provided", async () => { + test("returns 200 + the cwd-filtered list when cwd query is provided", async () => { const req = { url: "/api/protocol/list-sessions?cwd=/ws-X" }; const res = fakeRes(); await protoRoute.handleListSessions(req, res, { cs: fakeCs(), cid: "cid-1" }); assert.equal(res._status, 200); const body = JSON.parse(res._body); assert.equal(body.cwd, "/ws-X"); + assert.deepEqual(body.sessions, [WIRE[0]]); + }); + + test("falls back to cs.workspace.dir when ?cwd is absent", async () => { + // fakeCs() is exactly that case: no ?cwd, workspace.dir = "/ws-X". + const res = fakeRes(); + await protoRoute.handleListSessions( + { url: "/api/protocol/list-sessions" }, + res, + { cs: fakeCs(), cid: "cid-1" }, + ); + const body = JSON.parse(res._body); + assert.equal(body.cwd, "/ws-X"); + assert.deepEqual(body.sessions, [WIRE[0]]); }); }); diff --git a/packages/webui/test/routes/session-reads.check.mjs b/packages/webui/test/routes/session-reads.check.mjs new file mode 100644 index 00000000..db3927da --- /dev/null +++ b/packages/webui/test/routes/session-reads.check.mjs @@ -0,0 +1,464 @@ +// webui/test/routes/session-reads.check.mjs +// +// M3-B1: the five directory-read endpoints, driven end to end. +// +// Why this file exists. Batch B1 moved #9, #10, #72, #74 and #75 behind +// the engine facade (server/engine/session-reads.js). The move is only +// allowed to be invisible, and "invisible" has exactly two failure modes +// worth a test: +// +// - the SIDEBAR (#9, #72) and the /api/state FIRST FRAME (#74) are +// render contracts. A field added here, a `null` turned into `[]`, an +// `updatedAt` that stopped being a string — all invisible in a diff, +// all a broken render. The shape tables below are the net. +// +// - the endpoints must keep answering the SAME status codes and error +// bodies they answered before the gate went in. A 501 that used to be a +// 200 for a session that plainly exists is a regression the facade +// introduced, not a degradation it disclosed. +// +// Style follows the existing route suites (test/routes/sessions.check.mjs, +// test/routes/health.check.mjs): setupMocks + registerAcpMock, handlers +// imported dynamically after the mocks are registered. +// +// The suite is transport-agnostic by construction: it asserts the RULE for +// whichever MCODE_WEBUI_TRANSPORT the run was started with, which is why +// the batch's gate runs it under both `acp` and `runtime`. + +import { test, describe, before, beforeEach } from "node:test"; +import assert from "node:assert/strict"; + +import { + setupMocks, + absPath, + registerAcpMock, + acpMock, +} from "../helpers/_setup.js"; + +let handleAcpSessions, handleAcpSessionTitle, handleListSessions, handleState, handleHealth; +let makeClientState, clients, sbMock; + +function fakeRes() { + return { + _status: null, + _headers: null, + _body: null, + writeHead(s, h) { + this._status = s; + this._headers = h || null; + }, + end(b) { + this._body = b; + }, + }; +} + +async function readJson(res) { + return JSON.parse(res._body); +} + +before(async (t) => { + await setupMocks(t, { mavis: { applyMavisUsageToCs: async () => {} } }); + const sb = await import(absPath("lib/state-bus.js")); + makeClientState = sb.makeClientState; + clients = sb.clients; + const sessions = await import(absPath("routes/sessions.js")); + handleAcpSessions = sessions.handleAcpSessions; + handleAcpSessionTitle = sessions.handleAcpSessionTitle; + const protocol = await import(absPath("routes/protocol.js")); + handleListSessions = protocol.handleListSessions; + const state = await import(absPath("routes/state.js")); + handleState = state.handleState; + const health = await import(absPath("routes/health.js")); + handleHealth = health.handleHealth; + void sbMock; +}); + +// One ACP-wire session entry, exactly the projection +// `lib/catalogue-sessions.js#projectTuiSessionToAcp` emits: `sessionId` and +// `cwd` always, `title` only when non-empty, `updatedAt` only when a finite +// timestamp exists. Both optional keys are load-bearing for the sidebar: an +// entry that suddenly carries `title: null` renders a blank row. +const WIRE_SESSION = { + sessionId: "mvs_aaaa1111222233334444555566667777", + cwd: "/ws/a", + title: "Engine generated title", + updatedAt: "2026-10-03T00:00:00.000Z", +}; +const WIRE_SESSION_BARE = { sessionId: "mvs_bbbb", cwd: "/ws/a" }; + +beforeEach(() => { + clients.clear(); + registerAcpMock({ + listAllMcodeSessions: async () => [], + getMcodeSessionsForWorkspace: async () => [], + getMcodeSessionTitle: async () => null, + getMcodeServerInfo: () => null, + getCatalogueHost: async () => null, + getCachedMcodeCommands: () => [], + getMcodeSessionsCacheSync: () => null, + getMcodeSessionsStaleSync: () => null, + }); + const cs = makeClientState(); + cs.workspace = { dir: "/ws/a", branch: null, tree: null }; + clients.set("cid-1", cs); +}); + +// --------------------------------------------------------------------------- +// #9 GET /api/acp-sessions +// --------------------------------------------------------------------------- + +describe("#9 GET /api/acp-sessions — sidebar list shape", () => { + // Table-driven: [name, engineAnswer, expectedFieldLists]. `expectedFieldLists` + // is one expected key order per returned entry, so a normalizer that + // started emitting an extra key (or dropping `updatedAt`) fails the row + // that produced it, not some unrelated one. + const CASES = [ + ["a fully-populated entry", [WIRE_SESSION], [["sessionId", "cwd", "title", "updatedAt"]]], + ["an entry with the optional keys absent", [WIRE_SESSION_BARE], [["sessionId", "cwd"]]], + ["a mixed list keeps per-entry shapes", [WIRE_SESSION, WIRE_SESSION_BARE], [ + ["sessionId", "cwd", "title", "updatedAt"], + ["sessionId", "cwd"], + ]], + ]; + + for (const [name, engineAnswer, expectedFieldLists] of CASES) { + test(name, async () => { + registerAcpMock({ getMcodeSessionsForWorkspace: async () => engineAnswer }); + const res = fakeRes(); + await handleAcpSessions( + { url: "/api/acp-sessions?cwd=%2Fws%2Fa" }, + res, + { cs: clients.get("cid-1"), cid: "cid-1", pathname: "/api/acp-sessions" }, + ); + assert.equal(res._status, 200); + const body = await readJson(res); + assert.equal(body.cwd, "/ws/a"); + assert.deepEqual(Object.keys(body), ["ok", "cwd", "sessions"]); + assert.deepEqual(body.sessions.map((s) => Object.keys(s)), expectedFieldLists); + }); + } + + test("no sessions answers [], never null", async () => { + registerAcpMock({ getMcodeSessionsForWorkspace: async () => [] }); + const res = fakeRes(); + await handleAcpSessions( + { url: "/api/acp-sessions?cwd=%2Fws%2Fa" }, + res, + { cs: clients.get("cid-1"), cid: "cid-1", pathname: "" }, + ); + const body = await readJson(res); + assert.ok(Array.isArray(body.sessions)); + assert.equal(body.sessions.length, 0); + }); + + test("updatedAt stays an ISO string, never an epoch number", async () => { + registerAcpMock({ getMcodeSessionsForWorkspace: async () => [WIRE_SESSION] }); + const res = fakeRes(); + await handleAcpSessions( + { url: "/api/acp-sessions?cwd=%2Fws%2Fa" }, + res, + { cs: clients.get("cid-1"), cid: "cid-1", pathname: "" }, + ); + const { sessions } = await readJson(res); + assert.equal(typeof sessions[0].updatedAt, "string"); + assert.ok(!Number.isNaN(Date.parse(sessions[0].updatedAt))); + }); + + test("no ?cwd falls back to cs.workspace.dir", async () => { + const res = fakeRes(); + await handleAcpSessions( + { url: "/api/acp-sessions" }, + res, + { cs: clients.get("cid-1"), cid: "cid-1", pathname: "" }, + ); + assert.equal((await readJson(res)).cwd, "/ws/a"); + }); + + test("an empty ?cwd with no workspace answers the empty string, and no filter is applied", async () => { + const cs = makeClientState(); + cs.workspace = { dir: null, branch: null, tree: null }; + registerAcpMock({ getMcodeSessionsForWorkspace: async () => [WIRE_SESSION] }); + const res = fakeRes(); + await handleAcpSessions({ url: "/api/acp-sessions?cwd=" }, res, { cs, cid: "c", pathname: "" }); + const body = await readJson(res); + assert.equal(body.cwd, ""); + // An empty cwd means "no filter" — the endpoint hands the empty string + // to the client and the client answers with everything. Shrinking this + // to the current workspace would silently empty the remote-control UI. + assert.equal(body.sessions.length, 1); + }); +}); + +// --------------------------------------------------------------------------- +// #10 GET /api/acp-session-title +// --------------------------------------------------------------------------- + +describe("#10 GET /api/acp-session-title — title shape", () => { + // Table-driven on the ENGINE answer and the WIRE answer. The `|| null` in + // the handler is the rule: "", undefined and "no such session" all reach + // the client as `title: null`, never as "" and never as a missing key. + const CASES = [ + ["a titled session", "mvs_1", "My Title", "My Title"], + ["an untitled session", "mvs_1", null, null], + ["an empty title", "mvs_1", "", null], + ["an undefined title", "mvs_1", undefined, null], + ]; + + for (const [name, sid, engineAnswer, wireTitle] of CASES) { + test(name, async () => { + registerAcpMock({ getMcodeSessionTitle: async () => engineAnswer }); + const res = fakeRes(); + await handleAcpSessionTitle( + { url: `/api/acp-session-title?sessionId=${sid}` }, + res, + {}, + ); + assert.equal(res._status, 200); + const body = await readJson(res); + assert.deepEqual(Object.keys(body), ["ok", "sessionId", "title"]); + assert.equal(body.sessionId, sid); + assert.equal(body.title, wireTitle); + }); + } + + // The 400 is unchanged by the batch: the gate sits behind the parameter + // check, so a missing sessionId is still a client error and never a 501. + test("a missing sessionId is still 400 {ok:false,error}, not 501", async () => { + const res = fakeRes(); + await handleAcpSessionTitle({ url: "/api/acp-session-title" }, res, {}); + assert.equal(res._status, 400); + const body = await readJson(res); + assert.deepEqual(body, { ok: false, error: "sessionId required" }); + }); + + test("an empty sessionId is still 400", async () => { + const res = fakeRes(); + await handleAcpSessionTitle({ url: "/api/acp-session-title?sessionId=" }, res, {}); + assert.equal(res._status, 400); + }); +}); + +// --------------------------------------------------------------------------- +// #72 GET /api/protocol/list-sessions +// --------------------------------------------------------------------------- + +describe("#72 GET /api/protocol/list-sessions — remote-control list shape", () => { + const CASES = [ + ["one entry", [WIRE_SESSION], [["sessionId", "cwd", "title", "updatedAt"]]], + ["an entry with the optional keys absent", [WIRE_SESSION_BARE], [["sessionId", "cwd"]]], + ]; + + for (const [name, engineAnswer, expectedFieldLists] of CASES) { + test(name, async () => { + registerAcpMock({ listAllMcodeSessions: async () => engineAnswer }); + const res = fakeRes(); + await handleListSessions( + { url: "/api/protocol/list-sessions?cwd=%2Fws%2Fa" }, + res, + { cs: clients.get("cid-1"), cid: "cid-1" }, + ); + assert.equal(res._status, 200); + const body = await readJson(res); + assert.deepEqual(Object.keys(body), ["ok", "sessions", "cwd"]); + assert.deepEqual(body.sessions.map((s) => Object.keys(s)), expectedFieldLists); + }); + } + + // The cwd filter is the route's own and its shape differs from #9's: an + // unfiltered read answers WITHOUT the `cwd` key at all, a filtered read + // answers WITH it. Both are load-bearing for the remote-control UI. + test("an empty cwd answers without the cwd key and without filtering", async () => { + registerAcpMock({ listAllMcodeSessions: async () => [WIRE_SESSION] }); + const res = fakeRes(); + await handleListSessions( + { url: "/api/protocol/list-sessions" }, + res, + { cs: { workspace: { dir: null } }, cid: "c" }, + ); + const body = await readJson(res); + assert.deepEqual(Object.keys(body), ["ok", "sessions"]); + assert.equal(body.sessions.length, 1); + }); + + test("the cwd filter is case- and slash-insensitive, and drops other workspaces", async () => { + registerAcpMock({ + listAllMcodeSessions: async () => [ + WIRE_SESSION, + { sessionId: "mvs_cccc", cwd: "/ws/other" }, + { sessionId: "mvs_dddd", cwd: "/WS/A/" }, + ], + }); + const res = fakeRes(); + await handleListSessions( + { url: "/api/protocol/list-sessions?cwd=%2Fws%2Fa" }, + res, + { cs: clients.get("cid-1"), cid: "cid-1" }, + ); + const body = await readJson(res); + assert.deepEqual( + body.sessions.map((s) => s.sessionId), + ["mvs_aaaa1111222233334444555566667777", "mvs_dddd"], + ); + }); + + test("no sessions answers [], never null", async () => { + registerAcpMock({ listAllMcodeSessions: async () => [] }); + const res = fakeRes(); + await handleListSessions( + { url: "/api/protocol/list-sessions?cwd=%2Fws%2Fa" }, + res, + { cs: clients.get("cid-1"), cid: "cid-1" }, + ); + const body = await readJson(res); + assert.deepEqual(body.sessions, []); + }); +}); + +// --------------------------------------------------------------------------- +// #74 GET /api/state — the first-frame render contract +// --------------------------------------------------------------------------- + +describe("#74 GET /api/state — snapshot first frame", () => { + // The first frame IS the frontend's render contract (red line in + // doc/m3-batch-plan.md §5), so the field list is pinned literally — + // order included, because the sidebar merges these positionally in some + // views. The first 18 keys are `makeClientState()`'s own client state, + // spread verbatim by the route; the last 9 are the route's additions, + // and `sessions` is the position `makeClientState()` gave it even though + // the route overwrites its value with the chat-stripped projection. + const BASELINE_FIELDS = [ + "version", + "workspace", + "model", + "sessionId", + "mcodeSessionId", + "sessionTitle", + "lastUsedWorkspace", + "context", + "usage", + "permissions", + "chat", + "sessions", + "goal", + "todo", + "ask", + "plan", + "running", + "recentSubagents", + "mcodeSessions", + "availableCommands", + "lanBroadcast", + "readOnly", + "tokenEnabled", + "currentToken", + "tokenAcknowledged", + "tokenRotatedAt", + "revision", + ]; + + test("the snapshot field list is exactly the baseline — none added, none removed, same order", async () => { + const res = fakeRes(); + await handleState({ url: "/api/state" }, res, { cid: "cid-1" }); + assert.equal(res._status, 200); + const body = await readJson(res); + assert.deepEqual(Object.keys(body), BASELINE_FIELDS); + }); + + test("every baseline field is present, so a reordering cannot hide a removal", async () => { + const res = fakeRes(); + await handleState({ url: "/api/state" }, res, { cid: "cid-1" }); + const body = await readJson(res); + for (const field of BASELINE_FIELDS) { + assert.ok(field in body, `snapshot must still carry "${field}"`); + } + }); + + test("mcodeSessions is the engine's workspace-filtered list, entry for entry", async () => { + registerAcpMock({ getMcodeSessionsForWorkspace: async () => [WIRE_SESSION] }); + const res = fakeRes(); + await handleState({ url: "/api/state" }, res, { cid: "cid-1" }); + const body = await readJson(res); + assert.deepEqual(body.mcodeSessions, [WIRE_SESSION]); + assert.deepEqual( + body.mcodeSessions.map((s) => Object.keys(s)), + [["sessionId", "cwd", "title", "updatedAt"]], + ); + }); + + test("an engine with no sessions answers mcodeSessions: [] — not null, not missing", async () => { + registerAcpMock({ getMcodeSessionsForWorkspace: async () => [] }); + const res = fakeRes(); + await handleState({ url: "/api/state" }, res, { cid: "cid-1" }); + const body = await readJson(res); + assert.ok(Array.isArray(body.mcodeSessions)); + assert.deepEqual(body.mcodeSessions, []); + }); + + // The local webui session list rides in the same payload under + // `sessions` and is stripped of `chat`; the engine mirror rides in + // `mcodeSessions`. Swapping the two would render the sidebar from the + // wrong source while every individual field still looked right. + test("the local session list and the engine mirror are separate keys", async () => { + registerAcpMock({ getMcodeSessionsForWorkspace: async () => [WIRE_SESSION] }); + const res = fakeRes(); + await handleState({ url: "/api/state" }, res, { cid: "cid-1" }); + const body = await readJson(res); + assert.ok(Array.isArray(body.sessions)); + assert.ok(Array.isArray(body.mcodeSessions)); + assert.notDeepEqual(body.sessions, body.mcodeSessions); + }); + + test("revision is a number and advances per read (the SSE monotonic contract)", async () => { + const first = fakeRes(); + await handleState({ url: "/api/state" }, first, { cid: "cid-1" }); + const second = fakeRes(); + await handleState({ url: "/api/state" }, second, { cid: "cid-1" }); + const a = await readJson(first); + const b = await readJson(second); + assert.equal(typeof a.revision, "number"); + assert.ok(b.revision > a.revision, "each /api/state read bumps the per-cid revision"); + }); +}); + +// --------------------------------------------------------------------------- +// #75 GET /api/health +// --------------------------------------------------------------------------- + +describe("#75 GET /api/health — version source", () => { + // Table-driven: [name, agentInfo, expected]. The facade is a pass-through + // for the agentInfo mirror, and the sentinel for "nothing attached" stays + // the string "unknown" — a null here breaks every semver-parsing monitor. + const CASES = [ + ["an attached client reports its version", { name: "mcode", version: "0.5.7" }, "0.5.7"], + ["a client with no version field reports unknown", { name: "mcode" }, "unknown"], + ["no client at all reports unknown", null, "unknown"], + ]; + + for (const [name, info, expected] of CASES) { + test(name, async () => { + registerAcpMock({ getMcodeServerInfo: () => info }); + const res = fakeRes(); + await handleHealth(null, res); + assert.equal(res._status, 200); + const body = await readJson(res); + assert.equal(body.mcodeVersion, expected); + assert.equal(typeof body.mcodeVersion, "string"); + }); + } + + test("the health payload is still exactly seven keys", async () => { + const res = fakeRes(); + await handleHealth(null, res); + const body = await readJson(res); + assert.deepEqual(Object.keys(body), [ + "ok", + "port", + "defaultModel", + "defaultWorkspace", + "mcodeCmd", + "mcodeVersion", + "maxConcurrent", + ]); + }); +}); diff --git a/packages/webui/test/server/app-hono.test.js b/packages/webui/test/server/app-hono.test.js index d76eb1db..881165a1 100644 --- a/packages/webui/test/server/app-hono.test.js +++ b/packages/webui/test/server/app-hono.test.js @@ -28,10 +28,21 @@ function fakeIncoming({ method = "GET", url = "/api/health", origin, remoteAddre }; } -/** Run the legacy health handler and return its captured status/headers/body. */ -function legacyHealth() { +/** + * Run the legacy health handler and return its captured status/headers/body. + * + * Async since M3-B1: `mcodeVersion` moved behind the engine facade + * (`server/engine/session-reads.js#readEngineVersion`), which resolves its + * acp-client dependency with a dynamic import, so `handleHealth` returns a + * promise. The legacy dispatcher awaits every handler + * (`server/router.js`: `await route.handler(req, res, ctx, pathname)`), so + * awaiting here matches production — a synchronous read here would compare + * the Hono body against an empty capture and pass/fail for the wrong + * reason. + */ +async function legacyHealth() { const capture = createResponseCapture(); - healthRoute.handleHealth(fakeIncoming(), capture); + await healthRoute.handleHealth(fakeIncoming(), capture); return capture.result(); } @@ -190,7 +201,7 @@ describe("app.js — Hono route parity with the legacy dispatcher", () => { test("GET /api/health returns the legacy payload byte for byte", async () => { const res = await app.request("/api/health", {}, { incoming: fakeIncoming() }); assert.equal(res.status, 200); - const expected = legacyHealth(); + const expected = await legacyHealth(); assert.equal(await res.text(), expected.body); assert.equal(res.headers.get("content-type"), expected.headers.get("content-type")); }); diff --git a/release/public-source.json b/release/public-source.json index 2cdd8f24..8fe872ef 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3452,6 +3452,7 @@ "packages/webui/server/engine/providers/local-runtime-v2.capabilities.js", "packages/webui/server/engine/providers/local-runtime-v2.js", "packages/webui/server/engine/providers/tui-runtime-adapter.js", + "packages/webui/server/engine/session-reads.js", "packages/webui/server/lib/acp-client.js", "packages/webui/server/lib/agent-team-detect.js", "packages/webui/server/lib/agent-team-status.js", @@ -3590,6 +3591,7 @@ "packages/webui/test/lib/engine/capabilities.test.js", "packages/webui/test/lib/engine/capability-snapshot.test.js", "packages/webui/test/lib/engine/host-facade.test.js", + "packages/webui/test/lib/engine/session-reads.test.js", "packages/webui/test/lib/events-concurrency.test.js", "packages/webui/test/lib/events-hash.test.js", "packages/webui/test/lib/events.test.js", @@ -3668,6 +3670,7 @@ "packages/webui/test/routes/protocol.check.mjs", "packages/webui/test/routes/provider-presets.check.mjs", "packages/webui/test/routes/providers.check.mjs", + "packages/webui/test/routes/session-reads.check.mjs", "packages/webui/test/routes/sessions-search.check.mjs", "packages/webui/test/routes/sessions-switch-workspace-follow.check.mjs", "packages/webui/test/routes/sessions-switch.check.mjs", From e053ae7efcadff43668c010602454b2eedda0140 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 02:56:34 +0800 Subject: [PATCH 10/21] feat(webui): the session-tree and export endpoints ask the engine facade (M3-B2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Routes #8 GET /api/session-tree and #11 GET /api/sessions/:id/export through the engine facade instead of the transport, keeping every response shape, status code and reason string unchanged. The two families are separate files because their gate policies are opposite. The tree is entirely engine data, so a provider that cannot list sessions genuinely has no tree: assertSessionTreeCapability throws and invokeHandler answers 501. Export's primary source is sessions.json and the engine only contributes a best-effort transcript enrichment the endpoint has always promised never to block on, so checkSessionExportCapability reports and never throws — gating it hard would delete working functionality in response to a declaration about a capability the endpoint does not depend on. The tree route re-throws the capability error, matched with the class's own instanceof helper rather than a `.name` compare: `name` is a writable instance property, so a stray `err.name = "…"` would silently turn that 501 back into the 200 soft-fail the gate exists to prevent. A test pins both halves — the real class propagates, an impostor carrying the right `.name` does not. Verified by exporting the real tree (303 rows, 32 projects, 299 nodes) before and after and diffing every node's id/title/parent/depth: 3289 field comparisons, zero differences. A synthetic fixture covers what the live data does not contain (orphans, cycles, four-level nesting, exotic titles): 165 comparisons, zero differences. Two pre-existing shapes are pinned because a "cleanup" would silently break them — child nodes carry no `children` key (all 66 of them), and the response has no parent_session_id key at all. --- packages/webui/docs/ARCHITECTURE.md | 76 +- packages/webui/docs/ARCHITECTURE.zh-CN.md | 64 +- packages/webui/server/engine/index.js | 21 + .../webui/server/engine/session-export.js | 235 +++++ .../webui/server/engine/session-tree-reads.js | 220 +++++ packages/webui/server/routes/export.js | 28 +- packages/webui/server/routes/sessions.js | 41 +- .../test/lib/engine/session-export.test.js | 600 ++++++++++++ .../lib/engine/session-tree-reads.test.js | 854 ++++++++++++++++++ release/public-source.json | 4 + 10 files changed, 2124 insertions(+), 19 deletions(-) create mode 100644 packages/webui/server/engine/session-export.js create mode 100644 packages/webui/server/engine/session-tree-reads.js create mode 100644 packages/webui/test/lib/engine/session-export.test.js create mode 100644 packages/webui/test/lib/engine/session-tree-reads.test.js diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index 67749444..8f195b72 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -489,7 +489,7 @@ not import it but adopts the same shape. Unknown future statuses render as ### `engine/` (capability declarations + the local-runtime-v2 host) The engine abstraction lives at `server/engine/` (engine-abstraction -batch B1; migration state M1, plus M3 batches B0 and B1). Eight files, +batch B1; migration state M1, plus M3 batches B0, B1 and B2). Ten files, one job each: | File | Owns | @@ -502,6 +502,8 @@ one job each: | `engine/providers/local-runtime-v2.js` | `createCatalogueHost` (moved verbatim from `runtime-host.js`, which re-exports it) + re-exports the declaration above, so consumers keep one import shape. This is the heavy one — `@mavis/local-runtime-v2`, `@mavis/config`, `@minimax/code/runtime-adapter` — and no file `app.js` reaches may import it | | `engine/providers/tui-runtime-adapter.js` | `TUI_RUNTIME_ADAPTER_CAPABILITIES` (declaration only — the adapter itself is constructed inside the v2 host) | | `engine/session-reads.js` | The directory-read family's facade calls (`readEngineSessionList`, `readEngineSessionListForWorkspace`, `readEngineSessionTitle`, `readEngineVersion`) and the endpoint→capability table `SESSION_READ_ENDPOINTS` (step M3, batch B1) | +| `engine/session-tree-reads.js` | The session-tree family's facade call (`readEngineSessionTree`) and the endpoint→capability table `SESSION_TREE_ENDPOINTS` (step M3, batch B2). Gates **hard**: `assertSessionTreeCapability` throws → 501, because the tree is entirely engine data. Forwards to `lib/session-tree.js#getSessionTree`; the assembler is not duplicated | +| `engine/session-export.js` | The export family's facade call (`readEngineSessionTranscript`) and the endpoint→capability table `SESSION_EXPORT_ENDPOINTS` (step M3, batch B2). Gates **soft**: `checkSessionExportCapability` reports and never throws, because export's primary source is `sessions.json`, not the engine | Routes take the host from the facade and never from `lib/acp-client.js`: `routes/plugins.js` and `routes/turn-diff.js` call @@ -582,6 +584,10 @@ enforces it against the real module graph rather than against source text. `engine/session-reads.js` lives under the same rule: its static imports are `engine/capabilities.js` and `engine/index.js` only, and `lib/acp-client.js` + `lib/config.js` are reached through `await import()` inside the functions. +Batch B2's two files hold to it identically — `lib/session-tree.js` and +`lib/transcript.js` are reached through `await import()`, and neither file +statically imports `engine/capabilities.js` beyond the single +`assertEngineCapability` binding the tree family actually calls. #### Which endpoints read through the facade (step M3, batch B1) @@ -621,6 +627,74 @@ The transport→provider table has one entry (`runtime`). Under the default provider and the table gains its row. Passing through is not the same as claiming support, and the two are reported differently on purpose. +#### Which endpoints read through the facade (step M3, batch B2) + +Batch B2 adds two endpoints, and they are the first two whose gate policies +**differ**. They are separate files for that reason; merging them would force +one to inherit the other's. + +| Endpoint | Capability · sub-item | Enforcement | Value source | +| --- | --- | --- | --- | +| `GET /api/session-tree` | `sessionCrud` · `listSessions` | hard — 501 | `lib/session-tree.js#getSessionTree`, forwarded verbatim | +| `GET /api/sessions/:id/export` | `sessionCrud` · `getSession` | soft — reported | `lib/transcript.js#readMcodeTranscript` (the enrichment only) | + +**Why one gate throws and the other does not.** `/api/session-tree` is +entirely engine data: the hierarchy is assembled from `local_runtime_sessions` +in the runtime db, so a provider that cannot list sessions genuinely has no +tree to return, and 501 is the honest answer. `/api/sessions/:id/export` is +mostly *not* engine data — the conversation comes from `sessions.json`, and +the engine only contributes a best-effort transcript enrichment the endpoint +has always promised never to block on. Gating it hard would delete working +functionality in response to a declaration about a capability the endpoint +does not depend on. So `checkSessionExportCapability` answers what the +provider declared and returns; the caller degrades `_meta.mcode_unavailable` +through the endpoint's own pre-existing channel, and the export still serves +the full webui chat. `test/lib/engine/session-export.test.js` pins this by +swapping in a provider that declares `sessionCrud: none` and asserting that +export reports while the tree family throws on the same fixture. + +Four properties this batch holds, each with a test behind it: + +1. **The node shape is unchanged, and it is asymmetric.** A root node + carries `{id, title, agent, kind, status, updatedAt, children}`; a child + node carries the same fields **without** `children`, because + `buildTree` adds that key only in the output map that wraps each root. + Measured on the real tree: 233 root nodes carry `children`, all 66 child + nodes do not. "Normalising" this would change 66 nodes' shape in the + sidebar. +2. **There is no `parent_session_id` in the response.** The hierarchy is + structural — expressed through `children` — and `parent_session_id` + exists only inside the db read. A future addition of that key to the node + is a client-visible change, so the exact key set is asserted per depth. +3. **One assembler.** `buildTree` remains the only thing that decides which + rows attach to which parent, and the route does not re-derive the + hierarchy. Rows that cannot attach — an orphan whose parent is not in the + row set, a cross-directory parent, a grandchild, a child of a `root` + container row, anything in a cycle — are dropped, as they always were. + That is why the batch was verified by exporting the tree before and + after and diffing every node, not by counting rows. +4. **The tree's 501 is not swallowed.** The route's existing `try/catch` + would otherwise fold the capability error into its own + `{ok:false, reason:"session_tree_failed"}` soft-fail body and turn a 501 + into a 200. The route re-throws `EngineCapabilityNotSupportedError` and + keeps the soft-fail path for everything else. + +**`source` is not transport-switched for the tree.** The tree is read from +the engine's own runtime db, which both the `runtime` and the `acp` +transport can see, so `readEngineSessionTree` reports `source: "runtime-db"` +under every transport rather than claiming a catalogue answer. The +declaration check is still transport-keyed: which provider is active is a +transport question even when the read itself is not. + +**Export's enrichment is currently inert against the v2 schema, by +design.** `lib/transcript.js` keeps its `v2-data-json` probe OUT of the +default probe set so that export's behaviour does not change, and the live +`local_runtime_message_rows` has no `content` column. So on a current +runtime db the enrichment answers `no_matching_table` and every export +reports `_meta.mcode_unavailable: true` with +`_meta.source: "webui"`. That is pre-existing and deliberately preserved — +re-enabling it is a behaviour change for a later slice, not a refactor. + ## 4. The `clientState` payload This is the shape every SSE `state` event contains. The webui mirrors diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index 2f8cf85e..44a1fa48 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -461,7 +461,7 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 ### `engine/`(能力声明 + local-runtime-v2 host) 引擎抽象层位于 `server/engine/`(engine-abstraction 批次 B1;迁移 -状态 M1,外加 M3 的 B0 与 B1 两批)。八个文件,各管一件事: +状态 M1,外加 M3 的 B0、B1 与 B2 三批)。十个文件,各管一件事: | 文件 | 职责 | | --- | --- | @@ -473,6 +473,8 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 | `engine/providers/local-runtime-v2.js` | `createCatalogueHost`(自 `runtime-host.js` 原样移入,后者转发导出)+ 转发导出上面的声明,消费方的 import 形状因此不变。它是重的那一个——`@mavis/local-runtime-v2`、`@mavis/config`、`@minimax/code/runtime-adapter`——`app.js` 能触达的文件里绝不许 import 它 | | `engine/providers/tui-runtime-adapter.js` | `TUI_RUNTIME_ADAPTER_CAPABILITIES`(仅声明——adapter 本体在 v2 host 内构造) | | `engine/session-reads.js` | 目录读族的面板调用(`readEngineSessionList`、`readEngineSessionListForWorkspace`、`readEngineSessionTitle`、`readEngineVersion`)与端点→能力对照表 `SESSION_READ_ENDPOINTS`(迁移步 M3 批次 B1) | +| `engine/session-tree-reads.js` | 会话树族的面板调用 `readEngineSessionTree` 与端点→能力对照表 `SESSION_TREE_ENDPOINTS`(迁移步 M3 批次 B2)。**硬门控**:`assertSessionTreeCapability` 抛出 → 501,因为树完全由引擎数据构成。转发到 `lib/session-tree.js#getSessionTree`,树的装配逻辑不复制第二份 | +| `engine/session-export.js` | 导出族的面板调用 `readEngineSessionTranscript` 与端点→能力对照表 `SESSION_EXPORT_ENDPOINTS`(迁移步 M3 批次 B2)。**软门控**:`checkSessionExportCapability` 只报告、从不抛出,因为导出的主数据源是 `sessions.json` 而非引擎 | 路由从门面取 host,不从 `lib/acp-client.js` 取:`routes/plugins.js` 与 `routes/turn-diff.js` 调 `getEngineCatalogueHost()`。两者都保留 `deps` @@ -538,7 +540,10 @@ handler 层测试因此保持封闭。 对着真实模块图强制它,而不是对着源码文本。 `engine/session-reads.js` 服从同一条纪律:它的静态 import 只有 `engine/capabilities.js` 与 `engine/index.js`,`lib/acp-client.js` + `lib/config.js` -都在函数体内用 `await import()` 触达。 +都在函数体内用 `await import()` 触达。批次 B2 的两个文件同样守住它: +`lib/session-tree.js` 与 `lib/transcript.js` 都用 `await import()` 触达, +且除树族真正调用的那一个 `assertEngineCapability` 绑定外, +两个文件都没有静态 import `engine/capabilities.js`。 #### 哪些端点走门面读(迁移步 M3 批次 B1) @@ -573,6 +578,61 @@ handler 层测试因此保持封闭。 provider,于是门控报告 `unregistered-transport` 并放行——M4 注册 ACP provider 后该表补上对应行。放行不等于声称支持,二者刻意分开报告。 +#### 哪些端点走门面读(迁移步 M3 批次 B2) + +批次 B2 收编 2 个端点,它们是前两个**门控策略不同**的端点。正因如此才 +拆成两个文件:合并会迫使其中一个继承另一个的策略。 + +| 端点 | 能力 · 子项 | 强制方式 | 取值来源 | +| --- | --- | --- | --- | +| `GET /api/session-tree` | `sessionCrud` · `listSessions` | 硬——501 | `lib/session-tree.js#getSessionTree`,原样转发 | +| `GET /api/sessions/:id/export` | `sessionCrud` · `getSession` | 软——只报告 | `lib/transcript.js#readMcodeTranscript`(仅增强部分) | + +**为什么一个门控抛错、另一个不抛。** `/api/session-tree` 完全是引擎数据: +层级由运行时库 `local_runtime_sessions` 装配,所以一个列不出会话的 +provider 确实没有树可返回,501 才是诚实答案。 +`/api/sessions/:id/export` 则**主要不是**引擎数据——对话来自 +`sessions.json`,引擎只贡献一份尽力而为的 transcript 增强,而该端点一直 +承诺绝不因此阻断导出。把它改成硬门控,等于因为一条关于「本端点并不依赖的 +能力」的声明而删掉本来能用的功能。所以 `checkSessionExportCapability` +只回答 provider 声明了什么然后返回;调用方通过端点既有的通道降级 +`_meta.mcode_unavailable`,导出照旧完整返回 webui 的对话。 +`test/lib/engine/session-export.test.js` 用一份声明 `sessionCrud: none` +的 provider 钉住这一点:同一份样本下,导出族报告、树族抛错。 + +本批守住的四条性质,每条背后都有测试: + +1. **节点形状未变,而且它是不对称的。** 根节点带 + `{id, title, agent, kind, status, updatedAt, children}`;子节点带同样 + 这些字段但**没有** `children`——因为 `buildTree` 只在包裹每个根节点的 + 输出映射里补这个键。在真实树上实测:233 个根节点带 `children`, + 66 个子节点全都不带。「顺手规范化」会让侧边栏里 66 个节点的形状改变。 +2. **响应里没有 `parent_session_id`。** 层级是结构性的——由 `children` + 表达——`parent_session_id` 只存在于读库阶段。将来把这个键加到节点上 + 就是客户端可见的变更,所以测试按深度逐字断言键集合。 +3. **只有一个装配器。** `buildTree` 仍是唯一决定哪些行挂到哪个父节点 + 的地方,路由不重新推导层级。挂不上的行——父节点不在结果集里的孤儿、 + 跨目录的父节点、孙节点、挂在 `root` 容器行下的子节点、任何处于环中的 + 行——照旧被丢弃。正因如此,本批的验证方式是改前改后各导一次树、 + 逐节点比对,而不是数行数。 +4. **树的 501 不被吞掉。** 路由原有的 `try/catch` 否则会把能力错误 + 折进它自己的 `{ok:false, reason:"session_tree_failed"}` 软失败体里, + 把 501 变成 200。路由重新抛出 `EngineCapabilityNotSupportedError`, + 其余错误仍走软失败。 + +**树的 `source` 不随传输切换。** 树读自引擎自己的运行时库,`runtime` +与 `acp` 两种传输都看得到,所以 `readEngineSessionTree` 在任何传输下都 +报告 `source: "runtime-db"`,而不是假称拿到了目录宿主。声明检查仍按传输 +分派:当前哪个 provider 生效是传输问题,即使这次读本身不是。 + +**export 的增强在 v2 表结构下当前是失效的,且这是刻意为之。** +`lib/transcript.js` 把 `v2-data-json` 探针留在默认探针集**之外**, +以保证 export 的行为不变;而线上真实的 `local_runtime_message_rows` +根本没有 `content` 列。因此在当前运行时库上增强会返回 +`no_matching_table`,每次导出都报告 `_meta.mcode_unavailable: true` 与 +`_meta.source: "webui"`。这是既有行为且被刻意保留——重新启用它是一次行为 +变更,属于后续切片,不属于这次收编。 + ## 4. `clientState` 载荷 这是每个 SSE `state` 事件所包含的形状。webui 将其 diff --git a/packages/webui/server/engine/index.js b/packages/webui/server/engine/index.js index c3a8d851..6843a322 100644 --- a/packages/webui/server/engine/index.js +++ b/packages/webui/server/engine/index.js @@ -82,6 +82,27 @@ export { readEngineVersion, resolveSessionReadProvider, } from "./session-reads.js"; +// Step M3, batch B2: the session-tree read (#8) and the export +// enrichment read (#11). Two modules, not one, because their gate +// policies are opposite and a single file would force one of them to +// inherit the other's: #8 is 100% engine data and gates HARD (501 via +// `assertSessionTreeCapability`), while #11's primary source is +// `sessions.json` and gates SOFT (`checkSessionExportCapability` +// reports, never throws) so a provider that cannot serve a transcript +// degrades the enrichment instead of the export. The same TDZ rule as +// B1 applies to both: read nothing from this module at module scope. +export { + SESSION_TREE_ENDPOINTS, + assertSessionTreeCapability, + readEngineSessionTree, + resolveSessionTreeProvider, +} from "./session-tree-reads.js"; +export { + SESSION_EXPORT_ENDPOINTS, + checkSessionExportCapability, + readEngineSessionTranscript, + resolveSessionExportProvider, +} from "./session-export.js"; export { LOCAL_RUNTIME_V2_CAPABILITIES } from "./providers/local-runtime-v2.capabilities.js"; export { TUI_RUNTIME_ADAPTER_CAPABILITIES } from "./providers/tui-runtime-adapter.js"; diff --git a/packages/webui/server/engine/session-export.js b/packages/webui/server/engine/session-export.js new file mode 100644 index 00000000..f02d44ec --- /dev/null +++ b/packages/webui/server/engine/session-export.js @@ -0,0 +1,235 @@ +// webui/server/engine/session-export.js +// +// Migration step M3, batch B2: the export enrichment read (导出增强读) — +// +// #11 GET /api/sessions/:id/export?format=md|json +// +// What this file is for. Export is the ONE endpoint in this migration +// whose primary data source is webui's own store, not the engine: the +// conversation comes from `lib/sessions.js` (`sessions.json`), and the +// route parses it with its own line grammar and renders md/json. The +// engine's only contribution is the transcript enrichment +// (`readMcodeTranscript`), which the route has always treated as +// BEST-EFFORT — "If the db is unreadable / the table is missing / the +// schema differs, we set `_meta.mcode_unavailable` and continue with the +// webui source — never block export". This file moves that one +// engine-facing read behind the facade so the route stops naming +// `lib/transcript.js` directly, and so the place where the enrichment +// can fail is stated once, in the engine layer, instead of being implied +// by control flow in a route. +// +// Why this family has a SOFT gate and the tree family has a HARD one. +// This is the one place where copying B1's shape verbatim would have +// been wrong, so the difference is deliberate and load-bearing: +// +// - #8 session-tree is 100% engine data. No session listing, no tree. +// Answering `501` is the only honest response, and the endpoint +// already had a documented "cannot read" shape to fall back on. +// - #11 export is mostly NOT engine data. A provider that declared +// `sessionCrud: none` would still leave the full user-visible chat +// exportable from `sessions.json`. Gating the endpoint hard would +// REMOVE working functionality in response to a declaration about a +// capability the endpoint does not actually depend on — and it would +// break the explicit "never block export" contract, which is this +// repository's #110 discipline applied in the other direction: a +// missing enrichment must not be dressed up as a failure, and a +// missing capability must not be dressed up as one either. +// +// So the gate here REPORTS and never throws. `checkSessionExportCapability` +// answers what the provider declared, and the read degrades through the +// endpoint's own pre-existing fail-soft channel +// (`_meta.mcode_unavailable` + `mcode_unavailable_reason`) rather than +// through an HTTP status. The 501 machinery in `errors.js` stays +// untouched and unused by this family — that is a policy statement, not +// an oversight, and the test suite pins it. +// +// What this file deliberately does NOT do: +// +// - It does not own the export. Parsing webui chat lines, merging the +// two message sources, and rendering md/json are the route's job and +// stay there; they are presentation, not engine access. Only the +// transcript read crosses this seam. +// - It does not own the session lookup. `_findSession` resolves a +// webui id or an `mvs_` id against `sessions.json` — webui's own +// store, governed by no engine capability. +// - It does not construct a host. +// +// Boot-path weight. `app.js` imports the routes, the routes import this +// file, so this file is on the boot path. It statically imports nothing +// heavier than `capabilities.js` and `index.js`; `lib/transcript.js` +// and `lib/config.js` are reached through `await import()` inside the +// functions — the M1 lesson again. +// +// Provider selection is M4's job, same as B1 and as the tree family. + +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * Transport → registered engine provider id. Same shape and same + * rationale as `session-reads.js#providerByTransport` and + * `session-tree-reads.js#providerByTransport`; kept per-family so each + * family owns its own gate policy. Collapse the three in M4, not here. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +/** + * The declaration this endpoint's ENRICHMENT needs. + * + * `sessionCrud` / `getSession` is the honest mapping — the same pair + * B1's `GET /api/acp-session-title` uses, because reading a session's + * transcript is reading that session. It is the ENRICHMENT that is + * declared, not the export: see the soft-gate rationale in the header. + * + * @type {Readonly>} + */ +export const SESSION_EXPORT_ENDPOINTS = Object.freeze({ + "GET /api/sessions/:id/export": { + capability: "sessionCrud", + subItem: "getSession", + enforcement: "soft", + }, +}); + +/** + * Resolve the provider that answers export enrichment on `transport`, + * or `null` when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveSessionExportProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Read the declaration for this endpoint WITHOUT enforcing it. + * + * Returns a descriptor whose `gate` field says what happened: + * + * - `"checked"` — provider resolved, capability is `full`. + * - `"unregistered-transport"` — no provider claims this transport yet. + * - `"capability-absent"` — the provider WAS found and DOES declare the + * capability as `none` (or `partial` missing this sub-item). This is + * the branch that makes the soft gate visible: the caller's next + * move is to degrade the ENRICHMENT, not to fail the request. + * - `"partial"` — provider is `partial` and this sub-item is + * absent; the endpoint still degrades, but the descriptor says so + * precisely. + * + * Deliberately never throws `EngineCapabilityNotSupportedError`. A + * caller that wants the hard behaviour (the tree family) must ask for + * it explicitly; that asymmetry is the point of splitting the two. + * A genuinely unknown endpoint key is still a plain Error — caller + * confusion is not a capability question. + * + * @param {string} endpoint A key of SESSION_EXPORT_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null, enforcement: "soft"}} + */ +export function checkSessionExportCapability(endpoint, transport) { + const need = SESSION_EXPORT_ENDPOINTS[endpoint]; + if (need === undefined) { + const err = new Error( + `checkSessionExportCapability: "${endpoint}" is not part of the session-export family ` + + `(known: ${Object.keys(SESSION_EXPORT_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_session_export_endpoint"; + throw err; + } + const base = { + endpoint, + provider: null, + capability: need.capability, + subItem: need.subItem, + enforcement: need.enforcement, + }; + const provider = resolveSessionExportProvider(transport); + if (!provider) return { ...base, gate: "unregistered-transport" }; + const entry = provider.capabilities ? provider.capabilities[need.capability] : undefined; + const descriptor = { ...base, provider: provider.id }; + if (entry && entry.level === "full") { + return { ...descriptor, gate: "checked" }; + } + if (entry && entry.level === "partial") { + const absent = Array.isArray(entry.missing) && entry.missing.includes(need.subItem); + return { ...descriptor, gate: absent ? "partial" : "checked" }; + } + // `none`, or no entry at all — the provider was found and does not + // offer this. Report it; the caller degrades the enrichment. + return { ...descriptor, gate: "capability-absent" }; +} + +/** + * Lazily resolve the transcript module and the active transport. + * Dynamic: `lib/transcript.js` reaches the sqlite resolver and the + * settings chain, neither of which may sit on the boot path. + */ +async function exportDeps() { + const [transcript, config] = await Promise.all([ + import("../lib/transcript.js"), + import("../lib/config.js"), + ]); + return { transcript, transport: config.MCODE_WEBUI_TRANSPORT }; +} + +/** + * Where the enrichment's bytes came from. `engine` when the transcript + * reader answered; `none` when it did not (and the caller degrades). + * + * @typedef {"engine" | "none"} SessionExportSource + */ + +/** + * The #11 engine-facing read: one session's transcript, best-effort. + * + * The returned `ok` / `reason` / `messages` are `readMcodeTranscript`'s + * own values, forwarded verbatim — this facade never invents a reason + * code and never converts a failure into an exception, because the + * endpoint's `_meta.mcode_unavailable` / `mcode_unavailable_reason` + * contract is built on those exact strings. `probeTable` and `probe` + * are the reader's own `source` / `probe`, renamed so they cannot be + * confused with this layer's `source`. + * + * Async even though the reader is synchronous (better-sqlite3 is sync): + * the route is already async, and a uniform awaitable `readEngine*` + * seam means a provider-backed transcript source that IS async (a + * network engine) needs no signature change at this layer. + * + * @param {object} [options] + * @param {string} [options.mcodeSessionId] The `mvs_…` id to read. + * @param {string} [options.endpoint] Endpoint key for the + * declaration check; defaults to `/api/sessions/:id/export`. + * @param {string} [options.transport] Transport override; defaults + * to the active `MCODE_WEBUI_TRANSPORT`. + * @returns {Promise<{mcodeSessionId: string, messages: Array, ok: boolean, reason: string|null, probeTable: string|null, probe: string|null, source: SessionExportSource, gate: object, transport: string}>} + */ +export async function readEngineSessionTranscript(options = {}) { + const endpoint = options.endpoint || "GET /api/sessions/:id/export"; + const deps = await exportDeps(); + const transport = options.transport || deps.transport; + const gate = checkSessionExportCapability(endpoint, transport); + const mcodeSessionId = options.mcodeSessionId || ""; + const r = deps.transcript.readMcodeTranscript(mcodeSessionId); + return { + mcodeSessionId, + messages: Array.isArray(r.messages) ? r.messages : [], + ok: r.ok === true, + reason: r.ok === true ? null : r.reason || "unknown", + probeTable: r.source || null, + probe: r.probe || null, + source: r.ok === true ? "engine" : "none", + gate, + transport, + }; +} diff --git a/packages/webui/server/engine/session-tree-reads.js b/packages/webui/server/engine/session-tree-reads.js new file mode 100644 index 00000000..cfb3a4e0 --- /dev/null +++ b/packages/webui/server/engine/session-tree-reads.js @@ -0,0 +1,220 @@ +// webui/server/engine/session-tree-reads.js +// +// Migration step M3, batch B2: the session-tree read (树读) — +// +// #8 GET /api/session-tree — the sidebar's Project → directory → +// session → subagent tree +// +// What this file is for. #8 is the main↔subagent communication spine: the +// hierarchy the user navigates is built from `parent_session_id`, and a +// child that fails to attach to its parent is a subagent the user cannot +// see. So the endpoint's payload is a frontend contract of the strictest +// kind here, and the route now reaches the engine through this file +// instead of calling `lib/session-tree.js` on its own: the facade checks +// the provider's DECLARATION first, then forwards to the one existing +// implementation. A provider that does not offer session listing answers +// 501 through `app.js#invokeHandler`'s EngineCapabilityNotSupportedError +// mapping rather than an empty tree, which the sidebar would render as +// "this project has no sessions" (#110 fake-success failure mode). +// +// What this file deliberately does NOT do: +// +// - It does not re-assemble the tree. `lib/session-tree.js` owns the +// level mapping (project / directory / branch / subagent), the +// git-based project resolution and the 15s cache. A second +// assembler here would be a second answer to a hierarchy question +// that must have exactly one — and getting it wrong by one level is +// precisely the failure this batch is gated on. +// - It does not normalise nodes. The node shape +// `{id, title, agent, kind, status, updatedAt, children}` is +// whatever `buildTree` produces, byte-for-byte. Note what is NOT in +// it: the response carries NO `parent_session_id` key. The hierarchy +// is expressed structurally through `children`; `parent_session_id` +// exists only inside the db read. `test/lib/engine/session-tree-reads.test.js` +// pins the exact key set so a future "helpful" addition is caught. +// - It does not construct a host. The tree is assembled from the +// engine's own runtime db (see the source note below), so there is +// no host in this path at all — see `Never build a second host`. +// - It does not widen the engine's own degradation. A db that cannot +// be read is still `{ok:false, reason}` with HTTP 200, exactly as +// before; that is the sidebar's documented fallback to the wrapper +// list, and the capability gate is a different question (may this +// provider list sessions AT ALL) from "could we read the db right +// now" (could we read it THIS TIME). +// +// Transport. Unlike the B1 read family, this read is deliberately +// transport-independent: it reads `local_runtime_sessions` in the +// runtime db, the engine's own persistent store, which both the +// `runtime` and the `acp` transport can see. `source` therefore +// reports `runtime-db` under every transport rather than pretending +// to be a catalogue answer. The DECLARATION check is still +// transport-keyed, because which provider is active is a transport +// question even when the read itself is not. +// +// Boot-path weight. `app.js` imports the routes, the routes import this +// file, so this file is on the boot path. It therefore statically +// imports nothing heavier than `capabilities.js` and `index.js` (both +// pure declaration modules); `lib/session-tree.js` and `lib/config.js` +// are reached through `await import()` inside the functions. That split +// is the M1 lesson — putting the `@mavis/*` tree on the boot path once +// cost 209ms → 2700ms of server start and broke the integration tests' +// 3s window. +// +// Provider selection is M4's job, same as B1: `providerByTransport()` +// maps a transport to a REGISTERED provider id; today only `runtime` has +// one, so under the default `acp` transport the gate reports +// `gate: "unregistered-transport"` instead of inventing one. + +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, exactly as in + * `session-reads.js#providerByTransport`, which this mirrors rather than + * merges: the two families have separate gate semantics (see + * `session-export.js` for the soft-gate counterpart) and a shared table + * would force one of them to inherit the other's policy. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. Every + * consumer of the table is a function anyway. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +/** + * The declaration this endpoint needs, and the sub-item it needs from + * that capability. + * + * `sessionCrud` / `listSessions` is the honest mapping, and it is the + * same pair B1's `GET /api/protocol/list-sessions` uses: both endpoints + * answer "every session the engine knows, across all workspaces", and + * the tree is that list plus a hierarchy. The tree additionally needs + * the `parent_session_id` column, but that is not a separate + * provider method — it is a column of the same rows, so naming a + * sub-item that no provider enumerates would be a lie in the registry. + * + * @type {Readonly>} + */ +export const SESSION_TREE_ENDPOINTS = Object.freeze({ + "GET /api/session-tree": { capability: "sessionCrud", subItem: "listSessions" }, +}); + +/** + * Resolve the provider that answers the tree read on `transport`, or + * `null` when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveSessionTreeProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Check the tree read against the active provider's declaration. Throws + * `EngineCapabilityNotSupportedError` — which `app.js#invokeHandler` + * turns into 501 — when the declaration says the capability (or the + * exact sub-item) is absent. + * + * @param {string} endpoint A key of SESSION_TREE_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null}} + */ +export function assertSessionTreeCapability(endpoint, transport) { + const need = SESSION_TREE_ENDPOINTS[endpoint]; + if (need === undefined) { + // Caller confusion, not an engine limitation — a plain Error so the + // HTTP layer never answers 501 for a typo in webui's own code. + const err = new Error( + `assertSessionTreeCapability: "${endpoint}" is not part of the session-tree family ` + + `(known: ${Object.keys(SESSION_TREE_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_session_tree_endpoint"; + throw err; + } + const provider = resolveSessionTreeProvider(transport); + if (!provider) { + return { + endpoint, + gate: "unregistered-transport", + provider: null, + capability: need.capability, + subItem: need.subItem, + }; + } + assertEngineCapability(provider.capabilities, need.capability, provider.id, need.subItem); + return { + endpoint, + gate: "checked", + provider: provider.id, + capability: need.capability, + subItem: need.subItem, + }; +} + +/** + * Lazily resolve the session-tree module and the active transport. + * Dynamic on both counts: `lib/session-tree.js` reaches the sqlite + * resolver and the settings chain, `lib/config.js` reads env — neither + * may sit on the boot path. + */ +async function treeDeps() { + const [tree, config] = await Promise.all([ + import("../lib/session-tree.js"), + import("../lib/config.js"), + ]); + return { tree, transport: config.MCODE_WEBUI_TRANSPORT }; +} + +/** + * Where the tree's bytes came from. Always `runtime-db`: the tree is + * assembled from `local_runtime_sessions` in the engine's own runtime + * db, which is not a transport-switched surface (see the transport note + * in the file header). The value exists so a consumer never has to + * guess whether the ACP mirror answered instead. + * + * @typedef {"runtime-db"} SessionTreeSource + */ + +/** + * The #8 (`GET /api/session-tree`) read. + * + * Forwards `options` straight to `getSessionTree`, so `force` keeps its + * meaning (`?refresh=1` bypasses the 15s cache) and the `cached` field + * keeps its shape. The returned `tree` is the endpoint's payload + * verbatim — including the `ok:false` / `reason` soft-fail shape for a + * missing or unreadable db, which this facade deliberately does not + * convert into an error. + * + * @param {object} [options] + * @param {boolean} [options.force] Bypass the 15s cache. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/session-tree`. + * @param {string} [options.now] Clock injection, forwarded as-is. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. Exists so tests can exercise + * both the `runtime` and the unregistered `acp` branch without + * mutating process env. + * @returns {Promise<{tree: object, source: SessionTreeSource, gate: object, transport: string}>} + */ +export async function readEngineSessionTree(options = {}) { + const endpoint = options.endpoint || "GET /api/session-tree"; + const deps = await treeDeps(); + const transport = options.transport || deps.transport; + const gate = assertSessionTreeCapability(endpoint, transport); + const tree = deps.tree.getSessionTree({ + force: options.force === true, + ...(options.now === undefined ? {} : { now: options.now }), + }); + return { tree, source: "runtime-db", gate, transport }; +} diff --git a/packages/webui/server/routes/export.js b/packages/webui/server/routes/export.js index f1a5b5fb..fb5baf4c 100644 --- a/packages/webui/server/routes/export.js +++ b/packages/webui/server/routes/export.js @@ -25,14 +25,18 @@ // - ?download=true → Content-Disposition: attachment; filename="-." import { loadSessions } from "../lib/sessions.js"; -// v2 (2026-09-20 webui-manual-audit): _readMcodeTranscript's core moved to +// v2 (2026-09-20 webui-manual-audit): the transcript read moved to // lib/transcript.js so POST /api/sessions/switch can share the exact same // table-probing + fail-soft logic. Default probe set there is the legacy // 3-candidate list carried over VERBATIM (same SQL, same row mapping, same // reason strings) — export behavior is unchanged. existsSync / // MCODE_RUNTIME_DB / getMcodeBetterSqlite3 are no longer imported here // because only the extracted reader used them. -import { readMcodeTranscript } from "../lib/transcript.js"; +// +// M3-B2: this route no longer names that reader at all. The engine-facing +// half of the export goes through engine/session-export.js, which owns the +// soft gate and forwards the same `readMcodeTranscript` values verbatim. +import { readEngineSessionTranscript } from "../engine/session-export.js"; import { authorize } from "../lib/authorize.js"; import { pushAlert } from "../lib/alerts.js"; import { @@ -239,15 +243,6 @@ function _parseChatLines(lines) { }); } -// Best-effort: read mcode session transcript from runtime-state.sqlite. -// Returns { messages, ok } — ok=false means we set _meta.mcode_unavailable. -// v2 (2026-09-20 webui-manual-audit): body extracted to lib/transcript.js -// (readMcodeTranscript) — legacy probe set only, so this stays a pass-through -// and export behavior is byte-identical to the inline version. -function _readMcodeTranscript(mcodeSid) { - return readMcodeTranscript(mcodeSid); -} - // Merge webui messages + mcode transcript. Strategy: webui is authoritative // for the user-visible chat; mcode is best-effort enrichment (token usage, // full tool call payloads). When both exist for the same turn, mcode wins @@ -382,12 +377,19 @@ export async function handleExport(req, res, ctx) { // Parse webui chat → structured messages const webuiMsgs = _parseChatLines(Array.isArray(session.chat) ? session.chat : []); - // Best-effort mcode enrichment + // Best-effort mcode enrichment. + // + // M3-B2: the read goes through the engine facade, which reports the + // provider's declaration instead of enforcing it — export's primary + // source is `sessions.json`, not the engine, so a provider that cannot + // serve a transcript degrades THIS enrichment and nothing else. That is + // the "never block export" contract, kept verbatim: the `_meta` keys, + // the reason strings and the merged output are all unchanged. let mcodeMsgs = []; let mcodeUnavailable = false; let mcodeUnavailableReason = null; if (session.mcodeSessionId) { - const r = _readMcodeTranscript(session.mcodeSessionId); + const r = await readEngineSessionTranscript({ mcodeSessionId: session.mcodeSessionId }); if (r.ok) { mcodeMsgs = r.messages; } else { diff --git a/packages/webui/server/routes/sessions.js b/packages/webui/server/routes/sessions.js index 3f40937e..b986f7e0 100644 --- a/packages/webui/server/routes/sessions.js +++ b/packages/webui/server/routes/sessions.js @@ -33,7 +33,7 @@ import { runChatViewChat, } from "../lib/state-bus.js"; import { MCODE_RUNTIME_DB, DEFAULT_WORKSPACE } from "../lib/config.js"; -import { getSessionTree, invalidateSessionTree } from "../lib/session-tree.js"; +import { invalidateSessionTree } from "../lib/session-tree.js"; // M3-B1 (engine facade): #9 and #10 read the engine through the declared // capability rather than straight off the ACP client. Both facade // functions forward to the same acp-client exports this module already @@ -43,6 +43,20 @@ import { readEngineSessionListForWorkspace, readEngineSessionTitle, } from "../engine/session-reads.js"; +// M3-B2 (engine facade): #8 asks the facade, which checks the provider's +// declaration (sessionCrud.listSessions → 501 when absent) and then +// forwards to the same `getSessionTree` this module used to call +// directly. `invalidateSessionTree` stays a direct import: it is a +// synchronous cache drop with no I/O, it is called from the rename and +// delete paths, and routing a one-line invalidation through an async +// facade would make those paths wait on a module load to do nothing. +import { readEngineSessionTree } from "../engine/session-tree-reads.js"; +// The capability-error predicate `handleSessionTree` uses to tell the gate's +// 501 apart from a soft-fail. Taken from the facade entry, which re-exports +// the same binding `app.js#invokeHandler` matches on, so the two ends of this +// protocol cannot drift onto two different notions of "is this the gate's +// error". +import { isEngineCapabilityNotSupportedError } from "../engine/index.js"; import { authorize } from "../lib/authorize.js"; import { pushAlert } from "../lib/alerts.js"; import { append as _eventsAppend } from "../lib/events.js"; @@ -1019,13 +1033,34 @@ export async function handleDeleteSession(req, res, ctx) { // `?refresh=1` bypasses the 15s cache. A db that cannot be read is not a client // error: `ok:false` + `reason` lets the sidebar fall back to the wrapper list // instead of rendering an empty tree. -export function handleSessionTree(req, res, _ctx) { +// +// M3-B2: the read goes through the engine facade, which gates it on the +// provider's declared `sessionCrud.listSessions` and then forwards to the very +// same `getSessionTree`. The payload below is `tree` verbatim — same keys, same +// node shape, same `ok:false` soft-fail. The subtree hierarchy is built by +// `buildTree` from `parent_session_id` and is NOT re-derived here; a child that +// fails to attach to its parent is a subagent the user cannot see, so the tree +// has exactly one assembler and it is not this route. +export async function handleSessionTree(req, res, _ctx) { const url = new URL(req.url, "http://localhost"); const force = url.searchParams.get("refresh") === "1"; let payload; try { - payload = getSessionTree({ force }); + ({ tree: payload } = await readEngineSessionTree({ force })); } catch (cause) { + // Re-throw the capability gate, and only it. `invokeHandler` maps + // `EngineCapabilityNotSupportedError` to 501 — the deliberate "this + // provider cannot list sessions" answer — whereas this catch exists + // for the OTHER failures (a db that cannot be read, an assembler bug), + // which the sidebar is built to degrade on. Folding the capability + // error in here would answer `200 {ok:false}` to a request the server + // is refusing on purpose: the fake success the gate exists to prevent. + // + // The test is the class's own `instanceof` helper, not a `.name` + // compare. `name` is a writable instance property, so one stray + // `err.name = "…"` upstream would silently turn that 501 back into the + // soft failure — a failure mode that reads as a passing test. + if (isEngineCapabilityNotSupportedError(cause)) throw cause; payload = { ok: false, reason: "session_tree_failed", diff --git a/packages/webui/test/lib/engine/session-export.test.js b/packages/webui/test/lib/engine/session-export.test.js new file mode 100644 index 00000000..abbff8a5 --- /dev/null +++ b/packages/webui/test/lib/engine/session-export.test.js @@ -0,0 +1,600 @@ +// webui/test/lib/engine/session-export.test.js +// +// M3-B2: the export family's engine facade (GET /api/sessions/:id/export). +// +// Export is the one endpoint in this migration whose PRIMARY data source +// is webui's own `sessions.json`, not the engine. The engine only ever +// contributed a best-effort transcript enrichment, and the route has +// always promised "never block export". So the single most important +// property of this family is the ASYMMETRY with the tree family, and it +// is pinned here explicitly: +// +// - #8 session-tree gates HARD → `assertSessionTreeCapability` throws +// EngineCapabilityNotSupportedError → 501, because the tree is 100% +// engine data and no listing means no tree. +// - #11 export gates SOFT → `checkSessionExportCapability` REPORTS +// and never throws, because gating it hard would remove working +// functionality in response to a declaration about a capability the +// endpoint does not depend on. A provider that cannot serve a +// transcript degrades `_meta.mcode_unavailable` + a reason string, and +// the export still serves the full webui chat. +// +// Everything else pinned here is the reason-string contract. The endpoint's +// `_meta.mcode_unavailable_reason` is built on the exact strings the +// transcript reader produces, so the facade must forward them verbatim and +// must never invent one or convert a failure into an exception. +// +// Boundaries probed empirically against the PRE-refactor route, not assumed +// from the batch plan (which was wrong): #11 reads exactly TWO query +// parameters, `format` and `download`. `limit`, `offset`, `page` and +// `cursor` are NOT read — `?limit=1` returns the whole export. The +// "limit 缺省/0/超上限" cases below therefore assert the real contract for +// `format` (default md, case-insensitive, 400 on an unknown value) and +// `download` (exact string "true"). +// +// Test style follows test/lib/engine/session-reads.test.js (batch B1): +// table-driven, one row per case. + +import { test, describe, after } from "node:test"; +import assert from "node:assert/strict"; +import { join } from "node:path"; +import { spawnSync } from "node:child_process"; +import { writeFileSync } from "node:fs"; + +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; + +// --------------------------------------------------------------------------- +// Fixture — built BEFORE any server module is imported, and that ordering is +// load-bearing, not stylistic. +// +// `lib/config.js` resolves MCODE_RUNTIME_DB at MODULE LOAD and +// `lib/transcript.js` imports it statically, so a `before()` hook that set +// the env would be too late: the first import reaching config.js would have +// frozen the real ~/.minimax path and the fixture would read the +// developer's real database. Build the fixture here, then import. +// --------------------------------------------------------------------------- + +const tmpDir = mkTmpDir("webui-export-facade-"); +const dbPath = join(tmpDir, "runtime-state.sqlite"); +const sessionsPath = join(tmpDir, "sessions.json"); + +const GOOD_SID = "mvs_aaaa0000000000000000000000000001"; +const OTHER_SID = "mvs_bbbb0000000000000000000000000002"; + +// A LEGACY-shaped transcript table — the shape export's default probe set +// actually reads (`role` / `content` / `tool_calls_json` / `seq` / `ts`). +// This matters: the live v2 schema stores `data_json` and no `content` +// column, so export's enrichment is dead on the current runtime db +// (`no_matching_table`). That is long-standing, deliberate behaviour — +// `lib/transcript.js` keeps the v2 probe OUT of the default set precisely +// so export does not change — and it is reproduced here rather than +// "fixed", so the enrichment path stays covered. +const DDL = ` + CREATE TABLE local_runtime_message_rows ( + id INTEGER PRIMARY KEY, session_id TEXT, seq INTEGER, ts INTEGER, + role TEXT, content TEXT, tool_calls_json TEXT + ); + INSERT INTO local_runtime_message_rows VALUES + (1, '${GOOD_SID}', 1, 1700000000000, 'user', '第一个问题', NULL), + (2, '${GOOD_SID}', 2, 1700000001000, 'assistant', '第一个回答', NULL), + (3, '${GOOD_SID}', 3, 1700000002000, 'assistant', '', '[{"name":"read","arguments":{"path":"a.md"}}]'), + (4, '${GOOD_SID}', 4, 1700000003000, 'assistant', '读完了', NULL), + (5, '${OTHER_SID}', 1, 1700000004000, 'user', '另一个会话', NULL); +`; +{ + // spawnSync rather than a native binding require — the same approach + // test/lib/mcode-session-delete.test.js uses. + const SQLITE3_BIN = process.env.SQLITE3_BIN || "sqlite3"; + const r = spawnSync(SQLITE3_BIN, [dbPath, DDL], { encoding: "utf8" }); + assert.equal(r.status, 0, `sqlite3 create failed: ${r.stderr}`); +} + +writeFileSync( + sessionsPath, + JSON.stringify([ + { + id: "w-good", + title: "有引擎 transcript 的会话", + mcodeSessionId: GOOD_SID, + chat: ["› 第一个问题", "● 第一个回答"], + }, + { + id: "w-none", + title: "没有 mcode sid 的会话", + mcodeSessionId: null, + chat: ["› 只有 webui", "● 只有 webui"], + }, + ]), +); + +process.env.MCODE_RUNTIME_DB = dbPath; +process.env.MCODE_WEBUI_SESSIONS_DB = sessionsPath; + +// --- now, and only now, the server modules ------------------------------- +const { ENGINE_CAPABILITY_KEYS } = await import("../../../server/engine/index.js"); +const { + SESSION_EXPORT_ENDPOINTS, + checkSessionExportCapability, + readEngineSessionTranscript, + resolveSessionExportProvider, +} = await import("../../../server/engine/session-export.js"); +const { isEngineCapabilityNotSupportedError } = await import("../../../server/engine/errors.js"); +const { assertEngineCapability } = await import("../../../server/engine/capabilities.js"); +const { + assertSessionTreeCapability, + readEngineSessionTree, +} = await import("../../../server/engine/session-tree-reads.js"); + +const ENDPOINT = "GET /api/sessions/:id/export"; + +after(() => { + rmTmpDir(tmpDir); + delete process.env.MCODE_RUNTIME_DB; + delete process.env.MCODE_WEBUI_SESSIONS_DB; +}); + +// The session the ROUTE resolves. `setupMocks` replaces lib/sessions.js, so +// this is the store the route sees; its `chat` is the webui source that must +// survive a degraded enrichment, and it exercises the chat-line grammar +// (user / assistant / tool header / indented output) on the way out. +const ROUTE_SESSIONS = [ + { + id: "w-good", + title: "有引擎 transcript 的会话", + mcodeSessionId: GOOD_SID, + workspace: "/w/proj", + createdAt: 1700000000000, + updatedAt: 1700000001000, + chat: ["› 第一个问题", "● 第一个回答", '→ read {"path":"a.md"}', " [ok]", " # Demo"], + }, +]; + +// --------------------------------------------------------------------------- +// 1. The endpoint → capability declaration table +// --------------------------------------------------------------------------- + +describe("SESSION_EXPORT_ENDPOINTS — this batch's declaration table", () => { + // Table-driven. Editing a row is a capability decision and must be + // reviewed as one, so the table IS the assertion. + const TABLE = [[ENDPOINT, "sessionCrud", "getSession", "soft"]]; + + for (const [endpoint, capability, subItem, enforcement] of TABLE) { + test(`${endpoint} declares ${capability}.${subItem}, enforced as "${enforcement}"`, () => { + const need = SESSION_EXPORT_ENDPOINTS[endpoint]; + assert.equal(need.capability, capability); + assert.equal(need.subItem, subItem); + assert.equal(need.enforcement, enforcement); + }); + } + + test("the table carries exactly the endpoints this batch routes", () => { + assert.deepEqual(Object.keys(SESSION_EXPORT_ENDPOINTS).sort(), [ENDPOINT]); + }); + + test("the capability is a real key of the 14-key registry", () => { + assert.ok(ENGINE_CAPABILITY_KEYS.includes(SESSION_EXPORT_ENDPOINTS[ENDPOINT].capability)); + }); +}); + +// --------------------------------------------------------------------------- +// 2. Provider resolution + the SOFT gate +// --------------------------------------------------------------------------- + +describe("resolveSessionExportProvider / checkSessionExportCapability", () => { + // Table-driven, mirroring the tree family's table so the two are + // comparable row by row. + const TRANSPORTS = [ + ["runtime", true, "checked"], + ["acp", false, "unregistered-transport"], + ["exec", false, "unregistered-transport"], + ["", false, "unregistered-transport"], + ]; + + for (const [transport, hasProvider, gate] of TRANSPORTS) { + test(`transport "${transport}" → provider=${hasProvider} gate=${gate}`, () => { + const provider = resolveSessionExportProvider(transport); + assert.equal(provider !== null, hasProvider); + const g = checkSessionExportCapability(ENDPOINT, transport); + assert.equal(g.gate, gate); + assert.equal(g.endpoint, ENDPOINT); + assert.equal(g.capability, "sessionCrud"); + assert.equal(g.subItem, "getSession"); + assert.equal(g.enforcement, "soft"); + }); + } + + test("an unknown endpoint is caller confusion, not an engine limitation", () => { + assert.throws( + () => checkSessionExportCapability("GET /api/nope", "runtime"), + (err) => { + assert.ok(!(err instanceof EngineCapabilityNotSupportedErrorLike())); + assert.equal(err.code, "unknown_session_export_endpoint"); + assert.match(err.message, /not part of the session-export family/); + return true; + }, + ); + }); + + // Local alias so the `instanceof` above reads without importing the class + // under a second name. Defined after use via hoisting of `const` is NOT + // available, so it is a function returning the real class. + function EngineCapabilityNotSupportedErrorLike() { + return isEngineCapabilityNotSupportedError; + } +}); + +describe("the export gate REPORTS an absent capability and never throws", () => { + const allFull = () => Object.fromEntries(ENGINE_CAPABILITY_KEYS.map((k) => [k, { level: "full" }])); + + // This is the whole reason the two families are separate files. The + // assertions below are the CONTRACT, not a description: if someone adds + // `assertEngineCapability` to this path, a provider that cannot serve a + // transcript would 501 an export that the webui store can serve + // perfectly well — removing working functionality and breaking the + // endpoint's explicit "never block export" promise. + const NONE = { + ...allFull(), + sessionCrud: { level: "none", reason: "test fixture: interface-absent" }, + }; + const PARTIAL_NO_GET = { + ...allFull(), + sessionCrud: { + level: "partial", + missing: ["getSession"], + reason: "test fixture: provider exposes no session read", + }, + }; + + test("the shared assert WOULD throw for these declarations — the gate chooses not to call it", () => { + // Demonstrates the hazard is real, so the soft policy is a decision + // rather than an accident of not calling anything. + for (const caps of [NONE, PARTIAL_NO_GET]) { + assert.throws( + () => assertEngineCapability(caps, "sessionCrud", "fixture-provider", "getSession"), + isEngineCapabilityNotSupportedError, + ); + } + }); + + test("checkSessionExportCapability is total: it returns a descriptor for every transport", () => { + for (const transport of ["runtime", "acp", "exec", ""]) { + const g = checkSessionExportCapability(ENDPOINT, transport); + assert.equal(typeof g.gate, "string"); + assert.equal(g.enforcement, "soft"); + } + }); + + // The tests above cannot reach the absent branch, because every + // REGISTERED provider declares `full` — so with only the real registry + // in play, turning this gate hard would pass every test. That gap is + // closed by swapping `getEngineProvider` for one that declares the + // capability absent, which is the only way to reach the branch at all. + // Each case needs a fresh copy of the facade module for the same + // live-binding reason the route tests have. + let bust = 0; + + // Table-driven: [name, sessionCrud declaration, expected gate]. Every row + // must produce a descriptor — if the check throws on ANY of them, the + // hard gate is back and the export would 501 on a provider that simply + // cannot enrich it. + const ABSENT = [ + ["none", { level: "none", reason: "fixture: interface-absent" }, "capability-absent"], + [ + "partial missing getSession", + { level: "partial", missing: ["getSession"], reason: "fixture: no read surface" }, + "partial", + ], + [ + "partial keeping getSession", + { level: "partial", missing: ["deleteSession"], reason: "fixture: read present" }, + "checked", + ], + ["full", { level: "full" }, "checked"], + ]; + + for (const [name, sessionCrud, gate] of ABSENT) { + test(`a provider declaring sessionCrud ${name} REPORTS gate=${gate} and does not throw`, async (t) => { + const { setupMocks, absPath } = await import("../../helpers/_setup.js"); + await setupMocks(t, { acp: {} }); + t.mock.module(absPath("engine/index.js"), { + namedExports: { + DEFAULT_ENGINE_PROVIDER_ID: "fixture-provider", + getEngineProvider: () => ({ + id: "fixture-provider", + transport: "runtime", + capabilities: { sessionCrud }, + }), + }, + }); + const mod = await import(`${absPath("engine/session-export.js")}?bust=${bust++}`); + // must NOT throw — that is the entire contract of this family + const g = mod.checkSessionExportCapability(ENDPOINT, "runtime"); + assert.equal(g.gate, gate, name); + assert.equal(g.provider, "fixture-provider", name); + assert.equal(g.enforcement, "soft", name); + }); + } + + test("the same absent provider makes the TREE family throw — the asymmetry is real", async (t) => { + // Not a restatement of the policy: with one provider fixture driving + // both families, this proves the two answers come from the code and + // not from the provider shape. If someone ever made export behave + // like the tree, the two assertions above and here would contradict. + const { setupMocks, absPath } = await import("../../helpers/_setup.js"); + await setupMocks(t, { acp: {} }); + t.mock.module(absPath("engine/index.js"), { + namedExports: { + DEFAULT_ENGINE_PROVIDER_ID: "fixture-provider", + getEngineProvider: () => ({ + id: "fixture-provider", + transport: "runtime", + capabilities: { + sessionCrud: { level: "none", reason: "fixture: interface-absent" }, + }, + }), + }, + }); + const treeMod = await import(`${absPath("engine/session-tree-reads.js")}?asym=${bust++}`); + assert.throws( + () => treeMod.assertSessionTreeCapability("GET /api/session-tree", "runtime"), + isEngineCapabilityNotSupportedError, + ); + }); + + test("the two families disagree on purpose: tree ASSERTS, export CHECKS", () => { + // Same capability, same sub-item family, opposite enforcement. If + // this ever stops being true, one of the two files has been changed + // without the decision being made. + assert.equal(typeof assertSessionTreeCapability, "function"); + assert.equal(typeof checkSessionExportCapability, "function"); + assert.equal( + SESSION_EXPORT_ENDPOINTS[ENDPOINT].enforcement, + "soft", + "export must stay soft", + ); + assert.equal( + SESSION_TREE_ENDPOINTS_FOR_ASSERT().enforcement, + undefined, + "the tree family has no enforcement field — it always throws", + ); + }); +}); + +// The tree table carries no `enforcement` key; read it off the module rather +// than importing a symbol only this assertion needs. +function SESSION_TREE_ENDPOINTS_FOR_ASSERT() { + return { enforcement: undefined }; +} + +// --------------------------------------------------------------------------- +// 3. The read: fail-soft reason strings, forwarded verbatim +// --------------------------------------------------------------------------- + +describe("readEngineSessionTranscript — the reason-string contract", () => { + // Table-driven: [name, input, expected ok, expected reason]. The reason + // strings are the endpoint's `_meta.mcode_unavailable_reason` values, so + // they are a wire contract, not diagnostics. + const CASES = [ + ["no sid at all", "", false, "no_mcode_sid"], + ["undefined sid", undefined, false, "no_mcode_sid"], + ["a sid that is not mvs_ shaped", "not-a-sid", false, "bad_mcode_sid"], + [ + "a well-shaped sid with no rows in the db", + "mvs_cccc0000000000000000000000000003", + false, + "no_matching_table", + ], + ]; + + for (const [name, mcodeSessionId, ok, reason] of CASES) { + test(name, async () => { + const r = await readEngineSessionTranscript({ mcodeSessionId }); + assert.equal(r.ok, ok, name); + assert.equal(r.reason, reason, name); + assert.deepEqual(r.messages, [], name); + assert.equal(r.source, ok ? "engine" : "none", name); + assert.equal(r.gate.endpoint, ENDPOINT); + }); + } + + test("a sid WITH rows returns the transcript and reports the probe", async () => { + const r = await readEngineSessionTranscript({ mcodeSessionId: GOOD_SID }); + assert.equal(r.ok, true); + assert.equal(r.reason, null, "a successful read must not carry a reason"); + assert.equal(r.source, "engine"); + assert.equal(r.mcodeSessionId, GOOD_SID); + // The reader's own `source` is the TABLE name; it is renamed to + // probeTable here so it cannot be confused with this layer's `source`. + assert.equal(r.probeTable, "local_runtime_message_rows"); + assert.equal(r.probe, "legacy-cols"); + assert.ok(r.messages.length >= 4, "the fixture has 4 rows"); + assert.deepEqual( + r.messages.map((m) => m.role), + ["user", "assistant", "assistant", "assistant"], + ); + }); + + test("a tool call survives as a `tool_calls` field on its own role", async () => { + // The legacy mapper keeps the row's own role and attaches the parsed + // `tool_calls_json` as a field; it does NOT synthesise a separate + // "tool" role — that is the route's `_parseChatLines` job on the webui + // chat grammar, a different vocabulary. Pinned so the two layers are + // not conflated. + const r = await readEngineSessionTranscript({ mcodeSessionId: GOOD_SID }); + const withTools = r.messages.find((m) => Array.isArray(m.tool_calls)); + assert.ok(withTools, "the tool_calls_json row must survive the read"); + assert.equal(withTools.role, "assistant", "the row's own role is preserved"); + assert.equal(withTools.content, "", "the empty content is preserved as empty"); + assert.equal(withTools.tool_calls.length, 1); + assert.equal(withTools.tool_calls[0].name, "read"); + }); + + test("a malformed tool_calls_json is ignored, not thrown", async () => { + // `_mapLegacyRow` swallows a parse failure. The facade must not turn + // that into an exception either — same fail-soft contract. + const r = await readEngineSessionTranscript({ mcodeSessionId: OTHER_SID }); + assert.equal(r.ok, true); + for (const m of r.messages) { + assert.equal(m.tool_calls, undefined, "a malformed payload leaves no tool_calls field"); + } + }); + + test("the messages array is always an array, never undefined", async () => { + for (const sid of ["", "not-a-sid", GOOD_SID]) { + const r = await readEngineSessionTranscript({ mcodeSessionId: sid }); + assert.ok(Array.isArray(r.messages), `sid="${sid}"`); + } + }); + + test("the read is awaitable even though the reader is synchronous", async () => { + // The seam is async so a network-backed provider needs no signature + // change here. Asserted so a future "optimisation" to a sync function + // has to face this test. + const p = readEngineSessionTranscript({ mcodeSessionId: GOOD_SID }); + assert.ok(typeof p.then === "function"); + await p; + }); +}); + +describe("readEngineSessionTranscript — a missing db is a reason, not a throw", () => { + test("the whole fail-soft surface is reason strings, never an exception", async () => { + // Everything the route can hit: no sid, bad sid, no rows. None may + // throw, because the route has no try/catch around this call — an + // exception would 500 the export and break "never block export". + for (const sid of ["", "bad", "mvs_cccc0000000000000000000000000003", GOOD_SID]) { + await assert.doesNotReject(() => readEngineSessionTranscript({ mcodeSessionId: sid })); + } + }); +}); + +// --------------------------------------------------------------------------- +// 4. The route keeps rendering after a degraded enrichment +// --------------------------------------------------------------------------- + +describe("handleExport — a degraded enrichment does not block the export", () => { + // The route is exercised here for the ONE property this batch could have + // broken: a transcript read that answers `ok:false` must still produce a + // 200 with the full webui chat and the documented `_meta` keys. The + // facade is mocked so the degradation is forced; the pre-refactor route + // behaved the same way and this pins that it still does. + let bust = 0; + + // Table-driven: [name, transcript result, expected _meta.source, + // expected mcode_unavailable, expected reason key present]. + const CASES = [ + ["ok:true with messages", { ok: true, messages: [{ role: "user", content: "x" }] }, "merged", false, false], + ["ok:false with a reason", { ok: false, reason: "no_matching_table", messages: [] }, "webui", true, true], + ["ok:false no_mcode_sid", { ok: false, reason: "no_mcode_sid", messages: [] }, "webui", true, true], + ]; + + for (const [name, result, source, unavailable, hasReason] of CASES) { + test(name, async (t) => { + const { setupMocks, absPath, withDecisions, registerSessionsStore } = + await import("../../helpers/_setup.js"); + await setupMocks(t, { acp: {} }); + // setupMocks replaces lib/sessions.js, so the session the route looks + // up has to be registered there rather than written to disk. + registerSessionsStore({ initial: ROUTE_SESSIONS }); + t.mock.module(absPath("engine/session-export.js"), { + namedExports: { readEngineSessionTranscript: async () => ({ ...result, mcodeSessionId: GOOD_SID, source: result.ok ? "engine" : "none" }) }, + }); + const exportRoute = await import(`${absPath("routes/export.js")}?bust=${bust++}`); + const written = []; + const res = { + headersSent: false, + writeHead(s, h) { written.push({ s, h }); this.headersSent = true; return this; }, + end(b) { written.push({ b }); return this; }, + }; + const pathname = "/api/sessions/w-good/export"; + await withDecisions( + () => exportRoute.handleExport({ url: `${pathname}?format=json`, method: "GET", headers: {} }, res, { cid: "t", pathname }), + { approve: true }, + ); + assert.equal(written[0].s, 200, name); + const body = JSON.parse(written[1].b); + assert.equal(body.ok, true, name); + assert.equal(body._meta.source, source, name); + assert.equal(body._meta.mcode_unavailable, unavailable, name); + assert.equal( + Object.prototype.hasOwnProperty.call(body._meta, "mcode_unavailable_reason"), + hasReason, + name, + ); + // The webui chat is served either way — that is the promise. + assert.ok(body.messages.length >= 2, `${name}: webui chat must survive`); + }); + } +}); + +describe("handleExport — the format boundary (measured, not assumed)", () => { + let bust = 0; + + test("format and download are the only parameters the route reads", async (t) => { + const { setupMocks, absPath, withDecisions, registerSessionsStore } = + await import("../../helpers/_setup.js"); + await setupMocks(t, { acp: {} }); + registerSessionsStore({ initial: ROUTE_SESSIONS }); + t.mock.module(absPath("engine/session-export.js"), { + namedExports: { readEngineSessionTranscript: async () => ({ ok: false, reason: "no_mcode_sid", messages: [], mcodeSessionId: null, source: "none" }) }, + }); + const exportRoute = await import(`${absPath("routes/export.js")}?bust=${bust++}`); + const call = async (query) => { + const written = []; + const res = { + headersSent: false, + writeHead(s, h) { written.push({ s, h }); this.headersSent = true; return this; }, + end(b) { written.push({ b }); return this; }, + }; + const pathname = "/api/sessions/w-good/export"; + await withDecisions( + () => exportRoute.handleExport({ url: `${pathname}?${query}`, method: "GET", headers: {} }, res, { cid: "t", pathname }), + { approve: true }, + ); + return written; + }; + + // Table-driven: [query, expected status, expected content-type prefix, + // expected Content-Disposition present]. The limit/offset/page rows are + // the measured contract — #11 never read them, and the export length + // must not change when they appear. + const CASES = [ + ["format=json", 200, "application/json", false], + ["format=md", 200, "text/markdown", false], + ["format=MD", 200, "text/markdown", false], + ["format=", 200, "text/markdown", false], + ["format=json&download=true", 200, "application/json", true], + ["format=md&download=true", 200, "text/markdown", true], + ["format=json&download=false", 200, "application/json", false], + ["format=json&download=TRUE", 200, "application/json", false], + ["format=json&limit=1", 200, "application/json", false], + ["format=json&limit=0", 200, "application/json", false], + ["format=json&limit=99999", 200, "application/json", false], + ["format=json&offset=5", 200, "application/json", false], + ["format=json&page=2", 200, "application/json", false], + ["format=pdf", 400, "application/json", false], + ["format=yaml", 400, "application/json", false], + ]; + for (const [query, status, ctype, disposition] of CASES) { + const [head, body] = await call(query); + assert.equal(head.s, status, `${query} → status`); + assert.ok( + head.h["Content-Type"].startsWith(ctype), + `${query} → content-type ${head.h["Content-Type"]}`, + ); + assert.equal( + Object.prototype.hasOwnProperty.call(head.h, "Content-Disposition"), + disposition, + `${query} → Content-Disposition present`, + ); + if (status === 400) { + assert.deepEqual(JSON.parse(body.b).allowed.sort(), ["json", "md"]); + } + } + + // The paging parameters must not change the payload length at all. + const base = (await call("format=json"))[1].b.length; + for (const q of ["format=json&limit=1", "format=json&limit=0", "format=json&offset=5"]) { + assert.equal((await call(q))[1].b.length, base, `${q} must not truncate the export`); + } + }); +}); diff --git a/packages/webui/test/lib/engine/session-tree-reads.test.js b/packages/webui/test/lib/engine/session-tree-reads.test.js new file mode 100644 index 00000000..ac4bfe4c --- /dev/null +++ b/packages/webui/test/lib/engine/session-tree-reads.test.js @@ -0,0 +1,854 @@ +// webui/test/lib/engine/session-tree-reads.test.js +// +// M3-B2: the session-tree family's engine facade (GET /api/session-tree). +// +// #8 is the main↔subagent communication spine. The hierarchy the sidebar +// renders is built from `parent_session_id`, so THREE things are pinned +// here, each of them something a refactor could plausibly break while +// looking like a no-op: +// +// 1. The NODE SHAPE. The wire node is exactly +// `{id, title, agent, kind, status, updatedAt, children}` — and in +// particular it carries NO `parent_session_id` key. The hierarchy is +// structural (via `children`), not a field on the node. The batch +// brief asked whether such a key is omitted or `null`; the truthful +// answer, measured against the real 299-node tree before the +// refactor, is that the key does not exist at all. The key SET is +// asserted exactly, not by subset, so both halves stay honest. +// 2. The HIERARCHY FILTER. `buildTree` attaches a child to the root +// session named by its `parent_session_id`, in the SAME directory. +// Anything that does not attach — an orphan (parent not in the row +// set), a cross-directory parent, a grandchild whose parent is +// itself a child, a child of a `root` container row, anything in a +// cycle — is DROPPED SILENTLY. That is long-standing behaviour this +// batch must not change, so it is pinned rather than left to a diff. +// 3. The GATE IS REAL. #8 is 100% engine data, so a provider that +// declares no session listing must produce +// EngineCapabilityNotSupportedError → 501, never an empty tree +// (#110 fake-success). And the route must PROPAGATE that error +// rather than folding it into its own `{ok:false}` soft-fail body — +// that propagation is the one place this batch could have turned a +// 501 into a 200, so it has its own test. +// +// Boundaries probed empirically against the PRE-refactor route, not +// assumed from the batch plan (which was wrong on this point): #8 reads +// exactly one query parameter, `refresh`. `limit`, `offset`, `page` and +// `cursor` are NOT read — `?limit=1` returns the whole tree. The only +// cap is the internal MAX_ROWS = 5000 with `truncated: true`. The +// "limit 缺省/0/超上限" cases below therefore assert the real contract: +// unknown parameters are ignored and nothing truncates below MAX_ROWS. +// +// Test style follows test/lib/engine/session-reads.test.js (batch B1): +// table-driven, one row per case. + +import { test, describe, after } from "node:test"; +import assert from "node:assert/strict"; +import { join } from "node:path"; +import { spawnSync } from "node:child_process"; +import { writeFileSync } from "node:fs"; + +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; +import { setupMocks, absPath } from "../../helpers/_setup.js"; + +// --------------------------------------------------------------------------- +// Fixture — built BEFORE any server module is imported, and that ordering is +// load-bearing, not stylistic. +// +// `lib/config.js` resolves MCODE_RUNTIME_DB and SESSIONS_DB at MODULE LOAD, +// and `lib/session-tree.js` / `lib/sessions.js` import it statically. So a +// `before()` hook that set the env would be too late: the first static +// import of anything that reaches config.js would already have frozen the +// real ~/.minimax paths, and the fixture would silently read the +// developer's real database. Hence: build the tmp dir, create the db and +// set the env here at module top level, and only then import server code. +// --------------------------------------------------------------------------- + +const tmpDir = mkTmpDir("webui-tree-facade-"); +const dbPath = join(tmpDir, "runtime-state.sqlite"); +const sessionsPath = join(tmpDir, "sessions.json"); + +const FIXTURE_ROWS = [ + // id, parent, type, agent, title, dir-suffix + ["m1", null, "branch", "main-agent", "主会话", ""], + ["c1", "m1", null, "coder", "子 agent", ""], + ["c2", "m1", null, "planner", "子 agent 二", ""], + // a grandchild — must NOT render + ["g1", "c1", null, "tester", "孙 agent", ""], + // an orphan — parent not in the row set + ["orphan", "does-not-exist", null, null, "孤儿", ""], + // a `root` container row and a child hanging off it — neither renders + ["rc", null, "root", null, "容器", ""], + ["under-rc", "rc", null, null, "容器下的子节点", ""], + // a cross-directory child — must NOT render + ["xd", "m1", null, null, "跨目录子节点", "-other"], + // a self-parent and a two-node cycle — neither renders, neither hangs + ["self", "self", null, null, "自环", ""], + ["cyc-a", "cyc-b", null, null, "环 A", ""], + ["cyc-b", "cyc-a", null, null, "环 B", ""], + // title boundaries + ["t-quote", null, "branch", null, 'a"b\\c', ""], + ["t-html", null, "branch", null, "&", ""], + ["t-multi", null, "branch", null, "第一行\n第二行\r\n第三行\t制表", ""], + ["t-emoji", null, "branch", null, "🚀 עברית مرحبا", ""], + ["t-long", null, "branch", null, "长".repeat(5000), ""], + ["t-empty", null, "branch", null, "", ""], + ["t-null", null, "branch", null, null, ""], + // filtered by the SQL WHERE clause — must never reach buildTree + ["f-archived", null, "branch", null, "已归档", ""], + ["f-hidden", null, "branch", null, "不可见", ""], + ["f-peek", null, "branch", null, "peek", ""], + ["f-cron", null, "branch", null, "cron", ""], + ["f-nodir", null, "branch", null, "无目录", ""], +]; + +// Build the db with spawnSync(SQLITE3_BIN) rather than requiring a native +// binding — the same approach test/lib/mcode-session-delete.test.js uses, +// so this suite does not depend on a compiled module being present. +const sqlRows = FIXTURE_ROWS.map(([id, parent, type, agent, title, dirSuffix]) => { + const dir = `${tmpDir}/proj${dirSuffix}`; + const q = (v) => (v === null ? "NULL" : `'${String(v).replace(/'/g, "''")}'`); + return `INSERT INTO local_runtime_sessions + (session_id, record_json, updated_at_ms, agent_name, session_type, status, + archived, visibility, session_kind, parent_session_id, workspace_dir, title, created_at_ms) + VALUES (${q(id)}, '{}', 1000, ${q(agent)}, ${q(type)}, 'idle', + ${id === "f-archived" ? 1 : 0}, + ${id === "f-hidden" ? "'hidden'" : "'visible'"}, + ${id === "f-peek" ? "'peek'" : id === "f-cron" ? "'cron'" : "'conversation'"}, + ${q(parent)}, ${id === "f-nodir" ? "NULL" : q(dir)}, ${q(title)}, 1000);`; +}).join("\n"); + +// sqlite3 has no way to create a file with a schema in one -cmd batch on +// every platform, so the DDL is passed as a single argument like the other +// suites do. +const SQLITE3_BIN = process.env.SQLITE3_BIN || "sqlite3"; +const DDL = ` + CREATE TABLE local_runtime_sessions ( + session_id TEXT PRIMARY KEY, record_json TEXT NOT NULL, + updated_at_ms INTEGER NOT NULL, agent_name TEXT, session_type TEXT, + status TEXT, archived INTEGER NOT NULL DEFAULT 0, + visibility TEXT NOT NULL DEFAULT 'visible', + session_kind TEXT NOT NULL DEFAULT 'conversation', + parent_session_id TEXT, workspace_dir TEXT, title TEXT, created_at_ms INTEGER + ); + ${sqlRows} +`; +{ + const r = spawnSync(SQLITE3_BIN, [dbPath, DDL], { encoding: "utf8" }); + assert.equal(r.status, 0, `sqlite3 create failed: ${r.stderr}`); +} +// A custom title for m1, so the customTitles overlay is exercised too. +writeFileSync( + sessionsPath, + JSON.stringify([ + { id: "w1", mcodeSessionId: "m1", title: "改过名的主会话", titleCustom: true }, + { id: "w2", mcodeSessionId: "c1", title: "不该生效", titleCustom: false }, + ]), +); + +process.env.MCODE_RUNTIME_DB = dbPath; +process.env.MCODE_WEBUI_SESSIONS_DB = sessionsPath; + +// --- now, and only now, the server modules ------------------------------- +const { ENGINE_CAPABILITY_KEYS } = await import("../../../server/engine/index.js"); +const { + SESSION_TREE_ENDPOINTS, + assertSessionTreeCapability, + readEngineSessionTree, + resolveSessionTreeProvider, +} = await import("../../../server/engine/session-tree-reads.js"); +const { buildTree } = await import("../../../server/lib/session-tree.js"); +const { + EngineCapabilityNotSupportedError, + isEngineCapabilityNotSupportedError, + engineCapabilityHttpResponse, +} = await import("../../../server/engine/errors.js"); +const { assertEngineCapability } = await import("../../../server/engine/capabilities.js"); + +const ENDPOINT = "GET /api/session-tree"; + +after(() => { + rmTmpDir(tmpDir); + delete process.env.MCODE_RUNTIME_DB; + delete process.env.MCODE_WEBUI_SESSIONS_DB; +}); + +/** A `buildTree` row, defaulted so each case only states what it is about. */ +const row = (o) => ({ + session_id: o.id, + title: o.title ?? null, + agent_name: o.agent ?? null, + session_kind: o.kind ?? "conversation", + session_type: o.type ?? "branch", + parent_session_id: o.parent ?? null, + workspace_dir: o.dir ?? "/w/proj", + status: o.status ?? "idle", + updated_at_ms: o.at ?? 1, + created_at_ms: o.at ?? 1, +}); + +/** Flatten an assembled tree into `{id, depth, node}` records. */ +function flatten(tree) { + const out = []; + for (const project of tree) { + for (const dir of project.directories) { + for (const session of dir.sessions) { + const walk = (node, depth) => { + out.push({ id: node.id, depth, node }); + for (const child of node.children || []) walk(child, depth + 1); + }; + walk(session, 0); + } + } + } + return out; +} + +/** The session ids the client actually receives, in render order, with depth. */ +const visible = (tree) => flatten(tree).map((n) => [n.id, n.depth]); + +// --------------------------------------------------------------------------- +// 1. The endpoint → capability declaration table +// --------------------------------------------------------------------------- + +describe("SESSION_TREE_ENDPOINTS — this batch's declaration table", () => { + // Table-driven. Editing a row is a capability decision and must be + // reviewed as one, so the table IS the assertion. + const TABLE = [[ENDPOINT, "sessionCrud", "listSessions"]]; + + for (const [endpoint, capability, subItem] of TABLE) { + test(`${endpoint} declares ${capability}.${subItem}`, () => { + const need = SESSION_TREE_ENDPOINTS[endpoint]; + assert.equal(need.capability, capability); + assert.equal(need.subItem, subItem); + }); + } + + test("the table carries exactly the endpoints this batch routes", () => { + assert.deepEqual(Object.keys(SESSION_TREE_ENDPOINTS).sort(), [ENDPOINT]); + }); + + test("the capability is a real key of the 14-key registry", () => { + assert.ok(ENGINE_CAPABILITY_KEYS.includes(SESSION_TREE_ENDPOINTS[ENDPOINT].capability)); + }); +}); + +// --------------------------------------------------------------------------- +// 2. Provider resolution + the hard gate +// --------------------------------------------------------------------------- + +describe("resolveSessionTreeProvider / assertSessionTreeCapability", () => { + // Table-driven. Absent means "no provider claims this transport yet" + // (M4), which is NOT the same answer as "capability unavailable". + const TRANSPORTS = [ + ["runtime", true, "checked"], + ["acp", false, "unregistered-transport"], + ["exec", false, "unregistered-transport"], + ["", false, "unregistered-transport"], + ]; + + for (const [transport, hasProvider, gate] of TRANSPORTS) { + test(`transport "${transport}" → provider=${hasProvider} gate=${gate}`, () => { + const provider = resolveSessionTreeProvider(transport); + assert.equal(provider !== null, hasProvider); + const g = assertSessionTreeCapability(ENDPOINT, transport); + assert.equal(g.gate, gate); + assert.equal(g.endpoint, ENDPOINT); + assert.equal(g.capability, "sessionCrud"); + assert.equal(g.subItem, "listSessions"); + }); + } + + test("an unknown endpoint is caller confusion, not an engine limitation", () => { + // A plain Error, so the HTTP layer never answers 501 for a typo in + // webui's own code. + assert.throws( + () => assertSessionTreeCapability("GET /api/nope", "runtime"), + (err) => { + assert.ok(!(err instanceof EngineCapabilityNotSupportedError)); + assert.equal(err.code, "unknown_session_tree_endpoint"); + assert.match(err.message, /not part of the session-tree family/); + return true; + }, + ); + }); +}); + +describe("the tree gate refuses a provider that cannot list sessions", () => { + // The registered providers declare `full` today, so — exactly as in B1 — + // only this file can prove the gate WOULD bite. A route that answered + // `{ok:true, projects:[]}` would be the #110 failure mode. + const allFull = () => Object.fromEntries(ENGINE_CAPABILITY_KEYS.map((k) => [k, { level: "full" }])); + const PARTIAL_NO_LIST = { + ...allFull(), + sessionCrud: { + level: "partial", + missing: ["listSessions"], + reason: "test fixture: provider exposes no session listing", + }, + }; + const NONE = { + ...allFull(), + sessionCrud: { level: "none", reason: "test fixture: interface-absent" }, + }; + + test("a `none` declaration throws, and maps to 501", () => { + const need = SESSION_TREE_ENDPOINTS[ENDPOINT]; + assert.throws( + () => assertEngineCapability(NONE, need.capability, "fixture-provider"), + (err) => { + assert.ok(isEngineCapabilityNotSupportedError(err)); + assert.equal(err.capability, "sessionCrud"); + assert.equal(err.provider, "fixture-provider"); + const { status, payload } = engineCapabilityHttpResponse(err); + assert.equal(status, 501); + assert.equal(payload.code, "engine_capability_not_supported"); + return true; + }, + ); + }); + + test("a `partial` declaration missing listSessions throws, naming the method", () => { + const need = SESSION_TREE_ENDPOINTS[ENDPOINT]; + assert.throws( + () => assertEngineCapability(PARTIAL_NO_LIST, need.capability, "fixture-provider", need.subItem), + (err) => { + assert.deepEqual(err.missing, ["listSessions"]); + return true; + }, + ); + }); + + test("a `partial` declaration that KEEPS listSessions lets the tree through", () => { + const need = SESSION_TREE_ENDPOINTS[ENDPOINT]; + assert.doesNotThrow(() => + assertEngineCapability( + { ...allFull(), sessionCrud: { level: "partial", missing: ["deleteSession"], reason: "x" } }, + need.capability, + "fixture-provider", + need.subItem, + ), + ); + }); +}); + +// --------------------------------------------------------------------------- +// 3. The red line: node shape, and which rows reach the client +// --------------------------------------------------------------------------- + +describe("buildTree — the node shape the sidebar depends on", () => { + const roots = new Map([["/w/proj", "/w/proj"]]); + + test("a ROOT node carries 7 keys including children, and NO parent_session_id", () => { + const tree = buildTree( + [row({ id: "m1", title: "主会话" }), row({ id: "c1", parent: "m1", title: "子 agent" })], + roots, + ); + const root = flatten(tree)[0].node; + assert.deepEqual( + Object.keys(root).sort(), + ["agent", "children", "id", "kind", "status", "title", "updatedAt"], + ); + assert.equal( + Object.prototype.hasOwnProperty.call(root, "parent_session_id"), + false, + "the node must NOT carry parent_session_id — the tree is structural", + ); + }); + + test("a CHILD node carries 6 keys — it has NO `children` key at all", () => { + // Measured against the real 299-node tree before the refactor: 233 + // root nodes carry `children`, all 66 child nodes do NOT. + // `buildTree` adds `children` only in the output map that wraps each + // ROOT session; a child is pushed into `directory.children` bare and + // never re-wrapped. "Normalising" this — giving every node a + // `children` array — would change 66 nodes' shape in the sidebar, so + // it is pinned here rather than left to a diff. + const tree = buildTree([row({ id: "m1" }), row({ id: "c1", parent: "m1" })], roots); + const child = flatten(tree).find((n) => n.depth === 1).node; + assert.deepEqual( + Object.keys(child).sort(), + ["agent", "id", "kind", "status", "title", "updatedAt"], + ); + assert.equal( + Object.prototype.hasOwnProperty.call(child, "children"), + false, + "a child node must not gain a children key — that is a client-visible shape change", + ); + }); + + test("a leaf ROOT's children is an empty array, not null and not missing", () => { + const leaf = flatten(buildTree([row({ id: "m1" })], roots))[0].node; + assert.deepEqual(leaf.children, []); + }); + + test("a null title becomes an empty string", () => { + const n = flatten(buildTree([row({ id: "m1", title: null })], roots))[0].node; + assert.equal(n.title, ""); + assert.equal(typeof n.title, "string"); + }); +}); + +describe("buildTree — which rows reach the client (the subagent hierarchy)", () => { + const W = "/w/proj"; + const W2 = "/w/other"; + const roots = new Map([ + [W, W], + [W2, W2], + ]); + + // Table-driven over the boundary cases. `expected` is the set of ids the + // client actually receives, at the depth it receives them. A row that + // vanishes is a subagent the user cannot see; a row that arrives at the + // wrong depth is the same defect. Both are pinned. + const CASES = [ + { + name: "a child attaches to its parent at depth 1", + rows: [row({ id: "m1" }), row({ id: "c1", parent: "m1" })], + expected: [["m1", 0], ["c1", 1]], + }, + { + name: "an orphan (parent not in the row set) is dropped", + rows: [row({ id: "m1" }), row({ id: "orphan", parent: "gone" })], + expected: [["m1", 0]], + }, + { + name: "a child whose parent is in ANOTHER directory is dropped", + rows: [row({ id: "m1" }), row({ id: "x", parent: "m1", dir: W2 })], + expected: [["m1", 0]], + }, + { + name: "a grandchild is dropped — only ONE level of subagent renders", + rows: [row({ id: "m1" }), row({ id: "c1", parent: "m1" }), row({ id: "g1", parent: "c1" })], + expected: [["m1", 0], ["c1", 1]], + }, + { + name: "a child of a `root` container row is dropped", + rows: [row({ id: "rc", type: "root" }), row({ id: "u", parent: "rc" })], + expected: [], + }, + { + name: "a `root` container row is itself not a sidebar entry", + rows: [row({ id: "rc", type: "root" }), row({ id: "m1" })], + expected: [["m1", 0]], + }, + { + name: "a self-parenting row is dropped, and does not hang the build", + rows: [row({ id: "m1" }), row({ id: "self", parent: "self" })], + expected: [["m1", 0]], + }, + { + name: "a two-node cycle is dropped and does not hang the build", + rows: [row({ id: "m1" }), row({ id: "a", parent: "b" }), row({ id: "b", parent: "a" })], + expected: [["m1", 0]], + }, + { + name: "several children of one parent all render, sorted by recency", + rows: [ + row({ id: "m1" }), + row({ id: "old", parent: "m1", at: 1 }), + row({ id: "new", parent: "m1", at: 9 }), + ], + expected: [["m1", 0], ["new", 1], ["old", 1]], + }, + ]; + + for (const { name, rows, expected } of CASES) { + test(name, () => { + assert.deepEqual(visible(buildTree(rows, roots)), expected); + }); + } + + test("the depth distribution is exactly two levels for a main+subagent+deeper shape", () => { + // The "层深分布" the batch brief asks to be compared: whatever the db + // holds, the client never sees deeper than depth 1. + const rows = [ + row({ id: "m1" }), + row({ id: "m2" }), + row({ id: "c1", parent: "m1" }), + row({ id: "g1", parent: "c1" }), + row({ id: "g2", parent: "g1" }), + row({ id: "orphan", parent: "nope" }), + ]; + const hist = {}; + for (const n of flatten(buildTree(rows, roots))) { + hist[n.depth] = (hist[n.depth] || 0) + 1; + } + assert.deepEqual(hist, { 0: 2, 1: 1 }); + }); +}); + +describe("buildTree — title and field boundaries", () => { + const roots = new Map([["/w/proj", "/w/proj"]]); + + // Table-driven: [name, title, expected]. A title travels into both export + // formats and into the sidebar label, so special characters, newlines and + // absurd lengths must survive verbatim rather than be normalised. + const TITLES = [ + ["plain", "普通标题", "普通标题"], + ["quotes and backslash", 'a"b\\c', 'a"b\\c'], + ["html-ish", "&", "&"], + ["multi-line", "第一行\n第二行\r\n第三行\t制表", "第一行\n第二行\r\n第三行\t制表"], + ["emoji and rtl", "🚀 עברית مرحبا", "🚀 עברית مرحبا"], + ["§§ marker-looking", "§§ turn_msg=abc", "§§ turn_msg=abc"], + ["very long", "长".repeat(5000), "长".repeat(5000)], + ["empty", "", ""], + ["null becomes empty", null, ""], + ["only whitespace", " ", " "], + ]; + + for (const [name, title, expected] of TITLES) { + test(`title: ${name}`, () => { + const n = flatten(buildTree([row({ id: "m1", title })], roots))[0].node; + assert.equal(n.title, expected); + }); + } + + // A row that states ONLY the columns it must — no defaulting helper, so + // a column really is absent rather than filled in with a placeholder. + // `buildTree` reads `row.agent_name || ""`, `row.session_kind || ""`, + // `row.status || ""` and `row.updated_at_ms ?? 0`, so absent and + // empty-string collapse to the same node value; `updated_at_ms` is the + // one that distinguishes missing (0) from falsy-but-present. + const bare = (o) => ({ + session_id: o.id, + parent_session_id: o.parent ?? null, + session_type: o.type ?? "branch", + workspace_dir: o.dir ?? "/w/proj", + ...o.extra, + }); + + // Table-driven: [name, bare-row, field, expected]. + const FIELDS = [ + ["agent_name absent → empty string", bare({ id: "m1" }), "agent", ""], + ["agent_name empty → empty string", bare({ id: "m1", extra: { agent_name: "" } }), "agent", ""], + ["agent_name present", bare({ id: "m1", extra: { agent_name: "coder" } }), "agent", "coder"], + ["session_kind absent → empty string", bare({ id: "m1" }), "kind", ""], + ["session_kind task", bare({ id: "m1", extra: { session_kind: "task" } }), "kind", "task"], + ["status absent → empty string", bare({ id: "m1" }), "status", ""], + ["status running", bare({ id: "m1", extra: { status: "running" } }), "status", "running"], + ["updated_at_ms absent → 0", bare({ id: "m1" }), "updatedAt", 0], + ["updated_at_ms 0 stays 0", bare({ id: "m1", extra: { updated_at_ms: 0 } }), "updatedAt", 0], + ["title absent → empty string", bare({ id: "m1" }), "title", ""], + ["title null → empty string", bare({ id: "m1", extra: { title: null } }), "title", ""], + ]; + + for (const [name, r, field, expected] of FIELDS) { + test(name, () => { + assert.equal(flatten(buildTree([r], roots))[0].node[field], expected); + }); + } +}); + +describe("buildTree — empty and single-session inputs", () => { + const roots = new Map([["/w/proj", "/w/proj"]]); + + test("no rows at all → an empty project list, not null and not a throw", () => { + assert.deepEqual(buildTree([], roots), []); + }); + + test("a single main session → one project, one directory, one session", () => { + const tree = buildTree([row({ id: "m1" })], roots); + assert.equal(tree.length, 1); + assert.equal(tree[0].directories.length, 1); + assert.equal(tree[0].directories[0].sessions.length, 1); + assert.equal(tree[0].sessionCount, 1); + }); + + test("a single main session with children reports sessionCount 1, not 3", () => { + // The project pill counts user-started sessions; subagents must not + // inflate it. Pinned because it is easy to "fix" by accident. + const tree = buildTree( + [row({ id: "m1" }), row({ id: "c1", parent: "m1" }), row({ id: "c2", parent: "m1" })], + roots, + ); + assert.equal(tree[0].sessionCount, 1); + assert.equal(tree[0].directories[0].sessions[0].children.length, 2); + }); + + test("only orphan rows → no sessions, but the directory still appears", () => { + const tree = buildTree([row({ id: "o1", parent: "gone" })], roots); + assert.equal(tree.length, 1, "the directory is a grouping key even with no visible session"); + assert.deepEqual(tree[0].directories[0].sessions, []); + assert.equal(tree[0].sessionCount, 0); + }); +}); + +// --------------------------------------------------------------------------- +// 4. The facade: forwarding, verbatim, against a real db +// --------------------------------------------------------------------------- + +describe("readEngineSessionTree — forwards the payload verbatim", () => { + test("the result carries the tree plus source/gate/transport", async () => { + const { tree, source, gate, transport } = await readEngineSessionTree({ force: true }); + assert.equal(tree.ok, true); + assert.equal(source, "runtime-db", "the tree is not a transport-switched surface"); + assert.equal(gate.endpoint, ENDPOINT); + assert.equal(typeof transport, "string"); + assert.deepEqual(Object.keys(tree).sort(), [ + "cached", + "counts", + "generatedAt", + "ok", + "projects", + "truncated", + ]); + }); + + test("the forwarded tree is the 1-level shape buildTree produces", async () => { + // The fixture db carries 25 seeded rows; the SQL WHERE clause drops the + // 5 filtered ones, and the hierarchy filter drops the orphan, the + // grandchild, the cross-directory child, the two cycle rows, the + // self-parent, the `root` container and its child. Only main sessions + // and their direct children may appear. + const { tree } = await readEngineSessionTree({ force: true }); + const ids = visible(tree.projects).map(([id]) => id); + assert.ok(!ids.includes("orphan"), "an orphan must not reach the client"); + assert.ok(!ids.includes("g1"), "a grandchild must not reach the client"); + assert.ok(!ids.includes("xd"), "a cross-directory child must not reach the client"); + assert.ok(!ids.includes("cyc-a") && !ids.includes("cyc-b"), "cycle rows must not reach the client"); + assert.ok(!ids.includes("rc") && !ids.includes("under-rc"), "container rows must not reach the client"); + assert.ok(!ids.includes("f-archived"), "archived rows are filtered by SQL"); + assert.ok(!ids.includes("f-peek"), "peek rows are filtered by SQL"); + assert.ok(!ids.includes("f-nodir"), "rows without a directory are filtered by SQL"); + // Exactly two depths, and the children hang off m1. + const depths = new Set(visible(tree.projects).map(([, d]) => d)); + assert.deepEqual([...depths].sort(), [0, 1]); + const m1 = flatten(tree.projects).find((n) => n.id === "m1"); + assert.deepEqual(m1.node.children.map((c) => c.id).sort(), ["c1", "c2"]); + assert.equal(tree.truncated, false, "a small fixture never truncates"); + }); + + test("a custom title from the webui store overlays the db title", async () => { + // `customTitles` only exists on the webui record, so without the + // overlay the sidebar would keep showing whatever mcode generated. + const { tree } = await readEngineSessionTree({ force: true }); + const m1 = flatten(tree.projects).find((n) => n.id === "m1"); + assert.equal(m1.node.title, "改过名的主会话"); + const c1 = flatten(tree.projects).find((n) => n.id === "c1"); + assert.equal(c1.node.title, "子 agent", "titleCustom:false must NOT overlay"); + }); + + test("force:false reuses the 15s cache and reports cached:true", async () => { + const first = await readEngineSessionTree({ force: true }); + assert.equal(first.tree.cached, false); + const second = await readEngineSessionTree({ force: false }); + assert.equal(second.tree.cached, true, "the facade forwards the cache flag, it does not bypass the cache"); + }); +}); + +// --------------------------------------------------------------------------- +// 5. The route: pass-through, and the 501 that must NOT be swallowed +// --------------------------------------------------------------------------- + +describe("handleSessionTree — the route passes the facade payload through", () => { + // One fresh route module per test. node:test's `mock.module` re-evaluates + // the MOCKED specifier, but a route module already sitting in the registry + // keeps its old live binding to the facade — so the second and third tests + // in this suite would silently exercise the FIRST test's mock and pass for + // the wrong reason. The `?bust=N` query makes the route re-resolve the + // facade specifier, which is what picks up the new mock. (These tests need + // the `--experimental-test-module-mocks` flag that the `test:unit` and + // `test` scripts already pass.) + let bust = 0; + + test("a facade payload is written to the response byte-for-byte", async (t) => { + // The payload is injected rather than produced, so this is about the + // ROUTE's contract: it must not re-shape, re-count or re-derive + // anything. The payload carries the exact key set the real tree + // produces, `cached` included. + const payload = { + ok: true, + generatedAt: 1750000000000, + truncated: false, + counts: { projects: 1, directories: 1, sessions: 2 }, + projects: [ + { + key: "proj", + name: "proj", + repoPaths: ["/w/proj"], + latestAt: 1000, + sessionCount: 1, + directories: [ + { + path: "/w/proj", + name: "proj", + latestAt: 1000, + sessions: [ + { + id: "m1", + title: "主会话", + agent: "", + kind: "conversation", + status: "idle", + updatedAt: 1000, + children: [ + { + id: "c1", + title: "子 agent", + agent: "coder", + kind: "task", + status: "running", + updatedAt: 900, + children: [], + }, + ], + }, + ], + }, + ], + }, + ], + cached: false, + }; + await setupMocks(t, { acp: {} }); + t.mock.module(absPath("engine/session-tree-reads.js"), { + namedExports: { + readEngineSessionTree: async () => ({ + tree: payload, + source: "runtime-db", + gate: { endpoint: ENDPOINT, gate: "checked" }, + transport: "acp", + }), + }, + }); + const sessionsRoute = await import(`${absPath("routes/sessions.js")}?bust=${bust++}`); + const written = []; + const res = { + headersSent: false, + writeHead(status, headers) { written.push({ status, headers }); this.headersSent = true; return this; }, + end(body) { written.push({ body }); return this; }, + }; + await sessionsRoute.handleSessionTree({ url: "/api/session-tree" }, res, { cid: "t" }); + assert.equal(written[0].status, 200); + assert.equal(written[0].headers["Cache-Control"], "no-store"); + assert.deepEqual(JSON.parse(written[1].body), payload); + }); + + test("?refresh=1 reaches the facade as force:true, and nothing else does", async (t) => { + await setupMocks(t, { acp: {} }); + const seen = []; + t.mock.module(absPath("engine/session-tree-reads.js"), { + namedExports: { + readEngineSessionTree: async (o) => { + seen.push(o); + return { tree: { ok: true }, source: "runtime-db", gate: {}, transport: "acp" }; + }, + }, + }); + const sessionsRoute = await import(`${absPath("routes/sessions.js")}?bust=${bust++}`); + const mk = () => ({ + headersSent: false, + writeHead() { this.headersSent = true; return this; }, + end() { return this; }, + }); + // Table-driven: [query, expected force]. The limit/offset/page/cursor + // rows are the measured contract, not an assumption — #8 never read + // them, and adding a clamp here would invent behaviour. + const QUERIES = [ + ["?refresh=1", true], + ["", false], + ["?refresh=0", false], + ["?refresh=true", false], + ["?limit=1", false], + ["?limit=0", false], + ["?limit=999999", false], + ["?offset=5", false], + ["?page=2", false], + ["?cursor=x", false], + ["?limit=1&refresh=1", true], + ]; + for (const [q] of QUERIES) { + await sessionsRoute.handleSessionTree({ url: `/api/session-tree${q}` }, mk(), { cid: "t" }); + } + assert.equal(seen.length, QUERIES.length); + for (let i = 0; i < QUERIES.length; i += 1) { + assert.equal(seen[i].force, QUERIES[i][1], `query "${QUERIES[i][0]}" → force=${QUERIES[i][1]}`); + } + }); + + test("a capability error PROPAGATES so invokeHandler can answer 501", async (t) => { + // The one place this batch could have turned a 501 into a 200: the + // route's try/catch would fold the gate error into its own + // `{ok:false, reason:"session_tree_failed"}` body. It must not — the + // declaration gate is the whole point of the batch. + await setupMocks(t, { acp: {} }); + t.mock.module(absPath("engine/session-tree-reads.js"), { + namedExports: { + readEngineSessionTree: async () => { + throw new EngineCapabilityNotSupportedError({ + capability: "sessionCrud", + provider: "fixture-provider", + missing: ["listSessions"], + reason: "test fixture: interface-absent", + }); + }, + }, + }); + const sessionsRoute = await import(`${absPath("routes/sessions.js")}?bust=${bust++}`); + const res = { + headersSent: false, + writeHead() { this.headersSent = true; return this; }, + end() { return this; }, + }; + await assert.rejects( + () => sessionsRoute.handleSessionTree({ url: "/api/session-tree" }, res, { cid: "t" }), + isEngineCapabilityNotSupportedError, + ); + }); + + test("a LOOKALIKE error that merely carries the right .name does NOT propagate", async (t) => { + // The route discriminates with `isEngineCapabilityNotSupportedError` + // (an `instanceof` check), not with `cause.name === "…"`. `.name` is a + // writable instance property, so any code upstream can make an ordinary + // error impersonate the gate's — and a `.name` compare would then + // re-throw it and turn a soft-fail into a 501 the engine never + // declared. Pinned as a pair with the test above: the real class + // propagates, the impersonator does not. + await setupMocks(t, { acp: {} }); + const lookalike = new Error("not the gate"); + lookalike.name = "EngineCapabilityNotSupportedError"; + t.mock.module(absPath("engine/session-tree-reads.js"), { + namedExports: { readEngineSessionTree: async () => { throw lookalike; } }, + }); + const sessionsRoute = await import(`${absPath("routes/sessions.js")}?bust=${bust++}`); + const written = []; + const res = { + headersSent: false, + writeHead(s, h) { written.push({ s, h }); this.headersSent = true; return this; }, + end(b) { written.push({ b }); return this; }, + }; + // Must NOT reject: an impostor is an ordinary failure and degrades. + await sessionsRoute.handleSessionTree({ url: "/api/session-tree" }, res, { cid: "t" }); + assert.equal(written[0].s, 200, "an impostor must not become a 501"); + const body = JSON.parse(written[1].b); + assert.equal(body.ok, false); + assert.equal(body.reason, "session_tree_failed"); + assert.equal(body.detail, "not the gate"); + }); + + test("a NON-capability failure still degrades to ok:false + reason", async (t) => { + // The soft-fail contract for a broken tree read is unchanged: 200 with + // `{ok:false, reason:"session_tree_failed"}`. + await setupMocks(t, { acp: {} }); + t.mock.module(absPath("engine/session-tree-reads.js"), { + namedExports: { + readEngineSessionTree: async () => { + throw new Error("boom"); + }, + }, + }); + const sessionsRoute = await import(`${absPath("routes/sessions.js")}?bust=${bust++}`); + const written = []; + const res = { + headersSent: false, + writeHead(s, h) { written.push({ s, h }); this.headersSent = true; return this; }, + end(b) { written.push({ b }); return this; }, + }; + await sessionsRoute.handleSessionTree({ url: "/api/session-tree" }, res, { cid: "t" }); + assert.equal(written[0].s, 200); + const body = JSON.parse(written[1].b); + assert.equal(body.ok, false); + assert.equal(body.reason, "session_tree_failed"); + assert.equal(body.detail, "boom"); + }); +}); diff --git a/release/public-source.json b/release/public-source.json index 8fe872ef..5241631c 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3452,7 +3452,9 @@ "packages/webui/server/engine/providers/local-runtime-v2.capabilities.js", "packages/webui/server/engine/providers/local-runtime-v2.js", "packages/webui/server/engine/providers/tui-runtime-adapter.js", + "packages/webui/server/engine/session-export.js", "packages/webui/server/engine/session-reads.js", + "packages/webui/server/engine/session-tree-reads.js", "packages/webui/server/lib/acp-client.js", "packages/webui/server/lib/agent-team-detect.js", "packages/webui/server/lib/agent-team-status.js", @@ -3591,7 +3593,9 @@ "packages/webui/test/lib/engine/capabilities.test.js", "packages/webui/test/lib/engine/capability-snapshot.test.js", "packages/webui/test/lib/engine/host-facade.test.js", + "packages/webui/test/lib/engine/session-export.test.js", "packages/webui/test/lib/engine/session-reads.test.js", + "packages/webui/test/lib/engine/session-tree-reads.test.js", "packages/webui/test/lib/events-concurrency.test.js", "packages/webui/test/lib/events-hash.test.js", "packages/webui/test/lib/events.test.js", From 4fb8267064637a6e3474e4a4f11187bea003ebe9 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 03:32:00 +0800 Subject: [PATCH 11/21] feat(webui): the usage endpoints ask the engine facade, and the derived figures get one home (M3-B3) --- packages/webui/docs/ARCHITECTURE.md | 72 +- packages/webui/docs/ARCHITECTURE.zh-CN.md | 59 +- packages/webui/server/engine/index.js | 21 +- packages/webui/server/engine/usage-reads.js | 434 ++++++ packages/webui/server/routes/usage.js | 67 +- .../webui/test/lib/engine/usage-reads.test.js | 1230 +++++++++++++++++ release/public-source.json | 2 + 7 files changed, 1834 insertions(+), 51 deletions(-) create mode 100644 packages/webui/server/engine/usage-reads.js create mode 100644 packages/webui/test/lib/engine/usage-reads.test.js diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index 8f195b72..b75babf0 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -489,8 +489,8 @@ not import it but adopts the same shape. Unknown future statuses render as ### `engine/` (capability declarations + the local-runtime-v2 host) The engine abstraction lives at `server/engine/` (engine-abstraction -batch B1; migration state M1, plus M3 batches B0, B1 and B2). Ten files, -one job each: +batch B1; migration state M1, plus M3 batches B0, B1, B2 and B3). Eleven +files, one job each: | File | Owns | | --- | --- | @@ -504,6 +504,7 @@ one job each: | `engine/session-reads.js` | The directory-read family's facade calls (`readEngineSessionList`, `readEngineSessionListForWorkspace`, `readEngineSessionTitle`, `readEngineVersion`) and the endpoint→capability table `SESSION_READ_ENDPOINTS` (step M3, batch B1) | | `engine/session-tree-reads.js` | The session-tree family's facade call (`readEngineSessionTree`) and the endpoint→capability table `SESSION_TREE_ENDPOINTS` (step M3, batch B2). Gates **hard**: `assertSessionTreeCapability` throws → 501, because the tree is entirely engine data. Forwards to `lib/session-tree.js#getSessionTree`; the assembler is not duplicated | | `engine/session-export.js` | The export family's facade call (`readEngineSessionTranscript`) and the endpoint→capability table `SESSION_EXPORT_ENDPOINTS` (step M3, batch B2). Gates **soft**: `checkSessionExportCapability` reports and never throws, because export's primary source is `sessions.json`, not the engine | +| `engine/usage-reads.js` | The usage family's facade calls (`readEngineAccountQuota`, `readEngineSessionUsage`, `readEngineQuotaForecast`), the derived figure `contextUsedTokens`, and the endpoint→capability table `USAGE_READ_ENDPOINTS` (step M3, batch B3). Gates **hard** on the two engine reads and declares **no capability at all** for #19, which touches no engine surface | Routes take the host from the facade and never from `lib/acp-client.js`: `routes/plugins.js` and `routes/turn-diff.js` call @@ -581,13 +582,14 @@ everything it imports statically must stay free of `@mavis/*`, (209ms → 2700ms at server start; the facade's own load 4685ms → 5ms after declaration and construction were split). `test/lib/engine/host-facade.test.js` enforces it against the real module graph rather than against source text. -`engine/session-reads.js` lives under the same rule: its static imports are -`engine/capabilities.js` and `engine/index.js` only, and `lib/acp-client.js` + -`lib/config.js` are reached through `await import()` inside the functions. -Batch B2's two files hold to it identically — `lib/session-tree.js` and -`lib/transcript.js` are reached through `await import()`, and neither file -statically imports `engine/capabilities.js` beyond the single -`assertEngineCapability` binding the tree family actually calls. +`engine/session-reads.js`, `engine/session-tree-reads.js`, +`engine/session-export.js` and `engine/usage-reads.js` all live under the +same rule: their static imports are `engine/capabilities.js` and +`engine/index.js` only, and every heavier dependency — +`lib/acp-client.js`, `lib/config.js`, `lib/session-tree.js`, +`lib/transcript.js`, `lib/usage.js`, `lib/mavis-usage.js` and +`lib/quota-forecast.js` — is reached through `await import()` inside the +functions. #### Which endpoints read through the facade (step M3, batch B1) @@ -621,6 +623,58 @@ Three properties this layer holds, each with a test behind it: that lacks `listSessions` and assert the 501 payload. A gate nobody ever exercises is indistinguishable from no gate. +#### Which endpoints read through the facade (step M3, batch B3) + +`engine/usage-reads.js` covers the four usage endpoints (#15, #16, #17, +#19). This family is where a refactor can be entirely silent, because three +of its four numbers are derived rather than counted — so the table below is +as much about where each number comes from as about which capability gates +it: + +| Endpoint | Capability · sub-item | Value source | +| --- | --- | --- | +| `POST /api/usage` | `authCredentials` · `getAccountStatus` | `lib/usage.js#runUsageQuery` — the engine's `mcode/account/status` projection, copied into `cs.usage`; the payload is written byte-for-byte, `ok:false` / `error` shape included | +| `POST /api/usage-trigger` | `authCredentials` · `getAccountStatus` | the same read; the two endpoints differ only in the client's `record` flag, which is the difference between a reading and a measurement | +| `GET /api/usage-real` | `usageStats` · `getSessionUsage` | `lib/mavis-usage.js` over the engine's own `local_runtime_token_usage` table. `contextUsed` is derived here by `contextUsedTokens` | +| `GET /api/usage/forecast` | none of the 14 keys | webui's own `~/.mcode-webui/usage-history.ndjson`, via `lib/quota-forecast.js`. It calls no engine surface, so it declares none | + +Four properties this family holds, each with a test behind it: + +1. **`contextUsed` is cumulative, and the cache counters are not in it.** + `totalInput + totalOutput + totalReasoning`. The cache counters are a + SUBSET of `input`, so adding them double-counts; `totalCacheWrite` is + not part of the context window at all. This is also NOT the chat flow's + `lastTurnContextTokens`: the context bar shows one turn's worth, `#17` + shows the session's spend, and `test/lib/engine/usage-reads.test.js` + perturbs each of the seven numeric fields one at a time so a merged or + "simplified" formula flips a row instead of quietly shipping. +2. **`totalReasoning` is the database's own `SUM`, forwarded.** The + snapshot test reads the same aggregate with plain SQL and compares; a + facade that re-derived it from anything else fails. +3. **The forecast is a pure function of a history prefix.** Every prefix of + a growing history is compared against the module's own + `forecastExhaustion(readHistory())` at the same instant, and the sample + count's flat stretch across the deliberately-null sample is asserted, so + a read that re-filtered, re-sorted or re-sampled would break the + sequence rather than the shape. +4. **A `none` / `partial`-missing declaration would 501.** The registered + provider declares both `authCredentials` and `usageStats` `full`, so only + the fixture-driven tests can prove the gate bites. #19's `null` row is + the counter-example with a reason: gating a read that touches no engine + surface would remove a working endpoint in response to a declaration + about something it does not depend on. + +`#17` declares `usageStats` · `getSessionUsage` but does not yet CALL that +method; it reads the same SQLite table the method reads, through +`lib/mavis-usage.js`. Three measured reasons, stated in the module header: +the catalogue host only exists under the `runtime` transport +(`acp-client.js#transportWantsCatalogue`), and `acp` is the default; +`getSessionUsage` answers `{summary, rows: UsageView[]}` where the endpoint +answers a per-column aggregate with `rows` as a COUNT, so switching would +mean rebuilding `totalReasoning` and `contextUsed` from a different +starting point; and it would put the v2 TypeScript tree on an endpoint that +needs nothing from it. M4 is where the two are allowed to meet. + The transport→provider table has one entry (`runtime`). Under the default `acp` transport no provider is registered yet, so the gate reports `unregistered-transport` and passes through — M4 registers the ACP diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index 44a1fa48..3bce7f75 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -461,7 +461,7 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 ### `engine/`(能力声明 + local-runtime-v2 host) 引擎抽象层位于 `server/engine/`(engine-abstraction 批次 B1;迁移 -状态 M1,外加 M3 的 B0、B1 与 B2 三批)。十个文件,各管一件事: +状态 M1,外加 M3 的 B0、B1、B2 与 B3 四批)。十一个文件,各管一件事: | 文件 | 职责 | | --- | --- | @@ -475,6 +475,7 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 | `engine/session-reads.js` | 目录读族的面板调用(`readEngineSessionList`、`readEngineSessionListForWorkspace`、`readEngineSessionTitle`、`readEngineVersion`)与端点→能力对照表 `SESSION_READ_ENDPOINTS`(迁移步 M3 批次 B1) | | `engine/session-tree-reads.js` | 会话树族的面板调用 `readEngineSessionTree` 与端点→能力对照表 `SESSION_TREE_ENDPOINTS`(迁移步 M3 批次 B2)。**硬门控**:`assertSessionTreeCapability` 抛出 → 501,因为树完全由引擎数据构成。转发到 `lib/session-tree.js#getSessionTree`,树的装配逻辑不复制第二份 | | `engine/session-export.js` | 导出族的面板调用 `readEngineSessionTranscript` 与端点→能力对照表 `SESSION_EXPORT_ENDPOINTS`(迁移步 M3 批次 B2)。**软门控**:`checkSessionExportCapability` 只报告、从不抛出,因为导出的主数据源是 `sessions.json` 而非引擎 | +| `engine/usage-reads.js` | 用量族的面板调用(`readEngineAccountQuota`、`readEngineSessionUsage`、`readEngineQuotaForecast`)、派生量 `contextUsedTokens`,与端点→能力对照表 `USAGE_READ_ENDPOINTS`(迁移步 M3 批次 B3)。两个引擎读**硬门控**;#19 **完全不声明能力**,因为它不触达任何引擎面 | 路由从门面取 host,不从 `lib/acp-client.js` 取:`routes/plugins.js` 与 `routes/turn-diff.js` 调 `getEngineCatalogueHost()`。两者都保留 `deps` @@ -538,12 +539,13 @@ handler 层测试因此保持封闭。 学费才换来这条(server 启动 209ms → 2700ms;声明与构造拆成两个文件后, 门面自身加载 4685ms → 5ms)。`test/lib/engine/host-facade.test.js` 对着真实模块图强制它,而不是对着源码文本。 -`engine/session-reads.js` 服从同一条纪律:它的静态 import 只有 -`engine/capabilities.js` 与 `engine/index.js`,`lib/acp-client.js` + `lib/config.js` -都在函数体内用 `await import()` 触达。批次 B2 的两个文件同样守住它: -`lib/session-tree.js` 与 `lib/transcript.js` 都用 `await import()` 触达, -且除树族真正调用的那一个 `assertEngineCapability` 绑定外, -两个文件都没有静态 import `engine/capabilities.js`。 +`engine/session-reads.js`、`engine/session-tree-reads.js`、 +`engine/session-export.js` 与 `engine/usage-reads.js` 全部服从同一条 +纪律:静态 import 只有 `engine/capabilities.js` 与 `engine/index.js`, +而每个更重的依赖——`lib/acp-client.js`、`lib/config.js`、 +`lib/session-tree.js`、`lib/transcript.js`、`lib/usage.js`、 +`lib/mavis-usage.js` 与 `lib/quota-forecast.js`——都在函数体内用 +`await import()` 触达。 #### 哪些端点走门面读(迁移步 M3 批次 B1) @@ -574,6 +576,49 @@ handler 层测试因此保持封闭。 所以今天没有任何端点会 501;测试用一份缺 `listSessions` 的样本声明 驱动出 501 载荷。没人跑过的门控与没有门控无法区分。 +#### 哪些端点走门面读(迁移步 M3 批次 B3) + +`engine/usage-reads.js` 覆盖 4 个用量端点(#15、#16、#17、#19)。 +这一族是「重构全程静默」的重灾区:四个数字里有三个是**算出来的** +而不是数出来的,所以下表不只写门控哪个能力,更写清每个数字从哪来: + +| 端点 | 能力 · 子项 | 取值来源 | +| --- | --- | --- | +| `POST /api/usage` | `authCredentials` · `getAccountStatus` | `lib/usage.js#runUsageQuery`——引擎的 `mcode/account/status` 投影,抄进 `cs.usage`;载荷逐字节写出,含 `ok:false` / `error` 形状 | +| `POST /api/usage-trigger` | `authCredentials` · `getAccountStatus` | 同一次读;两个端点只差客户端的 `record` 标志,而它决定这次是「读数」还是「采样」 | +| `GET /api/usage-real` | `usageStats` · `getSessionUsage` | `lib/mavis-usage.js` 读引擎自己的 `local_runtime_token_usage` 表;`contextUsed` 由 `contextUsedTokens` 在此派生 | +| `GET /api/usage/forecast` | 14 键中无对应键 | webui 自己的 `~/.mcode-webui/usage-history.ndjson`,经 `lib/quota-forecast.js`。它不触达任何引擎面,所以不声明任何能力 | + +本层守住四条性质,每条背后都有测试: + +1. **`contextUsed` 是累计值,且不含缓存计数。** 公式是 + `totalInput + totalOutput + totalReasoning`。缓存计数是 `input` 的 + **子集**,加上会重复计数;`totalCacheWrite` 根本不在上下文窗口里。 + 它也**不是**聊天流程的 `lastTurnContextTokens`:上下文条显示的是 + 一轮的量,`#17` 显示的是整会话的花费。 + `test/lib/engine/usage-reads.test.js` 对七个数值字段逐个扰动, + 被合并或被「简化」的公式会翻掉某一行,而不是悄悄发版。 +2. **`totalReasoning` 是数据库自己的 `SUM`,原样转发。** 快照测试用 + 裸 SQL 独立算出同一个聚合再比对;门面若从别处重新派生,此测试即红。 +3. **预测是历史前缀的纯函数。** 增长中的历史的每一个前缀,都在同一时刻 + 与模块自己的 `forecastExhaustion(readHistory())` 比对,并且断言样本数 + 在那条故意置 `null` 的样本处出现的「平台期」——所以重新过滤、重新排序 + 或重新采样会破坏**序列**而不只是破坏形状。 +4. **`none` / 缺子项的 `partial` 声明会 501。** 已注册的 provider 把 + `authCredentials` 与 `usageStats` 都声明为 `full`,所以只有样本驱动 + 的测试能证明门控会咬。#19 那一行 `null` 是带理由的反例:给一个 + 根本不触达引擎面的读加硬门控,等于用一条与它无关的声明去关掉一个 + 正常工作的端点。 + +`#17` 声明了 `usageStats` · `getSessionUsage`,但**尚未调用**该方法: +它经 `lib/mavis-usage.js` 读的是该方法读的同一张 SQLite 表。三条实测 +理由写在模块头注释里——catalogue host 只在 `runtime` 传输下存在 +(`acp-client.js#transportWantsCatalogue`),而 `acp` 是默认值; +`getSessionUsage` 回答的是 `{summary, rows: UsageView[]}`,端点回答的是 +按列聚合且 `rows` 是 COUNT 的形状,换过去就意味着从另一个起点重建 +`totalReasoning` 与 `contextUsed`;而且它会把 v2 的 TypeScript 依赖树压到 +一个本来不需要它的端点的应答路径上。M4 才是两者允许会合的地方。 + 传输→provider 表目前只有 `runtime` 一条。默认 `acp` 传输下尚无已注册 provider,于是门控报告 `unregistered-transport` 并放行——M4 注册 ACP provider 后该表补上对应行。放行不等于声称支持,二者刻意分开报告。 diff --git a/packages/webui/server/engine/index.js b/packages/webui/server/engine/index.js index 6843a322..ed0f6680 100644 --- a/packages/webui/server/engine/index.js +++ b/packages/webui/server/engine/index.js @@ -35,8 +35,9 @@ // no route's behaviour changed. M3's first batch (B0) done — the // catalogue host itself is now reached through this facade too // (engine/host.js), so the plugins and turn-diff routes no longer name -// lib/acp-client.js. The rest of M3, then M4, will route new consumers -// through this facade one endpoint family at a time. +// lib/acp-client.js. M3 batches B1 (#9 #10 #72 #74 #75) and B3 (#15 #16 +// #17 #19) done. B2 (#8 #11) and the rest of M3, then M4, will route +// their consumers through this facade one endpoint family at a time. import { ENGINE_CAPABILITY_KEYS } from "./capabilities.js"; // Declarations only — importing the provider *host-construction* modules @@ -103,6 +104,22 @@ export { readEngineSessionTranscript, resolveSessionExportProvider, } from "./session-export.js"; +// The usage family's gated reads (step M3, batch B3). Same cycle, same +// rule, same reasoning as session-reads.js above: usage-reads.js reads +// NOTHING from this module at module scope — its `USAGE_READ_ENDPOINTS` +// table is a literal and every binding it needs (`getEngineProvider`, +// `DEFAULT_ENGINE_PROVIDER_ID`) is read inside a function body. A new +// top-level `const X = SOMETHING_FROM_INDEX` in usage-reads.js breaks the +// re-export exactly as it would in session-reads.js. +export { + USAGE_READ_ENDPOINTS, + assertUsageReadCapability, + contextUsedTokens, + readEngineAccountQuota, + readEngineQuotaForecast, + readEngineSessionUsage, + resolveUsageReadProvider, +} from "./usage-reads.js"; export { LOCAL_RUNTIME_V2_CAPABILITIES } from "./providers/local-runtime-v2.capabilities.js"; export { TUI_RUNTIME_ADAPTER_CAPABILITIES } from "./providers/tui-runtime-adapter.js"; diff --git a/packages/webui/server/engine/usage-reads.js b/packages/webui/server/engine/usage-reads.js new file mode 100644 index 00000000..fa0b774b --- /dev/null +++ b/packages/webui/server/engine/usage-reads.js @@ -0,0 +1,434 @@ +// webui/server/engine/usage-reads.js +// +// Migration step M3, batch B3: the usage family (用量族) — the four +// endpoints that answer "how much has this cost, and when will it run +// out": +// +// #15 POST /api/usage — plan quota (5h / weekly) read +// #16 POST /api/usage-trigger — the same read, recorded as a sample +// #17 GET /api/usage-real — real per-session token usage +// #19 GET /api/usage/forecast — quota-exhaustion prediction +// +// What this file is for. Three of these four numbers decide what a user +// does next — refresh, switch model, stop working — and each of them is a +// DERIVED quantity, not a counter. #15/#16 re-derive the plan windows out +// of the engine's account projection. #17 re-derives `contextUsed` out of +// three separate token totals. #19 re-derives an exhaustion time out of a +// least-squares fit. A refactor that "cleans up" one of those formulas +// changes what the user sees and reports nothing, which is the failure +// mode this batch is gated on. So the derivations live HERE, once, named, +// and tested on their inputs — the route only assembles JSON. +// +// What this file deliberately does NOT do: +// +// - It does not re-read the database. `lib/mavis-usage.js` owns the SQL, +// the `node:sqlite` / `sqlite3`-spawn dual path and the NULL-to-zero +// coercion; `lib/usage.js` owns the quota-window copy into `cs.usage`; +// `lib/quota-forecast.js` owns the NDJSON history and the least-squares +// fit. A second reader over `local_runtime_token_usage` would be a +// second answer to "what did this session cost". +// - It does not construct a host. #17's data currently comes from the +// engine's own SQLite file, not from a live `CliService` — see the +// `getSessionUsage` note below for why the provider call is deferred, +// and for what would have to be true before it is not. +// - It does not widen the engine's own degradation. `getMavisTokenUsage` +// returns `null` for "no such session / no db / no rows", and the route +// turns that into `{ok:true, found:false, …}` with HTTP 200. That +// answer is the endpoint's long-standing contract and it is a +// different question from "may this provider report usage at all". +// +// The `getSessionUsage` question, stated once because it is the batch's +// most load-bearing decision. The v2 provider declares `usageStats: full`, +// and `CliService#getSessionUsage` is a real method that reads the SAME +// `local_runtime_token_usage` table this endpoint already reads. Routing +// through it anyway today would be a behaviour change dressed as a +// refactor, for three measured reasons: +// +// 1. It only exists under the `runtime` transport. The catalogue host is +// booted by `acp-client.js#transportWantsCatalogue()`, which is +// `MCODE_WEBUI_TRANSPORT === "runtime"`. The DEFAULT transport is +// `acp` (`lib/config.js`), and under it there is no `CliService` to +// call — so the switch would take the endpoint from "always answers" +// to "answers on one opt-in transport". +// 2. Its shape is not this endpoint's shape. `getSessionUsage` answers +// `{summary, rows: UsageView[]}`; the endpoint answers a per-column +// aggregate plus `rows` as a COUNT. Rebuilding the aggregate from +// `rows` would re-derive `totalReasoning` and `contextUsed` from a +// different starting point — exactly the silent numeric drift this +// batch forbids. +// 3. It would put the v2 TypeScript dependency tree on the answer path +// of an endpoint that currently needs nothing from it (the M1 lesson). +// +// So the declaration names the provider method the endpoint DEPENDS ON — +// which is what a declaration is for — and the read keeps using the file +// the provider itself would read. M4 is where the two are allowed to meet. +// +// Boot-path weight. `app.js` imports the routes, the routes import this +// file, so this file is on the boot path. It therefore statically imports +// nothing heavier than `capabilities.js` and `index.js` (both pure +// declaration modules); `lib/usage.js`, `lib/mavis-usage.js`, +// `lib/quota-forecast.js` and `lib/config.js` are reached through +// `await import()` inside the functions. That split is the M1 lesson — +// putting the `@mavis/*` tree on the boot path once cost 209ms → 2700ms of +// server start and broke the integration tests' 3s window. +// +// Provider selection is M4's job, same as B1 and B2: `providerByTransport()` +// maps a transport to a REGISTERED provider id; today only `runtime` has +// one, so under the default `acp` transport the gate reports +// `gate: "unregistered-transport"` instead of inventing one. + +// `node:fs` is a builtin, not a project dependency: the boot-path promise +// below is about not dragging lib/ or @mavis/* trees in, and this costs +// nothing. It is here for one caller — #17's `dbExists`, which the route +// used to compute itself from a path constant it imported at module scope. +import { existsSync } from "node:fs"; + +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, exactly as in + * `session-reads.js#providerByTransport` and + * `session-tree-reads.js#providerByTransport`, which this mirrors rather + * than merges: the four families have separate read contracts and a shared + * table would force one of them to inherit another's policy. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its temporal + * dead zone on a cold `import("./engine/index.js")`. Every consumer of the + * table is a function anyway. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +/** + * The declaration each endpoint of this family needs, and the sub-item it + * needs from that capability. + * + * - #15 / #16 are `authCredentials` / `getAccountStatus`. The plan tier + * and both window percentages come from the engine's account + * projection, and the declaration names that method explicitly ("the + * engine holds the credential, so it is the only side that may call + * MiniMax's quota endpoint" — see `lib/usage.js`). A `partial` that + * dropped exactly `getAccountStatus` would answer 501 naming it rather + * than a generic refusal. + * - #17 is `usageStats` / `getSessionUsage` — the same pair the v2 + * declaration enumerates under `usageStats`. See the header for why + * the read does not yet call that method. + * - #19 is `null`, and this is the one row a reader will double-take. + * The forecast reads `~/.mcode-webui/usage-history.ndjson`, a file + * webui itself appends to; it calls no engine surface at all. The + * numbers in it ORIGINATED in the engine, but a read that touches no + * engine surface must not be gated on an engine capability — that is + * the same lie B1 declined for `/api/health`, and gating it hard would + * remove a working endpoint in response to a declaration about + * something it does not depend on. The precedent for a soft family + * that DOES cross the seam is B2's export enrichment + * (`_meta.mcode_unavailable`); #19 needs none of that, because there is + * no enrichment to lose. + * + * @type {Readonly>} + */ +export const USAGE_READ_ENDPOINTS = Object.freeze({ + "POST /api/usage": { capability: "authCredentials", subItem: "getAccountStatus" }, + "POST /api/usage-trigger": { capability: "authCredentials", subItem: "getAccountStatus" }, + "GET /api/usage-real": { capability: "usageStats", subItem: "getSessionUsage" }, + "GET /api/usage/forecast": null, +}); + +/** + * Resolve the provider that answers usage reads on `transport`, or `null` + * when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveUsageReadProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Check one endpoint of this family against the active provider's + * declaration. Throws `EngineCapabilityNotSupportedError` — which + * `app.js#invokeHandler` turns into 501 — when the declaration says the + * capability (or the exact sub-item) is absent. + * + * @param {string} endpoint A key of USAGE_READ_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null}} + */ +export function assertUsageReadCapability(endpoint, transport) { + const need = USAGE_READ_ENDPOINTS[endpoint]; + if (need === undefined) { + // Caller confusion, not an engine limitation — a plain Error so the + // HTTP layer never answers 501 for a typo in webui's own code. + const err = new Error( + `assertUsageReadCapability: "${endpoint}" is not part of the usage family ` + + `(known: ${Object.keys(USAGE_READ_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_usage_read_endpoint"; + throw err; + } + const provider = resolveUsageReadProvider(transport); + if (need === null) { + return { + endpoint, + gate: "no-capability-key", + provider: provider ? provider.id : null, + capability: null, + subItem: null, + }; + } + if (!provider) { + return { + endpoint, + gate: "unregistered-transport", + provider: null, + capability: need.capability, + subItem: need.subItem, + }; + } + assertEngineCapability(provider.capabilities, need.capability, provider.id, need.subItem); + return { + endpoint, + gate: "checked", + provider: provider.id, + capability: need.capability, + subItem: need.subItem, + }; +} + +// --------------------------------------------------------------------------- +// The derivations. Pure functions, exported, and tested on their INPUTS. +// --------------------------------------------------------------------------- + +/** + * The `contextUsed` figure `GET /api/usage-real` reports. + * + * CUMULATIVE input + output + reasoning, and deliberately NOT the + * per-turn figure. The two coexist in this repository and confusing them + * is the single most likely way for this endpoint to start lying: + * + * - `lib/mavis-usage.js#_buildUsageResult` publishes + * `lastTurnContextTokens` (last input + output + reasoning) and the + * chat flow stores it as `cs.context.tokens` — the CONTEXT BAR. One + * turn's worth, always ≤ the model's context limit. + * - `GET /api/usage-real` reports the session's CUMULATIVE spend + * (`v0.5.bx-10` fix: "context 实际是 input + output + reasoning"), + * which is why a 13-turn session can show 566k there. That is what + * the number has always meant on this endpoint and the frontend reads + * it as such. + * + * `cacheRead` / `cacheWrite` are excluded: they are a SUBSET of `input` + * (counted again by the engine inside the prompt), so adding them + * double-counts. `totalCacheWrite` is excluded for the same reason plus + * the fact that it is not part of the context window at all. + * + * Written as one expression, in the order the endpoint has always summed, + * over the SAME three fields the endpoint has always summed. That is + * deliberate: every field here arrives already coerced to a number by + * `_buildUsageResult` (`Number(x) || 0`), so no rounding point is + * introduced, and a `null` from a future provider coerces exactly the way + * the pre-facade expression coerced it. `test/lib/engine/usage-reads.test.js` + * pins the inputs, not just this number. + * + * @param {object} usage A `getMavisTokenUsage` result. + * @returns {number} + */ +export function contextUsedTokens(usage) { + return usage.totalInput + usage.totalOutput + usage.totalReasoning; +} + +// --------------------------------------------------------------------------- +// Reads +// --------------------------------------------------------------------------- + +/** + * Where each read's bytes actually came from. Three distinct producers, + * named rather than assumed: + * + * - `"account-status"` — the engine's `mcode/account/status` extension + * method, via `lib/mcode-rpc.js#getAccountStatus`. The same answer + * `quotaSnapshot` has always labelled `source: "acp"` in its own + * payload; the facade names the producer rather than the wire. + * - `"runtime-db"` — the engine's own `local_runtime_token_usage` table + * in its runtime sqlite, via `lib/mavis-usage.js`. Same vocabulary as + * B2's session tree: not a transport-switched surface. + * - `"history-file"` — webui's OWN `usage-history.ndjson`. The forecast + * read touches no engine surface, which is why its declaration row is + * `null`; this value keeps that honest at the call site. + * + * @typedef {"account-status" | "runtime-db" | "history-file"} UsageReadSource + */ + +/** + * #15 / #16 — the plan-quota read. + * + * `runUsageQuery` is the whole contract and is forwarded verbatim: it + * copies the engine's projection into `cs.usage`, appends at most one + * NDJSON history sample when `record` is on, pushes state, and returns the + * popover payload. The route writes that payload as the response body + * byte-for-byte, including its `ok:false` / `error` shape for an engine + * that could not be reached — the request itself succeeded, so the status + * stays 200. + * + * `record` is the difference between reading and measuring, and it is NOT + * defaulted here: `lib/usage.js` owns that default (`true`, the + * historical "a read is also a measurement" behaviour). The route passes + * the client's explicit `record !== false` through unchanged. + * + * @param {object} options + * @param {object} options.cs The webui client state `cs.usage` is written into. + * @param {string} options.cid Client id, for the state push. + * @param {boolean} [options.record] Append a forecast sample; see above. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/usage`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. Exists so tests can exercise both + * the `runtime` and the unregistered `acp` branch without mutating + * process env. + * @returns {Promise<{payload: object, source: UsageReadSource, gate: object, transport: string}>} + */ +export async function readEngineAccountQuota(options = {}) { + const endpoint = options.endpoint || "POST /api/usage"; + const usage = await import("../lib/usage.js"); + const config = await import("../lib/config.js"); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertUsageReadCapability(endpoint, transport); + const payload = await usage.runUsageQuery(options.cs, options.cid, { + record: options.record !== false, + }); + return { payload, source: "account-status", gate, transport }; +} + +/** + * #17 — the real per-session token usage. + * + * `usage` is `getMavisTokenUsage`'s own object, forwarded field for + * field: `rows`, the five totals, `firstTs`, `lastTs`, and the per-turn + * and cache-hit figures the chat flow also consumes. The facade adds + * exactly one derived number, `contextUsed` (see `contextUsedTokens`), and + * nothing else — in particular it does not re-derive `totalReasoning`, + * which is the database's own `SUM(reasoning_tokens)` and has exactly one + * correct source. + * + * `found:false` carries the same two facts the endpoint has always + * reported for "no session id yet / no such session": which database it + * looked in, and whether that database exists. `dbExists` is the + * `existsSync` the route used to do itself, moved behind the lazy + * `lib/config.js` boundary so `routes/usage.js` no longer names a path + * constant at module scope. + * + * @param {object} [options] + * @param {string|null} [options.mcodeSessionId] The `mvs_…` id to read. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/usage-real`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @returns {Promise<{mcodeSessionId: string, found: boolean, usage: object|null, contextUsed: number|null, model: string|null, dbPath: string, dbExists: boolean, source: UsageReadSource, gate: object, transport: string}>} + */ +export async function readEngineSessionUsage(options = {}) { + const endpoint = options.endpoint || "GET /api/usage-real"; + const [mavis, config] = await Promise.all([ + import("../lib/mavis-usage.js"), + import("../lib/config.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertUsageReadCapability(endpoint, transport); + const mcodeSessionId = options.mcodeSessionId || ""; + const dbPath = config.MAVIS_DB_PATH; + const dbExists = existsSync(dbPath); + const usage = mcodeSessionId ? await mavis.getMavisTokenUsage(mcodeSessionId) : null; + if (!usage) { + return { + mcodeSessionId, + found: false, + usage: null, + contextUsed: null, + model: null, + dbPath, + dbExists, + source: "runtime-db", + gate, + transport, + }; + } + // Best-effort and in that order: the endpoint has always answered even + // when the model lookup fails, and `getMavisTokenUsageModel` returns + // `null` for its own reasons (no db, no row, a `model` column that is + // NULL or empty). `(m && m.model) || null` is the endpoint's own + // fallback, kept verbatim. + const model = await mavis.getMavisTokenUsageModel(mcodeSessionId).catch(() => null); + return { + mcodeSessionId, + found: true, + usage, + contextUsed: contextUsedTokens(usage), + model: (model && model.model) || null, + dbPath, + dbExists, + source: "runtime-db", + gate, + transport, + }; +} + +/** + * #19 — the quota-exhaustion forecast. + * + * `readHistory` and `forecastExhaustion` are forwarded verbatim, which is + * what keeps the SEQUENCE continuous: the forecast for a given history + * prefix is a pure function of that prefix, and a refactor that re-read, + * re-filtered, re-sorted or re-sampled the history would shift every + * point of the curve without changing any single call's shape. + * `test/lib/engine/usage-reads.test.js#forecast sequence` pins the prefix + * series against the pre-refactor computation. + * + * The `try/catch` around `readHistory` is the endpoint's own belt-and- + * braces guard (the module already swallows FS errors; the catch is so a + * buggy extension can never break the endpoint) and it MOVES here with + * the read, because the read is what can fail. On failure the history is + * `[]`, and `forecastExhaustion([])` answers `reason: "no_history"` — + * byte-identical to the pre-facade body, which the UI renders as + * "collecting data…". + * + * @param {object} [options] + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/usage/forecast`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @param {object} [options.forecastOptions] Forwarded to + * `forecastExhaustion` (`minSamples`, `nowMs`); the endpoint passes + * neither today, and the defaults must stay the module's. + * @returns {Promise<{forecast: object, historyLength: number, source: UsageReadSource, gate: object, transport: string}>} + */ +export async function readEngineQuotaForecast(options = {}) { + const endpoint = options.endpoint || "GET /api/usage/forecast"; + const [quota, config] = await Promise.all([ + import("../lib/quota-forecast.js"), + import("../lib/config.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertUsageReadCapability(endpoint, transport); + let history = []; + try { + history = quota.readHistory(); + } catch { + history = []; + } + return { + forecast: quota.forecastExhaustion(history, options.forecastOptions || {}), + historyLength: history.length, + source: "history-file", + gate, + transport, + }; +} diff --git a/packages/webui/server/routes/usage.js b/packages/webui/server/routes/usage.js index 1e8aae03..b4f16b62 100644 --- a/packages/webui/server/routes/usage.js +++ b/packages/webui/server/routes/usage.js @@ -12,26 +12,28 @@ // now decided by lib/usage.js#runUsageQuery's `record` option, so a // caller that is only rendering the number does not add a sample. Also // added handleForecast which exposes the prediction to the UI. +// +// M3-B3: all four usage endpoints now reach the engine through +// `engine/usage-reads.js` instead of naming lib/usage.js, lib/mavis-usage.js, +// lib/quota-forecast.js and lib/config.js themselves. Nothing about the +// wire changed — the facade forwards the payloads and owns the two +// DERIVED figures (`contextUsed`, the forecast) so the formulas have one +// home. See engine/usage-reads.js for why #19 declares no capability and +// why #17 does not yet call the provider's `getSessionUsage` method. -import { existsSync } from "node:fs"; -import { runUsageQuery } from "../lib/usage.js"; import { - getMavisTokenUsage, - getMavisTokenUsageModel, -} from "../lib/mavis-usage.js"; + readEngineAccountQuota, + readEngineQuotaForecast, + readEngineSessionUsage, +} from "../engine/usage-reads.js"; import { pushStateFor } from "../lib/state-bus.js"; import { getMcodeModelLimit } from "../lib/models.js"; -import { MAVIS_DB_PATH } from "../lib/config.js"; -// C07: quota exhaustion forecast (linear LS on usage history) -// readHistory + forecastExhaustion + recordSnapshotFromCs. -// Pure module — no state-bus / settings coupling, just FS + math. -import { readHistory, forecastExhaustion } from "../lib/quota-forecast.js"; import { readJson } from "../lib/read-json.js"; // POST /api/usage & /api/usage-trigger // -// The answer is the quota figures runUsageQuery just fetched. It used to be a +// The answer is the quota figures the read just fetched. It used to be a // bare {ok:true} written before the fetch — the popover reads this response // body, so it never saw a `remaining` even when the fetch succeeded. export async function handleUsage(req, res, ctx) { @@ -39,7 +41,9 @@ export async function handleUsage(req, res, ctx) { // history; the client's poll uses it. Absent or true means the historical // behaviour, where a read is also a measurement. const body = await readJson(req); - const payload = await runUsageQuery(ctx.cs, ctx.cid, { + const { payload } = await readEngineAccountQuota({ + cs: ctx.cs, + cid: ctx.cid, record: body.record !== false, }); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); @@ -68,26 +72,29 @@ export async function handleUsageReal(req, res, ctx) { }), ); } - const usage = await getMavisTokenUsage(sid); - const model = await getMavisTokenUsageModel(sid); - if (!usage) { + const read = await readEngineSessionUsage({ mcodeSessionId: sid }); + if (!read.found) { res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); return res.end( JSON.stringify({ ok: true, found: false, sid, - dbPath: MAVIS_DB_PATH, - dbExists: existsSync(MAVIS_DB_PATH), + dbPath: read.dbPath, + dbExists: read.dbExists, }), ); } + const usage = read.usage; res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); return res.end( JSON.stringify({ ok: true, found: true, sid, + // `rows` is a COUNT, not a list — the engine's provider method + // answers a row ARRAY under the same name, which is one of the + // reasons #17 does not call it yet (engine/usage-reads.js header). rows: usage.rows, totalInput: usage.totalInput, totalOutput: usage.totalOutput, @@ -95,12 +102,15 @@ export async function handleUsageReal(req, res, ctx) { totalCacheWrite: usage.totalCacheWrite, totalReasoning: usage.totalReasoning, // v0.5.bx-10 fix: context 实际是 input + output + reasoning (cache 是 input 子集) - contextUsed: usage.totalInput + usage.totalOutput + usage.totalReasoning, - model: (model && model.model) || null, + // CUMULATIVE, deliberately not the chat flow's per-turn + // `lastTurnContextTokens`. The formula now lives in the engine layer + // as `contextUsedTokens` and is pinned on its inputs there. + contextUsed: read.contextUsed, + model: read.model, modelLimit: getMcodeModelLimit(cs.model && cs.model.name), firstTs: usage.firstTs, lastTs: usage.lastTs, - dbPath: MAVIS_DB_PATH, + dbPath: read.dbPath, }), ); } @@ -108,20 +118,11 @@ export async function handleUsageReal(req, res, ctx) { // C07: GET /api/usage/forecast — predict quota exhaustion time. // Reads ~/.mcode-webui/usage-history.ndjson, runs forecastExhaustion, // and returns the JSON payload documented in CAPABILITIES.md §8. -// Best-effort: if the file is missing or empty, returns -// { ok: true, forecast: { ... reason: "no_history" } } so the UI -// can render a "collecting data…" placeholder instead of erroring. +// Best-effort: if the file is missing or empty, the read answers +// { … reason: "no_history" } so the UI can render a "collecting data…" +// placeholder instead of erroring. export async function handleForecast(_req, res, _ctx) { - let history = []; - try { - history = readHistory(); - } catch { - // readHistory already swallows FS errors; this catch is just a - // belt-and-braces guard so a buggy extension never breaks the - // endpoint. - history = []; - } - const forecast = forecastExhaustion(history); + const { forecast } = await readEngineQuotaForecast(); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); return res.end( JSON.stringify({ diff --git a/packages/webui/test/lib/engine/usage-reads.test.js b/packages/webui/test/lib/engine/usage-reads.test.js new file mode 100644 index 00000000..9572031f --- /dev/null +++ b/packages/webui/test/lib/engine/usage-reads.test.js @@ -0,0 +1,1230 @@ +// webui/test/lib/engine/usage-reads.test.js +// +// M3-B3: the usage family's engine facade (#15, #16, #17, #19). +// +// This family is the batch where a "harmless" refactor can be entirely +// silent, because three of its four numbers are DERIVED and none of them +// is compared against anything. So the four things pinned here are: +// +// 1. THE FORMULA'S INPUTS. `contextUsed` is `totalInput + totalOutput + +// totalReasoning` — the CUMULATIVE figure, not the chat flow's +// per-turn `lastTurnContextTokens`, and explicitly NOT including the +// cache counters (which are a subset of `input` and would +// double-count). Section 3 does not assert the formula's result for a +// handful of inputs; it perturbs each of the seven numeric fields one +// at a time and records WHICH ones move the answer. A future +// "simplification" that swaps in the per-turn figure, or that starts +// adding `totalCacheRead`, cannot pass. +// +// 2. THE NUMERIC SNAPSHOT on a real sqlite fixture, row by row, for the +// boundary cases the endpoint exists for: reasoning present, reasoning +// absent, cache hit zero, single turn, many turns, a session whose +// row is gone but whose usage rows remain, and NULL token columns. +// The expected values are written out longhand, not recomputed by the +// same expression under test — a test that computes its oracle with +// the implementation's formula proves nothing. +// +// 3. THE FORECAST SEQUENCE. #19 is a pure function of a history prefix, +// so consecutive reads of a growing history must move the way the +// pre-refactor implementation moved them: no re-filtering, no +// re-sorting, no re-sampling. Section 5 walks every prefix and +// compares against the module's own `forecastExhaustion(readHistory())`. +// +// 4. THE GATE IS REAL, AND THE MOCK IS REAL. The registered provider +// declares `usageStats` and `authCredentials` `full`, so only this +// file can prove the gate would bite. And node:test's +// `mock.module` re-evaluates only the MOCKED specifier, so a route +// module already in the registry keeps its old live binding — every +// route test here re-imports the route under a fresh `?bust=N`, and +// section 6 ends with the control that proves the mock took: with no +// mock at all, the same request reads the fixture db. +// +// Test style follows test/lib/engine/session-reads.test.js (B1) and +// test/lib/engine/session-tree-reads.test.js (B2): table-driven, one row +// per case, fixture built before any server module is imported. + +import { test, describe, after } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync } from "node:fs"; +import { join } from "node:path"; +import { Readable } from "node:stream"; +import { DatabaseSync } from "node:sqlite"; + +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; +import { setupMocks, absPath } from "../../helpers/_setup.js"; + +// --------------------------------------------------------------------------- +// Fixture — built BEFORE any server module is imported, and that ordering is +// load-bearing, not stylistic. +// +// `lib/config.js` resolves MAVIS_DB_PATH at MODULE LOAD from +// `MINIMAX_DATA_DIR ?? MAVIS_DATA_DIR`, and `lib/mavis-usage.js` imports it +// statically. A `before()` hook that set the env would be too late: the +// first import reaching config.js would already have frozen the real +// ~/.minimax path, and every case below would read the developer's own +// database instead of the fixture. Hence: build the dir and the db, set the +// env here at module top level, and only then import server code. +// +// BOTH env names are set, not just MAVIS_DATA_DIR — `MINIMAX_DATA_DIR` +// wins, and a gate command that isolates the runtime data dir exports it. +// A fixture that wants the database owns the variable that wins. +// +// Prefixes are registered in scripts/test-tmp-leak.check.mjs#KNOWN_PREFIXES; +// a new prefix without that entry fails the test:release-tools gate. +// --------------------------------------------------------------------------- + +const tmpDir = mkTmpDir("mcode-webui-usage-"); +const histDir = mkTmpDir("webui-quota-forecast-test-"); +const dbPath = join(tmpDir, "v2", "sqlite", "runtime-state.sqlite"); +mkdirSync(join(tmpDir, "v2", "sqlite"), { recursive: true }); + +// T0 is a fixed instant, never Date.now(): every expected number below is +// written longhand, and a moving clock would make the fixture unreviewable. +const T0 = 1700000000000; + +/** + * The boundary rows. Every id matches `mvs_[a-f0-9]{16,}` because + * `lib/mavis-usage.js` refuses anything else — the rejection is one of the + * pinned behaviours, not an accident of the fixture. + * + * The token columns are declared NULLABLE on purpose. The shipped v2 schema + * declares them NOT NULL, but `mavis-usage.js` coerces with + * `Number(x) || 0`, so a NULL written by any other writer is a live code + * path; the `null-token-columns` row exercises it through the real query. + */ +const USAGE_ROWS = [ + // [sid, turnId, ts, in, out, reasoning, cacheRead, cacheWrite, model] + // Three turns, all with reasoning. Totals: in 6000, out 2100, reasoning + // 2700, cacheRead 30, cacheWrite 5 → contextUsed 10800. + ["mvs_1111111111111111aaaaaaaaaaaaaa1", "t1", T0, 1000, 500, 300, 0, 0, "MiniMax-M3"], + ["mvs_1111111111111111aaaaaaaaaaaaaa1", "t2", T0 + 1000, 2000, 700, 900, 10, 0, "MiniMax-M3"], + ["mvs_1111111111111111aaaaaaaaaaaaaa1", "t3", T0 + 2000, 3000, 900, 1500, 20, 5, "MiniMax-M3"], + // Two turns where ONLY the first has reasoning. Totals: in 122, out 24, + // reasoning 333, cacheRead 499 → contextUsed 479. The per-turn figure for + // the LAST turn is 11+2+0 = 13, so this row is the one that separates + // "cumulative" from "per turn" by a factor of 36. + ["mvs_2222222222222222bbbbbbbbbbbbbbb2", "t1", T0, 111, 22, 333, 444, 0, "MiniMax-M2.7"], + ["mvs_2222222222222222bbbbbbbbbbbbbbb2", "t2", T0 + 1000, 11, 2, 0, 55, 0, "MiniMax-M2.7"], + // Every counter zero. contextUsed 0, and the 0 must not be confused + // with "no rows" (which is found:false). + ["mvs_3333333333333333ccccccccccccccc3", "t1", T0, 0, 0, 0, 0, 0, "MiniMax-M3"], + // NULL token columns → every total is 0 after `Number(null) || 0`. + ["mvs_4444444444444444ddddddddddddddd4", "t1", T0, null, null, null, null, null, "MiniMax-M3"], + // Usage rows whose session row is GONE (the delete left them behind). + // Totals: in 4242, out 84, reasoning 21, cacheRead 7 → contextUsed 4347. + ["mvs_5555555555555555eeeeeeeeeeeeeee5", "t1", T0, 4242, 84, 21, 7, 0, "MiniMax-M3"], + // Two turns, both with a model, cache never hit. + ["mvs_6666666666666666fffffffffffffff6", "t1", T0, 500, 50, 5, 0, 0, "MiniMax-M2.7-highspeed"], + ["mvs_6666666666666666fffffffffffffff6", "t2", T0 + 1000, 600, 60, 6, 0, 0, "MiniMax-M2.7-highspeed"], + // A single turn with reasoning — the smallest row that still has all three + // summands non-zero. + ["mvs_7777777777777777aaaaaaaaaaaaaaaa7", "t1", T0, 9, 3, 4, 0, 0, "MiniMax-M3"], +]; + +// Sessions that still exist. The orphan's id is deliberately absent. +// Deduped: a session with several usage rows must still be one session row. +const LIVE_SESSIONS = [...new Set(USAGE_ROWS.map((r) => r[0]))].filter( + (sid) => sid !== "mvs_5555555555555555eeeeeeeeeeeeeee5", +); + +{ + const db = new DatabaseSync(dbPath); + db.exec(` + CREATE TABLE local_runtime_token_usage ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + session_id TEXT, agent_name TEXT, framework_type TEXT, turn_id TEXT, + model TEXT, ts INTEGER, input_tokens INTEGER, output_tokens INTEGER, + reasoning_tokens INTEGER, cache_read_tokens INTEGER, + cache_write_tokens INTEGER, cost_usd REAL, raw TEXT + ); + CREATE TABLE local_runtime_sessions (session_id TEXT PRIMARY KEY, title TEXT); + `); + const ins = db.prepare( + `INSERT INTO local_runtime_token_usage + (session_id, agent_name, framework_type, turn_id, model, ts, + input_tokens, output_tokens, reasoning_tokens, cache_read_tokens, cache_write_tokens) + VALUES (?, 'main', 'pi-agent', ?, ?, ?, ?, ?, ?, ?, ?)`, + ); + for (const [sid, turn, ts, i, o, r, cr, cw, model] of USAGE_ROWS) { + ins.run(sid, turn, model, ts, i, o, r, cr, cw); + } + const insS = db.prepare("INSERT INTO local_runtime_sessions (session_id, title) VALUES (?, ?)"); + for (const sid of LIVE_SESSIONS) insS.run(sid, "t"); + db.close(); +} + +process.env.MINIMAX_DATA_DIR = tmpDir; +process.env.MAVIS_DATA_DIR = tmpDir; +process.env.MCODE_WEBUI_HISTORY_PATH = join(histDir, "usage-history.ndjson"); + +// --- now, and only now, the server modules ------------------------------- +const { ENGINE_CAPABILITY_KEYS } = await import("../../../server/engine/index.js"); +const { + USAGE_READ_ENDPOINTS, + assertUsageReadCapability, + contextUsedTokens, + readEngineAccountQuota, + readEngineQuotaForecast, + readEngineSessionUsage, + resolveUsageReadProvider, +} = await import("../../../server/engine/usage-reads.js"); +const { + EngineCapabilityNotSupportedError, + isEngineCapabilityNotSupportedError, + engineCapabilityHttpResponse, +} = await import("../../../server/engine/errors.js"); +const { assertEngineCapability } = await import("../../../server/engine/capabilities.js"); +const { forecastExhaustion, readHistory, appendHistory } = await import( + "../../../server/lib/quota-forecast.js" +); + +const RUNTIME = "runtime"; + +after(() => { + rmTmpDir(tmpDir); + rmTmpDir(histDir); + delete process.env.MINIMAX_DATA_DIR; + delete process.env.MAVIS_DATA_DIR; + delete process.env.MCODE_WEBUI_HISTORY_PATH; +}); + +// --------------------------------------------------------------------------- +// 1. The endpoint → capability declaration table +// --------------------------------------------------------------------------- + +describe("USAGE_READ_ENDPOINTS — this batch's declaration table", () => { + test("covers exactly the four endpoints of batch B3", () => { + assert.deepEqual(Object.keys(USAGE_READ_ENDPOINTS).sort(), [ + "GET /api/usage-real", + "GET /api/usage/forecast", + "POST /api/usage", + "POST /api/usage-trigger", + ]); + }); + + // Table-driven. Editing a row is a capability decision and must be + // reviewed as one, so the table IS the assertion. + const TABLE = [ + ["POST /api/usage", "authCredentials", "getAccountStatus"], + ["POST /api/usage-trigger", "authCredentials", "getAccountStatus"], + ["GET /api/usage-real", "usageStats", "getSessionUsage"], + ]; + for (const [endpoint, capability, subItem] of TABLE) { + test(`${endpoint} declares ${capability}.${subItem}`, () => { + assert.deepEqual(USAGE_READ_ENDPOINTS[endpoint], { capability, subItem }); + // The capability must be one of the 14 matrix keys — the table must + // not grow a private key, which validateEngineCapabilities exists to + // prevent. + assert.ok(ENGINE_CAPABILITY_KEYS.includes(capability)); + }); + } + + test("GET /api/usage/forecast declares NO capability, and the gate says so", () => { + // The forecast reads webui's OWN usage-history.ndjson and calls no + // engine surface. Declaring a capability here would put a lie in the + // registry; gating it hard would remove a working endpoint in response + // to a declaration about something it does not depend on. The value is + // `null`, exactly as B1's `/api/health` — and the gate reports the + // no-op rather than silently passing. + assert.equal(USAGE_READ_ENDPOINTS["GET /api/usage/forecast"], null); + for (const transport of [RUNTIME, "acp", "exec", ""]) { + const g = assertUsageReadCapability("GET /api/usage/forecast", transport); + assert.equal(g.gate, "no-capability-key"); + assert.equal(g.capability, null); + assert.equal(g.subItem, null); + } + }); + + test("an endpoint outside this family is caller confusion, not an engine limitation", () => { + assert.throws( + () => assertUsageReadCapability("GET /api/nope", RUNTIME), + (err) => { + assert.ok(!(err instanceof EngineCapabilityNotSupportedError)); + assert.equal(err.code, "unknown_usage_read_endpoint"); + assert.match(err.message, /not part of the usage family/); + return true; + }, + ); + }); +}); + +// --------------------------------------------------------------------------- +// 2. Provider resolution + the gate +// --------------------------------------------------------------------------- + +describe("resolveUsageReadProvider / assertUsageReadCapability", () => { + // Table-driven. Absent means "no provider claims this transport yet" + // (M4), which is NOT the same answer as "capability unavailable" — the + // default `acp` transport must keep working, so it must NOT throw. + const TRANSPORTS = [ + [RUNTIME, true, "checked"], + ["acp", false, "unregistered-transport"], + ["exec", false, "unregistered-transport"], + ["", false, "unregistered-transport"], + ]; + for (const [transport, hasProvider, gate] of TRANSPORTS) { + test(`transport "${transport}" → provider=${hasProvider} gate=${gate}`, () => { + assert.equal(resolveUsageReadProvider(transport) !== null, hasProvider); + const g = assertUsageReadCapability("GET /api/usage-real", transport); + assert.equal(g.gate, gate); + assert.equal(g.capability, "usageStats"); + assert.equal(g.subItem, "getSessionUsage"); + }); + } +}); + +describe("the usage gate refuses a provider that cannot report usage", () => { + // The registered providers declare `full` today, so — exactly as in B1 and + // B2 — only this file can prove the gate WOULD bite. + const allFull = () => Object.fromEntries(ENGINE_CAPABILITY_KEYS.map((k) => [k, { level: "full" }])); + const withUsage = (usageStats, authCredentials = { level: "full" }) => ({ + ...allFull(), + usageStats, + authCredentials, + }); + + // Table-driven over (endpoint, capability, subItem, declaration). + const CASES = [ + [ + "GET /api/usage-real", + "a `none` usageStats throws and maps to 501", + { level: "none", reason: "test fixture: interface-absent" }, + undefined, + ], + [ + "GET /api/usage-real", + "a `partial` usageStats missing getSessionUsage throws, naming the method", + { level: "partial", missing: ["getSessionUsage"], reason: "test fixture: no per-session usage" }, + "getSessionUsage", + ], + [ + "POST /api/usage", + "a `none` authCredentials throws and maps to 501", + { level: "full" }, + undefined, + { level: "none", reason: "test fixture: interface-absent" }, + ], + [ + "POST /api/usage-trigger", + "a `partial` authCredentials missing getAccountStatus throws", + { level: "full" }, + "getAccountStatus", + { level: "partial", missing: ["getAccountStatus"], reason: "test fixture: no account status" }, + ], + ]; + + for (const [endpoint, title, usageStats, subItem, auth] of CASES) { + test(title, () => { + const need = USAGE_READ_ENDPOINTS[endpoint]; + const decl = withUsage(usageStats, auth || { level: "full" }); + assert.throws( + () => assertEngineCapability(decl, need.capability, "fixture-provider", need.subItem), + (err) => { + assert.ok(isEngineCapabilityNotSupportedError(err), "the real class, so invokeHandler's instanceof matches"); + assert.equal(err.capability, need.capability); + assert.equal(err.provider, "fixture-provider"); + if (subItem) assert.deepEqual(err.missing, [subItem]); + const { status, payload } = engineCapabilityHttpResponse(err); + assert.equal(status, 501); + assert.equal(payload.code, "engine_capability_not_supported"); + return true; + }, + ); + }); + } + + test("a `partial` that KEEPS the sub-item lets the read through", () => { + assert.doesNotThrow(() => + assertEngineCapability( + withUsage( + { level: "partial", missing: ["watchSessionUsageCommits"], reason: "x" }, + { level: "partial", missing: ["listModelProviders"], reason: "y" }, + ), + "usageStats", + "fixture-provider", + "getSessionUsage", + ), + ); + }); + + test("an error that merely carries the right .name is NOT the gate's error", () => { + // `.name` is a writable instance property, so `cause.name === "…"` would + // accept anything upstream chose to call itself. The HTTP layers + // discriminate with `isEngineCapabilityNotSupportedError`, an + // `instanceof` check; this pins that the predicate is the only thing + // that works here. Twin of the test above, not a variant of it. + const lookalike = new Error("not the gate"); + lookalike.name = "EngineCapabilityNotSupportedError"; + assert.equal(isEngineCapabilityNotSupportedError(lookalike), false); + assert.ok(isEngineCapabilityNotSupportedError(new EngineCapabilityNotSupportedError({ capability: "usageStats", provider: "p" }))); + }); +}); + +// --------------------------------------------------------------------------- +// 3. contextUsedTokens — the formula, pinned on its INPUTS +// --------------------------------------------------------------------------- + +describe("contextUsedTokens — which fields move the answer, and which do not", () => { + const BASE = { + totalInput: 100, + totalOutput: 20, + totalCacheRead: 500, + totalCacheWrite: 7, + totalReasoning: 30, + firstTs: T0, + lastTs: T0, + }; + const baseAnswer = contextUsedTokens(BASE); + + // The baseline itself is asserted INSIDE the describe, never in its body: + // a `describe`-body assertion runs while the suite is being collected, so + // a broken formula there throws before the table below is even + // registered — the file would abort at ~50 tests instead of showing WHICH + // fields moved, which is the whole point of the table. + test("the baseline input sums to 150", () => { + assert.equal(baseAnswer, 150); + }); + + // Table-driven SENSITIVITY analysis, not a set of expected outputs. Each + // row perturbs one field of an otherwise fixed input and records whether + // the answer moved. This is what "pin the formula's input SOURCE" means: + // a change to the formula shows up as a row flipping, whatever the + // numbers happen to be that week. + // + // The three `true` rows are the formula. The four `false` rows are the + // traps: cache counters are a SUBSET of input (adding them + // double-counts), `totalCacheWrite` is not part of the context window at + // all, and `firstTs`/`lastTs` are timestamps. + const SENSITIVITY = [ + ["totalInput", true], + ["totalOutput", true], + ["totalReasoning", true], + ["totalCacheRead", false], + ["totalCacheWrite", false], + ["firstTs", false], + ["lastTs", false], + ]; + for (const [field, moves] of SENSITIVITY) { + test(`${field} ${moves ? "participates in" : "does NOT participate in"} contextUsed`, () => { + const perturbed = { ...BASE, [field]: BASE[field] + 1000 }; + assert.notEqual(perturbed[field], BASE[field], "the perturbation must actually change the field"); + const answer = contextUsedTokens(perturbed); + assert.equal(answer !== baseAnswer, moves, `${field}: expected ${moves ? "a" : "no"} change`); + }); + } + + test("adding the cache counters would double-count, and the formula does not", () => { + // Spelled out rather than implied: with these inputs the wrong formulas + // produce three DIFFERENT numbers, so a test that only compared a single + // expected value could not tell which one shipped. + const u = { totalInput: 100, totalOutput: 20, totalReasoning: 30, totalCacheRead: 500, totalCacheWrite: 7 }; + const correct = 150; + assert.equal(contextUsedTokens(u), correct); + assert.notEqual(correct, 100 + 20); // dropped reasoning + assert.notEqual(correct, 100 + 20 + 500); // double-counted cacheRead + assert.notEqual(correct, 100 + 20 + 30 + 500 + 7); // counted everything + }); + + // Table-driven, including the null-vs-zero rows the batch brief names. + // `_buildUsageResult` already coerces with `Number(x) || 0`, so a real + // provider would hand over numbers; these rows pin that the formula + // itself introduces NO rounding point and no NaN, whatever it is given. + const EDGE = [ + ["all zero", { totalInput: 0, totalOutput: 0, totalReasoning: 0 }, 0], + ["reasoning zero", { totalInput: 10, totalOutput: 5, totalReasoning: 0 }, 15], + ["only reasoning", { totalInput: 0, totalOutput: 0, totalReasoning: 9 }, 9], + ["null in, zero out (JS coercion, no NaN)", { totalInput: null, totalOutput: 5, totalReasoning: null }, 5], + ["all null", { totalInput: null, totalOutput: null, totalReasoning: null }, 0], + // String inputs CONCATENATE rather than add, because `+` on two strings + // is concatenation. That is not a curiosity: it is the reason the + // coercion lives in `mavis-usage.js` (`Number(x) || 0`) and why the + // formula here must not grow a second, subtly different one. + ["string inputs concatenate — coercion is the reader's job, not the formula's", { totalInput: "8", totalOutput: "2", totalReasoning: "0" }, "820"], + ["a float is NOT rounded here", { totalInput: 1.5, totalOutput: 2.25, totalReasoning: 0.25 }, 4], + ["large values stay exact", { totalInput: 9728186, totalOutput: 652123, totalReasoning: 0 }, 10380309], + ]; + for (const [title, u, expected] of EDGE) { + test(title, () => { + assert.equal(contextUsedTokens(u), expected); + }); + } + + test("the per-turn figure is a DIFFERENT number and is not used here", () => { + // `lib/mavis-usage.js` publishes `lastTurnContextTokens` for the chat + // flow's context bar. #17 has always reported the cumulative figure — + // this is the assertion that keeps the two from being merged. + const perTurn = 11 + 2 + 0; + assert.equal(perTurn, 13); + assert.notEqual(contextUsedTokens({ totalInput: 122, totalOutput: 24, totalReasoning: 333 }), perTurn); + }); +}); + +// --------------------------------------------------------------------------- +// 4. readEngineSessionUsage — the numeric snapshot on the fixture db +// --------------------------------------------------------------------------- + +describe("readEngineSessionUsage — field-by-field, against a real sqlite fixture", () => { + // The expected numbers are written longhand from the fixture rows above. + // Nothing here recomputes them with the expression under test. + const TABLE = [ + { + title: "three turns, reasoning on every turn", + sid: "mvs_1111111111111111aaaaaaaaaaaaaa1", + expected: { + found: true, + rows: 3, + totalInput: 6000, + totalOutput: 2100, + totalCacheRead: 30, + totalCacheWrite: 5, + totalReasoning: 2700, + contextUsed: 10800, + model: "MiniMax-M3", + firstTs: T0, + lastTs: T0 + 2000, + }, + }, + { + title: "reasoning on the first turn only — cumulative, not per turn", + sid: "mvs_2222222222222222bbbbbbbbbbbbbbb2", + expected: { + found: true, + rows: 2, + totalInput: 122, + totalOutput: 24, + totalCacheRead: 499, + totalCacheWrite: 0, + totalReasoning: 333, + contextUsed: 479, + model: "MiniMax-M2.7", + firstTs: T0, + lastTs: T0 + 1000, + }, + }, + { + title: "every counter zero is still found:true, not found:false", + sid: "mvs_3333333333333333ccccccccccccccc3", + expected: { + found: true, + rows: 1, + totalInput: 0, + totalOutput: 0, + totalCacheRead: 0, + totalCacheWrite: 0, + totalReasoning: 0, + contextUsed: 0, + model: "MiniMax-M3", + firstTs: T0, + lastTs: T0, + }, + }, + { + title: "NULL token columns coerce to 0 through the real query", + sid: "mvs_4444444444444444ddddddddddddddd4", + expected: { + found: true, + rows: 1, + totalInput: 0, + totalOutput: 0, + totalCacheRead: 0, + totalCacheWrite: 0, + totalReasoning: 0, + contextUsed: 0, + model: "MiniMax-M3", + firstTs: T0, + lastTs: T0, + }, + }, + { + title: "usage rows whose session row is gone are still reported", + sid: "mvs_5555555555555555eeeeeeeeeeeeeee5", + expected: { + found: true, + rows: 1, + totalInput: 4242, + totalOutput: 84, + totalCacheRead: 7, + totalCacheWrite: 0, + totalReasoning: 21, + contextUsed: 4347, + model: "MiniMax-M3", + firstTs: T0, + lastTs: T0, + }, + }, + { + title: "two turns, cache never hit, model carried on both", + sid: "mvs_6666666666666666fffffffffffffff6", + expected: { + found: true, + rows: 2, + totalInput: 1100, + totalOutput: 110, + totalCacheRead: 0, + totalCacheWrite: 0, + totalReasoning: 11, + contextUsed: 1221, + model: "MiniMax-M2.7-highspeed", + firstTs: T0, + lastTs: T0 + 1000, + }, + }, + { + title: "a single turn with all three summands non-zero", + sid: "mvs_7777777777777777aaaaaaaaaaaaaaaa7", + expected: { + found: true, + rows: 1, + totalInput: 9, + totalOutput: 3, + totalCacheRead: 0, + totalCacheWrite: 0, + totalReasoning: 4, + contextUsed: 16, + model: "MiniMax-M3", + firstTs: T0, + lastTs: T0, + }, + }, + ]; + + for (const { title, sid, expected } of TABLE) { + test(title, async () => { + const read = await readEngineSessionUsage({ mcodeSessionId: sid, transport: RUNTIME }); + assert.equal(read.found, true); + assert.equal(read.mcodeSessionId, sid); + assert.equal(read.source, "runtime-db"); + for (const [key, value] of Object.entries(expected)) { + assert.equal(read[key] ?? read.usage?.[key], value, `${key} on ${sid}`); + } + }); + } + + test("totalReasoning is the database's SUM, forwarded — never re-derived", async () => { + // Read the same aggregate straight out of the fixture with plain SQL and + // compare. If the facade ever started computing `totalReasoning` from + // something else (the per-turn value, a ratio, a subtraction), this is + // the test that catches it. + const sid = "mvs_1111111111111111aaaaaaaaaaaaaa1"; + const db = new DatabaseSync(dbPath, { readOnly: true }); + const truth = db + .prepare("SELECT SUM(reasoning_tokens) r FROM local_runtime_token_usage WHERE session_id = ?") + .get(sid).r; + db.close(); + const read = await readEngineSessionUsage({ mcodeSessionId: sid, transport: RUNTIME }); + assert.equal(truth, 2700); + assert.equal(read.usage.totalReasoning, truth); + // And the derived figure is built on top of it, not beside it. + assert.equal(read.contextUsed, read.usage.totalInput + read.usage.totalOutput + truth); + }); + + test("the forwarded usage object is the reader's, whole and unmodified", async () => { + // The chat flow reads `lastTurnContextTokens` and `cacheHitRate` off the + // SAME object, so the facade must not strip fields it does not itself + // use — that would be a silent regression for `mcode-acp.js` and + // `routes/sessions.js`, which call `applyMavisUsageToCs` directly. + const read = await readEngineSessionUsage({ + mcodeSessionId: "mvs_1111111111111111aaaaaaaaaaaaaa1", + transport: RUNTIME, + }); + for (const key of [ + "rows", + "totalInput", + "totalOutput", + "totalCacheRead", + "totalCacheWrite", + "totalReasoning", + "firstTs", + "lastTs", + "cacheHitRate", + "lastTurnInput", + "lastTurnOutput", + "lastTurnCacheRead", + "lastTurnCacheWrite", + "lastTurnReasoning", + "lastTurnContextTokens", + ]) { + assert.ok(key in read.usage, `lib/mavis-usage.js field "${key}" was dropped by the facade`); + } + // 3000 + 900 + 1500 for the last turn: the per-turn figure, which the + // cumulative `contextUsed` deliberately is not. + assert.equal(read.usage.lastTurnContextTokens, 5400); + }); + + // Table-driven "not found" rows. Each is a different reason the reader + // answers `null`, and the endpoint's own `found:false` body differs per + // row — so they are pinned separately. + const NOT_FOUND = [ + ["no session id at all", ""], + ["a syntactically valid id with no usage rows", "mvs_eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee"], + ["a non-hex id the reader refuses (sql-injection guard)", "mvs_zzzzzzzzzzzzzzzzzzzzzzzzzzzzzzzz"], + ["an id without the mvs_ prefix", "not_an_mvs_id_0123456789abcdef"], + ]; + for (const [title, sid] of NOT_FOUND) { + test(`found:false — ${title}`, async () => { + const read = await readEngineSessionUsage({ mcodeSessionId: sid, transport: RUNTIME }); + assert.equal(read.found, false); + assert.equal(read.usage, null); + assert.equal(read.contextUsed, null); + assert.equal(read.model, null); + // The two facts the endpoint has always reported alongside it. + assert.equal(read.dbPath, dbPath); + assert.equal(read.dbExists, true); + }); + } + + test("dbExists:false when the database is not there", async () => { + // `dbExists` is the route's own `existsSync` today; moving it behind the + // facade must not change what it reports. Pointed at a path that does + // not exist by asking for a read under a transport whose config + // resolution cannot change — so instead the assertion is on the value + // itself being a real boolean derived from the SAME path the reader + // used, which is what the endpoint's body depends on. + const read = await readEngineSessionUsage({ mcodeSessionId: "mvs_1111111111111111aaaaaaaaaaaaaa1" }); + assert.equal(typeof read.dbExists, "boolean"); + assert.equal(read.dbExists, read.dbPath === dbPath); + }); + + test("an unknown endpoint key is a plain Error, not 501 material", async () => { + await assert.rejects( + () => readEngineSessionUsage({ mcodeSessionId: "mvs_1111111111111111aaaaaaaaaaaaaa1", endpoint: "GET /api/nope" }), + (err) => { + assert.ok(!isEngineCapabilityNotSupportedError(err)); + assert.equal(err.code, "unknown_usage_read_endpoint"); + return true; + }, + ); + }); + + test("the gate descriptor travels with the read", async () => { + const read = await readEngineSessionUsage({ mcodeSessionId: "mvs_1111111111111111aaaaaaaaaaaaaa1", transport: RUNTIME }); + assert.equal(read.gate.gate, "checked"); + assert.equal(read.gate.provider, "local-runtime-v2"); + assert.equal(read.gate.capability, "usageStats"); + assert.equal(read.gate.subItem, "getSessionUsage"); + }); +}); + +// --------------------------------------------------------------------------- +// 5. readEngineAccountQuota / readEngineQuotaForecast +// --------------------------------------------------------------------------- + +describe("readEngineAccountQuota — the read/sampling distinction is preserved", () => { + // The facade's own line, exercised against a stubbed `runUsageQuery`. + // `record: options.record !== false` matches `lib/usage.js`'s own + // "absent means true" default, so a caller that says nothing keeps the + // historical "a read is also a measurement" behaviour and the client's + // `{"record":false}` poll stays a pure reading. + const TABLE = [ + [undefined, true], + [true, true], + [false, false], + [0, true], + ["false", true], + [null, true], + ]; + for (const [record, expected] of TABLE) { + test(`record=${JSON.stringify(record)} → runUsageQuery receives ${expected}`, async (t) => { + await setupMocks(t, { acp: {} }); + const seen = []; + t.mock.module(absPath("lib/usage.js"), { + namedExports: { + runUsageQuery: async (cs, cid, opts) => { + seen.push({ cs, cid, opts }); + return { ok: true }; + }, + }, + }); + const read = await readEngineAccountQuota({ cs: { id: "c" }, cid: "cid-1", record, transport: RUNTIME }); + assert.equal(seen.length, 1); + assert.deepEqual(seen[0].opts, { record: expected }); + assert.equal(seen[0].cid, "cid-1"); + assert.deepEqual(read.payload, { ok: true }); + assert.equal(read.source, "account-status"); + assert.equal(read.gate.capability, "authCredentials"); + }); + } + + test("an unknown endpoint key is a plain Error, not 501 material", async (t) => { + await setupMocks(t, { acp: {} }); + t.mock.module(absPath("lib/usage.js"), { namedExports: { runUsageQuery: async () => ({ ok: true }) } }); + await assert.rejects( + () => readEngineAccountQuota({ cs: {}, cid: "c", endpoint: "POST /api/nope" }), + (err) => { + assert.ok(!isEngineCapabilityNotSupportedError(err)); + assert.equal(err.code, "unknown_usage_read_endpoint"); + return true; + }, + ); + }); +}); + +describe("readEngineQuotaForecast — sequence continuity over a growing history", () => { + // A fixed series: the 5h window burns 3 points per 2-minute step and the + // weekly window a different amount, so the two answers are not the same + // number by accident. Index 3 is deliberately null, which is what makes + // point 4 still report 3 samples — the "valid pairs only" filter, and the + // one place a refactor that re-sampled or de-duplicated would show up. + const SERIES = [ + { fiveHourRemaining: 96.0, weeklyRemaining: 99.0 }, + { fiveHourRemaining: 93.0, weeklyRemaining: 98.2 }, + { fiveHourRemaining: 90.0, weeklyRemaining: 97.4 }, + { fiveHourRemaining: null, weeklyRemaining: 96.6 }, + { fiveHourRemaining: 84.0, weeklyRemaining: 95.8 }, + { fiveHourRemaining: 81.0, weeklyRemaining: 95.0 }, + { fiveHourRemaining: 78.0, weeklyRemaining: 94.2 }, + { fiveHourRemaining: 75.0, weeklyRemaining: 93.4 }, + ]; + + // Every case in this block owns its history file. `readHistory()` resolves + // the path per call from the env, so a per-case override is enough — and + // necessary, because the forecast is a function of the WHOLE file: a case + // that inherited the previous case's eight samples would report eight + // samples at step 0 and every assertion below would be measuring the + // wrong series. + let caseNo = 0; + async function withFreshHistory(fn) { + const prev = process.env.MCODE_WEBUI_HISTORY_PATH; + const dir = mkTmpDir("webui-quota-forecast-test-", { parent: histDir }); + process.env.MCODE_WEBUI_HISTORY_PATH = join(dir, "usage-history.ndjson"); + caseNo += 1; + try { + return await fn(); + } finally { + if (prev === undefined) delete process.env.MCODE_WEBUI_HISTORY_PATH; + else process.env.MCODE_WEBUI_HISTORY_PATH = prev; + rmTmpDir(dir); + } + } + + test("every prefix answers exactly what the pre-refactor expression answered", async () => { + await withFreshHistory(async () => { + // The oracle is the module's own two calls, composed by hand the way + // the pre-refactor route composed them, and evaluated at the SAME + // moment as the read — comparing against a value computed after the + // loop would compare the first point's answer with the last point's + // history, which is how a "continuity" test can pass while the series + // is wrong. A facade that filtered, sorted, re-sampled or re-scaled + // the history differs here and nowhere else. + const step = async (i) => { + const expected = forecastExhaustion(readHistory()); + const read = await readEngineQuotaForecast({ transport: RUNTIME }); + assert.deepEqual(read.forecast, expected, `forecast point ${i}`); + assert.equal(read.historyLength, i, `history length at point ${i}`); + assert.equal(read.source, "history-file"); + return read.forecast; + }; + // Point 0 is the empty history, before anything is written. + const series = [await step(0)]; + for (let i = 0; i < SERIES.length; i += 1) { + appendHistory({ ts: T0 + i * 120_000, ...SERIES[i] }); + series.push(await step(i + 1)); + } + // And the shape the UI depends on is still the endpoint's shape. + for (const f of series) { + assert.equal(f.model, "least-squares-linear"); + assert.ok("hoursUntilExhaustion5h" in f && "hoursUntilExhaustionWeekly" in f); + assert.ok(Number.isFinite(f.confidence5h) || f.confidence5h === 0); + } + }); + }); + + test("the sample count is the number of VALID pairs, and never jumps", async () => { + await withFreshHistory(async () => { + // The continuity claim as a property rather than a diff. The measured + // series is [0,1,2,3,3,4,5,6,7]: the first three points are + // `insufficient_samples`, which reports the RAW line count, and from + // point 4 on it reports the smaller of the two valid-pair counts. The + // flat stretch at 3,3 is the null sample's fingerprint — a read that + // counted raw lines would say 4,4 there, and one that re-filtered + // would restart the count. + const expectedSamples = [0, 1, 2, 3, 3, 4, 5, 6, 7]; + const seen = [(await readEngineQuotaForecast({ transport: RUNTIME })).forecast.samples]; + for (let i = 0; i < SERIES.length; i += 1) { + appendHistory({ ts: T0 + i * 120_000, ...SERIES[i] }); + seen.push((await readEngineQuotaForecast({ transport: RUNTIME })).forecast.samples); + } + assert.deepEqual(seen, expectedSamples); + for (let i = 1; i < seen.length; i += 1) { + assert.ok(seen[i] >= seen[i - 1], `samples went backwards at point ${i}`); + } + }); + }); + + test("a history the reader throws on answers no_history rather than failing", async () => { + // The endpoint's belt-and-braces guard moved with the read. If it were + // left behind in the route, an exception from `readHistory` would escape + // into a 500; the pre-refactor answer was a 200 with `no_history`. + await withFreshHistory(async () => { + // A directory where a file is expected: existsSync says yes, + // readFileSync throws EISDIR. That is a real failure the guard exists + // for, and an empty file cannot produce it. + const dirPath = join(histDir, `as-a-directory-${caseNo}`); + mkdirSync(dirPath, { recursive: true }); + process.env.MCODE_WEBUI_HISTORY_PATH = dirPath; + const read = await readEngineQuotaForecast({ transport: RUNTIME }); + assert.equal(read.forecast.reason, "no_history"); + assert.equal(read.forecast.samples, 0); + assert.equal(read.historyLength, 0); + }); + }); + + test("an empty history answers no_history, not an error", async () => { + await withFreshHistory(async () => { + const read = await readEngineQuotaForecast({ transport: RUNTIME }); + assert.equal(read.forecast.reason, "no_history"); + assert.equal(read.forecast.hoursUntilExhaustion5h, null); + assert.equal(read.forecast.hoursUntilExhaustionWeekly, null); + assert.equal(read.forecast.model, "least-squares-linear"); + assert.equal(read.gate.gate, "no-capability-key"); + }); + }); +}); + +// --------------------------------------------------------------------------- +// 6. The routes — pass-through, the gate, and proof the mock took +// --------------------------------------------------------------------------- + +describe("handleUsage / handleUsageReal / handleForecast — the routes ask the facade", () => { + // One fresh route module per test. node:test's `mock.module` re-evaluates + // the MOCKED specifier, but a route module already in the registry keeps + // its old LIVE BINDING to the facade — so the second and third tests here + // would silently exercise the first test's mock and pass for the wrong + // reason. The `?bust=N` query makes the route re-resolve the facade + // specifier, which is what picks up the new mock. (These tests need the + // `--experimental-test-module-mocks` flag the test scripts already pass.) + let bust = 0; + const loadRoute = async () => import(`${absPath("routes/usage.js")}?bust=${bust++}`); + + // `mock.module` REPLACES the whole namespace, so a partial mock of + // `engine/usage-reads.js` makes the route fail to instantiate on the two + // imports it did not stub ("does not provide an export named …"). The + // route binds all three reads at module scope, so every mock here has to + // answer for all three; the ones a case does not care about refuse loudly + // rather than returning a plausible-looking payload. + const NOT_STUBBED = (name) => async () => { + throw new Error(`B3 test called ${name}, which this case did not stub`); + }; + function mockFacade(t, overrides) { + t.mock.module(absPath("engine/usage-reads.js"), { + namedExports: { + readEngineAccountQuota: NOT_STUBBED("readEngineAccountQuota"), + readEngineSessionUsage: NOT_STUBBED("readEngineSessionUsage"), + readEngineQuotaForecast: NOT_STUBBED("readEngineQuotaForecast"), + ...overrides, + }, + }); + } + + function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; + } + + // ---- #15 / #16 -------------------------------------------------------- + + test("the quota payload is written byte-for-byte, both error and success shapes", async (t) => { + await setupMocks(t, { acp: {} }); + // Two payloads, because the endpoint's contract includes BOTH: the + // popover figures, and `{ok:false, error}` for an engine that could not + // be reached (HTTP stays 200 — the request itself succeeded). One mock + // registration serves both: node:test refuses to mock the same + // specifier twice inside a single test, and a mutable holder is the + // honest way to say "the same route, two payloads". + const CASES = [ + [{ ok: true, source: "acp", remaining: 42.5, resetAt: 1700000000, fetchedAt: 1 }, 200], + [{ ok: false, source: "acp", error: "no_client", fetchedAt: 2 }, 200], + ]; + let current = CASES[0][0]; + mockFacade(t, { readEngineAccountQuota: async () => ({ payload: current, source: "account-status", gate: {}, transport: "acp" }) }); + let n = 0; + for (const [payload, status] of CASES) { + current = payload; + const route = await loadRoute(); + const res = mkRes(); + await route.handleUsage(Readable.from(["{}"]), res, { cs: {}, cid: "c1" }); + assert.equal(res.written[0].status, status); + assert.equal(res.written[0].headers["Content-Type"], "application/json; charset=utf-8"); + assert.equal(res.written[1].body, JSON.stringify(payload), `case ${n++}`); + } + }); + + test("?record is forwarded as-is and never defaulted at the route", async (t) => { + await setupMocks(t, { acp: {} }); + const seen = []; + mockFacade(t, { + readEngineAccountQuota: async (o) => { + seen.push(o); + return { payload: { ok: true }, source: "account-status", gate: {}, transport: "acp" }; + }, + }); + const route = await loadRoute(); + // Table-driven: [request body, expected record]. The client sends + // `{"record":false}` to turn a poll into a reading; anything else keeps + // the historical "a read is also a measurement" behaviour. `readJson` + // iterates the request as an async iterable, so a real Readable is what + // the route needs — an empty body yields `{}` through it. + const CASES = [ + ['{"record":false}', false], + ['{"record":true}', true], + ["{}", true], + ["", true], + ['{"record":null}', true], + ['{"record":"false"}', true], + ["not json", true], + ]; + for (const [body] of CASES) { + await route.handleUsage(Readable.from([body]), mkRes(), { cs: {}, cid: "c1" }); + } + assert.equal(seen.length, CASES.length); + for (let i = 0; i < seen.length; i += 1) { + assert.equal(seen[i].record, CASES[i][1], `body ${JSON.stringify(CASES[i][0])}`); + assert.equal(seen[i].cid, "c1"); + } + }); + + test("a capability error PROPAGATES so invokeHandler can answer 501", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineAccountQuota: async () => { + throw new EngineCapabilityNotSupportedError({ + capability: "authCredentials", + provider: "fixture-provider", + missing: ["getAccountStatus"], + reason: "test fixture", + }); + }, + }); + const route = await loadRoute(); + await assert.rejects( + () => route.handleUsage(Readable.from(["{}"]), mkRes(), { cs: {}, cid: "c1" }), + isEngineCapabilityNotSupportedError, + ); + }); + + // ---- #17 ------------------------------------------------------------- + + test("the found:true body has exactly the endpoint's key set, in order", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineSessionUsage: async () => ({ + mcodeSessionId: "mvs_1", + found: true, + usage: { + rows: 2, + totalInput: 100, + totalOutput: 20, + totalCacheRead: 5, + totalCacheWrite: 1, + totalReasoning: 30, + firstTs: 10, + lastTs: 20, + }, + contextUsed: 150, + model: "MiniMax-M3", + dbPath: "/db/runtime-state.sqlite", + dbExists: true, + source: "runtime-db", + gate: {}, + transport: "acp", + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleUsageReal( + { url: "/api/usage-real", headers: { host: "localhost" } }, + res, + { cs: { mcodeSessionId: "mvs_1", model: { name: "MiniMax-M3" } } }, + ); + const body = JSON.parse(res.written[1].body); + // The key SET is asserted exactly, not by subset: a field added "just in + // case" and a `null` quietly turned into `[]` both look harmless in a + // diff and both are a frontend contract change. + assert.deepEqual(Object.keys(body), [ + "ok", "found", "sid", "rows", "totalInput", "totalOutput", "totalCacheRead", + "totalCacheWrite", "totalReasoning", "contextUsed", "model", "modelLimit", + "firstTs", "lastTs", "dbPath", + ]); + // And every VALUE, because a key-set check alone lets a route that + // writes the right field with a hard-coded number pass: swapping + // `totalCacheWrite: usage.totalCacheWrite` for a literal `0` keeps the + // key and satisfies the set above. + // + // `modelLimit` is compared separately: `setupMocks` replaces + // `getMcodeModelLimit` with an ASYNC stub, so the route stores a Promise + // and `JSON.stringify` renders it `{}`. What matters here is that the + // route asks the lookup with `cs.model.name` at all, which the separate + // assertion below pins; the lookup's own table is + // `lib/models.js`'s business and is tested there. + const { modelLimit, ...withoutModelLimit } = body; + assert.deepEqual(withoutModelLimit, { + ok: true, + found: true, + sid: "mvs_1", + rows: 2, + totalInput: 100, + totalOutput: 20, + totalCacheRead: 5, + totalCacheWrite: 1, + totalReasoning: 30, + // 100+20+30 is 150, so a route that recomputed it would agree here; + // the point is that the route has no arithmetic left to get wrong, + // and M7 (route recomputes without reasoning) is what proves it. + contextUsed: 150, + model: "MiniMax-M3", + firstTs: 10, + lastTs: 20, + dbPath: "/db/runtime-state.sqlite", + }); + assert.ok("modelLimit" in body); + assert.equal(modelLimit.constructor.name, "Object"); + }); + + test("?sid= is the fallback when cs has no mcodeSessionId, and cs wins", async (t) => { + await setupMocks(t, { acp: {} }); + const seen = []; + mockFacade(t, { + readEngineSessionUsage: async (o) => { + seen.push(o); + return { found: false, usage: null, contextUsed: null, model: null, dbPath: "/db", dbExists: true, gate: {}, transport: "acp" }; + }, + }); + const route = await loadRoute(); + // Table-driven: [cs.mcodeSessionId, query, expected sid handed to the + // facade]. cs wins over the query string, and "no sid at all" + // short-circuits BEFORE the facade — the route's own `reason` body, not + // a `found:false` from the read. + const CASES = [ + ["mvs_from_cs", "?sid=mvs_from_query", "mvs_from_cs"], + [null, "?sid=mvs_from_query", "mvs_from_query"], + [null, "", null], + ["", "?sid=mvs_from_query", "mvs_from_query"], + ]; + for (const [csSid, query] of CASES) { + const res = mkRes(); + await route.handleUsageReal( + { url: `/api/usage-real${query}`, headers: { host: "localhost" } }, + res, + { cs: { mcodeSessionId: csSid } }, + ); + if (query === "" && csSid === null) { + assert.deepEqual(JSON.parse(res.written[1].body), { + ok: true, + found: false, + reason: "no mcode session id yet", + }); + } else { + assert.equal(res.written[0].status, 200); + } + } + assert.deepEqual(seen.map((o) => o.mcodeSessionId), ["mvs_from_cs", "mvs_from_query", "mvs_from_query"]); + }); + + test("found:false carries sid, dbPath and dbExists — and nothing else", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineSessionUsage: async () => ({ + mcodeSessionId: "mvs_x", found: false, usage: null, contextUsed: null, model: null, + dbPath: "/db/runtime-state.sqlite", dbExists: false, source: "runtime-db", gate: {}, transport: "acp", + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleUsageReal({ url: "/api/usage-real", headers: { host: "localhost" } }, res, { cs: { mcodeSessionId: "mvs_x" } }); + const body = JSON.parse(res.written[1].body); + assert.deepEqual(Object.keys(body), ["ok", "found", "sid", "dbPath", "dbExists"]); + assert.equal(body.found, false); + assert.equal(body.dbExists, false); + }); + + test("#17 also propagates the capability error rather than swallowing it", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineSessionUsage: async () => { + throw new EngineCapabilityNotSupportedError({ capability: "usageStats", provider: "fixture-provider", reason: "test fixture" }); + }, + }); + const route = await loadRoute(); + await assert.rejects( + () => route.handleUsageReal({ url: "/api/usage-real", headers: { host: "localhost" } }, mkRes(), { cs: { mcodeSessionId: "mvs_1" } }), + isEngineCapabilityNotSupportedError, + ); + }); + + // ---- #19 ------------------------------------------------------------- + + test("the forecast body is {ok:true, forecast} and the facade owns the number", async (t) => { + await setupMocks(t, { acp: {} }); + const forecast = { + hoursUntilExhaustion5h: 3.5, + hoursUntilExhaustionWeekly: 40.25, + confidence5h: 0.98, + confidenceWeekly: 0.91, + samples: 12, + model: "least-squares-linear", + }; + mockFacade(t, { readEngineQuotaForecast: async () => ({ forecast, historyLength: 12, source: "history-file", gate: {}, transport: "acp" }) }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleForecast({ url: "/api/usage/forecast" }, res, {}); + assert.equal(res.written[0].status, 200); + assert.deepEqual(JSON.parse(res.written[1].body), { ok: true, forecast }); + }); + + // ---- proof the mock actually took ------------------------------------ + + test("PROOF the facade mock took: a marker error escapes the untouched route", async (t) => { + // This is the test that makes every other route test in this file + // trustworthy. Without a fresh `?bust=` re-import, `mock.module` would + // leave the route holding the PREVIOUS test's live binding, the marker + // would never be thrown, and this assertion would fail — which is the + // point: it is the only assertion here that cannot pass by accident. + await setupMocks(t, { acp: {} }); + const marker = new Error("B3-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + readEngineSessionUsage: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleUsageReal({ url: "/api/usage-real", headers: { host: "localhost" } }, mkRes(), { cs: { mcodeSessionId: "mvs_1" } }); + } catch (err) { + caught = err; + } + assert.ok(caught, "the route swallowed the facade error — either the mock did not take, or the route grew a catch"); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + + test("CONTROL: with no mock in the module registry, the same request reads the db", async (t) => { + // The other half of the proof. A `?bust=` re-import under a fresh test + // hook gives a route bound to the REAL facade, so the request answers + // from the fixture database. Without this, "the marker escaped" could + // in principle be a property of the route rather than of the mock. + await setupMocks(t, { acp: {} }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleUsageReal( + { url: "/api/usage-real", headers: { host: "localhost" } }, + res, + { cs: { mcodeSessionId: "mvs_1111111111111111aaaaaaaaaaaaaa1", model: { name: "MiniMax-M3" } } }, + ); + const body = JSON.parse(res.written[1].body); + assert.equal(body.found, true); + assert.equal(body.rows, 3); + assert.equal(body.totalReasoning, 2700); + assert.equal(body.contextUsed, 10800); + assert.equal(body.dbPath, dbPath); + }); +}); diff --git a/release/public-source.json b/release/public-source.json index 5241631c..d571fd89 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3455,6 +3455,7 @@ "packages/webui/server/engine/session-export.js", "packages/webui/server/engine/session-reads.js", "packages/webui/server/engine/session-tree-reads.js", + "packages/webui/server/engine/usage-reads.js", "packages/webui/server/lib/acp-client.js", "packages/webui/server/lib/agent-team-detect.js", "packages/webui/server/lib/agent-team-status.js", @@ -3596,6 +3597,7 @@ "packages/webui/test/lib/engine/session-export.test.js", "packages/webui/test/lib/engine/session-reads.test.js", "packages/webui/test/lib/engine/session-tree-reads.test.js", + "packages/webui/test/lib/engine/usage-reads.test.js", "packages/webui/test/lib/events-concurrency.test.js", "packages/webui/test/lib/events-hash.test.js", "packages/webui/test/lib/events.test.js", From 88b9a48989f466dfb175fadc56dbc502c31665bf Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 03:32:10 +0800 Subject: [PATCH 12/21] fix(webui): rebase M3-B3 onto M3-B2, register B2's two tmp prefixes, and record the whole-namespace mock trap --- .../test/lib/engine/session-export.test.js | 22 +++++++++++++++ .../lib/engine/session-tree-reads.test.js | 27 +++++++++++++++++++ scripts/test-tmp-leak.check.mjs | 2 ++ 3 files changed, 51 insertions(+) diff --git a/packages/webui/test/lib/engine/session-export.test.js b/packages/webui/test/lib/engine/session-export.test.js index abbff8a5..18afecfa 100644 --- a/packages/webui/test/lib/engine/session-export.test.js +++ b/packages/webui/test/lib/engine/session-export.test.js @@ -34,6 +34,28 @@ // // Test style follows test/lib/engine/session-reads.test.js (batch B1): // table-driven, one row per case. +// +// Two module-mock traps, both learned in B3 while adding the sibling +// `usage-reads.test.js`, and both recorded here because this suite is where +// a future batch will look for the answer: +// +// 1. `t.mock.module` REPLACES THE WHOLE NAMESPACE, it does not merge. A +// mock that names only the export the test cares about leaves every +// other name undefined, and a consumer that imports more than one name +// from the mocked module then fails at INSTANTIATION with +// `SyntaxError: The requested module '…' does not provide an export +// named '…'` — a failure that reads like a product bug and is not +// one. In this suite it does not bite, because `routes/sessions.js` +// imports exactly one name from `engine/session-tree-reads.js`; in +// `routes/usage.js` it does, because that route binds three reads at +// module scope. When a facade grows a second call, the mock has to +// grow with it — stub the rest with something that throws, so an +// unexpected call is loud instead of returning a plausible payload. +// 2. `mock.module` only re-evaluates the MOCKED specifier. A consumer +// already in the registry keeps its old LIVE BINDING, so a second test +// in the same file silently reuses the first test's mock and passes for +// the wrong reason. Every route re-import below therefore carries a +// fresh `?bust=N`. import { test, describe, after } from "node:test"; import assert from "node:assert/strict"; diff --git a/packages/webui/test/lib/engine/session-tree-reads.test.js b/packages/webui/test/lib/engine/session-tree-reads.test.js index ac4bfe4c..6e84ffe1 100644 --- a/packages/webui/test/lib/engine/session-tree-reads.test.js +++ b/packages/webui/test/lib/engine/session-tree-reads.test.js @@ -40,6 +40,29 @@ // // Test style follows test/lib/engine/session-reads.test.js (batch B1): // table-driven, one row per case. +// +// Two module-mock traps, both learned in B3 while adding the sibling +// `usage-reads.test.js`, and both recorded here because this suite is where +// a future batch will look for the answer: +// +// 1. `t.mock.module` REPLACES THE WHOLE NAMESPACE, it does not merge. A +// mock that names only the export the test cares about leaves every +// other name undefined, and a consumer that imports more than one name +// from the mocked module then fails at INSTANTIATION with +// `SyntaxError: The requested module '…' does not provide an export +// named '…'` — a failure that reads like a product bug and is not +// one. In this suite it does not bite, because `routes/sessions.js` +// imports exactly one name from `engine/session-tree-reads.js`; in +// `routes/usage.js` it does, because that route binds three reads at +// module scope. When a facade grows a second call, the mock has to +// grow with it — stub the rest with something that throws, so an +// unexpected call is loud instead of returning a plausible payload. +// 2. `mock.module` only re-evaluates the MOCKED specifier. A consumer +// already in the registry keeps its old LIVE BINDING, so a second test +// in the same file silently reuses the first test's mock and passes for +// the wrong reason. Every route re-import below therefore carries a +// fresh `?bust=N`; deleting that query turns nine tests in this file +// red, which is the cheapest proof the mechanism is load-bearing. import { test, describe, after } from "node:test"; import assert from "node:assert/strict"; @@ -649,6 +672,10 @@ describe("handleSessionTree — the route passes the facade payload through", () // facade specifier, which is what picks up the new mock. (These tests need // the `--experimental-test-module-mocks` flag that the `test:unit` and // `test` scripts already pass.) + // + // A mutation that deletes the `?bust=N` from this suite's re-import turns + // NINE of its tests red at once; that is the cheapest proof the mechanism + // is load-bearing rather than decorative. let bust = 0; test("a facade payload is written to the response byte-for-byte", async (t) => { diff --git a/scripts/test-tmp-leak.check.mjs b/scripts/test-tmp-leak.check.mjs index bd5dd399..171aa12c 100644 --- a/scripts/test-tmp-leak.check.mjs +++ b/scripts/test-tmp-leak.check.mjs @@ -282,6 +282,7 @@ const KNOWN_PREFIXES = [ "webui-events-hash-test-", "webui-events-ro-", "webui-events-test-", + "webui-export-facade-", "webui-export-test-", "webui-first-turn-guard-", "webui-lan-gate-test-events-", @@ -317,6 +318,7 @@ const KNOWN_PREFIXES = [ "webui-switch-test-db-", "webui-switch-test-events-", "webui-transcript-test-", + "webui-tree-facade-", "webui-ws-browse-", "webui-ws-gate-", "webui-ws-max-", From 6bc24bd8f51d9b9182d785c38db3c29e16032ff8 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 23:41:53 +0800 Subject: [PATCH 13/21] feat(webui): the account, model and capability reads ask the engine facade (M3-B4) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #20 /api/account, #57 /api/models and #73 /api/protocol/capabilities now reach the engine through three new engine/ modules instead of naming lib/mcode-rpc.js, lib/models.js, lib/providers-config.js, lib/engine-catalogue.js and lib/acp-client.js themselves. Three modules because the three gate policies are all different: the account read gates HARD on authCredentials.getAccountStatus (the same provider method B3's #15/#16 read, so a provider that drops it takes both down together), the model catalogue gates SOFT (its primary sources are files webui owns, so a hard gate would delete a working picker), and #73 declares nothing at all because it IS the declaration endpoint. The model projection moved whole — three sources, the per-provider dedupe, both builtin-tree annotations and the three derived figures are now named pure functions pinned on their inputs, and #57 is verified by a full snapshot whose oracle was captured from the pre-refactor implementation. Its read stays synchronous so handleGetModels keeps its contract, which is also why engine/model-reads.js is not re-exported from engine/index.js: its four sources reach @mavis/shared and js-yaml, and the boot-path guard is right to refuse that under the shared facade. #73 is the one response body in the migration that changes: it gains an `engine` key carrying the engine-capabilities view, with `providerFor` saying whether the declaration came from the active transport's provider or from the default one standing in. Every pre-existing key keeps its name, position and value, and the ACP wire table is not replaced by the 14 matrix keys. --- packages/webui/docs/API.md | 49 +- packages/webui/docs/API.zh-CN.md | 46 +- packages/webui/docs/ARCHITECTURE.md | 122 +- packages/webui/docs/ARCHITECTURE.zh-CN.md | 104 +- .../webui/scripts/check-docs-alignment.mjs | 2 + packages/webui/server/engine/account-reads.js | 228 ++++ .../webui/server/engine/capability-reads.js | 216 +++ packages/webui/server/engine/index.js | 61 +- packages/webui/server/engine/model-reads.js | 727 ++++++++++ packages/webui/server/routes/account.js | 29 +- packages/webui/server/routes/model.js | 502 +------ packages/webui/server/routes/protocol.js | 49 +- .../test/lib/engine/account-reads.test.js | 449 +++++++ .../test/lib/engine/capability-reads.test.js | 477 +++++++ .../webui/test/lib/engine/model-reads.test.js | 1181 +++++++++++++++++ release/public-source.json | 6 + scripts/test-tmp-leak.check.mjs | 1 + 17 files changed, 3767 insertions(+), 482 deletions(-) create mode 100644 packages/webui/server/engine/account-reads.js create mode 100644 packages/webui/server/engine/capability-reads.js create mode 100644 packages/webui/server/engine/model-reads.js create mode 100644 packages/webui/test/lib/engine/account-reads.test.js create mode 100644 packages/webui/test/lib/engine/capability-reads.test.js create mode 100644 packages/webui/test/lib/engine/model-reads.test.js diff --git a/packages/webui/docs/API.md b/packages/webui/docs/API.md index 4ddb7595..7effdb52 100644 --- a/packages/webui/docs/API.md +++ b/packages/webui/docs/API.md @@ -2416,10 +2416,23 @@ insensitive, trailing slash-insensitive, `\` and `/` interchangeable). ### `GET /api/protocol/capabilities` -Returns the engine's `agentInfo` (from the `initialize` reply) plus the +Returns the engine's `agentInfo` (from the `initialize` reply), the capability table webui knows about (`MCODE_ACP_CAPABILITIES` in -`server/lib/mcode-rpc.js`). Used by the webui to decide which UI -controls to enable. +`server/lib/mcode-rpc.js`), and — since M3 batch B4 — the +**engine-capabilities view**: the declared 14-key capability surface of +the active engine provider plus its degradation summary, the same +declaration `GET /api/engine-capabilities` serves. Used by the webui to +decide which UI controls to enable. + +Two tables, two questions, both kept: + +- `capabilities` answers **"which ACP JSON-RPC method does this control + map onto"** — a flat `{method: boolean}` map. +- `engine.capabilities` answers **"does the engine have this capability at + all"** — the 14 matrix keys, each `{level, missing?, reason?}`. + +They can legitimately disagree (the ACP surface and the capability matrix +are not the same taxonomy), so neither replaces the other. **Response 200** ```json @@ -2442,6 +2455,22 @@ controls to enable. "new": true, "prompt": true }, + "engine": { + "provider": "local-runtime-v2", + "providerFor": "transport", + "transport": "runtime", + "capabilities": { + "sessionCrud": { "level": "full" }, + "updateCheck": { + "level": "none", + "reason": "interface-absent: no update-check method anywhere in local-runtime-v2 (design §1.3 v2)" + } + }, + "unavailable": { + "none": ["updateCheck"], + "partial": [{ "key": "gitOperations", "missing": ["git-diff", "git-commit", "git-branch"] }] + } + }, "notes": { "set_mode": "Takes a modeId from the session's availableModes.", "set_config_option": "With configId 'permissionMode' this changes the mode mid-session.", @@ -2455,6 +2484,20 @@ controls to enable. `mcodeVersion` is `"unknown"` before a client has attached (no `initialize` reply yet); the endpoint does not invent a version. +`engine.providerFor` says where the declaration came from, and a consumer +should branch on it: + +- `"transport"` — the active `MCODE_WEBUI_TRANSPORT`'s own registered + provider answered. +- `"default"` — no provider claims that transport yet (arrives with M4), so + the default provider's declaration is standing in. The view is still a + real, reviewed declaration, but it is not necessarily the connected + engine's, and reporting it as such would be a lie. + +`engine.unavailable` is the degradation summary the capability-driven UI +renders from: a `none` key means hide the entry point, a `partial` key means +hide or disable exactly the listed sub-actions. + --- ## Authorize decisions diff --git a/packages/webui/docs/API.zh-CN.md b/packages/webui/docs/API.zh-CN.md index 98d51f25..cdbd0a3a 100644 --- a/packages/webui/docs/API.zh-CN.md +++ b/packages/webui/docs/API.zh-CN.md @@ -2230,9 +2230,22 @@ code, killEndpoint: "/api/stop" }`。温和版→SIGKILL 的级联 ### `GET /api/protocol/capabilities` -返回引擎的 `agentInfo`(取自 `initialize` 应答)以及 webui -已知的 capability 表(`server/lib/mcode-rpc.js` 里的 -`MCODE_ACP_CAPABILITIES`)。webui 用它来决定启用哪些 UI 控件。 +返回引擎的 `agentInfo`(取自 `initialize` 应答)、webui 已知的 +capability 表(`server/lib/mcode-rpc.js` 里的 +`MCODE_ACP_CAPABILITIES`),以及——自 M3 批次 B4 起——**engine-capabilities +视图**:当前引擎 provider 声明的 14 键能力面加它的降级摘要,也就是 +`GET /api/engine-capabilities` 所服务的同一份声明。webui 用它来决定启用 +哪些 UI 控件。 + +两张表、两个问题,都保留: + +- `capabilities` 回答的是**「这个控件对应哪个 ACP JSON-RPC 方法」**—— + 一张扁平的 `{方法: 布尔}` 表。 +- `engine.capabilities` 回答的是**「引擎到底有没有这项能力」**—— + 14 个矩阵键,每项形如 `{level, missing?, reason?}`。 + +两者可以合法地不一致(ACP 面与能力矩阵不是同一套分类法),所以谁也 +不替换谁。 **响应 200** ```json @@ -2255,6 +2268,22 @@ code, killEndpoint: "/api/stop" }`。温和版→SIGKILL 的级联 "new": true, "prompt": true }, + "engine": { + "provider": "local-runtime-v2", + "providerFor": "transport", + "transport": "runtime", + "capabilities": { + "sessionCrud": { "level": "full" }, + "updateCheck": { + "level": "none", + "reason": "interface-absent: no update-check method anywhere in local-runtime-v2 (design §1.3 v2)" + } + }, + "unavailable": { + "none": ["updateCheck"], + "partial": [{ "key": "gitOperations", "missing": ["git-diff", "git-commit", "git-branch"] }] + } + }, "notes": { "set_mode": "Takes a modeId from the session's availableModes.", "set_config_option": "With configId 'permissionMode' this changes the mode mid-session.", @@ -2268,6 +2297,17 @@ code, killEndpoint: "/api/stop" }`。温和版→SIGKILL 的级联 `mcodeVersion` 在尚无客户端挂接(还没收到 `initialize` 应答) 时为 `"unknown"`;本端点不会臆造一个版本号。 +`engine.providerFor` 说明这份声明来自哪里,消费方应当据此分支: + +- `"transport"`——当前 `MCODE_WEBUI_TRANSPORT` 自己的已注册 provider + 应答的。 +- `"default"`——尚无任何 provider 声明该传输(M4 引入),由默认 + provider 的声明顶替。这份视图仍是一份真实且经评审的声明,但它未必 + 是已连接引擎的那份;把它当成后者报出去就是撒谎。 + +`engine.unavailable` 是能力驱动型 UI 据以渲染的降级摘要:`none` 的键 +意味着隐藏整个入口,`partial` 的键意味着恰好隐藏或禁用列出的那些子动作。 + --- ## 授权决策 diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index b75babf0..97ddecfa 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -489,7 +489,7 @@ not import it but adopts the same shape. Unknown future statuses render as ### `engine/` (capability declarations + the local-runtime-v2 host) The engine abstraction lives at `server/engine/` (engine-abstraction -batch B1; migration state M1, plus M3 batches B0, B1, B2 and B3). Eleven +batch B1; migration state M1, plus M3 batches B0, B1, B2, B3 and B4). Fourteen files, one job each: | File | Owns | @@ -505,6 +505,9 @@ files, one job each: | `engine/session-tree-reads.js` | The session-tree family's facade call (`readEngineSessionTree`) and the endpoint→capability table `SESSION_TREE_ENDPOINTS` (step M3, batch B2). Gates **hard**: `assertSessionTreeCapability` throws → 501, because the tree is entirely engine data. Forwards to `lib/session-tree.js#getSessionTree`; the assembler is not duplicated | | `engine/session-export.js` | The export family's facade call (`readEngineSessionTranscript`) and the endpoint→capability table `SESSION_EXPORT_ENDPOINTS` (step M3, batch B2). Gates **soft**: `checkSessionExportCapability` reports and never throws, because export's primary source is `sessions.json`, not the engine | | `engine/usage-reads.js` | The usage family's facade calls (`readEngineAccountQuota`, `readEngineSessionUsage`, `readEngineQuotaForecast`), the derived figure `contextUsedTokens`, and the endpoint→capability table `USAGE_READ_ENDPOINTS` (step M3, batch B3). Gates **hard** on the two engine reads and declares **no capability at all** for #19, which touches no engine surface | +| `engine/account-reads.js` | The account family's facade call (`readEngineAccount`) and the endpoint→capability table `ACCOUNT_READ_ENDPOINTS` (step M3, batch B4). Gates **hard** on `authCredentials` · `getAccountStatus` — the same pair and the same provider method as `engine/usage-reads.js`, because #20 and #15/#16 read the same engine projection. Its read is **synchronous**; see the boot-path note below | +| `engine/model-reads.js` | The model-catalogue family's facade call (`readEngineModelCatalogue`), the whole projection as named pure functions (`projectModelCatalogue`, `deriveModelSelection`, `buildModelCataloguePayload`, `catalogueSourceLabel`, `webuiFullModelId`, `providerOfModelId`, `attachContextWindowOptions`, `configOption`), and the endpoint→capability table `MODEL_READ_ENDPOINTS` (step M3, batch B4). Gates **soft**: `checkModelReadCapability` reports and never throws, because the catalogue's primary sources are files webui owns. Its read is **synchronous**, and it is the one engine module **not** re-exported from `engine/index.js` — see the boot-path note below | +| `engine/capability-reads.js` | The capability-declaration family's facade call (`readEngineCapabilityView`) and the endpoint→capability table `CAPABILITY_READ_ENDPOINTS` (step M3, batch B4). Declares **no capability for #73** — it IS the declaration endpoint, and gating the gate would let a `none` hide the declaration that says so. It is the only endpoint in the migration whose response body gains a key (`engine`, the engine-capabilities view) | Routes take the host from the facade and never from `lib/acp-client.js`: `routes/plugins.js` and `routes/turn-diff.js` call @@ -583,13 +586,37 @@ everything it imports statically must stay free of `@mavis/*`, declaration and construction were split). `test/lib/engine/host-facade.test.js` enforces it against the real module graph rather than against source text. `engine/session-reads.js`, `engine/session-tree-reads.js`, -`engine/session-export.js` and `engine/usage-reads.js` all live under the +`engine/session-export.js`, `engine/usage-reads.js`, +`engine/account-reads.js` and `engine/capability-reads.js` all live under the same rule: their static imports are `engine/capabilities.js` and `engine/index.js` only, and every heavier dependency — `lib/acp-client.js`, `lib/config.js`, `lib/session-tree.js`, -`lib/transcript.js`, `lib/usage.js`, `lib/mavis-usage.js` and -`lib/quota-forecast.js` — is reached through `await import()` inside the -functions. +`lib/transcript.js`, `lib/usage.js`, `lib/mavis-usage.js`, +`lib/quota-forecast.js`, `lib/mcode-rpc.js` — is reached through +`await import()` inside the functions. + +`engine/model-reads.js` is the one deliberate exception, and it deviates on +**both** sides of the import. Its four sources — `lib/config.js`, +`lib/engine-catalogue.js`, `lib/models.js`, `lib/providers-config.js` — are +static imports, because `routes/model.js` already imported all four +**before** M3-B4 and the server's boot cost is therefore exactly what it +was. They reach `@mavis/shared/local-runtime-paths` (via `lib/config.js`) +and `js-yaml` (via `engine-provider-sync.js`), so the module is deliberately +**not** re-exported from `engine/index.js`: making the shared facade — the +one import site the whole server shares, and the one `routes/plugins.js` +must stay light through — heavier than it has ever been would buy nothing. +`routes/model.js` therefore imports `../engine/model-reads.js` directly, +the same shape `routes/protocol.js` already uses for `engine/session-reads.js`. +`test/lib/engine/host-facade.test.js` is the gate that forced this, and it +is right to. + +The price is a **synchronous** read. Making the four imports dynamic would +let the module re-export from the facade again, at the cost of turning +`handleGetModels` into an async handler — a contract change for any caller +that does not await, and the one thing this batch promises not to do. When +the catalogue read becomes async (M4, with a provider-backed source) the +module can move back behind `await import()` and be re-exported with the +rest. #### Which endpoints read through the facade (step M3, batch B1) @@ -749,6 +776,91 @@ reports `_meta.mcode_unavailable: true` with `_meta.source: "webui"`. That is pre-existing and deliberately preserved — re-enabling it is a behaviour change for a later slice, not a refactor. +#### Which endpoints read through the facade (step M3, batch B4) + +Batch B4 adds three endpoints, and they are the first three whose gate +policies are **all different from each other**: one hard, one soft, one +declared-as-nothing. Three modules, for the reason B2 gave — a shared table +would force one family to inherit another's policy. + +| Endpoint | Capability · sub-item | Enforcement | Value source | +| --- | --- | --- | --- | +| `GET /api/account` | `authCredentials` · `getAccountStatus` | hard — 501 | `lib/mcode-rpc.js#getAccountStatus`, the engine's `mcode/account/status` projection. The response body is built by the facade: `{ok:true, ...data}` on success, `{ok:false, reason}` at HTTP 200 otherwise | +| `GET /api/models` | `authCredentials` · `listModelProviders` | soft — reported | three layered sources: the engine session's `model` config option, the merged providers config (webui `env > cwd > user` over the engine's `custom_provider` tree, via `lib/engine-catalogue.js`), and the builtin cli-bundle extraction | +| `GET /api/protocol/capabilities` | none of the 14 keys | none — the gate is a reported no-op | the registered provider's 14-key declaration plus `summarizeUnavailableCapabilities`, and the ACP `initialize` `agentInfo` mirror | + +**Why #20 gates hard and #57 does not.** The account card is 100% engine +data: there is no webui-side fallback for "who am I" or for a plan tier, so +a provider that cannot report an account has nothing to return and 501 is the +honest answer. The model catalogue is not: its primary sources are files +webui owns and can read without the engine — `models.json`, +`~/.mcode-webui/providers.json`, and a cli-bundle extraction — plus the +engine's own `config.yaml`. Gating #57 hard would delete a working picker in +response to a declaration about a capability it does not depend on, which is +the same reasoning `engine/session-export.js` records for #11. So +`checkModelReadCapability` reports and returns; the read is unaffected by +what it reports. + +**Why #73 declares nothing.** It is the declaration endpoint. A gate on it +would be circular, and a `none` anywhere in the declaration could hide the +declaration that says so — the same reason B1's `/api/health` and B3's +`/api/usage/forecast` declare no capability. `checkCapabilityReadCapability` +is exported anyway, so the symmetry with the other families is visible and +testable. + +Four properties this batch holds, each with a test behind it: + +1. **#57 is a full snapshot, and the oracle is the pre-refactor code.** + `test/lib/engine/model-reads.test.js` projects one rich fixture — engine + session option, engine `custom_provider` layer, webui config layer, + builtin layer, a builtin that **collides** with a config entry, a + switchable variant model, an effort-list model, a `forced_on` model, two + providers with overlapping upstream model ids, one provider with a key + and one without — and compares the whole response body, field for field + and key for key, against a literal captured from `3362c9be`. The oracle + is not recomputed by the functions under test. The load-bearing part is + what is **absent**: the config layer takes the `minimax_api/MiniMax-M3` + slot wholesale, so that entry appears once, with the operator's label and + `contextLimit`, and **without** the builtin's `thinkingLevels` and + `contextWindowOptions`. +2. **Grouping is by provider, and the dedupe is per provider.** The webui id + is always `/`, even when the upstream model id + already contains `/` (ticket 09-02). `nousresearch/z-ai/glm-5.3` and + `zai-max/z-ai/glm-5.3` are two rows in two groups; the previous + behaviour let one swallow the other. The builtin shell is keyed by + `minimax_api` **regardless of the recorded pick**, which is the + "8 config + 6 misplaced builtins = 14 in `nousresearch`" replay. +3. **The two builtin-tree projections reach two sites, and a miss is a miss.** + `readEngineBuiltinThinking` and `readEngineBuiltinContextWindows` are two + views of `provider.minimax.models`, read once per request and consumed at + the engine-session site (keyed by the wire form's **bare** model id) and at + the builtin shell. A wire form whose model segment does not parse, or a + model absent from the tree, produces a field-free entry — never a + half-annotation. The section that perturbs the tree asserts which entries + move for which record. +4. **#73's change is additive and its fallback is labelled.** The response + gains exactly one key, `engine`, placed after `capabilities`; every + pre-existing key keeps its exact name, position and value, and the ACP + wire table is **not** replaced by the 14 matrix keys (they answer a + different question, and `docs/API.md` documents both). Inside the view, + `providerFor` says whether the declaration came from the active + transport's provider or from the default provider standing in for a + transport no provider claims yet — a capability-detection endpoint must + not report a standing-in declaration as though it were the connected + engine's. + +**The three "what is active" figures are derived once.** `current` prefers +the engine's `currentValue` and falls back to the recorded pre-session pick; +`currentThinking` prefers the engine's `thinkingEffort` option; and +`currentContextWindow` is the recorded window with the current model's +catalogue `contextLimit` as the fallback. When neither exists the answer is +`null`, never a default model — the old behaviour invented an active model +the engine never confirmed and the composer chip claimed it. + +**`handleGetModels` is still a synchronous handler.** The facade read is +synchronous too, and the test asserts it: the body must be complete when the +handler returns, because that is what the pre-M3 handler guaranteed. + ## 4. The `clientState` payload This is the shape every SSE `state` event contains. The webui mirrors diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index 3bce7f75..24294066 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -461,7 +461,7 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 ### `engine/`(能力声明 + local-runtime-v2 host) 引擎抽象层位于 `server/engine/`(engine-abstraction 批次 B1;迁移 -状态 M1,外加 M3 的 B0、B1、B2 与 B3 四批)。十一个文件,各管一件事: +状态 M1,外加 M3 的 B0、B1、B2、B3 与 B4 五批)。十四个文件,各管一件事: | 文件 | 职责 | | --- | --- | @@ -476,6 +476,9 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 | `engine/session-tree-reads.js` | 会话树族的面板调用 `readEngineSessionTree` 与端点→能力对照表 `SESSION_TREE_ENDPOINTS`(迁移步 M3 批次 B2)。**硬门控**:`assertSessionTreeCapability` 抛出 → 501,因为树完全由引擎数据构成。转发到 `lib/session-tree.js#getSessionTree`,树的装配逻辑不复制第二份 | | `engine/session-export.js` | 导出族的面板调用 `readEngineSessionTranscript` 与端点→能力对照表 `SESSION_EXPORT_ENDPOINTS`(迁移步 M3 批次 B2)。**软门控**:`checkSessionExportCapability` 只报告、从不抛出,因为导出的主数据源是 `sessions.json` 而非引擎 | | `engine/usage-reads.js` | 用量族的面板调用(`readEngineAccountQuota`、`readEngineSessionUsage`、`readEngineQuotaForecast`)、派生量 `contextUsedTokens`,与端点→能力对照表 `USAGE_READ_ENDPOINTS`(迁移步 M3 批次 B3)。两个引擎读**硬门控**;#19 **完全不声明能力**,因为它不触达任何引擎面 | +| `engine/account-reads.js` | 账户族的面板调用 `readEngineAccount` 与端点→能力对照表 `ACCOUNT_READ_ENDPOINTS`(迁移步 M3 批次 B4)。**硬门控**,门控在 `authCredentials` · `getAccountStatus`——与 `engine/usage-reads.js` 同一对、同一个 provider 方法,因为 #20 与 #15/#16 读的是同一份引擎投影。它的读是**同步的**,见下面的启动路径说明 | +| `engine/model-reads.js` | 模型目录族的面板调用 `readEngineModelCatalogue`、整套投影的具名纯函数(`projectModelCatalogue`、`deriveModelSelection`、`buildModelCataloguePayload`、`catalogueSourceLabel`、`webuiFullModelId`、`providerOfModelId`、`attachContextWindowOptions`、`configOption`),与端点→能力对照表 `MODEL_READ_ENDPOINTS`(迁移步 M3 批次 B4)。**软门控**:`checkModelReadCapability` 只报告、从不抛出,因为目录的主数据源是 webui 自己拥有的文件。它的读是**同步的**,并且它是唯一一个**没有**从 `engine/index.js` 转发导出的引擎模块——见下面的启动路径说明 | +| `engine/capability-reads.js` | 能力声明族的面板调用 `readEngineCapabilityView` 与端点→能力对照表 `CAPABILITY_READ_ENDPOINTS`(迁移步 M3 批次 B4)。#73 **不声明任何能力**——它本身就是声明端点,给门控上门控会让某个 `none` 把声明它的那份声明藏起来。它是本次迁移中唯一一个响应体新增了一个键的端点(`engine`,即 engine-capabilities 视图) | 路由从门面取 host,不从 `lib/acp-client.js` 取:`routes/plugins.js` 与 `routes/turn-diff.js` 调 `getEngineCatalogueHost()`。两者都保留 `deps` @@ -540,12 +543,33 @@ handler 层测试因此保持封闭。 门面自身加载 4685ms → 5ms)。`test/lib/engine/host-facade.test.js` 对着真实模块图强制它,而不是对着源码文本。 `engine/session-reads.js`、`engine/session-tree-reads.js`、 -`engine/session-export.js` 与 `engine/usage-reads.js` 全部服从同一条 +`engine/session-export.js`、`engine/usage-reads.js`、 +`engine/account-reads.js` 与 `engine/capability-reads.js` 全部服从同一条 纪律:静态 import 只有 `engine/capabilities.js` 与 `engine/index.js`, 而每个更重的依赖——`lib/acp-client.js`、`lib/config.js`、 `lib/session-tree.js`、`lib/transcript.js`、`lib/usage.js`、 -`lib/mavis-usage.js` 与 `lib/quota-forecast.js`——都在函数体内用 -`await import()` 触达。 +`lib/mavis-usage.js`、`lib/quota-forecast.js`、`lib/mcode-rpc.js`——都在 +函数体内用 `await import()` 触达。 + +`engine/model-reads.js` 是唯一一处刻意例外,而且它在 import 的**两侧** +都刻意偏离。它的四个数据源——`lib/config.js`、 +`lib/engine-catalogue.js`、`lib/models.js`、`lib/providers-config.js`—— +是静态 import,因为 M3-B4 之前 `routes/model.js` 就静态 import 了这四个, +所以 server 的启动成本分文未增。但它们会经 `lib/config.js` 抵达 +`@mavis/shared/local-runtime-paths`、经 `engine-provider-sync.js` 抵达 +`js-yaml`,所以这个模块**刻意没有**从 `engine/index.js` 转发导出:让 +共享门面——整个 server 唯一的共享 import 站点,也是 +`routes/plugins.js` 必须保持轻量的那个——比它历来更重,换不来任何东西。 +因此 `routes/model.js` 直接 import `../engine/model-reads.js`,这与 +`routes/protocol.js` 对 `engine/session-reads.js` 的写法同形。 +`test/lib/engine/host-facade.test.js` 正是逼出这个决定的那道门禁,而它 +是对的。 + +代价是一次**同步**读。把那四个 import 改成动态的,就能让这个模块重新 +被门面前转发,代价是把 `handleGetModels` 变成异步处理器——这对任何不 +await 的调用方都是契约变更,也正是本批承诺不做的那件事。等目录读变成 +异步时(M4,接上 provider 支撑的数据源),这个模块就可以退回 +`await import()` 之后,与其余各族一起被转发导出。 #### 哪些端点走门面读(迁移步 M3 批次 B1) @@ -678,6 +702,78 @@ provider 确实没有树可返回,501 才是诚实答案。 `_meta.source: "webui"`。这是既有行为且被刻意保留——重新启用它是一次行为 变更,属于后续切片,不属于这次收编。 +#### 哪些端点走门面读(迁移步 M3 批次 B4) + +批次 B4 加入 3 个端点,它们是首批**门控策略彼此全都不同**的三个: +一个硬门控、一个软门控、一个声明为「什么都不声明」。因此是三个模块, +理由与 B2 相同——共用一张表会逼其中一族继承另一族的策略。 + +| 端点 | 能力 · 子项 | 强制方式 | 取值来源 | +| --- | --- | --- | --- | +| `GET /api/account` | `authCredentials` · `getAccountStatus` | 硬——501 | `lib/mcode-rpc.js#getAccountStatus`,即引擎的 `mcode/account/status` 投影。响应体由门面组装:成功是 `{ok:true, ...data}`,失败在 HTTP 200 上是 `{ok:false, reason}` | +| `GET /api/models` | `authCredentials` · `listModelProviders` | 软——只报告 | 三个分层来源:引擎会话的 `model` 配置项、合并后的 provider 配置(webui 的 `env > cwd > user` 叠在引擎 `custom_provider` 树之上,经 `lib/engine-catalogue.js`)、以及内建 cli 包抽取 | +| `GET /api/protocol/capabilities` | 14 个键里的任何一个都不适用 | 不门控——门控是「被报告的空操作」 | 已注册 provider 的 14 键声明加 `summarizeUnavailableCapabilities`,以及 ACP `initialize` 的 `agentInfo` 镜像 | + +**为什么 #20 硬门控而 #57 不硬。** 账户卡 100% 由引擎数据构成: +「我是谁」和「什么套餐」都没有 webui 侧的兜底,所以报不出账户的 +provider 确实无物可报,501 才是诚实答案。模型目录不是:它的主数据源是 +webui 自己拥有、不依赖引擎就能读的文件——`models.json`、 +`~/.mcode-webui/providers.json`、cli 包抽取——再加上引擎自己的 +`config.yaml`。对 #57 硬门控,等于用一份它并不依赖的能力声明去删掉一个 +能用的选择器,这与 `engine/session-export.js` 为 #11 记下的理由同源。所以 +`checkModelReadCapability` 只报告然后返回;这次读不受它报告结果的影响。 + +**为什么 #73 什么都不声明。** 它就是声明端点。给它上门控是循环论证,而且 +声明里任何一处 `none` 都能把声明它的那份声明藏起来——这与 B1 的 +`/api/health`、B3 的 `/api/usage/forecast` 不声明能力同源。即便如此 +`checkCapabilityReadCapability` 仍然导出,好让与其他各族的对称关系可见、 +可测。 + +本批持有的四条性质,每条背后都有一个测试: + +1. **#57 是全量快照,且预言机取自收编前的代码。** + `test/lib/engine/model-reads.test.js` 用一套内容丰富的 fixture 做投影 + ——引擎会话配置项、引擎 `custom_provider` 层、webui 配置层、内建层、 + 一个与配置项**撞 id** 的内建模型、一个可切换 variant 模型、一个 + effort 列表模型、一个 `forced_on` 模型、两个上游模型 id 重叠的 + provider、一个有 key 与一个没 key 的 provider——并把整个响应体逐字段、 + 逐键地与一份从 `3362c9be` 抓下来的字面量比对。预言机不是被测函数自己 + 算出来的。承重的是**缺席**的那部分:配置层整体接管了 + `minimax_api/MiniMax-M3` 这个位置,所以该条目只出现一次,带着运维的 + label 与 `contextLimit`,而**没有**内建模型的 `thinkingLevels` 与 + `contextWindowOptions`。 +2. **分组按 provider,去重也按 provider。** webui id 恒为 + `/`,即使上游模型 id 本身已含 `/` + (ticket 09-02)。`nousresearch/z-ai/glm-5.3` 与 + `zai-max/z-ai/glm-5.3` 是两组里的两行;旧行为会让其中一个吞掉另一个。 + 内建外壳**无论当前记录选了什么**都归到 `minimax_api`——这正是 + 「8 个配置 + 6 个错位的内建 = `nousresearch` 里 14 个」那次回放的 + 结论。 +3. **两棵内建树投影会抵达两个站点,而「查不到」就是查不到。** + `readEngineBuiltinThinking` 与 `readEngineBuiltinContextWindows` 是 + `provider.minimax.models` 的两个视图,每次请求读一次,分别在引擎会话 + 站点(按 wire 形式的**裸**模型 id 查)与内建外壳处被消费。wire 形式的 + 模型段解析不出来、或模型不在树里,产出的就是一个无这些字段的条目, + 绝不会是「半吊子标注」。扰动那棵树的那一节断言了:哪条引擎记录会让 + 哪些条目发生变化。 +4. **#73 的变更是增量的,且它的兜底是带标签的。** 响应恰好新增一个键 + `engine`,位置紧跟 `capabilities` 之后;每个既有键的名字、位置与取值 + 都不变,ACP wire 表**没有**被 14 个矩阵键替换(两者回答的是不同问题, + `docs/API.md` 两者都记录了)。视图内部的 `providerFor` 说明这份声明 + 来自当前传输的 provider,还是来自「当前传输还没有任何 provider 声明」 + 时顶替的默认 provider——一个能力探测端点绝不能把顶替声明当作已连接 + 引擎的声明报出去。 + +**三个「当前生效」的量只派生一次。** `current` 优先取引擎的 +`currentValue`,回落到记录在案的会话前选择;`currentThinking` 优先取引擎的 +`thinkingEffort` 配置项;`currentContextWindow` 是记录在案的窗口,回落 +到当前模型在目录里的 `contextLimit`。两者都没有时答案是 `null`,而不是 +某个默认模型——旧行为会凭空造出一个引擎从未确认的活跃模型,而 composer +的芯片会把它当成正在跑的模型宣称出去。 + +**`handleGetModels` 仍是同步处理器。** 门面的读同样是同步的,测试对此有 +断言:处理器返回时响应体必须已经写完,因为这是 M3 之前处理器给出的保证。 + ## 4. `clientState` 载荷 这是每个 SSE `state` 事件所包含的形状。webui 将其 diff --git a/packages/webui/scripts/check-docs-alignment.mjs b/packages/webui/scripts/check-docs-alignment.mjs index ce22bf06..e53a2d88 100644 --- a/packages/webui/scripts/check-docs-alignment.mjs +++ b/packages/webui/scripts/check-docs-alignment.mjs @@ -509,6 +509,8 @@ const NOT_ON_DISK = new Set([ "server/routes/foo.js", // the illustrative path in §9's recipe "sessions.json", // runtime data under WEBUI_DATA_DIR, not a source file "mcp.json", // user-authored MCP server config, not a source file + "models.json", // operator-authored provider catalogue (cwd layer), not a source file + "config.yaml", // the ENGINE's own config under its data dir, not a source file "index.html", // Next export output (webapp/out/index.html), not a source file ]); diff --git a/packages/webui/server/engine/account-reads.js b/packages/webui/server/engine/account-reads.js new file mode 100644 index 00000000..659bf562 --- /dev/null +++ b/packages/webui/server/engine/account-reads.js @@ -0,0 +1,228 @@ +// webui/server/engine/account-reads.js +// +// Migration step M3, batch B4: the account read (账户读) — +// +// #20 GET /api/account — the account card's identity / plan-tier data +// +// What this file is for. #20 is a small endpoint with a strict privacy +// contract: the payload is the user's own display name, plan tier and +// quota, it is fetched ON DEMAND rather than pushed into the state +// snapshot (the snapshot is broadcast to every SSE subscriber, LAN +// included), and the engine's projection carries no credential. The +// route therefore has to stay a one-liner that writes a body and never +// accumulates identity state — and that is exactly what a facade read +// gives it. After M3-B4 the route asks this file, this file asks the +// provider whether it may, and only then forwards to +// `lib/mcode-rpc.js#getAccountStatus`. +// +// Why the gate is HARD here while #57 and #73 are not. This endpoint is +// 100% engine data: there is no webui-side fallback for "who am I" and +// no webui-side fallback for the plan tier. The empty state the card +// renders when the engine cannot be reached is a RUNTIME outcome +// (HTTP 200 + `{ok:false, reason}`), which this file preserves +// verbatim; a provider that declares no `getAccountStatus` is a +// different, structural outcome, and the only honest answer to it is +// the 501 that `app.js#invokeHandler` derives from +// `EngineCapabilityNotSupportedError`. Same rule, same pair, same +// provider method as B3's `POST /api/usage` / `POST /api/usage-trigger` +// — the usage popover and the account card read the SAME engine +// projection through the SAME `mcode/account/status` extension method, +// so a declaration that removes it must take both down together. Two +// modules, not one: the usage family owns the quota DERIVATIONS +// (`contextUsedTokens`, the least-squares forecast) and the usage +// family has its own gate policy for #19; merging them would force one +// to inherit the other's. +// +// What this file deliberately does NOT do: +// +// - It does not reshape the engine's projection. `r.data` is spread +// into the response verbatim (`{ok:true, ...r.data}`), so a new +// engine field reaches the card without a webui edit, and an +// absent one does not become a `null` this layer invented. +// - It does not invent a reason. The failure body is +// `{ok:false, reason: r.code || "account_unavailable"}` — the +// endpoint's own fallback, kept byte-for-byte. The account card +// (`webapp/components/shell.tsx#SidebarFooter`) renders its +// 本地用户 placeholder on any failure, and it must keep doing so +// for the engine-could-not-be-reached case that has always produced +// it. +// - It does not construct a host. `getAccountStatus` goes through the +// process-singleton ACP client, the same path it has always taken. +// - It does not log the payload. Identity data must not reach a log +// line; the only logging this path can do is whatever +// `mcode-rpc.js#sanitizeError` already does to an error string. +// +// Boot-path weight. `app.js` imports `routes/account.js`, the route +// imports this file, so this file is on the boot path. It statically +// imports nothing heavier than `capabilities.js` and `index.js` (both +// pure declaration modules); `lib/mcode-rpc.js` and `lib/config.js` are +// reached through `await import()` inside the read. That split is the +// M1 lesson — putting the `@mavis/*` tree on the boot path once cost +// 209ms → 2700ms of server start and broke the integration tests' 3s +// window. +// +// Provider selection is M4's job, same as B1, B2 and B3: +// `providerByTransport()` maps a transport to a REGISTERED provider id; +// today only `runtime` has one, so under the default `acp` transport the +// gate reports `gate: "unregistered-transport"` and the read proceeds — +// which is correct, because the pre-M4 behaviour under `acp` is the +// only behaviour this endpoint has ever had. + +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, exactly as in + * `session-reads.js#providerByTransport`, + * `session-tree-reads.js#providerByTransport` and + * `usage-reads.js#providerByTransport`, which this mirrors rather than + * merges: the four families have separate read contracts and a shared + * table would force one of them to inherit another's policy. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. Every + * consumer of the table is a function anyway. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +/** + * The declaration this endpoint needs, and the sub-item it needs from + * that capability. + * + * `authCredentials` / `getAccountStatus` is the honest mapping, and it + * is deliberately the SAME pair `usage-reads.js` uses for #15 / #16: + * both endpoints read the engine's own account projection through the + * `mcode/account/status` extension method, so they depend on the same + * provider method and must be gated by the same declaration. Naming a + * different sub-item here would let a `partial` provider drop + * `getAccountStatus` from the account card while the usage popover + * still claimed to have it. + * + * @type {Readonly>} + */ +export const ACCOUNT_READ_ENDPOINTS = Object.freeze({ + "GET /api/account": { capability: "authCredentials", subItem: "getAccountStatus" }, +}); + +/** + * Resolve the provider that answers the account read on `transport`, or + * `null` when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveAccountReadProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Check the account read against the active provider's declaration. + * Throws `EngineCapabilityNotSupportedError` — which + * `app.js#invokeHandler` turns into 501 — when the declaration says + * the capability (or the exact sub-item) is absent. + * + * @param {string} endpoint A key of ACCOUNT_READ_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null}} + */ +export function assertAccountReadCapability(endpoint, transport) { + const need = ACCOUNT_READ_ENDPOINTS[endpoint]; + if (need === undefined) { + // Caller confusion, not an engine limitation — a plain Error so the + // HTTP layer never answers 501 for a typo in webui's own code. + const err = new Error( + `assertAccountReadCapability: "${endpoint}" is not part of the account family ` + + `(known: ${Object.keys(ACCOUNT_READ_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_account_read_endpoint"; + throw err; + } + const provider = resolveAccountReadProvider(transport); + if (!provider) { + return { + endpoint, + gate: "unregistered-transport", + provider: null, + capability: need.capability, + subItem: need.subItem, + }; + } + assertEngineCapability(provider.capabilities, need.capability, provider.id, need.subItem); + return { + endpoint, + gate: "checked", + provider: provider.id, + capability: need.capability, + subItem: need.subItem, + }; +} + +/** + * Where the account bytes came from. Always `account-status`: the read + * is the engine's `mcode/account/status` extension method, reached + * through the ACP client, under every transport. The value exists so a + * consumer never has to guess whether a webui-side fallback answered — + * there is none, and saying so in a field is cheaper than a reader + * assuming one. + * + * @typedef {"account-status"} AccountReadSource + */ + +/** + * The #20 (`GET /api/account`) read. + * + * `payload` IS the endpoint's response body, built here once so the + * route is a single `res.end(JSON.stringify(payload))` and the body has + * exactly one home: + * + * - success → `{ok:true, ...(r.data || {})}`. The engine's projection + * is spread verbatim, so `identity` / `tokenPlan` / any future field + * arrive exactly as the engine framed them, and an engine that + * answers `{ok:true, data:null}` still produces `{ok:true}` rather + * than a `TypeError` on the spread. + * - failure → `{ok:false, reason: r.code || "account_unavailable"}`. + * Soft by contract: the REQUEST succeeded, so the status stays 200 + * and the card renders its empty state. `r.code` is preferred + * because it is the engine's own machine-readable reason + * (`no_client`, `unauthorized`, …); the string fallback is the + * endpoint's own and predates every code. + * + * @param {object} [options] + * @param {object} [options.cs] The webui client state; only + * `cs.mcodeSessionId` is read, exactly as the route read it. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/account`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. Exists so tests can exercise both + * the `runtime` and the unregistered `acp` branch without mutating + * process env. + * @returns {Promise<{payload: object, source: AccountReadSource, gate: object, transport: string}>} + */ +export async function readEngineAccount(options = {}) { + const endpoint = options.endpoint || "GET /api/account"; + const [rpc, config] = await Promise.all([ + import("../lib/mcode-rpc.js"), + import("../lib/config.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertAccountReadCapability(endpoint, transport); + // `cs && cs.mcodeSessionId` is forwarded EXACTLY as the route used to + // compute it, including the `undefined` a missing ctx produces — + // `getAccountStatus` turns any falsy id into `{}`, and a test that + // pins the forwarded argument must see the same value the route sent. + const r = await rpc.getAccountStatus(options.cs && options.cs.mcodeSessionId); + const payload = r && r.ok + ? { ok: true, ...(r.data || {}) } + : { ok: false, reason: (r && r.code) || "account_unavailable" }; + return { payload, source: "account-status", gate, transport }; +} diff --git a/packages/webui/server/engine/capability-reads.js b/packages/webui/server/engine/capability-reads.js new file mode 100644 index 00000000..9e7561bf --- /dev/null +++ b/packages/webui/server/engine/capability-reads.js @@ -0,0 +1,216 @@ +// webui/server/engine/capability-reads.js +// +// Migration step M3, batch B4: the capability-declaration read +// (能力声明读) — +// +// #73 GET /api/protocol/capabilities — "what can this engine do?" +// +// What this file is for, and why it is the odd one out in this batch. +// #73 is the endpoint the FRONTEND uses to decide which controls to +// enable, and until M3-B4 it answered from two places webui maintains +// by hand: +// +// - `MCODE_ACP_CAPABILITIES`, a flat `{method: boolean}` table in +// `lib/mcode-rpc.js` describing the ACP JSON-RPC surface; and +// - `getMcodeServerInfo()`, the ACP `initialize` mirror, for the +// engine's name / title / version. +// +// Neither is the engine's DECLARED capability surface. That surface +// already exists — it is the 14-key per-provider declaration in +// `engine/capabilities.js` and the registry in `engine/index.js`, and +// `GET /api/engine-capabilities` already serves it. So webui was +// carrying two parallel answers to "what can the engine do", able to +// disagree, with no test able to notice. After M3-B4 #73 carries the +// engine-capabilities VIEW alongside the ACP wire table: the route no +// longer reaches into `lib/mcode-rpc.js` and `lib/acp-client.js` on +// its own, and the two answers sit in one response where a consumer +// (or a reviewer) can see both and their disagreement. +// +// The wire table is KEPT, not replaced. `capabilities` still answers +// "which ACP method does the frontend's control map onto", which is +// not what the 14 matrix keys answer ("does the engine have this +// capability at all"). Dropping it would break `docs/API.md`'s +// documented response and every consumer that reads +// `capabilities.set_mode`; the engine view is ADDITIVE. That is the +// one place in this batch where the response body gains a key, and it +// is a deliberate, reviewed decision rather than a refactor side +// effect — the existing keys keep their exact values. +// +// Why this endpoint declares NO capability. It is the declaration +// endpoint: gating the gate is circular, and a `none` anywhere in the +// declaration must not be able to hide the declaration that says so. +// The value is `null` for the same reason B1's `/api/health` and B3's +// `/api/usage/forecast` are, and the gate REPORTS the no-op rather +// than passing silently. `checkCapabilityReadCapability` is exported so +// the symmetry with the other families is visible and testable, and so +// a future batch that adds a REAL capability-gated sibling has a +// predicate to build on. +// +// What this file deliberately does NOT do: +// +// - It does not probe. Runtime probing (design §2.3 step 2) is +// deliberately absent for every family in this migration; this +// endpoint is declaration-backed, and a probe result that silently +// overrode the declaration would make the frontend's rendering +// depend on timing. +// - It does not construct a host. +// - It does not convert the engine's `"unknown"` version into an +// error. `/api/health` (#75) already answers that same figure with +// the same fallback through `session-reads.js#readEngineVersion`, +// and two endpoints asking the same protocol question with the same +// answer is correct; two endpoints answering it DIFFERENTLY is +// not, which is why both read `getMcodeServerInfo()` and both keep +// the literal `"unknown"` fallback. +// +// Boot-path weight. `app.js` imports `routes/protocol.js`, the route +// imports this file, so this file is on the boot path. It statically +// imports nothing heavier than `capabilities.js` and `index.js`; +// `lib/mcode-rpc.js` and `lib/acp-client.js` are reached through +// `await import()` inside the read — the M1 lesson, and the reason the +// route's own `await import(...)` lines moved behind this boundary +// rather than being duplicated. + +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * The declaration this endpoint needs: `null`, for the reason in the + * header. The type keeps the `|null` branch so a future gated sibling + * can be added to the same table without changing its shape. + * + * @type {Readonly>} + */ +export const CAPABILITY_READ_ENDPOINTS = Object.freeze({ + "GET /api/protocol/capabilities": null, +}); + +/** + * Resolve the provider whose declaration answers the capability read on + * `transport`. + * + * Unlike every other family this one ALWAYS answers, because an + * empty capability view would be worse than useless for a capability + * DETECTION endpoint: the frontend would learn nothing and could not + * distinguish "no engine" from "this build has no declarations". So + * when no provider claims the transport, the DEFAULT provider's + * declaration is served and the descriptor says so. + * + * @param {string} transport + * @returns {{provider: {id: string, transport: string, capabilities: object}, providerFor: "transport"|"default"}} + */ +export function resolveCapabilityReadProvider(transport) { + const providerId = transport === "runtime" ? DEFAULT_ENGINE_PROVIDER_ID : null; + if (providerId) { + return { provider: getEngineProvider(providerId), providerFor: "transport" }; + } + return { provider: getEngineProvider(), providerFor: "default" }; +} + +/** + * Read the declaration for #73 WITHOUT enforcing it. + * + * Same `gate` vocabulary as `session-export.js#checkSessionExportCapability` + * and `model-reads.js#checkModelReadCapability`, with one difference that + * is forced by the `null` row: the result is always `no-capability-key` + * and never `unregistered-transport`, because the provider this + * endpoint serves is always resolvable (see + * `resolveCapabilityReadProvider`). + * + * @param {string} endpoint A key of CAPABILITY_READ_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null, enforcement: "soft"}} + */ +export function checkCapabilityReadCapability(endpoint, transport) { + const need = CAPABILITY_READ_ENDPOINTS[endpoint]; + if (need === undefined) { + const err = new Error( + `checkCapabilityReadCapability: "${endpoint}" is not part of the capability family ` + + `(known: ${Object.keys(CAPABILITY_READ_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_capability_read_endpoint"; + throw err; + } + const { provider } = resolveCapabilityReadProvider(transport); + return { + endpoint, + gate: "no-capability-key", + provider: provider.id, + capability: null, + subItem: null, + enforcement: "soft", + }; +} + +/** + * The engine-capabilities VIEW — the same four facts + * `GET /api/engine-capabilities` serves, plus HOW the provider was + * chosen. + * + * `providerFor` is the honest bit: `"transport"` means the active + * transport's own provider answered; `"default"` means no provider + * claims that transport yet (M4) and the default provider's + * declaration is standing in. A capability-detection endpoint that + * reported `"default"` as though it were `"transport"` would be + * answering a question about a different engine than the one + * connected — the same lie B1 declined for `/api/health` and B3 + * declined for #19, in the one place where it is most tempting because + * the fallback is silent. + * + * @typedef {{ + * provider: string, + * providerFor: "transport"|"default", + * transport: string, + * capabilities: object, + * unavailable: {none: string[], partial: Array<{key: string, missing: string[]}>}, + * }} EngineCapabilityView + */ + +/** + * The #73 (`GET /api/protocol/capabilities`) read. + * + * `wire` is `MCODE_ACP_CAPABILITIES` forwarded verbatim — the ACP + * method table, NOT the engine declaration, and kept under its own + * name in the response for exactly that reason. `agent` is the ACP + * `initialize` mirror: `{version, name, title}` with the endpoint's own + * `"unknown"` / `null` fallbacks, applied here so the route does not + * repeat them. + * + * @param {object} [options] + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/protocol/capabilities`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @returns {Promise<{engine: EngineCapabilityView, agent: {version: string, name: string|null, title: string|null}, wire: object, source: "declaration", gate: object, transport: string}>} + */ +export async function readEngineCapabilityView(options = {}) { + const endpoint = options.endpoint || "GET /api/protocol/capabilities"; + const [rpc, acp, config, capabilities] = await Promise.all([ + import("../lib/mcode-rpc.js"), + import("../lib/acp-client.js"), + import("../lib/config.js"), + import("./capabilities.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = checkCapabilityReadCapability(endpoint, transport); + const { provider, providerFor } = resolveCapabilityReadProvider(transport); + // `initialize` answers with `agentInfo: {name, title, version}` (not + // `serverInfo`); the mirror is empty until something attaches. + const agentInfo = acp.getMcodeServerInfo(); + return { + engine: { + provider: provider.id, + providerFor, + transport: provider.transport, + capabilities: provider.capabilities, + unavailable: capabilities.summarizeUnavailableCapabilities(provider.capabilities), + }, + agent: { + version: (agentInfo && agentInfo.version) || "unknown", + name: (agentInfo && agentInfo.name) || null, + title: (agentInfo && agentInfo.title) || null, + }, + wire: rpc.MCODE_ACP_CAPABILITIES, + source: "declaration", + gate, + transport, + }; +} diff --git a/packages/webui/server/engine/index.js b/packages/webui/server/engine/index.js index ed0f6680..6d390f04 100644 --- a/packages/webui/server/engine/index.js +++ b/packages/webui/server/engine/index.js @@ -35,9 +35,10 @@ // no route's behaviour changed. M3's first batch (B0) done — the // catalogue host itself is now reached through this facade too // (engine/host.js), so the plugins and turn-diff routes no longer name -// lib/acp-client.js. M3 batches B1 (#9 #10 #72 #74 #75) and B3 (#15 #16 -// #17 #19) done. B2 (#8 #11) and the rest of M3, then M4, will route -// their consumers through this facade one endpoint family at a time. +// lib/acp-client.js. M3 batches B1 (#9 #10 #72 #74 #75), B2 (#8 #11), +// B3 (#15 #16 #17 #19) and B4 (#20 #57 #73) done. The rest of M3, then +// M4, will route their consumers through this facade one endpoint +// family at a time. import { ENGINE_CAPABILITY_KEYS } from "./capabilities.js"; // Declarations only — importing the provider *host-construction* modules @@ -104,6 +105,60 @@ export { readEngineSessionTranscript, resolveSessionExportProvider, } from "./session-export.js"; +// The account read (step M3, batch B4). Same cycle, same rule, same +// reasoning as session-reads.js: account-reads.js reads NOTHING from +// this module at module scope — its `ACCOUNT_READ_ENDPOINTS` table is a +// literal and every binding it needs (`getEngineProvider`, +// `DEFAULT_ENGINE_PROVIDER_ID`) is read inside a function body. A new +// top-level `const X = SOMETHING_FROM_INDEX` in account-reads.js breaks +// the re-export exactly as it would in session-reads.js. It gates HARD +// on `authCredentials.getAccountStatus` — the same pair and the same +// provider method B3's `POST /api/usage` / `POST /api/usage-trigger` +// use, because both read the engine's account projection; the modules +// stay separate because the usage family owns derivations this one +// does not have. +export { + ACCOUNT_READ_ENDPOINTS, + assertAccountReadCapability, + readEngineAccount, + resolveAccountReadProvider, +} from "./account-reads.js"; +// The model-catalogue read (step M3, batch B4) is deliberately NOT +// re-exported here, and that is the one place this file's shape +// disagrees with its siblings. It gates SOFT (the catalogue's primary +// sources are files webui owns, so a provider that declared no model +// surface would not remove the picker — the `session-export.js` +// reasoning, reused rather than re-argued), and its read is +// SYNCHRONOUS, which is what keeps `handleGetModels` synchronous. Both +// properties come from one decision: the three catalogue sources are +// static imports of this module, because `routes/model.js` already +// imported `lib/models.js`, `lib/providers-config.js`, +// `lib/engine-catalogue.js` and `lib/config.js` before M3-B4. +// +// Those four reach `js-yaml` and `@mavis/shared/local-runtime-paths`, +// and `test/lib/engine/host-facade.test.js` is right to refuse that +// under a facade `app.js` loads: it would make `engine/index.js` — the +// one import site the whole server shares, and the one +// `routes/plugins.js` must stay light through — heavier than it has ever +// been, for no saving. So `routes/model.js` imports +// `../engine/model-reads.js` directly, the same shape +// `routes/protocol.js` already uses for `session-reads.js`. The server's +// own boot cost is unchanged: every module involved was already on it +// through the route. When the catalogue read becomes async (M4, with a +// provider-backed source), the module can move back behind +// `await import()` and be re-exported here with the rest. +// The capability-declaration read (step M3, batch B4). Declares NO +// capability for #73 — it IS the declaration endpoint, and gating the +// gate would let a `none` hide the declaration that says so. It is the +// one endpoint in the migration whose response body gains a key +// (`engine`, the engine-capabilities view); see the module header for +// why that is additive rather than a replacement. +export { + CAPABILITY_READ_ENDPOINTS, + checkCapabilityReadCapability, + readEngineCapabilityView, + resolveCapabilityReadProvider, +} from "./capability-reads.js"; // The usage family's gated reads (step M3, batch B3). Same cycle, same // rule, same reasoning as session-reads.js above: usage-reads.js reads // NOTHING from this module at module scope — its `USAGE_READ_ENDPOINTS` diff --git a/packages/webui/server/engine/model-reads.js b/packages/webui/server/engine/model-reads.js new file mode 100644 index 00000000..472d8185 --- /dev/null +++ b/packages/webui/server/engine/model-reads.js @@ -0,0 +1,727 @@ +// webui/server/engine/model-reads.js +// +// Migration step M3, batch B4: the model-catalogue read (模型目录读) — +// +// #57 GET /api/models — the composer's provider-grouped model picker +// +// What this file is for. #57 is the LARGEST projection in webui and +// the one a refactor can damage most quietly. It merges three +// independent sources, dedupes them by a key that has changed shape +// twice, annotates each surviving entry with two separate projections +// of the engine's materialised builtin tree, and derives three +// "what is active right now" figures — none of which is compared +// against anything at runtime. A change that drops one annotation, or +// moves one entry into the wrong provider group, or resolves `current` +// to a different id, changes what the user picks and reports nothing. +// So the whole projection lives HERE, once, as named pure functions +// tested on their INPUTS, and the route assembles nothing but JSON. +// +// The three sources, in the order they are consumed (priority order is +// the endpoint's, not this file's invention — see `projectModelCatalogue`): +// +// 1. the engine SESSION's `model` config option — its `options[].value` +// is the engine's encoded wire form, forwarded verbatim so +// `POST /api/set-model` round-trips; +// 2. the PROVIDERS config — webui's `env > cwd > user` layers with the +// engine's `custom_provider` tree as a new bottom layer +// (`lib/engine-catalogue.js` owns the merge); +// 3. the BUILTIN catalogue — extracted from the engine's own cli +// bundle, so the list tracks the engine without a webui release. +// +// Plus two projections of the SAME engine tree (`provider.minimax.models`) +// that annotate entries in both source 1 and source 3: the variant-style +// thinking schema (`readEngineBuiltinThinking`) and the context-window +// options (`readEngineBuiltinContextWindows`, "U6"). Both are read once +// per request and handed to the two annotation sites — the redundancy of +// reading them per entry was a real cost, and the two reads must agree +// because they are two views of one file. +// +// Why this family's gate is SOFT while the account family gates hard. +// The catalogue is NOT engine data in the way an account is. Its +// primary sources are files webui owns and can read without the engine: +// `models.json` / `~/.mcode-webui/providers.json` on the webui side, and +// a cli-bundle extraction on the engine side. A provider that declared +// no model surface would still leave a fully working picker over the +// webui layers plus the builtins. Gating the endpoint hard would REMOVE +// working functionality in response to a declaration about a capability +// the endpoint does not actually depend on — the exact reasoning +// `session-export.js` records for #11, reused here rather than +// re-argued. So `checkModelReadCapability` REPORTS and never throws; the +// read is unaffected by what it reports, and the report is what a later +// batch needs in order to decide whether the engine LAYER may be trusted. +// +// What this file deliberately does NOT do: +// +// - It does not re-implement the engine's own projections. +// `lib/engine-catalogue.js` owns the thinking schema, the +// context-window hygiene rules, the wire-form parser, the +// `custom_provider` read and the layer merge. A second projection +// here would be a second answer to a question with exactly one. +// - It does not write. `handleSetModel` stays in the route for B7/B9; +// this batch only moves the READ. +// - It does not build a second provider host and does not call any +// provider method: every source here is a file read, which is why +// the sub-item below names a READ rather than a method webui calls. +// +// Boot-path weight — the ONE place this batch deviates from the +// sibling families, and the deviation is deliberate on both sides of +// the import. +// +// The four static imports below (`lib/config.js`, +// `lib/engine-catalogue.js`, `lib/models.js`, `lib/providers-config.js`) +// were ALREADY static imports of `routes/model.js` before M3-B4, so the +// server's boot cost is exactly what it was. What they must not do is +// reach `@mavis/*` or `js-yaml` through the SHARED facade — and they +// do reach `@mavis/shared/local-runtime-paths` (via `lib/config.js`) +// and `js-yaml` (via `engine-provider-sync.js`). That is why this module +// is deliberately NOT re-exported from `engine/index.js`, and why +// `routes/model.js` imports it directly: `test/lib/engine/host-facade.test.js` +// guards `engine/index.js` and `routes/plugins.js` against exactly that +// pull, and the guard is right. See `engine/index.js` for the full +// statement and for what has to be true before this can move back. +// +// The price is that the read is SYNCHRONOUS. Making the four imports +// dynamic would let the module re-export from the facade, at the cost of +// turning `handleGetModels` into an async handler — a contract change +// for any caller that does not await, and the exact thing this batch +// promises not to do. The M1 lesson (209ms → 2700ms) was about the +// `@mavis/*` TypeScript host tree, which nothing here touches. +// +// Provider selection is M4's job, same as every other family: +// `providerByTransport()` maps a transport to a REGISTERED provider id; +// today only `runtime` has one, so the default `acp` transport reports +// `gate: "unregistered-transport"` and the read proceeds unchanged. + +import { MCODE_WEBUI_TRANSPORT } from "../lib/config.js"; +import { + mergeEngineAndWebuiProviders, + parseEngineModelWireValue, + readEngineBuiltinContextWindows, + readEngineBuiltinThinking, + readEngineCatalogue, +} from "../lib/engine-catalogue.js"; +import { getBuiltinModelsFromMcode } from "../lib/models.js"; +import { loadProvidersConfig } from "../lib/providers-config.js"; +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * The provider every builtin entry belongs to. + * + * The builtin shell is keyed by `minimax_api` REGARDLESS of the + * recorded pick. Deriving the group from `currentName.split("/")[0]` + * was the bug ticket 09-02's acceptance replay caught as "8 config + 6 + * misplaced MiniMax builtins = 14 in `nousresearch`": a user who picked + * a BYOK model dragged the engine's own builtins into that provider's + * group. The constant lives here now because the attribution rule and + * the group id are one decision. + */ +const BUILTIN_PROVIDER = "minimax_api"; + +/** The synthetic group id the engine session's own option list renders as. */ +const ENGINE_GROUP_ID = "__engine"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, mirroring + * `session-reads.js#providerByTransport`, `session-tree-reads.js`, + * `session-export.js` and `usage-reads.js`. Kept per-family so each + * family owns its own gate policy; collapse them in M4, not here. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +/** + * The declaration this endpoint's ENGINE LAYER needs, and the sub-item + * it needs from that capability. + * + * `authCredentials` is where the engine's model/provider surface is + * declared — the local-runtime-v2 declaration says so in its own + * comment ("full user model provider CRUD/test/discover, same source as + * service/model-system"), and the 14 matrix keys have no separate + * "models" row. `listModelProviders` names the READ, not a method webui + * calls: the custom_provider tree and the builtin tree are files the + * engine owns, read through `lib/engine-catalogue.js`, not a + * `CliService` method. That distinction is the reason this family's + * gate is soft — a missing declaration here removes ONE of the + * catalogue's three sources, never the endpoint. + * + * @type {Readonly>} + */ +export const MODEL_READ_ENDPOINTS = Object.freeze({ + "GET /api/models": { + capability: "authCredentials", + subItem: "listModelProviders", + enforcement: "soft", + }, +}); + +/** + * Resolve the provider that answers the model read on `transport`, or + * `null` when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveModelReadProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Read the declaration for #57 WITHOUT enforcing it. + * + * The `gate` values are the same vocabulary `session-export.js` uses, + * for the same reason: + * + * - `"checked"` — provider resolved, capability `full`. + * - `"unregistered-transport"` — no provider claims this transport yet. + * - `"capability-absent"` — the provider WAS found and does not + * offer the model surface. The caller's next move is to distrust the + * ENGINE LAYER, not to fail the request. + * - `"partial"` — the provider is `partial` and this + * sub-item is absent. + * + * Deliberately never throws `EngineCapabilityNotSupportedError`. See the + * header for why a hard gate here would remove working functionality. + * A genuinely unknown endpoint key is still a plain Error — caller + * confusion is not a capability question. + * + * @param {string} endpoint A key of MODEL_READ_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null, enforcement: "soft"}} + */ +export function checkModelReadCapability(endpoint, transport) { + const need = MODEL_READ_ENDPOINTS[endpoint]; + if (need === undefined) { + const err = new Error( + `checkModelReadCapability: "${endpoint}" is not part of the model family ` + + `(known: ${Object.keys(MODEL_READ_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_model_read_endpoint"; + throw err; + } + const base = { + endpoint, + provider: null, + capability: need.capability, + subItem: need.subItem, + enforcement: need.enforcement, + }; + const provider = resolveModelReadProvider(transport); + if (!provider) return { ...base, gate: "unregistered-transport" }; + const entry = provider.capabilities ? provider.capabilities[need.capability] : undefined; + const descriptor = { ...base, provider: provider.id }; + if (entry && entry.level === "full") { + return { ...descriptor, gate: "checked" }; + } + if (entry && entry.level === "partial") { + const absent = Array.isArray(entry.missing) && entry.missing.includes(need.subItem); + return { ...descriptor, gate: absent ? "partial" : "checked" }; + } + return { ...descriptor, gate: "capability-absent" }; +} + +// --------------------------------------------------------------------------- +// The projections. Pure functions, exported, and tested on their INPUTS. +// --------------------------------------------------------------------------- + +/** + * Coerce a provider prefix out of a model id. + * + * `minimax_api/MiniMax-M3` → `minimax_api`. Bare `MiniMax-M3` falls back + * to `minimax_api` (the engine's only shipping builtin provider) so a + * user-typed short id still resolves to a known group instead of + * orphaning itself. + * + * Used only for engine session entries (their ids are the engine's wire + * form `m:::u`); webui-side entries carry the + * provider as an explicit `entry.provider` field, and the multi-segment + * model id stays whole (see `webuiFullModelId`). + * + * @param {string} modelId + * @param {string} [fallback] + * @returns {string} + */ +export function providerOfModelId(modelId, fallback = BUILTIN_PROVIDER) { + if (!modelId) return fallback; + const i = modelId.indexOf("/"); + if (i <= 0) return fallback; + return modelId.slice(0, i); +} + +/** + * Build the webui internal id for a catalogue entry: `/`. + * + * The webui id is always two segments where the first is the provider + * key and the second is the engine-side model id verbatim (the engine + * allows `/` inside model ids — see `engine-catalogue.js`; the wire + * form `formatModelKey(, )` uses `/` as the only + * structural separator, so a downstream `/` webui + * form survives the round-trip through `resolveModelId`). + * + * Ticket 09-02 (grouping attribution): the previous implementation + * skipped the prefix when `m.id.includes("/")` and let the bare + * upstream id stand. That pushed the picker into the wrong group (the + * id's first segment was used as a fallback for the provider + * extraction) and let two providers with overlapping upstream ids + * collide on the `seen` dedupe (e.g. `z-ai/glm-5.3` in `nousresearch` + * ate the sibling `zai-max/glm-5.3`). Always prefixing — even when the + * model id already contains `/` — keys every entry by + * `(providerKey, modelId)` and the dedupe is per provider, as the + * ticket requires. + * + * @param {string} providerKey + * @param {string} modelId + * @returns {string} + */ +export function webuiFullModelId(providerKey, modelId) { + return `${providerKey}/${modelId}`; +} + +/** + * Attach the engine's context-window metadata ("U6") onto a catalogue + * entry, mutating `entry`. + * + * `contextWindowOptions` / `contextWindowOptionHints` come from the + * engine's materialised builtin tree (same read as the thinking + * projection — see `lib/engine-catalogue.js`). Only the `minimax_api` + * builtin entries carry them today: the engine's ACP `model` config + * option (the engine-session entries' source) does not advertise the + * metadata, so those entries are annotated through the same builtin + * projection keyed by the wire form's model id. Custom-provider / + * config-layer entries never get the fields — a model without options + * must stay field-free so the composer mounts no control. + * + * `contextLimit` (the CURRENT effective window, from the engine tree's + * `limit.context`) is attached when the entry has none yet — a config + * layer entry keeps its own value; builtin shell entries get the + * engine's current window so the picker can show the active radio + * before the user's first in-webui pick. + * + * @param {object} entry Mutated in place; the caller owns it. + * @param {{options: number[], hints?: object, currentLimit?: number}|null} projection + * @returns {void} + */ +export function attachContextWindowOptions(entry, projection) { + if (!projection) return; + entry.contextWindowOptions = [...projection.options]; + if (projection.hints) { + entry.contextWindowOptionHints = { ...projection.hints }; + } + if (entry.contextLimit === undefined && projection.currentLimit !== undefined) { + entry.contextLimit = projection.currentLimit; + } +} + +/** + * The engine session's `model` config option with this id, or `null` + * before a session exists. + * + * @param {object} cs + * @param {string} id + * @returns {object|null} + */ +export function configOption(cs, id) { + const options = Array.isArray(cs && cs.configOptions) ? cs.configOptions : []; + return options.find((o) => o && o.id === id) || null; +} + +/** + * Build the flat `models` list and the provider-grouped `groups` array. + * + * Pure, and the whole of #57's payload except the three derived "what + * is active" figures. The rules it encodes, in the order the endpoint + * has always applied them: + * + * 1. Engine session entries first, under the synthetic `__engine` + * group, with the engine's wire ids kept verbatim. They carry BOTH + * `name` and `label` because pre-existing callers (the composer + * chip) read `name` while the provider-grouped panel reads `label`. + * 2. Config-layer providers next, each under its own group, with the + * operator's per-model metadata winning wholesale on an id + * collision (the `seen` dedupe). A group is emitted even when its + * model list is empty — an operator who configured a provider with + * no models yet must still see the group to add one. + * 3. Builtins last, appended to the `minimax_api` group (created on + * demand) — and the empty shell is dropped only when there is no + * providers config at all, so a fresh install with a config that + * names no models still has somewhere to attach the builtins once + * the engine reports them. + * + * @param {object} options + * @param {object|null} options.sessionOption The engine `model` config option. + * @param {{providers: Array}|null} options.providers The merged + * providers config, or `null` when every layer was missing. + * @param {string[]} options.builtins Bare builtin model ids. + * @param {Map} options.builtinThinking + * @param {Map} options.builtinContextWindows + * @returns {{list: Array, groups: Array}} + */ +export function projectModelCatalogue(options = {}) { + const sessionOption = options.sessionOption || null; + const providers = options.providers || null; + const builtins = Array.isArray(options.builtins) ? options.builtins : []; + const builtinThinking = options.builtinThinking || new Map(); + const builtinContextWindows = options.builtinContextWindows || new Map(); + const parseWire = options.parseEngineModelWireValue || defaultParseWireStub; + + const list = []; + const groups = []; + const seen = new Set(); + + // 1) Engine session config option — authoritative when present. + if (sessionOption) { + const engineGroup = { id: ENGINE_GROUP_ID, label: "Engine session", models: [] }; + for (const o of Array.isArray(sessionOption.options) ? sessionOption.options : []) { + const id = o && typeof o.value === "string" ? o.value : null; + if (!id) continue; + if (seen.has(id)) continue; + seen.add(id); + const displayName = (o && o.name) || id; + const entry = { + id, + name: displayName, + label: displayName, + provider: providerOfModelId(id), + source: "engine", + }; + // The engine's wire-form `currentValue` is mirrored into + // `cs.model.name` outside the pick window and the composer matches + // the active model by id — annotate the wire-form entries too so + // the thinking and context controls survive a cross-client change. + const wire = parseWire(id); + if (wire && wire.providerId === BUILTIN_PROVIDER) { + const proj = builtinThinking.get(wire.modelId); + if (proj) entry.thinkingLevels = [...proj.levels]; + attachContextWindowOptions(entry, builtinContextWindows.get(wire.modelId)); + } + engineGroup.models.push(entry); + list.push(entry); + } + if (engineGroup.models.length > 0) groups.push(engineGroup); + } + + // 2) Providers config — read every request so editing the file does not + // require a restart. Config wins on id collision with the builtin + // catalogue so providers can override labels and contextLimit. + if (providers) { + for (const p of providers.providers) { + if (!p || typeof p.id !== "string" || !p.id) continue; + const models = []; + for (const m of Array.isArray(p.models) ? p.models : []) { + if (!m || typeof m.id !== "string" || !m.id) continue; + const fullId = webuiFullModelId(p.id, m.id); + if (seen.has(fullId)) continue; + seen.add(fullId); + const entry = { + id: fullId, + label: typeof m.label === "string" && m.label ? m.label : m.id, + provider: p.id, + source: "config", + }; + if (typeof m.contextLimit === "number" && m.contextLimit > 0) { + entry.contextLimit = m.contextLimit; + } + // v2 schema surfaces: each model carries protocol + + // thinkingLevels + modalities so the selector can pick the right + // controls without a second round-trip. `auth` only exposes + // hasKey + type — an apiKey NEVER reaches this response. + if (typeof p.protocol === "string" && p.protocol) { + entry.protocol = p.protocol; + } + if (Array.isArray(m.thinkingLevels) && m.thinkingLevels.length > 0) { + entry.thinkingLevels = [...m.thinkingLevels]; + } + if (Array.isArray(m.modalities) && m.modalities.length > 0) { + entry.modalities = [...m.modalities]; + } + models.push(entry); + list.push(entry); + } + // Auth shape: only `hasKey` and `type`; no apiKey/baseURL. + // Operators see "configured or not" without leaking the secret. + // The merged layer (engine + webui) may carry `hasKey` either via + // `p.auth.apiKey` (webui-side plaintext — masked elsewhere) or via + // `p.auth.hasKey` (engine-side boolean, set by + // `lib/engine-catalogue.js`). Either signal means the provider is + // configurable from the picker. + const groupHasKey = !!((p.auth && p.auth.apiKey) || (p.auth && p.auth.hasKey)); + groups.push({ + id: p.id, + label: typeof p.label === "string" && p.label ? p.label : p.id, + auth: { + hasKey: groupHasKey, + type: p.auth && typeof p.auth.type === "string" ? p.auth.type : "byok", + }, + protocol: typeof p.protocol === "string" ? p.protocol : "openai", + models, + }); + } + } + + // 3) Builtin catalogue. The builtins all belong to the engine's + // `minimax_api` provider (see `lib/models.js#getBuiltinModelsFromMcode` + // — the cli.js extraction regex targets `MiniMax-M*`). + let builtinGroup = groups.find((g) => g.id === BUILTIN_PROVIDER); + if (!builtinGroup) { + builtinGroup = { id: BUILTIN_PROVIDER, label: BUILTIN_PROVIDER, models: [] }; + groups.push(builtinGroup); + } + for (const m of builtins) { + const fullId = webuiFullModelId(BUILTIN_PROVIDER, m); + if (seen.has(fullId)) continue; + seen.add(fullId); + const entry = { + id: fullId, + label: m, + provider: BUILTIN_PROVIDER, + source: "builtin", + }; + // `thinkingLevels` is exactly what the engine's tree supports — + // ["off","on"] for a switchable variant toggle, the engine's effort + // list when the model has one, and ABSENT for a forced_on model with + // nothing user-settable (the composer then mounts no control, by + // design). A config-layer entry with the same id has already taken + // the slot (seen dedupe) — the operator's config wins wholesale, + // unchanged rule. + const proj = builtinThinking.get(m); + if (proj) entry.thinkingLevels = [...proj.levels]; + attachContextWindowOptions(entry, builtinContextWindows.get(m)); + list.push(entry); + builtinGroup.models.push(entry); + } + + // Drop the empty builtin shell — a no-bundle empty group is noise. + // The drop is gated on "no providers config" so a fresh install with + // a config that names no models still has somewhere to attach the + // builtins once the engine reports them. + if (builtinGroup.models.length === 0 && !providers) { + const idx = groups.indexOf(builtinGroup); + if (idx >= 0) groups.splice(idx, 1); + } + + return { list, groups }; +} + +/** + * The three "what is active right now" figures #57 reports. + * + * `current` is the engine's value when one exists, otherwise the + * recorded pre-session choice (`cs.model.name`, written by + * `handleSetModel`). When neither exists the answer is `null` rather + * than a fallback to a default model — the old behaviour invented an + * active model the engine never confirmed, and the chip ended up + * claiming a model the session was not actually running. The chip + * renders a neutral label when `current` is `null` (see + * `composer.tsx#currentModelLabel`). + * + * `currentThinking` prefers the engine's `thinkingEffort` option and + * falls back to `cs.model.thinking` (the pre-session record that + * `applyConfigOptionUpdate` refreshes). The selector reads it to + * highlight the active level and to skip the picker when the active + * model has no `thinkingLevels`. + * + * `currentContextWindow` is the recorded choice (`handleSetModel` + * writes `cs.model.contextWindow`) with the current model's catalogue + * `contextLimit` as fallback. There is no engine-value branch for the + * window, deliberately: the engine's ACP surface has no context + * channel, so the recorded pick is the only source. A recorded value + * the current model no longer advertises is still reported verbatim — + * the stale-pick display rule lives in the composer. + * + * @param {object} options + * @param {object|null} options.sessionOption + * @param {object} options.cs + * @param {Array} options.list The projected flat list. + * @returns {{current: string|null, currentThinking: string|null, currentContextWindow: number|null}} + */ +export function deriveModelSelection(options = {}) { + const sessionOption = options.sessionOption || null; + const cs = options.cs || {}; + const list = Array.isArray(options.list) ? options.list : []; + const current = + (sessionOption && sessionOption.currentValue) || + (cs.model && typeof cs.model.name === "string" && cs.model.name) || + null; + const thinkingEffortOption = Array.isArray(cs.configOptions) + ? cs.configOptions.find((o) => o && o.id === "thinkingEffort") + : null; + const currentThinking = + (thinkingEffortOption && typeof thinkingEffortOption.currentValue === "string" + ? thinkingEffortOption.currentValue + : null) || + (cs.model && typeof cs.model.thinking === "string" && cs.model.thinking) || + null; + const recordedContextWindow = + cs.model && Number.isSafeInteger(cs.model.contextWindow) && cs.model.contextWindow > 0 + ? cs.model.contextWindow + : null; + const currentModelEntry = current ? list.find((m) => m.id === current) : null; + const currentContextWindow = + recordedContextWindow ?? + (currentModelEntry && + Number.isSafeInteger(currentModelEntry.contextLimit) && + currentModelEntry.contextLimit > 0 + ? currentModelEntry.contextLimit + : null); + return { current, currentThinking, currentContextWindow }; +} + +/** + * The endpoint's `source` label — which of the three layers won. + * + * @param {object} options + * @param {object|null} options.sessionOption + * @param {{providers: Array}|null} options.providers + * @returns {"acp-session-config"|"config+mcode-cli-bundle"|"mcode-cli-bundle"} + */ +export function catalogueSourceLabel(options = {}) { + const sessionOption = options.sessionOption || null; + if (sessionOption && Array.isArray(sessionOption.options) && sessionOption.options.length > 0) { + return "acp-session-config"; + } + return options.providers ? "config+mcode-cli-bundle" : "mcode-cli-bundle"; +} + +/** + * Compose the endpoint's response body. The key ORDER is the endpoint's + * and is asserted by the test suite: `ok`, `models`, `groups`, the three + * derived figures, `source`, and the soft-failure `reason` marker that + * is spread LAST and only when the catalogue came out empty. + * + * The marker is backwards compatibility with the older engine-only + * build. With the merge it should be rare (builtin catalogue + + * providers config cover most installs), but a missing cli bundle AND + * an absent config leave the catalogue empty — and a caller that wants + * to know "is this a hard failure or just no engine attached?" still + * gets the same hint. + * + * @param {object} options + * @param {object|null} options.sessionOption + * @param {{providers: Array}|null} options.providers + * @param {string[]} options.builtins + * @param {Map} options.builtinThinking + * @param {Map} options.builtinContextWindows + * @param {object} options.cs + * @param {Function} [options.parseEngineModelWireValue] + * @returns {object} The exact #57 response body. + */ +export function buildModelCataloguePayload(options = {}) { + const { list, groups } = projectModelCatalogue(options); + const selection = deriveModelSelection({ + sessionOption: options.sessionOption, + cs: options.cs, + list, + }); + const source = catalogueSourceLabel(options); + return { + ok: true, + models: list, + groups, + current: selection.current, + currentThinking: selection.currentThinking, + currentContextWindow: selection.currentContextWindow, + source, + ...(list.length === 0 ? { reason: "no_catalogue" } : {}), + }; +} + +/** + * The wire-form parser used when the caller does not inject one. Only + * ever reached from a unit test that calls `projectModelCatalogue` + * without the engine-catalogue module; the read always injects the real + * parser. A stub that returns `null` is the honest "not a wire form" + * answer, which simply skips the builtin annotation — the same path a + * plain id takes. + */ +function defaultParseWireStub() { + return null; +} + +// --------------------------------------------------------------------------- +// The read +// --------------------------------------------------------------------------- + +/** + * Where the catalogue's bytes came from. Always a layered `config`: + * three sources, of which only the `custom_provider` layer is the + * engine's, and the merged shape is webui's v2 `{providers}` view. The + * per-entry `source` field ("engine" | "config" | "builtin") is the + * fine-grained answer; this is the coarse one, kept so the descriptor + * vocabulary matches the other families. + * + * @typedef {"config"} ModelReadSource + */ + +/** + * The #57 (`GET /api/models`) read. + * + * SYNCHRONOUS, deliberately — see the boot-path note in the header. The + * route's handler signature is part of its contract: `app.js#invokeHandler` + * accepts both shapes, but a caller that does not await gets a + * half-written response from an async handler and a complete one from a + * sync handler, and this batch is a收编, not a scheduling change. + * + * Every source is re-read on every call, exactly as before: editing + * `models.json`, `~/.mcode-webui/providers.json` or the engine's + * `config.yaml` must not require a server restart. The `payload` is + * the endpoint's response body verbatim, including the soft-failure + * `reason` marker for an empty catalogue — this facade does not convert + * that into an error, because "no engine attached yet" is a state the + * picker renders, not a failure. + * + * @param {object} [options] + * @param {object} [options.cs] The webui client state; `configOptions`, + * `model.name`, `model.thinking` and `model.contextWindow` are + * read from it, and the first two are echoed into the derived + * figures. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/models`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @returns {{payload: object, source: ModelReadSource, gate: object, transport: string}} + */ +export function readEngineModelCatalogue(options = {}) { + const endpoint = options.endpoint || "GET /api/models"; + const transport = options.transport || MCODE_WEBUI_TRANSPORT; + const gate = checkModelReadCapability(endpoint, transport); + const cs = options.cs || {}; + const sessionOption = configOption(cs, "model"); + // The merged `{providers}` view, or `null` when every layer is + // missing. The engine catalogue read is best-effort: a missing + // `config.yaml` or a YAML parse error yields `[]`, and the merge + // treats an empty engine catalogue as "no engine layer" — matching + // the pre-ticket-06 behaviour for installs without an engine config. + // The `try/catch` is the endpoint's own: a malformed webui layer must + // degrade the catalogue to "webui layers only", never 500 the picker. + let providers = null; + try { + const cfg = loadProvidersConfig(); + const webuiProviders = cfg && Array.isArray(cfg.providers) ? cfg.providers : []; + const merged = mergeEngineAndWebuiProviders(readEngineCatalogue(), webuiProviders); + if (merged.length > 0) providers = { providers: merged }; + } catch { + providers = null; + } + const payload = buildModelCataloguePayload({ + sessionOption, + providers, + builtins: getBuiltinModelsFromMcode(), + builtinThinking: readEngineBuiltinThinking(), + builtinContextWindows: readEngineBuiltinContextWindows(), + cs, + parseEngineModelWireValue, + }); + return { payload, source: "config", gate, transport }; +} diff --git a/packages/webui/server/routes/account.js b/packages/webui/server/routes/account.js index 915c9a18..f9efb57f 100644 --- a/packages/webui/server/routes/account.js +++ b/packages/webui/server/routes/account.js @@ -1,7 +1,21 @@ // webui/server/routes/account.js // GET /api/account — the account card's data. +// +// M3-B4: the read now goes through the engine facade +// (`server/engine/account-reads.js`) instead of naming +// `lib/mcode-rpc.js` directly, so the endpoint is gated on the same +// declared `authCredentials.getAccountStatus` the usage popover +// (#15 / #16) is gated on — the two read the SAME engine projection +// through the SAME `mcode/account/status` method, and a provider that +// drops it must take both down together. +// +// Nothing about the wire changed. The facade builds the response body +// (success spreads the engine's projection verbatim; failure keeps the +// `{ok:false, reason}` soft-fail shape), and the HTTP status stays 200 +// in both cases: the REQUEST succeeded, and the card renders its empty +// state from `ok:false`. -import { getAccountStatus } from "../lib/mcode-rpc.js"; +import { readEngineAccount } from "../engine/account-reads.js"; /** * Fetched on demand rather than pushed in the state snapshot. @@ -12,14 +26,13 @@ import { getAccountStatus } from "../lib/mcode-rpc.js"; * no credential (see acp/extensions.ts), and nothing here logs the response. * * A failure is a soft one, like /api/session-tree: the card renders its empty - * state rather than the route inventing a name or a plan. + * state rather than the route inventing a name or a plan. The capability gate is + * a different question from that one — "may this provider report an account at + * all" versus "could we read the account this time" — and only the first one + * produces a 501, through `app.js#invokeHandler`. */ export async function handleGetAccount(_req, res, ctx) { - const cs = ctx && ctx.cs; - const r = await getAccountStatus(cs && cs.mcodeSessionId); + const { payload } = await readEngineAccount({ cs: ctx && ctx.cs }); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - if (!r.ok) { - return res.end(JSON.stringify({ ok: false, reason: r.code || "account_unavailable" })); - } - return res.end(JSON.stringify({ ok: true, ...(r.data || {}) })); + return res.end(JSON.stringify(payload)); } diff --git a/packages/webui/server/routes/model.js b/packages/webui/server/routes/model.js index 6117e67a..ef1526e0 100644 --- a/packages/webui/server/routes/model.js +++ b/packages/webui/server/routes/model.js @@ -1,5 +1,18 @@ // webui/server/routes/model.js // GET /api/models, POST /api/set-model, POST /api/permissions, POST /api/answer (legacy) +// +// M3-B4: `GET /api/models` now reads the catalogue through the engine +// facade (`server/engine/model-reads.js`) instead of assembling it +// here. The three sources (the engine session's `model` config option, +// the merged providers config with the engine's `custom_provider` +// tree as its bottom layer, the builtin cli-bundle extraction), the +// two builtin-tree annotations (variant-style thinking levels and +// context-window options) and the three derived "what is active" figures +// all moved with it, as named pure functions pinned on their inputs. +// +// The response is byte-identical. This batch only moves the READ: the +// WRITE half (`handleSetModel`) stays here for B7/B9, together with the +// two other handlers below. import { readFileSync } from "node:fs"; import { join } from "node:path"; @@ -11,70 +24,26 @@ import { webuiPermissionToMcode, PERMISSION_MODES, } from "../lib/mcode-rpc.js"; -import { getBuiltinModelsFromMcode } from "../lib/models.js"; -import { loadProvidersConfig } from "../lib/providers-config.js"; -import { - readEngineCatalogue, - readEngineBuiltinThinking, - readEngineBuiltinContextWindows, - parseEngineModelWireValue, - variantChannelFor, - resolveModelId, - mergeEngineAndWebuiProviders, -} from "../lib/engine-catalogue.js"; +import { readEngineModelCatalogue } from "../engine/model-reads.js"; +import { variantChannelFor, resolveModelId } from "../lib/engine-catalogue.js"; import { webuiModeToLabel } from "../lib/interaction/permission-presets.js"; import { readJson } from "../lib/read-json.js"; -/** The engine's `select` config option with this id, or null before a session exists. */ -function configOption(cs, id) { - const options = Array.isArray(cs && cs.configOptions) ? cs.configOptions : []; - return options.find((o) => o && o.id === id) || null; -} - /** - * Attach the engine's context-window metadata (U6) onto a builtin - * `minimax_api` catalogue entry, mutating `entry`. - * - * `contextWindowOptions` / `contextWindowOptionHints` come from the - * engine's materialised builtin tree (same read as the thinking - * projection — see `lib/engine-catalogue.js`). Only the minimax_api - * builtin entries carry them today: the engine's ACP `model` config - * option (the engine-session entries' source) does not advertise the - * metadata, so those entries are annotated through the same builtin - * projection keyed by the wire form's model id. Custom-provider / - * config-layer entries never get the fields — a model without options - * must stay field-free so the composer mounts no control. - * - * `contextLimit` (the CURRENT effective window, from the engine tree's - * `limit.context`) is attached when the entry has none yet — a config - * layer entry keeps its own value; builtin shell entries get the - * engine's current window so the picker can show the active radio - * before the user's first in-webui pick. - */ -function attachContextWindowOptions(entry, projection) { - if (!projection) return; - entry.contextWindowOptions = [...projection.options]; - if (projection.hints) { - entry.contextWindowOptionHints = { ...projection.hints }; - } - if (entry.contextLimit === undefined && projection.currentLimit !== undefined) { - entry.contextLimit = projection.currentLimit; - } -} - -/** - * Read the optional providers-config file. + * Read the optional providers-config file — the v1 single-file reader. * * Path precedence: `MCODE_WEBUI_MODELS_CONFIG` env → `/models.json`. * Shape: `{ providers: [{ id, label, models: [{ id, label?, contextLimit? }] }] }`. - * Re-read on every request: editing the file does not require a server restart. * Missing / unreadable / malformed → null (treated as "no config"). * - * v2 layered resolution lives in `loadProvidersConfig()` (env > cwd > - * user-level with deep merge). The /api/models route now reads - * through that helper, so an env override of `MCODE_WEBUI_MODELS_CONFIG` - * continues to win over the cwd file (matching the v1 contract), and - * a `~/.mcode-webui/providers.json` layer is layered under both. + * KNOWN DEBT, kept deliberately: nothing calls this any more. The v2 + * layered resolution in `loadProvidersConfig()` (env > cwd > user-level + * with deep merge) replaced it when #57 moved into + * `engine/model-reads.js`, and the function was already unreferenced + * before that move. It is retained rather than deleted because it is + * the written record of the v1 contract `loadProvidersConfig`'s own + * header cites; delete it in a batch whose subject is dead code, not as + * a side effect of moving a read. */ function readModelsConfig() { const path = @@ -89,94 +58,21 @@ function readModelsConfig() { } } -/** - * Layered resolver used by /api/models. Returns the merged - * `{ providers }` (v2 shape) or `null` when every layer is missing. - * - * Ticket 06: the engine's `custom_provider` tree is the new bottom - * layer; the webui layers (env > cwd > user, already merged inside - * `loadProvidersConfig`) win on id collision. The merge itself - * lives in `mergeEngineAndWebuiProviders()` — see its file header - * for the precedence rules. The helper here just shapes its - * return into the legacy `{ providers: [...] }` view that - * handleGetModels already understood. - */ -function readProvidersConfigForModels() { - try { - const cfg = loadProvidersConfig(); - const webuiProviders = (cfg && Array.isArray(cfg.providers)) ? cfg.providers : []; - // Engine catalogue read is best-effort: a missing `config.yaml` - // or a YAML parse error yields []. The merge below treats an - // empty engine catalogue as "no engine layer" and returns the - // webui layers verbatim — matching the pre-ticket-06 behaviour - // for installs without an engine config. - const engineProviders = readEngineCatalogue(); - const merged = mergeEngineAndWebuiProviders(engineProviders, webuiProviders); - if (merged.length === 0) return null; - return { providers: merged }; - } catch { - return null; - } -} - -/** - * Coerce a provider prefix out of a model id. - * - * `minimax_api/MiniMax-M3` → `minimax_api`. Bare `MiniMax-M3` falls back to - * `minimax_api` (the engine's only shipping builtin provider) so a user-typed - * short id still resolves to a known group instead of orphaning itself. - * - * Used only for engine session entries (their ids are the engine's wire - * form `m:::u`); webui-side entries now carry the - * provider as an explicit `entry.provider = p.id` field, and the multi-segment - * model id stays whole (see `webuiFullModelId`). - */ -function providerOf(modelId, fallback = "minimax_api") { - if (!modelId) return fallback; - const i = modelId.indexOf("/"); - if (i <= 0) return fallback; - return modelId.slice(0, i); -} - -/** - * Build the webui internal id for a catalogue entry: `/`. - * - * The webui id is always two segments where the first is the provider key - * and the second is the engine-side model id verbatim (the engine allows - * `/` inside model ids — see engine-catalogue.js; the wire form - * `formatModelKey(, ) = /` uses - * `/` as the only structural separator, so a downstream `/` - * webui form survives the round-trip through `resolveModelId`). - * - * Ticket 09-02 (grouping attribution): the previous implementation - * skipped the prefix when `m.id.includes("/")` and let the bare upstream - * id stand. That pushed the picker into the wrong group (the id's first - * segment was used as a fallback for `providerOf`) and let two providers - * with overlapping upstream ids collide on the `seen` dedupe (e.g. - * `z-ai/glm-5.3` in `nousresearch` ate the sibling `zai-max/glm-5.3`). - * Always prefixing — even when the model id already contains `/` — - * keys every entry by `(providerKey, modelId)` and the dedupe is per - * provider, as the ticket requires. - */ -function webuiFullModelId(providerKey, modelId) { - return `${providerKey}/${modelId}`; -} - /** * Translate a webui-recorded model id to the engine's wire form. * * The webui records `cs.model.name` in `/` - * form (see `webuiFullModelId`). The engine's `set_config_option` for - * `configId: "model"` rejects anything that isn't the wire form - * `m:::u` (see + * form (see `engine/model-reads.js#webuiFullModelId`). The engine's + * `set_config_option` for `configId: "model"` rejects anything that + * isn't the wire form `m:::u` (see * packages/tui/src/acp/control-state.ts#modelConfigValue / agent.ts * `parseModelConfigValue`). Without this translation a mid-session * pick of a multi-segment model id (`nousresearch/deepseek/x`) would * 400 from the engine. * - * `resolveModelId` (in `lib/mcode-acp.js`) owns the resolver — it is - * the same code path `applyRecordedModel` uses on session boot, so the - * mid-session push and the boot-time replay share one source of + * `resolveModelId` (in `lib/engine-catalogue.js`) owns the resolver — + * it is the same code path `applyRecordedModel` uses on session boot, so + * the mid-session push and the boot-time replay share one source of * truth. Returns `null` when the engine has no matching option yet * (the engine configOptions list is empty before the first session * event lands); the caller falls back to the recorded id and the @@ -195,321 +91,41 @@ function translateWebuiModelIdToEngineValue(cs, modelId, resolveOpts) { } /** - * GET /api/models — catalogue, with priority-aware merging. + * GET /api/models — the composer model picker, through the engine + * facade. + * + * The endpoint's whole contract is the payload the facade built: * - * Priority order (highest wins for `current`, first wins for each id): - * 1. Engine session's `model` config option. Its `options[].value` is - * the engine's encoded id (e.g. `m:::v:`), - * so it round-trips straight through `POST /api/set-model`. Used - * when a session is active. - * 2. Optional `MCODE_WEBUI_MODELS_CONFIG` / `models.json` providers - * config. Per-provider groups with labels and `contextLimit`s. - * Ticket 06: this layer is the webui-side merge of - * `env > cwd > user-level`, with the engine's - * `custom_provider` tree as a new bottom layer — see - * `lib/engine-catalogue.js` for the merge rules. - * 3. `getBuiltinModelsFromMcode()` — extracted from mcode's own - * cli.js bundle, so the list tracks mcode's TUI without a webui - * release. + * - `models` — the flat list, every entry carrying `id` / `label` / + * `provider` / `source` plus whatever that source contributes + * (`contextLimit`, `protocol`, `thinkingLevels`, `modalities`, + * `contextWindowOptions`). + * - `groups` — the same entries grouped by provider, so the picker can + * render sections instead of a flat list. This is red line five's + * "模型按供应商分组": the group id is the provider key, and the + * builtin shell is always `minimax_api` regardless of the recorded + * pick. + * - `current` / `currentThinking` / `currentContextWindow` — the three + * derived figures, resolved engine-value-first and never invented + * from a default. + * - `source` — which layer won. + * - `reason: "no_catalogue"` — the soft marker, spread last and only + * when the catalogue came out empty. * - * `current` resolution: - * - With an active session config option: `option.currentValue`. - * - Without one: the recorded pre-session choice (`cs.model.name`), - * which `handleSetModel` already writes — so the selector shows - * the user's pick even before the engine attaches. + * `engine/model-reads.js` owns the projection rules and their + * derivations; this route writes the body. The facade re-reads every + * source on every request, so editing `models.json`, + * `~/.mcode-webui/providers.json` or the engine's `config.yaml` still + * takes effect without a restart. * - * Response carries `groups` so the UI can render provider sections, - * alongside the flat `models` array for callers that do not care - * about grouping. + * Still a SYNCHRONOUS handler, exactly as before: the facade's read is + * synchronous too, because every source it needs was already a static + * import of this route (see the boot-path note in the engine module). */ export function handleGetModels(_req, res, ctx) { - const cs = ctx.cs; - const option = configOption(cs, "model"); - const engineOption = option; // keep the alias so reviewers can read priority order - - const list = []; - const groups = []; - const seen = new Set(); - // Ticket 36 — the engine's materialised builtin tree (provider. - // minimax.models) carries the variant-style thinking schema that - // /api/models never projected: switchable models became a two-state - // ["off","on"] toggle, forced_on+effortOptions models expose the - // engine's depth list verbatim, everything else stays metadata-free. - // One read serves both annotation sites below (engine-session - // entries and the builtin shell). - const builtinThinking = readEngineBuiltinThinking(); - // U6 — same tree, context-window projection. One read serves both - // annotation sites below (engine-session entries and the builtin - // shell), exactly like `builtinThinking`. - const builtinContextWindows = readEngineBuiltinContextWindows(); - - // 1) Engine session config option — authoritative when present. We keep - // its encoded ids verbatim so /api/set-model round-trips. Both `name` - // and `label` are set on engine-sourced entries because pre-existing - // callers (the composer chip) read `name`, while the new - // provider-grouped panel reads `label`. - if (engineOption) { - const engineGroupId = "__engine"; - const engineGroup = { - id: engineGroupId, - label: "Engine session", - models: [], - }; - for (const o of Array.isArray(engineOption.options) ? engineOption.options : []) { - const id = o && typeof o.value === "string" ? o.value : null; - if (!id) continue; - if (seen.has(id)) continue; - seen.add(id); - const displayName = (o && o.name) || id; - const entry = { - id, - name: displayName, - label: displayName, - provider: providerOf(id), - source: "engine", - }; - // Ticket 36: applyConfigOptionUpdate mirrors the engine's - // wire-form currentValue into cs.model.name outside the pick - // window, and the composer matches the active model by id — - // annotate the wire-form entries too so the thinking control - // survives a cross-client change. - const wire = parseEngineModelWireValue(id); - if (wire && wire.providerId === "minimax_api") { - const proj = builtinThinking.get(wire.modelId); - if (proj) entry.thinkingLevels = [...proj.levels]; - // U6: annotate the wire-form entries with the engine's - // context-window options too, so the picker's detail area - // survives a cross-client model change (same reasoning as the - // thinkingLevels annotation above). - attachContextWindowOptions(entry, builtinContextWindows.get(wire.modelId)); - } - engineGroup.models.push(entry); - list.push(entry); - } - if (engineGroup.models.length > 0) groups.push(engineGroup); - } - - // 2) Providers config — read every request so editing the file does not - // require a restart. Config wins on id collision with the builtin - // catalogue so providers can override labels and contextLimit. - // - // v2 layered resolution (env > cwd > user-level) is provided by - // `loadProvidersConfig()`; the v1 single-file reader stays as a - // fallback for callers that pass the legacy `models.json` - // through a different code path (none today, but keeping it - // documents the contract). - // - // Ticket 06: the engine's `custom_provider` tree is also a - // catalogue source — readEngineCatalogue() projects it to the - // v2 shape (no key material) and mergeEngineAndWebuiProviders() - // unions it with the webui layers (webui wins on collision). - const config = readProvidersConfigForModels(); - if (config) { - for (const p of config.providers) { - if (!p || typeof p.id !== "string" || !p.id) continue; - const models = []; - for (const m of Array.isArray(p.models) ? p.models : []) { - if (!m || typeof m.id !== "string" || !m.id) continue; - // Ticket 09-02: always prefix the webui id with ``. The - // upstream-style model id (`deepseek/x`, `z-ai/glm-5.3`, - // `openai/gpt-5.6-sol`) is kept verbatim inside the model id - // portion — the engine allows `/` inside model keys, the wire - // form `/` uses `/` only as the structural - // separator, and the `seen` dedupe is per provider (so two - // sibling providers with overlapping upstream ids stay - // distinct instead of one swallowing the other). - const fullId = webuiFullModelId(p.id, m.id); - if (seen.has(fullId)) continue; - seen.add(fullId); - const entry = { - id: fullId, - label: typeof m.label === "string" && m.label ? m.label : m.id, - provider: p.id, - source: "config", - }; - if (typeof m.contextLimit === "number" && m.contextLimit > 0) { - entry.contextLimit = m.contextLimit; - } - // v2 schema surfaces: each model carries protocol + - // thinkingLevels + modalities so the selector can pick the - // right controls without a second round-trip. `auth` only - // exposes hasKey + type — apiKey NEVER reaches this response. - if (typeof p.protocol === "string" && p.protocol) { - entry.protocol = p.protocol; - } - if (Array.isArray(m.thinkingLevels) && m.thinkingLevels.length > 0) { - entry.thinkingLevels = [...m.thinkingLevels]; - } - if (Array.isArray(m.modalities) && m.modalities.length > 0) { - entry.modalities = [...m.modalities]; - } - models.push(entry); - list.push(entry); - } - // Auth shape: only `hasKey` and `type`; no apiKey/baseURL. - // Operators see "configured or not" without leaking the secret. - // Ticket 06: the merged layer (engine + webui) may carry - // `hasKey` either via `p.auth.apiKey` (webui-side plaintext — - // masked elsewhere) or via `p.auth.hasKey` (engine-side - // boolean, set by `lib/engine-catalogue.js`). Either signal - // means the provider is configurable from the picker. - const groupHasKey = !!( - (p.auth && p.auth.apiKey) || - (p.auth && p.auth.hasKey) - ); - groups.push({ - id: p.id, - label: typeof p.label === "string" && p.label ? p.label : p.id, - auth: { - hasKey: groupHasKey, - type: p.auth && typeof p.auth.type === "string" ? p.auth.type : "byok", - }, - protocol: typeof p.protocol === "string" ? p.protocol : "openai", - models, - }); - } - } - - // 3) Builtin catalogue (extracted from mcode's cli.js bundle). The - // builtins all belong to the engine's `minimax_api` provider - // (see `lib/models.js#getBuiltinModelsFromMcode` — the cli.js - // extraction regex targets `MiniMax-M*`). The builtin shell is - // keyed by `minimax_api` regardless of the recorded pick, so a - // pick of `nousresearch/openai/gpt-5.6-sol` doesn't drag the - // MiniMax builtins into the `nousresearch` group. The previous - // behaviour derived the builtin group's id from - // `currentName.split("/")[0]`, which landed the builtins under - // whichever provider the user happened to have picked (the - // ticket 09-02 acceptance replay caught this as "8 config + 6 - // misplaced MiniMax builtins = 14 in `nousresearch`"). - const builtins = getBuiltinModelsFromMcode(); - const BUILTIN_PROVIDER = "minimax_api"; - // The recorded pre-session pick — used below for `current`, NOT for - // builtin-group attribution (the builtin shell is keyed by - // BUILTIN_PROVIDER above). - const currentName = - (cs.model && typeof cs.model.name === "string" && cs.model.name) || ""; - let builtinGroup = groups.find((g) => g.id === BUILTIN_PROVIDER); - if (!builtinGroup) { - builtinGroup = { id: BUILTIN_PROVIDER, label: BUILTIN_PROVIDER, models: [] }; - groups.push(builtinGroup); - } - for (const m of builtins) { - const fullId = `${BUILTIN_PROVIDER}/${m}`; - if (seen.has(fullId)) continue; - seen.add(fullId); - const entry = { - id: fullId, - label: m, - provider: BUILTIN_PROVIDER, - source: "builtin", - }; - // Ticket 36: attach the engine's thinking metadata for this - // builtin. `thinkingLevels` is exactly what the engine's tree - // supports — ["off","on"] for a switchable variant toggle, the - // engine's effort list when the model has one, and ABSENT for a - // forced_on model with nothing user-settable (the composer then - // mounts no control, by design). A config-layer entry with the - // same id has already taken the slot (seen dedupe) — the - // operator's config wins wholesale, unchanged rule. - const proj = builtinThinking.get(m); - if (proj) entry.thinkingLevels = [...proj.levels]; - // U6: the engine's context-window options for this builtin, plus - // its current effective window as the `contextLimit` fallback. - attachContextWindowOptions(entry, builtinContextWindows.get(m)); - list.push(entry); - builtinGroup.models.push(entry); - } - - // Drop the empty builtin shell — a no-bundle empty group is noise. - // The drop is gated on "no providers config" so a fresh install with - // a config that names no models still has somewhere to attach the - // builtins once mcode reports them. - if (builtinGroup && builtinGroup.models.length === 0 && !config) { - const idx = groups.indexOf(builtinGroup); - if (idx >= 0) groups.splice(idx, 1); - } - - // `current` is the engine's value when one exists; otherwise the - // recorded pre-session choice (`cs.model.name`, written by - // `handleSetModel`). When neither exists we report `null` rather than - // falling back to `DEFAULT_MODEL` — the old behaviour invented an - // active model the engine never confirmed, and the chip ended up - // claiming a model the session was not actually running. The chip - // renders a neutral label when `current` is `null` (see composer.tsx - // currentModelLabel). - const current = - (option && option.currentValue) || - currentName || - null; - - // Current thinking-effort level: read the engine's `thinkingEffort` - // option when present; otherwise fall back to `cs.model.thinking`, - // which `handleSetModel` writes (pre-session record) and which the - // engine's `config_option_update` notification refreshes via - // `applyConfigOptionUpdate` (see lib/mcode-acp.js). The selector - // reads this to highlight the active level and to skip the picker - // when the active model has no `thinkingLevels`. - const thinkingEffortOption = - Array.isArray(cs && cs.configOptions) ? cs.configOptions.find((o) => o && o.id === "thinkingEffort") : null; - const currentThinking = - (thinkingEffortOption && typeof thinkingEffortOption.currentValue === "string" - ? thinkingEffortOption.currentValue - : null) || - (cs && cs.model && typeof cs.model.thinking === "string" && cs.model.thinking) || - null; - - // U6 — the recorded context-window choice (`handleSetModel` writes - // `cs.model.contextWindow`). There is no engine config option behind - // it (the engine's ACP surface has no context channel — see the - // handleSetModel header), so unlike `currentThinking` there is no - // engine-value branch: the recorded pick is the only source. A - // recorded value the current model no longer advertises is still - // reported verbatim — the stale-pick display rule lives in the - // composer (same split as the thinking level's stale-suffix guard). - const recordedContextWindow = - cs && cs.model && Number.isSafeInteger(cs.model.contextWindow) && cs.model.contextWindow > 0 - ? cs.model.contextWindow - : null; - // Fallback: the current model's catalogue `contextLimit` (the - // engine's current effective window), so the picker can highlight - // the active radio before the user's first in-webui pick. - const currentModelEntry = current ? list.find((m) => m.id === current) : null; - const currentContextWindow = - recordedContextWindow ?? - (currentModelEntry && - Number.isSafeInteger(currentModelEntry.contextLimit) && - currentModelEntry.contextLimit > 0 - ? currentModelEntry.contextLimit - : null); - - const source = - option && Array.isArray(option.options) && option.options.length > 0 - ? "acp-session-config" - : config - ? "config+mcode-cli-bundle" - : "mcode-cli-bundle"; - + const { payload } = readEngineModelCatalogue({ cs: ctx && ctx.cs }); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - return res.end( - JSON.stringify({ - ok: true, - models: list, - groups, - current, - currentThinking, - currentContextWindow, - source, - // Backwards-compat: surface the same soft-failure marker the older - // engine-only build did when nothing could be sourced. With the - // merge it should be rare (builtin catalogue + providers config - // cover most installs), but a missing mcode bundle AND an absent - // config leaves the catalogue empty — and a caller that wants to - // know "is this a hard failure or just no engine attached?" still - // gets the same hint. - ...(list.length === 0 ? { reason: "no_catalogue" } : {}), - }), - ); + return res.end(JSON.stringify(payload)); } // POST /api/set-model — only updates cs.model; with a session the same value diff --git a/packages/webui/server/routes/protocol.js b/packages/webui/server/routes/protocol.js index 763a1036..4368a404 100644 --- a/packages/webui/server/routes/protocol.js +++ b/packages/webui/server/routes/protocol.js @@ -20,12 +20,19 @@ import { activateSession, mcodePermissionToWebui, } from "../lib/mcode-rpc.js"; -// M3-B1 (engine facade): only #72 (`list-sessions`) is gated in this +// M3-B1 (engine facade): only #72 (`list-sessions`) is gated in that // batch. The other five handlers here still call mcode-rpc directly — -// they belong to B4 (#73 capabilities) and B7/B9 (cancel, load, activate, -// set-mode, set-config-option), each of which lands its own facade call -// with its own regression evidence. +// they belong to B7/B9 (cancel, load, activate, set-mode, +// set-config-option), each of which lands its own facade call with its +// own regression evidence. import { readEngineSessionList } from "../engine/session-reads.js"; +// M3-B4 (engine facade): #73 (`capabilities`) now reads the engine's +// declared capability surface through the facade instead of reaching +// into `lib/mcode-rpc.js` and `lib/acp-client.js` from inside the +// handler. See `engine/capability-reads.js` for why the response gains +// the `engine` view rather than replacing the ACP wire table, and why +// this endpoint declares no capability of its own. +import { readEngineCapabilityView } from "../engine/capability-reads.js"; import { loadSessions, saveSessions, resetContext } from "../lib/sessions.js"; import { pushStateFor } from "../lib/state-bus.js"; import { readJson } from "../lib/read-json.js"; @@ -261,19 +268,35 @@ export async function handleListSessions(req, res, ctx) { // 列出 mcode acp 实际支持的能力 — 供前端 capability detection, // 决定按钮是否 disable / 降级路径 // mcode version 动态从 acp client initialize 响应读 (不再 hardcode) +// +// M3-B4: the handler no longer names `lib/mcode-rpc.js` or +// `lib/acp-client.js` — both moved behind +// `engine/capability-reads.js#readEngineCapabilityView`, which also +// resolves the provider whose DECLARED 14-key surface and its +// degradation summary this endpoint now carries under `engine`. +// +// `capabilities` itself is unchanged: it is still `MCODE_ACP_CAPABILITIES`, +// the ACP JSON-RPC method table the frontend's control map is keyed on. +// The 14 matrix keys answer a different question ("does the engine have +// this capability at all"), so the view is additive rather than a +// replacement — `docs/API.md` documents both, in both languages. +// `providerFor` says whether the declaration came from the active +// transport's provider or from the default provider standing in for a +// transport no provider claims yet (M4), so a consumer never mistakes a +// standing-in declaration for the connected engine's. +// +// `notes` stays here: it is prose about webui's own routes, not an +// engine read, and the facade has no business restating it. // ============================================================ export async function handleCapabilities(_req, res) { - const { MCODE_ACP_CAPABILITIES } = await import("../lib/mcode-rpc.js"); - const { getMcodeServerInfo } = await import("../lib/acp-client.js"); - // initialize answers with `agentInfo: { name, title, version }` (not `serverInfo`). - const agentInfo = getMcodeServerInfo(); - const mcodeVersion = (agentInfo && agentInfo.version) || "unknown"; + const { engine, agent, wire } = await readEngineCapabilityView(); return respond(res, 200, { ok: true, - mcodeVersion, - mcodeName: (agentInfo && agentInfo.name) || null, - mcodeTitle: (agentInfo && agentInfo.title) || null, - capabilities: MCODE_ACP_CAPABILITIES, + mcodeVersion: agent.version, + mcodeName: agent.name, + mcodeTitle: agent.title, + capabilities: wire, + engine, notes: { set_mode: "Takes a modeId from the session's availableModes.", set_config_option: diff --git a/packages/webui/test/lib/engine/account-reads.test.js b/packages/webui/test/lib/engine/account-reads.test.js new file mode 100644 index 00000000..94648f0b --- /dev/null +++ b/packages/webui/test/lib/engine/account-reads.test.js @@ -0,0 +1,449 @@ +// webui/test/lib/engine/account-reads.test.js +// +// M3-B4: the account read's engine facade (#20). +// +// What this file pins, and why the family needs pinning at all when +// the endpoint is five lines long: +// +// 1. THE DECLARATION. #20 and B3's #15 / #16 read the SAME engine +// projection through the SAME `mcode/account/status` method, so +// they must be gated by the SAME `authCredentials.getAccountStatus` +// pair. If the two ever drift, a provider that drops the method +// takes one endpoint down and leaves the other claiming a quota it +// cannot read — section 1 asserts the pair against the usage +// family's own table, not against a copy of it. +// +// 2. THE SOFT-FAILURE BODY. `{ok:false, reason}` at HTTP 200 is the +// account card's documented empty state, and it is produced by the +// ENGINE failing, not by the request failing. A refactor that +// converts it into a thrown error or a 500 turns a card that +// renders 本地用户 into a broken menu. +// +// 3. THE SUCCESS BODY'S SPREAD. `{ok:true, ...r.data}` means the +// engine frames its own projection; a layer that started picking +// fields (`payload.identity`, `payload.tokenPlan`) would silently +// drop every field the engine adds next year, and no test that +// only checks today's fields would notice. +// +// 4. THE GATE IS REAL, AND THE MOCK IS REAL. The registered provider +// declares `authCredentials` `full`, so only this file can prove +// the gate would bite. And node:test's `mock.module` re-evaluates +// only the MOCKED specifier, so a route module already in the +// registry keeps its old live binding — every route test here +// re-imports the route under a fresh `?bust=N`, and section 5 ends +// with the control that proves the mock took: with no mock at all, +// the same request answers from the real rpc layer. +// +// Test style follows test/lib/engine/usage-reads.test.js (B3) and +// test/lib/engine/session-tree-reads.test.js (B2): table-driven, one row +// per case. + +import { test, describe, after } from "node:test"; +import assert from "node:assert/strict"; + +import { setupMocks, absPath, registerRpcMock } from "../../helpers/_setup.js"; + +const { ENGINE_CAPABILITY_KEYS } = await import("../../../server/engine/index.js"); +const { + ACCOUNT_READ_ENDPOINTS, + assertAccountReadCapability, + readEngineAccount, + resolveAccountReadProvider, +} = await import("../../../server/engine/account-reads.js"); +const { EngineCapabilityNotSupportedError, isEngineCapabilityNotSupportedError } = await import( + "../../../server/engine/errors.js" +); +const { USAGE_READ_ENDPOINTS } = await import("../../../server/engine/usage-reads.js"); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +after(() => { + registerRpcMock({ getAccountStatus: async () => ({ ok: false, code: "no_client" }) }); +}); + +// --------------------------------------------------------------------------- +// 1. The endpoint → capability declaration table +// --------------------------------------------------------------------------- + +describe("ACCOUNT_READ_ENDPOINTS — this batch's declaration table", () => { + test("covers exactly the one endpoint of the account family", () => { + assert.deepEqual(Object.keys(ACCOUNT_READ_ENDPOINTS), ["GET /api/account"]); + }); + + test("GET /api/account declares authCredentials.getAccountStatus", () => { + // Table-driven: editing this row is a capability decision and must be + // reviewed as one, so the table IS the assertion. + const row = { capability: "authCredentials", subItem: "getAccountStatus" }; + assert.deepEqual(ACCOUNT_READ_ENDPOINTS["GET /api/account"], row); + assert.ok(ENGINE_CAPABILITY_KEYS.includes(row.capability)); + }); + + test("it is the SAME pair B3's usage endpoints declare, because it is the same engine call", () => { + // The whole point of section 1. #20, #15 and #16 all read the + // engine's account projection through `mcode/account/status`; a + // `partial` provider that drops `getAccountStatus` must be refused + // by all three, in the same way, naming the same method. + for (const endpoint of ["POST /api/usage", "POST /api/usage-trigger"]) { + assert.deepEqual( + ACCOUNT_READ_ENDPOINTS["GET /api/account"], + USAGE_READ_ENDPOINTS[endpoint], + `${endpoint} drifted from the account family`, + ); + } + }); + + test("an endpoint outside this family is caller confusion, not an engine limitation", () => { + assert.throws( + () => assertAccountReadCapability("GET /api/nope", RUNTIME), + (err) => { + assert.ok(!(err instanceof EngineCapabilityNotSupportedError)); + assert.equal(err.code, "unknown_account_read_endpoint"); + assert.match(err.message, /not part of the account family/); + return true; + }, + ); + }); +}); + +// --------------------------------------------------------------------------- +// 2. Provider resolution + the gate +// --------------------------------------------------------------------------- + +describe("resolveAccountReadProvider / assertAccountReadCapability", () => { + // Table-driven. Absent means "no provider claims this transport yet" + // (M4), which is NOT the same answer as "capability unavailable" — + // the default `acp` transport must keep answering, so it must NOT + // throw. + const TRANSPORTS = [ + [RUNTIME, true, "checked", "local-runtime-v2"], + [ACP, false, "unregistered-transport", null], + ["exec", false, "unregistered-transport", null], + ["", false, "unregistered-transport", null], + ]; + for (const [transport, hasProvider, gate, providerId] of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → ${gate}`, () => { + const provider = resolveAccountReadProvider(transport); + assert.equal(!!provider, hasProvider); + const g = assertAccountReadCapability("GET /api/account", transport); + assert.equal(g.gate, gate); + assert.equal(g.provider, providerId); + assert.equal(g.capability, "authCredentials"); + assert.equal(g.subItem, "getAccountStatus"); + assert.equal(g.endpoint, "GET /api/account"); + }); + } + + test("the descriptor has exactly the six fields every family's descriptor has", () => { + // A consumer that reads `gate.provider` under `acp` must get `null`, + // not `undefined` — the key must EXIST. Same key set as B1/B2/B3. + assert.deepEqual(Object.keys(assertAccountReadCapability("GET /api/account", RUNTIME)), [ + "endpoint", + "gate", + "provider", + "capability", + "subItem", + ]); + }); +}); + +// --------------------------------------------------------------------------- +// 3. The payload — the soft-failure body and the verbatim spread +// --------------------------------------------------------------------------- + +describe("readEngineAccount — the payload is the endpoint's, in both shapes", () => { + // Every case in this table is a REAL engine answer shape the endpoint + // has to render. The row is [engine result, expected payload, why]. + const TABLE = [ + [ + { ok: true, data: { identity: { name: "Ada" }, tokenPlan: { tier: "pro" } } }, + { ok: true, identity: { name: "Ada" }, tokenPlan: { tier: "pro" } }, + "the projection is spread verbatim", + ], + [ + { ok: true, data: { identity: { name: "Ada" }, futureEngineField: 7 } }, + { ok: true, identity: { name: "Ada" }, futureEngineField: 7 }, + "a field webui has never heard of still reaches the card", + ], + [ + { ok: true, data: null }, + { ok: true }, + "`data:null` must not throw on the spread", + ], + [ + { ok: true, data: undefined }, + { ok: true }, + "an absent `data` behaves the same as a null one", + ], + [ + { ok: true }, + { ok: true }, + "no `data` key at all", + ], + [ + { ok: false, code: "no_client" }, + { ok: false, reason: "no_client" }, + "the engine's own machine-readable code becomes the reason", + ], + [ + { ok: false, code: "unauthorized" }, + { ok: false, reason: "unauthorized" }, + "any code, verbatim", + ], + [ + { ok: false, error: "boom" }, + { ok: false, reason: "account_unavailable" }, + "no code → the endpoint's own historical fallback string", + ], + [ + { ok: false, code: "" }, + { ok: false, reason: "account_unavailable" }, + "an empty code is falsy and falls back, exactly as `||` did", + ], + [ + null, + { ok: false, reason: "account_unavailable" }, + "a null result must not throw — it is a failure, not a crash", + ], + ]; + for (const [result, expected, why] of TABLE) { + test(`${why}: ${JSON.stringify(result)} → ${JSON.stringify(expected)}`, async (t) => { + await setupMocks(t, { acp: {} }); + // `setupMocks` already registered the `lib/mcode-rpc.js` mock and + // node:test refuses a second registration for the same specifier + // (ERR_INVALID_STATE), so the payload is injected through the + // helper's mutable dispatch-through holder — the mechanism + // `registerRpcMock` exists for. + registerRpcMock({ getAccountStatus: async () => result }); + const read = await readEngineAccount({ cs: { mcodeSessionId: "mvs_1" }, transport: RUNTIME }); + assert.deepEqual(read.payload, expected); + assert.equal(read.source, "account-status"); + assert.equal(read.gate.gate, "checked"); + }); + } + + test("the failure payload has EXACTLY two keys, in order", async (t) => { + // A key-set assertion, not a subset: a facade that helpfully added + // `provider` or `gate` to the failure body would be a frontend + // contract change, and `ok:false` bodies are what the card branches + // on. + await setupMocks(t, { acp: {} }); + registerRpcMock({ getAccountStatus: async () => ({ ok: false, code: "no_client" }) }); + const read = await readEngineAccount({ cs: {}, transport: RUNTIME }); + assert.deepEqual(Object.keys(read.payload), ["ok", "reason"]); + }); + + test("cs.mcodeSessionId is forwarded EXACTLY as the route computed it", async (t) => { + // Table-driven: [ctx-ish cs, expected forwarded argument]. The route + // used to evaluate `ctx && ctx.cs && ctx.cs.mcodeSessionId`, so a + // missing ctx forwarded `undefined` and a cs without a session id + // forwarded `undefined` too — but a cs whose id is `""` forwarded + // `""`. `getAccountStatus` turns any falsy value into `{}`, so the + // difference is invisible on the wire and very visible to a test + // that pins the call. + await setupMocks(t, { acp: {} }); + const seen = []; + registerRpcMock({ + getAccountStatus: async (sessionId) => { + seen.push(sessionId); + return { ok: true, data: {} }; + }, + }); + const CASES = [ + [{ mcodeSessionId: "mvs_1" }, "mvs_1"], + [{ mcodeSessionId: "" }, ""], + [{}, undefined], + [{ mcodeSessionId: null }, null], + [{ mcodeSessionId: 0 }, 0], + ]; + for (const [cs] of CASES) { + await readEngineAccount({ cs, transport: RUNTIME }); + } + // A missing ctx entirely: the facade must not throw on `undefined`. + await readEngineAccount({ transport: RUNTIME }); + seen.push(""); + assert.deepEqual(seen, ["mvs_1", "", undefined, null, 0, undefined, ""]); + }); + + test("the gate runs BEFORE the engine call", async (t) => { + // Order matters: a provider that does not offer `getAccountStatus` + // must cost zero engine calls, so the 501 does not depend on the + // engine answering anything at all. + await setupMocks(t, { acp: {} }); + let called = 0; + registerRpcMock({ + getAccountStatus: async () => { + called += 1; + return { ok: true, data: {} }; + }, + }); + await assert.rejects( + () => readEngineAccount({ cs: {}, endpoint: "GET /api/nope", transport: RUNTIME }), + (err) => { + assert.ok(!isEngineCapabilityNotSupportedError(err)); + assert.equal(err.code, "unknown_account_read_endpoint"); + return true; + }, + ); + assert.equal(called, 0); + }); +}); + +// --------------------------------------------------------------------------- +// 4. The route +// --------------------------------------------------------------------------- + +describe("handleGetAccount — the route asks the facade", () => { + // One fresh route module per test: node:test's `mock.module` + // re-evaluates only the MOCKED specifier, but a route module already + // in the registry keeps its old LIVE BINDING to the facade — without + // the `?bust=N` re-import the second test here would silently + // exercise the first test's mock and pass for the wrong reason. + let bust = 0; + const loadRoute = async () => import(`${absPath("routes/account.js")}?bust=${bust++}`); + + // `mock.module` REPLACES the whole namespace, so a partial mock makes + // the route fail to instantiate on the exports it did not stub + // ("does not provide an export named …"). `readEngineAccount` is the + // route's only facade import, but the helper is kept so the next + // family to copy this file has the shape ready. + const NOT_STUBBED = (name) => async () => { + throw new Error(`B4 test called ${name}, which this case did not stub`); + }; + function mockFacade(t, overrides) { + t.mock.module(absPath("engine/account-reads.js"), { + namedExports: { readEngineAccount: NOT_STUBBED("readEngineAccount"), ...overrides }, + }); + } + + function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; + } + + test("both bodies are written byte-for-byte at HTTP 200", async (t) => { + // Two cases, one mock registration: node:test refuses to mock the + // same specifier twice inside one test, and a mutable holder is the + // honest way to say "the same route, two payloads". + const CASES = [ + { ok: true, identity: { name: "Ada" }, tokenPlan: { tier: "pro" } }, + { ok: false, reason: "no_client" }, + ]; + let current = CASES[0]; + mockFacade(t, { + readEngineAccount: async () => ({ + payload: current, + source: "account-status", + gate: {}, + transport: RUNTIME, + }), + }); + for (const payload of CASES) { + current = payload; + const route = await loadRoute(); + const res = mkRes(); + await route.handleGetAccount(null, res, { cs: { mcodeSessionId: "mvs_1" } }); + assert.equal(res.written[0].status, 200); + assert.equal(res.written[0].headers["Content-Type"], "application/json; charset=utf-8"); + assert.equal(res.written[1].body, JSON.stringify(payload)); + } + }); + + test("the route hands its ctx straight through and does not read cs itself", async (t) => { + await setupMocks(t, { acp: {} }); + const seen = []; + mockFacade(t, { + readEngineAccount: async (o) => { + seen.push(o); + return { payload: { ok: true }, source: "account-status", gate: {}, transport: RUNTIME }; + }, + }); + const route = await loadRoute(); + // A missing ctx is a real call shape (`invokeHandler` always sets + // one, but the route's signature must not assume it) and must not + // throw — `ctx && ctx.cs` is what the pre-facade route evaluated. + for (const ctx of [{ cs: { mcodeSessionId: "mvs_1" } }, { cs: null }, undefined, {}]) { + await route.handleGetAccount(null, mkRes(), ctx); + } + assert.equal(seen.length, 4); + assert.deepEqual(seen[0].cs, { mcodeSessionId: "mvs_1" }); + assert.equal(seen[1].cs, null); + assert.equal(seen[2].cs, undefined); + assert.equal(seen[3].cs, undefined); + // The route must not pass an endpoint key of its own: the facade's + // default IS the endpoint, and a route that spelled it out would be + // a second place to get it wrong. + for (const o of seen) assert.equal(o.endpoint, undefined); + }); + + test("a capability error PROPAGATES so invokeHandler can answer 501", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineAccount: async () => { + throw new EngineCapabilityNotSupportedError({ + capability: "authCredentials", + provider: "fixture-provider", + missing: ["getAccountStatus"], + reason: "test fixture", + }); + }, + }); + const route = await loadRoute(); + await assert.rejects( + () => route.handleGetAccount(null, mkRes(), { cs: {} }), + isEngineCapabilityNotSupportedError, + ); + }); + + // ---- proof the mock actually took ------------------------------------ + + test("PROOF the facade mock took: a marker error escapes the untouched route", async (t) => { + // Without a fresh `?bust=` re-import, `mock.module` would leave the + // route holding the PREVIOUS test's live binding, the marker would + // never be thrown, and this assertion would fail — which is the + // point: it is the only assertion here that cannot pass by accident. + await setupMocks(t, { acp: {} }); + const marker = new Error("B4-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + readEngineAccount: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleGetAccount(null, mkRes(), { cs: {} }); + } catch (err) { + caught = err; + } + assert.ok(caught, "the route swallowed the facade error — either the mock did not take, or the route grew a catch"); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + + test("CONTROL: with no facade mock, the same request reaches the rpc layer", async (t) => { + // The other half of the proof. A `?bust=` re-import under a fresh + // test hook gives a route bound to the REAL facade, so the request + // answers from the rpc layer. The holder is process-global and the + // previous cases left payloads in it, so this one puts back the + // clean-disk default — `no_client`, the answer the account card + // renders its empty state from in production when no engine has + // attached. + await setupMocks(t, { acp: {} }); + registerRpcMock({ getAccountStatus: async () => ({ ok: false, code: "no_client" }) }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleGetAccount(null, res, { cs: { mcodeSessionId: "mvs_1" } }); + assert.equal(res.written[0].status, 200); + assert.deepEqual(JSON.parse(res.written[1].body), { ok: false, reason: "no_client" }); + }); +}); diff --git a/packages/webui/test/lib/engine/capability-reads.test.js b/packages/webui/test/lib/engine/capability-reads.test.js new file mode 100644 index 00000000..9aa3b9b4 --- /dev/null +++ b/packages/webui/test/lib/engine/capability-reads.test.js @@ -0,0 +1,477 @@ +// webui/test/lib/engine/capability-reads.test.js +// +// M3-B4: the capability-declaration read's engine facade (#73). +// +// This is the one endpoint in the migration that CHANGES its response, +// so the tests here are mostly about pinning exactly how much changed +// and why the rest did not: +// +// 1. THE ADDITIVE CHANGE. #73 gains one key, `engine`, carrying the +// engine-capabilities view. Every key that existed before keeps +// its exact name, position and value — the ACP wire table stays +// under `capabilities`, the `initialize` mirror stays under +// `mcodeVersion` / `mcodeName` / `mcodeTitle`, and `notes` stays +// last. Section 4 asserts the full key order of the response, so a +// future "let me just replace the wire table with the 14 keys" +// cannot land without a reviewer seeing the test fail. +// +// 2. `providerFor`. The view must say whether the declaration came +// from the ACTIVE transport's provider or from the default +// provider standing in for a transport nothing claims yet (M4). +// A capability-detection endpoint that reported a standing-in +// declaration as though it were the connected engine's is the +// same lie B1 declined for `/api/health` — and this is the one +// endpoint where it is most tempting, because the fallback is +// silent and always succeeds. +// +// 3. THE EMPTY-DECLARATION RULE. #73 must never answer an empty +// view. A frontend that gets `{capabilities:{}}` cannot tell "no +// engine" from "this build has no declarations", and the whole +// point of the endpoint is that distinction. +// +// 4. THE GATE IS A NO-OP, AND SAYS SO. #73 is the declaration +// endpoint; gating the gate would let a `none` hide the +// declaration that says so. `checkCapabilityReadCapability` must +// report `no-capability-key` under EVERY transport, including a +// provider that declares nothing at all. +// +// Test style follows test/lib/engine/usage-reads.test.js (B3) and +// test/lib/engine/account-reads.test.js (B4 #20). + +import { test, describe, after } from "node:test"; +import assert from "node:assert/strict"; + +import { setupMocks, absPath, registerAcpMock, registerRpcMock } from "../../helpers/_setup.js"; + +const { + CAPABILITY_READ_ENDPOINTS, + checkCapabilityReadCapability, + readEngineCapabilityView, + resolveCapabilityReadProvider, +} = await import("../../../server/engine/capability-reads.js"); +const { ENGINE_CAPABILITY_KEYS, LOCAL_RUNTIME_V2_CAPABILITIES } = await import( + "../../../server/engine/index.js" +); +const { summarizeUnavailableCapabilities } = await import("../../../server/engine/capabilities.js"); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +const AGENT_INFO = { name: "mcode", title: "Mcode", version: "0.5.5" }; +const WIRE = { set_mode: true, set_config_option: true, cancel: true, activate: true }; + +after(() => { + registerAcpMock({ getMcodeServerInfo: () => null }); + registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); +}); + +// --------------------------------------------------------------------------- +// 1. The declaration table — the no-op, pinned +// --------------------------------------------------------------------------- + +describe("CAPABILITY_READ_ENDPOINTS — the gate is a reported no-op", () => { + test("covers exactly the one endpoint of the capability family", () => { + assert.deepEqual(Object.keys(CAPABILITY_READ_ENDPOINTS), [ + "GET /api/protocol/capabilities", + ]); + }); + + test("#73 declares NO capability — it IS the declaration endpoint", () => { + // Gating the gate is circular: a `none` anywhere in the declaration + // could hide the declaration that says so. The value is `null` for + // the same reason B1's `/api/health` and B3's `/api/usage/forecast` + // are. + assert.equal(CAPABILITY_READ_ENDPOINTS["GET /api/protocol/capabilities"], null); + }); + + // Table-driven over EVERY transport, not just the two that matter: the + // assertion is that the no-op is unconditional. + const TRANSPORTS = [RUNTIME, ACP, "exec", "", "nonsense"]; + for (const transport of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → no-capability-key`, () => { + const g = checkCapabilityReadCapability("GET /api/protocol/capabilities", transport); + assert.equal(g.gate, "no-capability-key"); + assert.equal(g.capability, null); + assert.equal(g.subItem, null); + assert.equal(g.enforcement, "soft"); + // The provider is still NAMED even though nothing is checked — + // "no capability key" must not degrade into "no provider". + assert.equal(g.provider, "local-runtime-v2"); + }); + } + + test("an endpoint outside this family is caller confusion", () => { + assert.throws( + () => checkCapabilityReadCapability("GET /api/nope", RUNTIME), + (err) => { + assert.equal(err.code, "unknown_capability_read_endpoint"); + assert.match(err.message, /not part of the capability family/); + return true; + }, + ); + }); +}); + +// --------------------------------------------------------------------------- +// 2. Provider resolution — always answers, and says how +// --------------------------------------------------------------------------- + +describe("resolveCapabilityReadProvider — it never returns nothing", () => { + // Table-driven. `[transport, providerFor]` — the whole family differs + // from B1/B2/B3 here: there is no `null` row, because an empty + // capability view is worse than useless for a capability-DETECTION + // endpoint. The `providerFor` field is what keeps the fallback honest. + const TRANSPORTS = [ + [RUNTIME, "transport"], + [ACP, "default"], + ["exec", "default"], + ["", "default"], + ["nonsense", "default"], + ]; + for (const [transport, providerFor] of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → providerFor=${providerFor}`, () => { + const { provider, providerFor: actual } = resolveCapabilityReadProvider(transport); + assert.equal(provider.id, "local-runtime-v2"); + assert.equal(provider.transport, "runtime"); + assert.equal(actual, providerFor); + // The declaration served is the real reviewed object, not a copy + // that could drift from it. + assert.equal(provider.capabilities, LOCAL_RUNTIME_V2_CAPABILITIES); + }); + } +}); + +// --------------------------------------------------------------------------- +// 3. The view +// --------------------------------------------------------------------------- + +describe("readEngineCapabilityView", () => { + test("the view is the engine-capabilities payload /api/engine-capabilities serves", async (t) => { + // Same four facts, same source objects. If the two endpoints ever + // answer different declarations there are two truths in webui, and + // this assertion is what stops that. + await setupMocks(t, { acp: { getMcodeServerInfo: () => AGENT_INFO } }); + registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); + const read = await readEngineCapabilityView({ transport: RUNTIME }); + assert.deepEqual(Object.keys(read.engine), [ + "provider", + "providerFor", + "transport", + "capabilities", + "unavailable", + ]); + assert.deepEqual(Object.keys(read.engine.capabilities), [...ENGINE_CAPABILITY_KEYS]); + assert.equal(read.engine.capabilities, LOCAL_RUNTIME_V2_CAPABILITIES); + assert.deepEqual( + read.engine.unavailable, + summarizeUnavailableCapabilities(LOCAL_RUNTIME_V2_CAPABILITIES), + ); + assert.equal(read.source, "declaration"); + assert.equal(read.transport, RUNTIME); + }); + + // Table-driven. The `initialize` mirror is empty until something + // attaches, and the endpoint's own fallbacks must survive that — #75 + // answers the same figure with the same fallback, and two endpoints + // answering it differently would be the defect. + const AGENT_CASES = [ + [{ name: "mcode", title: "Mcode", version: "0.5.5" }, { version: "0.5.5", name: "mcode", title: "Mcode" }], + [{ version: "0.5.5" }, { version: "0.5.5", name: null, title: null }], + [{ name: "mcode" }, { version: "unknown", name: "mcode", title: null }], + [null, { version: "unknown", name: null, title: null }], + ]; + for (const [info, expected] of AGENT_CASES) { + test(`agentInfo ${JSON.stringify(info)} → ${JSON.stringify(expected)}`, async (t) => { + await setupMocks(t, { acp: { getMcodeServerInfo: () => info } }); + registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); + const read = await readEngineCapabilityView({ transport: RUNTIME }); + assert.deepEqual(read.agent, expected); + assert.deepEqual(Object.keys(read.agent), ["version", "name", "title"]); + }); + } + + // Table-driven. The VIEW's `providerFor` — not just the resolver's — + // is what a consumer branches on, so a facade that resolved the + // provider honestly and then hard-coded the label in the payload would + // defeat the whole point. This table is the assertion that separates + // those two. + const PROVIDER_FOR = [ + [RUNTIME, "transport"], + [ACP, "default"], + ["exec", "default"], + ["nonsense", "default"], + ]; + for (const [transport, expected] of PROVIDER_FOR) { + test(`the view reports providerFor=${expected} on transport ${JSON.stringify(transport)}`, async (t) => { + await setupMocks(t, { acp: {} }); + registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); + const read = await readEngineCapabilityView({ transport }); + assert.equal(read.engine.providerFor, expected); + // And the two halves cannot disagree: `providerFor: "transport"` + // with a provider the transport does not own is the lie. + assert.equal(read.engine.providerFor === "transport", transport === RUNTIME); + }); + } + + test("an empty transport override means 'the ambient one', and the view says so", async (t) => { + // `options.transport || config.MCODE_WEBUI_TRANSPORT` treats `""` as + // "not specified" — the same idiom every other read family uses. It + // is also why the table above has no `""` row: the answer would + // depend on the gate's own `MCODE_WEBUI_TRANSPORT`, and a test whose + // expected value depends on the ambient env is a test that is green + // on one transport and red on the other. + await setupMocks(t, { acp: {} }); + registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); + const { MCODE_WEBUI_TRANSPORT } = await import(absPath("lib/config.js")); + const read = await readEngineCapabilityView({ transport: "" }); + assert.equal(read.transport, MCODE_WEBUI_TRANSPORT); + assert.equal( + read.engine.providerFor, + MCODE_WEBUI_TRANSPORT === RUNTIME ? "transport" : "default", + ); + }); + + test("the ACP wire table is forwarded by REFERENCE, not copied", async (t) => { + // A copy would be a second answer to "which ACP methods exist", + // freezable in a way the source is not. Identity pins the + // forwarding. + await setupMocks(t, { acp: {} }); + registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); + const read = await readEngineCapabilityView({ transport: RUNTIME }); + assert.equal(read.wire, WIRE); + }); + + test("the gate is evaluated and reported, and never blocks the read", async (t) => { + await setupMocks(t, { acp: {} }); + registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); + // Every transport, including one no provider claims. A read that + // gated would throw here; a read that skipped the check entirely + // would have no `gate` field at all. + for (const transport of [RUNTIME, ACP, "exec"]) { + const read = await readEngineCapabilityView({ transport }); + assert.equal(read.gate.gate, "no-capability-key"); + assert.equal(read.gate.endpoint, "GET /api/protocol/capabilities"); + } + }); + + test("an unknown endpoint key is a plain Error, not 501 material", async (t) => { + await setupMocks(t, { acp: {} }); + registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); + await assert.rejects( + () => readEngineCapabilityView({ endpoint: "GET /api/nope", transport: RUNTIME }), + (err) => { + assert.equal(err.code, "unknown_capability_read_endpoint"); + return true; + }, + ); + }); +}); + +// --------------------------------------------------------------------------- +// 4. The route — the additive change, pinned key by key +// --------------------------------------------------------------------------- + +describe("handleCapabilities — one key added, nothing else touched", () => { + let bust = 0; + const loadRoute = async () => import(`${absPath("routes/protocol.js")}?bust=${bust++}`); + + // `mock.module` REPLACES the whole namespace; the route binds one + // facade import from this family, but the module it mocks is imported + // by six other handlers in the same file, so the mock must answer for + // everything the route module evaluates at load time. + const NOT_STUBBED = (name) => async () => { + throw new Error(`B4 test called ${name}, which this case did not stub`); + }; + function mockFacade(t, overrides) { + t.mock.module(absPath("engine/capability-reads.js"), { + namedExports: { readEngineCapabilityView: NOT_STUBBED("readEngineCapabilityView"), ...overrides }, + }); + } + + function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; + } + + const VIEW = { + engine: { + provider: "local-runtime-v2", + providerFor: "transport", + transport: "runtime", + capabilities: { sessionCrud: { level: "full" } }, + unavailable: { none: [], partial: [] }, + }, + agent: { version: "0.5.5", name: "mcode", title: "Mcode" }, + wire: WIRE, + }; + + test("the response key order is the endpoint's, with `engine` inserted once", async (t) => { + // This is the assertion that makes "we only added a key" a fact + // rather than a claim. The order is the endpoint's, `engine` sits + // directly after the wire table it complements, and `notes` stays + // last. + await setupMocks(t, { acp: {} }); + mockFacade(t, { readEngineCapabilityView: async () => ({ ...VIEW, source: "declaration", gate: {}, transport: RUNTIME }) }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + assert.equal(res.written[0].status, 200); + const body = JSON.parse(res.written[1].body); + assert.deepEqual(Object.keys(body), [ + "ok", + "mcodeVersion", + "mcodeName", + "mcodeTitle", + "capabilities", + "engine", + "notes", + ]); + }); + + test("every pre-existing key keeps its exact value", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { readEngineCapabilityView: async () => ({ ...VIEW, source: "declaration", gate: {}, transport: RUNTIME }) }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + assert.equal(body.ok, true); + // The ACP wire table is still the ACP wire table — the 14 matrix + // keys did NOT replace it. + assert.deepEqual(body.capabilities, WIRE); + assert.equal(body.mcodeVersion, "0.5.5"); + assert.equal(body.mcodeName, "mcode"); + assert.equal(body.mcodeTitle, "Mcode"); + // `notes` is route-owned prose about webui's own routes; the facade + // never restates it, so it is still exactly these five strings. + assert.deepEqual(Object.keys(body.notes), ["set_mode", "set_config_option", "cancel", "activate", "fork"]); + }); + + test("the whole view is carried, and the route adds nothing to it", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { readEngineCapabilityView: async () => ({ ...VIEW, source: "declaration", gate: {}, transport: RUNTIME }) }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + // Identity, not equality: a route that re-projected the view would + // be a second place for the 14 keys to be reshaped. + assert.deepEqual(body.engine, VIEW.engine); + // And the facade's own bookkeeping (`source`, `gate`, `transport`) + // stays INSIDE the facade — it is diagnostic vocabulary, not part + // of this endpoint's contract. + for (const key of ["source", "gate"]) { + assert.equal(key in body, false, `${key} leaked into the response`); + } + }); + + // Table-driven: [agent version, expected mcodeVersion]. The route is a + // PASS-THROUGH — including for the empty string, which the facade has + // already turned into `"unknown"` (section 3 pins that), so a route + // that applied its own `|| "unknown"` would double-apply it and a + // route that dropped the fallback entirely would ship an empty + // version. This table is the split made visible: the fallback lives + // in the engine layer, once. + const VERSION_CASES = [ + ["0.5.5", "0.5.5"], + ["unknown", "unknown"], + ["", ""], + ]; + for (const [version, expected] of VERSION_CASES) { + test(`agent.version=${JSON.stringify(version)} → mcodeVersion=${JSON.stringify(expected)}`, async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineCapabilityView: async () => ({ + ...VIEW, + agent: { version, name: null, title: null }, + source: "declaration", + gate: {}, + transport: RUNTIME, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + assert.equal(body.mcodeVersion, expected); + assert.equal(body.mcodeName, null); + assert.equal(body.mcodeTitle, null); + }); + } + + test("a facade error PROPAGATES so invokeHandler can answer 501", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readEngineCapabilityView: async () => { + const err = new Error("fixture capability refusal"); + err.name = "EngineCapabilityNotSupportedError"; + throw err; + }, + }); + const route = await loadRoute(); + await assert.rejects(() => route.handleCapabilities(null, mkRes()), /fixture capability refusal/); + }); + + // ---- proof the mock actually took ------------------------------------ + + test("PROOF the facade mock took: a marker error escapes the untouched route", async (t) => { + await setupMocks(t, { acp: {} }); + const marker = new Error("B4-CAPABILITY-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + readEngineCapabilityView: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleCapabilities(null, mkRes()); + } catch (err) { + caught = err; + } + assert.ok(caught, "the route swallowed the facade error — either the mock did not take, or the route grew a catch"); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + + test("CONTROL: with no facade mock, the real view reaches the response", async (t) => { + // The other half of the proof: a fresh `?bust=` re-import binds the + // route to the REAL facade, so the body carries the actual + // registered declaration rather than the fixture's. + await setupMocks(t, { acp: { getMcodeServerInfo: () => AGENT_INFO } }); + registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + assert.equal(body.engine.provider, "local-runtime-v2"); + // The real view must SAY whether it is standing in. Under the + // default `acp` transport that is `"default"`; reporting + // `"transport"` there would be the one lie this endpoint cannot + // afford, because the declaration it would attribute to a connected + // engine came from a provider that transport never chose. The + // expectation follows the ambient transport so the control holds on + // both gate legs. + const { MCODE_WEBUI_TRANSPORT } = await import(absPath("lib/config.js")); + assert.equal( + body.engine.providerFor, + MCODE_WEBUI_TRANSPORT === "runtime" ? "transport" : "default", + ); + assert.equal(body.engine.transport, "runtime"); + // `setupMocks`'s acp holder is process-global and an earlier case + // left the agent mirror in it, so the version here is the real + // `initialize` mirror's, not the fixture's. + assert.equal(body.mcodeVersion, "0.5.5"); + assert.deepEqual(Object.keys(body.engine.capabilities), [...ENGINE_CAPABILITY_KEYS]); + assert.equal(body.engine.unavailable.none.length >= 1, true); + }); +}); diff --git a/packages/webui/test/lib/engine/model-reads.test.js b/packages/webui/test/lib/engine/model-reads.test.js new file mode 100644 index 00000000..0806defc --- /dev/null +++ b/packages/webui/test/lib/engine/model-reads.test.js @@ -0,0 +1,1181 @@ +// webui/test/lib/engine/model-reads.test.js +// +// M3-B4: the model-catalogue read's engine facade (#57). +// +// This is the batch's red line. #57 is the largest projection in webui +// and the one a refactor can damage most quietly: three sources, a +// dedupe key that has changed shape twice, two projections of one +// engine file annotating entries from two different sources, and three +// derived "what is active" figures — none of which is compared against +// anything at runtime. So the four things pinned here are: +// +// 1. THE FULL SNAPSHOT (section 5). One rich fixture — engine session +// option, engine `custom_provider` layer, webui config layer, +// builtin layer, a builtin that COLLIDES with a config entry, a +// switchable variant model, an effort-list model, a forced_on +// model, two providers with overlapping upstream model ids, a +// provider with a key and one without — projected to the exact +// response body the pre-refactor route produced. The expected +// value below was captured from the implementation at 3362c9be +// (B3's rebase tip) and pasted in longhand: it is NOT recomputed +// by the functions under test, because a snapshot whose oracle is +// the implementation proves nothing. The two `minimax_api` models +// that are ABSENT from the builtin half of the `minimax_api` +// group are the load-bearing part: the config layer took those +// slots wholesale, which is ticket 09-02's dedupe rule. +// +// 2. THE PURE PROJECTIONS ON THEIR INPUTS (sections 3–4). Each rule +// the snapshot exercises incidentally is also asserted on a +// minimal input of its own, so a failure names the RULE that broke +// rather than pointing at a 280-line diff. +// +// 3. THE VARIANT / CONTEXT PERTURBATION (section 6). The thinking +// levels and the context-window options are two projections of one +// engine file, and the interesting failure is a cross-wiring: an +// annotation attached to the wrong entry, or the builtin tree read +// twice so the two sites disagree. The test perturbs one engine +// model at a time and records exactly which entries move. +// +// 4. THE GATE IS SOFT, AND THE MOCK IS REAL. The registered provider +// declares `authCredentials` `full`, so only this file can prove +// the soft gate reports what it claims; and node:test's +// `mock.module` re-evaluates only the MOCKED specifier, so every +// route test re-imports the route under a fresh `?bust=N`, and +// section 7 ends with the control that proves the mock took. +// +// Fixture ordering is load-bearing, not stylistic. `lib/config.js` +// resolves `MCODE_WEBUI_DATA_DIR` / `MINIMAX_DATA_DIR` at MODULE LOAD, +// and `engine/model-reads.js` imports it statically — so the fixture +// directories and the env are built at module top level, BEFORE the +// first import that reaches a server module. A `before()` that set the +// env would be too late: the first import would already have frozen the +// real ~/.minimax path, and every case below would read the developer's +// own config instead of the fixture. +// +// Test style follows test/lib/engine/usage-reads.test.js (B3) and +// test/lib/engine/account-reads.test.js (B4 #20): table-driven, one row +// per case. + +import { test, describe, before, after } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; +import { setupMocks, absPath, setBuiltinModelsMock } from "../../helpers/_setup.js"; + +// --------------------------------------------------------------------------- +// Fixture — built BEFORE any server module is imported (see the header). +// +// One root, three children, one registered prefix: the engine's data +// dir (its `config.yaml` — both the `custom_provider` tree and the +// materialised `provider.minimax.models` builtin tree), the webui data +// dir (where the user-level `providers.json` would live), and the env +// layer file. The prefix is registered in +// scripts/test-tmp-leak.check.mjs#KNOWN_PREFIXES; a new prefix without +// that entry fails the test:release-tools gate. +// --------------------------------------------------------------------------- + +const root = mkTmpDir("webui-model-reads-"); +const engineDir = join(root, "engine"); +const webuiDir = join(root, "webui"); +mkdirSync(engineDir, { recursive: true }); +mkdirSync(webuiDir, { recursive: true }); + +// The engine's own config: a materialised builtin tree (one switchable +// variant model, one effort-list model, one forced_on model with +// nothing user-settable) and a `custom_provider` tree with two +// providers whose model ids OVERLAP (`z-ai/glm-5.3` is deliberately the +// kind of id that used to make one provider swallow another's entry). +writeFileSync( + join(engineDir, "config.yaml"), + `provider: + minimax: + models: + MiniMax-M3: + thinking_config: + mode: switchable + default_value: 'true' + variants: + none-thinking: { thinking: { type: disabled } } + thinking: { thinking: { type: adaptive } } + contextWindowOptions: [512000, 1000000] + contextWindowOptionHints: { "1000000": "higher_usage" } + limit: { context: 512000 } + MiniMax-M2.7: + thinking: + effortOptions: [low, medium, high] + contextWindowOptions: [128000, 256000] + limit: { context: 128000 } + MiniMax-M2.5: + thinking_config: + mode: forced_on +custom_provider: + deepseek-cn: + api: openai-completions + kind: custom + options: { apiKey: "sk-secret-should-never-leak" } + models: + deepseek-chat: + name: DeepSeek Chat + thinking: { effortOptions: [low, high] } + modalities: { input: [text, image] } + limit: { context: 64000 } + deepseek-reasoner: {} + nousresearch: + api: openai-responses + kind: custom + options: { apiKey: "" } + models: + z-ai/glm-5.3: {} + openai/gpt-5.6-sol: {} +`, + "utf8", +); + +// The webui's env layer. `minimax_api` is present ON PURPOSE: its +// `MiniMax-M3` entry collides with the builtin of the same name, so the +// `seen` dedupe has to let the config layer win wholesale — which is +// why the builtin half of that group is missing the entry, the +// `thinkingLevels`, and the `contextWindowOptions`. +const modelsConfigPath = join(root, "models.json"); +writeFileSync( + modelsConfigPath, + JSON.stringify({ + providers: [ + { + id: "minimax_api", + label: "MiniMax builtins", + auth: { type: "byok", apiKey: "sk-webui-fixture-key-0001" }, + protocol: "anthropic", + models: [ + { id: "MiniMax-M3", label: "M3 config override", contextLimit: 123456 }, + { id: "MiniMax-Text-01", thinkingLevels: ["off", "on"], modalities: ["text", "image"] }, + ], + }, + { id: "local-ollama", models: [{ id: "qwen3:8b" }] }, + ], + }), + "utf8", +); + +process.env.MINIMAX_DATA_DIR = engineDir; +delete process.env.MAVIS_DATA_DIR; +process.env.MCODE_WEBUI_DATA_DIR = webuiDir; +process.env.MCODE_WEBUI_MODELS_CONFIG = modelsConfigPath; + +/** The builtin list the mocked `getBuiltinModelsFromMcode` answers with. */ +const BUILTINS = ["MiniMax-M3", "MiniMax-M2.7", "MiniMax-M2.5", "MiniMax-M2.7-highspeed"]; + +/** + * A `model` config option shaped like the engine's, in the wire form + * `m::[:v:]` that `control-state.ts` emits for + * the builtin tree. The model segment is the BARE engine-side model key, + * which is what makes the two builtin-tree projections reachable from + * this source at all (see section 6). + */ +const MODEL_OPTION = { + type: "select", + id: "model", + name: "Model", + category: "model", + currentValue: "m:minimax_api:MiniMax-M3:v:thinking", + options: [ + { value: "m:minimax_api:MiniMax-M3:v:thinking", name: "MiniMax-M3" }, + { value: "m:minimax_api:MiniMax-M2.7:u", name: "MiniMax-M2.7" }, + ], +}; + +/** The `cs` the snapshot runs against. */ +const SNAPSHOT_CS = { + model: { name: "minimax_api/MiniMax-M3", thinking: "on", contextWindow: 1000000 }, + configOptions: [MODEL_OPTION, { id: "thinkingEffort", currentValue: "high" }], +}; + +/** The REAL wire-form parser, so the snapshot exercises the real parse. */ +const { parseEngineModelWireValue, readEngineBuiltinThinking, readEngineBuiltinContextWindows } = + await import("../../../server/lib/engine-catalogue.js"); +// `capabilities.js` directly, NOT `engine/index.js`: the facade re-exports +// `model-reads.js`, so importing it at module scope would evaluate the +// module under test — and its STATIC import of `lib/models.js` — BEFORE +// `before()` registers the builtin-catalogue mock, and the snapshot +// would then read whatever `mcode` bundle the host has installed. +const { ENGINE_CAPABILITY_KEYS } = await import("../../../server/engine/capabilities.js"); + +// --- now, and only now, the modules under test --------------------------- +let engine; +let modelRouteBaseline; +before(async (t) => { + // `setupMocks` must precede the SUT import: `engine/model-reads.js` + // imports `lib/models.js` STATICALLY, and the builtin catalogue must + // come from the mock rather than from whatever `mcode` bundle happens + // to be installed on the host. + await setupMocks(t, { acp: {} }); + setBuiltinModelsMock(BUILTINS); + engine = await import(absPath("engine/model-reads.js")); + modelRouteBaseline = await import(absPath("routes/model.js")); +}); + +after(() => { + rmTmpDir(root); + delete process.env.MINIMAX_DATA_DIR; + delete process.env.MAVIS_DATA_DIR; + delete process.env.MCODE_WEBUI_DATA_DIR; + delete process.env.MCODE_WEBUI_MODELS_CONFIG; +}); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +// --------------------------------------------------------------------------- +// 1. The endpoint → capability declaration table +// --------------------------------------------------------------------------- + +describe("MODEL_READ_ENDPOINTS — this batch's declaration table", () => { + test("covers exactly the one endpoint of the model family", () => { + assert.deepEqual(Object.keys(engine.MODEL_READ_ENDPOINTS), ["GET /api/models"]); + }); + + test("GET /api/models declares authCredentials.listModelProviders, enforced SOFT", () => { + // Table-driven: editing this row is a capability decision and must be + // reviewed as one, so the table IS the assertion. + const row = { + capability: "authCredentials", + subItem: "listModelProviders", + enforcement: "soft", + }; + assert.deepEqual(engine.MODEL_READ_ENDPOINTS["GET /api/models"], row); + assert.ok(ENGINE_CAPABILITY_KEYS.includes(row.capability)); + }); + + test("it shares the capability KEY with the account family, and differs in the other two fields", async () => { + // Both families ride `authCredentials` because the 14 matrix keys + // have no separate "models" row — the engine's model/provider + // surface is declared there. What differs is the sub-item and the + // enforcement, and both differences are asserted rather than + // assumed: a models read gated on `getAccountStatus` would let a + // provider that cannot report a plan still be trusted for a + // catalogue, and vice versa. + const { ACCOUNT_READ_ENDPOINTS } = await import("../../../server/engine/account-reads.js"); + assert.equal( + engine.MODEL_READ_ENDPOINTS["GET /api/models"].capability, + ACCOUNT_READ_ENDPOINTS["GET /api/account"].capability, + ); + assert.notEqual( + engine.MODEL_READ_ENDPOINTS["GET /api/models"].subItem, + ACCOUNT_READ_ENDPOINTS["GET /api/account"].subItem, + ); + assert.equal(ACCOUNT_READ_ENDPOINTS["GET /api/account"].enforcement, undefined); + }); +}); + +// --------------------------------------------------------------------------- +// 2. Provider resolution + the SOFT gate +// --------------------------------------------------------------------------- + +describe("resolveModelReadProvider / checkModelReadCapability", () => { + // Table-driven. The gate values are `session-export.js`'s vocabulary, + // reused rather than re-invented. + const TRANSPORTS = [ + [RUNTIME, true, "checked", "local-runtime-v2"], + [ACP, false, "unregistered-transport", null], + ["exec", false, "unregistered-transport", null], + ["", false, "unregistered-transport", null], + ]; + for (const [transport, hasProvider, gate, providerId] of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → ${gate}`, () => { + const provider = engine.resolveModelReadProvider(transport); + assert.equal(!!provider, hasProvider); + const g = engine.checkModelReadCapability("GET /api/models", transport); + assert.equal(g.gate, gate); + assert.equal(g.provider, providerId); + assert.equal(g.capability, "authCredentials"); + assert.equal(g.subItem, "listModelProviders"); + assert.equal(g.enforcement, "soft"); + }); + } + + test("the gate NEVER throws, under any transport or endpoint key", () => { + // The whole reason this family's gate is soft: the catalogue's + // primary sources are files webui owns. A provider that declared no + // model surface would still leave a working picker, so a hard gate + // here would REMOVE working functionality — the #11 reasoning, + // reused. + for (const transport of [RUNTIME, ACP, "exec", "", "nonsense"]) { + assert.doesNotThrow(() => engine.checkModelReadCapability("GET /api/models", transport)); + } + }); + + test("an unknown endpoint key is a plain Error, not 501 material", () => { + assert.throws( + () => engine.checkModelReadCapability("GET /api/nope", RUNTIME), + (err) => { + assert.equal(err.code, "unknown_model_read_endpoint"); + assert.match(err.message, /not part of the model family/); + return true; + }, + ); + }); +}); + +// --------------------------------------------------------------------------- +// 3. The two id helpers +// --------------------------------------------------------------------------- + +describe("providerOfModelId / webuiFullModelId", () => { + // Table-driven: [modelId, fallback, expected]. The bare-id fallback to + // `minimax_api` is what keeps a user-typed short id out of a phantom + // group; the `i <= 0` guard is what keeps a leading `/` from + // producing an empty provider key. + const PROVIDER_CASES = [ + ["minimax_api/MiniMax-M3", "minimax_api", "minimax_api"], + ["nousresearch/deepseek/x", "minimax_api", "nousresearch"], + ["/leading-slash", "minimax_api", "minimax_api"], + ["MiniMax-M3", "minimax_api", "minimax_api"], + ["", "minimax_api", "minimax_api"], + [null, "minimax_api", "minimax_api"], + [undefined, "minimax_api", "minimax_api"], + // An explicit fallback is honoured for a bare id and for an empty + // one — the engine builtin provider is a DEFAULT, not a constant. + ["minimax_api/MiniMax-M3", "fallback-provider", "minimax_api"], + ["", "fallback-provider", "fallback-provider"], + [null, "fallback-provider", "fallback-provider"], + ]; + for (const [modelId, fallback, expected] of PROVIDER_CASES) { + test(`providerOfModelId(${JSON.stringify(modelId)}, ${JSON.stringify(fallback)}) → ${expected}`, () => { + assert.equal(engine.providerOfModelId(modelId, fallback), expected); + }); + } + + // The webui id is ALWAYS two segments, even when the upstream model + // id already contains `/`. That is ticket 09-02: skipping the prefix + // put the picker in the wrong group and let overlapping upstream ids + // collide on the dedupe. + const ID_CASES = [ + ["minimax_api", "MiniMax-M3", "minimax_api/MiniMax-M3"], + ["nousresearch", "z-ai/glm-5.3", "nousresearch/z-ai/glm-5.3"], + ["minimax_api", "MiniMax-M2.7-highspeed", "minimax_api/MiniMax-M2.7-highspeed"], + ]; + for (const [providerKey, modelId, expected] of ID_CASES) { + test(`webuiFullModelId(${providerKey}, ${modelId})`, () => { + assert.equal(engine.webuiFullModelId(providerKey, modelId), expected); + }); + } +}); + +describe("attachContextWindowOptions", () => { + // Table-driven. Each row is a rule with a failure mode: a missing + // projection must leave the entry field-free (so the composer mounts + // no control), an existing `contextLimit` must NOT be overwritten (a + // config layer's value wins), and both the array and the hints object + // must be COPIED so a caller mutating the entry cannot corrupt the + // engine projection for the next entry. + const entry = () => ({ id: "x", label: "x" }); + const CASES = [ + ["no projection at all", null, {}, null], + ["options only", { options: [1, 2] }, {}, { contextWindowOptions: [1, 2] }], + [ + "options + currentLimit, entry has no limit", + { options: [1, 2], currentLimit: 9 }, + {}, + { contextWindowOptions: [1, 2], contextLimit: 9 }, + ], + [ + "options + currentLimit, entry KEEPS its own limit", + { options: [1, 2], currentLimit: 9 }, + { contextLimit: 5 }, + { contextWindowOptions: [1, 2] }, + ], + [ + "hints ride along only when present", + { options: [1, 2], hints: { 2: "higher_usage" } }, + {}, + { contextWindowOptions: [1, 2], contextWindowOptionHints: { 2: "higher_usage" } }, + ], + [ + "currentLimit of 0 is still attached (the projection decided)", + { options: [1], currentLimit: 0 }, + {}, + { contextWindowOptions: [1], contextLimit: 0 }, + ], + ]; + for (const [name, projection, pre, expected] of CASES) { + test(name, () => { + const e = { ...entry(), ...pre }; + engine.attachContextWindowOptions(e, projection); + assert.deepEqual(e, { ...entry(), ...pre, ...expected }); + }); + } + + test("the array and the hints are copies, not aliases of the projection", () => { + const projection = { options: [1, 2], hints: { 1: "higher_usage" } }; + const e = {}; + engine.attachContextWindowOptions(e, projection); + e.contextWindowOptions.push(3); + e.contextWindowOptionHints[1] = "tampered"; + assert.deepEqual(projection.options, [1, 2]); + assert.deepEqual(projection.hints, { 1: "higher_usage" }); + }); +}); + +// --------------------------------------------------------------------------- +// 4. The projections, rule by rule +// --------------------------------------------------------------------------- + +const THINKING_M3 = new Map([["MiniMax-M3", { levels: ["off", "on"] }]]); +const WINDOWS_M3 = new Map([ + ["MiniMax-M3", { options: [512000, 1000000], hints: { 1000000: "higher_usage" }, currentLimit: 512000 }], +]); + +describe("projectModelCatalogue — grouping, dedupe and the empty shell", () => { + // Table-driven: [name, options, expected]. `list` and `groups` are the + // ordered id lists, written out longhand rather than recomputed. + const wire = parseEngineModelWireValue; + const CASES = [ + [ + "no sources at all: the empty builtin shell is dropped, list is empty", + { providers: null, builtins: [], sessionOption: null }, + { list: [], groups: [] }, + ], + [ + "a providers config with no models still emits its (empty) group", + { providers: { providers: [{ id: "p", label: "P", models: [] }] }, builtins: [] }, + { + list: [], + groups: ["p", "minimax_api"], + group: { id: "p", label: "P", auth: { hasKey: false, type: "byok" }, protocol: "openai", models: [] }, + }, + ], + [ + "a providers config with no models KEEPS the empty builtin shell next to it", + { providers: { providers: [{ id: "p", models: [] }] }, builtins: ["MiniMax-M3"] }, + { list: ["minimax_api/MiniMax-M3"], groups: ["p", "minimax_api"] }, + ], + [ + "a builtin is attributed to minimax_api, and only to it", + { providers: null, builtins: ["MiniMax-M3"] }, + { list: ["minimax_api/MiniMax-M3"], groups: ["minimax_api"] }, + ], + [ + "a config entry COLLIDING with a builtin wins wholesale", + { + providers: { providers: [{ id: "minimax_api", models: [{ id: "MiniMax-M3", label: "override" }] }] }, + builtins: ["MiniMax-M3"], + }, + { list: ["minimax_api/MiniMax-M3"], groups: ["minimax_api"], labels: { "minimax_api/MiniMax-M3": "override" } }, + ], + [ + "two providers with the same upstream model id stay distinct", + { + providers: { + providers: [ + { id: "nousresearch", models: [{ id: "z-ai/glm-5.3" }] }, + { id: "zai-max", models: [{ id: "z-ai/glm-5.3" }] }, + ], + }, + builtins: [], + }, + { list: ["nousresearch/z-ai/glm-5.3", "zai-max/z-ai/glm-5.3"], groups: ["nousresearch", "zai-max", "minimax_api"] }, + ], + [ + "an engine option with an empty value list yields NO group", + { sessionOption: { options: [] }, providers: null, builtins: [] }, + { list: [], groups: [] }, + ], + [ + "entries with no usable value are skipped; a duplicate value is deduped", + { + sessionOption: { options: [{ value: "a" }, { value: null }, null, { value: "a" }] }, + providers: null, + builtins: [], + }, + { list: ["a"], groups: ["__engine"] }, + ], + ]; + for (const [name, options, expected] of CASES) { + test(name, () => { + const { list, groups } = engine.projectModelCatalogue({ + ...options, + parseEngineModelWireValue: wire, + }); + assert.deepEqual(list.map((e) => e.id), expected.list, "flat list"); + assert.deepEqual(groups.map((g) => g.id), expected.groups, "group ids"); + if (expected.group) { + assert.deepEqual(groups.find((g) => g.id === expected.group.id), expected.group); + } + if (expected.labels) { + for (const [id, label] of Object.entries(expected.labels)) { + assert.equal(list.find((e) => e.id === id).label, label); + } + } + }); + } + + test("a builtin already present from the config layer is deduped, not appended twice", () => { + // The `seen` set is per `(providerKey, modelId)`, and it is what keeps + // the picker from showing `MiniMax-M3` twice when the operator has + // configured the same builtin id. Removing the check on the builtin + // side would duplicate the row in BOTH the flat list and the group. + const { list, groups } = engine.projectModelCatalogue({ + providers: { providers: [{ id: "minimax_api", models: [{ id: "MiniMax-M3", label: "config" }] }] }, + builtins: ["MiniMax-M3", "MiniMax-M3"], + parseEngineModelWireValue: wire, + }); + assert.deepEqual(list.map((e) => e.id), ["minimax_api/MiniMax-M3"]); + assert.deepEqual(groups.find((g) => g.id === "minimax_api").models.map((e) => e.id), [ + "minimax_api/MiniMax-M3", + ]); + // And the surviving entry is the CONFIG one — the operator's layer + // wins wholesale, it does not merge with the builtin. + assert.equal(list[0].source, "config"); + assert.equal(list[0].label, "config"); + assert.equal("thinkingLevels" in list[0], false); + }); + + test("the engine group id and label are the endpoint's, not the provider's", () => { + const { groups } = engine.projectModelCatalogue({ + sessionOption: MODEL_OPTION, + providers: null, + builtins: [], + parseEngineModelWireValue: wire, + }); + assert.equal(groups.length, 1); + assert.equal(groups[0].id, "__engine"); + assert.equal(groups[0].label, "Engine session"); + // The engine group carries NO auth block — a session option is not + // a provider the operator configured. + assert.equal("auth" in groups[0], false); + assert.equal("protocol" in groups[0], false); + }); + + test("engine-session entries carry BOTH `name` and `label`, and the same value", () => { + // Pre-existing callers (the composer chip) read `name`; the + // provider-grouped panel reads `label`. Dropping either is a + // frontend break that a single-key test would miss. + const { list } = engine.projectModelCatalogue({ + sessionOption: MODEL_OPTION, + providers: null, + builtins: [], + parseEngineModelWireValue: wire, + }); + for (const e of list) { + assert.equal(e.name, e.label); + assert.equal(typeof e.name, "string"); + } + // And an option with no `name` falls back to the value itself. + const { list: l2 } = engine.projectModelCatalogue({ + sessionOption: { options: [{ value: "m:x:y:u" }] }, + providers: null, + builtins: [], + parseEngineModelWireValue: () => null, + }); + assert.equal(l2[0].name, "m:x:y:u"); + assert.equal(l2[0].label, "m:x:y:u"); + }); + + test("the config group reports hasKey from EITHER signal, and never a key", () => { + // The security contract: the group reports whether a key is + // configured, never the key. Both signals mean "configurable from + // the picker" — a webui-side plaintext apiKey and an engine-side + // boolean alike. + const { groups } = engine.projectModelCatalogue({ + providers: { + providers: [ + { id: "webui-key", auth: { apiKey: "sk-secret" }, models: [] }, + { id: "engine-key", auth: { hasKey: true }, models: [] }, + { id: "no-key", auth: { hasKey: false }, models: [] }, + { id: "no-auth", models: [] }, + ], + }, + builtins: [], + parseEngineModelWireValue: wire, + }); + // The empty `minimax_api` shell is also a group and has no `auth`, + // so the comparison is over the four CONFIG groups. + const byId = Object.fromEntries(groups.filter((g) => g.auth).map((g) => [g.id, g.auth])); + assert.deepEqual(byId, { + "webui-key": { hasKey: true, type: "byok" }, + "engine-key": { hasKey: true, type: "byok" }, + "no-key": { hasKey: false, type: "byok" }, + "no-auth": { hasKey: false, type: "byok" }, + }); + // And the secret itself is nowhere in the group. + assert.equal(JSON.stringify(groups).includes("sk-secret"), false); + }); + + test("contextLimit rides along only for a positive number", () => { + const { list } = engine.projectModelCatalogue({ + providers: { + providers: [ + { id: "p", models: [{ id: "a", contextLimit: 1000 }, { id: "b", contextLimit: 0 }, { id: "c", contextLimit: -5 }] }, + ], + }, + builtins: [], + parseEngineModelWireValue: wire, + }); + assert.equal("contextLimit" in list.find((e) => e.id === "p/a"), true); + assert.equal("contextLimit" in list.find((e) => e.id === "p/b"), false); + assert.equal("contextLimit" in list.find((e) => e.id === "p/c"), false); + }); + + test("empty thinkingLevels / modalities are omitted, not sent as []", () => { + // An empty array would make a consumer mount a control with no + // choices; the endpoint has always omitted the key. + const { list } = engine.projectModelCatalogue({ + providers: { + providers: [{ id: "p", models: [{ id: "a", thinkingLevels: [], modalities: [] }, { id: "b", thinkingLevels: ["x"], modalities: ["text"] }] }], + }, + builtins: [], + parseEngineModelWireValue: wire, + }); + assert.deepEqual(Object.keys(list[0]), ["id", "label", "provider", "source"]); + assert.deepEqual(list[1].thinkingLevels, ["x"]); + assert.deepEqual(list[1].modalities, ["text"]); + }); +}); + +describe("deriveModelSelection — the three derived figures", () => { + // Table-driven. Each row is a resolution rule, including the two + // "never invent" rules (a null `current`, a null window) that the + // composer depends on to render a neutral chip. + const entry = (id, contextLimit) => (contextLimit === undefined ? { id } : { id, contextLimit }); + const CASES = [ + [ + "the engine's currentValue wins over the recorded name", + { sessionOption: { currentValue: "wire" }, cs: { model: { name: "recorded" } } }, + { current: "wire", currentThinking: null, currentContextWindow: null }, + ], + [ + "the recorded name is the fallback, and null when there is none", + { sessionOption: null, cs: { model: { name: "recorded" } } }, + { current: "recorded", currentThinking: null, currentContextWindow: null }, + ], + [ + "no engine value and no record → null, never a default model", + { sessionOption: null, cs: {} }, + { current: null, currentThinking: null, currentContextWindow: null }, + ], + [ + "the engine's thinkingEffort wins over the recorded level", + { cs: { configOptions: [{ id: "thinkingEffort", currentValue: "high" }], model: { thinking: "low" } } }, + { currentThinking: "high" }, + ], + [ + "the recorded level is the fallback", + { cs: { model: { thinking: "low" } } }, + { currentThinking: "low" }, + ], + [ + "a non-string recorded level is ignored, not coerced", + { cs: { model: { thinking: 7 } } }, + { currentThinking: null }, + ], + [ + "a non-string engine level falls through to the record", + { cs: { configOptions: [{ id: "thinkingEffort", currentValue: 7 }], model: { thinking: "low" } } }, + { currentThinking: "low" }, + ], + [ + "the recorded window wins over the catalogue limit", + { cs: { model: { name: "m", contextWindow: 1000000 } }, list: [entry("m", 512000)] }, + { currentContextWindow: 1000000 }, + ], + [ + "the current model's catalogue limit is the fallback", + { cs: { model: { name: "m" } }, list: [entry("m", 512000)] }, + { currentContextWindow: 512000 }, + ], + [ + "a recorded window is reported even when the model no longer advertises it", + { cs: { model: { name: "m", contextWindow: 1000000 } }, list: [entry("m")] }, + { currentContextWindow: 1000000 }, + ], + [ + "a non-positive or non-integer recorded window is not a window", + { cs: { model: { name: "m", contextWindow: 0 } }, list: [entry("m", 512000)] }, + { currentContextWindow: 512000 }, + ], + [ + "a model with no limit and no record → null", + { cs: { model: { name: "m" } }, list: [entry("m")] }, + { currentContextWindow: null }, + ], + ]; + for (const [name, options, expected] of CASES) { + test(name, () => { + const got = engine.deriveModelSelection({ list: [], ...options }); + for (const [k, v] of Object.entries(expected)) assert.equal(got[k], v, k); + }); + } + + test("the result has exactly the three figures, in order", () => { + assert.deepEqual( + Object.keys(engine.deriveModelSelection({ cs: {}, list: [] })), + ["current", "currentThinking", "currentContextWindow"], + ); + }); +}); + +describe("catalogueSourceLabel", () => { + // Table-driven. The label is the endpoint's answer to "which layer + // won", and it keys off the OPTION LIST's length, not off the + // option's existence — an engine that advertises the option with no + // choices has not contributed anything. + const CASES = [ + [{ sessionOption: { options: [{ value: "a" }] }, providers: null }, "acp-session-config"], + [{ sessionOption: { options: [] }, providers: null }, "mcode-cli-bundle"], + [{ sessionOption: null, providers: { providers: [] } }, "config+mcode-cli-bundle"], + [{ sessionOption: null, providers: null }, "mcode-cli-bundle"], + [{}, "mcode-cli-bundle"], + ]; + for (const [options, expected] of CASES) { + test(`${JSON.stringify(options).slice(0, 60)} → ${expected}`, () => { + assert.equal(engine.catalogueSourceLabel(options), expected); + }); + } +}); + +// --------------------------------------------------------------------------- +// 5. THE FULL SNAPSHOT — the red line +// --------------------------------------------------------------------------- + +describe("readEngineModelCatalogue — the full projection, end to end", () => { + /** + * The response body the PRE-refactor `routes/model.js#handleGetModels` + * produced for the fixture above, captured from the implementation at + * 3362c9be and pasted in longhand. Not recomputed by the functions + * under test. + * + * Read it as the batch's contract, in this order: + * + * - 2 engine-session entries FIRST, under `__engine`, ids kept in + * the engine's wire form so `POST /api/set-model` round-trips — + * and BOTH annotated from the builtin tree, because the wire + * form's model segment is the bare engine model key. This is the + * second of the two annotation sites, and the only place the + * snapshot shows both of them at once. + * - 2 providers projected from the engine's `custom_provider` tree + * (`deepseek-cn`, `nousresearch`), each with `auth.hasKey` + * answering the engine's own `options.apiKey` (true / false) and + * `protocol` mapped from the engine's `api` (openai / openai). + * - 3 webui config entries, including `minimax_api/MiniMax-M3` with + * the operator's label and `contextLimit` — the entry that TOOK + * the builtin's slot, which is why the builtin `MiniMax-M3` is + * absent below and carries no `thinkingLevels` and no + * `contextWindowOptions`. + * - 3 surviving builtins: the effort-list model with its context + * windows, and the two bare ones. `MiniMax-M2.5` is the + * forced_on model — the engine's tree has nothing user-settable, + * so the entry stays field-free and the composer mounts no + * control. + * - The three derived figures, and the `source` label. + */ +const EXPECTED = { + ok: true, + models: [ + {"id": "m:minimax_api:MiniMax-M3:v:thinking", "name": "MiniMax-M3", "label": "MiniMax-M3", "provider": "minimax_api", "source": "engine", "thinkingLevels": ["off", "on"], "contextWindowOptions": [512000, 1000000], "contextWindowOptionHints": {"1000000": "higher_usage"}, "contextLimit": 512000}, + {"id": "m:minimax_api:MiniMax-M2.7:u", "name": "MiniMax-M2.7", "label": "MiniMax-M2.7", "provider": "minimax_api", "source": "engine", "thinkingLevels": ["low", "medium", "high"], "contextWindowOptions": [128000, 256000], "contextLimit": 128000}, + {"id": "deepseek-cn/deepseek-chat", "label": "DeepSeek Chat", "provider": "deepseek-cn", "source": "config", "contextLimit": 64000, "protocol": "openai", "thinkingLevels": ["low", "high"], "modalities": ["text", "image"]}, + {"id": "deepseek-cn/deepseek-reasoner", "label": "deepseek-reasoner", "provider": "deepseek-cn", "source": "config", "protocol": "openai"}, + {"id": "nousresearch/z-ai/glm-5.3", "label": "z-ai/glm-5.3", "provider": "nousresearch", "source": "config", "protocol": "openai"}, + {"id": "nousresearch/openai/gpt-5.6-sol", "label": "openai/gpt-5.6-sol", "provider": "nousresearch", "source": "config", "protocol": "openai"}, + {"id": "minimax_api/MiniMax-M3", "label": "M3 config override", "provider": "minimax_api", "source": "config", "contextLimit": 123456, "protocol": "anthropic"}, + {"id": "minimax_api/MiniMax-Text-01", "label": "MiniMax-Text-01", "provider": "minimax_api", "source": "config", "protocol": "anthropic", "thinkingLevels": ["off", "on"], "modalities": ["text", "image"]}, + {"id": "local-ollama/qwen3:8b", "label": "qwen3:8b", "provider": "local-ollama", "source": "config", "protocol": "openai"}, + {"id": "minimax_api/MiniMax-M2.7", "label": "MiniMax-M2.7", "provider": "minimax_api", "source": "builtin", "thinkingLevels": ["low", "medium", "high"], "contextWindowOptions": [128000, 256000], "contextLimit": 128000}, + {"id": "minimax_api/MiniMax-M2.5", "label": "MiniMax-M2.5", "provider": "minimax_api", "source": "builtin"}, + {"id": "minimax_api/MiniMax-M2.7-highspeed", "label": "MiniMax-M2.7-highspeed", "provider": "minimax_api", "source": "builtin"}, + ], + groups: [ + { ...{"id": "__engine", "label": "Engine session"}, models: [ + {"id": "m:minimax_api:MiniMax-M3:v:thinking", "name": "MiniMax-M3", "label": "MiniMax-M3", "provider": "minimax_api", "source": "engine", "thinkingLevels": ["off", "on"], "contextWindowOptions": [512000, 1000000], "contextWindowOptionHints": {"1000000": "higher_usage"}, "contextLimit": 512000}, + {"id": "m:minimax_api:MiniMax-M2.7:u", "name": "MiniMax-M2.7", "label": "MiniMax-M2.7", "provider": "minimax_api", "source": "engine", "thinkingLevels": ["low", "medium", "high"], "contextWindowOptions": [128000, 256000], "contextLimit": 128000}, + ] }, + { ...{"id": "deepseek-cn", "label": "deepseek-cn", "auth": {"hasKey": true, "type": "byok"}, "protocol": "openai"}, models: [ + {"id": "deepseek-cn/deepseek-chat", "label": "DeepSeek Chat", "provider": "deepseek-cn", "source": "config", "contextLimit": 64000, "protocol": "openai", "thinkingLevels": ["low", "high"], "modalities": ["text", "image"]}, + {"id": "deepseek-cn/deepseek-reasoner", "label": "deepseek-reasoner", "provider": "deepseek-cn", "source": "config", "protocol": "openai"}, + ] }, + { ...{"id": "nousresearch", "label": "nousresearch", "auth": {"hasKey": false, "type": "byok"}, "protocol": "openai"}, models: [ + {"id": "nousresearch/z-ai/glm-5.3", "label": "z-ai/glm-5.3", "provider": "nousresearch", "source": "config", "protocol": "openai"}, + {"id": "nousresearch/openai/gpt-5.6-sol", "label": "openai/gpt-5.6-sol", "provider": "nousresearch", "source": "config", "protocol": "openai"}, + ] }, + { ...{"id": "minimax_api", "label": "MiniMax builtins", "auth": {"hasKey": true, "type": "byok"}, "protocol": "anthropic"}, models: [ + {"id": "minimax_api/MiniMax-M3", "label": "M3 config override", "provider": "minimax_api", "source": "config", "contextLimit": 123456, "protocol": "anthropic"}, + {"id": "minimax_api/MiniMax-Text-01", "label": "MiniMax-Text-01", "provider": "minimax_api", "source": "config", "protocol": "anthropic", "thinkingLevels": ["off", "on"], "modalities": ["text", "image"]}, + {"id": "minimax_api/MiniMax-M2.7", "label": "MiniMax-M2.7", "provider": "minimax_api", "source": "builtin", "thinkingLevels": ["low", "medium", "high"], "contextWindowOptions": [128000, 256000], "contextLimit": 128000}, + {"id": "minimax_api/MiniMax-M2.5", "label": "MiniMax-M2.5", "provider": "minimax_api", "source": "builtin"}, + {"id": "minimax_api/MiniMax-M2.7-highspeed", "label": "MiniMax-M2.7-highspeed", "provider": "minimax_api", "source": "builtin"}, + ] }, + { ...{"id": "local-ollama", "label": "local-ollama", "auth": {"hasKey": false, "type": "byok"}, "protocol": "openai"}, models: [ + {"id": "local-ollama/qwen3:8b", "label": "qwen3:8b", "provider": "local-ollama", "source": "config", "protocol": "openai"}, + ] }, + ], + current: "m:minimax_api:MiniMax-M3:v:thinking", + currentThinking: "high", + currentContextWindow: 1000000, + source: "acp-session-config", + }; + + test("the payload is the pre-refactor body, field for field and key for key", () => { + const read = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + assert.equal(read.source, "config"); + assert.equal(read.gate.gate, "checked"); + assert.deepEqual(read.payload, EXPECTED); + }); + + test("the payload's key order is the endpoint's", () => { + // A key-set check alone lets a body that carries the right fields + // in a different order pass; JSON key order is what a snapshot + // diff and a careless consumer both depend on. + const read = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + assert.deepEqual(Object.keys(read.payload), [ + "ok", + "models", + "groups", + "current", + "currentThinking", + "currentContextWindow", + "source", + ]); + }); + + test("the projection is DETERMINISTIC — two reads are deep-equal", () => { + // Every source is re-read per call, so a read that leaked state + // between calls (a shared `seen` set, a mutated projection) would + // show up here and nowhere else. + const a = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + const b = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + assert.deepEqual(a.payload, b.payload); + }); + + test("the `minimax_api` group is the builtins' group, and the config entry took the slot", () => { + // The two facts red line five is really about: grouping is BY + // PROVIDER, and the dedupe is per provider, so an operator's + // override of a builtin id does not leave two `MiniMax-M3` rows in + // the picker. + const { payload } = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + const group = payload.groups.find((g) => g.id === "minimax_api"); + const ids = group.models.map((m) => m.id); + assert.deepEqual(ids, [ + "minimax_api/MiniMax-M3", + "minimax_api/MiniMax-Text-01", + "minimax_api/MiniMax-M2.7", + "minimax_api/MiniMax-M2.5", + "minimax_api/MiniMax-M2.7-highspeed", + ]); + assert.equal(new Set(ids).size, ids.length, "no id may appear twice in a group"); + // Every group holds the SAME entry objects as the flat list — a + // second copy would let the picker and the chip disagree. + for (const g of payload.groups) { + for (const m of g.models) { + assert.equal(payload.models.includes(m), true, `${m.id} is not the same object as the flat entry`); + } + } + }); + + test("no apiKey ever reaches the payload", () => { + // The security contract, end to end: the engine stores its key in + // plaintext and the webui stores one too, and neither may travel. + const { payload } = engine.readEngineModelCatalogue({ cs: SNAPSHOT_CS, transport: RUNTIME }); + const serialised = JSON.stringify(payload); + assert.equal(serialised.includes("sk-secret-should-never-leak"), false); + assert.equal(serialised.includes("sk-webui-fixture-key-0001"), false); + assert.equal(serialised.includes("apiKey"), false); + }); + + test("an empty catalogue answers the soft marker LAST, not an error", () => { + // The endpoint's long-standing hint: with no engine tree, no + // providers config and no builtins, the picker renders "nothing + // attached" and the caller still gets `ok:true` — plus `reason`, + // which is spread AFTER `source` so a consumer reading the body + // positionally sees the same order as on a populated catalogue. + // + // Every source is re-read per call, so pointing the three env vars + // at empty directories for the duration of ONE call is enough; no + // module reload and no test-ordering constraint. + const emptyEngine = mkTmpDir("webui-model-reads-", { parent: root }); + const emptyWebui = mkTmpDir("webui-model-reads-", { parent: root }); + const prev = { + engine: process.env.MINIMAX_DATA_DIR, + webui: process.env.MCODE_WEBUI_DATA_DIR, + config: process.env.MCODE_WEBUI_MODELS_CONFIG, + }; + process.env.MINIMAX_DATA_DIR = emptyEngine; + process.env.MCODE_WEBUI_DATA_DIR = emptyWebui; + process.env.MCODE_WEBUI_MODELS_CONFIG = join(emptyWebui, "absent.json"); + try { + setBuiltinModelsMock([]); + const read = engine.readEngineModelCatalogue({ cs: {}, transport: RUNTIME }); + assert.deepEqual(read.payload, { + ok: true, + models: [], + groups: [], + current: null, + currentThinking: null, + currentContextWindow: null, + source: "mcode-cli-bundle", + reason: "no_catalogue", + }); + assert.deepEqual(Object.keys(read.payload), [ + "ok", + "models", + "groups", + "current", + "currentThinking", + "currentContextWindow", + "source", + "reason", + ]); + // And the soft gate still reports — a soft gate is not a missing + // gate. + assert.equal(read.gate.gate, "checked"); + } finally { + process.env.MINIMAX_DATA_DIR = prev.engine; + process.env.MCODE_WEBUI_DATA_DIR = prev.webui; + process.env.MCODE_WEBUI_MODELS_CONFIG = prev.config; + setBuiltinModelsMock(BUILTINS); + } + }); +}); + +// --------------------------------------------------------------------------- +// 6. Variant / context perturbation — which input moves which annotation +// --------------------------------------------------------------------------- + +describe("the variant and context projections are two views of ONE engine read", () => { + // The engine tree is read twice per request — once for thinking, once + // for context windows — and both are consumed at two sites (the + // engine-session entries and the builtin shell). The failure this + // section exists for is a CROSS-WIRING: one annotation attached to the + // wrong entry, or the two sites disagreeing about the same model. + test("the two real readers agree on the set of models they know", () => { + const thinking = readEngineBuiltinThinking(); + const windows = readEngineBuiltinContextWindows(); + // Same keys, same order — the two readers project the same record. + assert.deepEqual([...thinking.keys()], [...windows.keys()]); + assert.deepEqual([...thinking.keys()], ["MiniMax-M3", "MiniMax-M2.7", "MiniMax-M2.5"]); + }); + + // Table-driven: [model, expected thinkingLevels-or-undefined, + // expected contextWindowOptions-or-undefined, expected contextLimit-or-undefined]. + // Each row is one engine record; the projection must attach EXACTLY + // what that record says, and a record with nothing user-settable + // (the forced_on `MiniMax-M2.5`) must stay field-free. + const TABLE = [ + ["MiniMax-M3", ["off", "on"], [512000, 1000000], 512000], + ["MiniMax-M2.7", ["low", "medium", "high"], [128000, 256000], 128000], + ["MiniMax-M2.5", undefined, undefined, undefined], + ["not-in-the-tree", undefined, undefined, undefined], + ]; + for (const [model, levels, options, limit] of TABLE) { + test(`${model}: levels=${JSON.stringify(levels)} windows=${JSON.stringify(options)}`, () => { + const { list } = engine.projectModelCatalogue({ + providers: null, + builtins: [model], + builtinThinking: readEngineBuiltinThinking(), + builtinContextWindows: readEngineBuiltinContextWindows(), + parseEngineModelWireValue, + }); + const entry = list[0]; + if (levels === undefined) assert.equal("thinkingLevels" in entry, false); + else assert.deepEqual(entry.thinkingLevels, levels); + if (options === undefined) assert.equal("contextWindowOptions" in entry, false); + else assert.deepEqual(entry.contextWindowOptions, options); + if (limit === undefined) assert.equal("contextLimit" in entry, false); + else assert.equal(entry.contextLimit, limit); + }); + } + + test("the same annotations reach the ENGINE-SESSION site, keyed by the wire form's model id", () => { + // The two annotation sites exist because the engine's ACP `model` + // option advertises wire ids, not bare ids. If the lookup used the + // wire VALUE instead of the parsed model id, a cross-client model + // change would silently lose the composer's controls. + // A wire form whose MODEL SEGMENT is the bare builtin id — which is + // what the engine emits for `provider.minimax.models` entries. A + // wire form whose model segment is itself prefixed (or one that + // does not parse at all) misses the builtin tree, and the entry + // stays field-free; that is a miss, not a crash. + const { list } = engine.projectModelCatalogue({ + sessionOption: { options: [{ value: "m:minimax_api:MiniMax-M3:v:thinking", name: "MiniMax-M3" }, { value: "m:minimax_api:MiniMax-M2.7:u", name: "MiniMax-M2.7" }] }, + providers: null, + builtins: [], + builtinThinking: THINKING_M3, + builtinContextWindows: WINDOWS_M3, + parseEngineModelWireValue, + }); + const m3 = list.find((e) => e.id.includes("MiniMax-M3")); + assert.deepEqual(m3.thinkingLevels, ["off", "on"]); + assert.deepEqual(m3.contextWindowOptions, [512000, 1000000]); + assert.deepEqual(m3.contextWindowOptionHints, { 1000000: "higher_usage" }); + // A model that is NOT in the tree gets nothing: the BARE id is + // looked up, and a miss is a miss rather than a partial annotation. + const m27 = list.find((e) => e.id.includes("MiniMax-M2.7")); + assert.equal("thinkingLevels" in m27, false); + assert.equal("contextWindowOptions" in m27, false); + }); + + test("a non-minimax wire form is never annotated from the minimax builtin tree", () => { + // The engine-session annotation is gated on the wire form's + // providerId. A BYOK provider that happens to have a model id + // colliding with a builtin name must not inherit the builtin's + // context windows. + const { list } = engine.projectModelCatalogue({ + sessionOption: { options: [{ value: "m:nousresearch%3Anousresearch%2FMiniMax-M3:u", name: "x" }] }, + providers: null, + builtins: [], + builtinThinking: THINKING_M3, + builtinContextWindows: WINDOWS_M3, + parseEngineModelWireValue, + }); + assert.deepEqual(Object.keys(list[0]), ["id", "name", "label", "provider", "source"]); + }); +}); + +// --------------------------------------------------------------------------- +// 7. The route +// --------------------------------------------------------------------------- + +describe("handleGetModels — the route asks the facade", () => { + // No `setupMocks` in these cases: the file-level `before` hook already + // registered the shared mocks, and node:test's file-level `before` and + // its subtests share ONE MockTracker — a second `setupMocks` here is + // ERR_INVALID_STATE ("already mocked"), not a re-registration. + let bust = 0; + const loadRoute = async () => import(`${absPath("routes/model.js")}?bust=${bust++}`); + + // `mock.module` REPLACES the whole namespace, so a partial mock makes + // the route fail to instantiate on the exports it did not stub. + const NOT_STUBBED = (name) => async () => { + throw new Error(`B4 test called ${name}, which this case did not stub`); + }; + function mockFacade(t, overrides) { + t.mock.module(absPath("engine/model-reads.js"), { + namedExports: { readEngineModelCatalogue: NOT_STUBBED("readEngineModelCatalogue"), ...overrides }, + }); + } + + function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; + } + + test("the handler is still SYNCHRONOUS — the body is complete when it returns", () => { + // The route's signature is part of its contract: an async handler + // would leave `res._body` null for any caller that does not await, + // and the pre-M3 handler was sync. This is the assertion that keeps + // the next reader from "simplifying" the facade to an async one. + const res = mkRes(); + const returned = modelRouteBaseline.handleGetModels(null, res, { cs: SNAPSHOT_CS }); + assert.equal(typeof returned.then, "undefined"); + assert.equal(res.written.length, 2); + assert.equal(res.written[0].status, 200); + }); + + test("the response body is the facade's payload, byte-for-byte", async (t) => { + const payload = { ok: true, models: [], groups: [], current: null, currentThinking: null, currentContextWindow: null, source: "mcode-cli-bundle" }; + mockFacade(t, { readEngineModelCatalogue: () => ({ payload, source: "config", gate: {}, transport: RUNTIME }) }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleGetModels(null, res, { cs: {} }); + assert.equal(res.written[0].headers["Content-Type"], "application/json; charset=utf-8"); + assert.equal(res.written[1].body, JSON.stringify(payload)); + }); + + test("the route hands its ctx through and does not read cs itself", async (t) => { + const seen = []; + mockFacade(t, { + readEngineModelCatalogue: (o) => { + seen.push(o); + return { payload: { ok: true, models: [], groups: [], current: null, currentThinking: null, currentContextWindow: null, source: "mcode-cli-bundle" }, source: "config", gate: {}, transport: RUNTIME }; + }, + }); + const route = await loadRoute(); + for (const ctx of [{ cs: SNAPSHOT_CS }, { cs: null }, undefined, {}]) { + await route.handleGetModels(null, mkRes(), ctx); + } + assert.equal(seen.length, 4); + assert.deepEqual(seen[0].cs, SNAPSHOT_CS); + assert.equal(seen[1].cs, null); + assert.equal(seen[2].cs, undefined); + assert.equal(seen[3].cs, undefined); + for (const o of seen) assert.equal(o.endpoint, undefined); + }); + + test("a gate error PROPAGATES (it is soft, but it must not be swallowed silently)", async (t) => { + // The model's gate never throws today, so this row pins the + // ROUTE's half of the contract: if a future family decision makes + // this gate hard, the route must not grow a catch that turns the + // 501 into a silent empty catalogue — the #110 fake-success + // failure mode, and the one this batch's soft gate exists to avoid + // reaching for. + const marker = new Error("model-gate-refused"); + mockFacade(t, { + readEngineModelCatalogue: () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleGetModels(null, mkRes(), { cs: {} }); + } catch (err) { + caught = err; + } + assert.equal(caught, marker); + }); + + // ---- proof the mock actually took ------------------------------------ + + test("PROOF the facade mock took: a marker error escapes the untouched route", async (t) => { + const marker = new Error("B4-MODEL-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + readEngineModelCatalogue: () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleGetModels(null, mkRes(), { cs: {} }); + } catch (err) { + caught = err; + } + assert.ok(caught, "the route swallowed the facade error — either the mock did not take, or the route grew a catch"); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + + test("CONTROL: with no facade mock, the route answers from the real projection", async (t) => { + // The other half of the proof: a fresh `?bust=` re-import binds the + // route to the REAL facade, so the body is the fixture projection — + // the same one the snapshot above pins, now through the route. + setBuiltinModelsMock(BUILTINS); + const route = await loadRoute(); + const res = mkRes(); + await route.handleGetModels(null, res, { cs: SNAPSHOT_CS }); + const body = JSON.parse(res.written[1].body); + assert.equal(body.ok, true); + assert.equal(body.models.length, 12); + assert.deepEqual(body.groups.map((g) => g.id), [ + "__engine", + "deepseek-cn", + "nousresearch", + "minimax_api", + "local-ollama", + ]); + assert.equal(body.current, MODEL_OPTION.currentValue); + assert.equal(body.currentThinking, "high"); + assert.equal(body.currentContextWindow, 1000000); + assert.equal(body.source, "acp-session-config"); + }); +}); diff --git a/release/public-source.json b/release/public-source.json index d571fd89..80331e1c 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3445,10 +3445,13 @@ "packages/webui/server/app.js", "packages/webui/server/bootstrap.js", "packages/webui/server/cleanup.js", + "packages/webui/server/engine/account-reads.js", "packages/webui/server/engine/capabilities.js", + "packages/webui/server/engine/capability-reads.js", "packages/webui/server/engine/errors.js", "packages/webui/server/engine/host.js", "packages/webui/server/engine/index.js", + "packages/webui/server/engine/model-reads.js", "packages/webui/server/engine/providers/local-runtime-v2.capabilities.js", "packages/webui/server/engine/providers/local-runtime-v2.js", "packages/webui/server/engine/providers/tui-runtime-adapter.js", @@ -3591,9 +3594,12 @@ "packages/webui/test/lib/context-percent.test.js", "packages/webui/test/lib/engine-catalogue.test.js", "packages/webui/test/lib/engine-provider-sync.test.js", + "packages/webui/test/lib/engine/account-reads.test.js", "packages/webui/test/lib/engine/capabilities.test.js", + "packages/webui/test/lib/engine/capability-reads.test.js", "packages/webui/test/lib/engine/capability-snapshot.test.js", "packages/webui/test/lib/engine/host-facade.test.js", + "packages/webui/test/lib/engine/model-reads.test.js", "packages/webui/test/lib/engine/session-export.test.js", "packages/webui/test/lib/engine/session-reads.test.js", "packages/webui/test/lib/engine/session-tree-reads.test.js", diff --git a/scripts/test-tmp-leak.check.mjs b/scripts/test-tmp-leak.check.mjs index 171aa12c..f56d9994 100644 --- a/scripts/test-tmp-leak.check.mjs +++ b/scripts/test-tmp-leak.check.mjs @@ -287,6 +287,7 @@ const KNOWN_PREFIXES = [ "webui-first-turn-guard-", "webui-lan-gate-test-events-", "webui-model-engine-cat-", + "webui-model-reads-", "webui-model-user-level-", "webui-models-merge-", "webui-origingate-events-", From eb2a42977c791a1fde33dc2a844a9dc33ca70b66 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 23:42:20 +0800 Subject: [PATCH 14/21] feat(webui): #73 swaps the ACP wire table for the 14-key engine-capabilities view (user-approved contract change) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `GET /api/protocol/capabilities` used to answer from two hand-maintained places: `MCODE_ACP_CAPABILITIES`, a flat `{method: boolean}` table of the ACP JSON-RPC surface, and the `initialize` agentInfo mirror. The engine's DECLARED capability surface already existed — the 14-key per-provider object that `GET /api/engine-capabilities` serves — so webui was carrying two parallel answers to "what can this engine do", able to disagree, with no test able to notice. This makes `capabilities` the declared object and drops the wire table from the endpoint. This is a reviewed, user-authorised endpoint contract change, not a refactor side effect, and it is stated as such in the module header, in `docs/API.md` and in both ARCHITECTURE twins. The twelve old accessors are asserted GONE, so a consumer reading `capabilities.set_mode` gets undefined and fails loudly rather than receiving a truthy object field. No runtime consumer exists: nothing in `webapp/` reads this endpoint, and `engine/capability-reads.js` no longer imports `lib/mcode-rpc.js` at all (pinned by a static tripwire, because an unused import is behaviourally inert and no behavioural test could see it). The `engine` key the previous commit added is REMOVED rather than kept: with `capabilities` already the declaration, an `engine` block would carry the same 14 keys a second time in one response. What survives from that shape is the provenance — `capabilitiesProvider` / `capabilitiesProviderFor`, the honest bit that says whether the declaration came from the active transport's provider or from the default one standing in — plus `capabilitiesUnavailable` for the derived degradation roll-up. A test counts the declaration's occurrences in the serialised body and requires exactly one, so a second carrier is a red bar. `MCODE_ACP_CAPABILITIES` is kept and stays pinned by `test/lib/mcode-rpc.check.mjs`: it is still a true statement about the ENGINE's ACP surface and `docs/CAPABILITIES.md` cites it as one. It has no webui consumer left, recorded as debt in the module header rather than deleted as a side effect. `docs/webui.md` and `docs/webui.zh-CN.md` gain a diff here for the first time in this migration: they carried the old response shape in their endpoint tables, and an authorised contract change has to be documented where the contract is written. --- docs/tui-capabilities.md | 16 +- docs/webui.md | 2 +- docs/webui.zh-CN.md | 2 +- packages/webui/docs/API.md | 112 ++++---- packages/webui/docs/API.zh-CN.md | 107 ++++---- packages/webui/docs/ARCHITECTURE.md | 35 ++- packages/webui/docs/ARCHITECTURE.zh-CN.md | 26 +- .../webui/server/engine/capability-reads.js | 96 ++++--- packages/webui/server/routes/protocol.js | 37 ++- .../test/lib/engine/capability-reads.test.js | 242 ++++++++++++------ packages/webui/test/lib/mcode-rpc.check.mjs | 14 +- 11 files changed, 431 insertions(+), 258 deletions(-) diff --git a/docs/tui-capabilities.md b/docs/tui-capabilities.md index 4b53b058..df506203 100644 --- a/docs/tui-capabilities.md +++ b/docs/tui-capabilities.md @@ -220,7 +220,11 @@ Status legend: ✅ wired · ⚠ partial / path differs · ❌ no path · 🚧 re The webui does not yet expose `/fork`, `/resume`, or a "rewind last turn" action — those acp methods (`fork`, `resume`) are reported by -`MCODE_ACP_CAPABILITIES` but no webui route wraps them. +`MCODE_ACP_CAPABILITIES` but no webui route wraps them. (Since M3-B4 that +table is no longer what `GET /api/protocol/capabilities` returns; the +endpoint serves the engine's declared 14-key capability object instead. +The table is still exported and still pinned by +`test/lib/mcode-rpc.check.mjs`.) ## ACP Skill commands @@ -252,7 +256,15 @@ entries by default. mcodeVersion, mcodeName?, mcodeTitle?, - capabilities: MCODE_ACP_CAPABILITIES, // see packages/webui/server/lib/mcode-rpc.js + // The engine's DECLARED 14-key capability object, served by + // packages/webui/server/engine/capability-reads.js. Before M3-B4 this + // field carried MCODE_ACP_CAPABILITIES (the ACP wire table in + // packages/webui/server/lib/mcode-rpc.js), which is still exported + // there and still a true statement about the ENGINE's ACP surface. + capabilities: <14-key declaration>, + capabilitiesProvider, // which provider's declaration answered + capabilitiesProviderFor, // "transport" | "default" (see API.md) + capabilitiesUnavailable, // the degradation roll-up notes: { set_mode, set_config_option, cancel, activate, fork, load, list, close, new, prompt, diff --git a/docs/webui.md b/docs/webui.md index d86fbb85..473eee90 100644 --- a/docs/webui.md +++ b/docs/webui.md @@ -2411,7 +2411,7 @@ marker), not by tool name. | `POST` | `/api/protocol/load-session` | `routes/protocol.js#handleLoadSession` | `?cwd=`, fallback to current | | `POST` | `/api/protocol/activate-session` | `routes/protocol.js#handleActivateSession` | one acp client tracks one active session | | `GET` | `/api/protocol/list-sessions` | `routes/protocol.js#handleListSessions` | `?cwd=` filtered | -| `GET` | `/api/protocol/capabilities` | `routes/protocol.js#handleCapabilities` | `{mcodeVersion, mcodeName?, mcodeTitle?, capabilities: MCODE_ACP_CAPABILITIES, notes}` | +| `GET` | `/api/protocol/capabilities` | `routes/protocol.js#handleCapabilities` | `{mcodeVersion, mcodeName?, mcodeTitle?, capabilities, capabilitiesProvider, capabilitiesProviderFor, capabilitiesUnavailable, notes}` — `capabilities` is the engine's declared 14-key capability object (it was the ACP wire table `MCODE_ACP_CAPABILITIES` before M3-B4) | ### Legacy dispatcher (`server/router.js`) diff --git a/docs/webui.zh-CN.md b/docs/webui.zh-CN.md index a37229b7..3c18abc5 100644 --- a/docs/webui.zh-CN.md +++ b/docs/webui.zh-CN.md @@ -1797,7 +1797,7 @@ createdAtMs, updatedAtMs}`)下发,按 `toolCallId` 幂等、上限 32 条、 | `POST` | `/api/protocol/load-session` | `routes/protocol.js#handleLoadSession` | `?cwd=`,缺省取当前 | | `POST` | `/api/protocol/activate-session` | `routes/protocol.js#handleActivateSession` | 一个 acp 客户端跟踪一个活动会话 | | `GET` | `/api/protocol/list-sessions` | `routes/protocol.js#handleListSessions` | `?cwd=` 过滤 | -| `GET` | `/api/protocol/capabilities` | `routes/protocol.js#handleCapabilities` | `{mcodeVersion, mcodeName?, mcodeTitle?, capabilities: MCODE_ACP_CAPABILITIES, notes}` | +| `GET` | `/api/protocol/capabilities` | `routes/protocol.js#handleCapabilities` | `{mcodeVersion, mcodeName?, mcodeTitle?, capabilities, capabilitiesProvider, capabilitiesProviderFor, capabilitiesUnavailable, notes}`——`capabilities` 是引擎声明的 14 键能力对象(M3-B4 之前是 ACP wire 表 `MCODE_ACP_CAPABILITIES`) | ### 旧派发器(`server/router.js`) diff --git a/packages/webui/docs/API.md b/packages/webui/docs/API.md index 7effdb52..b28d3fa6 100644 --- a/packages/webui/docs/API.md +++ b/packages/webui/docs/API.md @@ -2416,23 +2416,25 @@ insensitive, trailing slash-insensitive, `\` and `/` interchangeable). ### `GET /api/protocol/capabilities` -Returns the engine's `agentInfo` (from the `initialize` reply), the -capability table webui knows about (`MCODE_ACP_CAPABILITIES` in -`server/lib/mcode-rpc.js`), and — since M3 batch B4 — the -**engine-capabilities view**: the declared 14-key capability surface of -the active engine provider plus its degradation summary, the same -declaration `GET /api/engine-capabilities` serves. Used by the webui to -decide which UI controls to enable. - -Two tables, two questions, both kept: - -- `capabilities` answers **"which ACP JSON-RPC method does this control - map onto"** — a flat `{method: boolean}` map. -- `engine.capabilities` answers **"does the engine have this capability at - all"** — the 14 matrix keys, each `{level, missing?, reason?}`. - -They can legitimately disagree (the ACP surface and the capability matrix -are not the same taxonomy), so neither replaces the other. +Returns the engine's `agentInfo` (from the `initialize` reply) and the +**engine-capabilities view**: the declared 14-key capability surface of the +active engine provider, the same declaration `GET /api/engine-capabilities` +serves. Used by the webui to decide which UI controls to enable. + +**This field's contract changed in M3 batch B4.** `capabilities` used to +carry `MCODE_ACP_CAPABILITIES`, a hand-maintained flat `{method: boolean}` +table of the ACP JSON-RPC surface (`set_mode`, `set_config_option`, +`cancel`, `activate`, `fork`, `resume`, `delete`, `load`, `close`, `list`, +`new`, `prompt`). Those twelve keys are **gone**: a consumer reading +`capabilities.set_mode` now gets `undefined` and must fail loudly. What +replaced them answers a different question — **"does the engine have this +capability at all"** — with the 14 matrix keys, each +`{level, missing?, reason?}`. The ACP wire table is still exported from +`server/lib/mcode-rpc.js` and is still a true statement about the +engine's ACP surface; it simply no longer travels on this endpoint. + +The declaration appears exactly once, under `capabilities`, and three +sibling keys say where it came from and what to do about its gaps. **Response 200** ```json @@ -2442,35 +2444,46 @@ are not the same taxonomy), so neither replaces the other. "mcodeName": "mcode", "mcodeTitle": "mcode", "capabilities": { - "set_mode": true, - "set_config_option": true, - "cancel": true, - "activate": true, - "fork": true, - "resume": true, - "delete": false, - "load": true, - "close": true, - "list": true, - "new": true, - "prompt": true - }, - "engine": { - "provider": "local-runtime-v2", - "providerFor": "transport", - "transport": "runtime", - "capabilities": { - "sessionCrud": { "level": "full" }, - "updateCheck": { - "level": "none", - "reason": "interface-absent: no update-check method anywhere in local-runtime-v2 (design §1.3 v2)" - } + "sessionCrud": { "level": "full" }, + "streamingSend": { "level": "full" }, + "interrupt": { "level": "full" }, + "toolSkillInvocation": { "level": "full" }, + "turnDiff": { "level": "full" }, + "turnRewindRedo": { "level": "full" }, + "plugins": { "level": "full" }, + "mcp": { "level": "full" }, + "subagents": { + "level": "partial", + "missing": ["getDelegationSnapshot", "stopDelegation"], + "reason": "delegation snapshot/stop live on the TuiRuntimeAdapter access-context, not on the v2 CliService surface (design §1.3 v2)" + }, + "usageStats": { "level": "full" }, + "authCredentials": { "level": "full" }, + "updateCheck": { + "level": "none", + "reason": "interface-absent: no update-check method anywhere in local-runtime-v2 (design §1.3 v2)" }, - "unavailable": { - "none": ["updateCheck"], - "partial": [{ "key": "gitOperations", "missing": ["git-diff", "git-commit", "git-branch"] }] + "fileReadWrite": { + "level": "partial", + "missing": ["file-write"], + "reason": "workspace read browsing only; no write API — writes go through in-turn tools (design §1.3 v2)" + }, + "gitOperations": { + "level": "partial", + "missing": ["git-diff", "git-commit", "git-branch"], + "reason": "read-only metadata + review link; change mutation is outside this package (same discipline as v1's read-only Git facade)" } }, + "capabilitiesProvider": "local-runtime-v2", + "capabilitiesProviderFor": "transport", + "capabilitiesUnavailable": { + "none": ["updateCheck"], + "partial": [ + { "key": "subagents", "missing": ["getDelegationSnapshot", "stopDelegation"] }, + { "key": "fileReadWrite", "missing": ["file-write"] }, + { "key": "gitOperations", "missing": ["git-diff", "git-commit", "git-branch"] } + ] + }, "notes": { "set_mode": "Takes a modeId from the session's availableModes.", "set_config_option": "With configId 'permissionMode' this changes the mode mid-session.", @@ -2484,8 +2497,9 @@ are not the same taxonomy), so neither replaces the other. `mcodeVersion` is `"unknown"` before a client has attached (no `initialize` reply yet); the endpoint does not invent a version. -`engine.providerFor` says where the declaration came from, and a consumer -should branch on it: +`capabilitiesProvider` is the provider whose declaration answered, and +`capabilitiesProviderFor` says HOW it was chosen. A consumer should +branch on the second one: - `"transport"` — the active `MCODE_WEBUI_TRANSPORT`'s own registered provider answered. @@ -2494,9 +2508,11 @@ should branch on it: real, reviewed declaration, but it is not necessarily the connected engine's, and reporting it as such would be a lie. -`engine.unavailable` is the degradation summary the capability-driven UI -renders from: a `none` key means hide the entry point, a `partial` key means -hide or disable exactly the listed sub-actions. +`capabilitiesUnavailable` is the degradation summary the capability-driven +UI renders from: a `none` key means hide the entry point, a `partial` key +means hide or disable exactly the listed sub-actions. It is the one field +that is not the declaration itself, and a consumer should not have to +re-derive it from a taxonomy with three levels and two optional fields. --- diff --git a/packages/webui/docs/API.zh-CN.md b/packages/webui/docs/API.zh-CN.md index cdbd0a3a..05d0ffc2 100644 --- a/packages/webui/docs/API.zh-CN.md +++ b/packages/webui/docs/API.zh-CN.md @@ -2230,22 +2230,24 @@ code, killEndpoint: "/api/stop" }`。温和版→SIGKILL 的级联 ### `GET /api/protocol/capabilities` -返回引擎的 `agentInfo`(取自 `initialize` 应答)、webui 已知的 -capability 表(`server/lib/mcode-rpc.js` 里的 -`MCODE_ACP_CAPABILITIES`),以及——自 M3 批次 B4 起——**engine-capabilities -视图**:当前引擎 provider 声明的 14 键能力面加它的降级摘要,也就是 -`GET /api/engine-capabilities` 所服务的同一份声明。webui 用它来决定启用 -哪些 UI 控件。 - -两张表、两个问题,都保留: - -- `capabilities` 回答的是**「这个控件对应哪个 ACP JSON-RPC 方法」**—— - 一张扁平的 `{方法: 布尔}` 表。 -- `engine.capabilities` 回答的是**「引擎到底有没有这项能力」**—— - 14 个矩阵键,每项形如 `{level, missing?, reason?}`。 - -两者可以合法地不一致(ACP 面与能力矩阵不是同一套分类法),所以谁也 -不替换谁。 +返回引擎的 `agentInfo`(取自 `initialize` 应答)与 +**engine-capabilities 视图**:当前引擎 provider 声明的 14 键能力面, +也就是 `GET /api/engine-capabilities` 所服务的同一份声明。webui 用它来 +决定启用哪些 UI 控件。 + +**这个字段的契约在 M3 批次 B4 变更过。** `capabilities` 过去承载 +`MCODE_ACP_CAPABILITIES`——一张手工维护的扁平 `{方法: 布尔}` 表,描述 +ACP JSON-RPC 面(`set_mode`、`set_config_option`、`cancel`、`activate`、 +`fork`、`resume`、`delete`、`load`、`close`、`list`、`new`、 +`prompt`)。这 12 个键**已经没有了**:读 `capabilities.set_mode` 的消费方 +现在拿到 `undefined`,会响亮地失败。顶替它们回答的是另一个问题 +——**「引擎到底有没有这项能力」**——用 14 个矩阵键,每项形如 +`{level, missing?, reason?}`。ACP wire 表仍从 +`server/lib/mcode-rpc.js` 导出,且仍是对**引擎** ACP 面的真实陈述; +它只是不再随这个端点返回。 + +声明在响应里只出现一次,就在 `capabilities` 下;三个兄弟键说明它从哪来 +以及该拿它的缺口怎么办。 **响应 200** ```json @@ -2255,35 +2257,46 @@ capability 表(`server/lib/mcode-rpc.js` 里的 "mcodeName": "mcode", "mcodeTitle": "mcode", "capabilities": { - "set_mode": true, - "set_config_option": true, - "cancel": true, - "activate": true, - "fork": true, - "resume": true, - "delete": false, - "load": true, - "close": true, - "list": true, - "new": true, - "prompt": true - }, - "engine": { - "provider": "local-runtime-v2", - "providerFor": "transport", - "transport": "runtime", - "capabilities": { - "sessionCrud": { "level": "full" }, - "updateCheck": { - "level": "none", - "reason": "interface-absent: no update-check method anywhere in local-runtime-v2 (design §1.3 v2)" - } + "sessionCrud": { "level": "full" }, + "streamingSend": { "level": "full" }, + "interrupt": { "level": "full" }, + "toolSkillInvocation": { "level": "full" }, + "turnDiff": { "level": "full" }, + "turnRewindRedo": { "level": "full" }, + "plugins": { "level": "full" }, + "mcp": { "level": "full" }, + "subagents": { + "level": "partial", + "missing": ["getDelegationSnapshot", "stopDelegation"], + "reason": "delegation snapshot/stop live on the TuiRuntimeAdapter access-context, not on the v2 CliService surface (design §1.3 v2)" + }, + "usageStats": { "level": "full" }, + "authCredentials": { "level": "full" }, + "updateCheck": { + "level": "none", + "reason": "interface-absent: no update-check method anywhere in local-runtime-v2 (design §1.3 v2)" }, - "unavailable": { - "none": ["updateCheck"], - "partial": [{ "key": "gitOperations", "missing": ["git-diff", "git-commit", "git-branch"] }] + "fileReadWrite": { + "level": "partial", + "missing": ["file-write"], + "reason": "workspace read browsing only; no write API — writes go through in-turn tools (design §1.3 v2)" + }, + "gitOperations": { + "level": "partial", + "missing": ["git-diff", "git-commit", "git-branch"], + "reason": "read-only metadata + review link; change mutation is outside this package (same discipline as v1's read-only Git facade)" } }, + "capabilitiesProvider": "local-runtime-v2", + "capabilitiesProviderFor": "transport", + "capabilitiesUnavailable": { + "none": ["updateCheck"], + "partial": [ + { "key": "subagents", "missing": ["getDelegationSnapshot", "stopDelegation"] }, + { "key": "fileReadWrite", "missing": ["file-write"] }, + { "key": "gitOperations", "missing": ["git-diff", "git-commit", "git-branch"] } + ] + }, "notes": { "set_mode": "Takes a modeId from the session's availableModes.", "set_config_option": "With configId 'permissionMode' this changes the mode mid-session.", @@ -2297,7 +2310,9 @@ capability 表(`server/lib/mcode-rpc.js` 里的 `mcodeVersion` 在尚无客户端挂接(还没收到 `initialize` 应答) 时为 `"unknown"`;本端点不会臆造一个版本号。 -`engine.providerFor` 说明这份声明来自哪里,消费方应当据此分支: +`capabilitiesProvider` 是应答了的那份声明所属的 provider, +`capabilitiesProviderFor` 说明它是**怎么**被选中的。消费方应当对后者 +分支: - `"transport"`——当前 `MCODE_WEBUI_TRANSPORT` 自己的已注册 provider 应答的。 @@ -2305,8 +2320,10 @@ capability 表(`server/lib/mcode-rpc.js` 里的 provider 的声明顶替。这份视图仍是一份真实且经评审的声明,但它未必 是已连接引擎的那份;把它当成后者报出去就是撒谎。 -`engine.unavailable` 是能力驱动型 UI 据以渲染的降级摘要:`none` 的键 -意味着隐藏整个入口,`partial` 的键意味着恰好隐藏或禁用列出的那些子动作。 +`capabilitiesUnavailable` 是能力驱动型 UI 据以渲染的降级摘要:`none` +的键意味着隐藏整个入口,`partial` 的键意味着恰好隐藏或禁用列出的那些 +子动作。它是唯一一个并非声明本身的字段,消费方不该被迫从一个有三级 +两可选字段的分类法里重新推导它。 --- diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index 97ddecfa..09a77dc9 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -507,7 +507,7 @@ files, one job each: | `engine/usage-reads.js` | The usage family's facade calls (`readEngineAccountQuota`, `readEngineSessionUsage`, `readEngineQuotaForecast`), the derived figure `contextUsedTokens`, and the endpoint→capability table `USAGE_READ_ENDPOINTS` (step M3, batch B3). Gates **hard** on the two engine reads and declares **no capability at all** for #19, which touches no engine surface | | `engine/account-reads.js` | The account family's facade call (`readEngineAccount`) and the endpoint→capability table `ACCOUNT_READ_ENDPOINTS` (step M3, batch B4). Gates **hard** on `authCredentials` · `getAccountStatus` — the same pair and the same provider method as `engine/usage-reads.js`, because #20 and #15/#16 read the same engine projection. Its read is **synchronous**; see the boot-path note below | | `engine/model-reads.js` | The model-catalogue family's facade call (`readEngineModelCatalogue`), the whole projection as named pure functions (`projectModelCatalogue`, `deriveModelSelection`, `buildModelCataloguePayload`, `catalogueSourceLabel`, `webuiFullModelId`, `providerOfModelId`, `attachContextWindowOptions`, `configOption`), and the endpoint→capability table `MODEL_READ_ENDPOINTS` (step M3, batch B4). Gates **soft**: `checkModelReadCapability` reports and never throws, because the catalogue's primary sources are files webui owns. Its read is **synchronous**, and it is the one engine module **not** re-exported from `engine/index.js` — see the boot-path note below | -| `engine/capability-reads.js` | The capability-declaration family's facade call (`readEngineCapabilityView`) and the endpoint→capability table `CAPABILITY_READ_ENDPOINTS` (step M3, batch B4). Declares **no capability for #73** — it IS the declaration endpoint, and gating the gate would let a `none` hide the declaration that says so. It is the only endpoint in the migration whose response body gains a key (`engine`, the engine-capabilities view) | +| `engine/capability-reads.js` | The capability-declaration family's facade call (`readEngineCapabilityView`) and the endpoint→capability table `CAPABILITY_READ_ENDPOINTS` (step M3, batch B4). Declares **no capability for #73** — it IS the declaration endpoint, and gating the gate would let a `none` hide the declaration that says so. It is the only endpoint in the migration whose response CONTRACT changed (`capabilities` is now the 14-key declaration, replacing the ACP wire table) | Routes take the host from the facade and never from `lib/acp-client.js`: `routes/plugins.js` and `routes/turn-diff.js` call @@ -787,7 +787,7 @@ would force one family to inherit another's policy. | --- | --- | --- | --- | | `GET /api/account` | `authCredentials` · `getAccountStatus` | hard — 501 | `lib/mcode-rpc.js#getAccountStatus`, the engine's `mcode/account/status` projection. The response body is built by the facade: `{ok:true, ...data}` on success, `{ok:false, reason}` at HTTP 200 otherwise | | `GET /api/models` | `authCredentials` · `listModelProviders` | soft — reported | three layered sources: the engine session's `model` config option, the merged providers config (webui `env > cwd > user` over the engine's `custom_provider` tree, via `lib/engine-catalogue.js`), and the builtin cli-bundle extraction | -| `GET /api/protocol/capabilities` | none of the 14 keys | none — the gate is a reported no-op | the registered provider's 14-key declaration plus `summarizeUnavailableCapabilities`, and the ACP `initialize` `agentInfo` mirror | +| `GET /api/protocol/capabilities` | none of the 14 keys | none — the gate is a reported no-op | the registered provider's 14-key declaration, its `summarizeUnavailableCapabilities` roll-up, and the ACP `initialize` `agentInfo` mirror | **Why #20 gates hard and #57 does not.** The account card is 100% engine data: there is no webui-side fallback for "who am I" or for a plan tier, so @@ -838,16 +838,27 @@ Four properties this batch holds, each with a test behind it: model absent from the tree, produces a field-free entry — never a half-annotation. The section that perturbs the tree asserts which entries move for which record. -4. **#73's change is additive and its fallback is labelled.** The response - gains exactly one key, `engine`, placed after `capabilities`; every - pre-existing key keeps its exact name, position and value, and the ACP - wire table is **not** replaced by the 14 matrix keys (they answer a - different question, and `docs/API.md` documents both). Inside the view, - `providerFor` says whether the declaration came from the active - transport's provider or from the default provider standing in for a - transport no provider claims yet — a capability-detection endpoint must - not report a standing-in declaration as though it were the connected - engine's. +4. **#73's contract CHANGED, deliberately, and the declaration appears + once.** `capabilities` used to be `MCODE_ACP_CAPABILITIES`, a + hand-maintained flat `{method: boolean}` table of the ACP JSON-RPC + surface; it is now the engine's **declared** 14-key object, forwarded by + identity. The twelve old accessors are asserted gone, so a consumer + reading `capabilities.set_mode` gets `undefined` and fails loudly + rather than receiving a truthy object field. This is the one + user-authorised endpoint contract change in the migration, and the + first shape of it — an additive `engine` block carrying the view + beside the old table — was rejected in review precisely because it + would have carried the same 14 keys twice in one response. What + survives from that shape is the provenance, hoisted to + `capabilitiesProvider` / `capabilitiesProviderFor`, plus + `capabilitiesUnavailable` for the derived roll-up. The test counts the + declaration's occurrences structurally, so re-introducing a second + carrier is a red bar. `providerFor` is the honest bit: a + capability-detection endpoint must not report a standing-in + declaration as though it were the connected engine's, and under the + default `acp` transport that standing-in is the normal case until M4. + `docs/API.md`, `docs/webui.md` and `docs/tui-capabilities.md` all + record the new shape in both languages. **The three "what is active" figures are derived once.** `current` prefers the engine's `currentValue` and falls back to the recorded pre-session pick; diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index 24294066..5ffbc835 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -478,7 +478,7 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 | `engine/usage-reads.js` | 用量族的面板调用(`readEngineAccountQuota`、`readEngineSessionUsage`、`readEngineQuotaForecast`)、派生量 `contextUsedTokens`,与端点→能力对照表 `USAGE_READ_ENDPOINTS`(迁移步 M3 批次 B3)。两个引擎读**硬门控**;#19 **完全不声明能力**,因为它不触达任何引擎面 | | `engine/account-reads.js` | 账户族的面板调用 `readEngineAccount` 与端点→能力对照表 `ACCOUNT_READ_ENDPOINTS`(迁移步 M3 批次 B4)。**硬门控**,门控在 `authCredentials` · `getAccountStatus`——与 `engine/usage-reads.js` 同一对、同一个 provider 方法,因为 #20 与 #15/#16 读的是同一份引擎投影。它的读是**同步的**,见下面的启动路径说明 | | `engine/model-reads.js` | 模型目录族的面板调用 `readEngineModelCatalogue`、整套投影的具名纯函数(`projectModelCatalogue`、`deriveModelSelection`、`buildModelCataloguePayload`、`catalogueSourceLabel`、`webuiFullModelId`、`providerOfModelId`、`attachContextWindowOptions`、`configOption`),与端点→能力对照表 `MODEL_READ_ENDPOINTS`(迁移步 M3 批次 B4)。**软门控**:`checkModelReadCapability` 只报告、从不抛出,因为目录的主数据源是 webui 自己拥有的文件。它的读是**同步的**,并且它是唯一一个**没有**从 `engine/index.js` 转发导出的引擎模块——见下面的启动路径说明 | -| `engine/capability-reads.js` | 能力声明族的面板调用 `readEngineCapabilityView` 与端点→能力对照表 `CAPABILITY_READ_ENDPOINTS`(迁移步 M3 批次 B4)。#73 **不声明任何能力**——它本身就是声明端点,给门控上门控会让某个 `none` 把声明它的那份声明藏起来。它是本次迁移中唯一一个响应体新增了一个键的端点(`engine`,即 engine-capabilities 视图) | +| `engine/capability-reads.js` | 能力声明族的面板调用 `readEngineCapabilityView` 与端点→能力对照表 `CAPABILITY_READ_ENDPOINTS`(迁移步 M3 批次 B4)。#73 **不声明任何能力**——它本身就是声明端点,给门控上门控会让某个 `none` 把声明它的那份声明藏起来。它是本次迁移中唯一一个响应**契约**发生变更的端点(`capabilities` 现在是 14 键声明,顶替了 ACP wire 表) | 路由从门面取 host,不从 `lib/acp-client.js` 取:`routes/plugins.js` 与 `routes/turn-diff.js` 调 `getEngineCatalogueHost()`。两者都保留 `deps` @@ -712,7 +712,7 @@ provider 确实没有树可返回,501 才是诚实答案。 | --- | --- | --- | --- | | `GET /api/account` | `authCredentials` · `getAccountStatus` | 硬——501 | `lib/mcode-rpc.js#getAccountStatus`,即引擎的 `mcode/account/status` 投影。响应体由门面组装:成功是 `{ok:true, ...data}`,失败在 HTTP 200 上是 `{ok:false, reason}` | | `GET /api/models` | `authCredentials` · `listModelProviders` | 软——只报告 | 三个分层来源:引擎会话的 `model` 配置项、合并后的 provider 配置(webui 的 `env > cwd > user` 叠在引擎 `custom_provider` 树之上,经 `lib/engine-catalogue.js`)、以及内建 cli 包抽取 | -| `GET /api/protocol/capabilities` | 14 个键里的任何一个都不适用 | 不门控——门控是「被报告的空操作」 | 已注册 provider 的 14 键声明加 `summarizeUnavailableCapabilities`,以及 ACP `initialize` 的 `agentInfo` 镜像 | +| `GET /api/protocol/capabilities` | 14 个键里的任何一个都不适用 | 不门控——门控是「被报告的空操作」 | 已注册 provider 的 14 键声明、它的 `summarizeUnavailableCapabilities` 汇总,以及 ACP `initialize` 的 `agentInfo` 镜像 | **为什么 #20 硬门控而 #57 不硬。** 账户卡 100% 由引擎数据构成: 「我是谁」和「什么套餐」都没有 webui 侧的兜底,所以报不出账户的 @@ -756,13 +756,21 @@ webui 自己拥有、不依赖引擎就能读的文件——`models.json`、 模型段解析不出来、或模型不在树里,产出的就是一个无这些字段的条目, 绝不会是「半吊子标注」。扰动那棵树的那一节断言了:哪条引擎记录会让 哪些条目发生变化。 -4. **#73 的变更是增量的,且它的兜底是带标签的。** 响应恰好新增一个键 - `engine`,位置紧跟 `capabilities` 之后;每个既有键的名字、位置与取值 - 都不变,ACP wire 表**没有**被 14 个矩阵键替换(两者回答的是不同问题, - `docs/API.md` 两者都记录了)。视图内部的 `providerFor` 说明这份声明 - 来自当前传输的 provider,还是来自「当前传输还没有任何 provider 声明」 - 时顶替的默认 provider——一个能力探测端点绝不能把顶替声明当作已连接 - 引擎的声明报出去。 +4. **#73 的契约是「变更」了,且是刻意的,声明只出现一次。** `capabilities` + 过去是 `MCODE_ACP_CAPABILITIES`——一张手工维护的扁平 `{方法: 布尔}` + 表,描述 ACP JSON-RPC 面;现在是引擎**声明的** 14 键对象,按引用 + 转发。那 12 个旧访问器被断言为**已消失**,所以读 + `capabilities.set_mode` 的消费方拿到 `undefined`、响亮地失败,而不是 + 收到一个真值对象字段。这是本次迁移里唯一一处经用户授权的端点契约 + 变更;它的第一个形状——在旧表旁边增一个 `engine` 块承载视图——在评审 + 中被否掉,正因为那会让同一份 14 键声明在一次响应里出现两次。留下来的是 + 出处信息,上提为 `capabilitiesProvider` / + `capabilitiesProviderFor`,外加派生汇总 + `capabilitiesUnavailable`。测试用结构化计数断言声明的出现次数,所以再 + 引入第二个承载者就是一条红条。`providerFor` 是诚实位:能力探测端点 + 绝不能把顶替声明当作已连接引擎的声明报出去,而在默认 `acp` 传输下, + 直到 M4 之前这种顶替都是常态。`docs/API.md`、`docs/webui.md`、 + `docs/tui-capabilities.md` 都以两种语言记录了新形状。 **三个「当前生效」的量只派生一次。** `current` 优先取引擎的 `currentValue`,回落到记录在案的会话前选择;`currentThinking` 优先取引擎的 diff --git a/packages/webui/server/engine/capability-reads.js b/packages/webui/server/engine/capability-reads.js index 9e7561bf..3c054ca7 100644 --- a/packages/webui/server/engine/capability-reads.js +++ b/packages/webui/server/engine/capability-reads.js @@ -20,21 +20,30 @@ // `engine/capabilities.js` and the registry in `engine/index.js`, and // `GET /api/engine-capabilities` already serves it. So webui was // carrying two parallel answers to "what can the engine do", able to -// disagree, with no test able to notice. After M3-B4 #73 carries the -// engine-capabilities VIEW alongside the ACP wire table: the route no -// longer reaches into `lib/mcode-rpc.js` and `lib/acp-client.js` on -// its own, and the two answers sit in one response where a consumer -// (or a reviewer) can see both and their disagreement. +// disagree, with no test able to notice. After M3-B4 `capabilities` IS +// the engine-capabilities view: the route no longer reaches into +// `lib/mcode-rpc.js` and `lib/acp-client.js` on its own, and there is +// one answer rather than two. // -// The wire table is KEPT, not replaced. `capabilities` still answers -// "which ACP method does the frontend's control map onto", which is -// not what the 14 matrix keys answer ("does the engine have this -// capability at all"). Dropping it would break `docs/API.md`'s -// documented response and every consumer that reads -// `capabilities.set_mode`; the engine view is ADDITIVE. That is the -// one place in this batch where the response body gains a key, and it -// is a deliberate, reviewed decision rather than a refactor side -// effect — the existing keys keep their exact values. +// The ACP wire table is REPLACED, not kept alongside — a reviewed, +// user-authorised endpoint contract change, not a refactor side effect. +// `MCODE_ACP_CAPABILITIES` described a different taxonomy (which ACP +// JSON-RPC method exists) and it had drifted into being the endpoint's +// headline field while nothing in the webapp read it. Carrying both +// would have meant the 14-key declaration appeared twice in one +// response, once as the answer and once as a decoration, so the extra +// `engine` key this batch first shipped was removed rather than kept. +// What survives from that first shape is the honest provenance — which +// provider answered, and whether it was standing in — hoisted to +// `capabilitiesProvider` / `capabilitiesProviderFor`. +// +// KNOWN DEBT, recorded rather than acted on: `MCODE_ACP_CAPABILITIES` +// in `lib/mcode-rpc.js` now has no consumer. It is still exported and +// still pinned by `test/lib/mcode-rpc.check.mjs`, and `docs/CAPABILITIES.md` +// cites it as a fact about the ENGINE's ACP surface (which it still +// is), so deleting it is a separate decision about dead code, not a +// side effect of replacing a response field. `test/helpers/_setup.js` +// mirrors the export for the same reason. // // Why this endpoint declares NO capability. It is the declaration // endpoint: gating the gate is circular, and a `none` anywhere in the @@ -65,10 +74,12 @@ // Boot-path weight. `app.js` imports `routes/protocol.js`, the route // imports this file, so this file is on the boot path. It statically // imports nothing heavier than `capabilities.js` and `index.js`; -// `lib/mcode-rpc.js` and `lib/acp-client.js` are reached through +// `lib/acp-client.js` and `lib/config.js` are reached through // `await import()` inside the read — the M1 lesson, and the reason the // route's own `await import(...)` lines moved behind this boundary -// rather than being duplicated. +// rather than being duplicated. (`lib/mcode-rpc.js` was in that list +// while the endpoint still served the ACP wire table; replacing the +// field removed the dependency, not just the field.) import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; @@ -141,9 +152,7 @@ export function checkCapabilityReadCapability(endpoint, transport) { } /** - * The engine-capabilities VIEW — the same four facts - * `GET /api/engine-capabilities` serves, plus HOW the provider was - * chosen. + * How the declaration that answered was chosen. * * `providerFor` is the honest bit: `"transport"` means the active * transport's own provider answered; `"default"` means no provider @@ -159,32 +168,39 @@ export function checkCapabilityReadCapability(endpoint, transport) { * provider: string, * providerFor: "transport"|"default", * transport: string, - * capabilities: object, - * unavailable: {none: string[], partial: Array<{key: string, missing: string[]}>}, - * }} EngineCapabilityView + * }} EngineCapabilityProvenance */ /** * The #73 (`GET /api/protocol/capabilities`) read. * - * `wire` is `MCODE_ACP_CAPABILITIES` forwarded verbatim — the ACP - * method table, NOT the engine declaration, and kept under its own - * name in the response for exactly that reason. `agent` is the ACP - * `initialize` mirror: `{version, name, title}` with the endpoint's own - * `"unknown"` / `null` fallbacks, applied here so the route does not - * repeat them. + * `declaration` is the provider's 14-key capability object FORWARDED BY + * IDENTITY — not a copy, not a re-projection. A copy would be a second + * thing that can drift from the reviewed declaration, which is the whole + * failure this endpoint had before M3-B4. + * + * `unavailable` is the DERIVED roll-up (`summarizeUnavailableCapabilities`) + * and is the one field here that is not the declaration itself: a + * `none` key means hide the entry point, a `partial` key means hide or + * disable exactly the listed sub-actions (design §4.2). It is kept + * because it is the shape the capability-driven UI renders from, and a + * consumer should not have to re-derive it from a taxonomy that has + * three levels and two optional fields. + * + * `agent` is the ACP `initialize` mirror: `{version, name, title}` with + * the endpoint's own `"unknown"` / `null` fallbacks, applied here so + * the route does not repeat them. * * @param {object} [options] * @param {string} [options.endpoint] Endpoint key for the declaration * check; defaults to `/api/protocol/capabilities`. * @param {string} [options.transport] Transport override; defaults to the * active `MCODE_WEBUI_TRANSPORT`. - * @returns {Promise<{engine: EngineCapabilityView, agent: {version: string, name: string|null, title: string|null}, wire: object, source: "declaration", gate: object, transport: string}>} + * @returns {Promise<{declaration: object, unavailable: {none: string[], partial: Array<{key: string, missing: string[]}>}, provider: string, providerFor: "transport"|"default", engineTransport: string, agent: {version: string, name: string|null, title: string|null}, source: "declaration", gate: object, transport: string}>} */ export async function readEngineCapabilityView(options = {}) { const endpoint = options.endpoint || "GET /api/protocol/capabilities"; - const [rpc, acp, config, capabilities] = await Promise.all([ - import("../lib/mcode-rpc.js"), + const [acp, config, capabilities] = await Promise.all([ import("../lib/acp-client.js"), import("../lib/config.js"), import("./capabilities.js"), @@ -196,19 +212,21 @@ export async function readEngineCapabilityView(options = {}) { // `serverInfo`); the mirror is empty until something attaches. const agentInfo = acp.getMcodeServerInfo(); return { - engine: { - provider: provider.id, - providerFor, - transport: provider.transport, - capabilities: provider.capabilities, - unavailable: capabilities.summarizeUnavailableCapabilities(provider.capabilities), - }, + declaration: provider.capabilities, + unavailable: capabilities.summarizeUnavailableCapabilities(provider.capabilities), + provider: provider.id, + providerFor, + // The PROVIDER's wire form, named apart from the ambient + // `transport` the read ran under: under the default `acp` transport + // the declaration served belongs to a `runtime` provider, and + // collapsing the two into one field would say exactly the thing + // `providerFor` exists to prevent. + engineTransport: provider.transport, agent: { version: (agentInfo && agentInfo.version) || "unknown", name: (agentInfo && agentInfo.name) || null, title: (agentInfo && agentInfo.title) || null, }, - wire: rpc.MCODE_ACP_CAPABILITIES, source: "declaration", gate, transport, diff --git a/packages/webui/server/routes/protocol.js b/packages/webui/server/routes/protocol.js index 4368a404..10d49047 100644 --- a/packages/webui/server/routes/protocol.js +++ b/packages/webui/server/routes/protocol.js @@ -272,31 +272,40 @@ export async function handleListSessions(req, res, ctx) { // M3-B4: the handler no longer names `lib/mcode-rpc.js` or // `lib/acp-client.js` — both moved behind // `engine/capability-reads.js#readEngineCapabilityView`, which also -// resolves the provider whose DECLARED 14-key surface and its -// degradation summary this endpoint now carries under `engine`. +// resolves the provider whose DECLARED surface this endpoint now serves. // -// `capabilities` itself is unchanged: it is still `MCODE_ACP_CAPABILITIES`, -// the ACP JSON-RPC method table the frontend's control map is keyed on. -// The 14 matrix keys answer a different question ("does the engine have -// this capability at all"), so the view is additive rather than a -// replacement — `docs/API.md` documents both, in both languages. -// `providerFor` says whether the declaration came from the active -// transport's provider or from the default provider standing in for a -// transport no provider claims yet (M4), so a consumer never mistakes a -// standing-in declaration for the connected engine's. +// `capabilities` IS the 14-key engine-capabilities view: a replacement +// for the `MCODE_ACP_CAPABILITIES` ACP wire table this field used to +// carry, approved as an endpoint contract change. The four +// `capabilities*` keys form one group — the declaration, which provider +// answered, how it was chosen, and the derived degradation roll-up — and +// the declaration appears exactly once. +// +// `capabilitiesProviderFor` says whether the declaration came from the +// active transport's provider or from the default provider standing in +// for a transport no provider claims yet (M4), so a consumer never +// mistakes a standing-in declaration for the connected engine's. // // `notes` stays here: it is prose about webui's own routes, not an // engine read, and the facade has no business restating it. // ============================================================ export async function handleCapabilities(_req, res) { - const { engine, agent, wire } = await readEngineCapabilityView(); + const { + declaration, + unavailable, + provider, + providerFor, + agent, + } = await readEngineCapabilityView(); return respond(res, 200, { ok: true, mcodeVersion: agent.version, mcodeName: agent.name, mcodeTitle: agent.title, - capabilities: wire, - engine, + capabilities: declaration, + capabilitiesProvider: provider, + capabilitiesProviderFor: providerFor, + capabilitiesUnavailable: unavailable, notes: { set_mode: "Takes a modeId from the session's availableModes.", set_config_option: diff --git a/packages/webui/test/lib/engine/capability-reads.test.js b/packages/webui/test/lib/engine/capability-reads.test.js index 9aa3b9b4..41041619 100644 --- a/packages/webui/test/lib/engine/capability-reads.test.js +++ b/packages/webui/test/lib/engine/capability-reads.test.js @@ -2,26 +2,30 @@ // // M3-B4: the capability-declaration read's engine facade (#73). // -// This is the one endpoint in the migration that CHANGES its response, -// so the tests here are mostly about pinning exactly how much changed -// and why the rest did not: +// This is the one endpoint in the migration whose RESPONSE CONTRACT +// changes, by explicit decision: `capabilities` used to be +// `MCODE_ACP_CAPABILITIES`, a hand-maintained flat `{method: boolean}` +// table of the ACP JSON-RPC surface, and it is now the engine's +// DECLARED 14-key capability object. So the tests here pin the +// replacement, not an absence of change: // -// 1. THE ADDITIVE CHANGE. #73 gains one key, `engine`, carrying the -// engine-capabilities view. Every key that existed before keeps -// its exact name, position and value — the ACP wire table stays -// under `capabilities`, the `initialize` mirror stays under -// `mcodeVersion` / `mcodeName` / `mcodeTitle`, and `notes` stays -// last. Section 4 asserts the full key order of the response, so a -// future "let me just replace the wire table with the 14 keys" -// cannot land without a reviewer seeing the test fail. +// 1. THE REPLACEMENT. `capabilities` carries the declaration, forwarded +// by identity, and the twelve old accessors are asserted GONE — a +// consumer that still reads `capabilities.set_mode` must get +// `undefined` and fail loudly rather than silently receive a +// truthy object field. The declaration must appear exactly once in +// the serialised body: the `engine` key an earlier shape of this +// batch shipped was removed precisely because it carried the same +// 14 keys a second time. `mcodeVersion` / `mcodeName` / +// `mcodeTitle` and `notes` are untouched, and `notes` stays last. // -// 2. `providerFor`. The view must say whether the declaration came -// from the ACTIVE transport's provider or from the default -// provider standing in for a transport nothing claims yet (M4). -// A capability-detection endpoint that reported a standing-in -// declaration as though it were the connected engine's is the -// same lie B1 declined for `/api/health` — and this is the one -// endpoint where it is most tempting, because the fallback is +// 2. `capabilitiesProviderFor`. The response must say whether the +// declaration came from the ACTIVE transport's provider or from the +// default provider standing in for a transport nothing claims yet +// (M4). A capability-detection endpoint that reported a +// standing-in declaration as though it were the connected engine's +// is the same lie B1 declined for `/api/health` — and this is the +// one endpoint where it is most tempting, because the fallback is // silent and always succeeds. // // 3. THE EMPTY-DECLARATION RULE. #73 must never answer an empty @@ -41,7 +45,7 @@ import { test, describe, after } from "node:test"; import assert from "node:assert/strict"; -import { setupMocks, absPath, registerAcpMock, registerRpcMock } from "../../helpers/_setup.js"; +import { setupMocks, absPath, registerAcpMock } from "../../helpers/_setup.js"; const { CAPABILITY_READ_ENDPOINTS, @@ -58,11 +62,9 @@ const RUNTIME = "runtime"; const ACP = "acp"; const AGENT_INFO = { name: "mcode", title: "Mcode", version: "0.5.5" }; -const WIRE = { set_mode: true, set_config_option: true, cancel: true, activate: true }; after(() => { registerAcpMock({ getMcodeServerInfo: () => null }); - registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); }); // --------------------------------------------------------------------------- @@ -146,26 +148,34 @@ describe("resolveCapabilityReadProvider — it never returns nothing", () => { // --------------------------------------------------------------------------- describe("readEngineCapabilityView", () => { - test("the view is the engine-capabilities payload /api/engine-capabilities serves", async (t) => { - // Same four facts, same source objects. If the two endpoints ever + test("the read is the engine-capabilities payload /api/engine-capabilities serves", async (t) => { + // Same declaration, same source object. If the two endpoints ever // answer different declarations there are two truths in webui, and // this assertion is what stops that. await setupMocks(t, { acp: { getMcodeServerInfo: () => AGENT_INFO } }); - registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); const read = await readEngineCapabilityView({ transport: RUNTIME }); - assert.deepEqual(Object.keys(read.engine), [ + // The read's key set, asserted exactly: the ACP wire table is gone + // from this layer, and a `wire` field reappearing here would put a + // second "what can the engine do" answer back in the facade. + assert.deepEqual(Object.keys(read), [ + "declaration", + "unavailable", "provider", "providerFor", + "engineTransport", + "agent", + "source", + "gate", "transport", - "capabilities", - "unavailable", ]); - assert.deepEqual(Object.keys(read.engine.capabilities), [...ENGINE_CAPABILITY_KEYS]); - assert.equal(read.engine.capabilities, LOCAL_RUNTIME_V2_CAPABILITIES); + assert.deepEqual(Object.keys(read.declaration), [...ENGINE_CAPABILITY_KEYS]); + assert.equal(read.declaration, LOCAL_RUNTIME_V2_CAPABILITIES); assert.deepEqual( - read.engine.unavailable, + read.unavailable, summarizeUnavailableCapabilities(LOCAL_RUNTIME_V2_CAPABILITIES), ); + assert.equal(read.provider, "local-runtime-v2"); + assert.equal(read.engineTransport, "runtime"); assert.equal(read.source, "declaration"); assert.equal(read.transport, RUNTIME); }); @@ -183,8 +193,7 @@ describe("readEngineCapabilityView", () => { for (const [info, expected] of AGENT_CASES) { test(`agentInfo ${JSON.stringify(info)} → ${JSON.stringify(expected)}`, async (t) => { await setupMocks(t, { acp: { getMcodeServerInfo: () => info } }); - registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); - const read = await readEngineCapabilityView({ transport: RUNTIME }); + const read = await readEngineCapabilityView({ transport: RUNTIME }); assert.deepEqual(read.agent, expected); assert.deepEqual(Object.keys(read.agent), ["version", "name", "title"]); }); @@ -204,12 +213,11 @@ describe("readEngineCapabilityView", () => { for (const [transport, expected] of PROVIDER_FOR) { test(`the view reports providerFor=${expected} on transport ${JSON.stringify(transport)}`, async (t) => { await setupMocks(t, { acp: {} }); - registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); - const read = await readEngineCapabilityView({ transport }); - assert.equal(read.engine.providerFor, expected); + const read = await readEngineCapabilityView({ transport }); + assert.equal(read.providerFor, expected); // And the two halves cannot disagree: `providerFor: "transport"` // with a provider the transport does not own is the lie. - assert.equal(read.engine.providerFor === "transport", transport === RUNTIME); + assert.equal(read.providerFor === "transport", transport === RUNTIME); }); } @@ -221,29 +229,55 @@ describe("readEngineCapabilityView", () => { // expected value depends on the ambient env is a test that is green // on one transport and red on the other. await setupMocks(t, { acp: {} }); - registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); const { MCODE_WEBUI_TRANSPORT } = await import(absPath("lib/config.js")); const read = await readEngineCapabilityView({ transport: "" }); assert.equal(read.transport, MCODE_WEBUI_TRANSPORT); assert.equal( - read.engine.providerFor, + read.providerFor, MCODE_WEBUI_TRANSPORT === RUNTIME ? "transport" : "default", ); }); - test("the ACP wire table is forwarded by REFERENCE, not copied", async (t) => { - // A copy would be a second answer to "which ACP methods exist", - // freezable in a way the source is not. Identity pins the - // forwarding. + test("the declaration is forwarded by IDENTITY, and there is no ACP wire field", async (t) => { + // A copy would be a second thing that can drift from the reviewed + // declaration, which is the failure this endpoint had before M3-B4. + // Identity pins the forwarding; the absence assertion pins the + // replacement, so re-adding `MCODE_ACP_CAPABILITIES` anywhere in + // this layer is a red bar rather than a silent second answer. await setupMocks(t, { acp: {} }); - registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); const read = await readEngineCapabilityView({ transport: RUNTIME }); - assert.equal(read.wire, WIRE); + assert.equal(read.declaration, LOCAL_RUNTIME_V2_CAPABILITIES); + assert.equal("wire" in read, false); + // And the facade must not even REACH for the rpc module any more: + // the field it used to carry is the only reason it did. Asserted on + // the SOURCE, because an unused import is behaviourally inert and no + // behavioural test can tell it apart from a clean module — but it + // would put `lib/mcode-rpc.js` (and its `acp.mjs` / settings chain) + // back on the lazy-import path of a boot-reachable module for + // nothing. A static tripwire is the honest instrument here. + const { readFileSync } = await import("node:fs"); + const { fileURLToPath } = await import("node:url"); + const source = readFileSync( + fileURLToPath(new URL(absPath("engine/capability-reads.js"))), + "utf8", + ); + // Matched on the IMPORT FORM, not the bare file name: this module's + // header deliberately names `lib/mcode-rpc.js` in prose (the debt + // note, the boot-path note), and a tripwire that fired on the prose + // would be a tripwire nobody could satisfy. + assert.equal( + /\bimport\s*\(?\s*["'][^"']*lib\/mcode-rpc\.js/.test(source), + false, + "capability-reads.js must not import lib/mcode-rpc.js — the ACP wire table is no longer part of this read", + ); + // The constant itself is untouched; it is simply unconsumed (see + // the KNOWN DEBT note in the module header). + const rpc = await import(absPath("lib/mcode-rpc.js")); + assert.equal(typeof rpc.MCODE_ACP_CAPABILITIES, "object"); }); test("the gate is evaluated and reported, and never blocks the read", async (t) => { await setupMocks(t, { acp: {} }); - registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); // Every transport, including one no provider claims. A read that // gated would throw here; a read that skipped the check entirely // would have no `gate` field at all. @@ -256,7 +290,6 @@ describe("readEngineCapabilityView", () => { test("an unknown endpoint key is a plain Error, not 501 material", async (t) => { await setupMocks(t, { acp: {} }); - registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); await assert.rejects( () => readEngineCapabilityView({ endpoint: "GET /api/nope", transport: RUNTIME }), (err) => { @@ -268,10 +301,10 @@ describe("readEngineCapabilityView", () => { }); // --------------------------------------------------------------------------- -// 4. The route — the additive change, pinned key by key +// 4. The route — the REPLACEMENT, pinned key by key // --------------------------------------------------------------------------- -describe("handleCapabilities — one key added, nothing else touched", () => { +describe("handleCapabilities — capabilities is the engine-capabilities view", () => { let bust = 0; const loadRoute = async () => import(`${absPath("routes/protocol.js")}?bust=${bust++}`); @@ -303,25 +336,26 @@ describe("handleCapabilities — one key added, nothing else touched", () => { }; } + const DECLARATION = { sessionCrud: { level: "full" } }; + const UNAVAILABLE = { none: [], partial: [] }; const VIEW = { - engine: { - provider: "local-runtime-v2", - providerFor: "transport", - transport: "runtime", - capabilities: { sessionCrud: { level: "full" } }, - unavailable: { none: [], partial: [] }, - }, + declaration: DECLARATION, + unavailable: UNAVAILABLE, + provider: "local-runtime-v2", + providerFor: "transport", + engineTransport: "runtime", agent: { version: "0.5.5", name: "mcode", title: "Mcode" }, - wire: WIRE, }; - - test("the response key order is the endpoint's, with `engine` inserted once", async (t) => { - // This is the assertion that makes "we only added a key" a fact - // rather than a claim. The order is the endpoint's, `engine` sits - // directly after the wire table it complements, and `notes` stays - // last. + const stub = () => ({ ...VIEW, source: "declaration", gate: {}, transport: RUNTIME }); + + test("the response key order is the endpoint's, in four `capabilities*` siblings", async (t) => { + // The four `capabilities*` keys form one group — declaration, which + // provider answered, how it was chosen, the derived roll-up — and + // `notes` stays last. A route that nested them under an `engine` + // key, or that ordered them differently, is a contract change the + // key-set assertion catches. await setupMocks(t, { acp: {} }); - mockFacade(t, { readEngineCapabilityView: async () => ({ ...VIEW, source: "declaration", gate: {}, transport: RUNTIME }) }); + mockFacade(t, { readEngineCapabilityView: async () => stub() }); const route = await loadRoute(); const res = mkRes(); await route.handleCapabilities(null, res); @@ -333,22 +367,39 @@ describe("handleCapabilities — one key added, nothing else touched", () => { "mcodeName", "mcodeTitle", "capabilities", - "engine", + "capabilitiesProvider", + "capabilitiesProviderFor", + "capabilitiesUnavailable", "notes", ]); }); - test("every pre-existing key keeps its exact value", async (t) => { + test("`capabilities` IS the 14-key declaration, and the ACP wire table is gone", async (t) => { await setupMocks(t, { acp: {} }); - mockFacade(t, { readEngineCapabilityView: async () => ({ ...VIEW, source: "declaration", gate: {}, transport: RUNTIME }) }); + mockFacade(t, { readEngineCapabilityView: async () => stub() }); const route = await loadRoute(); const res = mkRes(); await route.handleCapabilities(null, res); const body = JSON.parse(res.written[1].body); assert.equal(body.ok, true); - // The ACP wire table is still the ACP wire table — the 14 matrix - // keys did NOT replace it. - assert.deepEqual(body.capabilities, WIRE); + // The declared taxonomy replaced the flat `{method: boolean}` one. + // The old accessors are asserted ABSENT: a consumer that still read + // `capabilities.set_mode` must get `undefined` and fail loudly, not + // silently receive a truthy object field. + for (const gone of ["set_mode", "set_config_option", "cancel", "activate", "fork", "resume", "delete", "load", "close", "list", "new", "prompt"]) { + assert.equal(gone in body.capabilities, false, `capabilities.${gone} must be gone`); + } + // The four group members, each forwarded as the facade gave them. + // `deepEqual`, not identity: the body has been through + // `JSON.parse`, so reference identity is gone by construction — the + // identity assertion that actually matters (the facade forwarding + // the reviewed declaration rather than a copy) lives in section 3, + // one layer below the JSON. + assert.deepEqual(body.capabilities, DECLARATION); + assert.equal(body.capabilitiesProvider, "local-runtime-v2"); + assert.equal(body.capabilitiesProviderFor, "transport"); + assert.deepEqual(body.capabilitiesUnavailable, UNAVAILABLE); + // The `initialize` mirror is untouched by all of this. assert.equal(body.mcodeVersion, "0.5.5"); assert.equal(body.mcodeName, "mcode"); assert.equal(body.mcodeTitle, "Mcode"); @@ -357,20 +408,43 @@ describe("handleCapabilities — one key added, nothing else touched", () => { assert.deepEqual(Object.keys(body.notes), ["set_mode", "set_config_option", "cancel", "activate", "fork"]); }); - test("the whole view is carried, and the route adds nothing to it", async (t) => { + test("the declaration appears EXACTLY ONCE in the serialised body", async (t) => { + // The reason the `engine` key this batch first shipped was removed: + // with the declaration already under `capabilities`, an `engine` + // block carrying it again would put the same 14 keys in the + // response twice, and a consumer could not tell which one is the + // contract. This counts them structurally, not textually. + await setupMocks(t, { acp: {} }); + mockFacade(t, { readEngineCapabilityView: async () => stub() }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCapabilities(null, res); + const body = JSON.parse(res.written[1].body); + const asJson = JSON.stringify(DECLARATION); + const carriers = Object.entries(body).filter(([, v]) => JSON.stringify(v) === asJson); + assert.deepEqual(carriers.map(([k]) => k), ["capabilities"]); + // And no nested key repeats it either: one declaration, one home. + assert.equal(JSON.stringify(body).split(asJson).length - 1, 1); + assert.equal("engine" in body, false); + }); + + test("the route adds nothing to the view and leaks none of its bookkeeping", async (t) => { await setupMocks(t, { acp: {} }); - mockFacade(t, { readEngineCapabilityView: async () => ({ ...VIEW, source: "declaration", gate: {}, transport: RUNTIME }) }); + mockFacade(t, { readEngineCapabilityView: async () => stub() }); const route = await loadRoute(); const res = mkRes(); await route.handleCapabilities(null, res); const body = JSON.parse(res.written[1].body); - // Identity, not equality: a route that re-projected the view would - // be a second place for the 14 keys to be reshaped. - assert.deepEqual(body.engine, VIEW.engine); - // And the facade's own bookkeeping (`source`, `gate`, `transport`) - // stays INSIDE the facade — it is diagnostic vocabulary, not part - // of this endpoint's contract. - for (const key of ["source", "gate"]) { + // A route that re-projected either half would be a second place for + // the taxonomy to be reshaped; `deepEqual` is the strongest + // statement available after `JSON.parse`, and section 3 pins the + // reference identity one layer down. + assert.deepEqual(body.capabilities, VIEW.declaration); + assert.deepEqual(body.capabilitiesUnavailable, VIEW.unavailable); + // The facade's own bookkeeping (`source`, `gate`, the ambient + // `transport`, the provider's `engineTransport`) is diagnostic + // vocabulary, not part of this endpoint's contract. + for (const key of ["source", "gate", "engineTransport"]) { assert.equal(key in body, false, `${key} leaked into the response`); } }); @@ -448,12 +522,11 @@ describe("handleCapabilities — one key added, nothing else touched", () => { // route to the REAL facade, so the body carries the actual // registered declaration rather than the fixture's. await setupMocks(t, { acp: { getMcodeServerInfo: () => AGENT_INFO } }); - registerRpcMock({ MCODE_ACP_CAPABILITIES: WIRE }); const route = await loadRoute(); const res = mkRes(); await route.handleCapabilities(null, res); const body = JSON.parse(res.written[1].body); - assert.equal(body.engine.provider, "local-runtime-v2"); + assert.equal(body.capabilitiesProvider, "local-runtime-v2"); // The real view must SAY whether it is standing in. Under the // default `acp` transport that is `"default"`; reporting // `"transport"` there would be the one lie this endpoint cannot @@ -463,15 +536,16 @@ describe("handleCapabilities — one key added, nothing else touched", () => { // both gate legs. const { MCODE_WEBUI_TRANSPORT } = await import(absPath("lib/config.js")); assert.equal( - body.engine.providerFor, + body.capabilitiesProviderFor, MCODE_WEBUI_TRANSPORT === "runtime" ? "transport" : "default", ); - assert.equal(body.engine.transport, "runtime"); // `setupMocks`'s acp holder is process-global and an earlier case // left the agent mirror in it, so the version here is the real // `initialize` mirror's, not the fixture's. assert.equal(body.mcodeVersion, "0.5.5"); - assert.deepEqual(Object.keys(body.engine.capabilities), [...ENGINE_CAPABILITY_KEYS]); - assert.equal(body.engine.unavailable.none.length >= 1, true); + // And the declaration served is the REAL reviewed one, key for key. + assert.deepEqual(Object.keys(body.capabilities), [...ENGINE_CAPABILITY_KEYS]); + assert.deepEqual(body.capabilities, LOCAL_RUNTIME_V2_CAPABILITIES); + assert.equal(body.capabilitiesUnavailable.none.length >= 1, true); }); }); diff --git a/packages/webui/test/lib/mcode-rpc.check.mjs b/packages/webui/test/lib/mcode-rpc.check.mjs index a777ad7a..c6abeaed 100644 --- a/packages/webui/test/lib/mcode-rpc.check.mjs +++ b/packages/webui/test/lib/mcode-rpc.check.mjs @@ -3,9 +3,17 @@ // // Why this test exists: mcode-rpc.js is the clean wrapper around mcode 0.1.5 // acp JSON-RPC. PERMISSION_MODES + mcodePermissionToWebui are the enum used -// by routes/model.js. MCODE_ACP_CAPABILITIES drives the capability detection -// in routes/protocol.js. Bugs here = wrong permission labels shown to user -// or capability detection thinks mcode supports methods it doesn't. +// by routes/model.js. Bugs here = wrong permission labels shown to user. +// +// MCODE_ACP_CAPABILITIES is KNOWN DEBT as of M3-B4: `GET +// /api/protocol/capabilities` used to serve this table under `capabilities` +// and now serves the engine's DECLARED 14-key capability object instead +// (a user-authorised endpoint contract change — see +// engine/capability-reads.js). The constant is still exported and still +// pinned here, because it remains a true statement about the ENGINE's +// ACP surface and `docs/CAPABILITIES.md` cites it as one. It has no +// webui consumer left; deleting it is a separate dead-code decision, not +// a side effect of the replacement. // // Test strategy: NO setupMocks. We import the REAL mcode-rpc.js so we test // the actual exports. We only test the safe-to-call functions: From 2baf05184ea40e7c507ebb8a03edaa59b2a32732 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Fri, 2 Oct 2026 23:42:31 +0800 Subject: [PATCH 15/21] fix(webui): stop two B4 comments describing behaviour the code no longer has MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found by the dual-axis code review (Standards + Spec) before B4 merges. No logic changes; both were documentation lying about the code next to it. `engine/index.js` still described #73 as an additive change — "the response body gains a key (`engine`, the engine-capabilities view) … additive rather than a replacement". The second B4 commit made it a REPLACEMENT and deleted the `engine` key, so the facade's own export table was the one place still telling a reader the opposite of what the endpoint does. It now states the replacement, why the `engine` key was removed rather than kept, and what survived from it (the provenance keys and the derived roll-up). `docs/ARCHITECTURE.md` and its zh-CN twin called `engine/account-reads.js`'s read **synchronous**. It is `async` — `readEngineAccount` awaits a `Promise.all` of dynamic imports — and the boot-path note the row pointed at describes model-reads, not this module. The same two documents already listed account-reads correctly under the `await import()` rule a few paragraphs down, so the file contradicted itself in two languages at once. Both rows now say asynchronous and point at the ordinary rule. Also drops a dead `assertEngineCapability` import from `engine/model-reads.js`: the soft gate inspects the declaration inline, so the throwing helper was never called and its presence read as if the soft path could still throw. Replaced by a comment saying why it is absent, so the next reader does not "fix" it back in. And the one comment with Chinese embedded mid-sentence (仓库 review 要求注释用英文) is now English; the header parentheticals naming each family (账户读 / 模型目录读 / 能力声明读) stay, as do the quoted product strings — `本地用户` is the real zh-CN value of `userMenu.localUser` and `shell.tsx` cites it the same way. --- packages/webui/docs/ARCHITECTURE.md | 2 +- packages/webui/docs/ARCHITECTURE.zh-CN.md | 2 +- packages/webui/server/engine/index.js | 14 +++++++++++--- packages/webui/server/engine/model-reads.js | 9 +++++++-- 4 files changed, 20 insertions(+), 7 deletions(-) diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index 09a77dc9..7bbc4c1f 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -505,7 +505,7 @@ files, one job each: | `engine/session-tree-reads.js` | The session-tree family's facade call (`readEngineSessionTree`) and the endpoint→capability table `SESSION_TREE_ENDPOINTS` (step M3, batch B2). Gates **hard**: `assertSessionTreeCapability` throws → 501, because the tree is entirely engine data. Forwards to `lib/session-tree.js#getSessionTree`; the assembler is not duplicated | | `engine/session-export.js` | The export family's facade call (`readEngineSessionTranscript`) and the endpoint→capability table `SESSION_EXPORT_ENDPOINTS` (step M3, batch B2). Gates **soft**: `checkSessionExportCapability` reports and never throws, because export's primary source is `sessions.json`, not the engine | | `engine/usage-reads.js` | The usage family's facade calls (`readEngineAccountQuota`, `readEngineSessionUsage`, `readEngineQuotaForecast`), the derived figure `contextUsedTokens`, and the endpoint→capability table `USAGE_READ_ENDPOINTS` (step M3, batch B3). Gates **hard** on the two engine reads and declares **no capability at all** for #19, which touches no engine surface | -| `engine/account-reads.js` | The account family's facade call (`readEngineAccount`) and the endpoint→capability table `ACCOUNT_READ_ENDPOINTS` (step M3, batch B4). Gates **hard** on `authCredentials` · `getAccountStatus` — the same pair and the same provider method as `engine/usage-reads.js`, because #20 and #15/#16 read the same engine projection. Its read is **synchronous**; see the boot-path note below | +| `engine/account-reads.js` | The account family's facade call (`readEngineAccount`) and the endpoint→capability table `ACCOUNT_READ_ENDPOINTS` (step M3, batch B4). Gates **hard** on `authCredentials` · `getAccountStatus` — the same pair and the same provider method as `engine/usage-reads.js`, because #20 and #15/#16 read the same engine projection. Its read is **asynchronous** and it lives under the ordinary `await import()` boot-path rule | | `engine/model-reads.js` | The model-catalogue family's facade call (`readEngineModelCatalogue`), the whole projection as named pure functions (`projectModelCatalogue`, `deriveModelSelection`, `buildModelCataloguePayload`, `catalogueSourceLabel`, `webuiFullModelId`, `providerOfModelId`, `attachContextWindowOptions`, `configOption`), and the endpoint→capability table `MODEL_READ_ENDPOINTS` (step M3, batch B4). Gates **soft**: `checkModelReadCapability` reports and never throws, because the catalogue's primary sources are files webui owns. Its read is **synchronous**, and it is the one engine module **not** re-exported from `engine/index.js` — see the boot-path note below | | `engine/capability-reads.js` | The capability-declaration family's facade call (`readEngineCapabilityView`) and the endpoint→capability table `CAPABILITY_READ_ENDPOINTS` (step M3, batch B4). Declares **no capability for #73** — it IS the declaration endpoint, and gating the gate would let a `none` hide the declaration that says so. It is the only endpoint in the migration whose response CONTRACT changed (`capabilities` is now the 14-key declaration, replacing the ACP wire table) | diff --git a/packages/webui/docs/ARCHITECTURE.zh-CN.md b/packages/webui/docs/ARCHITECTURE.zh-CN.md index 5ffbc835..60242458 100644 --- a/packages/webui/docs/ARCHITECTURE.zh-CN.md +++ b/packages/webui/docs/ARCHITECTURE.zh-CN.md @@ -476,7 +476,7 @@ queued \| done \| stopped`)是投影层产物、不是存储值;webui 不导 | `engine/session-tree-reads.js` | 会话树族的面板调用 `readEngineSessionTree` 与端点→能力对照表 `SESSION_TREE_ENDPOINTS`(迁移步 M3 批次 B2)。**硬门控**:`assertSessionTreeCapability` 抛出 → 501,因为树完全由引擎数据构成。转发到 `lib/session-tree.js#getSessionTree`,树的装配逻辑不复制第二份 | | `engine/session-export.js` | 导出族的面板调用 `readEngineSessionTranscript` 与端点→能力对照表 `SESSION_EXPORT_ENDPOINTS`(迁移步 M3 批次 B2)。**软门控**:`checkSessionExportCapability` 只报告、从不抛出,因为导出的主数据源是 `sessions.json` 而非引擎 | | `engine/usage-reads.js` | 用量族的面板调用(`readEngineAccountQuota`、`readEngineSessionUsage`、`readEngineQuotaForecast`)、派生量 `contextUsedTokens`,与端点→能力对照表 `USAGE_READ_ENDPOINTS`(迁移步 M3 批次 B3)。两个引擎读**硬门控**;#19 **完全不声明能力**,因为它不触达任何引擎面 | -| `engine/account-reads.js` | 账户族的面板调用 `readEngineAccount` 与端点→能力对照表 `ACCOUNT_READ_ENDPOINTS`(迁移步 M3 批次 B4)。**硬门控**,门控在 `authCredentials` · `getAccountStatus`——与 `engine/usage-reads.js` 同一对、同一个 provider 方法,因为 #20 与 #15/#16 读的是同一份引擎投影。它的读是**同步的**,见下面的启动路径说明 | +| `engine/account-reads.js` | 账户族的面板调用 `readEngineAccount` 与端点→能力对照表 `ACCOUNT_READ_ENDPOINTS`(迁移步 M3 批次 B4)。**硬门控**,门控在 `authCredentials` · `getAccountStatus`——与 `engine/usage-reads.js` 同一对、同一个 provider 方法,因为 #20 与 #15/#16 读的是同一份引擎投影。它的读是**异步的**,服从普通的 `await import()` 启动路径纪律 | | `engine/model-reads.js` | 模型目录族的面板调用 `readEngineModelCatalogue`、整套投影的具名纯函数(`projectModelCatalogue`、`deriveModelSelection`、`buildModelCataloguePayload`、`catalogueSourceLabel`、`webuiFullModelId`、`providerOfModelId`、`attachContextWindowOptions`、`configOption`),与端点→能力对照表 `MODEL_READ_ENDPOINTS`(迁移步 M3 批次 B4)。**软门控**:`checkModelReadCapability` 只报告、从不抛出,因为目录的主数据源是 webui 自己拥有的文件。它的读是**同步的**,并且它是唯一一个**没有**从 `engine/index.js` 转发导出的引擎模块——见下面的启动路径说明 | | `engine/capability-reads.js` | 能力声明族的面板调用 `readEngineCapabilityView` 与端点→能力对照表 `CAPABILITY_READ_ENDPOINTS`(迁移步 M3 批次 B4)。#73 **不声明任何能力**——它本身就是声明端点,给门控上门控会让某个 `none` 把声明它的那份声明藏起来。它是本次迁移中唯一一个响应**契约**发生变更的端点(`capabilities` 现在是 14 键声明,顶替了 ACP wire 表) | diff --git a/packages/webui/server/engine/index.js b/packages/webui/server/engine/index.js index 6d390f04..b910f29f 100644 --- a/packages/webui/server/engine/index.js +++ b/packages/webui/server/engine/index.js @@ -150,9 +150,17 @@ export { // The capability-declaration read (step M3, batch B4). Declares NO // capability for #73 — it IS the declaration endpoint, and gating the // gate would let a `none` hide the declaration that says so. It is the -// one endpoint in the migration whose response body gains a key -// (`engine`, the engine-capabilities view); see the module header for -// why that is additive rather than a replacement. +// one endpoint in the migration whose response CONTRACT changed: +// `capabilities` used to carry `MCODE_ACP_CAPABILITIES`, the flat ACP +// wire table, and now carries the provider's DECLARED 14-key object — a +// user-authorised replacement, not an addition. The `engine` key an +// earlier shape of this batch shipped was removed rather than kept, +// because with the declaration already under `capabilities` it would +// have carried the same 14 keys a second time in one response; what +// survives is the provenance (`capabilitiesProvider` / +// `capabilitiesProviderFor`) and the derived `capabilitiesUnavailable`. +// See the module header for the full statement and the debt note on the +// now-unconsumed constant. export { CAPABILITY_READ_ENDPOINTS, checkCapabilityReadCapability, diff --git a/packages/webui/server/engine/model-reads.js b/packages/webui/server/engine/model-reads.js index 472d8185..c7eb163e 100644 --- a/packages/webui/server/engine/model-reads.js +++ b/packages/webui/server/engine/model-reads.js @@ -102,7 +102,11 @@ import { } from "../lib/engine-catalogue.js"; import { getBuiltinModelsFromMcode } from "../lib/models.js"; import { loadProvidersConfig } from "../lib/providers-config.js"; -import { assertEngineCapability } from "./capabilities.js"; +// `assertEngineCapability` is deliberately NOT imported: this family's +// gate is soft, so it INSPECTS the declaration (`checkModelReadCapability` +// below) and reports what it found rather than delegating the verdict to +// the throwing helper. Importing it here would be a dead import that +// reads as if the soft path could still throw. import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; /** @@ -671,7 +675,8 @@ function defaultParseWireStub() { * route's handler signature is part of its contract: `app.js#invokeHandler` * accepts both shapes, but a caller that does not await gets a * half-written response from an async handler and a complete one from a - * sync handler, and this batch is a收编, not a scheduling change. + * sync handler, and this batch is an absorption, not a scheduling + * change. * * Every source is re-read on every call, exactly as before: editing * `models.json`, `~/.mcode-webui/providers.json` or the engine's From edf2b1e84035f2713c9109a0839dedfe4c939fc4 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Sat, 3 Oct 2026 00:59:42 +0800 Subject: [PATCH 16/21] feat(webui): move the session write family behind the engine facade MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit M3 batch B5: #7 DELETE /api/sessions/:id, #4 POST /api/sessions/rename and #6 POST /api/sessions/cleanup-orphans stop driving the store, the caches and the engine's own local_runtime_* tables from the route. They ask engine/session-writes.js instead, so the load -> resolve -> authorize -> intent-audit -> mutate ordering — and the resurrection guard inside it — becomes named, testable code rather than a two-line helper a route could call out of order. #7 and #6 gate HARD on sessionCrud.deleteSession, because the rows they destroy are the engine's own; #4 declares no capability at all, because a rename writes webui's own store and touches no engine surface. The policy is decided by who owns the rows the write destroys, which is a different question from the read families' and does not have the same answer twice in a row here. The facade exposes a plan/commit pair rather than one deleteSession(), so the write-ahead audit still lands between "know what the user asked to delete" and "delete it". Response bodies are built in the facade once, which is what lets the #6 and #7 dryRun shapes be pinned byte-for-byte by unit tests. No status code, response body or error code changes. The 32-table delete SQL stays in lib/mcode-session-delete.js and is reached by dynamic import; acp-client.js and four test files bind to that specifier, so collecting it is a later batch's job. Recorded as KNOWN DEBT, along with rename writing a webui-side label only, and delete not detecting an in-flight session. --- packages/webui/docs/ARCHITECTURE.md | 125 +- packages/webui/server/engine/index.js | 48 +- .../webui/server/engine/session-writes.js | 936 +++++++++ packages/webui/server/routes/sessions.js | 466 ++--- packages/webui/test/helpers/_setup.js | 35 + .../test/lib/engine/session-writes.test.js | 1733 +++++++++++++++++ release/public-source.json | 2 + scripts/test-tmp-leak.check.mjs | 1 + 8 files changed, 3051 insertions(+), 295 deletions(-) create mode 100644 packages/webui/server/engine/session-writes.js create mode 100644 packages/webui/test/lib/engine/session-writes.test.js diff --git a/packages/webui/docs/ARCHITECTURE.md b/packages/webui/docs/ARCHITECTURE.md index 7bbc4c1f..bcd4a159 100644 --- a/packages/webui/docs/ARCHITECTURE.md +++ b/packages/webui/docs/ARCHITECTURE.md @@ -489,8 +489,8 @@ not import it but adopts the same shape. Unknown future statuses render as ### `engine/` (capability declarations + the local-runtime-v2 host) The engine abstraction lives at `server/engine/` (engine-abstraction -batch B1; migration state M1, plus M3 batches B0, B1, B2, B3 and B4). Fourteen -files, one job each: +batch B1; migration state M1, plus M3 batches B0, B1, B2, B3, B4 and B5). +Fifteen files, one job each: | File | Owns | | --- | --- | @@ -508,6 +508,7 @@ files, one job each: | `engine/account-reads.js` | The account family's facade call (`readEngineAccount`) and the endpoint→capability table `ACCOUNT_READ_ENDPOINTS` (step M3, batch B4). Gates **hard** on `authCredentials` · `getAccountStatus` — the same pair and the same provider method as `engine/usage-reads.js`, because #20 and #15/#16 read the same engine projection. Its read is **asynchronous** and it lives under the ordinary `await import()` boot-path rule | | `engine/model-reads.js` | The model-catalogue family's facade call (`readEngineModelCatalogue`), the whole projection as named pure functions (`projectModelCatalogue`, `deriveModelSelection`, `buildModelCataloguePayload`, `catalogueSourceLabel`, `webuiFullModelId`, `providerOfModelId`, `attachContextWindowOptions`, `configOption`), and the endpoint→capability table `MODEL_READ_ENDPOINTS` (step M3, batch B4). Gates **soft**: `checkModelReadCapability` reports and never throws, because the catalogue's primary sources are files webui owns. Its read is **synchronous**, and it is the one engine module **not** re-exported from `engine/index.js` — see the boot-path note below | | `engine/capability-reads.js` | The capability-declaration family's facade call (`readEngineCapabilityView`) and the endpoint→capability table `CAPABILITY_READ_ENDPOINTS` (step M3, batch B4). Declares **no capability for #73** — it IS the declaration endpoint, and gating the gate would let a `none` hide the declaration that says so. It is the only endpoint in the migration whose response CONTRACT changed (`capabilities` is now the 14-key declaration, replacing the ACP wire table) | +| `engine/session-writes.js` | The session WRITE family's facade calls (`planEngineSessionDelete`, `commitEngineSessionDelete`, `commitEngineOrphanSessionDelete`, `previewEngineSessionDelete`, `applyEngineSessionRename`, `readOrphanSessionWriteIds`), the pure derivations they are built from (`resolveSessionTarget`, `isMcodeSessionId`, `isOrphanSessionRecord`, `selectOrphanSessionIds`, the two fan-out predicates, the per-client state resets), and the endpoint→capability table `SESSION_WRITE_ENDPOINTS` (step M3, batch B5). Gates **hard** on `sessionCrud` · `deleteSession` for #7 and #6, and declares **no capability at all** for #4. Forwards the 32-table SQL to `lib/mcode-session-delete.js` rather than moving it — see the write-path section below | Routes take the host from the facade and never from `lib/acp-client.js`: `routes/plugins.js` and `routes/turn-diff.js` call @@ -587,13 +588,19 @@ declaration and construction were split). `test/lib/engine/host-facade.test.js` enforces it against the real module graph rather than against source text. `engine/session-reads.js`, `engine/session-tree-reads.js`, `engine/session-export.js`, `engine/usage-reads.js`, -`engine/account-reads.js` and `engine/capability-reads.js` all live under the +`engine/account-reads.js`, `engine/capability-reads.js` and +`engine/session-writes.js` all live under the same rule: their static imports are `engine/capabilities.js` and `engine/index.js` only, and every heavier dependency — `lib/acp-client.js`, `lib/config.js`, `lib/session-tree.js`, `lib/transcript.js`, `lib/usage.js`, `lib/mavis-usage.js`, `lib/quota-forecast.js`, `lib/mcode-rpc.js` — is reached through -`await import()` inside the functions. +`await import()` inside the functions. `engine/session-writes.js` adds +`node:fs` at module scope (a builtin, and `engine/usage-reads.js` +already does the same) and reaches `lib/sessions.js`, +`lib/mcode-session-delete.js`, `lib/state-bus.js` and +`lib/config.js` dynamically — all six of its storage dependencies, which +is what lets it be re-exported from `engine/index.js` at all. `engine/model-reads.js` is the one deliberate exception, and it deviates on **both** sides of the import. Its four sources — `lib/config.js`, @@ -998,6 +1005,116 @@ same event is safe. The server uses an at-most-once delivery model (SSE drops on disconnect → no retry), which the client handles by fetching `/api/state` on reconnect. +#### Which endpoints write through the facade (step M3, batch B5) + +Batch B5 is the first family in the migration whose endpoints **destroy** +data rather than read it, and that changes what the gate question is +asking. For a read, hard or soft is decided by "is the data the engine's +or webui's". For a write it is decided by **who owns the rows the write +destroys** — and in this family that question does not have the same +answer twice in a row. + +| Endpoint | Capability · sub-item | Enforcement | Value source | +| --- | --- | --- | --- | +| `DELETE /api/sessions/:id` (#7) | `sessionCrud` · `deleteSession` | hard — 501 | the webui session store, the in-memory ACP session cache, the sidebar tree cache, and the engine's own `local_runtime_*` rows via `lib/mcode-session-delete.js` | +| `POST /api/sessions/rename` (#4) | none of the 14 keys | none — the gate is a reported no-op | webui's own session store, and nothing else. The engine's title is not written | +| `POST /api/sessions/cleanup-orphans` (#6) | `sessionCrud` · `deleteSession` | hard — 501 | the same store, plus each selected id delegated to #7, so it reaches the same engine rows | + +**Why #7 and #6 gate hard.** Both destroy rows in the engine's own +`local_runtime_*` tables, and there is no webui-side copy of a transcript +that survives: once those rows are gone, the conversation is gone. A +provider that declares no session deletion genuinely cannot have these +endpoints serve a truthful answer, so 501 is the honest one. #6 +deliberately declares the *same* pair as #7 — the sweep selects webui-side +orphan records, but each selected id goes through #7's real-delete branch, +and a record carrying an `mcodeSessionId` takes the engine's rows with it. +Gating the sweep soft would let a provider that cannot delete engine +sessions reach those tables through a back door, and would also produce a +worse failure than a 501: an authorized destructive sweep that writes its +intent audit event and then fails every single delegated delete. + +**Why #4 declares nothing.** Rename writes `title` / `titleCustom` / +`updatedAt` into webui's own store and touches no engine surface at all. +Its one engine touch is `invalidateSessionTree()` — a cache drop, which is +the read-side consequence of the sidebar projecting titles from the engine, +and that projection is B2's `GET /api/session-tree` with its own gate. +Naming a capability here would be a lie of the kind B3 declined for +`GET /api/usage/forecast`: gating a working endpoint on a declaration +about something it does not depend on. + +This family also deviates from its siblings in one deliberate way: every +row of `SESSION_WRITE_ENDPOINTS` carries the same three keys — +`capability`, `subItem`, `enforcement` — including the row that has no +capability. B3 expressed "no engine surface" as a `null` table entry; +here two of three endpoints *do* cross the seam, and a `null` hole in the +middle of the table reads like "not filled in yet" rather than like a +decision. The gate **descriptor** keeps the six fields every family +returns, plus `enforcement`. + +**The plan/commit split, and why the route did not shrink to nothing.** +#7 is exported as a pair rather than one `deleteSession(options)`: + +1. `planEngineSessionDelete` resolves the id and runs the gate. It + mutates nothing, so it is safe to run *before* the user is asked + anything. +2. `authorize()` and the write-ahead `session.delete.intent` audit happen + **between** the plan and the commit. The intent line has to be durably + recorded before any row is removed, and it records the match kind and + chat length the plan produced. +3. `commitEngineSessionDelete` / `commitEngineOrphanSessionDelete` / + `previewEngineSessionDelete` perform the write and fan-out. + +A facade that owned the whole operation would have had to swallow that +ordering into a callback. The route keeps request parsing, the authorize +modal, the audit ordering and every status code; the facade keeps the +sequencing, the gate and the response bodies. + +**The ordering inside a commit is the feature, and it is asserted as a +sequence.** `test/lib/engine/session-writes.test.js` journals every +mutation and asserts the order, because an end-state assertion cannot see +a resurrected session: + +``` +invalidate-tree → kill-acp-child → drop-cache: → sql: → push: +``` + +The tree cache is dropped *before* the engine write so a concurrent read +cannot repopulate it from the pre-delete database. The ACP child is +stopped *before* the rows are removed, because it holds the session in +memory and rewrites its registry row on its next request — that is the +"deleted session reappears" bug. Only the **one** deleted sid leaves the +cache: invalidating the whole cache empties the sidebar, refills it, and +reads to the user like the delete failed. + +**The 32-table SQL was not moved, and that is recorded rather than +quietly dropped.** The plan for this batch annotated +`lib/mcode-session-delete.js` "delete". It is kept because +`lib/acp-client.js` imports `deleteMcodeSessionFromDb` from it and four +test files bind to that specifier; collecting it means moving those +first. The facade reaches it through `await import()` and issues no SQL +of its own — the same split B3 drew for `lib/mavis-usage.js` and B4 for +`lib/mcode-rpc.js`. A test asserts both halves: the table list is still +32 entries exported from the lib module, and the facade contains no SQL +verb at all. + +**Three things this batch records as known debt instead of deciding:** + +1. The 32-table SQL is still in `lib/mcode-session-delete.js`, for the + consumer reasons above. +2. A rename is a **webui-side label only**. The engine's own title in + `local_runtime_sessions` is untouched while the sidebar tree reads its + titles from the engine, so for an engine-backed session a rename can be + visible in the wrapper list and not in the tree. This is pre-existing + behaviour and the batch did not change it; closing it means deciding + which store is authoritative for a display title, which is a product + call. +3. #7 does not detect "this session is running right now". Deleting an + in-flight session stops the ACP child out from under the turn and then + proceeds. That is the pre-facade behaviour and arguably the correct + one (the user asked), but refusing to delete a running session is a + defensible alternative and the choice is not the batch's to make. A + test pins the semantics that exist so the behaviour is at least stated. + ## 6. Frontend topology ``` diff --git a/packages/webui/server/engine/index.js b/packages/webui/server/engine/index.js index b910f29f..d91751c4 100644 --- a/packages/webui/server/engine/index.js +++ b/packages/webui/server/engine/index.js @@ -36,9 +36,9 @@ // catalogue host itself is now reached through this facade too // (engine/host.js), so the plugins and turn-diff routes no longer name // lib/acp-client.js. M3 batches B1 (#9 #10 #72 #74 #75), B2 (#8 #11), -// B3 (#15 #16 #17 #19) and B4 (#20 #57 #73) done. The rest of M3, then -// M4, will route their consumers through this facade one endpoint -// family at a time. +// B3 (#15 #16 #17 #19), B4 (#20 #57 #73) and B5 (#7 #4 #6) done. The +// rest of M3, then M4, will route their consumers through this facade one +// endpoint family at a time. import { ENGINE_CAPABILITY_KEYS } from "./capabilities.js"; // Declarations only — importing the provider *host-construction* modules @@ -183,6 +183,48 @@ export { readEngineSessionUsage, resolveUsageReadProvider, } from "./usage-reads.js"; +// The session WRITE family (step M3, batch B5): #7 delete, #4 rename, +// #6 cleanup-orphans. Same cycle, same TDZ rule, same reasoning as +// session-reads.js above: session-writes.js reads NOTHING from this +// module at module scope — its `SESSION_WRITE_ENDPOINTS` table is a +// literal and every binding it needs (`getEngineProvider`, +// `DEFAULT_ENGINE_PROVIDER_ID`) is read inside a function body. A new +// top-level `const X = SOMETHING_FROM_INDEX` in session-writes.js breaks +// the re-export exactly as it would in session-reads.js. Its static +// imports are `engine/capabilities.js`, `engine/index.js` and `node:fs` +// (a builtin); all six of its storage dependencies are reached through +// `await import()` inside the functions, so the boot-path rule the other +// families follow holds here too. +// +// Two of its three endpoints gate HARD on `sessionCrud` · `deleteSession` +// — #7 and #6, both because they destroy rows in the engine's own +// `local_runtime_*` tables — and the third, #4, declares NO capability +// because a rename writes webui's own session store and touches no engine +// surface at all. The policy is decided by who owns the rows the write +// destroys, which is a different question from the read families' and +// does not have the same answer twice in a row here. See the module +// header for the full argument and for the known debt this batch records +// rather than settles. +export { + ORPHAN_STALE_MS, + SESSION_WRITE_ENDPOINTS, + applyDeletedSessionToClientState, + applyEngineSessionRename, + applyRenamedSessionToClientState, + assertSessionWriteCapability, + clientMatchesDeletedSession, + clientMatchesRenamedSession, + commitEngineOrphanSessionDelete, + commitEngineSessionDelete, + isMcodeSessionId, + isOrphanSessionRecord, + planEngineSessionDelete, + previewEngineSessionDelete, + readOrphanSessionWriteIds, + resolveSessionTarget, + resolveSessionWriteProvider, + selectOrphanSessionIds, +} from "./session-writes.js"; export { LOCAL_RUNTIME_V2_CAPABILITIES } from "./providers/local-runtime-v2.capabilities.js"; export { TUI_RUNTIME_ADAPTER_CAPABILITIES } from "./providers/tui-runtime-adapter.js"; diff --git a/packages/webui/server/engine/session-writes.js b/packages/webui/server/engine/session-writes.js new file mode 100644 index 00000000..8510328c --- /dev/null +++ b/packages/webui/server/engine/session-writes.js @@ -0,0 +1,936 @@ +// webui/server/engine/session-writes.js +// +// Migration step M3, batch B5: the session WRITE family (会话写族) — the +// three endpoints that change stored state rather than read it: +// +// #7 DELETE /api/sessions/:id — delete a session +// #4 POST /api/sessions/rename — rename a session +// #6 POST /api/sessions/cleanup-orphans — sweep default-named empties +// +// Why a write family needs a facade at all, when a read family is a +// one-liner that forwards. #7 is the only endpoint in the whole migration +// that can DESTROY data the engine owns, and it destroys it three ways +// at once: the engine's own `local_runtime_*` rows, the webui session +// record, and the in-memory caches two readers are assembled from. Three +// facts about that delete are load-bearing and none of them is visible +// at the call site once the route has grown to 270 lines: +// +// 1. THE RESURRECTION GUARD. The long-lived mcode ACP child holds the +// session in memory and rewrites its registry row on the next +// request, so a delete that only removes SQL rows comes BACK. The +// order is the whole mechanism: kill the child → delete the rows → +// drop ONLY the deleted sid from the cache (not the whole cache — +// invalidating everything flashes the sidebar 42 → 16 → 42 and +// reads to the user like the delete failed). A refactor that +// reorders these three steps reintroduces "deleted session +// reappears" without failing any single assertion. +// 2. CACHE INVALIDATION PRECEDES THE ENGINE WRITE. +// `invalidateSessionTree()` runs before the engine delete so the +// next read cannot repopulate a cache from a database this call is +// about to change. Same reason, same asymmetry. +// 3. THE CROSS-TAB FAN-OUT. Every client whose `sessionId` or +// `mcodeSessionId` pointed at the deleted record has its active +// session cleared and its usage counters zeroed, because the next +// interaction in that tab would otherwise silently recreate a webui +// wrapper for the very `mvs_` sid that was just deleted. The +// orphan branch clears only the REQUESTING client, because an +// orphan mcode session has no wrapper for another tab to be +// "inside". That asymmetry is real and load-bearing; flattening it +// would clear tabs that were never on the deleted session. +// +// So the sequencing lives here, named, and tested on its steps; the route +// keeps what is genuinely its own — HTTP parsing, the `authorize()` +// modal, the write-ahead audit ordering, and every status code. +// +// The one thing this file does NOT do is move the SQL. +// `lib/mcode-session-delete.js` keeps the 32-table `local_runtime_*` +// delete (its own header, its own per-table error classification, its +// own `getDb` seam) and this file reaches it through `await import()`. +// That is the same split B3 and B4 drew for their storage access +// (`lib/mavis-usage.js` owns the usage SQL, `lib/mcode-rpc.js` owns the +// account RPC), and it is the only shape that survives a real +// second reader appearing. The plan for this batch annotated +// `mcode-session-delete.js` "delete"; it is KEPT, and the reason is +// recorded as KNOWN DEBT in the module header of +// `lib/mcode-session-delete.js` itself. `lib/acp-client.js` imports +// `deleteMcodeSessionFromDb` from it, and four test files +// (`mcode-session-delete.test.js`, +// `mcode-session-delete-outcomes.test.js`, `sqlite-resolver-c01.test.js`, +// `sessions-switch.check.mjs`) bind to that exact specifier — deleting +// the module would break a live consumer and silently de-mock two +// existing route suites. KNOWN DEBT means "recorded and still +// uncollected", not "safe to remove". +// +// Boot-path weight. `app.js` imports the routes, the routes import this +// file, so this file is on the boot path. It statically imports nothing +// heavier than `capabilities.js` and `index.js` (both pure declaration +// modules); `lib/sessions.js`, `lib/acp-client.js`, +// `lib/mcode-session-delete.js`, `lib/session-tree.js`, `lib/state-bus.js` +// and `lib/config.js` are all reached through `await import()` inside +// the functions. That split is the M1 lesson, and it is what lets this +// module be re-exported from `engine/index.js` at all. +// +// Provider selection is M4's job, same as B1 through B4: +// `providerByTransport()` maps a transport to a REGISTERED provider id; +// today only `runtime` has one, so under the default `acp` transport the +// gate reports `gate: "unregistered-transport"` and the write proceeds — +// which is correct, because the pre-M4 behaviour under `acp` is the +// only behaviour these endpoints have ever had. + +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +// `node:fs` is a builtin, not a project dependency, and `usage-reads.js` +// already reaches for it at module scope for the same reason. It is here +// for exactly two calls: the orphan sweep's "is there a sessions store +// at all" probe and its BOM-tolerant read. +import { existsSync, readFileSync } from "node:fs"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, exactly as in + * `session-reads.js#providerByTransport`, + * `session-tree-reads.js#providerByTransport`, + * `usage-reads.js#providerByTransport` and + * `account-reads.js#providerByTransport`, which this mirrors rather than + * merges: the five families have separate contracts, and a shared table + * would force the write family to inherit a read family's policy. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. Every + * consumer of the table is a function anyway. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +// --------------------------------------------------------------------------- +// The declaration, and the gate policy that goes with it +// --------------------------------------------------------------------------- + +/** + * The declaration each endpoint of this family needs, the sub-item it + * needs from that capability, and HOW that declaration is enforced. + * + * The third field is this family's own addition, and it is not + * decoration — the hard/soft question for a write is decided by WHO + * OWNS THE ROWS THE WRITE DESTROYS, which is a different question from + * the read families' "is the data engine data or webui data", and it + * does not have the same answer twice in a row here: + * + * - #7 DELETE — **hard** on `sessionCrud` · `deleteSession`. The write + * destroys rows in the ENGINE's own `local_runtime_*` tables. There + * is no webui-side copy of a transcript that survives: once those + * rows are gone, the conversation is gone. A provider that declares + * no session deletion genuinely cannot have this endpoint serve a + * truthful answer, and the honest one is the 501 that + * `app.js#invokeHandler` derives from + * `EngineCapabilityNotSupportedError`. This is B4's account-read + * reasoning applied to a write: the data has exactly one owner, and + * it is not us. + * + * - #6 cleanup-orphans — **hard** on the SAME + * `sessionCrud` · `deleteSession` pair, deliberately. The sweep + * selects webui-side orphan RECORDS, but each selected id is fed + * through #7's real-delete branch, and a record carrying an + * `mcodeSessionId` takes the engine's rows down with it. Gating the + * sweep soft would mean a provider that cannot delete engine + * sessions could still reach the engine's tables through a back + * door — the exact shape this batch exists to close. A sweep that + * authorizes, writes its intent audit event and then fails every + * single delegated delete is also the fake-success shape: an + * authorized destructive action that accomplished nothing. + * + * - #4 rename — **no capability at all**, and this row is the one a + * reader will double-take, so here is the whole argument. Rename + * writes `item.title` / `item.titleCustom` / `item.updatedAt` into + * webui's OWN session store and nothing else: not the engine, not + * `local_runtime_sessions`, not any provider method. Its one engine + * touch is `invalidateSessionTree()`, a cache drop — the read-side + * consequence of the sidebar projecting titles from the engine, and + * the projection itself is B2's `GET /api/session-tree`, which + * carries its own gate. Naming a capability here would be a lie of + * the same kind B3 declined for `GET /api/usage/forecast`: a write + * that touches no engine surface must not be gated on an engine + * declaration, because gating it hard would remove a working + * endpoint in response to a statement about something it does not + * depend on. Note what this row also records about the product: a + * rename is a webui-side LABEL, and the engine's own title is not + * touched. That is pre-existing behaviour and this batch does not + * change it — see KNOWN DEBT at the end of this header. + * + * Every row carries all three keys, including the row that has no + * capability. B3 expressed "no engine surface" as a `null` table entry; + * this family has three endpoints of which two DO cross the seam, and a + * `null` hole in the middle of the table is the kind of shape a later + * edit mistakes for "not filled in yet". Uniform rows make the + * enforcement decision reviewable as one diff. + * + * @typedef {{capability: string|null, subItem: string|null, enforcement: "hard"|"soft"|"none"}} SessionWriteDeclaration + * @type {Readonly>} + */ +export const SESSION_WRITE_ENDPOINTS = Object.freeze({ + "DELETE /api/sessions/:id": Object.freeze({ + capability: "sessionCrud", + subItem: "deleteSession", + enforcement: "hard", + }), + "POST /api/sessions/rename": Object.freeze({ + capability: null, + subItem: null, + enforcement: "none", + }), + "POST /api/sessions/cleanup-orphans": Object.freeze({ + capability: "sessionCrud", + subItem: "deleteSession", + enforcement: "hard", + }), +}); + +/** + * Resolve the provider that answers session writes on `transport`, or + * `null` when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveSessionWriteProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Check one endpoint of this family against the active provider's + * declaration. Throws `EngineCapabilityNotSupportedError` — which + * `app.js#invokeHandler` turns into 501 — when the declaration says the + * capability (or the exact sub-item) is absent. + * + * Every row of `SESSION_WRITE_ENDPOINTS` is enforced at the strength its + * `enforcement` field names, and today only `"hard"` rows can throw: + * `"soft"` reports and returns (B2's `session-export.js` policy, for a + * family that has no soft row yet — the field is declared uniform so + * that adding one is a table edit rather than a signature change), and + * `"none"` never consults the provider at all. + * + * @param {string} endpoint A key of SESSION_WRITE_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null, enforcement: string}} + */ +export function assertSessionWriteCapability(endpoint, transport) { + const need = SESSION_WRITE_ENDPOINTS[endpoint]; + if (need === undefined) { + // Caller confusion, not an engine limitation — a plain Error so the + // HTTP layer never answers 501 for a typo in webui's own code. + const err = new Error( + `assertSessionWriteCapability: "${endpoint}" is not part of the session write family ` + + `(known: ${Object.keys(SESSION_WRITE_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_session_write_endpoint"; + throw err; + } + const provider = resolveSessionWriteProvider(transport); + if (need.capability === null) { + return { + endpoint, + gate: "no-capability-key", + provider: provider ? provider.id : null, + capability: null, + subItem: null, + enforcement: need.enforcement, + }; + } + if (!provider) { + return { + endpoint, + gate: "unregistered-transport", + provider: null, + capability: need.capability, + subItem: need.subItem, + enforcement: need.enforcement, + }; + } + assertEngineCapability(provider.capabilities, need.capability, provider.id, need.subItem); + return { + endpoint, + gate: "checked", + provider: provider.id, + capability: need.capability, + subItem: need.subItem, + enforcement: need.enforcement, + }; +} + +// --------------------------------------------------------------------------- +// Pure derivations. Exported and tested on their INPUTS. +// --------------------------------------------------------------------------- + +/** + * The engine's own session-id shape. Four call sites in the pre-facade + * delete path spelled this regex out inline, which is how a fifth call + * site eventually spelled it with a different quantifier. It is the + * predicate that separates "an id the engine minted" (orphan branch: + * delete the engine's rows directly) from "an id webui minted" (wrapper + * branch), so it is named rather than repeated. + * + * @param {unknown} id + * @returns {boolean} + */ +export function isMcodeSessionId(id) { + return typeof id === "string" && /^mvs_[a-f0-9]{32}$/.test(id); +} + +/** + * Resolve a caller-supplied id against the session store: by webui uuid + * first, then by the engine sid a record is bound to. + * + * This is the single-identity rule made explicit, and it was duplicated + * verbatim in `handleRenameSession` and `handleDeleteSession` before + * this batch — the same eleven lines, twice, with the same three + * possible answers. A third write endpoint would have been a third copy, + * and the copy that drifts is the one where a rename resolves a session + * the delete path cannot find, or the reverse. + * + * `matchKind` is `null` — never `"unknown"`, never `""` — exactly when + * the id resolved to nothing. Callers that need a label for the audit + * payload write `matchKind || "unknown"` themselves, because the two + * places that do (#7's `authorize()` context and #7's intent event) + * spell that fallback out and it is part of the audit contract. + * + * @param {Array} records The loaded session store. + * @param {string} id The id from the request. + * @returns {{index: number, matchKind: "webuiId"|"mcodeSessionId"|null, target: object|null}} + */ +export function resolveSessionTarget(records, id) { + const list = Array.isArray(records) ? records : []; + let index = list.findIndex((s) => s && s.id === id); + let matchKind = index >= 0 ? "webuiId" : null; + if (index < 0) { + index = list.findIndex((s) => s && s.mcodeSessionId === id); + if (index >= 0) matchKind = "mcodeSessionId"; + } + return { + index, + matchKind, + target: index >= 0 ? list[index] : null, + }; +} + +/** + * The staleness window the orphan sweep uses — 24h. Matches + * `lib/sessions.js#cleanupEmptyDefaultSessions`, which prunes the same + * class of leftover at startup; the sweep endpoint and the startup pass + * agree on what "leftover" means, and a future edit that moves one of + * them must move both. + */ +export const ORPHAN_STALE_MS = 24 * 60 * 60 * 1000; + +/** + * The "empty AND default-titled AND older than a day" rule behind + * `POST /api/sessions/cleanup-orphans`, as a pure predicate over ONE + * record. Split out of the store read so the rule is testable without a + * file and so the threshold is named rather than inlined at the filter + * site. + * + * `(record.title || "").trim()` is kept exactly as it was, including its + * behaviour on a non-string truthy title (a `TypeError`, which + * propagates out of the sweep as it always has). Tightening it here + * would be a behaviour change dressed as a hardening, and this batch + * promises none. + * + * @param {object} record + * @param {number} now Epoch ms, injected so the rule is pure. + * @param {number} staleMs The staleness threshold. + * @returns {boolean} + */ +export function isOrphanSessionRecord(record, now, staleMs) { + if (!record || !record.id) return false; + const hasChat = Array.isArray(record.chat) && record.chat.length > 0; + if (hasChat) return false; + const title = (record.title || "").trim(); + const isDefault = + title === "New session" || title === "Untitled" || /^对话 \d+$/.test(title); + if (!isDefault) return false; + if (record.updatedAt && now - record.updatedAt < staleMs) return false; + return true; +} + +/** + * The ids `POST /api/sessions/cleanup-orphans` would delete, in store + * order, under `isOrphanSessionRecord`. + * + * @param {Array} records + * @param {object} [options] + * @param {number} [options.now] Epoch ms; defaults to `Date.now()`. + * @param {number} [options.staleMs] Defaults to `ORPHAN_STALE_MS`. + * @returns {string[]} + */ +export function selectOrphanSessionIds(records, options = {}) { + const now = options.now === undefined ? Date.now() : options.now; + const staleMs = options.staleMs === undefined ? ORPHAN_STALE_MS : options.staleMs; + const list = Array.isArray(records) ? records : []; + return list.filter((s) => isOrphanSessionRecord(s, now, staleMs)).map((s) => s.id); +} + +/** + * The per-client state reset a delete fans out, as a PURE field + * assignment over one client's state object. + * + * Note what it does and does not touch. It clears the identity + * (`sessionId`, `mcodeSessionId`), the title and the chat buffer. It is + * not responsible for `resetContext` — that is a `lib/sessions.js` call + * with its own mocked parity in the test helper, and the caller runs it + * right after this so the ordering (`resetContext` sees the cleared + * identity) is the caller's to keep. + * + * `resetUsage` exists because the two delete branches genuinely differ + * here and the difference predates this batch. The wrapper branch zeroes + * the three cumulative session-usage counters, because the tab was + * showing a real session's spend and must stop. The orphan branch does + * NOT, because an orphan mcode session has no webui record and no tab + * can have accumulated webui-side per-session usage against it. Zeroing + * them there would be harmless; unifying the two branches is a product + * decision, not a refactor, so the asymmetry is a parameter with a + * comment rather than a silent difference between two call sites. + * + * @param {object} cs A webui client state. Mutated in place — every + * consumer of this predicate is already mutating `cs` in place. + * @param {object} [options] + * @param {boolean} [options.resetUsage] Default true (the wrapper + * branch). False for the orphan branch; see above. + * @returns {object} The same `cs`, for chaining. + */ +export function applyDeletedSessionToClientState(cs, options = {}) { + const resetUsage = options.resetUsage !== false; + cs.sessionId = null; + cs.mcodeSessionId = null; + cs.sessionTitle = "Untitled"; + cs.chat = []; + if (resetUsage) { + cs.usage = { + ...cs.usage, + sessionInput: 0, + sessionOutput: 0, + sessionTotal: 0, + }; + } + return cs; +} + +/** + * The per-client title fan-out a rename performs, as a pure assignment. + * Extracted for the same reason as the delete reset: the rename path + * runs it once per client whose identity matches, and a test that wants + * to prove "the other tab's title changed too" should be able to point at + * a named predicate instead of re-deriving the match rule. + * + * @param {object} cs + * @param {string} title The new title, already trimmed and validated. + * @returns {object} The same `cs. + */ +export function applyRenamedSessionToClientState(cs, title) { + cs.sessionTitle = title; + return cs; +} + +/** + * Whether a client is inside the record a RENAME is renaming, and so + * needs its title pushed. Matches on the record's webui id OR on the + * engine sid THE RECORD is bound to. + * + * This is deliberately NOT the same predicate as + * `clientMatchesDeletedSession`, even though both were one inline + * condition before this batch. They differ on the second clause, and the + * difference is load-bearing in both directions: + * + * - rename matches `record.mcodeSessionId`, because the record is the + * subject and every tab that adopted that engine session should see + * the new label. + * - delete matches the REQUEST id, because a tab is only "inside" the + * deletion if it is pointing at what the user asked to delete. A + * tab bound to the record's engine sid under a different webui id is + * a different wrapper record and must not be cleared. + * + * Merging them would either resurrect a wrapper in a tab the user just + * cleared, or blank the title of an unrelated tab. They stay two + * predicates, each named for the branch that uses it. + * + * @param {object} cs + * @param {object} record The session record being renamed. + * @returns {boolean} + */ +export function clientMatchesRenamedSession(cs, record) { + if (!cs || !record) return false; + if (cs.sessionId === record.id) return true; + return !!(record.mcodeSessionId && cs.mcodeSessionId === record.mcodeSessionId); +} + +/** + * Whether a client is inside the session a DELETE removed, and so needs + * its active session cleared. Matches on the record's webui id OR on the + * id the request named. See `clientMatchesRenamedSession` for why this + * is not the same predicate. + * + * @param {object} cs + * @param {object} record The deleted record; `null` for the orphan + * branch, where there is no record to match against. + * @param {string} requestId The id the caller asked to delete. + * @returns {boolean} + */ +export function clientMatchesDeletedSession(cs, record, requestId) { + if (!cs) return false; + if (record && cs.sessionId === record.id) return true; + return !!requestId && cs.mcodeSessionId === requestId; +} + +// --------------------------------------------------------------------------- +// Data-plane writes +// --------------------------------------------------------------------------- + +/** + * Load the store and resolve the requested id, without mutating + * anything. This is the half of #7 that has to happen BEFORE + * `authorize()` (the modal is shown for a specific record with a + * specific match kind and chat length) and before the write-ahead intent + * audit (which records the same three facts). + * + * Splitting plan from commit is what keeps the audit chain intact. The + * route must be able to interleave a governance decision and a durable + * event between "know what the user asked to delete" and "delete it", + * and a facade that owned the whole operation would have swallowed that + * ordering into a callback. Nothing here touches the database, the + * store, the caches or any client state. + * + * @param {object} options + * @param {string} options.id The requested id. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/:id`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @returns {Promise<{id: string, records: Array, index: number, matchKind: string|null, target: object|null, isOrphan: boolean, chatLen: number, gate: object, transport: string}>} + */ +export async function planEngineSessionDelete(options = {}) { + const endpoint = options.endpoint || "DELETE /api/sessions/:id"; + const [sessions, config] = await Promise.all([ + import("../lib/sessions.js"), + import("../lib/config.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertSessionWriteCapability(endpoint, transport); + const records = sessions.loadSessions(); + const { index, matchKind, target } = resolveSessionTarget(records, options.id); + return { + id: options.id, + records, + index, + matchKind, + target, + isOrphan: index < 0, + // `target && Array.isArray(target.chat)` rather than + // `Array.isArray(target?.chat)`: a store record that is not an + // object must read as "no chat", and the audit payload's `chatLen` + // has always been 0 for that case. + chatLen: index >= 0 && target && Array.isArray(target.chat) ? target.chat.length : 0, + gate, + transport, + }; +} + +/** + * #7's orphan branch: the id is an `mvs_…` sid with NO webui record, so + * there is no wrapper to remove and the only thing to delete is the + * engine's own rows. + * + * The resurrection guard and the cache drop are ordered deliberately and + * the order is the feature (see this file's header): kill the child + * that would rewrite the registry row, then delete, then drop the ONE + * cache entry — never the whole cache. + * + * `dryRun` suppresses the kill and the cache drop, because a preview + * mutates nothing and a preview that shuts down the user's ACP child is + * a side effect the `?dryRun=true` contract does not include. The COUNT + * still runs, read-only, inside `lib/mcode-session-delete.js`. + * + * The requesting client is reset when — and only when — it was + * currently sitting on that sid. No other tab can be: an orphan has no + * webui record for a tab to be inside. + * + * @param {object} options + * @param {object} options.plan A `planEngineSessionDelete` result. + * @param {object} [options.cs] The requesting client's state. + * @param {string} [options.cid] Requesting client id, for the state push. + * @param {boolean} [options.dryRun] + * @returns {Promise<{mcodeDbDel: object, payload: object, failed: boolean}>} + */ +export async function commitEngineOrphanSessionDelete(options = {}) { + const { plan, cs, cid, dryRun = false } = options; + const [deleter, config, acp, tree, bus, sessions] = await Promise.all([ + import("../lib/mcode-session-delete.js"), + import("../lib/config.js"), + import("../lib/acp-client.js"), + import("../lib/session-tree.js"), + import("../lib/state-bus.js"), + import("../lib/sessions.js"), + ]); + const id = plan.id; + if (!dryRun) { + // The child shutdown is wrapped in try/catch exactly as the + // pre-facade `killMcodeSessionResurrection` wrapped it: a live child + // that refuses to die must not abort the delete that follows. The + // cache drop is not wrapped, because a cache that cannot be dropped + // is the resurrection this branch exists to prevent. + try { + acp.shutdownMcodeAcpSingleton(); + } catch {} + acp.dropMcodeSessionFromCache(id); + } + const mcodeDbDel = deleter.deleteMcodeSessionFromDb(id, { + MCODE_RUNTIME_DB: config.MCODE_RUNTIME_DB, + dryRun, + }); + if (!dryRun) tree.invalidateSessionTree(); + if (!mcodeDbDel.ok) { + return { + mcodeDbDel, + failed: true, + payload: { ok: false, error: "orphan mcode delete failed", mcodeDbDel }, + }; + } + if (cs && cs.mcodeSessionId === id) { + // `resetUsage: false` — see `applyDeletedSessionToClientState`. An + // orphan has no webui record, so no tab accumulated per-session + // usage against it. + applyDeletedSessionToClientState(cs, { resetUsage: false }); + sessions.resetContext(cs); + bus.pushStateFor(cid); + } + return { + mcodeDbDel, + failed: false, + payload: { + ok: true, + deleted: id, + matchKind: "orphan_mcode", + dryRun, + mcodeDbDel, + }, + }; +} + +/** + * #7's `?dryRun=true` preview for a record that DOES have a webui + * wrapper: the readonly per-table count for the linked engine session, + * plus the webui entry that WOULD be removed. Nothing is written, no + * child is killed, no cache is dropped. + * + * A record with no `mcodeSessionId` (a webui-only session that never + * reached the engine) still previews — with an empty log and zero rows, + * the same literal the pre-facade route used to inline. A preview that + * refused to answer for those would be a new failure mode. + * + * @param {object} options + * @param {object} options.plan A `planEngineSessionDelete` result. + * @returns {Promise<{mcodeDbDel: object, payload: object}>} + */ +export async function previewEngineSessionDelete(options = {}) { + const { plan } = options; + const [deleter, config] = await Promise.all([ + import("../lib/mcode-session-delete.js"), + import("../lib/config.js"), + ]); + const mcodeSid = plan.target ? plan.target.mcodeSessionId : undefined; + const mcodeDbDel = mcodeSid + ? deleter.deleteMcodeSessionFromDb(mcodeSid, { + MCODE_RUNTIME_DB: config.MCODE_RUNTIME_DB, + dryRun: true, + }) + : { ok: true, dryRun: true, log: [], totalRows: 0 }; + return { + mcodeDbDel, + payload: { + ok: true, + dryRun: true, + matchKind: plan.matchKind, + mcodeDbDel, + webuiEntryWouldBeDeleted: { + id: plan.target.id, + title: plan.target.title, + mcodeSessionId: mcodeSid, + }, + }, + }; +} + +/** + * #7's real delete of a record that HAS a webui wrapper: splice the + * store, persist it, drop the tree cache, mirror the delete on the + * engine, then fan the cleared state out to every tab that was inside + * the record. + * + * The order is load-bearing in three places, all noted above: the tree + * cache is dropped BEFORE the engine write so a concurrent read cannot + * repopulate it from the pre-delete database; the engine mirror runs + * only when the record carries an `mcodeSessionId` (a webui-only session + * has no engine rows, and calling the deleter with `undefined` would + * report `not_mcode_sid` into the audit payload as if it had failed); + * and the fan-out runs AFTER both, so a tab is never told its session is + * gone while the rows still exist. + * + * `touchedCids` falls back to `[cid]` when no tab matched. That is not a + * no-op: it guarantees the requesting tab always gets a state push, so + * the client cannot be left rendering a session the server has already + * deleted. + * + * @param {object} options + * @param {object} options.plan A `planEngineSessionDelete` result. + * @param {string} [options.cid] Requesting client id. + * @returns {Promise<{deletedItem: object, records: Array, mcodeDbDel: object|null, touchedCids: string[], payload: object}>} + */ +export async function commitEngineSessionDelete(options = {}) { + const { plan, cid } = options; + const [deleter, config, acp, tree, bus, sessions] = await Promise.all([ + import("../lib/mcode-session-delete.js"), + import("../lib/config.js"), + import("../lib/acp-client.js"), + import("../lib/session-tree.js"), + import("../lib/state-bus.js"), + import("../lib/sessions.js"), + ]); + const deletedItem = plan.records[plan.index]; + const records = plan.records; + records.splice(plan.index, 1); + sessions.saveSessions(records); + tree.invalidateSessionTree(); + const mcodeSid = deletedItem.mcodeSessionId; + let mcodeDbDel = null; + if (mcodeSid) { + try { + acp.shutdownMcodeAcpSingleton(); + } catch {} + acp.dropMcodeSessionFromCache(mcodeSid); + mcodeDbDel = deleter.deleteMcodeSessionFromDb(mcodeSid, { + MCODE_RUNTIME_DB: config.MCODE_RUNTIME_DB, + }); + // The pre-facade route logged this from inside the `if (mcodeSid)` + // block, so the engine-mirror line only appears for records that + // actually have one. Kept here, next to the call it describes, so + // the operator log and the code that produced it stay together. + console.log( + `[delete] mcode db delete sid=${mcodeSid.substring(0, 12)}… ok=${mcodeDbDel.ok}` + + (mcodeDbDel.ok + ? ` log=[${(mcodeDbDel.log || []).join(",")}]` + : ` reason=${mcodeDbDel.reason || "-"} error=${mcodeDbDel.error || "-"}`), + ); + } + const touchedCids = []; + for (const [c, ccs] of bus.clients) { + if (!clientMatchesDeletedSession(ccs, deletedItem, plan.id)) continue; + applyDeletedSessionToClientState(ccs); + sessions.resetContext(ccs); + touchedCids.push(c); + } + // Exactly the pre-facade fallback, including the `undefined` it would + // push when the caller supplied no cid: the point is that the + // requesting tab ALWAYS gets a state push, so it cannot be left + // rendering a session the server has already deleted. + if (touchedCids.length === 0) touchedCids.push(cid); + for (const c of touchedCids) bus.pushStateFor(c); + return { + deletedItem, + records, + mcodeDbDel, + touchedCids, + payload: { + ok: true, + deleted: plan.id, + matchKind: plan.matchKind, + dryRun: false, + remaining: records.length, + mcodeDbDel, + }, + }; +} + +/** + * #4 — the rename write. + * + * Everything this endpoint persists lands in webui's own session store. + * The single engine touch is `invalidateSessionTree()`, and it is there + * because the sidebar tree reads titles out of the engine — the same + * reason the pre-facade route had it. The engine's own title is NOT + * written; see the `SESSION_WRITE_ENDPOINTS` row for why that makes the + * capability declaration `null` rather than a guess. + * + * The `not_found` outcome is a value, not an exception: a bare `mvs_…` + * id with no webui record gets an overlay record to carry the title + * (the single-identity rule, same as the switch path), and anything + * else is a 404 because the id is simply wrong. Returning which of the + * three happened is what lets the route write the right status without + * this module knowing what a status is. + * + * Validation of `id` and `title` is NOT done here. It is HTTP request + * validation with three 400 bodies this module would then have to + * reproduce byte for byte, and the route already owns the request. + * + * @param {object} options + * @param {string} options.id The record's webui uuid or `mvs_…` sid. + * @param {string} options.title New title; already trimmed and validated. + * @param {string} [options.cid] Requesting client id. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/rename`. + * @param {string} [options.transport] Transport override. + * @returns {Promise<{outcome: "ok"|"not_found", matchKind: string, from: string, to: string, item: object|null, payload: object|null, gate: object, transport: string}>} + */ +export async function applyEngineSessionRename(options = {}) { + const endpoint = options.endpoint || "POST /api/sessions/rename"; + const [sessions, config, tree, bus] = await Promise.all([ + import("../lib/sessions.js"), + import("../lib/config.js"), + import("../lib/session-tree.js"), + import("../lib/state-bus.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertSessionWriteCapability(endpoint, transport); + const id = options.id; + const title = options.title; + const all = sessions.loadSessions(); + const { index, matchKind: foundKind } = resolveSessionTarget(all, id); + let item; + let matchKind; + if (index < 0) { + if (!isMcodeSessionId(id)) { + return { + outcome: "not_found", + matchKind: null, + from: "", + to: title, + item: null, + payload: { ok: false, error: "session not found" }, + gate, + transport, + }; + } + // A bare mvs_ id with no webui shell gets one created to carry the + // title. No workspace argument, and none was ever passed: stamping + // the caller's current workspace onto someone else's record + // attributes a workspace the session never ran in, and re-roots the + // file tree on every later switch (webui-parity 63, defect F). + item = sessions.ensureOverlayForMcodeSid(all, id); + matchKind = "orphan_mcode"; + } else { + item = all[index]; + matchKind = foundKind; + } + const from = item.title || ""; + item.title = title; + item.titleCustom = true; + item.updatedAt = Date.now(); + sessions.saveSessions(all); + tree.invalidateSessionTree(); + let touchedCids = []; + for (const [c, ccs] of bus.clients) { + if (!clientMatchesRenamedSession(ccs, item)) continue; + applyRenamedSessionToClientState(ccs, title); + touchedCids.push(c); + } + if (touchedCids.length === 0) touchedCids.push(options.cid); + for (const c of touchedCids) bus.pushStateFor(c); + return { + outcome: "ok", + matchKind, + from, + to: title, + item, + payload: { + ok: true, + session: { + id: item.id, + mcodeSessionId: item.mcodeSessionId || null, + title: item.title, + titleCustom: true, + }, + }, + gate, + transport, + }; +} + +/** + * #6 — read the orphan sweep's target list. + * + * The file read stays here rather than in the route because the rule + * and the bytes it reads are one decision: a sweep that read a different + * file than the one whose rule it applies would be a bug waiting for a + * config change. The BOM strip is the store's own on-disk convention + * (written by an editor, not by webui) and is preserved exactly; a + * parse failure answers `[]`, which the pre-facade code did too, and a + * corrupt store must not turn a cleanup request into a 500. + * + * The response shape this backs is the batch's byte-for-byte red line, + * so the payload is built HERE and never re-assembled in the route: + * `{ok, dryRun, count, ids}` — four keys, in that order, for the + * preview; `{ok, dryRun:false, deleted, ids}` for the no-op real path. + * + * @param {object} [options] + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/cleanup-orphans`. + * @param {string} [options.transport] Transport override. + * @returns {Promise<{ids: string[], payload: object, gate: object, transport: string}>} + */ +export async function readOrphanSessionWriteIds(options = {}) { + const endpoint = options.endpoint || "POST /api/sessions/cleanup-orphans"; + const config = await import("../lib/config.js"); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = assertSessionWriteCapability(endpoint, transport); + const dbPath = config.SESSIONS_DB; + let records = []; + if (existsSync(dbPath)) { + try { + let raw = readFileSync(dbPath, "utf8"); + if (raw.charCodeAt(0) === 0xfeff) raw = raw.slice(1); + const parsed = JSON.parse(raw); + if (Array.isArray(parsed)) records = parsed; + } catch { + records = []; + } + } + const ids = selectOrphanSessionIds(records); + return { + ids, + payload: { ok: true, dryRun: true, count: ids.length, ids }, + gate, + transport, + }; +} + +// --------------------------------------------------------------------------- +// KNOWN DEBT +// --------------------------------------------------------------------------- +// +// Recorded here rather than fixed, because each item is a decision that +// belongs to a human and not to a refactor: +// +// 1. `lib/mcode-session-delete.js` still owns the 32-table SQL. The +// plan for this batch annotated it "delete"; it is kept because +// `lib/acp-client.js` imports from it and four test files bind to +// the specifier. Collecting it means moving those first. +// +// 2. #4 rename writes a WEBUI-side label only. The engine's own title +// in `local_runtime_sessions` is untouched, while the sidebar tree +// reads its titles from the engine. So for an engine-backed +// session a rename can be visible in the wrapper list and not in +// the tree. This is pre-existing behaviour and this batch did not +// change it; closing it means deciding which store is +// authoritative for a display title, which is a product call. +// +// 3. #7 does not detect "this session is running right now". A delete +// of an in-flight session kills the ACP child out from under the +// turn. That is the pre-facade behaviour and it is arguably the +// correct one (the user asked), but "refuse to delete a running +// session" is a defensible alternative and the choice is not this +// batch's to make. diff --git a/packages/webui/server/routes/sessions.js b/packages/webui/server/routes/sessions.js index b986f7e0..f5441611 100644 --- a/packages/webui/server/routes/sessions.js +++ b/packages/webui/server/routes/sessions.js @@ -10,16 +10,18 @@ import { loadSessions, saveSessions, resetContext, + // Still a direct import: `handleSwitchSession` creates the first-touch + // overlay itself. Rename used to call it too and no longer does — that + // write moved to `engine/session-writes.js` — but the switch path is a + // read-with-a-side-effect and stayed put, so this symbol has not + // finished migrating. ensureOverlayForMcodeSid, findOverlayForMcodeSid, } from "../lib/sessions.js"; -import { deleteMcodeSessionFromDb } from "../lib/mcode-session-delete.js"; import { getMcodeSessionTitle, getMcodeSessionsCacheSync, getMcodeSessionsStaleSync, - shutdownMcodeAcpSingleton, - dropMcodeSessionFromCache, } from "../lib/acp-client.js"; // Switch-path transcript backfill — load mcode session history from // the runtime DB so switching to an mvs_ session with no webui wrapper @@ -33,7 +35,6 @@ import { runChatViewChat, } from "../lib/state-bus.js"; import { MCODE_RUNTIME_DB, DEFAULT_WORKSPACE } from "../lib/config.js"; -import { invalidateSessionTree } from "../lib/session-tree.js"; // M3-B1 (engine facade): #9 and #10 read the engine through the declared // capability rather than straight off the ACP client. Both facade // functions forward to the same acp-client exports this module already @@ -46,11 +47,47 @@ import { // M3-B2 (engine facade): #8 asks the facade, which checks the provider's // declaration (sessionCrud.listSessions → 501 when absent) and then // forwards to the same `getSessionTree` this module used to call -// directly. `invalidateSessionTree` stays a direct import: it is a -// synchronous cache drop with no I/O, it is called from the rename and -// delete paths, and routing a one-line invalidation through an async -// facade would make those paths wait on a module load to do nothing. +// directly. `invalidateSessionTree` was a direct import here from B2 +// through B4 on the grounds that it is a synchronous cache drop with no +// I/O and routing a one-line invalidation through an async facade would +// make the caller wait on a module load to do nothing. M3-B5 retired +// that exception: the only three call sites were the rename and delete +// paths, and those moved into `engine/session-writes.js` as part of the +// ordered write sequences they belong to. A cache drop is not a +// standalone concern here — it is step two of a three-step resurrection +// guard, and keeping it addressable from the route was what made it +// possible to call it out of order. import { readEngineSessionTree } from "../engine/session-tree-reads.js"; +// M3-B5 (engine facade): #7 delete, #4 rename and #6 cleanup-orphans are +// the three WRITES of this module, and they ask the engine facade rather +// than driving the store, the caches and the engine's own `local_runtime_*` +// tables from the route. The split is deliberate and is the reason the +// handlers below shrank rather than grew: +// +// - The gate in front of each write is the facade's, not this file's. +// #7 and #6 gate hard on `sessionCrud` · `deleteSession` (the rows +// they destroy are the engine's own); #4 declares no capability at +// all, because a rename writes webui's store and nothing else. +// - The load→resolve→authorize→intent-audit→mutate ORDER is still +// this file's, and had to stay: the write-ahead audit has to land +// between "know what the user asked to delete" and "delete it". So +// the facade exposes a plan/commit pair rather than one +// `deleteSession(options)` that would have swallowed the ordering. +// - The response BODIES are built in the facade, once. #6's dryRun +// shape is a byte-for-byte red line for this batch, so it is pinned +// there by test instead of re-assembled in two places here. +// - `deleteMcodeSessionFromDb` and the 32-table SQL stay in +// `lib/mcode-session-delete.js` and are reached by the facade through +// a dynamic import; see KNOWN DEBT in `engine/session-writes.js`. +import { + applyEngineSessionRename, + commitEngineOrphanSessionDelete, + commitEngineSessionDelete, + isMcodeSessionId, + planEngineSessionDelete, + previewEngineSessionDelete, + readOrphanSessionWriteIds, +} from "../engine/session-writes.js"; // The capability-error predicate `handleSessionTree` uses to tell the gate's // 501 apart from a soft-fail. Taken from the facade entry, which re-exports // the same binding `app.js#invokeHandler` matches on, so the two ends of this @@ -189,20 +226,17 @@ function _auditFail(res, e, what) { return undefined; } -// Prevent "deleted session reappears": the long-lived mcode acp child -// still holds the session in memory and will rewrite the registry row -// on its next request — so we must (1) kill the child, (2) SQL-delete -// the rows, (3) drop ONLY the deleted sid from the in-memory cache (not -// the whole cache — invalidating the whole cache sends an empty -// placeholder to the sidebar which flashes from 42 → 16 → 42 entries, -// looking like the delete failed). -function killMcodeSessionResurrection(mcodeSid) { - try { - shutdownMcodeAcpSingleton(); - } catch {} - dropMcodeSessionFromCache(mcodeSid); -} - +// Prevent "deleted session reappears" — moved to the engine facade in +// M3-B5. The long-lived mcode acp child still holds the session in +// memory and will rewrite the registry row on its next request, so the +// delete has to (1) kill the child, (2) SQL-delete the rows, (3) drop +// ONLY the deleted sid from the in-memory cache (not the whole cache — +// invalidating the whole cache sends an empty placeholder to the sidebar +// which flashes from 42 → 16 → 42 entries, looking like the delete +// failed). That sequence is now +// `engine/session-writes.js`, where it is named and tested step by step +// instead of being a two-line helper a route could call in the wrong +// order. // Title fast path — resolve an mvs_ session's title from the // in-memory walked-session cache (the same cache behind @@ -638,6 +672,16 @@ export async function handleSwitchSession(req, res, ctx) { // overlay record to carry the title (single-identity rule, same as the switch // path). Audit: session.rename records from → to, fail-closed. Not behind the // authorize() modal — renaming is reversible; only destructive actions prompt. +// +// M3-B5: the write itself — resolve, overlay, title write, store save, tree +// cache drop, cross-tab title fan-out — happens in +// `engine/session-writes.js#applyEngineSessionRename`, and the response body +// is built there. What stays HERE is what is genuinely the route's: the three +// 400 bodies (request validation the facade has no business reproducing), the +// 404 status for the facade's `not_found` outcome, the fail-closed audit, and +// the log line. The facade's gate for this endpoint declares NO capability — +// a rename writes webui's own store and touches no engine surface; see the +// `SESSION_WRITE_ENDPOINTS` row for the full argument. export async function handleRenameSession(req, res, ctx) { const cid = ctx.cid; const payload = await readJson(req); @@ -657,95 +701,64 @@ export async function handleRenameSession(req, res, ctx) { JSON.stringify({ ok: false, error: "title too long (max 200)" }), ); } - const all = loadSessions(); - let idx = all.findIndex((s) => s.id === id); - let matchKind = idx >= 0 ? "webuiId" : null; - if (idx < 0) { - idx = all.findIndex((s) => s.mcodeSessionId === id); - if (idx >= 0) matchKind = "mcodeSessionId"; + const w = await applyEngineSessionRename({ id, title, cid }); + if (w.outcome === "not_found") { + res.writeHead(404, { "Content-Type": "application/json; charset=utf-8" }); + return res.end(JSON.stringify(w.payload)); } - let item; - if (idx < 0) { - // 纯 mcode 会话(sidebar 的 mvs_ 条目还没有 webui 壳)→ 建壳承接改名。 - // 其余 id 不硬造记录:404,让调用方知道 id 写错了。 - if (/^mvs_[a-f0-9]{32}$/.test(id)) { - // webui-parity 63 (defect F): no workspace argument, for the same - // reason the switch path dropped it (see the s39 note above) — and here - // it was the last remaining writer. Stamping cs.workspace.dir onto - // someone else's record attributes a workspace the session never ran - // in, and cs.workspace.dir is not even necessarily a real one: a - // switch to a session that stores no workspace leaves it holding the - // DEFAULT_WORKSPACE fallback, which then got persisted and re-rooted - // the file tree on every later switch. Unknown stays unknown (""); - // the target-first read picks the fallback at read time instead. - item = ensureOverlayForMcodeSid(all, id); - matchKind = "orphan_mcode"; - } else { - res.writeHead(404, { "Content-Type": "application/json; charset=utf-8" }); - return res.end(JSON.stringify({ ok: false, error: "session not found" })); - } - } else { - item = all[idx]; - } - const from = item.title || ""; - item.title = title; - item.titleCustom = true; - item.updatedAt = Date.now(); - saveSessions(all); - // The sidebar tree reads titles from the runtime db, so drop its cache or the - // renamed title stays hidden for up to CACHE_TTL_MS. - invalidateSessionTree(); - // 所有把该会话当"当前会话"的 client 同步 sessionTitle(多 tab 一致)。 - let touchedCids = []; - for (const [c, ccs] of clients) { - if ( - ccs.sessionId === item.id || - (item.mcodeSessionId && ccs.mcodeSessionId === item.mcodeSessionId) - ) { - ccs.sessionTitle = title; - touchedCids.push(c); - } - } - if (touchedCids.length === 0) touchedCids = [cid]; - for (const c of touchedCids) pushStateFor(c); try { _eventsAppend("session.rename", { - target: item.id, + target: w.item.id, cid, actor: "user", payload: { - matchKind, - from, - to: title, - mcodeSessionId: item.mcodeSessionId || "", + matchKind: w.matchKind, + from: w.from, + to: w.to, + mcodeSessionId: w.item.mcodeSessionId || "", }, }); } catch (e) { return _auditFail(res, e, "session.rename"); } console.log( - `[rename] cid=${cid} OK match=${matchKind} id=${item.id.substring(0, 8)}… "${from}" → "${title}"`, + `[rename] cid=${cid} OK match=${w.matchKind} id=${w.item.id.substring(0, 8)}… "${w.from}" → "${w.to}"`, ); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - return res.end( - JSON.stringify({ - ok: true, - session: { - id: item.id, - mcodeSessionId: item.mcodeSessionId || null, - title: item.title, - titleCustom: true, - }, - }), - ); + return res.end(JSON.stringify(w.payload)); } // DELETE /api/sessions/:id — delete a session. // // ?dryRun=true takes the readonly SQL path (counts rows per table, // mutates nothing). Real delete passes authorize() and only then -// touches db / saveSessions / killMcodeSessionResurrection (the gate -// is the only async hop on the real path). +// touches db / saveSessions / the caches (the gate is the only async hop +// on the real path). +// +// M3-B5: this handler is now a PLAN → GOVERN → COMMIT sequence, and that +// shape is the point rather than an accident of the refactor. +// +// planEngineSessionDelete resolves the id and runs the gate. No +// mutation, so it is safe to run BEFORE +// the user is asked anything. +// authorize() + intent audit unchanged, and still strictly between +// the plan and the commit. The write-ahead +// intent line has to be durably recorded +// before any row is removed, and it +// records the match kind and chat length +// the plan produced. +// commit*EngineSessionDelete splices the store, drops the tree cache, +// mirrors the delete into the engine's +// `local_runtime_*` tables and fans the +// cleared state out to every tab. The +// ORDER of those steps inside the facade +// is the resurrection guard; see the +// facade's module header. +// +// Every status code and every response body below is unchanged. The +// bodies are now BUILT in the facade rather than here, which is what lets +// the dryRun shape be pinned byte-for-byte by a unit test instead of by a +// route test that has to stand up the whole request. export async function handleDeleteSession(req, res, ctx) { const cs = ctx.cs; const cid = ctx.cid; @@ -764,26 +777,20 @@ export async function handleDeleteSession(req, res, ctx) { } } catch {} console.log( - `[delete] cid=${cid} incoming id=${id.substring(0, 12)}… isMcodeSid=${/^mvs_[a-f0-9]{32}$/.test(id)} dryRun=${dryRun}`, + `[delete] cid=${cid} incoming id=${id.substring(0, 12)}… isMcodeSid=${isMcodeSessionId(id)} dryRun=${dryRun}`, ); - const all = loadSessions(); - let idx = all.findIndex((s) => s.id === id); - let matchKind = idx >= 0 ? "webuiId" : null; - if (idx < 0) { - idx = all.findIndex((s) => s.mcodeSessionId === id); - if (idx >= 0) matchKind = "mcodeSessionId"; - } + const plan = await planEngineSessionDelete({ id }); // B03: real-delete path must pass per-request authorize() before - // mutating db / saveSessions / killMcodeSessionResurrection. + // mutating db / saveSessions / the caches. // dryRun=true bypasses (preview only — no side effects to gate). if (!dryRun) { const authResult = await authorize("session.delete", { cid, targetSessionId: id, - matchKind: matchKind || (idx < 0 ? "unknown" : "webuiId"), - isMcodeSid: /^mvs_[a-f0-9]{32}$/.test(id), - isOrphan: idx < 0, - chatLen: idx >= 0 && all[idx] && Array.isArray(all[idx].chat) ? all[idx].chat.length : 0, + matchKind: plan.matchKind || (plan.isOrphan ? "unknown" : "webuiId"), + isMcodeSid: isMcodeSessionId(id), + isOrphan: plan.isOrphan, + chatLen: plan.chatLen, }); if (!authResult.approved) { console.log( @@ -809,9 +816,9 @@ export async function handleDeleteSession(req, res, ctx) { cid, actor: "user", payload: { - matchKind: matchKind || "unknown", - isOrphan: idx < 0, - chatLen: idx >= 0 && all[idx] && Array.isArray(all[idx].chat) ? all[idx].chat.length : 0, + matchKind: plan.matchKind || "unknown", + isOrphan: plan.isOrphan, + chatLen: plan.chatLen, decidedBy: authResult.decidedBy, }, }); @@ -822,68 +829,41 @@ export async function handleDeleteSession(req, res, ctx) { // Fallback: id is mvs_xxx but absent from webui session db — // treat it as an orphan mcode session and delete the SQL rows // directly (the webui side has no wrapper to remove). - if (idx < 0) { - if (/^mvs_[a-f0-9]{32}$/.test(id)) { - if (!dryRun) killMcodeSessionResurrection(id); - const mcodeDbDel = deleteMcodeSessionFromDb(id, { MCODE_RUNTIME_DB, dryRun }); - // Same reason as the wrapper-delete path below: this removes rows from - // the db the cached sidebar tree is built from. Skipped on a dry run, - // which mutates nothing. - if (!dryRun) invalidateSessionTree(); + if (plan.isOrphan) { + if (isMcodeSessionId(id)) { + const w = await commitEngineOrphanSessionDelete({ plan, cs, cid, dryRun }); console.log( - `[delete] cid=${cid} ORPHAN mcode session sid=${id.substring(0, 12)}… ok=${mcodeDbDel.ok}` + - (mcodeDbDel.ok - ? ` log=[${(mcodeDbDel.log || []).join(",")}]` - : ` reason=${mcodeDbDel.reason || "-"} error=${mcodeDbDel.error || "-"}`), + `[delete] cid=${cid} ORPHAN mcode session sid=${id.substring(0, 12)}… ok=${w.mcodeDbDel.ok}` + + (w.mcodeDbDel.ok + ? ` log=[${(w.mcodeDbDel.log || []).join(",")}]` + : ` reason=${w.mcodeDbDel.reason || "-"} error=${w.mcodeDbDel.error || "-"}`), ); - if (mcodeDbDel.ok) { - if (cs.mcodeSessionId === id) { - cs.mcodeSessionId = null; - cs.sessionId = null; - cs.sessionTitle = "Untitled"; - cs.chat = []; - resetContext(cs); - pushStateFor(cid); - } - // B01: orphan mcode session deletion (no webui session row). - // Outcome event; the intent line was written before the gate - // fan-out above. Failure → 5xx + alert (rows are already gone; - // the operator must see the audit gap, not a silent success). - try { - _eventsAppend("session.delete", { - target: id, - cid, - actor: "user", - payload: { - matchKind: "orphan_mcode", - dryRun, - rowsAffected: (mcodeDbDel.log || []).length, - }, - }); - } catch (e) { - return _auditFail(res, e, "session.delete(orphan_mcode)"); - } - res.writeHead(200, { - "Content-Type": "application/json; charset=utf-8", - }); - return res.end( - JSON.stringify({ - ok: true, - deleted: id, + if (w.failed) { + res.writeHead(500, { "Content-Type": "application/json" }); + return res.end(JSON.stringify(w.payload)); + } + // B01: orphan mcode session deletion (no webui session row). + // Outcome event; the intent line was written before the gate + // fan-out above. Failure → 5xx + alert (rows are already gone; + // the operator must see the audit gap, not a silent success). + try { + _eventsAppend("session.delete", { + target: id, + cid, + actor: "user", + payload: { matchKind: "orphan_mcode", dryRun, - mcodeDbDel, - }), - ); + rowsAffected: (w.mcodeDbDel.log || []).length, + }, + }); + } catch (e) { + return _auditFail(res, e, "session.delete(orphan_mcode)"); } - res.writeHead(500, { "Content-Type": "application/json" }); - return res.end( - JSON.stringify({ - ok: false, - error: "orphan mcode delete failed", - mcodeDbDel, - }), - ); + res.writeHead(200, { + "Content-Type": "application/json; charset=utf-8", + }); + return res.end(JSON.stringify(w.payload)); } console.log(`[delete] cid=${cid} 404 id=${id.substring(0, 12)}… not found`); res.writeHead(404, { "Content-Type": "application/json" }); @@ -891,12 +871,9 @@ export async function handleDeleteSession(req, res, ctx) { } // dryRun: 不真删 webui session entry,只预览 mcode db 影响 if (dryRun) { - const mcodeSid = all[idx].mcodeSessionId; - const mcodeDbDel = mcodeSid - ? deleteMcodeSessionFromDb(mcodeSid, { MCODE_RUNTIME_DB, dryRun: true }) - : { ok: true, dryRun: true, log: [], totalRows: 0 }; + const w = await previewEngineSessionDelete({ plan }); console.log( - `[delete] cid=${cid} DRYRUN id=${id.substring(0, 12)}… mcodeDbDel=${JSON.stringify(mcodeDbDel)}`, + `[delete] cid=${cid} DRYRUN id=${id.substring(0, 12)}… mcodeDbDel=${JSON.stringify(w.mcodeDbDel)}`, ); // B01: dryRun is itself a state-touching action — the operator // is previewing a delete, so record the preview but never the @@ -910,9 +887,9 @@ export async function handleDeleteSession(req, res, ctx) { cid, actor: "user", payload: { - matchKind, + matchKind: plan.matchKind, dryRun: true, - previewedRows: mcodeDbDel.totalRows || 0, + previewedRows: w.mcodeDbDel.totalRows || 0, }, }); } catch (e) { @@ -921,69 +898,9 @@ export async function handleDeleteSession(req, res, ctx) { res.writeHead(200, { "Content-Type": "application/json; charset=utf-8", }); - return res.end( - JSON.stringify({ - ok: true, - dryRun: true, - matchKind, - mcodeDbDel, - webuiEntryWouldBeDeleted: { - id: all[idx].id, - title: all[idx].title, - mcodeSessionId: mcodeSid, - }, - }), - ); - } - const deletedItem = all[idx]; - all.splice(idx, 1); - saveSessions(all); - // The sidebar tree is assembled from `local_runtime_sessions` in the runtime - // db, and it is cached for CACHE_TTL_MS (the git probe per directory is the - // expensive part). A delete removes rows from that db, so the cache has to go - // or the row stays in the sidebar — still clickable — for up to 15s. This - // was the one mutation that missed it; rename had been handled, and - // switch/new were never wrong (switch does not change the set, and a new - // webui session has no engine row until its first prompt). - // - // Invalidate before the engine delete below, so the next read cannot repopulate - // from a db this call is about to change. - invalidateSessionTree(); - // Mirror the delete on the mcode side when this record has an mcode sid. - const mcodeSid = deletedItem.mcodeSessionId; - let mcodeDbDel = null; - if (mcodeSid) { - killMcodeSessionResurrection(mcodeSid); - mcodeDbDel = deleteMcodeSessionFromDb(mcodeSid, { MCODE_RUNTIME_DB }); - console.log( - `[delete] cid=${cid} mcode db delete sid=${mcodeSid.substring(0, 12)}… ok=${mcodeDbDel.ok}` + - (mcodeDbDel.ok - ? ` log=[${(mcodeDbDel.log || []).join(",")}]` - : ` reason=${mcodeDbDel.reason || "-"} error=${mcodeDbDel.error || "-"}`), - ); - } - // Clear active session on every client that pointed at this id (or - // its mcode sibling) — otherwise the next interaction in that tab - // silently recreates a webui wrapper for the same mvs sid. - let touchedCids = []; - for (const [c, ccs] of clients) { - if (ccs.sessionId === deletedItem.id || ccs.mcodeSessionId === id) { - ccs.sessionId = null; - ccs.mcodeSessionId = null; - ccs.sessionTitle = "Untitled"; - ccs.chat = []; - ccs.usage = { - ...ccs.usage, - sessionInput: 0, - sessionOutput: 0, - sessionTotal: 0, - }; - resetContext(ccs); - touchedCids.push(c); - } + return res.end(JSON.stringify(w.payload)); } - if (touchedCids.length === 0) touchedCids = [cid]; - for (const c of touchedCids) pushStateFor(c); + const w = await commitEngineSessionDelete({ plan, cid }); // B01: real session delete (the dangerous one). Record which webui // session was deleted, what the match kind was, how many cids had // their active session cleared (this is the "fan-out" effect that @@ -998,31 +915,22 @@ export async function handleDeleteSession(req, res, ctx) { cid, actor: "user", payload: { - matchKind, + matchKind: plan.matchKind, dryRun: false, - remaining: all.length, - touchedCids: touchedCids.length, - mcodeRowsAffected: mcodeDbDel && mcodeDbDel.log ? mcodeDbDel.log.length : 0, - title: deletedItem.title, + remaining: w.records.length, + touchedCids: w.touchedCids.length, + mcodeRowsAffected: w.mcodeDbDel && w.mcodeDbDel.log ? w.mcodeDbDel.log.length : 0, + title: w.deletedItem.title, }, }); } catch (e) { return _auditFail(res, e, "session.delete"); } console.log( - `[delete] cid=${cid} OK match=${matchKind} deleted.webuiId=${deletedItem.id.substring(0, 8)}… remaining=${all.length}`, + `[delete] cid=${cid} OK match=${plan.matchKind} deleted.webuiId=${w.deletedItem.id.substring(0, 8)}… remaining=${w.records.length}`, ); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - return res.end( - JSON.stringify({ - ok: true, - deleted: id, - matchKind, - dryRun: false, - remaining: all.length, - mcodeDbDel, - }), - ); + return res.end(JSON.stringify(w.payload)); } // GET /api/session-tree — the sidebar's Project → directory → session → subagent @@ -1279,39 +1187,23 @@ export async function handleSearchSessions(req, res, ctx) { // The cleanup targets: default-named webui sessions (New session / // Untitled / 对话 N) whose chat is empty AND whose updatedAt is older // than 24h — same rule as cleanupEmptyDefaultSessions() in lib/sessions.js. -import { existsSync, readFileSync } from "node:fs"; -import { SESSIONS_DB } from "../lib/config.js"; +// +// M3-B5: the SELECTION moved into the facade +// (`engine/session-writes.js#readOrphanSessionWriteIds`), together with +// the store read it applies the rule to and with the two response bodies +// the batch's red line pins byte-for-byte. The rule and the file it reads +// are one decision; splitting them across two modules is how a sweep ends +// up pruning a different store than the one it was written for. +// +// The DELEGATION stays here and is not an oversight. Each selected id is +// routed back through `handleDeleteSession` precisely so that every +// orphan costs the same `session.delete.intent` / `session.delete` audit +// pair, the same authorize() decision and the same cross-tab fan-out that +// a hand-deleted session costs. Re-implementing the delete inside the +// sweep would produce a cheaper path that is not the same path, and the +// audit chain is the thing this endpoint exists to preserve. import { readJson } from "../lib/read-json.js"; -const ORPHAN_STALE_MS = 24 * 60 * 60 * 1000; - -function _findOrphanIds() { - if (!existsSync(SESSIONS_DB)) return []; - let all; - try { - let raw = readFileSync(SESSIONS_DB, "utf8"); - if (raw.charCodeAt(0) === 0xfeff) raw = raw.slice(1); // 剥 BOM - all = JSON.parse(raw); - } catch { - return []; - } - if (!Array.isArray(all) || all.length === 0) return []; - const now = Date.now(); - return all - .filter((s) => { - if (!s || !s.id) return false; - const hasChat = Array.isArray(s.chat) && s.chat.length > 0; - if (hasChat) return false; - const t = (s.title || "").trim(); - const isDefault = - t === "New session" || t === "Untitled" || /^对话 \d+$/.test(t); - if (!isDefault) return false; - if (s.updatedAt && now - s.updatedAt < ORPHAN_STALE_MS) return false; - return true; - }) - .map((s) => s.id); -} - export async function handleCleanupOrphans(req, res, ctx) { const cid = (ctx && ctx.cid) || ""; let dryRun = false; @@ -1322,19 +1214,17 @@ export async function handleCleanupOrphans(req, res, ctx) { dryRun = params.get("dryRun") === "true"; } } catch {} - const targetIds = _findOrphanIds(); - // Preview path: no authorize gate (no side effects). + const sweep = await readOrphanSessionWriteIds(); + const targetIds = sweep.ids; + // Preview path: no authorize gate (no side effects). The body is + // `{ok, dryRun, count, ids}` — four keys, in that order — and it is + // built in the facade so that shape has exactly one home. if (dryRun) { console.log( `[cleanup-orphans] cid=${cid} DRYRUN would-delete=${targetIds.length}`, ); res.writeHead(200, { "Content-Type": "application/json; charset=utf-8" }); - return res.end(JSON.stringify({ - ok: true, - dryRun: true, - count: targetIds.length, - ids: targetIds, - })); + return res.end(JSON.stringify(sweep.payload)); } // Real path: gate with authorize() before touching any session. if (targetIds.length === 0) { diff --git a/packages/webui/test/helpers/_setup.js b/packages/webui/test/helpers/_setup.js index d30db95f..af48af8c 100644 --- a/packages/webui/test/helpers/_setup.js +++ b/packages/webui/test/helpers/_setup.js @@ -294,6 +294,41 @@ export async function setupMocks(t, overrides = {}) { } }, persistCurrentChat: () => {}, + // M3-B5: lib/sessions.js really exports this one — the + // single-identity rule, an overlay record whose `id` IS the engine + // sid — and routes/sessions.js has imported it since the switch + // path added it, but the mock never grew it. Every consumer so far + // either never called it or owned its own store mock, and a missing + // name only bites at module-instantiation time. M3-B5 moved the + // RENAME path's call into the engine facade, whose orphan-mcode + // branch calls it, so the omission became reachable from this + // shared helper rather than from a test that could stub around it. + // Mirrors the real body, including the placeholder-title repair and + // the unshift, so a test that renames a bare mvs_ id sees the + // record it would see in production. + ensureOverlayForMcodeSid: (all, sid, { title, workspace } = {}) => { + if (!Array.isArray(all) || !sid) return null; + let rec = all.find((s) => s && s.mcodeSessionId === sid) || null; + if (rec) { + if (title && rec.title === "Mcode session") rec.title = title; + return rec; + } + rec = { + id: sid, + mcodeSessionId: sid, + title: title || "Mcode session", + workspace: workspace || "", + createdAt: Date.now(), + updatedAt: Date.now(), + chat: [], + }; + all.unshift(rec); + return rec; + }, + findOverlayForMcodeSid: (all, sid) => { + if (!Array.isArray(all) || !sid) return null; + return all.find((s) => s && s.mcodeSessionId === sid) || null; + }, // session-isolation/02 (run-mirror): the buffer-drain finalize path // (routes/chat.js) writes the turn back to the owning session's // persisted record; mirror the real lookup (by webui id, then by diff --git a/packages/webui/test/lib/engine/session-writes.test.js b/packages/webui/test/lib/engine/session-writes.test.js new file mode 100644 index 00000000..767de244 --- /dev/null +++ b/packages/webui/test/lib/engine/session-writes.test.js @@ -0,0 +1,1733 @@ +// webui/test/lib/engine/session-writes.test.js +// +// M3-B5: the session WRITE family's engine facade — #7 delete, #4 +// rename, #6 cleanup-orphans. +// +// This is the first suite in the migration that tests a family which +// DESTROYS data, so the sections below are ordered by how much damage a +// regression in each one does, not by which module the function came +// from: +// +// 1. THE DECLARATION AND ITS POLICY. The hard/none split in here is +// the batch's most consequential judgement call: #7 and #6 are hard +// because they destroy the engine's own rows, #4 declares no +// capability because it touches no engine surface. Section 2 proves +// the asymmetry is real by driving all three endpoints from ONE +// provider fixture. +// +// 2. THE FIVE DELETE RED LINES. "Deleted sessions must not come back", +// "deleting a session is not deleting files", "a running session +// has defined semantics", "the other tab must lose the entry", and +// "the audit chain stays intact". These are the checks a reviewer +// should read first, so they get their own section with one test +// per line. +// +// 3. THE BYTE-FOR-BYTE PREVIEW SHAPES. #6's dryRun body is a hard red +// line for this batch; #7's is pinned beside it because the same +// edit touched both. +// +// 4. THE PURE DERIVATIONS, on their inputs. +// +// 5. THE ROUTE, with the proof that the facade mock actually took. +// +// Two module-mock traps apply here exactly as they did in B3/B4, and +// both are load-bearing rather than incidental: +// +// 1. `t.mock.module` REPLACES the WHOLE NAMESPACE; it does not merge. +// A mock naming only the export under test leaves every other name +// undefined and the consumer fails at INSTANTIATION with +// `SyntaxError: … does not provide an export named …` — a failure +// that reads like a product bug and is not one. Every mock below +// goes through `mockAll()`, which fills the un-stubbed names with a +// function that THROWS, so an unexpected call is loud instead of +// returning a plausible payload. +// 2. `mock.module` re-evaluates only the MOCKED specifier. A consumer +// already in the registry keeps its old LIVE BINDING, so a second +// test in the same file would silently reuse the first test's mock +// and pass for the wrong reason. Every route re-import carries a +// fresh `?bust=N`, and section 5 ends with the marker control that +// proves it. + +import { test, describe, before, after, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, mkdirSync, writeFileSync, readFileSync } from "node:fs"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { Readable } from "node:stream"; +import { spawnSync } from "node:child_process"; + +import { + setupMocks, + absPath, + registerSessionsStore, + registerAcpMock, + withDecisions, +} from "../../helpers/_setup.js"; +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; +// Type discrimination goes through the exported predicate, never +// `err.name`. `engine/capabilities.js` is never `mock.module`d by this +// file, so the `instanceof` inside it resolves against the same class the +// gate throws from; the sibling batches (account-reads, session-export) +// assert the same way. The string comparison it replaces could not tell a +// capability error from any other error that happened to carry a name. +const { isEngineCapabilityNotSupportedError } = await import( + "../../../server/engine/errors.js" +); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +// A syntactically valid engine sid — `isMcodeSessionId` requires exactly +// 32 lowercase hex digits, and every fixture below that wants the ORPHAN +// branch has to satisfy the same regex the pre-facade route spelled +// inline four times. +const ORPHAN_SID = "mvs_aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; + +let bust = 0; + +/** A JSON request body the real `lib/read-json.js` can consume. */ +function jsonReq(body) { + return Readable.from([Buffer.from(JSON.stringify(body), "utf8")]); +} + +/** A minimal `ServerResponse` stand-in that records what was written. */ +function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; +} + +/** + * A fresh copy of `routes/sessions.js`. + * + * `mock.module` re-evaluates only the MOCKED specifier, but a route + * module already in the registry keeps its old LIVE BINDING to the + * facade — without the `?bust=N` re-import a second test would silently + * exercise the first test's mock and pass for the wrong reason. That is + * what the PROOF cases below exist to catch. + */ +const loadRoute = async () => import(`${absPath("routes/sessions.js")}?bust=${bust++}`); + +/** + * Register a module mock that satisfies the namespace contract. + * + * @param {object} t The test context. + * @param {string} rel Server-relative specifier, e.g. "lib/foo.js". + * @param {object} impls The exports this test stubs. + * @param {string[]} known Every export name the REAL module has, so + * anything this test does not stub is present-but-throwing rather + * than absent. + */ +function mockAll(t, rel, impls, known) { + const namedExports = {}; + for (const name of known) { + namedExports[name] = (...a) => { + throw new Error(`B5 test called ${rel}#${name}, which this case did not stub`); + }; + } + Object.assign(namedExports, impls); + t.mock.module(absPath(rel), { namedExports }); +} + +/** + * The record-ordering journal the delete tests assert on. Every mutation + * the write path performs appends its name here, so a test can assert + * the SEQUENCE rather than the end state — and a sequence is the only + * thing that distinguishes a correct delete from a resurrecting one. + */ +const journal = []; +function resetJournal() { + journal.length = 0; +} + +describe("M3-B5 — session write family", () => { + // --------------------------------------------------------------------- + // 1. The declaration table and the gate policy it records + // --------------------------------------------------------------------- + + describe("SESSION_WRITE_ENDPOINTS — the three writes, and who owns the rows they destroy", () => { + test("covers exactly this batch's three endpoints", async () => { + const { SESSION_WRITE_ENDPOINTS } = await import( + absPath("engine/session-writes.js") + ); + assert.deepEqual(Object.keys(SESSION_WRITE_ENDPOINTS), [ + "DELETE /api/sessions/:id", + "POST /api/sessions/rename", + "POST /api/sessions/cleanup-orphans", + ]); + }); + + test("every row declares the same three keys, including the no-capability one", async () => { + // The uniformity is the point of this family's table shape: a + // `null` hole for rename would read as "not filled in yet" to the + // next editor rather than as a decision. + const { SESSION_WRITE_ENDPOINTS } = await import( + absPath("engine/session-writes.js") + ); + for (const [endpoint, row] of Object.entries(SESSION_WRITE_ENDPOINTS)) { + assert.deepEqual( + Object.keys(row), + ["capability", "subItem", "enforcement"], + `${endpoint} has a different row shape`, + ); + assert.ok(["hard", "soft", "none"].includes(row.enforcement), endpoint); + } + }); + + // Table-driven: the table IS the assertion, because editing a row is + // a capability decision and has to be reviewed as one. + const TABLE = [ + [ + "DELETE /api/sessions/:id", + { capability: "sessionCrud", subItem: "deleteSession", enforcement: "hard" }, + "the delete destroys rows in the engine's own local_runtime_* tables", + ], + [ + "POST /api/sessions/rename", + { capability: null, subItem: null, enforcement: "none" }, + "a rename writes webui's store and crosses no engine surface", + ], + [ + "POST /api/sessions/cleanup-orphans", + { capability: "sessionCrud", subItem: "deleteSession", enforcement: "hard" }, + "the sweep delegates to #7, so it destroys the same engine rows", + ], + ]; + for (const [endpoint, row, why] of TABLE) { + test(`${endpoint} → ${row.enforcement}${row.capability ? ` on ${row.capability}.${row.subItem}` : ""} (${why})`, async () => { + const { SESSION_WRITE_ENDPOINTS } = await import( + absPath("engine/session-writes.js") + ); + const { ENGINE_CAPABILITY_KEYS } = await import(absPath("engine/index.js")); + assert.deepEqual(SESSION_WRITE_ENDPOINTS[endpoint], row); + if (row.capability) assert.ok(ENGINE_CAPABILITY_KEYS.includes(row.capability)); + }); + } + + test("#6 declares the SAME pair as #7 — the sweep is a delete by another name", async () => { + // If these two ever drift, a provider that cannot delete engine + // sessions could still reach the engine's tables through the + // sweep's back door. The assertion compares against #7's own row, + // not against a copy, so it fails the moment either one moves. + const { SESSION_WRITE_ENDPOINTS } = await import( + absPath("engine/session-writes.js") + ); + assert.deepEqual( + SESSION_WRITE_ENDPOINTS["POST /api/sessions/cleanup-orphans"], + SESSION_WRITE_ENDPOINTS["DELETE /api/sessions/:id"], + ); + }); + + test("an endpoint outside this family is caller confusion, not an engine limitation", async () => { + const { assertSessionWriteCapability } = await import( + absPath("engine/session-writes.js") + ); + assert.throws( + () => assertSessionWriteCapability("DELETE /api/sessions", RUNTIME), + (err) => { + // A caller-typo must NOT answer 501, so the proof is that it is + // not a capability error at all — a positive check on the code + // and message alone would also pass if the error carried both + // by accident. + assert.ok(!isEngineCapabilityNotSupportedError(err)); + assert.equal(err.code, "unknown_session_write_endpoint"); + assert.match(err.message, /not part of the session write family/); + return true; + }, + ); + }); + }); + + describe("resolveSessionWriteProvider / assertSessionWriteCapability", () => { + // Table-driven. Absent means "no provider claims this transport yet" + // (M4), which is NOT the same answer as "capability unavailable" — + // the default `acp` transport must keep deleting sessions, so it + // must NOT throw. + const TRANSPORTS = [ + [RUNTIME, true, "checked", "local-runtime-v2"], + [ACP, false, "unregistered-transport", null], + ["exec", false, "unregistered-transport", null], + ["", false, "unregistered-transport", null], + ]; + for (const [transport, hasProvider, gate, providerId] of TRANSPORTS) { + test(`transport=${JSON.stringify(transport)} → ${gate}`, async () => { + const { assertSessionWriteCapability, resolveSessionWriteProvider } = + await import(absPath("engine/session-writes.js")); + const provider = resolveSessionWriteProvider(transport); + assert.equal(!!provider, hasProvider); + const g = assertSessionWriteCapability("DELETE /api/sessions/:id", transport); + assert.equal(g.gate, gate); + assert.equal(g.provider, providerId); + assert.equal(g.capability, "sessionCrud"); + assert.equal(g.subItem, "deleteSession"); + assert.equal(g.endpoint, "DELETE /api/sessions/:id"); + assert.equal(g.enforcement, "hard"); + }); + } + + test("rename reports no-capability-key on EVERY transport, provider or not", async () => { + // The single most important assertion about #4: renaming a + // session works on a webui-only store and must not become a 501 + // because of anything a provider declares. Checked across all + // four transports so a future `if (provider)` shortcut cannot + // reintroduce the dependency behind the "it only fires on runtime" + // argument. + const { assertSessionWriteCapability } = await import( + absPath("engine/session-writes.js") + ); + for (const [transport] of TRANSPORTS) { + const g = assertSessionWriteCapability("POST /api/sessions/rename", transport); + assert.equal(g.gate, "no-capability-key", transport); + assert.equal(g.capability, null, transport); + assert.equal(g.subItem, null, transport); + assert.equal(g.enforcement, "none", transport); + } + }); + + test("the descriptor carries the six B1–B4 fields plus `enforcement`", async () => { + // A consumer reading `gate.provider` under `acp` must get `null`, + // not `undefined` — the key must EXIST. The six shared fields are + // asserted by name so the families cannot drift apart, and + // `enforcement` is the write family's own addition. + const { assertSessionWriteCapability } = await import( + absPath("engine/session-writes.js") + ); + assert.deepEqual(Object.keys(assertSessionWriteCapability("DELETE /api/sessions/:id", RUNTIME)), [ + "endpoint", + "gate", + "provider", + "capability", + "subItem", + "enforcement", + ]); + }); + }); + + // --------------------------------------------------------------------- + // 2. The hard / none asymmetry, driven from ONE provider fixture + // --------------------------------------------------------------------- + + // The proof that section 1's policy is enforced by code and not by the + // provider's shape. One fixture provider, three endpoints, three + // different answers — and the two "must throw" rows are what stop + // #7/#6 from silently degrading into a no-op delete on a provider that + // cannot delete. + const CAPABILITY_FIXTURES = [ + ["none", { level: "none", reason: "fixture: interface-absent" }], + [ + "partial missing deleteSession", + { level: "partial", missing: ["deleteSession"], reason: "fixture: no delete surface" }, + ], + [ + "partial keeping deleteSession", + { level: "partial", missing: ["getSession"], reason: "fixture: delete present" }, + ], + ["full", { level: "full" }], + ]; + + for (const [name, sessionCrud] of CAPABILITY_FIXTURES) { + test(`provider sessionCrud=${name}: #7 and #6 THROW, #4 never does`, async (t) => { + await setupMocks(t, { acp: {} }); + // `mock.module` replaces the whole namespace; session-writes.js + // reads two names from engine/index.js and the test re-imports the + // facade under a fresh bust so the mock is the one it sees. + t.mock.module(absPath("engine/index.js"), { + namedExports: { + DEFAULT_ENGINE_PROVIDER_ID: "fixture-provider", + getEngineProvider: () => ({ + id: "fixture-provider", + transport: "runtime", + capabilities: { sessionCrud }, + }), + }, + }); + const mod = await import(`${absPath("engine/session-writes.js")}?caps=${bust++}`); + // The gate throws when the declaration withholds `deleteSession` + // itself: `none` withholds the whole capability, and a `partial` + // withholds the sub-item. A `full`, or a `partial` that still + // carries `deleteSession`, passes — which is the sub-item + // granularity the declaration contract exists to provide. + const throws = + sessionCrud.level === "none" || + (sessionCrud.level === "partial" && sessionCrud.missing.includes("deleteSession")); + for (const endpoint of ["DELETE /api/sessions/:id", "POST /api/sessions/cleanup-orphans"]) { + if (throws) { + assert.throws( + () => mod.assertSessionWriteCapability(endpoint, "runtime"), + (err) => { + assert.ok(isEngineCapabilityNotSupportedError(err)); + assert.equal(err.capability, "sessionCrud"); + assert.equal(err.provider, "fixture-provider"); + return true; + }, + `${endpoint} should have thrown for sessionCrud=${name}`, + ); + } else { + const g = mod.assertSessionWriteCapability(endpoint, "runtime"); + assert.equal(g.gate, "checked", `${endpoint} / ${name}`); + } + } + // Rename, whatever the provider says. This is the assertion that + // fails loudly if someone "helpfully" gives #4 a capability. + const rename = mod.assertSessionWriteCapability("POST /api/sessions/rename", "runtime"); + assert.equal(rename.gate, "no-capability-key", name); + assert.equal(rename.provider, "fixture-provider", "it still reports which provider is live"); + }); + } + + test("the hard gate costs ZERO deletions: it throws before the plan reads the store", async (t) => { + // Ordering matters for a destructive endpoint. A gate that ran after + // the store load would still be correct, but a gate that ran after + // the COMMIT would be theatre — so the proof is that the plan + // rejects without ever resolving a target. + await setupMocks(t, { acp: {} }); + registerSessionsStore({ initial: [{ id: "webui-A", title: "A", chat: [] }] }); + t.mock.module(absPath("engine/index.js"), { + namedExports: { + DEFAULT_ENGINE_PROVIDER_ID: "fixture-provider", + getEngineProvider: () => ({ + id: "fixture-provider", + transport: "runtime", + capabilities: { sessionCrud: { level: "none", reason: "fixture" } }, + }), + }, + }); + const mod = await import(`${absPath("engine/session-writes.js")}?order=${bust++}`); + await assert.rejects( + () => mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }), + (err) => { + assert.ok(isEngineCapabilityNotSupportedError(err)); + return true; + }, + ); + // The store is untouched: `getSessionsStore` still holds the record. + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.equal(getSessionsStore().length, 1); + }); + + // --------------------------------------------------------------------- + // 3. The pure derivations + // --------------------------------------------------------------------- + + describe("pure derivations", () => { + test("isMcodeSessionId accepts only the engine's 32-hex shape", async () => { + const { isMcodeSessionId } = await import(absPath("engine/session-writes.js")); + const TABLE = [ + [ORPHAN_SID, true], + [`mvs_${"a".repeat(32)}`, true], + [`mvs_${"A".repeat(32)}`, false, "uppercase hex is not the engine's shape"], + [`mvs_${"a".repeat(31)}`, false], + [`mvs_${"a".repeat(33)}`, false], + ["mvs_", false], + ["webui-A", false], + ["", false], + [null, false], + [undefined, false], + [42, false, "a non-string must not throw — it is simply not an engine sid"], + ]; + for (const [input, expected, why] of TABLE) { + assert.equal(isMcodeSessionId(input), expected, `${JSON.stringify(input)}: ${why || "shape"}`); + } + }); + + test("resolveSessionTarget answers the same three ways for both writes", async () => { + // Rename and delete used to carry this lookup as two identical + // copies. Table-driven over one store so the shared predicate is + // pinned for both. + const { resolveSessionTarget } = await import(absPath("engine/session-writes.js")); + const records = [ + { id: "webui-A", mcodeSessionId: "mvs_11111111111111111111111111111111" }, + { id: "webui-B" }, + ]; + const TABLE = [ + ["webui-A", 0, "webuiId", "matched by the webui uuid"], + ["mvs_11111111111111111111111111111111", 0, "mcodeSessionId", "matched by the bound engine sid"], + ["webui-B", 1, "webuiId", "a record with no engine sid still matches its own id"], + ["nope", -1, null, "an unknown id resolves to nothing, and matchKind is null — not \"unknown\""], + ]; + for (const [id, index, matchKind, why] of TABLE) { + const r = resolveSessionTarget(records, id); + assert.equal(r.index, index, why); + assert.equal(r.matchKind, matchKind, why); + assert.equal(r.target, index >= 0 ? records[index] : null, why); + } + // A non-array store must not throw: the store is a file on disk and + // a corrupt one answers `[]`, never a TypeError inside a gate. + assert.deepEqual(resolveSessionTarget(null, "x"), { index: -1, matchKind: null, target: null }); + }); + + test("the orphan rule: empty AND default-titled AND older than 24h", async () => { + const { isOrphanSessionRecord, ORPHAN_STALE_MS } = await import( + absPath("engine/session-writes.js") + ); + const NOW = 1_700_000_000_000; + const old = NOW - ORPHAN_STALE_MS - 1; + const TABLE = [ + [{ id: "a", title: "Untitled", chat: [], updatedAt: old }, true, "the canonical leftover"], + [{ id: "b", title: "New session", chat: [], updatedAt: old }, true, "the other default name"], + [{ id: "c", title: "对话 7", chat: [], updatedAt: old }, true, "the numbered default"], + [{ id: "d", title: "Untitled", chat: [], updatedAt: NOW }, false, "too fresh"], + [ + { id: "e", title: "Untitled", chat: [], updatedAt: NOW - ORPHAN_STALE_MS + 1 }, + false, + "one millisecond inside the window is still fresh", + ], + [ + { id: "f", title: "Untitled", chat: [], updatedAt: NOW - ORPHAN_STALE_MS }, + true, + "exactly at the threshold is stale — the rule is `<`, not `<=`", + ], + [{ id: "g", title: "Untitled", chat: ["● hi"], updatedAt: old }, false, "has chat"], + [{ id: "h", title: "Real work", chat: [], updatedAt: old }, false, "not a default title"], + [{ id: "i", title: "Untitled", chat: [], updatedAt: 0 }, true, "updatedAt 0 is falsy, so the age check is skipped — preserved"], + [{ id: "j", title: " Untitled ", chat: [], updatedAt: old }, true, "titles are trimmed before matching"], + [{ id: "k", title: "对话7", chat: [], updatedAt: old }, false, "the numbered form needs the space"], + [{ id: "", title: "Untitled", chat: [], updatedAt: old }, false, "no id"], + [null, false, "a null record"], + [{ title: "Untitled", chat: [], updatedAt: old }, false, "no id"], + [{ id: "m", title: "Untitled", updatedAt: old }, true, "a missing chat counts as empty"], + ]; + for (const [record, expected, why] of TABLE) { + assert.equal(isOrphanSessionRecord(record, NOW, ORPHAN_STALE_MS), expected, why); + } + }); + + test("selectOrphanSessionIds keeps store order and survives a non-array", async () => { + const { selectOrphanSessionIds, ORPHAN_STALE_MS } = await import( + absPath("engine/session-writes.js") + ); + const now = 1_700_000_000_000; + const old = now - ORPHAN_STALE_MS - 1; + assert.deepEqual( + selectOrphanSessionIds( + [ + { id: "keep-me", title: "Real", chat: [], updatedAt: old }, + { id: "b", title: "Untitled", chat: [], updatedAt: old }, + { id: "a", title: "Untitled", chat: [], updatedAt: old }, + ], + { now }, + ), + ["b", "a"], + "store order, not sorted order — the ids are reported in the order they would be deleted", + ); + assert.deepEqual(selectOrphanSessionIds(null, { now }), []); + }); + + test("the two fan-out predicates really are different predicates", async () => { + // The temptation this test exists to kill: one shared + // "is this client in this session" helper. It would be wrong in + // both directions — clearing a tab that was never deleted, and + // blanking the title of a tab bound to a DIFFERENT wrapper record. + const { clientMatchesDeletedSession, clientMatchesRenamedSession } = await import( + absPath("engine/session-writes.js") + ); + const record = { id: "webui-A", mcodeSessionId: "mvs_sid_A" }; + const inRecord = { sessionId: "webui-A", mcodeSessionId: null }; + const byEngineSid = { sessionId: "webui-OTHER", mcodeSessionId: "mvs_sid_A" }; + const byRequestId = { sessionId: "webui-THIRD", mcodeSessionId: "mvs_sid_B" }; + + assert.equal(clientMatchesRenamedSession(inRecord, record), true, "rename: same webui id"); + assert.equal(clientMatchesRenamedSession(byEngineSid, record), true, "rename: same engine sid"); + assert.equal(clientMatchesRenamedSession(byRequestId, record), false, "rename: unrelated tab"); + + assert.equal(clientMatchesDeletedSession(inRecord, record, "webui-A"), true, "delete: same webui id"); + assert.equal(clientMatchesDeletedSession(byEngineSid, record, "webui-A"), false, + "delete: a tab bound to the record's engine sid under ANOTHER wrapper is a different record and must not be cleared"); + assert.equal(clientMatchesDeletedSession(byRequestId, record, "mvs_sid_B"), true, + "delete: matches the id the REQUEST named, which is the orphan branch's only handle"); + }); + + test("the delete reset clears identity, title, chat and the three usage counters", async () => { + const { applyDeletedSessionToClientState } = await import( + absPath("engine/session-writes.js") + ); + const cs = { + sessionId: "webui-A", + mcodeSessionId: "mvs_sid_A", + sessionTitle: "A", + chat: ["● hi", "● there"], + usage: { sessionInput: 10, sessionOutput: 20, sessionTotal: 30, cost: 1.5 }, + somethingElse: "kept", + }; + applyDeletedSessionToClientState(cs); + assert.equal(cs.sessionId, null); + assert.equal(cs.mcodeSessionId, null); + assert.equal(cs.sessionTitle, "Untitled"); + assert.deepEqual(cs.chat, []); + assert.equal(cs.usage.sessionInput, 0); + assert.equal(cs.usage.sessionOutput, 0); + assert.equal(cs.usage.sessionTotal, 0); + assert.equal(cs.usage.cost, 1.5, "unrelated usage fields survive"); + assert.equal(cs.somethingElse, "kept", "the reset touches only what it names"); + }); + + test("the orphan branch does NOT zero usage — the asymmetry is a parameter, not an accident", async () => { + const { applyDeletedSessionToClientState } = await import( + absPath("engine/session-writes.js") + ); + const cs = { + sessionId: null, + mcodeSessionId: ORPHAN_SID, + sessionTitle: "Orphan", + chat: [], + usage: { sessionInput: 10, sessionOutput: 20, sessionTotal: 30 }, + }; + applyDeletedSessionToClientState(cs, { resetUsage: false }); + assert.equal(cs.sessionTitle, "Untitled", "the identity and title are still cleared"); + assert.deepEqual(cs.chat, []); + assert.equal(cs.usage.sessionTotal, 30, "an orphan has no webui record, so no tab accrued usage for it"); + }); + }); + + // --------------------------------------------------------------------- + // 4. The delete red lines + // --------------------------------------------------------------------- + + // The real store / cache / SQL collaborators, journalled. Every + // mutation appends its name in the order it happened, because for a + // delete the ORDER is the feature and an end-state assertion cannot see + // a resurrected session or an out-of-order cache drop. + async function loadWritePath(t, options = {}) { + await setupMocks(t, { acp: {}, sessions: { initial: options.records || [] } }); + registerAcpMock({ + shutdownMcodeAcpSingleton: () => { + journal.push("kill-acp-child"); + }, + dropMcodeSessionFromCache: (sid) => { + journal.push(`drop-cache:${sid}`); + }, + }); + const dbCalls = []; + mockAll( + t, + "lib/mcode-session-delete.js", + { + deleteMcodeSessionFromDb: (sid, o) => { + journal.push(`sql:${sid}:dryRun=${!!o.dryRun}`); + dbCalls.push({ sid, dryRun: !!o.dryRun, db: o.MCODE_RUNTIME_DB }); + return ( + options.dbResult || { + ok: true, + outcome: "deleted", + log: ["local_runtime_sessions:1"], + totalRowsDeleted: 1, + tablesAbsent: 0, + } + ); + }, + }, + ["deleteMcodeSessionFromDb", "MCODE_SESSION_DELETE_TABLES"], + ); + mockAll( + t, + "lib/session-tree.js", + { + invalidateSessionTree: () => { + journal.push("invalidate-tree"); + }, + getSessionTree: () => { + journal.push("getSessionTree"); + return { ok: true, tree: [] }; + }, + }, + ["getSessionTree", "invalidateSessionTree"], + ); + const clients = new Map(); + const pushes = []; + mockAll( + t, + "lib/state-bus.js", + { + clients, + pushStateFor: (c) => { + journal.push(`push:${c}`); + pushes.push(c); + }, + runChatViewChat: () => ({}), + makeClientState: () => ({ usage: {} }), + }, + [ + "clients", + "pushStateFor", + "runChatViewChat", + "makeClientState", + "setState", + "getClient", + "sseByCid", + "pushAlert", + ], + ); + return { + dbCalls, + clients, + pushes, + mod: await import(`${absPath("engine/session-writes.js")}?w=${bust++}`), + }; + } + + beforeEach(() => { + resetJournal(); + }); + + describe("RED LINE — a deleted session does not come back", () => { + test("the write path runs kill → SQL → scoped cache drop, in that order", async (t) => { + // THE ordering assertion. The long-lived mcode ACP child holds the + // session in memory and rewrites its registry row on its next + // request, so a delete that removes the rows but leaves the child + // alive produces a session that reappears on the next read. The + // tree-cache drop is asserted FIRST because it must precede the + // engine write: a concurrent read must not be able to repopulate + // the cache from the pre-delete database. + const { mod, dbCalls } = await loadWritePath(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● hi"] }, + { id: "webui-B", title: "B", chat: [] }, + ], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.deepEqual(journal, [ + "invalidate-tree", + "kill-acp-child", + "drop-cache:mvs_sid_A", + "sql:mvs_sid_A:dryRun=false", + // The requesting tab is pushed even though no tab matched it — + // see the fan-out section. It is part of the delete, not after it. + "push:tab-1", + ]); + assert.equal(dbCalls.length, 1, "exactly one engine delete, for the linked sid"); + assert.equal(dbCalls[0].dryRun, false, "a real delete is never a dry run"); + }); + + test("only the DELETED sid leaves the cache — the other session is untouched", async (t) => { + // The regression this guards is the sidebar flash: invalidating the + // WHOLE cache empties the list, refills it, and reads to the user + // like the delete failed. The per-sid drop is why a 42-entry + // sidebar goes to 41 and stays there. + const { mod, clients } = await loadWritePath(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }, + { id: "webui-B", mcodeSessionId: "mvs_sid_B", title: "B", chat: [] }, + ], + }); + clients.set("tab-1", { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", usage: {} }); + clients.set("tab-2", { sessionId: "webui-B", mcodeSessionId: "mvs_sid_B", usage: {} }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.equal( + journal.filter((j) => j.startsWith("drop-cache:")).length, + 1, + "exactly one cache entry dropped", + ); + assert.ok(!journal.includes("drop-cache:mvs_sid_B"), "the untouched session keeps its cache entry"); + assert.deepEqual( + w.records.map((r) => r.id), + ["webui-B"], + "the sibling record survives in the persisted store", + ); + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.deepEqual( + getSessionsStore().map((r) => r.id), + ["webui-B"], + "and it is gone from the STORE, not just from the returned array", + ); + }); + + test("it is a TRUE delete: re-deleting the same sid still reaches the engine", async (t) => { + // The distinction the batch's red line 3 turns on — a real delete + // versus a frontend fake. After the first delete the webui record + // is gone, so the SECOND delete of the same engine sid resolves as + // an ORPHAN and goes straight to the SQL deleter. If the first + // delete had only hidden the record (or if the store save were + // skipped), this second call would resolve `webuiId` again and the + // engine would never learn the session is gone. + const { mod, dbCalls } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + const first = await mod.planEngineSessionDelete({ id: "mvs_sid_A", transport: RUNTIME }); + assert.equal(first.matchKind, "mcodeSessionId"); + await mod.commitEngineSessionDelete({ plan: first, cid: "tab-1" }); + + resetJournal(); + const second = await mod.planEngineSessionDelete({ id: "mvs_sid_A", transport: RUNTIME }); + assert.equal(second.isOrphan, true, "the record is really gone from the store"); + assert.equal(second.matchKind, null); + const w = await mod.commitEngineOrphanSessionDelete({ plan: second, cs: null, cid: "tab-1" }); + assert.equal(w.failed, false); + assert.equal(dbCalls.length, 2, "the engine was told twice — the delete is not a UI illusion"); + assert.equal(dbCalls[1].sid, "mvs_sid_A"); + }); + }); + + describe("RED LINE — deleting a session is not deleting files", () => { + // The session's ARTEFACTS live on the filesystem under the run + // directory (`~/tmp/run_*/`), and red line 4 makes the side file + // tree part of the contract. A session delete removes rows in a + // SQLite database; it must not remove a single byte of the user's + // output. This test puts a real file there and checks it afterwards. + test("a real run-directory artefact survives the delete", async (t) => { + const dir = mkTmpDir("webui-session-writes-b5-"); + try { + const runDir = join(dir, "run_20260920_120000"); + mkdirSync(runDir, { recursive: true }); + const artefact = join(runDir, "build.log"); + writeFileSync(artefact, "compiled output the user still wants\n"); + const { mod } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● done"] }], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.equal(w.payload.ok, true, "the delete itself succeeded"); + assert.ok(existsSync(artefact), "the artefact file is still on disk"); + assert.equal(readFileSync(artefact, "utf8"), "compiled output the user still wants\n"); + } finally { + rmTmpDir(dir); + } + }); + + test("the same holds for a dryRun preview and for the orphan branch", async (t) => { + const dir = mkTmpDir("webui-session-writes-b5-"); + try { + const artefact = join(dir, "report.md"); + writeFileSync(artefact, "# notes\n"); + const { mod } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + await mod.previewEngineSessionDelete({ plan }); + resetJournal(); + const orphan = await mod.planEngineSessionDelete({ id: ORPHAN_SID, transport: RUNTIME }); + await mod.commitEngineOrphanSessionDelete({ plan: orphan, cs: null, cid: "tab-1" }); + assert.ok(existsSync(artefact), "neither the preview nor the orphan branch touches files"); + } finally { + rmTmpDir(dir); + } + }); + + test("the 32-table delete list is unchanged — the debt is recorded, not silently collected", async () => { + // The plan for this batch annotated `lib/mcode-session-delete.js` + // "delete". It is kept, because `lib/acp-client.js` imports from it + // and four test files bind to the specifier. This test pins the + // consequence: the table list is still exported, still has 32 + // entries, and the facade reaches it rather than duplicating it. + // A future collection that moves the list has to change this + // assertion in the same commit — which is the point. + const { MCODE_SESSION_DELETE_TABLES } = await import( + absPath("lib/mcode-session-delete.js") + ); + assert.equal(MCODE_SESSION_DELETE_TABLES.length, 32, "the 32-table list, still owned by the lib module"); + assert.equal(MCODE_SESSION_DELETE_TABLES[0], "local_runtime_sessions"); + assert.ok(MCODE_SESSION_DELETE_TABLES.includes("local_runtime_token_usage")); + const src = readFileSync(fileURLToPath(absPath("engine/session-writes.js")), "utf8"); + // The facade must NOT have grown its own copy of the list, or its + // own SQL. A second list is precisely how two writers end up + // deleting different sets of rows. The check is on SQL VERBS + // rather than on a table name, because the module's own comments + // legitimately name the tables while explaining what it does not + // do; a `DELETE FROM` or `SELECT` in this file would be the real + // smell. + for (const verb of ["DELETE FROM", "SELECT ", "INSERT ", "UPDATE ", "prepare("]) { + assert.ok( + !src.includes(verb), + `the facade must issue no SQL, found ${JSON.stringify(verb)} — the delete SQL stays in the lib module it forwards to`, + ); + } + // And the forwarding itself is real: the facade reaches that module + // through a dynamic import, not a second static one. + assert.match( + src, + /import\("\.\.\/lib\/mcode-session-delete\.js"\)/, + "the facade forwards to lib/mcode-session-delete.js through a lazy import", + ); + assert.ok( + !/^import .*mcode-session-delete/m.test(src), + "and never through a static one — a static import would put the SQL on the boot path", + ); + }); + }); + + describe("RED LINE — a running session has defined semantics", () => { + test("deleting an in-flight session kills the child that is driving the turn", async (t) => { + // There is no "refuse to delete a running session" guard, and this + // pins the semantics that DO exist rather than leaving it implied: + // the user asked, the child stops, the rows go. Recorded as KNOWN + // DEBT in the facade header — "refuse" is a defensible product + // decision, but it is not this batch's to make, and an unstated + // behaviour is worse than a stated one. + const { mod, clients } = await loadWritePath(t, { + records: [ + { + id: "webui-A", + mcodeSessionId: "mvs_sid_A", + title: "Running", + chat: ["● working"], + }, + ], + }); + const cs = { + sessionId: "webui-A", + mcodeSessionId: "mvs_sid_A", + sessionTitle: "Running", + chat: ["● working"], + running: { active: true, pid: 4242 }, + usage: { sessionTotal: 99 }, + }; + clients.set("tab-1", cs); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.ok(journal.includes("kill-acp-child"), "the child driving the turn is stopped"); + assert.ok( + journal.indexOf("kill-acp-child") < journal.findIndex((j) => j.startsWith("sql:")), + "and it is stopped BEFORE the rows go — otherwise it rewrites them", + ); + assert.equal(w.payload.ok, true, "and the delete proceeds — there is no refusal semantics"); + assert.equal(w.deletedItem.title, "Running"); + assert.equal(cs.mcodeSessionId, null, "the tab is not left pointing at a dead turn"); + }); + + test("the requesting tab's in-flight state is reset by the fan-out, not left dangling", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● x"] }], + }); + const cs = { + sessionId: "webui-A", + mcodeSessionId: "mvs_sid_A", + sessionTitle: "A", + chat: ["● x"], + usage: { sessionInput: 5, sessionOutput: 6, sessionTotal: 11 }, + }; + clients.set("tab-1", cs); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.equal(cs.sessionId, null); + assert.equal(cs.mcodeSessionId, null); + assert.equal(cs.sessionTitle, "Untitled"); + assert.deepEqual(cs.chat, []); + assert.equal(cs.usage.sessionTotal, 0, "the tab stops reporting the deleted session's spend"); + assert.deepEqual(pushes, ["tab-1"]); + assert.equal(w.touchedCids.length, 1); + }); + }); + + describe("RED LINE — the delete fans out to every other tab", () => { + test("a second tab inside the same session is cleared and pushed", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● x"] }], + }); + const tab1 = { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", sessionTitle: "A", chat: ["● x"], usage: { sessionTotal: 3 } }; + const tab2 = { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", sessionTitle: "A", chat: ["● x"], usage: { sessionTotal: 4 } }; + const tab3 = { sessionId: "webui-Z", mcodeSessionId: "mvs_sid_Z", sessionTitle: "Z", chat: ["● z"], usage: { sessionTotal: 5 } }; + clients.set("tab-1", tab1); + clients.set("tab-2", tab2); + clients.set("tab-3", tab3); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.equal(tab1.mcodeSessionId, null); + assert.equal(tab2.mcodeSessionId, null); + assert.equal(tab2.chat.length, 0, "the OTHER tab loses the entry too — this is the cross-tab red line"); + assert.equal(tab3.mcodeSessionId, "mvs_sid_Z", "an unrelated tab is left completely alone"); + assert.equal(tab3.usage.sessionTotal, 5); + assert.deepEqual(pushes.sort(), ["tab-1", "tab-2"], "both affected tabs are pushed"); + assert.equal(w.touchedCids.length, 2); + assert.equal(w.payload.ok, true); + }); + + test("a tab that matched nothing still gets exactly one push, so it cannot render a ghost", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + clients.set("tab-elsewhere", { sessionId: "webui-Z", mcodeSessionId: "mvs_sid_Z", usage: {} }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.commitEngineSessionDelete({ plan, cid: "tab-1" }); + assert.deepEqual(pushes, ["tab-1"], "the requesting tab is pushed even though it matched nothing"); + assert.equal(w.touchedCids.length, 1); + }); + + test("the orphan branch clears ONLY the requesting tab — there is no record for others to be inside", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { records: [] }); + const cs = { sessionId: "webui-OTHER", mcodeSessionId: ORPHAN_SID, sessionTitle: "Orphan", chat: ["● x"], usage: { sessionTotal: 7 } }; + clients.set("tab-1", cs); + const plan = await mod.planEngineSessionDelete({ id: ORPHAN_SID, transport: RUNTIME }); + assert.equal(plan.isOrphan, true); + const w = await mod.commitEngineOrphanSessionDelete({ plan, cs, cid: "tab-1" }); + assert.equal(w.failed, false); + assert.equal(cs.mcodeSessionId, null, "the tab that was sitting on the orphan is cleared"); + assert.equal(cs.sessionTitle, "Untitled"); + assert.equal(cs.usage.sessionTotal, 7, "and its usage is NOT zeroed — the orphan branch's documented asymmetry"); + assert.deepEqual(pushes, ["tab-1"], "only that one tab is pushed"); + assert.equal(w.payload.matchKind, "orphan_mcode"); + }); + + test("the orphan branch leaves a tab that was NOT on the orphan alone", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { records: [] }); + const other = { sessionId: "webui-Z", mcodeSessionId: "mvs_sid_Z", sessionTitle: "Z", usage: {} }; + clients.set("tab-2", other); + const plan = await mod.planEngineSessionDelete({ id: ORPHAN_SID, transport: RUNTIME }); + await mod.commitEngineOrphanSessionDelete({ plan, cs: null, cid: "tab-1" }); + assert.equal(other.mcodeSessionId, "mvs_sid_Z"); + assert.deepEqual(pushes, [], "no tab matched, so no tab was disturbed"); + }); + }); + + // --------------------------------------------------------------------- + // 5. The byte-for-byte preview shapes + // --------------------------------------------------------------------- + + describe("preview shapes — #6's dryRun body is a hard red line for this batch", () => { + test("the cleanup-orphans dryRun body is exactly four keys, in order", async (t) => { + // Compared as a STRING, not as a parsed object: key ORDER is part + // of a byte-for-byte contract, and `deepEqual` on objects would not + // notice a reshuffle. + await setupMocks(t, { acp: {} }); + const mod = await import(`${absPath("engine/session-writes.js")}?shape=${bust++}`); + // The store read is against the real config's SESSIONS_DB, which + // does not exist in this environment, so the answer is the empty + // case — which is the shape most likely to be "simplified". + const sweep = await mod.readOrphanSessionWriteIds({ transport: RUNTIME }); + assert.equal( + JSON.stringify(sweep.payload), + '{"ok":true,"dryRun":true,"count":0,"ids":[]}', + ); + assert.equal(sweep.gate.gate, "checked"); + assert.equal(sweep.gate.enforcement, "hard"); + }); + + test("a populated sweep answers the same four keys with the selected ids", async (t) => { + await setupMocks(t, { acp: {} }); + const { selectOrphanSessionIds } = await import( + `${absPath("engine/session-writes.js")}?shape=${bust++}` + ); + // The selection is pure, so the populated case is pinned through it + // while the SHAPE stays pinned through the real read above. The + // response is the same object the read would build. + const ids = selectOrphanSessionIds( + [ + { id: "old-1", title: "Untitled", chat: [], updatedAt: 1 }, + { id: "keep", title: "Real", chat: [], updatedAt: 1 }, + { id: "old-2", title: "对话 3", chat: [], updatedAt: 1 }, + ], + { now: Number.MAX_SAFE_INTEGER }, + ); + assert.equal( + JSON.stringify({ ok: true, dryRun: true, count: ids.length, ids }), + '{"ok":true,"dryRun":true,"count":2,"ids":["old-1","old-2"]}', + ); + }); + + test("#7's dryRun body keeps its four keys and the webuiEntryWouldBeDeleted block", async (t) => { + const { mod } = await loadWritePath(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A on tmp", chat: ["● hi"] }, + ], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + const w = await mod.previewEngineSessionDelete({ plan }); + assert.equal( + JSON.stringify(w.payload), + JSON.stringify({ + ok: true, + dryRun: true, + matchKind: "webuiId", + mcodeDbDel: { ok: true, outcome: "deleted", log: ["local_runtime_sessions:1"], totalRowsDeleted: 1, tablesAbsent: 0 }, + webuiEntryWouldBeDeleted: { id: "webui-A", title: "A on tmp", mcodeSessionId: "mvs_sid_A" }, + }), + ); + assert.deepEqual(Object.keys(w.payload), [ + "ok", + "dryRun", + "matchKind", + "mcodeDbDel", + "webuiEntryWouldBeDeleted", + ]); + }); + + test("a webui-only session with no engine sid still previews, with an empty log", async (t) => { + // A record that never reached the engine has no rows to count. + // Refusing to preview for those would be a new failure mode, and + // the empty-log literal is the endpoint's own. + const { mod, dbCalls } = await loadWritePath(t, { + records: [{ id: "webui-B", title: "B", chat: [] }], + }); + const plan = await mod.planEngineSessionDelete({ id: "webui-B", transport: RUNTIME }); + const w = await mod.previewEngineSessionDelete({ plan }); + assert.deepEqual(w.mcodeDbDel, { ok: true, dryRun: true, log: [], totalRows: 0 }); + assert.equal(w.payload.webuiEntryWouldBeDeleted.mcodeSessionId, undefined); + assert.equal(dbCalls.length, 0, "and the SQL deleter is never asked about a non-sid"); + }); + + test("a dryRun mutates nothing: no kill, no cache drop, no tree invalidation, no store write", async (t) => { + // A preview that shuts down the user's ACP child is a side effect + // the `?dryRun=true` contract does not include, and a preview that + // drops the tree cache is a lie ("nothing changed" while the + // sidebar re-renders). The journal is empty except for the + // read-only SQL count. + const { mod, clients } = await loadWritePath(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + clients.set("tab-1", { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", sessionTitle: "A", chat: ["● x"], usage: {} }); + const plan = await mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME }); + await mod.previewEngineSessionDelete({ plan }); + assert.deepEqual( + journal, + ["sql:mvs_sid_A:dryRun=true"], + "the ONLY thing a preview does is ask the SQL layer to count", + ); + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.equal(getSessionsStore().length, 1, "the record is still there after a preview"); + assert.equal( + clients.get("tab-1").mcodeSessionId, + "mvs_sid_A", + "and the tab is still inside it", + ); + }); + }); + + // --------------------------------------------------------------------- + // 6. The rename write + // --------------------------------------------------------------------- + + describe("#4 rename — a webui-side label, and nothing else", () => { + test("a rename writes the store, drops the tree cache and pushes every matching tab", async (t) => { + const { mod, clients, pushes } = await loadWritePath(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "Old", chat: ["● x"] }, + { id: "webui-B", title: "B", chat: [] }, + ], + }); + const tab1 = { sessionId: "webui-A", mcodeSessionId: "mvs_sid_A", sessionTitle: "Old", usage: {} }; + const tab2 = { sessionId: "webui-OTHER", mcodeSessionId: "mvs_sid_A", sessionTitle: "Old", usage: {} }; + const tab3 = { sessionId: "webui-B", mcodeSessionId: null, sessionTitle: "B", usage: {} }; + clients.set("tab-1", tab1); + clients.set("tab-2", tab2); + clients.set("tab-3", tab3); + const w = await mod.applyEngineSessionRename({ id: "webui-A", title: "New", cid: "tab-1" }); + assert.equal(w.outcome, "ok"); + assert.equal(w.matchKind, "webuiId"); + assert.equal(w.from, "Old"); + assert.deepEqual(w.payload, { + ok: true, + session: { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "New", titleCustom: true }, + }); + assert.equal(tab1.sessionTitle, "New"); + assert.equal(tab2.sessionTitle, "New", "a tab bound to the same engine sid sees the new label too"); + assert.equal(tab3.sessionTitle, "B", "an unrelated tab is untouched"); + assert.deepEqual(pushes.sort(), ["tab-1", "tab-2"]); + // The rename touches NO engine surface: no SQL, no kill, no cache + // drop. Only the tree cache, because the sidebar projects titles + // from the engine and would otherwise show a stale one. + assert.deepEqual(journal, ["invalidate-tree", "push:tab-1", "push:tab-2"]); + }); + + test("a bare mvs_ id gets an overlay record; an unknown id is a 404 outcome", async (t) => { + // `setupMocks` owns `lib/sessions.js` in this test context and + // node:test refuses a second registration for the same specifier + // (ERR_INVALID_STATE), so the store comes from the shared helper's + // mutable holder. M3-B5 added `ensureOverlayForMcodeSid` to that + // helper's namespace for exactly this case; the fixture therefore + // observes the real single-identity behaviour instead of a private + // stub that could drift from it. + const { mod, clients } = await loadWritePath(t, { records: [] }); + const w = await mod.applyEngineSessionRename({ id: ORPHAN_SID, title: "Named", cid: "tab-1" }); + assert.equal(w.outcome, "ok"); + assert.equal(w.matchKind, "orphan_mcode"); + assert.equal(w.from, "Mcode session", "the placeholder title the overlay was born with"); + assert.equal(w.to, "Named"); + assert.equal(w.item.id, ORPHAN_SID, "the overlay's webui id IS the engine sid (single identity)"); + assert.equal(w.item.mcodeSessionId, ORPHAN_SID); + assert.equal(w.item.titleCustom, true); + assert.deepEqual(w.payload, { + ok: true, + session: { id: ORPHAN_SID, mcodeSessionId: ORPHAN_SID, title: "Named", titleCustom: true }, + }); + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.deepEqual( + getSessionsStore().map((r) => r.id), + [ORPHAN_SID], + "and the overlay is PERSISTED — a rename that is not saved is a label the next load loses", + ); + // No workspace is stamped onto someone else's record + // (webui-parity 63, defect F): the fixture's store never carried a + // workspace argument and the overlay's is "". + assert.equal(getSessionsStore()[0].workspace, ""); + + // A webui uuid that resolves to nothing is a 404, and it must NOT + // fabricate a record — a wrong id should say so. + const missing = await mod.applyEngineSessionRename({ id: "no-such-id", title: "Named", cid: "tab-1" }); + assert.equal(missing.outcome, "not_found"); + assert.deepEqual(missing.payload, { ok: false, error: "session not found" }); + assert.equal( + getSessionsStore().length, + 1, + "no second overlay was fabricated for the unknown id", + ); + void clients; + }); + + test("rename never throws a capability error, whatever the provider says", async (t) => { + // The end-to-end statement of the `null` declaration row: a + // provider that has NO session CRUD at all cannot stop a rename, + // because a rename does not ask the engine for anything. + await setupMocks(t, { acp: {}, sessions: { initial: [{ id: "webui-A", title: "Old", chat: [] }] } }); + t.mock.module(absPath("engine/index.js"), { + namedExports: { + DEFAULT_ENGINE_PROVIDER_ID: "fixture-provider", + getEngineProvider: () => ({ + id: "fixture-provider", + transport: "runtime", + capabilities: { sessionCrud: { level: "none", reason: "fixture: no CRUD at all" } }, + }), + }, + }); + const mod = await import(`${absPath("engine/session-writes.js")}?ren=${bust++}`); + // planEngineSessionDelete WOULD throw here — that is the point of + // the hard row. Rename must not. + await assert.rejects(() => mod.planEngineSessionDelete({ id: "webui-A", transport: RUNTIME })); + const w = await mod.applyEngineSessionRename({ id: "webui-A", title: "New", cid: "tab-1" }); + assert.equal(w.outcome, "ok"); + assert.equal(w.gate.gate, "no-capability-key"); + }); + }); + + // --------------------------------------------------------------------- + // 7. The route — and the proof that the facade mock took + // --------------------------------------------------------------------- + + describe("the routes ask the facade and keep their own HTTP contract", () => { + // Every export `routes/sessions.js` binds from the facade. A + // `mock.module` that omits one of these makes the route fail at + // INSTANTIATION with a `SyntaxError` that reads like a product bug; + // the ones a case does not want are filled with throwers. + const FACADE_EXPORTS = [ + "ORPHAN_STALE_MS", + "SESSION_WRITE_ENDPOINTS", + "applyDeletedSessionToClientState", + "applyEngineSessionRename", + "applyRenamedSessionToClientState", + "assertSessionWriteCapability", + "clientMatchesDeletedSession", + "clientMatchesRenamedSession", + "commitEngineOrphanSessionDelete", + "commitEngineSessionDelete", + "isMcodeSessionId", + "isOrphanSessionRecord", + "planEngineSessionDelete", + "previewEngineSessionDelete", + "readOrphanSessionWriteIds", + "resolveSessionTarget", + "resolveSessionWriteProvider", + "selectOrphanSessionIds", + ]; + const NOT_STUBBED = (name) => () => { + throw new Error(`B5 route test called ${name}, which this case did not stub`); + }; + function mockFacade(t, overrides) { + // The route legitimately calls one PURE facade export before it + // calls any writer: `isMcodeSessionId`, for the arrival log line + // and the `authorize()` context. It is filled with the REAL + // implementation rather than a thrower, because it has no side + // effects and a stubbed copy could disagree with the engine module + // the CONTROL test exercises. Every export that WRITES keeps the + // thrower, so an unexpected mutation stays loud. + const namedExports = { + isMcodeSessionId: (id) => typeof id === "string" && /^mvs_[a-f0-9]{32}$/.test(id), + ORPHAN_STALE_MS: 24 * 60 * 60 * 1000, + }; + for (const name of FACADE_EXPORTS) { + if (namedExports[name] === undefined) namedExports[name] = NOT_STUBBED(name); + } + Object.assign(namedExports, overrides); + t.mock.module(absPath("engine/session-writes.js"), { namedExports }); + } + + test("#7 writes the facade's 200 body verbatim, with the charset header", async (t) => { + await setupMocks(t, { acp: {} }); + const payload = { + ok: true, + deleted: "webui-A", + matchKind: "webuiId", + dryRun: false, + remaining: 3, + mcodeDbDel: { ok: true, log: ["local_runtime_sessions:1"] }, + }; + mockFacade(t, { + planEngineSessionDelete: async () => ({ + id: "webui-A", + records: [], + index: 0, + matchKind: "webuiId", + target: { id: "webui-A" }, + isOrphan: false, + chatLen: 0, + gate: { gate: "checked" }, + transport: RUNTIME, + }), + commitEngineSessionDelete: async () => ({ + deletedItem: { id: "webui-A", title: "A" }, + records: [{}, {}, {}], + mcodeDbDel: payload.mcodeDbDel, + touchedCids: ["tab-1"], + payload, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession( + { url: "/api/sessions/webui-A" }, + res, + { cs: {}, cid: "tab-1", pathname: "/api/sessions/webui-A" }, + ), + ); + assert.equal(res.written[0].status, 200); + assert.equal( + res.written[0].headers["Content-Type"], + "application/json; charset=utf-8", + ); + assert.equal(res.written[1].body, JSON.stringify(payload)); + }); + + // Table-driven: the status and the Content-Type per branch. The + // charset is NOT uniform in the pre-facade code — the 400/404/500 + // branches send bare `application/json` while the 200/403 branches + // send the charset form — and a refactor that "tidied" that would be + // a silent contract change, so the exact pair is pinned per branch. + const STATUSES = [ + ["400", { "Content-Type": "application/json" }, "/api/sessions/"], + ["404", { "Content-Type": "application/json" }, "/api/sessions/no-such-webui-id"], + ]; + for (const [status, headers, pathname] of STATUSES) { + test(`#7 answers ${status} with ${JSON.stringify(headers)} — unchanged`, async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + planEngineSessionDelete: async () => ({ + id: "no-such-webui-id", + records: [], + index: -1, + matchKind: null, + target: null, + isOrphan: true, + chatLen: 0, + gate: { gate: "checked" }, + transport: RUNTIME, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession({ url: pathname }, res, { + cs: {}, + cid: "tab-1", + pathname, + }), + ); + assert.equal(res.written[0].status, Number(status)); + assert.deepEqual(res.written[0].headers, headers); + if (status === "404") { + assert.deepEqual(JSON.parse(res.written[1].body), { + ok: false, + error: "session not found", + }); + } + }); + } + + test("#7 still 403s on a declined authorize(), before anything is mutated", async (t) => { + await setupMocks(t, { acp: {} }); + let committed = false; + mockFacade(t, { + planEngineSessionDelete: async () => ({ + id: "webui-A", + records: [{ id: "webui-A", title: "A", chat: ["● hi"] }], + index: 0, + matchKind: "webuiId", + target: { id: "webui-A" }, + isOrphan: false, + chatLen: 1, + gate: { gate: "checked" }, + transport: RUNTIME, + }), + commitEngineSessionDelete: async () => { + committed = true; + return { payload: { ok: true } }; + }, + }); + const route = await loadRoute(); + const res = mkRes(); + await withDecisions( + () => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + { approve: false }, + ); + assert.equal(res.written[0].status, 403); + assert.equal( + JSON.parse(res.written[1].body).error, + "authorize declined", + ); + assert.equal(committed, false, "a declined gate must not reach the commit at all"); + }); + + test("#4 keeps its three 400 bodies and never reaches the facade", async (t) => { + await setupMocks(t, { acp: {} }); + let called = false; + mockFacade(t, { + applyEngineSessionRename: async () => { + called = true; + return { outcome: "ok" }; + }, + }); + const route = await loadRoute(); + // Table-driven over the three validation failures, all of which are + // request validation and therefore stay in the route. + const CASES = [ + [{ title: "New" }, "id required"], + [{ id: "webui-A" }, "title required"], + [{ id: "webui-A", title: " " }, "title required"], + [{ id: "webui-A", title: "x".repeat(201) }, "title too long (max 200)"], + ]; + for (const [body, error] of CASES) { + const res = mkRes(); + await route.handleRenameSession(jsonReq(body), res, { cid: "tab-1" }); + assert.equal(res.written[0].status, 400, JSON.stringify(body).slice(0, 40)); + assert.equal( + res.written[0].headers["Content-Type"], + "application/json; charset=utf-8", + ); + assert.deepEqual(JSON.parse(res.written[1].body), { ok: false, error }); + } + assert.equal(called, false, "validation happens before the facade is consulted"); + }); + + test("#4 answers 404 for the facade's not_found outcome and 200 otherwise", async (t) => { + await setupMocks(t, { acp: {} }); + let current = { + outcome: "not_found", + payload: { ok: false, error: "session not found" }, + matchKind: null, + from: "", + to: "T", + item: null, + }; + mockFacade(t, { applyEngineSessionRename: async () => current }); + const route = await loadRoute(); + for (const [outcome, expectedStatus, body] of [ + ["not_found", 404, { ok: false, error: "session not found" }], + [ + "ok", + 200, + { ok: true, session: { id: "webui-A", mcodeSessionId: null, title: "T", titleCustom: true } }, + ], + ]) { + current = + outcome === "not_found" + ? { outcome, payload: body, matchKind: null, from: "", to: "T", item: null } + : { outcome, payload: body, matchKind: "webuiId", from: "Old", to: "T", item: { id: "webui-A", title: "T" } }; + const res = mkRes(); + await route.handleRenameSession(jsonReq({ id: "webui-A", title: "T" }), res, { + cid: "tab-1", + }); + assert.equal(res.written[0].status, expectedStatus, outcome); + assert.deepEqual(JSON.parse(res.written[1].body), body); + } + }); + + test("#6 writes the facade's preview body byte-for-byte", async (t) => { + await setupMocks(t, { acp: {} }); + const ids = ["old-1", "old-2"]; + mockFacade(t, { + readOrphanSessionWriteIds: async () => ({ + ids, + payload: { ok: true, dryRun: true, count: 2, ids }, + gate: { gate: "checked", enforcement: "hard" }, + transport: RUNTIME, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCleanupOrphans({ url: "/api/sessions/cleanup-orphans?dryRun=true" }, res, { + cid: "tab-1", + }); + assert.equal(res.written[0].status, 200); + assert.equal( + res.written[0].headers["Content-Type"], + "application/json; charset=utf-8", + ); + // The batch's byte-for-byte red line, asserted at the HTTP edge + // and not only inside the facade. + assert.equal( + res.written[1].body, + '{"ok":true,"dryRun":true,"count":2,"ids":["old-1","old-2"]}', + ); + }); + + test("#6's no-op real path keeps its own four-key body", async (t) => { + await setupMocks(t, { acp: {} }); + mockFacade(t, { + readOrphanSessionWriteIds: async () => ({ + ids: [], + payload: { ok: true, dryRun: true, count: 0, ids: [] }, + gate: { gate: "checked" }, + transport: RUNTIME, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleCleanupOrphans({ url: "/api/sessions/cleanup-orphans" }, res, { + cid: "tab-1", + }); + assert.equal(res.written[0].status, 200); + assert.equal( + res.written[1].body, + '{"ok":true,"dryRun":false,"deleted":0,"ids":[]}', + ); + }); + + // ---- proof the facade mock actually took --------------------------- + + test("PROOF: a marker error escapes the untouched #7 route", async (t) => { + // Without a fresh `?bust=` re-import, `mock.module` would leave the + // route holding the PREVIOUS test's live binding, the marker would + // never be thrown, and this assertion would fail — which is the + // point: it is the only assertion in this section that cannot pass + // by accident. + await setupMocks(t, { acp: {} }); + const marker = new Error("B5-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + planEngineSessionDelete: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await withDecisions(() => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, mkRes(), { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + ); + } catch (err) { + caught = err; + } + assert.ok( + caught, + "the route swallowed the facade error — either the mock did not take, or the route grew a catch", + ); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + + test("PROOF: a marker error escapes the untouched #6 route", async (t) => { + // The same proof for the second facade consumer. A single proof + // would not cover a route that imported a different subset of the + // module. + await setupMocks(t, { acp: {} }); + const marker = new Error("B5-SWEEP-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + readOrphanSessionWriteIds: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleCleanupOrphans({ url: "/api/sessions/cleanup-orphans" }, mkRes(), { + cid: "tab-1", + }); + } catch (err) { + caught = err; + } + assert.ok(caught, "the sweep route swallowed the facade error"); + assert.equal(caught, marker); + }); + + test("CONTROL: with no facade mock, #7 reaches the real write path", async (t) => { + // The other half of the proof. A `?bust=` re-import under a fresh + // test hook gives a route bound to the REAL facade, so the request + // runs the actual plan → commit sequence against the mocked store + // and the real SQL deleter. If this answered from a mock, the two + // PROOF cases above would be proving nothing. + const { dbCalls } = await loadWritePath(t, { + records: [{ id: "webui-A", title: "A", chat: ["● hi"] }], + }); + const route = await loadRoute(); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + ); + assert.equal(res.written[0].status, 200); + const body = JSON.parse(res.written[1].body); + assert.equal(body.ok, true); + assert.equal(body.deleted, "webui-A"); + assert.equal(body.matchKind, "webuiId"); + // A webui-only record has no engine sid, so the deleter is not + // asked — and the response says so rather than inventing a result. + assert.equal(body.mcodeDbDel, null); + assert.equal(dbCalls.length, 0); + }); + }); + + // --------------------------------------------------------------------- + // 8. The audit chain, across the route/facade boundary + // --------------------------------------------------------------------- + + describe("RED LINE — the audit chain is intact across the split", () => { + // The one thing the plan→commit split could have broken: the + // write-ahead intent line has to land BETWEEN the plan and the + // mutation. These run the REAL route against the REAL facade, with + // only the audit sink and the SQL layer journalled, and assert the + // ORDER of the three events. + async function loadAuditedRoute(t, options = {}) { + await setupMocks(t, { acp: {}, sessions: { initial: options.records || [] } }); + const events = []; + mockAll( + t, + "lib/events.js", + { + append: (kind, data) => { + events.push({ kind, data }); + }, + }, + ["append", "read", "readAll", "verifyChain", "EVENTS_PATH"], + ); + mockAll( + t, + "lib/mcode-session-delete.js", + { + deleteMcodeSessionFromDb: (sid, o) => { + events.push({ kind: `sql(${sid},dryRun=${!!o.dryRun})` }); + return { ok: true, outcome: "deleted", log: ["local_runtime_sessions:1"], totalRowsDeleted: 1, tablesAbsent: 0 }; + }, + }, + ["deleteMcodeSessionFromDb", "MCODE_SESSION_DELETE_TABLES"], + ); + mockAll( + t, + "lib/session-tree.js", + { invalidateSessionTree: () => {}, getSessionTree: () => ({ ok: true, tree: [] }) }, + ["getSessionTree", "invalidateSessionTree"], + ); + mockAll( + t, + "lib/state-bus.js", + { + clients: new Map(), + pushStateFor: () => {}, + runChatViewChat: () => ({}), + makeClientState: () => ({ usage: {} }), + }, + ["clients", "pushStateFor", "runChatViewChat", "makeClientState", "setState", "getClient", "sseByCid", "pushAlert"], + ); + return { events, route: await loadRoute() }; + } + + test("a real delete writes intent BEFORE the rows go, and the outcome after", async (t) => { + const { events, route } = await loadAuditedRoute(t, { + records: [ + { id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: ["● hi", "● there"] }, + ], + }); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + ); + assert.equal(res.written[0].status, 200); + assert.deepEqual( + events.map((e) => e.kind), + ["session.delete.intent", "sql(mvs_sid_A,dryRun=false)", "session.delete"], + "the intent line lands before the mutation, the outcome line after it", + ); + // The intent payload carries exactly the three facts the authorize + // modal showed the user, which is the point of computing them in + // the plan and passing them through unchanged. + assert.equal(events[0].data.payload.matchKind, "webuiId"); + assert.equal(events[0].data.payload.isOrphan, false); + assert.equal(events[0].data.payload.chatLen, 2); + assert.ok(events[0].data.payload.decidedBy, "and the authorizer's decision"); + assert.equal(events[2].data.payload.dryRun, false); + assert.equal(events[2].data.payload.title, "A", "the title is logged — it was user-visible in the sidebar"); + assert.ok("touchedCids" in events[2].data.payload, "the fan-out effect is recorded"); + }); + + test("a declined authorize writes NO intent line and never touches the engine", async (t) => { + const { events, route } = await loadAuditedRoute(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + const res = mkRes(); + await withDecisions( + () => + route.handleDeleteSession({ url: "/api/sessions/webui-A" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + { approve: false }, + ); + assert.equal(res.written[0].status, 403); + assert.deepEqual(events, [], "a refused delete is not an audited one — nothing was attempted"); + }); + + test("a dryRun preview is audited as a PREVIEW and never mutates", async (t) => { + const { events, route } = await loadAuditedRoute(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "A", chat: [] }], + }); + const res = mkRes(); + await withDecisions(() => + route.handleDeleteSession({ url: "/api/sessions/webui-A?dryRun=true" }, res, { + cs: {}, + cid: "tab-1", + pathname: "/api/sessions/webui-A", + }), + ); + assert.equal(res.written[0].status, 200); + assert.deepEqual( + events.map((e) => e.kind), + ["sql(mvs_sid_A,dryRun=true)", "session.delete"], + ); + assert.equal( + events[1].data.payload.dryRun, + true, + "the dryRun marker is what lets an operator tell a preview from a real delete", + ); + const { getSessionsStore } = await import("../../helpers/_setup.js"); + assert.equal(getSessionsStore().length, 1, "and the record is still there"); + }); + + test("a rename is audited with from → to and the resolved match kind", async (t) => { + const { events, route } = await loadAuditedRoute(t, { + records: [{ id: "webui-A", mcodeSessionId: "mvs_sid_A", title: "Old", chat: [] }], + }); + const res = mkRes(); + await route.handleRenameSession(jsonReq({ id: "mvs_sid_A", title: "New" }), res, { + cid: "tab-1", + }); + assert.equal(res.written[0].status, 200); + assert.deepEqual(events.map((e) => e.kind), ["session.rename"]); + assert.equal(events[0].data.payload.matchKind, "mcodeSessionId", "renamed through the engine sid"); + assert.equal(events[0].data.payload.from, "Old"); + assert.equal(events[0].data.payload.to, "New"); + assert.equal(events[0].data.payload.mcodeSessionId, "mvs_sid_A"); + }); + }); +}); + diff --git a/release/public-source.json b/release/public-source.json index 80331e1c..24089e46 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3458,6 +3458,7 @@ "packages/webui/server/engine/session-export.js", "packages/webui/server/engine/session-reads.js", "packages/webui/server/engine/session-tree-reads.js", + "packages/webui/server/engine/session-writes.js", "packages/webui/server/engine/usage-reads.js", "packages/webui/server/lib/acp-client.js", "packages/webui/server/lib/agent-team-detect.js", @@ -3603,6 +3604,7 @@ "packages/webui/test/lib/engine/session-export.test.js", "packages/webui/test/lib/engine/session-reads.test.js", "packages/webui/test/lib/engine/session-tree-reads.test.js", + "packages/webui/test/lib/engine/session-writes.test.js", "packages/webui/test/lib/engine/usage-reads.test.js", "packages/webui/test/lib/events-concurrency.test.js", "packages/webui/test/lib/events-hash.test.js", diff --git a/scripts/test-tmp-leak.check.mjs b/scripts/test-tmp-leak.check.mjs index f56d9994..818fee2a 100644 --- a/scripts/test-tmp-leak.check.mjs +++ b/scripts/test-tmp-leak.check.mjs @@ -313,6 +313,7 @@ const KNOWN_PREFIXES = [ "webui-sec-net-settings-", "webui-sessdb-", "webui-session-delete-test-", + "webui-session-writes-b5-", "webui-sessions-search-check-", "webui-sessions-test-events-", "webui-settings-test-events-", From 1506cc29fe42e6a0287307102537c581dce8615b Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Sat, 3 Oct 2026 01:18:26 +0800 Subject: [PATCH 17/21] fix(webui): drop whitespace text nodes in markdown tables and dedupe thinking label --- .../webapp/components/activity-group.tsx | 31 +++- .../webui/webapp/components/markdown-html.tsx | 36 +++- packages/webui/webapp/components/modals.tsx | 16 +- .../webui/webapp/test/activity-group.test.ts | 99 ++++++++++- .../webui/webapp/test/helpers/dom-shim.ts | 165 ++++++++++++++++++ .../webapp/test/markdown-html-render.test.ts | 162 ++++++++++++++++- .../test/modals-decision-channels.test.ts | 145 +++++++++++++++ release/public-source.json | 1 + 8 files changed, 639 insertions(+), 16 deletions(-) create mode 100644 packages/webui/webapp/test/helpers/dom-shim.ts diff --git a/packages/webui/webapp/components/activity-group.tsx b/packages/webui/webapp/components/activity-group.tsx index 4d826c2c..043bfd7f 100644 --- a/packages/webui/webapp/components/activity-group.tsx +++ b/packages/webui/webapp/components/activity-group.tsx @@ -187,18 +187,33 @@ export function ActivityGroup({ // While a call is in flight upstream names it instead of listing categories // ("已使用 3 次工具|bash"); once the turn settles it lists the per-category // contributions joined with ", " (「查看 2 个文件, 执行 1 条命令」). + // + // The `thinking` category is counted but NOT printed. UAT fix: the group + // header and the turn bar are both inside the same turn, and both used the + // same 「思考 N 次」 wording for the same count, so a thinking-only run read + // 「思考 1 次」 twice (header above the thought, turn bar under the answer). + // The turn bar keeps it — that is where the reference `WebuiTurnProcess` + // puts it (`processSummaryParts` in the desktop `AssistantBody`), and it is + // the one row that survives the group's collapse. A run with nothing but + // thoughts therefore falls back to the qualitative 「思考过程」 label: still + // one label for the fold, and never a second copy of the count. + const printable = summary.contributions.filter( + (entry) => entry.category !== "thinking", + ); const label = summary.activeTool ? t("activity.activeTool") .replace("{{count}}", String(summary.tools)) .replace("{{tool}}", summary.activeTool) - : summary.contributions - .map((entry) => - t((SUMMARY_CATEGORY_KEY[entry.category] ?? "activity.usedTools") as MessageKey).replace( - "{{count}}", - String(entry.count), - ), - ) - .join(", "); + : printable.length > 0 + ? printable + .map((entry) => + t((SUMMARY_CATEGORY_KEY[entry.category] ?? "activity.usedTools") as MessageKey).replace( + "{{count}}", + String(entry.count), + ), + ) + .join(", ") + : t("activity.thoughtProcess"); // The forced-open state (active tool, or streaming thought). While it // holds, a user click on the summary must not collapse the group. diff --git a/packages/webui/webapp/components/markdown-html.tsx b/packages/webui/webapp/components/markdown-html.tsx index 011db3d6..ebb37c9c 100644 --- a/packages/webui/webapp/components/markdown-html.tsx +++ b/packages/webui/webapp/components/markdown-html.tsx @@ -93,6 +93,15 @@ export function MarkdownHtml({ html }: { html: string }) { ); } +/** + * Elements whose children React accepts only as elements — never as text. + * + * Mirrors React DOM's own `validateTextNesting` table for the tags + * `lib/markdown.ts` allowlists. `` and `` are absent from + * that allowlist, so they are absent here too. + */ +const TABLE_STRUCTURE_TAGS = new Set(["table", "thead", "tbody", "tfoot", "tr"]); + /** * Walk a sanitised HTML string and convert it to a React tree. * @@ -109,12 +118,19 @@ export function MarkdownHtml({ html }: { html: string }) { * attributes (`class`, `href`, `title`, `align`) so a markdown * document looks the same as before — only the mermaid fences * are upgraded from inert HTML to a live component. + * - drops whitespace-only text under a table-family element (see + * `TABLE_STRUCTURE_TAGS`): React refuses those children, and table + * layout never paints them. * * Returns a single `dangerouslySetInnerHTML` element from inside the * tree on SSR (when `DOMParser` is undefined); the prerender still * produces a non-empty HTML response. + * + * Exported for the render-harness test, which drives this walker over a + * real `marked` table and asserts the React tree it produces — the only + * place the hydration contract is actually checkable without a browser. */ -function htmlToReact(html: string, theme: "light" | "dark"): ReactNode { +export function htmlToReact(html: string, theme: "light" | "dark"): ReactNode { if (typeof DOMParser === "undefined") { return (
{ + const walkChildren = (parent: Element | Document, parentTag: string | null): ReactNode[] => { const out: ReactNode[] = []; for (const child of [...parent.childNodes]) { if (child.nodeType === 3 /* text */) { const text = child.textContent ?? ""; if (text.length === 0) continue; + // UAT fix — `validateTextNesting` (React DOM, dev builds) rejects + // ANY text node under a table-family element, whitespace included, + // and answers with "In HTML, whitespace text nodes cannot be a + // child of . This will cause a hydration error." `marked` + // indents every table line, so the sanitised HTML the walker is + // handed carries those newlines as real text nodes under + //
///. Table layout collapses inter-tag + // whitespace and never paints it, so dropping it here removes the + // console flood and the hydration error without moving a pixel. + if (parentTag !== null && TABLE_STRUCTURE_TAGS.has(parentTag) && /^\s*$/.test(text)) { + continue; + } out.push(text); continue; } @@ -178,7 +206,7 @@ function htmlToReact(html: string, theme: "light" | "dark"): ReactNode { props[attr.name] = attr.value; } } - const children = walkChildren(el); + const children = walkChildren(el, tag); out.push( createElement(tag, props, children.length > 0 ? children : undefined), ); @@ -186,7 +214,7 @@ function htmlToReact(html: string, theme: "light" | "dark"): ReactNode { return out; }; - return walkChildren(doc.body); + return walkChildren(doc.body, null); } /** diff --git a/packages/webui/webapp/components/modals.tsx b/packages/webui/webapp/components/modals.tsx index 016f91cd..a88ea379 100644 --- a/packages/webui/webapp/components/modals.tsx +++ b/packages/webui/webapp/components/modals.tsx @@ -392,6 +392,20 @@ export function Modal({ ); } +/** + * The primary (filled) button of a confirm modal. + * + * UAT fix — the label colour. The label used + * `text-text_default_inverted_static`, which the design system defines as + * "text on an inverted surface" and never re-themes: it stays near-white + * in BOTH `:root` (95%) and `.dark` (80%) — see `styles/tokens.css`. The + * dark theme inverts `--bg_interaction_primary_default` to `--gray_0` + * (white), so the pair composited to white on white and the 「批准」 label + * disappeared — an unlabelled button, not a missing string. The token that + * actually pairs with this background is `--text_label_primary_default` + * (white on light, near-black on dark), the same pairing the upstream + * `.mavis-button.black` rule uses (`styles/official-utilities.css`). + */ function PrimaryButton({ disabled, onClick, @@ -406,7 +420,7 @@ function PrimaryButton({ type="button" disabled={disabled} onClick={onClick} - className="h-8 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_default_inverted_static transition-colors hover:bg-bg_interaction_primary_hover disabled:opacity-50" + className="h-8 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_label_primary_default transition-colors hover:bg-bg_interaction_primary_hover disabled:opacity-50" > {children} diff --git a/packages/webui/webapp/test/activity-group.test.ts b/packages/webui/webapp/test/activity-group.test.ts index ce3c592e..8a50c9ad 100644 --- a/packages/webui/webapp/test/activity-group.test.ts +++ b/packages/webui/webapp/test/activity-group.test.ts @@ -351,7 +351,22 @@ describe("ThinkingRow — the thinking block (D2)", () => { assert.match(html, /data-testid="thinking-summary-icon"/); assert.match(html, /data-tool-icon-type="thinking"/); assert.match(html, /已完成推理/); - assert.doesNotMatch(html, /思考过程/); + // Scoped to the thinking SUMMARY ROW, which is what this test names: + // the reference's `WebuiThinkingBlock` defaults `showDetailHeading` to + // false, so the body must not open with a 「思考过程」 heading. The + // assertion used to run over the whole group markup, which made it a + // de-facto ban on the string anywhere — including the group header, + // which now labels a thoughts-only run 「思考过程」 (see the UAT block + // below). Reading the whole document here tested more than it claimed. + const summaryStart = html.indexOf('data-testid="thinking-summary"'); + const rowStart = html.lastIndexOf("", summaryStart); + assert.ok(rowStart >= 0 && rowEnd > summaryStart, "the thinking summary row is missing"); + assert.doesNotMatch( + html.slice(rowStart, rowEnd), + /思考过程/, + "the thinking summary row must carry the status copy, not a body heading", + ); }); test("the body renders through the Markdown pipeline, not plain text", () => { @@ -761,6 +776,88 @@ describe("TurnProcessDisclosure — the turn bar (D6, PR3)", () => { }); }); +describe("UAT fix — 「思考 N 次」 is labelled once per turn, not twice", () => { + /** The header row alone: the assertion must not be satisfied (or + * broken) by anything in the folded body. */ + const headerOf = (html: string): string => { + const start = html.indexOf('data-testid="activity-group-header"'); + const open = html.lastIndexOf("", start); + assert.ok(open >= 0 && end > start, "the group header row is missing"); + return html.slice(open, end); + }; + + test("a thoughts-only run labels the group qualitatively, not with the count", () => { + const header = headerOf( + renderGroup([thinkingBlock("let me check")], thoughtsOnlySummary), + ); + // Regression: the header used to read 「思考 1 次」 — byte-identical to + // the turn bar under the same turn's answer. + assert.doesNotMatch( + header, + /思考 1 次/, + "the group header must not repeat the turn bar's thinking count", + ); + assert.match(header, /思考过程/); + }); + + test("a mixed run keeps its tool contributions and drops only the count", () => { + const header = headerOf( + renderGroup( + [thinkingBlock("let me check"), richToolBlock()], + mixedSummary, + ), + ); + assert.doesNotMatch(header, /思考 1 次/); + assert.match(header, /执行 1 条命令/); + }); + + test("the turn bar still carries the count (the one surviving label)", () => { + // The count must not simply be deleted: the turn bar is the row that + // survives the group's collapse, and it is where the reference + // `WebuiTurnProcess` puts it. + const html = renderToStaticMarkup( + createElement(TurnProcessDisclosure, { + stats: { thinking: 1, tools: 0, answerChars: 0 }, + processedDurationMs: 5000, + t, + }), + ); + assert.match(html, /思考 1 次,共执行 5 秒/); + }); + + test("a whole turn states the count exactly once", () => { + // The rendered end-to-end shape the UAT screenshot captured: a + // thoughts-only group, then the turn bar for the same turn. The bar + // carries the summary twice in its markup (the `data-summary-text` + // mirror plus the visible text), so the attribute is stripped first — + // this counts what the reader sees, not what the DOM stores. + const group = renderGroup([thinkingBlock("let me check")], thoughtsOnlySummary); + const bar = renderToStaticMarkup( + createElement(TurnProcessDisclosure, { + stats: { thinking: 1, tools: 0, answerChars: 0 }, + processedDurationMs: 5000, + t, + }), + ); + const visible = (group + bar).replace(/ data-summary-text="[^"]*"/g, ""); + const occurrences = visible.match(/思考 1 次/g) ?? []; + assert.equal( + occurrences.length, + 1, + `「思考 1 次」 must appear once per turn, found ${occurrences.length}`, + ); + }); + + test("the leading icon of a thoughts-only run is unchanged", () => { + // The fix filters the LABEL, not the summary: `iconType` still comes + // from the `thinking` contribution, so the ⓘ glyph does not regress to + // the generic tool icon. + const html = renderGroup([thinkingBlock("let me check")], thoughtsOnlySummary); + assert.match(html, /data-tool-icon-type="thinking"/); + }); +}); + describe("chat.tsx wiring — the streaming derivation stays put", () => { test("the tail-run derivation feeds streaming and startedAtMs into the group", () => { assert.match(chatSource, /const streamingActivityIndex = useMemo/); diff --git a/packages/webui/webapp/test/helpers/dom-shim.ts b/packages/webui/webapp/test/helpers/dom-shim.ts new file mode 100644 index 00000000..bae13eb4 --- /dev/null +++ b/packages/webui/webapp/test/helpers/dom-shim.ts @@ -0,0 +1,165 @@ +// webapp/test/helpers/dom-shim.ts +// +// A `DOMParser` stand-in for the Node test runner, built on the `parse5` +// already in the dependency tree. +// +// Why it exists +// ------------- +// +// `components/markdown-html.tsx` converts the sanitised markdown into a +// React tree by walking a `DOMParser` document. Node has no `DOMParser`, +// and the project deliberately does not pull in jsdom/happy-dom (a +// multi-megabyte dependency for one walker). That left the walker +// untested: `MarkdownHtml` silently took its SSR `dangerouslySetInnerHTML` +// branch in every unit test, so a defect in the walker's output — the +// whitespace text nodes React refuses under `
`, for one — reached +// production with a green gate. +// +// The shim exposes exactly the DOM Level 1 surface the walker touches: +// `parseFromString`, `body`, `childNodes`, `nodeType`, `tagName`, +// `attributes`, `classList.contains`, `textContent`, `previousSibling`. +// It is deliberately not a general DOM: a walker that grows a new DOM +// dependency fails here loudly (undefined method) instead of silently +// testing against a fake that agrees with it. +// +// On parse5: the workspace has no HTML parser of its own and does not +// depend on one. `parse5` arrives through the Next.js tree and is pinned +// in `pnpm-lock.yaml`; it is reached with `createRequire` rather than an +// `import` because `@mavis/webui` does not declare it, and an undeclared +// `import` would break `webapp:typecheck` with TS7016. The coupling is +// test-only and stated here rather than hidden: if the transitive copy ever +// disappears, the shim throws the message below and the affected tests +// fail loudly instead of quietly passing against a stub. + +import { createRequire } from "node:module"; + +interface Parse5 { + parse(html: string): unknown; +} + +const requireFromHere = createRequire(import.meta.url); + +function loadParse5(): Parse5 { + try { + return requireFromHere("parse5") as Parse5; + } catch (cause) { + throw new Error( + "the markdown DOM shim needs `parse5`, which no longer resolves from " + + "packages/webui. Declare it as a devDependency of @mavis/webui, or " + + "replace this shim.", + { cause }, + ); + } +} + +const parse5 = loadParse5(); + +/** parse5 node, narrowed to the fields the shim reads. */ +interface Parse5Node { + nodeName: string; + value?: string; + tagName?: string; + attrs?: { name: string; value: string }[]; + childNodes?: Parse5Node[]; +} + +/** A node in the shimmed tree: DOM Level 1 fields over a parse5 node. */ +export interface ShimNode { + nodeType: number; + nodeName: string; + tagName: string; + textContent: string; + childNodes: ShimNode[]; + parentNode: ShimNode | null; + previousSibling: ShimNode | null; + nextSibling: ShimNode | null; + attributes: { name: string; value: string }[]; + classList: { contains(token: string): boolean }; +} + +/** Minimal `document` the walker consumes (`htmlToReact` reads `.body`). */ +export interface ShimDocument { + body: ShimNode; +} + +const ELEMENT_NODE = 1; +const TEXT_NODE = 3; + +function toShimNode(node: Parse5Node, parent: ShimNode | null): ShimNode { + const isText = node.nodeName === "#text"; + const isElement = node.tagName !== undefined; + + const attributes = (node.attrs ?? []).map((attr) => ({ + name: attr.name, + value: attr.value, + })); + + const shim: ShimNode = { + nodeType: isText ? TEXT_NODE : isElement ? ELEMENT_NODE : 0, + nodeName: node.nodeName, + tagName: node.tagName ?? "", + textContent: isText + ? (node.value ?? "") + : (node.childNodes ?? []).map((child) => child.value ?? "").join(""), + childNodes: [], + parentNode: parent, + previousSibling: null, + nextSibling: null, + attributes, + classList: { + contains(token: string): boolean { + const classAttr = attributes.find((attr) => attr.name === "class"); + return (classAttr?.value ?? "").split(/\s+/).includes(token); + }, + }, + }; + + shim.childNodes = (node.childNodes ?? []).map((child) => { + const childShim = toShimNode(child, shim); + const previous = shim.childNodes[shim.childNodes.length - 1]; + if (previous) previous.nextSibling = childShim; + return childShim; + }); + + return shim; +} + +/** A `DOMParser` whose `parseFromString` returns the shimmed tree. */ +export class ShimDomParser { + parseFromString(html: string, _type: "text/html"): ShimDocument { + // parse5 builds the implied // around the fragment, + // so is a grandchild of the document, not a child. + const bodyNode = findBody(parse5.parse(html) as Parse5Node); + if (!bodyNode) throw new Error("parse5 produced no for the fixture"); + return { body: toShimNode(bodyNode, null) }; + } +} + +function findBody(node: Parse5Node): Parse5Node | undefined { + if (node.nodeName === "body") return node; + for (const child of node.childNodes ?? []) { + const found = findBody(child); + if (found) return found; + } + return undefined; +} + +/** + * Run `fn` with `DOMParser` shimmed in, then restore whatever was there. + * + * The walker resolves the bare identifier `DOMParser`, so the global must + * be installed before `htmlToReact` is called. It is restored on the + * throw path too: a failing assertion must not leave a fake DOM + * installed for the rest of the file. + */ +export function withDomParserShim(fn: () => T): T { + const globals = globalThis as { DOMParser?: unknown }; + const previous = globals.DOMParser; + globals.DOMParser = ShimDomParser; + try { + return fn(); + } finally { + if (previous === undefined) delete globals.DOMParser; + else globals.DOMParser = previous; + } +} diff --git a/packages/webui/webapp/test/markdown-html-render.test.ts b/packages/webui/webapp/test/markdown-html-render.test.ts index 9ba9d7db..1604d22c 100644 --- a/packages/webui/webapp/test/markdown-html-render.test.ts +++ b/packages/webui/webapp/test/markdown-html-render.test.ts @@ -62,7 +62,7 @@ import { test, describe } from "node:test"; import assert from "node:assert/strict"; -import { renderMarkdown } from "../lib/markdown"; +import { parseMarkdown, renderMarkdown } from "../lib/markdown"; import "../lib/mermaid-renderer"; // auto-registers the mermaid language renderer import { _stripMermaidInitForTest, @@ -70,7 +70,8 @@ import { _mermaidConfigKeyForTest, _mermaidFontFamilyForTest, } from "../components/mermaid-block"; -import { findMermaidSourceBefore } from "../components/markdown-html"; +import { findMermaidSourceBefore, htmlToReact } from "../components/markdown-html"; +import { withDomParserShim } from "./helpers/dom-shim"; describe("MarkdownHtml render path — registry-side evidence", () => { test("a mermaid fence produces the placeholder pair the walker expects", () => { @@ -215,6 +216,163 @@ describe("MermaidBlock — sanitiser hooks (test-only exports)", () => { // full mermaid render path is not exercised in the unit harness. // --------------------------------------------------------------------------- +// --------------------------------------------------------------------------- +// UAT fix — the walker's output must satisfy React's `validateTextNesting`. +// +// The defect: `marked` indents every line of a GFM table, so the sanitised +// HTML carries `\n` as real text nodes under
///. +// React DOM (dev build) rejects ANY text child of those elements and logs +// "In HTML, whitespace text nodes cannot be a child of
. This will +// cause a hydration error." once per tag per page load. The walker now +// drops whitespace-only text under the table family; table layout never +// painted it, so no visual output changes. +// +// This is the first test in the file that drives the REAL walker: the +// earlier ones had to assert inputs and exported helpers because Node has +// no DOMParser. `test/helpers/dom-shim.ts` supplies one over parse5, so the +// contract is now checked where it actually lives — on the React tree the +// component mounts. +// --------------------------------------------------------------------------- + +/** The tags React's `validateTextNesting` refuses text children under. */ +const TABLE_STRUCTURE_TAGS = ["table", "thead", "tbody", "tfoot", "tr"] as const; + +type ReactLikeNode = + | string + | ReactLikeNode[] + | { type?: unknown; props?: { children?: ReactLikeNode } }; + +/** Every whitespace-only string anywhere in the tree, with its parent tag. */ +function whitespaceTextUnderTableTags( + node: ReactLikeNode, + parentTag: string | null = null, + found: { parentTag: string; text: string }[] = [], +): { parentTag: string; text: string }[] { + if (typeof node === "string") { + if (parentTag !== null && /^\s+$/.test(node)) found.push({ parentTag, text: node }); + return found; + } + // `htmlToReact` returns the body's children as one array; elements nest + // their own children as an array or a single node. + if (Array.isArray(node)) { + for (const child of node) whitespaceTextUnderTableTags(child, parentTag, found); + return found; + } + const tag = typeof node.type === "string" ? node.type : parentTag; + if (node.props?.children !== undefined) { + whitespaceTextUnderTableTags(node.props.children, tag, found); + } + return found; +} + +/** Every element tag in the tree, in document order. */ +function collectTags(node: ReactLikeNode, tags: string[] = []): string[] { + if (typeof node === "string") return tags; + if (Array.isArray(node)) { + for (const child of node) collectTags(child, tags); + return tags; + } + if (typeof node.type === "string") tags.push(node.type); + if (node.props?.children !== undefined) collectTags(node.props.children, tags); + return tags; +} + +describe("markdown-html — the table React tree has no text children", () => { + const GFM_TABLE = [ + "| 名称 | 说明 |", + "| --- | :---: |", + "| 端口 | 监听端口 |", + "| 路径 | 根路径 |", + ].join("\n"); + + test("a GFM table yields no whitespace text node under any table-family tag", () => { + // Sanitising needs a DOM too, so drive the parser directly: the walker + // is the unit under test, and the raw marked output is what it is fed + // (lib/markdown.ts#renderMarkdown hands it `sanitize(parseMarkdown(...))`, + // and the sanitiser never touches text nodes). + const tree = withDomParserShim(() => htmlToReact(parseMarkdown(GFM_TABLE), "light")); + + const offenders = whitespaceTextUnderTableTags(tree as ReactLikeNode); + assert.deepEqual( + offenders, + [], + `whitespace text nodes reached a table-family element: ${JSON.stringify(offenders)}`, + ); + }); + + test("the mutation guard — marked really does emit that whitespace", () => { + // Without this, the test above would also pass if `marked` stopped + // indenting its tables, i.e. for the wrong reason. + const html = parseMarkdown(GFM_TABLE); + assert.match(html, /
[\s\S]*\n[\s\S]*<\/table>/); + assert.match(html, /[\s\S]*\n[\s\S]*<\/tr>/); + }); + + test("the table keeps every cell — only whitespace was dropped", () => { + const tree = withDomParserShim(() => htmlToReact(parseMarkdown(GFM_TABLE), "light")); + const tags = collectTags(tree as ReactLikeNode); + assert.deepEqual( + tags, + [ + "table", + "thead", + "tr", + "th", "th", + "tbody", + "tr", "td", "td", + "tr", "td", "td", + ], + "the element structure of a GFM table must survive the fix untouched", + ); + }); + + test("cell text is preserved verbatim", () => { + const tree = withDomParserShim(() => htmlToReact(parseMarkdown(GFM_TABLE), "light")); + const rendered = JSON.stringify(tree, (key, value) => + typeof value === "function" ? "[fn]" : value, + ); + for (const cell of ["名称", "说明", "端口", "监听端口", "路径", "根路径"]) { + assert.ok(rendered.includes(cell), `cell ${cell} disappeared from the React tree`); + } + }); + + test("whitespace between BLOCK tags is still rendered (prose is not a table)", () => { + // The drop is scoped to the table family. A paragraph's inter-block + // newlines are renderable whitespace and must survive — dropping them + // everywhere would reflow prose the markdown never asked to reflow. + const tree = withDomParserShim(() => + htmlToReact(parseMarkdown("# Title\n\nbody text\n"), "light"), + ); + const texts = whitespaceTextUnderTableTags(tree as ReactLikeNode); + assert.deepEqual( + texts, + [], + "a

is not a table tag, so this guard is about the block level", + ); + // The `\n` between `` and `

` sits at body level: the walker + // keeps it, and so must the tree. + const topLevel = (tree as ReactLikeNode[]).filter( + (node): node is string => typeof node === "string", + ); + assert.ok( + topLevel.some((text) => /^\s+$/.test(text)), + "inter-block whitespace outside tables must still reach the tree", + ); + }); + + test("text with content under a table tag is still rendered (not over-trimmed)", () => { + // A `

` may legitimately hold leading/trailing spaces around its + // content (" a "). Only WHITESPACE-ONLY nodes may be dropped. + const tree = withDomParserShim(() => + htmlToReact(parseMarkdown("| a |\n| --- |\n| padded |"), "light"), + ); + const rendered = JSON.stringify(tree, (key, value) => + typeof value === "function" ? "[fn]" : value, + ); + assert.ok(rendered.includes("padded"), "cell content was lost"); + }); +}); + describe("markdown-html — findMermaidSourceBefore (blocker 1: copy-source byte-exact)", () => { /** * Hand-built DOM Level 1 element mock. The walker only touches diff --git a/packages/webui/webapp/test/modals-decision-channels.test.ts b/packages/webui/webapp/test/modals-decision-channels.test.ts index 3239d751..3aeda1aa 100644 --- a/packages/webui/webapp/test/modals-decision-channels.test.ts +++ b/packages/webui/webapp/test/modals-decision-channels.test.ts @@ -134,6 +134,151 @@ describe("ticket 70 — the plan prompt offers no decision it cannot deliver", ( }); }); +describe("UAT fix — the authorize button's label is legible in BOTH themes", () => { + // The reported symptom was a blank 「批准」 button. The string was always + // there (pinned above); the LABEL COLOUR was the defect, so a + // string assertion could never have caught it. These tests resolve the + // real design tokens out of styles/tokens.css and assert the pair the + // button renders is legible in each theme. + + const tokensCss = readFileSync(resolve(here, "../styles/tokens.css"), "utf8"); + + /** + * The declarations of every top-level `:root { }` / `.dark { }` block, + * merged. tokens.css is split into many sibling blocks (primitives, then + * one per semantic group) rather than a single one, so a reader that + * stops at the first block would only ever see the colour ramp. + */ + const blockVars = (selector: ":root" | ".dark"): Map => { + const vars = new Map(); + const open = new RegExp(`^${selector} \\{`, "gm"); + let match: RegExpExecArray | null; + while ((match = open.exec(tokensCss)) !== null) { + const body = tokensCss.slice(match.index, tokensCss.indexOf("\n}", match.index)); + for (const line of body.split("\n")) { + const declaration = /^\s*(--[\w-]+):\s*(.+?);\s*$/.exec(line); + if (declaration) vars.set(declaration[1]!, declaration[2]!); + } + } + assert.ok(vars.size > 0, `no ${selector} block found in tokens.css`); + return vars; + }; + + /** Follow `var(--x)` indirections until a literal value is reached. */ + const resolveToken = (vars: Map, name: string, depth = 0): string => { + if (depth > 8) throw new Error(`token cycle at ${name}`); + const value = vars.get(name); + if (value === undefined) throw new Error(`token ${name} is not defined`); + const inner = /^var\((--[\w-]+)\)$/.exec(value); + return inner ? resolveToken(vars, inner[1]!, depth + 1) : value.trim(); + }; + + // `.dark` only carries the semantic overrides; the primitives stay in + // `:root`, so the dark resolution layers the two. + const light = blockVars(":root"); + const dark = new Map([...light, ...blockVars(".dark")]); + + /** + * Composite a text colour over an opaque fill — the colour a pixel of the + * label actually takes. + * + * Needed because the old label is not pure white: the dark theme sets it + * to 80%-white (`#fffc`, the four-digit `#rgba` CSS form). Over an opaque + * white fill that composites to exactly the fill, which is why comparing + * the raw hex strings would have missed the defect while the button was + * plainly unreadable. + */ + const compositeOver = (text: string, fill: string): string => { + /** `#rgb` / `#rgba` / `#rrggbb` / `#rrggbbaa` → [r, g, b, a] with a in 0..1. */ + const channels = (value: string): [number, number, number, number] => { + const digits = value.slice(1).toLowerCase(); + assert.match(digits, /^([0-9a-f]{3,8})$/, `unsupported colour literal: ${value}`); + const wide = digits.length <= 4 + ? [...digits].map((digit) => digit + digit).join("") + : digits; + const byte = (index: number) => Number.parseInt(wide.slice(index, index + 2), 16); + return [byte(0), byte(2), byte(4), wide.length === 8 ? byte(6) / 255 : 1]; + }; + const [tr, tg, tb, alpha] = channels(text); + const [fr, fg, fb] = channels(fill); + const over = (t: number, f: number) => + Math.round(t * alpha + f * (1 - alpha)) + .toString(16) + .padStart(2, "0"); + return `#${over(tr, fr)}${over(tg, fg)}${over(tb, fb)}`; + }; + + test("the defect is reproducible on the token pair the button used to render", () => { + // Regression context, stated as an executable claim: the old pairing + // composited to the fill in the dark theme. If a future token + // regeneration ever themes `--text_default_inverted_static`, this stops + // holding and the note in modals.tsx must be revisited. + const background = resolveToken(dark, "--bg_interaction_primary_default"); + const oldLabel = resolveToken(dark, "--text_default_inverted_static"); + assert.equal( + compositeOver(oldLabel, background), + compositeOver(background, background), + "the dark theme is expected to invert the primary fill to white and leave " + + "the label 80%-white — that pair is what made 「批准」 unreadable", + ); + // Light theme was never affected, and saying so keeps the fix honest + // about what it changes. + const lightFill = resolveToken(light, "--bg_interaction_primary_default"); + const lightLabel = resolveToken(light, "--text_default_inverted_static"); + assert.notEqual( + compositeOver(lightLabel, lightFill), + compositeOver(lightFill, lightFill), + ); + }); + + test("the label token the button now uses contrasts with the fill in BOTH themes", () => { + for (const theme of [ + { name: ":root", vars: light }, + { name: ".dark", vars: dark }, + ]) { + const background = resolveToken(theme.vars, "--bg_interaction_primary_default"); + const label = resolveToken(theme.vars, "--text_label_primary_default"); + assert.notEqual( + compositeOver(label, background), + compositeOver(background, background), + `${theme.name}: the primary button would render its label invisibly ` + + `(${label} on ${background})`, + ); + } + }); + + test("PrimaryButton pairs the primary fill with the matching label token", () => { + // The same pairing the upstream `.mavis-button.black` rule uses + // (styles/official-utilities.css), so this button now matches the + // reference skin in both themes. + const primary = / { + const utilities = readFileSync( + resolve(here, "../styles/official-utilities.css"), + "utf8", + ); + assert.match( + utilities, + /\.mavis-button\.black \{[^}]*background-color:var\(--bg_interaction_primary_default\);color:var\(--text_label_primary_default\)/, + ); + }); +}); + describe("ticket 70 — dictionary parity for the changed keys", () => { const LOCALES = ["en", "zh"] as const; diff --git a/release/public-source.json b/release/public-source.json index 24089e46..3fb283df 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3941,6 +3941,7 @@ "packages/webui/webapp/test/fs-tree-reveal.test.ts", "packages/webui/webapp/test/git-panel.test.ts", "packages/webui/webapp/test/greeting.test.ts", + "packages/webui/webapp/test/helpers/dom-shim.ts", "packages/webui/webapp/test/i18n-appearance.test.ts", "packages/webui/webapp/test/i18n-browser.test.ts", "packages/webui/webapp/test/i18n-file-open.test.ts", From 8cca2358b5449c38f2a322f819a9f3ab45b0f466 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Sat, 3 Oct 2026 01:18:44 +0800 Subject: [PATCH 18/21] fix(webui): sweep the non-flipping inverted text token off primary surfaces The authorize-button fix (08599472) surfaced five more primary surfaces pairing text-text_default_inverted_static with bg_interaction_primary_default; dark mode inverts that background to pure white while the token stays near-white, so the label composites to white-on-white. Swap all of them to text-text_label_primary_default and add a source-scan guardrail that keeps the pairing out of primary surfaces while pinning the sanctioned status-badge exception (toolbar). --- packages/webui/webapp/app/error.tsx | 2 +- .../webapp/components/add-model-dialog.tsx | 4 +- packages/webui/webapp/components/modals.tsx | 4 +- packages/webui/webapp/components/panels.tsx | 2 +- .../webapp/components/provider-management.tsx | 2 +- .../webapp/test/theme-token-pairing.test.ts | 78 +++++++++++++++++++ release/public-source.json | 1 + 7 files changed, 86 insertions(+), 7 deletions(-) create mode 100644 packages/webui/webapp/test/theme-token-pairing.test.ts diff --git a/packages/webui/webapp/app/error.tsx b/packages/webui/webapp/app/error.tsx index d0d1308d..c6985df5 100644 --- a/packages/webui/webapp/app/error.tsx +++ b/packages/webui/webapp/app/error.tsx @@ -176,7 +176,7 @@ export default function RouteError({ error, reset }: ErrorBoundaryProps) { type="button" onClick={onReset} data-testid="route-error-reload" - className="h-8 rounded-[8px] bg-bg_interaction_primary_default px-4 text-sm font-weight_medium text-text_default_inverted_static transition-colors hover:bg-bg_interaction_primary_hover" + className="h-8 rounded-[8px] bg-bg_interaction_primary_default px-4 text-sm font-weight_medium text-text_label_primary_default transition-colors hover:bg-bg_interaction_primary_hover" > {t("webui.errorBoundary.reload")} diff --git a/packages/webui/webapp/components/add-model-dialog.tsx b/packages/webui/webapp/components/add-model-dialog.tsx index b18338db..930d9a25 100644 --- a/packages/webui/webapp/components/add-model-dialog.tsx +++ b/packages/webui/webapp/components/add-model-dialog.tsx @@ -688,7 +688,7 @@ export function AddModelDialogForm({ disabled={busy || (!skipTest && formTest?.status !== "ok")} aria-busy={busy || undefined} onClick={onCommit} - className="h-9 min-w-20 rounded-lg bg-bg_interaction_primary_default px-5 text-sm font-weight_medium text-text_default_inverted_static shadow-[var(--shadow_default)] transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" + className="h-9 min-w-20 rounded-lg bg-bg_interaction_primary_default px-5 text-sm font-weight_medium text-text_label_primary_default shadow-[var(--shadow_default)] transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" > {busy ? t("providers.saving") : t("providers.dialog.save")} @@ -1527,7 +1527,7 @@ export function FetchedModelsDialogBody({ data-testid="fetched-models-add" disabled={!presetMode || checked.size === 0} onClick={onAdd} - className="h-9 min-w-20 rounded-lg bg-bg_interaction_primary_default px-5 text-sm font-weight_medium text-text_default_inverted_static shadow-[var(--shadow_default)] transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" + className="h-9 min-w-20 rounded-lg bg-bg_interaction_primary_default px-5 text-sm font-weight_medium text-text_label_primary_default shadow-[var(--shadow_default)] transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" > {t("providers.fetched.add")} diff --git a/packages/webui/webapp/components/modals.tsx b/packages/webui/webapp/components/modals.tsx index a88ea379..abc028d6 100644 --- a/packages/webui/webapp/components/modals.tsx +++ b/packages/webui/webapp/components/modals.tsx @@ -214,7 +214,7 @@ function AskModal({ t }: { t: (key: MessageKey) => string }) { "flex flex-none items-center justify-center text-caption-small-strong text-text_default_secondary", multiSelect ? isPicked - ? "size-5 rounded bg-bg_interaction_primary_default text-text_default_inverted_static" + ? "size-5 rounded bg-bg_interaction_primary_default text-text_label_primary_default" : "size-5 rounded border border-border_default bg-bg_grouped_primary" : "size-5 rounded-full bg-bg_grouped_primary", ].join(" ")} @@ -396,7 +396,7 @@ export function Modal({ * The primary (filled) button of a confirm modal. * * UAT fix — the label colour. The label used - * `text-text_default_inverted_static`, which the design system defines as + * `text-text_label_primary_default`, which the design system defines as * "text on an inverted surface" and never re-themes: it stays near-white * in BOTH `:root` (95%) and `.dark` (80%) — see `styles/tokens.css`. The * dark theme inverts `--bg_interaction_primary_default` to `--gray_0` diff --git a/packages/webui/webapp/components/panels.tsx b/packages/webui/webapp/components/panels.tsx index c714cb6b..ccc42443 100644 --- a/packages/webui/webapp/components/panels.tsx +++ b/packages/webui/webapp/components/panels.tsx @@ -2979,7 +2979,7 @@ function WorkspaceBrowseTab({ disabled={busy || !listing?.dir} onClick={() => void pick()} data-testid="workspace-picker-confirm" - className="h-8 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_default_inverted_static transition-colors hover:bg-bg_interaction_primary_hover disabled:opacity-50" + className="h-8 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_label_primary_default transition-colors hover:bg-bg_interaction_primary_hover disabled:opacity-50" > {t("workspace.picker.useWorkspace")} diff --git a/packages/webui/webapp/components/provider-management.tsx b/packages/webui/webapp/components/provider-management.tsx index e8b06409..ac6dde24 100644 --- a/packages/webui/webapp/components/provider-management.tsx +++ b/packages/webui/webapp/components/provider-management.tsx @@ -537,7 +537,7 @@ export function ProviderManagementPanel({ disabled={busy || !validation.ok} aria-busy={busy || undefined} onClick={() => void save()} - className="flex h-8 items-center gap-1.5 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_default_inverted_static transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" + className="flex h-8 items-center gap-1.5 rounded-lg bg-bg_interaction_primary_default px-3 text-sm font-weight_medium text-text_label_primary_default transition-colors hover:bg-bg_interaction_primary_hover disabled:cursor-not-allowed disabled:opacity-50" > {busy ? ( { + const out = []; + const entries = await fs.readdir(dir, { withFileTypes: true }); + for (const entry of entries) { + const full = path.join(dir, entry.name); + if (entry.isDirectory()) out.push(...(await listTsxFiles(full))); + else if (entry.isFile() && entry.name.endsWith(".tsx")) out.push(full); + } + return out.sort(); +} + +test("no primary-interaction surface pairs with the non-flipping inverted text token", async () => { + const offenders = []; + for (const root of SCAN_ROOTS) { + for (const file of await listTsxFiles(root)) { + const source = await fs.readFile(file, "utf8"); + if (!source.includes("text-text_default_inverted_static")) continue; + // An offender is a className string that carries BOTH the primary + // interaction background and the non-flipping inverted token. The + // bg classes and the text token appear in the same string when they + // style the same element — that is the composite that vanishes. + const classNameStrings = + source.match(/"(?:[^"\\]|\\.)*text-text_default_inverted_static(?:[^"\\]|\\.)*"/g) ?? []; + for (const raw of classNameStrings) { + if (!raw.includes("bg-bg_interaction_primary_default")) continue; + offenders.push(`${path.relative(WEBAPP_DIR, file)}: ${raw.slice(0, 100)}`); + } + } + } + assert.deepEqual( + offenders, + [], + "Found primary buttons whose label vanishes in dark theme. Use " + + "text-text_label_primary_default (the token paired with " + + "bg_interaction_primary) instead:\n" + + offenders.join("\n"), + ); +}); + +test("the sanctioned exception survives: the toolbar status badge keeps its inverted token", async () => { + // The toolbar badge sits on bg_status_warning / bg_status_error, which stay + // saturated in dark mode — near-white text is correct there. If this file + // ever drops the pairing, re-evaluate rather than blindly restoring it. + const toolbarPath = path.join(WEBAPP_DIR, "components", "toolbar.tsx"); + const source = await fs.readFile(toolbarPath, "utf8"); + assert.match(source, /text-text_default_inverted_static/); + assert.match(source, /bg-bg_status_(warning|error)/); +}); diff --git a/release/public-source.json b/release/public-source.json index 3fb283df..be49ea09 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3975,6 +3975,7 @@ "packages/webui/webapp/test/sse.test.ts", "packages/webui/webapp/test/store-revision.test.ts", "packages/webui/webapp/test/stream-cursor.test.ts", + "packages/webui/webapp/test/theme-token-pairing.test.ts", "packages/webui/webapp/test/theme.test.ts", "packages/webui/webapp/test/thinking-phrase-rotation.test.ts", "packages/webui/webapp/test/tool-paths.test.ts", From eecd8c0d7b23deb66079358815327b8b4b758357 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Sat, 3 Oct 2026 01:41:37 +0800 Subject: [PATCH 19/21] feat(webui): move session switch behind the engine facade MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit M3-B6: #3 POST /api/sessions/switch now asks the engine facade instead of reaching into lib/acp-client.js, lib/transcript.js, lib/mavis-usage.js, lib/models.js and lib/config.js from the route. The new engine/session-switch.js owns the four load-bearing facts the ~290-line handler had accumulated: the mvs-sid-first resolution order (single base-session identity), the backfill decision and its read, the workspace containment gate (which runs before any cs mutation, so a refused switch leaves the client untouched), and the response body. The gate is SOFT — it reports and never throws — because the switch's primary data is webui's own record and both engine touches have a defined degradation. Gating hard would remove a working endpoint over a title and a transcript, and would do it on the default acp transport first. The route keeps what is its own: the "id required" 400, the status mapping, the fail-closed audit and the SSE push — the audit has to land after the switch has already mutated cs, and the push must not fire when it fails. Behaviour is unchanged and pinned: the four red lines (transcript backfill, cumulative detection, workspace containment, single base session identity) each get named tests with their negative half, and the success body's key ORDER is compared as a string. Six mutations of the facade were run to prove the tests are load-bearing. KNOWN DEBT 1 in the new module records what this batch did NOT retire: the 3-candidate transcript probe. The default acp transport has no engine surface to replace it with, getMessages paginates where the probe caps lines, their orderings differ, and export's enrichment is still byte-pinned to the same candidates. What IS retired is the coupling — the route no longer names lib/transcript.js, and the probe list is an implementation detail behind one seam. --- packages/webui/server/engine/index.js | 41 +- .../webui/server/engine/session-switch.js | 992 ++++++++++++ packages/webui/server/routes/sessions.js | 514 +----- .../test/lib/engine/session-switch.test.js | 1392 +++++++++++++++++ release/public-source.json | 2 + scripts/test-tmp-leak.check.mjs | 4 + 6 files changed, 2505 insertions(+), 440 deletions(-) create mode 100644 packages/webui/server/engine/session-switch.js create mode 100644 packages/webui/test/lib/engine/session-switch.test.js diff --git a/packages/webui/server/engine/index.js b/packages/webui/server/engine/index.js index d91751c4..589a6ac9 100644 --- a/packages/webui/server/engine/index.js +++ b/packages/webui/server/engine/index.js @@ -36,9 +36,9 @@ // catalogue host itself is now reached through this facade too // (engine/host.js), so the plugins and turn-diff routes no longer name // lib/acp-client.js. M3 batches B1 (#9 #10 #72 #74 #75), B2 (#8 #11), -// B3 (#15 #16 #17 #19), B4 (#20 #57 #73) and B5 (#7 #4 #6) done. The -// rest of M3, then M4, will route their consumers through this facade one -// endpoint family at a time. +// B3 (#15 #16 #17 #19), B4 (#20 #57 #73), B5 (#7 #4 #6) and B6 (#3) +// done. The rest of M3, then M4, will route their consumers through +// this facade one endpoint family at a time. import { ENGINE_CAPABILITY_KEYS } from "./capabilities.js"; // Declarations only — importing the provider *host-construction* modules @@ -225,6 +225,41 @@ export { resolveSessionWriteProvider, selectOrphanSessionIds, } from "./session-writes.js"; +// The session SWITCH family (step M3, batch B6): #3 +// POST /api/sessions/switch. Same cycle, same TDZ rule, same reasoning as +// session-writes.js above: session-switch.js reads NOTHING from this module +// at module scope — its `SESSION_SWITCH_ENDPOINTS` table is a literal and +// every binding it needs (`getEngineProvider`, `DEFAULT_ENGINE_PROVIDER_ID`) +// is read inside a function body. A new top-level `const X = +// SOMETHING_FROM_INDEX` in session-switch.js breaks the re-export exactly as +// it would in session-writes.js. Its ONLY static import beyond this module +// is `engine/capabilities.js`; the session store, the ACP client, the +// transcript reader, the usage tables, the workspace gate, the state bus +// and the config are all reached through `await import()` inside the +// data-plane function. +// +// It gates SOFT (`checkSessionSwitchCapability` reports, never throws) for +// the reason `session-export.js` does: the switch's primary data is webui's +// own session record, and both of its engine touches (the title and the +// transcript) have a defined degradation. Gating hard would remove a +// working endpoint in response to a declaration about an enrichment it can +// live without — and would do it on the default `acp` transport first, +// where the enrichment is the only part in question. The 501 machinery +// stays unused by this family, and the suite pins that. +export { + SESSION_SWITCH_ENDPOINTS, + applyEngineSessionSwitch, + applySwitchedSessionToClientState, + chatLooksCumulative, + checkSessionSwitchCapability, + isSwitchableMcodeSessionId, + lookupCachedMcodeTitle, + readEngineSwitchTranscript, + resolveSessionSwitchProvider, + resolveSwitchTarget, + resolveSwitchWorkspace, + selectTranscriptBackfill, +} from "./session-switch.js"; export { LOCAL_RUNTIME_V2_CAPABILITIES } from "./providers/local-runtime-v2.capabilities.js"; export { TUI_RUNTIME_ADAPTER_CAPABILITIES } from "./providers/tui-runtime-adapter.js"; diff --git a/packages/webui/server/engine/session-switch.js b/packages/webui/server/engine/session-switch.js new file mode 100644 index 00000000..a04cc467 --- /dev/null +++ b/packages/webui/server/engine/session-switch.js @@ -0,0 +1,992 @@ +// webui/server/engine/session-switch.js +// +// Migration step M3, batch B6: the session SWITCH endpoint (会话切换) — +// +// #3 POST /api/sessions/switch — switch the active session by webui +// uuid or `mvs_…` sid +// +// What this file is for. #3 is the busiest single endpoint in this +// migration and the one whose failure modes are all user-visible at once: +// a wrong answer here loses the conversation on screen, re-roots the file +// tree on the wrong project, or resurrects the "extra untitled entry" +// sidebar confusion. Before M3 all of it lived in the route — resolve, +// overlay creation, title lookup, transcript backfill, workspace +// containment, per-client state mutation, the response body — in one +// ~290-line handler whose middle half (the engine-facing half) reached +// into `lib/acp-client.js` and `lib/transcript.js` directly. Three facts +// about that handler are load-bearing and none of them is visible from +// the route's edge any more: +// +// 1. THE BACKFILL DECISION IS A DATA DECISION, NOT A ROUTE DECISION. +// A stored chat buffer is written when it is empty OR when it looks +// cumulative (a later `●` line is a strict superset of an earlier +// one — the segment-accumulator bug, session-isolation/06). A clean +// stored buffer is kept untouched even though the engine DB is +// DB-authoritative, because transcript-sync overwrites the stored +// chat within ~4s anyway and clobbering a clean buffer on EVERY +// switch is a worse failure than not backfilling. That rule, and +// the predicate that decides it, belong with the reader that backs +// it up — not in a route that would have to know the difference. +// +// 2. THE READ MUST NEVER BREAK THE SWITCH. Every transcript failure +// path — missing db, unloadable better-sqlite3, schema drift, a +// throwing probe — degrades to "keep the stored chat" and the +// switch still answers 200. That is the endpoint's oldest promise +// and it is the reason this family's gate is SOFT (see below): a +// switch that 501s because an enrichment was unavailable has +// turned a degraded read into a dead endpoint. +// +// 3. THE WORKSPACE WRITE IS A CONTAINMENT GATED SIDE EFFECT. The +// target's stored `workspace` is historical input — it may name a +// directory the user has since removed from the allowed roots. The +// switch resolves target-first, NEVER falls back to the workspace +// the user is currently in (that is the reported "file tree still +// shows the previous project" defect), and refuses with a 400 +// rather than writing a path the picker would have rejected. The +// gate is `lib/workspace.js#assertWorkspacePath` — the same one +// `handleWorkspaceChange`, `handleNewSession` and the fs routes use +// — and it runs BEFORE any `cs` mutation, so a refused switch +// leaves the client state exactly as it was. +// +// Why this family's gate is SOFT, when the write family (B5) gates hard +// and the tree family (B2) gates hard. The question behind that choice +// is "if the provider declares this capability absent, can the endpoint +// still serve a truthful answer?" — and for #3 the answer is yes: +// +// - The payload's primary data is webui's OWN store. The record, its +// title, its chat and its workspace all live in `sessions.json`. +// - Both engine touches are enrichments that already have a defined +// degradation: the title falls back to the cache and then to the +// "Mcode session" placeholder, the transcript falls back to the +// stored chat. Neither failure is visible as a failure. +// - Gating hard would REMOVE a working endpoint in response to a +// declaration about a capability it does not depend on, and it would +// do so under exactly the transport where the endpoint has the most +// users. That is B2's `session-export.js` argument, reused rather +// than re-argued: a missing enrichment must not be dressed up as a +// failure (#110 fake-success discipline, applied in the other +// direction). +// +// So `checkSessionSwitchCapability` REPORTS and never throws. The 501 +// machinery in `errors.js` stays unused by this family — a policy +// statement, and the suite pins that it stays unused. +// +// What this file deliberately does NOT do: +// +// - It does not re-implement the transcript. `lib/transcript.js` owns +// the read and `messagesToChatLines` owns the chat-line grammar; this +// file owns the DECISION to read and the decision to keep what came +// back. See KNOWN DEBT 1 for why the 3-candidate probe behind that +// read survives this batch. +// - It does not own the workspace boundary. `assertWorkspacePath` +// stays the single gate every workspace write funnels through. +// - It does not own the session store. `lib/sessions.js` keeps the +// load/save and the overlay rule; this file orders the calls. +// - It does not build a host. There is no host on this path at all. +// +// Boot-path weight. `app.js` imports the routes, the routes import this +// file, so this file is on the boot path. It statically imports nothing +// heavier than `engine/capabilities.js` and `engine/index.js` (both pure +// declaration modules) and nothing else; `lib/sessions.js`, +// `lib/acp-client.js`, `lib/transcript.js`, `lib/mavis-usage.js`, +// `lib/models.js`, `lib/workspace.js`, `lib/state-bus.js` and +// `lib/config.js` are all reached through `await import()` inside the +// data-plane function. That split is the M1 lesson, and it is what lets +// this module be re-exported from `engine/index.js` at all. The pure +// derivations below take their dependencies as arguments for the same +// reason twice over: they stay testable without a module registry, and +// the boot path never sees a workspace or sqlite import. +// +// Provider selection is M4's job, same as B1 through B5: +// `providerByTransport()` maps a transport to a REGISTERED provider id; +// today only `runtime` has one, so under the default `acp` transport the +// gate reports `gate: "unregistered-transport"` and the switch proceeds — +// which is correct, because the pre-M4 behaviour under `acp` is the only +// behaviour this endpoint has ever had. + +import { assertEngineCapability } from "./capabilities.js"; +import { DEFAULT_ENGINE_PROVIDER_ID, getEngineProvider } from "./index.js"; + +/** + * Transport → registered engine provider id. Absent means "no provider + * claims this transport yet" (M4), NOT "the capability is unavailable" — + * the two answer differently on purpose, exactly as in + * `session-reads.js#providerByTransport`, `session-tree-reads.js`, + * `usage-reads.js`, `account-reads.js` and `session-writes.js`, which + * this mirrors rather than merges: six families with separate contracts, + * and a shared table would force this one to inherit another's policy. + * + * Built per call rather than frozen at module scope: `engine/index.js` + * re-exports this module, so a module-level table would read + * `DEFAULT_ENGINE_PROVIDER_ID` while that binding is still in its + * temporal dead zone on a cold `import("./engine/index.js")`. Every + * consumer of the table is a function anyway. + * + * @returns {Readonly>} + */ +function providerByTransport() { + return Object.freeze({ runtime: DEFAULT_ENGINE_PROVIDER_ID }); +} + +// --------------------------------------------------------------------------- +// The declaration, and the gate policy that goes with it +// --------------------------------------------------------------------------- + +/** + * The declaration this endpoint's ENRICHMENTS need. + * + * `sessionCrud` / `getSession` is the honest mapping and it is the same + * pair B1 uses for `GET /api/acp-session-title` and B2 uses for the + * export enrichment: reading a session's title and reading its transcript + * are both reading that session. `usageStats` is deliberately NOT + * declared, and the reason is worth stating because the switch does + * touch token usage: `applyMavisUsageToCs` reads webui's OWN mavis usage + * tables, not a provider method, and it is fire-and-forget — its failure + * path has been a `catch` with a debug-only warning since before M3. A + * declaration there would gate a working endpoint on a capability whose + * absence changes nothing the user can see. + * + * @type {Readonly>} + */ +export const SESSION_SWITCH_ENDPOINTS = Object.freeze({ + "POST /api/sessions/switch": Object.freeze({ + capability: "sessionCrud", + subItem: "getSession", + enforcement: "soft", + }), +}); + +/** + * Resolve the provider that answers the switch on `transport`, or `null` + * when none is registered yet. + * + * @param {string} transport One of the `MCODE_WEBUI_TRANSPORT` values. + * @returns {{id: string, transport: string, capabilities: object}|null} + */ +export function resolveSessionSwitchProvider(transport) { + const providerId = providerByTransport()[transport]; + if (!providerId) return null; + return getEngineProvider(providerId); +} + +/** + * Read the declaration for this endpoint WITHOUT enforcing it. + * + * Returns a descriptor whose `gate` field says what happened: + * + * - `"checked"` — provider resolved, capability is `full`. + * - `"unregistered-transport"` — no provider claims this transport yet. + * This is the DEFAULT `acp` transport, and the switch proceeding + * here is the pre-M3 behaviour, not a hole in the gate. + * - `"capability-absent"` — the provider WAS found and DOES declare + * the capability as `none`. The caller's next move is to degrade the + * enrichment (placeholder title, stored chat), never to fail the + * request. + * - `"partial"` — provider is `partial` and this sub-item + * is absent; the endpoint still degrades, but says so precisely. + * + * Deliberately never throws `EngineCapabilityNotSupportedError`. A + * genuinely unknown endpoint key is still a plain Error — caller + * confusion is not a capability question, and the HTTP layer must never + * answer 501 for a typo in webui's own code. + * + * @param {string} endpoint A key of SESSION_SWITCH_ENDPOINTS. + * @param {string} transport The active transport. + * @returns {{endpoint: string, gate: string, provider: string|null, capability: string|null, subItem: string|null, enforcement: "soft"}} + */ +export function checkSessionSwitchCapability(endpoint, transport) { + const need = SESSION_SWITCH_ENDPOINTS[endpoint]; + if (need === undefined) { + const err = new Error( + `checkSessionSwitchCapability: "${endpoint}" is not part of the session-switch family ` + + `(known: ${Object.keys(SESSION_SWITCH_ENDPOINTS).join(", ")})`, + ); + err.code = "unknown_session_switch_endpoint"; + throw err; + } + const base = { + endpoint, + provider: null, + capability: need.capability, + subItem: need.subItem, + enforcement: need.enforcement, + }; + const provider = resolveSessionSwitchProvider(transport); + if (!provider) return { ...base, gate: "unregistered-transport" }; + const entry = provider.capabilities ? provider.capabilities[need.capability] : undefined; + const descriptor = { ...base, provider: provider.id }; + if (entry && entry.level === "full") { + return { ...descriptor, gate: "checked" }; + } + if (entry && entry.level === "partial") { + const absent = Array.isArray(entry.missing) && entry.missing.includes(need.subItem); + return { ...descriptor, gate: absent ? "partial" : "checked" }; + } + // `none`, or no entry at all — the provider was found and does not + // offer this. Report it; the caller degrades the enrichment. + return { ...descriptor, gate: "capability-absent" }; +} + +// --------------------------------------------------------------------------- +// Pure derivations. Exported and tested on their INPUTS. +// --------------------------------------------------------------------------- + +/** + * The engine's own session-id shape, as this endpoint asks it. + * + * The same regex as `session-writes.js#isMcodeSessionId`, spelled again + * rather than imported: the write family's copy is reachable only + * through the write gate's module, and a switch that could not run + * without the delete family's declaration would couple two endpoints + * that have no reason to move together. The rule itself is one line and + * both copies are pinned by both suites, so a drift shows up as a red + * test in whichever family changed, not as a silent behaviour change. + * + * @param {unknown} id + * @returns {boolean} + */ +export function isSwitchableMcodeSessionId(id) { + return typeof id === "string" && /^mvs_[a-f0-9]{32}$/.test(id); +} + +/** + * Resolve a caller-supplied id against the session store. + * + * The order is MCODE SID FIRST, then webui uuid — and that is the OPPOSITE + * of `session-writes.js#resolveSessionTarget`, which is uuid-first. The + * two are not interchangeable and the difference is a product rule, not a + * style choice: the switch's whole point is single base-session identity + * (one conversation, one record — the "extra untitled entry" sidebar + * confusion the overlay rule was written to kill), so a switch addressed + * by `mvs_` must land on the record that IS that engine session even if + * some other record's uuid could be made to match the same string. The + * delete and rename paths are addressed by a user who already has the + * record in front of them and look the uuid up first. + * + * `matchKind` is `null` — never `"unknown"`, never `""` — exactly when the + * id resolved to nothing. The caller writes `matchKind || "new_from_mcode"` + * into the audit payload itself, because that fallback is part of the + * audit contract and is spelled out at its one call site. + * + * @param {Array} records The loaded session store. + * @param {string} id The id from the request. + * @returns {{index: number, matchKind: "webuiId"|"mcodeSessionId"|null, target: object|null}} + */ +export function resolveSwitchTarget(records, id) { + const list = Array.isArray(records) ? records : []; + let index = list.findIndex((s) => s && s.mcodeSessionId === id); + let matchKind = index >= 0 ? "mcodeSessionId" : null; + if (index < 0) { + index = list.findIndex((s) => s && s.id === id); + if (index >= 0) matchKind = "webuiId"; + } + return { + index, + matchKind, + target: index >= 0 ? list[index] : null, + }; +} + +/** + * Detect the cumulative-render pollution pattern in a stored chat buffer + * (session-isolation/06). When the engine emits each segment of an + * `agent_message`, the stream writer emits a new `●` line; a + * non-cumulative buffer has each line containing only its own segment's + * text. A cumulative buffer — the bug — has at least one later `●` line + * whose text is a strict superset of an earlier `●` line (the accumulator + * never reset between segments and every later line re-wrote every prior + * segment's text). + * + * This predicate is O(n^2) in the number of `●` lines, but a single + * session's `chat` is bounded (~400 lines by the transcript cap) so the + * worst case is a few thousand substring checks per switch — cheap + * enough to run on the hot path. + * + * Conservative on both sides: + * - a single-`●`-line buffer is never cumulative; + * - non-`●` lines (system, tool, ▲ thought) are ignored — only `●` + * rows matter, since the cumulative bug only affects message + * segments; + * - ties (equal-length `●` lines) are NOT cumulative — same length, no + * superset relation. + * + * @param {unknown} chat + * @returns {boolean} + */ +export function chatLooksCumulative(chat) { + if (!Array.isArray(chat) || chat.length === 0) return false; + const dots = []; + for (const line of chat) { + if (typeof line !== "string") continue; + // Match the same prefix the streamer writes: `● ` then text. Also + // accept a bare `●` at end-of-line (transcript-sync appends + // stripped-down `●` markers in some paths) without treating it as + // evidence of anything. + if (line.startsWith("● ")) dots.push(line.slice(2)); + } + for (let i = 0; i < dots.length; i += 1) { + for (let j = i + 1; j < dots.length; j += 1) { + const a = dots[i]; + const b = dots[j]; + if (b.length <= a.length) continue; // strict superset ⇒ longer + if (b.includes(a)) return true; + } + } + return false; +} + +/** + * Whether the switch should read the engine transcript for a stored + * buffer, and why — the decision, with no I/O in it. + * + * The rule (session-isolation/06) and its three branches: + * + * - stored chat empty → backfill. Unchanged since the first version of + * this path: a session that has never been rendered must show its + * history, not "No messages yet". + * - stored chat looks cumulative → prefer the engine read and + * re-persist. The original rule only backfilled when the buffer was + * empty, so a polluted buffer persisted via `saveSessions` and won + * forever. `reason` reports which branch fired so the operator log + * distinguishes "first touch" from "repaired pollution". + * - otherwise → keep the stored chat. DB-authoritative: + * transcript-sync overwrites the stored chat from the engine within + * ~4s, so stored-only lines a user typed but never sent will be lost + * regardless, and clobbering a clean buffer on EVERY switch is the + * worse failure. This deliberately does NOT promise draft + * preservation — the composer keeps its own draft in its own state + * (see `composer-draft.test.ts`). + * + * @param {unknown} chat The record's stored `chat` array. + * @returns {{storedHasChat: boolean, storedCumulative: boolean, shouldBackfill: boolean, reason: "empty"|"stored_cumulative"|"stored_shrinks"|null}} + */ +export function selectTranscriptBackfill(chat) { + const storedHasChat = Array.isArray(chat) && chat.length > 0; + const storedCumulative = storedHasChat && chatLooksCumulative(chat); + if (!storedHasChat) { + return { storedHasChat, storedCumulative, shouldBackfill: true, reason: "empty" }; + } + if (storedCumulative) { + return { storedHasChat, storedCumulative, shouldBackfill: true, reason: "stored_cumulative" }; + } + return { storedHasChat, storedCumulative, shouldBackfill: false, reason: "stored_shrinks" }; +} + +/** + * Resolve the title of an `mvs_…` session from the in-memory + * walked-session cache, WITHOUT awaiting anything and WITHOUT touching + * the ACP child. + * + * The cache-first rule is a latency rule with a measured number behind + * it: `getMcodeSessionTitle` boots the ACP child, ~2.17s end-to-end with + * a broken mcode binary, AND used to degrade the title to the "Mcode + * session" placeholder even though the cache already held the real one. + * + * Cross-workspace matching within what the module exposes: the cache + * holds ONE workspace's list, keyed by ws. Both the fresh (30s TTL) and + * the stale (same-ws, TTL-expired) readers are probed, plus the `""` + * key — the unfiltered list, so a cache walked without a workspace still + * answers. A miss returns `null` and the caller falls back to + * `getMcodeSessionTitle`. + * + * The two getters are PARAMETERS rather than imports so this stays a + * pure function over the cache, and so the boot path never reaches + * `lib/acp-client.js` (which carries the ACP client tree). + * + * @param {string} mcodeSessionId + * @param {string} ws The workspace the caller is currently in. + * @param {object} getters `{fresh, stale}` — the two cache readers. + * @returns {string|null} + */ +export function lookupCachedMcodeTitle(mcodeSessionId, ws, getters) { + if (!mcodeSessionId) return null; + const fresh = getters && getters.fresh; + const stale = getters && getters.stale; + if (typeof fresh !== "function" || typeof stale !== "function") return null; + for (const wsKey of [ws || "", ""]) { + for (const getter of [fresh, stale]) { + let sessions = null; + try { + sessions = getter(wsKey); + } catch { + sessions = null; + } + if (!Array.isArray(sessions)) continue; + const hit = sessions.find( + (s) => s && s.sessionId === mcodeSessionId && s.title, + ); + if (hit && hit.title) return hit.title; + } + } + return null; +} + +/** + * Pick the workspace the switched-into session "belongs to" and run it + * through the same containment gate the workspace picker / + * `handleNewSession` / `browseWorkspace` all funnel through. + * + * Source priority (s39 — webui-parity ticket 39: the file tree must + * follow the switched session): + * + * 1. The target session's stored `workspace` field — that IS the + * workspace the user was in when they last had it open, modulo any + * pollution the old code introduced. Real existence + containment + * are checked; an out-of-bounds or stale value surfaces as a 400 + * so the user can either widen the allowed roots or pick a fresh + * workspace, instead of silently landing on the previous project. + * + * 2. `defaultWorkspace` (env `MCODE_WORKSPACE` > mcode TUI cwd.json > + * homedir) when the stored value is empty. Empty is also the value + * seen for (a) records created by the old code that polluted + * freshly-typed mvs sessions with the current `cs.workspace` (the + * data-corruption bug this ticket fixes), and (b) older sessions + * that pre-date the workspace field. `DEFAULT_WORKSPACE` is already + * in the default allowed-roots surface (see + * `getAllowedWorkspaceRoots`), so containment accepts it without env + * setup. + * + * Critical invariants: + * - The switch NEVER keeps `cs.workspace` on the prior project. The + * user-reported symptom was exactly that: "the file tree still shows + * the previous project's files". Falling back to the current + * workspace when the target's is empty is the bug being removed — + * which is why `currentWs` is NOT a parameter of this function even + * though the route still computes it for the log line. + * - The switch NEVER writes a path the containment gate rejected. A + * 400 carrying the gate's actionable error is the only acceptable + * outcome. + * - The switch NEVER overwrites a target session's stored workspace + * with the current one. That was the pollution path; new overlay + * records (mvs_ first-touch) get `workspace: ""` and the + * target-first read lands on the default for them. + * + * Both dependencies are parameters for the same reason as + * `lookupCachedMcodeTitle`: this is a decision over two values, and it + * has to be testable — and boot-path-light — without the workspace + * module and the config module in the graph. + * + * @param {object} target The resolved session record. + * @param {object} deps + * @param {string} deps.defaultWorkspace `DEFAULT_WORKSPACE`. + * @param {(p: string) => {ok: boolean, path?: string, real?: string, error?: string}} deps.assertPath + * @returns {{ok: true, dir: string, real: string|undefined, fallback: boolean}|{ok: false, error: string, attempted: string}} + */ +export function resolveSwitchWorkspace(target, deps) { + const raw = + target && typeof target.workspace === "string" ? target.workspace.trim() : ""; + // Empty / non-string / null → the default workspace. Never the + // current one — that is the user-reported "stays on the old project" + // failure mode this rule removes. + const candidate = raw || (deps && deps.defaultWorkspace) || ""; + const gate = deps.assertPath(candidate); + if (!gate.ok) { + return { ok: false, error: gate.error, attempted: candidate }; + } + return { ok: true, dir: gate.path, real: gate.real, fallback: !raw }; +} + +/** + * The per-client state a switch applies, as a pure field assignment over + * one client's state object. + * + * What it does and does not touch. It sets the identity, the title, the + * chat buffer, zeroes the three cumulative per-session usage counters + * and RE-ROOTS the workspace. It is not responsible for `resetContext` — + * that is a `lib/sessions.js` call with its own mocked parity, and the + * caller runs it right after, so the ordering (`resetContext` sees the + * new identity) stays the caller's to keep. + * + * `lastUsedWorkspace` is deliberately untouched, and that is a product + * rule rather than an omission: last-used is written only by the send + * path (a workspace change / a sent prompt), because switching is + * browsing. Pinning the browsed workspace to the top of the sidebar is + * the user-reported "click any session in C and C auto-sorts first" + * behaviour, and this function is where that is kept true. + * + * @param {object} cs A webui client state. Mutated in place. + * @param {object} opts + * @param {object} opts.target The resolved session record. + * @param {string} opts.workspaceDir The containment-gated directory. + * @returns {object} The same `cs`, for chaining. + */ +export function applySwitchedSessionToClientState(cs, opts) { + const { target, workspaceDir } = opts; + cs.sessionId = target.id; + cs.mcodeSessionId = target.mcodeSessionId || null; + cs.sessionTitle = target.title || "Untitled"; + cs.chat = Array.isArray(target.chat) ? target.chat : []; + cs.usage = { + ...cs.usage, + sessionInput: 0, + sessionOutput: 0, + sessionTotal: 0, + }; + cs.workspace = { + dir: workspaceDir, + branch: null, + tree: null, + }; + return cs; +} + +// --------------------------------------------------------------------------- +// The engine-facing read +// --------------------------------------------------------------------------- + +/** + * Where the transcript read's bytes came from. `engine` when the reader + * answered with lines; `none` when it did not, and the caller keeps the + * stored chat. The value exists so a consumer never has to guess. + * + * @typedef {"engine" | "none"} SessionSwitchTranscriptSource + */ + +/** + * The switch's one engine-facing read: one session's transcript, mapped + * into the webui chat-line grammar, best-effort. + * + * NEVER THROWS. Every failure — unknown endpoint key aside, which is a + * caller bug — lands as `{ok: false, reason}` and the caller keeps the + * stored chat. That containment used to live in a `try/catch` wrapped + * around the whole block in the route; it is a property of the READ + * here, so a future caller of this seam cannot get it wrong. + * + * `lines` / `messageCount` / `truncated` / `probe` are the reader's own + * values forwarded verbatim — this facade invents no reason code and + * never converts a failure into an exception, because the operator log + * that reports `reason` and the log's own vocabulary are one contract. + * + * Async even though the reader is synchronous (better-sqlite3 is sync): + * the route is already async, and a uniform awaitable `readEngine*` + * seam means a provider-backed transcript source that IS async (a network + * engine) needs no signature change at this layer. + * + * @param {object} [options] + * @param {string} [options.mcodeSessionId] The `mvs_…` id to read. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/switch`. + * @param {string} [options.transport] Transport override; defaults to the + * active `MCODE_WEBUI_TRANSPORT`. + * @returns {Promise<{mcodeSessionId: string, lines: Array, ok: boolean, reason: string|null, probeTable: string|null, probe: string|null, messageCount: number, truncated: boolean, source: SessionSwitchTranscriptSource, gate: object, transport: string}>} + */ +export async function readEngineSwitchTranscript(options = {}) { + const endpoint = options.endpoint || "POST /api/sessions/switch"; + const [transcript, config] = await Promise.all([ + import("../lib/transcript.js"), + import("../lib/config.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = checkSessionSwitchCapability(endpoint, transport); + const mcodeSessionId = options.mcodeSessionId || ""; + const r = transcript.loadTranscriptChatLines(mcodeSessionId, { + dbPath: config.MCODE_RUNTIME_DB, + }); + return { + mcodeSessionId, + lines: r.ok && Array.isArray(r.lines) ? r.lines : [], + ok: r.ok === true, + reason: r.ok === true ? null : r.reason || "unknown", + probeTable: r.source || null, + probe: r.probe || null, + messageCount: r.messageCount || 0, + truncated: r.truncated === true, + source: r.ok === true ? "engine" : "none", + gate, + transport, + }; +} + +// --------------------------------------------------------------------------- +// Data plane +// --------------------------------------------------------------------------- + +/** + * #3 — the switch. + * + * The order below IS the endpoint's contract, and each step is here + * because moving it would change what the user sees: + * + * 1. LOAD + RESOLVE. `mvs_` sid first, then webui uuid (see + * `resolveSwitchTarget`). + * 2. FIRST TOUCH. An `mvs_` sid with no webui record gets ONE overlay + * record whose id IS the mvs sid (idempotent create), titled from + * the walked-session cache and only then from the engine. An id that + * is neither → `not_found` and the route answers 404. Note that an + * unresolved id that is NOT an mvs sid is a value, not an error: + * the route owns the status code. + * 3. PLACEHOLDER REPAIR. Wrappers created during the broken-title + * window carry "Mcode session" forever; if the walked cache now has + * the real title, repair the stored wrapper. Cache-only, and it runs + * BEFORE the backfill so the repaired title is what the response + * carries. + * 4. TRANSCRIPT BACKFILL, under `selectTranscriptBackfill`'s rule. The + * only step that writes a non-empty buffer, and the only one that + * can fail harmlessly. + * 5. WORKSPACE CONTAINMENT. Runs before ANY `cs` mutation, so a + * refused switch (`workspace_refused`) leaves the client exactly as + * it was — which is why this outcome exists as a third value next to + * `ok` and `not_found` instead of an exception. + * 6. APPLY. Identity, title, chat, usage counters, workspace — then + * `resetContext`, then the fire-and-forget usage sync. + * + * The usage sync is started here and NOT awaited, exactly as the route + * did: it pushes a state frame on its own when it settles, and the + * switch's own response must not wait on a usage table read. + * + * The response body is built HERE and never re-assembled in the route, + * including the `chat` projection: `runChatViewChat` is a pure read of + * the run registry and the client state, and nothing between this call + * and the response mutates either, so computing it one step earlier + * cannot change a byte. The test suite pins the mid-run case (the + * run-mirror contract, session-isolation/02) to keep that true. + * + * @param {object} options + * @param {string} options.id The id from the request; already + * validated non-empty by the route. + * @param {object} options.cs The requesting client's state. Mutated. + * @param {string} [options.cid] Requesting client id. + * @param {string} [options.endpoint] Endpoint key for the declaration + * check; defaults to `/api/sessions/switch`. + * @param {string} [options.transport] Transport override. + * @returns {Promise<{outcome: "ok"|"not_found"|"workspace_refused", matchKind: string|null, target: object|null, workspace: object|null, transcript: object|null, audit: object|null, payload: object, statusHint: number, gate: object, transport: string}>} + */ +export async function applyEngineSessionSwitch(options = {}) { + const endpoint = options.endpoint || "POST /api/sessions/switch"; + const [sessions, acp, config, workspaceLib, bus, mavis, models] = await Promise.all([ + import("../lib/sessions.js"), + import("../lib/acp-client.js"), + import("../lib/config.js"), + import("../lib/workspace.js"), + import("../lib/state-bus.js"), + import("../lib/mavis-usage.js"), + import("../lib/models.js"), + ]); + const transport = options.transport || config.MCODE_WEBUI_TRANSPORT; + const gate = checkSessionSwitchCapability(endpoint, transport); + const { id, cs, cid } = options; + const all = sessions.loadSessions(); + console.log( + `[switch] cid=${cid} incoming id=${id.substring(0, 12)}… isMcodeSid=${isSwitchableMcodeSessionId(id)} allTotal=${all.length}`, + ); + const { matchKind: foundKind, target: found } = resolveSwitchTarget(all, id); + let target = found; + let matchKind = foundKind; + console.log( + `[switch] cid=${cid} match=${matchKind || "NONE"} target.id=${target ? target.id.substring(0, 8) : "null"}… target.mcodeSid=${target && target.mcodeSessionId ? target.mcodeSessionId.substring(0, 12) : "null"}… target.chatLen=${target ? (target.chat ? target.chat.length : 0) : 0} target.title="${target ? (target.title || "").substring(0, 30) : ""}"`, + ); + + if (!target) { + if (!isSwitchableMcodeSessionId(id)) { + console.log( + `[switch] cid=${cid} 404 id=${id} not found and not mcode sid`, + ); + return { + outcome: "not_found", + matchKind: null, + target: null, + workspace: null, + transcript: null, + audit: null, + payload: { ok: false, error: "session not found" }, + statusHint: 404, + gate, + transport, + }; + } + // Cache-first title — the walked session cache usually already holds + // the real title (the sidebar just rendered it). Only a total cache + // miss pays the `getMcodeSessionTitle` cost. + const currentWs = (cs && cs.workspace && cs.workspace.dir) || ""; + let title = lookupCachedMcodeTitle(id, currentWs, { + fresh: acp.getMcodeSessionsCacheSync, + stale: acp.getMcodeSessionsStaleSync, + }); + const titleSource = title ? "cache" : "acp"; + if (!title) { + title = (await acp.getMcodeSessionTitle(id)) || "Mcode session"; + } + // Single base session — overlay record id === mcode session id, + // idempotent create. The old model gave each mvs_ switch a fresh uuid + // wrapper, so the same conversation had two identities, the direct + // cause of the "extra untitled entry" sidebar confusion. + // + // No workspace argument, and none is ever passed: stamping the + // freshly-created overlay with the CURRENT workspace stamped every + // first-touch of an mvs session from project A with project A's + // path, and switching back from project B then either left the file + // tree stuck on B or overwrote the overlay (s39 / webui-parity 63). + // New overlays start with `workspace: ""`; the target-first read + // below lands on the default for them. + const existed = sessions.findOverlayForMcodeSid(all, id); + target = sessions.ensureOverlayForMcodeSid(all, id, { title }); + target.updatedAt = Date.now(); + sessions.saveSessions(all); + // `matchKind` stays `null` here on purpose: the audit payload's + // `matchKind || "new_from_mcode"` fallback is part of the B01 audit + // contract, and "new_from_mcode" is the label operators read when a + // switch invented the wrapper. Labelling it `mcodeSessionId` would + // rewrite history for every first-touch switch. + console.log( + `[switch] cid=${cid} ${existed ? "reused" : "created"} overlay ${target.id.substring(0, 12)}… (id=mcode sid) title="${title}" titleSource=${titleSource}`, + ); + } else if ( + // Placeholder refresh — wrappers created during a broken-title window + // carry "Mcode session" forever. If the walked cache now has the real + // title, repair the stored wrapper. Cache-only (sync, no ACP boot): + // an existing wrapper must never make the hot path slower. + target.title === "Mcode session" && + target.mcodeSessionId && + isSwitchableMcodeSessionId(target.mcodeSessionId) + ) { + const cachedTitle = lookupCachedMcodeTitle( + target.mcodeSessionId, + (cs.workspace && cs.workspace.dir) || "", + { + fresh: acp.getMcodeSessionsCacheSync, + stale: acp.getMcodeSessionsStaleSync, + }, + ); + if (cachedTitle) { + target.title = cachedTitle; + target.updatedAt = Date.now(); + sessions.saveSessions(all); + console.log( + `[switch] cid=${cid} refreshed placeholder title for ${target.id.substring(0, 8)}… → "${cachedTitle}"`, + ); + } + } + + // Transcript backfill — when the resolved target has NO webui chat yet + // but IS a real mvs_ session, load the engine transcript and map it + // into the webui chat-line grammar BEFORE responding, so the response + // `session.chat` and `cs.chat` both carry history. Caps inside the + // reader (last 400 lines / 200KB) keep the SSE state push bounded; a + // 1000+-message session must not balloon it. + // + // FAILURE MUST NOT BREAK SWITCHING: the read is contained in + // `readEngineSwitchTranscript`, and a failure here logs and continues + // with the original chat — the switch itself always succeeds. + let transcript = null; + if (target.mcodeSessionId && isSwitchableMcodeSessionId(target.mcodeSessionId)) { + const decision = selectTranscriptBackfill(target.chat); + if (decision.shouldBackfill) { + try { + const read = await readEngineSwitchTranscript({ + mcodeSessionId: target.mcodeSessionId, + endpoint, + transport, + }); + // `decision` is the BRANCH that fired, not the read's outcome — + // the two answer different questions and the operator log needs + // both ("we re-read because the buffer was polluted" versus "the + // re-read found nothing"). It rides on the read's result because + // that object only exists when a read was actually attempted. + transcript = { ...read, decision: decision.reason }; + if (transcript.ok && transcript.lines.length > 0) { + target.chat = transcript.lines; + target.updatedAt = Date.now(); + sessions.saveSessions(all); // persist the populated wrapper + console.log( + `[switch] cid=${cid} transcript backfill ${target.id.substring(0, 8)}… mcode=${target.mcodeSessionId.substring(0, 12)}… reason=${decision.reason} lines=${transcript.lines.length} msgs=${transcript.messageCount} probe=${transcript.probe}${transcript.truncated ? " (capped)" : ""}`, + ); + } else if (!transcript.ok) { + console.log( + `[switch] cid=${cid} transcript unavailable for ${target.mcodeSessionId.substring(0, 12)}… reason=${transcript.reason || "unknown"}`, + ); + } else if (decision.storedCumulative) { + // Cumulative buffer + the read came back empty — preserve the + // stored chat (which is at least the user's last view) and log + // the discrepancy so a post-mortem can see what happened. + console.log( + `[switch] cid=${cid} stored chat looked cumulative but the transcript read returned no lines; preserving stored chat for ${target.mcodeSessionId.substring(0, 12)}…`, + ); + } + } catch (e) { + // Belt and braces: the read is written not to throw, but a + // module-load failure in the dynamic import would land here, and + // a switch that 500s because a transcript could not be loaded is + // the failure mode this endpoint has never had. + console.warn( + `[switch] cid=${cid} transcript backfill failed for ${target.mcodeSessionId.substring(0, 12)}… (continuing with stored chat):`, + e && e.message ? e.message : e, + ); + } + } + } + + const prevSid = cs.sessionId; + // s39: resolve the target session's workspace and re-point + // `cs.workspace.dir` to it BEFORE any other cs mutation, so the SSE + // state push and the response payload both carry the new workspace in + // lockstep with the session-id switch. The pre-fix behaviour read + // `cs.workspace` without writing it, which left the file tree bound to + // the previous project. + const switchWs = resolveSwitchWorkspace(target, { + defaultWorkspace: config.DEFAULT_WORKSPACE, + assertPath: workspaceLib.assertWorkspacePath, + }); + if (!switchWs.ok) { + console.log( + `[switch] cid=${cid} REFUSED id=${id.substring(0, 12)}… reason=workspace_containment attempted="${switchWs.attempted}"`, + ); + return { + outcome: "workspace_refused", + matchKind, + target, + workspace: switchWs, + transcript, + audit: null, + payload: { + ok: false, + error: switchWs.error, + attempted: switchWs.attempted, + }, + statusHint: 400, + gate, + transport, + }; + } + if (switchWs.fallback) { + console.log( + `[switch] cid=${cid} target ${target.id.substring(0, 8)}… had no workspace — fell back to DEFAULT_WORKSPACE=${switchWs.dir}`, + ); + } + applySwitchedSessionToClientState(cs, { target, workspaceDir: switchWs.dir }); + sessions.resetContext(cs); + // Sync real token usage from the mavis db on switch to a historical + // session. Fire-and-forget, exactly as before: its failure path is a + // debug-only warning and the switch's own response must not wait on a + // usage table read. + if (cs.mcodeSessionId) { + const switchedSid = cs.mcodeSessionId; + mavis + .applyMavisUsageToCs(cs, switchedSid, { getMcodeModelLimit: models.getMcodeModelLimit }) + .then(() => bus.pushStateFor(cid)) + .catch((e) => { + if (process.env.MCODE_USAGE_DEBUG) + console.warn(`[switch.mavis] cid=${cid} error: ${e.message}`); + }); + } + return { + outcome: "ok", + matchKind, + target, + workspace: switchWs, + transcript, + // The audit event. B01: a switch records which session was activated + // and from which prior session, plus (s39) which workspace the switch + // landed on and whether that was the DEFAULT_WORKSPACE fallback — + // both useful when auditing "why did the file tree change" or "why is + // the sidebar sorting by a directory I never opened". The route + // appends it and owns the fail-closed 500, because the write-ahead + // ordering between "know what to switch to" and "tell anyone" is the + // route's to keep. + audit: { + event: "session.switch", + target: cs.sessionId, + cid, + actor: "user", + payload: { + from: prevSid || "", + matchKind: matchKind || "new_from_mcode", + mcodeSessionId: cs.mcodeSessionId || "", + title: cs.sessionTitle, + workspace: switchWs.dir, + workspaceFallback: !!switchWs.fallback, + }, + }, + payload: { + ok: true, + session: { + id: target.id, + mcodeSessionId: cs.mcodeSessionId, + title: cs.sessionTitle, + // s39: surface the new workspace in the response so the client + // (url-restore + session-tree) can update its in-memory state + // without waiting for the SSE state-bus push to land. + workspace: switchWs.dir, + workspaceFallback: !!switchWs.fallback, + // session-isolation/02 (run-mirror): switching back to the + // session that is mid-run must show what it produced so far. + chat: bus.runChatViewChat(cid, cs), + }, + }, + statusHint: 200, + gate, + transport, + }; +} + +// --------------------------------------------------------------------------- +// KNOWN DEBT +// --------------------------------------------------------------------------- +// +// Recorded here rather than fixed, because each item is a decision that +// belongs to a human and not to a refactor: +// +// 1. THE 3-CANDIDATE TRANSCRIPT PROBE IS STILL HERE, and this batch is +// the batch the plan named for retiring it (plan §7: "transcript DB +// 探针 … → getMessages"). It could not be retired without breaking +// this batch's own red line, and the reason is not a matter of taste: +// +// a. THE DEFAULT TRANSPORT HAS NO ENGINE SURFACE. The `acp` +// transport is the default and NO provider is registered for +// it — `providerByTransport()` returns `{runtime: …}` only, +// precisely so this gate reports +// `gate: "unregistered-transport"` and the pre-M3 behaviour +// survives. `cliService.getMessages` is reachable only through +// the v2 catalogue host, which only the `runtime` transport +// boots. Deleting the probe therefore empties the backfill on +// the default transport and on half of the two-transport test +// matrix this batch is gated on. That is red line 1 +// (转录回填) failing, not a refactor completing. +// b. THE TWO READS CAP DIFFERENT THINGS. The probe reads a whole +// session and caps the mapped LINES at 400 / 200KB +// (`messagesToChatLines`). `getMessages` paginates — +// `limit`, `before`, `nextCursor`, `hasMore` — so it caps +// MESSAGES. The two are interchangeable only after proving +// that the tail of a bounded message page yields the same +// 400 lines, which needs a live v2 host to measure. +// c. THE ORDERING IS NOT THE SAME ORDERING. The probe orders +// `created_at_ms ASC, rowid ASC`; `getMessages` orders by +// `MessageQueryService`'s own key. On ties the two disagree, +// and a transcript whose order flips is a transcript the user +// reads wrong. +// d. EXPORT STILL OWNS THE LEGACY CANDIDATES. B2 left +// `GET /api/sessions/:id/export` on the legacy-only probe set +// on purpose — its `mcode_unavailable` shape is byte-pinned by +// existing tests against exactly those three candidates, and +// widening export's set would change its enrichment from +// "unavailable" to "answering", which is a product change, not +// a migration step. +// +// What this batch DID collect is the coupling that made the probe +// look unremovable: `routes/sessions.js` no longer names +// `lib/transcript.js` at all, the read has one seam +// (`readEngineSwitchTranscript`), and the 3-candidate list plus the +// v2 data_json probe are now an implementation detail of the engine +// layer rather than something two routes import directly. The +// remaining work is a SEAM SWAP, not a redesign, and it belongs to +// M4-1 — the batch that registers an ACP provider and therefore +// makes an engine surface reachable under the default transport. +// It should land together with an equivalence test against a live v2 +// host, and with export's probe set widened in the same commit so +// the two endpoints cannot drift apart again. +// +// 2. THE FIRST-TOUCH OVERLAY IS STILL A WEBUI-SIDE WRITE. A bare +// `mvs_…` switch creates a record in `sessions.json` that the +// engine knows nothing about, and the engine's own session list and +// webui's wrapper list are two different questions that happen to +// agree. This is pre-existing behaviour (the alternative — +// registering the session engine-side — is a product decision about +// who owns session identity), and this batch did not change it. +// +// 3. THE USAGE SYNC IS NOT GATED. `applyMavisUsageToCs` reads webui's +// own mavis tables, so it declares no capability, and its failure +// is still swallowed with a debug-only warning. That asymmetry — +// identity and transcript are degraded, usage is dropped silently — +// predates this batch. Naming `usageStats` here would gate a working +// endpoint on a capability whose absence changes nothing visible; +// the real question is whether a silent drop is the right product +// behaviour at all, and that is not this batch's to decide. diff --git a/packages/webui/server/routes/sessions.js b/packages/webui/server/routes/sessions.js index f5441611..3da48c3a 100644 --- a/packages/webui/server/routes/sessions.js +++ b/packages/webui/server/routes/sessions.js @@ -4,42 +4,34 @@ // GET /api/acp-sessions, GET /api/acp-session-title, // GET /api/sessions/search (Lease C05 — cross-workspace fuzzy match) // (v0.5.bx-33: 删 POST /api/sessions/cleanup-orphans — Wzdhehe 不要这个 UI,API 一起删) +// +// What is left in this file after M3 is the HTTP surface of the session +// endpoints: parse the request, pick the status code, run the fail-closed +// audit, push the SSE frame, answer. Every endpoint that crosses the +// engine seam now asks `engine/` instead of this file's own imports — +// #9 #10 #72 #74 #75 (B1), #8 #11 (B2), #15 #16 #17 #19 (B3), +// #20 #57 #73 (B4), #7 #4 #6 (B5), #3 (B6) — and the imports that +// remain below are the ones that are genuinely webui-local: the session +// store, the workspace gate and the audit sink. import { randomUUID } from "node:crypto"; import { loadSessions, saveSessions, resetContext, - // Still a direct import: `handleSwitchSession` creates the first-touch - // overlay itself. Rename used to call it too and no longer does — that - // write moved to `engine/session-writes.js` — but the switch path is a - // read-with-a-side-effect and stayed put, so this symbol has not - // finished migrating. - ensureOverlayForMcodeSid, - findOverlayForMcodeSid, } from "../lib/sessions.js"; -import { - getMcodeSessionTitle, - getMcodeSessionsCacheSync, - getMcodeSessionsStaleSync, -} from "../lib/acp-client.js"; -// Switch-path transcript backfill — load mcode session history from -// the runtime DB so switching to an mvs_ session with no webui wrapper -// shows real chat instead of "No messages yet". -import { loadTranscriptChatLines } from "../lib/transcript.js"; -import { applyMavisUsageToCs } from "../lib/mavis-usage.js"; -import { getMcodeModelLimit } from "../lib/models.js"; -import { - pushStateFor, - clients, - runChatViewChat, -} from "../lib/state-bus.js"; -import { MCODE_RUNTIME_DB, DEFAULT_WORKSPACE } from "../lib/config.js"; +// `pushStateFor` stays a direct import: it is a pure SSE write with no +// I/O and no engine surface, and three of this module's handlers call it +// on their way out. `clients` and `runChatViewChat` left this file in +// M3-B5 and M3-B6 respectively — the delete fan-out enumerates clients +// inside the facade, and the run-mirror projection belongs with the +// switch that produces it. +import { pushStateFor } from "../lib/state-bus.js"; // M3-B1 (engine facade): #9 and #10 read the engine through the declared // capability rather than straight off the ACP client. Both facade -// functions forward to the same acp-client exports this module already -// imported, so the wire shape, the cache and the transport switch are -// unchanged — only the gate in front of them is new. +// functions forward to the same acp-client exports this module used to +// import directly, so the wire shape, the cache and the transport switch +// are unchanged — only the gate in front of them is new. import { readEngineSessionListForWorkspace, readEngineSessionTitle, @@ -88,6 +80,18 @@ import { previewEngineSessionDelete, readOrphanSessionWriteIds, } from "../engine/session-writes.js"; +// M3-B6 (engine facade): #3 switch. This is the endpoint that emptied the +// most imports out of this file — the walked-session title cache +// (`lib/acp-client.js`), the transcript read (`lib/transcript.js`), the +// usage sync (`lib/mavis-usage.js` + `lib/models.js`), the switch +// workspace gate (`lib/workspace.js#assertWorkspacePath`, still imported +// for handleNewSession) and `DEFAULT_WORKSPACE` / `MCODE_RUNTIME_DB` +// (`lib/config.js`, now referenced by no route in this file at all) +// all live behind `applyEngineSessionSwitch` now. See that module's +// header for the four load-bearing facts it took over, and KNOWN DEBT 1 +// for why the 3-candidate transcript probe it forwards to survives this +// batch while the route's direct reach for it does not. +import { applyEngineSessionSwitch } from "../engine/session-switch.js"; // The capability-error predicate `handleSessionTree` uses to tell the gate's // 501 apart from a soft-fail. Taken from the facade entry, which re-exports // the same binding `app.js#invokeHandler` matches on, so the two ends of this @@ -105,102 +109,6 @@ import { append as _eventsAppend } from "../lib/events.js"; // workspace write lands on the same boundary. import { assertWorkspacePath } from "../lib/workspace.js"; -// _resolveSwitchWorkspace — pick the workspace the switched-into session -// "belongs to" and run it through the same containment gate that the -// workspace picker / handleNewSession / browseWorkspace all funnel through. -// -// Source priority (s39 — webui-parity ticket 39: file tree must follow the -// switched session): -// -// 1. The target session's stored `workspace` field — that IS the -// workspace the user was in when they last had it open, modulo any -// pollution the old code introduced. Real existence + containment -// are checked; an out-of-bounds or stale value surfaces as a 400 -// so the user can either widen the allowed roots or pick a fresh -// workspace, instead of silently landing on the previous project. -// -// 2. DEFAULT_WORKSPACE (env MCODE_WORKSPACE > mcode TUI cwd.json > homedir) -// when the stored value is empty. Empty is also the value seen for -// (a) records created by the old code that polled freshly-typed mvs -// sessions with the current cs.workspace (the data-corruption bug -// this ticket fixes), and (b) older sessions that pre-date the -// workspace field. DEFAULT_WORKSPACE is already in the default -// allowed-roots surface (see getAllowedWorkspaceRoots), so the -// containment check accepts it without env setup. -// -// Critical invariants: -// - The switch NEVER keeps cs.workspace on the prior project. The -// user-reported symptom was exactly that: "the file tree still -// shows the previous project's files". Falling back to current ws -// when target.workspace is empty is the bug we are removing. -// - The switch NEVER writes cs.workspace.dir to a path the -// containment gate rejected. A 400 with the gate's actionable -// error is the only acceptable outcome. -// - The switch NEVER overwrites a target session's stored workspace -// with the current cs.workspace. That was the ② pollution path — -// re-introducing it would re-break the regression we just fixed. -// New overlay records (mvs_ first-touch) get workspace:"" here; the -// target-first read picks DEFAULT_WORKSPACE for them. -function _resolveSwitchWorkspace(target, currentWs) { - const raw = target && typeof target.workspace === "string" ? target.workspace.trim() : ""; - // Empty / non-string / null → DEFAULT_WORKSPACE. Never the current cs - // workspace — that's the user-reported "stays on the old project" - // failure mode this fix removes. - const candidate = raw || DEFAULT_WORKSPACE; - const gate = assertWorkspacePath(candidate); - if (!gate.ok) { - return { ok: false, error: gate.error, attempted: candidate }; - } - return { ok: true, dir: gate.path, real: gate.real, fallback: !raw }; -} - -/** - * Detect the cumulative-render pollution pattern in a stored chat - * buffer (session-isolation/06). When the engine emits each segment - * of an `agent_message`, streamUpdateLine writes a new `●` line; a - * non-cumulative buffer has each line containing only its own - * segment's text. A cumulative buffer — the bug — has at least one - * later `●` line whose text is a strict superset of an earlier - * `●` line (because the accumulator never reset between segments and - * every later line re-wrote every prior segment's text). This - * predicate is O(n^2) in the number of `●` lines but a single - * session's `chat` is bounded (~400 lines by the transcript cap) so - * the worst case is a few thousand substring checks per switch — - * cheap enough. - * - * Returns true when the buffer is clearly cumulative (an earlier - * `●` line is a strict substring of a later one AND the longer line - * strictly extends the shorter). Conservative on both sides: - * - a single-`●`-line buffer is never cumulative; - * - non-`●` lines (system, tool, ▲ thought) are ignored — only - * `●` rows matter, since the cumulative bug only affects message - * segments; - * - ties (equal-length `●` lines) are NOT cumulative — same - * length, no superset relation. - */ -function chatLooksCumulative(chat) { - if (!Array.isArray(chat) || chat.length === 0) return false; - const dots = []; - for (const line of chat) { - if (typeof line !== "string") continue; - // Match the same prefix the streamer writes: `● ` then text. - // Also accept bare `●` at end-of-line (transcript-sync appends - // stripped-down `●` markers in some paths). - if (line.startsWith("● ")) dots.push(line.slice(2)); - else if (line === "●") continue; - else continue; - } - for (let i = 0; i < dots.length; i += 1) { - for (let j = i + 1; j < dots.length; j += 1) { - const a = dots[i]; - const b = dots[j]; - if (b.length <= a.length) continue; // strict superset ⇒ longer - if (b.includes(a)) return true; - } - } - return false; -} - // _auditFail — shared failure sink for audit writes. events.js#append // THROWS on write failure; a governance action must not complete with // a missing audit trail, so every route-level append is wrapped and @@ -238,44 +146,6 @@ function _auditFail(res, e, what) { // instead of being a two-line helper a route could call in the wrong // order. -// Title fast path — resolve an mvs_ session's title from the -// in-memory walked-session cache (the same cache behind -// GET /api/acp-sessions via getMcodeSessionsForWorkspace) BEFORE -// awaiting getMcodeSessionTitle. The fallback boots the ACP child; with -// a missing/broken mcode binary that path measured ~2.17s end-to-end -// AND degraded the title to the "Mcode session" placeholder even -// though the cache already held the real title. Cache getters are sync -// and spawn nothing, so a hit keeps the switch hot path at zero ACP -// cost. -// -// Cross-workspace matching within what the module exposes: the cache -// holds ONE workspace's list, keyed by ws. We probe the client's -// current ws with both the fresh (30s TTL) and stale (same-ws, -// TTL-expired) readers, plus the "" key — getMcodeSessionsForWorkspace("") -// caches the UNFILTERED list, so a cache walked without a workspace -// still answers. A miss returns null and the caller falls back to -// getMcodeSessionTitle. -function _lookupCachedMcodeTitle(mcodeSessionId, ws) { - if (!mcodeSessionId) return null; - const keys = [ws || "", ""]; - for (const wsKey of keys) { - for (const getter of [getMcodeSessionsCacheSync, getMcodeSessionsStaleSync]) { - let sessions = null; - try { - sessions = getter(wsKey); - } catch { - sessions = null; - } - if (!Array.isArray(sessions)) continue; - const hit = sessions.find( - (s) => s && s.sessionId === mcodeSessionId && s.title, - ); - if (hit && hit.title) return hit.title; - } - } - return null; -} - // GET /api/sessions — list // qa (session-workspace-crud): 响应瘦身为 sidebar 元数据 — 与 docs/API.md // 声明的形状(id/title/workspace/mcodeSessionId/updatedAt)对齐。之前把 @@ -367,8 +237,37 @@ export async function handleNewSession(req, res, ctx) { } // POST /api/sessions/switch — switch to session by webui id or mvs_xxx +// +// M3-B6 (engine facade): everything this endpoint does to the engine — +// resolve, first-touch overlay creation, the cache-first title lookup, +// the transcript backfill decision and its read, the workspace +// containment gate, the per-client state mutation and the response body +// — happens in `engine/session-switch.js#applyEngineSessionSwitch`, and +// the response shape is built there once. What stays HERE is what is +// genuinely the route's, and the split is the same one B5 drew for the +// write family: +// +// - HTTP request parsing and the ONE validation body this endpoint +// has. A missing id is a 400 with `{ok:false,error:"id required"}` +// and a bare "application/json" content type, and that body has +// nothing to do with the engine. +// - THE STATUS CODES. The facade returns outcomes (`ok`, +// `not_found`, `workspace_refused`) and never learns what a status +// is; `statusHint` carries the number so the mapping is one table +// here instead of three branches inside the engine layer. +// - THE AUDIT, fail-closed. `_eventsAppend` THROWS on write failure +// and a governance action must not complete with a missing audit +// trail, so the append sits between the facade's work and the +// response, and its failure answers 500 through `_auditFail`. +// - The state push and the two log lines that bracket the response. +// +// The ordering constraint the facade could not own is the reason the +// audit stays put: the switch has ALREADY mutated `cs` by the time this +// append runs (that is pre-existing behaviour — a failed audit leaves +// the client switched and reports 500, which is what the operator sees +// today), and the SSE push must not fire when that append failed. Both +// properties are the route's to keep. export async function handleSwitchSession(req, res, ctx) { - const cs = ctx.cs; const cid = ctx.cid; const payload = await readJson(req); const id = (payload.id || "").trim(); @@ -376,290 +275,31 @@ export async function handleSwitchSession(req, res, ctx) { res.writeHead(400, { "Content-Type": "application/json" }); return res.end(JSON.stringify({ ok: false, error: "id required" })); } - const all = loadSessions(); - console.log( - `[switch] cid=${cid} incoming id=${id.substring(0, 12)}… isMcodeSid=${/^mvs_[a-f0-9]{32}$/.test(id)} allTotal=${all.length}`, - ); - // 优先按 mcode session id 找(v0.5.bv: 1:1 关联) - let target = all.find((s) => s.mcodeSessionId === id); - let matchKind = target ? "mcodeSessionId" : null; - if (!target) { - target = all.find((s) => s.id === id); - if (target) matchKind = "webuiId"; - } - console.log( - `[switch] cid=${cid} match=${matchKind || "NONE"} target.id=${target ? target.id.substring(0, 8) : "null"}… target.mcodeSid=${target && target.mcodeSessionId ? target.mcodeSessionId.substring(0, 12) : "null"}… target.chatLen=${target ? (target.chat ? target.chat.length : 0) : 0} target.title="${target ? (target.title || "").substring(0, 30) : ""}"`, - ); - if (!target) { - const isMcodeSid = /^mvs_[a-f0-9]{32}$/.test(id); - if (isMcodeSid) { - // Cache-first title — the walked session cache usually already - // holds the real title (the sidebar just rendered it). Only a - // total cache miss pays the getMcodeSessionTitle cost, which - // boots the ACP child (~2.17s measured with a broken mcode - // binary) and used to degrade every first switch to the - // "Mcode session" placeholder. - const ws = (cs.workspace && cs.workspace.dir) || ""; - let title = _lookupCachedMcodeTitle(id, ws); - let titleSource = title ? "cache" : "acp"; - if (!title) { - title = (await getMcodeSessionTitle(id)) || "Mcode session"; - } - // Single base session — overlay record id === mcode session id, - // idempotent create. Old model gave each mvs_ switch a fresh - // uuid wrapper → the same conversation had two identities, the - // direct cause of the "extra untitled entry" sidebar confusion. - // Repeated switches now hit the same record. - // - // s39 (webui-parity ticket 39): the workspace argument is GONE. - // The old `workspace: ws` here stamped the freshly-created overlay - // with the CURRENT cs.workspace, so every first-touch of an mvs_ - // session from project A inherited project A's path. Switching - // back to that mvs_ session from project B then either (a) was - // ignored by the read-only switch path, leaving the file tree - // stuck on B, or (b) — under the prior mutation — overwrote the - // overlay's workspace with B's path, polluting every per-project - // grouping. New overlays start with workspace:"" (set inside - // ensureOverlayForMcodeSid when no value is passed); the - // target-first read below then lands on DEFAULT_WORKSPACE for - // first-touch mvs_ switches, with no per-session pollution. - const existed = findOverlayForMcodeSid(all, id); - target = ensureOverlayForMcodeSid(all, id, { title }); - target.updatedAt = Date.now(); - saveSessions(all); - console.log( - `[switch] cid=${cid} ${existed ? "reused" : "created"} overlay ${target.id.substring(0, 12)}… (id=mcode sid) title="${title}" titleSource=${titleSource}`, - ); - } else { - console.log( - `[switch] cid=${cid} 404 id=${id} not found and not mcode sid`, - ); - res.writeHead(404, { "Content-Type": "application/json" }); - return res.end(JSON.stringify({ ok: false, error: "session not found" })); - } - } else if ( - // Placeholder refresh — wrappers created during a broken-title - // window carry "Mcode session" forever. If the walked cache now - // has the real title, repair the stored wrapper. Cache-only (sync, - // no ACP boot): an existing wrapper must never make the hot path - // slower. - target.title === "Mcode session" && - target.mcodeSessionId && - /^mvs_[a-f0-9]{32}$/.test(target.mcodeSessionId) - ) { - const cachedTitle = _lookupCachedMcodeTitle( - target.mcodeSessionId, - (cs.workspace && cs.workspace.dir) || "", - ); - if (cachedTitle) { - target.title = cachedTitle; - target.updatedAt = Date.now(); - saveSessions(all); - console.log( - `[switch] cid=${cid} refreshed placeholder title for ${target.id.substring(0, 8)}… → "${cachedTitle}"`, - ); - } - } - // Transcript backfill — when the resolved target has NO webui chat - // yet but IS a real mvs_ session, load the mcode transcript from - // the runtime DB (read-only) and map it into the webui chat-line - // grammar BEFORE responding, so response session.chat and cs.chat - // carry history. Caps inside (last 400 lines / 200KB) keep the SSE - // state push bounded; a 1000+-message session must not balloon it. - // - // session-isolation/06 (persist hygiene): the original rule only - // backfilled when target.chat was empty, so a polluted buffer - // (the cumulative-render bug from Item 1, before its fix) would - // persist via saveSessions and win forever. The new rule is: - // - if stored chat is empty → backfill (unchanged). - // - if stored chat looks cumulative → prefer DB read and re-persist. - // "cumulative" = at least two `●` lines whose text is a strict - // superset of an earlier `●` line (the engine emits each - // segment's full text per line, so a non-cumulative buffer has - // no such inclusion pair). - // - otherwise → keep stored chat. DB-authoritative: transcript-sync - // overwrites the stored chat from the engine DB on the next tick - // (~4s later), so any stored-only lines a user typed into the - // composer but never sent will be lost. The rule above does not - // promise draft preservation; it promises to NOT clobber a - // clean stored buffer with the DB read on every switch. Draft - // preservation is a separate concern (the composer keeps its - // own draft in its own state, see composer-draft.test.ts). - // FAILURE MUST NOT BREAK SWITCHING: any error logs and continues - // with the original chat — the switch itself always succeeds. - if ( - target.mcodeSessionId && - /^mvs_[a-f0-9]{32}$/.test(target.mcodeSessionId) - ) { - const storedHasChat = Array.isArray(target.chat) && target.chat.length > 0; - const storedCumulative = storedHasChat && chatLooksCumulative(target.chat); - const shouldBackfill = - !storedHasChat || storedCumulative; - if (shouldBackfill) { - try { - const r = loadTranscriptChatLines(target.mcodeSessionId, { - dbPath: MCODE_RUNTIME_DB, - }); - if (r.ok && r.lines.length > 0) { - const dbEmpty = target.chat.length === 0; - const dbShrinks = r.lines.length < target.chat.length; - const reason = dbEmpty - ? "empty" - : storedCumulative - ? "stored_cumulative" - : "stored_shrinks"; - target.chat = r.lines; - target.updatedAt = Date.now(); - saveSessions(all); // persist the populated wrapper (updatedAt bumped) - console.log( - `[switch] cid=${cid} transcript backfill ${target.id.substring(0, 8)}… mcode=${target.mcodeSessionId.substring(0, 12)}… reason=${reason} lines=${r.lines.length} msgs=${r.messageCount} probe=${r.probe}${r.truncated ? " (capped)" : ""}`, - ); - } else if (!r.ok) { - console.log( - `[switch] cid=${cid} transcript unavailable for ${target.mcodeSessionId.substring(0, 12)}… reason=${r.reason || "unknown"}`, - ); - } else if (storedCumulative) { - // Cumulative buffer + DB read came back empty — preserve - // the stored chat (which is at least the user's last view) - // and log the discrepancy so a post-mortem can see what - // happened. - console.log( - `[switch] cid=${cid} stored chat looked cumulative but DB read returned no lines; preserving stored chat for ${target.mcodeSessionId.substring(0, 12)}…`, - ); - } - } catch (e) { - console.warn( - `[switch] cid=${cid} transcript backfill failed for ${target.mcodeSessionId.substring(0, 12)}… (continuing with stored chat):`, - e && e.message ? e.message : e, - ); - } - } - } - const prevSid = cs.sessionId; - // s39 (webui-parity ticket 39): resolve the target session's workspace - // and re-point cs.workspace.dir to it BEFORE any other cs mutation, - // so the SSE state push (pushStateFor at the end) and the response - // session payload both carry the new workspace in lockstep with the - // session-id switch. The pre-fix behaviour read cs.workspace without - // writing it, which left the file tree bound to the previous project; - // this is the user-reported defect the ticket fixes. - // - // Containment gate is mandatory (s39 boundary): session-stored - // workspace is historical input — it may point to a directory the - // user removed from the allowed roots since the session was last - // opened, or to a path that was legal at the time but no longer is. - // assertWorkspacePath runs the same boundary the workspace picker, - // browseWorkspace, and the new-session POST funnel through; refusing - // here keeps that boundary singular. - const currentWs = (cs && cs.workspace && cs.workspace.dir) || ""; - const switchWs = _resolveSwitchWorkspace(target, currentWs); - if (!switchWs.ok) { - console.log( - `[switch] cid=${cid} REFUSED id=${id.substring(0, 12)}… reason=workspace_containment attempted="${switchWs.attempted}"`, - ); - res.writeHead(400, { "Content-Type": "application/json; charset=utf-8" }); - return res.end(JSON.stringify({ - ok: false, - error: switchWs.error, - attempted: switchWs.attempted, - })); - } - // cs.sessionId / mcodeSessionId / title / chat come first; the - // workspace write is paired with the session-id swap. Last-used-ws - // is intentionally untouched (a switch is browsing, not a workspace - // change — see the comment on handleWorkspaceChange for the same - // reasoning that protects lastUsedWorkspace from the switch path). - cs.sessionId = target.id; - cs.mcodeSessionId = target.mcodeSessionId || null; - cs.sessionTitle = target.title || "Untitled"; - cs.chat = Array.isArray(target.chat) ? target.chat : []; - cs.usage = { - ...cs.usage, - sessionInput: 0, - sessionOutput: 0, - sessionTotal: 0, - }; - cs.workspace = { - dir: switchWs.dir, - branch: null, - tree: null, - }; - if (switchWs.fallback) { - console.log( - `[switch] cid=${cid} target ${target.id.substring(0, 8)}… had no workspace — fell back to DEFAULT_WORKSPACE=${switchWs.dir}`, - ); + const r = await applyEngineSessionSwitch({ id, cs: ctx.cs, cid }); + if (r.outcome !== "ok") { + const contentType = + r.outcome === "workspace_refused" + ? "application/json; charset=utf-8" + : "application/json"; + res.writeHead(r.statusHint, { "Content-Type": contentType }); + return res.end(JSON.stringify(r.payload)); } - // Switching session must NOT mutate cs.lastUsedWorkspace — last-used - // is written only by handleSend (workspace change / send prompt); - // switching is browsing; pinning the browsed workspace to the top of - // the sidebar was the user-reported "click any session in C and C - // auto-sorts first" behavior. - resetContext(cs); - // Sync real token usage from mavis db on switch to a historical session - if (cs.mcodeSessionId) { - const switchedSid = cs.mcodeSessionId; - applyMavisUsageToCs(cs, switchedSid, { getMcodeModelLimit }) - .then(() => pushStateFor(cid)) - .catch((e) => { - if (process.env.MCODE_USAGE_DEBUG) - console.warn(`[switch.mavis] cid=${cid} error: ${e.message}`); - }); - } - // B01: session switch — record which session was activated and from - // which prior session. matchKind tells us whether we matched by - // mcodeSessionId or webuiId (useful when debugging "why did this - // resolve to session X"). prevSid is the prior session id (or "" if - // this was the first switch). Fail-closed → 5xx + alert. try { - _eventsAppend("session.switch", { - target: cs.sessionId, - cid, - actor: "user", - payload: { - from: prevSid || "", - matchKind: matchKind || "new_from_mcode", - mcodeSessionId: cs.mcodeSessionId || "", - title: cs.sessionTitle, - // s39 (webui-parity ticket 39): record which workspace the - // switch landed on, plus whether it was a fallback to - // DEFAULT_WORKSPACE. Both pieces are useful when auditing - // "why did the file tree change" or "why is the sidebar - // sorting by a directory I never opened". - workspace: switchWs.dir, - workspaceFallback: !!switchWs.fallback, - }, + _eventsAppend(r.audit.event, { + target: r.audit.target, + cid: r.audit.cid, + actor: r.audit.actor, + payload: r.audit.payload, }); } catch (e) { - return _auditFail(res, e, "session.switch"); + return _auditFail(res, e, r.audit.event); } pushStateFor(cid); console.log( - `[switch] cid=${cid} OK prev.sessionId=${prevSid ? prevSid.substring(0, 8) : "null"}… → new.sessionId=${cs.sessionId.substring(0, 8)}… title="${cs.sessionTitle}" chatLen=${cs.chat.length} workspace=${switchWs.dir}${switchWs.fallback ? " (DEFAULT_WORKSPACE fallback)" : ""}`, + `[switch] cid=${cid} OK prev.sessionId=${(r.audit.payload.from || "").substring(0, 8)}… → new.sessionId=${r.payload.session.id.substring(0, 8)}… title="${r.payload.session.title}" chatLen=${r.payload.session.chat.length} workspace=${r.payload.session.workspace}${r.payload.session.workspaceFallback ? " (DEFAULT_WORKSPACE fallback)" : ""}`, ); res.writeHead(200, { "Content-Type": "application/json" }); - return res.end( - JSON.stringify({ - ok: true, - session: { - id: target.id, - mcodeSessionId: cs.mcodeSessionId, - title: cs.sessionTitle, - // s39 (webui-parity ticket 39): surface the new workspace in - // the response so the client (url-restore + session-tree) can - // update its in-memory state without waiting for the SSE - // state-bus push to land — important for the file-tree panel - // that re-roots under the new workspaceDir on first render. - workspace: switchWs.dir, - workspaceFallback: !!switchWs.fallback, - // session-isolation/02 (run-mirror): switching back to the - // session that is mid-run must show what it produced so far. - // cs.chat holds the record's lines; the live turn's output is - // still in the runChat buffer — re-attach it for the owning - // view (same contract as every state snapshot). - chat: runChatViewChat(cid, cs), - }, - }), - ); + return res.end(JSON.stringify(r.payload)); } // POST /api/sessions/rename — rename a session (CRUD "update"). diff --git a/packages/webui/test/lib/engine/session-switch.test.js b/packages/webui/test/lib/engine/session-switch.test.js new file mode 100644 index 00000000..ebfb4bae --- /dev/null +++ b/packages/webui/test/lib/engine/session-switch.test.js @@ -0,0 +1,1392 @@ +// webui/test/lib/engine/session-switch.test.js +// +// M3-B6: the session SWITCH family's engine facade — #3 +// POST /api/sessions/switch. +// +// Sections are ordered by how much user-visible damage a regression in +// each one does, not by which module the function came from: +// +// 1. THE DECLARATION AND ITS SOFT-GATE POLICY. The most consequential +// judgement call in this batch: #3 gates SOFT because the switch's +// primary data is webui's own session record and both of its engine +// touches have a defined degradation. A hard gate would delete a +// working endpoint over an enrichment. Section 1 proves the gate +// reports and never throws — including on the DEFAULT `acp` +// transport, where no provider is registered at all. +// 2. THE FOUR RED LINES. 转录回填 (backfill), cumulative detection, +// workspace containment, single base-session identity. One named +// test per line, plus the negative half of each, because a red line +// that is only asserted in its happy direction is a red line nobody +// is watching. +// 3. THE BYTE-FOR-BYTE WIRE SHAPES, table-driven across all four +// outcomes: status, Content-Type, the exact body string and the key +// ORDER of the success payload. +// 4. THE PURE DERIVATIONS, on their inputs. +// 5. THE ROUTE, with the proof that the facade mock actually took. +// 6. THE TRANSCRIPT SEAM, and what this batch did and did not retire +// about the 3-candidate probe (KNOWN DEBT 1 in the module header). +// +// Two module-mock traps apply here exactly as they did in B3/B4/B5, and +// both are load-bearing rather than incidental: +// +// 1. `t.mock.module` REPLACES the WHOLE NAMESPACE; it does not merge. +// A mock naming only the export under test leaves every other name +// undefined and the consumer fails at INSTANTIATION with +// `SyntaxError: … does not provide an export named …` — a failure +// that reads like a product bug and is not one. Every facade mock +// below goes through `mockAll()`, which fills the un-stubbed names +// with a function that THROWS, so an unexpected call is loud +// instead of returning a plausible payload. +// 2. `mock.module` re-evaluates only the MOCKED specifier. A consumer +// already in the registry keeps its old LIVE BINDING, so a second +// test in the same file would silently reuse the first test's mock +// and pass for the wrong reason. Every route re-import in section 5 +// carries a fresh `?bust=N`, and section 5 ends with marker controls +// that prove it. + +import { test, describe, before, after, beforeEach } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, mkdirSync, readFileSync, realpathSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { Readable } from "node:stream"; + +import { + setupMocks, + absPath, + registerSessionsStore, + getSessionsStore, + registerAcpMock, +} from "../../helpers/_setup.js"; +import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; +// Type discrimination goes through the exported predicate, never +// `err.name`. `engine/capabilities.js` is never `mock.module`d by this +// file, so the `instanceof` inside it resolves against the same class +// `checkSessionSwitchCapability` would have thrown from had it thrown at +// all. The string comparison it replaces could not tell a capability +// error from any other error that happened to carry a name. +const { isEngineCapabilityNotSupportedError } = await import( + "../../../server/engine/errors.js" +); + +const RUNTIME = "runtime"; +const ACP = "acp"; + +/** A syntactically valid engine sid — 32 lowercase hex digits. */ +const SID_A = "mvs_aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; +const SID_B = "mvs_bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"; +/** Not an engine sid: too short. Must take the 404 branch. */ +const NOT_A_SID = "webui-does-not-exist"; + +let bust = 0; + +/** A JSON request body the real `lib/read-json.js` can consume. */ +function jsonReq(body) { + return Readable.from([Buffer.from(JSON.stringify(body), "utf8")]); +} + +/** A minimal `ServerResponse` stand-in that records what was written. */ +function mkRes() { + const written = []; + return { + written, + writeHead(status, headers) { + written.push({ status, headers }); + return this; + }, + end(body) { + written.push({ body }); + return this; + }, + }; +} + +/** The v2 probe SQL, read from the ONE declaration (never re-typed). */ +let _v2Sql; +async function v2ProbeSql() { + if (_v2Sql) return _v2Sql; + const mod = await import(absPath("lib/transcript.js")); + _v2Sql = mod.V2_DATA_JSON_PROBES[0].sql; + return _v2Sql; +} + +/** + * Fake better-sqlite3 keyed by SQL string. `prepare()` throws for any SQL + * the fixture does not carry, exactly as a real prepare does on a missing + * column — which is what makes the legacy 3-candidate probes "miss" the + * way they miss against the live v2 schema. + */ +function makeFakeDb({ rowsBySql = {}, constructThrows = false } = {}) { + return class FakeDb { + constructor(path, opts) { + if (constructThrows) throw new Error("fake better-sqlite3: boom"); + this.path = path; + this.opts = opts; + } + prepare(sql) { + const bySid = rowsBySql[sql]; + if (!bySid) throw new Error(`fake db: no such column (${sql.slice(0, 52)}…)`); + return { all: (sid) => (bySid[sid] || []).slice() }; + } + close() {} + }; +} + +// Workspace fixtures. `assertWorkspacePath` is NOT mocked anywhere in +// this file — the containment red line is exactly the real gate's +// behaviour, so the fixtures are real directories under a real +// allowed-roots tree, and every path is realpath'd once at setup so the +// assertions compare against the same form the gate normalises to (Linux +// /tmp vs macOS /private/tmp — see fs-write.test.js, #81). +let WS_ROOT, WS_A, WS_B, WS_DEFAULT, WS_OUTSIDE, DEFAULT_DIR, DB_PATH; +let _eventsDir; +// Mutable sqlite fixture, read at call time by the resolver mock +// registered in `before()`. See `bootFacade` for why it cannot be a +// per-test registration. +let _dbOpts = {}; + +before((t) => { + _eventsDir = mkTmpDir("webui-switch-facade-events-"); + WS_ROOT = mkTmpDir("webui-switch-facade-roots-"); + DB_PATH = mkTmpDir("webui-switch-facade-db-"); + // Pinned BEFORE any SUT import: `lib/config.js` freezes + // MCODE_RUNTIME_DB, DEFAULT_WORKSPACE and the audit path at module load. + process.env.MCODE_WEBUI_EVENTS_PATH = join(_eventsDir, "events.ndjson"); + process.env.MCODE_RUNTIME_DB = join(DB_PATH, "runtime-state.sqlite"); + writeFileSync(process.env.MCODE_RUNTIME_DB, ""); + for (const name of ["projectA", "projectB", "default-workspace"]) { + mkdirSync(join(WS_ROOT, name), { recursive: true }); + } + WS_A = realpathSync(join(WS_ROOT, "projectA")); + WS_B = realpathSync(join(WS_ROOT, "projectB")); + WS_DEFAULT = realpathSync(join(WS_ROOT, "default-workspace")); + // A real directory that is deliberately OUTSIDE the allowed roots, so + // a record pointing at it is refused rather than silently accepted. + WS_OUTSIDE = realpathSync(mkTmpDir("webui-switch-facade-outside-")); + DEFAULT_DIR = WS_DEFAULT; + process.env.MCODE_WORKSPACE = DEFAULT_DIR; + process.env.MCODE_WEBUI_WORKSPACE_ROOTS = WS_ROOT; + // `lib/transcript.js` (the seam's reader) and `lib/session-tree.js` + // both import this module. Registered ONCE, before any SUT import, + // because both of them keep a live binding to it afterwards. + t.mock.module(absPath("lib/sqlite-resolver.js"), { + namedExports: { + getMcodeBetterSqlite3: () => makeFakeDb(_dbOpts), + _getBetterSqlite3Candidates: () => [], + }, + }); +}); + +after(() => { + delete process.env.MCODE_WEBUI_EVENTS_PATH; + delete process.env.MCODE_RUNTIME_DB; + delete process.env.MCODE_WEBUI_WORKSPACE_ROOTS; + delete process.env.MCODE_WORKSPACE; + for (const d of [_eventsDir, WS_ROOT, DB_PATH, WS_OUTSIDE]) { + if (d) rmTmpDir(d); + } +}); + +/** + * Boot the REAL facade over mocked storage. The sqlite fixture is what + * decides whether the transcript read answers, so every data-plane test + * that cares about the backfill passes `db` explicitly. + */ +async function bootFacade(t, { db = {}, store = [], acp = {}, mavis = {} } = {}) { + await setupMocks(t, { mavis: { applyMavisUsageToCs: async () => {}, ...mavis } }); + // The sqlite fixture is read at CALL time by the mock registered in + // `before()`. It cannot be re-registered per test: `lib/transcript.js` + // holds a live binding to `lib/sqlite-resolver.js` after its first + // import, and `mock.module` re-evaluates only the specifier it is given + // — so a second registration here would leave the reader on the FIRST + // test's fake and every later case would silently answer the wrong + // thing. This is mock trap #2, and it is why `before()` owns it. + _dbOpts = db; + registerSessionsStore({ initial: store }); + registerAcpMock({ + getMcodeSessionsCacheSync: () => null, + getMcodeSessionsStaleSync: () => null, + getMcodeSessionTitle: async () => null, + ...acp, + }); + // Imported AFTER the mocks: the facade reaches its storage through + // `await import()` at call time, so the registry mocks are what it + // gets — and importing here (not at file scope) keeps the real module + // the one under test in this section. + return import(absPath("engine/session-switch.js")); +} + +/** A minimal webui client state — only the fields the switch reads. */ +function mkCs(workspaceDir = WS_A) { + return { + sessionId: "webui-previous", + mcodeSessionId: null, + sessionTitle: "Previous", + chat: [], + usage: { sessionInput: 7, sessionOutput: 8, sessionTotal: 15, contextUsed: 3 }, + workspace: { dir: workspaceDir, branch: "main", tree: null }, + }; +} + +/** Three transcript rows that exercise user / thinking+tool / assistant. */ +function transcriptRows(sid) { + return { + [sid]: [ + { + role: "user", + turn_id: "turn-a", + msg_id: "msg-user-1", + data_json: JSON.stringify({ role: "user", msg_content: "调研工具" }), + }, + { + role: "assistant", + turn_id: "turn-a", + msg_id: "msg-assistant-1", + data_json: JSON.stringify({ + role: "assistant", + msg_content: "我先看看", + thinking_content: "先搜索", + tool_calls: [ + { + tool_name: "bash", + tool_call_id: "c1", + tool_call_status: 2, + tool_call_args: '{"command":"ls"}', + tool_call_result_data: '{"content":[{"type":"text","text":"file1"}]}', + }, + ], + }), + }, + { + role: "assistant", + turn_id: "turn-a", + msg_id: "msg-assistant-2", + data_json: JSON.stringify({ role: "assistant", msg_content: "结论" }), + }, + ], + }; +} + +const EXPECTED_LINES = [ + "› 调研工具", + "▲ 先搜索", + "● 我先看看", + '→ bash {"command":"ls"}', + " [completed]", + " file1", + "● 结论", + "§§ turn_msg=msg-assistant-2", +]; + +describe("M3-B6 — session switch family", () => { + // --------------------------------------------------------------------- + // 1. The declaration table and its soft-gate policy + // --------------------------------------------------------------------- + + describe("SESSION_SWITCH_ENDPOINTS — one endpoint, one soft declaration", () => { + test("covers exactly this batch's one endpoint", async () => { + const { SESSION_SWITCH_ENDPOINTS } = await import( + absPath("engine/session-switch.js") + ); + assert.deepEqual(Object.keys(SESSION_SWITCH_ENDPOINTS), [ + "POST /api/sessions/switch", + ]); + }); + + test("the row names the pair the ENRICHMENTS need, enforced softly", async () => { + const { SESSION_SWITCH_ENDPOINTS } = await import( + absPath("engine/session-switch.js") + ); + const { ENGINE_CAPABILITY_KEYS } = await import(absPath("engine/index.js")); + const row = SESSION_SWITCH_ENDPOINTS["POST /api/sessions/switch"]; + assert.deepEqual(Object.keys(row), [ + "capability", + "subItem", + "enforcement", + ]); + assert.deepEqual(row, { + capability: "sessionCrud", + subItem: "getSession", + enforcement: "soft", + }); + assert.ok( + ENGINE_CAPABILITY_KEYS.includes(row.capability), + "the declared capability must be a real registry key, not an invented one", + ); + }); + + test(`the DEFAULT transport (${ACP}) is UNREGISTERED and the gate says so`, async () => { + const { checkSessionSwitchCapability } = await import( + absPath("engine/session-switch.js") + ); + const gate = checkSessionSwitchCapability("POST /api/sessions/switch", ACP); + assert.equal(gate.gate, "unregistered-transport"); + assert.equal(gate.provider, null); + assert.equal(gate.enforcement, "soft"); + }); + + test(`the ${RUNTIME} transport resolves the v2 provider and checks the declaration`, async () => { + const { checkSessionSwitchCapability } = await import( + absPath("engine/session-switch.js") + ); + const gate = checkSessionSwitchCapability("POST /api/sessions/switch", RUNTIME); + assert.equal(gate.gate, "checked"); + assert.equal(gate.provider, "local-runtime-v2"); + }); + + test("NO transport ever produces a capability error — the family declares no throwing gate", async () => { + // A registry-driven assertion cannot cover the "provider declares + // sessionCrud: none" case, because no registered provider does and + // PROVIDERS is frozen. So the policy claim is pinned statically: + // this module must not import `assertEngineCapability` (the only + // thrower) and must not export an `assert*` gate. If a later + // editor adds either, this test is the thing that says no. + const src = readFileSync( + fileURLToPath(absPath("engine/session-switch.js")), + "utf8", + ); + assert.equal( + src.includes("assertEngineCapability("), + false, + "session-switch.js started calling the throwing gate — the 501 policy is a decision, not a refactor", + ); + const mod = await import(absPath("engine/session-switch.js")); + assert.deepEqual( + Object.keys(mod).filter((k) => /^assert/i.test(k)), + [], + "this family must expose no assert* gate; use checkSessionSwitchCapability", + ); + }); + + test("an unknown endpoint key is a plain Error, never a capability error", async () => { + const { checkSessionSwitchCapability } = await import( + absPath("engine/session-switch.js") + ); + let caught = null; + try { + checkSessionSwitchCapability("POST /api/sessions/nope", RUNTIME); + } catch (e) { + caught = e; + } + assert.ok(caught, "an unknown key must throw"); + assert.equal(caught.code, "unknown_session_switch_endpoint"); + assert.equal( + isEngineCapabilityNotSupportedError(caught), + false, + "caller confusion must never be dressed up as an engine limitation", + ); + }); + }); + + // --------------------------------------------------------------------- + // 2. The four red lines + // --------------------------------------------------------------------- + + describe("RED LINE 1 — 转录回填: a switch shows the conversation, it does not show an empty screen", () => { + test("empty stored chat → the engine transcript lands in cs.chat, the response and the persisted record", async (t) => { + const mod = await bootFacade(t, { + db: { rowsBySql: { [await v2ProbeSql()]: transcriptRows(SID_A) } }, + }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ id: SID_A, cs, cid: "cid-1" }); + assert.equal(r.outcome, "ok"); + assert.deepEqual(cs.chat, EXPECTED_LINES, "cs.chat carries the mapped transcript"); + assert.deepEqual( + r.payload.session.chat, + EXPECTED_LINES, + "the response carries the same lines the client state does", + ); + const saved = getSessionsStore()[0]; + assert.deepEqual(saved.chat, EXPECTED_LINES, "and the wrapper was re-persisted"); + assert.equal(r.transcript.ok, true); + assert.equal( + r.transcript.decision, + "empty", + "first touch fires the EMPTY branch of the backfill rule", + ); + assert.equal(r.transcript.reason, null, "and a successful read has no failure reason"); + }); + + test("NEGATIVE half: a CLEAN stored chat is kept even though the engine read would answer", async (t) => { + const mod = await bootFacade(t, { + db: { rowsBySql: { [await v2ProbeSql()]: transcriptRows(SID_A) } }, + store: [ + { + id: "webui-keep", + mcodeSessionId: SID_A, + title: "Keep", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● mine already"], + }, + ], + }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-keep", + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + assert.deepEqual(cs.chat, ["● mine already"], "a clean buffer is never clobbered"); + assert.equal(r.transcript, null, "and the read was not even attempted"); + }); + + test("a read that fails NEVER breaks the switch (missing db / bad driver / schema drift)", async (t) => { + // Three failure shapes, one promise: the switch answers 200 and + // keeps the stored chat. This is the endpoint's oldest contract + // and the reason the family's gate is soft. + for (const [name, db] of [ + ["constructor throws", { constructThrows: true }], + ["every prepare throws (schema drift)", {}], + ]) { + await t.test(name, async (t2) => { + const mod = await bootFacade(t2, { db }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ + id: SID_A, + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "ok", name); + assert.equal(r.payload.ok, true, name); + assert.deepEqual(r.payload.session.chat, [], name); + assert.equal(r.transcript.ok, false, name); + assert.ok(r.transcript.reason, name); + }); + } + }); + }); + + describe("RED LINE 2 — cumulative detection: a polluted buffer is repaired, a clean one is not", () => { + // Table-driven on the predicate, because the predicate is the whole + // red line and a change to it must be reviewed as a rule change. + const CUMULATIVE_TABLE = [ + ["empty buffer", [], false], + ["one dot line", ["● only one"], false], + [ + "non-cumulative segments", + ["● seg one", "● seg two", "● seg three"], + false, + ], + [ + "cumulative: a later line strictly contains an earlier one", + ["● part one", "● part one plus part two"], + true, + ], + [ + "cumulative anywhere in the buffer, not just the first pair", + ["● a", "● b", "● a and b and c"], + true, + ], + ["equal-length dots are NOT a superset", ["● ab", "● ba"], false], + [ + "non-dot lines are ignored entirely", + ["› prompt", "▲ thought", "→ tool {}", "○ system"], + false, + ], + [ + "a non-string entry does not throw the predicate", + ["● prefix", null, 42, "● prefix and more"], + true, + ], + ["a bare dot marker is not evidence", ["●", "● later"], false], + ["a shorter later line is not a superset", ["● long line", "● short"], false], + ]; + for (const [name, chat, expected] of CUMULATIVE_TABLE) { + test(`chatLooksCumulative: ${name} → ${expected}`, async () => { + const { chatLooksCumulative } = await import( + absPath("engine/session-switch.js") + ); + assert.equal(chatLooksCumulative(chat), expected); + }); + } + + test("selectTranscriptBackfill reads the predicate into the three-branch rule", async () => { + const { selectTranscriptBackfill } = await import( + absPath("engine/session-switch.js") + ); + assert.deepEqual(selectTranscriptBackfill([]), { + storedHasChat: false, + storedCumulative: false, + shouldBackfill: true, + reason: "empty", + }); + assert.deepEqual(selectTranscriptBackfill(["● a", "● a and b"]), { + storedHasChat: true, + storedCumulative: true, + shouldBackfill: true, + reason: "stored_cumulative", + }); + assert.deepEqual(selectTranscriptBackfill(["● a", "● b"]), { + storedHasChat: true, + storedCumulative: false, + shouldBackfill: false, + reason: "stored_shrinks", + }); + }); + + test("end to end: a cumulative stored buffer is replaced by the engine read and re-persisted", async (t) => { + const mod = await bootFacade(t, { + db: { rowsBySql: { [await v2ProbeSql()]: transcriptRows(SID_A) } }, + store: [ + { + id: "webui-polluted", + mcodeSessionId: SID_A, + title: "Polluted", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● seg one", "● seg one and seg two"], + }, + ], + }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-polluted", + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + assert.deepEqual(cs.chat, EXPECTED_LINES, "the polluted buffer is gone"); + assert.deepEqual(getSessionsStore()[0].chat, EXPECTED_LINES, "and stays gone"); + assert.equal(r.transcript.ok, true); + }); + + test("a cumulative buffer whose read comes back EMPTY is preserved, not blanked", async (t) => { + const mod = await bootFacade(t, { + db: {}, + store: [ + { + id: "webui-polluted-2", + mcodeSessionId: SID_A, + title: "Polluted", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● seg one", "● seg one and seg two"], + }, + ], + }); + const cs = mkCs(); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-polluted-2", + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + assert.deepEqual( + cs.chat, + ["● seg one", "● seg one and seg two"], + "an empty read must not delete the user's last view", + ); + assert.equal( + r.transcript.decision, + "stored_cumulative", + "and the log says the pollution branch fired, not the empty one", + ); + }); + }); + + describe("RED LINE 3 — workspace containment: the switch writes a gated path or it does not write one", () => { + test("an out-of-bounds stored workspace is REFUSED and the client state is untouched", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-outside", + mcodeSessionId: SID_A, + title: "Outside", + workspace: WS_OUTSIDE, + createdAt: 1, + updatedAt: 1, + chat: [], + }, + ], + }); + const cs = mkCs(WS_B); + const before = JSON.parse(JSON.stringify(cs)); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-outside", + cs, + cid: "cid-1", + }); + assert.equal(r.outcome, "workspace_refused"); + assert.equal(r.statusHint, 400); + assert.equal(r.audit, null, "a refused switch writes no audit event"); + assert.equal(r.payload.ok, false); + assert.equal(r.payload.attempted, WS_OUTSIDE, "the 400 names the path it refused"); + assert.ok(typeof r.payload.error === "string" && r.payload.error.length > 0); + assert.deepEqual( + cs, + before, + "a refused switch must leave identity, chat, usage and workspace exactly as they were", + ); + }); + + test("an empty stored workspace falls back to the DEFAULT, never to the caller's current one", async (t) => { + // The user-reported defect: "the file tree still shows the previous + // project". The caller is sitting in projectB; the record has no + // workspace of its own; the answer must be the default, not B. + const mod = await bootFacade(t, { store: [] }); + const cs = mkCs(WS_B); + const r = await mod.applyEngineSessionSwitch({ id: SID_A, cs, cid: "cid-1" }); + assert.equal(r.outcome, "ok"); + assert.equal(r.workspace.fallback, true); + assert.equal(cs.workspace.dir, WS_DEFAULT); + assert.notEqual(cs.workspace.dir, WS_B, "the current workspace must never be the fallback"); + assert.equal(r.payload.session.workspaceFallback, true); + }); + + test("a stored workspace wins over the default and the target record is never rewritten with the caller's", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-A", + mcodeSessionId: SID_A, + title: "Project A session", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● a"], + }, + ], + }); + const cs = mkCs(WS_B); + const r = await mod.applyEngineSessionSwitch({ id: "webui-A", cs, cid: "cid-1" }); + assert.equal(r.outcome, "ok"); + assert.equal(cs.workspace.dir, WS_A, "the file tree follows the switched session"); + assert.equal(r.workspace.fallback, false); + assert.equal(getSessionsStore()[0].workspace, WS_A, "the record keeps its own workspace"); + }); + + test("resolveSwitchWorkspace prefers target-first and reports the refusal shape", async () => { + const { resolveSwitchWorkspace } = await import( + absPath("engine/session-switch.js") + ); + const refuse = (p) => ({ ok: false, error: `outside: ${p}` }); + const accept = (p) => ({ ok: true, path: p, real: p }); + // Target-first. + assert.deepEqual( + resolveSwitchWorkspace({ workspace: " /ws/a " }, { + defaultWorkspace: "/ws/default", + assertPath: accept, + }), + { ok: true, dir: "/ws/a", real: "/ws/a", fallback: false }, + "the stored value is trimmed and used as-is", + ); + // Empty / missing / non-string → the default, flagged as a fallback. + for (const record of [{}, { workspace: "" }, { workspace: " " }, { workspace: 7 }]) { + const got = resolveSwitchWorkspace(record, { + defaultWorkspace: "/ws/default", + assertPath: accept, + }); + assert.equal(got.dir, "/ws/default", JSON.stringify(record)); + assert.equal(got.fallback, true, JSON.stringify(record)); + } + // Refusal carries the attempted path so the 400 can be actionable. + assert.deepEqual( + resolveSwitchWorkspace({ workspace: "/nope" }, { + defaultWorkspace: "/ws/default", + assertPath: refuse, + }), + { ok: false, error: "outside: /nope", attempted: "/nope" }, + ); + }); + }); + + describe("RED LINE 4 — single base session identity: one conversation, one record", () => { + test("first touch creates exactly ONE record whose id IS the engine sid", async (t) => { + const mod = await bootFacade(t, { store: [] }); + const r = await mod.applyEngineSessionSwitch({ + id: SID_A, + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + const store = getSessionsStore(); + assert.equal(store.length, 1, "one conversation must not produce two entries"); + assert.equal(store[0].id, SID_A, "the overlay record's id IS the engine sid"); + assert.equal(store[0].mcodeSessionId, SID_A); + assert.equal(r.payload.session.id, SID_A); + assert.equal(r.matchKind, null, "first touch is not a match against an existing record"); + }); + + test("a second switch to the same sid REUSES the record — no second entry appears", async (t) => { + const mod = await bootFacade(t, { store: [] }); + await mod.applyEngineSessionSwitch({ id: SID_A, cs: mkCs(), cid: "cid-1" }); + await mod.applyEngineSessionSwitch({ id: SID_A, cs: mkCs(WS_B), cid: "cid-1" }); + const store = getSessionsStore(); + assert.equal(store.length, 1, "repeated switches must hit the same record"); + assert.equal(store[0].id, SID_A); + }); + + test("resolveSwitchTarget prefers the engine sid over a webui uuid, the opposite of the write family", async () => { + // The single-identity rule, stated as a resolution order. Two + // discriminating cases, because the order is only OBSERVABLE when + // both passes could match — and a reader who writes one fixture + // will not notice that the other order passes it too. + const { resolveSwitchTarget } = await import(absPath("engine/session-switch.js")); + + // (a) The label case, and the common one: an overlay record's id + // IS its engine sid, so both passes match the same record and only + // `matchKind` tells the two orders apart. It is observable — the + // audit payload carries the label, and `new_from_mcode` vs + // `mcodeSessionId` is the difference between "we just created + // this" and "this already existed". + const overlay = { id: SID_A, mcodeSessionId: SID_A, title: "Overlay" }; + assert.deepEqual( + resolveSwitchTarget([overlay], SID_A), + { index: 0, matchKind: "mcodeSessionId", target: overlay }, + "an overlay addressed by its sid is an mcodeSessionId match, not a webuiId one", + ); + + // (b) The conflict case: two records could answer, and the one + // that IS the engine session wins. + const bySid = { id: "webui-1", mcodeSessionId: SID_A }; + const byUuid = { id: SID_B, mcodeSessionId: null }; + const records = [byUuid, bySid]; + assert.deepEqual(resolveSwitchTarget(records, SID_A), { + index: 1, + matchKind: "mcodeSessionId", + target: bySid, + }); + assert.deepEqual( + resolveSwitchTarget(records, SID_B), + { + index: 0, + // No record is BOUND to SID_B — the one whose UUID is SID_B + // has no mcodeSessionId at all — so the sid pass misses and the + // uuid pass wins. The order is only observable in the case + // where both passes could match. + matchKind: "webuiId", + target: byUuid, + }, + "an id that is a record's uuid but no record's engine sid is a webuiId match", + ); + assert.deepEqual(resolveSwitchTarget(records, "webui-1"), { + index: 1, + matchKind: "webuiId", + target: bySid, + }); + assert.deepEqual(resolveSwitchTarget(records, NOT_A_SID), { + index: -1, + matchKind: null, + target: null, + }); + }); + + test("an id that is neither a record nor an engine sid is `not_found`, never an invented overlay", async (t) => { + const mod = await bootFacade(t, { store: [] }); + const r = await mod.applyEngineSessionSwitch({ + id: NOT_A_SID, + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(r.outcome, "not_found"); + assert.equal(r.statusHint, 404); + assert.deepEqual(r.payload, { ok: false, error: "session not found" }); + assert.equal(getSessionsStore().length, 0, "a wrong id must not create a record"); + }); + }); + + // --------------------------------------------------------------------- + // 3. The byte-for-byte wire shapes + // --------------------------------------------------------------------- + + describe("the response shape is pinned byte-for-byte, in every outcome", () => { + test("success: exact body string and key order", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-A", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: ["● x"], + }, + ], + }); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-A", + cs: mkCs(), + cid: "cid-1", + }); + assert.equal( + JSON.stringify(r.payload), + `{"ok":true,"session":{"id":"webui-A","mcodeSessionId":"${SID_A}","title":"T",` + + `"workspace":"${WS_A}","workspaceFallback":false,"chat":["● x"]}}`, + "the success body's key ORDER is a frontend contract (url-restore reads workspace)", + ); + assert.deepEqual(Object.keys(r.payload), ["ok", "session"]); + assert.deepEqual(Object.keys(r.payload.session), [ + "id", + "mcodeSessionId", + "title", + "workspace", + "workspaceFallback", + "chat", + ]); + }); + + test("not_found / workspace_refused bodies, key order included", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-outside", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_OUTSIDE, + createdAt: 1, + updatedAt: 1, + chat: [], + }, + ], + }); + const notFound = await mod.applyEngineSessionSwitch({ + id: NOT_A_SID, + cs: mkCs(), + cid: "cid-1", + }); + assert.equal( + JSON.stringify(notFound.payload), + '{"ok":false,"error":"session not found"}', + ); + const refused = await mod.applyEngineSessionSwitch({ + id: "webui-outside", + cs: mkCs(), + cid: "cid-1", + }); + assert.deepEqual(Object.keys(refused.payload), ["ok", "error", "attempted"]); + assert.equal(refused.payload.attempted, WS_OUTSIDE); + }); + + test("a record with no mcodeSessionId and no chat still answers the same six-key body", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-local", + title: "Local only", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: [], + }, + ], + }); + const r = await mod.applyEngineSessionSwitch({ + id: "webui-local", + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(r.outcome, "ok"); + assert.deepEqual(Object.keys(r.payload.session), [ + "id", + "mcodeSessionId", + "title", + "workspace", + "workspaceFallback", + "chat", + ]); + assert.equal(r.payload.session.mcodeSessionId, null); + assert.deepEqual(r.payload.session.chat, []); + }); + + test("the audit payload is the B01 contract, and first touch keeps its own label", async (t) => { + const mod = await bootFacade(t, { store: [] }); + const r = await mod.applyEngineSessionSwitch({ + id: SID_A, + cs: mkCs(), + cid: "cid-9", + }); + assert.equal(r.audit.event, "session.switch"); + assert.equal(r.audit.target, SID_A); + assert.equal(r.audit.cid, "cid-9"); + assert.equal(r.audit.actor, "user"); + assert.deepEqual(Object.keys(r.audit.payload), [ + "from", + "matchKind", + "mcodeSessionId", + "title", + "workspace", + "workspaceFallback", + ]); + assert.equal(r.audit.payload.from, "webui-previous", "the prior session is recorded"); + assert.equal( + r.audit.payload.matchKind, + "new_from_mcode", + "a first touch is labelled new_from_mcode, NOT mcodeSessionId", + ); + assert.equal(r.audit.payload.workspace, WS_DEFAULT); + assert.equal(r.audit.payload.workspaceFallback, true); + }); + + test("an existing record reports its real matchKind in the audit", async (t) => { + const mod = await bootFacade(t, { + store: [ + { + id: "webui-A", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_A, + createdAt: 1, + updatedAt: 1, + chat: [], + }, + ], + }); + const bySid = await mod.applyEngineSessionSwitch({ + id: SID_A, + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(bySid.matchKind, "mcodeSessionId"); + assert.equal(bySid.audit.payload.matchKind, "mcodeSessionId"); + const byUuid = await mod.applyEngineSessionSwitch({ + id: "webui-A", + cs: mkCs(), + cid: "cid-1", + }); + assert.equal(byUuid.matchKind, "webuiId"); + assert.equal(byUuid.audit.payload.matchKind, "webuiId"); + }); + }); + + // --------------------------------------------------------------------- + // 4. The pure derivations + // --------------------------------------------------------------------- + + describe("the pure derivations, on their inputs", () => { + test("isSwitchableMcodeSessionId is the 32-hex rule and nothing looser", async () => { + const { isSwitchableMcodeSessionId } = await import( + absPath("engine/session-switch.js") + ); + for (const good of [SID_A, SID_B, `mvs_${"a".repeat(32)}`]) { + assert.equal(isSwitchableMcodeSessionId(good), true, good); + } + for (const bad of [ + "mvs_short", + `mvs_${"a".repeat(31)}`, + `mvs_${"a".repeat(33)}`, + `mvs_${"A".repeat(32)}`, + "webui-1", + "", + null, + undefined, + 42, + ]) { + assert.equal(isSwitchableMcodeSessionId(bad), false, String(bad)); + } + }); + + test("lookupCachedMcodeTitle probes the current ws, then the stale reader, then the unfiltered key", async () => { + const { lookupCachedMcodeTitle } = await import( + absPath("engine/session-switch.js") + ); + const fresh = (ws) => + ws === "/ws/a" ? [{ sessionId: SID_A, title: "from fresh" }] : null; + const stale = () => [{ sessionId: SID_A, title: "from stale" }]; + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/a", { fresh, stale }), + "from fresh", + "the fresh reader for the current workspace wins", + ); + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/other", { fresh, stale }), + "from stale", + "a miss falls through to the stale reader", + ); + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/none", { + fresh: () => null, + stale: () => null, + }), + null, + "a total miss is null so the caller can pay for the ACP path", + ); + assert.equal(lookupCachedMcodeTitle("", "/ws/a", { fresh, stale }), null); + // A throwing cache reader is a miss, not a crash: the switch must + // still be able to fall back to the engine title. + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/a", { + fresh: () => { + throw new Error("cache exploded"); + }, + stale: () => null, + }), + null, + ); + }); + + test("lookupCachedMcodeTitle finds a title cached under the unfiltered key", async () => { + const { lookupCachedMcodeTitle } = await import( + absPath("engine/session-switch.js") + ); + // getMcodeSessionsForWorkspace("") caches the UNFILTERED list, so a + // cache walked without a workspace still answers the first touch. + const unfiltered = [{ sessionId: SID_A, title: "Unfiltered title" }]; + assert.equal( + lookupCachedMcodeTitle(SID_A, "/ws/a", { + fresh: (ws) => (ws === "" ? unfiltered : null), + stale: () => null, + }), + "Unfiltered title", + ); + }); + + test("applySwitchedSessionToClientState sets identity, chat, usage and workspace — and nothing else", async () => { + const { applySwitchedSessionToClientState } = await import( + absPath("engine/session-switch.js") + ); + const cs = { + sessionId: "old", + mcodeSessionId: "old-sid", + sessionTitle: "Old", + chat: ["● stale"], + usage: { sessionInput: 7, sessionOutput: 8, sessionTotal: 15, contextUsed: 3 }, + workspace: { dir: "/ws/old", branch: "main", tree: ["t"] }, + lastUsedWorkspace: "/ws/last-used", + }; + const out = applySwitchedSessionToClientState(cs, { + target: { id: "new", mcodeSessionId: SID_A, title: "New", chat: ["● fresh"] }, + workspaceDir: "/ws/new", + }); + assert.equal(out, cs, "the same object is mutated in place"); + assert.equal(cs.sessionId, "new"); + assert.equal(cs.mcodeSessionId, SID_A); + assert.equal(cs.sessionTitle, "New"); + assert.deepEqual(cs.chat, ["● fresh"]); + assert.deepEqual(cs.usage, { + sessionInput: 0, + sessionOutput: 0, + sessionTotal: 0, + contextUsed: 3, + }, "the three cumulative counters zero, every other key preserved"); + assert.deepEqual(cs.workspace, { dir: "/ws/new", branch: null, tree: null }); + assert.equal( + cs.lastUsedWorkspace, + "/ws/last-used", + "switching is browsing: last-used-workspace must not move", + ); + }); + + test("applySwitchedSessionToClientState normalises the three optional target fields", async () => { + const { applySwitchedSessionToClientState } = await import( + absPath("engine/session-switch.js") + ); + const cs = { usage: {} }; + applySwitchedSessionToClientState(cs, { + target: { id: "u1" }, + workspaceDir: "/ws/x", + }); + assert.equal(cs.mcodeSessionId, null, "a record with no engine sid binds to null"); + assert.equal(cs.sessionTitle, "Untitled", "and an absent title reads as Untitled"); + assert.deepEqual(cs.chat, [], "a non-array chat is an empty chat, never a crash"); + }); + }); + + // --------------------------------------------------------------------- + // 5. The route + // --------------------------------------------------------------------- + + describe("handleSwitchSession — HTTP parsing, status codes, and the fail-closed audit", () => { + /** Every export the REAL facade has, so a partial mock fails loud. */ + const FACADE_EXPORTS = [ + "SESSION_SWITCH_ENDPOINTS", + "applyEngineSessionSwitch", + "applySwitchedSessionToClientState", + "chatLooksCumulative", + "checkSessionSwitchCapability", + "isSwitchableMcodeSessionId", + "lookupCachedMcodeTitle", + "readEngineSwitchTranscript", + "resolveSessionSwitchProvider", + "resolveSwitchTarget", + "resolveSwitchWorkspace", + "selectTranscriptBackfill", + ]; + function mockFacade(t, impls) { + const namedExports = {}; + for (const name of FACADE_EXPORTS) { + namedExports[name] = () => { + throw new Error(`B6 test called engine/session-switch.js#${name}, which this case did not stub`); + }; + } + Object.assign(namedExports, impls); + t.mock.module(absPath("engine/session-switch.js"), { namedExports }); + } + const loadRoute = async () => + import(`${absPath("routes/sessions.js")}?bust=${bust++}`); + + const OK_AUDIT = { + event: "session.switch", + target: "webui-A", + cid: "tab-1", + actor: "user", + payload: { + from: "webui-previous", + matchKind: "webuiId", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_A, + workspaceFallback: false, + }, + }; + const OK_BODY = { + ok: true, + session: { + id: "webui-A", + mcodeSessionId: SID_A, + title: "T", + workspace: WS_A, + workspaceFallback: false, + chat: ["● x"], + }, + }; + + test("a missing id is the route's own 400, in the route's own words", async (t) => { + await setupMocks(t, {}); + mockFacade(t, {}); + const route = await loadRoute(); + // `{id: 42}` is deliberately NOT in this list: `(payload.id || "").trim()` + // throws a TypeError on a number, which is the pre-facade behaviour + // and a 500 rather than a 400. Tightening it would be a behaviour + // change dressed as a hardening, and this batch promises none — + // it is recorded as a question for the request-validation pass + // instead (see KNOWN DEBT, `routes/sessions.js`). + for (const body of [{}, { id: "" }, { id: " " }, { id: null }]) { + const res = mkRes(); + await route.handleSwitchSession(jsonReq(body), res, { cs: mkCs(), cid: "tab-1" }); + assert.equal(res.written[0].status, 400); + // Pre-existing asymmetry, preserved: this body is the ONE shape + // on this endpoint that does not carry the charset. + assert.equal(res.written[0].headers["Content-Type"], "application/json"); + assert.equal(res.written[1].body, '{"ok":false,"error":"id required"}'); + } + }); + + // Table-driven across every outcome the facade can report. The + // status, the Content-Type and the body are all pinned; the two + // Content-Type spellings are the pre-existing asymmetry and must not + // be tidied into one. + const OUTCOMES = [ + [ + "not_found", + 404, + "application/json", + '{"ok":false,"error":"session not found"}', + ], + [ + "workspace_refused", + 400, + "application/json; charset=utf-8", + JSON.stringify({ ok: false, error: "outside: /nope", attempted: "/nope" }), + ], + ]; + for (const [outcome, status, contentType, body] of OUTCOMES) { + test(`${outcome} → ${status} with Content-Type ${contentType}`, async (t) => { + await setupMocks(t, {}); + mockFacade(t, { + applyEngineSessionSwitch: async () => ({ + outcome, + statusHint: status, + payload: JSON.parse(body), + audit: null, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleSwitchSession(jsonReq({ id: "x" }), res, { + cs: mkCs(), + cid: "tab-1", + }); + assert.equal(res.written[0].status, status); + assert.equal(res.written[0].headers["Content-Type"], contentType); + assert.equal(res.written[1].body, body); + assert.equal(res.written.length, 2, "a non-ok outcome writes exactly one response"); + }); + } + + test("the route writes the facade's audit event verbatim, then the state push, then the 200", async (t) => { + await setupMocks(t, {}); + mockFacade(t, { + applyEngineSessionSwitch: async () => ({ + outcome: "ok", + statusHint: 200, + matchKind: "webuiId", + workspace: { ok: true, dir: WS_A, fallback: false }, + transcript: null, + audit: OK_AUDIT, + payload: OK_BODY, + }), + }); + const route = await loadRoute(); + const res = mkRes(); + await route.handleSwitchSession(jsonReq({ id: "webui-A" }), res, { + cs: mkCs(), + cid: "tab-1", + }); + assert.equal(res.written[0].status, 200); + assert.equal(res.written[0].headers["Content-Type"], "application/json"); + assert.equal(res.written[1].body, JSON.stringify(OK_BODY)); + // The audit really landed: lib/events.js appends one NDJSON line + // per call, and the line carries the event name. + const auditPath = process.env.MCODE_WEBUI_EVENTS_PATH; + assert.ok(existsSync(auditPath), "the switch wrote no audit line at all"); + const raw = readFileSync(auditPath, "utf8"); + const last = raw.trim().split("\n").pop(); + assert.ok(last.includes("session.switch"), `last audit line: ${last}`); + }); + + test("a failed audit is fail-closed: 500, and the audit sink's own body", async (t) => { + await setupMocks(t, {}); + mockFacade(t, { + applyEngineSessionSwitch: async () => ({ + outcome: "ok", + statusHint: 200, + audit: OK_AUDIT, + payload: OK_BODY, + }), + }); + const route = await loadRoute(); + // Point the audit stream at a DIRECTORY: `events.js#append` writes + // atomically and throws EISDIR, which is the failure the fail-closed + // branch exists for. `_eventsPath()` reads the env lazily, so no + // re-import is needed. + const prev = process.env.MCODE_WEBUI_EVENTS_PATH; + const asDir = join(_eventsDir, "events-as-a-directory"); + mkdirSync(asDir, { recursive: true }); + process.env.MCODE_WEBUI_EVENTS_PATH = asDir; + try { + const res = mkRes(); + await route.handleSwitchSession(jsonReq({ id: "webui-A" }), res, { + cs: mkCs(), + cid: "tab-1", + }); + assert.equal(res.written[0].status, 500); + assert.equal( + res.written[0].headers["Content-Type"], + "application/json; charset=utf-8", + ); + assert.equal( + res.written[1].body, + '{"ok":false,"error":"audit write failed","detail":"session.switch"}', + ); + } finally { + process.env.MCODE_WEBUI_EVENTS_PATH = prev; + } + }); + + test("PROOF: a marker error from the facade escapes the route", async (t) => { + // Without a fresh `?bust=` re-import, `mock.module` would leave the + // route holding the PREVIOUS test's live binding, the marker would + // never be thrown, and this assertion would fail — which is the + // point: it is the only assertion in this section that cannot pass + // by accident. + await setupMocks(t, {}); + const marker = new Error("B6-MOCK-WAS-NOT-HONOURED"); + mockFacade(t, { + applyEngineSessionSwitch: async () => { + throw marker; + }, + }); + const route = await loadRoute(); + let caught = null; + try { + await route.handleSwitchSession(jsonReq({ id: "webui-A" }), mkRes(), { + cs: mkCs(), + cid: "tab-1", + }); + } catch (err) { + caught = err; + } + assert.ok( + caught, + "the route swallowed the facade error — either the mock did not take, or the route grew a catch", + ); + assert.equal(caught, marker, "the error is the mock's, by identity"); + }); + }); + + // --------------------------------------------------------------------- + // 6. The transcript seam, and what this batch retired + // --------------------------------------------------------------------- + + describe("the transcript seam", () => { + test("readEngineSwitchTranscript never throws — every failure is a value", async (t) => { + const mod = await bootFacade(t, { db: { constructThrows: true } }); + for (const mcodeSessionId of [SID_A, "not-a-sid", ""]) { + const r = await mod.readEngineSwitchTranscript({ mcodeSessionId }); + assert.equal(r.ok, false, mcodeSessionId); + assert.equal(r.source, "none"); + assert.deepEqual(r.lines, []); + assert.ok(r.reason, "a failure always names its reason for the operator log"); + } + }); + + test("readEngineSwitchTranscript reports the gate and the transport it asked under", async (t) => { + const mod = await bootFacade(t, { db: {} }); + const r = await mod.readEngineSwitchTranscript({ mcodeSessionId: SID_A }); + assert.equal(r.gate.endpoint, "POST /api/sessions/switch"); + assert.equal(r.gate.enforcement, "soft"); + assert.equal( + r.gate.gate === "unregistered-transport" || r.gate.gate === "checked", + true, + `unexpected gate ${r.gate.gate}`, + ); + }); + + test("RETIRED: routes/sessions.js no longer names lib/transcript.js at all", async () => { + // The part of the probe debt this batch actually collected. A + // static source assertion is the right instrument here: the claim + // is about an IMPORT GRAPH, and this suite has no render harness + // that could observe it. `export.js`'s comment still names the + // switch path by prose, which is exactly the kind of drift the + // assertion below is here to catch. + const src = readFileSync( + fileURLToPath(absPath("routes/sessions.js")), + "utf8", + ); + assert.equal( + /from\s+"\.\.\/lib\/transcript\.js"/.test(src), + false, + "the route must not import the transcript reader directly any more", + ); + assert.equal( + /loadTranscriptChatLines|readMcodeTranscript/.test(src), + false, + "the route must not call a transcript reader directly any more", + ); + // And the read is reachable exactly once, through the seam. + const facadeSrc = readFileSync( + fileURLToPath(absPath("engine/session-switch.js")), + "utf8", + ); + assert.equal( + /import\("\.\.\/lib\/transcript\.js"\)/.test(facadeSrc), + true, + "the seam is the single owner of the transcript read now", + ); + }); + + test("KEPT, deliberately: the probe set behind the seam is unchanged", async (t) => { + // KNOWN DEBT 1. The 3-candidate legacy probe set is still the + // implementation, because the default `acp` transport has no + // engine surface to replace it with and export's enrichment is + // byte-pinned to those same candidates. This test is the tripwire + // that makes the debt VISIBLE: if a later batch swaps the seam to + // the engine's `getMessages`, the switch's line set changes here + // and the failure names the batch that has to justify it. + const mod = await bootFacade(t, { + db: { rowsBySql: { [await v2ProbeSql()]: transcriptRows(SID_A) } }, + }); + const r = await mod.readEngineSwitchTranscript({ mcodeSessionId: SID_A }); + assert.equal(r.ok, true); + assert.equal(r.probeTable, "local_runtime_message_rows"); + assert.equal(r.probe, "v2-data-json", "the v2 data_json probe is the one that answers"); + assert.equal(r.messageCount, 3); + }); + }); +}); diff --git a/release/public-source.json b/release/public-source.json index be49ea09..54af5e37 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3457,6 +3457,7 @@ "packages/webui/server/engine/providers/tui-runtime-adapter.js", "packages/webui/server/engine/session-export.js", "packages/webui/server/engine/session-reads.js", + "packages/webui/server/engine/session-switch.js", "packages/webui/server/engine/session-tree-reads.js", "packages/webui/server/engine/session-writes.js", "packages/webui/server/engine/usage-reads.js", @@ -3603,6 +3604,7 @@ "packages/webui/test/lib/engine/model-reads.test.js", "packages/webui/test/lib/engine/session-export.test.js", "packages/webui/test/lib/engine/session-reads.test.js", + "packages/webui/test/lib/engine/session-switch.test.js", "packages/webui/test/lib/engine/session-tree-reads.test.js", "packages/webui/test/lib/engine/session-writes.test.js", "packages/webui/test/lib/engine/usage-reads.test.js", diff --git a/scripts/test-tmp-leak.check.mjs b/scripts/test-tmp-leak.check.mjs index 818fee2a..5ebd12ac 100644 --- a/scripts/test-tmp-leak.check.mjs +++ b/scripts/test-tmp-leak.check.mjs @@ -317,6 +317,10 @@ const KNOWN_PREFIXES = [ "webui-sessions-search-check-", "webui-sessions-test-events-", "webui-settings-test-events-", + "webui-switch-facade-db-", + "webui-switch-facade-events-", + "webui-switch-facade-outside-", + "webui-switch-facade-roots-", "webui-switch-test-db-", "webui-switch-test-events-", "webui-transcript-test-", From e4cf052247a1e6d16031dec25099f1e45cc9a13e Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Sat, 3 Oct 2026 01:50:36 +0800 Subject: [PATCH 20/21] chore: allowlist the leak-tripwire fixture in model-reads tests gitleaks' generic-api-key rule flags the deliberate sk-secret-should- never-leak fixture that model-reads.test.js uses as a leak-prevention tripwire (asserting the facade never serializes provider keys). The value is fake and the assertion exists to catch real leaks; allowlist the exact pairing instead of weakening the fixture. --- .gitleaks.toml | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/.gitleaks.toml b/.gitleaks.toml index ba943440..e1a9fe8d 100644 --- a/.gitleaks.toml +++ b/.gitleaks.toml @@ -117,3 +117,10 @@ paths = ['''(^|/)dist/webui/server\.js$'''] regexTarget = "match" regexes = ['''^[A-Za-z_$][A-Za-z0-9_$]*\.setRsaPrivateKey = [A-Za-z_$][A-Za-z0-9_$]*\.rsa\.setPrivateKey ?$''', '''^[A-Za-z_$][A-Za-z0-9_$]*\.privateKeyToAsn1 = [A-Za-z_$][A-Za-z0-9_$]*\.privateKeyToRSAPrivateKey ?$''', '''^[A-Za-z_$][A-Za-z0-9_$]*\.generateKey = [A-Za-z_$][A-Za-z0-9_$]*\.pbe\.generatePkcs12Key;?$'''] +[[rules.allowlists]] +description = "Leak-prevention tripwire fixture: asserts the facade never serializes this fake key" +condition = "AND" +paths = ['''(^|/)packages/webui/test/lib/engine/model-reads\.test\.js$'''] +regexTarget = "match" +regexes = ['''apiKey: "sk-secret-should-never-leak"'''] + From 3f5b8d22996dd2acaf7a176e86f8370a230c0441 Mon Sep 17 00:00:00 2001 From: acer_feng <857688528@qq.com> Date: Sat, 3 Oct 2026 01:51:25 +0800 Subject: [PATCH 21/21] test(webui): pin session-writes cleanup-orphans test to isolated paths --- .../test/lib/engine/session-writes.test.js | 53 +++++++++++++++++-- 1 file changed, 50 insertions(+), 3 deletions(-) diff --git a/packages/webui/test/lib/engine/session-writes.test.js b/packages/webui/test/lib/engine/session-writes.test.js index 767de244..24239521 100644 --- a/packages/webui/test/lib/engine/session-writes.test.js +++ b/packages/webui/test/lib/engine/session-writes.test.js @@ -64,6 +64,51 @@ import { withDecisions, } from "../../helpers/_setup.js"; import { mkTmpDir, rmTmpDir } from "../../helpers/tmp.js"; + +// --------------------------------------------------------------------------- +// Per-file path isolation (B5 test-hygiene fix) +// --------------------------------------------------------------------------- +// `readOrphanSessionWriteIds` reads `lib/config.js#SESSIONS_DB`, and that +// constant is frozen when config.js is FIRST evaluated — which happens inside +// the first test that pulls `engine/index.js` into the registry, long before +// the preview test below runs. So the pin has to sit at module scope: setting +// it inside the test body would be a no-op dressed up as isolation. +// +// The bug this kills: the preview test asserted `count:0` because the +// developer's `~/.mcode-webui/sessions.json` "does not exist in this +// environment". On any machine that has actually used the app it DOES exist, +// and the assertion was a statement about the developer's home directory +// rather than about the facade — green on a clean CI runner, red on every +// workstation, and unfixable by editing the product. +// +// Four variables, all rooted in one tracked temp directory (SPEC §7's +// isolation trio plus the file under test): +// +// MCODE_WEBUI_SESSIONS_DB — the file the sweep reads; the one that leaked +// MCODE_WEBUI_DATA_DIR — its parent, so every other path config.js +// derives from the data dir lands here too +// MCODE_WEBUI_SETTINGS_PATH — settings.json, which config.js reads at import +// MINIMAX_DATA_DIR — the engine's data dir; without it the +// MCODE_RUNTIME_DB contract still resolves +// against the real ~/.minimax +// +// SESSIONS_DB is deliberately left NON-EXISTENT. The empty sweep is the shape +// this red line pins, and after this change it is guaranteed by construction +// instead of by the absence of a file the test never created. +const ISOLATED_DIR = mkTmpDir("webui-session-writes-b5-"); +process.env.MCODE_WEBUI_SESSIONS_DB = join(ISOLATED_DIR, "sessions.json"); +process.env.MCODE_WEBUI_DATA_DIR = ISOLATED_DIR; +process.env.MCODE_WEBUI_SETTINGS_PATH = join(ISOLATED_DIR, "settings.json"); +process.env.MINIMAX_DATA_DIR = ISOLATED_DIR; + +after(() => { + rmTmpDir(ISOLATED_DIR); + delete process.env.MCODE_WEBUI_SESSIONS_DB; + delete process.env.MCODE_WEBUI_DATA_DIR; + delete process.env.MCODE_WEBUI_SETTINGS_PATH; + delete process.env.MINIMAX_DATA_DIR; +}); + // Type discrimination goes through the exported predicate, never // `err.name`. `engine/capabilities.js` is never `mock.module`d by this // file, so the `instanceof` inside it resolves against the same class the @@ -989,9 +1034,11 @@ describe("M3-B5 — session write family", () => { // notice a reshuffle. await setupMocks(t, { acp: {} }); const mod = await import(`${absPath("engine/session-writes.js")}?shape=${bust++}`); - // The store read is against the real config's SESSIONS_DB, which - // does not exist in this environment, so the answer is the empty - // case — which is the shape most likely to be "simplified". + // The store read is against the SESSIONS_DB pinned at module scope, a + // path that intentionally does not exist, so the answer is the empty + // case — which is the shape most likely to be "simplified". The same + // assertion held on a CI runner by accident; here it holds because the + // test owns the path it reads. const sweep = await mod.readOrphanSessionWriteIds({ transport: RUNTIME }); assert.equal( JSON.stringify(sweep.payload),