From b150a225d5e6256e9f04d7bbbfe12d2c7aca54a6 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 01:55:58 -0700 Subject: [PATCH 01/75] fix(server): a keyless daemon starts a new chat on its configured private model MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit biorouter serve spawns biorouterd with no user-action key (SD-7), and the new-chat gate demanded that key's proof before binding a private default, so every POST /agent/start on a serve daemon configured with a private provider was refused 409 — the 2026-09-10 QA run's F1. The configured default is the person's choice, made out of band with biorouter configure (open question 24 put the raise at the write), so on a daemon that holds no key the new-chat bind no longer asks for a proof nobody can give. A daemon that holds a key (the desktop's) still refuses a proof-less first bind: the renderer sends the proof for free, and a model holding the recovered secret should not mint a private chat. The exemption covers one provider at one moment. On a keyless daemon, /agent/update_provider now measures every move onto a private model from Public (raise_baseline), so a new chat cannot be moved sideways to a private model nobody configured. --- crates/biorouter-server/src/routes/agent.rs | 123 +++++- .../tests/new_chat_no_user_key.rs | 364 ++++++++++++++++++ 2 files changed, 480 insertions(+), 7 deletions(-) create mode 100644 crates/biorouter-server/tests/new_chat_no_user_key.rs diff --git a/crates/biorouter-server/src/routes/agent.rs b/crates/biorouter-server/src/routes/agent.rs index 8dc1604f8..629edf7e9 100644 --- a/crates/biorouter-server/src/routes/agent.rs +++ b/crates/biorouter-server/src/routes/agent.rs @@ -341,6 +341,59 @@ fn configured_new_session_provider() -> Result, Er } } +/// SD-9 (`docs/deployment/serve-decisions.md`): does binding the operator's +/// configured default provider to a brand-new chat need a person's proof? +/// +/// A new chat has no capability of its own yet, so a private default reads as a +/// raise from Public — the reading `update_agent_provider` gives any first bind. +/// What differs is who chose the model. `/agent/start` names no provider; it +/// binds `BIOROUTER_PROVIDER`, a key only a proven person may write over HTTP +/// (DR-16, open question 24) or the operator wrote out of band with `biorouter +/// configure`. The proof is therefore asked for only where it can be given: +/// +/// * `Proven` — the desktop renderer, which sends `X-User-Action` on every start +/// (`ui/desktop/src/sessions.ts`). Binds. +/// * `Unproven` — a daemon that holds a key, and a caller that did not present +/// it: a script, or the model holding the daemon secret AR-11 found +/// recoverable. Refused, as before. The person proves themselves here at no +/// cost, and without the refusal a model could mint a private-capability chat +/// with an extension set of its own choosing. +/// * `NoKeyInstalled` — `biorouter serve` (SD-7) or a hand-run `biorouterd`. +/// Nobody on this daemon can prove anything, so a refusal here refuses the +/// person too, on every new chat, always — the 2026-09-10 QA's F1. Binds. +/// +/// Only the configured default is exempt, and only at creation: `/agent/start` +/// cannot name any other provider, and on a keyless daemon +/// `update_agent_provider` refuses every private bind ([`raise_baseline`]), so a +/// new chat there never reaches a private model the operator did not choose. +fn new_chat_bind_needs_user(enforced: bool, tier: ProviderTier, proof: UserActionProof) -> bool { + enforced + && raise_needs_user_action(ProviderTier::Public, tier) + && match proof { + UserActionProof::Proven | UserActionProof::NoKeyInstalled => false, + UserActionProof::Unproven => true, + } +} + +/// The capability `update_agent_provider` measures a raise from — SD-9's other +/// half. +/// +/// On a daemon that holds a user-action key it is the chat's live capability, +/// as it always was. On one that holds none, no chat's private capability came +/// from anything a person proved over HTTP: the only private binding such a +/// daemon hands out through its routes is [`new_chat_bind_needs_user`]'s +/// creation-time bind to the configured default. There is no private floor for +/// a request to build on, so a bind to ANY private provider is measured from +/// Public, and refused, since no proof can arrive. Without this, SD-9 would let +/// a new chat on a private default be moved sideways, `Private -> Private`, to a +/// private model nobody configured. +fn raise_baseline(current: ProviderTier, proof: UserActionProof) -> ProviderTier { + match proof { + UserActionProof::NoKeyInstalled => ProviderTier::Public, + UserActionProof::Proven | UserActionProof::Unproven => current, + } +} + async fn bind_new_session_provider( state: &AppState, session: &Session, @@ -355,10 +408,12 @@ async fn bind_new_session_provider( message: format!("Failed to configure the selected provider for the new chat: {error}"), status: StatusCode::BAD_REQUEST, })?; - if biorouter::privacy::privacy_tiers_enabled() - && raise_needs_user_action(ProviderTier::Public, provider.tier()) - && !is_user_action(headers) - { + // DR-15's master opt-out, read inside the gate as every #56 surface does. + if new_chat_bind_needs_user( + biorouter::privacy::privacy_tiers_enabled(), + provider.tier(), + user_action_proof(headers), + ) { return Err(ErrorResponse { message: PrivacyRefusal::TierRaiseNeedsUser { requested: provider_name, @@ -619,7 +674,7 @@ pub struct RestartAgentResponse { (status = 200, description = "Agent started successfully", body = Session), (status = 400, description = "Bad request", body = ErrorResponse), (status = 401, description = "Unauthorized - invalid secret key"), - (status = 409, description = "The selected private provider requires user-action proof", body = ErrorResponse), + (status = 409, description = "The configured provider is private and this daemon holds a user-action key, but the request carried no proof it came from the user (SD-9). A daemon with no user-action key binds its configured provider without one.", body = ErrorResponse), (status = 500, description = "Internal server error", body = ErrorResponse) ) )] @@ -1297,7 +1352,9 @@ async fn get_callable_tool_count( a public model cannot be bound to a private chat \ (body = PrivacyBarrierBody). DR-16: the bind raises this \ chat's capability to Private and the request carried no \ - proof it came from the user (body = plain text)", + proof it came from the user; on a daemon with no \ + user-action key, any bind to a private model (SD-9) \ + (body = plain text)", body = PrivacyBarrierBody), (status = 424, description = "Agent not initialized"), (status = 500, description = "Internal server error") @@ -1370,6 +1427,10 @@ async fn update_agent_provider( // DR-16 rejected. Sideways and downward binds are untouched for every // caller, which is what keeps Gate A's path, the CLI, // `restore_provider_from_session` and every apps-runtime bind working. + // The one exception is this route on a daemon with no user-action key, + // where a move onto a private model is measured from Public however the + // chat is bound today — `raise_baseline`, SD-9. The predicate itself is + // unchanged, and none of the in-process binds above passes through here. // // An unbound session reads as Public — `Agent::provider` errors when nothing // is bound (and when Gate B' refuses what is), and the conservative reading @@ -1380,11 +1441,13 @@ async fn update_agent_provider( .await .map(|p| p.tier()) .unwrap_or(ProviderTier::Public); + // SD-9's other half — see `raise_baseline`. + let baseline = raise_baseline(current, user_action_proof(&headers)); // DR-15's master opt-out, read INSIDE the gate. A direct read, not a // `CallCapability`: a provider raise over HTTP is not a tool call and has no // admitted capability to inherit. if biorouter::privacy::privacy_tiers_enabled() - && raise_needs_user_action(current, new_provider.tier()) + && raise_needs_user_action(baseline, new_provider.tier()) && !is_user_action(&headers) { return Err(( @@ -3117,6 +3180,52 @@ mod new_session_provider_binding_tests { .await .unwrap(); } + + /// SD-9, every proof verdict against both tiers. The keyless arm cannot be + /// reached through a route in this binary — the installed digest is a + /// process-global `OnceLock` and the test above installs one — so the route + /// half lives in `tests/new_chat_no_user_key.rs`, a binary that never does. + #[test] + fn only_a_daemon_that_can_check_a_proof_asks_a_new_chat_for_one() { + use UserActionProof::{NoKeyInstalled, Proven, Unproven}; + assert!(!new_chat_bind_needs_user(true, ProviderTier::Private, Proven)); + assert!(new_chat_bind_needs_user(true, ProviderTier::Private, Unproven)); + assert!( + !new_chat_bind_needs_user(true, ProviderTier::Private, NoKeyInstalled), + "a keyless daemon refusing its own configured default refuses every person, always" + ); + for proof in [Proven, Unproven, NoKeyInstalled] { + // A public default raises nothing, for anyone. + assert!(!new_chat_bind_needs_user(true, ProviderTier::Public, proof)); + // DR-15's master opt-out turns the gate off, not the question. + assert!(!new_chat_bind_needs_user(false, ProviderTier::Private, proof)); + } + } + + /// SD-9's other half: a keyless daemon measures every move onto a private + /// model from Public, so its exemption for the configured default cannot be + /// carried sideways to a private model nobody configured. + #[test] + fn a_keyless_daemon_has_no_private_floor_for_a_switch_to_build_on() { + use UserActionProof::{NoKeyInstalled, Proven, Unproven}; + for current in [ProviderTier::Private, ProviderTier::Public] { + assert_eq!(raise_baseline(current, NoKeyInstalled), ProviderTier::Public); + // A daemon that can check a proof keeps measuring from the live binding. + assert_eq!(raise_baseline(current, Proven), current); + assert_eq!(raise_baseline(current, Unproven), current); + } + // The composition `update_agent_provider` asks. Sideways onto a private + // model is a raise only where no proof can be checked. + let sideways = |proof| { + raise_needs_user_action( + raise_baseline(ProviderTier::Private, proof), + ProviderTier::Private, + ) + }; + assert!(sideways(NoKeyInstalled)); + assert!(!sideways(Unproven)); + assert!(!sideways(Proven)); + } } #[cfg(test)] diff --git a/crates/biorouter-server/tests/new_chat_no_user_key.rs b/crates/biorouter-server/tests/new_chat_no_user_key.rs new file mode 100644 index 000000000..4eb7f8264 --- /dev/null +++ b/crates/biorouter-server/tests/new_chat_no_user_key.rs @@ -0,0 +1,364 @@ +//! SD-9: on a daemon that holds no proof-of-user key — `biorouter serve` (SD-7) +//! or a hand-run `biorouterd` — a new chat starts on the operator's configured +//! provider, private or not, and nothing on the HTTP surface can move it onto a +//! private model the operator did not configure. +//! +//! The 2026-09-10 QA run (finding F1) measured the first half failing: with +//! `versa_azure` configured, every `POST /agent/start` on a `serve` daemon was +//! refused 409, because the new-chat gate asked for a proof this daemon can +//! never check. The browser showed nothing at all. +//! +//! ⚠ **Its own test binary on purpose**, for the reason `approval_no_user_key.rs` +//! gives: the installed digest is a process-global `OnceLock`, the lib's tests +//! install one, and inside that binary the keyless state is unreachable once the +//! first of them wins. Nothing here installs a digest — which is exactly how +//! `biorouter serve` starts its daemon. + +// Redirects this binary's Biorouter data/config/state dirs at a throwaway root +// before `main`, so nothing here can open the developer's real `sessions.db` +// or write the developer's real `config.yaml`. +#[path = "../src/test_sandbox.rs"] +mod test_sandbox; + +use std::collections::HashMap; +use std::sync::Arc; +use std::time::Duration; + +use axum::body::Body; +use axum::http::{HeaderMap, Request, StatusCode}; +use axum::Router; +use biorouter::config::{with_config_overrides, Config}; +use biorouter::conversation::message::Message; +use biorouter::privacy::refusal::USER_ACTION_REFUSAL_MARKER; +use biorouter::privacy::SessionClassification; +use biorouter::providers::versa_azure::VERSA_AZURE_DEPLOYMENT; +use biorouter_server::auth::{user_action_proof, UserActionProof}; +use biorouter_server::state::AppState; +use serde_json::{json, Value}; +use serial_test::serial; +use tower::ServiceExt; +use wiremock::matchers::{body_string_contains, method, path}; +use wiremock::{Mock, MockServer, ResponseTemplate}; + +/// The posture the QA run used: `biorouter configure` chose Versa. The key is a +/// placeholder — these tests construct the provider and never send it a request. +fn versa_is_the_configured_default() -> HashMap { + HashMap::from([ + ("BIOROUTER_PROVIDER".to_string(), "versa_azure".to_string()), + ( + "BIOROUTER_MODEL".to_string(), + VERSA_AZURE_DEPLOYMENT.to_string(), + ), + ( + "VERSA_AZURE_API_KEY".to_string(), + "placeholder-never-sent".to_string(), + ), + ]) +} + +/// Every test here stands on this: the daemon under test holds no key. +fn assert_the_daemon_is_keyless() { + assert_eq!( + user_action_proof(&HeaderMap::new()), + UserActionProof::NoKeyInstalled, + "something in this binary installed a user-action digest, so these tests would be \ + measuring a desktop daemon rather than a `biorouter serve` one" + ); + assert!( + biorouter::privacy::privacy_tiers_enabled(), + "privacy tiers are off, so no gate below would fire either way" + ); +} + +async fn post_json(app: Router, uri: &str, body: Value) -> (StatusCode, String) { + let response = app + .oneshot( + Request::builder() + .uri(uri) + .method("POST") + .header("content-type", "application/json") + .body(Body::from(body.to_string())) + .unwrap(), + ) + .await + .unwrap(); + let status = response.status(); + let bytes = tokio::time::timeout( + Duration::from_secs(120), + axum::body::to_bytes(response.into_body(), usize::MAX), + ) + .await + .expect("the response body did not finish within two minutes") + .unwrap(); + (status, String::from_utf8_lossy(&bytes).into_owned()) +} + +fn start_request(working_dir: &std::path::Path) -> Value { + // No extensions: the chat under test needs a model and nothing else, and + // the machine default set is not this test's subject. + json!({ "working_dir": working_dir, "extension_overrides": [] }) +} + +async fn discard(state: &Arc, session_id: &str) { + // The tests run serially, so every cached agent is this test's. + state.clear_cached_agents().await; + let _ = state.session_manager().delete_session(session_id).await; +} + +/// F1, the half the QA run measured: the configured private provider is the +/// person's choice, made at the terminal, so binding it to a brand-new chat is +/// not a switch and needs no proof — on the one kind of daemon where no proof +/// can exist. +#[tokio::test(flavor = "multi_thread")] +#[serial] +async fn a_keyless_daemon_starts_a_new_chat_on_its_configured_private_model() { + assert_the_daemon_is_keyless(); + let state = AppState::new().await.unwrap(); + let dir = tempfile::tempdir().unwrap(); + + let (status, body) = with_config_overrides( + versa_is_the_configured_default(), + post_json( + biorouter_server::routes::agent::routes(Arc::clone(&state)), + "/agent/start", + start_request(dir.path()), + ), + ) + .await; + assert_eq!( + status, + StatusCode::OK, + "a keyless daemon refused to start a chat on its own configured model: {body}" + ); + let id = serde_json::from_str::(&body).unwrap()["id"] + .as_str() + .expect("the started session carries an id") + .to_string(); + + let row = state.session_manager().get_session(&id, false).await.unwrap(); + assert_eq!(row.provider_name.as_deref(), Some("versa_azure")); + assert_eq!( + row.model_config.map(|config| config.model_name).as_deref(), + Some(VERSA_AZURE_DEPLOYMENT) + ); + // O5: the ratchet fires on the first turn, never on the bind. A chat that + // has touched nothing is not yet private — it is private-CAPABLE. + assert_eq!(row.privacy_tier, SessionClassification::Public); + + let agent = state.get_agent_for_route(id.clone()).await.unwrap(); + assert_eq!( + agent + .provider() + .await + .expect("the first turn must not fail with `Provider not set`") + .get_name(), + "versa_azure" + ); + + discard(&state, &id).await; +} + +/// The complement the exemption must not leak into: a request that asks a new +/// chat for a DIFFERENT private model than the one the operator configured is +/// still refused. `/agent/start` cannot name one, so the request that can is +/// `/agent/update_provider` on the chat it just made — `Private -> Private`, +/// which the raise predicate alone would call sideways and wave through. +#[tokio::test(flavor = "multi_thread")] +#[serial] +async fn a_keyless_daemon_will_not_move_a_new_chat_to_a_private_model_nobody_configured() { + assert_the_daemon_is_keyless(); + let state = AppState::new().await.unwrap(); + let dir = tempfile::tempdir().unwrap(); + + let (status, body) = with_config_overrides( + versa_is_the_configured_default(), + post_json( + biorouter_server::routes::agent::routes(Arc::clone(&state)), + "/agent/start", + start_request(dir.path()), + ), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + let id = serde_json::from_str::(&body).unwrap()["id"] + .as_str() + .unwrap() + .to_string(); + + // A loopback Ollama is Private (`self_hosted_tier`), and constructing one + // opens no connection, so port 1 is never dialled. + let (status, body) = with_config_overrides( + HashMap::from([( + "OLLAMA_HOST".to_string(), + "http://127.0.0.1:1".to_string(), + )]), + post_json( + biorouter_server::routes::agent::routes(Arc::clone(&state)), + "/agent/update_provider", + json!({ "session_id": id, "provider": "ollama", "model": "stub-model" }), + ), + ) + .await; + assert_eq!( + status, + StatusCode::CONFLICT, + "a keyless daemon moved a new chat onto a private model the operator never configured: \ + {body}" + ); + assert!( + body.contains(USER_ACTION_REFUSAL_MARKER), + "refused, but not by the tier gate: {body}" + ); + + let row = state.session_manager().get_session(&id, false).await.unwrap(); + assert_eq!( + row.provider_name.as_deref(), + Some("versa_azure"), + "the refused switch rewrote the row anyway" + ); + let agent = state.get_agent_for_route(id.clone()).await.unwrap(); + assert_eq!(agent.provider().await.unwrap().get_name(), "versa_azure"); + + discard(&state, &id).await; +} + +/// SD-1 is untouched by SD-9: the configured default itself still cannot be +/// changed from a keyless daemon, which is what makes it the operator's choice. +#[tokio::test(flavor = "multi_thread")] +#[serial] +async fn a_keyless_daemon_still_refuses_to_change_the_configured_default() { + assert_the_daemon_is_keyless(); + let state = AppState::new().await.unwrap(); + let (status, body) = post_json( + biorouter_server::routes::config_management::routes(state), + "/config/set_provider", + json!({ "provider": "versa_azure", "model": VERSA_AZURE_DEPLOYMENT }), + ) + .await; + assert_eq!(status, StatusCode::CONFLICT, "{body}"); +} + +/// "privacy_tier ratchets on the first turn as usual": the exemption changes who +/// may bind the configured model, and nothing about what a turn on it does. +/// +/// The configured default here is an Ollama endpoint on loopback, because a +/// Versa module re-pointed at a stub server is no longer Private +/// (`ucsf_gateway_tier` reads the resolved host) and a test must not send +/// traffic to the real gateway. ⚠ No model runs: the endpoint is a `wiremock` +/// stub that returns one canned completion. +#[tokio::test(flavor = "multi_thread")] +#[serial] +async fn the_first_turn_on_a_keyless_default_chat_ratchets_it_as_usual() { + assert_the_daemon_is_keyless(); + + let stub = MockServer::start().await; + let chunk = |delta: Value, finish: Value| { + json!({ + "id": "stub", "object": "chat.completion.chunk", "model": "stub-model", + "choices": [{ "index": 0, "delta": delta, "finish_reason": finish }] + }) + }; + let sse = format!( + "data: {}\n\ndata: {}\n\ndata: [DONE]\n\n", + chunk(json!({ "role": "assistant", "content": "ready" }), Value::Null), + chunk(json!({ "content": "" }), json!("stop")), + ); + Mock::given(method("POST")) + .and(path("/v1/chat/completions")) + .and(body_string_contains("\"stream\":true")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("content-type", "text/event-stream") + .set_body_string(sse), + ) + .with_priority(1) + .mount(&stub) + .await; + // Anything that asks without streaming — the chat's auto-title — gets a + // plain completion rather than an event stream it cannot parse. + Mock::given(method("POST")) + .and(path("/v1/chat/completions")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "id": "stub", "object": "chat.completion", "model": "stub-model", + "choices": [{ "index": 0, "finish_reason": "stop", + "message": { "role": "assistant", "content": "Stub title" } }], + "usage": { "prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2 } + }))) + .mount(&stub) + .await; + + // Written to this binary's sandboxed config.yaml rather than scoped to a + // task: the turn runs on a spawned task, which a task-local override would + // not reach. + let config = Config::global(); + config.set_param("BIOROUTER_PROVIDER", "ollama").unwrap(); + config.set_param("BIOROUTER_MODEL", "stub-model").unwrap(); + config.set_param("OLLAMA_HOST", stub.uri()).unwrap(); + + let state = AppState::new().await.unwrap(); + let dir = tempfile::tempdir().unwrap(); + let (status, body) = post_json( + biorouter_server::routes::agent::routes(Arc::clone(&state)), + "/agent/start", + start_request(dir.path()), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + let id = serde_json::from_str::(&body).unwrap()["id"] + .as_str() + .unwrap() + .to_string(); + let before = state.session_manager().get_session(&id, false).await.unwrap(); + assert_eq!(before.provider_name.as_deref(), Some("ollama")); + assert_eq!(before.privacy_tier, SessionClassification::Public); + + let message = Message::user().with_text("Reply with the single word ready."); + let (status, stream) = post_json( + biorouter_server::routes::reply::routes(Arc::clone(&state)), + "/reply", + json!({ "user_message": message, "session_id": id }), + ) + .await; + assert_eq!(status, StatusCode::OK, "{stream}"); + assert!( + stream.contains("\"Finish\""), + "the turn did not finish: {stream}" + ); + assert!( + stream.contains("ready"), + "the turn did not run on the stub: {stream}" + ); + + let after = state.session_manager().get_session(&id, false).await.unwrap(); + assert_eq!( + after.privacy_tier, + SessionClassification::Private, + "a turn on the configured private model did not ratchet the chat" + ); + assert_eq!(after.privacy_reason.as_deref(), Some("turn:ollama")); + + // The chat's NEXT request, now that it is private. A keyless daemon reaches a + // private chat only for a caller whose stated capability covers it, so a + // browser tab that states nothing loses the chat it just started; the host's + // provider, which is what a tab on this host states (SD-9), keeps it. + let reach = |caller: Option<&str>| { + let mut request = Request::builder().uri(format!("/sessions/{id}")); + if let Some(provider) = caller { + request = request.header("X-Caller-Provider", provider); + } + let app = biorouter_server::routes::session::routes(Arc::clone(&state)); + async move { + app.oneshot(request.body(Body::empty()).unwrap()) + .await + .unwrap() + .status() + } + }; + assert_eq!(reach(None).await, StatusCode::FORBIDDEN); + assert_eq!(reach(Some("ollama")).await, StatusCode::OK); + + discard(&state, &id).await; + for key in ["BIOROUTER_PROVIDER", "BIOROUTER_MODEL", "OLLAMA_HOST"] { + let _ = config.delete(key); + } +} From 7b77f9dadcbdb94ef8d8b034f1bbcc9898930514 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 01:55:58 -0700 Subject: [PATCH 02/75] test(privacy): point AR-15's closure scan at update_agent_provider the_documented_closure_is_the_one_the_code_performs took the FIRST TierRaiseNeedsUser in routes/agent.rs. Since eb594ded that has been the new-chat gate, not update_agent_provider's, so deleting the proof check from update_provider left the scan green (measured). The scan now starts at the handler AR-15 is about and is bounded by the next route. --- .../tests/privacy_ar15_is_retired.rs | 21 ++++++++++++++++--- 1 file changed, 18 insertions(+), 3 deletions(-) diff --git a/crates/biorouter-server/tests/privacy_ar15_is_retired.rs b/crates/biorouter-server/tests/privacy_ar15_is_retired.rs index 6862181b4..057683be3 100644 --- a/crates/biorouter-server/tests/privacy_ar15_is_retired.rs +++ b/crates/biorouter-server/tests/privacy_ar15_is_retired.rs @@ -345,9 +345,24 @@ fn the_documented_closure_is_the_one_the_code_performs() { // The docs now assert a specific gate. If the gate goes, the docs are // wrong in the *dangerous* direction — claiming a hole is closed when it is // open — and no other test in this tree ties the two together. - let refusal = AGENT_ROUTE - .find("PrivacyRefusal::TierRaiseNeedsUser") - .expect("routes/agent.rs no longer refuses an unproven tier raise at all"); + // + // ⚠ The scan starts at `update_agent_provider`, the handler AR-15 is about. + // It used to take the FIRST refusal in the file, which stopped being this + // handler's when `eb594ded` put the new-chat gate above it: from then on the + // scan read that gate, and deleting the proof from this one left it green. + // The new-chat gate is SD-9's, a different rule with its own tests in + // `routes/agent.rs`. + let handler = AGENT_ROUTE + .find("async fn update_agent_provider") + .expect("update_agent_provider moved; AR-15's gate is the one inside it"); + let body = cut_from(AGENT_ROUTE, handler); + // Bounded by the next route, so a refusal that left this handler cannot be + // found in the one after it. + let body = cut_to(body, body.find("\n#[utoipa::path(").unwrap_or(body.len())); + let refusal = handler + + body + .find("PrivacyRefusal::TierRaiseNeedsUser") + .expect("update_agent_provider no longer refuses an unproven tier raise at all"); let guard = cut_to(AGENT_ROUTE, refusal) .rfind(" if ") .expect("the tier-raise refusal is not inside an `if`"); From 51d863cb698a11119c22554f5d925aff8514313a Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 01:55:58 -0700 Subject: [PATCH 03/75] fix(desktop): say when a chat fails to start, and keep what was typed A failed POST /agent/start reached console.error and nothing else on the Home composer (the QA run's F1): the text vanished and nothing appeared. One pure notice, startChatFailureNotice, now words every failure for a person (the daemon's own text stays behind Copy error), and every surface that starts a chat reports through it: Home, a fresh tab, a workflow window, the launcher, both Ask Biorouter buttons and Workflows. A source scan keeps an eighth surface from picking its own way to fail. Also fixed on the way: a workflow window whose start failed re-ran its creation effect in a loop, and a failed workflow start replaced the Workflows list with a list-load error. A browser tab now states the host's configured model (X-Caller-Provider) where the desktop proves the person: measured, a chat started on a private host model 403s its next request once its first reply makes it private, and 200s with the host provider stated. --- ui/desktop/src/App.tsx | 10 ++ ui/desktop/src/components/BaseChat.tsx | 16 +- .../GroupedExtensionLoadingToast.tsx | 12 +- .../src/components/Hub.startFailure.test.tsx | 140 +++++++++++++++++ ui/desktop/src/components/Hub.tsx | 22 ++- .../components/workflows/WorkflowsView.tsx | 9 +- ui/desktop/src/toasts.tsx | 7 +- ui/desktop/src/utils/launcherMessage.ts | 7 +- ui/desktop/src/utils/startChatFailure.test.ts | 146 ++++++++++++++++++ ui/desktop/src/utils/startChatFailure.ts | 92 +++++++++++ .../src/utils/userAction.surface.test.ts | 75 +++++++++ ui/desktop/src/utils/userAction.ts | 63 ++++++++ 12 files changed, 580 insertions(+), 19 deletions(-) create mode 100644 ui/desktop/src/components/Hub.startFailure.test.tsx create mode 100644 ui/desktop/src/utils/startChatFailure.test.ts create mode 100644 ui/desktop/src/utils/startChatFailure.ts create mode 100644 ui/desktop/src/utils/userAction.surface.test.ts diff --git a/ui/desktop/src/App.tsx b/ui/desktop/src/App.tsx index 8c43452cb..237e2c913 100644 --- a/ui/desktop/src/App.tsx +++ b/ui/desktop/src/App.tsx @@ -58,6 +58,8 @@ import { View, ViewOptions } from './utils/navigationUtils'; import { useNavigation } from './hooks/useNavigation'; import { errorMessage } from './utils/conversionUtils'; +import { startChatFailureNotice } from './utils/startChatFailure'; +import { toastError } from './toasts'; import { getInitialWorkingDir } from './utils/workingDir'; import { deliverLauncherMessage } from './utils/launcherMessage'; import { ChatStreamProvider } from './hooks/chatStreamStore'; @@ -168,6 +170,14 @@ const PairRouteContent = ({ setChat }: { setChat: (chat: ChatType) => void }) => }); } catch (error) { console.error('Failed to create session:', error); + toastError(startChatFailureNotice(error, { kept: false })); + // Leave. Every input this effect keys on is still true after a + // failure, so staying would re-run it the moment `isCreatingSession` + // drops — a `POST /agent/start` loop for as long as the refusal + // holds. Home is the resting state for a layout with no tabs (#38). + // The cargo here is a workflow window's; a launcher message arrives + // with a session id and never reaches this branch. + navigate('/', { replace: true }); } finally { setIsCreatingSession(false); } diff --git a/ui/desktop/src/components/BaseChat.tsx b/ui/desktop/src/components/BaseChat.tsx index d3abed351..5519df2c9 100644 --- a/ui/desktop/src/components/BaseChat.tsx +++ b/ui/desktop/src/components/BaseChat.tsx @@ -72,7 +72,8 @@ import { useBoundAffiliation } from './privacy/useBoundAffiliation'; import { getSessionTitlePadding } from './Layout/TitlebarControls'; import { announceSessionName, renameSession } from '../utils/sessionNameSync'; import { toastError, toastWarning } from '../toasts'; -import { errorMessage, isConnectionError } from '../utils/conversionUtils'; +import { errorMessage } from '../utils/conversionUtils'; +import { startChatFailureNotice } from '../utils/startChatFailure'; import { Greeting } from './common/Greeting'; import { navigateWithViewTransition } from '../utils/navigationUtils'; import { unwrapGuardrailFrameInContent } from '../utils/guardrailFrame'; @@ -736,8 +737,9 @@ export function collectArtifactsFromMessages( * is unreachable the awaited createSession rejects *after* the text is already * gone — and the bare catch used to show nothing, so the message silently * vanished. Restore the typed text (via a `restore-chat-input` event the composer - * listens for) and surface a visible toast. Connection detection only picks the - * wording; the toast + restore fire on ANY rejection, so no silent path remains. + * listens for) and surface a visible toast. The words are + * `startChatFailureNotice`'s, shared with every other surface that starts a + * chat; the toast + restore fire on ANY rejection, so no silent path remains. * Exported so it can be unit-tested without Electron. */ export function handleCreateSessionError( @@ -754,13 +756,7 @@ export function handleCreateSessionError( }, }) ); - const connection = isConnectionError(err); - toastError({ - title: connection ? 'Backend disconnected' : 'Failed to start chat', - msg: connection - ? 'Biorouter could not reach its backend. Your message was kept - try again in a moment.' - : errorMessage(err), - }); + toastError(startChatFailureNotice(err, { kept: true })); } /** diff --git a/ui/desktop/src/components/GroupedExtensionLoadingToast.tsx b/ui/desktop/src/components/GroupedExtensionLoadingToast.tsx index ed2552cd9..4a6fc4328 100644 --- a/ui/desktop/src/components/GroupedExtensionLoadingToast.tsx +++ b/ui/desktop/src/components/GroupedExtensionLoadingToast.tsx @@ -4,6 +4,11 @@ import { Button } from './ui/button'; import { ModalShell } from './ModalShell'; import { NotificationContent, type NotificationStatus } from './alerts/NotificationSurface'; import { startNewSession } from '../sessions'; +// ⚠ `toasts.tsx` renders this component, so this import closes a cycle. It is +// the one `utils/extensionErrorUtils` already closes, and `toastError` is only +// read inside a click handler, long after both modules have evaluated. +import { toastError } from '../toasts'; +import { startChatFailureNotice } from '../utils/startChatFailure'; import { useNavigation } from '../hooks/useNavigation'; import { formatExtensionErrorMessage } from '../utils/extensionErrorUtils'; import { getInitialWorkingDir } from '../utils/workingDir'; @@ -232,7 +237,12 @@ export function GroupedExtensionLoadingToast({ onOpenChange={setReportOpen} extensions={extensions} onAskBiorouter={ - setView ? (hints) => startNewSession(getInitialWorkingDir(), hints, setView) : null + setView + ? (hints) => + void startNewSession(getInitialWorkingDir(), hints, setView).catch((error) => + toastError(startChatFailureNotice(error, { kept: false })) + ) + : null } /> )} diff --git a/ui/desktop/src/components/Hub.startFailure.test.tsx b/ui/desktop/src/components/Hub.startFailure.test.tsx new file mode 100644 index 000000000..3f9715245 --- /dev/null +++ b/ui/desktop/src/components/Hub.startFailure.test.tsx @@ -0,0 +1,140 @@ +import React from 'react'; +import { describe, it, expect, vi, beforeEach } from 'vitest'; +import { render, screen, fireEvent, waitFor } from '@testing-library/react'; + +/** + * F1 of the 2026-09-10 QA run, on the surface where it was measured: type into + * the Home composer, press Enter, and have `POST /agent/start` fail. The + * composer had already wiped itself, and the failure went to `console.error` + * and nowhere else — no toast, no message, the text gone. + * + * The real Hub and the real ChatInput are rendered, because the property lives + * in the handshake between them: ChatInput restores its box only when the + * submit resolves `false`, and Hub decides what it resolves. Only the heavy or + * context-hungry children are stubbed, as in `ChatInput.workingDir.test.tsx`. + */ + +const { mockCreateSession, mockToastError } = vi.hoisted(() => ({ + mockCreateSession: vi.fn(), + mockToastError: vi.fn(), +})); + +vi.mock('../sessions', () => ({ createSession: mockCreateSession })); +vi.mock('../toasts', async (importOriginal) => ({ + ...(await importOriginal()), + toastError: mockToastError, +})); +vi.mock('./sessions/SessionsInsights', () => ({ SessionInsights: () => null })); +vi.mock('./ConfigContext', () => ({ + useConfig: () => ({ + extensionsList: [], + getProviders: vi.fn(async () => []), + read: vi.fn(async () => null), + }), +})); +vi.mock('./ModelAndProviderContext', () => ({ + useModelAndProvider: () => ({ + getCurrentModelAndProvider: vi.fn(async () => ({ model: null, provider: null })), + currentModel: null, + currentProvider: null, + currentModelSupportsVision: false, + currentModelSupportedInputMimeTypes: null, + }), +})); +vi.mock('../hooks/useDiverge', () => ({ + useDiverge: () => ({ diverge: vi.fn() }), +})); +vi.mock('./settings/models/bottom_bar/ModelsBottomBar', () => ({ default: () => null })); +vi.mock('./bottom_menu/BottomMenuExtensionSelection', () => ({ + BottomMenuExtensionSelection: () => null, +})); +vi.mock('./bottom_menu/BottomMenuSkillSelection', () => ({ + BottomMenuSkillSelection: () => null, +})); +vi.mock('./bottom_menu/BottomMenuKnowledgeSelection', () => ({ + BottomMenuKnowledgeSelection: () => null, +})); +vi.mock('./bottom_menu/BottomMenuReasoningEffort', () => ({ + BottomMenuReasoningEffort: () => null, +})); +vi.mock('./bottom_menu/CostTracker', () => ({ CostTracker: () => null })); +vi.mock('./MessageQueue', () => ({ default: () => null })); +vi.mock('./MentionPopover', () => { + const MentionPopoverMock = React.forwardRef(() => null); + MentionPopoverMock.displayName = 'MentionPopoverMock'; + return { default: MentionPopoverMock }; +}); +vi.mock('../api', () => ({ + getSession: vi.fn(async () => ({ data: null })), + llamacppStatus: vi.fn(async () => ({ data: {} })), + updateWorkingDir: vi.fn(async () => ({ data: {} })), +})); + +import Hub from './Hub'; + +/** What `POST /agent/start` answered on the QA run's `biorouter serve` daemon. */ +const SERVE_DAEMON_REFUSAL = { + message: + "Switching this chat to a private model is the user's decision, not yours. The request to " + + "switch it to 'versa_azure' did not come from the model picker, so the chat is unchanged and " + + 'still on its current model. Do not retry; the same call will be refused again.', +}; + +beforeEach(() => { + vi.clearAllMocks(); + Object.assign(window, { + appConfig: { + get: (key: string) => (key === 'BIOROUTER_WORKING_DIR' ? '/default/workdir' : undefined), + }, + electron: { + directoryChooser: vi.fn(async () => ({ canceled: true, filePaths: [] })), + addRecentDir: vi.fn(), + logInfo: vi.fn(), + getPathForFile: vi.fn(() => ''), + on: vi.fn(), + off: vi.fn(), + }, + }); +}); + +describe('Hub: a chat that fails to start', () => { + it('says so in words for a person, and the composer keeps what was typed', async () => { + mockCreateSession.mockRejectedValueOnce(SERVE_DAEMON_REFUSAL); + const setView = vi.fn(); + const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}); + render(); + + const composer = screen.getByPlaceholderText('Ask Biorouter anything…') as HTMLTextAreaElement; + fireEvent.change(composer, { target: { value: 'Reply with the single word ready.' } }); + fireEvent.keyDown(composer, { key: 'Enter' }); + + await waitFor(() => expect(mockCreateSession).toHaveBeenCalledTimes(1)); + await waitFor(() => expect(mockToastError).toHaveBeenCalledTimes(1)); + const notice = mockToastError.mock.calls[0][0] as { title: string; msg: string }; + expect(notice.title).toBe('Failed to start chat'); + expect(notice.msg).not.toContain('Do not retry'); + expect(notice.msg).toContain('Your message was kept.'); + + await waitFor(() => expect(composer.value).toBe('Reply with the single word ready.')); + expect(setView).not.toHaveBeenCalled(); + consoleError.mockRestore(); + }); + + it('opens the chat when the start succeeds, with nothing to report', async () => { + mockCreateSession.mockResolvedValueOnce({ id: 'session-1' }); + const setView = vi.fn(); + render(); + + const composer = screen.getByPlaceholderText('Ask Biorouter anything…') as HTMLTextAreaElement; + fireEvent.change(composer, { target: { value: 'hello' } }); + fireEvent.keyDown(composer, { key: 'Enter' }); + + await waitFor(() => + expect(setView).toHaveBeenCalledWith( + 'pair', + expect.objectContaining({ resumeSessionId: 'session-1', initialMessage: 'hello' }) + ) + ); + expect(mockToastError).not.toHaveBeenCalled(); + }); +}); diff --git a/ui/desktop/src/components/Hub.tsx b/ui/desktop/src/components/Hub.tsx index 8b4228478..9e4b2f506 100644 --- a/ui/desktop/src/components/Hub.tsx +++ b/ui/desktop/src/components/Hub.tsx @@ -29,6 +29,8 @@ import { getInitialWorkingDir } from '../utils/workingDir'; import { createSession } from '../sessions'; import LoadingBioRouter from './LoadingBioRouter'; import type { UserAttachment } from '../types/message'; +import { toastError } from '../toasts'; +import { startChatFailureNotice } from '../utils/startChatFailure'; export default function Hub({ setView, @@ -39,7 +41,16 @@ export default function Hub({ const [workingDir, setWorkingDir] = useState(getInitialWorkingDir()); const [isCreatingSession, setIsCreatingSession] = useState(false); - const handleSubmit = async (e: React.FormEvent) => { + /** + * Resolves FALSE when no chat was started, which is ChatInput's signal to put + * the box back exactly as the user left it — text, reference chips and pasted + * images — since it wiped itself before this awaited anything. + * + * ⚠ The failure used to reach `console.error` and nothing else: the text + * vanished and the Home screen sat unchanged (the 2026-09-10 QA run, F1). It + * is now a toast in words for a person — see `startChatFailureNotice`. + */ + const handleSubmit = async (e: React.FormEvent): Promise => { const customEvent = e as unknown as CustomEvent; const combinedTextFromInput = customEvent.detail?.value || ''; const attachments = (customEvent.detail?.attachments ?? []) as UserAttachment[]; @@ -49,6 +60,7 @@ export default function Hub({ const extensionConfigs = getExtensionConfigsWithOverrides(extensionsList); clearExtensionOverrides(); setIsCreatingSession(true); + e.preventDefault(); try { const session = await createSession(workingDir, { @@ -61,13 +73,17 @@ export default function Hub({ initialMessage: combinedTextFromInput, initialAttachments: attachments, }); + return true; } catch (error) { console.error('Failed to create session:', error); setIsCreatingSession(false); + toastError(startChatFailureNotice(error, { kept: true })); + return false; } - - e.preventDefault(); } + // A second send while the first is still creating the chat: refused, so + // the composer keeps it. + return false; }; return ( diff --git a/ui/desktop/src/components/workflows/WorkflowsView.tsx b/ui/desktop/src/components/workflows/WorkflowsView.tsx index fdea8dc64..032bee99a 100644 --- a/ui/desktop/src/components/workflows/WorkflowsView.tsx +++ b/ui/desktop/src/components/workflows/WorkflowsView.tsx @@ -46,6 +46,7 @@ import { import { SearchView } from '../conversation/SearchView'; import cronstrue from 'cronstrue'; import { getInitialWorkingDir } from '../../utils/workingDir'; +import { startChatFailureNotice } from '../../utils/startChatFailure'; import { DropdownMenu, DropdownMenuContent, @@ -147,9 +148,11 @@ export default function WorkflowsView() { resumeSessionId: session.id, }); } catch (error) { - console.error('Failed to load workflow:', error); - const errorMsg = error instanceof Error ? error.message : 'Failed to load workflow'; - setError(errorMsg); + // A toast, not `setError`: that state is the LIST's load error, and + // setting it replaced a list that had loaded fine with "Couldn't load + // workflows", whose Try again reloads the list rather than the chat. + console.error('Failed to start workflow chat:', error); + toastError(startChatFailureNotice(error, { kept: false })); } }; diff --git a/ui/desktop/src/toasts.tsx b/ui/desktop/src/toasts.tsx index 2671b6340..de32255df 100644 --- a/ui/desktop/src/toasts.tsx +++ b/ui/desktop/src/toasts.tsx @@ -8,6 +8,7 @@ import { ExtensionLoadingStatus, } from './components/GroupedExtensionLoadingToast'; import { getInitialWorkingDir } from './utils/workingDir'; +import { startChatFailureNotice } from './utils/startChatFailure'; import { launchDependencyDebugSession } from './utils/launchDependencyDebug'; import type { DependencyFailure } from './utils/dependencyDebugPrompt'; @@ -340,7 +341,11 @@ function ToastErrorContent({ diff --git a/ui/desktop/src/utils/launcherMessage.ts b/ui/desktop/src/utils/launcherMessage.ts index de2924669..832e3f5b5 100644 --- a/ui/desktop/src/utils/launcherMessage.ts +++ b/ui/desktop/src/utils/launcherMessage.ts @@ -1,5 +1,7 @@ import type { NavigateFunction } from 'react-router-dom'; import { createSession } from '../sessions'; +import { toastError } from '../toasts'; +import { startChatFailureNotice } from './startChatFailure'; import { getInitialWorkingDir } from './workingDir'; import type { PairRouteState } from '../components/Pair'; @@ -31,7 +33,9 @@ import type { PairRouteState } from '../components/Pair'; * The message itself still travels as location.state, which is where the IN * effect reads cargo from once the param has opened the gate. On failure we * navigate nowhere: the marker keeps the window parked on the empty pane (the - * pre-#38 resting state) rather than silently discarding the launch intent. + * pre-#38 resting state) rather than silently discarding the launch intent — + * and the failure is said out loud, in `startChatFailureNotice`'s words, since + * a parked pane explains nothing by itself. */ export async function deliverLauncherMessage( navigate: NavigateFunction, @@ -46,5 +50,6 @@ export async function deliverLauncherMessage( }); } catch (error) { console.error('Failed to create session for launcher message:', error); + toastError(startChatFailureNotice(error, { kept: false })); } } diff --git a/ui/desktop/src/utils/startChatFailure.test.ts b/ui/desktop/src/utils/startChatFailure.test.ts new file mode 100644 index 000000000..2e4d56bff --- /dev/null +++ b/ui/desktop/src/utils/startChatFailure.test.ts @@ -0,0 +1,146 @@ +import { readdirSync, readFileSync, statSync } from 'node:fs'; +import { join, relative, sep } from 'node:path'; +import { describe, expect, it } from 'vitest'; +import { + BACKEND_DISCONNECTED_TITLE, + START_CHAT_FAILED_TITLE, + isStartRefusedForWantOfProof, + startChatFailureNotice, +} from './startChatFailure'; + +/** + * The body the 2026-09-10 QA run captured from `POST /agent/start` on a + * `biorouter serve` daemon with `versa_azure` configured (finding F1), verbatim. + * Under `throwOnError` the generated client throws the parsed body, so this + * object IS what a caller's `catch` receives. + */ +const QA_REFUSAL = { + message: + "Switching this chat to a private model is the user's decision, not yours. The request to " + + "switch it to 'versa_azure' did not come from the model picker, so the chat is unchanged and " + + 'still on its current model. Do not retry; the same call will be refused again. If this task ' + + 'genuinely needs a private model, stop and ask the user to switch this chat to a private ' + + 'model first, in the desktop app under Settings > Models, or with the model chip in the ' + + 'composer.', +}; + +describe('startChatFailureNotice', () => { + it('words the privacy refusal for a person and keeps the daemon text behind Copy error', () => { + const notice = startChatFailureNotice(QA_REFUSAL, { kept: true }); + expect(notice.title).toBe(START_CHAT_FAILED_TITLE); + // Every instruction in the refusal is addressed to a model, and the last + // one names a control the browser surface deliberately disables (SD-1). + for (const forAModel of [ + 'Do not retry', + 'not yours', + 'model picker', + 'stop and ask the user', + 'model chip', + ]) { + expect(notice.msg).not.toContain(forAModel); + } + expect(notice.msg).toContain('Your message was kept.'); + expect(notice.traceback).toBe(QA_REFUSAL.message); + }); + + it('only claims the message was kept when the caller put it back', () => { + expect(startChatFailureNotice(QA_REFUSAL, { kept: false }).msg).not.toContain('kept'); + }); + + it('shows a failure already written for a person as it came', () => { + // The A/B leg of the QA run: the same sandbox with a public provider and no key. + const body = { + message: + 'Failed to configure the selected provider for the new chat: Configuration value not ' + + 'found: OPENAI_API_KEY', + }; + expect(startChatFailureNotice(body, { kept: false })).toEqual({ + title: START_CHAT_FAILED_TITLE, + msg: body.message, + }); + expect(startChatFailureNotice(body, { kept: true }).msg).toBe( + `${body.message} Your message was kept.` + ); + }); + + it('says the backend is unreachable when the request never got an answer', () => { + const notice = startChatFailureNotice(new TypeError('Failed to fetch'), { kept: true }); + expect(notice.title).toBe(BACKEND_DISCONNECTED_TITLE); + expect(notice.msg).toContain('could not reach its backend'); + expect(notice.msg).toContain('Your message was kept'); + }); + + it('recognizes the refusal by its marker, in the shapes the daemon sends, and nothing else', () => { + expect(isStartRefusedForWantOfProof(QA_REFUSAL)).toBe(true); + expect(isStartRefusedForWantOfProof(QA_REFUSAL.message)).toBe(true); + // A thrown Error that happens to carry the words is a bug, not a policy. + expect(isStartRefusedForWantOfProof(new Error(QA_REFUSAL.message))).toBe(false); + expect(isStartRefusedForWantOfProof({ message: 'Failed to create session: disk full' })).toBe( + false + ); + expect(isStartRefusedForWantOfProof(null)).toBe(false); + expect(isStartRefusedForWantOfProof(undefined)).toBe(false); + }); +}); + +/** + * Every surface that starts a chat reports a failure through the notice. + * + * Seven surfaces started a chat when this was written, and before it six of + * them handled a failure four different ways: a console line (the Home composer + * the QA run hit, the launcher), an unhandled rejection (both "Ask Biorouter" + * buttons), a retry loop (a window opened for a workflow), and a list-load + * error over a list that had loaded (Workflows). An eighth surface must not get + * to pick a fifth. + */ +describe('every surface that starts a chat', () => { + const SRC = join(__dirname, '..'); + const STARTS_A_CHAT = /\b(?:createSession|startNewSession|startAgent)\(/; + // `sessions.ts` DEFINES the first two and wraps the third; it reports nothing + // because it has no one to report to. + const DEFINITIONS = new Set(['sessions.ts']); + + const sourceFiles = (dir: string): string[] => + readdirSync(dir).flatMap((name) => { + const path = join(dir, name); + if (statSync(path).isDirectory()) { + // The generated client is not a surface. + return name === 'api' || name === 'node_modules' ? [] : sourceFiles(path); + } + return /\.tsx?$/.test(name) && !/\.test\.tsx?$/.test(name) ? [path] : []; + }); + + // Prose about a call is not a call: `navigationUtils.ts` names + // `startNewSession()` in a comment and starts nothing. + const withoutComments = (source: string) => + source.replace(/\/\*[\s\S]*?\*\//g, '').replace(/(^|\s)\/\/.*$/gm, '$1'); + + const surfaces = sourceFiles(SRC) + .map((path) => ({ + rel: relative(SRC, path).split(sep).join('/'), + source: withoutComments(readFileSync(path, 'utf8')), + })) + .filter(({ rel, source }) => !DEFINITIONS.has(rel) && STARTS_A_CHAT.test(source)); + + it('finds the surfaces it is guarding, so the scan is not vacuous', () => { + const found = surfaces.map(({ rel }) => rel); + for (const known of [ + 'App.tsx', + 'toasts.tsx', + 'utils/launcherMessage.ts', + 'components/Hub.tsx', + 'components/BaseChat.tsx', + 'components/GroupedExtensionLoadingToast.tsx', + 'components/workflows/WorkflowsView.tsx', + ]) { + expect(found).toContain(known); + } + }); + + it.each(surfaces.map(({ rel, source }) => [rel, source] as const))( + '%s reports a failed start in the shared words', + (_rel, source) => { + expect(source).toMatch(/startChatFailureNotice\(|handleCreateSessionError\(/); + } + ); +}); diff --git a/ui/desktop/src/utils/startChatFailure.ts b/ui/desktop/src/utils/startChatFailure.ts new file mode 100644 index 000000000..2fdcb3aff --- /dev/null +++ b/ui/desktop/src/utils/startChatFailure.ts @@ -0,0 +1,92 @@ +import { errorMessage, isConnectionError } from './conversionUtils'; +import { USER_ACTION_REFUSAL_MARKER } from './userAction'; + +/** + * What a person is told when `POST /agent/start` fails — on every surface that + * starts a chat: the Home composer, a fresh tab's composer, a window opened for + * a workflow, a launcher message, the "Ask Biorouter" buttons. + * + * The 2026-09-10 QA run (finding F1) found the Home composer swallowing this + * failure whole: the typed text vanished, nothing appeared, and the only trace + * was a console line. The daemon's refusal had been correct, and was written + * for a model — "Do not retry", "ask the user to switch this chat" — so showing + * it verbatim would have been the second half of the same bug. Hence one pure + * mapping, shared by every caller, that picks words for a person and keeps the + * daemon's own text in the toast's copyable details. + * + * Pure (no toast, no DOM) so the words are tested without rendering anything, + * and so `toasts.tsx`, which itself starts chats, can use it without an import + * cycle. Each caller hands the result to `toastError`. + */ +export type StartChatFailureNotice = { + title: string; + msg: string; + /** The daemon's own words, behind "Copy error", when `msg` replaced them. */ + traceback?: string; +}; + +export const START_CHAT_FAILED_TITLE = 'Failed to start chat'; +export const BACKEND_DISCONNECTED_TITLE = 'Backend disconnected'; + +/** + * The daemon refused to bind its private default because the request carried + * no proof it came from a person, on a backend that holds a user-action key + * (SD-9). The `serve` daemon holds none and binds its default, so this reaches + * a person only on a desktop app pointed at a backend started elsewhere — the + * case `NO_USER_PROOF_TOAST_MSG` in `ModelAndProviderContext` words the same way. + * + * ⚠ The route answers with an `ErrorResponse`, so under `throwOnError` the + * thrown value is the parsed `{ message }` object — not the plain string + * `isUserActionRefusal` tests for, which is `/agent/update_provider`'s shape. + * Both are accepted, keyed on the marker. A real `Error` carrying the same words + * is not a policy refusal, whatever it reads. + */ +export const isStartRefusedForWantOfProof = (error: unknown): boolean => { + if (error instanceof Error) return false; + const text = + typeof error === 'string' + ? error + : typeof error === 'object' && + error !== null && + 'message' in error && + typeof error.message === 'string' + ? error.message + : null; + return text !== null && text.includes(USER_ACTION_REFUSAL_MARKER); +}; + +/** + * @param kept whether the caller has put the message back where the person can + * see it (the composer). Say so only when it is true. + */ +export function startChatFailureNotice( + error: unknown, + { kept }: { kept: boolean } +): StartChatFailureNotice { + // `errorMessage` answers its default, not the text, for a bare string. + const daemonText = typeof error === 'string' ? error : errorMessage(error); + if (isConnectionError(error)) { + return { + title: BACKEND_DISCONNECTED_TITLE, + msg: kept + ? 'Biorouter could not reach its backend. Your message was kept - try again in a moment.' + : 'Biorouter could not reach its backend. Try again in a moment.', + traceback: daemonText, + }; + } + const keptSentence = kept ? ' Your message was kept.' : ''; + if (isStartRefusedForWantOfProof(error)) { + return { + title: START_CHAT_FAILED_TITLE, + msg: + 'Biorouter is connected to a backend started outside the app, which could not confirm ' + + 'the request came from you, so it did not start a chat on its private model. To use a ' + + `private model, start the chat in the Biorouter app.${keptSentence}`, + traceback: daemonText, + }; + } + // Every other refusal on this route is already written for a person — + // "Failed to configure the selected provider for the new chat: …" — so it is + // shown as it came, rather than replaced by something vaguer. + return { title: START_CHAT_FAILED_TITLE, msg: `${daemonText}${keptSentence}` }; +} diff --git a/ui/desktop/src/utils/userAction.surface.test.ts b/ui/desktop/src/utils/userAction.surface.test.ts new file mode 100644 index 000000000..70851962e --- /dev/null +++ b/ui/desktop/src/utils/userAction.surface.test.ts @@ -0,0 +1,75 @@ +import { readFileSync } from 'node:fs'; +import { join } from 'node:path'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; + +const { mockReadConfig } = vi.hoisted(() => ({ mockReadConfig: vi.fn() })); +vi.mock('../api', () => ({ readConfig: mockReadConfig })); + +import { + CALLER_PROVIDER_HEADER, + resetHostProviderForTests, + userActionHeaders, +} from './userAction'; +import { BROWSER_SURFACE_MARKER } from './surface'; + +/** + * SD-9: what each surface says on the requests the daemon's reach gate reads. + * + * The browser half is the one that was missing. On a `biorouter serve` daemon a + * chat started on the host's private model is private from its first reply, and + * the daemon reaches a private chat only for a caller whose stated capability + * covers it — so a tab that stated nothing lost its own chat after one answer. + */ +const onBrowser = () => { + document.documentElement.dataset.biorouterSurface = BROWSER_SURFACE_MARKER; +}; + +beforeEach(() => { + vi.clearAllMocks(); + resetHostProviderForTests(); + delete document.documentElement.dataset.biorouterSurface; + Object.assign(window, { + electron: { getUserActionKey: vi.fn(async () => 'desktop-user-action-key') }, + }); +}); + +afterEach(() => { + delete document.documentElement.dataset.biorouterSurface; +}); + +describe('userActionHeaders', () => { + it('proves the person on the desktop, and states no capability', async () => { + expect(await userActionHeaders()).toEqual({ 'X-User-Action': 'desktop-user-action-key' }); + expect(mockReadConfig).not.toHaveBeenCalled(); + }); + + it("states the host's configured model in a browser, and claims no proof", async () => { + onBrowser(); + mockReadConfig.mockResolvedValue({ data: 'versa_azure' }); + + expect(await userActionHeaders()).toEqual({ [CALLER_PROVIDER_HEADER]: 'versa_azure' }); + expect(mockReadConfig).toHaveBeenCalledWith({ + body: { key: 'BIOROUTER_PROVIDER', is_secret: false }, + }); + // Read once per page: the host's model cannot change under a running tab. + await userActionHeaders(); + expect(mockReadConfig).toHaveBeenCalledTimes(1); + }); + + it('says nothing when the host model cannot be read, and asks again next time', async () => { + onBrowser(); + mockReadConfig.mockRejectedValueOnce(new TypeError('Failed to fetch')); + expect(await userActionHeaders()).toEqual({}); + + mockReadConfig.mockResolvedValueOnce({ data: 'versa_azure' }); + expect(await userActionHeaders()).toEqual({ [CALLER_PROVIDER_HEADER]: 'versa_azure' }); + }); + + it('uses the header name the daemon reads', () => { + const gate = readFileSync( + join(__dirname, '../../../../crates/biorouter-server/src/routes/session_reach.rs'), + 'utf8' + ); + expect(gate).toContain(`pub const CALLER_PROVIDER_HEADER: &str = "${CALLER_PROVIDER_HEADER}";`); + }); +}); diff --git a/ui/desktop/src/utils/userAction.ts b/ui/desktop/src/utils/userAction.ts index dd3c5da4c..2e6cb6c47 100644 --- a/ui/desktop/src/utils/userAction.ts +++ b/ui/desktop/src/utils/userAction.ts @@ -1,3 +1,6 @@ +import { readConfig } from '../api'; +import { isBrowserSurface } from './surface'; + /** * Issue #56 DR-16: the header that proves a request came from the person at the * keyboard rather than from the model. @@ -96,7 +99,67 @@ export const COPY_OF_PRIVATE_REFUSAL_MARKER = 'only the person at the keyboard m export const isPrivateCopyRefusal = (error: unknown): boolean => typeof error === 'string' && error.includes(COPY_OF_PRIVATE_REFUSAL_MARKER); +/** + * Mirrored from `CALLER_PROVIDER_HEADER` in + * `crates/biorouter-server/src/routes/session_reach.rs`. It carries the NAME of + * the provider the caller runs under; the daemon resolves the tier itself. + */ +export const CALLER_PROVIDER_HEADER = 'X-Caller-Provider'; + +/** The host's `BIOROUTER_PROVIDER`, once it has been read successfully. */ +let hostProvider: string | undefined; + +/** + * The provider the machine running `biorouter serve` was configured with. + * + * Only a successful read is cached. A failure answers `null` and is asked again + * next time, so a transient error cannot pin the page to the public side for as + * long as it stays open. + */ +async function hostConfiguredProvider(): Promise { + if (hostProvider) return hostProvider; + try { + const { data } = await readConfig({ body: { key: 'BIOROUTER_PROVIDER', is_secret: false } }); + if (typeof data === 'string' && data.trim()) { + hostProvider = data.trim(); + return hostProvider; + } + } catch { + // Say nothing, which the daemon reads as the public side — fail-safe. + } + return null; +} + +/** For tests: forget the cached host provider. */ +export const resetHostProviderForTests = (): void => { + hostProvider = undefined; +}; + +/** + * The headers that answer the daemon's "may this caller reach this chat?" on + * the requests that need an answer — one surface at a time. + * + * * **The desktop app proves the person**: `X-User-Action`, the key the + * Electron main process minted and handed the daemon's digest on stdin. + * * **A browser states its model** (SD-9). The `biorouter serve` daemon holds no + * key (SD-7), so there is no person to prove, and a browser session runs the + * model the host was configured with (SD-1). It says so the way `biorouter + * session` does from a terminal: `X-Caller-Provider` naming that provider. + * Without it, a chat started on a private host model became unreachable from + * the tab the moment its first reply made it private — measured: the next + * request answered 403, the same request stating the host's provider 200. + * + * ⚠ **Not authentication, and not a widening.** Anything holding the daemon + * secret can send that header already (`session_reach.rs` says as much); what + * this adds is that the one legitimate browser client says what is true of it. + * On a host configured with a public model it states a public one, and private + * chats stay out of reach exactly as before. + */ export const userActionHeaders = async (): Promise> => { + if (isBrowserSurface()) { + const provider = await hostConfiguredProvider(); + return provider ? { [CALLER_PROVIDER_HEADER]: provider } : {}; + } try { return { 'X-User-Action': await window.electron.getUserActionKey() }; } catch { From a80ca7116f2fa5e86128e01c9e2b6d0085b0c7df Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 02:06:48 -0700 Subject: [PATCH 04/75] fix(desktop): every start-failure notice offers Copy error MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The notice set a traceback only when it replaced the daemon's words, so the commonest failure on a serve host — the configured model has no credential there — showed no Copy error. It is always set now. --- ui/desktop/src/utils/startChatFailure.test.ts | 2 ++ ui/desktop/src/utils/startChatFailure.ts | 13 +++++++++---- 2 files changed, 11 insertions(+), 4 deletions(-) diff --git a/ui/desktop/src/utils/startChatFailure.test.ts b/ui/desktop/src/utils/startChatFailure.test.ts index 2e4d56bff..d9936bbce 100644 --- a/ui/desktop/src/utils/startChatFailure.test.ts +++ b/ui/desktop/src/utils/startChatFailure.test.ts @@ -57,6 +57,8 @@ describe('startChatFailureNotice', () => { expect(startChatFailureNotice(body, { kept: false })).toEqual({ title: START_CHAT_FAILED_TITLE, msg: body.message, + // Always copyable: the troubleshooting guide sends people to "Copy error". + traceback: body.message, }); expect(startChatFailureNotice(body, { kept: true }).msg).toBe( `${body.message} Your message was kept.` diff --git a/ui/desktop/src/utils/startChatFailure.ts b/ui/desktop/src/utils/startChatFailure.ts index 2fdcb3aff..93fb26761 100644 --- a/ui/desktop/src/utils/startChatFailure.ts +++ b/ui/desktop/src/utils/startChatFailure.ts @@ -21,8 +21,8 @@ import { USER_ACTION_REFUSAL_MARKER } from './userAction'; export type StartChatFailureNotice = { title: string; msg: string; - /** The daemon's own words, behind "Copy error", when `msg` replaced them. */ - traceback?: string; + /** The daemon's own words, behind the toast's "Copy error". */ + traceback: string; }; export const START_CHAT_FAILED_TITLE = 'Failed to start chat'; @@ -87,6 +87,11 @@ export function startChatFailureNotice( } // Every other refusal on this route is already written for a person — // "Failed to configure the selected provider for the new chat: …" — so it is - // shown as it came, rather than replaced by something vaguer. - return { title: START_CHAT_FAILED_TITLE, msg: `${daemonText}${keptSentence}` }; + // shown as it came, rather than replaced by something vaguer. It is still the + // traceback too, which is what puts "Copy error" on the toast. + return { + title: START_CHAT_FAILED_TITLE, + msg: `${daemonText}${keptSentence}`, + traceback: daemonText, + }; } From fe977583726b6e0893b0470a0851cbc5ce783f7e Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 02:06:49 -0700 Subject: [PATCH 05/75] =?UTF-8?q?docs(serve):=20SD-9=20=E2=80=94=20a=20new?= =?UTF-8?q?=20chat=20starts=20on=20the=20operator's=20model=20without=20a?= =?UTF-8?q?=20proof?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Records the ruling behind the new-chat fix: what it exempts (the configured default, at creation, on a daemon with no user-action key), what it does not (any other private model; any daemon that holds a key), the two halves that keep it from opening more, and the consequence to accept, with the privacy checklist's two questions answered as an enumeration. browser-access.md no longer promises a private-provider serve host works while it refused every chat, and says what a browser chat on one does. programmatic-session-access.md names the browser as a sender of X-Caller-Provider and the one proof-less bind. SD-1 named a route that does not exist (/config/provider); it is /config/set_provider. The two 409 descriptions in the OpenAPI spec now say which daemon refuses. --- CLAUDE.md | 16 +++ docs/deployment/browser-access.md | 28 ++++- .../deployment/programmatic-session-access.md | 15 ++- docs/deployment/serve-decisions.md | 101 +++++++++++++++++- ui/desktop/openapi.json | 4 +- ui/desktop/src/api/types.gen.ts | 4 +- 6 files changed, 156 insertions(+), 12 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 17e8c1799..e689d88c6 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1177,6 +1177,22 @@ replaced a standalone `biorouter-headless` binary and its Linux tarball, both de **agent**, so every surface that writes a capability key asks `isBrowserSurface()` (`ui/desktop/src/utils/surface.ts`) and explains *before* the user can reach the 409. Do not "fix" browser mode by weakening the refusal. +- **A new chat starts on the configured model without a proof, and nothing else does** (SD-9). + Until it, a `serve` daemon with a private provider configured refused EVERY `/agent/start` + (the 2026-09-10 QA's F1): the new-chat bind asked for a proof a keyless daemon cannot check. + Three pieces, each load-bearing — measured by removing it: on a keyless daemon + `new_chat_bind_needs_user` (`routes/agent.rs`) lets the configured default bind, while a keyed + daemon still refuses a proof-less private first bind; `raise_baseline` makes a keyless + daemon's `/agent/update_provider` measure every move onto a private model from Public, or the + exemption would carry sideways to a private model nobody configured; and the browser states the + host's model as `X-Caller-Provider` (`userActionHeaders()` on `isBrowserSurface()`), without + which a chat's first reply ratcheted it private and its next request 403'd. Tests: + `cargo test -p biorouter-server --test new_chat_no_user_key` (its own binary: the digest is a + process-global `OnceLock`). ⚠ **Still unreachable in a browser, and out of SD-9's scope:** + `/agent/cancel` and `/interrupt` require the proof unconditionally, so Stop and mid-turn + steering cannot work on a keyless daemon. ⚠ `privacy_ar15_is_retired.rs`'s closure scan took + the FIRST `TierRaiseNeedsUser` in `routes/agent.rs`, which from `eb594ded` was the new-chat gate + and not AR-15's — it now starts at `update_agent_provider`. - **A control that can never work here says so, before it is touched** (SD-8). The same `Stdio::null()` that closes SD-1 means NO approval carrying `requires_user_proof` can ever be granted on a `serve` daemon — for anyone, always. So `confirm_tool_action` answers a diff --git a/docs/deployment/browser-access.md b/docs/deployment/browser-access.md index 299acf6b2..259b8c568 100644 --- a/docs/deployment/browser-access.md +++ b/docs/deployment/browser-access.md @@ -22,7 +22,9 @@ first and then [Headless Linux deployment](headless-linux.md). ## Quickstart Choose the provider and model **before** you start serving — a browser session cannot change them -(see [The model is fixed before you start](#the-model-is-fixed-before-you-start)): +(see [The model is fixed before you start](#the-model-is-fixed-before-you-start)). Either kind +works: a commercial model, or a private one your institution hosts or that runs on the machine, +which is the choice to make for patient data. ```bash biorouter configure @@ -165,6 +167,21 @@ provider is chosen once, at the terminal, and the tier that choice implies holds session in that daemon. A run started against an institutional model is private for its whole life; one started against a commercial model is public for its whole life. Neither can drift. +What that means for a chat you start in the browser: + +- **It starts on the configured model**, private or not, with nothing asked of you — choosing it + at the terminal was the decision. +- **A private model makes it private with its first reply.** The tab you are in keeps working with + it: the browser tells the daemon which model it runs under, and a chat is open to anything + running under a model at least as private as the chat. +- **Nothing in the browser can move it onto a different private model.** The configured one is the + only private model a chat started here ever reaches. +- **A private chat started in the desktop application opens here only when the host's model is + private too.** On a host configured with a commercial model it stays out of reach, with a card + saying so; open it in the desktop application instead. + +The reasoning is recorded as [decision SD-9](serve-decisions.md#sd-9--a-new-chat-starts-on-the-operators-model-without-a-proof-and-nothing-else-does). + **The fix is to choose the provider before you start serving:** ```bash @@ -184,7 +201,7 @@ differs: | Area | In a browser | |---|---| -| Chat, sessions, history, extensions, skills, knowledge bases, workflows | Work as they do in the desktop application. | +| Chat, sessions, history, extensions, skills, knowledge bases, workflows | Work as they do in the desktop application, with two differences that follow from the fixed model: a new chat starts on the host's configured model, and a private chat opens only on a host whose configured model is private. See [The model is fixed before you start](#the-model-is-fixed-before-you-start). | | Workspace control, several conversations at once, live app agents | Work — these are WebSocket-backed daemon routes, reached on the same origin. | | Model and provider selection | **Not available.** See [The model is fixed before you start](#the-model-is-fixed-before-you-start). | | File and folder pickers | No native dialog. You type a path, and it is a path **on the machine running the daemon**, not on the machine holding the browser. | @@ -210,6 +227,13 @@ you ran, and finds it either next to that CLI or next to the application the CLI recorded). If the application has moved or been reinstalled since, run `biorouter setup-path` again from inside it — `\resources\bin\biorouter.exe setup-path`. +**A message does not start a chat.** The composer keeps what you typed, and a notice in the corner +says why. The commonest cause is on the host rather than in the browser — for example *Failed to +configure the selected provider for the new chat: Configuration value not found: +OPENAI_API_KEY* means the model chosen with `biorouter configure` has no credential on the serving +machine. Fix it there, restart `serve`, and open the new address it prints. **Copy error** on the +notice copies the daemon's own words, for a bug report. + **The tab says the link needs its access token.** The `?t=` part was dropped — from a copy-paste, a chat client shortening the link, or a bookmark saved after the redirect. Use the full address as printed. If the launch has since restarted, the token has changed; read the new one from the diff --git a/docs/deployment/programmatic-session-access.md b/docs/deployment/programmatic-session-access.md index 7d976e883..3ce60bb08 100644 --- a/docs/deployment/programmatic-session-access.md +++ b/docs/deployment/programmatic-session-access.md @@ -53,7 +53,10 @@ case with its own logic — it is that rule, stated on a request. The header is read by [`crates/biorouter-server/src/routes/session_reach.rs`](../../crates/biorouter-server/src/routes/session_reach.rs); `biorouter session watch`, `send` and `attach` already send it, which is why those commands reach a -private chat from a terminal that can never prove a human is present. +private chat from a terminal that can never prove a human is present. So does the browser interface +`biorouter serve` serves, naming the model the host was configured with +([SD-9](serve-decisions.md#sd-9--a-new-chat-starts-on-the-operators-model-without-a-proof-and-nothing-else-does)): +a browser, like a terminal, can never carry the proof, and runs the model its host chose. ## What the header is *not* @@ -67,9 +70,13 @@ which got the answer backwards in both directions: a terminal running an institu refused, while the desktop app was admitted for the same chat while running a public one. **It is not a way to raise or lower a tier.** Reaching a chat and *reclassifying* one are separate -decisions. Raising a session's classification, declassifying it, and binding a private model all -still require proof that the person at the keyboard acted (`X-User-Action`), and no header changes -that. A capability is a fact about a model; neither of those is a decision a model may make. +decisions. Raising a session's classification, declassifying it, and binding a private model to a +chat all still require proof that the person at the keyboard acted (`X-User-Action`), and no header +changes that. A capability is a fact about a model; neither of those is a decision a model may make. +The one bind that needs no proof is not a header's doing either: on a daemon with no user-action +key, a **new** chat starts on the model the operator configured, private or not, because choosing +it with `biorouter configure` was the decision +([SD-9](serve-decisions.md#sd-9--a-new-chat-starts-on-the-operators-model-without-a-proof-and-nothing-else-does)). **It is not a per-request opt-out.** There is no header that turns the gate off. The only machine-wide switch is the privacy master switch, which lives in its own record beside diff --git a/docs/deployment/serve-decisions.md b/docs/deployment/serve-decisions.md index f413d2bef..92205c5b5 100644 --- a/docs/deployment/serve-decisions.md +++ b/docs/deployment/serve-decisions.md @@ -17,7 +17,8 @@ This page records the decisions that replaced that arrangement. They were taken several of them only make sense as a set: the reason a browser session cannot switch models (SD-1) is also the reason it needs no proof-of-user mechanism, which is the reason the daemon can be spawned with a closed stdin (SD-7) — and the reason every control that needs that proof -must say so before the user reaches for it (SD-8). Read [the architecture](serve-architecture.md) +must say so before the user reaches for it (SD-8), and the reason the one model the operator +chose must not need that proof at all (SD-9). Read [the architecture](serve-architecture.md) for how the result is built, and [browser access](browser-access.md) for how to use it. Records are identified `SD-n` — *serve decision*. The numbering is stable; a superseded record @@ -27,7 +28,7 @@ keeps its number and says what replaced it. ## SD-1 — A browser session cannot change its model or provider, and that is the point -**Ruling.** `POST /config/provider` continues to refuse a request that carries no proof a human +**Ruling.** `POST /config/set_provider` continues to refuse a request that carries no proof a human made it. Browser-served Biorouter installs no such proof. A browser session therefore runs whatever provider and model the machine was already configured with, and the model picker is inert. @@ -46,6 +47,12 @@ anyone opens a tab — and the tier that choice implies holds for every session A run started against an institutional Bedrock model is private for its whole life; one started against a commercial model is public for its whole life. Neither can drift. +> ⚠ **The first half of that sentence was unreachable until SD-9.** A `serve` daemon holds no +> user-action key, and the new-chat bind asked for that key's proof before binding a private +> model — so with an institutional model configured, no chat could be started at all, and the +> interface showed nothing (the 2026-09-10 QA run, finding F1). See +> [SD-9](#sd-9--a-new-chat-starts-on-the-operators-model-without-a-proof-and-nothing-else-does). + **Displaced alternatives.** - *Mint a digest scoped to a loopback bind.* Rejected: it makes the guarantee depend on the @@ -227,6 +234,96 @@ can never half-believe a person is reachable. --- +## SD-9 — A new chat starts on the operator's model without a proof, and nothing else does + +**Ruling.** On a daemon that holds no user-action key — the one `biorouter serve` starts (SD-7), +or a `biorouterd` started by hand — `POST /agent/start` binds the operator's configured provider +to the new chat without asking for proof of a person, whether that provider is public or private. +Three things hold beside it: + +- On that daemon, `POST /agent/update_provider` refuses every move onto a private model, whatever + the chat runs on now. The configured model is the only private model a chat there can reach. +- A daemon that holds a key — the desktop application's — is unchanged. Its renderer sends the + proof on every start, and a start that lacks it is refused as before. +- The browser interface states the host's configured model on the requests that reach into a chat + (`X-Caller-Provider`), the way `biorouter session` already does from a terminal, so a chat its + first reply made private stays reachable from the tab that started it. + +**Why.** The configured model is the person's decision, made out of band. `/agent/start` names no +provider: it binds `BIOROUTER_PROVIDER`, which only a proven person may write over HTTP, or which +the operator wrote at the terminal with `biorouter configure`. Open question 24 of the privacy plan +already put the raise at that write — *a raise of every future session* — and SD-1 already says the +tier that choice implies *holds for every session in that daemon*. A new chat taking that tier is +the choice being honored, not a switch. DR-16 governs raising a chat that exists, and a chat that +did not exist a moment ago has nothing to raise. + +On a daemon with no key, asking for the proof can only refuse everyone. The 2026-09-10 QA run +measured it: with an institutional model configured, every new chat on a `serve` daemon was +refused 409, in a sentence written for a model that pointed at a model picker SD-1 disables, and +the interface showed nothing at all. A control nobody can pass is not a boundary; it is the +product not working. + +The other two halves close what the exemption would otherwise open. Without the rule on +`/agent/update_provider`, a chat bound to the private default could be moved to a different private +model — `Private → Private`, which DR-16's raise predicate calls sideways and allows — that nobody +configured. Without the capability statement, the tab that started a chat lost it after one reply: +the reply makes the chat private (the classification ratchets on the turn, never on the bind), and a +keyless daemon reaches a private chat only for a caller whose stated capability covers it. Measured: +the chat's next request answered 403 with nothing stated, and 200 with the host's provider stated. + +**Why not on every daemon.** On a daemon that holds a key the proof costs the person nothing — the +renderer attaches it to every start — and it still refuses a caller that cannot present it. That +includes a model holding the daemon secret, which AR-11 found recoverable and which could otherwise +mint a private-capability chat through `/agent/start` with an extension set of its own choosing. +Relaxing the gate there buys the person nothing and gives that model something. + +**Who can do this, and what else reaches the same place.** The two questions every privacy +control answers in writing ([privacy tiers §3.1](../security/privacy-tiers.md)): + +- *Who can initiate it.* On a keyless daemon, anything holding the daemon secret: the person in + the browser and, indistinguishably, a model running in a chat on that daemon that has recovered + the secret. Both get the configured model and nothing else. +- *What else reaches a chat running on the configured private model:* + + | Door | Proof asked | Changed here | + |---|---|---| + | `workspace_open { new: … }` — binds the machine default through `restore_provider_from_session` | None, on every daemon (privacy tiers, "Did not ship") | No | + | `POST /agent/restart` on a row that names no provider — `restore_provider_from_session` falls back to the configured default | None | No | + | An app session's creation bind (DR-21) | None, deliberately | No | + | `POST /agent/update_provider` onto a private model | The proof; on a keyless daemon, refused outright | Yes | + | `POST /config/set_provider`, and `/config/upsert` or `/config/remove` on a capability key | The proof (SD-1, open question 24) | No — the configured model stays the operator's to choose | + +**Displaced alternatives.** + +- *Keep the refusal, and explain it in the interface.* Rejected. SD-8's explanation is for a + control that can never work; this one is the product's core. A `serve` deployment whose only + model is institutional would be a chat application that cannot chat. +- *Exempt every new chat, whatever provider it asks for.* Rejected. The operator's choice is what + makes the bind legitimate, so a provider the request picked would be a switch. `/agent/start` + names none today, and a field that ever let it name one must not inherit this exemption. +- *Exempt the configured model on every daemon.* Rejected; see *Why not on every daemon*. +- *Let a keyless daemon treat its configured model as the capability of any request that states + none.* Rejected. An absent header resolving to Public is the fail-safe the reach gate is built on, + and a default that raised it would speak for every caller rather than for the client that says + what it runs. + +**Consequence to accept.** On a keyless daemon whose configured model is private, a model running +in a chat on that daemon — a public-model chat resumed from the shared session store — that has +recovered the daemon secret can start a private-capability chat through `/agent/start` with +extensions it chose, and can reach private chats by stating the host's provider. It could already +do the first through `workspace_open { new }`, and the second by spelling a provider name (the +header is not authentication, as `session_reach.rs` records); and the filesystem read-deny that +would stop it carrying anything back out did not ship. It is recorded rather than closed: on a +daemon that cannot tell a person from a model, closing it means refusing the person. + +And one visible change: a browser tab on a host configured with a private model now opens private +chats started in the desktop application on the same machine, which it was refused before. That is +the reach rule — *the caller's capability must be at least the chat's classification* — admitting +it, exactly as it admits `biorouter session` configured with the same model. On a host configured +with a public model nothing changes, and private chats stay out of the browser's reach. + +--- + ## Related documentation - [Architecture of the serving path](serve-architecture.md) — how the decisions above are built. diff --git a/ui/desktop/openapi.json b/ui/desktop/openapi.json index 1a66a7468..0b269a025 100644 --- a/ui/desktop/openapi.json +++ b/ui/desktop/openapi.json @@ -689,7 +689,7 @@ "description": "Unauthorized - invalid secret key" }, "409": { - "description": "The selected private provider requires user-action proof", + "description": "The configured provider is private and this daemon holds a user-action key, but the request carried no proof it came from the user (SD-9). A daemon with no user-action key binds its configured provider without one.", "content": { "application/json": { "schema": { @@ -881,7 +881,7 @@ "description": "The session is out of reach, or the target is a subagent and the request lacks user-action proof" }, "409": { - "description": "Refused by a privacy boundary (issue #56). Gate A: a public model cannot be bound to a private chat (body = PrivacyBarrierBody). DR-16: the bind raises this chat's capability to Private and the request carried no proof it came from the user (body = plain text)", + "description": "Refused by a privacy boundary (issue #56). Gate A: a public model cannot be bound to a private chat (body = PrivacyBarrierBody). DR-16: the bind raises this chat's capability to Private and the request carried no proof it came from the user; on a daemon with no user-action key, any bind to a private model (SD-9) (body = plain text)", "content": { "application/json": { "schema": { diff --git a/ui/desktop/src/api/types.gen.ts b/ui/desktop/src/api/types.gen.ts index 68b657939..82535f4d0 100644 --- a/ui/desktop/src/api/types.gen.ts +++ b/ui/desktop/src/api/types.gen.ts @@ -4719,7 +4719,7 @@ export type StartAgentErrors = { */ 401: unknown; /** - * The selected private provider requires user-action proof + * The configured provider is private and this daemon holds a user-action key, but the request carried no proof it came from the user (SD-9). A daemon with no user-action key binds its configured provider without one. */ 409: ErrorResponse; /** @@ -4872,7 +4872,7 @@ export type UpdateAgentProviderErrors = { */ 403: unknown; /** - * Refused by a privacy boundary (issue #56). Gate A: a public model cannot be bound to a private chat (body = PrivacyBarrierBody). DR-16: the bind raises this chat's capability to Private and the request carried no proof it came from the user (body = plain text) + * Refused by a privacy boundary (issue #56). Gate A: a public model cannot be bound to a private chat (body = PrivacyBarrierBody). DR-16: the bind raises this chat's capability to Private and the request carried no proof it came from the user; on a daemon with no user-action key, any bind to a private model (SD-9) (body = plain text) */ 409: PrivacyBarrierBody; /** From ddaa856c8b3c2d1781faf4e0855f309ca537063a Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 02:11:00 -0700 Subject: [PATCH 06/75] fix(server): the keyless-daemon warning says what SD-9 still allows It said the daemon refuses every request that raises a session's privacy capability. A new chat binding the configured provider is no longer one of them, so an operator reading it would conclude a private model cannot work. --- crates/biorouter-server/src/commands/agent.rs | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/crates/biorouter-server/src/commands/agent.rs b/crates/biorouter-server/src/commands/agent.rs index e9e784051..b08ae34db 100644 --- a/crates/biorouter-server/src/commands/agent.rs +++ b/crates/biorouter-server/src/commands/agent.rs @@ -124,8 +124,9 @@ pub async fn run() -> Result<()> { let user_action_digest = read_user_action_digest().await; if user_action_digest.is_none() { tracing::warn!( - "no user-action key on stdin: this daemon will refuse every request that raises a \ - session's privacy capability, including one made by the person at the keyboard" + "no user-action key on stdin: this daemon will refuse every request that raises an \ + existing chat's privacy capability, including one made by the person at the \ + keyboard; a new chat still starts on the configured provider (SD-9)" ); } // A tool whose approval can never be granted must not be offered. `serve` From 3b1336500ae0cf91462775396b4867bd912cdae1 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 02:28:22 -0700 Subject: [PATCH 07/75] style(desktop): prettier on the surface test --- ui/desktop/src/utils/userAction.surface.test.ts | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/ui/desktop/src/utils/userAction.surface.test.ts b/ui/desktop/src/utils/userAction.surface.test.ts index 70851962e..7254a5d84 100644 --- a/ui/desktop/src/utils/userAction.surface.test.ts +++ b/ui/desktop/src/utils/userAction.surface.test.ts @@ -5,11 +5,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; const { mockReadConfig } = vi.hoisted(() => ({ mockReadConfig: vi.fn() })); vi.mock('../api', () => ({ readConfig: mockReadConfig })); -import { - CALLER_PROVIDER_HEADER, - resetHostProviderForTests, - userActionHeaders, -} from './userAction'; +import { CALLER_PROVIDER_HEADER, resetHostProviderForTests, userActionHeaders } from './userAction'; import { BROWSER_SURFACE_MARKER } from './surface'; /** From 7ea48e73ac58ef822738d9edae94f9ae6fc17a79 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 02:29:47 -0700 Subject: [PATCH 08/75] style(server): rustfmt the SD-9 tests --- crates/biorouter-server/src/routes/agent.rs | 23 ++++++++++--- .../tests/new_chat_no_user_key.rs | 34 ++++++++++++++----- 2 files changed, 44 insertions(+), 13 deletions(-) diff --git a/crates/biorouter-server/src/routes/agent.rs b/crates/biorouter-server/src/routes/agent.rs index 629edf7e9..36ace3292 100644 --- a/crates/biorouter-server/src/routes/agent.rs +++ b/crates/biorouter-server/src/routes/agent.rs @@ -3188,8 +3188,16 @@ mod new_session_provider_binding_tests { #[test] fn only_a_daemon_that_can_check_a_proof_asks_a_new_chat_for_one() { use UserActionProof::{NoKeyInstalled, Proven, Unproven}; - assert!(!new_chat_bind_needs_user(true, ProviderTier::Private, Proven)); - assert!(new_chat_bind_needs_user(true, ProviderTier::Private, Unproven)); + assert!(!new_chat_bind_needs_user( + true, + ProviderTier::Private, + Proven + )); + assert!(new_chat_bind_needs_user( + true, + ProviderTier::Private, + Unproven + )); assert!( !new_chat_bind_needs_user(true, ProviderTier::Private, NoKeyInstalled), "a keyless daemon refusing its own configured default refuses every person, always" @@ -3198,7 +3206,11 @@ mod new_session_provider_binding_tests { // A public default raises nothing, for anyone. assert!(!new_chat_bind_needs_user(true, ProviderTier::Public, proof)); // DR-15's master opt-out turns the gate off, not the question. - assert!(!new_chat_bind_needs_user(false, ProviderTier::Private, proof)); + assert!(!new_chat_bind_needs_user( + false, + ProviderTier::Private, + proof + )); } } @@ -3209,7 +3221,10 @@ mod new_session_provider_binding_tests { fn a_keyless_daemon_has_no_private_floor_for_a_switch_to_build_on() { use UserActionProof::{NoKeyInstalled, Proven, Unproven}; for current in [ProviderTier::Private, ProviderTier::Public] { - assert_eq!(raise_baseline(current, NoKeyInstalled), ProviderTier::Public); + assert_eq!( + raise_baseline(current, NoKeyInstalled), + ProviderTier::Public + ); // A daemon that can check a proof keeps measuring from the live binding. assert_eq!(raise_baseline(current, Proven), current); assert_eq!(raise_baseline(current, Unproven), current); diff --git a/crates/biorouter-server/tests/new_chat_no_user_key.rs b/crates/biorouter-server/tests/new_chat_no_user_key.rs index 4eb7f8264..ae168e9ea 100644 --- a/crates/biorouter-server/tests/new_chat_no_user_key.rs +++ b/crates/biorouter-server/tests/new_chat_no_user_key.rs @@ -135,7 +135,11 @@ async fn a_keyless_daemon_starts_a_new_chat_on_its_configured_private_model() { .expect("the started session carries an id") .to_string(); - let row = state.session_manager().get_session(&id, false).await.unwrap(); + let row = state + .session_manager() + .get_session(&id, false) + .await + .unwrap(); assert_eq!(row.provider_name.as_deref(), Some("versa_azure")); assert_eq!( row.model_config.map(|config| config.model_name).as_deref(), @@ -188,10 +192,7 @@ async fn a_keyless_daemon_will_not_move_a_new_chat_to_a_private_model_nobody_con // A loopback Ollama is Private (`self_hosted_tier`), and constructing one // opens no connection, so port 1 is never dialled. let (status, body) = with_config_overrides( - HashMap::from([( - "OLLAMA_HOST".to_string(), - "http://127.0.0.1:1".to_string(), - )]), + HashMap::from([("OLLAMA_HOST".to_string(), "http://127.0.0.1:1".to_string())]), post_json( biorouter_server::routes::agent::routes(Arc::clone(&state)), "/agent/update_provider", @@ -210,7 +211,11 @@ async fn a_keyless_daemon_will_not_move_a_new_chat_to_a_private_model_nobody_con "refused, but not by the tier gate: {body}" ); - let row = state.session_manager().get_session(&id, false).await.unwrap(); + let row = state + .session_manager() + .get_session(&id, false) + .await + .unwrap(); assert_eq!( row.provider_name.as_deref(), Some("versa_azure"), @@ -260,7 +265,10 @@ async fn the_first_turn_on_a_keyless_default_chat_ratchets_it_as_usual() { }; let sse = format!( "data: {}\n\ndata: {}\n\ndata: [DONE]\n\n", - chunk(json!({ "role": "assistant", "content": "ready" }), Value::Null), + chunk( + json!({ "role": "assistant", "content": "ready" }), + Value::Null + ), chunk(json!({ "content": "" }), json!("stop")), ); Mock::given(method("POST")) @@ -308,7 +316,11 @@ async fn the_first_turn_on_a_keyless_default_chat_ratchets_it_as_usual() { .as_str() .unwrap() .to_string(); - let before = state.session_manager().get_session(&id, false).await.unwrap(); + let before = state + .session_manager() + .get_session(&id, false) + .await + .unwrap(); assert_eq!(before.provider_name.as_deref(), Some("ollama")); assert_eq!(before.privacy_tier, SessionClassification::Public); @@ -329,7 +341,11 @@ async fn the_first_turn_on_a_keyless_default_chat_ratchets_it_as_usual() { "the turn did not run on the stub: {stream}" ); - let after = state.session_manager().get_session(&id, false).await.unwrap(); + let after = state + .session_manager() + .get_session(&id, false) + .await + .unwrap(); assert_eq!( after.privacy_tier, SessionClassification::Private, From 0dcc5095fbebd0ae53f6253fca33a8115ce6cb30 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 02:30:50 -0700 Subject: [PATCH 09/75] fix(privacy): one reach gate for every HTTP route that names a chat or a knowledge base QA on merged main 7c96d796 (2026-09-10) measured four places where a caller holding nothing but the daemon secret - which a public chat's shell recovers with ps eww - was answered by the daemon while the tool path refused it: - H2: every /knowledge/bases/{id}/... route served a private base's pages, graph, history, location and a .brkb export; GET /knowledge/bases listed it. - M1: GET /sessions returned every chat on the machine, private ones titled. - M2: GET /agent/tools?session_id= returned private-extension tool names while add_extension on the same chat refused. - F0: DELETE /sessions/{id} deleted a private chat the read refused, 4 of 4. Each is now the singular read's own gate (session_reach's pure decision), composed rather than re-derived: - Knowledge bases: one route_layer (session_reach::gate_knowledge_base) on a sub-router holding exactly the routes that name a base by {id}, reads and writes alike. An absent or malformed id is answered as a private one. The bases list and /knowledge/active omit what the caller cannot reach, and a selection write cannot move what its caller cannot see (KnowledgeService::set_selection_within). - Chats: session_reach on DELETE, rename, workflow values, the in-place edit arm (it truncates), extensions, usage, /agent/tools, callable_tool_count, /workflows/create and /skills/session; every refusal is GET /sessions/{id}'s byte for byte. GET /sessions, /sessions/sidebar and /schedule/{id}/sessions filter to the rows that gate admits (the sidebar scans so paging stays whole). Conversation ingest checks every named chat. - biorouter serve: its own interface (the served document's cookie) keeps the operator's configured-provider tier on listings and knowledge bases, which were open to it before; the transcript gate never reads it. Nothing refused before is permitted now. Tests show each route failing before and passing after; the wiring census names every new call site. --- crates/biorouter-mcp/src/knowledge/service.rs | 195 ++- crates/biorouter-server/src/auth.rs | 87 ++ crates/biorouter-server/src/commands/agent.rs | 43 + crates/biorouter-server/src/routes/agent.rs | 55 + .../biorouter-server/src/routes/knowledge.rs | 147 +- .../biorouter-server/src/routes/schedule.rs | 22 +- crates/biorouter-server/src/routes/session.rs | 328 ++++- .../src/routes/session_reach.rs | 1273 ++++++++++++++++- crates/biorouter-server/src/routes/skills.rs | 17 + crates/biorouter-server/src/routes/web_ui.rs | 35 +- .../biorouter-server/src/routes/workflow.rs | 58 +- .../tests/knowledge_routes.rs | 653 ++++++++- .../tests/serve_operator_reach.rs | 193 +++ .../biorouter/tests/privacy_guard_wiring.rs | 179 ++- 14 files changed, 3129 insertions(+), 156 deletions(-) create mode 100644 crates/biorouter-server/tests/serve_operator_reach.rs diff --git a/crates/biorouter-mcp/src/knowledge/service.rs b/crates/biorouter-mcp/src/knowledge/service.rs index f71d85bb4..0a27ad874 100644 --- a/crates/biorouter-mcp/src/knowledge/service.rs +++ b/crates/biorouter-mcp/src/knowledge/service.rs @@ -4091,7 +4091,80 @@ impl KnowledgeService { primary: PrimaryUpdate<'_>, ) -> anyhow::Result { let _lock = self.lock_root()?; - self.apply_selection_unlocked(session_id, hidden, primary) + self.apply_selection_unlocked(session_id, hidden, primary, &|_| true) + } + + /// [`Self::set_selection`] for a caller that cannot reach every base (issue + /// #56, QA 2026-09-10 H2): **a caller changes only what it can see.** + /// + /// `reachable` is the caller's reach, decided by the daemon's HTTP gate. + /// For a caller that reaches everything it admits every id and this is + /// exactly [`Self::set_selection`]. For one that does not: + /// + /// * `hidden` is taken literally for the bases `reachable` admits, and every + /// base it does not admit keeps the state it already had in this scope — + /// neither hidden nor revealed by a list its caller was never shown. That + /// is the case that matters: a renderer prunes ids missing from the list + /// it was given, and a filtered list would otherwise un-hide every private + /// base on the machine as a side effect of one click. + /// * `Clear` is a no-op when the scope's effective primary is a base the + /// caller cannot reach. It was shown no primary, so it asked to clear none. + /// `Inherit` likewise leaves a pin this scope holds on such a base: it + /// would drop a choice the caller was never shown. + /// * `Set(id)` naming a base the caller cannot reach is refused. The route + /// answers that case first, with the gate's own refusal; this is the + /// backstop, and it names nothing. + /// * A refusal's list of available bases names only reachable ones. + /// + /// One root lock across the read of the stored state and the write, so the + /// merge cannot interleave with another writer (see [`Self::hide_kb`] for + /// why a read-modify-write across two calls loses an edit). + pub fn set_selection_within( + &self, + session_id: Option<&str>, + hidden: Option<&[String]>, + primary: PrimaryUpdate<'_>, + reachable: &dyn Fn(&str) -> bool, + ) -> anyhow::Result { + let _lock = self.lock_root()?; + let hidden = match hidden { + None => None, + Some(submitted) => { + let mut next = Self::sanitize_kb_id_list(submitted)? + .into_iter() + .filter(|id| reachable(id)) + .collect::>(); + next.extend( + self.hidden_for_scope_unlocked(session_id)? + .into_iter() + .filter(|id| !reachable(id)), + ); + Some(next) + } + }; + let primary = match primary { + PrimaryUpdate::Set(id) if !reachable(id) => { + anyhow::bail!("knowledge base '{id}' is not available") + } + PrimaryUpdate::Clear | PrimaryUpdate::Inherit => { + let own = + self.read_primary_file_unlocked(&self.primary_path_for_scope(session_id))?; + // `Clear` is judged against the pointer the scope is USING and + // `Inherit` against the one it HOLDS: clearing hides what is + // shown, and inheriting drops only this scope's own pin. + let judged = match primary { + PrimaryUpdate::Clear => self.effective_primary_unlocked(&own, session_id)?, + _ => own, + }; + if judged.pinned().is_some_and(|id| !reachable(id)) { + PrimaryUpdate::Unchanged + } else { + primary + } + } + other => other, + }; + self.apply_selection_unlocked(session_id, hidden.as_deref(), primary, reachable) } /// Drop one base from this scope's set, in one root-locked step. @@ -4118,7 +4191,7 @@ impl KnowledgeService { if !hidden.iter().any(|id| id == kb_id) { hidden.push(kb_id.to_string()); } - self.apply_selection_unlocked(session_id, Some(&hidden), primary) + self.apply_selection_unlocked(session_id, Some(&hidden), primary, &|_| true) } /// Add one base to this scope's set (un-hide it), in one root-locked step. @@ -4151,7 +4224,7 @@ impl KnowledgeService { .into_iter() .filter(|id| id != kb_id) .collect::>(); - self.apply_selection_unlocked(session_id, Some(&hidden), primary) + self.apply_selection_unlocked(session_id, Some(&hidden), primary, &|_| true) } /// Set this scope's set from the ids that should be **visible** — the @@ -4176,7 +4249,7 @@ impl KnowledgeService { .into_iter() .filter(|id| !visible.contains(id)) .collect::>(); - self.apply_selection_unlocked(session_id, Some(&hidden), primary) + self.apply_selection_unlocked(session_id, Some(&hidden), primary, &|_| true) } /// The engine behind every selection write: decide, validate, *then* write. @@ -4190,11 +4263,18 @@ impl KnowledgeService { /// "commit" line can fail on anything but I/O. /// /// Callers must already hold the root lock. + /// + /// `listed` decides which bases a refusal may NAME when it lists what is + /// available: every base for the in-process callers, the reachable ones for + /// an HTTP caller that cannot reach them all (see + /// [`Self::set_selection_within`]). A refusal that enumerated the rest would + /// hand over the ids the caller was just refused. fn apply_selection_unlocked( &self, session_id: Option<&str>, hidden: Option<&[String]>, primary: PrimaryUpdate<'_>, + listed: &dyn Fn(&str) -> bool, ) -> anyhow::Result { // ---- decide: touch nothing on disk until every branch has succeeded ---- let installed = self.installed_kb_ids_unlocked()?; @@ -4225,10 +4305,15 @@ impl KnowledgeService { PrimaryUpdate::Inherit => Some(StoredPrimary::Inherit), PrimaryUpdate::Set(id) => { if !next_ids.iter().any(|known| known == id) { - let available = if next_ids.is_empty() { + let named = next_ids + .iter() + .filter(|known| listed(known)) + .map(String::as_str) + .collect::>(); + let available = if named.is_empty() { "none".to_string() } else { - next_ids.join(", ") + named.join(", ") }; // Scope-appropriate vocabulary: the CLI and scheduled jobs // pass `None` and have no session concept at all (D11), so @@ -7320,6 +7405,104 @@ mod tests { Ok(()) } + /// Issue #56, QA 2026-09-10 H2: a caller that cannot reach every base + /// changes only the bases it can. Each rule is driven against the one a + /// plausible wrong implementation would break — taking the submitted set + /// literally, clearing a primary it was never shown, dropping a pin it could + /// not see, and naming the rest of the machine's bases in a refusal. + #[test] + fn a_limited_caller_changes_only_what_it_can_see() -> anyhow::Result<()> { + let tmp = tempfile::TempDir::new()?; + let svc = KnowledgeService::new(tmp.path().to_path_buf()); + for id in ["alpha", "beta", "secret"] { + svc.create_base(id, id, None)?; + } + let sees = |id: &str| id != "secret"; + + // The user pins `secret` as this chat's primary, with nothing hidden. + svc.set_selection(Some("s1"), Some(&[]), PrimaryUpdate::Set("secret"))?; + + // A limited caller rewrites the set naming only what it saw: `secret` + // stays visible (it was not hidden) and `beta` is hidden as asked. + let sel = svc.set_selection_within( + Some("s1"), + Some(&["beta".to_string()]), + PrimaryUpdate::Unchanged, + &sees, + )?; + assert_eq!(sel.hidden_kbs, vec!["beta".to_string()]); + assert_eq!(sel.primary_kb.as_deref(), Some("secret")); + + // It asks to clear the primary it was shown as none: `secret` stays. + let sel = svc.set_selection_within(Some("s1"), None, PrimaryUpdate::Clear, &sees)?; + assert_eq!( + sel.primary_kb.as_deref(), + Some("secret"), + "cleared a hidden primary" + ); + // …and to inherit, which would drop this chat's pin on `secret`: stays. + let sel = svc.set_selection_within(Some("s1"), None, PrimaryUpdate::Inherit, &sees)?; + assert_eq!( + sel.primary_kb.as_deref(), + Some("secret"), + "dropped a hidden pin" + ); + + // It may not hide `secret` by naming it, nor pin it. + let sel = svc.set_selection_within( + Some("s1"), + Some(&["secret".to_string()]), + PrimaryUpdate::Unchanged, + &sees, + )?; + assert!( + sel.hidden_kbs.is_empty(), + "hid a base it could not see: {sel:?}" + ); + let err = svc + .set_selection_within(Some("s1"), None, PrimaryUpdate::Set("secret"), &sees) + .unwrap_err() + .to_string(); + assert!(!err.contains("alpha") && !err.contains("beta"), "{err}"); + + // The user hides `secret`; a limited caller that "un-hides everything" + // leaves it hidden. + svc.set_selection( + Some("s1"), + Some(&["secret".to_string()]), + PrimaryUpdate::Set("alpha"), + )?; + let sel = + svc.set_selection_within(Some("s1"), Some(&[]), PrimaryUpdate::Unchanged, &sees)?; + assert_eq!(sel.hidden_kbs, vec!["secret".to_string()]); + + // A refusal names only what the caller can see: hiding `beta` while + // pinning it fails, and the list of what IS available omits `secret` + // even though `secret` is not hidden from the set at this point. + svc.set_selection(Some("s1"), Some(&[]), PrimaryUpdate::Set("alpha"))?; + let err = svc + .set_selection_within( + Some("s1"), + Some(&["beta".to_string()]), + PrimaryUpdate::Set("beta"), + &sees, + ) + .unwrap_err() + .to_string(); + assert!(err.contains("alpha"), "{err}"); + assert!( + !err.contains("secret"), + "a refusal named a base the caller cannot see: {err}" + ); + + // A caller that sees everything is `set_selection`, byte for byte. + let everything = |_: &str| true; + let sel = + svc.set_selection_within(Some("s1"), Some(&[]), PrimaryUpdate::Clear, &everything)?; + assert_eq!(sel.primary_kb, None); + Ok(()) + } + /// The membership primitives every caller actually needs, so none of them /// has to read the hidden list, edit it and write it back. Each takes the /// whole gesture and applies it under one root lock. diff --git a/crates/biorouter-server/src/auth.rs b/crates/biorouter-server/src/auth.rs index 29470f71e..b89b5b015 100644 --- a/crates/biorouter-server/src/auth.rs +++ b/crates/biorouter-server/src/auth.rs @@ -127,6 +127,78 @@ pub fn is_user_action(headers: &axum::http::HeaderMap) -> bool { matches!(user_action_proof(headers), UserActionProof::Proven) } +/// The standing a `biorouter serve` daemon gives its OWN web interface (issue +/// #56, the QA follow-up of 2026-09-10 that closed H2 and M1). +/// +/// A serve daemon holds no user-action digest (SD-7) and pins the provider for +/// every session it runs (SD-1), so the tier the operator's configured provider +/// implies is the only capability its interface can be said to have. This keeps +/// that tier beside the browser token whose cookie marks a request as coming +/// from the document this daemon served — which is how a request from the +/// operator's browser is told from one that merely holds the secret. +/// +/// ⚠ **It widens nothing that was refused.** It is read only by the listing and +/// knowledge-base gates in `routes::session_reach`, which were open to this +/// interface before they existed; the transcript gate, `session_reach` itself, +/// never reads it. A serve daemon's browser therefore keeps exactly the reach it +/// had, and a caller holding only the secret loses it. +/// +/// ⚠ **Not authentication, and not a proof of a person.** `biorouter serve` +/// hands this daemon the token in its environment, beside the secret, so a +/// caller that can read one can read the other — the residual `X-Caller-Provider` +/// already carries (#47). It never satisfies a proof-of-user check: SD-1 and +/// SD-8 stand exactly as they were. +struct ServedOperator { + browser_token: String, + capability: biorouter::privacy::ProviderTier, +} + +static SERVED_OPERATOR: OnceLock = OnceLock::new(); + +/// Record a serve daemon's operator standing. Called once, from +/// `commands::agent::run`, and only when the web interface is served behind a +/// browser token: a `--no-token` daemon cannot tell its own interface from any +/// other local caller, so it gives none. +pub fn install_served_operator( + browser_token: String, + capability: biorouter::privacy::ProviderTier, +) { + let _ = SERVED_OPERATOR.set(ServedOperator { + browser_token, + capability, + }); +} + +/// The capability a request earns by presenting the served document's cookie: +/// the operator's tier on a serve daemon, `Public` for every other request on +/// every other daemon. +pub fn served_operator_capability( + headers: &axum::http::HeaderMap, +) -> biorouter::privacy::ProviderTier { + match SERVED_OPERATOR.get() { + Some(operator) + if served_document_matches( + crate::routes::web_ui::session_cookie(headers), + &operator.browser_token, + ) => + { + operator.capability + } + _ => biorouter::privacy::ProviderTier::Public, + } +} + +/// Does the presented cookie carry the served document's token? +/// +/// Pure, so the rule is testable without the process global; compared without +/// an early return, the same way the secret is. An empty token matches nothing. +pub fn served_document_matches(presented: Option<&str>, browser_token: &str) -> bool { + match presented { + Some(presented) if !browser_token.is_empty() => secret_matches(presented, browser_token), + _ => false, + } +} + fn get_failed_attempts() -> &'static Mutex>> { FAILED_ATTEMPTS.get_or_init(|| Mutex::new(HashMap::new())) } @@ -704,6 +776,21 @@ mod tests { assert!(!is_unauthenticated_path("/tool_bridgeX/abc")); } + /// The serve daemon's operator standing is earned by the served document's + /// cookie and by nothing else: the whole token, not a prefix; not an empty + /// one; not an absent one. + #[test] + fn only_the_served_documents_cookie_earns_the_operator_standing() { + use super::served_document_matches; + assert!(served_document_matches(Some("0123abcd"), "0123abcd")); + assert!(!served_document_matches(Some("0123abc"), "0123abcd")); + assert!(!served_document_matches(Some(""), "0123abcd")); + assert!(!served_document_matches(None, "0123abcd")); + // An empty token is "no token", and "no token" earns nothing — never + // the equality of two empty strings. + assert!(!served_document_matches(Some(""), "")); + } + #[test] fn secret_compare_is_exact() { assert!(secret_matches("abc", "abc")); diff --git a/crates/biorouter-server/src/commands/agent.rs b/crates/biorouter-server/src/commands/agent.rs index e9e784051..75f42d6cd 100644 --- a/crates/biorouter-server/src/commands/agent.rs +++ b/crates/biorouter-server/src/commands/agent.rs @@ -70,6 +70,34 @@ async fn read_user_action_digest() -> Option<[u8; 32]> { <[u8; 32]>::try_from(bytes.as_slice()).ok() } +/// The tier SD-1 pins for every session a serve daemon runs: the DECLARED tier +/// of the provider the operator configured, reduced with `least` over the lead +/// provider when a lead model is configured — the reduction a bound lead/worker +/// pair gets, since its transcript reaches both. +/// +/// Read ONCE, at launch. The operator made this choice at the terminal before +/// anyone opened a tab (SD-1), and `config.yaml` is agent-writable (DR-17), so a +/// value re-read per request would be one a model could raise by editing a file. +/// Unconfigured, and a name this install does not publish, both read Public — +/// the fail-safe side, and the reach this interface had for every private chat +/// before it had any. +async fn served_operator_capability() -> biorouter::privacy::ProviderTier { + use biorouter::privacy::ProviderTier; + use biorouter::workflow::privacy::declared_provider_tier; + let config = biorouter::config::Config::global(); + let Ok(provider) = config.get_biorouter_provider() else { + return ProviderTier::Public; + }; + let mut capability = declared_provider_tier(&provider).await; + if config.get_param::("BIOROUTER_LEAD_MODEL").is_ok() { + let lead = config + .get_param::("BIOROUTER_LEAD_PROVIDER") + .unwrap_or_else(|_| provider.clone()); + capability = ProviderTier::least(capability, declared_provider_tier(&lead).await); + } + capability +} + pub async fn run() -> Result<()> { crate::logging::setup_logging(Some("biorouterd"))?; @@ -173,6 +201,21 @@ pub async fn run() -> Result<()> { // there, so its absence here means a loopback bind whose launcher // chose not to require one. let browser_token = std::env::var("BIOROUTER_BROWSER_TOKEN").ok(); + // Issue #56, QA 2026-09-10 (SD-9): the interface this daemon serves is + // the operator's, and SD-1 pins the provider every session here runs + // on — so that provider's tier is the reach the listing and + // knowledge-base gates give a request carrying the served document's + // cookie. Without a token there is no such cookie, and the interface + // cannot be told from any other local caller, so it gets none. + if let Some(token) = browser_token.as_deref().filter(|t| !t.is_empty()) { + let capability = served_operator_capability().await; + info!( + ?capability, + "the served interface is given the configured provider's tier on listings \ + and knowledge bases" + ); + biorouter_server::auth::install_served_operator(token.to_string(), capability); + } let ui = crate::routes::web_ui::WebUi::new(&web_dir, &secret_key, browser_token) .map_err(|e| { anyhow::anyhow!( diff --git a/crates/biorouter-server/src/routes/agent.rs b/crates/biorouter-server/src/routes/agent.rs index 8dc1604f8..ab588e1f3 100644 --- a/crates/biorouter-server/src/routes/agent.rs +++ b/crates/biorouter-server/src/routes/agent.rs @@ -1114,6 +1114,10 @@ async fn update_from_session( responses( (status = 200, description = "Tools retrieved successfully", body = Vec), (status = 401, description = "Unauthorized - invalid secret key"), + (status = 403, description = "Refused by a privacy boundary: `session_id` names a chat \ + this caller may not reach, answered with the same refusal, \ + word for word, that `GET /sessions/{session_id}` gives \ + (body = plain text)"), (status = 408, description = "Extension timed out while loading for settings"), (status = 424, description = "Agent not initialized"), (status = 500, description = "Internal server error") @@ -1122,6 +1126,34 @@ async fn update_from_session( async fn get_tools( State(state): State>, Query(query): Query, + headers: axum::http::HeaderMap, +) -> axum::response::Response { + // Issue #56, QA 2026-09-10 M2. Naming a private chat here handed a caller + // holding only the daemon secret that chat's private-extension tool names, + // while `add_extension` on the same chat refused it — and, worse, `get_agent` + // below MINTS an agent for the named chat, loading its extensions, on that + // caller's say-so. So the read's own gate runs first. The comment further + // down, about Gate E, is about which tools a MODEL is shown; this is about + // whether the CALLER may address the chat at all, and the empty id — the + // settings page's one global extension — names no chat and is not gated. + if !query.session_id.is_empty() { + if let Err(refusal) = crate::routes::session_reach::session_reach( + state.session_manager(), + &query.session_id, + &headers, + ) + .await + { + return refusal.into_response(); + } + } + permission_editor_tools(state, query).await.into_response() +} + +/// The body of [`get_tools`], once the caller may address the named chat. +async fn permission_editor_tools( + state: Arc, + query: GetToolsQuery, ) -> Result>, StatusCode> { let config = Config::global(); let biorouter_mode = config.get_biorouter_mode().unwrap_or(BioRouterMode::Auto); @@ -1248,12 +1280,35 @@ async fn get_tools( responses( (status = 200, description = "Model-visible callable tool count", body = CallableToolCountResponse), (status = 401, description = "Unauthorized - invalid secret key"), + (status = 403, description = "Refused by a privacy boundary: the same refusal, word for \ + word, that `GET /sessions/{session_id}` gives (body = plain \ + text)"), (status = 424, description = "Agent not initialized") ) )] async fn get_callable_tool_count( State(state): State>, Query(query): Query, + headers: axum::http::HeaderMap, +) -> axum::response::Response { + // Issue #56, QA 2026-09-10 — M2's sibling: the same named chat, and the + // same agent minted for it below, so the same gate before either. + if let Err(refusal) = crate::routes::session_reach::session_reach( + state.session_manager(), + &query.session_id, + &headers, + ) + .await + { + return refusal.into_response(); + } + model_visible_tool_count(state, query).await.into_response() +} + +/// The body of [`get_callable_tool_count`], once the caller may address the chat. +async fn model_visible_tool_count( + state: Arc, + query: CallableToolCountQuery, ) -> Result, StatusCode> { let session_id = query.session_id; let child_initializing = biorouter::agents::subagent_handle::is_child_initializing(&session_id); diff --git a/crates/biorouter-server/src/routes/knowledge.rs b/crates/biorouter-server/src/routes/knowledge.rs index 46251e6bf..1f15ee648 100644 --- a/crates/biorouter-server/src/routes/knowledge.rs +++ b/crates/biorouter-server/src/routes/knowledge.rs @@ -34,15 +34,16 @@ use utoipa::ToSchema; /// Build the knowledge router. The router owns an `Arc` directly so /// it can be tested without constructing a full `AppState`. +/// +/// ⚠ **Every route that names a base by `{id}` lives in `base_routes`, and +/// nothing else does.** That sub-router carries +/// `session_reach::gate_knowledge_base` as a `route_layer`, so each of its +/// routes — and any added to it later — answers a caller who may not reach the +/// named base with the same refusal before its handler runs (issue #56, QA +/// 2026-09-10 H2). A route that names a base and is registered on the outer +/// router instead is ungated: put it here. pub fn router(svc: Arc) -> Router { - Router::new() - .route("/bases", get(list_bases).post(create_base)) - .route( - "/bases/import", - post(import_brkb).layer(DefaultBodyLimit::max( - biorouter_mcp::knowledge::brkb::MAX_ARCHIVE_HTTP_BODY_BYTES, - )), - ) + let base_routes = Router::new() .route( "/bases/{id}", get(get_base).put(update_base).delete(delete_base), @@ -60,7 +61,6 @@ pub fn router(svc: Arc) -> Router { .route("/bases/{id}/history", get(list_history)) .route("/bases/{id}/preview", post(preview_state)) .route("/bases/{id}/restore", post(restore_state)) - .route("/expand-path", post(expand_path)) .route("/bases/{id}/raw", post(add_raw_source)) .route("/bases/{id}/ingest", post(ingest)) .route("/bases/{id}/ingest-conversation", post(ingest_conversation)) @@ -73,8 +73,23 @@ pub fn router(svc: Arc) -> Router { "/bases/{id}/sources/{sid}/credibility", put(override_credibility), ) + .route_layer(axum::middleware::from_fn_with_state( + svc.clone(), + crate::routes::session_reach::gate_knowledge_base, + )); + + Router::new() + .route("/bases", get(list_bases).post(create_base)) + .route( + "/bases/import", + post(import_brkb).layer(DefaultBodyLimit::max( + biorouter_mcp::knowledge::brkb::MAX_ARCHIVE_HTTP_BODY_BYTES, + )), + ) + .route("/expand-path", post(expand_path)) .route("/active", get(get_active).post(set_active)) .route("/check-model", post(check_model)) + .merge(base_routes) .with_state(svc) } @@ -440,8 +455,13 @@ pub struct LintBody { /// store already answers — and it would also appear on `kb_list_bases`, a /// model-facing tool whose payload Task 10D's metadata register governs. /// -/// This route is user-facing: the renderer is the only caller, and Task 10C -/// already removes private bases from the model's own listing entirely. +/// ⚠ **"The renderer is the only caller" was this doc's premise, and QA +/// measured it false on 2026-09-10 (H2):** a public chat's shell recovered the +/// daemon secret and read this list, private bases included. So the rows are +/// now the bases the caller could open — the desktop app, which sends the +/// user's proof, still sees every one, with its tier — and a private base is +/// OMITTED for anyone else, as Task 10C already omits it from the model's own +/// listing: a base's id and name are user-authored content. #[derive(Serialize, ToSchema)] pub struct KbListEntry { #[serde(flatten)] @@ -451,17 +471,28 @@ pub struct KbListEntry { #[utoipa::path( get, path = "/knowledge/bases", - responses((status = 200, description = "List of knowledge bases", body = Vec)) + responses((status = 200, description = "The knowledge bases this caller may open: every base \ + for the desktop app (the user-action proof) or a \ + caller stating a private provider, the public ones \ + for anyone else. A private base is omitted, never \ + redacted.", body = Vec)) )] pub async fn list_bases( State(svc): State>, + headers: HeaderMap, ) -> Result>, (StatusCode, String)> { + let caller = crate::routes::session_reach::http_caller(&headers).await; let bases = svc .list_bases() .map_err(|e| (StatusCode::INTERNAL_SERVER_ERROR, e.to_string()))?; Ok(Json( bases .into_iter() + .filter(|manifest| { + caller + .reach_knowledge_base(svc.root(), &manifest.id) + .is_ok() + }) .map(|manifest| KbListEntry { tier: tier::entry(svc.root(), &manifest.id).tier, manifest, @@ -1115,19 +1146,28 @@ pub struct GetActiveQuery { pub session_id: Option, } -fn selection_response( - svc: &KnowledgeService, - session_id: Option<&str>, -) -> Result { - let selection = svc - .selection(session_id) - .map_err(|e| (StatusCode::INTERNAL_SERVER_ERROR, format!("{e:#}")))?; - Ok(ActiveKbResponse { - kb_ids: selection.kb_ids, - active_kb: selection.primary_kb.clone(), - primary_kb: selection.primary_kb, - hidden_kbs: selection.hidden_kbs, - }) +/// A selection as THIS caller may see it (issue #56, QA 2026-09-10 H2). +/// +/// The second listing of base ids beside `GET /knowledge/bases`, and filtered +/// by the same gate: a base the caller cannot reach is dropped from the set and +/// from the hidden list, and a primary on one reads `null`. The result is the +/// selection the Knowledge view would hold if those bases did not exist — which +/// is exactly what the list it was given says — so nothing in it points at a +/// base the caller would then be refused. For the desktop app, which sends the +/// user's proof, nothing is dropped. +fn active_response( + selection: biorouter_mcp::knowledge::service::KbSelection, + root: &std::path::Path, + caller: &crate::routes::session_reach::HttpCaller, +) -> ActiveKbResponse { + let reachable = |id: &String| caller.reach_knowledge_base(root, id).is_ok(); + let primary_kb = selection.primary_kb.filter(reachable); + ActiveKbResponse { + kb_ids: selection.kb_ids.into_iter().filter(reachable).collect(), + active_kb: primary_kb.clone(), + primary_kb, + hidden_kbs: selection.hidden_kbs.into_iter().filter(reachable).collect(), + } } #[utoipa::path( @@ -1136,15 +1176,23 @@ fn selection_response( ("session_id" = Option, Query, description = "Optional chat session id for the session-scoped selection"), ), responses( - (status = 200, description = "The session's knowledge bases and its primary", body = ActiveKbResponse), + (status = 200, description = "The session's knowledge bases and its primary, showing only \ + the bases this caller may open: a private base is omitted \ + from both lists, and a private primary reads null, for a \ + caller without the user's proof or a private capability", body = ActiveKbResponse), (status = 403, description = "The named session is outside the caller's privacy reach") ) )] pub async fn get_active( State(svc): State>, Query(q): Query, + headers: HeaderMap, ) -> Result, (StatusCode, String)> { - Ok(Json(selection_response(&svc, q.session_id.as_deref())?)) + let caller = crate::routes::session_reach::http_caller(&headers).await; + let selection = svc + .selection(q.session_id.as_deref()) + .map_err(|e| (StatusCode::INTERNAL_SERVER_ERROR, format!("{e:#}")))?; + Ok(Json(active_response(selection, svc.root(), &caller))) } #[utoipa::path( @@ -1160,30 +1208,42 @@ pub async fn get_active( (status = 403, description = "Refused by a privacy boundary (issue #56 Task 58 / #47): \ `session_id` names a private chat (or an absent one, and an \ unproven caller is told the same thing for both) and the \ - request carried no proof it came from the user \ + request carried no proof it came from the user; or \ + `primary_kb` names a knowledge base this caller may not \ + reach, answered exactly as a base that does not exist \ (body = plain text)"), ) )] pub async fn set_active( State(svc): State>, + // Before `Json`, which consumes the body and must be last. + headers: HeaderMap, Json(body): Json, ) -> Result, (StatusCode, String)> { let primary = body .primary_update() .map_err(|message| (StatusCode::BAD_REQUEST, message))?; + let caller = crate::routes::session_reach::http_caller(&headers).await; + // Naming a base the caller may not reach is answered by the gate, in the + // gate's words, before the service sees it — the same refusal a base that + // does not exist gets, so pinning is not a way to ask which ids are private. + if let PrimaryUpdate::Set(id) = primary { + caller + .reach_knowledge_base(svc.root(), id) + .map_err(|refusal| (refusal.status, refusal.message.to_string()))?; + } + // A caller changes only what it can see: see `set_selection_within`. For a + // caller that reaches every base this is exactly `set_selection`. + let reachable = |id: &str| caller.reach_knowledge_base(svc.root(), id).is_ok(); let selection = svc - .set_selection( + .set_selection_within( body.session_id.as_deref(), body.hidden_kbs.as_deref(), primary, + &reachable, ) .map_err(|e| (StatusCode::BAD_REQUEST, format!("{e:#}")))?; - Ok(Json(ActiveKbResponse { - kb_ids: selection.kb_ids, - active_kb: selection.primary_kb.clone(), - primary_kb: selection.primary_kb, - hidden_kbs: selection.hidden_kbs, - })) + Ok(Json(active_response(selection, svc.root(), &caller))) } #[utoipa::path( @@ -1716,6 +1776,8 @@ pub async fn ingest( pub async fn ingest_conversation( State(svc): State>, Path(id): Path, + // Before `Json`, which consumes the body and must be last. + headers: HeaderMap, Json(body): Json, ) -> Result { if body.session_ids.is_empty() { @@ -1734,6 +1796,21 @@ pub async fn ingest_conversation( // and one binding is what makes that visible instead of argued. let session_manager = std::sync::Arc::new(biorouter::session::session_manager::SessionManager::instance()); + + // Issue #56, QA 2026-09-10 H2. This route NAMES chats, and streams what the + // macro makes of them back to whoever asked — so the caller must be able to + // reach each one, by the gate `GET /sessions/{id}` uses and in its words, + // before a single transcript is read. Gate G below is a different question + // (may the MODEL read them), and a caller holding only the daemon secret can + // name a private model: without this it read a private chat through one. + // Every id is checked before any is loaded, so the refusal cannot say which + // of several named chats exist. + for sid in &body.session_ids { + crate::routes::session_reach::session_reach(&session_manager, sid, &headers) + .await + .map_err(|refusal| (refusal.status, refusal.message.to_string()))?; + } + let mut sessions = Vec::new(); for sid in &body.session_ids { match session_manager.get_session(sid, true).await { diff --git a/crates/biorouter-server/src/routes/schedule.rs b/crates/biorouter-server/src/routes/schedule.rs index f9500e6f7..818cbe8e9 100644 --- a/crates/biorouter-server/src/routes/schedule.rs +++ b/crates/biorouter-server/src/routes/schedule.rs @@ -373,7 +373,10 @@ fn classify_run_now_error(id: &str, error: &biorouter::scheduler::SchedulerError SessionsQuery // This will automatically pick up 'limit' as a query parameter ), responses( - (status = 200, description = "A list of session display info", body = Vec), + (status = 200, description = "A list of session display info, holding only the runs this \ + caller could open: a private run is omitted for a caller \ + with neither the user-action proof nor a private capability, \ + as it is from `GET /sessions`", body = Vec), (status = 500, description = "Internal server error") ), tag = "schedule" @@ -383,16 +386,23 @@ async fn sessions_handler( State(state): State>, Path(schedule_id_param): Path, // Renamed to avoid confusion with session_id Query(query_params): Query, + headers: axum::http::HeaderMap, ) -> Result>, StatusCode> { let scheduler = state.scheduler(); + // Issue #56, QA 2026-09-10 M1: a schedule's runs, by name and working + // directory — the rows `GET /sessions` lists, through another door. Filtered + // by the same rule, and BEFORE the limit, so a page of private runs does not + // leave a caller with an empty page and the impression there were none. + let caller = crate::routes::session_reach::http_caller(&headers).await; - match scheduler - .sessions(&schedule_id_param, query_params.limit) - .await - { + match scheduler.sessions(&schedule_id_param, usize::MAX).await { Ok(session_tuples) => { let mut display_infos = Vec::new(); - for (session_name, session) in session_tuples { + for (session_name, session) in session_tuples + .into_iter() + .filter(|(_, session)| caller.lists_session(session.privacy_tier)) + .take(query_params.limit) + { display_infos.push(SessionDisplayInfo { id: session_name.clone(), name: session.name, diff --git a/crates/biorouter-server/src/routes/session.rs b/crates/biorouter-server/src/routes/session.rs index f9a6dffeb..2f0ed7e5a 100644 --- a/crates/biorouter-server/src/routes/session.rs +++ b/crates/biorouter-server/src/routes/session.rs @@ -337,7 +337,10 @@ fn is_valid_session_id(id: &str) -> bool { ("include_subagents" = Option, Query, description = "Include sub_agent sessions (grouped under parent_session_id); default false") ), responses( - (status = 200, description = "List of available sessions retrieved successfully", body = SessionListResponse), + (status = 200, description = "The sessions this caller could open. A private session is \ + omitted — never redacted — for a caller that carries neither \ + the user-action proof nor a private capability, exactly as \ + `GET /sessions/{session_id}` would refuse it", body = SessionListResponse), (status = 401, description = "Unauthorized - Invalid or missing API key"), (status = 500, description = "Internal server error") ), @@ -349,12 +352,18 @@ fn is_valid_session_id(id: &str) -> bool { async fn list_sessions( State(state): State>, Query(query): Query, + headers: axum::http::HeaderMap, ) -> Result, StatusCode> { - let sessions = state + // Issue #56, QA 2026-09-10 M1: this returned every row on the machine — + // title, working directory, privacy reason — to a caller the singular read + // refuses. It now returns the rows that read would admit, and nothing else. + let caller = crate::routes::session_reach::http_caller(&headers).await; + let mut sessions = state .session_manager() .list_sessions_by_types(listed_session_types(query.include_subagents)) .await .map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?; + sessions.retain(|session| caller.lists_session(session.privacy_tier)); Ok(Json(SessionListResponse { sessions })) } @@ -368,7 +377,13 @@ async fn list_sessions( ("include_subagents" = Option, Query, description = "Include sub_agent sessions (grouped under parent_session_id); default false") ), responses( - (status = 200, description = "Paginated lightweight session summaries for the sidebar", body = SidebarSessionListResponse), + (status = 200, description = "Paginated lightweight session summaries for the sidebar, \ + holding only the sessions this caller could open (see \ + `GET /sessions`). `next_offset` is where the next page \ + starts; for a caller shown every session it is `offset + \ + limit` as before, and for one shown a filtered view it is a \ + position in the underlying ordering, so pass it back as \ + given rather than computing it", body = SidebarSessionListResponse), (status = 401, description = "Unauthorized - Invalid or missing API key"), (status = 500, description = "Internal server error") ), @@ -380,26 +395,88 @@ async fn list_sessions( async fn list_sidebar_sessions( State(state): State>, Query(query): Query, + headers: axum::http::HeaderMap, ) -> Result, StatusCode> { let limit = query.limit.clamp(1, MAX_SIDEBAR_SESSION_LIMIT); - let mut sessions = state - .session_manager() - .list_session_summaries( - limit.saturating_add(1), - query.offset, - query.include_subagents, - false, - ) - .await - .map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?; + let caller = crate::routes::session_reach::http_caller(&headers).await; - let has_more = sessions.len() > limit as usize; - sessions.truncate(limit as usize); - let next_offset = has_more.then(|| query.offset.saturating_add(limit)); + // A caller shown every row — the desktop app, a private-capability program, + // or any caller with tiers switched off — pages exactly as it always did, + // one query per page. + if caller.lists_session(SessionClassification::Private) { + let mut sessions = state + .session_manager() + .list_session_summaries( + limit.saturating_add(1), + query.offset, + query.include_subagents, + false, + ) + .await + .map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?; + + let has_more = sessions.len() > limit as usize; + sessions.truncate(limit as usize); + let next_offset = has_more.then(|| query.offset.saturating_add(limit)); + + return Ok(Json(SidebarSessionListResponse { + sessions, + has_more, + next_offset, + })); + } + + // Issue #56, QA 2026-09-10 M1: every other caller is shown the public rows + // only, so the page is assembled by SCANNING the ordering rather than by + // filtering one `LIMIT` window — a window filtered after the fact hands back + // short, ragged pages, and a `has_more` counted before the filter would + // report the private rows it hid, which is the count oracle omission exists + // to close. `workspace_list` pages a filtered view the same way. + // + // `offset` and `next_offset` are therefore positions in the UNFILTERED + // ordering: the next page starts exactly where this one stopped, so a walk + // that passes `next_offset` back sees every visible row once. + // + // The scan is bounded per request. Hitting the bound is not the end of the + // list: the page says where to resume, so a machine whose history is mostly + // private is walked in several requests rather than silently cut short. + const SCAN_CHUNK: u32 = 200; + const MAX_SCANNED_ROWS: u32 = 20_000; + let manager = state.session_manager(); + let mut sessions = Vec::with_capacity(limit as usize); + let mut next_offset = None; + let mut position = query.offset; + 'scan: loop { + if position.saturating_sub(query.offset) >= MAX_SCANNED_ROWS { + next_offset = Some(position); + break; + } + let chunk = manager + .list_session_summaries(SCAN_CHUNK, position, query.include_subagents, false) + .await + .map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?; + let fetched = chunk.len() as u32; + for (index, summary) in chunk.into_iter().enumerate() { + if !caller.lists_session(summary.privacy_tier) { + continue; + } + if sessions.len() == limit as usize { + // A visible row beyond this page exists, so there is a next + // page, and it starts at this row. + next_offset = Some(position.saturating_add(index as u32)); + break 'scan; + } + sessions.push(summary); + } + if fetched < SCAN_CHUNK { + break; + } + position = position.saturating_add(fetched); + } Ok(Json(SidebarSessionListResponse { sessions, - has_more, + has_more: next_offset.is_some(), next_offset, })) } @@ -548,6 +625,9 @@ pub struct SessionModelUsageResponse { (status = 200, description = "Per-model usage for the session", body = SessionModelUsageResponse), (status = 400, description = "Invalid session id"), (status = 401, description = "Unauthorized - Invalid or missing API key"), + (status = 403, description = "Refused by a privacy boundary: the same refusal, word for \ + word, that `GET /sessions/{session_id}` gives (body = plain \ + text)"), (status = 404, description = "Session not found"), (status = 500, description = "Internal server error") ), @@ -559,22 +639,30 @@ pub struct SessionModelUsageResponse { async fn get_session_usage( State(state): State>, Path(session_id): Path, -) -> Result, StatusCode> { + headers: axum::http::HeaderMap, +) -> Response { if !is_valid_session_id(&session_id) { - return Err(StatusCode::BAD_REQUEST); + return StatusCode::BAD_REQUEST.into_response(); + } + // Issue #56, QA 2026-09-10: a named chat's metadata, and a 200/404 that told + // an unproven caller whether the id existed. The read's own gate, first. + if let Err(refusal) = + crate::routes::session_reach::session_reach(state.session_manager(), &session_id, &headers) + .await + { + return refusal.into_response(); } - let models = state + match state .session_manager() .get_session_model_usage(&session_id) .await - .map_err(|error| { - if error.to_string().contains("not found") { - StatusCode::NOT_FOUND - } else { - StatusCode::INTERNAL_SERVER_ERROR - } - })?; - Ok(Json(SessionModelUsageResponse { models })) + { + Ok(models) => Json(SessionModelUsageResponse { models }).into_response(), + Err(error) if error.to_string().contains("not found") => { + StatusCode::NOT_FOUND.into_response() + } + Err(_) => StatusCode::INTERNAL_SERVER_ERROR.into_response(), + } } #[utoipa::path( @@ -588,6 +676,9 @@ async fn get_session_usage( (status = 200, description = "Session name updated successfully"), (status = 400, description = "Bad request - Name too long (max 200 characters)"), (status = 401, description = "Unauthorized - Invalid or missing API key"), + (status = 403, description = "Refused by a privacy boundary: the same refusal, word for \ + word, that `GET /sessions/{session_id}` gives (body = plain \ + text)"), (status = 404, description = "Session not found"), (status = 500, description = "Internal server error") ), @@ -599,28 +690,36 @@ async fn get_session_usage( async fn update_session_name( State(state): State>, Path(session_id): Path, + // Before `Json`, which consumes the body and must be last. + headers: axum::http::HeaderMap, Json(request): Json, -) -> Result { +) -> Response { if !is_valid_session_id(&session_id) { - return Err(StatusCode::BAD_REQUEST); + return StatusCode::BAD_REQUEST.into_response(); } - let name = request.name.trim(); - if name.is_empty() { - return Err(StatusCode::BAD_REQUEST); + // Issue #56, QA 2026-09-10 (F0's sweep): renaming a chat is a write into it, + // and a write may never be cheaper than the read. + if let Err(refusal) = + crate::routes::session_reach::session_reach(state.session_manager(), &session_id, &headers) + .await + { + return refusal.into_response(); } - if name.len() > MAX_NAME_LENGTH { - return Err(StatusCode::BAD_REQUEST); + let name = request.name.trim(); + if name.is_empty() || name.len() > MAX_NAME_LENGTH { + return StatusCode::BAD_REQUEST.into_response(); } - state + match state .session_manager() .update(&session_id) .user_provided_name(name.to_string()) .apply() .await - .map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?; - - Ok(StatusCode::OK) + { + Ok(_) => StatusCode::OK.into_response(), + Err(_) => StatusCode::INTERNAL_SERVER_ERROR.into_response(), + } } #[utoipa::path( @@ -633,6 +732,9 @@ async fn update_session_name( responses( (status = 200, description = "Session user workflow values updated successfully", body = UpdateSessionUserWorkflowValuesResponse), (status = 401, description = "Unauthorized - Invalid or missing API key"), + (status = 403, description = "Refused by a privacy boundary: the same refusal, word for \ + word, that `GET /sessions/{session_id}` gives (body = plain \ + text)"), (status = 404, description = "Session not found", body = ErrorResponse), (status = 500, description = "Internal server error", body = ErrorResponse) ), @@ -645,14 +747,41 @@ async fn update_session_name( async fn update_session_user_workflow_values( State(state): State>, Path(session_id): Path, + // Before `Json`, which consumes the body and must be last. + headers: axum::http::HeaderMap, Json(request): Json, -) -> Result, ErrorResponse> { +) -> Response { if !is_valid_session_id(&session_id) { - return Err(ErrorResponse { + return ErrorResponse { message: "Invalid session ID".to_string(), status: StatusCode::BAD_REQUEST, - }); + } + .into_response(); } + // Issue #56, QA 2026-09-10 (F0's sweep): this rewrites the chat's workflow + // values and re-applies the workflow to its live agent — a write into the + // chat — so it asks the read's gate before it touches the row or the agent. + // The refusal is the read's plain text, not this route's JSON envelope, so + // a client recognises one boundary by one body. + if let Err(refusal) = + crate::routes::session_reach::session_reach(state.session_manager(), &session_id, &headers) + .await + { + return refusal.into_response(); + } + apply_user_workflow_values(&state, &session_id, request) + .await + .into_response() +} + +/// The body of [`update_session_user_workflow_values`] once the caller may +/// address the chat. +async fn apply_user_workflow_values( + state: &Arc, + session_id: &str, + request: UpdateSessionUserWorkflowValuesRequest, +) -> Result, ErrorResponse> { + let session_id = session_id.to_string(); state .session_manager() .update(&session_id) @@ -730,6 +859,10 @@ async fn update_session_user_workflow_values( responses( (status = 200, description = "Session deleted successfully"), (status = 401, description = "Unauthorized - Invalid or missing API key"), + (status = 403, description = "Refused by a privacy boundary (issue #56, QA 2026-09-10 \ + F0): the same refusal, word for word, that `GET \ + /sessions/{session_id}` gives — including for a chat that \ + does not exist (body = plain text)"), (status = 404, description = "Session not found"), (status = 500, description = "Internal server error") ), @@ -741,9 +874,23 @@ async fn update_session_user_workflow_values( async fn delete_session( State(state): State>, Path(session_id): Path, -) -> Result { + headers: axum::http::HeaderMap, +) -> Response { if !is_valid_session_id(&session_id) { - return Err(StatusCode::BAD_REQUEST); + return StatusCode::BAD_REQUEST.into_response(); + } + // Issue #56, QA 2026-09-10 F0. A caller holding nothing but the daemon + // secret was refused this chat's transcript and could delete it — four of + // four, measured — so the delete now asks the read's own gate, FIRST: before + // the turn is cancelled and before anything parked on a person is released, + // because each of those is itself an effect on the chat. The refusal is the + // read's, byte for byte, so it no more confirms the chat exists than the read + // does — where the old 200/404 pair confirmed it and then destroyed it. + if let Err(refusal) = + crate::routes::session_reach::session_reach(state.session_manager(), &session_id, &headers) + .await + { + return refusal.into_response(); } // Deleting a chat stops its turn. This used to happen by accident and the @@ -778,19 +925,11 @@ async fn delete_session( ); } - state - .session_manager() - .delete_session(&session_id) - .await - .map_err(|e| { - if e.to_string().contains("not found") { - StatusCode::NOT_FOUND - } else { - StatusCode::INTERNAL_SERVER_ERROR - } - })?; - - Ok(StatusCode::OK) + match state.session_manager().delete_session(&session_id).await { + Ok(()) => StatusCode::OK.into_response(), + Err(e) if e.to_string().contains("not found") => StatusCode::NOT_FOUND.into_response(), + Err(_) => StatusCode::INTERNAL_SERVER_ERROR.into_response(), + } } #[utoipa::path( @@ -971,7 +1110,27 @@ async fn edit_message( } } } - EditType::Edit => edit_in_place(&state, &session_id, &request).await, + EditType::Edit => { + // Issue #56, QA 2026-09-10 (F0's sweep). The in-place arm TRUNCATES + // this chat's history, and it asked nothing of the caller — so a + // caller the read refuses could cut a private transcript it could + // not see. It asks the read's gate now, before the turn lock (whose + // 409 would say the chat is busy) and before the snapshot. + // + // The `Diverge` arm is left on DR-19's gate above, which is strictly + // stronger for a private source (the proof, not merely reach) and + // already answers an unreadable one as private. + if let Err(refusal) = crate::routes::session_reach::session_reach( + state.session_manager(), + &session_id, + &headers, + ) + .await + { + return refusal.into_response(); + } + edit_in_place(&state, &session_id, &request).await + } } } @@ -1486,6 +1645,9 @@ pub struct SessionExtensionsResponse { responses( (status = 200, description = "Session extensions retrieved successfully", body = SessionExtensionsResponse), (status = 401, description = "Unauthorized - Invalid or missing API key"), + (status = 403, description = "Refused by a privacy boundary: the same refusal, word for \ + word, that `GET /sessions/{session_id}` gives (body = plain \ + text)"), (status = 404, description = "Session not found"), (status = 500, description = "Internal server error") ), @@ -1497,13 +1659,35 @@ pub struct SessionExtensionsResponse { async fn get_session_extensions( State(state): State>, Path(session_id): Path, -) -> Result, StatusCode> { + headers: axum::http::HeaderMap, +) -> Response { if !is_valid_session_id(&session_id) { - return Err(StatusCode::BAD_REQUEST); + return StatusCode::BAD_REQUEST.into_response(); + } + // Issue #56, QA 2026-09-10 — M2's sibling. A private chat's enabled + // extensions name, by name, the private connectors Gate E hides from a + // public model's own tool list (`cdwagent`, `ucsfomopagent`), so this asks + // the read's gate before it reads the row. + if let Err(refusal) = + crate::routes::session_reach::session_reach(state.session_manager(), &session_id, &headers) + .await + { + return refusal.into_response(); } + match session_extensions(&state, &session_id).await { + Ok(extensions) => Json(SessionExtensionsResponse { extensions }).into_response(), + Err(status) => status.into_response(), + } +} + +/// The enabled extension list of a chat the caller may address. +async fn session_extensions( + state: &Arc, + session_id: &str, +) -> Result, StatusCode> { let session = state .session_manager() - .get_session(&session_id, false) + .get_session(session_id, false) .await .map_err(|_| StatusCode::NOT_FOUND)?; @@ -1521,7 +1705,7 @@ async fn get_session_extensions( .unwrap_or_else(biorouter::config::get_enabled_extensions) }; - Ok(Json(SessionExtensionsResponse { extensions })) + Ok(extensions) } /// BR-71: the sessions holding a turn right now. @@ -1998,12 +2182,30 @@ pub(crate) mod diverge_tests { assert_eq!(status, axum::http::StatusCode::BAD_REQUEST); } + /// The person at the keyboard is told a missing chat is missing. A caller + /// holding only the daemon secret is told what the read tells it — the same + /// refusal it gets for a private chat — since QA measured this route's + /// 200/404 pair to be an oracle for which ids exist (2026-09-10). #[tokio::test(flavor = "multi_thread")] #[serial] async fn usage_route_returns_not_found_for_missing_session() { + install_test_user_action_key(); let state = AppState::new().await.unwrap(); + let res = routes(state.clone()) + .oneshot( + Request::builder() + .method("GET") + .uri("/sessions/29990101_99999/usage") + .header("X-User-Action", TEST_USER_ACTION_KEY) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(res.status(), axum::http::StatusCode::NOT_FOUND); + let (status, _) = get_usage(state, "29990101_99999").await; - assert_eq!(status, axum::http::StatusCode::NOT_FOUND); + assert_eq!(status, axum::http::StatusCode::FORBIDDEN); } /// `days` is attacker-controlled; the server clamps it rather than building a diff --git a/crates/biorouter-server/src/routes/session_reach.rs b/crates/biorouter-server/src/routes/session_reach.rs index 09dbb875e..abcc081ca 100644 --- a/crates/biorouter-server/src/routes/session_reach.rs +++ b/crates/biorouter-server/src/routes/session_reach.rs @@ -175,10 +175,13 @@ use axum::http::{HeaderMap, StatusCode}; use axum::response::{IntoResponse, Response}; use biorouter::privacy::{ProviderTier, SessionClassification}; use biorouter::session::session_manager::SessionManager; +use biorouter_mcp::knowledge::service::KnowledgeService; // Issue #56 DR-16. `src/routes/` is compiled into the `biorouterd` binary as // well as the lib and cannot name `crate::auth`, so this is the shared // direction — the same import `routes::session` and `routes::knowledge` use. -use biorouter_server::auth::{user_action_proof, UserActionProof}; +use biorouter_server::auth::{served_operator_capability, user_action_proof, UserActionProof}; +use std::path::Path; +use std::sync::Arc; /// The header a Biorouter client names the model it is running under. /// @@ -300,6 +303,51 @@ pub const SESSION_REACH_NO_KEY: &str = Nothing was read and nothing was changed. This control is unavailable on this daemon; use \ the desktop app."; +/// [`SESSION_OUT_OF_REACH`] for a knowledge base the caller named — the same +/// decision, from the same function, with the subject's noun changed and +/// nothing else (issue #56, QA 2026-09-10 H2). +/// +/// ⚠ **ONE sentence for "that base is private" and for "there is no such +/// base"**, for the reason the chat constant gives. A base's id and name are +/// user-authored content — the plan's Task 10D ruled that directly enumerating +/// them is the content crossing, not a side channel — so a refusal that told a +/// private base from an absent one would enumerate the machine's private bases +/// one guess at a time. The existence oracle AR-5 accepts is a different door +/// (`create_base`'s "already exists") and nothing here widens it. +/// +/// ⚠ Every constraint on [`SESSION_OUT_OF_REACH`] binds this one, and the leak +/// guards below are run against both: it names no base, no page and no path; it +/// is fixed text; it signposts the operator page without naming the header; and +/// its last words are the stop. +/// +/// ⚠ **The KB tool path says something different, deliberately.** A model +/// calling `kb_read_page` is told [`biorouter_mcp::knowledge::tier::KB_PRIVATE_REFUSAL`] +/// ("switch this chat to a private model"), which is the remedy for a chat. An +/// HTTP caller has no chat to switch; what it has is this daemon's reach rule, +/// the one [`SESSION_OUT_OF_REACH`] states for a chat. +pub const KNOWLEDGE_BASE_OUT_OF_REACH: &str = + "That knowledge base is private, or there is no knowledge base with that id. This request was \ + made on a public model and carried no proof it came from the person at the keyboard, and the \ + two answers are deliberately the same so that nothing about the knowledge base is disclosed. \ + Nothing was read and nothing was changed. Do not retry as you are; the same call will be \ + refused again, and no setting, hook or permission mode changes it. A private knowledge base \ + is reachable from a session running a private model, one the institution hosts or one that \ + runs on this machine, or from the desktop app when the person at the keyboard acts. Pointing \ + a program that already runs under such a model at this daemon is a setup decision for \ + whoever operates it, and the Biorouter documentation covers it under 'Reaching a private chat \ + from a script'. If this task genuinely needs that knowledge base, stop and ask the user to \ + open it for you."; + +/// …and [`SESSION_REACH_NO_KEY`]'s sibling, for a daemon that was handed no +/// user-action key at all — a `biorouter serve` daemon among them (SD-7), whose +/// browser reads this when it is pointed at a private base its operator's tier +/// does not cover. +pub const KNOWLEDGE_BASE_REACH_NO_KEY: &str = + "This daemon was started without a user-action key, so it cannot verify that a request came \ + from the person at the keyboard, and reaching into a private knowledge base requires that \ + proof. Nothing was read and nothing was changed. This control is unavailable on this daemon; \ + use the desktop app."; + /// The named session, reduced to the one bit this gate turns on. /// /// Three states rather than two because the third has to be *represented* in @@ -343,6 +391,35 @@ impl From for super::errors::ErrorResponse { } } +impl SessionOutOfReach { + /// The same refusal, worded for a knowledge base. + /// + /// A mapping between the constant pairs rather than a second decision: the + /// verdict — which of the two arms, and that it refused at all — is + /// [`refuse_unless_reachable`]'s, and this changes only the noun. Private, + /// because nothing outside this module should be choosing a refusal's words + /// apart from the decision that produced it. + fn for_knowledge_base(self) -> Self { + let message = if self.message == SESSION_REACH_NO_KEY { + KNOWLEDGE_BASE_REACH_NO_KEY + } else { + KNOWLEDGE_BASE_OUT_OF_REACH + }; + Self { message, ..self } + } +} + +impl From for TargetTier { + /// A row the caller already holds — a listing's — is readable by + /// construction, so it is never [`TargetTier::Unreadable`]. + fn from(classification: SessionClassification) -> Self { + match classification { + SessionClassification::Private => Self::Private, + SessionClassification::Public => Self::Public, + } + } +} + /// May a caller in this credential state reach a session in this state? /// /// ⚠ **Extracted so the claim is asserted rather than grepped for.** None of the @@ -488,6 +565,124 @@ pub async fn session_reach( ) } +/// Who is asking, resolved ONCE per request and threaded through every decision +/// that request needs — the HTTP counterpart of `CallCapability`, and for the +/// same reason: a listing that re-read the master switch or re-resolved the +/// caller per row could half-believe two answers. +/// +/// It carries the two facts [`session_reach`] turns on — the capability the +/// request states ([`CALLER_PROVIDER_HEADER`]) and the user-action proof — and a +/// third that only a `biorouter serve` daemon ever sets: +/// `auth::served_operator_capability`, the operator's configured tier, earned by +/// presenting the served document's cookie. +/// +/// ⚠ **The third input is read by the surfaces this type serves, and never by +/// [`session_reach`].** Listings and knowledge bases were fully open to a serve +/// daemon's browser before they were gated, so honouring the operator's tier +/// there keeps that browser's reach exactly where it was. The transcript gate +/// refused that browser every private chat before this type existed, and +/// feeding the operator's tier into it would admit what it refused — the one +/// thing this change may not do. Whether a serve operator on a private provider +/// should reach a private transcript is a decision still to be made, and it is +/// recorded as open in `docs/deployment/serve-decisions.md` SD-9, not taken here. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct HttpCaller { + /// DR-15's master opt-out, sampled with everything else. + enforced: bool, + /// What the request states it runs under, resolved by this daemon's + /// registry — [`caller_capability`]. + stated: ProviderTier, + /// A serve daemon's operator tier, for a request from its served document. + /// `Public` on every other daemon and for every other request. + served_operator: ProviderTier, + proof: UserActionProof, +} + +/// Resolve the caller behind one request. See [`HttpCaller`]. +pub async fn http_caller(headers: &HeaderMap) -> HttpCaller { + HttpCaller { + enforced: biorouter::privacy::privacy_tiers_enabled(), + stated: caller_capability(headers).await, + served_operator: served_operator_capability(headers), + proof: user_action_proof(headers), + } +} + +impl HttpCaller { + /// Private if either capability input is: a program stating a private + /// provider, or a serve daemon's own interface on a private one. + fn capability(&self) -> ProviderTier { + if self.stated.is_private() || self.served_operator.is_private() { + ProviderTier::Private + } else { + ProviderTier::Public + } + } + + /// May this caller be shown a chat of this classification in a listing? + /// + /// Exactly [`refuse_unless_reachable`]'s answer for the row, so a listing is + /// the union of what the singular gate admits one id at a time and cannot + /// tell a caller anything per-id probing is worded to withhold. **Omission, + /// not redaction**: a row carries an LLM-written title and a working + /// directory, both content (§11.4), which is the rule `workspace_list` + /// already applies to a model. + pub fn lists_session(&self, classification: SessionClassification) -> bool { + refuse_unless_reachable( + self.enforced, + TargetTier::from(classification), + self.capability(), + self.proof, + ) + .is_ok() + } + + /// The reach gate for a knowledge base the caller named — the same pure + /// decision a chat gets, with the base's tier as the target and + /// [`KNOWLEDGE_BASE_OUT_OF_REACH`] as its words. + /// + /// An id that is not well-formed, and one that names no base, are + /// [`TargetTier::Unreadable`] and so are refused exactly as a private base + /// is — to a caller that proves nothing. A caller that does prove it is the + /// user is let through to the handler, which tells them the truth (400 or + /// 404). DR-15's opt-out is inert all the way down, including for the + /// absent id, so a user who turned tiers off still gets their 404. + pub fn reach_knowledge_base(&self, root: &Path, kb_id: &str) -> Result<(), SessionOutOfReach> { + if !self.enforced { + return Ok(()); + } + refuse_unless_reachable( + self.enforced, + knowledge_base_tier(root, kb_id), + self.capability(), + self.proof, + ) + .map_err(SessionOutOfReach::for_knowledge_base) + } +} + +/// A named knowledge base, reduced to the bit the gate turns on. +/// +/// ⚠ **Absent is not public here**, though it is in +/// [`biorouter_mcp::knowledge::tier::is_private`], and both are right for their +/// callers. The tier store reads an absent base as public because "nothing is +/// there to leak" and refusing would stop a public chat creating one. At this +/// gate the question is what a REFUSAL says, and a caller told "private" for one +/// id and "not found" for another has been handed an oracle; so an absent (or +/// malformed) id is answered as a private one. Creating a base is `POST +/// /knowledge/bases`, which names no existing id and is not behind this gate. +fn knowledge_base_tier(root: &Path, kb_id: &str) -> TargetTier { + use biorouter_mcp::knowledge::{paths, tier}; + if paths::validate_kb_id(kb_id).is_err() || !paths::kb_root(root, kb_id).is_dir() { + return TargetTier::Unreadable; + } + if tier::is_private(root, kb_id) { + TargetTier::Private + } else { + TargetTier::Public + } +} + /// `GET|POST /knowledge/active` — the gated route whose router does not have an /// [`AppState`](crate::state::AppState) to resolve a tier with. /// @@ -570,6 +765,53 @@ pub async fn gate_knowledge_active( .await } +/// Every `/knowledge/bases/{id}…` route, behind ONE layer (issue #56, QA +/// 2026-09-10 H2). +/// +/// The tool path refused a public caller a private base at +/// `KnowledgeServer::call_tool`; these routes called the service directly and +/// handed the same base's pages, graph, history, location and a `.brkb` of the +/// whole tree to a caller holding nothing but the daemon secret. The plan had +/// left them ungated on the premise that "the Knowledge view is the user, not a +/// model" — true of the renderer, and false of the secret, which a public chat's +/// own shell recovered with `ps eww` (AR-11). The user is now told apart the +/// way every other private surface tells them apart: by the proof the desktop +/// sends, or by the private capability a program states. +/// +/// ⚠ **A layer on a sub-router of exactly the routes that name a base, not a +/// list of routes.** `knowledge::router` puts every `{id}` route in one router +/// and `route_layer`s this onto it, so the gate reads the `id` the router +/// itself matched — percent-decoded exactly as each handler's `Path` sees it — +/// and a route added there later is gated by construction. Reads and writes +/// alike: a caller that may not read a base may not rewrite, restore, merge or +/// delete it either, which is F0's lesson applied here before anyone measured +/// it. +/// +/// It runs before the handler's own extractors, so a refused request never has +/// its body parsed, its model constructed or its base looked up. +pub async fn gate_knowledge_base( + axum::extract::State(svc): axum::extract::State>, + params: axum::extract::RawPathParams, + request: axum::extract::Request, + next: axum::middleware::Next, +) -> Response { + let kb_id = params + .iter() + .find(|(key, _)| *key == "id") + .map(|(_, value)| value.to_owned()); + // Unreachable through `knowledge::router`, where every route this layer + // wraps captures `{id}`. Refused rather than waved through, so that a route + // moved in here without the capture fails closed instead of open. + let Some(kb_id) = kb_id else { + return (StatusCode::FORBIDDEN, KNOWLEDGE_BASE_OUT_OF_REACH).into_response(); + }; + let caller = http_caller(request.headers()).await; + if let Err(refusal) = caller.reach_knowledge_base(svc.root(), &kb_id) { + return refusal.into_response(); + } + next.run(request).await +} + #[cfg(test)] mod tests { use super::*; @@ -1080,6 +1322,9 @@ mod tests { let agent_rs = include_str!("agent.rs"); let events_rs = include_str!("session_events.rs"); let status_rs = include_str!("status.rs"); + let workflow_rs = include_str!("workflow.rs"); + let skills_rs = include_str!("skills.rs"); + let knowledge_rs = include_str!("knowledge.rs"); for (src, func, gate_call, first_touch, what) in [ ( reply_rs, @@ -1138,6 +1383,85 @@ mod tests { "try_begin_turn_idempotent(", "the turn lock, whose 409 says whether this chat is busy", ), + // ── QA 2026-09-10: F0, M2, and the sweep F0 asked for ── + ( + session_rs, + "async fn delete_session(", + "session_reach(", + "cancel_turn(", + "the turn cancel and the parked-card release, each an effect on the chat, \ + ahead of the delete itself", + ), + ( + session_rs, + "async fn update_session_name(", + "session_reach(", + ".user_provided_name(", + "the rename", + ), + ( + session_rs, + "async fn update_session_user_workflow_values(", + "session_reach(", + "apply_user_workflow_values(", + "the row write and the workflow re-applied to the live agent", + ), + ( + session_rs, + "async fn edit_message(", + "session_reach(", + "edit_in_place(", + "the in-place truncation", + ), + ( + session_rs, + "async fn get_session_extensions(", + "session_reach(", + "session_extensions(", + "the row read that names the chat's extensions", + ), + ( + session_rs, + "async fn get_session_usage(", + "session_reach(", + "get_session_model_usage(", + "the usage read, whose 200/404 said whether the id existed", + ), + ( + agent_rs, + "async fn get_tools(", + "session_reach(", + "permission_editor_tools(", + "the agent fetch, which mints an agent for the chat", + ), + ( + agent_rs, + "async fn get_callable_tool_count(", + "session_reach(", + "model_visible_tool_count(", + "the agent fetch, which mints an agent for the chat", + ), + ( + workflow_rs, + "async fn create_workflow(", + "session_reach(", + "workflow_from_session(", + "the transcript load and the model that summarises it", + ), + ( + skills_rs, + "pub async fn set_session_skills(", + "session_reach(", + "session_skills::apply(", + "the per-chat skill write", + ), + ( + knowledge_rs, + "pub async fn ingest_conversation(", + "session_reach(", + ".get_session(sid, true)", + "the transcript load", + ), ] { let handler = body_of(src, func); let gate = handler.find(gate_call).unwrap_or_else(|| { @@ -1164,10 +1488,12 @@ mod tests { // reads the row — and measured live against a private session each // answers 403 without the capability header and proceeds with it. They // are controls for the EXTRACTOR, not exemptions from the gate, and the - // comment here said otherwise until 2026-09-04. `interrupt` and - // `get_session_extensions` are the genuinely ungated pair: `interrupt` - // requires the user's proof instead, and `get_session_extensions` is on - // the module header's open residual. + // comment here said otherwise until 2026-09-04. `interrupt` requires + // the user's proof instead of reach, so it is a genuinely ungated + // control. `get_session_extensions` was this file's other one until + // QA's 2026-09-10 sweep gated it; `get_session_insights` and + // `running_sessions` replace it — machine-wide aggregates that name no + // chat — on the two sides of this file's gated handlers. // // BOTH sides in `agent.rs`: `agent_remove_extension` sits after the two // gated handlers' neighbourhood and `update_agent_provider` before it, @@ -1175,7 +1501,8 @@ mod tests { // over-reads towards the other. for (src, control) in [ (reply_rs, "pub async fn interrupt"), - (session_rs, "async fn get_session_extensions"), + (session_rs, "async fn get_session_insights("), + (session_rs, "async fn running_sessions("), (agent_rs, "async fn agent_remove_extension"), (agent_rs, "async fn update_agent_provider"), // BOTH sides in the two files this sweep added, for the same reason: @@ -1195,6 +1522,250 @@ mod tests { } } + // ─── QA 2026-09-10: the caller, the knowledge-base target, the words ─── + + fn caller( + stated: ProviderTier, + served_operator: ProviderTier, + proof: UserActionProof, + ) -> HttpCaller { + HttpCaller { + enforced: true, + stated, + served_operator, + proof, + } + } + + /// A listing admits exactly what the singular gate admits, at every corner + /// — so it can never tell a caller more than per-id probing does, and never + /// less than the desktop app and a private program are owed. + #[test] + fn a_listing_is_the_singular_gate_applied_row_by_row() { + for stated in CAPABILITIES { + for proof in PROOFS { + let who = caller(stated, ProviderTier::Public, proof); + for classification in [ + SessionClassification::Public, + SessionClassification::Private, + ] { + assert_eq!( + who.lists_session(classification), + refuse_unless_reachable( + true, + TargetTier::from(classification), + stated, + proof + ) + .is_ok(), + "{stated:?} {proof:?} {classification:?}" + ); + } + } + } + // The two shapes QA cares about, spelled out. + let secret_only = caller( + ProviderTier::Public, + ProviderTier::Public, + UserActionProof::Unproven, + ); + assert!(secret_only.lists_session(SessionClassification::Public)); + assert!(!secret_only.lists_session(SessionClassification::Private)); + let desktop = caller( + ProviderTier::Public, + ProviderTier::Public, + UserActionProof::Proven, + ); + assert!(desktop.lists_session(SessionClassification::Private)); + } + + /// A serve daemon's own interface keeps the reach its operator's provider + /// implies on the surfaces this type serves — and a serve daemon on a public + /// provider gives it none, which is the same answer as a secret-only caller. + #[test] + fn the_served_operator_standing_is_a_capability_and_only_that() { + let private_operator = caller( + ProviderTier::Public, + ProviderTier::Private, + UserActionProof::NoKeyInstalled, + ); + assert!(private_operator.lists_session(SessionClassification::Private)); + let public_operator = caller( + ProviderTier::Public, + ProviderTier::Public, + UserActionProof::NoKeyInstalled, + ); + assert!(!public_operator.lists_session(SessionClassification::Private)); + assert!(public_operator.lists_session(SessionClassification::Public)); + } + + /// ⚠ **The transcript gate never reads the served-operator standing**, and + /// this is the assertion that keeps it so: feeding it there would admit a + /// serve daemon's browser to private transcripts it has always been refused + /// — the one direction this change may not move. `session_reach` resolves + /// its capability from the header alone; the served input is read by + /// `http_caller`, which `session_reach` does not call. + #[test] + fn the_transcript_gate_does_not_read_the_served_operator_standing() { + let session_reach_body = crate::routes::body_of( + include_str!("session_reach.rs"), + "pub async fn session_reach(", + ); + assert!( + !session_reach_body.contains("served_operator") + && !session_reach_body.contains("http_caller("), + "the transcript gate now reads the serve operator's standing, which would admit a \ + browser to private transcripts it was always refused" + ); + assert!(session_reach_body.contains("caller_capability(headers)")); + } + + /// A knowledge base's target, at each of its corners: a private base; a + /// public one; one that does not exist; and an id that could not name one. + /// The last two are answered as the first, to a caller that proves nothing. + #[test] + fn a_knowledge_base_target_answers_absent_and_malformed_as_private() { + let root = tempfile::tempdir().unwrap(); + let svc = + biorouter_mcp::knowledge::service::KnowledgeService::new(root.path().to_path_buf()); + svc.create_base("notes", "Notes", None).unwrap(); + svc.create_base("omop", "OMOP", None).unwrap(); + biorouter_mcp::knowledge::tier::raise_unlocked(root.path(), "omop", true).unwrap(); + + assert_eq!( + knowledge_base_tier(root.path(), "notes"), + TargetTier::Public + ); + assert_eq!( + knowledge_base_tier(root.path(), "omop"), + TargetTier::Private + ); + assert_eq!( + knowledge_base_tier(root.path(), "no-such-base"), + TargetTier::Unreadable + ); + for malformed in ["../sessions", "Bad--Id", "", "a/b"] { + assert_eq!( + knowledge_base_tier(root.path(), malformed), + TargetTier::Unreadable, + "{malformed:?}" + ); + } + + let secret_only = caller( + ProviderTier::Public, + ProviderTier::Public, + UserActionProof::Unproven, + ); + let private_refusal = secret_only + .reach_knowledge_base(root.path(), "omop") + .unwrap_err(); + assert_eq!(private_refusal.message, KNOWLEDGE_BASE_OUT_OF_REACH); + assert_eq!(private_refusal.status, StatusCode::FORBIDDEN); + for other in ["no-such-base", "../sessions"] { + assert_eq!( + secret_only.reach_knowledge_base(root.path(), other), + Err(private_refusal), + "{other:?} was answered differently from a private base" + ); + } + assert!(secret_only + .reach_knowledge_base(root.path(), "notes") + .is_ok()); + + // The person at the keyboard reaches all of them; the handler then tells + // them the truth about the absent and malformed ones. + let desktop = caller( + ProviderTier::Public, + ProviderTier::Public, + UserActionProof::Proven, + ); + for id in ["omop", "notes", "no-such-base", "../sessions"] { + assert!( + desktop.reach_knowledge_base(root.path(), id).is_ok(), + "{id}" + ); + } + + // A keyless daemon says so in the knowledge base's words. + let keyless = caller( + ProviderTier::Public, + ProviderTier::Public, + UserActionProof::NoKeyInstalled, + ); + assert_eq!( + keyless + .reach_knowledge_base(root.path(), "omop") + .unwrap_err() + .message, + KNOWLEDGE_BASE_REACH_NO_KEY + ); + + // DR-15: with tiers off nothing is refused — not even the absent id, so a + // user who opted out still gets their 404 from the handler. + let off = HttpCaller { + enforced: false, + ..secret_only + }; + for id in ["omop", "no-such-base"] { + assert!(off.reach_knowledge_base(root.path(), id).is_ok(), "{id}"); + } + } + + /// The knowledge-base refusals obey every rule the chat ones do, checked by + /// the same predicates: fixed text, no digit, quote or path, the stop + /// clause last, the operator page named without the header, and neither + /// renderer marker. + #[test] + fn the_knowledge_base_refusals_keep_every_rule_the_chat_refusals_keep() { + for message in [KNOWLEDGE_BASE_OUT_OF_REACH, KNOWLEDGE_BASE_REACH_NO_KEY] { + assert!(!message.chars().any(|c| c.is_ascii_digit()), "{message}"); + assert!( + !message.contains('"') && !message.contains('\u{201c}'), + "{message}" + ); + assert!( + !message.contains('/') && !message.contains('\\'), + "{message}" + ); + assert!(!message.contains(CALLER_PROVIDER_HEADER), "{message}"); + assert!(!message.contains("versa_azure"), "{message}"); + assert!( + !message.contains(biorouter::privacy::refusal::USER_ACTION_REFUSAL_MARKER), + "{message}" + ); + assert!( + !message.contains(crate::routes::session::COPY_OF_PRIVATE_REFUSAL_MARKER), + "{message}" + ); + // It may call a base private only while offering "no such base". + assert!( + !message.contains("base is private") + || message.contains("or there is no knowledge base with that id"), + "{message}" + ); + } + let doc = include_str!("../../../../docs/deployment/programmatic-session-access.md"); + let title = doc + .lines() + .next() + .and_then(|l| l.strip_prefix("# ")) + .unwrap(); + assert!(KNOWLEDGE_BASE_OUT_OF_REACH.contains(title)); + assert!(KNOWLEDGE_BASE_OUT_OF_REACH.contains( + "Do not retry as you are; the same call will be refused again, and no setting, hook \ + or permission mode changes it." + )); + assert!(KNOWLEDGE_BASE_OUT_OF_REACH + .trim_end() + .ends_with("stop and ask the user to open it for you.")); + // "Started without a user-action key" is what the keyless knowledge-base + // tier binary keys on, and what a serve operator's browser reads. + assert!(KNOWLEDGE_BASE_REACH_NO_KEY.contains("started without a user-action key")); + assert_ne!(KNOWLEDGE_BASE_OUT_OF_REACH, SESSION_OUT_OF_REACH); + assert_ne!(KNOWLEDGE_BASE_REACH_NO_KEY, SESSION_REACH_NO_KEY); + } + /// The knowledge route's gate is a middleware, so the scan above cannot see /// it — but the wiring can still be lost in a refactor of `configure`, and a /// layer that is never applied is a security control that silently does @@ -1211,6 +1782,39 @@ mod tests { "the knowledge router no longer carries the session-reach gate" ); } + + /// Every route that names a base by `{id}` sits in `base_routes`, behind + /// `gate_knowledge_base`, and none sits on the outer router. The HTTP tests + /// in `tests/knowledge_routes.rs` prove the layer FIRES on the routes that + /// exist today; this is what stops a route added tomorrow landing on the + /// wrong router, where it would be ungated and nothing would say so. + #[test] + fn every_route_that_names_a_base_sits_behind_the_knowledge_base_gate() { + let router = body_of(include_str!("knowledge.rs"), "pub fn router("); + let (gated, outer) = router + .split_once(".route_layer(") + .expect("the knowledge router no longer layers the base-reach gate"); + assert!( + outer.contains("session_reach::gate_knowledge_base"), + "the knowledge router's route layer is no longer the base-reach gate" + ); + let (layer, outer) = outer + .split_once("Router::new()") + .expect("the outer knowledge router moved"); + assert!(layer.contains("gate_knowledge_base")); + assert!( + gated.matches("\"/bases/{id}").count() >= 20, + "fewer routes than expected sit behind the gate:\n{gated}" + ); + assert!( + !outer.contains("{id}"), + "a route naming a base by `{{id}}` is registered on the ungated outer router:\n{outer}" + ); + assert!( + !gated.contains("\"/bases\"") && !gated.contains("\"/active\""), + "a route that names no base was put behind the base gate" + ); + } } #[cfg(test)] @@ -2172,39 +2776,652 @@ mod bypass_tests { // Step 4.1's other half, for this route: a PUBLIC chat is untouched by // the layer and reaches the handler, which answers on its own terms. - let (status, body) = post_knowledge_active( + // + // ⚠ Since QA's 2026-09-10 sweep the handler's own terms, for an + // unproven caller naming a base that does not exist, are the + // KNOWLEDGE-BASE refusal — the one it gives for a private base, so that + // pinning is not a way to ask which ids exist. That body is still one + // only the handler can produce (the layer's is `SESSION_OUT_OF_REACH`), + // so it proves the layer let the request through as well as the old + // 400 did. The person at the keyboard still gets the 400 that names + // the id, from `set_selection`. + for session in [Some(public.id()), None] { + let mut body = serde_json::json!({ "primary_kb": NO_SUCH_KB }); + if let Some(id) = session { + body["session_id"] = serde_json::json!(id); + } + let (status, answer) = post_knowledge_active(state.clone(), body.clone(), None).await; + assert_eq!( + (status, answer.as_str()), + (StatusCode::FORBIDDEN, KNOWLEDGE_BASE_OUT_OF_REACH), + "{session:?}: the layer refused an unproven caller the session gate should have \ + let through, or the handler told it whether the base exists" + ); + let (status, answer) = + post_knowledge_active(state.clone(), body, Some(TEST_USER_ACTION_KEY)).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "{session:?}: {answer}"); + assert!( + answer.contains(NO_SUCH_KB), + "this 400 did not come from `set_selection`: only it echoes the kb id: {answer}" + ); + } + } + + // ─── QA 2026-09-10 (H2 / M1 / M2 / F0): the rest of the chat surface ─── + // + // Every test below drives the REAL router tree with the headers each + // caller really sends. "Secret only" is the caller QA measured: a public + // chat's shell that recovered the daemon secret with `ps eww`. The daemon + // cannot tell it from any other client, so it is a public model. + + /// The proof-of-user header, exactly as the desktop app sends it. + const PROOF: (&str, &str) = ("X-User-Action", TEST_USER_ACTION_KEY); + + /// A caller stating that it runs under an institution-hosted model — the + /// CLI's shape, and the capability half of the gate. + const PRIVATE_CAPABILITY: (&str, &str) = (CALLER_PROVIDER_HEADER, "versa_azure"); + + /// One request through `routes::configure`, the tree `commands::agent` + /// serves, so a gate wired onto the wrong router is measured rather than + /// assumed. `check_token` is layered outside `configure`, so every request + /// here already holds the daemon secret — which is the whole premise. + async fn call( + state: Arc, + method: &str, + uri: &str, + body: Option, + headers: &[(&str, &str)], + ) -> (StatusCode, String) { + let app = crate::routes::configure(state, "qa-h2-f0-sweep-secret".to_string()); + let mut builder = Request::builder().method(method).uri(uri); + for (name, value) in headers { + builder = builder.header(*name, *value); + } + let body = match body { + Some(json) => { + builder = builder.header("content-type", "application/json"); + Body::from(serde_json::to_vec(&json).unwrap()) + } + None => Body::empty(), + }; + let res = app.oneshot(builder.body(body).unwrap()).await.unwrap(); + let status = res.status(); + let bytes = to_bytes(res.into_body(), usize::MAX).await.unwrap(); + (status, String::from_utf8_lossy(&bytes).into_owned()) + } + + /// Every route that names ONE chat and answered a secret-only caller when + /// QA measured it, as `(method, uri, body)` for a given id. + /// + /// ⚠ **Destructive last.** Before this change the first row deleted the + /// chat outright, which would turn every later row into a probe of an + /// absent id and hide what each of them did to a real one. + fn chat_addressing_routes(id: &str) -> Vec<(&'static str, String, Option)> { + vec![ + ("GET", format!("/sessions/{id}/extensions"), None), + ("GET", format!("/sessions/{id}/usage"), None), + ("GET", format!("/agent/tools?session_id={id}"), None), + ( + "GET", + format!("/agent/callable_tool_count?session_id={id}"), + None, + ), + ( + "POST", + "/workflows/create".to_string(), + Some(serde_json::json!({ "session_id": id })), + ), + ( + "PUT", + format!("/sessions/{id}/name"), + Some(serde_json::json!({ "name": "renamed by an unproven caller" })), + ), + ( + "PUT", + format!("/sessions/{id}/user_workflow_values"), + Some(serde_json::json!({ "userWorkflowValues": {} })), + ), + ( + "POST", + "/skills/session".to_string(), + Some(serde_json::json!({ "sessionId": id, "add": ["qa-h2-probe-skill"] })), + ), + ( + "POST", + format!("/sessions/{id}/edit_message"), + Some(serde_json::json!({ "timestamp": 0, "editType": "edit" })), + ), + ("DELETE", format!("/sessions/{id}"), None), + ] + } + + /// **F0, and the sweep it asked for.** QA held nothing but the daemon + /// secret and was refused a private chat's transcript — then deleted the + /// same chat, four of four. Every route that names a chat now asks the + /// read's own gate, so each one answers an unproven caller exactly as + /// `GET /sessions/{id}` does: the same status, the same bytes, and the same + /// answer for a chat that does not exist. + /// + /// Mismatches are collected rather than asserted one at a time, so a + /// regression reports every door it reopened instead of the first. + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn every_route_that_names_a_private_chat_refuses_it_exactly_as_the_read_does() { + install_test_user_action_key(); + let state = AppState::new().await.unwrap(); + let private = seed_private_chat(&state, "QA F0 sweep (test fixture)").await; + // Syntactically a session id, and not a row on this machine. + let absent = "29990101_424242"; + + let before = state + .session_manager() + .get_session(private.id(), true) + .await + .unwrap(); + + let (read_status, read_body) = get_session_with(state.clone(), private.id(), None).await; + assert_eq!(read_status, StatusCode::FORBIDDEN); + assert_eq!( + read_body, SESSION_OUT_OF_REACH, + "the read path's refusal is what every route below is compared against" + ); + + let mut leaks = Vec::new(); + for target in [private.id(), absent] { + for (method, uri, body) in chat_addressing_routes(target) { + let (status, got) = call(state.clone(), method, &uri, body, &[]).await; + if status != read_status || got != read_body { + leaks.push(format!("{method} {uri} -> {status}: {got:.160}")); + } + } + } + assert!( + leaks.is_empty(), + "a caller holding nothing but the daemon secret was answered differently from \ + `GET /sessions/{{id}}` by {} route(s):\n {}", + leaks.len(), + leaks.join("\n ") + ); + + // …and nothing moved: the chat is still there, under its own name, with + // its transcript and its extension state. + let after = state + .session_manager() + .get_session(private.id(), true) + .await + .expect("an unproven caller removed a private chat"); + assert_eq!( + after.name, before.name, + "an unproven caller renamed a private chat" + ); + assert_eq!( + serde_json::to_value(&after.conversation).unwrap(), + serde_json::to_value(&before.conversation).unwrap(), + "an unproven caller changed a private chat's transcript" + ); + assert_eq!( + serde_json::to_value(&after.extension_data).unwrap(), + serde_json::to_value(&before.extension_data).unwrap(), + "an unproven caller wrote into a private chat's per-chat state" + ); + } + + /// The other half, which "refuse the unproven caller" alone would satisfy + /// by refusing everyone: the person at the keyboard (the proof) and a + /// program running under a private model (the capability) both still get + /// through. Each route is driven to a status only its own body can produce, + /// chosen so nothing expensive or irreversible runs: the turn lock (409), a + /// queued child (424), a chat with no workflow (404) or no transcript (an + /// `error` field). + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn the_person_at_the_keyboard_and_a_private_caller_still_reach_each_one() { + install_test_user_action_key(); + let state = AppState::new().await.unwrap(); + + for credential in [PROOF, PRIVATE_CAPABILITY] { + let private = seed_private_chat(&state, "QA F0 admitted arm (test fixture)").await; + let id = private.id(); + let headers = [credential]; + + let (status, body) = call( + state.clone(), + "GET", + &format!("/sessions/{id}/extensions"), + None, + &headers, + ) + .await; + assert_eq!(status, StatusCode::OK, "{credential:?} extensions: {body}"); + let (status, body) = call( + state.clone(), + "GET", + &format!("/sessions/{id}/usage"), + None, + &headers, + ) + .await; + assert_eq!(status, StatusCode::OK, "{credential:?} usage: {body}"); + + let (status, body) = call( + state.clone(), + "PUT", + &format!("/sessions/{id}/name"), + Some(serde_json::json!({ "name": "renamed by the user" })), + &headers, + ) + .await; + assert_eq!(status, StatusCode::OK, "{credential:?} rename: {body}"); + + // No workflow was ever attached, so the handler's own 404 is the + // proof it ran. + let (status, body) = call( + state.clone(), + "PUT", + &format!("/sessions/{id}/user_workflow_values"), + Some(serde_json::json!({ "userWorkflowValues": {} })), + &headers, + ) + .await; + assert_eq!( + status, + StatusCode::NOT_FOUND, + "{credential:?} workflow values: {body}" + ); + + let (status, body) = call( + state.clone(), + "POST", + "/skills/session", + Some(serde_json::json!({ "sessionId": id, "add": ["qa-h2-probe-skill"] })), + &headers, + ) + .await; + assert_eq!(status, StatusCode::OK, "{credential:?} skills: {body}"); + + // Held so an admitted in-place edit stops at the lock instead of + // truncating the chat. + let turn_guard = state + .try_begin_turn_idempotent(id, tokio_util::sync::CancellationToken::new(), None) + .expect("no turn is running in a session created a moment ago"); + let (status, body) = call( + state.clone(), + "POST", + &format!("/sessions/{id}/edit_message"), + Some(serde_json::json!({ "timestamp": 0, "editType": "edit" })), + &headers, + ) + .await; + assert_eq!(status, StatusCode::CONFLICT, "{credential:?} edit: {body}"); + drop(turn_guard); + + // DELETE last: admitted, it removes the row, which is the point. + let (status, body) = call( + state.clone(), + "DELETE", + &format!("/sessions/{id}"), + None, + &headers, + ) + .await; + assert_eq!(status, StatusCode::OK, "{credential:?} delete: {body}"); + assert!( + state + .session_manager() + .get_session(id, false) + .await + .is_err(), + "an admitted delete left the row behind" + ); + } + + // `/workflows/create` on a chat whose provider cannot be built here + // (no credentials in the sandbox) answers with a 200 whose `error` + // field is the handler's own — measured before this change as + // "Failed to create workflow: Provider not set". Nothing reaches a + // model, and the gate cannot produce that body. + let empty = seed_private_chat_without_messages(&state, "QA F0 empty (test fixture)").await; + let (status, body) = call( state.clone(), - serde_json::json!({ "session_id": public.id(), "primary_kb": NO_SUCH_KB }), - None, + "POST", + "/workflows/create", + Some(serde_json::json!({ "session_id": empty.id() })), + &[PROOF], ) .await; - assert_eq!( - status, - StatusCode::BAD_REQUEST, - "the layer refused an unproven caller on a PUBLIC chat: {body}" + assert_eq!(status, StatusCode::OK, "workflows/create: {body}"); + let answer: serde_json::Value = serde_json::from_str(&body).unwrap(); + assert!( + answer["error"].is_string() && !body.contains(SESSION_OUT_OF_REACH), + "workflows/create did not reach its own handler: {body}" ); + // The admitted request built an agent for the chat. Dropped here: this + // database recycles `YYYYMMDD_N` ids once a row is deleted, and a + // cached agent left under this id would be found by the next test's + // fresh chat and read as something that test's request created. + let _ = state.agent_manager.remove_session(empty.id()).await; + + // The two tool routes, on a QUEUED child: admitted, each reaches the + // not-ready answer (424) rather than minting an agent for the chat. + let child = seed_queued_private_child(&state).await; + for uri in [ + format!("/agent/tools?session_id={}", child.chat.id()), + format!("/agent/callable_tool_count?session_id={}", child.chat.id()), + ] { + let (status, body) = call(state.clone(), "GET", &uri, None, &[PROOF]).await; + assert_eq!(status, StatusCode::FAILED_DEPENDENCY, "{uri}: {body}"); + } + } + + /// **M2, as QA measured it.** `GET /agent/tools?session_id=` + /// handed a secret-only caller the private chat's tool names while + /// `add_extension` on the same chat refused. Asserted on the queued-child + /// shape so the admitted arm is observable without an agent being built. + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn a_private_chats_tool_surface_is_refused_as_its_transcript_is() { + install_test_user_action_key(); + let state = AppState::new().await.unwrap(); + let child = seed_queued_private_child(&state).await; + // Measure THIS request: an agent cached under a recycled id by an + // earlier test is not one this request created. + let _ = state.agent_manager.remove_session(child.chat.id()).await; + assert!(state.peek_agent(child.chat.id()).await.is_none()); + for uri in [ + format!("/agent/tools?session_id={}", child.chat.id()), + format!("/agent/callable_tool_count?session_id={}", child.chat.id()), + ] { + let (status, body) = call(state.clone(), "GET", &uri, None, &[]).await; + assert_eq!( + (status, body.as_str()), + (StatusCode::FORBIDDEN, SESSION_OUT_OF_REACH), + "{uri} answered a secret-only caller" + ); + } assert!( - body.contains(NO_SUCH_KB), - "this 400 did not come from `set_selection`: only it echoes the kb id: {body}" + state.peek_agent(child.chat.id()).await.is_none(), + "a refused caller still materialised an agent for the chat" ); + } - // A body naming NO session addresses the machine-wide scope, not a - // chat, so the gate has nothing to resolve and must let it through to - // the handler that owns it. - let (status, body) = post_knowledge_active( - state.clone(), - serde_json::json!({ "primary_kb": NO_SUCH_KB }), - None, + /// A public chat is untouched on every one of these routes, for a caller + /// that proves nothing — the gate is a condition on the target, never a + /// wall in front of the client. + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn a_public_chat_is_untouched_by_the_sweep() { + install_test_user_action_key(); + let state = AppState::new().await.unwrap(); + let public = seed_chat( + &state, + "QA F0 public (test fixture)", + SessionClassification::Public, ) .await; + let id = public.id(); + for (method, uri, expected) in [ + ("GET", format!("/sessions/{id}/extensions"), StatusCode::OK), + ("GET", format!("/sessions/{id}/usage"), StatusCode::OK), + ("DELETE", format!("/sessions/{id}"), StatusCode::OK), + ] { + let (status, body) = call(state.clone(), method, &uri, None, &[]).await; + assert_eq!(status, expected, "{method} {uri}: {body}"); + } + } + + /// **M1.** `GET /sessions` returned every row — 5,543 of them, 792 private, + /// each with its title, directory and privacy reason — to a caller the + /// singular read refuses. A listing now shows a caller exactly the rows the + /// singular gate would admit it to, so it cannot learn from the list what + /// per-id probing is worded not to tell it. + /// + /// Answered here is `session_reach.rs`'s open question: **filter, not + /// refuse.** A refused list would break every client for the public chats + /// the gate is deliberately inert on. + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn every_listing_shows_a_caller_only_the_chats_it_could_open() { + install_test_user_action_key(); + let state = AppState::new().await.unwrap(); + let private = seed_private_chat(&state, "QA M1 private (test fixture)").await; + let public = seed_chat( + &state, + "QA M1 public (test fixture)", + SessionClassification::Public, + ) + .await; + + for (headers, sees_private) in [ + (&[][..], false), + (&[PROOF][..], true), + (&[PRIVATE_CAPABILITY][..], true), + ] { + for uri in ["/sessions", "/sessions?include_subagents=true"] { + let (status, body) = call(state.clone(), "GET", uri, None, headers).await; + assert_eq!(status, StatusCode::OK, "{uri}: {body}"); + assert!( + body.contains(public.id()), + "{uri} {headers:?} lost a public chat" + ); + assert_eq!( + body.contains(private.id()), + sees_private, + "{uri} {headers:?}: private chat listed = {}", + body.contains(private.id()) + ); + if !sees_private { + assert!( + !body.contains("QA M1 private"), + "{uri} leaked the private chat's title without its id" + ); + } + } + let ids = sidebar_ids(&state, 50, headers).await; + assert!(ids.contains(&public.id().to_string())); + assert_eq!(ids.contains(&private.id().to_string()), sees_private); + } + } + + /// Paging a FILTERED sidebar must still walk every visible row exactly + /// once: a filter applied after `LIMIT` would hand back short, ragged pages + /// and let `has_more` count the rows it hid. + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn a_filtered_sidebar_pages_through_every_visible_chat_exactly_once() { + install_test_user_action_key(); + let state = AppState::new().await.unwrap(); + let mut seeded = Vec::new(); + for i in 0..4 { + seeded.push( + seed_chat( + &state, + &format!("QA M1 paging public {i} (test fixture)"), + SessionClassification::Public, + ) + .await, + ); + seeded.push( + seed_private_chat(&state, &format!("QA M1 paging private {i} (test fixture)")) + .await, + ); + } + let rows = sidebar_ids(&state, 3, &[]).await; + let mut deduped = rows.clone(); + deduped.sort(); + deduped.dedup(); assert_eq!( - status, - StatusCode::BAD_REQUEST, - "the gate refused a request that names no chat at all: {body}" + rows.len(), + deduped.len(), + "a filtered page repeated a row: {rows:?}" ); - assert!( - body.contains(NO_SUCH_KB), - "this 400 did not come from `set_selection`: only it echoes the kb id: {body}" + for chat in &seeded { + let tier = state + .session_manager() + .get_session(chat.id(), false) + .await + .unwrap() + .privacy_tier; + assert_eq!( + rows.contains(&chat.id().to_string()), + tier == SessionClassification::Public, + "{} ({tier:?}) was {} the unproven sidebar", + chat.id(), + if rows.contains(&chat.id().to_string()) { + "in" + } else { + "missing from" + } + ); + } + } + + /// `GET /schedule/{id}/sessions` lists a schedule's runs by name and + /// directory — the same rows, through a different door. + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn a_schedules_run_list_is_filtered_like_every_other_listing() { + install_test_user_action_key(); + let state = AppState::new().await.unwrap(); + const SCHEDULE: &str = "qa-m1-probe-schedule"; + let private = seed_private_chat(&state, "QA M1 scheduled private (test fixture)").await; + let public = seed_chat( + &state, + "QA M1 scheduled public (test fixture)", + SessionClassification::Public, + ) + .await; + for chat in [&private, &public] { + state + .session_manager() + .update(chat.id()) + .schedule_id(Some(SCHEDULE.to_string())) + .apply() + .await + .unwrap(); + } + for (headers, sees_private) in [(&[][..], false), (&[PROOF][..], true)] { + let (status, body) = call( + state.clone(), + "GET", + &format!("/schedule/{SCHEDULE}/sessions?limit=50"), + None, + headers, + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert!(body.contains(public.id())); + assert_eq!( + body.contains(private.id()), + sees_private, + "{headers:?}: {body}" + ); + } + } + + /// Every id the sidebar hands this caller, walking `next_offset` to the end. + async fn sidebar_ids( + state: &Arc, + limit: u32, + headers: &[(&str, &str)], + ) -> Vec { + let mut ids = Vec::new(); + let mut offset = 0u64; + for _ in 0..10_000 { + let (status, body) = call( + state.clone(), + "GET", + &format!("/sessions/sidebar?limit={limit}&offset={offset}"), + None, + headers, + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + let page: serde_json::Value = serde_json::from_str(&body).unwrap(); + for row in page["sessions"].as_array().unwrap() { + ids.push(row["id"].as_str().unwrap().to_string()); + } + if page["has_more"] != serde_json::Value::Bool(true) { + return ids; + } + offset = page["next_offset"] + .as_u64() + .expect("has_more without next_offset"); + } + panic!("the sidebar never reported its last page"); + } + + /// A private chat with no message at all — `/workflows/create` answers such + /// a chat before it builds an agent. + async fn seed_private_chat_without_messages(state: &Arc, label: &str) -> SeededChat { + let manager = state.session_manager(); + let session = manager + .create_session( + PathBuf::from("/tmp/task58_session_reach"), + label.to_string(), + SessionType::User, + ) + .await + .unwrap(); + manager + .update(&session.id) + .provider_name("versa_azure") + .model_config(ModelConfig::new("gpt-4o").unwrap()) + .raise_privacy(SessionClassification::Private, "turn:versa_azure") + .apply() + .await + .unwrap(); + SeededChat { + state: state.clone(), + id: session.id, + } + } + + /// A private subagent registered as still initializing: its tool routes, + /// once admitted, answer 424 without building an agent. + struct QueuedChild { + chat: SeededChat, + handle: Arc, + } + + impl Drop for QueuedChild { + fn drop(&mut self) { + self.handle + .complete(biorouter::agents::SubagentResult::from_error( + "QA M2 queued-child fixture cleaned up", + )); + } + } + + async fn seed_queued_private_child(state: &Arc) -> QueuedChild { + let manager = state.session_manager(); + let session = manager + .create_session( + PathBuf::from("/tmp/task58_session_reach"), + "QA M2 queued child (test fixture)".to_string(), + SessionType::SubAgent, + ) + .await + .unwrap(); + manager + .update(&session.id) + .provider_name("versa_azure") + .model_config(ModelConfig::new("gpt-4o").unwrap()) + .raise_privacy(SessionClassification::Private, "turn:versa_azure") + .apply() + .await + .unwrap(); + let handle = biorouter::agents::subagent_handle::BackgroundSubagent::register_initializing( + "qa-m2-parent", + session.id.clone(), + "QA M2 queued child", + tokio_util::sync::CancellationToken::new(), ); + QueuedChild { + chat: SeededChat { + state: state.clone(), + id: session.id, + }, + handle, + } } } diff --git a/crates/biorouter-server/src/routes/skills.rs b/crates/biorouter-server/src/routes/skills.rs index 08aac5f7e..85706f3ae 100644 --- a/crates/biorouter-server/src/routes/skills.rs +++ b/crates/biorouter-server/src/routes/skills.rs @@ -167,6 +167,10 @@ pub async fn skill_catalog_handler( responses( (status = 200, description = "Applied", body = SessionSkillsResponse), (status = 401, description = "Unauthorized - invalid or missing secret key"), + (status = 403, description = "Refused by a privacy boundary: `sessionId` names a chat \ + this caller may not reach, answered with the same refusal, \ + word for word, that `GET /sessions/{session_id}` gives \ + (body = plain text)"), (status = 404, description = "No such conversation"), (status = 500, description = "The override could not be persisted"), ), @@ -175,8 +179,21 @@ pub async fn skill_catalog_handler( )] pub async fn set_session_skills( State(state): State>, + // Before `Json`, which consumes the body and must be last. + headers: axum::http::HeaderMap, Json(request): Json, ) -> Result, (StatusCode, String)> { + // Issue #56, QA 2026-09-10 (F0's sweep). Enabling a skill in a chat puts its + // instructions into that chat's next turn — a write into the chat — so a + // caller the read refuses may not do it. Asked first, before the request is + // validated against anything the chat holds. + crate::routes::session_reach::session_reach( + state.session_manager(), + &request.session_id, + &headers, + ) + .await + .map_err(|refusal| (refusal.status, refusal.message.to_string()))?; if request.add.is_empty() && request.remove.is_empty() { return Err(( StatusCode::BAD_REQUEST, diff --git a/crates/biorouter-server/src/routes/web_ui.rs b/crates/biorouter-server/src/routes/web_ui.rs index f452c422b..ba509fea8 100644 --- a/crates/biorouter-server/src/routes/web_ui.rs +++ b/crates/biorouter-server/src/routes/web_ui.rs @@ -34,11 +34,21 @@ //! 4. From then on the application presents `X-Secret-Key` exactly as the //! desktop renderer does, and every API route is guarded exactly as before. //! -//! **The cookie gates the document and nothing else.** It is not accepted as -//! authentication on any API route. Accepting it there would make every API -//! route reachable by a credential the browser attaches automatically, which is -//! a cross-site request forgery surface the header scheme does not have. Keeping -//! the cookie's authority to one request is why `check_token` needed no change. +//! **The cookie gates the document, and authenticates nothing else.** It is not +//! accepted as authentication on any API route. Accepting it there would make +//! every API route reachable by a credential the browser attaches automatically, +//! which is a cross-site request forgery surface the header scheme does not +//! have. Keeping the cookie's authority to one request is why `check_token` +//! needed no change. +//! +//! It has one other reader, and it is a narrowing rather than an admission: an +//! API request that already passed `check_token` and ALSO carries this cookie +//! came from the document this daemon served, so `auth::served_operator_capability` +//! gives it the operator's configured tier on the listing and knowledge-base +//! gates (`routes::session_reach`). A request holding only the secret is a +//! public caller there. `SameSite=Strict` keeps the cookie off every cross-site +//! request, and a forged request still needs the secret, so no CSRF surface +//! appears. See `docs/deployment/serve-decisions.md` SD-9. //! //! # Why there is no brute-force throttle here //! @@ -186,6 +196,19 @@ fn cookie_value<'a>(headers: &'a HeaderMap, name: &str) -> Option<&'a str> { .map(|(_, v)| v.trim()) } +/// The session cookie the token exchange set, if the request carries one. +/// +/// The one reader of [`SESSION_COOKIE`]: the shell below asks it whether to +/// serve the document, and `auth::served_operator_capability` asks it whether a +/// request came from that document — which earns a serve daemon's own interface +/// the operator's tier on the listing and knowledge-base gates, and nothing +/// else. It is never accepted as authentication on an API route: `check_token` +/// still demands `X-Secret-Key`, so the cookie can only narrow a caller that +/// already holds the secret, never admit one that does not. +pub(crate) fn session_cookie(headers: &HeaderMap) -> Option<&str> { + cookie_value(headers, SESSION_COOKIE) +} + /// The application shell, and the token-for-cookie exchange that gates it. /// /// This handler also serves every unmatched path, so a deep link into the @@ -214,7 +237,7 @@ async fn index( return unauthorized(); } - if !ui.token_matches(cookie_value(&headers, SESSION_COOKIE)) { + if !ui.token_matches(session_cookie(&headers)) { return unauthorized(); } diff --git a/crates/biorouter-server/src/routes/workflow.rs b/crates/biorouter-server/src/routes/workflow.rs index acac229ab..9afe3bfee 100644 --- a/crates/biorouter-server/src/routes/workflow.rs +++ b/crates/biorouter-server/src/routes/workflow.rs @@ -150,8 +150,13 @@ pub struct WorkflowToYamlResponse { path = "/workflows/create", request_body = CreateWorkflowRequest, responses( - (status = 200, description = "Workflow created successfully", body = CreateWorkflowResponse), + (status = 200, description = "Workflow created successfully. Its `knowledge_bases` names \ + only the bases this caller may open", body = CreateWorkflowResponse), (status = 400, description = "Bad request"), + (status = 403, description = "Refused by a privacy boundary: `session_id` names a chat \ + this caller may not reach, answered with the same refusal, \ + word for word, that `GET /sessions/{session_id}` gives \ + (body = plain text)"), (status = 412, description = "Precondition failed - Agent not available"), (status = 500, description = "Internal server error") ), @@ -159,7 +164,58 @@ pub struct WorkflowToYamlResponse { )] async fn create_workflow( State(state): State>, + // Before `Json`, which consumes the body and must be last. + headers: axum::http::HeaderMap, Json(request): Json, +) -> axum::response::Response { + use axum::response::IntoResponse; + // Issue #56, QA 2026-09-10 (F0's sweep). This loads the named chat's WHOLE + // transcript and hands back a workflow a model wrote from it — the + // transcript again, summarised — so it asks the read's gate first, before + // the chat is loaded or an agent is built for it. + if let Err(refusal) = crate::routes::session_reach::session_reach( + state.session_manager(), + &request.session_id, + &headers, + ) + .await + { + return refusal.into_response(); + } + let caller = crate::routes::session_reach::http_caller(&headers).await; + match workflow_from_session(&state, request).await { + Ok(Json(mut response)) => { + // The enrichment records the chat's visible knowledge bases, which + // can include a private base even for a public chat. Named only as + // far as this caller may open them — the rule `GET + // /knowledge/active` applies to the same list. + if let Some(bases) = response + .workflow + .as_mut() + .and_then(|workflow| workflow.knowledge_bases.as_mut()) + { + let root = state.knowledge_service.root(); + bases + .visible + .retain(|id| caller.reach_knowledge_base(root, id).is_ok()); + if bases + .default + .as_deref() + .is_some_and(|id| caller.reach_knowledge_base(root, id).is_err()) + { + bases.default = None; + } + } + Json(response).into_response() + } + Err(status) => status.into_response(), + } +} + +/// The body of [`create_workflow`], once the caller may address the chat. +async fn workflow_from_session( + state: &Arc, + request: CreateWorkflowRequest, ) -> Result, StatusCode> { tracing::info!( "Workflow creation request received for session_id: {}", diff --git a/crates/biorouter-server/tests/knowledge_routes.rs b/crates/biorouter-server/tests/knowledge_routes.rs index 7c1a9f337..5f3228a1a 100644 --- a/crates/biorouter-server/tests/knowledge_routes.rs +++ b/crates/biorouter-server/tests/knowledge_routes.rs @@ -20,7 +20,15 @@ fn build_test_router() -> (tempfile::TempDir, Router) { (dir, router) } +/// `POST /active` as the Knowledge view sends it — carrying the user's proof. +/// +/// ⚠ Since QA's 2026-09-10 H2 sweep the selection is filtered for a caller +/// WITHOUT that proof (private and absent bases dropped, a write unable to move +/// what it cannot see), so these mechanics tests speak as the user, which is who +/// the renderer is. What an unproven caller sees and may change is +/// `h2_http_barrier`'s subject. async fn post_active(app: &Router, body: serde_json::Value) -> (u16, serde_json::Value) { + tier_route::install_test_user_action_key(); let res = app .clone() .oneshot( @@ -28,6 +36,7 @@ async fn post_active(app: &Router, body: serde_json::Value) -> (u16, serde_json: .method("POST") .uri("/active") .header("content-type", "application/json") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .body(Body::from(serde_json::to_vec(&body).unwrap())) .unwrap(), ) @@ -43,14 +52,22 @@ async fn post_active(app: &Router, body: serde_json::Value) -> (u16, serde_json: ) } +/// `GET /active`, with the user's proof — see [`post_active`]. async fn get_active(app: &Router, session_id: Option<&str>) -> serde_json::Value { + tier_route::install_test_user_action_key(); let uri = match session_id { Some(sid) => format!("/active?session_id={sid}"), None => "/active".to_string(), }; let res = app .clone() - .oneshot(Request::builder().uri(uri).body(Body::empty()).unwrap()) + .oneshot( + Request::builder() + .uri(uri) + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) + .body(Body::empty()) + .unwrap(), + ) .await .unwrap(); assert_eq!(res.status(), 200); @@ -115,17 +132,33 @@ async fn get_location_returns_kb_path() { #[tokio::test] async fn get_location_404_for_unknown_kb() { + // The person at the keyboard is told the base is not there. A caller + // without the proof is told what it is told for a private base (403) — + // QA 2026-09-10 H2 — so the 404 is not an oracle for which ids exist. + tier_route::install_test_user_action_key(); let (_d, app) = build_test_router(); let res = app + .clone() .oneshot( Request::builder() .uri("/bases/nope/location") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) .await .unwrap(); assert_eq!(res.status(), 404); + let res = app + .oneshot( + Request::builder() + .uri("/bases/nope/location") + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(res.status(), 403); } // ────────────────────────────────────────────────────────────────────────────── @@ -302,10 +335,14 @@ async fn update_base_metadata_roundtrip() { assert_eq!(manifest["name"], "Renamed Knowledge Base"); assert_eq!(manifest["color"], "#123456"); + // Asked as the user: an unproven caller is told nothing about an id that + // names no base (QA 2026-09-10 H2), so only the user can see the 404. + tier_route::install_test_user_action_key(); let res = app .oneshot( Request::builder() .uri("/bases/rename") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) @@ -1752,10 +1789,18 @@ async fn read_page_rejects_invalid_kb_id_with_400() { // "INVALID--KB" violates both the lowercase rule and the `--` rule. We do // not need to create the KB; validation fires before any filesystem touch. + // + // Asked as the user, who is owed the handler's 400. A caller without the + // proof never reaches the handler: a malformed id is answered as a private + // one is (QA 2026-09-10 H2), which also keeps a `..` out of every path + // join below the gate for that caller. + tier_route::install_test_user_action_key(); let res = app + .clone() .oneshot( Request::builder() .uri("/bases/INVALID--KB/page?path=knowledge/x.md") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) @@ -1766,6 +1811,16 @@ async fn read_page_rejects_invalid_kb_id_with_400() { 400, "invalid kb-id must return 400, not 500 (regression test)" ); + let res = app + .oneshot( + Request::builder() + .uri("/bases/INVALID--KB/page?path=knowledge/x.md") + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(res.status(), 403); } #[tokio::test] @@ -1825,6 +1880,8 @@ async fn fresh_selection_reports_soul_as_the_default_primary_when_bootstrapped() #[tokio::test] async fn active_kb_roundtrip() { + // The Knowledge view, which sends the user's proof; see `post_active`. + tier_route::install_test_user_action_key(); let (_d, app) = build_test_router(); // Empty initially. @@ -1833,6 +1890,7 @@ async fn active_kb_roundtrip() { .oneshot( Request::builder() .uri("/active") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) @@ -1881,6 +1939,7 @@ async fn active_kb_roundtrip() { Request::builder() .method("POST") .uri("/active") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .header("content-type", "application/json") .body(Body::from(set_body)) .unwrap(), @@ -1899,6 +1958,7 @@ async fn active_kb_roundtrip() { .oneshot( Request::builder() .uri("/active") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) @@ -1930,6 +1990,7 @@ async fn active_kb_roundtrip() { Request::builder() .method("POST") .uri("/active") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .header("content-type", "application/json") .body(Body::from(clear_body)) .unwrap(), @@ -1943,6 +2004,7 @@ async fn active_kb_roundtrip() { .oneshot( Request::builder() .uri("/active") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) @@ -1969,6 +2031,7 @@ async fn active_kb_roundtrip() { Request::builder() .method("POST") .uri("/active") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .header("content-type", "application/json") .body(Body::from(bad_body)) .unwrap(), @@ -2385,10 +2448,16 @@ async fn hiding_the_primary_promotes_for_an_inheriting_chat_too() { /// down, in `KnowledgeService::export_brkb`, would change this route too — the /// user would stop being able to download a private base from their own /// Knowledge view. So assert the bytes come back. +/// +/// ⚠ "The user" is the request carrying the user's proof, which the desktop's +/// Knowledge view sends. Until QA's 2026-09-10 H2 sweep this test's export +/// carried nothing and was served — the same request a public chat's shell makes +/// with a recovered daemon secret, which is now refused (`h2_http_barrier`). #[tokio::test] async fn the_users_own_export_route_is_not_subject_to_the_models_location_rule() { use axum::http::header; + tier_route::install_test_user_action_key(); let (_d, root, app) = build_test_router_with_root(); let create_body = serde_json::to_vec(&serde_json::json!({"id": "omop", "name": "Omop"})).unwrap(); @@ -2415,6 +2484,7 @@ async fn the_users_own_export_route_is_not_subject_to_the_models_location_rule() .oneshot( Request::builder() .uri("/bases/omop/export") + .header("X-User-Action", tier_route::TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) @@ -2614,7 +2684,15 @@ mod privacy_ratchet { // ── Issue #56, Task 10C: the barrier at CP2, over HTTP ─────────────────── + /// A macro run as the Knowledge view starts one: with the user's proof. + /// + /// ⚠ Since QA's 2026-09-10 H2 sweep a private base answers a caller WITHOUT + /// that proof before the macro route runs at all (`gate_knowledge_base`), + /// so the tests below — which are about CP2, the MODEL's capability — speak + /// as the user in order to reach it. The two gates ask different questions: + /// may this caller address the base, and may this model read it. async fn post_json_raw(app: &Router, uri: &str, body: serde_json::Value) -> (u16, String) { + super::tier_route::install_test_user_action_key(); let res = app .clone() .oneshot( @@ -2622,6 +2700,7 @@ mod privacy_ratchet { .method("POST") .uri(uri) .header("content-type", "application/json") + .header("X-User-Action", super::tier_route::TEST_USER_ACTION_KEY) .body(Body::from(serde_json::to_vec(&body).unwrap())) .unwrap(), ) @@ -2668,12 +2747,17 @@ mod privacy_ratchet { ); assert!(body.contains("private"), "{body}"); - // And the GUI's own read routes are untouched: the user is not a model. + // And the Knowledge view still reads the page: the user is not a model. + // ⚠ "The user" is now the request carrying the user's proof, which is + // what the desktop sends. Until QA's 2026-09-10 H2 sweep this read + // carried nothing at all and was served anyway — which is the same + // request a public chat's shell makes with a recovered daemon secret. let res = app .clone() .oneshot( Request::builder() .uri("/bases/omop/page?path=knowledge/x.md") + .header("X-User-Action", super::tier_route::TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) @@ -2997,12 +3081,15 @@ mod tier_route { let (_d, root, app) = guarded_router(); seed(&root, &app).await; + // The Knowledge view's listing — with the user's proof, which is what + // lists a private base at all since QA's 2026-09-10 H2 sweep. let res = app .clone() .oneshot( Request::builder() .uri("/bases") .header("X-Secret-Key", TEST_SECRET) + .header("X-User-Action", TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) @@ -3042,12 +3129,15 @@ mod tier_route { ) .unwrap(); + // As the publicize dialog asks it: with the user's proof. A caller + // without it is refused this private base's tier and counts outright. let res = app .clone() .oneshot( Request::builder() .uri("/bases/omop/tier") .header("X-Secret-Key", TEST_SECRET) + .header("X-User-Action", TEST_USER_ACTION_KEY) .body(Body::empty()) .unwrap(), ) @@ -3225,8 +3315,21 @@ mod okf_surface { !root.join("lit").exists(), "a refused create must not leave a half-scaffolded base on disk" ); - let (status, _) = get_json(&app, "/bases/lit").await; - assert_eq!(status, 404, "and the base must not be readable"); + // Asked as the user, who is told it is not there; a caller without the + // proof gets the refusal it gets for a private base (QA 2026-09-10 H2). + super::tier_route::install_test_user_action_key(); + let res = app + .clone() + .oneshot( + Request::builder() + .uri("/bases/lit") + .header("X-User-Action", super::tier_route::TEST_USER_ACTION_KEY) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(res.status(), 404, "and the base must not be readable"); } /// The typed graph, on the wire. @@ -3624,3 +3727,545 @@ mod merge_route { ); } } + +// ────────────────────────────────────────────────────────────────────────────── +// QA 2026-09-10, H2 — a private knowledge base over HTTP +// +// The tool path refused a public caller (`kb_read_page`, `kb_search`, +// `kb_list_pages`, `kb_export`; `kb_list_bases` omits the base), while every +// `/knowledge/bases/{id}/…` route handed the same base's pages, graph, history +// and a `.brkb` of the whole tree to a caller holding nothing but the daemon +// secret — which a public chat's own shell recovered with `ps eww`. These tests +// are that caller, and the person at the keyboard beside it. +// ────────────────────────────────────────────────────────────────────────────── +mod h2_http_barrier { + use super::tier_route::{install_test_user_action_key, TEST_USER_ACTION_KEY}; + use super::*; + use biorouter_mcp::knowledge::tier; + + /// Appears in the seeded pages and nowhere else, so "the content came + /// back" is an assertion rather than an impression. + const SENTINEL: &str = "qa-h2-private-page-marker-not-real-data"; + const PRIVATE_KB: &str = "omop"; + const PUBLIC_KB: &str = "notes"; + /// A well-formed id that names no base on this machine. + const ABSENT_KB: &str = "no-such-base"; + + async fn call( + app: &Router, + method: &str, + uri: &str, + body: Option, + proof: bool, + ) -> (u16, String) { + let mut builder = Request::builder().method(method).uri(uri); + if proof { + builder = builder.header("X-User-Action", TEST_USER_ACTION_KEY); + } + let body = match body { + Some(json) => { + builder = builder.header("content-type", "application/json"); + Body::from(serde_json::to_vec(&json).unwrap()) + } + None => Body::empty(), + }; + let res = app + .clone() + .oneshot(builder.body(body).unwrap()) + .await + .unwrap(); + let status = res.status().as_u16(); + let bytes = axum::body::to_bytes(res.into_body(), usize::MAX) + .await + .unwrap(); + (status, String::from_utf8_lossy(&bytes).into_owned()) + } + + /// Two bases through the real routes, each with one page and one commit; + /// the first is then ratcheted private the way a private chat's ingest + /// leaves it. Returns that base's commit for the history-shaped routes. + async fn seed(app: &Router, root: &std::path::Path) -> String { + for (id, name) in [(PRIVATE_KB, "OMOP"), (PUBLIC_KB, "Notes")] { + create_kb(app.clone(), id, name).await; + let (status, body) = call( + app, + "PUT", + &format!("/bases/{id}/pages/knowledge/x.md"), + Some(serde_json::json!({ + "content": valid_page("note", "X", &format!("# X\n\n{SENTINEL} in {id}")), + "commit_message": "seed", + })), + false, + ) + .await; + assert_eq!(status, 200, "seeding {id}: {body}"); + } + tier::raise_unlocked(root, PRIVATE_KB, true).unwrap(); + let (status, body) = call( + app, + "GET", + &format!("/bases/{PRIVATE_KB}/history"), + None, + true, + ) + .await; + assert_eq!(status, 200, "{body}"); + let history: serde_json::Value = serde_json::from_str(&body).unwrap(); + history[0]["commit_sha"].as_str().unwrap().to_string() + } + + fn model() -> serde_json::Value { + // Unknown to the registry: an admitted macro stops at `build_completer` + // with a 400, long before any model is reached. + serde_json::json!({ "provider": "qa-h2-no-such-provider", "model": "m" }) + } + + /// Every route under `/bases/{id}`, as `(method, uri, body)`. + /// + /// ⚠ **Destructive last**, for the reason the chat sweep gives: before this + /// change `DELETE` removed the base outright, and every row after it would + /// then have been probing an absent id. + fn base_addressing_routes( + id: &str, + sha: &str, + ) -> Vec<(&'static str, String, Option)> { + vec![ + ("GET", format!("/bases/{id}"), None), + ("GET", format!("/bases/{id}/tier"), None), + ("GET", format!("/bases/{id}/graph"), None), + ("GET", format!("/bases/{id}/location"), None), + ("GET", format!("/bases/{id}/page?path=knowledge/x.md"), None), + ("GET", format!("/bases/{id}/pages"), None), + ("GET", format!("/bases/{id}/pages/knowledge/x.md"), None), + ("GET", format!("/bases/{id}/history"), None), + ( + "POST", + format!("/bases/{id}/preview"), + Some(serde_json::json!({ "commit_sha": sha, "path": "knowledge/x.md" })), + ), + ("GET", format!("/bases/{id}/export"), None), + ( + "POST", + format!("/bases/{id}/query"), + Some(serde_json::json!({ "question": "what is in it?", "model": model() })), + ), + ( + "POST", + format!("/bases/{id}/lint"), + Some(serde_json::json!({ "model": model() })), + ), + ("POST", format!("/bases/{id}/sources/s1/reclassify"), None), + ( + "POST", + format!("/bases/{id}/tier"), + Some(serde_json::json!({ "tier": "public" })), + ), + ( + "POST", + format!("/bases/{id}/merge"), + Some(serde_json::json!({ "source_kb_id": PUBLIC_KB })), + ), + ( + "PUT", + format!("/bases/{id}"), + Some(serde_json::json!({ "name": "renamed by an unproven caller" })), + ), + ( + "PUT", + format!("/bases/{id}/default-model"), + Some(serde_json::json!({ "model": model() })), + ), + ( + "PUT", + format!("/bases/{id}/pages/knowledge/x.md"), + Some(serde_json::json!({ + "content": valid_page("note", "X", "overwritten by an unproven caller"), + "commit_message": "overwrite", + })), + ), + ( + "POST", + format!("/bases/{id}/raw"), + Some(serde_json::json!({ "text": "an unproven raw source", "title": "t" })), + ), + ( + "POST", + format!("/bases/{id}/ingest"), + Some(serde_json::json!({ "source": { "text": "t" }, "model": model() })), + ), + ( + "POST", + format!("/bases/{id}/ingest-conversation"), + Some(serde_json::json!({ "session_ids": ["29990101_1"], "model": model() })), + ), + ( + "POST", + format!("/bases/{id}/restore"), + Some(serde_json::json!({ "commit_sha": sha })), + ), + ("DELETE", format!("/bases/{id}"), None), + ] + } + + /// **H2.** Every route under `/bases/{id}` answers an unproven caller on a + /// private base exactly as the page read does — the same status and the + /// same bytes — and answers a base that does not exist the same way, so the + /// refusal is not an oracle for which ids name a private base. + /// + /// Collected rather than asserted row by row, so a regression names every + /// door it reopened. + #[tokio::test] + async fn every_route_that_names_a_private_base_refuses_an_unproven_caller_as_the_read_does() { + install_test_user_action_key(); + let (_d, root, app) = build_test_router_with_root(); + let sha = seed(&app, &root).await; + + let (read_status, read_body) = call( + &app, + "GET", + &format!("/bases/{PRIVATE_KB}/page?path=knowledge/x.md"), + None, + false, + ) + .await; + assert_eq!(read_status, 403, "the private page was served: {read_body}"); + assert!( + !read_body.contains(SENTINEL), + "the refusal carried the page" + ); + + let mut leaks = Vec::new(); + for id in [PRIVATE_KB, ABSENT_KB] { + for (method, uri, body) in base_addressing_routes(id, &sha) { + let (status, got) = call(&app, method, &uri, body, false).await; + if status != read_status || got != read_body { + leaks.push(format!("{method} {uri} -> {status}: {got:.160}")); + } + } + } + assert!( + leaks.is_empty(), + "a caller holding nothing but the daemon secret was answered differently from the \ + page read by {} route(s):\n {}", + leaks.len(), + leaks.join("\n ") + ); + + // …and nothing moved: the base is still there, still private, and its + // page still says what it said. + assert!(root.join(PRIVATE_KB).join("knowledge/x.md").exists()); + assert!(tier::is_private(&root, PRIVATE_KB)); + let page = std::fs::read_to_string(root.join(PRIVATE_KB).join("knowledge/x.md")).unwrap(); + assert!( + page.contains(SENTINEL), + "an unproven caller rewrote a private page" + ); + } + + /// The other half — "refuse the unproven caller" is satisfied by "refuse + /// everyone", and the Knowledge view must keep working. The person at the + /// keyboard reads the private base in full, and gets the honest 404 for a + /// base that is not there. + #[tokio::test] + async fn the_person_at_the_keyboard_still_reads_their_own_private_base() { + install_test_user_action_key(); + let (_d, root, app) = build_test_router_with_root(); + let sha = seed(&app, &root).await; + + for uri in [ + format!("/bases/{PRIVATE_KB}/page?path=knowledge/x.md"), + format!("/bases/{PRIVATE_KB}/pages/knowledge/x.md"), + ] { + let (status, body) = call(&app, "GET", &uri, None, true).await; + assert_eq!(status, 200, "{uri}: {body}"); + assert!( + body.contains(SENTINEL), + "{uri} came back without the page: {body}" + ); + } + for uri in [ + format!("/bases/{PRIVATE_KB}"), + format!("/bases/{PRIVATE_KB}/tier"), + format!("/bases/{PRIVATE_KB}/graph"), + format!("/bases/{PRIVATE_KB}/location"), + format!("/bases/{PRIVATE_KB}/pages"), + format!("/bases/{PRIVATE_KB}/history"), + format!("/bases/{PRIVATE_KB}/export"), + ] { + let (status, body) = call(&app, "GET", &uri, None, true).await; + assert_eq!(status, 200, "{uri}: {body:.200}"); + } + let (status, body) = call( + &app, + "POST", + &format!("/bases/{PRIVATE_KB}/preview"), + Some(serde_json::json!({ "commit_sha": sha, "path": "knowledge/x.md" })), + true, + ) + .await; + assert_eq!(status, 200, "{body}"); + assert!(body.contains(SENTINEL)); + + let (status, _) = call(&app, "GET", &format!("/bases/{ABSENT_KB}"), None, true).await; + assert_eq!( + status, 404, + "the user is entitled to know the base is not there" + ); + } + + /// A public base is untouched for a caller that proves nothing. + #[tokio::test] + async fn a_public_base_is_untouched_for_an_unproven_caller() { + install_test_user_action_key(); + let (_d, root, app) = build_test_router_with_root(); + seed(&app, &root).await; + let (status, body) = call( + &app, + "GET", + &format!("/bases/{PUBLIC_KB}/page?path=knowledge/x.md"), + None, + false, + ) + .await; + assert_eq!(status, 200, "{body}"); + assert!(body.contains(SENTINEL)); + let (status, body) = call( + &app, + "GET", + &format!("/bases/{PUBLIC_KB}/export"), + None, + false, + ) + .await; + assert_eq!(status, 200, "{body:.200}"); + } + + /// `GET /knowledge/bases` OMITS a private base from an unproven caller — + /// omission, not a 404 for the list and not a redacted row, because a + /// base's id and name are user-authored content (the tool path's + /// `kb_list_bases` makes the same choice). + #[tokio::test] + async fn the_bases_listing_omits_a_private_base_from_an_unproven_caller() { + install_test_user_action_key(); + let (_d, root, app) = build_test_router_with_root(); + seed(&app, &root).await; + + let (status, body) = call(&app, "GET", "/bases", None, false).await; + assert_eq!(status, 200, "{body}"); + let ids: Vec = serde_json::from_str::(&body) + .unwrap() + .as_array() + .unwrap() + .iter() + .map(|row| row["id"].as_str().unwrap().to_string()) + .collect(); + assert!(ids.contains(&PUBLIC_KB.to_string()), "{ids:?}"); + assert!( + !ids.contains(&PRIVATE_KB.to_string()), + "an unproven caller was listed a private base: {ids:?}" + ); + assert!( + !body.contains("OMOP"), + "the private base's name leaked: {body}" + ); + + let (status, body) = call(&app, "GET", "/bases", None, true).await; + assert_eq!(status, 200); + assert!( + body.contains(PRIVATE_KB) && body.contains(PUBLIC_KB), + "{body}" + ); + } + + /// `/knowledge/active` is the second listing of base ids, and the one the + /// Knowledge view hydrates from. An unproven caller sees only what it can + /// reach — and may not change what it cannot see: its writes leave a + /// private base's hidden state and a private primary exactly where they + /// were. Without that, a renderer that prunes ids missing from its + /// (filtered) list would silently rewrite the machine-wide selection. + #[tokio::test] + async fn the_selection_shows_and_changes_only_what_an_unproven_caller_can_reach() { + install_test_user_action_key(); + let (_d, root, app) = build_test_router_with_root(); + seed(&app, &root).await; + + // The user pins the private base as the machine-wide primary. + let (status, body) = call( + &app, + "POST", + "/active", + Some(serde_json::json!({ "primary_kb": PRIVATE_KB, "hidden_kbs": [] })), + true, + ) + .await; + assert_eq!(status, 200, "{body}"); + + // An unproven reader is told nothing about it. + let (status, body) = call(&app, "GET", "/active", None, false).await; + assert_eq!(status, 200, "{body}"); + let seen: serde_json::Value = serde_json::from_str(&body).unwrap(); + assert!( + !body.contains(PRIVATE_KB), + "an unproven caller was shown a private base: {body}" + ); + assert_eq!(seen["primary_kb"], serde_json::Value::Null); + + // It cannot clear what it cannot see… + let (status, body) = call( + &app, + "POST", + "/active", + Some(serde_json::json!({ "clear_primary": true })), + false, + ) + .await; + assert_eq!(status, 200, "{body}"); + // …cannot name it… + let (named, named_body) = call( + &app, + "POST", + "/active", + Some(serde_json::json!({ "primary_kb": PRIVATE_KB })), + false, + ) + .await; + let (absent, absent_body) = call( + &app, + "POST", + "/active", + Some(serde_json::json!({ "primary_kb": ABSENT_KB })), + false, + ) + .await; + assert_eq!(named, 403, "{named_body}"); + assert_eq!( + (named, named_body.as_str()), + (absent, absent_body.as_str()), + "naming a private base and naming no base answered differently" + ); + assert!( + !absent_body.contains(PRIVATE_KB), + "the refusal enumerated a private id: {absent_body}" + ); + + // …and cannot hide it: an id it cannot reach is not its to move. + let (status, body) = call( + &app, + "POST", + "/active", + Some(serde_json::json!({ "hidden_kbs": [PRIVATE_KB] })), + false, + ) + .await; + assert_eq!(status, 200, "{body}"); + + let (status, body) = call(&app, "GET", "/active", None, true).await; + assert_eq!(status, 200, "{body}"); + let truth: serde_json::Value = serde_json::from_str(&body).unwrap(); + assert_eq!(truth["primary_kb"], serde_json::json!(PRIVATE_KB), "{body}"); + assert!( + !truth["hidden_kbs"] + .as_array() + .unwrap() + .iter() + .any(|id| id == PRIVATE_KB), + "an unproven caller hid a private base: {body}" + ); + + // The user hides it; an unproven caller that rewrites the set cannot + // bring it back. + let (status, _) = call( + &app, + "POST", + "/active", + Some(serde_json::json!({ "hidden_kbs": [PRIVATE_KB], "clear_primary": true })), + true, + ) + .await; + assert_eq!(status, 200); + let (status, body) = call( + &app, + "POST", + "/active", + Some(serde_json::json!({ "hidden_kbs": [] })), + false, + ) + .await; + assert_eq!(status, 200, "{body}"); + assert!(!body.contains(PRIVATE_KB), "{body}"); + let (_, body) = call(&app, "GET", "/active", None, true).await; + let truth: serde_json::Value = serde_json::from_str(&body).unwrap(); + assert_eq!( + truth["hidden_kbs"], + serde_json::json!([PRIVATE_KB]), + "an unproven caller un-hid a private base it could not see: {body}" + ); + } + + /// `POST /bases/{id}/ingest-conversation` names chats as well as a base, + /// and streams what the macro makes of them back to the caller. So the + /// caller must be able to reach every chat it names — the same gate, with + /// the same refusal, as `GET /sessions/{id}` — before a transcript is read. + #[tokio::test] + async fn conversation_ingest_refuses_a_private_chat_to_an_unproven_caller() { + use biorouter::session::session_manager::{SessionManager, SessionType}; + install_test_user_action_key(); + let (_d, root, app) = build_test_router_with_root(); + seed(&app, &root).await; + + let manager = SessionManager::instance(); + let chat = manager + .create_session( + std::path::PathBuf::from("/tmp/qa_h2_ingest"), + "QA H2 ingest (test fixture)".to_string(), + SessionType::User, + ) + .await + .unwrap(); + manager + .add_message( + &chat.id, + &biorouter::conversation::message::Message::user().with_text(SENTINEL), + ) + .await + .unwrap(); + manager + .update(&chat.id) + .provider_name("versa_azure") + .model_config(biorouter::model::ModelConfig::new("gpt-4o").unwrap()) + .raise_privacy( + biorouter::privacy::SessionClassification::Private, + "turn:versa_azure", + ) + .apply() + .await + .unwrap(); + + let ingest = |id: String| serde_json::json!({ "session_ids": [id], "model": model() }); + let uri = format!("/bases/{PUBLIC_KB}/ingest-conversation"); + let (private_status, private_body) = + call(&app, "POST", &uri, Some(ingest(chat.id.clone())), false).await; + let (absent_status, absent_body) = call( + &app, + "POST", + &uri, + Some(ingest("29990101_424242".into())), + false, + ) + .await; + assert_eq!(private_status, 403, "{private_body}"); + assert_eq!( + (private_status, private_body.as_str()), + (absent_status, absent_body.as_str()), + "a private chat and an absent one answered differently" + ); + assert!(!private_body.contains(SENTINEL)); + + // The person at the keyboard gets past the gate to the handler's own + // answer — the unknown provider's 400. + let (status, body) = call(&app, "POST", &uri, Some(ingest(chat.id.clone())), true).await; + assert_eq!(status, 400, "{body}"); + + manager.delete_session(&chat.id).await.unwrap(); + } +} diff --git a/crates/biorouter-server/tests/serve_operator_reach.rs b/crates/biorouter-server/tests/serve_operator_reach.rs new file mode 100644 index 000000000..e020839e3 --- /dev/null +++ b/crates/biorouter-server/tests/serve_operator_reach.rs @@ -0,0 +1,193 @@ +//! Issue #56, QA 2026-09-10 (SD-9): a `biorouter serve` daemon's own web +//! interface keeps the reach its operator's provider implies — on the listing +//! and knowledge-base surfaces, which were open to it before they were gated — +//! and a caller holding only the daemon secret does not. +//! +//! ⚠ **Its own test binary on purpose.** The operator standing and the +//! user-action digest are both process-global `OnceLock`s. Nothing here +//! installs a digest, which is exactly how `biorouter serve` starts its daemon +//! (`Stdio::null()`, SD-7), and the operator standing installed below must not +//! leak into any other binary's view of the gates. + +// Redirects this binary's Biorouter data/config/state dirs at a throwaway root +// before `main`, so nothing here can open the developer's real `sessions.db`. +#[path = "../src/test_sandbox.rs"] +mod test_sandbox; + +use axum::{body::Body, http::Request, Router}; +use biorouter::conversation::message::Message; +use biorouter::model::ModelConfig; +use biorouter::privacy::{ProviderTier, SessionClassification}; +use biorouter::session::SessionType; +use biorouter_mcp::knowledge::service::KnowledgeService; +use biorouter_server::routes::session_reach::{KNOWLEDGE_BASE_REACH_NO_KEY, SESSION_REACH_NO_KEY}; +use biorouter_server::state::AppState; +use std::sync::Arc; +use tower::ServiceExt; + +/// The browser token `biorouter serve` would have minted for this launch. +const BROWSER_TOKEN: &str = "9f1c2e7a5b3d4c6e8f0a1b2c3d4e5f60"; +const SENTINEL: &str = "sd9-served-operator-marker-not-real-data"; + +/// The operator configured a private provider — institution-hosted — so SD-1 +/// pins every session this daemon runs to a private model. +fn install_private_operator() { + biorouter_server::auth::install_served_operator( + BROWSER_TOKEN.to_string(), + ProviderTier::Private, + ); +} + +fn served_document_cookie() -> String { + format!("biorouter_session={BROWSER_TOKEN}") +} + +async fn send(app: &Router, uri: &str, cookie: Option<&str>) -> (u16, String) { + let mut builder = Request::builder().uri(uri); + if let Some(cookie) = cookie { + builder = builder.header("cookie", cookie); + } + let res = app + .clone() + .oneshot(builder.body(Body::empty()).unwrap()) + .await + .unwrap(); + let status = res.status().as_u16(); + let bytes = axum::body::to_bytes(res.into_body(), usize::MAX) + .await + .unwrap(); + (status, String::from_utf8_lossy(&bytes).into_owned()) +} + +/// The Knowledge view in the operator's browser still reads a private base in +/// full; a caller holding the same secret without the served document's cookie +/// — or with a cookie that is not it — is refused, in the keyless daemon's own +/// words, and is listed only the public base. +#[tokio::test] +async fn the_served_interface_keeps_the_operators_reach_on_knowledge_bases() { + install_private_operator(); + let dir = tempfile::tempdir().unwrap(); + let root = dir.path().to_path_buf(); + let svc = Arc::new(KnowledgeService::new(root.clone())); + svc.create_base("omop", "OMOP", None).unwrap(); + svc.create_base("notes", "Notes", None).unwrap(); + let page = root.join("omop").join("knowledge").join("x.md"); + std::fs::create_dir_all(page.parent().unwrap()).unwrap(); + std::fs::write(&page, format!("# x\n\n{SENTINEL}\n")).unwrap(); + biorouter_mcp::knowledge::tier::raise_unlocked(&root, "omop", true).unwrap(); + let app = biorouter_server::routes::knowledge::router(svc); + + let cookie = served_document_cookie(); + let (status, body) = send(&app, "/bases/omop/page?path=knowledge/x.md", Some(&cookie)).await; + assert_eq!( + status, 200, + "the operator's own browser lost its Knowledge view: {body}" + ); + assert!(body.contains(SENTINEL)); + let (status, body) = send(&app, "/bases", Some(&cookie)).await; + assert_eq!(status, 200); + assert!(body.contains("omop") && body.contains("notes"), "{body}"); + + for (label, cookie) in [ + ("no cookie", None), + ( + "a cookie that is not the served document's", + Some("biorouter_session=guessed"), + ), + ( + "a cookie under another name", + Some("other_session=9f1c2e7a5b3d4c6e8f0a1b2c3d4e5f60"), + ), + ] { + let (status, body) = send(&app, "/bases/omop/page?path=knowledge/x.md", cookie).await; + assert_eq!( + (status, body.as_str()), + (403, KNOWLEDGE_BASE_REACH_NO_KEY), + "{label}: a caller holding only the secret read a private base" + ); + let (status, body) = send(&app, "/bases", cookie).await; + assert_eq!(status, 200); + assert!( + body.contains("notes") && !body.contains("omop"), + "{label}: listed a private base: {body}" + ); + } +} + +/// On chats, the operator standing preserves the History list — and nothing +/// else. The transcript gate refused this browser every private chat before +/// this change and still does, and so does every route that names a chat: +/// deleting one is never cheaper than reading it. Widening the transcript gate +/// for a serve operator is recorded as an open decision (SD-9), not taken. +#[tokio::test(flavor = "multi_thread")] +async fn the_served_interface_keeps_its_history_list_and_gains_nothing_else() { + install_private_operator(); + let state = AppState::new().await.unwrap(); + let manager = state.session_manager(); + let chat = manager + .create_session( + std::path::PathBuf::from("/tmp/sd9_served_operator"), + "SD-9 private (test fixture)".to_string(), + SessionType::User, + ) + .await + .unwrap(); + manager + .add_message(&chat.id, &Message::user().with_text(SENTINEL)) + .await + .unwrap(); + manager + .update(&chat.id) + .provider_name("versa_azure") + .model_config(ModelConfig::new("gpt-4o").unwrap()) + .raise_privacy(SessionClassification::Private, "turn:versa_azure") + .apply() + .await + .unwrap(); + let app = biorouter_server::routes::configure(state.clone(), "sd9-secret".to_string()); + let cookie = served_document_cookie(); + + let (status, body) = send(&app, "/sessions", Some(&cookie)).await; + assert_eq!(status, 200); + assert!( + body.contains(&chat.id), + "the operator's history lost a private chat" + ); + let (status, body) = send(&app, "/sessions", None).await; + assert_eq!(status, 200); + assert!( + !body.contains(&chat.id), + "a secret-only caller was listed a private chat" + ); + + // The transcript: refused before this change, refused after — with the + // cookie or without it. + for cookie in [Some(cookie.as_str()), None] { + let (status, body) = send(&app, &format!("/sessions/{}", chat.id), cookie).await; + assert_eq!( + (status, body.as_str()), + (403, SESSION_REACH_NO_KEY), + "{cookie:?}: the served-operator standing reached a private transcript" + ); + } + let res = app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/sessions/{}", chat.id)) + .header("cookie", served_document_cookie()) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!( + res.status(), + 403, + "a route that names a chat admitted what the transcript read refuses" + ); + assert!(manager.get_session(&chat.id, false).await.is_ok()); + + manager.delete_session(&chat.id).await.unwrap(); +} diff --git a/crates/biorouter/tests/privacy_guard_wiring.rs b/crates/biorouter/tests/privacy_guard_wiring.rs index ae20fc417..354df44f0 100644 --- a/crates/biorouter/tests/privacy_guard_wiring.rs +++ b/crates/biorouter/tests/privacy_guard_wiring.rs @@ -318,11 +318,28 @@ const REGISTRY: &[Guard] = &[ // read as refs-only and stand out. Site { file: "crates/biorouter-server/src/routes/agent.rs", - counts: c(4, 4, 0), + counts: c(6, 6, 0), kind: SiteKind::Guard, what: "`POST /agent/resume`, `POST /agent/update_from_session`, and `POST \ /agent/update_working_dir`, plus the shared `authorize_agent_control` \ - gate used by provider, extension, stop, and restart mutations", + gate used by provider, extension, stop, and restart mutations — and, \ + since QA's 2026-09-10 M2, `GET /agent/tools` (a private chat's \ + private-extension tool names, handed to a secret-only caller while \ + `add_extension` on the same chat refused) and `GET \ + /agent/callable_tool_count`, both of which mint an agent for the chat \ + they name", + }, + Site { + file: "crates/biorouter-server/src/routes/knowledge.rs", + counts: c(1, 6, 0), + kind: SiteKind::Guard, + what: "`POST /knowledge/bases/{id}/ingest-conversation`, one call inside the \ + loop over the chats the request names, before any transcript is read: \ + the route streams what a model makes of those chats back to its caller, \ + and a caller holding only the daemon secret could name a private model \ + (QA 2026-09-10 H2). The other five refs are the MODULE qualifier on \ + `session_reach::gate_knowledge_base`, `http_caller` (three handlers) and \ + `HttpCaller` — names that live beside the gate, not the gate", }, Site { file: "crates/biorouter-server/src/routes/mod.rs", @@ -339,13 +356,42 @@ const REGISTRY: &[Guard] = &[ session, plus the explicit continuation takeover and group-abandon \ recovery mutation", }, + Site { + file: "crates/biorouter-server/src/routes/schedule.rs", + counts: c(0, 1, 0), + kind: SiteKind::Unrelated, + what: "the MODULE qualifier on `session_reach::http_caller`, which filters \ + `GET /schedule/{id}/sessions` — a listing, gated by `lists_session`, not \ + by this function", + }, Site { file: "crates/biorouter-server/src/routes/session.rs", - counts: c(2, 2, 0), + counts: c(8, 10, 0), kind: SiteKind::Guard, what: "`GET /sessions/{id}` (the transcript) and `GET /sessions/{id}/export` \ - (the same transcript, `to_string_pretty`); the export sibling was \ - ungated until this sweep", + (the same transcript, `to_string_pretty`), and — QA 2026-09-10 F0 and \ + the sweep it asked for — every other route that names a chat: `DELETE \ + /sessions/{id}` (measured deleting a private chat the read refused, four \ + of four), `PUT …/name`, `PUT …/user_workflow_values`, the in-place arm \ + of `POST …/edit_message` (it truncates), `GET …/extensions` and `GET \ + …/usage`. Ten refs: the module qualifier on each of the eight calls, \ + and on `http_caller` for the two listings", + }, + Site { + file: "crates/biorouter-server/src/routes/skills.rs", + counts: c(1, 1, 0), + kind: SiteKind::Guard, + what: "`POST /skills/session`, which writes a skill's instructions into the \ + named chat's next turn (QA 2026-09-10, F0's sweep)", + }, + Site { + file: "crates/biorouter-server/src/routes/workflow.rs", + counts: c(1, 2, 0), + kind: SiteKind::Guard, + what: "`POST /workflows/create`, which loads the named chat's whole transcript \ + and returns a workflow a model wrote from it (QA 2026-09-10, F0's \ + sweep). The second ref is the module qualifier on `http_caller`, which \ + filters the knowledge bases the workflow names", }, Site { file: "crates/biorouter-server/src/routes/session_events.rs", @@ -412,10 +458,129 @@ const REGISTRY: &[Guard] = &[ status: Status::WiredThrough("session_reach"), sites: &[Site { file: SESSION_REACH, - counts: c(1, 0, 0), + counts: c(3, 0, 0), kind: SiteKind::Guard, what: "`session_reach` itself, which is this predicate plus the two lookups that \ - feed it", + feed it; and since QA's 2026-09-10 sweep `HttpCaller::lists_session` (a \ + listing is the rows this decision admits, one at a time) and \ + `HttpCaller::reach_knowledge_base` (the same decision with a knowledge \ + base as the target). ONE decision, three subjects: a second spelling of it \ + is what this census exists to stop", + }], + }, + Guard { + ident: "http_caller", + defined_in: SESSION_REACH, + decides: "who is asking, resolved ONCE per request: the stated capability, the \ + user-action proof, DR-15's switch, and — on a serve daemon only — the \ + operator's tier for a request carrying the served document's cookie", + status: Status::Wired, + sites: &[ + Site { + file: "crates/biorouter-server/src/routes/knowledge.rs", + counts: c(3, 0, 0), + kind: SiteKind::Guard, + what: "`GET /knowledge/bases` (a private base omitted from a caller that \ + cannot open it) and both halves of `/knowledge/active` (the selection \ + filtered, and a write unable to move what its caller cannot see)", + }, + Site { + file: "crates/biorouter-server/src/routes/schedule.rs", + counts: c(1, 0, 0), + kind: SiteKind::Guard, + what: "`GET /schedule/{id}/sessions`, a schedule's runs by name and directory", + }, + Site { + file: "crates/biorouter-server/src/routes/session.rs", + counts: c(2, 0, 0), + kind: SiteKind::Guard, + what: "`GET /sessions` and `GET /sessions/sidebar` — QA 2026-09-10 M1, every \ + chat on the machine, titled, to a secret-only caller", + }, + Site { + file: SESSION_REACH, + counts: c(1, 0, 0), + kind: SiteKind::Guard, + what: "`gate_knowledge_base`, the layer on every `/knowledge/bases/{id}` route", + }, + Site { + file: "crates/biorouter-server/src/routes/workflow.rs", + counts: c(1, 0, 0), + kind: SiteKind::Guard, + what: "`POST /workflows/create`, whose workflow names the chat's knowledge \ + bases — filtered to the ones its caller may open", + }, + ], + }, + Guard { + ident: "lists_session", + defined_in: SESSION_REACH, + decides: "whether a listing may show a caller a chat of a given classification: \ + exactly the singular gate's answer for that row, so omission and never \ + redaction", + status: Status::Wired, + sites: &[ + Site { + file: "crates/biorouter-server/src/routes/schedule.rs", + counts: c(1, 0, 0), + kind: SiteKind::Guard, + what: "`GET /schedule/{id}/sessions`, filtered BEFORE its limit", + }, + Site { + file: "crates/biorouter-server/src/routes/session.rs", + counts: c(3, 0, 0), + kind: SiteKind::Guard, + what: "`GET /sessions`, and `GET /sessions/sidebar` twice: once to take the \ + one-query fast path for a caller shown every row, once per row of the \ + scan that pages a filtered view without ragged pages or a count oracle", + }, + ], + }, + Guard { + ident: "reach_knowledge_base", + defined_in: SESSION_REACH, + decides: "whether an HTTP caller naming a knowledge base may reach it: the chat gate's \ + decision with the base's tier as the target, an absent or malformed id \ + answered as a private one", + status: Status::Wired, + sites: &[ + Site { + file: "crates/biorouter-server/src/routes/knowledge.rs", + counts: c(4, 0, 0), + kind: SiteKind::Guard, + what: "the bases listing's filter, the selection response's filter, and \ + `POST /knowledge/active`'s two uses: the refusal for pinning a base the \ + caller cannot reach, and the predicate `set_selection_within` merges by", + }, + Site { + file: SESSION_REACH, + counts: c(1, 0, 0), + kind: SiteKind::Guard, + what: "`gate_knowledge_base`, which asks it for every `/knowledge/bases/{id}` \ + route before the handler runs", + }, + Site { + file: "crates/biorouter-server/src/routes/workflow.rs", + counts: c(2, 0, 0), + kind: SiteKind::Guard, + what: "`POST /workflows/create`: a generated workflow's visible bases, and its \ + default one", + }, + ], + }, + Guard { + ident: "gate_knowledge_base", + defined_in: SESSION_REACH, + decides: "the knowledge-base reach gate, as ONE route layer on the sub-router holding \ + exactly the routes that name a base by `{id}`", + status: Status::Wired, + sites: &[Site { + file: "crates/biorouter-server/src/routes/knowledge.rs", + counts: c(0, 1, 0), + kind: SiteKind::Guard, + what: "`route_layer(from_fn_with_state(svc, session_reach::gate_knowledge_base))` \ + in `knowledge::router`. A REFERENCE, as `gate_knowledge_active`'s is: a \ + middleware never appears with parentheses", }], }, Guard { From e3896e5d3e8cb1d5aef1981aa69822f1e4a31065 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 02:31:24 -0700 Subject: [PATCH 10/75] docs(desktop): say what the capability header changes for a browser tab MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit It claimed 'not a widening'. It adds no reach for a caller holding the daemon secret, but a tab on a private-model host now opens private chats, including desktop-started ones — which SD-9 records. --- ui/desktop/src/utils/userAction.ts | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/ui/desktop/src/utils/userAction.ts b/ui/desktop/src/utils/userAction.ts index 2e6cb6c47..90c21db65 100644 --- a/ui/desktop/src/utils/userAction.ts +++ b/ui/desktop/src/utils/userAction.ts @@ -149,11 +149,13 @@ export const resetHostProviderForTests = (): void => { * the tab the moment its first reply made it private — measured: the next * request answered 403, the same request stating the host's provider 200. * - * ⚠ **Not authentication, and not a widening.** Anything holding the daemon - * secret can send that header already (`session_reach.rs` says as much); what - * this adds is that the one legitimate browser client says what is true of it. - * On a host configured with a public model it states a public one, and private - * chats stay out of reach exactly as before. + * ⚠ **Not authentication, and no new reach for anything holding the secret** — + * that caller could always send the header (`session_reach.rs` says as much). + * What changes is the browser tab's own reach: on a host configured with a + * private model it now opens private chats, including ones started in the + * desktop app, which SD-9 records as a consequence. On a host configured with a + * public model it states a public one, and private chats stay out of reach + * exactly as before. */ export const userActionHeaders = async (): Promise> => { if (isBrowserSurface()) { From 936f7f8b272e942d42e971abc66af278af594fe1 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 10:15:20 -0700 Subject: [PATCH 11/75] fix(desktop): send the user's proof on every call the reach gate now covers The daemon now answers a request without X-User-Action as a public model on every route that names a chat or a knowledge base (previous commit), so the desktop - the person at the keyboard - says so on each of them: - Knowledge view: knowledgeFetch and the ingest/lint SSE stream attach the proof centrally; listBases, getActive, getGraph, getLocation, getPageBody, previewState, listHistory, restoreState, getKbTier and deleteBase carry it. The listing matters twice: KnowledgeContext prunes its selection against it, so a filtered list would read as deleted bases. - Chats: the session list cache, sidebar, first-run privacy notice, delete, rename, workflow values, in-place edit, extensions, usage, tool count, skills, schedule runs and create-workflow. - History's Export also sends it: the export route was gated in an earlier sweep and exporting a private chat from History had no proof to show. sessionListCache captures include_subagents before the proof's async hop, so an orphaned request still asks for the list it was issued for. Tests that pinned exact call arguments now assert the proof is sent. Docs: privacy-tiers.md records what shipped and what did not change; the execution plan's 'the Knowledge view is the user' scope note is marked superseded and open question 15(b) answered; serve-decisions.md gains SD-9 (the served interface keeps its operator's reach on listings and knowledge bases, and gains no transcript); programmatic-session-access.md's route tables match the tree; session_reach.rs answers its open listing question. OpenAPI spec and TS client regenerated for the new 403 responses. --- CLAUDE.md | 16 +++ .../src/routes/session_reach.rs | 106 +++++++++++++++--- docs/deployment/browser-access.md | 11 +- .../deployment/programmatic-session-access.md | 51 ++++++--- docs/deployment/serve-architecture.md | 7 ++ docs/deployment/serve-decisions.md | 58 ++++++++++ docs/security/privacy-tiers-execution-plan.md | 33 +++++- docs/security/privacy-tiers.md | 37 ++++++ ui/desktop/openapi.json | 43 +++++-- ui/desktop/src/api/types.gen.ts | 59 ++++++++-- .../useSidebarSessions.test.ts | 10 ++ .../BioRouterSidebar/useSidebarSessions.ts | 4 + ui/desktop/src/components/MentionPopover.tsx | 11 +- .../src/components/alerts/useToolCount.ts | 4 + .../BottomMenuExtensionSelection.test.tsx | 9 ++ .../BottomMenuExtensionSelection.tsx | 14 ++- .../components/knowledge/KbTierControl.tsx | 6 +- .../components/knowledge/KnowledgeContext.tsx | 18 ++- .../components/knowledge/KnowledgeView.tsx | 7 +- .../knowledge/hooks/knowledgeRequest.ts | 15 +++ .../components/knowledge/hooks/useHistory.ts | 3 + .../knowledge/hooks/useIngestStream.ts | 6 + .../knowledge/hooks/useKnowledgeBases.ts | 3 +- .../knowledge/hooks/useKnowledgeGraph.ts | 9 +- .../knowledge/hooks/usePagePreview.ts | 6 + .../privacy/FirstRunPrivacyNotice.tsx | 4 + .../privacy/FirstRunPrivacyNoticeGate.tsx | 8 +- .../sessions/SessionListView.test.tsx | 16 ++- .../components/sessions/SessionListView.tsx | 8 ++ .../src/components/skills/useSkillCatalog.ts | 4 + .../components/subagent/useSubagentSession.ts | 8 +- .../CreateWorkflowFromSessionModal.tsx | 85 ++++++++------ ui/desktop/src/hooks/chatStreamStore.test.ts | 4 + ui/desktop/src/hooks/chatStreamStore.tsx | 20 +++- ui/desktop/src/hooks/useCostTracking.ts | 4 + ui/desktop/src/hooks/useWorkflowManager.ts | 4 + ui/desktop/src/schedule.ts | 4 + ui/desktop/src/utils/sessionListCache.test.ts | 23 +++- ui/desktop/src/utils/sessionListCache.ts | 23 +++- ui/desktop/src/utils/sessionNameSync.test.ts | 8 ++ ui/desktop/src/utils/sessionNameSync.ts | 4 + 41 files changed, 657 insertions(+), 116 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 17e8c1799..934e8061a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -274,6 +274,22 @@ what did not" section first**; the rest of that document is the design, not the - **Knowledge bases ratchet too.** A base takes the tier of the most sensitive session that wrote to it (four write choke points), is refused to a public caller at the read choke points, and a refusal names what it refused. `biorouter-mcp/src/knowledge/tier*.rs`. +- **Holding the daemon secret does not make a caller the user.** That was the premise behind + leaving the `/knowledge/*` read routes, `GET /sessions` and `DELETE /sessions/{id}` ungated. A + public chat's shell recovers the secret with `ps eww`, and QA used it to read a private base and + delete a private chat (H2/M1/M2/F0, 2026-09-10). Every HTTP route that names a chat or a + knowledge base now asks `routes::session_reach`'s one decision: a private target needs the + user-action proof or a stated private capability. The rules that follow: + - A route that names one chat calls `session_reach`, and refuses with its exact plain text. + - Listings filter through `HttpCaller::lists_session`. + - Every `/knowledge/bases/{id}` route sits in `knowledge::router`'s `base_routes`, behind + `gate_knowledge_base`. Put any new `{id}` route there. + - ⚠ **The renderer must send `userActionHeaders()` on every such call.** A missing proof is not an + error: private rows silently vanish, and the Knowledge view's prune effects then read them as + deleted. + - A `biorouter serve` browser gets its operator's tier on listings and knowledge bases only + (SD-9). + - The wiring census (`crates/biorouter/tests/privacy_guard_wiring.rs`) counts every call site. - **Affiliation is a third axis** (DR-26, plan Phase 6): tier asks *how sensitive*, affiliation asks *whose*. HIPAA compliance does not transfer between institutions, so a UCSF model reaching another institution's private connector is warned/refused even though both endpoints are Private. diff --git a/crates/biorouter-server/src/routes/session_reach.rs b/crates/biorouter-server/src/routes/session_reach.rs index abcc081ca..ba65df6ec 100644 --- a/crates/biorouter-server/src/routes/session_reach.rs +++ b/crates/biorouter-server/src/routes/session_reach.rs @@ -24,15 +24,21 @@ //! inert there; //! * it still reaches every session-addressing route NOT on //! [the gated list](self#the-gated-list). `POST /interrupt` and `POST -//! /agent/cancel` now require user-action proof; `GET +//! /agent/cancel` now require user-action proof. `GET //! /sessions/{id}/extensions`, `GET /sessions/{id}/usage`, `PUT //! /sessions/{id}/name`, `PUT /sessions/{id}/user_workflow_values` and -//! `DELETE /sessions/{id}` remain open, as do `GET /active_work` and `POST -//! /active_work/{id}/cancel` — which name no session id in their path and so -//! enumerate, in the manner of `GET /sessions` below, but carry a `title` and -//! `detail` holding the SHELL COMMAND or TASK PROMPT of every running job. -//! That is content rather than metadata, and it is the one row here that a -//! reader should not file mentally beside "titles and directories". +//! `DELETE /sessions/{id}` were open until QA's 2026-09-10 sweep, which +//! measured the last one deleting a private chat the read refused (F0), and +//! they are on the list now. `GET /active_work` and `POST +//! /active_work/{id}/cancel` remain open — they name no session id in their +//! path and so enumerate, but carry a `title` and `detail` holding the SHELL +//! COMMAND or TASK PROMPT of every running job. That is content rather than +//! metadata, and it is the one row here that a reader should not file +//! mentally beside "titles and directories". `GET /sessions/running` (ids +//! only, and `biorouter session list` needs it whole to report liveness +//! truthfully), `GET /sessions/changes` (a watched row's provider, model and +//! tier columns), `GET /sessions/insights` and `GET /sessions/activity` +//! (aggregates) remain open too. //! ⚠ **This bullet listed `POST /agent/resume` as open until 2026-09-04, and //! it was wrong** — measured against a live private session, `/agent/resume` //! answers 403 without the capability header and 200 with it, because @@ -55,16 +61,29 @@ //! * both read and write halves of `/knowledge/active` are gated when they name //! a session. Machine-wide selection requests name no chat and remain outside //! the session boundary; -//! * **`GET /sessions` and `GET /sessions/sidebar` are still open, and they -//! enumerate wholesale.** `SessionSummary` carries `id`, `name`, `working_dir` -//! and `privacy_tier`, so one unproven request returns every private chat on -//! the machine, titled, with the directory it runs in. This does not weaken the -//! gate — none of those rows carries a transcript — but it does undercut the -//! *reason* [`SESSION_OUT_OF_REACH`] is worded as one sentence for two -//! answers. That wording closes an oracle that enumerates private chats one id -//! at a time; the bigger one, which returns them all at once, is still there. -//! Closing it is a listing-route decision (what a caller with no proof may be -//! shown), not a reach decision, and it is not made here; +//! * ~~**`GET /sessions` and `GET /sessions/sidebar` are still open, and they +//! enumerate wholesale.**~~ **ANSWERED 2026-09-11 (QA M1): they FILTER.** QA +//! measured `GET /sessions` returning all 5,543 rows — 792 private, each with +//! id, title, working directory and privacy reason — to a caller the singular +//! read refuses, which undercut the whole reason [`SESSION_OUT_OF_REACH`] is +//! one sentence for two answers. The decision this bullet left open is now +//! made: a listing shows a caller exactly the rows this gate would admit it to +//! ([`HttpCaller::lists_session`]), so the list is the union of what per-id +//! probing could learn and nothing more. **Filter, not refuse**: a refused +//! list would break every client on the public chats the gate is deliberately +//! inert on. The sidebar pages its filtered view by scanning, so `has_more` +//! cannot count the rows it hid. `GET /schedule/{id}/sessions` takes the same +//! filter; +//! * **knowledge bases take the same decision** since the same sweep (QA H2): +//! every `/knowledge/bases/{id}…` route sits behind [`gate_knowledge_base`], +//! and `GET /knowledge/bases` and `/knowledge/active` omit what the caller +//! cannot reach. The target is the base's tier, an absent or malformed id is +//! [`TargetTier::Unreadable`], and the words are [`KNOWLEDGE_BASE_OUT_OF_REACH`]; +//! * a `biorouter serve` daemon's own interface — a request carrying the served +//! document's cookie — is given its operator's configured tier on those +//! listing and knowledge-base surfaces, which were open to it before they +//! were gated, and on NOTHING this function decides ([`HttpCaller`], +//! `docs/deployment/serve-decisions.md` SD-9); //! * **`workspace_read_conversation` was open too, and it is CLOSED — but by a //! different instrument, and a reader must not credit this module for it.** //! That MCP tool (`crates/biorouter/src/agents/workspace_extension.rs`) used @@ -149,6 +168,18 @@ //! | `POST /agent/continuation/recover` | Resumes a parked continuation in the named session. Gates directly. | //! | `POST /agent/update_from_session` | Adopts another session's provider configuration. Gates directly. | //! | `POST /agent/update_provider` · `restart` · `stop` · `remove_extension` | Gate through [`authorize_agent_control`](../agent/fn.authorize_agent_control.html), which calls [`session_reach`] and then reads the row. | +//! | `DELETE /sessions/{session_id}` | QA 2026-09-10 F0: deleted a private chat the read refused, four of four. Gated before the turn is cancelled or anything parked is released. | +//! | `PUT /sessions/{session_id}/name` · `user_workflow_values` | Writes into the chat; the second re-applies its workflow to the live agent. | +//! | `POST /sessions/{session_id}/edit_message`, `editType: edit` | Truncates the chat in place. (`diverge` keeps DR-19's stricter proof gate.) | +//! | `GET /sessions/{session_id}/extensions` · `usage` | The chat's extensions by name (M2's sibling); its usage, whose 200/404 was an existence oracle. | +//! | `GET /agent/tools` · `GET /agent/callable_tool_count` | QA M2: a private chat's private-connector tool names. Both mint an agent for the chat, so the gate runs first. The empty `session_id` of the settings page names no chat. | +//! | `POST /workflows/create` | Loads the chat's whole transcript and returns what a model makes of it. | +//! | `POST /skills/session` | Writes a skill's instructions into the chat's next turn. | +//! | `POST /knowledge/bases/{id}/ingest-conversation` | Every chat the request names, checked before any is loaded. | +//! +//! Every row since the 2026-09-10 sweep answers with [`SESSION_OUT_OF_REACH`] +//! as PLAIN TEXT — the bytes `GET /sessions/{session_id}` returns — rather than +//! through the route's own error envelope, so one boundary has one body. //! //! ⚠ **Two spellings, one list.** The last row reaches the gate through a helper //! rather than by naming it, which is why a scan for the literal `session_reach(` @@ -3318,6 +3349,47 @@ mod bypass_tests { } } + /// The knowledge-base layer, through the tree the daemon SERVES: nested + /// under `/knowledge` by `configure`, beneath `gate_knowledge_active`. Every + /// other test of it drives `knowledge::router` bare, and a layer that reads + /// its `{id}` from the matched route is exactly the kind of thing `nest` can + /// change underneath it. + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn the_knowledge_base_gate_fires_under_the_served_router_tree() { + install_test_user_action_key(); + let state = AppState::new().await.unwrap(); + let kb = format!("qa-h2-served-{}", std::process::id()); + let root = state.knowledge_service.root().to_path_buf(); + state + .knowledge_service + .create_base(&kb, "QA H2", None) + .unwrap(); + let page = root.join(&kb).join("knowledge").join("x.md"); + std::fs::create_dir_all(page.parent().unwrap()).unwrap(); + std::fs::write(&page, "# x\n\nqa-h2-served-marker\n").unwrap(); + biorouter_mcp::knowledge::tier::raise_unlocked(&root, &kb, true).unwrap(); + + let uri = format!("/knowledge/bases/{kb}/page?path=knowledge/x.md"); + let (status, body) = call(state.clone(), "GET", &uri, None, &[]).await; + assert_eq!( + (status, body.as_str()), + (StatusCode::FORBIDDEN, KNOWLEDGE_BASE_OUT_OF_REACH), + "the served tree handed a secret-only caller a private base's page" + ); + let (status, body) = call(state.clone(), "GET", &uri, None, &[PROOF]).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert!(body.contains("qa-h2-served-marker")); + let (status, body) = call(state.clone(), "GET", "/knowledge/bases", None, &[]).await; + assert_eq!(status, StatusCode::OK); + assert!( + !body.contains(&kb), + "the served list named a private base: {body}" + ); + + let _ = state.knowledge_service.delete_base_async(&kb, None).await; + } + /// Every id the sidebar hands this caller, walking `next_offset` to the end. async fn sidebar_ids( state: &Arc, diff --git a/docs/deployment/browser-access.md b/docs/deployment/browser-access.md index 299acf6b2..16192ae90 100644 --- a/docs/deployment/browser-access.md +++ b/docs/deployment/browser-access.md @@ -184,7 +184,7 @@ differs: | Area | In a browser | |---|---| -| Chat, sessions, history, extensions, skills, knowledge bases, workflows | Work as they do in the desktop application. | +| Chat, sessions, history, extensions, skills, knowledge bases, workflows | Work as they do in the desktop application, for everything public. **Private** chats and knowledge bases appear in History and the Knowledge view only when the provider you configured is private, and a private chat cannot be opened from the browser at all. See [decision SD-9](serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). | | Workspace control, several conversations at once, live app agents | Work — these are WebSocket-backed daemon routes, reached on the same origin. | | Model and provider selection | **Not available.** See [The model is fixed before you start](#the-model-is-fixed-before-you-start). | | File and folder pickers | No native dialog. You type a path, and it is a path **on the machine running the daemon**, not on the machine holding the browser. | @@ -223,6 +223,15 @@ interface is served at the root of the daemon's own origin and nowhere else. browser is on a different computer, its local files are not visible to the agent; copy them to the serving machine first. +**A private chat or knowledge base you can see in the desktop app is missing from the browser.** +Private chats and knowledge bases are shown in the browser only when the provider `serve` was +started with is itself private, meaning institution-hosted or running on this machine, and only +when `serve` was started with its access token (the default). The desktop app proves a person is at +the keyboard; a browser cannot, so it is given the reach of the model its daemon runs on and no +more. On a private provider the chat is listed but still cannot be opened from the browser. Open +it in the desktop app. The reasoning is +[decision SD-9](serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). + ### When the interface cannot be found `serve` looks for the built interface in a fixed order, and names every location it tried when it diff --git a/docs/deployment/programmatic-session-access.md b/docs/deployment/programmatic-session-access.md index 7d976e883..8809b216b 100644 --- a/docs/deployment/programmatic-session-access.md +++ b/docs/deployment/programmatic-session-access.md @@ -175,6 +175,30 @@ one of them resolves the target's tier **before** it touches the session, so a r | `POST /agent/update_working_dir` | Repoints the session at a directory. | | `POST /agent/add_extension` · `remove_extension` | Attaches or detaches tools. | | `GET`/`POST /knowledge/active` | Reads or repoints the session's knowledge bases. | +| `DELETE /sessions/{id}` | Deletes the chat. Ungated until 2026-09-10, when QA deleted a private chat the read refused (F0). | +| `PUT /sessions/{id}/name` · `PUT /sessions/{id}/user_workflow_values` | Renames the chat; rewrites its workflow values and re-applies the workflow. | +| `POST /sessions/{id}/edit_message` with `editType: edit` | Truncates the chat's history in place. (`diverge` keeps its stricter gate: see below.) | +| `GET /sessions/{id}/extensions` · `GET /sessions/{id}/usage` | The chat's enabled extensions, and its per-model token counts. | +| `GET /agent/tools` · `GET /agent/callable_tool_count`, naming a `session_id` | The chat's tool surface. Both build an agent for the chat, so both are gated before that happens. | +| `POST /workflows/create` | A workflow a model writes from the chat's whole transcript. | +| `POST /skills/session` | The chat's per-chat skill overrides. | +| `POST /knowledge/bases/{id}/ingest-conversation` | Every chat the request names, each checked before any transcript is read. | + +Each of these refuses a caller exactly as `GET /sessions/{id}` does, with the same status and the +same words, and answers a chat that does not exist the same way. Deleting, renaming or editing a +chat is never easier than reading it. + +**Listings and knowledge bases apply the same rule.** They do not refuse a list; they leave out what +the caller could not open: + +| Route | What a caller without the header or the proof gets | +|---|---| +| `GET /sessions`, `GET /sessions/sidebar`, `GET /schedule/{id}/sessions` | The public chats only. A private chat is omitted, never redacted. It is not shown with its title removed. The sidebar still pages cleanly: follow `next_offset` as returned rather than computing it. | +| Every `/knowledge/bases/{id}…` route: pages, graph, history, location, export, preview, and the writes | A private base is refused with a knowledge-base twin of the chat refusal. A base that does not exist, and a malformed id, get the same refusal. | +| `GET /knowledge/bases`, `GET`/`POST /knowledge/active` | The public bases only. A write to the selection cannot hide, reveal or unpin a base the caller cannot see. | + +A browser pointed at `biorouter serve` is a special case of this, described in +[decision SD-9](serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). ## What the header does *not* cover @@ -193,26 +217,25 @@ it would be wrong: | `POST /agent/call_tool` | Privacy Gate C at the extension-manager dispatch point, plus the uninspected-boundary refusals. | | `POST /agent/read_resource` | Gate C's sibling at the extension-manager resource read. Like `call_tool` it has no caller identity, so it declares `CallCapability::public_enforced()` rather than sampling the named session: naming a private chat buys nothing, and a private extension is refused with `403`. | -**Ungated, and low-yield.** These name a session but return only its tool surface, not its contents: -`GET /agent/tools`, `GET /agent/callable_tool_count`, `GET /skills/catalog`, `POST /skills/refresh`. -They are listed as a measurement, not as a ruling — nothing in the source records a decision to -exempt them, so read this row as "not gated" rather than "deliberately not gated". `POST -/agent/read_resource` was on this list until 2026-09-09 and is now gated; the row above says how. +**Ungated, and low-yield.** These name a session but return only skill state, not its contents: +`GET /skills/catalog`, `POST /skills/refresh`. They are listed as a measurement, not as a ruling: +nothing in the source records a decision to exempt them, so read this row as "not gated" rather +than "deliberately not gated". `POST /agent/read_resource` was on this list until 2026-09-09 and is +now gated, as the row above explains. `GET /agent/tools` and `GET /agent/callable_tool_count` were on +it until 2026-09-10. QA then measured the first handing a private chat's private-connector tool names +to a caller holding only the secret (M2), and both are now gated. **Ungated, and a known residual.** These reach or describe a private session without the gate. None -returns a transcript, so none is the boundary this feature defends — but none is closed either, and -a reader should not infer from this page that the surface is complete: +returns a transcript, so none is the boundary this feature defends. None is closed either, and a +reader should not infer from this page that the surface is complete: | Route | What an ungated caller gets | |---|---| -| `GET /sessions`, `GET /sessions/sidebar` | Every session on the machine — id, name, working directory and tier. Enumerates wholesale; recorded as an open residual in `session_reach.rs`. | -| `GET /sessions/running` | The ids of sessions with a turn in flight. | -| `GET /active_work` | Every running background job, subagent, detached turn and scheduled run — with `sessionId`, and a `title`/`detail` that carries the **shell command or task prompt**. This is content rather than metadata, and it is not named in `session_reach.rs`'s residual list. | +| `GET /sessions/running` | The ids of sessions with a turn in flight. Left unfiltered on purpose: `biorouter session list` reads it to report whether a run is still going, and a filtered answer would report a running private chat as finished. | +| `GET /sessions/changes` | For the ids a caller names, and any other row that changed, the provider, model and tier columns. Metadata, not titles or transcripts. | +| `GET /sessions/insights`, `GET /sessions/activity` | Machine-wide counts and per-day usage. Aggregates that name no chat. | +| `GET /active_work` | Every running background job, subagent, detached turn and scheduled run. Each comes with its `sessionId` and a `title`/`detail` that carries the **shell command or task prompt**. This is content rather than metadata, and it is the most significant item on this list. | | `POST /active_work/{id}/cancel` | Cancels any of the above by its registry id. The id is not a session id, so the gate cannot be applied without a reverse lookup. | -| `GET /sessions/{id}/usage` | Per-model token counts for a named session, and a `200`/`404` that tells the caller whether the id exists. | -| `GET /sessions/{id}/extensions` | The session's enabled extension list. | -| `PUT /sessions/{id}/name`, `PUT /sessions/{id}/user_workflow_values`, `DELETE /sessions/{id}` | Renames, edits workflow values, or deletes the session. | -| `POST /skills/session` | Rewrites a session's per-chat skill overrides. | | `GET /schedule/{id}/inspect`, `POST /schedule/{id}/run_now`, `POST /schedule/create` | Inspects or launches scheduled work that may run in a private session. | The daemon has no principal, so none of this is a *tier* bypass in the strict sense — a caller diff --git a/docs/deployment/serve-architecture.md b/docs/deployment/serve-architecture.md index 28c0cde6e..1795f1300 100644 --- a/docs/deployment/serve-architecture.md +++ b/docs/deployment/serve-architecture.md @@ -130,6 +130,13 @@ API routes: doing so would make every API route reachable by a cookie the browse automatically, which is a cross-site request forgery surface that the header scheme does not have. Keeping the cookie's job to one request means `check_token` is unchanged. +The cookie has one other reader, and it narrows rather than admits. An API request that has +already passed `check_token` and also carries the cookie came from the document this daemon +served, so the listing and knowledge-base gates give it the tier of the provider the operator +configured. A request holding only the secret is a public caller there. The transcript gate never +reads the cookie. See +[decision SD-9](serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). + > **Warning.** `check_token` records a failed attempt for every request without the secret and > refuses after twenty inside sixty seconds, keyed on the peer address. The browser-token check > must not feed that same counter — a mistyped URL would otherwise lock the user out of their own diff --git a/docs/deployment/serve-decisions.md b/docs/deployment/serve-decisions.md index f413d2bef..6d3723803 100644 --- a/docs/deployment/serve-decisions.md +++ b/docs/deployment/serve-decisions.md @@ -227,6 +227,64 @@ can never half-believe a person is reachable. --- +## SD-9 — The served interface keeps its operator's reach on listings and knowledge bases, and gains nothing else + +**Ruling (2026-09-11).** Since the privacy fix for QA findings H2 and M1 (2026-09-10), every +daemon route that lists chats, or names, lists or reads a knowledge base, answers a caller that +holds only the daemon secret as a **public model**: private chats are left out of lists, and a +private knowledge base is refused. The desktop application is told apart by the proof-of-user +header it sends. A `serve` daemon holds no such proof (SD-7), so it recognises its **own +interface** another way. A request that carries the served document's session cookie, on a daemon +started with a browser token, is given the tier implied by the provider the operator configured +(SD-1). That tier is read once at launch: the declared tier of the configured provider, reduced with +`least` over a configured lead provider, which is the reduction a bound lead/worker pair gets. + +- On a **private** provider (institution-hosted, or local), the History list and the Knowledge + view show private chats and knowledge bases, as they did before the fix. +- On a **public** provider they show public ones only. That is also what any caller holding just + the secret sees. + +**Why.** SD-1 already makes every session in a `serve` daemon run on the operator's provider, so +that provider's tier is the only capability the interface can be said to have. The cookie is what +separates the interface from anything else holding the secret. Without it the fix would have had +to go one of two ways, and both are wrong. One strips an operator on a private provider of their +own history and knowledge, which is a hard regression. The other hands every holder of the secret +the operator's reach, which reopens H2 on every `serve` daemon. + +**What it does not do.** + +- **It reaches no private transcript.** The transcript gate, and every route that names one chat + (open, export, the live event stream, delete, rename, and the rest), never read this standing. + A `serve` browser was refused every private transcript before this ruling and still is. So on a + private provider the History list shows private chats that cannot be opened from the browser, + and cannot be deleted or renamed from it either. That is SD-7's limitation, unchanged, and it + keeps deleting a chat from ever being easier than reading it. Letting the transcript gate honour + a served operator would be the first time a gate widened. It is an **open decision**, recorded + here and not taken. +- **It is not authentication, and not a proof of a person.** `biorouter serve` passes both the + secret and the browser token in the daemon's environment. A caller that can read one can read + the other, which is the residual the `X-Caller-Provider` header already carries + ([issue #47](https://github.com/BaranziniLab/biorouter/issues/47)). It never satisfies a + proof-of-user check, so SD-1 and SD-8 stand exactly as they were. +- **A `--no-token` daemon gives it to nobody.** Without a token there is no cookie, and the + interface cannot be told apart from any other local caller. Such a daemon shows public chats and + knowledge bases only. +- **It creates no cross-site request forgery surface.** The cookie is `SameSite=Strict`, so no + cross-site request carries it, and every API request still needs `X-Secret-Key` to reach this + standing at all. It can only narrow a caller that already holds the secret, never admit one that + does not. + +**Consequence to accept.** Two `serve` daemons on one machine, configured with providers of +different tiers, show different subsets of one shared history and knowledge store. That follows +from SD-1, which already made the provider a property of the daemon rather than of the tab. + +Implemented in `crates/biorouter-server/src/auth.rs` (`install_served_operator`, +`served_operator_capability`) and `routes::session_reach::HttpCaller`. Pinned by +`crates/biorouter-server/tests/serve_operator_reach.rs`, which asserts both halves: the interface +keeps its listing and knowledge-base reach, and gains no transcript. + +--- + ## Related documentation - [Architecture of the serving path](serve-architecture.md) — how the decisions above are built. diff --git a/docs/security/privacy-tiers-execution-plan.md b/docs/security/privacy-tiers-execution-plan.md index 827830c01..bf3650f59 100644 --- a/docs/security/privacy-tiers-execution-plan.md +++ b/docs/security/privacy-tiers-execution-plan.md @@ -1393,10 +1393,13 @@ The cost, stated plainly: it; the privacy barrier does not. The app surfaces the refusal string in its `kb_result` error frame rather than failing silently, but it is a working app that stops answering for a reason the app author did not cause and cannot fix. -- **The Knowledge view itself keeps working.** `GET /knowledge/bases/{id}/page`, `/pages`, `/graph`, +- **The Knowledge view itself keeps working.** ~~`GET /knowledge/bases/{id}/page`, `/pages`, `/graph`, `/history`, `/preview`, `/export` are not gated (Task 10C's second ⚠): the user reading their own - notes is not a model. So the base is not *lost* — it is unreachable to models on a public - capability, and readable by hand. + notes is not a model.~~ **Since 2026-09-11 they are gated on who is asking. The Knowledge view + sends the user-action proof and still reads everything; a caller holding only the daemon secret + is refused a private base** (QA finding H2; see Task 10C's superseded ⚠). So the base is not + *lost*. It is unreachable to models on a public capability, and to anything that merely holds the + secret, and readable by hand. - The repair is the same one every other private surface offers — switch the chat to a private model — and it is discoverable, because the refusal string names it. It is still a real loss of ergonomics for a user whose default model is commercial and whose knowledge base has one private @@ -5155,7 +5158,7 @@ choke points cover everything" (they do not — they cover everything they were | `export_app` → `export_brkb` | **CP4** | Complete for the drafter's **content** door: `knowledge_service_for_export` has exactly one caller. | | A base's **id and name** — `list_platform_catalog`, `validate::check_*` rejection strings, `capability_report` | **CP5** (Task 10D) | **Found in round four, not derived in round three.** CP1–CP4 were derived over content and CP5 was not in the enumeration; both of Task 10C's new-surface detectors are structurally blind to it, because neither pattern names `list_bases`. Task 10D adds a metadata detector. | | The no-target/no-primary error id lists — `kb_id_or_primary` `:323-341`, `resolve_target_kb` `:149-159` | **Task 10C** and **Task 11** | Same class as CP5, same blind spot, two more instances. Both were found by sweeping `session_kb_ids` callers by hand; no detector in this plan would have found either. | -| The 7 `/knowledge/*` GUI **read** routes | **nothing, by decision** | The Knowledge view is the user, not a model (Task 10C's ⚠). [Open question 15](#open-questions) records that the asymmetry is undecided in the UI. | +| The 7 `/knowledge/*` GUI **read** routes, and every other route naming a base by `{id}` | ~~nothing, by decision~~ **the HTTP reach gate, since 2026-09-11** | ~~The Knowledge view is the user, not a model (Task 10C's ⚠).~~ QA's H2 measured that premise false: a public chat's shell recovered the secret and read a private base over HTTP. `routes::session_reach::gate_knowledge_base` now covers every `{id}` route, reads and writes, as one route layer. The bases list and `/knowledge/active` omit what the caller cannot reach. The desktop's Knowledge view sends the user-action proof. | | The `/knowledge/*` **write** routes, the CLI's write commands, `soul.rs`, `reset.rs` | **nothing, by decision** | No model is involved; there is no service-level write choke point to hang a raise on (Task 10B's second exclusion list). | | Existence of a base, from a *guessed* id (`create_base`'s "already exists", `resolve_target_kb:141`) | **nothing, by decision** | DR-7 puts side channels out of scope. [AR-5](#ar-5--the-existence-of-a-private-knowledge-base-is-still-inferable). | | A **future** surface of either kind | **a detector, not a construction** | Task 10C's Step 5 has two content detectors (expect 4 and 4); Task 10D's Step 5 has the metadata one, in **two** sweeps (27 hits / 18 production outside `knowledge/`, 22 / 5 inside it — the second added after a single-sweep version proved structurally unable to see `kb_get_active`), plus the metadata register, which classifies *tools* rather than call sites. All are counted enumerations that fail when they grow. That is a tripwire, not coverage. | @@ -7292,6 +7295,26 @@ out of their own notes with no model involved anywhere. The four macro routes ** because those run a model. If you find yourself adding a check to `get_page_body` or `list_pages`, stop: that is a different product decision and it is [Open question 15](#open-questions). +> ⚠ **SUPERSEDED 2026-09-11. The routes are gated now, and the reason this note gave was the defect.** +> The note assumed that a request carrying the daemon secret came from the user. AR-11 had already +> measured that secret to be recoverable from inside the daemon, and DR-17 left the filesystem +> open. On merged `main` at `7c96d796`, QA drove it end to end (finding H2): a public chat's shell +> ran `ps eww`, took the secret, and `curl`ed `/knowledge/bases/{id}/page` for a private base's page. +> The note also clashed with the tier route beside it, which already treated a caller holding only +> the secret as "not a human" for changing a tier (AR-11/AR-15). +> +> The user is now told apart the way every other private surface tells them apart: by the +> user-action proof the desktop sends, or by the private capability a program states +> (`X-Caller-Provider`). One route layer, `routes::session_reach::gate_knowledge_base`, covers every +> route that names a base by `{id}`, **reads and writes alike**, so a caller that may not read a base +> cannot rewrite, restore or delete it either. `GET /knowledge/bases` and `/knowledge/active` omit +> what the caller cannot reach. The Knowledge view still reads everything, because it sends the +> proof. A `biorouter serve` browser keeps its operator's reach under SD-9. So "a barrier there +> would lock a user out of their own notes" did not come true: the barrier is on the caller who +> proves nothing, and the user proves it on every request. Half (b) of +> [Open question 15](#open-questions) is answered by this. Record: +> [`privacy-tiers.md`](privacy-tiers.md#shipped), the knowledge-base tier entry. + - [ ] **Step 1: Write the failing tests** ```rust @@ -22223,7 +22246,7 @@ independent follow-ups. | **12** | **Does `ensure_privacy_schema` co-landing with BR-71 need a merge-order decision?** Both branches add `parent_session_id`; both would take migration 17. The shape-guarded arm plus the unconditional reconcile makes either order safe **in the database**, but the two diffs conflict textually in `session_manager.rs`. Resolution guidance: take either side — the columns are identical — and keep the **shape-guarded** form. | Task 6, and BR-71 Task 1. | | **13** | **Does `medcp`'s continued reachability need a first-run notice, or is the badge enough?** §13.5 specifies a one-time notice naming any **enabled** extension that is Public and declares clinical-looking credentials. On the operator's machine that names exactly one extension, `medcp`, and nothing else changes. | Task 38's notice copy. Hard-code that expectation into its test fixture. | | **14** | **How does `memory`'s local store get a tier?** AR-3: `compose_instructions` (`memory/mod.rs:277`) inlines local memories in full (`:310-322`) into every session opened in that directory, including one on a public model, and Task 19 ships only a disclosure. The design's §9.3 B3 names the fix — "classify memory entries and filter `retrieve_all` by the session's capability tier at init" — but the on-disk format carries no provenance (`:387-388` writes a `# {tags}` line and bare lines; `:414-418` reads them back keyed by the *tag string*), and `compose_instructions` runs once at `MemoryServer::new` (`:108`) rather than per turn, so a naive capability filter there freezes across a mid-session model swap — the O6 hazard. A real fix needs per-entry provenance **and** a per-turn recompute. | Nothing in this plan. Open it as a follow-up issue at Task 40 Step 6. | -| **15** | **(a) ANSWERED by [DR-18](#dr-18--the-knowledge-base-tier-is-user-controllable-and-a-private-session-creates-a-private-base) — half (b) still open.** ~~**Does a knowledge base need a declassification path,~~ and does the barrier belong on the GUI's own read routes?** Two halves of the same scope question. (a) AR-1: a session can be declassified (Task 29, user-only, graded, audited) and a KB cannot, so a user who ratchets their only base by accident has no in-product exit.~~ **It does, and it has one: [Task 29A](#task-29a-knowledge-base-publicize--privatize--user-only-graded-audited).** Half (b) is untouched and still open. (b) Task 10C gates the four `/knowledge/*` **macro** routes (they run a model) and deliberately leaves the GUI's read routes alone (the Knowledge view is the user, not a model) — a defensible line, but it means the *app* shows a private base that the *agent* in the next tab cannot read, and nobody has decided whether that asymmetry should be visible in the UI. | (a) is now [DR-18](#dr-18--the-knowledge-base-tier-is-user-controllable-and-a-private-session-creates-a-private-base) and [Task 29A](#task-29a-knowledge-base-publicize--privatize--user-only-graded-audited). (b) remains a follow-up — and [Task 29A](#task-29a-knowledge-base-publicize--privatize--user-only-graded-audited) makes the asymmetry *more* visible, not less, because the Knowledge view now shows a tier chip the agent obeys and the read routes do not. | +| **15** | **(a) ANSWERED by [DR-18](#dr-18--the-knowledge-base-tier-is-user-controllable-and-a-private-session-creates-a-private-base); (b) ANSWERED 2026-09-11 (QA H2 — see the last column).** ~~**Does a knowledge base need a declassification path,~~ and does the barrier belong on the GUI's own read routes?** Two halves of the same scope question. (a) AR-1: a session can be declassified (Task 29, user-only, graded, audited) and a KB cannot, so a user who ratchets their only base by accident has no in-product exit.~~ **It does, and it has one: [Task 29A](#task-29a-knowledge-base-publicize--privatize--user-only-graded-audited).** Half (b) is untouched and still open. (b) Task 10C gates the four `/knowledge/*` **macro** routes (they run a model) and deliberately leaves the GUI's read routes alone (the Knowledge view is the user, not a model) — a defensible line, but it means the *app* shows a private base that the *agent* in the next tab cannot read, and nobody has decided whether that asymmetry should be visible in the UI. | (a) is now [DR-18](#dr-18--the-knowledge-base-tier-is-user-controllable-and-a-private-session-creates-a-private-base) and [Task 29A](#task-29a-knowledge-base-publicize--privatize--user-only-graded-audited). ~~(b) remains a follow-up — and [Task 29A](#task-29a-knowledge-base-publicize--privatize--user-only-graded-audited) makes the asymmetry *more* visible, not less, because the Knowledge view now shows a tier chip the agent obeys and the read routes do not.~~ **(b) ANSWERED 2026-09-11: yes, for a caller without the user's proof.** The read routes had no barrier on the premise that the Knowledge view is the user. QA's H2 showed that a caller holding only the recoverable secret was treated as that user. The routes now take the reach gate that `GET /sessions/{id}` takes. The desktop, which sends the proof, sees what it always saw, so the app and the agent in the next tab still differ — as the user and a public model should. See Task 10C's superseded ⚠. | | **17** | ✅ **RESOLVED by [DR-17](#scope-ruling--dr-17-narrows-this-plan-to-the-session-store) — the question has no subject.** There is no Linux read-deny in v1, so there is nothing for Landlock to express and no `bubblewrap` dependency to remove. The analysis is kept because it is the measured reason Landlock cannot do this at all, and a revival would otherwise re-derive it. ~~**Should Linux get a Landlock read-deny by granting the complement?** Landlock has no deny rule, so hiding a subpath means handling read accesses and granting read to every sibling of every ancestor of every deny root. Task 14A declines it in v1 for three measured reasons, the disqualifying one being that anything created in an enumerated ancestor *after* the ruleset is built is unreadable for that command's lifetime — `cd ~ && mkdir out && echo x > out/f && cat out/f` fails.~~ | **Nothing — Task 14A is deferred.** Previously: Task 14A makes `bubblewrap` the only Linux mechanism that can express the read-deny, and the refusal names `apt install bubblewrap` as the fix. A Landlock complement would remove that dependency; it needs a real ergonomics trial on a populated `$HOME` before it is worth the failure mode. | | **18** | ⚠ **WIDENED by [DR-17](#scope-ruling--dr-17-narrows-this-plan-to-the-session-store), not resolved.** DR-14 used to remove two of the three local sources of an app id; with the barrier deferred, **all three are open again** — `GET /apps` needs only the secret, the app tree is an ordinary directory, and `agent_drafter__list_apps` is unfiltered because Task 14E is deferred. So any loopback client that can list apps can drive any app's agent socket with no credential. This is squarely inside DR-17's accepted risk and inside [Task 30A](#task-30a-the-non-private-model-disclosure)'s disclosure. Original text: **Should the per-app agent WebSocket be authenticated by something a shell cannot obtain?** `GET /apps/{id}` and `GET /apps/{id}/agent` are deliberately unauthenticated (`auth.rs:52-78`), and `serve_index` (`apps.rs:168-184`) embeds the socket token in the page it serves, so any loopback client that knows an app id can read the token and drive that app's agent. ⚠ **There are THREE local sources of app ids, not two, and this row said two until this round.** DR-14 removes the first two — `GET /apps` needs the secret, and the app tree is deny root #4 — but the third is `agent_drafter__list_apps` (`agent_drafter/mod.rs:2636` → `ArtifactStore::list`, `store.rs:606`), a tool on a **public** extension that takes no path argument, so neither Layer A nor a filesystem deny can see it. Task 14C withdrew that premise; this row had not caught up. What Task 14E changes is narrower than "removes": a public-capability session's `list_apps` no longer returns a **private** app's id, so what stays reachable is that any loopback client — including a public-capability session — can drive a **public** app's agent socket with no credential at all. | Nothing in this plan; the residual is stated in [AR-6](#ar-6--retired-by-dr-17--on-a-host-that-cannot-express-the-read-deny-a-public-session-loses-the-shell-and-two-costs-come-with-the-sandbox-itself) and pinned by Task 14C's `the_unauthenticated_app_surface_does_not_grow_by_accident`. | | **20** | ⚠ **WIDENED by [DR-17](#scope-ruling--dr-17-narrows-this-plan-to-the-session-store), not resolved.** Layer A used to cover the biggest local route, `POST /agent/call_tool`; with it deferred, that route is covered by **Gate C** for private *extensions* and by nothing for private *paths*. The route list below is unchanged and is now the full extent of what a local caller holding the secret can read. Original text: **Should the daemon's HTTP API authenticate a caller that is on the same machine?** [AR-11](#ar-11--amended-by-dr-17--the-daemons-own-api-secret-is-recoverable): the secret is recoverable from the daemon's own environment (`ps -Ewww -p $PPID` on macOS, `/proc/self/environ` in-process on Linux), so `check_token`'s header comparison stops a remote caller and not a local one. Layer A covers the biggest local route, `POST /agent/call_tool`, because that route dispatches through the same choke point. It does **not** cover the routes that return private content without running a tool: `GET /sessions/{id}/export` and the rest of the transcript family, the `/knowledge/*` read routes, `GET /apps/{id}/export`, and `GET /diagnostics/{id}` — which returns a zip of `session.json`, recent `logs/*.jsonl` and a verbatim `config.yaml`, and is the widest single route in the API. | Nothing in this plan. Task 14C states the residual instead of the old "no way to authenticate" claim, and pins the strip so the *remote* half stays closed. Closing the local half needs a per-caller credential the daemon does not hand to its own children — the same shape as [Open question 18](#open-questions), and probably the same fix. | diff --git a/docs/security/privacy-tiers.md b/docs/security/privacy-tiers.md index 8153afa87..7b8a214ef 100644 --- a/docs/security/privacy-tiers.md +++ b/docs/security/privacy-tiers.md @@ -64,6 +64,43 @@ this section is the ledger. written to it, the ratchet fires at four write choke points, the barrier refuses at the read ones, and a refusal names what it refused rather than returning a silently short answer. The user can publicize or privatize a base themselves, graded and audited. + + ⚠ **"The read ones" were the tool path's until 2026-09-11.** Every `/knowledge/bases/{id}/…` HTTP + route was left ungated. The plan's scope note gave the reason: *"the Knowledge view is the user, + not a model"*. That premise was that holding the daemon secret meant being the user, and QA + measured it false on merged `main` at `7c96d796` (H2). A public chat's own shell recovered the + secret with `ps eww`, then read a private base's page, graph, history and `.brkb` export over + HTTP, while `kb_read_page` refused the same chat. The same run found three siblings. `GET + /sessions` listed every private chat with its title and directory (M1). `GET /agent/tools` named + a private chat's private-connector tools (M2). `DELETE /sessions/{id}` deleted a private chat the + read refused, four times out of four (F0). + + They now share **one gate**, `routes::session_reach`'s pure decision: a private target needs the + caller's stated private capability or the user-action proof. It is applied as follows. + + - **Chats.** Every route that names one chat asks `session_reach`: delete, rename, workflow values, + the in-place edit arm, extensions, usage, `/agent/tools`, `/agent/callable_tool_count`, + `/workflows/create`, `/skills/session`, and each chat `ingest-conversation` names. Each refuses + with `GET /sessions/{id}`'s own words, and answers a chat that does not exist the same way. + - **Chat listings.** `GET /sessions`, `/sessions/sidebar` and `/schedule/{id}/sessions` omit the + rows that gate would refuse. + - **Knowledge bases.** One route layer covers every route that names a base by `{id}`, reads and + writes alike. An absent or malformed id is answered as a private one. + - **Knowledge-base listings.** `GET /knowledge/bases` and `/knowledge/active` omit what the + caller cannot reach, and a selection write cannot move a base its caller cannot see. + + The desktop app sends the proof on each of these calls and sees exactly what it saw before. A + `biorouter serve` browser keeps its operator's reach on listings and knowledge bases and gains + no transcript ([SD-9](../deployment/serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else)). + Nothing refused before is permitted now. + + ⚠ **What it does not change**, stated so it is not over-read. Privacy remains a safety boundary + for a cooperating agent, not a security boundary. A public chat with a shell can still read the + knowledge base's files and `sessions.db` directly (DR-17 left the filesystem open; see *Did not + ship*), and a caller holding the secret can still state a private provider in `X-Caller-Provider` + (issue #47). What closed is the path through Biorouter's own API. The routes still open are + listed in `routes/session_reach.rs`'s module header and + [Reaching a private chat from a script](../deployment/programmatic-session-access.md#what-the-header-does-not-cover). - **Declassification (§12), graded** — a `turn:*` chat keeps its single click; every other provenance owes both the typed phrase and R18 / DR-20's operating-system authentication, and one predicate decides both so they cannot drift apart. In the desktop app, and as diff --git a/ui/desktop/openapi.json b/ui/desktop/openapi.json index 1a66a7468..f59304b65 100644 --- a/ui/desktop/openapi.json +++ b/ui/desktop/openapi.json @@ -266,6 +266,9 @@ "401": { "description": "Unauthorized - invalid secret key" }, + "403": { + "description": "Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text)" + }, "424": { "description": "Agent not initialized" } @@ -797,6 +800,9 @@ "401": { "description": "Unauthorized - invalid secret key" }, + "403": { + "description": "Refused by a privacy boundary: `session_id` names a chat this caller may not reach, answered with the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text)" + }, "408": { "description": "Extension timed out while loading for settings" }, @@ -1895,7 +1901,7 @@ ], "responses": { "200": { - "description": "The session's knowledge bases and its primary", + "description": "The session's knowledge bases and its primary, showing only the bases this caller may open: a private base is omitted from both lists, and a private primary reads null, for a caller without the user's proof or a private capability", "content": { "application/json": { "schema": { @@ -1939,7 +1945,7 @@ "description": "Unknown kb id, a primary outside the resulting set, or conflicting primary-KB fields" }, "403": { - "description": "Refused by a privacy boundary (issue #56 Task 58 / #47): `session_id` names a private chat (or an absent one, and an unproven caller is told the same thing for both) and the request carried no proof it came from the user (body = plain text)" + "description": "Refused by a privacy boundary (issue #56 Task 58 / #47): `session_id` names a private chat (or an absent one, and an unproven caller is told the same thing for both) and the request carried no proof it came from the user; or `primary_kb` names a knowledge base this caller may not reach, answered exactly as a base that does not exist (body = plain text)" } } } @@ -1952,7 +1958,7 @@ "operationId": "list_bases", "responses": { "200": { - "description": "List of knowledge bases", + "description": "The knowledge bases this caller may open: every base for the desktop app (the user-action proof) or a caller stating a private provider, the public ones for anyone else. A private base is omitted, never redacted.", "content": { "application/json": { "schema": { @@ -3780,7 +3786,7 @@ ], "responses": { "200": { - "description": "A list of session display info", + "description": "A list of session display info, holding only the runs this caller could open: a private run is omitted for a caller with neither the user-action proof nor a private capability, as it is from `GET /sessions`", "content": { "application/json": { "schema": { @@ -3848,7 +3854,7 @@ ], "responses": { "200": { - "description": "List of available sessions retrieved successfully", + "description": "The sessions this caller could open. A private session is omitted — never redacted — for a caller that carries neither the user-action proof nor a private capability, exactly as `GET /sessions/{session_id}` would refuse it", "content": { "application/json": { "schema": { @@ -4117,7 +4123,7 @@ ], "responses": { "200": { - "description": "Paginated lightweight session summaries for the sidebar", + "description": "Paginated lightweight session summaries for the sidebar, holding only the sessions this caller could open (see `GET /sessions`). `next_offset` is where the next page starts; for a caller shown every session it is `offset + limit` as before, and for one shown a filtered view it is a position in the underlying ordering, so pass it back as given rather than computing it", "content": { "application/json": { "schema": { @@ -4220,6 +4226,9 @@ "401": { "description": "Unauthorized - Invalid or missing API key" }, + "403": { + "description": "Refused by a privacy boundary (issue #56, QA 2026-09-10 F0): the same refusal, word for word, that `GET /sessions/{session_id}` gives — including for a chat that does not exist (body = plain text)" + }, "404": { "description": "Session not found" }, @@ -4545,6 +4554,9 @@ "401": { "description": "Unauthorized - Invalid or missing API key" }, + "403": { + "description": "Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text)" + }, "404": { "description": "Session not found" }, @@ -4596,6 +4608,9 @@ "401": { "description": "Unauthorized - Invalid or missing API key" }, + "403": { + "description": "Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text)" + }, "404": { "description": "Session not found" }, @@ -4644,6 +4659,9 @@ "401": { "description": "Unauthorized - Invalid or missing API key" }, + "403": { + "description": "Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text)" + }, "404": { "description": "Session not found" }, @@ -4699,6 +4717,9 @@ "401": { "description": "Unauthorized - Invalid or missing API key" }, + "403": { + "description": "Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text)" + }, "404": { "description": "Session not found", "content": { @@ -4999,6 +5020,9 @@ "401": { "description": "Unauthorized - invalid or missing secret key" }, + "403": { + "description": "Refused by a privacy boundary: `sessionId` names a chat this caller may not reach, answered with the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text)" + }, "404": { "description": "No such conversation" }, @@ -5257,7 +5281,7 @@ }, "responses": { "200": { - "description": "Workflow created successfully", + "description": "Workflow created successfully. Its `knowledge_bases` names only the bases this caller may open", "content": { "application/json": { "schema": { @@ -5269,6 +5293,9 @@ "400": { "description": "Bad request" }, + "403": { + "description": "Refused by a privacy boundary: `session_id` names a chat this caller may not reach, answered with the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text)" + }, "412": { "description": "Precondition failed - Agent not available" }, @@ -8693,7 +8720,7 @@ } } ], - "description": "One row of `GET /knowledge/bases`: the stored manifest plus the privacy tier\n(issue #56).\n\nThe tier is **flattened alongside** the manifest rather than added to it,\nbecause `manifest.yaml` is the on-disk record and the tier lives in\n`.kb-tiers`. A `tier` field on [`Manifest`] would be persisted by the next\n`manifest::save` and become a second, staler answer to a question the tier\nstore already answers — and it would also appear on `kb_list_bases`, a\nmodel-facing tool whose payload Task 10D's metadata register governs.\n\nThis route is user-facing: the renderer is the only caller, and Task 10C\nalready removes private bases from the model's own listing entirely." + "description": "One row of `GET /knowledge/bases`: the stored manifest plus the privacy tier\n(issue #56).\n\nThe tier is **flattened alongside** the manifest rather than added to it,\nbecause `manifest.yaml` is the on-disk record and the tier lives in\n`.kb-tiers`. A `tier` field on [`Manifest`] would be persisted by the next\n`manifest::save` and become a second, staler answer to a question the tier\nstore already answers — and it would also appear on `kb_list_bases`, a\nmodel-facing tool whose payload Task 10D's metadata register governs.\n\n⚠ **\"The renderer is the only caller\" was this doc's premise, and QA\nmeasured it false on 2026-09-10 (H2):** a public chat's shell recovered the\ndaemon secret and read this list, private bases included. So the rows are\nnow the bases the caller could open — the desktop app, which sends the\nuser's proof, still sees every one, with its tier — and a private base is\nOMITTED for anyone else, as Task 10C already omits it from the model's own\nlisting: a base's id and name are user-authored content." }, "KbTier": { "type": "string", diff --git a/ui/desktop/src/api/types.gen.ts b/ui/desktop/src/api/types.gen.ts index 68b657939..bee349467 100644 --- a/ui/desktop/src/api/types.gen.ts +++ b/ui/desktop/src/api/types.gen.ts @@ -1561,8 +1561,13 @@ export type KbFormat = 'okf' | 'biookf'; * store already answers — and it would also appear on `kb_list_bases`, a * model-facing tool whose payload Task 10D's metadata register governs. * - * This route is user-facing: the renderer is the only caller, and Task 10C - * already removes private bases from the model's own listing entirely. + * ⚠ **"The renderer is the only caller" was this doc's premise, and QA + * measured it false on 2026-09-10 (H2):** a public chat's shell recovered the + * daemon secret and read this list, private bases included. So the rows are + * now the bases the caller could open — the desktop app, which sends the + * user's proof, still sees every one, with its tier — and a private base is + * OMITTED for anyone else, as Task 10C already omits it from the model's own + * listing: a base's id and name are user-authored content. */ export type KbListEntry = Manifest & { tier: KbTier; @@ -4381,6 +4386,10 @@ export type GetCallableToolCountErrors = { * Unauthorized - invalid secret key */ 401: unknown; + /** + * Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text) + */ + 403: unknown; /** * Agent not initialized */ @@ -4795,6 +4804,10 @@ export type GetToolsErrors = { * Unauthorized - invalid secret key */ 401: unknown; + /** + * Refused by a privacy boundary: `session_id` names a chat this caller may not reach, answered with the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text) + */ + 403: unknown; /** * Extension timed out while loading for settings */ @@ -5688,7 +5701,7 @@ export type GetActiveErrors = { export type GetActiveResponses = { /** - * The session's knowledge bases and its primary + * The session's knowledge bases and its primary, showing only the bases this caller may open: a private base is omitted from both lists, and a private primary reads null, for a caller without the user's proof or a private capability */ 200: ActiveKbResponse; }; @@ -5708,7 +5721,7 @@ export type SetActiveErrors = { */ 400: unknown; /** - * Refused by a privacy boundary (issue #56 Task 58 / #47): `session_id` names a private chat (or an absent one, and an unproven caller is told the same thing for both) and the request carried no proof it came from the user (body = plain text) + * Refused by a privacy boundary (issue #56 Task 58 / #47): `session_id` names a private chat (or an absent one, and an unproven caller is told the same thing for both) and the request carried no proof it came from the user; or `primary_kb` names a knowledge base this caller may not reach, answered exactly as a base that does not exist (body = plain text) */ 403: unknown; }; @@ -5731,7 +5744,7 @@ export type ListBasesData = { export type ListBasesResponses = { /** - * List of knowledge bases + * The knowledge bases this caller may open: every base for the desktop app (the user-action proof) or a caller stating a private provider, the public ones for anyone else. A private base is omitted, never redacted. */ 200: Array; }; @@ -7118,7 +7131,7 @@ export type SessionsHandlerErrors = { export type SessionsHandlerResponses = { /** - * A list of session display info + * A list of session display info, holding only the runs this caller could open: a private run is omitted for a caller with neither the user-action proof nor a private capability, as it is from `GET /sessions` */ 200: Array; }; @@ -7182,7 +7195,7 @@ export type ListSessionsErrors = { export type ListSessionsResponses = { /** - * List of available sessions retrieved successfully + * The sessions this caller could open. A private session is omitted — never redacted — for a caller that carries neither the user-action proof nor a private capability, exactly as `GET /sessions/{session_id}` would refuse it */ 200: SessionListResponse; }; @@ -7379,7 +7392,7 @@ export type ListSidebarSessionsErrors = { export type ListSidebarSessionsResponses = { /** - * Paginated lightweight session summaries for the sidebar + * Paginated lightweight session summaries for the sidebar, holding only the sessions this caller could open (see `GET /sessions`). `next_offset` is where the next page starts; for a caller shown every session it is `offset + limit` as before, and for one shown a filtered view it is a position in the underlying ordering, so pass it back as given rather than computing it */ 200: SidebarSessionListResponse; }; @@ -7403,6 +7416,10 @@ export type DeleteSessionErrors = { * Unauthorized - Invalid or missing API key */ 401: unknown; + /** + * Refused by a privacy boundary (issue #56, QA 2026-09-10 F0): the same refusal, word for word, that `GET /sessions/{session_id}` gives — including for a chat that does not exist (body = plain text) + */ + 403: unknown; /** * Session not found */ @@ -7694,6 +7711,10 @@ export type GetSessionExtensionsErrors = { * Unauthorized - Invalid or missing API key */ 401: unknown; + /** + * Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text) + */ + 403: unknown; /** * Session not found */ @@ -7734,6 +7755,10 @@ export type UpdateSessionNameErrors = { * Unauthorized - Invalid or missing API key */ 401: unknown; + /** + * Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text) + */ + 403: unknown; /** * Session not found */ @@ -7772,6 +7797,10 @@ export type GetSessionUsageErrors = { * Unauthorized - Invalid or missing API key */ 401: unknown; + /** + * Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text) + */ + 403: unknown; /** * Session not found */ @@ -7808,6 +7837,10 @@ export type UpdateSessionUserWorkflowValuesErrors = { * Unauthorized - Invalid or missing API key */ 401: unknown; + /** + * Refused by a privacy boundary: the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text) + */ + 403: unknown; /** * Session not found */ @@ -8012,6 +8045,10 @@ export type SetSessionSkillsErrors = { * Unauthorized - invalid or missing secret key */ 401: unknown; + /** + * Refused by a privacy boundary: `sessionId` names a chat this caller may not reach, answered with the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text) + */ + 403: unknown; /** * No such conversation */ @@ -8210,6 +8247,10 @@ export type CreateWorkflowErrors = { * Bad request */ 400: unknown; + /** + * Refused by a privacy boundary: `session_id` names a chat this caller may not reach, answered with the same refusal, word for word, that `GET /sessions/{session_id}` gives (body = plain text) + */ + 403: unknown; /** * Precondition failed - Agent not available */ @@ -8222,7 +8263,7 @@ export type CreateWorkflowErrors = { export type CreateWorkflowResponses = { /** - * Workflow created successfully + * Workflow created successfully. Its `knowledge_bases` names only the bases this caller may open */ 200: CreateWorkflowResponse; }; diff --git a/ui/desktop/src/components/BioRouterSidebar/useSidebarSessions.test.ts b/ui/desktop/src/components/BioRouterSidebar/useSidebarSessions.test.ts index 177faad99..90d6187e1 100644 --- a/ui/desktop/src/components/BioRouterSidebar/useSidebarSessions.test.ts +++ b/ui/desktop/src/components/BioRouterSidebar/useSidebarSessions.test.ts @@ -14,6 +14,13 @@ vi.mock('../../api', () => ({ listSidebarSessions: mocks.listSidebarSessions, })); +// The proof the desktop sends. Since issue #56's QA sweep (2026-09-10) the +// daemon answers a request without it as a public model — private chats and +// knowledge bases omitted or refused — so each call here must carry it. +vi.mock('../../utils/userAction', () => ({ + userActionHeaders: async () => ({ 'X-User-Action': 'test-proof' }), +})); + function makeSummary(index: number): SessionSummary { const timestamp = new Date(Date.parse('2026-07-15T12:00:00.000Z') - index * 60_000).toISOString(); return { @@ -61,6 +68,7 @@ describe('useSidebarSessions', () => { expect(result.current.hasMore).toBe(true); expect(mocks.listSidebarSessions).toHaveBeenNthCalledWith(1, { query: { limit: 10, offset: 0 }, + headers: { 'X-User-Action': 'test-proof' }, throwOnError: true, }); @@ -70,6 +78,7 @@ describe('useSidebarSessions', () => { expect(result.current.hasMore).toBe(false); expect(mocks.listSidebarSessions).toHaveBeenNthCalledWith(2, { query: { limit: 10, offset: 10 }, + headers: { 'X-User-Action': 'test-proof' }, throwOnError: true, }); }); @@ -107,6 +116,7 @@ describe('useSidebarSessions', () => { expect(result.current.sessions).toHaveLength(20); expect(mocks.listSidebarSessions).toHaveBeenNthCalledWith(3, { query: { limit: 10, offset: 0 }, + headers: { 'X-User-Action': 'test-proof' }, throwOnError: true, }); }); diff --git a/ui/desktop/src/components/BioRouterSidebar/useSidebarSessions.ts b/ui/desktop/src/components/BioRouterSidebar/useSidebarSessions.ts index 4acef7279..9cfdf5d4f 100644 --- a/ui/desktop/src/components/BioRouterSidebar/useSidebarSessions.ts +++ b/ui/desktop/src/components/BioRouterSidebar/useSidebarSessions.ts @@ -1,5 +1,6 @@ import { useCallback, useEffect, useRef, useState } from 'react'; import { listSidebarSessions, type SessionSummary } from '../../api'; +import { userActionHeaders } from '../../utils/userAction'; import { subscribeSessionNameChanges } from '../../utils/sessionNameSync'; import { subscribeSessionListChanges } from '../../utils/sessionListCache'; @@ -50,8 +51,11 @@ export default function useSidebarSessions(): SidebarSessionsState { setIsLoading(true); try { + // With the user's proof: without it the daemon pages a view with every + // private chat omitted (issue #56, QA 2026-09-10 M1). const response = await listSidebarSessions({ query: { limit: SIDEBAR_SESSION_PAGE_SIZE, offset }, + headers: await userActionHeaders(), throwOnError: true, }); const page = response.data; diff --git a/ui/desktop/src/components/MentionPopover.tsx b/ui/desktop/src/components/MentionPopover.tsx index b49b56b7f..e06972434 100644 --- a/ui/desktop/src/components/MentionPopover.tsx +++ b/ui/desktop/src/components/MentionPopover.tsx @@ -11,6 +11,7 @@ import { createPortal } from 'react-dom'; import { ItemIcon } from './ItemIcon'; import BuiltInBadge from './ui/BuiltInBadge'; import { CommandType, getActive, getSessionExtensions, getSlashCommands, listBases } from '../api'; +import { userActionHeaders } from '../utils/userAction'; import type { CatalogView } from '../api'; import { getInitialWorkingDir } from '../utils/workingDir'; import { IMAGE_EXTENSIONS } from '../utils/imageFormats'; @@ -543,14 +544,20 @@ const MentionPopover = forwardRef< const loadReferenceItems = useCallback( async (includeCommands: boolean) => { + // The user's proof on the three reads below that name a chat or its + // knowledge bases: since issue #56's QA sweep (2026-09-10) a request + // without it is answered as a public model, and would be shown no + // private base and no private chat's extensions. + const headers = await userActionHeaders(); const [commandsResponse, basesResponse, activeResponse, skillsResult, sessionExtensions] = await Promise.all([ includeCommands ? getSlashCommands({ throwOnError: true }) : Promise.resolve({ data: { commands: [] } }), - listBases({ throwOnError: false }), + listBases({ headers, throwOnError: false }), getActive({ query: sessionId ? { session_id: sessionId } : undefined, + headers, throwOnError: false, }), // The daemon's catalog, not a renderer scan: a skill bundled inside @@ -560,7 +567,7 @@ const MentionPopover = forwardRef< () => ({ generation: 0, roots: [], skills: [], bundles: [] }) as CatalogView ), sessionId - ? getSessionExtensions({ path: { session_id: sessionId } }).catch(() => null) + ? getSessionExtensions({ path: { session_id: sessionId }, headers }).catch(() => null) : Promise.resolve(null), ]); const commandItems: DisplayItem[] = (commandsResponse.data?.commands || []) diff --git a/ui/desktop/src/components/alerts/useToolCount.ts b/ui/desktop/src/components/alerts/useToolCount.ts index ed9ee2834..bdb942b93 100644 --- a/ui/desktop/src/components/alerts/useToolCount.ts +++ b/ui/desktop/src/components/alerts/useToolCount.ts @@ -1,5 +1,6 @@ import { useState, useEffect } from 'react'; import { getCallableToolCount } from '../../api'; +import { userActionHeaders } from '../../utils/userAction'; import { CATALOG_CHANGED_EVENT } from '../../utils/catalogSubscription'; import { SESSION_TOOLS_CHANGED_EVENT, @@ -35,8 +36,11 @@ export const useToolCount = (sessionId: string, agentReady: boolean = true) => { controller?.abort(); controller = new AbortController(); try { + // With the user's proof: a private chat refuses this read to a caller + // without it (issue #56, QA 2026-09-10 M2). const response = await getCallableToolCount({ query: { session_id: sessionId }, + headers: await userActionHeaders(), signal: controller.signal, }); if (cancelled || requestRevision !== revision) return; diff --git a/ui/desktop/src/components/bottom_menu/BottomMenuExtensionSelection.test.tsx b/ui/desktop/src/components/bottom_menu/BottomMenuExtensionSelection.test.tsx index 9c213dbdd..f95f60a40 100644 --- a/ui/desktop/src/components/bottom_menu/BottomMenuExtensionSelection.test.tsx +++ b/ui/desktop/src/components/bottom_menu/BottomMenuExtensionSelection.test.tsx @@ -106,6 +106,13 @@ vi.mock('../../api', () => ({ getSessionExtensions: mocks.getSessionExtensions, })); +// The proof the desktop sends. Since issue #56's QA sweep (2026-09-10) the +// daemon answers a request without it as a public model — private chats and +// knowledge bases omitted or refused — so each call here must carry it. +vi.mock('../../utils/userAction', () => ({ + userActionHeaders: async () => ({ 'X-User-Action': 'test-proof' }), +})); + vi.mock('../settings/extensions/agent-api', () => ({ addToAgent: mocks.addToAgent, removeFromAgent: mocks.removeFromAgent, @@ -355,6 +362,7 @@ describe('BottomMenuExtensionSelection', () => { ); expect(mocks.getSessionExtensions).toHaveBeenLastCalledWith({ path: { session_id: 'session-1' }, + headers: { 'X-User-Action': 'test-proof' }, }); await waitFor(() => expect(example).toHaveAttribute('aria-checked', 'true')); expect(screen.getByLabelText('Manage extensions (1 enabled)')).toBeInTheDocument(); @@ -400,6 +408,7 @@ describe('BottomMenuExtensionSelection', () => { ); expect(mocks.getSessionExtensions).toHaveBeenLastCalledWith({ path: { session_id: 'session-1' }, + headers: { 'X-User-Action': 'test-proof' }, }); await waitFor(() => expect(screen.getByLabelText('Manage extensions (1 enabled)')).toBeInTheDocument() diff --git a/ui/desktop/src/components/bottom_menu/BottomMenuExtensionSelection.tsx b/ui/desktop/src/components/bottom_menu/BottomMenuExtensionSelection.tsx index 43aaa14fa..d964f7527 100644 --- a/ui/desktop/src/components/bottom_menu/BottomMenuExtensionSelection.tsx +++ b/ui/desktop/src/components/bottom_menu/BottomMenuExtensionSelection.tsx @@ -19,6 +19,7 @@ import { isBuiltInExtension, } from '../settings/extensions/subcomponents/ExtensionList'; import { ExtensionConfig, getSessionExtensions } from '../../api'; +import { userActionHeaders } from '../../utils/userAction'; import type { SessionClassification } from '../../api/types.gen'; import { addToAgent, removeFromAgent } from '../settings/extensions/agent-api'; import { extensionPairingRefused } from '../settings/extensions/extensionPrivacy'; @@ -140,8 +141,11 @@ export const BottomMenuExtensionSelection = ({ } try { + // With the user's proof: a private chat's extensions are refused to a + // caller without it (issue #56, QA 2026-09-10). const response = await getSessionExtensions({ path: { session_id: sessionId }, + headers: await userActionHeaders(), }); if (current && response.data?.extensions) { @@ -214,7 +218,10 @@ export const BottomMenuExtensionSelection = ({ if (sessionToggleChainsRef.current.get(name) !== operation) return; try { - const response = await getSessionExtensions({ path: { session_id: sessionId } }); + const response = await getSessionExtensions({ + path: { session_id: sessionId }, + headers: await userActionHeaders(), + }); if (sessionToggleChainsRef.current.get(name) !== operation) return; if (response.data?.extensions) { setSessionExtensions(response.data.extensions); @@ -425,7 +432,10 @@ export const BottomMenuExtensionSelection = ({ : removeFromAgent(ext.name, sessionId, true) ) ); - const response = await getSessionExtensions({ path: { session_id: sessionId } }); + const response = await getSessionExtensions({ + path: { session_id: sessionId }, + headers: await userActionHeaders(), + }); if (response.data?.extensions) { setSessionExtensions(response.data.extensions); setSessionExtensionsLoaded(true); diff --git a/ui/desktop/src/components/knowledge/KbTierControl.tsx b/ui/desktop/src/components/knowledge/KbTierControl.tsx index f3f6f58f2..e6ea4ef20 100644 --- a/ui/desktop/src/components/knowledge/KbTierControl.tsx +++ b/ui/desktop/src/components/knowledge/KbTierControl.tsx @@ -167,7 +167,11 @@ export function KbTierPanel({ kb }: { kb: { id: string; name: string; tier: KbTi setRadius(null); void (async () => { try { - const res = await getKbTier({ path: { id: kb.id }, throwOnError: true }); + const res = await getKbTier({ + path: { id: kb.id }, + headers: await userActionHeaders(), + throwOnError: true, + }); if (cancelled) return; setRadius({ pageCount: res.data.page_count, diff --git a/ui/desktop/src/components/knowledge/KnowledgeContext.tsx b/ui/desktop/src/components/knowledge/KnowledgeContext.tsx index 34b513075..2ee5554d6 100644 --- a/ui/desktop/src/components/knowledge/KnowledgeContext.tsx +++ b/ui/desktop/src/components/knowledge/KnowledgeContext.tsx @@ -186,6 +186,7 @@ export function KnowledgeProvider({ try { const res = await getActive({ query: sessionId ? { session_id: sessionId } : undefined, + headers: await userActionHeaders(), throwOnError: true, }); if (generation !== selectionGenerationRef.current) return; @@ -323,7 +324,11 @@ export function KnowledgeProvider({ return; } try { - const res = await getActive({ query: undefined, throwOnError: true }); + const res = await getActive({ + query: undefined, + headers: await userActionHeaders(), + throwOnError: true, + }); setDefaultPrimaryKbId(readPrimary(res.data)); } catch (err) { // Keep the last known default: a failed read is not evidence that there @@ -373,7 +378,15 @@ export function KnowledgeProvider({ const refresh = useCallback(async () => { setLoading(true); try { - const res = await listBases({ throwOnError: true }); + // ⚠ With the user's proof, and it is load-bearing twice over. Since + // issue #56's QA sweep (2026-09-10) the daemon OMITS a private base from + // a caller without it — and the two effects below prune the selection + // against this list, so a list missing a base reads as "that base was + // deleted" and would drop it from the primary and the hidden set. The + // daemon also refuses to let such a caller move what it cannot see + // (`set_selection_within`), but the list the user is shown must be the + // whole one. + const res = await listBases({ headers: await userActionHeaders(), throwOnError: true }); setBases(res.data || []); setBasesLoaded(true); setBasesError(null); @@ -455,6 +468,7 @@ export function KnowledgeProvider({ try { const res = await getActive({ query: sessionId ? { session_id: sessionId } : undefined, + headers: await userActionHeaders(), throwOnError: true, }); if (cancelled || generation !== selectionGenerationRef.current) return; diff --git a/ui/desktop/src/components/knowledge/KnowledgeView.tsx b/ui/desktop/src/components/knowledge/KnowledgeView.tsx index 41757c7a9..ef694f668 100644 --- a/ui/desktop/src/components/knowledge/KnowledgeView.tsx +++ b/ui/desktop/src/components/knowledge/KnowledgeView.tsx @@ -14,6 +14,7 @@ import { Trash2, } from '../icons/app-icons'; import { getLocation } from '../../api'; +import { userActionHeaders } from '../../utils/userAction'; import { Button } from '../ui/button'; import { PrivacyBadge } from '../ui/PrivacyBadge'; import { EmptyState } from '../ui/empty-state'; @@ -117,7 +118,11 @@ function KnowledgeViewInner() { async function openKbFolder() { if (!primaryKbId) return; try { - const res = await getLocation({ path: { id: primaryKbId }, throwOnError: true }); + const res = await getLocation({ + path: { id: primaryKbId }, + headers: await userActionHeaders(), + throwOnError: true, + }); const path = res.data?.path; if (path) await window.electron.openDirectoryInExplorer(path); } catch (err) { diff --git a/ui/desktop/src/components/knowledge/hooks/knowledgeRequest.ts b/ui/desktop/src/components/knowledge/hooks/knowledgeRequest.ts index 1d43fbb53..6a342e316 100644 --- a/ui/desktop/src/components/knowledge/hooks/knowledgeRequest.ts +++ b/ui/desktop/src/components/knowledge/hooks/knowledgeRequest.ts @@ -1,4 +1,5 @@ import { client } from '../../../api/client.gen'; +import { userActionHeaders } from '../../../utils/userAction'; type ElectronBridge = { getBiorouterdHostPort?: () => Promise; @@ -42,12 +43,26 @@ export async function buildKnowledgeUrl(path: string): Promise { return `${await getBackendBaseUrl()}${path}`; } +/** + * A request to the daemon's `/knowledge/*` routes, as the Knowledge view makes + * it: with the secret, and with the user-action proof. + * + * Issue #56, QA 2026-09-10 H2: every route that names a knowledge base now + * answers a caller WITHOUT the proof as a public model — a private base is + * refused and omitted from listings — because a public chat's shell could + * recover the secret and read private bases with it. The desktop is the person + * at the keyboard, so it says so on every knowledge request; a request that + * forgot would see private bases vanish, not an error. + */ export async function knowledgeFetch(path: string, init: RequestInit = {}): Promise { const headers = new Headers(init.headers ?? {}); const secret = await getSecretKey(); if (secret) { headers.set('X-Secret-Key', secret); } + for (const [name, value] of Object.entries(await userActionHeaders())) { + headers.set(name, value); + } return fetch(await buildKnowledgeUrl(path), { ...init, diff --git a/ui/desktop/src/components/knowledge/hooks/useHistory.ts b/ui/desktop/src/components/knowledge/hooks/useHistory.ts index 0d569d1f2..b7ad10c0e 100644 --- a/ui/desktop/src/components/knowledge/hooks/useHistory.ts +++ b/ui/desktop/src/components/knowledge/hooks/useHistory.ts @@ -1,5 +1,6 @@ import { useCallback, useEffect, useState } from 'react'; import { listHistory, restoreState } from '../../../api'; +import { userActionHeaders } from '../../../utils/userAction'; import type { HistoryEntry, RestoreResponse } from '../../../api/types.gen'; export interface UseHistoryResult { @@ -26,6 +27,7 @@ export function useHistory(kbId: string | null): UseHistoryResult { const res = await listHistory({ path: { id: kbId }, query: { limit: 200 }, + headers: await userActionHeaders(), throwOnError: true, }); // ListHistoryResponses[200] is typed `unknown` in the generated SDK, @@ -46,6 +48,7 @@ export function useHistory(kbId: string | null): UseHistoryResult { const res = await restoreState({ path: { id: kbId }, body: { commit_sha: commitSha }, + headers: await userActionHeaders(), throwOnError: true, }); const sha = (res.data as RestoreResponse | undefined)?.new_commit_sha ?? ''; diff --git a/ui/desktop/src/components/knowledge/hooks/useIngestStream.ts b/ui/desktop/src/components/knowledge/hooks/useIngestStream.ts index 6b5c1fb93..0b454e0dc 100644 --- a/ui/desktop/src/components/knowledge/hooks/useIngestStream.ts +++ b/ui/desktop/src/components/knowledge/hooks/useIngestStream.ts @@ -1,5 +1,6 @@ import { useCallback, useRef, useState } from 'react'; import { buildKnowledgeUrl, getSecretKey } from './knowledgeRequest'; +import { userActionHeaders } from '../../../utils/userAction'; export type SubAgentEvent = | { kind: 'step'; index: number; assistant_text: string } @@ -74,12 +75,17 @@ export function useIngestStream() { // `cfg.headers as Record` is unreliable because // HeadersInit can be a Headers instance or a [string,string][] array. const xSecretKey = await getSecretKey(); + // The user-action proof, as `knowledgeFetch` sends it: a macro names a + // base, and since issue #56's QA sweep (2026-09-10) a request without + // the proof is answered as a public model, which a private base refuses. + const proof = await userActionHeaders(); try { const res = await fetch(url, { method: 'POST', headers: { 'X-Secret-Key': xSecretKey, + ...proof, ...(requestInit.headers ?? {}), }, body: requestInit.body, diff --git a/ui/desktop/src/components/knowledge/hooks/useKnowledgeBases.ts b/ui/desktop/src/components/knowledge/hooks/useKnowledgeBases.ts index b9be11a26..fc2ae969e 100644 --- a/ui/desktop/src/components/knowledge/hooks/useKnowledgeBases.ts +++ b/ui/desktop/src/components/knowledge/hooks/useKnowledgeBases.ts @@ -3,6 +3,7 @@ import { createBase as apiCreate, deleteBase as apiDelete } from '../../../api'; import { useKnowledge } from '../KnowledgeContext'; import type { KbFormat, Manifest } from '../../../api/types.gen'; import { knowledgeFetch } from './knowledgeRequest'; +import { userActionHeaders } from '../../../utils/userAction'; export function useKnowledgeBases() { const { refresh, setPrimaryKbId, primaryKbId } = useKnowledge(); @@ -62,7 +63,7 @@ export function useKnowledgeBases() { const remove = useCallback( async (id: string): Promise => { - await apiDelete({ throwOnError: true, path: { id } }); + await apiDelete({ throwOnError: true, path: { id }, headers: await userActionHeaders() }); if (primaryKbId === id) setPrimaryKbId(null); await refresh(); }, diff --git a/ui/desktop/src/components/knowledge/hooks/useKnowledgeGraph.ts b/ui/desktop/src/components/knowledge/hooks/useKnowledgeGraph.ts index 793cf2746..7bb2e544e 100644 --- a/ui/desktop/src/components/knowledge/hooks/useKnowledgeGraph.ts +++ b/ui/desktop/src/components/knowledge/hooks/useKnowledgeGraph.ts @@ -1,5 +1,6 @@ import { useCallback, useEffect, useState } from 'react'; import { getGraph } from '../../../api'; +import { userActionHeaders } from '../../../utils/userAction'; import type { Graph } from '../../../api/types.gen'; export interface UseKnowledgeGraphResult { @@ -22,7 +23,13 @@ export function useKnowledgeGraph(kbId: string | null): UseKnowledgeGraphResult setLoading(true); setError(null); try { - const res = await getGraph({ path: { id: kbId }, throwOnError: true }); + // With the user's proof: a private base is refused to any caller without + // it (issue #56, QA 2026-09-10 H2), and this view is the user. + const res = await getGraph({ + path: { id: kbId }, + headers: await userActionHeaders(), + throwOnError: true, + }); setGraph(res.data ?? null); } catch (err) { setError(err instanceof Error ? err.message : String(err)); diff --git a/ui/desktop/src/components/knowledge/hooks/usePagePreview.ts b/ui/desktop/src/components/knowledge/hooks/usePagePreview.ts index a214bfbb6..28f17d929 100644 --- a/ui/desktop/src/components/knowledge/hooks/usePagePreview.ts +++ b/ui/desktop/src/components/knowledge/hooks/usePagePreview.ts @@ -1,6 +1,7 @@ // ui/desktop/src/components/knowledge/hooks/usePagePreview.ts import { useEffect, useState } from 'react'; import { getPageBody, previewState } from '../../../api'; +import { userActionHeaders } from '../../../utils/userAction'; export interface UsePagePreviewResult { content: string | null; @@ -28,15 +29,20 @@ export function usePagePreview( setError(null); (async () => { try { + // With the user's proof: a private base's pages are refused to any + // caller without it (issue #56, QA 2026-09-10 H2). + const headers = await userActionHeaders(); const res = previewSha ? await previewState({ path: { id: kbId }, body: { commit_sha: previewSha, path }, + headers, throwOnError: true, }) : await getPageBody({ path: { id: kbId }, query: { path }, + headers, throwOnError: true, }); if (!cancelled) setContent(res.data?.content ?? null); diff --git a/ui/desktop/src/components/privacy/FirstRunPrivacyNotice.tsx b/ui/desktop/src/components/privacy/FirstRunPrivacyNotice.tsx index b3a25ca7c..cc83ed298 100644 --- a/ui/desktop/src/components/privacy/FirstRunPrivacyNotice.tsx +++ b/ui/desktop/src/components/privacy/FirstRunPrivacyNotice.tsx @@ -2,6 +2,7 @@ import { useEffect, useState } from 'react'; import { Dialog, DialogContent, DialogDescription, DialogHeader, DialogTitle } from '../ui/dialog'; import { Button } from '../ui/button'; import { listSessions, type Session } from '../../api'; +import { userActionHeaders } from '../../utils/userAction'; /** * The numbers the day-one notice quotes, over the population **History actually @@ -187,8 +188,11 @@ export function shouldShowFirstRunNotice(counts: NoticeCounts): boolean { * answer, and touches no shared state. It costs one GET, once per install. */ async function fetchVisibleSessions(): Promise { + // With the user's proof: without it the daemon omits every private chat — + // the chats this notice exists to count (issue #56, QA 2026-09-10 M1). const response = await listSessions({ throwOnError: true, + headers: await userActionHeaders(), query: { include_subagents: false }, }); return response.data.sessions; diff --git a/ui/desktop/src/components/privacy/FirstRunPrivacyNoticeGate.tsx b/ui/desktop/src/components/privacy/FirstRunPrivacyNoticeGate.tsx index f500d0d82..93f6c11d1 100644 --- a/ui/desktop/src/components/privacy/FirstRunPrivacyNoticeGate.tsx +++ b/ui/desktop/src/components/privacy/FirstRunPrivacyNoticeGate.tsx @@ -8,6 +8,7 @@ import { type NoticeCounts, } from './FirstRunPrivacyNotice'; import { listSessions, type Session } from '../../api'; +import { userActionHeaders } from '../../utils/userAction'; /** * Where "this machine has already been told" is recorded. @@ -83,7 +84,12 @@ export function FirstRunPrivacyNoticeGate() { useEffect(() => { if (dismissed) return; let cancelled = false; - listSessions({ throwOnError: true, query: { include_subagents: false } }) + // With the user's proof: without it the daemon omits every private chat, + // and the notice would count none (issue #56, QA 2026-09-10 M1). + userActionHeaders() + .then((headers) => + listSessions({ throwOnError: true, headers, query: { include_subagents: false } }) + ) .then((response) => { if (!cancelled) setCounts(computeNoticeCounts(response.data.sessions as Session[])); }) diff --git a/ui/desktop/src/components/sessions/SessionListView.test.tsx b/ui/desktop/src/components/sessions/SessionListView.test.tsx index 48cc1fc82..3a3beb44c 100644 --- a/ui/desktop/src/components/sessions/SessionListView.test.tsx +++ b/ui/desktop/src/components/sessions/SessionListView.test.tsx @@ -30,6 +30,13 @@ vi.mock('../../toasts', () => ({ toastError: mocks.toastError, })); +// The proof the desktop sends. Since issue #56's QA sweep (2026-09-10) the +// daemon answers a request without it as a public model — private chats and +// knowledge bases omitted or refused — so each call here must carry it. +vi.mock('../../utils/userAction', () => ({ + userActionHeaders: async () => ({ 'X-User-Action': 'test-proof' }), +})); + vi.mock('../conversation/SearchView', () => ({ SearchView: ({ children }: { children: ReactNode }) => <>{children}, })); @@ -126,7 +133,12 @@ describe('SessionListView loading and cache', () => { expect(screen.getByText('Cached conversation')).toBeInTheDocument(); expect(screen.queryByRole('status', { name: 'Loading chat history' })).not.toBeInTheDocument(); - expect(mocks.listSessions).toHaveBeenCalledTimes(2); + // The revalidation leaves one async hop after mount — it waits for the + // user's proof, which it must carry — so it is awaited rather than assumed. + await waitFor(() => expect(mocks.listSessions).toHaveBeenCalledTimes(2)); + expect(mocks.listSessions).toHaveBeenLastCalledWith( + expect.objectContaining({ headers: { 'X-User-Action': 'test-proof' } }) + ); await act(async () => { finishRefresh?.({ data: { sessions: [session] } }); @@ -284,6 +296,7 @@ describe('SessionListView row actions', () => { ); expect(mocks.deleteSession).toHaveBeenCalledWith({ path: { session_id: session.id }, + headers: { 'X-User-Action': 'test-proof' }, throwOnError: true, }); }); @@ -511,6 +524,7 @@ describe('SessionListView row actions', () => { expect(mocks.updateSessionName).toHaveBeenCalledWith({ path: { session_id: session.id }, body: { name: 'Updated session name' }, + headers: { 'X-User-Action': 'test-proof' }, throwOnError: true, }); }); diff --git a/ui/desktop/src/components/sessions/SessionListView.tsx b/ui/desktop/src/components/sessions/SessionListView.tsx index 083e84ee6..4452c84df 100644 --- a/ui/desktop/src/components/sessions/SessionListView.tsx +++ b/ui/desktop/src/components/sessions/SessionListView.tsx @@ -47,6 +47,7 @@ import { ExtensionConfig, ExtensionData, } from '../../api'; +import { userActionHeaders } from '../../utils/userAction'; import { formatExtensionName } from '../settings/extensions/subcomponents/ExtensionList'; import { getSearchShortcutText } from '../../utils/keyboardShortcuts'; import { ReadableContent } from '../Layout/ReadableContent'; @@ -954,8 +955,11 @@ const SessionListView: React.FC = React.memo(({ onSelectSe setSessionToDelete(null); try { + // With the user's proof: deleting a private chat is refused, exactly as + // reading it is, to a caller without it (issue #56, QA 2026-09-10 F0). await deleteSession({ path: { session_id: sessionToDeleteId }, + headers: await userActionHeaders(), throwOnError: true, }); const removeDeletedSession = (currentSessions: Session[]) => @@ -985,8 +989,12 @@ const SessionListView: React.FC = React.memo(({ onSelectSe const handleExportSession = useCallback(async (session: Session, e: React.MouseEvent) => { e.stopPropagation(); + // With the user's proof, like every read of a chat's transcript: the + // export route has refused a private chat to a caller without it since the + // reach gate's export sweep, and this is the person at the keyboard. const response = await exportSession({ path: { session_id: session.id }, + headers: await userActionHeaders(), throwOnError: true, }); diff --git a/ui/desktop/src/components/skills/useSkillCatalog.ts b/ui/desktop/src/components/skills/useSkillCatalog.ts index e5a5f4d8a..db0e29191 100644 --- a/ui/desktop/src/components/skills/useSkillCatalog.ts +++ b/ui/desktop/src/components/skills/useSkillCatalog.ts @@ -49,6 +49,7 @@ import { useCallback, useEffect, useMemo, useRef, useState } from 'react'; import type { CatalogBundle, CatalogSkill, CatalogView, SkillRoot } from '../../api'; import { refreshSkillCatalog, setSessionSkills, skillCatalogHandler } from '../../api'; +import { userActionHeaders } from '../../utils/userAction'; import { CATALOG_CHANGED_EVENT } from '../../utils/catalogSubscription'; import { isContextBundle, isContextSkill } from '../settings/contexts/contexts'; import { @@ -255,12 +256,15 @@ export function useSkillCatalog(sessionId: string | null): SkillCatalogState { try { if (sessionId) { + // With the user's proof: a private chat refuses this write to a + // caller without it (issue #56, QA 2026-09-10). const response = await setSessionSkills({ body: { sessionId, add: enabled ? keys : [], remove: enabled ? [] : keys, }, + headers: await userActionHeaders(), throwOnError: true, }); commit(response.data.catalog); diff --git a/ui/desktop/src/components/subagent/useSubagentSession.ts b/ui/desktop/src/components/subagent/useSubagentSession.ts index 029c3ac64..b28531bae 100644 --- a/ui/desktop/src/components/subagent/useSubagentSession.ts +++ b/ui/desktop/src/components/subagent/useSubagentSession.ts @@ -112,8 +112,12 @@ export function useSubagentSession(sessionId: string): SubagentSessionInfo { (m) => m?.metadata?.provenance?.kind === 'spawn_context' ); const spawnContext = record?.content?.map((c) => ('text' in c ? c.text : '')).join('\n'); - const extensionsResponse = (await getSessionExtensions({ path: { session_id: sessionId } })) - .data; + const extensionsResponse = ( + await getSessionExtensions({ + path: { session_id: sessionId }, + headers: await userActionHeaders(), + }) + ).data; if (cancelled) return; setInfo({ isSubagent: true, diff --git a/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx b/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx index b17c85653..d042b1dbb 100644 --- a/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx +++ b/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx @@ -6,6 +6,7 @@ import { Button } from '../ui/button'; import { WorkflowFormFields } from './shared/WorkflowFormFields'; import { WorkflowFormData } from './shared/workflowFormSchema'; import { createWorkflow, getActive, getSessionExtensions, listBases } from '../../api/sdk.gen'; +import { userActionHeaders } from '../../utils/userAction'; import { WorkflowParameter } from './shared/workflowFormSchema'; import { toastError } from '../../toasts'; import { saveWorkflow } from '../../workflow/workflow_management'; @@ -97,37 +98,51 @@ export default function CreateWorkflowFromSessionModal({ setAnalysisStage(stages[currentStageIndex]); }, 800); + // The user's proof, on every request below that names this chat or its + // knowledge bases: since issue #56's QA sweep (2026-09-10) a request + // without it is answered as a public model, and a private chat — the + // chat this modal is opened from — would refuse all four. + const proof = userActionHeaders(); + // Pre-select session extensions immediately — independent of workflow analysis - getSessionExtensions({ path: { session_id: sessionId }, throwOnError: false }).then((res) => { - if (cancelled) return; - if (res.data?.extensions) { - setWorkflowExtensions(res.data.extensions); - } - }); + void proof + .then((headers) => + getSessionExtensions({ path: { session_id: sessionId }, headers, throwOnError: false }) + ) + .then((res) => { + if (cancelled) return; + if (res.data?.extensions) { + setWorkflowExtensions(res.data.extensions); + } + }); - Promise.all([ - listBases({ throwOnError: false }), - getActive({ query: { session_id: sessionId }, throwOnError: false }), - ]).then(([basesRes, activeRes]) => { - if (cancelled) return; - const bases: Manifest[] = basesRes.data ?? []; - const hidden = new Set(activeRes.data?.hidden_kbs ?? []); - const visible = bases.filter((base) => !hidden.has(base.id)).map((base) => base.id); - // The captured default is the session's primary; `active_kb` is the - // deprecated mirror, read so a fresh renderer survives an older daemon. - const primary = activeRes.data?.primary_kb ?? activeRes.data?.active_kb ?? null; - const defaultId = primary && visible.includes(primary) ? primary : (visible[0] ?? null); - - setKnowledgeBaseItems( - bases.map((base) => ({ - id: base.id, - label: base.name, - description: base.id, - })) - ); - setWorkflowKnowledgeBaseIds(visible); - setDefaultKnowledgeBaseId(defaultId); - }); + void proof + .then((headers) => + Promise.all([ + listBases({ headers, throwOnError: false }), + getActive({ query: { session_id: sessionId }, headers, throwOnError: false }), + ]) + ) + .then(([basesRes, activeRes]) => { + if (cancelled) return; + const bases: Manifest[] = basesRes.data ?? []; + const hidden = new Set(activeRes.data?.hidden_kbs ?? []); + const visible = bases.filter((base) => !hidden.has(base.id)).map((base) => base.id); + // The captured default is the session's primary; `active_kb` is the + // deprecated mirror, read so a fresh renderer survives an older daemon. + const primary = activeRes.data?.primary_kb ?? activeRes.data?.active_kb ?? null; + const defaultId = primary && visible.includes(primary) ? primary : (visible[0] ?? null); + + setKnowledgeBaseItems( + bases.map((base) => ({ + id: base.id, + label: base.name, + description: base.id, + })) + ); + setWorkflowKnowledgeBaseIds(visible); + setDefaultKnowledgeBaseId(defaultId); + }); // The daemon's catalog, so a skill bundled inside an installed extension // can be attached to a workflow like any other (#113). @@ -154,10 +169,14 @@ export default function CreateWorkflowFromSessionModal({ }); // Analyze the conversation to generate a suggested workflow - createWorkflow({ - body: { session_id: sessionId }, - throwOnError: true, - }) + proof + .then((headers) => + createWorkflow({ + body: { session_id: sessionId }, + headers, + throwOnError: true, + }) + ) .then((response) => { if (cancelled) return; clearInterval(stageInterval); diff --git a/ui/desktop/src/hooks/chatStreamStore.test.ts b/ui/desktop/src/hooks/chatStreamStore.test.ts index ee26ab45f..4f824c176 100644 --- a/ui/desktop/src/hooks/chatStreamStore.test.ts +++ b/ui/desktop/src/hooks/chatStreamStore.test.ts @@ -942,6 +942,7 @@ describe('ChatStreamRegistry', () => { editType: 'edit', expectedMessageIds: ['u1', 'a1', 'a2'], }, + headers: { 'X-User-Action': 'test-key' }, throwOnError: true, }); }); @@ -992,6 +993,9 @@ describe('ChatStreamRegistry', () => { expect(editMessage).toHaveBeenCalledWith({ path: { session_id: sessionId }, body: { timestamp: 10, editType: 'edit' }, + // The in-place edit asks the read's reach gate since QA's 2026-09-10 + // sweep, so it carries the proof as a branch does. + headers: { 'X-User-Action': 'test-key' }, throwOnError: true, }); const body = vi.mocked(editMessage).mock.calls[0][0].body as Record; diff --git a/ui/desktop/src/hooks/chatStreamStore.tsx b/ui/desktop/src/hooks/chatStreamStore.tsx index 20c084a8e..caf85f520 100644 --- a/ui/desktop/src/hooks/chatStreamStore.tsx +++ b/ui/desktop/src/hooks/chatStreamStore.tsx @@ -332,7 +332,9 @@ async function fetchAllSessions(): Promise<{ id: string; name?: string | null }[ } sessionListInflightAt = now; sessionListInflight = (async () => { - const response = await listSessions({ throwOnError: true }); + // With the user's proof: without it the daemon omits private chats + // (issue #56, QA 2026-09-10 M1). + const response = await listSessions({ throwOnError: true, headers: await userActionHeaders() }); return (response.data?.sessions ?? []) as { id: string; name?: string | null }[]; })(); sessionListInflight.catch(() => { @@ -3653,6 +3655,9 @@ class ChatStreamController { setWorkflowUserParams = async (user_workflow_values: Record): Promise => { if (this.snapshot.session) { + // With the user's proof: this writes into the chat and re-applies its + // workflow, which a private chat refuses to a caller without it (issue + // #56, QA 2026-09-10). await updateSessionUserWorkflowValues({ path: { session_id: this.sessionId, @@ -3660,6 +3665,7 @@ class ChatStreamController { body: { userWorkflowValues: user_workflow_values, }, + headers: await userActionHeaders(), throwOnError: true, }); this.updateSnapshot((prev) => @@ -4177,13 +4183,15 @@ class ChatStreamController { editType, ...(expectedMessageIds ? { expectedMessageIds } : {}), }, + // The proof that the person at the keyboard asked, on both edit types. // Issue #56 DR-19: `diverge` branches this chat into a NEW session that // inherits its provider, so on a private chat it mints a new - // private-capability session and the daemon refuses it without proof the - // request came from the person at the keyboard. `edit` truncates this - // session in place and mints nothing, so it is not gated and does not - // carry the proof. - ...(editType === 'diverge' ? { headers: await userActionHeaders() } : {}), + // private-capability session and the daemon refuses it without the + // proof. `edit` truncates this session in place; it mints nothing, but + // since QA's 2026-09-10 sweep it asks the read's reach gate, because a + // caller that may not read a private chat may not cut its history + // either. + headers: await userActionHeaders(), throwOnError: true, }); diff --git a/ui/desktop/src/hooks/useCostTracking.ts b/ui/desktop/src/hooks/useCostTracking.ts index 2c3f7aa00..a847e6e33 100644 --- a/ui/desktop/src/hooks/useCostTracking.ts +++ b/ui/desktop/src/hooks/useCostTracking.ts @@ -1,6 +1,7 @@ import { useEffect, useState } from 'react'; import { fetchModelPricing } from '../utils/pricing'; import { getSessionUsage, ModelUsageRow, Session } from '../api'; +import { userActionHeaders } from '../utils/userAction'; import { billedTokens } from '../utils/usageAccounting'; export interface ModelCostRow { @@ -166,8 +167,11 @@ export const useCostTracking = ({ session }: UseCostTrackingProps) => { return; } try { + // With the user's proof: a private chat's usage is refused to a caller + // without it (issue #56, QA 2026-09-10). const response = await getSessionUsage({ path: { session_id: sessionId }, + headers: await userActionHeaders(), throwOnError: false, }); const rows = response.data?.models ?? []; diff --git a/ui/desktop/src/hooks/useWorkflowManager.ts b/ui/desktop/src/hooks/useWorkflowManager.ts index accf3e5ec..ccc1928f9 100644 --- a/ui/desktop/src/hooks/useWorkflowManager.ts +++ b/ui/desktop/src/hooks/useWorkflowManager.ts @@ -5,6 +5,7 @@ import { Message } from '../api'; import { substituteParameters } from '../utils/providerUtils'; import { updateSessionUserWorkflowValues } from '../api'; +import { userActionHeaders } from '../utils/userAction'; import { useChatContext } from '../contexts/ChatContext'; import { ChatType } from '../types/chat'; import { toastError, toastSuccess } from '../toasts'; @@ -198,6 +199,9 @@ export const useWorkflowManager = (chat: ChatType, workflow?: Workflow | null) = body: { userWorkflowValues: inputValues, }, + // With the user's proof: a private chat refuses this write to a caller + // without it (issue #56, QA 2026-09-10). + headers: await userActionHeaders(), throwOnError: true, }); let resolvedWorkflow = response.data?.workflow; diff --git a/ui/desktop/src/schedule.ts b/ui/desktop/src/schedule.ts index 2d39debb9..a8a797811 100644 --- a/ui/desktop/src/schedule.ts +++ b/ui/desktop/src/schedule.ts @@ -11,6 +11,7 @@ import { inspectRunningJob as apiInspectRunningJob, SessionDisplayInfo, } from './api'; +import { userActionHeaders } from './utils/userAction'; export interface ScheduledJob { id: string; @@ -142,9 +143,12 @@ export async function getScheduleSessions( scheduleId: string, limit: number ): Promise> { + // With the user's proof: a schedule's private runs are omitted from a caller + // without it (issue #56, QA 2026-09-10 M1). const response = await apiGetScheduleSessions({ path: { id: scheduleId }, query: { limit }, + headers: await userActionHeaders(), throwOnError: true, }); diff --git a/ui/desktop/src/utils/sessionListCache.test.ts b/ui/desktop/src/utils/sessionListCache.test.ts index 6e4f19945..63f3c65b8 100644 --- a/ui/desktop/src/utils/sessionListCache.test.ts +++ b/ui/desktop/src/utils/sessionListCache.test.ts @@ -19,6 +19,14 @@ vi.mock('../api', () => ({ updateSessionName: mocks.updateSessionName, })); +// The proof the desktop sends. Since issue #56's QA sweep (2026-09-10) a list +// request without it is shown no private chat, so every request here must carry +// it — and it arrives one async hop after the call, which is why the assertions +// below wait for `listSessions` rather than expecting it synchronously. +vi.mock('./userAction', () => ({ + userActionHeaders: async () => ({ 'X-User-Action': 'test-proof' }), +})); + beforeEach(() => { vi.clearAllMocks(); clearSessionListCache(); @@ -36,7 +44,10 @@ describe('sessionListCache', () => { preloadSessionList(); const viewLoad = refreshSessionList(); - expect(mocks.listSessions).toHaveBeenCalledTimes(1); + await vi.waitFor(() => expect(mocks.listSessions).toHaveBeenCalledTimes(1)); + expect(mocks.listSessions).toHaveBeenCalledWith( + expect.objectContaining({ headers: { 'X-User-Action': 'test-proof' } }) + ); finishRequest?.({ data: { sessions: [] } }); await viewLoad; expect(getCachedSessionList()).toEqual([]); @@ -103,7 +114,13 @@ describe('sessionListCache', () => { const first = refreshSessionList(); const second = refreshSessionList(true); - expect(mocks.listSessions).toHaveBeenCalledTimes(2); + await vi.waitFor(() => expect(mocks.listSessions).toHaveBeenCalledTimes(2)); + // Each request asks for the list it was issued for, even the orphan whose + // flag changed during the proof's async hop. + expect(mocks.listSessions.mock.calls.map(([options]) => options.query)).toEqual([ + { include_subagents: false }, + { include_subagents: true }, + ]); finishSecond?.({ data: { sessions: [{ id: 'with-subagents' }] } }); await second; @@ -160,7 +177,7 @@ describe('sessionListCache', () => { notifySessionListChanged(); expect(listener).toHaveBeenCalledTimes(1); - expect(mocks.listSessions).toHaveBeenCalled(); + await vi.waitFor(() => expect(mocks.listSessions).toHaveBeenCalled()); unsub(); }); }); diff --git a/ui/desktop/src/utils/sessionListCache.ts b/ui/desktop/src/utils/sessionListCache.ts index 7ae7b2732..f0a649b0c 100644 --- a/ui/desktop/src/utils/sessionListCache.ts +++ b/ui/desktop/src/utils/sessionListCache.ts @@ -1,4 +1,5 @@ import { listSessions, type Session } from '../api'; +import { userActionHeaders } from './userAction'; import { subscribeSessionNameChanges } from './sessionNameSync'; let cachedSessions: Session[] | null = null; @@ -115,12 +116,22 @@ export async function refreshSessionList(includeSubagents?: boolean): Promise({ - throwOnError: true, - // `cachedIncludeSubagents`, not the parameter: a keyless call must send the - // identity the cache is holding, not `undefined`. - query: { include_subagents: cachedIncludeSubagents }, - }) + // With the user's proof: since issue #56's QA sweep (2026-09-10) a listing + // omits every private chat from a caller without it, as the singular read + // refuses one — and this app is the person at the keyboard. + // `cachedIncludeSubagents`, not the parameter: a keyless call must send the + // identity the cache is holding, not `undefined`. Read NOW, before the + // proof's async hop: a flag change in that gap orphans this request, and + // an orphan must still ask for the list it was issued for. + const issuedFor = cachedIncludeSubagents; + inFlightRequest = userActionHeaders() + .then((headers) => + listSessions({ + throwOnError: true, + headers, + query: { include_subagents: issuedFor }, + }) + ) .then((response) => { // Superseded while in flight: hand the answer back to whoever awaited // this exact call, but publish nothing — the cache and its subscribers diff --git a/ui/desktop/src/utils/sessionNameSync.test.ts b/ui/desktop/src/utils/sessionNameSync.test.ts index 8d50b1157..dcf41db1c 100644 --- a/ui/desktop/src/utils/sessionNameSync.test.ts +++ b/ui/desktop/src/utils/sessionNameSync.test.ts @@ -6,6 +6,13 @@ vi.mock('../api', () => ({ updateSessionName: vi.fn(async () => ({ data: {} })), })); +// The proof the desktop sends. Since issue #56's QA sweep (2026-09-10) the +// daemon answers a request without it as a public model — private chats and +// knowledge bases omitted or refused — so each call here must carry it. +vi.mock('./userAction', () => ({ + userActionHeaders: async () => ({ 'X-User-Action': 'test-proof' }), +})); + import { updateSessionName } from '../api'; import { announceSessionName, @@ -127,6 +134,7 @@ describe('renameSession', () => { expect(updateSessionName).toHaveBeenCalledWith({ path: { session_id: 's1' }, body: { name: 'Q1 Plans' }, + headers: { 'X-User-Action': 'test-proof' }, throwOnError: true, }); }); diff --git a/ui/desktop/src/utils/sessionNameSync.ts b/ui/desktop/src/utils/sessionNameSync.ts index 874d43eee..2456f52ab 100644 --- a/ui/desktop/src/utils/sessionNameSync.ts +++ b/ui/desktop/src/utils/sessionNameSync.ts @@ -28,6 +28,7 @@ import type { Message, Session } from '../api'; import { updateSessionName } from '../api'; +import { userActionHeaders } from './userAction'; export const DEFAULT_SESSION_NAME = 'New chat'; @@ -144,9 +145,12 @@ export async function renameSession( const trimmed = newName.trim(); if (!trimmed) throw new Error('Chat name cannot be empty'); + // With the user's proof: renaming a private chat is refused, exactly as + // reading it is, to a caller without it (issue #56, QA 2026-09-10). await updateSessionName({ path: { session_id: sessionId }, body: { name: trimmed }, + headers: await userActionHeaders(), throwOnError: true, }); From 5a9f3fb8d57789e7db104099c9504ca27e0a93f1 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 10:31:51 -0700 Subject: [PATCH 12/75] test(privacy): run the knowledge-base sweep and the serve standing where CI looks MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CI runs cargo test --workspace --lib --bins, so the integration binaries that held the H2 route sweep (tests/knowledge_routes.rs) and the served-operator standing (tests/serve_operator_reach.rs) were not what kept either door shut. Both now have a copy in routes::session_reach's lib tests, driven through routes::configure — the tree the daemon serves. The serve decision record becomes SD-10 (SD-9 is claimed by PR #229), and its transcript bullet is reworded so it holds whether or not the interface states a capability: the cookie never reaches a transcript, and a stated capability is judged at those routes exactly as any caller's is. --- CLAUDE.md | 2 +- crates/biorouter-server/src/commands/agent.rs | 2 +- .../src/routes/session_reach.rs | 385 +++++++++++++++++- crates/biorouter-server/src/routes/web_ui.rs | 2 +- .../tests/serve_operator_reach.rs | 6 +- docs/deployment/browser-access.md | 8 +- .../deployment/programmatic-session-access.md | 2 +- docs/deployment/serve-architecture.md | 2 +- docs/deployment/serve-decisions.md | 25 +- docs/security/privacy-tiers-execution-plan.md | 2 +- docs/security/privacy-tiers.md | 2 +- 11 files changed, 412 insertions(+), 26 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 934e8061a..c89030a97 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -288,7 +288,7 @@ what did not" section first**; the rest of that document is the design, not the error: private rows silently vanish, and the Knowledge view's prune effects then read them as deleted. - A `biorouter serve` browser gets its operator's tier on listings and knowledge bases only - (SD-9). + (SD-10). - The wiring census (`crates/biorouter/tests/privacy_guard_wiring.rs`) counts every call site. - **Affiliation is a third axis** (DR-26, plan Phase 6): tier asks *how sensitive*, affiliation asks *whose*. HIPAA compliance does not transfer between institutions, so a UCSF model reaching another diff --git a/crates/biorouter-server/src/commands/agent.rs b/crates/biorouter-server/src/commands/agent.rs index 75f42d6cd..001dc06ba 100644 --- a/crates/biorouter-server/src/commands/agent.rs +++ b/crates/biorouter-server/src/commands/agent.rs @@ -201,7 +201,7 @@ pub async fn run() -> Result<()> { // there, so its absence here means a loopback bind whose launcher // chose not to require one. let browser_token = std::env::var("BIOROUTER_BROWSER_TOKEN").ok(); - // Issue #56, QA 2026-09-10 (SD-9): the interface this daemon serves is + // Issue #56, QA 2026-09-10 (SD-10): the interface this daemon serves is // the operator's, and SD-1 pins the provider every session here runs // on — so that provider's tier is the reach the listing and // knowledge-base gates give a request carrying the served document's diff --git a/crates/biorouter-server/src/routes/session_reach.rs b/crates/biorouter-server/src/routes/session_reach.rs index ba65df6ec..bc2f4035a 100644 --- a/crates/biorouter-server/src/routes/session_reach.rs +++ b/crates/biorouter-server/src/routes/session_reach.rs @@ -83,7 +83,7 @@ //! document's cookie — is given its operator's configured tier on those //! listing and knowledge-base surfaces, which were open to it before they //! were gated, and on NOTHING this function decides ([`HttpCaller`], -//! `docs/deployment/serve-decisions.md` SD-9); +//! `docs/deployment/serve-decisions.md` SD-10); //! * **`workspace_read_conversation` was open too, and it is CLOSED — but by a //! different instrument, and a reader must not credit this module for it.** //! That MCP tool (`crates/biorouter/src/agents/workspace_extension.rs`) used @@ -615,7 +615,7 @@ pub async fn session_reach( /// feeding the operator's tier into it would admit what it refused — the one /// thing this change may not do. Whether a serve operator on a private provider /// should reach a private transcript is a decision still to be made, and it is -/// recorded as open in `docs/deployment/serve-decisions.md` SD-9, not taken here. +/// recorded as open in `docs/deployment/serve-decisions.md` SD-10, not taken here. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct HttpCaller { /// DR-15's master opt-out, sampled with everything else. @@ -3390,6 +3390,387 @@ mod bypass_tests { let _ = state.knowledge_service.delete_base_async(&kb, None).await; } + /// Appears in the seeded knowledge pages and nowhere else. + const KB_SENTINEL: &str = "qa-h2-lib-sweep-marker-not-real-data"; + + /// Two bases in the served tree's knowledge store, each with one page and + /// one commit; the first is then ratcheted private the way a private chat's + /// ingest leaves it. Deleted on drop — including when an assertion fails — + /// because every test in this binary shares that store. + struct SeededBases { + state: Arc, + private: String, + public: String, + /// The private base's newest commit, for the history-shaped routes. + sha: String, + } + + impl Drop for SeededBases { + fn drop(&mut self) { + let root = self.state.knowledge_service.root().to_path_buf(); + for id in [&self.private, &self.public] { + let _ = self.state.knowledge_service.delete_base(id); + let _ = std::fs::remove_dir_all(root.join(id)); + } + } + } + + async fn seed_bases(state: &Arc, label: &str) -> SeededBases { + let pid = std::process::id(); + let mut seeded = SeededBases { + state: state.clone(), + private: format!("qa-{label}-private-{pid}"), + public: format!("qa-{label}-public-{pid}"), + sha: String::new(), + }; + for (id, name) in [ + (seeded.private.clone(), "QA private base (test fixture)"), + (seeded.public.clone(), "QA public base (test fixture)"), + ] { + let (status, body) = call( + state.clone(), + "POST", + "/knowledge/bases", + Some(serde_json::json!({ "id": id, "name": name })), + &[PROOF], + ) + .await; + assert_eq!(status, StatusCode::OK, "creating {id}: {body}"); + let (status, body) = call( + state.clone(), + "PUT", + &format!("/knowledge/bases/{id}/pages/knowledge/x.md"), + Some(serde_json::json!({ + "content": biorouter_mcp::knowledge::page_fixtures::valid_page( + "note", + "X", + &format!("# X\n\n{KB_SENTINEL} in {id}"), + ), + "commit_message": "seed", + })), + &[PROOF], + ) + .await; + assert_eq!(status, StatusCode::OK, "seeding {id}: {body}"); + } + let root = state.knowledge_service.root().to_path_buf(); + biorouter_mcp::knowledge::tier::raise_unlocked(&root, &seeded.private, true).unwrap(); + let (status, body) = call( + state.clone(), + "GET", + &format!("/knowledge/bases/{}/history", seeded.private), + None, + &[PROOF], + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + let history: serde_json::Value = serde_json::from_str(&body).unwrap(); + seeded.sha = history[0]["commit_sha"].as_str().unwrap().to_string(); + seeded + } + + /// Every route under `/knowledge/bases/{id}` in the served tree, as + /// `(method, uri, body)`. Macros name a provider the registry does not + /// know, so an admitted one stops with a 400 long before any model. + /// + /// ⚠ **Destructive last**, for the reason the chat sweep gives. + fn base_addressing_routes( + id: &str, + sha: &str, + other: &str, + ) -> Vec<(&'static str, String, Option)> { + let model = serde_json::json!({ "provider": "qa-h2-no-such-provider", "model": "m" }); + let page = biorouter_mcp::knowledge::page_fixtures::valid_page( + "note", + "X", + "overwritten by an unproven caller", + ); + let base = format!("/knowledge/bases/{id}"); + vec![ + ("GET", base.clone(), None), + ("GET", format!("{base}/tier"), None), + ("GET", format!("{base}/graph"), None), + ("GET", format!("{base}/location"), None), + ("GET", format!("{base}/page?path=knowledge/x.md"), None), + ("GET", format!("{base}/pages"), None), + ("GET", format!("{base}/pages/knowledge/x.md"), None), + ("GET", format!("{base}/history"), None), + ( + "POST", + format!("{base}/preview"), + Some(serde_json::json!({ "commit_sha": sha, "path": "knowledge/x.md" })), + ), + ("GET", format!("{base}/export"), None), + ( + "POST", + format!("{base}/query"), + Some(serde_json::json!({ "question": "what is in it?", "model": model })), + ), + ( + "POST", + format!("{base}/lint"), + Some(serde_json::json!({ "model": model })), + ), + ("POST", format!("{base}/sources/s1/reclassify"), None), + ( + "POST", + format!("{base}/tier"), + Some(serde_json::json!({ "tier": "public" })), + ), + ( + "POST", + format!("{base}/merge"), + Some(serde_json::json!({ "source_kb_id": other })), + ), + ( + "PUT", + base.clone(), + Some(serde_json::json!({ "name": "renamed by an unproven caller" })), + ), + ( + "PUT", + format!("{base}/default-model"), + Some(serde_json::json!({ "model": model })), + ), + ( + "PUT", + format!("{base}/pages/knowledge/x.md"), + Some(serde_json::json!({ "content": page, "commit_message": "overwrite" })), + ), + ( + "POST", + format!("{base}/raw"), + Some(serde_json::json!({ "text": "an unproven raw source", "title": "t" })), + ), + ( + "POST", + format!("{base}/ingest"), + Some(serde_json::json!({ "source": { "text": "t" }, "model": model })), + ), + ( + "POST", + format!("{base}/ingest-conversation"), + Some(serde_json::json!({ "session_ids": ["29990101_1"], "model": model })), + ), + ( + "POST", + format!("{base}/restore"), + Some(serde_json::json!({ "commit_sha": sha })), + ), + ("DELETE", base, None), + ] + } + + /// **H2, through the tree the daemon serves and in the binary CI runs.** + /// Every route that names a knowledge base answers a caller holding only + /// the daemon secret, on a private base, exactly as the page read does — + /// the same status and the same bytes — and answers a base that does not + /// exist the same way. The person at the keyboard still reads all of it, + /// and a public base is untouched. + /// + /// `tests/knowledge_routes.rs` (`h2_http_barrier`) sweeps the bare router + /// as well, but CI runs `cargo test --workspace --lib --bins`, so that + /// binary is not what keeps this door shut; this test is. + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn every_route_that_names_a_private_base_refuses_it_exactly_as_the_read_does() { + install_test_user_action_key(); + let state = AppState::new().await.unwrap(); + let bases = seed_bases(&state, "h2-sweep").await; + let absent = format!("qa-h2-sweep-absent-{}", std::process::id()); + let page_read = |id: &str| format!("/knowledge/bases/{id}/page?path=knowledge/x.md"); + + let (read_status, read_body) = + call(state.clone(), "GET", &page_read(&bases.private), None, &[]).await; + assert_eq!( + (read_status, read_body.as_str()), + (StatusCode::FORBIDDEN, KNOWLEDGE_BASE_OUT_OF_REACH), + "the read path's refusal is what every route below is compared against" + ); + + let mut leaks = Vec::new(); + for id in [bases.private.as_str(), absent.as_str()] { + for (method, uri, body) in base_addressing_routes(id, &bases.sha, &bases.public) { + let (status, got) = call(state.clone(), method, &uri, body, &[]).await; + if status != read_status || got != read_body { + leaks.push(format!("{method} {uri} -> {status}: {got:.160}")); + } + } + } + assert!( + leaks.is_empty(), + "a caller holding nothing but the daemon secret was answered differently from the \ + page read by {} route(s):\n {}", + leaks.len(), + leaks.join("\n ") + ); + + // …and nothing moved: still there, still private, same page. + let root = state.knowledge_service.root().to_path_buf(); + assert!(biorouter_mcp::knowledge::tier::is_private( + &root, + &bases.private + )); + let on_disk = + std::fs::read_to_string(root.join(&bases.private).join("knowledge/x.md")).unwrap(); + assert!( + on_disk.contains(KB_SENTINEL), + "an unproven caller rewrote a private page" + ); + + // The listing omits the private base — its id and its name — from the + // same caller, and shows it to the user. + let (status, body) = call(state.clone(), "GET", "/knowledge/bases", None, &[]).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert!(body.contains(&bases.public), "{body}"); + assert!( + !body.contains(&bases.private) && !body.contains("QA private base"), + "the served list named a private base to a secret-only caller: {body}" + ); + let (status, body) = call(state.clone(), "GET", "/knowledge/bases", None, &[PROOF]).await; + assert_eq!(status, StatusCode::OK); + assert!(body.contains(&bases.private), "{body}"); + + // The other half: "refuse everyone" would pass everything above. + for (method, uri, body) in base_addressing_routes(&bases.private, &bases.sha, "") + .into_iter() + .filter(|(method, _, _)| *method == "GET") + { + let (status, got) = call(state.clone(), method, &uri, body, &[PROOF]).await; + assert_eq!(status, StatusCode::OK, "{uri}: {got:.200}"); + } + let (status, got) = call( + state.clone(), + "GET", + &page_read(&bases.private), + None, + &[PROOF], + ) + .await; + assert_eq!(status, StatusCode::OK, "{got}"); + assert!(got.contains(KB_SENTINEL), "{got}"); + let (status, _) = call( + state.clone(), + "GET", + &format!("/knowledge/bases/{absent}"), + None, + &[PROOF], + ) + .await; + assert_eq!( + status, + StatusCode::NOT_FOUND, + "the user is entitled to know the base is not there" + ); + let (status, got) = call(state.clone(), "GET", &page_read(&bases.public), None, &[]).await; + assert_eq!(status, StatusCode::OK, "a public base was refused: {got}"); + assert!(got.contains(KB_SENTINEL)); + } + + /// The browser token a `biorouter serve` launch would have minted. Distinct + /// from every other cookie value in this binary's tests. + const SERVED_TOKEN: &str = "5d0c9b8a7f6e5d4c3b2a19f8e7d6c5b4"; + + /// **SD-10, in the binary CI runs.** A `serve` daemon's own interface — + /// told apart by the served document's cookie — keeps the listing and + /// knowledge-base reach its operator's private provider implies; the same + /// request without the cookie, or with the wrong one, is a public caller; + /// and the cookie opens no transcript — `GET /sessions/{id}` and `DELETE` + /// refuse it exactly as they refuse the secret alone. + /// + /// ⚠ It installs the operator standing into this test binary for good (a + /// `OnceLock`, as in the daemon). That is harmless to every other test here + /// because the standing is earned only by a request carrying this exact + /// cookie, and none of them sends it. The keyless arm — how `serve` really + /// starts its daemon — needs a binary with no user-action key, and is + /// `tests/serve_operator_reach.rs`. + #[tokio::test(flavor = "multi_thread")] + #[serial] + async fn a_served_interface_keeps_its_listing_reach_and_gains_no_transcript() { + install_test_user_action_key(); + crate::auth::install_served_operator(SERVED_TOKEN.to_string(), ProviderTier::Private); + let cookie = format!("biorouter_session={SERVED_TOKEN}"); + let mut probe = HeaderMap::new(); + probe.insert(axum::http::header::COOKIE, cookie.parse().unwrap()); + assert_eq!( + crate::auth::served_operator_capability(&probe), + ProviderTier::Private, + "a different serve operator was installed into this binary first; this test's \ + premise does not hold" + ); + + let state = AppState::new().await.unwrap(); + let private = seed_private_chat(&state, "SD-10 served private (test fixture)").await; + let bases = seed_bases(&state, "sd10").await; + let served = [("cookie", cookie.as_str())]; + let wrong = [( + "cookie", + "biorouter_session=00000000000000000000000000000000", + )]; + + for (headers, operator) in [(&served[..], true), (&[][..], false), (&wrong[..], false)] { + let (status, body) = call(state.clone(), "GET", "/sessions", None, headers).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!( + body.contains(private.id()), + operator, + "GET /sessions {headers:?}" + ); + let ids = sidebar_ids(&state, 50, headers).await; + assert_eq!(ids.contains(&private.id().to_string()), operator); + + let (status, body) = + call(state.clone(), "GET", "/knowledge/bases", None, headers).await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!( + body.contains(&bases.private), + operator, + "GET /knowledge/bases {headers:?}" + ); + let (status, body) = call( + state.clone(), + "GET", + &format!( + "/knowledge/bases/{}/page?path=knowledge/x.md", + bases.private + ), + None, + headers, + ) + .await; + if operator { + assert_eq!(status, StatusCode::OK, "{body}"); + assert!(body.contains(KB_SENTINEL), "{body}"); + } else { + assert_eq!( + (status, body.as_str()), + (StatusCode::FORBIDDEN, KNOWLEDGE_BASE_OUT_OF_REACH), + "{headers:?}" + ); + } + } + + // The cookie earns nothing at the transcript gate: the read and the + // delete refuse the served interface exactly as the secret alone. + for method in ["GET", "DELETE"] { + let uri = format!("/sessions/{}", private.id()); + let (status, body) = call(state.clone(), method, &uri, None, &served).await; + assert_eq!( + (status, body.as_str()), + (StatusCode::FORBIDDEN, SESSION_OUT_OF_REACH), + "{method} {uri} with the served cookie" + ); + } + assert!( + state + .session_manager() + .get_session(private.id(), false) + .await + .is_ok(), + "the served cookie deleted a private chat" + ); + } + /// Every id the sidebar hands this caller, walking `next_offset` to the end. async fn sidebar_ids( state: &Arc, diff --git a/crates/biorouter-server/src/routes/web_ui.rs b/crates/biorouter-server/src/routes/web_ui.rs index ba509fea8..3fd9f5a5d 100644 --- a/crates/biorouter-server/src/routes/web_ui.rs +++ b/crates/biorouter-server/src/routes/web_ui.rs @@ -48,7 +48,7 @@ //! gates (`routes::session_reach`). A request holding only the secret is a //! public caller there. `SameSite=Strict` keeps the cookie off every cross-site //! request, and a forged request still needs the secret, so no CSRF surface -//! appears. See `docs/deployment/serve-decisions.md` SD-9. +//! appears. See `docs/deployment/serve-decisions.md` SD-10. //! //! # Why there is no brute-force throttle here //! diff --git a/crates/biorouter-server/tests/serve_operator_reach.rs b/crates/biorouter-server/tests/serve_operator_reach.rs index e020839e3..35c496b8c 100644 --- a/crates/biorouter-server/tests/serve_operator_reach.rs +++ b/crates/biorouter-server/tests/serve_operator_reach.rs @@ -1,4 +1,4 @@ -//! Issue #56, QA 2026-09-10 (SD-9): a `biorouter serve` daemon's own web +//! Issue #56, QA 2026-09-10 (SD-10): a `biorouter serve` daemon's own web //! interface keeps the reach its operator's provider implies — on the listing //! and knowledge-base surfaces, which were open to it before they were gated — //! and a caller holding only the daemon secret does not. @@ -118,7 +118,7 @@ async fn the_served_interface_keeps_the_operators_reach_on_knowledge_bases() { /// else. The transcript gate refused this browser every private chat before /// this change and still does, and so does every route that names a chat: /// deleting one is never cheaper than reading it. Widening the transcript gate -/// for a serve operator is recorded as an open decision (SD-9), not taken. +/// for a serve operator is recorded as an open decision (SD-10), not taken. #[tokio::test(flavor = "multi_thread")] async fn the_served_interface_keeps_its_history_list_and_gains_nothing_else() { install_private_operator(); @@ -127,7 +127,7 @@ async fn the_served_interface_keeps_its_history_list_and_gains_nothing_else() { let chat = manager .create_session( std::path::PathBuf::from("/tmp/sd9_served_operator"), - "SD-9 private (test fixture)".to_string(), + "SD-10 private (test fixture)".to_string(), SessionType::User, ) .await diff --git a/docs/deployment/browser-access.md b/docs/deployment/browser-access.md index 16192ae90..7d1a03da8 100644 --- a/docs/deployment/browser-access.md +++ b/docs/deployment/browser-access.md @@ -184,7 +184,7 @@ differs: | Area | In a browser | |---|---| -| Chat, sessions, history, extensions, skills, knowledge bases, workflows | Work as they do in the desktop application, for everything public. **Private** chats and knowledge bases appear in History and the Knowledge view only when the provider you configured is private, and a private chat cannot be opened from the browser at all. See [decision SD-9](serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). | +| Chat, sessions, history, extensions, skills, knowledge bases, workflows | Work as they do in the desktop application, for everything public. **Private** chats and knowledge bases appear in History and the Knowledge view only when the provider you configured is private. Being listed does not by itself make a private chat openable from the browser. See [decision SD-10](serve-decisions.md#sd-10--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). | | Workspace control, several conversations at once, live app agents | Work — these are WebSocket-backed daemon routes, reached on the same origin. | | Model and provider selection | **Not available.** See [The model is fixed before you start](#the-model-is-fixed-before-you-start). | | File and folder pickers | No native dialog. You type a path, and it is a path **on the machine running the daemon**, not on the machine holding the browser. | @@ -228,9 +228,9 @@ Private chats and knowledge bases are shown in the browser only when the provide started with is itself private, meaning institution-hosted or running on this machine, and only when `serve` was started with its access token (the default). The desktop app proves a person is at the keyboard; a browser cannot, so it is given the reach of the model its daemon runs on and no -more. On a private provider the chat is listed but still cannot be opened from the browser. Open -it in the desktop app. The reasoning is -[decision SD-9](serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). +more. On a private provider the chat is listed, but being listed does not by itself let the +browser open, rename or delete it; when it cannot, open the chat in the desktop app. The reasoning is +[decision SD-10](serve-decisions.md#sd-10--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). ### When the interface cannot be found diff --git a/docs/deployment/programmatic-session-access.md b/docs/deployment/programmatic-session-access.md index 8809b216b..cace2f3e2 100644 --- a/docs/deployment/programmatic-session-access.md +++ b/docs/deployment/programmatic-session-access.md @@ -198,7 +198,7 @@ the caller could not open: | `GET /knowledge/bases`, `GET`/`POST /knowledge/active` | The public bases only. A write to the selection cannot hide, reveal or unpin a base the caller cannot see. | A browser pointed at `biorouter serve` is a special case of this, described in -[decision SD-9](serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). +[decision SD-10](serve-decisions.md#sd-10--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). ## What the header does *not* cover diff --git a/docs/deployment/serve-architecture.md b/docs/deployment/serve-architecture.md index 1795f1300..dca065e68 100644 --- a/docs/deployment/serve-architecture.md +++ b/docs/deployment/serve-architecture.md @@ -135,7 +135,7 @@ already passed `check_token` and also carries the cookie came from the document served, so the listing and knowledge-base gates give it the tier of the provider the operator configured. A request holding only the secret is a public caller there. The transcript gate never reads the cookie. See -[decision SD-9](serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). +[decision SD-10](serve-decisions.md#sd-10--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else). > **Warning.** `check_token` records a failed attempt for every request without the secret and > refuses after twenty inside sixty seconds, keyed on the peer address. The browser-token check diff --git a/docs/deployment/serve-decisions.md b/docs/deployment/serve-decisions.md index 6d3723803..292d802c5 100644 --- a/docs/deployment/serve-decisions.md +++ b/docs/deployment/serve-decisions.md @@ -227,7 +227,7 @@ can never half-believe a person is reachable. --- -## SD-9 — The served interface keeps its operator's reach on listings and knowledge bases, and gains nothing else +## SD-10 — The served interface keeps its operator's reach on listings and knowledge bases, and gains nothing else **Ruling (2026-09-11).** Since the privacy fix for QA findings H2 and M1 (2026-09-10), every daemon route that lists chats, or names, lists or reads a knowledge base, answers a caller that @@ -255,12 +255,14 @@ the operator's reach, which reopens H2 on every `serve` daemon. - **It reaches no private transcript.** The transcript gate, and every route that names one chat (open, export, the live event stream, delete, rename, and the rest), never read this standing. - A `serve` browser was refused every private transcript before this ruling and still is. So on a - private provider the History list shows private chats that cannot be opened from the browser, - and cannot be deleted or renamed from it either. That is SD-7's limitation, unchanged, and it - keeps deleting a chat from ever being easier than reading it. Letting the transcript gate honour - a served operator would be the first time a gate widened. It is an **open decision**, recorded - here and not taken. + They judge a `serve` browser exactly as they judged it before this ruling: on the proof it + carries, which is none (SD-7), and on the capability it states with `X-Caller-Provider`, which + they judge as they judge any caller's. An interface that states no capability — the case this + ruling was written against — sees private chats in its History list that it cannot open, delete + or rename. That is SD-7's limitation, left where this ruling found it, and it keeps deleting a + chat from ever being easier than reading it. Letting the transcript gate honour the cookie + itself would be the first time a gate widened. It is an **open decision**, recorded here and not + taken. - **It is not authentication, and not a proof of a person.** `biorouter serve` passes both the secret and the browser token in the daemon's environment. A caller that can read one can read the other, which is the residual the `X-Caller-Provider` header already carries @@ -279,9 +281,12 @@ different tiers, show different subsets of one shared history and knowledge stor from SD-1, which already made the provider a property of the daemon rather than of the tab. Implemented in `crates/biorouter-server/src/auth.rs` (`install_served_operator`, -`served_operator_capability`) and `routes::session_reach::HttpCaller`. Pinned by -`crates/biorouter-server/tests/serve_operator_reach.rs`, which asserts both halves: the interface -keeps its listing and knowledge-base reach, and gains no transcript. +`served_operator_capability`) and `routes::session_reach::HttpCaller`. Pinned in two places, +each asserting both halves — the interface keeps its listing and knowledge-base reach, and the +cookie gains it no transcript: `a_served_interface_keeps_its_listing_reach_and_gains_no_transcript` +in `routes::session_reach`'s lib tests, which is the copy CI runs, and +`crates/biorouter-server/tests/serve_operator_reach.rs`, which adds the keyless arm — a daemon with +no user-action key, as `serve` really starts it. --- diff --git a/docs/security/privacy-tiers-execution-plan.md b/docs/security/privacy-tiers-execution-plan.md index bf3650f59..54e750e1f 100644 --- a/docs/security/privacy-tiers-execution-plan.md +++ b/docs/security/privacy-tiers-execution-plan.md @@ -7309,7 +7309,7 @@ stop: that is a different product decision and it is [Open question 15](#open-qu > route that names a base by `{id}`, **reads and writes alike**, so a caller that may not read a base > cannot rewrite, restore or delete it either. `GET /knowledge/bases` and `/knowledge/active` omit > what the caller cannot reach. The Knowledge view still reads everything, because it sends the -> proof. A `biorouter serve` browser keeps its operator's reach under SD-9. So "a barrier there +> proof. A `biorouter serve` browser keeps its operator's reach under SD-10. So "a barrier there > would lock a user out of their own notes" did not come true: the barrier is on the caller who > proves nothing, and the user proves it on every request. Half (b) of > [Open question 15](#open-questions) is answered by this. Record: diff --git a/docs/security/privacy-tiers.md b/docs/security/privacy-tiers.md index 7b8a214ef..4799fda3b 100644 --- a/docs/security/privacy-tiers.md +++ b/docs/security/privacy-tiers.md @@ -91,7 +91,7 @@ this section is the ledger. The desktop app sends the proof on each of these calls and sees exactly what it saw before. A `biorouter serve` browser keeps its operator's reach on listings and knowledge bases and gains - no transcript ([SD-9](../deployment/serve-decisions.md#sd-9--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else)). + no transcript ([SD-10](../deployment/serve-decisions.md#sd-10--the-served-interface-keeps-its-operators-reach-on-listings-and-knowledge-bases-and-gains-nothing-else)). Nothing refused before is permitted now. ⚠ **What it does not change**, stated so it is not over-read. Privacy remains a safety boundary From ccc3e7985904d978229b66c25af9eb24da2fdc0a Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 10:33:27 -0700 Subject: [PATCH 13/75] test(privacy): the knowledge-base sweep covers the credibility override too --- crates/biorouter-server/src/routes/session_reach.rs | 5 +++++ crates/biorouter-server/tests/knowledge_routes.rs | 5 +++++ 2 files changed, 10 insertions(+) diff --git a/crates/biorouter-server/src/routes/session_reach.rs b/crates/biorouter-server/src/routes/session_reach.rs index bc2f4035a..28d2f2080 100644 --- a/crates/biorouter-server/src/routes/session_reach.rs +++ b/crates/biorouter-server/src/routes/session_reach.rs @@ -3522,6 +3522,11 @@ mod bypass_tests { format!("{base}/merge"), Some(serde_json::json!({ "source_kb_id": other })), ), + ( + "PUT", + format!("{base}/sources/s1/credibility"), + Some(serde_json::json!({})), + ), ( "PUT", base.clone(), diff --git a/crates/biorouter-server/tests/knowledge_routes.rs b/crates/biorouter-server/tests/knowledge_routes.rs index 5f3228a1a..6b8692318 100644 --- a/crates/biorouter-server/tests/knowledge_routes.rs +++ b/crates/biorouter-server/tests/knowledge_routes.rs @@ -3855,6 +3855,11 @@ mod h2_http_barrier { Some(serde_json::json!({ "model": model() })), ), ("POST", format!("/bases/{id}/sources/s1/reclassify"), None), + ( + "PUT", + format!("/bases/{id}/sources/s1/credibility"), + Some(serde_json::json!({})), + ), ( "POST", format!("/bases/{id}/tier"), From 01115e7a328c21ac6d4b877167859ad6d182cd7b Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 10:52:51 -0700 Subject: [PATCH 14/75] test(privacy): reach the serve standing through the library path the bin also links routes::session_reach is compiled into the biorouterd bin as well as the library, and the bin has no auth module of its own, so the new served-operator test named crate::auth and failed to build there (clippy caught it; cargo test --lib alone did not). It now goes through biorouter_server::auth, which is the static http_caller reads in either binary. --- crates/biorouter-server/src/routes/session_reach.rs | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/crates/biorouter-server/src/routes/session_reach.rs b/crates/biorouter-server/src/routes/session_reach.rs index 28d2f2080..83fdaec76 100644 --- a/crates/biorouter-server/src/routes/session_reach.rs +++ b/crates/biorouter-server/src/routes/session_reach.rs @@ -3693,12 +3693,18 @@ mod bypass_tests { #[serial] async fn a_served_interface_keeps_its_listing_reach_and_gains_no_transcript() { install_test_user_action_key(); - crate::auth::install_served_operator(SERVED_TOKEN.to_string(), ProviderTier::Private); + // `biorouter_server::`, not `crate::`: this module is also compiled into + // the `biorouterd` bin, which has no `auth` module of its own and reads + // the library's — the same static `http_caller` reads in either binary. + biorouter_server::auth::install_served_operator( + SERVED_TOKEN.to_string(), + ProviderTier::Private, + ); let cookie = format!("biorouter_session={SERVED_TOKEN}"); let mut probe = HeaderMap::new(); probe.insert(axum::http::header::COOKIE, cookie.parse().unwrap()); assert_eq!( - crate::auth::served_operator_capability(&probe), + served_operator_capability(&probe), ProviderTier::Private, "a different serve operator was installed into this binary first; this test's \ premise does not hold" From 3f9ae470b472026e9281c29badf253365834e064 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 11:47:57 -0700 Subject: [PATCH 15/75] sessionBindingSync: announce the app-wide model selection across windows MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit F3 (provider QA, 2026-09-10): the per-chat binding already crossed windows on `biorouter:session-binding`, but the app-wide selection — BIOROUTER_PROVIDER / BIOROUTER_MODEL, the pair `/agent/start` binds a new chat to — had no announcement at all. The module header claimed each window "picks it up from its own config read"; each window read it once, at mount. Add `announceAppModelSelection` / `subscribeAppModelSelectionChanges` on the same channel, told apart from a binding by shape. The message is a nudge with no provider and no model: two windows' writes can be announced in the opposite order from the one they landed in, so a receiver must re-read the daemon rather than apply a payload. Local listeners run synchronously so the writing window re-reads too. --- .../src/utils/sessionBindingSync.test.ts | 121 +++++++++++++++++- ui/desktop/src/utils/sessionBindingSync.ts | 91 ++++++++++++- 2 files changed, 206 insertions(+), 6 deletions(-) diff --git a/ui/desktop/src/utils/sessionBindingSync.test.ts b/ui/desktop/src/utils/sessionBindingSync.test.ts index c42b159a0..c3d5eba74 100644 --- a/ui/desktop/src/utils/sessionBindingSync.test.ts +++ b/ui/desktop/src/utils/sessionBindingSync.test.ts @@ -1,5 +1,12 @@ import { beforeEach, describe, expect, it, vi } from 'vitest'; -import { announceSessionBinding, subscribeSessionBindingChanges } from './sessionBindingSync'; +import { + announceAppModelSelection, + announceSessionBinding, + subscribeAppModelSelectionChanges, + subscribeSessionBindingChanges, +} from './sessionBindingSync'; + +const deliver = () => new Promise((resolve) => setTimeout(resolve, 0)); /** * Handoff 04 — a per-chat model switch made in one window has to reach the @@ -89,3 +96,115 @@ describe('sessionBindingSync', () => { expect(seen).toEqual([]); }); }); + +/** + * F3 (provider QA, 2026-09-10). The app-wide selection — the pair `/agent/start` + * binds a new chat to — crosses windows on the same channel as the binding, and + * the difference between the two messages is the whole design: a binding is a + * fact about one row, the selection announcement is a NUDGE to re-read. See + * "The second fact crosses too" in the module header. + */ +describe('sessionBindingSync — the app-wide selection', () => { + beforeEach(() => { + vi.restoreAllMocks(); + }); + + it('wakes local listeners synchronously, so the writing window re-reads too', () => { + let woken = 0; + const unsubscribe = subscribeAppModelSelectionChanges(() => { + woken += 1; + }); + + announceAppModelSelection(); + + expect(woken).toBe(1); + unsubscribe(); + }); + + /** + * ⚠ A kind and nothing else. A receiver that could read a provider and a + * model off the message would eventually apply them, and two windows' writes + * can be announced in the opposite order from the one they landed in. + */ + it('posts a nudge carrying no provider and no model', () => { + const posted: unknown[] = []; + vi.spyOn(BroadcastChannel.prototype, 'postMessage').mockImplementation((message: unknown) => { + posted.push(message); + }); + + const unsubscribe = subscribeAppModelSelectionChanges(() => {}); + announceAppModelSelection(); + unsubscribe(); + + expect(posted).toEqual([{ kind: 'app-model-selection' }]); + }); + + it('wakes on a nudge another window posted', async () => { + let woken = 0; + const unsubscribe = subscribeAppModelSelectionChanges(() => { + woken += 1; + }); + + const other = new BroadcastChannel('biorouter:session-binding'); + other.postMessage({ kind: 'app-model-selection' }); + await deliver(); + other.close(); + unsubscribe(); + + expect(woken).toBe(1); + }); + + /** + * One channel, two facts, told apart by shape — and neither may be mistaken + * for the other. A nudge must not reach a row patcher (it names no session to + * patch), and a binding must not make every window re-read its selection: a + * per-chat switch over there says nothing about new chats over here. + */ + it('keeps the two facts apart on the one channel', async () => { + const bindings: unknown[] = []; + let nudges = 0; + const offBinding = subscribeSessionBindingChanges((change) => bindings.push(change)); + const offSelection = subscribeAppModelSelectionChanges(() => { + nudges += 1; + }); + + const other = new BroadcastChannel('biorouter:session-binding'); + other.postMessage({ kind: 'app-model-selection' }); + other.postMessage({ sessionId: 's5', provider: 'codex', model: 'gpt-6-astra' }); + await deliver(); + other.close(); + offBinding(); + offSelection(); + + expect(nudges).toBe(1); + expect(bindings).toEqual([{ sessionId: 's5', provider: 'codex', model: 'gpt-6-astra' }]); + }); + + it('ignores a message of some other kind', async () => { + let nudges = 0; + const unsubscribe = subscribeAppModelSelectionChanges(() => { + nudges += 1; + }); + + const other = new BroadcastChannel('biorouter:session-binding'); + other.postMessage({ kind: 'something-else' }); + other.postMessage('app-model-selection'); + await deliver(); + other.close(); + unsubscribe(); + + expect(nudges).toBe(0); + }); + + it('stops waking a listener once it unsubscribes', () => { + let woken = 0; + const unsubscribe = subscribeAppModelSelectionChanges(() => { + woken += 1; + }); + unsubscribe(); + + announceAppModelSelection(); + + expect(woken).toBe(0); + }); +}); diff --git a/ui/desktop/src/utils/sessionBindingSync.ts b/ui/desktop/src/utils/sessionBindingSync.ts index 3ac6e1fb3..6fbb3efa1 100644 --- a/ui/desktop/src/utils/sessionBindingSync.ts +++ b/ui/desktop/src/utils/sessionBindingSync.ts @@ -66,13 +66,34 @@ * the selection says what a NEW chat will run on. A window that hears "chat X is * now bound to Y" learns something true about chat X and nothing at all about * its own next new chat — which is exactly right, because a per-chat switch made - * over there is not a statement about new chats over here. The global default - * does move underneath both windows, and each picks it up from its own config - * read; that is unchanged, and unrelated. + * over there is not a statement about new chats over here. * * What must NOT cross is a claim the receiver cannot check. So the receiver * treats an announcement about a chat it holds as authoritative (the daemon * accepted the write before it was announced) and ignores everything else. + * + * # The second fact crosses too — F3 + * + * This header used to end the objection above with "the global default does + * move underneath both windows, and each picks it up from its own config read". + * **Each window read it once, when it mounted, and never again.** Measured on + * 2026-09-10 (provider QA, finding F3): switch the app-wide model in window 1 and + * window 2's chip never moved — it went on reading `gpt-5.5-2026-04-24 (Private + * model, UCSF)` while the chat it started bound `claude_code`, because + * `/agent/start` binds whatever `config.yaml` says at that instant. The session + * was classified `public` correctly; the label the user acted on was the lie. + * + * So the app-wide selection has its own announcement on this same channel + * ({@link announceAppModelSelection}). It differs from the binding in one way + * that matters: it is a **nudge, never a payload**. A binding is a fact about + * one row that the daemon accepted before it was announced; the selection is + * one pair of config keys that any window, the CLI or a hand edit may be + * rewriting at the same moment, and two announcements can arrive in the + * opposite order from the two writes they describe. A receiver that applied the + * values would end on whichever message arrived last; one that re-reads the + * daemon ends on whichever WRITE landed last, which is what `/agent/start` will + * bind. The same reason `catalogSubscription` refetches rather than applying a + * delta. */ export interface SessionBindingChange { @@ -97,10 +118,25 @@ type Listener = (change: SessionBindingChange) => void; const listeners = new Set(); +/** + * The wire form of {@link announceAppModelSelection}: a kind and nothing else. + * + * ⚠ No provider, no model — deliberately, and not for brevity. See "The second + * fact crosses too" above: a receiver that could read the values off the message + * would eventually be written to apply them, and applying them is the race. + */ +export const APP_MODEL_SELECTION_MESSAGE = { kind: 'app-model-selection' } as const; + +type AppModelSelectionListener = () => void; + +const appSelectionListeners = new Set(); + // ── Cross-window broadcast ──────────────────────────────────────────────── // Same mechanism and same channel shape as `sessionNameSync`: BroadcastChannel // reaches every React subtree in this renderer AND every other BrowserWindow of -// the same origin, which is what a second Biorouter window is. +// the same origin, which is what a second Biorouter window is. ONE channel +// carries both facts, told apart by shape: a binding names a session, the +// selection nudge names a `kind` and no session. let channel: BroadcastChannel | null = null; function getChannel(): BroadcastChannel | null { @@ -108,7 +144,16 @@ function getChannel(): BroadcastChannel | null { if (typeof BroadcastChannel === 'undefined') return null; channel = new BroadcastChannel('biorouter:session-binding'); channel.onmessage = (event: MessageEvent) => { - const change = event.data as SessionBindingChange | undefined; + const data = event.data as + | SessionBindingChange + | typeof APP_MODEL_SELECTION_MESSAGE + | null + | undefined; + if (data && (data as { kind?: unknown }).kind === APP_MODEL_SELECTION_MESSAGE.kind) { + for (const listener of [...appSelectionListeners]) listener(); + return; + } + const change = data as SessionBindingChange | null | undefined; // Shape-checked, not trusted: this arrives from another window and a // malformed message must not patch a row with `undefined`. if (!change || !change.sessionId || !change.provider || !change.model) return; @@ -141,3 +186,39 @@ export function announceSessionBinding(change: SessionBindingChange): void { for (const listener of [...listeners]) listener(change); getChannel()?.postMessage(change); } + +/** + * Subscribe to "the app-wide model selection may have moved". Returns the + * unsubscribe. + * + * The listener gets no values, only the nudge: re-read `BIOROUTER_PROVIDER` / + * `BIOROUTER_MODEL` from the daemon and state what comes back. Subscribe from a + * MOUNT (`ModelAndProviderProvider`'s effect), never from a lookup — a + * subscription is a side effect, and one hung off a getter runs in every test + * that calls the getter. + */ +export function subscribeAppModelSelectionChanges(listener: () => void): () => void { + getChannel(); + appSelectionListeners.add(listener); + return () => { + appSelectionListeners.delete(listener); + }; +} + +/** + * Announce that `BIOROUTER_PROVIDER` / `BIOROUTER_MODEL` were just written. + * + * Call it AFTER the write resolved: a receiver re-reads the daemon, and a nudge + * that outran its write would re-read the value it was sent to replace — and + * then, with nothing further to wake it, keep stating it. + * + * Local listeners run synchronously, as {@link announceSessionBinding}'s do, so + * the window that wrote re-reads too. That is not redundant: the writer is not + * always `ModelAndProviderContext` itself (onboarding and Lead/Worker write + * these keys through `ConfigContext.upsert`), and a read issued after the write + * is what settles a race with another window's write. + */ +export function announceAppModelSelection(): void { + for (const listener of [...appSelectionListeners]) listener(); + getChannel()?.postMessage({ ...APP_MODEL_SELECTION_MESSAGE }); +} From 750d3d788d7dc38ce1fecbc4e6df0c5c1d47afe8 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 11:48:09 -0700 Subject: [PATCH 16/75] desktop: every window states the model its next new chat will run on (F3) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit F3 (provider QA, 2026-09-10, HIGH). Change the app-wide model in window 1 and window 2's chip never moved. Window 2 read `gpt-5.5-2026-04-24 (Private model, UCSF)` at the instant of send; the chat it created bound `claude_code`, was classified public — correctly — and its turn went to a consumer subscription with no BAA. `ModelAndProviderContext` read BIOROUTER_PROVIDER/BIOROUTER_MODEL once, on mount, while `/agent/start` binds whatever those keys say on the daemon at that instant. - ModelAndProviderContext re-reads the pair (a pure read, never the fallback seeding) on the app-wide announcement and when its window regains focus or becomes visible. Every statement of the selection — mount read, re-read, own switch — is ticketed and publishes only if nothing issued after it has been published, compared against what was last APPLIED (a failed newer read must not condemn an older good one). A read that returns no body keeps the label rather than erasing it. - Every renderer write of the two keys announces: `changeModel`, the first-run default seeding, and ConfigContext's `upsert`/`remove` — which covers onboarding's local and coding-agent cards, Lead/Worker and reset, none of which updated even their own window's chip before. - A switch made from inside a chat now changes THAT chat only, unless the new "Also use for new chats" box is ticked (unticked by default). This is privacy-tiers §14.3 P4's recommended decoupling: QA F bound Claude Code in one chat for one check and the next chat it opened came up public. With no chat (Home, a chat not yet started, Settings, onboarding) a switch sets the model new chats start on, and the dialog, the success toast and the chip's dropdown ("Model for new chats") now say so, including "in every window". - Both new-chat composers (Home and a not-yet-started chat) re-read the pair immediately before `createSession`. If the chip was stale — a `biorouter configure` in the terminal docked inside the window never takes its focus — the send is refused, the fresh model and its tier go on screen, a toast names what changed, and the composer gets the text back. The pin still outranks a stale row: the app-wide selection touches neither. --- ui/desktop/src/components/BaseChat.tsx | 17 +- .../src/components/ConfigContext.test.tsx | 97 ++- ui/desktop/src/components/ConfigContext.tsx | 19 + ui/desktop/src/components/Hub.tsx | 9 +- ...delAndProviderContext.crossWindow.test.tsx | 619 ++++++++++++++++++ .../ModelAndProviderContext.test.tsx | 47 +- .../components/ModelAndProviderContext.tsx | 256 +++++++- .../privacy/useConfirmNewChatModel.test.tsx | 192 ++++++ .../privacy/useConfirmNewChatModel.ts | 101 +++ .../ModelsBottomBar.pinned.test.tsx | 41 +- .../models/bottom_bar/ModelsBottomBar.tsx | 25 +- .../SwitchModelModal.privacy.test.tsx | 15 +- .../subcomponents/SwitchModelModal.test.tsx | 125 +++- .../models/subcomponents/SwitchModelModal.tsx | 56 +- 14 files changed, 1565 insertions(+), 54 deletions(-) create mode 100644 ui/desktop/src/components/ModelAndProviderContext.crossWindow.test.tsx create mode 100644 ui/desktop/src/components/privacy/useConfirmNewChatModel.test.tsx create mode 100644 ui/desktop/src/components/privacy/useConfirmNewChatModel.ts diff --git a/ui/desktop/src/components/BaseChat.tsx b/ui/desktop/src/components/BaseChat.tsx index d3abed351..0b2711a11 100644 --- a/ui/desktop/src/components/BaseChat.tsx +++ b/ui/desktop/src/components/BaseChat.tsx @@ -43,6 +43,7 @@ import { WorkflowWarningModal } from './ui/WorkflowWarningModal'; import { NonPrivateModelDisclosureGate } from './privacy/NonPrivateModelDisclosureGate'; import { PinnedModelNote } from './privacy/PinnedModelNote'; import { usePinnedModel } from './privacy/usePinnedModel'; +import { useConfirmNewChatModel } from './privacy/useConfirmNewChatModel'; import { scanWorkflow } from '../workflow'; import { useCostTracking } from '../hooks/useCostTracking'; import { useDiverge } from '../hooks/useDiverge'; @@ -1203,6 +1204,8 @@ function BaseChatContent({ const [hasNotAcceptedWorkflow, setHasNotAcceptedWorkflow] = useState(); const [hasWorkflowSecurityWarnings, setHasWorkflowSecurityWarnings] = useState(false); const [isCreatingSession, setIsCreatingSession] = useState(false); + // F3 — the model this chat is about to be created on is the one on screen. + const confirmNewChatModel = useConfirmNewChatModel(); // #39 — the working directory chosen in the composer BEFORE a session // exists (sidebar "New chat" mounts this chat with no sessionId, so // DirSwitcher has nothing to persist to yet). Read exactly once, by the @@ -1575,10 +1578,12 @@ function BaseChatContent({ /** * Resolves FALSE when the message was refused and the composer still owns the * text (ChatInput puts it back). The pre-session branch returns TRUE on both - * of its outcomes: a created session has navigated with the message as its - * cargo, and a failed `createSession` has already restored the composer and - * toasted through `handleCreateSessionError`, so a second restore would be a - * duplicate rather than a rescue. + * of its outcomes once a session is attempted: a created session has + * navigated with the message as its cargo, and a failed `createSession` has + * already restored the composer and toasted through `handleCreateSessionError`, + * so a second restore would be a duplicate rather than a rescue. It returns + * FALSE only when F3's model check refused BEFORE anything was attempted — + * the one case where the composer's own restore is the rescue. */ const handleFormSubmit = async (e: React.FormEvent): Promise => { const customEvent = e as unknown as CustomEvent; @@ -1591,6 +1596,10 @@ function BaseChatContent({ // If no session exists, create one and navigate with the initial message const hasAttachments = Array.isArray(attachments) && attachments.length > 0; if (!session && !sessionId && (textValue.trim() || hasAttachments) && !isCreatingSession) { + // F3. `/agent/start` binds whatever the app-wide selection is NOW, and the + // composer's chip is this window's copy of it. A refusal here has already + // put the fresh model on screen; resolving `false` hands the text back. + if (!(await confirmNewChatModel())) return false; setIsCreatingSession(true); try { // #39 — honour the directory picked in the composer before the diff --git a/ui/desktop/src/components/ConfigContext.test.tsx b/ui/desktop/src/components/ConfigContext.test.tsx index 03d38176a..3bc40ab03 100644 --- a/ui/desktop/src/components/ConfigContext.test.tsx +++ b/ui/desktop/src/components/ConfigContext.test.tsx @@ -1,6 +1,6 @@ import { act, fireEvent, render, screen, waitFor } from '@testing-library/react'; import { useState } from 'react'; -import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; import { ConfigProvider, useConfig } from './ConfigContext'; // Issue #52 — the cached `config` object was only ever re-read when a write @@ -412,3 +412,98 @@ describe('ConfigContext catalogue subscription (#112)', () => { expect(unhandled).toEqual([]); }); }); + +/** + * F3 (provider QA, 2026-09-10). `BIOROUTER_PROVIDER` and `BIOROUTER_MODEL` are + * what `/agent/start` binds a new chat to, so a write of either through this + * context is announced to every window — each of which re-reads the pair. The + * writers this catches are the ones that never pass through + * `ModelAndProviderContext.changeModel`: onboarding's local and coding-agent + * cards, Lead/Worker settings and Settings' reset, which until now left even + * their own window's chip naming the previous model. + */ +describe('ConfigContext announces writes of the app-wide model selection (F3)', () => { + function WriteProbe() { + const { upsert, remove } = useConfig(); + const [result, setResult] = useState('idle'); + const run = (write: () => Promise) => { + setResult('pending'); + write().then( + () => setResult('ok'), + (error: unknown) => setResult(`failed: ${String(error)}`) + ); + }; + return ( +
+ {result} + + + + +
+ ); + } + + let nudges = 0; + let unsubscribe: () => void = () => {}; + + beforeEach(async () => { + nudges = 0; + const { subscribeAppModelSelectionChanges } = await import('../utils/sessionBindingSync'); + unsubscribe = subscribeAppModelSelectionChanges(() => { + nudges += 1; + }); + mocks.removeConfig.mockResolvedValue({ data: {} }); + }); + + afterEach(() => unsubscribe()); + + const renderWriteProbe = () => + render( + + + + ); + + it.each([['Write provider'], ['Write model'], ['Remove model']])( + '%s announces, once the write has resolved', + async (button) => { + renderWriteProbe(); + fireEvent.click(screen.getByRole('button', { name: button })); + await waitFor(() => expect(screen.getByTestId('write-result')).toHaveTextContent('ok')); + expect(nudges).toBe(1); + } + ); + + it('says nothing about a key a new chat does not bind', async () => { + renderWriteProbe(); + fireEvent.click(screen.getByRole('button', { name: 'Write mode' })); + await waitFor(() => expect(screen.getByTestId('write-result')).toHaveTextContent('ok')); + expect(nudges).toBe(0); + }); + + /** A refused write moved nothing, so there is nothing to re-read. */ + it('says nothing when the write was refused', async () => { + mocks.upsertConfig.mockRejectedValue(new Error('409 Conflict')); + renderWriteProbe(); + fireEvent.click(screen.getByRole('button', { name: 'Write provider' })); + await waitFor(() => + expect(screen.getByTestId('write-result')).toHaveTextContent('failed: Error: 409 Conflict') + ); + expect(nudges).toBe(0); + }); +}); diff --git a/ui/desktop/src/components/ConfigContext.tsx b/ui/desktop/src/components/ConfigContext.tsx index 68de4ae61..5fcd1f9b5 100644 --- a/ui/desktop/src/components/ConfigContext.tsx +++ b/ui/desktop/src/components/ConfigContext.tsx @@ -30,6 +30,7 @@ import { shouldDefaultEnablePromotedCapability, } from './settings/capabilities/capabilities'; import { PRIVACY_TIERS_KEY, privacyTiersEnabledFromConfig } from './settings/privacy/privacyTiers'; +import { announceAppModelSelection } from '../utils/sessionBindingSync'; import type { ConfigResponse, UpsertConfigQuery, @@ -95,6 +96,21 @@ export class MalformedConfigError extends Error { const ConfigContext = createContext(undefined); +/** + * F3 — the two keys `/agent/start` binds a new chat to. + * + * A write of either changes what every window's composer must state, so it is + * announced to all of them (`utils/sessionBindingSync`), each of which re-reads + * the pair. `ModelAndProviderContext.changeModel` announces its own writes; the + * ones caught HERE are those that never pass through it — the local and + * coding-agent onboarding cards, Lead/Worker settings, Settings' reset — which + * until now left even their own window's chip naming the previous model. + */ +const APP_MODEL_SELECTION_KEYS: ReadonlySet = new Set([ + 'BIOROUTER_PROVIDER', + 'BIOROUTER_MODEL', +]); + export const ConfigProvider: React.FC = ({ children }) => { const [config, setConfig] = useState({}); const [providersList, setProvidersList] = useState([]); @@ -201,6 +217,8 @@ export const ConfigProvider: React.FC = ({ children }) => { headers: await userActionHeaders(), }); await reloadConfigAfterWrite(); + // After the write resolved — a refused one threw above and moved nothing. + if (APP_MODEL_SELECTION_KEYS.has(key)) announceAppModelSelection(); }, [reloadConfigAfterWrite] ); @@ -226,6 +244,7 @@ export const ConfigProvider: React.FC = ({ children }) => { headers: await userActionHeaders(), }); await reloadConfigAfterWrite(); + if (APP_MODEL_SELECTION_KEYS.has(key)) announceAppModelSelection(); }, [reloadConfigAfterWrite] ); diff --git a/ui/desktop/src/components/Hub.tsx b/ui/desktop/src/components/Hub.tsx index 8b4228478..3e15b77ed 100644 --- a/ui/desktop/src/components/Hub.tsx +++ b/ui/desktop/src/components/Hub.tsx @@ -29,6 +29,7 @@ import { getInitialWorkingDir } from '../utils/workingDir'; import { createSession } from '../sessions'; import LoadingBioRouter from './LoadingBioRouter'; import type { UserAttachment } from '../types/message'; +import { useConfirmNewChatModel } from './privacy/useConfirmNewChatModel'; export default function Hub({ setView, @@ -38,14 +39,20 @@ export default function Hub({ const { extensionsList } = useConfig(); const [workingDir, setWorkingDir] = useState(getInitialWorkingDir()); const [isCreatingSession, setIsCreatingSession] = useState(false); + const confirmNewChatModel = useConfirmNewChatModel(); - const handleSubmit = async (e: React.FormEvent) => { + const handleSubmit = async (e: React.FormEvent): Promise => { const customEvent = e as unknown as CustomEvent; const combinedTextFromInput = customEvent.detail?.value || ''; const attachments = (customEvent.detail?.attachments ?? []) as UserAttachment[]; const hasAttachments = attachments.length > 0; if ((combinedTextFromInput.trim() || hasAttachments) && !isCreatingSession) { + // F3. Before anything is consumed — the extension overrides below are + // cleared as they are read — so a refused send leaves nothing behind but + // the text, which `ChatInput` puts back when this resolves `false`. + if (!(await confirmNewChatModel())) return false; + const extensionConfigs = getExtensionConfigsWithOverrides(extensionsList); clearExtensionOverrides(); setIsCreatingSession(true); diff --git a/ui/desktop/src/components/ModelAndProviderContext.crossWindow.test.tsx b/ui/desktop/src/components/ModelAndProviderContext.crossWindow.test.tsx new file mode 100644 index 000000000..0497a8e7d --- /dev/null +++ b/ui/desktop/src/components/ModelAndProviderContext.crossWindow.test.tsx @@ -0,0 +1,619 @@ +import { act, fireEvent, render, screen, waitFor, within } from '@testing-library/react'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { useState } from 'react'; +import { + ModelAndProviderProvider, + useModelAndProvider, + type ChangeModelOptions, +} from './ModelAndProviderContext'; +import ModelsBottomBar from './settings/models/bottom_bar/ModelsBottomBar'; +import { __resetDisclosureStoreForTests } from './privacy/disclosureCopy'; +import { usePinnedModel } from './privacy/usePinnedModel'; +import { useConfirmNewChatModel } from './privacy/useConfirmNewChatModel'; +import type Model from './settings/models/modelInterface'; +import type { Session } from '../api/types.gen'; +import type { PinnedModelView } from '../hooks/chatStreamStore'; + +/** + * F3 (provider QA, 2026-09-10) — a second window's model chip was stale, and the + * chat it started ran on a model it never showed. + * + * Measured on merged main `7c96d796`, both directions: change the app-wide + * model in window 1 and window 2's chip never moved (8 s). Window 2 read + * `gpt-5.5-2026-04-24 (Private model, UCSF)` at the instant of send; the chat it + * created bound `claude_code`, was classified `public` — correctly — and its + * turn went to a consumer subscription with no BAA. The privacy machinery held; + * the label the human acted on did not. + * + * Root cause: `ModelAndProviderContext` read `BIOROUTER_PROVIDER` / + * `BIOROUTER_MODEL` once, on mount, while `/agent/start` binds a new chat to + * whatever those keys say on the daemon at that instant. + * + * Every test here mounts TWO `ModelAndProviderProvider` trees — two windows' + * worth of state in one document — each rendering the real composer chip, over + * one fake daemon whose two keys are what `/agent/start` would bind. The + * assertion that matters throughout is the one the QA run could not make: the + * chip a window shows equals what its next new chat would run on. + */ + +const mocks = vi.hoisted(() => ({ + read: vi.fn(), + getProviders: vi.fn(), + refreshConfig: vi.fn(), + setConfigProvider: vi.fn(), + updateAgentProvider: vi.fn(), + llamacppStatus: vi.fn(), + llamacppWarmup: vi.fn(), + getPrivacyDisclosure: vi.fn(), + ackPrivacyDisclosure: vi.fn(), + toastSuccess: vi.fn(), + toastError: vi.fn(), + toastWarning: vi.fn(), +})); + +vi.mock('../api', async (importOriginal) => ({ + ...(await importOriginal>()), + setConfigProvider: mocks.setConfigProvider, + updateAgentProvider: mocks.updateAgentProvider, + llamacppStatus: mocks.llamacppStatus, + llamacppWarmup: mocks.llamacppWarmup, + getPrivacyDisclosure: mocks.getPrivacyDisclosure, + ackPrivacyDisclosure: mocks.ackPrivacyDisclosure, +})); + +vi.mock('../toasts', async (importOriginal) => ({ + ...(await importOriginal>()), + toastSuccess: mocks.toastSuccess, + toastError: mocks.toastError, + toastWarning: mocks.toastWarning, +})); + +vi.mock('../utils/userAction', async (importOriginal) => ({ + ...(await importOriginal>()), + userActionHeaders: async () => ({ 'X-User-Action': 'test-key' }), +})); + +// `usePrivacyTiersEnabled` too: the chip's padlock reads the master switch. +vi.mock('./ConfigContext', () => ({ + useConfig: () => ({ + read: mocks.read, + getProviders: mocks.getProviders, + refreshConfig: mocks.refreshConfig, + }), + usePrivacyTiersEnabled: () => true, +})); + +// `BaseChat` is the whole chat surface; the chip imports it for one dead context. +vi.mock('./BaseChat', () => ({ useCurrentModelInfo: () => null })); +vi.mock('./settings/models/subcomponents/SwitchModelModal', () => ({ + SwitchModelModal: () => null, +})); +vi.mock('./settings/models/subcomponents/LeadWorkerSettings', () => ({ + LeadWorkerSettings: () => null, +})); + +Object.defineProperty(window, 'appConfig', { + writable: true, + value: { get: () => undefined }, +}); + +// ── The daemon ───────────────────────────────────────────────────────────── + +const PRIVATE = { provider: 'versa_azure', model: 'gpt-5.5-2026-04-24' }; +const PUBLIC = { provider: 'claude_code', model: 'claude-fable-5-1' }; +const CODEX = { provider: 'codex', model: 'gpt-6-astra' }; + +/** `config.yaml`'s two keys, as the daemon holds them. */ +const daemon: { provider: string | null; model: string | null } = { ...PRIVATE }; + +/** + * What the next `/agent/start` binds: `configured_new_session_provider` reads + * exactly these two keys and nothing the renderer sends. + */ +const nextNewChatBinding = () => ({ provider: daemon.provider, model: daemon.model }); + +/** A write made by anything other than the renderer: the CLI, a hand edit. */ +const writeOutsideTheRenderer = (next: { provider: string; model: string }) => { + daemon.provider = next.provider; + daemon.model = next.model; +}; + +const UCSF = { kind: 'institutions', institutions: [{ id: 'ucsf', display_name: 'UCSF' }] }; + +const providerRow = ( + name: string, + display: string, + tier: 'private' | 'public', + affiliation: unknown = null +) => ({ + name, + is_configured: true, + provider_type: 'Builtin', + metadata: { name, display_name: display, tier, runs_locally: false, known_models: [] }, + affiliation, + resolved_tier: tier, +}); + +const PROVIDER_ROWS = [ + providerRow('versa_azure', 'Versa API Azure', 'private', UCSF), + providerRow('claude_code', 'Claude Code', 'public'), + providerRow('codex', 'Codex', 'public'), +]; + +/** The chip's accessible name, exactly as a screen reader — or QA — reads it. */ +const CHIP = { + [PRIVATE.model]: `Current model: ${PRIVATE.model} (Private model, UCSF)`, + [PUBLIC.model]: `Current model: ${PUBLIC.model} (Public model)`, + [CODEX.model]: `Current model: ${CODEX.model} (Public model)`, +}; + +/** The chip's label for whatever pair the daemon holds right now. */ +const chipForDaemon = () => CHIP[nextNewChatBinding().model as string]; + +/** A second window speaking on the channel, as `BroadcastChannel` delivers it. */ +async function announceFromAnotherWindow() { + const other = new BroadcastChannel('biorouter:session-binding'); + other.postMessage({ kind: 'app-model-selection' }); + // Delivery is a task, not a microtask; close only once it has happened. + await act(async () => { + await new Promise((resolve) => setTimeout(resolve, 0)); + }); + other.close(); +} + +// ── The windows ──────────────────────────────────────────────────────────── + +const dropdownRef = { current: null } as unknown as React.RefObject; + +const asModel = (pair: { provider: string; model: string }): Model => ({ + name: pair.model, + provider: pair.provider, + subtext: pair.provider, +}); + +/** + * One window's Home composer: its chip, plus the two acts that matter — the + * model switcher's commit and a send that would create a new chat. + */ +function HomeWindow({ label }: { label: string }) { + const { changeModel } = useModelAndProvider(); + const confirmNewChatModel = useConfirmNewChatModel(); + const [sendResult, setSendResult] = useState('idle'); + return ( +
+ + + + + {sendResult} +
+ ); +} + +/** One window showing an existing chat, with the switcher committing for it. */ +function ChatWindow({ + label, + session, + pin, + switchOptions, +}: { + label: string; + session: Session; + pin?: PinnedModelView; + switchOptions?: ChangeModelOptions; +}) { + const { changeModel } = useModelAndProvider(); + const { effectiveModel } = usePinnedModel(session, pin); + return ( +
+ + +
+ ); +} + +const inWindow = (label: string) => within(screen.getByRole('region', { name: label })); + +const chipIn = (label: string) => + inWindow(label).getByRole('button', { name: /^Current model:/ }) as HTMLElement; + +async function expectChip(label: string, name: string) { + await waitFor(() => expect(chipIn(label)).toHaveAccessibleName(name)); +} + +function renderTwoHomeWindows() { + return render( + <> + + + + + + + + ); +} + +const session = (overrides: Partial): Session => + ({ + id: 'chat-a', + working_dir: '/tmp', + name: 'A chat', + message_count: 3, + privacy_tier: 'private', + ...overrides, + }) as Session; + +beforeEach(() => { + vi.clearAllMocks(); + __resetDisclosureStoreForTests(); + Object.assign(daemon, PRIVATE); + mocks.read.mockImplementation(async (key: string) => { + if (key === 'BIOROUTER_MODEL') return daemon.model; + if (key === 'BIOROUTER_PROVIDER') return daemon.provider; + return null; + }); + mocks.getProviders.mockResolvedValue(PROVIDER_ROWS); + mocks.refreshConfig.mockResolvedValue(undefined); + // `/config/set_provider`, as the daemon applies it. + mocks.setConfigProvider.mockImplementation( + async ({ body }: { body: { provider: string; model: string } }) => { + daemon.provider = body.provider; + daemon.model = body.model; + return { data: null }; + } + ); + mocks.updateAgentProvider.mockResolvedValue({ data: '' }); + mocks.getPrivacyDisclosure.mockResolvedValue({ + data: { + title_template: '{provider} is not hosted by your institution.', + long: 'LONG', + short: 'SHORT', + acknowledged: true, + }, + }); +}); + +describe('F3 — the app-wide selection reaches every window', () => { + /** + * The brief's first test, as stated: two consumers, the broadcast, and the + * second one's model, provider and privacy label updating without a remount. + * The announcement arrives the way another BrowserWindow's does — on the + * channel, from a different `BroadcastChannel` object. + * + * Fails on 7c96d796: nothing listens for it, and window 2 keeps + * `gpt-5.5-2026-04-24 (Private model, UCSF)` for as long as it lives. + */ + it("updates a second window's chip — model, provider and privacy — without a remount", async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + const before = chipIn('window 2'); + + writeOutsideTheRenderer(PUBLIC); + await announceFromAnotherWindow(); + + await expectChip('window 2', CHIP[PUBLIC.model]); + expect(chipIn('window 2')).not.toHaveAccessibleName(/Private model/); + // The same element, re-rendered — not a fresh mount that re-read on mount. + expect(chipIn('window 2')).toBe(before); + }); + + /** + * The brief's second test: a stale label cannot survive a change, in either + * direction. "The value the next `/agent/start` would bind" is the fake + * daemon's pair, which is exactly what `configured_new_session_provider` + * reads. + */ + it('states what the next new chat would bind, after a switch in either direction', async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + // Private → public. The direction QA measured going to a no-BAA plan. + fireEvent.click(inWindow('window 1').getByRole('button', { name: /to Claude Code/ })); + await waitFor(() => expect(nextNewChatBinding()).toEqual(PUBLIC)); + await expectChip('window 2', chipForDaemon()); + expect(chipIn('window 2')).toHaveAccessibleName(CHIP[PUBLIC.model]); + expect(chipIn('window 1')).toHaveAccessibleName(CHIP[PUBLIC.model]); + + // And back. + fireEvent.click(inWindow('window 1').getByRole('button', { name: /to Versa/ })); + await waitFor(() => expect(nextNewChatBinding()).toEqual(PRIVATE)); + await expectChip('window 2', chipForDaemon()); + expect(chipIn('window 2')).toHaveAccessibleName(CHIP[PRIVATE.model]); + }); + + /** + * ⚠ The nudge is not a payload, and this is why: two windows' writes can be + * announced in the opposite order from the one in which they landed. A + * receiver that applied values would end on the last MESSAGE; one that + * re-reads ends on the last WRITE — the one `/agent/start` binds. + */ + it('ends on the write that landed last, not the announcement that arrived last', async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + // Two writes land — Codex, then Claude Code — and both are announced. + writeOutsideTheRenderer(CODEX); + writeOutsideTheRenderer(PUBLIC); + await announceFromAnotherWindow(); + await announceFromAnotherWindow(); + + await expectChip('window 2', CHIP[PUBLIC.model]); + }); +}); + +describe('F3 — reads are ordered by when they were issued', () => { + /** + * A read that left before a newer statement was published must not come back + * afterwards and restore the model the newer one replaced. Window 2 re-reads + * twice; the first answer is held until after the second has landed. + */ + it('does not let a slower, older read overwrite a newer one', async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + let releaseOld: (value: string) => void = () => {}; + const heldModel = new Promise((resolve) => { + releaseOld = resolve; + }); + // While `holding`, every read answers with the OLD pair, and slowly. + let holding = true; + mocks.read.mockImplementation(async (key: string) => { + if (holding && key === 'BIOROUTER_MODEL') return heldModel; + if (holding && key === 'BIOROUTER_PROVIDER') return PRIVATE.provider; + if (key === 'BIOROUTER_MODEL') return daemon.model; + if (key === 'BIOROUTER_PROVIDER') return daemon.provider; + return null; + }); + + await announceFromAnotherWindow(); + holding = false; + writeOutsideTheRenderer(PUBLIC); + await announceFromAnotherWindow(); + await expectChip('window 2', CHIP[PUBLIC.model]); + + // The first read finally answers — with the pair the second one replaced. + await act(async () => { + releaseOld(PRIVATE.model); + await heldModel; + }); + expect(chipIn('window 2')).toHaveAccessibleName(CHIP[PUBLIC.model]); + }); + + /** + * ⚠ The other half of the rule: a NEWER read that fails publishes nothing, so + * it must not condemn an older one that succeeded + * (`renderer-testing-traps.md`, "Newest issued is the wrong rule"). + */ + it('keeps an older read that succeeded when a newer one fails', async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + writeOutsideTheRenderer(PUBLIC); + let release: () => void = () => {}; + const gate = new Promise((resolve) => { + release = resolve; + }); + // The first nudge's reads are slow and right; the second's are fast and + // failed — the generated client resolves a dead daemon with no body. + let phase: 'slow' | 'failing' = 'slow'; + mocks.read.mockImplementation(async (key: string) => { + if (phase === 'failing') return undefined; + await gate; + if (key === 'BIOROUTER_MODEL') return daemon.model; + if (key === 'BIOROUTER_PROVIDER') return daemon.provider; + return null; + }); + + await announceFromAnotherWindow(); + phase = 'failing'; + await announceFromAnotherWindow(); + await act(async () => { + release(); + await gate; + }); + + await expectChip('window 2', CHIP[PUBLIC.model]); + }); + + /** A failed read is not evidence that nothing is configured. */ + it('keeps the label it has when a re-read fails, rather than erasing it', async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + mocks.read.mockResolvedValue(undefined); + await announceFromAnotherWindow(); + + expect(chipIn('window 2')).toHaveAccessibleName(CHIP[PRIVATE.model]); + expect(screen.queryByRole('button', { name: 'Choose a model' })).toBeNull(); + }); +}); + +describe('F3 — a write nothing announces', () => { + /** + * `biorouter configure` in a terminal writes `config.yaml`, the daemon's + * cache is keyed on the file's stamp, and nothing tells any window. Coming + * back to the window is when the user acts, so that is when it re-reads. + */ + it('re-reads when the window regains focus', async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + writeOutsideTheRenderer(PUBLIC); + await act(async () => { + window.dispatchEvent(new Event('focus')); + }); + + await expectChip('window 2', CHIP[PUBLIC.model]); + }); + + it('re-reads when the window becomes visible again', async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + writeOutsideTheRenderer(CODEX); + await act(async () => { + document.dispatchEvent(new Event('visibilitychange')); + }); + + await expectChip('window 2', CHIP[CODEX.model]); + }); + + /** + * A terminal docked INSIDE the window never takes its focus, so a send can + * still be the first thing to notice. It is refused, the fresh model is put + * on screen, and the toast says what changed in words — including the tier, + * which is the reason any of this matters. + */ + it('refuses a send made on a stale chip, and shows the model it would have used', async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + writeOutsideTheRenderer(PUBLIC); + fireEvent.click(inWindow('window 2').getByRole('button', { name: 'Send' })); + + await waitFor(() => expect(screen.getByTestId('window 2-send')).toHaveTextContent('false')); + await expectChip('window 2', CHIP[PUBLIC.model]); + expect(mocks.toastWarning).toHaveBeenCalledWith( + expect.objectContaining({ + title: 'Message not sent', + msg: expect.stringContaining( + `New chats now start on ${PUBLIC.model} (Claude Code, a public model), not ${PRIVATE.model}` + ), + }) + ); + }); + + it('lets a send through when the chip already states what a new chat binds', async () => { + renderTwoHomeWindows(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + fireEvent.click(inWindow('window 2').getByRole('button', { name: 'Send' })); + + await waitFor(() => expect(screen.getByTestId('window 2-send')).toHaveTextContent('true')); + expect(mocks.toastWarning).not.toHaveBeenCalled(); + }); +}); + +describe('F3 / privacy-tiers P4 — a switch made in a chat is about that chat', () => { + function renderChatAndHome(switchOptions?: ChangeModelOptions) { + return render( + <> + + + + + + + + ); + } + + /** + * QA F bound Claude Code in one chat for one check, and the next chat it + * opened came up public: the switch had silently moved the app-wide default. + * Unticked, it moves the chat and nothing else — so the other window's + * new-chat chip does not move, and it is RIGHT not to. + */ + it('leaves the model new chats start on alone, in every window', async () => { + renderChatAndHome(); + await expectChip('window 2', CHIP[PRIVATE.model]); + + fireEvent.click(inWindow('window 1').getByRole('button', { name: /this chat to Codex/ })); + + await waitFor(() => expect(mocks.updateAgentProvider).toHaveBeenCalledTimes(1)); + await waitFor(() => expect(mocks.toastSuccess).toHaveBeenCalled()); + expect(mocks.setConfigProvider).not.toHaveBeenCalled(); + expect(nextNewChatBinding()).toEqual(PRIVATE); + expect(chipIn('window 2')).toHaveAccessibleName(chipForDaemon()); + expect(mocks.toastSuccess).toHaveBeenCalledWith( + expect.objectContaining({ + msg: `This chat now uses ${CODEX.model} from ${CODEX.provider}. Other chats, and new ones, are unchanged.`, + }) + ); + }); + + it('moves every window when the user asks for new chats too', async () => { + renderChatAndHome({ alsoForNewChats: true }); + await expectChip('window 2', CHIP[PRIVATE.model]); + + fireEvent.click(inWindow('window 1').getByRole('button', { name: /this chat to Codex/ })); + + await waitFor(() => expect(nextNewChatBinding()).toEqual(CODEX)); + await expectChip('window 2', chipForDaemon()); + expect(chipIn('window 2')).toHaveAccessibleName(CHIP[CODEX.model]); + }); +}); + +describe('F3 — the pin still outranks a stale row', () => { + /** + * A chat's composer states the chat's own binding, and `chatBinding` prefers + * the turn-reported pin over the cached row (`privacy/pinnedModel.ts`, and the + * PIN note in `utils/sessionBindingSync.ts`). + * An app-wide change crossing windows must not disturb that: the chat below + * last RAN on Versa (the pin), its cached row still names Codex, and the + * selection moves to Claude Code. Its chip names Versa before and after. + */ + it('keeps naming the pinned model when the app-wide selection moves under it', async () => { + const stale = session({ + id: 'chat-b', + privacy_tier: 'private', + provider_name: CODEX.provider, + model_config: { model_name: CODEX.model } as Session['model_config'], + }); + render( + <> + + + + + + + + ); + await expectChip( + 'window 2', + `${CHIP[PRIVATE.model]}. Private chat. Biorouter only lets a private model open it.` + ); + + fireEvent.click(inWindow('window 1').getByRole('button', { name: /to Claude Code/ })); + await waitFor(() => expect(nextNewChatBinding()).toEqual(PUBLIC)); + await expectChip('window 1', CHIP[PUBLIC.model]); + + expect(chipIn('window 2')).toHaveAccessibleName( + `${CHIP[PRIVATE.model]}. Private chat. Biorouter only lets a private model open it.` + ); + expect(chipIn('window 2')).not.toHaveAccessibleName(new RegExp(CODEX.model)); + expect(chipIn('window 2')).not.toHaveAccessibleName(new RegExp(PUBLIC.model)); + }); +}); diff --git a/ui/desktop/src/components/ModelAndProviderContext.test.tsx b/ui/desktop/src/components/ModelAndProviderContext.test.tsx index 84eed2516..75409ee43 100644 --- a/ui/desktop/src/components/ModelAndProviderContext.test.tsx +++ b/ui/desktop/src/components/ModelAndProviderContext.test.tsx @@ -4,7 +4,12 @@ import { act, fireEvent, render, screen, waitFor } from '@testing-library/react'; import { useState } from 'react'; import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; -import { ModelAndProviderProvider, useModelAndProvider } from './ModelAndProviderContext'; +import { + ModelAndProviderProvider, + switchedModelMessage, + useModelAndProvider, + type ChangeModelOptions, +} from './ModelAndProviderContext'; import { subscribeSessionBindingChanges } from '../utils/sessionBindingSync'; import type Model from './settings/models/modelInterface'; @@ -167,14 +172,16 @@ const clientRejecting = (body: unknown) => async (options?: { throwOnError?: boo * asserted on the boolean the callers branch on rather than on a rendered toast * alone. */ -function SessionSwitchHarness() { +function SessionSwitchHarness({ options }: { options?: ChangeModelOptions } = {}) { const { changeModel } = useModelAndProvider(); const [result, setResult] = useState('pending'); return ( <> @@ -482,12 +489,12 @@ describe('ModelAndProviderProvider announces the binding it just wrote', () => { }); /** - * ⚠ **Ordering, not just occurrence.** The announcement lands BEFORE - * `setConfigProvider` moves the global default, so in the only render where - * the row and the selection can disagree it is the ROW that holds the new - * binding. Announcing afterwards would invert that window and flash the model - * the user had just switched away from — the regression PR #192 narrowed its - * rule to avoid. + * ⚠ **Ordering, not just occurrence.** When the switch moves the new-chat + * default too, the announcement lands BEFORE `setConfigProvider` moves it, so + * in the only render where the row and the selection can disagree it is the + * ROW that holds the new binding. Announcing afterwards would invert that + * window and flash the model the user had just switched away from — the + * regression PR #192 narrowed its rule to avoid. */ it('before the global default moves, not after', async () => { const order: string[] = []; @@ -503,7 +510,7 @@ describe('ModelAndProviderProvider announces the binding it just wrote', () => { render( - + ); @@ -689,3 +696,23 @@ describe('ModelAndProviderProvider config readiness', () => { await waitFor(() => expect(screen.getByTestId('status')).toHaveTextContent('ready:none')); }); }); + +/** + * F3 / privacy-tiers P4 — the success toast names where a switch landed. It + * used to say "Switched models — using X from Y" whichever of the chat and the + * new-chat default had moved, which is how a switch in one chat could quietly + * become what every new chat started on. + */ +describe('switchedModelMessage', () => { + it('names each of the three places a switch can land', () => { + expect(switchedModelMessage('Codex 6', 'Codex', { chat: true, newChats: false })).toBe( + 'This chat now uses Codex 6 from Codex. Other chats, and new ones, are unchanged.' + ); + expect(switchedModelMessage('Codex 6', 'Codex', { chat: true, newChats: true })).toBe( + 'This chat, and new chats in every window, now use Codex 6 from Codex.' + ); + expect(switchedModelMessage('Codex 6', 'Codex', { chat: false, newChats: true })).toBe( + 'New chats in every window now start on Codex 6 from Codex. Existing chats keep their own model.' + ); + }); +}); diff --git a/ui/desktop/src/components/ModelAndProviderContext.tsx b/ui/desktop/src/components/ModelAndProviderContext.tsx index 1b71ab9b2..775e19a2b 100644 --- a/ui/desktop/src/components/ModelAndProviderContext.tsx +++ b/ui/desktop/src/components/ModelAndProviderContext.tsx @@ -1,4 +1,12 @@ -import React, { createContext, useContext, useState, useEffect, useMemo, useCallback } from 'react'; +import React, { + createContext, + useContext, + useState, + useEffect, + useMemo, + useCallback, + useRef, +} from 'react'; import { toastError, toastSuccess } from '../toasts'; import Model, { getProviderMetadata, @@ -32,7 +40,11 @@ import { } from './ui/dialog'; import { Button } from './ui/button'; import { notifySessionToolsChanged } from '../utils/sessionToolEvents'; -import { announceSessionBinding } from '../utils/sessionBindingSync'; +import { + announceAppModelSelection, + announceSessionBinding, + subscribeAppModelSelectionChanges, +} from '../utils/sessionBindingSync'; // titles export const UNKNOWN_PROVIDER_TITLE = 'Provider name lookup'; @@ -42,7 +54,29 @@ export const UNKNOWN_PROVIDER_MSG = 'Unknown provider in config. Check your conf // success const CHANGE_MODEL_TOAST_TITLE = 'Model changed'; -const SWITCH_MODEL_SUCCESS_MSG = 'Switched models'; + +/** + * What the success toast says a switch changed. + * + * It used to say "Switched models — using X from Y" whatever had moved, which + * was how a switch made in one chat could quietly become the model every new + * chat started on. A switch now lands in one of three places (see + * `ChangeModelOptions`), and the toast names the one it landed in. + */ +export function switchedModelMessage( + label: string, + source: string, + scope: { chat: boolean; newChats: boolean } +): string { + const using = `${label} from ${source}`; + if (scope.chat && scope.newChats) { + return `This chat, and new chats in every window, now use ${using}.`; + } + if (scope.chat) { + return `This chat now uses ${using}. Other chats, and new ones, are unchanged.`; + } + return `New chats in every window now start on ${using}. Existing chats keep their own model.`; +} /** * Issue #56 DR-16. The one refusal in this feature addressed to the USER rather @@ -71,19 +105,66 @@ export const NO_USER_PROOF_TOAST_MSG = */ export type ModelConfigStatus = 'loading' | 'ready'; +/** + * The app-wide selection — `BIOROUTER_PROVIDER` / `BIOROUTER_MODEL` — as the + * daemon holds it. This is exactly what `/agent/start` binds a new chat to + * (`configured_new_session_provider`), and nothing else about it is implied: an + * existing chat runs on its own session row. `null` is "not set". + */ +export interface AppModelSelection { + provider: string | null; + model: string | null; +} + +/** + * Where a model switch lands. + * + * ⚠ **A switch made from inside a chat changes THAT CHAT, and nothing else, + * unless the user says otherwise.** Until 2026-09-11 it also rewrote the + * app-wide default, silently: provider QA F bound Claude Code in one chat for + * one check, and the next chat it opened came up public. That is the coupling + * `docs/security/privacy-tiers.md` §14.3 P4 asked to be undone — "pick Versa + * once in a scratch chat privatises not one session but every session created + * afterwards", and the mirror image makes every new chat public. The coupling is + * now the explicit opt-in below, offered in the dialog where the choice is made. + * + * A switch with no chat (Home's composer, a chat not yet started, Settings → + * Models, onboarding) has only one thing it can change — the model new chats + * start on — so it always does, and its dialog says so. + */ +export interface ChangeModelOptions { + /** Also make this the model every new chat starts on, in every window. */ + alsoForNewChats?: boolean; +} + interface ModelAndProviderContextType { currentModel: string | null; currentProvider: string | null; modelConfigStatus: ModelConfigStatus; currentModelSupportsVision: boolean; currentModelSupportedInputMimeTypes: string[] | null; - changeModel: (sessionId: string | null, model: Model) => Promise; + changeModel: ( + sessionId: string | null, + model: Model, + options?: ChangeModelOptions + ) => Promise; getCurrentModelAndProvider: () => Promise<{ model: string; provider: string }>; getFallbackModelAndProvider: () => Promise<{ model: string; provider: string }>; getCurrentModelAndProviderForDisplay: () => Promise<{ model: string; provider: string }>; getCurrentModelDisplayName: () => Promise; getCurrentProviderDisplayName: () => Promise; // Gets provider display name from subtext refreshCurrentModelAndProvider: () => Promise; + /** + * F3. Re-read the app-wide selection from the daemon and state it, now. + * + * Resolves with what was read, or `null` when the daemon could not answer — + * in which case nothing on screen changed, because a failed read is not + * evidence that nothing is configured. A pure read: unlike the mount-time + * {@link refreshCurrentModelAndProvider}, it never seeds the bundled default, + * so neither a window gaining focus nor another window's announcement can + * write config. + */ + syncAppModelSelection: () => Promise; } interface ModelAndProviderProviderProps { @@ -209,6 +290,37 @@ export const ModelAndProviderProvider: React.FC = const [llamaWarmupDialog, setLlamaWarmupDialog] = useState(null); const { read, getProviders, refreshConfig } = useConfig(); + /** + * F3 — the order in which statements of the app-wide selection may land. + * + * Three things now set `currentModel`/`currentProvider`: the mount read, a + * re-read (another window's announcement, this window regaining focus), and + * this window's own switch. Reads are async and overlap, so each takes a + * ticket when it is ISSUED and publishes only if no statement issued after it + * has already been published — a read that left before a switch landed cannot + * come back afterwards and restore the model the user switched away from. + * + * ⚠ Compared against what was last APPLIED, never against what was last + * issued: a newer read that FAILS publishes nothing, and must not thereby + * condemn an older one that succeeded (`docs/desktop-ui/renderer-testing-traps.md`, + * "Newest issued is the wrong rule"). + */ + const selectionIssued = useRef(0); + const selectionApplied = useRef(0); + + const takeSelectionTicket = useCallback(() => ++selectionIssued.current, []); + + const publishSelection = useCallback( + (ticket: number, model: string | null, provider: string | null): boolean => { + if (ticket < selectionApplied.current) return false; + selectionApplied.current = ticket; + setCurrentModel(model); + setCurrentProvider(provider); + return true; + }, + [] + ); + /** * Invalidate ConfigContext's cached snapshot after a write that bypassed it. * @@ -319,9 +431,12 @@ export const ModelAndProviderProvider: React.FC = }, [llamaWarmupDialog]); const changeModel = useCallback( - async (sessionId: string | null, model: Model) => { + async (sessionId: string | null, model: Model, options?: ChangeModelOptions) => { const modelName = model.name; const providerName = model.provider; + // See `ChangeModelOptions`: from a chat, the app-wide default moves only + // when the user asked for it; with no chat, it is the only thing to move. + const setsNewChatDefault = !sessionId || options?.alsoForNewChats === true; let phase = 'agent'; try { @@ -371,11 +486,13 @@ export const ModelAndProviderProvider: React.FC = // differs from the app-wide selection, would keep naming the model the // user just switched away from. // - // ⚠ **Here, not below.** This lands BEFORE `setConfigProvider` and - // before `setCurrentProvider`/`setCurrentModel`, so in the only render - // where the two can disagree the ROW holds the new binding and the - // selection still holds the old one. Announcing after the selection - // moved would invert that window and flash the old model. + // ⚠ **Here, not below.** When this switch moves the new-chat default + // as well, this lands BEFORE `setConfigProvider` and before the + // selection is published, so in the only render where the two can + // disagree the ROW holds the new binding and the selection still + // holds the old one. Announcing after the selection moved would invert + // that window and flash the old model. (A switch that moves only this + // chat never moves the selection, and the row alone is the change.) // // ⚠ And only after `updateAgentProvider` RESOLVED: a refusal (Gate A's // 409 for a public model on a private chat) throws past this line, and @@ -388,24 +505,34 @@ export const ModelAndProviderProvider: React.FC = }); } - phase = 'config'; - await setConfigProvider({ - body: { - provider: providerName, - model: modelName, - }, - headers: await userActionHeaders(), - throwOnError: true, - }); + if (setsNewChatDefault) { + phase = 'config'; + await setConfigProvider({ + body: { + provider: providerName, + model: modelName, + }, + headers: await userActionHeaders(), + throwOnError: true, + }); - setCurrentProvider(providerName); - setCurrentModel(modelName); - setModelConfigStatus('ready'); - await refreshCachedConfig(); + // A statement like any read, and ticketed like one: a read that left + // before this write landed is older than what is now on screen, and + // must not be allowed to come back and restore the previous model. + publishSelection(takeSelectionTicket(), modelName, providerName); + setModelConfigStatus('ready'); + await refreshCachedConfig(); + // F3. Every other window — and each one's next new chat — follows. + // After the write, never before: a receiver re-reads the daemon. + announceAppModelSelection(); + } toastSuccess({ title: CHANGE_MODEL_TOAST_TITLE, - msg: `${SWITCH_MODEL_SUCCESS_MSG} — using ${model.alias ?? modelName} from ${model.subtext ?? providerName}`, + msg: switchedModelMessage(model.alias ?? modelName, model.subtext ?? providerName, { + chat: !!sessionId, + newChats: setsNewChatDefault, + }), }); // Issue #56 DR-26 at the BIND surface. Binding a model covered by one // institution's agreements into a chat holding another institution's @@ -458,7 +585,7 @@ export const ModelAndProviderProvider: React.FC = return false; } }, - [prepareLlamaModel, refreshCachedConfig] + [prepareLlamaModel, refreshCachedConfig, publishSelection, takeSelectionTicket] ); const getFallbackModelAndProvider = useCallback(async () => { @@ -481,6 +608,9 @@ export const ModelAndProviderProvider: React.FC = }); // Same API-mediated write, same stale cache (#52). await refreshCachedConfig(); + // F3. A seeded default is a new-chat default like any other; a window + // that mounted before it was written would otherwise go on naming none. + announceAppModelSelection(); } catch (error) { console.error('[getFallbackModelAndProvider] Failed to write to config', error); } @@ -558,10 +688,10 @@ export const ModelAndProviderProvider: React.FC = }, [read, getCurrentModelAndProviderForDisplay]); const refreshCurrentModelAndProvider = useCallback(async () => { + const ticket = takeSelectionTicket(); try { const { model, provider } = await getCurrentModelAndProvider(); - setCurrentModel(model); - setCurrentProvider(provider); + publishSelection(ticket, model, provider); } catch (_error) { console.error('Failed to refresh current model and provider:', _error); } finally { @@ -570,7 +700,75 @@ export const ModelAndProviderProvider: React.FC = // would park every consumer on a spinner that never resolves. setModelConfigStatus('ready'); } - }, [getCurrentModelAndProvider]); + }, [getCurrentModelAndProvider, publishSelection, takeSelectionTicket]); + + const syncAppModelSelection = useCallback(async (): Promise => { + const ticket = takeSelectionTicket(); + let fresh: AppModelSelection; + try { + const [model, provider] = await Promise.all([ + read('BIOROUTER_MODEL', false), + read('BIOROUTER_PROVIDER', false), + ]); + // `/config/read` answers an unset key with `null`, and a failed read — + // a 500, an unreachable daemon — resolves with no body at all, which the + // generated client hands back as `undefined`. Only the first is a fact. + if (model === undefined || provider === undefined) return null; + fresh = { + model: typeof model === 'string' && model ? model : null, + provider: typeof provider === 'string' && provider ? provider : null, + }; + } catch (error) { + console.error('Failed to re-read the app-wide model selection:', error); + return null; + } + publishSelection(ticket, fresh.model, fresh.provider); + return fresh; + }, [read, publishSelection, takeSelectionTicket]); + + /** + * F3 — follow the app-wide selection for the life of this window. + * + * Two ears, one action (a pure re-read): + * + * - **Another window, or this one's `ConfigContext`, wrote it.** Every + * renderer write of `BIOROUTER_PROVIDER`/`BIOROUTER_MODEL` announces on + * `sessionBindingSync`'s channel, which reaches every window of the app. + * - **Something outside the renderer wrote it** — `biorouter configure` in a + * terminal (the app's own included), a hand-edited `config.yaml`. Nothing + * announces those; the daemon's config cache is keyed on the file's stamp, + * so `/agent/start` binds them at once. The window is re-read when it + * regains focus or becomes visible, which is when a user who made the + * change elsewhere comes back to act on it. The send path checks once more + * (`useConfirmNewChatModel`), because a terminal docked INSIDE the window + * never takes the window's focus away. + * + * ⚠ Mount-once, with the handler read through a ref at call time — the + * subscription must not be torn down and re-made because a callback's + * identity moved (the same rule `ConfigContext`'s catalogue subscription + * records, for the same reason: the subscription belongs to the mount). + */ + const syncAppModelSelectionRef = useRef(syncAppModelSelection); + useEffect(() => { + syncAppModelSelectionRef.current = syncAppModelSelection; + }, [syncAppModelSelection]); + + useEffect(() => { + const resync = () => { + void syncAppModelSelectionRef.current(); + }; + const onVisibility = () => { + if (document.visibilityState === 'visible') resync(); + }; + const unsubscribe = subscribeAppModelSelectionChanges(resync); + window.addEventListener('focus', resync); + document.addEventListener('visibilitychange', onVisibility); + return () => { + unsubscribe(); + window.removeEventListener('focus', resync); + document.removeEventListener('visibilitychange', onVisibility); + }; + }, []); // Derive vision support whenever the active model/provider changes useEffect(() => { @@ -658,6 +856,7 @@ export const ModelAndProviderProvider: React.FC = getCurrentModelDisplayName, getCurrentProviderDisplayName, refreshCurrentModelAndProvider, + syncAppModelSelection, }), [ currentModel, @@ -672,6 +871,7 @@ export const ModelAndProviderProvider: React.FC = getCurrentModelDisplayName, getCurrentProviderDisplayName, refreshCurrentModelAndProvider, + syncAppModelSelection, ] ); diff --git a/ui/desktop/src/components/privacy/useConfirmNewChatModel.test.tsx b/ui/desktop/src/components/privacy/useConfirmNewChatModel.test.tsx new file mode 100644 index 000000000..bf14f7ebd --- /dev/null +++ b/ui/desktop/src/components/privacy/useConfirmNewChatModel.test.tsx @@ -0,0 +1,192 @@ +import { act, renderHook } from '@testing-library/react'; +import { readFileSync } from 'node:fs'; +import { resolve } from 'node:path'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { + NEW_CHAT_MODEL_CHANGED_TITLE, + newChatModelChangedMessage, + useConfirmNewChatModel, +} from './useConfirmNewChatModel'; + +/** + * F3 — the last look before a composer creates a new chat. + * + * The cross-window suite (`ModelAndProviderContext.crossWindow.test.tsx`) drives + * this hook end to end against the real context. What is pinned here is the + * part that must NOT refuse: a check that blocked sends on a guess would be + * routed around, which is worse than the stale label it exists to catch. + */ + +const mocks = vi.hoisted(() => ({ + state: { + currentModel: 'gpt-5.5-2026-04-24' as string | null, + currentProvider: 'versa_azure' as string | null, + modelConfigStatus: 'ready' as 'ready' | 'loading', + }, + syncAppModelSelection: vi.fn(), + getProviders: vi.fn(), + toastWarning: vi.fn(), +})); + +vi.mock('../ModelAndProviderContext', () => ({ + useModelAndProvider: () => ({ + ...mocks.state, + syncAppModelSelection: mocks.syncAppModelSelection, + }), +})); + +vi.mock('../ConfigContext', () => ({ + useConfig: () => ({ getProviders: mocks.getProviders }), +})); + +vi.mock('../../toasts', () => ({ toastWarning: mocks.toastWarning })); + +const confirm = async () => { + const { result } = renderHook(() => useConfirmNewChatModel()); + let answer: boolean | undefined; + await act(async () => { + answer = await result.current(); + }); + return answer; +}; + +beforeEach(() => { + vi.clearAllMocks(); + mocks.state.currentModel = 'gpt-5.5-2026-04-24'; + mocks.state.currentProvider = 'versa_azure'; + mocks.state.modelConfigStatus = 'ready'; + mocks.getProviders.mockResolvedValue([ + { + name: 'claude_code', + metadata: { name: 'claude_code', display_name: 'Claude Code' }, + resolved_tier: 'public', + }, + ]); +}); + +describe('useConfirmNewChatModel', () => { + it('proceeds when the chip already states what a new chat binds', async () => { + mocks.syncAppModelSelection.mockResolvedValue({ + provider: 'versa_azure', + model: 'gpt-5.5-2026-04-24', + }); + expect(await confirm()).toBe(true); + expect(mocks.toastWarning).not.toHaveBeenCalled(); + }); + + it('refuses a known mismatch, naming the new model, its provider and its tier', async () => { + mocks.syncAppModelSelection.mockResolvedValue({ + provider: 'claude_code', + model: 'claude-fable-5-1', + }); + expect(await confirm()).toBe(false); + expect(mocks.toastWarning).toHaveBeenCalledWith({ + title: NEW_CHAT_MODEL_CHANGED_TITLE, + msg: + 'New chats now start on claude-fable-5-1 (Claude Code, a public model), not ' + + 'gpt-5.5-2026-04-24, which this window was still showing. Your message is back in the ' + + 'composer, and the model shown below is the one it will use.', + }); + }); + + /** No evidence is not a mismatch: `createSession` reports a dead daemon. */ + it('proceeds when the daemon could not be read', async () => { + mocks.syncAppModelSelection.mockResolvedValue(null); + expect(await confirm()).toBe(true); + expect(mocks.toastWarning).not.toHaveBeenCalled(); + }); + + /** + * The first frames after launch have no label on screen to be wrong about, + * and refusing there would block the fastest send for nothing. + */ + it('does not even look while nothing is named on screen', async () => { + mocks.state.modelConfigStatus = 'loading'; + mocks.state.currentModel = null; + mocks.state.currentProvider = null; + expect(await confirm()).toBe(true); + expect(mocks.syncAppModelSelection).not.toHaveBeenCalled(); + }); + + it('still refuses when the catalog cannot name the new provider', async () => { + mocks.getProviders.mockRejectedValue(new Error('catalog down')); + mocks.syncAppModelSelection.mockResolvedValue({ provider: 'codex', model: 'gpt-6-astra' }); + expect(await confirm()).toBe(false); + expect(mocks.toastWarning).toHaveBeenCalledWith( + expect.objectContaining({ + msg: expect.stringContaining('New chats now start on gpt-6-astra (codex), not'), + }) + ); + }); +}); + +describe('newChatModelChangedMessage', () => { + it('says so when no model is set for new chats any more', () => { + expect( + newChatModelChangedMessage( + 'gpt-5.5-2026-04-24', + { provider: null, model: null }, + null, + undefined + ) + ).toBe( + 'No model is set for new chats any more — this window was still showing ' + + 'gpt-5.5-2026-04-24. Your message is back in the composer.' + ); + }); + + it('states a private tier as plainly as a public one', () => { + expect( + newChatModelChangedMessage( + 'claude-fable-5-1', + { provider: 'versa_azure', model: 'gpt-5.5-2026-04-24' }, + 'Versa API Azure', + 'private' + ) + ).toContain('(Versa API Azure, a private model)'); + }); + + /** An unresolved tier is not "public", and it is not said to be. */ + it('says nothing about a tier it could not resolve', () => { + expect( + newChatModelChangedMessage('x', { provider: 'ollama', model: 'llama' }, 'Ollama', undefined) + ).toContain('(Ollama)'); + }); +}); + +/** + * ⚠ Both composers, and before `createSession`. Home (`Hub.tsx`) renders its + * OWN `ChatInput`, not `BaseChat`'s — the app launches onto it — so a check + * wired into one of them only is absent from half the ways a chat is started. + * And a check placed after + * `createSession` would run once `/agent/start` had already bound the chat. + * jsdom mounts neither surface cheaply, so this is pinned at the source. + */ +describe('both new-chat composers look before they create', () => { + const source = (file: string) => readFileSync(resolve(__dirname, '..', file), 'utf8'); + const CHECK = 'if (!(await confirmNewChatModel())) return false;'; + + /** + * Each composer's first side effect of a send, which the check must precede: + * a refused send has to leave everything as it found it. Home clears the + * pending extension overrides as it reads them; a chat marks itself as + * creating, which blocks the composer. + */ + it.each([ + ['Hub.tsx', 'clearExtensionOverrides();'], + ['BaseChat.tsx', 'setIsCreatingSession(true);'], + ])('%s checks the model before createSession and before %s', (file, firstSideEffect) => { + const text = source(file); + const check = text.indexOf(CHECK); + expect(check).toBeGreaterThan(-1); + // Exactly one check per composer — a second one would be a second policy. + expect(text.indexOf(CHECK, check + CHECK.length)).toBe(-1); + + const sideEffect = text.indexOf(firstSideEffect, check); + const create = text.indexOf('await createSession(', check); + expect(sideEffect).toBeGreaterThan(check); + expect(create).toBeGreaterThan(sideEffect); + // And nothing of the kind slipped in ahead of it. + expect(text.slice(Math.max(0, check - 400), check)).not.toContain(firstSideEffect); + }); +}); diff --git a/ui/desktop/src/components/privacy/useConfirmNewChatModel.ts b/ui/desktop/src/components/privacy/useConfirmNewChatModel.ts new file mode 100644 index 000000000..7322ed60b --- /dev/null +++ b/ui/desktop/src/components/privacy/useConfirmNewChatModel.ts @@ -0,0 +1,101 @@ +import { useCallback } from 'react'; +import { useConfig } from '../ConfigContext'; +import { useModelAndProvider, type AppModelSelection } from '../ModelAndProviderContext'; +import { readResolvedProviderTier } from './useBoundProviderTier'; +import { toastWarning } from '../../toasts'; +import type { ProviderTier } from '../../api/types.gen'; + +export const NEW_CHAT_MODEL_CHANGED_TITLE = 'Message not sent'; + +/** + * The toast for a send refused because the model on screen was not the model a + * new chat would have started on. + * + * Written for a person, and it states the tier in words because the tier is + * the reason this check exists: a window reading "Private model, UCSF" must not + * start a chat on a public model with nothing said. It names what is true now, + * what the window had been showing, and where the message went — and stops. + */ +export function newChatModelChangedMessage( + shownModel: string, + fresh: AppModelSelection, + freshProviderName: string | null, + freshTier: ProviderTier | undefined +): string { + if (!fresh.model || !fresh.provider) { + return ( + `No model is set for new chats any more — this window was still showing ${shownModel}. ` + + 'Your message is back in the composer.' + ); + } + const tier = + freshTier === 'private' + ? ', a private model' + : freshTier === 'public' + ? ', a public model' + : ''; + return ( + `New chats now start on ${fresh.model} (${freshProviderName ?? fresh.provider}${tier}), ` + + `not ${shownModel}, which this window was still showing. Your message is back in the ` + + 'composer, and the model shown below is the one it will use.' + ); +} + +/** + * F3 — the last look before a composer creates a NEW chat. + * + * `/agent/start` binds a new chat to whatever `BIOROUTER_PROVIDER` / + * `BIOROUTER_MODEL` say on the daemon at that instant, and it takes no provider + * of its own. Everything the composer states about the model — name, gauge, + * cost, the "Private model, UCSF" padlock — comes from this window's copy of + * those two keys. `ModelAndProviderContext` keeps that copy current: every + * window hears every renderer write, and re-reads on focus. What it cannot hear + * is a write from outside the renderer while this window keeps its focus — a + * `biorouter configure` in the terminal docked inside this very window is the + * ordinary case. So the send path asks once more, at the only moment the answer + * decides anything. + * + * Resolves `true` to proceed. `false` means the window was stale: the fresh + * selection is already on screen (the re-read published it), a toast says what + * changed, and the caller returns `false` so `ChatInput` puts the text back. + * + * ⚠ It refuses only a KNOWN mismatch. Nothing named on screen yet (the first + * frames after launch, or no model at all) and a read that failed both proceed + * exactly as before: the first had no label to be wrong about, and the second + * has no evidence — `createSession` reports a daemon that cannot answer. + * + * ⚠ Not a gate. The daemon classifies the chat by what it binds, correctly, + * whatever this does; this only keeps the human's last look honest. + */ +export function useConfirmNewChatModel(): () => Promise { + const { currentModel, currentProvider, modelConfigStatus, syncAppModelSelection } = + useModelAndProvider(); + const { getProviders } = useConfig(); + + return useCallback(async () => { + if (modelConfigStatus !== 'ready' || !currentProvider || !currentModel) return true; + + const fresh = await syncAppModelSelection(); + if (!fresh) return true; + if (fresh.provider === currentProvider && fresh.model === currentModel) return true; + + let providerName: string | null = null; + let tier: ProviderTier | undefined; + if (fresh.provider) { + try { + const row = (await getProviders(false)).find( + (candidate) => candidate.name === fresh.provider + ); + providerName = row?.metadata?.display_name ?? null; + tier = readResolvedProviderTier(row); + } catch { + // The sentence is still complete and still true without either. + } + } + toastWarning({ + title: NEW_CHAT_MODEL_CHANGED_TITLE, + msg: newChatModelChangedMessage(currentModel, fresh, providerName, tier), + }); + return false; + }, [currentModel, currentProvider, modelConfigStatus, syncAppModelSelection, getProviders]); +} diff --git a/ui/desktop/src/components/settings/models/bottom_bar/ModelsBottomBar.pinned.test.tsx b/ui/desktop/src/components/settings/models/bottom_bar/ModelsBottomBar.pinned.test.tsx index f8a2a93b1..5a3bacb0c 100644 --- a/ui/desktop/src/components/settings/models/bottom_bar/ModelsBottomBar.pinned.test.tsx +++ b/ui/desktop/src/components/settings/models/bottom_bar/ModelsBottomBar.pinned.test.tsx @@ -1,6 +1,10 @@ import { fireEvent, render, screen, waitFor } from '@testing-library/react'; import { beforeEach, describe, expect, it, vi } from 'vitest'; -import ModelsBottomBar, { CHAT_KEEPS_ITS_MODEL_NOTE } from './ModelsBottomBar'; +import ModelsBottomBar, { + CHAT_KEEPS_ITS_MODEL_NOTE, + NEW_CHATS_MODEL_HEADING, + NEW_CHATS_MODEL_NOTE, +} from './ModelsBottomBar'; import { __resetDisclosureStoreForTests } from '../../../privacy/disclosureCopy'; /** @@ -174,3 +178,38 @@ describe('a chat bound to something other than the app-wide selection', () => { expect(screen.queryByTestId('chat-binding-note')).toBeNull(); }); }); + +/** + * F3 — where there is no chat yet (Home, a chat not started), the chip names + * the APP-WIDE selection, and a switch from it changes that selection for every + * window. The dropdown says whose model it is and how far a change reaches, + * beside the control that makes the change. + */ +describe('the chip where there is no chat yet', () => { + const renderSessionless = () => + render( + + ); + + const openDropdown = async () => { + await screen.findByRole('button', { name: /Current model:/ }); + fireEvent.pointerDown(screen.getByLabelText(/Current model/), { button: 0, ctrlKey: false }); + }; + + it('heads its dropdown as the model for new chats, reaching every window', async () => { + renderSessionless(); + await openDropdown(); + + expect(await screen.findByText(NEW_CHATS_MODEL_HEADING)).toBeInTheDocument(); + expect(screen.getByTestId('new-chats-model-note')).toHaveTextContent(NEW_CHATS_MODEL_NOTE); + expect(screen.queryByText('Current model')).toBeNull(); + }); + + it('keeps "Current model", and no such line, in a chat', async () => { + renderBar(undefined); + await openDropdown(); + + expect(await screen.findByText('Current model')).toBeInTheDocument(); + expect(screen.queryByTestId('new-chats-model-note')).toBeNull(); + }); +}); diff --git a/ui/desktop/src/components/settings/models/bottom_bar/ModelsBottomBar.tsx b/ui/desktop/src/components/settings/models/bottom_bar/ModelsBottomBar.tsx index 31813e60a..743710bd3 100644 --- a/ui/desktop/src/components/settings/models/bottom_bar/ModelsBottomBar.tsx +++ b/ui/desktop/src/components/settings/models/bottom_bar/ModelsBottomBar.tsx @@ -45,6 +45,19 @@ import type { PinnedModelView } from '../../../../hooks/chatStreamStore'; export const CHAT_KEEPS_ITS_MODEL_NOTE = 'This chat keeps the model it was last set to. A model chosen elsewhere applies to new chats.'; +/** + * F3 — the heading and line this chip's dropdown carries where there is no chat + * yet (Home, a chat not started). + * + * There the chip names the APP-WIDE selection — the pair `/agent/start` will + * bind — and switching from it changes that pair for every window. "Current + * model" read as a property of this screen; the heading says whose model it is, + * and the line says how far a change reaches, beside the control that makes it. + */ +export const NEW_CHATS_MODEL_HEADING = 'Model for new chats'; +export const NEW_CHATS_MODEL_NOTE = + 'New chats in every window start on this model. Existing chats keep their own.'; + interface ModelsBottomBarProps { sessionId: string | null; dropdownRef: React.RefObject; @@ -518,11 +531,21 @@ export default function ModelsBottomBar({
-
Current model
+
+ {sessionId ? 'Current model' : NEW_CHATS_MODEL_HEADING} +
{shownModelName} {shownProviderName && ` · ${shownProviderName}`}
+ {!sessionId && ( +
+ {NEW_CHATS_MODEL_NOTE} +
+ )} {/* Under the heading "Current model", so it must be about the model. It used to be `privacyLine`. */} {modelTierWords && ( diff --git a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.privacy.test.tsx b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.privacy.test.tsx index eb5f44bf6..b7febdf6d 100644 --- a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.privacy.test.tsx +++ b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.privacy.test.tsx @@ -278,11 +278,16 @@ describe('SwitchModelModal — pre-flight, not post-refusal', () => { fireEvent.click(confirm); await waitFor(() => expect(mocks.changeModel).toHaveBeenCalledTimes(1)); - expect(mocks.changeModel).toHaveBeenCalledWith('s1', { - name: 'gpt-5.6-sol', - provider: 'codex', - subtext: 'Codex', - }); + expect(mocks.changeModel).toHaveBeenCalledWith( + 's1', + { + name: 'gpt-5.6-sol', + provider: 'codex', + subtext: 'Codex', + }, + // Opened from a chat, the box is offered and starts unticked. + { alsoForNewChats: false } + ); }); // The control case: a public provider has no affiliation at all, so the row is diff --git a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.test.tsx b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.test.tsx index edaf1894e..58b25cd51 100644 --- a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.test.tsx +++ b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.test.tsx @@ -1,6 +1,12 @@ import { act, fireEvent, render, screen, waitFor } from '@testing-library/react'; import { beforeEach, describe, expect, it, vi } from 'vitest'; -import { SwitchModelModal } from './SwitchModelModal'; +import { + ALSO_FOR_NEW_CHATS_HINT, + ALSO_FOR_NEW_CHATS_LABEL, + SWITCH_SCOPE_NEW_CHATS, + SWITCH_SCOPE_THIS_CHAT, + SwitchModelModal, +} from './SwitchModelModal'; const mocks = vi.hoisted(() => ({ getProviders: vi.fn(), @@ -221,3 +227,120 @@ describe('SwitchModelModal switch feedback', () => { expect(unhandled).toEqual([]); }); }); + +/** + * F3 / `docs/security/privacy-tiers.md` §14.3 P4 — the dialog says what a + * switch changes, before the user commits. + * + * Until 2026-09-11 a switch made from a chat's composer also rewrote the model + * every new chat starts on, in every window, with nothing on screen to say so: + * provider QA F bound Claude Code in one chat for one check, and the next chat + * it opened came up public. The coupling is now an explicit, unticked box, and + * the dialog opened with no chat says plainly that it is the app-wide choice. + */ +describe('SwitchModelModal — what the switch changes', () => { + beforeEach(() => { + vi.clearAllMocks(); + mocks.getProviders.mockResolvedValue([ + { + name: 'versa_azure', + is_configured: true, + provider_type: 'Institutional', + metadata: { + name: 'versa_azure', + display_name: 'Versa API Azure', + default_model: 'gpt-5.5-2026-04-24', + known_models: [{ name: 'gpt-5.5-2026-04-24' }], + allows_unlisted_models: false, + config_keys: [], + }, + }, + ]); + mocks.getProviderModels.mockResolvedValue(['gpt-5.5-2026-04-24']); + mocks.read.mockResolvedValue(''); + mocks.changeModel.mockResolvedValue(true); + }); + + const renderModal = (sessionId: string | null) => + render( + + ); + + const confirm = () => + waitFor(() => { + const found = screen.getAllByRole('button').find((el) => el.textContent === 'Select model'); + if (!found) throw new Error('Select model button not rendered'); + return found; + }); + + /** Let the provider list land, and the model list after it, inside act. */ + const settle = async () => { + await waitFor(() => + expect(screen.getByTestId('provider-select')).toHaveTextContent('versa_azure') + ); + await act(async () => { + await new Promise((resolve) => setTimeout(resolve, 0)); + }); + }; + + it('from a chat, says it switches this chat and offers new chats as an unticked box', async () => { + renderModal('s-1'); + + expect(screen.getByText(SWITCH_SCOPE_THIS_CHAT)).toBeInTheDocument(); + const box = screen.getByRole('checkbox', { name: new RegExp(ALSO_FOR_NEW_CHATS_LABEL) }); + expect(box).not.toBeChecked(); + expect(screen.getByText(ALSO_FOR_NEW_CHATS_HINT)).toBeInTheDocument(); + await settle(); + }); + + it('leaves new chats alone unless the box is ticked', async () => { + renderModal('s-1'); + fireEvent.click(await confirm()); + + await waitFor(() => expect(mocks.changeModel).toHaveBeenCalledTimes(1)); + expect(mocks.changeModel).toHaveBeenCalledWith( + 's-1', + expect.objectContaining({ name: 'gpt-5.5-2026-04-24', provider: 'versa_azure' }), + { alsoForNewChats: false } + ); + }); + + it('carries a ticked box through to the switch', async () => { + renderModal('s-1'); + fireEvent.click(screen.getByRole('checkbox', { name: new RegExp(ALSO_FOR_NEW_CHATS_LABEL) })); + fireEvent.click(await confirm()); + + await waitFor(() => expect(mocks.changeModel).toHaveBeenCalledTimes(1)); + expect(mocks.changeModel).toHaveBeenCalledWith( + 's-1', + expect.objectContaining({ name: 'gpt-5.5-2026-04-24' }), + { alsoForNewChats: true } + ); + }); + + /** + * With no chat — Home, a chat not started, Settings → Models, onboarding — + * the only thing a switch can change is the model new chats start on, so + * there is no box to offer, and the description says how far it reaches. + */ + it('with no chat, says it sets the model new chats start on in every window', async () => { + renderModal(null); + + expect(screen.getByText(SWITCH_SCOPE_NEW_CHATS)).toBeInTheDocument(); + expect(screen.queryByRole('checkbox')).toBeNull(); + + await settle(); + fireEvent.click(await confirm()); + await waitFor(() => expect(mocks.changeModel).toHaveBeenCalledTimes(1)); + expect(mocks.changeModel).toHaveBeenCalledWith( + null, + expect.objectContaining({ name: 'gpt-5.5-2026-04-24' }) + ); + }); +}); diff --git a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.tsx b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.tsx index 5fead65d5..7cc37e1bc 100644 --- a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.tsx +++ b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.tsx @@ -10,6 +10,7 @@ import { DialogTitle, } from '../../../ui/dialog'; import { Button } from '../../../ui/button'; +import { Checkbox } from '../../../ui/Checkbox'; import { QUICKSTART_GUIDE_URL } from '../../providers/modal/constants'; import { Input } from '../../../ui/input'; import { Select } from '../../../ui/Select'; @@ -135,6 +136,23 @@ const modelOptionSearchText = (option: ModelOption) => const PUBLIC_MODEL_IN_PRIVATE_CHAT = 'Unavailable: this is a private chat, so only private models may run in it'; +/** + * F3 / privacy-tiers §14.3 P4 — what a switch from THIS dialog changes, said in + * the dialog, before the user commits. + * + * Opened from a chat, the dialog changes that chat and nothing else unless the + * box below the pickers is ticked; opened with no chat — Home's composer, a chat + * not started yet, Settings → Models, onboarding — the only thing it can change + * is the model new chats start on, in every window. The old description, "for + * your chats", fitted neither, and the switch it described did both. + */ +export const SWITCH_SCOPE_THIS_CHAT = 'Select a provider and model for this chat.'; +export const SWITCH_SCOPE_NEW_CHATS = + 'Select the provider and model new chats start on, in every window. Existing chats keep their own model.'; +export const ALSO_FOR_NEW_CHATS_LABEL = 'Also use for new chats'; +export const ALSO_FOR_NEW_CHATS_HINT = + 'New chats in every window will start on this model. Left unticked, only this chat changes.'; + const renderModelOptionLabel = ( rawOption: unknown, meta: { context: 'menu' | 'value' }, @@ -380,6 +398,12 @@ export const SwitchModelModal = ({ */ const [switching, setSwitching] = useState(false); const [submitError, setSubmitError] = useState(null); + /** + * P4's "Also make this my default for new chats", as an explicit and + * unticked box. Only offered from a chat: with no chat there is nothing else + * the switch could change. See `ChangeModelOptions`. + */ + const [alsoForNewChats, setAlsoForNewChats] = useState(false); const handleSubmit = async () => { // The button below is already disabled on a browser surface; this is the @@ -412,7 +436,9 @@ export const SwitchModelModal = ({ modelObj = { name: model, provider: provider, subtext: providerDisplayName } as Model; } - const changed = await changeModel(sessionId, modelObj); + const changed = sessionId + ? await changeModel(sessionId, modelObj, { alsoForNewChats }) + : await changeModel(null, modelObj); if (!changed) { // `changeModel` has already raised the toast that explains *why* — a // privacy barrier, a missing user proof, a provider failure. This says @@ -699,7 +725,9 @@ export const SwitchModelModal = ({ {hostManaged ? HOST_MANAGED_MODEL_TITLE - : 'Select a provider and model to use for your chats.'} + : sessionId + ? SWITCH_SCOPE_THIS_CHAT + : SWITCH_SCOPE_NEW_CHATS} @@ -909,6 +937,30 @@ export const SwitchModelModal = ({ )}
+ {/* + P4's opt-in, directly above the confirm it modifies. Unticked by + default: a switch made in a chat is a statement about that chat, and a + public model chosen for one scratch chat must not become what every + new chat — in every window — silently starts on. + */} + {sessionId && !hostManaged && ( +
+ setAlsoForNewChats(event.target.checked)} + disabled={switching} + className="mt-0.5" + /> + +
+ )} + {submitError && (
Date: Fri, 11 Sep 2026 11:54:54 -0700 Subject: [PATCH 17/75] docs: model selection across windows (F3), and P4's decoupling in the ledger MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New docs/desktop-ui/model-selection-across-windows.md: the two facts a model chip can state (a chat's binding vs the app-wide selection /agent/start binds), what a switch changes from each surface and why (privacy-tiers §14.3 P4), how each window stays current (announcements, focus re-reads, ticketed reads), the last look before a new chat, what it does not cover, the tests, and how to check it in the running app. Indexed in docs/desktop-ui/README.md. privacy-tiers.md's "What shipped" ledger gains a dated line for P4, so §14.3 no longer reads as open. --- docs/desktop-ui/README.md | 1 + .../model-selection-across-windows.md | 147 ++++++++++++++++++ docs/security/privacy-tiers.md | 7 + 3 files changed, 155 insertions(+) create mode 100644 docs/desktop-ui/model-selection-across-windows.md diff --git a/docs/desktop-ui/README.md b/docs/desktop-ui/README.md index ff74729d8..612f94e24 100644 --- a/docs/desktop-ui/README.md +++ b/docs/desktop-ui/README.md @@ -34,6 +34,7 @@ arrived looking for one of those, leave now. | [Where a generated artifact is displayed](artifact-display-surfaces.md) | The rule that a figure, an app card or any generated artifact has exactly ONE display surface — the artifact side panel — on all three transcript surfaces (live chat, saved session, shared session), and why that is enforced by a required prop rather than by convention. Covers what the removed inline renderer actually cost (a second CSP, a second action channel, a second resize contract, a fabricated session id), what was deleted with it, and what deliberately stayed. Current. | | [The preview panel](preview-panel/README.md) | The working documents for expanding the artifact side panel: a measured survey of every render branch, image list and guard as it stands, and the plan to widen it along five axes — more image formats, the Office gap around the renderers that already ship, live websites in their own native view, an annotation channel back into the chat, and agent access to what the panel is showing. Plan **executed**; the implementation record covers what shipped, what was verified against real Electron, and what is still open. | | [The settings visual vocabulary](settings-visual-vocabulary.md) | The ten rules the Settings view (Models, Chat, App), the chat-history surfaces, the Scheduler and the component views (Workflows, Extensions, Skills, Built apps) are built to, and the primitives they lean on: a row's fill never depends on its state, rows are direct children of their list, one note shape with a ceiling on it, type roles rather than sizes, the button ladder, and one shared page header whose actions sit on their own line under the description. Covers why six of them are asserted at the source rather than in a render test — jsdom never runs Tailwind, and two of the defects only appear in the cascade. Current. | +| [Model selection across windows](model-selection-across-windows.md) | What keeps the composer's model chip — name, gauge, cost and privacy padlock — equal to what the next turn will run on in every window (provider-QA F3): a chat's own binding versus the app-wide selection `/agent/start` binds, the rule that a switch made in a chat changes that chat unless "Also use for new chats" is ticked (privacy-tiers §14.3 P4), the nudge every renderer write announces across windows, the ticketed re-reads, and the last look before a new chat is created. Current. | | [The provider catalog](provider-catalog.md) | The one surface listing every provider the daemon serves: three tabs carrying §14.5's privacy taxonomy, institutions named from the daemon's affiliation payload rather than from a literal, AI agents ahead of the API providers, and a default tab computed from the bound provider / a subscription-ready CLI / Local. Also the first-run screen, which is the same component in `mode="onboarding"`, the `BIOROUTER_ONBOARDING_SKIPPED` escape from it, and the composer's no-model state that makes that escape honest. Current. | | [The settings visual vocabulary](settings-visual-vocabulary.md) | The nine rules the Settings view (Models, Chat, App) is built to, and the two primitives they lean on: a row's fill never depends on its state, rows are direct children of their list, one note shape with a ceiling on it, type roles rather than sizes, and the button ladder. Covers why six of them are asserted at the source rather than in a render test — jsdom never runs Tailwind, and two of the defects only appear in the cascade. Current. | | [Diverge behavior checklist](diverge-behavior-checklist.md) | A catalog of 68 user actions for Diverge — the feature that branches a conversation into a new session — each paired with the behavior BioRouter must exhibit, serving as both a manual QA script and the spec the automated tests encode. Current; last revised 2026-07-18, when the dashboard-canvas items were deleted alongside dashboard mode itself. | diff --git a/docs/desktop-ui/model-selection-across-windows.md b/docs/desktop-ui/model-selection-across-windows.md new file mode 100644 index 000000000..64170201e --- /dev/null +++ b/docs/desktop-ui/model-selection-across-windows.md @@ -0,0 +1,147 @@ +# Model selection across windows + +> **What this is.** The rules that keep the composer's model chip — its name, gauge, cost line and "Private model, UCSF" padlock — equal to what the next turn will actually run on, in every window, and the decision about what a model switch changes. +> **Status:** Current. Shipped for provider-QA finding F3 (2026-09-10) on top of `main` at `7c96d796`. +> **Audience:** developers working on the desktop renderer's model selection, the composer, or privacy tiers. + +A chat runs on one of two things, and the chip has to state the right one at the moment a +person acts on it. Provider QA measured the failure on 2026-09-10: with two windows open, +changing the model in window 1 left window 2's chip reading `gpt-5.5-2026-04-24 (Private +model, UCSF)`, and the chat window 2 started bound `claude_code` — a consumer subscription +with no BAA (business associate agreement). The privacy barrier held: the chat was classified +`public`. What failed was the label the user read before typing. + +## The two facts a chip can state + +| Fact | Where it lives | What it decides | Who states it | +|---|---|---|---| +| **A chat's binding** | the session row (`provider_name`, `model_config`), outranked by the pin a turn reports | what an existing chat's next turn runs on — Gate B rebinds from the row | the chip inside a chat that has one | +| **The app-wide selection** | `BIOROUTER_PROVIDER` / `BIOROUTER_MODEL` in `config.yaml` | what a **new** chat binds — `/agent/start` reads exactly these two keys (`configured_new_session_provider`) and accepts no provider of its own | the chip on Home and in a chat not yet started | + +`privacy/pinnedModel.ts` (`chatBinding`) and `privacy/usePinnedModel.ts` choose between the +two for a chat. This document is about keeping each one *current* — before F3 the second was +read once, when `ModelAndProviderContext` mounted, and never again. + +## What a model switch changes + +The model switcher (`SwitchModelModal`) is reached from the chip, from Settings → Models and +from onboarding, and every one of them ends in `ModelAndProviderContext.changeModel`. + +| Opened from | What the switch changes | What the dialog says | +|---|---|---| +| a chat that exists | **that chat only** — its session row, through `/agent/update_provider` | "Select a provider and model for this chat." plus an unticked **Also use for new chats** box | +| a chat, with the box ticked | that chat **and** the app-wide selection | the box's hint: new chats in every window will start on it | +| Home, a chat not yet started, Settings → Models, onboarding | the app-wide selection — the only thing there is to change | "Select the provider and model new chats start on, in every window. Existing chats keep their own model." | + +The success toast names which of the three happened (`switchedModelMessage`), and the chip's +dropdown, where there is no chat, is headed **Model for new chats** with the line "New chats +in every window start on this model. Existing chats keep their own." + +> **Why.** Until 2026-09-11 a switch made in a chat also rewrote the app-wide selection, +> silently. QA F bound Claude Code in one chat for one check, and the next chat it opened came +> up public. [`privacy-tiers.md`](../security/privacy-tiers.md) §14.3 **P4** had already asked +> for this coupling to be undone — "pick Versa once in a scratch chat privatises not one +> session but every session created afterwards" — with an explicit "also make this my default +> for new chats" control. The box is that control, and it starts unticked because a switch +> made inside a chat reads as a statement about that chat. + +## How each window stays current + +Every window of the app is its own renderer with its own `ModelAndProviderContext`, over one +daemon. + +- **Every renderer write announces.** `changeModel`, the first-run seeding of the bundled + default (`getFallbackModelAndProvider`), and `ConfigContext.upsert` / `remove` of either key + call `announceAppModelSelection` (`utils/sessionBindingSync.ts`). The last one covers the + writers that never pass through `changeModel`: the local and coding-agent onboarding cards, + Lead/Worker settings and Settings' reset. Before F3 those did not update even their own + window's chip. +- **The announcement is a nudge, not a payload.** It travels on the same `BroadcastChannel` + as the per-chat binding (`biorouter:session-binding`), shaped `{ kind: 'app-model-selection' }` + and carrying no provider and no model. Two windows' writes can be announced in the opposite + order from the one they landed in, so every receiver re-reads the daemon and ends on the + write that landed last — the one `/agent/start` will bind. +- **A window re-reads when it regains focus or becomes visible.** Nothing announces a write + made outside the renderer: `biorouter configure` in a terminal, or a hand-edited + `config.yaml`. The daemon's config cache is keyed on the file's stamp, so `/agent/start` + binds such a write at once. +- **Reads are ticketed.** The mount read, every re-read and the window's own switch each take + a ticket when issued, and one publishes only if nothing issued after it has already been + published. The comparison is against what was last *applied*, never what was last + *issued* — see [renderer testing traps](renderer-testing-traps.md). A re-read that comes + back with no body (a 500, a daemon that is restarting) changes nothing on screen: a failed + read is not evidence that nothing is configured. +- **A re-read never writes.** `syncAppModelSelection` is a pure read. Only the mount-time + `refreshCurrentModelAndProvider` may seed the bundled default, so neither a focus event nor + another window's announcement can write config. + +## The last look before a new chat + +Both composers that create a chat — Home (`Hub.tsx`) and a chat not yet started +(`BaseChat.tsx`) — call `useConfirmNewChatModel` immediately before `createSession`, ahead +of anything the send consumes. It re-reads the pair and compares it with what the chip +showed. On a mismatch it: + +1. publishes the fresh pair, so the chip, gauge, cost and padlock change; +2. raises a **Message not sent** toast naming the new model, its provider and its tier in + words ("New chats now start on claude-fable-5-1 (Claude Code, a public model), not + gpt-5.5-2026-04-24, which this window was still showing…"); +3. resolves `false`, and `ChatInput` puts the text back. + +It exists for the one write no ear hears in time: a `biorouter configure` in the terminal docked +**inside** the window, which never takes the window's focus. It refuses only a known mismatch — +nothing named on screen yet, or a read that failed, both proceed as before. It is not a gate: +the daemon classifies the chat by what it binds, whatever this check does. + +## What this does not cover + +- **The pin still outranks the row.** An app-wide change touches neither a session row nor a + turn-reported pin, and a chat's chip goes on naming its own binding. The cross-window suite + pins this. +- **Lead/Worker's `(lead)` / `(worker)` suffix** is read by `ModelsBottomBar` when it mounts. + The model name beside it is live; the suffix in a second window is not. +- **An unannounced write while the window keeps focus** leaves the chip stale until the next + focus change or the next new-chat send, which re-reads before it creates anything. An + existing chat is unaffected either way: it runs on its own row. +- **The window between the last look and `/agent/start`** — two loopback round trips — is not + closed. Closing it needs `/agent/start` to accept an expected binding and refuse a + mismatch, which is daemon work. + +## Tests + +| Suite | What it pins | +|---|---| +| `components/ModelAndProviderContext.crossWindow.test.tsx` | two providers, each rendering the real chip over one fake daemon: a second window's chip — model, provider and privacy — follows the nudge without a remount; it equals what the next `/agent/start` would bind after a switch in either direction; ordering (a slow older read, a failed newer read); focus and visibility re-reads; the send refusal; a per-chat switch leaving new chats alone; the pin outranking a stale row | +| `utils/sessionBindingSync.test.ts` | the nudge is synchronous locally, carries no values, crosses windows, and never mixes with a binding | +| `components/ConfigContext.test.tsx` | `upsert` / `remove` of the two keys announce once the write resolved; other keys and refused writes do not | +| `components/privacy/useConfirmNewChatModel.test.tsx` | when the last look refuses and when it must not; both composers call it before `createSession`, pinned at the source | +| `settings/models/subcomponents/SwitchModelModal.test.tsx` | the dialog's scope copy and the unticked box | +| `settings/models/bottom_bar/ModelsBottomBar.pinned.test.tsx` | the "Model for new chats" heading where there is no chat | + +```bash +cd ui/desktop && npx vitest run src/components/ModelAndProviderContext* src/utils/sessionBindingSync* +``` + +## Checking it in the running app + +Launch a sandboxed instance with the dev GUI launcher on a free CDP port, open a second window +with `window.electron.createChatWindow()` from the first window's DevTools, and read each +window's chip by its accessible name (`button[aria-label^="Current model:"]`). + +1. Change the model from window 1's Home chip (or tick **Also use for new chats** in a chat). + Window 2's chip must read the new model within two seconds. +2. Send from window 2's Home composer, then read what the turn really ran on: + + ```bash + sqlite3 "$SANDBOX/sessions/sessions.db" "select provider, model_id from token_events order by id desc limit 1" + ``` + +3. Repeat in the private → public direction. At no point may window 2's chip read "Private + model, UCSF" while its next turn goes to a public model. + +## Related documentation + +- [Privacy tiers](../security/privacy-tiers.md) — §14.3 P4 is the decoupling this ships; the ledger at the top says what else shipped. +- [Renderer testing traps](renderer-testing-traps.md) — why the ticket compares against the last applied read. +- [The provider catalog](provider-catalog.md) — the other surface that opens the model switcher, including onboarding. +- [Launching the dev GUI from a shell without a TTY](launching-the-dev-gui.md) — how to put two windows in front of you for the runtime check. diff --git a/docs/security/privacy-tiers.md b/docs/security/privacy-tiers.md index 8153afa87..2f2fbfc04 100644 --- a/docs/security/privacy-tiers.md +++ b/docs/security/privacy-tiers.md @@ -112,6 +112,13 @@ this section is the ledger. higher price for a yes, never the absence of one** — a build that withheld the control there would restore the hard block DR-26 exists to prevent, for exactly the deployments careful enough to choose `strict`. +- **§14.3 P4's decoupling — added 2026-09-11, after this ledger was written.** A model switch made + in a chat changes that chat only; making it the model new chats start on is an explicit, + unticked "Also use for new chats" box in the switcher. Provider QA F measured the coupling it + replaces: one chat switched to Claude Code for one check, and the next chat opened came up + public. The same change (QA finding F3) makes every window's chip follow the app-wide + selection live, so no window names a private model while its next new chat would bind a public + one. See [model selection across windows](../desktop-ui/model-selection-across-windows.md). ### Did not ship From 6579877258d4a632264bdfa5d16f66366de1aa41 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 12:00:05 -0700 Subject: [PATCH 18/75] feat(agent): enforce the checklist for multi-step turns (planning gate) The Todo capability could always keep a checklist; nothing made the model keep one. The only trigger was an advisory paragraph in system.md, the todo MOIM rendered nothing while the list was empty, the old "don't stop with unchecked todos" gate had been removed, and under Code Execution mode the todo tools were reachable only from inside a script. Three multi-step QA sessions on 2026-09-10 produced zero checklists. This makes the checklist a harness behaviour rather than a prompt hint: - Code Execution filter: the five todo__* tools stay directly callable (exact names via todo_extension::TODO_TOOL_NAMES, never a prefix). - agents::planning_gate: a small tested classifier reads the user's prompt (numbered/bulleted list of actions, 3+ instructions, or 2+ joined by then/after that/finally/steps; how-to questions and code never count). For a multi-step turn with an empty checklist: * the MOIM carries a "write the checklist first" reminder until it exists; * the first non-todo tool batch is refused with a pointer to todo_write, once per turn, as a synthetic inspection denial in inspect_and_gate_tool_requests (not a registered inspector: the bridge and approval relay would run it too, and the bridge flattens reasons); * a turn that created or changed the list cannot end while items are unfinished unless the final message names each one; at most STOP_HOOK_BLOCK_CAP blocks per turn, on a count tools do not reset. - Scope is one predicate (enforcement_applies) read by the turn and by the system prompt, so the prompt's new clause describes exactly what runs: not Chat mode, not subagents, not coding-agent bridge providers, only with todo__todo_write on the roster. Todo disabled: none of it fires. - The Stop-hook arms move out of the reply_internal generator into decide_turn_stop (checklist check, then hooks), shrinking its poll frame. - Docs: todo.md, code-execution.md and hooks-reference.md say what is enforced; stale sessionTodos/ChatSummary comments corrected; a ChatSummary test pins a direct todo_write appearing and ticking in the summary. --- crates/biorouter/src/agents/agent.rs | 224 ++- crates/biorouter/src/agents/mod.rs | 3 + crates/biorouter/src/agents/moim.rs | 70 +- crates/biorouter/src/agents/planning_gate.rs | 1712 +++++++++++++++++ crates/biorouter/src/agents/prompt_manager.rs | 38 +- crates/biorouter/src/agents/reply_parts.rs | 66 +- crates/biorouter/src/agents/todo_extension.rs | 55 +- crates/biorouter/src/prompts/system.md | 11 + docs/agent-loop/hooks/hooks-reference.md | 21 + docs/extensions/built-in/code-execution.md | 10 + docs/extensions/built-in/todo.md | 38 +- .../src/components/ChatSummary.test.tsx | 70 + ui/desktop/src/utils/sessionTodos.ts | 20 +- 13 files changed, 2258 insertions(+), 80 deletions(-) create mode 100644 crates/biorouter/src/agents/planning_gate.rs diff --git a/crates/biorouter/src/agents/agent.rs b/crates/biorouter/src/agents/agent.rs index f471c0ea3..53bd2d0a8 100644 --- a/crates/biorouter/src/agents/agent.rs +++ b/crates/biorouter/src/agents/agent.rs @@ -692,13 +692,27 @@ fn skill_already_loaded_pointer() -> &'static str { // `user` message every turn. That over-reached: when the agent was genuinely stuck // (e.g. an unrecoverable provider error), it re-injected the same message forever // and never resolved the root cause — and it polluted the conversation with fake -// user input. "Don't stop while work is unfinished" is now left to the proper, -// bounded, user-configurable mechanisms: the Stop-hook system (`StopHookVerdict`, -// capped by `STOP_HOOK_BLOCK_CAP`, delivered as hidden-visibility feedback + a -// user-facing system notification) and the `/goal` loop (whose stall budget does -// NOT reset when tools run, so it gives up when progress stalls). A user who wants -// "keep going until the todos are done" sets a `/goal` or a Stop hook — both go -// through that bounded, stall-aware path instead of an unbounded loop injection. +// user input. "Don't stop while work is unfinished" now goes through bounded +// paths only: the Stop-hook system (`StopHookVerdict`, capped by +// `STOP_HOOK_BLOCK_CAP`, delivered as hidden-visibility feedback + a user-facing +// system notification), the `/goal` loop (whose stall budget does NOT reset when +// tools run, so it gives up when progress stalls), and — for the checklist +// itself — the planning gate's stop check (`agents::planning_gate`). That check +// is the old gate's intent without its defects: it runs only on a turn that +// worked the list, is satisfied by a final message that NAMES the open items, +// speaks through the same hidden feedback + notice as a Stop hook, and blocks at +// most `STOP_HOOK_BLOCK_CAP` times per turn on a count that does not reset when +// tools run. + +/// How a turn's attempt to finish resolves — decided by +/// [`Agent::decide_turn_stop`], acted on by the reply loop. +enum TurnStop { + /// Let the turn end, after showing the user `notices`. + Finish { notices: Vec }, + /// Keep working: `feedback` goes to the model as a hidden steer, `notice` + /// to the user. + KeepWorking { feedback: String, notice: String }, +} /// Context needed for the reply function pub struct ReplyContext { @@ -3462,6 +3476,9 @@ pub struct Agent { pub(super) hooks_manager: Arc, /// Active `/goal` conditions per session (see [`crate::agents::goal`]). pub(super) goals: crate::agents::goal::GoalRegistry, + /// The planning gate's per-turn state, per session (see + /// [`crate::agents::planning_gate`]). + pub(super) planning: crate::agents::planning_gate::PlanningRegistry, /// Lazily-created scheduler for `/loop`/`/schedule` when no /// `scheduler_service` was injected (plain CLI/TUI sessions). pub(super) fallback_scheduler: tokio::sync::OnceCell>, @@ -4293,6 +4310,7 @@ impl Agent { )), hooks_manager, goals: Default::default(), + planning: Default::default(), fallback_scheduler: tokio::sync::OnceCell::new(), vault: Mutex::new(None), soft_interrupts: Arc::new(std::sync::Mutex::new(SoftInterrupts::new())), @@ -5509,6 +5527,9 @@ impl Agent { let _phase = super::phase_timing::Phase::start("agent.assemble_turn_context"); let moim_phase = super::phase_timing::Phase::start("agent.inject_moim"); + // The planning gate's "write the checklist first", for as long as a + // multi-step turn's list is empty. See `planning_gate`. + let reminder = self.planning_reminder(session_id).await; let (conversation, moim_injected) = super::moim::inject_moim( session_id, conversation.clone(), @@ -5516,6 +5537,7 @@ impl Agent { working_dir, &self.normalizer, cancel, + reminder.as_deref(), ) .await; drop(moim_phase); @@ -5589,6 +5611,18 @@ impl Agent { inspection_results.append(&mut revalidated); } + // The planning gate's once-per-turn redirect, added as inspection + // results so the permission merge, the denial path and the transcript + // treat it like any other refusal. Deliberately NOT a registered + // inspector: the coding-agent bridge and the approval relay run the + // registered set too, and the bridge flattens every denial to a + // generic "denied by policy" — a pointer to `todo_write` that never + // reaches the model is a refusal with no way forward. + inspection_results.extend( + self.planning_gate_denials(&session.id, remaining_requests) + .await, + ); + let permission_check_result = self .tool_inspection_manager .process_inspection_results_with_permission_inspector( @@ -6412,6 +6446,14 @@ impl Agent { { result.reason.clone() } + // The planning gate: nobody declined, and the reason is + // the instruction — write the checklist, then repeat. + Some(result) + if result.inspector_name + == crate::agents::planning_gate::PLANNING_GATE_NAME => + { + result.reason.clone() + } _ => DECLINED_RESPONSE.to_string(), }; let mut response = response_msg.lock().await; @@ -8617,6 +8659,14 @@ impl Agent { self.restore_goal(&session_config.id).await; let message_text = user_message.as_concat_text(); + // The planning gate's reading of this prompt. Classified here, where + // the prompt text is known, and armed in `reply_internal`, where the + // turn's tool roster is. A slash command is not a request to plan. + let planning_signal = if message_text.trim().starts_with('/') { + None + } else { + crate::agents::planning_gate::classify_request(&message_text) + }; // User-configured hooks: SessionStart fires once per session, then // UserPromptSubmit may block the prompt or inject context. Slash @@ -9018,7 +9068,7 @@ impl Agent { } }; - let mut reply_stream = self.reply_internal(final_conversation, rewrite_basis, session_config, session, cancel_token).await?; + let mut reply_stream = self.reply_internal(final_conversation, rewrite_basis, session_config, session, cancel_token, planning_signal).await?; while let Some(event) = reply_stream.next().await { yield event?; } @@ -9180,6 +9230,79 @@ impl Agent { } } + /// Everything that decides whether a turn may end, in order: the planning + /// gate's checklist check, then the Stop hooks (a `/goal` judge is one). + /// + /// The checklist goes first because it is deterministic and cheap — a + /// command hook may run a test suite and a prompt hook costs a model call, + /// neither worth paying on a stop the checklist is about to refuse. It is + /// skipped while a `/goal` is active: that session already has a judge + /// re-reading the work on every stop, and a checklist block would be + /// counted against the goal's own budget by `stop_hook_block_feedback`. + /// + /// Split out of the `reply_internal` generator to keep its `poll` frame + /// small — see [`ToolBatchMaps`]. + async fn decide_turn_stop( + &self, + session_id: &str, + working_dir: &std::path::Path, + conversation: &Conversation, + active_goal: Option, + ) -> TurnStop { + let mut notices = Vec::new(); + if active_goal.is_none() { + match self.checklist_stop(session_id, conversation).await { + crate::agents::planning_gate::ChecklistStop::Block { feedback, notice } => { + return TurnStop::KeepWorking { feedback, notice }; + } + crate::agents::planning_gate::ChecklistStop::GiveUp { notice } => { + notices.push(notice); + } + crate::agents::planning_gate::ChecklistStop::Clear => {} + } + } + + let transcript_tail = crate::agents::goal::transcript_tail(conversation); + match self + .hooks_manager + .stop(session_id, working_dir, transcript_tail) + .await + { + crate::hooks::StopHookVerdict::Proceed => { + // An active goal whose evaluator let the stop proceed is met: + // clear it and tell the user. + if let Some(goal) = active_goal { + self.clear_goal(session_id).await; + notices.push(format!( + "🎯 Goal met and cleared: {}", + crate::agents::goal::ellipsize(&goal.condition, 200) + )); + } + TurnStop::Finish { notices } + } + crate::hooks::StopHookVerdict::CapReached => { + let goal_hint = if active_goal.is_some() { + " The /goal stays active and will be re-evaluated next turn; run /goal clear to stop it." + } else { + "" + }; + notices.push(format!( + "Stop hook block limit ({}) reached; finishing anyway.{}", + crate::hooks::STOP_HOOK_BLOCK_CAP, + goal_hint + )); + TurnStop::Finish { notices } + } + crate::hooks::StopHookVerdict::Blocked { reason } => { + // The goal-budget accounting lives on `stop_hook_block_feedback`. + let (feedback, notice) = self + .stop_hook_block_feedback(session_id, &reason, active_goal.is_some()) + .await; + TurnStop::KeepWorking { feedback, notice } + } + } + } + /// Bill one overflow-recovery compaction's provider round-trips to both the /// reply budget and the session gauge. /// @@ -9221,6 +9344,7 @@ impl Agent { session_config: SessionConfig, session: Session, cancel_token: Option, + planning_signal: Option, ) -> Result>> { let session_manager = self.config.session_manager.clone(); let provider_conversation = crate::conversation::without_bedrock_reasoning(&conversation); @@ -9249,6 +9373,15 @@ impl Agent { } = context; let reply_span = tracing::Span::current(); self.reset_retry_attempts().await; + // Opened here, outside the generator: the turn's roster is final and + // `session` still holds the checklist as the turn found it. + self.begin_planning_turn( + &session, + planning_signal, + &tools, + &toolshim_tools, + reply_provider.uses_tool_bridge_for_tool_surface(), + ); // Freshness basis for this turn's overflow-recovery compactions. // @@ -11270,69 +11403,40 @@ impl Agent { } } - let transcript_tail = crate::agents::goal::transcript_tail(&conversation); - match self.hooks_manager.stop(&session_config.id, &session.working_dir, transcript_tail).await { - crate::hooks::StopHookVerdict::Proceed => { - // An active goal whose evaluator let the stop - // proceed is met: clear it and tell the user. - if let Some(goal) = active_goal { - self.clear_goal(&session_config.id).await; - yield AgentEvent::Message( - inline_notice_user_only( - format!( - "🎯 Goal met and cleared: {}", - crate::agents::goal::ellipsize(&goal.condition, 200) - ), - ), - ); + // The planning gate's checklist check, then the Stop hooks + // (a /goal judge is one). The deciding lives in + // `decide_turn_stop`, out of this generator's `poll` frame; + // only the yields are left here. + match self.decide_turn_stop( + &session_config.id, + &session.working_dir, + &conversation, + active_goal, + ).await { + TurnStop::Finish { notices } => { + for notice in notices { + yield AgentEvent::Message(inline_notice_user_only(notice)); } break; } - crate::hooks::StopHookVerdict::CapReached => { - let goal_hint = if active_goal.is_some() { - " The /goal stays active and will be re-evaluated next turn; run /goal clear to stop it." - } else { - "" - }; - yield AgentEvent::Message( - inline_notice_user_only( - format!( - "Stop hook block limit ({}) reached; finishing anyway.{}", - crate::hooks::STOP_HOOK_BLOCK_CAP, - goal_hint - ), - ), - ); - break; - } - crate::hooks::StopHookVerdict::Blocked { reason } => { - // The goal-budget accounting lives on - // `stop_hook_block_feedback`. - let (feedback_text, notice) = self.stop_hook_block_feedback( - &session_config.id, - &reason, - active_goal.is_some(), - ).await; - + TurnStop::KeepWorking { feedback, notice } => { // #59 / #66 SHAPE 2: hidden from the user, named for // the client. let (feedback, published) = persist_steering_message( &session_manager, &session_config.id, - feedback_text, + feedback, ).await?; if let Some(published) = published { yield published; } conversation.push(feedback); - yield AgentEvent::Message( - inline_notice_user_only(notice,), - ); + yield AgentEvent::Message(inline_notice_user_only(notice)); // Keep looping: the model sees the feedback next turn. - // After a give-up the goal is cleared, so the next stop + // After a goal gives up it is cleared, so the next stop // proceeds once the agent delivers its wrap-up. // - // #69: a blocked Stop reverses the exit the queue was + // #69: a blocked stop reverses the exit the queue was // closed for, so re-open it for the extra work. self.reopen_for_more_work(); } @@ -16661,6 +16765,16 @@ mod tests { script, so Code Execution mode must leave it directly callable: {names:?}" ); } + // The checklist is the second exemption, and the reason is the model + // rather than the plumbing: behind a script wrapper it went unused. + // `agent_for_tests` loads Todo, so all five are on this roster. + for expected in crate::agents::todo_extension::TODO_TOOL_NAMES { + assert!( + names.contains(&expected), + "{expected} must stay a direct call in Code Execution mode, or the planning \ + gate points the model at a tool it can only reach from a script: {names:?}" + ); + } } /// The exemption is written against the ONE predicate that also decides the diff --git a/crates/biorouter/src/agents/mod.rs b/crates/biorouter/src/agents/mod.rs index 430b9c16e..3e68e865b 100644 --- a/crates/biorouter/src/agents/mod.rs +++ b/crates/biorouter/src/agents/mod.rs @@ -38,6 +38,9 @@ pub mod post_edit_diagnostics; // Stage 0 of the tool-call latency work: opt-in per-phase timing behind // `BIOROUTER_PHASE_TIMING=1`, free when off. pub mod phase_timing; +// Native checklist control for a multi-step turn: the reminder, the +// once-per-turn redirect to `todo_write`, and the bounded stop check. +pub(crate) mod planning_gate; pub mod prompt_manager; mod recurring; // BR-12: `pub(crate)` so `context_mgmt::run_eager_compaction` can reuse diff --git a/crates/biorouter/src/agents/moim.rs b/crates/biorouter/src/agents/moim.rs index c2d60f62d..c396e7c5a 100644 --- a/crates/biorouter/src/agents/moim.rs +++ b/crates/biorouter/src/agents/moim.rs @@ -59,9 +59,26 @@ fn strip_existing_moim(messages: &mut Vec) { }); } +/// Put `reminder` at the head of the block, directly after the opening tag. +/// +/// The head, not the tail, because [`cap_moim_block`] keeps the head: a +/// reminder appended after a large workspace map would be the first thing the +/// cap cut. +fn with_reminder(moim: String, reminder: Option<&str>) -> String { + match reminder.map(str::trim).filter(|text| !text.is_empty()) { + Some(reminder) => moim.replacen(MOIM_OPEN_TAG, &format!("{MOIM_OPEN_TAG}\n{reminder}"), 1), + None => moim, + } +} + /// Inject the MOIM `` block into the conversation handed to the model, /// returning the (re-normalized) conversation and whether a block was injected. /// +/// `reminder` is a first-party line the agent loop wants in front of the model +/// for this call only — the planning gate's "write the checklist first". It +/// rides inside the block, so it is never persisted and disappears the moment +/// the loop stops passing it. +/// /// BR-56: normalization goes through the agent's [`SharedNormalizer`], which /// re-fixes only the messages appended since the last call instead of the whole /// history — this runs on every provider call, so in a long multi-tool turn it was @@ -73,6 +90,7 @@ pub async fn inject_moim( working_dir: &Path, normalizer: &SharedNormalizer, cancel: Option<&CancellationToken>, + reminder: Option<&str>, ) -> (Conversation, bool) { if SKIP.with(|f| f.get()) { return (conversation, false); @@ -82,7 +100,7 @@ pub async fn inject_moim( .collect_moim(session_id, working_dir, cancel) .await { - let moim = cap_moim_block(moim, max_moim_tokens()); + let moim = cap_moim_block(with_reminder(moim, reminder), max_moim_tokens()); let mut messages = conversation.messages().clone(); // Drop any stale MOIM from a prior loop iteration first, so a long // multi-tool turn never accumulates several near-identical (and @@ -145,6 +163,7 @@ mod tests { &working_dir, &SharedNormalizer::new(), None, + None, ) .await; let msgs = result.messages(); @@ -184,6 +203,7 @@ mod tests { &working_dir, &SharedNormalizer::new(), None, + None, ) .await; @@ -257,6 +277,7 @@ mod tests { &working_dir, &SharedNormalizer::new(), None, + None, ) .await; let msgs = result.messages(); @@ -358,6 +379,52 @@ mod tests { assert_eq!(cap_moim_block(moim.clone(), 8_000), moim); } + /// The planning gate's reminder rides inside the one block, at its head, so + /// the size cap — which keeps the head — cannot be what removes it. + #[tokio::test] + async fn a_reminder_rides_at_the_head_of_the_block_and_survives_the_cap() { + let temp_dir = tempfile::tempdir().unwrap(); + let em = ExtensionManager::new_without_provider(temp_dir.path().to_path_buf()); + let conv = Conversation::new_unvalidated(vec![Message::user().with_text("do it")]); + let (result, injected) = inject_moim( + "test-session-id", + conv, + &em, + &PathBuf::from("/test/dir"), + &SharedNormalizer::new(), + None, + Some("Planning required: write the checklist first."), + ) + .await; + assert!(injected); + assert_eq!(count_info_msgs(&result), 1, "one block, not two"); + let block = result.messages()[0] + .content + .iter() + .filter_map(|c| c.as_text()) + .find(|t| t.contains(MOIM_OPEN_TAG)) + .expect("the block") + .to_string(); + assert!( + block.starts_with(&format!( + "{MOIM_OPEN_TAG}\nPlanning required: write the checklist first." + )), + "{block}" + ); + + let huge = format!( + "{MOIM_OPEN_TAG}\nIt is currently now\n{}\n{MOIM_CLOSE_TAG}", + "x".repeat(40_000) + ); + let capped = cap_moim_block(with_reminder(huge, Some("KEEP ME")), 100); + assert!(capped.contains("KEEP ME"), "{capped}"); + assert!(is_moim_block(&capped)); + + // No reminder, or a blank one, leaves the block exactly as it was. + assert_eq!(with_reminder(sample_moim(), None), sample_moim()); + assert_eq!(with_reminder(sample_moim(), Some(" ")), sample_moim()); + } + /// BR-2: a cap of 0 disables MOIM truncation. #[test] fn test_cap_moim_block_disabled_with_zero() { @@ -385,6 +452,7 @@ mod tests { &working_dir, &SharedNormalizer::new(), None, + None, ) .await; diff --git a/crates/biorouter/src/agents/planning_gate.rs b/crates/biorouter/src/agents/planning_gate.rs new file mode 100644 index 000000000..f7ce01441 --- /dev/null +++ b/crates/biorouter/src/agents/planning_gate.rs @@ -0,0 +1,1712 @@ +//! The planning gate: native control of a multi-step turn's checklist. +//! +//! The Todo capability could always keep a checklist; nothing made the model +//! keep one. The only trigger was an advisory paragraph in `system.md`, +//! `TodoClient::get_moim` renders nothing while the list is empty, and the +//! hard-coded "don't stop with unchecked todos" gate was removed for +//! re-injecting a fake user message forever (the NOTE near the top of +//! `agent.rs`). Measured on 2026-09-10: three multi-step QA sessions, zero +//! checklists. This module is the harness half of the fix — three behaviours, +//! each bounded, each of which a model that disagrees can get past: +//! +//! 1. **Reminder.** A turn whose prompt [`classify_request`] reads as several +//! steps, in a session whose checklist is empty, carries a "write the +//! checklist first" line in its MOIM block until the list exists. +//! 2. **Redirect.** The first batch of non-Todo tool calls in such a turn is +//! refused with a pointer back to `todo__todo_write` — ONCE per turn. A model +//! that states a reason and repeats the call gets it run. +//! 3. **Stop check.** A turn that created or changed the checklist cannot end +//! while items are unfinished, unless its final message names each one. It +//! blocks at most [`STOP_HOOK_BLOCK_CAP`] times per turn and, unlike the +//! Stop-hook counter, the count does NOT reset when tools run — so a model +//! that keeps stopping cannot cycle until `max_turns`. +//! +//! **Scope** is one predicate, [`enforcement_applies`], read by both the turn +//! and the system prompt so the prompt can never promise what the turn does not +//! do: tool-running modes only (not Chat), not a subagent, not a coding-agent +//! provider (its tool calls run through the bridge, which never passes the +//! reply loop's gate), and only while `todo__todo_write` is on the model's +//! roster. Disable the Todo capability and none of it fires. +//! +//! Everything that runs inside the reply loop lives here, as `Agent` methods, +//! so the `reply_internal` generator only calls them — its `poll` frame sits a +//! few percent under the thread stack in debug builds, and every line kept out +//! of it is paid for two or three times over during delegation. + +use std::collections::{HashMap, HashSet}; +use std::hash::{Hash, Hasher}; +use std::sync::PoisonError; + +use once_cell::sync::Lazy; +use regex::Regex; +use rmcp::model::{Role, Tool}; + +use crate::agents::final_output_tool::FINAL_OUTPUT_TOOL_NAME; +use crate::agents::todo_extension::{is_todo_tool_name, TODO_WRITE_TOOL_NAME}; +use crate::config::BioRouterMode; +use crate::conversation::message::ToolRequest; +use crate::conversation::Conversation; +use crate::hooks::STOP_HOOK_BLOCK_CAP; +use crate::session::extension_data::{TodoItem, TodoState, TodoStatus}; +use crate::session::session_manager::SessionType; +use crate::session::Session; +use crate::tool_inspection::{InspectionAction, InspectionResult}; + +use super::Agent; + +/// `InspectionResult::inspector_name` on the gate's refusals, so the denial +/// path hands the model the real reason instead of claiming the user declined. +pub(crate) const PLANNING_GATE_NAME: &str = "planning_gate"; + +/// The two Todo calls that put items on an empty checklist. A batch carrying +/// one is writing the plan in the same step as the work, so the gate lets the +/// whole batch through. +const CHECKLIST_SEEDING_TOOLS: [&str; 2] = ["todo__todo_write", "todo__todo_add"]; + +/// Sessions whose turn state one [`Agent`] keeps at once. An entry outlives its +/// turn only until the session's next turn replaces it; the bound is for a +/// daemon hosting many chats on one agent (`biorouter web`). +const MAX_TRACKED_SESSIONS: usize = 256; + +// --------------------------------------------------------------------------- +// The classifier +// --------------------------------------------------------------------------- + +/// Why a request reads as several steps. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MultiStep { + /// A numbered (`1.`, `2)`, `(3)`, `Step 4:`) or bulleted list of actions. + List { items: usize }, + /// Three or more instructions in a row, each opening with an action verb. + Instructions { count: usize }, + /// Two or more instructions joined by an explicit sequencing word — "then", + /// "after that", "finally", "steps". + Sequenced { count: usize }, +} + +impl MultiStep { + fn describe(self) -> String { + match self { + Self::List { items } => format!("a list of {items} steps"), + Self::Instructions { count } => format!("{count} separate instructions"), + Self::Sequenced { count } => format!("{count} instructions in sequence"), + } + } +} + +/// Code is pasted data, never the request's own steps: a stack trace with +/// numbered frames is not a plan. +static FENCED_CODE: Lazy = + Lazy::new(|| Regex::new(r"(?s)```.*?(?:```|\z)").expect("valid regex")); + +/// Inline code, replaced by a neutral word so `write `x.txt`` keeps its verb. +static INLINE_CODE: Lazy = Lazy::new(|| Regex::new(r"`[^`\n]+`").expect("valid regex")); + +/// A numbered-list marker: `1.` `2)` `(3)` or `Step 4:`, after the start of the +/// text, whitespace or a bracket, and before whitespace. The whitespace on both +/// sides is what keeps `3.12` and `v1.2` out. +static NUMBERED_MARKER: Lazy = Lazy::new(|| { + Regex::new(r"(?i)(?:^|[\s(\[])(?:step\s+)?\(?(\d{1,2})[.):]\s+").expect("valid regex") +}); + +/// A bulleted line. +static BULLET: Lazy = + Lazy::new(|| Regex::new(r"(?m)^[ \t]*[-*+•][ \t]+(\S[^\n]*)$").expect("valid regex")); + +/// Where one instruction ends and the next may begin. `.` counts only before +/// whitespace, so `hello.txt` and `3.12` stay whole. +static CLAUSE_BREAK: Lazy = Lazy::new(|| { + Regex::new( + r"(?i)[.!?;:](?:\s+|$)|[,\n]|\b(?:and|then|after\s+that|afterwards?|finally|lastly|also|plus)\b", + ) + .expect("valid regex") +}); + +/// An explicit statement that the work comes in order. +static SEQUENCE_CUE: Lazy = Lazy::new(|| { + Regex::new( + r"(?i)\b(?:then|after\s+that|afterwards?|finally|lastly|followed\s+by|once\s+(?:that|this|it)(?:'s|\s+is)?\s+(?:done|finished|complete)|steps?)\b", + ) + .expect("valid regex") +}); + +/// A request for an explanation, not for work. "How do I create a venv and +/// then install the deps?" names two actions in sequence and wants neither +/// performed; the system prompt already says to answer those first. +static INFORMATIONAL: Lazy = Lazy::new(|| { + Regex::new( + r"(?i)\b(?:how\s+(?:to|do|does|did|can|could|would|should|might|is|are)|what(?:'s|\s+is)\s+the\s+(?:best\s+)?way\s+to|explain\s+how|walk\s+me\s+through)\b", + ) + .expect("valid regex") +}); + +/// Verbs that open an instruction to DO something — to files, data, code, a +/// service. Base forms only: an imperative uses them, a description ("it +/// validates each row") does not. Answer-shaped verbs (explain, describe, +/// summarize, tell, compare, show, translate) are left out on purpose: three +/// questions in a row are not three steps of work. +const ACTION_VERB_LIST: &str = "\ + add adjust aggregate align analyze analyse annotate append apply archive assemble \ + attach audit automate backup benchmark bisect build bump calculate cancel capture cd \ + change check checkout chmod clean clear clone close cluster collect combine commit \ + compile compress compute concatenate configure connect convert copy correct count \ + crawl create crop curl debug decompress decrypt deduplicate delete deploy detect \ + diff disable download draw drop dump duplicate edit email embed enable encode \ + encrypt erase estimate evaluate execute expand export extend extract fetch filter \ + find fit fix flatten fork format gather generate get grep group gunzip gzip hash \ + identify implement import increase index ingest initialize initialise insert inspect \ + install integrate join kill label launch lint link list load locate lookup make map \ + mark measure merge migrate mkdir modify monitor mount move mv normalize normalise \ + notify open optimize optimise organize organise package parse paste patch ping plot \ + populate post prepare preprocess print process profile prune publish pull push put \ + query rank read rebase rebuild record recompute redact redo reduce refactor refresh \ + regenerate register reindex reinstall release reload remove rename render reorder \ + reorganize repair replace replicate reproduce request rerun resample reset reshape \ + resize resolve restart restore restructure retrain retrieve retry revert review \ + rewrite rm rotate run sample save scaffold scale scan schedule scrape search seed \ + select send separate serialize set setup share shuffle sign simulate slice smooth \ + sort split stage standardize start stash stop store strip submit subset subtract sum \ + swap switch symlink sync tabulate tag tar test tidy tokenize touch trace track train \ + transfer transform transpose trigger trim truncate tune uncompress undo uninstall \ + unpack unzip update upgrade upload validate verify visualize visualise watch wget \ + wire wrap write zip"; + +static ACTION_VERBS: Lazy> = + Lazy::new(|| ACTION_VERB_LIST.split_whitespace().collect()); + +/// Words that can precede an instruction's verb without changing it. Longest +/// first, so "i need you to" is stripped before "i need to" is tried. +const FILLERS: &[&[&str]] = &[ + &["i", "would", "like", "you", "to"], + &["i'd", "like", "you", "to"], + &["i", "want", "you", "to"], + &["i", "need", "you", "to"], + &["go", "ahead", "and"], + &["make", "sure", "to"], + &["make", "sure", "you"], + &["i", "need", "to"], + &["i", "want", "to"], + &["we", "need", "to"], + &["you", "need", "to"], + &["after", "that"], + &["can", "you"], + &["could", "you"], + &["would", "you"], + &["will", "you"], + &["you", "should"], + &["let", "us"], + &["try", "to"], + &["please"], + &["kindly"], + &["then"], + &["and"], + &["also"], + &["now"], + &["just"], + &["next"], + &["finally"], + &["lastly"], + &["first"], + &["firstly"], + &["second"], + &["secondly"], + &["third"], + &["thirdly"], + &["afterwards"], + &["afterward"], + &["let's"], + &["lets"], + &["step"], + &["so"], +]; + +/// Read the user's prompt and say whether it asks for several steps of work. +/// +/// Deliberately a small, conservative heuristic — English cues only, and a +/// false negative is cheap (the system prompt still asks the model to plan) +/// where a false positive costs one redirected tool call. It fires on: +/// +/// * a numbered or bulleted list whose items open with action verbs — or of +/// three or more items introduced by an instruction ("Build a page with: +/// 1. a header 2. a footer 3. a nav bar"); +/// * three or more instructions, each opening with an action verb; +/// * two or more instructions joined by a sequencing word ("then", "after +/// that", "finally", "steps"). +/// +/// A request for an explanation ("how do I …") never fires, and fenced or +/// inline code is ignored. +pub fn classify_request(text: &str) -> Option { + let prose = prose_only(text); + if INFORMATIONAL.is_match(&prose) { + return None; + } + if let Some(items) = listed_steps(&prose) { + return Some(MultiStep::List { items }); + } + let count = action_clause_count(&prose); + if count >= 3 { + Some(MultiStep::Instructions { count }) + } else if count >= 2 && SEQUENCE_CUE.is_match(&prose) { + Some(MultiStep::Sequenced { count }) + } else { + None + } +} + +fn prose_only(text: &str) -> String { + let without_fences = FENCED_CODE.replace_all(text, "\n"); + INLINE_CODE.replace_all(&without_fences, "x").into_owned() +} + +/// The longest list in `prose` that reads as steps, as its item count. +fn listed_steps(prose: &str) -> Option { + [numbered_list(prose), bulleted_list(prose)] + .into_iter() + .flatten() + .filter(|(intro, items)| list_is_steps(intro, items)) + .map(|(_, items)| items.len()) + .max() +} + +fn list_is_steps(intro: &str, items: &[&str]) -> bool { + if items.len() < 2 { + return false; + } + let actions = items.iter().filter(|item| starts_with_action(item)).count(); + actions >= 2 || (items.len() >= 3 && intro_is_instruction(intro)) +} + +/// Does the line that introduces a list tell the model to do something? +fn intro_is_instruction(intro: &str) -> bool { + let line = intro.trim_end().rsplit('\n').next().unwrap_or_default(); + action_clause_count(line) >= 1 +} + +/// The longest `1, 2, 3, …` run of numbered markers, as the text before it and +/// the items it numbers. +fn numbered_list(prose: &str) -> Option<(&str, Vec<&str>)> { + // (marker start, item start) for each marker in the run being built. + let mut best: Vec<(usize, usize)> = Vec::new(); + let mut run: Vec<(usize, usize)> = Vec::new(); + for captures in NUMBERED_MARKER.captures_iter(prose) { + let (Some(whole), Some(number)) = (captures.get(0), captures.get(1)) else { + continue; + }; + let Ok(number) = number.as_str().parse::() else { + continue; + }; + if number == run.len() + 1 { + run.push((whole.start(), whole.end())); + } else if number == 1 { + if run.len() > best.len() { + best = std::mem::take(&mut run); + } else { + run.clear(); + } + run.push((whole.start(), whole.end())); + } + } + if run.len() > best.len() { + best = run; + } + if best.len() < 2 { + return None; + } + let items = best + .iter() + .enumerate() + .map(|(index, &(_, item_start))| { + let end = match best.get(index + 1) { + Some(&(next_marker, _)) => next_marker, + // The last item ends with its line: an inline list has one line, + // and prose after a line list is not part of its last step. + None => prose[item_start..] + .find('\n') + .map_or(prose.len(), |offset| item_start + offset), + }; + prose[item_start..end].trim() + }) + .collect(); + Some((&prose[..best[0].0], items)) +} + +fn bulleted_list(prose: &str) -> Option<(&str, Vec<&str>)> { + let mut first_start = None; + let mut items = Vec::new(); + for captures in BULLET.captures_iter(prose) { + let (Some(whole), Some(item)) = (captures.get(0), captures.get(1)) else { + continue; + }; + first_start.get_or_insert(whole.start()); + items.push(item.as_str().trim()); + } + let start = first_start?; + (items.len() >= 2).then(|| (&prose[..start], items)) +} + +fn action_clause_count(prose: &str) -> usize { + CLAUSE_BREAK + .split(prose) + .filter(|clause| starts_with_action(clause)) + .count() +} + +fn starts_with_action(clause: &str) -> bool { + let words = words(clause); + strip_fillers(&words) + .first() + .is_some_and(|word| ACTION_VERBS.contains(word.as_str())) +} + +/// Lower-cased words, keeping an inner apostrophe ("let's", "i'd"). +fn words(text: &str) -> Vec { + text.split(|c: char| !(c.is_alphanumeric() || c == '\'' || c == '’')) + .map(|word| { + word.trim_matches(|c| c == '\'' || c == '’') + .replace('’', "'") + .to_lowercase() + }) + .filter(|word| !word.is_empty()) + .collect() +} + +fn strip_fillers(mut words: &[String]) -> &[String] { + loop { + // A list number or a stray count ("2 files") is not the verb. + if words + .first() + .is_some_and(|word| word.chars().all(|c| c.is_ascii_digit())) + { + words = &words[1..]; + continue; + } + let Some(filler) = FILLERS.iter().find(|filler| { + words.len() >= filler.len() && filler.iter().zip(words).all(|(a, b)| *a == b) + }) else { + return words; + }; + words = &words[filler.len()..]; + } +} + +// --------------------------------------------------------------------------- +// Scope +// --------------------------------------------------------------------------- + +/// Does the gate run for this conversation? The ONE answer, read by the turn +/// (`Agent::begin_planning_turn`) and by the system prompt +/// (`prepare_tools_and_prompt_for_provider`), so the prompt describes exactly +/// what the turn enforces. +/// +/// * Chat mode runs no tools, so there is no tool call to redirect. +/// * A subagent's turn ends with an observe-only `SubagentStop`, never a +/// blockable Stop, so the stop check could not run there anyway; its task +/// was written by a parent model that keeps its own checklist. +/// * A coding-agent provider's tool calls run through the tool bridge, which +/// inspects them on its own path and never reaches the reply loop's gate. +/// * No `todo__todo_write` on the roster means nothing the gate could point at. +pub(crate) fn enforcement_applies<'a>( + mode: BioRouterMode, + is_subagent: bool, + bridge_surface: bool, + tool_names: impl IntoIterator, +) -> bool { + mode != BioRouterMode::Chat + && !is_subagent + && !bridge_surface + && tool_names + .into_iter() + .any(|name| name == TODO_WRITE_TOOL_NAME) +} + +// --------------------------------------------------------------------------- +// Turn state +// --------------------------------------------------------------------------- + +/// One turn's gate state. +#[derive(Debug, Clone)] +struct TurnPlan { + /// Why this turn's prompt reads as several steps, when it does. + signal: Option, + /// The checklist had no items when the turn opened. + list_was_empty: bool, + /// [`fingerprint`] of the checklist when the turn opened. + start_fingerprint: u64, + /// The once-per-turn redirect has fired. + redirect_spent: bool, + /// The checklist has been seen with items since the turn opened. + list_seen: bool, + /// Checklist stop blocks this turn. Never reset by a tool call. + stop_blocks: u32, + /// Insertion order, for the bound. + serial: u64, +} + +impl TurnPlan { + fn armed(&self) -> Option { + (self.list_was_empty && !self.list_seen && !self.redirect_spent) + .then_some(self.signal) + .flatten() + } +} + +#[derive(Debug, Default)] +struct Turns { + plans: HashMap, + serial: u64, +} + +/// Per-session turn state for one [`Agent`]. +/// +/// ⚠ Owned by the agent, never process-global. Session ids are `YYYYMMDD_N` +/// per DATABASE, so two tests with their own stores routinely share one, and a +/// global keyed by it would let one test's turn arm another test's gate. +#[derive(Debug, Default)] +pub(crate) struct PlanningRegistry { + turns: std::sync::Mutex, +} + +impl PlanningRegistry { + fn lock(&self) -> std::sync::MutexGuard<'_, Turns> { + self.turns.lock().unwrap_or_else(PoisonError::into_inner) + } + + /// Replace the session's state with `plan`, or clear it when the gate does + /// not run this turn. + fn begin(&self, session_id: &str, plan: Option) { + let mut turns = self.lock(); + let Some(mut plan) = plan else { + turns.plans.remove(session_id); + return; + }; + turns.serial += 1; + plan.serial = turns.serial; + turns.plans.insert(session_id.to_string(), plan); + while turns.plans.len() > MAX_TRACKED_SESSIONS { + let Some(oldest) = turns + .plans + .iter() + .min_by_key(|(_, plan)| plan.serial) + .map(|(id, _)| id.clone()) + else { + break; + }; + turns.plans.remove(&oldest); + } + } + + /// The turn's signal while the reminder and the redirect are live: a + /// multi-step prompt, a list that was empty and has not been seen since, + /// and the redirect not yet spent. + fn armed(&self, session_id: &str) -> Option { + self.lock().plans.get(session_id).and_then(TurnPlan::armed) + } + + fn note_list_exists(&self, session_id: &str) { + if let Some(plan) = self.lock().plans.get_mut(session_id) { + plan.list_seen = true; + } + } + + /// Spend the turn's one redirect. `None` when it is not armed — including + /// when it already fired, which is the once-per-turn bound. + fn spend_redirect(&self, session_id: &str) -> Option { + let mut turns = self.lock(); + let plan = turns.plans.get_mut(session_id)?; + let signal = plan.armed()?; + plan.redirect_spent = true; + Some(signal) + } + + /// `(start fingerprint, stop blocks so far)`, when the gate runs. + fn stop_state(&self, session_id: &str) -> Option<(u64, u32)> { + self.lock() + .plans + .get(session_id) + .map(|plan| (plan.start_fingerprint, plan.stop_blocks)) + } + + fn record_stop_block(&self, session_id: &str) { + if let Some(plan) = self.lock().plans.get_mut(session_id) { + plan.stop_blocks += 1; + } + } +} + +/// A stable fingerprint of a checklist — plan, ids, statuses and texts — so the +/// stop check can tell "this turn worked the list" from "the list is left over +/// from an earlier turn". An absent list and an empty one read the same. +fn fingerprint(state: Option<&TodoState>) -> u64 { + let mut hasher = std::collections::hash_map::DefaultHasher::new(); + state + .map(TodoState::render) + .unwrap_or_default() + .hash(&mut hasher); + hasher.finish() +} + +// --------------------------------------------------------------------------- +// Texts +// --------------------------------------------------------------------------- + +/// The MOIM line a multi-step turn carries while its checklist is empty. +fn reminder_text(signal: MultiStep) -> String { + format!( + "Planning required: this request has {}, and your checklist is empty. Before you do \ + anything else, call `{TODO_WRITE_TOOL_NAME}` with one `- [ ]` item per step. Until \ + the checklist exists, the first other tool call you make this turn will be refused.", + signal.describe() + ) +} + +/// What a redirected call returns to the model. +fn redirect_reason(signal: MultiStep) -> String { + format!( + "Not run: this request has {}, and the checklist is still empty, so Biorouter needs \ + the plan first. Call `{TODO_WRITE_TOOL_NAME}` with one `- [ ]` item per step, then \ + make this call again. This refusal happens once per turn: if you have a reason not \ + to keep a checklist for this request, say so in your reply and repeat the call — it \ + will run.", + signal.describe() + ) +} + +fn status_label(status: TodoStatus) -> &'static str { + match status { + TodoStatus::Pending => "not started", + TodoStatus::InProgress => "in progress", + TodoStatus::Blocked => "blocked", + TodoStatus::Completed => "completed", + } +} + +fn id_list<'a>(items: impl IntoIterator) -> String { + const SHOWN: usize = 6; + let ids: Vec = items + .into_iter() + .map(|item| format!("#{}", item.id)) + .collect(); + if ids.len() > SHOWN { + format!("{}, …", ids[..SHOWN].join(", ")) + } else { + ids.join(", ") + } +} + +/// The notice for a turn that ends with the checklist open because the stop +/// check has spent its budget. +fn give_up_notice(open: usize) -> String { + format!( + "📋 The checklist still has {open} unfinished item(s) after {STOP_HOOK_BLOCK_CAP} \ + reminders; finishing anyway." + ) +} + +// --------------------------------------------------------------------------- +// The redirect +// --------------------------------------------------------------------------- + +/// Which calls in one batch the redirect refuses: every call except the Todo +/// tools and the workflow's structured-output tool — and none at all when the +/// batch itself seeds the checklist, because the plan then lands in the same +/// step as the work. A malformed call is left alone; it fails on its own. +fn redirect_targets(requests: &[ToolRequest]) -> Vec { + let calls: Vec<(&str, &str)> = requests + .iter() + .filter_map(|request| { + let call = request.tool_call.as_ref().ok()?; + Some((request.id.as_str(), call.name.as_ref())) + }) + .collect(); + if calls + .iter() + .any(|(_, name)| CHECKLIST_SEEDING_TOOLS.contains(name)) + { + return Vec::new(); + } + calls + .into_iter() + .filter(|(_, name)| !is_todo_tool_name(name) && *name != FINAL_OUTPUT_TOOL_NAME) + .map(|(id, _)| id.to_string()) + .collect() +} + +// --------------------------------------------------------------------------- +// The stop check +// --------------------------------------------------------------------------- + +/// What the stop check found wrong with ending the turn now. +#[derive(Debug, Clone, PartialEq, Eq)] +struct ChecklistObjection { + /// For the model: every unfinished item, and the two ways out. + feedback: String, + /// For the user. + notice: String, + /// How many items are unfinished. + open: usize, +} + +/// The turn may not end while `state` has unfinished items, unless +/// `final_text` names every one of them — by `#N` id or by its text. +/// +/// The feedback asks for a reason too, but only the naming is checked: a +/// deterministic check cannot tell a reason from a status recap, and a named +/// item is one the user can see was left open, which is the point. +fn checklist_objection(state: &TodoState, final_text: &str) -> Option { + let open: Vec<&TodoItem> = state + .items + .iter() + .filter(|item| item.status != TodoStatus::Completed) + .collect(); + if open.is_empty() { + return None; + } + let final_words = words(final_text); + if open + .iter() + .all(|item| names_item(final_text, &final_words, item)) + { + return None; + } + let listed = open + .iter() + .map(|item| { + format!( + "- #{} ({}) {}", + item.id, + status_label(item.status), + item.text + ) + }) + .collect::>() + .join("\n"); + Some(ChecklistObjection { + feedback: format!( + "Before you finish: your checklist still has {} unfinished item(s):\n{listed}\n\ + Finish them, marking each one completed with `todo__todo_update` as you go. If an \ + item cannot or should not be done now, end your turn with a message that names \ + each unfinished item by its #N id and says why it is not done.", + open.len() + ), + notice: format!( + "📋 The checklist still has {} unfinished item(s) ({}); asking the agent to finish \ + them or say why not.", + open.len(), + id_list(open.iter().copied()) + ), + open: open.len(), + }) +} + +fn names_item(final_text: &str, final_words: &[String], item: &TodoItem) -> bool { + mentions_id(final_text, &item.id) || contains_words(final_words, &words(&item.text)) +} + +/// `#3` names item 3; `#30` does not. +fn mentions_id(text: &str, id: &str) -> bool { + let needle = format!("#{id}"); + text.match_indices(&needle).any(|(at, _)| { + !text[at + needle.len()..] + .chars() + .next() + .is_some_and(|c| c.is_ascii_digit()) + }) +} + +fn contains_words(haystack: &[String], needle: &[String]) -> bool { + !needle.is_empty() + && haystack + .windows(needle.len()) + .any(|window| window == needle) +} + +/// The text of the turn's final answer: every assistant message after the +/// last user-role message. A tool result and a steer are user-role here, so +/// this is exactly what the model said since it last heard anything. +fn final_reply_text(conversation: &Conversation) -> String { + let messages = conversation.messages(); + let start = messages + .iter() + .rposition(|message| message.role == Role::User) + .map_or(0, |index| index + 1); + messages[start..] + .iter() + .filter(|message| message.role == Role::Assistant) + .map(|message| message.as_concat_text()) + .collect::>() + .join("\n") +} + +/// The checklist half of a turn's stop decision. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(super) enum ChecklistStop { + /// Nothing to say: the gate does not run, the turn did not work the list, + /// or every item is completed or named. + Clear, + /// Keep working: `feedback` to the model, `notice` to the user. + Block { feedback: String, notice: String }, + /// Still open, but the per-turn budget is spent: finish, and say so. + GiveUp { notice: String }, +} + +// --------------------------------------------------------------------------- +// The agent's side +// --------------------------------------------------------------------------- + +impl Agent { + /// Open this turn's gate state, once per reply, before the loop starts. + /// + /// `session` is the row `reply` read at the top of the turn, so its + /// `extension_data` is the checklist as the turn found it. + pub(super) fn begin_planning_turn( + &self, + session: &Session, + signal: Option, + tools: &[Tool], + toolshim_tools: &[Tool], + bridge_surface: bool, + ) { + let enforced = enforcement_applies( + self.config.biorouter_mode, + session.session_type == SessionType::SubAgent, + bridge_surface, + // A toolshim turn hands the provider no tools and keeps them here. + tools + .iter() + .chain(toolshim_tools) + .map(|tool| tool.name.as_ref()), + ); + if !enforced { + self.planning.begin(&session.id, None); + return; + } + // An unreadable blob (a newer build wrote it) counts as a list that + // exists: the gate never nags about a checklist it cannot see. + let (list_was_empty, start_fingerprint) = match TodoState::try_load(&session.extension_data) + { + Ok(state) => ( + state.as_ref().is_none_or(|state| state.items.is_empty()), + fingerprint(state.as_ref()), + ), + Err(_) => (false, fingerprint(None)), + }; + self.planning.begin( + &session.id, + Some(TurnPlan { + signal, + list_was_empty, + start_fingerprint, + redirect_spent: false, + list_seen: false, + stop_blocks: 0, + serial: 0, + }), + ); + } + + /// The reminder for this provider call, while the turn is armed and the + /// checklist is still empty. + pub(super) async fn planning_reminder(&self, session_id: &str) -> Option { + let signal = self.planning.armed(session_id)?; + if self.checklist_exists(session_id).await { + self.planning.note_list_exists(session_id); + return None; + } + Some(reminder_text(signal)) + } + + /// The once-per-turn redirect, as inspection results the permission merge + /// turns into ordinary refusals. Empty unless the turn is armed, the batch + /// has a call to refuse, and the checklist is still empty. + pub(super) async fn planning_gate_denials( + &self, + session_id: &str, + requests: &[ToolRequest], + ) -> Vec { + if self.planning.armed(session_id).is_none() { + return Vec::new(); + } + let targets = redirect_targets(requests); + if targets.is_empty() { + return Vec::new(); + } + if self.checklist_exists(session_id).await { + self.planning.note_list_exists(session_id); + return Vec::new(); + } + let Some(signal) = self.planning.spend_redirect(session_id) else { + return Vec::new(); + }; + tracing::info!( + session_id, + refused = targets.len(), + "planning gate: redirected the turn's first tool batch to todo_write" + ); + let reason = redirect_reason(signal); + targets + .into_iter() + .map(|tool_request_id| InspectionResult { + tool_request_id, + action: InspectionAction::Deny, + reason: reason.clone(), + confidence: 1.0, + inspector_name: PLANNING_GATE_NAME.to_string(), + finding_id: None, + }) + .collect() + } + + /// Whether the turn may end with the checklist as it stands. + pub(super) async fn checklist_stop( + &self, + session_id: &str, + conversation: &Conversation, + ) -> ChecklistStop { + let Some((start_fingerprint, blocks)) = self.planning.stop_state(session_id) else { + return ChecklistStop::Clear; + }; + // Disabled mid-turn: no tool left to finish the list with. + if !self + .extension_manager + .is_extension_enabled(crate::agents::todo_extension::EXTENSION_NAME) + .await + { + return ChecklistStop::Clear; + } + // Fail open on a read error or a blob this build cannot parse: a check + // that cannot see the list does not keep a turn alive on its behalf. + let Ok(session) = self + .config + .session_manager + .get_session(session_id, false) + .await + else { + return ChecklistStop::Clear; + }; + let Ok(Some(state)) = TodoState::try_load(&session.extension_data) else { + return ChecklistStop::Clear; + }; + if fingerprint(Some(&state)) == start_fingerprint { + // Not this turn's list: left over from an earlier turn and untouched. + return ChecklistStop::Clear; + } + let Some(objection) = checklist_objection(&state, &final_reply_text(conversation)) else { + return ChecklistStop::Clear; + }; + if blocks >= STOP_HOOK_BLOCK_CAP { + return ChecklistStop::GiveUp { + notice: give_up_notice(objection.open), + }; + } + self.planning.record_stop_block(session_id); + ChecklistStop::Block { + feedback: objection.feedback, + notice: objection.notice, + } + } + + /// Does the session's checklist have items right now? `true` on any error + /// or an unreadable blob, so neither the reminder nor the redirect ever + /// fires on a list it could not read. + async fn checklist_exists(&self, session_id: &str) -> bool { + match self + .config + .session_manager + .get_session(session_id, false) + .await + { + Ok(session) => match TodoState::try_load(&session.extension_data) { + Ok(state) => state.is_some_and(|state| !state.items.is_empty()), + Err(_) => true, + }, + Err(_) => true, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::conversation::message::Message; + use rmcp::model::CallToolRequestParams; + + // -- classifier ------------------------------------------------------- + + #[test] + fn the_classifier_recognises_requests_that_have_several_steps() { + let cases: &[(&str, MultiStep)] = &[ + ( + "1. create a temp dir 2. write hello.txt into it 3. count its bytes 4. delete it", + MultiStep::List { items: 4 }, + ), + ( + "1. Create a temp dir\n2. Write hello.txt into it\n3. Count its bytes\n4. Delete it", + MultiStep::List { items: 4 }, + ), + ( + "Please do the following:\n- clone the repo\n- build it\n- run the tests", + MultiStep::List { items: 3 }, + ), + ( + "Step 1: fetch the data. Step 2: plot it.", + MultiStep::List { items: 2 }, + ), + ( + "Build a landing page with: 1. a header 2. a pricing table 3. a footer", + MultiStep::List { items: 3 }, + ), + ( + "Create a temp dir, write hello.txt into it, count its bytes and delete it.", + MultiStep::Instructions { count: 4 }, + ), + ( + "Download the CSV, clean the missing values, and plot the distribution.", + MultiStep::Instructions { count: 3 }, + ), + ( + "Create the directory, then write hello.txt into it.", + MultiStep::Sequenced { count: 2 }, + ), + ( + "First download the dataset. After that, normalize the columns.", + MultiStep::Sequenced { count: 2 }, + ), + ( + "Can you run the tests and then fix any failures?", + MultiStep::Sequenced { count: 2 }, + ), + ( + "Write a script that downloads the data, then plot the results", + MultiStep::Sequenced { count: 2 }, + ), + ]; + for (prompt, expected) in cases { + assert_eq!( + classify_request(prompt), + Some(*expected), + "should read as several steps: {prompt:?}" + ); + } + } + + #[test] + fn the_classifier_leaves_single_asks_questions_and_pasted_lists_alone() { + let cases = [ + "", + " ", + "What is the capital of France?", + "Fix the typo in README.md", + "Read README.md and tell me what this project does", + "Explain the steps of glycolysis", + "If the build fails then fix it", + "Python 3.12 is out. Should I upgrade?", + "Thanks, that worked!", + "Which is better? 1. Postgres 2. SQLite", + "Summarize this paper: 1. Introduction 2. Methods 3. Results", + // Two instructions and no word that sequences them. + "Create a new branch and commit these changes", + // One task with requirements, described in the third person. + "Please write a function that parses the file, validates each row, and saves it", + // A how-to question names steps it does not want performed. + "Can you tell me how to create a directory, write a file and then delete it in bash?", + "How do I: 1. create a venv 2. install the deps 3. run the tests?", + // Steps inside a code fence are pasted data. + "Why does this fail?\n```\n1. open the file\n2. write the header\n3. close it\n```", + ]; + for prompt in cases { + assert_eq!( + classify_request(prompt), + None, + "should not read as several steps: {prompt:?}" + ); + } + } + + // -- scope ------------------------------------------------------------ + + #[test] + fn the_gate_runs_only_where_it_can_do_what_it_says() { + let roster = ["developer__shell", TODO_WRITE_TOOL_NAME]; + assert!(enforcement_applies( + BioRouterMode::Auto, + false, + false, + roster + )); + assert!(enforcement_applies( + BioRouterMode::Approve, + false, + false, + roster + )); + // No tool runs in Chat mode, so there is nothing to redirect. + assert!(!enforcement_applies( + BioRouterMode::Chat, + false, + false, + roster + )); + assert!(!enforcement_applies( + BioRouterMode::Auto, + true, + false, + roster + )); + assert!(!enforcement_applies( + BioRouterMode::Auto, + false, + true, + roster + )); + // The capability is off, or its seeding tool is not granted. + assert!(!enforcement_applies( + BioRouterMode::Auto, + false, + false, + ["developer__shell", "todo__todo_update"] + )); + } + + // -- the redirect ----------------------------------------------------- + + fn request(id: &str, name: &str) -> ToolRequest { + ToolRequest { + id: id.to_string(), + tool_call: Ok(CallToolRequestParams { + task: None, + meta: None, + name: name.to_string().into(), + arguments: Some(serde_json::Map::new()), + }), + metadata: None, + tool_meta: None, + } + } + + #[test] + fn the_redirect_refuses_every_non_todo_call_unless_the_batch_seeds_the_list() { + assert_eq!( + redirect_targets(&[ + request("a", "developer__shell"), + request("b", "fixture__step") + ]), + vec!["a".to_string(), "b".to_string()] + ); + // The plan lands in the same step as the work: let the batch run. + for seeding in CHECKLIST_SEEDING_TOOLS { + assert!( + redirect_targets(&[request("a", seeding), request("b", "developer__shell")]) + .is_empty(), + "{seeding} seeds the checklist" + ); + } + // A plan with no checklist does not satisfy the gate, and is not refused. + assert_eq!( + redirect_targets(&[ + request("a", "todo__plan_write"), + request("b", "developer__shell") + ]), + vec!["b".to_string()] + ); + assert!(redirect_targets(&[request("a", FINAL_OUTPUT_TOOL_NAME)]).is_empty()); + assert!(redirect_targets(&[request("a", "todo__todo_update")]).is_empty()); + } + + // -- turn state ------------------------------------------------------- + + fn plan(signal: Option, list_was_empty: bool) -> TurnPlan { + TurnPlan { + signal, + list_was_empty, + start_fingerprint: fingerprint(None), + redirect_spent: false, + list_seen: false, + stop_blocks: 0, + serial: 0, + } + } + + #[test] + fn the_redirect_is_spent_once_and_the_list_disarms_the_gate() { + let registry = PlanningRegistry::default(); + let steps = Some(MultiStep::List { items: 4 }); + + registry.begin("s", Some(plan(steps, true))); + assert_eq!(registry.armed("s"), steps); + assert_eq!(registry.spend_redirect("s"), steps); + assert_eq!(registry.spend_redirect("s"), None, "once per turn"); + assert_eq!(registry.armed("s"), None, "no reminder after the refusal"); + + // A new turn re-arms it. + registry.begin("s", Some(plan(steps, true))); + assert_eq!(registry.armed("s"), steps); + registry.note_list_exists("s"); + assert_eq!(registry.armed("s"), None); + assert_eq!(registry.spend_redirect("s"), None); + + // A one-line turn, or a list that already existed, never arms. + registry.begin("s", Some(plan(None, true))); + assert_eq!(registry.armed("s"), None); + registry.begin("s", Some(plan(steps, false))); + assert_eq!(registry.armed("s"), None); + // …but the stop check still has its baseline. + assert!(registry.stop_state("s").is_some()); + + // A turn the gate does not run for leaves nothing behind. + registry.begin("s", None); + assert_eq!(registry.stop_state("s"), None); + } + + #[test] + fn the_registry_is_bounded_and_evicts_the_oldest_session() { + let registry = PlanningRegistry::default(); + for n in 0..(MAX_TRACKED_SESSIONS + 10) { + registry.begin(&format!("s{n}"), Some(plan(None, true))); + } + let turns = registry.lock(); + assert_eq!(turns.plans.len(), MAX_TRACKED_SESSIONS); + assert!(!turns.plans.contains_key("s0")); + assert!(turns + .plans + .contains_key(&format!("s{}", MAX_TRACKED_SESSIONS + 9))); + } + + // -- the stop check --------------------------------------------------- + + fn checklist(markdown: &str) -> TodoState { + let mut state = TodoState::default(); + state.set_from_markdown(markdown); + state + } + + #[test] + fn unfinished_items_block_the_stop_until_they_are_named() { + let state = checklist( + "- [x] create a temp dir\n- [x] write hello.txt into it\n\ + - [~] count its bytes\n- [ ] delete it", + ); + + let objection = checklist_objection(&state, "All done!").expect("two items are open"); + assert_eq!(objection.open, 2); + assert!(objection + .feedback + .contains("#3 (in progress) count its bytes")); + assert!(objection.feedback.contains("#4 (not started) delete it")); + assert!(objection.feedback.contains("todo__todo_update")); + assert!(objection.notice.contains("#3, #4"), "{}", objection.notice); + + // Naming only one of them is not enough. + assert!(checklist_objection(&state, "#3 is still running.").is_some()); + // Every one, by id… + assert!(checklist_objection( + &state, + "I stopped early: #3 needs the dir to exist and #4 would delete your data." + ) + .is_none()); + // …or by its text. + assert!(checklist_objection( + &state, + "I did not count its bytes, and I did not delete it: the disk is read-only." + ) + .is_none()); + // `#30` is not `#3`. + assert!(checklist_objection(&state, "#30 and #40 are open").is_some()); + } + + #[test] + fn a_finished_or_empty_checklist_never_blocks_and_blocked_items_count_as_open() { + assert!(checklist_objection(&checklist("- [x] one\n- [x] two"), "done").is_none()); + assert!(checklist_objection(&TodoState::default(), "done").is_none()); + let objection = checklist_objection(&checklist("- [x] one\n- [!] ask the user"), "done") + .expect("a blocked item is unfinished"); + assert!(objection.feedback.contains("#2 (blocked) ask the user")); + } + + #[test] + fn the_fingerprint_moves_with_any_change_and_not_otherwise() { + let before = checklist("- [ ] one\n- [ ] two"); + let mut after = before.clone(); + assert_eq!(fingerprint(Some(&before)), fingerprint(Some(&after))); + after.update_item("2", Some(TodoStatus::Completed), None); + assert_ne!(fingerprint(Some(&before)), fingerprint(Some(&after))); + assert_eq!(fingerprint(None), fingerprint(Some(&TodoState::default()))); + } + + #[test] + fn the_final_reply_is_what_the_model_said_after_it_last_heard_anything() { + let conversation = Conversation::new_unvalidated(vec![ + Message::user().with_text("do the thing"), + Message::assistant().with_text("I'll start with #1"), + Message::user().with_text("tool result"), + Message::assistant().with_text("Finished #1."), + ]); + assert_eq!(final_reply_text(&conversation), "Finished #1."); + } +} + +/// The gate driven through the real reply loop: a scripted provider, the real +/// Todo capability, and an in-process fixture tool standing in for work. +#[cfg(test)] +mod agent_loop_tests { + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::sync::{Arc, Mutex}; + + use futures::StreamExt; + use rmcp::handler::server::router::tool::ToolRouter; + use rmcp::handler::server::wrapper::Parameters; + use rmcp::model::{CallToolRequestParams, CallToolResult, Content, ServerCapabilities}; + use rmcp::{object, tool, tool_handler, tool_router}; + + use crate::agents::extension::ExtensionConfig; + use crate::agents::{Agent, AgentConfig, AgentEvent, SessionConfig}; + use crate::config::permission::PermissionManager; + use crate::config::BioRouterMode; + use crate::conversation::message::{Message, MessageContent}; + use crate::model::ModelConfig; + use crate::providers::base::{Provider, ProviderMetadata, ProviderUsage, Usage}; + use crate::providers::errors::ProviderError; + use crate::session::extension_data::{TodoState, TodoStatus}; + use crate::session::session_manager::SessionType; + use crate::session::SessionManager; + use rmcp::model::Tool; + + /// One scripted reply per provider call; records what each call was shown. + struct ScriptedProvider { + script: Vec, + calls: AtomicUsize, + seen: Mutex)>>, + } + + impl ScriptedProvider { + fn new(script: Vec) -> Arc { + Arc::new(Self { + script, + calls: AtomicUsize::new(0), + seen: Mutex::new(Vec::new()), + }) + } + + fn calls(&self) -> usize { + self.calls.load(Ordering::SeqCst) + } + + /// Every text the provider was shown on call `n` — the conversation, + /// MOIM included — plus the tool results, flattened. + fn shown(&self, n: usize) -> String { + let seen = self.seen.lock().unwrap(); + seen[n] + .1 + .iter() + .flat_map(|message| message.content.iter()) + .map(|content| match content { + MessageContent::ToolResponse(response) => response + .tool_result + .as_ref() + .map(|result| { + result + .content + .iter() + .filter_map(|c| c.as_text().map(|t| t.text.clone())) + .collect::>() + .join("\n") + }) + .unwrap_or_default(), + other => other.as_text().map(str::to_string).unwrap_or_default(), + }) + .collect::>() + .join("\n") + } + + fn system_prompt(&self, n: usize) -> String { + self.seen.lock().unwrap()[n].0.clone() + } + } + + #[async_trait::async_trait] + impl Provider for ScriptedProvider { + fn metadata() -> ProviderMetadata { + ProviderMetadata::empty() + } + + fn get_name(&self) -> &str { + "scripted" + } + + fn get_model_config(&self) -> ModelConfig { + ModelConfig::new_or_fail("scripted-model") + } + + async fn complete_with_model( + &self, + _model_config: &ModelConfig, + system: &str, + messages: &[Message], + _tools: &[Tool], + ) -> Result<(Message, ProviderUsage), ProviderError> { + let n = self.calls.fetch_add(1, Ordering::SeqCst); + self.seen + .lock() + .unwrap() + .push((system.to_string(), messages.to_vec())); + let reply = self + .script + .get(n) + .cloned() + .unwrap_or_else(|| Message::assistant().with_text("(script exhausted)")); + Ok(( + reply, + ProviderUsage::new( + "scripted-model".to_string(), + Usage::new(Some(10), Some(5), Some(15)), + ), + )) + } + } + + /// The work: one tool that counts how often it really ran. + #[derive(Clone)] + struct StepServer { + tool_router: ToolRouter, + runs: Arc, + } + + /// A step number, so a script that calls the tool many times does not trip + /// the repetition guard — a different gate, with its own tests. + #[derive(Debug, serde::Deserialize, schemars::JsonSchema)] + struct StepArgs { + #[serde(default)] + n: u32, + } + + #[tool_router(router = tool_router)] + impl StepServer { + fn new(runs: Arc) -> Self { + Self { + tool_router: Self::tool_router(), + runs, + } + } + + #[tool(description = "Do one step of the fixture task")] + fn step(&self, args: Parameters) -> Result { + self.runs.fetch_add(1, Ordering::SeqCst); + Ok(CallToolResult::success(vec![Content::text(format!( + "step {} done", + args.0.n + ))])) + } + } + + #[tool_handler(router = self.tool_router)] + impl rmcp::ServerHandler for StepServer { + fn get_info(&self) -> rmcp::model::ServerInfo { + rmcp::model::ServerInfo { + capabilities: ServerCapabilities::builder().enable_tools().build(), + ..Default::default() + } + } + } + + fn call(id: &str, name: &str, arguments: serde_json::Value) -> Message { + Message::assistant().with_tool_request( + id, + Ok(CallToolRequestParams { + task: None, + meta: None, + name: name.to_string().into(), + arguments: arguments.as_object().cloned(), + }), + ) + } + + struct Fixture { + agent: Agent, + session_id: String, + runs: Arc, + _dirs: Vec, + } + + async fn fixture(todo: bool, provider: Arc) -> Fixture { + let data = tempfile::tempdir().unwrap(); + let work = tempfile::tempdir().unwrap(); + let permissions = tempfile::tempdir().unwrap(); + let session_manager = Arc::new(SessionManager::new(data.path().to_path_buf())); + let agent = Agent::with_config( + AgentConfig::new( + Arc::clone(&session_manager), + Arc::new(PermissionManager::new(permissions.path().to_path_buf())), + None, + BioRouterMode::Auto, + ) + .with_project_hooks(false), + ); + if todo { + agent + .add_extension(ExtensionConfig::Platform { + name: "todo".into(), + description: "todo".into(), + bundled: Some(true), + available_tools: vec![], + }) + .await + .expect("enable the Todo capability"); + } + let runs = Arc::new(AtomicUsize::new(0)); + agent + .extension_manager + .add_inprocess_server("fixture", StepServer::new(Arc::clone(&runs))) + .await + .expect("inject the fixture tool"); + let session = session_manager + .create_session( + work.path().to_path_buf(), + "planning gate".to_string(), + SessionType::User, + ) + .await + .unwrap(); + agent + .update_provider(provider, &session.id) + .await + .expect("bind the scripted provider"); + Fixture { + agent, + session_id: session.id, + runs, + _dirs: vec![data, work, permissions], + } + } + + /// Drive one user turn; return every message the stream yielded. + async fn turn(fixture: &Fixture, prompt: &str) -> Vec { + let stream = fixture + .agent + .reply( + Message::user().with_text(prompt), + SessionConfig { + id: fixture.session_id.clone(), + schedule_id: None, + max_turns: Some(20), + max_tool_calls: None, + budget: None, + retry_config: None, + reasoning_effort: None, + }, + None, + ) + .await + .expect("the turn starts"); + tokio::pin!(stream); + let mut yielded = Vec::new(); + while let Some(event) = stream.next().await { + if let AgentEvent::Message(message) = event.expect("the turn runs") { + yielded.push(message); + } + } + yielded + } + + fn notices(messages: &[Message]) -> Vec { + messages + .iter() + .flat_map(|message| message.content.iter()) + .filter_map(|content| match content { + MessageContent::SystemNotification(notice) => Some(notice.msg.clone()), + _ => None, + }) + .collect() + } + + async fn checklist(fixture: &Fixture) -> TodoState { + let session = fixture + .agent + .config + .session_manager + .get_session(&fixture.session_id, false) + .await + .unwrap(); + TodoState::load(&session.extension_data).unwrap_or_default() + } + + const FOUR_STEPS: &str = + "1. create a temp dir 2. write hello.txt into it 3. count its bytes 4. delete it"; + + #[tokio::test] + async fn a_multi_step_turn_is_planned_redirected_worked_and_finished() { + let provider = ScriptedProvider::new(vec![ + // 0: straight to work — refused once, pointed at the checklist. + call("work-1", "fixture__step", serde_json::json!({"n": 1})), + // 1: the plan. + call( + "plan", + "todo__todo_write", + serde_json::json!({"content": "- [ ] make the dir\n- [ ] write the file"}), + ), + // 2: the same work, which now runs. + call("work-2", "fixture__step", serde_json::json!({"n": 1})), + // 3: stops with the list open and nothing named — sent back. + Message::assistant().with_text("All finished."), + // 4: ticks the items, in one batch. + call( + "tick-1", + "todo__todo_update", + serde_json::json!({"id": "1", "status": "completed"}), + ) + .with_tool_request( + "tick-2", + Ok(CallToolRequestParams { + task: None, + meta: None, + name: "todo__todo_update".into(), + arguments: Some(object!({"id": "2", "status": "completed"})), + }), + ), + // 5: stops with every item completed. + Message::assistant().with_text("Done: made the dir and wrote the file."), + ]); + let fixture = fixture(true, Arc::clone(&provider)).await; + + let yielded = turn(&fixture, FOUR_STEPS).await; + + assert_eq!( + provider.calls(), + 6, + "exactly the scripted turn, no more and no less" + ); + // The prompt states what the gate enforces, because the gate runs. + assert!(provider + .system_prompt(0) + .contains("Biorouter enforces the checklist")); + // 1. The reminder rode the first call's context… + assert!( + provider.shown(0).contains("Planning required"), + "{}", + provider.shown(0) + ); + // 2. …the first work call was refused with a pointer to todo_write… + let after_redirect = provider.shown(1); + assert!( + after_redirect.contains("Not run: this request has a list of 4 steps"), + "{after_redirect}" + ); + assert!(after_redirect.contains("todo__todo_write")); + // …and never ran; the repeat after the plan did, exactly once. + assert_eq!(fixture.runs.load(Ordering::SeqCst), 1); + // The reminder is gone once the list exists. + assert!(!provider.shown(2).contains("Planning required")); + // 3. The early stop was sent back with the open items named. + let after_block = provider.shown(4); + assert!( + after_block.contains("your checklist still has 2 unfinished item(s)"), + "{after_block}" + ); + assert!(after_block.contains("#1 (not started) make the dir")); + let shown_to_user = notices(&yielded); + assert!( + shown_to_user + .iter() + .any(|notice| notice.contains("📋") && notice.contains("#1, #2")), + "{shown_to_user:?}" + ); + // The list the chat summary reads: both items, both ticked. + let state = checklist(&fixture).await; + assert_eq!(state.items.len(), 2); + assert!(state + .items + .iter() + .all(|item| item.status == TodoStatus::Completed)); + } + + #[tokio::test] + async fn a_one_line_turn_gets_no_reminder_no_redirect_and_no_list() { + let provider = ScriptedProvider::new(vec![ + call("work", "fixture__step", serde_json::json!({"n": 1})), + Message::assistant().with_text("Done."), + ]); + let fixture = fixture(true, Arc::clone(&provider)).await; + + turn(&fixture, "Run the fixture step.").await; + + assert_eq!(provider.calls(), 2); + assert!(!provider.shown(0).contains("Planning required")); + assert!(!provider.shown(1).contains("Not run:")); + assert_eq!(fixture.runs.load(Ordering::SeqCst), 1, "the call ran"); + assert!(checklist(&fixture).await.items.is_empty()); + } + + #[tokio::test] + async fn with_the_todo_capability_off_the_plain_behaviour_returns() { + let provider = ScriptedProvider::new(vec![ + call("work", "fixture__step", serde_json::json!({"n": 1})), + Message::assistant().with_text("All finished."), + ]); + let fixture = fixture(false, Arc::clone(&provider)).await; + + turn(&fixture, FOUR_STEPS).await; + + assert_eq!(provider.calls(), 2); + assert!(!provider.shown(0).contains("Planning required")); + assert!(!provider + .system_prompt(0) + .contains("Biorouter enforces the checklist")); + assert_eq!(fixture.runs.load(Ordering::SeqCst), 1, "not redirected"); + } + + /// The stop check's two ways out: naming the open items ends the turn at + /// once, and a model that never does is let go after the cap — which does + /// not reset between blocks, because every block here is followed by tool + /// calls that would reset the Stop-hook counter. + #[tokio::test] + async fn the_stop_check_accepts_named_items_and_gives_up_at_the_cap() { + let seed = call( + "plan", + "todo__todo_write", + serde_json::json!({"content": "- [ ] one\n- [ ] two"}), + ); + + // Named: the open items are explained, so the first stop ends the turn. + let provider = ScriptedProvider::new(vec![ + seed.clone(), + Message::assistant().with_text("I stopped: #1 and #2 need your credentials."), + ]); + let named = fixture(true, Arc::clone(&provider)).await; + let yielded = turn(&named, FOUR_STEPS).await; + assert_eq!(provider.calls(), 2); + assert!(notices(&yielded) + .iter() + .all(|notice| !notice.contains("📋"))); + + // Stubborn: works, stops unnamed, works, stops unnamed, … + let mut script = vec![seed]; + for n in 0..(crate::hooks::STOP_HOOK_BLOCK_CAP + 1) { + script.push(call( + &format!("work-{n}"), + "fixture__step", + serde_json::json!({"n": n}), + )); + script.push(Message::assistant().with_text("All finished.")); + } + let provider = ScriptedProvider::new(script); + let stubborn = fixture(true, Arc::clone(&provider)).await; + let yielded = turn(&stubborn, FOUR_STEPS).await; + let cap = crate::hooks::STOP_HOOK_BLOCK_CAP as usize; + // The seed, then (work, stop) for each block, then the last (work, + // stop) that the cap lets through. + assert_eq!(provider.calls(), 1 + 2 * (cap + 1)); + let shown = notices(&yielded); + assert_eq!( + shown + .iter() + .filter(|notice| notice.contains("asking the agent to finish")) + .count(), + cap, + "{shown:?}" + ); + assert!( + shown + .iter() + .any(|notice| notice.contains("finishing anyway")), + "{shown:?}" + ); + } +} diff --git a/crates/biorouter/src/agents/prompt_manager.rs b/crates/biorouter/src/agents/prompt_manager.rs index c7add748a..0a35680ed 100644 --- a/crates/biorouter/src/agents/prompt_manager.rs +++ b/crates/biorouter/src/agents/prompt_manager.rs @@ -125,6 +125,9 @@ struct SystemPromptContext { is_autonomous: bool, enable_subagents: bool, code_execution_mode: bool, + /// The planning gate runs for this conversation (`agents::planning_gate`): + /// the "Working on Tasks" section then states exactly what it enforces. + checklist_enforcement: bool, } pub struct SystemPromptBuilder<'a, M> { @@ -135,6 +138,7 @@ pub struct SystemPromptBuilder<'a, M> { subagents_enabled: bool, hints: Option, code_execution_mode: bool, + checklist_enforcement: bool, variant: PromptVariant, } @@ -167,6 +171,14 @@ impl<'a> SystemPromptBuilder<'a, PromptManager> { self } + /// Whether the planning gate enforces the checklist for this turn. Pass + /// `planning_gate::enforcement_applies`, never a value of your own: the + /// clause this renders is a description of that gate. + pub fn with_checklist_enforcement(mut self, enabled: bool) -> Self { + self.checklist_enforcement = enabled; + self + } + pub fn with_hints(mut self, working_dir: &Path) -> Self { let config = Config::global(); let hints_filenames = config @@ -207,19 +219,21 @@ impl<'a> SystemPromptBuilder<'a, PromptManager> { subagents_enabled, hints, code_execution_mode, + checklist_enforcement, variant, } = self; let (extensions_info, hints) = prepare_injected_context(extensions_info, frontend_instructions, hints); let config = Config::global(); let biorouter_mode = config.get_biorouter_mode().unwrap_or(BioRouterMode::Auto); - let context = build_system_prompt_context( + let mut context = build_system_prompt_context( manager, extensions_info, biorouter_mode, subagents_enabled, code_execution_mode, ); + context.checklist_enforcement = checklist_enforcement; let base_prompt = render_base_prompt(manager, variant, &context); append_system_prompt_extras(manager, base_prompt, hints, biorouter_mode) } @@ -352,6 +366,7 @@ fn build_system_prompt_context( is_autonomous: biorouter_mode == BioRouterMode::Auto, enable_subagents: subagents_enabled, code_execution_mode, + checklist_enforcement: false, } } @@ -481,6 +496,7 @@ impl PromptManager { subagents_enabled: false, hints: None, code_execution_mode: false, + checklist_enforcement: false, variant: PromptVariant::Default, } } @@ -924,6 +940,26 @@ mod tests { assert!(resources_only.contains("Extension Manager operations are not available")); } + /// The checklist clause describes the planning gate, so it renders exactly + /// when the gate runs (`planning_gate::enforcement_applies`) and never + /// otherwise — a prompt promising a refusal the turn does not make is the + /// defect this flag exists to prevent. + #[test] + fn the_checklist_clause_renders_only_while_the_planning_gate_runs() { + let manager = PromptManager::with_timestamp(DateTime::::from_timestamp(0, 0).unwrap()); + + let off = manager.builder().build(); + assert!(!off.contains("Biorouter enforces the checklist"), "{off}"); + + let on = manager.builder().with_checklist_enforcement(true).build(); + assert!(on.contains("Biorouter enforces the checklist"), "{on}"); + assert!(on.contains("your first action is `todo__todo_write`")); + assert!(on.contains("That refusal happens once per turn")); + assert!(on.contains("names each unfinished item by its `#N` id")); + // It extends the planning bullet rather than replacing it. + assert!(on.contains("plan before acting")); + } + /// Contract test for the agentic-behavior clauses added to `system.md`. /// Each assertion guards one intentional instruction against silent /// removal/regression. These are the table-stakes behaviors the prompt diff --git a/crates/biorouter/src/agents/reply_parts.rs b/crates/biorouter/src/agents/reply_parts.rs index 4a5dfe798..9c9c2270e 100644 --- a/crates/biorouter/src/agents/reply_parts.rs +++ b/crates/biorouter/src/agents/reply_parts.rs @@ -255,6 +255,16 @@ fn coerce_tool_arguments( /// `every_tool_absent_from_the_code_execution_catalogue_stays_directly_callable` /// below. /// +/// The five Todo tools are the second superset exemption: in the catalogue AND +/// kept, for a reason about models rather than plumbing. A planning tool that +/// is reachable only by writing JavaScript inside `execute_code` is one a model +/// does not reach for — measured in the 2026-09-10 composer QA run, where the +/// collapsed roster had 18 tools, no `todo__*`, and the model built a checklist +/// exactly once, when told to. They are cheap structural calls, the same class +/// as the platform tools, and the planning gate (`agents::planning_gate`) +/// points a multi-step turn at `todo__todo_write` by name, which only works if +/// the model can call it. +/// /// Both name forms of the spawn tool are kept — models strip prefixes. /// /// [`ExtensionManager::get_prefixed_tools_excluding`]: crate::agents::ExtensionManager::get_prefixed_tools_excluding @@ -297,6 +307,9 @@ pub(crate) fn survives_code_execution_filter( // roster" — which is the state that reaches nowhere, and the state this // tool shipped in for exactly one live run. || crate::security::knowledge_delete::is_knowledge_delete_tool(tool_name) + // The checklist: exact names, never a `todo__` prefix — see + // `TODO_TOOL_NAMES` for why the prefix is not the builtin's to claim. + || crate::agents::todo_extension::is_todo_tool_name(tool_name) } fn code_execution_mode_is_active(loaded: bool, tools: &[Tool]) -> bool { @@ -467,6 +480,26 @@ impl Agent { &model_config.model_name, ); + let is_subagent = matches!( + self.config + .session_manager + .get_session(session_id, false) + .await + .ok() + .map(|session| session.session_type), + Some(SessionType::SubAgent) + ); + // The same predicate, over the same roster, that decides whether the + // turn enforces anything (`Agent::begin_planning_turn`) — so the prompt + // can never describe a gate this turn does not run. Read before the + // toolshim branch below empties `tools`. + let checklist_enforcement = crate::agents::planning_gate::enforcement_applies( + self.config.biorouter_mode, + is_subagent, + bridge_replaces_tool_surface, + tools.iter().map(|tool| tool.name.as_ref()), + ); + let prompt_manager = self.prompt_manager.lock().await; let enable_subagents = match active_bridge_plan { Some(plan) => plan.delegation_available, @@ -477,20 +510,12 @@ impl Agent { .with_extensions(extensions_info.into_iter()) .with_frontend_instructions(self.frontend_instructions.lock().await.clone()) .with_code_execution_mode(code_execution_active) + .with_checklist_enforcement(checklist_enforcement) .with_hints(working_dir) .with_enable_subagents(enable_subagents) .with_prompt_variant(prompt_variant) .build(); - let is_subagent = matches!( - self.config - .session_manager - .get_session(session_id, false) - .await - .ok() - .map(|session| session.session_type), - Some(SessionType::SubAgent) - ); if is_subagent { system_prompt.push_str("\n\n"); system_prompt.push_str(SUBAGENT_STEERING_INSTRUCTIONS); @@ -1279,6 +1304,29 @@ mod tests { )); } + /// The checklist stays a direct call in Code Execution mode. Before this, + /// `todo__*` collapsed into the JS catalogue with everything else and a + /// model reached it only by scripting — which it did once, when told to. + #[test] + fn the_code_execution_filter_keeps_every_todo_tool_directly_callable() { + let prefix = format!("{CODE_EXECUTION_EXTENSION}__"); + let no_frontend = HashSet::new(); + for name in crate::agents::todo_extension::TODO_TOOL_NAMES { + assert!( + survives_code_execution_filter(name, &prefix, &no_frontend), + "{name} is a cheap structural call the planning gate points at by name, so \ + Code Execution mode must keep it directly callable" + ); + } + // Exact names, not a prefix: a third-party server keyed `todo` gets no + // exemption from sharing the builtin's key. + assert!(!survives_code_execution_filter( + "todo__some_other_tool", + &prefix, + &no_frontend + )); + } + /// The CLASS guard, and the thing that was missing while issue #141 was /// closed for exactly one of the families it applies to. /// diff --git a/crates/biorouter/src/agents/todo_extension.rs b/crates/biorouter/src/agents/todo_extension.rs index 79bacd606..0d6e7cb37 100644 --- a/crates/biorouter/src/agents/todo_extension.rs +++ b/crates/biorouter/src/agents/todo_extension.rs @@ -15,6 +15,30 @@ use tokio_util::sync::CancellationToken; pub static EXTENSION_NAME: &str = "todo"; +/// The Todo tools as the model calls them: the extension key, `__`, the tool. +/// +/// Spelled out rather than matched by a `todo__` prefix, because the prefix is +/// not this builtin's to own: a user who disables the capability and installs +/// an MCP server keyed `todo` would otherwise inherit every exemption these +/// names carry. `the_todo_tool_names_are_the_tools_this_extension_lists` pins +/// the list to [`TodoClient::get_tools`], so a sixth tool cannot ship unnamed. +pub const TODO_TOOL_NAMES: [&str; 5] = [ + "todo__todo_write", + "todo__todo_add", + "todo__todo_expand", + "todo__todo_update", + "todo__plan_write", +]; + +/// The tool that seeds a checklist, and the one the planning gate points a +/// multi-step turn at (`agents::planning_gate`). +pub const TODO_WRITE_TOOL_NAME: &str = "todo__todo_write"; + +/// Is `tool_name` one of this capability's tools, as the model calls it? +pub fn is_todo_tool_name(tool_name: &str) -> bool { + TODO_TOOL_NAMES.contains(&tool_name) +} + /// Default cap on the number of items a checklist may hold (per session). const DEFAULT_MAX_ITEMS: usize = 200; @@ -726,9 +750,12 @@ impl McpClientTrait for TodoClient { .await .ok()?; - // Only the live plan/task state belongs here; the behavioral rule (plan - // up front, keep a todo list) lives in system.md so it holds even - // without this extension. See BR-4 / BR-60. + // Only the live plan/task state belongs here. The behavioural rule + // lives in two places, neither of them this extension: system.md + // states it, so it holds even without this extension (BR-4 / BR-60), + // and the planning gate (`agents::planning_gate`) enforces it for a + // multi-step turn — adding its own "write the checklist first" line to + // the same MOIM block while this one has nothing to render. let state = extension_data::TodoState::load(&metadata.extension_data)?; if state.is_empty() { return None; @@ -778,6 +805,28 @@ mod tests { .join("\n") } + /// `TODO_TOOL_NAMES` carries exemptions (the Code Execution filter, the + /// planning gate), so it must be exactly this extension's tools under the + /// key the manager gives them — no more, no fewer. + #[test] + fn the_todo_tool_names_are_the_tools_this_extension_lists() { + let key = crate::config::extensions::name_to_key(EXTENSION_NAME); + let mut listed: Vec = TodoClient::get_tools() + .iter() + .map(|tool| format!("{key}__{}", tool.name)) + .collect(); + listed.sort(); + let mut named: Vec = TODO_TOOL_NAMES.iter().map(|n| n.to_string()).collect(); + named.sort(); + assert_eq!(listed, named); + assert!(is_todo_tool_name(TODO_WRITE_TOOL_NAME)); + assert!( + !is_todo_tool_name("todo_write"), + "only the name the model calls" + ); + assert!(!is_todo_tool_name("todo__something_else")); + } + #[tokio::test] async fn all_advertised_todo_tools_dispatch_and_reinject_their_state() { let temp = tempfile::tempdir().unwrap(); diff --git a/crates/biorouter/src/prompts/system.md b/crates/biorouter/src/prompts/system.md index b35c36470..7b388e08a 100644 --- a/crates/biorouter/src/prompts/system.md +++ b/crates/biorouter/src/prompts/system.md @@ -143,6 +143,17 @@ session-scoped tool state; do not imply that Extension Manager is the only such through them in order, and keep track of progress so nothing is dropped. When todo/plan tools are available, keep a living plan and a per-item checklist: update each item's status as you go (in progress → completed) rather than rewriting the whole list, and before yielding confirm every item is completed or say why not. +{% if checklist_enforcement %} +- Biorouter enforces the checklist when a request has several steps (a numbered or bulleted list of actions, three or + more instructions, or instructions joined by "then", "after that", "finally" or "steps"): + - While the checklist is empty, your first action is `todo__todo_write`, with one `- [ ]` item per step. The first + other tool call you make in that turn is refused with a pointer back to it. That refusal happens once per turn: if + you have a reason not to keep a checklist, say so and repeat the call, and it runs. + - As you work, mark each item `in_progress` when you start it and `completed` when it is done, with + `todo__todo_update`. + - A turn that created or changed the checklist cannot end while items are unfinished, unless your final message + names each unfinished item by its `#N` id and says why it is not done. Otherwise you are sent back to finish. +{% endif %} - Once you start a task, carry it through to completion before yielding. Don't stop half-done, and don't gold-plate beyond what was asked. - Before editing a file, read the relevant parts, and don't guess its contents. Don't fabricate file paths, APIs, or diff --git a/docs/agent-loop/hooks/hooks-reference.md b/docs/agent-loop/hooks/hooks-reference.md index c35768b38..ac3df5081 100644 --- a/docs/agent-loop/hooks/hooks-reference.md +++ b/docs/agent-loop/hooks/hooks-reference.md @@ -279,6 +279,27 @@ spelling is accepted as an alias. - **PostToolUse blocks are capped** at 3 consecutive blocks per session (see [Blocking a tool result](#blocking-a-tool-result-posttooluse)). +### Built-in checks that run before your Stop hooks + +When the agent tries to finish a turn, Biorouter runs its own checks first, in +this order, and only then consults your `Stop` hooks: + +1. the done gate, when you configured one (`BIOROUTER_DONE_GATE`); +2. the self-critique pass, when you enabled it; +3. the Todo checklist check: a turn that created or changed the checklist may + not end while items are unfinished, unless the final message names each open + item. See [What Biorouter enforces](../../extensions/built-in/todo.md#what-biorouter-enforces). + +A built-in check that blocks ends that stop attempt, so your hooks run on the +next one. The checklist check runs ahead of your hooks because it is +deterministic and cheap, while a command hook may run a whole test suite; there +is no point paying for a hook on a stop that is already refused. It has its own +budget, 5 blocks per turn, which does not reset when the agent runs tools and +does not count against the Stop-hook cap above. Its feedback reaches the model +the same way a Stop-hook block does: as hidden feedback, with a notice for you. +It stands aside while a `/goal` is active, because the goal's own judge decides +then. + ### Scheduling of observe-only events **Observe-only events run detached, but are not discarded.** `Notification`, diff --git a/docs/extensions/built-in/code-execution.md b/docs/extensions/built-in/code-execution.md index 09c41807e..8d6cc9f49 100644 --- a/docs/extensions/built-in/code-execution.md +++ b/docs/extensions/built-in/code-execution.md @@ -77,6 +77,15 @@ The syntax rules are: > **Warning.** `execute_code` is annotated as destructive and non-idempotent, and it can reach every effective tool exposed by enabled capabilities and loaded extensions — including `developer`'s `shell` and `text_editor`. It inherits the same blast radius as those tools, so the permission controls in the [Developer capability guide](developer.md) and [permission modes](../../security/permission-modes.md) apply to it too. +## Tools that stay directly callable + +While Code Mode is on, the model's direct tool list shrinks to this capability's own tools, and everything else is reached by importing it into a script. A few tools stay direct calls anyway: + +- **Tools a script cannot reach.** These are the core `platform__*` operations and the workflow's structured-output tool, which the agent loop runs rather than any importable module, plus any tool the interface registered. +- **Tools a script must not run.** Deleting a knowledge base is shown to you for approval first, and a script call cannot raise that approval. +- **Delegation and workspace control.** The subagent tool and the `workspace__*` tools stay direct calls, because driving other conversations from inside a script is not what the sandbox is for. +- **The [Todo](todo.md) checklist tools.** All five stay direct calls, even though scripts can import them too. A checklist the model could only update by writing JavaScript was one it did not keep, and BioRouter's plan-first rule points a multi-step request at `todo_write` by name. + ## Example usage In this example, BioRouter compiles a report that would otherwise take several separate tool calls. @@ -110,6 +119,7 @@ The file has been saved to the root directory as `LOG.md`. ## Related documentation - [Developer capability](developer.md) — the `shell` and `text_editor` tools most Code Mode scripts import, and the access controls that constrain them. +- [Todo capability](todo.md) — the checklist tools that stay directly callable in Code Mode, and the plan-first rules that use them. - [Extension Manager capability](extension-manager.md) — the other lever for keeping the active tool count and context usage down. - [Context engineering](../../agent-loop/context-engineering.md) — the broader picture of how BioRouter manages its context window. - [Permission modes](../../security/permission-modes.md) — how to require approval before a script runs shell commands or edits files. diff --git a/docs/extensions/built-in/todo.md b/docs/extensions/built-in/todo.md index 9e43cc040..9ef262038 100644 --- a/docs/extensions/built-in/todo.md +++ b/docs/extensions/built-in/todo.md @@ -1,10 +1,40 @@ # Todo capability > **What this is.** User guide to the built-in Todo capability, which makes BioRouter break multi-step work into a tracked checklist and report progress as it goes. -> **Status:** Current. The capability is enabled by default, so no manual setup is normally needed. +> **Status:** Current. The capability is enabled by default, so no manual setup is normally needed. The rules under [What BioRouter enforces](#what-biorouter-enforces) are enforced by the agent loop as of 2026-09-11; before that they were advice in the system prompt that nothing checked. > **Audience:** end users. -The Todo capability keeps BioRouter organized on long tasks. BioRouter reaches for it automatically when a task has two or more steps, touches multiple files or components, or has uncertain scope. At the start it creates a checklist, updates the checklist as it works, and verifies at the end that every item is done — so you can see where it is rather than waiting for a single opaque answer. +The Todo capability keeps BioRouter organized on long tasks. When your request has several steps, BioRouter writes a checklist before it starts, ticks items off as it works, and does not end the turn with items left open unless it tells you which ones and why. The checklist appears in the chat summary's **To Do** section as soon as it exists, so you can see where BioRouter is rather than waiting for a single opaque answer. + +## What BioRouter enforces + +For a request with several steps, keeping the checklist is not left to the model's judgement. Three rules apply, and each one is bounded so that a model with a good reason can still proceed: + +1. **Plan first.** When your message reads as several steps and the chat's checklist is empty, BioRouter tells the model, for that turn, to write the checklist with `todo_write` before doing anything else. +2. **One redirect.** If the model reaches for another tool anyway, BioRouter refuses that first call (or that first batch of parallel calls) and points it back at `todo_write`. The refused call never runs, and the model sees why. A batch that writes the checklist alongside its other calls is not refused. This happens **once per turn**: if the model repeats the call — for instance after explaining why a checklist is not worth keeping here — it runs. +3. **No silent stop.** When a turn created or changed the checklist, BioRouter does not let it end while items are unfinished, unless the model's final message names each open item, by its `#N` id or by its text. Otherwise BioRouter sends the model back to finish, and shows you a notice saying which items are open. It does this at most five times per turn, and the count does not reset when the model runs tools. After the fifth time the turn ends anyway, with a notice that items were left open. + +The model is asked to say *why* an item is left open, but only the naming is checked. A mechanical check cannot tell a reason from a status recap; a named item is one you can see was not done, which is what the rule protects. + +### What counts as "several steps" + +- A numbered or bulleted list of actions, such as `1. create a temp dir 2. write hello.txt into it 3. count its bytes 4. delete it`, or a list of three or more items introduced by an instruction ("Build a page with: 1. a header 2. a pricing table 3. a footer"). +- Three or more instructions in a row ("Download the CSV, clean the missing values, and plot the distribution"). +- Two or more instructions joined by "then", "after that", "afterwards", "finally", "followed by" or "steps" ("Create the directory, then write hello.txt into it"). + +The detection is deliberately conservative, and it reads English cues only. A question about how to do something ("How do I create a venv and then install the dependencies?") never counts, and neither does text inside code blocks or a single instruction. When the detection does not fire, the model can still keep a checklist on its own, and the system prompt asks it to. + +### When none of this applies + +- The Todo capability is disabled, or `todo_write` is not on the model's tool list for this chat. +- Chat mode, where no tools run. +- Subagents. A delegated task's turn ends with an observe-only `SubagentStop` rather than a blockable stop, and its instructions come from the parent model. +- The Claude Code and Codex providers. Their tool calls reach BioRouter through the coding-agent bridge rather than BioRouter's own loop. +- Slash commands. + +While a `/goal` is active, the goal's evaluator decides when a turn may end, so the "no silent stop" rule stands aside; the first two rules still apply. + +The system prompt describes these rules only in a chat where they are enforced. > **Note.** This capability is **enabled by default**. Its internal registration still uses the legacy `PlatformExtensionDef` type; that storage name does not make it an installed extension. The configuration walkthrough below is only needed if you previously disabled it, or want to confirm its state. @@ -31,6 +61,8 @@ The Todo capability keeps BioRouter organized on long tasks. BioRouter reaches f ## Available tools +These five tools stay directly callable when Code Execution mode is on. Most other tools are then reached only by writing a script, and a checklist the model could only update from inside a script was one it did not keep. + | Tool | Description | |------|-------------| | `todo_write` | Replace the entire checklist with a markdown checklist. Used to seed the initial list. It discards every existing item and renumbers the `#N` ids, so the tools below are what BioRouter uses to change a list that already exists. | @@ -151,3 +183,5 @@ The documentation now follows a consistent pattern and provides a clear, organiz - [Installation](../../getting-started/installation.md) — the list of capabilities enabled out of the box. - [Subagents](../../agent-loop/subagents.md) — the other mechanism for structuring long, multi-step work. - [Context engineering](../../agent-loop/context-engineering.md) — how the checklist and living plan are re-injected into the model's context each turn. +- [Hooks reference](../../agent-loop/hooks/hooks-reference.md) — where the "no silent stop" check sits among the other checks that run when the agent tries to finish a turn. +- [Code Execution capability](code-execution.md) — the mode that routes most tools through scripts, and the tools it leaves directly callable. diff --git a/ui/desktop/src/components/ChatSummary.test.tsx b/ui/desktop/src/components/ChatSummary.test.tsx index 5d632e446..eaa2ce9ff 100644 --- a/ui/desktop/src/components/ChatSummary.test.tsx +++ b/ui/desktop/src/components/ChatSummary.test.tsx @@ -167,6 +167,12 @@ describe('compact chat summary', () => { * The live-refresh path itself already existed and is NOT what these tests add: * `useSessionTodos.ts:14` computes `revision` and lists it in the effect's deps * at `:63`. + * + * ⚠ The exemption described as missing above now exists (2026-09-11): the five + * `todo__*` tools stay directly callable in Code Execution mode, and the + * planning gate sends a multi-step turn to `todo__todo_write` first. So the + * COMMON path is now a top-level request, pinned by the last test below; the + * scripted path above still happens whenever a model imports the tools. */ const FOUR_TASK_SESSION = { id: 'chat', @@ -245,4 +251,68 @@ describe('the summary panel over persisted To Do state', () => { within(screen.getByRole('list', { name: 'To Do tasks' })).getAllByRole('listitem') ).toHaveLength(4); }); + + // The path the planning gate makes the usual one: the model's FIRST action on + // a multi-step request is a direct `todo__todo_write`, and the list has to + // appear the moment that call is acknowledged — then tick as updates land. + it('shows a checklist the moment a direct todo_write lands, then ticks it off', async () => { + mocks.getSession.mockResolvedValue({ data: NO_TASK_SESSION }); + const { rerender } = render(); + await waitFor(() => expect(mocks.getSession).toHaveBeenCalledTimes(1)); + expect(screen.queryByRole('region', { name: 'To Do progress' })).not.toBeInTheDocument(); + + const direct = (id: string, name: string): Message[] => + [ + { + role: 'assistant', + created: 0, + metadata: { agentVisible: true, userVisible: true }, + content: [ + { type: 'toolRequest', id, toolCall: { status: 'success', value: { name } } }, + { + type: 'toolResponse', + id, + toolResult: { status: 'success', value: { content: [], isError: false } }, + }, + ], + }, + ] as Message[]; + + const planned = { + id: 'chat', + extension_data: { + 'todo.v1': { + items: [ + { id: '1', text: 'create a temp dir', status: 'pending' }, + { id: '2', text: 'write hello.txt', status: 'pending' }, + ], + }, + }, + } as unknown as Session; + mocks.getSession.mockResolvedValue({ data: planned }); + const afterPlan = direct('plan', 'todo__todo_write'); + rerender(); + await waitFor(() => expect(screen.getByText('0 of 2 complete')).toBeInTheDocument()); + + const ticked = { + id: 'chat', + extension_data: { + 'todo.v1': { + items: [ + { id: '1', text: 'create a temp dir', status: 'completed' }, + { id: '2', text: 'write hello.txt', status: 'pending' }, + ], + }, + }, + } as unknown as Session; + mocks.getSession.mockResolvedValue({ data: ticked }); + rerender( + + ); + await waitFor(() => expect(screen.getByText('1 of 2 complete')).toBeInTheDocument()); + expect(screen.getByRole('progressbar')).toHaveAttribute('aria-valuenow', '1'); + }); }); diff --git a/ui/desktop/src/utils/sessionTodos.ts b/ui/desktop/src/utils/sessionTodos.ts index 19d3b0b14..d2a3636eb 100644 --- a/ui/desktop/src/utils/sessionTodos.ts +++ b/ui/desktop/src/utils/sessionTodos.ts @@ -109,16 +109,18 @@ const EXECUTED_CALLS_META_KEY = 'biorouter/tool-calls'; /** * How many acknowledged checklist mutations a tool result ran as SUB-calls. * - * `todo__*` is NOT exempt from `reply_parts::survives_code_execution_filter` - * (reply_parts.rs:206-219, applied at :286), and `code_execution` is - * `default_enabled: true` — so in a default chat the model cannot call - * `todo__todo_write` directly at all. It reaches it only from a script, and the - * transcript therefore carries ZERO top-level `todo__*` requests. A revision - * derived from top-level requests alone is then a constant for the whole - * session: the effect never re-runs and the panel is frozen at open time. - * Reopening flips `open`, which IS a dep, so it shows the truth again — which - * is exactly the reported symptom. + * Both channels are live, so both are read. Since 2026-09-11 the five + * `todo__*` tools are exempt from `reply_parts::survives_code_execution_filter` + * (`todo_extension::TODO_TOOL_NAMES`) and the planning gate points a multi-step + * turn at `todo__todo_write` by name, so the common case is now an ordinary + * top-level request, counted in `todoMutationRevision` below. But a script can + * still import the same tools, and a checklist it writes leaves NO top-level + * `todo__*` request in the transcript — only this meta. * + * That second channel is why this function exists. Before the exemption it was + * the ONLY one in a default chat (`code_execution` is `default_enabled: true`), + * and a revision derived from top-level requests alone stayed constant for the + * whole session: the effect never re-ran and the panel was frozen at open time. * Measured on the drive session `20260901_1`: top-level requests were * `code_execution` x39 and `workspace` x12 and nothing else, while 15 separate * `execute_code` runs carried `todo__*` sub-calls. From 5ca597804923c31cdf1dcaee95e681a78a2aa23f Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 12:10:47 -0700 Subject: [PATCH 19/75] fix(planning-gate): checked string access for clippy::string_slice; trace the stop check Every offset was a regex match boundary or a find() result, so a char boundary, but the workspace denies clippy::string_slice; str::get states the same thing without an index that could panic. The stop check now logs its block and give-up decisions, so a runtime run can be read from the daemon log. --- crates/biorouter/src/agents/planning_gate.rs | 34 +++++++++++++------- 1 file changed, 23 insertions(+), 11 deletions(-) diff --git a/crates/biorouter/src/agents/planning_gate.rs b/crates/biorouter/src/agents/planning_gate.rs index f7ce01441..4f37baf0f 100644 --- a/crates/biorouter/src/agents/planning_gate.rs +++ b/crates/biorouter/src/agents/planning_gate.rs @@ -309,9 +309,9 @@ fn numbered_list(prose: &str) -> Option<(&str, Vec<&str>)> { if run.len() > best.len() { best = run; } - if best.len() < 2 { - return None; - } + let &(list_start, _) = best.first().filter(|_| best.len() >= 2)?; + // Every offset here is a regex match boundary or a `find` result, so a + // char boundary; `get` only states that without an index that could panic. let items = best .iter() .enumerate() @@ -320,14 +320,15 @@ fn numbered_list(prose: &str) -> Option<(&str, Vec<&str>)> { Some(&(next_marker, _)) => next_marker, // The last item ends with its line: an inline list has one line, // and prose after a line list is not part of its last step. - None => prose[item_start..] - .find('\n') + None => prose + .get(item_start..) + .and_then(|rest| rest.find('\n')) .map_or(prose.len(), |offset| item_start + offset), }; - prose[item_start..end].trim() + prose.get(item_start..end).unwrap_or_default().trim() }) .collect(); - Some((&prose[..best[0].0], items)) + Some((prose.get(..list_start).unwrap_or_default(), items)) } fn bulleted_list(prose: &str) -> Option<(&str, Vec<&str>)> { @@ -341,7 +342,7 @@ fn bulleted_list(prose: &str) -> Option<(&str, Vec<&str>)> { items.push(item.as_str().trim()); } let start = first_start?; - (items.len() >= 2).then(|| (&prose[..start], items)) + (items.len() >= 2).then(|| (prose.get(..start).unwrap_or_default(), items)) } fn action_clause_count(prose: &str) -> usize { @@ -706,9 +707,9 @@ fn names_item(final_text: &str, final_words: &[String], item: &TodoItem) -> bool fn mentions_id(text: &str, id: &str) -> bool { let needle = format!("#{id}"); text.match_indices(&needle).any(|(at, _)| { - !text[at + needle.len()..] - .chars() - .next() + !text + .get(at + needle.len()..) + .and_then(|rest| rest.chars().next()) .is_some_and(|c| c.is_ascii_digit()) }) } @@ -894,11 +895,22 @@ impl Agent { return ChecklistStop::Clear; }; if blocks >= STOP_HOOK_BLOCK_CAP { + tracing::info!( + session_id, + open = objection.open, + "planning gate: checklist still open at the stop-check cap; letting the turn end" + ); return ChecklistStop::GiveUp { notice: give_up_notice(objection.open), }; } self.planning.record_stop_block(session_id); + tracing::info!( + session_id, + open = objection.open, + block = blocks + 1, + "planning gate: sent the turn back to finish or name its open checklist items" + ); ChecklistStop::Block { feedback: objection.feedback, notice: objection.notice, From 6b6f508cd184375e48f5567235c7ef7b73ac73ec Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 12:11:44 -0700 Subject: [PATCH 20/75] desktop: "Also use for new chats" ticks when its square is clicked MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found by driving the running app. `Checkbox` draws its square beside an `sr-only` input, and the switcher's label was a SIBLING (`htmlFor`), so only the words toggled the box — a click on the square itself did nothing. Wrap the box and its words in one `label`, as SessionListView's checkbox already is. The new test clicks the square (Checkbox's own target) and fails against the sibling-label markup with `expect(element).toBeChecked()`. --- .../subcomponents/SwitchModelModal.test.tsx | 15 +++++++++++++++ .../models/subcomponents/SwitchModelModal.tsx | 16 ++++++++++++---- 2 files changed, 27 insertions(+), 4 deletions(-) diff --git a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.test.tsx b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.test.tsx index 58b25cd51..cc20e7350 100644 --- a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.test.tsx +++ b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.test.tsx @@ -299,6 +299,21 @@ describe('SwitchModelModal — what the switch changes', () => { await settle(); }); + /** + * ⚠ Found by driving the running app, not by a test. `Checkbox` draws its + * square beside an `sr-only` input; with the label as a SIBLING (`htmlFor`) + * only the words toggled it, and a click on the square — the thing a person + * aims at — did nothing at all. + */ + it('ticks when the square itself is clicked, not only its words', async () => { + renderModal('s-1'); + const box = screen.getByRole('checkbox', { name: new RegExp(ALSO_FOR_NEW_CHATS_LABEL) }); + // The square: Checkbox's own 24px target, which holds the hidden input. + fireEvent.click(box.parentElement as HTMLElement); + expect(box).toBeChecked(); + await settle(); + }); + it('leaves new chats alone unless the box is ticked', async () => { renderModal('s-1'); fireEvent.click(await confirm()); diff --git a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.tsx b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.tsx index 7cc37e1bc..850d2ade0 100644 --- a/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.tsx +++ b/ui/desktop/src/components/settings/models/subcomponents/SwitchModelModal.tsx @@ -942,9 +942,17 @@ export const SwitchModelModal = ({ default: a switch made in a chat is a statement about that chat, and a public model chosen for one scratch chat must not become what every new chat — in every window — silently starts on. + + ⚠ The `label` WRAPS the box, as `SessionListView`'s does. `Checkbox` + draws a square beside an `sr-only` input, so with a sibling label + (`htmlFor`) only the text toggled it and clicking the square itself did + nothing — found in the running app, which is where it showed. */} {sessionId && !hostManaged && ( -
+
+ + )} {submitError && ( From 21abab80c0a8c90d6e81849b5ca918b360b6d51d Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 12:11:45 -0700 Subject: [PATCH 21/75] =?UTF-8?q?docs:=20model=20selection=20across=20wind?= =?UTF-8?q?ows=20=E2=80=94=20measured=20runtime=20results?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Correct the sandbox store path (`/data/sessions/sessions.db`), add the recipe for seeing the last look refuse an unannounced hand edit, record the toast-class trap (`TOAST_SURFACE_CLASS_NAME`, not `Toastify__toast`), and the 2026-09-11 measurements: 310 ms and 390 ms cross-window lag, both turns' token_events equal to the chip at send, the last look refusing both directions within 100 ms with no focus event. Also records, under "What this does not cover", that `/config/set_provider` writes provider then model as two writes — `config.yaml` held a mixed pair for ~55 ms — which a `/agent/start` in the gap would bind. Daemon work. --- .../model-selection-across-windows.md | 25 +++++++++++++++++-- 1 file changed, 23 insertions(+), 2 deletions(-) diff --git a/docs/desktop-ui/model-selection-across-windows.md b/docs/desktop-ui/model-selection-across-windows.md index 64170201e..99401e27a 100644 --- a/docs/desktop-ui/model-selection-across-windows.md +++ b/docs/desktop-ui/model-selection-across-windows.md @@ -106,6 +106,13 @@ the daemon classifies the chat by what it binds, whatever this check does. - **The window between the last look and `/agent/start`** — two loopback round trips — is not closed. Closing it needs `/agent/start` to accept an expected binding and refuse a mismatch, which is daemon work. +- **`/config/set_provider` is not atomic.** `set_config_provider` writes + `BIOROUTER_PROVIDER` and then `BIOROUTER_MODEL` as two config writes; measured on + 2026-09-11, `config.yaml` held `versa_azure` beside `gpt-6-astra` for about 55 ms of a + Codex → Versa switch. A re-read caused by an announcement never sees it, because the + announcement follows the write; one caused by a focus change landing in the gap could, and + the announcement right behind it corrects the chip. A `/agent/start` landing in the gap has + no such second chance and would bind the mixed pair. Daemon work. ## Tests @@ -130,14 +137,28 @@ window's chip by its accessible name (`button[aria-label^="Current model:"]`). 1. Change the model from window 1's Home chip (or tick **Also use for new chats** in a chat). Window 2's chip must read the new model within two seconds. -2. Send from window 2's Home composer, then read what the turn really ran on: +2. Send from window 2's Home composer, then read what the turn really ran on. The store sits + under the sandbox's `data/`, not beside `config/`: ```bash - sqlite3 "$SANDBOX/sessions/sessions.db" "select provider, model_id from token_events order by id desc limit 1" + sqlite3 -readonly ~/biorouter-runs//data/sessions/sessions.db "select provider, model_id from token_events order by id desc limit 1" ``` 3. Repeat in the private → public direction. At no point may window 2's chip read "Private model, UCSF" while its next turn goes to a public model. +4. To see the last look refuse, hand-edit the two keys in `~/biorouter-runs//config/config.yaml` + and send from window 2 without giving it focus. The chip changes, the text stays in the + composer, and no session is created. ⚠ The toast carries the app's own class + (`TOAST_SURFACE_CLASS_NAME`), not `Toastify__toast`: watch `section.Toastify`, or an + observer will report "no toast" for one that rendered. + +Measured on 2026-09-11 in a sandboxed instance at load average ~120, with both chips logged +every 50 ms against one clock: window 2's chip followed window 1's switch in 310 ms +(Versa → Codex) and 390 ms (Codex → Versa), in both cases directly from one correct label to +the other. The two turns sent from window 2 recorded `codex / gpt-6-astra` and +`versa_azure / gpt-5.5-2026-04-24` in `token_events`, each equal to the chip at the moment of +sending. The last look refused both directions of an unannounced hand edit within 100 ms, +with no focus or visibility event involved. ## Related documentation From 9dd93194d7372bca22095c1011e5ccf2ad93b882 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 12:32:35 -0700 Subject: [PATCH 22/75] feat(planning-gate): log whether a turn is gated A private provider's request log is metadata-only, so neither the reminder nor the prompt clause can be read back from it, and the first live run could not tell a gated turn from a model that planned on its own. The turn now logs one line saying whether the gate is out of scope, in scope, or armed (with the signal's kind and counts, never the prompt), and debug lines when the reminder is added or the checklist disarms it. --- crates/biorouter/src/agents/planning_gate.rs | 26 ++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/crates/biorouter/src/agents/planning_gate.rs b/crates/biorouter/src/agents/planning_gate.rs index 4f37baf0f..d3817bdfb 100644 --- a/crates/biorouter/src/agents/planning_gate.rs +++ b/crates/biorouter/src/agents/planning_gate.rs @@ -778,6 +778,7 @@ impl Agent { .map(|tool| tool.name.as_ref()), ); if !enforced { + tracing::debug!(session_id = %session.id, "planning gate: not in scope for this turn"); self.planning.begin(&session.id, None); return; } @@ -791,6 +792,23 @@ impl Agent { ), Err(_) => (false, fingerprint(None)), }; + // The one line that says, from outside, whether this turn is gated: + // a private provider's request log is metadata-only, so neither the + // reminder nor the prompt clause can be read back from it. Carries the + // signal's kind and counts, never the prompt. + match signal.filter(|_| list_was_empty) { + Some(signal) => tracing::info!( + session_id = %session.id, + ?signal, + "planning gate: multi-step turn with an empty checklist; reminder and redirect armed" + ), + None => tracing::debug!( + session_id = %session.id, + ?signal, + list_was_empty, + "planning gate: in scope, not armed (stop check only)" + ), + } self.planning.begin( &session.id, Some(TurnPlan { @@ -811,8 +829,16 @@ impl Agent { let signal = self.planning.armed(session_id)?; if self.checklist_exists(session_id).await { self.planning.note_list_exists(session_id); + tracing::debug!( + session_id, + "planning gate: checklist exists; reminder and redirect disarmed" + ); return None; } + tracing::debug!( + session_id, + "planning gate: reminder added to this call's context" + ); Some(reminder_text(signal)) } From aec4078df46b5cc7df53c771065c880f4d54aeb6 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 12:34:24 -0700 Subject: [PATCH 23/75] docs(todo): the intro claims no more than the stop check enforces Only the naming of an open item is checked, not the reason, so the intro no longer says the agent 'tells you which ones and why'. --- docs/extensions/built-in/todo.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/extensions/built-in/todo.md b/docs/extensions/built-in/todo.md index 9ef262038..456fef011 100644 --- a/docs/extensions/built-in/todo.md +++ b/docs/extensions/built-in/todo.md @@ -4,7 +4,7 @@ > **Status:** Current. The capability is enabled by default, so no manual setup is normally needed. The rules under [What BioRouter enforces](#what-biorouter-enforces) are enforced by the agent loop as of 2026-09-11; before that they were advice in the system prompt that nothing checked. > **Audience:** end users. -The Todo capability keeps BioRouter organized on long tasks. When your request has several steps, BioRouter writes a checklist before it starts, ticks items off as it works, and does not end the turn with items left open unless it tells you which ones and why. The checklist appears in the chat summary's **To Do** section as soon as it exists, so you can see where BioRouter is rather than waiting for a single opaque answer. +The Todo capability keeps BioRouter organized on long tasks. When your request has several steps, BioRouter writes a checklist before it starts, ticks items off as it works, and does not end the turn with items left open without naming them for you. The checklist appears in the chat summary's **To Do** section as soon as it exists, so you can see where BioRouter is rather than waiting for a single opaque answer. ## What BioRouter enforces From 88c26d51622ce306370bcc9c6a6cf7f6cf47c3e7 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 13:59:40 -0700 Subject: [PATCH 24/75] test(desktop): stub the privacy-off note in the Hub start-failure test main's H3 note (PrivacyTiersOffNote) now renders in Hub and needs a router and two ConfigContext hooks this test's mocks do not provide; it has nothing to do with starting a chat. --- ui/desktop/src/components/Hub.startFailure.test.tsx | 3 +++ 1 file changed, 3 insertions(+) diff --git a/ui/desktop/src/components/Hub.startFailure.test.tsx b/ui/desktop/src/components/Hub.startFailure.test.tsx index 3f9715245..21e2649b0 100644 --- a/ui/desktop/src/components/Hub.startFailure.test.tsx +++ b/ui/desktop/src/components/Hub.startFailure.test.tsx @@ -25,6 +25,9 @@ vi.mock('../toasts', async (importOriginal) => ({ toastError: mockToastError, })); vi.mock('./sessions/SessionsInsights', () => ({ SessionInsights: () => null })); +// The privacy-off note (H3) needs a router and two more ConfigContext hooks, and +// says nothing about starting a chat. +vi.mock('./privacy/PrivacyTiersOffNote', () => ({ PrivacyTiersOffNote: () => null })); vi.mock('./ConfigContext', () => ({ useConfig: () => ({ extensionsList: [], From 970e73071a042d08c643003e2e0b0863395e1005 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 14:06:44 -0700 Subject: [PATCH 25/75] fix(desktop): a workflow captured from a chat with no primary has no primary The create-workflow modal computes the captured default in two places: from the chat's own selection read, and from the generated workflow's `knowledge_bases` block. Both fell back to `visible[0]` when the chat named no primary, so the saved workflow's `default` was the first visible base. `apply_knowledge_selection` maps `default: Some(id)` to `PrimaryUpdate::Set(id)`, so every chat the workflow started got a primary, the target of KB-less writes, that the chat it was captured from never had. The daemon's rule is the opposite. `plan_knowledge_selection` never infers the primary from `visible`, and the doc on `plan_workflow_knowledge_selection` says why: a promoted primary turns "I did not say where to write" into a commit into someone's base. With no primary, a KB-less write fails and names the candidates. Both paths now go through `primaryAmong`: the named primary if it is among the bases the workflow will see, and otherwise `null`. The daemon's block for a chat with no primary omits `default` entirely (`skip_serializing_if`), and that reads as `null` too. A primary outside `visible` becomes `null`, not unioned in the way the daemon unions an author's `default`. The daemon does that because somebody wrote the workflow and meant it. A captured primary outside its own set is an inconsistent read. The block is one locked snapshot and never is one, but the modal's own read is two requests (the base list and the selection), and a base created or deleted between their answers leaves the selection naming a base the list lacks. Unioning could save a deleted base as the default, which `set_visible_kbs` refuses, so every chat the workflow starts would fail with a 400. After a failed list it would save the primary as the only visible base, hiding every other one. Tests: the read path with no generated block; the generated block with `default: null` and with `default` absent (the daemon's own wire shape); and a primary outside the visible bases, once from the read and once from the block. All five fail before this change: each saved the first visible base. --- .../CreateWorkflowFromSessionModal.tsx | 51 +++-- .../CreateWorkflowFromSessionModal.test.tsx | 184 +++++++++++++++--- 2 files changed, 196 insertions(+), 39 deletions(-) diff --git a/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx b/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx index d7b4361a7..c11c4f4a2 100644 --- a/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx +++ b/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx @@ -26,6 +26,35 @@ interface CreateWorkflowFromSessionModalProps { const KNOWLEDGE_SELECTION_UNREAD = "Could not load this chat's knowledge bases, so none were selected automatically."; +/** + * The primary a captured selection keeps: the one it names, if that base is + * among the ones the workflow will see, and otherwise none. + * + * ⚠ **Never the first visible base.** A saved `default` becomes the primary of + * every chat the workflow starts (`apply_knowledge_selection`), and the primary + * is where KB-less writes go. The daemon never infers it from `visible` + * (`plan_knowledge_selection` in `crates/biorouter/src/workflow/runtime.rs`), + * so a chat with no primary gives a workflow with no primary. Falling back to + * `visible[0]` gave each of those chats a write target the chat it was captured + * from never had. + * + * A primary outside `visible` is dropped, where the daemon would union it in. + * The daemon unions a `default` because somebody wrote that workflow and meant + * it; a captured primary outside its own set is an inconsistent read instead. + * The generated block is one locked snapshot and never is one, but this + * modal's own read is two requests, and a base created or deleted between their + * answers leaves the selection naming a base the list lacks. Unioning it could + * save a deleted base as the default, which `set_visible_kbs` refuses, so every + * chat the workflow starts would fail; after a failed list, it would save the + * primary as the only visible base and hide every other one. + */ +function primaryAmong( + primary: string | null | undefined, + visible: readonly string[] +): string | null { + return primary && visible.includes(primary) ? primary : null; +} + export default function CreateWorkflowFromSessionModal({ isOpen, onClose, @@ -143,10 +172,8 @@ export default function CreateWorkflowFromSessionModal({ const visible = bases .filter((base) => !selection.hiddenKbIds.has(base.id)) .map((base) => base.id); - const primary = selection.primaryKbId; - const defaultId = primary && visible.includes(primary) ? primary : (visible[0] ?? null); setWorkflowKnowledgeBaseIds(visible); - setDefaultKnowledgeBaseId(defaultId); + setDefaultKnowledgeBaseId(primaryAmong(selection.primaryKbId, visible)); }); // The daemon's catalog, so a skill bundled inside an installed extension @@ -233,21 +260,11 @@ export default function CreateWorkflowFromSessionModal({ // modal's own read said, the selection is known now. setKnowledgeSelectionUnread(false); const visible = workflow.knowledge_bases.visible ?? []; - generatedResourcesRef.current.knowledgeBases = { - default: - workflow.knowledge_bases.default && - visible.includes(workflow.knowledge_bases.default) - ? workflow.knowledge_bases.default - : (visible[0] ?? null), - visible, - }; + // The daemon sends no `default` at all for a chat with no primary. + const defaultId = primaryAmong(workflow.knowledge_bases.default, visible); + generatedResourcesRef.current.knowledgeBases = { default: defaultId, visible }; setWorkflowKnowledgeBaseIds(visible); - setDefaultKnowledgeBaseId( - workflow.knowledge_bases.default && - visible.includes(workflow.knowledge_bases.default) - ? workflow.knowledge_bases.default - : (visible[0] ?? null) - ); + setDefaultKnowledgeBaseId(defaultId); } if (workflow.skills && workflow.skills.length > 0) { diff --git a/ui/desktop/src/components/workflows/__tests__/CreateWorkflowFromSessionModal.test.tsx b/ui/desktop/src/components/workflows/__tests__/CreateWorkflowFromSessionModal.test.tsx index d06994d93..2ffc1f05d 100644 --- a/ui/desktop/src/components/workflows/__tests__/CreateWorkflowFromSessionModal.test.tsx +++ b/ui/desktop/src/components/workflows/__tests__/CreateWorkflowFromSessionModal.test.tsx @@ -3,7 +3,7 @@ import { act, render, screen, waitFor } from '@testing-library/react'; import userEvent from '@testing-library/user-event'; import CreateWorkflowFromSessionModal from '../CreateWorkflowFromSessionModal'; import { createWorkflow, getActive, listBases, skillCatalogHandler } from '../../../api/sdk.gen'; -import type { CreateWorkflowResponse } from '../../../api/types.gen'; +import type { CreateWorkflowResponse, WorkflowKnowledgeBases } from '../../../api/types.gen'; import { saveWorkflow } from '../../../workflow/workflow_management'; import { reachGatedGetActive, USER_ACTION_KEY } from '../../../test/reachGate'; @@ -85,6 +85,28 @@ const mockListBases = vi.mocked(listBases); const mockSkillCatalog = vi.mocked(skillCatalogHandler); const mockSaveWorkflow = vi.mocked(saveWorkflow); +/** Cross a macrotask boundary, so every `.then` already queued has run. */ +async function settleReads() { + await act(async () => { + await new Promise((resolve) => setTimeout(resolve, 0)); + }); +} + +/** Wait for the form, press "Create workflow", and return what was saved. */ +async function saveTheWorkflow(user: ReturnType) { + await waitFor( + () => { + expect(screen.getByTestId('create-workflow-button')).toBeEnabled(); + }, + { timeout: 2000 } + ); + await user.click(screen.getByTestId('create-workflow-button')); + await waitFor(() => { + expect(mockSaveWorkflow).toHaveBeenCalled(); + }); + return mockSaveWorkflow.mock.calls[0]?.[0]; +} + describe('CreateWorkflowFromSessionModal', () => { const defaultProps = { isOpen: true, @@ -627,27 +649,6 @@ describe('CreateWorkflowFromSessionModal', () => { warn.mockRestore(); }); - /** Cross a macrotask boundary, so every `.then` already queued has run. */ - async function settleReads() { - await act(async () => { - await new Promise((resolve) => setTimeout(resolve, 0)); - }); - } - - async function saveTheWorkflow(user: ReturnType) { - await waitFor( - () => { - expect(screen.getByTestId('create-workflow-button')).toBeEnabled(); - }, - { timeout: 2000 } - ); - await user.click(screen.getByTestId('create-workflow-button')); - await waitFor(() => { - expect(mockSaveWorkflow).toHaveBeenCalled(); - }); - return mockSaveWorkflow.mock.calls[0]?.[0]; - } - it('captures no knowledge bases rather than every base', async () => { const user = userEvent.setup(); render(); @@ -723,4 +724,143 @@ describe('CreateWorkflowFromSessionModal', () => { }); }); }); + + /** + * A chat with no primary knowledge base gives a workflow with no primary. + * + * The modal filled the gap with the first visible base, on the read path and + * the generation path alike, and `apply_knowledge_selection` + * (`crates/biorouter/src/workflow/runtime.rs`) turns a saved `default` into + * `PrimaryUpdate::Set`. So every chat the workflow started got a write target + * the chat it was captured from never had. The daemon's rule is the opposite + * (`plan_knowledge_selection`): the primary comes only from `default`, and is + * never inferred from `visible`. + */ + describe("the workflow's primary knowledge base", () => { + const kb = (id: string) => ({ id, name: id, color: '#cf6d47', created_at: '' }); + /** What the daemon answers for this chat: two bases, and neither is primary. */ + const NO_PRIMARY = { + kb_ids: ['lab-notes', 'soul'], + primary_kb: null, + active_kb: null, + hidden_kbs: ['grant-drafts'], + }; + const generation = (knowledgeBases?: WorkflowKnowledgeBases) => ({ + data: { + workflow: { + title: 'Analyzed Workflow Title', + description: 'Analyzed description', + instructions: 'Analyzed instructions', + ...(knowledgeBases ? { knowledge_bases: knowledgeBases } : {}), + }, + error: undefined, + }, + error: undefined, + request: new globalThis.Request('http://localhost/test'), + response: new globalThis.Response(), + }); + /** Leave the generation's block as the only statement of the chat's selection. */ + const failTheSelectionRead = () => + mockGetActive.mockResolvedValue({ data: undefined, error: 'Failed to fetch' } as never); + let warn: MockInstance; + + beforeEach(() => { + warn = vi.spyOn(console, 'warn').mockImplementation(() => {}); + mockListBases.mockResolvedValue({ + data: [kb('soul'), kb('lab-notes'), kb('grant-drafts')], + error: undefined, + } as never); + mockGetActive.mockResolvedValue({ data: NO_PRIMARY, error: undefined } as never); + // No block of its own, so the chat's selection as the modal read it is the + // only place the saved workflow can take its bases from. + mockCreateWorkflow.mockResolvedValue(generation()); + }); + + afterEach(() => { + warn.mockRestore(); + mockListBases.mockResolvedValue({ data: [], error: undefined } as never); + mockGetActive.mockResolvedValue({ + data: { active_kb: null, hidden_kbs: [] }, + error: undefined, + } as never); + }); + + it('saves no primary when the chat has none', async () => { + const user = userEvent.setup(); + render(); + + const saved = await saveTheWorkflow(user); + + expect(mockGetActive).toHaveBeenCalledWith( + expect.objectContaining({ query: { session_id: defaultProps.sessionId } }) + ); + expect(saved?.knowledge_bases).toEqual({ default: null, visible: ['soul', 'lab-notes'] }); + }); + + // The daemon never sends `default: null` itself: the field is + // `skip_serializing_if = "Option::is_none"`, so its own block for a chat + // with no primary has no `default` at all. The two mean the same thing. + it.each<[string, WorkflowKnowledgeBases]>([ + ['null', { default: null, visible: ['lab-notes', 'soul'] }], + ['absent', { visible: ['lab-notes', 'soul'] }], + ])('saves no primary when the generated block has none (default %s)', async (_, block) => { + const user = userEvent.setup(); + failTheSelectionRead(); + mockCreateWorkflow.mockResolvedValue(generation(block)); + render(); + + const saved = await saveTheWorkflow(user); + + expect(saved?.knowledge_bases).toEqual({ default: null, visible: ['lab-notes', 'soul'] }); + }); + + /** + * A primary that is not among the bases the workflow will see is dropped, + * not unioned into them the way `plan_knowledge_selection` unions a + * `default` missing from `visible`. The daemon does that for a workflow + * somebody wrote, whose author plainly meant that `default`. Nobody wrote + * this one: a captured primary outside its own set is an inconsistent read. + */ + describe('a primary outside the visible bases', () => { + // The modal reads the base list and the selection as two requests, which + // can answer in either order, so a base created or deleted between them + // leaves the selection naming a primary the list does not hold. Unioning it + // in could save a base that no longer exists as the default, and + // `set_visible_kbs` refuses a primary outside the set, so every chat the + // workflow starts would then fail. + it("is not saved from the chat's selection", async () => { + const user = userEvent.setup(); + mockGetActive.mockResolvedValue({ + data: { + kb_ids: ['lab-notes', 'new-notes', 'soul'], + primary_kb: 'new-notes', + active_kb: 'new-notes', + hidden_kbs: ['grant-drafts'], + }, + error: undefined, + } as never); + render(); + + const saved = await saveTheWorkflow(user); + + expect(saved?.knowledge_bases).toEqual({ default: null, visible: ['soul', 'lab-notes'] }); + }); + + // The daemon's block is one locked snapshot whose primary is always a + // member of its set (`selection_unlocked`), so this guards against a block + // that breaks that; today's daemon never sends one. + it('is not saved from the generated block', async () => { + const user = userEvent.setup(); + failTheSelectionRead(); + mockCreateWorkflow.mockResolvedValue( + generation({ default: 'grant-drafts', visible: ['lab-notes', 'soul'] }) + ); + render(); + + const saved = await saveTheWorkflow(user); + + expect(saved?.knowledge_bases).toEqual({ default: null, visible: ['lab-notes', 'soul'] }); + }); + }); + }); }); From 7c76088ed58fb386713e2cd47253d2c1ae1a8c17 Mon Sep 17 00:00:00 2001 From: Wanjun Gu Date: Fri, 11 Sep 2026 14:07:54 -0700 Subject: [PATCH 26/75] fix(desktop): only the Default control names a workflow's primary knowledge base The previous commit stops the create-workflow modal from inventing a primary when it captures a chat's selection. Editing the selection afterwards still invented one, in two places: - The picker (`WorkflowResourcePicker`). Switching a base on made it the default whenever none was set, and switching the default off handed the role to `next[0]`. - The modal's `onKnowledgeBaseIdsChange`. It promoted `ids[0]` on any change while no default was set, including switching off an unrelated base. So a user who captured a chat with no primary and then removed one base got the first remaining base as the write target of every chat the workflow starts. These are user gestures, not captures, which is the case for keeping them. They go anyway. The gesture is "this workflow may search this base", and the promotion adds "and KB-less writes go here", which the user did not ask for. It is the promotion the daemon removed in 3a6e6888 ("never promote a sole visible base to the primary"): "exactly one candidate" was treated as consent, which it is not, and the author who wants that base as the write target has a field to say so. In the picker, that field is the Default control, one click away on every selected row. - Switching a base on changes the selection only. - Switching the default off leaves no default, rather than passing it on. - The modal's handler keeps only the membership rule: a named default stays while its base is selected and becomes `null` when it is not. - The Default control is now a toggle (`aria-pressed`), so pressing the current default clears it. Without that, "these bases, and no default", the state a chat with no primary now captures, could not be restored once any base had been made the default, short of switching it off and on again. Each control is named for its row ("Default KB: lab-notes") so the pressed state says which base it belongs to. Tests, in the modal: switching a base on, switching another off, and switching the primary off each save no primary. All three fail on the previous commit. Switching a base on still fails with only the handler fixed (the picker names the base) or only the picker fixed (the handler names `ids[0]`). In the picker: switching on leaves the default unset, switching the default off clears it, and the Default control names a base and clears it when pressed again. All three fail before this change. A fourth, switching another base off leaves the default alone, passes either way as a guard against clearing it on every toggle. --- .../CreateWorkflowFromSessionModal.tsx | 10 +- .../CreateWorkflowFromSessionModal.test.tsx | 79 +++++++++++- .../shared/WorkflowResourcePicker.tsx | 23 ++-- .../__tests__/WorkflowResourcePicker.test.tsx | 122 ++++++++++++++++++ 4 files changed, 219 insertions(+), 15 deletions(-) create mode 100644 ui/desktop/src/components/workflows/shared/__tests__/WorkflowResourcePicker.test.tsx diff --git a/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx b/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx index c11c4f4a2..45c5f9b57 100644 --- a/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx +++ b/ui/desktop/src/components/workflows/CreateWorkflowFromSessionModal.tsx @@ -509,11 +509,11 @@ export default function CreateWorkflowFromSessionModal({ onKnowledgeBaseIdsChange={(ids) => { resourceEditsRef.current.knowledgeBases = true; setWorkflowKnowledgeBaseIds(ids); - if (defaultKnowledgeBaseId && !ids.includes(defaultKnowledgeBaseId)) { - setDefaultKnowledgeBaseId(ids[0] ?? null); - } else if (!defaultKnowledgeBaseId && ids.length > 0) { - setDefaultKnowledgeBaseId(ids[0]); - } + // Switching bases on or off never names a primary: only the + // picker's Default control does. A primary already named stays + // while its base is selected, and goes when it is not, rather + // than passing to whichever base is left. + setDefaultKnowledgeBaseId((current) => primaryAmong(current, ids)); }} defaultKnowledgeBaseId={defaultKnowledgeBaseId} onDefaultKnowledgeBaseIdChange={(id) => { diff --git a/ui/desktop/src/components/workflows/__tests__/CreateWorkflowFromSessionModal.test.tsx b/ui/desktop/src/components/workflows/__tests__/CreateWorkflowFromSessionModal.test.tsx index 2ffc1f05d..4a8f963a3 100644 --- a/ui/desktop/src/components/workflows/__tests__/CreateWorkflowFromSessionModal.test.tsx +++ b/ui/desktop/src/components/workflows/__tests__/CreateWorkflowFromSessionModal.test.tsx @@ -1,4 +1,14 @@ -import { describe, it, expect, vi, beforeEach, afterEach, type MockInstance } from 'vitest'; +import { + describe, + it, + expect, + vi, + beforeAll, + afterAll, + beforeEach, + afterEach, + type MockInstance, +} from 'vitest'; import { act, render, screen, waitFor } from '@testing-library/react'; import userEvent from '@testing-library/user-event'; import CreateWorkflowFromSessionModal from '../CreateWorkflowFromSessionModal'; @@ -764,6 +774,22 @@ describe('CreateWorkflowFromSessionModal', () => { mockGetActive.mockResolvedValue({ data: undefined, error: 'Failed to fetch' } as never); let warn: MockInstance; + beforeAll(() => { + // The picker is a Radix popover, and floating-ui measures it. + vi.stubGlobal( + 'ResizeObserver', + class { + observe() {} + unobserve() {} + disconnect() {} + } + ); + }); + + afterAll(() => { + vi.unstubAllGlobals(); + }); + beforeEach(() => { warn = vi.spyOn(console, 'warn').mockImplementation(() => {}); mockListBases.mockResolvedValue({ @@ -862,5 +888,56 @@ describe('CreateWorkflowFromSessionModal', () => { expect(saved?.knowledge_bases).toEqual({ default: null, visible: ['lab-notes', 'soul'] }); }); }); + + /** + * Only the picker's Default control names a primary. Switching a base on + * says the workflow may search it, not that KB-less writes go there, and + * switching one off must not hand the role to whichever base is left. + */ + describe('when the user edits the knowledge bases', () => { + async function openThePicker(user: ReturnType) { + await user.click(await screen.findByText('2 KBs selected')); + } + + it('switching a base on does not make it the primary', async () => { + const user = userEvent.setup(); + render(); + + await openThePicker(user); + await user.click(screen.getByRole('switch', { name: 'Toggle grant-drafts' })); + + const saved = await saveTheWorkflow(user); + expect(saved?.knowledge_bases).toEqual({ + default: null, + visible: ['soul', 'lab-notes', 'grant-drafts'], + }); + }); + + it('switching a base off does not make another the primary', async () => { + const user = userEvent.setup(); + render(); + + await openThePicker(user); + await user.click(screen.getByRole('switch', { name: 'Toggle lab-notes' })); + + const saved = await saveTheWorkflow(user); + expect(saved?.knowledge_bases).toEqual({ default: null, visible: ['soul'] }); + }); + + it('switching the primary off leaves no primary', async () => { + const user = userEvent.setup(); + mockGetActive.mockResolvedValue({ + data: { ...NO_PRIMARY, primary_kb: 'lab-notes', active_kb: 'lab-notes' }, + error: undefined, + } as never); + render(); + + await openThePicker(user); + await user.click(screen.getByRole('switch', { name: 'Toggle lab-notes' })); + + const saved = await saveTheWorkflow(user); + expect(saved?.knowledge_bases).toEqual({ default: null, visible: ['soul'] }); + }); + }); }); }); diff --git a/ui/desktop/src/components/workflows/shared/WorkflowResourcePicker.tsx b/ui/desktop/src/components/workflows/shared/WorkflowResourcePicker.tsx index 8b9ab5b87..a6e609aca 100644 --- a/ui/desktop/src/components/workflows/shared/WorkflowResourcePicker.tsx +++ b/ui/desktop/src/components/workflows/shared/WorkflowResourcePicker.tsx @@ -67,21 +67,22 @@ export function WorkflowResourcePicker({ }); }, [items, query, selected]); + // Switching an item on or off changes the selection, never the default: only + // the Default control names one. For knowledge bases the default becomes the + // primary of every chat the workflow starts, which is where KB-less writes + // go, and the daemon never infers that pointer (`plan_knowledge_selection`). + // So a base switched on is not made the default, and switching the default + // off leaves none rather than passing the role to the first base left. const toggleSelected = (id: string) => { if (selected.has(id)) { - const next = selectedIds.filter((selectedId) => selectedId !== id); - onSelectedIdsChange(next); + onSelectedIdsChange(selectedIds.filter((selectedId) => selectedId !== id)); if (defaultId === id) { - onDefaultIdChange?.(next[0] ?? null); + onDefaultIdChange?.(null); } return; } - const next = [...selectedIds, id]; - onSelectedIdsChange(next); - if (!defaultId) { - onDefaultIdChange?.(id); - } + onSelectedIdsChange([...selectedIds, id]); }; return ( @@ -161,8 +162,12 @@ export function WorkflowResourcePicker({ )}
{onDefaultIdChange && isSelected && ( + // A toggle, so "selected, and no default" stays reachable + // once a default has been named.