From 611bea242cabf88f23dfbd82b467c41a2f637970 Mon Sep 17 00:00:00 2001 From: Phil Merrell Date: Fri, 24 Jul 2026 09:04:57 -0600 Subject: [PATCH] Release/1.11.0 (#722) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(memory): wire memory-spaces table/bucket names onto app-api app-api owns the Memory Spaces CRUD surface (`/memory/spaces/*`) but its container environment only set MEMORY_SPACES_ENABLED — never the table or bucket names the service reads (DYNAMODB_MEMORY_SPACES_TABLE_NAME / S3_MEMORY_SPACES_BUCKET_NAME). Without them the repository falls back to the default "memory-spaces" table name, which doesn't exist, so every read throws a boto3 ResourceNotFoundException that the centralized handler maps to a 502 "Upstream service error." (inference-api already sets the identical trio, but per the service-boundary rule it isn't the one serving these routes.) Thread `refs.memorySpacesTable`/`.memorySpacesBucket` through AppApiSsmParams and emit both names next to the flag in buildAppApiEnvironment. Names are always wired (read lazily); only MEMORY_SPACES_ENABLED gates route mounting, so flipping the switch on later needs no env change. New unit test guards the wiring; tsc + 431 infra jest tests green. Co-Authored-By: Claude Opus 4.8 * docs(cdk): capture the "wire resource name to every compute" rule Fold the lesson from the app-api env-wiring fix into the cdk-infrastructure skill so the next construct-author sees it while wiring, not after a 502. Adds a "Cross-Construct References" subsection: set a resource's name env var on every compute that reads it (one doesn't imply the other), the silent-502 failure mode (default-name fallback → ResourceNotFoundException → generic 502, invisible to synth/CI), and the env-map test guard + service-boundary caveat. Co-Authored-By: Claude Opus 4.8 * feat(memory): deterministic consolidation health pass (A6) The safe, non-LLM slice of Workstream A6. `MemorySpaceService.consolidate` (editor+) + `POST /memory/spaces/{id}/consolidate` → a `ConsolidationReport`. Auto-fixes only storage hygiene: orphaned content-addressed objects — keys under a space's prefix that no manifest entry or the index pointer references (leaks from crashed/raced writes) — are GC'd (new `MemorySpaceStore.list_keys` drives it). Everything that needs a judgment call is *reported, not mutated*: - duplicate content across slugs (same content hash) — which slug survives is semantic, so it's flagged, never auto-merged; - dead `[[slug]]` wikilinks in MEMORY.md — reported; opt-in `stripDeadLinks` unlinks them (they point nowhere) while preserving the surrounding prose; - over-cap entry counts (`MEMORY_SPACE_INDEX_CAP`, default 200) — flagged, never auto-evicted. This deliberately does not merge/evict/rewrite durable memory — that's deferred to the LLM consolidation pass (Workstream B era), which extends this exact `consolidate()` seam once agentic writes create real duplication/staleness to act on. On-demand only for now; scheduler/threshold auto-run and SPA surfacing are follow-ups. Tests: 8 service (healthy report, orphan GC + skip, dup-report-no-merge, dead-link report + strip-keeps-prose, over-cap flag, editor gate) + 4 route (report shape, no-body, viewer 403, flag-off 404) + 3 store (list_keys prefix scoping / empty / disabled). 104 memory + boundary tests green. Co-Authored-By: Claude Opus 4.8 * docs(agent): Agent Designer spec — unified primitive-binding surface Captures the strategy for the "Agent Designer" (Agent Harness Editor): a new authoring surface that composes an Agent from RBAC-governed primitives (instructions, model, KBs, tools, skills, Memory Spaces, + future), replacing the term/feature "Assistant." Locks the load-bearing decisions: own a primitive-agnostic Agent contract and federate AgentCore Registry later rather than build on it (adopt-with-boundary precedent); evolve the assistant store in place (no parallel table); a uniform bindings[] model with the model as a governed single-select; RBAC = compose the five existing per-primitive access checks (incl. ModelAccessService), not a new system; design-time filter + run-time re-resolution per invoker with block-on- missing v1; ship memory-consumption as a thin vertical slice before the full Designer. Phasing 0–5 + later AWS federation; supersedes the memory spec's "extend the Assistant" §B1 framing. Co-Authored-By: Claude Opus 4.8 * feat(agents): Agent contract + compat mapping in shared assistants models Phase 1 (PR-1) of the Agent Designer. Pure library, zero behavior change: legacy Assistants read unchanged and no caller passes the new fields yet. - AgentModelConfig (D3 governed single-select; field is model_settings/ alias modelConfig to dodge pydantic's reserved model_config — R3) - AgentBinding (open kind on read, KNOWN_BINDING_KINDS for request validation) - optional model_settings + bindings on Assistant (additive) - compat.effective_bindings/to_agent_view (D2): absent bindings synthesize a knowledge_base binding reffing the assistant id (KB's only stable identity, F4 deferred — R4); absent model maps to None, never fabricated (R1) - Decimal-safe serialization for modelConfig.params floats Co-Authored-By: Claude Opus 4.8 * feat(agents): persist bindings + modelConfig with design-time validation Phase 1 (PR-2). The Agent fields now round-trip through the rag-assistants store and are validated at write time by composing existing RBAC checks (D4). Legacy clients are unaffected: the SPA sends none of the new fields and the AssistantResponse surface is unchanged. - service.create_assistant/update_assistant thread bindings + model_settings; to_ddb_safe on write / from_ddb on read so modelConfig.params floats survive DynamoDB (Decimal); explicit [] replaces bindings, absent leaves them untouched - app_api/agents/services/binding_validation.py composes model access (ModelAccessService), memory resolve_permission (viewer+/editor+), the implicit-KB rejection, and inert shape-only checks for tool/skill (D4/D5) - assistants POST/PUT validate then pass through; validation raises 4xx outside the create handler's generic except so it isn't masked as 500 - tests: validation matrix (incl. inert no-RBAC guarantee), persistence round trip, legacy no-field read Co-Authored-By: Claude Opus 4.8 * feat(agents): /agents alias router behind AGENTS_API_ENABLED (dark) Phase 1 (PR-3). A governed Agent read/write surface over the evolved assistant store: same shared service functions and identity-based access gates as /assistants, but returning the Agent shape (compat.to_agent_view -> AgentResponse) so callers see modelConfig + bindings. Legacy ids valid unchanged. - feature_flags.agents_enabled(): AGENTS_API_ENABLED, default OFF (memory-spaces pattern) — surface 404s while off, ships incrementally, /assistants unaffected - app_api/agents/routes.py: require_agents_enabled 404-gate; draft/create/list/ get/update/delete + 4 shares endpoints, delegating to apis.shared.assistants service and reprojecting via to_agent_view; create/update run binding_validation - AgentResponse/AgentsListResponse/AgentSharesResponse (agentId == assistantId) - main.py mounts the router - test-chat + document sub-routes deliberately excluded (would force a 2nd architecture import-boundary exception); list is owner+shared (public/pagination parity deferred to the Phase-4 Designer) - tests: 404-gate, agentId/bindings projection, CRUD permission gating, shares Co-Authored-By: Claude Opus 4.8 * feat(agents): wire AGENTS_API_ENABLED through CDK + Phase 1 docs Phase 1 (PR-4). Completes Phase 1 deployability: the /agents surface can now be turned on per environment. No new AWS resources — the flag only gates whether the routes 404 (the assistant store it reads is always present). - config.ts: AgentsConfig { enabled }; CDK_AGENTS_API_ENABLED (default off, empty-string-safe) or an `agents.enabled` cdk.json context, mirroring memorySpaces exactly - app-api-environment.ts: AGENTS_API_ENABLED env on app-api - infra tests: default-off / opt-on assertion; mock-config default - docs: agent-designer.md Phase-1 status (+ the two refinements and the Oliver-dogfood-gated-on-Phase-3 note); CHANGELOG [Unreleased] The live Oliver dogfood (D6) is deliberately NOT included: it needs Phase 3 harness resolution + Memory Spaces deployed to the target env before a memory_space binding resolves at invocation. Tracked as the Phase 3 payoff. Co-Authored-By: Claude Opus 4.8 * feat(agents): thread AGENTS_API_ENABLED to the inference runtime (Phase 3 PR-0) Phase 3 harness resolution runs inside inference-api, so the runtime needs the same flag the app-api surface got in #593. Default off, mirrors the app-api wiring; without it the harness ignores Agent bindings entirely (today's behavior). Co-Authored-By: Claude Opus 4.8 * feat(agents): resolve Agent modelConfig at invocation, per invoker (Phase 3 PR-A) The Harness now re-resolves an Agent's governed modelConfig against the INVOKING user (D5) and applies it to model selection. Absent modelConfig ⇒ the model resolves exactly as today; gated on AGENTS_API_ENABLED (off in all envs still). - agent_binding_resolver.py: resolve_agent_invocation() checks the pinned model against AppRoleService.can_access_model for the invoker (R2 — same gate the harness uses elsewhere), returns a model_override or raises AgentBindingBlockedError. inference-api imports apis.shared only (boundary-safe) - routes.py /invocations: resolve after assistant load, before the KB search; on block, stream a conversational stream_error via stream_conversational_message (D5 block-with-message, no silent downgrade). Override wins at model resolution; agent params sit beneath request params, still flowing through admin bounds/locks - tests: allowed→override, denied→block (checked vs invoker), no-modelConfig→ empty plan (no RBAC call); 57 existing inference/chat tests unchanged Co-Authored-By: Claude Opus 4.8 * feat(agents): Memory-Space hydration helper for prompt injection (Phase 3 PR-B) Shared, sync helper that resolves a memory_space binding's alwaysLoad specs into injectable text fragments — the read side of Workstream B. - resolve_always_load(): MEMORY.md → index; latest:/ → most-recent matching manifest entry (defines that scheme, which had no resolver); bare slug → entry. Missing entries skipped (never fails a turn). Byte-budgeted with a truncation marker pointing at memory_read (MEMORY_INJECTION_MAX_BYTES, ~24KB) - render_memory_block(): delimited system-prompt block; empty for a fresh space - reads go through MemorySpaceService (re-checks viewer+ internally) — no leak - 11 unit tests against a fake service Co-Authored-By: Claude Opus 4.8 * feat(agents): inject bound Memory Space into the prompt, per invoker (Phase 3 PR-C) The Harness now resolves an Agent's memory_space binding against the invoking user and injects the space's alwaysLoad content (read-only) into the system prompt — the first half of the Workstream B / Oliver payoff. - agent_binding_resolver: _resolve_memory() checks the invoker's grant via MemorySpaceService.resolve_permission (D4); blocks (D5) when the flag is off, the space is gone, or a readwrite binding meets a below-editor invoker (no silent read-only downgrade). Returns ResolvedMemoryBinding (v1: first binding) - routes.py: after prompt assembly, hydrate via resolve_always_load (asyncio.to_ thread; MemorySpaceService re-checks viewer+) and append render_memory_block; best-effort — a memory-read hiccup never fails the turn - tests: memory grant matrix (none/flag-off/missing/read-viewer/readwrite-viewer- block/readwrite-editor/invoker-identity); 78 inference+compat tests green Co-Authored-By: Claude Opus 4.8 * fix(agents): rename app_api.agents package to avoid shadowing top-level agents run-app-api.sh launches app-api with `cd src/apis/app_api && python main.py`, putting that directory on sys.path[0]. The new apis/app_api/agents/ package (Agent Designer surface, #591/#592) then shadowed the top-level `agents` package, so `admin/quota/routes.py`'s `from agents.main_agent...` resolved into it and crashed startup with `ModuleNotFoundError: No module named 'agents.main_agent'`. Tests never caught it — pytest runs from backend/ where `agents` resolves correctly. Production is unaffected (the container runs `uvicorn apis.app_api.main:app` from WORKDIR /app, so sys.path[0] is /app). Rename the package apis/app_api/agents → apis/app_api/agent_designer (and its test dir) so its name can't collide with the top-level `agents` package. Pure rename: the /agents URL surface, router, and behavior are unchanged. Verified: `import agents.main_agent.quota.repository` and the full app-api module now load from src/apis/app_api; 34 agent/boundary tests pass. Co-Authored-By: Claude Opus 4.8 * feat(agents): memory_* tools scoped to an Agent's bound Memory Space (Phase 3) Completes the Workstream B write side: an Agent with a memory_space binding now gets memory_list / memory_read (always) and memory_write (readwrite bindings only) at invocation — Oliver can read AND write his space. - agents/builtin_tools/memory_spaces/: closure-scoped factories capturing the binding's space id + invoker identity, MemorySpaceService via asyncio.to_thread (artifact-tools pattern). Every call re-checks the grant inside the service (viewer+ read / editor+ write), so a revoked grant becomes an error tool-result mid-session, never a leak - routes.py _build_memory_tools(): appended to the extra_tools seam only when a memory binding resolved; write tool gated on access==readwrite. extra_tools agents are never cached → tools closed over user A can't be served to user B - not gated on enabled_tools: the governing capability is the Agent's binding, not the user's tool picker (same reasoning as artifact tools) - tests: tool success/permission-error/not-found matrix + seam counts (none=0, read=2, readwrite=3); 84 inference+tool tests green Co-Authored-By: Claude Opus 4.8 * chore(memory): default Memory Spaces ON with a kill switch Memory Spaces is a complete feature (CRUD + SPA panel + agent binding), so it should ship enabled for every deployer/forker — opt-out, not opt-in — matching the kbSync / scheduledRuns convention. The table + bucket are already provisioned unconditionally in PlatformStack, so this only flips the runtime MEMORY_SPACES_ENABLED env var; no new infra footprint. - config.ts: memorySpaces.enabled default true (empty/unset workflow var = on, only literal "false" disables); interface + block docs updated - platform.yml: forward CDK_MEMORY_SPACES_ENABLED (kill switch) and CDK_AGENTS_API_ENABLED (per-env enable for the still-default-off /agents surface, so dev can turn it on for dogfooding without a code change) - app-api-environment.ts: comment reflects default-on - config.test.ts: 5 tests locking default-on + kill-switch + context override Agent Designer (AGENTS_API_ENABLED) stays default OFF until the Phase-4 Designer UI ships — a headless /agents surface helps no forker. Co-Authored-By: Claude Opus 4.8 * feat(agent-designer): Phase 4 — Agent Designer UI + bindable catalog API Ships the Agent Designer authoring surface (Phase 4) and its Phase-2 precursor, the bindable-primitives catalog. The pickers can't exist without the catalog, so both land together. Backend (Phase 2 catalog): - GET /agents/bindable?kind=model|tool|skill|knowledge_base|memory_space returns an RBAC-filtered palette, composing the 5 existing per-primitive access services (D4); no new RBAC invented. Uniform BindableItem shape so every picker consumes one contract. Route declared before /{agent_id} so the literal path isn't captured. knowledge_base → empty (welded/synthesized); skill/memory_space → empty when their feature flag is off. Behind AGENTS_API_ENABLED. - Fix binding_validation._validate_model: it resolved models via get_managed_model() — a primary-key lookup on the internal UUID — but modelConfig.modelId is the Bedrock model_id that the runtime resolver, RBAC (permissions.models) and invocation all key on. A valid model would have been rejected 400 on save the moment the picker set one. Now matches by model_id, consistent with the whole chain. Frontend (Phase 4 UI): - New agents/ feature dir (separate from the Assistants editor): Agent + Binding + BindableItem TS contracts, a thin AgentApiService, and an AgentService signal facade with the accessible$ 404-probe idiom + a per-kind bindable cache. - Agents list page (model + binding-count badges) and an agent-form page: persona/emoji/tags/starters, a required single-select model picker (D3), tool/skill multi-select chips, and a memory-space picker with access (read/read+write — write disabled unless editor+ on the space, per D5) and an alwaysLoad MEMORY.md toggle. KB shown read-only. Sharing reuses the assistants share dialog (agentId == assistantId). - Routes agents / agents/new / agents/:id/edit, plus a sidenav "Agents" entry gated on the accessible$ probe. Tests: backend 1552 pass (9 new catalog + 4 new route + 3 updated model-validation); SPA build + tsc clean, ng test 7 AgentService + 11 sidenav specs pass. Co-Authored-By: Claude Opus 4.8 * test(sidenav): stub AgentService probe to fix unhandled HTTP rejection The sidenav constructor now probes agent accessibility (void agentService.loadAgents()), but sidenav.spec.ts didn't provide a mock AgentService, so the real service fired an unstubbed HTTP GET /agents that rejected with status 0 — vitest fails the run on unhandled errors even though all assertions passed. Provide a mock AgentService (accessible$ signal + no-op loadAgents) mirroring the schedule/memory stubs, plus parity tests for the probe + showAgents gate. Co-Authored-By: Claude Opus 4.8 * fix(agent-designer): align model write-check with the bindable catalog The catalog lists models via ModelAccessService.filter_accessible_models, but design-time write validation used can_access_model — and the two disagree. filter_accessible_models grants access whenever the model id is in the user's AppRole permissions.models; can_access_model only honors that membership when the model record ALSO carries a non-empty allowed_app_roles. So a model granted purely via the user's AppRole (empty allowed_app_roles) was listed by the picker but rejected on save with a 403 — and the runtime resolver (membership-based) would actually have allowed it. Validate the model with the same filter_accessible_models predicate the catalog uses, so 'if the palette offers it, the write accepts it' holds by construction. Adds a regression test for the empty-allowed_app_roles grant. Co-Authored-By: Claude Opus 4.8 * feat(sidenav): gate Memory Spaces + Agents to system-admin, add Preview badges Match the Scheduled Runs treatment for the two other preview surfaces: Memory Spaces and Agents now also require the system_admin AppRole (showX() && isAdmin()) in addition to their accessibility probe. Adds a small amber 'Preview' badge to all three nav entries (Agents, Memory Spaces, Scheduled Runs) so their preview status is visible. Co-Authored-By: Claude Opus 4.8 * feat(agent-designer): resolve tool bindings at invocation (replace + per-invoker RBAC) An Agent's `tool` bindings were stored by the Designer but inert at run time — the free-select tool picker fully drove the toolset regardless of what the Agent bound. This resolves them, mirroring the shipped `modelConfig` override: - Run-time (inference-api): `resolve_agent_invocation` now returns `plan.tools` (`ResolvedTools`). When an Agent binds tools they *replace* the request's `enabled_tools` for the turn; each bound tool is re-checked against the INVOKING user via `AppRoleService.can_access_tool` (the same AppRole gate the harness uses for model, R2) and a missing tool blocks the turn with a message (D5). No tool binding ⇒ `plan.tools is None` ⇒ the request drives the toolset exactly as today. Wired at the existing `extra_tools`/`get_agent` seam via `effective_enabled_tools` (also feeds the spreadsheet/artifact tool gates + attachment guidance/inventory). - Design-time (app-api): `tool` dropped from `_INERT_KINDS`; a bound tool must be in the author's palette (`ToolCatalogService.get_user_accessible_tools`, the same source the picker fetches — "if the palette offers it, the write accepts it", cf. the model check). The palette is resolved once per write. `skill` bindings stay inert here (their run-time fold interacts with agent_type/skill resolution — a follow-up slice). Tests: 6 resolver cases (override, dedupe, block-on-missing, per-invoker, none→passthrough) + 5 validation cases (accessible/inaccessible/empty-ref/fetch-once/lazy). Full backend suite green (4621 passed). Co-Authored-By: Claude Opus 4.8 * feat(topnav): surface active assistant in the top nav Move the assistant/agent indicator out of the chat-input footer and into the top nav, beside the session title, so an attached assistant is visible throughout the conversation. - Add a compact 'variant' to app-assistant-indicator: a subtle name-only pill (emoji + name) that opens the same actions menu (New session / Edit / Share) on click. The full card style is preserved behind variant="card". - Add a menuPlacement input so the actions dropdown opens downward in the top nav instead of clipping off-screen. - Thread the assistant/owner/loading state and action outputs from the chat container into app-topnav; render the pill (with a loading shimmer) to the right of the title. - Remove the now-orphaned footer indicator and loading skeletons from the full-page chat container (embedded preview footer left intact). - Assistant card: move conversation starters into a collapsible accordion (expanded by default) to keep the card compact. Co-Authored-By: Claude Opus 4.8 * feat(agent-designer): resolve skill bindings at invocation (replace + force skill-mode) Completes the tool/skill runtime-resolution gap (tools landed in #601). An Agent's `skill` bindings were stored by the Designer but inert at run time. This resolves them, mirroring the tool/model overrides: - Run-time (inference-api): `resolve_agent_invocation` now returns `plan.skills` (`ResolvedSkills`). When an Agent binds skills they *replace* the request's skills for the turn AND the route forces `agent_type="skill"` so the SkillAgent discloses exactly the bound set. Each bound skill is re-checked against the INVOKING user via `AppRoleService.can_access_skill`; a missing skill — or the Skills feature being disabled in this environment — blocks the turn with a message (D5). No skill binding ⇒ `plan.skills is None` ⇒ the request's agent_type/enabled_skills drive the turn as today. Wired by reassigning `effective_agent_type`/`effective_skill_ids` before the main-turn get_agent, so the values flow into the construction snapshot and a bound-skill agent resumes on the same skills_hash (resume-safe, same mechanism the tool slice relies on). - Design-time (app-api): `skill` dropped from inert (no inert kinds remain). A bound skill is flag-gated (`skills_enabled()`) and must be in the author's palette (`resolve_accessible_skill_ids`, the same source the picker fetches — cf. the tool check); the palette is resolved once per write and only when skills are enabled. Tests: +6 resolver (override, dedupe, flag-off block, block-on-missing, per-invoker, none →passthrough) + 6 validation (accessible/inaccessible/empty-ref/flag-off/fetch-once/lazy). Full backend suite green (4631 passed). Co-Authored-By: Claude Opus 4.8 * feat(agent-designer): reflect governed agent bindings in the chat-input (lock pickers) The backend governs an Agent's model/tool/skill bindings at invocation (#601, #602) — the agent's set wins regardless of what the client sends. The chat-input still showed the model/tool/skill pickers as free-select, which was dishonest (a change the backend ignores). This locks each picker to the active Agent's bindings, per primitive. - Session page (`session.page.ts`): inject AgentService/ToolService/SkillService; fetch the governed Agent alongside the assistant (agentId == assistantId) in `loadAssistant`; apply per-primitive locks from `modelConfig`/`bindings`, and release them when navigating to plain chat. Best-effort: the /agents surface may be disabled (404) or the assistant may be a legacy assistant with no bindings — every failure leaves the pickers free-select. - ModelService/ToolService/SkillService: add a small agent-lock API (`lockToAgent*` / `clearAgentLock` + `agentLocked`/`agentModelLocked`). While locked, `enabledToolIds`/ `enabledSkillIds` return the bound set (replace semantics, matching the backend), toggles no-op, and `isToolShownEnabled`/`isSkillShownEnabled` render the bound set honestly. - UI: model-dropdown shows a locked read-only chip ("set by this agent"); model-settings shows a "Set by agent" model row and a "This agent uses a fixed set of tools/skills" banner, with tool/skill/sub-tool toggles disabled + greyed while locked. This is UI honesty, not enforcement — the backend remains the authority. Per-primitive: an agent that binds a model but no tools locks only the model; the rest stay free-select. Tests: +5 tool-lock, +5 skill-lock, +4 model-lock service specs (ng test, 51 pass); `tsc` clean; production build (AOT template check) clean. Known limitations (documented for follow-up): a model race if the pinned model isn't in the user's loaded set yet (dropdown disables but may show the fallback name until models load); the skill-lock banner only shows in skills chat-mode. Co-Authored-By: Claude Opus 4.8 * fix(agent-designer): release chat-input picker locks on new conversation The agent-binding picker locks live in root singleton services (Model/Tool/Skill Service) that outlive the session component. Clicking "New chat" navigates to `/`, which recreates the session component with fresh assistant()/agent() signals (both null). The lock-release lived inside the `if (loadedAssistant || … || agent())` guard, which is false on that fresh component — so the stale locks from the previous agent conversation were never released, leaving the model + tools pickers stuck. Move `clearAgentBindingLocks()` out of the guard so it always runs when there is no assistant in the URL. Idempotent — a no-op when nothing is locked. Co-Authored-By: Claude Opus 4.8 * feat(agent-designer): show only the bound tools/skills when an agent locks the settings When an Agent dictates a fixed toolset/skillset, the settings panel listed every accessible tool/skill with the bound ones toggled on and the rest greyed off — a long, noisy list. Filter to show ONLY the bound (enabled) tools/skills so the panel reflects exactly what the agent uses. - ToolService.visibleTools / SkillService.visibleSkills: agent-locked → filter to the bound ids; otherwise the full accessible list. - model-settings template iterates the visible* lists. Tests: +1 tool-lock, +1 skill-lock spec (ng test green); tsc + AOT build clean. Co-Authored-By: Claude Opus 4.8 * chore(agent-designer): default AGENTS_API_ENABLED on with a kill switch The Agent Designer is complete (contract → surface → resolution → Designer UI → binding reflection), so flip the feature flag from opt-in to default-on, matching the house style for shipped features (scheduled_runs / memorySpaces). - backend `agents_enabled()`: empty-string-safe default-on — unset/empty ⇒ enabled, only the literal "false" disables (was `== "true"`, default off). - CDK `config.agents.enabled`: mirror the memorySpaces/scheduledRuns ternary (`!== 'false'` + context fallback `?? true`), so an unset/empty GitHub Actions var can't silently disable it. - Tests: add the Agents API default-on/empty/kill-switch/context suite to config.test.ts (mirrors Memory Spaces); rename the app-api-environment threading test (no longer "default off"). The `/agents/*` API now ships everywhere; the SPA nav stays preview-gated (system-admin + "Preview" badge) until Assistants are deprecated, so this doesn't broaden user-facing exposure — it just stops the API 404ing per-environment. Co-Authored-By: Claude Opus 4.8 * feat(agent-designer): manage an agent's knowledge base from the Agent Designer Extract the assistant editor's inline "Knowledge base" section into a standalone, reusable KnowledgeBaseSectionComponent and use it in both the assistant form and the Agent Designer — replacing the agent form's read-only "managed automatically" card with the live document/web-crawl/connector flow. This closes the last Agent migration blocker. The gap was frontend-only: the document upload/ingestion/retrieval pipeline already keys on the record id and agentId == assistantId, so /assistants/{id}/documents backs an agent unchanged. No backend or data-model changes (Option 1, not the deferred F4 first-class KB primitive). The component owns record identity via a createDraft callback so the first content-adding action can mint a draft in create mode; a permissionResolved input gates the edit-only sync-policy calls so a viewer never 403s on the default owner guess. The assistant form keeps createDraftAssistant as its callback (shedding ~1000 lines); the agent form adds createDraftAgent and drops the read-only kbBinding path. Verified: ng build clean, ng test 1449 specs green (incl. assistant-form spec). Co-Authored-By: Claude Opus 4.8 * test(scheduled-runs): freeze dispatcher clock to de-flake cadence rearm test test_next_run_at_uses_schedule_cadence asserted the daily-9am re-arm delta fell in (1h, 48h), which fails when CI runs in the hour before 9am Boise (the next daily run is legitimately <1h away). Freeze dispatcher._now to a fixed instant and assert next_run_at equals compute_next_run_at recomputed from the same instant, making the test time-of-day independent. Co-Authored-By: Claude Opus 4.8 * feat(agent-designer): govern model params + live editor preview Model-params governance: - binding_validation._validate_model_params rejects params that are unsupported / locked / out-of-[min,max] / out-of-allowed against the model's admin supported_params (belt-and-suspenders to the runtime merge; author-facing 400 instead of a silent clamp). +9 tests. - Data-driven Parameters subsection under the model picker reading meta.supportedParams (numeric inputs, enum selects, locked read-only); empty params omit `params` (today's exact resolution). Live side-by-side preview in the agent editor: - New AgentPreviewComponent reuses PreviewChatService and streams the SAVED agent through the real /chat/stream invocation path, so all bindings (model/params/tools/skills/memory) resolve server-side. Capability strip + dirty banner make the resolved context and the save-to-apply semantics explicit. - Agents send a minimal request body (message/session_id/agent id) and opt out of the assistant preview's system_prompt + owner-tools injection, which fought the bindings and blew the 8KB system_prompt cap for long personas (422). PreviewChatService gains a backward- compatible opts flag; assistant preview behavior unchanged. - Two-column editor shell mirroring the assistant editor. Verified: backend 59 pass (9 new), ng build clean, 16 SPA specs (7 agents + 9 preview-chat). Co-Authored-By: Claude Opus 4.8 * feat(agent-designer): lock preview model picker; trim preview nav The Agent Designer preview reused the main chat-input, whose model dropdown reads the root ModelService — so it showed the user's global model (e.g. Sonnet 5) and let them switch it, even though the harness resolves the model from the agent's binding server-side. Wire the preview to lock that picker to the agent's model via the same lockToAgentModel mechanism the session page uses for a real agent conversation, released on destroy (and idempotently on the next plain chat via the session page's self-heal effect). Also hide the Memory Spaces and Scheduled Runs side-nav entries for now (routes/pages and their capability probes are unchanged, so re-enabling is just re-adding the template blocks). Agents stays system-admin only. Co-Authored-By: Claude Opus 4.8 * fix(memory-spaces): route namespaced entry slugs via :path converter Entry slugs are namespaced with a slash (e.g. `people/brian-bolt`), but the app-api entry routes declared a plain `{slug}` param whose converter stops at `/`. Uvicorn percent-decodes `%2F`→`/` before routing, so `/entries/people/ brian-bolt` never matched `/entries/{slug}` and returned 404 on view/edit/delete. Switch the GET/PUT/DELETE entry routes to the `{slug:path}` converter so the embedded slash is captured and the slug arrives matching the manifest. Adds a route test exercising upsert→read→delete with a slashed slug. Co-Authored-By: Claude Opus 4.8 * feat(memory-spaces): let the agent read/write MEMORY.md via reserved slug MEMORY.md is the space's human-readable index — a standalone S3 object outside the entries manifest, injected into the agent's context each session via hydration. It never appears in `memory_list`, and the agent had no tool to read it back or keep it in sync with the entries it writes, so the machine-readable manifest and the human-readable index could silently drift. Route the reserved `"MEMORY.md"` slug (case-insensitive) through the existing service methods: `memory_read("MEMORY.md")` → `read_index` (viewer+), `memory_write("MEMORY.md", body)` → `update_index` (editor+, body only). No new tool surface; matches the literal hydration already uses. The slug is reserved — the agent cannot create an ordinary entry named MEMORY.md. Write stays gated identically to entry writes (only bound when the binding grants readwrite; service re-checks editor+). Docstrings + spec §4/§5 updated. Co-Authored-By: Claude Opus 4.8 * feat(schedules): target Agents instead of Assistants on scheduled runs The scheduled-run form's target selector now lists Agents (the Agent Designer primitive that supersedes the Assistant) instead of Assistants. Same underlying record — agentId == assistantId — so the wire field stays `assistantId` and no backend change is needed for the swap. Because an Agent's `tool` bindings replace the run's `enabled_tools` at invocation (agent_binding_resolver / routes.py effective_enabled_tools), the manual tool picker is now hidden whenever an Agent is selected — showing it would let the user pick tools that get silently discarded. The picker (and its snapshot semantics) remains only for the "Default agent" case. Submit drops any stale snapshot when an Agent is targeted. Also fixes "Run now" to target the selected Agent via ragAssistantId (the /runs/now backend already accepts it) — previously it ignored the target, so the attended test surface didn't match what the schedule would actually run. Co-Authored-By: Claude Opus 4.8 * chore(kaizen): weekly research scan 2026-07-10 Generated by the kaizen-research skill. Top 5 ideas appended to docs/kaizen/review-queue.md for the kaizen-review-prep run later this morning. Co-Authored-By: Claude Opus 4.8 (1M context) * chore(deps): upgrade Strands to 1.47.0 and add aws-bedrock-token-generator Bumps strands-agents 1.40.0 -> 1.47.0 (and the [bidi] extra to match) and adds aws-bedrock-token-generator==1.1.0 (bounded >=1.1.0,<2.0.0 by strands' openai extra). strands-agents-tools stays at 0.5.2 (resolver-confirmed compatible). Unblocks Bedrock Mantle work that needs the newer SDK: - OpenAIResponsesModel (Responses API) for models that don't support Chat Completions (e.g. openai.gpt-5.x on Mantle). - bedrock_mantle_config, which mints the Mantle bearer token via aws-bedrock-token-generator and derives the base URL + model-family base path (openai.gpt-5.* -> /openai/v1, else -> /v1). Full backend suite green on 1.47.0 (2306 passed). Co-Authored-By: Claude Opus 4.8 * fix(scheduled-runs): remove RBAC gate causing prod 403 "Access Denied" Regular users hit a 403 "You do not have access to scheduled runs" toast on page load. The `/schedules` and `/runs/*` surfaces were gated by the `scheduled-runs` RBAC capability, granted only to a beta cohort's AppRole (admins passed via the `*` wildcard). The sidenav ran a background `loadSchedules()` probe on every load, and the global errorInterceptor popped the toast on the 403 before the schedule service's graceful catch ran. The feature doesn't need admin/beta gating — keep it low-key and reachable only by direct URL for now: - Drop the capability check from both `require_scheduled_runs_user` gates; only the `SCHEDULED_RUNS_ENABLED` kill switch remains (404 when off). Runs still execute with the caller's own RBAC-allowed tools, so this widens who can reach the surface, not what any one caller can do. - Remove the vestigial sidenav schedules probe and dead showSchedules/navigateToSchedules wiring (the template never rendered a "Scheduled runs" link). - Update route + sidenav tests accordingly. `apis/shared/rbac/capabilities.py` is now unreferenced; left in place as generic RBAC infra so re-gating is a two-line revert. Co-Authored-By: Claude Opus 4.8 * docs: consolidate release workflow into one auto-invoked steering doc + skill Fold the versioning and release-notes guidance into a single 'cutting a release' guide covering the branch workflow, SemVer bump + version sync, change identification across the divergent main/develop histories, writing both release docs, the squash-merge PR into main, and the required backmerge into develop. - Add .kiro/steering/cutting-a-release.md (inclusion: auto — name + description, intent-triggered) - Add .claude/skills/cutting-a-release/ (SKILL.md auto-invoked via description) with references/{release-notes-format,changelog-format}.md for progressive disclosure - Remove superseded .kiro/steering/{versioning,release-notes}.md and .claude/skills/{versioning,release-notes}/ - Repoint .github/copilot-instructions.md at the consolidated skill/steering * feat(models): Mantle Responses API + per-model region; drop endpoint-path knob Refactors the admin "mantle" provider onto Strands' bedrock_mantle_config so the SDK owns the base URL, model-family base path, and bearer-token minting — removing hand-rolled inference plumbing. Adds the two things the library can't infer as declarative per-model fields: - apiMode (chat | responses): selects OpenAIModel vs OpenAIResponsesModel. Some Mantle models (e.g. openai.gpt-5.x) only serve the Responses API and reject Chat Completions, which the endpoint-path knob could never satisfy. - region: optional override into bedrock_mantle_config["region"], driving both the Mantle endpoint host and the SigV4 region the token is signed for — so a model can pin inference to its host region (e.g. gpt-5.x in us-east-1) independent of where the app runs. mantleEndpointPath is kept as an accept-but-ignore deprecated schema field (no stored record breaks) and removed from the UI + runtime. The Responses API uses different native param names, so to_mantle_config selects a Responses map (max_output_tokens, nested reasoning.effort) by mode. Runtime fields (mantle_api_mode/mantle_region) thread through model_config, the agent factory, base_agent, the paused-turn snapshot, stream_coordinator, and the chat service/routes. get_mantle_base_url/generate_bedrock_bearer_token are retained for the admin model-browse list (not inference). Gemma 4 (google.gemma-4-31b) is temporarily un-curated: it needs the /openai/v1 base path but the SDK only routes openai.gpt-5.* there, and bedrock_mantle_config forbids a base_url override. Re-add once the "google.gemma-" family prefix lands upstream in strands-agents/sdk-python. Backend suite green (2342). Frontend typecheck + manage-models specs green. Co-Authored-By: Claude Opus 4.8 * fix(api-converse): serve /chat/api-converse from app-api, not via inference proxy The API-key converse endpoint was broken in cloud. app-api proxied POST /chat/api-converse to `{INFERENCE_API_URL}/chat/api-converse`, but inference-api now runs inside an AgentCore Runtime whose data plane only serves POST /invocations and GET /ping — any other path returns UnknownOperationException (404) before reaching the container. It worked locally only because localhost:8001 bypasses the runtime gateway. Relocate the handler onto app-api as a self-contained route (validate key -> RBAC -> bedrock-runtime.converse -> cost accounting), reusing the shared services it already depends on. app-api reaches Bedrock directly via its task role, so there is no inference-api hop and no INFERENCE_API_URL dependency. Delete the proxy, the now-dead inference-api route, and its DTOs (moved to app_api/chat/models.py). Repoint the converse tests at the app-api module. Verified: 263 backend tests pass, import-boundary test clean, and a real un-mocked smoke against a Bedrock model returns 200 (stream + non-stream). Co-Authored-By: Claude Opus 4.8 * fix(app-api): grant Bedrock streaming + inference-profile invoke The relocated /chat/api-converse handler calls Bedrock Converse from app-api, so the task role's invoke grant must cover what the catalog's model IDs need. Expand the BedrockInvokeModel statement to add bedrock:InvokeModelWithResponseStream (the stream=true path) and broaden resources to all-region foundation models plus the account-level inference-profile ARN, since the catalog uses `us.*` cross-region inference profiles. Mirrors inference-api's BedrockModelInvocation grant. Verified: infra tsc clean, 442 infra jest tests pass. Co-Authored-By: Claude Opus 4.8 * fix(settings): point API-key snippets at /api/chat/api-converse After the BFF refactor, CloudFront only routes /api/* to the backend; other paths hit the SPA origin, which rejects POST with a CloudFront 403. The generated curl/Python/JS examples emitted the bare origin, producing `/chat/api-converse`. Resolve a relative/empty appApiUrl against the current origin so snippets target `/api/chat/api-converse`; leave an already-absolute value (local dev's http://localhost:8000) untouched. Co-Authored-By: Claude Opus 4.8 * feat(api-converse): route Bedrock Mantle models via a shared builder The API-key /chat/api-converse handler was Bedrock-only; provider="mantle" models (e.g. openai.gpt-5.4) 400'd because it always called bedrock-runtime.converse. Add a Mantle path so the full model catalog works. Extract the Mantle model construction (class-pick + bedrock_mantle_config) and its param maps + MantleApiMode enum out of agents/main_agent/core into a new apis/shared/models/mantle.py, so the agent factory and the API-key handler share ONE implementation (app-api can't import agents/). The factory now delegates to build_mantle_model. The handler resolves the requested model's provider from the catalog and branches: bedrock -> boto3 converse (unchanged); mantle -> the shared builder + the bare Strands model's .stream(), which yields the same Converse-shaped events the Bedrock path already emits — so SSE translation and usage/cost accounting are shared (cost is now tagged with the real provider). Unknown / lookup-failure ids fail safe to the Bedrock path. Verified: shared builder + factory-delegation + handler mantle-path unit tests; and real dev-ai smokes — chat-mode Mantle and Responses-API Mantle (openai.gpt-5.4) both return 200 (stream + non-stream) against the live endpoint. Co-Authored-By: Claude Opus 4.8 * feat(app-api): grant bedrock-mantle:CreateInference for api-converse The api-converse Mantle path invokes a Mantle model directly from app-api, so the task role needs bedrock-mantle:CreateInference (Mantle's own IAM namespace) — without it, mantle requests AccessDeny. Fold it into the existing project-scoped Mantle statement (was browse-only Get*/List*), renamed BedrockMantleInference to mirror the runtime role's grant. Verified: infra tsc clean, integration jest green (24 passed). Co-Authored-By: Claude Opus 4.8 * feat(identity): MCP user identity forwarding via access-token enrichment Add an opt-in Cognito Pre-Token-Generation v2 Lambda that copies configured user-pool attributes into namespaced claims on the ACCESS token, so personalized MCP tools can identify the caller. The access token is the only token forwarded end-to-end to MCP servers, so enrichment needs no changes to the SPA -> app-api -> inference-api -> MCP forwarding path. Shipped disabled by default (opt-in): a fork that configures nothing gets zero resources and the token is forwarded as before. Enabling requires the Cognito Essentials feature plan (pinned on the pool) plus two GitHub Actions variables (CDK_MCP_TOKEN_ENRICHMENT_ENABLED + CDK_MCP_TOKEN_ENRICHMENT_CLAIMS), keeping the committed cdk.context.json inert. - config: McpIdentityConfig (enabled + accessTokenClaims); claim map settable via JSON env var or context; parseJsonRecordEnv helper. - handler: stdlib-only, fail-open Pre-Token-Gen v2 trigger (returns event unchanged on any error so login is never blocked). - construct: real-code Lambda (fromAsset) attached via addTrigger V2_0; pool featurePlan pinned to ESSENTIALS. - wired conditionally into PlatformStack; platform.yml job-level env. - docs: spec updated (open questions resolved) + implementation summary, incl. the mcp-servers follow-on handoff. Ref: docs/specs/MCP_USER_IDENTITY_FORWARDING_SPEC.md * fix(app-api): grant bedrock-agentcore:CreateTokenVault for OAuth provider create Admin "add OAuth provider" (POST /admin/oauth-providers/) returned a 502 Bad Gateway. dev-ai app-api logs showed the real cause: an AccessDeniedException on bedrock-agentcore:CreateTokenVault against token-vault/default. AgentCore's CreateOauth2CredentialProvider ensures the default token vault exists on the first provider create, which requires CreateTokenVault (+ GetTokenVault) on the caller. The app-api task role had the ...Oauth2CredentialProvider actions but not the TokenVault ones. The shared error handler maps an uncaught AWS ClientError to HTTP 502, so the missing permission surfaced as a 502 rather than a 403. Add CreateTokenVault + GetTokenVault to the AgentCoreWorkloadIdentityAccess statement. The resource scope (token-vault/*) already covered token-vault/default; only the actions were missing. Requires a platform.yml (CDK) redeploy to take effect. Co-Authored-By: Claude Opus 4.8 * fix(scripts): make sync-version.sh portable across GNU and BSD tools The version-sync script only ran inside the dev container / CI (GNU coreutils); on macOS (BSD sed/grep) it errored out and silently left the manifests un-synced, so a release cut locally had to hand-edit every manifest. Replace the three GNU-only constructs with POSIX equivalents: - `grep -oP ... \K` (Perl regex) -> `sed -n 's/.../\1/p'` / awk field split - `sed -i "expr"` (GNU in-place) -> sed_inplace helper (temp file + mv) - `sed "0,/re/s/..."` (GNU-only address) -> awk first-match replace Behavior is unchanged on GNU; the script now runs identically on macOS. Verified both --check and the write path (incl. shields.io `--` hyphen doubling and SemVer->PEP 440 lock conversion) round-trip on BSD tools. Co-Authored-By: Claude Opus 4.8 * feat(admin): make admin sidebar nav sticky on desktop Pin the admin layout aside below the sticky top bar so the section nav stays in view while the content area scrolls. Uses lg:self-start so the aside shrinks to its content (flex items stretch to full height by default, which defeats position:sticky), plus a max-height + overflow so a long nav scrolls internally. Mobile dropdown is untouched. Co-Authored-By: Claude Opus 4.8 * feat(frontend): redesign 404 page to match auth screens Rework the not-found page onto the same design system as the login and first-boot pages: the primary-derived lava-lamp parallax backdrop (six depth-tiered morphing blobs), the masked graph-paper grid overlay, and the frosted-glass card. The oversized 404 sits above the card where the auth pages place the logo, so all three screens read as one system. Preserves existing behavior (sidenav hide/show, Return Home, Go Back) and respects prefers-reduced-motion. Classes are nf-prefixed and component-scoped via view encapsulation. Co-Authored-By: Claude Opus 4.8 * fix(frontend): make shell scroll container real so sticky nav engages The admin aside's lg:sticky never engaged because its nearest scrolling ancestor was the app shell's `flex-1 overflow-y-auto` div, which had no bounded height — it grew to content and the window scrolled instead, so sticky bound to a box that never moved. Pin
to h-dvh so that div becomes a genuine scroll container; the admin aside and top bar now stick. Also apply the admin bar's frosted-glass treatment (bg-*/opacity + backdrop-blur-sm) to the session topnav so the two surfaces match. Co-Authored-By: Claude Opus 4.8 * docs(specs): quota cooldown windows + platform ceiling spec and committee one-pager Replaces the hard monthly quota cutoff with a three-layer model: anchored 5-hour cooldown windows (Claude-style, exact reset times), a hard admin-adjustable platform-wide monthly ceiling as the fiscal guarantee, and the per-user monthly limit demoted to a generous anti-runaway backstop with degrade-to-economy-model as the target behavior. Backstop horizon (monthly vs weekly) is a per-tier choice. Includes an admin pilot tuning playbook with an observe-only phase, a user-facing quota status endpoint, recommended opening numbers, and a 7-PR implementation breakdown. The one-pager is the committee-facing rationale. Co-Authored-By: Claude Fable 5 * fix(frontend): guarantee JIT compiler in vitest runs to stop PlatformLocation flake The unit-test builder keeps Angular packages external, so vitest evaluates raw fesm2022 chunks whose partial declarations (ɵɵngDeclareInjectable/ ɵɵngDeclareFactory) compile eagerly and require @angular/compiler. Its presence was incidental — loaded transitively via @angular/core/testing in the builder's init-testbed setup — so specs with no static Angular imports (app.spec.ts dynamic-imports './app') could evaluate an unlinked @angular/common chunk first and fail with "The injectable 'PlatformLocation' needs to be compiled using the JIT compiler, but '@angular/compiler' is not available" (angular/angular-cli#31993). - add src/test-setup.ts importing @angular/compiler, wired via the test target's setupFiles and included in tsconfig.spec.json - add src/test-setup.spec.ts guarding the invariant deterministically - bump the first shared-view.page spec to 15s: it pays the one-time dynamic page-chunk import, which can exceed 5s under full-suite load Co-Authored-By: Claude Fable 5 * fix(frontend): size chat scroll space to the response, adapt to shell scroll container Replace the fixed viewport-tall bottom spacer in the message list with a min-height on the last turn group (user message + its assistant responses). The response streams into the reserved space instead of pushing a static spacer further down: a short response leaves exactly the room needed to pin the user message at the top, and a response taller than the viewport leaves zero dead scroll below it. Turn groups are keyed by their first message id so a finished turn's DOM (including live MCP App iframes) never remounts when the next turn starts, and the end-of-conversation sections (loader, consent/approval prompts, compaction, orphan artifacts) render inside the reserved space so they stay visible next to the response. Also adapt the session page to the real shell scroll container introduced by #634 (frosted sticky nav): the window no longer scrolls, which had silently broken submit scroll-to-message and scroll save/restore. scrollToMessage now uses scrollIntoView with a scroll-mt-20 header offset, and save/restore reads the shell container's scrollTop via a stable #app-scroll-container hook. Co-Authored-By: Claude Fable 5 * feat(settings): make user settings sidebar nav sticky on desktop Mirror the admin layout change (#632): pin the settings aside below the sticky top bar so the section nav stays in view while the content area scrolls. Uses lg:self-start so the aside shrinks to its content (grid items stretch to full row height by default, which defeats position:sticky), plus a max-height + overflow so a long nav scrolls internally. Mobile dropdown is untouched. Co-Authored-By: Claude Opus 4.8 * feat(admin-tools): discover OAuth-gated MCP servers with the admin's vaulted token The admin tool "Discover" flow refused OAuth-gated MCP servers outright, so servers like the GitHub remote MCP server (api.githubcopilot.com/mcp/) could not be discovered — discovery either 400'd on auth_type=oauth2 or connected unauthenticated and got a 401 from the server (wrapped to a 400). Discovery now accepts the OAuth provider id and connects using the admin's own vaulted 3LO token for that provider, fetched via AgentCore Identity (get_token_for_user) and injected as a bearer — mirroring how the agent loop attaches the end-user's provider token at runtime, and reusing the exact path connector_status already uses. This validates the admin's own connection and lists the tools their token can see (providers such as GitHub scope-filter the tool list to the token's grants). It fetches the admin's token only; it cannot mint an arbitrary end-user's token. Backend: - Add requires_oauth_provider (alias requiresOauthProvider) to MCPDiscoverRequest. - Handler loads the provider, fetches the admin's vaulted token, injects it as oauth_token into create_external_mcp_client. requires_consent -> 409, unknown provider / conflict with forward_auth / oauth2-without-provider -> 400. Frontend: - Send requiresOauthProvider in the discover payload (the form control already existed) and the OAuth2CallbackUrl header (bare /oauth-complete, no query string) so the backend can resolve the admin's token. Tests: 5 backend tests for the OAuth-provider discovery path; 2 SPA specs for the discover payload. Co-Authored-By: Claude Opus 4.8 * feat(manage-models): add Sonnet 5 + GPT-5.4 curated cards, order by capability Add two curated model catalog cards: - Claude Sonnet 5 (bedrock, global.anthropic.claude-sonnet-5) — 1M context, effort-based reasoning, caching on. - GPT-5.4 (mantle, openai.gpt-5.4) — Responses API surface; the openai.gpt-5.* model id matches the SDK's /openai/v1 routing prefixes, so one-click create routes correctly (unlike the commented-out Gemma card). Order the Bedrock Claude cards most-capable-first (Opus 4.7, Sonnet 5, Sonnet 4.6, Haiku 4.5) and place GPT-5.4 ahead of Qwen in the Mantle list. Move the "Bedrock Mantle" provider tab next to "Bedrock" in the catalog selector. Co-Authored-By: Claude Opus 4.8 * fix(mantle): route google.gemma-4-* to /openai/v1 base path Gemma 4 is served ONLY on Mantle's /openai/v1 path (per its AWS model card), but the Strands SDK's _OPENAI_PATH_MODEL_PREFIXES ships only "openai.gpt-5.", so google.gemma-4-* fell through to /v1 and inference 401'd with access_denied ("... is not enabled for this account"). Append "google.gemma-4-" to the SDK's prefix table at build time (_ensure_gemma4_openai_v1_routing: lazy, idempotent, guarded) until it lands upstream. Scoped to the 4.x family — Gemma 3 stays on /v1. - Guard tests: prefix registers on build, all three Gemma 4 variants resolve to /openai/v1, Gemma 3 stays on /v1, registration idempotent. - Correct the stale curated-models.ts note (the "would fail at chat time" claim is obsolete; its "google.gemma-" re-add hint would have misrouted Gemma 3). - Add design note proposing mantleEndpointPath as a live admin setting as the durable alternative to chasing the SDK's hardcoded table. Co-Authored-By: Claude Opus 4.8 * feat(manage-models): make max output tokens optional Newer reasoning / Responses-API models (GPT-5.x, Claude with adaptive thinking) don't publish a discrete max-output-tokens value — output shares the context budget with reasoning tokens, so there's no fixed cap to enter. Our own GPT-5.4 curated card already carries a decorative value with no backing max_tokens spec. maxOutputTokens is only a ceiling for the admin-configured max_tokens inference param and is never sent to the provider, so leaving it unset is safe at inference time. This makes the admin form field optional to match. - ManagedModelCreate / ManagedModel: max_output_tokens -> Optional[int] - DynamoDB write: omit maxOutputTokens when absent (matches other optionals) - Form control: drop Validators.required, default null (number | null); the 0-default + min(1) combo would otherwise still block submit - SPA interfaces typed number | null; catalog card null-guarded (shows "— out") - Both ceiling validators already skipped an absent value — no change needed Co-Authored-By: Claude Opus 4.8 * fix(docker): float curl security patch to survive Debian mirror purges Debian removes the superseded point version of curl from the trixie mirror on each security update, so an exact +deb13uN pin breaks every build once the next CVE lands. Pin to +deb13u* to track the live patch while keeping the minor version fixed; the digest-pinned base image is what actually provides reproducibility. Co-Authored-By: Claude Opus 4.8 * feat(web-sources): allow removing a web source A web source could be added but never removed. There was no DELETE route, no client method, and no UI affordance — the only way to drop one was to delete every page document it produced and let the orphan cascade in cleanup_service pick up the crawl row as a side effect. Add the operation as a first-class one, inverting that existing cascade: DELETE /assistants/{id}/web-sources/crawls/{crawl_id} removes the crawl's sync policy, soft-deletes every page under its root URL (vectors and S3 teardown hand off to the same background cleanup the single-document path uses), then hard-deletes the crawl row. A crawl that is genuinely in flight is refused with a 409 rather than raced — the crawler would keep writing pages we just enumerated. A crawl stuck at 'running' because its process died is not in flight and stays deletable, so a zombie source can't become permanently undeletable. The route is edit-gated (owner or editor), matching the documents surface that renders the list. Co-Authored-By: Claude Opus 4.8 * fix(web-sources): let editors start and view crawls start_crawl, list_crawls and get_crawl gated on the owner-keyed get_assistant(), which returns None for a user holding only an editor share — so an editor got a 404 from the "Add web content" button the SPA already renders for them (canManageSync() shows it to anyone who isn't a viewer). Route them through the same _require_edit_permission helper the documents, sync-policies and delete-crawl surfaces use, so owner|editor is the gate and a viewer gets a 403 instead of a misleading 404. No owner_id threading is needed on these three: the document writes are keyed on the assistant (PK=AST#), not its owner, and imported_by_user_id/started_by_user_id intentionally record the *acting* user — substituting the owner there would credit an editor's import to the owner. owner_id stays confined to the delete path, whose soft_delete_document/_list_crawl_pages calls really are owner-keyed. Co-Authored-By: Claude Opus 4.8 * fix(rbac): make AppRole the single source of truth for model access The model admin page and the role admin page wrote to two different, unlinked fields. Enabling a model for a role on the model page wrote `allowedAppRoles` onto the model record — a field no access check ever read — so the grant silently did nothing: the role page still showed the model unchecked, and users never saw it in the chat picker. Only editing the role's `grantedModels` had any effect. Make the role record the single source of truth, matching the pattern tools and skills already use: - The model form's role picker now writes THROUGH to each selected role's `grantedModels` (new ModelRoleService.set_roles_for_model), mirroring set_roles_for_tool. Create/update/delete routes wire it up, migrating grants on a modelId rename and revoking them on delete. - `allowedAppRoles` is no longer persisted on the model; it is derived from the role records on read (hydrate_model_roles), so the model page and role page can no longer disagree. Adds `inheritedAppRoles` for wildcard/inherited grants, surfaced read-only in the form. - can_access_model and filter_accessible_models both delegate to one `_grants_access` predicate. They previously diverged (one gated on allowed_app_roles, one didn't), so a model could be listed by the catalog yet denied on use. - Removes the dead POST /sync-roles endpoint (never called; only existed to paper over the drift); replaces it with GET /managed-models/{id}/roles. - Drops Validators.required on the picker, since a model reachable only via a wildcard grant legitimately has zero direct grants. Adds regression coverage for the write-through, the derived read, and the two access checks agreeing. Full backend + frontend suites green (the 8 pre-existing get_metadata_storage failures are unrelated). Co-Authored-By: Claude Opus 4.8 * fix(tests): repoint storage patch target and harden integration gate Two pre-existing failures on develop, both unrelated to the code under test: - test_cache_savings.py patched apis.app_api.storage.get_metadata_storage, but that accessor moved to apis.shared.storage (the app_api.storage module is now an empty stub). Repoint all 5 patch targets. Production code in sessions/services/metadata.py already imports from the new location. - test_compaction_integration.py gated its real-AWS integration tests on AGENTCORE_MEMORY_ID. That variable leaks into the process mid-suite when other tests reload apis.app_api.main (load_dotenv(override=True) injects a local backend/src/.env), so the tests ran order-dependently against invalid credentials instead of skipping. Gate on an explicit RUN_AGENTCORE_INTEGRATION_TESTS=1 opt-in instead. Full suite: 4771 passed, 6 skipped. Co-Authored-By: Claude Opus 4.8 * fix(chat): keep SSE stream open across tab switches @microsoft/fetch-event-source defaults to openWhenHidden:false, which aborts the SSE connection on visibilitychange-to-hidden and reopens it — issuing a fresh POST /invocations for the SAME turn — when the tab becomes visible again. That reopen happens inside the library, reusing the request and bypassing the SPA's per-session double-submit and streamId supersession guards. Because a client abort does not propagate through the AgentCore Runtime data plane, the original backend agent keeps running while the reopened one runs the same turn concurrently. Both persist tool-use/tool-result events to the same AgentCore Memory session, corrupting history with duplicate / interleaved toolResult turns and bricking the conversation with a Bedrock "toolResult blocks exceed toolUse blocks" ValidationException. Set openWhenHidden:true on both fetchEventSource call sites so a single stream stays alive across tab switches (also correct for long agentic turns). The server-side restore-time repair is the safety net for already -corrupted histories. Co-Authored-By: Claude Opus 4.8 * fix(sessions): repair tool-use/tool-result pairing on restore Bedrock Converse rejects any history where a user turn's toolResult blocks do not exactly match the preceding assistant turn's toolUse blocks ("The number of toolResult blocks at messages.N exceeds the number of toolUse blocks of previous turn"). A single such violation anywhere in a session's persisted history makes every subsequent turn fail, permanently bricking the conversation. Such corruption can be written by concurrent/interrupted turns with parallel tool calls (e.g. a duplicate invocation spawned by a tab switch): duplicate toolResult turns, toolResults reordered away from their toolUse turn (assistant/assistant/user/user), or toolResults orphaned after a synthetic error turn. The SDK's own _fix_broken_tool_use only rebuilds the single message after each toolUse turn, so it does not repair these shapes. Add TurnBasedSessionManager._repair_tool_pairing, an unconditional restore-time normalizer (sibling to _strip_document_bytes) that rebuilds a Bedrock-valid history on the final agent.messages: every toolUse turn is immediately followed by exactly one matching result turn (missing ones synthesized as errors), duplicate/orphaned result turns are dropped, and consecutive same-role turns are merged. No-op (identity) on healthy history. Kill switch: AGENTCORE_MEMORY_HISTORY_REPAIR_ENABLED=false. Validated against a real bricked production history (24 violations -> 0, idempotent). Self-heals affected sessions on their next turn. Co-Authored-By: Claude Opus 4.8 * test(sessions): make compaction fixtures valid Converse histories Two pre-existing compaction tests fed the session manager a user toolResult turn with no preceding assistant toolUse (make_tool_result_message alone) — an invalid Converse history that Bedrock would also reject. The new restore-time _repair_tool_pairing correctly drops/merges those orphaned turns, changing the message counts the tests asserted. Give each fixture a matching toolUse turn before the toolResult so the repair no-ops and the tests exercise compaction/truncation and checkpoint slicing in isolation. Counts updated accordingly (4->5 kept; slice 2->3). Co-Authored-By: Claude Opus 4.8 * fix(sessions): guard synthetic error persistence against role-alternation breaks When a turn errors inside the agent stream, the handler persists a synthetic "⚠️ Something went wrong" assistant turn. If the last persisted message was already an assistant turn (a dangling assistant toolUse, or a prior synthetic error turn), this appends a second consecutive assistant message, breaking Bedrock's strict user/assistant alternation. The next turn then fails and persists yet another assistant error turn — an amplifier that turns one bad turn into a permanently bricked session. Add a centralized role-alternation guard in persist_synthetic_messages via a new last_persisted_role param: any synthetic turn that would land adjacent to a same-role turn is dropped (the error stays a live-only UI affordance, the same choice the max_tokens path already makes). Callers in stream_coordinator pass the history tail role via a new _last_persisted_role(agent) helper. Complements PR #653's restore-time _repair_tool_pairing, which masks this on the model-request path; this fixes the write side so storage and the message display stay clean too. Co-Authored-By: Claude Opus 4.8 * fix(chat): reject duplicate concurrent turns with a per-session single-flight lease A client-side abort (Stop, tab switch, dropped socket, retry) does not propagate through the AgentCore Runtime data plane, and the Runtime can route a duplicate POST /invocations to a different container. Two agent loops then run concurrently against one AgentCore Memory session and corrupt tool-pairing history, which Bedrock Converse rejects on every subsequent turn ("toolResult blocks exceed toolUse blocks"). This bricked prod session f761f59b. Follow-up to PR #653, which closed the frontend tab-switch vector. Add a distributed single-flight guard at the inference-api /invocations turn-start chokepoint: - session_lease.py: acquire/renew/release on a dedicated sessions-metadata item (PK=USER#{uid}, SK=LEASE#{sid}) via an atomic conditional write. leaseExpiresAt is the app-level check; ttl is a coarse auto-reap backstop. Owner-scoped renew and release. Fail-open on any non-conflict DynamoDB error. - routes.py: acquire at turn-start; reject a duplicate with 409. Resume / max-tokens continuation re-enter an already-ended loop, so they acquire with force=True (never blocked, still install a lease). Heartbeat renews the lease while the turn streams; release in the generator finally + both except handlers. Preview / no-DynamoDB paths skip the guard. - SPA: handle the 409 as a soft "Already responding" notice (AlreadyStreamingError) instead of a hard "Chat Request Failed" toast; unwrap the BFF's double-encoded detail; loading clears so the user can retry once the prior turn finishes. Design note + distributed-cancel follow-on: docs/specs/session-single-flight-guard.md Co-Authored-By: Claude Opus 4.8 * feat(chat): distributed turn cancellation — make Stop actually stop the server turn Follow-on to the single-flight lease (#655). A client abort doesn't propagate through the AgentCore Runtime data plane, so Stop was cosmetic server-side: the container ran to completion, held the lease, and burned model/tool spend. That left "Stop → resend" returning 409 until the prior turn finished naturally. Reuse the lease as the cross-container signalling channel: - Signal: the app-api user_stopped endpoint calls request_session_cancel, which stamps cancelRequestedFor= on the lease item (owner-scoped, so a stale Stop can't kill a later turn). Best-effort — never fails the Stop. - Observe: the inference-api lease heartbeat (tightened 30s→10s) renews with ReturnValues=ALL_NEW and, on cancelRequestedFor==owner, flips session_manager.cancelled. - Effect A (tools): the always-on StopHook cancels the next tool call. - Effect B (model stream): a cooperative check at the top of the StreamCoordinator loop raises _CooperativeStopSignal; a dedicated arm persists the partial via _persist_interruption (marked user_stopped), emits terminal SSE frames, and ends cleanly (no re-raise) so a still-connected client closes and the lease releases. This is what ends a pure-chat turn, which has no tool boundary for StopHook. - acquire clears any stale cancel marker (REMOVE) on takeover. Net: Stop ends the server turn; the 409-on-resend window shrinks from a full turn to ~one heartbeat (10s), and wasted spend after Stop is halted. Rides the existing interrupted-turn teardown/persist path (hardened by #653's _repair_tool_pairing), so stopping mid-stream never orphans or corrupts history. Residual (documented): in-flight tool calls finish before cancel is seen; already- generated Bedrock tokens are billed. Design note: docs/specs/session-single-flight-guard.md Co-Authored-By: Claude Opus 4.8 * fix(infra): grant app-api task role access to shared-conversations table The shared-conversations DynamoDB table was threaded into the app-api container as an env var (SHARED_CONVERSATIONS_TABLE_NAME) but never granted on the task role. Every conversation-share operation therefore failed against DynamoDB: - POST /conversations/{id}/share -> PutItem AccessDeniedException - GET /conversations/{id}/shares -> Query AccessDeniedException on the SessionShareIndex GSI Both surfaced to users as a generic 500 "Failed to create share". Add `SharedConversationsAccess` to the app-api coreTables grant list so the role gets the standard DynamoDB action set on the table and its GSIs (index/*), matching every other table the app-api touches. Also add a regression test that synthesizes PlatformStack and asserts the app-api role has a SharedConversationsAccess statement granting PutItem/Query/GetItem with a GSI resource and no wildcard. Verified the test fails without the grant. Co-Authored-By: Claude Opus 4.8 * feat(shares): offload large-conversation snapshots to S3 Sharing a large conversation failed: ShareService.create_share inlined the full message list into a single DynamoDB item, exceeding the 400 KB item limit and surfacing to users as a bare 500 (observed in prod-ai as a PutItem ValidationException). This is separate from the IAM-grant bug in PR #657. Offload the snapshot body (messages + metadata) to a new private shared-conversations S3 bucket, keeping only control fields plus a body_ref pointer in DynamoDB — mirroring the Memory Spaces / Artifacts / Skills S3-offload pattern. Reads fall back to inline for legacy shares, so existing shares keep working with no migration and the SPA contract is unchanged. - New ShareSnapshotStore (content-addressed S3 put/get/delete, SSE-S3, dedupe) - create_share writes body to S3 + body_ref item; revoke/session-cleanup best-effort delete the object - _load_snapshot_body reads from S3 or falls back to legacy inline items - ShareStorageUnavailableError -> friendly 503 instead of a bare 500 - CDK: shared-conversations bucket + SSM param, compute-ref, app-api env (SHARED_CONVERSATIONS_BUCKET_NAME), and SharedConversationsBucketReadWrite IAM grant (app-api only) - Tests: store round-trip/dedupe, >400 KB regression, S3 + legacy reads, export-from-S3, revoke cleanup, storage-unavailable Spec: docs/specs/share-large-conversations-s3-offload.md Co-Authored-By: Claude Opus 4.8 * fix(agents): resolve model provider for agent-bound invocations Agent (assistant) model bindings persist only `model_id` — never `provider` — so previewing/invoking an agent bound to a Mantle model (e.g. `openai.gpt-5.4`) resolved to provider=None. That misroutes the model to Bedrock ConverseStream, which rejects it with "The provided model identifier is invalid", even though the same model works from the normal chat path (which always sends `provider` alongside `model_id`). Two complementary fixes: - Backend (server-authoritative): `_resolve_model_settings` now also returns the model's registered `provider` from the managed-model registry, and the invocation path backfills `effective_provider` from it when the request/binding didn't carry one. This fixes all existing agents with a provider-less stored binding — no data backfill needed — and mirrors how `mantle_api_mode`/`mantle_region` are already recovered. The app-tool-call / app-context-update rebuild paths get the same fallback so a rebuilt agent keys on the same provider as its main turn. - Frontend: the Agent Designer save payload now persists the selected model's `provider` (from the catalog `meta.provider`) alongside `modelId`, so newly created/edited bindings are self-describing. Co-Authored-By: Claude Opus 4.8 * fix(inference): bind effective_enabled_tools on resume path Resume turns (interrupt_responses set — OAuth-gated MCP consent or tool-approval) crashed with `NameError: cannot access free variable 'effective_enabled_tools'`. The variable is referenced unconditionally by the `stream_with_quota_warning` streaming closure (attachment guidance + tabular inventory) but was only assigned in the non-resume branch. On resume the closure raised before its first yield, the inference-api container returned 500, and the AgentCore Runtime data plane translated that into a 424 Failed Dependency to app-api and the SPA. This broke every interrupt-resume turn since the agent-designer tool-binding refactor (0b9b039a) — most visibly "connect to Gmail for employees", which completes via an OAuth-consent resume. Bind effective_enabled_tools from the paused-turn snapshot on the resume branch (the same source the resume get_agent call uses). Adds a resume-path regression test to tests/routes/test_inference.py that drives /invocations with interrupt_responses and asserts a 200 stream; without the fix it fails with the NameError. Co-Authored-By: Claude Opus 4.8 * chore(kaizen): weekly research scan 2026-07-17 Generated by the kaizen-research skill. Top 5 ideas appended to docs/kaizen/review-queue.md for the kaizen-review-prep run later this morning. Co-Authored-By: Claude Opus 4.8 (1M context) * docs: add session-metadata static sort key spec (issue #175) Root-cause spec for the SessionMetadata parse-failure warnings: the session row's sort key encodes lastMessageAt, forcing a delete+put row move every turn. Concurrent writers race that move and upsert bare ghost rows. Also drives the first-turn duplicate-row race. Fix: static SK (S#{session_id}) + sparse SessionRecencyIndex GSI for recency listing. Covers the expand -> migrate -> backfill -> contract migration, downstream/forked-deployment safety (marker gate + graceful GSI-missing fallback), pagination-token compatibility, and the test matrix. Co-Authored-By: Claude Opus 4.8 * feat(infra): add SessionRecencyIndex GSI to sessions-metadata (issue #175 Phase 0) Sparse recency index (GSI4_PK=USER#{id}, GSI4_SK={lastMessageAt}#{session_id}, projection ALL) for newest-first active-session listing once the base sort key becomes static. Phase 0 of the static-sort-key migration: adding the index is a no-op until rows populate GSI4 keys, so it deploys safely ahead of any code change. IAM already covers it via the SessionsMetadataAccess /index/* wildcard. Update tables-detailed test to assert all four GSIs (the "2 GSIs" title was already stale after DueScheduleIndex) and the new index's key schema. Co-Authored-By: Claude Opus 4.8 * feat(sessions): dual-scheme union read for session listing (issue #175 Phase 1a) Expand-read step of the static-sort-key migration. list_user_sessions now reads the UNION of two disjoint sources so a session is visible whether or not its base sort key has been migrated to the static S#{session_id} form: - legacy (un-migrated): base table, SK begins_with 'S#ACTIVE#' - migrated: SessionRecencyIndex GSI (GSI4_PK=USER#{id}, GSI4_SK={lastMessageAt}#{id}) Pagination switches to a value cursor ({lastMessageAt}#{session_id}) so each page is derived independently from the last returned position, with no cross-page buffering; fetching limit+1 valid rows per source is provably enough to detect a next page. The cursor decoder is tolerant — legacy/undecodable tokens fall back to first page (a harmless reset across the deploy boundary). Degrades to legacy-only if SessionRecencyIndex doesn't exist yet (code ahead of the CDK GSI): the GSI query's ResourceNotFoundException is caught. No writes change and no row migrates in this phase — this only teaches every reader to cope with both schemes, which must be fully rolled out before Phase 1b turns on self-migrating writes. Tests: union ordering, cross-union pagination (no dupes/gaps), migrated-only via GSI, ghost/preview skip, and graceful fallback when the index is absent. conftest sessions_metadata_table fixture gains the SessionRecencyIndex GSI to match prod. Co-Authored-By: Claude Opus 4.8 * fix(deps): bump strands-agents to 1.48.0 for cachePoint-attachment fix Auto prompt caching (CacheConfig strategy=auto) appended its cachePoint after the last user message's content, so any turn attaching a non-PDF document (txt/docx/csv/...) sent [text, document, cachePoint] and Bedrock's Anthropic adapter rejected it with "ValidationException ... messages.N.content.M.type: Field required", surfacing to users as "Agent force-stopped" (prod incidents Jul 14-16, e.g. session dd1a647a on a .txt transcript upload). strands 1.48.0 places the cache point before the first non-PDF document block instead (upstream issue #1966); every placement it produces was verified live against global.anthropic.claude-sonnet-4-6 ConverseStream. Also corrects the model_config comment that credited PR #1438/1.39.0 with this fix - #1438 was the auto-caching feature itself. Co-Authored-By: Claude Fable 5 * fix(sessions): degrade to legacy-only on real ValidationException for missing GSI (issue #175) The Phase 1a dual-scheme read (PR #667) catches a missing SessionRecencyIndex to fall back to legacy-only listing, but only handled ResourceNotFoundException — what moto raises. Real DynamoDB raises ValidationException ("The table does not have the specified index") for a missing GSI (verified against the prod table). So if the 1a backend deployed to an environment before the CDK GSI existed, list_user_sessions would 503 instead of degrading. Broaden the catch to also handle ValidationException (scoped by the "specified index" message so genuinely malformed queries still surface). This restores the intended order-independence: 1a is safe whether or not SessionRecencyIndex exists yet, which matters for prod deploy ordering (backend.yml vs platform.yml) and for forked deployments. Add a test that reproduces the real ValidationException on the index query (moto masks it), asserting fallback to legacy-only results. Co-Authored-By: Claude Opus 4.8 * Add Word document tools (create/modify/list/read) Provision a full Word (.docx) toolset behind the single create_word_document capability toggle. Each tool runs python-docx in Bedrock Code Interpreter and uses the existing user-files store (S3 + DynamoDB) for persistence and delivery. - create/modify/list/read tools in agents/builtin_tools/word_document_tool.py, injected per-request via _build_word_document_tools (inference_api/chat/routes.py). - Frontend inline-visual 'word_document' renderer with an accessible download button (Tailwind utilities, no scoped CSS). - Restore-time content-block sanitizer in TurnBasedSessionManager: drops empty/typeless blocks from restored history that caused Bedrock ConverseStream 'messages.N.content.M.type: Field required'. - Seed create_word_document in bootstrap DEFAULT_TOOLS ('Word Documents') + updated seed tests. * feat(sessions): static-SK write path — born static, self-migrate, no rotation (issue #175 Phase 1b) Turns on the write side of the static-sort-key migration. Sessions stop encoding lastMessageAt in the sort key, so the row never moves and the ghost-row race that produced "Failed to parse session item" warnings is structurally eliminated for every migrated row. Changed (all resolve the row via GSI, which is SK-scheme-agnostic): - ensure_session_metadata_exists: new sessions born at static SK S#{id} + GSI4 keys, with a real attribute_not_exists(PK) conditional put. The deterministic SK makes the guard meaningful, closing the first-turn duplicate-row race the old timestamped SK made impossible to gate. - update_session_activity: drops the per-turn Phase-B rotation. Static rows update in place (SET GSI4_SK re-positions the sparse recency index — no row move); a still-legacy row does its one-time final rotation to the static SK, carrying any concurrent write. - _store_session_metadata_cloud: static SK; migrate legacy->static on move; SET GSI4 for active, REMOVE for deleted; never un-migrates a static row. - session_service.delete_session: resolves the raw SK via _get_session_by_gsi instead of reconstructing S#ACTIVE#{lastMessageAt}#{id} (which misses migrated rows). Non-rotating soft-delete: SET status=deleted + REMOVE GSI4 in place, or migrate a legacy row to a static tombstone. Drops the S#DELETED# prefix (nothing reads it). The ~10 other writers resolve-then-update-in-place on the current SK and need no change — they already work on a static SK and never rotate. Tests: TestWriteSideMigration (born-static, one-time migrate, no rotation, soft-delete in-place/legacy, end-to-end create->activity->list->delete) plus the real ConditionalCheckFailedException contract (moto raises it for a failed conditional put). Updated three tests that encoded the old rotation contract. Full shared+routes suites: 1689 passed. Co-Authored-By: Claude Opus 4.8 * Fix S3 PutObject PermanentRedirect in Word tools The user-files S3 client pinned its endpoint to https://s3.{AWS_REGION}.amazonaws.com. In the AgentCore Runtime AWS_REGION does not reliably match the bucket region, and the explicit endpoint_url disables botocore's automatic S3 region redirect, so PutObject failed with PermanentRedirect. Resolve the bucket's real region via HeadBucket (x-amz-bucket-region header; maps to s3:ListBucket, which the runtime role already has — GetBucketLocation is not granted) and pin the client to it, dropping the hardcoded endpoint_url. Fixes both the save and the presigned download URL region. * feat(scripts): static-SK backfill for the cold tail + ghost cleanup (issue #175 Phase 2) One-shot, idempotent, throttled backfill that finishes the migration for rows the lazy write-path (Phase 1b) hasn't touched: rewrites legacy S#ACTIVE#/S#DELETED# session rows to the static S#{id} scheme (populating GSI4 for active, none for deleted), deletes the ghost/stub rows the old rotating-SK writers produced, and — only once a fresh scan finds zero legacy rows — writes the migration-complete marker that unblocks Phase 3. Safety: - Dry-run by default; --apply required to write. - Static put uses attribute_not_exists(SK) so it never clobbers a row a live writer already migrated with fresher data; the legacy delete is an idempotent no-op if already gone. - --sleep throttles; re-runnable to convergence. - Marker gated: --set-marker re-scans and withholds the marker while any legacy row remains, so Phase 3 can't be unblocked on partially-migrated data. Tested: 10 moto cases (classification, dry-run no-op, active+deleted migrate, ghost delete, idempotency, conditional-put skip of a live-migrated row, marker gating). Also validated as a dry-run against real dev-ai data (145 legacy rows, 0 ghosts) to confirm the real-DynamoDB scan/filter behavior. Co-Authored-By: Claude Opus 4.8 * feat(sessions): contract session list to GSI-only once migration completes (issue #175 Phase 3) Final phase of the static-sort-key migration. list_user_sessions now reads the SessionRecencyIndex GSI alone once the Phase 2 backfill has set the migration- complete marker (PK=MIGRATION#session-sk, SK=STATE, complete=true) — the legacy S#ACTIVE# union branch is only queried until then. - The marker check is memoised per-process (the marker only ever goes unset->set, never back), so migrated deployments pay no extra read after the first observation; a container that started pre-backfill picks up the flip on a later call. - Fails open: any error reading the marker keeps dual-read. Downstream/forked deployments that haven't run the backfill stay in dual-read, so removing the legacy branch here can never blank an un-migrated sidebar. - Safety net: even with the marker set, if the GSI query itself errors (transient ValidationException/ResourceNotFound) the legacy branch is still queried, so a flaky index never returns an empty list. The legacy code path is retained behind the marker rather than deleted, per the downstream-safety design; a later release can drop it once all deployments report the marker set. Tests: dual-read when marker absent, GSI-only (legacy row excluded) when set, marker memoisation, and GSI-failure-falls-back-to-legacy-even-with-marker. Full sessions+routes+backfill+architecture suites: 136 passed. Co-Authored-By: Claude Opus 4.8 * docs(specs): skills v2 — skills as a pure knowledge primitive bound on Agents Supersedes the tool-binding sections of admin-skills-rbac-tool-binding.md and the mode-toggle design in skills-mode.md. Skills become agentskills.io knowledge bundles (no bound_tool_ids), bound on Agents via the Designer, with a user-uploaded tier, opt-in chat selection, Strands AgentSkills plugin runtime, and invoke-through sharing for shared Agents. Co-Authored-By: Claude Fable 5 * refactor(agents): delete MCP tool-folding machinery and hook shims (skills v2 PR-1) Skills no longer bind tools, so the entire fold stack that served bound_tool_ids goes: mcp_binding.py (FoldedMCPTool, resolve_mcp_bindings, the two folded-tool lookup factories), mcp_tool_folding.py and its drop_folded_tools calls in FilteredMCPClient / UICapableMCPClient, the tool_use_provider_lookup / tool_use_approval_lookup shims on OAuthConsentHook / MCPExternalApprovalHook (+ FoldedToolApproval), and their wiring in base_agent. SkillAgent is neutered, not deleted (PR-2): DB-backed skills are instructions-only; the file/dev @skill binding path is unchanged. SkillRegistry loses all_bound_tool_ids/bind_catalog_tools. Note: the spec's stream_coordinator skill_executor unwrap item has no corresponding code — nothing existed to delete there. Per docs/specs/skills-as-agent-primitive.md §2/§8 (PR-1). Co-Authored-By: Claude Opus 4.8 * refactor(skills): remove bound_tool_ids end-to-end (skills v2 PR-1) A skill is a pure knowledge bundle: drop bound_tool_ids from SkillDefinition, the create/update/response DTOs and the Dynamo (de)serialization; delete _validate_bound_tools and the boundToolCount projection; stop emitting boundToolIds in the bindable-catalog meta; seed web_research as an instructions+reference-file bundle with no bound tool. SPA: remove the boundToolIds field/model plumbing, the skill-form bound tools section, the list-page badge, and the tool-picker dialog. Existing rows with boundToolIds deserialize fine (attribute ignored); no production data binds tools to skills (SKILLS_ENABLED=false everywhere). Co-Authored-By: Claude Opus 4.8 * refactor(chat): remove skills mode — policy, toggle, and preference (skills v2 PR-1) Skills are opt-in per turn, not a mode. Delete the admin chat-mode policy (platform_settings sentinel + admin /settings/chat routes + public /system/chat-settings), _resolve_effective_agent_type, and the preferred_agent_mode user setting. DEFAULT_AGENT_TYPE flips to "chat"; an explicit agent_type="skill" (future picker / agent-binding resolver) still resolves skills, and the enabled_skills request plumbing + skills_hash caching are kept verbatim for the PR-4 picker. SPA: delete ChatModeService and the Skills/Tools capabilities toggle; Skills + Tools sections render unconditionally (Skills gated on having skills or an agent lock, so it stays inert while the feature is off); stop sending agent_type/enabled_skills; drop preferredAgentMode and the session-preference agentType plumbing. Agent-bound skill locks now load the skill list on demand so locked rows render their names. Co-Authored-By: Claude Opus 4.8 * docs(specs): mark skills-mode and tool-binding specs superseded by skills v2 Co-Authored-By: Claude Opus 4.8 * feat(skills): swap runtime to Strands AgentSkills plugin (skills v2 PR-2) Spike gate PASSED (20/20, docs/specs/skills-v2-pr2-spike-findings.md): the vended AgentSkills plugin composes with our prompt assembly (block-level injection preserves cache points), the skills_hash cache key, paused-turn resume, and agent.state round-trips our TurnBasedSessionManager unchanged. Runtime swap: - Map DB SkillDefinition -> strands.Skill via a new skills/strands_mapping.py (slugged name for agentskills.io validity + S3/harness portability; true skill_id + human display_name carried in metadata). Add advisory allowed_tools + skill_metadata frontmatter passthrough to the model (D1/D4). - ChatAgent conditionally adds AgentSkills(skills=[...]) when the turn carries accessible_skill_ids; AgentFactory.create_agent gains a plugins param. - Retire the homegrown disclosure stack: SkillAgent, skill_registry.py, skill_tools.py (skill_dispatcher/skill_executor), the @skill decorator + file/dev definitions, and their tests. "skill" stays a registered alias -> ChatAgent so the existing agent_type/skills_hash cache-key + resume path (which the spec keeps) resolve to a ChatAgent-with-plugin unchanged. read_skill_file (L3 reference bytes) and the S3 SKILL.md write-through projection follow as separate PR-2 commits. Full backend suite green (4611 passed, 3 skipped). Co-Authored-By: Claude Opus 4.8 * feat(skills): standard bundle layout + read_skill_file (skills v2 PR-2) Completes the PR-2 runtime: reference-file disclosure (L3) and the agentskills.io bundle layout that makes each skill a portable artifact. S3 bundle layout + SKILL.md projection (#6): - SkillResourceRef gains `kind` (reference|script|asset); dynamo round-trips it, old rows default to reference. - resource_store keys files at skills/{id}/{references|scripts|assets}/{filename} (path-based, dedupe dropped for the readable standard layout); add put_skill_md and resource_key/skill_md_key helpers. - create_skill/update_skill write a SKILL.md projection generated from the row (new apis/shared/skills/bundle.py: slugify + generate_skill_md) — best-effort, never fails the catalog write. Admin upload route gains an optional `kind`. read_skill_file (#4): - New per-turn tool (agents/main_agent/skills/strands_mapping.build_skills_runtime returns plugin + tool from one record fetch). Resolves `path` against the skill's manifest (no filesystem/traversal), serves bytes from SkillResourceStore, labels scripts inert (D5), describes binary assets instead of dumping bytes, and is implicitly access-gated (bound only to the turn's effective records — richer §6 invoke-through is PR-4). Skill.instructions gain an "Available reference files" listing so the model knows what to request. - ChatAgent wires read_skill_file alongside the AgentSkills plugin. slugify moved to apis/shared/skills/bundle (shared by the app-api projection and the agents runtime; import-boundary safe). Full backend suite green (4634 passed). Co-Authored-By: Claude Opus 4.8 * docs(specs): mark skills v2 PR-2 done in the plan Co-Authored-By: Claude Opus 4.8 * fix(agents): correct stale skills copy and surface invalid-save feedback Two issues found while smoke-testing skills v2 PR-2 (#681). Skills v2 decision D1 removed skill->tool binding entirely: a skill is a pure knowledge bundle and never grants or carries a tool. Two surfaces still claimed otherwise -- the Agent Designer skills section ("with their own bound tools") and the admin skill edit form ("and bound tools"). Saving an invalid agent form also no-opped silently: persist() marked the controls touched and returned, so the only feedback was an inline error that is usually below the fold once the author has scrolled to the Model/Skills sections. The dirty banner stayed up, making the click look ignored. Now an invalid save also toasts and scrolls the first invalid control into view. The reveal helper matches on input/textarea/select rather than [formControlName]: the starters array binds [formControlName]="$index", a property binding that renders no attribute to select on. Verified with npm run build and ng test (127 files, 1452 tests passing). Co-Authored-By: Claude Opus 4.8 * feat(skills): user-authored skills tier (skills v2 PR-3) Adds the owner-scoped half of the skill catalog: any user can author their own agentskills.io knowledge bundles and reach them at runtime, without an admin RBAC grant. Backend - `list_skills_by_owner` — GSI4 (SkillOwnerIndex) partition query, the "list my skills" path. `list_skills` gains an `owner_id` filter. - `UserSkillService` — owner-scoped CRUD. Ownership is resolved on every path; a skill you do not own is 404, never 403, so the surface never confirms someone else's skill exists. Resource handling (caps, manifest, bundle layout, orphan GC) delegates to SkillCatalogService so both tiers emit identical bundles. - `/skills/mine/*` routes on the existing session-auth router, so they inherit the SKILLS_ENABLED mount gate. - Skill ids are allocated server-side from the display name and suffixed on collision (docx -> docx_2). Ids stay globally unique because the runtime activation key is the slugified id, and a 409 would disclose the existence of a skill the user cannot see. - `resolve_accessible_skill_ids` now returns catalog ∪ own — ownership is its own grant. This is what makes an authored skill usable at all. Two tier boundaries closed, both of which would have leaked private skills: - `get_all_skill_ids` (RBAC "*" wildcard expansion) now lists only catalog skills; a wildcard grant must not sweep in other users' authored skills. - The admin role-grant endpoints refuse user-authored skills, since granting one to an AppRole would hand a private document to a whole role. The admin catalog list is likewise scoped to owner_id == "system". Frontend - My Skills page (list + create/edit form) with SKILL.md import prefill and reference/script/asset uploads; scripts are labeled non-executable. - Nav entry gated on the same 404 accessibility probe memory-spaces uses. Co-Authored-By: Claude Opus 4.8 * fix(skills): preserve SKILL.md frontmatter through import (skills v2 PR-3) Live clickthrough found an imported bundle losing everything outside name/description: a SKILL.md carrying `license: MIT` round-tripped back out of S3 without it. Spec D2 requires import/export to be round-trip-faithful, and the backend already had `skill_metadata` + `allowed_tools` columns for exactly this — only the client-side import never populated them, leaving both fields dead on the user tier. - `parseSkillMarkdown` now also returns `allowedTools` (comma-separated or inline-array forms) and `metadata` (every non-reserved frontmatter key). Additive to the DTO, so the admin form is unaffected. - The My Skills form carries both through create and update, including for a loaded skill, so an edit never silently drops them. - Advisory tools render as chips with the D1 disclaimer — skills never grant tools; the bound agent decides. Also fixes a `capitalize` on the staged-file line title-casing the whole string ("Uploads When You Save"); only the kind should capitalize. Verified: a bundle with license + compatibility + allowed-tools now emerges from the S3 SKILL.md projection intact. Co-Authored-By: Claude Opus 4.8 * chore(skills): backfill script for v1 skill bundles (skills v2 PR-3) Skills authored before PR-2 have neither the SKILL.md write-through projection nor the standard bundle layout — v1 stored resources content-addressed (skills/{id}/{sha256}) with no `kind`. Their S3 prefix is therefore not a valid agentskills.io bundle: it can't be handed to a managed Harness or exported as-is, which is the whole point of the projection. The script fixes both per skill: copies each legacy object to its standard path, rewrites the row's manifest to point there (adding `kind`), and writes the SKILL.md generated from the row. It imports `generate_skill_md` and the key helpers from the live write path, so a backfilled bundle is byte-identical to one the app writes today. Follows the backfill_session_static_sk conventions: dry-run by default, idempotent, throttled, scopeable to one skill. Copies are non-destructive — legacy objects survive unless --delete-legacy, and a manifest entry whose bytes are missing is left untouched rather than repointed at nothing. Applied to dev-ai/web_research: manifest now points at references/extraction_tips.md, SKILL.md written, read path verified at 200. Co-Authored-By: Claude Opus 4.8 * fix(skills): seed example skill as a v2 bundle (skills v2 PR-3) Backfilling dev's web_research row surfaced that the seeder itself still emits v1 shapes, so every fresh environment reproduces exactly the state the backfill just repaired: - resources landed at the content-addressed key (skills/{id}/{sha256}) with no `kind`, instead of skills/{id}/references/{filename} - no SKILL.md was ever written, so the seeded prefix was not a valid agentskills.io bundle and could not be handed to a managed Harness or exported as-is Both fixed. The slug and frontmatter rules are duplicated from apis/shared/skills/bundle.py rather than imported, because seed.sh runs this script standalone after infra deploy without the app package on the path — the existing content-hash logic was duplicated for the same reason. Tests now pin the standard layout and the projection, plus a guard that the seed prose never again names the retired v1 meta-tools (skill_executor / skill_dispatcher). That drift is what left dev's row instructing the model to call tools deleted in PR-1/PR-2. Note the seeders are skip-if-exists, so this repairs new environments only; existing ones need backfill_skill_bundles.py (dev-ai: applied). Co-Authored-By: Claude Opus 4.8 * feat(skills): selection surfaces + invoke-through access (skills v2 PR-4) Wires the chat opt-in picker end-to-end and lands the §6 invoke-through access predicate. Per spec docs/specs/skills-as-agent-primitive.md §8. The picker's markup shipped in PR-1 but nothing ever sent its selection, so the plain-chat skills path had never actually run. Turning it on surfaced three latent bugs: - Skill resolution was gated on agent_type == "skill", so the picker could never have taken effect. Skills are now driven by the selection on any turn; agent_type gates nothing ("skill" stays a ChatAgent alias only so stale SPA sessions don't 422). - The binding resolver gated on AppRoleService.can_access_skill, which has no ownership clause — an author was blocked on their own authored skill when invoking their own Agent — and whose "*" wildcard matched any id at all, including another user's private skill. Clauses 1+2 now route through resolve_accessible_skill_ids, which expands "*" over the catalog only. - Paused-turn resume and the construction snapshot both keyed skills off agent_type == "skill", which would have orphaned the paused agent of any plain-chat turn carrying skills. Both now key off the snapshot's own enabled_skills. Invoke-through (D7) is deliberately AGENT-scoped: it lives in the binding resolver, not in resolve_accessible_skill_ids, because widening the shared resolver would leak an Agent owner's private skills into every invoker's plain-chat picker and bindable palette. The owner-match clause blocks chain-sharing, and a system-owned Agent gets no invoke-through at all so RBAC stays the sole gate on catalog skills. D6 default flip: an absent or empty enabled_skills means no skills, on both the runtime filter and the picker's untouched-preference default. The two must agree or the UI would show skills as active that the turn never loads. It also keeps skills free for turns that don't want them — an absent selection short-circuits before any RBAC or skill-table read. read_skill_file needed no per-call predicate: its record set IS the turn's effective skill set, so there is no id the model can name to reach a skill the invoker cannot use. The Designer palette union needed no code — /agents/bindable already delegates to resolve_accessible_skill_ids, which PR-3 widened. Backend 4714 passed; SPA 1469 passed across 128 spec files. Co-Authored-By: Claude Opus 4.8 * chore(assets): add GitHub connector logo variants Light/dark Octocat marks alongside the existing google-* connector logos. Follows the repo's theme convention: -light is the black glyph (for light backgrounds), -dark the white one. Nothing references these yet — they're staged for a GitHub connector. Unrelated to the skills work in this branch; riding along rather than sitting untracked. Co-Authored-By: Claude Opus 4.8 * feat(skills): enable Skills v2 + admin-only capability gate (skills v2 PR-5) Flips SKILLS_ENABLED to default-ON with a kill switch and adds the infrastructure wiring it never had, closing out the Skills v2 epic (spec docs/specs/skills-as-agent-primitive.md §8). SKILLS_ENABLED had zero CDK/workflow plumbing, so "enable it per environment" was not previously expressible. Adds SkillsConfig to config.ts, threads SKILLS_ENABLED into both app-api and inference-api (they must stay in step — design-time refuses to bind a skill while the flag is off, so a mismatch would let an Agent be built with skills the runtime then blocks), and forwards CDK_SKILLS_ENABLED in platform.yml with the empty-string-safe ternary an unset GitHub variable requires. Feature existence and audience are two independent controls. The flag says the feature exists in an environment; the new `skills` RBAC capability says who sees the user-facing surfaces. system_admin holds it implicitly via its "*" tools grant, so the picker and My Skills stay admin-only during rollout; GA is one grant of `skills` to the `default` role, no redeploy. The gate raises 404, not 403. The SPA hides the My Skills nav entry by riding the list call, so a 404 hides the surface while a 403 surfaces an error toast — the failure mode that got the scheduled-runs capability gate reverted in prod. It also deliberately does not gate the runtime: an Agent shared to an ordinary user must still resolve its bound skills (invoke-through, §6/D7), and a capability check there would break exactly that path. Verified live end to end against a real agentskills.io bundle (Anthropic's docx) uploaded as a user skill and bound to an Agent: L1 8,075 -> L2 9,835 (skills tool, SKILL.md body) -> L3 10,737 (read_skill_file on references/LICENSE.txt). Invoke-through confirmed with a second non-admin account — the grant resolves through the shared Agent while the same skill stays absent from that user's own picker and /skills/mine. The "session auth, not Bearer" assertion now walks the transitive dependency tree rather than each route's direct dependencies, since the routes hang off the capability gate which in turn depends on the session. Pins the invariant that actually matters instead of the shape. Co-Authored-By: Claude Opus 4.8 * refactor(rbac): remove dead AppRoleService.can_access_skill Skills v2 moved skill authorization to apis/shared/skills/access.py (resolve_accessible_skill_ids = catalog ∪ own, and resolve_invocable_skill_ids which adds the Agent-owner invoke-through clause). AppRoleService.can_access_skill has had zero production callers since PR-4 and is wrong on two axes for anything user-tier: no ownership clause, and its "*" wildcard matches ANY skill id including another user's private authored skill. - Delete the method and its three tests. test_can_access_skill_with_wildcard asserted can_access_skill(user, "any_skill") is True — it enshrined the wildcard over-expansion bug as expected behavior. - Fix two stale docstring/comment references in agent_designer's binding_validation.py that still named it as the live run-time mechanism; since PR-4 that is resolve_invocable_skill_ids. - Reword the intentional "deliberately NOT can_access_skill" rationale in skills/access.py and agent_binding_resolver.py to past tense so they no longer imply the function still exists. Co-Authored-By: Claude Opus 4.8 * refactor(skills): delete dead SkillAccessService Skills v2 moved skill authorization to apis/shared/skills/access.py (resolve_accessible_skill_ids / resolve_invocable_skill_ids). SkillAccessService was left behind with zero live callers — not exported from admin/services/__init__.py, not DI-wired, and reached by no dynamic import. Its can_access_skill carried the same "*"-wildcard over-expansion flaw that got AppRoleService.can_access_skill deleted in #686. Also updates the stale comment in admin/skills/routes.py that named the service as the consumer of the all-skill-ids snapshot; that snapshot is still live, but its reader is now skills.access. Co-Authored-By: Claude Opus 4.8 * docs(agents): draft the Agent Directory spec A browse-and-discover surface for published Agents: a directory page and a detail page modeled on the ChatGPT/Claude plugin-detail layout, built as a read-view over the Agent record rather than a new primitive. The central decision (D1) is that we do NOT introduce a "Plugin" noun. Both vendors need a bundle layer because their capabilities install into a workspace separately from any persona; our Agent's `bindings[]` already IS that bundle, attached to the persona. Every field a vendor plugin-detail page renders already exists on our record. Grounding findings that shaped the design: - `VisibilityStatusIndex` (GSI2) is live and populated, and PUBLIC *access* still resolves to "viewer" — only the *listing* was switched off in ad4437e9 when email sharing superseded a public index. The read path is mostly built. - Listing is nonetheless a new sparse GSI5, not a re-enable: `VISIBILITY#PUBLIC` is one hot partition and can't be filtered by category. GSI5 is the next free slot (GSI4 is DueSyncIndex), and DueSyncIndex on the same table is the precedent — unlisted agents have no key, so the query physically can't see them. - `listed` is deliberately separate from `visibility` (D3). Deriving listing from PUBLIC would retroactively publish every existing PUBLIC agent to the whole institution with no author consent; backfill is listed=false. - Publishing amplifies Skills v2 invoke-through from a typed email list to everyone, so the publish dialog must enumerate which authored skills it exposes, and memory_space bindings block publication outright (D5). - The Designer's block-on-missing rule (D5) strains under open browsing, so the detail page previews per-invoker runnability up front — which resolves the "per-invoker capability preview" open question parked in agent-designer.md. Also notes a behavior change needing a call: GET /agents/{id} currently returns `instructions` to any PUBLIC viewer, which is a much larger exposure once agents are broadly listed than it was under link-sharing. Co-Authored-By: Claude Opus 4.8 * refactor(artifacts): collapse the two catalog rows into one Artifacts toggle Artifacts shipped as two independent `protocol: local` catalog rows — `create_artifact` and `update_artifact` — so the tool picker listed them as two unrelated entries. The picker only groups children under a parent for MCP protocols (driven by `serverTools`), so there was nothing to nest them under; the flat listing was a data-model fact, not a template gap. Adopt the Word-documents idiom already used one entry below in the seed list: a single catalog row whose id is the gate key, with the runtime injecting the full toolset. `create_artifact` is now that key and provisions both the create and update tools. `seed_default_tools` is create-only, so a seed run does nothing to an environment seeded before this change. Add a backfill script that retitles the surviving row, promotes `update_artifact` to `create_artifact` everywhere a grant can hide — role TOOL_GRANT# items *and* the grantedTools/effectivePermissions.tools arrays on DEFINITION, user toolPreferences, assistant bindings — then deletes the retired row. Promote-before-delete, so an aborted run degrades to "both granted", never "neither". Schedule snapshots are left alone: a stale id there is an inert no-op since the runtime only reads `create_artifact`. Two judgement calls worth recording: - User prefs are a sparse override map, so only an explicit *enable* of the retired id carries over. Someone who switched update off while leaving create at its default-on never asked to lose artifacts, so an explicit disable just drops the key. - A role granting `*` gains no concrete grant — the wildcard already covers the keeper and narrowing it would be a silent scope change. Behavior change: anyone with create enabled but update disabled now gains update. That is inherent to collapsing the toggle. Co-Authored-By: Claude Opus 4.8 * chore(sidenav): hide the My Skills nav entry The /my-skills route, page, service, and the whole /skills/mine backend surface stay fully functional — only the sidenav link is removed, until we decide how users should actually navigate to their skills. Drops the now-dead showMySkills computed, the MySkillService injection, and its loadSkills() accessibility probe (that call existed solely to decide whether to render the link). Co-Authored-By: Claude Opus 4.8 * docs(skills): draft the Skill Creator spec Adopts the authoring half of Anthropic's open-source skill-creator and scopes out the eval half, which assumes a filesystem, subagent spawning, and script execution — none of which exist here (scripts are inert by design). Evals hang off F1's headless lane instead. PR-1 ships the methodology as an admin-catalog skill with zero code, routing the handoff through the My Skills form's existing frontmatter parsing. PR-2..PR-4 add the missing primitive: agent tools that write the user's own skills, executing as the invoking user and re-checking ownership per call, mirroring memory_write. Depends on skill-bundle-import's nested paths and server-side SKILL.md parser rather than restating them. Co-Authored-By: Claude Opus 4.8 * feat(skills): drop the `skills` capability gate from user-facing routes The user-facing skills surfaces (`GET /skills/`, `PUT /skills/preferences`, and all of `/skills/mine/*`) hung off `require_skills_capability`, which 404'd anyone not holding the `skills` RBAC capability. Its job was to keep skills admin-only during the v2 rollout, with GA framed as "one grant of `skills` to the `default` role, no redeploy." That GA path does not work, for two independent reasons: 1. `default` is a *fallback* role. `resolve_user_permissions` consults it only when a user matches zero AppRoles (service.py "Step 3"); it is not merged alongside a matched role. Prod's `default` carries no JWT mappings at all, so granting there would reach only unmapped users — never the faculty/staff/student cohorts. 2. A capability id cannot be granted from the admin roles UI regardless. That form builds `grantedTools` from the tool catalog, and a capability is not a tool, so there is no way to select it and no free-text entry. Net effect: an admin who granted a catalog skill to a role would find it silently invisible to that role's users, with no in-product way to fix it. Remove the gate. Skills are governed by `SKILLS_ENABLED` per environment and by a role's `grantedSkills` per cohort — a complete model that the admin UI can actually operate. Both surfaces are already self-limiting: `GET /skills/` returns only what `resolve_accessible_skill_ids` grants (no grants means an empty list and no rendered picker), and every `/skills/mine/*` route is owner-scoped inside `UserSkillService`. The route-coverage control is kept rather than dropped, retargeted from "every route is capability-gated" to "every route requires a session" — the invariant that still matters now that the session dependency is the only thing between these routes and an anonymous caller. A second test pins the removal so the gate cannot creep back without also making capabilities grantable from the roles UI. Also corrects the "GA = grant to `default`" claim where it appeared in capabilities.py, infrastructure/lib/config.ts, and platform.yml, and notes that `SCHEDULED_RUNS_CAPABILITY` is itself unused (that gate was dropped after 403ing in prod), which leaves capabilities.py with no consumers. Co-Authored-By: Claude Opus 4.8 * refactor(rbac): delete the dead capabilities module `apis/shared/rbac/capabilities.py` has no consumers left. It defined two capability ids granted through the `grantedTools` axis: * `SCHEDULED_RUNS_CAPABILITY` was orphaned when the RBAC gate on `/schedules` and `/runs` was dropped after 403ing in prod. * `SKILLS_CAPABILITY` was the last live caller, removed in the preceding commit along with `require_skills_capability` and its 12 route deps. Delete it rather than keeping it as a reference. The mechanism it documented is not one we want reached for again: a capability id cannot be granted from the admin roles UI (that form builds `grantedTools` from the tool catalog and offers no free-text entry), so any gate built on it is operable only by hand-writing DynamoDB items. Both live gates were removed for that reason. Its docstring carried two findings that cost real effort to establish, so they move to `AppRoleService.resolve_user_permissions` — the code they actually describe — rather than dying with the file: 1. `default` is a *fallback*, not a universal role. Step 3 substitutes it only when a user matched zero AppRoles, and prod's `default` carries no `jwtRoleMappings`, so granting there reaches only unmapped users. The "GA = one grant to `default`, no redeploy" framing that appeared in several comments was wrong. 2. The roles-UI limitation above, recorded where someone would see it before routing a new grant through this axis. The algorithm list in that docstring is renumbered to match the code's own step comments, so the "Step 3" reference is unambiguous. Also corrects both `feature_flags.py` docstrings, which still described the now-deleted capability as the companion "who may use it" control: `skills_enabled` points at a role's `grantedSkills`; `scheduled_runs_enabled` notes the flag is now the only control and the routes are deliberately ungated. Co-Authored-By: Claude Opus 4.8 * fix(chat-input): make the textarea scrollable and reset it after submit Three defects in the chat input's auto-sizing: - The textarea carried `overflow-hidden`, so once content exceeded the visible area there was no way to scroll within it. - `onTextareaInput` set `height = scrollHeight` with no clamp. Past 200px the inline height kept growing while `max-height` capped the rendered height, leaving the two diverged and the scrollbar unreachable. The growth is now clamped in a shared `autoResize()`. - `submitChatRequest` cleared the value but left the stale inline height, so the input stayed expanded after sending. Adds `resetTextareaHeight()`. Also drops the `.chat-textarea` CSS block: it declared its own min-height/max-height/field-sizing but the class is applied nowhere in the template, so the rules were dead while contradicting the real inline styles. The JS path is now the sole sizing authority. The `isExpanded` signal goes too — it was written on submit and never read. Co-Authored-By: Claude Opus 4.8 * fix(compaction): make restored history byte-stable to preserve Bedrock prompt-cache hits Tool-content truncation previously ran on every session restore behind a sliding protected-turns window, so each new turn re-mutated the turn that just aged past the window. Bedrock prompt caching requires an exact prefix match, so this forced a full prefix cache re-write (~$2.5/MTok on a 35k-150k prefix) nearly every turn — costing far more than the read tokens truncation saved (evidence: prod session aecd387d, inter-turn prefix shrinkages of -382/-1035/-1513 with cacheRead=0 inside the cache TTL). Redesign: truncation is now driven only by a persisted truncation_anchor in the compaction state (sessions-metadata `compaction` attribute): - The anchor moves when the checkpoint advances (update_after_turn), where the slice already pays the single cache re-write — one mutation per compaction event instead of one per turn. - It also advances opportunistically at restore when more than cache_ttl_seconds (default 300s, AGENTCORE_MEMORY_COMPACTION_CACHE_TTL_SECONDS) have passed since the previous turn: the cache entry has expired anyway, so pending truncations are applied for free. - Legacy state records without the field default the anchor to the checkpoint, so retained history stops being mutated immediately on upgrade. - The compaction-failure path in initialize() now resets _compaction_state_loaded so update_after_turn re-loads persisted state instead of overwriting checkpoint/anchor with defaults. Removes the now-dead _find_protected_indices sliding-window helper and adds tests/agents/main_agent/session/test_compaction_stability.py asserting agent.messages is byte-identical across consecutive restores whenever no compaction-state change occurs. Co-Authored-By: Claude Fable 5 * fix(skills): deterministic skill ordering to preserve Bedrock prompt cache Skill records reached the AgentSkills system-prompt block in nondeterministic order, changing the prompt between turns of the same session and invalidating the Bedrock prompt cache (exact-prefix match) — forcing full cache re-writes on turns well inside the TTL. Two sources, fixed at three layers: - batch_get_skills returned raw DynamoDB batch_get_item response order; now sorted by skill_id. - resolve_user_permissions built grant unions as sets and returned list(set) — iteration order varies per process via hash randomization; tools/models/skills now sorted. - build_skills_runtime sorts records before constructing AgentSkills as defense in depth at the injection point. Regression tests force a descending batch_get_item response order and a reversed fetch order, both verified to fail without the fix. Co-Authored-By: Claude Fable 5 * fix(cache): add tools + system cachePoints so message-level misses read the stable prefix CacheConfig(strategy="auto") places exactly one message-level cachePoint; when its lookup misses, nothing is read and the whole prefix re-writes at the cache-write premium. One proven miss mode is structural: Anthropic's cache lookback checks only ~20 content blocks behind the breakpoint, so a wide parallel tool fan-out (18 parallel calls = ~38 new blocks) pushes the previous checkpoint out of range — prod session aecd387d observed cacheRead=0 / cacheWrite=134k mid-turn (~$0.34). Now the request carries 3 of Bedrock's max-4 cachePoints: - toolConfig tail via cache_tools="default" - system tail via SystemContentBlock list with trailing cachePoint (built in AgentFactory; the cache_prompt config key is deprecated) - last-user-message point via the existing auto strategy (which strips only message-level points, never system/tools ones) Both new points are gated on ModelConfig.bedrock_cache_points_supported() (mirrors Strands' _cache_strategy predicate) because unlike auto mode they would be sent verbatim to non-Anthropic models and rejected. Verified live on global.anthropic.claude-sonnet-5 via ConverseStream: simulated message-level miss reads the 6.5k-token system+tools prefix from cache (cacheRead=6497, cacheWrite=83) instead of re-writing it. CountTokens accepts the cachePoint-bearing request, so context attribution is unaffected. Position/budget test asserts exactly 3 cachePoints. Upstream: strands-agents/harness-sdk#3348 proposes auto mode keep a rolling pair of message cachePoints so fan-outs stay within the lookback. Co-Authored-By: Claude Fable 5 * feat(observability): make prompt-cache economics measurable per model call A $1.60 prod conversation audit (session aecd387d) showed 75% of spend was avoidable Bedrock prompt-cache re-writes, and diagnosing the causes took hours of manual forensics against raw DynamoDB rows. This makes the whole class measurable end to end: - PrefixFingerprintHook (BeforeModelCallEvent) hashes the three cacheable prefix components per model call — toolConfig (order-sensitive canonical JSON), effective system prompt (captured after AgentSkills injection), and message history excluding the newest message. The stream coordinator persists entry N on the turn's Nth assistant-message cost row, so a miss is diagnosable with a column diff instead of row forensics. - Write-time cacheStatus per cost row (first_write | hit | miss_ttl_expired | miss_avoidable | uncached) derived from the session's previous C# row (one GSI read), plus wastedUsd for avoidable misses priced at the cache-write premium over cache-read from the row's own pricingSnapshot. Turn rows now write sequentially (was parallel) so each call classifies against its true predecessor. - Session-row rollups next to totalCost: totalCacheReadTokens, totalCacheWriteTokens, avoidableMissCount, wastedUsd — cache-efficiency ratio for lists/admin without scanning cost rows. - Admin cost anatomy: GET /admin/costs/sessions/{sessionId}/calls (require_admin) returns chronological per-call rows with token splits, cost, cacheStatus, and fingerprints, plus session-level cache summary. - CloudWatch EMF per call (CacheReadTokens / CacheWriteTokens / AvoidableMiss / WastedUsd) via a raw-JSON stdout logger — dashboard and alarm on fleet cache-write share with no SDK calls or extra IAM. - CI determinism guard: builds the chat-agent surface twice from shuffled skill/role/tool record orders (RBAC merge -> skills runtime -> Agent -> AgentSkills injection) and asserts identical system-prompt and toolConfig fingerprints; verified to fail when the skill-ordering sort is removed. Complements the sorting + byte-stability fixes from feature/prompt-cache-stability (merged in). Co-Authored-By: Claude Fable 5 * feat(observability): kill switch + prompt-cache conventions in CLAUDE.md PROMPT_CACHE_OBSERVABILITY_ENABLED=false disables the fingerprint hook, per-call cacheStatus derivation (and its GSI read), and EMF emission — default ON per house convention; empty string stays enabled. Raw cacheRead/cacheWrite token rollups are unaffected (usage passthrough, not derived). CLAUDE.md now encodes the determinism + byte-stability contract and the fingerprint-based debugging recipe. Co-Authored-By: Claude Fable 5 * chore: PR #697 follow-up breadcrumbs + token-cost-effectiveness tenet - CLAUDE.md: add "token cost effectiveness is a design tenet" bullet under Key Conventions — prefix determinism / bounded per-turn payloads / verify via the #697 observability, balanced so quality wins on genuine conflict. - kaizen review-queue: queue the two deferred #697 follow-ups — track harness-sdk#3348 (rolling-pair cachePoints; local workaround gated on dashboard evidence) and the ContextOffloader S3 adoption spike (with its four known gotchas). Co-Authored-By: Claude Fable 5 * feat(observability): prompt-cache CloudWatch dashboard + alarms (PR #697 follow-up) New cross-service construct area lib/constructs/observability/ with a PromptCacheObservabilityConstruct composed into PlatformStack. Graphs the dimension-less EMF metrics both APIs emit into AgentCoreStack/PromptCache (cache read/write tokens, a cache-efficiency MathExpression, AvoidableMiss, WastedUsd) plus a Logs Insights widget over the runtime log group grouped by cacheStatus. Console-only alarms on AvoidableMiss and WastedUsd Sums (stricter in prod, NOT_BREACHING on missing data so the PROMPT_CACHE_OBSERVABILITY_ENABLED kill switch stays quiet). No SNS — alerting infra is deliberately out of scope, matching kb-sync and scheduled-runs. Co-Authored-By: Claude Fable 5 * feat(admin-costs): add per-session cost-anatomy drill-down page Consumes GET /admin/costs/sessions/{id}/calls (backend PR #697): - SessionCostAnatomy / SessionCallRow / PrefixFingerprints models + CacheStatus union - AdminCostHttpService.getSessionCostAnatomy with URL-encoded session id - New /admin/costs/sessions/:id route (lazy, component input binding) - Drill-down page: summary rollups (total cost, cache efficiency incl. null, avoidable misses, wasted USD, cache read/write tokens), chronological calls table with color-coded cacheStatus badges, and prefix-fingerprint diffing that flags which hash (tools/system/history) flipped vs the previous fingerprinted call — the cache-buster diagnosis on miss_avoidable rows. Expandable rows show full hashes + messageCount; 404 renders a no-cost-rows empty state. - Session-id lookup form on the Cost Analytics dashboard as the entry point - Vitest specs for the diff util, HTTP method, and page states Co-Authored-By: Claude Fable 5 * fix(observability): don't flag first cache write after below-threshold calls as miss_avoidable When every prior call in a session was uncached (prompt below the model's minimum cacheable prefix, e.g. ~4096 tokens on Claude Haiku 4.5), the first call that crosses the threshold does cacheWrite>0/cacheRead=0 and was classified miss_avoidable — inflating the AvoidableMiss and WastedUsd EMF metrics and the admin Session Cost Anatomy page. Verified live in dev-ai session 9a1f25b2 (calls 1-5 uncached at 3.5-4k tokens, call 6 falsely flagged with write=4122/read=0). classify_cache_status now takes the previous call's cached-prefix token total: when the immediately preceding call had zero cache activity there was no entry to read from, so the write is classified first_write (the expected initial population) and excluded from waste pricing. Unknown (None) keeps the previous behavior. Co-Authored-By: Claude Fable 5 * Set S3_USER_FILES_BUCKET_NAME on inference-api runtime The AgentCore Runtime env block set DYNAMODB_USER_FILES_TABLE_NAME but not S3_USER_FILES_BUCKET_NAME, so the Word-document tools' _user_files_bucket() fell back to the literal 'user-files' default and PutObject failed with AccessDenied (and earlier PermanentRedirect against that unrelated bucket). The runtime role's UserFilesBucketAccess already grants Get/Put/Delete/List on the real bucket; this just points the runtime at it. tsc build passes. * Fail loudly when Word doc storage bucket is unconfigured _user_files_bucket() previously defaulted to a literal 'user-files' bucket when S3_USER_FILES_BUCKET_NAME was unset, which surfaced a missing runtime env var as a confusing S3 PermanentRedirect/AccessDenied. It now raises _StorageNotConfiguredError, and create/modify/read short-circuit before the Code Interpreter run with a clear 'storage is not configured' message. * feat(office-tools): add Excel spreadsheet creation and editing tools - Add excel_spreadsheet_tool.py with create/modify/list/read tools for .xlsx files - Create office/_storage.py module with shared Code Interpreter and S3 storage utilities for Word and Excel - Implement Excel toolset integration in inference_api chat routes with tool injection - Add "Excel Spreadsheets" catalog entry to DEFAULT_TOOLS in seed_bootstrap_data.py - Replace word-document-renderer with generic file-download-renderer for all generated office documents - Extend Word document tool to use shared office storage module - Update test fixtures and chat routes to support Excel tool provisioning - Generated Excel files are persisted to S3_USER_FILES_BUCKET_NAME and appear in chat Files panel * feat(office-tools): add PowerPoint presentation toolset Add powerpoint_presentation_tool.py with create/modify/list/read tools for .pptx files, mirroring the Word/Excel toolsets and reusing the shared office/_storage.py module (Code Interpreter + user-files storage). python-pptx runs in the sandbox; generated decks persist to S3_USER_FILES_BUCKET_NAME and render inline via the shared file_download card. - Gate key create_powerpoint_presentation injects the full toolset via _build_powerpoint_presentation_tools in inference_api chat routes - Seed 'PowerPoint Presentations' into DEFAULT_TOOLS; update seed test counts (7->8) and assertions - Add pptx MIME/extension to ALLOWED_MIME_TYPES/ALLOWED_EXTENSIONS - Frontend: file-download-renderer picks an orange presentation icon for .pptx/.ppt - Fold slide-design guidance (palettes, typography, layout variety, anti-patterns) into the create docstring in lieu of porting AWS's JS/Deno skills framework * feat(office-tools): add PowerPoint template support + layout introspection create_powerpoint_presentation gains an optional template_name: when given, the deck is built on top of that .pptx (resolved from this chat's files) so it inherits the template's slide masters, layouts, theme colors, fonts, and master/layout branding. The template's example slides are stripped first (removing slide-id refs leaves masters/layouts intact), so it starts themed-but-empty. Add list_powerpoint_layouts tool (equivalent of the AWS sample's get_presentation_layouts): reports a template's layout indices, names, and placeholders so the model targets the right layout/placeholders when building on it. Provisioned by the same create_powerpoint_presentation gate key (no new catalog entry). Wired the new tool into _build_powerpoint_presentation_tools in inference_api chat routes. * fix(infra): allow the mcp-sandbox origin in the SPA frame-src CSP The SPA distribution's CloudFront ResponseHeadersPolicy only ever opened frame-src to 'self' and the artifacts origin, so every domained deploy CSP-blocked the MCP App sandbox iframe (mcp-sandbox.{domain}) — the sandbox side's frame-ancestors was locked to the SPA origin from the start, but the SPA side was never extended. Localhost dev bypasses CloudFront's response headers entirely, which masked the gap through live verification. Thread the already-computed mcpSandboxProxyOrigin from PlatformStack into SpaDistributionConstruct and append it to frame-src. Add a regression test that synths a domained PlatformStack and asserts both iframe origins are present in the frontend headers policy. Co-Authored-By: Claude Fable 5 * feat: session workspace tools (workspace_list/read/write) Generic agent file surface over the existing user-files store, per docs/specs/session-workspace-tools.md: - apis/shared/files/workspace.py: DynamoDB-metadata-first list (session + user scope), bounded ranged text reads (48KB/call with offset continuation), binary by presigned-URL reference only (never base64), and a write path mirroring word_document_tool._store_document (READY FileMetadata row, source="agent", quota check + increment). Missing identity raises — no default-user fallback. - FileMetadata gains an additive display-only `source` field (default "upload"). - agents/builtin_tools/workspace_tools.py: make_workspace_{list,read, write}_tool closure factories; workspace_write returns the inline download-card contract (ui_type "workspace_file"). - Wiring: _build_workspace_tools in inference chat routes behind the single "workspace_files" catalog gate key (seeded, enabledByDefault false) and the WORKSPACE_TOOLS_ENABLED default-on kill switch (apis/shared/feature_flags.py). - SPA: workspace_file ui_type routes to the existing download-card renderer. - 48 new backend tests; seed tests updated for the 7th catalog entry. Co-Authored-By: Claude Fable 5 * fix: render markdown artifacts authored under the default HTML type create_artifact's content_type defaults to text/html, so a request like "create a markdown recipe artifact" produced raw Markdown stored verbatim as HTML — the iframe then showed run-together `#`/`**` source and the card badge read "HTML" instead of rendering the document. Add a server-side safety net: HTML-typed content that lacks a standalone HTML document shell (`` / ``) is reclassified as text/markdown before storage, so the writer wraps it into a proper render document and the DynamoDB row (badge + render-Lambda mapping) stays truthful. Real HTML documents, Markdown, and other MIME types pass through untouched. Applied in both create_artifact_record and update_artifact_record (the update path covers the inherited-type case where there is no argument for the model to get right). Also steer the model up front: the create_artifact docstring now says to author prose (reports, docs, recipes, mostly-text) as Markdown and to honor an explicit "markdown" request, reducing how often the backstop fires. Co-Authored-By: Claude Opus 4.8 * fix: don't render empty "Thinking" blocks from signature-only reasoning Sonnet 5 sometimes persists a reasoningContent block with empty reasoningText.text and no redactedContent (a signature-only thinking block). The signature must stay in the message because Bedrock requires it on follow-up calls, but the block has nothing to display. The live stream parser already guards on reasoningText, but the history- rehydration path bypasses it and reaches displayBlocks, which pushed a Thinking display block for any truthy reasoningContent object. The reasoning-content component then renders its "Thinking" header unconditionally, so an empty collapsible appeared on reload. Guard the render decision with hasRenderableReasoning() so a reasoningContent block is only painted when it has reasoning text or redacted content -- mirroring the component's own visibility logic and the live parser's guard. Display-only: the block is left untouched in the persisted message to preserve the signature/prompt-cache contract. Co-Authored-By: Claude Opus 4.8 * Release/1.11.0 Feature release adding two new agent capabilities to chat: a PowerPoint (.pptx) presentation toolset and a generic File Workspace toolset, plus two rendering fixes. No new AWS resources and no CDK deploy required. - feat: PowerPoint presentation toolset (create/modify/list/read + layouts) behind the "PowerPoint Presentations" catalog toggle, off by default (#713) - feat: File Workspace toolset (workspace_list/read/write) behind the "File Workspace" catalog toggle + WORKSPACE_TOOLS_ENABLED kill switch (#716) - fix: render Markdown artifacts authored under the default HTML type (#720) - fix: don't paint empty "Thinking" blocks from signature-only reasoning (#721) Bump VERSION 1.10.0 -> 1.11.0 and sync manifests + lockfiles. Co-Authored-By: Claude Opus 4.8 --------- Co-authored-by: Claude Opus 4.8 Co-authored-by: Colin Smith <7762103+colinmxs@users.noreply.github.com> Co-authored-by: derrickfink Co-authored-by: Derrick Fink Co-authored-by: Roman Meredith Co-authored-by: Roman meredith <48036775+ramenNoodles1998@users.noreply.github.com> --- CHANGELOG.md | 14 + README.md | 4 +- RELEASE_NOTES.md | 58 ++ VERSION | 2 +- backend/pyproject.toml | 2 +- backend/scripts/seed_bootstrap_data.py | 30 + .../agents/builtin_tools/artifacts/service.py | 48 +- .../agents/builtin_tools/artifacts/tools.py | 7 + .../powerpoint_presentation_tool.py | 750 ++++++++++++++++++ .../agents/builtin_tools/workspace_tools.py | 190 +++++ backend/src/apis/inference_api/chat/routes.py | 93 +++ backend/src/apis/shared/feature_flags.py | 17 + backend/src/apis/shared/files/__init__.py | 20 + backend/src/apis/shared/files/models.py | 9 + backend/src/apis/shared/files/workspace.py | 430 ++++++++++ .../artifacts/test_artifact_tools.py | 54 ++ .../builtin_tools/test_workspace_tools.py | 154 ++++ backend/tests/shared/test_workspace.py | 278 +++++++ backend/tests/test_seed_system_admin_jwt.py | 34 +- backend/uv.lock | 2 +- docs/specs/session-workspace-tools.md | 244 ++++++ frontend/ai.client/package-lock.json | 4 +- frontend/ai.client/package.json | 2 +- .../assistant-message.component.spec.ts | 32 + .../components/assistant-message.component.ts | 26 +- .../file-download-renderer.component.ts | 9 + infrastructure/package-lock.json | 4 +- infrastructure/package.json | 2 +- 28 files changed, 2499 insertions(+), 20 deletions(-) create mode 100644 backend/src/agents/builtin_tools/powerpoint_presentation_tool.py create mode 100644 backend/src/agents/builtin_tools/workspace_tools.py create mode 100644 backend/src/apis/shared/files/workspace.py create mode 100644 backend/tests/agents/builtin_tools/test_workspace_tools.py create mode 100644 backend/tests/shared/test_workspace.py create mode 100644 docs/specs/session-workspace-tools.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 6a8a39eaf..526bfad45 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,20 @@ All notable changes to this project are documented in this file. Format follows For narrative release notes written for operators and product owners, see [RELEASE_NOTES.md](RELEASE_NOTES.md). +## [1.11.0] - 2026-07-24 + +Feature release adding **two new agent capabilities to chat**: a PowerPoint (`.pptx`) presentation toolset that mirrors the existing Excel/Word tools, and a generic **File Workspace** toolset that lets the agent list, read, and save text files in a conversation's workspace. Both ship as single catalog toggles, off by default. Also fixes Markdown artifacts that rendered as raw source when authored under the default HTML type, and empty "Thinking" blocks that appeared on conversation reload. No new AWS resources, no dependency changes, no CDK deploy required — ships via `backend.yml` + `frontend-deploy.yml`; two new tool catalog entries must be seeded per environment. + +### 🚀 Added + +- PowerPoint presentation toolset — `create_powerpoint_presentation`, `modify_powerpoint_presentation`, `list_powerpoint_presentations`, `read_powerpoint_presentation`, and `list_powerpoint_layouts` build and edit `.pptx` decks via python-pptx in the sandboxed Code Interpreter; generated files persist to the user-files S3 bucket and appear in the chat Files panel with a download link. One catalog entry ("PowerPoint Presentations", gate key `create_powerpoint_presentation`, off by default) provisions the whole set (#713) +- File Workspace toolset — `workspace_list`, `workspace_read`, `workspace_write` give the agent a generic file surface over the conversation's user-files store: read uploaded text files on demand and save text deliverables (Markdown, CSV, JSON) to the chat Files panel with a download link. One catalog entry ("File Workspace", gate key `workspace_files`, off by default) provisions the set, gated per environment by the `WORKSPACE_TOOLS_ENABLED` kill switch (default ON). Backed by a new `apis/shared/files/workspace.py` module; file records gain a display-only `source` provenance field (#716) + +### 🐛 Fixed + +- Markdown artifacts authored under the default `text/html` content type now render as formatted documents instead of run-together `#`/`**` source — HTML-typed content that lacks a full HTML document shell is reclassified as Markdown, and the `create_artifact` tool guidance now steers prose deliverables to Markdown mode (#720) +- Empty "Thinking" blocks no longer appear on conversation reload — signature-only reasoning blocks (which some models, e.g. Sonnet 5, emit and persist for API correctness) are kept in the message for the Bedrock signature/prompt-cache contract but are no longer painted as an empty collapsible, matching the live stream parser's guard (#721) + ## [1.10.0] - 2026-07-21 Feature release adding **Excel spreadsheet creation and editing to chat** — a four-tool `.xlsx` toolset behind a single "Excel Spreadsheets" catalog toggle, built on a new shared office-document storage module — and fixing the CSP gap that **blocked every MCP App iframe on deployed environments**. Requires a CDK deploy (SPA CloudFront response-headers change); no data migration. diff --git a/README.md b/README.md index 9742c0d8f..ace2eaa2f 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ **An open-source, production-ready Generative AI platform for institutions** *Built by Boise State University, designed for everyone.* -[![Release](https://img.shields.io/badge/Release-v1.10.0-6366f1?style=flat&logo=github&logoColor=white)](RELEASE_NOTES.md) +[![Release](https://img.shields.io/badge/Release-v1.11.0-6366f1?style=flat&logo=github&logoColor=white)](RELEASE_NOTES.md) [![Nightly](https://github.com/Boise-State-Development/agentcore-public-stack/actions/workflows/nightly.yml/badge.svg)](https://github.com/Boise-State-Development/agentcore-public-stack/actions/workflows/nightly.yml) ![Python](https://img.shields.io/badge/Python-3.13+-3776AB?style=flat&logo=python&logoColor=white) @@ -296,7 +296,7 @@ agentcore-public-stack/ See [RELEASE_NOTES.md](RELEASE_NOTES.md) for the full changelog, including new features, bug fixes, platform upgrades, and deployment notes for each release. -**Current release:** v1.10.0 +**Current release:** v1.11.0 --- diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index 69e3b73a4..bbd90e369 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -1,3 +1,61 @@ +# Release Notes — v1.11.0 + +**Release Date:** July 24, 2026 +**Previous Release:** v1.10.0 (July 21, 2026) + +--- + +> 🚀 **No CDK deploy required this release** — no new AWS resources, no dependency changes, no infrastructure edits. Ship backend code via `backend.yml` and the SPA via `frontend-deploy.yml`. **Per-environment action:** two new tool catalog entries ("PowerPoint Presentations", "File Workspace") must be seeded and granted via RBAC before users can enable them — see Deployment notes. + +--- + +## Highlights + +v1.11.0 adds **two new agent capabilities to chat**. First, a **PowerPoint presentation toolset** — the agent can create, edit, read, and list real `.pptx` decks, built with python-pptx inside the sandboxed Code Interpreter and delivered through the chat Files panel with a download link — completing the office-document trio alongside the existing Excel and Word tools. Second, a generic **File Workspace toolset** that gives the agent a first-class way to list, read, and save text files (Markdown, CSV, JSON) in a conversation's workspace, over the same user-files store. Both ship as single catalog toggles, off by default. The release also fixes two rendering bugs: Markdown artifacts that came out as raw `#`/`**` source when authored under the default HTML type, and empty "Thinking" collapsibles that appeared on conversation reload for signature-only reasoning blocks. + +## PowerPoint presentations in chat + +Users can ask the agent to build or revise real PowerPoint decks mid-conversation — slide outlines, briefing decks, templated layouts — and get a downloadable `.pptx` back in the chat's Files panel. The capability mirrors the Excel and Word toolsets: one admin toggle provisions the whole round-trip. + +### Backend + +- `agents/builtin_tools/powerpoint_presentation_tool.py` (750+ lines) — five tool factories: `make_create_powerpoint_presentation_tool`, `make_modify_powerpoint_presentation_tool`, `make_list_powerpoint_presentations_tool`, `make_read_powerpoint_presentation_tool`, and `make_list_powerpoint_layouts_tool`. Generation and edits run python-pptx inside the sandboxed AgentCore Code Interpreter; nothing executes in the API container, and identity is captured by closure (same pattern as the Word/Excel tools, since the runtime does not populate `ToolContext`). +- `apis/inference_api/chat/routes.py` — `_build_powerpoint_presentation_tools` injects the toolset at runtime when the catalog toggle is enabled. One catalog entry ("PowerPoint Presentations", gate key `create_powerpoint_presentation`, `enabledByDefault: false`) provisions all five tools. +- Generated files persist to the user-files bucket (`S3_USER_FILES_BUCKET_NAME`) and surface in the session's Files panel. + +### Frontend + +- The generic `file-download-renderer` component (introduced in v1.10.0 for Word/Excel) now also handles `.pptx`, so generated presentations render through the same inline download card. + +## File Workspace toolset + +Gives the agent a durable, generic file surface over a conversation's workspace — distinct from the format-specific office tools. The model can enumerate what files exist, read uploaded text files on demand instead of front-loading them into context, and save text deliverables back to the conversation. + +### Backend + +- `agents/builtin_tools/workspace_tools.py` — three tools: `workspace_list`, `workspace_read`, `workspace_write`. Reads uploaded text files on demand and writes text deliverables (Markdown, CSV, JSON) to the user-files store, where they appear in the chat Files panel with a download link. +- `apis/shared/files/workspace.py` (430+ lines) — new shared module implementing the workspace read/write surface over the user-files store. +- `apis/shared/feature_flags.py` — `workspace_tools_enabled()` gates the feature per environment via `WORKSPACE_TOOLS_ENABLED` (**default ON, kill switch** — only the literal `false` disables). This is independent of the `workspace_files` catalog entry, which governs *who* may use the tools via RBAC. +- `apis/shared/files/models.py` — file records gain a display-only `source` provenance field ("upload" or the id of the tool that produced the file); never part of an access decision. +- `apis/inference_api/chat/routes.py` — `_build_workspace_tools` injects the set when the catalog toggle is enabled. One catalog entry ("File Workspace", gate key `workspace_files`, `enabledByDefault: false`). + +### Test Coverage + +430+ lines of new tests across `tests/agents/builtin_tools/test_workspace_tools.py` and `tests/shared/test_workspace.py` covering the tool surface and the workspace store. Design captured in `docs/specs/session-workspace-tools.md`. + +## 🐛 Bug fixes + +- **Markdown artifacts rendered as raw source.** `create_artifact`'s `content_type` defaults to `text/html`, so a request like "make a markdown recipe" easily produced raw Markdown stored under the HTML type — which then rendered as run-together `#`/`**` source instead of a formatted document. The service now reclassifies HTML-typed content that lacks a full HTML document shell (`` / ``) as Markdown, which the writer wraps into a proper render document; the tool guidance also now steers prose deliverables (reports, articles, notes, recipes) to Markdown mode. Non-HTML and genuine HTML-document content pass through untouched (#720) +- **Empty "Thinking" blocks appeared on reload.** Some models (e.g. Sonnet 5) persist a signature-only `reasoningContent` block — empty `reasoningText.text`, no redacted content — which Bedrock requires kept in the message for follow-up calls. The live stream parser already guarded on this, but the history-rehydration path bypassed it and painted an empty "Thinking" collapsible on reload. A `hasRenderableReasoning()` guard now mirrors the component's own visibility logic and the parser's guard, so the block is only painted when it has reasoning text or redacted content. Display-only: the block is left untouched in the persisted message to preserve the signature/prompt-cache contract (#721) + +## 🚀 Deployment notes + +- **No CDK deploy needed** — no new AWS resources, no infrastructure changes, no dependency changes (python-pptx runs in the sandboxed Code Interpreter, like openpyxl for Excel). Run `backend.yml` (app-api / inference-api) and `frontend-deploy.yml` as usual. +- **Seed and grant the two new tool catalog entries per environment** — "PowerPoint Presentations" (gate key `create_powerpoint_presentation`) and "File Workspace" (gate key `workspace_files`) ship in the bootstrap seed data with `enabledByDefault: false`. Environments seeded before this release won't have the rows: add them via the admin Tools page (or re-run the tools seeding) and grant them to the appropriate roles via RBAC. +- **File Workspace kill switch** — `WORKSPACE_TOOLS_ENABLED` defaults ON; no configuration is required to enable the feature. Set it to `false` on the inference-api environment to disable the workspace tools entirely for an environment, independent of the catalog grant. + +--- + # Release Notes — v1.10.0 **Release Date:** July 21, 2026 diff --git a/VERSION b/VERSION index 81c871de4..1cac385c6 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -1.10.0 +1.11.0 diff --git a/backend/pyproject.toml b/backend/pyproject.toml index f153f28a4..c44a9e031 100644 --- a/backend/pyproject.toml +++ b/backend/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "agentcore-stack" -version = "1.10.0" +version = "1.11.0" requires-python = ">=3.10" description = "Multi-agent conversational AI system with AWS Bedrock AgentCore" readme = "README.md" diff --git a/backend/scripts/seed_bootstrap_data.py b/backend/scripts/seed_bootstrap_data.py index 26f9fb06c..8477eb543 100644 --- a/backend/scripts/seed_bootstrap_data.py +++ b/backend/scripts/seed_bootstrap_data.py @@ -442,6 +442,21 @@ def seed_default_models( "isPublic": True, "forwardAuthToken": False, }, + { + # Single catalog entry / toggle that provisions the whole workspace + # toolset. Enabling this one id injects list/read/write at runtime — + # see WORKSPACE_TOOL_IDS and _build_workspace_tools in + # apis/inference_api/chat/routes.py. Keep the toolId as + # "workspace_files": it is the gate key. + "toolId": "workspace_files", + "displayName": "File Workspace", + "description": "List, read, and save files in the conversation's workspace. Reads uploaded text files on demand and saves text deliverables (markdown, CSV, JSON) to the chat's Files with a download link.", + "category": "document", + "protocol": "local", + "enabledByDefault": False, + "isPublic": True, + "forwardAuthToken": False, + }, { # Single catalog entry / toggle that provisions the whole Excel # spreadsheet toolset. Enabling this one id injects create/modify/list/ @@ -459,6 +474,21 @@ def seed_default_models( "isPublic": True, "forwardAuthToken": False, }, + { + # Single catalog entry / toggle that provisions the whole PowerPoint + # presentation toolset. Enabling this one id injects create/modify/list/ + # read at runtime — see POWERPOINT_PRESENTATION_TOOL_IDS and + # _build_powerpoint_presentation_tools in apis/inference_api/chat/routes.py. + # Keep the toolId as "create_powerpoint_presentation": it is the gate key. + "toolId": "create_powerpoint_presentation", + "displayName": "PowerPoint Presentations", + "description": "Create, edit, read, and list PowerPoint (.pptx) presentations using python-pptx in a sandboxed environment. Generated files are saved to the chat's Files with a download link.", + "category": "document", + "protocol": "local", + "enabledByDefault": False, + "isPublic": True, + "forwardAuthToken": False, + }, ] diff --git a/backend/src/agents/builtin_tools/artifacts/service.py b/backend/src/agents/builtin_tools/artifacts/service.py index 86a5cb232..a47c3e264 100644 --- a/backend/src/agents/builtin_tools/artifacts/service.py +++ b/backend/src/agents/builtin_tools/artifacts/service.py @@ -43,6 +43,7 @@ # for a Markdown artifact: a self-contained HTML render wrapper. _RENDERED_CONTENT_TYPE = "text/html; charset=utf-8" _MARKDOWN_MIME_TYPES = frozenset({"text/markdown", "text/x-markdown"}) +_HTML_MIME_TYPES = frozenset({"text/html", "application/xhtml+xml"}) # Markdown is base64-embedded so no character ever needs HTML/JS escaping # and there is no second network fetch (the artifact-origin CSP sets @@ -132,10 +133,47 @@ """ +def _bare_type(content_type: Optional[str]) -> str: + """MIME type with any `; charset=` suffix stripped, lowercased.""" + return (content_type or "").split(";")[0].strip().lower() + + def _is_markdown(content_type: Optional[str]) -> bool: """True for a Markdown MIME type, ignoring any `; charset=` suffix.""" - bare = (content_type or "").split(";")[0].strip().lower() - return bare in _MARKDOWN_MIME_TYPES + return _bare_type(content_type) in _MARKDOWN_MIME_TYPES + + +def _looks_like_html_document(content: str) -> bool: + """True if `content` opens like a full standalone HTML document. + + The create_artifact contract requires HTML artifacts to be a complete + document (`` + a full `…`), which is what + the sandboxed iframe needs to render. + """ + head = content.lstrip()[:1024].lower() + return head.startswith(" str: + """Safety net for HTML-typed content that isn't an HTML document. + + `content_type` defaults to `text/html`, so a request like "make a + markdown recipe" easily produces raw Markdown authored under the HTML + type. Stored verbatim as HTML, that renders as run-together `#`/`**` + source instead of a formatted document. When HTML-typed content lacks + the required document shell it is almost always Markdown the model + forgot to type, so reclassify it as Markdown (which the writer wraps + into a proper render document). Markdown and non-HTML types pass + through untouched. + """ + if _bare_type(content_type) not in _HTML_MIME_TYPES: + return content_type + if _looks_like_html_document(content): + return content_type + logger.info( + "reclassifying HTML-typed artifact as markdown (no HTML document shell)" + ) + return "text/markdown" def _wrap_markdown(title: str, markdown: str) -> str: @@ -265,7 +303,7 @@ def create_artifact_record( """Create v1 of a new artifact. Returns (artifact_id, version).""" artifact_id = uuid.uuid4().hex version = 1 - content_type = content_type or _DEFAULT_CONTENT_TYPE + content_type = _coerce_content_type(content, content_type or _DEFAULT_CONTENT_TYPE) now = _now_iso() content_key = _put_object( user_id, artifact_id, version, content, content_type, title @@ -337,7 +375,9 @@ def update_artifact_record( current = int(head["version"]) version = current + 1 title = title or head.get("title", "") - content_type = content_type or head.get("content_type") or _DEFAULT_CONTENT_TYPE + content_type = _coerce_content_type( + content, content_type or head.get("content_type") or _DEFAULT_CONTENT_TYPE + ) now = _now_iso() content_key = _put_object( user_id, artifact_id, version, content, content_type, title diff --git a/backend/src/agents/builtin_tools/artifacts/tools.py b/backend/src/agents/builtin_tools/artifacts/tools.py index a202e0349..ef247ff32 100644 --- a/backend/src/agents/builtin_tools/artifacts/tools.py +++ b/backend/src/agents/builtin_tools/artifacts/tools.py @@ -32,6 +32,13 @@ async def create_artifact( will want to view, keep, or iterate on — an HTML page, a chart, an interactive widget, a formatted report, or a written document. + Choosing a mode: if the deliverable is prose — a report, an + article, notes, documentation, a recipe, any mostly-text + document — author it as Markdown (`content_type="text/markdown"`), + NOT HTML. Only reach for HTML when you genuinely need custom + layout, styling, charts, or interactivity. When the user asks for + "markdown", always use the Markdown mode. + Two authoring modes: - HTML (default): `content` MUST be a complete standalone HTML diff --git a/backend/src/agents/builtin_tools/powerpoint_presentation_tool.py b/backend/src/agents/builtin_tools/powerpoint_presentation_tool.py new file mode 100644 index 000000000..b64d988e3 --- /dev/null +++ b/backend/src/agents/builtin_tools/powerpoint_presentation_tool.py @@ -0,0 +1,750 @@ +"""PowerPoint presentation tools (create / modify / list / read). + +Each tool runs python-pptx code inside AWS Bedrock Code Interpreter and uses the +existing user-files store (``apis.shared.files``) for persistence and delivery +— generated/modified ``.pptx`` files land in ``S3_USER_FILES_BUCKET_NAME`` with +a ``FileMetadata`` row (status READY) in ``DYNAMODB_USER_FILES_TABLE_NAME``, so +they appear in the chat's Files panel and are downloadable via the app-api +``/files/{id}/preview-url`` route. + +Tools +----- +* ``create_powerpoint_presentation`` — build a new deck from python-pptx code. +* ``modify_powerpoint_presentation`` — edit an existing deck with python-pptx. +* ``list_powerpoint_presentations`` — list the .pptx files in this chat. +* ``read_powerpoint_presentation`` — extract a deck's slide text + notes. + +(A slide-screenshot/preview tool is intentionally omitted: rasterizing a .pptx +requires LibreOffice/poppler, which the Python-only Code Interpreter sandbox +does not provide — the same reason the Word toolset omits one.) + +Design notes +------------ +* The Code Interpreter + user-files storage plumbing is shared with the Word + and Excel toolsets and lives in ``builtin_tools.office._storage``; this module + keeps only the python-pptx specifics (preamble, generate/modify/extract) and + the four tool factories. +* Identity (``user_id`` / ``session_id``) is captured by closure via the + ``make_*`` factories — the same pattern used by the artifacts, Word document, + and Excel spreadsheet tools (the Strands runtime here does NOT populate + ``ToolContext.invocation_state`` with identity). The tools are injected + per-request through ``extra_tools`` (see ``_build_powerpoint_presentation_tools`` + in ``apis/inference_api/chat/routes.py``); they are deliberately NOT registered + in ``builtin_tools/__init__`` because they need request-scoped identity. +""" + +from __future__ import annotations + +import asyncio +import logging +from typing import Any, Dict, Optional + +from strands import tool + +from agents.builtin_tools.office._storage import ( + _DocGenError, + _ci_exec, + _ci_read_bytes, + _ci_write_bytes, + _download_card, + _download_s3_bytes, + _error, + _get_code_interpreter_id, + _NO_CI_MESSAGE, + _region, + _storage_configured, + _store_document, + _validate_document_name, +) + +logger = logging.getLogger(__name__) + +# PowerPoint presentation MIME type (matches apis.shared.files.ALLOWED_MIME_TYPES). +_PPTX_MIME = ( + "application/vnd.openxmlformats-officedocument.presentationml.presentation" +) + +# Sandbox path used to stage a source presentation loaded from S3. +_SANDBOX_SOURCE = "_source.pptx" + +# Sandbox path used to stage a template presentation loaded from S3. +_SANDBOX_TEMPLATE = "_template.pptx" + +_NO_STORAGE_MESSAGE = ( + "❌ PowerPoint presentation storage is not configured " + "(S3_USER_FILES_BUCKET_NAME is not set on the runtime)." +) + + +# --------------------------------------------------------------------------- +# python-pptx presentation builders (run in Code Interpreter) +# --------------------------------------------------------------------------- + + +_PPTX_PREAMBLE = ( + "from pptx import Presentation\n" + "from pptx.util import Inches, Pt, Emu\n" + "from pptx.dml.color import RGBColor\n" + "from pptx.enum.text import PP_ALIGN, MSO_ANCHOR\n" + "from pptx.enum.shapes import MSO_SHAPE\n" +) + + +def _generate_pptx_bytes( + code_interpreter_id: str, + python_code: str, + filename: str, + template_bytes: Optional[bytes] = None, +) -> bytes: + """Build a new .pptx from user code and return its bytes. + + When ``template_bytes`` is given, the deck is built on top of that template + so generated slides inherit its slide masters, layouts, theme colors, fonts, + and any master/layout branding (logo, footer). Otherwise a blank 16:9 deck + is used. + + Blocking (boto3 / Code Interpreter) — call via ``asyncio.to_thread``. + """ + from bedrock_agentcore.tools.code_interpreter_client import CodeInterpreter + + code_interpreter = CodeInterpreter(_region()) + code_interpreter.start(identifier=code_interpreter_id) + try: + if template_bytes is not None: + # Start from the uploaded template, then strip its own example + # slides — removing the slide-id references leaves the slide + # masters and layouts (and their theme/branding) intact, so the + # deck starts themed-but-empty and the model's code adds fresh + # slides via the template's layouts (``prs.slide_layouts[...]``). + # Slide size is inherited from the template (not forced to 16:9). + _ci_write_bytes(code_interpreter, _SANDBOX_TEMPLATE, template_bytes) + init = ( + f"prs = Presentation({_SANDBOX_TEMPLATE!r})\n" + "for _sid in list(prs.slides._sldIdLst):\n" + " prs.slides._sldIdLst.remove(_sid)\n" + ) + else: + init = ( + "prs = Presentation()\n" + "prs.slide_width = Inches(13.333)\n" + "prs.slide_height = Inches(7.5)\n" + ) + # The user's code operates on a pre-initialized ``prs`` and must not + # call Presentation()/prs.save() itself — we own the lifecycle. + _ci_exec( + code_interpreter, + ( + f"{_PPTX_PREAMBLE}\n" + f"{init}\n" + f"{python_code}\n\n" + f"prs.save({filename!r})\n" + ), + ) + data = _ci_read_bytes(code_interpreter, filename) + if data is None: + raise _DocGenError( + f"Presentation '{filename}' was not produced. Make sure your " + "code adds slides to `prs`." + ) + return data + finally: + try: + code_interpreter.stop() + except Exception: # pragma: no cover - cleanup best-effort + pass + + +def _modify_pptx_bytes( + code_interpreter_id: str, + source_bytes: bytes, + python_code: str, + output_filename: str, +) -> bytes: + """Load an existing .pptx, apply user edits, return the new bytes. + + Blocking — call via ``asyncio.to_thread``. + """ + from bedrock_agentcore.tools.code_interpreter_client import CodeInterpreter + + code_interpreter = CodeInterpreter(_region()) + code_interpreter.start(identifier=code_interpreter_id) + try: + _ci_write_bytes(code_interpreter, _SANDBOX_SOURCE, source_bytes) + _ci_exec( + code_interpreter, + ( + f"{_PPTX_PREAMBLE}\n" + f"prs = Presentation({_SANDBOX_SOURCE!r})\n\n" + f"{python_code}\n\n" + f"prs.save({output_filename!r})\n" + ), + ) + data = _ci_read_bytes(code_interpreter, output_filename) + if data is None: + raise _DocGenError( + f"Modified presentation '{output_filename}' was not produced." + ) + return data + finally: + try: + code_interpreter.stop() + except Exception: # pragma: no cover - cleanup best-effort + pass + + +def _extract_pptx_text(code_interpreter_id: str, source_bytes: bytes) -> str: + """Extract readable text (per slide: shapes, tables, notes) from a .pptx. + + Blocking — call via ``asyncio.to_thread``. + """ + from bedrock_agentcore.tools.code_interpreter_client import CodeInterpreter + + code_interpreter = CodeInterpreter(_region()) + code_interpreter.start(identifier=code_interpreter_id) + try: + _ci_write_bytes(code_interpreter, _SANDBOX_SOURCE, source_bytes) + extraction = ( + "from pptx import Presentation\n" + f"prs = Presentation({_SANDBOX_SOURCE!r})\n" + "lines = []\n" + "for i, slide in enumerate(prs.slides):\n" + " lines.append('## Slide %d' % (i + 1))\n" + " for shape in slide.shapes:\n" + " if shape.has_text_frame:\n" + " t = shape.text_frame.text.strip()\n" + " if t:\n" + " lines.append(t)\n" + " if shape.has_table:\n" + " for row in shape.table.rows:\n" + " cells = [c.text.strip() for c in row.cells]\n" + " lines.append(' | '.join(cells))\n" + " if slide.has_notes_slide:\n" + " notes = slide.notes_slide.notes_text_frame.text.strip()\n" + " if notes:\n" + " lines.append('[Notes] ' + notes)\n" + " lines.append('')\n" + "print('\\n'.join(lines))\n" + ) + return _ci_exec(code_interpreter, extraction).strip() + finally: + try: + code_interpreter.stop() + except Exception: # pragma: no cover - cleanup best-effort + pass + + +def _extract_pptx_layouts(code_interpreter_id: str, source_bytes: bytes) -> str: + """List a presentation's slide layouts (index, name, placeholders). + + Useful before building on a template: the model can see which layout + indices exist and which placeholders each one exposes, so + ``prs.slides.add_slide(prs.slide_layouts[i])`` targets the right layout and + ``slide.placeholders[idx]`` fills the right slot. Blocking — call via + ``asyncio.to_thread``. + """ + from bedrock_agentcore.tools.code_interpreter_client import CodeInterpreter + + code_interpreter = CodeInterpreter(_region()) + code_interpreter.start(identifier=code_interpreter_id) + try: + _ci_write_bytes(code_interpreter, _SANDBOX_SOURCE, source_bytes) + extraction = ( + "from pptx import Presentation\n" + f"prs = Presentation({_SANDBOX_SOURCE!r})\n" + "layouts = prs.slide_layouts\n" + "lines = ['Total layouts: %d' % len(layouts)]\n" + "for i, layout in enumerate(layouts):\n" + " phs = []\n" + " for ph in layout.placeholders:\n" + " phs.append('%d=%s' % (ph.placeholder_format.idx, ph.name))\n" + " detail = ', '.join(phs) if phs else '(no placeholders)'\n" + " lines.append('[%d] %s | placeholders: %s' % (i, layout.name, detail))\n" + "print('\\n'.join(lines))\n" + ) + return _ci_exec(code_interpreter, extraction).strip() + finally: + try: + code_interpreter.stop() + except Exception: # pragma: no cover - cleanup best-effort + pass + + +# --------------------------------------------------------------------------- +# User-files lookup +# --------------------------------------------------------------------------- + + +async def _find_powerpoint_presentation( + user_id: str, session_id: str, presentation_name: str +): + """Find the newest READY .pptx in this session matching ``presentation_name``. + + Returns the ``FileMetadata`` or ``None``. ``list_session_files`` returns + newest-first, so the first match is the latest version. + """ + from apis.shared.files import FileStatus, get_file_upload_repository + + target = ( + presentation_name + if presentation_name.lower().endswith(".pptx") + else f"{presentation_name}.pptx" + ) + files = await get_file_upload_repository().list_session_files( + session_id, status=FileStatus.READY + ) + for meta in files: + if ( + meta.user_id == user_id + and meta.mime_type == _PPTX_MIME + and meta.filename.lower() == target.lower() + ): + return meta + return None + + +# --------------------------------------------------------------------------- +# Tool factories +# --------------------------------------------------------------------------- + + +def make_create_powerpoint_presentation_tool(session_id: str, user_id: str): + """Create a ``create_powerpoint_presentation`` tool bound to the identity.""" + + @tool + async def create_powerpoint_presentation( + python_code: str, + presentation_name: str, + template_name: Optional[str] = None, + ) -> Any: + """Create a new PowerPoint (.pptx) presentation using python-pptx code. + + Executes python-pptx code in a sandboxed Code Interpreter to build a + 16:9 widescreen deck, saves it to the user's files, and returns a + download card. Great for pitch decks, reports, and summaries with + titled slides, bullet content, tables, and embedded charts. + + Optionally builds on a template (see ``template_name``) so the deck + inherits a branded theme, fonts, and layouts instead of the plain + default. Prefer a template when the user has uploaded one or asked for + their branding. + + Available libraries in the sandbox: python-pptx, matplotlib, pandas, + numpy. + + Args: + python_code: python-pptx code that builds the deck. A blank 16:9 + presentation is already available as ``prs = Presentation()`` + (slide size preset to 13.33" x 7.5") — do NOT call + ``Presentation()`` or ``prs.save()`` yourself; the tool saves it + for you. ``Inches``, ``Pt``, ``Emu``, ``RGBColor``, ``PP_ALIGN``, + ``MSO_ANCHOR`` and ``MSO_SHAPE`` are already imported. + + Add slides from the built-in layouts (0=title, 1=title+content, + 5=title only, 6=blank), e.g.: + slide = prs.slides.add_slide(prs.slide_layouts[0]) + slide.shapes.title.text = 'Quarterly Review' + slide.placeholders[1].text = 'FY2026 — Q4' + + A dark title slide with a custom text box: + slide = prs.slides.add_slide(prs.slide_layouts[6]) + slide.background.fill.solid() + slide.background.fill.fore_color.rgb = RGBColor(0x1E, 0x27, 0x61) + box = slide.shapes.add_textbox(Inches(0.8), Inches(2.6), Inches(11.7), Inches(2)) + tf = box.text_frame + tf.text = 'Product Strategy' + r = tf.paragraphs[0].runs[0] + r.font.size, r.font.bold = Pt(44), True + r.font.color.rgb = RGBColor(0xFF, 0xFF, 0xFF) + + A bullet content slide (keep to <= 4 bullets): + slide = prs.slides.add_slide(prs.slide_layouts[1]) + slide.shapes.title.text = 'Highlights' + body = slide.placeholders[1].text_frame + body.text = 'Revenue up 15%' + p = body.add_paragraph(); p.text = 'Churn down to 2%' + + A table: + tbl = slide.shapes.add_table(2, 2, Inches(1), Inches(2), + Inches(8), Inches(2)).table + tbl.cell(0, 0).text = 'Quarter'; tbl.cell(0, 1).text = 'Revenue' + + A matplotlib chart image: + import matplotlib.pyplot as plt + plt.figure(figsize=(8, 4.5)) + plt.bar(['Q1', 'Q2'], [100, 120]) + plt.savefig('chart.png', dpi=200, bbox_inches='tight') + plt.close() + slide.shapes.add_picture('chart.png', Inches(2.5), Inches(1.5), width=Inches(8)) + + Design guidance (aim for a polished, cohesive deck): + - Pick ONE color palette for the whole deck and reuse it. Good + options (primary / secondary / accent hex): + Midnight Executive 1E2761 / CADCFC / FFFFFF; + Teal Trust 028090 / 00A896 / 02C39A; + Forest & Moss 2C5F2D / 97BC62 / F5F5F5; + Coral Energy F96167 / F9E795 / 2F3C7E; + Charcoal Minimal 36454F / F2F2F2 / 212121. + One color dominates (60-70%); dark backgrounds for the + title/closing slides, light for content. Never plain white, + never default PowerPoint blue. + - Typography: titles 36-44pt bold, section headers 20-24pt, + body 14-16pt, big stat callouts 48-120pt. Left-align body + text; center only titles and stats. + - Every slide should carry a visual element (a shape, colored + accent bar, icon circle, table, or chart) — avoid text-only + slides, and vary the layout slide to slide. + + presentation_name: File name WITHOUT extension (.pptx is added + automatically). Use only letters, numbers, hyphens, and + underscores (e.g. "sales-deck", "Q4_review"). + + template_name: Optional name of a .pptx template already available + in this chat (an uploaded deck or a previously generated one, + with or without the .pptx extension). When given, the new deck + is built ON TOP of that template so it inherits the template's + slide masters, layouts, theme colors, fonts, and any master- or + layout-level branding (logo, footer). The template's own example + slides are stripped first, so you start themed-but-empty. + + When using a template: + - Add slides from the TEMPLATE's layouts, e.g. + ``slide = prs.slides.add_slide(prs.slide_layouts[1])``, and + fill the layout's placeholders (``slide.shapes.title``, + ``slide.placeholders[idx]``) rather than drawing everything + from scratch — that's what preserves the branded look. + - Do NOT override slide backgrounds/fonts with the palette + below; let the template's theme drive the design. + - Use ``list_powerpoint_presentations`` to see available names, + and ``read_powerpoint_presentation`` to inspect a template's + existing content if helpful. + + Returns: + An inline download card. The presentation is also saved to this + chat's Files. + """ + is_valid, error_msg = _validate_document_name(presentation_name) + if not is_valid: + return _error( + f"❌ Invalid presentation name '{presentation_name}': {error_msg}\n\n" + "Examples: sales-deck, Q4_review, pitch-final" + ) + + filename = f"{presentation_name}.pptx" + code_interpreter_id = _get_code_interpreter_id() + if not code_interpreter_id: + return _error(_NO_CI_MESSAGE) + if not _storage_configured(): + return _error(_NO_STORAGE_MESSAGE) + + # Resolve an optional template from this chat's files. When provided, + # the deck is built on top of it so it inherits the template's theme. + template_bytes = None + if template_name: + template_src = await _find_powerpoint_presentation( + user_id, session_id, template_name + ) + if template_src is None: + return _error( + f"❌ No PowerPoint template named '{template_name}' was found " + "in this chat. Upload a .pptx template first, or use " + "list_powerpoint_presentations to see what's available." + ) + try: + template_bytes = await asyncio.to_thread( + _download_s3_bytes, template_src.s3_bucket, template_src.s3_key + ) + except Exception as exc: # noqa: BLE001 - surface storage errors + logger.error(f"create_powerpoint_presentation template load error: {exc}") + return _error( + f"❌ Failed to load template '{template_src.filename}': {exc}" + ) + + try: + file_bytes = await asyncio.to_thread( + _generate_pptx_bytes, + code_interpreter_id, + python_code, + filename, + template_bytes, + ) + except _DocGenError as exc: + return _error( + f"❌ Failed to create '{filename}'.\n\n```\n{exc}\n```\n\n" + "Check the python-pptx code for errors." + ) + except Exception as exc: # noqa: BLE001 - surface any sandbox error + logger.error(f"create_powerpoint_presentation sandbox error: {exc}") + return _error(f"❌ Failed to create '{filename}': {exc}") + + try: + _id, download_url, size_kb = await _store_document( + user_id, session_id, filename, file_bytes, _PPTX_MIME + ) + except Exception as exc: # noqa: BLE001 - storage failure is terminal + logger.error(f"create_powerpoint_presentation storage error: {exc}") + return _error(f"❌ Created '{filename}' but failed to save it: {exc}") + + return _download_card(filename, download_url, size_kb, "Created") + + return create_powerpoint_presentation + + +def make_modify_powerpoint_presentation_tool(session_id: str, user_id: str): + """Create a ``modify_powerpoint_presentation`` tool bound to the identity.""" + + @tool + async def modify_powerpoint_presentation( + presentation_name: str, + python_code: str, + output_name: Optional[str] = None, + ) -> Any: + """Modify an existing PowerPoint (.pptx) presentation with python-pptx code. + + Loads a deck previously created in this chat, runs your python-pptx code + against it, and saves the result (as a new file so the original is + preserved). Returns a download card. + + Use ``list_powerpoint_presentations`` first if you are unsure of the + exact name. + + Args: + presentation_name: Name of the existing deck to edit (with or + without the .pptx extension), e.g. "sales-deck". + python_code: python-pptx code that edits the deck. The loaded + presentation is available as ``prs = Presentation(...)`` — do + NOT call ``Presentation()`` or ``prs.save()`` yourself. Existing + slides are ``prs.slides``; add new ones with + ``prs.slides.add_slide(prs.slide_layouts[...])``. ``Inches``, + ``Pt``, ``Emu``, ``RGBColor``, ``PP_ALIGN``, ``MSO_ANCHOR`` and + ``MSO_SHAPE`` are already imported. + + Example (append a closing slide): + slide = prs.slides.add_slide(prs.slide_layouts[5]) + slide.shapes.title.text = 'Thank You' + + Example (edit the first slide's title): + prs.slides[0].shapes.title.text = 'Updated Title' + + output_name: Optional name (without extension) for the edited copy. + Defaults to the source name (a new versioned copy is saved). + + Returns: + An inline download card for the edited presentation. + """ + code_interpreter_id = _get_code_interpreter_id() + if not code_interpreter_id: + return _error(_NO_CI_MESSAGE) + if not _storage_configured(): + return _error(_NO_STORAGE_MESSAGE) + + source = await _find_powerpoint_presentation( + user_id, session_id, presentation_name + ) + if source is None: + return _error( + f"❌ No PowerPoint presentation named '{presentation_name}' was " + "found in this chat. Use list_powerpoint_presentations to see " + "what's available." + ) + + out_base = output_name or source.filename + if out_base.lower().endswith(".pptx"): + out_base = out_base[: -len(".pptx")] + is_valid, error_msg = _validate_document_name(out_base) + if not is_valid: + return _error( + f"❌ Invalid output name '{out_base}': {error_msg}" + ) + output_filename = f"{out_base}.pptx" + + try: + source_bytes = await asyncio.to_thread( + _download_s3_bytes, source.s3_bucket, source.s3_key + ) + file_bytes = await asyncio.to_thread( + _modify_pptx_bytes, + code_interpreter_id, + source_bytes, + python_code, + output_filename, + ) + except _DocGenError as exc: + return _error( + f"❌ Failed to modify '{source.filename}'.\n\n```\n{exc}\n```\n\n" + "Check the python-pptx code for errors." + ) + except Exception as exc: # noqa: BLE001 - surface any sandbox error + logger.error(f"modify_powerpoint_presentation error: {exc}") + return _error(f"❌ Failed to modify '{source.filename}': {exc}") + + try: + _id, download_url, size_kb = await _store_document( + user_id, session_id, output_filename, file_bytes, _PPTX_MIME + ) + except Exception as exc: # noqa: BLE001 - storage failure is terminal + logger.error(f"modify_powerpoint_presentation storage error: {exc}") + return _error( + f"❌ Modified '{source.filename}' but failed to save it: {exc}" + ) + + return _download_card(output_filename, download_url, size_kb, "Updated") + + return modify_powerpoint_presentation + + +def make_list_powerpoint_presentations_tool(session_id: str, user_id: str): + """Create a ``list_powerpoint_presentations`` tool bound to the identity.""" + + @tool + async def list_powerpoint_presentations() -> Dict[str, Any]: + """List the PowerPoint (.pptx) presentations available in this chat. + + Returns the file names and sizes of decks created or modified in this + conversation. Use the names with modify_powerpoint_presentation or + read_powerpoint_presentation. + """ + from apis.shared.files import FileStatus, get_file_upload_repository + + files = await get_file_upload_repository().list_session_files( + session_id, status=FileStatus.READY + ) + seen: set[str] = set() + rows = [] + for meta in files: # newest-first + if meta.user_id != user_id or meta.mime_type != _PPTX_MIME: + continue + if meta.filename in seen: + continue + seen.add(meta.filename) + rows.append(f"- {meta.filename} ({meta.size_bytes / 1024:.1f} KB)") + + if not rows: + text = ( + "No PowerPoint presentations in this chat yet. Use " + "create_powerpoint_presentation to make one." + ) + else: + text = "PowerPoint presentations in this chat:\n" + "\n".join(rows) + return {"content": [{"text": text}], "status": "success"} + + return list_powerpoint_presentations + + +def make_read_powerpoint_presentation_tool(session_id: str, user_id: str): + """Create a ``read_powerpoint_presentation`` tool bound to the identity.""" + + @tool + async def read_powerpoint_presentation(presentation_name: str) -> Dict[str, Any]: + """Read the text content of an existing PowerPoint (.pptx) presentation. + + Extracts each slide's text, tables, and speaker notes from a deck + created in this chat so you can reference or summarize its contents. Use + list_powerpoint_presentations first if unsure of the exact name. + + Args: + presentation_name: Name of the deck to read (with or without the + .pptx extension), e.g. "sales-deck". + + Returns: + The presentation's text content, grouped by slide. + """ + code_interpreter_id = _get_code_interpreter_id() + if not code_interpreter_id: + return _error(_NO_CI_MESSAGE) + if not _storage_configured(): + return _error(_NO_STORAGE_MESSAGE) + + source = await _find_powerpoint_presentation( + user_id, session_id, presentation_name + ) + if source is None: + return _error( + f"❌ No PowerPoint presentation named '{presentation_name}' was " + "found in this chat. Use list_powerpoint_presentations to see " + "what's available." + ) + + try: + source_bytes = await asyncio.to_thread( + _download_s3_bytes, source.s3_bucket, source.s3_key + ) + text = await asyncio.to_thread( + _extract_pptx_text, code_interpreter_id, source_bytes + ) + except _DocGenError as exc: + return _error(f"❌ Failed to read '{source.filename}': {exc}") + except Exception as exc: # noqa: BLE001 - surface any sandbox error + logger.error(f"read_powerpoint_presentation error: {exc}") + return _error(f"❌ Failed to read '{source.filename}': {exc}") + + body = text or "(The presentation has no extractable text.)" + return { + "content": [ + {"text": f"Content of {source.filename}:\n\n{body}"} + ], + "status": "success", + } + + return read_powerpoint_presentation + + +def make_list_powerpoint_layouts_tool(session_id: str, user_id: str): + """Create a ``list_powerpoint_layouts`` tool bound to the identity.""" + + @tool + async def list_powerpoint_layouts(presentation_name: str) -> Dict[str, Any]: + """List the slide layouts of a PowerPoint (.pptx) file in this chat. + + Reports each layout's index, name, and placeholder slots. Call this on a + template BEFORE building on it (create_powerpoint_presentation with + template_name) so you add slides from the right layout + (``prs.slide_layouts[index]``) and fill the correct placeholders + (``slide.placeholders[idx]``) — that's what preserves the template's + branded design. Works on any .pptx available in this chat (an uploaded + template or a deck generated here). + + Args: + presentation_name: Name of the .pptx to inspect (with or without the + .pptx extension), e.g. "brand-template". + + Returns: + The layout inventory (index, name, placeholders per layout). + """ + code_interpreter_id = _get_code_interpreter_id() + if not code_interpreter_id: + return _error(_NO_CI_MESSAGE) + if not _storage_configured(): + return _error(_NO_STORAGE_MESSAGE) + + source = await _find_powerpoint_presentation( + user_id, session_id, presentation_name + ) + if source is None: + return _error( + f"❌ No PowerPoint presentation named '{presentation_name}' was " + "found in this chat. Use list_powerpoint_presentations to see " + "what's available." + ) + + try: + source_bytes = await asyncio.to_thread( + _download_s3_bytes, source.s3_bucket, source.s3_key + ) + text = await asyncio.to_thread( + _extract_pptx_layouts, code_interpreter_id, source_bytes + ) + except _DocGenError as exc: + return _error(f"❌ Failed to inspect '{source.filename}': {exc}") + except Exception as exc: # noqa: BLE001 - surface any sandbox error + logger.error(f"list_powerpoint_layouts error: {exc}") + return _error(f"❌ Failed to inspect '{source.filename}': {exc}") + + body = text or "(No layouts found.)" + return { + "content": [ + {"text": f"Layouts in {source.filename}:\n\n{body}"} + ], + "status": "success", + } + + return list_powerpoint_layouts diff --git a/backend/src/agents/builtin_tools/workspace_tools.py b/backend/src/agents/builtin_tools/workspace_tools.py new file mode 100644 index 000000000..1b15e8e54 --- /dev/null +++ b/backend/src/agents/builtin_tools/workspace_tools.py @@ -0,0 +1,190 @@ +"""Workspace tools (list / read / write) over the user-files store. + +A generic file surface for the agent: enumerate the user's files, read text +content on demand (bounded), and write text deliverables the user can +download. All storage flows through ``apis.shared.files.workspace`` — the +DynamoDB user-files table is the source of truth, so workspace files appear +in the chat's Files panel alongside uploads. + +Design notes +------------ +* Identity (``user_id`` / ``session_id``) is captured by closure via the + ``make_*`` factories — the same pattern as the artifact, spreadsheet, and + word-document tools (the Strands runtime here does NOT populate + ``ToolContext.invocation_state`` with identity). The tools are injected + per-request through ``extra_tools`` (see ``_build_workspace_tools`` in + ``apis/inference_api/chat/routes.py``); they are deliberately NOT registered + in ``builtin_tools/__init__`` because they need request-scoped identity. +* Binary files are returned by reference (presigned URL), never base64 — file + bytes do not flow through the model (token-cost tenet, CLAUDE.md). +* ``workspace_write`` returns the same inline download-card contract as the + word tool (``ui_type: "file_download"``) so the SPA renders a first-class + download card. +""" + +from __future__ import annotations + +import json +import logging +from typing import Any, Dict + +from strands import tool + +from apis.shared.files.workspace import ( + WorkspaceError, + WorkspaceStorageNotConfiguredError, + list_workspace_files, + read_workspace_file, + write_workspace_file, +) + +logger = logging.getLogger(__name__) + +_NO_STORAGE_MESSAGE = ( + "❌ File workspace storage is not configured " + "(S3_USER_FILES_BUCKET_NAME is not set on the runtime)." +) + + +def _error(text: str) -> Dict[str, Any]: + return {"content": [{"text": text}], "status": "error"} + + +def _success(payload: Dict[str, Any]) -> Dict[str, Any]: + return {"content": [{"json": payload}], "status": "success"} + + +def make_workspace_list_tool(session_id: str, user_id: str): + """Create a ``workspace_list`` tool bound to the given identity.""" + + @tool + async def workspace_list(scope: str = "session") -> Any: + """List the files available in the user's workspace. + + Covers files the user attached and files tools have produced (Word + documents, workspace writes, …). Entries with ``readable: true`` can + be opened with ``workspace_read``; others are binary and move by + reference. + + Args: + scope: "session" (default) lists files from this conversation; + "user" lists the user's files across all conversations + (newest first) — use it when the user refers to a file from + an earlier conversation. + + Returns: + Files with upload_id, filename, mime_type, size_bytes, source, + and readable. Capped at the newest ~100 entries + (``truncated: true`` when more exist). + """ + try: + result = await list_workspace_files(user_id, session_id, scope=scope) + except WorkspaceStorageNotConfiguredError: + return _error(_NO_STORAGE_MESSAGE) + except WorkspaceError as exc: + return _error(f"❌ {exc}") + except Exception as exc: # noqa: BLE001 - surface conversationally + logger.error(f"workspace_list error: {exc}") + return _error(f"❌ Failed to list workspace files: {exc}") + return _success(result) + + return workspace_list + + +def make_workspace_read_tool(session_id: str, user_id: str): + """Create a ``workspace_read`` tool bound to the given identity.""" + + @tool + async def workspace_read(upload_id: str, offset: int = 0) -> Any: + """Read a file from the user's workspace. + + Text files (plain, markdown, CSV, JSON, HTML) return their content + inline, up to ~48KB per call — when ``truncated`` is true, call again + with ``offset`` set to ``next_offset`` to continue. Binary files + (PDF, Office, images) return metadata plus a download URL instead of + content; hand tabular files to ``analyze_spreadsheet``. + + Args: + upload_id: The file's id, as returned by ``workspace_list``. + offset: Byte offset to continue a previous truncated read + (default 0). + + Returns: + For text: content, truncated flag, and next_offset. For binary: + metadata and a short-lived download URL. + """ + try: + result = await read_workspace_file(user_id, upload_id, offset=offset) + except WorkspaceStorageNotConfiguredError: + return _error(_NO_STORAGE_MESSAGE) + except WorkspaceError as exc: + return _error(f"❌ {exc}") + except Exception as exc: # noqa: BLE001 - surface conversationally + logger.error(f"workspace_read error: {exc}") + return _error(f"❌ Failed to read file '{upload_id}': {exc}") + return _success(result) + + return workspace_read + + +def make_workspace_write_tool(session_id: str, user_id: str): + """Create a ``workspace_write`` tool bound to the given identity.""" + + @tool + async def workspace_write( + filename: str, + content: str, + mime_type: str = "text/plain", + ) -> Any: + """Save a text file to the user's workspace with a download card. + + Use this for text deliverables the user should keep: markdown + reports, CSV exports, JSON data, code listings. The file is saved to + this chat's Files and presented with a download button. For Word + documents use ``create_word_document``; for interactive documents + use ``create_artifact``. + + Args: + filename: File name, a single path segment (letters, numbers, + dots, hyphens, underscores, spaces). The extension must match + ``mime_type`` and is added automatically if omitted. + content: The file content as plain text (max 1MB). + mime_type: One of text/plain (default), text/markdown, text/csv, + text/html, application/json. + + Returns: + An inline download card. The file is also saved to this chat's + Files, and each write creates a new file version (no overwrite). + """ + try: + result = await write_workspace_file( + user_id, session_id, filename, content, mime_type=mime_type + ) + except WorkspaceStorageNotConfiguredError: + return _error(_NO_STORAGE_MESSAGE) + except WorkspaceError as exc: + return _error(f"❌ {exc}") + except Exception as exc: # noqa: BLE001 - surface conversationally + logger.error(f"workspace_write error: {exc}") + return _error(f"❌ Failed to save '{filename}': {exc}") + + # Same promoted download-card contract as word_document_tool — the + # SPA routes ui_type "file_download" to the inline download card. + return json.dumps( + { + "success": True, + "ui_type": "file_download", + "ui_display": "inline", + "payload": { + "filename": result["filename"], + "download_url": result["download_url"], + "size_kb": result["size_kb"], + }, + "summary": ( + f"Saved {result['filename']} ({result['size_kb']}). " + "Also saved to this chat's Files." + ), + } + ) + + return workspace_write diff --git a/backend/src/apis/inference_api/chat/routes.py b/backend/src/apis/inference_api/chat/routes.py index b69b535d3..fe8204ea8 100644 --- a/backend/src/apis/inference_api/chat/routes.py +++ b/backend/src/apis/inference_api/chat/routes.py @@ -487,6 +487,47 @@ def _build_word_document_tools( return tools +# ============================================================ +# Workspace Tool Injection +# ============================================================ + +WORKSPACE_TOOL_IDS = {"workspace_files"} + + +def _build_workspace_tools( + enabled_tools: list | None, + session_id: str, + user_id: str, +) -> list: + """Create context-bound workspace file tools if enabled by the user. + + Identity is captured by closure (same pattern as the artifact and word + document tools). The "workspace_files" catalog entry is a single toggle + that provisions the full toolset (list/read/write). + """ + from apis.shared.feature_flags import workspace_tools_enabled + + if not workspace_tools_enabled(): + return [] + if not enabled_tools or not WORKSPACE_TOOL_IDS.intersection(enabled_tools): + return [] + + from agents.builtin_tools.workspace_tools import ( + make_workspace_list_tool, + make_workspace_read_tool, + make_workspace_write_tool, + ) + + tools = [ + make_workspace_list_tool(session_id, user_id), + make_workspace_read_tool(session_id, user_id), + make_workspace_write_tool(session_id, user_id), + ] + + logger.info(f"Created {len(tools)} workspace tools") + return tools + + # ============================================================ # Excel Spreadsheet Tool Injection # ============================================================ @@ -531,6 +572,50 @@ def _build_excel_spreadsheet_tools( return tools +# ============================================================ +# PowerPoint Presentation Tool Injection +# ============================================================ + +POWERPOINT_PRESENTATION_TOOL_IDS = {"create_powerpoint_presentation"} + + +def _build_powerpoint_presentation_tools( + enabled_tools: list | None, + session_id: str, + user_id: str, +) -> list: + """Create context-bound PowerPoint presentation tools if enabled by the user. + + Identity is captured by closure (same pattern as the Word document and Excel + spreadsheet tools) since the runtime does not populate ToolContext. + """ + if not enabled_tools or not POWERPOINT_PRESENTATION_TOOL_IDS.intersection(enabled_tools): + return [] + + # The PowerPoint capability is a single toggle: enabling + # create_powerpoint_presentation provisions the full deck toolset + # (create/modify/list/read) so the model can round-trip on a presentation + # without extra admin catalog entries. + from agents.builtin_tools.powerpoint_presentation_tool import ( + make_create_powerpoint_presentation_tool, + make_list_powerpoint_layouts_tool, + make_list_powerpoint_presentations_tool, + make_modify_powerpoint_presentation_tool, + make_read_powerpoint_presentation_tool, + ) + + tools = [ + make_create_powerpoint_presentation_tool(session_id, user_id), + make_modify_powerpoint_presentation_tool(session_id, user_id), + make_list_powerpoint_presentations_tool(session_id, user_id), + make_read_powerpoint_presentation_tool(session_id, user_id), + make_list_powerpoint_layouts_tool(session_id, user_id), + ] + + logger.info(f"Created {len(tools)} powerpoint presentation tools") + return tools + + def _build_memory_tools(agent_memory, user_id: str, user_email: str) -> list: """Context-bound Memory-Space tools for an Agent's resolved memory binding. @@ -1837,10 +1922,18 @@ async def invocations(request: InvocationRequest, current_user: User = Depends(g enabled_tools=effective_enabled_tools, session_id=input_data.session_id, user_id=user_id, + ) + _build_workspace_tools( + enabled_tools=effective_enabled_tools, + session_id=input_data.session_id, + user_id=user_id, ) + _build_excel_spreadsheet_tools( enabled_tools=effective_enabled_tools, session_id=input_data.session_id, user_id=user_id, + ) + _build_powerpoint_presentation_tools( + enabled_tools=effective_enabled_tools, + session_id=input_data.session_id, + user_id=user_id, ) + _build_memory_tools( agent_memory=agent_memory, user_id=user_id, diff --git a/backend/src/apis/shared/feature_flags.py b/backend/src/apis/shared/feature_flags.py index 2392c0f73..ca31783ea 100644 --- a/backend/src/apis/shared/feature_flags.py +++ b/backend/src/apis/shared/feature_flags.py @@ -75,6 +75,23 @@ def memory_spaces_enabled() -> bool: return os.environ.get("MEMORY_SPACES_ENABLED", "false").lower() == "true" +def workspace_tools_enabled() -> bool: + """Whether the workspace file tools are enabled for this environment. + + Covers the ``workspace_list`` / ``workspace_read`` / ``workspace_write`` + agent tools (the generic file surface over the user-files store — see + ``docs/specs/session-workspace-tools.md``). **Default ON with a kill + switch** (house style, mirroring ``scheduled_runs_enabled``): unset or + empty resolves to enabled; only the literal ``"false"`` + (case-insensitive) disables. + + Note this flag gates *feature existence* per environment; *who* may use + the tools is the ``workspace_files`` catalog entry granted via roles — + two independent controls. + """ + return os.environ.get("WORKSPACE_TOOLS_ENABLED", "").strip().lower() != "false" + + def agents_enabled() -> bool: """Whether the Agent Designer surface is enabled for this environment. diff --git a/backend/src/apis/shared/files/__init__.py b/backend/src/apis/shared/files/__init__.py index a29ba6306..de83b81e5 100644 --- a/backend/src/apis/shared/files/__init__.py +++ b/backend/src/apis/shared/files/__init__.py @@ -37,6 +37,17 @@ get_file_resolver, ) +from .workspace import ( + WorkspaceError, + WorkspaceFileNotFoundError, + WorkspaceQuotaExceededError, + WorkspaceStorageNotConfiguredError, + WorkspaceValidationError, + list_workspace_files, + read_workspace_file, + write_workspace_file, +) + __all__ = [ # Models "FileStatus", @@ -65,4 +76,13 @@ "FileResolverError", "FileResolver", "get_file_resolver", + # Workspace + "WorkspaceError", + "WorkspaceFileNotFoundError", + "WorkspaceQuotaExceededError", + "WorkspaceStorageNotConfiguredError", + "WorkspaceValidationError", + "list_workspace_files", + "read_workspace_file", + "write_workspace_file", ] diff --git a/backend/src/apis/shared/files/models.py b/backend/src/apis/shared/files/models.py index 1d0210aaf..2306814d3 100644 --- a/backend/src/apis/shared/files/models.py +++ b/backend/src/apis/shared/files/models.py @@ -33,6 +33,7 @@ class FileStatus(str, Enum): "text/csv": "csv", "application/vnd.ms-excel": "xls", "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": "xlsx", + "application/vnd.openxmlformats-officedocument.presentationml.presentation": "pptx", "text/markdown": "md", # Images (Bedrock-supported) "image/png": "png", @@ -50,6 +51,7 @@ class FileStatus(str, Enum): ".csv": "text/csv", ".xls": "application/vnd.ms-excel", ".xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", + ".pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation", ".md": "text/markdown", # Images ".png": "image/png", @@ -153,6 +155,11 @@ class FileMetadata(BaseModel): # Status status: FileStatus = Field(default=FileStatus.PENDING) + # Provenance: "upload" (SPA attachment flow) or the id of the agent tool + # that produced the file (e.g. "agent", "word_document"). Display-only — + # never part of an access decision. + source: str = Field(default="upload", description="Origin of the file") + # Timestamps created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc)) updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc)) @@ -193,6 +200,7 @@ def to_dynamo_item(self) -> dict: "s3Key": self.s3_key, "s3Bucket": self.s3_bucket, "s3Uri": self.s3_uri, + "source": self.source, "status": self.status if isinstance(self.status, str) else self.status.value, "createdAt": self.created_at.isoformat() + "Z", "updatedAt": self.updated_at.isoformat() + "Z", @@ -215,6 +223,7 @@ def from_dynamo_item(cls, item: dict) -> "FileMetadata": s3_key=item.get("s3Key", ""), s3_bucket=item.get("s3Bucket", ""), status=item.get("status", FileStatus.PENDING), + source=item.get("source", "upload"), created_at=datetime.fromisoformat(created_at.rstrip("Z")) if created_at else datetime.now(timezone.utc), updated_at=datetime.fromisoformat(updated_at.rstrip("Z")) if updated_at else datetime.now(timezone.utc), ttl=item.get("ttl"), diff --git a/backend/src/apis/shared/files/workspace.py b/backend/src/apis/shared/files/workspace.py new file mode 100644 index 000000000..d72c5627d --- /dev/null +++ b/backend/src/apis/shared/files/workspace.py @@ -0,0 +1,430 @@ +"""Session workspace service — bounded list/read/write over the user-files store. + +Backs the ``workspace_list`` / ``workspace_read`` / ``workspace_write`` agent +tools (see ``agents/builtin_tools/workspace_tools.py`` and +``docs/specs/session-workspace-tools.md``). Every operation goes through the +DynamoDB user-files table (`FileMetadata` + `FileUploadRepository`) — the table +is the source of truth, never a raw S3 listing — so workspace files appear in +the SPA Files panel and participate in quota accounting exactly like uploads. + +Hard rules encoded here: +* Reads and writes are bounded (`WORKSPACE_READ_MAX_BYTES`, + `WORKSPACE_WRITE_MAX_BYTES`) — file bytes never flow through the model + unbounded, and binary files move by reference (presigned URL) only. +* Identity is caller-supplied and mandatory: a missing ``user_id`` / + ``session_id`` raises instead of defaulting (a silent default would collapse + sessions into a shared namespace). +* Ownership is enforced by the table's key shape (``PK = USER#{userId}``); + no operation accepts a model-supplied S3 key. +""" + +from __future__ import annotations + +import asyncio +import logging +import os +import re +import uuid +from datetime import datetime, timezone +from typing import Any, Dict, List, Optional + +import boto3 +from botocore.config import Config +from botocore.exceptions import ClientError + +from .models import FileMetadata, FileStatus +from .repository import get_file_upload_repository + +logger = logging.getLogger(__name__) + +# Per-call byte cap for text reads; continuation via `offset`. +WORKSPACE_READ_MAX_BYTES = int( + os.environ.get("WORKSPACE_READ_MAX_BYTES", 48 * 1024) # 48KB +) +# Per-call byte cap for writes. Model-generated text is inherently small; the +# cap bounds runaway loops, not legitimate deliverables. +WORKSPACE_WRITE_MAX_BYTES = int( + os.environ.get("WORKSPACE_WRITE_MAX_BYTES", 1024 * 1024) # 1MB +) +# A tool result is a per-turn payload too — cap listings. +WORKSPACE_LIST_MAX_ENTRIES = int(os.environ.get("WORKSPACE_LIST_MAX_ENTRIES", 100)) + +# Same ceiling the upload flow enforces (apis/app_api/files/service.py). +_USER_QUOTA_BYTES = int( + os.environ.get("FILE_UPLOAD_USER_QUOTA_BYTES", 1024 * 1024 * 1024) # 1GB +) + +_DOWNLOAD_URL_TTL = 60 * 60 # 1 hour, matches word_document_tool + +# MIME types whose content may be returned inline as text. +_TEXT_MIME_PREFIXES = ("text/",) +_TEXT_MIME_EXACT = frozenset({"application/json"}) + +# MIME types workspace_write accepts, with their canonical extension. +WRITABLE_MIME_TYPES: Dict[str, str] = { + "text/plain": ".txt", + "text/markdown": ".md", + "text/csv": ".csv", + "text/html": ".html", + "application/json": ".json", +} + +# Filename: single path segment, no traversal, sane characters. +_FILENAME_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._\- ]{0,120}$") + + +class WorkspaceError(Exception): + """Base for workspace failures surfaced conversationally by the tools.""" + + +class WorkspaceStorageNotConfiguredError(WorkspaceError): + """S3_USER_FILES_BUCKET_NAME is not set on the runtime.""" + + +class WorkspaceFileNotFoundError(WorkspaceError): + """No READY file with that upload_id belongs to this user.""" + + +class WorkspaceQuotaExceededError(WorkspaceError): + """The write would push the user past their storage quota.""" + + +class WorkspaceValidationError(WorkspaceError): + """Bad filename / MIME type / size / offset.""" + + +def is_text_mime(mime_type: str) -> bool: + """True when the MIME type's content can be returned inline as text.""" + mt = (mime_type or "").lower().split(";")[0].strip() + return mt.startswith(_TEXT_MIME_PREFIXES) or mt in _TEXT_MIME_EXACT + + +def _require_identity(user_id: str, session_id: Optional[str] = None) -> None: + """Fail loudly on missing identity — never default to a shared namespace.""" + if not user_id: + raise WorkspaceError("workspace called without a user_id") + if session_id is not None and not session_id: + raise WorkspaceError("workspace called without a session_id") + + +def _bucket() -> str: + bucket = os.environ.get("S3_USER_FILES_BUCKET_NAME") + if not bucket: + raise WorkspaceStorageNotConfiguredError( + "S3_USER_FILES_BUCKET_NAME is not set on the runtime" + ) + return bucket + + +_s3_client = None +_bucket_region: Optional[str] = None + + +def _region() -> str: + return ( + os.environ.get("AWS_REGION") + or os.environ.get("AWS_DEFAULT_REGION") + or "us-west-2" + ) + + +def _resolve_bucket_region(bucket: str) -> str: + """Discover the user-files bucket's real region (see word_document_tool: + a client pinned to the wrong region fails PutObject with + PermanentRedirect; ``head_bucket`` reports the true region either way). + """ + global _bucket_region + if _bucket_region: + return _bucket_region + + region = None + probe = boto3.client("s3", region_name=_region()) + try: + resp = probe.head_bucket(Bucket=bucket) + region = ( + resp.get("ResponseMetadata", {}) + .get("HTTPHeaders", {}) + .get("x-amz-bucket-region") + ) + except ClientError as exc: + region = ( + exc.response.get("ResponseMetadata", {}) + .get("HTTPHeaders", {}) + .get("x-amz-bucket-region") + ) + if not region: + logger.warning(f"Could not resolve region for bucket {bucket}: {exc}") + except Exception as exc: # pragma: no cover - network edge + logger.warning(f"Could not resolve region for bucket {bucket}: {exc}") + _bucket_region = region or _region() + return _bucket_region + + +def _s3(): + """SigV4 S3 client pinned to the user-files bucket's actual region.""" + global _s3_client + if _s3_client is None: + region = _resolve_bucket_region(_bucket()) + _s3_client = boto3.client( + "s3", + region_name=region, + config=Config(signature_version="s3v4", s3={"addressing_style": "virtual"}), + ) + return _s3_client + + +def _entry(meta: FileMetadata, include_session: bool) -> Dict[str, Any]: + entry: Dict[str, Any] = { + "upload_id": meta.upload_id, + "filename": meta.filename, + "mime_type": meta.mime_type, + "size_bytes": meta.size_bytes, + "source": meta.source, + "created_at": meta.created_at.isoformat(), + "readable": is_text_mime(meta.mime_type), + } + if include_session: + entry["session_id"] = meta.session_id + return entry + + +async def list_workspace_files( + user_id: str, session_id: str, scope: str = "session" +) -> Dict[str, Any]: + """List the user's READY files — this conversation or all conversations. + + DynamoDB-only (never an S3 listing). Output is capped at + ``WORKSPACE_LIST_MAX_ENTRIES`` newest-first entries with a ``truncated`` + flag. + """ + _require_identity(user_id, session_id) + if scope not in ("session", "user"): + raise WorkspaceValidationError( + f"Unknown scope '{scope}' — use 'session' or 'user'" + ) + + repo = get_file_upload_repository() + truncated = False + + if scope == "session": + files = await repo.list_session_files(session_id, status=FileStatus.READY) + # The GSI partition is the session; enforce ownership explicitly. + files = [m for m in files if m.user_id == user_id] + if len(files) > WORKSPACE_LIST_MAX_ENTRIES: + files = files[:WORKSPACE_LIST_MAX_ENTRIES] + truncated = True + else: + files, next_cursor = await repo.list_user_files( + user_id, limit=WORKSPACE_LIST_MAX_ENTRIES, status=FileStatus.READY + ) + truncated = next_cursor is not None + + return { + "scope": scope, + "files": [_entry(m, include_session=scope == "user") for m in files], + "count": len(files), + "truncated": truncated, + } + + +async def _get_owned_ready_file(user_id: str, upload_id: str) -> FileMetadata: + meta = await get_file_upload_repository().get_file(user_id, upload_id) + if meta is None: + raise WorkspaceFileNotFoundError( + f"No file with id '{upload_id}' found in your workspace" + ) + status = meta.status if isinstance(meta.status, str) else meta.status.value + if status != FileStatus.READY.value: + raise WorkspaceFileNotFoundError( + f"No file with id '{upload_id}' found in your workspace" + ) + return meta + + +def _ranged_get(bucket: str, key: str, offset: int, length: int) -> bytes: + """Blocking ranged S3 GET — only `length` bytes ever enter memory.""" + resp = _s3().get_object( + Bucket=bucket, Key=key, Range=f"bytes={offset}-{offset + length - 1}" + ) + return resp["Body"].read() + + +async def read_workspace_file( + user_id: str, upload_id: str, offset: int = 0 +) -> Dict[str, Any]: + """Read a file's content (text, bounded) or mint a reference (binary). + + Text MIME types return up to ``WORKSPACE_READ_MAX_BYTES`` UTF-8 bytes from + ``offset`` with ``truncated`` + ``next_offset`` for continuation. All other + types return metadata plus a short-lived presigned GET URL — never base64. + """ + _require_identity(user_id) + if offset < 0: + raise WorkspaceValidationError("offset must be >= 0") + + meta = await _get_owned_ready_file(user_id, upload_id) + base = { + "upload_id": meta.upload_id, + "filename": meta.filename, + "mime_type": meta.mime_type, + "size_bytes": meta.size_bytes, + "source": meta.source, + } + + if not is_text_mime(meta.mime_type): + url = await asyncio.to_thread( + _s3().generate_presigned_url, + "get_object", + Params={"Bucket": meta.s3_bucket, "Key": meta.s3_key}, + ExpiresIn=_DOWNLOAD_URL_TTL, + ) + return { + **base, + "encoding": "reference", + "download_url": url, + "note": ( + "Binary file — content is not returned inline. Use the URL for " + "delivery, analyze_spreadsheet for tabular data, or the code " + "interpreter for byte-level processing." + ), + } + + if offset >= meta.size_bytes: + raise WorkspaceValidationError( + f"offset {offset} is beyond the end of the file " + f"({meta.size_bytes} bytes)" + ) + + data = await asyncio.to_thread( + _ranged_get, meta.s3_bucket, meta.s3_key, offset, WORKSPACE_READ_MAX_BYTES + ) + end = offset + len(data) + truncated = end < meta.size_bytes + return { + **base, + "encoding": "text", + "content": data.decode("utf-8", errors="replace"), + "offset": offset, + "truncated": truncated, + "next_offset": end if truncated else None, + } + + +def _validate_filename(filename: str, mime_type: str) -> str: + """Sanitize the filename and reconcile its extension with the MIME type. + + Returns the final filename. A missing extension gets the MIME type's + canonical one appended; a mismatched extension is rejected. + """ + if mime_type not in WRITABLE_MIME_TYPES: + allowed = ", ".join(sorted(WRITABLE_MIME_TYPES)) + raise WorkspaceValidationError( + f"Unsupported mime_type '{mime_type}'. Workspace writes are " + f"text-only: {allowed}. Use the dedicated document tools for " + "binary formats." + ) + if not filename or not _FILENAME_RE.match(filename) or ".." in filename: + raise WorkspaceValidationError( + f"Invalid filename '{filename}'. Use a single name (no path " + "separators) with letters, numbers, dots, hyphens, underscores, " + "or spaces." + ) + + expected_ext = WRITABLE_MIME_TYPES[mime_type] + root, dot, ext = filename.rpartition(".") + if not root: + return f"{filename}{expected_ext}" + if f".{ext.lower()}" != expected_ext: + raise WorkspaceValidationError( + f"Filename extension '.{ext}' does not match mime_type " + f"'{mime_type}' (expected '{expected_ext}')" + ) + return filename + + +async def write_workspace_file( + user_id: str, + session_id: str, + filename: str, + content: str, + mime_type: str = "text/plain", + source: str = "agent", +) -> Dict[str, Any]: + """Write a text file into the current session's workspace. + + Mirrors the canonical agent write path + (``word_document_tool._store_document``): put_object under the session's + prefix → READY ``FileMetadata`` row → quota increment → presigned + download URL. Each write is a new ``upload_id`` (no in-place overwrite); + listings return newest-first, so a same-name write supersedes. + """ + _require_identity(user_id, session_id) + final_name = _validate_filename(filename, mime_type) + + data = content.encode("utf-8") + if len(data) > WORKSPACE_WRITE_MAX_BYTES: + raise WorkspaceValidationError( + f"Content is {len(data)} bytes; the per-write limit is " + f"{WORKSPACE_WRITE_MAX_BYTES} bytes" + ) + + repo = get_file_upload_repository() + quota = await repo.get_user_quota(user_id) + if quota.total_bytes + len(data) > _USER_QUOTA_BYTES: + raise WorkspaceQuotaExceededError( + f"Storage quota exceeded ({quota.total_bytes} of " + f"{_USER_QUOTA_BYTES} bytes used)" + ) + + bucket = _bucket() + timestamp_hex = format(int(datetime.now(timezone.utc).timestamp() * 1000), "x") + upload_id = f"{timestamp_hex}_{uuid.uuid4().hex[:16]}" + s3_key = f"user-files/{user_id}/{session_id}/{upload_id}/{final_name}" + + await asyncio.to_thread( + _s3().put_object, + Bucket=bucket, + Key=s3_key, + Body=data, + ContentType=mime_type, + ) + + metadata = FileMetadata( + upload_id=upload_id, + user_id=user_id, + session_id=session_id, + filename=final_name, + mime_type=mime_type, + size_bytes=len(data), + s3_key=s3_key, + s3_bucket=bucket, + status=FileStatus.READY, + source=source, + ) + await repo.create_file(metadata) + await repo.increment_quota(user_id, len(data)) + + download_url = await asyncio.to_thread( + _s3().generate_presigned_url, + "get_object", + Params={ + "Bucket": bucket, + "Key": s3_key, + "ResponseContentType": mime_type, + "ResponseContentDisposition": f'attachment; filename="{final_name}"', + }, + ExpiresIn=_DOWNLOAD_URL_TTL, + ) + + logger.info( + f"[workspace_write] {len(data)} bytes → {final_name} " + f"(upload_id={upload_id}, source={source})" + ) + return { + "upload_id": upload_id, + "filename": final_name, + "mime_type": mime_type, + "size_bytes": len(data), + "size_kb": f"{len(data) / 1024:.1f} KB", + "download_url": download_url, + } diff --git a/backend/tests/agents/builtin_tools/artifacts/test_artifact_tools.py b/backend/tests/agents/builtin_tools/artifacts/test_artifact_tools.py index 02d6cceb4..152e1b324 100644 --- a/backend/tests/agents/builtin_tools/artifacts/test_artifact_tools.py +++ b/backend/tests/agents/builtin_tools/artifacts/test_artifact_tools.py @@ -188,6 +188,60 @@ def test_markdown_update_rewraps_inherited_type(aws) -> None: assert _embedded_markdown(body) == new_md +def test_markdown_content_under_default_type_is_reclassified(aws) -> None: + """Regression: raw Markdown authored under the default HTML type (the + model forgot content_type="text/markdown") must be treated as Markdown + so it renders, not stored verbatim as run-together HTML source.""" + ddb, s3 = aws + aid, _ = service.create_artifact_record(USER, SESSION, "Recipe", MD, "") + + # Row is corrected to Markdown → card badge reads "MD", not "HTML". + assert _item(ddb, aid, "V#00001")["content_type"] == "text/markdown" + + body = s3.get_object( + Bucket=BUCKET, Key=f"{USER}/{aid}/v1/index.html" + )["Body"].read().decode() + assert body.lstrip().startswith("") + assert "https://esm.sh/marked@14.1.4" in body + assert _embedded_markdown(body) == MD + + +def test_markdown_content_under_explicit_html_type_is_reclassified(aws) -> None: + ddb, s3 = aws + aid, _ = service.create_artifact_record(USER, SESSION, "Recipe", MD, "text/html") + assert _item(ddb, aid, "V#00001")["content_type"] == "text/markdown" + body = s3.get_object( + Bucket=BUCKET, Key=f"{USER}/{aid}/v1/index.html" + )["Body"].read().decode() + assert _embedded_markdown(body) == MD + + +def test_full_html_document_not_reclassified(aws) -> None: + """A genuine standalone HTML document keeps its HTML type and is + stored verbatim (the safety net must not touch real HTML).""" + ddb, s3 = aws + aid, _ = service.create_artifact_record(USER, SESSION, "Page", DOC, "text/html") + assert _item(ddb, aid, "V#00001")["content_type"] == "text/html" + assert s3.get_object( + Bucket=BUCKET, Key=f"{USER}/{aid}/v1/index.html" + )["Body"].read().decode() == DOC + + +def test_update_reclassifies_markdown_under_inherited_html_type(aws) -> None: + """An HTML artifact updated with raw Markdown (content_type omitted → + inherits HTML from HEAD) is reclassified so the new version renders.""" + ddb, s3 = aws + aid, _ = service.create_artifact_record(USER, SESSION, "Page", DOC, "text/html") + new_md = "# Rewritten\n\nNow it's **markdown**.\n" + ver = service.update_artifact_record(USER, aid, new_md, None, None) + assert ver == 2 + assert _item(ddb, aid, "V#00002")["content_type"] == "text/markdown" + body = s3.get_object( + Bucket=BUCKET, Key=f"{USER}/{aid}/v2/index.html" + )["Body"].read().decode() + assert _embedded_markdown(body) == new_md + + def test_html_artifact_not_wrapped(aws) -> None: _, s3 = aws aid, _ = service.create_artifact_record(USER, SESSION, "Page", DOC, "text/html") diff --git a/backend/tests/agents/builtin_tools/test_workspace_tools.py b/backend/tests/agents/builtin_tools/test_workspace_tools.py new file mode 100644 index 000000000..9b63f0e1d --- /dev/null +++ b/backend/tests/agents/builtin_tools/test_workspace_tools.py @@ -0,0 +1,154 @@ +"""Workspace tool factories (agents/builtin_tools/workspace_tools.py). + +Each tool is closed over the request identity; the shared workspace service +functions are patched. Verifies success payload shapes, the workspace_file +download-card contract on write, and that service failures surface as error +tool-results rather than raising. +""" + +import json +from unittest.mock import AsyncMock + +import pytest + +from agents.builtin_tools.workspace_tools import ( + make_workspace_list_tool, + make_workspace_read_tool, + make_workspace_write_tool, +) +from apis.shared.files.workspace import ( + WorkspaceFileNotFoundError, + WorkspaceStorageNotConfiguredError, + WorkspaceValidationError, +) + +MODULE = "agents.builtin_tools.workspace_tools" + + +class TestRoutesGating: + """_build_workspace_tools in apis/inference_api/chat/routes.py.""" + + @pytest.mark.parametrize( + "enabled,expected", + [(None, 0), ([], 0), (["calculator"], 0), (["workspace_files"], 3)], + ) + def test_gate_key_provisions_full_toolset(self, enabled, expected): + from apis.inference_api.chat.routes import _build_workspace_tools + + tools = _build_workspace_tools(enabled, "s1", "u1") + assert len(tools) == expected + + def test_kill_switch_disables(self, monkeypatch): + from apis.inference_api.chat.routes import _build_workspace_tools + + monkeypatch.setenv("WORKSPACE_TOOLS_ENABLED", "false") + assert _build_workspace_tools(["workspace_files"], "s1", "u1") == [] + + def test_empty_flag_value_stays_enabled(self, monkeypatch): + from apis.inference_api.chat.routes import _build_workspace_tools + + monkeypatch.setenv("WORKSPACE_TOOLS_ENABLED", "") + assert len(_build_workspace_tools(["workspace_files"], "s1", "u1")) == 3 + + +async def _call(tool, *args, **kwargs): + fn = getattr(tool, "__wrapped__", None) or tool + return await fn(*args, **kwargs) + + +class TestWorkspaceList: + @pytest.mark.asyncio + async def test_lists_files(self, monkeypatch): + svc = AsyncMock(return_value={"scope": "session", "files": [], "count": 0, "truncated": False}) + monkeypatch.setattr(f"{MODULE}.list_workspace_files", svc) + tool = make_workspace_list_tool("s1", "u1") + result = await _call(tool) + assert result["status"] == "success" + assert result["content"][0]["json"]["scope"] == "session" + svc.assert_awaited_once_with("u1", "s1", scope="session") + + @pytest.mark.asyncio + async def test_scope_passthrough(self, monkeypatch): + svc = AsyncMock(return_value={"scope": "user", "files": [], "count": 0, "truncated": False}) + monkeypatch.setattr(f"{MODULE}.list_workspace_files", svc) + tool = make_workspace_list_tool("s1", "u1") + await _call(tool, scope="user") + svc.assert_awaited_once_with("u1", "s1", scope="user") + + @pytest.mark.asyncio + async def test_storage_not_configured_is_error_result(self, monkeypatch): + svc = AsyncMock(side_effect=WorkspaceStorageNotConfiguredError("no bucket")) + monkeypatch.setattr(f"{MODULE}.list_workspace_files", svc) + tool = make_workspace_list_tool("s1", "u1") + result = await _call(tool) + assert result["status"] == "error" + assert "not configured" in result["content"][0]["text"] + + +class TestWorkspaceRead: + @pytest.mark.asyncio + async def test_reads_with_offset(self, monkeypatch): + svc = AsyncMock(return_value={"encoding": "text", "content": "hi"}) + monkeypatch.setattr(f"{MODULE}.read_workspace_file", svc) + tool = make_workspace_read_tool("s1", "u1") + result = await _call(tool, upload_id="f1", offset=8) + assert result["status"] == "success" + assert result["content"][0]["json"]["content"] == "hi" + svc.assert_awaited_once_with("u1", "f1", offset=8) + + @pytest.mark.asyncio + async def test_not_found_is_error_result(self, monkeypatch): + svc = AsyncMock(side_effect=WorkspaceFileNotFoundError("No file with id 'ghost'")) + monkeypatch.setattr(f"{MODULE}.read_workspace_file", svc) + tool = make_workspace_read_tool("s1", "u1") + result = await _call(tool, upload_id="ghost") + assert result["status"] == "error" + assert "ghost" in result["content"][0]["text"] + + @pytest.mark.asyncio + async def test_unexpected_exception_is_error_result(self, monkeypatch): + svc = AsyncMock(side_effect=RuntimeError("boom")) + monkeypatch.setattr(f"{MODULE}.read_workspace_file", svc) + tool = make_workspace_read_tool("s1", "u1") + result = await _call(tool, upload_id="f1") + assert result["status"] == "error" + + +class TestWorkspaceWrite: + @pytest.mark.asyncio + async def test_returns_workspace_file_download_card(self, monkeypatch): + svc = AsyncMock( + return_value={ + "upload_id": "up1", + "filename": "report.md", + "mime_type": "text/markdown", + "size_bytes": 4, + "size_kb": "0.0 KB", + "download_url": "https://signed.example/f", + } + ) + monkeypatch.setattr(f"{MODULE}.write_workspace_file", svc) + tool = make_workspace_write_tool("s1", "u1") + result = await _call(tool, filename="report.md", content="# Hi", mime_type="text/markdown") + + card = json.loads(result) + assert card["success"] is True + assert card["ui_type"] == "file_download" + assert card["ui_display"] == "inline" + assert card["payload"] == { + "filename": "report.md", + "download_url": "https://signed.example/f", + "size_kb": "0.0 KB", + } + svc.assert_awaited_once_with( + "u1", "s1", "report.md", "# Hi", mime_type="text/markdown" + ) + + @pytest.mark.asyncio + async def test_validation_failure_is_error_result(self, monkeypatch): + svc = AsyncMock(side_effect=WorkspaceValidationError("bad extension")) + monkeypatch.setattr(f"{MODULE}.write_workspace_file", svc) + tool = make_workspace_write_tool("s1", "u1") + result = await _call(tool, filename="x.csv", content="a", mime_type="text/markdown") + assert result["status"] == "error" + assert "bad extension" in result["content"][0]["text"] diff --git a/backend/tests/shared/test_workspace.py b/backend/tests/shared/test_workspace.py new file mode 100644 index 000000000..d930d3b7b --- /dev/null +++ b/backend/tests/shared/test_workspace.py @@ -0,0 +1,278 @@ +"""Session workspace service (apis/shared/files/workspace.py). + +The DynamoDB repository and S3 client are mocked; these tests pin the +contract the workspace tools rely on: metadata-table-first listing, bounded +ranged text reads with continuation, binary-by-reference, the +_store_document-style write path (READY row + quota increment), and the +fail-loudly identity / validation rules. +""" + +from datetime import datetime, timezone +from unittest.mock import AsyncMock, MagicMock + +import pytest + +import apis.shared.files.workspace as workspace +from apis.shared.files.models import FileMetadata, FileStatus, UserFileQuota +from apis.shared.files.workspace import ( + WorkspaceError, + WorkspaceFileNotFoundError, + WorkspaceQuotaExceededError, + WorkspaceValidationError, + list_workspace_files, + read_workspace_file, + write_workspace_file, +) + +USER = "user-1" +SESSION = "sess-1" + + +def _meta( + upload_id: str = "f1", + user_id: str = USER, + session_id: str = SESSION, + filename: str = "notes.md", + mime_type: str = "text/markdown", + size_bytes: int = 10, + status: FileStatus = FileStatus.READY, + source: str = "upload", +) -> FileMetadata: + return FileMetadata( + upload_id=upload_id, + user_id=user_id, + session_id=session_id, + filename=filename, + mime_type=mime_type, + size_bytes=size_bytes, + s3_key=f"user-files/{user_id}/{session_id}/{upload_id}/{filename}", + s3_bucket="bucket", + status=status, + source=source, + created_at=datetime(2026, 7, 21, tzinfo=timezone.utc), + ) + + +@pytest.fixture +def repo(monkeypatch) -> MagicMock: + repo = MagicMock() + repo.list_session_files = AsyncMock(return_value=[]) + repo.list_user_files = AsyncMock(return_value=([], None)) + repo.get_file = AsyncMock(return_value=None) + repo.create_file = AsyncMock() + repo.increment_quota = AsyncMock() + repo.get_user_quota = AsyncMock(return_value=UserFileQuota(user_id=USER)) + monkeypatch.setattr(workspace, "get_file_upload_repository", lambda: repo) + return repo + + +@pytest.fixture +def s3(monkeypatch) -> MagicMock: + client = MagicMock() + client.generate_presigned_url.return_value = "https://signed.example/f" + monkeypatch.setattr(workspace, "_s3", lambda: client) + monkeypatch.setenv("S3_USER_FILES_BUCKET_NAME", "bucket") + return client + + +def _s3_body(data: bytes) -> dict: + body = MagicMock() + body.read.return_value = data + return {"Body": body} + + +class TestListWorkspaceFiles: + @pytest.mark.asyncio + async def test_session_scope_filters_ownership(self, repo, s3): + repo.list_session_files.return_value = [ + _meta(upload_id="mine"), + _meta(upload_id="theirs", user_id="someone-else"), + ] + result = await list_workspace_files(USER, SESSION) + assert [f["upload_id"] for f in result["files"]] == ["mine"] + assert result["scope"] == "session" + assert result["truncated"] is False + repo.list_session_files.assert_awaited_once_with( + SESSION, status=FileStatus.READY + ) + + @pytest.mark.asyncio + async def test_entry_shape_and_readable_flag(self, repo, s3): + repo.list_session_files.return_value = [ + _meta(upload_id="t", mime_type="text/plain"), + _meta(upload_id="j", mime_type="application/json"), + _meta(upload_id="b", filename="deck.pdf", mime_type="application/pdf"), + ] + result = await list_workspace_files(USER, SESSION) + by_id = {f["upload_id"]: f for f in result["files"]} + assert by_id["t"]["readable"] and by_id["j"]["readable"] + assert not by_id["b"]["readable"] + assert by_id["t"]["source"] == "upload" + # session scope omits session_id (it's implied) + assert "session_id" not in by_id["t"] + + @pytest.mark.asyncio + async def test_user_scope_includes_session_and_truncation(self, repo, s3): + repo.list_user_files.return_value = ([_meta()], "next-cursor") + result = await list_workspace_files(USER, SESSION, scope="user") + assert result["truncated"] is True + assert result["files"][0]["session_id"] == SESSION + repo.list_user_files.assert_awaited_once_with( + USER, limit=workspace.WORKSPACE_LIST_MAX_ENTRIES, status=FileStatus.READY + ) + + @pytest.mark.asyncio + async def test_session_scope_caps_entries(self, repo, s3, monkeypatch): + monkeypatch.setattr(workspace, "WORKSPACE_LIST_MAX_ENTRIES", 2) + repo.list_session_files.return_value = [ + _meta(upload_id=f"f{i}") for i in range(4) + ] + result = await list_workspace_files(USER, SESSION) + assert result["count"] == 2 and result["truncated"] is True + + @pytest.mark.asyncio + async def test_unknown_scope_rejected(self, repo, s3): + with pytest.raises(WorkspaceValidationError): + await list_workspace_files(USER, SESSION, scope="everything") + + @pytest.mark.asyncio + async def test_missing_identity_fails_loudly(self, repo, s3): + with pytest.raises(WorkspaceError): + await list_workspace_files("", SESSION) + with pytest.raises(WorkspaceError): + await list_workspace_files(USER, "") + + +class TestReadWorkspaceFile: + @pytest.mark.asyncio + async def test_text_read_full(self, repo, s3): + repo.get_file.return_value = _meta(size_bytes=5) + s3.get_object.return_value = _s3_body(b"hello") + result = await read_workspace_file(USER, "f1") + assert result["encoding"] == "text" + assert result["content"] == "hello" + assert result["truncated"] is False and result["next_offset"] is None + # ranged GET, never a full-object read + assert "Range" in s3.get_object.call_args.kwargs + + @pytest.mark.asyncio + async def test_text_read_truncates_with_continuation(self, repo, s3, monkeypatch): + monkeypatch.setattr(workspace, "WORKSPACE_READ_MAX_BYTES", 4) + repo.get_file.return_value = _meta(size_bytes=10) + s3.get_object.return_value = _s3_body(b"abcd") + result = await read_workspace_file(USER, "f1") + assert result["truncated"] is True and result["next_offset"] == 4 + assert s3.get_object.call_args.kwargs["Range"] == "bytes=0-3" + + @pytest.mark.asyncio + async def test_text_read_offset_continues(self, repo, s3, monkeypatch): + monkeypatch.setattr(workspace, "WORKSPACE_READ_MAX_BYTES", 4) + repo.get_file.return_value = _meta(size_bytes=6) + s3.get_object.return_value = _s3_body(b"ef") + result = await read_workspace_file(USER, "f1", offset=4) + assert result["content"] == "ef" + assert result["truncated"] is False and result["next_offset"] is None + assert s3.get_object.call_args.kwargs["Range"] == "bytes=4-7" + + @pytest.mark.asyncio + async def test_binary_returns_reference_never_content(self, repo, s3): + repo.get_file.return_value = _meta( + filename="deck.pdf", mime_type="application/pdf" + ) + result = await read_workspace_file(USER, "f1") + assert result["encoding"] == "reference" + assert result["download_url"] == "https://signed.example/f" + assert "content" not in result + s3.get_object.assert_not_called() + + @pytest.mark.asyncio + async def test_missing_and_non_ready_are_not_found(self, repo, s3): + repo.get_file.return_value = None + with pytest.raises(WorkspaceFileNotFoundError): + await read_workspace_file(USER, "ghost") + repo.get_file.return_value = _meta(status=FileStatus.PENDING) + with pytest.raises(WorkspaceFileNotFoundError): + await read_workspace_file(USER, "f1") + + @pytest.mark.asyncio + async def test_offset_beyond_end_rejected(self, repo, s3): + repo.get_file.return_value = _meta(size_bytes=5) + with pytest.raises(WorkspaceValidationError): + await read_workspace_file(USER, "f1", offset=5) + with pytest.raises(WorkspaceValidationError): + await read_workspace_file(USER, "f1", offset=-1) + + +class TestWriteWorkspaceFile: + @pytest.mark.asyncio + async def test_happy_path_follows_store_document_contract(self, repo, s3): + result = await write_workspace_file( + USER, SESSION, "report.md", "# Hi", mime_type="text/markdown" + ) + + put_kwargs = s3.put_object.call_args.kwargs + assert put_kwargs["Bucket"] == "bucket" + assert put_kwargs["ContentType"] == "text/markdown" + assert put_kwargs["Key"].startswith(f"user-files/{USER}/{SESSION}/") + assert put_kwargs["Key"].endswith("/report.md") + + created: FileMetadata = repo.create_file.call_args.args[0] + assert created.source == "agent" + status = created.status if isinstance(created.status, str) else created.status.value + assert status == FileStatus.READY.value + repo.increment_quota.assert_awaited_once_with(USER, len(b"# Hi")) + + assert result["filename"] == "report.md" + assert result["download_url"] == "https://signed.example/f" + assert result["size_bytes"] == 4 + + @pytest.mark.asyncio + async def test_extension_appended_when_missing(self, repo, s3): + result = await write_workspace_file( + USER, SESSION, "report", "x", mime_type="text/markdown" + ) + assert result["filename"] == "report.md" + + @pytest.mark.asyncio + async def test_extension_mismatch_rejected(self, repo, s3): + with pytest.raises(WorkspaceValidationError): + await write_workspace_file( + USER, SESSION, "report.csv", "x", mime_type="text/markdown" + ) + + @pytest.mark.asyncio + @pytest.mark.parametrize("mime", ["application/pdf", "image/png", "junk"]) + async def test_non_text_mime_rejected(self, repo, s3, mime): + with pytest.raises(WorkspaceValidationError): + await write_workspace_file(USER, SESSION, "f.bin", "x", mime_type=mime) + + @pytest.mark.asyncio + @pytest.mark.parametrize( + "name", ["../evil.txt", "a/b.txt", "a\\b.txt", "", ".hidden"] + ) + async def test_bad_filenames_rejected(self, repo, s3, name): + with pytest.raises(WorkspaceValidationError): + await write_workspace_file(USER, SESSION, name, "x") + + @pytest.mark.asyncio + async def test_oversized_content_rejected(self, repo, s3, monkeypatch): + monkeypatch.setattr(workspace, "WORKSPACE_WRITE_MAX_BYTES", 8) + with pytest.raises(WorkspaceValidationError): + await write_workspace_file(USER, SESSION, "big.txt", "123456789") + s3.put_object.assert_not_called() + + @pytest.mark.asyncio + async def test_quota_exceeded_rejected_before_write(self, repo, s3, monkeypatch): + monkeypatch.setattr(workspace, "_USER_QUOTA_BYTES", 10) + repo.get_user_quota.return_value = UserFileQuota(user_id=USER, total_bytes=9) + with pytest.raises(WorkspaceQuotaExceededError): + await write_workspace_file(USER, SESSION, "f.txt", "abc") + s3.put_object.assert_not_called() + repo.create_file.assert_not_awaited() + + @pytest.mark.asyncio + async def test_missing_identity_fails_loudly(self, repo, s3): + with pytest.raises(WorkspaceError): + await write_workspace_file("", SESSION, "f.txt", "x") + with pytest.raises(WorkspaceError): + await write_workspace_file(USER, "", "f.txt", "x") diff --git a/backend/tests/test_seed_system_admin_jwt.py b/backend/tests/test_seed_system_admin_jwt.py index 4eeb6e112..41f709392 100644 --- a/backend/tests/test_seed_system_admin_jwt.py +++ b/backend/tests/test_seed_system_admin_jwt.py @@ -131,7 +131,7 @@ def test_creates_default_tools(self, dynamodb_table): """Creates the default tool entries.""" result = seed_default_tools(TABLE_NAME, REGION) - assert result.created == 7 + assert result.created == 9 assert result.failed == 0 # Verify fetch_url_content @@ -216,6 +216,20 @@ def test_creates_default_tools(self, dynamodb_table): assert item["GSI1PK"] == "CATEGORY#document" assert item["GSI1SK"] == "TOOL#create_word_document" + # Verify workspace_files (single toggle for the workspace toolset) + resp = dynamodb_table.get_item( + Key={"PK": "TOOL#workspace_files", "SK": "METADATA"} + ) + item = resp["Item"] + assert item["toolId"] == "workspace_files" + assert item["displayName"] == "File Workspace" + assert item["category"] == "document" + assert item["protocol"] == "local" + assert item["enabledByDefault"] is False + assert item["isPublic"] is True + assert item["GSI1PK"] == "CATEGORY#document" + assert item["GSI1SK"] == "TOOL#workspace_files" + # Verify create_excel_spreadsheet (single toggle for the whole Excel toolset) resp = dynamodb_table.get_item( Key={"PK": "TOOL#create_excel_spreadsheet", "SK": "METADATA"} @@ -230,13 +244,27 @@ def test_creates_default_tools(self, dynamodb_table): assert item["GSI1PK"] == "CATEGORY#document" assert item["GSI1SK"] == "TOOL#create_excel_spreadsheet" + # Verify create_powerpoint_presentation (single toggle for the whole PowerPoint toolset) + resp = dynamodb_table.get_item( + Key={"PK": "TOOL#create_powerpoint_presentation", "SK": "METADATA"} + ) + item = resp["Item"] + assert item["toolId"] == "create_powerpoint_presentation" + assert item["displayName"] == "PowerPoint Presentations" + assert item["category"] == "document" + assert item["protocol"] == "local" + assert item["enabledByDefault"] is False + assert item["isPublic"] is True + assert item["GSI1PK"] == "CATEGORY#document" + assert item["GSI1SK"] == "TOOL#create_powerpoint_presentation" + def test_skips_existing_tools(self, dynamodb_table): """Skips tools that already exist.""" seed_default_tools(TABLE_NAME, REGION) result = seed_default_tools(TABLE_NAME, REGION) - assert result.skipped == 7 + assert result.skipped == 9 assert result.created == 0 def test_partial_skip(self, dynamodb_table): @@ -250,7 +278,7 @@ def test_partial_skip(self, dynamodb_table): result = seed_default_tools(TABLE_NAME, REGION) - assert result.created == 6 + assert result.created == 8 assert result.skipped == 1 diff --git a/backend/uv.lock b/backend/uv.lock index 4c6376832..ee1b2cc65 100644 --- a/backend/uv.lock +++ b/backend/uv.lock @@ -12,7 +12,7 @@ resolution-markers = [ [[package]] name = "agentcore-stack" -version = "1.10.0" +version = "1.11.0" source = { editable = "." } dependencies = [ { name = "aiofiles" }, diff --git a/docs/specs/session-workspace-tools.md b/docs/specs/session-workspace-tools.md new file mode 100644 index 000000000..5e7828c36 --- /dev/null +++ b/docs/specs/session-workspace-tools.md @@ -0,0 +1,244 @@ +# Session workspace tools (`workspace_list` / `workspace_read` / `workspace_write`) + +**Status:** PR-1 implemented (branch `feature/session-workspace-tools`, off `develop`) +**Inspired by:** the shared-workspace tool pattern in +`aws-samples/sample-strands-agent-with-agentcore` (adapted, not adopted — see +"Why not the reference implementation as-is") + +## Problem + +The agent can already work with files — but only through vertical slices, each +of which reimplements the same substrate: + +- `word_document_tool.py` writes generated `.docx` to the user-files bucket, + registers `FileMetadata`, and returns an inline download card. +- `spreadsheet_analysis/` lists tabular attachments from DynamoDB and analyzes + them in the AgentCore Code Interpreter sandbox. +- `code_interpreter_diagram_tool.py` produces images. +- The SPA uploads attachments via presigned URLs + (`apis/app_api/files/service.py`) into + `user-files/{userId}/{sessionId}/{uploadId}/{filename}`. + +What's missing is the **horizontal primitive**: a generic list/read/write +surface over the user's files that any tool, skill, or future harness lane can +compose with. Concretely, today the agent cannot: + +- enumerate what the user has uploaded (except spreadsheets) or what other + tools have produced this session; +- read a text/markdown/CSV attachment *on demand* mid-turn (attachments are + push-only: inlined as content blocks on the user message by + `multimodal/prompt_builder.py`); +- write a plain text/markdown/CSV/JSON deliverable the user can download + (only `.docx` has a write path); +- chain tools through files (code-interpreter output → document tool → + download) without each pair growing a bespoke bridge. + +This is also a prerequisite primitive for the agentic-platform roadmap: skills +that produce files, the F1 headless/harness entrypoint, and any future +general-purpose code-interpreter tool all need a durable file surface between +ephemeral sandbox sessions. + +## Why not the reference implementation as-is + +The reference repo's `workspace_*` tools are raw S3 wrappers. Three conflicts +with this stack's conventions: + +1. **They bypass the metadata layer.** In this stack the DynamoDB user-files + table is the source of truth (SPA listing, status, quota, thumbnails, TTL). + A raw `put_object` produces a file invisible to the Files UI and outside + quota accounting; a raw `list_objects_v2` sees a different world than the + SPA does. Every workspace operation must go through + `apis.shared.files` (`FileMetadata` + `FileUploadRepository`), exactly like + `word_document_tool._store_document` already does. +2. **Base64 file content through the model violates the token-cost tenet.** + `workspace_read` returning base64 of a binary file is an unbounded per-turn + payload — a 2 MB PPTX ≈ 2.7 MB of base64 in a tool result, at model prices, + likely blowing context outright. Binary files move **by reference** + (upload_id / presigned URL); byte processing belongs in the code-interpreter + sandbox. Text reads must be bounded. +3. **`invocation_state.get('user_id', 'default_user')` is a cross-tenant bug + waiting to happen.** A missing identity would silently collapse every + affected session into one shared namespace. This stack injects identity by + closure — `make_*_tool(session_id, user_id)` factories instantiated + per-request in `apis/inference_api/chat/routes.py` — so identity is + guaranteed present or the tool is never built. Same "fail loudly" posture + as PR #706 (unconfigured bucket). + +Also dropped: the parallel `code-agent-workspace/` / `documents//` +namespace scheme. That's an artifact of the reference repo's multiple +sub-agents. Here, the existing key layout plus a `source` attribute on the +metadata row carries the same information without a second key convention. + +## Decision summary + +| Question | Decision | +|----------|----------| +| Source of truth | **DynamoDB user-files table** — never raw S3 listing | +| Write path | Through `FileMetadata` + repository (the `_store_document` pattern), status `READY` | +| Read scope | **User-scoped** — all the user's `READY` files, any session | +| Write scope | **Session-scoped** — new files land under the current session's prefix | +| Binary handling | **By reference only** — metadata + presigned URL; no base64 through the model | +| Text read bound | `WORKSPACE_READ_MAX_BYTES` (default **48 KB**) per call, with `offset` for continuation | +| Identity | Closure-injected via `make_workspace_*_tool(session_id, user_id)` factories | +| Key layout | Existing `user-files/{userId}/{sessionId}/{uploadId}/{filename}` — no new namespaces | +| Provenance | New optional `source` field on `FileMetadata` (`"upload"` default, `"agent"`, `"word_document"`, …) | +| Quota | Agent-written files **count against** the user's existing quota; quota-exceeded is a friendly tool error | +| Overwrite semantics | **No in-place overwrite** — each write is a new `uploadId`; same-name writes supersede in listings | +| Lifecycle | Inherit existing 365-day DynamoDB TTL + matching S3 expiration — nothing new | +| Module home | Service in `apis/shared/files/workspace.py`; tools in `agents/builtin_tools/workspace_tools.py` | +| RBAC | One catalog entry (`workspace_files`) as the gate key provisioning all three tools (the `create_word_document` pattern) | +| Feature flag | `WORKSPACE_TOOLS_ENABLED`, **default true**, `=false` kill switch | +| UI | `workspace_write` returns the same inline download-card contract as the word tool | + +## Tool surface + +All three tools are `make_*` factories (closure identity), registered in +`ToolRegistry` alongside the word-document tools and instantiated per-request +in `apis/inference_api/chat/routes.py`. All returns are compact JSON strings; +errors return `{"error": ..., "status": "error"}` conversationally (never +raise through the agent loop). + +### `workspace_list` + +``` +workspace_list(scope: str = "session") -> str + scope: "session" (this conversation) | "user" (all conversations) +``` + +- Queries DynamoDB only: GSI1 (`CONV#{sessionId}`) for session scope, the + `USER#{userId}` partition for user scope. Never touches S3. +- Filters to `status == READY`. +- Returns per file: `upload_id`, `filename`, `mime_type`, `size_bytes`, + `source`, `session_id` (user scope only), `created_at`, and a `readable` + bool (text-type per the MIME allowlist). +- Bounded output: cap at ~100 most-recent entries with a `truncated` flag — + a tool result is a per-turn payload too. + +### `workspace_read` + +``` +workspace_read(upload_id: str, offset: int = 0) -> str +``` + +- Looks up `FileMetadata` by `(user_id, upload_id)` — ownership is enforced by + the table key shape (`PK = USER#{userId}`), so cross-user reads are + impossible by construction. Cross-*session* reads are allowed (precedent: + `word_document_tool._find_word_document` already reads across sessions). +- **Text MIME types** (`text/*`, `application/json`, csv/md/html): return + UTF-8 content from `offset`, capped at `WORKSPACE_READ_MAX_BYTES` (48 KB + default), with `truncated` + `next_offset` for continuation. Ranged S3 GET + (`Range` header) so a 200 MB file never enters process memory. +- **Binary / everything else**: return metadata + a short-lived presigned GET + URL and a hint naming the right tool for the bytes + (`analyze_spreadsheet` for tabular, code interpreter for the rest). Never + base64. PDFs and images already reach the model as content blocks via the + attachment flow; this tool does not duplicate that path in v1. + +### `workspace_write` + +``` +workspace_write(filename: str, content: str, mime_type: str = "text/plain") -> str +``` + +- **Text-only in v1.** `mime_type` must be in a text allowlist (plain, + markdown, csv, json, html). Binary production stays with the dedicated + tools (`create_word_document`, diagram tool); when a general + code-interpreter tool lands, its sandbox→S3 sync becomes the binary write + path and this tool stays as-is. +- Size-capped per call (`WORKSPACE_WRITE_MAX_BYTES`, default 1 MB — content + the model just generated is inherently small). +- Follows `_store_document` exactly: quota check → `put_object` under + `user-files/{userId}/{sessionId}/{uploadId}/{filename}` → `FileMetadata` + row (`source="agent"`, `status=READY`) → `increment_quota` → presigned + download URL. +- Returns the inline download-card contract + (`ui_type` / `ui_display: "inline"` / payload) so the SPA renders a + first-class download card. Reuses the shared `file_download` `ui_type` + already routed to `FileDownloadRendererComponent` in + `inline-visual.component.ts`, so no frontend change is needed. +- Filename validation mirrors the word tool's (`_validate_document_name` + generalized): no path separators, no traversal, extension must match + `mime_type`. + +## Design notes + +### Read scope: why user-scoped + +`FileMetadata`'s native key shape is user-partitioned (`PK = USER#{userId}`, +GSI1 by session), so "all my files" is the cheap query, not the expensive one. +The compelling UX is *"use the CSV I uploaded yesterday"* — without +cross-session read, users must re-upload per conversation. The word tool +already crossed this line quietly; this spec just makes it the stated policy. +Writes stay session-scoped so provenance ("this file came from that +conversation") stays intact and the S3 key layout is unchanged. + +### Quota: why agent files count + +Simplest and safest: same repository path as uploads, no second accounting +regime, and it bounds a runaway agent loop writing files. The write cap plus +quota makes the worst case boring. If agent-generated deliverables ever crowd +out upload quota in practice, carve-out is a follow-up, not a v1 concern. + +### Token-cost audit (per CLAUDE.md tenet) + +- Nothing added to the cacheable prefix beyond three small static tool + definitions in `toolConfig` (deterministic ordering via the existing + registry path). +- Per-turn payloads are all bounded: list ≤ ~100 rows, read ≤ 48 KB/call, + write returns a small card. No unbounded pass-through anywhere. +- The pull model is a net token *saving* vs. today's push model for large + text attachments: the model reads the 5 KB it needs instead of receiving + the whole document as a content block on every restored turn. + +### Attachment-flow interaction (explicitly out of scope for v1) + +Today `prompt_builder.py` inlines every attachment as content blocks. Once +`workspace_read` exists, a future change could stop inlining large text files +and inject a one-line pointer ("attached: `report.md`, upload_id …") instead — +a meaningful prompt-size win, but it changes restored-history bytes and +therefore interacts with the prompt-cache byte-stability contract. Do it as +its own PR with `cacheStatus` verification, not as a rider on this one. + +## Security + +- **Tenant isolation by construction:** every repository call is keyed by the + closure-injected `user_id`; there is no path where model-supplied input + selects the partition. +- **No model-supplied S3 keys:** tools accept `upload_id` / `filename` only; + S3 keys are always derived server-side. +- **Filename sanitization** on write (no separators/traversal, extension ↔ + MIME agreement). +- **Presigned URLs** are short-TTL GET-only, same posture as the existing + preview-url endpoint. +- Governance posture per platform norm: identity-claims gating, no content + inspection. + +## Phasing + +**PR-1 — service + tools (backend) + download card (frontend).** +`apis/shared/files/workspace.py` (bounded ranged read, list queries, write +helper extracted/shared with `_store_document`), `source` field on +`FileMetadata` (additive, default `"upload"` on read), the three tool +factories + registry/catalog/RBAC wiring, `file_download` inline card, +tests (unit + import-boundary clean). + +**PR-2 (optional, later) — word tool convergence.** Reimplement +`_store_document` and `_find_word_document` on the workspace service so +there is one write path. Pure refactor, no behavior change. + +**Deferred:** attachment-flow pull-model change (cache-sensitive, own PR); +binary write via code-interpreter sandbox sync; SPA "Files" page surfacing +`source` provenance. + +## Open questions + +1. Does the SPA Files page need any change for v1 beyond the download card? + (Agent-written files will simply appear in the existing list; showing a + "generated" badge off `source` is nice-to-have.) +2. Should `workspace_read` serve image files as proper image content blocks + (size-gated) instead of URL-by-reference? Useful for "look at the diagram + you made", but content-block emission from a tool result needs Strands + plumbing verification first. +3. RBAC granularity: one `workspace_files` catalog entry vs. three. One entry is + recommended (they're useless separately), but confirm the admin-tools page + renders a multi-tool entry cleanly. diff --git a/frontend/ai.client/package-lock.json b/frontend/ai.client/package-lock.json index 87f5f8574..8dffdfabf 100644 --- a/frontend/ai.client/package-lock.json +++ b/frontend/ai.client/package-lock.json @@ -1,12 +1,12 @@ { "name": "ai.client", - "version": "1.10.0", + "version": "1.11.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "ai.client", - "version": "1.10.0", + "version": "1.11.0", "dependencies": { "@angular/cdk": "21.2.14", "@angular/common": "21.2.17", diff --git a/frontend/ai.client/package.json b/frontend/ai.client/package.json index 69bae0039..a175c970d 100644 --- a/frontend/ai.client/package.json +++ b/frontend/ai.client/package.json @@ -1,6 +1,6 @@ { "name": "ai.client", - "version": "1.10.0", + "version": "1.11.0", "scripts": { "ng": "ng", "start": "ng serve", diff --git a/frontend/ai.client/src/app/session/components/message-list/components/assistant-message.component.spec.ts b/frontend/ai.client/src/app/session/components/message-list/components/assistant-message.component.spec.ts index 700e95cef..798c31e62 100644 --- a/frontend/ai.client/src/app/session/components/message-list/components/assistant-message.component.spec.ts +++ b/frontend/ai.client/src/app/session/components/message-list/components/assistant-message.component.spec.ts @@ -334,6 +334,38 @@ describe('AssistantMessageComponent', () => { expect(blocks[2].type).toBe('tool_group'); expect(blocks[2].group!.calls.length).toBe(1); }); + + it('should not render a reasoning block with empty text and no redaction', () => { + // Signature-only / empty thinking blocks (e.g. Sonnet 5) are kept in the + // message for API correctness but must not paint an empty "Thinking" + // collapsible. + fixture.componentRef.setInput('message', makeMessage([ + { + type: 'reasoningContent', + reasoningContent: { reasoningText: { text: '', signature: 'abc123' } }, + }, + { type: 'text', text: 'Here is your answer.' }, + ])); + fixture.detectChanges(); + + const blocks = component.displayBlocks(); + expect(blocks.length).toBe(1); + expect(blocks[0].type).toBe('text'); + }); + + it('should render a reasoning block that has only redacted content', () => { + fixture.componentRef.setInput('message', makeMessage([ + { + type: 'reasoningContent', + reasoningContent: { redactedContent: 'encrypted-blob' }, + }, + ])); + fixture.detectChanges(); + + const blocks = component.displayBlocks(); + expect(blocks.length).toBe(1); + expect(blocks[0].type).toBe('reasoningContent'); + }); }); describe('tool call data mapping', () => { diff --git a/frontend/ai.client/src/app/session/components/message-list/components/assistant-message.component.ts b/frontend/ai.client/src/app/session/components/message-list/components/assistant-message.component.ts index 9fccd23ae..2e38fbae7 100644 --- a/frontend/ai.client/src/app/session/components/message-list/components/assistant-message.component.ts +++ b/frontend/ai.client/src/app/session/components/message-list/components/assistant-message.component.ts @@ -351,8 +351,16 @@ export class AssistantMessageComponent { }; for (const block of blocks) { - // Handle reasoning content - if (block.type === 'reasoningContent' && block.reasoningContent) { + // Handle reasoning content. Only render a "Thinking" block when it has + // something to show -- reasoning text or a redacted notice. A block with + // an empty `reasoningText.text` and no `redactedContent` (e.g. a + // signature-only thinking block that some models, like Sonnet 5, still + // emit/persist) would otherwise render an empty collapsible header. The + // block is kept in the persisted message for API correctness (the + // signature is required on subsequent Bedrock calls); we just don't paint + // it. This mirrors the live stream parser's `reasoningText` guard, which + // the history-rehydration path bypasses. + if (block.type === 'reasoningContent' && this.hasRenderableReasoning(block)) { flushToolGroup(); result.push({ type: 'reasoningContent', data: block }); continue; @@ -468,6 +476,20 @@ export class AssistantMessageComponent { return toolUse.result ?? { content: [], status: 'success' }; } + /** + * Whether a reasoningContent block has anything worth rendering as a + * "Thinking" section: actual reasoning text or a redacted-content notice. + * Signature-only / empty-text blocks are kept in the message (the signature + * is required for subsequent Bedrock calls) but paint nothing, so they must + * not produce an empty collapsible. Mirrors ReasoningContentComponent's own + * `hasReasoningText || hasRedactedContent` visibility checks. + */ + private hasRenderableReasoning(block: ContentBlock): boolean { + const data = block.reasoningContent; + if (!data) return false; + return !!data.reasoningText?.text || !!data.redactedContent; + } + /** * Extract promoted visual data from a tool use result. * Returns null if not a promoted visual (no ui_type or ui_display !== 'inline'). diff --git a/frontend/ai.client/src/app/session/components/message-list/components/inline-visual/renderers/file-download-renderer.component.ts b/frontend/ai.client/src/app/session/components/message-list/components/inline-visual/renderers/file-download-renderer.component.ts index 82518709b..643f77a26 100644 --- a/frontend/ai.client/src/app/session/components/message-list/components/inline-visual/renderers/file-download-renderer.component.ts +++ b/frontend/ai.client/src/app/session/components/message-list/components/inline-visual/renderers/file-download-renderer.component.ts @@ -25,6 +25,8 @@ const DOCUMENT_ICON = 'M9 12h6m-6 4h6m2 5H7a2 2 0 01-2-2V5a2 2 0 012-2h5.586a1 1 0 01.707.293l5.414 5.414a1 1 0 01.293.707V19a2 2 0 01-2 2z'; const SPREADSHEET_ICON = 'M9 17V7m0 10a2 2 0 01-2 2H5a2 2 0 01-2-2V7a2 2 0 012-2h2a2 2 0 012 2m0 10a2 2 0 002 2h2a2 2 0 002-2M9 7a2 2 0 012-2h2a2 2 0 012 2m0 10V7m0 10a2 2 0 002 2h2a2 2 0 002-2V7a2 2 0 00-2-2h-2a2 2 0 00-2 2'; +const PRESENTATION_ICON = + 'M16 8v8m-4-5v5m-4-2v2m-2 4h12a2 2 0 002-2V6a2 2 0 00-2-2H6a2 2 0 00-2 2v10a2 2 0 002 2z'; const WORD_STYLE: FileKindStyle = { iconPath: DOCUMENT_ICON, @@ -34,6 +36,10 @@ const EXCEL_STYLE: FileKindStyle = { iconPath: SPREADSHEET_ICON, badgeClass: 'bg-green-50 text-green-600 dark:bg-green-900/30 dark:text-green-400', }; +const POWERPOINT_STYLE: FileKindStyle = { + iconPath: PRESENTATION_ICON, + badgeClass: 'bg-orange-50 text-orange-600 dark:bg-orange-900/30 dark:text-orange-400', +}; const GENERIC_STYLE: FileKindStyle = { iconPath: DOCUMENT_ICON, badgeClass: 'bg-gray-100 text-gray-600 dark:bg-gray-700 dark:text-gray-300', @@ -47,6 +53,9 @@ function styleForFilename(filename: string): FileKindStyle { if (lower.endsWith('.docx') || lower.endsWith('.doc')) { return WORD_STYLE; } + if (lower.endsWith('.pptx') || lower.endsWith('.ppt')) { + return POWERPOINT_STYLE; + } return GENERIC_STYLE; } diff --git a/infrastructure/package-lock.json b/infrastructure/package-lock.json index c694c14a3..74167e76b 100644 --- a/infrastructure/package-lock.json +++ b/infrastructure/package-lock.json @@ -1,12 +1,12 @@ { "name": "infrastructure", - "version": "1.10.0", + "version": "1.11.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "infrastructure", - "version": "1.10.0", + "version": "1.11.0", "dependencies": { "aws-cdk-lib": "2.260.0", "constructs": "10.6.0" diff --git a/infrastructure/package.json b/infrastructure/package.json index 4f0ed2a23..22f2830da 100644 --- a/infrastructure/package.json +++ b/infrastructure/package.json @@ -1,6 +1,6 @@ { "name": "infrastructure", - "version": "1.10.0", + "version": "1.11.0", "bin": { "infrastructure": "bin/infrastructure.js" },