From 81a7fcb484c9dbdeaebd559847a96102cd86e390 Mon Sep 17 00:00:00 2001 From: Chojan Shang Date: Fri, 2 Oct 2026 02:20:01 +0800 Subject: [PATCH 1/8] refactor(tools): unify tool discovery and preparation in catalogs --- src/bub/builtin/agent.py | 52 ++++++------------- src/bub/builtin/catalogs.py | 52 +++++++++++++++++++ src/bub/builtin/tools.py | 2 +- src/bub/tools.py | 9 +++- tests/test_builtin_agent.py | 3 +- tests/test_subagent_tool.py | 1 + ...ool_providers.py => test_tool_catalogs.py} | 42 +++++++++------ website/src/content/docs/docs/build/tools.mdx | 6 +-- .../content/docs/zh-cn/docs/build/tools.mdx | 6 +-- 9 files changed, 111 insertions(+), 62 deletions(-) create mode 100644 src/bub/builtin/catalogs.py rename tests/{test_tool_providers.py => test_tool_catalogs.py} (59%) diff --git a/src/bub/builtin/agent.py b/src/bub/builtin/agent.py index 1142ff05..8336d3f2 100644 --- a/src/bub/builtin/agent.py +++ b/src/bub/builtin/agent.py @@ -17,6 +17,7 @@ from loguru import logger +from bub.builtin.catalogs import BuiltinToolCatalog, CodeModeCatalog from bub.builtin.commands import strip_command_prefix, validate_command_prefix from bub.builtin.model_runner import ( ModelRunner, @@ -29,7 +30,7 @@ from bub.store import AsyncTapeStore, AsyncTapeStoreAdapter, InMemoryTapeStore, TapeStore, is_async_tape_store from bub.streaming import AsyncStreamEvents, StreamEvent, StreamState from bub.tape import Tape -from bub.tools import REGISTRY, Tool, ToolContext, ToolProvider, model_tools +from bub.tools import REGISTRY, Tool, ToolCatalog, ToolContext, model_tools from bub.tracing import Span, current_span from bub.turn import TurnState from bub.utils import workspace_from_state @@ -73,11 +74,16 @@ def __init__( ) self.framework = framework self.tools = {tool.name: tool for tool in tools} if tools is not None else REGISTRY.copy() - self.tool_providers: list[ToolProvider] = [] + self.catalogs: list[ToolCatalog] = [BuiltinToolCatalog(self.tools), CodeModeCatalog(self.tools)] self.tape_store = tape_store self.skill_dirs = skill_dirs self.model_runner = ModelRunner(self.settings, hooks=framework.get_agent_hooks()) + @property + def known_tools(self) -> dict[str, Tool]: + """Current execution tools plus catalog entries, with later sources taking precedence.""" + return {name: item for catalog in self.catalogs for name, item in catalog.tools.items()} + @cached_property def tape(self) -> Tape: """Return the lazily constructed, cached tape factory for this agent. @@ -247,6 +253,9 @@ async def _run_command(self, tape: Tape, *, line: str) -> str: output = "" status = "ok" try: + known_tools = self.known_tools + if name in known_tools: + self.tools[name] = known_tools[name] if name not in self.tools: if "bash" not in self.tools: raise ValueError("bash tool is not available") # noqa: TRY301 @@ -445,17 +454,18 @@ async def _run_once( prompt_text = "" else: prompt_text = _extract_text_from_parts(prompt) + known_tools = self.known_tools if allowed_tools is not None: from bub.builtin.tools import resolve_tool_names - allowed_tools = resolve_tool_names(allowed_tools, all_names=self.tools) + allowed_tools = resolve_tool_names(allowed_tools, all_names=known_tools) if allowed_skills is not None: allowed_skills = {name.casefold() for name in allowed_skills} tape.context.state["allowed_skills"] = list(allowed_skills) if allowed_tools is not None: - tools = [tool for tool in self.tools.values() if tool.name in allowed_tools] + tools = [tool for tool in known_tools.values() if tool.name in allowed_tools] else: - tools = list(self.tools.values()) + tools = list(known_tools.values()) return await self._run_once_stream( tape=tape, prompt=prompt, @@ -476,8 +486,8 @@ async def _run_once_stream( tools: list[Tool], ) -> AsyncStreamEvents: tools_prompts: list[str] = [] - for provider in (self._prepare_code_mode, *self.tool_providers): - tools, tools_prompt = await provider(tools, tape) + for catalog in self.catalogs: + tools, tools_prompt = await catalog.prepare(tools, tape) if tools_prompt: tools_prompts.append(tools_prompt) system_prompt = self._system_prompt( @@ -514,34 +524,6 @@ async def _run_once_stream( steering_messages=steering_messages, ) - async def _prepare_code_mode(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - """Split tools for code mode and return model-facing tools and the stub prompt. - - Code mode is a session setting (``state["code_mode"]``, switched by the ``code_mode`` - command) and applies only when ``run_code`` is among the allowed tools: the model then - sees preserved tools directly, and every other tool is callable only from code. - """ - from bub.builtin.codemode import ( - CODE_MODE_STATE_KEY, - CODE_TOOLS_STATE_KEY, - RUN_CODE_TOOL_NAME, - render_code_mode_prompt, - write_tool_stub, - ) - - state = tape.context.state - direct_tools = [tool for tool in tools if tool.name != RUN_CODE_TOOL_NAME] - if not state.get(CODE_MODE_STATE_KEY) or len(direct_tools) == len(tools): - state.pop(CODE_TOOLS_STATE_KEY, None) - return direct_tools, "" - - code_tools = [tool for tool in direct_tools if tool.code_use] - state[CODE_TOOLS_STATE_KEY] = model_tools(code_tools) - stub_path = write_tool_stub( - code_tools, session_id=str(state.get("session_id", "")), workspace=workspace_from_state(state) - ) - return [tool for tool in tools if tool.preserve], render_code_mode_prompt(stub_path) - def _system_prompt( self, prompt: str, diff --git a/src/bub/builtin/catalogs.py b/src/bub/builtin/catalogs.py new file mode 100644 index 00000000..52421823 --- /dev/null +++ b/src/bub/builtin/catalogs.py @@ -0,0 +1,52 @@ +"""Default tool catalogs: direct tools first, code-mode presentation last.""" + +from __future__ import annotations + +from dataclasses import dataclass + +from bub.tape import Tape +from bub.tools import Tool, model_tools +from bub.utils import workspace_from_state + + +@dataclass +class BuiltinToolCatalog: + """Expose the agent's registered tools directly.""" + + tools: dict[str, Tool] + + async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: + return tools, "" + + +@dataclass +class CodeModeCatalog: + """Present the selected tools as a code stub when code mode is enabled.""" + + registry: dict[str, Tool] + + @property + def tools(self) -> dict[str, Tool]: + return {name: item for name, item in self.registry.items() if name == "run_code"} + + async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: + from bub.builtin.codemode import ( + CODE_MODE_STATE_KEY, + CODE_TOOLS_STATE_KEY, + RUN_CODE_TOOL_NAME, + render_code_mode_prompt, + write_tool_stub, + ) + + state = tape.context.state + direct_tools = [item for item in tools if item.name != RUN_CODE_TOOL_NAME] + if not state.get(CODE_MODE_STATE_KEY) or len(direct_tools) == len(tools): + state.pop(CODE_TOOLS_STATE_KEY, None) + return direct_tools, "" + + code_tools = [item for item in direct_tools if item.code_use] + state[CODE_TOOLS_STATE_KEY] = model_tools(code_tools) + stub_path = write_tool_stub( + code_tools, session_id=str(state.get("session_id", "")), workspace=workspace_from_state(state) + ) + return [item for item in tools if item.preserve], render_code_mode_prompt(stub_path) diff --git a/src/bub/builtin/tools.py b/src/bub/builtin/tools.py index 12bfd7f3..ff6740c2 100644 --- a/src/bub/builtin/tools.py +++ b/src/bub/builtin/tools.py @@ -434,7 +434,7 @@ async def run_subagent(param: SubAgentInput, *, context: ToolContext) -> SubAgen else: subagent_session = param.session state = {**context.state, "session_id": subagent_session} - allowed_tools = resolve_tool_names(param.allowed_tools or None, exclude={"subagent"}, all_names=agent.tools) + allowed_tools = resolve_tool_names(param.allowed_tools or None, exclude={"subagent"}, all_names=agent.known_tools) output = "" errors: list[str] = [] stream = await agent.run_stream( diff --git a/src/bub/tools.py b/src/bub/tools.py index c9346417..9d5e589a 100644 --- a/src/bub/tools.py +++ b/src/bub/tools.py @@ -219,8 +219,13 @@ def validated(*args: Any, **kwargs: Any) -> Any: ) -type ToolProvider = Callable[[list[Tool], Tape], Awaitable[tuple[list[Tool], str]]] -"""Prepare registered tools and a prompt fragment for one model request.""" +class ToolCatalog(Protocol): + """Known tools and their per-request preparation, owned by one source.""" + + @property + def tools(self) -> dict[str, Tool]: ... + + async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: ... def model_tools(tools: Iterable[Tool]) -> list[Tool]: diff --git a/tests/test_builtin_agent.py b/tests/test_builtin_agent.py index bedcec6a..33936465 100644 --- a/tests/test_builtin_agent.py +++ b/tests/test_builtin_agent.py @@ -12,6 +12,7 @@ import bub.builtin.codemode import bub.builtin.tools # noqa: F401 — registers builtin tools (incl. `model`) from bub.builtin.agent import Agent +from bub.builtin.catalogs import BuiltinToolCatalog, CodeModeCatalog from bub.builtin.model_runner import ModelRunner from bub.builtin.settings import AgentSettings from bub.builtin.steering import InMemorySteeringInbox @@ -55,7 +56,7 @@ async def build_prompt(message: dict[str, Any], session_id: str, state: dict[str agent.command_prefix = agent.settings.command_prefix agent.framework = framework agent.tools = REGISTRY.copy() - agent.tool_providers = [] + agent.catalogs = [BuiltinToolCatalog(agent.tools), CodeModeCatalog(agent.tools)] agent.tape_store = None agent.skill_dirs = None agent.model_runner = _FakeModelRunner(agent.settings) diff --git a/tests/test_subagent_tool.py b/tests/test_subagent_tool.py index 1826de24..d77f5930 100644 --- a/tests/test_subagent_tool.py +++ b/tests/test_subagent_tool.py @@ -21,6 +21,7 @@ def __init__(self, state: dict[str, Any]) -> None: class FakeAgent: def __init__(self) -> None: self.tools = REGISTRY.copy() + self.known_tools = self.tools self.run_stream = AsyncMock(side_effect=self._run_stream) async def _run_stream(self, **kwargs: Any) -> AsyncStreamEvents: diff --git a/tests/test_tool_providers.py b/tests/test_tool_catalogs.py similarity index 59% rename from tests/test_tool_providers.py rename to tests/test_tool_catalogs.py index 18329053..b15440e4 100644 --- a/tests/test_tool_providers.py +++ b/tests/test_tool_catalogs.py @@ -8,14 +8,16 @@ from any_llm.types.completion import ChatCompletion from bub.builtin.agent import Agent +from bub.builtin.codemode import run_code from bub.framework import BubFramework from bub.tape import Tape from bub.tools import Tool @pytest.mark.asyncio -async def test_provider_prompt_reaches_the_model_and_registered_tools_remain_callable( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch +@pytest.mark.parametrize("code_mode", [False, True]) +async def test_catalog_tools_and_prompt_reach_the_model_within_scope( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, code_mode: bool ) -> None: monkeypatch.setenv("BUB_HOME", str(tmp_path)) framework = BubFramework(config_file=tmp_path / "config.yml") @@ -29,10 +31,15 @@ def lookup(name: str) -> str: calls.append(name) return f"Hello {name}" - supplied = Tool.from_callable(lookup, name="provider.lookup") + supplied = Tool.from_callable(lookup, name="catalog.lookup") - async def provide(tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - return tools, "Use provider_lookup to greet the requested person." + class Catalog: + def __init__(self) -> None: + self.tools = {supplied.name: supplied} + + async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: + tape.context.state["_runtime_agent"].tools[supplied.name] = supplied + return tools, "Use catalog_lookup to greet the requested person." requests: list[dict[str, Any]] = [] @@ -43,13 +50,15 @@ async def acompletion(self, **kwargs: Any) -> ChatCompletion: requests.append(kwargs) message: dict[str, Any] = {"role": "assistant", "content": "Hello Ada"} if len(requests) == 1: + name = "run_code" if code_mode else "catalog_lookup" + arguments = {"code": "print(await tools.catalog_lookup(name='Ada'))"} if code_mode else {"name": "Ada"} message = { "role": "assistant", "tool_calls": [ { "id": "lookup", "type": "function", - "function": {"name": "provider_lookup", "arguments": json.dumps({"name": "Ada"})}, + "function": {"name": name, "arguments": json.dumps(arguments)}, } ], } @@ -68,24 +77,23 @@ async def acompletion(self, **kwargs: Any) -> ChatCompletion: }) monkeypatch.setattr("bub.builtin.model_runner.AnyLLM.create", lambda *args, **kwargs: Provider()) - agent = Agent(framework, tools=[direct, denied, supplied], skill_dirs=[]) - agent.tool_providers.append(provide) + agent = Agent(framework, tools=[direct, denied, run_code], skill_dirs=[]) + agent.catalogs.insert(-1, Catalog()) stream = await agent.run_stream( - session_id="provider", + session_id="catalog", prompt="Greet Ada.", model="openrouter:test-model", - allowed_tools=["direct", "provider_lookup"], + allowed_tools=["direct", "catalog_lookup", "run_code"], + state={"code_mode": code_mode}, ) - events = [event async for event in stream] - assert any(event.data.get("text") == "Hello Ada" for event in events if event.kind == "final") + async for _ in stream: + pass assert calls == ["Ada"] definitions = {item["function"]["name"]: item["function"] for item in requests[0]["tools"]} - assert definitions.keys() == {"direct", "provider_lookup"} + assert definitions.keys() == ({"run_code"} if code_mode else {"direct", "catalog_lookup"}) assert any( - "Use provider_lookup" in message["content"] - for message in requests[0]["messages"] - if message["role"] == "system" + "Use catalog_lookup" in message["content"] for message in requests[0]["messages"] if message["role"] == "system" ) assert any( - message.get("content") == "Hello Ada" for message in requests[1]["messages"] if message["role"] == "tool" + "Hello Ada" in message.get("content", "") for message in requests[1]["messages"] if message["role"] == "tool" ) diff --git a/website/src/content/docs/docs/build/tools.mdx b/website/src/content/docs/docs/build/tools.mdx index de281adf..a72d4091 100644 --- a/website/src/content/docs/docs/build/tools.mdx +++ b/website/src/content/docs/docs/build/tools.mdx @@ -64,11 +64,11 @@ If `run_code` is not allowed (for example, a subagent restricted with `allowed_t `run_code` hands the code to the session [environment](#run-tools-in-an-environment)'s `run_code`, together with a `call_tool` callback. Tool calls always run on the host through that callback, so hooks still see every call. `timeout_seconds` cancels the environment call. The builtin `LocalEnvironment` starts a fresh Python process for every call (`sys.executable`, working directory is the workspace) and forwards tool calls as JSON lines over its stdin and stdout, so arguments and results must be JSON-serializable. When the code finishes, fails or times out, it kills the process and everything it started. That process runs on the host with the same permissions as `bash`, so enable code mode only where `bash` would be acceptable. The return annotation of a tool function (or `Tool.output_schema`, for tools built by hand) determines the result type shown in the stub. -### Per-request tool presentation +### Tool catalogs -`Agent.tool_providers` accepts async callbacks `(tools, tape) -> (tools, tool_prompt)`. Bub applies the current tool scope and prepares code mode first, then runs these callbacks in registration order. They receive runtime names; model aliases are applied afterward. The returned tools and prompt fragment are used for that request. +`Agent.catalogs` contains a builtin catalog and a final code-mode catalog. Insert plugin catalogs before code mode with `agent.catalogs.insert(-1, source)` to make their tools available for discovery without immediately registering them in `Agent.tools`. The `ToolCatalog` protocol has two members: a `tools` mapping of runtime names to `Tool` instances, and async `prepare(tools, tape) -> (tools, tool_prompt)`. -Providers prepare already registered tools; they do not register tools or change connection lifecycles. A plugin binds its discovery helper alongside its tools and keeps discovery within the supplied scope. Full native definitions are sent through the tool schema instead of repeated in the system prompt. +`Agent.known_tools` combines current catalog entries for commands and `allowed_tools`. Bub applies that scope before calling each catalog's `prepare` in order, then converts model aliases. Builtins remain directly available; code mode presents the final selected tools through its full stub. A plugin catalog registers selected tools in `Agent.tools` before returning them; direct comma commands register the named tool when called. Plugins own catalog removal and connection cleanup. Full native definitions are sent through tool schemas instead of repeated in the system prompt. ## Run tools in an environment diff --git a/website/src/content/docs/zh-cn/docs/build/tools.mdx b/website/src/content/docs/zh-cn/docs/build/tools.mdx index a819fb5b..8e589f5c 100644 --- a/website/src/content/docs/zh-cn/docs/build/tools.mdx +++ b/website/src/content/docs/zh-cn/docs/build/tools.mdx @@ -65,11 +65,11 @@ Code mode 让模型在 Python 中调用工具,而不必一次只发起一个 `run_code` 会把代码和一个 `call_tool` 回调一起交给当前 session [执行环境](#在执行环境中运行工具)的 `run_code` 执行。工具调用始终经这个回调在宿主上执行,所以 hook 仍能看到每一次调用。超过 `timeout_seconds` 时会取消执行环境里的执行。builtin 的 `LocalEnvironment` 每次调用都会启动一个新的 Python 进程(使用 `sys.executable`,工作目录为 workspace),工具调用以 JSON 行的形式经进程的 stdin/stdout 转发,因此参数和结果都必须能 JSON 序列化。代码执行完、出错或超时后,它会杀掉该进程及其启动的所有子进程。这个进程直接跑在宿主上,权限与 `bash` 相同,因此只应在允许使用 `bash` 的场景下开启 code mode。stub 中的结果类型来自工具函数的返回注解(手动构造的工具则来自 `Tool.output_schema`)。 -### 每次请求的工具呈现 +### 工具目录 -`Agent.tool_providers` 接受异步回调 `(tools, tape) -> (tools, tool_prompt)`。Bub 先应用当前工具范围、准备 code mode,再按注册顺序执行这些回调。输入使用运行时名称,之后才转换模型别名;返回的工具及提示片段用于本次请求。 +`Agent.catalogs` 默认包含 builtin catalog 和末尾的 code-mode catalog。通过 `agent.catalogs.insert(-1, source)` 将插件目录插入 code mode 之前,就能让工具参与发现,而无需立即注册到 `Agent.tools`。`ToolCatalog` 协议包含两个成员:以运行时名称映射到 `Tool` 的 `tools`,以及异步方法 `prepare(tools, tape) -> (tools, tool_prompt)`。 -provider 准备已注册的工具,工具注册与连接生命周期仍由插件管理。插件绑定自己的发现 helper,并将查询限制在传入范围内。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。 +`Agent.known_tools` 合并当前目录条目,供命令查找及 `allowed_tools` 使用。Bub 先应用工具范围,再依次调用各目录的 `prepare`,最后转换模型别名。内置工具直接可用;code mode 通过完整 stub 呈现最终选中的工具。插件目录在返回选中的工具前将其注册到 `Agent.tools`;直接执行逗号命令时才注册点名的工具。目录移除与连接清理由插件管理。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。 ## 在执行环境中运行工具 From 21e4f0f52927e5686ce8b6fce7d07f947bd8d47b Mon Sep 17 00:00:00 2001 From: Chojan Shang Date: Fri, 2 Oct 2026 02:42:46 +0800 Subject: [PATCH 2/8] refactor(tools): clarify catalog registration and presentation --- src/bub/builtin/agent.py | 13 +++-- src/bub/builtin/catalogs.py | 52 ------------------- src/bub/builtin/codemode/__init__.py | 24 +++++++++ src/bub/tools.py | 16 +++++- tests/test_builtin_agent.py | 6 +-- tests/test_tool_catalogs.py | 26 ++++++---- website/src/content/docs/docs/build/tools.mdx | 8 +-- .../content/docs/zh-cn/docs/build/tools.mdx | 8 +-- 8 files changed, 77 insertions(+), 76 deletions(-) delete mode 100644 src/bub/builtin/catalogs.py diff --git a/src/bub/builtin/agent.py b/src/bub/builtin/agent.py index 8336d3f2..31337f50 100644 --- a/src/bub/builtin/agent.py +++ b/src/bub/builtin/agent.py @@ -17,7 +17,7 @@ from loguru import logger -from bub.builtin.catalogs import BuiltinToolCatalog, CodeModeCatalog +from bub.builtin.codemode import CodeModeCatalog from bub.builtin.commands import strip_command_prefix, validate_command_prefix from bub.builtin.model_runner import ( ModelRunner, @@ -30,7 +30,7 @@ from bub.store import AsyncTapeStore, AsyncTapeStoreAdapter, InMemoryTapeStore, TapeStore, is_async_tape_store from bub.streaming import AsyncStreamEvents, StreamEvent, StreamState from bub.tape import Tape -from bub.tools import REGISTRY, Tool, ToolCatalog, ToolContext, model_tools +from bub.tools import REGISTRY, DirectToolCatalog, Tool, ToolCatalog, ToolContext, model_tools from bub.tracing import Span, current_span from bub.turn import TurnState from bub.utils import workspace_from_state @@ -74,7 +74,7 @@ def __init__( ) self.framework = framework self.tools = {tool.name: tool for tool in tools} if tools is not None else REGISTRY.copy() - self.catalogs: list[ToolCatalog] = [BuiltinToolCatalog(self.tools), CodeModeCatalog(self.tools)] + self.catalogs: list[ToolCatalog] = [DirectToolCatalog(self.tools), CodeModeCatalog()] self.tape_store = tape_store self.skill_dirs = skill_dirs self.model_runner = ModelRunner(self.settings, hooks=framework.get_agent_hooks()) @@ -84,6 +84,11 @@ def known_tools(self) -> dict[str, Tool]: """Current execution tools plus catalog entries, with later sources taking precedence.""" return {name: item for catalog in self.catalogs for name, item in catalog.tools.items()} + def add_catalog(self, catalog: ToolCatalog) -> None: + """Register a source once, after existing sources and before code-mode presentation.""" + if all(item is not catalog for item in self.catalogs): + self.catalogs.insert(-1, catalog) + @cached_property def tape(self) -> Tape: """Return the lazily constructed, cached tape factory for this agent. @@ -487,7 +492,9 @@ async def _run_once_stream( ) -> AsyncStreamEvents: tools_prompts: list[str] = [] for catalog in self.catalogs: + owned = catalog.tools tools, tools_prompt = await catalog.prepare(tools, tape) + self.tools.update({item.name: item for item in tools if owned.get(item.name) is item}) if tools_prompt: tools_prompts.append(tools_prompt) system_prompt = self._system_prompt( diff --git a/src/bub/builtin/catalogs.py b/src/bub/builtin/catalogs.py deleted file mode 100644 index 52421823..00000000 --- a/src/bub/builtin/catalogs.py +++ /dev/null @@ -1,52 +0,0 @@ -"""Default tool catalogs: direct tools first, code-mode presentation last.""" - -from __future__ import annotations - -from dataclasses import dataclass - -from bub.tape import Tape -from bub.tools import Tool, model_tools -from bub.utils import workspace_from_state - - -@dataclass -class BuiltinToolCatalog: - """Expose the agent's registered tools directly.""" - - tools: dict[str, Tool] - - async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - return tools, "" - - -@dataclass -class CodeModeCatalog: - """Present the selected tools as a code stub when code mode is enabled.""" - - registry: dict[str, Tool] - - @property - def tools(self) -> dict[str, Tool]: - return {name: item for name, item in self.registry.items() if name == "run_code"} - - async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - from bub.builtin.codemode import ( - CODE_MODE_STATE_KEY, - CODE_TOOLS_STATE_KEY, - RUN_CODE_TOOL_NAME, - render_code_mode_prompt, - write_tool_stub, - ) - - state = tape.context.state - direct_tools = [item for item in tools if item.name != RUN_CODE_TOOL_NAME] - if not state.get(CODE_MODE_STATE_KEY) or len(direct_tools) == len(tools): - state.pop(CODE_TOOLS_STATE_KEY, None) - return direct_tools, "" - - code_tools = [item for item in direct_tools if item.code_use] - state[CODE_TOOLS_STATE_KEY] = model_tools(code_tools) - stub_path = write_tool_stub( - code_tools, session_id=str(state.get("session_id", "")), workspace=workspace_from_state(state) - ) - return [item for item in tools if item.preserve], render_code_mode_prompt(stub_path) diff --git a/src/bub/builtin/codemode/__init__.py b/src/bub/builtin/codemode/__init__.py index fc6462a1..503763da 100644 --- a/src/bub/builtin/codemode/__init__.py +++ b/src/bub/builtin/codemode/__init__.py @@ -17,7 +17,9 @@ from bub.environment import CodeFailed from bub.errors import BubError, ErrorKind from bub.hooks.interception import AgentHooks +from bub.tape import Tape from bub.tools import Tool, ToolContext, ToolExecutor, model_tools, tool +from bub.utils import workspace_from_state RUN_CODE_TOOL_NAME = "run_code" CODE_MODE_STATE_KEY = "code_mode" @@ -308,3 +310,25 @@ async def set_code_mode(enable: bool, *, context: ToolContext) -> str: """ await set_session_setting(context, CODE_MODE_STATE_KEY, enable) return f"Session code mode {'enabled' if enable else 'disabled'} (applies from the next turn)." + + +class CodeModeCatalog: + """Adapt the final scoped toolset to code mode; contribute no tool definitions.""" + + @property + def tools(self) -> dict[str, Tool]: + return {} + + async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: + state = tape.context.state + direct_tools = [item for item in tools if item.name != RUN_CODE_TOOL_NAME] + if not state.get(CODE_MODE_STATE_KEY) or len(direct_tools) == len(tools): + state.pop(CODE_TOOLS_STATE_KEY, None) + return direct_tools, "" + + code_tools = [item for item in direct_tools if item.code_use] + state[CODE_TOOLS_STATE_KEY] = model_tools(code_tools) + stub_path = write_tool_stub( + code_tools, session_id=str(state.get("session_id", "")), workspace=workspace_from_state(state) + ) + return [item for item in tools if item.preserve], render_code_mode_prompt(stub_path) diff --git a/src/bub/tools.py b/src/bub/tools.py index 9d5e589a..0b912e77 100644 --- a/src/bub/tools.py +++ b/src/bub/tools.py @@ -220,7 +220,11 @@ def validated(*args: Any, **kwargs: Any) -> Any: class ToolCatalog(Protocol): - """Known tools and their per-request preparation, owned by one source.""" + """Declare known tools and prepare the scoped toolset and guidance for a request. + + Tools use runtime names. Returned declared tool instances are registered by + the agent; discovery state and resource cleanup remain owned by the catalog. + """ @property def tools(self) -> dict[str, Tool]: ... @@ -228,6 +232,16 @@ def tools(self) -> dict[str, Tool]: ... async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: ... +@dataclass +class DirectToolCatalog: + """Make a collection of tools directly available within the current scope.""" + + tools: dict[str, Tool] + + async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: + return tools, "" + + def model_tools(tools: Iterable[Tool]) -> list[Tool]: """Convert agent-enabled runtime tools into model-safe aliases.""" return [replace(tool_item, name=tool_item.name.replace(".", "_")) for tool_item in tools if tool_item.agent_use] diff --git a/tests/test_builtin_agent.py b/tests/test_builtin_agent.py index 33936465..54054ab5 100644 --- a/tests/test_builtin_agent.py +++ b/tests/test_builtin_agent.py @@ -12,14 +12,14 @@ import bub.builtin.codemode import bub.builtin.tools # noqa: F401 — registers builtin tools (incl. `model`) from bub.builtin.agent import Agent -from bub.builtin.catalogs import BuiltinToolCatalog, CodeModeCatalog +from bub.builtin.codemode import CodeModeCatalog from bub.builtin.model_runner import ModelRunner from bub.builtin.settings import AgentSettings from bub.builtin.steering import InMemorySteeringInbox from bub.errors import BubError from bub.streaming import AsyncStreamEvents, StreamEvent, StreamState from bub.tape import TapeContext -from bub.tools import REGISTRY, tool +from bub.tools import REGISTRY, DirectToolCatalog, tool # --------------------------------------------------------------------------- # Agent.run() tests: merge_back logic and model passthrough @@ -56,7 +56,7 @@ async def build_prompt(message: dict[str, Any], session_id: str, state: dict[str agent.command_prefix = agent.settings.command_prefix agent.framework = framework agent.tools = REGISTRY.copy() - agent.catalogs = [BuiltinToolCatalog(agent.tools), CodeModeCatalog(agent.tools)] + agent.catalogs = [DirectToolCatalog(agent.tools), CodeModeCatalog()] agent.tape_store = None agent.skill_dirs = None agent.model_runner = _FakeModelRunner(agent.settings) diff --git a/tests/test_tool_catalogs.py b/tests/test_tool_catalogs.py index b15440e4..1ccecae3 100644 --- a/tests/test_tool_catalogs.py +++ b/tests/test_tool_catalogs.py @@ -11,12 +11,12 @@ from bub.builtin.codemode import run_code from bub.framework import BubFramework from bub.tape import Tape -from bub.tools import Tool +from bub.tools import DirectToolCatalog, Tool @pytest.mark.asyncio @pytest.mark.parametrize("code_mode", [False, True]) -async def test_catalog_tools_and_prompt_reach_the_model_within_scope( +async def test_catalog_selection_registers_scoped_tools_for_native_and_code_calls( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, code_mode: bool ) -> None: monkeypatch.setenv("BUB_HOME", str(tmp_path)) @@ -25,6 +25,7 @@ async def test_catalog_tools_and_prompt_reach_the_model_within_scope( framework.load_builtin_hooks() direct = Tool.from_callable(lambda: "direct", name="direct") denied = Tool.from_callable(lambda: "denied", name="denied") + pending = Tool.from_callable(lambda: "pending", name="catalog.pending") calls: list[str] = [] def lookup(name: str) -> str: @@ -33,13 +34,9 @@ def lookup(name: str) -> str: supplied = Tool.from_callable(lookup, name="catalog.lookup") - class Catalog: - def __init__(self) -> None: - self.tools = {supplied.name: supplied} - + class Catalog(DirectToolCatalog): async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - tape.context.state["_runtime_agent"].tools[supplied.name] = supplied - return tools, "Use catalog_lookup to greet the requested person." + return [item for item in tools if item is not pending], "Use catalog_lookup to greet the requested person." requests: list[dict[str, Any]] = [] @@ -77,20 +74,27 @@ async def acompletion(self, **kwargs: Any) -> ChatCompletion: }) monkeypatch.setattr("bub.builtin.model_runner.AnyLLM.create", lambda *args, **kwargs: Provider()) - agent = Agent(framework, tools=[direct, denied, run_code], skill_dirs=[]) - agent.catalogs.insert(-1, Catalog()) + agent = Agent(framework, tools=[direct, run_code], skill_dirs=[]) + agent.add_catalog(Catalog({supplied.name: supplied, denied.name: denied, pending.name: pending})) + assert supplied.name not in agent.tools stream = await agent.run_stream( session_id="catalog", prompt="Greet Ada.", model="openrouter:test-model", - allowed_tools=["direct", "catalog_lookup", "run_code"], + allowed_tools=["direct", "catalog_lookup", "catalog_pending", "run_code"], state={"code_mode": code_mode}, ) async for _ in stream: pass assert calls == ["Ada"] + assert supplied.name in agent.tools + assert denied.name not in agent.tools + assert pending.name not in agent.tools definitions = {item["function"]["name"]: item["function"] for item in requests[0]["tools"]} assert definitions.keys() == ({"run_code"} if code_mode else {"direct", "catalog_lookup"}) + if code_mode: + stub = next((tmp_path / "codemode").rglob("*.pyi")).read_text() + assert "catalog_pending" not in stub assert any( "Use catalog_lookup" in message["content"] for message in requests[0]["messages"] if message["role"] == "system" ) diff --git a/website/src/content/docs/docs/build/tools.mdx b/website/src/content/docs/docs/build/tools.mdx index a72d4091..55009861 100644 --- a/website/src/content/docs/docs/build/tools.mdx +++ b/website/src/content/docs/docs/build/tools.mdx @@ -66,9 +66,11 @@ If `run_code` is not allowed (for example, a subagent restricted with `allowed_t ### Tool catalogs -`Agent.catalogs` contains a builtin catalog and a final code-mode catalog. Insert plugin catalogs before code mode with `agent.catalogs.insert(-1, source)` to make their tools available for discovery without immediately registering them in `Agent.tools`. The `ToolCatalog` protocol has two members: a `tools` mapping of runtime names to `Tool` instances, and async `prepare(tools, tape) -> (tools, tool_prompt)`. +`Tool` defines one callable capability. `ToolCatalog` declares known tools and prepares the tools and guidance available for each request. Both contracts live in `bub.tools`; `DirectToolCatalog(tools)` provides immediate availability for a collection of tools. -`Agent.known_tools` combines current catalog entries for commands and `allowed_tools`. Bub applies that scope before calling each catalog's `prepare` in order, then converts model aliases. Builtins remain directly available; code mode presents the final selected tools through its full stub. A plugin catalog registers selected tools in `Agent.tools` before returning them; direct comma commands register the named tool when called. Plugins own catalog removal and connection cleanup. Full native definitions are sent through tool schemas instead of repeated in the system prompt. +Use `agent.add_catalog(source)` to register a source once, after existing sources and before the final code-mode catalog. Its `tools` mapping uses runtime names and existing `Tool` instances; `prepare(tools, tape) -> (tools, tool_prompt)` receives the whole scoped toolset. A source preserves tools it does not own while selecting its own definitions and optional discovery helpers. Presentation catalogs such as code mode can transform the selected toolset. Deferred exposure is a catalog policy, not the default. + +`Agent.known_tools` combines catalog entries for commands and `allowed_tools`. The Agent applies scope, prepares each catalog, and registers its returned declared tool instances in `Agent.tools` before continuing. Catalog implementations do not need to mutate the execution registry. Builtins remain directly available; code mode uses the final selected tools in its full stub. Direct comma commands register the named tool when called. Plugins own discovery state, catalog removal and connection cleanup. Full native definitions are sent through tool schemas instead of repeated in the system prompt. ## Run tools in an environment @@ -91,7 +93,7 @@ The `REGISTRY` lives in [`bub.tools`](https://github.com/bubbuild/bub/blob/main/ REGISTRY: dict[str, Tool] = {} ``` -Every `@tool` call mutates this dict at **import time**. Bub's builtin agent reads from `REGISTRY` when assembling the tool list for the model. There is no separate registration step. +Every `@tool` call mutates this dict at **import time**. `Agent(tools=None)` snapshots it when created; later imports do not update existing agents. Add tools to an instance's `Agent.tools`, or use `agent.add_catalog(source)` for a tool collection with its own preparation policy. ## Import the tools module from your plugin diff --git a/website/src/content/docs/zh-cn/docs/build/tools.mdx b/website/src/content/docs/zh-cn/docs/build/tools.mdx index 8e589f5c..a93acf15 100644 --- a/website/src/content/docs/zh-cn/docs/build/tools.mdx +++ b/website/src/content/docs/zh-cn/docs/build/tools.mdx @@ -67,9 +67,11 @@ Code mode 让模型在 Python 中调用工具,而不必一次只发起一个 ### 工具目录 -`Agent.catalogs` 默认包含 builtin catalog 和末尾的 code-mode catalog。通过 `agent.catalogs.insert(-1, source)` 将插件目录插入 code mode 之前,就能让工具参与发现,而无需立即注册到 `Agent.tools`。`ToolCatalog` 协议包含两个成员:以运行时名称映射到 `Tool` 的 `tools`,以及异步方法 `prepare(tools, tape) -> (tools, tool_prompt)`。 +`Tool` 定义一项可调用能力,`ToolCatalog` 声明已知工具并准备每次请求可用的工具与使用提示。两个契约都位于 `bub.tools`;`DirectToolCatalog(tools)` 让一组工具立即可用。 -`Agent.known_tools` 合并当前目录条目,供命令查找及 `allowed_tools` 使用。Bub 先应用工具范围,再依次调用各目录的 `prepare`,最后转换模型别名。内置工具直接可用;code mode 通过完整 stub 呈现最终选中的工具。插件目录在返回选中的工具前将其注册到 `Agent.tools`;直接执行逗号命令时才注册点名的工具。目录移除与连接清理由插件管理。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。 +通过 `agent.add_catalog(source)` 注册工具来源。同一个实例只注册一次,顺序位于已有来源之后、末尾的 code-mode catalog 之前。`tools` 使用运行时名称和已有的 `Tool` 实例;`prepare(tools, tape) -> (tools, tool_prompt)` 接收整个经过 scope 过滤的工具集合。工具来源保留其他来源的工具,选择自己的完整定义,并可添加发现工具。Code mode 等呈现目录可以转换最终工具集合。按需暴露由目录自身决定,并非默认行为。 + +`Agent.known_tools` 合并目录条目,供命令查找及 `allowed_tools` 使用。Agent 应用 scope、准备各目录,并在继续处理前将返回的已声明工具实例注册到 `Agent.tools`,目录实现无需修改执行 registry。内置工具直接可用;code mode 的完整 stub 使用最终选中的工具。直接执行逗号命令时才注册点名的工具。发现状态、目录移除与连接清理由插件管理。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。 ## 在执行环境中运行工具 @@ -93,7 +95,7 @@ Code mode 让模型在 Python 中调用工具,而不必一次只发起一个 REGISTRY: dict[str, Tool] = {} ``` -每一次 `@tool` 调用都会在**导入时**修改这个字典。Bub 内置代理在为模型组装工具列表时从 `REGISTRY` 读取。没有独立的注册步骤。 +每一次 `@tool` 调用都会在**导入时**修改这个字典。`Agent(tools=None)` 在创建时取得其快照,之后导入的工具不会自动更新已有 Agent。可直接向实例的 `Agent.tools` 添加工具,或通过 `agent.add_catalog(source)` 接入具有独立准备策略的工具集合。 ## 在插件里导入工具模块 From 9b0518d20e0400ebb17aad12aa88413cbeb2d610 Mon Sep 17 00:00:00 2001 From: Chojan Shang Date: Fri, 2 Oct 2026 03:24:41 +0800 Subject: [PATCH 3/8] fix: preserve native tool loading history for prompt caching --- src/bub/builtin/agent.py | 11 +- src/bub/builtin/codex_provider.py | 37 ++++- src/bub/builtin/model_runner.py | 26 +++- src/bub/builtin/tool_loading.py | 24 ++++ src/bub/tools.py | 1 + tests/test_tool_loading.py | 130 ++++++++++++++++++ website/src/content/docs/docs/build/tools.mdx | 2 + .../content/docs/zh-cn/docs/build/tools.mdx | 2 + 8 files changed, 226 insertions(+), 7 deletions(-) create mode 100644 src/bub/builtin/tool_loading.py create mode 100644 tests/test_tool_loading.py diff --git a/src/bub/builtin/agent.py b/src/bub/builtin/agent.py index 31337f50..47afb617 100644 --- a/src/bub/builtin/agent.py +++ b/src/bub/builtin/agent.py @@ -82,7 +82,12 @@ def __init__( @property def known_tools(self) -> dict[str, Tool]: """Current execution tools plus catalog entries, with later sources taking precedence.""" - return {name: item for catalog in self.catalogs for name, item in catalog.tools.items()} + tools: dict[str, Tool] = {} + for catalog in self.catalogs: + for name, item in catalog.tools.items(): + tools.pop(name, None) + tools[name] = item + return tools def add_catalog(self, catalog: ToolCatalog) -> None: """Register a source once, after existing sources and before code-mode presentation.""" @@ -541,11 +546,11 @@ def _system_prompt( blocks: list[str] = [] if result := self.framework.get_system_prompt(prompt=prompt, state=state): blocks.append(result) - if tools_prompt: - blocks.append(tools_prompt) workspace = workspace_from_state(state) if skills_prompt := self._load_skills_prompt(prompt, workspace, allowed_skills): blocks.append(skills_prompt) + if tools_prompt: + blocks.append(tools_prompt) return "\n\n".join(blocks) def _has_steering_messages(self, state: TurnState) -> bool: diff --git a/src/bub/builtin/codex_provider.py b/src/bub/builtin/codex_provider.py index 2aa523fb..f0ed65b8 100644 --- a/src/bub/builtin/codex_provider.py +++ b/src/bub/builtin/codex_provider.py @@ -2,6 +2,7 @@ from __future__ import annotations +import re import time from collections.abc import AsyncIterator, Mapping, Sequence from dataclasses import dataclass @@ -128,10 +129,28 @@ def _completion_params_to_responses_params(self, params: CompletionParams) -> Re if params.reasoning_effort not in {None, "auto"}: reasoning = {"effort": params.reasoning_effort} + messages = params.messages + tools = cast("Sequence[Any] | None", params.tools) + resets = [ + index + for index, message in enumerate(messages) + if message.get("type") == "tool_definitions" and message.get("reset") + ] + if not supports_codex_tool_loading(params.model_id) or params.tool_choice not in (None, "auto", "none"): + messages = [message for message in messages if message.get("type") != "tool_definitions"] + resets = [] + if resets: + start = resets[-1] + tools = messages[start]["tools"] + messages = [ + message + for index, message in enumerate(messages) + if message.get("type") != "tool_definitions" or index > start + ] return ResponsesParams( model=params.model_id, - input=cast("Any", _completion_messages_to_responses_input(params.messages)), - tools=self._completion_tools_to_response_tools(cast("Sequence[Any] | None", params.tools)), + input=cast("Any", _completion_messages_to_responses_input(messages)), + tools=self._completion_tools_to_response_tools(tools), tool_choice=self._completion_tool_choice_to_response_tool_choice(params.tool_choice), response_format=params.response_format, stream=params.stream, @@ -405,6 +424,14 @@ def _completion_messages_to_responses_input(messages: Sequence[Any]) -> list[dic if not payload: continue + if payload.get("type") == "tool_definitions": + response_input.append({ + "type": "additional_tools", + "role": "developer", + "tools": OpenaiCodexProvider._completion_tools_to_response_tools(payload["tools"]), + }) + continue + role = payload.get("role") if role == "tool": tool_result = _completion_tool_result_to_response_item(payload) @@ -467,6 +494,12 @@ def _mapping_from_value(value: Any) -> Mapping[str, Any]: return {} +def supports_codex_tool_loading(model_id: str) -> bool: + """Keep older models on the existing top-level tool definition path.""" + version = re.match(r"^gpt-(\d+)(?:\.(\d+))?(?:-|$)", model_id) + return bool(version and (int(version[1]), int(version[2] or 0)) >= (5, 4)) + + def should_use_openai_codex_provider( provider: str, model_id: str, *, api_key: str | None, api_base: str | None ) -> bool: diff --git a/src/bub/builtin/model_runner.py b/src/bub/builtin/model_runner.py index bcf38fcf..c5a37319 100644 --- a/src/bub/builtin/model_runner.py +++ b/src/bub/builtin/model_runner.py @@ -29,8 +29,13 @@ from loguru import logger from pydantic import TypeAdapter, ValidationError -from bub.builtin.codex_provider import OpenaiCodexProvider, should_use_openai_codex_provider +from bub.builtin.codex_provider import ( + OpenaiCodexProvider, + should_use_openai_codex_provider, + supports_codex_tool_loading, +) from bub.builtin.settings import AgentSettings, ModelCandidate +from bub.builtin.tool_loading import tool_definition_update from bub.channels.message import audio_mime_type_from_format from bub.errors import BubError, ErrorKind from bub.hooks.interception import ( @@ -149,7 +154,10 @@ async def completion_response( "gen_ai.request.model": candidate.model_id, }) streaming = llm.SUPPORTS_COMPLETION_STREAMING - completion_messages = _adapt_messages_for_provider(messages, candidate.provider) + native_messages = messages + if not isinstance(llm, OpenaiCodexProvider): + native_messages = [item for item in messages if item.get("type") != "tool_definitions"] + completion_messages = _adapt_messages_for_provider(native_messages, candidate.provider) completion_kwargs = { **self.settings.completion_args, **_extra_options(llm, stream=streaming), @@ -194,6 +202,20 @@ async def iterator() -> AsyncGenerator[StreamEvent, None]: model=model, steering_messages=steering_messages, ) + client_kwargs = self.settings.model_client_kwargs("openai") + if ( + model.startswith("openai:") + and supports_codex_tool_loading(model.partition(":")[2]) + and should_use_openai_codex_provider( + "openai", + model.partition(":")[2], + api_key=client_kwargs.get("api_key"), + api_base=client_kwargs.get("api_base"), + ) + and (definitions := tool_definition_update(messages, tools)) + ): + messages.insert(len(messages) - len(new_messages), definitions) + new_messages.insert(0, definitions) output = ModelOutputAccumulator() request = LlmCallRequest( run_id=run_id, diff --git a/src/bub/builtin/tool_loading.py b/src/bub/builtin/tool_loading.py new file mode 100644 index 00000000..b4e35c8b --- /dev/null +++ b/src/bub/builtin/tool_loading.py @@ -0,0 +1,24 @@ +"""Record native definition additions at their position in a conversation.""" + +from __future__ import annotations + +from copy import deepcopy +from typing import Any + +from bub.tools import Tool + + +def tool_definition_update(messages: list[dict[str, Any]], tools: list[Tool]) -> dict[str, Any] | None: + """Append additions; start a new definition prefix when scope or schemas change.""" + declared: dict[str, dict[str, Any]] = {} + history = [message for message in messages if message.get("type") == "tool_definitions"] + for message in history: + if message.get("reset"): + declared.clear() + declared.update((item["function"]["name"], item) for item in message["tools"]) + current = {item.name: item.to_schema() for item in tools} + reset = not history or any(name not in current or current[name] != item for name, item in declared.items()) + added = list(current.values()) if reset else [item for name, item in current.items() if name not in declared] + if not added and not reset: + return None + return {"role": "developer", "type": "tool_definitions", "tools": deepcopy(added), "reset": reset} diff --git a/src/bub/tools.py b/src/bub/tools.py index 0b912e77..8f72ef1c 100644 --- a/src/bub/tools.py +++ b/src/bub/tools.py @@ -224,6 +224,7 @@ class ToolCatalog(Protocol): Tools use runtime names. Returned declared tool instances are registered by the agent; discovery state and resource cleanup remain owned by the catalog. + Mapping order determines discovery order independently of execution registration. """ @property diff --git a/tests/test_tool_loading.py b/tests/test_tool_loading.py new file mode 100644 index 00000000..38e6ae33 --- /dev/null +++ b/tests/test_tool_loading.py @@ -0,0 +1,130 @@ +from __future__ import annotations + +import json +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import pytest +from any_llm.types.responses import ResponsesParams + +from bub.builtin.agent import Agent +from bub.builtin.codex_provider import OpenaiCodexProvider +from bub.framework import BubFramework +from bub.store import FileTapeStore +from bub.tape import InMemoryTapeStore, Tape +from bub.tools import DirectToolCatalog, Tool, ToolContext + + +@pytest.mark.asyncio +@pytest.mark.parametrize("persistent", [False, True]) +async def test_responses_appends_native_definitions_and_reuses_them_after_restart( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + persistent: bool, +) -> None: + monkeypatch.setenv("BUB_HOME", str(tmp_path)) + framework = BubFramework(config_file=tmp_path / "config.yml") + framework.workspace = tmp_path + framework.load_builtin_hooks() + calls: list[str] = [] + remote = { + name: Tool.from_callable(lambda name=name: calls.append(name) or name, name=f"catalog.{name}") + for name in ("first", "second") + } + + async def discover(name: str, *, context: ToolContext) -> str: + await context.tape.append_event("catalog.selected", {"name": name}, context=False) + return f"Loaded {name}" + + class Catalog(DirectToolCatalog): + async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: + entries = await tape.store.fetch_all(tape.context.build_query(tape.query()).kinds("event")) + loaded = { + entry.payload["data"]["name"] for entry in entries if entry.payload.get("name") == "catalog.selected" + } + return [ + item for item in tools if item.name not in self.tools or item.name in loaded + ], "Discover tools by name." + + wire: list[ResponsesParams] = [] + replies: list[str | tuple[str, dict[str, Any]]] = [ + ("discover", {"name": "catalog.second"}), + ("catalog_second", {}), + ("discover", {"name": "catalog.first"}), + ("catalog_first", {}), + "done", + ] + monkeypatch.setattr(OpenaiCodexProvider, "_init_client", lambda *a, **k: None) + client = OpenaiCodexProvider(api_key="fixture") + + async def responses(params: ResponsesParams, **kwargs: Any): + wire.append(params) + reply = replies.pop(0) + + async def events(): + if isinstance(reply, tuple): + name, arguments = reply + yield SimpleNamespace( + type="response.output_item.added", + output_index=0, + item=SimpleNamespace( + type="function_call", id="item", call_id=f"call-{len(wire)}", name=name, arguments="" + ), + ) + yield SimpleNamespace( + type="response.function_call_arguments.delta", output_index=0, delta=json.dumps(arguments) + ) + else: + yield SimpleNamespace(type="response.output_text.delta", delta=reply) + yield SimpleNamespace( + type="response.completed", response=SimpleNamespace(id="reply", created_at=0, usage=None) + ) + + return events() + + monkeypatch.setattr(client, "_aresponses", responses) + monkeypatch.setattr("bub.builtin.model_runner.should_use_openai_codex_provider", lambda *a, **k: True) + monkeypatch.setattr("bub.builtin.model_runner.ModelRunner.create_llm_client", lambda *a, **k: client) + memory_store = InMemoryTapeStore() + + def agent() -> Agent: + result = Agent( + framework, + tools=[Tool.from_callable(discover, context=True)], + skill_dirs=[], + tape_store=FileTapeStore(tmp_path / "store") if persistent else memory_store, + ) + result.add_catalog(Catalog({item.name: item for item in remote.values()})) + return result + + async def run(instance: Agent, **kwargs: Any) -> None: + stream = await instance.run_stream( + session_id="loading", prompt="Use both tools.", model="openai:gpt-5.5", **kwargs + ) + async for _ in stream: + pass + + await run(agent()) + assert calls == ["second", "first"] + assert all(request.tools == wire[0].tools for request in wire) + assert [item["name"] for item in wire[0].tools or []] == ["discover"] + loaded = [item for item in wire[3].input if item.get("type") == "additional_tools"] + assert [item["tools"][0]["name"] for item in loaded] == ["catalog_second", "catalog_first"] + assert wire[3].input[: len(wire[2].input)] == wire[2].input + replies.extend([("catalog_second", {}), "done"]) + await run(agent()) + assert calls == ["second", "first", "second"] + assert [item for item in wire[-1].input if item.get("type") == "additional_tools"] == loaded + + remote["second"].parameters["description"] = "Updated lookup arguments." + replies.append("done") + await run(agent()) + assert "catalog_second" in {item["name"] for item in wire[-1].tools or []} + assert not any(item.get("type") == "additional_tools" for item in wire[-1].input) + + # A narrower scope must not keep earlier definitions callable. + replies.append("done") + await run(agent(), allowed_tools=["catalog_second"]) + assert [item["name"] for item in wire[-1].tools or []] == ["catalog_second"] + assert not any(item.get("type") == "additional_tools" for item in wire[-1].input) diff --git a/website/src/content/docs/docs/build/tools.mdx b/website/src/content/docs/docs/build/tools.mdx index 55009861..d3618f89 100644 --- a/website/src/content/docs/docs/build/tools.mdx +++ b/website/src/content/docs/docs/build/tools.mdx @@ -72,6 +72,8 @@ Use `agent.add_catalog(source)` to register a source once, after existing source `Agent.known_tools` combines catalog entries for commands and `allowed_tools`. The Agent applies scope, prepares each catalog, and registers its returned declared tool instances in `Agent.tools` before continuing. Catalog implementations do not need to mutate the execution registry. Builtins remain directly available; code mode uses the final selected tools in its full stub. Direct comma commands register the named tool when called. Plugins own discovery state, catalog removal and connection cleanup. Full native definitions are sent through tool schemas instead of repeated in the system prompt. +Catalog mappings determine discovery order independently of the execution registry. Keep discovery summaries stable as definitions load. With Bub's Codex Responses transport on GPT-5.4 and later models, the first native definitions form a fixed prefix and newly selected definitions are appended at their position in the conversation. This history survives restarts. Narrowing scope or changing a schema rebuilds the definition prefix. Other transports, older models and forced tool choices use the current complete top-level toolset. Code mode keeps its native tools fixed and adds stub content to history when the model reads it. Tool handlers and catalog signatures are unchanged. + ## Run tools in an environment `bash`, `bash.output`, `bash.kill` and `fs.*` do not touch the host directly: they run on the session's `Environment` (from [`bub.environment`](https://github.com/bubbuild/bub/blob/main/src/bub/environment.py)), which spawns processes, reads and writes text files, and runs code for `run_code`. Bub keeps the tool behavior — background shells, timeouts, rendering, hooks — on the host, so an environment only needs to implement a few operations: diff --git a/website/src/content/docs/zh-cn/docs/build/tools.mdx b/website/src/content/docs/zh-cn/docs/build/tools.mdx index a93acf15..a7125253 100644 --- a/website/src/content/docs/zh-cn/docs/build/tools.mdx +++ b/website/src/content/docs/zh-cn/docs/build/tools.mdx @@ -73,6 +73,8 @@ Code mode 让模型在 Python 中调用工具,而不必一次只发起一个 `Agent.known_tools` 合并目录条目,供命令查找及 `allowed_tools` 使用。Agent 应用 scope、准备各目录,并在继续处理前将返回的已声明工具实例注册到 `Agent.tools`,目录实现无需修改执行 registry。内置工具直接可用;code mode 的完整 stub 使用最终选中的工具。直接执行逗号命令时才注册点名的工具。发现状态、目录移除与连接清理由插件管理。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。 +目录映射决定发现顺序,独立于执行 registry。加载完整定义时应保持目录摘要稳定。Bub 的 Codex Responses 传输在 GPT-5.4 及后续模型上将首次原生定义放入固定前缀,后续选中的定义按加载位置追加到会话历史,重启后仍保留这些位置。缩小 scope 或修改 schema 会重新建立定义前缀。其他传输、较早模型以及强制选择工具时,使用当前完整的顶层工具集合。Code mode 保持原生工具集合固定,模型读取 stub 时才将其内容追加到历史。工具处理函数和目录接口的签名保持不变。 + ## 在执行环境中运行工具 From e33c350e1dc10bef217acccd4420d07c466254d4 Mon Sep 17 00:00:00 2001 From: Chojan Shang Date: Fri, 2 Oct 2026 03:30:31 +0800 Subject: [PATCH 4/8] refactor: keep progressive tool loading provider independent --- src/bub/builtin/codex_provider.py | 37 +---- src/bub/builtin/model_runner.py | 26 +--- src/bub/builtin/tool_loading.py | 24 ---- tests/test_tool_loading.py | 130 ------------------ website/src/content/docs/docs/build/tools.mdx | 2 +- .../content/docs/zh-cn/docs/build/tools.mdx | 2 +- 6 files changed, 6 insertions(+), 215 deletions(-) delete mode 100644 src/bub/builtin/tool_loading.py delete mode 100644 tests/test_tool_loading.py diff --git a/src/bub/builtin/codex_provider.py b/src/bub/builtin/codex_provider.py index f0ed65b8..2aa523fb 100644 --- a/src/bub/builtin/codex_provider.py +++ b/src/bub/builtin/codex_provider.py @@ -2,7 +2,6 @@ from __future__ import annotations -import re import time from collections.abc import AsyncIterator, Mapping, Sequence from dataclasses import dataclass @@ -129,28 +128,10 @@ def _completion_params_to_responses_params(self, params: CompletionParams) -> Re if params.reasoning_effort not in {None, "auto"}: reasoning = {"effort": params.reasoning_effort} - messages = params.messages - tools = cast("Sequence[Any] | None", params.tools) - resets = [ - index - for index, message in enumerate(messages) - if message.get("type") == "tool_definitions" and message.get("reset") - ] - if not supports_codex_tool_loading(params.model_id) or params.tool_choice not in (None, "auto", "none"): - messages = [message for message in messages if message.get("type") != "tool_definitions"] - resets = [] - if resets: - start = resets[-1] - tools = messages[start]["tools"] - messages = [ - message - for index, message in enumerate(messages) - if message.get("type") != "tool_definitions" or index > start - ] return ResponsesParams( model=params.model_id, - input=cast("Any", _completion_messages_to_responses_input(messages)), - tools=self._completion_tools_to_response_tools(tools), + input=cast("Any", _completion_messages_to_responses_input(params.messages)), + tools=self._completion_tools_to_response_tools(cast("Sequence[Any] | None", params.tools)), tool_choice=self._completion_tool_choice_to_response_tool_choice(params.tool_choice), response_format=params.response_format, stream=params.stream, @@ -424,14 +405,6 @@ def _completion_messages_to_responses_input(messages: Sequence[Any]) -> list[dic if not payload: continue - if payload.get("type") == "tool_definitions": - response_input.append({ - "type": "additional_tools", - "role": "developer", - "tools": OpenaiCodexProvider._completion_tools_to_response_tools(payload["tools"]), - }) - continue - role = payload.get("role") if role == "tool": tool_result = _completion_tool_result_to_response_item(payload) @@ -494,12 +467,6 @@ def _mapping_from_value(value: Any) -> Mapping[str, Any]: return {} -def supports_codex_tool_loading(model_id: str) -> bool: - """Keep older models on the existing top-level tool definition path.""" - version = re.match(r"^gpt-(\d+)(?:\.(\d+))?(?:-|$)", model_id) - return bool(version and (int(version[1]), int(version[2] or 0)) >= (5, 4)) - - def should_use_openai_codex_provider( provider: str, model_id: str, *, api_key: str | None, api_base: str | None ) -> bool: diff --git a/src/bub/builtin/model_runner.py b/src/bub/builtin/model_runner.py index c5a37319..bcf38fcf 100644 --- a/src/bub/builtin/model_runner.py +++ b/src/bub/builtin/model_runner.py @@ -29,13 +29,8 @@ from loguru import logger from pydantic import TypeAdapter, ValidationError -from bub.builtin.codex_provider import ( - OpenaiCodexProvider, - should_use_openai_codex_provider, - supports_codex_tool_loading, -) +from bub.builtin.codex_provider import OpenaiCodexProvider, should_use_openai_codex_provider from bub.builtin.settings import AgentSettings, ModelCandidate -from bub.builtin.tool_loading import tool_definition_update from bub.channels.message import audio_mime_type_from_format from bub.errors import BubError, ErrorKind from bub.hooks.interception import ( @@ -154,10 +149,7 @@ async def completion_response( "gen_ai.request.model": candidate.model_id, }) streaming = llm.SUPPORTS_COMPLETION_STREAMING - native_messages = messages - if not isinstance(llm, OpenaiCodexProvider): - native_messages = [item for item in messages if item.get("type") != "tool_definitions"] - completion_messages = _adapt_messages_for_provider(native_messages, candidate.provider) + completion_messages = _adapt_messages_for_provider(messages, candidate.provider) completion_kwargs = { **self.settings.completion_args, **_extra_options(llm, stream=streaming), @@ -202,20 +194,6 @@ async def iterator() -> AsyncGenerator[StreamEvent, None]: model=model, steering_messages=steering_messages, ) - client_kwargs = self.settings.model_client_kwargs("openai") - if ( - model.startswith("openai:") - and supports_codex_tool_loading(model.partition(":")[2]) - and should_use_openai_codex_provider( - "openai", - model.partition(":")[2], - api_key=client_kwargs.get("api_key"), - api_base=client_kwargs.get("api_base"), - ) - and (definitions := tool_definition_update(messages, tools)) - ): - messages.insert(len(messages) - len(new_messages), definitions) - new_messages.insert(0, definitions) output = ModelOutputAccumulator() request = LlmCallRequest( run_id=run_id, diff --git a/src/bub/builtin/tool_loading.py b/src/bub/builtin/tool_loading.py deleted file mode 100644 index b4e35c8b..00000000 --- a/src/bub/builtin/tool_loading.py +++ /dev/null @@ -1,24 +0,0 @@ -"""Record native definition additions at their position in a conversation.""" - -from __future__ import annotations - -from copy import deepcopy -from typing import Any - -from bub.tools import Tool - - -def tool_definition_update(messages: list[dict[str, Any]], tools: list[Tool]) -> dict[str, Any] | None: - """Append additions; start a new definition prefix when scope or schemas change.""" - declared: dict[str, dict[str, Any]] = {} - history = [message for message in messages if message.get("type") == "tool_definitions"] - for message in history: - if message.get("reset"): - declared.clear() - declared.update((item["function"]["name"], item) for item in message["tools"]) - current = {item.name: item.to_schema() for item in tools} - reset = not history or any(name not in current or current[name] != item for name, item in declared.items()) - added = list(current.values()) if reset else [item for name, item in current.items() if name not in declared] - if not added and not reset: - return None - return {"role": "developer", "type": "tool_definitions", "tools": deepcopy(added), "reset": reset} diff --git a/tests/test_tool_loading.py b/tests/test_tool_loading.py deleted file mode 100644 index 38e6ae33..00000000 --- a/tests/test_tool_loading.py +++ /dev/null @@ -1,130 +0,0 @@ -from __future__ import annotations - -import json -from pathlib import Path -from types import SimpleNamespace -from typing import Any - -import pytest -from any_llm.types.responses import ResponsesParams - -from bub.builtin.agent import Agent -from bub.builtin.codex_provider import OpenaiCodexProvider -from bub.framework import BubFramework -from bub.store import FileTapeStore -from bub.tape import InMemoryTapeStore, Tape -from bub.tools import DirectToolCatalog, Tool, ToolContext - - -@pytest.mark.asyncio -@pytest.mark.parametrize("persistent", [False, True]) -async def test_responses_appends_native_definitions_and_reuses_them_after_restart( - tmp_path: Path, - monkeypatch: pytest.MonkeyPatch, - persistent: bool, -) -> None: - monkeypatch.setenv("BUB_HOME", str(tmp_path)) - framework = BubFramework(config_file=tmp_path / "config.yml") - framework.workspace = tmp_path - framework.load_builtin_hooks() - calls: list[str] = [] - remote = { - name: Tool.from_callable(lambda name=name: calls.append(name) or name, name=f"catalog.{name}") - for name in ("first", "second") - } - - async def discover(name: str, *, context: ToolContext) -> str: - await context.tape.append_event("catalog.selected", {"name": name}, context=False) - return f"Loaded {name}" - - class Catalog(DirectToolCatalog): - async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - entries = await tape.store.fetch_all(tape.context.build_query(tape.query()).kinds("event")) - loaded = { - entry.payload["data"]["name"] for entry in entries if entry.payload.get("name") == "catalog.selected" - } - return [ - item for item in tools if item.name not in self.tools or item.name in loaded - ], "Discover tools by name." - - wire: list[ResponsesParams] = [] - replies: list[str | tuple[str, dict[str, Any]]] = [ - ("discover", {"name": "catalog.second"}), - ("catalog_second", {}), - ("discover", {"name": "catalog.first"}), - ("catalog_first", {}), - "done", - ] - monkeypatch.setattr(OpenaiCodexProvider, "_init_client", lambda *a, **k: None) - client = OpenaiCodexProvider(api_key="fixture") - - async def responses(params: ResponsesParams, **kwargs: Any): - wire.append(params) - reply = replies.pop(0) - - async def events(): - if isinstance(reply, tuple): - name, arguments = reply - yield SimpleNamespace( - type="response.output_item.added", - output_index=0, - item=SimpleNamespace( - type="function_call", id="item", call_id=f"call-{len(wire)}", name=name, arguments="" - ), - ) - yield SimpleNamespace( - type="response.function_call_arguments.delta", output_index=0, delta=json.dumps(arguments) - ) - else: - yield SimpleNamespace(type="response.output_text.delta", delta=reply) - yield SimpleNamespace( - type="response.completed", response=SimpleNamespace(id="reply", created_at=0, usage=None) - ) - - return events() - - monkeypatch.setattr(client, "_aresponses", responses) - monkeypatch.setattr("bub.builtin.model_runner.should_use_openai_codex_provider", lambda *a, **k: True) - monkeypatch.setattr("bub.builtin.model_runner.ModelRunner.create_llm_client", lambda *a, **k: client) - memory_store = InMemoryTapeStore() - - def agent() -> Agent: - result = Agent( - framework, - tools=[Tool.from_callable(discover, context=True)], - skill_dirs=[], - tape_store=FileTapeStore(tmp_path / "store") if persistent else memory_store, - ) - result.add_catalog(Catalog({item.name: item for item in remote.values()})) - return result - - async def run(instance: Agent, **kwargs: Any) -> None: - stream = await instance.run_stream( - session_id="loading", prompt="Use both tools.", model="openai:gpt-5.5", **kwargs - ) - async for _ in stream: - pass - - await run(agent()) - assert calls == ["second", "first"] - assert all(request.tools == wire[0].tools for request in wire) - assert [item["name"] for item in wire[0].tools or []] == ["discover"] - loaded = [item for item in wire[3].input if item.get("type") == "additional_tools"] - assert [item["tools"][0]["name"] for item in loaded] == ["catalog_second", "catalog_first"] - assert wire[3].input[: len(wire[2].input)] == wire[2].input - replies.extend([("catalog_second", {}), "done"]) - await run(agent()) - assert calls == ["second", "first", "second"] - assert [item for item in wire[-1].input if item.get("type") == "additional_tools"] == loaded - - remote["second"].parameters["description"] = "Updated lookup arguments." - replies.append("done") - await run(agent()) - assert "catalog_second" in {item["name"] for item in wire[-1].tools or []} - assert not any(item.get("type") == "additional_tools" for item in wire[-1].input) - - # A narrower scope must not keep earlier definitions callable. - replies.append("done") - await run(agent(), allowed_tools=["catalog_second"]) - assert [item["name"] for item in wire[-1].tools or []] == ["catalog_second"] - assert not any(item.get("type") == "additional_tools" for item in wire[-1].input) diff --git a/website/src/content/docs/docs/build/tools.mdx b/website/src/content/docs/docs/build/tools.mdx index d3618f89..060c93d3 100644 --- a/website/src/content/docs/docs/build/tools.mdx +++ b/website/src/content/docs/docs/build/tools.mdx @@ -72,7 +72,7 @@ Use `agent.add_catalog(source)` to register a source once, after existing source `Agent.known_tools` combines catalog entries for commands and `allowed_tools`. The Agent applies scope, prepares each catalog, and registers its returned declared tool instances in `Agent.tools` before continuing. Catalog implementations do not need to mutate the execution registry. Builtins remain directly available; code mode uses the final selected tools in its full stub. Direct comma commands register the named tool when called. Plugins own discovery state, catalog removal and connection cleanup. Full native definitions are sent through tool schemas instead of repeated in the system prompt. -Catalog mappings determine discovery order independently of the execution registry. Keep discovery summaries stable as definitions load. With Bub's Codex Responses transport on GPT-5.4 and later models, the first native definitions form a fixed prefix and newly selected definitions are appended at their position in the conversation. This history survives restarts. Narrowing scope or changing a schema rebuilds the definition prefix. Other transports, older models and forced tool choices use the current complete top-level toolset. Code mode keeps its native tools fixed and adds stub content to history when the model reads it. Tool handlers and catalog signatures are unchanged. +Catalog mappings determine discovery order independently of the execution registry. Keep discovery summaries stable as definitions load, and append selected tools in the current conversation's discovery order. Put catalog guidance after stable instructions and skills. This preserves existing content and tool order without model-version checks or provider-specific loading messages. Native additions still grow the tool list; the provider's request format determines how much of the earlier prefix remains cacheable. Code mode adds stub content to history when the model reads it. ## Run tools in an environment diff --git a/website/src/content/docs/zh-cn/docs/build/tools.mdx b/website/src/content/docs/zh-cn/docs/build/tools.mdx index a7125253..b4a7b847 100644 --- a/website/src/content/docs/zh-cn/docs/build/tools.mdx +++ b/website/src/content/docs/zh-cn/docs/build/tools.mdx @@ -73,7 +73,7 @@ Code mode 让模型在 Python 中调用工具,而不必一次只发起一个 `Agent.known_tools` 合并目录条目,供命令查找及 `allowed_tools` 使用。Agent 应用 scope、准备各目录,并在继续处理前将返回的已声明工具实例注册到 `Agent.tools`,目录实现无需修改执行 registry。内置工具直接可用;code mode 的完整 stub 使用最终选中的工具。直接执行逗号命令时才注册点名的工具。发现状态、目录移除与连接清理由插件管理。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。 -目录映射决定发现顺序,独立于执行 registry。加载完整定义时应保持目录摘要稳定。Bub 的 Codex Responses 传输在 GPT-5.4 及后续模型上将首次原生定义放入固定前缀,后续选中的定义按加载位置追加到会话历史,重启后仍保留这些位置。缩小 scope 或修改 schema 会重新建立定义前缀。其他传输、较早模型以及强制选择工具时,使用当前完整的顶层工具集合。Code mode 保持原生工具集合固定,模型读取 stub 时才将其内容追加到历史。工具处理函数和目录接口的签名保持不变。 +目录映射决定发现顺序,独立于执行 registry。加载完整定义时保持目录摘要稳定,按当前会话的发现顺序追加选中的工具,并把目录提示放在稳定的指令和技能之后。这样无需模型版本判断或 provider 专用加载消息,就能保持已有内容和工具顺序。原生工具列表仍会随发现而增长;此前前缀有多少可以命中缓存,取决于 provider 的请求格式。Code mode 在模型读取 stub 时才将其内容追加到历史。 ## 在执行环境中运行工具 From 1c768086997fc3509df0f0f9f0fe00ddba054639 Mon Sep 17 00:00:00 2001 From: Chojan Shang Date: Fri, 2 Oct 2026 05:30:25 +0800 Subject: [PATCH 5/8] fix: keep selected skill instructions in conversation history --- src/bub/builtin/agent.py | 24 ++++- src/bub/builtin/codemode/__init__.py | 4 +- src/bub/builtin/hook_impl.py | 1 + tests/test_skill_loading.py | 98 +++++++++++++++++++ .../src/content/docs/docs/build/skills.mdx | 2 + .../content/docs/zh-cn/docs/build/skills.mdx | 2 + 6 files changed, 126 insertions(+), 5 deletions(-) create mode 100644 tests/test_skill_loading.py diff --git a/src/bub/builtin/agent.py b/src/bub/builtin/agent.py index 47afb617..998dbe15 100644 --- a/src/bub/builtin/agent.py +++ b/src/bub/builtin/agent.py @@ -446,8 +446,7 @@ def _load_skills_prompt(self, prompt: str, workspace: Path, allowed_skills: set[ for skill in discover_skills(workspace, skill_dirs=self.skill_dirs) if allowed_skills is None or skill.name.casefold() in allowed_skills } - expanded_skills = set(HINT_RE.findall(prompt)) & set(skill_index.keys()) - return render_skills_prompt(list(skill_index.values()), expanded_skills=expanded_skills) + return render_skills_prompt(list(skill_index.values())) async def _run_once( self, @@ -472,6 +471,23 @@ async def _run_once( if allowed_skills is not None: allowed_skills = {name.casefold() for name in allowed_skills} tape.context.state["allowed_skills"] = list(allowed_skills) + if prompt is not None: + hinted = {name.casefold() for name in HINT_RE.findall(prompt_text)} + selected = [ + skill + for skill in discover_skills(workspace_from_state(tape.context.state), skill_dirs=self.skill_dirs) + if skill.name.casefold() in hinted + and (allowed_skills is None or skill.name.casefold() in allowed_skills) + ] + if selected: + bodies = "\n\n".join( + f'\n{skill.body()}\n' + for skill in selected + ) + if isinstance(prompt, str): + prompt = f"{prompt}\n\n{bodies}" + else: + prompt = [*prompt, {"type": "text", "text": bodies}] if allowed_tools is not None: tools = [tool for tool in known_tools.values() if tool.name in allowed_tools] else: @@ -545,12 +561,12 @@ def _system_prompt( ) -> str: blocks: list[str] = [] if result := self.framework.get_system_prompt(prompt=prompt, state=state): - blocks.append(result) + blocks.append(f"\n{result}\n") workspace = workspace_from_state(state) if skills_prompt := self._load_skills_prompt(prompt, workspace, allowed_skills): blocks.append(skills_prompt) if tools_prompt: - blocks.append(tools_prompt) + blocks.append(f"\n{tools_prompt}\n") return "\n\n".join(blocks) def _has_steering_messages(self, state: TurnState) -> bool: diff --git a/src/bub/builtin/codemode/__init__.py b/src/bub/builtin/codemode/__init__.py index 503763da..6fae6cf6 100644 --- a/src/bub/builtin/codemode/__init__.py +++ b/src/bub/builtin/codemode/__init__.py @@ -245,7 +245,8 @@ def render_code_mode_prompt(stub_path: Path) -> str: "\n" f"More tools are available as async Python functions `tools.(...)` inside `{RUN_CODE_TOOL_NAME}`. " f"Their signatures, result types and documentation are in the stub file: {stub_path}\n" - "Read the stub before calling a tool you have not used yet. Always `await` tool calls (top-level `await` " + "Locate and read only the relevant declarations in the stub before calling an unfamiliar tool. " + "Always `await` tool calls (top-level `await` " "is allowed) and pass keyword arguments; they return structured values and raise on failure. " f"`{RUN_CODE_TOOL_NAME}` returns only what the code prints, so print the results you need, and combine " "several tool calls in one run when possible.\n" @@ -257,6 +258,7 @@ def render_code_mode_prompt(stub_path: Path) -> str: async def run_code(code: str, timeout_seconds: int = DEFAULT_RUN_CODE_TIMEOUT_SECONDS, *, context: ToolContext) -> str: """Run Python code in the environment and return everything it prints. + By default, each call starts a fresh Python process; keep operations that share variables in the same call. Tools are async functions available as `tools.(...)`: await them with keyword arguments (top-level `await` is allowed). See the tool stub file referenced in the system prompt for their signatures and result types. The code is stopped after timeout_seconds. diff --git a/src/bub/builtin/hook_impl.py b/src/bub/builtin/hook_impl.py index 70f05ab3..f9897a18 100644 --- a/src/bub/builtin/hook_impl.py +++ b/src/bub/builtin/hook_impl.py @@ -37,6 +37,7 @@ DEFAULT_SYSTEM_PROMPT = """\ Call tools or skills to finish the task. +Issue independent read calls together in the same model step. Before ending this run, you MUST determine whether a response needs to be sent via channel, checking the following conditions: diff --git a/tests/test_skill_loading.py b/tests/test_skill_loading.py new file mode 100644 index 00000000..74a93675 --- /dev/null +++ b/tests/test_skill_loading.py @@ -0,0 +1,98 @@ +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +import pytest +from any_llm.types.completion import ChatCompletion + +from bub.builtin.agent import Agent +from bub.framework import BubFramework +from bub.tools import Tool + + +@pytest.mark.asyncio +@pytest.mark.parametrize("multimodal", [False, True]) +@pytest.mark.parametrize("allowed", [True, False]) +async def test_explicit_skill_scope_and_history_survive_tool_calls( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, multimodal: bool, allowed: bool +) -> None: + monkeypatch.setenv("BUB_HOME", str(tmp_path / "home")) + skill_dir = tmp_path / ".agents/skills/review" + skill_dir.mkdir(parents=True) + body = "Keep every evidence reference." + (skill_dir / "SKILL.md").write_text(f"---\nname: review\ndescription: Review records.\n---\n{body}\n") + framework = BubFramework(config_file=tmp_path / "config.yml") + framework.workspace = tmp_path + framework.load_builtin_hooks() + requests: list[dict[str, Any]] = [] + calls: list[str] = [] + + def lookup() -> str: + calls.append("lookup") + return "evidence found" + + class Provider: + SUPPORTS_COMPLETION_STREAMING = False + + async def acompletion(self, **kwargs: Any) -> ChatCompletion: + requests.append(kwargs) + message: dict[str, Any] = {"role": "assistant", "content": "reviewed"} + if len(requests) == 1: + message = { + "role": "assistant", + "tool_calls": [ + { + "id": "lookup", + "type": "function", + "function": {"name": "lookup", "arguments": "{}"}, + } + ], + } + return ChatCompletion.model_validate({ + "id": "reply", + "model": "test-model", + "created": 0, + "object": "chat.completion", + "choices": [ + { + "index": 0, + "finish_reason": "tool_calls" if "tool_calls" in message else "stop", + "message": message, + } + ], + }) + + provider = Provider() + monkeypatch.setattr("bub.builtin.model_runner.AnyLLM.create", lambda *args, **kwargs: provider) + agent = Agent(framework, tools=[Tool.from_callable(lookup)], skill_dirs=[skill_dir.parent]) + image = {"type": "image_url", "image_url": {"url": "https://example.test/evidence.png"}} + prompt: str | list[dict] = "$REVIEW Check the records." + if multimodal: + prompt = [{"type": "text", "text": prompt}, image] + for current in (prompt, "Continue the review."): + stream = await agent.run_stream( + session_id="review", + prompt=current, + model="openrouter:test-model", + allowed_skills=["REVIEW"] if allowed else [], + state={"_runtime_workspace": str(tmp_path)}, + ) + async for _ in stream: + pass + + assert calls == ["lookup"] + systems = [request["messages"][0]["content"] for request in requests] + assert all(system == systems[0] and body not in system for system in systems) + for request in requests: + users = [message["content"] for message in request["messages"] if message["role"] == "user"] + assert json.dumps(users).count(body) == int(allowed) + if multimodal: + first_user = next(message for message in requests[0]["messages"] if message["role"] == "user") + assert image in first_user["content"] + assert any( + "evidence found" in message.get("content", "") + for message in requests[-1]["messages"] + if message["role"] == "tool" + ) diff --git a/website/src/content/docs/docs/build/skills.mdx b/website/src/content/docs/docs/build/skills.mdx index b8a00b1f..ed8dfa24 100644 --- a/website/src/content/docs/docs/build/skills.mdx +++ b/website/src/content/docs/docs/build/skills.mdx @@ -9,6 +9,8 @@ This guide shows how to author a **skill** — a discoverable directory centered A skill is a unit of model-facing instruction Bub exposes via the `,skill` comma command and through prompt rendering. Bub's loader implements the [Agent Skills](https://agentskills.io/) format; this page focuses on the contract Bub enforces and the packaging patterns that put skills on the discovery path. +The system prompt lists skill summaries. An explicit `$my-skill` hint appends the rendered body to the current user message, so it stays in tape history after tool calls. `allowed_skills` limits both summaries and explicit expansion; the `,skill` command is unchanged. + ## Before you begin - A plugin or distribution package built with a backend that supports custom file inclusion (Hatch, uv-build, PDM all work). diff --git a/website/src/content/docs/zh-cn/docs/build/skills.mdx b/website/src/content/docs/zh-cn/docs/build/skills.mdx index f69935ca..dbdaf48f 100644 --- a/website/src/content/docs/zh-cn/docs/build/skills.mdx +++ b/website/src/content/docs/zh-cn/docs/build/skills.mdx @@ -9,6 +9,8 @@ sidebar: 技能是 Bub 通过 `,skill` 逗号命令以及提示渲染暴露给模型的指令单元。Bub 的加载器实现了 [Agent Skills](https://agentskills.io/) 格式;本页聚焦 Bub 强制的契约以及把技能放上发现路径的打包模式。 +系统提示词列出技能摘要。显式 `$my-skill` 提示会把渲染后的正文追加到当前用户消息中,因此工具调用后正文仍保留在 tape 历史中。`allowed_skills` 同时限制摘要和显式展开,`,skill` 命令保持不变。 + ## 开始之前 - 一个使用支持自定义文件包含的构建后端的插件或发行版包(Hatch、uv-build、PDM 都可)。 From 65d70cedc4ea05a865c946c2d320612be61dc6d4 Mon Sep 17 00:00:00 2001 From: Chojan Shang Date: Sun, 4 Oct 2026 08:36:48 +0800 Subject: [PATCH 6/8] refactor: separate tool discovery from request preparation --- src/bub/builtin/agent.py | 60 +++++++++++++------ src/bub/builtin/codemode/__init__.py | 24 -------- src/bub/tools.py | 24 +------- tests/test_builtin_agent.py | 6 +- ...ool_catalogs.py => test_tool_providers.py} | 31 ++++++---- website/src/content/docs/docs/build/tools.mdx | 12 ++-- .../content/docs/zh-cn/docs/build/tools.mdx | 13 ++-- 7 files changed, 76 insertions(+), 94 deletions(-) rename tests/{test_tool_catalogs.py => test_tool_providers.py} (74%) diff --git a/src/bub/builtin/agent.py b/src/bub/builtin/agent.py index 998dbe15..a561c22b 100644 --- a/src/bub/builtin/agent.py +++ b/src/bub/builtin/agent.py @@ -7,7 +7,7 @@ import re import shlex import time -from collections.abc import AsyncGenerator, AsyncIterator, Collection, Iterable +from collections.abc import AsyncGenerator, AsyncIterator, Collection, Iterable, Mapping from contextlib import AsyncExitStack, aclosing from dataclasses import dataclass, replace from datetime import UTC, datetime @@ -17,7 +17,6 @@ from loguru import logger -from bub.builtin.codemode import CodeModeCatalog from bub.builtin.commands import strip_command_prefix, validate_command_prefix from bub.builtin.model_runner import ( ModelRunner, @@ -30,7 +29,7 @@ from bub.store import AsyncTapeStore, AsyncTapeStoreAdapter, InMemoryTapeStore, TapeStore, is_async_tape_store from bub.streaming import AsyncStreamEvents, StreamEvent, StreamState from bub.tape import Tape -from bub.tools import REGISTRY, DirectToolCatalog, Tool, ToolCatalog, ToolContext, model_tools +from bub.tools import REGISTRY, Tool, ToolContext, ToolProvider, model_tools from bub.tracing import Span, current_span from bub.turn import TurnState from bub.utils import workspace_from_state @@ -74,26 +73,22 @@ def __init__( ) self.framework = framework self.tools = {tool.name: tool for tool in tools} if tools is not None else REGISTRY.copy() - self.catalogs: list[ToolCatalog] = [DirectToolCatalog(self.tools), CodeModeCatalog()] + self.tool_sources: dict[object, Mapping[str, Tool]] = {} + self.tool_providers: list[ToolProvider] = [] self.tape_store = tape_store self.skill_dirs = skill_dirs self.model_runner = ModelRunner(self.settings, hooks=framework.get_agent_hooks()) @property def known_tools(self) -> dict[str, Tool]: - """Current execution tools plus catalog entries, with later sources taking precedence.""" - tools: dict[str, Tool] = {} - for catalog in self.catalogs: - for name, item in catalog.tools.items(): + """Known runtime tools; later registered sources take precedence.""" + tools = self.tools.copy() + for source in self.tool_sources.values(): + for name, item in source.items(): tools.pop(name, None) tools[name] = item return tools - def add_catalog(self, catalog: ToolCatalog) -> None: - """Register a source once, after existing sources and before code-mode presentation.""" - if all(item is not catalog for item in self.catalogs): - self.catalogs.insert(-1, catalog) - @cached_property def tape(self) -> Tape: """Return the lazily constructed, cached tape factory for this agent. @@ -512,12 +507,15 @@ async def _run_once_stream( tools: list[Tool], ) -> AsyncStreamEvents: tools_prompts: list[str] = [] - for catalog in self.catalogs: - owned = catalog.tools - tools, tools_prompt = await catalog.prepare(tools, tape) - self.tools.update({item.name: item for item in tools if owned.get(item.name) is item}) + scoped = {item.name: item for item in tools} + for provider in self.tool_providers: + tools, tools_prompt = await provider(tools, tape) if tools_prompt: tools_prompts.append(tools_prompt) + self.tools.update({item.name: item for item in tools if scoped.get(item.name) is item}) + tools, tools_prompt = await self._prepare_code_mode(tools, tape) + if tools_prompt: + tools_prompts.append(tools_prompt) system_prompt = self._system_prompt( prompt_text, state=tape.context.state, @@ -552,6 +550,34 @@ async def _run_once_stream( steering_messages=steering_messages, ) + async def _prepare_code_mode(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: + """Split tools for code mode and return model-facing tools and the stub prompt. + + Code mode is a session setting (``state["code_mode"]``, switched by the ``code_mode`` + command) and applies only when ``run_code`` is among the allowed tools: the model then + sees preserved tools directly, and every other tool is callable only from code. + """ + from bub.builtin.codemode import ( + CODE_MODE_STATE_KEY, + CODE_TOOLS_STATE_KEY, + RUN_CODE_TOOL_NAME, + render_code_mode_prompt, + write_tool_stub, + ) + + state = tape.context.state + direct_tools = [tool for tool in tools if tool.name != RUN_CODE_TOOL_NAME] + if not state.get(CODE_MODE_STATE_KEY) or len(direct_tools) == len(tools): + state.pop(CODE_TOOLS_STATE_KEY, None) + return direct_tools, "" + + code_tools = [tool for tool in direct_tools if tool.code_use] + state[CODE_TOOLS_STATE_KEY] = model_tools(code_tools) + stub_path = write_tool_stub( + code_tools, session_id=str(state.get("session_id", "")), workspace=workspace_from_state(state) + ) + return [tool for tool in tools if tool.preserve], render_code_mode_prompt(stub_path) + def _system_prompt( self, prompt: str, diff --git a/src/bub/builtin/codemode/__init__.py b/src/bub/builtin/codemode/__init__.py index 6fae6cf6..fb5439f7 100644 --- a/src/bub/builtin/codemode/__init__.py +++ b/src/bub/builtin/codemode/__init__.py @@ -17,9 +17,7 @@ from bub.environment import CodeFailed from bub.errors import BubError, ErrorKind from bub.hooks.interception import AgentHooks -from bub.tape import Tape from bub.tools import Tool, ToolContext, ToolExecutor, model_tools, tool -from bub.utils import workspace_from_state RUN_CODE_TOOL_NAME = "run_code" CODE_MODE_STATE_KEY = "code_mode" @@ -312,25 +310,3 @@ async def set_code_mode(enable: bool, *, context: ToolContext) -> str: """ await set_session_setting(context, CODE_MODE_STATE_KEY, enable) return f"Session code mode {'enabled' if enable else 'disabled'} (applies from the next turn)." - - -class CodeModeCatalog: - """Adapt the final scoped toolset to code mode; contribute no tool definitions.""" - - @property - def tools(self) -> dict[str, Tool]: - return {} - - async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - state = tape.context.state - direct_tools = [item for item in tools if item.name != RUN_CODE_TOOL_NAME] - if not state.get(CODE_MODE_STATE_KEY) or len(direct_tools) == len(tools): - state.pop(CODE_TOOLS_STATE_KEY, None) - return direct_tools, "" - - code_tools = [item for item in direct_tools if item.code_use] - state[CODE_TOOLS_STATE_KEY] = model_tools(code_tools) - stub_path = write_tool_stub( - code_tools, session_id=str(state.get("session_id", "")), workspace=workspace_from_state(state) - ) - return [item for item in tools if item.preserve], render_code_mode_prompt(stub_path) diff --git a/src/bub/tools.py b/src/bub/tools.py index 8f72ef1c..110107de 100644 --- a/src/bub/tools.py +++ b/src/bub/tools.py @@ -219,28 +219,8 @@ def validated(*args: Any, **kwargs: Any) -> Any: ) -class ToolCatalog(Protocol): - """Declare known tools and prepare the scoped toolset and guidance for a request. - - Tools use runtime names. Returned declared tool instances are registered by - the agent; discovery state and resource cleanup remain owned by the catalog. - Mapping order determines discovery order independently of execution registration. - """ - - @property - def tools(self) -> dict[str, Tool]: ... - - async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: ... - - -@dataclass -class DirectToolCatalog: - """Make a collection of tools directly available within the current scope.""" - - tools: dict[str, Tool] - - async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - return tools, "" +type ToolProvider = Callable[[list[Tool], Tape], Awaitable[tuple[list[Tool], str]]] +"""Prepare scoped tools and a prompt fragment for one model request.""" def model_tools(tools: Iterable[Tool]) -> list[Tool]: diff --git a/tests/test_builtin_agent.py b/tests/test_builtin_agent.py index 54054ab5..836aa86c 100644 --- a/tests/test_builtin_agent.py +++ b/tests/test_builtin_agent.py @@ -12,14 +12,13 @@ import bub.builtin.codemode import bub.builtin.tools # noqa: F401 — registers builtin tools (incl. `model`) from bub.builtin.agent import Agent -from bub.builtin.codemode import CodeModeCatalog from bub.builtin.model_runner import ModelRunner from bub.builtin.settings import AgentSettings from bub.builtin.steering import InMemorySteeringInbox from bub.errors import BubError from bub.streaming import AsyncStreamEvents, StreamEvent, StreamState from bub.tape import TapeContext -from bub.tools import REGISTRY, DirectToolCatalog, tool +from bub.tools import REGISTRY, tool # --------------------------------------------------------------------------- # Agent.run() tests: merge_back logic and model passthrough @@ -56,7 +55,8 @@ async def build_prompt(message: dict[str, Any], session_id: str, state: dict[str agent.command_prefix = agent.settings.command_prefix agent.framework = framework agent.tools = REGISTRY.copy() - agent.catalogs = [DirectToolCatalog(agent.tools), CodeModeCatalog()] + agent.tool_sources = {} + agent.tool_providers = [] agent.tape_store = None agent.skill_dirs = None agent.model_runner = _FakeModelRunner(agent.settings) diff --git a/tests/test_tool_catalogs.py b/tests/test_tool_providers.py similarity index 74% rename from tests/test_tool_catalogs.py rename to tests/test_tool_providers.py index 1ccecae3..8a57e848 100644 --- a/tests/test_tool_catalogs.py +++ b/tests/test_tool_providers.py @@ -11,13 +11,14 @@ from bub.builtin.codemode import run_code from bub.framework import BubFramework from bub.tape import Tape -from bub.tools import DirectToolCatalog, Tool +from bub.tools import Tool @pytest.mark.asyncio @pytest.mark.parametrize("code_mode", [False, True]) -async def test_catalog_selection_registers_scoped_tools_for_native_and_code_calls( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, code_mode: bool +@pytest.mark.parametrize("reverse_providers", [False, True]) +async def test_discovery_precedence_is_independent_of_provider_order_for_native_and_code_calls( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, code_mode: bool, reverse_providers: bool ) -> None: monkeypatch.setenv("BUB_HOME", str(tmp_path)) framework = BubFramework(config_file=tmp_path / "config.yml") @@ -34,9 +35,11 @@ def lookup(name: str) -> str: supplied = Tool.from_callable(lookup, name="catalog.lookup") - class Catalog(DirectToolCatalog): - async def prepare(self, tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - return [item for item in tools if item is not pending], "Use catalog_lookup to greet the requested person." + async def select(tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: + return [item for item in tools if item is not pending], "Use catalog_lookup to greet the requested person." + + async def guide(tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: + return tools, "Keep the greeting brief." requests: list[dict[str, Any]] = [] @@ -75,8 +78,9 @@ async def acompletion(self, **kwargs: Any) -> ChatCompletion: monkeypatch.setattr("bub.builtin.model_runner.AnyLLM.create", lambda *args, **kwargs: Provider()) agent = Agent(framework, tools=[direct, run_code], skill_dirs=[]) - agent.add_catalog(Catalog({supplied.name: supplied, denied.name: denied, pending.name: pending})) - assert supplied.name not in agent.tools + agent.tool_sources["earlier"] = {supplied.name: Tool.from_callable(lambda: "wrong", name=supplied.name)} + agent.tool_sources["later"] = {supplied.name: supplied, denied.name: denied, pending.name: pending} + agent.tool_providers = [guide, select] if reverse_providers else [select, guide] stream = await agent.run_stream( session_id="catalog", prompt="Greet Ada.", @@ -87,17 +91,18 @@ async def acompletion(self, **kwargs: Any) -> ChatCompletion: async for _ in stream: pass assert calls == ["Ada"] - assert supplied.name in agent.tools - assert denied.name not in agent.tools - assert pending.name not in agent.tools definitions = {item["function"]["name"]: item["function"] for item in requests[0]["tools"]} assert definitions.keys() == ({"run_code"} if code_mode else {"direct", "catalog_lookup"}) if code_mode: stub = next((tmp_path / "codemode").rglob("*.pyi")).read_text() assert "catalog_pending" not in stub - assert any( - "Use catalog_lookup" in message["content"] for message in requests[0]["messages"] if message["role"] == "system" + system = "\n".join(message["content"] for message in requests[0]["messages"] if message["role"] == "system") + guidance = ( + ["Keep the greeting brief.", "Use catalog_lookup"] + if reverse_providers + else ["Use catalog_lookup", "Keep the greeting brief."] ) + assert system.index(guidance[0]) < system.index(guidance[1]) assert any( "Hello Ada" in message.get("content", "") for message in requests[1]["messages"] if message["role"] == "tool" ) diff --git a/website/src/content/docs/docs/build/tools.mdx b/website/src/content/docs/docs/build/tools.mdx index 060c93d3..02fa3e00 100644 --- a/website/src/content/docs/docs/build/tools.mdx +++ b/website/src/content/docs/docs/build/tools.mdx @@ -64,15 +64,13 @@ If `run_code` is not allowed (for example, a subagent restricted with `allowed_t `run_code` hands the code to the session [environment](#run-tools-in-an-environment)'s `run_code`, together with a `call_tool` callback. Tool calls always run on the host through that callback, so hooks still see every call. `timeout_seconds` cancels the environment call. The builtin `LocalEnvironment` starts a fresh Python process for every call (`sys.executable`, working directory is the workspace) and forwards tool calls as JSON lines over its stdin and stdout, so arguments and results must be JSON-serializable. When the code finishes, fails or times out, it kills the process and everything it started. That process runs on the host with the same permissions as `bash`, so enable code mode only where `bash` would be acceptable. The return annotation of a tool function (or `Tool.output_schema`, for tools built by hand) determines the result type shown in the stub. -### Tool catalogs +### Tool discovery and request preparation -`Tool` defines one callable capability. `ToolCatalog` declares known tools and prepares the tools and guidance available for each request. Both contracts live in `bub.tools`; `DirectToolCatalog(tools)` provides immediate availability for a collection of tools. +Register directly available tools in `Agent.tools`. For a separate discovery inventory, set `agent.tool_sources[source]` to a mapping of runtime names to `Tool` instances. `source` is a plugin-owned key used to update or remove that inventory. `Agent.known_tools` combines direct tools and source inventories for commands and `allowed_tools`; later registered sources win duplicate names, independently of provider order. Updating an existing source preserves its priority. -Use `agent.add_catalog(source)` to register a source once, after existing sources and before the final code-mode catalog. Its `tools` mapping uses runtime names and existing `Tool` instances; `prepare(tools, tape) -> (tools, tool_prompt)` receives the whole scoped toolset. A source preserves tools it does not own while selecting its own definitions and optional discovery helpers. Presentation catalogs such as code mode can transform the selected toolset. Deferred exposure is a catalog policy, not the default. +Append request-time preparation callables to `agent.tool_providers`. The `ToolProvider` contract in `bub.tools` is `async (tools, tape) -> (tools, prompt_fragment)`. Providers receive the filtered toolset, run in their supplied order, and pass their result to the next provider. They may select definitions and contribute discovery guidance, but must not reintroduce tools excluded by the scope. Builtin preparation runs last, so code mode uses the final selected toolset. -`Agent.known_tools` combines catalog entries for commands and `allowed_tools`. The Agent applies scope, prepares each catalog, and registers its returned declared tool instances in `Agent.tools` before continuing. Catalog implementations do not need to mutate the execution registry. Builtins remain directly available; code mode uses the final selected tools in its full stub. Direct comma commands register the named tool when called. Plugins own discovery state, catalog removal and connection cleanup. Full native definitions are sent through tool schemas instead of repeated in the system prompt. - -Catalog mappings determine discovery order independently of the execution registry. Keep discovery summaries stable as definitions load, and append selected tools in the current conversation's discovery order. Put catalog guidance after stable instructions and skills. This preserves existing content and tool order without model-version checks or provider-specific loading messages. Native additions still grow the tool list; the provider's request format determines how much of the earlier prefix remains cacheable. Code mode adds stub content to history when the model reads it. +The Agent registers selected declared tool instances in `Agent.tools` before builtin preparation. Discovery inventories alone do not expose definitions to the model. Plugins own discovery state, source removal and connection cleanup. Full native definitions are sent through tool schemas instead of repeated in system. Keep discovery summaries stable and append native definitions in the conversation's first-discovery order; request serialization determines how much of the prefix remains cacheable. ## Run tools in an environment @@ -95,7 +93,7 @@ The `REGISTRY` lives in [`bub.tools`](https://github.com/bubbuild/bub/blob/main/ REGISTRY: dict[str, Tool] = {} ``` -Every `@tool` call mutates this dict at **import time**. `Agent(tools=None)` snapshots it when created; later imports do not update existing agents. Add tools to an instance's `Agent.tools`, or use `agent.add_catalog(source)` for a tool collection with its own preparation policy. +Every `@tool` call mutates this dict at **import time**. `Agent(tools=None)` snapshots it when created; later imports do not update existing agents. Add tools to an instance's `Agent.tools`, or register a discovery inventory in `agent.tool_sources` and a preparation callable in `agent.tool_providers`. ## Import the tools module from your plugin diff --git a/website/src/content/docs/zh-cn/docs/build/tools.mdx b/website/src/content/docs/zh-cn/docs/build/tools.mdx index b4a7b847..f4cabf03 100644 --- a/website/src/content/docs/zh-cn/docs/build/tools.mdx +++ b/website/src/content/docs/zh-cn/docs/build/tools.mdx @@ -65,16 +65,13 @@ Code mode 让模型在 Python 中调用工具,而不必一次只发起一个 `run_code` 会把代码和一个 `call_tool` 回调一起交给当前 session [执行环境](#在执行环境中运行工具)的 `run_code` 执行。工具调用始终经这个回调在宿主上执行,所以 hook 仍能看到每一次调用。超过 `timeout_seconds` 时会取消执行环境里的执行。builtin 的 `LocalEnvironment` 每次调用都会启动一个新的 Python 进程(使用 `sys.executable`,工作目录为 workspace),工具调用以 JSON 行的形式经进程的 stdin/stdout 转发,因此参数和结果都必须能 JSON 序列化。代码执行完、出错或超时后,它会杀掉该进程及其启动的所有子进程。这个进程直接跑在宿主上,权限与 `bash` 相同,因此只应在允许使用 `bash` 的场景下开启 code mode。stub 中的结果类型来自工具函数的返回注解(手动构造的工具则来自 `Tool.output_schema`)。 -### 工具目录 +### 工具发现与请求准备 -`Tool` 定义一项可调用能力,`ToolCatalog` 声明已知工具并准备每次请求可用的工具与使用提示。两个契约都位于 `bub.tools`;`DirectToolCatalog(tools)` 让一组工具立即可用。 +直接可用的工具注册在 `Agent.tools`。独立的发现清单通过 `agent.tool_sources[source]` 登记,值为运行时名称到 `Tool` 实例的映射。`source` 是插件管理的键,用于更新或移除自己的清单。`Agent.known_tools` 合并直接工具和各来源清单,供命令查找及 `allowed_tools` 使用;后注册的来源优先处理重名,与 provider 顺序无关。更新已有来源不改变优先级。 -通过 `agent.add_catalog(source)` 注册工具来源。同一个实例只注册一次,顺序位于已有来源之后、末尾的 code-mode catalog 之前。`tools` 使用运行时名称和已有的 `Tool` 实例;`prepare(tools, tape) -> (tools, tool_prompt)` 接收整个经过 scope 过滤的工具集合。工具来源保留其他来源的工具,选择自己的完整定义,并可添加发现工具。Code mode 等呈现目录可以转换最终工具集合。按需暴露由目录自身决定,并非默认行为。 - -`Agent.known_tools` 合并目录条目,供命令查找及 `allowed_tools` 使用。Agent 应用 scope、准备各目录,并在继续处理前将返回的已声明工具实例注册到 `Agent.tools`,目录实现无需修改执行 registry。内置工具直接可用;code mode 的完整 stub 使用最终选中的工具。直接执行逗号命令时才注册点名的工具。发现状态、目录移除与连接清理由插件管理。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。 - -目录映射决定发现顺序,独立于执行 registry。加载完整定义时保持目录摘要稳定,按当前会话的发现顺序追加选中的工具,并把目录提示放在稳定的指令和技能之后。这样无需模型版本判断或 provider 专用加载消息,就能保持已有内容和工具顺序。原生工具列表仍会随发现而增长;此前前缀有多少可以命中缓存,取决于 provider 的请求格式。Code mode 在模型读取 stub 时才将其内容追加到历史。 +向 `agent.tool_providers` 追加请求准备函数。`bub.tools` 中的 `ToolProvider` 契约是 `async (tools, tape) -> (tools, prompt_fragment)`。Provider 接收过滤后的工具集合,按提供的顺序执行,并将结果交给下一项。它们可以选择完整定义并贡献发现提示,但不得重新加入被 scope 排除的工具。Builtin preparation 最后执行,因此 code mode 使用最终选中的工具集合。 +Agent 在 builtin preparation 前把选中的已声明工具实例注册到 `Agent.tools`。发现清单本身不会向模型暴露完整定义。发现状态、来源移除和连接清理由插件管理。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。保持发现摘要稳定,并按当前会话的首次发现顺序追加原生定义;此前前缀有多少可以命中缓存,取决于请求序列化方式。 ## 在执行环境中运行工具 @@ -97,7 +94,7 @@ Code mode 让模型在 Python 中调用工具,而不必一次只发起一个 REGISTRY: dict[str, Tool] = {} ``` -每一次 `@tool` 调用都会在**导入时**修改这个字典。`Agent(tools=None)` 在创建时取得其快照,之后导入的工具不会自动更新已有 Agent。可直接向实例的 `Agent.tools` 添加工具,或通过 `agent.add_catalog(source)` 接入具有独立准备策略的工具集合。 +每一次 `@tool` 调用都会在**导入时**修改这个字典。`Agent(tools=None)` 在创建时取得其快照,之后导入的工具不会自动更新已有 Agent。可直接向实例的 `Agent.tools` 添加工具,或分别通过 `agent.tool_sources` 登记发现清单、通过 `agent.tool_providers` 登记请求准备函数。 ## 在插件里导入工具模块 From 98654082dae0f0d0f44a3a35e79d1abcbf727634 Mon Sep 17 00:00:00 2001 From: Chojan Shang Date: Sun, 4 Oct 2026 09:18:45 +0800 Subject: [PATCH 7/8] refactor: limit core changes to tool discovery and preparation --- src/bub/builtin/agent.py | 28 ++---- src/bub/builtin/codemode/__init__.py | 4 +- src/bub/builtin/hook_impl.py | 1 - src/bub/tools.py | 2 +- tests/test_skill_loading.py | 98 ------------------- tests/test_tool_providers.py | 34 ++++--- .../src/content/docs/docs/build/skills.mdx | 2 - .../content/docs/zh-cn/docs/build/skills.mdx | 2 - 8 files changed, 27 insertions(+), 144 deletions(-) delete mode 100644 tests/test_skill_loading.py diff --git a/src/bub/builtin/agent.py b/src/bub/builtin/agent.py index a561c22b..b7f5cc34 100644 --- a/src/bub/builtin/agent.py +++ b/src/bub/builtin/agent.py @@ -441,7 +441,8 @@ def _load_skills_prompt(self, prompt: str, workspace: Path, allowed_skills: set[ for skill in discover_skills(workspace, skill_dirs=self.skill_dirs) if allowed_skills is None or skill.name.casefold() in allowed_skills } - return render_skills_prompt(list(skill_index.values())) + expanded_skills = set(HINT_RE.findall(prompt)) & set(skill_index.keys()) + return render_skills_prompt(list(skill_index.values()), expanded_skills=expanded_skills) async def _run_once( self, @@ -466,23 +467,6 @@ async def _run_once( if allowed_skills is not None: allowed_skills = {name.casefold() for name in allowed_skills} tape.context.state["allowed_skills"] = list(allowed_skills) - if prompt is not None: - hinted = {name.casefold() for name in HINT_RE.findall(prompt_text)} - selected = [ - skill - for skill in discover_skills(workspace_from_state(tape.context.state), skill_dirs=self.skill_dirs) - if skill.name.casefold() in hinted - and (allowed_skills is None or skill.name.casefold() in allowed_skills) - ] - if selected: - bodies = "\n\n".join( - f'\n{skill.body()}\n' - for skill in selected - ) - if isinstance(prompt, str): - prompt = f"{prompt}\n\n{bodies}" - else: - prompt = [*prompt, {"type": "text", "text": bodies}] if allowed_tools is not None: tools = [tool for tool in known_tools.values() if tool.name in allowed_tools] else: @@ -512,7 +496,7 @@ async def _run_once_stream( tools, tools_prompt = await provider(tools, tape) if tools_prompt: tools_prompts.append(tools_prompt) - self.tools.update({item.name: item for item in tools if scoped.get(item.name) is item}) + self.tools.update({item.name: scoped[item.name] for item in tools if item.name in scoped}) tools, tools_prompt = await self._prepare_code_mode(tools, tape) if tools_prompt: tools_prompts.append(tools_prompt) @@ -587,12 +571,12 @@ def _system_prompt( ) -> str: blocks: list[str] = [] if result := self.framework.get_system_prompt(prompt=prompt, state=state): - blocks.append(f"\n{result}\n") + blocks.append(result) + if tools_prompt: + blocks.append(tools_prompt) workspace = workspace_from_state(state) if skills_prompt := self._load_skills_prompt(prompt, workspace, allowed_skills): blocks.append(skills_prompt) - if tools_prompt: - blocks.append(f"\n{tools_prompt}\n") return "\n\n".join(blocks) def _has_steering_messages(self, state: TurnState) -> bool: diff --git a/src/bub/builtin/codemode/__init__.py b/src/bub/builtin/codemode/__init__.py index fb5439f7..fc6462a1 100644 --- a/src/bub/builtin/codemode/__init__.py +++ b/src/bub/builtin/codemode/__init__.py @@ -243,8 +243,7 @@ def render_code_mode_prompt(stub_path: Path) -> str: "\n" f"More tools are available as async Python functions `tools.(...)` inside `{RUN_CODE_TOOL_NAME}`. " f"Their signatures, result types and documentation are in the stub file: {stub_path}\n" - "Locate and read only the relevant declarations in the stub before calling an unfamiliar tool. " - "Always `await` tool calls (top-level `await` " + "Read the stub before calling a tool you have not used yet. Always `await` tool calls (top-level `await` " "is allowed) and pass keyword arguments; they return structured values and raise on failure. " f"`{RUN_CODE_TOOL_NAME}` returns only what the code prints, so print the results you need, and combine " "several tool calls in one run when possible.\n" @@ -256,7 +255,6 @@ def render_code_mode_prompt(stub_path: Path) -> str: async def run_code(code: str, timeout_seconds: int = DEFAULT_RUN_CODE_TIMEOUT_SECONDS, *, context: ToolContext) -> str: """Run Python code in the environment and return everything it prints. - By default, each call starts a fresh Python process; keep operations that share variables in the same call. Tools are async functions available as `tools.(...)`: await them with keyword arguments (top-level `await` is allowed). See the tool stub file referenced in the system prompt for their signatures and result types. The code is stopped after timeout_seconds. diff --git a/src/bub/builtin/hook_impl.py b/src/bub/builtin/hook_impl.py index f9897a18..70f05ab3 100644 --- a/src/bub/builtin/hook_impl.py +++ b/src/bub/builtin/hook_impl.py @@ -37,7 +37,6 @@ DEFAULT_SYSTEM_PROMPT = """\ Call tools or skills to finish the task. -Issue independent read calls together in the same model step. Before ending this run, you MUST determine whether a response needs to be sent via channel, checking the following conditions: diff --git a/src/bub/tools.py b/src/bub/tools.py index 110107de..c9346417 100644 --- a/src/bub/tools.py +++ b/src/bub/tools.py @@ -220,7 +220,7 @@ def validated(*args: Any, **kwargs: Any) -> Any: type ToolProvider = Callable[[list[Tool], Tape], Awaitable[tuple[list[Tool], str]]] -"""Prepare scoped tools and a prompt fragment for one model request.""" +"""Prepare registered tools and a prompt fragment for one model request.""" def model_tools(tools: Iterable[Tool]) -> list[Tool]: diff --git a/tests/test_skill_loading.py b/tests/test_skill_loading.py deleted file mode 100644 index 74a93675..00000000 --- a/tests/test_skill_loading.py +++ /dev/null @@ -1,98 +0,0 @@ -from __future__ import annotations - -import json -from pathlib import Path -from typing import Any - -import pytest -from any_llm.types.completion import ChatCompletion - -from bub.builtin.agent import Agent -from bub.framework import BubFramework -from bub.tools import Tool - - -@pytest.mark.asyncio -@pytest.mark.parametrize("multimodal", [False, True]) -@pytest.mark.parametrize("allowed", [True, False]) -async def test_explicit_skill_scope_and_history_survive_tool_calls( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, multimodal: bool, allowed: bool -) -> None: - monkeypatch.setenv("BUB_HOME", str(tmp_path / "home")) - skill_dir = tmp_path / ".agents/skills/review" - skill_dir.mkdir(parents=True) - body = "Keep every evidence reference." - (skill_dir / "SKILL.md").write_text(f"---\nname: review\ndescription: Review records.\n---\n{body}\n") - framework = BubFramework(config_file=tmp_path / "config.yml") - framework.workspace = tmp_path - framework.load_builtin_hooks() - requests: list[dict[str, Any]] = [] - calls: list[str] = [] - - def lookup() -> str: - calls.append("lookup") - return "evidence found" - - class Provider: - SUPPORTS_COMPLETION_STREAMING = False - - async def acompletion(self, **kwargs: Any) -> ChatCompletion: - requests.append(kwargs) - message: dict[str, Any] = {"role": "assistant", "content": "reviewed"} - if len(requests) == 1: - message = { - "role": "assistant", - "tool_calls": [ - { - "id": "lookup", - "type": "function", - "function": {"name": "lookup", "arguments": "{}"}, - } - ], - } - return ChatCompletion.model_validate({ - "id": "reply", - "model": "test-model", - "created": 0, - "object": "chat.completion", - "choices": [ - { - "index": 0, - "finish_reason": "tool_calls" if "tool_calls" in message else "stop", - "message": message, - } - ], - }) - - provider = Provider() - monkeypatch.setattr("bub.builtin.model_runner.AnyLLM.create", lambda *args, **kwargs: provider) - agent = Agent(framework, tools=[Tool.from_callable(lookup)], skill_dirs=[skill_dir.parent]) - image = {"type": "image_url", "image_url": {"url": "https://example.test/evidence.png"}} - prompt: str | list[dict] = "$REVIEW Check the records." - if multimodal: - prompt = [{"type": "text", "text": prompt}, image] - for current in (prompt, "Continue the review."): - stream = await agent.run_stream( - session_id="review", - prompt=current, - model="openrouter:test-model", - allowed_skills=["REVIEW"] if allowed else [], - state={"_runtime_workspace": str(tmp_path)}, - ) - async for _ in stream: - pass - - assert calls == ["lookup"] - systems = [request["messages"][0]["content"] for request in requests] - assert all(system == systems[0] and body not in system for system in systems) - for request in requests: - users = [message["content"] for message in request["messages"] if message["role"] == "user"] - assert json.dumps(users).count(body) == int(allowed) - if multimodal: - first_user = next(message for message in requests[0]["messages"] if message["role"] == "user") - assert image in first_user["content"] - assert any( - "evidence found" in message.get("content", "") - for message in requests[-1]["messages"] - if message["role"] == "tool" - ) diff --git a/tests/test_tool_providers.py b/tests/test_tool_providers.py index 8a57e848..a37c4fef 100644 --- a/tests/test_tool_providers.py +++ b/tests/test_tool_providers.py @@ -1,6 +1,7 @@ from __future__ import annotations import json +from dataclasses import replace from pathlib import Path from typing import Any @@ -26,20 +27,22 @@ async def test_discovery_precedence_is_independent_of_provider_order_for_native_ framework.load_builtin_hooks() direct = Tool.from_callable(lambda: "direct", name="direct") denied = Tool.from_callable(lambda: "denied", name="denied") - pending = Tool.from_callable(lambda: "pending", name="catalog.pending") + pending = Tool.from_callable(lambda: "pending", name="provider.pending") calls: list[str] = [] def lookup(name: str) -> str: calls.append(name) return f"Hello {name}" - supplied = Tool.from_callable(lookup, name="catalog.lookup") + supplied = Tool.from_callable(lookup, name="provider.lookup") async def select(tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - return [item for item in tools if item is not pending], "Use catalog_lookup to greet the requested person." + return [item for item in tools if item is not pending], "Use provider_lookup to greet the requested person." async def guide(tools: list[Tool], tape: Tape) -> tuple[list[Tool], str]: - return tools, "Keep the greeting brief." + return [ + replace(item, renderer=str.upper) if item.name == supplied.name else item for item in tools + ], "Keep the greeting brief." requests: list[dict[str, Any]] = [] @@ -50,8 +53,8 @@ async def acompletion(self, **kwargs: Any) -> ChatCompletion: requests.append(kwargs) message: dict[str, Any] = {"role": "assistant", "content": "Hello Ada"} if len(requests) == 1: - name = "run_code" if code_mode else "catalog_lookup" - arguments = {"code": "print(await tools.catalog_lookup(name='Ada'))"} if code_mode else {"name": "Ada"} + name = "run_code" if code_mode else "provider_lookup" + arguments = {"code": "print(await tools.provider_lookup(name='Ada'))"} if code_mode else {"name": "Ada"} message = { "role": "assistant", "tool_calls": [ @@ -82,27 +85,28 @@ async def acompletion(self, **kwargs: Any) -> ChatCompletion: agent.tool_sources["later"] = {supplied.name: supplied, denied.name: denied, pending.name: pending} agent.tool_providers = [guide, select] if reverse_providers else [select, guide] stream = await agent.run_stream( - session_id="catalog", + session_id="provider", prompt="Greet Ada.", model="openrouter:test-model", - allowed_tools=["direct", "catalog_lookup", "catalog_pending", "run_code"], + allowed_tools=["direct", "provider_lookup", "provider_pending", "run_code"], state={"code_mode": code_mode}, ) - async for _ in stream: - pass + events = [event async for event in stream] + assert any(event.data.get("text") == "Hello Ada" for event in events if event.kind == "final") assert calls == ["Ada"] definitions = {item["function"]["name"]: item["function"] for item in requests[0]["tools"]} - assert definitions.keys() == ({"run_code"} if code_mode else {"direct", "catalog_lookup"}) + assert definitions.keys() == ({"run_code"} if code_mode else {"direct", "provider_lookup"}) if code_mode: stub = next((tmp_path / "codemode").rglob("*.pyi")).read_text() - assert "catalog_pending" not in stub + assert "provider_pending" not in stub system = "\n".join(message["content"] for message in requests[0]["messages"] if message["role"] == "system") guidance = ( - ["Keep the greeting brief.", "Use catalog_lookup"] + ["Keep the greeting brief.", "Use provider_lookup"] if reverse_providers - else ["Use catalog_lookup", "Keep the greeting brief."] + else ["Use provider_lookup", "Keep the greeting brief."] ) assert system.index(guidance[0]) < system.index(guidance[1]) + expected = "Hello Ada" if code_mode else "HELLO ADA" assert any( - "Hello Ada" in message.get("content", "") for message in requests[1]["messages"] if message["role"] == "tool" + expected in message.get("content", "") for message in requests[1]["messages"] if message["role"] == "tool" ) diff --git a/website/src/content/docs/docs/build/skills.mdx b/website/src/content/docs/docs/build/skills.mdx index ed8dfa24..b8a00b1f 100644 --- a/website/src/content/docs/docs/build/skills.mdx +++ b/website/src/content/docs/docs/build/skills.mdx @@ -9,8 +9,6 @@ This guide shows how to author a **skill** — a discoverable directory centered A skill is a unit of model-facing instruction Bub exposes via the `,skill` comma command and through prompt rendering. Bub's loader implements the [Agent Skills](https://agentskills.io/) format; this page focuses on the contract Bub enforces and the packaging patterns that put skills on the discovery path. -The system prompt lists skill summaries. An explicit `$my-skill` hint appends the rendered body to the current user message, so it stays in tape history after tool calls. `allowed_skills` limits both summaries and explicit expansion; the `,skill` command is unchanged. - ## Before you begin - A plugin or distribution package built with a backend that supports custom file inclusion (Hatch, uv-build, PDM all work). diff --git a/website/src/content/docs/zh-cn/docs/build/skills.mdx b/website/src/content/docs/zh-cn/docs/build/skills.mdx index dbdaf48f..f69935ca 100644 --- a/website/src/content/docs/zh-cn/docs/build/skills.mdx +++ b/website/src/content/docs/zh-cn/docs/build/skills.mdx @@ -9,8 +9,6 @@ sidebar: 技能是 Bub 通过 `,skill` 逗号命令以及提示渲染暴露给模型的指令单元。Bub 的加载器实现了 [Agent Skills](https://agentskills.io/) 格式;本页聚焦 Bub 强制的契约以及把技能放上发现路径的打包模式。 -系统提示词列出技能摘要。显式 `$my-skill` 提示会把渲染后的正文追加到当前用户消息中,因此工具调用后正文仍保留在 tape 历史中。`allowed_skills` 同时限制摘要和显式展开,`,skill` 命令保持不变。 - ## 开始之前 - 一个使用支持自定义文件包含的构建后端的插件或发行版包(Hatch、uv-build、PDM 都可)。 From 52fb7d05725f3b031c8ee43f7db7390e4df3d81c Mon Sep 17 00:00:00 2001 From: Frost Ming Date: Wed, 7 Oct 2026 16:23:13 +0800 Subject: [PATCH 8/8] fix: scope unknown-tool recovery to the request toolset Record the model aliases of the tools left after providers in the turn state and let the unknown-tool interceptor check them, instead of registering selected tools on the shared Agent.tools. Commands resolve through known_tools without mutating Agent.tools either. This keeps Agent.tools stable across sessions: tools loaded in one session no longer pass the interceptor in another, and code mode no longer registers every discovered tool permanently. Co-Authored-By: Claude Opus 5.5 (1M context) --- src/bub/builtin/agent.py | 15 +++++++-------- src/bub/builtin/hook_impl.py | 10 ++++++---- tests/test_builtin_hook_impl.py | 19 +++++++++++++++++++ tests/test_tool_providers.py | 2 ++ website/src/content/docs/docs/build/tools.mdx | 2 +- .../content/docs/zh-cn/docs/build/tools.mdx | 2 +- 6 files changed, 36 insertions(+), 14 deletions(-) diff --git a/src/bub/builtin/agent.py b/src/bub/builtin/agent.py index b7f5cc34..2249f93a 100644 --- a/src/bub/builtin/agent.py +++ b/src/bub/builtin/agent.py @@ -36,6 +36,8 @@ HINT_RE = re.compile(r"\$([A-Za-z0-9_.-]+)") MAX_AUTO_HANDOFF_RETRIES = 1 +# Model aliases of the tools prepared for the current request, used to recover unknown tool calls. +REQUEST_TOOLS_STATE_KEY = "_runtime_request_tools" class Agent: @@ -259,15 +261,13 @@ async def _run_command(self, tape: Tape, *, line: str) -> str: status = "ok" try: known_tools = self.known_tools - if name in known_tools: - self.tools[name] = known_tools[name] - if name not in self.tools: - if "bash" not in self.tools: + if name not in known_tools: + if "bash" not in known_tools: raise ValueError("bash tool is not available") # noqa: TRY301 - bash_tool = self.tools["bash"] + bash_tool = known_tools["bash"] output = bash_tool.render(await bash_tool.run(context=context, command=line)) else: - command_tool = self.tools[name] + command_tool = known_tools[name] args = _parse_args(arg_tokens) if command_tool.context: args.kwargs["context"] = context @@ -491,12 +491,11 @@ async def _run_once_stream( tools: list[Tool], ) -> AsyncStreamEvents: tools_prompts: list[str] = [] - scoped = {item.name: item for item in tools} for provider in self.tool_providers: tools, tools_prompt = await provider(tools, tape) if tools_prompt: tools_prompts.append(tools_prompt) - self.tools.update({item.name: scoped[item.name] for item in tools if item.name in scoped}) + tape.context.state[REQUEST_TOOLS_STATE_KEY] = [item.name for item in model_tools(tools)] tools, tools_prompt = await self._prepare_code_mode(tools, tape) if tools_prompt: tools_prompts.append(tools_prompt) diff --git a/src/bub/builtin/hook_impl.py b/src/bub/builtin/hook_impl.py index 70f05ab3..2ec79708 100644 --- a/src/bub/builtin/hook_impl.py +++ b/src/bub/builtin/hook_impl.py @@ -9,7 +9,7 @@ from loguru import logger from bub import inquirer as bub_inquirer -from bub.builtin.agent import Agent +from bub.builtin.agent import REQUEST_TOOLS_STATE_KEY, Agent from bub.builtin.commands import strip_command_prefix from bub.builtin.context import default_tape_context from bub.builtin.onboarding import collect_model_config @@ -358,9 +358,11 @@ async def before_tool_call( """ from bub.tools import model_tools - agent = self._get_agent(state) - - available_tools = tuple(tool_item.name for tool_item in model_tools(agent.tools.values())) + if (request_tools := state.get(REQUEST_TOOLS_STATE_KEY)) is not None: + available_tools = tuple(request_tools) + else: + agent = self._get_agent(state) + available_tools = tuple(tool_item.name for tool_item in model_tools(agent.known_tools.values())) if call.tool in available_tools: return None diff --git a/tests/test_builtin_hook_impl.py b/tests/test_builtin_hook_impl.py index 7747a3f1..031130ef 100644 --- a/tests/test_builtin_hook_impl.py +++ b/tests/test_builtin_hook_impl.py @@ -41,6 +41,7 @@ def __init__(self, home: Path, *, tape: Tape | None = None) -> None: self.command_prefix = "," self.settings = SimpleNamespace(home=home) self.tools = REGISTRY.copy() + self.known_tools = self.tools # A real in-memory async tape so load_state's recovery path runs against # the same store the tests write `model_switch` events to. self.tape = tape if tape is not None else _fake_tape(home) @@ -493,3 +494,21 @@ async def _do(): assert decision is not None assert "fs_reed" in decision.result assert "fs_read" in decision.result + + +def test_before_tool_call_recovers_tool_outside_current_request(tmp_path: Path) -> None: + _, impl, _ = _build_impl(tmp_path) + import asyncio + + from bub.builtin.agent import REQUEST_TOOLS_STATE_KEY + from bub.hooks.interception import ToolCall + + state = {REQUEST_TOOLS_STATE_KEY: ["bash"]} + + async def _do(name: str): + return await impl.before_tool_call(ToolCall(run_id="r", tool=name, arguments={}), state=state) + + assert asyncio.run(_do("bash")) is None + decision = asyncio.run(_do("bash_output")) + assert decision is not None and decision.action == "replace" + assert "bash_output" in decision.result diff --git a/tests/test_tool_providers.py b/tests/test_tool_providers.py index a37c4fef..a5e5ea4b 100644 --- a/tests/test_tool_providers.py +++ b/tests/test_tool_providers.py @@ -94,6 +94,8 @@ async def acompletion(self, **kwargs: Any) -> ChatCompletion: events = [event async for event in stream] assert any(event.data.get("text") == "Hello Ada" for event in events if event.kind == "final") assert calls == ["Ada"] + # Request preparation does not register discovered tools on the agent. + assert agent.tools.keys() == {"direct", "run_code"} definitions = {item["function"]["name"]: item["function"] for item in requests[0]["tools"]} assert definitions.keys() == ({"run_code"} if code_mode else {"direct", "provider_lookup"}) if code_mode: diff --git a/website/src/content/docs/docs/build/tools.mdx b/website/src/content/docs/docs/build/tools.mdx index 02fa3e00..1b73791c 100644 --- a/website/src/content/docs/docs/build/tools.mdx +++ b/website/src/content/docs/docs/build/tools.mdx @@ -70,7 +70,7 @@ Register directly available tools in `Agent.tools`. For a separate discovery inv Append request-time preparation callables to `agent.tool_providers`. The `ToolProvider` contract in `bub.tools` is `async (tools, tape) -> (tools, prompt_fragment)`. Providers receive the filtered toolset, run in their supplied order, and pass their result to the next provider. They may select definitions and contribute discovery guidance, but must not reintroduce tools excluded by the scope. Builtin preparation runs last, so code mode uses the final selected toolset. -The Agent registers selected declared tool instances in `Agent.tools` before builtin preparation. Discovery inventories alone do not expose definitions to the model. Plugins own discovery state, source removal and connection cleanup. Full native definitions are sent through tool schemas instead of repeated in system. Keep discovery summaries stable and append native definitions in the conversation's first-discovery order; request serialization determines how much of the prefix remains cacheable. +Request preparation does not change `Agent.tools`. Known tools that pass the scope reach the providers, so a provider that defers discovered tools must remove them from the request itself. The tools left after the providers form the request toolset: only those calls run, and calls to other names receive a guidance result. Plugins own discovery state, source removal and connection cleanup. Full native definitions are sent through tool schemas instead of repeated in system. Keep discovery summaries stable and append native definitions in the conversation's first-discovery order; request serialization determines how much of the prefix remains cacheable. ## Run tools in an environment diff --git a/website/src/content/docs/zh-cn/docs/build/tools.mdx b/website/src/content/docs/zh-cn/docs/build/tools.mdx index f4cabf03..6475f87f 100644 --- a/website/src/content/docs/zh-cn/docs/build/tools.mdx +++ b/website/src/content/docs/zh-cn/docs/build/tools.mdx @@ -71,7 +71,7 @@ Code mode 让模型在 Python 中调用工具,而不必一次只发起一个 向 `agent.tool_providers` 追加请求准备函数。`bub.tools` 中的 `ToolProvider` 契约是 `async (tools, tape) -> (tools, prompt_fragment)`。Provider 接收过滤后的工具集合,按提供的顺序执行,并将结果交给下一项。它们可以选择完整定义并贡献发现提示,但不得重新加入被 scope 排除的工具。Builtin preparation 最后执行,因此 code mode 使用最终选中的工具集合。 -Agent 在 builtin preparation 前把选中的已声明工具实例注册到 `Agent.tools`。发现清单本身不会向模型暴露完整定义。发现状态、来源移除和连接清理由插件管理。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。保持发现摘要稳定,并按当前会话的首次发现顺序追加原生定义;此前前缀有多少可以命中缓存,取决于请求序列化方式。 +请求准备不会修改 `Agent.tools`。通过 scope 的已知工具都会交给 provider,因此需要延迟暴露的发现工具必须由 provider 自己从本次请求中移除。provider 处理后剩下的工具构成本次请求的工具集合:只有这些调用会执行,调用其他名称会收到引导结果。发现状态、来源移除和连接清理由插件管理。完整原生定义通过工具 schema 发送,系统提示词不再重复列出。保持发现摘要稳定,并按当前会话的首次发现顺序追加原生定义;此前前缀有多少可以命中缓存,取决于请求序列化方式。 ## 在执行环境中运行工具