diff --git a/src/bub/builtin/agent.py b/src/bub/builtin/agent.py index 43047247..864b69e0 100644 --- a/src/bub/builtin/agent.py +++ b/src/bub/builtin/agent.py @@ -481,7 +481,7 @@ async def _prepare_deferred_tools(self, tools: list[Tool], tape: Tape) -> tuple[ ) direct = [tool for tool in tools if not tool.deferred and tool.name != TOOL_DESCRIBE_TOOL_NAME] - deferred = {tool.name: tool for tool in tools if tool.deferred and tool.agent_use} + deferred = {tool.name: tool for tool in tools if tool.deferred and tool.exposure in ("auto", "direct")} if not deferred: return direct, "" direct.append(self.tools.get(TOOL_DESCRIBE_TOOL_NAME, tool_describe)) @@ -540,7 +540,8 @@ async def _prepare_code_mode(self, tools: list[Tool], tape: Tape) -> tuple[list[ Code mode is a session setting (``state["code_mode"]``, switched by the ``code_mode`` command) and applies only when ``run_code`` is among the allowed tools: the model then - sees preserved tools directly, and every other tool is callable only from code. + sees ``direct`` tools, and ``auto`` and ``code`` tools are callable only from code. Outside + code mode, ``code`` tools are dropped. """ from bub.builtin.codemode import ( CODE_MODE_STATE_KEY, @@ -554,14 +555,14 @@ async def _prepare_code_mode(self, tools: list[Tool], tape: Tape) -> tuple[list[ direct_tools = [tool for tool in tools if tool.name != RUN_CODE_TOOL_NAME] if not state.get(CODE_MODE_STATE_KEY) or len(direct_tools) == len(tools): state.pop(CODE_TOOLS_STATE_KEY, None) - return direct_tools, "" + return [tool for tool in direct_tools if tool.exposure != "code"], "" - code_tools = [tool for tool in direct_tools if tool.code_use] + code_tools = [tool for tool in direct_tools if tool.exposure in ("auto", "code")] state[CODE_TOOLS_STATE_KEY] = model_tools(code_tools) stub_path = write_tool_stub( code_tools, session_id=str(state.get("session_id", "")), workspace=workspace_from_state(state) ) - return [tool for tool in tools if tool.preserve], render_code_mode_prompt(stub_path) + return [tool for tool in tools if tool.exposure == "direct"], render_code_mode_prompt(stub_path) def _system_prompt( self, diff --git a/src/bub/builtin/codemode/__init__.py b/src/bub/builtin/codemode/__init__.py index fc6462a1..d9038796 100644 --- a/src/bub/builtin/codemode/__init__.py +++ b/src/bub/builtin/codemode/__init__.py @@ -251,7 +251,7 @@ def render_code_mode_prompt(stub_path: Path) -> str: ) -@tool(name=RUN_CODE_TOOL_NAME, context=True, preserve=True) +@tool(name=RUN_CODE_TOOL_NAME, context=True, exposure="direct") async def run_code(code: str, timeout_seconds: int = DEFAULT_RUN_CODE_TIMEOUT_SECONDS, *, context: ToolContext) -> str: """Run Python code in the environment and return everything it prints. @@ -299,11 +299,11 @@ async def call_tool(name: str, arguments: dict[str, Any]) -> Any: return "".join(output) -@tool(name="code_mode", context=True, agent_use=False) +@tool(name="code_mode", context=True, exposure="command") async def set_code_mode(enable: bool, *, context: ToolContext) -> str: """Enable or disable code mode for THIS session. Invoke as the `,code_mode enable=true` command. - In code mode the model calls preserved tools directly and every other tool from + In code mode the model calls `direct` tools directly and every other tool from Python through `run_code`. Takes effect on the NEXT turn and persists across restarts. """ await set_session_setting(context, CODE_MODE_STATE_KEY, enable) diff --git a/src/bub/builtin/hook_impl.py b/src/bub/builtin/hook_impl.py index 6592fb5d..89c2b1f7 100644 --- a/src/bub/builtin/hook_impl.py +++ b/src/bub/builtin/hook_impl.py @@ -367,7 +367,7 @@ async def before_tool_call( code_tools = state.get(CODE_TOOLS_STATE_KEY) or () available_tools = (*state["_runtime_tool_names"], *(tool_item.name for tool_item in code_tools)) else: - available_tools = tuple(tool_item.name for tool_item in agent_tools) + available_tools = tuple(tool_item.name for tool_item in agent_tools if tool_item.exposure != "code") if call.tool in available_tools: return None diff --git a/src/bub/builtin/spill.py b/src/bub/builtin/spill.py index 91fb538f..c024e243 100644 --- a/src/bub/builtin/spill.py +++ b/src/bub/builtin/spill.py @@ -262,7 +262,7 @@ async def read( return SpillPage(manifest, "".join(chunks), start, stop, next_cursor, complete) -@tool(context=True, name=SPILL_READ_TOOL_NAME, preserve=True) +@tool(context=True, name=SPILL_READ_TOOL_NAME, exposure="direct") async def spill_read( handle: str, cursor: int = 0, diff --git a/src/bub/builtin/tools.py b/src/bub/builtin/tools.py index 31b5a062..b3391cbb 100644 --- a/src/bub/builtin/tools.py +++ b/src/bub/builtin/tools.py @@ -96,7 +96,7 @@ def _tool_signature(tool_item: Tool) -> str: def render_tools_prompt(tools: Iterable[Tool]) -> str: """Render a human-readable description of tools for builtin agent prompts.""" - agent_tools = [tool_item for tool_item in tools if tool_item.agent_use] + agent_tools = [tool_item for tool_item in tools if tool_item.exposure in ("auto", "direct")] if not agent_tools: return "" lines = [] @@ -279,7 +279,7 @@ def _render_subagent(result: SubAgentResult) -> str: return result["output"] + "".join(f"[Error: {message}]" for message in result["errors"]) -@tool(context=True, preserve=True) +@tool(context=True, exposure="direct") async def bash( command: str, cwd: str | None = None, @@ -317,7 +317,7 @@ async def bash( return shell.output.strip() or "(no output)" -@tool(name="bash.output", preserve=True) +@tool(name="bash.output", exposure="direct") async def bash_output(shell_id: str, offset: int = 0, limit: int | None = None) -> str: """Read buffered output from a background shell, with optional offset/limit for incremental polling.""" shell = shell_manager.get(shell_id) @@ -332,14 +332,14 @@ async def bash_output(shell_id: str, offset: int = 0, limit: int | None = None) return f"id: {shell.shell_id}\nstatus: {shell.status}\nexit_code: {exit_code}\nnext_offset: {end}\noutput:\n{body}" -@tool(name="bash.kill", preserve=True) +@tool(name="bash.kill", exposure="direct") async def kill_bash(shell_id: str) -> str: """Terminate a background shell process.""" shell = await shell_manager.terminate(shell_id) return f"id: {shell.shell_id}\nstatus: {shell.status}\nexit_code: {shell.returncode}" -@tool(context=True, name="fs.read", preserve=True) +@tool(context=True, name="fs.read", exposure="direct") async def fs_read(path: str, offset: int = 0, limit: int | None = None, *, context: ToolContext) -> str: """Read a text file and return its content. Supports optional pagination with offset and limit.""" environment = environment_from_state(context.state) @@ -350,7 +350,7 @@ async def fs_read(path: str, offset: int = 0, limit: int | None = None, *, conte return "\n".join(lines[start:end]) -@tool(context=True, name="fs.write", preserve=True) +@tool(context=True, name="fs.write", exposure="direct") async def fs_write(path: str, content: str, *, context: ToolContext) -> str: """Write content to a text file.""" environment = environment_from_state(context.state) @@ -359,7 +359,7 @@ async def fs_write(path: str, content: str, *, context: ToolContext) -> str: return f"wrote: {resolved_path}" -@tool(context=True, name="fs.edit", preserve=True) +@tool(context=True, name="fs.edit", exposure="direct") async def fs_edit(path: str, old: str, new: str, start: int = 0, *, context: ToolContext) -> str: """Edit a text file by replacing old text with new text. You can specify the line number to start searching for the old text.""" environment = environment_from_state(context.state) @@ -398,7 +398,7 @@ def skill_describe(name: str | None = None, *, context: ToolContext) -> SkillLis return {"name": skill.name, "location": str(skill.location), "content": skill.body() or ""} -@tool(context=True, name=TOOL_DESCRIBE_TOOL_NAME, preserve=True) +@tool(context=True, name=TOOL_DESCRIBE_TOOL_NAME, exposure="direct") async def tool_describe(names: list[str], *, context: ToolContext) -> ToolDescriptions: """Load tools by name and return their definitions. Deferred tools become callable from the next step.""" agent = _get_agent(context) @@ -406,7 +406,7 @@ async def tool_describe(names: list[str], *, context: ToolContext) -> ToolDescri available = { name: tool_item for name, tool_item in agent.tools.items() - if tool_item.agent_use and (allowed_tools is None or name in allowed_tools) + if tool_item.exposure in ("auto", "direct") and (allowed_tools is None or name in allowed_tools) } index = _tool_name_index(available) loaded = set(await loaded_tool_names(context.tape)) @@ -528,7 +528,7 @@ async def run_subagent(param: SubAgentInput, *, context: ToolContext) -> SubAgen return {"session_id": subagent_session, "output": output, "errors": errors} -@tool(name="help", context=True, agent_use=False) +@tool(name="help", context=True, exposure="command") def show_help(*, context: ToolContext | None = None) -> str: """Show a help message.""" agent = context.state.get("_runtime_agent") if context is not None else None @@ -554,7 +554,7 @@ def show_help(*, context: ToolContext | None = None) -> str: ) -@tool(name="quit", context=True, agent_use=False) +@tool(name="quit", context=True, exposure="command") async def quit_tool(*, context: ToolContext) -> str: """Abort the tasks of the current session. DO NOT use it in a normal workflow.""" agent = _get_agent(context) @@ -564,7 +564,7 @@ async def quit_tool(*, context: ToolContext) -> str: return "Session tasks stopped." -@tool(name="model", context=True, agent_use=False) +@tool(name="model", context=True, exposure="command") async def set_model(model_id: str, *, context: ToolContext) -> str: """Switch the model for THIS session. Invoke as the `,model ` command. @@ -577,7 +577,7 @@ async def set_model(model_id: str, *, context: ToolContext) -> str: return f"Session model set to {model_id} (applies from the next turn)." -@tool(name="reasoning_effort", context=True, agent_use=False) +@tool(name="reasoning_effort", context=True, exposure="command") async def set_reasoning_effort(reasoning_effort: str, *, context: ToolContext) -> str: """Set the reasoning effort for this session starting from the next turn.""" reasoning_effort = reasoning_effort.strip() diff --git a/src/bub/tools.py b/src/bub/tools.py index 1b873fb3..0d60a399 100644 --- a/src/bub/tools.py +++ b/src/bub/tools.py @@ -8,7 +8,7 @@ import time from collections.abc import Awaitable, Callable, Iterable, Sequence from dataclasses import dataclass, field, replace -from typing import TYPE_CHECKING, Any, Protocol, get_type_hints, overload +from typing import TYPE_CHECKING, Any, Literal, Protocol, get_args, get_type_hints, overload from loguru import logger from pydantic import BaseModel, ConfigDict, TypeAdapter, ValidationError, validate_call @@ -134,6 +134,16 @@ def validate_target(*args: Any, **kwargs: Any) -> tuple[tuple[Any, ...], dict[st return validate_call(validate_target) +ToolExposure = Literal["auto", "direct", "code", "command"] +"""Where a tool can be called from. + +- ``auto``: by the model directly, or only from ``run_code`` in code mode. +- ``direct``: by the model directly, also in code mode; never from ``run_code``. +- ``code``: only from ``run_code`` in code mode; never by the model directly. +- ``command``: only as a comma command; never by the model or from code. +""" + + @dataclass(frozen=True) class Tool: """A callable unit the model can invoke.""" @@ -143,22 +153,17 @@ class Tool: description: str = "" parameters: dict[str, Any] = field(default_factory=dict) context: bool = False - agent_use: bool = True + exposure: ToolExposure = "auto" + """Where the tool can be called from, see ``ToolExposure``.""" renderer: Callable[[Any], str] | None = None - preserve: bool = False - """Keep the tool directly callable by the model in code mode; others are reachable only from code.""" output_schema: dict[str, Any] | None = None """JSON schema of the structured result, used to describe the tool to model-written code.""" deferred: bool = False """Load the tool on demand: only its name is listed until ``tool.describe`` loads its definition.""" - @property - def code_use(self) -> bool: - """Whether the tool is callable from model-written code (``tools.*`` in ``run_code``). - - Preserved tools stay model-facing only, and tools hidden from the agent are never exposed to code. - """ - return self.agent_use and not self.preserve + def __post_init__(self) -> None: + if self.exposure not in get_args(ToolExposure): + raise ValueError(f"Tool '{self.name}' has unknown exposure {self.exposure!r}.") def run(self, *args: Any, **kwargs: Any) -> Any: return self.handler(*args, **kwargs) @@ -188,9 +193,8 @@ def from_callable( name: str | None = None, description: str | None = None, context: bool = False, - agent_use: bool = True, + exposure: ToolExposure = "auto", renderer: Callable[[Any], str] | None = None, - preserve: bool = False, deferred: bool = False, ) -> Tool: signature = inspect.signature(func) @@ -215,17 +219,20 @@ def validated(*args: Any, **kwargs: Any) -> Any: parameters=parameters, handler=validated, context=context, - agent_use=agent_use, + exposure=exposure, renderer=renderer, - preserve=preserve, output_schema=_output_schema(func), deferred=deferred, ) def model_tools(tools: Iterable[Tool]) -> list[Tool]: - """Convert agent-enabled runtime tools into model-safe aliases.""" - return [replace(tool_item, name=tool_item.name.replace(".", "_")) for tool_item in tools if tool_item.agent_use] + """Convert tools callable by the model or code into model-safe aliases; comma commands are dropped.""" + return [ + replace(tool_item, name=tool_item.name.replace(".", "_")) + for tool_item in tools + if tool_item.exposure != "command" + ] @dataclass(frozen=True) @@ -559,9 +566,8 @@ def tool( model: type[BaseModel] | None = ..., description: str | None = ..., context: bool = ..., - agent_use: bool = ..., + exposure: ToolExposure = ..., renderer: Callable[[Any], str] | None = ..., - preserve: bool = ..., deferred: bool = ..., ) -> Tool: ... @@ -574,9 +580,8 @@ def tool( model: type[BaseModel] | None = ..., description: str | None = ..., context: bool = ..., - agent_use: bool = ..., + exposure: ToolExposure = ..., renderer: Callable[[Any], str] | None = ..., - preserve: bool = ..., deferred: bool = ..., ) -> Callable[[Callable], Tool]: ... @@ -588,15 +593,15 @@ def tool( model: type[BaseModel] | None = None, description: str | None = None, context: bool = False, - agent_use: bool = True, + exposure: ToolExposure = "auto", renderer: Callable[[Any], str] | None = None, - preserve: bool = False, deferred: bool = False, ) -> Tool | Callable[[Callable], Tool]: """Decorator to convert a function into a Tool instance. Tools should return structured results; ``renderer`` turns such a result into the plain text shown to the model outside code mode (defaults to JSON for non-strings). + ``exposure`` decides where the tool can be called from (see ``ToolExposure``), and ``deferred`` tools are loaded on demand through ``tool.describe``. """ @@ -618,9 +623,8 @@ def handler(*args: Any, **kwargs: Any) -> Any: parameters=model.model_json_schema(), handler=handler, context=context, - agent_use=agent_use, + exposure=exposure, renderer=renderer, - preserve=preserve, output_schema=_output_schema(func), deferred=deferred, ) @@ -630,9 +634,8 @@ def handler(*args: Any, **kwargs: Any) -> Any: name=name, description=description, context=context, - agent_use=agent_use, + exposure=exposure, renderer=renderer, - preserve=preserve, deferred=deferred, ) tool_instance = _add_logging(result) diff --git a/tests/test_builtin_agent.py b/tests/test_builtin_agent.py index f36b2eb1..519c1c66 100644 --- a/tests/test_builtin_agent.py +++ b/tests/test_builtin_agent.py @@ -411,7 +411,7 @@ def denied_agent_tool() -> str: @pytest.mark.asyncio -async def test_agent_run_excludes_tools_disabled_for_agent_use() -> None: +async def test_agent_run_excludes_command_tools() -> None: visible_name = "tests.visible_agent_tool" internal_name = "tests.internal_agent_tool" REGISTRY.pop(visible_name, None) @@ -421,7 +421,7 @@ async def test_agent_run_excludes_tools_disabled_for_agent_use() -> None: def visible_agent_tool() -> str: return "visible" - @tool(name=internal_name, description="Internal tool", agent_use=False) + @tool(name=internal_name, description="Internal tool", exposure="command") def internal_agent_tool() -> str: return "internal" diff --git a/tests/test_builtin_tools.py b/tests/test_builtin_tools.py index c321d7b7..343775d2 100644 --- a/tests/test_builtin_tools.py +++ b/tests/test_builtin_tools.py @@ -103,8 +103,8 @@ def test_render_tools_prompt_returns_empty_string_for_empty_input() -> None: assert render_tools_prompt([]) == "" -def test_render_tools_prompt_excludes_tools_disabled_for_agent_use() -> None: - internal_tool = Tool(name="tests.internal", handler=lambda: None, agent_use=False) +def test_render_tools_prompt_excludes_command_tools() -> None: + internal_tool = Tool(name="tests.internal", handler=lambda: None, exposure="command") assert render_tools_prompt([internal_tool]) == "" @@ -188,7 +188,7 @@ async def test_set_model_overwrites_previous_model(tmp_path) -> None: def test_set_reasoning_effort_is_registered_for_internal_use() -> None: assert REGISTRY["reasoning_effort"] is set_reasoning_effort assert set_reasoning_effort.context is True - assert set_reasoning_effort.agent_use is False + assert set_reasoning_effort.exposure == "command" assert set_reasoning_effort.parameters == { "type": "object", "properties": {"reasoning_effort": {"type": "string"}}, diff --git a/tests/test_codemode.py b/tests/test_codemode.py index c98b8ac8..44ee8e24 100644 --- a/tests/test_codemode.py +++ b/tests/test_codemode.py @@ -116,7 +116,7 @@ def test_stub_path_is_stable_per_session_and_tool_set(tmp_path: Path, monkeypatc def test_run_code_is_a_preserved_tool() -> None: assert run_code.name == RUN_CODE_TOOL_NAME - assert run_code.preserve is True + assert run_code.exposure == "direct" assert run_code.parameters["required"] == ["code"] @@ -210,7 +210,7 @@ async def test_code_mode_command_records_session_switch(tmp_path: Path) -> None: result = await set_code_mode.run(enable=True, context=context) assert set_code_mode.name == "code_mode" - assert set_code_mode.agent_use is False + assert set_code_mode.exposure == "command" assert result == "Session code mode enabled (applies from the next turn)." assert context.state["code_mode"] is True entries = list(await context.tape.store.fetch_all(context.tape.query().kinds("event"))) diff --git a/tests/test_spill.py b/tests/test_spill.py index cc800453..1e05cb01 100644 --- a/tests/test_spill.py +++ b/tests/test_spill.py @@ -308,7 +308,7 @@ async def test_unknown_handle_and_invalid_read_bounds_are_friendly(tmp_path: Pat def test_spill_read_uses_the_builtin_tool_naming_convention() -> None: assert spill_read.name == SPILL_READ_TOOL_NAME == "spill.read" - assert spill_read.preserve is True + assert spill_read.exposure == "direct" assert model_tools([spill_read])[0].name == SPILL_READ_MODEL_NAME == "spill_read" assert "spill_read(handle, cursor?, count?, from_end?)" in render_tools_prompt([spill_read]) diff --git a/tests/test_tool_loading.py b/tests/test_tool_loading.py index a0008a82..2ef62e4c 100644 --- a/tests/test_tool_loading.py +++ b/tests/test_tool_loading.py @@ -196,3 +196,67 @@ async def test_code_mode_exposes_deferred_tools_to_code_without_loading( assert _tool_names(requests[0]) == ["run_code"] assert [tool.name for tool in state[CODE_TOOLS_STATE_KEY]] == ["direct", "provider_lookup"] assert "" not in _system_prompt(requests[0]) + + +@pytest.mark.asyncio +async def test_code_exposure_tools_are_hidden_from_the_model_outside_code_mode( + framework: BubFramework, monkeypatch: pytest.MonkeyPatch +) -> None: + from bub.builtin.tools import tool_describe + + direct = Tool.from_callable(lambda: "direct", name="direct") + code_tool = Tool.from_callable(lambda: "secret", name="secret", exposure="code") + deferred_code_tool = Tool.from_callable(lambda: "lookup", name="provider.lookup", deferred=True, exposure="code") + requests = _scripted_provider( + monkeypatch, + [_tool_call("secret", "secret", {}), {"role": "assistant", "content": "done"}], + ) + agent = Agent(framework, tools=[direct, code_tool, deferred_code_tool, tool_describe], skill_dirs=[]) + + stream = await agent.run_stream(session_id="plain", prompt="Hi.", model="openrouter:test-model") + _ = [event async for event in stream] + + assert _tool_names(requests[0]) == ["direct"] + assert "" not in _system_prompt(requests[0]) + result = next(message for message in requests[1]["messages"] if message["role"] == "tool") + assert result["content"].startswith("Tool `secret` does not exist.") + + +@pytest.mark.asyncio +async def test_code_exposure_tools_are_exposed_to_code_in_code_mode( + framework: BubFramework, monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + from bub.builtin.codemode import CODE_TOOLS_STATE_KEY, run_code + + direct = Tool.from_callable(lambda: "direct", name="direct") + code_tool = Tool.from_callable(lambda: "secret", name="secret", exposure="code") + requests = _scripted_provider(monkeypatch, [{"role": "assistant", "content": "done"}]) + agent = Agent(framework, tools=[direct, code_tool, run_code], skill_dirs=[]) + state: dict[str, Any] = {"code_mode": True, "_runtime_workspace": str(tmp_path)} + + stream = await agent.run_stream(session_id="code", prompt="Hi.", model="openrouter:test-model", state=state) + _ = [event async for event in stream] + + assert _tool_names(requests[0]) == ["run_code"] + assert [tool.name for tool in state[CODE_TOOLS_STATE_KEY]] == ["direct", "secret"] + + +@pytest.mark.asyncio +async def test_tool_describe_does_not_describe_code_exposure_tools( + framework: BubFramework, monkeypatch: pytest.MonkeyPatch +) -> None: + from bub.builtin.tools import tool_describe + + deferred = Tool.from_callable(lambda: "lookup", name="provider.lookup", deferred=True) + code_tool = Tool.from_callable(lambda: "secret", name="secret", exposure="code") + requests = _scripted_provider( + monkeypatch, + [_tool_call("describe", "tool_describe", {"names": ["secret"]}), {"role": "assistant", "content": "done"}], + ) + agent = Agent(framework, tools=[deferred, code_tool, tool_describe], skill_dirs=[]) + + stream = await agent.run_stream(session_id="describe", prompt="Hi.", model="openrouter:test-model") + _ = [event async for event in stream] + + describe_result = next(message for message in requests[1]["messages"] if message["role"] == "tool") + assert json.loads(describe_result["content"]) == {"tools": [], "unknown": ["secret"]} diff --git a/tests/test_tools.py b/tests/test_tools.py index e9e89adc..ebf30e45 100644 --- a/tests/test_tools.py +++ b/tests/test_tools.py @@ -55,13 +55,14 @@ def rename_me(value: str) -> str: assert "additionalProperties" not in rename_me.parameters -def test_model_tools_excludes_tools_disabled_for_agent_use() -> None: +def test_model_tools_excludes_command_tools() -> None: visible_tool = Tool(name="tests.visible", handler=lambda: None) - internal_tool = Tool(name="tests.internal", handler=lambda: None, agent_use=False) + code_tool = Tool(name="tests.code", handler=lambda: None, exposure="code") + internal_tool = Tool(name="tests.internal", handler=lambda: None, exposure="command") - rewritten = model_tools([visible_tool, internal_tool]) + rewritten = model_tools([visible_tool, code_tool, internal_tool]) - assert [item.name for item in rewritten] == ["tests_visible"] + assert [item.name for item in rewritten] == ["tests_visible", "tests_code"] @pytest.mark.asyncio @@ -75,21 +76,21 @@ def sync_tool(payload: EchoInput) -> str: assert sync_tool.name == tool_name assert sync_tool.description == "Sync test tool" - assert sync_tool.agent_use is True + assert sync_tool.exposure == "auto" assert REGISTRY[tool_name] is sync_tool assert await sync_tool.run(value="hello") == "HELLO" @pytest.mark.asyncio -async def test_tool_decorator_can_disable_agent_use_without_disabling_direct_calls() -> None: +async def test_tool_decorator_command_exposure_keeps_direct_calls() -> None: tool_name = "tests.internal_tool" REGISTRY.pop(tool_name, None) - @tool(name=tool_name, agent_use=False) + @tool(name=tool_name, exposure="command") def internal_tool(value: str) -> str: return value.upper() - assert internal_tool.agent_use is False + assert internal_tool.exposure == "command" assert REGISTRY[tool_name] is internal_tool assert await internal_tool.run("hello") == "HELLO" @@ -205,16 +206,9 @@ def test_tool_render_defaults_to_json_for_structured_results() -> None: assert sample.render(EchoInput(value="x")) == '{"value":"x"}' -@pytest.mark.parametrize( - ("agent_use", "preserve", "code_use"), - [(True, False, True), (True, True, False), (False, False, False), (False, True, False)], -) -def test_tool_code_use_excludes_preserved_and_agent_hidden_tools( - agent_use: bool, preserve: bool, code_use: bool -) -> None: - sample = Tool(name="tests.code_use", handler=lambda: None, agent_use=agent_use, preserve=preserve) - - assert sample.code_use is code_use +def test_tool_rejects_unknown_exposure() -> None: + with pytest.raises(ValueError, match="unknown exposure 'model'"): + Tool(name="tests.exposure", handler=lambda: None, exposure="model") # type: ignore[arg-type] def test_tool_decorator_accepts_renderer() -> None: diff --git a/website/src/content/docs/docs/build/tools.mdx b/website/src/content/docs/docs/build/tools.mdx index b0f1a580..3d0c5cbb 100644 --- a/website/src/content/docs/docs/build/tools.mdx +++ b/website/src/content/docs/docs/build/tools.mdx @@ -30,7 +30,7 @@ The tool's name defaults to the function name. Override it with `@tool(name="mat ## Return structured results -Tools should return structured data — a `dict`, `TypedDict`, list, or Pydantic model — rather than pre-formatted prose. The exceptions are tools marked `preserve=True` (see [Code mode](#code-mode)) and comma-command-only tools (`agent_use=False`): code never calls them, so they return plain text. Pass `renderer=` to control how that result is turned into the plain text the model reads: +Tools should return structured data — a `dict`, `TypedDict`, list, or Pydantic model — rather than pre-formatted prose. The exceptions are tools with `exposure="direct"` or `exposure="command"` (see [Code mode](#code-mode)): code never calls them, so they return plain text. Pass `renderer=` to control how that result is turned into the plain text the model reads: ```python from typing import TypedDict @@ -55,12 +55,23 @@ def get_order(order_id: str) -> Order: Code mode lets the model call tools from Python instead of one tool call at a time. It is a per-session switch: send the comma command `,code_mode enable=true` (or `,code_mode enable=false`) to turn it on or off. The change applies from the next turn and persists across restarts. SDK callers can set `state["code_mode"] = True` instead. In a code-mode turn where `run_code` is among the allowed tools: -- The model sees only tools marked `preserve=True` (`bash`, `bash.output`, `bash.kill`, `fs.read`, `fs.write`, `fs.edit`, `spill.read`) plus `run_code(code: str) -> str`. Preserved tools are called directly only; they are not available under `tools.*`. +- The model sees only tools with `exposure="direct"` (`bash`, `bash.output`, `bash.kill`, `fs.read`, `fs.write`, `fs.edit`, `spill.read`) plus `run_code(code: str) -> str`. Direct tools are called directly only; they are not available under `tools.*`. - Every other allowed tool is an async function inside `run_code`, named after its model-facing name (`tape.info` becomes `await tools.tape_info()`). Top-level `await` is allowed and `asyncio.gather` runs calls concurrently. Calls take keyword arguments, go through the usual `before_tool_call`/`after_tool_call` hooks, return structured results, and raise on failure. -- Bub writes a Python stub with one function per non-preserved allowed tool — parameter and return types plus docstring — under `~/.bub/codemode/` and puts its path in the system prompt. The path stays the same for a session as long as its tool set does not change. +- Bub writes a Python stub with one function per allowed `auto` or `code` tool — parameter and return types plus docstring — under `~/.bub/codemode/` and puts its path in the system prompt. The path stays the same for a session as long as its tool set does not change. - `run_code(code, timeout_seconds=120)` returns what the code writes to stdout. An uncaught exception becomes a tool error whose details carry the printed output and the traceback. -If `run_code` is not allowed (for example, a subagent restricted with `allowed_tools`), that turn falls back to direct tool calls. Declare `preserve=True` on your own tools to keep them directly callable. Nested calls run with `ToolContext.code_mode` set to `True` (hooks see it as `call.context.code_mode`); their results are not rendered or spilled. +If `run_code` is not allowed (for example, a subagent restricted with `allowed_tools`), that turn falls back to direct tool calls. + +The `exposure` option of `@tool` and `Tool.from_callable` decides where a tool can be called from: + +| `exposure` | Outside code mode | In code mode | +| --- | --- | --- | +| `"auto"` (default) | Model calls it directly | Only from `run_code` | +| `"direct"` | Model calls it directly | Model calls it directly; not under `tools.*` | +| `"code"` | Unavailable; `tool_describe` does not describe it | Only from `run_code` | +| `"command"` | Comma command only | Comma command only | + +Nested calls run with `ToolContext.code_mode` set to `True` (hooks see it as `call.context.code_mode`); their results are not rendered or spilled. `run_code` hands the code to the session [environment](#run-tools-in-an-environment)'s `run_code`, together with a `call_tool` callback. Tool calls always run on the host through that callback, so hooks still see every call. `timeout_seconds` cancels the environment call. The builtin `LocalEnvironment` starts a fresh Python process for every call (`sys.executable`, working directory is the workspace) and forwards tool calls as JSON lines over its stdin and stdout, so arguments and results must be JSON-serializable. When the code finishes, fails or times out, it kills the process and everything it started. That process runs on the host with the same permissions as `bash`, so enable code mode only where `bash` would be acceptable. The return annotation of a tool function (or `Tool.output_schema`, for tools built by hand) determines the result type shown in the stub. @@ -72,7 +83,7 @@ Mark a tool `deferred=True` (on `@tool`, `Tool.from_callable`, or with `dataclas - `tool_describe` returns the name, description, and parameters of each requested tool and appends a `tool.loaded` tape event with the runtime names of newly loaded deferred tools. Unknown or disallowed names are returned under `unknown`. - On every model request Bub reads all `tool.loaded` events from the tape and appends those tools, in load order, after the direct tools. Loaded tools stay available for later steps and turns of the session until the tape is reset. - `allowed_tools` filters direct and deferred tools alike: a deferred tool outside the scope is neither listed nor loadable. -- Calling a deferred tool before loading it returns guidance to call `tool_describe` first. Code mode sees every allowed tool: deferred tools that are not preserved appear in the stub without loading, and deferral applies only to the remaining model-facing tools. +- Calling a deferred tool before loading it returns guidance to call `tool_describe` first. Code mode sees every allowed tool: deferred `auto` and `code` tools appear in the stub without loading, and deferral applies only to the remaining model-facing tools. ## Run tools in an environment diff --git a/website/src/content/docs/docs/concepts/tape-and-context.mdx b/website/src/content/docs/docs/concepts/tape-and-context.mdx index 9656de9d..8fc73204 100644 --- a/website/src/content/docs/docs/concepts/tape-and-context.mdx +++ b/website/src/content/docs/docs/concepts/tape-and-context.mdx @@ -52,7 +52,7 @@ The default `provide_tape_store` returns a `FileTapeStore` rooted at `~/.bub/tap The builtin spill plugin mounts a `SpillStore` through `provide_tape_sidecar`, registers `spill.read`, and bounds large results through the existing `after_tool_call` hook. Bub measures the result exactly as the model would see it: strings remain plain text, structured values render as JSON, and errors render from their structured payload. Small results retain their original value; large results become a spill reference without clearing the execution's failure state. The sidecar itself has no tool-result interception contract. Large results are stored in a sibling tape named `__sidecar__spill`. The sidecar uses the same `TapeStore` as the session tape, so existing storage plugins do not need a spill-specific interface. Bub writes UTF-8-safe chunks followed by a manifest; the manifest is the completion marker for the stored result. -The main tape keeps a bounded preview and an opaque handle instead of the complete result. The `spill.read` tool reads a bounded page by handle and cursor, including pages counted from the end. Content explicitly returned by `spill.read` is recorded as a normal bounded tool result. `spill.read` is preserved, so it returns plain text and stays directly callable in code mode instead of being routed through `run_code`. +The main tape keeps a bounded preview and an opaque handle instead of the complete result. The `spill.read` tool reads a bounded page by handle and cursor, including pages counted from the end. Content explicitly returned by `spill.read` is recorded as a normal bounded tool result. `spill.read` has `exposure="direct"`, so it returns plain text and stays directly callable in code mode instead of being routed through `run_code`. The sidecar follows the session tape through forks, merges, archive, and reset, but remains a separate tape and is never scanned while constructing the main context. It can also be archived or reset independently with `Tape.archive_sidecar("spill")` and `Tape.reset_sidecar("spill")`. When reset requests an archive, Bub preserves the sidecar if that archive fails. diff --git a/website/src/content/docs/zh-cn/docs/build/tools.mdx b/website/src/content/docs/zh-cn/docs/build/tools.mdx index dbf841c8..d13cb53e 100644 --- a/website/src/content/docs/zh-cn/docs/build/tools.mdx +++ b/website/src/content/docs/zh-cn/docs/build/tools.mdx @@ -30,7 +30,7 @@ def add(a: int, b: int) -> int: ## 返回结构化结果 -工具应返回结构化数据 —— `dict`、`TypedDict`、列表或 Pydantic 模型 —— 而不是预先排版好的文本。例外是标记为 `preserve=True` 的工具(见 [Code mode](#code-mode))和只作为逗号命令的工具(`agent_use=False`):代码永远不会调用它们,因此返回纯文本。通过 `renderer=` 控制如何把结果转换成模型读取的纯文本: +工具应返回结构化数据 —— `dict`、`TypedDict`、列表或 Pydantic 模型 —— 而不是预先排版好的文本。例外是 `exposure="direct"` 或 `exposure="command"` 的工具(见 [Code mode](#code-mode)):代码永远不会调用它们,因此返回纯文本。通过 `renderer=` 控制如何把结果转换成模型读取的纯文本: ```python from typing import TypedDict @@ -55,12 +55,23 @@ def get_order(order_id: str) -> Order: Code mode 让模型在 Python 中调用工具,而不必一次只发起一个工具调用。它是按 session 生效的开关:发送逗号命令 `,code_mode enable=true`(或 `,code_mode enable=false`)来开启或关闭,从下一个 turn 起生效,并在重启后保留。SDK 调用方也可以直接设置 `state["code_mode"] = True`。在开启 code mode 的 turn 中,只要 `run_code` 在允许的工具之内: -- 模型只直接看到标记为 `preserve=True` 的工具(`bash`、`bash.output`、`bash.kill`、`fs.read`、`fs.write`、`fs.edit`、`spill.read`)以及 `run_code(code: str) -> str`。preserve 工具只能直接调用,不会出现在 `tools.*` 中。 +- 模型只直接看到 `exposure="direct"` 的工具(`bash`、`bash.output`、`bash.kill`、`fs.read`、`fs.write`、`fs.edit`、`spill.read`)以及 `run_code(code: str) -> str`。direct 工具只能直接调用,不会出现在 `tools.*` 中。 - 其他允许的工具在 `run_code` 中都是异步函数,名称使用面向模型的形式(`tape.info` 对应 `await tools.tape_info()`)。支持顶层 `await`,可以用 `asyncio.gather` 并发调用。调用只接受关键字参数,照常经过 `before_tool_call`/`after_tool_call` hook,返回结构化结果,失败时抛出异常。 -- Bub 会在 `~/.bub/codemode/` 下为允许的非 preserve 工具生成一个 Python stub 文件,每个工具对应一个函数,包含参数类型、返回类型和 docstring,并把文件路径写入 system prompt。只要工具集合不变,同一个 session 中这个路径保持不变。 +- Bub 会在 `~/.bub/codemode/` 下为允许的 `auto` 和 `code` 工具生成一个 Python stub 文件,每个工具对应一个函数,包含参数类型、返回类型和 docstring,并把文件路径写入 system prompt。只要工具集合不变,同一个 session 中这个路径保持不变。 - `run_code(code, timeout_seconds=120)` 返回代码写到 stdout 的内容。未捕获的异常会变成工具错误,错误详情中包含已打印的输出和 traceback。 -如果 `run_code` 不在允许范围内(例如被 `allowed_tools` 限制的 subagent),该 turn 会退回直接调用工具的方式。给自己的工具声明 `preserve=True` 可以让它保持直接可调用。嵌套调用时 `ToolContext.code_mode`(hook 中通过 `call.context.code_mode` 读取)为 `True`,结果不会被渲染或 spill。 +如果 `run_code` 不在允许范围内(例如被 `allowed_tools` 限制的 subagent),该 turn 会退回直接调用工具的方式。 + +`@tool` 和 `Tool.from_callable` 的 `exposure` 选项决定工具能从哪里调用: + +| `exposure` | code mode 之外 | code mode 下 | +| --- | --- | --- | +| `"auto"`(默认) | 模型直接调用 | 只能在 `run_code` 中调用 | +| `"direct"` | 模型直接调用 | 模型直接调用;不出现在 `tools.*` 中 | +| `"code"` | 不可用;`tool_describe` 也不会描述它 | 只能在 `run_code` 中调用 | +| `"command"` | 只作为逗号命令 | 只作为逗号命令 | + +嵌套调用时 `ToolContext.code_mode`(hook 中通过 `call.context.code_mode` 读取)为 `True`,结果不会被渲染或 spill。 `run_code` 会把代码和一个 `call_tool` 回调一起交给当前 session [执行环境](#在执行环境中运行工具)的 `run_code` 执行。工具调用始终经这个回调在宿主上执行,所以 hook 仍能看到每一次调用。超过 `timeout_seconds` 时会取消执行环境里的执行。builtin 的 `LocalEnvironment` 每次调用都会启动一个新的 Python 进程(使用 `sys.executable`,工作目录为 workspace),工具调用以 JSON 行的形式经进程的 stdin/stdout 转发,因此参数和结果都必须能 JSON 序列化。代码执行完、出错或超时后,它会杀掉该进程及其启动的所有子进程。这个进程直接跑在宿主上,权限与 `bash` 相同,因此只应在允许使用 `bash` 的场景下开启 code mode。stub 中的结果类型来自工具函数的返回注解(手动构造的工具则来自 `Tool.output_schema`)。 @@ -73,7 +84,7 @@ Code mode 让模型在 Python 中调用工具,而不必一次只发起一个 - `tool_describe` 返回所请求工具的名称、描述和参数,并把新加载的按需工具的运行时名称作为 `tool.loaded` 事件写入 tape。未知或不允许的名称放在 `unknown` 中返回。 - 每次请求模型时,Bub 从 tape 读取所有 `tool.loaded` 事件,按加载顺序把这些工具追加到直接加载的工具之后。已加载的工具在该 session 后续的步骤和轮次中持续可用,直到 tape 被重置。 - `allowed_tools` 同时过滤直接加载和按需加载的工具:范围外的按需工具既不会被列出,也无法加载。 -- 未加载就调用按需工具时,会返回提示,要求先调用 `tool_describe`。code mode 能看到所有允许的工具:非 preserve 的按需工具无需加载即出现在 stub 中,按需加载只作用于剩下的模型可见工具。 +- 未加载就调用按需工具时,会返回提示,要求先调用 `tool_describe`。code mode 能看到所有允许的工具:`auto` 和 `code` 的按需工具无需加载即出现在 stub 中,按需加载只作用于剩下的模型可见工具。 ## 在执行环境中运行工具 diff --git a/website/src/content/docs/zh-cn/docs/concepts/tape-and-context.mdx b/website/src/content/docs/zh-cn/docs/concepts/tape-and-context.mdx index 3b4fc86e..0415add1 100644 --- a/website/src/content/docs/zh-cn/docs/concepts/tape-and-context.mdx +++ b/website/src/content/docs/zh-cn/docs/concepts/tape-and-context.mdx @@ -52,7 +52,7 @@ tape_name = f"{workspace_hash}__{session_hash}" builtin spill 插件通过 `provide_tape_sidecar` 挂载 `SpillStore`、注册 `spill.read`,并通过现有 `after_tool_call` hook 限制大结果。Bub 按模型实际看到的形式计算结果大小:字符串保留为纯文本,结构化值渲染为 JSON,错误从结构化 payload 渲染。小结果保留原值;大结果替换为 spill 引用,同时不清除执行的失败状态。sidecar 本身没有工具结果拦截约定。较大的结果存放在名为 `__sidecar__spill` 的 sibling tape 中。sidecar 与 session tape 使用同一个 `TapeStore`,因此现有存储插件不需要实现 spill 专用接口。Bub 依次写入 UTF-8 安全的 chunk,最后写入 manifest;manifest 是该结果已完整存储的提交标记。 -主 tape 只保留有界预览和 opaque handle,不保存完整结果。`spill.read` 工具按 handle 与 cursor 有界读取,也支持从末尾开始读取。由 `spill.read` 明确返回的内容会作为普通的有界 tool result 记录。`spill.read` 是 preserve 工具,因此返回纯文本,并且在 code mode 下仍然直接可调用,而不必经过 `run_code`。 +主 tape 只保留有界预览和 opaque handle,不保存完整结果。`spill.read` 工具按 handle 与 cursor 有界读取,也支持从末尾开始读取。由 `spill.read` 明确返回的内容会作为普通的有界 tool result 记录。`spill.read` 的 `exposure` 是 `"direct"`,因此返回纯文本,并且在 code mode 下仍然直接可调用,而不必经过 `run_code`。 sidecar 与 session tape 一起参与 fork、merge、archive 和 reset,但始终是独立 tape,构造主 context 时不会扫描它。也可以通过 `Tape.archive_sidecar("spill")` 和 `Tape.reset_sidecar("spill")` 单独 archive 或 reset sidecar。当 reset 要求先 archive 时,如果 archive 失败,Bub 会保留 sidecar。