From 631c63f09bfa8dbce901437f333579a64cc3be63 Mon Sep 17 00:00:00 2001 From: Niels Rogge Date: Fri, 18 Sep 2026 13:55:20 +0000 Subject: [PATCH] Default MCP search_papers to hybrid mode like pwc search The hosted MCP server defaulted search_papers to keyword mode while the CLI's `pwc search` defaults to hybrid, so the same agent query returned different results depending on which surface it used. Share one DEFAULT_SEARCH_MODE between the tool schema and the rate limiter so an omitted mode is served as hybrid and counted toward the semantic search limit, and document keyword mode as the opt-out for exact terminology or when that limit is close. Bump pwc-mcp to 0.2.3. Co-Authored-By: Claude Fable 5.1 --- mcp_server/README.md | 8 ++--- mcp_server/SKILL.md | 10 ++++--- mcp_server/SPEC.md | 4 +-- mcp_server/pyproject.toml | 2 +- mcp_server/src/pwc_mcp/__init__.py | 2 +- mcp_server/src/pwc_mcp/app.py | 6 ++-- mcp_server/src/pwc_mcp/server.py | 6 ++-- mcp_server/tests/test_app.py | 47 +++++++++++++++++++++++++++++- mcp_server/tests/test_server.py | 4 +++ mcp_server/uv.lock | 2 +- 10 files changed, 72 insertions(+), 19 deletions(-) diff --git a/mcp_server/README.md b/mcp_server/README.md index e5bb11b..2027631 100644 --- a/mcp_server/README.md +++ b/mcp_server/README.md @@ -58,13 +58,13 @@ Parameter names follow the CLI flags except for the established MCP names (`--include-evals`). Terminal-only flags (`--json`, `--implementation-coverage`, `--flat`) have no parameter because MCP output is always structured. Hosted differences from the CLI: `limit` is capped at 25, -`search_papers` defaults to `keyword` mode, `get_paper_info` returns compact -official-first resources, and `read_paper` is chunked. +`get_paper_info` returns compact official-first resources, and `read_paper` +is chunked. `search_papers` defaults to `hybrid` mode like `pwc search`. `tests/test_parity.py` fails when the CLI parser and the tool schemas drift. All tools are annotated read-only and idempotent. Search is deterministic; -the caller controls keyword, hybrid, or semantic mode. `read_paper` fetches at -most one 64 KiB catalog chunk per call and returns a signed, one-hour +the caller controls hybrid (default), keyword, or semantic mode. `read_paper` +fetches at most one 64 KiB catalog chunk per call and returns a signed, one-hour continuation cursor when more Markdown remains. Continuations stay pinned to the resolved paper and content version, so a changed paper fails with an explicit restart response. diff --git a/mcp_server/SKILL.md b/mcp_server/SKILL.md index d279512..8218277 100644 --- a/mcp_server/SKILL.md +++ b/mcp_server/SKILL.md @@ -4,7 +4,7 @@ description: "Papers With Code MCP tools for searching and reading AI/ML papers, compatibility: "Requires an MCP client connected to https://paperswithcode.co/mcp with the Papers With Code tools available." --- -Generated for `pwc-mcp v0.2.2` and stock-client MCP protocol `2025-11-25`. +Generated for `pwc-mcp v0.2.3` and stock-client MCP protocol `2025-11-25`. The tools query the public [Papers With Code](https://paperswithcode.co) catalog anonymously and are read-only. Every tool runs the matching `pwc` CLI research @@ -108,9 +108,11 @@ available and the user needs one of those capabilities. - Tool results use stable, versioned structured output with compact text fallbacks. Prefer structured fields over parsing the text fallback. -- Search mode is deterministic: `keyword` is the default; `hybrid` adds dense - retrieval and `semantic` uses it alone. Both `hybrid` and `semantic` count - toward the semantic search rate limit. +- Search mode is deterministic: `hybrid` is the default, matching + `pwc search`; `keyword` uses lexical retrieval alone and `semantic` uses + dense retrieval alone. Both `hybrid` and `semantic` count toward the semantic + search rate limit, so pass `mode: "keyword"` for exact terminology or when + that limit is close. - Catalog-filtered paper lists fail closed unless the server confirms every requested filter; never treat results from an older server as filtered. Parameter-filtered leaderboards fail closed unless the server confirms diff --git a/mcp_server/SPEC.md b/mcp_server/SPEC.md index edd8811..10bfae7 100644 --- a/mcp_server/SPEC.md +++ b/mcp_server/SPEC.md @@ -49,8 +49,8 @@ Tools run the CLI handlers in-process through the shared cached transport, so validation, fail-closed filter confirmation, and the JSON payload are the CLI's. Every result includes that payload as `data` beside typed projections. Terminal-only flags have no parameter. The hosted service caps `limit` at 25, -defaults `search_papers` to keyword mode, returns compact official-first paper -resources, and serves `read_paper` in chunks. +returns compact official-first paper resources, and serves `read_paper` in +chunks. `search_papers` defaults to hybrid mode, the same as `pwc search`. Expose these resource templates: diff --git a/mcp_server/pyproject.toml b/mcp_server/pyproject.toml index 0ef7902..7fb5b70 100644 --- a/mcp_server/pyproject.toml +++ b/mcp_server/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "pwc-mcp" -version = "0.2.2" +version = "0.2.3" description = "Read-only Papers With Code MCP server" readme = "README.md" requires-python = ">=3.10" diff --git a/mcp_server/src/pwc_mcp/__init__.py b/mcp_server/src/pwc_mcp/__init__.py index 0bfe2fb..13777c6 100644 --- a/mcp_server/src/pwc_mcp/__init__.py +++ b/mcp_server/src/pwc_mcp/__init__.py @@ -1,3 +1,3 @@ """Read-only Papers With Code MCP server.""" -__version__ = "0.2.2" +__version__ = "0.2.3" diff --git a/mcp_server/src/pwc_mcp/app.py b/mcp_server/src/pwc_mcp/app.py index fca6998..5310eee 100644 --- a/mcp_server/src/pwc_mcp/app.py +++ b/mcp_server/src/pwc_mcp/app.py @@ -25,7 +25,7 @@ from pwc_mcp import __version__ from pwc_mcp.catalog import CatalogClient -from pwc_mcp.server import TOOL_COMMANDS, Catalog, build_server +from pwc_mcp.server import DEFAULT_SEARCH_MODE, TOOL_COMMANDS, Catalog, build_server LOGGER = logging.getLogger("pwc_mcp.requests") # MCP SDK diagnostics can include peer-supplied tool names and resource URIs. @@ -270,7 +270,7 @@ async def replay() -> Message: def _tool_and_semantic(headers: Headers, body: bytes) -> tuple[str | None, bool]: header_tool = headers.get("mcp-name") body_tool = None - mode = None + mode = DEFAULT_SEARCH_MODE try: payload = json.loads(body) except (UnicodeDecodeError, json.JSONDecodeError): @@ -286,7 +286,7 @@ def _tool_and_semantic(headers: Headers, body: bytes) -> tuple[str | None, bool] params = payload.get("params") arguments = params.get("arguments") if isinstance(params, dict) else None if isinstance(arguments, dict): - mode = arguments.get("mode") + mode = arguments.get("mode", DEFAULT_SEARCH_MODE) tool = body_tool or header_tool tool = tool if tool in KNOWN_TOOLS else None # Hybrid retrieval also runs the dense embedding search upstream. diff --git a/mcp_server/src/pwc_mcp/server.py b/mcp_server/src/pwc_mcp/server.py index 546911f..79982d0 100644 --- a/mcp_server/src/pwc_mcp/server.py +++ b/mcp_server/src/pwc_mcp/server.py @@ -67,6 +67,8 @@ ) # Hosted ceiling on rows per response (SPEC.md); the CLI allows up to 100. MAX_ROWS = 25 +# Matches `pwc search --mode`; app.py counts omitted modes with this value. +DEFAULT_SEARCH_MODE = "hybrid" # CLI lookup failures the caller can act on (an unknown or ambiguous name, an # unconfirmed filter). Transport, HTTP, and response-shape failures stay generic. CLIENT_FACING_ERRORS = ( @@ -382,14 +384,14 @@ def run(tool: str, **parameters: Any) -> Any: @server.tool(annotations=READ_ONLY, structured_output=True) def search_papers( query: Query, - mode: Literal["hybrid", "keyword", "semantic"] = "keyword", + mode: Literal["hybrid", "keyword", "semantic"] = DEFAULT_SEARCH_MODE, page: Page = 1, limit: Limit = 10, published_after: IsoDate | None = None, published_before: IsoDate | None = None, has_official_implementation: bool = False, ) -> PaperPage: - """Search papers by title, topic, author, or arXiv ID (`pwc search`). Broad discovery only: for best, top, or state-of-the-art model questions start with get_task, list_benchmarks, and get_benchmark, which return leaderboard evidence that search cannot.""" + """Search papers by title, topic, author, or arXiv ID (`pwc search`). Broad discovery only: for best, top, or state-of-the-art model questions start with get_task, list_benchmarks, and get_benchmark, which return leaderboard evidence that search cannot. Mode defaults to hybrid like `pwc search`; use keyword for exact terminology or to stay under the semantic search rate limit.""" _validate_date_range(published_after, published_before) result = _paper_page( run( diff --git a/mcp_server/tests/test_app.py b/mcp_server/tests/test_app.py index 3732c5d..3a2e165 100644 --- a/mcp_server/tests/test_app.py +++ b/mcp_server/tests/test_app.py @@ -43,7 +43,7 @@ def test_health_and_browser_origin_policy_are_explicit(): assert health.json() == { "status": "ok", "service": "pwc-mcp", - "version": "0.2.2", + "version": "0.2.3", "protocol": "2025-11-25", } assert rejected.status_code == 403 @@ -294,6 +294,51 @@ def test_hybrid_search_counts_toward_the_semantic_limit(): assert limited.status_code == 429 +def test_omitted_search_mode_defaults_to_hybrid_and_counts_toward_the_semantic_limit(): + app = create_app( + StubCatalog(), + allowed_hosts=["testserver"], + semantic_limit=1, + ) + body = { + "jsonrpc": "2.0", + "id": 1, + "method": "tools/call", + "params": {"name": "search_papers", "arguments": {"query": "attention"}}, + } + + with TestClient(app) as client: + first = client.post("/mcp", json=body) + limited = client.post("/mcp", json=body) + + assert first.status_code != 429 + assert limited.status_code == 429 + + +def test_keyword_search_does_not_count_toward_the_semantic_limit(): + app = create_app( + StubCatalog(), + allowed_hosts=["testserver"], + semantic_limit=1, + ) + body = { + "jsonrpc": "2.0", + "id": 1, + "method": "tools/call", + "params": { + "name": "search_papers", + "arguments": {"query": "attention", "mode": "keyword"}, + }, + } + + with TestClient(app) as client: + first = client.post("/mcp", json=body) + second = client.post("/mcp", json=body) + + assert first.status_code != 429 + assert second.status_code != 429 + + def test_rate_limits_are_env_configurable_with_hosted_defaults(monkeypatch): from pwc_mcp.app import ( DEFAULT_GLOBAL_CONCURRENCY_LIMIT, diff --git a/mcp_server/tests/test_server.py b/mcp_server/tests/test_server.py index 6df8b9e..00ef157 100644 --- a/mcp_server/tests/test_server.py +++ b/mcp_server/tests/test_server.py @@ -256,6 +256,10 @@ async def exercise(): "keyword", "semantic", ] + # Same default as `pwc search --mode`. + assert ( + tools["search_papers"].input_schema["properties"]["mode"]["default"] == "hybrid" + ) assert result.is_error is False assert result.structured_content == { "schema_version": "v1", diff --git a/mcp_server/uv.lock b/mcp_server/uv.lock index dc3a56b..e54f7b6 100644 --- a/mcp_server/uv.lock +++ b/mcp_server/uv.lock @@ -443,7 +443,7 @@ source = { editable = "../standalone_cli" } [[package]] name = "pwc-mcp" -version = "0.2.2" +version = "0.2.3" source = { editable = "." } dependencies = [ { name = "mcp" },