diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 1c121f74c..a072f6886 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -40,7 +40,7 @@ jobs: cache: true - name: Initialize CodeQL - uses: github/codeql-action/init@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4 + uses: github/codeql-action/init@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4 with: languages: ${{ matrix.language }} build-mode: ${{ matrix.build-mode }} @@ -51,4 +51,4 @@ jobs: run: go build ./... - name: Perform CodeQL analysis - uses: github/codeql-action/analyze@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4 + uses: github/codeql-action/analyze@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4 diff --git a/cmd/gomodel/docs/docs.go b/cmd/gomodel/docs/docs.go index 24ffb6836..90f72c5dc 100644 --- a/cmd/gomodel/docs/docs.go +++ b/cmd/gomodel/docs/docs.go @@ -252,6 +252,12 @@ const docTemplate = `{ "name": "stream", "in": "query" }, + { + "type": "string", + "description": "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay", + "name": "exclude_operation", + "in": "query" + }, { "type": "string", "description": "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message", @@ -381,6 +387,12 @@ const docTemplate = `{ "name": "stream", "in": "query" }, + { + "type": "string", + "description": "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay", + "name": "exclude_operation", + "in": "query" + }, { "type": "string", "description": "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message", @@ -1161,7 +1173,7 @@ const docTemplate = `{ }, "/admin/mcp-servers/{name}/catalog": { "get": { - "description": "Lists the tools, prompts, resources, and resource templates the named server currently exposes through the gateway, after operator tool filters. Names are the upstream originals; the aggregated /mcp endpoint prefixes them with the server slug.", + "description": "Lists the tools, prompts, resources, and resource templates the named server currently exposes through the gateway, after operator tool filters. Discovered tools the filters hide are listed separately under excluded_tools. Names are the upstream originals; the aggregated /mcp endpoint prefixes them with the server slug.", "produces": [ "application/json" ], @@ -7492,6 +7504,213 @@ const docTemplate = `{ ] } }, + "/v1/systemone": { + "post": { + "description": "Available when a jev or openrouter provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", + "consumes": [ + "application/json" + ], + "produces": [ + "application/json" + ], + "tags": [ + "systemone" + ], + "summary": "Evaluate a System One decision request (Jev / Kev)", + "parameters": [ + { + "description": "System One request: model, state, and questions", + "name": "request", + "in": "body", + "required": true, + "schema": { + "type": "object" + } + } + ], + "responses": { + "200": { + "description": "System One answers, in the provider's shape", + "schema": { + "type": "object" + } + }, + "400": { + "description": "Bad Request", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "401": { + "description": "Unauthorized", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "404": { + "description": "Not Found", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "429": { + "description": "Too Many Requests", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "502": { + "description": "Bad Gateway", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ] + } + }, + "/v1/systemone/permute": { + "post": { + "description": "A Kev server diagnostic: the request is a System One request, and n_perm (1 to 64, default 6) sets how many option orders run. Only jev providers pointing at a Kev server serve it.", + "consumes": [ + "application/json" + ], + "produces": [ + "application/json" + ], + "tags": [ + "systemone" + ], + "summary": "Run one Choice question with several option orders (Kev)", + "parameters": [ + { + "description": "System One request with one Choice question", + "name": "request", + "in": "body", + "required": true, + "schema": { + "type": "object" + } + } + ], + "responses": { + "200": { + "description": "Kev's answer, in the provider's shape", + "schema": { + "type": "object" + } + }, + "400": { + "description": "Bad Request", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "401": { + "description": "Unauthorized", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "404": { + "description": "Not Found", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "429": { + "description": "Too Many Requests", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "502": { + "description": "Bad Gateway", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ] + } + }, + "/v1/systemone/separate": { + "post": { + "description": "A Kev server diagnostic that answers each question separately. Only jev providers pointing at a Kev server serve it.", + "consumes": [ + "application/json" + ], + "produces": [ + "application/json" + ], + "tags": [ + "systemone" + ], + "summary": "Run each System One question in its own forward pass (Kev)", + "parameters": [ + { + "description": "System One request: model, state, and questions", + "name": "request", + "in": "body", + "required": true, + "schema": { + "type": "object" + } + } + ], + "responses": { + "200": { + "description": "Kev's answer, in the provider's shape", + "schema": { + "type": "object" + } + }, + "400": { + "description": "Bad Request", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "401": { + "description": "Unauthorized", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "404": { + "description": "Not Found", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "429": { + "description": "Too Many Requests", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "502": { + "description": "Bad Gateway", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ] + } + }, "/v1/usage": { "get": { "description": "Returns recorded usage, budget statuses, and rate limit statuses for the caller's effective user path (the path bound to the managed API key, or the user-path header for master-key callers).", @@ -7982,9 +8201,18 @@ const docTemplate = `{ "type": "string" } }, + "disallowed_user_paths": { + "type": "array", + "items": { + "type": "string" + } + }, "enabled": { "type": "boolean" }, + "excluded_tool_count": { + "type": "integer" + }, "headers": { "type": "object", "additionalProperties": { @@ -8474,6 +8702,12 @@ const docTemplate = `{ "type": "string" } }, + "disallowed_user_paths": { + "type": "array", + "items": { + "type": "string" + } + }, "enabled": { "type": "boolean" }, @@ -9015,6 +9249,9 @@ const docTemplate = `{ "name": { "type": "string" }, + "strict": { + "type": "boolean" + }, "type": { "type": "string" } @@ -11296,8 +11533,14 @@ const docTemplate = `{ "description": { "type": "string" }, + "destructive": { + "type": "boolean" + }, "name": { "type": "string" + }, + "read_only": { + "type": "boolean" } } }, @@ -11332,6 +11575,12 @@ const docTemplate = `{ "mcpgateway.CatalogView": { "type": "object", "properties": { + "excluded_tools": { + "type": "array", + "items": { + "$ref": "#/definitions/mcpgateway.CatalogFeature" + } + }, "instructions": { "type": "string" }, diff --git a/config/config.example.yaml b/config/config.example.yaml index 3391d595f..1a177aaa6 100644 --- a/config/config.example.yaml +++ b/config/config.example.yaml @@ -144,9 +144,10 @@ models: # headers: # Authorization: "Bearer ${GITHUB_PAT}" # description: GitHub tools -# allowed_tools: [] # allowlist of upstream tool names; empty = all -# disallowed_tools: [] # blocklist, applied after the allowlist +# allowed_tools: [] # allowlist of upstream tool names; empty = all (new upstream tools stay hidden when set) +# disallowed_tools: [] # tools to exclude, applied after the allowlist; edits apply without a reconnect # user_paths: [] # restrict visibility to these user-path subtrees; empty = everyone +# disallowed_user_paths: [] # hide from these subtrees, even inside user_paths # tool_timeout: 30s # per tools/call upper bound # local-files: # stdio servers spawn a subprocess and are declarative-only: # transport: stdio # the admin API and dashboard reject them by design @@ -590,10 +591,14 @@ providers: type: jev api_key: "${JEV_API_KEY}" # base_url defaults to "https://api.typesafe.ai". TypeSafe's System One - # API is a decision API with no OpenAI-compatible surface: requests go to - # POST /p/jev/v1/systemone, or point the TypeSafe SDK at /p/jev. A - # self-hosted Kev server speaks the same API without authentication: - # set base_url (e.g. "http://localhost:8009") and omit api_key. + # API is a decision API with no OpenAI-compatible surface: configuring + # this provider (or openrouter, which serves Jev natively) enables + # POST /v1/systemone, which forwards requests natively (point the + # TypeSafe SDK at the gateway root). A self-hosted Kev + # server speaks the same API without authentication: set base_url + # (e.g. "http://localhost:8009") and omit api_key. Name it "kev" to see + # that name in logs and usage; no separate provider type is needed. + # Pinned versions such as "jev-1.13.0" route here without being listed. # Jev is priced per input token and is not in the upstream model catalog; # declare its pricing here to have the gateway cost System One requests. # models: diff --git a/config/env.go b/config/env.go index 17c6dfcd5..00919d895 100644 --- a/config/env.go +++ b/config/env.go @@ -66,7 +66,7 @@ func applyPluginsLoadEnv(cfg *Config) { return } load := make([]PluginFileConfig, 0, 4) - for _, item := range strings.Split(v, ",") { + for item := range strings.SplitSeq(v, ",") { if entry := parsePluginLoadEntry(item); entry.File != "" { load = append(load, entry) } diff --git a/config/mcp.go b/config/mcp.go index 6e8a161d6..d4d1f0592 100644 --- a/config/mcp.go +++ b/config/mcp.go @@ -135,6 +135,10 @@ type MCPServerConfig struct { // (subtree match, same semantics as virtual models). Empty means all. UserPaths []string `yaml:"user_paths,omitempty" json:"user_paths,omitempty"` + // DisallowedUserPaths hides the server from these user-path subtrees, + // even inside an allowed UserPaths subtree. Empty excludes nobody. + DisallowedUserPaths []string `yaml:"disallowed_user_paths,omitempty" json:"disallowed_user_paths,omitempty"` + // ToolTimeout bounds a single tools/call against this server. // Default: 30s. ToolTimeout time.Duration `yaml:"tool_timeout,omitempty" json:"tool_timeout,omitempty"` @@ -205,6 +209,9 @@ func expandMCPServerEnv(server *MCPServerConfig) { for i := range server.UserPaths { server.UserPaths[i] = expandString(server.UserPaths[i]) } + for i := range server.DisallowedUserPaths { + server.DisallowedUserPaths[i] = expandString(server.DisallowedUserPaths[i]) + } for key, value := range server.Headers { server.Headers[key] = expandString(value) } @@ -370,23 +377,35 @@ func ValidateMCPServerConfig(server *MCPServerConfig) error { if server.ToolTimeout == 0 { server.ToolTimeout = DefaultMCPToolTimeout } - if len(server.UserPaths) > 0 { - normalized := make([]string, 0, len(server.UserPaths)) - for _, raw := range server.UserPaths { - path, err := core.NormalizeUserPath(raw) - if err != nil { - return fmt.Errorf("invalid user_paths value %q: %w", raw, err) - } - if path != "" && !slices.Contains(normalized, path) { - normalized = append(normalized, path) - } - } - slices.Sort(normalized) - server.UserPaths = normalized + var err error + if server.UserPaths, err = normalizeMCPUserPaths("user_paths", server.UserPaths); err != nil { + return err + } + if server.DisallowedUserPaths, err = normalizeMCPUserPaths("disallowed_user_paths", server.DisallowedUserPaths); err != nil { + return err } return nil } +// normalizeMCPUserPaths canonicalizes, dedupes, and sorts one user-path list. +func normalizeMCPUserPaths(field string, paths []string) ([]string, error) { + if len(paths) == 0 { + return paths, nil + } + normalized := make([]string, 0, len(paths)) + for _, raw := range paths { + path, err := core.NormalizeUserPath(raw) + if err != nil { + return nil, fmt.Errorf("invalid %s value %q: %w", field, raw, err) + } + if path != "" && !slices.Contains(normalized, path) { + normalized = append(normalized, path) + } + } + slices.Sort(normalized) + return normalized, nil +} + // MCPServerEnabled reports the effective enabled state (default true). func MCPServerEnabled(server MCPServerConfig) bool { return server.Enabled == nil || *server.Enabled diff --git a/config/mcp_test.go b/config/mcp_test.go index 1c38902d6..2be771e81 100644 --- a/config/mcp_test.go +++ b/config/mcp_test.go @@ -98,6 +98,11 @@ func TestNormalizeMCPConfigRejectsInvalid(t *testing.T) { servers: map[string]MCPServerConfig{"a": {URL: "https://x/mcp", UserPaths: []string{"/team/../admin"}}}, wantErr: "invalid user_paths", }, + { + name: "invalid disallowed user path", + servers: map[string]MCPServerConfig{"a": {URL: "https://x/mcp", DisallowedUserPaths: []string{"/team/../admin"}}}, + wantErr: "invalid disallowed_user_paths", + }, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { @@ -112,8 +117,9 @@ func TestNormalizeMCPConfigRejectsInvalid(t *testing.T) { func TestNormalizeMCPConfigCanonicalizesUserPaths(t *testing.T) { cfg := MCPConfig{Servers: map[string]MCPServerConfig{ "a": { - URL: "https://x/mcp", - UserPaths: []string{" team/a ", "/team/a", "/team/b/"}, + URL: "https://x/mcp", + UserPaths: []string{" team/a ", "/team/a", "/team/b/"}, + DisallowedUserPaths: []string{"/team/a/contractors/", " team/a/contractors"}, }, }} err := normalizeMCPConfig(&cfg) @@ -122,6 +128,7 @@ func TestNormalizeMCPConfigCanonicalizesUserPaths(t *testing.T) { got := cfg.Servers["a"].UserPaths want := []string{"/team/a", "/team/b"} require.Equal(t, want, got) + assert.Equal(t, []string{"/team/a/contractors"}, cfg.Servers["a"].DisallowedUserPaths) } func TestApplyMCPEnvMergesOverYAML(t *testing.T) { diff --git a/docs/about/roadmap.mdx b/docs/about/roadmap.mdx index 254e33bab..19c577d21 100644 --- a/docs/about/roadmap.mdx +++ b/docs/about/roadmap.mdx @@ -49,11 +49,11 @@ keywords: ["roadmap", "milestones", "upcoming features", "releases"] - [x] Intelligent routing ([GoModel Pro](/pro/intelligent-routing), beta) - [x] Context window compression ([GoModel Pro](/pro/compression)) - [x] OIDC single sign-on ([GoModel Pro](/pro/sso)) +- [x] Egress proxy pools with health checks and failover ([GoModel Pro](/pro/egress), beta; per-provider `proxy_url` is in [core](/providers/outbound-proxy)) - [ ] Cluster mode - [ ] LDAP integration - [ ] Role-based access control (RBAC) for the dashboard and admin API - [ ] Secret manager integrations (Vault, AWS Secrets Manager/KMS, Azure Key Vault, GCP Secret Manager) -- [ ] Managed outbound proxy pools: health checks, failover between proxies, and assignment rules across providers (per-provider `proxy_url` is in [core](/providers/outbound-proxy)) - [ ] Audit-log export/streaming to object storage and data lakes - [ ] IP / CIDR allowlists - [ ] LTS channel with signed artifacts, SBOM, and CVE pre-notification diff --git a/docs/advanced/api-endpoints.mdx b/docs/advanced/api-endpoints.mdx index f39cc87da..ff9b452fc 100644 --- a/docs/advanced/api-endpoints.mdx +++ b/docs/advanced/api-endpoints.mdx @@ -12,8 +12,8 @@ documented separately in [Admin Endpoints](/advanced/admin-endpoints). For request and response details, see the dedicated guides: [Responses API](/advanced/responses-api), [Conversations API](/advanced/conversations-api), [Anthropic Messages API](/advanced/anthropic-messages-api), -[Audio API](/advanced/audio-api), [Images API](/advanced/images-api), and -[Usage API](/advanced/usage-api). +[Audio API](/advanced/audio-api), [Images API](/advanced/images-api), +[System One API](/advanced/systemone-api), and [Usage API](/advanced/usage-api). ## OpenAI-Compatible API @@ -111,6 +111,17 @@ metered. | `/v1/messages` | POST | Anthropic Messages API through translated model routing (streaming supported) | | `/v1/messages/count_tokens` | POST | Heuristic Anthropic Messages input token estimate | +## System One API + +Available when a `jev` or `openrouter` provider is configured; see +[System One API](/advanced/systemone-api). + +| Endpoint | Method | Description | +| ---------------------------- | ------ | ------------------------------------------------------------------------ | +| `/v1/systemone` | POST | Evaluate a state against typed questions (Jev, Kev), forwarded natively | +| `/v1/systemone/permute` | POST | Kev only: run one Choice question with several option orders | +| `/v1/systemone/separate` | POST | Kev only: run each question in its own forward pass | + ## Gateway Extensions | Endpoint | Method | Description | diff --git a/docs/advanced/audio-api.mdx b/docs/advanced/audio-api.mdx index 65c699395..73fa45bef 100644 --- a/docs/advanced/audio-api.mdx +++ b/docs/advanced/audio-api.mdx @@ -165,6 +165,27 @@ JSON object; `text`, `srt`, and `vtt` return a `text/plain` body. Add `stream=true` to receive the transcript as it is produced — see [Streaming](#streaming). +### Self-hosted speech-to-text servers + +Many self-hosted STT servers serve `/v1/audio/transcriptions` but no `/v1/models`. +Register one as an `openai` provider and list the model it serves: + +```yaml +providers: + local-stt: + type: openai + base_url: "http://localhost:8000/v1" + api_key: "unused" + models: + - id: "whisper-1" # the model name your server expects + metadata: + modes: [audio_transcription] # lists it as an audio model +``` + +When `/models` answers 404 or 405, GoModel serves the configured models and +reports the provider as healthy. Other `/models` failures still mark it as +degraded. + ## Streaming GoModel forwards an audio body to the client **as the provider produces it**, so diff --git a/docs/advanced/configuration.mdx b/docs/advanced/configuration.mdx index 3fe2eaad9..8d7e6e7d6 100644 --- a/docs/advanced/configuration.mdx +++ b/docs/advanced/configuration.mdx @@ -452,7 +452,10 @@ Every provider type also accepts a comma-separated configured model list via `_MODELS`, for example `OPENROUTER_MODELS`, `ORACLE_MODELS`, `AZURE_MODELS`, `SGLANG_MODELS`, `VLLM_MODELS`, or `LLMD_MODELS`. By default, `CONFIGURED_PROVIDER_MODELS_MODE=fallback` uses configured lists only when -upstream `/models` fails, returns nil, or returns an empty list. Set +upstream `/models` fails, returns nil, or returns an empty list. An +OpenAI-compatible provider without a `/models` endpoint (404 or 405) is served +from its configured list and reported healthy, which suits single-API servers +such as speech-to-text. Set `CONFIGURED_PROVIDER_MODELS_MODE=allowlist` to expose only configured models for providers that define a list and skip their upstream `/models` calls. YAML `providers..models` provides the same model-list input for named provider diff --git a/docs/advanced/systemone-api.mdx b/docs/advanced/systemone-api.mdx new file mode 100644 index 000000000..8e765fe86 --- /dev/null +++ b/docs/advanced/systemone-api.mdx @@ -0,0 +1,179 @@ +--- +title: "System One API" +description: "Send TypeSafe System One decision requests (Jev, Kev) through GoModel, forwarded natively with virtual models, guardrails, caching, failover, audit, and usage." +icon: "scale" +keywords: ["System One", "systemone", "Jev", "Kev", "TypeSafe", "decision model", "noul", "choice", "score", "OpenRouter"] +--- + +`POST /v1/systemone` serves TypeSafe's System One API: a request carries a +`state` (the text or record to evaluate) and a map of typed questions, and the +answer is a calibrated probability per question. It is a decision API, not a +text generator, so GoModel forwards it **natively** and never translates it to +or from chat. + +The endpoint is available once a [`jev` provider](/providers/jev) (hosted Jev +or a self-hosted Kev server) or an `openrouter` provider is configured. Without +one, it answers `404`. + +## Request and answer + +```bash +curl -s http://localhost:8080/v1/systemone \ + -H "Authorization: Bearer $GOMODEL_MASTER_KEY" \ + -H "Content-Type: application/json" \ + -d '{ + "model": "jev-latest", + "state": "Shoes arrived two weeks late and in the wrong size. Also I see two charges on my card.", + "questions": { + "department": {"type": "choice", "instructions": "Which team should handle this?", + "criteria": {"returns": "Exchanges, refunds, wrong items", + "shipping": "Delivery status, delays", + "billing": "Charges, invoices"}}, + "escalate": {"type": "noul", "instructions": "Does this need urgent human attention?"}, + "frustration": {"type": "score", "instructions": "How frustrated is the customer?", + "criteria": ["Calm", "Frustrated", "Very angry"]} + } + }' +``` + +The answer is the provider's own, relayed unchanged: + +```json +{ + "model": "jev-1.13.0", + "answers": { + "department": {"type": "choice", "choice": "returns", "confidence": 0.21, + "probabilities": {"returns": 0.47, "shipping": 0.28, "billing": 0.25}}, + "escalate": {"type": "noul", "noul": 0.93}, + "frustration": {"type": "score", "score": 1.44, "confidence": 0.78, + "legend": {"0": "Calm", "1": "Frustrated", "2": "Very angry"}, + "probabilities": {"0": 0.00, "1": 0.56, "2": 0.44}} + }, + "usage": {"input_tokens": 101, "output_tokens": 161} +} +``` + +The TypeSafe SDKs send `POST {base_url}/v1/systemone`, so point them at the +gateway root (`base_url="http://localhost:8080"`) with your GoModel key; see +[Jev / Kev](/providers/jev#using-the-typesafe-sdks). + +## Routes + +| Route | What it does | +| --- | --- | +| `POST /v1/systemone` | Evaluate a state against a map of questions | +| `POST /v1/systemone/permute` | Kev only: run one Choice question with several option orders (`n_perm`, 1 to 64, default 6) | +| `POST /v1/systemone/separate` | Kev only: run each question in its own forward pass | + +The Kev routes behave like `/v1/systemone`. They are refused for OpenRouter, +which answers only the evaluation route; a hosted TypeSafe `jev` provider +returns its own `404` for them. + +## What the gateway does + +1. Resolves `model` like any other endpoint: a bare name, a provider-qualified + name (`jev/jev-latest`), or a [virtual model](/features/virtual-models), + then applies the caller's [model allowlist](/features/users), + [rate limits](/features/rate-limits), and [budgets](/features/budgets). +2. Runs the workflow's prompt [guardrails](/advanced/guardrails) over `state`. +3. Serves an identical earlier request from the [response cache](/features/cache). +4. Forwards the body with only `model` (the resolved name) and `state` (if a + guardrail edited it) changed. Questions, criteria, and every other field + reach the provider byte for byte. +5. Relays the answer unchanged and records it in the audit log (request type + **System One**) and in usage. + +## Models + +| Provider | Model names | +| --- | --- | +| `jev` (hosted) | `jev/jev-latest`, `jev/jev-preview`, and any versioned ID such as `jev/jev-1.13.0` | +| `jev` (Kev server) | `kev/kev-latest` and the checkpoint's aliases, for a provider named `kev` | +| `openrouter` | `openrouter/typesafe/jev-1.13`, `openrouter/~typesafe/jev-latest`, and OpenRouter's other decision models, such as `openrouter/jaredpalmer/kev-4b` | + +System One models are listed in `GET /v1/models` as utility models with no +generation mode. + +TypeSafe lists only its aliases but accepts any versioned ID, so a pinned +version works without being declared: GoModel routes a model it does not list +to a `jev` provider when the name says which one (`jev/jev-1.13.0`), or, for a +bare name, when exactly one `jev` provider is configured. A virtual model can +pin a version the same way. + +OpenRouter accepts `jev-latest` itself, but GoModel routes on its catalog IDs. +To keep a plain `jev-latest` (the TypeSafe SDKs' default) working through +OpenRouter, add a virtual model: + +```yaml +virtual_models: + - source: jev-latest + target: openrouter/~typesafe/jev-latest +``` + +## Caching + +With the [response cache](/features/cache) enabled, an identical request (same +route, resolved model, guardrails, and body after guardrail edits) is answered +from the exact cache (`X-Cache: HIT (exact)`) and recorded in usage as a cache +hit. The semantic cache never serves System One: a state that is merely +similar is not the same decision. Send `Cache-Control: no-cache` to skip the +cache for one request. + +## Failover + +A virtual model with the `failover` strategy moves a request to its next +target when the current one fails with an availability error (`429` or `5xx`, +including TypeSafe's `529`, by default; see [Failover](/features/failover)). +Every target receives the request in its own System One form. A target without +the API, such as a chat model, is skipped without using a failover attempt, +and client errors such as a malformed question (`422`) are returned without +failover: + +```yaml +virtual_models: + - source: decider + strategy: failover + targets: + - { model: kev/kev-latest } # local Kev first + - { model: openrouter/typesafe/jev-1.13 } # hosted Jev when Kev is down +``` + +The audit log shows each attempt, usage is recorded under the target that +answered, and a failover answer is not cached. + +## Guardrails + +Guardrails see `state` as a single user message: a string state as its text, +any other JSON value as its encoded JSON, which must still be valid JSON after +an edit. That is what anonymizing and blocking guardrails need; for example, a +`string_replace` rule that masks card numbers applies to `state` before it +leaves the gateway. The questions are your application's fixed schema and are +not exposed. + +Edits a decision request has no place for, such as a system prompt injected by +a guardrail that also covers chat models, are dropped. The gateway logs one +warning per kind of dropped edit, then logs repeats at debug level. A +guardrail that would answer the request itself blocks it instead, since System +One callers expect typed answers, not text. + +## Errors and misuse + +The endpoint never translates, and it says so when a request cannot work: + +| Situation | Result | +| --- | --- | +| No `jev` or `openrouter` provider configured | `404` | +| `model` missing | `400` | +| Model on a provider without System One, or a chat, embedding, or other generation model (including through a virtual model) | `400 invalid_request_error` explaining why; the gateway logs a warning | +| A System One model sent to `/v1/chat/completions`, `/v1/responses`, or `/v1/embeddings` | `400 invalid_request_error` pointing at `/v1/systemone` | +| Upstream error, such as a malformed question | The provider's status, with its message | + +## Audit, usage, and cost + +Each call is an audit entry under its route, with the requested and resolved +model, provider, request and response bodies, guardrail outcomes, and failover +attempts; filter the audit log by the **System One** request type. Usage +records the answer's `input_tokens` and `output_tokens` under the model that +answered. OpenRouter reports its own `usage.cost`, which is recorded as the +request's cost; for hosted Jev, declare pricing on the provider (see +[Jev / Kev](/providers/jev#models-access-control-and-cost)). diff --git a/docs/docs.json b/docs/docs.json index 171286377..511eb42e5 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -114,7 +114,8 @@ "pro/overview", "pro/compression", "pro/sso", - "pro/intelligent-routing" + "pro/intelligent-routing", + "pro/egress" ] }, { @@ -142,6 +143,7 @@ "advanced/responses-compatibility", "advanced/conversations-api", "advanced/anthropic-messages-api", + "advanced/systemone-api", "advanced/extra-content", "advanced/audio-api", "advanced/images-api", diff --git a/docs/features/cache.mdx b/docs/features/cache.mdx index 75bd9f4d3..512773685 100644 --- a/docs/features/cache.mdx +++ b/docs/features/cache.mdx @@ -14,6 +14,7 @@ requests on: - `/v1/responses` - `/v1/messages` - `/v1/embeddings` +- `/v1/systemone` (and Kev's `/permute` and `/separate`) Streaming and non-streaming variants of the same request are cached independently: a streaming miss stores the raw SSE bytes and a streaming hit @@ -37,7 +38,8 @@ X-Cache: HIT (semantic) represent the exact text it was requested for, so replaying the vector of a merely similar input would be a wrong answer rather than an equivalent one. Embeddings requests are also never streamed, so only the JSON response is - cached. + cached. The same holds for [System One](/advanced/systemone-api#caching) + decisions: a similar state is not the same decision. ## Enable the exact cache diff --git a/docs/features/mcp-gateway.mdx b/docs/features/mcp-gateway.mdx index e65f3ee51..0d6cca447 100644 --- a/docs/features/mcp-gateway.mdx +++ b/docs/features/mcp-gateway.mdx @@ -17,8 +17,8 @@ the same way it already sits between your apps and model providers: - **Aggregation with namespacing.** Tools and prompts from every configured server appear in one catalog as `{slug}_{name}` (for example `github_create_issue`), in a stable, deterministic order. -- **Least-privilege discovery.** A server can be scoped to `user_paths`; - callers outside the subtree do not merely get errors — the tools never +- **Least-privilege discovery.** A server can be scoped to `user_paths` and + carved out with `disallowed_user_paths`; hidden callers do not merely get errors — the tools never appear in their `tools/list` at all. Operator-level `allowed_tools` / `disallowed_tools` filters trim noisy servers, which keeps agent context small. @@ -82,8 +82,10 @@ Open **MCP Servers** and click **Add MCP Server**. Fill in: - **URL** — the upstream MCP endpoint. - **Headers** — upstream credentials (for example `Authorization: Bearer …`); saved values are shown redacted afterward. -- Optional **allowed/disallowed tools** and **user paths** to scope - visibility, the same fields available in `config.yaml`. +- **Tools** — check the tools clients may use; see + [Choose which tools each server exposes](#choose-which-tools-each-server-exposes). +- Optional **user paths** and **excluded user paths** under **Advanced**; see + [Limit who can see a server](#limit-who-can-see-a-server). Save to connect immediately. Use the row's refresh icon to force a **Reconnect**, the list icon to open the catalog inspector, and the pencil @@ -110,6 +112,7 @@ mcp: headers: Authorization: "Bearer ${GITHUB_PAT}" user_paths: ["/engineering"] # optional visibility scope + disallowed_user_paths: ["/engineering/contractors"] # carve-out; wins over user_paths disallowed_tools: ["delete_repo"] local-files: transport: stdio # declarative-only; see security notes @@ -135,6 +138,7 @@ Per-server fields: | `allowed_tools` | all | Allowlist of upstream tool names | | `disallowed_tools` | none | Blocklist, applied after the allowlist | | `user_paths` | everyone | Visibility subtrees, like virtual models | +| `disallowed_user_paths` | none | Subtrees that never see the server; wins over `user_paths` | | `tool_timeout` | `30s` | Upper bound for one `tools/call` | Invalid declarations (bad slug, missing `url`/`command`, unknown transport) @@ -224,14 +228,78 @@ A last error mentioning `x509: certificate signed by unknown authority` means the server's TLS certificate is issued by a CA the container does not trust. See [private CA certificates](/guides/production#upstream-certificates-behind-a-private-ca). +## Choose which tools each server exposes + +By default every tool a server reports is exposed. Trim a server's tools to +keep agent context small and to keep risky tools, such as deletes or merges, +away from clients. + +In the dashboard, edit a server and use its **Tools** section. It lists every +tool the server reports, with a checkbox for each, a filter box, and +**Expose all** / **Exclude all** for the filtered rows. Tools the server +annotates as `read-only` or `destructive` carry a badge. These hints come from +the upstream; GoModel does not verify them. + +**Tools the server adds later** controls what happens when the upstream grows: + +| Choice | Saved as | New upstream tools | +| --- | --- | --- | +| **Expose automatically** (default) | `disallowed_tools`: the unchecked tools | Exposed | +| **Keep hidden** | `allowed_tools`: the checked tools | Hidden until you check them | + +Choose **Keep hidden** for servers you do not control, so a new tool can never +reach clients without review. + +You can also add a tool by name, for example before the server first +connects. Names the server does not report stay in the list, marked +**Not reported by the server**, so typos and removed tools stay visible. + +Changing tool filters applies immediately without reconnecting to the server. +The change also covers MCP sessions that are already open: an excluded tool +fails with an error even for a client that listed it before the change. + +In `config.yaml` the same filters use original, unprefixed tool names: + +```yaml +mcp: + servers: + github: + url: https://api.githubcopilot.com/mcp + disallowed_tools: ["delete_repository", "merge_pull_request"] +``` + +## Limit who can see a server + +`user_paths` lists who may see a server; `disallowed_user_paths` carves callers +out. Both match subtrees, so `/contractors` also covers `/contractors/acme`, +and an exclusion wins over an allowed path. + +| Goal | Configuration | +| --- | --- | +| Everyone | leave both empty (default) | +| Only engineering | `user_paths: ["/engineering"]` | +| Everyone except contractors | `disallowed_user_paths: ["/contractors"]` | +| Engineering except its contractors | both of the above, scoped under `/engineering` | + +The caller's user path comes from its API key when the key has one. Otherwise, +including with the master key, it comes from the `X-GoModel-User-Path` header +(or the header named by `USER_PATH_HEADER`), which is a quick way to check what +a given subtree sees. + +Hidden servers are left out of `tools/list`, and `/mcp/{slug}` returns 404 for +them. Like tool filters, visibility edits apply without reconnecting to the +server and are checked on every call, so they also reach MCP sessions that are +already open. + ## Inspect a server's catalog The dashboard's catalog inspector (or -`GET /admin/mcp-servers/{slug}/catalog`) lists exactly what a server -currently exposes through the gateway — tools, prompts, resources, and -resource templates with their descriptions, after your `allowed_tools` / -`disallowed_tools` filters. Names are the upstream originals; the aggregated -`/mcp` endpoint serves them as `{slug}_{name}`. +`GET /admin/mcp-servers/{slug}/catalog`) lists what a server currently exposes +through the gateway: tools, prompts, resources, and resource templates with +their descriptions, after your `allowed_tools` / `disallowed_tools` filters. +Tools that the server reports but the filters hide are listed separately under +`excluded_tools`. Names are the upstream originals; the aggregated `/mcp` +endpoint serves them as `{slug}_{name}`. ## Security notes diff --git a/docs/openapi.json b/docs/openapi.json index 4d8e196fa..a6cb55a95 100644 --- a/docs/openapi.json +++ b/docs/openapi.json @@ -330,6 +330,14 @@ "type": "boolean" } }, + { + "description": "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay", + "name": "exclude_operation", + "in": "query", + "schema": { + "type": "string" + } + }, { "description": "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message", "name": "search", @@ -507,6 +515,14 @@ "type": "boolean" } }, + { + "description": "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay", + "name": "exclude_operation", + "in": "query", + "schema": { + "type": "string" + } + }, { "description": "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message", "name": "search", @@ -1596,7 +1612,7 @@ }, "/admin/mcp-servers/{name}/catalog": { "get": { - "description": "Lists the tools, prompts, resources, and resource templates the named server currently exposes through the gateway, after operator tool filters. Names are the upstream originals; the aggregated /mcp endpoint prefixes them with the server slug.", + "description": "Lists the tools, prompts, resources, and resource templates the named server currently exposes through the gateway, after operator tool filters. Discovered tools the filters hide are listed separately under excluded_tools. Names are the upstream originals; the aggregated /mcp endpoint prefixes them with the server slug.", "tags": [ "admin" ], @@ -11177,6 +11193,272 @@ } } }, + "/v1/systemone": { + "post": { + "description": "Available when a jev or openrouter provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", + "tags": [ + "systemone" + ], + "summary": "Evaluate a System One decision request (Jev / Kev)", + "requestBody": { + "$ref": "#/components/requestBodies/Request2" + }, + "responses": { + "200": { + "description": "System One answers, in the provider's shape", + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + } + }, + "400": { + "description": "Bad Request", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "401": { + "description": "Unauthorized", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "404": { + "description": "Not Found", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "429": { + "description": "Too Many Requests", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "502": { + "description": "Bad Gateway", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ], + "x-mint": { + "metadata": { + "sidebarTitle": "/v1/systemone", + "title": "Evaluate a System One decision request (Jev / Kev)", + "description": "GoModel API reference for POST /v1/systemone: Evaluate a System One decision request (Jev / Kev)." + } + } + } + }, + "/v1/systemone/permute": { + "post": { + "description": "A Kev server diagnostic: the request is a System One request, and n_perm (1 to 64, default 6) sets how many option orders run. Only jev providers pointing at a Kev server serve it.", + "tags": [ + "systemone" + ], + "summary": "Run one Choice question with several option orders (Kev)", + "requestBody": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + }, + "description": "System One request with one Choice question", + "required": true + }, + "responses": { + "200": { + "description": "Kev's answer, in the provider's shape", + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + } + }, + "400": { + "description": "Bad Request", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "401": { + "description": "Unauthorized", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "404": { + "description": "Not Found", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "429": { + "description": "Too Many Requests", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "502": { + "description": "Bad Gateway", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ], + "x-mint": { + "metadata": { + "sidebarTitle": "/v1/systemone/permute", + "title": "Run one Choice question with several option orders (Kev)", + "description": "GoModel API reference for POST /v1/systemone/permute: Run one Choice question with several option orders (Kev)." + } + } + } + }, + "/v1/systemone/separate": { + "post": { + "description": "A Kev server diagnostic that answers each question separately. Only jev providers pointing at a Kev server serve it.", + "tags": [ + "systemone" + ], + "summary": "Run each System One question in its own forward pass (Kev)", + "requestBody": { + "$ref": "#/components/requestBodies/Request2" + }, + "responses": { + "200": { + "description": "Kev's answer, in the provider's shape", + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + } + }, + "400": { + "description": "Bad Request", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "401": { + "description": "Unauthorized", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "404": { + "description": "Not Found", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "429": { + "description": "Too Many Requests", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "502": { + "description": "Bad Gateway", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ], + "x-mint": { + "metadata": { + "sidebarTitle": "/v1/systemone/separate", + "title": "Run each System One question in its own forward pass (Kev)", + "description": "GoModel API reference for POST /v1/systemone/separate: Run each System One question in its own forward pass (Kev)." + } + } + } + }, "/v1/usage": { "get": { "description": "Returns recorded usage, budget statuses, and rate limit statuses for the caller's effective user path (the path bound to the managed API key, or the user-path header for master-key callers).", @@ -11342,6 +11624,17 @@ }, "description": "Anthropic Messages request", "required": true + }, + "Request2": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + }, + "description": "System One request: model, state, and questions", + "required": true } }, "securitySchemes": { @@ -11770,9 +12063,18 @@ "type": "string" } }, + "disallowed_user_paths": { + "type": "array", + "items": { + "type": "string" + } + }, "enabled": { "type": "boolean" }, + "excluded_tool_count": { + "type": "integer" + }, "headers": { "type": "object", "additionalProperties": { @@ -12269,6 +12571,12 @@ "type": "string" } }, + "disallowed_user_paths": { + "type": "array", + "items": { + "type": "string" + } + }, "enabled": { "type": "boolean" }, @@ -12826,6 +13134,9 @@ "name": { "type": "string" }, + "strict": { + "type": "boolean" + }, "type": { "type": "string" } @@ -15146,8 +15457,14 @@ "description": { "type": "string" }, + "destructive": { + "type": "boolean" + }, "name": { "type": "string" + }, + "read_only": { + "type": "boolean" } } }, @@ -15182,6 +15499,12 @@ "mcpgateway.CatalogView": { "type": "object", "properties": { + "excluded_tools": { + "type": "array", + "items": { + "$ref": "#/components/schemas/mcpgateway.CatalogFeature" + } + }, "instructions": { "type": "string" }, diff --git a/docs/pro/egress.mdx b/docs/pro/egress.mdx new file mode 100644 index 000000000..f53008704 --- /dev/null +++ b/docs/pro/egress.mdx @@ -0,0 +1,118 @@ +--- +title: "Egress Proxy Pools" +sidebarTitle: "Egress proxy pools" +description: "Send provider traffic through named pools of HTTP or SOCKS5 proxies with health checks, failover, and rules that assign providers to pools." +icon: "route" +tag: "Pro" +keywords: ["GoModel Pro", "egress", "proxy pool", "SOCKS5", "HTTP proxy", "failover", "static IP", "health check"] +--- + +## Overview + +Open-source GoModel sends one provider through one proxy with +[`proxy_url`](/providers/outbound-proxy). That is enough for a single +geo-restricted provider, but not for a fleet that must always leave through +an allowlisted egress IP: when that one proxy goes down, every request behind +it fails. + +Egress proxy pools add the missing pieces: + +- **Pools** of HTTP, HTTPS, SOCKS5, or SOCKS5h proxies, picked by `failover` + (first healthy, in order) or `round_robin`. +- **Health checks** that probe every proxy on a schedule and skip members that + fail, then bring them back once they pass again. +- **Rules** that assign providers to pools by name pattern or provider type, + so a new provider is covered without touching its own configuration. + +A provider's own `proxy_url` still wins over any rule. Providers that no rule +matches keep the gateway-wide `HTTP_PROXY`, `HTTPS_PROXY`, and `NO_PROXY` +behaviour. + + + Like SSO, enabled egress fails closed: startup aborts when the `egress` + entitlement is missing or the configuration is invalid. Traffic that was + meant to leave through a proxy never silently leaves from the gateway's own + address instead. + + +## Configure pools + +Configure them under `extensions.egress` in the main GoModel YAML +configuration: + +```yaml +extensions: + egress: + enabled: true + proxies: + eu: + urls: + - socks5://user:${EU_PROXY_PASSWORD}@10.0.0.1:1080 + - socks5://user:${EU_PROXY_PASSWORD}@10.0.0.2:1080 + strategy: failover + check_url: https://api.openai.com/v1/models + us: + urls: [http://egress-us.internal:3128] + rules: + - providers: ["openai*", "type:anthropic"] + proxy: eu + - providers: ["*"] + proxy: us +``` + +Rules are evaluated in order and the first match wins. A pattern matches the +provider name (`openai-eu`, `openai*`); with a `type:` prefix it matches the +provider type instead (`type:gemini`). With a single pool and no rules every +provider uses that pool. + +The same settings are available as environment variables, which override the +YAML values. One pool per `PRO_EGRESS_PROXY_`; the name becomes the pool +name in lowercase with underscores as hyphens (`EU_STATIC` becomes +`eu-static`): + +```bash +PRO_EGRESS_ENABLED=true +PRO_EGRESS_PROXY_EU=socks5://user:pass@10.0.0.1:1080,socks5://user:pass@10.0.0.2:1080 +PRO_EGRESS_PROXY_EU_STRATEGY=failover +PRO_EGRESS_PROXY_EU_CHECK_URL=https://api.openai.com/v1/models +PRO_EGRESS_PROXY_US=http://egress-us.internal:3128 +PRO_EGRESS_RULES=openai*=eu,type:anthropic=eu,*=us +``` + +| Setting | Default | Change it when | +| --- | --- | --- | +| `strategy` | `failover` | Use `round_robin` to spread load across members instead of preferring the first. | +| `check_url` | unset (TCP dial) | Set a URL to prove the proxy can actually reach the internet, not just accept connections. | +| `health_check_interval` | `30s` | Lower it when a dead proxy must be noticed faster than half a minute. | +| `health_check_timeout` | `5s` | Raise it for slow proxies far from the gateway. | +| `failure_threshold` | `2` | Consecutive failed probes before a member is skipped. Raise it on flaky links. | + +## Health checks and failover + +Every member is probed before the gateway serves its first request, so a +proxy that is already down never receives traffic. Without `check_url` a probe +is a TCP connection to the proxy port. With `check_url` it is a `HEAD` request +through the proxy; any relayed response counts as healthy, including `401` or +`404` from the target, while a `5xx` counts as failure because forward proxies +answer `502`, `503`, and `504` themselves when they cannot reach the target. + +After the first probe a member needs `failure_threshold` consecutive failures +to be skipped and one success to return. When every member of a pool is +unhealthy the pool is still used, starting with its first member, and a +warning is logged once per outage. A matched pool never falls back to a direct +connection. + +## Inspect + +Health transitions are logged with the pool name, member position, and the +proxy URL with any password masked. Prometheus exposes: + +- `gomodel_pro_egress_proxy_healthy{pool,member,proxy}`: `1` while the member + is in rotation. +- `gomodel_pro_egress_health_checks_total{pool,member,proxy,result}`: probes + by outcome. +- `gomodel_pro_egress_selections_total{pool,member,proxy}`: requests each + member was chosen for. + +Alert on `gomodel_pro_egress_proxy_healthy == 0` to learn about a dead proxy +before the pool runs out of members. diff --git a/docs/pro/overview.mdx b/docs/pro/overview.mdx index a44851951..a11eeba3b 100644 --- a/docs/pro/overview.mdx +++ b/docs/pro/overview.mdx @@ -1,10 +1,10 @@ --- title: "GoModel Pro" sidebarTitle: "Overview" -description: "GoModel Pro adds licensed prompt compression, intelligent routing, and OIDC SSO to the GoModel gateway." +description: "GoModel Pro adds licensed prompt compression, intelligent routing, OIDC SSO, and egress proxy pools to the GoModel gateway." icon: "gem" tag: "Pro" -keywords: ["GoModel Pro", "prompt compression", "intelligent routing", "OIDC", "SSO", "quota templates", "license"] +keywords: ["GoModel Pro", "prompt compression", "intelligent routing", "OIDC", "SSO", "quota templates", "egress proxy pools", "license"] --- ## What Pro is @@ -12,14 +12,15 @@ keywords: ["GoModel Pro", "prompt compression", "intelligent routing", "OIDC", " GoModel Pro is the commercial distribution of GoModel. It uses the same gateway, configuration, providers, dashboard, and APIs as open-source GoModel, with licensed extensions for prompt compression, intelligent routing, -OIDC single sign-on, and per-child quota templates. +OIDC single sign-on, egress proxy pools, and per-child quota templates. Without a valid license, the Pro binary starts as the open-source gateway and -does not block traffic. An explicitly enabled SSO configuration is the -exception: startup fails closed when its SSO entitlement or configuration is -invalid, so an authentication boundary cannot disappear silently. +does not block traffic. Explicitly enabled SSO or egress pools are the +exception: startup fails closed when their entitlement or configuration is +invalid, so an authentication boundary cannot disappear silently and proxied +traffic never leaves from the gateway's own address. - + Deterministic, fail-open request rewriting that removes repeated context before it reaches the provider. @@ -32,6 +33,10 @@ invalid, so an authentication boundary cannot disappear silently. Easy/hard tier classification per request plus adaptive, health-aware provider selection inside a virtual model. Beta. + + Health-checked pools of HTTP or SOCKS5 proxies with failover and rules + that assign providers to pools. + ## Quota templates diff --git a/docs/providers/anthropic.mdx b/docs/providers/anthropic.mdx index 6b9d9b885..a51b63d0f 100644 --- a/docs/providers/anthropic.mdx +++ b/docs/providers/anthropic.mdx @@ -233,6 +233,17 @@ the completion before parsing `message.content`: Only `finish_reason: "stop"` with non-empty content is worth handing to a JSON parser; treat anything else as an error rather than parsing it. +## Developer messages and strict tools + +`developer` messages are OpenAI's newer name for `system`; GoModel sends them to +Anthropic as system content, following the same placement rules as `system` +messages. + +A function tool with `"strict": true` is forwarded as an Anthropic strict tool. +Strict tool schemas are held to the same limits as structured output, so GoModel +adapts them with the same rules as in [Structured output](#structured-output). +Tools without `strict` keep their schema exactly as sent. + ## Verbosity OpenAI's `verbosity` (and `text.verbosity` on `/v1/responses`) has no Anthropic diff --git a/docs/providers/gemini.mdx b/docs/providers/gemini.mdx index 8402420f3..d432cf57a 100644 --- a/docs/providers/gemini.mdx +++ b/docs/providers/gemini.mdx @@ -162,6 +162,34 @@ and Cloud Storage URIs are sent as a file reference Gemini resolves itself. Any other remote `https://...` URL is rejected rather than dropped — upload it through the Gemini Files API first. +## Messages and tools + +In native mode, OpenAI chat fields map onto `generateContent` like this: + +| OpenAI field | Gemini `generateContent` | +| ------------ | ------------------------ | +| `system` and `developer` messages | `system_instruction` (Gemini has no separate developer role) | +| `tool_choice: "auto"` / `"required"` / `"none"` | `functionCallingConfig.mode` `AUTO` / `ANY` / `NONE` | +| `tool_choice` naming one function | `ANY` with `allowedFunctionNames: [name]` | +| `tool_choice: {"type": "allowed_tools", ...}` with mode `required` | `ANY` with `allowedFunctionNames` | +| `tool_choice: {"type": "allowed_tools", ...}` with mode `auto` | `VALIDATED` with `allowedFunctionNames` | +| `strict: true` on any function tool | `VALIDATED` instead of `AUTO`; `ANY` already enforces the schema | +| `parallel_tool_calls: false` | dropped; Gemini has no equivalent | + +Gemini accepts `allowedFunctionNames` only with `ANY` or `VALIDATED`, which is +why an `auto` subset becomes `VALIDATED`: the model can still answer in text, +but any call it makes is limited to the listed tools. `VALIDATED` applies to +the whole request, so one strict tool makes every declared tool +schema-validated. + +Gemini has no way to limit a turn to one function call, and naming a single +tool does not help: Gemini can call the same function more than once in a turn. +When `parallel_tool_calls: false` (or Anthropic's `disable_parallel_tool_use`) +matters, handle or reject the extra calls in your client. + +An `allowed_tools` choice with an empty `tools` list is rejected with a 400, as +OpenAI does. + ## Not yet integrated - Automatic fetching of remote `image_url` values in native mode. diff --git a/docs/providers/jev.mdx b/docs/providers/jev.mdx index c2c2b16c0..c983f5485 100644 --- a/docs/providers/jev.mdx +++ b/docs/providers/jev.mdx @@ -2,7 +2,7 @@ title: "Jev / Kev (TypeSafe System One)" sidebarTitle: "Jev / Kev" description: "Route TypeSafe System One decision requests through GoModel, to the hosted Jev API or a self-hosted Kev server." -icon: "scale-balanced" +icon: "scale" keywords: ["Jev", "Kev", "TypeSafe", "System One", "decision model", "classification", "noul", "choice", "score", "self-hosted"] --- @@ -22,10 +22,11 @@ There are three question types: | `score` | Rate against ordered levels | `score`, plus `legend`, `probabilities` and `confidence` | The API is not OpenAI-compatible, and its answers have no chat equivalent, so -GoModel does not translate it: System One requests go through -[passthrough](/features/passthrough-api) at `/p/jev/...`, which is enabled by -default for this provider. Chat, `/responses`, and `/v1/embeddings` return -`invalid_request_error` for `jev` models. +GoModel forwards it natively instead of translating it: `POST /v1/systemone` +is available as soon as a `jev` or `openrouter` provider is configured, and +[passthrough](/features/passthrough-api) at `/p/jev/...` reaches every other +upstream route. Chat, `/responses`, and `/v1/embeddings` return +`invalid_request_error` for `jev` models, pointing at `/v1/systemone`. ## Configure @@ -49,13 +50,15 @@ GOMODEL_MASTER_KEY=change-me use; a trailing `/v1` is accepted and trimmed, so both spellings address the same server. To run the hosted API and a local Kev side by side, register the second under a suffixed name: `JEV_KEV_BASE_URL=...` creates provider - `jev-kev`, reached at `/p/jev-kev/...`. + `jev-kev`, reached at `/p/jev-kev/...`. In `config.yaml`, any name works, + such as `kev: {type: jev, base_url: ...}`, and that name is what logs, + usage, and model prefixes show. ## Verify ```bash -curl -s http://localhost:8080/p/jev/v1/systemone \ +curl -s http://localhost:8080/v1/systemone \ -H "Authorization: Bearer change-me" \ -H "Content-Type: application/json" \ -d '{ @@ -88,19 +91,21 @@ curl -s http://localhost:8080/p/jev/v1/systemone \ } ``` -The `/v1` segment is optional: `/p/jev/systemone` is the same route. Use -`kev-latest` as the model on a Kev server; it also answers to `jev-latest`. +Use `kev-latest` as the model on a Kev server; it also answers to +`jev-latest`. The same request works at `/p/jev/v1/systemone`, but that +passthrough route skips virtual models and guardrails; see +[the native endpoint](#the-native-endpoint). ## Using the TypeSafe SDKs -The SDKs send `POST {base_url}/v1/systemone`, so point them at the provider's -passthrough root and authenticate with your GoModel key: +The SDKs send `POST {base_url}/v1/systemone`, so point them at the gateway +itself and authenticate with your GoModel key: ```python Python from typesafe_sdk import Noul, TypeSafeClient -client = TypeSafeClient(api_key="change-me", base_url="http://localhost:8080/p/jev") +client = TypeSafeClient(api_key="change-me", base_url="http://localhost:8080") response = client.system_one( state="I was charged twice. Please fix this ASAP.", questions={"billing": Noul(instructions="Is this ticket about billing?")}, @@ -111,7 +116,7 @@ print(response.nouls["billing"].noul) ```typescript JavaScript import { TypeSafeClient, noul } from "@typesafe-ai/sdk"; -const client = new TypeSafeClient({ apiKey: "change-me", baseURL: "http://localhost:8080/p/jev" }); +const client = new TypeSafeClient({ apiKey: "change-me", baseURL: "http://localhost:8080" }); const result = await client.systemOne({ state: "I was charged twice. Please fix this ASAP.", questions: { billing: noul({ instructions: "Is this ticket about billing?" }) }, @@ -120,17 +125,30 @@ console.log(result.answers.billing.noul); ``` -The same works with `TYPESAFE_BASE_URL=http://localhost:8080/p/jev` and -`TYPESAFE_API_KEY=change-me` in the environment. +The same works with `TYPESAFE_BASE_URL=http://localhost:8080` and +`TYPESAFE_API_KEY=change-me` in the environment. With several System One +providers, name the model with its provider (`kev/kev-latest`) or a +[virtual model](/features/virtual-models). +The SDKs' model listing expects TypeSafe's shape, while the gateway's +`/v1/models` is OpenAI-shaped; list upstream models at `/p/jev/v1/models`. -## Native routes +## System One API + +`POST /v1/systemone` is a gateway endpoint, not a raw proxy: it applies +virtual models, guardrails on `state`, the response cache, failover, audit, +and usage, and it forwards the request natively without translating it. Kev's +`/v1/systemone/permute` and `/v1/systemone/separate` work the same way. See +[System One API](/advanced/systemone-api) for the full behavior, including +OpenRouter, which serves Jev natively too. + +Every other upstream route is reachable through +[passthrough](/features/passthrough-api), without virtual models, guardrails, +or caching: | Route | What it does | | --- | --- | -| `POST /p/jev/v1/systemone` | Evaluate a state against a map of questions | | `GET /p/jev/v1/models` | The names the `model` field accepts, in the upstream's own shape | -| `POST /p/jev/v1/systemone/permute` | Kev only: run one Choice question with several option orders | -| `POST /p/jev/v1/systemone/separate` | Kev only: run each question in its own forward pass | +| `POST /p/jev/v1/systemone` | The evaluation route, forwarded as sent | Upstream errors keep their status code, with the provider's body carried in the gateway error message: a malformed question comes back as TypeSafe's `422` @@ -145,9 +163,12 @@ IDs such as `jev-1.13.0` are accepted by the `model` field whether or not they are listed. The models are categorized as utility models with no generation mode, since there is no OpenAI endpoint to route them to. -Every System One request names its model, so the passthrough surface applies -the caller's [model allowlist](/features/users) to it like any other -request. +Every System One request names its model, so both `/v1/systemone` and the +passthrough surface apply the caller's [model allowlist](/features/users) to +it like any other request. + +A pinned version such as `jev-1.13.0` works without being declared; see +[Models](/advanced/systemone-api#models) for how unlisted names are routed. The response's `usage.input_tokens` and `usage.output_tokens` are recorded, so System One calls appear in the usage API and dashboard under the model that diff --git a/docs/providers/kimicode.mdx b/docs/providers/kimicode.mdx index 214ccff17..cb91f2358 100644 --- a/docs/providers/kimicode.mdx +++ b/docs/providers/kimicode.mdx @@ -7,8 +7,17 @@ keywords: ["Kimi Code", "Moonshot", "quota", "provider setup"] Kimi Code is an OpenAI-compatible coding assistant served at `https://api.kimi.com/coding/v1`. GoModel routes chat, model listing, embeddings, and passthrough requests through the shared -OpenAI adapter. The `/v1/responses` endpoint is translated through chat completions, while -files and batches are not supported by the upstream endpoint. +OpenAI adapter. The `/v1/responses` endpoint is forwarded natively to the upstream Responses +API instead of being translated through chat completions, while files and batches are not +supported by the upstream endpoint. + +Kimi Code retains no responses. Requests with `store: true` are rewritten to `store: false` +(the upstream rejects `store: true` with a 400). Chaining works only through GoModel: with a +response store configured, the gateway expands a `previous_response_id` chain by replaying the +stored history into the request before dispatch, and a `conversation` reference resolves +through the conversation store the same way. Without those stores, a request carrying +`previous_response_id` or `conversation` is rejected with an invalid-request error, because +the upstream can never resolve the referenced state. ## Configure diff --git a/docs/providers/outbound-proxy.mdx b/docs/providers/outbound-proxy.mdx index b0b8dc89d..0dc69b793 100644 --- a/docs/providers/outbound-proxy.mdx +++ b/docs/providers/outbound-proxy.mdx @@ -91,6 +91,6 @@ nothing reaches the provider from the gateway's own IP. Pooled proxies with health checks, automatic failover between proxies, and - assignment rules that cover many providers at once are on the - [GoModel Pro roadmap](/about/roadmap). + assignment rules that cover many providers at once are a + [GoModel Pro feature](/pro/egress). diff --git a/docs/providers/overview.mdx b/docs/providers/overview.mdx index 055ac06c8..d8ea5d3bb 100644 --- a/docs/providers/overview.mdx +++ b/docs/providers/overview.mdx @@ -211,10 +211,14 @@ support, not every individual model capability exposed by an upstream provider. through passthrough. See [audio.cpp](/providers/audiocpp). - **Jev / Kev** — TypeSafe's System One API is a decision API (state plus typed questions in, calibrated probabilities out) with no OpenAI-compatible - surface, so it is reached only through passthrough at - `POST /p/jev/v1/systemone`. `JEV_API_KEY` configures the hosted API; for a - self-hosted Kev server, which speaks the same API without authentication, - set `JEV_BASE_URL` and leave the key unset. See [Jev](/providers/jev). + surface, so GoModel forwards it natively, untranslated, at + `POST /v1/systemone` (with virtual models, guardrails, audit, and usage) and + through passthrough at `/p/jev/...`. The endpoint is available once a `jev` + or `openrouter` provider is configured; OpenRouter serves Jev natively, and + its decision models are listed as utility models. + `JEV_API_KEY` configures the hosted API; for a self-hosted Kev server, which + speaks the same API without authentication, set `JEV_BASE_URL` and leave the + key unset. See [Jev](/providers/jev). - **llama.cpp / LM Studio** — `LLAMACPP_BASE_URL` is required (llama-server's default port collides with GoModel's own 8080, so there is no default); `LLAMACPP_API_KEY` is optional. Do not register these servers as `ollama`, diff --git a/go.mod b/go.mod index f7b55650f..b42e17b0b 100644 --- a/go.mod +++ b/go.mod @@ -8,10 +8,10 @@ go 1.27.1 retract [v0.1.52, v0.1.79] require ( - github.com/aws/aws-sdk-go-v2 v1.47.0 - github.com/aws/aws-sdk-go-v2/config v1.33.5 - github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.0 - github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.0 + github.com/aws/aws-sdk-go-v2 v1.47.1 + github.com/aws/aws-sdk-go-v2/config v1.33.6 + github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.1 + github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.1 github.com/cespare/xxhash/v2 v2.3.0 github.com/coder/websocket v1.8.15 github.com/goccy/go-json v0.10.6 @@ -57,17 +57,17 @@ require ( cloud.google.com/go/compute/metadata v0.9.0 // indirect github.com/KyleBanks/depth v1.2.1 // indirect github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 // indirect - github.com/aws/aws-sdk-go-v2/credentials v1.20.5 // indirect - github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 // indirect - github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 // indirect - github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 // indirect - github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 // indirect + github.com/aws/aws-sdk-go-v2/credentials v1.20.6 // indirect + github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1 // indirect + github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4 // indirect + github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4 // indirect + github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4 // indirect github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 // indirect - github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 // indirect - github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 // indirect - github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 // indirect - github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 // indirect - github.com/aws/aws-sdk-go-v2/service/sts v1.51.0 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4 // indirect + github.com/aws/aws-sdk-go-v2/service/signin v1.10.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sso v1.38.1 // indirect + github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sts v1.51.1 // indirect github.com/aws/smithy-go v1.28.1 // indirect github.com/beorn7/perks v1.0.1 // indirect github.com/cenkalti/backoff/v5 v5.0.3 // indirect diff --git a/go.sum b/go.sum index 4fd97edbb..cf9708fbc 100644 --- a/go.sum +++ b/go.sum @@ -2,38 +2,38 @@ cloud.google.com/go/compute/metadata v0.9.0 h1:pDUj4QMoPejqq20dK0Pg2N4yG9zIkYGdB cloud.google.com/go/compute/metadata v0.9.0/go.mod h1:E0bWwX5wTnLPedCKqk3pJmVgCBSM6qQI1yTBdEb3C10= github.com/KyleBanks/depth v1.2.1 h1:5h8fQADFrWtarTdtDudMmGsC7GPbOAu6RVB3ffsVFHc= github.com/KyleBanks/depth v1.2.1/go.mod h1:jzSb9d0L43HxTQfT+oSA1EEp2q+ne2uh6XgeJcm8brE= -github.com/aws/aws-sdk-go-v2 v1.47.0 h1:0jsHallhJCeaU0Ko48c/3FK1ctOQ7NpzggxriJOQ8MQ= -github.com/aws/aws-sdk-go-v2 v1.47.0/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU= +github.com/aws/aws-sdk-go-v2 v1.47.1 h1:uOIZnp4PK3ZhKI0dNrJrhTEsLxbpXHTAJlwoS1pvAtw= +github.com/aws/aws-sdk-go-v2 v1.47.1/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU= github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 h1:GPRlPwz40I2B2VrBEASOA3Bi77NyeqejNLkifosX0rs= github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20/go.mod h1:g7PNzKcsOKWb4fkSRBA7BZVAS6Y8IcxzN+nRohhQ1Q8= -github.com/aws/aws-sdk-go-v2/config v1.33.5 h1:UA1dmokBFOLFoOyVBhO6HjM6edy0MIk5AZSkJVcksQw= -github.com/aws/aws-sdk-go-v2/config v1.33.5/go.mod h1:Dop8axzz0xx38GExIYWXdeyc8QQ7Cr+nPsxpD/LYy4U= -github.com/aws/aws-sdk-go-v2/credentials v1.20.5 h1:wklUVvHMc9xTQ3rcp49/ISpiMnhbCicJcA6n6S8m7J8= -github.com/aws/aws-sdk-go-v2/credentials v1.20.5/go.mod h1:fyEdrn6ccLFOkoK84j5bQyGTxp9zPt5l2XMhxf4DVZs= -github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 h1:AM4hHjww+PSFtt6E+UrBrPlZkWsePCLEt9AjkfQX+yM= -github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0/go.mod h1:3x/yXezeQjpOvBb4jEMxrS8SXvpdvJ5abv6l5c1gWM8= -github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 h1:Hp/VgjP0BysR3OgLlR057Vz2LcbbVnoWeJ+3qWiS/fY= -github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3/go.mod h1:nwGV5qw7F1IZPgxCvA/ph8N2TAuz+BkRG/bXn808qMA= -github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 h1:MUaM4f+kj1ZIBPZfUS8cxP1GKXXZtHJjAthy93AN7SM= -github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3/go.mod h1:6YmVmEVRI5ZZzRjCSsb9SryKH0hAlMRdgA7kG9aDvBU= -github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 h1:fuSCw4Z2qfRCztMPO3GXJNSiEp6Wee+WOLwrHHUMy9c= -github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3/go.mod h1:6SxcHheD1pPR5+kWm1wGvjlL/YqUsh267sAfEmN4K7A= -github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.0 h1:GiM/TNCIawTZvs0lLC3meQuTTD1dZxlo0BIH7xGR/AY= -github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.0/go.mod h1:tFtu0iACN2cRrgcRrrBdQRtKEGK0lRrmEI2yC13/Ygw= -github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.0 h1:0YDtf7baintcPG68AG0sMdAVoV1Iej6eXIBh38sd1m4= -github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.0/go.mod h1:7P+JIqwRqUIlzGn7Ba0qbSbPV/zsdZk7VkJTEFO6Z3U= +github.com/aws/aws-sdk-go-v2/config v1.33.6 h1:MBjkSTLczek/UgiK+EYPIoRTqE7gP8vtW3OFbFo7Nug= +github.com/aws/aws-sdk-go-v2/config v1.33.6/go.mod h1:grRAFzdAZJrwcbasJRg2MPvIrVjtlfXllHssN6+E1JE= +github.com/aws/aws-sdk-go-v2/credentials v1.20.6 h1:NpAFXCU7NzXNkdGK3zQTtsRJ+3v9tZQV0xcdRw8uBdw= +github.com/aws/aws-sdk-go-v2/credentials v1.20.6/go.mod h1:mcZCoiPnyMvP8VMNbygNX5lLqSlkYJIMPODylQMurOk= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1 h1:8gALAAmacnIXh+z6VkdDanv4/IkG5APdg4DZLDTmLog= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1/go.mod h1:Z7IJhJU+poOdJjUR2wpyY21ossQ1XS/R3Lk9Msq5kM4= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4 h1:CLq4+8UHCI+ZZYl/EuJxXovaIVN2xeeT8JV+dsApQ5E= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4/go.mod h1:Wv4q5sAM04xAMkoOedxLx2inVf6K5FdxYp+A61L+q/0= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4 h1:dD4MR81I7YkpEBRk6UP9rocC2QnT3qVuXwzlYTtfGEs= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4/go.mod h1:EcXV1kAFd5XwSkDHlj94gnF3q5CkJyYiIJfH8N0VmrE= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4 h1:7Wo47d/xn/7KttCSBd8EGYeZ7ULRFRkUHr6vkZPBzVQ= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4/go.mod h1:tDB2IVC1xC3vX8o+6uRlzhTxP3g1b77CZXFX/oD2FnQ= +github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.1 h1:Qz8Ptgep8qW6gOt28objap/3HaEYhjtkD0yP23Nxy4Q= +github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.1/go.mod h1:rTbux1fJj4skkaUl6R/cSdaf6R9/0UwNo8VEyolo+B0= +github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.1 h1:tVg987qhntW9rVFTYyVjU+HnIkrmXzOf7Tqw+Iq+398= +github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.1/go.mod h1:BHpwIwobMDKpDzoTnpdpGOp0rtfpFlAz6X/C2PpJTcA= github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 h1:bAdDl/HkGCcGPoe25ToSHEw23VIxt6CT5fLcg111BKg= github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19/go.mod h1:KaUzbLxv4CeSxh6ZCl9B4m7CuFenS8kUEaDs+f/DQr4= -github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 h1:bON1rJf67TSTDCKg816AAIE4xSTtoo9tl0XRkO72R+I= -github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3/go.mod h1:c5BBpjJcQXpfeq9iASyVKA3T6vX6B6LEXY4mL/gklDY= -github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 h1:ZD5qFpWcaOKdTuhBi431pIDkCgrMkMlMT6jlpSPoIRI= -github.com/aws/aws-sdk-go-v2/service/signin v1.10.0/go.mod h1:8Nuuf+tR346PjJ3MvZPh9pekbLiLQFWJhzMXfwy7alA= -github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 h1:JGeeBcMlhg1xtOXYpeCaTQBZObtXMPQCUqBcmr65NRA= -github.com/aws/aws-sdk-go-v2/service/sso v1.38.0/go.mod h1:XwteswG9EOMRFm73UT0t+MbTwyLxMrEXkU6e+v92Lzo= -github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 h1:obhahQXDEdVEv8y5bTKXR30LVaxYe1kyYM0L7l2Iq+k= -github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0/go.mod h1:6twZZ/aXHNy1vXUO8koUbp++MYzMASkOgEBdkbJYmO0= -github.com/aws/aws-sdk-go-v2/service/sts v1.51.0 h1:Zpnqa6XtrNzXZnwbdCqHOXpXhMsa01ql/pcRQ1sb4hk= -github.com/aws/aws-sdk-go-v2/service/sts v1.51.0/go.mod h1:/8JRcdTt//hG0Q4BTmGbuOplT7ABe+5rdtqUHqXvYIM= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4 h1:29SvnfGhXjTl8ONxFwbj2rs6lbhiFXD2CgFQmbT/bXY= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4/go.mod h1:wm04I5DMuNVvZHFe/dHnUxincvNbbK7AiNBbYsQivek= +github.com/aws/aws-sdk-go-v2/service/signin v1.10.1 h1:DzCCWLzcIRQ77F3DEUljud7bEjTgFOIKXP52NmVRyhU= +github.com/aws/aws-sdk-go-v2/service/signin v1.10.1/go.mod h1:xpo/geVldu8payT375WekctUzopG/hBU7miiqItMUlw= +github.com/aws/aws-sdk-go-v2/service/sso v1.38.1 h1:Umtl/0YZhng4xndfW3lKJrYYP7NLEjI6bGXVomwLcs0= +github.com/aws/aws-sdk-go-v2/service/sso v1.38.1/go.mod h1:rRD/dnm7q0HYE/I5TMaPgkWyyUGLcwuxHLABsLnQ3e0= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1 h1:orIWdNiLgzrhu/11RcPPKO/SBzUUymbUQuZbSPImghg= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1/go.mod h1:skwM/xsbR/1ReUTesv9BhpJp1VjajR7DWQnuVLwiXsQ= +github.com/aws/aws-sdk-go-v2/service/sts v1.51.1 h1:0HOqZXRvMytH6bFHVIc0oJX07sZjfhz0zXtjs6gdE8s= +github.com/aws/aws-sdk-go-v2/service/sts v1.51.1/go.mod h1:26zA0GhDrLo+yiLI2yXWxqB1PdsShfLikoI7GOEgugM= github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ= github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc= github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM= diff --git a/internal/admin/handler_audit.go b/internal/admin/handler_audit.go index 124732f62..9f8712e3a 100644 --- a/internal/admin/handler_audit.go +++ b/internal/admin/handler_audit.go @@ -51,6 +51,7 @@ const conversationBuildTimeout = 10 * time.Second // @Param error_type query string false "Filter by error type" // @Param status_code query int false "Filter by status code" // @Param stream query bool false "Filter by stream mode (true/false)" +// @Param exclude_operation query string false "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay" // @Param search query string false "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message" // @Param limit query int false "Page size (default 25, max 100)" // @Param offset query int false "Offset for pagination" @@ -169,6 +170,14 @@ func parseAuditLogQueryParams(c *echo.Context) (auditlog.LogQueryParams, error) params.Stream = &parsed } + if raw := c.QueryParam("exclude_operation"); raw != "" { + ops, unknown, ok := core.ParseOperations(raw) + if !ok { + return params, core.NewInvalidRequestError("invalid exclude_operation: "+unknown, nil) + } + params.ExcludeOperations = ops + } + if l := c.QueryParam("limit"); l != "" { parsed, err := strconv.Atoi(l) if err != nil || parsed <= 0 { @@ -211,6 +220,7 @@ func parseAuditLogQueryParams(c *echo.Context) (auditlog.LogQueryParams, error) // @Param error_type query string false "Filter by error type" // @Param status_code query int false "Filter by status code" // @Param stream query bool false "Filter by stream mode (true/false)" +// @Param exclude_operation query string false "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay" // @Param search query string false "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message" // @Param limit query int false "Page size in threads (default 25, max 100)" // @Param offset query int false "Offset for pagination" diff --git a/internal/admin/handler_audit_sessions_test.go b/internal/admin/handler_audit_sessions_test.go index 6c4f2fa2b..fb8af79db 100644 --- a/internal/admin/handler_audit_sessions_test.go +++ b/internal/admin/handler_audit_sessions_test.go @@ -7,6 +7,7 @@ import ( "time" "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/echotest" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -137,3 +138,17 @@ func TestAuditLog_SessionIDSkipsDefaultDateWindow(t *testing.T) { require.False(t, reader.lastQuery.StartDate.IsZero()) require.False(t, reader.lastQuery.EndDate.IsZero()) } + +func TestAuditLog_ExcludeOperationFilter(t *testing.T) { + reader := &mockAuditReader{logResult: &auditlog.LogListResult{}} + h := NewHandler(nil, nil, WithAuditReader(reader)) + + c, _ := echotest.Get(t, "/admin/audit/log?exclude_operation=mcp,audio_speech") + require.NoError(t, h.AuditLog(c)) + assert.Equal(t, []core.Operation{core.OperationMCP, core.OperationAudioSpeech}, reader.lastQuery.ExcludeOperations) + + c, rec := echotest.Get(t, "/admin/audit/log?exclude_operation=mcp,nope") + require.NoError(t, h.AuditLog(c)) + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "invalid exclude_operation: nope") +} diff --git a/internal/admin/handler_mcpservers.go b/internal/admin/handler_mcpservers.go index b71f3af7c..b83a36738 100644 --- a/internal/admin/handler_mcpservers.go +++ b/internal/admin/handler_mcpservers.go @@ -37,41 +37,44 @@ const redactedMCPHeaderValue = "***" // Header values equal to "***" preserve the stored value; enabled defaults to // true, preserving the existing value when omitted on an update. type upsertMCPServerRequest struct { - Name string `json:"name"` - Slug string `json:"slug,omitempty"` - URL string `json:"url"` - Transport string `json:"transport,omitempty"` - Headers map[string]string `json:"headers,omitempty"` - Description string `json:"description,omitempty"` - Enabled *bool `json:"enabled,omitempty"` - AllowedTools []string `json:"allowed_tools,omitempty"` - DisallowedTools []string `json:"disallowed_tools,omitempty"` - UserPaths []string `json:"user_paths,omitempty"` - ToolTimeoutSeconds int `json:"tool_timeout_seconds,omitempty"` + Name string `json:"name"` + Slug string `json:"slug,omitempty"` + URL string `json:"url"` + Transport string `json:"transport,omitempty"` + Headers map[string]string `json:"headers,omitempty"` + Description string `json:"description,omitempty"` + Enabled *bool `json:"enabled,omitempty"` + AllowedTools []string `json:"allowed_tools,omitempty"` + DisallowedTools []string `json:"disallowed_tools,omitempty"` + UserPaths []string `json:"user_paths,omitempty"` + DisallowedUserPaths []string `json:"disallowed_user_paths,omitempty"` + ToolTimeoutSeconds int `json:"tool_timeout_seconds,omitempty"` } // mcpServerViewResponse is the admin view of one MCP server: its definition // (headers redacted) plus runtime connection state. Managed marks // config/env-declared servers, which are read-only in the dashboard. type mcpServerViewResponse struct { - Name string `json:"name"` - Slug string `json:"slug"` - URL string `json:"url"` - Transport string `json:"transport"` - Description string `json:"description,omitempty"` - Enabled bool `json:"enabled"` - AllowedTools []string `json:"allowed_tools,omitempty"` - DisallowedTools []string `json:"disallowed_tools,omitempty"` - UserPaths []string `json:"user_paths,omitempty"` - ToolTimeoutSeconds int `json:"tool_timeout_seconds,omitempty"` - Headers map[string]string `json:"headers,omitempty"` - Managed bool `json:"managed"` - Status string `json:"status"` - LastError string `json:"last_error,omitempty"` - ToolCount int `json:"tool_count"` - PromptCount int `json:"prompt_count"` - ResourceCount int `json:"resource_count"` - ConnectedAt *time.Time `json:"connected_at,omitempty"` + Name string `json:"name"` + Slug string `json:"slug"` + URL string `json:"url"` + Transport string `json:"transport"` + Description string `json:"description,omitempty"` + Enabled bool `json:"enabled"` + AllowedTools []string `json:"allowed_tools,omitempty"` + DisallowedTools []string `json:"disallowed_tools,omitempty"` + UserPaths []string `json:"user_paths,omitempty"` + DisallowedUserPaths []string `json:"disallowed_user_paths,omitempty"` + ToolTimeoutSeconds int `json:"tool_timeout_seconds,omitempty"` + Headers map[string]string `json:"headers,omitempty"` + Managed bool `json:"managed"` + Status string `json:"status"` + LastError string `json:"last_error,omitempty"` + ToolCount int `json:"tool_count"` + ExcludedToolCount int `json:"excluded_tool_count"` + PromptCount int `json:"prompt_count"` + ResourceCount int `json:"resource_count"` + ConnectedAt *time.Time `json:"connected_at,omitempty"` } // ListMCPServers handles GET /admin/mcp-servers. @@ -203,7 +206,7 @@ func (h *Handler) ReconnectMCPServer(c *echo.Context) error { // MCPServerCatalog handles GET /admin/mcp-servers/:name/catalog. // // @Summary Inspect one MCP server's current catalog -// @Description Lists the tools, prompts, resources, and resource templates the named server currently exposes through the gateway, after operator tool filters. Names are the upstream originals; the aggregated /mcp endpoint prefixes them with the server slug. +// @Description Lists the tools, prompts, resources, and resource templates the named server currently exposes through the gateway, after operator tool filters. Discovered tools the filters hide are listed separately under excluded_tools. Names are the upstream originals; the aggregated /mcp endpoint prefixes them with the server slug. // @Tags admin // @Produce json // @Security BearerAuth @@ -249,17 +252,18 @@ func (h *Handler) buildMCPServerUpsert(ctx context.Context, slug, displayName st } server := mcpgateway.ManagedServer{ - Name: slug, - DisplayName: displayName, - URL: strings.TrimSpace(req.URL), - Transport: strings.TrimSpace(req.Transport), - Headers: headers, - Description: strings.TrimSpace(req.Description), - Enabled: enabled, - AllowedTools: req.AllowedTools, - DisallowedTools: req.DisallowedTools, - UserPaths: req.UserPaths, - ToolTimeoutSeconds: req.ToolTimeoutSeconds, + Name: slug, + DisplayName: displayName, + URL: strings.TrimSpace(req.URL), + Transport: strings.TrimSpace(req.Transport), + Headers: headers, + Description: strings.TrimSpace(req.Description), + Enabled: enabled, + AllowedTools: req.AllowedTools, + DisallowedTools: req.DisallowedTools, + UserPaths: req.UserPaths, + DisallowedUserPaths: req.DisallowedUserPaths, + ToolTimeoutSeconds: req.ToolTimeoutSeconds, } if current != nil { server.CreatedAt = current.CreatedAt @@ -303,23 +307,25 @@ func (h *Handler) mcpServerView(view mcpgateway.ServerView) mcpServerViewRespons displayName = spec.Name } resp := mcpServerViewResponse{ - Name: displayName, - Slug: spec.Name, - URL: spec.URL, - Transport: spec.Transport, - Description: spec.Description, - Enabled: spec.Enabled, - AllowedTools: spec.AllowedTools, - DisallowedTools: spec.DisallowedTools, - UserPaths: spec.UserPaths, - ToolTimeoutSeconds: int(spec.ToolTimeout / time.Second), - Headers: redactMCPHeaders(spec.Headers), - Managed: h.mcpServers.IsManaged(spec.Name), - Status: string(view.Status), - LastError: view.LastError, - ToolCount: view.ToolCount, - PromptCount: view.PromptCount, - ResourceCount: view.ResourceCount, + Name: displayName, + Slug: spec.Name, + URL: spec.URL, + Transport: spec.Transport, + Description: spec.Description, + Enabled: spec.Enabled, + AllowedTools: spec.AllowedTools, + DisallowedTools: spec.DisallowedTools, + UserPaths: spec.UserPaths, + DisallowedUserPaths: spec.DisallowedUserPaths, + ToolTimeoutSeconds: int(spec.ToolTimeout / time.Second), + Headers: redactMCPHeaders(spec.Headers), + Managed: h.mcpServers.IsManaged(spec.Name), + Status: string(view.Status), + LastError: view.LastError, + ToolCount: view.ToolCount, + ExcludedToolCount: view.ExcludedToolCount, + PromptCount: view.PromptCount, + ResourceCount: view.ResourceCount, } if !view.ConnectedAt.IsZero() { connectedAt := view.ConnectedAt diff --git a/internal/admin/handler_mcpservers_test.go b/internal/admin/handler_mcpservers_test.go index c2b690b99..56110a803 100644 --- a/internal/admin/handler_mcpservers_test.go +++ b/internal/admin/handler_mcpservers_test.go @@ -150,9 +150,10 @@ func TestListMCPServers_RedactsHeadersAndFlagsManaged(t *testing.T) { Enabled: true, ToolTimeout: 30 * time.Second, }, - Status: mcpgateway.StatusConnected, - ToolCount: 3, - ConnectedAt: connectedAt, + Status: mcpgateway.StatusConnected, + ToolCount: 3, + ExcludedToolCount: 2, + ConnectedAt: connectedAt, }) fake.addStored(mcpgateway.ManagedServer{ Name: "notion", @@ -184,6 +185,7 @@ func TestListMCPServers_RedactsHeadersAndFlagsManaged(t *testing.T) { assert.Equal(t, "***", github.Headers["Authorization"]) assert.Equal(t, string(mcpgateway.StatusConnected), github.Status) assert.Equal(t, 3, github.ToolCount) + assert.Equal(t, 2, github.ExcludedToolCount) require.NotNil(t, github.ConnectedAt) assert.True(t, github.ConnectedAt.Equal(connectedAt)) @@ -221,7 +223,7 @@ func TestUpsertMCPServer_CreatesAndReturnsRedactedView(t *testing.T) { fake := newMCPAdminFake() h := newMCPHandler(fake) - body := `{"name":"notion","url":"https://mcp.notion.com/mcp","headers":{"Authorization":"Bearer real-token"},"description":"notes","user_paths":["/team"]}` + body := `{"name":"notion","url":"https://mcp.notion.com/mcp","headers":{"Authorization":"Bearer real-token"},"description":"notes","user_paths":["/team"],"disallowed_user_paths":[" team/contractors/ "]}` c, rec := echotest.Request(t, http.MethodPut, "/admin/mcp-servers", body) err := h.UpsertMCPServer(c) require.NoError(t, err) @@ -238,6 +240,8 @@ func TestUpsertMCPServer_CreatesAndReturnsRedactedView(t *testing.T) { stored, ok := fake.stored["notion"] require.True(t, ok, "upsert did not reach the service") assert.Equal(t, "Bearer real-token", stored.Headers["Authorization"]) + assert.Equal(t, []string{"/team/contractors"}, stored.DisallowedUserPaths) + assert.Equal(t, []string{"/team/contractors"}, view.DisallowedUserPaths) } func TestUpsertMCPServer_PreservesRedactedHeadersAndEnabled(t *testing.T) { @@ -436,8 +440,9 @@ func TestMCPServerCatalog(t *testing.T) { }, mcpgateway.StatusConnected) fake.catalogs = map[string]mcpgateway.CatalogView{ "github": { - Tools: []mcpgateway.CatalogFeature{{Name: "create_issue", Description: "Create an issue"}}, - Prompts: []mcpgateway.CatalogFeature{{Name: "triage"}}, + Tools: []mcpgateway.CatalogFeature{{Name: "create_issue", Description: "Create an issue"}}, + ExcludedTools: []mcpgateway.CatalogFeature{{Name: "delete_repo", Destructive: true}}, + Prompts: []mcpgateway.CatalogFeature{{Name: "triage"}}, Resources: []mcpgateway.CatalogResource{ {URI: "repo://readme", Name: "readme"}, }, @@ -475,6 +480,9 @@ func TestMCPServerCatalog(t *testing.T) { assert.Equal(t, mcpgateway.StatusConnected, catalog.Status) require.Len(t, catalog.Tools, 1) assert.Equal(t, "create_issue", catalog.Tools[0].Name) + require.Len(t, catalog.ExcludedTools, 1) + assert.Equal(t, "delete_repo", catalog.ExcludedTools[0].Name) + assert.True(t, catalog.ExcludedTools[0].Destructive) assert.Len(t, catalog.Prompts, 1) assert.Len(t, catalog.Resources, 1) }) diff --git a/internal/admin/handler_providers.go b/internal/admin/handler_providers.go index 4a3b628ca..931c95b02 100644 --- a/internal/admin/handler_providers.go +++ b/internal/admin/handler_providers.go @@ -262,6 +262,9 @@ func classifyProviderStatus(cfg providers.SanitizedProviderConfig, runtime provi if usingCachedModels { return "degraded", "Starting", "serving cached model inventory while live refresh finishes", lastError } + if runtime.ModelListingUnsupported { + return "healthy", "Healthy", "provider does not list models; serving configured models", lastError + } return "healthy", "Healthy", "configured and model discovery succeeded", lastError case modelFetchError != "" && runtime.DiscoveredModelCount > 0: // Refresh failed but the inventory was deliberately kept fresh (no diff --git a/internal/admin/handler_providers_test.go b/internal/admin/handler_providers_test.go index 53f669fa2..9b1bdcf13 100644 --- a/internal/admin/handler_providers_test.go +++ b/internal/admin/handler_providers_test.go @@ -32,6 +32,29 @@ func TestClassifyProviderStatus_HealthyForAllowlistInventory(t *testing.T) { require.Equal(t, "Healthy", label) } +// A provider without a /models endpoint (e.g. a speech-to-text server) is +// fully described by its configured models and must not look degraded. +func TestClassifyProviderStatus_HealthyWhenModelListingUnsupported(t *testing.T) { + now := time.Now().UTC() + cfg := providers.SanitizedProviderConfig{Name: "stt", Type: "openai"} + runtime := providers.ProviderRuntimeSnapshot{ + Name: "stt", + Type: "openai", + Registered: true, + RegistryInitialized: true, + DiscoveredModelCount: 1, + LastModelFetchAt: &now, + LastModelFetchSuccessAt: &now, + ModelListingUnsupported: true, + } + + status, label, reason, lastError := classifyProviderStatus(cfg, runtime) + require.Equal(t, "healthy", status) + require.Equal(t, "Healthy", label) + require.Equal(t, "provider does not list models; serving configured models", reason) + require.Empty(t, lastError) +} + // A provider retired from load balancing by a failed availability probe has a // clean model-fetch record but must not be reported healthy: the routing layer // is actively skipping it and its models are hidden from the model list. diff --git a/internal/anthropicapi/request.go b/internal/anthropicapi/request.go index 11c7c38eb..a77417568 100644 --- a/internal/anthropicapi/request.go +++ b/internal/anthropicapi/request.go @@ -750,6 +750,9 @@ func convertTools(tools []Tool, lenient bool) ([]map[string]any, error) { } function["parameters"] = schema } + if tool.Strict != nil { + function["strict"] = *tool.Strict + } converted := map[string]any{"type": "function", "function": function} raw, err := validatedCacheControlJSON(tool.CacheControl) if err != nil { diff --git a/internal/anthropicapi/request_test.go b/internal/anthropicapi/request_test.go index 0084f68d2..c4c0b02b8 100644 --- a/internal/anthropicapi/request_test.go +++ b/internal/anthropicapi/request_test.go @@ -244,6 +244,26 @@ func TestToChatRequestTools(t *testing.T) { require.Equal(t, "function", choice["type"], "tool_choice = %#v", chat.ToolChoice) } +func TestToChatRequestToolStrict(t *testing.T) { + chat, err := ToChatRequest(mustDecode(t, `{ + "model":"m","max_tokens":10, + "messages":[{"role":"user","content":"hi"}], + "tools":[ + {"name":"strict_tool","strict":true,"input_schema":{"type":"object","properties":{}}}, + {"name":"plain_tool","input_schema":{"type":"object","properties":{}}} + ] + }`)) + require.NoError(t, err) + require.Len(t, chat.Tools, 2) + + strictFn, ok := chat.Tools[0]["function"].(map[string]any) + require.True(t, ok) + assert.Equal(t, true, strictFn["strict"]) + plainFn, ok := chat.Tools[1]["function"].(map[string]any) + require.True(t, ok) + assert.NotContains(t, plainFn, "strict") +} + func TestToChatRequestRejectsServerTool(t *testing.T) { _, err := ToChatRequest(mustDecode(t, `{ "model":"m","max_tokens":10, diff --git a/internal/anthropicapi/types.go b/internal/anthropicapi/types.go index ae8236906..5a0dea72e 100644 --- a/internal/anthropicapi/types.go +++ b/internal/anthropicapi/types.go @@ -88,6 +88,7 @@ type Tool struct { Name string `json:"name"` Description string `json:"description,omitempty"` InputSchema json.RawMessage `json:"input_schema,omitempty" swaggertype:"object"` + Strict *bool `json:"strict,omitempty"` CacheControl json.RawMessage `json:"cache_control,omitempty" swaggertype:"object"` } diff --git a/internal/auditlog/reader.go b/internal/auditlog/reader.go index 9554a6b24..6dfcaa056 100644 --- a/internal/auditlog/reader.go +++ b/internal/auditlog/reader.go @@ -3,6 +3,8 @@ package auditlog import ( "context" "time" + + "github.com/enterpilot/gomodel/internal/core" ) // QueryParams specifies the date range for audit log retrieval. @@ -24,8 +26,12 @@ type LogQueryParams struct { Search string StatusCode *int Stream *bool - Limit int - Offset int + // ExcludeOperations drops entries whose path belongs to one of these + // operations. Entries outside every operation (e.g. authentication + // events) always stay. + ExcludeOperations []core.Operation + Limit int + Offset int // OmitAttempts excludes provider attempts from returned entries. The default is false. OmitAttempts bool // ExactUserPath matches only UserPath instead of its subtree. The default is false. diff --git a/internal/auditlog/reader_mongodb.go b/internal/auditlog/reader_mongodb.go index 0069df70b..76555e42f 100644 --- a/internal/auditlog/reader_mongodb.go +++ b/internal/auditlog/reader_mongodb.go @@ -283,6 +283,9 @@ func mongoLogMatchFilters(params LogQueryParams) (bson.D, error) { if params.Stream != nil { matchFilters = append(matchFilters, bson.E{Key: "stream", Value: *params.Stream}) } + if len(params.ExcludeOperations) > 0 { + matchFilters = append(matchFilters, mongoExcludeOperationsFilter(params.ExcludeOperations)) + } if params.Search != "" && isCanonicalUUID(params.Search) { // A full canonical UUID is a pasted identifier: match the indexed // identity fields by equality (both spellings — stored ids are @@ -417,3 +420,25 @@ func (r *MongoDBReader) findConversationEntry(ctx context.Context, filter bson.D return row.toLogEntry(), nil } + +// mongoExcludeOperationsFilter drops paths belonging to any of the +// operations; entries without a path stay. Exact paths also match with one +// trailing slash, as DescribeEndpoint does. +func mongoExcludeOperationsFilter(ops []core.Operation) bson.E { + var exact bson.A + var nor bson.A + for _, op := range ops { + paths, _ := core.PathsForOperation(op) + for _, path := range paths.Exact { + exact = append(exact, path, path+"/") + } + for _, prefix := range paths.Prefixes { + exact = append(exact, prefix) + nor = append(nor, bson.D{{Key: "path", Value: bson.D{ + {Key: "$regex", Value: "^" + regexp.QuoteMeta(prefix+"/")}, + }}}) + } + } + nor = append(nor, bson.D{{Key: "path", Value: bson.D{{Key: "$in", Value: exact}}}}) + return bson.E{Key: "$nor", Value: nor} +} diff --git a/internal/auditlog/reader_sql.go b/internal/auditlog/reader_sql.go index 05ea05160..e47548052 100644 --- a/internal/auditlog/reader_sql.go +++ b/internal/auditlog/reader_sql.go @@ -12,6 +12,7 @@ import ( "github.com/goccy/go-json" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/storage/sqlutil" "github.com/enterpilot/gomodel/internal/storage/sqlx" ) @@ -243,6 +244,10 @@ func (r *SQLReader) logFilters(ctx context.Context, params LogQueryParams) ([]st if params.Stream != nil { add("stream = ?", *params.Stream) } + if len(params.ExcludeOperations) > 0 { + condition, values := excludeOperationsSQLFilter(params.ExcludeOperations) + add(condition, values...) + } if params.Search != "" { condition, values := r.searchFilter(params.Search, r.searchIsIndexed(ctx)) add(condition, values...) @@ -560,3 +565,23 @@ func isMissingAuditAttemptsTable(err error) bool { return strings.Contains(message, "audit_log_attempts") && (strings.Contains(message, "no such table") || strings.Contains(message, "does not exist")) } + +// excludeOperationsSQLFilter drops paths belonging to any of the operations. +// Exact paths also match with one trailing slash, as DescribeEndpoint does. +// The prefixes hold no LIKE wildcards, so they need no escaping. +func excludeOperationsSQLFilter(ops []core.Operation) (string, []any) { + var clauses []string + var args []any + for _, op := range ops { + paths, _ := core.PathsForOperation(op) + for _, exact := range paths.Exact { + clauses = append(clauses, "path = ?", "path = ?") + args = append(args, exact, exact+"/") + } + for _, prefix := range paths.Prefixes { + clauses = append(clauses, "path = ?", "path LIKE ?") + args = append(args, prefix, prefix+"/%") + } + } + return "(path IS NULL OR NOT (" + strings.Join(clauses, " OR ") + "))", args +} diff --git a/internal/auditlog/reader_suite_test.go b/internal/auditlog/reader_suite_test.go index decbeb639..c40be48f3 100644 --- a/internal/auditlog/reader_suite_test.go +++ b/internal/auditlog/reader_suite_test.go @@ -2,11 +2,13 @@ package auditlog import ( "context" + "fmt" "testing" "time" "go.mongodb.org/mongo-driver/v2/mongo" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/storage/mongotest" "github.com/enterpilot/gomodel/internal/storage/sqlx" "github.com/enterpilot/gomodel/internal/storage/sqlx/sqlxtest" @@ -103,3 +105,57 @@ func TestReader_GetLastUsedByAuthKeys(t *testing.T) { assert.False(t, ok) }) } + +func TestReader_GetLogsExcludesOperations(t *testing.T) { + runReaderSuite(t, func(t *testing.T, store LogStore, reader Reader) { + ctx := context.Background() + base := time.Date(2026, 1, 16, 12, 0, 0, 0, time.UTC) + paths := []string{ + "/v1/chat/completions", "/mcp", "/mcp/github", "/mcpx", + "/v1/audio/speech", "/v1/audio/speech/", "/v1/audio/transcriptions", + "/p/openai/v1/models", "/sso/callback", "", + } + entries := make([]*LogEntry, 0, len(paths)) + for i, path := range paths { + entries = append(entries, &LogEntry{ + ID: fmt.Sprintf("op-%d", i), + Timestamp: base.Add(time.Duration(i) * time.Minute), + Path: path, + }) + } + require.NoError(t, store.WriteBatch(ctx, entries)) + + tests := []struct { + name string + ops []core.Operation + want []string + }{ + {name: "mcp prefix", ops: []core.Operation{core.OperationMCP}, want: []string{ + "/v1/chat/completions", "/mcpx", "/v1/audio/speech", "/v1/audio/speech/", + "/v1/audio/transcriptions", "/p/openai/v1/models", "/sso/callback", "", + }}, + {name: "exact with trailing slash and prefix", ops: []core.Operation{ + core.OperationAudioSpeech, core.OperationProviderPassthrough, + }, want: []string{ + "/v1/chat/completions", "/mcp", "/mcp/github", "/mcpx", + "/v1/audio/transcriptions", "/sso/callback", "", + }}, + {name: "every classified type keeps unclassified rows", ops: []core.Operation{ + core.OperationChatCompletions, core.OperationMCP, core.OperationAudioSpeech, + core.OperationAudioTranscriptions, core.OperationProviderPassthrough, + }, want: []string{"/mcpx", "/sso/callback", ""}}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result, err := reader.GetLogs(ctx, LogQueryParams{ExcludeOperations: tt.ops, Limit: 50}) + require.NoError(t, err) + got := make([]string, 0, len(result.Entries)) + for _, entry := range result.Entries { + got = append(got, entry.Path) + } + assert.ElementsMatch(t, tt.want, got) + assert.Equal(t, len(tt.want), result.Total) + }) + } + }) +} diff --git a/internal/core/endpoint_operations.go b/internal/core/endpoint_operations.go new file mode 100644 index 000000000..26ac74342 --- /dev/null +++ b/internal/core/endpoint_operations.go @@ -0,0 +1,59 @@ +package core + +import "strings" + +// OperationPaths lists the request paths that DescribeEndpoint classifies as +// one operation, in a shape storage filters can match: Exact paths compare by +// equality, and each Prefix matches itself or anything under "Prefix/". +type OperationPaths struct { + Exact []string + Prefixes []string +} + +// operationPaths mirrors describeEndpointPath. TestOperationPathsMatchDescribeEndpoint +// keeps the two in sync. +var operationPaths = map[Operation]OperationPaths{ + OperationChatCompletions: {Exact: []string{"/v1/chat/completions", "/v1/messages", "/v1/messages/count_tokens"}}, + OperationResponses: {Prefixes: []string{"/v1/responses"}}, + OperationConversations: {Prefixes: []string{"/v1/conversations"}}, + OperationEmbeddings: {Exact: []string{"/v1/embeddings"}}, + OperationBatches: {Prefixes: []string{"/v1/batches", "/v1/messages/batches"}}, + OperationFiles: {Prefixes: []string{"/v1/files"}}, + OperationAudioSpeech: {Exact: []string{"/v1/audio/speech"}}, + OperationAudioTranscriptions: {Exact: []string{"/v1/audio/transcriptions"}}, + OperationAudioTranslations: {Exact: []string{"/v1/audio/translations"}}, + OperationImageGenerations: {Exact: []string{"/v1/images/generations"}}, + OperationImageEdits: {Exact: []string{"/v1/images/edits"}}, + OperationRealtime: {Exact: []string{ + "/v1/realtime", "/v1/realtime/calls", "/v1/realtime/client_secrets", + "/v1/realtime/translations", "/v1/realtime/translations/calls", "/v1/realtime/translations/client_secrets", + }}, + OperationMCP: {Prefixes: []string{"/mcp"}}, + OperationSystemOne: {Exact: []string{"/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate"}}, + OperationProviderPassthrough: {Prefixes: []string{"/p"}}, +} + +// PathsForOperation returns the paths of a known operation. +func PathsForOperation(op Operation) (OperationPaths, bool) { + paths, ok := operationPaths[op] + return paths, ok +} + +// ParseOperations parses a comma-separated operation list, ignoring blanks +// and duplicates. It reports the first unknown name. +func ParseOperations(raw string) ([]Operation, string, bool) { + var ops []Operation + seen := map[Operation]bool{} + for part := range strings.SplitSeq(raw, ",") { + op := Operation(strings.ToLower(strings.TrimSpace(part))) + if op == "" || seen[op] { + continue + } + if _, ok := operationPaths[op]; !ok { + return nil, string(op), false + } + seen[op] = true + ops = append(ops, op) + } + return ops, "", true +} diff --git a/internal/core/endpoint_operations_test.go b/internal/core/endpoint_operations_test.go new file mode 100644 index 000000000..ba2bd8edd --- /dev/null +++ b/internal/core/endpoint_operations_test.go @@ -0,0 +1,72 @@ +package core + +import ( + "slices" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestOperationPathsMatchDescribeEndpoint(t *testing.T) { + for op, paths := range operationPaths { + for _, path := range paths.Exact { + assert.Equal(t, op, DescribeEndpointPath(path).Operation, path) + } + for _, prefix := range paths.Prefixes { + assert.Equal(t, op, DescribeEndpointPath(prefix+"/x").Operation, prefix+"/x") + } + } +} + +func TestOperationPathsCoverClassifiedPaths(t *testing.T) { + paths := []string{ + "/v1/chat/completions", "/v1/messages", "/v1/messages/count_tokens", + "/v1/responses", "/v1/responses/resp_1/input_items", "/v1/conversations/conv_1", + "/v1/embeddings", "/v1/batches/b_1/cancel", "/v1/messages/batches/b_1", + "/v1/files/f_1/content", "/v1/audio/speech", "/v1/audio/transcriptions", + "/v1/audio/translations", "/v1/images/generations", "/v1/images/edits", + "/v1/realtime", "/v1/realtime/translations/calls", "/mcp", "/mcp/github", + "/p/openai/v1/models", + } + for _, path := range paths { + want := DescribeEndpointPath(path).Operation + require.NotEmpty(t, want, path) + rule, ok := PathsForOperation(want) + require.True(t, ok, path) + assert.True(t, rule.matches(path), path) + } +} + +func (p OperationPaths) matches(path string) bool { + if slices.Contains(p.Exact, path) { + return true + } + for _, prefix := range p.Prefixes { + if path == prefix || len(path) > len(prefix) && path[:len(prefix)+1] == prefix+"/" { + return true + } + } + return false +} + +func TestParseOperations(t *testing.T) { + tests := []struct { + name string + raw string + want []Operation + unknown string + }{ + {name: "empty", raw: ""}, + {name: "list", raw: " MCP, audio_speech,,mcp ", want: []Operation{OperationMCP, OperationAudioSpeech}}, + {name: "unknown", raw: "mcp,nope", unknown: "nope"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, unknown, ok := ParseOperations(tt.raw) + assert.Equal(t, tt.unknown == "", ok) + assert.Equal(t, tt.unknown, unknown) + assert.Equal(t, tt.want, got) + }) + } +} diff --git a/internal/core/endpoints.go b/internal/core/endpoints.go index b75404322..0e2edc494 100644 --- a/internal/core/endpoints.go +++ b/internal/core/endpoints.go @@ -33,6 +33,7 @@ const ( OperationRealtime Operation = "realtime" OperationProviderPassthrough Operation = "provider_passthrough" OperationMCP Operation = "mcp" + OperationSystemOne Operation = "systemone" ) // EndpointDescriptor centralizes the transport-facing classification of model and provider routes. @@ -169,6 +170,17 @@ func describeEndpointPath(path string) EndpointDescriptor { Dialect: "openai_compat", Operation: OperationImageEdits, } + case path == "/v1/systemone" || path == "/v1/systemone/permute" || path == "/v1/systemone/separate": + // TypeSafe's System One decision API (Jev, Kev) and the diagnostic + // variants Kev servers add. It has no canonical translation: the body + // is forwarded to a System One provider unchanged, apart from the + // routed model and guardrail edits to state. + return EndpointDescriptor{ + ModelInteraction: true, + IngressManaged: true, + Dialect: "systemone", + Operation: OperationSystemOne, + } case isRealtimePath(path): // The realtime endpoints relay the provider's schema verbatim: /v1/realtime // upgrades to a websocket, /v1/realtime/calls exchanges WebRTC SDP, and @@ -238,7 +250,7 @@ func bodyModeForEndpoint(method, path string, operation Operation) BodyMode { return BodyModeMultipart } return BodyModeNone - case OperationAudioSpeech, OperationImageGenerations: + case OperationAudioSpeech, OperationImageGenerations, OperationSystemOne: return BodyModeJSON case OperationAudioTranscriptions, OperationAudioTranslations, OperationImageEdits: return BodyModeMultipart diff --git a/internal/core/endpoints_test.go b/internal/core/endpoints_test.go index 6c48356b2..3fd2b0046 100644 --- a/internal/core/endpoints_test.go +++ b/internal/core/endpoints_test.go @@ -42,6 +42,10 @@ func TestDescribeEndpointPath(t *testing.T) { {path: "/v1/realtime/translations/client_secrets", managed: false, dialect: "openai_compat", operation: OperationRealtime, bodyMode: BodyModeNone, interaction: true}, {path: "/mcp", managed: false, dialect: "mcp", operation: OperationMCP, bodyMode: BodyModeNone, interaction: true}, {path: "/mcp/linear", managed: false, dialect: "mcp", operation: OperationMCP, bodyMode: BodyModeNone, interaction: true}, + {path: "/v1/systemone", managed: true, dialect: "systemone", operation: OperationSystemOne, bodyMode: BodyModeJSON, interaction: true}, + {path: "/v1/systemone/permute", managed: true, dialect: "systemone", operation: OperationSystemOne, bodyMode: BodyModeJSON, interaction: true}, + {path: "/v1/systemone/separate", managed: true, dialect: "systemone", operation: OperationSystemOne, bodyMode: BodyModeJSON, interaction: true}, + {path: "/v1/systemone/other", managed: false, dialect: "", operation: "", bodyMode: BodyModeNone, interaction: false}, {path: "/p/openai/responses", managed: true, dialect: "provider_passthrough", operation: OperationProviderPassthrough, bodyMode: BodyModeOpaque, interaction: true}, {path: "/v1/models", managed: false, dialect: "", operation: "", bodyMode: BodyModeNone, interaction: false}, } diff --git a/internal/core/errors.go b/internal/core/errors.go index 3640d3dda..92d4aa6fc 100644 --- a/internal/core/errors.go +++ b/internal/core/errors.go @@ -188,6 +188,35 @@ func NewEmptyProviderResponseError(provider string) *GatewayError { // an empty 200 response apart from other 502s. var ErrNoChoices = errors.New("provider returned no choices") +// ErrModelListingUnsupported marks a model-list failure meaning the upstream +// has no /models endpoint, as opposed to a listing that failed; test with +// errors.Is. +var ErrModelListingUnsupported = errors.New("provider does not list models") + +// MarkModelListingUnsupported tags a 404 or 405 from an OpenAI-compatible +// /models call with ErrModelListingUnsupported. The message is unchanged and +// errors.As still finds the GatewayError; other errors pass through as is. +// Only call it on errors from a /models request: the same statuses from other +// APIs do not mean the endpoint is missing. +func MarkModelListingUnsupported(err error) error { + gatewayErr, ok := errors.AsType[*GatewayError](err) + if !ok { + return err + } + switch gatewayErr.HTTPStatusCode() { + case http.StatusNotFound, http.StatusMethodNotAllowed: + return modelListingUnsupportedError{err} + default: + return err + } +} + +type modelListingUnsupportedError struct{ error } + +func (e modelListingUnsupportedError) Unwrap() []error { + return []error{e.error, ErrModelListingUnsupported} +} + // NewNoChoicesProviderError reports a chat completion that succeeded upstream // but carried no choices (502), so failover treats it as a failed attempt. func NewNoChoicesProviderError(provider string) *GatewayError { diff --git a/internal/core/interfaces.go b/internal/core/interfaces.go index 8e03490c6..3dd20210f 100644 --- a/internal/core/interfaces.go +++ b/internal/core/interfaces.go @@ -207,6 +207,13 @@ type MessagesTokenCounter interface { CountMessagesTokens(ctx context.Context, model string, body []byte) (int, error) } +// UnlistedModelAcceptor is implemented by providers that serve model IDs +// their listing omits, such as TypeSafe's versioned Jev IDs (jev-1.13.0), so a +// virtual model can target a provider-qualified name the catalog lacks. +type UnlistedModelAcceptor interface { + AcceptsUnlistedModels() bool +} + // ErrMessagesTokenCountUnsupported reports that the provider owning a model // has no token counting endpoint. var ErrMessagesTokenCountUnsupported = errors.New("provider has no token counting endpoint") diff --git a/internal/core/systemone.go b/internal/core/systemone.go new file mode 100644 index 000000000..8ee6d5bd1 --- /dev/null +++ b/internal/core/systemone.go @@ -0,0 +1,13 @@ +package core + +import "github.com/goccy/go-json" + +// SystemOneRequest is the part of a System One decision request the gateway +// reads: the model it routes on and the state guardrails inspect. The rest of +// the body (the questions and their criteria) reaches the provider unchanged. +type SystemOneRequest struct { + Model string `json:"model"` + // State is the text or record the questions are asked about, as sent: a + // JSON string in TypeSafe's examples, but any JSON value is forwarded. + State json.RawMessage `json:"state,omitempty"` +} diff --git a/internal/core/workflow.go b/internal/core/workflow.go index 4d9a25f37..a6cc38ac5 100644 --- a/internal/core/workflow.go +++ b/internal/core/workflow.go @@ -57,6 +57,12 @@ func CapabilitiesForEndpoint(desc EndpointDescriptor) CapabilitySet { return CapabilitySet{ SemanticExtraction: true, } + case OperationSystemOne: + return CapabilitySet{ + AliasResolution: true, + Guardrails: true, + UsageTracking: true, + } case OperationProviderPassthrough: return CapabilitySet{ SemanticExtraction: true, diff --git a/internal/gateway/failover.go b/internal/gateway/failover.go index 370db8477..4a1ede930 100644 --- a/internal/gateway/failover.go +++ b/internal/gateway/failover.go @@ -30,6 +30,9 @@ func (o *InferenceOrchestrator) ProviderTypeForSelector(selector core.ModelSelec if providerType := strings.TrimSpace(o.provider.GetProviderType(selector.QualifiedModel())); providerType != "" { return providerType } + if _, providerType := configuredSelectorProvider(o.provider, selector); providerType != "" { + return providerType + } if provider := strings.TrimSpace(selector.Provider); provider != "" { return provider } @@ -42,6 +45,7 @@ func tryFailoverResponse[T any]( workflow *core.Workflow, model, provider string, primaryErr error, + eligible func(selector core.ModelSelector, providerType string) bool, call func(selector core.ModelSelector, providerType, providerName string) (T, string, error), ) (T, ExecutionMeta, error) { var zero T @@ -75,6 +79,16 @@ func tryFailoverResponse[T any]( qualified := selector.QualifiedModel() providerType := o.ProviderTypeForSelector(selector, ProviderTypeFromWorkflow(workflow)) providerName := ResolvedProviderName(o.provider, selector, ProviderNameFromWorkflow(workflow)) + // A target that cannot serve the request is skipped before it counts + // against the attempt cap, so it never crowds out a later valid one. + if eligible != nil && !eligible(selector, providerType) { + slog.Info("skipping failover target that cannot serve the request", + "request_id", requestID, + "to", qualified, + "provider_type", providerType, + ) + continue + } if o.routeGate != nil && !o.routeGate.RouteAvailable(providerName, qualified) { slog.Info("skipping rate-limited failover target", "request_id", requestID, @@ -121,13 +135,14 @@ func executeWithFailoverResponse[T any]( workflow *core.Workflow, model, provider string, primary func() (T, string, string, error), + eligible func(selector core.ModelSelector, providerType string) bool, failoverFn func(selector core.ModelSelector, providerType, providerName string) (T, string, error), ) (T, ExecutionMeta, error) { resp, resolvedProviderType, resolvedProviderName, err := primary() if err == nil { return resp, ExecutionMeta{ProviderType: resolvedProviderType, ProviderName: resolvedProviderName}, nil } - return tryFailoverResponse(ctx, o, workflow, model, provider, err, failoverFn) + return tryFailoverResponse(ctx, o, workflow, model, provider, err, eligible, failoverFn) } func executeTranslatedWithFailover[Req any, Resp any]( @@ -158,6 +173,7 @@ func executeTranslatedWithFailover[Req any, Resp any]( } return resp, ResponseProviderType(ProviderTypeFromWorkflow(workflow), responseProvider), ProviderNameFromWorkflow(workflow), nil }, + nil, func(selector core.ModelSelector, providerType, providerName string) (Resp, string, error) { // A failover target gets a different request body, so it must not // reuse the client's idempotency key. @@ -255,3 +271,57 @@ func firstNonEmptyString(values ...string) string { } return "" } + +// PassthroughCall sends one native request to selector's provider. Provider +// error statuses must come back as errors so the failover policy can judge +// them. +type PassthroughCall func(ctx context.Context, selector core.ModelSelector, providerType, providerName string) (*core.PassthroughResponse, error) + +// ExecutePassthroughWithFailover runs a native, untranslated request against +// the workflow's resolved route and then, while the failover policy allows, +// against its failover targets. It is the native-endpoint counterpart of the +// translated failover path: attempts are recorded the same way, but every +// target receives the client's own dialect, so eligible must reject a +// failover target that cannot serve it; rejected targets are skipped without +// counting against the attempt cap. The selector that answered is returned +// with the response. +func (o *InferenceOrchestrator) ExecutePassthroughWithFailover(ctx context.Context, workflow *core.Workflow, eligible func(selector core.ModelSelector, providerType string) bool, call PassthroughCall) (*core.PassthroughResponse, core.ModelSelector, ExecutionMeta, error) { + primary := core.ModelSelector{} + if workflow != nil && workflow.Resolution != nil { + primary = workflow.Resolution.ResolvedSelector + } + type answer struct { + resp *core.PassthroughResponse + selector core.ModelSelector + } + result, meta, err := executeWithFailoverResponse(ctx, o, workflow, primary.Model, primary.Provider, + func() (answer, string, string, error) { + started := time.Now() + providerType, providerName := ProviderTypeFromWorkflow(workflow), ProviderNameFromWorkflow(workflow) + qualified := primary.QualifiedModel() + // A rate-saturated primary route must not reach the provider; its + // stored 429 becomes the primary failure that starts the sweep. + if saturated := core.PrimaryRouteSaturated(ctx); saturated != nil { + recordProviderAttempt(ctx, providerAttemptFromResult(AttemptKindPrimary, providerType, providerName, qualified, started, saturated)) + return answer{}, "", "", saturated + } + resp, err := call(ctx, primary, providerType, providerName) + recordProviderAttempt(ctx, providerAttemptFromResult(AttemptKindPrimary, providerType, providerName, qualified, started, err)) + if err != nil { + return answer{}, "", "", err + } + return answer{resp: resp, selector: primary}, providerType, providerName, nil + }, + eligible, + func(selector core.ModelSelector, providerType, providerName string) (answer, string, error) { + // A failover target gets a different body, so it must not reuse + // the client's idempotency key. + resp, err := call(core.WithIdempotencyKey(ctx, ""), selector, providerType, providerName) + if err != nil { + return answer{}, "", err + } + return answer{resp: resp, selector: selector}, providerType, nil + }, + ) + return result.resp, result.selector, meta, err +} diff --git a/internal/gateway/failover_policy_test.go b/internal/gateway/failover_policy_test.go index fdf4f9176..cdb96d676 100644 --- a/internal/gateway/failover_policy_test.go +++ b/internal/gateway/failover_policy_test.go @@ -92,7 +92,7 @@ func TestTryFailoverResponseHonorsMaxAttempts(t *testing.T) { return "", "", core.NewProviderError("openai", http.StatusBadGateway, selector.Model+" down", nil) } - _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.False(t, meta.UsedFailover) require.Error(t, err) @@ -104,6 +104,25 @@ func TestTryFailoverResponseHonorsMaxAttempts(t *testing.T) { } } +// Targets that cannot serve the request (a chat model in a System One chain) +// are skipped before a call, so they do not consume attempts either. +func TestTryFailoverResponseIneligibleTargetsDoNotConsumeAttempts(t *testing.T) { + o, workflow := threeTargetFixture(&FailoverPolicy{MaxAttempts: 1}) + primaryErr := core.NewProviderError("openai", http.StatusBadGateway, "primary down", nil) + eligible := func(selector core.ModelSelector, _ string) bool { return selector.Model != "a" } + var calls []string + call := func(selector core.ModelSelector, _, _ string) (string, string, error) { + calls = append(calls, selector.QualifiedModel()) + return "ok", "openai", nil + } + + _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, eligible, call) + + require.NoError(t, err) + require.True(t, meta.UsedFailover) + require.Equal(t, []string{"openai/b"}, calls) +} + // Targets skipped before a call (rate-limited routes) do not consume attempts. func TestTryFailoverResponseMaxAttemptsCountsCallsOnly(t *testing.T) { o, workflow := threeTargetFixture(&FailoverPolicy{MaxAttempts: 1}) @@ -115,7 +134,7 @@ func TestTryFailoverResponseMaxAttemptsCountsCallsOnly(t *testing.T) { return "ok", "openai", nil } - _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.True(t, meta.UsedFailover) require.NoError(t, err) @@ -150,7 +169,7 @@ func TestTryFailoverResponseSkipsWhenPolicyDoesNotMatch(t *testing.T) { return "ok", "openai", nil } - _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.False(t, called) require.False(t, meta.UsedFailover) diff --git a/internal/gateway/failover_test.go b/internal/gateway/failover_test.go index db345517e..f6ac6e541 100644 --- a/internal/gateway/failover_test.go +++ b/internal/gateway/failover_test.go @@ -48,7 +48,7 @@ func TestTryFailoverResponseSkipsWhenContextCanceled(t *testing.T) { return "", "", core.NewProviderError("openai", http.StatusBadGateway, "unexpected failover call", nil) } - _, meta, err := tryFailoverResponse(ctx, o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + _, meta, err := tryFailoverResponse(ctx, o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.False(t, called) require.False(t, meta.UsedFailover) @@ -66,7 +66,7 @@ func TestTryFailoverResponseAttemptsWhenContextLive(t *testing.T) { return "ok", "openai", nil } - resp, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + resp, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.True(t, called) require.True(t, meta.UsedFailover) @@ -100,7 +100,7 @@ func TestTryFailoverResponseSkipsRateLimitedTargets(t *testing.T) { return "ok", "anthropic", nil } - resp, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + resp, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.Len(t, attempted, 1) require.Equal(t, "anthropic/claude", attempted[0]) diff --git a/internal/gateway/inference_orchestrator_test.go b/internal/gateway/inference_orchestrator_test.go index 3e2c2e27c..63d031d97 100644 --- a/internal/gateway/inference_orchestrator_test.go +++ b/internal/gateway/inference_orchestrator_test.go @@ -8,6 +8,7 @@ import ( "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/usage" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -169,6 +170,30 @@ func TestInferenceOrchestratorProviderTypeForSelectorCanonicalizesProviderNameSe require.Equal(t, "openai", got) } +// namedProviderStub knows configured provider names and their types, but no +// catalog entry for the models under test. +type namedProviderStub struct { + providerTypeResolverStub + typesByName map[string]string +} + +func (p *namedProviderStub) GetProviderTypeForName(name string) string { return p.typesByName[name] } + +// A failover target the catalog does not list, such as a pinned jev version +// on a provider named kev, is routed to the provider its selector names, with +// that provider's type, never to the primary's provider. +func TestUnlistedSelectorResolvesToTheProviderItNames(t *testing.T) { + provider := &namedProviderStub{typesByName: map[string]string{"kev": "jev", "jev-down": "jev"}} + orchestrator := NewInferenceOrchestrator(InferenceConfig{Provider: provider}) + + pinned := core.ModelSelector{Provider: "kev", Model: "kev-4b-2026-09"} + assert.Equal(t, "jev", orchestrator.ProviderTypeForSelector(pinned, "openai")) + assert.Equal(t, "kev", ResolvedProviderName(provider, pinned, "jev-down")) + + unknown := core.ModelSelector{Provider: "nope", Model: "x"} + assert.Equal(t, "jev-down", ResolvedProviderName(provider, unknown, "jev-down")) +} + func TestQualifyModelWithProviderPrefixesSlashModelIDs(t *testing.T) { got := QualifyModelWithProvider("openai/gpt-4o-mini", "openrouter") require.Equal(t, "openrouter/openai/gpt-4o-mini", got) diff --git a/internal/gateway/interfaces.go b/internal/gateway/interfaces.go index 60fa8783e..cc4cf8da6 100644 --- a/internal/gateway/interfaces.go +++ b/internal/gateway/interfaces.go @@ -53,6 +53,12 @@ type TranslatedRequestPatcher interface { PatchResponsesRequest(ctx context.Context, req *core.ResponsesRequest) (*core.ResponsesRequest, error) } +// SystemOneRequestPatcher is an optional TranslatedRequestPatcher capability: +// it runs the prompt phase over a System One decision request's state. +type SystemOneRequestPatcher interface { + PatchSystemOneRequest(ctx context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) +} + // PromptContentEditor is an optional TranslatedRequestPatcher capability: it // reports whether the prompt phase may rewrite the content of this request. // Only a rewriting prompt phase (anonymization, redaction) needs the replayed diff --git a/internal/gateway/request_model_resolution.go b/internal/gateway/request_model_resolution.go index da77d7cfa..fdf1ca2cb 100644 --- a/internal/gateway/request_model_resolution.go +++ b/internal/gateway/request_model_resolution.go @@ -30,9 +30,30 @@ func ResolvedProviderName(provider core.RoutableProvider, selector core.ModelSel return providerName } } + // A model the catalog does not list (a pinned jev version) still belongs to + // the provider its selector names, not to the fallback's. + if providerName, providerType := configuredSelectorProvider(provider, selector); providerType != "" { + return providerName + } return fallback } +// configuredSelectorProvider returns the provider a selector names explicitly +// and that provider's type, or empty strings when the selector names no +// configured provider. +func configuredSelectorProvider(provider core.RoutableProvider, selector core.ModelSelector) (string, string) { + providerName := strings.TrimSpace(selector.Provider) + named, ok := provider.(core.ProviderNameTypeResolver) + if providerName == "" || !ok { + return "", "" + } + providerType := strings.TrimSpace(named.GetProviderTypeForName(providerName)) + if providerType == "" { + return "", "" + } + return providerName, providerType +} + // ResolvedWorkflowProviderName returns the configured provider name recorded in a resolution. func ResolvedWorkflowProviderName(resolution *core.RequestModelResolution) string { if resolution == nil { diff --git a/internal/guardrails/integration_test.go b/internal/guardrails/integration_test.go index 03f1b096d..fdb696698 100644 --- a/internal/guardrails/integration_test.go +++ b/internal/guardrails/integration_test.go @@ -11,6 +11,7 @@ import ( "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/plugins" "github.com/enterpilot/gomodel/pluginapi" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -329,3 +330,31 @@ func TestWorkflowBatchPreparerRejectsRespondDecisions(t *testing.T) { require.ErrorAs(t, err, &gatewayErr) require.Equal(t, http.StatusBadRequest, gatewayErr.HTTPStatusCode()) } + +// A System One request exposes only its state: an anonymizing guardrail +// rewrites it, while a system prompt a workflow injects for chat models has +// no place in a decision request and is dropped rather than failing it. +func TestWorkflowRequestPatcherSystemOneGuardsState(t *testing.T) { + store := newTestStore( + systemPromptDefinition("safety", "be safe"), + Definition{Name: "privacy", Type: "llm_based_altering", Config: rawConfig(t, map[string]any{"model": "openai/gpt-4o-mini", "roles": []string{"user"}})}, + ) + service := newService(t, store, chatFunc(func(_ context.Context, req *core.ChatRequest) (*core.ChatResponse, error) { + text := core.ExtractTextContent(req.Messages[1].Content) + inner := strings.TrimSuffix(strings.TrimPrefix(text, "\n"), "\n") + return replyChat(strings.ReplaceAll(inner, "John", "[PERSON]"))(context.Background(), req) + })) + patcher := NewWorkflowRequestPatcher(staticChains{chainsFor(t, service, + StepReference{Ref: "privacy", Step: 10}, + StepReference{Ref: "safety", Step: 20}, + )}) + + req := &core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(`"John was charged twice"`)} + ctx, state := plugins.WithRequestState(context.Background()) + got, err := patcher.PatchSystemOneRequest(ctx, req) + require.NoError(t, err) + assert.JSONEq(t, `"[PERSON] was charged twice"`, string(got.State)) + assert.Equal(t, "kev-latest", got.Model) + assert.JSONEq(t, `"John was charged twice"`, string(req.State), "the original request must not change") + require.Len(t, state.Snapshot(), 2) +} diff --git a/internal/guardrails/workflow_executor.go b/internal/guardrails/workflow_executor.go index 886e99fda..01920a1bc 100644 --- a/internal/guardrails/workflow_executor.go +++ b/internal/guardrails/workflow_executor.go @@ -2,6 +2,9 @@ package guardrails import ( "context" + "log/slog" + "strings" + "sync" "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/plugins" @@ -36,6 +39,36 @@ func (p *WorkflowRequestPatcher) PatchResponsesRequest(ctx context.Context, req return processGuardedResponses(ctx, p.chain(ctx), req) } +// PatchSystemOneRequest runs the prompt chain over a System One request's +// state. Edits a decision request has no place for, such as an injected +// system prompt, are dropped with a warning rather than failing the request: +// a guardrail scoped to every model is not wrong for System One models, it +// only has nothing to change there. +func (p *WorkflowRequestPatcher) PatchSystemOneRequest(ctx context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) { + if req == nil { + return nil, nil + } + return processGuarded(ctx, p.chain(ctx), req, "System One", exchange.FromSystemOneRequest, applySystemOneEdits) +} + +// systemOneDropsWarned records the kinds of dropped guardrail edits already +// logged at warning level. A guardrail scoped to every model edits every +// System One request the same way, so the misconfiguration is reported once +// per kind; repeats are logged at debug level. +var systemOneDropsWarned sync.Map + +func applySystemOneEdits(req *core.SystemOneRequest, prompt *pluginapi.Prompt) (*core.SystemOneRequest, error) { + if dropped := exchange.SystemOneUncarriedEdits(prompt); len(dropped) > 0 { + const message = "guardrail edits a System One request cannot carry were dropped; only the state is guarded" + if _, repeated := systemOneDropsWarned.LoadOrStore(strings.Join(dropped, ","), struct{}{}); repeated { + slog.Debug(message, "model", req.Model, "dropped", dropped) + } else { + slog.Warn(message+" (repeats are logged at debug level)", "model", req.Model, "dropped", dropped) + } + } + return exchange.ApplyToSystemOneRequest(req, prompt) +} + // EditsPromptContent reports whether the request's prompt chain holds an // instance that edits content, such as an anonymizing guardrail. A chained // Responses request only needs its stored history expanded into the input diff --git a/internal/mcpgateway/catalog.go b/internal/mcpgateway/catalog.go index f541ae930..5af49c17e 100644 --- a/internal/mcpgateway/catalog.go +++ b/internal/mcpgateway/catalog.go @@ -19,6 +19,10 @@ const namespaceSeparator = "_" // and prompts keep their original (un-prefixed) names and raw schemas; the // per-session server view applies namespacing. type catalog struct { + // discovered is every tool the upstream lists; tools is the subset the + // operator tool filters expose. Keeping both lets filter edits apply + // without re-listing and lets the admin inspector show hidden tools. + discovered []*mcp.Tool tools []*mcp.Tool prompts []*mcp.Prompt resources []*mcp.Resource @@ -33,6 +37,24 @@ func (c *catalog) toolCount() int { return len(c.tools) } +func (c *catalog) excludedToolCount() int { + if c == nil { + return 0 + } + return len(c.discovered) - len(c.tools) +} + +// withToolFilters returns a copy whose exposed tools reflect the given +// filters. Catalogs are immutable, so the copy shares every other list. +func (c *catalog) withToolFilters(allowed, disallowed []string) *catalog { + if c == nil { + return nil + } + next := *c + next.tools = filterTools(c.discovered, allowed, disallowed) + return &next +} + func (c *catalog) promptCount() int { if c == nil { return 0 @@ -52,24 +74,33 @@ func NamespacedName(server, name string) string { return server + namespaceSeparator + name } -// filterTools applies the operator-level allow/deny lists to original tool -// names and returns tools in deterministic (sorted) order. Deterministic -// ordering keeps downstream tools/list stable, which keeps provider prompt -// caches warm for clients that embed the tool list in prompts. -func filterTools(tools []*mcp.Tool, allowed, disallowed []string) []*mcp.Tool { - filtered := make([]*mcp.Tool, 0, len(tools)) +// discoverTools drops unnamed tools, normalizes schemas, and returns tools in +// deterministic (sorted) order. Deterministic ordering keeps downstream +// tools/list stable, which keeps provider prompt caches warm for clients that +// embed the tool list in prompts. +func discoverTools(tools []*mcp.Tool) []*mcp.Tool { + discovered := make([]*mcp.Tool, 0, len(tools)) for _, tool := range tools { if tool == nil || tool.Name == "" { continue } - if !toolAllowed(tool.Name, allowed, disallowed) { - continue - } - filtered = append(filtered, normalizeToolSchemas(tool)) + discovered = append(discovered, normalizeToolSchemas(tool)) } - slices.SortFunc(filtered, func(a, b *mcp.Tool) int { + slices.SortFunc(discovered, func(a, b *mcp.Tool) int { return strings.Compare(a.Name, b.Name) }) + return discovered +} + +// filterTools applies the operator-level allow/deny lists to original tool +// names, preserving input order. +func filterTools(tools []*mcp.Tool, allowed, disallowed []string) []*mcp.Tool { + filtered := make([]*mcp.Tool, 0, len(tools)) + for _, tool := range tools { + if toolAllowed(tool.Name, allowed, disallowed) { + filtered = append(filtered, tool) + } + } return filtered } @@ -136,10 +167,17 @@ func normalizeUserPaths(paths []string) []string { // userPathAllowed reports whether userPath falls inside one of the allowed // subtrees. An empty allow list means everyone. Mirrors virtual models. func userPathAllowed(userPath string, allowed []string) bool { - if len(allowed) == 0 { - return true + return len(allowed) == 0 || userPathWithin(userPath, allowed) +} + +// userPathWithin reports whether userPath falls inside one of the sorted, +// normalized subtrees. "/" matches every caller, including one without a +// user path; an empty list matches nobody. +func userPathWithin(userPath string, subtrees []string) bool { + if len(subtrees) == 0 { + return false } - if _, ok := slices.BinarySearch(allowed, "/"); ok { + if _, ok := slices.BinarySearch(subtrees, "/"); ok { return true } userPath, err := core.NormalizeUserPath(userPath) @@ -147,7 +185,7 @@ func userPathAllowed(userPath string, allowed []string) bool { return false } for _, candidate := range core.UserPathAncestors(userPath) { - if _, ok := slices.BinarySearch(allowed, candidate); ok { + if _, ok := slices.BinarySearch(subtrees, candidate); ok { return true } } diff --git a/internal/mcpgateway/catalog_test.go b/internal/mcpgateway/catalog_test.go index 2886fd85d..c788689d5 100644 --- a/internal/mcpgateway/catalog_test.go +++ b/internal/mcpgateway/catalog_test.go @@ -5,6 +5,7 @@ import ( "testing" "github.com/modelcontextprotocol/go-sdk/mcp" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -80,7 +81,7 @@ func TestFilterTools(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { t.Parallel() - filtered := filterTools(tools, tt.allowed, tt.disallowed) + filtered := filterTools(discoverTools(tools), tt.allowed, tt.disallowed) names := make([]string, 0, len(filtered)) for _, tool := range filtered { names = append(names, tool.Name) @@ -90,7 +91,7 @@ func TestFilterTools(t *testing.T) { } } -func TestFilterToolsNormalizesSchemasForDownstream(t *testing.T) { +func TestDiscoverToolsNormalizesSchemasForDownstream(t *testing.T) { t.Parallel() validInput := map[string]any{"type": "object", "properties": map[string]any{"q": map[string]any{"type": "string"}}} tools := []*mcp.Tool{ @@ -99,9 +100,9 @@ func TestFilterToolsNormalizesSchemasForDownstream(t *testing.T) { {Name: "valid", InputSchema: validInput, OutputSchema: map[string]any{"type": "object"}}, } - filtered := filterTools(tools, nil, nil) - byName := make(map[string]*mcp.Tool, len(filtered)) - for _, tool := range filtered { + discovered := discoverTools(tools) + byName := make(map[string]*mcp.Tool, len(discovered)) + for _, tool := range discovered { byName[tool.Name] = tool } for _, name := range []string{"missing", "invalid", "valid"} { @@ -112,6 +113,75 @@ func TestFilterToolsNormalizesSchemasForDownstream(t *testing.T) { require.True(t, isObjectSchema(byName["valid"].OutputSchema), "valid output schema = %#v, want preserved object", byName["valid"].OutputSchema) } +func TestCatalogWithToolFiltersKeepsDiscoveredTools(t *testing.T) { + t.Parallel() + base := &catalog{discovered: discoverTools([]*mcp.Tool{{Name: "write"}, {Name: "read"}})} + + filtered := base.withToolFilters(nil, []string{"write"}) + require.Len(t, filtered.tools, 1) + assert.Equal(t, "read", filtered.tools[0].Name) + assert.Equal(t, 1, filtered.excludedToolCount()) + assert.Len(t, filtered.discovered, 2) + assert.Empty(t, base.tools, "withToolFilters must not mutate the receiver") + + var missing *catalog + assert.Nil(t, missing.withToolFilters(nil, []string{"write"})) +} + +func TestCatalogToolRelaysExplicitSafetyHints(t *testing.T) { + t.Parallel() + destructive := true + tests := []struct { + name string + annotations *mcp.ToolAnnotations + wantReadOnly bool + wantDestructive bool + }{ + {name: "unannotated tool has no hints"}, + {name: "read-only hint", annotations: &mcp.ToolAnnotations{ReadOnlyHint: true}, wantReadOnly: true}, + {name: "explicit destructive hint", annotations: &mcp.ToolAnnotations{DestructiveHint: &destructive}, wantDestructive: true}, + {name: "read-only wins over destructive", annotations: &mcp.ToolAnnotations{ReadOnlyHint: true, DestructiveHint: &destructive}, wantReadOnly: true}, + {name: "implicit destructive default is not shown", annotations: &mcp.ToolAnnotations{}}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + feature := catalogTool(&mcp.Tool{Name: "tool", Annotations: tt.annotations}) + assert.Equal(t, tt.wantReadOnly, feature.ReadOnly) + assert.Equal(t, tt.wantDestructive, feature.Destructive) + }) + } +} + +func TestServerSpecVisibleTo(t *testing.T) { + t.Parallel() + tests := []struct { + name string + userPath string + allowed []string + disallowed []string + want bool + }{ + {name: "no scope admits everyone", userPath: "/x", want: true}, + {name: "excluded subtree is hidden", userPath: "/contractors/acme", disallowed: []string{"/contractors"}, want: false}, + {name: "sibling of excluded subtree stays visible", userPath: "/staff", disallowed: []string{"/contractors"}, want: true}, + {name: "exclusion wins inside an allowed subtree", userPath: "/eng/contractors", allowed: []string{"/eng"}, disallowed: []string{"/eng/contractors"}, want: false}, + {name: "rest of allowed subtree stays visible", userPath: "/eng/platform", allowed: []string{"/eng"}, disallowed: []string{"/eng/contractors"}, want: true}, + {name: "caller without user path is not in an excluded subtree", userPath: "", disallowed: []string{"/contractors"}, want: true}, + {name: "root exclusion hides from every caller", userPath: "", disallowed: []string{"/"}, want: false}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + spec := ServerSpec{ + UserPaths: normalizeUserPaths(tt.allowed), + DisallowedUserPaths: normalizeUserPaths(tt.disallowed), + } + assert.Equal(t, tt.want, spec.visibleTo(tt.userPath)) + }) + } +} + func TestUserPathAllowed(t *testing.T) { t.Parallel() tests := []struct { diff --git a/internal/mcpgateway/catalog_view.go b/internal/mcpgateway/catalog_view.go index f410afecc..86ee76ebb 100644 --- a/internal/mcpgateway/catalog_view.go +++ b/internal/mcpgateway/catalog_view.go @@ -3,10 +3,14 @@ package mcpgateway import "github.com/modelcontextprotocol/go-sdk/mcp" // CatalogFeature is one listed tool or prompt in a catalog view, using the -// upstream's original (un-prefixed) name. +// upstream's original (un-prefixed) name. ReadOnly and Destructive relay the +// upstream's tool annotations only when it set them explicitly; they are +// hints for operators choosing tools to exclude, not guarantees. type CatalogFeature struct { Name string `json:"name"` Description string `json:"description,omitempty"` + ReadOnly bool `json:"read_only,omitempty"` + Destructive bool `json:"destructive,omitempty"` } // CatalogResource is one listed resource in a catalog view. @@ -24,15 +28,17 @@ type CatalogTemplate struct { } // CatalogView is the admin-facing snapshot of what one upstream currently -// exposes through the gateway, after the operator tool filters were applied. +// exposes through the gateway. Tools are the ones the operator tool filters +// expose; ExcludedTools are discovered upstream but hidden by those filters. type CatalogView struct { - Server string `json:"server"` - Status ServerStatus `json:"status"` - Instructions string `json:"instructions,omitempty"` - Tools []CatalogFeature `json:"tools"` - Prompts []CatalogFeature `json:"prompts"` - Resources []CatalogResource `json:"resources"` - Templates []CatalogTemplate `json:"templates"` + Server string `json:"server"` + Status ServerStatus `json:"status"` + Instructions string `json:"instructions,omitempty"` + Tools []CatalogFeature `json:"tools"` + ExcludedTools []CatalogFeature `json:"excluded_tools"` + Prompts []CatalogFeature `json:"prompts"` + Resources []CatalogResource `json:"resources"` + Templates []CatalogTemplate `json:"templates"` } // Catalog returns the current catalog snapshot for one server, for the admin @@ -45,19 +51,29 @@ func (s *Service) Catalog(name string) (CatalogView, bool) { } snapshot, status := u.snapshot() view := CatalogView{ - Server: name, - Status: status, - Tools: []CatalogFeature{}, - Prompts: []CatalogFeature{}, - Resources: []CatalogResource{}, - Templates: []CatalogTemplate{}, + Server: name, + Status: status, + Tools: []CatalogFeature{}, + ExcludedTools: []CatalogFeature{}, + Prompts: []CatalogFeature{}, + Resources: []CatalogResource{}, + Templates: []CatalogTemplate{}, } if snapshot == nil { return view, true } view.Instructions = snapshot.instructions + exposed := make(map[string]struct{}, len(snapshot.tools)) for _, tool := range snapshot.tools { - view.Tools = append(view.Tools, catalogFeature(tool.Name, tool.Description, tool.Annotations, tool.Title)) + exposed[tool.Name] = struct{}{} + } + for _, tool := range snapshot.discovered { + feature := catalogTool(tool) + if _, ok := exposed[tool.Name]; ok { + view.Tools = append(view.Tools, feature) + } else { + view.ExcludedTools = append(view.ExcludedTools, feature) + } } for _, prompt := range snapshot.prompts { view.Prompts = append(view.Prompts, CatalogFeature{Name: prompt.Name, Description: prompt.Description}) @@ -71,16 +87,24 @@ func (s *Service) Catalog(name string) (CatalogView, bool) { return view, true } -// catalogFeature prefers the human-facing description, falling back to the +// catalogTool prefers the human-facing description, falling back to the // annotation title so the inspector never shows a blank row for tools that // only set display metadata. -func catalogFeature(name, description string, annotations *mcp.ToolAnnotations, title string) CatalogFeature { - feature := CatalogFeature{Name: name, Description: description} - if feature.Description == "" && title != "" { - feature.Description = title +func catalogTool(tool *mcp.Tool) CatalogFeature { + feature := CatalogFeature{Name: tool.Name, Description: tool.Description} + if feature.Description == "" && tool.Title != "" { + feature.Description = tool.Title + } + annotations := tool.Annotations + if annotations == nil { + return feature } - if feature.Description == "" && annotations != nil && annotations.Title != "" { + if feature.Description == "" && annotations.Title != "" { feature.Description = annotations.Title } + // The MCP spec defaults destructiveHint to true for non-read-only tools; + // only an explicit hint is shown, so unannotated tools stay unbadged. + feature.ReadOnly = annotations.ReadOnlyHint + feature.Destructive = !annotations.ReadOnlyHint && annotations.DestructiveHint != nil && *annotations.DestructiveHint return feature } diff --git a/internal/mcpgateway/manager.go b/internal/mcpgateway/manager.go index 8e8e1876b..60b35a01b 100644 --- a/internal/mcpgateway/manager.go +++ b/internal/mcpgateway/manager.go @@ -45,8 +45,11 @@ func NewManager(httpClient *http.Client) *Manager { // Apply reconciles the running upstreams with the desired specs: removed // servers are closed, new servers are added, changed servers are redialed. -// Unchanged servers keep their live session and catalog. Initial connects run -// asynchronously so startup and admin edits never block on upstream IO. +// Unchanged servers keep their live session and catalog; servers whose only +// change is the access policy (tool filters, user-path scopes) keep their +// session and apply the new policy in place. +// Initial connects run asynchronously so startup and admin edits never block +// on upstream IO. func (m *Manager) Apply(specs []ServerSpec) { desired := make(map[string]ServerSpec, len(specs)) for _, spec := range specs { @@ -58,14 +61,20 @@ func (m *Manager) Apply(specs []ServerSpec) { m.mu.Lock() for name, existing := range m.upstreams { - spec, keep := desired[name] - if keep && existing.spec.equal(spec) { - delete(desired, name) - continue + if spec, keep := desired[name]; keep { + current := existing.currentSpec() + if current.equal(spec) { + delete(desired, name) + continue + } + if current.withoutAccessPolicy().equal(spec.withoutAccessPolicy()) { + existing.setAccessPolicy(spec) + delete(desired, name) + continue + } } toClose = append(toClose, existing) delete(m.upstreams, name) - _ = spec } for name, spec := range desired { fresh := newUpstream(spec, m.httpClient) diff --git a/internal/mcpgateway/service.go b/internal/mcpgateway/service.go index 87e893247..205157366 100644 --- a/internal/mcpgateway/service.go +++ b/internal/mcpgateway/service.go @@ -3,6 +3,7 @@ package mcpgateway import ( "context" "crypto/rand" + "errors" "fmt" "log/slog" "net/http" @@ -61,6 +62,10 @@ type Service struct { stop chan struct{} } +// ErrServerNotVisible rejects a call from a session whose user path may no +// longer use the server. +var ErrServerNotVisible = errors.New("server is not available for this user path") + // sessionBinding pins a downstream MCP session to the user path it was // initialized under. Bearer auth still runs on every request; the binding // additionally stops a *different* principal from riding a leaked session ID, @@ -452,7 +457,7 @@ func (s *Service) visibleServers(scope requestScope) []ServerView { continue } } - if !userPathAllowed(scope.userPath, view.Spec.UserPaths) { + if !view.Spec.visibleTo(scope.userPath) { continue } visible = append(visible, view) @@ -465,7 +470,7 @@ func (s *Service) findVisibleServer(name, userPath string) (ServerView, bool) { if view.Spec.Name != name { continue } - if !userPathAllowed(userPath, view.Spec.UserPaths) { + if !view.Spec.visibleTo(userPath) { return ServerView{}, false } return view, true @@ -528,6 +533,9 @@ func (s *Service) registerTools(server *mcp.Server, upstreamName string, snapsho func (s *Service) toolHandler(upstreamName, toolName, exposedName, endpoint string) mcp.ToolHandler { return func(ctx context.Context, req *mcp.CallToolRequest) (*mcp.CallToolResult, error) { + if err := s.authorizeSession(req.Session, upstreamName); err != nil { + return nil, err + } started := time.Now() result, err := s.manager.CallTool(ctx, upstreamName, toolName, req.Params.Arguments) s.recordToolCall(req, upstreamName, exposedName, endpoint, started, result, err) @@ -555,6 +563,9 @@ func (s *Service) registerPrompts(server *mcp.Server, upstreamName string, snaps clone.Name = exposed originalName := prompt.Name server.AddPrompt(&clone, func(ctx context.Context, req *mcp.GetPromptRequest) (*mcp.GetPromptResult, error) { + if err := s.authorizeSession(req.Session, upstreamName); err != nil { + return nil, err + } params := *req.Params params.Name = originalName result, err := s.manager.GetPrompt(ctx, upstreamName, ¶ms) @@ -571,6 +582,9 @@ func (s *Service) registerPrompts(server *mcp.Server, upstreamName string, snaps // skipped with a warning rather than silently re-routed. func (s *Service) registerResources(server *mcp.Server, upstreamName string, snapshot *catalog, owners map[string]string) { read := func(ctx context.Context, req *mcp.ReadResourceRequest) (*mcp.ReadResourceResult, error) { + if err := s.authorizeSession(req.Session, upstreamName); err != nil { + return nil, err + } result, err := s.manager.ReadResource(ctx, upstreamName, req.Params) if err != nil { return nil, fmt.Errorf("mcp server %q: %w", upstreamName, err) @@ -598,6 +612,34 @@ func (s *Service) registerResources(server *mcp.Server, upstreamName string, sna } } +// authorizeSession re-checks, on every call, that the user path a session +// was bound to may still use upstreamName. Sessions snapshot their catalog at +// initialize, so without this a narrowed user_paths or a new +// disallowed_user_paths entry would not reach sessions that are already open. +// A session without a binding is rejected: bindings live as long as the +// session, so a missing one means it was deleted or expired mid-call, and +// guessing a user path could slip past disallowed_user_paths. +func (s *Service) authorizeSession(session *mcp.ServerSession, upstreamName string) error { + sessionID := "" + if session != nil { + sessionID = session.ID() + } + return s.authorizeSessionID(sessionID, upstreamName) +} + +func (s *Service) authorizeSessionID(sessionID, upstreamName string) error { + s.bindMu.Lock() + binding, ok := s.bindings[sessionID] + s.bindMu.Unlock() + if !ok { + return fmt.Errorf("mcp server %q: %w", upstreamName, ErrServerNotVisible) + } + if _, visible := s.findVisibleServer(upstreamName, binding.userPath); !visible { + return fmt.Errorf("mcp server %q: %w", upstreamName, ErrServerNotVisible) + } + return nil +} + // bindSession records the principal a new session was initialized under. func (s *Service) bindSession(sessionID, authKeyID, userPath, pinned string) { s.bindMu.Lock() diff --git a/internal/mcpgateway/service_test.go b/internal/mcpgateway/service_test.go index 0a017b57f..87a43f500 100644 --- a/internal/mcpgateway/service_test.go +++ b/internal/mcpgateway/service_test.go @@ -12,6 +12,7 @@ import ( "time" "github.com/modelcontextprotocol/go-sdk/mcp" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" "github.com/enterpilot/gomodel/config" @@ -411,6 +412,127 @@ func TestToolFiltersHideTools(t *testing.T) { require.Equal(t, "alpha_read", names[0]) } +func TestToolFilterEditAppliesInPlaceAndBlocksOpenSessions(t *testing.T) { + url := newTestUpstream(t, "alpha", func(server *mcp.Server) { + addEchoTool("read")(server) + addEchoTool("write")(server) + }) + spec := testSpec("alpha", url, nil) + service, gatewayURL := newTestService(t, nil, spec) + + openSession := connectClient(t, gatewayURL+"/mcp", nil) + require.Equal(t, []string{"alpha_read", "alpha_write"}, listToolNames(t, openSession)) + + before, ok := service.manager.get("alpha") + require.True(t, ok) + upstreamSession := before.session + + spec.DisallowedTools = []string{"write"} + service.manager.Apply([]ServerSpec{spec}) + + after, ok := service.manager.get("alpha") + require.True(t, ok) + assert.Same(t, before, after, "a filter-only edit must not replace the upstream") + assert.Same(t, upstreamSession, after.session, "a filter-only edit must not redial") + view := after.view() + assert.Equal(t, StatusConnected, view.Status) + assert.Equal(t, 1, view.ToolCount) + assert.Equal(t, 1, view.ExcludedToolCount) + assert.Equal(t, []string{"write"}, view.Spec.DisallowedTools) + + service.manager.Apply([]ServerSpec{spec}) + unchanged, ok := service.manager.get("alpha") + require.True(t, ok) + assert.Same(t, after, unchanged, "re-applying an identical spec must keep the upstream") + + _, err := openSession.CallTool(context.Background(), &mcp.CallToolParams{Name: "alpha_write"}) + require.Error(t, err, "an excluded tool must not be callable from a session opened before the edit") + assert.Contains(t, err.Error(), "excluded") + + fresh := connectClient(t, gatewayURL+"/mcp", nil) + assert.Equal(t, []string{"alpha_read"}, listToolNames(t, fresh)) + + catalog, ok := service.Catalog("alpha") + require.True(t, ok) + require.Len(t, catalog.Tools, 1) + assert.Equal(t, "read", catalog.Tools[0].Name) + require.Len(t, catalog.ExcludedTools, 1) + assert.Equal(t, "write", catalog.ExcludedTools[0].Name) +} + +func TestDisallowedUserPathsHideServers(t *testing.T) { + alphaURL := newTestUpstream(t, "alpha", addEchoTool("echo")) + betaURL := newTestUpstream(t, "beta", addEchoTool("search")) + _, gatewayURL := newTestService(t, nil, + testSpec("alpha", alphaURL, nil), + testSpec("beta", betaURL, func(spec *ServerSpec) { + spec.UserPaths = []string{"/eng"} + spec.DisallowedUserPaths = []string{"/eng/contractors"} + }), + ) + + contractor := connectClient(t, gatewayURL+"/mcp", map[string]string{core.UserPathHeader: "/eng/contractors/acme"}) + assert.Equal(t, []string{"alpha_echo"}, listToolNames(t, contractor)) + + staff := connectClient(t, gatewayURL+"/mcp", map[string]string{core.UserPathHeader: "/eng/platform"}) + assert.Equal(t, []string{"alpha_echo", "beta_search"}, listToolNames(t, staff)) + + req, err := http.NewRequest(http.MethodPost, gatewayURL+"/mcp/beta", strings.NewReader(`{}`)) + require.NoError(t, err) + req.Header.Set(core.UserPathHeader, "/eng/contractors") + resp, err := http.DefaultClient.Do(req) + require.NoError(t, err) + defer resp.Body.Close() + assert.Equal(t, http.StatusNotFound, resp.StatusCode) +} + +func TestUserPathEditAppliesInPlaceAndReachesOpenSessions(t *testing.T) { + url := newTestUpstream(t, "alpha", addEchoTool("echo")) + spec := testSpec("alpha", url, nil) + service, gatewayURL := newTestService(t, nil, spec) + + session := connectClient(t, gatewayURL+"/mcp", map[string]string{core.UserPathHeader: "/contractors"}) + require.Equal(t, []string{"alpha_echo"}, listToolNames(t, session)) + + before, ok := service.manager.get("alpha") + require.True(t, ok) + upstreamSession := before.session + + spec.DisallowedUserPaths = []string{"/contractors"} + service.manager.Apply([]ServerSpec{spec}) + + after, ok := service.manager.get("alpha") + require.True(t, ok) + assert.Same(t, before, after, "a user-path edit must not replace the upstream") + assert.Same(t, upstreamSession, after.session, "a user-path edit must not redial") + + _, err := session.CallTool(context.Background(), &mcp.CallToolParams{Name: "alpha_echo"}) + require.Error(t, err, "a session opened before the exclusion must lose access") + assert.Contains(t, err.Error(), "not available for this user path") + + other := connectClient(t, gatewayURL+"/mcp", map[string]string{core.UserPathHeader: "/staff"}) + result, err := other.CallTool(context.Background(), &mcp.CallToolParams{Name: "alpha_echo"}) + require.NoError(t, err) + assert.False(t, result.IsError) +} + +func TestAuthorizeSessionFailsClosedWithoutBinding(t *testing.T) { + url := newTestUpstream(t, "alpha", addEchoTool("echo")) + service, _ := newTestService(t, nil, testSpec("alpha", url, func(spec *ServerSpec) { + spec.DisallowedUserPaths = []string{"/contractors"} + })) + + // A binding removed mid-call (DELETE or expiry) must not be read as a + // caller without a user path, which the denylist alone would admit. + err := service.authorizeSessionID("deleted-session", "alpha") + require.ErrorIs(t, err, ErrServerNotVisible) + + service.bindSession("live", "", "/staff", "") + require.NoError(t, service.authorizeSessionID("live", "alpha")) + service.bindSession("contractor", "", "/contractors/acme", "") + require.ErrorIs(t, service.authorizeSessionID("contractor", "alpha"), ErrServerNotVisible) +} + func TestSessionBindingRejectsForeignUserPath(t *testing.T) { alphaURL := newTestUpstream(t, "alpha", addEchoTool("echo")) _, gatewayURL := newTestService(t, nil, testSpec("alpha", alphaURL, nil)) diff --git a/internal/mcpgateway/store.go b/internal/mcpgateway/store.go index 1f740be5b..694443873 100644 --- a/internal/mcpgateway/store.go +++ b/internal/mcpgateway/store.go @@ -35,17 +35,18 @@ type ManagedServer struct { Name string `json:"slug"` DisplayName string `json:"name"` - URL string `json:"url"` - Transport string `json:"transport"` - Headers map[string]string `json:"headers,omitempty"` - Description string `json:"description,omitempty"` - Enabled bool `json:"enabled"` - AllowedTools []string `json:"allowed_tools,omitempty"` - DisallowedTools []string `json:"disallowed_tools,omitempty"` - UserPaths []string `json:"user_paths,omitempty"` - ToolTimeoutSeconds int `json:"tool_timeout_seconds,omitempty"` - CreatedAt time.Time `json:"created_at"` - UpdatedAt time.Time `json:"updated_at"` + URL string `json:"url"` + Transport string `json:"transport"` + Headers map[string]string `json:"headers,omitempty"` + Description string `json:"description,omitempty"` + Enabled bool `json:"enabled"` + AllowedTools []string `json:"allowed_tools,omitempty"` + DisallowedTools []string `json:"disallowed_tools,omitempty"` + UserPaths []string `json:"user_paths,omitempty"` + DisallowedUserPaths []string `json:"disallowed_user_paths,omitempty"` + ToolTimeoutSeconds int `json:"tool_timeout_seconds,omitempty"` + CreatedAt time.Time `json:"created_at"` + UpdatedAt time.Time `json:"updated_at"` } // Validate checks the row against the same rules as declarative config, @@ -70,11 +71,12 @@ func (m *ManagedServer) Validate() error { return fmt.Errorf("tool_timeout_seconds must not be negative") } cfg := config.MCPServerConfig{ - URL: m.URL, - Transport: m.Transport, - Headers: m.Headers, - UserPaths: m.UserPaths, - ToolTimeout: time.Duration(m.ToolTimeoutSeconds) * time.Second, + URL: m.URL, + Transport: m.Transport, + Headers: m.Headers, + UserPaths: m.UserPaths, + DisallowedUserPaths: m.DisallowedUserPaths, + ToolTimeout: time.Duration(m.ToolTimeoutSeconds) * time.Second, } if err := config.ValidateMCPServerConfig(&cfg); err != nil { return err @@ -82,6 +84,7 @@ func (m *ManagedServer) Validate() error { m.Transport = cfg.Transport m.URL = cfg.URL m.UserPaths = cfg.UserPaths + m.DisallowedUserPaths = cfg.DisallowedUserPaths return nil } @@ -92,18 +95,19 @@ func (m ManagedServer) Spec() ServerSpec { timeout = config.DefaultMCPToolTimeout } return ServerSpec{ - Name: m.Name, - DisplayName: m.DisplayName, - URL: m.URL, - Transport: m.Transport, - Headers: maps.Clone(m.Headers), - Description: m.Description, - Enabled: m.Enabled, - AllowedTools: slices.Clone(m.AllowedTools), - DisallowedTools: slices.Clone(m.DisallowedTools), - UserPaths: normalizeUserPaths(m.UserPaths), - ToolTimeout: timeout, - Managed: false, + Name: m.Name, + DisplayName: m.DisplayName, + URL: m.URL, + Transport: m.Transport, + Headers: maps.Clone(m.Headers), + Description: m.Description, + Enabled: m.Enabled, + AllowedTools: slices.Clone(m.AllowedTools), + DisallowedTools: slices.Clone(m.DisallowedTools), + UserPaths: normalizeUserPaths(m.UserPaths), + DisallowedUserPaths: normalizeUserPaths(m.DisallowedUserPaths), + ToolTimeout: timeout, + Managed: false, } } diff --git a/internal/mcpgateway/store_mongodb.go b/internal/mcpgateway/store_mongodb.go index 3aff033f5..bc1722f8d 100644 --- a/internal/mcpgateway/store_mongodb.go +++ b/internal/mcpgateway/store_mongodb.go @@ -13,19 +13,20 @@ import ( ) type mongoMCPServerDocument struct { - ID string `bson:"_id"` - DisplayName string `bson:"display_name,omitempty"` - URL string `bson:"url,omitempty"` - Transport string `bson:"transport,omitempty"` - Headers map[string]string `bson:"headers,omitempty"` - Description string `bson:"description,omitempty"` - Enabled bool `bson:"enabled"` - AllowedTools []string `bson:"allowed_tools,omitempty"` - DisallowedTools []string `bson:"disallowed_tools,omitempty"` - UserPaths []string `bson:"user_paths,omitempty"` - ToolTimeoutSeconds int `bson:"tool_timeout_seconds,omitempty"` - CreatedAt time.Time `bson:"created_at"` - UpdatedAt time.Time `bson:"updated_at"` + ID string `bson:"_id"` + DisplayName string `bson:"display_name,omitempty"` + URL string `bson:"url,omitempty"` + Transport string `bson:"transport,omitempty"` + Headers map[string]string `bson:"headers,omitempty"` + Description string `bson:"description,omitempty"` + Enabled bool `bson:"enabled"` + AllowedTools []string `bson:"allowed_tools,omitempty"` + DisallowedTools []string `bson:"disallowed_tools,omitempty"` + UserPaths []string `bson:"user_paths,omitempty"` + DisallowedUserPaths []string `bson:"disallowed_user_paths,omitempty"` + ToolTimeoutSeconds int `bson:"tool_timeout_seconds,omitempty"` + CreatedAt time.Time `bson:"created_at"` + UpdatedAt time.Time `bson:"updated_at"` } type mongoMCPServerIDFilter struct { @@ -94,17 +95,18 @@ func (s *MongoDBStore) Upsert(ctx context.Context, server ManagedServer) error { stampUpsert(&server) update := bson.M{ "$set": bson.M{ - "display_name": server.DisplayName, - "url": server.URL, - "transport": server.Transport, - "headers": server.Headers, - "description": server.Description, - "enabled": server.Enabled, - "allowed_tools": server.AllowedTools, - "disallowed_tools": server.DisallowedTools, - "user_paths": server.UserPaths, - "tool_timeout_seconds": server.ToolTimeoutSeconds, - "updated_at": server.UpdatedAt, + "display_name": server.DisplayName, + "url": server.URL, + "transport": server.Transport, + "headers": server.Headers, + "description": server.Description, + "enabled": server.Enabled, + "allowed_tools": server.AllowedTools, + "disallowed_tools": server.DisallowedTools, + "user_paths": server.UserPaths, + "disallowed_user_paths": server.DisallowedUserPaths, + "tool_timeout_seconds": server.ToolTimeoutSeconds, + "updated_at": server.UpdatedAt, }, "$setOnInsert": bson.M{ "created_at": server.CreatedAt, @@ -159,5 +161,8 @@ func managedServerFromMongo(doc mongoMCPServerDocument) ManagedServer { if len(doc.UserPaths) > 0 { server.UserPaths = append([]string(nil), doc.UserPaths...) } + if len(doc.DisallowedUserPaths) > 0 { + server.DisallowedUserPaths = append([]string(nil), doc.DisallowedUserPaths...) + } return server } diff --git a/internal/mcpgateway/store_sql.go b/internal/mcpgateway/store_sql.go index cc1190439..c6d925e1a 100644 --- a/internal/mcpgateway/store_sql.go +++ b/internal/mcpgateway/store_sql.go @@ -26,6 +26,7 @@ var sqlTable = `CREATE TABLE IF NOT EXISTS mcp_servers ( allowed_tools TEXT NOT NULL DEFAULT '[]', disallowed_tools TEXT NOT NULL DEFAULT '[]', user_paths TEXT NOT NULL DEFAULT '[]', + disallowed_user_paths TEXT NOT NULL DEFAULT '[]', tool_timeout_seconds INTEGER NOT NULL DEFAULT 0, created_at ` + sqlx.TypeInt64 + ` NOT NULL, updated_at ` + sqlx.TypeInt64 + ` NOT NULL @@ -39,10 +40,11 @@ var sqlIndexes = []string{ // sqlMigrations backfill columns added after the table's first release. var sqlMigrations = []string{ `ALTER TABLE mcp_servers ADD COLUMN display_name TEXT NOT NULL DEFAULT ''`, + `ALTER TABLE mcp_servers ADD COLUMN disallowed_user_paths TEXT NOT NULL DEFAULT '[]'`, } const selectMCPServerColumns = `name, display_name, url, transport, headers, description, enabled, ` + - `allowed_tools, disallowed_tools, user_paths, tool_timeout_seconds, created_at, updated_at` + `allowed_tools, disallowed_tools, user_paths, disallowed_user_paths, tool_timeout_seconds, created_at, updated_at` // NewSQLStore creates the mcp_servers table and indexes if needed. func NewSQLStore(ctx context.Context, db sqlx.DB) (*SQLStore, error) { @@ -119,12 +121,16 @@ func (s *SQLStore) Upsert(ctx context.Context, server ManagedServer) error { if err != nil { return err } + disallowedPathsJSON, err := encodeJSONList(server.DisallowedUserPaths) + if err != nil { + return err + } _, err = s.db.Exec(ctx, ` INSERT INTO mcp_servers ( name, display_name, url, transport, headers, description, enabled, - allowed_tools, disallowed_tools, user_paths, tool_timeout_seconds, created_at, updated_at + allowed_tools, disallowed_tools, user_paths, disallowed_user_paths, tool_timeout_seconds, created_at, updated_at ) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) ON CONFLICT(name) DO UPDATE SET display_name = excluded.display_name, url = excluded.url, @@ -135,6 +141,7 @@ func (s *SQLStore) Upsert(ctx context.Context, server ManagedServer) error { allowed_tools = excluded.allowed_tools, disallowed_tools = excluded.disallowed_tools, user_paths = excluded.user_paths, + disallowed_user_paths = excluded.disallowed_user_paths, tool_timeout_seconds = excluded.tool_timeout_seconds, updated_at = excluded.updated_at `, @@ -148,6 +155,7 @@ func (s *SQLStore) Upsert(ctx context.Context, server ManagedServer) error { allowedJSON, disallowedJSON, pathsJSON, + disallowedPathsJSON, server.ToolTimeoutSeconds, server.CreatedAt.Unix(), server.UpdatedAt.Unix(), @@ -175,7 +183,7 @@ func (s *SQLStore) Close() error { func scanSQLMCPServer(scanner sqlx.Row) (ManagedServer, error) { var server ManagedServer - var headers, allowed, disallowed, userPaths []byte + var headers, allowed, disallowed, userPaths, disallowedUserPaths []byte var createdAt, updatedAt int64 if err := scanner.Scan( &server.Name, @@ -188,6 +196,7 @@ func scanSQLMCPServer(scanner sqlx.Row) (ManagedServer, error) { &allowed, &disallowed, &userPaths, + &disallowedUserPaths, &server.ToolTimeoutSeconds, &createdAt, &updatedAt, @@ -207,6 +216,9 @@ func scanSQLMCPServer(scanner sqlx.Row) (ManagedServer, error) { if server.UserPaths, err = decodeJSONList(userPaths); err != nil { return ManagedServer{}, err } + if server.DisallowedUserPaths, err = decodeJSONList(disallowedUserPaths); err != nil { + return ManagedServer{}, err + } if server.DisplayName == "" { server.DisplayName = server.Name } diff --git a/internal/mcpgateway/store_sql_test.go b/internal/mcpgateway/store_sql_test.go index 6741924b5..f79e7c056 100644 --- a/internal/mcpgateway/store_sql_test.go +++ b/internal/mcpgateway/store_sql_test.go @@ -38,17 +38,18 @@ func TestStoreRoundTrip(t *testing.T) { ctx := context.Background() server := ManagedServer{ - Name: "github", - DisplayName: "GitHub MCP", - URL: "https://api.githubcopilot.com/mcp", - Transport: "http", - Headers: map[string]string{"Authorization": "Bearer secret"}, - Description: "GitHub tools", - Enabled: true, - AllowedTools: []string{"create_issue"}, - DisallowedTools: []string{"delete_repo"}, - UserPaths: []string{"/team-a"}, - ToolTimeoutSeconds: 45, + Name: "github", + DisplayName: "GitHub MCP", + URL: "https://api.githubcopilot.com/mcp", + Transport: "http", + Headers: map[string]string{"Authorization": "Bearer secret"}, + Description: "GitHub tools", + Enabled: true, + AllowedTools: []string{"create_issue"}, + DisallowedTools: []string{"delete_repo"}, + UserPaths: []string{"/team-a"}, + DisallowedUserPaths: []string{"/team-a/contractors"}, + ToolTimeoutSeconds: 45, } err := store.Upsert(ctx, server) require.NoError(t, err) @@ -64,6 +65,7 @@ func TestStoreRoundTrip(t *testing.T) { require.Len(t, got.AllowedTools, 1) require.Equal(t, "create_issue", got.AllowedTools[0]) require.Equal(t, 45, got.ToolTimeoutSeconds) + require.Equal(t, []string{"/team-a/contractors"}, got.DisallowedUserPaths) require.False(t, got.CreatedAt.IsZero()) require.False(t, got.UpdatedAt.IsZero(), "Get() timestamps not stamped: %+v", got) @@ -119,6 +121,7 @@ func TestSQLStoreMigratesDisplayName(t *testing.T) { server, err := store.Get(ctx, "linear") require.NoError(t, err) require.Equal(t, "linear", server.DisplayName) + require.Empty(t, server.DisallowedUserPaths) }) } diff --git a/internal/mcpgateway/types.go b/internal/mcpgateway/types.go index e6a908e32..78b1a8b2a 100644 --- a/internal/mcpgateway/types.go +++ b/internal/mcpgateway/types.go @@ -46,7 +46,10 @@ type ServerSpec struct { AllowedTools []string DisallowedTools []string UserPaths []string - ToolTimeout time.Duration + // DisallowedUserPaths hides the server from these subtrees; it wins over + // UserPaths. + DisallowedUserPaths []string + ToolTimeout time.Duration // Managed marks specs declared in config.yaml / MCP_SERVERS. They override // admin-store rows with the same name and are read-only in the dashboard. @@ -56,21 +59,22 @@ type ServerSpec struct { // SpecFromConfig converts one declarative config entry into a runtime spec. func SpecFromConfig(name string, cfg config.MCPServerConfig) ServerSpec { return ServerSpec{ - Name: name, - DisplayName: name, - URL: cfg.URL, - Transport: cfg.Transport, - Headers: maps.Clone(cfg.Headers), - Command: cfg.Command, - Args: slices.Clone(cfg.Args), - Env: maps.Clone(cfg.Env), - Description: cfg.Description, - Enabled: config.MCPServerEnabled(cfg), - AllowedTools: slices.Clone(cfg.AllowedTools), - DisallowedTools: slices.Clone(cfg.DisallowedTools), - UserPaths: normalizeUserPaths(cfg.UserPaths), - ToolTimeout: cfg.ToolTimeout, - Managed: true, + Name: name, + DisplayName: name, + URL: cfg.URL, + Transport: cfg.Transport, + Headers: maps.Clone(cfg.Headers), + Command: cfg.Command, + Args: slices.Clone(cfg.Args), + Env: maps.Clone(cfg.Env), + Description: cfg.Description, + Enabled: config.MCPServerEnabled(cfg), + AllowedTools: slices.Clone(cfg.AllowedTools), + DisallowedTools: slices.Clone(cfg.DisallowedTools), + UserPaths: normalizeUserPaths(cfg.UserPaths), + DisallowedUserPaths: normalizeUserPaths(cfg.DisallowedUserPaths), + ToolTimeout: cfg.ToolTimeout, + Managed: true, } } @@ -90,18 +94,39 @@ func (s ServerSpec) equal(other ServerSpec) bool { slices.Equal(s.AllowedTools, other.AllowedTools) && slices.Equal(s.DisallowedTools, other.DisallowedTools) && slices.Equal(s.UserPaths, other.UserPaths) && + slices.Equal(s.DisallowedUserPaths, other.DisallowedUserPaths) && s.ToolTimeout == other.ToolTimeout && s.Managed == other.Managed } +// withoutAccessPolicy clears the gateway-side access policy (tool filters and +// user-path scopes), so callers can tell a policy-only edit, applied in place, +// from one that changes the upstream connection and needs a redial. +func (s ServerSpec) withoutAccessPolicy() ServerSpec { + s.AllowedTools = nil + s.DisallowedTools = nil + s.UserPaths = nil + s.DisallowedUserPaths = nil + return s +} + +// visibleTo reports whether a caller with userPath may see and use this +// server: inside UserPaths (empty means everyone) and outside every +// DisallowedUserPaths subtree. +func (s ServerSpec) visibleTo(userPath string) bool { + return userPathAllowed(userPath, s.UserPaths) && !userPathWithin(userPath, s.DisallowedUserPaths) +} + // ServerView is a point-in-time snapshot of one upstream for admin and // dashboard consumption. type ServerView struct { - Spec ServerSpec - Status ServerStatus - LastError string - ToolCount int - PromptCount int + Spec ServerSpec + Status ServerStatus + LastError string + ToolCount int + // ExcludedToolCount counts discovered tools hidden by the tool filters. + ExcludedToolCount int + PromptCount int // ResourceCount includes resource templates. ResourceCount int ConnectedAt time.Time diff --git a/internal/mcpgateway/types_test.go b/internal/mcpgateway/types_test.go new file mode 100644 index 000000000..01245fa5d --- /dev/null +++ b/internal/mcpgateway/types_test.go @@ -0,0 +1,48 @@ +package mcpgateway + +import ( + "testing" + "time" + + "github.com/stretchr/testify/assert" + + "github.com/enterpilot/gomodel/config" +) + +func TestSpecFromConfigMapsAccessPolicy(t *testing.T) { + t.Parallel() + disabled := false + cfg := config.MCPServerConfig{ + URL: "https://mcp.example.com/mcp", + Transport: config.MCPTransportHTTP, + Headers: map[string]string{"Authorization": "Bearer x"}, + Description: "GitHub tools", + Enabled: &disabled, + AllowedTools: []string{"search"}, + DisallowedTools: []string{"delete_repo"}, + UserPaths: []string{"/eng/", "/eng"}, + DisallowedUserPaths: []string{" eng/contractors "}, + ToolTimeout: 45 * time.Second, + } + + spec := SpecFromConfig("github", cfg) + + assert.Equal(t, "github", spec.Name) + assert.Equal(t, "github", spec.DisplayName) + assert.Equal(t, "https://mcp.example.com/mcp", spec.URL) + assert.Equal(t, "GitHub tools", spec.Description) + assert.False(t, spec.Enabled) + assert.True(t, spec.Managed) + assert.Equal(t, 45*time.Second, spec.ToolTimeout) + assert.Equal(t, []string{"search"}, spec.AllowedTools) + assert.Equal(t, []string{"delete_repo"}, spec.DisallowedTools) + assert.Equal(t, []string{"/eng"}, spec.UserPaths) + assert.Equal(t, []string{"/eng/contractors"}, spec.DisallowedUserPaths) + assert.False(t, spec.visibleTo("/eng/contractors/acme"), "config-declared exclusions must reach the runtime spec") + + // The spec owns its lists: later edits to the config must not leak in. + cfg.AllowedTools[0] = "mutated" + cfg.Headers["Authorization"] = "mutated" + assert.Equal(t, []string{"search"}, spec.AllowedTools) + assert.Equal(t, "Bearer x", spec.Headers["Authorization"]) +} diff --git a/internal/mcpgateway/upstream.go b/internal/mcpgateway/upstream.go index 708982b41..5ce883296 100644 --- a/internal/mcpgateway/upstream.go +++ b/internal/mcpgateway/upstream.go @@ -28,6 +28,9 @@ const connectTimeout = 15 * time.Second // listTimeout bounds one full catalog listing pass. const listTimeout = 30 * time.Second +// ErrToolExcluded rejects a call to a tool the operator filters hide. +var ErrToolExcluded = errors.New("tool is excluded by the gateway tool filters") + // upstream owns the client session and catalog snapshot for one server. One // shared session serves all downstream sessions: v1 forwards no per-user // upstream credentials and bridges no server-to-client requests, so @@ -65,16 +68,47 @@ func (u *upstream) view() ServerView { u.stateMu.Lock() defer u.stateMu.Unlock() return ServerView{ - Spec: u.spec, - Status: u.status, - LastError: u.lastErr, - ToolCount: u.catalog.toolCount(), - PromptCount: u.catalog.promptCount(), - ResourceCount: u.catalog.resourceCount(), - ConnectedAt: u.connectedAt, + Spec: u.spec, + Status: u.status, + LastError: u.lastErr, + ToolCount: u.catalog.toolCount(), + ExcludedToolCount: u.catalog.excludedToolCount(), + PromptCount: u.catalog.promptCount(), + ResourceCount: u.catalog.resourceCount(), + ConnectedAt: u.connectedAt, } } +// currentSpec returns the spec under the state lock; the access policy can +// change in place (setAccessPolicy), so readers of it must not race it. +func (u *upstream) currentSpec() ServerSpec { + u.stateMu.Lock() + defer u.stateMu.Unlock() + return u.spec +} + +// setAccessPolicy swaps the tool filters and user-path scopes from spec and +// re-filters the cached catalog without touching the upstream session. They +// are gateway policy, so changing them never needs a redial. +func (u *upstream) setAccessPolicy(spec ServerSpec) { + u.stateMu.Lock() + defer u.stateMu.Unlock() + u.spec.AllowedTools = spec.AllowedTools + u.spec.DisallowedTools = spec.DisallowedTools + u.spec.UserPaths = spec.UserPaths + u.spec.DisallowedUserPaths = spec.DisallowedUserPaths + u.catalog = u.catalog.withToolFilters(spec.AllowedTools, spec.DisallowedTools) +} + +// toolExposed reports whether the current filters expose name. Downstream +// sessions snapshot their tool list at initialize, so calls re-check here to +// make an exclusion effective for sessions that are already open. +func (u *upstream) toolExposed(name string) bool { + u.stateMu.Lock() + defer u.stateMu.Unlock() + return toolAllowed(name, u.spec.AllowedTools, u.spec.DisallowedTools) +} + // snapshot returns the current catalog (nil when never listed). func (u *upstream) snapshot() (*catalog, ServerStatus) { u.stateMu.Lock() @@ -113,7 +147,7 @@ func (u *upstream) refresh(ctx context.Context) error { } u.stateMu.Lock() - u.catalog = fresh + u.catalog = fresh.withToolFilters(u.spec.AllowedTools, u.spec.DisallowedTools) u.status = StatusConnected u.lastErr = "" u.stateMu.Unlock() @@ -366,7 +400,8 @@ func requestOrigin(raw string) string { // list rebuilds the catalog from the upstream's declared capabilities. Valid // tool schemas and metadata pass through untouched; malformed schemas are -// made safe for the stricter downstream SDK. +// made safe for the stricter downstream SDK. Tool filters are applied by the +// caller, under the state lock. func (u *upstream) list(ctx context.Context, session *mcp.ClientSession) (*catalog, error) { fresh := &catalog{} init := session.InitializeResult() @@ -384,7 +419,7 @@ func (u *upstream) list(ctx context.Context, session *mcp.ClientSession) (*catal } tools = append(tools, tool) } - fresh.tools = filterTools(tools, u.spec.AllowedTools, u.spec.DisallowedTools) + fresh.discovered = discoverTools(tools) } if caps != nil && caps.Prompts != nil { @@ -438,6 +473,9 @@ func (u *upstream) list(ctx context.Context, session *mcp.ClientSession) (*catal // callTool forwards one tools/call with the original tool name. A session // that died since the last call is redialed once, transparently. func (u *upstream) callTool(ctx context.Context, name string, args json.RawMessage) (*mcp.CallToolResult, error) { + if !u.toolExposed(name) { + return nil, fmt.Errorf("%w: %q", ErrToolExcluded, name) + } params := &mcp.CallToolParams{Name: name} if len(args) > 0 { params.Arguments = args diff --git a/internal/plugins/exchange/systemone_request.go b/internal/plugins/exchange/systemone_request.go new file mode 100644 index 000000000..eaef9b685 --- /dev/null +++ b/internal/plugins/exchange/systemone_request.go @@ -0,0 +1,117 @@ +package exchange + +import ( + "bytes" + "fmt" + "sort" + + "github.com/goccy/go-json" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/pluginapi" +) + +// SystemOneStateMessageID is the ID of the user message built from a System +// One request's state. +const SystemOneStateMessageID = "state" + +// FromSystemOneRequest builds the unified prompt for a System One decision +// request. The state is the only content a caller supplies per request, so it +// becomes the prompt's single user message: a string state as its text, any +// other JSON value as its encoded JSON. The questions are the application's +// fixed schema and are not exposed. +func FromSystemOneRequest(req *core.SystemOneRequest) (*pluginapi.Prompt, error) { + if req == nil { + return nil, fmt.Errorf("exchange: nil System One request") + } + raw, err := json.Marshal(req) + if err != nil { + return nil, fmt.Errorf("exchange: encode System One request: %w", err) + } + p := &pluginapi.Prompt{ + Raw: raw, + Messages: []pluginapi.Message{pluginapi.TextMessage(pluginapi.RoleUser, systemOneStateText(req.State))}, + Params: pluginapi.Params{Model: req.Model}, + } + p.Messages[0].ID = SystemOneStateMessageID + p.Reset() + return p, nil +} + +func systemOneStateText(state json.RawMessage) string { + var text string + if trimmed := bytes.TrimSpace(state); len(trimmed) > 0 && trimmed[0] == '"' && json.Unmarshal(trimmed, &text) == nil { + return text + } + return string(state) +} + +// ApplyToSystemOneRequest returns a copy of original with the state +// message's edits applied. A string state stays a string; a state sent as +// another JSON value must still be valid JSON after the edit. Removing the +// state is an error, since the request would have nothing to decide on. +// Edits System One has no place for, such as inserted messages or parameter +// changes, are not applied; SystemOneUncarriedEdits lists them. +func ApplyToSystemOneRequest(original *core.SystemOneRequest, p *pluginapi.Prompt) (*core.SystemOneRequest, error) { + if original == nil || p == nil { + return nil, fmt.Errorf("exchange: nil System One request or prompt") + } + result := *original + switch p.Changes().Messages[SystemOneStateMessageID] { + case "": + return &result, nil + case pluginapi.ChangeRemoved: + return nil, fmt.Errorf("exchange: the System One state was removed") + } + msg := p.Message(SystemOneStateMessageID) + if msg == nil { + return nil, fmt.Errorf("exchange: the System One state was removed") + } + text := msg.Text() + // A missing state becomes a string; null, numbers, and records keep + // their JSON type (json.Unmarshal would accept null into a string). + state := bytes.TrimSpace(original.State) + if len(state) == 0 || state[0] == '"' { + encoded, err := json.Marshal(text) + if err != nil { + return nil, fmt.Errorf("exchange: encode System One state: %w", err) + } + result.State = encoded + return &result, nil + } + if !json.Valid([]byte(text)) { + return nil, fmt.Errorf("exchange: the edited System One state is no longer valid JSON") + } + result.State = json.RawMessage(text) + return &result, nil +} + +// SystemOneUncarriedEdits describes the kinds of prompt edits +// ApplyToSystemOneRequest does not apply ("inserted message", parameter +// "temperature"), deduplicated and sorted, or nil when every edit was +// carried. Message IDs are left out: they differ per request and name +// nothing an operator can act on. +func SystemOneUncarriedEdits(p *pluginapi.Prompt) []string { + if p == nil { + return nil + } + changes := p.Changes() + seen := map[string]struct{}{} + for id, kind := range changes.Messages { + if id != SystemOneStateMessageID { + seen[string(kind)+" message"] = struct{}{} + } + } + for name := range changes.Params { + seen[fmt.Sprintf("parameter %q", name)] = struct{}{} + } + if len(seen) == 0 { + return nil + } + uncarried := make([]string, 0, len(seen)) + for description := range seen { + uncarried = append(uncarried, description) + } + sort.Strings(uncarried) + return uncarried +} diff --git a/internal/plugins/exchange/systemone_request_test.go b/internal/plugins/exchange/systemone_request_test.go new file mode 100644 index 000000000..4cf0d4078 --- /dev/null +++ b/internal/plugins/exchange/systemone_request_test.go @@ -0,0 +1,129 @@ +package exchange + +import ( + "testing" + + "github.com/goccy/go-json" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/pluginapi" +) + +func TestFromSystemOneRequestExposesStateAsUserMessage(t *testing.T) { + tests := []struct { + name string + state string + want string + }{ + {name: "string state", state: `"charged twice"`, want: "charged twice"}, + {name: "record state", state: `{"ticket":"charged twice"}`, want: `{"ticket":"charged twice"}`}, + {name: "null state", state: `null`, want: "null"}, + {name: "no state", state: ``, want: ""}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + p, err := FromSystemOneRequest(&core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(tt.state)}) + require.NoError(t, err) + require.Len(t, p.Messages, 1) + + assert.Equal(t, SystemOneStateMessageID, p.Messages[0].ID) + assert.Equal(t, pluginapi.RoleUser, p.Messages[0].Role) + assert.Equal(t, tt.want, p.Messages[0].Text()) + assert.Equal(t, "kev-latest", p.Params.Model) + assert.False(t, p.Changes().Dirty) + }) + } +} + +func TestApplyToSystemOneRequest(t *testing.T) { + tests := []struct { + name string + state string + edit func(p *pluginapi.Prompt) error + wantState string + wantErr string + }{ + { + name: "untouched", + state: `"John"`, + edit: func(*pluginapi.Prompt) error { return nil }, + wantState: `"John"`, + }, + { + name: "string state stays a string", + state: `"John \"J\" Doe"`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `[PERSON] "J"`) }, + wantState: `"[PERSON] \"J\""`, + }, + { + name: "record state stays a record", + state: `{"name":"John"}`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `{"name":"[PERSON]"}`) }, + wantState: `{"name":"[PERSON]"}`, + }, + { + name: "null state keeps its JSON type", + state: `null`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `{"redacted":true}`) }, + wantState: `{"redacted":true}`, + }, + { + name: "record state must stay valid JSON", + state: `{"name":"John"}`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `name: [PERSON]`) }, + wantErr: "no longer valid JSON", + }, + { + name: "state cannot be removed", + state: `"John"`, + edit: func(p *pluginapi.Prompt) error { return p.Remove(SystemOneStateMessageID) }, + wantErr: "state was removed", + }, + { + name: "uncarried edits are not applied", + state: `"John"`, + edit: func(p *pluginapi.Prompt) error { + p.Insert(0, pluginapi.TextMessage(pluginapi.RoleSystem, "be safe")) + p.SetParam("temperature", 0.1) + return nil + }, + wantState: `"John"`, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + original := &core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(tt.state)} + p, err := FromSystemOneRequest(original) + require.NoError(t, err) + require.NoError(t, tt.edit(p)) + + got, err := ApplyToSystemOneRequest(original, p) + if tt.wantErr != "" { + require.Error(t, err) + assert.Contains(t, err.Error(), tt.wantErr) + return + } + require.NoError(t, err) + assert.JSONEq(t, tt.wantState, string(got.State)) + assert.Equal(t, "kev-latest", got.Model) + assert.JSONEq(t, tt.state, string(original.State), "the original request must not change") + }) + } +} + +func TestSystemOneUncarriedEdits(t *testing.T) { + p, err := FromSystemOneRequest(&core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(`"John"`)}) + require.NoError(t, err) + assert.Nil(t, SystemOneUncarriedEdits(p)) + + require.NoError(t, p.SetText(SystemOneStateMessageID, 0, "[PERSON]")) + assert.Nil(t, SystemOneUncarriedEdits(p), "a state edit is carried") + + p.Insert(0, pluginapi.TextMessage(pluginapi.RoleSystem, "be safe")) + p.Append(pluginapi.TextMessage(pluginapi.RoleSystem, "be brief")) + p.SetParam("temperature", 0.1) + assert.Equal(t, []string{"inserted message", `parameter "temperature"`}, SystemOneUncarriedEdits(p), + "kinds are listed once, without per-request message IDs") +} diff --git a/internal/pricingoverrides/resolver.go b/internal/pricingoverrides/resolver.go index 78b60836c..945f2b0d8 100644 --- a/internal/pricingoverrides/resolver.go +++ b/internal/pricingoverrides/resolver.go @@ -29,6 +29,29 @@ func (s *Service) ResolvePricing(model, providerName string) *core.ModelPricing return cloneBasePricing(basePricing) } +// HasModelPricing reports whether pricing is declared for exactly this model: +// catalog pricing, or an override scoped to the model rather than to its +// provider or to every model. +func (s *Service) HasModelPricing(model, providerName string) bool { + if s == nil { + return false + } + providerName = strings.TrimSpace(providerName) + rawModel := strings.TrimSpace(model) + model = modelIDFromSelector(rawModel, providerName) + if model == "" { + return false + } + if s.snapshot().hasModelScopedOverride(providerName, model) { + return true + } + if s.base == nil { + return false + } + return s.base.ResolvePricing(model, providerName) != nil || + (rawModel != model && s.base.ResolvePricing(rawModel, providerName) != nil) +} + func cloneBasePricing(base *core.ModelPricing) *core.ModelPricing { if base == nil { return nil diff --git a/internal/pricingoverrides/service_test.go b/internal/pricingoverrides/service_test.go index e977e1455..f39659872 100644 --- a/internal/pricingoverrides/service_test.go +++ b/internal/pricingoverrides/service_test.go @@ -7,6 +7,7 @@ import ( "time" "github.com/enterpilot/gomodel/internal/core" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -353,3 +354,35 @@ func TestNormalizedRefreshIntervalClampsBelowRefreshTimeout(t *testing.T) { }) } } + +func TestServiceHasModelPricing(t *testing.T) { + baseRate := 1.0 + service, err := NewService( + newTestStore( + Override{Selector: "/", Pricing: Pricing{InputPerMtok: new(float64(10))}}, + Override{Selector: "jev/", Pricing: Pricing{InputPerMtok: new(float64(20))}}, + Override{Selector: "jev/jev-1.13.0", Pricing: Pricing{InputPerMtok: new(float64(42))}}, + Override{Selector: "kev-4b", Pricing: Pricing{InputPerMtok: new(float64(0))}}, + ), + testCatalog{providerNames: []string{"jev", "openai"}}, + selectivePricingResolver{"openai/gpt-4o": {InputPerMtok: &baseRate}}, + ) + require.NoError(t, err) + require.NoError(t, service.Refresh(context.Background())) + + tests := []struct { + model, provider string + want bool + }{ + {model: "jev-1.13.0", provider: "jev", want: true}, + {model: "jev/jev-1.13.0", provider: "jev", want: true}, + {model: "kev-4b", provider: "kev", want: true}, + {model: "gpt-4o", provider: "openai", want: true}, + // Only the provider-wide and global overrides match these. + {model: "jev-latest", provider: "jev", want: false}, + {model: "gpt-9", provider: "openai", want: false}, + } + for _, tt := range tests { + assert.Equal(t, tt.want, service.HasModelPricing(tt.model, tt.provider), "%s/%s", tt.provider, tt.model) + } +} diff --git a/internal/pricingoverrides/snapshot.go b/internal/pricingoverrides/snapshot.go index 5a4338680..c8b743b30 100644 --- a/internal/pricingoverrides/snapshot.go +++ b/internal/pricingoverrides/snapshot.go @@ -108,6 +108,18 @@ func (snap snapshot) matchingOverride(providerName, model string) (compiledOverr return compiledOverride{}, false } +// hasModelScopedOverride reports whether an override names this model, +// either for one provider or model-wide. +func (snap snapshot) hasModelScopedOverride(providerName, model string) bool { + if key := modelselectors.ExactMatchKey(providerName, model); key != "" { + if _, ok := snap.exact[key]; ok { + return true + } + } + _, ok := snap.modelWide[model] + return ok +} + func snapshotOverrides(snap snapshot) []Override { result := make([]Override, 0, len(snap.order)) for _, selector := range snap.order { diff --git a/internal/providers/anthropic/anthropic_test.go b/internal/providers/anthropic/anthropic_test.go index 0f6a0f405..0202c0dc7 100644 --- a/internal/providers/anthropic/anthropic_test.go +++ b/internal/providers/anthropic/anthropic_test.go @@ -950,6 +950,21 @@ func TestConvertToAnthropicRequest(t *testing.T) { assert.Len(t, req.Messages, 1) }, }, + { + name: "developer message becomes system", + input: &core.ChatRequest{ + Model: "claude-haiku-4-5-20251001", + Messages: []core.Message{ + {Role: "developer", Content: "You are a helpful assistant"}, + {Role: "user", Content: "Hello"}, + }, + }, + checkFn: func(t *testing.T, req *anthropicRequest) { + assert.Equal(t, "You are a helpful assistant", req.System) + require.Len(t, req.Messages, 1) + assert.Equal(t, "user", req.Messages[0].Role) + }, + }, { name: "request with parameters", input: &core.ChatRequest{ @@ -1679,6 +1694,56 @@ func TestConvertOpenAIToolsToAnthropic(t *testing.T) { require.True(t, ok, "InputSchema.properties = %#v, want object map", tools[0].InputSchema["properties"]) }, }, + { + name: "strict tool keeps strict and sanitizes schema", + tools: []map[string]any{ + { + "type": "function", + "function": map[string]any{ + "name": "lookup_weather", + "strict": true, + "parameters": map[string]any{ + "type": "object", + "properties": map[string]any{ + "city": map[string]any{"type": "string", "minLength": 1}, + }, + }, + }, + }, + }, + wantLen: 1, + checkFn: func(t *testing.T, tools []anthropicTool) { + assert.True(t, tools[0].Strict) + assert.Equal(t, false, tools[0].InputSchema["additionalProperties"]) + properties, ok := tools[0].InputSchema["properties"].(map[string]any) + require.True(t, ok) + city, ok := properties["city"].(map[string]any) + require.True(t, ok) + assert.NotContains(t, city, "minLength") + }, + }, + { + name: "non strict tool keeps schema as sent", + tools: []map[string]any{ + { + "type": "function", + "function": map[string]any{ + "name": "lookup_weather", + "parameters": map[string]any{ + "type": "object", + "properties": map[string]any{ + "city": map[string]any{"type": "string", "minLength": 1}, + }, + }, + }, + }, + }, + wantLen: 1, + checkFn: func(t *testing.T, tools []anthropicTool) { + assert.False(t, tools[0].Strict) + assert.NotContains(t, tools[0].InputSchema, "additionalProperties") + }, + }, { name: "unsupported tool type returns error", tools: []map[string]any{ diff --git a/internal/providers/anthropic/request_translation.go b/internal/providers/anthropic/request_translation.go index 1c412e71c..34011ac3e 100644 --- a/internal/providers/anthropic/request_translation.go +++ b/internal/providers/anthropic/request_translation.go @@ -132,10 +132,18 @@ func convertOpenAIToolsToAnthropic(tools []map[string]any) ([]anthropicTool, err if err != nil { return nil, err } + // strict tools are held to the same schema subset as structured + // outputs, so the schema gets the same sanitizing. + strict, _ := function["strict"].(bool) + schema := inputSchema.(map[string]any) + if strict { + schema = sanitizeAnthropicSchema(schema) + } out = append(out, anthropicTool{ Name: name, Description: description, - InputSchema: inputSchema.(map[string]any), + InputSchema: schema, + Strict: strict, CacheControl: cacheControl, }) } @@ -445,7 +453,8 @@ func convertToAnthropicRequest(req *core.ChatRequest) (*anthropicRequest, error) conversationStarted := false for _, msg := range req.Messages { - if msg.Role == "system" { + // "developer" is OpenAI's newer name for "system"; Anthropic rejects it. + if msg.Role == "system" || msg.Role == "developer" { systemContent, err := buildAnthropicSystemContent(msg.Content) if err != nil { return nil, err diff --git a/internal/providers/anthropic/types.go b/internal/providers/anthropic/types.go index 837ee293b..28261ae43 100644 --- a/internal/providers/anthropic/types.go +++ b/internal/providers/anthropic/types.go @@ -39,6 +39,7 @@ type anthropicTool struct { Name string `json:"name"` Description string `json:"description,omitempty"` InputSchema map[string]any `json:"input_schema"` + Strict bool `json:"strict,omitempty"` CacheControl json.RawMessage `json:"cache_control,omitempty"` } diff --git a/internal/providers/configured_models.go b/internal/providers/configured_models.go index b167d9835..f2e33b7a8 100644 --- a/internal/providers/configured_models.go +++ b/internal/providers/configured_models.go @@ -1,6 +1,7 @@ package providers import ( + "errors" "sort" "strings" "time" @@ -16,8 +17,13 @@ const ( configuredProviderModelsAllowlist configuredProviderModelsApplyReason = "allowlist" configuredProviderModelsMerge configuredProviderModelsApplyReason = "merge" configuredProviderModelsUpstreamError configuredProviderModelsApplyReason = "upstream_error" - configuredProviderModelsUpstreamNil configuredProviderModelsApplyReason = "upstream_nil" - configuredProviderModelsUpstreamEmpty configuredProviderModelsApplyReason = "upstream_empty" + // configuredProviderModelsUpstreamUnlisted means the provider has no model + // listing endpoint (404/405 on /models). Servers that only expose a single + // API, such as speech-to-text servers, are fully described by the + // configured list, so it counts as authoritative rather than a fallback. + configuredProviderModelsUpstreamUnlisted configuredProviderModelsApplyReason = "upstream_unlisted" + configuredProviderModelsUpstreamNil configuredProviderModelsApplyReason = "upstream_nil" + configuredProviderModelsUpstreamEmpty configuredProviderModelsApplyReason = "upstream_empty" ) func normalizeConfiguredProviderModels(models []string) []string { @@ -62,6 +68,9 @@ func applyConfiguredProviderModels( return configuredProviderModelsResponse(providerName, providerType, configuredModels, upstream, fallbackCreated), configuredProviderModelsAllowlist } + if modelListingUnsupported(upstreamErr) { + return configuredProviderModelsResponse(providerName, providerType, configuredModels, upstream, fallbackCreated), configuredProviderModelsUpstreamUnlisted + } if upstreamErr != nil { return configuredProviderModelsResponse(providerName, providerType, configuredModels, upstream, fallbackCreated), configuredProviderModelsUpstreamError } @@ -77,6 +86,14 @@ func applyConfiguredProviderModels( return upstream, configuredProviderModelsNotApplied } +// modelListingUnsupported reports whether err means the upstream has no model +// listing endpoint at all, as opposed to a listing that failed. Providers mark +// such errors at the /models call, so a 404 from another API (e.g. the Bedrock +// control plane) still counts as a failure. +func modelListingUnsupported(err error) bool { + return errors.Is(err, core.ErrModelListingUnsupported) +} + // configuredModelOwner picks the owned_by value for synthesized entries. func configuredModelOwner(providerName, providerType string) string { owner := strings.TrimSpace(providerType) diff --git a/internal/providers/configured_models_test.go b/internal/providers/configured_models_test.go index e379e24d5..6cb2e24f6 100644 --- a/internal/providers/configured_models_test.go +++ b/internal/providers/configured_models_test.go @@ -2,6 +2,8 @@ package providers import ( "errors" + "fmt" + "net/http" "testing" "github.com/enterpilot/gomodel/config" @@ -92,3 +94,32 @@ func TestApplyConfiguredProviderModels_MergeFallsBackWhenUpstreamFails(t *testin }) } } + +func TestApplyConfiguredProviderModels_MissingModelsEndpointIsAuthoritative(t *testing.T) { + notFound := core.MarkModelListingUnsupported(core.ParseProviderError("openai", http.StatusNotFound, []byte("404 Not Found"), nil)) + tests := []struct { + name string + mode config.ConfiguredProviderModelsMode + err error + wantReason configuredProviderModelsApplyReason + }{ + {name: "fallback 404", mode: config.ConfiguredProviderModelsModeFallback, err: notFound, wantReason: configuredProviderModelsUpstreamUnlisted}, + {name: "merge 404", mode: config.ConfiguredProviderModelsModeMerge, err: notFound, wantReason: configuredProviderModelsUpstreamUnlisted}, + {name: "wrapped 404", mode: config.ConfiguredProviderModelsModeFallback, err: fmt.Errorf("list models: %w", notFound), wantReason: configuredProviderModelsUpstreamUnlisted}, + {name: "405", mode: config.ConfiguredProviderModelsModeFallback, err: core.MarkModelListingUnsupported(core.ParseProviderError("openai", http.StatusMethodNotAllowed, nil, nil)), wantReason: configuredProviderModelsUpstreamUnlisted}, + // A 404 from an API other than /models (e.g. the Bedrock control plane) is not marked. + {name: "unmarked 404", mode: config.ConfiguredProviderModelsModeFallback, err: core.ParseProviderError("bedrock", http.StatusNotFound, nil, nil), wantReason: configuredProviderModelsUpstreamError}, + {name: "500", mode: config.ConfiguredProviderModelsModeFallback, err: core.ParseProviderError("openai", http.StatusInternalServerError, nil, nil), wantReason: configuredProviderModelsUpstreamError}, + {name: "401", mode: config.ConfiguredProviderModelsModeFallback, err: core.ParseProviderError("openai", http.StatusUnauthorized, nil, nil), wantReason: configuredProviderModelsUpstreamError}, + {name: "plain error", mode: config.ConfiguredProviderModelsModeFallback, err: errors.New("connection refused"), wantReason: configuredProviderModelsUpstreamError}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + resp, reason := applyConfiguredProviderModels("stt", "openai", tt.mode, []string{"whisper-1"}, nil, tt.err, 123) + require.Equal(t, tt.wantReason, reason) + require.NotNil(t, resp) + require.Len(t, resp.Data, 1) + require.Equal(t, "whisper-1", resp.Data[0].ID) + }) + } +} diff --git a/internal/providers/gemini/native.go b/internal/providers/gemini/native.go index 9c0079be6..019941317 100644 --- a/internal/providers/gemini/native.go +++ b/internal/providers/gemini/native.go @@ -264,7 +264,10 @@ func convertChatRequestToGemini(req *core.ChatRequest) (*geminiGenerateContentRe return nil, err } out.Tools = tools - out.ToolConfig = geminiToolConfigFromOpenAI(req.ToolChoice) + out.ToolConfig, err = geminiToolConfigFromOpenAI(req.ToolChoice, hasStrictTool(req.Tools)) + if err != nil { + return nil, err + } out.GenerationConfig = geminiGenerationConfig(req) out.SafetySettings = geminiSafetySettings(req) out.CachedContent = geminiCachedContent(req) @@ -642,7 +645,11 @@ func validateGeminiParametersJSONSchema(encoded json.RawMessage) (json.RawMessag return stripped, nil } -func geminiToolConfigFromOpenAI(choice any) *geminiToolConfig { +// geminiToolConfigFromOpenAI maps tool_choice onto functionCallingConfig. With +// strict set (any tool declared strict: true), the free-choice mode becomes +// VALIDATED, Gemini's AUTO with schema-adherent function calls; ANY already +// guarantees schema adherence. +func geminiToolConfigFromOpenAI(choice any, strict bool) (*geminiToolConfig, error) { mode := "" var allowed []string @@ -658,23 +665,73 @@ func geminiToolConfigFromOpenAI(choice any) *geminiToolConfig { } case map[string]any: choiceType, _ := value["type"].(string) - if strings.TrimSpace(choiceType) == "function" { + switch strings.TrimSpace(choiceType) { + case "function": mode = "ANY" if fn, ok := value["function"].(map[string]any); ok { if name, _ := fn["name"].(string); name != "" { allowed = []string{name} } } + case "allowed_tools": + var err error + mode, allowed, err = geminiAllowedToolsConfig(value) + if err != nil { + return nil, err + } } } + if strict && (mode == "" || mode == "AUTO") { + mode = "VALIDATED" + } if mode == "" { - return nil + return nil, nil } return &geminiToolConfig{FunctionCallingConfig: geminiFunctionCallingConfig{ Mode: mode, AllowedFunctionNames: allowed, - }} + }}, nil +} + +// geminiAllowedToolsConfig maps an allowed_tools choice onto a mode and the +// allowedFunctionNames subset. Gemini accepts allowedFunctionNames only with +// ANY or VALIDATED, so "auto" becomes VALIDATED: the model may still answer in +// text, but any call is limited to the subset. An empty subset is rejected, as +// OpenAI does, rather than widened to every declared tool. +func geminiAllowedToolsConfig(choice map[string]any) (string, []string, error) { + spec, _ := choice["allowed_tools"].(map[string]any) + tools, _ := spec["tools"].([]any) + var names []string + for _, raw := range tools { + tool, _ := raw.(map[string]any) + fn, _ := tool["function"].(map[string]any) + if name, _ := fn["name"].(string); strings.TrimSpace(name) != "" { + names = append(names, name) + } + } + + if len(names) == 0 { + return "", nil, core.NewInvalidRequestError("tool_choice.allowed_tools.tools must list at least one function", nil) + } + + if mode, _ := spec["mode"].(string); strings.TrimSpace(mode) == "required" { + return "ANY", names, nil + } + return "VALIDATED", names, nil +} + +// hasStrictTool reports whether any function tool asks for strict schema +// adherence. +func hasStrictTool(tools []map[string]any) bool { + for _, tool := range tools { + if fn, ok := tool["function"].(map[string]any); ok { + if strict, _ := fn["strict"].(bool); strict { + return true + } + } + } + return false } func geminiGenerationConfig(req *core.ChatRequest) map[string]any { diff --git a/internal/providers/gemini/native_tool_config_test.go b/internal/providers/gemini/native_tool_config_test.go new file mode 100644 index 000000000..ecba04326 --- /dev/null +++ b/internal/providers/gemini/native_tool_config_test.go @@ -0,0 +1,101 @@ +package gemini + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/core" +) + +func TestGeminiToolConfigFromOpenAI(t *testing.T) { + allowedTools := func(mode string, names ...string) map[string]any { + tools := make([]any, 0, len(names)) + for _, name := range names { + tools = append(tools, map[string]any{"type": "function", "function": map[string]any{"name": name}}) + } + return map[string]any{ + "type": "allowed_tools", + "allowed_tools": map[string]any{"mode": mode, "tools": tools}, + } + } + + tests := []struct { + name string + choice any + strict bool + wantMode string + wantAllowed []string + }{ + {name: "unset", choice: nil}, + {name: "auto", choice: "auto", wantMode: "AUTO"}, + {name: "required", choice: "required", wantMode: "ANY"}, + {name: "none", choice: "none", wantMode: "NONE"}, + { + name: "named function", + choice: map[string]any{"type": "function", "function": map[string]any{"name": "tool_a"}}, + wantMode: "ANY", + wantAllowed: []string{"tool_a"}, + }, + { + name: "allowed_tools required", + choice: allowedTools("required", "tool_a", "tool_b"), + wantMode: "ANY", + wantAllowed: []string{"tool_a", "tool_b"}, + }, + { + // Gemini rejects allowedFunctionNames with AUTO. + name: "allowed_tools auto", + choice: allowedTools("auto", "tool_b"), + wantMode: "VALIDATED", + wantAllowed: []string{"tool_b"}, + }, + {name: "strict unset", choice: nil, strict: true, wantMode: "VALIDATED"}, + {name: "strict auto", choice: "auto", strict: true, wantMode: "VALIDATED"}, + {name: "strict required", choice: "required", strict: true, wantMode: "ANY"}, + {name: "strict none", choice: "none", strict: true, wantMode: "NONE"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, err := geminiToolConfigFromOpenAI(tt.choice, tt.strict) + require.NoError(t, err) + if tt.wantMode == "" { + assert.Nil(t, got) + return + } + require.NotNil(t, got) + assert.Equal(t, tt.wantMode, got.FunctionCallingConfig.Mode) + assert.Equal(t, tt.wantAllowed, got.FunctionCallingConfig.AllowedFunctionNames) + }) + } +} + +func TestGeminiToolConfigFromOpenAIRejectsEmptyAllowedTools(t *testing.T) { + for _, mode := range []string{"auto", "required"} { + t.Run(mode, func(t *testing.T) { + _, err := geminiToolConfigFromOpenAI(map[string]any{ + "type": "allowed_tools", + "allowed_tools": map[string]any{"mode": mode, "tools": []any{}}, + }, false) + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + }) + } +} + +func TestConvertChatRequestToGeminiStrictToolUsesValidatedMode(t *testing.T) { + out, err := convertChatRequestToGemini(&core.ChatRequest{ + Model: "gemini-2.5-flash", + Messages: []core.Message{{Role: "user", Content: "hi"}}, + Tools: []map[string]any{ + {"type": "function", "function": map[string]any{"name": "tool_a"}}, + {"type": "function", "function": map[string]any{"name": "tool_b", "strict": true}}, + }, + }) + require.NoError(t, err) + require.NotNil(t, out.ToolConfig) + assert.Equal(t, "VALIDATED", out.ToolConfig.FunctionCallingConfig.Mode) +} diff --git a/internal/providers/jev/jev.go b/internal/providers/jev/jev.go index 32b3ea576..9f999d72c 100644 --- a/internal/providers/jev/jev.go +++ b/internal/providers/jev/jev.go @@ -3,8 +3,8 @@ // API. System One is a decision API rather than a text-generation one: a // request carries a state and a map of typed questions (noul, choice, score) // and the answer is a calibrated probability per question. It has no -// OpenAI-compatible surface, so the gateway reaches it through native -// passthrough at /p/jev/systemone. +// OpenAI-compatible surface, so the gateway forwards it natively, at +// POST /v1/systemone or through passthrough at /p/jev/systemone. package jev import ( @@ -42,8 +42,9 @@ type Provider struct { } var ( - _ core.Provider = (*Provider)(nil) - _ core.PassthroughProvider = (*Provider)(nil) + _ core.Provider = (*Provider)(nil) + _ core.PassthroughProvider = (*Provider)(nil) + _ core.UnlistedModelAcceptor = (*Provider)(nil) ) // New creates a Jev provider. The client is rooted at the API origin, which @@ -64,6 +65,10 @@ func New(cfg providers.ProviderConfig, opts providers.ProviderOptions) core.Prov return p } +// AcceptsUnlistedModels reports that the upstream accepts versioned IDs +// (jev-1.13.0) it does not list: TypeSafe lists only its aliases. +func (p *Provider) AcceptsUnlistedModels() bool { return true } + // SetBaseURL allows configuring a custom base URL for the provider. func (p *Provider) SetBaseURL(url string) { p.client.SetBaseURL(baseURL(url)) @@ -110,12 +115,12 @@ func (p *Provider) Embeddings(_ context.Context, _ *core.EmbeddingRequest) (*cor } func unsupported(surface string) error { - return core.NewInvalidRequestError("jev does not support "+surface+"; send System One requests to /p/jev/systemone", nil) + return core.NewInvalidRequestError("jev does not support "+surface+"; it answers System One decision requests, which GoModel does not translate: send them to POST /v1/systemone", nil) } -// Passthrough forwards a System One request as the client wrote it. It is the -// only way to reach the evaluation endpoint, since the request and answer -// shapes have no OpenAI equivalent. +// Passthrough forwards a System One request as the client wrote it. Both +// /v1/systemone and /p/jev/... reach the evaluation endpoint through it, +// since the request and answer shapes have no OpenAI equivalent. func (p *Provider) Passthrough(ctx context.Context, req *core.PassthroughRequest) (*core.PassthroughResponse, error) { if req == nil { return nil, core.NewInvalidRequestError("passthrough request is required", nil) diff --git a/internal/providers/jev/jev_test.go b/internal/providers/jev/jev_test.go index 43c603028..9475fcaf8 100644 --- a/internal/providers/jev/jev_test.go +++ b/internal/providers/jev/jev_test.go @@ -45,7 +45,7 @@ func TestUnsupportedCapabilities_ReturnInvalidRequestErrors(t *testing.T) { _, err := provider.ChatCompletion(context.Background(), &core.ChatRequest{Model: "jev-latest"}) providertest.AssertUnsupported(t, err) - assert.Contains(t, err.Error(), "/p/jev/systemone") + assert.Contains(t, err.Error(), "/v1/systemone") _, err = provider.StreamChatCompletion(context.Background(), &core.ChatRequest{Model: "jev-latest"}) providertest.AssertUnsupported(t, err) _, err = provider.Responses(context.Background(), &core.ResponsesRequest{Model: "jev-latest"}) diff --git a/internal/providers/kimicode/kimicode.go b/internal/providers/kimicode/kimicode.go index 2899e6acb..e9b0f610c 100644 --- a/internal/providers/kimicode/kimicode.go +++ b/internal/providers/kimicode/kimicode.go @@ -1,11 +1,17 @@ // Package kimicode provides Kimi Code API integration for the LLM gateway. // -// The "kimicode" provider routes to Kimi Code's OpenAI-compatible chat -// completions endpoint, so all transport goes through the shared chat-centric -// adapter and model IDs are forwarded unchanged. +// The "kimicode" provider routes to Kimi Code's OpenAI-compatible API: chat +// completions, model listing, embeddings, and passthrough go through the +// shared chat-centric adapter, while the Responses API is served natively by +// the upstream /responses endpoint. Kimi Code retains no responses, so +// store=true is pinned to false. package kimicode import ( + "context" + "io" + "strings" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/providers" "github.com/enterpilot/gomodel/internal/providers/openai" @@ -23,9 +29,11 @@ var Registration = providers.Registration{ } // Provider implements the core.Provider interface for Kimi Code. Kimi Code is -// OpenAI-compatible, so all transport goes through the shared chat-centric +// OpenAI-compatible, so most transport goes through the shared chat-centric // adapter: chat completions, model listing, embeddings, and passthrough are -// exposed via the embedded *openai.ChatCompatible. +// exposed via the embedded *openai.ChatCompatible. The Responses API is +// forwarded natively to the upstream /responses endpoint through the same +// adapter instance. type Provider struct { *openai.ChatCompatible } @@ -39,3 +47,72 @@ func New(cfg providers.ProviderConfig, opts providers.ProviderOptions) core.Prov BaseURL: providers.ResolveBaseURL(cfg.BaseURL, defaultBaseURL), })} } + +// Responses serves the Responses API natively through the upstream /responses +// endpoint. Kimi Code retains no responses, so a non-empty +// previous_response_id is rejected before any upstream call (see +// rejectPreviousResponseID); store=true is pinned to false by +// adaptResponsesRequest. +func (p *Provider) Responses(ctx context.Context, req *core.ResponsesRequest) (*core.ResponsesResponse, error) { + if err := rejectPreviousResponseID(req); err != nil { + return nil, err + } + return p.Compatible().Responses(ctx, adaptResponsesRequest(req)) +} + +// StreamResponses forwards the request to the upstream /responses endpoint +// with stream enabled, returning its Responses SSE stream. Like Responses, it +// rejects a non-empty previous_response_id before any upstream call. +func (p *Provider) StreamResponses(ctx context.Context, req *core.ResponsesRequest) (io.ReadCloser, error) { + if err := rejectPreviousResponseID(req); err != nil { + return nil, err + } + return p.Compatible().StreamResponses(ctx, adaptResponsesRequest(req)) +} + +// rejectPreviousResponseID fails requests chaining from earlier state: +// Kimi Code cannot resolve a previous response ID or a gateway-local +// conversation upstream, and answering statelessly would silently drop the +// conversation context the caller expects. The rejection only fires when the +// gateway has no store to expand the chain with; requests whose state the +// gateway already replayed into input (both fields cleared) pass through. +// The ID check mirrors the gateway and the chat-translation validator, both +// of which treat a whitespace-only ID as empty. +func rejectPreviousResponseID(req *core.ResponsesRequest) error { + if req == nil { + return nil + } + if req.Conversation != nil { + return core.NewInvalidRequestError( + "kimicode does not retain responses: conversation is not supported", nil) + } + if strings.TrimSpace(req.PreviousResponseID) == "" { + return nil + } + return core.NewInvalidRequestError( + "kimicode does not retain responses: previous_response_id is not supported", nil) +} + +// adaptResponsesRequest pins store to false: the service retains no +// responses, so store=true fails upstream with a 400 (Postel's law — adapt +// instead of failing). A whitespace-only previous_response_id is treated as +// empty by rejectPreviousResponseID and cleared here, because omitempty does +// not omit a non-empty whitespace string and the upstream cannot resolve it. +func adaptResponsesRequest(req *core.ResponsesRequest) *core.ResponsesRequest { + if req == nil { + return nil + } + whitespaceID := req.PreviousResponseID != "" && strings.TrimSpace(req.PreviousResponseID) == "" + if (req.Store == nil || !*req.Store) && !whitespaceID { + return req + } + cp := *req + if req.Store != nil && *req.Store { + disabled := false + cp.Store = &disabled + } + if whitespaceID { + cp.PreviousResponseID = "" + } + return &cp +} diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index 54eecf1ce..23eea1d8a 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -1,9 +1,14 @@ package kimicode import ( + "context" "net/http" + "net/http/httptest" "testing" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/llmclient" "github.com/enterpilot/gomodel/internal/providers" @@ -12,7 +17,8 @@ import ( // Kimi Code is a thin wrapper over the shared chat-centric adapter and // forwards embeddings upstream unchanged, so the shared contract covers its -// surface. +// surface. The Responses API is forwarded natively to the upstream /responses +// endpoint. func TestChatCompatibleContract(t *testing.T) { providertest.AssertChatCompatible(t, providertest.ChatCompatible{ Registration: Registration, @@ -23,6 +29,270 @@ func TestChatCompatibleContract(t *testing.T) { opts.HTTPClient = client return New(providers.ProviderConfig{APIKey: apiKey, BaseURL: baseURL}, opts) }, - Embeddings: true, + Embeddings: true, + NativeResponses: true, + }) +} + +func boolPtr(b bool) *bool { return &b } + +// newTestProvider builds a provider wired to the test server through the +// injected HTTP client, matching how the shared contract constructs it. +func newTestProvider(server *httptest.Server) core.Provider { + opts := providertest.Options(llmclient.Hooks{}) + opts.HTTPClient = server.Client() + return New(providers.ProviderConfig{APIKey: "kimi-key", BaseURL: server.URL}, opts) +} + +func TestRejectPreviousResponseID(t *testing.T) { + t.Run("nil request passes through", func(t *testing.T) { + assert.NoError(t, rejectPreviousResponseID(nil)) + }) + + t.Run("clean request passes through", func(t *testing.T) { + assert.NoError(t, rejectPreviousResponseID(&core.ResponsesRequest{Model: "kimi-for-coding", Input: "hi"})) + }) +} + +func TestAdaptResponsesRequest(t *testing.T) { + t.Run("nil passes through", func(t *testing.T) { + assert.Nil(t, adaptResponsesRequest(nil)) + }) + + t.Run("clean request is returned unchanged", func(t *testing.T) { + req := &core.ResponsesRequest{Model: "kimi-for-coding", Input: "hi"} + assert.Same(t, req, adaptResponsesRequest(req)) + }) + + t.Run("store true is pinned to false", func(t *testing.T) { + req := &core.ResponsesRequest{Model: "kimi-for-coding", Input: "hi", Store: boolPtr(true)} + got := adaptResponsesRequest(req) + require.NotSame(t, req, got, "adapted request should be a copy") + require.NotNil(t, got.Store) + assert.False(t, *got.Store) + // The caller's request must not be mutated. + assert.True(t, *req.Store, "original request Store was mutated") + }) + + t.Run("previous_response_id is preserved", func(t *testing.T) { + // adaptResponsesRequest does not touch PreviousResponseID; the + // Responses/StreamResponses methods reject it instead (tested below). + req := &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "hi", + PreviousResponseID: "resp_old", + Store: boolPtr(false), + } + got := adaptResponsesRequest(req) + assert.Equal(t, "resp_old", got.PreviousResponseID) + require.NotNil(t, got.Store) + assert.False(t, *got.Store, "explicit store=false should stay false") + }) + + t.Run("whitespace-only previous_response_id is cleared", func(t *testing.T) { + req := &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "hi", + PreviousResponseID: " ", + } + got := adaptResponsesRequest(req) + require.NotSame(t, req, got, "adapted request should be a copy") + assert.Empty(t, got.PreviousResponseID) + assert.Equal(t, " ", req.PreviousResponseID, "original request was mutated") + }) +} + +// responsesGoldenBody mirrors a real non-streaming /responses reply from the +// Kimi Code upstream (recorded 2026-09-08, trimmed to the members GoModel +// consumes). The upstream reply keeps extra members (prompt_cache_key, +// safety_identifier, service_tier); unknown members are ignored on decode. +const responsesGoldenBody = `{ + "id": "resp_golden", + "object": "response", + "created_at": 1788866012, + "completed_at": 1788866014, + "status": "completed", + "output": [ + { + "type": "reasoning", + "id": "rs_golden", + "status": "completed", + "summary": [{"type": "summary_text", "text": "Simple request."}] + }, + { + "type": "message", + "id": "msg_golden", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "OK", "annotations": []}] + } + ], + "usage": { + "input_tokens": 88, + "input_tokens_details": {"cache_write_tokens": 12, "cached_tokens": 88}, + "output_tokens": 53, + "output_tokens_details": {"reasoning_tokens": 37}, + "total_tokens": 141 + }, + "store": false, + "model": "kimi-for-coding" +}` + +// TestResponses_ForwardsGatewayReplayedHistory covers what the gateway +// dispatches after expanding a previous_response_id chain against its +// response store: the stored history is replayed into input as items +// (reasoning and message items among them, IDs stripped) and +// previous_response_id is cleared. Kimi Code must forward that replayed +// input to /responses as stored instead of rejecting it. +func TestResponses_ForwardsGatewayReplayedHistory(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) + + provider := newTestProvider(server) + + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: []any{ + map[string]any{ + "type": "message", + "role": "user", + "content": []any{map[string]any{"type": "input_text", "text": "remember: zebra"}}, + }, + map[string]any{ + "type": "reasoning", + "status": "completed", + "summary": []any{ + map[string]any{"type": "summary_text", "text": "thinking about zebras"}, + }, + }, + map[string]any{ + "type": "message", + "role": "assistant", + "status": "completed", + "content": []any{map[string]any{"type": "output_text", "text": "the word is zebra"}}, + }, + map[string]any{ + "type": "message", + "role": "user", + "content": []any{map[string]any{"type": "input_text", "text": "what is the word?"}}, + }, + }, + Store: boolPtr(true), + }) + require.NoError(t, err) + require.NotNil(t, resp) + + req := capture.Last(t) + assert.Equal(t, "/responses", req.Path) + wire := req.JSON(t) + assert.Equal(t, false, wire["store"], "store is pinned to false on the wire") + assert.NotContains(t, wire, "previous_response_id") + + items, ok := wire["input"].([]any) + require.True(t, ok, "wire input = %#v, want replayed items", wire["input"]) + require.Len(t, items, 4) + reasoning, ok := items[1].(map[string]any) + require.True(t, ok) + assert.Equal(t, "reasoning", reasoning["type"], "replayed reasoning item must be forwarded unchanged") + assert.Equal(t, "user", items[3].(map[string]any)["role"], "the client's own turn is replayed last") +} + +// Kimi Code retains no responses, so a request chaining from an earlier +// response that the gateway could not expand (no store configured) must be +// rejected before any upstream call instead of being answered statelessly. +func TestResponses_RejectsPreviousResponseID(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) + + provider := newTestProvider(server) + + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + PreviousResponseID: "resp_old", + }) + require.Error(t, err) + assert.Nil(t, resp) + + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr, "error type = %T, want *core.GatewayError", err) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + assert.Contains(t, gatewayErr.Error(), "previous_response_id") + assert.Equal(t, 0, capture.Count(), "rejected request must not reach the upstream") + + t.Run("whitespace-only ID passes through", func(t *testing.T) { + // The gateway and the chat-translation validator treat a + // whitespace-only ID as empty; the provider must behave the same, + // and the unresolvable value must not reach the wire (omitempty + // does not omit a non-empty whitespace string). + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + PreviousResponseID: " ", + }) + require.NoError(t, err) + require.NotNil(t, resp) + assert.Equal(t, 1, capture.Count(), "whitespace-only ID is treated as empty") + wire := capture.Last(t).JSON(t) + _, present := wire["previous_response_id"] + assert.False(t, present, "whitespace-only ID must be omitted from the wire") + }) + + t.Run("conversation reference is rejected", func(t *testing.T) { + before := capture.Count() + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + Conversation: &core.ResponsesConversationRef{ID: "conv_old"}, + }) + require.Error(t, err) + assert.Nil(t, resp) + + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + assert.Contains(t, gatewayErr.Error(), "conversation") + assert.Equal(t, before, capture.Count(), "conversation request must not reach the upstream") + }) +} + +func TestStreamResponses_RejectsPreviousResponseID(t *testing.T) { + server, capture := providertest.SSEServer(t, "") + + provider := newTestProvider(server) + + stream, err := provider.StreamResponses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + PreviousResponseID: "resp_old", + }) + require.Error(t, err) + assert.Nil(t, stream) + + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr, "error type = %T, want *core.GatewayError", err) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + assert.Contains(t, gatewayErr.Error(), "previous_response_id") + assert.Equal(t, 0, capture.Count(), "rejected request must not reach the upstream") +} + +// SetBaseURL must retarget the single adapter serving both the chat-centric +// surface and the native Responses endpoint. +func TestSetBaseURL(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) + + p := New(providers.ProviderConfig{APIKey: "kimi-key"}, providertest.Options(llmclient.Hooks{})) + kp, ok := p.(*Provider) + require.True(t, ok) + require.Equal(t, defaultBaseURL, kp.GetBaseURL()) + + kp.SetBaseURL(server.URL) + + assert.Equal(t, server.URL, kp.GetBaseURL()) + + _, err := p.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", }) + require.NoError(t, err) + require.Equal(t, 1, capture.Count()) + assert.Equal(t, "/responses", capture.Last(t).Path) } diff --git a/internal/providers/openai/chat_compatible.go b/internal/providers/openai/chat_compatible.go index 437af6437..aaec07f09 100644 --- a/internal/providers/openai/chat_compatible.go +++ b/internal/providers/openai/chat_compatible.go @@ -47,6 +47,14 @@ func bearerHeaders(req *http.Request, apiKey string) { providers.SetAuthHeaders(req, apiKey, providers.AuthHeaderConfig{AuthScheme: "Bearer "}) } +// Compatible exposes the underlying OpenAI-compatible adapter, so a provider +// that embeds ChatCompatible can serve a native Responses endpoint through +// the same instance instead of building a second adapter with the same +// configuration. +func (c *ChatCompatible) Compatible() *CompatibleProvider { + return c.compatible +} + // SetBaseURL allows configuring a custom base URL for the provider. func (c *ChatCompatible) SetBaseURL(url string) { c.compatible.SetBaseURL(url) diff --git a/internal/providers/openai/chat_compatible_test.go b/internal/providers/openai/chat_compatible_test.go new file mode 100644 index 000000000..975094664 --- /dev/null +++ b/internal/providers/openai/chat_compatible_test.go @@ -0,0 +1,21 @@ +package openai + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/providers" +) + +// Compatible must expose the adapter the chat-centric surface was built +// with, so providers serving native Responses use the same instance. +func TestChatCompatible_Compatible(t *testing.T) { + chat := NewChatCompatible("key", providers.ProviderOptions{}, CompatibleProviderConfig{ + ProviderName: "test", + BaseURL: "https://example.com/v1", + }) + require.NotNil(t, chat) + assert.Same(t, chat.compatible, chat.Compatible()) +} diff --git a/internal/providers/openai/compatible_provider.go b/internal/providers/openai/compatible_provider.go index 4cb7b48b7..98247d31f 100644 --- a/internal/providers/openai/compatible_provider.go +++ b/internal/providers/openai/compatible_provider.go @@ -220,7 +220,7 @@ func (p *CompatibleProvider) ListModels(ctx context.Context) (*core.ModelsRespon Endpoint: "/models", }, &resp) if err != nil { - return nil, err + return nil, core.MarkModelListingUnsupported(err) } normalizeModelsResponse(&resp) return &resp, nil @@ -242,7 +242,7 @@ func (p *CompatibleProvider) ListModelsWithMaxModelLen(ctx context.Context) (*co Method: http.MethodGet, Endpoint: "/models", }, &upstream); err != nil { - return nil, err + return nil, core.MarkModelListingUnsupported(err) } resp := &core.ModelsResponse{Object: upstream.Object, Data: make([]core.Model, 0, len(upstream.Data))} for _, entry := range upstream.Data { diff --git a/internal/providers/openai/compatible_provider_test.go b/internal/providers/openai/compatible_provider_test.go index 000537b12..ac53ef160 100644 --- a/internal/providers/openai/compatible_provider_test.go +++ b/internal/providers/openai/compatible_provider_test.go @@ -2,6 +2,7 @@ package openai import ( "context" + "errors" "io" "net/http" "strings" @@ -69,6 +70,38 @@ func TestCompatibleProvider_ListModels_ReturnsUpstreamError(t *testing.T) { assert.Contains(t, []core.ErrorType{core.ErrorTypeProvider, core.ErrorTypeNotFound}, gatewayErr.Type) } +func TestCompatibleProvider_ListModels_MarksMissingModelsEndpoint(t *testing.T) { + tests := []struct { + name string + status int + unsupported bool + }{ + {name: "404", status: http.StatusNotFound, unsupported: true}, + {name: "405", status: http.StatusMethodNotAllowed, unsupported: true}, + {name: "500", status: http.StatusInternalServerError}, + {name: "401", status: http.StatusUnauthorized}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + server, _ := providertest.Server(t, func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(tt.status) + }) + provider := NewCompatibleProvider( + "test-key", + providers.ProviderOptions{HTTPClient: server.Client(), Resilience: providertest.Resilience()}, + CompatibleProviderConfig{ProviderName: "stt", BaseURL: server.URL}, + ) + + _, err := provider.ListModels(context.Background()) + require.Error(t, err) + assert.Equal(t, tt.unsupported, errors.Is(err, core.ErrModelListingUnsupported)) + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr) + assert.Equal(t, gatewayErr.Error(), err.Error()) + }) + } +} + func TestCompatibleProvider_AdaptChatRequest_RewritesBodyOnChatAndStream(t *testing.T) { server, capture := providertest.JSONServer(t, http.StatusOK, `{"id":"resp","model":"quirk-1","choices":[]}`) provider := NewCompatibleProvider( diff --git a/internal/providers/openrouter/openrouter.go b/internal/providers/openrouter/openrouter.go index 70ee13f6d..1222473a7 100644 --- a/internal/providers/openrouter/openrouter.go +++ b/internal/providers/openrouter/openrouter.go @@ -141,8 +141,9 @@ func (p *Provider) ListModels(ctx context.Context) (*core.ModelsResponse, error) // servableOpenRouterModalities are output modalities the gateway can reach on // OpenRouter: text and image generation flow through chat completions, -// embeddings through /embeddings, and speech/transcription through the -// /audio endpoints. A model listing none of these (rerank-only, video) has no +// embeddings through /embeddings, speech/transcription through the /audio +// endpoints, and decisions (System One models such as Jev) through +// /v1/systemone. A model listing none of these (rerank-only, video) has no // working endpoint here. var servableOpenRouterModalities = map[string]struct{}{ "text": {}, @@ -150,6 +151,7 @@ var servableOpenRouterModalities = map[string]struct{}{ "embeddings": {}, "speech": {}, "transcription": {}, + "decisions": {}, } func openrouterServable(m openrouterModel) bool { @@ -171,6 +173,7 @@ func openrouterServable(m openrouterModel) bool { // ID inference. func openrouterMetadata(m openrouterModel) *core.ModelMetadata { modes := make([]string, 0, 2) + decisions := false // "rerank" is deliberately not mapped: the gateway has no rerank surface // on OpenRouter, and the rerank mode would sort the model into the // Embeddings category despite being unreachable here. @@ -186,6 +189,8 @@ func openrouterMetadata(m openrouterModel) *core.ModelMetadata { modes = append(modes, "audio_speech") case "transcription": modes = append(modes, "audio_transcription") + case "decisions": + decisions = true } } pricing := openrouterPricing(m) @@ -196,9 +201,15 @@ func openrouterMetadata(m openrouterModel) *core.ModelMetadata { Capabilities: capabilities, Pricing: pricing, } - if len(modes) > 0 { + switch { + case len(modes) > 0: meta.Modes = modes meta.Categories = core.CategoriesForModes(modes) + case decisions: + // A decision model answers System One requests only. It has no + // generation mode to claim, so like the jev provider's models it is + // a utility model that no OpenAI endpoint routes to. + meta.Categories = []core.ModelCategory{core.CategoryUtility} } if m.ContextLength > 0 { contextWindow := m.ContextLength @@ -207,7 +218,7 @@ func openrouterMetadata(m openrouterModel) *core.ModelMetadata { if m.TopProvider.MaxCompletionTokens > 0 { meta.MaxOutputTokens = new(m.TopProvider.MaxCompletionTokens) } - if len(modes) == 0 && meta.ContextWindow == nil && meta.MaxOutputTokens == nil && pricing == nil && + if len(meta.Categories) == 0 && meta.ContextWindow == nil && meta.MaxOutputTokens == nil && pricing == nil && capabilities == nil && meta.DisplayName == "" && meta.Description == "" { return nil } diff --git a/internal/providers/openrouter/openrouter_test.go b/internal/providers/openrouter/openrouter_test.go index 0a50da65f..39bffeb7d 100644 --- a/internal/providers/openrouter/openrouter_test.go +++ b/internal/providers/openrouter/openrouter_test.go @@ -59,6 +59,11 @@ func TestListModels_StampsArchitectureModalities(t *testing.T) { "architecture":{"input_modalities":["text"],"output_modalities":["rerank"]}}, {"id":"acme/video-only","created":1721260800, "architecture":{"input_modalities":["text"],"output_modalities":["video"]}}, + {"id":"typesafe/jev-1.13","name":"TypeSafe: Jev 1.13","created":1789689684,"context_length":32000, + "architecture":{"input_modalities":["text"],"output_modalities":["decisions"]}, + "pricing":{"prompt":"0.000000042","completion":"0"}}, + {"id":"~typesafe/jev-latest","created":1789689684, + "architecture":{"input_modalities":["text"],"output_modalities":["decisions"]}}, {"id":"mystery/no-architecture","created":1721260800} ]}`) provider := newTestProvider(server.URL, server.Client()) @@ -72,7 +77,7 @@ func TestListModels_StampsArchitectureModalities(t *testing.T) { // embedding models would never enter the catalog. assert.Equal(t, "all", req.Query.Get("output_modalities")) - require.Len(t, resp.Data, 6) + require.Len(t, resp.Data, 8) byID := modelsByID(resp) chat := byID["openai/gpt-4o-mini"] @@ -114,6 +119,17 @@ func TestListModels_StampsArchitectureModalities(t *testing.T) { require.NotNil(t, stt.Metadata) assert.Equal(t, []string{"audio_transcription"}, stt.Metadata.Modes) + // Decision models serve /v1/systemone only: listed, but as utility models + // with no mode that would route an OpenAI request to them. + for _, id := range []string{"typesafe/jev-1.13", "~typesafe/jev-latest"} { + decision := byID[id] + require.NotNil(t, decision.Metadata, id) + assert.Empty(t, decision.Metadata.Modes, id) + assert.Equal(t, []core.ModelCategory{core.CategoryUtility}, decision.Metadata.Categories, id) + } + require.NotNil(t, byID["typesafe/jev-1.13"].Metadata.Pricing) + assert.InDelta(t, 0.042, *byID["typesafe/jev-1.13"].Metadata.Pricing.InputPerMtok, 1e-9) + assert.NotContains(t, byID, "cohere/rerank-only") assert.NotContains(t, byID, "acme/video-only") diff --git a/internal/providers/provider_availability.go b/internal/providers/provider_availability.go index dafa6e341..1c8c4f27e 100644 --- a/internal/providers/provider_availability.go +++ b/internal/providers/provider_availability.go @@ -2,6 +2,7 @@ package providers import ( "context" + "strings" "time" "github.com/enterpilot/gomodel/internal/core" @@ -29,6 +30,18 @@ func (r *ModelRegistry) probeAvailability(ctx context.Context, provider core.Pro defer cancel() err := checker.CheckAvailability(probeCtx) + if modelListingUnsupported(err) && r.hasConfiguredProviderModels(providerName) { + // Probes that list models (Ollama, Bedrock Mantle) fail on servers + // without a /models endpoint. The server answered, and its configured + // models are the inventory, so it is reachable. + err = nil + } r.RecordAvailabilityCheck(providerName, err) return err } + +func (r *ModelRegistry) hasConfiguredProviderModels(providerName string) bool { + r.mu.RLock() + defer r.mu.RUnlock() + return len(r.configuredProviderModels[strings.TrimSpace(providerName)]) > 0 +} diff --git a/internal/providers/provider_status.go b/internal/providers/provider_status.go index 83d6d684c..daabd19e6 100644 --- a/internal/providers/provider_status.go +++ b/internal/providers/provider_status.go @@ -64,6 +64,7 @@ type ProviderRuntimeSnapshot struct { LastAvailabilityOKAt *time.Time `json:"last_availability_ok_at,omitempty"` LastAvailabilityError string `json:"last_availability_error,omitempty"` InventoryStale bool `json:"inventory_stale,omitempty"` + ModelListingUnsupported bool `json:"model_listing_unsupported,omitempty"` } type providerRuntimeState struct { @@ -80,6 +81,9 @@ type providerRuntimeState struct { // provider with an honest 502/503) but are skipped by ModelAvailable, // which load balancing uses to route around the provider. inventoryStale bool + // modelListingUnsupported marks a provider without a /models endpoint + // whose inventory comes from its configured model list. + modelListingUnsupported bool } // SanitizeProviderConfigs converts effective provider configs into a stable, diff --git a/internal/providers/registry_availability.go b/internal/providers/registry_availability.go index b15d5f615..35678b0a1 100644 --- a/internal/providers/registry_availability.go +++ b/internal/providers/registry_availability.go @@ -141,6 +141,7 @@ func (r *ModelRegistry) ProviderRuntimeSnapshots() []ProviderRuntimeSnapshot { LastAvailabilityOKAt: timePtrUTC(state.lastAvailabilityOKAt), LastAvailabilityError: state.lastAvailabilityError, InventoryStale: state.inventoryStale, + ModelListingUnsupported: state.modelListingUnsupported, }) } r.mu.RUnlock() diff --git a/internal/providers/registry_init.go b/internal/providers/registry_init.go index 1fc315ef4..f3aa4dbbc 100644 --- a/internal/providers/registry_init.go +++ b/internal/providers/registry_init.go @@ -175,7 +175,9 @@ func (r *ModelRegistry) fetchAllProviderModels( "reason", string(configuredReason), "configured_models", len(configuredModels), } - if err != nil { + if configuredReason == configuredProviderModelsUpstreamUnlisted { + slog.Debug("provider does not list models, using configured provider models", attrs...) + } else if err != nil { configuredUpstreamError = err.Error() attrs = append(attrs, "error", err) slog.Warn("upstream ListModels failed, using configured provider models", attrs...) @@ -234,30 +236,37 @@ func (r *ModelRegistry) fetchAllProviderModels( } runtimeUpdate := providerRuntimeState{ - registered: true, - lastModelFetchAt: fetchAt, - lastModelFetchError: configuredUpstreamError, + registered: true, + lastModelFetchAt: fetchAt, + lastModelFetchError: configuredUpstreamError, + modelListingUnsupported: configuredReason == configuredProviderModelsUpstreamUnlisted, } // Mark the inventory as authoritatively populated when this fetch is the - // last word on the provider's model list. That covers two cases: + // last word on the provider's model list. That covers these cases: // - upstream succeeded (no allowlist, or allowlist overlaid on a real // response) — reason is configuredProviderModelsNotApplied // - allowlist mode intentionally skipped upstream and produced the // inventory from configuration — reason is configuredProviderModelsAllowlist // - merge mode overlaid configured models on a healthy upstream // response — reason is configuredProviderModelsMerge + // - the upstream has no /models endpoint, so the configured list is + // the whole inventory — reason is configuredProviderModelsUpstreamUnlisted // Fallback cases (configured*UpstreamError, *Nil, *Empty) keep // lastModelFetchSuccessAt unset so health surfaces "live refresh failed, // serving configured fallback". if configuredReason == configuredProviderModelsNotApplied || configuredReason == configuredProviderModelsAllowlist || - configuredReason == configuredProviderModelsMerge { + configuredReason == configuredProviderModelsMerge || + configuredReason == configuredProviderModelsUpstreamUnlisted { runtimeUpdate.lastModelFetchSuccessAt = fetchAt } - // Merge keeps availability signals too: the upstream call actually - // succeeded, unlike the fallback reasons. + // Merge and unlisted keep availability signals too: the upstream + // answered, unlike the fallback reasons. For unlisted this also clears + // a startup probe that hit the same missing /models endpoint before + // configured models were known. if configuredReason == configuredProviderModelsNotApplied || - configuredReason == configuredProviderModelsMerge { + configuredReason == configuredProviderModelsMerge || + configuredReason == configuredProviderModelsUpstreamUnlisted { runtimeUpdate.lastAvailabilityCheckAt = fetchAt runtimeUpdate.lastAvailabilityOKAt = fetchAt } @@ -470,6 +479,7 @@ func (r *ModelRegistry) applyProviderRuntimeUpdatesLocked(updates map[string]pro // this matters in particular for allowlist-mode refreshes which // don't bump SuccessAt but still produce usable models. current.lastModelFetchError = optionalFailureMessage(update.lastModelFetchError) + current.modelListingUnsupported = update.modelListingUnsupported } if !update.lastModelFetchSuccessAt.IsZero() { current.lastModelFetchSuccessAt = update.lastModelFetchSuccessAt diff --git a/internal/providers/registry_lookup.go b/internal/providers/registry_lookup.go index 5a7b68264..64db52c2a 100644 --- a/internal/providers/registry_lookup.go +++ b/internal/providers/registry_lookup.go @@ -171,6 +171,28 @@ func (r *ModelRegistry) ModelAvailable(model string) bool { return !r.providerRuntime[info.ProviderName].inventoryStale } +// AcceptsUnlistedModel reports whether a provider-qualified model the catalog +// does not list can still be served, because its provider accepts IDs it does +// not list (see core.UnlistedModelAcceptor) and its inventory is fresh. A bare +// name never qualifies: it does not say which provider to use. +func (r *ModelRegistry) AcceptsUnlistedModel(model string) bool { + providerName, _ := splitModelSelector(strings.TrimSpace(model)) + if providerName == "" { + return false + } + r.mu.RLock() + defer r.mu.RUnlock() + + for _, provider := range r.providers { + if r.providerNames[provider] != providerName { + continue + } + acceptor, ok := provider.(core.UnlistedModelAcceptor) + return ok && acceptor.AcceptsUnlistedModels() && !r.providerRuntime[providerName].inventoryStale + } + return false +} + // GetProviderType returns the provider type string for the given model. // Returns empty string if the model is not found. func (r *ModelRegistry) GetProviderType(model string) string { diff --git a/internal/providers/registry_normalization_test.go b/internal/providers/registry_normalization_test.go index 2ce0e6270..3d29b31d5 100644 --- a/internal/providers/registry_normalization_test.go +++ b/internal/providers/registry_normalization_test.go @@ -6,6 +6,7 @@ import ( "testing" "github.com/enterpilot/gomodel/internal/core" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -340,3 +341,44 @@ func TestRecordAvailabilityCheckKeepsFailureMarker(t *testing.T) { }) } } + +// LookupModel describes one catalog model by any selector the router resolves, +// including OpenRouter's "~"-prefixed alias IDs. +func TestRouterLookupModel(t *testing.T) { + registry := newTestRegistryWithModels(registryModelEntry{ + provider: &mockProvider{name: "openrouter"}, + providerName: "openrouter", + providerType: "openrouter", + modelID: "~typesafe/jev-latest", + }) + router, err := NewRouter(registry) + require.NoError(t, err) + + for _, selector := range []string{"openrouter/~typesafe/jev-latest", "~typesafe/jev-latest"} { + model, ok := router.LookupModel(selector) + require.True(t, ok, selector) + assert.Equal(t, "~typesafe/jev-latest", model.ID, selector) + } + _, ok := router.LookupModel("openrouter/unknown") + assert.False(t, ok) + + _, ok = (&Router{}).LookupModel("openrouter/~typesafe/jev-latest") + assert.False(t, ok, "a lookup without single-model access describes nothing") +} + +// ProviderNamesForType lists every configured instance of one type, so a +// caller can tell a single jev provider from several. +func TestRouterProviderNamesForType(t *testing.T) { + registry := newTestRegistryWithModels( + registryModelEntry{provider: &mockProvider{name: "kev"}, providerName: "kev", providerType: "jev", modelID: "kev-latest"}, + registryModelEntry{provider: &mockProvider{name: "jev"}, providerName: "jev", providerType: "jev", modelID: "jev-latest"}, + registryModelEntry{provider: &mockProvider{name: "openrouter"}, providerName: "openrouter", providerType: "openrouter", modelID: "typesafe/jev-1.13"}, + ) + router, err := NewRouter(registry) + require.NoError(t, err) + + assert.Equal(t, []string{"jev", "kev"}, router.ProviderNamesForType("jev")) + assert.Equal(t, []string{"openrouter"}, router.ProviderNamesForType("openrouter")) + assert.Empty(t, router.ProviderNamesForType("anthropic")) + assert.Empty(t, router.ProviderNamesForType("")) +} diff --git a/internal/providers/registry_test.go b/internal/providers/registry_test.go index 58ad21876..cdea37ca7 100644 --- a/internal/providers/registry_test.go +++ b/internal/providers/registry_test.go @@ -179,6 +179,39 @@ func TestModelRegistry(t *testing.T) { require.Nil(t, snapshots[0].LastModelFetchSuccessAt) }) + t.Run("ConfiguredModelsWithoutModelsEndpointAreHealthy", func(t *testing.T) { + registry := NewModelRegistry() + mock := ®istryMockProvider{ + name: "stt", + err: core.MarkModelListingUnsupported(core.ParseProviderError("openai", http.StatusNotFound, []byte("404 Not Found"), nil)), + } + registry.RegisterProviderWithNameAndType(mock, "stt", "openai") + registry.SetProviderConfiguredModels("stt", []string{"whisper-1"}) + + err := registry.Initialize(context.Background()) + require.NoError(t, err) + require.True(t, registry.Supports("whisper-1")) + require.Empty(t, registry.FailedProviderNames()) + + snapshots := registry.ProviderRuntimeSnapshots() + require.Len(t, snapshots, 1) + assert.Empty(t, snapshots[0].LastModelFetchError) + assert.NotNil(t, snapshots[0].LastModelFetchSuccessAt) + assert.True(t, snapshots[0].ModelListingUnsupported) + + // A server that later starts listing models drops the marker. + mock.err = nil + mock.modelsResponse = &core.ModelsResponse{ + Object: "list", + Data: []core.Model{{ID: "whisper-1", Object: "model", OwnedBy: "stt"}}, + } + err = registry.Initialize(context.Background()) + require.NoError(t, err) + snapshots = registry.ProviderRuntimeSnapshots() + require.Len(t, snapshots, 1) + assert.False(t, snapshots[0].ModelListingUnsupported) + }) + t.Run("SuccessfulLiveModelFetchClearsAvailabilityError", func(t *testing.T) { registry := NewModelRegistry() mock := ®istryMockProvider{ @@ -1122,6 +1155,69 @@ func (p *availabilityFailingProvider) CheckAvailability(context.Context) error { return p.availabilityErr } +// Providers whose availability probe lists models (Ollama, Bedrock Mantle) hit +// the same 404 as the refresh on a server without /models. With configured +// models the provider must still leave the recheck set and pass the refresh +// gate. +func TestRefreshProviderModels_MissingModelsEndpointPassesAvailabilityGate(t *testing.T) { + notFound := core.MarkModelListingUnsupported(core.ParseProviderError("ollama", http.StatusNotFound, []byte("404 page not found"), nil)) + registry := NewModelRegistry() + stt := &availabilityFailingProvider{ + registryMockProvider: ®istryMockProvider{name: "stt", err: notFound}, + availabilityErr: notFound, + } + registry.RegisterProviderWithNameAndType(stt, "stt", "ollama") + // Startup probes before configured models are registered. + registry.RecordAvailabilityCheck("stt", notFound) + registry.SetProviderConfiguredModels("stt", []string{"whisper-1"}) + + err := registry.Initialize(context.Background()) + require.NoError(t, err) + require.Empty(t, registry.FailedProviderNames()) + + count, err := registry.RefreshProviderModels(context.Background(), "stt") + require.NoError(t, err) + assert.Equal(t, 1, count) + assert.True(t, registry.ModelAvailable("stt/whisper-1")) + + snapshots := registry.ProviderRuntimeSnapshots() + require.Len(t, snapshots, 1) + assert.Empty(t, snapshots[0].LastAvailabilityError) + assert.Empty(t, snapshots[0].LastModelFetchError) +} + +// Without configured models a 404 probe is still a failure. +func TestRefreshProviderModels_MissingModelsEndpointWithoutConfiguredModelsFails(t *testing.T) { + notFound := core.MarkModelListingUnsupported(core.ParseProviderError("ollama", http.StatusNotFound, []byte("404 page not found"), nil)) + registry := NewModelRegistry() + stt := &availabilityFailingProvider{ + registryMockProvider: ®istryMockProvider{name: "stt", err: notFound}, + availabilityErr: notFound, + } + registry.RegisterProviderWithNameAndType(stt, "stt", "ollama") + + _, err := registry.RefreshProviderModels(context.Background(), "stt") + require.Error(t, err) + assert.Equal(t, []string{"stt"}, registry.FailedProviderNames()) +} + +// A 404 from a probe that does not call /models (Bedrock's control plane) is a +// real failure even when models are configured. +func TestRefreshProviderModels_UnmarkedNotFoundProbeFailsWithConfiguredModels(t *testing.T) { + notFound := core.ParseProviderError("bedrock", http.StatusNotFound, nil, nil) + registry := NewModelRegistry() + bedrock := &availabilityFailingProvider{ + registryMockProvider: ®istryMockProvider{name: "bedrock", err: notFound}, + availabilityErr: notFound, + } + registry.RegisterProviderWithNameAndType(bedrock, "bedrock", "bedrock") + registry.SetProviderConfiguredModels("bedrock", []string{"anthropic.claude"}) + + _, err := registry.RefreshProviderModels(context.Background(), "bedrock") + require.Error(t, err) + assert.Equal(t, []string{"bedrock"}, registry.FailedProviderNames()) +} + // A failed availability check during a per-provider refresh marks the // provider stale just like a failed model fetch. func TestRefreshProviderModels_AvailabilityFailureMarksStale(t *testing.T) { @@ -2233,3 +2329,30 @@ func TestSetModelList_ClearsETag(t *testing.T) { got := registry.currentModelListETag("https://example.test/models.min.json") require.Empty(t, got) } + +// unlistedAcceptingProvider serves model IDs it does not list, as a jev +// provider serves pinned versions. +type unlistedAcceptingProvider struct { + registryMockProvider +} + +func (p *unlistedAcceptingProvider) AcceptsUnlistedModels() bool { return true } + +func TestModelRegistryAcceptsUnlistedModel(t *testing.T) { + registry := NewModelRegistry() + registry.RegisterProviderWithNameAndType(&unlistedAcceptingProvider{}, "jev", "jev") + registry.RegisterProviderWithNameAndType(®istryMockProvider{name: "openai"}, "openai", "openai") + + tests := []struct { + model string + want bool + }{ + {model: "jev/jev-1.13.0", want: true}, + {model: "jev-1.13.0", want: false}, + {model: "openai/gpt-9", want: false}, + {model: "unknown/jev-1.13.0", want: false}, + } + for _, tt := range tests { + assert.Equal(t, tt.want, registry.AcceptsUnlistedModel(tt.model), tt.model) + } +} diff --git a/internal/providers/router_models.go b/internal/providers/router_models.go index 4f943569e..80779bb49 100644 --- a/internal/providers/router_models.go +++ b/internal/providers/router_models.go @@ -182,3 +182,34 @@ func (r *Router) NativeResponseProviderTypes() []string { return ok }) } + +// LookupModel returns a copy of the catalog entry for a model selector, or +// false when the model is unknown or the lookup cannot describe one model. +func (r *Router) LookupModel(model string) (*core.Model, bool) { + if r.caps.modelInfo == nil { + return nil, false + } + info := r.caps.modelInfo.GetModel(model) + if info == nil { + return nil, false + } + cloned := info.Model + return &cloned, true +} + +// ProviderNamesForType lists the configured provider instance names of one +// type, sorted, or nil when the lookup cannot enumerate its providers. +func (r *Router) ProviderNamesForType(providerType string) []string { + providerType = strings.TrimSpace(providerType) + if providerType == "" || r.caps.nameLister == nil { + return nil + } + var names []string + for _, name := range r.caps.nameLister.ProviderNames() { + if r.GetProviderTypeForName(name) == providerType { + names = append(names, name) + } + } + sort.Strings(names) + return names +} diff --git a/internal/responsecache/responsecache.go b/internal/responsecache/responsecache.go index c43c9832c..a12124e10 100644 --- a/internal/responsecache/responsecache.go +++ b/internal/responsecache/responsecache.go @@ -304,3 +304,11 @@ func NewResponseCacheMiddlewareWithStore(store cache.Store, ttl time.Duration) * simple: newSimpleCacheMiddleware(store, ttl, nil), } } + +// NewResponseCacheMiddlewareWithStoreAndUsage creates middleware with a custom +// store that records cache hits in usage (for testing). +func NewResponseCacheMiddlewareWithStoreAndUsage(store cache.Store, ttl time.Duration, usageLogger usage.LoggerInterface, pricingResolver usage.PricingResolver) *ResponseCacheMiddleware { + return &ResponseCacheMiddleware{ + simple: newSimpleCacheMiddleware(store, ttl, newUsageHitRecorder(usageLogger, pricingResolver)), + } +} diff --git a/internal/responsecache/usage_hit.go b/internal/responsecache/usage_hit.go index 903fc3fa4..b232cb7ec 100644 --- a/internal/responsecache/usage_hit.go +++ b/internal/responsecache/usage_hit.go @@ -4,6 +4,8 @@ import ( "log/slog" "strings" + "github.com/goccy/go-json" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/usage" ) @@ -45,10 +47,8 @@ func newUsageHitRecorder(logger usage.LoggerInterface, pricingResolver usage.Pri requestID = ex.RequestHeader(core.RequestIDHeader) } - var pricing *core.ModelPricing - if pricingResolver != nil { - pricing = pricingResolver.ResolvePricing(model, cacheHitPricingProvider(provider, providerName)) - } + pricing := usage.ResolveServedModelPricing(pricingResolver, model, cacheHitPricingProvider(provider, providerName), + func() string { return cachedAnsweredModel(body) }) entry := usage.ExtractFromCachedResponseBody(body, requestID, model, provider, endpoint, cacheType, pricing) if entry == nil { @@ -62,6 +62,18 @@ func newUsageHitRecorder(logger usage.LoggerInterface, pricingResolver usage.Pri } } +// cachedAnsweredModel returns the model a cached JSON answer names, such as +// jev-1.13.0 for a request routed to jev-latest, or "" for other bodies. +func cachedAnsweredModel(body []byte) string { + var answer struct { + Model string `json:"model"` + } + if err := json.Unmarshal(body, &answer); err != nil { + return "" + } + return answer.Model +} + func cacheHitPricingProvider(provider, providerName string) string { if name := strings.TrimSpace(providerName); name != "" { return name diff --git a/internal/server/auth.go b/internal/server/auth.go index a6f57580c..7d73db6b9 100644 --- a/internal/server/auth.go +++ b/internal/server/auth.go @@ -100,10 +100,11 @@ func NewAuthMiddleware(cfg AuthMiddlewareConfig) echo.MiddlewareFunc { // sessions. Hide every identity value installed by outer extension // middleware before validating the selected credential; clearing only // the response header would leave downstream context consumers scoped - // to the wrong principal. + // to the wrong principal. The caller's own user-path header is not + // such an identity and is restored (see transportOwnedUserPath). setAuthenticationUserHeader(c, "") ctx := ext.WithoutAuthentication(c.Request().Context()) - ctx = core.WithEffectiveUserPath(ctx, "") + ctx = core.WithEffectiveUserPath(ctx, transportOwnedUserPath(c.Request(), userPathHeaderName)) ctx = core.WithCredentialAllowedModels(ctx, nil) ctx = core.WithAccessScope(ctx, core.AccessScope{}) c.SetRequest(c.Request().WithContext(ctx)) @@ -158,6 +159,27 @@ func NewAuthMiddleware(cfg AuthMiddlewareConfig) echo.MiddlewareFunc { } } +// transportOwnedUserPath returns the request's user-path header for model +// endpoints that own their transport (MCP, realtime, audio uploads), and "" +// for every other endpoint. Those endpoints take no request snapshot, so +// RequestSnapshotCapture seeds the header path as the effective user path; +// without restoring it here, master-key and unbound-key callers lose their +// path there while ingress-managed endpoints keep it through the snapshot. +// Only this middleware and RequestSnapshotCapture write the header, so the +// value is the caller's own, never an outer extension session's. A key with +// a bound user path still overrides it in applyAuthKeyResult. +func transportOwnedUserPath(req *http.Request, headerName string) string { + desc := core.DescribeEndpoint(req.Method, req.URL.Path) + if desc.IngressManaged || !desc.ModelInteraction { + return "" + } + userPath, err := core.NormalizeUserPath(req.Header.Get(headerName)) + if err != nil { + return "" + } + return userPath +} + func hasRequestAuthenticators(authenticators []ext.RequestAuthenticator) bool { for _, authenticator := range authenticators { if !requestAuthenticatorIsNil(authenticator) { diff --git a/internal/server/http.go b/internal/server/http.go index 560683152..3815dff75 100644 --- a/internal/server/http.go +++ b/internal/server/http.go @@ -496,6 +496,11 @@ func New(provider core.RoutableProvider, cfg *Config) *Server { e.POST("/v1/audio/translations", handler.AudioTranslations) e.POST("/v1/images/generations", handler.ImageGenerations) e.POST("/v1/images/edits", handler.ImageEdits) + // System One decisions (Jev / Kev). The handler answers 404 until a jev + // or openrouter provider is configured. + e.POST("/v1/systemone", handler.SystemOne) + e.POST("/v1/systemone/permute", handler.SystemOnePermute) + e.POST("/v1/systemone/separate", handler.SystemOneSeparate) if cfg == nil || cfg.RealtimeEnabled { e.GET("/v1/realtime", handler.Realtime) e.POST("/v1/realtime/calls", handler.RealtimeCalls) diff --git a/internal/server/master_key_user_path_test.go b/internal/server/master_key_user_path_test.go index ff17be0cf..d016d449e 100644 --- a/internal/server/master_key_user_path_test.go +++ b/internal/server/master_key_user_path_test.go @@ -101,3 +101,67 @@ func TestMasterKeyUserPathHeaderScopesRestrictedModelAccess(t *testing.T) { }) } } + +// TestTransportOwnedEndpointsKeepHeaderUserPath covers model endpoints that +// own their transport (MCP, realtime, audio uploads). They take no request +// snapshot, so the header path lives only in the effective user path, and the +// explicit-credential reset must not erase it: a master-key or unbound-key +// caller keeps its header path, a bound key still wins, and identity from an +// outer extension session never survives an explicit credential. +func TestTransportOwnedEndpointsKeepHeaderUserPath(t *testing.T) { + authenticator := mockAuthenticator{ + enabled: true, + tokenToID: map[string]string{"sk_bound": "key-bound", "sk_unbound": "key-unbound"}, + tokenPath: map[string]string{"sk_bound": "/team/bound"}, + } + tests := []struct { + name string + path string + token string + headerPath string + outerIdentity string + // configuredHeader is the server's USER_PATH_HEADER; empty keeps the default. + configuredHeader string + want string + }{ + {name: "master key on /mcp keeps header path", path: "/mcp", token: "master-key", headerPath: "/eng/platform", want: "/eng/platform"}, + {name: "master key on pinned /mcp/{server}", path: "/mcp/github", token: "master-key", headerPath: "/eng", want: "/eng"}, + {name: "master key on audio transcription", path: "/v1/audio/transcriptions", token: "master-key", headerPath: "/team/x", want: "/team/x"}, + {name: "master key on realtime", path: "/v1/realtime", token: "master-key", headerPath: "/team/x", want: "/team/x"}, + {name: "master key without header stays global", path: "/mcp", token: "master-key", want: ""}, + {name: "unbound managed key keeps header path", path: "/mcp", token: "sk_unbound", headerPath: "/eng/platform", want: "/eng/platform"}, + {name: "bound managed key wins over header", path: "/mcp", token: "sk_bound", headerPath: "/eng/platform", want: "/team/bound"}, + {name: "outer extension identity is dropped", path: "/mcp", token: "master-key", outerIdentity: "/ext/session", want: ""}, + {name: "header replaces outer extension identity", path: "/mcp", token: "master-key", outerIdentity: "/ext/session", headerPath: "/eng", want: "/eng"}, + {name: "configured header name on /mcp", path: "/mcp", token: "master-key", configuredHeader: "X-Tenant-Path", headerPath: "/eng", want: "/eng"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + var got string + auth := AuthMiddlewareWithAuthenticator("master-key", authenticator, nil, tt.configuredHeader)(func(c *echo.Context) error { + got = core.UserPathFromContext(c.Request().Context()) + return c.String(http.StatusOK, "ok") + }) + // An outer extension session installs its identity before the + // gateway's auth middleware runs. + outer := func(c *echo.Context) error { + if tt.outerIdentity != "" { + req := c.Request() + c.SetRequest(req.WithContext(core.WithEffectiveUserPath(req.Context(), tt.outerIdentity))) + } + return auth(c) + } + chain := RequestSnapshotCapture(tt.configuredHeader)(outer) + + opts := []echotest.Option{echotest.WithHeader("Authorization", "Bearer "+tt.token)} + if tt.headerPath != "" { + opts = append(opts, echotest.WithHeader(core.UserPathHeaderName(tt.configuredHeader), tt.headerPath)) + } + c, rec := echotest.Post(t, tt.path, `{}`, opts...) + + require.NoError(t, chain(c)) + require.Equal(t, http.StatusOK, rec.Code) + assert.Equal(t, tt.want, got) + }) + } +} diff --git a/internal/server/messages_native.go b/internal/server/messages_native.go index d460297a1..d451a9319 100644 --- a/internal/server/messages_native.go +++ b/internal/server/messages_native.go @@ -3,8 +3,8 @@ package server import ( "bytes" "context" - // encoding/json rather than goccy: rewriteMessagesModel needs the - // decoder's InputOffset to splice the model value in place. + // encoding/json rather than goccy: replaceTopLevelMember needs the + // decoder's InputOffset to splice a value in place. "encoding/json" "errors" "io" @@ -138,6 +138,17 @@ func rewriteMessagesModel(body []byte, model string) ([]byte, error) { if strings.TrimSpace(model) == "" { return body, nil } + encoded, err := json.Marshal(model) + if err != nil { + return nil, err + } + return replaceTopLevelMember(body, "model", encoded) +} + +// replaceTopLevelMember returns body with the value of its top-level key +// member replaced by value, splicing only those bytes. The body is returned +// unchanged when the member is absent or already holds value. +func replaceTopLevelMember(body []byte, key string, value []byte) ([]byte, error) { dec := json.NewDecoder(bytes.NewReader(body)) tok, err := dec.Token() if err != nil { @@ -146,45 +157,36 @@ func rewriteMessagesModel(body []byte, model string) ([]byte, error) { if delim, ok := tok.(json.Delim); !ok || delim != '{' { return nil, errors.New("request body is not a JSON object") } - // Walk every top-level member and remember the span of the last "model" + // Walk every top-level member and remember the span of the last matching // value: decoders keep the last duplicate member, so that is the one the - // resolved model came from and the one to rewrite. - var modelRaw json.RawMessage - var modelEnd int64 + // gateway read and the one to rewrite. + var found json.RawMessage + var foundEnd int64 for dec.More() { keyTok, err := dec.Token() if err != nil { return nil, err } - key, _ := keyTok.(string) + name, _ := keyTok.(string) var raw json.RawMessage if err := dec.Decode(&raw); err != nil { return nil, err } - if key != "model" { + if name != key { continue } - modelRaw = raw - modelEnd = dec.InputOffset() + found = raw + foundEnd = dec.InputOffset() } - if modelRaw == nil { + if found == nil || bytes.Equal(found, value) { return body, nil } - var current string - _ = json.Unmarshal(modelRaw, ¤t) - if current == model { - return body, nil - } - encoded, err := json.Marshal(model) - if err != nil { - return nil, err - } - // The model value is a scalar, so modelRaw holds its exact source bytes - // and modelEnd points just past them. - start := modelEnd - int64(len(modelRaw)) - rewritten := make([]byte, 0, int64(len(body))-int64(len(modelRaw))+int64(len(encoded))) + // The decoder hands back the value's exact source bytes, and foundEnd + // points just past them. + start := foundEnd - int64(len(found)) + rewritten := make([]byte, 0, int64(len(body))-int64(len(found))+int64(len(value))) rewritten = append(rewritten, body[:start]...) - rewritten = append(rewritten, encoded...) - rewritten = append(rewritten, body[modelEnd:]...) + rewritten = append(rewritten, value...) + rewritten = append(rewritten, body[foundEnd:]...) return rewritten, nil } diff --git a/internal/server/model_validation.go b/internal/server/model_validation.go index 4a2d18d60..dbcd57b09 100644 --- a/internal/server/model_validation.go +++ b/internal/server/model_validation.go @@ -103,6 +103,13 @@ func deriveWorkflowWithPolicy( } return workflow, nil + case core.OperationSystemOne: + // The System One handler resolves the model itself: only it knows + // whether the endpoint is available (answering 404 before any model + // error) and when an unlisted pinned version may still route to a + // jev provider. + return nil, nil + case core.OperationChatCompletions, core.OperationResponses, core.OperationEmbeddings: workflow.Mode = core.ExecutionModeTranslated if desc.BodyMode != core.BodyModeJSON { @@ -126,6 +133,9 @@ func deriveWorkflowWithPolicy( } return workflow, nil } + if systemOneOnlyModel(provider, resolution) { + return nil, systemOneOnlyModelError(desc.Operation, resolution) + } return translatedWorkflow(c.Request().Context(), requestID, desc, resolution, policyResolver) default: diff --git a/internal/server/previous_response_test.go b/internal/server/previous_response_test.go index e22376ae2..1296fa0e0 100644 --- a/internal/server/previous_response_test.go +++ b/internal/server/previous_response_test.go @@ -121,6 +121,57 @@ func TestResponsesWithPreviousResponseID_ChainCarriesFullHistory(t *testing.T) { require.Len(t, second.InputItems, 1) } +// TestResponsesWithPreviousResponseID_KimicodeChainReplaysHistory covers the +// native-Responses kimicode provider chaining through the gateway: kimicode +// has no Responses lifecycle, so the gateway treats it as translated and, +// with a response store, replays the stored chain into input instead of +// forwarding the id. The replayed items — reasoning output included, item +// ids stripped — are what reaches kimicode's /responses upstream. +func TestResponsesWithPreviousResponseID_KimicodeChainReplaysHistory(t *testing.T) { + provider := previousResponseTestProvider(t, "kimicode") + srv := New(provider, nil) + + // A turn the gateway served for kimicode earlier: the snapshot holds the + // client's input items and the provider's output, reasoning included. + err := srv.handler.currentResponseStore().Create(context.Background(), &responsestore.StoredResponse{ + Response: &core.ResponsesResponse{ + ID: "resp_kimi_1", Object: "response", Status: "completed", + Output: []core.ResponsesOutputItem{ + { + ID: "rs_1", Type: "reasoning", Status: "completed", + ExtraFields: core.UnknownJSONFieldsFromMap(map[string]json.RawMessage{ + "summary": json.RawMessage(`[{"type":"summary_text","text":"thinking about zebras"}]`), + }), + }, + {ID: "msg_1", Type: "message", Role: "assistant", Content: []core.ResponsesContentItem{{Type: "output_text", Text: "the word is zebra"}}}, + }, + }, + InputItems: []json.RawMessage{json.RawMessage(`{"id":"in_1","type":"message","role":"user","content":[{"type":"input_text","text":"remember: zebra"}]}`)}, + Provider: "kimicode", + }) + require.NoError(t, err) + + rec := postResponses(t, srv, `{"model":"gpt-5-mini","input":"what is the word?","previous_response_id":"resp_kimi_1"}`) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + forwarded := provider.capturedResponsesReq + require.NotNil(t, forwarded) + require.Empty(t, forwarded.PreviousResponseID, "translated providers get the id stripped before dispatch") + + items := forwardedInputItems(t, provider.capturingProvider) + require.Len(t, items, 4) + require.Equal(t, "reasoning", items[1]["type"], "stored reasoning output must replay unchanged: %#v", items[1]) + summary, _ := json.Marshal(items[1]["summary"]) + require.Contains(t, string(summary), "thinking about zebras") + for i, item := range items[:3] { + _, hasID := item["id"] + require.False(t, hasID, "stored item id must be stripped before dispatch (item %d): %#v", i, item) + } + text, _ := json.Marshal(items[2]["content"]) + require.Contains(t, string(text), "the word is zebra") + require.Equal(t, "user", items[3]["role"], "the client's own turn is replayed last") +} + func TestResponsesWithPreviousResponseID_StreamingChainedTurn(t *testing.T) { provider := previousResponseTestProvider(t, "anthropic") provider.streamData = "event: response.completed\ndata: {\"type\":\"response.completed\",\"response\":{\"id\":\"resp_s\",\"object\":\"response\",\"status\":\"completed\",\"output\":[]}}\n\ndata: [DONE]\n\n" diff --git a/internal/server/systemone_dispatch.go b/internal/server/systemone_dispatch.go new file mode 100644 index 000000000..e9ad381a2 --- /dev/null +++ b/internal/server/systemone_dispatch.go @@ -0,0 +1,158 @@ +package server + +import ( + "bytes" + "context" + "errors" + "io" + "net/http" + "strings" + + "github.com/labstack/echo/v5" + + "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/gateway" + "github.com/enterpilot/gomodel/internal/responsecache" +) + +// dispatchSystemOneWithCache serves a guarded System One body from the +// response cache when the workflow allows it, and forwards it otherwise. +// Only the exact layer applies: the key covers the route, the resolved model, +// the guardrail chain, and the forwarded body, and the semantic layer never +// serves these paths, since a similar state is not the same decision. +func (s *translatedInferenceService) dispatchSystemOneWithCache(c *echo.Context, route systemOneRoute, workflow *core.Workflow, body []byte) error { + dispatch := func() error { return s.dispatchSystemOne(c, route, workflow, body) } + if s.responseCache == nil || !workflow.CacheEnabled() { + return dispatch() + } + c.SetRequest(c.Request().WithContext(s.inference().WithCacheRequestContext(c.Request().Context(), workflow))) + err := s.responseCache.HandleRequest(c, body, dispatch) + if replayErr, ok := errors.AsType[*responsecache.ReplayError](err); ok { + recordCachedStreamError(c, replayErr.Err) + return nil + } + return err +} + +// dispatchSystemOne forwards the body to the resolved route, and to the +// workflow's failover targets while the failover policy allows, then relays +// the answer unchanged with audit and usage accounting. +func (s *translatedInferenceService) dispatchSystemOne(c *echo.Context, route systemOneRoute, workflow *core.Workflow, body []byte) error { + passthroughProvider, ok := s.provider.(core.RoutablePassthrough) + if !ok { + return handleError(c, core.NewInvalidRequestError("provider passthrough is not supported by the current provider router", nil)) + } + // Record each provider attempt so the audit entry shows a failed primary + // and the failover that answered, as it does for chat. + c.SetRequest(c.Request().WithContext(gateway.WithAttemptRecorder(c.Request().Context()))) + s.observeLiveProviderAttempts(c, workflow) + + failovers := len(s.inference().FailoverSelectors(workflow)) + adm, err := enforceAdmission(c, s.rateLimiter, s.budgetChecker, rateLimitRouteFromWorkflow(workflow).withFailovers(failovers)) + if err != nil { + return handleError(c, err) + } + defer adm.release() + ctx := adm.dispatchContext(c.Request().Context()) + + headers := buildPassthroughHeaders(ctx, c.Request().Header) + // The client's Idempotency-Key reaches the primary through the request + // context, which failover attempts clear; forwarded as an explicit header + // it would also mark every failover target's different body. + headers.Del(core.IdempotencyKeyHeader) + eligible := func(selector core.ModelSelector, providerType string) bool { + return s.systemOneUnsupportedReason(route, selector, providerType) == "" + } + resp, executed, meta, err := s.inference().ExecutePassthroughWithFailover(ctx, workflow, eligible, + func(ctx context.Context, selector core.ModelSelector, providerType, providerName string) (*core.PassthroughResponse, error) { + return s.sendSystemOne(ctx, passthroughProvider, route, selector, providerType, providerName, headers, body) + }) + enrichAuditEntryWithProviderAttempts(c) + if err != nil { + return handleError(c, err) + } + + if meta.UsedFailover { + markRequestFailoverUsed(c) + auditlog.EnrichEntryWithFailover(c, meta.FailoverModel) + workflow = executedSystemOneWorkflow(workflow, executed, meta) + storeWorkflow(c, workflow) + } + auditlog.EnrichEntryWithWorkflow(c, workflow) + auditlog.EnrichEntryWithResolvedRoute(c, executed.QualifiedModel(), meta.ProviderType, meta.ProviderName) + info := &core.PassthroughRouteInfo{ + Provider: meta.ProviderType, + ProviderName: meta.ProviderName, + NormalizedEndpoint: route.endpoint, + SemanticOperation: route.operation, + AuditPath: route.path, + Model: executed.Model, + } + return proxyPassthroughResponse(c, s.logger, s.usageLogger, s.pricingResolver, meta.ProviderType, meta.ProviderName, route.endpoint, info, resp) +} + +// maxSystemOneErrorBodyBytes caps how much of an upstream error body is read +// to build the gateway error, so a misbehaving upstream cannot make the +// gateway buffer an unbounded body. +const maxSystemOneErrorBodyBytes = 64 << 10 + +// sendSystemOne sends the body to one target under its own model name. Only +// targets that serve the route reach it: the handler checks the primary and +// the failover sweep skips ineligible targets. An upstream error status comes +// back as an error the failover policy can judge. +func (s *translatedInferenceService) sendSystemOne( + ctx context.Context, + passthroughProvider core.RoutablePassthrough, + route systemOneRoute, + selector core.ModelSelector, + providerType, providerName string, + headers http.Header, + body []byte, +) (*core.PassthroughResponse, error) { + forwarded, err := rewriteMessagesModel(body, selector.Model) + if err != nil { + return nil, core.NewInvalidRequestError("invalid request body: "+err.Error(), err) + } + resp, err := passthroughProvider.Passthrough(ctx, providerType, &core.PassthroughRequest{ + Method: http.MethodPost, + Endpoint: route.endpoint, + Operation: route.operation, + Model: selector.Model, + Body: io.NopCloser(bytes.NewReader(forwarded)), + Headers: headers.Clone(), + ProviderName: providerName, + }) + if err != nil { + return nil, err + } + if resp == nil || resp.Body == nil { + return nil, core.NewProviderError(providerType, http.StatusBadGateway, "provider returned empty passthrough response", nil) + } + if resp.StatusCode < http.StatusBadRequest { + return resp, nil + } + defer func() { _ = resp.Body.Close() }() + errorBody, err := io.ReadAll(io.LimitReader(resp.Body, maxSystemOneErrorBodyBytes)) + if err != nil { + return nil, core.NewProviderError(providerType, http.StatusBadGateway, "failed to read provider error response", err) + } + return nil, core.ParseProviderError(providerType, resp.StatusCode, errorBody, nil) +} + +// executedSystemOneWorkflow returns a copy of workflow routed to the failover +// target that answered, so usage is priced and audited under the model that +// did the work rather than the primary that failed. +func executedSystemOneWorkflow(workflow *core.Workflow, executed core.ModelSelector, meta gateway.ExecutionMeta) *core.Workflow { + if workflow == nil || workflow.Resolution == nil { + return workflow + } + resolution := *workflow.Resolution + resolution.ResolvedSelector = executed + resolution.ProviderType = strings.TrimSpace(meta.ProviderType) + resolution.ProviderName = strings.TrimSpace(meta.ProviderName) + cloned := *workflow + cloned.ProviderType = resolution.ProviderType + cloned.Resolution = &resolution + return &cloned +} diff --git a/internal/server/systemone_dispatch_test.go b/internal/server/systemone_dispatch_test.go new file mode 100644 index 000000000..1198bf12d --- /dev/null +++ b/internal/server/systemone_dispatch_test.go @@ -0,0 +1,376 @@ +package server + +import ( + "context" + "io" + "net/http" + "slices" + "strings" + "testing" + "time" + + "github.com/goccy/go-json" + "github.com/labstack/echo/v5" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/cache" + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/echotest" + "github.com/enterpilot/gomodel/internal/gateway" + "github.com/enterpilot/gomodel/internal/responsecache" + "github.com/enterpilot/gomodel/internal/usage" +) + +// scriptedSystemOneProvider answers each passthrough by the model the body +// forwards, and records every call as " ". +type scriptedSystemOneProvider struct { + *mockProvider + // statuses maps a forwarded model to the status it answers with; + // 200 by default. + statuses map[string]int + catalog map[string]core.Model + calls []string + // idempotencyKeys records the explicit Idempotency-Key of each call. + idempotencyKeys []string +} + +// newScriptedSystemOneProvider configures the given "/" +// selectors, keyed to their provider types. +func newScriptedSystemOneProvider(models map[string]string) *scriptedSystemOneProvider { + mock := &mockProvider{providerTypes: map[string]string{}, providerNames: map[string]string{}} + for qualified, providerType := range models { + providerName, model, _ := strings.Cut(qualified, "/") + mock.supportedModels = append(mock.supportedModels, model) + mock.providerTypes[qualified] = providerType + mock.providerNames[qualified] = providerName + } + return &scriptedSystemOneProvider{mockProvider: mock, statuses: map[string]int{}} +} + +func (p *scriptedSystemOneProvider) Passthrough(_ context.Context, _ string, req *core.PassthroughRequest) (*core.PassthroughResponse, error) { + raw, err := io.ReadAll(req.Body) + if err != nil { + return nil, err + } + var sent struct { + Model string `json:"model"` + } + if err := json.Unmarshal(raw, &sent); err != nil { + return nil, err + } + p.calls = append(p.calls, req.ProviderName+" "+req.Endpoint+" "+sent.Model) + p.idempotencyKeys = append(p.idempotencyKeys, req.Headers.Get(core.IdempotencyKeyHeader)) + + status := p.statuses[sent.Model] + body := `{"error":{"message":"overloaded"}}` + if status == 0 { + status = http.StatusOK + body = `{"model":"` + sent.Model + `-answered","answers":{},"usage":{"input_tokens":10,"output_tokens":1}}` + } + return &core.PassthroughResponse{ + StatusCode: status, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(body)), + }, nil +} + +func (p *scriptedSystemOneProvider) LookupModel(model string) (*core.Model, bool) { + found, ok := p.catalog[model] + return &found, ok +} + +func (p *scriptedSystemOneProvider) ProviderNamesForType(providerType string) []string { + var names []string + for qualified, candidate := range p.providerTypes { + if name := p.providerNames[qualified]; candidate == providerType && !slices.Contains(names, name) { + names = append(names, name) + } + } + slices.Sort(names) + return names +} + +func systemOneRequest(model string) string { + return `{"model":"` + model + `","state":"I was charged twice.","questions":{"refund":{"type":"noul","instructions":"Refund?"}}}` +} + +// An identical request is answered from the exact cache without reaching the +// provider. +func TestSystemOne_ServesRepeatsFromTheExactCache(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{"kev/kev-latest": "jev"}) + store := cache.NewMapStore() + defer store.Close() + mw := responsecache.NewResponseCacheMiddlewareWithStore(store, time.Hour) + handler := NewHandler(provider, nil, nil, nil) + handler.responseCache = mw + + c, first := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, first.Code, first.Body.String()) + // The cache write is asynchronous; drain it before the repeat. + require.NoError(t, mw.Close()) + + c, second := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, second.Code, second.Body.String()) + + assert.Equal(t, "HIT (exact)", second.Header().Get("X-Cache")) + assert.JSONEq(t, first.Body.String(), second.Body.String()) + assert.Len(t, provider.calls, 1, "the repeat must not reach the provider") +} + +// answeredModelPricing prices only the models it names, as an operator who +// declares a price for the versioned model that answers an alias. +type answeredModelPricing map[string]*core.ModelPricing + +func (r answeredModelPricing) ResolvePricing(model, _ string) *core.ModelPricing { return r[model] } + +// A cache hit is recorded in usage as an exact hit, with the tokens of the +// replayed answer, priced like the live answer: by the model that answered +// when only it carries a price. +func TestSystemOne_RecordsCacheHitsInUsage(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{"kev/kev-latest": "jev"}) + store := cache.NewMapStore() + defer store.Close() + rate := 1_000_000.0 + pricing := answeredModelPricing{"kev-latest-answered": {InputPerMtok: &rate}} + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + mw := responsecache.NewResponseCacheMiddlewareWithStoreAndUsage(store, time.Hour, usageLogger, pricing) + handler := newHandler(provider, nil, usageLogger, pricing, nil, nil, nil, nil) + handler.responseCache = mw + + c, first := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, first.Code, first.Body.String()) + require.NoError(t, mw.Close()) + + c, second := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, "HIT (exact)", second.Header().Get("X-Cache")) + + require.Len(t, usageLogger.entries, 2) + hit := usageLogger.entries[1] + assert.Equal(t, usage.CacheTypeExact, hit.CacheType) + assert.Equal(t, "/v1/systemone", hit.Endpoint) + assert.Equal(t, "jev", hit.Provider) + assert.Equal(t, 10, hit.InputTokens) + assert.Equal(t, 1, hit.OutputTokens) + for _, entry := range usageLogger.entries { + require.NotNil(t, entry.InputCost, "cache type %q", entry.CacheType) + assert.InDelta(t, 10.0, *entry.InputCost, 1e-9) + } +} + +// A different state is a different decision: it misses the cache. +func TestSystemOne_CacheKeyCoversTheState(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{"kev/kev-latest": "jev"}) + store := cache.NewMapStore() + defer store.Close() + mw := responsecache.NewResponseCacheMiddlewareWithStore(store, time.Hour) + handler := NewHandler(provider, nil, nil, nil) + handler.responseCache = mw + + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + require.NoError(t, mw.Close()) + + c, rec = echotest.Post(t, "/v1/systemone", strings.Replace(systemOneRequest("kev-latest"), "twice", "once", 1)) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Empty(t, rec.Header().Get("X-Cache")) + assert.Len(t, provider.calls, 2) +} + +// When the primary fails with an availability error, the request moves to +// the virtual model's next target in its own dialect. A target without the +// System One API is skipped rather than called, and usage and audit carry the +// model that answered. +func TestSystemOne_FailsOverToTheNextSystemOneTarget(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "kev/kev-latest": "jev", + "openai/gpt-5-mini": "openai", + "openrouter/typesafe/jev-1.13": "openrouter", + }) + provider.statuses["kev-latest"] = http.StatusServiceUnavailable + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + handler := newHandler(provider, nil, usageLogger, nil, nil, nil, failoverResolverStub{selectors: []core.ModelSelector{ + {Provider: "openai", Model: "gpt-5-mini"}, + {Provider: "openrouter", Model: "typesafe/jev-1.13"}, + }}, nil) + + entry := &auditlog.LogEntry{Data: &auditlog.LogData{}} + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest("kev/kev-latest"), echotest.WithValue(string(auditlog.LogEntryKey), entry)) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, []string{"kev systemone kev-latest", "openrouter systemone typesafe/jev-1.13"}, provider.calls, + "the chat model must be skipped, not called") + require.NotNil(t, entry.Data.Failover) + assert.Equal(t, "openrouter/typesafe/jev-1.13", entry.Data.Failover.TargetModel) + assert.Equal(t, "openrouter/typesafe/jev-1.13", entry.ResolvedModel) + assert.Equal(t, "openrouter", entry.Provider) + require.Len(t, entry.Data.Attempts, 2, "a skipped target is not an attempt") + assert.Equal(t, http.StatusServiceUnavailable, entry.Data.Attempts[0].StatusCode) + assert.True(t, entry.Data.Attempts[1].Success) + + require.Len(t, usageLogger.entries, 1) + assert.Equal(t, "openrouter", usageLogger.entries[0].Provider) + assert.Equal(t, "typesafe/jev-1.13-answered", usageLogger.entries[0].Model) +} + +// A skipped target does not count against max_attempts, so a chat model ahead +// of a valid target in the chain cannot use up the only failover attempt. The +// client's Idempotency-Key is not forwarded as a header: it reaches the +// primary through the request context, and a failover target's different +// body must not carry it. +func TestSystemOne_FailoverSkipsTargetsWithoutUsingAttempts(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "kev/kev-latest": "jev", + "openai/gpt-5-mini": "openai", + "openrouter/typesafe/jev-1.13": "openrouter", + }) + provider.statuses["kev-latest"] = http.StatusServiceUnavailable + handler := newHandler(provider, nil, nil, nil, nil, nil, failoverResolverStub{selectors: []core.ModelSelector{ + {Provider: "openai", Model: "gpt-5-mini"}, + {Provider: "openrouter", Model: "typesafe/jev-1.13"}, + }}, nil) + handler.failoverPolicy = &gateway.FailoverPolicy{MaxAttempts: 1} + + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest("kev/kev-latest"), echotest.WithHeader(core.IdempotencyKeyHeader, "client-key-1")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, []string{"kev systemone kev-latest", "openrouter systemone typesafe/jev-1.13"}, provider.calls) + assert.Equal(t, []string{"", ""}, provider.idempotencyKeys, "the key must not travel as an explicit header") +} + +// A client error such as a malformed question is not an availability +// problem: it is returned as the upstream reported it, without failover. +func TestSystemOne_DoesNotFailOverOnClientErrors(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "kev/kev-latest": "jev", + "openrouter/typesafe/jev-1.13": "openrouter", + }) + provider.statuses["kev-latest"] = http.StatusUnprocessableEntity + handler := newHandler(provider, nil, nil, nil, nil, nil, failoverResolverStub{selectors: []core.ModelSelector{ + {Provider: "openrouter", Model: "typesafe/jev-1.13"}, + }}, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest("kev/kev-latest")) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusUnprocessableEntity, rec.Code, rec.Body.String()) + assert.Equal(t, []string{"kev systemone kev-latest"}, provider.calls) +} + +// Kev's diagnostic routes are served natively on jev providers and refused +// on OpenRouter, which serves only the evaluation route. +func TestSystemOne_KevDiagnosticRoutes(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "kev/kev-latest": "jev", + "openrouter/typesafe/jev-1.13": "openrouter", + }) + handler := NewHandler(provider, nil, nil, nil) + + for path, serve := range map[string]func(*echo.Context) error{ + "/v1/systemone/permute": handler.SystemOnePermute, + "/v1/systemone/separate": handler.SystemOneSeparate, + } { + t.Run(path, func(t *testing.T) { + provider.calls = nil + c, rec := echotest.Post(t, path, systemOneRequest("kev/kev-latest")) + require.NoError(t, serve(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.Equal(t, []string{"kev " + strings.TrimPrefix(path, "/v1/") + " kev-latest"}, provider.calls) + + c, rec = echotest.Post(t, path, systemOneRequest("openrouter/typesafe/jev-1.13")) + require.NoError(t, serve(c)) + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "this route is served by Kev servers") + }) + } +} + +// TypeSafe lists only its aliases but accepts any versioned ID, so a pinned +// version routes to a jev provider without being declared: named with its +// provider, bare when one jev provider is configured, or through a virtual +// model. A bare name stays unrouted when several jev providers could own it. +func TestSystemOne_RoutesUnlistedPinnedVersions(t *testing.T) { + aliases := systemOneAliasResolver{"pinned": {Provider: "jev", Model: "jev-1.13.0"}} + tests := []struct { + name string + models map[string]string + model string + wantCall string + wantCode int + }{ + {name: "provider-qualified", models: map[string]string{"jev/jev-latest": "jev"}, model: "jev/jev-1.13.0", wantCall: "jev systemone jev-1.13.0"}, + {name: "bare with one jev provider", models: map[string]string{"jev/jev-latest": "jev", "openrouter/typesafe/jev-1.13": "openrouter"}, model: "jev-1.13.0", wantCall: "jev systemone jev-1.13.0"}, + {name: "virtual model", models: map[string]string{"jev/jev-latest": "jev"}, model: "pinned", wantCall: "jev systemone jev-1.13.0"}, + {name: "self-hosted provider name", models: map[string]string{"kev/kev-latest": "jev"}, model: "kev/kev-4b-2026-09", wantCall: "kev systemone kev-4b-2026-09"}, + {name: "bare with two jev providers", models: map[string]string{"jev/jev-latest": "jev", "kev/kev-latest": "jev"}, model: "jev-1.13.0", wantCode: http.StatusNotFound}, + {name: "unknown on OpenRouter", models: map[string]string{"openrouter/typesafe/jev-1.13": "openrouter"}, model: "openrouter/typesafe/jev-9", wantCode: http.StatusNotFound}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + provider := newScriptedSystemOneProvider(tt.models) + handler := newHandler(provider, nil, nil, nil, aliases, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest(tt.model)) + require.NoError(t, handler.SystemOne(c)) + + if tt.wantCode != 0 { + assert.Equal(t, tt.wantCode, rec.Code, rec.Body.String()) + assert.Empty(t, provider.calls) + return + } + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.Equal(t, []string{tt.wantCall}, provider.calls) + }) + } +} + +// A System One model sent to an OpenAI-compatible route is refused by the +// gateway with a pointer at /v1/systemone, from the catalog alone, so the +// caller never sees the upstream's advice to use the upstream's own endpoint. +// Chat models on the same provider are unaffected. +func TestSystemOneModels_OnOpenAIRoutesPointAtSystemOne(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "openrouter/~typesafe/jev-latest": "openrouter", + "openrouter/openai/gpt-4o-mini": "openrouter", + }) + provider.catalog = map[string]core.Model{ + "openrouter/~typesafe/jev-latest": {ID: "~typesafe/jev-latest", Metadata: &core.ModelMetadata{ + Categories: []core.ModelCategory{core.CategoryUtility}, + }}, + "openrouter/openai/gpt-4o-mini": {ID: "openai/gpt-4o-mini", Metadata: &core.ModelMetadata{ + Modes: []string{"chat"}, Categories: []core.ModelCategory{core.CategoryTextGeneration}, + }}, + } + provider.response = &core.ChatResponse{ID: "c1", Object: "chat.completion", Model: "openai/gpt-4o-mini", + Choices: []core.Choice{{Message: core.ResponseMessage{Role: "assistant", Content: "ok"}, FinishReason: "stop"}}} + srv := New(provider, &Config{ModelResolver: systemOneAliasResolver{"jev-latest": {Provider: "openrouter", Model: "~typesafe/jev-latest"}}}) + + requests := map[string]string{ + "/v1/chat/completions": `{"model":"jev-latest","messages":[{"role":"user","content":"hi"}]}`, + "/v1/responses": `{"model":"openrouter/~typesafe/jev-latest","input":"hi"}`, + "/v1/embeddings": `{"model":"openrouter/~typesafe/jev-latest","input":"hi"}`, + "/v1/messages": `{"model":"jev-latest","max_tokens":8,"messages":[{"role":"user","content":"hi"}]}`, + } + for path, body := range requests { + t.Run(path, func(t *testing.T) { + rec := postJSON(t, srv, path, body) + assert.Equal(t, http.StatusBadRequest, rec.Code, rec.Body.String()) + assert.Contains(t, rec.Body.String(), "System One decision model") + assert.Contains(t, rec.Body.String(), "POST /v1/systemone") + }) + } + assert.Zero(t, provider.chatCompletionCalls) + + rec := postJSON(t, srv, "/v1/chat/completions", `{"model":"openrouter/openai/gpt-4o-mini","messages":[{"role":"user","content":"hi"}]}`) + assert.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) +} diff --git a/internal/server/systemone_handler.go b/internal/server/systemone_handler.go new file mode 100644 index 000000000..6c43c262e --- /dev/null +++ b/internal/server/systemone_handler.go @@ -0,0 +1,355 @@ +package server + +import ( + "bytes" + "context" + "fmt" + "log/slog" + "net/http" + "slices" + "strings" + + "github.com/goccy/go-json" + "github.com/labstack/echo/v5" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/gateway" + "github.com/enterpilot/gomodel/internal/plugins" +) + +// systemOneRoute is one System One route the gateway serves natively. +type systemOneRoute struct { + // path is the gateway route; endpoint is the same route as a provider's + // passthrough spells it, without the /v1 prefix. + path string + endpoint string + // operation names the call in provider metrics and logs. + operation string + // kevOnly marks the diagnostic routes only Kev servers implement; the + // hosted API and OpenRouter serve the evaluation route alone. + kevOnly bool +} + +var ( + systemOneEvaluate = systemOneRoute{path: "/v1/systemone", endpoint: "systemone", operation: "systemone"} + systemOnePermute = systemOneRoute{path: "/v1/systemone/permute", endpoint: "systemone/permute", operation: "systemone_permute", kevOnly: true} + systemOneSeparate = systemOneRoute{path: "/v1/systemone/separate", endpoint: "systemone/separate", operation: "systemone_separate", kevOnly: true} +) + +const jevProviderType = "jev" + +// systemOneProviderTypes are the provider types that serve the System One API +// natively: jev (TypeSafe's hosted Jev and self-hosted Kev servers) and +// OpenRouter, which serves Jev and Kev at the same path with the same request +// and answer shapes. Configuring either makes /v1/systemone available. +var systemOneProviderTypes = []string{jevProviderType, "openrouter"} + +// SystemOne handles POST /v1/systemone. +// +// It accepts TypeSafe's System One decision request (a state plus a map of +// typed questions) and forwards it natively: the body reaches the provider +// unchanged apart from the routed model name and guardrail edits to the +// state. Requests are never translated to another API, so a model whose +// provider has no System One API is rejected. +// +// @Summary Evaluate a System One decision request (Jev / Kev) +// @Description Available when a jev or openrouter provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated. +// @Tags systemone +// @Accept json +// @Produce json +// @Security BearerAuth +// @Param request body object true "System One request: model, state, and questions" +// @Success 200 {object} object "System One answers, in the provider's shape" +// @Failure 400 {object} core.OpenAIErrorEnvelope +// @Failure 401 {object} core.OpenAIErrorEnvelope +// @Failure 404 {object} core.OpenAIErrorEnvelope +// @Failure 429 {object} core.OpenAIErrorEnvelope +// @Failure 502 {object} core.OpenAIErrorEnvelope +// @Router /v1/systemone [post] +func (h *Handler) SystemOne(c *echo.Context) error { + return h.translatedInference().serveSystemOne(c, systemOneEvaluate) +} + +// SystemOnePermute handles POST /v1/systemone/permute. +// +// @Summary Run one Choice question with several option orders (Kev) +// @Description A Kev server diagnostic: the request is a System One request, and n_perm (1 to 64, default 6) sets how many option orders run. Only jev providers pointing at a Kev server serve it. +// @Tags systemone +// @Accept json +// @Produce json +// @Security BearerAuth +// @Param request body object true "System One request with one Choice question" +// @Success 200 {object} object "Kev's answer, in the provider's shape" +// @Failure 400 {object} core.OpenAIErrorEnvelope +// @Failure 401 {object} core.OpenAIErrorEnvelope +// @Failure 404 {object} core.OpenAIErrorEnvelope +// @Failure 429 {object} core.OpenAIErrorEnvelope +// @Failure 502 {object} core.OpenAIErrorEnvelope +// @Router /v1/systemone/permute [post] +func (h *Handler) SystemOnePermute(c *echo.Context) error { + return h.translatedInference().serveSystemOne(c, systemOnePermute) +} + +// SystemOneSeparate handles POST /v1/systemone/separate. +// +// @Summary Run each System One question in its own forward pass (Kev) +// @Description A Kev server diagnostic that answers each question separately. Only jev providers pointing at a Kev server serve it. +// @Tags systemone +// @Accept json +// @Produce json +// @Security BearerAuth +// @Param request body object true "System One request: model, state, and questions" +// @Success 200 {object} object "Kev's answer, in the provider's shape" +// @Failure 400 {object} core.OpenAIErrorEnvelope +// @Failure 401 {object} core.OpenAIErrorEnvelope +// @Failure 404 {object} core.OpenAIErrorEnvelope +// @Failure 429 {object} core.OpenAIErrorEnvelope +// @Failure 502 {object} core.OpenAIErrorEnvelope +// @Router /v1/systemone/separate [post] +func (h *Handler) SystemOneSeparate(c *echo.Context) error { + return h.translatedInference().serveSystemOne(c, systemOneSeparate) +} + +// serveSystemOne resolves, guards, and forwards one System One request. +func (s *translatedInferenceService) serveSystemOne(c *echo.Context, route systemOneRoute) error { + if !systemOneAvailable(s.provider) { + return handleError(c, core.NewNotFoundError("POST "+route.path+" is available only when a jev or openrouter provider is configured")) + } + body, err := requestBodyBytes(c) + if err != nil { + return handleError(c, core.NewInvalidRequestError("invalid request body: "+err.Error(), err)) + } + var req core.SystemOneRequest + if err := json.Unmarshal(body, &req); err != nil { + return handleError(c, core.NewInvalidRequestError("invalid request body: "+err.Error(), err)) + } + if strings.TrimSpace(req.Model) == "" { + return handleError(c, core.NewInvalidRequestError("model is required", nil).WithParam("model")) + } + + workflow, err := s.systemOneWorkflow(c, req.Model) + if err != nil { + return handleError(c, err) + } + resolution := workflow.Resolution + if s.modelAuthorizer != nil { + if err := s.modelAuthorizer.ValidateModelAccess(c.Request().Context(), resolution.ResolvedSelector); err != nil { + return handleError(c, err) + } + } + if reason := s.systemOneUnsupportedReason(route, resolution.ResolvedSelector, resolution.ProviderType); reason != "" { + return handleError(c, systemOneUnsupportedModelError(c, route, resolution, reason)) + } + + body, err = s.guardSystemOneState(c, workflow, &req, body) + if err != nil { + return handleError(c, err) + } + return s.dispatchSystemOneWithCache(c, route, workflow, body) +} + +// systemOneAvailable reports whether a provider that serves System One is +// configured. It is checked per request rather than at route registration so +// a provider added at runtime makes the endpoint available without a restart. +func systemOneAvailable(provider core.RoutableProvider) bool { + named, ok := provider.(core.ProviderTypeNameResolver) + if !ok { + return false + } + for _, providerType := range systemOneProviderTypes { + if strings.TrimSpace(named.GetProviderNameForType(providerType)) != "" { + return true + } + } + return false +} + +// systemOneWorkflow resolves the request's model (virtual models first, then +// the catalog) and builds its workflow. A model the catalog does not list may +// still be a pinned version a jev provider accepts; see unlistedJevResolution. +func (s *translatedInferenceService) systemOneWorkflow(c *echo.Context, model string) (*core.Workflow, error) { + if workflow := core.GetWorkflow(c.Request().Context()); workflow != nil && workflow.Resolution != nil { + return workflow, nil + } + requested := core.NewRequestedModelSelector(model, "") + resolution, ok := s.unlistedJevResolution(c.Request().Context(), requested) + if ok { + enrichAuditEntryWithRequestedModel(c, requested) + } else { + var err error + resolution, err = resolveAndStoreRequestModelResolution(c, s.provider, s.modelResolver, nil, model, "") + if err != nil { + return nil, err + } + } + workflow, err := translatedWorkflowForRequest(c, resolution, s.workflowPolicyResolver) + if err != nil { + return nil, err + } + storeWorkflow(c, workflow) + return workflow, nil +} + +// providerNamesByType lists configured provider instances of one type; the +// provider router implements it. +type providerNamesByType interface { + ProviderNamesForType(providerType string) []string +} + +// unlistedJevResolution routes a model the catalog does not list to a jev +// provider. TypeSafe lists only its aliases, yet accepts every versioned ID +// (jev-1.13.0) in the model field, so a pinned version must work without +// being declared first. The provider is the one the model names +// (jev/jev-1.13.0, kev/...), or for a bare name the only jev provider +// configured; virtual models apply first, so one can pin a version too. It +// is checked before the regular resolution, which would refresh the +// provider's model list on every such request; the upstream reports a name +// it rejects. +func (s *translatedInferenceService) unlistedJevResolution(ctx context.Context, requested core.RequestedModelSelector) (*core.RequestModelResolution, bool) { + selector, aliasApplied, err := gateway.ResolveExecutionSelector(ctx, s.provider, s.modelResolver, requested) + if err != nil || selector.Model == "" || s.provider.Supports(selector.QualifiedModel()) { + return nil, false + } + providerName := "" + switch named, _ := s.provider.(core.ProviderNameTypeResolver); { + case selector.Provider == "": + if lister, ok := s.provider.(providerNamesByType); ok { + if names := lister.ProviderNamesForType(jevProviderType); len(names) == 1 { + providerName = names[0] + } + } + case named != nil && named.GetProviderTypeForName(selector.Provider) == jevProviderType: + providerName = selector.Provider + case selector.Provider == jevProviderType: + if byType, ok := s.provider.(core.ProviderTypeNameResolver); ok { + providerName = byType.GetProviderNameForType(jevProviderType) + } + } + if strings.TrimSpace(providerName) == "" { + return nil, false + } + return &core.RequestModelResolution{ + Requested: requested, + ResolvedSelector: core.ModelSelector{Provider: providerName, Model: selector.Model}, + ProviderType: jevProviderType, + ProviderName: providerName, + AliasApplied: aliasApplied, + }, true +} + +// modelCatalog describes single catalog models; the provider router +// implements it. +type modelCatalog interface { + LookupModel(model string) (*core.Model, bool) +} + +// systemOneUnsupportedReason explains why a model cannot answer a request on +// route, or returns "" when it can. The provider must serve the API, a Kev +// diagnostic route needs a jev provider, and since OpenRouter also serves +// chat models, the model must not be catalogued with a generation mode. A +// model the catalog does not describe is given the benefit of the doubt: the +// upstream reports it if it is wrong. +func (s *translatedInferenceService) systemOneUnsupportedReason(route systemOneRoute, selector core.ModelSelector, providerType string) string { + providerType = strings.TrimSpace(providerType) + if !slices.Contains(systemOneProviderTypes, providerType) { + return fmt.Sprintf("is served by provider type %s, which has no System One API", providerType) + } + if route.kevOnly && providerType != jevProviderType { + return fmt.Sprintf("is served by provider type %s, which answers only %s; this route is served by Kev servers", providerType, systemOneEvaluate.path) + } + catalog, ok := s.provider.(modelCatalog) + if !ok { + return "" + } + model, ok := catalog.LookupModel(selector.QualifiedModel()) + if !ok || model == nil || model.Metadata == nil || len(model.Metadata.Modes) == 0 { + return "" + } + return fmt.Sprintf("is a %s model, not a System One model", strings.Join(model.Metadata.Modes, "/")) +} + +// systemOneOnlyModel reports whether the resolved model answers only System +// One requests: served by a System One provider and catalogued as a utility +// model with no generation mode, as jev models and OpenRouter's decision +// models are. The catalog is consulted rather than the provider, so the check +// holds from startup, before any provider has listed its models again. +func systemOneOnlyModel(provider core.RoutableProvider, resolution *core.RequestModelResolution) bool { + if resolution == nil || !slices.Contains(systemOneProviderTypes, strings.TrimSpace(resolution.ProviderType)) { + return false + } + catalog, ok := provider.(modelCatalog) + if !ok { + return false + } + model, ok := catalog.LookupModel(resolution.ResolvedQualifiedModel()) + if !ok || model == nil || model.Metadata == nil || len(model.Metadata.Modes) > 0 { + return false + } + return slices.Contains(model.Metadata.Categories, core.CategoryUtility) +} + +// systemOneOnlyModelError points a chat, Responses, or embeddings request for +// a System One model at /v1/systemone. Without it the caller would see the +// upstream's own advice, which names the upstream's endpoint, not the +// gateway's. +func systemOneOnlyModelError(operation core.Operation, resolution *core.RequestModelResolution) error { + surface := strings.ReplaceAll(string(operation), "_", " ") + return core.NewInvalidRequestError(fmt.Sprintf( + "model %q is a System One decision model and does not support %s; it answers decision requests, which GoModel does not translate: send them to POST %s", + resolution.RequestedQualifiedModel(), surface, systemOneEvaluate.path, + ), nil).WithParam("model") +} + +// systemOneUnsupportedModelError explains a request whose model cannot answer +// System One. It is also logged: a virtual model that sends System One +// traffic to a chat model is an operator mistake the caller cannot fix. +func systemOneUnsupportedModelError(c *echo.Context, route systemOneRoute, resolution *core.RequestModelResolution, reason string) error { + requested := resolution.RequestedQualifiedModel() + resolved := resolution.ResolvedQualifiedModel() + slog.Warn("System One request routed to a model without the System One API", + "request_id", requestIDFromContextOrHeader(c.Request()), + "path", route.path, + "requested_model", requested, + "resolved_model", resolved, + "provider_type", resolution.ProviderType, + "reason", reason, + ) + target := fmt.Sprintf("%q", requested) + if resolved != requested { + target += fmt.Sprintf(" (resolved to %q)", resolved) + } + return core.NewInvalidRequestError(fmt.Sprintf( + "model %s %s; %s forwards requests natively and does not translate them to other APIs, so use a System One model such as a jev model or OpenRouter's typesafe/jev-1.13", + target, reason, route.path, + ), nil).WithParam("model") +} + +// guardSystemOneState runs the workflow's prompt guardrails over the state +// and returns the body carrying their edits. A guardrail that answers the +// request itself blocks it instead: its answer is chat text, and a System +// One caller expects typed answers. +func (s *translatedInferenceService) guardSystemOneState(c *echo.Context, workflow *core.Workflow, req *core.SystemOneRequest, body []byte) ([]byte, error) { + patcher, ok := s.translatedRequestPatcher.(gateway.SystemOneRequestPatcher) + if !ok || !workflow.GuardrailsEnabled() { + return body, nil + } + // Compare against a copy: a patcher may redact the state in place and + // return the same request, and that edit must still reach the body. + original := bytes.Clone(req.State) + patched, err := patcher.PatchSystemOneRequest(c.Request().Context(), req) + s.recordGuardrailOutcomes(c) + if err != nil { + if short := shortCircuitOf(err); short != nil { + return nil, plugins.BlockError(short.Decision, http.StatusBadRequest) + } + return nil, err + } + if patched == nil || bytes.Equal(patched.State, original) { + return body, nil + } + rewritten, err := replaceTopLevelMember(body, "state", patched.State) + if err != nil { + return nil, core.NewInvalidRequestError("invalid request body: "+err.Error(), err) + } + return rewritten, nil +} diff --git a/internal/server/systemone_handler_test.go b/internal/server/systemone_handler_test.go new file mode 100644 index 000000000..45e82b6cc --- /dev/null +++ b/internal/server/systemone_handler_test.go @@ -0,0 +1,367 @@ +package server + +import ( + "context" + "io" + "net/http" + "strings" + "testing" + + "github.com/goccy/go-json" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/echotest" + "github.com/enterpilot/gomodel/internal/usage" +) + +const ( + systemOneQuestions = `{"refund":{"type":"noul","instructions":"Is the customer asking for money back?"}}` + systemOneAnswer = `{"model":"kev-1.0","answers":{"refund":{"type":"noul","noul":0.98}},"usage":{"input_tokens":275,"output_tokens":20}}` +) + +func systemOneBody(model string) string { + return `{"model":"` + model + `","state":"I was charged twice for card 4111.","questions":` + systemOneQuestions + `}` +} + +// systemOneAliasResolver maps virtual model names to concrete selectors. +type systemOneAliasResolver map[string]core.ModelSelector + +func (r systemOneAliasResolver) ResolveModel(requested core.RequestedModelSelector) (core.ModelSelector, bool, error) { + if selector, ok := r[requested.RequestedQualifiedModel()]; ok { + return selector, true, nil + } + selector, err := requested.Normalize() + return selector, false, err +} + +// newSystemOneProvider configures a local Kev server (type jev, named kev), +// OpenRouter, and an OpenAI chat model, answering every passthrough with body. +func newSystemOneProvider(body string) *mockProvider { + return &mockProvider{ + supportedModels: []string{"kev-latest", "typesafe/jev-1.13", "gpt-5-mini"}, + providerTypes: map[string]string{ + "kev/kev-latest": "jev", + "openrouter/typesafe/jev-1.13": "openrouter", + "openai/gpt-5-mini": "openai", + }, + providerNames: map[string]string{ + "kev/kev-latest": "kev", + "openrouter/typesafe/jev-1.13": "openrouter", + "openai/gpt-5-mini": "openai", + }, + passthroughResponse: &core.PassthroughResponse{ + StatusCode: http.StatusOK, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(body)), + }, + } +} + +var systemOneAliases = systemOneAliasResolver{ + "decider": {Provider: "kev", Model: "kev-latest"}, + "chatty": {Provider: "openai", Model: "gpt-5-mini"}, +} + +func forwardedSystemOneBody(t *testing.T, provider *mockProvider) map[string]any { + t.Helper() + require.NotNil(t, provider.lastPassthroughReq) + raw, err := io.ReadAll(provider.lastPassthroughReq.Body) + require.NoError(t, err) + var body map[string]any + require.NoError(t, json.Unmarshal(raw, &body)) + + return body +} + +// Without a provider that serves System One the endpoint does not exist, +// whatever the model. +func TestSystemOne_UnavailableWithoutSystemOneProvider(t *testing.T) { + provider := &mockProvider{ + supportedModels: []string{"gpt-5-mini"}, + providerTypes: map[string]string{"openai/gpt-5-mini": "openai"}, + providerNames: map[string]string{"openai/gpt-5-mini": "openai"}, + } + handler := NewHandler(provider, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("gpt-5-mini")) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusNotFound, rec.Code) + assert.Contains(t, rec.Body.String(), "jev or openrouter provider") + assert.Nil(t, provider.lastPassthroughReq) +} + +// A virtual model resolves to its System One target, and the body reaches the +// provider unchanged except for the concrete model name. +func TestSystemOne_ForwardsNativelyAndRecordsUsage(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + handler := newHandlerWithAuthorizer(provider, nil, usageLogger, nil, systemOneAliases, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("decider")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.JSONEq(t, systemOneAnswer, rec.Body.String()) + + assert.Equal(t, "jev", provider.lastPassthroughProvider) + assert.Equal(t, "systemone", provider.lastPassthroughReq.Endpoint) + assert.Equal(t, "kev", provider.lastPassthroughReq.ProviderName) + body := forwardedSystemOneBody(t, provider) + assert.Equal(t, "kev-latest", body["model"]) + assert.Equal(t, "I was charged twice for card 4111.", body["state"]) + questions, err := json.Marshal(body["questions"]) + require.NoError(t, err) + assert.JSONEq(t, systemOneQuestions, string(questions)) + + require.Len(t, usageLogger.entries, 1) + entry := usageLogger.entries[0] + assert.Equal(t, 275, entry.InputTokens) + assert.Equal(t, 20, entry.OutputTokens) + assert.Equal(t, "/v1/systemone", entry.Endpoint) + assert.Equal(t, "jev", entry.Provider) + assert.Equal(t, "kev", entry.ProviderName) + assert.Equal(t, "kev-1.0", entry.Model, "usage is recorded under the model that answered") +} + +// OpenRouter serves Jev at the same path, so it is a native target too; its +// reported cost is kept with the usage entry. +func TestSystemOne_ForwardsOpenRouterJevNatively(t *testing.T) { + answer := `{"id":"gen-dec-1","model":"typesafe/jev-1.13-20260917","answers":{},"usage":{"input_tokens":275,"output_tokens":20,"cost":0.00003}}` + provider := newSystemOneProvider(answer) + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + handler := newHandlerWithAuthorizer(provider, nil, usageLogger, nil, nil, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("openrouter/typesafe/jev-1.13")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, "openrouter", provider.lastPassthroughProvider) + assert.Equal(t, "typesafe/jev-1.13", forwardedSystemOneBody(t, provider)["model"]) + require.Len(t, usageLogger.entries, 1) + assert.InDelta(t, 0.00003, usageLogger.entries[0].RawData["cost"], 1e-12) +} + +// OpenRouter serves System One natively, so it enables the endpoint on its own. +// Its catalog names Jev "~typesafe/jev-latest"; a virtual model gives SDK +// callers the "jev-latest" name they send by default. +func TestSystemOne_WorksWithOpenRouterAlone(t *testing.T) { + answer := `{"id":"gen-dec-1","model":"typesafe/jev-1.13-20260917","answers":{},"usage":{"input_tokens":10,"output_tokens":1}}` + for _, model := range []string{"openrouter/~typesafe/jev-latest", "jev-latest"} { + t.Run(model, func(t *testing.T) { + provider := &mockProvider{ + supportedModels: []string{"~typesafe/jev-latest"}, + providerTypes: map[string]string{"openrouter/~typesafe/jev-latest": "openrouter"}, + providerNames: map[string]string{"openrouter/~typesafe/jev-latest": "openrouter"}, + passthroughResponse: &core.PassthroughResponse{ + StatusCode: http.StatusOK, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(answer)), + }, + } + aliases := systemOneAliasResolver{"jev-latest": {Provider: "openrouter", Model: "~typesafe/jev-latest"}} + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, aliases, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody(model)) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, "openrouter", provider.lastPassthroughProvider) + assert.Equal(t, "systemone", provider.lastPassthroughReq.Endpoint) + assert.Equal(t, "~typesafe/jev-latest", forwardedSystemOneBody(t, provider)["model"]) + }) + } +} + +// The endpoint never translates: a model on a provider without the System One +// API is rejected with an explanation, whether named directly or through a +// virtual model. +func TestSystemOne_RejectsModelsWithoutSystemOneAPI(t *testing.T) { + for _, model := range []string{"openai/gpt-5-mini", "chatty"} { + t.Run(model, func(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, systemOneAliases, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody(model)) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "no System One API") + assert.Contains(t, rec.Body.String(), "does not translate") + assert.Nil(t, provider.lastPassthroughReq) + }) + } +} + +// catalogProvider adds the router's single-model catalog lookup to the mock. +type catalogProvider struct { + *mockProvider + models map[string]core.Model +} + +func (p catalogProvider) LookupModel(model string) (*core.Model, bool) { + found, ok := p.models[model] + return &found, ok +} + +// OpenRouter serves chat and decision models from one provider, so the model +// itself must be a System One model: one catalogued with a generation mode is +// rejected, while a decision model (a utility model with no mode) is forwarded. +func TestSystemOne_RejectsOpenRouterChatModels(t *testing.T) { + provider := catalogProvider{ + mockProvider: &mockProvider{ + supportedModels: []string{"typesafe/jev-1.13", "openai/gpt-4o-mini"}, + providerTypes: map[string]string{ + "openrouter/typesafe/jev-1.13": "openrouter", + "openrouter/openai/gpt-4o-mini": "openrouter", + }, + providerNames: map[string]string{ + "openrouter/typesafe/jev-1.13": "openrouter", + "openrouter/openai/gpt-4o-mini": "openrouter", + }, + passthroughResponse: &core.PassthroughResponse{ + StatusCode: http.StatusOK, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(systemOneAnswer)), + }, + }, + models: map[string]core.Model{ + "openrouter/typesafe/jev-1.13": {ID: "typesafe/jev-1.13", Metadata: &core.ModelMetadata{ + Categories: []core.ModelCategory{core.CategoryUtility}, + }}, + "openrouter/openai/gpt-4o-mini": {ID: "openai/gpt-4o-mini", Metadata: &core.ModelMetadata{ + Modes: []string{"chat"}, Categories: []core.ModelCategory{core.CategoryTextGeneration}, + }}, + }, + } + handler := NewHandler(provider, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("openrouter/openai/gpt-4o-mini")) + require.NoError(t, handler.SystemOne(c)) + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "is a chat model, not a System One model") + assert.Contains(t, rec.Body.String(), "does not translate") + assert.Nil(t, provider.lastPassthroughReq) + + c, rec = echotest.Post(t, "/v1/systemone", systemOneBody("openrouter/typesafe/jev-1.13")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.Equal(t, "openrouter", provider.lastPassthroughProvider) +} + +func TestSystemOne_RequiresModel(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := NewHandler(provider, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", `{"state":"hi","questions":{}}`) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "model is required") +} + +// stateRedactingPatcher stands in for an anonymizing guardrail. +type stateRedactingPatcher struct{} + +func (stateRedactingPatcher) PatchChatRequest(_ context.Context, req *core.ChatRequest) (*core.ChatRequest, error) { + return req, nil +} + +func (stateRedactingPatcher) PatchResponsesRequest(_ context.Context, req *core.ResponsesRequest) (*core.ResponsesRequest, error) { + return req, nil +} + +func (stateRedactingPatcher) PatchSystemOneRequest(_ context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) { + patched := *req + patched.State = json.RawMessage(strings.ReplaceAll(string(req.State), "4111", "[card]")) + return &patched, nil +} + +// inPlaceRedactingPatcher edits the request it was given and returns it; the +// patcher contract allows that, and the edit must still reach the provider. +type inPlaceRedactingPatcher struct{ stateRedactingPatcher } + +func (inPlaceRedactingPatcher) PatchSystemOneRequest(_ context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) { + req.State = json.RawMessage(strings.ReplaceAll(string(req.State), "4111", "[card]")) + return req, nil +} + +// Guardrails see the state and their edits reach the provider, whether the +// patcher returns a copy or edits in place; the rest of the body is untouched. +func TestSystemOne_GuardrailsEditState(t *testing.T) { + patchers := map[string]TranslatedRequestPatcher{ + "copy": stateRedactingPatcher{}, + "in place": inPlaceRedactingPatcher{}, + } + for name, patcher := range patchers { + t.Run(name, func(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, systemOneAliases, nil, nil, nil, patcher) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("decider")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + body := forwardedSystemOneBody(t, provider) + assert.Equal(t, "I was charged twice for card [card].", body["state"]) + assert.Equal(t, "kev-latest", body["model"]) + }) + } +} + +// Through the full middleware stack an unavailable endpoint answers 404 +// before the model is resolved, so a missing or unknown model is not +// reported for a route that is not there. +func TestSystemOne_UnavailableBeforeModelResolution(t *testing.T) { + provider := &mockProvider{ + supportedModels: []string{"gpt-5-mini"}, + providerTypes: map[string]string{"openai/gpt-5-mini": "openai"}, + providerNames: map[string]string{"openai/gpt-5-mini": "openai"}, + } + srv := New(provider, &Config{}) + + for _, body := range []string{`{"state":"hi","questions":{}}`, systemOneBody("no-such-model")} { + rec := postJSON(t, srv, "/v1/systemone", body) + assert.Equal(t, http.StatusNotFound, rec.Code, rec.Body.String()) + assert.Contains(t, rec.Body.String(), "jev or openrouter provider") + } +} + +// Through the full middleware stack a System One call is audited under its +// own path with the requested and resolved routes, and its usage recorded. +func TestSystemOne_AuditsAndRecordsUsageThroughServer(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + auditLogger := &capturingAuditLogger{config: auditlog.Config{Enabled: true, LogBodies: true}} + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + srv := New(provider, &Config{ + AuditLogger: auditLogger, + UsageLogger: usageLogger, + ModelResolver: systemOneAliases, + }) + + rec := postJSON(t, srv, "/v1/systemone", systemOneBody("decider")) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + require.Len(t, auditLogger.entries, 1) + entry := auditLogger.entries[0] + assert.Equal(t, "/v1/systemone", entry.Path) + assert.Equal(t, http.StatusOK, entry.StatusCode) + assert.Equal(t, "decider", entry.RequestedModel) + assert.Equal(t, "kev/kev-latest", entry.ResolvedModel) + assert.True(t, entry.AliasUsed) + assert.Equal(t, "jev", entry.Provider) + assert.Equal(t, "kev", entry.ProviderName) + require.NotNil(t, entry.Data) + requestBody, err := json.Marshal(entry.Data.RequestBody) + require.NoError(t, err) + assert.Contains(t, string(requestBody), "charged twice") + responseBody, err := json.Marshal(entry.Data.ResponseBody) + require.NoError(t, err) + assert.Contains(t, string(responseBody), "noul") + + require.Len(t, usageLogger.entries, 1) + assert.Equal(t, "/v1/systemone", usageLogger.entries[0].Endpoint) + assert.Equal(t, entry.RequestID, usageLogger.entries[0].RequestID) +} diff --git a/internal/testconventions/assertions_test.go b/internal/testconventions/assertions_test.go index e28f4deea..d30b2793f 100644 --- a/internal/testconventions/assertions_test.go +++ b/internal/testconventions/assertions_test.go @@ -21,6 +21,7 @@ import ( var skippedDirs = map[string]bool{ ".git": true, ".claude": true, // local agent worktrees, gitignored + ".worktrees": true, // local git worktrees, gitignored ".cache": true, "node_modules": true, "third_party": true, diff --git a/internal/usage/extractor.go b/internal/usage/extractor.go index 46ed07b3d..732a45b5d 100644 --- a/internal/usage/extractor.go +++ b/internal/usage/extractor.go @@ -223,6 +223,11 @@ func ExtractFromSSEUsage( requestID, model, provider, endpoint string, pricing ...*core.ModelPricing, ) *UsageEntry { + // Anthropic-style usage (and System One answers) report input and output + // tokens without a total. + if totalTokens == 0 { + totalTokens = inputTokens + outputTokens + } entry := &UsageEntry{ ID: uuid.New().String(), RequestID: requestID, @@ -281,6 +286,9 @@ func ExtractFromCachedResponseBody( if entry == nil { entry = extractFromCachedSSEBody(body, requestID, model, provider, endpoint, pricing...) } + if entry == nil { + entry = extractFromCachedJSONBody(body, requestID, model, provider, endpoint, pricing...) + } if entry == nil { entry = &UsageEntry{ @@ -346,6 +354,31 @@ func extractFromCachedSSEBody( return observer.cachedEntry } +// extractFromCachedJSONBody reads usage from a cached JSON body of an +// endpoint without a typed response, such as a native /v1/systemone answer, +// the same way a live passthrough response is read. +func extractFromCachedJSONBody( + body []byte, + requestID, model, provider, endpoint string, + pricing ...*core.ModelPricing, +) *UsageEntry { + var payload map[string]any + if err := json.Unmarshal(body, &payload); err != nil || payload == nil { + return nil + } + observer := &StreamUsageObserver{ + model: strings.TrimSpace(model), + provider: strings.TrimSpace(provider), + requestID: strings.TrimSpace(requestID), + endpoint: endpoint, + } + if len(pricing) > 0 && pricing[0] != nil { + observer.pricingResolver = staticPricingResolver{pricing: pricing[0]} + } + observer.OnJSONEvent(payload) + return observer.cachedEntry +} + func normalizeCachedResponseEndpoint(endpoint string) string { normalized := strings.TrimSpace(endpoint) if normalized == "" { diff --git a/internal/usage/extractor_test.go b/internal/usage/extractor_test.go index 86bfed117..c781d0185 100644 --- a/internal/usage/extractor_test.go +++ b/internal/usage/extractor_test.go @@ -436,6 +436,19 @@ func TestExtractFromSSEUsage(t *testing.T) { assert.Equal(t, 25, entry.RawData["cached_tokens"]) } +func TestExtractFromSSEUsageDerivesMissingTotal(t *testing.T) { + // Anthropic-style usage and System One answers carry no total_tokens. + entry := ExtractFromSSEUsage( + "", + 27, 9, 0, + nil, + "req-systemone", "jev-1.13.0", "jev", "/v1/systemone", + ) + + require.NotNil(t, entry) + assert.Equal(t, 36, entry.TotalTokens) +} + func TestExtractFromSSEUsageEmptyRawData(t *testing.T) { entry := ExtractFromSSEUsage( "chatcmpl-789", @@ -493,6 +506,21 @@ func TestExtractFromCachedResponseBody(t *testing.T) { require.Equal(t, 10, entry.TotalTokens) }) + // A native endpoint without a typed response (System One) is read like a + // live passthrough answer, so a cache hit keeps its token counts. + t.Run("reads usage from an untyped JSON body", func(t *testing.T) { + body := []byte(`{"model":"jev-1.13.0","answers":{"refund":{"type":"noul","noul":0.98}},"usage":{"input_tokens":275,"output_tokens":20}}`) + + entry := ExtractFromCachedResponseBody(body, "req-systemone", "jev-latest", "jev", "/v1/systemone", CacheTypeExact) + require.NotNil(t, entry) + require.Equal(t, CacheTypeExact, entry.CacheType) + require.Equal(t, "/v1/systemone", entry.Endpoint) + require.Equal(t, "jev", entry.Provider) + require.Equal(t, 275, entry.InputTokens) + require.Equal(t, 20, entry.OutputTokens) + require.Equal(t, 295, entry.TotalTokens) + }) + t.Run("falls back to synthetic entry when body cannot be parsed", func(t *testing.T) { entry := ExtractFromCachedResponseBody([]byte("{"), "req-cache-fallback", "gpt-4o", "openai", "/v1/chat/completions", CacheTypeExact) require.NotNil(t, entry) diff --git a/internal/usage/pricing.go b/internal/usage/pricing.go index db56717d4..d6c1502d2 100644 --- a/internal/usage/pricing.go +++ b/internal/usage/pricing.go @@ -1,6 +1,10 @@ package usage -import "github.com/enterpilot/gomodel/internal/core" +import ( + "strings" + + "github.com/enterpilot/gomodel/internal/core" +) // PricingResolver resolves pricing metadata for a given model and provider type. // Implementations should check the registry first and fall back to a reverse-index @@ -8,3 +12,50 @@ import "github.com/enterpilot/gomodel/internal/core" type PricingResolver interface { ResolvePricing(model, providerType string) *core.ModelPricing } + +// ModelPricingChecker is implemented by pricing resolvers that can tell +// pricing declared for exactly one model (catalog pricing or a model-scoped +// override) from a broad provider-wide or global rule. +type ModelPricingChecker interface { + HasModelPricing(model, providerType string) bool +} + +// ResolveServedModelPricing prices a response routed to one model and +// answered by another, such as the alias jev-latest answered by jev-1.13.0. +// Pricing declared for exactly the routed model wins, then pricing declared +// for exactly the answering model, then any broader rule matching the routed +// model, then the answering model. answered is called at most once, and only +// when the routed model has no exact pricing. +func ResolveServedModelPricing(resolver PricingResolver, routed, providerType string, answered func() string) *core.ModelPricing { + if resolver == nil { + return nil + } + routed = strings.TrimSpace(routed) + checker, canCheck := resolver.(ModelPricingChecker) + if canCheck && checker.HasModelPricing(routed, providerType) { + return resolver.ResolvePricing(routed, providerType) + } + + answeredModel, resolved := "", false + answeredOnce := func() string { + if !resolved && answered != nil { + if model := strings.TrimSpace(answered()); model != routed { + answeredModel = model + } + } + resolved = true + return answeredModel + } + if canCheck { + if model := answeredOnce(); model != "" && checker.HasModelPricing(model, providerType) { + return resolver.ResolvePricing(model, providerType) + } + } + if pricing := resolver.ResolvePricing(routed, providerType); pricing != nil { + return pricing + } + if model := answeredOnce(); model != "" { + return resolver.ResolvePricing(model, providerType) + } + return nil +} diff --git a/internal/usage/pricing_test.go b/internal/usage/pricing_test.go new file mode 100644 index 000000000..d9bb15f12 --- /dev/null +++ b/internal/usage/pricing_test.go @@ -0,0 +1,83 @@ +package usage + +import ( + "testing" + + "github.com/stretchr/testify/assert" + + "github.com/enterpilot/gomodel/internal/core" +) + +// checkedPricingResolver prices models from exact entries, falling back to a +// broad (provider-wide) rule, and reports which prices are exact. +type checkedPricingResolver struct { + exact map[string]*core.ModelPricing + broad *core.ModelPricing +} + +func (r checkedPricingResolver) ResolvePricing(model, _ string) *core.ModelPricing { + if pricing, ok := r.exact[model]; ok { + return pricing + } + return r.broad +} + +func (r checkedPricingResolver) HasModelPricing(model, _ string) bool { + _, ok := r.exact[model] + return ok +} + +func TestResolveServedModelPricing(t *testing.T) { + rate := func(v float64) *core.ModelPricing { return &core.ModelPricing{InputPerMtok: &v} } + routedExact, answeredExact, broad := rate(1), rate(2), rate(3) + + tests := []struct { + name string + resolver PricingResolver + want *core.ModelPricing + }{ + { + name: "exact routed price wins", + resolver: checkedPricingResolver{exact: map[string]*core.ModelPricing{"jev-latest": routedExact, "jev-1.13.0": answeredExact}, broad: broad}, + want: routedExact, + }, + { + name: "exact answering price beats a broad routed rule", + resolver: checkedPricingResolver{exact: map[string]*core.ModelPricing{"jev-1.13.0": answeredExact}, broad: broad}, + want: answeredExact, + }, + { + name: "broad rule when neither is exact", + resolver: checkedPricingResolver{broad: broad}, + want: broad, + }, + { + name: "resolver without exact checks tries routed then answering", + resolver: mapPricingResolver{"jev-1.13.0/jev": answeredExact}, + want: answeredExact, + }, + { + name: "no resolver", + want: nil, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := ResolveServedModelPricing(tt.resolver, "jev-latest", "jev", func() string { return "jev-1.13.0" }) + assert.Same(t, tt.want, got) + }) + } +} + +// The answering model is only looked up when the routed model has no exact +// price, so a cache hit does not decode its body for the common case. +func TestResolveServedModelPricingSkipsAnsweredLookupForExactRoutedPrice(t *testing.T) { + v := 1.0 + resolver := checkedPricingResolver{exact: map[string]*core.ModelPricing{"gpt-4o": {InputPerMtok: &v}}} + calls := 0 + ResolveServedModelPricing(resolver, "gpt-4o", "openai", func() string { + calls++ + return "gpt-4o-2024-08-06" + }) + assert.Zero(t, calls) +} diff --git a/internal/usage/stream_observer.go b/internal/usage/stream_observer.go index 927750078..ae8e99c65 100644 --- a/internal/usage/stream_observer.go +++ b/internal/usage/stream_observer.go @@ -169,10 +169,8 @@ func (o *StreamUsageObserver) mergeWithCachedEntry(entry *UsageEntry) *UsageEntr entry.TotalTokens = entry.InputTokens + entry.OutputTokens } var pricingArgs []*core.ModelPricing - if o.pricingResolver != nil { - if p := o.pricingResolver.ResolvePricing(o.pricingModel(entry.Model), o.pricingProvider()); p != nil { - pricingArgs = append(pricingArgs, p) - } + if p := o.resolvePricing(entry.Model); p != nil { + pricingArgs = append(pricingArgs, p) } applyUsageCosts(entry, o.provider, o.endpoint, pricingArgs...) return entry @@ -267,10 +265,8 @@ func (o *StreamUsageObserver) extractUsageFromEvent(chunk map[string]any) *Usage } var pricingArgs []*core.ModelPricing - if o.pricingResolver != nil { - if p := o.pricingResolver.ResolvePricing(o.pricingModel(model), o.pricingProvider()); p != nil { - pricingArgs = append(pricingArgs, p) - } + if p := o.resolvePricing(model); p != nil { + pricingArgs = append(pricingArgs, p) } entry := ExtractFromSSEUsage( @@ -304,6 +300,16 @@ func (o *StreamUsageObserver) pricingModel(responseModel string) string { return strings.TrimSpace(responseModel) } +// resolvePricing prices the routed model or, when only it carries pricing, +// the model that answered; see ResolveServedModelPricing. +func (o *StreamUsageObserver) resolvePricing(responseModel string) *core.ModelPricing { + if o == nil { + return nil + } + return ResolveServedModelPricing(o.pricingResolver, o.pricingModel(responseModel), o.pricingProvider(), + func() string { return responseModel }) +} + func (o *StreamUsageObserver) pricingProvider() string { if o == nil { return "" diff --git a/internal/usage/stream_observer_test.go b/internal/usage/stream_observer_test.go index 576ebdbd2..79c27c896 100644 --- a/internal/usage/stream_observer_test.go +++ b/internal/usage/stream_observer_test.go @@ -617,3 +617,38 @@ func TestStreamUsageObserverAnthropicNativeEvents(t *testing.T) { assert.Equal(t, 100, entry.RawData["cache_creation_input_tokens"]) assert.Equal(t, 200, entry.RawData["cache_read_input_tokens"]) } + +// A routed alias (jev-latest) is answered by a versioned model (jev-1.13.0); +// the routed model's price wins, and the answered model's price applies when +// only it is declared. +func TestStreamUsageObserverPricesAnsweredModelWhenRoutedHasNone(t *testing.T) { + routedRate, answeredRate := 10.0, 42.0 + routed := &core.ModelPricing{InputPerMtok: &routedRate} + answered := &core.ModelPricing{InputPerMtok: &answeredRate} + + tests := []struct { + name string + resolver mapPricingResolver + wantInput float64 + }{ + {name: "routed model priced", resolver: mapPricingResolver{"jev-latest/jev": routed, "jev-1.13.0/jev": answered}, wantInput: 10}, + {name: "only answered model priced", resolver: mapPricingResolver{"jev-1.13.0/jev": answered}, wantInput: 42}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + logger := &trackingLogger{enabled: true} + observer := NewStreamUsageObserver(logger, "jev-latest", "jev", "req-s1", "/v1/systemone", tt.resolver) + observer.OnJSONEvent(map[string]any{ + "model": "jev-1.13.0", + "usage": map[string]any{"input_tokens": float64(1_000_000), "output_tokens": float64(0)}, + }) + observer.OnStreamClose() + + entries := logger.getEntries() + require.Len(t, entries, 1) + require.NotNil(t, entries[0].InputCost) + assert.InDelta(t, tt.wantInput, *entries[0].InputCost, 1e-9) + assert.Equal(t, "jev-1.13.0", entries[0].Model) + }) + } +} diff --git a/internal/virtualmodels/chain.go b/internal/virtualmodels/chain.go index 3f2c461a9..e3c3b8e2e 100644 --- a/internal/virtualmodels/chain.go +++ b/internal/virtualmodels/chain.go @@ -66,7 +66,7 @@ func (s *snapshot) viableTargets(entry *redirectEntry, catalog Catalog) []resolv func (s *snapshot) viable(owner *redirectEntry, target resolvedTarget, catalog Catalog) bool { inner, ok := s.chained(owner.vm.Source, target) if !ok { - return catalog.ModelAvailable(target.qualified) + return modelServable(catalog, target.qualified) } if !inner.vm.Enabled { return false @@ -100,7 +100,7 @@ func (s *snapshot) leafTargets(entry *redirectEntry, catalog Catalog) []resolved func (s *snapshot) leaves(owner *redirectEntry, target resolvedTarget, catalog Catalog) []resolvedTarget { inner, ok := s.chained(owner.vm.Source, target) if !ok { - if catalog.ModelAvailable(target.qualified) { + if modelServable(catalog, target.qualified) { return []resolvedTarget{target} } return nil diff --git a/internal/virtualmodels/service.go b/internal/virtualmodels/service.go index c0a13af99..881f16e29 100644 --- a/internal/virtualmodels/service.go +++ b/internal/virtualmodels/service.go @@ -679,7 +679,7 @@ func (s *Service) firstUnsupportedTarget(current *snapshot, vm VirtualModel) (st if _, chained := current.chained(vm.Source, candidate); chained { continue } - if !s.catalog.Supports(qualified) { + if !s.catalog.Supports(qualified) && !modelServable(s.catalog, qualified) { return qualified, true } } diff --git a/internal/virtualmodels/types.go b/internal/virtualmodels/types.go index 82f7cfaa5..eb07bf91a 100644 --- a/internal/virtualmodels/types.go +++ b/internal/virtualmodels/types.go @@ -254,3 +254,20 @@ type Catalog interface { LookupModel(model string) (*core.Model, bool) ProviderNames() []string } + +// unlistedModelCatalog is implemented by catalogs whose providers serve +// provider-qualified model IDs they do not list, such as a jev provider's +// pinned versions (jev/jev-1.13.0). +type unlistedModelCatalog interface { + AcceptsUnlistedModel(model string) bool +} + +// modelServable reports whether a concrete target can serve a request now: +// listed and available, or unlisted on a provider that accepts such IDs. +func modelServable(catalog Catalog, model string) bool { + if catalog.ModelAvailable(model) { + return true + } + unlisted, ok := catalog.(unlistedModelCatalog) + return ok && unlisted.AcceptsUnlistedModel(model) +} diff --git a/internal/virtualmodels/unlisted_target_test.go b/internal/virtualmodels/unlisted_target_test.go new file mode 100644 index 000000000..8c1211855 --- /dev/null +++ b/internal/virtualmodels/unlisted_target_test.go @@ -0,0 +1,52 @@ +package virtualmodels + +import ( + "context" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/core" +) + +// unlistedCatalog is a fakeCatalog whose named providers serve IDs they do +// not list, as a jev provider serves pinned versions. +type unlistedCatalog struct { + fakeCatalog + acceptsUnlisted map[string]bool +} + +func (c unlistedCatalog) AcceptsUnlistedModel(model string) bool { + provider, _, ok := strings.Cut(model, "/") + return ok && c.acceptsUnlisted[provider] +} + +// A redirect may pin a version its provider accepts without listing it; the +// same target on a provider that lists everything it serves is still refused. +func TestService_RedirectToUnlistedModelOfAcceptingProvider(t *testing.T) { + t.Parallel() + catalog := unlistedCatalog{ + providers: []string{"openai", "jev"}, + supported: map[string]core.Model{ + "openai/gpt-4o": {ID: "openai/gpt-4o"}, + "jev/jev-latest": {ID: "jev/jev-latest"}, + }, + acceptsUnlisted: map[string]bool{"jev": true}, + } + svc, err := NewService(newSQLVMStore(t), catalog, true) + require.NoError(t, err) + ctx := context.Background() + + err = svc.Upsert(ctx, VirtualModel{Source: "pinned", Targets: []Target{{Model: "jev/jev-1.13.0"}}, Enabled: true}) + require.NoError(t, err) + selector, changed, err := svc.ResolveModel(core.NewRequestedModelSelector("pinned", "")) + require.NoError(t, err) + assert.True(t, changed) + assert.Equal(t, "jev/jev-1.13.0", selector.QualifiedModel()) + + err = svc.Upsert(ctx, VirtualModel{Source: "missing", Targets: []Target{{Model: "openai/gpt-9"}}, Enabled: true}) + require.Error(t, err) + assert.Contains(t, err.Error(), "target model not found: openai/gpt-9") +} diff --git a/internal/workflows/service.go b/internal/workflows/service.go index f6b1181b8..01d710519 100644 --- a/internal/workflows/service.go +++ b/internal/workflows/service.go @@ -277,6 +277,8 @@ func (s *Service) Deactivate(ctx context.Context, id string) error { } // GetView returns one workflow version view, including inactive historical versions. +// GetView returns one workflow version by ID. A version that no longer +// compiles is returned with CompileError set rather than as an error. func (s *Service) GetView(ctx context.Context, id string) (View, error) { if s == nil { return View{}, fmt.Errorf("workflow service is required") @@ -293,7 +295,13 @@ func (s *Service) GetView(ctx context.Context, id string) (View, error) { return View{}, ErrNotFound } - return s.viewForVersion(*version) + view, err := s.viewForVersion(*version) + if err != nil { + // Historical versions may reference guardrails that no longer exist; + // still show them, flagged like ListViews does. + return viewWithError(*version, err), nil + } + return view, nil } // ListViews returns the active workflows together with their effective diff --git a/internal/workflows/service_test.go b/internal/workflows/service_test.go index 8c2ac05be..c818cb8b4 100644 --- a/internal/workflows/service_test.go +++ b/internal/workflows/service_test.go @@ -14,6 +14,7 @@ import ( "github.com/enterpilot/gomodel/internal/plugins" "github.com/enterpilot/gomodel/internal/plugins/builtin" "github.com/enterpilot/gomodel/pluginapi" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -894,6 +895,31 @@ func TestServiceListViews_AnnotatesCompileFailuresPerRow(t *testing.T) { require.Equal(t, "openai", views[1].ScopeDisplay) } +func TestServiceGetView_ReturnsCompileFailureAsView(t *testing.T) { + store := &staticStore{ + versions: []Version{{ + ID: "provider-v1", + Scope: Scope{Provider: "openai"}, + ScopeKey: "provider:openai", + Version: 1, + Name: "stale-guardrail", + Payload: Payload{SchemaVersion: 1, Features: FeatureFlags{Audit: true}}, + }}, + } + service, err := NewService(store, &versionFailingCompiler{ + delegate: NewCompilerWithFeatureCaps(nil, core.DefaultWorkflowFeatures()), + version: "provider-v1", + err: errors.New("unknown guardrail ref: sys-tone"), + }) + require.NoError(t, err) + + view, err := service.GetView(context.Background(), "provider-v1") + require.NoError(t, err) + assert.Equal(t, "provider-v1", view.ID) + assert.Equal(t, "provider", view.ScopeType) + assert.Equal(t, "compile workflow \"provider-v1\": unknown guardrail ref: sys-tone", view.CompileError) +} + func TestViewScopeSpecificity_PathExceedsProvider(t *testing.T) { require.Greater(t, viewScopeSpecificity("path"), viewScopeSpecificity("provider")) } diff --git a/tests/e2e/manage-release-e2e-stack.sh b/tests/e2e/manage-release-e2e-stack.sh index 102cb9871..2fdcf4efd 100755 --- a/tests/e2e/manage-release-e2e-stack.sh +++ b/tests/e2e/manage-release-e2e-stack.sh @@ -11,6 +11,9 @@ MONGO_DATABASE="${GOMODEL_RELEASE_MONGO_DATABASE:-gomodel_release_e2e}" MOCK_MCP_BIN="${GOMODEL_RELEASE_MOCK_MCP_BINARY:-$REPO_ROOT/bin/mockmcp}" MOCK_MCP_PORT="${GOMODEL_RELEASE_MOCK_MCP_PORT:-18090}" MOCK_MCP_TOKEN="${GOMODEL_RELEASE_MOCK_MCP_TOKEN:-qa-mock-mcp-secret}" +MOCK_JEV_BIN="${GOMODEL_RELEASE_MOCK_JEV_BINARY:-$REPO_ROOT/bin/mockjev}" +MOCK_JEV_PORT="${GOMODEL_RELEASE_MOCK_JEV_PORT:-18091}" +MOCK_JEV_KEY="${GOMODEL_RELEASE_MOCK_JEV_KEY:-qa-mock-jev-key}" BUILD_BEFORE_START=0 @@ -37,6 +40,7 @@ Gateways: Helpers: mock-mcp http://localhost:18090 (mock MCP upstream: /alpha token-gated, /beta open) + mock-jev http://localhost:18091 (mock System One upstreams: /jev keyed, /kev keyless Kev, /down always 529) EOF } @@ -99,7 +103,24 @@ load_env() { export OPENROUTER_MODEL_FILTER_INCLUDE="${OPENROUTER_MODEL_FILTER_INCLUDE:-*:free}" export XAI_MODELS="${XAI_MODELS:-grok-4.3,grok-voice-latest}" export BAILIAN_MODELS="${BAILIAN_MODELS:-qwen3-omni-flash-realtime}" - export ENABLED_PASSTHROUGH_PROVIDERS="${ENABLED_PASSTHROUGH_PROVIDERS:-openai,anthropic,openrouter,zai,vllm,deepseek,bailian,xai}" + export ENABLED_PASSTHROUGH_PROVIDERS="${ENABLED_PASSTHROUGH_PROVIDERS:-openai,anthropic,openrouter,zai,vllm,deepseek,bailian,xai,jev}" + # System One (Jev / Kev) providers backed by the local mockjev upstream: + # "jev" is hosted-shaped and keyed, "jev-kev" a keyless Kev server, and + # "jev-down" answers 529 so System One failover can be exercised. Every + # JEV_* value from .env is dropped first (suffixed keys, model lists): the + # scenarios assert what the mock echoes back, and a key left on a keyless + # provider would reach the mock. The Kev URL keeps a trailing /v1, which the + # provider trims. + local name + for name in $(compgen -e); do + if [[ "$name" == JEV_* ]]; then + unset "$name" + fi + done + export JEV_API_KEY="$MOCK_JEV_KEY" + export JEV_BASE_URL="http://localhost:$MOCK_JEV_PORT/jev" + export JEV_KEV_BASE_URL="http://localhost:$MOCK_JEV_PORT/kev/v1" + export JEV_DOWN_BASE_URL="http://localhost:$MOCK_JEV_PORT/down" } ensure_binary() { @@ -109,54 +130,62 @@ ensure_binary() { if (( BUILD_BEFORE_START == 1 )) || [[ ! -x "$MOCK_MCP_BIN" ]]; then (cd "$REPO_ROOT" && go build -o "$MOCK_MCP_BIN" ./tests/e2e/mockmcp) fi + if (( BUILD_BEFORE_START == 1 )) || [[ ! -x "$MOCK_JEV_BIN" ]]; then + (cd "$REPO_ROOT" && go build -o "$MOCK_JEV_BIN" ./tests/e2e/mockjev) + fi } -start_mock_mcp() { - local dir="$STACK_DIR/mock-mcp" +# Starts one mock upstream binary on its port and waits for /healthz. +# usage: start_mock NAME PORT BINARY [ENV=VALUE...] +start_mock() { + local name="$1" port="$2" bin="$3" + shift 3 + local dir="$STACK_DIR/$name" local log_file="$dir/logs/server.log" local pid_file="$dir/server.pid" mkdir -p "$dir/logs" if is_pid_running "$pid_file"; then - printf 'mock-mcp already running pid=%s url=http://localhost:%s\n' "$(cat "$pid_file")" "$MOCK_MCP_PORT" + printf '%s already running pid=%s url=http://localhost:%s\n' "$name" "$(cat "$pid_file")" "$port" return 0 fi # A foreign process on the port would answer the health probe and mask a - # failed bind (e.g. a manually started mockmcp with a different token). - if curl -fsS "http://localhost:$MOCK_MCP_PORT/healthz" >/dev/null 2>&1; then - die "port $MOCK_MCP_PORT is already in use by an unmanaged process; stop it before starting mock-mcp" + # failed bind (e.g. a manually started mock with different settings). + if curl -fsS "http://localhost:$port/healthz" >/dev/null 2>&1; then + die "port $port is already in use by an unmanaged process; stop it before starting $name" fi rm -f "$pid_file" ( cd "$dir" - nohup env PORT="$MOCK_MCP_PORT" MOCK_MCP_TOKEN="$MOCK_MCP_TOKEN" "$MOCK_MCP_BIN" >"$log_file" 2>&1 < /dev/null & + nohup env PORT="$port" "$@" "$bin" >"$log_file" 2>&1 < /dev/null & echo $! >"$pid_file" ) local attempt for attempt in $(seq 1 15); do - if curl -fsS "http://localhost:$MOCK_MCP_PORT/healthz" >/dev/null 2>&1; then - printf 'started mock-mcp pid=%s url=http://localhost:%s\n' "$(cat "$pid_file")" "$MOCK_MCP_PORT" + if curl -fsS "http://localhost:$port/healthz" >/dev/null 2>&1; then + printf 'started %s pid=%s url=http://localhost:%s\n' "$name" "$(cat "$pid_file")" "$port" return 0 fi sleep 1 done - echo "failed to start mock-mcp on port $MOCK_MCP_PORT" >&2 + echo "failed to start $name on port $port" >&2 [[ -f "$log_file" ]] && tail -n 40 "$log_file" >&2 exit 1 } -stop_mock_mcp() { - local pid_file="$STACK_DIR/mock-mcp/server.pid" +stop_mock() { + local name="$1" + local pid_file="$STACK_DIR/$name/server.pid" local pid if [[ ! -f "$pid_file" ]]; then - printf 'mock-mcp not running\n' + printf '%s not running\n' "$name" return 0 fi @@ -165,22 +194,23 @@ stop_mock_mcp() { kill "$pid" 2>/dev/null || true fi rm -f "$pid_file" - printf 'stopped mock-mcp\n' + printf 'stopped %s\n' "$name" } -status_mock_mcp() { - local pid_file="$STACK_DIR/mock-mcp/server.pid" +status_mock() { + local name="$1" port="$2" + local pid_file="$STACK_DIR/$name/server.pid" local health="down" local pid="stopped" if is_pid_running "$pid_file"; then pid="$(cat "$pid_file")" - if curl -fsS "http://localhost:$MOCK_MCP_PORT/healthz" >/dev/null 2>&1; then + if curl -fsS "http://localhost:$port/healthz" >/dev/null 2>&1; then health="ok" fi fi - printf '%-12s pid=%-8s url=http://localhost:%s health=%s\n' "mock-mcp" "$pid" "$MOCK_MCP_PORT" "$health" + printf '%-12s pid=%-8s url=http://localhost:%s health=%s\n' "$name" "$pid" "$port" "$health" } ensure_pg_database() { @@ -360,7 +390,8 @@ start_stack() { mkdir -p "$STACK_DIR" ensure_pg_database write_guardrail_config - start_mock_mcp + start_mock mock-mcp "$MOCK_MCP_PORT" "$MOCK_MCP_BIN" MOCK_MCP_TOKEN="$MOCK_MCP_TOKEN" + start_mock mock-jev "$MOCK_JEV_PORT" "$MOCK_JEV_BIN" MOCK_JEV_KEY="$MOCK_JEV_KEY" start_gateway sqlite-main \ -u GOMODEL_MASTER_KEY \ @@ -457,11 +488,13 @@ stop_stack() { stop_gateway mongo-smoke stop_gateway pg-smoke stop_gateway sqlite-main - stop_mock_mcp + stop_mock mock-jev + stop_mock mock-mcp } status_stack() { - status_mock_mcp + status_mock mock-mcp "$MOCK_MCP_PORT" + status_mock mock-jev "$MOCK_JEV_PORT" status_gateway sqlite-main status_gateway pg-smoke status_gateway mongo-smoke diff --git a/tests/e2e/mockjev/main.go b/tests/e2e/mockjev/main.go new file mode 100644 index 000000000..892ddb690 --- /dev/null +++ b/tests/e2e/mockjev/main.go @@ -0,0 +1,348 @@ +// Command mockjev serves deterministic TypeSafe System One upstreams for the +// release E2E curl matrix, one per path prefix, so a single process backs +// several jev providers: +// +// /jev hosted-API shape: requires "Authorization: Bearer $MOCK_JEV_KEY"; +// lists jev-latest and jev-preview by "name"; accepts any versioned +// ID (jev-1.13.0); answers jev-latest as jev-1.13.0; has no Kev +// diagnostic routes (404, like the hosted API). +// /kev Kev-server shape: no authentication; lists checkpoint kev-latest +// by "id" with alias kev-4b; serves /v1/systemone/permute and +// /v1/systemone/separate. +// /down lists kev-down but answers every System One route with 529, the +// status TypeSafe uses for overload, so failover can be exercised. +// +// Each answer carries a "mock" object echoing what reached the upstream (the +// model, state, questions, extra top-level fields, whether an Authorization +// header arrived, the X-Request-Id, and a per-upstream request sequence), so +// scenarios can assert what the gateway forwarded and whether a cached +// answer was replayed. Malformed questions get a FastAPI-style 422, as the +// hosted API returns. +// +// PORT selects the listen port (default 18091). GET /healthz reports liveness. +package main + +import ( + "bytes" + "encoding/json" + "fmt" + "log" + "net/http" + "os" + "regexp" + "sort" + "strings" + "sync" +) + +type upstream struct { + name string + key string // required bearer key; empty means no authentication + models any // GET /v1/models body + kevRoute bool // serves permute and separate + down bool // answers System One routes with 529 + accepts func(model string) (answeredAs string, ok bool) + + mu sync.Mutex + seq int +} + +// maxBodyBytes bounds a request body; the largest the matrix sends is about +// 70 KiB. +const maxBodyBytes = 1 << 20 + +var versionedJev = regexp.MustCompile(`^jev-\d+\.\d+\.\d+$`) + +func newUpstreams(jevKey string) []*upstream { + return []*upstream{ + { + name: "jev", + key: jevKey, + models: map[string]any{"models": []map[string]any{ + {"name": "jev-latest", "description": "Latest Jev (mock)", "release_date": "2026-06-01"}, + {"name": "jev-preview", "description": "Preview Jev (mock)"}, + }}, + accepts: func(model string) (string, bool) { + switch { + case model == "jev-latest": + return "jev-1.13.0", true + case model == "jev-preview": + return "jev-1.14.0-preview", true + case versionedJev.MatchString(model): + return model, true + } + return "", false + }, + }, + { + name: "kev", + kevRoute: true, + models: map[string]any{"models": []map[string]any{ + {"id": "kev-latest", "aliases": []string{"kev-4b"}, "release_date": "2026-09-01T00:00:00Z"}, + }}, + accepts: func(model string) (string, bool) { + if model == "kev-latest" || model == "kev-4b" { + return "kev-4b-e2e", true + } + return "", false + }, + }, + { + name: "down", + kevRoute: true, + down: true, + models: map[string]any{"models": []map[string]any{{"id": "kev-down"}}}, + accepts: func(string) (string, bool) { return "", false }, + }, + } +} + +// question keeps instructions and criteria as raw JSON: the SDKs accept +// structured values there, and an answer needs only option names and the +// number of score levels. +type question struct { + Type string `json:"type"` + Instructions json.RawMessage `json:"instructions"` + Criteria json.RawMessage `json:"criteria"` +} + +type request struct { + Model string `json:"model"` + State json.RawMessage `json:"state"` + Questions map[string]question `json:"questions"` + NPerm *int `json:"n_perm"` +} + +func writeJSON(w http.ResponseWriter, status int, body any) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(status) + _ = json.NewEncoder(w).Encode(body) +} + +// validationError mirrors the FastAPI 422 body the hosted API returns. +func validationError(w http.ResponseWriter, loc []any, msg string) { + writeJSON(w, http.StatusUnprocessableEntity, map[string]any{ + "detail": []map[string]any{{"loc": append([]any{"body"}, loc...), "msg": msg, "type": "value_error"}}, + }) +} + +func (u *upstream) authorized(r *http.Request) bool { + return u.key == "" || r.Header.Get("Authorization") == "Bearer "+u.key +} + +func (u *upstream) authSeen(r *http.Request) string { + switch auth := r.Header.Get("Authorization"); { + case auth == "": + return "none" + case u.key != "" && auth == "Bearer "+u.key: + return "provider-key" + default: + return "other" + } +} + +func (u *upstream) serveModels(w http.ResponseWriter, r *http.Request) { + if !u.authorized(r) { + writeJSON(w, http.StatusUnauthorized, map[string]any{"detail": "Invalid API key"}) + return + } + writeJSON(w, http.StatusOK, u.models) +} + +func (u *upstream) serveSystemOne(route string) http.HandlerFunc { + return func(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodPost { + writeJSON(w, http.StatusMethodNotAllowed, map[string]any{"detail": "Method Not Allowed"}) + return + } + if route != "evaluate" && !u.kevRoute { + writeJSON(w, http.StatusNotFound, map[string]any{"detail": "Not Found"}) + return + } + if !u.authorized(r) { + writeJSON(w, http.StatusUnauthorized, map[string]any{"detail": "Invalid API key"}) + return + } + if u.down { + writeJSON(w, 529, map[string]any{"detail": "Overloaded (mock " + u.name + ")"}) + return + } + var raw bytes.Buffer + if _, err := raw.ReadFrom(http.MaxBytesReader(w, r.Body, maxBodyBytes)); err != nil { + writeJSON(w, http.StatusRequestEntityTooLarge, map[string]any{"detail": err.Error()}) + return + } + var req request + if err := json.Unmarshal(raw.Bytes(), &req); err != nil { + validationError(w, nil, "invalid JSON: "+err.Error()) + return + } + answeredAs, ok := u.accepts(req.Model) + if !ok { + writeJSON(w, http.StatusNotFound, map[string]any{"detail": fmt.Sprintf("Model %q not found", req.Model)}) + return + } + if len(req.State) == 0 || string(req.State) == "null" { + validationError(w, []any{"state"}, "Field required") + return + } + if len(req.Questions) == 0 { + validationError(w, []any{"questions"}, "At least one question is required") + return + } + names := make([]string, 0, len(req.Questions)) + for name := range req.Questions { + names = append(names, name) + } + sort.Strings(names) + + nPerm := 0 + if route == "permute" { + if len(req.Questions) != 1 || req.Questions[names[0]].Type != "choice" { + validationError(w, []any{"questions"}, "permute takes exactly one choice question") + return + } + nPerm = 6 + if req.NPerm != nil { + nPerm = *req.NPerm + } + if nPerm < 1 || nPerm > 64 { + validationError(w, []any{"n_perm"}, "n_perm must be between 1 and 64") + return + } + } + + answers := make(map[string]any, len(names)) + for _, name := range names { + answer, msg := answerFor(req.Questions[name]) + if msg != "" { + validationError(w, []any{"questions", name}, msg) + return + } + answers[name] = answer + } + + u.mu.Lock() + u.seq++ + seq := u.seq + u.mu.Unlock() + + var top map[string]json.RawMessage + _ = json.Unmarshal(raw.Bytes(), &top) + extra := map[string]json.RawMessage{} + for key, value := range top { + switch key { + case "model", "state", "questions", "n_perm": + default: + extra[key] = value + } + } + + body := map[string]any{ + "model": answeredAs, + "answers": answers, + "usage": map[string]any{ + "input_tokens": 10 + len(req.State)/4 + 5*len(names), + "output_tokens": 3 * len(names), + }, + "mock": map[string]any{ + "upstream": u.name, + "route": route, + "received_model": req.Model, + "received_state": req.State, + "questions": top["questions"], + "extra": extra, + "authorization": u.authSeen(r), + "request_id": r.Header.Get("X-Request-Id"), + "request_seq": seq, + "received_length": raw.Len(), + }, + } + if route == "permute" { + body["n_perm"] = nPerm + } + if route == "separate" { + body["separate"] = true + } + writeJSON(w, http.StatusOK, body) + } +} + +// answerFor builds a deterministic, well-formed answer for one question, or +// returns the validation message the hosted API would reject it with. +func answerFor(q question) (any, string) { + switch q.Type { + case "noul": + return map[string]any{"type": "noul", "noul": 0.93}, "" + case "choice": + var criteria map[string]json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil || len(criteria) < 2 { + return nil, "choice criteria must map at least two option names to descriptions" + } + options := make([]string, 0, len(criteria)) + for option := range criteria { + options = append(options, option) + } + sort.Strings(options) + probabilities := make(map[string]float64, len(options)) + rest := 0.4 / float64(len(options)-1) + for i, option := range options { + probabilities[option] = rest + if i == 0 { + probabilities[option] = 0.6 + } + } + return map[string]any{"type": "choice", "choice": options[0], "confidence": 0.5, "probabilities": probabilities}, "" + case "score": + var levels []json.RawMessage + if err := json.Unmarshal(q.Criteria, &levels); err != nil || len(levels) < 2 { + return nil, "score criteria must list at least two ordered levels" + } + legend := make(map[string]any, len(levels)) + probabilities := make(map[string]float64, len(levels)) + for i, level := range levels { + var label string + if json.Unmarshal(level, &label) == nil { + legend[fmt.Sprint(i)] = label + } else { + legend[fmt.Sprint(i)] = level + } + probabilities[fmt.Sprint(i)] = 0 + } + probabilities["1"] = 1 + return map[string]any{"type": "score", "score": 1.0, "confidence": 0.8, "legend": legend, "probabilities": probabilities}, "" + case "": + return nil, "Field required: type" + default: + return nil, fmt.Sprintf("Input tag %q found using 'type' does not match any of the expected tags: 'noul', 'choice', 'score'", q.Type) + } +} + +func main() { + port := os.Getenv("PORT") + if port == "" { + port = "18091" + } + jevKey := os.Getenv("MOCK_JEV_KEY") + if jevKey == "" { + jevKey = "qa-mock-jev-key" + } + + mux := http.NewServeMux() + for _, u := range newUpstreams(jevKey) { + prefix := "/" + u.name + mux.HandleFunc(prefix+"/v1/models", u.serveModels) + mux.HandleFunc(prefix+"/v1/systemone", u.serveSystemOne("evaluate")) + mux.HandleFunc(prefix+"/v1/systemone/permute", u.serveSystemOne("permute")) + mux.HandleFunc(prefix+"/v1/systemone/separate", u.serveSystemOne("separate")) + } + mux.HandleFunc("/healthz", func(w http.ResponseWriter, _ *http.Request) { + fmt.Fprintln(w, "ok") + }) + mux.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) { + writeJSON(w, http.StatusNotFound, map[string]any{"detail": "Not Found: " + strings.TrimSpace(r.URL.Path)}) + }) + + log.Printf("mockjev listening on :%s", port) + log.Fatal(http.ListenAndServe(":"+port, mux)) +} diff --git a/tests/e2e/release-e2e-scenarios.md b/tests/e2e/release-e2e-scenarios.md index d51206e89..137228fac 100644 --- a/tests/e2e/release-e2e-scenarios.md +++ b/tests/e2e/release-e2e-scenarios.md @@ -11,6 +11,9 @@ These scenarios are prepared for execution across these local gateways: - `http://localhost:18090` - mock MCP upstream (`tests/e2e/mockmcp`, started by the stack manager; `/alpha` requires the `X-Mock-Token` header, `/beta` is open) +- `http://localhost:18091` - mock System One upstreams (`tests/e2e/mockjev`, + started by the stack manager and registered on every gateway as the `jev`, + `jev-kev`, and `jev-down` providers) ## Recommended runner @@ -171,6 +174,22 @@ Stateful note: Gemini 3 tool call replayed with its thought signature. They create and clean up their own artifacts and are rerunnable in any order; `S227` reloads the SQLite gateway and therefore stays sequential +- `S229`-`S241` exercise the Jev / Kev System One API (`/v1/systemone`, Kev's + `/permute` and `/separate`, passthrough, pinned versions, misuse negatives, + audit and usage, failover, exact cache on the auth gateway, state guardrails + on the guardrail gateway, managed-key allowlists) against the mock upstream + on port 18091, since no hosted Jev key or Kev server is available; each + prints `SKIPPED:` and exits 0 when the mock is down. They create and delete + their own `$QA_SUFFIX`-scoped virtual models, guardrails, workflows, keys, + and pricing overrides and are rerunnable in any order. `S237` sets a pricing + override and `S240` a guardrail workflow, so both stay sequential +- `S242`-`S244` exercise MCP per-server tool filters and + `disallowed_user_paths` (in-place edits reaching open sessions) and the + master key keeping the caller's user-path header on `/mcp` and audio + uploads; they register `$QA_SUFFIX`-scoped servers and delete them, but + mutate the shared MCP catalog, so they stay sequential +- `S245`-`S246` exercise `developer` messages, `strict` tools, and Gemini's + `allowed_tools` tool choice; they are read-only and rerunnable in any order - `S218` exercises Gemini's native `batchEmbedContents` path (batch input, `dimensions`); read-only and rerunnable in any order - `S219` asserts the effective resilience configuration on @@ -487,6 +506,58 @@ mcp_cleanup_release_servers() { curl -sS -o /dev/null -X DELETE "$base/admin/mcp-servers/$QA_MCP_BETA" || true } +# System One (Jev / Kev) upstreams served by tests/e2e/mockjev: the stack +# manager registers "jev" (hosted shape, keyed), "jev-kev" (keyless Kev +# server), and "jev-down" (always 529) on every gateway. +export JEV_MOCK_BASE="${JEV_MOCK_BASE:-http://localhost:${GOMODEL_RELEASE_MOCK_JEV_PORT:-18091}}" +export QA_SYSTEMONE_QUESTIONS='{"department":{"type":"choice","instructions":"Which team should handle this?","criteria":{"returns":"Exchanges and refunds","shipping":"Delivery delays","billing":"Charges and invoices"}},"escalate":{"type":"noul","instructions":"Does this need urgent human attention?"},"frustration":{"type":"score","instructions":"How frustrated is the customer?","criteria":["Calm","Frustrated","Very angry"]}}' +export QA_SYSTEMONE_CHOICE='{"department":{"type":"choice","instructions":"Which team?","criteria":{"returns":"Returns","billing":"Billing"}}}' + +# Skips when the mock upstream is down, and fails when the gateway was started +# without the mock-backed jev providers (an outdated stack manager). +# usage: systemone_require_mock BASE_URL [curl args...] +systemone_require_mock() { + local base="$1" + shift + if ! curl -fsS "$JEV_MOCK_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock System One upstream is not running on $JEV_MOCK_BASE" + exit 0 + fi + if ! curl -fsS "$base/v1/models" "$@" | jq -e ' + any(.data[]; .id == "jev/jev-latest") and any(.data[]; .id == "jev-kev/kev-latest") + ' >/dev/null; then + echo "error: $base has no mock-backed jev providers; restart it with tests/e2e/manage-release-e2e-stack.sh" >&2 + exit 1 + fi +} + +# Asserts an HTTP status and prints the body on a mismatch. +# usage: assert_http_status WANT GOT BODY_FILE +assert_http_status() { + if [ "$2" != "$1" ]; then + echo "error: expected HTTP $1, got $2" >&2 + cat "$3" >&2 || true + exit 1 + fi +} + +# Polls the audit or usage log until an entry for the request id appears. +# usage: wait_log_entry BASE_URL audit|usage REQUEST_ID OUTPUT_FILE [curl args...] +wait_log_entry() { + local base="$1" kind="$2" rid="$3" out="$4" + shift 4 + for _ in $(seq 1 15); do + curl -fsS "$base/admin/$kind/log?search=$rid&limit=5" "$@" > "$out" + if jq -e --arg rid "$rid" 'any(.entries[]?; .request_id == $rid)' "$out" >/dev/null; then + return 0 + fi + sleep 1 + done + jq . "$out" >&2 || true + echo "error: no $kind entry for $rid on $base" >&2 + exit 1 +} + run_release_budget_enforcement() { local base_url="$1" local budget_path="$2" @@ -3565,7 +3636,7 @@ if jq -e '.providers[] | select(.name == "fireworks") | (.status != "healthy") a exit 0 fi FIREWORKS_MODEL=$(curl -fsS "$BASE_URL/v1/models" \ - | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("llama-v3p1-8b-instruct$"))) + .)[0]') + | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("gpt-oss-120b$"))) + .)[0]') RESP_FILE="$QA_RUN_DIR/s153.chat.json" curl -fsS "$BASE_URL/v1/chat/completions" \ -H 'Content-Type: application/json' \ @@ -3589,7 +3660,7 @@ if jq -e '.providers[] | select(.name == "fireworks") | (.status != "healthy") a exit 0 fi FIREWORKS_MODEL=$(curl -fsS "$BASE_URL/v1/models" \ - | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("llama-v3p1-8b-instruct$"))) + .)[0]') + | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("gpt-oss-120b$"))) + .)[0]') SSE_FILE="$QA_RUN_DIR/s154.chat.sse" curl -fsS --no-buffer "$BASE_URL/v1/chat/completions" \ -H 'Content-Type: application/json' \ @@ -6166,3 +6237,946 @@ jq -c --argjson tools "$TOOLS" '{ jq '{provider,answer:.choices[0].message.content}' "$FOLLOW_FILE" assert_chat_response_contains "$FOLLOW_FILE" "gemini" "22" ``` + +## 35. Jev / Kev System One API + +`POST /v1/systemone` (and Kev's `/permute` and `/separate`) forwards TypeSafe +System One decision requests natively. No hosted Jev key or Kev server is +available to the matrix, so the stack manager starts `tests/e2e/mockjev` on +port 18091 and registers three `jev` providers against it on every gateway: +`jev` (hosted-API shape, keyed, lists `jev-latest`/`jev-preview`, accepts any +versioned `jev-X.Y.Z`), `jev-kev` (keyless Kev server whose base URL keeps a +trailing `/v1`, checkpoint `kev-latest` with alias `kev-4b`), and `jev-down` +(lists `kev-down`, answers every System One route with `529`). Each mock +answer carries a `mock` object echoing what reached the upstream (model, +state, questions, extra fields, whether an `Authorization` header arrived, +`X-Request-Id`, and a per-upstream request sequence), so the scenarios can +assert exactly what the gateway forwarded and whether an answer was replayed +from cache. + +### S229 System One providers register and list utility models + +```bash +systemone_require_mock "$BASE_URL" + +MODELS_FILE="$QA_RUN_DIR/s229.models.json" +curl -fsS "$BASE_URL/v1/models" > "$MODELS_FILE" +jq -c '[.data[] | select(.owned_by | startswith("jev")) | {id, categories: .metadata.categories, modes: .metadata.modes}]' "$MODELS_FILE" +jq -e ' + ([.data[] | select(.owned_by | startswith("jev")) | .id] | sort) + == ["jev-down/kev-down","jev-kev/kev-4b","jev-kev/kev-latest","jev/jev-latest","jev/jev-preview"] + and all(.data[] | select(.owned_by | startswith("jev")); .metadata.categories == ["utility"] and ((.metadata.modes // []) | length == 0)) + and any(.data[]; .id == "jev/jev-latest" and .metadata.description == "Latest Jev (mock)" and .created > 0) +' "$MODELS_FILE" >/dev/null + +STATUS_FILE="$QA_RUN_DIR/s229.status.json" +curl -fsS "$BASE_URL/admin/providers/status" > "$STATUS_FILE" +# jev-down turns degraded once failover scenarios have sent it traffic (its +# 529s count against request health), so it only has to be registered. +jq -e ' + [.. | objects | select(.type? == "jev" and has("status")) | {name, status}] | sort_by(.name) as $s + | ($s | map(.name)) == ["jev","jev-down","jev-kev"] + and all($s[]; if .name == "jev-down" then (.status | IN("healthy","degraded")) else .status == "healthy" end) +' "$STATUS_FILE" >/dev/null + +# Passthrough lists models in each upstream's own shape. +curl -fsS "$BASE_URL/p/jev/v1/models" \ + | jq -e '[.models[].name] == ["jev-latest","jev-preview"]' >/dev/null +curl -fsS "$BASE_URL/p/jev-kev/v1/models" \ + | jq -e '.models[0].id == "kev-latest" and .models[0].aliases == ["kev-4b"]' >/dev/null +``` + +### S230 Native `/v1/systemone` answers every question type on hosted Jev + +Sends choice, noul, and score questions plus an extra top-level field. The +answer is relayed unchanged, only `model` is rewritten to the resolved name, +the questions and extra field reach the upstream byte for byte, the client's +`Authorization` header is replaced by the provider key, and the request ID is +forwarded. + +```bash +systemone_require_mock "$BASE_URL" + +for MODEL in jev-latest jev/jev-latest; do + RID="qa-s1-hosted-$QA_SUFFIX-${MODEL//\//-}" + RESP_FILE="$QA_RUN_DIR/s230.${MODEL//\//-}.json" + CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -H 'Authorization: Bearer qa-client-token-must-not-reach-upstream' \ + -H "X-Request-ID: $RID" \ + -d "{\"model\":\"$MODEL\",\"state\":\"Shoes arrived late and I see two charges on my card.\",\"questions\":$QA_SYSTEMONE_QUESTIONS,\"qa_marker\":{\"nested\":[1,2,3]}}") + assert_http_status 200 "$CODE" "$RESP_FILE" + jq -c '{model, answers, usage, mock: (.mock | {upstream, received_model, authorization, request_id})}' "$RESP_FILE" + jq -e --arg rid "$RID" --argjson questions "$QA_SYSTEMONE_QUESTIONS" ' + .model == "jev-1.13.0" + and .answers.department.type == "choice" and (.answers.department.choice | IN("returns","shipping","billing")) + and (.answers.department.probabilities | keys | sort) == ["billing","returns","shipping"] + and .answers.escalate.type == "noul" and (.answers.escalate.noul | type == "number") + and .answers.frustration.type == "score" and .answers.frustration.legend == {"0":"Calm","1":"Frustrated","2":"Very angry"} + and .usage.input_tokens > 0 and .usage.output_tokens > 0 + and .mock.upstream == "jev" + and .mock.received_model == "jev-latest" + and .mock.questions == $questions + and .mock.extra == {"qa_marker":{"nested":[1,2,3]}} + and .mock.authorization == "provider-key" + and .mock.request_id == $rid + ' "$RESP_FILE" >/dev/null +done +``` + +### S231 Keyless Kev server by checkpoint and alias + +The `jev-kev` provider has no key and its base URL ends in `/v1`, which the +provider trims. No `Authorization` header may reach it, not even the client's. + +```bash +systemone_require_mock "$BASE_URL" + +for MODEL in jev-kev/kev-latest kev-latest jev-kev/kev-4b kev-4b; do + RESP_FILE="$QA_RUN_DIR/s231.${MODEL//\//-}.json" + CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -H 'Authorization: Bearer qa-client-token-must-not-reach-upstream' \ + -d "{\"model\":\"$MODEL\",\"state\":\"I was charged twice.\",\"questions\":$QA_SYSTEMONE_CHOICE}") + assert_http_status 200 "$CODE" "$RESP_FILE" + jq -e --arg sent "${MODEL#jev-kev/}" ' + .model == "kev-4b-e2e" + and .mock.upstream == "kev" + and .mock.received_model == $sent + and .mock.authorization == "none" + and .answers.department.type == "choice" + ' "$RESP_FILE" >/dev/null +done +``` + +### S232 Pinned Jev versions route without being listed + +TypeSafe lists only its aliases but accepts any versioned ID. A name that +says which `jev` provider to use reaches it unlisted; a bare unlisted name is +not guessed while several `jev` providers are configured; a model the +upstream rejects comes back with the upstream's status. + +```bash +systemone_require_mock "$BASE_URL" + +for MODEL in jev/jev-1.13.0 jev/jev-1.12.0; do + RESP_FILE="$QA_RUN_DIR/s232.${MODEL//\//-}.json" + CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"$MODEL\",\"state\":\"pinned\",\"questions\":$QA_SYSTEMONE_CHOICE}") + assert_http_status 200 "$CODE" "$RESP_FILE" + jq -e --arg version "${MODEL#jev/}" '.model == $version and .mock.upstream == "jev" and .mock.received_model == $version' "$RESP_FILE" >/dev/null +done + +# Bare and unlisted, with jev, jev-kev, and jev-down all configured. +RESP_FILE="$QA_RUN_DIR/s232.bare.json" +CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-1.13.0\",\"state\":\"pinned\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 404 "$CODE" "$RESP_FILE" +jq -e '.error.code == "model_not_found"' "$RESP_FILE" >/dev/null + +# Routed to jev because the name says so; the upstream rejects it. +RESP_FILE="$QA_RUN_DIR/s232.unknown.json" +CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev/not-a-jev-model\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 404 "$CODE" "$RESP_FILE" +jq -e '.error.type == "not_found_error" and .error.provider == "jev" and (.error.message | contains("not-a-jev-model"))' "$RESP_FILE" >/dev/null +``` + +### S233 A virtual model pins an unlisted Jev version + +The System One docs state that a virtual model can pin a version the same way +a provider-qualified name does (`virtual_models: [{source: ..., target: +jev/jev-1.13.0}]`). This creates one through the admin API and sends a request +through it. + +```bash +systemone_require_mock "$BASE_URL" + +NAME="qa-jev-pinned-$QA_SUFFIX" +cleanup_s233() { + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\"}" || true +} +trap cleanup_s233 EXIT + +VM_FILE="$QA_RUN_DIR/s233.vm.json" +CODE=$(curl -sS -o "$VM_FILE" -w '%{http_code}' -X PUT "$BASE_URL/admin/virtual-models" \ + -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\",\"target_model\":\"jev/jev-1.12.0\"}") +assert_http_status 200 "$CODE" "$VM_FILE" + +RESP_FILE="$QA_RUN_DIR/s233.answer.json" +CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"$NAME\",\"state\":\"pinned through a virtual model\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$RESP_FILE" +jq -e '.model == "jev-1.12.0" and .mock.upstream == "jev" and .mock.received_model == "jev-1.12.0"' "$RESP_FILE" >/dev/null +``` + +### S234 Kev diagnostic routes `/permute` and `/separate` + +Kev serves both diagnostic routes; the hosted-shaped `jev` answers them with +its own `404`, and an OpenRouter model is refused before any upstream call +since OpenRouter serves only the evaluation route. + +```bash +systemone_require_mock "$BASE_URL" + +post_systemone() { + local route="$1" body="$2" out="$3" + curl -sS -o "$out" -w '%{http_code}' "$BASE_URL/v1/systemone$route" -H 'Content-Type: application/json' -d "$body" +} + +F="$QA_RUN_DIR/s234.permute.json" +CODE=$(post_systemone /permute "{\"model\":\"jev-kev/kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"n_perm\":3}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.n_perm == 3 and .mock.route == "permute" and .mock.received_model == "kev-latest"' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.permute-default.json" +CODE=$(post_systemone /permute "{\"model\":\"kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.n_perm == 6' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.permute-bad.json" +CODE=$(post_systemone /permute "{\"model\":\"kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"n_perm\":65}" "$F") +assert_http_status 422 "$CODE" "$F" +jq -e '.error.type == "invalid_request_error" and (.error.message | contains("n_perm"))' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.separate.json" +CODE=$(post_systemone /separate "{\"model\":\"jev-kev/kev-4b\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_QUESTIONS}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.separate == true and .mock.route == "separate" and (.answers | keys | length) == 3' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.hosted-permute.json" +CODE=$(post_systemone /permute "{\"model\":\"jev/jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 404 "$CODE" "$F" +jq -e '.error.type == "not_found_error" and .error.provider == "jev"' "$F" >/dev/null + +OPENROUTER_MODEL=$(curl -fsS "$BASE_URL/v1/models" | jq -r '[.data[].id | select(startswith("openrouter/"))][0] // empty') +if [ -n "$OPENROUTER_MODEL" ]; then + F="$QA_RUN_DIR/s234.openrouter-permute.json" + CODE=$(post_systemone /permute "{\"model\":\"$OPENROUTER_MODEL\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") + assert_http_status 400 "$CODE" "$F" + jq -e '.error.param == "model" and (.error.message | contains("answers only /v1/systemone"))' "$F" >/dev/null +else + echo "note: no openrouter model in the catalog; OpenRouter permute refusal not checked" +fi +``` + +### S235 System One misuse is rejected with an explanation (negatives) + +The endpoint never translates: missing or malformed input, chat models (direct +or through a virtual model), and System One models on OpenAI routes are all +`400 invalid_request_error` naming the fix. A malformed question reaches the +upstream and comes back as its `422`. + +```bash +systemone_require_mock "$BASE_URL" + +NAME="qa-s1-chat-vm-$QA_SUFFIX" +cleanup_s235() { + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\"}" || true +} +trap cleanup_s235 EXIT +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\",\"target_model\":\"openai/gpt-4.1-nano\"}" >/dev/null + +# expect_invalid PATH BODY JQ_MESSAGE_FILTER +expect_invalid() { + local path="$1" body="$2" filter="$3" out + out=$(mktemp "$QA_RUN_DIR/s235.XXXXXX") + local code + code=$(curl -sS -o "$out" -w '%{http_code}' "$BASE_URL$path" -H 'Content-Type: application/json' -d "$body") + assert_http_status 400 "$code" "$out" + if ! jq -e ".error.type == \"invalid_request_error\" and ($filter)" "$out" >/dev/null; then + echo "error: unexpected 400 body for $path" >&2 + cat "$out" >&2 + exit 1 + fi +} + +expect_invalid /v1/systemone "{\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.param == "model" and .error.message == "model is required"' +expect_invalid /v1/systemone '{"model":' \ + '.error.message | startswith("invalid request body")' +expect_invalid /v1/systemone "{\"model\":\"openai/gpt-4.1-nano\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.param == "model" and (.error.message | contains("provider type openai, which has no System One API"))' +expect_invalid /v1/systemone "{\"model\":\"$NAME\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.message | contains("(resolved to \"openai/gpt-4.1-nano\")")' +OPENROUTER_MODEL=$(curl -fsS "$BASE_URL/v1/models" | jq -r '[.data[] | select((.id | startswith("openrouter/")) and ((.metadata.modes // []) | index("chat")))][0].id // empty') +if [ -n "$OPENROUTER_MODEL" ]; then + expect_invalid /v1/systemone "{\"model\":\"$OPENROUTER_MODEL\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.message | contains("not a System One model")' +fi + +expect_invalid /v1/chat/completions '{"model":"jev/jev-latest","messages":[{"role":"user","content":"hi"}]}' \ + '.error.param == "model" and (.error.message | contains("does not support chat completions") and contains("POST /v1/systemone"))' +expect_invalid /v1/responses '{"model":"jev-kev/kev-latest","input":"hi"}' \ + '.error.message | contains("does not support responses") and contains("POST /v1/systemone")' +expect_invalid /v1/embeddings '{"model":"jev-latest","input":"hi"}' \ + '.error.message | contains("does not support embeddings") and contains("POST /v1/systemone")' + +# A misrouted System One request is an operator mistake, so it is logged. +grep -Fq 'System One request routed to a model without the System One API' \ + "$RELEASE_STACK_DIR/sqlite-main/logs/server.log" + +F="$QA_RUN_DIR/s235.bad-question.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d '{"model":"jev-kev/kev-latest","state":"x","questions":{"q":{"type":"maybe","instructions":"?"}}}') +assert_http_status 422 "$CODE" "$F" +jq -e '.error.provider == "jev" and (.error.message | contains("questions") and contains("maybe"))' "$F" >/dev/null + +F="$QA_RUN_DIR/s235.get.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone") +assert_http_status 405 "$CODE" "$F" +``` + +### S236 Passthrough reaches the same upstreams under `/p/jev*` + +Passthrough forwards the body as sent (no model rewrite), rejects a body that +names the model twice (the upstream parser could pick a value the gateway +never checked), and reaches Kev's diagnostic routes on the suffixed provider. + +```bash +systemone_require_mock "$BASE_URL" + +F="$QA_RUN_DIR/s236.systemone.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/p/jev/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-latest\",\"state\":\"via passthrough\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$F" +jq -e '.model == "jev-1.13.0" and .mock.received_model == "jev-latest" and .mock.authorization == "provider-key"' "$F" >/dev/null + +# The /v1 prefix is optional on passthrough routes. +F="$QA_RUN_DIR/s236.permute.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/p/jev-kev/systemone/permute" -H 'Content-Type: application/json' \ + -d "{\"model\":\"kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"n_perm\":2}") +assert_http_status 200 "$CODE" "$F" +jq -e '.n_perm == 2 and .mock.upstream == "kev" and .mock.authorization == "none"' "$F" >/dev/null + +F="$QA_RUN_DIR/s236.dup-model.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/p/jev/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"model\":\"jev-preview\"}") +assert_http_status 400 "$CODE" "$F" +jq -e '.error.message | contains("model field is repeated")' "$F" >/dev/null +``` + +### S237 System One calls are audited, filterable, metered, and priced + +Checks the audit entry (route, requested and resolved model, provider, one +successful primary attempt), the `exclude_operation=systemone` request-type +filter, and the usage entry: tokens copied from the answer, recorded under +the model that answered, and priced by an operator override. + +```bash +systemone_require_mock "$BASE_URL" + +# A provider-wide rate must not shadow the rate declared for the version that +# answers the jev-latest alias. +SELECTOR="jev/jev-1.13.0" +PROVIDER_SELECTOR="jev/" +cleanup_s237() { + for selector in "$SELECTOR" "$PROVIDER_SELECTOR"; do + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/model-pricing-overrides" \ + -H 'Content-Type: application/json' -d "{\"selector\":\"$selector\"}" || true + done +} +trap cleanup_s237 EXIT +curl -fsS -X PUT "$BASE_URL/admin/model-pricing-overrides" -H 'Content-Type: application/json' \ + -d "{\"selector\":\"$SELECTOR\",\"pricing\":{\"input_per_mtok\":42,\"output_per_mtok\":0}}" >/dev/null +curl -fsS -X PUT "$BASE_URL/admin/model-pricing-overrides" -H 'Content-Type: application/json' \ + -d "{\"selector\":\"$PROVIDER_SELECTOR\",\"pricing\":{\"input_per_mtok\":1,\"output_per_mtok\":1}}" >/dev/null + +RID="qa-s1-audit-$QA_SUFFIX" +ANSWER_FILE="$QA_RUN_DIR/s237.answer.json" +curl -fsS "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' -H "X-Request-ID: $RID" \ + -d "{\"model\":\"jev/jev-latest\",\"state\":\"audit me\",\"questions\":$QA_SYSTEMONE_QUESTIONS}" > "$ANSWER_FILE" +IN=$(jq -er '.usage.input_tokens' "$ANSWER_FILE") +OUT=$(jq -er '.usage.output_tokens' "$ANSWER_FILE") + +AUDIT_FILE="$QA_RUN_DIR/s237.audit.json" +wait_log_entry "$BASE_URL" audit "$RID" "$AUDIT_FILE" +jq -e --arg rid "$RID" ' + any(.entries[]; .request_id == $rid + and .path == "/v1/systemone" and .method == "POST" and .status_code == 200 + and .requested_model == "jev/jev-latest" and .resolved_model == "jev/jev-latest" + and .provider == "jev" and .provider_name == "jev" + and ([.data.attempts[]? | {kind, provider_name, success}] == [{"kind":"primary","provider_name":"jev","success":true}])) +' "$AUDIT_FILE" >/dev/null + +curl -fsS "$BASE_URL/admin/audit/log?search=$RID&limit=5&exclude_operation=systemone" \ + | jq -e '(.entries // []) | length == 0' >/dev/null +curl -fsS "$BASE_URL/admin/audit/log?search=$RID&limit=5&exclude_operation=chat_completions,provider_passthrough" \ + | jq -e --arg rid "$RID" 'any(.entries[]; .request_id == $rid)' >/dev/null +CODE=$(curl -sS -o "$QA_RUN_DIR/s237.bad-filter.json" -w '%{http_code}' "$BASE_URL/admin/audit/log?exclude_operation=not_an_operation") +assert_http_status 400 "$CODE" "$QA_RUN_DIR/s237.bad-filter.json" + +USAGE_FILE="$QA_RUN_DIR/s237.usage.json" +wait_log_entry "$BASE_URL" usage "$RID" "$USAGE_FILE" +jq -c --arg rid "$RID" '.entries[] | select(.request_id == $rid)' "$USAGE_FILE" +jq -e --arg rid "$RID" --argjson in "$IN" --argjson out "$OUT" ' + any(.entries[]; .request_id == $rid + and .endpoint == "/v1/systemone" and .model == "jev-1.13.0" + and .provider == "jev" and .provider_name == "jev" + and .input_tokens == $in and .output_tokens == $out) +' "$USAGE_FILE" >/dev/null +# The two checks below are independent, so both are reported before failing. +FAILED=0 +if ! jq -e --arg rid "$RID" --argjson total "$((IN + OUT))" \ + 'any(.entries[]; .request_id == $rid and .total_tokens == $total)' "$USAGE_FILE" >/dev/null; then + echo "error: usage total_tokens is not input_tokens + output_tokens ($IN + $OUT) for a System One answer" >&2 + FAILED=1 +fi +# Jev's documented pricing: per input token, output_per_mtok 0. +if ! jq -e --arg rid "$RID" --argjson in "$IN" ' + any(.entries[]; .request_id == $rid and ((((.input_cost // -1) - ($in * 42 / 1000000)) | fabs) < 0.000000001)) + ' "$USAGE_FILE" >/dev/null; then + echo "error: the pricing override (input 42/Mtok, output 0) did not cost the System One usage entry" >&2 + FAILED=1 +fi +[ "$FAILED" = 0 ] +``` + +### S238 Failover moves System One requests between System One targets + +A `failover` virtual model whose primary answers `529` moves to its next +target, skips a chat model without spending an attempt, and records the +answer under the target that did the work. A client error (`422`) is returned +without failover. + +```bash +systemone_require_mock "$BASE_URL" + +FO="qa-s1-failover-$QA_SUFFIX" +FO422="qa-s1-failover-422-$QA_SUFFIX" +FOPIN="qa-s1-failover-pinned-$QA_SUFFIX" +cleanup_s238() { + for name in "$FO" "$FO422" "$FOPIN"; do + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$name\"}" || true + done +} +trap cleanup_s238 EXIT + +# The down target on its own relays the upstream overload status (or 503 +# once its circuit breaker has opened after earlier reruns). +F="$QA_RUN_DIR/s238.down.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-down/kev-down\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}") +case "$CODE" in 529|503) ;; *) assert_http_status 529 "$CODE" "$F" ;; esac + +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$FO\",\"strategy\":\"failover\",\"targets\":[{\"model\":\"jev-down/kev-down\"},{\"model\":\"openai/gpt-4.1-nano\"},{\"model\":\"jev-kev/kev-latest\"}]}" >/dev/null +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$FO422\",\"strategy\":\"failover\",\"targets\":[{\"model\":\"jev-kev/kev-latest\"},{\"model\":\"jev/jev-latest\"}]}" >/dev/null + +RID="qa-s1-failover-$QA_SUFFIX" +F="$QA_RUN_DIR/s238.failover.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -H "X-Request-ID: $RID" \ + -d "{\"model\":\"$FO\",\"state\":\"fail over please\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$F" +jq -e '.model == "kev-4b-e2e" and .mock.upstream == "kev" and .mock.received_model == "kev-latest"' "$F" >/dev/null + +AUDIT_FILE="$QA_RUN_DIR/s238.audit.json" +wait_log_entry "$BASE_URL" audit "$RID" "$AUDIT_FILE" +jq -c --arg rid "$RID" '.entries[] | select(.request_id == $rid) | [.data.attempts[] | {kind, provider_name, status_code, success}]' "$AUDIT_FILE" +jq -e --arg rid "$RID" --arg fo "$FO" ' + any(.entries[]; .request_id == $rid + and .requested_model == $fo and .resolved_model == "jev-kev/kev-latest" and .provider_name == "jev-kev" + and .data.failover != null + and ([.data.attempts[] | .provider_name] == ["jev-down","jev-kev"]) + and .data.attempts[0].success == false and .data.attempts[1].kind == "failover" and .data.attempts[1].success == true) +' "$AUDIT_FILE" >/dev/null + +USAGE_FILE="$QA_RUN_DIR/s238.usage.json" +wait_log_entry "$BASE_URL" usage "$RID" "$USAGE_FILE" +jq -e --arg rid "$RID" 'any(.entries[]; .request_id == $rid and .provider_name == "jev-kev" and .model == "kev-4b-e2e")' "$USAGE_FILE" >/dev/null + +# A pinned version the catalog does not list is a valid failover target and +# is sent to the provider it names, not to the failed primary's. +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$FOPIN\",\"strategy\":\"failover\",\"targets\":[{\"model\":\"jev-down/kev-down\"},{\"model\":\"jev/jev-1.12.0\"}]}" >/dev/null +F="$QA_RUN_DIR/s238.failover-pinned.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"$FOPIN\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$F" +jq -e '.model == "jev-1.12.0" and .mock.upstream == "jev" and .mock.received_model == "jev-1.12.0"' "$F" >/dev/null + +RID422="qa-s1-failover-422-$QA_SUFFIX" +F="$QA_RUN_DIR/s238.no-failover.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -H "X-Request-ID: $RID422" \ + -d "{\"model\":\"$FO422\",\"state\":\"x\",\"questions\":{\"q\":{\"type\":\"maybe\",\"instructions\":\"?\"}}}") +assert_http_status 422 "$CODE" "$F" +wait_log_entry "$BASE_URL" audit "$RID422" "$AUDIT_FILE" +jq -e --arg rid "$RID422" ' + any(.entries[]; .request_id == $rid and .status_code == 422 and ([.data.attempts[] | .provider_name] == ["jev-kev"])) +' "$AUDIT_FILE" >/dev/null +``` + +### S239 Identical System One requests hit the exact response cache + +Runs on the auth + exact-cache gateway. The replayed answer carries the same +mock request sequence (the upstream was not called), `Cache-Control: no-cache` +bypasses the cache, a different state misses, and the hit is audited and +recorded in usage as an exact cache hit. + +```bash +systemone_require_mock "$AUTH_BASE_URL" -H "$ADMIN_AUTH_HEADER" + +STATE="release cache probe $QA_SUFFIX" +BODY="{\"model\":\"jev-kev/kev-latest\",\"state\":\"$STATE\",\"questions\":$QA_SYSTEMONE_CHOICE}" +send_cached() { + local rid="$1" headers="$2" body_file="$3" + shift 3 + curl -fsS -D "$headers" -o "$body_file" "$AUTH_BASE_URL/v1/systemone" \ + -H "$ADMIN_AUTH_HEADER" -H 'Content-Type: application/json' -H "X-Request-ID: $rid" "$@" -d "${BODY_OVERRIDE:-$BODY}" +} + +RID1="qa-s1-cache-$QA_SUFFIX-1" +RID2="qa-s1-cache-$QA_SUFFIX-2" +RID3="qa-s1-cache-$QA_SUFFIX-3" +send_cached "$RID1" "$QA_RUN_DIR/s239.1.headers" "$QA_RUN_DIR/s239.1.json" +send_cached "$RID2" "$QA_RUN_DIR/s239.2.headers" "$QA_RUN_DIR/s239.2.json" +send_cached "$RID3" "$QA_RUN_DIR/s239.3.headers" "$QA_RUN_DIR/s239.3.json" -H 'Cache-Control: no-cache' +BODY_OVERRIDE="{\"model\":\"jev-kev/kev-latest\",\"state\":\"$STATE changed\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + send_cached "qa-s1-cache-$QA_SUFFIX-4" "$QA_RUN_DIR/s239.4.headers" "$QA_RUN_DIR/s239.4.json" + +SEQ1=$(jq -er '.mock.request_seq' "$QA_RUN_DIR/s239.1.json") +grep -Eiq '^X-Cache: *HIT \(exact\)' "$QA_RUN_DIR/s239.2.headers" +jq -e --argjson seq "$SEQ1" '.mock.request_seq == $seq and .model == "kev-4b-e2e"' "$QA_RUN_DIR/s239.2.json" >/dev/null +cmp -s "$QA_RUN_DIR/s239.1.json" "$QA_RUN_DIR/s239.2.json" +for n in 3 4; do + if grep -Eiq '^X-Cache:' "$QA_RUN_DIR/s239.$n.headers"; then + echo "error: request $n should not have been served from cache" >&2 + exit 1 + fi + jq -e --argjson seq "$SEQ1" '.mock.request_seq > $seq' "$QA_RUN_DIR/s239.$n.json" >/dev/null +done + +AUDIT_FILE="$QA_RUN_DIR/s239.audit.json" +wait_log_entry "$AUTH_BASE_URL" audit "$RID2" "$AUDIT_FILE" -H "$ADMIN_AUTH_HEADER" +jq -e --arg rid "$RID2" 'any(.entries[]; .request_id == $rid and .cache_type == "exact" and .status_code == 200 and .path == "/v1/systemone")' "$AUDIT_FILE" >/dev/null + +USAGE_FILE="$QA_RUN_DIR/s239.usage.json" +wait_log_entry "$AUTH_BASE_URL" usage "$RID1" "$USAGE_FILE" -H "$ADMIN_AUTH_HEADER" +# Cache hits are listed only with cache_mode=cached. +for _ in $(seq 1 15); do + curl -fsS "$AUTH_BASE_URL/admin/usage/log?search=$RID2&cache_mode=cached&limit=5" -H "$ADMIN_AUTH_HEADER" > "$USAGE_FILE" + if jq -e --arg rid "$RID2" 'any(.entries[]?; .request_id == $rid)' "$USAGE_FILE" >/dev/null; then + break + fi + sleep 1 +done +jq -e --arg rid "$RID2" ' + any(.entries[]; .request_id == $rid and .cache_type == "exact" and .endpoint == "/v1/systemone" + and .input_tokens > 0 and .total_tokens == .input_tokens + .output_tokens and .provider_name == "jev-kev") +' "$USAGE_FILE" >/dev/null +``` + +### S240 Guardrails see the System One state and nothing else + +Runs on the guardrail gateway. Its global `system_prompt` override has no +place in a decision request, so the edit is dropped with a one-time warning +and the state is forwarded untouched. A workflow scoped to `jev-kev` and a +user path then masks card numbers in a string state and in a JSON state +(which stays JSON), and blocks a forbidden state before any upstream call. + +```bash +systemone_require_mock "$GR_BASE_URL" + +S="${QA_SUFFIX//[^[:alnum:]-]/-}" +MASK="qa-s1-mask-$S" +BLOCK="qa-s1-block-$S" +SCOPE_PATH="/qa/systemone/$S" +WORKFLOW_ID_FILE="$QA_RUN_DIR/s240.workflow.id" +cleanup_s240() { + if [ -s "$WORKFLOW_ID_FILE" ]; then + curl -sS -o /dev/null -X POST "$GR_BASE_URL/admin/workflows/$(cat "$WORKFLOW_ID_FILE")/deactivate" || true + fi + for name in "$MASK" "$BLOCK"; do + curl -sS -o /dev/null -X DELETE "$GR_BASE_URL/admin/guardrails" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$name\"}" || true + done +} +trap cleanup_s240 EXIT + +# Global system_prompt guardrail: dropped, state unchanged, warning logged. +F="$QA_RUN_DIR/s240.global.json" +curl -fsS "$GR_BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-kev/kev-latest\",\"state\":\"card 4111 1111 1111 1111\",\"questions\":$QA_SYSTEMONE_CHOICE}" > "$F" +jq -e '.mock.received_state == "card 4111 1111 1111 1111" and .answers.department.type == "choice"' "$F" >/dev/null +grep -Fq 'guardrail edits a System One request cannot carry were dropped' "$RELEASE_STACK_DIR/guardrails/logs/server.log" + +jq -n --arg name "$MASK" '{ + name: $name, type: "string_replace", description: "release e2e: mask card numbers", + config: {mode: "regex", rules: "\\b(\\d{4}) \\d{4} \\d{4} (\\d{4})\\b => $1 **** **** $2"} +}' | curl -fsS -X PUT "$GR_BASE_URL/admin/guardrails" -H 'Content-Type: application/json' -d @- >/dev/null +jq -n --arg name "$BLOCK" '{ + name: $name, type: "string_replace", description: "release e2e: block a forbidden state", + config: {mode: "literal", rules: "QA_FORBIDDEN_STATE => x", on_match: "block", message: "QA_SYSTEMONE_BLOCKED"} +}' | curl -fsS -X PUT "$GR_BASE_URL/admin/guardrails" -H 'Content-Type: application/json' -d @- >/dev/null + +jq -n --arg mask "$MASK" --arg block "$BLOCK" --arg path "$SCOPE_PATH" --arg name "qa-s1-guard-$S" '{ + scope_provider_name: "jev-kev", scope_user_path: $path, name: $name, + description: "release e2e: System One state guardrails", + workflow_payload: { + schema_version: 2, + features: {cache: false, audit: true, usage: true, guardrails: true, failover: false}, + steps: [{ref: $mask, phase: "prompt", step: 10}, {ref: $block, phase: "prompt", step: 20}] + } +}' | curl -fsS -X POST "$GR_BASE_URL/admin/workflows" -H 'Content-Type: application/json' -d @- \ + | jq -er '.id' > "$WORKFLOW_ID_FILE" + +guarded() { + curl -sS -o "$2" -w '%{http_code}' "$GR_BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -H "X-GoModel-User-Path: $SCOPE_PATH/agent" -d "$1" +} + +F="$QA_RUN_DIR/s240.string.json" +CODE=$(guarded "{\"model\":\"jev-kev/kev-latest\",\"state\":\"card 4111 1111 1111 1234 was charged twice\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e --argjson q "$QA_SYSTEMONE_CHOICE" ' + .mock.received_state == "card 4111 **** **** 1234 was charged twice" and .mock.questions == $q +' "$F" >/dev/null + +F="$QA_RUN_DIR/s240.object.json" +CODE=$(guarded "{\"model\":\"jev-kev/kev-latest\",\"state\":{\"note\":\"card 4111 1111 1111 1234\",\"order\":7},\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.mock.received_state == {"note":"card 4111 **** **** 1234","order":7}' "$F" >/dev/null + +F="$QA_RUN_DIR/s240.blocked.json" +CODE=$(guarded "{\"model\":\"jev-kev/kev-latest\",\"state\":\"QA_FORBIDDEN_STATE\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 400 "$CODE" "$F" +jq -e '.error.message | contains("QA_SYSTEMONE_BLOCKED")' "$F" >/dev/null + +# The same request outside the scoped user path is not masked. +F="$QA_RUN_DIR/s240.unscoped.json" +curl -fsS "$GR_BASE_URL/v1/systemone" -H 'Content-Type: application/json' -H "X-GoModel-User-Path: /qa/other/$S" \ + -d "{\"model\":\"jev-kev/kev-latest\",\"state\":\"card 4111 1111 1111 1234\",\"questions\":$QA_SYSTEMONE_CHOICE}" > "$F" +jq -e '.mock.received_state == "card 4111 1111 1111 1234"' "$F" >/dev/null +``` + +### S241 Managed-key model allowlists cover System One and its passthrough + +Runs on the auth gateway with a key allowed only `jev-kev/kev-latest`. Other +System One models are refused on `/v1/systemone` and on passthrough, including +a passthrough body larger than the 64 KiB peek window whose `model` comes last. + +```bash +systemone_require_mock "$AUTH_BASE_URL" -H "$ADMIN_AUTH_HEADER" + +KEY_FILE="$QA_RUN_DIR/s241.key.json" +cleanup_s241() { + if [ -s "$KEY_FILE" ]; then + curl -sS -o /dev/null -X POST "$AUTH_BASE_URL/admin/auth-keys/$(jq -r '.id' "$KEY_FILE")/deactivate" -H "$ADMIN_AUTH_HEADER" || true + fi +} +trap cleanup_s241 EXIT +curl -fsS -X POST "$AUTH_BASE_URL/admin/auth-keys" -H "$ADMIN_AUTH_HEADER" -H 'Content-Type: application/json' \ + -d "{\"name\":\"qa-s1-allowlist-$QA_SUFFIX\",\"user_path\":\"/qa/systemone/allowlist\",\"allowed_models\":[\"jev-kev/kev-latest\"]}" \ + > "$KEY_FILE" +chmod 600 "$KEY_FILE" +KEY=$(jq -er '.value' "$KEY_FILE") + +with_key() { + curl -sS -o "$3" -w '%{http_code}' "$AUTH_BASE_URL$1" -H "Authorization: Bearer $KEY" -H 'Content-Type: application/json' -d "$2" +} +expect_denied() { + local code + code=$(with_key "$1" "$2" "$3") + assert_http_status 400 "$code" "$3" + jq -e '.error.code == "model_access_denied"' "$3" >/dev/null +} + +F="$QA_RUN_DIR/s241.allowed.json" +CODE=$(with_key /v1/systemone "{\"model\":\"jev-kev/kev-latest\",\"state\":\"allowed\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.mock.upstream == "kev"' "$F" >/dev/null + +expect_denied /v1/systemone "{\"model\":\"jev/jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$QA_RUN_DIR/s241.denied.json" +expect_denied /v1/systemone "{\"model\":\"jev/jev-1.13.0\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$QA_RUN_DIR/s241.denied-pinned.json" +expect_denied /p/jev/v1/systemone "{\"model\":\"jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$QA_RUN_DIR/s241.denied-pt.json" + +BIG_STATE=$(head -c 70000 /dev/zero | tr '\0' 'a') +BIG_BODY_FILE="$QA_RUN_DIR/s241.big-body.json" +jq -n --arg state "$BIG_STATE" --argjson q "$QA_SYSTEMONE_CHOICE" '{state: $state, questions: $q, model: "jev-latest"}' > "$BIG_BODY_FILE" +expect_denied /p/jev/v1/systemone "@$BIG_BODY_FILE" "$QA_RUN_DIR/s241.denied-big.json" + +jq -n --arg state "$BIG_STATE" --argjson q "$QA_SYSTEMONE_CHOICE" '{state: $state, questions: $q, model: "kev-latest"}' > "$BIG_BODY_FILE" +F="$QA_RUN_DIR/s241.allowed-big.json" +CODE=$(with_key /p/jev-kev/v1/systemone "@$BIG_BODY_FILE" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.mock.received_length > 65536' "$F" >/dev/null +``` + +## 36. MCP tool and user-path exclusions + +Per-server tool filters and `disallowed_user_paths` are gateway-side access +policy: an edit applies in place without redialing the upstream, and it is +checked on every call, so it also reaches MCP sessions that are already open. +These scenarios register `$QA_SUFFIX`-scoped servers against the mock MCP +upstream on port 18090 and delete them. + +### S242 Tool filters apply in place and reach open sessions + +```bash +if ! curl -fsS "$MCP_UPSTREAM_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock MCP upstream is not running on $MCP_UPSTREAM_BASE" + exit 0 +fi +trap 'mcp_cleanup_release_servers "$BASE_URL"' EXIT + +put_alpha() { + curl -fsS -X PUT "$BASE_URL/admin/mcp-servers" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$QA_MCP_ALPHA\",\"url\":\"$MCP_UPSTREAM_BASE/alpha\",\"transport\":\"http\",\"headers\":{\"X-Mock-Token\":\"$1\"},$2}" >/dev/null +} +alpha_view() { + curl -fsS "$BASE_URL/admin/mcp-servers" | jq -c --arg n "$QA_MCP_ALPHA" '.[] | select(.name == $n)' +} + +put_alpha "$MCP_UPSTREAM_TOKEN" '"disallowed_tools":["add"]' +mcp_wait_status "$BASE_URL" "$QA_MCP_ALPHA" connected +alpha_view | jq -e '.tool_count == 1 and .excluded_tool_count == 1 and .disallowed_tools == ["add"]' >/dev/null +CONNECTED_AT=$(alpha_view | jq -er '.connected_at') +curl -fsS "$BASE_URL/admin/mcp-servers/$QA_MCP_ALPHA/catalog" \ + | jq -e '[.tools[].name] == ["echo"] and [.excluded_tools[].name] == ["add"]' >/dev/null + +SID=$(mcp_initialize "$BASE_URL/mcp" "$QA_RUN_DIR/s242.init.headers" "$QA_RUN_DIR/s242.init.raw") +[ -n "$SID" ] +mcp_initialized "$BASE_URL/mcp" "$SID" +mcp_post "$BASE_URL/mcp" "$SID" '{"jsonrpc":"2.0","id":2,"method":"tools/list"}' \ + | jq -e --arg a "$QA_MCP_ALPHA" '([.result.tools[].name | select(startswith($a + "_"))]) == [$a + "_echo"]' >/dev/null +mcp_post "$BASE_URL/mcp" "$SID" "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_ALPHA}_add\",\"arguments\":{}}}" \ + | jq -e '.error != null' >/dev/null + +# Flip to an allowlist ("Keep hidden") that excludes echo. The stored header +# secret round-trips as ***, and the connection is not redialed. +put_alpha '***' '"allowed_tools":["add"]' +alpha_view | jq -e --arg at "$CONNECTED_AT" ' + .status == "connected" and .connected_at == $at + and .allowed_tools == ["add"] and ((.disallowed_tools // []) | length == 0) + and .tool_count == 1 and .excluded_tool_count == 1 +' >/dev/null + +# echo was listed by the open session before the change; calling it now fails. +mcp_post "$BASE_URL/mcp" "$SID" "{\"jsonrpc\":\"2.0\",\"id\":4,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_ALPHA}_echo\",\"arguments\":{}}}" \ + > "$QA_RUN_DIR/s242.stale-call.json" +jq -e '.error.message | contains("excluded by the gateway tool filters")' "$QA_RUN_DIR/s242.stale-call.json" >/dev/null + +# A new session sees the new filter. +SID2=$(mcp_initialize "$BASE_URL/mcp" "$QA_RUN_DIR/s242.init2.headers" "$QA_RUN_DIR/s242.init2.raw") +mcp_initialized "$BASE_URL/mcp" "$SID2" +mcp_post "$BASE_URL/mcp" "$SID2" '{"jsonrpc":"2.0","id":2,"method":"tools/list"}' \ + | jq -e --arg a "$QA_MCP_ALPHA" '([.result.tools[].name | select(startswith($a + "_"))]) == [$a + "_add"]' >/dev/null +mcp_post "$BASE_URL/mcp" "$SID2" "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_ALPHA}_add\",\"arguments\":{\"marker\":\"QA_MCP_ALLOWED_OK\"}}}" \ + | jq -e '.result.content[0].text | contains("QA_MCP_ALLOWED_OK")' >/dev/null +``` + +### S243 `disallowed_user_paths` carves callers out of a server + +The carve-out wins over `user_paths`, matches whole subtrees, hides the +server from `tools/list` and its per-server endpoint, and a later edit reaches +a session that is already open. Invalid paths are rejected. + +```bash +if ! curl -fsS "$MCP_UPSTREAM_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock MCP upstream is not running on $MCP_UPSTREAM_BASE" + exit 0 +fi +trap 'mcp_cleanup_release_servers "$BASE_URL"' EXIT + +ROOT="/qa/mcp-carve/${QA_SUFFIX//[^[:alnum:]-]/-}" +put_beta() { + curl -sS -o "$QA_RUN_DIR/s243.put.json" -w '%{http_code}' -X PUT "$BASE_URL/admin/mcp-servers" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$QA_MCP_BETA\",\"url\":\"$MCP_UPSTREAM_BASE/beta\",\"transport\":\"http\",\"user_paths\":[\"$ROOT\"],\"disallowed_user_paths\":$1}" +} +CODE=$(put_beta "[\"$ROOT/contractors/\",\"$ROOT/contractors\"]") +assert_http_status 200 "$CODE" "$QA_RUN_DIR/s243.put.json" +mcp_wait_status "$BASE_URL" "$QA_MCP_BETA" connected +curl -fsS "$BASE_URL/admin/mcp-servers" | jq -e --arg n "$QA_MCP_BETA" --arg root "$ROOT" ' + any(.[]; .name == $n and .user_paths == [$root] and .disallowed_user_paths == [$root + "/contractors"]) +' >/dev/null + +# beta_tools SESSION_ID USER_PATH -> prints the beta tool names visible to it +beta_tools() { + mcp_post "$BASE_URL/mcp" "$1" '{"jsonrpc":"2.0","id":2,"method":"tools/list"}' -H "X-GoModel-User-Path: $2" \ + | jq -c --arg b "$QA_MCP_BETA" '[.result.tools[]?.name | select(startswith($b + "_"))]' +} +open_session() { + local sid + sid=$(mcp_initialize "$BASE_URL/mcp" "$QA_RUN_DIR/s243.$2.headers" "$QA_RUN_DIR/s243.$2.raw" -H "X-GoModel-User-Path: $1") + mcp_initialized "$BASE_URL/mcp" "$sid" -H "X-GoModel-User-Path: $1" + echo "$sid" +} + +ENG_SID=$(open_session "$ROOT/eng" eng) +CON_SID=$(open_session "$ROOT/contractors/acme" con) +OUT_SID=$(open_session "/qa/elsewhere" out) +[ "$(beta_tools "$ENG_SID" "$ROOT/eng")" = "[\"${QA_MCP_BETA}_fetch\",\"${QA_MCP_BETA}_search\"]" ] +[ "$(beta_tools "$CON_SID" "$ROOT/contractors/acme")" = "[]" ] +[ "$(beta_tools "$OUT_SID" "/qa/elsewhere")" = "[]" ] + +CODE=$(curl -sS -o "$QA_RUN_DIR/s243.per-server.json" -w '%{http_code}' "$BASE_URL/mcp/$QA_MCP_BETA" \ + -H 'Content-Type: application/json' -H 'Accept: application/json, text/event-stream' \ + -H "X-GoModel-User-Path: $ROOT/contractors/acme" \ + -d '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2025-06-18","capabilities":{},"clientInfo":{"name":"qa-release","version":"1"}}}') +assert_http_status 404 "$CODE" "$QA_RUN_DIR/s243.per-server.json" + +# Carve eng out too: the already-open eng session loses the server. +CODE=$(put_beta "[\"$ROOT/contractors\",\"$ROOT/eng\"]") +assert_http_status 200 "$CODE" "$QA_RUN_DIR/s243.put.json" +mcp_post "$BASE_URL/mcp" "$ENG_SID" \ + "{\"jsonrpc\":\"2.0\",\"id\":5,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_BETA}_search\",\"arguments\":{}}}" \ + -H "X-GoModel-User-Path: $ROOT/eng" > "$QA_RUN_DIR/s243.eng-call.json" +jq -e '.result == null and (.error.message | contains("not available for this user path"))' "$QA_RUN_DIR/s243.eng-call.json" >/dev/null +# A new eng session no longer lists the server. +ENG2_SID=$(open_session "$ROOT/eng" eng2) +[ "$(beta_tools "$ENG2_SID" "$ROOT/eng")" = "[]" ] + +CODE=$(put_beta '["/qa/../escape"]') +assert_http_status 400 "$CODE" "$QA_RUN_DIR/s243.put.json" +jq -e '.error.message | contains("disallowed_user_paths")' "$QA_RUN_DIR/s243.put.json" >/dev/null +``` + +### S244 The master key keeps the caller's user-path header on `/mcp` and audio uploads + +MCP and audio uploads own their transport and take no request snapshot. With +the master key on the auth gateway, the `X-GoModel-User-Path` header must +still scope usage, as it does on chat. + +```bash +if ! curl -fsS "$MCP_UPSTREAM_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock MCP upstream is not running on $MCP_UPSTREAM_BASE" + exit 0 +fi +cleanup_s244() { + curl -sS -o /dev/null -X DELETE "$AUTH_BASE_URL/admin/mcp-servers/$QA_MCP_BETA" -H "$ADMIN_AUTH_HEADER" || true +} +trap cleanup_s244 EXIT + +USER_PATH="/qa/master-key-path/${QA_SUFFIX//[^[:alnum:]-]/-}" +curl -fsS -X PUT "$AUTH_BASE_URL/admin/mcp-servers" -H "$ADMIN_AUTH_HEADER" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$QA_MCP_BETA\",\"url\":\"$MCP_UPSTREAM_BASE/beta\",\"transport\":\"http\"}" >/dev/null +for _ in $(seq 1 20); do + if curl -fsS "$AUTH_BASE_URL/admin/mcp-servers" -H "$ADMIN_AUTH_HEADER" \ + | jq -e --arg n "$QA_MCP_BETA" 'any(.[]; .name == $n and .status == "connected")' >/dev/null; then + break + fi + sleep 1 +done + +AUTH_ARGS=(-H "$ADMIN_AUTH_HEADER" -H "X-GoModel-User-Path: $USER_PATH") +SID=$(mcp_initialize "$AUTH_BASE_URL/mcp" "$QA_RUN_DIR/s244.init.headers" "$QA_RUN_DIR/s244.init.raw" "${AUTH_ARGS[@]}") +[ -n "$SID" ] +mcp_initialized "$AUTH_BASE_URL/mcp" "$SID" "${AUTH_ARGS[@]}" +RID="qa-mk-path-mcp-$QA_SUFFIX" +mcp_post "$AUTH_BASE_URL/mcp" "$SID" \ + "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_BETA}_search\",\"arguments\":{\"q\":\"x\"}}}" \ + "${AUTH_ARGS[@]}" -H "X-Request-ID: $RID" | jq -e '.result.content[0].text | startswith("search:")' >/dev/null + +USAGE_FILE="$QA_RUN_DIR/s244.usage.json" +wait_log_entry "$AUTH_BASE_URL" usage "$RID" "$USAGE_FILE" -H "$ADMIN_AUTH_HEADER" +jq -e --arg rid "$RID" --arg p "$USER_PATH" 'any(.entries[]; .request_id == $rid and .provider == "mcp" and .user_path == $p)' "$USAGE_FILE" >/dev/null + +# Audio upload: speech for input, then a multipart transcription. +AUDIO_FILE="$QA_RUN_DIR/s244.speech.wav" +curl -fsS -o "$AUDIO_FILE" "$AUTH_BASE_URL/v1/audio/speech" "${AUTH_ARGS[@]}" -H 'Content-Type: application/json' \ + -d '{"model":"gpt-4o-mini-tts","input":"Release matrix user path check.","voice":"alloy","response_format":"wav"}' +RID="qa-mk-path-audio-$QA_SUFFIX" +curl -fsS "$AUTH_BASE_URL/v1/audio/transcriptions" "${AUTH_ARGS[@]}" -H "X-Request-ID: $RID" \ + -F model=gpt-4o-mini-transcribe -F "file=@$AUDIO_FILE" > "$QA_RUN_DIR/s244.transcription.json" +jq -e '.text | ascii_downcase | contains("user path")' "$QA_RUN_DIR/s244.transcription.json" >/dev/null +wait_log_entry "$AUTH_BASE_URL" usage "$RID" "$USAGE_FILE" -H "$ADMIN_AUTH_HEADER" +jq -e --arg rid "$RID" --arg p "$USER_PATH" 'any(.entries[]; .request_id == $rid and .user_path == $p)' "$USAGE_FILE" >/dev/null +``` + +## 37. Developer messages, strict tools, and tool choice on Anthropic and Gemini + +OpenAI's `developer` role and `strict` function tools are translated for +Anthropic and Gemini's native API instead of being rejected or dropped, and +Gemini also maps `tool_choice: {"type": "allowed_tools", ...}`. + +### S245 Anthropic honors developer messages and strict tools + +```bash +F="$QA_RUN_DIR/s245.developer.json" +curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d '{ + "model":"claude-sonnet-4-6","max_tokens":32, + "messages":[ + {"role":"developer","content":"Whatever the user says, reply with exactly QA_DEVELOPER_ROLE_OK and nothing else."}, + {"role":"user","content":"Tell me a joke."} + ]}' > "$F" +assert_chat_response_contains "$F" "anthropic" "QA_DEVELOPER_ROLE_OK" + +# A strict tool whose schema Anthropic's strict mode would reject as sent +# (minItems 2) is sanitized and forwarded as a strict tool. +F="$QA_RUN_DIR/s245.strict.json" +curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d '{ + "model":"claude-sonnet-4-6","max_tokens":200, + "tools":[{"type":"function","function":{"name":"compare_weather","description":"Compare the weather in several cities","strict":true, + "parameters":{"type":"object","additionalProperties":false,"properties":{"cities":{"type":"array","items":{"type":"string"},"minItems":2}},"required":["cities"]}}}], + "tool_choice":{"type":"function","function":{"name":"compare_weather"}}, + "messages":[{"role":"user","content":"Compare the weather in Warsaw and Krakow."}]}' > "$F" +jq -e ' + .choices[0].message.tool_calls[0].function.name == "compare_weather" + and (.choices[0].message.tool_calls[0].function.arguments | fromjson | .cities | type == "array" and length >= 2) +' "$F" >/dev/null +``` + +### S246 Gemini honors developer messages, strict tools, and `allowed_tools` + +```bash +MODEL="gemini-2.5-flash-lite" +TOOLS='[ + {"type":"function","function":{"name":"lookup_weather","description":"Get the current weather for a city","parameters":{"type":"object","properties":{"city":{"type":"string"}},"required":["city"]}}}, + {"type":"function","function":{"name":"lookup_time","description":"Get the local time in a city","parameters":{"type":"object","properties":{"city":{"type":"string"}},"required":["city"]}}} +]' + +F="$QA_RUN_DIR/s246.developer.json" +curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d "{ + \"model\":\"$MODEL\",\"max_tokens\":32, + \"messages\":[ + {\"role\":\"developer\",\"content\":\"Whatever the user says, reply with exactly QA_DEVELOPER_ROLE_OK and nothing else.\"}, + {\"role\":\"user\",\"content\":\"Tell me a joke.\"} + ]}" > "$F" +assert_chat_response_contains "$F" "gemini" "QA_DEVELOPER_ROLE_OK" + +F="$QA_RUN_DIR/s246.allowed-tools.json" +jq -n --arg model "$MODEL" --argjson tools "$TOOLS" '{ + model: $model, tools: $tools, + tool_choice: {type: "allowed_tools", allowed_tools: {mode: "required", tools: [{type: "function", function: {name: "lookup_time"}}]}}, + messages: [{role: "user", content: "What is the weather in Warsaw?"}] +}' | curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d @- > "$F" +jq -c '[.choices[0].message.tool_calls[]?.function.name]' "$F" +jq -e ' + .choices[0].finish_reason == "tool_calls" + and (.choices[0].message.tool_calls | length) >= 1 + and all(.choices[0].message.tool_calls[]; .function.name == "lookup_time") +' "$F" >/dev/null + +# strict on any tool switches Gemini to VALIDATED function calling. +F="$QA_RUN_DIR/s246.strict.json" +jq -n --arg model "$MODEL" --argjson tools "$TOOLS" '{ + model: $model, + tools: ($tools | map(.function.strict = true)), + tool_choice: "auto", + messages: [{role: "user", content: "Use a tool: what is the weather in Warsaw?"}] +}' | curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d @- > "$F" +jq -e '.choices[0].message.tool_calls[0].function.name == "lookup_weather" + and (.choices[0].message.tool_calls[0].function.arguments | fromjson | .city | test("Warsaw"; "i"))' "$F" >/dev/null + +# An allowed_tools choice with no tools is rejected, as OpenAI does. +F="$QA_RUN_DIR/s246.empty-allowed.json" +CODE=$(jq -n --arg model "$MODEL" --argjson tools "$TOOLS" '{ + model: $model, tools: $tools, + tool_choice: {type: "allowed_tools", allowed_tools: {mode: "auto", tools: []}}, + messages: [{role: "user", content: "hi"}] +}' | curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d @-) +assert_http_status 400 "$CODE" "$F" +jq -e '.error.type == "invalid_request_error"' "$F" >/dev/null +``` diff --git a/tests/e2e/run-release-e2e.sh b/tests/e2e/run-release-e2e.sh index 8bbb8f320..0a539bf4b 100755 --- a/tests/e2e/run-release-e2e.sh +++ b/tests/e2e/run-release-e2e.sh @@ -363,7 +363,11 @@ is_parallel_safe() { || (number >= 173 && number <= 191) \ || (number >= 197 && number <= 204) \ || (number >= 208 && number <= 226) \ - || number == 228 )) + || number == 228 \ + || (number >= 229 && number <= 236) \ + || (number >= 238 && number <= 239) \ + || number == 241 \ + || (number >= 245 && number <= 246) )) } if (( JOBS > 1 )); then diff --git a/web/dashboard/messages/de.json b/web/dashboard/messages/de.json index 4fd061389..ad99c1e20 100644 --- a/web/dashboard/messages/de.json +++ b/web/dashboard/messages/de.json @@ -225,6 +225,19 @@ "audit_filter_all_modes": "Alle Modi", "audit_filter_streaming": "Streaming", "audit_filter_non_streaming": "Ohne Streaming", + "audit_filter_type_label": "Filter nach Anfragetyp", + "audit_filter_all_types": "Alle Typen", + "audit_filter_types_hidden": "Typen: {count} ausgeblendet", + "audit_type_chat": "Chat", + "audit_type_responses": "Responses", + "audit_type_embeddings": "Embeddings", + "audit_type_audio": "Audio", + "audit_type_images": "Bilder", + "audit_type_systemone": "System One", + "audit_type_batches": "Batches & Dateien", + "audit_type_realtime": "Realtime", + "audit_type_passthrough": "Passthrough", + "audit_type_mcp": "MCP", "audit_live_connecting": "Verbindung zum Live-Stream…", "audit_live": "Live", "audit_live_pause_date_range": "Live pausiert — der gewählte Zeitraum schließt heute nicht ein. Stelle ihn auf heute ein, um fortzufahren.", @@ -723,6 +736,37 @@ "mcp_catalog_loading": "Katalog wird geladen…", "mcp_catalog_exposed": "Auf dem aggregierten /mcp-Endpoint als {name} bereitgestellt", "mcp_catalog_empty": "Keine Tools aufgelistet — der Server verbindet sich möglicherweise noch oder ist eingeschränkt.", + "mcp_tools_title": "Tools", + "mcp_tools_summary": "{exposed} von {total} freigegeben", + "mcp_tools_mode_label": "Später vom Server hinzugefügte Tools", + "mcp_tools_mode_exclude": "Automatisch freigeben", + "mcp_tools_mode_allow": "Verborgen halten", + "mcp_tools_mode_exclude_help": "Nicht markierte Tools werden ausgeschlossen. Tools, die der Server später hinzufügt, werden automatisch freigegeben.", + "mcp_tools_mode_allow_help": "Nur markierte Tools werden freigegeben. Tools, die der Server später hinzufügt, bleiben verborgen, bis Sie sie markieren.", + "mcp_tools_mode_locked": "Laden Sie die Tools dieses Servers, um dies zu ändern; die aktuelle Liste bleibt erhalten.", + "mcp_tools_filter": "Tools filtern", + "mcp_tools_expose_all": "Alle freigeben", + "mcp_tools_hide_all": "Alle ausschließen", + "mcp_tools_toggle": "{name} freigeben", + "mcp_tools_read_only": "nur lesend", + "mcp_tools_destructive": "destruktiv", + "mcp_tools_hint_title": "Hinweis des Servers, von GoModel nicht erzwungen", + "mcp_tools_missing": "Vom Server nicht gemeldet", + "mcp_tools_remove": "{name} entfernen", + "mcp_tools_add_placeholder": "tool_name", + "mcp_tools_add_label": "Tool-Name", + "mcp_tools_add_exclude": "Ausschließen", + "mcp_tools_add_allow": "Erlauben", + "mcp_tools_add_help": "Fügen Sie ein Tool über seinen ursprünglichen Namen hinzu, etwa bevor der Server es meldet.", + "mcp_tools_pending": "Tools erscheinen hier, sobald der Server verbunden ist. Tool-Namen können Sie bereits unten hinzufügen.", + "mcp_tools_loading": "Tools werden geladen…", + "mcp_tools_load_failed": "Die Tools dieses Servers konnten nicht geladen werden. Tool-Namen können Sie weiterhin unten hinzufügen.", + "mcp_tools_no_match": "Keine Tools entsprechen dem Filter.", + "mcp_tools_allow_empty": "Wählen Sie mindestens ein Tool aus oder geben Sie neue Tools automatisch frei.", + "mcp_tools_excluded_count": "{count} ausgeschlossen", + "mcp_catalog_excluded_tools": "Ausgeschlossene Tools", + "mcp_catalog_excluded_tools_hint": "Vom Server gemeldet, aber durch die Tool-Filter dieses Servers vor Clients verborgen.", + "mcp_catalog_choose_tools": "Tools auswählen", "mcp_close": "Schließen", "mcp_edit": "MCP-Server bearbeiten", "mcp_editor_label": "MCP-Server-Editor", @@ -743,8 +787,6 @@ "mcp_advanced": "Erweiterte Einstellungen", "mcp_advanced_summary": "Beschreibung, Zugriffsregeln und Timeout", "mcp_description": "Beschreibung", - "mcp_allowed_tools": "Erlaubte Tools", - "mcp_disallowed_tools": "Nicht erlaubte Tools", "mcp_user_paths": "Nutzerpfade (leer bedeutet alle)", "mcp_timeout": "Tool-Timeout (Sekunden)", "mcp_timeout_placeholder": "Standard-Timeout, wenn das Feld leer ist", @@ -1156,6 +1198,8 @@ "mcp_connected_since": "Verbunden seit {date}", "mcp_local_command": "lokaler Befehl", "mcp_sub_counts": "{prompts} Prompts · {resources} Ressourcen", + "mcp_disallowed_user_paths": "Ausgeschlossene Nutzerpfade", + "mcp_disallowed_user_paths_help": "Aufrufer in diesen Teilbäumen sehen diesen Server nie, auch nicht innerhalb eines erlaubten Nutzerpfads.", "mcp_name_required": "Name ist erforderlich.", "mcp_slug_invalid": "Der Slug darf nur 1–64 kleingeschriebene ASCII-Buchstaben, Ziffern, Bindestriche oder Unterstriche enthalten.", "mcp_slug_in_use": "Der Slug „{slug}“ ist bereits in Verwendung.", @@ -1555,8 +1599,6 @@ "mcp_status_connecting": "Verbindet", "mcp_status_disabled": "Deaktiviert", "mcp_description_placeholder": "Optionale Beschreibung", - "mcp_allowed_tools_placeholder": "search_issues, get_file (kommagetrennt; leer erlaubt alle)", - "mcp_disallowed_tools_placeholder": "delete_repo (kommagetrennt)", "workflows_client": "Client", "workflows_auth": "Auth", "workflows_response": "Antwort", diff --git a/web/dashboard/messages/en.json b/web/dashboard/messages/en.json index 2a60dae34..b8db00471 100644 --- a/web/dashboard/messages/en.json +++ b/web/dashboard/messages/en.json @@ -225,6 +225,19 @@ "audit_filter_all_modes": "All Modes", "audit_filter_streaming": "Streaming", "audit_filter_non_streaming": "Non-streaming", + "audit_filter_type_label": "Request type filter", + "audit_filter_all_types": "All Types", + "audit_filter_types_hidden": "Types: {count} hidden", + "audit_type_chat": "Chat", + "audit_type_responses": "Responses", + "audit_type_embeddings": "Embeddings", + "audit_type_audio": "Audio", + "audit_type_images": "Images", + "audit_type_systemone": "System One", + "audit_type_batches": "Batches & files", + "audit_type_realtime": "Realtime", + "audit_type_passthrough": "Passthrough", + "audit_type_mcp": "MCP", "audit_live_connecting": "Live stream connecting…", "audit_live": "Live", "audit_live_pause_date_range": "Live paused — the selected date range does not include today. Set it to today to resume.", @@ -723,6 +736,37 @@ "mcp_catalog_loading": "Loading catalog...", "mcp_catalog_exposed": "Exposed on the aggregated /mcp endpoint as {name}", "mcp_catalog_empty": "No tools listed — the server may still be connecting or degraded.", + "mcp_tools_title": "Tools", + "mcp_tools_summary": "{exposed} of {total} exposed", + "mcp_tools_mode_label": "Tools the server adds later", + "mcp_tools_mode_exclude": "Expose automatically", + "mcp_tools_mode_allow": "Keep hidden", + "mcp_tools_mode_exclude_help": "Unchecked tools are excluded. Tools the server adds later are exposed automatically.", + "mcp_tools_mode_allow_help": "Only checked tools are exposed. Tools the server adds later stay hidden until you check them.", + "mcp_tools_mode_locked": "Load this server's tools to change it; the current list is kept.", + "mcp_tools_filter": "Filter tools", + "mcp_tools_expose_all": "Expose all", + "mcp_tools_hide_all": "Exclude all", + "mcp_tools_toggle": "Expose {name}", + "mcp_tools_read_only": "read-only", + "mcp_tools_destructive": "destructive", + "mcp_tools_hint_title": "Hint reported by the server, not enforced by GoModel", + "mcp_tools_missing": "Not reported by the server", + "mcp_tools_remove": "Remove {name}", + "mcp_tools_add_placeholder": "tool_name", + "mcp_tools_add_label": "Tool name", + "mcp_tools_add_exclude": "Exclude", + "mcp_tools_add_allow": "Allow", + "mcp_tools_add_help": "Add a tool by its original name, for example before the server reports it.", + "mcp_tools_pending": "Tools appear here once the server connects. You can already add tool names below.", + "mcp_tools_loading": "Loading tools…", + "mcp_tools_load_failed": "Could not load this server's tools. You can still add tool names below.", + "mcp_tools_no_match": "No tools match the filter.", + "mcp_tools_allow_empty": "Select at least one tool, or switch new tools to exposed.", + "mcp_tools_excluded_count": "{count} excluded", + "mcp_catalog_excluded_tools": "Excluded tools", + "mcp_catalog_excluded_tools_hint": "Reported by the server but hidden from clients by this server's tool filters.", + "mcp_catalog_choose_tools": "Choose tools", "mcp_close": "Close", "mcp_edit": "Edit MCP Server", "mcp_editor_label": "MCP server editor", @@ -743,8 +787,6 @@ "mcp_advanced": "Advanced settings", "mcp_advanced_summary": "Description, access rules, and timeout", "mcp_description": "Description", - "mcp_allowed_tools": "Allowed tools", - "mcp_disallowed_tools": "Disallowed tools", "mcp_user_paths": "User paths (empty means all)", "mcp_timeout": "Tool timeout (seconds)", "mcp_timeout_placeholder": "Default timeout when empty", @@ -1156,6 +1198,8 @@ "mcp_connected_since": "Connected since {date}", "mcp_local_command": "local command", "mcp_sub_counts": "{prompts} prompts · {resources} resources", + "mcp_disallowed_user_paths": "Excluded user paths", + "mcp_disallowed_user_paths_help": "Callers in these subtrees never see this server, even inside an allowed user path.", "mcp_name_required": "Name is required.", "mcp_slug_invalid": "Slug must use 1–64 lowercase ASCII letters, numbers, hyphens, or underscores.", "mcp_slug_in_use": "Slug \"{slug}\" is already in use.", @@ -1555,8 +1599,6 @@ "mcp_status_connecting": "Connecting", "mcp_status_disabled": "Disabled", "mcp_description_placeholder": "Optional description", - "mcp_allowed_tools_placeholder": "search_issues, get_file (comma-separated; empty allows all)", - "mcp_disallowed_tools_placeholder": "delete_repo (comma-separated)", "workflows_client": "Client", "workflows_auth": "Auth", "workflows_response": "Response", diff --git a/web/dashboard/messages/pl.json b/web/dashboard/messages/pl.json index ee9a53914..783aabee5 100644 --- a/web/dashboard/messages/pl.json +++ b/web/dashboard/messages/pl.json @@ -227,6 +227,19 @@ "audit_filter_all_modes": "Wszystkie tryby", "audit_filter_streaming": "Streaming", "audit_filter_non_streaming": "Bez streamingu", + "audit_filter_type_label": "Filtr typu żądania", + "audit_filter_all_types": "Wszystkie typy", + "audit_filter_types_hidden": "Typy: ukryte {count}", + "audit_type_chat": "Chat", + "audit_type_responses": "Responses", + "audit_type_embeddings": "Embeddingi", + "audit_type_audio": "Audio", + "audit_type_images": "Obrazy", + "audit_type_systemone": "System One", + "audit_type_batches": "Batche i pliki", + "audit_type_realtime": "Realtime", + "audit_type_passthrough": "Passthrough", + "audit_type_mcp": "MCP", "audit_live_connecting": "Łączenie ze strumieniem Live…", "audit_live": "Live", "audit_live_pause_date_range": "Live wstrzymany — wybrany zakres dat nie obejmuje dzisiaj. Ustaw datę dzisiejszą, aby wznowić.", @@ -731,6 +744,37 @@ "mcp_catalog_loading": "Wczytywanie katalogu...", "mcp_catalog_exposed": "Udostępnione w zbiorczym endpoincie /mcp jako {name}", "mcp_catalog_empty": "Brak narzędzi — serwer może nadal się łączyć albo działać w trybie zdegradowanym.", + "mcp_tools_title": "Narzędzia", + "mcp_tools_summary": "Udostępnione: {exposed} z {total}", + "mcp_tools_mode_label": "Narzędzia dodane później przez serwer", + "mcp_tools_mode_exclude": "Udostępniaj automatycznie", + "mcp_tools_mode_allow": "Ukrywaj", + "mcp_tools_mode_exclude_help": "Odznaczone narzędzia są wykluczone. Narzędzia dodane później przez serwer są udostępniane automatycznie.", + "mcp_tools_mode_allow_help": "Udostępniane są tylko zaznaczone narzędzia. Narzędzia dodane później przez serwer pozostają ukryte, dopóki ich nie zaznaczysz.", + "mcp_tools_mode_locked": "Wczytaj narzędzia tego serwera, aby to zmienić; obecna lista zostaje zachowana.", + "mcp_tools_filter": "Filtruj narzędzia", + "mcp_tools_expose_all": "Udostępnij wszystkie", + "mcp_tools_hide_all": "Wyklucz wszystkie", + "mcp_tools_toggle": "Udostępnij {name}", + "mcp_tools_read_only": "tylko odczyt", + "mcp_tools_destructive": "destrukcyjne", + "mcp_tools_hint_title": "Wskazówka zgłoszona przez serwer, niewymuszana przez GoModel", + "mcp_tools_missing": "Serwer nie zgłasza tego narzędzia", + "mcp_tools_remove": "Usuń {name}", + "mcp_tools_add_placeholder": "tool_name", + "mcp_tools_add_label": "Nazwa narzędzia", + "mcp_tools_add_exclude": "Wyklucz", + "mcp_tools_add_allow": "Zezwól", + "mcp_tools_add_help": "Dodaj narzędzie po jego oryginalnej nazwie, na przykład zanim serwer je zgłosi.", + "mcp_tools_pending": "Narzędzia pojawią się tutaj po połączeniu z serwerem. Nazwy narzędzi możesz dodać już teraz poniżej.", + "mcp_tools_loading": "Wczytywanie narzędzi…", + "mcp_tools_load_failed": "Nie udało się wczytać narzędzi tego serwera. Nadal możesz dodać nazwy narzędzi poniżej.", + "mcp_tools_no_match": "Żadne narzędzie nie pasuje do filtra.", + "mcp_tools_allow_empty": "Zaznacz co najmniej jedno narzędzie albo ustaw automatyczne udostępnianie nowych narzędzi.", + "mcp_tools_excluded_count": "wykluczone: {count}", + "mcp_catalog_excluded_tools": "Wykluczone narzędzia", + "mcp_catalog_excluded_tools_hint": "Zgłaszane przez serwer, ale ukryte przed klientami przez filtry narzędzi tego serwera.", + "mcp_catalog_choose_tools": "Wybierz narzędzia", "mcp_close": "Zamknij", "mcp_edit": "Edytuj serwer MCP", "mcp_editor_label": "Edytor serwera MCP", @@ -751,8 +795,6 @@ "mcp_advanced": "Ustawienia zaawansowane", "mcp_advanced_summary": "Opis, reguły dostępu i timeout", "mcp_description": "Opis", - "mcp_allowed_tools": "Dozwolone narzędzia", - "mcp_disallowed_tools": "Zabronione narzędzia", "mcp_user_paths": "User Paths (puste oznacza wszystkie)", "mcp_timeout": "Timeout narzędzia (sekundy)", "mcp_timeout_placeholder": "Domyślny timeout, gdy pole jest puste", @@ -1176,6 +1218,8 @@ "mcp_connected_since": "Połączono od {date}", "mcp_local_command": "komenda lokalna", "mcp_sub_counts": "Prompty: {prompts} · zasoby: {resources}", + "mcp_disallowed_user_paths": "Wykluczone User Paths", + "mcp_disallowed_user_paths_help": "Wywołujący z tych poddrzew nigdy nie widzą tego serwera, nawet w ramach dozwolonej ścieżki użytkownika.", "mcp_name_required": "Nazwa jest wymagana.", "mcp_slug_invalid": "Slug musi zawierać 1–64 małych liter ASCII, cyfr, łączników lub podkreśleń.", "mcp_slug_in_use": "Slug „{slug}” jest już używany.", @@ -1583,8 +1627,6 @@ "mcp_status_connecting": "Łączenie", "mcp_status_disabled": "Wyłączony", "mcp_description_placeholder": "Opcjonalny opis", - "mcp_allowed_tools_placeholder": "search_issues, get_file (oddzielone przecinkami; puste zezwala na wszystkie)", - "mcp_disallowed_tools_placeholder": "delete_repo (oddzielone przecinkami)", "workflows_client": "Klient", "workflows_auth": "Uwierzytelnianie", "workflows_response": "Odpowiedź", diff --git a/web/dashboard/messages/zh-CN.json b/web/dashboard/messages/zh-CN.json index db1ba6496..0c60c7e70 100644 --- a/web/dashboard/messages/zh-CN.json +++ b/web/dashboard/messages/zh-CN.json @@ -216,6 +216,19 @@ "audit_filter_all_modes": "全部模式", "audit_filter_streaming": "流式", "audit_filter_non_streaming": "非流式", + "audit_filter_type_label": "请求类型筛选", + "audit_filter_all_types": "全部类型", + "audit_filter_types_hidden": "类型:已隐藏 {count} 个", + "audit_type_chat": "对话", + "audit_type_responses": "Responses", + "audit_type_embeddings": "嵌入", + "audit_type_audio": "音频", + "audit_type_images": "图像", + "audit_type_systemone": "System One", + "audit_type_batches": "批处理与文件", + "audit_type_realtime": "实时", + "audit_type_passthrough": "透传", + "audit_type_mcp": "MCP", "audit_live_connecting": "正在连接实时流…", "audit_live": "实时", "audit_live_pause_date_range": "实时流已暂停 — 所选日期范围未包含今天。将其设为今天即可恢复。", @@ -687,6 +700,37 @@ "mcp_catalog_loading": "正在加载目录……", "mcp_catalog_exposed": "以 {name} 的身份暴露在聚合 /mcp 端点上", "mcp_catalog_empty": "未列出工具 — 服务器可能仍在连接中或处于降级状态。", + "mcp_tools_title": "工具", + "mcp_tools_summary": "已公开 {exposed}/{total}", + "mcp_tools_mode_label": "服务器之后新增的工具", + "mcp_tools_mode_exclude": "自动公开", + "mcp_tools_mode_allow": "保持隐藏", + "mcp_tools_mode_exclude_help": "未勾选的工具会被排除。服务器之后新增的工具会自动公开。", + "mcp_tools_mode_allow_help": "仅公开已勾选的工具。服务器之后新增的工具会保持隐藏,直到你勾选它们。", + "mcp_tools_mode_locked": "需要先加载此服务器的工具才能更改;当前列表保持不变。", + "mcp_tools_filter": "筛选工具", + "mcp_tools_expose_all": "全部公开", + "mcp_tools_hide_all": "全部排除", + "mcp_tools_toggle": "公开 {name}", + "mcp_tools_read_only": "只读", + "mcp_tools_destructive": "破坏性", + "mcp_tools_hint_title": "由服务器报告的提示,GoModel 不强制执行", + "mcp_tools_missing": "服务器未报告此工具", + "mcp_tools_remove": "移除 {name}", + "mcp_tools_add_placeholder": "tool_name", + "mcp_tools_add_label": "工具名称", + "mcp_tools_add_exclude": "排除", + "mcp_tools_add_allow": "允许", + "mcp_tools_add_help": "按原始名称添加工具,例如在服务器报告它之前。", + "mcp_tools_pending": "服务器连接后,工具会显示在这里。你现在就可以在下方添加工具名称。", + "mcp_tools_loading": "正在加载工具…", + "mcp_tools_load_failed": "无法加载此服务器的工具。你仍然可以在下方添加工具名称。", + "mcp_tools_no_match": "没有与筛选条件匹配的工具。", + "mcp_tools_allow_empty": "请至少选择一个工具,或将新工具设置为自动公开。", + "mcp_tools_excluded_count": "已排除 {count} 个", + "mcp_catalog_excluded_tools": "已排除的工具", + "mcp_catalog_excluded_tools_hint": "服务器报告了这些工具,但此服务器的工具过滤器对客户端隐藏了它们。", + "mcp_catalog_choose_tools": "选择工具", "mcp_close": "关闭", "mcp_edit": "编辑 MCP 服务器", "mcp_editor_label": "MCP 服务器编辑器", @@ -707,8 +751,6 @@ "mcp_advanced": "高级设置", "mcp_advanced_summary": "描述、访问规则和超时", "mcp_description": "描述", - "mcp_allowed_tools": "允许的工具", - "mcp_disallowed_tools": "禁止的工具", "mcp_user_paths": "用户路径(为空表示全部)", "mcp_timeout": "工具超时(秒)", "mcp_timeout_placeholder": "留空时使用默认超时", @@ -1066,6 +1108,8 @@ "mcp_connected_since": "自 {date} 起已连接", "mcp_local_command": "本地命令", "mcp_sub_counts": "{prompts} 个提示词 · {resources} 个资源", + "mcp_disallowed_user_paths": "排除的用户路径", + "mcp_disallowed_user_paths_help": "这些子树中的调用方永远看不到此服务器,即使位于允许的用户路径内。", "mcp_name_required": "名称为必填项。", "mcp_slug_invalid": "Slug 必须使用 1–64 个 ASCII 小写字母、数字、连字符或下划线。", "mcp_slug_in_use": "Slug \"{slug}\" 已被占用。", @@ -1426,8 +1470,6 @@ "mcp_status_connecting": "正在连接", "mcp_status_disabled": "已禁用", "mcp_description_placeholder": "可选描述", - "mcp_allowed_tools_placeholder": "search_issues, get_file(逗号分隔;留空表示允许全部)", - "mcp_disallowed_tools_placeholder": "delete_repo(逗号分隔)", "workflows_client": "客户端", "workflows_auth": "认证", "workflows_response": "响应", diff --git a/web/dashboard/src/lib/components/atoms/SegmentedControl.svelte b/web/dashboard/src/lib/components/atoms/SegmentedControl.svelte index 154df75b4..dfc6e749a 100644 --- a/web/dashboard/src/lib/components/atoms/SegmentedControl.svelte +++ b/web/dashboard/src/lib/components/atoms/SegmentedControl.svelte @@ -2,11 +2,13 @@ // Segmented button group for switching between a few exclusive options // (Tokens/Costs, chart intervals, throughput granularity, ...). // options: array of {value, label}; onchange fires with the picked value. + // disabled locks every option (the active one stays highlighted). let { options = [], value, onchange, ariaLabel = "", + disabled = false, class: className = "", } = $props(); @@ -18,6 +20,7 @@ class="segmented-btn" class:active={value === option.value} aria-pressed={value === option.value} + {disabled} onclick={() => onchange?.(option.value)} >{option.label} {/each} @@ -51,10 +54,15 @@ white-space: nowrap; } - .segmented-btn:hover { + .segmented-btn:hover:not(:disabled) { color: var(--text); } + .segmented-btn:disabled { + cursor: not-allowed; + opacity: 0.6; + } + .segmented-btn.active { background: var(--accent); color: #fff; diff --git a/web/dashboard/src/pages/audit-logs/AuditFilters.svelte b/web/dashboard/src/pages/audit-logs/AuditFilters.svelte index 3f7744b96..f3adf91ca 100644 --- a/web/dashboard/src/pages/audit-logs/AuditFilters.svelte +++ b/web/dashboard/src/pages/audit-logs/AuditFilters.svelte @@ -1,15 +1,35 @@
@@ -69,6 +89,25 @@ +
+ + {hiddenCount > 0 ? m.audit_filter_types_hidden({ count: hiddenCount }) : m.audit_filter_all_types()} + +
+ {m.audit_filter_type_label()} + {#each AUDIT_TYPES as type (type.key)} + {@const visible = !auditList.auditHiddenTypes.includes(type.key)} + + {/each} +
+
+ {/if}
@@ -107,11 +133,38 @@ gap: 12px; } + .mcp-catalog-section-hint { + margin: 0 0 8px; + font-size: 12px; + } + .mcp-catalog-item-name { font-size: 13px; overflow-wrap: anywhere; } + .mcp-catalog-list-excluded .mcp-catalog-item-name { + color: var(--text-muted); + text-decoration: line-through; + } + + .mcp-catalog-badge { + display: inline-block; + margin-left: 6px; + padding: 0 6px; + border: 1px solid var(--border); + border-radius: 999px; + color: var(--text-muted); + font-size: 11px; + line-height: 18px; + vertical-align: middle; + } + + .mcp-catalog-badge-danger { + border-color: color-mix(in srgb, var(--danger) 35%, var(--border)); + color: var(--danger); + } + .mcp-catalog-item-aggregated { margin-top: 2px; color: var(--text-muted); diff --git a/web/dashboard/src/pages/mcp-servers/McpServerEditor.svelte b/web/dashboard/src/pages/mcp-servers/McpServerEditor.svelte index 8122b2da0..4dc408fad 100644 --- a/web/dashboard/src/pages/mcp-servers/McpServerEditor.svelte +++ b/web/dashboard/src/pages/mcp-servers/McpServerEditor.svelte @@ -2,12 +2,14 @@ // MCP server editor modal (create + edit), built on the shared EditorDialog // shell. The slug is derived from the name until manually edited and becomes // immutable once the server exists. Saved header values arrive masked as - // "***"; leaving them unchanged keeps the stored secret on save. + // "***"; leaving them unchanged keeps the stored secret on save. Tool + // exposure lives in McpToolPicker. import TableActionButton from "$lib/components/atoms/TableActionButton.svelte"; import Icon from "$lib/components/atoms/Icon.svelte"; import EnabledToggle from "$lib/components/atoms/EnabledToggle.svelte"; import FormField from "$lib/components/molecules/FormField.svelte"; import EditorDialog from "$lib/components/organisms/EditorDialog.svelte"; + import McpToolPicker from "./McpToolPicker.svelte"; import { mcpServers } from "./mcpServers.svelte.js"; import { Plus, Trash2 } from "lucide"; import * as m from "$lib/paraglide/messages.js"; @@ -125,6 +127,8 @@ + +
-
- - -
- -
- - -
-
+
+ + + {m.mcp_disallowed_user_paths_help()} +
+
{formatNumber(server.tool_count || 0)} + {#if server.excluded_tool_count > 0} + {m.mcp_tools_excluded_count({ count: formatNumber(server.excluded_tool_count) })} + {/if}
{mcpServerSubCountsLabel(server)}
@@ -138,6 +143,12 @@ white-space: nowrap; } + .mcp-server-excluded-count { + margin-left: 6px; + color: var(--text-muted); + font-size: 12px; + } + /* The server table has seven information-dense columns. Preserve readable cells on narrow screens and let the wrapper scroll instead of squeezing them. */ diff --git a/web/dashboard/src/pages/mcp-servers/McpToolPicker.svelte b/web/dashboard/src/pages/mcp-servers/McpToolPicker.svelte new file mode 100644 index 000000000..11a9a8c51 --- /dev/null +++ b/web/dashboard/src/pages/mcp-servers/McpToolPicker.svelte @@ -0,0 +1,321 @@ + + +
+
+ {m.mcp_tools_title()} + {#if summary.total > 0} + {m.mcp_tools_summary(summary)} + {/if} +
+ +
+ {m.mcp_tools_mode_label()} + mcpServers.switchToolMode(mode)} + /> +
+ + {allowMode ? m.mcp_tools_mode_allow_help() : m.mcp_tools_mode_exclude_help()} + {#if modeLocked} + {m.mcp_tools_mode_locked()} + {/if} + + + {#if mcpServers.editorTools.loading} + + {:else if discovered.length === 0} +

+ {mcpServers.editorTools.error ? m.mcp_tools_load_failed() : m.mcp_tools_pending()} +

+ {:else} +
+ (mcpServers.toolQuery = "")} + /> + + +
+ {/if} + + {#if rows.length > 0} +
    + {#each rows as row (row.name)} +
  • + {#if row.missing} + + {row.name} + {m.mcp_tools_missing()} + + mcpServers.removeToolName(row.name)} + > + + + {:else} + + {/if} +
  • + {/each} +
+ {:else if discovered.length > 0} +

{m.mcp_tools_no_match()}

+ {/if} + +
+ + +
+ {m.mcp_tools_add_help()} +
+ + diff --git a/web/dashboard/src/pages/mcp-servers/mcp-servers.js b/web/dashboard/src/pages/mcp-servers/mcp-servers.js index 57dc34b8c..42111893d 100644 --- a/web/dashboard/src/pages/mcp-servers/mcp-servers.js +++ b/web/dashboard/src/pages/mcp-servers/mcp-servers.js @@ -3,6 +3,13 @@ import * as m from "../../lib/paraglide/messages.js"; +// Tool filter modes. "exclude" stores disallowed_tools: every tool is exposed +// except the listed ones, so tools the server adds later appear automatically. +// "allow" stores allowed_tools: only the listed tools are exposed, so new +// tools stay hidden until selected. +export const MCP_TOOL_MODE_EXCLUDE = "exclude"; +export const MCP_TOOL_MODE_ALLOW = "allow"; + export function defaultMcpServerForm() { return { name: "", @@ -12,9 +19,10 @@ export function defaultMcpServerForm() { description: "", enabled: true, headers: [], - allowed_tools: "", - disallowed_tools: "", + tool_mode: MCP_TOOL_MODE_EXCLUDE, + tool_names: [], user_paths: "", + disallowed_user_paths: "", tool_timeout_seconds: "", }; } @@ -25,6 +33,7 @@ export function defaultMcpCatalog() { status: "", instructions: "", tools: [], + excluded_tools: [], prompts: [], resources: [], templates: [], @@ -202,17 +211,14 @@ export function mcpServerFormFromServer(server) { description: String(server.description || "").trim(), enabled: server.enabled !== false, headers: mcpHeadersToRows(server.headers), - allowed_tools: (Array.isArray(server.allowed_tools) - ? server.allowed_tools - : [] - ).join(", "), - disallowed_tools: (Array.isArray(server.disallowed_tools) - ? server.disallowed_tools - : [] - ).join(", "), + ...mcpToolFilterFromServer(server), user_paths: (Array.isArray(server.user_paths) ? server.user_paths : []).join( "\n", ), + disallowed_user_paths: (Array.isArray(server.disallowed_user_paths) + ? server.disallowed_user_paths + : [] + ).join("\n"), tool_timeout_seconds: server.tool_timeout_seconds ? String(server.tool_timeout_seconds) : "", @@ -247,6 +253,13 @@ export function buildMcpServerPayload(form, mode, servers) { if (!url) { return { error: m.mcp_url_required() }; } + const toolNames = uniqueToolNames(form.tool_names); + const allowMode = form.tool_mode === MCP_TOOL_MODE_ALLOW; + if (allowMode && toolNames.length === 0) { + // An empty allowed_tools list means "no restriction" on the gateway, the + // opposite of what an operator who unchecked everything expects. + return { error: m.mcp_tools_allow_empty() }; + } let toolTimeoutSeconds; const rawTimeout = String(form.tool_timeout_seconds || "").trim(); if (rawTimeout !== "") { @@ -267,9 +280,10 @@ export function buildMcpServerPayload(form, mode, servers) { headers: mcpHeaderRowsToObject(form.headers), description: String(form.description || "").trim(), enabled: Boolean(form.enabled), - allowed_tools: splitCommaList(form.allowed_tools), - disallowed_tools: splitCommaList(form.disallowed_tools), + allowed_tools: allowMode ? toolNames : [], + disallowed_tools: allowMode ? [] : toolNames, user_paths: normalizeMcpUserPaths(form.user_paths), + disallowed_user_paths: normalizeMcpUserPaths(form.disallowed_user_paths), tool_timeout_seconds: toolTimeoutSeconds, }, }; @@ -289,6 +303,7 @@ export function normalizeMcpCatalog(name, payload) { status: String(source.status || "").trim(), instructions: String(source.instructions || "").trim(), tools: list(source.tools), + excluded_tools: list(source.excluded_tools), prompts: list(source.prompts), resources: list(source.resources), templates: list(source.templates), @@ -316,9 +331,21 @@ export function mcpCatalogSections(catalog) { name: String(item.name || ""), aggregated: mcpNamespacedName(source, item.name), description: String(item.description || "").trim(), + readOnly: item.read_only === true, + destructive: item.destructive === true, }); const sections = [ { key: "tools", title: m.mcp_catalog_tools(), items: (source.tools || []).map(feature("tool")) }, + { + key: "excluded_tools", + title: m.mcp_catalog_excluded_tools(), + hint: m.mcp_catalog_excluded_tools_hint(), + excluded: true, + items: (source.excluded_tools || []).map((item) => ({ + ...feature("excluded")(item), + aggregated: "", + })), + }, { key: "prompts", title: m.mcp_catalog_prompts(), @@ -351,3 +378,141 @@ export function mcpCatalogSections(catalog) { export function mcpCatalogIsEmpty(catalog) { return mcpCatalogSections(catalog).length === 0; } + +// --- tool picker ----------------------------------------------------------- + +function uniqueToolNames(names) { + const seen = new Set(); + const result = []; + (Array.isArray(names) ? names : []).forEach((value) => { + const name = String(value || "").trim(); + if (name && !seen.has(name)) { + seen.add(name); + result.push(name); + } + }); + return result; +} + +// mcpToolFilterFromServer maps the stored allow/deny lists onto one editor +// mode. Config accepts both lists at once (deny applies after allow); the +// allow mode with allowed − disallowed exposes exactly the same tools. +export function mcpToolFilterFromServer(server) { + const allowed = uniqueToolNames(server && server.allowed_tools); + const disallowed = uniqueToolNames(server && server.disallowed_tools); + if (allowed.length > 0) { + return { + tool_mode: MCP_TOOL_MODE_ALLOW, + tool_names: allowed.filter((name) => !disallowed.includes(name)), + }; + } + return { tool_mode: MCP_TOOL_MODE_EXCLUDE, tool_names: disallowed }; +} + +// mcpDiscoveredTools merges the exposed and excluded catalog lists into one +// name-sorted list: the full set the upstream reports, whatever the filters. +export function mcpDiscoveredTools(catalog) { + const source = catalog || {}; + const byName = new Map(); + [...(source.tools || []), ...(source.excluded_tools || [])].forEach((item) => { + const name = String((item && item.name) || "").trim(); + if (name && !byName.has(name)) { + byName.set(name, { + name, + description: String(item.description || "").trim(), + readOnly: item.read_only === true, + destructive: item.destructive === true, + }); + } + }); + return [...byName.values()].sort((a, b) => a.name.localeCompare(b.name)); +} + +export function mcpToolExposed(form, name) { + const listed = (form.tool_names || []).includes(name); + return form.tool_mode === MCP_TOOL_MODE_ALLOW ? listed : !listed; +} + +// setMcpToolsExposed returns the tool_names list after exposing or hiding +// every given tool in the form's current mode. +export function setMcpToolsExposed(form, names, exposed) { + const targets = uniqueToolNames(names); + const listed = form.tool_mode === MCP_TOOL_MODE_ALLOW ? exposed : !exposed; + const current = uniqueToolNames(form.tool_names); + if (listed) { + return uniqueToolNames([...current, ...targets]); + } + return current.filter((name) => !targets.includes(name)); +} + +// mcpToolModeSwitchable reports whether the mode can flip without changing +// what is exposed. Converting between an allowlist and a denylist needs the +// full tool set; without it, an allowlist would become an empty denylist, +// which exposes every tool. An empty list is safe to flip either way. +export function mcpToolModeSwitchable(form, discovered) { + return (discovered || []).length > 0 || uniqueToolNames(form.tool_names).length === 0; +} + +// switchMcpToolMode flips the filter mode while keeping every discovered +// tool's exposure unchanged; only the treatment of future tools changes. +// Names the server does not report are meaningful only in the old mode. +// When the flip is not safe (see mcpToolModeSwitchable) the form is kept. +export function switchMcpToolMode(form, mode, discovered) { + const next = mode === MCP_TOOL_MODE_ALLOW ? MCP_TOOL_MODE_ALLOW : MCP_TOOL_MODE_EXCLUDE; + if (next === form.tool_mode || !mcpToolModeSwitchable(form, discovered)) { + return { tool_mode: form.tool_mode, tool_names: uniqueToolNames(form.tool_names) }; + } + const names = (discovered || []).map((tool) => tool.name); + const exposed = names.filter((name) => mcpToolExposed(form, name)); + return { + tool_mode: next, + tool_names: + next === MCP_TOOL_MODE_ALLOW + ? exposed + : names.filter((name) => !exposed.includes(name)), + }; +} + +// mcpToolPickerRows lists every discovered tool with its exposure in the +// form, followed by listed names the server does not report (typos, removed +// tools, or names added before the server connected). query narrows by name +// or description. +export function mcpToolPickerRows(form, discovered, query) { + const needle = String(query || "").trim().toLowerCase(); + const known = new Set((discovered || []).map((tool) => tool.name)); + const rows = (discovered || []).map((tool) => ({ + ...tool, + exposed: mcpToolExposed(form, tool.name), + missing: false, + })); + uniqueToolNames(form.tool_names).forEach((name) => { + if (!known.has(name)) { + rows.push({ + name, + description: "", + readOnly: false, + destructive: false, + exposed: form.tool_mode === MCP_TOOL_MODE_ALLOW, + missing: true, + }); + } + }); + if (!needle) { + return rows; + } + return rows.filter( + (row) => + row.name.toLowerCase().includes(needle) || + row.description.toLowerCase().includes(needle), + ); +} + +// mcpToolSelectionSummary counts discovered tools only: a listed name the +// server does not report is neither exposed nor hidden today. +export function mcpToolSelectionSummary(form, discovered) { + const total = (discovered || []).length; + const exposed = (discovered || []).filter((tool) => + mcpToolExposed(form, tool.name), + ).length; + return { exposed, total }; +} diff --git a/web/dashboard/src/pages/mcp-servers/mcpServers.svelte.js b/web/dashboard/src/pages/mcp-servers/mcpServers.svelte.js index e825b43a8..9a143a291 100644 --- a/web/dashboard/src/pages/mcp-servers/mcpServers.svelte.js +++ b/web/dashboard/src/pages/mcp-servers/mcpServers.svelte.js @@ -13,13 +13,17 @@ import { defaultMcpServerForm, deriveMcpServerSlug, filterMcpServers, + mcpDiscoveredTools, mcpServerFormFromServer, mcpPollShouldRetry, mcpServerSlug, mcpServersNeedPolling, mcpServerStatus, normalizeMcpCatalog, + setMcpToolsExposed, + switchMcpToolMode, MCP_SERVERS_POLL_MS, + MCP_TOOL_MODE_ALLOW, } from "./mcp-servers.js"; class McpServersState { @@ -38,6 +42,15 @@ class McpServersState { advancedOpen = $state(false); form = $state(defaultMcpServerForm()); + // Editor tool picker: every tool the server reports, exposed or not. A new + // server, or one that never listed, has none yet. + editorTools = $state({ loading: false, error: "", tools: [] }); + toolQuery = $state(""); + toolDraft = $state(""); + // Bumped per editor catalog load, so a response for an editor session that + // was closed and reopened (same slug) cannot overwrite the newer one. + #toolLoadSeq = 0; + deletingName = $state(""); reconnectingName = $state(""); @@ -179,6 +192,7 @@ class McpServersState { this.advancedOpen = false; this.error = ""; this.form = defaultMcpServerForm(); + this.#resetEditorTools(); this.formOpen = true; } @@ -191,7 +205,9 @@ class McpServersState { this.advancedOpen = false; this.error = ""; this.form = mcpServerFormFromServer(server); + this.#resetEditorTools(); this.formOpen = true; + void this.#loadEditorTools(server); } closeForm() { @@ -201,6 +217,57 @@ class McpServersState { this.advancedOpen = false; this.error = ""; this.form = defaultMcpServerForm(); + this.#resetEditorTools(); + } + + #resetEditorTools() { + this.#toolLoadSeq += 1; + this.editorTools = { loading: false, error: "", tools: [] }; + this.toolQuery = ""; + this.toolDraft = ""; + } + + async #loadEditorTools(server) { + const seq = ++this.#toolLoadSeq; + this.editorTools = { ...this.editorTools, loading: true, error: "" }; + const loaded = await this.#fetchCatalog(server); + // The editor may have closed, reopened, or moved to another server. + if (seq !== this.#toolLoadSeq) { + return; + } + if (loaded.stale) { + this.editorTools = { ...this.editorTools, loading: false }; + return; + } + this.editorTools = { + loading: false, + error: loaded.error || "", + tools: loaded.error ? [] : mcpDiscoveredTools(loaded.catalog), + }; + } + + setToolsExposed(names, exposed) { + this.form.tool_names = setMcpToolsExposed(this.form, names, exposed); + } + + switchToolMode(mode) { + const next = switchMcpToolMode(this.form, mode, this.editorTools.tools); + this.form.tool_mode = next.tool_mode; + this.form.tool_names = next.tool_names; + } + + addToolDraft() { + const name = this.toolDraft.trim(); + if (!name) { + return; + } + // Adding a name lists it in the current mode: excluded or allowed. + this.setToolsExposed([name], this.form.tool_mode === MCP_TOOL_MODE_ALLOW); + this.toolDraft = ""; + } + + removeToolName(name) { + this.form.tool_names = (this.form.tool_names || []).filter((item) => item !== name); } syncSlugFromName() { @@ -377,7 +444,6 @@ class McpServersState { // so it does not map onto the shared list/mutation ladder. async openCatalog(server) { - const name = String((server && server.name) || "").trim(); const slug = mcpServerSlug(server); if (!slug) { return; @@ -392,39 +458,62 @@ class McpServersState { status: mcpServerStatus(server), }; + const loaded = await this.#fetchCatalog(server); + this.catalogLoading = false; + if (loaded.stale) { + return; + } + if (loaded.error) { + this.catalogError = loaded.error; + return; + } + this.catalog = loaded.catalog; + } + + // chooseToolsFromCatalog jumps from the read-only inspector to the editor's + // tool picker for the same server. + chooseToolsFromCatalog() { + const server = (this.servers || []).find( + (item) => mcpServerSlug(item) === this.catalog.server, + ); + if (!server || server.managed) { + return; + } + this.closeCatalog(); + this.openEdit(server); + } + + // #fetchCatalog resolves to { catalog }, { error }, or { stale }. + async #fetchCatalog(server) { + const name = String((server && server.name) || "").trim(); + const slug = mcpServerSlug(server); try { const result = await getJSON( "/admin/mcp-servers/" + encodeURIComponent(slug) + "/catalog", { label: "mcp server catalog" }, ); if (result.stale) { - return; + return { stale: true }; } if (result.status === 503) { this.available = false; - this.catalogError = m.mcp_unavailable(); - return; + return { error: m.mcp_unavailable() }; } if (result.status === 404) { - this.catalogError = m.mcp_not_found({ name }); - return; + return { error: m.mcp_not_found({ name }) }; } if (!result.ok) { - this.catalogError = - result.status === 401 - ? m.common_authentication_required() - : errorPayloadMessage( - result.data, - m.mcp_catalog_load_failed(), - ); - return; + return { + error: + result.status === 401 + ? m.common_authentication_required() + : errorPayloadMessage(result.data, m.mcp_catalog_load_failed()), + }; } - this.catalog = normalizeMcpCatalog(slug, result.data); + return { catalog: normalizeMcpCatalog(slug, result.data) }; } catch (e) { console.error("Failed to load MCP server catalog:", e); - this.catalogError = m.mcp_catalog_load_failed(); - } finally { - this.catalogLoading = false; + return { error: m.mcp_catalog_load_failed() }; } } diff --git a/web/dashboard/src/pages/models/modelDetails.js b/web/dashboard/src/pages/models/modelDetails.js index d49fa007d..5f90a0393 100644 --- a/web/dashboard/src/pages/models/modelDetails.js +++ b/web/dashboard/src/pages/models/modelDetails.js @@ -56,14 +56,14 @@ function pushSection(sections, key, title, items) { } } -// rowHasModelDetails reports whether the accordion has anything to show. A -// virtual model whose target is not in the inventory only carries a stub -// model ({ id, object }), so it gets no toggle. +// rowHasModelDetails reports whether the accordion has anything to show. +// Virtual-model rows get no toggle: their model is only the first resolvable +// target's, which its own row already details and which misdescribes a +// virtual model spreading requests over several targets. export function rowHasModelDetails(row) { const model = row && row.model; - if (!model) return false; - if (!row.is_alias) return Boolean(text(model.id)); - return Boolean(model.metadata || text(model.owned_by) || createdDate(model.created)); + if (!model || row.is_alias) return false; + return Boolean(text(model.id)); } // ---- Layers and views ---- diff --git a/web/dashboard/tests/audit-operations.test.js b/web/dashboard/tests/audit-operations.test.js new file mode 100644 index 000000000..2c2b7d885 --- /dev/null +++ b/web/dashboard/tests/audit-operations.test.js @@ -0,0 +1,81 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; + +import { + AUDIT_TYPES, + auditEntryTypeVisible, + auditExcludeOperationsQuery, + auditTypeForPath, + normalizeHiddenTypes, +} from "../src/pages/audit-logs/audit-operations.js"; +import { auditLogWithLiveEntries, buildAuditLogQuery, buildAuditSessionQuery } from "../src/pages/audit-logs/audit-logic.js"; + +test("auditTypeForPath mirrors the gateway endpoint classification", () => { + const cases = [ + ["/v1/chat/completions", "chat"], + ["/v1/messages", "chat"], + ["/v1/messages/batches/b_1", "batches"], + ["/v1/responses/resp_1/input_items", "responses"], + ["/v1/conversations", "responses"], + ["/v1/files/f_1/content", "batches"], + ["/v1/audio/transcriptions?x=1", "audio"], + ["/v1/images/edits/", "images"], + ["/v1/systemone", "systemone"], + ["/v1/systemone/permute", "systemone"], + ["/v1/systemone/separate/", "systemone"], + ["/v1/systemone/other", ""], + ["/v1/realtime/translations/calls", "realtime"], + ["/mcp", "mcp"], + ["/mcp/github", "mcp"], + ["/mcpx", ""], + ["/p/openai/v1/models", "passthrough"], + ["/admin/audit/log", ""], + ["", ""], + ]; + for (const [path, want] of cases) { + assert.equal(auditTypeForPath(path), want, path); + } +}); + +test("normalizeHiddenTypes keeps known unique keys", () => { + assert.deepEqual(normalizeHiddenTypes(["mcp", "nope", "mcp"]), ["mcp"]); + const all = AUDIT_TYPES.map((type) => type.key); + assert.deepEqual(normalizeHiddenTypes(all), all); + assert.deepEqual(normalizeHiddenTypes("mcp"), []); +}); + +test("auditExcludeOperationsQuery lists the operations of the hidden types", () => { + assert.equal(auditExcludeOperationsQuery([]), ""); + assert.equal( + auditExcludeOperationsQuery(["mcp", "audio", "passthrough"]), + "audio_speech,audio_transcriptions,audio_translations,provider_passthrough,mcp", + ); +}); + +test("audit list and session queries send exclude_operation only when a type is hidden", () => { + const base = { dateQuery: "days=7", limit: 25, offset: 0 }; + assert.doesNotMatch(buildAuditLogQuery(base), /exclude_operation=/); + assert.match(buildAuditLogQuery({ ...base, hiddenTypes: ["mcp"] }), /&exclude_operation=mcp$/); + assert.doesNotMatch(buildAuditSessionQuery({ sessionId: "s" }), /exclude_operation=/); + assert.match( + buildAuditSessionQuery({ sessionId: "s", hiddenTypes: ["passthrough"] }), + /&exclude_operation=provider_passthrough$/, + ); +}); + +test("auditEntryTypeVisible drops hidden types and keeps unclassified rows", () => { + assert.equal(auditEntryTypeVisible({ path: "/mcp" }, []), true); + assert.equal(auditEntryTypeVisible({ path: "/mcp" }, ["mcp"]), false); + assert.equal(auditEntryTypeVisible({ path: "/v1/chat/completions" }, ["mcp"]), true); + assert.equal(auditEntryTypeVisible({ path: "/sso/callback" }, ["mcp"]), true); + assert.equal(auditEntryTypeVisible({}, ["mcp"]), true); +}); + +test("auditLogWithLiveEntries keeps pending previews of hidden types off the page", () => { + const payload = { entries: [], total: 0, limit: 25, offset: 0 }; + const preview = (id, path) => ({ id, request_id: id, path, _live: true, _live_pending: true, _audit_flushed: false }); + const current = [preview("chat", "/v1/chat/completions"), preview("tool", "/mcp")]; + const next = auditLogWithLiveEntries(payload, current, { hiddenTypes: ["mcp"] }); + assert.deepEqual(next.entries.map((entry) => entry.id), ["chat"]); + assert.equal(next.total, 1); +}); diff --git a/web/dashboard/tests/mcp-servers.test.js b/web/dashboard/tests/mcp-servers.test.js index e886c389b..3266609e2 100644 --- a/web/dashboard/tests/mcp-servers.test.js +++ b/web/dashboard/tests/mcp-servers.test.js @@ -28,6 +28,13 @@ import { normalizeMcpCatalog, splitCommaList, normalizeMcpUserPaths, + mcpDiscoveredTools, + mcpToolFilterFromServer, + mcpToolModeSwitchable, + mcpToolPickerRows, + mcpToolSelectionSummary, + setMcpToolsExposed, + switchMcpToolMode, } from "../src/pages/mcp-servers/mcp-servers.js"; test("deriveMcpServerSlug normalizes display names and falls back to a hash", () => { @@ -242,9 +249,10 @@ test("buildMcpServerPayload produces the normalized PUT payload", () => { { name: "Authorization", value: "***" }, { name: "", value: "ignored" }, ], - allowed_tools: "search_issues, get_file", - disallowed_tools: "", + tool_mode: "allow", + tool_names: ["search_issues", " get_file ", "search_issues"], user_paths: "/team/alpha\n/team/beta", + disallowed_user_paths: " /team/alpha/contractors \n\n", tool_timeout_seconds: "45", }, "edit", @@ -263,6 +271,7 @@ test("buildMcpServerPayload produces the normalized PUT payload", () => { allowed_tools: ["search_issues", "get_file"], disallowed_tools: [], user_paths: ["/team/alpha", "/team/beta"], + disallowed_user_paths: ["/team/alpha/contractors"], tool_timeout_seconds: 45, }); }); @@ -302,6 +311,30 @@ test("buildMcpServerPayload validates required fields and timeout", () => { } }); +test("buildMcpServerPayload maps the tool mode onto one filter list", () => { + const base = { + ...defaultMcpServerForm(), + name: "github", + url: "https://mcp.example.com/mcp", + }; + + const excluded = buildMcpServerPayload( + { ...base, tool_names: ["delete_repo"] }, + "create", + [], + ); + assert.deepEqual(excluded.payload.allowed_tools, []); + assert.deepEqual(excluded.payload.disallowed_tools, ["delete_repo"]); + + // An empty allowlist would expose every tool on the gateway, so the + // "only selected" mode requires at least one selection. + assert.equal( + buildMcpServerPayload({ ...base, tool_mode: "allow", tool_names: [] }, "create", []) + .error, + "Select at least one tool, or switch new tools to exposed.", + ); +}); + test("buildMcpServerPayload rejects duplicate slugs only when creating", () => { const form = { ...defaultMcpServerForm(), @@ -330,6 +363,7 @@ test("mcpServerFormFromServer prefills the editor form", () => { allowed_tools: ["search_issues"], disallowed_tools: ["delete_repo"], user_paths: ["/team/alpha"], + disallowed_user_paths: ["/team/alpha/contractors"], tool_timeout_seconds: 45, }), { @@ -340,14 +374,136 @@ test("mcpServerFormFromServer prefills the editor form", () => { description: "Issue tools", enabled: false, headers: [{ name: "Authorization", value: "***" }], - allowed_tools: "search_issues", - disallowed_tools: "delete_repo", + tool_mode: "allow", + tool_names: ["search_issues"], user_paths: "/team/alpha", + disallowed_user_paths: "/team/alpha/contractors", tool_timeout_seconds: "45", }, ); }); +test("mcpToolFilterFromServer picks one mode without changing exposure", () => { + assert.deepEqual(mcpToolFilterFromServer({}), { tool_mode: "exclude", tool_names: [] }); + assert.deepEqual(mcpToolFilterFromServer({ disallowed_tools: ["delete_repo"] }), { + tool_mode: "exclude", + tool_names: ["delete_repo"], + }); + // Deny applies after allow on the gateway, so allowed − disallowed exposes + // the same tools. + assert.deepEqual( + mcpToolFilterFromServer({ + allowed_tools: ["read", "write"], + disallowed_tools: ["write"], + }), + { tool_mode: "allow", tool_names: ["read"] }, + ); +}); + +test("mcpDiscoveredTools merges exposed and excluded tools by name", () => { + const discovered = mcpDiscoveredTools( + normalizeMcpCatalog("github", { + tools: [{ name: "search", read_only: true }], + excluded_tools: [{ name: "delete_repo", description: "Delete", destructive: true }], + }), + ); + assert.deepEqual(discovered, [ + { name: "delete_repo", description: "Delete", readOnly: false, destructive: true }, + { name: "search", description: "", readOnly: true, destructive: false }, + ]); +}); + +test("tool picker toggles, bulk actions, and summary follow the mode", () => { + const discovered = [{ name: "a" }, { name: "b" }, { name: "c" }].map((tool) => ({ + ...tool, + description: "", + readOnly: false, + destructive: false, + })); + + const exclude = { tool_mode: "exclude", tool_names: [] }; + exclude.tool_names = setMcpToolsExposed(exclude, ["b"], false); + assert.deepEqual(exclude.tool_names, ["b"]); + assert.deepEqual(mcpToolSelectionSummary(exclude, discovered), { exposed: 2, total: 3 }); + assert.deepEqual(setMcpToolsExposed(exclude, ["a", "b", "c"], true), []); + assert.deepEqual(setMcpToolsExposed(exclude, ["a", "b", "c"], false), ["b", "a", "c"]); + + const allow = { tool_mode: "allow", tool_names: ["a"] }; + assert.deepEqual(setMcpToolsExposed(allow, ["c"], true), ["a", "c"]); + assert.deepEqual(setMcpToolsExposed(allow, ["a"], false), []); + assert.deepEqual(mcpToolSelectionSummary(allow, discovered), { exposed: 1, total: 3 }); +}); + +test("switchMcpToolMode keeps every discovered tool's exposure", () => { + const discovered = [{ name: "a" }, { name: "b" }, { name: "c" }]; + const toAllow = switchMcpToolMode( + { tool_mode: "exclude", tool_names: ["b", "ghost"] }, + "allow", + discovered, + ); + assert.deepEqual(toAllow, { tool_mode: "allow", tool_names: ["a", "c"] }); + + const back = switchMcpToolMode(toAllow, "exclude", discovered); + assert.deepEqual(back, { tool_mode: "exclude", tool_names: ["b"] }); +}); + +test("switchMcpToolMode keeps the list when the catalog is unknown", () => { + // Without the full tool set, an allowlist would become an empty denylist, + // which exposes every tool on save. + const allow = { tool_mode: "allow", tool_names: ["read"] }; + assert.equal(mcpToolModeSwitchable(allow, []), false); + assert.deepEqual(switchMcpToolMode(allow, "exclude", []), { + tool_mode: "allow", + tool_names: ["read"], + }); + + // An empty list flips safely, so a brand-new server can still pick a mode. + const fresh = { tool_mode: "exclude", tool_names: [] }; + assert.equal(mcpToolModeSwitchable(fresh, []), true); + assert.deepEqual(switchMcpToolMode(fresh, "allow", []), { + tool_mode: "allow", + tool_names: [], + }); +}); + +test("mcpToolPickerRows flags listed names the server does not report and filters", () => { + const discovered = [ + { name: "create_issue", description: "Create an issue", readOnly: false, destructive: false }, + { name: "delete_repo", description: "", readOnly: false, destructive: true }, + ]; + const form = { tool_mode: "exclude", tool_names: ["delete_repo", "delet_repo"] }; + + const rows = mcpToolPickerRows(form, discovered, ""); + assert.deepEqual( + rows.map((row) => [row.name, row.exposed, row.missing]), + [ + ["create_issue", true, false], + ["delete_repo", false, false], + ["delet_repo", false, true], + ], + ); + assert.deepEqual( + mcpToolPickerRows(form, discovered, "ISSUE").map((row) => row.name), + ["create_issue"], + ); +}); + +test("mcpCatalogSections lists excluded tools without an aggregated name", () => { + const sections = mcpCatalogSections( + normalizeMcpCatalog("github", { + tools: [{ name: "search" }], + excluded_tools: [{ name: "delete_repo", destructive: true }], + }), + ); + assert.deepEqual( + sections.map((section) => section.key), + ["tools", "excluded_tools"], + ); + assert.equal(sections[1].excluded, true); + assert.equal(sections[1].items[0].aggregated, ""); + assert.equal(sections[1].items[0].destructive, true); +}); + test("mcpCatalogSections derives aggregated /mcp names for tools and prompts only", () => { const catalog = normalizeMcpCatalog("github", { server: "github", @@ -378,6 +534,8 @@ test("mcpCatalogSections derives aggregated /mcp names for tools and prompts onl name: "create_issue", aggregated: "github_create_issue", description: "Create a GitHub issue", + readOnly: false, + destructive: false, }); assert.equal(tools[1].aggregated, "github_search_issues"); assert.equal(tools[1].description, ""); diff --git a/web/dashboard/tests/models-details.test.js b/web/dashboard/tests/models-details.test.js index 106011a60..20b7dc8ef 100644 --- a/web/dashboard/tests/models-details.test.js +++ b/web/dashboard/tests/models-details.test.js @@ -54,13 +54,13 @@ function itemValue(details, key, label) { return entry ? entry.value : undefined; } -test("rowHasModelDetails: real models expand, alias stubs do not", () => { +test("rowHasModelDetails: real models expand, virtual models do not", () => { assert.equal(rowHasModelDetails(modelRow()), true); assert.equal(rowHasModelDetails(modelRow({ model: { id: "bare" } })), true); assert.equal(rowHasModelDetails({ is_alias: true, model: { id: "alias", object: "model" } }), false); assert.equal( - rowHasModelDetails({ is_alias: true, model: { id: "gpt-4o", owned_by: "openai" } }), - true, + rowHasModelDetails({ is_alias: true, model: { id: "gpt-4o", owned_by: "openai", metadata: {} } }), + false, ); assert.equal(rowHasModelDetails({ is_alias: false, model: null }), false); assert.equal(rowHasModelDetails(null), false);