From f1e92dacd86288f4b3aff11435ea063dde680f38 Mon Sep 17 00:00:00 2001 From: "Jakub A. W" Date: Sat, 26 Sep 2026 18:54:49 +0200 Subject: [PATCH 1/3] feat(jev): add native /v1/systemone endpoint --- cmd/gomodel/docs/docs.go | 69 +++++ config/config.example.yaml | 10 +- docs/openapi.json | 94 +++++++ docs/providers/jev.mdx | 95 +++++-- docs/providers/overview.mdx | 11 +- internal/core/endpoint_operations.go | 1 + internal/core/endpoints.go | 13 +- internal/core/endpoints_test.go | 2 + internal/core/systemone.go | 13 + internal/core/workflow.go | 6 + internal/gateway/interfaces.go | 6 + internal/guardrails/integration_test.go | 29 +++ internal/guardrails/workflow_executor.go | 20 ++ .../plugins/exchange/systemone_request.go | 104 ++++++++ .../exchange/systemone_request_test.go | 120 +++++++++ internal/providers/jev/jev.go | 12 +- internal/providers/jev/jev_test.go | 2 +- internal/server/http.go | 3 + internal/server/messages_native.go | 54 ++-- internal/server/model_validation.go | 2 +- internal/server/systemone_handler.go | 227 ++++++++++++++++ internal/server/systemone_handler_test.go | 244 ++++++++++++++++++ web/dashboard/messages/de.json | 1 + web/dashboard/messages/en.json | 1 + web/dashboard/messages/pl.json | 1 + web/dashboard/messages/zh-CN.json | 1 + .../src/pages/audit-logs/audit-operations.js | 2 + web/dashboard/tests/audit-operations.test.js | 2 + 28 files changed, 1083 insertions(+), 62 deletions(-) create mode 100644 internal/core/systemone.go create mode 100644 internal/plugins/exchange/systemone_request.go create mode 100644 internal/plugins/exchange/systemone_request_test.go create mode 100644 internal/server/systemone_handler.go create mode 100644 internal/server/systemone_handler_test.go diff --git a/cmd/gomodel/docs/docs.go b/cmd/gomodel/docs/docs.go index d45b61404..769ec2064 100644 --- a/cmd/gomodel/docs/docs.go +++ b/cmd/gomodel/docs/docs.go @@ -7504,6 +7504,75 @@ const docTemplate = `{ ] } }, + "/v1/systemone": { + "post": { + "description": "Available when a jev provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", + "consumes": [ + "application/json" + ], + "produces": [ + "application/json" + ], + "tags": [ + "systemone" + ], + "summary": "Evaluate a System One decision request (Jev / Kev)", + "parameters": [ + { + "description": "System One request: model, state, and questions", + "name": "request", + "in": "body", + "required": true, + "schema": { + "type": "object" + } + } + ], + "responses": { + "200": { + "description": "System One answers, in the provider's shape", + "schema": { + "type": "object" + } + }, + "400": { + "description": "Bad Request", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "401": { + "description": "Unauthorized", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "404": { + "description": "Not Found", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "429": { + "description": "Too Many Requests", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "502": { + "description": "Bad Gateway", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ] + } + }, "/v1/usage": { "get": { "description": "Returns recorded usage, budget statuses, and rate limit statuses for the caller's effective user path (the path bound to the managed API key, or the user-path header for master-key callers).", diff --git a/config/config.example.yaml b/config/config.example.yaml index 9f3e7e239..9b0a08d46 100644 --- a/config/config.example.yaml +++ b/config/config.example.yaml @@ -591,10 +591,12 @@ providers: type: jev api_key: "${JEV_API_KEY}" # base_url defaults to "https://api.typesafe.ai". TypeSafe's System One - # API is a decision API with no OpenAI-compatible surface: requests go to - # POST /p/jev/v1/systemone, or point the TypeSafe SDK at /p/jev. A - # self-hosted Kev server speaks the same API without authentication: - # set base_url (e.g. "http://localhost:8009") and omit api_key. + # API is a decision API with no OpenAI-compatible surface: configuring + # this provider enables POST /v1/systemone, which forwards requests + # natively (point the TypeSafe SDK at the gateway root). A self-hosted Kev + # server speaks the same API without authentication: set base_url + # (e.g. "http://localhost:8009") and omit api_key. Name it "kev" to see + # that name in logs and usage; no separate provider type is needed. # Jev is priced per input token and is not in the upstream model catalog; # declare its pricing here to have the gateway cost System One requests. # models: diff --git a/docs/openapi.json b/docs/openapi.json index f90416500..6f7ef6e20 100644 --- a/docs/openapi.json +++ b/docs/openapi.json @@ -11193,6 +11193,100 @@ } } }, + "/v1/systemone": { + "post": { + "description": "Available when a jev provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", + "tags": [ + "systemone" + ], + "summary": "Evaluate a System One decision request (Jev / Kev)", + "requestBody": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + }, + "description": "System One request: model, state, and questions", + "required": true + }, + "responses": { + "200": { + "description": "System One answers, in the provider's shape", + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + } + }, + "400": { + "description": "Bad Request", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "401": { + "description": "Unauthorized", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "404": { + "description": "Not Found", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "429": { + "description": "Too Many Requests", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "502": { + "description": "Bad Gateway", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ], + "x-mint": { + "metadata": { + "sidebarTitle": "/v1/systemone", + "title": "Evaluate a System One decision request (Jev / Kev)", + "description": "GoModel API reference for POST /v1/systemone: Evaluate a System One decision request (Jev / Kev)." + } + } + } + }, "/v1/usage": { "get": { "description": "Returns recorded usage, budget statuses, and rate limit statuses for the caller's effective user path (the path bound to the managed API key, or the user-path header for master-key callers).", diff --git a/docs/providers/jev.mdx b/docs/providers/jev.mdx index c2c2b16c0..4004393e6 100644 --- a/docs/providers/jev.mdx +++ b/docs/providers/jev.mdx @@ -2,7 +2,7 @@ title: "Jev / Kev (TypeSafe System One)" sidebarTitle: "Jev / Kev" description: "Route TypeSafe System One decision requests through GoModel, to the hosted Jev API or a self-hosted Kev server." -icon: "scale-balanced" +icon: "scale" keywords: ["Jev", "Kev", "TypeSafe", "System One", "decision model", "classification", "noul", "choice", "score", "self-hosted"] --- @@ -22,10 +22,11 @@ There are three question types: | `score` | Rate against ordered levels | `score`, plus `legend`, `probabilities` and `confidence` | The API is not OpenAI-compatible, and its answers have no chat equivalent, so -GoModel does not translate it: System One requests go through -[passthrough](/features/passthrough-api) at `/p/jev/...`, which is enabled by -default for this provider. Chat, `/responses`, and `/v1/embeddings` return -`invalid_request_error` for `jev` models. +GoModel forwards it natively instead of translating it: `POST /v1/systemone` +is available as soon as a `jev` provider is configured, and +[passthrough](/features/passthrough-api) at `/p/jev/...` reaches every other +upstream route. Chat, `/responses`, and `/v1/embeddings` return +`invalid_request_error` for `jev` models, pointing at `/v1/systemone`. ## Configure @@ -49,13 +50,15 @@ GOMODEL_MASTER_KEY=change-me use; a trailing `/v1` is accepted and trimmed, so both spellings address the same server. To run the hosted API and a local Kev side by side, register the second under a suffixed name: `JEV_KEV_BASE_URL=...` creates provider - `jev-kev`, reached at `/p/jev-kev/...`. + `jev-kev`, reached at `/p/jev-kev/...`. In `config.yaml`, any name works, + such as `kev: {type: jev, base_url: ...}`, and that name is what logs, + usage, and model prefixes show. ## Verify ```bash -curl -s http://localhost:8080/p/jev/v1/systemone \ +curl -s http://localhost:8080/v1/systemone \ -H "Authorization: Bearer change-me" \ -H "Content-Type: application/json" \ -d '{ @@ -88,19 +91,21 @@ curl -s http://localhost:8080/p/jev/v1/systemone \ } ``` -The `/v1` segment is optional: `/p/jev/systemone` is the same route. Use -`kev-latest` as the model on a Kev server; it also answers to `jev-latest`. +Use `kev-latest` as the model on a Kev server; it also answers to +`jev-latest`. The same request works at `/p/jev/v1/systemone`, but that +passthrough route skips virtual models and guardrails; see +[the native endpoint](#the-native-endpoint). ## Using the TypeSafe SDKs -The SDKs send `POST {base_url}/v1/systemone`, so point them at the provider's -passthrough root and authenticate with your GoModel key: +The SDKs send `POST {base_url}/v1/systemone`, so point them at the gateway +itself and authenticate with your GoModel key: ```python Python from typesafe_sdk import Noul, TypeSafeClient -client = TypeSafeClient(api_key="change-me", base_url="http://localhost:8080/p/jev") +client = TypeSafeClient(api_key="change-me", base_url="http://localhost:8080") response = client.system_one( state="I was charged twice. Please fix this ASAP.", questions={"billing": Noul(instructions="Is this ticket about billing?")}, @@ -111,7 +116,7 @@ print(response.nouls["billing"].noul) ```typescript JavaScript import { TypeSafeClient, noul } from "@typesafe-ai/sdk"; -const client = new TypeSafeClient({ apiKey: "change-me", baseURL: "http://localhost:8080/p/jev" }); +const client = new TypeSafeClient({ apiKey: "change-me", baseURL: "http://localhost:8080" }); const result = await client.systemOne({ state: "I was charged twice. Please fix this ASAP.", questions: { billing: noul({ instructions: "Is this ticket about billing?" }) }, @@ -120,14 +125,61 @@ console.log(result.answers.billing.noul); ``` -The same works with `TYPESAFE_BASE_URL=http://localhost:8080/p/jev` and -`TYPESAFE_API_KEY=change-me` in the environment. +The same works with `TYPESAFE_BASE_URL=http://localhost:8080` and +`TYPESAFE_API_KEY=change-me` in the environment. With several System One +providers, name the model with its provider (`kev/kev-latest`) or a +[virtual model](/features/virtual-models). +The SDKs' model listing expects TypeSafe's shape, while the gateway's +`/v1/models` is OpenAI-shaped; list upstream models at `/p/jev/v1/models`. + +## The native endpoint + +`POST /v1/systemone` is a gateway endpoint, not a raw proxy. For each request +GoModel: + +1. Resolves `model` like any other endpoint: a bare name, a provider-qualified + name (`jev/jev-latest`), or a virtual model, then applies the caller's + model allowlist, rate limits, and budgets. +2. Runs the workflow's prompt [guardrails](/advanced/guardrails) over `state`. +3. Forwards the body with only `model` (to the resolved name) and `state` (if + a guardrail edited it) changed. Questions, criteria, and every other field + reach the provider byte for byte. +4. Relays the answer unchanged, and records it in the audit log (as a + **System One** request) and in usage. + +The endpoint never translates. If a model resolves to a provider without the +System One API, for example a virtual model pointing at a chat model, the +request fails with `400 invalid_request_error` explaining why, and the gateway +logs a warning. Without a `jev` provider, the route answers `404`. + +Two targets serve the API natively: `jev` providers (hosted Jev or a Kev +server) and, once the endpoint is available, +[OpenRouter](https://openrouter.ai/docs/guides/community/jev), which serves +Jev at the same path (`openrouter/typesafe/jev-1.13`). A virtual model can therefore front a local +Kev server and OpenRouter's `typesafe/jev-1.13` together. OpenRouter adds `id`, +`provider`, and `usage.cost` to the answer, and GoModel records the reported +cost. + +Response caching and failover do not apply to this endpoint yet. + +### Guardrails + +Guardrails see `state` as a single user message: a string state as its text, +any other JSON value as its encoded JSON (which must still be valid JSON after +an edit). This is what anonymizing or blocking guardrails need. The questions +are your application's fixed schema and are not exposed. + +Guardrail edits a decision request has no place for, such as a system prompt +injected by a workflow that also covers chat models, are dropped with a +warning in the logs rather than failing the request. A guardrail that would +answer the request itself blocks it instead, since System One callers expect +typed answers, not text. ## Native routes | Route | What it does | | --- | --- | -| `POST /p/jev/v1/systemone` | Evaluate a state against a map of questions | +| `POST /p/jev/v1/systemone` | Evaluate a state against a map of questions, without virtual models or guardrails | | `GET /p/jev/v1/models` | The names the `model` field accepts, in the upstream's own shape | | `POST /p/jev/v1/systemone/permute` | Kev only: run one Choice question with several option orders | | `POST /p/jev/v1/systemone/separate` | Kev only: run each question in its own forward pass | @@ -145,9 +197,14 @@ IDs such as `jev-1.13.0` are accepted by the `model` field whether or not they are listed. The models are categorized as utility models with no generation mode, since there is no OpenAI endpoint to route them to. -Every System One request names its model, so the passthrough surface applies -the caller's [model allowlist](/features/users) to it like any other -request. +Every System One request names its model, so both `/v1/systemone` and the +passthrough surface apply the caller's [model allowlist](/features/users) to +it like any other request. + +`/v1/systemone` routes only to models in the catalog. To pin a version the +upstream does not list, such as `jev-1.13.0`, declare it under the provider's +`models` (as in the pricing example below) and set +`CONFIGURED_PROVIDER_MODELS_MODE=merge`; passthrough accepts any name. The response's `usage.input_tokens` and `usage.output_tokens` are recorded, so System One calls appear in the usage API and dashboard under the model that diff --git a/docs/providers/overview.mdx b/docs/providers/overview.mdx index 055ac06c8..93506fd8d 100644 --- a/docs/providers/overview.mdx +++ b/docs/providers/overview.mdx @@ -211,10 +211,13 @@ support, not every individual model capability exposed by an upstream provider. through passthrough. See [audio.cpp](/providers/audiocpp). - **Jev / Kev** — TypeSafe's System One API is a decision API (state plus typed questions in, calibrated probabilities out) with no OpenAI-compatible - surface, so it is reached only through passthrough at - `POST /p/jev/v1/systemone`. `JEV_API_KEY` configures the hosted API; for a - self-hosted Kev server, which speaks the same API without authentication, - set `JEV_BASE_URL` and leave the key unset. See [Jev](/providers/jev). + surface, so GoModel forwards it natively, untranslated, at + `POST /v1/systemone` (with virtual models, guardrails, audit, and usage) and + through passthrough at `/p/jev/...`. The endpoint is available once a `jev` + provider is configured, and can also route to OpenRouter's Jev models. + `JEV_API_KEY` configures the hosted API; for a self-hosted Kev server, which + speaks the same API without authentication, set `JEV_BASE_URL` and leave the + key unset. See [Jev](/providers/jev). - **llama.cpp / LM Studio** — `LLAMACPP_BASE_URL` is required (llama-server's default port collides with GoModel's own 8080, so there is no default); `LLAMACPP_API_KEY` is optional. Do not register these servers as `ollama`, diff --git a/internal/core/endpoint_operations.go b/internal/core/endpoint_operations.go index 6ab6df986..8bb2b1395 100644 --- a/internal/core/endpoint_operations.go +++ b/internal/core/endpoint_operations.go @@ -29,6 +29,7 @@ var operationPaths = map[Operation]OperationPaths{ "/v1/realtime/translations", "/v1/realtime/translations/calls", "/v1/realtime/translations/client_secrets", }}, OperationMCP: {Prefixes: []string{"/mcp"}}, + OperationSystemOne: {Exact: []string{"/v1/systemone"}}, OperationProviderPassthrough: {Prefixes: []string{"/p"}}, } diff --git a/internal/core/endpoints.go b/internal/core/endpoints.go index b75404322..386d3c726 100644 --- a/internal/core/endpoints.go +++ b/internal/core/endpoints.go @@ -33,6 +33,7 @@ const ( OperationRealtime Operation = "realtime" OperationProviderPassthrough Operation = "provider_passthrough" OperationMCP Operation = "mcp" + OperationSystemOne Operation = "systemone" ) // EndpointDescriptor centralizes the transport-facing classification of model and provider routes. @@ -169,6 +170,16 @@ func describeEndpointPath(path string) EndpointDescriptor { Dialect: "openai_compat", Operation: OperationImageEdits, } + case path == "/v1/systemone": + // TypeSafe's System One decision API (Jev, Kev). It has no canonical + // translation: the body is forwarded to a System One provider + // unchanged, apart from the routed model and guardrail edits to state. + return EndpointDescriptor{ + ModelInteraction: true, + IngressManaged: true, + Dialect: "systemone", + Operation: OperationSystemOne, + } case isRealtimePath(path): // The realtime endpoints relay the provider's schema verbatim: /v1/realtime // upgrades to a websocket, /v1/realtime/calls exchanges WebRTC SDP, and @@ -238,7 +249,7 @@ func bodyModeForEndpoint(method, path string, operation Operation) BodyMode { return BodyModeMultipart } return BodyModeNone - case OperationAudioSpeech, OperationImageGenerations: + case OperationAudioSpeech, OperationImageGenerations, OperationSystemOne: return BodyModeJSON case OperationAudioTranscriptions, OperationAudioTranslations, OperationImageEdits: return BodyModeMultipart diff --git a/internal/core/endpoints_test.go b/internal/core/endpoints_test.go index 6c48356b2..b881965de 100644 --- a/internal/core/endpoints_test.go +++ b/internal/core/endpoints_test.go @@ -42,6 +42,8 @@ func TestDescribeEndpointPath(t *testing.T) { {path: "/v1/realtime/translations/client_secrets", managed: false, dialect: "openai_compat", operation: OperationRealtime, bodyMode: BodyModeNone, interaction: true}, {path: "/mcp", managed: false, dialect: "mcp", operation: OperationMCP, bodyMode: BodyModeNone, interaction: true}, {path: "/mcp/linear", managed: false, dialect: "mcp", operation: OperationMCP, bodyMode: BodyModeNone, interaction: true}, + {path: "/v1/systemone", managed: true, dialect: "systemone", operation: OperationSystemOne, bodyMode: BodyModeJSON, interaction: true}, + {path: "/v1/systemone/permute", managed: false, dialect: "", operation: "", bodyMode: BodyModeNone, interaction: false}, {path: "/p/openai/responses", managed: true, dialect: "provider_passthrough", operation: OperationProviderPassthrough, bodyMode: BodyModeOpaque, interaction: true}, {path: "/v1/models", managed: false, dialect: "", operation: "", bodyMode: BodyModeNone, interaction: false}, } diff --git a/internal/core/systemone.go b/internal/core/systemone.go new file mode 100644 index 000000000..8ee6d5bd1 --- /dev/null +++ b/internal/core/systemone.go @@ -0,0 +1,13 @@ +package core + +import "github.com/goccy/go-json" + +// SystemOneRequest is the part of a System One decision request the gateway +// reads: the model it routes on and the state guardrails inspect. The rest of +// the body (the questions and their criteria) reaches the provider unchanged. +type SystemOneRequest struct { + Model string `json:"model"` + // State is the text or record the questions are asked about, as sent: a + // JSON string in TypeSafe's examples, but any JSON value is forwarded. + State json.RawMessage `json:"state,omitempty"` +} diff --git a/internal/core/workflow.go b/internal/core/workflow.go index 4d9a25f37..a6cc38ac5 100644 --- a/internal/core/workflow.go +++ b/internal/core/workflow.go @@ -57,6 +57,12 @@ func CapabilitiesForEndpoint(desc EndpointDescriptor) CapabilitySet { return CapabilitySet{ SemanticExtraction: true, } + case OperationSystemOne: + return CapabilitySet{ + AliasResolution: true, + Guardrails: true, + UsageTracking: true, + } case OperationProviderPassthrough: return CapabilitySet{ SemanticExtraction: true, diff --git a/internal/gateway/interfaces.go b/internal/gateway/interfaces.go index 60fa8783e..cc4cf8da6 100644 --- a/internal/gateway/interfaces.go +++ b/internal/gateway/interfaces.go @@ -53,6 +53,12 @@ type TranslatedRequestPatcher interface { PatchResponsesRequest(ctx context.Context, req *core.ResponsesRequest) (*core.ResponsesRequest, error) } +// SystemOneRequestPatcher is an optional TranslatedRequestPatcher capability: +// it runs the prompt phase over a System One decision request's state. +type SystemOneRequestPatcher interface { + PatchSystemOneRequest(ctx context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) +} + // PromptContentEditor is an optional TranslatedRequestPatcher capability: it // reports whether the prompt phase may rewrite the content of this request. // Only a rewriting prompt phase (anonymization, redaction) needs the replayed diff --git a/internal/guardrails/integration_test.go b/internal/guardrails/integration_test.go index 03f1b096d..fdb696698 100644 --- a/internal/guardrails/integration_test.go +++ b/internal/guardrails/integration_test.go @@ -11,6 +11,7 @@ import ( "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/plugins" "github.com/enterpilot/gomodel/pluginapi" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -329,3 +330,31 @@ func TestWorkflowBatchPreparerRejectsRespondDecisions(t *testing.T) { require.ErrorAs(t, err, &gatewayErr) require.Equal(t, http.StatusBadRequest, gatewayErr.HTTPStatusCode()) } + +// A System One request exposes only its state: an anonymizing guardrail +// rewrites it, while a system prompt a workflow injects for chat models has +// no place in a decision request and is dropped rather than failing it. +func TestWorkflowRequestPatcherSystemOneGuardsState(t *testing.T) { + store := newTestStore( + systemPromptDefinition("safety", "be safe"), + Definition{Name: "privacy", Type: "llm_based_altering", Config: rawConfig(t, map[string]any{"model": "openai/gpt-4o-mini", "roles": []string{"user"}})}, + ) + service := newService(t, store, chatFunc(func(_ context.Context, req *core.ChatRequest) (*core.ChatResponse, error) { + text := core.ExtractTextContent(req.Messages[1].Content) + inner := strings.TrimSuffix(strings.TrimPrefix(text, "\n"), "\n") + return replyChat(strings.ReplaceAll(inner, "John", "[PERSON]"))(context.Background(), req) + })) + patcher := NewWorkflowRequestPatcher(staticChains{chainsFor(t, service, + StepReference{Ref: "privacy", Step: 10}, + StepReference{Ref: "safety", Step: 20}, + )}) + + req := &core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(`"John was charged twice"`)} + ctx, state := plugins.WithRequestState(context.Background()) + got, err := patcher.PatchSystemOneRequest(ctx, req) + require.NoError(t, err) + assert.JSONEq(t, `"[PERSON] was charged twice"`, string(got.State)) + assert.Equal(t, "kev-latest", got.Model) + assert.JSONEq(t, `"John was charged twice"`, string(req.State), "the original request must not change") + require.Len(t, state.Snapshot(), 2) +} diff --git a/internal/guardrails/workflow_executor.go b/internal/guardrails/workflow_executor.go index 886e99fda..82783ec31 100644 --- a/internal/guardrails/workflow_executor.go +++ b/internal/guardrails/workflow_executor.go @@ -2,6 +2,7 @@ package guardrails import ( "context" + "log/slog" "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/plugins" @@ -36,6 +37,25 @@ func (p *WorkflowRequestPatcher) PatchResponsesRequest(ctx context.Context, req return processGuardedResponses(ctx, p.chain(ctx), req) } +// PatchSystemOneRequest runs the prompt chain over a System One request's +// state. Edits a decision request has no place for, such as an injected +// system prompt, are dropped with a warning rather than failing the request: +// a guardrail scoped to every model is not wrong for System One models, it +// only has nothing to change there. +func (p *WorkflowRequestPatcher) PatchSystemOneRequest(ctx context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) { + if req == nil { + return nil, nil + } + return processGuarded(ctx, p.chain(ctx), req, "System One", exchange.FromSystemOneRequest, applySystemOneEdits) +} + +func applySystemOneEdits(req *core.SystemOneRequest, prompt *pluginapi.Prompt) (*core.SystemOneRequest, error) { + if dropped := exchange.SystemOneUncarriedEdits(prompt); len(dropped) > 0 { + slog.Warn("guardrail edits a System One request cannot carry were dropped; only the state is guarded", "model", req.Model, "dropped", dropped) + } + return exchange.ApplyToSystemOneRequest(req, prompt) +} + // EditsPromptContent reports whether the request's prompt chain holds an // instance that edits content, such as an anonymizing guardrail. A chained // Responses request only needs its stored history expanded into the input diff --git a/internal/plugins/exchange/systemone_request.go b/internal/plugins/exchange/systemone_request.go new file mode 100644 index 000000000..e10d28732 --- /dev/null +++ b/internal/plugins/exchange/systemone_request.go @@ -0,0 +1,104 @@ +package exchange + +import ( + "fmt" + "sort" + + "github.com/goccy/go-json" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/pluginapi" +) + +// SystemOneStateMessageID is the ID of the user message built from a System +// One request's state. +const SystemOneStateMessageID = "state" + +// FromSystemOneRequest builds the unified prompt for a System One decision +// request. The state is the only content a caller supplies per request, so it +// becomes the prompt's single user message: a string state as its text, any +// other JSON value as its encoded JSON. The questions are the application's +// fixed schema and are not exposed. +func FromSystemOneRequest(req *core.SystemOneRequest) (*pluginapi.Prompt, error) { + if req == nil { + return nil, fmt.Errorf("exchange: nil System One request") + } + raw, err := json.Marshal(req) + if err != nil { + return nil, fmt.Errorf("exchange: encode System One request: %w", err) + } + p := &pluginapi.Prompt{ + Raw: raw, + Messages: []pluginapi.Message{pluginapi.TextMessage(pluginapi.RoleUser, systemOneStateText(req.State))}, + Params: pluginapi.Params{Model: req.Model}, + } + p.Messages[0].ID = SystemOneStateMessageID + p.Reset() + return p, nil +} + +func systemOneStateText(state json.RawMessage) string { + var text string + if err := json.Unmarshal(state, &text); err == nil { + return text + } + return string(state) +} + +// ApplyToSystemOneRequest returns a copy of original with the state +// message's edits applied. A string state stays a string; a state sent as +// another JSON value must still be valid JSON after the edit. Removing the +// state is an error, since the request would have nothing to decide on. +// Edits System One has no place for, such as inserted messages or parameter +// changes, are not applied; SystemOneUncarriedEdits lists them. +func ApplyToSystemOneRequest(original *core.SystemOneRequest, p *pluginapi.Prompt) (*core.SystemOneRequest, error) { + if original == nil || p == nil { + return nil, fmt.Errorf("exchange: nil System One request or prompt") + } + result := *original + switch p.Changes().Messages[SystemOneStateMessageID] { + case "": + return &result, nil + case pluginapi.ChangeRemoved: + return nil, fmt.Errorf("exchange: the System One state was removed") + } + msg := p.Message(SystemOneStateMessageID) + if msg == nil { + return nil, fmt.Errorf("exchange: the System One state was removed") + } + text := msg.Text() + var current string + if err := json.Unmarshal(original.State, ¤t); err == nil || len(original.State) == 0 { + encoded, err := json.Marshal(text) + if err != nil { + return nil, fmt.Errorf("exchange: encode System One state: %w", err) + } + result.State = encoded + return &result, nil + } + if !json.Valid([]byte(text)) { + return nil, fmt.Errorf("exchange: the edited System One state is no longer valid JSON") + } + result.State = json.RawMessage(text) + return &result, nil +} + +// SystemOneUncarriedEdits describes the prompt edits ApplyToSystemOneRequest +// does not apply, in a stable order, or nil when every edit was carried. +func SystemOneUncarriedEdits(p *pluginapi.Prompt) []string { + if p == nil { + return nil + } + changes := p.Changes() + var uncarried []string + for id, kind := range changes.Messages { + if id != SystemOneStateMessageID { + uncarried = append(uncarried, fmt.Sprintf("%s message %q", kind, id)) + } + } + for name := range changes.Params { + uncarried = append(uncarried, fmt.Sprintf("parameter %q", name)) + } + sort.Strings(uncarried) + return uncarried +} diff --git a/internal/plugins/exchange/systemone_request_test.go b/internal/plugins/exchange/systemone_request_test.go new file mode 100644 index 000000000..b2ec317e1 --- /dev/null +++ b/internal/plugins/exchange/systemone_request_test.go @@ -0,0 +1,120 @@ +package exchange + +import ( + "testing" + + "github.com/goccy/go-json" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/pluginapi" +) + +func TestFromSystemOneRequestExposesStateAsUserMessage(t *testing.T) { + tests := []struct { + name string + state string + want string + }{ + {name: "string state", state: `"charged twice"`, want: "charged twice"}, + {name: "record state", state: `{"ticket":"charged twice"}`, want: `{"ticket":"charged twice"}`}, + {name: "no state", state: ``, want: ""}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + p, err := FromSystemOneRequest(&core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(tt.state)}) + require.NoError(t, err) + require.Len(t, p.Messages, 1) + + assert.Equal(t, SystemOneStateMessageID, p.Messages[0].ID) + assert.Equal(t, pluginapi.RoleUser, p.Messages[0].Role) + assert.Equal(t, tt.want, p.Messages[0].Text()) + assert.Equal(t, "kev-latest", p.Params.Model) + assert.False(t, p.Changes().Dirty) + }) + } +} + +func TestApplyToSystemOneRequest(t *testing.T) { + tests := []struct { + name string + state string + edit func(p *pluginapi.Prompt) error + wantState string + wantErr string + }{ + { + name: "untouched", + state: `"John"`, + edit: func(*pluginapi.Prompt) error { return nil }, + wantState: `"John"`, + }, + { + name: "string state stays a string", + state: `"John \"J\" Doe"`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `[PERSON] "J"`) }, + wantState: `"[PERSON] \"J\""`, + }, + { + name: "record state stays a record", + state: `{"name":"John"}`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `{"name":"[PERSON]"}`) }, + wantState: `{"name":"[PERSON]"}`, + }, + { + name: "record state must stay valid JSON", + state: `{"name":"John"}`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `name: [PERSON]`) }, + wantErr: "no longer valid JSON", + }, + { + name: "state cannot be removed", + state: `"John"`, + edit: func(p *pluginapi.Prompt) error { return p.Remove(SystemOneStateMessageID) }, + wantErr: "state was removed", + }, + { + name: "uncarried edits are not applied", + state: `"John"`, + edit: func(p *pluginapi.Prompt) error { + p.Insert(0, pluginapi.TextMessage(pluginapi.RoleSystem, "be safe")) + p.SetParam("temperature", 0.1) + return nil + }, + wantState: `"John"`, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + original := &core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(tt.state)} + p, err := FromSystemOneRequest(original) + require.NoError(t, err) + require.NoError(t, tt.edit(p)) + + got, err := ApplyToSystemOneRequest(original, p) + if tt.wantErr != "" { + require.Error(t, err) + assert.Contains(t, err.Error(), tt.wantErr) + return + } + require.NoError(t, err) + assert.JSONEq(t, tt.wantState, string(got.State)) + assert.Equal(t, "kev-latest", got.Model) + assert.JSONEq(t, tt.state, string(original.State), "the original request must not change") + }) + } +} + +func TestSystemOneUncarriedEdits(t *testing.T) { + p, err := FromSystemOneRequest(&core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(`"John"`)}) + require.NoError(t, err) + assert.Nil(t, SystemOneUncarriedEdits(p)) + + require.NoError(t, p.SetText(SystemOneStateMessageID, 0, "[PERSON]")) + assert.Nil(t, SystemOneUncarriedEdits(p), "a state edit is carried") + + id := p.Insert(0, pluginapi.TextMessage(pluginapi.RoleSystem, "be safe")) + p.SetParam("temperature", 0.1) + assert.Equal(t, []string{`inserted message "` + id + `"`, `parameter "temperature"`}, SystemOneUncarriedEdits(p)) +} diff --git a/internal/providers/jev/jev.go b/internal/providers/jev/jev.go index 32b3ea576..93dd4acef 100644 --- a/internal/providers/jev/jev.go +++ b/internal/providers/jev/jev.go @@ -3,8 +3,8 @@ // API. System One is a decision API rather than a text-generation one: a // request carries a state and a map of typed questions (noul, choice, score) // and the answer is a calibrated probability per question. It has no -// OpenAI-compatible surface, so the gateway reaches it through native -// passthrough at /p/jev/systemone. +// OpenAI-compatible surface, so the gateway forwards it natively, at +// POST /v1/systemone or through passthrough at /p/jev/systemone. package jev import ( @@ -110,12 +110,12 @@ func (p *Provider) Embeddings(_ context.Context, _ *core.EmbeddingRequest) (*cor } func unsupported(surface string) error { - return core.NewInvalidRequestError("jev does not support "+surface+"; send System One requests to /p/jev/systemone", nil) + return core.NewInvalidRequestError("jev does not support "+surface+"; it answers System One decision requests, which GoModel does not translate: send them to POST /v1/systemone", nil) } -// Passthrough forwards a System One request as the client wrote it. It is the -// only way to reach the evaluation endpoint, since the request and answer -// shapes have no OpenAI equivalent. +// Passthrough forwards a System One request as the client wrote it. Both +// /v1/systemone and /p/jev/... reach the evaluation endpoint through it, +// since the request and answer shapes have no OpenAI equivalent. func (p *Provider) Passthrough(ctx context.Context, req *core.PassthroughRequest) (*core.PassthroughResponse, error) { if req == nil { return nil, core.NewInvalidRequestError("passthrough request is required", nil) diff --git a/internal/providers/jev/jev_test.go b/internal/providers/jev/jev_test.go index 43c603028..9475fcaf8 100644 --- a/internal/providers/jev/jev_test.go +++ b/internal/providers/jev/jev_test.go @@ -45,7 +45,7 @@ func TestUnsupportedCapabilities_ReturnInvalidRequestErrors(t *testing.T) { _, err := provider.ChatCompletion(context.Background(), &core.ChatRequest{Model: "jev-latest"}) providertest.AssertUnsupported(t, err) - assert.Contains(t, err.Error(), "/p/jev/systemone") + assert.Contains(t, err.Error(), "/v1/systemone") _, err = provider.StreamChatCompletion(context.Background(), &core.ChatRequest{Model: "jev-latest"}) providertest.AssertUnsupported(t, err) _, err = provider.Responses(context.Background(), &core.ResponsesRequest{Model: "jev-latest"}) diff --git a/internal/server/http.go b/internal/server/http.go index 560683152..09a7ad174 100644 --- a/internal/server/http.go +++ b/internal/server/http.go @@ -496,6 +496,9 @@ func New(provider core.RoutableProvider, cfg *Config) *Server { e.POST("/v1/audio/translations", handler.AudioTranslations) e.POST("/v1/images/generations", handler.ImageGenerations) e.POST("/v1/images/edits", handler.ImageEdits) + // System One decisions (Jev / Kev). The handler answers 404 until a jev + // provider is configured, so the route costs nothing otherwise. + e.POST("/v1/systemone", handler.SystemOne) if cfg == nil || cfg.RealtimeEnabled { e.GET("/v1/realtime", handler.Realtime) e.POST("/v1/realtime/calls", handler.RealtimeCalls) diff --git a/internal/server/messages_native.go b/internal/server/messages_native.go index d460297a1..d451a9319 100644 --- a/internal/server/messages_native.go +++ b/internal/server/messages_native.go @@ -3,8 +3,8 @@ package server import ( "bytes" "context" - // encoding/json rather than goccy: rewriteMessagesModel needs the - // decoder's InputOffset to splice the model value in place. + // encoding/json rather than goccy: replaceTopLevelMember needs the + // decoder's InputOffset to splice a value in place. "encoding/json" "errors" "io" @@ -138,6 +138,17 @@ func rewriteMessagesModel(body []byte, model string) ([]byte, error) { if strings.TrimSpace(model) == "" { return body, nil } + encoded, err := json.Marshal(model) + if err != nil { + return nil, err + } + return replaceTopLevelMember(body, "model", encoded) +} + +// replaceTopLevelMember returns body with the value of its top-level key +// member replaced by value, splicing only those bytes. The body is returned +// unchanged when the member is absent or already holds value. +func replaceTopLevelMember(body []byte, key string, value []byte) ([]byte, error) { dec := json.NewDecoder(bytes.NewReader(body)) tok, err := dec.Token() if err != nil { @@ -146,45 +157,36 @@ func rewriteMessagesModel(body []byte, model string) ([]byte, error) { if delim, ok := tok.(json.Delim); !ok || delim != '{' { return nil, errors.New("request body is not a JSON object") } - // Walk every top-level member and remember the span of the last "model" + // Walk every top-level member and remember the span of the last matching // value: decoders keep the last duplicate member, so that is the one the - // resolved model came from and the one to rewrite. - var modelRaw json.RawMessage - var modelEnd int64 + // gateway read and the one to rewrite. + var found json.RawMessage + var foundEnd int64 for dec.More() { keyTok, err := dec.Token() if err != nil { return nil, err } - key, _ := keyTok.(string) + name, _ := keyTok.(string) var raw json.RawMessage if err := dec.Decode(&raw); err != nil { return nil, err } - if key != "model" { + if name != key { continue } - modelRaw = raw - modelEnd = dec.InputOffset() + found = raw + foundEnd = dec.InputOffset() } - if modelRaw == nil { + if found == nil || bytes.Equal(found, value) { return body, nil } - var current string - _ = json.Unmarshal(modelRaw, ¤t) - if current == model { - return body, nil - } - encoded, err := json.Marshal(model) - if err != nil { - return nil, err - } - // The model value is a scalar, so modelRaw holds its exact source bytes - // and modelEnd points just past them. - start := modelEnd - int64(len(modelRaw)) - rewritten := make([]byte, 0, int64(len(body))-int64(len(modelRaw))+int64(len(encoded))) + // The decoder hands back the value's exact source bytes, and foundEnd + // points just past them. + start := foundEnd - int64(len(found)) + rewritten := make([]byte, 0, int64(len(body))-int64(len(found))+int64(len(value))) rewritten = append(rewritten, body[:start]...) - rewritten = append(rewritten, encoded...) - rewritten = append(rewritten, body[modelEnd:]...) + rewritten = append(rewritten, value...) + rewritten = append(rewritten, body[foundEnd:]...) return rewritten, nil } diff --git a/internal/server/model_validation.go b/internal/server/model_validation.go index 4a2d18d60..2d85ebcbb 100644 --- a/internal/server/model_validation.go +++ b/internal/server/model_validation.go @@ -103,7 +103,7 @@ func deriveWorkflowWithPolicy( } return workflow, nil - case core.OperationChatCompletions, core.OperationResponses, core.OperationEmbeddings: + case core.OperationChatCompletions, core.OperationResponses, core.OperationEmbeddings, core.OperationSystemOne: workflow.Mode = core.ExecutionModeTranslated if desc.BodyMode != core.BodyModeJSON { // Responses lifecycle routes (GET/DELETE /v1/responses/{id}, diff --git a/internal/server/systemone_handler.go b/internal/server/systemone_handler.go new file mode 100644 index 000000000..428fa0c0b --- /dev/null +++ b/internal/server/systemone_handler.go @@ -0,0 +1,227 @@ +package server + +import ( + "bytes" + "fmt" + "io" + "log/slog" + "net/http" + "strings" + + "github.com/goccy/go-json" + "github.com/labstack/echo/v5" + + "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/gateway" + "github.com/enterpilot/gomodel/internal/plugins" +) + +const ( + systemOnePath = "/v1/systemone" + systemOneEndpoint = "systemone" + // jevProviderType serves TypeSafe's hosted Jev API and self-hosted Kev + // servers; configuring one is what makes /v1/systemone available. + jevProviderType = "jev" +) + +// systemOneProviderTypes are the provider types that serve the System One API +// natively. OpenRouter serves Jev at the same path with the same request and +// answer shapes, so it can back a virtual model next to a jev provider. +var systemOneProviderTypes = map[string]struct{}{ + jevProviderType: {}, + "openrouter": {}, +} + +// SystemOne handles POST /v1/systemone. +// +// It accepts TypeSafe's System One decision request (a state plus a map of +// typed questions) and forwards it natively: the body reaches the provider +// unchanged apart from the routed model name and guardrail edits to the +// state. Requests are never translated to another API, so a model whose +// provider has no System One API is rejected. +// +// @Summary Evaluate a System One decision request (Jev / Kev) +// @Description Available when a jev provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated. +// @Tags systemone +// @Accept json +// @Produce json +// @Security BearerAuth +// @Param request body object true "System One request: model, state, and questions" +// @Success 200 {object} object "System One answers, in the provider's shape" +// @Failure 400 {object} core.OpenAIErrorEnvelope +// @Failure 401 {object} core.OpenAIErrorEnvelope +// @Failure 404 {object} core.OpenAIErrorEnvelope +// @Failure 429 {object} core.OpenAIErrorEnvelope +// @Failure 502 {object} core.OpenAIErrorEnvelope +// @Router /v1/systemone [post] +func (h *Handler) SystemOne(c *echo.Context) error { + return h.translatedInference().SystemOne(c) +} + +// SystemOne resolves, guards, and forwards one System One request. +func (s *translatedInferenceService) SystemOne(c *echo.Context) error { + if !s.systemOneAvailable() { + return handleError(c, core.NewNotFoundError("POST "+systemOnePath+" is available only when a jev provider is configured")) + } + body, err := requestBodyBytes(c) + if err != nil { + return handleError(c, core.NewInvalidRequestError("invalid request body: "+err.Error(), err)) + } + var req core.SystemOneRequest + if err := json.Unmarshal(body, &req); err != nil { + return handleError(c, core.NewInvalidRequestError("invalid request body: "+err.Error(), err)) + } + if strings.TrimSpace(req.Model) == "" { + return handleError(c, core.NewInvalidRequestError("model is required", nil).WithParam("model")) + } + + workflow, err := s.systemOneWorkflow(c, req.Model) + if err != nil { + return handleError(c, err) + } + ctx := c.Request().Context() + resolution := workflow.Resolution + if s.modelAuthorizer != nil { + if err := s.modelAuthorizer.ValidateModelAccess(ctx, resolution.ResolvedSelector); err != nil { + return handleError(c, err) + } + } + providerType := strings.TrimSpace(resolution.ProviderType) + if _, ok := systemOneProviderTypes[providerType]; !ok { + return handleError(c, systemOneUnsupportedModelError(c, resolution)) + } + + body, err = s.guardSystemOneState(c, workflow, &req, body) + if err != nil { + return handleError(c, err) + } + model := resolution.ResolvedSelector.Model + if body, err = rewriteMessagesModel(body, model); err != nil { + return handleError(c, core.NewInvalidRequestError("invalid request body: "+err.Error(), err)) + } + return s.dispatchSystemOne(c, workflow, model, body) +} + +// systemOneAvailable reports whether a jev provider is configured. It is +// checked per request rather than at route registration so a provider added +// at runtime makes the endpoint available without a restart. +func (s *translatedInferenceService) systemOneAvailable() bool { + named, ok := s.provider.(core.ProviderTypeNameResolver) + return ok && strings.TrimSpace(named.GetProviderNameForType(jevProviderType)) != "" +} + +// systemOneWorkflow returns the request's workflow with its model resolved. +// The workflow middleware resolves it from the body; a request that reached +// the handler without one (the body was not parsed there) is resolved here, +// so virtual models and workflow policy apply either way. +func (s *translatedInferenceService) systemOneWorkflow(c *echo.Context, model string) (*core.Workflow, error) { + if workflow := core.GetWorkflow(c.Request().Context()); workflow != nil && workflow.Resolution != nil { + return workflow, nil + } + resolution, err := resolveAndStoreRequestModelResolution(c, s.provider, s.modelResolver, nil, model, "") + if err != nil { + return nil, err + } + workflow, err := translatedWorkflowForRequest(c, resolution, s.workflowPolicyResolver) + if err != nil { + return nil, err + } + storeWorkflow(c, workflow) + return workflow, nil +} + +// systemOneUnsupportedModelError explains a request whose model routes to a +// provider without the System One API. It is also logged: a virtual model +// that sends System One traffic to a chat model is an operator mistake the +// caller cannot fix. +func systemOneUnsupportedModelError(c *echo.Context, resolution *core.RequestModelResolution) error { + requested := resolution.RequestedQualifiedModel() + resolved := resolution.ResolvedQualifiedModel() + slog.Warn("System One request routed to a provider without the System One API", + "request_id", requestIDFromContextOrHeader(c.Request()), + "requested_model", requested, + "resolved_model", resolved, + "provider_type", resolution.ProviderType, + ) + target := fmt.Sprintf("%q", requested) + if resolved != requested { + target += fmt.Sprintf(" (resolved to %q)", resolved) + } + return core.NewInvalidRequestError(fmt.Sprintf( + "model %s is served by a %s provider, which has no System One API; %s forwards requests natively and does not translate them to other APIs, so use a jev model", + target, resolution.ProviderType, systemOnePath, + ), nil).WithParam("model") +} + +// guardSystemOneState runs the workflow's prompt guardrails over the state +// and returns the body carrying their edits. A guardrail that answers the +// request itself blocks it instead: its answer is chat text, and a System +// One caller expects typed answers. +func (s *translatedInferenceService) guardSystemOneState(c *echo.Context, workflow *core.Workflow, req *core.SystemOneRequest, body []byte) ([]byte, error) { + patcher, ok := s.translatedRequestPatcher.(gateway.SystemOneRequestPatcher) + if !ok || !workflow.GuardrailsEnabled() { + return body, nil + } + patched, err := patcher.PatchSystemOneRequest(c.Request().Context(), req) + s.recordGuardrailOutcomes(c) + if err != nil { + if short := shortCircuitOf(err); short != nil { + return nil, plugins.BlockError(short.Decision, http.StatusBadRequest) + } + return nil, err + } + if patched == nil || patched == req || bytes.Equal(patched.State, req.State) { + return body, nil + } + rewritten, err := replaceTopLevelMember(body, "state", patched.State) + if err != nil { + return nil, core.NewInvalidRequestError("invalid request body: "+err.Error(), err) + } + return rewritten, nil +} + +// dispatchSystemOne forwards the body to the resolved provider and relays its +// answer unchanged, with admission, audit, and usage accounting. +func (s *translatedInferenceService) dispatchSystemOne(c *echo.Context, workflow *core.Workflow, model string, body []byte) error { + passthroughProvider, ok := s.provider.(core.RoutablePassthrough) + if !ok { + return handleError(c, core.NewInvalidRequestError("provider passthrough is not supported by the current provider router", nil)) + } + s.observeLiveProviderAttempts(c, workflow) + + adm, err := enforceAdmission(c, s.rateLimiter, s.budgetChecker, rateLimitRouteFromWorkflow(workflow)) + if err != nil { + return handleError(c, err) + } + defer adm.release() + ctx := adm.dispatchContext(c.Request().Context()) + + resolution := workflow.Resolution + providerType := strings.TrimSpace(resolution.ProviderType) + providerName := strings.TrimSpace(resolution.ProviderName) + resp, err := passthroughProvider.Passthrough(ctx, providerType, &core.PassthroughRequest{ + Method: http.MethodPost, + Endpoint: systemOneEndpoint, + Operation: "systemone", + Model: model, + Body: io.NopCloser(bytes.NewReader(body)), + Headers: buildPassthroughHeaders(ctx, c.Request().Header), + ProviderName: providerName, + }) + if err != nil { + return handleError(c, err) + } + + auditlog.EnrichEntryWithWorkflow(c, workflow) + auditlog.EnrichEntryWithResolvedRoute(c, resolution.ResolvedQualifiedModel(), providerType, providerName) + info := &core.PassthroughRouteInfo{ + Provider: providerType, + ProviderName: providerName, + NormalizedEndpoint: systemOneEndpoint, + SemanticOperation: "systemone", + AuditPath: systemOnePath, + Model: model, + } + return proxyPassthroughResponse(c, s.logger, s.usageLogger, s.pricingResolver, providerType, providerName, systemOneEndpoint, info, resp) +} diff --git a/internal/server/systemone_handler_test.go b/internal/server/systemone_handler_test.go new file mode 100644 index 000000000..b06a53145 --- /dev/null +++ b/internal/server/systemone_handler_test.go @@ -0,0 +1,244 @@ +package server + +import ( + "context" + "io" + "net/http" + "strings" + "testing" + + "github.com/goccy/go-json" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/echotest" + "github.com/enterpilot/gomodel/internal/usage" +) + +const ( + systemOneQuestions = `{"refund":{"type":"noul","instructions":"Is the customer asking for money back?"}}` + systemOneAnswer = `{"model":"kev-1.0","answers":{"refund":{"type":"noul","noul":0.98}},"usage":{"input_tokens":275,"output_tokens":20}}` +) + +func systemOneBody(model string) string { + return `{"model":"` + model + `","state":"I was charged twice for card 4111.","questions":` + systemOneQuestions + `}` +} + +// systemOneAliasResolver maps virtual model names to concrete selectors. +type systemOneAliasResolver map[string]core.ModelSelector + +func (r systemOneAliasResolver) ResolveModel(requested core.RequestedModelSelector) (core.ModelSelector, bool, error) { + if selector, ok := r[requested.RequestedQualifiedModel()]; ok { + return selector, true, nil + } + selector, err := requested.Normalize() + return selector, false, err +} + +// newSystemOneProvider configures a local Kev server (type jev, named kev), +// OpenRouter, and an OpenAI chat model, answering every passthrough with body. +func newSystemOneProvider(body string) *mockProvider { + return &mockProvider{ + supportedModels: []string{"kev-latest", "typesafe/jev-1.13", "gpt-5-mini"}, + providerTypes: map[string]string{ + "kev/kev-latest": "jev", + "openrouter/typesafe/jev-1.13": "openrouter", + "openai/gpt-5-mini": "openai", + }, + providerNames: map[string]string{ + "kev/kev-latest": "kev", + "openrouter/typesafe/jev-1.13": "openrouter", + "openai/gpt-5-mini": "openai", + }, + passthroughResponse: &core.PassthroughResponse{ + StatusCode: http.StatusOK, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(body)), + }, + } +} + +var systemOneAliases = systemOneAliasResolver{ + "decider": {Provider: "kev", Model: "kev-latest"}, + "chatty": {Provider: "openai", Model: "gpt-5-mini"}, +} + +func forwardedSystemOneBody(t *testing.T, provider *mockProvider) map[string]any { + t.Helper() + require.NotNil(t, provider.lastPassthroughReq) + raw, err := io.ReadAll(provider.lastPassthroughReq.Body) + require.NoError(t, err) + var body map[string]any + require.NoError(t, json.Unmarshal(raw, &body)) + + return body +} + +// Without a jev provider the endpoint does not exist, whatever the model. +func TestSystemOne_UnavailableWithoutJevProvider(t *testing.T) { + provider := &mockProvider{ + supportedModels: []string{"gpt-5-mini"}, + providerTypes: map[string]string{"openai/gpt-5-mini": "openai"}, + providerNames: map[string]string{"openai/gpt-5-mini": "openai"}, + } + handler := NewHandler(provider, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("gpt-5-mini")) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusNotFound, rec.Code) + assert.Contains(t, rec.Body.String(), "jev provider") + assert.Nil(t, provider.lastPassthroughReq) +} + +// A virtual model resolves to its System One target, and the body reaches the +// provider unchanged except for the concrete model name. +func TestSystemOne_ForwardsNativelyAndRecordsUsage(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + handler := newHandlerWithAuthorizer(provider, nil, usageLogger, nil, systemOneAliases, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("decider")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.JSONEq(t, systemOneAnswer, rec.Body.String()) + + assert.Equal(t, "jev", provider.lastPassthroughProvider) + assert.Equal(t, "systemone", provider.lastPassthroughReq.Endpoint) + assert.Equal(t, "kev", provider.lastPassthroughReq.ProviderName) + body := forwardedSystemOneBody(t, provider) + assert.Equal(t, "kev-latest", body["model"]) + assert.Equal(t, "I was charged twice for card 4111.", body["state"]) + questions, err := json.Marshal(body["questions"]) + require.NoError(t, err) + assert.JSONEq(t, systemOneQuestions, string(questions)) + + require.Len(t, usageLogger.entries, 1) + entry := usageLogger.entries[0] + assert.Equal(t, 275, entry.InputTokens) + assert.Equal(t, 20, entry.OutputTokens) + assert.Equal(t, "/v1/systemone", entry.Endpoint) + assert.Equal(t, "jev", entry.Provider) + assert.Equal(t, "kev", entry.ProviderName) + assert.Equal(t, "kev-1.0", entry.Model, "usage is recorded under the model that answered") +} + +// OpenRouter serves Jev at the same path, so it is a native target too; its +// reported cost is kept with the usage entry. +func TestSystemOne_ForwardsOpenRouterJevNatively(t *testing.T) { + answer := `{"id":"gen-dec-1","model":"typesafe/jev-1.13-20260917","answers":{},"usage":{"input_tokens":275,"output_tokens":20,"cost":0.00003}}` + provider := newSystemOneProvider(answer) + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + handler := newHandlerWithAuthorizer(provider, nil, usageLogger, nil, nil, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("openrouter/typesafe/jev-1.13")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, "openrouter", provider.lastPassthroughProvider) + assert.Equal(t, "typesafe/jev-1.13", forwardedSystemOneBody(t, provider)["model"]) + require.Len(t, usageLogger.entries, 1) + assert.InDelta(t, 0.00003, usageLogger.entries[0].RawData["cost"], 1e-12) +} + +// The endpoint never translates: a model on a provider without the System One +// API is rejected with an explanation, whether named directly or through a +// virtual model. +func TestSystemOne_RejectsModelsWithoutSystemOneAPI(t *testing.T) { + for _, model := range []string{"openai/gpt-5-mini", "chatty"} { + t.Run(model, func(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, systemOneAliases, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody(model)) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "no System One API") + assert.Contains(t, rec.Body.String(), "does not translate") + assert.Nil(t, provider.lastPassthroughReq) + }) + } +} + +func TestSystemOne_RequiresModel(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := NewHandler(provider, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", `{"state":"hi","questions":{}}`) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "model is required") +} + +// stateRedactingPatcher stands in for an anonymizing guardrail. +type stateRedactingPatcher struct{} + +func (stateRedactingPatcher) PatchChatRequest(_ context.Context, req *core.ChatRequest) (*core.ChatRequest, error) { + return req, nil +} + +func (stateRedactingPatcher) PatchResponsesRequest(_ context.Context, req *core.ResponsesRequest) (*core.ResponsesRequest, error) { + return req, nil +} + +func (stateRedactingPatcher) PatchSystemOneRequest(_ context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) { + patched := *req + patched.State = json.RawMessage(strings.ReplaceAll(string(req.State), "4111", "[card]")) + return &patched, nil +} + +// Guardrails see the state and their edits reach the provider; the rest of +// the body is untouched. +func TestSystemOne_GuardrailsEditState(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, systemOneAliases, nil, nil, nil, stateRedactingPatcher{}) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("decider")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + body := forwardedSystemOneBody(t, provider) + assert.Equal(t, "I was charged twice for card [card].", body["state"]) + assert.Equal(t, "kev-latest", body["model"]) +} + +// Through the full middleware stack a System One call is audited under its +// own path with the requested and resolved routes, and its usage recorded. +func TestSystemOne_AuditsAndRecordsUsageThroughServer(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + auditLogger := &capturingAuditLogger{config: auditlog.Config{Enabled: true, LogBodies: true}} + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + srv := New(provider, &Config{ + AuditLogger: auditLogger, + UsageLogger: usageLogger, + ModelResolver: systemOneAliases, + }) + + rec := postJSON(t, srv, "/v1/systemone", systemOneBody("decider")) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + require.Len(t, auditLogger.entries, 1) + entry := auditLogger.entries[0] + assert.Equal(t, "/v1/systemone", entry.Path) + assert.Equal(t, http.StatusOK, entry.StatusCode) + assert.Equal(t, "decider", entry.RequestedModel) + assert.Equal(t, "kev/kev-latest", entry.ResolvedModel) + assert.True(t, entry.AliasUsed) + assert.Equal(t, "jev", entry.Provider) + assert.Equal(t, "kev", entry.ProviderName) + require.NotNil(t, entry.Data) + requestBody, err := json.Marshal(entry.Data.RequestBody) + require.NoError(t, err) + assert.Contains(t, string(requestBody), "charged twice") + responseBody, err := json.Marshal(entry.Data.ResponseBody) + require.NoError(t, err) + assert.Contains(t, string(responseBody), "noul") + + require.Len(t, usageLogger.entries, 1) + assert.Equal(t, "/v1/systemone", usageLogger.entries[0].Endpoint) + assert.Equal(t, entry.RequestID, usageLogger.entries[0].RequestID) +} diff --git a/web/dashboard/messages/de.json b/web/dashboard/messages/de.json index b4697d221..ad99c1e20 100644 --- a/web/dashboard/messages/de.json +++ b/web/dashboard/messages/de.json @@ -233,6 +233,7 @@ "audit_type_embeddings": "Embeddings", "audit_type_audio": "Audio", "audit_type_images": "Bilder", + "audit_type_systemone": "System One", "audit_type_batches": "Batches & Dateien", "audit_type_realtime": "Realtime", "audit_type_passthrough": "Passthrough", diff --git a/web/dashboard/messages/en.json b/web/dashboard/messages/en.json index ff304603f..b8db00471 100644 --- a/web/dashboard/messages/en.json +++ b/web/dashboard/messages/en.json @@ -233,6 +233,7 @@ "audit_type_embeddings": "Embeddings", "audit_type_audio": "Audio", "audit_type_images": "Images", + "audit_type_systemone": "System One", "audit_type_batches": "Batches & files", "audit_type_realtime": "Realtime", "audit_type_passthrough": "Passthrough", diff --git a/web/dashboard/messages/pl.json b/web/dashboard/messages/pl.json index 3e9a218a3..783aabee5 100644 --- a/web/dashboard/messages/pl.json +++ b/web/dashboard/messages/pl.json @@ -235,6 +235,7 @@ "audit_type_embeddings": "Embeddingi", "audit_type_audio": "Audio", "audit_type_images": "Obrazy", + "audit_type_systemone": "System One", "audit_type_batches": "Batche i pliki", "audit_type_realtime": "Realtime", "audit_type_passthrough": "Passthrough", diff --git a/web/dashboard/messages/zh-CN.json b/web/dashboard/messages/zh-CN.json index dbf13cfc7..0c60c7e70 100644 --- a/web/dashboard/messages/zh-CN.json +++ b/web/dashboard/messages/zh-CN.json @@ -224,6 +224,7 @@ "audit_type_embeddings": "嵌入", "audit_type_audio": "音频", "audit_type_images": "图像", + "audit_type_systemone": "System One", "audit_type_batches": "批处理与文件", "audit_type_realtime": "实时", "audit_type_passthrough": "透传", diff --git a/web/dashboard/src/pages/audit-logs/audit-operations.js b/web/dashboard/src/pages/audit-logs/audit-operations.js index 7c493dbd1..7bbd43321 100644 --- a/web/dashboard/src/pages/audit-logs/audit-operations.js +++ b/web/dashboard/src/pages/audit-logs/audit-operations.js @@ -17,6 +17,7 @@ export const AUDIT_TYPES = [ operations: ["audio_speech", "audio_transcriptions", "audio_translations"], }, { key: "images", label: () => m.audit_type_images(), operations: ["image_generations", "image_edits"] }, + { key: "systemone", label: () => m.audit_type_systemone(), operations: ["systemone"] }, { key: "batches", label: () => m.audit_type_batches(), operations: ["batches", "files"] }, { key: "realtime", label: () => m.audit_type_realtime(), operations: ["realtime"] }, { key: "passthrough", label: () => m.audit_type_passthrough(), operations: ["provider_passthrough"] }, @@ -35,6 +36,7 @@ const EXACT_PATHS = { "/v1/audio/translations": "audio", "/v1/images/generations": "images", "/v1/images/edits": "images", + "/v1/systemone": "systemone", "/v1/realtime": "realtime", "/v1/realtime/calls": "realtime", "/v1/realtime/client_secrets": "realtime", diff --git a/web/dashboard/tests/audit-operations.test.js b/web/dashboard/tests/audit-operations.test.js index 1b8914269..e94d99faf 100644 --- a/web/dashboard/tests/audit-operations.test.js +++ b/web/dashboard/tests/audit-operations.test.js @@ -20,6 +20,8 @@ test("auditTypeForPath mirrors the gateway endpoint classification", () => { ["/v1/files/f_1/content", "batches"], ["/v1/audio/transcriptions?x=1", "audio"], ["/v1/images/edits/", "images"], + ["/v1/systemone", "systemone"], + ["/v1/systemone/permute", ""], ["/v1/realtime/translations/calls", "realtime"], ["/mcp", "mcp"], ["/mcp/github", "mcp"], From 3e2c79f14a213f496bf4b89baf0c37edb4e92998 Mon Sep 17 00:00:00 2001 From: "Jakub A. W" Date: Sat, 26 Sep 2026 19:06:16 +0200 Subject: [PATCH 2/3] feat(openrouter): serve System One decision models natively --- cmd/gomodel/docs/docs.go | 2 +- config/config.example.yaml | 5 +- docs/openapi.json | 2 +- docs/providers/jev.mdx | 44 ++++++--- docs/providers/overview.mdx | 3 +- internal/providers/openrouter/openrouter.go | 19 +++- .../providers/openrouter/openrouter_test.go | 18 +++- .../providers/registry_normalization_test.go | 25 +++++ internal/providers/router_models.go | 14 +++ internal/server/http.go | 2 +- internal/server/systemone_handler.go | 82 +++++++++++----- internal/server/systemone_handler_test.go | 94 ++++++++++++++++++- 12 files changed, 257 insertions(+), 53 deletions(-) diff --git a/cmd/gomodel/docs/docs.go b/cmd/gomodel/docs/docs.go index 769ec2064..dacf4d64d 100644 --- a/cmd/gomodel/docs/docs.go +++ b/cmd/gomodel/docs/docs.go @@ -7506,7 +7506,7 @@ const docTemplate = `{ }, "/v1/systemone": { "post": { - "description": "Available when a jev provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", + "description": "Available when a jev or openrouter provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", "consumes": [ "application/json" ], diff --git a/config/config.example.yaml b/config/config.example.yaml index 9b0a08d46..d1213070e 100644 --- a/config/config.example.yaml +++ b/config/config.example.yaml @@ -592,8 +592,9 @@ providers: api_key: "${JEV_API_KEY}" # base_url defaults to "https://api.typesafe.ai". TypeSafe's System One # API is a decision API with no OpenAI-compatible surface: configuring - # this provider enables POST /v1/systemone, which forwards requests - # natively (point the TypeSafe SDK at the gateway root). A self-hosted Kev + # this provider (or openrouter, which serves Jev natively) enables + # POST /v1/systemone, which forwards requests natively (point the + # TypeSafe SDK at the gateway root). A self-hosted Kev # server speaks the same API without authentication: set base_url # (e.g. "http://localhost:8009") and omit api_key. Name it "kev" to see # that name in logs and usage; no separate provider type is needed. diff --git a/docs/openapi.json b/docs/openapi.json index 6f7ef6e20..efa2de15f 100644 --- a/docs/openapi.json +++ b/docs/openapi.json @@ -11195,7 +11195,7 @@ }, "/v1/systemone": { "post": { - "description": "Available when a jev provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", + "description": "Available when a jev or openrouter provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", "tags": [ "systemone" ], diff --git a/docs/providers/jev.mdx b/docs/providers/jev.mdx index 4004393e6..ca23eabce 100644 --- a/docs/providers/jev.mdx +++ b/docs/providers/jev.mdx @@ -23,7 +23,7 @@ There are three question types: The API is not OpenAI-compatible, and its answers have no chat equivalent, so GoModel forwards it natively instead of translating it: `POST /v1/systemone` -is available as soon as a `jev` provider is configured, and +is available as soon as a `jev` or `openrouter` provider is configured, and [passthrough](/features/passthrough-api) at `/p/jev/...` reaches every other upstream route. Chat, `/responses`, and `/v1/embeddings` return `invalid_request_error` for `jev` models, pointing at `/v1/systemone`. @@ -147,21 +147,39 @@ GoModel: 4. Relays the answer unchanged, and records it in the audit log (as a **System One** request) and in usage. -The endpoint never translates. If a model resolves to a provider without the -System One API, for example a virtual model pointing at a chat model, the -request fails with `400 invalid_request_error` explaining why, and the gateway -logs a warning. Without a `jev` provider, the route answers `404`. - -Two targets serve the API natively: `jev` providers (hosted Jev or a Kev -server) and, once the endpoint is available, -[OpenRouter](https://openrouter.ai/docs/guides/community/jev), which serves -Jev at the same path (`openrouter/typesafe/jev-1.13`). A virtual model can therefore front a local -Kev server and OpenRouter's `typesafe/jev-1.13` together. OpenRouter adds `id`, -`provider`, and `usage.cost` to the answer, and GoModel records the reported -cost. +The endpoint never translates. A model that cannot answer System One fails +with `400 invalid_request_error` explaining why, and the gateway logs a +warning: one on a provider without the API, or one the catalog lists as a +chat, embedding, or other generation model, such as a virtual model pointing +at a chat model. Without a `jev` or `openrouter` provider, the route answers +`404`. Response caching and failover do not apply to this endpoint yet. +### Through OpenRouter + +[OpenRouter serves Jev natively](https://openrouter.ai/docs/guides/community/jev) +at the same path, so an OpenRouter key alone is enough: + +```bash +OPENROUTER_API_KEY=sk-or-... +``` + +OpenRouter's decision models appear in `GET /v1/models` as utility models, +priced from OpenRouter's listing: `openrouter/typesafe/jev-1.13`, +`openrouter/~typesafe/jev-latest` (tracks the newest Jev), and other decision +models such as Kev 4B (`openrouter/jaredpalmer/kev-4b`). Name them that way in +`model`. OpenRouter accepts `jev-latest` itself, but GoModel routes on its +catalog IDs, so to keep the TypeSafe SDK's plain `jev-latest` working, add a +[virtual model](/features/virtual-models) `jev-latest` that targets +`openrouter/~typesafe/jev-latest`. Chat models on the same provider are +rejected, since they have no System One API. + +The answer carries OpenRouter's `id`, `provider`, and `usage.cost`, and GoModel +records that reported cost with the request's usage. With a `jev` provider +configured as well, one virtual model can front a local Kev server and +OpenRouter's Jev together. + ### Guardrails Guardrails see `state` as a single user message: a string state as its text, diff --git a/docs/providers/overview.mdx b/docs/providers/overview.mdx index 93506fd8d..d8ea5d3bb 100644 --- a/docs/providers/overview.mdx +++ b/docs/providers/overview.mdx @@ -214,7 +214,8 @@ support, not every individual model capability exposed by an upstream provider. surface, so GoModel forwards it natively, untranslated, at `POST /v1/systemone` (with virtual models, guardrails, audit, and usage) and through passthrough at `/p/jev/...`. The endpoint is available once a `jev` - provider is configured, and can also route to OpenRouter's Jev models. + or `openrouter` provider is configured; OpenRouter serves Jev natively, and + its decision models are listed as utility models. `JEV_API_KEY` configures the hosted API; for a self-hosted Kev server, which speaks the same API without authentication, set `JEV_BASE_URL` and leave the key unset. See [Jev](/providers/jev). diff --git a/internal/providers/openrouter/openrouter.go b/internal/providers/openrouter/openrouter.go index 70ee13f6d..1222473a7 100644 --- a/internal/providers/openrouter/openrouter.go +++ b/internal/providers/openrouter/openrouter.go @@ -141,8 +141,9 @@ func (p *Provider) ListModels(ctx context.Context) (*core.ModelsResponse, error) // servableOpenRouterModalities are output modalities the gateway can reach on // OpenRouter: text and image generation flow through chat completions, -// embeddings through /embeddings, and speech/transcription through the -// /audio endpoints. A model listing none of these (rerank-only, video) has no +// embeddings through /embeddings, speech/transcription through the /audio +// endpoints, and decisions (System One models such as Jev) through +// /v1/systemone. A model listing none of these (rerank-only, video) has no // working endpoint here. var servableOpenRouterModalities = map[string]struct{}{ "text": {}, @@ -150,6 +151,7 @@ var servableOpenRouterModalities = map[string]struct{}{ "embeddings": {}, "speech": {}, "transcription": {}, + "decisions": {}, } func openrouterServable(m openrouterModel) bool { @@ -171,6 +173,7 @@ func openrouterServable(m openrouterModel) bool { // ID inference. func openrouterMetadata(m openrouterModel) *core.ModelMetadata { modes := make([]string, 0, 2) + decisions := false // "rerank" is deliberately not mapped: the gateway has no rerank surface // on OpenRouter, and the rerank mode would sort the model into the // Embeddings category despite being unreachable here. @@ -186,6 +189,8 @@ func openrouterMetadata(m openrouterModel) *core.ModelMetadata { modes = append(modes, "audio_speech") case "transcription": modes = append(modes, "audio_transcription") + case "decisions": + decisions = true } } pricing := openrouterPricing(m) @@ -196,9 +201,15 @@ func openrouterMetadata(m openrouterModel) *core.ModelMetadata { Capabilities: capabilities, Pricing: pricing, } - if len(modes) > 0 { + switch { + case len(modes) > 0: meta.Modes = modes meta.Categories = core.CategoriesForModes(modes) + case decisions: + // A decision model answers System One requests only. It has no + // generation mode to claim, so like the jev provider's models it is + // a utility model that no OpenAI endpoint routes to. + meta.Categories = []core.ModelCategory{core.CategoryUtility} } if m.ContextLength > 0 { contextWindow := m.ContextLength @@ -207,7 +218,7 @@ func openrouterMetadata(m openrouterModel) *core.ModelMetadata { if m.TopProvider.MaxCompletionTokens > 0 { meta.MaxOutputTokens = new(m.TopProvider.MaxCompletionTokens) } - if len(modes) == 0 && meta.ContextWindow == nil && meta.MaxOutputTokens == nil && pricing == nil && + if len(meta.Categories) == 0 && meta.ContextWindow == nil && meta.MaxOutputTokens == nil && pricing == nil && capabilities == nil && meta.DisplayName == "" && meta.Description == "" { return nil } diff --git a/internal/providers/openrouter/openrouter_test.go b/internal/providers/openrouter/openrouter_test.go index 0a50da65f..39bffeb7d 100644 --- a/internal/providers/openrouter/openrouter_test.go +++ b/internal/providers/openrouter/openrouter_test.go @@ -59,6 +59,11 @@ func TestListModels_StampsArchitectureModalities(t *testing.T) { "architecture":{"input_modalities":["text"],"output_modalities":["rerank"]}}, {"id":"acme/video-only","created":1721260800, "architecture":{"input_modalities":["text"],"output_modalities":["video"]}}, + {"id":"typesafe/jev-1.13","name":"TypeSafe: Jev 1.13","created":1789689684,"context_length":32000, + "architecture":{"input_modalities":["text"],"output_modalities":["decisions"]}, + "pricing":{"prompt":"0.000000042","completion":"0"}}, + {"id":"~typesafe/jev-latest","created":1789689684, + "architecture":{"input_modalities":["text"],"output_modalities":["decisions"]}}, {"id":"mystery/no-architecture","created":1721260800} ]}`) provider := newTestProvider(server.URL, server.Client()) @@ -72,7 +77,7 @@ func TestListModels_StampsArchitectureModalities(t *testing.T) { // embedding models would never enter the catalog. assert.Equal(t, "all", req.Query.Get("output_modalities")) - require.Len(t, resp.Data, 6) + require.Len(t, resp.Data, 8) byID := modelsByID(resp) chat := byID["openai/gpt-4o-mini"] @@ -114,6 +119,17 @@ func TestListModels_StampsArchitectureModalities(t *testing.T) { require.NotNil(t, stt.Metadata) assert.Equal(t, []string{"audio_transcription"}, stt.Metadata.Modes) + // Decision models serve /v1/systemone only: listed, but as utility models + // with no mode that would route an OpenAI request to them. + for _, id := range []string{"typesafe/jev-1.13", "~typesafe/jev-latest"} { + decision := byID[id] + require.NotNil(t, decision.Metadata, id) + assert.Empty(t, decision.Metadata.Modes, id) + assert.Equal(t, []core.ModelCategory{core.CategoryUtility}, decision.Metadata.Categories, id) + } + require.NotNil(t, byID["typesafe/jev-1.13"].Metadata.Pricing) + assert.InDelta(t, 0.042, *byID["typesafe/jev-1.13"].Metadata.Pricing.InputPerMtok, 1e-9) + assert.NotContains(t, byID, "cohere/rerank-only") assert.NotContains(t, byID, "acme/video-only") diff --git a/internal/providers/registry_normalization_test.go b/internal/providers/registry_normalization_test.go index 2ce0e6270..0af8eb32b 100644 --- a/internal/providers/registry_normalization_test.go +++ b/internal/providers/registry_normalization_test.go @@ -6,6 +6,7 @@ import ( "testing" "github.com/enterpilot/gomodel/internal/core" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -340,3 +341,27 @@ func TestRecordAvailabilityCheckKeepsFailureMarker(t *testing.T) { }) } } + +// LookupModel describes one catalog model by any selector the router resolves, +// including OpenRouter's "~"-prefixed alias IDs. +func TestRouterLookupModel(t *testing.T) { + registry := newTestRegistryWithModels(registryModelEntry{ + provider: &mockProvider{name: "openrouter"}, + providerName: "openrouter", + providerType: "openrouter", + modelID: "~typesafe/jev-latest", + }) + router, err := NewRouter(registry) + require.NoError(t, err) + + for _, selector := range []string{"openrouter/~typesafe/jev-latest", "~typesafe/jev-latest"} { + model, ok := router.LookupModel(selector) + require.True(t, ok, selector) + assert.Equal(t, "~typesafe/jev-latest", model.ID, selector) + } + _, ok := router.LookupModel("openrouter/unknown") + assert.False(t, ok) + + _, ok = (&Router{}).LookupModel("openrouter/~typesafe/jev-latest") + assert.False(t, ok, "a lookup without single-model access describes nothing") +} diff --git a/internal/providers/router_models.go b/internal/providers/router_models.go index 4f943569e..b49d87f70 100644 --- a/internal/providers/router_models.go +++ b/internal/providers/router_models.go @@ -182,3 +182,17 @@ func (r *Router) NativeResponseProviderTypes() []string { return ok }) } + +// LookupModel returns a copy of the catalog entry for a model selector, or +// false when the model is unknown or the lookup cannot describe one model. +func (r *Router) LookupModel(model string) (*core.Model, bool) { + if r.caps.modelInfo == nil { + return nil, false + } + info := r.caps.modelInfo.GetModel(model) + if info == nil { + return nil, false + } + cloned := info.Model + return &cloned, true +} diff --git a/internal/server/http.go b/internal/server/http.go index 09a7ad174..f6e4dab94 100644 --- a/internal/server/http.go +++ b/internal/server/http.go @@ -497,7 +497,7 @@ func New(provider core.RoutableProvider, cfg *Config) *Server { e.POST("/v1/images/generations", handler.ImageGenerations) e.POST("/v1/images/edits", handler.ImageEdits) // System One decisions (Jev / Kev). The handler answers 404 until a jev - // provider is configured, so the route costs nothing otherwise. + // or openrouter provider is configured. e.POST("/v1/systemone", handler.SystemOne) if cfg == nil || cfg.RealtimeEnabled { e.GET("/v1/realtime", handler.Realtime) diff --git a/internal/server/systemone_handler.go b/internal/server/systemone_handler.go index 428fa0c0b..ab23f0c2c 100644 --- a/internal/server/systemone_handler.go +++ b/internal/server/systemone_handler.go @@ -6,6 +6,7 @@ import ( "io" "log/slog" "net/http" + "slices" "strings" "github.com/goccy/go-json" @@ -20,18 +21,13 @@ import ( const ( systemOnePath = "/v1/systemone" systemOneEndpoint = "systemone" - // jevProviderType serves TypeSafe's hosted Jev API and self-hosted Kev - // servers; configuring one is what makes /v1/systemone available. - jevProviderType = "jev" ) // systemOneProviderTypes are the provider types that serve the System One API -// natively. OpenRouter serves Jev at the same path with the same request and -// answer shapes, so it can back a virtual model next to a jev provider. -var systemOneProviderTypes = map[string]struct{}{ - jevProviderType: {}, - "openrouter": {}, -} +// natively: jev (TypeSafe's hosted Jev and self-hosted Kev servers) and +// OpenRouter, which serves Jev and Kev at the same path with the same request +// and answer shapes. Configuring either makes /v1/systemone available. +var systemOneProviderTypes = []string{"jev", "openrouter"} // SystemOne handles POST /v1/systemone. // @@ -42,7 +38,7 @@ var systemOneProviderTypes = map[string]struct{}{ // provider has no System One API is rejected. // // @Summary Evaluate a System One decision request (Jev / Kev) -// @Description Available when a jev provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated. +// @Description Available when a jev or openrouter provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated. // @Tags systemone // @Accept json // @Produce json @@ -62,7 +58,7 @@ func (h *Handler) SystemOne(c *echo.Context) error { // SystemOne resolves, guards, and forwards one System One request. func (s *translatedInferenceService) SystemOne(c *echo.Context) error { if !s.systemOneAvailable() { - return handleError(c, core.NewNotFoundError("POST "+systemOnePath+" is available only when a jev provider is configured")) + return handleError(c, core.NewNotFoundError("POST "+systemOnePath+" is available only when a jev or openrouter provider is configured")) } body, err := requestBodyBytes(c) if err != nil { @@ -87,9 +83,8 @@ func (s *translatedInferenceService) SystemOne(c *echo.Context) error { return handleError(c, err) } } - providerType := strings.TrimSpace(resolution.ProviderType) - if _, ok := systemOneProviderTypes[providerType]; !ok { - return handleError(c, systemOneUnsupportedModelError(c, resolution)) + if reason := s.systemOneUnsupportedReason(resolution); reason != "" { + return handleError(c, systemOneUnsupportedModelError(c, resolution, reason)) } body, err = s.guardSystemOneState(c, workflow, &req, body) @@ -103,12 +98,20 @@ func (s *translatedInferenceService) SystemOne(c *echo.Context) error { return s.dispatchSystemOne(c, workflow, model, body) } -// systemOneAvailable reports whether a jev provider is configured. It is -// checked per request rather than at route registration so a provider added -// at runtime makes the endpoint available without a restart. +// systemOneAvailable reports whether a provider that serves System One is +// configured. It is checked per request rather than at route registration so +// a provider added at runtime makes the endpoint available without a restart. func (s *translatedInferenceService) systemOneAvailable() bool { named, ok := s.provider.(core.ProviderTypeNameResolver) - return ok && strings.TrimSpace(named.GetProviderNameForType(jevProviderType)) != "" + if !ok { + return false + } + for _, providerType := range systemOneProviderTypes { + if strings.TrimSpace(named.GetProviderNameForType(providerType)) != "" { + return true + } + } + return false } // systemOneWorkflow returns the request's workflow with its model resolved. @@ -131,26 +134,53 @@ func (s *translatedInferenceService) systemOneWorkflow(c *echo.Context, model st return workflow, nil } -// systemOneUnsupportedModelError explains a request whose model routes to a -// provider without the System One API. It is also logged: a virtual model -// that sends System One traffic to a chat model is an operator mistake the -// caller cannot fix. -func systemOneUnsupportedModelError(c *echo.Context, resolution *core.RequestModelResolution) error { +// modelCatalog describes single catalog models; the provider router +// implements it. +type modelCatalog interface { + LookupModel(model string) (*core.Model, bool) +} + +// systemOneUnsupportedReason explains why the resolved model cannot answer a +// System One request, or returns "" when it can. The provider must serve the +// API, and since OpenRouter also serves chat models, the model must not be +// catalogued with a generation mode. A model the catalog does not describe is +// given the benefit of the doubt: the upstream reports it if it is wrong. +func (s *translatedInferenceService) systemOneUnsupportedReason(resolution *core.RequestModelResolution) string { + providerType := strings.TrimSpace(resolution.ProviderType) + if !slices.Contains(systemOneProviderTypes, providerType) { + return fmt.Sprintf("is served by a %s provider, which has no System One API", providerType) + } + catalog, ok := s.provider.(modelCatalog) + if !ok { + return "" + } + model, ok := catalog.LookupModel(resolution.ResolvedQualifiedModel()) + if !ok || model == nil || model.Metadata == nil || len(model.Metadata.Modes) == 0 { + return "" + } + return fmt.Sprintf("is a %s model, not a System One model", strings.Join(model.Metadata.Modes, "/")) +} + +// systemOneUnsupportedModelError explains a request whose model cannot answer +// System One. It is also logged: a virtual model that sends System One +// traffic to a chat model is an operator mistake the caller cannot fix. +func systemOneUnsupportedModelError(c *echo.Context, resolution *core.RequestModelResolution, reason string) error { requested := resolution.RequestedQualifiedModel() resolved := resolution.ResolvedQualifiedModel() - slog.Warn("System One request routed to a provider without the System One API", + slog.Warn("System One request routed to a model without the System One API", "request_id", requestIDFromContextOrHeader(c.Request()), "requested_model", requested, "resolved_model", resolved, "provider_type", resolution.ProviderType, + "reason", reason, ) target := fmt.Sprintf("%q", requested) if resolved != requested { target += fmt.Sprintf(" (resolved to %q)", resolved) } return core.NewInvalidRequestError(fmt.Sprintf( - "model %s is served by a %s provider, which has no System One API; %s forwards requests natively and does not translate them to other APIs, so use a jev model", - target, resolution.ProviderType, systemOnePath, + "model %s %s; %s forwards requests natively and does not translate them to other APIs, so use a System One model such as a jev model or OpenRouter's typesafe/jev-1.13", + target, reason, systemOnePath, ), nil).WithParam("model") } diff --git a/internal/server/systemone_handler_test.go b/internal/server/systemone_handler_test.go index b06a53145..af6e4c55a 100644 --- a/internal/server/systemone_handler_test.go +++ b/internal/server/systemone_handler_test.go @@ -76,8 +76,9 @@ func forwardedSystemOneBody(t *testing.T, provider *mockProvider) map[string]any return body } -// Without a jev provider the endpoint does not exist, whatever the model. -func TestSystemOne_UnavailableWithoutJevProvider(t *testing.T) { +// Without a provider that serves System One the endpoint does not exist, +// whatever the model. +func TestSystemOne_UnavailableWithoutSystemOneProvider(t *testing.T) { provider := &mockProvider{ supportedModels: []string{"gpt-5-mini"}, providerTypes: map[string]string{"openai/gpt-5-mini": "openai"}, @@ -89,7 +90,7 @@ func TestSystemOne_UnavailableWithoutJevProvider(t *testing.T) { require.NoError(t, handler.SystemOne(c)) assert.Equal(t, http.StatusNotFound, rec.Code) - assert.Contains(t, rec.Body.String(), "jev provider") + assert.Contains(t, rec.Body.String(), "jev or openrouter provider") assert.Nil(t, provider.lastPassthroughReq) } @@ -143,6 +144,37 @@ func TestSystemOne_ForwardsOpenRouterJevNatively(t *testing.T) { assert.InDelta(t, 0.00003, usageLogger.entries[0].RawData["cost"], 1e-12) } +// OpenRouter serves System One natively, so it enables the endpoint on its own. +// Its catalog names Jev "~typesafe/jev-latest"; a virtual model gives SDK +// callers the "jev-latest" name they send by default. +func TestSystemOne_WorksWithOpenRouterAlone(t *testing.T) { + answer := `{"id":"gen-dec-1","model":"typesafe/jev-1.13-20260917","answers":{},"usage":{"input_tokens":10,"output_tokens":1}}` + for _, model := range []string{"openrouter/~typesafe/jev-latest", "jev-latest"} { + t.Run(model, func(t *testing.T) { + provider := &mockProvider{ + supportedModels: []string{"~typesafe/jev-latest"}, + providerTypes: map[string]string{"openrouter/~typesafe/jev-latest": "openrouter"}, + providerNames: map[string]string{"openrouter/~typesafe/jev-latest": "openrouter"}, + passthroughResponse: &core.PassthroughResponse{ + StatusCode: http.StatusOK, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(answer)), + }, + } + aliases := systemOneAliasResolver{"jev-latest": {Provider: "openrouter", Model: "~typesafe/jev-latest"}} + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, aliases, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody(model)) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, "openrouter", provider.lastPassthroughProvider) + assert.Equal(t, "systemone", provider.lastPassthroughReq.Endpoint) + assert.Equal(t, "~typesafe/jev-latest", forwardedSystemOneBody(t, provider)["model"]) + }) + } +} + // The endpoint never translates: a model on a provider without the System One // API is rejected with an explanation, whether named directly or through a // virtual model. @@ -163,6 +195,62 @@ func TestSystemOne_RejectsModelsWithoutSystemOneAPI(t *testing.T) { } } +// catalogProvider adds the router's single-model catalog lookup to the mock. +type catalogProvider struct { + *mockProvider + models map[string]core.Model +} + +func (p catalogProvider) LookupModel(model string) (*core.Model, bool) { + found, ok := p.models[model] + return &found, ok +} + +// OpenRouter serves chat and decision models from one provider, so the model +// itself must be a System One model: one catalogued with a generation mode is +// rejected, while a decision model (a utility model with no mode) is forwarded. +func TestSystemOne_RejectsOpenRouterChatModels(t *testing.T) { + provider := catalogProvider{ + mockProvider: &mockProvider{ + supportedModels: []string{"typesafe/jev-1.13", "openai/gpt-4o-mini"}, + providerTypes: map[string]string{ + "openrouter/typesafe/jev-1.13": "openrouter", + "openrouter/openai/gpt-4o-mini": "openrouter", + }, + providerNames: map[string]string{ + "openrouter/typesafe/jev-1.13": "openrouter", + "openrouter/openai/gpt-4o-mini": "openrouter", + }, + passthroughResponse: &core.PassthroughResponse{ + StatusCode: http.StatusOK, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(systemOneAnswer)), + }, + }, + models: map[string]core.Model{ + "openrouter/typesafe/jev-1.13": {ID: "typesafe/jev-1.13", Metadata: &core.ModelMetadata{ + Categories: []core.ModelCategory{core.CategoryUtility}, + }}, + "openrouter/openai/gpt-4o-mini": {ID: "openai/gpt-4o-mini", Metadata: &core.ModelMetadata{ + Modes: []string{"chat"}, Categories: []core.ModelCategory{core.CategoryTextGeneration}, + }}, + }, + } + handler := NewHandler(provider, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("openrouter/openai/gpt-4o-mini")) + require.NoError(t, handler.SystemOne(c)) + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "is a chat model, not a System One model") + assert.Contains(t, rec.Body.String(), "does not translate") + assert.Nil(t, provider.lastPassthroughReq) + + c, rec = echotest.Post(t, "/v1/systemone", systemOneBody("openrouter/typesafe/jev-1.13")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.Equal(t, "openrouter", provider.lastPassthroughProvider) +} + func TestSystemOne_RequiresModel(t *testing.T) { provider := newSystemOneProvider(systemOneAnswer) handler := NewHandler(provider, nil, nil, nil) From 3eddb1b008c9137f9d7dee6e8a145964757bfb3c Mon Sep 17 00:00:00 2001 From: "Jakub A. W" Date: Sat, 26 Sep 2026 19:09:42 +0200 Subject: [PATCH 3/3] fix(jev): guard in-place state edits and answer 404 before model resolution --- .../plugins/exchange/systemone_request.go | 9 ++- .../exchange/systemone_request_test.go | 7 +++ internal/server/model_validation.go | 5 ++ internal/server/systemone_handler.go | 11 ++-- internal/server/systemone_handler_test.go | 55 +++++++++++++++---- 5 files changed, 70 insertions(+), 17 deletions(-) diff --git a/internal/plugins/exchange/systemone_request.go b/internal/plugins/exchange/systemone_request.go index e10d28732..73ad3ad33 100644 --- a/internal/plugins/exchange/systemone_request.go +++ b/internal/plugins/exchange/systemone_request.go @@ -1,6 +1,7 @@ package exchange import ( + "bytes" "fmt" "sort" @@ -39,7 +40,7 @@ func FromSystemOneRequest(req *core.SystemOneRequest) (*pluginapi.Prompt, error) func systemOneStateText(state json.RawMessage) string { var text string - if err := json.Unmarshal(state, &text); err == nil { + if trimmed := bytes.TrimSpace(state); len(trimmed) > 0 && trimmed[0] == '"' && json.Unmarshal(trimmed, &text) == nil { return text } return string(state) @@ -67,8 +68,10 @@ func ApplyToSystemOneRequest(original *core.SystemOneRequest, p *pluginapi.Promp return nil, fmt.Errorf("exchange: the System One state was removed") } text := msg.Text() - var current string - if err := json.Unmarshal(original.State, ¤t); err == nil || len(original.State) == 0 { + // A missing state becomes a string; null, numbers, and records keep + // their JSON type (json.Unmarshal would accept null into a string). + state := bytes.TrimSpace(original.State) + if len(state) == 0 || state[0] == '"' { encoded, err := json.Marshal(text) if err != nil { return nil, fmt.Errorf("exchange: encode System One state: %w", err) diff --git a/internal/plugins/exchange/systemone_request_test.go b/internal/plugins/exchange/systemone_request_test.go index b2ec317e1..62d5ca08b 100644 --- a/internal/plugins/exchange/systemone_request_test.go +++ b/internal/plugins/exchange/systemone_request_test.go @@ -19,6 +19,7 @@ func TestFromSystemOneRequestExposesStateAsUserMessage(t *testing.T) { }{ {name: "string state", state: `"charged twice"`, want: "charged twice"}, {name: "record state", state: `{"ticket":"charged twice"}`, want: `{"ticket":"charged twice"}`}, + {name: "null state", state: `null`, want: "null"}, {name: "no state", state: ``, want: ""}, } for _, tt := range tests { @@ -62,6 +63,12 @@ func TestApplyToSystemOneRequest(t *testing.T) { edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `{"name":"[PERSON]"}`) }, wantState: `{"name":"[PERSON]"}`, }, + { + name: "null state keeps its JSON type", + state: `null`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `{"redacted":true}`) }, + wantState: `{"redacted":true}`, + }, { name: "record state must stay valid JSON", state: `{"name":"John"}`, diff --git a/internal/server/model_validation.go b/internal/server/model_validation.go index 2d85ebcbb..52b54be42 100644 --- a/internal/server/model_validation.go +++ b/internal/server/model_validation.go @@ -104,6 +104,11 @@ func deriveWorkflowWithPolicy( return workflow, nil case core.OperationChatCompletions, core.OperationResponses, core.OperationEmbeddings, core.OperationSystemOne: + if desc.Operation == core.OperationSystemOne && !systemOneAvailable(provider) { + // The handler answers 404; resolving the model first would + // report a model error for an endpoint that is not there. + return nil, nil + } workflow.Mode = core.ExecutionModeTranslated if desc.BodyMode != core.BodyModeJSON { // Responses lifecycle routes (GET/DELETE /v1/responses/{id}, diff --git a/internal/server/systemone_handler.go b/internal/server/systemone_handler.go index ab23f0c2c..b297e7fcf 100644 --- a/internal/server/systemone_handler.go +++ b/internal/server/systemone_handler.go @@ -57,7 +57,7 @@ func (h *Handler) SystemOne(c *echo.Context) error { // SystemOne resolves, guards, and forwards one System One request. func (s *translatedInferenceService) SystemOne(c *echo.Context) error { - if !s.systemOneAvailable() { + if !systemOneAvailable(s.provider) { return handleError(c, core.NewNotFoundError("POST "+systemOnePath+" is available only when a jev or openrouter provider is configured")) } body, err := requestBodyBytes(c) @@ -101,8 +101,8 @@ func (s *translatedInferenceService) SystemOne(c *echo.Context) error { // systemOneAvailable reports whether a provider that serves System One is // configured. It is checked per request rather than at route registration so // a provider added at runtime makes the endpoint available without a restart. -func (s *translatedInferenceService) systemOneAvailable() bool { - named, ok := s.provider.(core.ProviderTypeNameResolver) +func systemOneAvailable(provider core.RoutableProvider) bool { + named, ok := provider.(core.ProviderTypeNameResolver) if !ok { return false } @@ -193,6 +193,9 @@ func (s *translatedInferenceService) guardSystemOneState(c *echo.Context, workfl if !ok || !workflow.GuardrailsEnabled() { return body, nil } + // Compare against a copy: a patcher may redact the state in place and + // return the same request, and that edit must still reach the body. + original := bytes.Clone(req.State) patched, err := patcher.PatchSystemOneRequest(c.Request().Context(), req) s.recordGuardrailOutcomes(c) if err != nil { @@ -201,7 +204,7 @@ func (s *translatedInferenceService) guardSystemOneState(c *echo.Context, workfl } return nil, err } - if patched == nil || patched == req || bytes.Equal(patched.State, req.State) { + if patched == nil || bytes.Equal(patched.State, original) { return body, nil } rewritten, err := replaceTopLevelMember(body, "state", patched.State) diff --git a/internal/server/systemone_handler_test.go b/internal/server/systemone_handler_test.go index af6e4c55a..45e82b6cc 100644 --- a/internal/server/systemone_handler_test.go +++ b/internal/server/systemone_handler_test.go @@ -279,19 +279,54 @@ func (stateRedactingPatcher) PatchSystemOneRequest(_ context.Context, req *core. return &patched, nil } -// Guardrails see the state and their edits reach the provider; the rest of -// the body is untouched. +// inPlaceRedactingPatcher edits the request it was given and returns it; the +// patcher contract allows that, and the edit must still reach the provider. +type inPlaceRedactingPatcher struct{ stateRedactingPatcher } + +func (inPlaceRedactingPatcher) PatchSystemOneRequest(_ context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) { + req.State = json.RawMessage(strings.ReplaceAll(string(req.State), "4111", "[card]")) + return req, nil +} + +// Guardrails see the state and their edits reach the provider, whether the +// patcher returns a copy or edits in place; the rest of the body is untouched. func TestSystemOne_GuardrailsEditState(t *testing.T) { - provider := newSystemOneProvider(systemOneAnswer) - handler := newHandlerWithAuthorizer(provider, nil, nil, nil, systemOneAliases, nil, nil, nil, stateRedactingPatcher{}) + patchers := map[string]TranslatedRequestPatcher{ + "copy": stateRedactingPatcher{}, + "in place": inPlaceRedactingPatcher{}, + } + for name, patcher := range patchers { + t.Run(name, func(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, systemOneAliases, nil, nil, nil, patcher) - c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("decider")) - require.NoError(t, handler.SystemOne(c)) - require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("decider")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) - body := forwardedSystemOneBody(t, provider) - assert.Equal(t, "I was charged twice for card [card].", body["state"]) - assert.Equal(t, "kev-latest", body["model"]) + body := forwardedSystemOneBody(t, provider) + assert.Equal(t, "I was charged twice for card [card].", body["state"]) + assert.Equal(t, "kev-latest", body["model"]) + }) + } +} + +// Through the full middleware stack an unavailable endpoint answers 404 +// before the model is resolved, so a missing or unknown model is not +// reported for a route that is not there. +func TestSystemOne_UnavailableBeforeModelResolution(t *testing.T) { + provider := &mockProvider{ + supportedModels: []string{"gpt-5-mini"}, + providerTypes: map[string]string{"openai/gpt-5-mini": "openai"}, + providerNames: map[string]string{"openai/gpt-5-mini": "openai"}, + } + srv := New(provider, &Config{}) + + for _, body := range []string{`{"state":"hi","questions":{}}`, systemOneBody("no-such-model")} { + rec := postJSON(t, srv, "/v1/systemone", body) + assert.Equal(t, http.StatusNotFound, rec.Code, rec.Body.String()) + assert.Contains(t, rec.Body.String(), "jev or openrouter provider") + } } // Through the full middleware stack a System One call is audited under its