From b8cb81c034f77e2a3d28b93c3368c205bec8079f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jakub=20A=2E=20W=C4=85sek?= Date: Thu, 24 Sep 2026 11:06:48 +0200 Subject: [PATCH 01/23] fix(dashboard): drop the details accordion from virtual model rows (#1079) --- web/dashboard/src/pages/models/modelDetails.js | 12 ++++++------ web/dashboard/tests/models-details.test.js | 6 +++--- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/web/dashboard/src/pages/models/modelDetails.js b/web/dashboard/src/pages/models/modelDetails.js index d49fa007d..5f90a0393 100644 --- a/web/dashboard/src/pages/models/modelDetails.js +++ b/web/dashboard/src/pages/models/modelDetails.js @@ -56,14 +56,14 @@ function pushSection(sections, key, title, items) { } } -// rowHasModelDetails reports whether the accordion has anything to show. A -// virtual model whose target is not in the inventory only carries a stub -// model ({ id, object }), so it gets no toggle. +// rowHasModelDetails reports whether the accordion has anything to show. +// Virtual-model rows get no toggle: their model is only the first resolvable +// target's, which its own row already details and which misdescribes a +// virtual model spreading requests over several targets. export function rowHasModelDetails(row) { const model = row && row.model; - if (!model) return false; - if (!row.is_alias) return Boolean(text(model.id)); - return Boolean(model.metadata || text(model.owned_by) || createdDate(model.created)); + if (!model || row.is_alias) return false; + return Boolean(text(model.id)); } // ---- Layers and views ---- diff --git a/web/dashboard/tests/models-details.test.js b/web/dashboard/tests/models-details.test.js index 106011a60..20b7dc8ef 100644 --- a/web/dashboard/tests/models-details.test.js +++ b/web/dashboard/tests/models-details.test.js @@ -54,13 +54,13 @@ function itemValue(details, key, label) { return entry ? entry.value : undefined; } -test("rowHasModelDetails: real models expand, alias stubs do not", () => { +test("rowHasModelDetails: real models expand, virtual models do not", () => { assert.equal(rowHasModelDetails(modelRow()), true); assert.equal(rowHasModelDetails(modelRow({ model: { id: "bare" } })), true); assert.equal(rowHasModelDetails({ is_alias: true, model: { id: "alias", object: "model" } }), false); assert.equal( - rowHasModelDetails({ is_alias: true, model: { id: "gpt-4o", owned_by: "openai" } }), - true, + rowHasModelDetails({ is_alias: true, model: { id: "gpt-4o", owned_by: "openai", metadata: {} } }), + false, ); assert.equal(rowHasModelDetails({ is_alias: false, model: null }), false); assert.equal(rowHasModelDetails(null), false); From 61408599deb8dbead9485c067ef76715d84bc238 Mon Sep 17 00:00:00 2001 From: Ben <50115212+weselben@users.noreply.github.com> Date: Thu, 24 Sep 2026 12:10:26 +0200 Subject: [PATCH 02/23] fix(tests): skip .worktrees in testconventions scan (#1087) --- internal/testconventions/assertions_test.go | 1 + 1 file changed, 1 insertion(+) diff --git a/internal/testconventions/assertions_test.go b/internal/testconventions/assertions_test.go index e28f4deea..d30b2793f 100644 --- a/internal/testconventions/assertions_test.go +++ b/internal/testconventions/assertions_test.go @@ -21,6 +21,7 @@ import ( var skippedDirs = map[string]bool{ ".git": true, ".claude": true, // local agent worktrees, gitignored + ".worktrees": true, // local git worktrees, gitignored ".cache": true, "node_modules": true, "third_party": true, From d2b00e944a5ed31c06132cf6da5bc5da5ddfa7fb Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jakub=20A=2E=20W=C4=85sek?= Date: Thu, 24 Sep 2026 12:11:27 +0200 Subject: [PATCH 03/23] feat(audit): filter audit logs by request type (#1082) * feat(audit): filter audit logs by request type * fix(audit): hide request types by exclusion so unclassified entries stay --- cmd/gomodel/docs/docs.go | 12 +++ config/env.go | 2 +- docs/openapi.json | 16 +++ internal/admin/handler_audit.go | 10 ++ internal/admin/handler_audit_sessions_test.go | 15 +++ internal/auditlog/reader.go | 10 +- internal/auditlog/reader_mongodb.go | 25 +++++ internal/auditlog/reader_sql.go | 25 +++++ internal/auditlog/reader_suite_test.go | 56 ++++++++++ internal/core/endpoint_operations.go | 58 ++++++++++ internal/core/endpoint_operations_test.go | 72 +++++++++++++ web/dashboard/messages/de.json | 12 +++ web/dashboard/messages/en.json | 12 +++ web/dashboard/messages/pl.json | 12 +++ web/dashboard/messages/zh-CN.json | 12 +++ .../src/pages/audit-logs/AuditFilters.svelte | 101 +++++++++++++++++- .../src/pages/audit-logs/audit-logic.js | 25 +++-- .../src/pages/audit-logs/audit-operations.js | 86 +++++++++++++++ .../src/pages/audit-logs/auditList.svelte.js | 28 +++++ .../src/pages/audit-logs/live-logs-logic.js | 5 +- .../src/pages/audit-logs/liveLogs.svelte.js | 13 +++ web/dashboard/tests/audit-operations.test.js | 77 +++++++++++++ 22 files changed, 671 insertions(+), 13 deletions(-) create mode 100644 internal/core/endpoint_operations.go create mode 100644 internal/core/endpoint_operations_test.go create mode 100644 web/dashboard/src/pages/audit-logs/audit-operations.js create mode 100644 web/dashboard/tests/audit-operations.test.js diff --git a/cmd/gomodel/docs/docs.go b/cmd/gomodel/docs/docs.go index 24ffb6836..ae15c7e4d 100644 --- a/cmd/gomodel/docs/docs.go +++ b/cmd/gomodel/docs/docs.go @@ -252,6 +252,12 @@ const docTemplate = `{ "name": "stream", "in": "query" }, + { + "type": "string", + "description": "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay", + "name": "exclude_operation", + "in": "query" + }, { "type": "string", "description": "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message", @@ -381,6 +387,12 @@ const docTemplate = `{ "name": "stream", "in": "query" }, + { + "type": "string", + "description": "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay", + "name": "exclude_operation", + "in": "query" + }, { "type": "string", "description": "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message", diff --git a/config/env.go b/config/env.go index 17c6dfcd5..00919d895 100644 --- a/config/env.go +++ b/config/env.go @@ -66,7 +66,7 @@ func applyPluginsLoadEnv(cfg *Config) { return } load := make([]PluginFileConfig, 0, 4) - for _, item := range strings.Split(v, ",") { + for item := range strings.SplitSeq(v, ",") { if entry := parsePluginLoadEntry(item); entry.File != "" { load = append(load, entry) } diff --git a/docs/openapi.json b/docs/openapi.json index 4d8e196fa..f425f6bee 100644 --- a/docs/openapi.json +++ b/docs/openapi.json @@ -330,6 +330,14 @@ "type": "boolean" } }, + { + "description": "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay", + "name": "exclude_operation", + "in": "query", + "schema": { + "type": "string" + } + }, { "description": "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message", "name": "search", @@ -507,6 +515,14 @@ "type": "boolean" } }, + { + "description": "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay", + "name": "exclude_operation", + "in": "query", + "schema": { + "type": "string" + } + }, { "description": "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message", "name": "search", diff --git a/internal/admin/handler_audit.go b/internal/admin/handler_audit.go index 124732f62..9f8712e3a 100644 --- a/internal/admin/handler_audit.go +++ b/internal/admin/handler_audit.go @@ -51,6 +51,7 @@ const conversationBuildTimeout = 10 * time.Second // @Param error_type query string false "Filter by error type" // @Param status_code query int false "Filter by status code" // @Param stream query bool false "Filter by stream mode (true/false)" +// @Param exclude_operation query string false "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay" // @Param search query string false "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message" // @Param limit query int false "Page size (default 25, max 100)" // @Param offset query int false "Offset for pagination" @@ -169,6 +170,14 @@ func parseAuditLogQueryParams(c *echo.Context) (auditlog.LogQueryParams, error) params.Stream = &parsed } + if raw := c.QueryParam("exclude_operation"); raw != "" { + ops, unknown, ok := core.ParseOperations(raw) + if !ok { + return params, core.NewInvalidRequestError("invalid exclude_operation: "+unknown, nil) + } + params.ExcludeOperations = ops + } + if l := c.QueryParam("limit"); l != "" { parsed, err := strconv.Atoi(l) if err != nil || parsed <= 0 { @@ -211,6 +220,7 @@ func parseAuditLogQueryParams(c *echo.Context) (auditlog.LogQueryParams, error) // @Param error_type query string false "Filter by error type" // @Param status_code query int false "Filter by status code" // @Param stream query bool false "Filter by stream mode (true/false)" +// @Param exclude_operation query string false "Comma-separated endpoint operations to hide, e.g. mcp,provider_passthrough,audio_speech; other entries, including unclassified ones, stay" // @Param search query string false "Search across request_id/requested_model/provider/method/path/session_id/error_type/error_message" // @Param limit query int false "Page size in threads (default 25, max 100)" // @Param offset query int false "Offset for pagination" diff --git a/internal/admin/handler_audit_sessions_test.go b/internal/admin/handler_audit_sessions_test.go index 6c4f2fa2b..fb8af79db 100644 --- a/internal/admin/handler_audit_sessions_test.go +++ b/internal/admin/handler_audit_sessions_test.go @@ -7,6 +7,7 @@ import ( "time" "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/echotest" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -137,3 +138,17 @@ func TestAuditLog_SessionIDSkipsDefaultDateWindow(t *testing.T) { require.False(t, reader.lastQuery.StartDate.IsZero()) require.False(t, reader.lastQuery.EndDate.IsZero()) } + +func TestAuditLog_ExcludeOperationFilter(t *testing.T) { + reader := &mockAuditReader{logResult: &auditlog.LogListResult{}} + h := NewHandler(nil, nil, WithAuditReader(reader)) + + c, _ := echotest.Get(t, "/admin/audit/log?exclude_operation=mcp,audio_speech") + require.NoError(t, h.AuditLog(c)) + assert.Equal(t, []core.Operation{core.OperationMCP, core.OperationAudioSpeech}, reader.lastQuery.ExcludeOperations) + + c, rec := echotest.Get(t, "/admin/audit/log?exclude_operation=mcp,nope") + require.NoError(t, h.AuditLog(c)) + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "invalid exclude_operation: nope") +} diff --git a/internal/auditlog/reader.go b/internal/auditlog/reader.go index 9554a6b24..6dfcaa056 100644 --- a/internal/auditlog/reader.go +++ b/internal/auditlog/reader.go @@ -3,6 +3,8 @@ package auditlog import ( "context" "time" + + "github.com/enterpilot/gomodel/internal/core" ) // QueryParams specifies the date range for audit log retrieval. @@ -24,8 +26,12 @@ type LogQueryParams struct { Search string StatusCode *int Stream *bool - Limit int - Offset int + // ExcludeOperations drops entries whose path belongs to one of these + // operations. Entries outside every operation (e.g. authentication + // events) always stay. + ExcludeOperations []core.Operation + Limit int + Offset int // OmitAttempts excludes provider attempts from returned entries. The default is false. OmitAttempts bool // ExactUserPath matches only UserPath instead of its subtree. The default is false. diff --git a/internal/auditlog/reader_mongodb.go b/internal/auditlog/reader_mongodb.go index 0069df70b..76555e42f 100644 --- a/internal/auditlog/reader_mongodb.go +++ b/internal/auditlog/reader_mongodb.go @@ -283,6 +283,9 @@ func mongoLogMatchFilters(params LogQueryParams) (bson.D, error) { if params.Stream != nil { matchFilters = append(matchFilters, bson.E{Key: "stream", Value: *params.Stream}) } + if len(params.ExcludeOperations) > 0 { + matchFilters = append(matchFilters, mongoExcludeOperationsFilter(params.ExcludeOperations)) + } if params.Search != "" && isCanonicalUUID(params.Search) { // A full canonical UUID is a pasted identifier: match the indexed // identity fields by equality (both spellings — stored ids are @@ -417,3 +420,25 @@ func (r *MongoDBReader) findConversationEntry(ctx context.Context, filter bson.D return row.toLogEntry(), nil } + +// mongoExcludeOperationsFilter drops paths belonging to any of the +// operations; entries without a path stay. Exact paths also match with one +// trailing slash, as DescribeEndpoint does. +func mongoExcludeOperationsFilter(ops []core.Operation) bson.E { + var exact bson.A + var nor bson.A + for _, op := range ops { + paths, _ := core.PathsForOperation(op) + for _, path := range paths.Exact { + exact = append(exact, path, path+"/") + } + for _, prefix := range paths.Prefixes { + exact = append(exact, prefix) + nor = append(nor, bson.D{{Key: "path", Value: bson.D{ + {Key: "$regex", Value: "^" + regexp.QuoteMeta(prefix+"/")}, + }}}) + } + } + nor = append(nor, bson.D{{Key: "path", Value: bson.D{{Key: "$in", Value: exact}}}}) + return bson.E{Key: "$nor", Value: nor} +} diff --git a/internal/auditlog/reader_sql.go b/internal/auditlog/reader_sql.go index 05ea05160..e47548052 100644 --- a/internal/auditlog/reader_sql.go +++ b/internal/auditlog/reader_sql.go @@ -12,6 +12,7 @@ import ( "github.com/goccy/go-json" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/storage/sqlutil" "github.com/enterpilot/gomodel/internal/storage/sqlx" ) @@ -243,6 +244,10 @@ func (r *SQLReader) logFilters(ctx context.Context, params LogQueryParams) ([]st if params.Stream != nil { add("stream = ?", *params.Stream) } + if len(params.ExcludeOperations) > 0 { + condition, values := excludeOperationsSQLFilter(params.ExcludeOperations) + add(condition, values...) + } if params.Search != "" { condition, values := r.searchFilter(params.Search, r.searchIsIndexed(ctx)) add(condition, values...) @@ -560,3 +565,23 @@ func isMissingAuditAttemptsTable(err error) bool { return strings.Contains(message, "audit_log_attempts") && (strings.Contains(message, "no such table") || strings.Contains(message, "does not exist")) } + +// excludeOperationsSQLFilter drops paths belonging to any of the operations. +// Exact paths also match with one trailing slash, as DescribeEndpoint does. +// The prefixes hold no LIKE wildcards, so they need no escaping. +func excludeOperationsSQLFilter(ops []core.Operation) (string, []any) { + var clauses []string + var args []any + for _, op := range ops { + paths, _ := core.PathsForOperation(op) + for _, exact := range paths.Exact { + clauses = append(clauses, "path = ?", "path = ?") + args = append(args, exact, exact+"/") + } + for _, prefix := range paths.Prefixes { + clauses = append(clauses, "path = ?", "path LIKE ?") + args = append(args, prefix, prefix+"/%") + } + } + return "(path IS NULL OR NOT (" + strings.Join(clauses, " OR ") + "))", args +} diff --git a/internal/auditlog/reader_suite_test.go b/internal/auditlog/reader_suite_test.go index decbeb639..c40be48f3 100644 --- a/internal/auditlog/reader_suite_test.go +++ b/internal/auditlog/reader_suite_test.go @@ -2,11 +2,13 @@ package auditlog import ( "context" + "fmt" "testing" "time" "go.mongodb.org/mongo-driver/v2/mongo" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/storage/mongotest" "github.com/enterpilot/gomodel/internal/storage/sqlx" "github.com/enterpilot/gomodel/internal/storage/sqlx/sqlxtest" @@ -103,3 +105,57 @@ func TestReader_GetLastUsedByAuthKeys(t *testing.T) { assert.False(t, ok) }) } + +func TestReader_GetLogsExcludesOperations(t *testing.T) { + runReaderSuite(t, func(t *testing.T, store LogStore, reader Reader) { + ctx := context.Background() + base := time.Date(2026, 1, 16, 12, 0, 0, 0, time.UTC) + paths := []string{ + "/v1/chat/completions", "/mcp", "/mcp/github", "/mcpx", + "/v1/audio/speech", "/v1/audio/speech/", "/v1/audio/transcriptions", + "/p/openai/v1/models", "/sso/callback", "", + } + entries := make([]*LogEntry, 0, len(paths)) + for i, path := range paths { + entries = append(entries, &LogEntry{ + ID: fmt.Sprintf("op-%d", i), + Timestamp: base.Add(time.Duration(i) * time.Minute), + Path: path, + }) + } + require.NoError(t, store.WriteBatch(ctx, entries)) + + tests := []struct { + name string + ops []core.Operation + want []string + }{ + {name: "mcp prefix", ops: []core.Operation{core.OperationMCP}, want: []string{ + "/v1/chat/completions", "/mcpx", "/v1/audio/speech", "/v1/audio/speech/", + "/v1/audio/transcriptions", "/p/openai/v1/models", "/sso/callback", "", + }}, + {name: "exact with trailing slash and prefix", ops: []core.Operation{ + core.OperationAudioSpeech, core.OperationProviderPassthrough, + }, want: []string{ + "/v1/chat/completions", "/mcp", "/mcp/github", "/mcpx", + "/v1/audio/transcriptions", "/sso/callback", "", + }}, + {name: "every classified type keeps unclassified rows", ops: []core.Operation{ + core.OperationChatCompletions, core.OperationMCP, core.OperationAudioSpeech, + core.OperationAudioTranscriptions, core.OperationProviderPassthrough, + }, want: []string{"/mcpx", "/sso/callback", ""}}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result, err := reader.GetLogs(ctx, LogQueryParams{ExcludeOperations: tt.ops, Limit: 50}) + require.NoError(t, err) + got := make([]string, 0, len(result.Entries)) + for _, entry := range result.Entries { + got = append(got, entry.Path) + } + assert.ElementsMatch(t, tt.want, got) + assert.Equal(t, len(tt.want), result.Total) + }) + } + }) +} diff --git a/internal/core/endpoint_operations.go b/internal/core/endpoint_operations.go new file mode 100644 index 000000000..6ab6df986 --- /dev/null +++ b/internal/core/endpoint_operations.go @@ -0,0 +1,58 @@ +package core + +import "strings" + +// OperationPaths lists the request paths that DescribeEndpoint classifies as +// one operation, in a shape storage filters can match: Exact paths compare by +// equality, and each Prefix matches itself or anything under "Prefix/". +type OperationPaths struct { + Exact []string + Prefixes []string +} + +// operationPaths mirrors describeEndpointPath. TestOperationPathsMatchDescribeEndpoint +// keeps the two in sync. +var operationPaths = map[Operation]OperationPaths{ + OperationChatCompletions: {Exact: []string{"/v1/chat/completions", "/v1/messages", "/v1/messages/count_tokens"}}, + OperationResponses: {Prefixes: []string{"/v1/responses"}}, + OperationConversations: {Prefixes: []string{"/v1/conversations"}}, + OperationEmbeddings: {Exact: []string{"/v1/embeddings"}}, + OperationBatches: {Prefixes: []string{"/v1/batches", "/v1/messages/batches"}}, + OperationFiles: {Prefixes: []string{"/v1/files"}}, + OperationAudioSpeech: {Exact: []string{"/v1/audio/speech"}}, + OperationAudioTranscriptions: {Exact: []string{"/v1/audio/transcriptions"}}, + OperationAudioTranslations: {Exact: []string{"/v1/audio/translations"}}, + OperationImageGenerations: {Exact: []string{"/v1/images/generations"}}, + OperationImageEdits: {Exact: []string{"/v1/images/edits"}}, + OperationRealtime: {Exact: []string{ + "/v1/realtime", "/v1/realtime/calls", "/v1/realtime/client_secrets", + "/v1/realtime/translations", "/v1/realtime/translations/calls", "/v1/realtime/translations/client_secrets", + }}, + OperationMCP: {Prefixes: []string{"/mcp"}}, + OperationProviderPassthrough: {Prefixes: []string{"/p"}}, +} + +// PathsForOperation returns the paths of a known operation. +func PathsForOperation(op Operation) (OperationPaths, bool) { + paths, ok := operationPaths[op] + return paths, ok +} + +// ParseOperations parses a comma-separated operation list, ignoring blanks +// and duplicates. It reports the first unknown name. +func ParseOperations(raw string) ([]Operation, string, bool) { + var ops []Operation + seen := map[Operation]bool{} + for part := range strings.SplitSeq(raw, ",") { + op := Operation(strings.ToLower(strings.TrimSpace(part))) + if op == "" || seen[op] { + continue + } + if _, ok := operationPaths[op]; !ok { + return nil, string(op), false + } + seen[op] = true + ops = append(ops, op) + } + return ops, "", true +} diff --git a/internal/core/endpoint_operations_test.go b/internal/core/endpoint_operations_test.go new file mode 100644 index 000000000..ba2bd8edd --- /dev/null +++ b/internal/core/endpoint_operations_test.go @@ -0,0 +1,72 @@ +package core + +import ( + "slices" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestOperationPathsMatchDescribeEndpoint(t *testing.T) { + for op, paths := range operationPaths { + for _, path := range paths.Exact { + assert.Equal(t, op, DescribeEndpointPath(path).Operation, path) + } + for _, prefix := range paths.Prefixes { + assert.Equal(t, op, DescribeEndpointPath(prefix+"/x").Operation, prefix+"/x") + } + } +} + +func TestOperationPathsCoverClassifiedPaths(t *testing.T) { + paths := []string{ + "/v1/chat/completions", "/v1/messages", "/v1/messages/count_tokens", + "/v1/responses", "/v1/responses/resp_1/input_items", "/v1/conversations/conv_1", + "/v1/embeddings", "/v1/batches/b_1/cancel", "/v1/messages/batches/b_1", + "/v1/files/f_1/content", "/v1/audio/speech", "/v1/audio/transcriptions", + "/v1/audio/translations", "/v1/images/generations", "/v1/images/edits", + "/v1/realtime", "/v1/realtime/translations/calls", "/mcp", "/mcp/github", + "/p/openai/v1/models", + } + for _, path := range paths { + want := DescribeEndpointPath(path).Operation + require.NotEmpty(t, want, path) + rule, ok := PathsForOperation(want) + require.True(t, ok, path) + assert.True(t, rule.matches(path), path) + } +} + +func (p OperationPaths) matches(path string) bool { + if slices.Contains(p.Exact, path) { + return true + } + for _, prefix := range p.Prefixes { + if path == prefix || len(path) > len(prefix) && path[:len(prefix)+1] == prefix+"/" { + return true + } + } + return false +} + +func TestParseOperations(t *testing.T) { + tests := []struct { + name string + raw string + want []Operation + unknown string + }{ + {name: "empty", raw: ""}, + {name: "list", raw: " MCP, audio_speech,,mcp ", want: []Operation{OperationMCP, OperationAudioSpeech}}, + {name: "unknown", raw: "mcp,nope", unknown: "nope"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, unknown, ok := ParseOperations(tt.raw) + assert.Equal(t, tt.unknown == "", ok) + assert.Equal(t, tt.unknown, unknown) + assert.Equal(t, tt.want, got) + }) + } +} diff --git a/web/dashboard/messages/de.json b/web/dashboard/messages/de.json index 4fd061389..5add5e321 100644 --- a/web/dashboard/messages/de.json +++ b/web/dashboard/messages/de.json @@ -225,6 +225,18 @@ "audit_filter_all_modes": "Alle Modi", "audit_filter_streaming": "Streaming", "audit_filter_non_streaming": "Ohne Streaming", + "audit_filter_type_label": "Filter nach Anfragetyp", + "audit_filter_all_types": "Alle Typen", + "audit_filter_types_hidden": "Typen: {count} ausgeblendet", + "audit_type_chat": "Chat", + "audit_type_responses": "Responses", + "audit_type_embeddings": "Embeddings", + "audit_type_audio": "Audio", + "audit_type_images": "Bilder", + "audit_type_batches": "Batches & Dateien", + "audit_type_realtime": "Realtime", + "audit_type_passthrough": "Passthrough", + "audit_type_mcp": "MCP", "audit_live_connecting": "Verbindung zum Live-Stream…", "audit_live": "Live", "audit_live_pause_date_range": "Live pausiert — der gewählte Zeitraum schließt heute nicht ein. Stelle ihn auf heute ein, um fortzufahren.", diff --git a/web/dashboard/messages/en.json b/web/dashboard/messages/en.json index 2a60dae34..9e12f9e6e 100644 --- a/web/dashboard/messages/en.json +++ b/web/dashboard/messages/en.json @@ -225,6 +225,18 @@ "audit_filter_all_modes": "All Modes", "audit_filter_streaming": "Streaming", "audit_filter_non_streaming": "Non-streaming", + "audit_filter_type_label": "Request type filter", + "audit_filter_all_types": "All Types", + "audit_filter_types_hidden": "Types: {count} hidden", + "audit_type_chat": "Chat", + "audit_type_responses": "Responses", + "audit_type_embeddings": "Embeddings", + "audit_type_audio": "Audio", + "audit_type_images": "Images", + "audit_type_batches": "Batches & files", + "audit_type_realtime": "Realtime", + "audit_type_passthrough": "Passthrough", + "audit_type_mcp": "MCP", "audit_live_connecting": "Live stream connecting…", "audit_live": "Live", "audit_live_pause_date_range": "Live paused — the selected date range does not include today. Set it to today to resume.", diff --git a/web/dashboard/messages/pl.json b/web/dashboard/messages/pl.json index ee9a53914..ba9c432e1 100644 --- a/web/dashboard/messages/pl.json +++ b/web/dashboard/messages/pl.json @@ -227,6 +227,18 @@ "audit_filter_all_modes": "Wszystkie tryby", "audit_filter_streaming": "Streaming", "audit_filter_non_streaming": "Bez streamingu", + "audit_filter_type_label": "Filtr typu żądania", + "audit_filter_all_types": "Wszystkie typy", + "audit_filter_types_hidden": "Typy: ukryte {count}", + "audit_type_chat": "Chat", + "audit_type_responses": "Responses", + "audit_type_embeddings": "Embeddingi", + "audit_type_audio": "Audio", + "audit_type_images": "Obrazy", + "audit_type_batches": "Batche i pliki", + "audit_type_realtime": "Realtime", + "audit_type_passthrough": "Passthrough", + "audit_type_mcp": "MCP", "audit_live_connecting": "Łączenie ze strumieniem Live…", "audit_live": "Live", "audit_live_pause_date_range": "Live wstrzymany — wybrany zakres dat nie obejmuje dzisiaj. Ustaw datę dzisiejszą, aby wznowić.", diff --git a/web/dashboard/messages/zh-CN.json b/web/dashboard/messages/zh-CN.json index db1ba6496..16c2eb5f1 100644 --- a/web/dashboard/messages/zh-CN.json +++ b/web/dashboard/messages/zh-CN.json @@ -216,6 +216,18 @@ "audit_filter_all_modes": "全部模式", "audit_filter_streaming": "流式", "audit_filter_non_streaming": "非流式", + "audit_filter_type_label": "请求类型筛选", + "audit_filter_all_types": "全部类型", + "audit_filter_types_hidden": "类型:已隐藏 {count} 个", + "audit_type_chat": "对话", + "audit_type_responses": "Responses", + "audit_type_embeddings": "嵌入", + "audit_type_audio": "音频", + "audit_type_images": "图像", + "audit_type_batches": "批处理与文件", + "audit_type_realtime": "实时", + "audit_type_passthrough": "透传", + "audit_type_mcp": "MCP", "audit_live_connecting": "正在连接实时流…", "audit_live": "实时", "audit_live_pause_date_range": "实时流已暂停 — 所选日期范围未包含今天。将其设为今天即可恢复。", diff --git a/web/dashboard/src/pages/audit-logs/AuditFilters.svelte b/web/dashboard/src/pages/audit-logs/AuditFilters.svelte index 3f7744b96..f3adf91ca 100644 --- a/web/dashboard/src/pages/audit-logs/AuditFilters.svelte +++ b/web/dashboard/src/pages/audit-logs/AuditFilters.svelte @@ -1,15 +1,35 @@
@@ -69,6 +89,25 @@ +
+ + {hiddenCount > 0 ? m.audit_filter_types_hidden({ count: hiddenCount }) : m.audit_filter_all_types()} + +
+ {m.audit_filter_type_label()} + {#each AUDIT_TYPES as type (type.key)} + {@const visible = !auditList.auditHiddenTypes.includes(type.key)} + + {/each} +
+
{/each} @@ -51,10 +54,15 @@ white-space: nowrap; } - .segmented-btn:hover { + .segmented-btn:hover:not(:disabled) { color: var(--text); } + .segmented-btn:disabled { + cursor: not-allowed; + opacity: 0.6; + } + .segmented-btn.active { background: var(--accent); color: #fff; diff --git a/web/dashboard/src/pages/mcp-servers/McpCatalogModal.svelte b/web/dashboard/src/pages/mcp-servers/McpCatalogModal.svelte index 447f4921d..0ec1d488e 100644 --- a/web/dashboard/src/pages/mcp-servers/McpCatalogModal.svelte +++ b/web/dashboard/src/pages/mcp-servers/McpCatalogModal.svelte @@ -1,7 +1,8 @@ mcpServers.closeCatalog()}> @@ -47,12 +54,26 @@ {#each sections as section (section.key)}

{section.title}

-
    + {#if section.hint} +

    {section.hint}

    + {/if} +
      {#each section.items as item (item.key)}
    • {item.name} + {#if item.readOnly} + {m.mcp_tools_read_only()} + {/if} + {#if item.destructive} + {m.mcp_tools_destructive()} + {/if} {#if item.aggregated}
      + {#if editable && !mcpServers.catalogLoading && !mcpServers.catalogError} + + {/if}
@@ -107,11 +133,38 @@ gap: 12px; } + .mcp-catalog-section-hint { + margin: 0 0 8px; + font-size: 12px; + } + .mcp-catalog-item-name { font-size: 13px; overflow-wrap: anywhere; } + .mcp-catalog-list-excluded .mcp-catalog-item-name { + color: var(--text-muted); + text-decoration: line-through; + } + + .mcp-catalog-badge { + display: inline-block; + margin-left: 6px; + padding: 0 6px; + border: 1px solid var(--border); + border-radius: 999px; + color: var(--text-muted); + font-size: 11px; + line-height: 18px; + vertical-align: middle; + } + + .mcp-catalog-badge-danger { + border-color: color-mix(in srgb, var(--danger) 35%, var(--border)); + color: var(--danger); + } + .mcp-catalog-item-aggregated { margin-top: 2px; color: var(--text-muted); diff --git a/web/dashboard/src/pages/mcp-servers/McpServerEditor.svelte b/web/dashboard/src/pages/mcp-servers/McpServerEditor.svelte index 8122b2da0..4dc408fad 100644 --- a/web/dashboard/src/pages/mcp-servers/McpServerEditor.svelte +++ b/web/dashboard/src/pages/mcp-servers/McpServerEditor.svelte @@ -2,12 +2,14 @@ // MCP server editor modal (create + edit), built on the shared EditorDialog // shell. The slug is derived from the name until manually edited and becomes // immutable once the server exists. Saved header values arrive masked as - // "***"; leaving them unchanged keeps the stored secret on save. + // "***"; leaving them unchanged keeps the stored secret on save. Tool + // exposure lives in McpToolPicker. import TableActionButton from "$lib/components/atoms/TableActionButton.svelte"; import Icon from "$lib/components/atoms/Icon.svelte"; import EnabledToggle from "$lib/components/atoms/EnabledToggle.svelte"; import FormField from "$lib/components/molecules/FormField.svelte"; import EditorDialog from "$lib/components/organisms/EditorDialog.svelte"; + import McpToolPicker from "./McpToolPicker.svelte"; import { mcpServers } from "./mcpServers.svelte.js"; import { Plus, Trash2 } from "lucide"; import * as m from "$lib/paraglide/messages.js"; @@ -125,6 +127,8 @@ + +
-
- - -
- -
- - -
-
+
+ + + {m.mcp_disallowed_user_paths_help()} +
+
{formatNumber(server.tool_count || 0)} + {#if server.excluded_tool_count > 0} + {m.mcp_tools_excluded_count({ count: formatNumber(server.excluded_tool_count) })} + {/if}
{mcpServerSubCountsLabel(server)}
@@ -138,6 +143,12 @@ white-space: nowrap; } + .mcp-server-excluded-count { + margin-left: 6px; + color: var(--text-muted); + font-size: 12px; + } + /* The server table has seven information-dense columns. Preserve readable cells on narrow screens and let the wrapper scroll instead of squeezing them. */ diff --git a/web/dashboard/src/pages/mcp-servers/McpToolPicker.svelte b/web/dashboard/src/pages/mcp-servers/McpToolPicker.svelte new file mode 100644 index 000000000..11a9a8c51 --- /dev/null +++ b/web/dashboard/src/pages/mcp-servers/McpToolPicker.svelte @@ -0,0 +1,321 @@ + + +
+
+ {m.mcp_tools_title()} + {#if summary.total > 0} + {m.mcp_tools_summary(summary)} + {/if} +
+ +
+ {m.mcp_tools_mode_label()} + mcpServers.switchToolMode(mode)} + /> +
+ + {allowMode ? m.mcp_tools_mode_allow_help() : m.mcp_tools_mode_exclude_help()} + {#if modeLocked} + {m.mcp_tools_mode_locked()} + {/if} + + + {#if mcpServers.editorTools.loading} + + {:else if discovered.length === 0} +

+ {mcpServers.editorTools.error ? m.mcp_tools_load_failed() : m.mcp_tools_pending()} +

+ {:else} +
+ (mcpServers.toolQuery = "")} + /> + + +
+ {/if} + + {#if rows.length > 0} +
    + {#each rows as row (row.name)} +
  • + {#if row.missing} + + {row.name} + {m.mcp_tools_missing()} + + mcpServers.removeToolName(row.name)} + > + + + {:else} + + {/if} +
  • + {/each} +
+ {:else if discovered.length > 0} +

{m.mcp_tools_no_match()}

+ {/if} + +
+ + +
+ {m.mcp_tools_add_help()} +
+ + diff --git a/web/dashboard/src/pages/mcp-servers/mcp-servers.js b/web/dashboard/src/pages/mcp-servers/mcp-servers.js index 57dc34b8c..42111893d 100644 --- a/web/dashboard/src/pages/mcp-servers/mcp-servers.js +++ b/web/dashboard/src/pages/mcp-servers/mcp-servers.js @@ -3,6 +3,13 @@ import * as m from "../../lib/paraglide/messages.js"; +// Tool filter modes. "exclude" stores disallowed_tools: every tool is exposed +// except the listed ones, so tools the server adds later appear automatically. +// "allow" stores allowed_tools: only the listed tools are exposed, so new +// tools stay hidden until selected. +export const MCP_TOOL_MODE_EXCLUDE = "exclude"; +export const MCP_TOOL_MODE_ALLOW = "allow"; + export function defaultMcpServerForm() { return { name: "", @@ -12,9 +19,10 @@ export function defaultMcpServerForm() { description: "", enabled: true, headers: [], - allowed_tools: "", - disallowed_tools: "", + tool_mode: MCP_TOOL_MODE_EXCLUDE, + tool_names: [], user_paths: "", + disallowed_user_paths: "", tool_timeout_seconds: "", }; } @@ -25,6 +33,7 @@ export function defaultMcpCatalog() { status: "", instructions: "", tools: [], + excluded_tools: [], prompts: [], resources: [], templates: [], @@ -202,17 +211,14 @@ export function mcpServerFormFromServer(server) { description: String(server.description || "").trim(), enabled: server.enabled !== false, headers: mcpHeadersToRows(server.headers), - allowed_tools: (Array.isArray(server.allowed_tools) - ? server.allowed_tools - : [] - ).join(", "), - disallowed_tools: (Array.isArray(server.disallowed_tools) - ? server.disallowed_tools - : [] - ).join(", "), + ...mcpToolFilterFromServer(server), user_paths: (Array.isArray(server.user_paths) ? server.user_paths : []).join( "\n", ), + disallowed_user_paths: (Array.isArray(server.disallowed_user_paths) + ? server.disallowed_user_paths + : [] + ).join("\n"), tool_timeout_seconds: server.tool_timeout_seconds ? String(server.tool_timeout_seconds) : "", @@ -247,6 +253,13 @@ export function buildMcpServerPayload(form, mode, servers) { if (!url) { return { error: m.mcp_url_required() }; } + const toolNames = uniqueToolNames(form.tool_names); + const allowMode = form.tool_mode === MCP_TOOL_MODE_ALLOW; + if (allowMode && toolNames.length === 0) { + // An empty allowed_tools list means "no restriction" on the gateway, the + // opposite of what an operator who unchecked everything expects. + return { error: m.mcp_tools_allow_empty() }; + } let toolTimeoutSeconds; const rawTimeout = String(form.tool_timeout_seconds || "").trim(); if (rawTimeout !== "") { @@ -267,9 +280,10 @@ export function buildMcpServerPayload(form, mode, servers) { headers: mcpHeaderRowsToObject(form.headers), description: String(form.description || "").trim(), enabled: Boolean(form.enabled), - allowed_tools: splitCommaList(form.allowed_tools), - disallowed_tools: splitCommaList(form.disallowed_tools), + allowed_tools: allowMode ? toolNames : [], + disallowed_tools: allowMode ? [] : toolNames, user_paths: normalizeMcpUserPaths(form.user_paths), + disallowed_user_paths: normalizeMcpUserPaths(form.disallowed_user_paths), tool_timeout_seconds: toolTimeoutSeconds, }, }; @@ -289,6 +303,7 @@ export function normalizeMcpCatalog(name, payload) { status: String(source.status || "").trim(), instructions: String(source.instructions || "").trim(), tools: list(source.tools), + excluded_tools: list(source.excluded_tools), prompts: list(source.prompts), resources: list(source.resources), templates: list(source.templates), @@ -316,9 +331,21 @@ export function mcpCatalogSections(catalog) { name: String(item.name || ""), aggregated: mcpNamespacedName(source, item.name), description: String(item.description || "").trim(), + readOnly: item.read_only === true, + destructive: item.destructive === true, }); const sections = [ { key: "tools", title: m.mcp_catalog_tools(), items: (source.tools || []).map(feature("tool")) }, + { + key: "excluded_tools", + title: m.mcp_catalog_excluded_tools(), + hint: m.mcp_catalog_excluded_tools_hint(), + excluded: true, + items: (source.excluded_tools || []).map((item) => ({ + ...feature("excluded")(item), + aggregated: "", + })), + }, { key: "prompts", title: m.mcp_catalog_prompts(), @@ -351,3 +378,141 @@ export function mcpCatalogSections(catalog) { export function mcpCatalogIsEmpty(catalog) { return mcpCatalogSections(catalog).length === 0; } + +// --- tool picker ----------------------------------------------------------- + +function uniqueToolNames(names) { + const seen = new Set(); + const result = []; + (Array.isArray(names) ? names : []).forEach((value) => { + const name = String(value || "").trim(); + if (name && !seen.has(name)) { + seen.add(name); + result.push(name); + } + }); + return result; +} + +// mcpToolFilterFromServer maps the stored allow/deny lists onto one editor +// mode. Config accepts both lists at once (deny applies after allow); the +// allow mode with allowed − disallowed exposes exactly the same tools. +export function mcpToolFilterFromServer(server) { + const allowed = uniqueToolNames(server && server.allowed_tools); + const disallowed = uniqueToolNames(server && server.disallowed_tools); + if (allowed.length > 0) { + return { + tool_mode: MCP_TOOL_MODE_ALLOW, + tool_names: allowed.filter((name) => !disallowed.includes(name)), + }; + } + return { tool_mode: MCP_TOOL_MODE_EXCLUDE, tool_names: disallowed }; +} + +// mcpDiscoveredTools merges the exposed and excluded catalog lists into one +// name-sorted list: the full set the upstream reports, whatever the filters. +export function mcpDiscoveredTools(catalog) { + const source = catalog || {}; + const byName = new Map(); + [...(source.tools || []), ...(source.excluded_tools || [])].forEach((item) => { + const name = String((item && item.name) || "").trim(); + if (name && !byName.has(name)) { + byName.set(name, { + name, + description: String(item.description || "").trim(), + readOnly: item.read_only === true, + destructive: item.destructive === true, + }); + } + }); + return [...byName.values()].sort((a, b) => a.name.localeCompare(b.name)); +} + +export function mcpToolExposed(form, name) { + const listed = (form.tool_names || []).includes(name); + return form.tool_mode === MCP_TOOL_MODE_ALLOW ? listed : !listed; +} + +// setMcpToolsExposed returns the tool_names list after exposing or hiding +// every given tool in the form's current mode. +export function setMcpToolsExposed(form, names, exposed) { + const targets = uniqueToolNames(names); + const listed = form.tool_mode === MCP_TOOL_MODE_ALLOW ? exposed : !exposed; + const current = uniqueToolNames(form.tool_names); + if (listed) { + return uniqueToolNames([...current, ...targets]); + } + return current.filter((name) => !targets.includes(name)); +} + +// mcpToolModeSwitchable reports whether the mode can flip without changing +// what is exposed. Converting between an allowlist and a denylist needs the +// full tool set; without it, an allowlist would become an empty denylist, +// which exposes every tool. An empty list is safe to flip either way. +export function mcpToolModeSwitchable(form, discovered) { + return (discovered || []).length > 0 || uniqueToolNames(form.tool_names).length === 0; +} + +// switchMcpToolMode flips the filter mode while keeping every discovered +// tool's exposure unchanged; only the treatment of future tools changes. +// Names the server does not report are meaningful only in the old mode. +// When the flip is not safe (see mcpToolModeSwitchable) the form is kept. +export function switchMcpToolMode(form, mode, discovered) { + const next = mode === MCP_TOOL_MODE_ALLOW ? MCP_TOOL_MODE_ALLOW : MCP_TOOL_MODE_EXCLUDE; + if (next === form.tool_mode || !mcpToolModeSwitchable(form, discovered)) { + return { tool_mode: form.tool_mode, tool_names: uniqueToolNames(form.tool_names) }; + } + const names = (discovered || []).map((tool) => tool.name); + const exposed = names.filter((name) => mcpToolExposed(form, name)); + return { + tool_mode: next, + tool_names: + next === MCP_TOOL_MODE_ALLOW + ? exposed + : names.filter((name) => !exposed.includes(name)), + }; +} + +// mcpToolPickerRows lists every discovered tool with its exposure in the +// form, followed by listed names the server does not report (typos, removed +// tools, or names added before the server connected). query narrows by name +// or description. +export function mcpToolPickerRows(form, discovered, query) { + const needle = String(query || "").trim().toLowerCase(); + const known = new Set((discovered || []).map((tool) => tool.name)); + const rows = (discovered || []).map((tool) => ({ + ...tool, + exposed: mcpToolExposed(form, tool.name), + missing: false, + })); + uniqueToolNames(form.tool_names).forEach((name) => { + if (!known.has(name)) { + rows.push({ + name, + description: "", + readOnly: false, + destructive: false, + exposed: form.tool_mode === MCP_TOOL_MODE_ALLOW, + missing: true, + }); + } + }); + if (!needle) { + return rows; + } + return rows.filter( + (row) => + row.name.toLowerCase().includes(needle) || + row.description.toLowerCase().includes(needle), + ); +} + +// mcpToolSelectionSummary counts discovered tools only: a listed name the +// server does not report is neither exposed nor hidden today. +export function mcpToolSelectionSummary(form, discovered) { + const total = (discovered || []).length; + const exposed = (discovered || []).filter((tool) => + mcpToolExposed(form, tool.name), + ).length; + return { exposed, total }; +} diff --git a/web/dashboard/src/pages/mcp-servers/mcpServers.svelte.js b/web/dashboard/src/pages/mcp-servers/mcpServers.svelte.js index e825b43a8..9a143a291 100644 --- a/web/dashboard/src/pages/mcp-servers/mcpServers.svelte.js +++ b/web/dashboard/src/pages/mcp-servers/mcpServers.svelte.js @@ -13,13 +13,17 @@ import { defaultMcpServerForm, deriveMcpServerSlug, filterMcpServers, + mcpDiscoveredTools, mcpServerFormFromServer, mcpPollShouldRetry, mcpServerSlug, mcpServersNeedPolling, mcpServerStatus, normalizeMcpCatalog, + setMcpToolsExposed, + switchMcpToolMode, MCP_SERVERS_POLL_MS, + MCP_TOOL_MODE_ALLOW, } from "./mcp-servers.js"; class McpServersState { @@ -38,6 +42,15 @@ class McpServersState { advancedOpen = $state(false); form = $state(defaultMcpServerForm()); + // Editor tool picker: every tool the server reports, exposed or not. A new + // server, or one that never listed, has none yet. + editorTools = $state({ loading: false, error: "", tools: [] }); + toolQuery = $state(""); + toolDraft = $state(""); + // Bumped per editor catalog load, so a response for an editor session that + // was closed and reopened (same slug) cannot overwrite the newer one. + #toolLoadSeq = 0; + deletingName = $state(""); reconnectingName = $state(""); @@ -179,6 +192,7 @@ class McpServersState { this.advancedOpen = false; this.error = ""; this.form = defaultMcpServerForm(); + this.#resetEditorTools(); this.formOpen = true; } @@ -191,7 +205,9 @@ class McpServersState { this.advancedOpen = false; this.error = ""; this.form = mcpServerFormFromServer(server); + this.#resetEditorTools(); this.formOpen = true; + void this.#loadEditorTools(server); } closeForm() { @@ -201,6 +217,57 @@ class McpServersState { this.advancedOpen = false; this.error = ""; this.form = defaultMcpServerForm(); + this.#resetEditorTools(); + } + + #resetEditorTools() { + this.#toolLoadSeq += 1; + this.editorTools = { loading: false, error: "", tools: [] }; + this.toolQuery = ""; + this.toolDraft = ""; + } + + async #loadEditorTools(server) { + const seq = ++this.#toolLoadSeq; + this.editorTools = { ...this.editorTools, loading: true, error: "" }; + const loaded = await this.#fetchCatalog(server); + // The editor may have closed, reopened, or moved to another server. + if (seq !== this.#toolLoadSeq) { + return; + } + if (loaded.stale) { + this.editorTools = { ...this.editorTools, loading: false }; + return; + } + this.editorTools = { + loading: false, + error: loaded.error || "", + tools: loaded.error ? [] : mcpDiscoveredTools(loaded.catalog), + }; + } + + setToolsExposed(names, exposed) { + this.form.tool_names = setMcpToolsExposed(this.form, names, exposed); + } + + switchToolMode(mode) { + const next = switchMcpToolMode(this.form, mode, this.editorTools.tools); + this.form.tool_mode = next.tool_mode; + this.form.tool_names = next.tool_names; + } + + addToolDraft() { + const name = this.toolDraft.trim(); + if (!name) { + return; + } + // Adding a name lists it in the current mode: excluded or allowed. + this.setToolsExposed([name], this.form.tool_mode === MCP_TOOL_MODE_ALLOW); + this.toolDraft = ""; + } + + removeToolName(name) { + this.form.tool_names = (this.form.tool_names || []).filter((item) => item !== name); } syncSlugFromName() { @@ -377,7 +444,6 @@ class McpServersState { // so it does not map onto the shared list/mutation ladder. async openCatalog(server) { - const name = String((server && server.name) || "").trim(); const slug = mcpServerSlug(server); if (!slug) { return; @@ -392,39 +458,62 @@ class McpServersState { status: mcpServerStatus(server), }; + const loaded = await this.#fetchCatalog(server); + this.catalogLoading = false; + if (loaded.stale) { + return; + } + if (loaded.error) { + this.catalogError = loaded.error; + return; + } + this.catalog = loaded.catalog; + } + + // chooseToolsFromCatalog jumps from the read-only inspector to the editor's + // tool picker for the same server. + chooseToolsFromCatalog() { + const server = (this.servers || []).find( + (item) => mcpServerSlug(item) === this.catalog.server, + ); + if (!server || server.managed) { + return; + } + this.closeCatalog(); + this.openEdit(server); + } + + // #fetchCatalog resolves to { catalog }, { error }, or { stale }. + async #fetchCatalog(server) { + const name = String((server && server.name) || "").trim(); + const slug = mcpServerSlug(server); try { const result = await getJSON( "/admin/mcp-servers/" + encodeURIComponent(slug) + "/catalog", { label: "mcp server catalog" }, ); if (result.stale) { - return; + return { stale: true }; } if (result.status === 503) { this.available = false; - this.catalogError = m.mcp_unavailable(); - return; + return { error: m.mcp_unavailable() }; } if (result.status === 404) { - this.catalogError = m.mcp_not_found({ name }); - return; + return { error: m.mcp_not_found({ name }) }; } if (!result.ok) { - this.catalogError = - result.status === 401 - ? m.common_authentication_required() - : errorPayloadMessage( - result.data, - m.mcp_catalog_load_failed(), - ); - return; + return { + error: + result.status === 401 + ? m.common_authentication_required() + : errorPayloadMessage(result.data, m.mcp_catalog_load_failed()), + }; } - this.catalog = normalizeMcpCatalog(slug, result.data); + return { catalog: normalizeMcpCatalog(slug, result.data) }; } catch (e) { console.error("Failed to load MCP server catalog:", e); - this.catalogError = m.mcp_catalog_load_failed(); - } finally { - this.catalogLoading = false; + return { error: m.mcp_catalog_load_failed() }; } } diff --git a/web/dashboard/tests/mcp-servers.test.js b/web/dashboard/tests/mcp-servers.test.js index e886c389b..3266609e2 100644 --- a/web/dashboard/tests/mcp-servers.test.js +++ b/web/dashboard/tests/mcp-servers.test.js @@ -28,6 +28,13 @@ import { normalizeMcpCatalog, splitCommaList, normalizeMcpUserPaths, + mcpDiscoveredTools, + mcpToolFilterFromServer, + mcpToolModeSwitchable, + mcpToolPickerRows, + mcpToolSelectionSummary, + setMcpToolsExposed, + switchMcpToolMode, } from "../src/pages/mcp-servers/mcp-servers.js"; test("deriveMcpServerSlug normalizes display names and falls back to a hash", () => { @@ -242,9 +249,10 @@ test("buildMcpServerPayload produces the normalized PUT payload", () => { { name: "Authorization", value: "***" }, { name: "", value: "ignored" }, ], - allowed_tools: "search_issues, get_file", - disallowed_tools: "", + tool_mode: "allow", + tool_names: ["search_issues", " get_file ", "search_issues"], user_paths: "/team/alpha\n/team/beta", + disallowed_user_paths: " /team/alpha/contractors \n\n", tool_timeout_seconds: "45", }, "edit", @@ -263,6 +271,7 @@ test("buildMcpServerPayload produces the normalized PUT payload", () => { allowed_tools: ["search_issues", "get_file"], disallowed_tools: [], user_paths: ["/team/alpha", "/team/beta"], + disallowed_user_paths: ["/team/alpha/contractors"], tool_timeout_seconds: 45, }); }); @@ -302,6 +311,30 @@ test("buildMcpServerPayload validates required fields and timeout", () => { } }); +test("buildMcpServerPayload maps the tool mode onto one filter list", () => { + const base = { + ...defaultMcpServerForm(), + name: "github", + url: "https://mcp.example.com/mcp", + }; + + const excluded = buildMcpServerPayload( + { ...base, tool_names: ["delete_repo"] }, + "create", + [], + ); + assert.deepEqual(excluded.payload.allowed_tools, []); + assert.deepEqual(excluded.payload.disallowed_tools, ["delete_repo"]); + + // An empty allowlist would expose every tool on the gateway, so the + // "only selected" mode requires at least one selection. + assert.equal( + buildMcpServerPayload({ ...base, tool_mode: "allow", tool_names: [] }, "create", []) + .error, + "Select at least one tool, or switch new tools to exposed.", + ); +}); + test("buildMcpServerPayload rejects duplicate slugs only when creating", () => { const form = { ...defaultMcpServerForm(), @@ -330,6 +363,7 @@ test("mcpServerFormFromServer prefills the editor form", () => { allowed_tools: ["search_issues"], disallowed_tools: ["delete_repo"], user_paths: ["/team/alpha"], + disallowed_user_paths: ["/team/alpha/contractors"], tool_timeout_seconds: 45, }), { @@ -340,14 +374,136 @@ test("mcpServerFormFromServer prefills the editor form", () => { description: "Issue tools", enabled: false, headers: [{ name: "Authorization", value: "***" }], - allowed_tools: "search_issues", - disallowed_tools: "delete_repo", + tool_mode: "allow", + tool_names: ["search_issues"], user_paths: "/team/alpha", + disallowed_user_paths: "/team/alpha/contractors", tool_timeout_seconds: "45", }, ); }); +test("mcpToolFilterFromServer picks one mode without changing exposure", () => { + assert.deepEqual(mcpToolFilterFromServer({}), { tool_mode: "exclude", tool_names: [] }); + assert.deepEqual(mcpToolFilterFromServer({ disallowed_tools: ["delete_repo"] }), { + tool_mode: "exclude", + tool_names: ["delete_repo"], + }); + // Deny applies after allow on the gateway, so allowed − disallowed exposes + // the same tools. + assert.deepEqual( + mcpToolFilterFromServer({ + allowed_tools: ["read", "write"], + disallowed_tools: ["write"], + }), + { tool_mode: "allow", tool_names: ["read"] }, + ); +}); + +test("mcpDiscoveredTools merges exposed and excluded tools by name", () => { + const discovered = mcpDiscoveredTools( + normalizeMcpCatalog("github", { + tools: [{ name: "search", read_only: true }], + excluded_tools: [{ name: "delete_repo", description: "Delete", destructive: true }], + }), + ); + assert.deepEqual(discovered, [ + { name: "delete_repo", description: "Delete", readOnly: false, destructive: true }, + { name: "search", description: "", readOnly: true, destructive: false }, + ]); +}); + +test("tool picker toggles, bulk actions, and summary follow the mode", () => { + const discovered = [{ name: "a" }, { name: "b" }, { name: "c" }].map((tool) => ({ + ...tool, + description: "", + readOnly: false, + destructive: false, + })); + + const exclude = { tool_mode: "exclude", tool_names: [] }; + exclude.tool_names = setMcpToolsExposed(exclude, ["b"], false); + assert.deepEqual(exclude.tool_names, ["b"]); + assert.deepEqual(mcpToolSelectionSummary(exclude, discovered), { exposed: 2, total: 3 }); + assert.deepEqual(setMcpToolsExposed(exclude, ["a", "b", "c"], true), []); + assert.deepEqual(setMcpToolsExposed(exclude, ["a", "b", "c"], false), ["b", "a", "c"]); + + const allow = { tool_mode: "allow", tool_names: ["a"] }; + assert.deepEqual(setMcpToolsExposed(allow, ["c"], true), ["a", "c"]); + assert.deepEqual(setMcpToolsExposed(allow, ["a"], false), []); + assert.deepEqual(mcpToolSelectionSummary(allow, discovered), { exposed: 1, total: 3 }); +}); + +test("switchMcpToolMode keeps every discovered tool's exposure", () => { + const discovered = [{ name: "a" }, { name: "b" }, { name: "c" }]; + const toAllow = switchMcpToolMode( + { tool_mode: "exclude", tool_names: ["b", "ghost"] }, + "allow", + discovered, + ); + assert.deepEqual(toAllow, { tool_mode: "allow", tool_names: ["a", "c"] }); + + const back = switchMcpToolMode(toAllow, "exclude", discovered); + assert.deepEqual(back, { tool_mode: "exclude", tool_names: ["b"] }); +}); + +test("switchMcpToolMode keeps the list when the catalog is unknown", () => { + // Without the full tool set, an allowlist would become an empty denylist, + // which exposes every tool on save. + const allow = { tool_mode: "allow", tool_names: ["read"] }; + assert.equal(mcpToolModeSwitchable(allow, []), false); + assert.deepEqual(switchMcpToolMode(allow, "exclude", []), { + tool_mode: "allow", + tool_names: ["read"], + }); + + // An empty list flips safely, so a brand-new server can still pick a mode. + const fresh = { tool_mode: "exclude", tool_names: [] }; + assert.equal(mcpToolModeSwitchable(fresh, []), true); + assert.deepEqual(switchMcpToolMode(fresh, "allow", []), { + tool_mode: "allow", + tool_names: [], + }); +}); + +test("mcpToolPickerRows flags listed names the server does not report and filters", () => { + const discovered = [ + { name: "create_issue", description: "Create an issue", readOnly: false, destructive: false }, + { name: "delete_repo", description: "", readOnly: false, destructive: true }, + ]; + const form = { tool_mode: "exclude", tool_names: ["delete_repo", "delet_repo"] }; + + const rows = mcpToolPickerRows(form, discovered, ""); + assert.deepEqual( + rows.map((row) => [row.name, row.exposed, row.missing]), + [ + ["create_issue", true, false], + ["delete_repo", false, false], + ["delet_repo", false, true], + ], + ); + assert.deepEqual( + mcpToolPickerRows(form, discovered, "ISSUE").map((row) => row.name), + ["create_issue"], + ); +}); + +test("mcpCatalogSections lists excluded tools without an aggregated name", () => { + const sections = mcpCatalogSections( + normalizeMcpCatalog("github", { + tools: [{ name: "search" }], + excluded_tools: [{ name: "delete_repo", destructive: true }], + }), + ); + assert.deepEqual( + sections.map((section) => section.key), + ["tools", "excluded_tools"], + ); + assert.equal(sections[1].excluded, true); + assert.equal(sections[1].items[0].aggregated, ""); + assert.equal(sections[1].items[0].destructive, true); +}); + test("mcpCatalogSections derives aggregated /mcp names for tools and prompts only", () => { const catalog = normalizeMcpCatalog("github", { server: "github", @@ -378,6 +534,8 @@ test("mcpCatalogSections derives aggregated /mcp names for tools and prompts onl name: "create_issue", aggregated: "github_create_issue", description: "Create a GitHub issue", + readOnly: false, + destructive: false, }); assert.equal(tools[1].aggregated, "github_search_issues"); assert.equal(tools[1].description, ""); From e9d274e24b9c13043df8df84cb29b1abaa9512b1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jakub=20A=2E=20W=C4=85sek?= Date: Sat, 26 Sep 2026 18:22:25 +0200 Subject: [PATCH 09/23] fix(auth): keep header user path on MCP, realtime, and audio endpoints (#1096) * fix(auth): keep header user path on MCP, realtime, and audio endpoints * docs(mcp): mention configurable user path header * test(auth): cover configured user path header on /mcp --- docs/features/mcp-gateway.mdx | 5 ++ internal/server/auth.go | 26 +++++++- internal/server/master_key_user_path_test.go | 64 ++++++++++++++++++++ 3 files changed, 93 insertions(+), 2 deletions(-) diff --git a/docs/features/mcp-gateway.mdx b/docs/features/mcp-gateway.mdx index 393a4c85b..0d6cca447 100644 --- a/docs/features/mcp-gateway.mdx +++ b/docs/features/mcp-gateway.mdx @@ -281,6 +281,11 @@ and an exclusion wins over an allowed path. | Everyone except contractors | `disallowed_user_paths: ["/contractors"]` | | Engineering except its contractors | both of the above, scoped under `/engineering` | +The caller's user path comes from its API key when the key has one. Otherwise, +including with the master key, it comes from the `X-GoModel-User-Path` header +(or the header named by `USER_PATH_HEADER`), which is a quick way to check what +a given subtree sees. + Hidden servers are left out of `tools/list`, and `/mcp/{slug}` returns 404 for them. Like tool filters, visibility edits apply without reconnecting to the server and are checked on every call, so they also reach MCP sessions that are diff --git a/internal/server/auth.go b/internal/server/auth.go index a6f57580c..7d73db6b9 100644 --- a/internal/server/auth.go +++ b/internal/server/auth.go @@ -100,10 +100,11 @@ func NewAuthMiddleware(cfg AuthMiddlewareConfig) echo.MiddlewareFunc { // sessions. Hide every identity value installed by outer extension // middleware before validating the selected credential; clearing only // the response header would leave downstream context consumers scoped - // to the wrong principal. + // to the wrong principal. The caller's own user-path header is not + // such an identity and is restored (see transportOwnedUserPath). setAuthenticationUserHeader(c, "") ctx := ext.WithoutAuthentication(c.Request().Context()) - ctx = core.WithEffectiveUserPath(ctx, "") + ctx = core.WithEffectiveUserPath(ctx, transportOwnedUserPath(c.Request(), userPathHeaderName)) ctx = core.WithCredentialAllowedModels(ctx, nil) ctx = core.WithAccessScope(ctx, core.AccessScope{}) c.SetRequest(c.Request().WithContext(ctx)) @@ -158,6 +159,27 @@ func NewAuthMiddleware(cfg AuthMiddlewareConfig) echo.MiddlewareFunc { } } +// transportOwnedUserPath returns the request's user-path header for model +// endpoints that own their transport (MCP, realtime, audio uploads), and "" +// for every other endpoint. Those endpoints take no request snapshot, so +// RequestSnapshotCapture seeds the header path as the effective user path; +// without restoring it here, master-key and unbound-key callers lose their +// path there while ingress-managed endpoints keep it through the snapshot. +// Only this middleware and RequestSnapshotCapture write the header, so the +// value is the caller's own, never an outer extension session's. A key with +// a bound user path still overrides it in applyAuthKeyResult. +func transportOwnedUserPath(req *http.Request, headerName string) string { + desc := core.DescribeEndpoint(req.Method, req.URL.Path) + if desc.IngressManaged || !desc.ModelInteraction { + return "" + } + userPath, err := core.NormalizeUserPath(req.Header.Get(headerName)) + if err != nil { + return "" + } + return userPath +} + func hasRequestAuthenticators(authenticators []ext.RequestAuthenticator) bool { for _, authenticator := range authenticators { if !requestAuthenticatorIsNil(authenticator) { diff --git a/internal/server/master_key_user_path_test.go b/internal/server/master_key_user_path_test.go index ff17be0cf..d016d449e 100644 --- a/internal/server/master_key_user_path_test.go +++ b/internal/server/master_key_user_path_test.go @@ -101,3 +101,67 @@ func TestMasterKeyUserPathHeaderScopesRestrictedModelAccess(t *testing.T) { }) } } + +// TestTransportOwnedEndpointsKeepHeaderUserPath covers model endpoints that +// own their transport (MCP, realtime, audio uploads). They take no request +// snapshot, so the header path lives only in the effective user path, and the +// explicit-credential reset must not erase it: a master-key or unbound-key +// caller keeps its header path, a bound key still wins, and identity from an +// outer extension session never survives an explicit credential. +func TestTransportOwnedEndpointsKeepHeaderUserPath(t *testing.T) { + authenticator := mockAuthenticator{ + enabled: true, + tokenToID: map[string]string{"sk_bound": "key-bound", "sk_unbound": "key-unbound"}, + tokenPath: map[string]string{"sk_bound": "/team/bound"}, + } + tests := []struct { + name string + path string + token string + headerPath string + outerIdentity string + // configuredHeader is the server's USER_PATH_HEADER; empty keeps the default. + configuredHeader string + want string + }{ + {name: "master key on /mcp keeps header path", path: "/mcp", token: "master-key", headerPath: "/eng/platform", want: "/eng/platform"}, + {name: "master key on pinned /mcp/{server}", path: "/mcp/github", token: "master-key", headerPath: "/eng", want: "/eng"}, + {name: "master key on audio transcription", path: "/v1/audio/transcriptions", token: "master-key", headerPath: "/team/x", want: "/team/x"}, + {name: "master key on realtime", path: "/v1/realtime", token: "master-key", headerPath: "/team/x", want: "/team/x"}, + {name: "master key without header stays global", path: "/mcp", token: "master-key", want: ""}, + {name: "unbound managed key keeps header path", path: "/mcp", token: "sk_unbound", headerPath: "/eng/platform", want: "/eng/platform"}, + {name: "bound managed key wins over header", path: "/mcp", token: "sk_bound", headerPath: "/eng/platform", want: "/team/bound"}, + {name: "outer extension identity is dropped", path: "/mcp", token: "master-key", outerIdentity: "/ext/session", want: ""}, + {name: "header replaces outer extension identity", path: "/mcp", token: "master-key", outerIdentity: "/ext/session", headerPath: "/eng", want: "/eng"}, + {name: "configured header name on /mcp", path: "/mcp", token: "master-key", configuredHeader: "X-Tenant-Path", headerPath: "/eng", want: "/eng"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + var got string + auth := AuthMiddlewareWithAuthenticator("master-key", authenticator, nil, tt.configuredHeader)(func(c *echo.Context) error { + got = core.UserPathFromContext(c.Request().Context()) + return c.String(http.StatusOK, "ok") + }) + // An outer extension session installs its identity before the + // gateway's auth middleware runs. + outer := func(c *echo.Context) error { + if tt.outerIdentity != "" { + req := c.Request() + c.SetRequest(req.WithContext(core.WithEffectiveUserPath(req.Context(), tt.outerIdentity))) + } + return auth(c) + } + chain := RequestSnapshotCapture(tt.configuredHeader)(outer) + + opts := []echotest.Option{echotest.WithHeader("Authorization", "Bearer "+tt.token)} + if tt.headerPath != "" { + opts = append(opts, echotest.WithHeader(core.UserPathHeaderName(tt.configuredHeader), tt.headerPath)) + } + c, rec := echotest.Post(t, tt.path, `{}`, opts...) + + require.NoError(t, chain(c)) + require.Equal(t, http.StatusOK, rec.Code) + assert.Equal(t, tt.want, got) + }) + } +} From 5d9af0798bcf0ba3ba8907241e0621c5b4237819 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jakub=20A=2E=20W=C4=85sek?= Date: Sat, 26 Sep 2026 19:59:07 +0200 Subject: [PATCH 10/23] feat(jev): add native /v1/systemone endpoint (#1097) * feat(jev): add native /v1/systemone endpoint * feat(openrouter): serve System One decision models natively * fix(jev): guard in-place state edits and answer 404 before model resolution --- cmd/gomodel/docs/docs.go | 69 ++++ config/config.example.yaml | 11 +- docs/openapi.json | 94 +++++ docs/providers/jev.mdx | 113 +++++- docs/providers/overview.mdx | 12 +- internal/core/endpoint_operations.go | 1 + internal/core/endpoints.go | 13 +- internal/core/endpoints_test.go | 2 + internal/core/systemone.go | 13 + internal/core/workflow.go | 6 + internal/gateway/interfaces.go | 6 + internal/guardrails/integration_test.go | 29 ++ internal/guardrails/workflow_executor.go | 20 + .../plugins/exchange/systemone_request.go | 107 +++++ .../exchange/systemone_request_test.go | 127 ++++++ internal/providers/jev/jev.go | 12 +- internal/providers/jev/jev_test.go | 2 +- internal/providers/openrouter/openrouter.go | 19 +- .../providers/openrouter/openrouter_test.go | 18 +- .../providers/registry_normalization_test.go | 25 ++ internal/providers/router_models.go | 14 + internal/server/http.go | 3 + internal/server/messages_native.go | 54 +-- internal/server/model_validation.go | 7 +- internal/server/systemone_handler.go | 260 +++++++++++++ internal/server/systemone_handler_test.go | 367 ++++++++++++++++++ web/dashboard/messages/de.json | 1 + web/dashboard/messages/en.json | 1 + web/dashboard/messages/pl.json | 1 + web/dashboard/messages/zh-CN.json | 1 + .../src/pages/audit-logs/audit-operations.js | 2 + web/dashboard/tests/audit-operations.test.js | 2 + 32 files changed, 1345 insertions(+), 67 deletions(-) create mode 100644 internal/core/systemone.go create mode 100644 internal/plugins/exchange/systemone_request.go create mode 100644 internal/plugins/exchange/systemone_request_test.go create mode 100644 internal/server/systemone_handler.go create mode 100644 internal/server/systemone_handler_test.go diff --git a/cmd/gomodel/docs/docs.go b/cmd/gomodel/docs/docs.go index d45b61404..dacf4d64d 100644 --- a/cmd/gomodel/docs/docs.go +++ b/cmd/gomodel/docs/docs.go @@ -7504,6 +7504,75 @@ const docTemplate = `{ ] } }, + "/v1/systemone": { + "post": { + "description": "Available when a jev or openrouter provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", + "consumes": [ + "application/json" + ], + "produces": [ + "application/json" + ], + "tags": [ + "systemone" + ], + "summary": "Evaluate a System One decision request (Jev / Kev)", + "parameters": [ + { + "description": "System One request: model, state, and questions", + "name": "request", + "in": "body", + "required": true, + "schema": { + "type": "object" + } + } + ], + "responses": { + "200": { + "description": "System One answers, in the provider's shape", + "schema": { + "type": "object" + } + }, + "400": { + "description": "Bad Request", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "401": { + "description": "Unauthorized", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "404": { + "description": "Not Found", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "429": { + "description": "Too Many Requests", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "502": { + "description": "Bad Gateway", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ] + } + }, "/v1/usage": { "get": { "description": "Returns recorded usage, budget statuses, and rate limit statuses for the caller's effective user path (the path bound to the managed API key, or the user-path header for master-key callers).", diff --git a/config/config.example.yaml b/config/config.example.yaml index 9f3e7e239..d1213070e 100644 --- a/config/config.example.yaml +++ b/config/config.example.yaml @@ -591,10 +591,13 @@ providers: type: jev api_key: "${JEV_API_KEY}" # base_url defaults to "https://api.typesafe.ai". TypeSafe's System One - # API is a decision API with no OpenAI-compatible surface: requests go to - # POST /p/jev/v1/systemone, or point the TypeSafe SDK at /p/jev. A - # self-hosted Kev server speaks the same API without authentication: - # set base_url (e.g. "http://localhost:8009") and omit api_key. + # API is a decision API with no OpenAI-compatible surface: configuring + # this provider (or openrouter, which serves Jev natively) enables + # POST /v1/systemone, which forwards requests natively (point the + # TypeSafe SDK at the gateway root). A self-hosted Kev + # server speaks the same API without authentication: set base_url + # (e.g. "http://localhost:8009") and omit api_key. Name it "kev" to see + # that name in logs and usage; no separate provider type is needed. # Jev is priced per input token and is not in the upstream model catalog; # declare its pricing here to have the gateway cost System One requests. # models: diff --git a/docs/openapi.json b/docs/openapi.json index f90416500..efa2de15f 100644 --- a/docs/openapi.json +++ b/docs/openapi.json @@ -11193,6 +11193,100 @@ } } }, + "/v1/systemone": { + "post": { + "description": "Available when a jev or openrouter provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated.", + "tags": [ + "systemone" + ], + "summary": "Evaluate a System One decision request (Jev / Kev)", + "requestBody": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + }, + "description": "System One request: model, state, and questions", + "required": true + }, + "responses": { + "200": { + "description": "System One answers, in the provider's shape", + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + } + }, + "400": { + "description": "Bad Request", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "401": { + "description": "Unauthorized", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "404": { + "description": "Not Found", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "429": { + "description": "Too Many Requests", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "502": { + "description": "Bad Gateway", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ], + "x-mint": { + "metadata": { + "sidebarTitle": "/v1/systemone", + "title": "Evaluate a System One decision request (Jev / Kev)", + "description": "GoModel API reference for POST /v1/systemone: Evaluate a System One decision request (Jev / Kev)." + } + } + } + }, "/v1/usage": { "get": { "description": "Returns recorded usage, budget statuses, and rate limit statuses for the caller's effective user path (the path bound to the managed API key, or the user-path header for master-key callers).", diff --git a/docs/providers/jev.mdx b/docs/providers/jev.mdx index c2c2b16c0..ca23eabce 100644 --- a/docs/providers/jev.mdx +++ b/docs/providers/jev.mdx @@ -2,7 +2,7 @@ title: "Jev / Kev (TypeSafe System One)" sidebarTitle: "Jev / Kev" description: "Route TypeSafe System One decision requests through GoModel, to the hosted Jev API or a self-hosted Kev server." -icon: "scale-balanced" +icon: "scale" keywords: ["Jev", "Kev", "TypeSafe", "System One", "decision model", "classification", "noul", "choice", "score", "self-hosted"] --- @@ -22,10 +22,11 @@ There are three question types: | `score` | Rate against ordered levels | `score`, plus `legend`, `probabilities` and `confidence` | The API is not OpenAI-compatible, and its answers have no chat equivalent, so -GoModel does not translate it: System One requests go through -[passthrough](/features/passthrough-api) at `/p/jev/...`, which is enabled by -default for this provider. Chat, `/responses`, and `/v1/embeddings` return -`invalid_request_error` for `jev` models. +GoModel forwards it natively instead of translating it: `POST /v1/systemone` +is available as soon as a `jev` or `openrouter` provider is configured, and +[passthrough](/features/passthrough-api) at `/p/jev/...` reaches every other +upstream route. Chat, `/responses`, and `/v1/embeddings` return +`invalid_request_error` for `jev` models, pointing at `/v1/systemone`. ## Configure @@ -49,13 +50,15 @@ GOMODEL_MASTER_KEY=change-me use; a trailing `/v1` is accepted and trimmed, so both spellings address the same server. To run the hosted API and a local Kev side by side, register the second under a suffixed name: `JEV_KEV_BASE_URL=...` creates provider - `jev-kev`, reached at `/p/jev-kev/...`. + `jev-kev`, reached at `/p/jev-kev/...`. In `config.yaml`, any name works, + such as `kev: {type: jev, base_url: ...}`, and that name is what logs, + usage, and model prefixes show. ## Verify ```bash -curl -s http://localhost:8080/p/jev/v1/systemone \ +curl -s http://localhost:8080/v1/systemone \ -H "Authorization: Bearer change-me" \ -H "Content-Type: application/json" \ -d '{ @@ -88,19 +91,21 @@ curl -s http://localhost:8080/p/jev/v1/systemone \ } ``` -The `/v1` segment is optional: `/p/jev/systemone` is the same route. Use -`kev-latest` as the model on a Kev server; it also answers to `jev-latest`. +Use `kev-latest` as the model on a Kev server; it also answers to +`jev-latest`. The same request works at `/p/jev/v1/systemone`, but that +passthrough route skips virtual models and guardrails; see +[the native endpoint](#the-native-endpoint). ## Using the TypeSafe SDKs -The SDKs send `POST {base_url}/v1/systemone`, so point them at the provider's -passthrough root and authenticate with your GoModel key: +The SDKs send `POST {base_url}/v1/systemone`, so point them at the gateway +itself and authenticate with your GoModel key: ```python Python from typesafe_sdk import Noul, TypeSafeClient -client = TypeSafeClient(api_key="change-me", base_url="http://localhost:8080/p/jev") +client = TypeSafeClient(api_key="change-me", base_url="http://localhost:8080") response = client.system_one( state="I was charged twice. Please fix this ASAP.", questions={"billing": Noul(instructions="Is this ticket about billing?")}, @@ -111,7 +116,7 @@ print(response.nouls["billing"].noul) ```typescript JavaScript import { TypeSafeClient, noul } from "@typesafe-ai/sdk"; -const client = new TypeSafeClient({ apiKey: "change-me", baseURL: "http://localhost:8080/p/jev" }); +const client = new TypeSafeClient({ apiKey: "change-me", baseURL: "http://localhost:8080" }); const result = await client.systemOne({ state: "I was charged twice. Please fix this ASAP.", questions: { billing: noul({ instructions: "Is this ticket about billing?" }) }, @@ -120,14 +125,79 @@ console.log(result.answers.billing.noul); ``` -The same works with `TYPESAFE_BASE_URL=http://localhost:8080/p/jev` and -`TYPESAFE_API_KEY=change-me` in the environment. +The same works with `TYPESAFE_BASE_URL=http://localhost:8080` and +`TYPESAFE_API_KEY=change-me` in the environment. With several System One +providers, name the model with its provider (`kev/kev-latest`) or a +[virtual model](/features/virtual-models). +The SDKs' model listing expects TypeSafe's shape, while the gateway's +`/v1/models` is OpenAI-shaped; list upstream models at `/p/jev/v1/models`. + +## The native endpoint + +`POST /v1/systemone` is a gateway endpoint, not a raw proxy. For each request +GoModel: + +1. Resolves `model` like any other endpoint: a bare name, a provider-qualified + name (`jev/jev-latest`), or a virtual model, then applies the caller's + model allowlist, rate limits, and budgets. +2. Runs the workflow's prompt [guardrails](/advanced/guardrails) over `state`. +3. Forwards the body with only `model` (to the resolved name) and `state` (if + a guardrail edited it) changed. Questions, criteria, and every other field + reach the provider byte for byte. +4. Relays the answer unchanged, and records it in the audit log (as a + **System One** request) and in usage. + +The endpoint never translates. A model that cannot answer System One fails +with `400 invalid_request_error` explaining why, and the gateway logs a +warning: one on a provider without the API, or one the catalog lists as a +chat, embedding, or other generation model, such as a virtual model pointing +at a chat model. Without a `jev` or `openrouter` provider, the route answers +`404`. + +Response caching and failover do not apply to this endpoint yet. + +### Through OpenRouter + +[OpenRouter serves Jev natively](https://openrouter.ai/docs/guides/community/jev) +at the same path, so an OpenRouter key alone is enough: + +```bash +OPENROUTER_API_KEY=sk-or-... +``` + +OpenRouter's decision models appear in `GET /v1/models` as utility models, +priced from OpenRouter's listing: `openrouter/typesafe/jev-1.13`, +`openrouter/~typesafe/jev-latest` (tracks the newest Jev), and other decision +models such as Kev 4B (`openrouter/jaredpalmer/kev-4b`). Name them that way in +`model`. OpenRouter accepts `jev-latest` itself, but GoModel routes on its +catalog IDs, so to keep the TypeSafe SDK's plain `jev-latest` working, add a +[virtual model](/features/virtual-models) `jev-latest` that targets +`openrouter/~typesafe/jev-latest`. Chat models on the same provider are +rejected, since they have no System One API. + +The answer carries OpenRouter's `id`, `provider`, and `usage.cost`, and GoModel +records that reported cost with the request's usage. With a `jev` provider +configured as well, one virtual model can front a local Kev server and +OpenRouter's Jev together. + +### Guardrails + +Guardrails see `state` as a single user message: a string state as its text, +any other JSON value as its encoded JSON (which must still be valid JSON after +an edit). This is what anonymizing or blocking guardrails need. The questions +are your application's fixed schema and are not exposed. + +Guardrail edits a decision request has no place for, such as a system prompt +injected by a workflow that also covers chat models, are dropped with a +warning in the logs rather than failing the request. A guardrail that would +answer the request itself blocks it instead, since System One callers expect +typed answers, not text. ## Native routes | Route | What it does | | --- | --- | -| `POST /p/jev/v1/systemone` | Evaluate a state against a map of questions | +| `POST /p/jev/v1/systemone` | Evaluate a state against a map of questions, without virtual models or guardrails | | `GET /p/jev/v1/models` | The names the `model` field accepts, in the upstream's own shape | | `POST /p/jev/v1/systemone/permute` | Kev only: run one Choice question with several option orders | | `POST /p/jev/v1/systemone/separate` | Kev only: run each question in its own forward pass | @@ -145,9 +215,14 @@ IDs such as `jev-1.13.0` are accepted by the `model` field whether or not they are listed. The models are categorized as utility models with no generation mode, since there is no OpenAI endpoint to route them to. -Every System One request names its model, so the passthrough surface applies -the caller's [model allowlist](/features/users) to it like any other -request. +Every System One request names its model, so both `/v1/systemone` and the +passthrough surface apply the caller's [model allowlist](/features/users) to +it like any other request. + +`/v1/systemone` routes only to models in the catalog. To pin a version the +upstream does not list, such as `jev-1.13.0`, declare it under the provider's +`models` (as in the pricing example below) and set +`CONFIGURED_PROVIDER_MODELS_MODE=merge`; passthrough accepts any name. The response's `usage.input_tokens` and `usage.output_tokens` are recorded, so System One calls appear in the usage API and dashboard under the model that diff --git a/docs/providers/overview.mdx b/docs/providers/overview.mdx index 055ac06c8..d8ea5d3bb 100644 --- a/docs/providers/overview.mdx +++ b/docs/providers/overview.mdx @@ -211,10 +211,14 @@ support, not every individual model capability exposed by an upstream provider. through passthrough. See [audio.cpp](/providers/audiocpp). - **Jev / Kev** — TypeSafe's System One API is a decision API (state plus typed questions in, calibrated probabilities out) with no OpenAI-compatible - surface, so it is reached only through passthrough at - `POST /p/jev/v1/systemone`. `JEV_API_KEY` configures the hosted API; for a - self-hosted Kev server, which speaks the same API without authentication, - set `JEV_BASE_URL` and leave the key unset. See [Jev](/providers/jev). + surface, so GoModel forwards it natively, untranslated, at + `POST /v1/systemone` (with virtual models, guardrails, audit, and usage) and + through passthrough at `/p/jev/...`. The endpoint is available once a `jev` + or `openrouter` provider is configured; OpenRouter serves Jev natively, and + its decision models are listed as utility models. + `JEV_API_KEY` configures the hosted API; for a self-hosted Kev server, which + speaks the same API without authentication, set `JEV_BASE_URL` and leave the + key unset. See [Jev](/providers/jev). - **llama.cpp / LM Studio** — `LLAMACPP_BASE_URL` is required (llama-server's default port collides with GoModel's own 8080, so there is no default); `LLAMACPP_API_KEY` is optional. Do not register these servers as `ollama`, diff --git a/internal/core/endpoint_operations.go b/internal/core/endpoint_operations.go index 6ab6df986..8bb2b1395 100644 --- a/internal/core/endpoint_operations.go +++ b/internal/core/endpoint_operations.go @@ -29,6 +29,7 @@ var operationPaths = map[Operation]OperationPaths{ "/v1/realtime/translations", "/v1/realtime/translations/calls", "/v1/realtime/translations/client_secrets", }}, OperationMCP: {Prefixes: []string{"/mcp"}}, + OperationSystemOne: {Exact: []string{"/v1/systemone"}}, OperationProviderPassthrough: {Prefixes: []string{"/p"}}, } diff --git a/internal/core/endpoints.go b/internal/core/endpoints.go index b75404322..386d3c726 100644 --- a/internal/core/endpoints.go +++ b/internal/core/endpoints.go @@ -33,6 +33,7 @@ const ( OperationRealtime Operation = "realtime" OperationProviderPassthrough Operation = "provider_passthrough" OperationMCP Operation = "mcp" + OperationSystemOne Operation = "systemone" ) // EndpointDescriptor centralizes the transport-facing classification of model and provider routes. @@ -169,6 +170,16 @@ func describeEndpointPath(path string) EndpointDescriptor { Dialect: "openai_compat", Operation: OperationImageEdits, } + case path == "/v1/systemone": + // TypeSafe's System One decision API (Jev, Kev). It has no canonical + // translation: the body is forwarded to a System One provider + // unchanged, apart from the routed model and guardrail edits to state. + return EndpointDescriptor{ + ModelInteraction: true, + IngressManaged: true, + Dialect: "systemone", + Operation: OperationSystemOne, + } case isRealtimePath(path): // The realtime endpoints relay the provider's schema verbatim: /v1/realtime // upgrades to a websocket, /v1/realtime/calls exchanges WebRTC SDP, and @@ -238,7 +249,7 @@ func bodyModeForEndpoint(method, path string, operation Operation) BodyMode { return BodyModeMultipart } return BodyModeNone - case OperationAudioSpeech, OperationImageGenerations: + case OperationAudioSpeech, OperationImageGenerations, OperationSystemOne: return BodyModeJSON case OperationAudioTranscriptions, OperationAudioTranslations, OperationImageEdits: return BodyModeMultipart diff --git a/internal/core/endpoints_test.go b/internal/core/endpoints_test.go index 6c48356b2..b881965de 100644 --- a/internal/core/endpoints_test.go +++ b/internal/core/endpoints_test.go @@ -42,6 +42,8 @@ func TestDescribeEndpointPath(t *testing.T) { {path: "/v1/realtime/translations/client_secrets", managed: false, dialect: "openai_compat", operation: OperationRealtime, bodyMode: BodyModeNone, interaction: true}, {path: "/mcp", managed: false, dialect: "mcp", operation: OperationMCP, bodyMode: BodyModeNone, interaction: true}, {path: "/mcp/linear", managed: false, dialect: "mcp", operation: OperationMCP, bodyMode: BodyModeNone, interaction: true}, + {path: "/v1/systemone", managed: true, dialect: "systemone", operation: OperationSystemOne, bodyMode: BodyModeJSON, interaction: true}, + {path: "/v1/systemone/permute", managed: false, dialect: "", operation: "", bodyMode: BodyModeNone, interaction: false}, {path: "/p/openai/responses", managed: true, dialect: "provider_passthrough", operation: OperationProviderPassthrough, bodyMode: BodyModeOpaque, interaction: true}, {path: "/v1/models", managed: false, dialect: "", operation: "", bodyMode: BodyModeNone, interaction: false}, } diff --git a/internal/core/systemone.go b/internal/core/systemone.go new file mode 100644 index 000000000..8ee6d5bd1 --- /dev/null +++ b/internal/core/systemone.go @@ -0,0 +1,13 @@ +package core + +import "github.com/goccy/go-json" + +// SystemOneRequest is the part of a System One decision request the gateway +// reads: the model it routes on and the state guardrails inspect. The rest of +// the body (the questions and their criteria) reaches the provider unchanged. +type SystemOneRequest struct { + Model string `json:"model"` + // State is the text or record the questions are asked about, as sent: a + // JSON string in TypeSafe's examples, but any JSON value is forwarded. + State json.RawMessage `json:"state,omitempty"` +} diff --git a/internal/core/workflow.go b/internal/core/workflow.go index 4d9a25f37..a6cc38ac5 100644 --- a/internal/core/workflow.go +++ b/internal/core/workflow.go @@ -57,6 +57,12 @@ func CapabilitiesForEndpoint(desc EndpointDescriptor) CapabilitySet { return CapabilitySet{ SemanticExtraction: true, } + case OperationSystemOne: + return CapabilitySet{ + AliasResolution: true, + Guardrails: true, + UsageTracking: true, + } case OperationProviderPassthrough: return CapabilitySet{ SemanticExtraction: true, diff --git a/internal/gateway/interfaces.go b/internal/gateway/interfaces.go index 60fa8783e..cc4cf8da6 100644 --- a/internal/gateway/interfaces.go +++ b/internal/gateway/interfaces.go @@ -53,6 +53,12 @@ type TranslatedRequestPatcher interface { PatchResponsesRequest(ctx context.Context, req *core.ResponsesRequest) (*core.ResponsesRequest, error) } +// SystemOneRequestPatcher is an optional TranslatedRequestPatcher capability: +// it runs the prompt phase over a System One decision request's state. +type SystemOneRequestPatcher interface { + PatchSystemOneRequest(ctx context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) +} + // PromptContentEditor is an optional TranslatedRequestPatcher capability: it // reports whether the prompt phase may rewrite the content of this request. // Only a rewriting prompt phase (anonymization, redaction) needs the replayed diff --git a/internal/guardrails/integration_test.go b/internal/guardrails/integration_test.go index 03f1b096d..fdb696698 100644 --- a/internal/guardrails/integration_test.go +++ b/internal/guardrails/integration_test.go @@ -11,6 +11,7 @@ import ( "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/plugins" "github.com/enterpilot/gomodel/pluginapi" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -329,3 +330,31 @@ func TestWorkflowBatchPreparerRejectsRespondDecisions(t *testing.T) { require.ErrorAs(t, err, &gatewayErr) require.Equal(t, http.StatusBadRequest, gatewayErr.HTTPStatusCode()) } + +// A System One request exposes only its state: an anonymizing guardrail +// rewrites it, while a system prompt a workflow injects for chat models has +// no place in a decision request and is dropped rather than failing it. +func TestWorkflowRequestPatcherSystemOneGuardsState(t *testing.T) { + store := newTestStore( + systemPromptDefinition("safety", "be safe"), + Definition{Name: "privacy", Type: "llm_based_altering", Config: rawConfig(t, map[string]any{"model": "openai/gpt-4o-mini", "roles": []string{"user"}})}, + ) + service := newService(t, store, chatFunc(func(_ context.Context, req *core.ChatRequest) (*core.ChatResponse, error) { + text := core.ExtractTextContent(req.Messages[1].Content) + inner := strings.TrimSuffix(strings.TrimPrefix(text, "\n"), "\n") + return replyChat(strings.ReplaceAll(inner, "John", "[PERSON]"))(context.Background(), req) + })) + patcher := NewWorkflowRequestPatcher(staticChains{chainsFor(t, service, + StepReference{Ref: "privacy", Step: 10}, + StepReference{Ref: "safety", Step: 20}, + )}) + + req := &core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(`"John was charged twice"`)} + ctx, state := plugins.WithRequestState(context.Background()) + got, err := patcher.PatchSystemOneRequest(ctx, req) + require.NoError(t, err) + assert.JSONEq(t, `"[PERSON] was charged twice"`, string(got.State)) + assert.Equal(t, "kev-latest", got.Model) + assert.JSONEq(t, `"John was charged twice"`, string(req.State), "the original request must not change") + require.Len(t, state.Snapshot(), 2) +} diff --git a/internal/guardrails/workflow_executor.go b/internal/guardrails/workflow_executor.go index 886e99fda..82783ec31 100644 --- a/internal/guardrails/workflow_executor.go +++ b/internal/guardrails/workflow_executor.go @@ -2,6 +2,7 @@ package guardrails import ( "context" + "log/slog" "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/plugins" @@ -36,6 +37,25 @@ func (p *WorkflowRequestPatcher) PatchResponsesRequest(ctx context.Context, req return processGuardedResponses(ctx, p.chain(ctx), req) } +// PatchSystemOneRequest runs the prompt chain over a System One request's +// state. Edits a decision request has no place for, such as an injected +// system prompt, are dropped with a warning rather than failing the request: +// a guardrail scoped to every model is not wrong for System One models, it +// only has nothing to change there. +func (p *WorkflowRequestPatcher) PatchSystemOneRequest(ctx context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) { + if req == nil { + return nil, nil + } + return processGuarded(ctx, p.chain(ctx), req, "System One", exchange.FromSystemOneRequest, applySystemOneEdits) +} + +func applySystemOneEdits(req *core.SystemOneRequest, prompt *pluginapi.Prompt) (*core.SystemOneRequest, error) { + if dropped := exchange.SystemOneUncarriedEdits(prompt); len(dropped) > 0 { + slog.Warn("guardrail edits a System One request cannot carry were dropped; only the state is guarded", "model", req.Model, "dropped", dropped) + } + return exchange.ApplyToSystemOneRequest(req, prompt) +} + // EditsPromptContent reports whether the request's prompt chain holds an // instance that edits content, such as an anonymizing guardrail. A chained // Responses request only needs its stored history expanded into the input diff --git a/internal/plugins/exchange/systemone_request.go b/internal/plugins/exchange/systemone_request.go new file mode 100644 index 000000000..73ad3ad33 --- /dev/null +++ b/internal/plugins/exchange/systemone_request.go @@ -0,0 +1,107 @@ +package exchange + +import ( + "bytes" + "fmt" + "sort" + + "github.com/goccy/go-json" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/pluginapi" +) + +// SystemOneStateMessageID is the ID of the user message built from a System +// One request's state. +const SystemOneStateMessageID = "state" + +// FromSystemOneRequest builds the unified prompt for a System One decision +// request. The state is the only content a caller supplies per request, so it +// becomes the prompt's single user message: a string state as its text, any +// other JSON value as its encoded JSON. The questions are the application's +// fixed schema and are not exposed. +func FromSystemOneRequest(req *core.SystemOneRequest) (*pluginapi.Prompt, error) { + if req == nil { + return nil, fmt.Errorf("exchange: nil System One request") + } + raw, err := json.Marshal(req) + if err != nil { + return nil, fmt.Errorf("exchange: encode System One request: %w", err) + } + p := &pluginapi.Prompt{ + Raw: raw, + Messages: []pluginapi.Message{pluginapi.TextMessage(pluginapi.RoleUser, systemOneStateText(req.State))}, + Params: pluginapi.Params{Model: req.Model}, + } + p.Messages[0].ID = SystemOneStateMessageID + p.Reset() + return p, nil +} + +func systemOneStateText(state json.RawMessage) string { + var text string + if trimmed := bytes.TrimSpace(state); len(trimmed) > 0 && trimmed[0] == '"' && json.Unmarshal(trimmed, &text) == nil { + return text + } + return string(state) +} + +// ApplyToSystemOneRequest returns a copy of original with the state +// message's edits applied. A string state stays a string; a state sent as +// another JSON value must still be valid JSON after the edit. Removing the +// state is an error, since the request would have nothing to decide on. +// Edits System One has no place for, such as inserted messages or parameter +// changes, are not applied; SystemOneUncarriedEdits lists them. +func ApplyToSystemOneRequest(original *core.SystemOneRequest, p *pluginapi.Prompt) (*core.SystemOneRequest, error) { + if original == nil || p == nil { + return nil, fmt.Errorf("exchange: nil System One request or prompt") + } + result := *original + switch p.Changes().Messages[SystemOneStateMessageID] { + case "": + return &result, nil + case pluginapi.ChangeRemoved: + return nil, fmt.Errorf("exchange: the System One state was removed") + } + msg := p.Message(SystemOneStateMessageID) + if msg == nil { + return nil, fmt.Errorf("exchange: the System One state was removed") + } + text := msg.Text() + // A missing state becomes a string; null, numbers, and records keep + // their JSON type (json.Unmarshal would accept null into a string). + state := bytes.TrimSpace(original.State) + if len(state) == 0 || state[0] == '"' { + encoded, err := json.Marshal(text) + if err != nil { + return nil, fmt.Errorf("exchange: encode System One state: %w", err) + } + result.State = encoded + return &result, nil + } + if !json.Valid([]byte(text)) { + return nil, fmt.Errorf("exchange: the edited System One state is no longer valid JSON") + } + result.State = json.RawMessage(text) + return &result, nil +} + +// SystemOneUncarriedEdits describes the prompt edits ApplyToSystemOneRequest +// does not apply, in a stable order, or nil when every edit was carried. +func SystemOneUncarriedEdits(p *pluginapi.Prompt) []string { + if p == nil { + return nil + } + changes := p.Changes() + var uncarried []string + for id, kind := range changes.Messages { + if id != SystemOneStateMessageID { + uncarried = append(uncarried, fmt.Sprintf("%s message %q", kind, id)) + } + } + for name := range changes.Params { + uncarried = append(uncarried, fmt.Sprintf("parameter %q", name)) + } + sort.Strings(uncarried) + return uncarried +} diff --git a/internal/plugins/exchange/systemone_request_test.go b/internal/plugins/exchange/systemone_request_test.go new file mode 100644 index 000000000..62d5ca08b --- /dev/null +++ b/internal/plugins/exchange/systemone_request_test.go @@ -0,0 +1,127 @@ +package exchange + +import ( + "testing" + + "github.com/goccy/go-json" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/pluginapi" +) + +func TestFromSystemOneRequestExposesStateAsUserMessage(t *testing.T) { + tests := []struct { + name string + state string + want string + }{ + {name: "string state", state: `"charged twice"`, want: "charged twice"}, + {name: "record state", state: `{"ticket":"charged twice"}`, want: `{"ticket":"charged twice"}`}, + {name: "null state", state: `null`, want: "null"}, + {name: "no state", state: ``, want: ""}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + p, err := FromSystemOneRequest(&core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(tt.state)}) + require.NoError(t, err) + require.Len(t, p.Messages, 1) + + assert.Equal(t, SystemOneStateMessageID, p.Messages[0].ID) + assert.Equal(t, pluginapi.RoleUser, p.Messages[0].Role) + assert.Equal(t, tt.want, p.Messages[0].Text()) + assert.Equal(t, "kev-latest", p.Params.Model) + assert.False(t, p.Changes().Dirty) + }) + } +} + +func TestApplyToSystemOneRequest(t *testing.T) { + tests := []struct { + name string + state string + edit func(p *pluginapi.Prompt) error + wantState string + wantErr string + }{ + { + name: "untouched", + state: `"John"`, + edit: func(*pluginapi.Prompt) error { return nil }, + wantState: `"John"`, + }, + { + name: "string state stays a string", + state: `"John \"J\" Doe"`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `[PERSON] "J"`) }, + wantState: `"[PERSON] \"J\""`, + }, + { + name: "record state stays a record", + state: `{"name":"John"}`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `{"name":"[PERSON]"}`) }, + wantState: `{"name":"[PERSON]"}`, + }, + { + name: "null state keeps its JSON type", + state: `null`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `{"redacted":true}`) }, + wantState: `{"redacted":true}`, + }, + { + name: "record state must stay valid JSON", + state: `{"name":"John"}`, + edit: func(p *pluginapi.Prompt) error { return p.SetText(SystemOneStateMessageID, 0, `name: [PERSON]`) }, + wantErr: "no longer valid JSON", + }, + { + name: "state cannot be removed", + state: `"John"`, + edit: func(p *pluginapi.Prompt) error { return p.Remove(SystemOneStateMessageID) }, + wantErr: "state was removed", + }, + { + name: "uncarried edits are not applied", + state: `"John"`, + edit: func(p *pluginapi.Prompt) error { + p.Insert(0, pluginapi.TextMessage(pluginapi.RoleSystem, "be safe")) + p.SetParam("temperature", 0.1) + return nil + }, + wantState: `"John"`, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + original := &core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(tt.state)} + p, err := FromSystemOneRequest(original) + require.NoError(t, err) + require.NoError(t, tt.edit(p)) + + got, err := ApplyToSystemOneRequest(original, p) + if tt.wantErr != "" { + require.Error(t, err) + assert.Contains(t, err.Error(), tt.wantErr) + return + } + require.NoError(t, err) + assert.JSONEq(t, tt.wantState, string(got.State)) + assert.Equal(t, "kev-latest", got.Model) + assert.JSONEq(t, tt.state, string(original.State), "the original request must not change") + }) + } +} + +func TestSystemOneUncarriedEdits(t *testing.T) { + p, err := FromSystemOneRequest(&core.SystemOneRequest{Model: "kev-latest", State: json.RawMessage(`"John"`)}) + require.NoError(t, err) + assert.Nil(t, SystemOneUncarriedEdits(p)) + + require.NoError(t, p.SetText(SystemOneStateMessageID, 0, "[PERSON]")) + assert.Nil(t, SystemOneUncarriedEdits(p), "a state edit is carried") + + id := p.Insert(0, pluginapi.TextMessage(pluginapi.RoleSystem, "be safe")) + p.SetParam("temperature", 0.1) + assert.Equal(t, []string{`inserted message "` + id + `"`, `parameter "temperature"`}, SystemOneUncarriedEdits(p)) +} diff --git a/internal/providers/jev/jev.go b/internal/providers/jev/jev.go index 32b3ea576..93dd4acef 100644 --- a/internal/providers/jev/jev.go +++ b/internal/providers/jev/jev.go @@ -3,8 +3,8 @@ // API. System One is a decision API rather than a text-generation one: a // request carries a state and a map of typed questions (noul, choice, score) // and the answer is a calibrated probability per question. It has no -// OpenAI-compatible surface, so the gateway reaches it through native -// passthrough at /p/jev/systemone. +// OpenAI-compatible surface, so the gateway forwards it natively, at +// POST /v1/systemone or through passthrough at /p/jev/systemone. package jev import ( @@ -110,12 +110,12 @@ func (p *Provider) Embeddings(_ context.Context, _ *core.EmbeddingRequest) (*cor } func unsupported(surface string) error { - return core.NewInvalidRequestError("jev does not support "+surface+"; send System One requests to /p/jev/systemone", nil) + return core.NewInvalidRequestError("jev does not support "+surface+"; it answers System One decision requests, which GoModel does not translate: send them to POST /v1/systemone", nil) } -// Passthrough forwards a System One request as the client wrote it. It is the -// only way to reach the evaluation endpoint, since the request and answer -// shapes have no OpenAI equivalent. +// Passthrough forwards a System One request as the client wrote it. Both +// /v1/systemone and /p/jev/... reach the evaluation endpoint through it, +// since the request and answer shapes have no OpenAI equivalent. func (p *Provider) Passthrough(ctx context.Context, req *core.PassthroughRequest) (*core.PassthroughResponse, error) { if req == nil { return nil, core.NewInvalidRequestError("passthrough request is required", nil) diff --git a/internal/providers/jev/jev_test.go b/internal/providers/jev/jev_test.go index 43c603028..9475fcaf8 100644 --- a/internal/providers/jev/jev_test.go +++ b/internal/providers/jev/jev_test.go @@ -45,7 +45,7 @@ func TestUnsupportedCapabilities_ReturnInvalidRequestErrors(t *testing.T) { _, err := provider.ChatCompletion(context.Background(), &core.ChatRequest{Model: "jev-latest"}) providertest.AssertUnsupported(t, err) - assert.Contains(t, err.Error(), "/p/jev/systemone") + assert.Contains(t, err.Error(), "/v1/systemone") _, err = provider.StreamChatCompletion(context.Background(), &core.ChatRequest{Model: "jev-latest"}) providertest.AssertUnsupported(t, err) _, err = provider.Responses(context.Background(), &core.ResponsesRequest{Model: "jev-latest"}) diff --git a/internal/providers/openrouter/openrouter.go b/internal/providers/openrouter/openrouter.go index 70ee13f6d..1222473a7 100644 --- a/internal/providers/openrouter/openrouter.go +++ b/internal/providers/openrouter/openrouter.go @@ -141,8 +141,9 @@ func (p *Provider) ListModels(ctx context.Context) (*core.ModelsResponse, error) // servableOpenRouterModalities are output modalities the gateway can reach on // OpenRouter: text and image generation flow through chat completions, -// embeddings through /embeddings, and speech/transcription through the -// /audio endpoints. A model listing none of these (rerank-only, video) has no +// embeddings through /embeddings, speech/transcription through the /audio +// endpoints, and decisions (System One models such as Jev) through +// /v1/systemone. A model listing none of these (rerank-only, video) has no // working endpoint here. var servableOpenRouterModalities = map[string]struct{}{ "text": {}, @@ -150,6 +151,7 @@ var servableOpenRouterModalities = map[string]struct{}{ "embeddings": {}, "speech": {}, "transcription": {}, + "decisions": {}, } func openrouterServable(m openrouterModel) bool { @@ -171,6 +173,7 @@ func openrouterServable(m openrouterModel) bool { // ID inference. func openrouterMetadata(m openrouterModel) *core.ModelMetadata { modes := make([]string, 0, 2) + decisions := false // "rerank" is deliberately not mapped: the gateway has no rerank surface // on OpenRouter, and the rerank mode would sort the model into the // Embeddings category despite being unreachable here. @@ -186,6 +189,8 @@ func openrouterMetadata(m openrouterModel) *core.ModelMetadata { modes = append(modes, "audio_speech") case "transcription": modes = append(modes, "audio_transcription") + case "decisions": + decisions = true } } pricing := openrouterPricing(m) @@ -196,9 +201,15 @@ func openrouterMetadata(m openrouterModel) *core.ModelMetadata { Capabilities: capabilities, Pricing: pricing, } - if len(modes) > 0 { + switch { + case len(modes) > 0: meta.Modes = modes meta.Categories = core.CategoriesForModes(modes) + case decisions: + // A decision model answers System One requests only. It has no + // generation mode to claim, so like the jev provider's models it is + // a utility model that no OpenAI endpoint routes to. + meta.Categories = []core.ModelCategory{core.CategoryUtility} } if m.ContextLength > 0 { contextWindow := m.ContextLength @@ -207,7 +218,7 @@ func openrouterMetadata(m openrouterModel) *core.ModelMetadata { if m.TopProvider.MaxCompletionTokens > 0 { meta.MaxOutputTokens = new(m.TopProvider.MaxCompletionTokens) } - if len(modes) == 0 && meta.ContextWindow == nil && meta.MaxOutputTokens == nil && pricing == nil && + if len(meta.Categories) == 0 && meta.ContextWindow == nil && meta.MaxOutputTokens == nil && pricing == nil && capabilities == nil && meta.DisplayName == "" && meta.Description == "" { return nil } diff --git a/internal/providers/openrouter/openrouter_test.go b/internal/providers/openrouter/openrouter_test.go index 0a50da65f..39bffeb7d 100644 --- a/internal/providers/openrouter/openrouter_test.go +++ b/internal/providers/openrouter/openrouter_test.go @@ -59,6 +59,11 @@ func TestListModels_StampsArchitectureModalities(t *testing.T) { "architecture":{"input_modalities":["text"],"output_modalities":["rerank"]}}, {"id":"acme/video-only","created":1721260800, "architecture":{"input_modalities":["text"],"output_modalities":["video"]}}, + {"id":"typesafe/jev-1.13","name":"TypeSafe: Jev 1.13","created":1789689684,"context_length":32000, + "architecture":{"input_modalities":["text"],"output_modalities":["decisions"]}, + "pricing":{"prompt":"0.000000042","completion":"0"}}, + {"id":"~typesafe/jev-latest","created":1789689684, + "architecture":{"input_modalities":["text"],"output_modalities":["decisions"]}}, {"id":"mystery/no-architecture","created":1721260800} ]}`) provider := newTestProvider(server.URL, server.Client()) @@ -72,7 +77,7 @@ func TestListModels_StampsArchitectureModalities(t *testing.T) { // embedding models would never enter the catalog. assert.Equal(t, "all", req.Query.Get("output_modalities")) - require.Len(t, resp.Data, 6) + require.Len(t, resp.Data, 8) byID := modelsByID(resp) chat := byID["openai/gpt-4o-mini"] @@ -114,6 +119,17 @@ func TestListModels_StampsArchitectureModalities(t *testing.T) { require.NotNil(t, stt.Metadata) assert.Equal(t, []string{"audio_transcription"}, stt.Metadata.Modes) + // Decision models serve /v1/systemone only: listed, but as utility models + // with no mode that would route an OpenAI request to them. + for _, id := range []string{"typesafe/jev-1.13", "~typesafe/jev-latest"} { + decision := byID[id] + require.NotNil(t, decision.Metadata, id) + assert.Empty(t, decision.Metadata.Modes, id) + assert.Equal(t, []core.ModelCategory{core.CategoryUtility}, decision.Metadata.Categories, id) + } + require.NotNil(t, byID["typesafe/jev-1.13"].Metadata.Pricing) + assert.InDelta(t, 0.042, *byID["typesafe/jev-1.13"].Metadata.Pricing.InputPerMtok, 1e-9) + assert.NotContains(t, byID, "cohere/rerank-only") assert.NotContains(t, byID, "acme/video-only") diff --git a/internal/providers/registry_normalization_test.go b/internal/providers/registry_normalization_test.go index 2ce0e6270..0af8eb32b 100644 --- a/internal/providers/registry_normalization_test.go +++ b/internal/providers/registry_normalization_test.go @@ -6,6 +6,7 @@ import ( "testing" "github.com/enterpilot/gomodel/internal/core" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -340,3 +341,27 @@ func TestRecordAvailabilityCheckKeepsFailureMarker(t *testing.T) { }) } } + +// LookupModel describes one catalog model by any selector the router resolves, +// including OpenRouter's "~"-prefixed alias IDs. +func TestRouterLookupModel(t *testing.T) { + registry := newTestRegistryWithModels(registryModelEntry{ + provider: &mockProvider{name: "openrouter"}, + providerName: "openrouter", + providerType: "openrouter", + modelID: "~typesafe/jev-latest", + }) + router, err := NewRouter(registry) + require.NoError(t, err) + + for _, selector := range []string{"openrouter/~typesafe/jev-latest", "~typesafe/jev-latest"} { + model, ok := router.LookupModel(selector) + require.True(t, ok, selector) + assert.Equal(t, "~typesafe/jev-latest", model.ID, selector) + } + _, ok := router.LookupModel("openrouter/unknown") + assert.False(t, ok) + + _, ok = (&Router{}).LookupModel("openrouter/~typesafe/jev-latest") + assert.False(t, ok, "a lookup without single-model access describes nothing") +} diff --git a/internal/providers/router_models.go b/internal/providers/router_models.go index 4f943569e..b49d87f70 100644 --- a/internal/providers/router_models.go +++ b/internal/providers/router_models.go @@ -182,3 +182,17 @@ func (r *Router) NativeResponseProviderTypes() []string { return ok }) } + +// LookupModel returns a copy of the catalog entry for a model selector, or +// false when the model is unknown or the lookup cannot describe one model. +func (r *Router) LookupModel(model string) (*core.Model, bool) { + if r.caps.modelInfo == nil { + return nil, false + } + info := r.caps.modelInfo.GetModel(model) + if info == nil { + return nil, false + } + cloned := info.Model + return &cloned, true +} diff --git a/internal/server/http.go b/internal/server/http.go index 560683152..f6e4dab94 100644 --- a/internal/server/http.go +++ b/internal/server/http.go @@ -496,6 +496,9 @@ func New(provider core.RoutableProvider, cfg *Config) *Server { e.POST("/v1/audio/translations", handler.AudioTranslations) e.POST("/v1/images/generations", handler.ImageGenerations) e.POST("/v1/images/edits", handler.ImageEdits) + // System One decisions (Jev / Kev). The handler answers 404 until a jev + // or openrouter provider is configured. + e.POST("/v1/systemone", handler.SystemOne) if cfg == nil || cfg.RealtimeEnabled { e.GET("/v1/realtime", handler.Realtime) e.POST("/v1/realtime/calls", handler.RealtimeCalls) diff --git a/internal/server/messages_native.go b/internal/server/messages_native.go index d460297a1..d451a9319 100644 --- a/internal/server/messages_native.go +++ b/internal/server/messages_native.go @@ -3,8 +3,8 @@ package server import ( "bytes" "context" - // encoding/json rather than goccy: rewriteMessagesModel needs the - // decoder's InputOffset to splice the model value in place. + // encoding/json rather than goccy: replaceTopLevelMember needs the + // decoder's InputOffset to splice a value in place. "encoding/json" "errors" "io" @@ -138,6 +138,17 @@ func rewriteMessagesModel(body []byte, model string) ([]byte, error) { if strings.TrimSpace(model) == "" { return body, nil } + encoded, err := json.Marshal(model) + if err != nil { + return nil, err + } + return replaceTopLevelMember(body, "model", encoded) +} + +// replaceTopLevelMember returns body with the value of its top-level key +// member replaced by value, splicing only those bytes. The body is returned +// unchanged when the member is absent or already holds value. +func replaceTopLevelMember(body []byte, key string, value []byte) ([]byte, error) { dec := json.NewDecoder(bytes.NewReader(body)) tok, err := dec.Token() if err != nil { @@ -146,45 +157,36 @@ func rewriteMessagesModel(body []byte, model string) ([]byte, error) { if delim, ok := tok.(json.Delim); !ok || delim != '{' { return nil, errors.New("request body is not a JSON object") } - // Walk every top-level member and remember the span of the last "model" + // Walk every top-level member and remember the span of the last matching // value: decoders keep the last duplicate member, so that is the one the - // resolved model came from and the one to rewrite. - var modelRaw json.RawMessage - var modelEnd int64 + // gateway read and the one to rewrite. + var found json.RawMessage + var foundEnd int64 for dec.More() { keyTok, err := dec.Token() if err != nil { return nil, err } - key, _ := keyTok.(string) + name, _ := keyTok.(string) var raw json.RawMessage if err := dec.Decode(&raw); err != nil { return nil, err } - if key != "model" { + if name != key { continue } - modelRaw = raw - modelEnd = dec.InputOffset() + found = raw + foundEnd = dec.InputOffset() } - if modelRaw == nil { + if found == nil || bytes.Equal(found, value) { return body, nil } - var current string - _ = json.Unmarshal(modelRaw, ¤t) - if current == model { - return body, nil - } - encoded, err := json.Marshal(model) - if err != nil { - return nil, err - } - // The model value is a scalar, so modelRaw holds its exact source bytes - // and modelEnd points just past them. - start := modelEnd - int64(len(modelRaw)) - rewritten := make([]byte, 0, int64(len(body))-int64(len(modelRaw))+int64(len(encoded))) + // The decoder hands back the value's exact source bytes, and foundEnd + // points just past them. + start := foundEnd - int64(len(found)) + rewritten := make([]byte, 0, int64(len(body))-int64(len(found))+int64(len(value))) rewritten = append(rewritten, body[:start]...) - rewritten = append(rewritten, encoded...) - rewritten = append(rewritten, body[modelEnd:]...) + rewritten = append(rewritten, value...) + rewritten = append(rewritten, body[foundEnd:]...) return rewritten, nil } diff --git a/internal/server/model_validation.go b/internal/server/model_validation.go index 4a2d18d60..52b54be42 100644 --- a/internal/server/model_validation.go +++ b/internal/server/model_validation.go @@ -103,7 +103,12 @@ func deriveWorkflowWithPolicy( } return workflow, nil - case core.OperationChatCompletions, core.OperationResponses, core.OperationEmbeddings: + case core.OperationChatCompletions, core.OperationResponses, core.OperationEmbeddings, core.OperationSystemOne: + if desc.Operation == core.OperationSystemOne && !systemOneAvailable(provider) { + // The handler answers 404; resolving the model first would + // report a model error for an endpoint that is not there. + return nil, nil + } workflow.Mode = core.ExecutionModeTranslated if desc.BodyMode != core.BodyModeJSON { // Responses lifecycle routes (GET/DELETE /v1/responses/{id}, diff --git a/internal/server/systemone_handler.go b/internal/server/systemone_handler.go new file mode 100644 index 000000000..b297e7fcf --- /dev/null +++ b/internal/server/systemone_handler.go @@ -0,0 +1,260 @@ +package server + +import ( + "bytes" + "fmt" + "io" + "log/slog" + "net/http" + "slices" + "strings" + + "github.com/goccy/go-json" + "github.com/labstack/echo/v5" + + "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/gateway" + "github.com/enterpilot/gomodel/internal/plugins" +) + +const ( + systemOnePath = "/v1/systemone" + systemOneEndpoint = "systemone" +) + +// systemOneProviderTypes are the provider types that serve the System One API +// natively: jev (TypeSafe's hosted Jev and self-hosted Kev servers) and +// OpenRouter, which serves Jev and Kev at the same path with the same request +// and answer shapes. Configuring either makes /v1/systemone available. +var systemOneProviderTypes = []string{"jev", "openrouter"} + +// SystemOne handles POST /v1/systemone. +// +// It accepts TypeSafe's System One decision request (a state plus a map of +// typed questions) and forwards it natively: the body reaches the provider +// unchanged apart from the routed model name and guardrail edits to the +// state. Requests are never translated to another API, so a model whose +// provider has no System One API is rejected. +// +// @Summary Evaluate a System One decision request (Jev / Kev) +// @Description Available when a jev or openrouter provider is configured. The request and answer follow TypeSafe's System One API; models on providers without that API are rejected rather than translated. +// @Tags systemone +// @Accept json +// @Produce json +// @Security BearerAuth +// @Param request body object true "System One request: model, state, and questions" +// @Success 200 {object} object "System One answers, in the provider's shape" +// @Failure 400 {object} core.OpenAIErrorEnvelope +// @Failure 401 {object} core.OpenAIErrorEnvelope +// @Failure 404 {object} core.OpenAIErrorEnvelope +// @Failure 429 {object} core.OpenAIErrorEnvelope +// @Failure 502 {object} core.OpenAIErrorEnvelope +// @Router /v1/systemone [post] +func (h *Handler) SystemOne(c *echo.Context) error { + return h.translatedInference().SystemOne(c) +} + +// SystemOne resolves, guards, and forwards one System One request. +func (s *translatedInferenceService) SystemOne(c *echo.Context) error { + if !systemOneAvailable(s.provider) { + return handleError(c, core.NewNotFoundError("POST "+systemOnePath+" is available only when a jev or openrouter provider is configured")) + } + body, err := requestBodyBytes(c) + if err != nil { + return handleError(c, core.NewInvalidRequestError("invalid request body: "+err.Error(), err)) + } + var req core.SystemOneRequest + if err := json.Unmarshal(body, &req); err != nil { + return handleError(c, core.NewInvalidRequestError("invalid request body: "+err.Error(), err)) + } + if strings.TrimSpace(req.Model) == "" { + return handleError(c, core.NewInvalidRequestError("model is required", nil).WithParam("model")) + } + + workflow, err := s.systemOneWorkflow(c, req.Model) + if err != nil { + return handleError(c, err) + } + ctx := c.Request().Context() + resolution := workflow.Resolution + if s.modelAuthorizer != nil { + if err := s.modelAuthorizer.ValidateModelAccess(ctx, resolution.ResolvedSelector); err != nil { + return handleError(c, err) + } + } + if reason := s.systemOneUnsupportedReason(resolution); reason != "" { + return handleError(c, systemOneUnsupportedModelError(c, resolution, reason)) + } + + body, err = s.guardSystemOneState(c, workflow, &req, body) + if err != nil { + return handleError(c, err) + } + model := resolution.ResolvedSelector.Model + if body, err = rewriteMessagesModel(body, model); err != nil { + return handleError(c, core.NewInvalidRequestError("invalid request body: "+err.Error(), err)) + } + return s.dispatchSystemOne(c, workflow, model, body) +} + +// systemOneAvailable reports whether a provider that serves System One is +// configured. It is checked per request rather than at route registration so +// a provider added at runtime makes the endpoint available without a restart. +func systemOneAvailable(provider core.RoutableProvider) bool { + named, ok := provider.(core.ProviderTypeNameResolver) + if !ok { + return false + } + for _, providerType := range systemOneProviderTypes { + if strings.TrimSpace(named.GetProviderNameForType(providerType)) != "" { + return true + } + } + return false +} + +// systemOneWorkflow returns the request's workflow with its model resolved. +// The workflow middleware resolves it from the body; a request that reached +// the handler without one (the body was not parsed there) is resolved here, +// so virtual models and workflow policy apply either way. +func (s *translatedInferenceService) systemOneWorkflow(c *echo.Context, model string) (*core.Workflow, error) { + if workflow := core.GetWorkflow(c.Request().Context()); workflow != nil && workflow.Resolution != nil { + return workflow, nil + } + resolution, err := resolveAndStoreRequestModelResolution(c, s.provider, s.modelResolver, nil, model, "") + if err != nil { + return nil, err + } + workflow, err := translatedWorkflowForRequest(c, resolution, s.workflowPolicyResolver) + if err != nil { + return nil, err + } + storeWorkflow(c, workflow) + return workflow, nil +} + +// modelCatalog describes single catalog models; the provider router +// implements it. +type modelCatalog interface { + LookupModel(model string) (*core.Model, bool) +} + +// systemOneUnsupportedReason explains why the resolved model cannot answer a +// System One request, or returns "" when it can. The provider must serve the +// API, and since OpenRouter also serves chat models, the model must not be +// catalogued with a generation mode. A model the catalog does not describe is +// given the benefit of the doubt: the upstream reports it if it is wrong. +func (s *translatedInferenceService) systemOneUnsupportedReason(resolution *core.RequestModelResolution) string { + providerType := strings.TrimSpace(resolution.ProviderType) + if !slices.Contains(systemOneProviderTypes, providerType) { + return fmt.Sprintf("is served by a %s provider, which has no System One API", providerType) + } + catalog, ok := s.provider.(modelCatalog) + if !ok { + return "" + } + model, ok := catalog.LookupModel(resolution.ResolvedQualifiedModel()) + if !ok || model == nil || model.Metadata == nil || len(model.Metadata.Modes) == 0 { + return "" + } + return fmt.Sprintf("is a %s model, not a System One model", strings.Join(model.Metadata.Modes, "/")) +} + +// systemOneUnsupportedModelError explains a request whose model cannot answer +// System One. It is also logged: a virtual model that sends System One +// traffic to a chat model is an operator mistake the caller cannot fix. +func systemOneUnsupportedModelError(c *echo.Context, resolution *core.RequestModelResolution, reason string) error { + requested := resolution.RequestedQualifiedModel() + resolved := resolution.ResolvedQualifiedModel() + slog.Warn("System One request routed to a model without the System One API", + "request_id", requestIDFromContextOrHeader(c.Request()), + "requested_model", requested, + "resolved_model", resolved, + "provider_type", resolution.ProviderType, + "reason", reason, + ) + target := fmt.Sprintf("%q", requested) + if resolved != requested { + target += fmt.Sprintf(" (resolved to %q)", resolved) + } + return core.NewInvalidRequestError(fmt.Sprintf( + "model %s %s; %s forwards requests natively and does not translate them to other APIs, so use a System One model such as a jev model or OpenRouter's typesafe/jev-1.13", + target, reason, systemOnePath, + ), nil).WithParam("model") +} + +// guardSystemOneState runs the workflow's prompt guardrails over the state +// and returns the body carrying their edits. A guardrail that answers the +// request itself blocks it instead: its answer is chat text, and a System +// One caller expects typed answers. +func (s *translatedInferenceService) guardSystemOneState(c *echo.Context, workflow *core.Workflow, req *core.SystemOneRequest, body []byte) ([]byte, error) { + patcher, ok := s.translatedRequestPatcher.(gateway.SystemOneRequestPatcher) + if !ok || !workflow.GuardrailsEnabled() { + return body, nil + } + // Compare against a copy: a patcher may redact the state in place and + // return the same request, and that edit must still reach the body. + original := bytes.Clone(req.State) + patched, err := patcher.PatchSystemOneRequest(c.Request().Context(), req) + s.recordGuardrailOutcomes(c) + if err != nil { + if short := shortCircuitOf(err); short != nil { + return nil, plugins.BlockError(short.Decision, http.StatusBadRequest) + } + return nil, err + } + if patched == nil || bytes.Equal(patched.State, original) { + return body, nil + } + rewritten, err := replaceTopLevelMember(body, "state", patched.State) + if err != nil { + return nil, core.NewInvalidRequestError("invalid request body: "+err.Error(), err) + } + return rewritten, nil +} + +// dispatchSystemOne forwards the body to the resolved provider and relays its +// answer unchanged, with admission, audit, and usage accounting. +func (s *translatedInferenceService) dispatchSystemOne(c *echo.Context, workflow *core.Workflow, model string, body []byte) error { + passthroughProvider, ok := s.provider.(core.RoutablePassthrough) + if !ok { + return handleError(c, core.NewInvalidRequestError("provider passthrough is not supported by the current provider router", nil)) + } + s.observeLiveProviderAttempts(c, workflow) + + adm, err := enforceAdmission(c, s.rateLimiter, s.budgetChecker, rateLimitRouteFromWorkflow(workflow)) + if err != nil { + return handleError(c, err) + } + defer adm.release() + ctx := adm.dispatchContext(c.Request().Context()) + + resolution := workflow.Resolution + providerType := strings.TrimSpace(resolution.ProviderType) + providerName := strings.TrimSpace(resolution.ProviderName) + resp, err := passthroughProvider.Passthrough(ctx, providerType, &core.PassthroughRequest{ + Method: http.MethodPost, + Endpoint: systemOneEndpoint, + Operation: "systemone", + Model: model, + Body: io.NopCloser(bytes.NewReader(body)), + Headers: buildPassthroughHeaders(ctx, c.Request().Header), + ProviderName: providerName, + }) + if err != nil { + return handleError(c, err) + } + + auditlog.EnrichEntryWithWorkflow(c, workflow) + auditlog.EnrichEntryWithResolvedRoute(c, resolution.ResolvedQualifiedModel(), providerType, providerName) + info := &core.PassthroughRouteInfo{ + Provider: providerType, + ProviderName: providerName, + NormalizedEndpoint: systemOneEndpoint, + SemanticOperation: "systemone", + AuditPath: systemOnePath, + Model: model, + } + return proxyPassthroughResponse(c, s.logger, s.usageLogger, s.pricingResolver, providerType, providerName, systemOneEndpoint, info, resp) +} diff --git a/internal/server/systemone_handler_test.go b/internal/server/systemone_handler_test.go new file mode 100644 index 000000000..45e82b6cc --- /dev/null +++ b/internal/server/systemone_handler_test.go @@ -0,0 +1,367 @@ +package server + +import ( + "context" + "io" + "net/http" + "strings" + "testing" + + "github.com/goccy/go-json" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/echotest" + "github.com/enterpilot/gomodel/internal/usage" +) + +const ( + systemOneQuestions = `{"refund":{"type":"noul","instructions":"Is the customer asking for money back?"}}` + systemOneAnswer = `{"model":"kev-1.0","answers":{"refund":{"type":"noul","noul":0.98}},"usage":{"input_tokens":275,"output_tokens":20}}` +) + +func systemOneBody(model string) string { + return `{"model":"` + model + `","state":"I was charged twice for card 4111.","questions":` + systemOneQuestions + `}` +} + +// systemOneAliasResolver maps virtual model names to concrete selectors. +type systemOneAliasResolver map[string]core.ModelSelector + +func (r systemOneAliasResolver) ResolveModel(requested core.RequestedModelSelector) (core.ModelSelector, bool, error) { + if selector, ok := r[requested.RequestedQualifiedModel()]; ok { + return selector, true, nil + } + selector, err := requested.Normalize() + return selector, false, err +} + +// newSystemOneProvider configures a local Kev server (type jev, named kev), +// OpenRouter, and an OpenAI chat model, answering every passthrough with body. +func newSystemOneProvider(body string) *mockProvider { + return &mockProvider{ + supportedModels: []string{"kev-latest", "typesafe/jev-1.13", "gpt-5-mini"}, + providerTypes: map[string]string{ + "kev/kev-latest": "jev", + "openrouter/typesafe/jev-1.13": "openrouter", + "openai/gpt-5-mini": "openai", + }, + providerNames: map[string]string{ + "kev/kev-latest": "kev", + "openrouter/typesafe/jev-1.13": "openrouter", + "openai/gpt-5-mini": "openai", + }, + passthroughResponse: &core.PassthroughResponse{ + StatusCode: http.StatusOK, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(body)), + }, + } +} + +var systemOneAliases = systemOneAliasResolver{ + "decider": {Provider: "kev", Model: "kev-latest"}, + "chatty": {Provider: "openai", Model: "gpt-5-mini"}, +} + +func forwardedSystemOneBody(t *testing.T, provider *mockProvider) map[string]any { + t.Helper() + require.NotNil(t, provider.lastPassthroughReq) + raw, err := io.ReadAll(provider.lastPassthroughReq.Body) + require.NoError(t, err) + var body map[string]any + require.NoError(t, json.Unmarshal(raw, &body)) + + return body +} + +// Without a provider that serves System One the endpoint does not exist, +// whatever the model. +func TestSystemOne_UnavailableWithoutSystemOneProvider(t *testing.T) { + provider := &mockProvider{ + supportedModels: []string{"gpt-5-mini"}, + providerTypes: map[string]string{"openai/gpt-5-mini": "openai"}, + providerNames: map[string]string{"openai/gpt-5-mini": "openai"}, + } + handler := NewHandler(provider, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("gpt-5-mini")) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusNotFound, rec.Code) + assert.Contains(t, rec.Body.String(), "jev or openrouter provider") + assert.Nil(t, provider.lastPassthroughReq) +} + +// A virtual model resolves to its System One target, and the body reaches the +// provider unchanged except for the concrete model name. +func TestSystemOne_ForwardsNativelyAndRecordsUsage(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + handler := newHandlerWithAuthorizer(provider, nil, usageLogger, nil, systemOneAliases, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("decider")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.JSONEq(t, systemOneAnswer, rec.Body.String()) + + assert.Equal(t, "jev", provider.lastPassthroughProvider) + assert.Equal(t, "systemone", provider.lastPassthroughReq.Endpoint) + assert.Equal(t, "kev", provider.lastPassthroughReq.ProviderName) + body := forwardedSystemOneBody(t, provider) + assert.Equal(t, "kev-latest", body["model"]) + assert.Equal(t, "I was charged twice for card 4111.", body["state"]) + questions, err := json.Marshal(body["questions"]) + require.NoError(t, err) + assert.JSONEq(t, systemOneQuestions, string(questions)) + + require.Len(t, usageLogger.entries, 1) + entry := usageLogger.entries[0] + assert.Equal(t, 275, entry.InputTokens) + assert.Equal(t, 20, entry.OutputTokens) + assert.Equal(t, "/v1/systemone", entry.Endpoint) + assert.Equal(t, "jev", entry.Provider) + assert.Equal(t, "kev", entry.ProviderName) + assert.Equal(t, "kev-1.0", entry.Model, "usage is recorded under the model that answered") +} + +// OpenRouter serves Jev at the same path, so it is a native target too; its +// reported cost is kept with the usage entry. +func TestSystemOne_ForwardsOpenRouterJevNatively(t *testing.T) { + answer := `{"id":"gen-dec-1","model":"typesafe/jev-1.13-20260917","answers":{},"usage":{"input_tokens":275,"output_tokens":20,"cost":0.00003}}` + provider := newSystemOneProvider(answer) + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + handler := newHandlerWithAuthorizer(provider, nil, usageLogger, nil, nil, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("openrouter/typesafe/jev-1.13")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, "openrouter", provider.lastPassthroughProvider) + assert.Equal(t, "typesafe/jev-1.13", forwardedSystemOneBody(t, provider)["model"]) + require.Len(t, usageLogger.entries, 1) + assert.InDelta(t, 0.00003, usageLogger.entries[0].RawData["cost"], 1e-12) +} + +// OpenRouter serves System One natively, so it enables the endpoint on its own. +// Its catalog names Jev "~typesafe/jev-latest"; a virtual model gives SDK +// callers the "jev-latest" name they send by default. +func TestSystemOne_WorksWithOpenRouterAlone(t *testing.T) { + answer := `{"id":"gen-dec-1","model":"typesafe/jev-1.13-20260917","answers":{},"usage":{"input_tokens":10,"output_tokens":1}}` + for _, model := range []string{"openrouter/~typesafe/jev-latest", "jev-latest"} { + t.Run(model, func(t *testing.T) { + provider := &mockProvider{ + supportedModels: []string{"~typesafe/jev-latest"}, + providerTypes: map[string]string{"openrouter/~typesafe/jev-latest": "openrouter"}, + providerNames: map[string]string{"openrouter/~typesafe/jev-latest": "openrouter"}, + passthroughResponse: &core.PassthroughResponse{ + StatusCode: http.StatusOK, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(answer)), + }, + } + aliases := systemOneAliasResolver{"jev-latest": {Provider: "openrouter", Model: "~typesafe/jev-latest"}} + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, aliases, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody(model)) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, "openrouter", provider.lastPassthroughProvider) + assert.Equal(t, "systemone", provider.lastPassthroughReq.Endpoint) + assert.Equal(t, "~typesafe/jev-latest", forwardedSystemOneBody(t, provider)["model"]) + }) + } +} + +// The endpoint never translates: a model on a provider without the System One +// API is rejected with an explanation, whether named directly or through a +// virtual model. +func TestSystemOne_RejectsModelsWithoutSystemOneAPI(t *testing.T) { + for _, model := range []string{"openai/gpt-5-mini", "chatty"} { + t.Run(model, func(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, systemOneAliases, nil, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody(model)) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "no System One API") + assert.Contains(t, rec.Body.String(), "does not translate") + assert.Nil(t, provider.lastPassthroughReq) + }) + } +} + +// catalogProvider adds the router's single-model catalog lookup to the mock. +type catalogProvider struct { + *mockProvider + models map[string]core.Model +} + +func (p catalogProvider) LookupModel(model string) (*core.Model, bool) { + found, ok := p.models[model] + return &found, ok +} + +// OpenRouter serves chat and decision models from one provider, so the model +// itself must be a System One model: one catalogued with a generation mode is +// rejected, while a decision model (a utility model with no mode) is forwarded. +func TestSystemOne_RejectsOpenRouterChatModels(t *testing.T) { + provider := catalogProvider{ + mockProvider: &mockProvider{ + supportedModels: []string{"typesafe/jev-1.13", "openai/gpt-4o-mini"}, + providerTypes: map[string]string{ + "openrouter/typesafe/jev-1.13": "openrouter", + "openrouter/openai/gpt-4o-mini": "openrouter", + }, + providerNames: map[string]string{ + "openrouter/typesafe/jev-1.13": "openrouter", + "openrouter/openai/gpt-4o-mini": "openrouter", + }, + passthroughResponse: &core.PassthroughResponse{ + StatusCode: http.StatusOK, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(systemOneAnswer)), + }, + }, + models: map[string]core.Model{ + "openrouter/typesafe/jev-1.13": {ID: "typesafe/jev-1.13", Metadata: &core.ModelMetadata{ + Categories: []core.ModelCategory{core.CategoryUtility}, + }}, + "openrouter/openai/gpt-4o-mini": {ID: "openai/gpt-4o-mini", Metadata: &core.ModelMetadata{ + Modes: []string{"chat"}, Categories: []core.ModelCategory{core.CategoryTextGeneration}, + }}, + }, + } + handler := NewHandler(provider, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("openrouter/openai/gpt-4o-mini")) + require.NoError(t, handler.SystemOne(c)) + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "is a chat model, not a System One model") + assert.Contains(t, rec.Body.String(), "does not translate") + assert.Nil(t, provider.lastPassthroughReq) + + c, rec = echotest.Post(t, "/v1/systemone", systemOneBody("openrouter/typesafe/jev-1.13")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.Equal(t, "openrouter", provider.lastPassthroughProvider) +} + +func TestSystemOne_RequiresModel(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := NewHandler(provider, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", `{"state":"hi","questions":{}}`) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "model is required") +} + +// stateRedactingPatcher stands in for an anonymizing guardrail. +type stateRedactingPatcher struct{} + +func (stateRedactingPatcher) PatchChatRequest(_ context.Context, req *core.ChatRequest) (*core.ChatRequest, error) { + return req, nil +} + +func (stateRedactingPatcher) PatchResponsesRequest(_ context.Context, req *core.ResponsesRequest) (*core.ResponsesRequest, error) { + return req, nil +} + +func (stateRedactingPatcher) PatchSystemOneRequest(_ context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) { + patched := *req + patched.State = json.RawMessage(strings.ReplaceAll(string(req.State), "4111", "[card]")) + return &patched, nil +} + +// inPlaceRedactingPatcher edits the request it was given and returns it; the +// patcher contract allows that, and the edit must still reach the provider. +type inPlaceRedactingPatcher struct{ stateRedactingPatcher } + +func (inPlaceRedactingPatcher) PatchSystemOneRequest(_ context.Context, req *core.SystemOneRequest) (*core.SystemOneRequest, error) { + req.State = json.RawMessage(strings.ReplaceAll(string(req.State), "4111", "[card]")) + return req, nil +} + +// Guardrails see the state and their edits reach the provider, whether the +// patcher returns a copy or edits in place; the rest of the body is untouched. +func TestSystemOne_GuardrailsEditState(t *testing.T) { + patchers := map[string]TranslatedRequestPatcher{ + "copy": stateRedactingPatcher{}, + "in place": inPlaceRedactingPatcher{}, + } + for name, patcher := range patchers { + t.Run(name, func(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + handler := newHandlerWithAuthorizer(provider, nil, nil, nil, systemOneAliases, nil, nil, nil, patcher) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneBody("decider")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + body := forwardedSystemOneBody(t, provider) + assert.Equal(t, "I was charged twice for card [card].", body["state"]) + assert.Equal(t, "kev-latest", body["model"]) + }) + } +} + +// Through the full middleware stack an unavailable endpoint answers 404 +// before the model is resolved, so a missing or unknown model is not +// reported for a route that is not there. +func TestSystemOne_UnavailableBeforeModelResolution(t *testing.T) { + provider := &mockProvider{ + supportedModels: []string{"gpt-5-mini"}, + providerTypes: map[string]string{"openai/gpt-5-mini": "openai"}, + providerNames: map[string]string{"openai/gpt-5-mini": "openai"}, + } + srv := New(provider, &Config{}) + + for _, body := range []string{`{"state":"hi","questions":{}}`, systemOneBody("no-such-model")} { + rec := postJSON(t, srv, "/v1/systemone", body) + assert.Equal(t, http.StatusNotFound, rec.Code, rec.Body.String()) + assert.Contains(t, rec.Body.String(), "jev or openrouter provider") + } +} + +// Through the full middleware stack a System One call is audited under its +// own path with the requested and resolved routes, and its usage recorded. +func TestSystemOne_AuditsAndRecordsUsageThroughServer(t *testing.T) { + provider := newSystemOneProvider(systemOneAnswer) + auditLogger := &capturingAuditLogger{config: auditlog.Config{Enabled: true, LogBodies: true}} + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + srv := New(provider, &Config{ + AuditLogger: auditLogger, + UsageLogger: usageLogger, + ModelResolver: systemOneAliases, + }) + + rec := postJSON(t, srv, "/v1/systemone", systemOneBody("decider")) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + require.Len(t, auditLogger.entries, 1) + entry := auditLogger.entries[0] + assert.Equal(t, "/v1/systemone", entry.Path) + assert.Equal(t, http.StatusOK, entry.StatusCode) + assert.Equal(t, "decider", entry.RequestedModel) + assert.Equal(t, "kev/kev-latest", entry.ResolvedModel) + assert.True(t, entry.AliasUsed) + assert.Equal(t, "jev", entry.Provider) + assert.Equal(t, "kev", entry.ProviderName) + require.NotNil(t, entry.Data) + requestBody, err := json.Marshal(entry.Data.RequestBody) + require.NoError(t, err) + assert.Contains(t, string(requestBody), "charged twice") + responseBody, err := json.Marshal(entry.Data.ResponseBody) + require.NoError(t, err) + assert.Contains(t, string(responseBody), "noul") + + require.Len(t, usageLogger.entries, 1) + assert.Equal(t, "/v1/systemone", usageLogger.entries[0].Endpoint) + assert.Equal(t, entry.RequestID, usageLogger.entries[0].RequestID) +} diff --git a/web/dashboard/messages/de.json b/web/dashboard/messages/de.json index b4697d221..ad99c1e20 100644 --- a/web/dashboard/messages/de.json +++ b/web/dashboard/messages/de.json @@ -233,6 +233,7 @@ "audit_type_embeddings": "Embeddings", "audit_type_audio": "Audio", "audit_type_images": "Bilder", + "audit_type_systemone": "System One", "audit_type_batches": "Batches & Dateien", "audit_type_realtime": "Realtime", "audit_type_passthrough": "Passthrough", diff --git a/web/dashboard/messages/en.json b/web/dashboard/messages/en.json index ff304603f..b8db00471 100644 --- a/web/dashboard/messages/en.json +++ b/web/dashboard/messages/en.json @@ -233,6 +233,7 @@ "audit_type_embeddings": "Embeddings", "audit_type_audio": "Audio", "audit_type_images": "Images", + "audit_type_systemone": "System One", "audit_type_batches": "Batches & files", "audit_type_realtime": "Realtime", "audit_type_passthrough": "Passthrough", diff --git a/web/dashboard/messages/pl.json b/web/dashboard/messages/pl.json index 3e9a218a3..783aabee5 100644 --- a/web/dashboard/messages/pl.json +++ b/web/dashboard/messages/pl.json @@ -235,6 +235,7 @@ "audit_type_embeddings": "Embeddingi", "audit_type_audio": "Audio", "audit_type_images": "Obrazy", + "audit_type_systemone": "System One", "audit_type_batches": "Batche i pliki", "audit_type_realtime": "Realtime", "audit_type_passthrough": "Passthrough", diff --git a/web/dashboard/messages/zh-CN.json b/web/dashboard/messages/zh-CN.json index dbf13cfc7..0c60c7e70 100644 --- a/web/dashboard/messages/zh-CN.json +++ b/web/dashboard/messages/zh-CN.json @@ -224,6 +224,7 @@ "audit_type_embeddings": "嵌入", "audit_type_audio": "音频", "audit_type_images": "图像", + "audit_type_systemone": "System One", "audit_type_batches": "批处理与文件", "audit_type_realtime": "实时", "audit_type_passthrough": "透传", diff --git a/web/dashboard/src/pages/audit-logs/audit-operations.js b/web/dashboard/src/pages/audit-logs/audit-operations.js index 7c493dbd1..7bbd43321 100644 --- a/web/dashboard/src/pages/audit-logs/audit-operations.js +++ b/web/dashboard/src/pages/audit-logs/audit-operations.js @@ -17,6 +17,7 @@ export const AUDIT_TYPES = [ operations: ["audio_speech", "audio_transcriptions", "audio_translations"], }, { key: "images", label: () => m.audit_type_images(), operations: ["image_generations", "image_edits"] }, + { key: "systemone", label: () => m.audit_type_systemone(), operations: ["systemone"] }, { key: "batches", label: () => m.audit_type_batches(), operations: ["batches", "files"] }, { key: "realtime", label: () => m.audit_type_realtime(), operations: ["realtime"] }, { key: "passthrough", label: () => m.audit_type_passthrough(), operations: ["provider_passthrough"] }, @@ -35,6 +36,7 @@ const EXACT_PATHS = { "/v1/audio/translations": "audio", "/v1/images/generations": "images", "/v1/images/edits": "images", + "/v1/systemone": "systemone", "/v1/realtime": "realtime", "/v1/realtime/calls": "realtime", "/v1/realtime/client_secrets": "realtime", diff --git a/web/dashboard/tests/audit-operations.test.js b/web/dashboard/tests/audit-operations.test.js index 1b8914269..e94d99faf 100644 --- a/web/dashboard/tests/audit-operations.test.js +++ b/web/dashboard/tests/audit-operations.test.js @@ -20,6 +20,8 @@ test("auditTypeForPath mirrors the gateway endpoint classification", () => { ["/v1/files/f_1/content", "batches"], ["/v1/audio/transcriptions?x=1", "audio"], ["/v1/images/edits/", "images"], + ["/v1/systemone", "systemone"], + ["/v1/systemone/permute", ""], ["/v1/realtime/translations/calls", "realtime"], ["/mcp", "mcp"], ["/mcp/github", "mcp"], From 667453caa520e5af19cc04e4628cc9be1296385f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jakub=20A=2E=20W=C4=85sek?= Date: Sat, 26 Sep 2026 21:33:27 +0200 Subject: [PATCH 11/23] feat(jev): cache, fail over, pin versions, and serve Kev routes on /v1/systemone (#1098) * feat(jev): cache, fail over, pin versions, and serve Kev routes on /v1/systemone * fix(jev): skip ineligible failover targets without using attempts * fix(jev): point OpenAI routes at /v1/systemone for decision models and warn once on dropped guardrail edits --- cmd/gomodel/docs/docs.go | 138 ++++++++ config/config.example.yaml | 1 + docs/advanced/api-endpoints.mdx | 15 +- docs/advanced/systemone-api.mdx | 179 ++++++++++ docs/docs.json | 1 + docs/features/cache.mdx | 4 +- docs/openapi.json | 193 +++++++++- docs/providers/jev.mdx | 80 +---- internal/core/endpoint_operations.go | 2 +- internal/core/endpoints.go | 9 +- internal/core/endpoints_test.go | 4 +- internal/gateway/failover.go | 69 +++- internal/gateway/failover_policy_test.go | 25 +- internal/gateway/failover_test.go | 6 +- internal/guardrails/workflow_executor.go | 15 +- .../plugins/exchange/systemone_request.go | 20 +- .../exchange/systemone_request_test.go | 6 +- .../providers/registry_normalization_test.go | 17 + internal/providers/router_models.go | 17 + internal/server/http.go | 2 + internal/server/model_validation.go | 17 +- internal/server/systemone_dispatch.go | 158 +++++++++ internal/server/systemone_dispatch_test.go | 334 ++++++++++++++++++ internal/server/systemone_handler.go | 259 +++++++++----- internal/usage/extractor.go | 28 ++ internal/usage/extractor_test.go | 14 + .../src/pages/audit-logs/audit-operations.js | 2 + web/dashboard/tests/audit-operations.test.js | 4 +- 28 files changed, 1434 insertions(+), 185 deletions(-) create mode 100644 docs/advanced/systemone-api.mdx create mode 100644 internal/server/systemone_dispatch.go create mode 100644 internal/server/systemone_dispatch_test.go diff --git a/cmd/gomodel/docs/docs.go b/cmd/gomodel/docs/docs.go index dacf4d64d..90f72c5dc 100644 --- a/cmd/gomodel/docs/docs.go +++ b/cmd/gomodel/docs/docs.go @@ -7573,6 +7573,144 @@ const docTemplate = `{ ] } }, + "/v1/systemone/permute": { + "post": { + "description": "A Kev server diagnostic: the request is a System One request, and n_perm (1 to 64, default 6) sets how many option orders run. Only jev providers pointing at a Kev server serve it.", + "consumes": [ + "application/json" + ], + "produces": [ + "application/json" + ], + "tags": [ + "systemone" + ], + "summary": "Run one Choice question with several option orders (Kev)", + "parameters": [ + { + "description": "System One request with one Choice question", + "name": "request", + "in": "body", + "required": true, + "schema": { + "type": "object" + } + } + ], + "responses": { + "200": { + "description": "Kev's answer, in the provider's shape", + "schema": { + "type": "object" + } + }, + "400": { + "description": "Bad Request", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "401": { + "description": "Unauthorized", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "404": { + "description": "Not Found", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "429": { + "description": "Too Many Requests", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "502": { + "description": "Bad Gateway", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ] + } + }, + "/v1/systemone/separate": { + "post": { + "description": "A Kev server diagnostic that answers each question separately. Only jev providers pointing at a Kev server serve it.", + "consumes": [ + "application/json" + ], + "produces": [ + "application/json" + ], + "tags": [ + "systemone" + ], + "summary": "Run each System One question in its own forward pass (Kev)", + "parameters": [ + { + "description": "System One request: model, state, and questions", + "name": "request", + "in": "body", + "required": true, + "schema": { + "type": "object" + } + } + ], + "responses": { + "200": { + "description": "Kev's answer, in the provider's shape", + "schema": { + "type": "object" + } + }, + "400": { + "description": "Bad Request", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "401": { + "description": "Unauthorized", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "404": { + "description": "Not Found", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "429": { + "description": "Too Many Requests", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + }, + "502": { + "description": "Bad Gateway", + "schema": { + "$ref": "#/definitions/core.OpenAIErrorEnvelope" + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ] + } + }, "/v1/usage": { "get": { "description": "Returns recorded usage, budget statuses, and rate limit statuses for the caller's effective user path (the path bound to the managed API key, or the user-path header for master-key callers).", diff --git a/config/config.example.yaml b/config/config.example.yaml index d1213070e..1a177aaa6 100644 --- a/config/config.example.yaml +++ b/config/config.example.yaml @@ -598,6 +598,7 @@ providers: # server speaks the same API without authentication: set base_url # (e.g. "http://localhost:8009") and omit api_key. Name it "kev" to see # that name in logs and usage; no separate provider type is needed. + # Pinned versions such as "jev-1.13.0" route here without being listed. # Jev is priced per input token and is not in the upstream model catalog; # declare its pricing here to have the gateway cost System One requests. # models: diff --git a/docs/advanced/api-endpoints.mdx b/docs/advanced/api-endpoints.mdx index f39cc87da..ff9b452fc 100644 --- a/docs/advanced/api-endpoints.mdx +++ b/docs/advanced/api-endpoints.mdx @@ -12,8 +12,8 @@ documented separately in [Admin Endpoints](/advanced/admin-endpoints). For request and response details, see the dedicated guides: [Responses API](/advanced/responses-api), [Conversations API](/advanced/conversations-api), [Anthropic Messages API](/advanced/anthropic-messages-api), -[Audio API](/advanced/audio-api), [Images API](/advanced/images-api), and -[Usage API](/advanced/usage-api). +[Audio API](/advanced/audio-api), [Images API](/advanced/images-api), +[System One API](/advanced/systemone-api), and [Usage API](/advanced/usage-api). ## OpenAI-Compatible API @@ -111,6 +111,17 @@ metered. | `/v1/messages` | POST | Anthropic Messages API through translated model routing (streaming supported) | | `/v1/messages/count_tokens` | POST | Heuristic Anthropic Messages input token estimate | +## System One API + +Available when a `jev` or `openrouter` provider is configured; see +[System One API](/advanced/systemone-api). + +| Endpoint | Method | Description | +| ---------------------------- | ------ | ------------------------------------------------------------------------ | +| `/v1/systemone` | POST | Evaluate a state against typed questions (Jev, Kev), forwarded natively | +| `/v1/systemone/permute` | POST | Kev only: run one Choice question with several option orders | +| `/v1/systemone/separate` | POST | Kev only: run each question in its own forward pass | + ## Gateway Extensions | Endpoint | Method | Description | diff --git a/docs/advanced/systemone-api.mdx b/docs/advanced/systemone-api.mdx new file mode 100644 index 000000000..8e765fe86 --- /dev/null +++ b/docs/advanced/systemone-api.mdx @@ -0,0 +1,179 @@ +--- +title: "System One API" +description: "Send TypeSafe System One decision requests (Jev, Kev) through GoModel, forwarded natively with virtual models, guardrails, caching, failover, audit, and usage." +icon: "scale" +keywords: ["System One", "systemone", "Jev", "Kev", "TypeSafe", "decision model", "noul", "choice", "score", "OpenRouter"] +--- + +`POST /v1/systemone` serves TypeSafe's System One API: a request carries a +`state` (the text or record to evaluate) and a map of typed questions, and the +answer is a calibrated probability per question. It is a decision API, not a +text generator, so GoModel forwards it **natively** and never translates it to +or from chat. + +The endpoint is available once a [`jev` provider](/providers/jev) (hosted Jev +or a self-hosted Kev server) or an `openrouter` provider is configured. Without +one, it answers `404`. + +## Request and answer + +```bash +curl -s http://localhost:8080/v1/systemone \ + -H "Authorization: Bearer $GOMODEL_MASTER_KEY" \ + -H "Content-Type: application/json" \ + -d '{ + "model": "jev-latest", + "state": "Shoes arrived two weeks late and in the wrong size. Also I see two charges on my card.", + "questions": { + "department": {"type": "choice", "instructions": "Which team should handle this?", + "criteria": {"returns": "Exchanges, refunds, wrong items", + "shipping": "Delivery status, delays", + "billing": "Charges, invoices"}}, + "escalate": {"type": "noul", "instructions": "Does this need urgent human attention?"}, + "frustration": {"type": "score", "instructions": "How frustrated is the customer?", + "criteria": ["Calm", "Frustrated", "Very angry"]} + } + }' +``` + +The answer is the provider's own, relayed unchanged: + +```json +{ + "model": "jev-1.13.0", + "answers": { + "department": {"type": "choice", "choice": "returns", "confidence": 0.21, + "probabilities": {"returns": 0.47, "shipping": 0.28, "billing": 0.25}}, + "escalate": {"type": "noul", "noul": 0.93}, + "frustration": {"type": "score", "score": 1.44, "confidence": 0.78, + "legend": {"0": "Calm", "1": "Frustrated", "2": "Very angry"}, + "probabilities": {"0": 0.00, "1": 0.56, "2": 0.44}} + }, + "usage": {"input_tokens": 101, "output_tokens": 161} +} +``` + +The TypeSafe SDKs send `POST {base_url}/v1/systemone`, so point them at the +gateway root (`base_url="http://localhost:8080"`) with your GoModel key; see +[Jev / Kev](/providers/jev#using-the-typesafe-sdks). + +## Routes + +| Route | What it does | +| --- | --- | +| `POST /v1/systemone` | Evaluate a state against a map of questions | +| `POST /v1/systemone/permute` | Kev only: run one Choice question with several option orders (`n_perm`, 1 to 64, default 6) | +| `POST /v1/systemone/separate` | Kev only: run each question in its own forward pass | + +The Kev routes behave like `/v1/systemone`. They are refused for OpenRouter, +which answers only the evaluation route; a hosted TypeSafe `jev` provider +returns its own `404` for them. + +## What the gateway does + +1. Resolves `model` like any other endpoint: a bare name, a provider-qualified + name (`jev/jev-latest`), or a [virtual model](/features/virtual-models), + then applies the caller's [model allowlist](/features/users), + [rate limits](/features/rate-limits), and [budgets](/features/budgets). +2. Runs the workflow's prompt [guardrails](/advanced/guardrails) over `state`. +3. Serves an identical earlier request from the [response cache](/features/cache). +4. Forwards the body with only `model` (the resolved name) and `state` (if a + guardrail edited it) changed. Questions, criteria, and every other field + reach the provider byte for byte. +5. Relays the answer unchanged and records it in the audit log (request type + **System One**) and in usage. + +## Models + +| Provider | Model names | +| --- | --- | +| `jev` (hosted) | `jev/jev-latest`, `jev/jev-preview`, and any versioned ID such as `jev/jev-1.13.0` | +| `jev` (Kev server) | `kev/kev-latest` and the checkpoint's aliases, for a provider named `kev` | +| `openrouter` | `openrouter/typesafe/jev-1.13`, `openrouter/~typesafe/jev-latest`, and OpenRouter's other decision models, such as `openrouter/jaredpalmer/kev-4b` | + +System One models are listed in `GET /v1/models` as utility models with no +generation mode. + +TypeSafe lists only its aliases but accepts any versioned ID, so a pinned +version works without being declared: GoModel routes a model it does not list +to a `jev` provider when the name says which one (`jev/jev-1.13.0`), or, for a +bare name, when exactly one `jev` provider is configured. A virtual model can +pin a version the same way. + +OpenRouter accepts `jev-latest` itself, but GoModel routes on its catalog IDs. +To keep a plain `jev-latest` (the TypeSafe SDKs' default) working through +OpenRouter, add a virtual model: + +```yaml +virtual_models: + - source: jev-latest + target: openrouter/~typesafe/jev-latest +``` + +## Caching + +With the [response cache](/features/cache) enabled, an identical request (same +route, resolved model, guardrails, and body after guardrail edits) is answered +from the exact cache (`X-Cache: HIT (exact)`) and recorded in usage as a cache +hit. The semantic cache never serves System One: a state that is merely +similar is not the same decision. Send `Cache-Control: no-cache` to skip the +cache for one request. + +## Failover + +A virtual model with the `failover` strategy moves a request to its next +target when the current one fails with an availability error (`429` or `5xx`, +including TypeSafe's `529`, by default; see [Failover](/features/failover)). +Every target receives the request in its own System One form. A target without +the API, such as a chat model, is skipped without using a failover attempt, +and client errors such as a malformed question (`422`) are returned without +failover: + +```yaml +virtual_models: + - source: decider + strategy: failover + targets: + - { model: kev/kev-latest } # local Kev first + - { model: openrouter/typesafe/jev-1.13 } # hosted Jev when Kev is down +``` + +The audit log shows each attempt, usage is recorded under the target that +answered, and a failover answer is not cached. + +## Guardrails + +Guardrails see `state` as a single user message: a string state as its text, +any other JSON value as its encoded JSON, which must still be valid JSON after +an edit. That is what anonymizing and blocking guardrails need; for example, a +`string_replace` rule that masks card numbers applies to `state` before it +leaves the gateway. The questions are your application's fixed schema and are +not exposed. + +Edits a decision request has no place for, such as a system prompt injected by +a guardrail that also covers chat models, are dropped. The gateway logs one +warning per kind of dropped edit, then logs repeats at debug level. A +guardrail that would answer the request itself blocks it instead, since System +One callers expect typed answers, not text. + +## Errors and misuse + +The endpoint never translates, and it says so when a request cannot work: + +| Situation | Result | +| --- | --- | +| No `jev` or `openrouter` provider configured | `404` | +| `model` missing | `400` | +| Model on a provider without System One, or a chat, embedding, or other generation model (including through a virtual model) | `400 invalid_request_error` explaining why; the gateway logs a warning | +| A System One model sent to `/v1/chat/completions`, `/v1/responses`, or `/v1/embeddings` | `400 invalid_request_error` pointing at `/v1/systemone` | +| Upstream error, such as a malformed question | The provider's status, with its message | + +## Audit, usage, and cost + +Each call is an audit entry under its route, with the requested and resolved +model, provider, request and response bodies, guardrail outcomes, and failover +attempts; filter the audit log by the **System One** request type. Usage +records the answer's `input_tokens` and `output_tokens` under the model that +answered. OpenRouter reports its own `usage.cost`, which is recorded as the +request's cost; for hosted Jev, declare pricing on the provider (see +[Jev / Kev](/providers/jev#models-access-control-and-cost)). diff --git a/docs/docs.json b/docs/docs.json index f01997512..511eb42e5 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -143,6 +143,7 @@ "advanced/responses-compatibility", "advanced/conversations-api", "advanced/anthropic-messages-api", + "advanced/systemone-api", "advanced/extra-content", "advanced/audio-api", "advanced/images-api", diff --git a/docs/features/cache.mdx b/docs/features/cache.mdx index 75bd9f4d3..512773685 100644 --- a/docs/features/cache.mdx +++ b/docs/features/cache.mdx @@ -14,6 +14,7 @@ requests on: - `/v1/responses` - `/v1/messages` - `/v1/embeddings` +- `/v1/systemone` (and Kev's `/permute` and `/separate`) Streaming and non-streaming variants of the same request are cached independently: a streaming miss stores the raw SSE bytes and a streaming hit @@ -37,7 +38,8 @@ X-Cache: HIT (semantic) represent the exact text it was requested for, so replaying the vector of a merely similar input would be a wrong answer rather than an equivalent one. Embeddings requests are also never streamed, so only the JSON response is - cached. + cached. The same holds for [System One](/advanced/systemone-api#caching) + decisions: a similar state is not the same decision. ## Enable the exact cache diff --git a/docs/openapi.json b/docs/openapi.json index efa2de15f..a6cb55a95 100644 --- a/docs/openapi.json +++ b/docs/openapi.json @@ -11200,6 +11200,92 @@ "systemone" ], "summary": "Evaluate a System One decision request (Jev / Kev)", + "requestBody": { + "$ref": "#/components/requestBodies/Request2" + }, + "responses": { + "200": { + "description": "System One answers, in the provider's shape", + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + } + }, + "400": { + "description": "Bad Request", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "401": { + "description": "Unauthorized", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "404": { + "description": "Not Found", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "429": { + "description": "Too Many Requests", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "502": { + "description": "Bad Gateway", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ], + "x-mint": { + "metadata": { + "sidebarTitle": "/v1/systemone", + "title": "Evaluate a System One decision request (Jev / Kev)", + "description": "GoModel API reference for POST /v1/systemone: Evaluate a System One decision request (Jev / Kev)." + } + } + } + }, + "/v1/systemone/permute": { + "post": { + "description": "A Kev server diagnostic: the request is a System One request, and n_perm (1 to 64, default 6) sets how many option orders run. Only jev providers pointing at a Kev server serve it.", + "tags": [ + "systemone" + ], + "summary": "Run one Choice question with several option orders (Kev)", "requestBody": { "content": { "application/json": { @@ -11208,12 +11294,12 @@ } } }, - "description": "System One request: model, state, and questions", + "description": "System One request with one Choice question", "required": true }, "responses": { "200": { - "description": "System One answers, in the provider's shape", + "description": "Kev's answer, in the provider's shape", "content": { "application/json": { "schema": { @@ -11280,9 +11366,95 @@ ], "x-mint": { "metadata": { - "sidebarTitle": "/v1/systemone", - "title": "Evaluate a System One decision request (Jev / Kev)", - "description": "GoModel API reference for POST /v1/systemone: Evaluate a System One decision request (Jev / Kev)." + "sidebarTitle": "/v1/systemone/permute", + "title": "Run one Choice question with several option orders (Kev)", + "description": "GoModel API reference for POST /v1/systemone/permute: Run one Choice question with several option orders (Kev)." + } + } + } + }, + "/v1/systemone/separate": { + "post": { + "description": "A Kev server diagnostic that answers each question separately. Only jev providers pointing at a Kev server serve it.", + "tags": [ + "systemone" + ], + "summary": "Run each System One question in its own forward pass (Kev)", + "requestBody": { + "$ref": "#/components/requestBodies/Request2" + }, + "responses": { + "200": { + "description": "Kev's answer, in the provider's shape", + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + } + }, + "400": { + "description": "Bad Request", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "401": { + "description": "Unauthorized", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "404": { + "description": "Not Found", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "429": { + "description": "Too Many Requests", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + }, + "502": { + "description": "Bad Gateway", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/core.OpenAIErrorEnvelope" + } + } + } + } + }, + "security": [ + { + "BearerAuth": [] + } + ], + "x-mint": { + "metadata": { + "sidebarTitle": "/v1/systemone/separate", + "title": "Run each System One question in its own forward pass (Kev)", + "description": "GoModel API reference for POST /v1/systemone/separate: Run each System One question in its own forward pass (Kev)." } } } @@ -11452,6 +11624,17 @@ }, "description": "Anthropic Messages request", "required": true + }, + "Request2": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + } + }, + "description": "System One request: model, state, and questions", + "required": true } }, "securitySchemes": { diff --git a/docs/providers/jev.mdx b/docs/providers/jev.mdx index ca23eabce..c983f5485 100644 --- a/docs/providers/jev.mdx +++ b/docs/providers/jev.mdx @@ -132,75 +132,23 @@ providers, name the model with its provider (`kev/kev-latest`) or a The SDKs' model listing expects TypeSafe's shape, while the gateway's `/v1/models` is OpenAI-shaped; list upstream models at `/p/jev/v1/models`. -## The native endpoint +## System One API -`POST /v1/systemone` is a gateway endpoint, not a raw proxy. For each request -GoModel: +`POST /v1/systemone` is a gateway endpoint, not a raw proxy: it applies +virtual models, guardrails on `state`, the response cache, failover, audit, +and usage, and it forwards the request natively without translating it. Kev's +`/v1/systemone/permute` and `/v1/systemone/separate` work the same way. See +[System One API](/advanced/systemone-api) for the full behavior, including +OpenRouter, which serves Jev natively too. -1. Resolves `model` like any other endpoint: a bare name, a provider-qualified - name (`jev/jev-latest`), or a virtual model, then applies the caller's - model allowlist, rate limits, and budgets. -2. Runs the workflow's prompt [guardrails](/advanced/guardrails) over `state`. -3. Forwards the body with only `model` (to the resolved name) and `state` (if - a guardrail edited it) changed. Questions, criteria, and every other field - reach the provider byte for byte. -4. Relays the answer unchanged, and records it in the audit log (as a - **System One** request) and in usage. - -The endpoint never translates. A model that cannot answer System One fails -with `400 invalid_request_error` explaining why, and the gateway logs a -warning: one on a provider without the API, or one the catalog lists as a -chat, embedding, or other generation model, such as a virtual model pointing -at a chat model. Without a `jev` or `openrouter` provider, the route answers -`404`. - -Response caching and failover do not apply to this endpoint yet. - -### Through OpenRouter - -[OpenRouter serves Jev natively](https://openrouter.ai/docs/guides/community/jev) -at the same path, so an OpenRouter key alone is enough: - -```bash -OPENROUTER_API_KEY=sk-or-... -``` - -OpenRouter's decision models appear in `GET /v1/models` as utility models, -priced from OpenRouter's listing: `openrouter/typesafe/jev-1.13`, -`openrouter/~typesafe/jev-latest` (tracks the newest Jev), and other decision -models such as Kev 4B (`openrouter/jaredpalmer/kev-4b`). Name them that way in -`model`. OpenRouter accepts `jev-latest` itself, but GoModel routes on its -catalog IDs, so to keep the TypeSafe SDK's plain `jev-latest` working, add a -[virtual model](/features/virtual-models) `jev-latest` that targets -`openrouter/~typesafe/jev-latest`. Chat models on the same provider are -rejected, since they have no System One API. - -The answer carries OpenRouter's `id`, `provider`, and `usage.cost`, and GoModel -records that reported cost with the request's usage. With a `jev` provider -configured as well, one virtual model can front a local Kev server and -OpenRouter's Jev together. - -### Guardrails - -Guardrails see `state` as a single user message: a string state as its text, -any other JSON value as its encoded JSON (which must still be valid JSON after -an edit). This is what anonymizing or blocking guardrails need. The questions -are your application's fixed schema and are not exposed. - -Guardrail edits a decision request has no place for, such as a system prompt -injected by a workflow that also covers chat models, are dropped with a -warning in the logs rather than failing the request. A guardrail that would -answer the request itself blocks it instead, since System One callers expect -typed answers, not text. - -## Native routes +Every other upstream route is reachable through +[passthrough](/features/passthrough-api), without virtual models, guardrails, +or caching: | Route | What it does | | --- | --- | -| `POST /p/jev/v1/systemone` | Evaluate a state against a map of questions, without virtual models or guardrails | | `GET /p/jev/v1/models` | The names the `model` field accepts, in the upstream's own shape | -| `POST /p/jev/v1/systemone/permute` | Kev only: run one Choice question with several option orders | -| `POST /p/jev/v1/systemone/separate` | Kev only: run each question in its own forward pass | +| `POST /p/jev/v1/systemone` | The evaluation route, forwarded as sent | Upstream errors keep their status code, with the provider's body carried in the gateway error message: a malformed question comes back as TypeSafe's `422` @@ -219,10 +167,8 @@ Every System One request names its model, so both `/v1/systemone` and the passthrough surface apply the caller's [model allowlist](/features/users) to it like any other request. -`/v1/systemone` routes only to models in the catalog. To pin a version the -upstream does not list, such as `jev-1.13.0`, declare it under the provider's -`models` (as in the pricing example below) and set -`CONFIGURED_PROVIDER_MODELS_MODE=merge`; passthrough accepts any name. +A pinned version such as `jev-1.13.0` works without being declared; see +[Models](/advanced/systemone-api#models) for how unlisted names are routed. The response's `usage.input_tokens` and `usage.output_tokens` are recorded, so System One calls appear in the usage API and dashboard under the model that diff --git a/internal/core/endpoint_operations.go b/internal/core/endpoint_operations.go index 8bb2b1395..26ac74342 100644 --- a/internal/core/endpoint_operations.go +++ b/internal/core/endpoint_operations.go @@ -29,7 +29,7 @@ var operationPaths = map[Operation]OperationPaths{ "/v1/realtime/translations", "/v1/realtime/translations/calls", "/v1/realtime/translations/client_secrets", }}, OperationMCP: {Prefixes: []string{"/mcp"}}, - OperationSystemOne: {Exact: []string{"/v1/systemone"}}, + OperationSystemOne: {Exact: []string{"/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate"}}, OperationProviderPassthrough: {Prefixes: []string{"/p"}}, } diff --git a/internal/core/endpoints.go b/internal/core/endpoints.go index 386d3c726..0e2edc494 100644 --- a/internal/core/endpoints.go +++ b/internal/core/endpoints.go @@ -170,10 +170,11 @@ func describeEndpointPath(path string) EndpointDescriptor { Dialect: "openai_compat", Operation: OperationImageEdits, } - case path == "/v1/systemone": - // TypeSafe's System One decision API (Jev, Kev). It has no canonical - // translation: the body is forwarded to a System One provider - // unchanged, apart from the routed model and guardrail edits to state. + case path == "/v1/systemone" || path == "/v1/systemone/permute" || path == "/v1/systemone/separate": + // TypeSafe's System One decision API (Jev, Kev) and the diagnostic + // variants Kev servers add. It has no canonical translation: the body + // is forwarded to a System One provider unchanged, apart from the + // routed model and guardrail edits to state. return EndpointDescriptor{ ModelInteraction: true, IngressManaged: true, diff --git a/internal/core/endpoints_test.go b/internal/core/endpoints_test.go index b881965de..3fd2b0046 100644 --- a/internal/core/endpoints_test.go +++ b/internal/core/endpoints_test.go @@ -43,7 +43,9 @@ func TestDescribeEndpointPath(t *testing.T) { {path: "/mcp", managed: false, dialect: "mcp", operation: OperationMCP, bodyMode: BodyModeNone, interaction: true}, {path: "/mcp/linear", managed: false, dialect: "mcp", operation: OperationMCP, bodyMode: BodyModeNone, interaction: true}, {path: "/v1/systemone", managed: true, dialect: "systemone", operation: OperationSystemOne, bodyMode: BodyModeJSON, interaction: true}, - {path: "/v1/systemone/permute", managed: false, dialect: "", operation: "", bodyMode: BodyModeNone, interaction: false}, + {path: "/v1/systemone/permute", managed: true, dialect: "systemone", operation: OperationSystemOne, bodyMode: BodyModeJSON, interaction: true}, + {path: "/v1/systemone/separate", managed: true, dialect: "systemone", operation: OperationSystemOne, bodyMode: BodyModeJSON, interaction: true}, + {path: "/v1/systemone/other", managed: false, dialect: "", operation: "", bodyMode: BodyModeNone, interaction: false}, {path: "/p/openai/responses", managed: true, dialect: "provider_passthrough", operation: OperationProviderPassthrough, bodyMode: BodyModeOpaque, interaction: true}, {path: "/v1/models", managed: false, dialect: "", operation: "", bodyMode: BodyModeNone, interaction: false}, } diff --git a/internal/gateway/failover.go b/internal/gateway/failover.go index 370db8477..f20951283 100644 --- a/internal/gateway/failover.go +++ b/internal/gateway/failover.go @@ -42,6 +42,7 @@ func tryFailoverResponse[T any]( workflow *core.Workflow, model, provider string, primaryErr error, + eligible func(selector core.ModelSelector, providerType string) bool, call func(selector core.ModelSelector, providerType, providerName string) (T, string, error), ) (T, ExecutionMeta, error) { var zero T @@ -75,6 +76,16 @@ func tryFailoverResponse[T any]( qualified := selector.QualifiedModel() providerType := o.ProviderTypeForSelector(selector, ProviderTypeFromWorkflow(workflow)) providerName := ResolvedProviderName(o.provider, selector, ProviderNameFromWorkflow(workflow)) + // A target that cannot serve the request is skipped before it counts + // against the attempt cap, so it never crowds out a later valid one. + if eligible != nil && !eligible(selector, providerType) { + slog.Info("skipping failover target that cannot serve the request", + "request_id", requestID, + "to", qualified, + "provider_type", providerType, + ) + continue + } if o.routeGate != nil && !o.routeGate.RouteAvailable(providerName, qualified) { slog.Info("skipping rate-limited failover target", "request_id", requestID, @@ -121,13 +132,14 @@ func executeWithFailoverResponse[T any]( workflow *core.Workflow, model, provider string, primary func() (T, string, string, error), + eligible func(selector core.ModelSelector, providerType string) bool, failoverFn func(selector core.ModelSelector, providerType, providerName string) (T, string, error), ) (T, ExecutionMeta, error) { resp, resolvedProviderType, resolvedProviderName, err := primary() if err == nil { return resp, ExecutionMeta{ProviderType: resolvedProviderType, ProviderName: resolvedProviderName}, nil } - return tryFailoverResponse(ctx, o, workflow, model, provider, err, failoverFn) + return tryFailoverResponse(ctx, o, workflow, model, provider, err, eligible, failoverFn) } func executeTranslatedWithFailover[Req any, Resp any]( @@ -158,6 +170,7 @@ func executeTranslatedWithFailover[Req any, Resp any]( } return resp, ResponseProviderType(ProviderTypeFromWorkflow(workflow), responseProvider), ProviderNameFromWorkflow(workflow), nil }, + nil, func(selector core.ModelSelector, providerType, providerName string) (Resp, string, error) { // A failover target gets a different request body, so it must not // reuse the client's idempotency key. @@ -255,3 +268,57 @@ func firstNonEmptyString(values ...string) string { } return "" } + +// PassthroughCall sends one native request to selector's provider. Provider +// error statuses must come back as errors so the failover policy can judge +// them. +type PassthroughCall func(ctx context.Context, selector core.ModelSelector, providerType, providerName string) (*core.PassthroughResponse, error) + +// ExecutePassthroughWithFailover runs a native, untranslated request against +// the workflow's resolved route and then, while the failover policy allows, +// against its failover targets. It is the native-endpoint counterpart of the +// translated failover path: attempts are recorded the same way, but every +// target receives the client's own dialect, so eligible must reject a +// failover target that cannot serve it; rejected targets are skipped without +// counting against the attempt cap. The selector that answered is returned +// with the response. +func (o *InferenceOrchestrator) ExecutePassthroughWithFailover(ctx context.Context, workflow *core.Workflow, eligible func(selector core.ModelSelector, providerType string) bool, call PassthroughCall) (*core.PassthroughResponse, core.ModelSelector, ExecutionMeta, error) { + primary := core.ModelSelector{} + if workflow != nil && workflow.Resolution != nil { + primary = workflow.Resolution.ResolvedSelector + } + type answer struct { + resp *core.PassthroughResponse + selector core.ModelSelector + } + result, meta, err := executeWithFailoverResponse(ctx, o, workflow, primary.Model, primary.Provider, + func() (answer, string, string, error) { + started := time.Now() + providerType, providerName := ProviderTypeFromWorkflow(workflow), ProviderNameFromWorkflow(workflow) + qualified := primary.QualifiedModel() + // A rate-saturated primary route must not reach the provider; its + // stored 429 becomes the primary failure that starts the sweep. + if saturated := core.PrimaryRouteSaturated(ctx); saturated != nil { + recordProviderAttempt(ctx, providerAttemptFromResult(AttemptKindPrimary, providerType, providerName, qualified, started, saturated)) + return answer{}, "", "", saturated + } + resp, err := call(ctx, primary, providerType, providerName) + recordProviderAttempt(ctx, providerAttemptFromResult(AttemptKindPrimary, providerType, providerName, qualified, started, err)) + if err != nil { + return answer{}, "", "", err + } + return answer{resp: resp, selector: primary}, providerType, providerName, nil + }, + eligible, + func(selector core.ModelSelector, providerType, providerName string) (answer, string, error) { + // A failover target gets a different body, so it must not reuse + // the client's idempotency key. + resp, err := call(core.WithIdempotencyKey(ctx, ""), selector, providerType, providerName) + if err != nil { + return answer{}, "", err + } + return answer{resp: resp, selector: selector}, providerType, nil + }, + ) + return result.resp, result.selector, meta, err +} diff --git a/internal/gateway/failover_policy_test.go b/internal/gateway/failover_policy_test.go index fdf4f9176..cdb96d676 100644 --- a/internal/gateway/failover_policy_test.go +++ b/internal/gateway/failover_policy_test.go @@ -92,7 +92,7 @@ func TestTryFailoverResponseHonorsMaxAttempts(t *testing.T) { return "", "", core.NewProviderError("openai", http.StatusBadGateway, selector.Model+" down", nil) } - _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.False(t, meta.UsedFailover) require.Error(t, err) @@ -104,6 +104,25 @@ func TestTryFailoverResponseHonorsMaxAttempts(t *testing.T) { } } +// Targets that cannot serve the request (a chat model in a System One chain) +// are skipped before a call, so they do not consume attempts either. +func TestTryFailoverResponseIneligibleTargetsDoNotConsumeAttempts(t *testing.T) { + o, workflow := threeTargetFixture(&FailoverPolicy{MaxAttempts: 1}) + primaryErr := core.NewProviderError("openai", http.StatusBadGateway, "primary down", nil) + eligible := func(selector core.ModelSelector, _ string) bool { return selector.Model != "a" } + var calls []string + call := func(selector core.ModelSelector, _, _ string) (string, string, error) { + calls = append(calls, selector.QualifiedModel()) + return "ok", "openai", nil + } + + _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, eligible, call) + + require.NoError(t, err) + require.True(t, meta.UsedFailover) + require.Equal(t, []string{"openai/b"}, calls) +} + // Targets skipped before a call (rate-limited routes) do not consume attempts. func TestTryFailoverResponseMaxAttemptsCountsCallsOnly(t *testing.T) { o, workflow := threeTargetFixture(&FailoverPolicy{MaxAttempts: 1}) @@ -115,7 +134,7 @@ func TestTryFailoverResponseMaxAttemptsCountsCallsOnly(t *testing.T) { return "ok", "openai", nil } - _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.True(t, meta.UsedFailover) require.NoError(t, err) @@ -150,7 +169,7 @@ func TestTryFailoverResponseSkipsWhenPolicyDoesNotMatch(t *testing.T) { return "ok", "openai", nil } - _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + _, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.False(t, called) require.False(t, meta.UsedFailover) diff --git a/internal/gateway/failover_test.go b/internal/gateway/failover_test.go index db345517e..f6ac6e541 100644 --- a/internal/gateway/failover_test.go +++ b/internal/gateway/failover_test.go @@ -48,7 +48,7 @@ func TestTryFailoverResponseSkipsWhenContextCanceled(t *testing.T) { return "", "", core.NewProviderError("openai", http.StatusBadGateway, "unexpected failover call", nil) } - _, meta, err := tryFailoverResponse(ctx, o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + _, meta, err := tryFailoverResponse(ctx, o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.False(t, called) require.False(t, meta.UsedFailover) @@ -66,7 +66,7 @@ func TestTryFailoverResponseAttemptsWhenContextLive(t *testing.T) { return "ok", "openai", nil } - resp, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + resp, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.True(t, called) require.True(t, meta.UsedFailover) @@ -100,7 +100,7 @@ func TestTryFailoverResponseSkipsRateLimitedTargets(t *testing.T) { return "ok", "anthropic", nil } - resp, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, call) + resp, meta, err := tryFailoverResponse(context.Background(), o, workflow, "openai/gpt-4o", "openai", primaryErr, nil, call) require.Len(t, attempted, 1) require.Equal(t, "anthropic/claude", attempted[0]) diff --git a/internal/guardrails/workflow_executor.go b/internal/guardrails/workflow_executor.go index 82783ec31..01920a1bc 100644 --- a/internal/guardrails/workflow_executor.go +++ b/internal/guardrails/workflow_executor.go @@ -3,6 +3,8 @@ package guardrails import ( "context" "log/slog" + "strings" + "sync" "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/plugins" @@ -49,9 +51,20 @@ func (p *WorkflowRequestPatcher) PatchSystemOneRequest(ctx context.Context, req return processGuarded(ctx, p.chain(ctx), req, "System One", exchange.FromSystemOneRequest, applySystemOneEdits) } +// systemOneDropsWarned records the kinds of dropped guardrail edits already +// logged at warning level. A guardrail scoped to every model edits every +// System One request the same way, so the misconfiguration is reported once +// per kind; repeats are logged at debug level. +var systemOneDropsWarned sync.Map + func applySystemOneEdits(req *core.SystemOneRequest, prompt *pluginapi.Prompt) (*core.SystemOneRequest, error) { if dropped := exchange.SystemOneUncarriedEdits(prompt); len(dropped) > 0 { - slog.Warn("guardrail edits a System One request cannot carry were dropped; only the state is guarded", "model", req.Model, "dropped", dropped) + const message = "guardrail edits a System One request cannot carry were dropped; only the state is guarded" + if _, repeated := systemOneDropsWarned.LoadOrStore(strings.Join(dropped, ","), struct{}{}); repeated { + slog.Debug(message, "model", req.Model, "dropped", dropped) + } else { + slog.Warn(message+" (repeats are logged at debug level)", "model", req.Model, "dropped", dropped) + } } return exchange.ApplyToSystemOneRequest(req, prompt) } diff --git a/internal/plugins/exchange/systemone_request.go b/internal/plugins/exchange/systemone_request.go index 73ad3ad33..eaef9b685 100644 --- a/internal/plugins/exchange/systemone_request.go +++ b/internal/plugins/exchange/systemone_request.go @@ -86,21 +86,31 @@ func ApplyToSystemOneRequest(original *core.SystemOneRequest, p *pluginapi.Promp return &result, nil } -// SystemOneUncarriedEdits describes the prompt edits ApplyToSystemOneRequest -// does not apply, in a stable order, or nil when every edit was carried. +// SystemOneUncarriedEdits describes the kinds of prompt edits +// ApplyToSystemOneRequest does not apply ("inserted message", parameter +// "temperature"), deduplicated and sorted, or nil when every edit was +// carried. Message IDs are left out: they differ per request and name +// nothing an operator can act on. func SystemOneUncarriedEdits(p *pluginapi.Prompt) []string { if p == nil { return nil } changes := p.Changes() - var uncarried []string + seen := map[string]struct{}{} for id, kind := range changes.Messages { if id != SystemOneStateMessageID { - uncarried = append(uncarried, fmt.Sprintf("%s message %q", kind, id)) + seen[string(kind)+" message"] = struct{}{} } } for name := range changes.Params { - uncarried = append(uncarried, fmt.Sprintf("parameter %q", name)) + seen[fmt.Sprintf("parameter %q", name)] = struct{}{} + } + if len(seen) == 0 { + return nil + } + uncarried := make([]string, 0, len(seen)) + for description := range seen { + uncarried = append(uncarried, description) } sort.Strings(uncarried) return uncarried diff --git a/internal/plugins/exchange/systemone_request_test.go b/internal/plugins/exchange/systemone_request_test.go index 62d5ca08b..4cf0d4078 100644 --- a/internal/plugins/exchange/systemone_request_test.go +++ b/internal/plugins/exchange/systemone_request_test.go @@ -121,7 +121,9 @@ func TestSystemOneUncarriedEdits(t *testing.T) { require.NoError(t, p.SetText(SystemOneStateMessageID, 0, "[PERSON]")) assert.Nil(t, SystemOneUncarriedEdits(p), "a state edit is carried") - id := p.Insert(0, pluginapi.TextMessage(pluginapi.RoleSystem, "be safe")) + p.Insert(0, pluginapi.TextMessage(pluginapi.RoleSystem, "be safe")) + p.Append(pluginapi.TextMessage(pluginapi.RoleSystem, "be brief")) p.SetParam("temperature", 0.1) - assert.Equal(t, []string{`inserted message "` + id + `"`, `parameter "temperature"`}, SystemOneUncarriedEdits(p)) + assert.Equal(t, []string{"inserted message", `parameter "temperature"`}, SystemOneUncarriedEdits(p), + "kinds are listed once, without per-request message IDs") } diff --git a/internal/providers/registry_normalization_test.go b/internal/providers/registry_normalization_test.go index 0af8eb32b..3d29b31d5 100644 --- a/internal/providers/registry_normalization_test.go +++ b/internal/providers/registry_normalization_test.go @@ -365,3 +365,20 @@ func TestRouterLookupModel(t *testing.T) { _, ok = (&Router{}).LookupModel("openrouter/~typesafe/jev-latest") assert.False(t, ok, "a lookup without single-model access describes nothing") } + +// ProviderNamesForType lists every configured instance of one type, so a +// caller can tell a single jev provider from several. +func TestRouterProviderNamesForType(t *testing.T) { + registry := newTestRegistryWithModels( + registryModelEntry{provider: &mockProvider{name: "kev"}, providerName: "kev", providerType: "jev", modelID: "kev-latest"}, + registryModelEntry{provider: &mockProvider{name: "jev"}, providerName: "jev", providerType: "jev", modelID: "jev-latest"}, + registryModelEntry{provider: &mockProvider{name: "openrouter"}, providerName: "openrouter", providerType: "openrouter", modelID: "typesafe/jev-1.13"}, + ) + router, err := NewRouter(registry) + require.NoError(t, err) + + assert.Equal(t, []string{"jev", "kev"}, router.ProviderNamesForType("jev")) + assert.Equal(t, []string{"openrouter"}, router.ProviderNamesForType("openrouter")) + assert.Empty(t, router.ProviderNamesForType("anthropic")) + assert.Empty(t, router.ProviderNamesForType("")) +} diff --git a/internal/providers/router_models.go b/internal/providers/router_models.go index b49d87f70..80779bb49 100644 --- a/internal/providers/router_models.go +++ b/internal/providers/router_models.go @@ -196,3 +196,20 @@ func (r *Router) LookupModel(model string) (*core.Model, bool) { cloned := info.Model return &cloned, true } + +// ProviderNamesForType lists the configured provider instance names of one +// type, sorted, or nil when the lookup cannot enumerate its providers. +func (r *Router) ProviderNamesForType(providerType string) []string { + providerType = strings.TrimSpace(providerType) + if providerType == "" || r.caps.nameLister == nil { + return nil + } + var names []string + for _, name := range r.caps.nameLister.ProviderNames() { + if r.GetProviderTypeForName(name) == providerType { + names = append(names, name) + } + } + sort.Strings(names) + return names +} diff --git a/internal/server/http.go b/internal/server/http.go index f6e4dab94..3815dff75 100644 --- a/internal/server/http.go +++ b/internal/server/http.go @@ -499,6 +499,8 @@ func New(provider core.RoutableProvider, cfg *Config) *Server { // System One decisions (Jev / Kev). The handler answers 404 until a jev // or openrouter provider is configured. e.POST("/v1/systemone", handler.SystemOne) + e.POST("/v1/systemone/permute", handler.SystemOnePermute) + e.POST("/v1/systemone/separate", handler.SystemOneSeparate) if cfg == nil || cfg.RealtimeEnabled { e.GET("/v1/realtime", handler.Realtime) e.POST("/v1/realtime/calls", handler.RealtimeCalls) diff --git a/internal/server/model_validation.go b/internal/server/model_validation.go index 52b54be42..dbcd57b09 100644 --- a/internal/server/model_validation.go +++ b/internal/server/model_validation.go @@ -103,12 +103,14 @@ func deriveWorkflowWithPolicy( } return workflow, nil - case core.OperationChatCompletions, core.OperationResponses, core.OperationEmbeddings, core.OperationSystemOne: - if desc.Operation == core.OperationSystemOne && !systemOneAvailable(provider) { - // The handler answers 404; resolving the model first would - // report a model error for an endpoint that is not there. - return nil, nil - } + case core.OperationSystemOne: + // The System One handler resolves the model itself: only it knows + // whether the endpoint is available (answering 404 before any model + // error) and when an unlisted pinned version may still route to a + // jev provider. + return nil, nil + + case core.OperationChatCompletions, core.OperationResponses, core.OperationEmbeddings: workflow.Mode = core.ExecutionModeTranslated if desc.BodyMode != core.BodyModeJSON { // Responses lifecycle routes (GET/DELETE /v1/responses/{id}, @@ -131,6 +133,9 @@ func deriveWorkflowWithPolicy( } return workflow, nil } + if systemOneOnlyModel(provider, resolution) { + return nil, systemOneOnlyModelError(desc.Operation, resolution) + } return translatedWorkflow(c.Request().Context(), requestID, desc, resolution, policyResolver) default: diff --git a/internal/server/systemone_dispatch.go b/internal/server/systemone_dispatch.go new file mode 100644 index 000000000..e9ad381a2 --- /dev/null +++ b/internal/server/systemone_dispatch.go @@ -0,0 +1,158 @@ +package server + +import ( + "bytes" + "context" + "errors" + "io" + "net/http" + "strings" + + "github.com/labstack/echo/v5" + + "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/gateway" + "github.com/enterpilot/gomodel/internal/responsecache" +) + +// dispatchSystemOneWithCache serves a guarded System One body from the +// response cache when the workflow allows it, and forwards it otherwise. +// Only the exact layer applies: the key covers the route, the resolved model, +// the guardrail chain, and the forwarded body, and the semantic layer never +// serves these paths, since a similar state is not the same decision. +func (s *translatedInferenceService) dispatchSystemOneWithCache(c *echo.Context, route systemOneRoute, workflow *core.Workflow, body []byte) error { + dispatch := func() error { return s.dispatchSystemOne(c, route, workflow, body) } + if s.responseCache == nil || !workflow.CacheEnabled() { + return dispatch() + } + c.SetRequest(c.Request().WithContext(s.inference().WithCacheRequestContext(c.Request().Context(), workflow))) + err := s.responseCache.HandleRequest(c, body, dispatch) + if replayErr, ok := errors.AsType[*responsecache.ReplayError](err); ok { + recordCachedStreamError(c, replayErr.Err) + return nil + } + return err +} + +// dispatchSystemOne forwards the body to the resolved route, and to the +// workflow's failover targets while the failover policy allows, then relays +// the answer unchanged with audit and usage accounting. +func (s *translatedInferenceService) dispatchSystemOne(c *echo.Context, route systemOneRoute, workflow *core.Workflow, body []byte) error { + passthroughProvider, ok := s.provider.(core.RoutablePassthrough) + if !ok { + return handleError(c, core.NewInvalidRequestError("provider passthrough is not supported by the current provider router", nil)) + } + // Record each provider attempt so the audit entry shows a failed primary + // and the failover that answered, as it does for chat. + c.SetRequest(c.Request().WithContext(gateway.WithAttemptRecorder(c.Request().Context()))) + s.observeLiveProviderAttempts(c, workflow) + + failovers := len(s.inference().FailoverSelectors(workflow)) + adm, err := enforceAdmission(c, s.rateLimiter, s.budgetChecker, rateLimitRouteFromWorkflow(workflow).withFailovers(failovers)) + if err != nil { + return handleError(c, err) + } + defer adm.release() + ctx := adm.dispatchContext(c.Request().Context()) + + headers := buildPassthroughHeaders(ctx, c.Request().Header) + // The client's Idempotency-Key reaches the primary through the request + // context, which failover attempts clear; forwarded as an explicit header + // it would also mark every failover target's different body. + headers.Del(core.IdempotencyKeyHeader) + eligible := func(selector core.ModelSelector, providerType string) bool { + return s.systemOneUnsupportedReason(route, selector, providerType) == "" + } + resp, executed, meta, err := s.inference().ExecutePassthroughWithFailover(ctx, workflow, eligible, + func(ctx context.Context, selector core.ModelSelector, providerType, providerName string) (*core.PassthroughResponse, error) { + return s.sendSystemOne(ctx, passthroughProvider, route, selector, providerType, providerName, headers, body) + }) + enrichAuditEntryWithProviderAttempts(c) + if err != nil { + return handleError(c, err) + } + + if meta.UsedFailover { + markRequestFailoverUsed(c) + auditlog.EnrichEntryWithFailover(c, meta.FailoverModel) + workflow = executedSystemOneWorkflow(workflow, executed, meta) + storeWorkflow(c, workflow) + } + auditlog.EnrichEntryWithWorkflow(c, workflow) + auditlog.EnrichEntryWithResolvedRoute(c, executed.QualifiedModel(), meta.ProviderType, meta.ProviderName) + info := &core.PassthroughRouteInfo{ + Provider: meta.ProviderType, + ProviderName: meta.ProviderName, + NormalizedEndpoint: route.endpoint, + SemanticOperation: route.operation, + AuditPath: route.path, + Model: executed.Model, + } + return proxyPassthroughResponse(c, s.logger, s.usageLogger, s.pricingResolver, meta.ProviderType, meta.ProviderName, route.endpoint, info, resp) +} + +// maxSystemOneErrorBodyBytes caps how much of an upstream error body is read +// to build the gateway error, so a misbehaving upstream cannot make the +// gateway buffer an unbounded body. +const maxSystemOneErrorBodyBytes = 64 << 10 + +// sendSystemOne sends the body to one target under its own model name. Only +// targets that serve the route reach it: the handler checks the primary and +// the failover sweep skips ineligible targets. An upstream error status comes +// back as an error the failover policy can judge. +func (s *translatedInferenceService) sendSystemOne( + ctx context.Context, + passthroughProvider core.RoutablePassthrough, + route systemOneRoute, + selector core.ModelSelector, + providerType, providerName string, + headers http.Header, + body []byte, +) (*core.PassthroughResponse, error) { + forwarded, err := rewriteMessagesModel(body, selector.Model) + if err != nil { + return nil, core.NewInvalidRequestError("invalid request body: "+err.Error(), err) + } + resp, err := passthroughProvider.Passthrough(ctx, providerType, &core.PassthroughRequest{ + Method: http.MethodPost, + Endpoint: route.endpoint, + Operation: route.operation, + Model: selector.Model, + Body: io.NopCloser(bytes.NewReader(forwarded)), + Headers: headers.Clone(), + ProviderName: providerName, + }) + if err != nil { + return nil, err + } + if resp == nil || resp.Body == nil { + return nil, core.NewProviderError(providerType, http.StatusBadGateway, "provider returned empty passthrough response", nil) + } + if resp.StatusCode < http.StatusBadRequest { + return resp, nil + } + defer func() { _ = resp.Body.Close() }() + errorBody, err := io.ReadAll(io.LimitReader(resp.Body, maxSystemOneErrorBodyBytes)) + if err != nil { + return nil, core.NewProviderError(providerType, http.StatusBadGateway, "failed to read provider error response", err) + } + return nil, core.ParseProviderError(providerType, resp.StatusCode, errorBody, nil) +} + +// executedSystemOneWorkflow returns a copy of workflow routed to the failover +// target that answered, so usage is priced and audited under the model that +// did the work rather than the primary that failed. +func executedSystemOneWorkflow(workflow *core.Workflow, executed core.ModelSelector, meta gateway.ExecutionMeta) *core.Workflow { + if workflow == nil || workflow.Resolution == nil { + return workflow + } + resolution := *workflow.Resolution + resolution.ResolvedSelector = executed + resolution.ProviderType = strings.TrimSpace(meta.ProviderType) + resolution.ProviderName = strings.TrimSpace(meta.ProviderName) + cloned := *workflow + cloned.ProviderType = resolution.ProviderType + cloned.Resolution = &resolution + return &cloned +} diff --git a/internal/server/systemone_dispatch_test.go b/internal/server/systemone_dispatch_test.go new file mode 100644 index 000000000..106dc5466 --- /dev/null +++ b/internal/server/systemone_dispatch_test.go @@ -0,0 +1,334 @@ +package server + +import ( + "context" + "io" + "net/http" + "slices" + "strings" + "testing" + "time" + + "github.com/goccy/go-json" + "github.com/labstack/echo/v5" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/auditlog" + "github.com/enterpilot/gomodel/internal/cache" + "github.com/enterpilot/gomodel/internal/core" + "github.com/enterpilot/gomodel/internal/echotest" + "github.com/enterpilot/gomodel/internal/gateway" + "github.com/enterpilot/gomodel/internal/responsecache" + "github.com/enterpilot/gomodel/internal/usage" +) + +// scriptedSystemOneProvider answers each passthrough by the model the body +// forwards, and records every call as " ". +type scriptedSystemOneProvider struct { + *mockProvider + // statuses maps a forwarded model to the status it answers with; + // 200 by default. + statuses map[string]int + catalog map[string]core.Model + calls []string + // idempotencyKeys records the explicit Idempotency-Key of each call. + idempotencyKeys []string +} + +// newScriptedSystemOneProvider configures the given "/" +// selectors, keyed to their provider types. +func newScriptedSystemOneProvider(models map[string]string) *scriptedSystemOneProvider { + mock := &mockProvider{providerTypes: map[string]string{}, providerNames: map[string]string{}} + for qualified, providerType := range models { + providerName, model, _ := strings.Cut(qualified, "/") + mock.supportedModels = append(mock.supportedModels, model) + mock.providerTypes[qualified] = providerType + mock.providerNames[qualified] = providerName + } + return &scriptedSystemOneProvider{mockProvider: mock, statuses: map[string]int{}} +} + +func (p *scriptedSystemOneProvider) Passthrough(_ context.Context, _ string, req *core.PassthroughRequest) (*core.PassthroughResponse, error) { + raw, err := io.ReadAll(req.Body) + if err != nil { + return nil, err + } + var sent struct { + Model string `json:"model"` + } + if err := json.Unmarshal(raw, &sent); err != nil { + return nil, err + } + p.calls = append(p.calls, req.ProviderName+" "+req.Endpoint+" "+sent.Model) + p.idempotencyKeys = append(p.idempotencyKeys, req.Headers.Get(core.IdempotencyKeyHeader)) + + status := p.statuses[sent.Model] + body := `{"error":{"message":"overloaded"}}` + if status == 0 { + status = http.StatusOK + body = `{"model":"` + sent.Model + `-answered","answers":{},"usage":{"input_tokens":10,"output_tokens":1}}` + } + return &core.PassthroughResponse{ + StatusCode: status, + Headers: map[string][]string{"Content-Type": {"application/json"}}, + Body: io.NopCloser(strings.NewReader(body)), + }, nil +} + +func (p *scriptedSystemOneProvider) LookupModel(model string) (*core.Model, bool) { + found, ok := p.catalog[model] + return &found, ok +} + +func (p *scriptedSystemOneProvider) ProviderNamesForType(providerType string) []string { + var names []string + for qualified, candidate := range p.providerTypes { + if name := p.providerNames[qualified]; candidate == providerType && !slices.Contains(names, name) { + names = append(names, name) + } + } + slices.Sort(names) + return names +} + +func systemOneRequest(model string) string { + return `{"model":"` + model + `","state":"I was charged twice.","questions":{"refund":{"type":"noul","instructions":"Refund?"}}}` +} + +// An identical request is answered from the exact cache without reaching the +// provider. +func TestSystemOne_ServesRepeatsFromTheExactCache(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{"kev/kev-latest": "jev"}) + store := cache.NewMapStore() + defer store.Close() + mw := responsecache.NewResponseCacheMiddlewareWithStore(store, time.Hour) + handler := NewHandler(provider, nil, nil, nil) + handler.responseCache = mw + + c, first := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, first.Code, first.Body.String()) + // The cache write is asynchronous; drain it before the repeat. + require.NoError(t, mw.Close()) + + c, second := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, second.Code, second.Body.String()) + + assert.Equal(t, "HIT (exact)", second.Header().Get("X-Cache")) + assert.JSONEq(t, first.Body.String(), second.Body.String()) + assert.Len(t, provider.calls, 1, "the repeat must not reach the provider") +} + +// A different state is a different decision: it misses the cache. +func TestSystemOne_CacheKeyCoversTheState(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{"kev/kev-latest": "jev"}) + store := cache.NewMapStore() + defer store.Close() + mw := responsecache.NewResponseCacheMiddlewareWithStore(store, time.Hour) + handler := NewHandler(provider, nil, nil, nil) + handler.responseCache = mw + + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + require.NoError(t, mw.Close()) + + c, rec = echotest.Post(t, "/v1/systemone", strings.Replace(systemOneRequest("kev-latest"), "twice", "once", 1)) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Empty(t, rec.Header().Get("X-Cache")) + assert.Len(t, provider.calls, 2) +} + +// When the primary fails with an availability error, the request moves to +// the virtual model's next target in its own dialect. A target without the +// System One API is skipped rather than called, and usage and audit carry the +// model that answered. +func TestSystemOne_FailsOverToTheNextSystemOneTarget(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "kev/kev-latest": "jev", + "openai/gpt-5-mini": "openai", + "openrouter/typesafe/jev-1.13": "openrouter", + }) + provider.statuses["kev-latest"] = http.StatusServiceUnavailable + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + handler := newHandler(provider, nil, usageLogger, nil, nil, nil, failoverResolverStub{selectors: []core.ModelSelector{ + {Provider: "openai", Model: "gpt-5-mini"}, + {Provider: "openrouter", Model: "typesafe/jev-1.13"}, + }}, nil) + + entry := &auditlog.LogEntry{Data: &auditlog.LogData{}} + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest("kev/kev-latest"), echotest.WithValue(string(auditlog.LogEntryKey), entry)) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, []string{"kev systemone kev-latest", "openrouter systemone typesafe/jev-1.13"}, provider.calls, + "the chat model must be skipped, not called") + require.NotNil(t, entry.Data.Failover) + assert.Equal(t, "openrouter/typesafe/jev-1.13", entry.Data.Failover.TargetModel) + assert.Equal(t, "openrouter/typesafe/jev-1.13", entry.ResolvedModel) + assert.Equal(t, "openrouter", entry.Provider) + require.Len(t, entry.Data.Attempts, 2, "a skipped target is not an attempt") + assert.Equal(t, http.StatusServiceUnavailable, entry.Data.Attempts[0].StatusCode) + assert.True(t, entry.Data.Attempts[1].Success) + + require.Len(t, usageLogger.entries, 1) + assert.Equal(t, "openrouter", usageLogger.entries[0].Provider) + assert.Equal(t, "typesafe/jev-1.13-answered", usageLogger.entries[0].Model) +} + +// A skipped target does not count against max_attempts, so a chat model ahead +// of a valid target in the chain cannot use up the only failover attempt. The +// client's Idempotency-Key is not forwarded as a header: it reaches the +// primary through the request context, and a failover target's different +// body must not carry it. +func TestSystemOne_FailoverSkipsTargetsWithoutUsingAttempts(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "kev/kev-latest": "jev", + "openai/gpt-5-mini": "openai", + "openrouter/typesafe/jev-1.13": "openrouter", + }) + provider.statuses["kev-latest"] = http.StatusServiceUnavailable + handler := newHandler(provider, nil, nil, nil, nil, nil, failoverResolverStub{selectors: []core.ModelSelector{ + {Provider: "openai", Model: "gpt-5-mini"}, + {Provider: "openrouter", Model: "typesafe/jev-1.13"}, + }}, nil) + handler.failoverPolicy = &gateway.FailoverPolicy{MaxAttempts: 1} + + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest("kev/kev-latest"), echotest.WithHeader(core.IdempotencyKeyHeader, "client-key-1")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + assert.Equal(t, []string{"kev systemone kev-latest", "openrouter systemone typesafe/jev-1.13"}, provider.calls) + assert.Equal(t, []string{"", ""}, provider.idempotencyKeys, "the key must not travel as an explicit header") +} + +// A client error such as a malformed question is not an availability +// problem: it is returned as the upstream reported it, without failover. +func TestSystemOne_DoesNotFailOverOnClientErrors(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "kev/kev-latest": "jev", + "openrouter/typesafe/jev-1.13": "openrouter", + }) + provider.statuses["kev-latest"] = http.StatusUnprocessableEntity + handler := newHandler(provider, nil, nil, nil, nil, nil, failoverResolverStub{selectors: []core.ModelSelector{ + {Provider: "openrouter", Model: "typesafe/jev-1.13"}, + }}, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest("kev/kev-latest")) + require.NoError(t, handler.SystemOne(c)) + + assert.Equal(t, http.StatusUnprocessableEntity, rec.Code, rec.Body.String()) + assert.Equal(t, []string{"kev systemone kev-latest"}, provider.calls) +} + +// Kev's diagnostic routes are served natively on jev providers and refused +// on OpenRouter, which serves only the evaluation route. +func TestSystemOne_KevDiagnosticRoutes(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "kev/kev-latest": "jev", + "openrouter/typesafe/jev-1.13": "openrouter", + }) + handler := NewHandler(provider, nil, nil, nil) + + for path, serve := range map[string]func(*echo.Context) error{ + "/v1/systemone/permute": handler.SystemOnePermute, + "/v1/systemone/separate": handler.SystemOneSeparate, + } { + t.Run(path, func(t *testing.T) { + provider.calls = nil + c, rec := echotest.Post(t, path, systemOneRequest("kev/kev-latest")) + require.NoError(t, serve(c)) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.Equal(t, []string{"kev " + strings.TrimPrefix(path, "/v1/") + " kev-latest"}, provider.calls) + + c, rec = echotest.Post(t, path, systemOneRequest("openrouter/typesafe/jev-1.13")) + require.NoError(t, serve(c)) + assert.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "this route is served by Kev servers") + }) + } +} + +// TypeSafe lists only its aliases but accepts any versioned ID, so a pinned +// version routes to a jev provider without being declared: named with its +// provider, bare when one jev provider is configured, or through a virtual +// model. A bare name stays unrouted when several jev providers could own it. +func TestSystemOne_RoutesUnlistedPinnedVersions(t *testing.T) { + aliases := systemOneAliasResolver{"pinned": {Provider: "jev", Model: "jev-1.13.0"}} + tests := []struct { + name string + models map[string]string + model string + wantCall string + wantCode int + }{ + {name: "provider-qualified", models: map[string]string{"jev/jev-latest": "jev"}, model: "jev/jev-1.13.0", wantCall: "jev systemone jev-1.13.0"}, + {name: "bare with one jev provider", models: map[string]string{"jev/jev-latest": "jev", "openrouter/typesafe/jev-1.13": "openrouter"}, model: "jev-1.13.0", wantCall: "jev systemone jev-1.13.0"}, + {name: "virtual model", models: map[string]string{"jev/jev-latest": "jev"}, model: "pinned", wantCall: "jev systemone jev-1.13.0"}, + {name: "self-hosted provider name", models: map[string]string{"kev/kev-latest": "jev"}, model: "kev/kev-4b-2026-09", wantCall: "kev systemone kev-4b-2026-09"}, + {name: "bare with two jev providers", models: map[string]string{"jev/jev-latest": "jev", "kev/kev-latest": "jev"}, model: "jev-1.13.0", wantCode: http.StatusNotFound}, + {name: "unknown on OpenRouter", models: map[string]string{"openrouter/typesafe/jev-1.13": "openrouter"}, model: "openrouter/typesafe/jev-9", wantCode: http.StatusNotFound}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + provider := newScriptedSystemOneProvider(tt.models) + handler := newHandler(provider, nil, nil, nil, aliases, nil, nil, nil) + + c, rec := echotest.Post(t, "/v1/systemone", systemOneRequest(tt.model)) + require.NoError(t, handler.SystemOne(c)) + + if tt.wantCode != 0 { + assert.Equal(t, tt.wantCode, rec.Code, rec.Body.String()) + assert.Empty(t, provider.calls) + return + } + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.Equal(t, []string{tt.wantCall}, provider.calls) + }) + } +} + +// A System One model sent to an OpenAI-compatible route is refused by the +// gateway with a pointer at /v1/systemone, from the catalog alone, so the +// caller never sees the upstream's advice to use the upstream's own endpoint. +// Chat models on the same provider are unaffected. +func TestSystemOneModels_OnOpenAIRoutesPointAtSystemOne(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{ + "openrouter/~typesafe/jev-latest": "openrouter", + "openrouter/openai/gpt-4o-mini": "openrouter", + }) + provider.catalog = map[string]core.Model{ + "openrouter/~typesafe/jev-latest": {ID: "~typesafe/jev-latest", Metadata: &core.ModelMetadata{ + Categories: []core.ModelCategory{core.CategoryUtility}, + }}, + "openrouter/openai/gpt-4o-mini": {ID: "openai/gpt-4o-mini", Metadata: &core.ModelMetadata{ + Modes: []string{"chat"}, Categories: []core.ModelCategory{core.CategoryTextGeneration}, + }}, + } + provider.response = &core.ChatResponse{ID: "c1", Object: "chat.completion", Model: "openai/gpt-4o-mini", + Choices: []core.Choice{{Message: core.ResponseMessage{Role: "assistant", Content: "ok"}, FinishReason: "stop"}}} + srv := New(provider, &Config{ModelResolver: systemOneAliasResolver{"jev-latest": {Provider: "openrouter", Model: "~typesafe/jev-latest"}}}) + + requests := map[string]string{ + "/v1/chat/completions": `{"model":"jev-latest","messages":[{"role":"user","content":"hi"}]}`, + "/v1/responses": `{"model":"openrouter/~typesafe/jev-latest","input":"hi"}`, + "/v1/embeddings": `{"model":"openrouter/~typesafe/jev-latest","input":"hi"}`, + "/v1/messages": `{"model":"jev-latest","max_tokens":8,"messages":[{"role":"user","content":"hi"}]}`, + } + for path, body := range requests { + t.Run(path, func(t *testing.T) { + rec := postJSON(t, srv, path, body) + assert.Equal(t, http.StatusBadRequest, rec.Code, rec.Body.String()) + assert.Contains(t, rec.Body.String(), "System One decision model") + assert.Contains(t, rec.Body.String(), "POST /v1/systemone") + }) + } + assert.Zero(t, provider.chatCompletionCalls) + + rec := postJSON(t, srv, "/v1/chat/completions", `{"model":"openrouter/openai/gpt-4o-mini","messages":[{"role":"user","content":"hi"}]}`) + assert.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) +} diff --git a/internal/server/systemone_handler.go b/internal/server/systemone_handler.go index b297e7fcf..6c43c262e 100644 --- a/internal/server/systemone_handler.go +++ b/internal/server/systemone_handler.go @@ -2,8 +2,8 @@ package server import ( "bytes" + "context" "fmt" - "io" "log/slog" "net/http" "slices" @@ -12,22 +12,37 @@ import ( "github.com/goccy/go-json" "github.com/labstack/echo/v5" - "github.com/enterpilot/gomodel/internal/auditlog" "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/gateway" "github.com/enterpilot/gomodel/internal/plugins" ) -const ( - systemOnePath = "/v1/systemone" - systemOneEndpoint = "systemone" +// systemOneRoute is one System One route the gateway serves natively. +type systemOneRoute struct { + // path is the gateway route; endpoint is the same route as a provider's + // passthrough spells it, without the /v1 prefix. + path string + endpoint string + // operation names the call in provider metrics and logs. + operation string + // kevOnly marks the diagnostic routes only Kev servers implement; the + // hosted API and OpenRouter serve the evaluation route alone. + kevOnly bool +} + +var ( + systemOneEvaluate = systemOneRoute{path: "/v1/systemone", endpoint: "systemone", operation: "systemone"} + systemOnePermute = systemOneRoute{path: "/v1/systemone/permute", endpoint: "systemone/permute", operation: "systemone_permute", kevOnly: true} + systemOneSeparate = systemOneRoute{path: "/v1/systemone/separate", endpoint: "systemone/separate", operation: "systemone_separate", kevOnly: true} ) +const jevProviderType = "jev" + // systemOneProviderTypes are the provider types that serve the System One API // natively: jev (TypeSafe's hosted Jev and self-hosted Kev servers) and // OpenRouter, which serves Jev and Kev at the same path with the same request // and answer shapes. Configuring either makes /v1/systemone available. -var systemOneProviderTypes = []string{"jev", "openrouter"} +var systemOneProviderTypes = []string{jevProviderType, "openrouter"} // SystemOne handles POST /v1/systemone. // @@ -52,13 +67,53 @@ var systemOneProviderTypes = []string{"jev", "openrouter"} // @Failure 502 {object} core.OpenAIErrorEnvelope // @Router /v1/systemone [post] func (h *Handler) SystemOne(c *echo.Context) error { - return h.translatedInference().SystemOne(c) + return h.translatedInference().serveSystemOne(c, systemOneEvaluate) } -// SystemOne resolves, guards, and forwards one System One request. -func (s *translatedInferenceService) SystemOne(c *echo.Context) error { +// SystemOnePermute handles POST /v1/systemone/permute. +// +// @Summary Run one Choice question with several option orders (Kev) +// @Description A Kev server diagnostic: the request is a System One request, and n_perm (1 to 64, default 6) sets how many option orders run. Only jev providers pointing at a Kev server serve it. +// @Tags systemone +// @Accept json +// @Produce json +// @Security BearerAuth +// @Param request body object true "System One request with one Choice question" +// @Success 200 {object} object "Kev's answer, in the provider's shape" +// @Failure 400 {object} core.OpenAIErrorEnvelope +// @Failure 401 {object} core.OpenAIErrorEnvelope +// @Failure 404 {object} core.OpenAIErrorEnvelope +// @Failure 429 {object} core.OpenAIErrorEnvelope +// @Failure 502 {object} core.OpenAIErrorEnvelope +// @Router /v1/systemone/permute [post] +func (h *Handler) SystemOnePermute(c *echo.Context) error { + return h.translatedInference().serveSystemOne(c, systemOnePermute) +} + +// SystemOneSeparate handles POST /v1/systemone/separate. +// +// @Summary Run each System One question in its own forward pass (Kev) +// @Description A Kev server diagnostic that answers each question separately. Only jev providers pointing at a Kev server serve it. +// @Tags systemone +// @Accept json +// @Produce json +// @Security BearerAuth +// @Param request body object true "System One request: model, state, and questions" +// @Success 200 {object} object "Kev's answer, in the provider's shape" +// @Failure 400 {object} core.OpenAIErrorEnvelope +// @Failure 401 {object} core.OpenAIErrorEnvelope +// @Failure 404 {object} core.OpenAIErrorEnvelope +// @Failure 429 {object} core.OpenAIErrorEnvelope +// @Failure 502 {object} core.OpenAIErrorEnvelope +// @Router /v1/systemone/separate [post] +func (h *Handler) SystemOneSeparate(c *echo.Context) error { + return h.translatedInference().serveSystemOne(c, systemOneSeparate) +} + +// serveSystemOne resolves, guards, and forwards one System One request. +func (s *translatedInferenceService) serveSystemOne(c *echo.Context, route systemOneRoute) error { if !systemOneAvailable(s.provider) { - return handleError(c, core.NewNotFoundError("POST "+systemOnePath+" is available only when a jev or openrouter provider is configured")) + return handleError(c, core.NewNotFoundError("POST "+route.path+" is available only when a jev or openrouter provider is configured")) } body, err := requestBodyBytes(c) if err != nil { @@ -76,26 +131,21 @@ func (s *translatedInferenceService) SystemOne(c *echo.Context) error { if err != nil { return handleError(c, err) } - ctx := c.Request().Context() resolution := workflow.Resolution if s.modelAuthorizer != nil { - if err := s.modelAuthorizer.ValidateModelAccess(ctx, resolution.ResolvedSelector); err != nil { + if err := s.modelAuthorizer.ValidateModelAccess(c.Request().Context(), resolution.ResolvedSelector); err != nil { return handleError(c, err) } } - if reason := s.systemOneUnsupportedReason(resolution); reason != "" { - return handleError(c, systemOneUnsupportedModelError(c, resolution, reason)) + if reason := s.systemOneUnsupportedReason(route, resolution.ResolvedSelector, resolution.ProviderType); reason != "" { + return handleError(c, systemOneUnsupportedModelError(c, route, resolution, reason)) } body, err = s.guardSystemOneState(c, workflow, &req, body) if err != nil { return handleError(c, err) } - model := resolution.ResolvedSelector.Model - if body, err = rewriteMessagesModel(body, model); err != nil { - return handleError(c, core.NewInvalidRequestError("invalid request body: "+err.Error(), err)) - } - return s.dispatchSystemOne(c, workflow, model, body) + return s.dispatchSystemOneWithCache(c, route, workflow, body) } // systemOneAvailable reports whether a provider that serves System One is @@ -114,17 +164,23 @@ func systemOneAvailable(provider core.RoutableProvider) bool { return false } -// systemOneWorkflow returns the request's workflow with its model resolved. -// The workflow middleware resolves it from the body; a request that reached -// the handler without one (the body was not parsed there) is resolved here, -// so virtual models and workflow policy apply either way. +// systemOneWorkflow resolves the request's model (virtual models first, then +// the catalog) and builds its workflow. A model the catalog does not list may +// still be a pinned version a jev provider accepts; see unlistedJevResolution. func (s *translatedInferenceService) systemOneWorkflow(c *echo.Context, model string) (*core.Workflow, error) { if workflow := core.GetWorkflow(c.Request().Context()); workflow != nil && workflow.Resolution != nil { return workflow, nil } - resolution, err := resolveAndStoreRequestModelResolution(c, s.provider, s.modelResolver, nil, model, "") - if err != nil { - return nil, err + requested := core.NewRequestedModelSelector(model, "") + resolution, ok := s.unlistedJevResolution(c.Request().Context(), requested) + if ok { + enrichAuditEntryWithRequestedModel(c, requested) + } else { + var err error + resolution, err = resolveAndStoreRequestModelResolution(c, s.provider, s.modelResolver, nil, model, "") + if err != nil { + return nil, err + } } workflow, err := translatedWorkflowForRequest(c, resolution, s.workflowPolicyResolver) if err != nil { @@ -134,41 +190,125 @@ func (s *translatedInferenceService) systemOneWorkflow(c *echo.Context, model st return workflow, nil } +// providerNamesByType lists configured provider instances of one type; the +// provider router implements it. +type providerNamesByType interface { + ProviderNamesForType(providerType string) []string +} + +// unlistedJevResolution routes a model the catalog does not list to a jev +// provider. TypeSafe lists only its aliases, yet accepts every versioned ID +// (jev-1.13.0) in the model field, so a pinned version must work without +// being declared first. The provider is the one the model names +// (jev/jev-1.13.0, kev/...), or for a bare name the only jev provider +// configured; virtual models apply first, so one can pin a version too. It +// is checked before the regular resolution, which would refresh the +// provider's model list on every such request; the upstream reports a name +// it rejects. +func (s *translatedInferenceService) unlistedJevResolution(ctx context.Context, requested core.RequestedModelSelector) (*core.RequestModelResolution, bool) { + selector, aliasApplied, err := gateway.ResolveExecutionSelector(ctx, s.provider, s.modelResolver, requested) + if err != nil || selector.Model == "" || s.provider.Supports(selector.QualifiedModel()) { + return nil, false + } + providerName := "" + switch named, _ := s.provider.(core.ProviderNameTypeResolver); { + case selector.Provider == "": + if lister, ok := s.provider.(providerNamesByType); ok { + if names := lister.ProviderNamesForType(jevProviderType); len(names) == 1 { + providerName = names[0] + } + } + case named != nil && named.GetProviderTypeForName(selector.Provider) == jevProviderType: + providerName = selector.Provider + case selector.Provider == jevProviderType: + if byType, ok := s.provider.(core.ProviderTypeNameResolver); ok { + providerName = byType.GetProviderNameForType(jevProviderType) + } + } + if strings.TrimSpace(providerName) == "" { + return nil, false + } + return &core.RequestModelResolution{ + Requested: requested, + ResolvedSelector: core.ModelSelector{Provider: providerName, Model: selector.Model}, + ProviderType: jevProviderType, + ProviderName: providerName, + AliasApplied: aliasApplied, + }, true +} + // modelCatalog describes single catalog models; the provider router // implements it. type modelCatalog interface { LookupModel(model string) (*core.Model, bool) } -// systemOneUnsupportedReason explains why the resolved model cannot answer a -// System One request, or returns "" when it can. The provider must serve the -// API, and since OpenRouter also serves chat models, the model must not be -// catalogued with a generation mode. A model the catalog does not describe is -// given the benefit of the doubt: the upstream reports it if it is wrong. -func (s *translatedInferenceService) systemOneUnsupportedReason(resolution *core.RequestModelResolution) string { - providerType := strings.TrimSpace(resolution.ProviderType) +// systemOneUnsupportedReason explains why a model cannot answer a request on +// route, or returns "" when it can. The provider must serve the API, a Kev +// diagnostic route needs a jev provider, and since OpenRouter also serves +// chat models, the model must not be catalogued with a generation mode. A +// model the catalog does not describe is given the benefit of the doubt: the +// upstream reports it if it is wrong. +func (s *translatedInferenceService) systemOneUnsupportedReason(route systemOneRoute, selector core.ModelSelector, providerType string) string { + providerType = strings.TrimSpace(providerType) if !slices.Contains(systemOneProviderTypes, providerType) { - return fmt.Sprintf("is served by a %s provider, which has no System One API", providerType) + return fmt.Sprintf("is served by provider type %s, which has no System One API", providerType) + } + if route.kevOnly && providerType != jevProviderType { + return fmt.Sprintf("is served by provider type %s, which answers only %s; this route is served by Kev servers", providerType, systemOneEvaluate.path) } catalog, ok := s.provider.(modelCatalog) if !ok { return "" } - model, ok := catalog.LookupModel(resolution.ResolvedQualifiedModel()) + model, ok := catalog.LookupModel(selector.QualifiedModel()) if !ok || model == nil || model.Metadata == nil || len(model.Metadata.Modes) == 0 { return "" } return fmt.Sprintf("is a %s model, not a System One model", strings.Join(model.Metadata.Modes, "/")) } +// systemOneOnlyModel reports whether the resolved model answers only System +// One requests: served by a System One provider and catalogued as a utility +// model with no generation mode, as jev models and OpenRouter's decision +// models are. The catalog is consulted rather than the provider, so the check +// holds from startup, before any provider has listed its models again. +func systemOneOnlyModel(provider core.RoutableProvider, resolution *core.RequestModelResolution) bool { + if resolution == nil || !slices.Contains(systemOneProviderTypes, strings.TrimSpace(resolution.ProviderType)) { + return false + } + catalog, ok := provider.(modelCatalog) + if !ok { + return false + } + model, ok := catalog.LookupModel(resolution.ResolvedQualifiedModel()) + if !ok || model == nil || model.Metadata == nil || len(model.Metadata.Modes) > 0 { + return false + } + return slices.Contains(model.Metadata.Categories, core.CategoryUtility) +} + +// systemOneOnlyModelError points a chat, Responses, or embeddings request for +// a System One model at /v1/systemone. Without it the caller would see the +// upstream's own advice, which names the upstream's endpoint, not the +// gateway's. +func systemOneOnlyModelError(operation core.Operation, resolution *core.RequestModelResolution) error { + surface := strings.ReplaceAll(string(operation), "_", " ") + return core.NewInvalidRequestError(fmt.Sprintf( + "model %q is a System One decision model and does not support %s; it answers decision requests, which GoModel does not translate: send them to POST %s", + resolution.RequestedQualifiedModel(), surface, systemOneEvaluate.path, + ), nil).WithParam("model") +} + // systemOneUnsupportedModelError explains a request whose model cannot answer // System One. It is also logged: a virtual model that sends System One // traffic to a chat model is an operator mistake the caller cannot fix. -func systemOneUnsupportedModelError(c *echo.Context, resolution *core.RequestModelResolution, reason string) error { +func systemOneUnsupportedModelError(c *echo.Context, route systemOneRoute, resolution *core.RequestModelResolution, reason string) error { requested := resolution.RequestedQualifiedModel() resolved := resolution.ResolvedQualifiedModel() slog.Warn("System One request routed to a model without the System One API", "request_id", requestIDFromContextOrHeader(c.Request()), + "path", route.path, "requested_model", requested, "resolved_model", resolved, "provider_type", resolution.ProviderType, @@ -180,7 +320,7 @@ func systemOneUnsupportedModelError(c *echo.Context, resolution *core.RequestMod } return core.NewInvalidRequestError(fmt.Sprintf( "model %s %s; %s forwards requests natively and does not translate them to other APIs, so use a System One model such as a jev model or OpenRouter's typesafe/jev-1.13", - target, reason, systemOnePath, + target, reason, route.path, ), nil).WithParam("model") } @@ -213,48 +353,3 @@ func (s *translatedInferenceService) guardSystemOneState(c *echo.Context, workfl } return rewritten, nil } - -// dispatchSystemOne forwards the body to the resolved provider and relays its -// answer unchanged, with admission, audit, and usage accounting. -func (s *translatedInferenceService) dispatchSystemOne(c *echo.Context, workflow *core.Workflow, model string, body []byte) error { - passthroughProvider, ok := s.provider.(core.RoutablePassthrough) - if !ok { - return handleError(c, core.NewInvalidRequestError("provider passthrough is not supported by the current provider router", nil)) - } - s.observeLiveProviderAttempts(c, workflow) - - adm, err := enforceAdmission(c, s.rateLimiter, s.budgetChecker, rateLimitRouteFromWorkflow(workflow)) - if err != nil { - return handleError(c, err) - } - defer adm.release() - ctx := adm.dispatchContext(c.Request().Context()) - - resolution := workflow.Resolution - providerType := strings.TrimSpace(resolution.ProviderType) - providerName := strings.TrimSpace(resolution.ProviderName) - resp, err := passthroughProvider.Passthrough(ctx, providerType, &core.PassthroughRequest{ - Method: http.MethodPost, - Endpoint: systemOneEndpoint, - Operation: "systemone", - Model: model, - Body: io.NopCloser(bytes.NewReader(body)), - Headers: buildPassthroughHeaders(ctx, c.Request().Header), - ProviderName: providerName, - }) - if err != nil { - return handleError(c, err) - } - - auditlog.EnrichEntryWithWorkflow(c, workflow) - auditlog.EnrichEntryWithResolvedRoute(c, resolution.ResolvedQualifiedModel(), providerType, providerName) - info := &core.PassthroughRouteInfo{ - Provider: providerType, - ProviderName: providerName, - NormalizedEndpoint: systemOneEndpoint, - SemanticOperation: "systemone", - AuditPath: systemOnePath, - Model: model, - } - return proxyPassthroughResponse(c, s.logger, s.usageLogger, s.pricingResolver, providerType, providerName, systemOneEndpoint, info, resp) -} diff --git a/internal/usage/extractor.go b/internal/usage/extractor.go index 46ed07b3d..250b74461 100644 --- a/internal/usage/extractor.go +++ b/internal/usage/extractor.go @@ -281,6 +281,9 @@ func ExtractFromCachedResponseBody( if entry == nil { entry = extractFromCachedSSEBody(body, requestID, model, provider, endpoint, pricing...) } + if entry == nil { + entry = extractFromCachedJSONBody(body, requestID, model, provider, endpoint, pricing...) + } if entry == nil { entry = &UsageEntry{ @@ -346,6 +349,31 @@ func extractFromCachedSSEBody( return observer.cachedEntry } +// extractFromCachedJSONBody reads usage from a cached JSON body of an +// endpoint without a typed response, such as a native /v1/systemone answer, +// the same way a live passthrough response is read. +func extractFromCachedJSONBody( + body []byte, + requestID, model, provider, endpoint string, + pricing ...*core.ModelPricing, +) *UsageEntry { + var payload map[string]any + if err := json.Unmarshal(body, &payload); err != nil || payload == nil { + return nil + } + observer := &StreamUsageObserver{ + model: strings.TrimSpace(model), + provider: strings.TrimSpace(provider), + requestID: strings.TrimSpace(requestID), + endpoint: endpoint, + } + if len(pricing) > 0 && pricing[0] != nil { + observer.pricingResolver = staticPricingResolver{pricing: pricing[0]} + } + observer.OnJSONEvent(payload) + return observer.cachedEntry +} + func normalizeCachedResponseEndpoint(endpoint string) string { normalized := strings.TrimSpace(endpoint) if normalized == "" { diff --git a/internal/usage/extractor_test.go b/internal/usage/extractor_test.go index 86bfed117..01cc313a8 100644 --- a/internal/usage/extractor_test.go +++ b/internal/usage/extractor_test.go @@ -493,6 +493,20 @@ func TestExtractFromCachedResponseBody(t *testing.T) { require.Equal(t, 10, entry.TotalTokens) }) + // A native endpoint without a typed response (System One) is read like a + // live passthrough answer, so a cache hit keeps its token counts. + t.Run("reads usage from an untyped JSON body", func(t *testing.T) { + body := []byte(`{"model":"jev-1.13.0","answers":{"refund":{"type":"noul","noul":0.98}},"usage":{"input_tokens":275,"output_tokens":20}}`) + + entry := ExtractFromCachedResponseBody(body, "req-systemone", "jev-latest", "jev", "/v1/systemone", CacheTypeExact) + require.NotNil(t, entry) + require.Equal(t, CacheTypeExact, entry.CacheType) + require.Equal(t, "/v1/systemone", entry.Endpoint) + require.Equal(t, "jev", entry.Provider) + require.Equal(t, 275, entry.InputTokens) + require.Equal(t, 20, entry.OutputTokens) + }) + t.Run("falls back to synthetic entry when body cannot be parsed", func(t *testing.T) { entry := ExtractFromCachedResponseBody([]byte("{"), "req-cache-fallback", "gpt-4o", "openai", "/v1/chat/completions", CacheTypeExact) require.NotNil(t, entry) diff --git a/web/dashboard/src/pages/audit-logs/audit-operations.js b/web/dashboard/src/pages/audit-logs/audit-operations.js index 7bbd43321..67da43c21 100644 --- a/web/dashboard/src/pages/audit-logs/audit-operations.js +++ b/web/dashboard/src/pages/audit-logs/audit-operations.js @@ -37,6 +37,8 @@ const EXACT_PATHS = { "/v1/images/generations": "images", "/v1/images/edits": "images", "/v1/systemone": "systemone", + "/v1/systemone/permute": "systemone", + "/v1/systemone/separate": "systemone", "/v1/realtime": "realtime", "/v1/realtime/calls": "realtime", "/v1/realtime/client_secrets": "realtime", diff --git a/web/dashboard/tests/audit-operations.test.js b/web/dashboard/tests/audit-operations.test.js index e94d99faf..2c2b7d885 100644 --- a/web/dashboard/tests/audit-operations.test.js +++ b/web/dashboard/tests/audit-operations.test.js @@ -21,7 +21,9 @@ test("auditTypeForPath mirrors the gateway endpoint classification", () => { ["/v1/audio/transcriptions?x=1", "audio"], ["/v1/images/edits/", "images"], ["/v1/systemone", "systemone"], - ["/v1/systemone/permute", ""], + ["/v1/systemone/permute", "systemone"], + ["/v1/systemone/separate/", "systemone"], + ["/v1/systemone/other", ""], ["/v1/realtime/translations/calls", "realtime"], ["/mcp", "mcp"], ["/mcp/github", "mcp"], From 12861be4633f1c2ebd654dca1a0efe669446cb06 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jakub=20A=2E=20W=C4=85sek?= Date: Sat, 26 Sep 2026 23:45:28 +0200 Subject: [PATCH 12/23] fix(jev): record System One token totals and prices, allow pinned versions in virtual models (#1099) * fix(jev): record System One token totals and prices, allow pinned versions in virtual models * test(e2e): cover Jev/Kev System One, MCP exclusions, and tool choice in the release matrix * fix(jev): prefer exact version pricing and route pinned failover targets to their provider --- internal/core/interfaces.go | 7 + internal/gateway/failover.go | 3 + .../gateway/inference_orchestrator_test.go | 25 + internal/gateway/request_model_resolution.go | 21 + internal/pricingoverrides/resolver.go | 23 + internal/pricingoverrides/service_test.go | 33 + internal/pricingoverrides/snapshot.go | 12 + internal/providers/jev/jev.go | 9 +- internal/providers/registry_lookup.go | 22 + internal/providers/registry_test.go | 27 + internal/responsecache/responsecache.go | 8 + internal/responsecache/usage_hit.go | 20 +- internal/server/systemone_dispatch_test.go | 42 + internal/usage/extractor.go | 5 + internal/usage/extractor_test.go | 14 + internal/usage/pricing.go | 53 +- internal/usage/pricing_test.go | 83 ++ internal/usage/stream_observer.go | 22 +- internal/usage/stream_observer_test.go | 35 + internal/virtualmodels/chain.go | 4 +- internal/virtualmodels/service.go | 2 +- internal/virtualmodels/types.go | 17 + .../virtualmodels/unlisted_target_test.go | 52 + tests/e2e/manage-release-e2e-stack.sh | 77 +- tests/e2e/mockjev/main.go | 348 ++++++ tests/e2e/release-e2e-scenarios.md | 1018 ++++++++++++++++- tests/e2e/run-release-e2e.sh | 6 +- 27 files changed, 1945 insertions(+), 43 deletions(-) create mode 100644 internal/usage/pricing_test.go create mode 100644 internal/virtualmodels/unlisted_target_test.go create mode 100644 tests/e2e/mockjev/main.go diff --git a/internal/core/interfaces.go b/internal/core/interfaces.go index 8e03490c6..3dd20210f 100644 --- a/internal/core/interfaces.go +++ b/internal/core/interfaces.go @@ -207,6 +207,13 @@ type MessagesTokenCounter interface { CountMessagesTokens(ctx context.Context, model string, body []byte) (int, error) } +// UnlistedModelAcceptor is implemented by providers that serve model IDs +// their listing omits, such as TypeSafe's versioned Jev IDs (jev-1.13.0), so a +// virtual model can target a provider-qualified name the catalog lacks. +type UnlistedModelAcceptor interface { + AcceptsUnlistedModels() bool +} + // ErrMessagesTokenCountUnsupported reports that the provider owning a model // has no token counting endpoint. var ErrMessagesTokenCountUnsupported = errors.New("provider has no token counting endpoint") diff --git a/internal/gateway/failover.go b/internal/gateway/failover.go index f20951283..4a1ede930 100644 --- a/internal/gateway/failover.go +++ b/internal/gateway/failover.go @@ -30,6 +30,9 @@ func (o *InferenceOrchestrator) ProviderTypeForSelector(selector core.ModelSelec if providerType := strings.TrimSpace(o.provider.GetProviderType(selector.QualifiedModel())); providerType != "" { return providerType } + if _, providerType := configuredSelectorProvider(o.provider, selector); providerType != "" { + return providerType + } if provider := strings.TrimSpace(selector.Provider); provider != "" { return provider } diff --git a/internal/gateway/inference_orchestrator_test.go b/internal/gateway/inference_orchestrator_test.go index 3e2c2e27c..63d031d97 100644 --- a/internal/gateway/inference_orchestrator_test.go +++ b/internal/gateway/inference_orchestrator_test.go @@ -8,6 +8,7 @@ import ( "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/usage" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -169,6 +170,30 @@ func TestInferenceOrchestratorProviderTypeForSelectorCanonicalizesProviderNameSe require.Equal(t, "openai", got) } +// namedProviderStub knows configured provider names and their types, but no +// catalog entry for the models under test. +type namedProviderStub struct { + providerTypeResolverStub + typesByName map[string]string +} + +func (p *namedProviderStub) GetProviderTypeForName(name string) string { return p.typesByName[name] } + +// A failover target the catalog does not list, such as a pinned jev version +// on a provider named kev, is routed to the provider its selector names, with +// that provider's type, never to the primary's provider. +func TestUnlistedSelectorResolvesToTheProviderItNames(t *testing.T) { + provider := &namedProviderStub{typesByName: map[string]string{"kev": "jev", "jev-down": "jev"}} + orchestrator := NewInferenceOrchestrator(InferenceConfig{Provider: provider}) + + pinned := core.ModelSelector{Provider: "kev", Model: "kev-4b-2026-09"} + assert.Equal(t, "jev", orchestrator.ProviderTypeForSelector(pinned, "openai")) + assert.Equal(t, "kev", ResolvedProviderName(provider, pinned, "jev-down")) + + unknown := core.ModelSelector{Provider: "nope", Model: "x"} + assert.Equal(t, "jev-down", ResolvedProviderName(provider, unknown, "jev-down")) +} + func TestQualifyModelWithProviderPrefixesSlashModelIDs(t *testing.T) { got := QualifyModelWithProvider("openai/gpt-4o-mini", "openrouter") require.Equal(t, "openrouter/openai/gpt-4o-mini", got) diff --git a/internal/gateway/request_model_resolution.go b/internal/gateway/request_model_resolution.go index da77d7cfa..fdf1ca2cb 100644 --- a/internal/gateway/request_model_resolution.go +++ b/internal/gateway/request_model_resolution.go @@ -30,9 +30,30 @@ func ResolvedProviderName(provider core.RoutableProvider, selector core.ModelSel return providerName } } + // A model the catalog does not list (a pinned jev version) still belongs to + // the provider its selector names, not to the fallback's. + if providerName, providerType := configuredSelectorProvider(provider, selector); providerType != "" { + return providerName + } return fallback } +// configuredSelectorProvider returns the provider a selector names explicitly +// and that provider's type, or empty strings when the selector names no +// configured provider. +func configuredSelectorProvider(provider core.RoutableProvider, selector core.ModelSelector) (string, string) { + providerName := strings.TrimSpace(selector.Provider) + named, ok := provider.(core.ProviderNameTypeResolver) + if providerName == "" || !ok { + return "", "" + } + providerType := strings.TrimSpace(named.GetProviderTypeForName(providerName)) + if providerType == "" { + return "", "" + } + return providerName, providerType +} + // ResolvedWorkflowProviderName returns the configured provider name recorded in a resolution. func ResolvedWorkflowProviderName(resolution *core.RequestModelResolution) string { if resolution == nil { diff --git a/internal/pricingoverrides/resolver.go b/internal/pricingoverrides/resolver.go index 78b60836c..945f2b0d8 100644 --- a/internal/pricingoverrides/resolver.go +++ b/internal/pricingoverrides/resolver.go @@ -29,6 +29,29 @@ func (s *Service) ResolvePricing(model, providerName string) *core.ModelPricing return cloneBasePricing(basePricing) } +// HasModelPricing reports whether pricing is declared for exactly this model: +// catalog pricing, or an override scoped to the model rather than to its +// provider or to every model. +func (s *Service) HasModelPricing(model, providerName string) bool { + if s == nil { + return false + } + providerName = strings.TrimSpace(providerName) + rawModel := strings.TrimSpace(model) + model = modelIDFromSelector(rawModel, providerName) + if model == "" { + return false + } + if s.snapshot().hasModelScopedOverride(providerName, model) { + return true + } + if s.base == nil { + return false + } + return s.base.ResolvePricing(model, providerName) != nil || + (rawModel != model && s.base.ResolvePricing(rawModel, providerName) != nil) +} + func cloneBasePricing(base *core.ModelPricing) *core.ModelPricing { if base == nil { return nil diff --git a/internal/pricingoverrides/service_test.go b/internal/pricingoverrides/service_test.go index e977e1455..f39659872 100644 --- a/internal/pricingoverrides/service_test.go +++ b/internal/pricingoverrides/service_test.go @@ -7,6 +7,7 @@ import ( "time" "github.com/enterpilot/gomodel/internal/core" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -353,3 +354,35 @@ func TestNormalizedRefreshIntervalClampsBelowRefreshTimeout(t *testing.T) { }) } } + +func TestServiceHasModelPricing(t *testing.T) { + baseRate := 1.0 + service, err := NewService( + newTestStore( + Override{Selector: "/", Pricing: Pricing{InputPerMtok: new(float64(10))}}, + Override{Selector: "jev/", Pricing: Pricing{InputPerMtok: new(float64(20))}}, + Override{Selector: "jev/jev-1.13.0", Pricing: Pricing{InputPerMtok: new(float64(42))}}, + Override{Selector: "kev-4b", Pricing: Pricing{InputPerMtok: new(float64(0))}}, + ), + testCatalog{providerNames: []string{"jev", "openai"}}, + selectivePricingResolver{"openai/gpt-4o": {InputPerMtok: &baseRate}}, + ) + require.NoError(t, err) + require.NoError(t, service.Refresh(context.Background())) + + tests := []struct { + model, provider string + want bool + }{ + {model: "jev-1.13.0", provider: "jev", want: true}, + {model: "jev/jev-1.13.0", provider: "jev", want: true}, + {model: "kev-4b", provider: "kev", want: true}, + {model: "gpt-4o", provider: "openai", want: true}, + // Only the provider-wide and global overrides match these. + {model: "jev-latest", provider: "jev", want: false}, + {model: "gpt-9", provider: "openai", want: false}, + } + for _, tt := range tests { + assert.Equal(t, tt.want, service.HasModelPricing(tt.model, tt.provider), "%s/%s", tt.provider, tt.model) + } +} diff --git a/internal/pricingoverrides/snapshot.go b/internal/pricingoverrides/snapshot.go index 5a4338680..c8b743b30 100644 --- a/internal/pricingoverrides/snapshot.go +++ b/internal/pricingoverrides/snapshot.go @@ -108,6 +108,18 @@ func (snap snapshot) matchingOverride(providerName, model string) (compiledOverr return compiledOverride{}, false } +// hasModelScopedOverride reports whether an override names this model, +// either for one provider or model-wide. +func (snap snapshot) hasModelScopedOverride(providerName, model string) bool { + if key := modelselectors.ExactMatchKey(providerName, model); key != "" { + if _, ok := snap.exact[key]; ok { + return true + } + } + _, ok := snap.modelWide[model] + return ok +} + func snapshotOverrides(snap snapshot) []Override { result := make([]Override, 0, len(snap.order)) for _, selector := range snap.order { diff --git a/internal/providers/jev/jev.go b/internal/providers/jev/jev.go index 93dd4acef..9f999d72c 100644 --- a/internal/providers/jev/jev.go +++ b/internal/providers/jev/jev.go @@ -42,8 +42,9 @@ type Provider struct { } var ( - _ core.Provider = (*Provider)(nil) - _ core.PassthroughProvider = (*Provider)(nil) + _ core.Provider = (*Provider)(nil) + _ core.PassthroughProvider = (*Provider)(nil) + _ core.UnlistedModelAcceptor = (*Provider)(nil) ) // New creates a Jev provider. The client is rooted at the API origin, which @@ -64,6 +65,10 @@ func New(cfg providers.ProviderConfig, opts providers.ProviderOptions) core.Prov return p } +// AcceptsUnlistedModels reports that the upstream accepts versioned IDs +// (jev-1.13.0) it does not list: TypeSafe lists only its aliases. +func (p *Provider) AcceptsUnlistedModels() bool { return true } + // SetBaseURL allows configuring a custom base URL for the provider. func (p *Provider) SetBaseURL(url string) { p.client.SetBaseURL(baseURL(url)) diff --git a/internal/providers/registry_lookup.go b/internal/providers/registry_lookup.go index 5a7b68264..64db52c2a 100644 --- a/internal/providers/registry_lookup.go +++ b/internal/providers/registry_lookup.go @@ -171,6 +171,28 @@ func (r *ModelRegistry) ModelAvailable(model string) bool { return !r.providerRuntime[info.ProviderName].inventoryStale } +// AcceptsUnlistedModel reports whether a provider-qualified model the catalog +// does not list can still be served, because its provider accepts IDs it does +// not list (see core.UnlistedModelAcceptor) and its inventory is fresh. A bare +// name never qualifies: it does not say which provider to use. +func (r *ModelRegistry) AcceptsUnlistedModel(model string) bool { + providerName, _ := splitModelSelector(strings.TrimSpace(model)) + if providerName == "" { + return false + } + r.mu.RLock() + defer r.mu.RUnlock() + + for _, provider := range r.providers { + if r.providerNames[provider] != providerName { + continue + } + acceptor, ok := provider.(core.UnlistedModelAcceptor) + return ok && acceptor.AcceptsUnlistedModels() && !r.providerRuntime[providerName].inventoryStale + } + return false +} + // GetProviderType returns the provider type string for the given model. // Returns empty string if the model is not found. func (r *ModelRegistry) GetProviderType(model string) string { diff --git a/internal/providers/registry_test.go b/internal/providers/registry_test.go index e0a6483f6..cdea37ca7 100644 --- a/internal/providers/registry_test.go +++ b/internal/providers/registry_test.go @@ -2329,3 +2329,30 @@ func TestSetModelList_ClearsETag(t *testing.T) { got := registry.currentModelListETag("https://example.test/models.min.json") require.Empty(t, got) } + +// unlistedAcceptingProvider serves model IDs it does not list, as a jev +// provider serves pinned versions. +type unlistedAcceptingProvider struct { + registryMockProvider +} + +func (p *unlistedAcceptingProvider) AcceptsUnlistedModels() bool { return true } + +func TestModelRegistryAcceptsUnlistedModel(t *testing.T) { + registry := NewModelRegistry() + registry.RegisterProviderWithNameAndType(&unlistedAcceptingProvider{}, "jev", "jev") + registry.RegisterProviderWithNameAndType(®istryMockProvider{name: "openai"}, "openai", "openai") + + tests := []struct { + model string + want bool + }{ + {model: "jev/jev-1.13.0", want: true}, + {model: "jev-1.13.0", want: false}, + {model: "openai/gpt-9", want: false}, + {model: "unknown/jev-1.13.0", want: false}, + } + for _, tt := range tests { + assert.Equal(t, tt.want, registry.AcceptsUnlistedModel(tt.model), tt.model) + } +} diff --git a/internal/responsecache/responsecache.go b/internal/responsecache/responsecache.go index c43c9832c..a12124e10 100644 --- a/internal/responsecache/responsecache.go +++ b/internal/responsecache/responsecache.go @@ -304,3 +304,11 @@ func NewResponseCacheMiddlewareWithStore(store cache.Store, ttl time.Duration) * simple: newSimpleCacheMiddleware(store, ttl, nil), } } + +// NewResponseCacheMiddlewareWithStoreAndUsage creates middleware with a custom +// store that records cache hits in usage (for testing). +func NewResponseCacheMiddlewareWithStoreAndUsage(store cache.Store, ttl time.Duration, usageLogger usage.LoggerInterface, pricingResolver usage.PricingResolver) *ResponseCacheMiddleware { + return &ResponseCacheMiddleware{ + simple: newSimpleCacheMiddleware(store, ttl, newUsageHitRecorder(usageLogger, pricingResolver)), + } +} diff --git a/internal/responsecache/usage_hit.go b/internal/responsecache/usage_hit.go index 903fc3fa4..b232cb7ec 100644 --- a/internal/responsecache/usage_hit.go +++ b/internal/responsecache/usage_hit.go @@ -4,6 +4,8 @@ import ( "log/slog" "strings" + "github.com/goccy/go-json" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/usage" ) @@ -45,10 +47,8 @@ func newUsageHitRecorder(logger usage.LoggerInterface, pricingResolver usage.Pri requestID = ex.RequestHeader(core.RequestIDHeader) } - var pricing *core.ModelPricing - if pricingResolver != nil { - pricing = pricingResolver.ResolvePricing(model, cacheHitPricingProvider(provider, providerName)) - } + pricing := usage.ResolveServedModelPricing(pricingResolver, model, cacheHitPricingProvider(provider, providerName), + func() string { return cachedAnsweredModel(body) }) entry := usage.ExtractFromCachedResponseBody(body, requestID, model, provider, endpoint, cacheType, pricing) if entry == nil { @@ -62,6 +62,18 @@ func newUsageHitRecorder(logger usage.LoggerInterface, pricingResolver usage.Pri } } +// cachedAnsweredModel returns the model a cached JSON answer names, such as +// jev-1.13.0 for a request routed to jev-latest, or "" for other bodies. +func cachedAnsweredModel(body []byte) string { + var answer struct { + Model string `json:"model"` + } + if err := json.Unmarshal(body, &answer); err != nil { + return "" + } + return answer.Model +} + func cacheHitPricingProvider(provider, providerName string) string { if name := strings.TrimSpace(providerName); name != "" { return name diff --git a/internal/server/systemone_dispatch_test.go b/internal/server/systemone_dispatch_test.go index 106dc5466..1198bf12d 100644 --- a/internal/server/systemone_dispatch_test.go +++ b/internal/server/systemone_dispatch_test.go @@ -121,6 +121,48 @@ func TestSystemOne_ServesRepeatsFromTheExactCache(t *testing.T) { assert.Len(t, provider.calls, 1, "the repeat must not reach the provider") } +// answeredModelPricing prices only the models it names, as an operator who +// declares a price for the versioned model that answers an alias. +type answeredModelPricing map[string]*core.ModelPricing + +func (r answeredModelPricing) ResolvePricing(model, _ string) *core.ModelPricing { return r[model] } + +// A cache hit is recorded in usage as an exact hit, with the tokens of the +// replayed answer, priced like the live answer: by the model that answered +// when only it carries a price. +func TestSystemOne_RecordsCacheHitsInUsage(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{"kev/kev-latest": "jev"}) + store := cache.NewMapStore() + defer store.Close() + rate := 1_000_000.0 + pricing := answeredModelPricing{"kev-latest-answered": {InputPerMtok: &rate}} + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + mw := responsecache.NewResponseCacheMiddlewareWithStoreAndUsage(store, time.Hour, usageLogger, pricing) + handler := newHandler(provider, nil, usageLogger, pricing, nil, nil, nil, nil) + handler.responseCache = mw + + c, first := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, first.Code, first.Body.String()) + require.NoError(t, mw.Close()) + + c, second := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, "HIT (exact)", second.Header().Get("X-Cache")) + + require.Len(t, usageLogger.entries, 2) + hit := usageLogger.entries[1] + assert.Equal(t, usage.CacheTypeExact, hit.CacheType) + assert.Equal(t, "/v1/systemone", hit.Endpoint) + assert.Equal(t, "jev", hit.Provider) + assert.Equal(t, 10, hit.InputTokens) + assert.Equal(t, 1, hit.OutputTokens) + for _, entry := range usageLogger.entries { + require.NotNil(t, entry.InputCost, "cache type %q", entry.CacheType) + assert.InDelta(t, 10.0, *entry.InputCost, 1e-9) + } +} + // A different state is a different decision: it misses the cache. func TestSystemOne_CacheKeyCoversTheState(t *testing.T) { provider := newScriptedSystemOneProvider(map[string]string{"kev/kev-latest": "jev"}) diff --git a/internal/usage/extractor.go b/internal/usage/extractor.go index 250b74461..732a45b5d 100644 --- a/internal/usage/extractor.go +++ b/internal/usage/extractor.go @@ -223,6 +223,11 @@ func ExtractFromSSEUsage( requestID, model, provider, endpoint string, pricing ...*core.ModelPricing, ) *UsageEntry { + // Anthropic-style usage (and System One answers) report input and output + // tokens without a total. + if totalTokens == 0 { + totalTokens = inputTokens + outputTokens + } entry := &UsageEntry{ ID: uuid.New().String(), RequestID: requestID, diff --git a/internal/usage/extractor_test.go b/internal/usage/extractor_test.go index 01cc313a8..c781d0185 100644 --- a/internal/usage/extractor_test.go +++ b/internal/usage/extractor_test.go @@ -436,6 +436,19 @@ func TestExtractFromSSEUsage(t *testing.T) { assert.Equal(t, 25, entry.RawData["cached_tokens"]) } +func TestExtractFromSSEUsageDerivesMissingTotal(t *testing.T) { + // Anthropic-style usage and System One answers carry no total_tokens. + entry := ExtractFromSSEUsage( + "", + 27, 9, 0, + nil, + "req-systemone", "jev-1.13.0", "jev", "/v1/systemone", + ) + + require.NotNil(t, entry) + assert.Equal(t, 36, entry.TotalTokens) +} + func TestExtractFromSSEUsageEmptyRawData(t *testing.T) { entry := ExtractFromSSEUsage( "chatcmpl-789", @@ -505,6 +518,7 @@ func TestExtractFromCachedResponseBody(t *testing.T) { require.Equal(t, "jev", entry.Provider) require.Equal(t, 275, entry.InputTokens) require.Equal(t, 20, entry.OutputTokens) + require.Equal(t, 295, entry.TotalTokens) }) t.Run("falls back to synthetic entry when body cannot be parsed", func(t *testing.T) { diff --git a/internal/usage/pricing.go b/internal/usage/pricing.go index db56717d4..d6c1502d2 100644 --- a/internal/usage/pricing.go +++ b/internal/usage/pricing.go @@ -1,6 +1,10 @@ package usage -import "github.com/enterpilot/gomodel/internal/core" +import ( + "strings" + + "github.com/enterpilot/gomodel/internal/core" +) // PricingResolver resolves pricing metadata for a given model and provider type. // Implementations should check the registry first and fall back to a reverse-index @@ -8,3 +12,50 @@ import "github.com/enterpilot/gomodel/internal/core" type PricingResolver interface { ResolvePricing(model, providerType string) *core.ModelPricing } + +// ModelPricingChecker is implemented by pricing resolvers that can tell +// pricing declared for exactly one model (catalog pricing or a model-scoped +// override) from a broad provider-wide or global rule. +type ModelPricingChecker interface { + HasModelPricing(model, providerType string) bool +} + +// ResolveServedModelPricing prices a response routed to one model and +// answered by another, such as the alias jev-latest answered by jev-1.13.0. +// Pricing declared for exactly the routed model wins, then pricing declared +// for exactly the answering model, then any broader rule matching the routed +// model, then the answering model. answered is called at most once, and only +// when the routed model has no exact pricing. +func ResolveServedModelPricing(resolver PricingResolver, routed, providerType string, answered func() string) *core.ModelPricing { + if resolver == nil { + return nil + } + routed = strings.TrimSpace(routed) + checker, canCheck := resolver.(ModelPricingChecker) + if canCheck && checker.HasModelPricing(routed, providerType) { + return resolver.ResolvePricing(routed, providerType) + } + + answeredModel, resolved := "", false + answeredOnce := func() string { + if !resolved && answered != nil { + if model := strings.TrimSpace(answered()); model != routed { + answeredModel = model + } + } + resolved = true + return answeredModel + } + if canCheck { + if model := answeredOnce(); model != "" && checker.HasModelPricing(model, providerType) { + return resolver.ResolvePricing(model, providerType) + } + } + if pricing := resolver.ResolvePricing(routed, providerType); pricing != nil { + return pricing + } + if model := answeredOnce(); model != "" { + return resolver.ResolvePricing(model, providerType) + } + return nil +} diff --git a/internal/usage/pricing_test.go b/internal/usage/pricing_test.go new file mode 100644 index 000000000..d9bb15f12 --- /dev/null +++ b/internal/usage/pricing_test.go @@ -0,0 +1,83 @@ +package usage + +import ( + "testing" + + "github.com/stretchr/testify/assert" + + "github.com/enterpilot/gomodel/internal/core" +) + +// checkedPricingResolver prices models from exact entries, falling back to a +// broad (provider-wide) rule, and reports which prices are exact. +type checkedPricingResolver struct { + exact map[string]*core.ModelPricing + broad *core.ModelPricing +} + +func (r checkedPricingResolver) ResolvePricing(model, _ string) *core.ModelPricing { + if pricing, ok := r.exact[model]; ok { + return pricing + } + return r.broad +} + +func (r checkedPricingResolver) HasModelPricing(model, _ string) bool { + _, ok := r.exact[model] + return ok +} + +func TestResolveServedModelPricing(t *testing.T) { + rate := func(v float64) *core.ModelPricing { return &core.ModelPricing{InputPerMtok: &v} } + routedExact, answeredExact, broad := rate(1), rate(2), rate(3) + + tests := []struct { + name string + resolver PricingResolver + want *core.ModelPricing + }{ + { + name: "exact routed price wins", + resolver: checkedPricingResolver{exact: map[string]*core.ModelPricing{"jev-latest": routedExact, "jev-1.13.0": answeredExact}, broad: broad}, + want: routedExact, + }, + { + name: "exact answering price beats a broad routed rule", + resolver: checkedPricingResolver{exact: map[string]*core.ModelPricing{"jev-1.13.0": answeredExact}, broad: broad}, + want: answeredExact, + }, + { + name: "broad rule when neither is exact", + resolver: checkedPricingResolver{broad: broad}, + want: broad, + }, + { + name: "resolver without exact checks tries routed then answering", + resolver: mapPricingResolver{"jev-1.13.0/jev": answeredExact}, + want: answeredExact, + }, + { + name: "no resolver", + want: nil, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := ResolveServedModelPricing(tt.resolver, "jev-latest", "jev", func() string { return "jev-1.13.0" }) + assert.Same(t, tt.want, got) + }) + } +} + +// The answering model is only looked up when the routed model has no exact +// price, so a cache hit does not decode its body for the common case. +func TestResolveServedModelPricingSkipsAnsweredLookupForExactRoutedPrice(t *testing.T) { + v := 1.0 + resolver := checkedPricingResolver{exact: map[string]*core.ModelPricing{"gpt-4o": {InputPerMtok: &v}}} + calls := 0 + ResolveServedModelPricing(resolver, "gpt-4o", "openai", func() string { + calls++ + return "gpt-4o-2024-08-06" + }) + assert.Zero(t, calls) +} diff --git a/internal/usage/stream_observer.go b/internal/usage/stream_observer.go index 927750078..ae8e99c65 100644 --- a/internal/usage/stream_observer.go +++ b/internal/usage/stream_observer.go @@ -169,10 +169,8 @@ func (o *StreamUsageObserver) mergeWithCachedEntry(entry *UsageEntry) *UsageEntr entry.TotalTokens = entry.InputTokens + entry.OutputTokens } var pricingArgs []*core.ModelPricing - if o.pricingResolver != nil { - if p := o.pricingResolver.ResolvePricing(o.pricingModel(entry.Model), o.pricingProvider()); p != nil { - pricingArgs = append(pricingArgs, p) - } + if p := o.resolvePricing(entry.Model); p != nil { + pricingArgs = append(pricingArgs, p) } applyUsageCosts(entry, o.provider, o.endpoint, pricingArgs...) return entry @@ -267,10 +265,8 @@ func (o *StreamUsageObserver) extractUsageFromEvent(chunk map[string]any) *Usage } var pricingArgs []*core.ModelPricing - if o.pricingResolver != nil { - if p := o.pricingResolver.ResolvePricing(o.pricingModel(model), o.pricingProvider()); p != nil { - pricingArgs = append(pricingArgs, p) - } + if p := o.resolvePricing(model); p != nil { + pricingArgs = append(pricingArgs, p) } entry := ExtractFromSSEUsage( @@ -304,6 +300,16 @@ func (o *StreamUsageObserver) pricingModel(responseModel string) string { return strings.TrimSpace(responseModel) } +// resolvePricing prices the routed model or, when only it carries pricing, +// the model that answered; see ResolveServedModelPricing. +func (o *StreamUsageObserver) resolvePricing(responseModel string) *core.ModelPricing { + if o == nil { + return nil + } + return ResolveServedModelPricing(o.pricingResolver, o.pricingModel(responseModel), o.pricingProvider(), + func() string { return responseModel }) +} + func (o *StreamUsageObserver) pricingProvider() string { if o == nil { return "" diff --git a/internal/usage/stream_observer_test.go b/internal/usage/stream_observer_test.go index 576ebdbd2..79c27c896 100644 --- a/internal/usage/stream_observer_test.go +++ b/internal/usage/stream_observer_test.go @@ -617,3 +617,38 @@ func TestStreamUsageObserverAnthropicNativeEvents(t *testing.T) { assert.Equal(t, 100, entry.RawData["cache_creation_input_tokens"]) assert.Equal(t, 200, entry.RawData["cache_read_input_tokens"]) } + +// A routed alias (jev-latest) is answered by a versioned model (jev-1.13.0); +// the routed model's price wins, and the answered model's price applies when +// only it is declared. +func TestStreamUsageObserverPricesAnsweredModelWhenRoutedHasNone(t *testing.T) { + routedRate, answeredRate := 10.0, 42.0 + routed := &core.ModelPricing{InputPerMtok: &routedRate} + answered := &core.ModelPricing{InputPerMtok: &answeredRate} + + tests := []struct { + name string + resolver mapPricingResolver + wantInput float64 + }{ + {name: "routed model priced", resolver: mapPricingResolver{"jev-latest/jev": routed, "jev-1.13.0/jev": answered}, wantInput: 10}, + {name: "only answered model priced", resolver: mapPricingResolver{"jev-1.13.0/jev": answered}, wantInput: 42}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + logger := &trackingLogger{enabled: true} + observer := NewStreamUsageObserver(logger, "jev-latest", "jev", "req-s1", "/v1/systemone", tt.resolver) + observer.OnJSONEvent(map[string]any{ + "model": "jev-1.13.0", + "usage": map[string]any{"input_tokens": float64(1_000_000), "output_tokens": float64(0)}, + }) + observer.OnStreamClose() + + entries := logger.getEntries() + require.Len(t, entries, 1) + require.NotNil(t, entries[0].InputCost) + assert.InDelta(t, tt.wantInput, *entries[0].InputCost, 1e-9) + assert.Equal(t, "jev-1.13.0", entries[0].Model) + }) + } +} diff --git a/internal/virtualmodels/chain.go b/internal/virtualmodels/chain.go index 3f2c461a9..e3c3b8e2e 100644 --- a/internal/virtualmodels/chain.go +++ b/internal/virtualmodels/chain.go @@ -66,7 +66,7 @@ func (s *snapshot) viableTargets(entry *redirectEntry, catalog Catalog) []resolv func (s *snapshot) viable(owner *redirectEntry, target resolvedTarget, catalog Catalog) bool { inner, ok := s.chained(owner.vm.Source, target) if !ok { - return catalog.ModelAvailable(target.qualified) + return modelServable(catalog, target.qualified) } if !inner.vm.Enabled { return false @@ -100,7 +100,7 @@ func (s *snapshot) leafTargets(entry *redirectEntry, catalog Catalog) []resolved func (s *snapshot) leaves(owner *redirectEntry, target resolvedTarget, catalog Catalog) []resolvedTarget { inner, ok := s.chained(owner.vm.Source, target) if !ok { - if catalog.ModelAvailable(target.qualified) { + if modelServable(catalog, target.qualified) { return []resolvedTarget{target} } return nil diff --git a/internal/virtualmodels/service.go b/internal/virtualmodels/service.go index c0a13af99..881f16e29 100644 --- a/internal/virtualmodels/service.go +++ b/internal/virtualmodels/service.go @@ -679,7 +679,7 @@ func (s *Service) firstUnsupportedTarget(current *snapshot, vm VirtualModel) (st if _, chained := current.chained(vm.Source, candidate); chained { continue } - if !s.catalog.Supports(qualified) { + if !s.catalog.Supports(qualified) && !modelServable(s.catalog, qualified) { return qualified, true } } diff --git a/internal/virtualmodels/types.go b/internal/virtualmodels/types.go index 82f7cfaa5..eb07bf91a 100644 --- a/internal/virtualmodels/types.go +++ b/internal/virtualmodels/types.go @@ -254,3 +254,20 @@ type Catalog interface { LookupModel(model string) (*core.Model, bool) ProviderNames() []string } + +// unlistedModelCatalog is implemented by catalogs whose providers serve +// provider-qualified model IDs they do not list, such as a jev provider's +// pinned versions (jev/jev-1.13.0). +type unlistedModelCatalog interface { + AcceptsUnlistedModel(model string) bool +} + +// modelServable reports whether a concrete target can serve a request now: +// listed and available, or unlisted on a provider that accepts such IDs. +func modelServable(catalog Catalog, model string) bool { + if catalog.ModelAvailable(model) { + return true + } + unlisted, ok := catalog.(unlistedModelCatalog) + return ok && unlisted.AcceptsUnlistedModel(model) +} diff --git a/internal/virtualmodels/unlisted_target_test.go b/internal/virtualmodels/unlisted_target_test.go new file mode 100644 index 000000000..8c1211855 --- /dev/null +++ b/internal/virtualmodels/unlisted_target_test.go @@ -0,0 +1,52 @@ +package virtualmodels + +import ( + "context" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/core" +) + +// unlistedCatalog is a fakeCatalog whose named providers serve IDs they do +// not list, as a jev provider serves pinned versions. +type unlistedCatalog struct { + fakeCatalog + acceptsUnlisted map[string]bool +} + +func (c unlistedCatalog) AcceptsUnlistedModel(model string) bool { + provider, _, ok := strings.Cut(model, "/") + return ok && c.acceptsUnlisted[provider] +} + +// A redirect may pin a version its provider accepts without listing it; the +// same target on a provider that lists everything it serves is still refused. +func TestService_RedirectToUnlistedModelOfAcceptingProvider(t *testing.T) { + t.Parallel() + catalog := unlistedCatalog{ + providers: []string{"openai", "jev"}, + supported: map[string]core.Model{ + "openai/gpt-4o": {ID: "openai/gpt-4o"}, + "jev/jev-latest": {ID: "jev/jev-latest"}, + }, + acceptsUnlisted: map[string]bool{"jev": true}, + } + svc, err := NewService(newSQLVMStore(t), catalog, true) + require.NoError(t, err) + ctx := context.Background() + + err = svc.Upsert(ctx, VirtualModel{Source: "pinned", Targets: []Target{{Model: "jev/jev-1.13.0"}}, Enabled: true}) + require.NoError(t, err) + selector, changed, err := svc.ResolveModel(core.NewRequestedModelSelector("pinned", "")) + require.NoError(t, err) + assert.True(t, changed) + assert.Equal(t, "jev/jev-1.13.0", selector.QualifiedModel()) + + err = svc.Upsert(ctx, VirtualModel{Source: "missing", Targets: []Target{{Model: "openai/gpt-9"}}, Enabled: true}) + require.Error(t, err) + assert.Contains(t, err.Error(), "target model not found: openai/gpt-9") +} diff --git a/tests/e2e/manage-release-e2e-stack.sh b/tests/e2e/manage-release-e2e-stack.sh index 102cb9871..2fdcf4efd 100755 --- a/tests/e2e/manage-release-e2e-stack.sh +++ b/tests/e2e/manage-release-e2e-stack.sh @@ -11,6 +11,9 @@ MONGO_DATABASE="${GOMODEL_RELEASE_MONGO_DATABASE:-gomodel_release_e2e}" MOCK_MCP_BIN="${GOMODEL_RELEASE_MOCK_MCP_BINARY:-$REPO_ROOT/bin/mockmcp}" MOCK_MCP_PORT="${GOMODEL_RELEASE_MOCK_MCP_PORT:-18090}" MOCK_MCP_TOKEN="${GOMODEL_RELEASE_MOCK_MCP_TOKEN:-qa-mock-mcp-secret}" +MOCK_JEV_BIN="${GOMODEL_RELEASE_MOCK_JEV_BINARY:-$REPO_ROOT/bin/mockjev}" +MOCK_JEV_PORT="${GOMODEL_RELEASE_MOCK_JEV_PORT:-18091}" +MOCK_JEV_KEY="${GOMODEL_RELEASE_MOCK_JEV_KEY:-qa-mock-jev-key}" BUILD_BEFORE_START=0 @@ -37,6 +40,7 @@ Gateways: Helpers: mock-mcp http://localhost:18090 (mock MCP upstream: /alpha token-gated, /beta open) + mock-jev http://localhost:18091 (mock System One upstreams: /jev keyed, /kev keyless Kev, /down always 529) EOF } @@ -99,7 +103,24 @@ load_env() { export OPENROUTER_MODEL_FILTER_INCLUDE="${OPENROUTER_MODEL_FILTER_INCLUDE:-*:free}" export XAI_MODELS="${XAI_MODELS:-grok-4.3,grok-voice-latest}" export BAILIAN_MODELS="${BAILIAN_MODELS:-qwen3-omni-flash-realtime}" - export ENABLED_PASSTHROUGH_PROVIDERS="${ENABLED_PASSTHROUGH_PROVIDERS:-openai,anthropic,openrouter,zai,vllm,deepseek,bailian,xai}" + export ENABLED_PASSTHROUGH_PROVIDERS="${ENABLED_PASSTHROUGH_PROVIDERS:-openai,anthropic,openrouter,zai,vllm,deepseek,bailian,xai,jev}" + # System One (Jev / Kev) providers backed by the local mockjev upstream: + # "jev" is hosted-shaped and keyed, "jev-kev" a keyless Kev server, and + # "jev-down" answers 529 so System One failover can be exercised. Every + # JEV_* value from .env is dropped first (suffixed keys, model lists): the + # scenarios assert what the mock echoes back, and a key left on a keyless + # provider would reach the mock. The Kev URL keeps a trailing /v1, which the + # provider trims. + local name + for name in $(compgen -e); do + if [[ "$name" == JEV_* ]]; then + unset "$name" + fi + done + export JEV_API_KEY="$MOCK_JEV_KEY" + export JEV_BASE_URL="http://localhost:$MOCK_JEV_PORT/jev" + export JEV_KEV_BASE_URL="http://localhost:$MOCK_JEV_PORT/kev/v1" + export JEV_DOWN_BASE_URL="http://localhost:$MOCK_JEV_PORT/down" } ensure_binary() { @@ -109,54 +130,62 @@ ensure_binary() { if (( BUILD_BEFORE_START == 1 )) || [[ ! -x "$MOCK_MCP_BIN" ]]; then (cd "$REPO_ROOT" && go build -o "$MOCK_MCP_BIN" ./tests/e2e/mockmcp) fi + if (( BUILD_BEFORE_START == 1 )) || [[ ! -x "$MOCK_JEV_BIN" ]]; then + (cd "$REPO_ROOT" && go build -o "$MOCK_JEV_BIN" ./tests/e2e/mockjev) + fi } -start_mock_mcp() { - local dir="$STACK_DIR/mock-mcp" +# Starts one mock upstream binary on its port and waits for /healthz. +# usage: start_mock NAME PORT BINARY [ENV=VALUE...] +start_mock() { + local name="$1" port="$2" bin="$3" + shift 3 + local dir="$STACK_DIR/$name" local log_file="$dir/logs/server.log" local pid_file="$dir/server.pid" mkdir -p "$dir/logs" if is_pid_running "$pid_file"; then - printf 'mock-mcp already running pid=%s url=http://localhost:%s\n' "$(cat "$pid_file")" "$MOCK_MCP_PORT" + printf '%s already running pid=%s url=http://localhost:%s\n' "$name" "$(cat "$pid_file")" "$port" return 0 fi # A foreign process on the port would answer the health probe and mask a - # failed bind (e.g. a manually started mockmcp with a different token). - if curl -fsS "http://localhost:$MOCK_MCP_PORT/healthz" >/dev/null 2>&1; then - die "port $MOCK_MCP_PORT is already in use by an unmanaged process; stop it before starting mock-mcp" + # failed bind (e.g. a manually started mock with different settings). + if curl -fsS "http://localhost:$port/healthz" >/dev/null 2>&1; then + die "port $port is already in use by an unmanaged process; stop it before starting $name" fi rm -f "$pid_file" ( cd "$dir" - nohup env PORT="$MOCK_MCP_PORT" MOCK_MCP_TOKEN="$MOCK_MCP_TOKEN" "$MOCK_MCP_BIN" >"$log_file" 2>&1 < /dev/null & + nohup env PORT="$port" "$@" "$bin" >"$log_file" 2>&1 < /dev/null & echo $! >"$pid_file" ) local attempt for attempt in $(seq 1 15); do - if curl -fsS "http://localhost:$MOCK_MCP_PORT/healthz" >/dev/null 2>&1; then - printf 'started mock-mcp pid=%s url=http://localhost:%s\n' "$(cat "$pid_file")" "$MOCK_MCP_PORT" + if curl -fsS "http://localhost:$port/healthz" >/dev/null 2>&1; then + printf 'started %s pid=%s url=http://localhost:%s\n' "$name" "$(cat "$pid_file")" "$port" return 0 fi sleep 1 done - echo "failed to start mock-mcp on port $MOCK_MCP_PORT" >&2 + echo "failed to start $name on port $port" >&2 [[ -f "$log_file" ]] && tail -n 40 "$log_file" >&2 exit 1 } -stop_mock_mcp() { - local pid_file="$STACK_DIR/mock-mcp/server.pid" +stop_mock() { + local name="$1" + local pid_file="$STACK_DIR/$name/server.pid" local pid if [[ ! -f "$pid_file" ]]; then - printf 'mock-mcp not running\n' + printf '%s not running\n' "$name" return 0 fi @@ -165,22 +194,23 @@ stop_mock_mcp() { kill "$pid" 2>/dev/null || true fi rm -f "$pid_file" - printf 'stopped mock-mcp\n' + printf 'stopped %s\n' "$name" } -status_mock_mcp() { - local pid_file="$STACK_DIR/mock-mcp/server.pid" +status_mock() { + local name="$1" port="$2" + local pid_file="$STACK_DIR/$name/server.pid" local health="down" local pid="stopped" if is_pid_running "$pid_file"; then pid="$(cat "$pid_file")" - if curl -fsS "http://localhost:$MOCK_MCP_PORT/healthz" >/dev/null 2>&1; then + if curl -fsS "http://localhost:$port/healthz" >/dev/null 2>&1; then health="ok" fi fi - printf '%-12s pid=%-8s url=http://localhost:%s health=%s\n' "mock-mcp" "$pid" "$MOCK_MCP_PORT" "$health" + printf '%-12s pid=%-8s url=http://localhost:%s health=%s\n' "$name" "$pid" "$port" "$health" } ensure_pg_database() { @@ -360,7 +390,8 @@ start_stack() { mkdir -p "$STACK_DIR" ensure_pg_database write_guardrail_config - start_mock_mcp + start_mock mock-mcp "$MOCK_MCP_PORT" "$MOCK_MCP_BIN" MOCK_MCP_TOKEN="$MOCK_MCP_TOKEN" + start_mock mock-jev "$MOCK_JEV_PORT" "$MOCK_JEV_BIN" MOCK_JEV_KEY="$MOCK_JEV_KEY" start_gateway sqlite-main \ -u GOMODEL_MASTER_KEY \ @@ -457,11 +488,13 @@ stop_stack() { stop_gateway mongo-smoke stop_gateway pg-smoke stop_gateway sqlite-main - stop_mock_mcp + stop_mock mock-jev + stop_mock mock-mcp } status_stack() { - status_mock_mcp + status_mock mock-mcp "$MOCK_MCP_PORT" + status_mock mock-jev "$MOCK_JEV_PORT" status_gateway sqlite-main status_gateway pg-smoke status_gateway mongo-smoke diff --git a/tests/e2e/mockjev/main.go b/tests/e2e/mockjev/main.go new file mode 100644 index 000000000..892ddb690 --- /dev/null +++ b/tests/e2e/mockjev/main.go @@ -0,0 +1,348 @@ +// Command mockjev serves deterministic TypeSafe System One upstreams for the +// release E2E curl matrix, one per path prefix, so a single process backs +// several jev providers: +// +// /jev hosted-API shape: requires "Authorization: Bearer $MOCK_JEV_KEY"; +// lists jev-latest and jev-preview by "name"; accepts any versioned +// ID (jev-1.13.0); answers jev-latest as jev-1.13.0; has no Kev +// diagnostic routes (404, like the hosted API). +// /kev Kev-server shape: no authentication; lists checkpoint kev-latest +// by "id" with alias kev-4b; serves /v1/systemone/permute and +// /v1/systemone/separate. +// /down lists kev-down but answers every System One route with 529, the +// status TypeSafe uses for overload, so failover can be exercised. +// +// Each answer carries a "mock" object echoing what reached the upstream (the +// model, state, questions, extra top-level fields, whether an Authorization +// header arrived, the X-Request-Id, and a per-upstream request sequence), so +// scenarios can assert what the gateway forwarded and whether a cached +// answer was replayed. Malformed questions get a FastAPI-style 422, as the +// hosted API returns. +// +// PORT selects the listen port (default 18091). GET /healthz reports liveness. +package main + +import ( + "bytes" + "encoding/json" + "fmt" + "log" + "net/http" + "os" + "regexp" + "sort" + "strings" + "sync" +) + +type upstream struct { + name string + key string // required bearer key; empty means no authentication + models any // GET /v1/models body + kevRoute bool // serves permute and separate + down bool // answers System One routes with 529 + accepts func(model string) (answeredAs string, ok bool) + + mu sync.Mutex + seq int +} + +// maxBodyBytes bounds a request body; the largest the matrix sends is about +// 70 KiB. +const maxBodyBytes = 1 << 20 + +var versionedJev = regexp.MustCompile(`^jev-\d+\.\d+\.\d+$`) + +func newUpstreams(jevKey string) []*upstream { + return []*upstream{ + { + name: "jev", + key: jevKey, + models: map[string]any{"models": []map[string]any{ + {"name": "jev-latest", "description": "Latest Jev (mock)", "release_date": "2026-06-01"}, + {"name": "jev-preview", "description": "Preview Jev (mock)"}, + }}, + accepts: func(model string) (string, bool) { + switch { + case model == "jev-latest": + return "jev-1.13.0", true + case model == "jev-preview": + return "jev-1.14.0-preview", true + case versionedJev.MatchString(model): + return model, true + } + return "", false + }, + }, + { + name: "kev", + kevRoute: true, + models: map[string]any{"models": []map[string]any{ + {"id": "kev-latest", "aliases": []string{"kev-4b"}, "release_date": "2026-09-01T00:00:00Z"}, + }}, + accepts: func(model string) (string, bool) { + if model == "kev-latest" || model == "kev-4b" { + return "kev-4b-e2e", true + } + return "", false + }, + }, + { + name: "down", + kevRoute: true, + down: true, + models: map[string]any{"models": []map[string]any{{"id": "kev-down"}}}, + accepts: func(string) (string, bool) { return "", false }, + }, + } +} + +// question keeps instructions and criteria as raw JSON: the SDKs accept +// structured values there, and an answer needs only option names and the +// number of score levels. +type question struct { + Type string `json:"type"` + Instructions json.RawMessage `json:"instructions"` + Criteria json.RawMessage `json:"criteria"` +} + +type request struct { + Model string `json:"model"` + State json.RawMessage `json:"state"` + Questions map[string]question `json:"questions"` + NPerm *int `json:"n_perm"` +} + +func writeJSON(w http.ResponseWriter, status int, body any) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(status) + _ = json.NewEncoder(w).Encode(body) +} + +// validationError mirrors the FastAPI 422 body the hosted API returns. +func validationError(w http.ResponseWriter, loc []any, msg string) { + writeJSON(w, http.StatusUnprocessableEntity, map[string]any{ + "detail": []map[string]any{{"loc": append([]any{"body"}, loc...), "msg": msg, "type": "value_error"}}, + }) +} + +func (u *upstream) authorized(r *http.Request) bool { + return u.key == "" || r.Header.Get("Authorization") == "Bearer "+u.key +} + +func (u *upstream) authSeen(r *http.Request) string { + switch auth := r.Header.Get("Authorization"); { + case auth == "": + return "none" + case u.key != "" && auth == "Bearer "+u.key: + return "provider-key" + default: + return "other" + } +} + +func (u *upstream) serveModels(w http.ResponseWriter, r *http.Request) { + if !u.authorized(r) { + writeJSON(w, http.StatusUnauthorized, map[string]any{"detail": "Invalid API key"}) + return + } + writeJSON(w, http.StatusOK, u.models) +} + +func (u *upstream) serveSystemOne(route string) http.HandlerFunc { + return func(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodPost { + writeJSON(w, http.StatusMethodNotAllowed, map[string]any{"detail": "Method Not Allowed"}) + return + } + if route != "evaluate" && !u.kevRoute { + writeJSON(w, http.StatusNotFound, map[string]any{"detail": "Not Found"}) + return + } + if !u.authorized(r) { + writeJSON(w, http.StatusUnauthorized, map[string]any{"detail": "Invalid API key"}) + return + } + if u.down { + writeJSON(w, 529, map[string]any{"detail": "Overloaded (mock " + u.name + ")"}) + return + } + var raw bytes.Buffer + if _, err := raw.ReadFrom(http.MaxBytesReader(w, r.Body, maxBodyBytes)); err != nil { + writeJSON(w, http.StatusRequestEntityTooLarge, map[string]any{"detail": err.Error()}) + return + } + var req request + if err := json.Unmarshal(raw.Bytes(), &req); err != nil { + validationError(w, nil, "invalid JSON: "+err.Error()) + return + } + answeredAs, ok := u.accepts(req.Model) + if !ok { + writeJSON(w, http.StatusNotFound, map[string]any{"detail": fmt.Sprintf("Model %q not found", req.Model)}) + return + } + if len(req.State) == 0 || string(req.State) == "null" { + validationError(w, []any{"state"}, "Field required") + return + } + if len(req.Questions) == 0 { + validationError(w, []any{"questions"}, "At least one question is required") + return + } + names := make([]string, 0, len(req.Questions)) + for name := range req.Questions { + names = append(names, name) + } + sort.Strings(names) + + nPerm := 0 + if route == "permute" { + if len(req.Questions) != 1 || req.Questions[names[0]].Type != "choice" { + validationError(w, []any{"questions"}, "permute takes exactly one choice question") + return + } + nPerm = 6 + if req.NPerm != nil { + nPerm = *req.NPerm + } + if nPerm < 1 || nPerm > 64 { + validationError(w, []any{"n_perm"}, "n_perm must be between 1 and 64") + return + } + } + + answers := make(map[string]any, len(names)) + for _, name := range names { + answer, msg := answerFor(req.Questions[name]) + if msg != "" { + validationError(w, []any{"questions", name}, msg) + return + } + answers[name] = answer + } + + u.mu.Lock() + u.seq++ + seq := u.seq + u.mu.Unlock() + + var top map[string]json.RawMessage + _ = json.Unmarshal(raw.Bytes(), &top) + extra := map[string]json.RawMessage{} + for key, value := range top { + switch key { + case "model", "state", "questions", "n_perm": + default: + extra[key] = value + } + } + + body := map[string]any{ + "model": answeredAs, + "answers": answers, + "usage": map[string]any{ + "input_tokens": 10 + len(req.State)/4 + 5*len(names), + "output_tokens": 3 * len(names), + }, + "mock": map[string]any{ + "upstream": u.name, + "route": route, + "received_model": req.Model, + "received_state": req.State, + "questions": top["questions"], + "extra": extra, + "authorization": u.authSeen(r), + "request_id": r.Header.Get("X-Request-Id"), + "request_seq": seq, + "received_length": raw.Len(), + }, + } + if route == "permute" { + body["n_perm"] = nPerm + } + if route == "separate" { + body["separate"] = true + } + writeJSON(w, http.StatusOK, body) + } +} + +// answerFor builds a deterministic, well-formed answer for one question, or +// returns the validation message the hosted API would reject it with. +func answerFor(q question) (any, string) { + switch q.Type { + case "noul": + return map[string]any{"type": "noul", "noul": 0.93}, "" + case "choice": + var criteria map[string]json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil || len(criteria) < 2 { + return nil, "choice criteria must map at least two option names to descriptions" + } + options := make([]string, 0, len(criteria)) + for option := range criteria { + options = append(options, option) + } + sort.Strings(options) + probabilities := make(map[string]float64, len(options)) + rest := 0.4 / float64(len(options)-1) + for i, option := range options { + probabilities[option] = rest + if i == 0 { + probabilities[option] = 0.6 + } + } + return map[string]any{"type": "choice", "choice": options[0], "confidence": 0.5, "probabilities": probabilities}, "" + case "score": + var levels []json.RawMessage + if err := json.Unmarshal(q.Criteria, &levels); err != nil || len(levels) < 2 { + return nil, "score criteria must list at least two ordered levels" + } + legend := make(map[string]any, len(levels)) + probabilities := make(map[string]float64, len(levels)) + for i, level := range levels { + var label string + if json.Unmarshal(level, &label) == nil { + legend[fmt.Sprint(i)] = label + } else { + legend[fmt.Sprint(i)] = level + } + probabilities[fmt.Sprint(i)] = 0 + } + probabilities["1"] = 1 + return map[string]any{"type": "score", "score": 1.0, "confidence": 0.8, "legend": legend, "probabilities": probabilities}, "" + case "": + return nil, "Field required: type" + default: + return nil, fmt.Sprintf("Input tag %q found using 'type' does not match any of the expected tags: 'noul', 'choice', 'score'", q.Type) + } +} + +func main() { + port := os.Getenv("PORT") + if port == "" { + port = "18091" + } + jevKey := os.Getenv("MOCK_JEV_KEY") + if jevKey == "" { + jevKey = "qa-mock-jev-key" + } + + mux := http.NewServeMux() + for _, u := range newUpstreams(jevKey) { + prefix := "/" + u.name + mux.HandleFunc(prefix+"/v1/models", u.serveModels) + mux.HandleFunc(prefix+"/v1/systemone", u.serveSystemOne("evaluate")) + mux.HandleFunc(prefix+"/v1/systemone/permute", u.serveSystemOne("permute")) + mux.HandleFunc(prefix+"/v1/systemone/separate", u.serveSystemOne("separate")) + } + mux.HandleFunc("/healthz", func(w http.ResponseWriter, _ *http.Request) { + fmt.Fprintln(w, "ok") + }) + mux.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) { + writeJSON(w, http.StatusNotFound, map[string]any{"detail": "Not Found: " + strings.TrimSpace(r.URL.Path)}) + }) + + log.Printf("mockjev listening on :%s", port) + log.Fatal(http.ListenAndServe(":"+port, mux)) +} diff --git a/tests/e2e/release-e2e-scenarios.md b/tests/e2e/release-e2e-scenarios.md index d51206e89..137228fac 100644 --- a/tests/e2e/release-e2e-scenarios.md +++ b/tests/e2e/release-e2e-scenarios.md @@ -11,6 +11,9 @@ These scenarios are prepared for execution across these local gateways: - `http://localhost:18090` - mock MCP upstream (`tests/e2e/mockmcp`, started by the stack manager; `/alpha` requires the `X-Mock-Token` header, `/beta` is open) +- `http://localhost:18091` - mock System One upstreams (`tests/e2e/mockjev`, + started by the stack manager and registered on every gateway as the `jev`, + `jev-kev`, and `jev-down` providers) ## Recommended runner @@ -171,6 +174,22 @@ Stateful note: Gemini 3 tool call replayed with its thought signature. They create and clean up their own artifacts and are rerunnable in any order; `S227` reloads the SQLite gateway and therefore stays sequential +- `S229`-`S241` exercise the Jev / Kev System One API (`/v1/systemone`, Kev's + `/permute` and `/separate`, passthrough, pinned versions, misuse negatives, + audit and usage, failover, exact cache on the auth gateway, state guardrails + on the guardrail gateway, managed-key allowlists) against the mock upstream + on port 18091, since no hosted Jev key or Kev server is available; each + prints `SKIPPED:` and exits 0 when the mock is down. They create and delete + their own `$QA_SUFFIX`-scoped virtual models, guardrails, workflows, keys, + and pricing overrides and are rerunnable in any order. `S237` sets a pricing + override and `S240` a guardrail workflow, so both stay sequential +- `S242`-`S244` exercise MCP per-server tool filters and + `disallowed_user_paths` (in-place edits reaching open sessions) and the + master key keeping the caller's user-path header on `/mcp` and audio + uploads; they register `$QA_SUFFIX`-scoped servers and delete them, but + mutate the shared MCP catalog, so they stay sequential +- `S245`-`S246` exercise `developer` messages, `strict` tools, and Gemini's + `allowed_tools` tool choice; they are read-only and rerunnable in any order - `S218` exercises Gemini's native `batchEmbedContents` path (batch input, `dimensions`); read-only and rerunnable in any order - `S219` asserts the effective resilience configuration on @@ -487,6 +506,58 @@ mcp_cleanup_release_servers() { curl -sS -o /dev/null -X DELETE "$base/admin/mcp-servers/$QA_MCP_BETA" || true } +# System One (Jev / Kev) upstreams served by tests/e2e/mockjev: the stack +# manager registers "jev" (hosted shape, keyed), "jev-kev" (keyless Kev +# server), and "jev-down" (always 529) on every gateway. +export JEV_MOCK_BASE="${JEV_MOCK_BASE:-http://localhost:${GOMODEL_RELEASE_MOCK_JEV_PORT:-18091}}" +export QA_SYSTEMONE_QUESTIONS='{"department":{"type":"choice","instructions":"Which team should handle this?","criteria":{"returns":"Exchanges and refunds","shipping":"Delivery delays","billing":"Charges and invoices"}},"escalate":{"type":"noul","instructions":"Does this need urgent human attention?"},"frustration":{"type":"score","instructions":"How frustrated is the customer?","criteria":["Calm","Frustrated","Very angry"]}}' +export QA_SYSTEMONE_CHOICE='{"department":{"type":"choice","instructions":"Which team?","criteria":{"returns":"Returns","billing":"Billing"}}}' + +# Skips when the mock upstream is down, and fails when the gateway was started +# without the mock-backed jev providers (an outdated stack manager). +# usage: systemone_require_mock BASE_URL [curl args...] +systemone_require_mock() { + local base="$1" + shift + if ! curl -fsS "$JEV_MOCK_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock System One upstream is not running on $JEV_MOCK_BASE" + exit 0 + fi + if ! curl -fsS "$base/v1/models" "$@" | jq -e ' + any(.data[]; .id == "jev/jev-latest") and any(.data[]; .id == "jev-kev/kev-latest") + ' >/dev/null; then + echo "error: $base has no mock-backed jev providers; restart it with tests/e2e/manage-release-e2e-stack.sh" >&2 + exit 1 + fi +} + +# Asserts an HTTP status and prints the body on a mismatch. +# usage: assert_http_status WANT GOT BODY_FILE +assert_http_status() { + if [ "$2" != "$1" ]; then + echo "error: expected HTTP $1, got $2" >&2 + cat "$3" >&2 || true + exit 1 + fi +} + +# Polls the audit or usage log until an entry for the request id appears. +# usage: wait_log_entry BASE_URL audit|usage REQUEST_ID OUTPUT_FILE [curl args...] +wait_log_entry() { + local base="$1" kind="$2" rid="$3" out="$4" + shift 4 + for _ in $(seq 1 15); do + curl -fsS "$base/admin/$kind/log?search=$rid&limit=5" "$@" > "$out" + if jq -e --arg rid "$rid" 'any(.entries[]?; .request_id == $rid)' "$out" >/dev/null; then + return 0 + fi + sleep 1 + done + jq . "$out" >&2 || true + echo "error: no $kind entry for $rid on $base" >&2 + exit 1 +} + run_release_budget_enforcement() { local base_url="$1" local budget_path="$2" @@ -3565,7 +3636,7 @@ if jq -e '.providers[] | select(.name == "fireworks") | (.status != "healthy") a exit 0 fi FIREWORKS_MODEL=$(curl -fsS "$BASE_URL/v1/models" \ - | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("llama-v3p1-8b-instruct$"))) + .)[0]') + | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("gpt-oss-120b$"))) + .)[0]') RESP_FILE="$QA_RUN_DIR/s153.chat.json" curl -fsS "$BASE_URL/v1/chat/completions" \ -H 'Content-Type: application/json' \ @@ -3589,7 +3660,7 @@ if jq -e '.providers[] | select(.name == "fireworks") | (.status != "healthy") a exit 0 fi FIREWORKS_MODEL=$(curl -fsS "$BASE_URL/v1/models" \ - | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("llama-v3p1-8b-instruct$"))) + .)[0]') + | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("gpt-oss-120b$"))) + .)[0]') SSE_FILE="$QA_RUN_DIR/s154.chat.sse" curl -fsS --no-buffer "$BASE_URL/v1/chat/completions" \ -H 'Content-Type: application/json' \ @@ -6166,3 +6237,946 @@ jq -c --argjson tools "$TOOLS" '{ jq '{provider,answer:.choices[0].message.content}' "$FOLLOW_FILE" assert_chat_response_contains "$FOLLOW_FILE" "gemini" "22" ``` + +## 35. Jev / Kev System One API + +`POST /v1/systemone` (and Kev's `/permute` and `/separate`) forwards TypeSafe +System One decision requests natively. No hosted Jev key or Kev server is +available to the matrix, so the stack manager starts `tests/e2e/mockjev` on +port 18091 and registers three `jev` providers against it on every gateway: +`jev` (hosted-API shape, keyed, lists `jev-latest`/`jev-preview`, accepts any +versioned `jev-X.Y.Z`), `jev-kev` (keyless Kev server whose base URL keeps a +trailing `/v1`, checkpoint `kev-latest` with alias `kev-4b`), and `jev-down` +(lists `kev-down`, answers every System One route with `529`). Each mock +answer carries a `mock` object echoing what reached the upstream (model, +state, questions, extra fields, whether an `Authorization` header arrived, +`X-Request-Id`, and a per-upstream request sequence), so the scenarios can +assert exactly what the gateway forwarded and whether an answer was replayed +from cache. + +### S229 System One providers register and list utility models + +```bash +systemone_require_mock "$BASE_URL" + +MODELS_FILE="$QA_RUN_DIR/s229.models.json" +curl -fsS "$BASE_URL/v1/models" > "$MODELS_FILE" +jq -c '[.data[] | select(.owned_by | startswith("jev")) | {id, categories: .metadata.categories, modes: .metadata.modes}]' "$MODELS_FILE" +jq -e ' + ([.data[] | select(.owned_by | startswith("jev")) | .id] | sort) + == ["jev-down/kev-down","jev-kev/kev-4b","jev-kev/kev-latest","jev/jev-latest","jev/jev-preview"] + and all(.data[] | select(.owned_by | startswith("jev")); .metadata.categories == ["utility"] and ((.metadata.modes // []) | length == 0)) + and any(.data[]; .id == "jev/jev-latest" and .metadata.description == "Latest Jev (mock)" and .created > 0) +' "$MODELS_FILE" >/dev/null + +STATUS_FILE="$QA_RUN_DIR/s229.status.json" +curl -fsS "$BASE_URL/admin/providers/status" > "$STATUS_FILE" +# jev-down turns degraded once failover scenarios have sent it traffic (its +# 529s count against request health), so it only has to be registered. +jq -e ' + [.. | objects | select(.type? == "jev" and has("status")) | {name, status}] | sort_by(.name) as $s + | ($s | map(.name)) == ["jev","jev-down","jev-kev"] + and all($s[]; if .name == "jev-down" then (.status | IN("healthy","degraded")) else .status == "healthy" end) +' "$STATUS_FILE" >/dev/null + +# Passthrough lists models in each upstream's own shape. +curl -fsS "$BASE_URL/p/jev/v1/models" \ + | jq -e '[.models[].name] == ["jev-latest","jev-preview"]' >/dev/null +curl -fsS "$BASE_URL/p/jev-kev/v1/models" \ + | jq -e '.models[0].id == "kev-latest" and .models[0].aliases == ["kev-4b"]' >/dev/null +``` + +### S230 Native `/v1/systemone` answers every question type on hosted Jev + +Sends choice, noul, and score questions plus an extra top-level field. The +answer is relayed unchanged, only `model` is rewritten to the resolved name, +the questions and extra field reach the upstream byte for byte, the client's +`Authorization` header is replaced by the provider key, and the request ID is +forwarded. + +```bash +systemone_require_mock "$BASE_URL" + +for MODEL in jev-latest jev/jev-latest; do + RID="qa-s1-hosted-$QA_SUFFIX-${MODEL//\//-}" + RESP_FILE="$QA_RUN_DIR/s230.${MODEL//\//-}.json" + CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -H 'Authorization: Bearer qa-client-token-must-not-reach-upstream' \ + -H "X-Request-ID: $RID" \ + -d "{\"model\":\"$MODEL\",\"state\":\"Shoes arrived late and I see two charges on my card.\",\"questions\":$QA_SYSTEMONE_QUESTIONS,\"qa_marker\":{\"nested\":[1,2,3]}}") + assert_http_status 200 "$CODE" "$RESP_FILE" + jq -c '{model, answers, usage, mock: (.mock | {upstream, received_model, authorization, request_id})}' "$RESP_FILE" + jq -e --arg rid "$RID" --argjson questions "$QA_SYSTEMONE_QUESTIONS" ' + .model == "jev-1.13.0" + and .answers.department.type == "choice" and (.answers.department.choice | IN("returns","shipping","billing")) + and (.answers.department.probabilities | keys | sort) == ["billing","returns","shipping"] + and .answers.escalate.type == "noul" and (.answers.escalate.noul | type == "number") + and .answers.frustration.type == "score" and .answers.frustration.legend == {"0":"Calm","1":"Frustrated","2":"Very angry"} + and .usage.input_tokens > 0 and .usage.output_tokens > 0 + and .mock.upstream == "jev" + and .mock.received_model == "jev-latest" + and .mock.questions == $questions + and .mock.extra == {"qa_marker":{"nested":[1,2,3]}} + and .mock.authorization == "provider-key" + and .mock.request_id == $rid + ' "$RESP_FILE" >/dev/null +done +``` + +### S231 Keyless Kev server by checkpoint and alias + +The `jev-kev` provider has no key and its base URL ends in `/v1`, which the +provider trims. No `Authorization` header may reach it, not even the client's. + +```bash +systemone_require_mock "$BASE_URL" + +for MODEL in jev-kev/kev-latest kev-latest jev-kev/kev-4b kev-4b; do + RESP_FILE="$QA_RUN_DIR/s231.${MODEL//\//-}.json" + CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -H 'Authorization: Bearer qa-client-token-must-not-reach-upstream' \ + -d "{\"model\":\"$MODEL\",\"state\":\"I was charged twice.\",\"questions\":$QA_SYSTEMONE_CHOICE}") + assert_http_status 200 "$CODE" "$RESP_FILE" + jq -e --arg sent "${MODEL#jev-kev/}" ' + .model == "kev-4b-e2e" + and .mock.upstream == "kev" + and .mock.received_model == $sent + and .mock.authorization == "none" + and .answers.department.type == "choice" + ' "$RESP_FILE" >/dev/null +done +``` + +### S232 Pinned Jev versions route without being listed + +TypeSafe lists only its aliases but accepts any versioned ID. A name that +says which `jev` provider to use reaches it unlisted; a bare unlisted name is +not guessed while several `jev` providers are configured; a model the +upstream rejects comes back with the upstream's status. + +```bash +systemone_require_mock "$BASE_URL" + +for MODEL in jev/jev-1.13.0 jev/jev-1.12.0; do + RESP_FILE="$QA_RUN_DIR/s232.${MODEL//\//-}.json" + CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"$MODEL\",\"state\":\"pinned\",\"questions\":$QA_SYSTEMONE_CHOICE}") + assert_http_status 200 "$CODE" "$RESP_FILE" + jq -e --arg version "${MODEL#jev/}" '.model == $version and .mock.upstream == "jev" and .mock.received_model == $version' "$RESP_FILE" >/dev/null +done + +# Bare and unlisted, with jev, jev-kev, and jev-down all configured. +RESP_FILE="$QA_RUN_DIR/s232.bare.json" +CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-1.13.0\",\"state\":\"pinned\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 404 "$CODE" "$RESP_FILE" +jq -e '.error.code == "model_not_found"' "$RESP_FILE" >/dev/null + +# Routed to jev because the name says so; the upstream rejects it. +RESP_FILE="$QA_RUN_DIR/s232.unknown.json" +CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev/not-a-jev-model\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 404 "$CODE" "$RESP_FILE" +jq -e '.error.type == "not_found_error" and .error.provider == "jev" and (.error.message | contains("not-a-jev-model"))' "$RESP_FILE" >/dev/null +``` + +### S233 A virtual model pins an unlisted Jev version + +The System One docs state that a virtual model can pin a version the same way +a provider-qualified name does (`virtual_models: [{source: ..., target: +jev/jev-1.13.0}]`). This creates one through the admin API and sends a request +through it. + +```bash +systemone_require_mock "$BASE_URL" + +NAME="qa-jev-pinned-$QA_SUFFIX" +cleanup_s233() { + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\"}" || true +} +trap cleanup_s233 EXIT + +VM_FILE="$QA_RUN_DIR/s233.vm.json" +CODE=$(curl -sS -o "$VM_FILE" -w '%{http_code}' -X PUT "$BASE_URL/admin/virtual-models" \ + -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\",\"target_model\":\"jev/jev-1.12.0\"}") +assert_http_status 200 "$CODE" "$VM_FILE" + +RESP_FILE="$QA_RUN_DIR/s233.answer.json" +CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"$NAME\",\"state\":\"pinned through a virtual model\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$RESP_FILE" +jq -e '.model == "jev-1.12.0" and .mock.upstream == "jev" and .mock.received_model == "jev-1.12.0"' "$RESP_FILE" >/dev/null +``` + +### S234 Kev diagnostic routes `/permute` and `/separate` + +Kev serves both diagnostic routes; the hosted-shaped `jev` answers them with +its own `404`, and an OpenRouter model is refused before any upstream call +since OpenRouter serves only the evaluation route. + +```bash +systemone_require_mock "$BASE_URL" + +post_systemone() { + local route="$1" body="$2" out="$3" + curl -sS -o "$out" -w '%{http_code}' "$BASE_URL/v1/systemone$route" -H 'Content-Type: application/json' -d "$body" +} + +F="$QA_RUN_DIR/s234.permute.json" +CODE=$(post_systemone /permute "{\"model\":\"jev-kev/kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"n_perm\":3}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.n_perm == 3 and .mock.route == "permute" and .mock.received_model == "kev-latest"' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.permute-default.json" +CODE=$(post_systemone /permute "{\"model\":\"kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.n_perm == 6' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.permute-bad.json" +CODE=$(post_systemone /permute "{\"model\":\"kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"n_perm\":65}" "$F") +assert_http_status 422 "$CODE" "$F" +jq -e '.error.type == "invalid_request_error" and (.error.message | contains("n_perm"))' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.separate.json" +CODE=$(post_systemone /separate "{\"model\":\"jev-kev/kev-4b\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_QUESTIONS}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.separate == true and .mock.route == "separate" and (.answers | keys | length) == 3' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.hosted-permute.json" +CODE=$(post_systemone /permute "{\"model\":\"jev/jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 404 "$CODE" "$F" +jq -e '.error.type == "not_found_error" and .error.provider == "jev"' "$F" >/dev/null + +OPENROUTER_MODEL=$(curl -fsS "$BASE_URL/v1/models" | jq -r '[.data[].id | select(startswith("openrouter/"))][0] // empty') +if [ -n "$OPENROUTER_MODEL" ]; then + F="$QA_RUN_DIR/s234.openrouter-permute.json" + CODE=$(post_systemone /permute "{\"model\":\"$OPENROUTER_MODEL\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") + assert_http_status 400 "$CODE" "$F" + jq -e '.error.param == "model" and (.error.message | contains("answers only /v1/systemone"))' "$F" >/dev/null +else + echo "note: no openrouter model in the catalog; OpenRouter permute refusal not checked" +fi +``` + +### S235 System One misuse is rejected with an explanation (negatives) + +The endpoint never translates: missing or malformed input, chat models (direct +or through a virtual model), and System One models on OpenAI routes are all +`400 invalid_request_error` naming the fix. A malformed question reaches the +upstream and comes back as its `422`. + +```bash +systemone_require_mock "$BASE_URL" + +NAME="qa-s1-chat-vm-$QA_SUFFIX" +cleanup_s235() { + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\"}" || true +} +trap cleanup_s235 EXIT +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\",\"target_model\":\"openai/gpt-4.1-nano\"}" >/dev/null + +# expect_invalid PATH BODY JQ_MESSAGE_FILTER +expect_invalid() { + local path="$1" body="$2" filter="$3" out + out=$(mktemp "$QA_RUN_DIR/s235.XXXXXX") + local code + code=$(curl -sS -o "$out" -w '%{http_code}' "$BASE_URL$path" -H 'Content-Type: application/json' -d "$body") + assert_http_status 400 "$code" "$out" + if ! jq -e ".error.type == \"invalid_request_error\" and ($filter)" "$out" >/dev/null; then + echo "error: unexpected 400 body for $path" >&2 + cat "$out" >&2 + exit 1 + fi +} + +expect_invalid /v1/systemone "{\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.param == "model" and .error.message == "model is required"' +expect_invalid /v1/systemone '{"model":' \ + '.error.message | startswith("invalid request body")' +expect_invalid /v1/systemone "{\"model\":\"openai/gpt-4.1-nano\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.param == "model" and (.error.message | contains("provider type openai, which has no System One API"))' +expect_invalid /v1/systemone "{\"model\":\"$NAME\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.message | contains("(resolved to \"openai/gpt-4.1-nano\")")' +OPENROUTER_MODEL=$(curl -fsS "$BASE_URL/v1/models" | jq -r '[.data[] | select((.id | startswith("openrouter/")) and ((.metadata.modes // []) | index("chat")))][0].id // empty') +if [ -n "$OPENROUTER_MODEL" ]; then + expect_invalid /v1/systemone "{\"model\":\"$OPENROUTER_MODEL\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.message | contains("not a System One model")' +fi + +expect_invalid /v1/chat/completions '{"model":"jev/jev-latest","messages":[{"role":"user","content":"hi"}]}' \ + '.error.param == "model" and (.error.message | contains("does not support chat completions") and contains("POST /v1/systemone"))' +expect_invalid /v1/responses '{"model":"jev-kev/kev-latest","input":"hi"}' \ + '.error.message | contains("does not support responses") and contains("POST /v1/systemone")' +expect_invalid /v1/embeddings '{"model":"jev-latest","input":"hi"}' \ + '.error.message | contains("does not support embeddings") and contains("POST /v1/systemone")' + +# A misrouted System One request is an operator mistake, so it is logged. +grep -Fq 'System One request routed to a model without the System One API' \ + "$RELEASE_STACK_DIR/sqlite-main/logs/server.log" + +F="$QA_RUN_DIR/s235.bad-question.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d '{"model":"jev-kev/kev-latest","state":"x","questions":{"q":{"type":"maybe","instructions":"?"}}}') +assert_http_status 422 "$CODE" "$F" +jq -e '.error.provider == "jev" and (.error.message | contains("questions") and contains("maybe"))' "$F" >/dev/null + +F="$QA_RUN_DIR/s235.get.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone") +assert_http_status 405 "$CODE" "$F" +``` + +### S236 Passthrough reaches the same upstreams under `/p/jev*` + +Passthrough forwards the body as sent (no model rewrite), rejects a body that +names the model twice (the upstream parser could pick a value the gateway +never checked), and reaches Kev's diagnostic routes on the suffixed provider. + +```bash +systemone_require_mock "$BASE_URL" + +F="$QA_RUN_DIR/s236.systemone.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/p/jev/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-latest\",\"state\":\"via passthrough\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$F" +jq -e '.model == "jev-1.13.0" and .mock.received_model == "jev-latest" and .mock.authorization == "provider-key"' "$F" >/dev/null + +# The /v1 prefix is optional on passthrough routes. +F="$QA_RUN_DIR/s236.permute.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/p/jev-kev/systemone/permute" -H 'Content-Type: application/json' \ + -d "{\"model\":\"kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"n_perm\":2}") +assert_http_status 200 "$CODE" "$F" +jq -e '.n_perm == 2 and .mock.upstream == "kev" and .mock.authorization == "none"' "$F" >/dev/null + +F="$QA_RUN_DIR/s236.dup-model.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/p/jev/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"model\":\"jev-preview\"}") +assert_http_status 400 "$CODE" "$F" +jq -e '.error.message | contains("model field is repeated")' "$F" >/dev/null +``` + +### S237 System One calls are audited, filterable, metered, and priced + +Checks the audit entry (route, requested and resolved model, provider, one +successful primary attempt), the `exclude_operation=systemone` request-type +filter, and the usage entry: tokens copied from the answer, recorded under +the model that answered, and priced by an operator override. + +```bash +systemone_require_mock "$BASE_URL" + +# A provider-wide rate must not shadow the rate declared for the version that +# answers the jev-latest alias. +SELECTOR="jev/jev-1.13.0" +PROVIDER_SELECTOR="jev/" +cleanup_s237() { + for selector in "$SELECTOR" "$PROVIDER_SELECTOR"; do + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/model-pricing-overrides" \ + -H 'Content-Type: application/json' -d "{\"selector\":\"$selector\"}" || true + done +} +trap cleanup_s237 EXIT +curl -fsS -X PUT "$BASE_URL/admin/model-pricing-overrides" -H 'Content-Type: application/json' \ + -d "{\"selector\":\"$SELECTOR\",\"pricing\":{\"input_per_mtok\":42,\"output_per_mtok\":0}}" >/dev/null +curl -fsS -X PUT "$BASE_URL/admin/model-pricing-overrides" -H 'Content-Type: application/json' \ + -d "{\"selector\":\"$PROVIDER_SELECTOR\",\"pricing\":{\"input_per_mtok\":1,\"output_per_mtok\":1}}" >/dev/null + +RID="qa-s1-audit-$QA_SUFFIX" +ANSWER_FILE="$QA_RUN_DIR/s237.answer.json" +curl -fsS "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' -H "X-Request-ID: $RID" \ + -d "{\"model\":\"jev/jev-latest\",\"state\":\"audit me\",\"questions\":$QA_SYSTEMONE_QUESTIONS}" > "$ANSWER_FILE" +IN=$(jq -er '.usage.input_tokens' "$ANSWER_FILE") +OUT=$(jq -er '.usage.output_tokens' "$ANSWER_FILE") + +AUDIT_FILE="$QA_RUN_DIR/s237.audit.json" +wait_log_entry "$BASE_URL" audit "$RID" "$AUDIT_FILE" +jq -e --arg rid "$RID" ' + any(.entries[]; .request_id == $rid + and .path == "/v1/systemone" and .method == "POST" and .status_code == 200 + and .requested_model == "jev/jev-latest" and .resolved_model == "jev/jev-latest" + and .provider == "jev" and .provider_name == "jev" + and ([.data.attempts[]? | {kind, provider_name, success}] == [{"kind":"primary","provider_name":"jev","success":true}])) +' "$AUDIT_FILE" >/dev/null + +curl -fsS "$BASE_URL/admin/audit/log?search=$RID&limit=5&exclude_operation=systemone" \ + | jq -e '(.entries // []) | length == 0' >/dev/null +curl -fsS "$BASE_URL/admin/audit/log?search=$RID&limit=5&exclude_operation=chat_completions,provider_passthrough" \ + | jq -e --arg rid "$RID" 'any(.entries[]; .request_id == $rid)' >/dev/null +CODE=$(curl -sS -o "$QA_RUN_DIR/s237.bad-filter.json" -w '%{http_code}' "$BASE_URL/admin/audit/log?exclude_operation=not_an_operation") +assert_http_status 400 "$CODE" "$QA_RUN_DIR/s237.bad-filter.json" + +USAGE_FILE="$QA_RUN_DIR/s237.usage.json" +wait_log_entry "$BASE_URL" usage "$RID" "$USAGE_FILE" +jq -c --arg rid "$RID" '.entries[] | select(.request_id == $rid)' "$USAGE_FILE" +jq -e --arg rid "$RID" --argjson in "$IN" --argjson out "$OUT" ' + any(.entries[]; .request_id == $rid + and .endpoint == "/v1/systemone" and .model == "jev-1.13.0" + and .provider == "jev" and .provider_name == "jev" + and .input_tokens == $in and .output_tokens == $out) +' "$USAGE_FILE" >/dev/null +# The two checks below are independent, so both are reported before failing. +FAILED=0 +if ! jq -e --arg rid "$RID" --argjson total "$((IN + OUT))" \ + 'any(.entries[]; .request_id == $rid and .total_tokens == $total)' "$USAGE_FILE" >/dev/null; then + echo "error: usage total_tokens is not input_tokens + output_tokens ($IN + $OUT) for a System One answer" >&2 + FAILED=1 +fi +# Jev's documented pricing: per input token, output_per_mtok 0. +if ! jq -e --arg rid "$RID" --argjson in "$IN" ' + any(.entries[]; .request_id == $rid and ((((.input_cost // -1) - ($in * 42 / 1000000)) | fabs) < 0.000000001)) + ' "$USAGE_FILE" >/dev/null; then + echo "error: the pricing override (input 42/Mtok, output 0) did not cost the System One usage entry" >&2 + FAILED=1 +fi +[ "$FAILED" = 0 ] +``` + +### S238 Failover moves System One requests between System One targets + +A `failover` virtual model whose primary answers `529` moves to its next +target, skips a chat model without spending an attempt, and records the +answer under the target that did the work. A client error (`422`) is returned +without failover. + +```bash +systemone_require_mock "$BASE_URL" + +FO="qa-s1-failover-$QA_SUFFIX" +FO422="qa-s1-failover-422-$QA_SUFFIX" +FOPIN="qa-s1-failover-pinned-$QA_SUFFIX" +cleanup_s238() { + for name in "$FO" "$FO422" "$FOPIN"; do + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$name\"}" || true + done +} +trap cleanup_s238 EXIT + +# The down target on its own relays the upstream overload status (or 503 +# once its circuit breaker has opened after earlier reruns). +F="$QA_RUN_DIR/s238.down.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-down/kev-down\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}") +case "$CODE" in 529|503) ;; *) assert_http_status 529 "$CODE" "$F" ;; esac + +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$FO\",\"strategy\":\"failover\",\"targets\":[{\"model\":\"jev-down/kev-down\"},{\"model\":\"openai/gpt-4.1-nano\"},{\"model\":\"jev-kev/kev-latest\"}]}" >/dev/null +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$FO422\",\"strategy\":\"failover\",\"targets\":[{\"model\":\"jev-kev/kev-latest\"},{\"model\":\"jev/jev-latest\"}]}" >/dev/null + +RID="qa-s1-failover-$QA_SUFFIX" +F="$QA_RUN_DIR/s238.failover.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -H "X-Request-ID: $RID" \ + -d "{\"model\":\"$FO\",\"state\":\"fail over please\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$F" +jq -e '.model == "kev-4b-e2e" and .mock.upstream == "kev" and .mock.received_model == "kev-latest"' "$F" >/dev/null + +AUDIT_FILE="$QA_RUN_DIR/s238.audit.json" +wait_log_entry "$BASE_URL" audit "$RID" "$AUDIT_FILE" +jq -c --arg rid "$RID" '.entries[] | select(.request_id == $rid) | [.data.attempts[] | {kind, provider_name, status_code, success}]' "$AUDIT_FILE" +jq -e --arg rid "$RID" --arg fo "$FO" ' + any(.entries[]; .request_id == $rid + and .requested_model == $fo and .resolved_model == "jev-kev/kev-latest" and .provider_name == "jev-kev" + and .data.failover != null + and ([.data.attempts[] | .provider_name] == ["jev-down","jev-kev"]) + and .data.attempts[0].success == false and .data.attempts[1].kind == "failover" and .data.attempts[1].success == true) +' "$AUDIT_FILE" >/dev/null + +USAGE_FILE="$QA_RUN_DIR/s238.usage.json" +wait_log_entry "$BASE_URL" usage "$RID" "$USAGE_FILE" +jq -e --arg rid "$RID" 'any(.entries[]; .request_id == $rid and .provider_name == "jev-kev" and .model == "kev-4b-e2e")' "$USAGE_FILE" >/dev/null + +# A pinned version the catalog does not list is a valid failover target and +# is sent to the provider it names, not to the failed primary's. +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$FOPIN\",\"strategy\":\"failover\",\"targets\":[{\"model\":\"jev-down/kev-down\"},{\"model\":\"jev/jev-1.12.0\"}]}" >/dev/null +F="$QA_RUN_DIR/s238.failover-pinned.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"$FOPIN\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$F" +jq -e '.model == "jev-1.12.0" and .mock.upstream == "jev" and .mock.received_model == "jev-1.12.0"' "$F" >/dev/null + +RID422="qa-s1-failover-422-$QA_SUFFIX" +F="$QA_RUN_DIR/s238.no-failover.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -H "X-Request-ID: $RID422" \ + -d "{\"model\":\"$FO422\",\"state\":\"x\",\"questions\":{\"q\":{\"type\":\"maybe\",\"instructions\":\"?\"}}}") +assert_http_status 422 "$CODE" "$F" +wait_log_entry "$BASE_URL" audit "$RID422" "$AUDIT_FILE" +jq -e --arg rid "$RID422" ' + any(.entries[]; .request_id == $rid and .status_code == 422 and ([.data.attempts[] | .provider_name] == ["jev-kev"])) +' "$AUDIT_FILE" >/dev/null +``` + +### S239 Identical System One requests hit the exact response cache + +Runs on the auth + exact-cache gateway. The replayed answer carries the same +mock request sequence (the upstream was not called), `Cache-Control: no-cache` +bypasses the cache, a different state misses, and the hit is audited and +recorded in usage as an exact cache hit. + +```bash +systemone_require_mock "$AUTH_BASE_URL" -H "$ADMIN_AUTH_HEADER" + +STATE="release cache probe $QA_SUFFIX" +BODY="{\"model\":\"jev-kev/kev-latest\",\"state\":\"$STATE\",\"questions\":$QA_SYSTEMONE_CHOICE}" +send_cached() { + local rid="$1" headers="$2" body_file="$3" + shift 3 + curl -fsS -D "$headers" -o "$body_file" "$AUTH_BASE_URL/v1/systemone" \ + -H "$ADMIN_AUTH_HEADER" -H 'Content-Type: application/json' -H "X-Request-ID: $rid" "$@" -d "${BODY_OVERRIDE:-$BODY}" +} + +RID1="qa-s1-cache-$QA_SUFFIX-1" +RID2="qa-s1-cache-$QA_SUFFIX-2" +RID3="qa-s1-cache-$QA_SUFFIX-3" +send_cached "$RID1" "$QA_RUN_DIR/s239.1.headers" "$QA_RUN_DIR/s239.1.json" +send_cached "$RID2" "$QA_RUN_DIR/s239.2.headers" "$QA_RUN_DIR/s239.2.json" +send_cached "$RID3" "$QA_RUN_DIR/s239.3.headers" "$QA_RUN_DIR/s239.3.json" -H 'Cache-Control: no-cache' +BODY_OVERRIDE="{\"model\":\"jev-kev/kev-latest\",\"state\":\"$STATE changed\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + send_cached "qa-s1-cache-$QA_SUFFIX-4" "$QA_RUN_DIR/s239.4.headers" "$QA_RUN_DIR/s239.4.json" + +SEQ1=$(jq -er '.mock.request_seq' "$QA_RUN_DIR/s239.1.json") +grep -Eiq '^X-Cache: *HIT \(exact\)' "$QA_RUN_DIR/s239.2.headers" +jq -e --argjson seq "$SEQ1" '.mock.request_seq == $seq and .model == "kev-4b-e2e"' "$QA_RUN_DIR/s239.2.json" >/dev/null +cmp -s "$QA_RUN_DIR/s239.1.json" "$QA_RUN_DIR/s239.2.json" +for n in 3 4; do + if grep -Eiq '^X-Cache:' "$QA_RUN_DIR/s239.$n.headers"; then + echo "error: request $n should not have been served from cache" >&2 + exit 1 + fi + jq -e --argjson seq "$SEQ1" '.mock.request_seq > $seq' "$QA_RUN_DIR/s239.$n.json" >/dev/null +done + +AUDIT_FILE="$QA_RUN_DIR/s239.audit.json" +wait_log_entry "$AUTH_BASE_URL" audit "$RID2" "$AUDIT_FILE" -H "$ADMIN_AUTH_HEADER" +jq -e --arg rid "$RID2" 'any(.entries[]; .request_id == $rid and .cache_type == "exact" and .status_code == 200 and .path == "/v1/systemone")' "$AUDIT_FILE" >/dev/null + +USAGE_FILE="$QA_RUN_DIR/s239.usage.json" +wait_log_entry "$AUTH_BASE_URL" usage "$RID1" "$USAGE_FILE" -H "$ADMIN_AUTH_HEADER" +# Cache hits are listed only with cache_mode=cached. +for _ in $(seq 1 15); do + curl -fsS "$AUTH_BASE_URL/admin/usage/log?search=$RID2&cache_mode=cached&limit=5" -H "$ADMIN_AUTH_HEADER" > "$USAGE_FILE" + if jq -e --arg rid "$RID2" 'any(.entries[]?; .request_id == $rid)' "$USAGE_FILE" >/dev/null; then + break + fi + sleep 1 +done +jq -e --arg rid "$RID2" ' + any(.entries[]; .request_id == $rid and .cache_type == "exact" and .endpoint == "/v1/systemone" + and .input_tokens > 0 and .total_tokens == .input_tokens + .output_tokens and .provider_name == "jev-kev") +' "$USAGE_FILE" >/dev/null +``` + +### S240 Guardrails see the System One state and nothing else + +Runs on the guardrail gateway. Its global `system_prompt` override has no +place in a decision request, so the edit is dropped with a one-time warning +and the state is forwarded untouched. A workflow scoped to `jev-kev` and a +user path then masks card numbers in a string state and in a JSON state +(which stays JSON), and blocks a forbidden state before any upstream call. + +```bash +systemone_require_mock "$GR_BASE_URL" + +S="${QA_SUFFIX//[^[:alnum:]-]/-}" +MASK="qa-s1-mask-$S" +BLOCK="qa-s1-block-$S" +SCOPE_PATH="/qa/systemone/$S" +WORKFLOW_ID_FILE="$QA_RUN_DIR/s240.workflow.id" +cleanup_s240() { + if [ -s "$WORKFLOW_ID_FILE" ]; then + curl -sS -o /dev/null -X POST "$GR_BASE_URL/admin/workflows/$(cat "$WORKFLOW_ID_FILE")/deactivate" || true + fi + for name in "$MASK" "$BLOCK"; do + curl -sS -o /dev/null -X DELETE "$GR_BASE_URL/admin/guardrails" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$name\"}" || true + done +} +trap cleanup_s240 EXIT + +# Global system_prompt guardrail: dropped, state unchanged, warning logged. +F="$QA_RUN_DIR/s240.global.json" +curl -fsS "$GR_BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-kev/kev-latest\",\"state\":\"card 4111 1111 1111 1111\",\"questions\":$QA_SYSTEMONE_CHOICE}" > "$F" +jq -e '.mock.received_state == "card 4111 1111 1111 1111" and .answers.department.type == "choice"' "$F" >/dev/null +grep -Fq 'guardrail edits a System One request cannot carry were dropped' "$RELEASE_STACK_DIR/guardrails/logs/server.log" + +jq -n --arg name "$MASK" '{ + name: $name, type: "string_replace", description: "release e2e: mask card numbers", + config: {mode: "regex", rules: "\\b(\\d{4}) \\d{4} \\d{4} (\\d{4})\\b => $1 **** **** $2"} +}' | curl -fsS -X PUT "$GR_BASE_URL/admin/guardrails" -H 'Content-Type: application/json' -d @- >/dev/null +jq -n --arg name "$BLOCK" '{ + name: $name, type: "string_replace", description: "release e2e: block a forbidden state", + config: {mode: "literal", rules: "QA_FORBIDDEN_STATE => x", on_match: "block", message: "QA_SYSTEMONE_BLOCKED"} +}' | curl -fsS -X PUT "$GR_BASE_URL/admin/guardrails" -H 'Content-Type: application/json' -d @- >/dev/null + +jq -n --arg mask "$MASK" --arg block "$BLOCK" --arg path "$SCOPE_PATH" --arg name "qa-s1-guard-$S" '{ + scope_provider_name: "jev-kev", scope_user_path: $path, name: $name, + description: "release e2e: System One state guardrails", + workflow_payload: { + schema_version: 2, + features: {cache: false, audit: true, usage: true, guardrails: true, failover: false}, + steps: [{ref: $mask, phase: "prompt", step: 10}, {ref: $block, phase: "prompt", step: 20}] + } +}' | curl -fsS -X POST "$GR_BASE_URL/admin/workflows" -H 'Content-Type: application/json' -d @- \ + | jq -er '.id' > "$WORKFLOW_ID_FILE" + +guarded() { + curl -sS -o "$2" -w '%{http_code}' "$GR_BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -H "X-GoModel-User-Path: $SCOPE_PATH/agent" -d "$1" +} + +F="$QA_RUN_DIR/s240.string.json" +CODE=$(guarded "{\"model\":\"jev-kev/kev-latest\",\"state\":\"card 4111 1111 1111 1234 was charged twice\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e --argjson q "$QA_SYSTEMONE_CHOICE" ' + .mock.received_state == "card 4111 **** **** 1234 was charged twice" and .mock.questions == $q +' "$F" >/dev/null + +F="$QA_RUN_DIR/s240.object.json" +CODE=$(guarded "{\"model\":\"jev-kev/kev-latest\",\"state\":{\"note\":\"card 4111 1111 1111 1234\",\"order\":7},\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.mock.received_state == {"note":"card 4111 **** **** 1234","order":7}' "$F" >/dev/null + +F="$QA_RUN_DIR/s240.blocked.json" +CODE=$(guarded "{\"model\":\"jev-kev/kev-latest\",\"state\":\"QA_FORBIDDEN_STATE\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 400 "$CODE" "$F" +jq -e '.error.message | contains("QA_SYSTEMONE_BLOCKED")' "$F" >/dev/null + +# The same request outside the scoped user path is not masked. +F="$QA_RUN_DIR/s240.unscoped.json" +curl -fsS "$GR_BASE_URL/v1/systemone" -H 'Content-Type: application/json' -H "X-GoModel-User-Path: /qa/other/$S" \ + -d "{\"model\":\"jev-kev/kev-latest\",\"state\":\"card 4111 1111 1111 1234\",\"questions\":$QA_SYSTEMONE_CHOICE}" > "$F" +jq -e '.mock.received_state == "card 4111 1111 1111 1234"' "$F" >/dev/null +``` + +### S241 Managed-key model allowlists cover System One and its passthrough + +Runs on the auth gateway with a key allowed only `jev-kev/kev-latest`. Other +System One models are refused on `/v1/systemone` and on passthrough, including +a passthrough body larger than the 64 KiB peek window whose `model` comes last. + +```bash +systemone_require_mock "$AUTH_BASE_URL" -H "$ADMIN_AUTH_HEADER" + +KEY_FILE="$QA_RUN_DIR/s241.key.json" +cleanup_s241() { + if [ -s "$KEY_FILE" ]; then + curl -sS -o /dev/null -X POST "$AUTH_BASE_URL/admin/auth-keys/$(jq -r '.id' "$KEY_FILE")/deactivate" -H "$ADMIN_AUTH_HEADER" || true + fi +} +trap cleanup_s241 EXIT +curl -fsS -X POST "$AUTH_BASE_URL/admin/auth-keys" -H "$ADMIN_AUTH_HEADER" -H 'Content-Type: application/json' \ + -d "{\"name\":\"qa-s1-allowlist-$QA_SUFFIX\",\"user_path\":\"/qa/systemone/allowlist\",\"allowed_models\":[\"jev-kev/kev-latest\"]}" \ + > "$KEY_FILE" +chmod 600 "$KEY_FILE" +KEY=$(jq -er '.value' "$KEY_FILE") + +with_key() { + curl -sS -o "$3" -w '%{http_code}' "$AUTH_BASE_URL$1" -H "Authorization: Bearer $KEY" -H 'Content-Type: application/json' -d "$2" +} +expect_denied() { + local code + code=$(with_key "$1" "$2" "$3") + assert_http_status 400 "$code" "$3" + jq -e '.error.code == "model_access_denied"' "$3" >/dev/null +} + +F="$QA_RUN_DIR/s241.allowed.json" +CODE=$(with_key /v1/systemone "{\"model\":\"jev-kev/kev-latest\",\"state\":\"allowed\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.mock.upstream == "kev"' "$F" >/dev/null + +expect_denied /v1/systemone "{\"model\":\"jev/jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$QA_RUN_DIR/s241.denied.json" +expect_denied /v1/systemone "{\"model\":\"jev/jev-1.13.0\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$QA_RUN_DIR/s241.denied-pinned.json" +expect_denied /p/jev/v1/systemone "{\"model\":\"jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$QA_RUN_DIR/s241.denied-pt.json" + +BIG_STATE=$(head -c 70000 /dev/zero | tr '\0' 'a') +BIG_BODY_FILE="$QA_RUN_DIR/s241.big-body.json" +jq -n --arg state "$BIG_STATE" --argjson q "$QA_SYSTEMONE_CHOICE" '{state: $state, questions: $q, model: "jev-latest"}' > "$BIG_BODY_FILE" +expect_denied /p/jev/v1/systemone "@$BIG_BODY_FILE" "$QA_RUN_DIR/s241.denied-big.json" + +jq -n --arg state "$BIG_STATE" --argjson q "$QA_SYSTEMONE_CHOICE" '{state: $state, questions: $q, model: "kev-latest"}' > "$BIG_BODY_FILE" +F="$QA_RUN_DIR/s241.allowed-big.json" +CODE=$(with_key /p/jev-kev/v1/systemone "@$BIG_BODY_FILE" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.mock.received_length > 65536' "$F" >/dev/null +``` + +## 36. MCP tool and user-path exclusions + +Per-server tool filters and `disallowed_user_paths` are gateway-side access +policy: an edit applies in place without redialing the upstream, and it is +checked on every call, so it also reaches MCP sessions that are already open. +These scenarios register `$QA_SUFFIX`-scoped servers against the mock MCP +upstream on port 18090 and delete them. + +### S242 Tool filters apply in place and reach open sessions + +```bash +if ! curl -fsS "$MCP_UPSTREAM_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock MCP upstream is not running on $MCP_UPSTREAM_BASE" + exit 0 +fi +trap 'mcp_cleanup_release_servers "$BASE_URL"' EXIT + +put_alpha() { + curl -fsS -X PUT "$BASE_URL/admin/mcp-servers" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$QA_MCP_ALPHA\",\"url\":\"$MCP_UPSTREAM_BASE/alpha\",\"transport\":\"http\",\"headers\":{\"X-Mock-Token\":\"$1\"},$2}" >/dev/null +} +alpha_view() { + curl -fsS "$BASE_URL/admin/mcp-servers" | jq -c --arg n "$QA_MCP_ALPHA" '.[] | select(.name == $n)' +} + +put_alpha "$MCP_UPSTREAM_TOKEN" '"disallowed_tools":["add"]' +mcp_wait_status "$BASE_URL" "$QA_MCP_ALPHA" connected +alpha_view | jq -e '.tool_count == 1 and .excluded_tool_count == 1 and .disallowed_tools == ["add"]' >/dev/null +CONNECTED_AT=$(alpha_view | jq -er '.connected_at') +curl -fsS "$BASE_URL/admin/mcp-servers/$QA_MCP_ALPHA/catalog" \ + | jq -e '[.tools[].name] == ["echo"] and [.excluded_tools[].name] == ["add"]' >/dev/null + +SID=$(mcp_initialize "$BASE_URL/mcp" "$QA_RUN_DIR/s242.init.headers" "$QA_RUN_DIR/s242.init.raw") +[ -n "$SID" ] +mcp_initialized "$BASE_URL/mcp" "$SID" +mcp_post "$BASE_URL/mcp" "$SID" '{"jsonrpc":"2.0","id":2,"method":"tools/list"}' \ + | jq -e --arg a "$QA_MCP_ALPHA" '([.result.tools[].name | select(startswith($a + "_"))]) == [$a + "_echo"]' >/dev/null +mcp_post "$BASE_URL/mcp" "$SID" "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_ALPHA}_add\",\"arguments\":{}}}" \ + | jq -e '.error != null' >/dev/null + +# Flip to an allowlist ("Keep hidden") that excludes echo. The stored header +# secret round-trips as ***, and the connection is not redialed. +put_alpha '***' '"allowed_tools":["add"]' +alpha_view | jq -e --arg at "$CONNECTED_AT" ' + .status == "connected" and .connected_at == $at + and .allowed_tools == ["add"] and ((.disallowed_tools // []) | length == 0) + and .tool_count == 1 and .excluded_tool_count == 1 +' >/dev/null + +# echo was listed by the open session before the change; calling it now fails. +mcp_post "$BASE_URL/mcp" "$SID" "{\"jsonrpc\":\"2.0\",\"id\":4,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_ALPHA}_echo\",\"arguments\":{}}}" \ + > "$QA_RUN_DIR/s242.stale-call.json" +jq -e '.error.message | contains("excluded by the gateway tool filters")' "$QA_RUN_DIR/s242.stale-call.json" >/dev/null + +# A new session sees the new filter. +SID2=$(mcp_initialize "$BASE_URL/mcp" "$QA_RUN_DIR/s242.init2.headers" "$QA_RUN_DIR/s242.init2.raw") +mcp_initialized "$BASE_URL/mcp" "$SID2" +mcp_post "$BASE_URL/mcp" "$SID2" '{"jsonrpc":"2.0","id":2,"method":"tools/list"}' \ + | jq -e --arg a "$QA_MCP_ALPHA" '([.result.tools[].name | select(startswith($a + "_"))]) == [$a + "_add"]' >/dev/null +mcp_post "$BASE_URL/mcp" "$SID2" "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_ALPHA}_add\",\"arguments\":{\"marker\":\"QA_MCP_ALLOWED_OK\"}}}" \ + | jq -e '.result.content[0].text | contains("QA_MCP_ALLOWED_OK")' >/dev/null +``` + +### S243 `disallowed_user_paths` carves callers out of a server + +The carve-out wins over `user_paths`, matches whole subtrees, hides the +server from `tools/list` and its per-server endpoint, and a later edit reaches +a session that is already open. Invalid paths are rejected. + +```bash +if ! curl -fsS "$MCP_UPSTREAM_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock MCP upstream is not running on $MCP_UPSTREAM_BASE" + exit 0 +fi +trap 'mcp_cleanup_release_servers "$BASE_URL"' EXIT + +ROOT="/qa/mcp-carve/${QA_SUFFIX//[^[:alnum:]-]/-}" +put_beta() { + curl -sS -o "$QA_RUN_DIR/s243.put.json" -w '%{http_code}' -X PUT "$BASE_URL/admin/mcp-servers" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$QA_MCP_BETA\",\"url\":\"$MCP_UPSTREAM_BASE/beta\",\"transport\":\"http\",\"user_paths\":[\"$ROOT\"],\"disallowed_user_paths\":$1}" +} +CODE=$(put_beta "[\"$ROOT/contractors/\",\"$ROOT/contractors\"]") +assert_http_status 200 "$CODE" "$QA_RUN_DIR/s243.put.json" +mcp_wait_status "$BASE_URL" "$QA_MCP_BETA" connected +curl -fsS "$BASE_URL/admin/mcp-servers" | jq -e --arg n "$QA_MCP_BETA" --arg root "$ROOT" ' + any(.[]; .name == $n and .user_paths == [$root] and .disallowed_user_paths == [$root + "/contractors"]) +' >/dev/null + +# beta_tools SESSION_ID USER_PATH -> prints the beta tool names visible to it +beta_tools() { + mcp_post "$BASE_URL/mcp" "$1" '{"jsonrpc":"2.0","id":2,"method":"tools/list"}' -H "X-GoModel-User-Path: $2" \ + | jq -c --arg b "$QA_MCP_BETA" '[.result.tools[]?.name | select(startswith($b + "_"))]' +} +open_session() { + local sid + sid=$(mcp_initialize "$BASE_URL/mcp" "$QA_RUN_DIR/s243.$2.headers" "$QA_RUN_DIR/s243.$2.raw" -H "X-GoModel-User-Path: $1") + mcp_initialized "$BASE_URL/mcp" "$sid" -H "X-GoModel-User-Path: $1" + echo "$sid" +} + +ENG_SID=$(open_session "$ROOT/eng" eng) +CON_SID=$(open_session "$ROOT/contractors/acme" con) +OUT_SID=$(open_session "/qa/elsewhere" out) +[ "$(beta_tools "$ENG_SID" "$ROOT/eng")" = "[\"${QA_MCP_BETA}_fetch\",\"${QA_MCP_BETA}_search\"]" ] +[ "$(beta_tools "$CON_SID" "$ROOT/contractors/acme")" = "[]" ] +[ "$(beta_tools "$OUT_SID" "/qa/elsewhere")" = "[]" ] + +CODE=$(curl -sS -o "$QA_RUN_DIR/s243.per-server.json" -w '%{http_code}' "$BASE_URL/mcp/$QA_MCP_BETA" \ + -H 'Content-Type: application/json' -H 'Accept: application/json, text/event-stream' \ + -H "X-GoModel-User-Path: $ROOT/contractors/acme" \ + -d '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2025-06-18","capabilities":{},"clientInfo":{"name":"qa-release","version":"1"}}}') +assert_http_status 404 "$CODE" "$QA_RUN_DIR/s243.per-server.json" + +# Carve eng out too: the already-open eng session loses the server. +CODE=$(put_beta "[\"$ROOT/contractors\",\"$ROOT/eng\"]") +assert_http_status 200 "$CODE" "$QA_RUN_DIR/s243.put.json" +mcp_post "$BASE_URL/mcp" "$ENG_SID" \ + "{\"jsonrpc\":\"2.0\",\"id\":5,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_BETA}_search\",\"arguments\":{}}}" \ + -H "X-GoModel-User-Path: $ROOT/eng" > "$QA_RUN_DIR/s243.eng-call.json" +jq -e '.result == null and (.error.message | contains("not available for this user path"))' "$QA_RUN_DIR/s243.eng-call.json" >/dev/null +# A new eng session no longer lists the server. +ENG2_SID=$(open_session "$ROOT/eng" eng2) +[ "$(beta_tools "$ENG2_SID" "$ROOT/eng")" = "[]" ] + +CODE=$(put_beta '["/qa/../escape"]') +assert_http_status 400 "$CODE" "$QA_RUN_DIR/s243.put.json" +jq -e '.error.message | contains("disallowed_user_paths")' "$QA_RUN_DIR/s243.put.json" >/dev/null +``` + +### S244 The master key keeps the caller's user-path header on `/mcp` and audio uploads + +MCP and audio uploads own their transport and take no request snapshot. With +the master key on the auth gateway, the `X-GoModel-User-Path` header must +still scope usage, as it does on chat. + +```bash +if ! curl -fsS "$MCP_UPSTREAM_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock MCP upstream is not running on $MCP_UPSTREAM_BASE" + exit 0 +fi +cleanup_s244() { + curl -sS -o /dev/null -X DELETE "$AUTH_BASE_URL/admin/mcp-servers/$QA_MCP_BETA" -H "$ADMIN_AUTH_HEADER" || true +} +trap cleanup_s244 EXIT + +USER_PATH="/qa/master-key-path/${QA_SUFFIX//[^[:alnum:]-]/-}" +curl -fsS -X PUT "$AUTH_BASE_URL/admin/mcp-servers" -H "$ADMIN_AUTH_HEADER" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$QA_MCP_BETA\",\"url\":\"$MCP_UPSTREAM_BASE/beta\",\"transport\":\"http\"}" >/dev/null +for _ in $(seq 1 20); do + if curl -fsS "$AUTH_BASE_URL/admin/mcp-servers" -H "$ADMIN_AUTH_HEADER" \ + | jq -e --arg n "$QA_MCP_BETA" 'any(.[]; .name == $n and .status == "connected")' >/dev/null; then + break + fi + sleep 1 +done + +AUTH_ARGS=(-H "$ADMIN_AUTH_HEADER" -H "X-GoModel-User-Path: $USER_PATH") +SID=$(mcp_initialize "$AUTH_BASE_URL/mcp" "$QA_RUN_DIR/s244.init.headers" "$QA_RUN_DIR/s244.init.raw" "${AUTH_ARGS[@]}") +[ -n "$SID" ] +mcp_initialized "$AUTH_BASE_URL/mcp" "$SID" "${AUTH_ARGS[@]}" +RID="qa-mk-path-mcp-$QA_SUFFIX" +mcp_post "$AUTH_BASE_URL/mcp" "$SID" \ + "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_BETA}_search\",\"arguments\":{\"q\":\"x\"}}}" \ + "${AUTH_ARGS[@]}" -H "X-Request-ID: $RID" | jq -e '.result.content[0].text | startswith("search:")' >/dev/null + +USAGE_FILE="$QA_RUN_DIR/s244.usage.json" +wait_log_entry "$AUTH_BASE_URL" usage "$RID" "$USAGE_FILE" -H "$ADMIN_AUTH_HEADER" +jq -e --arg rid "$RID" --arg p "$USER_PATH" 'any(.entries[]; .request_id == $rid and .provider == "mcp" and .user_path == $p)' "$USAGE_FILE" >/dev/null + +# Audio upload: speech for input, then a multipart transcription. +AUDIO_FILE="$QA_RUN_DIR/s244.speech.wav" +curl -fsS -o "$AUDIO_FILE" "$AUTH_BASE_URL/v1/audio/speech" "${AUTH_ARGS[@]}" -H 'Content-Type: application/json' \ + -d '{"model":"gpt-4o-mini-tts","input":"Release matrix user path check.","voice":"alloy","response_format":"wav"}' +RID="qa-mk-path-audio-$QA_SUFFIX" +curl -fsS "$AUTH_BASE_URL/v1/audio/transcriptions" "${AUTH_ARGS[@]}" -H "X-Request-ID: $RID" \ + -F model=gpt-4o-mini-transcribe -F "file=@$AUDIO_FILE" > "$QA_RUN_DIR/s244.transcription.json" +jq -e '.text | ascii_downcase | contains("user path")' "$QA_RUN_DIR/s244.transcription.json" >/dev/null +wait_log_entry "$AUTH_BASE_URL" usage "$RID" "$USAGE_FILE" -H "$ADMIN_AUTH_HEADER" +jq -e --arg rid "$RID" --arg p "$USER_PATH" 'any(.entries[]; .request_id == $rid and .user_path == $p)' "$USAGE_FILE" >/dev/null +``` + +## 37. Developer messages, strict tools, and tool choice on Anthropic and Gemini + +OpenAI's `developer` role and `strict` function tools are translated for +Anthropic and Gemini's native API instead of being rejected or dropped, and +Gemini also maps `tool_choice: {"type": "allowed_tools", ...}`. + +### S245 Anthropic honors developer messages and strict tools + +```bash +F="$QA_RUN_DIR/s245.developer.json" +curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d '{ + "model":"claude-sonnet-4-6","max_tokens":32, + "messages":[ + {"role":"developer","content":"Whatever the user says, reply with exactly QA_DEVELOPER_ROLE_OK and nothing else."}, + {"role":"user","content":"Tell me a joke."} + ]}' > "$F" +assert_chat_response_contains "$F" "anthropic" "QA_DEVELOPER_ROLE_OK" + +# A strict tool whose schema Anthropic's strict mode would reject as sent +# (minItems 2) is sanitized and forwarded as a strict tool. +F="$QA_RUN_DIR/s245.strict.json" +curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d '{ + "model":"claude-sonnet-4-6","max_tokens":200, + "tools":[{"type":"function","function":{"name":"compare_weather","description":"Compare the weather in several cities","strict":true, + "parameters":{"type":"object","additionalProperties":false,"properties":{"cities":{"type":"array","items":{"type":"string"},"minItems":2}},"required":["cities"]}}}], + "tool_choice":{"type":"function","function":{"name":"compare_weather"}}, + "messages":[{"role":"user","content":"Compare the weather in Warsaw and Krakow."}]}' > "$F" +jq -e ' + .choices[0].message.tool_calls[0].function.name == "compare_weather" + and (.choices[0].message.tool_calls[0].function.arguments | fromjson | .cities | type == "array" and length >= 2) +' "$F" >/dev/null +``` + +### S246 Gemini honors developer messages, strict tools, and `allowed_tools` + +```bash +MODEL="gemini-2.5-flash-lite" +TOOLS='[ + {"type":"function","function":{"name":"lookup_weather","description":"Get the current weather for a city","parameters":{"type":"object","properties":{"city":{"type":"string"}},"required":["city"]}}}, + {"type":"function","function":{"name":"lookup_time","description":"Get the local time in a city","parameters":{"type":"object","properties":{"city":{"type":"string"}},"required":["city"]}}} +]' + +F="$QA_RUN_DIR/s246.developer.json" +curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d "{ + \"model\":\"$MODEL\",\"max_tokens\":32, + \"messages\":[ + {\"role\":\"developer\",\"content\":\"Whatever the user says, reply with exactly QA_DEVELOPER_ROLE_OK and nothing else.\"}, + {\"role\":\"user\",\"content\":\"Tell me a joke.\"} + ]}" > "$F" +assert_chat_response_contains "$F" "gemini" "QA_DEVELOPER_ROLE_OK" + +F="$QA_RUN_DIR/s246.allowed-tools.json" +jq -n --arg model "$MODEL" --argjson tools "$TOOLS" '{ + model: $model, tools: $tools, + tool_choice: {type: "allowed_tools", allowed_tools: {mode: "required", tools: [{type: "function", function: {name: "lookup_time"}}]}}, + messages: [{role: "user", content: "What is the weather in Warsaw?"}] +}' | curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d @- > "$F" +jq -c '[.choices[0].message.tool_calls[]?.function.name]' "$F" +jq -e ' + .choices[0].finish_reason == "tool_calls" + and (.choices[0].message.tool_calls | length) >= 1 + and all(.choices[0].message.tool_calls[]; .function.name == "lookup_time") +' "$F" >/dev/null + +# strict on any tool switches Gemini to VALIDATED function calling. +F="$QA_RUN_DIR/s246.strict.json" +jq -n --arg model "$MODEL" --argjson tools "$TOOLS" '{ + model: $model, + tools: ($tools | map(.function.strict = true)), + tool_choice: "auto", + messages: [{role: "user", content: "Use a tool: what is the weather in Warsaw?"}] +}' | curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d @- > "$F" +jq -e '.choices[0].message.tool_calls[0].function.name == "lookup_weather" + and (.choices[0].message.tool_calls[0].function.arguments | fromjson | .city | test("Warsaw"; "i"))' "$F" >/dev/null + +# An allowed_tools choice with no tools is rejected, as OpenAI does. +F="$QA_RUN_DIR/s246.empty-allowed.json" +CODE=$(jq -n --arg model "$MODEL" --argjson tools "$TOOLS" '{ + model: $model, tools: $tools, + tool_choice: {type: "allowed_tools", allowed_tools: {mode: "auto", tools: []}}, + messages: [{role: "user", content: "hi"}] +}' | curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d @-) +assert_http_status 400 "$CODE" "$F" +jq -e '.error.type == "invalid_request_error"' "$F" >/dev/null +``` diff --git a/tests/e2e/run-release-e2e.sh b/tests/e2e/run-release-e2e.sh index 8bbb8f320..0a539bf4b 100755 --- a/tests/e2e/run-release-e2e.sh +++ b/tests/e2e/run-release-e2e.sh @@ -363,7 +363,11 @@ is_parallel_safe() { || (number >= 173 && number <= 191) \ || (number >= 197 && number <= 204) \ || (number >= 208 && number <= 226) \ - || number == 228 )) + || number == 228 \ + || (number >= 229 && number <= 236) \ + || (number >= 238 && number <= 239) \ + || number == 241 \ + || (number >= 245 && number <= 246) )) } if (( JOBS > 1 )); then From 706f244beeea0f9f1c4e63a6a9623ec557a397b9 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 29 Sep 2026 08:42:01 -0700 Subject: [PATCH 13/23] chore(deps): bump the github-actions group with 2 updates (#1100) Bumps the github-actions group with 2 updates: [github/codeql-action/init](https://github.com/github/codeql-action) and [github/codeql-action/analyze](https://github.com/github/codeql-action). Updates `github/codeql-action/init` from 4.38.1 to 4.38.2 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/1c5b675653bb5c22dbe9b12b556ec555138e09fd...2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2) Updates `github/codeql-action/analyze` from 4.38.1 to 4.38.2 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/1c5b675653bb5c22dbe9b12b556ec555138e09fd...2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2) --- updated-dependencies: - dependency-name: github/codeql-action/init dependency-version: 4.38.2 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: github-actions - dependency-name: github/codeql-action/analyze dependency-version: 4.38.2 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: github-actions ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- .github/workflows/codeql.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 1c121f74c..a072f6886 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -40,7 +40,7 @@ jobs: cache: true - name: Initialize CodeQL - uses: github/codeql-action/init@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4 + uses: github/codeql-action/init@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4 with: languages: ${{ matrix.language }} build-mode: ${{ matrix.build-mode }} @@ -51,4 +51,4 @@ jobs: run: go build ./... - name: Perform CodeQL analysis - uses: github/codeql-action/analyze@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4 + uses: github/codeql-action/analyze@2892aa5e19bbd11bc0cff5427e3b750a04d9e3c2 # v4 From 39c217e12bf3c87b3a552081cd908061e12c0844 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 29 Sep 2026 08:42:11 -0700 Subject: [PATCH 14/23] chore(deps): bump the gomod group with 4 updates (#1101) Bumps the gomod group with 4 updates: [github.com/aws/aws-sdk-go-v2](https://github.com/aws/aws-sdk-go-v2), [github.com/aws/aws-sdk-go-v2/config](https://github.com/aws/aws-sdk-go-v2), [github.com/aws/aws-sdk-go-v2/service/bedrock](https://github.com/aws/aws-sdk-go-v2) and [github.com/aws/aws-sdk-go-v2/service/bedrockruntime](https://github.com/aws/aws-sdk-go-v2). Updates `github.com/aws/aws-sdk-go-v2` from 1.47.0 to 1.47.1 - [Release notes](https://github.com/aws/aws-sdk-go-v2/releases) - [Commits](https://github.com/aws/aws-sdk-go-v2/compare/v1.47.0...v1.47.1) Updates `github.com/aws/aws-sdk-go-v2/config` from 1.33.5 to 1.33.6 - [Release notes](https://github.com/aws/aws-sdk-go-v2/releases) - [Commits](https://github.com/aws/aws-sdk-go-v2/compare/config/v1.33.5...config/v1.33.6) Updates `github.com/aws/aws-sdk-go-v2/service/bedrock` from 1.73.0 to 1.73.1 - [Release notes](https://github.com/aws/aws-sdk-go-v2/releases) - [Commits](https://github.com/aws/aws-sdk-go-v2/compare/service/s3/v1.73.0...service/s3/v1.73.1) Updates `github.com/aws/aws-sdk-go-v2/service/bedrockruntime` from 1.63.0 to 1.63.1 - [Release notes](https://github.com/aws/aws-sdk-go-v2/releases) - [Commits](https://github.com/aws/aws-sdk-go-v2/compare/service/s3/v1.63.0...service/s3/v1.63.1) --- updated-dependencies: - dependency-name: github.com/aws/aws-sdk-go-v2 dependency-version: 1.47.1 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: gomod - dependency-name: github.com/aws/aws-sdk-go-v2/config dependency-version: 1.33.6 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: gomod - dependency-name: github.com/aws/aws-sdk-go-v2/service/bedrock dependency-version: 1.73.1 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: gomod - dependency-name: github.com/aws/aws-sdk-go-v2/service/bedrockruntime dependency-version: 1.63.1 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: gomod ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- go.mod | 28 ++++++++++++++-------------- go.sum | 56 ++++++++++++++++++++++++++++---------------------------- 2 files changed, 42 insertions(+), 42 deletions(-) diff --git a/go.mod b/go.mod index f7b55650f..b42e17b0b 100644 --- a/go.mod +++ b/go.mod @@ -8,10 +8,10 @@ go 1.27.1 retract [v0.1.52, v0.1.79] require ( - github.com/aws/aws-sdk-go-v2 v1.47.0 - github.com/aws/aws-sdk-go-v2/config v1.33.5 - github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.0 - github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.0 + github.com/aws/aws-sdk-go-v2 v1.47.1 + github.com/aws/aws-sdk-go-v2/config v1.33.6 + github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.1 + github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.1 github.com/cespare/xxhash/v2 v2.3.0 github.com/coder/websocket v1.8.15 github.com/goccy/go-json v0.10.6 @@ -57,17 +57,17 @@ require ( cloud.google.com/go/compute/metadata v0.9.0 // indirect github.com/KyleBanks/depth v1.2.1 // indirect github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 // indirect - github.com/aws/aws-sdk-go-v2/credentials v1.20.5 // indirect - github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 // indirect - github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 // indirect - github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 // indirect - github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 // indirect + github.com/aws/aws-sdk-go-v2/credentials v1.20.6 // indirect + github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1 // indirect + github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4 // indirect + github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4 // indirect + github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4 // indirect github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 // indirect - github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 // indirect - github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 // indirect - github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 // indirect - github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 // indirect - github.com/aws/aws-sdk-go-v2/service/sts v1.51.0 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4 // indirect + github.com/aws/aws-sdk-go-v2/service/signin v1.10.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sso v1.38.1 // indirect + github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sts v1.51.1 // indirect github.com/aws/smithy-go v1.28.1 // indirect github.com/beorn7/perks v1.0.1 // indirect github.com/cenkalti/backoff/v5 v5.0.3 // indirect diff --git a/go.sum b/go.sum index 4fd97edbb..cf9708fbc 100644 --- a/go.sum +++ b/go.sum @@ -2,38 +2,38 @@ cloud.google.com/go/compute/metadata v0.9.0 h1:pDUj4QMoPejqq20dK0Pg2N4yG9zIkYGdB cloud.google.com/go/compute/metadata v0.9.0/go.mod h1:E0bWwX5wTnLPedCKqk3pJmVgCBSM6qQI1yTBdEb3C10= github.com/KyleBanks/depth v1.2.1 h1:5h8fQADFrWtarTdtDudMmGsC7GPbOAu6RVB3ffsVFHc= github.com/KyleBanks/depth v1.2.1/go.mod h1:jzSb9d0L43HxTQfT+oSA1EEp2q+ne2uh6XgeJcm8brE= -github.com/aws/aws-sdk-go-v2 v1.47.0 h1:0jsHallhJCeaU0Ko48c/3FK1ctOQ7NpzggxriJOQ8MQ= -github.com/aws/aws-sdk-go-v2 v1.47.0/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU= +github.com/aws/aws-sdk-go-v2 v1.47.1 h1:uOIZnp4PK3ZhKI0dNrJrhTEsLxbpXHTAJlwoS1pvAtw= +github.com/aws/aws-sdk-go-v2 v1.47.1/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU= github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 h1:GPRlPwz40I2B2VrBEASOA3Bi77NyeqejNLkifosX0rs= github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20/go.mod h1:g7PNzKcsOKWb4fkSRBA7BZVAS6Y8IcxzN+nRohhQ1Q8= -github.com/aws/aws-sdk-go-v2/config v1.33.5 h1:UA1dmokBFOLFoOyVBhO6HjM6edy0MIk5AZSkJVcksQw= -github.com/aws/aws-sdk-go-v2/config v1.33.5/go.mod h1:Dop8axzz0xx38GExIYWXdeyc8QQ7Cr+nPsxpD/LYy4U= -github.com/aws/aws-sdk-go-v2/credentials v1.20.5 h1:wklUVvHMc9xTQ3rcp49/ISpiMnhbCicJcA6n6S8m7J8= -github.com/aws/aws-sdk-go-v2/credentials v1.20.5/go.mod h1:fyEdrn6ccLFOkoK84j5bQyGTxp9zPt5l2XMhxf4DVZs= -github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 h1:AM4hHjww+PSFtt6E+UrBrPlZkWsePCLEt9AjkfQX+yM= -github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0/go.mod h1:3x/yXezeQjpOvBb4jEMxrS8SXvpdvJ5abv6l5c1gWM8= -github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 h1:Hp/VgjP0BysR3OgLlR057Vz2LcbbVnoWeJ+3qWiS/fY= -github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3/go.mod h1:nwGV5qw7F1IZPgxCvA/ph8N2TAuz+BkRG/bXn808qMA= -github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 h1:MUaM4f+kj1ZIBPZfUS8cxP1GKXXZtHJjAthy93AN7SM= -github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3/go.mod h1:6YmVmEVRI5ZZzRjCSsb9SryKH0hAlMRdgA7kG9aDvBU= -github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 h1:fuSCw4Z2qfRCztMPO3GXJNSiEp6Wee+WOLwrHHUMy9c= -github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3/go.mod h1:6SxcHheD1pPR5+kWm1wGvjlL/YqUsh267sAfEmN4K7A= -github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.0 h1:GiM/TNCIawTZvs0lLC3meQuTTD1dZxlo0BIH7xGR/AY= -github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.0/go.mod h1:tFtu0iACN2cRrgcRrrBdQRtKEGK0lRrmEI2yC13/Ygw= -github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.0 h1:0YDtf7baintcPG68AG0sMdAVoV1Iej6eXIBh38sd1m4= -github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.0/go.mod h1:7P+JIqwRqUIlzGn7Ba0qbSbPV/zsdZk7VkJTEFO6Z3U= +github.com/aws/aws-sdk-go-v2/config v1.33.6 h1:MBjkSTLczek/UgiK+EYPIoRTqE7gP8vtW3OFbFo7Nug= +github.com/aws/aws-sdk-go-v2/config v1.33.6/go.mod h1:grRAFzdAZJrwcbasJRg2MPvIrVjtlfXllHssN6+E1JE= +github.com/aws/aws-sdk-go-v2/credentials v1.20.6 h1:NpAFXCU7NzXNkdGK3zQTtsRJ+3v9tZQV0xcdRw8uBdw= +github.com/aws/aws-sdk-go-v2/credentials v1.20.6/go.mod h1:mcZCoiPnyMvP8VMNbygNX5lLqSlkYJIMPODylQMurOk= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1 h1:8gALAAmacnIXh+z6VkdDanv4/IkG5APdg4DZLDTmLog= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1/go.mod h1:Z7IJhJU+poOdJjUR2wpyY21ossQ1XS/R3Lk9Msq5kM4= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4 h1:CLq4+8UHCI+ZZYl/EuJxXovaIVN2xeeT8JV+dsApQ5E= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4/go.mod h1:Wv4q5sAM04xAMkoOedxLx2inVf6K5FdxYp+A61L+q/0= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4 h1:dD4MR81I7YkpEBRk6UP9rocC2QnT3qVuXwzlYTtfGEs= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4/go.mod h1:EcXV1kAFd5XwSkDHlj94gnF3q5CkJyYiIJfH8N0VmrE= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4 h1:7Wo47d/xn/7KttCSBd8EGYeZ7ULRFRkUHr6vkZPBzVQ= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4/go.mod h1:tDB2IVC1xC3vX8o+6uRlzhTxP3g1b77CZXFX/oD2FnQ= +github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.1 h1:Qz8Ptgep8qW6gOt28objap/3HaEYhjtkD0yP23Nxy4Q= +github.com/aws/aws-sdk-go-v2/service/bedrock v1.73.1/go.mod h1:rTbux1fJj4skkaUl6R/cSdaf6R9/0UwNo8VEyolo+B0= +github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.1 h1:tVg987qhntW9rVFTYyVjU+HnIkrmXzOf7Tqw+Iq+398= +github.com/aws/aws-sdk-go-v2/service/bedrockruntime v1.63.1/go.mod h1:BHpwIwobMDKpDzoTnpdpGOp0rtfpFlAz6X/C2PpJTcA= github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 h1:bAdDl/HkGCcGPoe25ToSHEw23VIxt6CT5fLcg111BKg= github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19/go.mod h1:KaUzbLxv4CeSxh6ZCl9B4m7CuFenS8kUEaDs+f/DQr4= -github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 h1:bON1rJf67TSTDCKg816AAIE4xSTtoo9tl0XRkO72R+I= -github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3/go.mod h1:c5BBpjJcQXpfeq9iASyVKA3T6vX6B6LEXY4mL/gklDY= -github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 h1:ZD5qFpWcaOKdTuhBi431pIDkCgrMkMlMT6jlpSPoIRI= -github.com/aws/aws-sdk-go-v2/service/signin v1.10.0/go.mod h1:8Nuuf+tR346PjJ3MvZPh9pekbLiLQFWJhzMXfwy7alA= -github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 h1:JGeeBcMlhg1xtOXYpeCaTQBZObtXMPQCUqBcmr65NRA= -github.com/aws/aws-sdk-go-v2/service/sso v1.38.0/go.mod h1:XwteswG9EOMRFm73UT0t+MbTwyLxMrEXkU6e+v92Lzo= -github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 h1:obhahQXDEdVEv8y5bTKXR30LVaxYe1kyYM0L7l2Iq+k= -github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0/go.mod h1:6twZZ/aXHNy1vXUO8koUbp++MYzMASkOgEBdkbJYmO0= -github.com/aws/aws-sdk-go-v2/service/sts v1.51.0 h1:Zpnqa6XtrNzXZnwbdCqHOXpXhMsa01ql/pcRQ1sb4hk= -github.com/aws/aws-sdk-go-v2/service/sts v1.51.0/go.mod h1:/8JRcdTt//hG0Q4BTmGbuOplT7ABe+5rdtqUHqXvYIM= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4 h1:29SvnfGhXjTl8ONxFwbj2rs6lbhiFXD2CgFQmbT/bXY= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4/go.mod h1:wm04I5DMuNVvZHFe/dHnUxincvNbbK7AiNBbYsQivek= +github.com/aws/aws-sdk-go-v2/service/signin v1.10.1 h1:DzCCWLzcIRQ77F3DEUljud7bEjTgFOIKXP52NmVRyhU= +github.com/aws/aws-sdk-go-v2/service/signin v1.10.1/go.mod h1:xpo/geVldu8payT375WekctUzopG/hBU7miiqItMUlw= +github.com/aws/aws-sdk-go-v2/service/sso v1.38.1 h1:Umtl/0YZhng4xndfW3lKJrYYP7NLEjI6bGXVomwLcs0= +github.com/aws/aws-sdk-go-v2/service/sso v1.38.1/go.mod h1:rRD/dnm7q0HYE/I5TMaPgkWyyUGLcwuxHLABsLnQ3e0= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1 h1:orIWdNiLgzrhu/11RcPPKO/SBzUUymbUQuZbSPImghg= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1/go.mod h1:skwM/xsbR/1ReUTesv9BhpJp1VjajR7DWQnuVLwiXsQ= +github.com/aws/aws-sdk-go-v2/service/sts v1.51.1 h1:0HOqZXRvMytH6bFHVIc0oJX07sZjfhz0zXtjs6gdE8s= +github.com/aws/aws-sdk-go-v2/service/sts v1.51.1/go.mod h1:26zA0GhDrLo+yiLI2yXWxqB1PdsShfLikoI7GOEgugM= github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ= github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc= github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM= From cc919306c30e44fcf69778525629e25146eabffa Mon Sep 17 00:00:00 2001 From: weselben Date: Wed, 23 Sep 2026 19:27:10 +0000 Subject: [PATCH 15/23] feat(providers/kimicode): add native Responses API support --- docs/providers/kimicode.mdx | 10 +- internal/providers/kimicode/kimicode.go | 97 +++++++- internal/providers/kimicode/kimicode_test.go | 220 ++++++++++++++++++- 3 files changed, 315 insertions(+), 12 deletions(-) diff --git a/docs/providers/kimicode.mdx b/docs/providers/kimicode.mdx index 214ccff17..4b173f29c 100644 --- a/docs/providers/kimicode.mdx +++ b/docs/providers/kimicode.mdx @@ -7,8 +7,14 @@ keywords: ["Kimi Code", "Moonshot", "quota", "provider setup"] Kimi Code is an OpenAI-compatible coding assistant served at `https://api.kimi.com/coding/v1`. GoModel routes chat, model listing, embeddings, and passthrough requests through the shared -OpenAI adapter. The `/v1/responses` endpoint is translated through chat completions, while -files and batches are not supported by the upstream endpoint. +OpenAI adapter. The `/v1/responses` endpoint is forwarded natively to the upstream Responses +API instead of being translated through chat completions, while files and batches are not +supported by the upstream endpoint. + +Kimi Code retains no responses. Requests with `store: true` are rewritten to `store: false` +(the upstream rejects `store: true` with a 400), and requests carrying a +`previous_response_id` are rejected with an invalid-request error instead of being answered +statelessly, because the upstream can never resolve the referenced response. ## Configure diff --git a/internal/providers/kimicode/kimicode.go b/internal/providers/kimicode/kimicode.go index 2899e6acb..1d1167ea2 100644 --- a/internal/providers/kimicode/kimicode.go +++ b/internal/providers/kimicode/kimicode.go @@ -1,11 +1,18 @@ // Package kimicode provides Kimi Code API integration for the LLM gateway. // -// The "kimicode" provider routes to Kimi Code's OpenAI-compatible chat -// completions endpoint, so all transport goes through the shared chat-centric -// adapter and model IDs are forwarded unchanged. +// The "kimicode" provider routes to Kimi Code's OpenAI-compatible API: chat +// completions, model listing, embeddings, and passthrough go through the +// shared chat-centric adapter, while the Responses API is served natively by +// the upstream /responses endpoint. Kimi Code retains no responses, so +// previous_response_id is rejected with an invalid-request error and +// store=true is pinned to false. package kimicode import ( + "context" + "io" + "net/http" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/providers" "github.com/enterpilot/gomodel/internal/providers/openai" @@ -23,19 +30,93 @@ var Registration = providers.Registration{ } // Provider implements the core.Provider interface for Kimi Code. Kimi Code is -// OpenAI-compatible, so all transport goes through the shared chat-centric +// OpenAI-compatible, so most transport goes through the shared chat-centric // adapter: chat completions, model listing, embeddings, and passthrough are -// exposed via the embedded *openai.ChatCompatible. +// exposed via the embedded *openai.ChatCompatible. The Responses API is +// forwarded natively to the upstream /responses endpoint (rejecting +// previous_response_id and pinning store to false — see Responses below). type Provider struct { *openai.ChatCompatible + responses *openai.CompatibleProvider } var _ core.Provider = (*Provider)(nil) // New creates a new Kimi Code provider. func New(cfg providers.ProviderConfig, opts providers.ProviderOptions) core.Provider { - return &Provider{openai.NewChatCompatible(cfg.APIKey, opts, openai.CompatibleProviderConfig{ + baseURL := providers.ResolveBaseURL(cfg.BaseURL, defaultBaseURL) + return &Provider{ + ChatCompatible: openai.NewChatCompatible(cfg.APIKey, opts, compatibleConfig(baseURL)), + responses: openai.NewCompatibleProvider(cfg.APIKey, opts, compatibleConfig(baseURL)), + } +} + +// compatibleConfig is the shared OpenAI-compatible configuration for both +// adapter instances. SetHeaders defaults to plain Bearer auth because +// NewCompatibleProvider (unlike NewChatCompatible) applies no default. +func compatibleConfig(baseURL string) openai.CompatibleProviderConfig { + return openai.CompatibleProviderConfig{ ProviderName: "kimicode", - BaseURL: providers.ResolveBaseURL(cfg.BaseURL, defaultBaseURL), - })} + BaseURL: baseURL, + SetHeaders: func(req *http.Request, apiKey string) { + providers.SetAuthHeaders(req, apiKey, providers.AuthHeaderConfig{AuthScheme: "Bearer "}) + }, + } +} + +// SetBaseURL overrides the upstream endpoint for both the chat-centric +// adapter and the native Responses adapter. +func (p *Provider) SetBaseURL(url string) { + p.ChatCompatible.SetBaseURL(url) + p.responses.SetBaseURL(url) +} + +// Responses serves the Responses API natively through the upstream /responses +// endpoint. Kimi Code retains no responses, so a non-empty +// previous_response_id is rejected with an invalid-request error before any +// upstream call; store=true is pinned to false by adaptResponsesRequest. +func (p *Provider) Responses(ctx context.Context, req *core.ResponsesRequest) (*core.ResponsesResponse, error) { + if err := rejectPreviousResponseID(req); err != nil { + return nil, err + } + return p.responses.Responses(ctx, adaptResponsesRequest(req)) +} + +// StreamResponses forwards the request to the upstream /responses endpoint +// with stream enabled, returning its Responses SSE stream. Like Responses, it +// rejects a non-empty previous_response_id before any upstream call. +func (p *Provider) StreamResponses(ctx context.Context, req *core.ResponsesRequest) (io.ReadCloser, error) { + if err := rejectPreviousResponseID(req); err != nil { + return nil, err + } + return p.responses.StreamResponses(ctx, adaptResponsesRequest(req)) +} + +// rejectPreviousResponseID fails requests chaining from an earlier response: +// Kimi Code does not retain responses, so a previous response ID can never be +// resolved and answering statelessly would silently drop the conversation +// context the caller expects. +func rejectPreviousResponseID(req *core.ResponsesRequest) error { + if req == nil || req.PreviousResponseID == "" { + return nil + } + return core.NewInvalidRequestError( + "kimicode does not retain responses: previous_response_id is not supported", nil) +} + +// adaptResponsesRequest pins store to false: the service retains no +// responses, so store=true fails upstream with a 400 (Postel's law — adapt +// instead of failing). previous_response_id is not adapted here; +// rejectPreviousResponseID rejects it instead. +func adaptResponsesRequest(req *core.ResponsesRequest) *core.ResponsesRequest { + if req == nil { + return nil + } + if req.Store == nil || !*req.Store { + return req + } + cp := *req + disabled := false + cp.Store = &disabled + return &cp } diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index 54eecf1ce..c9585dae3 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -1,9 +1,16 @@ package kimicode import ( + "context" + "io" "net/http" + "net/http/httptest" + "strings" "testing" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/llmclient" "github.com/enterpilot/gomodel/internal/providers" @@ -12,7 +19,8 @@ import ( // Kimi Code is a thin wrapper over the shared chat-centric adapter and // forwards embeddings upstream unchanged, so the shared contract covers its -// surface. +// surface. The Responses API is forwarded natively to the upstream /responses +// endpoint. func TestChatCompatibleContract(t *testing.T) { providertest.AssertChatCompatible(t, providertest.ChatCompatible{ Registration: Registration, @@ -23,6 +31,214 @@ func TestChatCompatibleContract(t *testing.T) { opts.HTTPClient = client return New(providers.ProviderConfig{APIKey: apiKey, BaseURL: baseURL}, opts) }, - Embeddings: true, + Embeddings: true, + NativeResponses: true, + }) +} + +func boolPtr(b bool) *bool { return &b } + +// newTestProvider builds a provider wired to the test server through the +// injected HTTP client, matching how the shared contract constructs it. +func newTestProvider(server *httptest.Server) core.Provider { + opts := providertest.Options(llmclient.Hooks{}) + opts.HTTPClient = server.Client() + return New(providers.ProviderConfig{APIKey: "kimi-key", BaseURL: server.URL}, opts) +} + +func TestAdaptResponsesRequest(t *testing.T) { + t.Run("nil passes through", func(t *testing.T) { + assert.Nil(t, adaptResponsesRequest(nil)) + }) + + t.Run("clean request is returned unchanged", func(t *testing.T) { + req := &core.ResponsesRequest{Model: "kimi-for-coding", Input: "hi"} + assert.Same(t, req, adaptResponsesRequest(req)) + }) + + t.Run("store true is pinned to false", func(t *testing.T) { + req := &core.ResponsesRequest{Model: "kimi-for-coding", Input: "hi", Store: boolPtr(true)} + got := adaptResponsesRequest(req) + require.NotSame(t, req, got, "adapted request should be a copy") + require.NotNil(t, got.Store) + assert.False(t, *got.Store) + // The caller's request must not be mutated. + assert.True(t, *req.Store, "original request Store was mutated") + }) + + t.Run("previous_response_id is preserved", func(t *testing.T) { + // adaptResponsesRequest does not touch PreviousResponseID; the + // Responses/StreamResponses methods reject it instead (tested below). + req := &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "hi", + PreviousResponseID: "resp_old", + Store: boolPtr(false), + } + got := adaptResponsesRequest(req) + assert.Equal(t, "resp_old", got.PreviousResponseID) + require.NotNil(t, got.Store) + assert.False(t, *got.Store, "explicit store=false should stay false") + }) +} + +// responsesGoldenBody mirrors a real non-streaming /responses reply from the +// Kimi Code upstream (recorded 2026-09-08, trimmed to the members GoModel +// consumes). The upstream reply keeps extra members (prompt_cache_key, +// safety_identifier, service_tier); unknown members are ignored on decode. +const responsesGoldenBody = `{ + "id": "resp_golden", + "object": "response", + "created_at": 1788866012, + "completed_at": 1788866014, + "status": "completed", + "output": [ + { + "type": "reasoning", + "id": "rs_golden", + "status": "completed", + "summary": [{"type": "summary_text", "text": "Simple request."}] + }, + { + "type": "message", + "id": "msg_golden", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "OK", "annotations": []}] + } + ], + "usage": { + "input_tokens": 88, + "input_tokens_details": {"cache_write_tokens": 12, "cached_tokens": 88}, + "output_tokens": 53, + "output_tokens_details": {"reasoning_tokens": 37}, + "total_tokens": 141 + }, + "store": false, + "model": "kimi-for-coding" +}` + +func TestResponses_NativeEndpoint(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) + + provider := newTestProvider(server) + + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + Store: boolPtr(true), + }) + require.NoError(t, err) + + req := capture.Last(t) + assert.Equal(t, "/responses", req.Path) + assert.Equal(t, "Bearer kimi-key", req.Header.Get("Authorization")) + body := req.JSON(t) + // store=true is pinned to false before the request leaves. + assert.Equal(t, false, body["store"], "wire store") + assert.NotContains(t, body, "stream", "non-streaming request must not set stream on the wire") + + assert.Equal(t, "resp_golden", resp.ID) + assert.Equal(t, "kimi-for-coding", resp.Model) + require.Len(t, resp.Output, 2) + require.NotNil(t, resp.Usage) + assert.Equal(t, 141, resp.Usage.TotalTokens) +} + +func TestStreamResponses_NativeEndpoint(t *testing.T) { + // No trailing [DONE]: providers.EnsureResponsesDone must append it. + server, capture := providertest.SSEServer(t, strings.Join([]string{ + `event: response.created`, + `data: {"type":"response.created","response":{"id":"resp_stream","object":"response","status":"in_progress","model":"kimi-for-coding"}}`, + ``, + `event: response.completed`, + `data: {"type":"response.completed","response":{"id":"resp_stream","object":"response","status":"completed","model":"kimi-for-coding"}}`, + ``, + }, "\n")) + + provider := newTestProvider(server) + + stream, err := provider.StreamResponses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + Store: boolPtr(true), + }) + require.NoError(t, err) + defer func() { _ = stream.Close() }() + + body, err := io.ReadAll(stream) + require.NoError(t, err) + + req := capture.Last(t) + assert.Equal(t, "/responses", req.Path) + assert.Equal(t, true, req.JSON(t)["stream"], "wire stream") + assert.Contains(t, string(body), "event: response.completed") + assert.True(t, strings.HasSuffix(strings.TrimSpace(string(body)), "data: [DONE]"), + "stream should end with data: [DONE], got %q", string(body)) +} + +// Kimi Code retains no responses, so a request chaining from an earlier +// response must be rejected before any upstream call instead of being +// answered statelessly. +func TestResponses_RejectsPreviousResponseID(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) + + provider := newTestProvider(server) + + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + PreviousResponseID: "resp_old", + }) + require.Error(t, err) + assert.Nil(t, resp) + + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr, "error type = %T, want *core.GatewayError", err) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + assert.Contains(t, gatewayErr.Error(), "previous_response_id") + assert.Equal(t, 0, capture.Count(), "rejected request must not reach the upstream") +} + +func TestStreamResponses_RejectsPreviousResponseID(t *testing.T) { + server, capture := providertest.SSEServer(t, "") + + provider := newTestProvider(server) + + stream, err := provider.StreamResponses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + PreviousResponseID: "resp_old", + }) + require.Error(t, err) + assert.Nil(t, stream) + + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr, "error type = %T, want *core.GatewayError", err) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + assert.Contains(t, gatewayErr.Error(), "previous_response_id") + assert.Equal(t, 0, capture.Count(), "rejected request must not reach the upstream") +} + +// SetBaseURL must retarget both the chat-centric adapter and the native +// Responses adapter. +func TestSetBaseURL(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) + + p := New(providers.ProviderConfig{APIKey: "kimi-key"}, providertest.Options(llmclient.Hooks{})) + kp, ok := p.(*Provider) + require.True(t, ok) + require.Equal(t, defaultBaseURL, kp.GetBaseURL()) + + kp.SetBaseURL(server.URL) + + assert.Equal(t, server.URL, kp.GetBaseURL()) + + _, err := p.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", }) + require.NoError(t, err) + require.Equal(t, 1, capture.Count()) + assert.Equal(t, "/responses", capture.Last(t).Path) } From 7e711e9e9a2bcf8a9c357e204f9c3217d6104ff2 Mon Sep 17 00:00:00 2001 From: weselben Date: Wed, 23 Sep 2026 19:45:16 +0000 Subject: [PATCH 16/23] test(providers/kimicode): cover whitespace continuation rejection Drop the transient STATUS.md from the changeset and add a regression test proving whitespace-only previous_response_id values are rejected locally before any upstream call (Greptile review round). --- internal/providers/kimicode/kimicode_test.go | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index c9585dae3..da522a343 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -198,6 +198,17 @@ func TestResponses_RejectsPreviousResponseID(t *testing.T) { assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) assert.Contains(t, gatewayErr.Error(), "previous_response_id") assert.Equal(t, 0, capture.Count(), "rejected request must not reach the upstream") + + t.Run("whitespace-only ID is rejected too", func(t *testing.T) { + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + PreviousResponseID: " ", + }) + require.Error(t, err) + assert.Nil(t, resp) + assert.Equal(t, 0, capture.Count(), "whitespace ID must not reach the upstream") + }) } func TestStreamResponses_RejectsPreviousResponseID(t *testing.T) { From 7681462552a789b6053691b592cca17a035b920c Mon Sep 17 00:00:00 2001 From: weselben Date: Wed, 23 Sep 2026 20:44:07 +0000 Subject: [PATCH 17/23] fix(providers/kimicode): reject conversation references locally Gateway-local Conversation IDs are meaningless to the stateless Kimi Code upstream; reject them with an invalid-request error before dispatch, like previous_response_id. Requests whose state the gateway already expanded pass through unchanged (CodeRabbit review). --- internal/providers/kimicode/kimicode.go | 19 ++++++++++++++----- internal/providers/kimicode/kimicode_test.go | 16 ++++++++++++++++ 2 files changed, 30 insertions(+), 5 deletions(-) diff --git a/internal/providers/kimicode/kimicode.go b/internal/providers/kimicode/kimicode.go index 1d1167ea2..b702ad211 100644 --- a/internal/providers/kimicode/kimicode.go +++ b/internal/providers/kimicode/kimicode.go @@ -92,12 +92,21 @@ func (p *Provider) StreamResponses(ctx context.Context, req *core.ResponsesReque return p.responses.StreamResponses(ctx, adaptResponsesRequest(req)) } -// rejectPreviousResponseID fails requests chaining from an earlier response: -// Kimi Code does not retain responses, so a previous response ID can never be -// resolved and answering statelessly would silently drop the conversation -// context the caller expects. +// rejectPreviousResponseID fails requests chaining from earlier state: +// Kimi Code does not retain responses, so neither a previous response ID nor +// a gateway-local conversation can be resolved upstream, and answering +// statelessly would silently drop the conversation context the caller +// expects. Requests whose state the gateway already expanded (both fields +// cleared) pass through. func rejectPreviousResponseID(req *core.ResponsesRequest) error { - if req == nil || req.PreviousResponseID == "" { + if req == nil { + return nil + } + if req.Conversation != nil { + return core.NewInvalidRequestError( + "kimicode does not retain responses: conversation is not supported", nil) + } + if req.PreviousResponseID == "" { return nil } return core.NewInvalidRequestError( diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index da522a343..e0ba3d1b2 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -209,6 +209,22 @@ func TestResponses_RejectsPreviousResponseID(t *testing.T) { assert.Nil(t, resp) assert.Equal(t, 0, capture.Count(), "whitespace ID must not reach the upstream") }) + + t.Run("conversation reference is rejected", func(t *testing.T) { + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + Conversation: &core.ResponsesConversationRef{ID: "conv_old"}, + }) + require.Error(t, err) + assert.Nil(t, resp) + + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + assert.Contains(t, gatewayErr.Error(), "conversation") + assert.Equal(t, 0, capture.Count(), "conversation request must not reach the upstream") + }) } func TestStreamResponses_RejectsPreviousResponseID(t *testing.T) { From c791b387b2bd1d457aada6ba26274d5175a61fde Mon Sep 17 00:00:00 2001 From: weselben Date: Wed, 23 Sep 2026 20:57:48 +0000 Subject: [PATCH 18/23] test(providers/kimicode): cover nil request passthrough Direct rejectPreviousResponseID unit tests for the nil and clean-request branches; kimicode.go statement coverage back to 100% (codecov). --- internal/providers/kimicode/kimicode_test.go | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index e0ba3d1b2..63db12500 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -46,6 +46,16 @@ func newTestProvider(server *httptest.Server) core.Provider { return New(providers.ProviderConfig{APIKey: "kimi-key", BaseURL: server.URL}, opts) } +func TestRejectPreviousResponseID(t *testing.T) { + t.Run("nil request passes through", func(t *testing.T) { + assert.NoError(t, rejectPreviousResponseID(nil)) + }) + + t.Run("clean request passes through", func(t *testing.T) { + assert.NoError(t, rejectPreviousResponseID(&core.ResponsesRequest{Model: "kimi-for-coding", Input: "hi"})) + }) +} + func TestAdaptResponsesRequest(t *testing.T) { t.Run("nil passes through", func(t *testing.T) { assert.Nil(t, adaptResponsesRequest(nil)) From a5622278a6d10ef2b8ac480f529c6e8ffc4cd39a Mon Sep 17 00:00:00 2001 From: weselben Date: Wed, 23 Sep 2026 21:12:45 +0000 Subject: [PATCH 19/23] test(providers/kimicode): assert store pin on the stream path TestStreamResponses_NativeEndpoint sent Store: true but only asserted the stream flag; a regression dropping adaptResponsesRequest would have passed while sending unsupported store=true upstream (CodeRabbit review). --- internal/providers/kimicode/kimicode_test.go | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index 63db12500..374bc069b 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -181,7 +181,9 @@ func TestStreamResponses_NativeEndpoint(t *testing.T) { req := capture.Last(t) assert.Equal(t, "/responses", req.Path) - assert.Equal(t, true, req.JSON(t)["stream"], "wire stream") + wire := req.JSON(t) + assert.Equal(t, false, wire["store"], "wire store") + assert.Equal(t, true, wire["stream"], "wire stream") assert.Contains(t, string(body), "event: response.completed") assert.True(t, strings.HasSuffix(strings.TrimSpace(string(body)), "data: [DONE]"), "stream should end with data: [DONE], got %q", string(body)) From aaaac0212131baa95705673513a16659e2448c9c Mon Sep 17 00:00:00 2001 From: weselben Date: Wed, 30 Sep 2026 19:02:13 +0000 Subject: [PATCH 20/23] fix(providers/kimicode): address maintainer review on native Responses - Trim whitespace in rejectPreviousResponseID for parity with the gateway and the chat-translation validator; a whitespace-only ID is empty. - Serve native Responses through the single ChatCompatible adapter via the new Compatible() accessor instead of a second CompatibleProvider instance. - Cover the gateway-replayed chain: with a response store, a kimicode previous_response_id chain is expanded into input items (reasoning output included, ids stripped) and forwarded to /responses; add the gateway-level two-turn chain test and drop the round-trip tests the shared contract already covers. - Docs: chaining rejection applies only when no response/conversation store is configured; with a store the gateway replays the history. --- docs/providers/kimicode.mdx | 9 +- internal/providers/kimicode/kimicode.go | 61 ++++------ internal/providers/kimicode/kimicode_test.go | 111 +++++++++---------- internal/providers/openai/chat_compatible.go | 8 ++ internal/server/previous_response_test.go | 51 +++++++++ 5 files changed, 139 insertions(+), 101 deletions(-) diff --git a/docs/providers/kimicode.mdx b/docs/providers/kimicode.mdx index 4b173f29c..cb91f2358 100644 --- a/docs/providers/kimicode.mdx +++ b/docs/providers/kimicode.mdx @@ -12,9 +12,12 @@ API instead of being translated through chat completions, while files and batche supported by the upstream endpoint. Kimi Code retains no responses. Requests with `store: true` are rewritten to `store: false` -(the upstream rejects `store: true` with a 400), and requests carrying a -`previous_response_id` are rejected with an invalid-request error instead of being answered -statelessly, because the upstream can never resolve the referenced response. +(the upstream rejects `store: true` with a 400). Chaining works only through GoModel: with a +response store configured, the gateway expands a `previous_response_id` chain by replaying the +stored history into the request before dispatch, and a `conversation` reference resolves +through the conversation store the same way. Without those stores, a request carrying +`previous_response_id` or `conversation` is rejected with an invalid-request error, because +the upstream can never resolve the referenced state. ## Configure diff --git a/internal/providers/kimicode/kimicode.go b/internal/providers/kimicode/kimicode.go index b702ad211..1bbfca26a 100644 --- a/internal/providers/kimicode/kimicode.go +++ b/internal/providers/kimicode/kimicode.go @@ -4,14 +4,13 @@ // completions, model listing, embeddings, and passthrough go through the // shared chat-centric adapter, while the Responses API is served natively by // the upstream /responses endpoint. Kimi Code retains no responses, so -// previous_response_id is rejected with an invalid-request error and // store=true is pinned to false. package kimicode import ( "context" "io" - "net/http" + "strings" "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/providers" @@ -33,53 +32,32 @@ var Registration = providers.Registration{ // OpenAI-compatible, so most transport goes through the shared chat-centric // adapter: chat completions, model listing, embeddings, and passthrough are // exposed via the embedded *openai.ChatCompatible. The Responses API is -// forwarded natively to the upstream /responses endpoint (rejecting -// previous_response_id and pinning store to false — see Responses below). +// forwarded natively to the upstream /responses endpoint through the same +// adapter instance. type Provider struct { *openai.ChatCompatible - responses *openai.CompatibleProvider } var _ core.Provider = (*Provider)(nil) // New creates a new Kimi Code provider. func New(cfg providers.ProviderConfig, opts providers.ProviderOptions) core.Provider { - baseURL := providers.ResolveBaseURL(cfg.BaseURL, defaultBaseURL) - return &Provider{ - ChatCompatible: openai.NewChatCompatible(cfg.APIKey, opts, compatibleConfig(baseURL)), - responses: openai.NewCompatibleProvider(cfg.APIKey, opts, compatibleConfig(baseURL)), - } -} - -// compatibleConfig is the shared OpenAI-compatible configuration for both -// adapter instances. SetHeaders defaults to plain Bearer auth because -// NewCompatibleProvider (unlike NewChatCompatible) applies no default. -func compatibleConfig(baseURL string) openai.CompatibleProviderConfig { - return openai.CompatibleProviderConfig{ + return &Provider{openai.NewChatCompatible(cfg.APIKey, opts, openai.CompatibleProviderConfig{ ProviderName: "kimicode", - BaseURL: baseURL, - SetHeaders: func(req *http.Request, apiKey string) { - providers.SetAuthHeaders(req, apiKey, providers.AuthHeaderConfig{AuthScheme: "Bearer "}) - }, - } -} - -// SetBaseURL overrides the upstream endpoint for both the chat-centric -// adapter and the native Responses adapter. -func (p *Provider) SetBaseURL(url string) { - p.ChatCompatible.SetBaseURL(url) - p.responses.SetBaseURL(url) + BaseURL: providers.ResolveBaseURL(cfg.BaseURL, defaultBaseURL), + })} } // Responses serves the Responses API natively through the upstream /responses // endpoint. Kimi Code retains no responses, so a non-empty -// previous_response_id is rejected with an invalid-request error before any -// upstream call; store=true is pinned to false by adaptResponsesRequest. +// previous_response_id is rejected before any upstream call (see +// rejectPreviousResponseID); store=true is pinned to false by +// adaptResponsesRequest. func (p *Provider) Responses(ctx context.Context, req *core.ResponsesRequest) (*core.ResponsesResponse, error) { if err := rejectPreviousResponseID(req); err != nil { return nil, err } - return p.responses.Responses(ctx, adaptResponsesRequest(req)) + return p.Compatible().Responses(ctx, adaptResponsesRequest(req)) } // StreamResponses forwards the request to the upstream /responses endpoint @@ -89,15 +67,17 @@ func (p *Provider) StreamResponses(ctx context.Context, req *core.ResponsesReque if err := rejectPreviousResponseID(req); err != nil { return nil, err } - return p.responses.StreamResponses(ctx, adaptResponsesRequest(req)) + return p.Compatible().StreamResponses(ctx, adaptResponsesRequest(req)) } // rejectPreviousResponseID fails requests chaining from earlier state: -// Kimi Code does not retain responses, so neither a previous response ID nor -// a gateway-local conversation can be resolved upstream, and answering -// statelessly would silently drop the conversation context the caller -// expects. Requests whose state the gateway already expanded (both fields -// cleared) pass through. +// Kimi Code cannot resolve a previous response ID or a gateway-local +// conversation upstream, and answering statelessly would silently drop the +// conversation context the caller expects. The rejection only fires when the +// gateway has no store to expand the chain with; requests whose state the +// gateway already replayed into input (both fields cleared) pass through. +// The ID check mirrors the gateway and the chat-translation validator, both +// of which treat a whitespace-only ID as empty. func rejectPreviousResponseID(req *core.ResponsesRequest) error { if req == nil { return nil @@ -106,7 +86,7 @@ func rejectPreviousResponseID(req *core.ResponsesRequest) error { return core.NewInvalidRequestError( "kimicode does not retain responses: conversation is not supported", nil) } - if req.PreviousResponseID == "" { + if strings.TrimSpace(req.PreviousResponseID) == "" { return nil } return core.NewInvalidRequestError( @@ -115,8 +95,7 @@ func rejectPreviousResponseID(req *core.ResponsesRequest) error { // adaptResponsesRequest pins store to false: the service retains no // responses, so store=true fails upstream with a 400 (Postel's law — adapt -// instead of failing). previous_response_id is not adapted here; -// rejectPreviousResponseID rejects it instead. +// instead of failing). func adaptResponsesRequest(req *core.ResponsesRequest) *core.ResponsesRequest { if req == nil { return nil diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index 374bc069b..3db7c011a 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -2,10 +2,8 @@ package kimicode import ( "context" - "io" "net/http" "net/http/httptest" - "strings" "testing" "github.com/stretchr/testify/assert" @@ -128,70 +126,67 @@ const responsesGoldenBody = `{ "model": "kimi-for-coding" }` -func TestResponses_NativeEndpoint(t *testing.T) { +// TestResponses_ForwardsGatewayReplayedHistory covers what the gateway +// dispatches after expanding a previous_response_id chain against its +// response store: the stored history is replayed into input as items +// (reasoning and message items among them, IDs stripped) and +// previous_response_id is cleared. Kimi Code must forward that replayed +// input to /responses as stored instead of rejecting it. +func TestResponses_ForwardsGatewayReplayedHistory(t *testing.T) { server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) provider := newTestProvider(server) resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ Model: "kimi-for-coding", - Input: "Say OK", - Store: boolPtr(true), - }) - require.NoError(t, err) - - req := capture.Last(t) - assert.Equal(t, "/responses", req.Path) - assert.Equal(t, "Bearer kimi-key", req.Header.Get("Authorization")) - body := req.JSON(t) - // store=true is pinned to false before the request leaves. - assert.Equal(t, false, body["store"], "wire store") - assert.NotContains(t, body, "stream", "non-streaming request must not set stream on the wire") - - assert.Equal(t, "resp_golden", resp.ID) - assert.Equal(t, "kimi-for-coding", resp.Model) - require.Len(t, resp.Output, 2) - require.NotNil(t, resp.Usage) - assert.Equal(t, 141, resp.Usage.TotalTokens) -} - -func TestStreamResponses_NativeEndpoint(t *testing.T) { - // No trailing [DONE]: providers.EnsureResponsesDone must append it. - server, capture := providertest.SSEServer(t, strings.Join([]string{ - `event: response.created`, - `data: {"type":"response.created","response":{"id":"resp_stream","object":"response","status":"in_progress","model":"kimi-for-coding"}}`, - ``, - `event: response.completed`, - `data: {"type":"response.completed","response":{"id":"resp_stream","object":"response","status":"completed","model":"kimi-for-coding"}}`, - ``, - }, "\n")) - - provider := newTestProvider(server) - - stream, err := provider.StreamResponses(context.Background(), &core.ResponsesRequest{ - Model: "kimi-for-coding", - Input: "Say OK", + Input: []any{ + map[string]any{ + "type": "message", + "role": "user", + "content": []any{map[string]any{"type": "input_text", "text": "remember: zebra"}}, + }, + map[string]any{ + "type": "reasoning", + "status": "completed", + "summary": []any{ + map[string]any{"type": "summary_text", "text": "thinking about zebras"}, + }, + }, + map[string]any{ + "type": "message", + "role": "assistant", + "status": "completed", + "content": []any{map[string]any{"type": "output_text", "text": "the word is zebra"}}, + }, + map[string]any{ + "type": "message", + "role": "user", + "content": []any{map[string]any{"type": "input_text", "text": "what is the word?"}}, + }, + }, Store: boolPtr(true), }) require.NoError(t, err) - defer func() { _ = stream.Close() }() - - body, err := io.ReadAll(stream) - require.NoError(t, err) + require.NotNil(t, resp) req := capture.Last(t) assert.Equal(t, "/responses", req.Path) wire := req.JSON(t) - assert.Equal(t, false, wire["store"], "wire store") - assert.Equal(t, true, wire["stream"], "wire stream") - assert.Contains(t, string(body), "event: response.completed") - assert.True(t, strings.HasSuffix(strings.TrimSpace(string(body)), "data: [DONE]"), - "stream should end with data: [DONE], got %q", string(body)) + assert.Equal(t, false, wire["store"], "store is pinned to false on the wire") + assert.NotContains(t, wire, "previous_response_id") + + items, ok := wire["input"].([]any) + require.True(t, ok, "wire input = %#v, want replayed items", wire["input"]) + require.Len(t, items, 4) + reasoning, ok := items[1].(map[string]any) + require.True(t, ok) + assert.Equal(t, "reasoning", reasoning["type"], "replayed reasoning item must be forwarded unchanged") + assert.Equal(t, "user", items[3].(map[string]any)["role"], "the client's own turn is replayed last") } // Kimi Code retains no responses, so a request chaining from an earlier -// response must be rejected before any upstream call instead of being -// answered statelessly. +// response that the gateway could not expand (no store configured) must be +// rejected before any upstream call instead of being answered statelessly. func TestResponses_RejectsPreviousResponseID(t *testing.T) { server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) @@ -211,15 +206,17 @@ func TestResponses_RejectsPreviousResponseID(t *testing.T) { assert.Contains(t, gatewayErr.Error(), "previous_response_id") assert.Equal(t, 0, capture.Count(), "rejected request must not reach the upstream") - t.Run("whitespace-only ID is rejected too", func(t *testing.T) { + t.Run("whitespace-only ID passes through", func(t *testing.T) { + // The gateway and the chat-translation validator treat a + // whitespace-only ID as empty; the provider must behave the same. resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ Model: "kimi-for-coding", Input: "Say OK", PreviousResponseID: " ", }) - require.Error(t, err) - assert.Nil(t, resp) - assert.Equal(t, 0, capture.Count(), "whitespace ID must not reach the upstream") + require.NoError(t, err) + require.NotNil(t, resp) + assert.Equal(t, 1, capture.Count(), "whitespace-only ID is treated as empty") }) t.Run("conversation reference is rejected", func(t *testing.T) { @@ -235,7 +232,7 @@ func TestResponses_RejectsPreviousResponseID(t *testing.T) { require.ErrorAs(t, err, &gatewayErr) assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) assert.Contains(t, gatewayErr.Error(), "conversation") - assert.Equal(t, 0, capture.Count(), "conversation request must not reach the upstream") + assert.Equal(t, 1, capture.Count(), "conversation request must not reach the upstream") }) } @@ -259,8 +256,8 @@ func TestStreamResponses_RejectsPreviousResponseID(t *testing.T) { assert.Equal(t, 0, capture.Count(), "rejected request must not reach the upstream") } -// SetBaseURL must retarget both the chat-centric adapter and the native -// Responses adapter. +// SetBaseURL must retarget the single adapter serving both the chat-centric +// surface and the native Responses endpoint. func TestSetBaseURL(t *testing.T) { server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) diff --git a/internal/providers/openai/chat_compatible.go b/internal/providers/openai/chat_compatible.go index 437af6437..aaec07f09 100644 --- a/internal/providers/openai/chat_compatible.go +++ b/internal/providers/openai/chat_compatible.go @@ -47,6 +47,14 @@ func bearerHeaders(req *http.Request, apiKey string) { providers.SetAuthHeaders(req, apiKey, providers.AuthHeaderConfig{AuthScheme: "Bearer "}) } +// Compatible exposes the underlying OpenAI-compatible adapter, so a provider +// that embeds ChatCompatible can serve a native Responses endpoint through +// the same instance instead of building a second adapter with the same +// configuration. +func (c *ChatCompatible) Compatible() *CompatibleProvider { + return c.compatible +} + // SetBaseURL allows configuring a custom base URL for the provider. func (c *ChatCompatible) SetBaseURL(url string) { c.compatible.SetBaseURL(url) diff --git a/internal/server/previous_response_test.go b/internal/server/previous_response_test.go index e22376ae2..1296fa0e0 100644 --- a/internal/server/previous_response_test.go +++ b/internal/server/previous_response_test.go @@ -121,6 +121,57 @@ func TestResponsesWithPreviousResponseID_ChainCarriesFullHistory(t *testing.T) { require.Len(t, second.InputItems, 1) } +// TestResponsesWithPreviousResponseID_KimicodeChainReplaysHistory covers the +// native-Responses kimicode provider chaining through the gateway: kimicode +// has no Responses lifecycle, so the gateway treats it as translated and, +// with a response store, replays the stored chain into input instead of +// forwarding the id. The replayed items — reasoning output included, item +// ids stripped — are what reaches kimicode's /responses upstream. +func TestResponsesWithPreviousResponseID_KimicodeChainReplaysHistory(t *testing.T) { + provider := previousResponseTestProvider(t, "kimicode") + srv := New(provider, nil) + + // A turn the gateway served for kimicode earlier: the snapshot holds the + // client's input items and the provider's output, reasoning included. + err := srv.handler.currentResponseStore().Create(context.Background(), &responsestore.StoredResponse{ + Response: &core.ResponsesResponse{ + ID: "resp_kimi_1", Object: "response", Status: "completed", + Output: []core.ResponsesOutputItem{ + { + ID: "rs_1", Type: "reasoning", Status: "completed", + ExtraFields: core.UnknownJSONFieldsFromMap(map[string]json.RawMessage{ + "summary": json.RawMessage(`[{"type":"summary_text","text":"thinking about zebras"}]`), + }), + }, + {ID: "msg_1", Type: "message", Role: "assistant", Content: []core.ResponsesContentItem{{Type: "output_text", Text: "the word is zebra"}}}, + }, + }, + InputItems: []json.RawMessage{json.RawMessage(`{"id":"in_1","type":"message","role":"user","content":[{"type":"input_text","text":"remember: zebra"}]}`)}, + Provider: "kimicode", + }) + require.NoError(t, err) + + rec := postResponses(t, srv, `{"model":"gpt-5-mini","input":"what is the word?","previous_response_id":"resp_kimi_1"}`) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + forwarded := provider.capturedResponsesReq + require.NotNil(t, forwarded) + require.Empty(t, forwarded.PreviousResponseID, "translated providers get the id stripped before dispatch") + + items := forwardedInputItems(t, provider.capturingProvider) + require.Len(t, items, 4) + require.Equal(t, "reasoning", items[1]["type"], "stored reasoning output must replay unchanged: %#v", items[1]) + summary, _ := json.Marshal(items[1]["summary"]) + require.Contains(t, string(summary), "thinking about zebras") + for i, item := range items[:3] { + _, hasID := item["id"] + require.False(t, hasID, "stored item id must be stripped before dispatch (item %d): %#v", i, item) + } + text, _ := json.Marshal(items[2]["content"]) + require.Contains(t, string(text), "the word is zebra") + require.Equal(t, "user", items[3]["role"], "the client's own turn is replayed last") +} + func TestResponsesWithPreviousResponseID_StreamingChainedTurn(t *testing.T) { provider := previousResponseTestProvider(t, "anthropic") provider.streamData = "event: response.completed\ndata: {\"type\":\"response.completed\",\"response\":{\"id\":\"resp_s\",\"object\":\"response\",\"status\":\"completed\",\"output\":[]}}\n\ndata: [DONE]\n\n" From 5e2e972f4d16b3c825e618f3a7fd1f11ada10da9 Mon Sep 17 00:00:00 2001 From: weselben Date: Wed, 30 Sep 2026 19:25:41 +0000 Subject: [PATCH 21/23] fix(providers/kimicode): omit whitespace-only previous_response_id on the wire A whitespace-only previous_response_id passes the TrimSpace guard but was forwarded unchanged: omitempty does not omit a non-empty whitespace string, and the upstream cannot resolve it. Clear the field on the copied request and assert its omission on the wire. --- internal/providers/kimicode/kimicode.go | 16 ++++++++++++---- internal/providers/kimicode/kimicode_test.go | 19 ++++++++++++++++++- 2 files changed, 30 insertions(+), 5 deletions(-) diff --git a/internal/providers/kimicode/kimicode.go b/internal/providers/kimicode/kimicode.go index 1bbfca26a..e9b0f610c 100644 --- a/internal/providers/kimicode/kimicode.go +++ b/internal/providers/kimicode/kimicode.go @@ -95,16 +95,24 @@ func rejectPreviousResponseID(req *core.ResponsesRequest) error { // adaptResponsesRequest pins store to false: the service retains no // responses, so store=true fails upstream with a 400 (Postel's law — adapt -// instead of failing). +// instead of failing). A whitespace-only previous_response_id is treated as +// empty by rejectPreviousResponseID and cleared here, because omitempty does +// not omit a non-empty whitespace string and the upstream cannot resolve it. func adaptResponsesRequest(req *core.ResponsesRequest) *core.ResponsesRequest { if req == nil { return nil } - if req.Store == nil || !*req.Store { + whitespaceID := req.PreviousResponseID != "" && strings.TrimSpace(req.PreviousResponseID) == "" + if (req.Store == nil || !*req.Store) && !whitespaceID { return req } cp := *req - disabled := false - cp.Store = &disabled + if req.Store != nil && *req.Store { + disabled := false + cp.Store = &disabled + } + if whitespaceID { + cp.PreviousResponseID = "" + } return &cp } diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index 3db7c011a..8e488eb6a 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -88,6 +88,18 @@ func TestAdaptResponsesRequest(t *testing.T) { require.NotNil(t, got.Store) assert.False(t, *got.Store, "explicit store=false should stay false") }) + + t.Run("whitespace-only previous_response_id is cleared", func(t *testing.T) { + req := &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "hi", + PreviousResponseID: " ", + } + got := adaptResponsesRequest(req) + require.NotSame(t, req, got, "adapted request should be a copy") + assert.Empty(t, got.PreviousResponseID) + assert.Equal(t, " ", req.PreviousResponseID, "original request was mutated") + }) } // responsesGoldenBody mirrors a real non-streaming /responses reply from the @@ -208,7 +220,9 @@ func TestResponses_RejectsPreviousResponseID(t *testing.T) { t.Run("whitespace-only ID passes through", func(t *testing.T) { // The gateway and the chat-translation validator treat a - // whitespace-only ID as empty; the provider must behave the same. + // whitespace-only ID as empty; the provider must behave the same, + // and the unresolvable value must not reach the wire (omitempty + // does not omit a non-empty whitespace string). resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ Model: "kimi-for-coding", Input: "Say OK", @@ -217,6 +231,9 @@ func TestResponses_RejectsPreviousResponseID(t *testing.T) { require.NoError(t, err) require.NotNil(t, resp) assert.Equal(t, 1, capture.Count(), "whitespace-only ID is treated as empty") + wire := capture.Last(t).JSON(t) + _, present := wire["previous_response_id"] + assert.False(t, present, "whitespace-only ID must be omitted from the wire") }) t.Run("conversation reference is rejected", func(t *testing.T) { From 02fe4ff0520789273c37e4e6c36c4d4b69e62511 Mon Sep 17 00:00:00 2001 From: weselben Date: Wed, 30 Sep 2026 20:03:57 +0000 Subject: [PATCH 22/23] test(providers/kimicode): make request-count assertion order-independent The conversation-reference subtest asserted a fixed capture count, which only holds when the whitespace-only subtest ran first. Record the count before the rejected request instead (CodeRabbit review). --- internal/providers/kimicode/kimicode_test.go | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index 8e488eb6a..23eea1d8a 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -237,6 +237,7 @@ func TestResponses_RejectsPreviousResponseID(t *testing.T) { }) t.Run("conversation reference is rejected", func(t *testing.T) { + before := capture.Count() resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ Model: "kimi-for-coding", Input: "Say OK", @@ -249,7 +250,7 @@ func TestResponses_RejectsPreviousResponseID(t *testing.T) { require.ErrorAs(t, err, &gatewayErr) assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) assert.Contains(t, gatewayErr.Error(), "conversation") - assert.Equal(t, 1, capture.Count(), "conversation request must not reach the upstream") + assert.Equal(t, before, capture.Count(), "conversation request must not reach the upstream") }) } From 4c4e7a3a843dee29927e735b39684be125f96dea Mon Sep 17 00:00:00 2001 From: weselben Date: Wed, 30 Sep 2026 20:05:28 +0000 Subject: [PATCH 23/23] test(providers/openai): cover ChatCompatible.Compatible accessor --- .../providers/openai/chat_compatible_test.go | 21 +++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 internal/providers/openai/chat_compatible_test.go diff --git a/internal/providers/openai/chat_compatible_test.go b/internal/providers/openai/chat_compatible_test.go new file mode 100644 index 000000000..975094664 --- /dev/null +++ b/internal/providers/openai/chat_compatible_test.go @@ -0,0 +1,21 @@ +package openai + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/providers" +) + +// Compatible must expose the adapter the chat-centric surface was built +// with, so providers serving native Responses use the same instance. +func TestChatCompatible_Compatible(t *testing.T) { + chat := NewChatCompatible("key", providers.ProviderOptions{}, CompatibleProviderConfig{ + ProviderName: "test", + BaseURL: "https://example.com/v1", + }) + require.NotNil(t, chat) + assert.Same(t, chat.compatible, chat.Compatible()) +}