diff --git a/internal/core/interfaces.go b/internal/core/interfaces.go index 8e03490c6..3dd20210f 100644 --- a/internal/core/interfaces.go +++ b/internal/core/interfaces.go @@ -207,6 +207,13 @@ type MessagesTokenCounter interface { CountMessagesTokens(ctx context.Context, model string, body []byte) (int, error) } +// UnlistedModelAcceptor is implemented by providers that serve model IDs +// their listing omits, such as TypeSafe's versioned Jev IDs (jev-1.13.0), so a +// virtual model can target a provider-qualified name the catalog lacks. +type UnlistedModelAcceptor interface { + AcceptsUnlistedModels() bool +} + // ErrMessagesTokenCountUnsupported reports that the provider owning a model // has no token counting endpoint. var ErrMessagesTokenCountUnsupported = errors.New("provider has no token counting endpoint") diff --git a/internal/gateway/failover.go b/internal/gateway/failover.go index f20951283..4a1ede930 100644 --- a/internal/gateway/failover.go +++ b/internal/gateway/failover.go @@ -30,6 +30,9 @@ func (o *InferenceOrchestrator) ProviderTypeForSelector(selector core.ModelSelec if providerType := strings.TrimSpace(o.provider.GetProviderType(selector.QualifiedModel())); providerType != "" { return providerType } + if _, providerType := configuredSelectorProvider(o.provider, selector); providerType != "" { + return providerType + } if provider := strings.TrimSpace(selector.Provider); provider != "" { return provider } diff --git a/internal/gateway/inference_orchestrator_test.go b/internal/gateway/inference_orchestrator_test.go index 3e2c2e27c..63d031d97 100644 --- a/internal/gateway/inference_orchestrator_test.go +++ b/internal/gateway/inference_orchestrator_test.go @@ -8,6 +8,7 @@ import ( "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/usage" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -169,6 +170,30 @@ func TestInferenceOrchestratorProviderTypeForSelectorCanonicalizesProviderNameSe require.Equal(t, "openai", got) } +// namedProviderStub knows configured provider names and their types, but no +// catalog entry for the models under test. +type namedProviderStub struct { + providerTypeResolverStub + typesByName map[string]string +} + +func (p *namedProviderStub) GetProviderTypeForName(name string) string { return p.typesByName[name] } + +// A failover target the catalog does not list, such as a pinned jev version +// on a provider named kev, is routed to the provider its selector names, with +// that provider's type, never to the primary's provider. +func TestUnlistedSelectorResolvesToTheProviderItNames(t *testing.T) { + provider := &namedProviderStub{typesByName: map[string]string{"kev": "jev", "jev-down": "jev"}} + orchestrator := NewInferenceOrchestrator(InferenceConfig{Provider: provider}) + + pinned := core.ModelSelector{Provider: "kev", Model: "kev-4b-2026-09"} + assert.Equal(t, "jev", orchestrator.ProviderTypeForSelector(pinned, "openai")) + assert.Equal(t, "kev", ResolvedProviderName(provider, pinned, "jev-down")) + + unknown := core.ModelSelector{Provider: "nope", Model: "x"} + assert.Equal(t, "jev-down", ResolvedProviderName(provider, unknown, "jev-down")) +} + func TestQualifyModelWithProviderPrefixesSlashModelIDs(t *testing.T) { got := QualifyModelWithProvider("openai/gpt-4o-mini", "openrouter") require.Equal(t, "openrouter/openai/gpt-4o-mini", got) diff --git a/internal/gateway/request_model_resolution.go b/internal/gateway/request_model_resolution.go index da77d7cfa..fdf1ca2cb 100644 --- a/internal/gateway/request_model_resolution.go +++ b/internal/gateway/request_model_resolution.go @@ -30,9 +30,30 @@ func ResolvedProviderName(provider core.RoutableProvider, selector core.ModelSel return providerName } } + // A model the catalog does not list (a pinned jev version) still belongs to + // the provider its selector names, not to the fallback's. + if providerName, providerType := configuredSelectorProvider(provider, selector); providerType != "" { + return providerName + } return fallback } +// configuredSelectorProvider returns the provider a selector names explicitly +// and that provider's type, or empty strings when the selector names no +// configured provider. +func configuredSelectorProvider(provider core.RoutableProvider, selector core.ModelSelector) (string, string) { + providerName := strings.TrimSpace(selector.Provider) + named, ok := provider.(core.ProviderNameTypeResolver) + if providerName == "" || !ok { + return "", "" + } + providerType := strings.TrimSpace(named.GetProviderTypeForName(providerName)) + if providerType == "" { + return "", "" + } + return providerName, providerType +} + // ResolvedWorkflowProviderName returns the configured provider name recorded in a resolution. func ResolvedWorkflowProviderName(resolution *core.RequestModelResolution) string { if resolution == nil { diff --git a/internal/pricingoverrides/resolver.go b/internal/pricingoverrides/resolver.go index 78b60836c..945f2b0d8 100644 --- a/internal/pricingoverrides/resolver.go +++ b/internal/pricingoverrides/resolver.go @@ -29,6 +29,29 @@ func (s *Service) ResolvePricing(model, providerName string) *core.ModelPricing return cloneBasePricing(basePricing) } +// HasModelPricing reports whether pricing is declared for exactly this model: +// catalog pricing, or an override scoped to the model rather than to its +// provider or to every model. +func (s *Service) HasModelPricing(model, providerName string) bool { + if s == nil { + return false + } + providerName = strings.TrimSpace(providerName) + rawModel := strings.TrimSpace(model) + model = modelIDFromSelector(rawModel, providerName) + if model == "" { + return false + } + if s.snapshot().hasModelScopedOverride(providerName, model) { + return true + } + if s.base == nil { + return false + } + return s.base.ResolvePricing(model, providerName) != nil || + (rawModel != model && s.base.ResolvePricing(rawModel, providerName) != nil) +} + func cloneBasePricing(base *core.ModelPricing) *core.ModelPricing { if base == nil { return nil diff --git a/internal/pricingoverrides/service_test.go b/internal/pricingoverrides/service_test.go index e977e1455..f39659872 100644 --- a/internal/pricingoverrides/service_test.go +++ b/internal/pricingoverrides/service_test.go @@ -7,6 +7,7 @@ import ( "time" "github.com/enterpilot/gomodel/internal/core" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -353,3 +354,35 @@ func TestNormalizedRefreshIntervalClampsBelowRefreshTimeout(t *testing.T) { }) } } + +func TestServiceHasModelPricing(t *testing.T) { + baseRate := 1.0 + service, err := NewService( + newTestStore( + Override{Selector: "/", Pricing: Pricing{InputPerMtok: new(float64(10))}}, + Override{Selector: "jev/", Pricing: Pricing{InputPerMtok: new(float64(20))}}, + Override{Selector: "jev/jev-1.13.0", Pricing: Pricing{InputPerMtok: new(float64(42))}}, + Override{Selector: "kev-4b", Pricing: Pricing{InputPerMtok: new(float64(0))}}, + ), + testCatalog{providerNames: []string{"jev", "openai"}}, + selectivePricingResolver{"openai/gpt-4o": {InputPerMtok: &baseRate}}, + ) + require.NoError(t, err) + require.NoError(t, service.Refresh(context.Background())) + + tests := []struct { + model, provider string + want bool + }{ + {model: "jev-1.13.0", provider: "jev", want: true}, + {model: "jev/jev-1.13.0", provider: "jev", want: true}, + {model: "kev-4b", provider: "kev", want: true}, + {model: "gpt-4o", provider: "openai", want: true}, + // Only the provider-wide and global overrides match these. + {model: "jev-latest", provider: "jev", want: false}, + {model: "gpt-9", provider: "openai", want: false}, + } + for _, tt := range tests { + assert.Equal(t, tt.want, service.HasModelPricing(tt.model, tt.provider), "%s/%s", tt.provider, tt.model) + } +} diff --git a/internal/pricingoverrides/snapshot.go b/internal/pricingoverrides/snapshot.go index 5a4338680..c8b743b30 100644 --- a/internal/pricingoverrides/snapshot.go +++ b/internal/pricingoverrides/snapshot.go @@ -108,6 +108,18 @@ func (snap snapshot) matchingOverride(providerName, model string) (compiledOverr return compiledOverride{}, false } +// hasModelScopedOverride reports whether an override names this model, +// either for one provider or model-wide. +func (snap snapshot) hasModelScopedOverride(providerName, model string) bool { + if key := modelselectors.ExactMatchKey(providerName, model); key != "" { + if _, ok := snap.exact[key]; ok { + return true + } + } + _, ok := snap.modelWide[model] + return ok +} + func snapshotOverrides(snap snapshot) []Override { result := make([]Override, 0, len(snap.order)) for _, selector := range snap.order { diff --git a/internal/providers/jev/jev.go b/internal/providers/jev/jev.go index 93dd4acef..9f999d72c 100644 --- a/internal/providers/jev/jev.go +++ b/internal/providers/jev/jev.go @@ -42,8 +42,9 @@ type Provider struct { } var ( - _ core.Provider = (*Provider)(nil) - _ core.PassthroughProvider = (*Provider)(nil) + _ core.Provider = (*Provider)(nil) + _ core.PassthroughProvider = (*Provider)(nil) + _ core.UnlistedModelAcceptor = (*Provider)(nil) ) // New creates a Jev provider. The client is rooted at the API origin, which @@ -64,6 +65,10 @@ func New(cfg providers.ProviderConfig, opts providers.ProviderOptions) core.Prov return p } +// AcceptsUnlistedModels reports that the upstream accepts versioned IDs +// (jev-1.13.0) it does not list: TypeSafe lists only its aliases. +func (p *Provider) AcceptsUnlistedModels() bool { return true } + // SetBaseURL allows configuring a custom base URL for the provider. func (p *Provider) SetBaseURL(url string) { p.client.SetBaseURL(baseURL(url)) diff --git a/internal/providers/registry_lookup.go b/internal/providers/registry_lookup.go index 5a7b68264..64db52c2a 100644 --- a/internal/providers/registry_lookup.go +++ b/internal/providers/registry_lookup.go @@ -171,6 +171,28 @@ func (r *ModelRegistry) ModelAvailable(model string) bool { return !r.providerRuntime[info.ProviderName].inventoryStale } +// AcceptsUnlistedModel reports whether a provider-qualified model the catalog +// does not list can still be served, because its provider accepts IDs it does +// not list (see core.UnlistedModelAcceptor) and its inventory is fresh. A bare +// name never qualifies: it does not say which provider to use. +func (r *ModelRegistry) AcceptsUnlistedModel(model string) bool { + providerName, _ := splitModelSelector(strings.TrimSpace(model)) + if providerName == "" { + return false + } + r.mu.RLock() + defer r.mu.RUnlock() + + for _, provider := range r.providers { + if r.providerNames[provider] != providerName { + continue + } + acceptor, ok := provider.(core.UnlistedModelAcceptor) + return ok && acceptor.AcceptsUnlistedModels() && !r.providerRuntime[providerName].inventoryStale + } + return false +} + // GetProviderType returns the provider type string for the given model. // Returns empty string if the model is not found. func (r *ModelRegistry) GetProviderType(model string) string { diff --git a/internal/providers/registry_test.go b/internal/providers/registry_test.go index e0a6483f6..cdea37ca7 100644 --- a/internal/providers/registry_test.go +++ b/internal/providers/registry_test.go @@ -2329,3 +2329,30 @@ func TestSetModelList_ClearsETag(t *testing.T) { got := registry.currentModelListETag("https://example.test/models.min.json") require.Empty(t, got) } + +// unlistedAcceptingProvider serves model IDs it does not list, as a jev +// provider serves pinned versions. +type unlistedAcceptingProvider struct { + registryMockProvider +} + +func (p *unlistedAcceptingProvider) AcceptsUnlistedModels() bool { return true } + +func TestModelRegistryAcceptsUnlistedModel(t *testing.T) { + registry := NewModelRegistry() + registry.RegisterProviderWithNameAndType(&unlistedAcceptingProvider{}, "jev", "jev") + registry.RegisterProviderWithNameAndType(®istryMockProvider{name: "openai"}, "openai", "openai") + + tests := []struct { + model string + want bool + }{ + {model: "jev/jev-1.13.0", want: true}, + {model: "jev-1.13.0", want: false}, + {model: "openai/gpt-9", want: false}, + {model: "unknown/jev-1.13.0", want: false}, + } + for _, tt := range tests { + assert.Equal(t, tt.want, registry.AcceptsUnlistedModel(tt.model), tt.model) + } +} diff --git a/internal/responsecache/responsecache.go b/internal/responsecache/responsecache.go index c43c9832c..a12124e10 100644 --- a/internal/responsecache/responsecache.go +++ b/internal/responsecache/responsecache.go @@ -304,3 +304,11 @@ func NewResponseCacheMiddlewareWithStore(store cache.Store, ttl time.Duration) * simple: newSimpleCacheMiddleware(store, ttl, nil), } } + +// NewResponseCacheMiddlewareWithStoreAndUsage creates middleware with a custom +// store that records cache hits in usage (for testing). +func NewResponseCacheMiddlewareWithStoreAndUsage(store cache.Store, ttl time.Duration, usageLogger usage.LoggerInterface, pricingResolver usage.PricingResolver) *ResponseCacheMiddleware { + return &ResponseCacheMiddleware{ + simple: newSimpleCacheMiddleware(store, ttl, newUsageHitRecorder(usageLogger, pricingResolver)), + } +} diff --git a/internal/responsecache/usage_hit.go b/internal/responsecache/usage_hit.go index 903fc3fa4..b232cb7ec 100644 --- a/internal/responsecache/usage_hit.go +++ b/internal/responsecache/usage_hit.go @@ -4,6 +4,8 @@ import ( "log/slog" "strings" + "github.com/goccy/go-json" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/usage" ) @@ -45,10 +47,8 @@ func newUsageHitRecorder(logger usage.LoggerInterface, pricingResolver usage.Pri requestID = ex.RequestHeader(core.RequestIDHeader) } - var pricing *core.ModelPricing - if pricingResolver != nil { - pricing = pricingResolver.ResolvePricing(model, cacheHitPricingProvider(provider, providerName)) - } + pricing := usage.ResolveServedModelPricing(pricingResolver, model, cacheHitPricingProvider(provider, providerName), + func() string { return cachedAnsweredModel(body) }) entry := usage.ExtractFromCachedResponseBody(body, requestID, model, provider, endpoint, cacheType, pricing) if entry == nil { @@ -62,6 +62,18 @@ func newUsageHitRecorder(logger usage.LoggerInterface, pricingResolver usage.Pri } } +// cachedAnsweredModel returns the model a cached JSON answer names, such as +// jev-1.13.0 for a request routed to jev-latest, or "" for other bodies. +func cachedAnsweredModel(body []byte) string { + var answer struct { + Model string `json:"model"` + } + if err := json.Unmarshal(body, &answer); err != nil { + return "" + } + return answer.Model +} + func cacheHitPricingProvider(provider, providerName string) string { if name := strings.TrimSpace(providerName); name != "" { return name diff --git a/internal/server/systemone_dispatch_test.go b/internal/server/systemone_dispatch_test.go index 106dc5466..1198bf12d 100644 --- a/internal/server/systemone_dispatch_test.go +++ b/internal/server/systemone_dispatch_test.go @@ -121,6 +121,48 @@ func TestSystemOne_ServesRepeatsFromTheExactCache(t *testing.T) { assert.Len(t, provider.calls, 1, "the repeat must not reach the provider") } +// answeredModelPricing prices only the models it names, as an operator who +// declares a price for the versioned model that answers an alias. +type answeredModelPricing map[string]*core.ModelPricing + +func (r answeredModelPricing) ResolvePricing(model, _ string) *core.ModelPricing { return r[model] } + +// A cache hit is recorded in usage as an exact hit, with the tokens of the +// replayed answer, priced like the live answer: by the model that answered +// when only it carries a price. +func TestSystemOne_RecordsCacheHitsInUsage(t *testing.T) { + provider := newScriptedSystemOneProvider(map[string]string{"kev/kev-latest": "jev"}) + store := cache.NewMapStore() + defer store.Close() + rate := 1_000_000.0 + pricing := answeredModelPricing{"kev-latest-answered": {InputPerMtok: &rate}} + usageLogger := &collectingUsageLogger{config: usage.Config{Enabled: true}} + mw := responsecache.NewResponseCacheMiddlewareWithStoreAndUsage(store, time.Hour, usageLogger, pricing) + handler := newHandler(provider, nil, usageLogger, pricing, nil, nil, nil, nil) + handler.responseCache = mw + + c, first := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, http.StatusOK, first.Code, first.Body.String()) + require.NoError(t, mw.Close()) + + c, second := echotest.Post(t, "/v1/systemone", systemOneRequest("kev-latest")) + require.NoError(t, handler.SystemOne(c)) + require.Equal(t, "HIT (exact)", second.Header().Get("X-Cache")) + + require.Len(t, usageLogger.entries, 2) + hit := usageLogger.entries[1] + assert.Equal(t, usage.CacheTypeExact, hit.CacheType) + assert.Equal(t, "/v1/systemone", hit.Endpoint) + assert.Equal(t, "jev", hit.Provider) + assert.Equal(t, 10, hit.InputTokens) + assert.Equal(t, 1, hit.OutputTokens) + for _, entry := range usageLogger.entries { + require.NotNil(t, entry.InputCost, "cache type %q", entry.CacheType) + assert.InDelta(t, 10.0, *entry.InputCost, 1e-9) + } +} + // A different state is a different decision: it misses the cache. func TestSystemOne_CacheKeyCoversTheState(t *testing.T) { provider := newScriptedSystemOneProvider(map[string]string{"kev/kev-latest": "jev"}) diff --git a/internal/usage/extractor.go b/internal/usage/extractor.go index 250b74461..732a45b5d 100644 --- a/internal/usage/extractor.go +++ b/internal/usage/extractor.go @@ -223,6 +223,11 @@ func ExtractFromSSEUsage( requestID, model, provider, endpoint string, pricing ...*core.ModelPricing, ) *UsageEntry { + // Anthropic-style usage (and System One answers) report input and output + // tokens without a total. + if totalTokens == 0 { + totalTokens = inputTokens + outputTokens + } entry := &UsageEntry{ ID: uuid.New().String(), RequestID: requestID, diff --git a/internal/usage/extractor_test.go b/internal/usage/extractor_test.go index 01cc313a8..c781d0185 100644 --- a/internal/usage/extractor_test.go +++ b/internal/usage/extractor_test.go @@ -436,6 +436,19 @@ func TestExtractFromSSEUsage(t *testing.T) { assert.Equal(t, 25, entry.RawData["cached_tokens"]) } +func TestExtractFromSSEUsageDerivesMissingTotal(t *testing.T) { + // Anthropic-style usage and System One answers carry no total_tokens. + entry := ExtractFromSSEUsage( + "", + 27, 9, 0, + nil, + "req-systemone", "jev-1.13.0", "jev", "/v1/systemone", + ) + + require.NotNil(t, entry) + assert.Equal(t, 36, entry.TotalTokens) +} + func TestExtractFromSSEUsageEmptyRawData(t *testing.T) { entry := ExtractFromSSEUsage( "chatcmpl-789", @@ -505,6 +518,7 @@ func TestExtractFromCachedResponseBody(t *testing.T) { require.Equal(t, "jev", entry.Provider) require.Equal(t, 275, entry.InputTokens) require.Equal(t, 20, entry.OutputTokens) + require.Equal(t, 295, entry.TotalTokens) }) t.Run("falls back to synthetic entry when body cannot be parsed", func(t *testing.T) { diff --git a/internal/usage/pricing.go b/internal/usage/pricing.go index db56717d4..d6c1502d2 100644 --- a/internal/usage/pricing.go +++ b/internal/usage/pricing.go @@ -1,6 +1,10 @@ package usage -import "github.com/enterpilot/gomodel/internal/core" +import ( + "strings" + + "github.com/enterpilot/gomodel/internal/core" +) // PricingResolver resolves pricing metadata for a given model and provider type. // Implementations should check the registry first and fall back to a reverse-index @@ -8,3 +12,50 @@ import "github.com/enterpilot/gomodel/internal/core" type PricingResolver interface { ResolvePricing(model, providerType string) *core.ModelPricing } + +// ModelPricingChecker is implemented by pricing resolvers that can tell +// pricing declared for exactly one model (catalog pricing or a model-scoped +// override) from a broad provider-wide or global rule. +type ModelPricingChecker interface { + HasModelPricing(model, providerType string) bool +} + +// ResolveServedModelPricing prices a response routed to one model and +// answered by another, such as the alias jev-latest answered by jev-1.13.0. +// Pricing declared for exactly the routed model wins, then pricing declared +// for exactly the answering model, then any broader rule matching the routed +// model, then the answering model. answered is called at most once, and only +// when the routed model has no exact pricing. +func ResolveServedModelPricing(resolver PricingResolver, routed, providerType string, answered func() string) *core.ModelPricing { + if resolver == nil { + return nil + } + routed = strings.TrimSpace(routed) + checker, canCheck := resolver.(ModelPricingChecker) + if canCheck && checker.HasModelPricing(routed, providerType) { + return resolver.ResolvePricing(routed, providerType) + } + + answeredModel, resolved := "", false + answeredOnce := func() string { + if !resolved && answered != nil { + if model := strings.TrimSpace(answered()); model != routed { + answeredModel = model + } + } + resolved = true + return answeredModel + } + if canCheck { + if model := answeredOnce(); model != "" && checker.HasModelPricing(model, providerType) { + return resolver.ResolvePricing(model, providerType) + } + } + if pricing := resolver.ResolvePricing(routed, providerType); pricing != nil { + return pricing + } + if model := answeredOnce(); model != "" { + return resolver.ResolvePricing(model, providerType) + } + return nil +} diff --git a/internal/usage/pricing_test.go b/internal/usage/pricing_test.go new file mode 100644 index 000000000..d9bb15f12 --- /dev/null +++ b/internal/usage/pricing_test.go @@ -0,0 +1,83 @@ +package usage + +import ( + "testing" + + "github.com/stretchr/testify/assert" + + "github.com/enterpilot/gomodel/internal/core" +) + +// checkedPricingResolver prices models from exact entries, falling back to a +// broad (provider-wide) rule, and reports which prices are exact. +type checkedPricingResolver struct { + exact map[string]*core.ModelPricing + broad *core.ModelPricing +} + +func (r checkedPricingResolver) ResolvePricing(model, _ string) *core.ModelPricing { + if pricing, ok := r.exact[model]; ok { + return pricing + } + return r.broad +} + +func (r checkedPricingResolver) HasModelPricing(model, _ string) bool { + _, ok := r.exact[model] + return ok +} + +func TestResolveServedModelPricing(t *testing.T) { + rate := func(v float64) *core.ModelPricing { return &core.ModelPricing{InputPerMtok: &v} } + routedExact, answeredExact, broad := rate(1), rate(2), rate(3) + + tests := []struct { + name string + resolver PricingResolver + want *core.ModelPricing + }{ + { + name: "exact routed price wins", + resolver: checkedPricingResolver{exact: map[string]*core.ModelPricing{"jev-latest": routedExact, "jev-1.13.0": answeredExact}, broad: broad}, + want: routedExact, + }, + { + name: "exact answering price beats a broad routed rule", + resolver: checkedPricingResolver{exact: map[string]*core.ModelPricing{"jev-1.13.0": answeredExact}, broad: broad}, + want: answeredExact, + }, + { + name: "broad rule when neither is exact", + resolver: checkedPricingResolver{broad: broad}, + want: broad, + }, + { + name: "resolver without exact checks tries routed then answering", + resolver: mapPricingResolver{"jev-1.13.0/jev": answeredExact}, + want: answeredExact, + }, + { + name: "no resolver", + want: nil, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := ResolveServedModelPricing(tt.resolver, "jev-latest", "jev", func() string { return "jev-1.13.0" }) + assert.Same(t, tt.want, got) + }) + } +} + +// The answering model is only looked up when the routed model has no exact +// price, so a cache hit does not decode its body for the common case. +func TestResolveServedModelPricingSkipsAnsweredLookupForExactRoutedPrice(t *testing.T) { + v := 1.0 + resolver := checkedPricingResolver{exact: map[string]*core.ModelPricing{"gpt-4o": {InputPerMtok: &v}}} + calls := 0 + ResolveServedModelPricing(resolver, "gpt-4o", "openai", func() string { + calls++ + return "gpt-4o-2024-08-06" + }) + assert.Zero(t, calls) +} diff --git a/internal/usage/stream_observer.go b/internal/usage/stream_observer.go index 927750078..ae8e99c65 100644 --- a/internal/usage/stream_observer.go +++ b/internal/usage/stream_observer.go @@ -169,10 +169,8 @@ func (o *StreamUsageObserver) mergeWithCachedEntry(entry *UsageEntry) *UsageEntr entry.TotalTokens = entry.InputTokens + entry.OutputTokens } var pricingArgs []*core.ModelPricing - if o.pricingResolver != nil { - if p := o.pricingResolver.ResolvePricing(o.pricingModel(entry.Model), o.pricingProvider()); p != nil { - pricingArgs = append(pricingArgs, p) - } + if p := o.resolvePricing(entry.Model); p != nil { + pricingArgs = append(pricingArgs, p) } applyUsageCosts(entry, o.provider, o.endpoint, pricingArgs...) return entry @@ -267,10 +265,8 @@ func (o *StreamUsageObserver) extractUsageFromEvent(chunk map[string]any) *Usage } var pricingArgs []*core.ModelPricing - if o.pricingResolver != nil { - if p := o.pricingResolver.ResolvePricing(o.pricingModel(model), o.pricingProvider()); p != nil { - pricingArgs = append(pricingArgs, p) - } + if p := o.resolvePricing(model); p != nil { + pricingArgs = append(pricingArgs, p) } entry := ExtractFromSSEUsage( @@ -304,6 +300,16 @@ func (o *StreamUsageObserver) pricingModel(responseModel string) string { return strings.TrimSpace(responseModel) } +// resolvePricing prices the routed model or, when only it carries pricing, +// the model that answered; see ResolveServedModelPricing. +func (o *StreamUsageObserver) resolvePricing(responseModel string) *core.ModelPricing { + if o == nil { + return nil + } + return ResolveServedModelPricing(o.pricingResolver, o.pricingModel(responseModel), o.pricingProvider(), + func() string { return responseModel }) +} + func (o *StreamUsageObserver) pricingProvider() string { if o == nil { return "" diff --git a/internal/usage/stream_observer_test.go b/internal/usage/stream_observer_test.go index 576ebdbd2..79c27c896 100644 --- a/internal/usage/stream_observer_test.go +++ b/internal/usage/stream_observer_test.go @@ -617,3 +617,38 @@ func TestStreamUsageObserverAnthropicNativeEvents(t *testing.T) { assert.Equal(t, 100, entry.RawData["cache_creation_input_tokens"]) assert.Equal(t, 200, entry.RawData["cache_read_input_tokens"]) } + +// A routed alias (jev-latest) is answered by a versioned model (jev-1.13.0); +// the routed model's price wins, and the answered model's price applies when +// only it is declared. +func TestStreamUsageObserverPricesAnsweredModelWhenRoutedHasNone(t *testing.T) { + routedRate, answeredRate := 10.0, 42.0 + routed := &core.ModelPricing{InputPerMtok: &routedRate} + answered := &core.ModelPricing{InputPerMtok: &answeredRate} + + tests := []struct { + name string + resolver mapPricingResolver + wantInput float64 + }{ + {name: "routed model priced", resolver: mapPricingResolver{"jev-latest/jev": routed, "jev-1.13.0/jev": answered}, wantInput: 10}, + {name: "only answered model priced", resolver: mapPricingResolver{"jev-1.13.0/jev": answered}, wantInput: 42}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + logger := &trackingLogger{enabled: true} + observer := NewStreamUsageObserver(logger, "jev-latest", "jev", "req-s1", "/v1/systemone", tt.resolver) + observer.OnJSONEvent(map[string]any{ + "model": "jev-1.13.0", + "usage": map[string]any{"input_tokens": float64(1_000_000), "output_tokens": float64(0)}, + }) + observer.OnStreamClose() + + entries := logger.getEntries() + require.Len(t, entries, 1) + require.NotNil(t, entries[0].InputCost) + assert.InDelta(t, tt.wantInput, *entries[0].InputCost, 1e-9) + assert.Equal(t, "jev-1.13.0", entries[0].Model) + }) + } +} diff --git a/internal/virtualmodels/chain.go b/internal/virtualmodels/chain.go index 3f2c461a9..e3c3b8e2e 100644 --- a/internal/virtualmodels/chain.go +++ b/internal/virtualmodels/chain.go @@ -66,7 +66,7 @@ func (s *snapshot) viableTargets(entry *redirectEntry, catalog Catalog) []resolv func (s *snapshot) viable(owner *redirectEntry, target resolvedTarget, catalog Catalog) bool { inner, ok := s.chained(owner.vm.Source, target) if !ok { - return catalog.ModelAvailable(target.qualified) + return modelServable(catalog, target.qualified) } if !inner.vm.Enabled { return false @@ -100,7 +100,7 @@ func (s *snapshot) leafTargets(entry *redirectEntry, catalog Catalog) []resolved func (s *snapshot) leaves(owner *redirectEntry, target resolvedTarget, catalog Catalog) []resolvedTarget { inner, ok := s.chained(owner.vm.Source, target) if !ok { - if catalog.ModelAvailable(target.qualified) { + if modelServable(catalog, target.qualified) { return []resolvedTarget{target} } return nil diff --git a/internal/virtualmodels/service.go b/internal/virtualmodels/service.go index c0a13af99..881f16e29 100644 --- a/internal/virtualmodels/service.go +++ b/internal/virtualmodels/service.go @@ -679,7 +679,7 @@ func (s *Service) firstUnsupportedTarget(current *snapshot, vm VirtualModel) (st if _, chained := current.chained(vm.Source, candidate); chained { continue } - if !s.catalog.Supports(qualified) { + if !s.catalog.Supports(qualified) && !modelServable(s.catalog, qualified) { return qualified, true } } diff --git a/internal/virtualmodels/types.go b/internal/virtualmodels/types.go index 82f7cfaa5..eb07bf91a 100644 --- a/internal/virtualmodels/types.go +++ b/internal/virtualmodels/types.go @@ -254,3 +254,20 @@ type Catalog interface { LookupModel(model string) (*core.Model, bool) ProviderNames() []string } + +// unlistedModelCatalog is implemented by catalogs whose providers serve +// provider-qualified model IDs they do not list, such as a jev provider's +// pinned versions (jev/jev-1.13.0). +type unlistedModelCatalog interface { + AcceptsUnlistedModel(model string) bool +} + +// modelServable reports whether a concrete target can serve a request now: +// listed and available, or unlisted on a provider that accepts such IDs. +func modelServable(catalog Catalog, model string) bool { + if catalog.ModelAvailable(model) { + return true + } + unlisted, ok := catalog.(unlistedModelCatalog) + return ok && unlisted.AcceptsUnlistedModel(model) +} diff --git a/internal/virtualmodels/unlisted_target_test.go b/internal/virtualmodels/unlisted_target_test.go new file mode 100644 index 000000000..8c1211855 --- /dev/null +++ b/internal/virtualmodels/unlisted_target_test.go @@ -0,0 +1,52 @@ +package virtualmodels + +import ( + "context" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/core" +) + +// unlistedCatalog is a fakeCatalog whose named providers serve IDs they do +// not list, as a jev provider serves pinned versions. +type unlistedCatalog struct { + fakeCatalog + acceptsUnlisted map[string]bool +} + +func (c unlistedCatalog) AcceptsUnlistedModel(model string) bool { + provider, _, ok := strings.Cut(model, "/") + return ok && c.acceptsUnlisted[provider] +} + +// A redirect may pin a version its provider accepts without listing it; the +// same target on a provider that lists everything it serves is still refused. +func TestService_RedirectToUnlistedModelOfAcceptingProvider(t *testing.T) { + t.Parallel() + catalog := unlistedCatalog{ + providers: []string{"openai", "jev"}, + supported: map[string]core.Model{ + "openai/gpt-4o": {ID: "openai/gpt-4o"}, + "jev/jev-latest": {ID: "jev/jev-latest"}, + }, + acceptsUnlisted: map[string]bool{"jev": true}, + } + svc, err := NewService(newSQLVMStore(t), catalog, true) + require.NoError(t, err) + ctx := context.Background() + + err = svc.Upsert(ctx, VirtualModel{Source: "pinned", Targets: []Target{{Model: "jev/jev-1.13.0"}}, Enabled: true}) + require.NoError(t, err) + selector, changed, err := svc.ResolveModel(core.NewRequestedModelSelector("pinned", "")) + require.NoError(t, err) + assert.True(t, changed) + assert.Equal(t, "jev/jev-1.13.0", selector.QualifiedModel()) + + err = svc.Upsert(ctx, VirtualModel{Source: "missing", Targets: []Target{{Model: "openai/gpt-9"}}, Enabled: true}) + require.Error(t, err) + assert.Contains(t, err.Error(), "target model not found: openai/gpt-9") +} diff --git a/tests/e2e/manage-release-e2e-stack.sh b/tests/e2e/manage-release-e2e-stack.sh index 102cb9871..2fdcf4efd 100755 --- a/tests/e2e/manage-release-e2e-stack.sh +++ b/tests/e2e/manage-release-e2e-stack.sh @@ -11,6 +11,9 @@ MONGO_DATABASE="${GOMODEL_RELEASE_MONGO_DATABASE:-gomodel_release_e2e}" MOCK_MCP_BIN="${GOMODEL_RELEASE_MOCK_MCP_BINARY:-$REPO_ROOT/bin/mockmcp}" MOCK_MCP_PORT="${GOMODEL_RELEASE_MOCK_MCP_PORT:-18090}" MOCK_MCP_TOKEN="${GOMODEL_RELEASE_MOCK_MCP_TOKEN:-qa-mock-mcp-secret}" +MOCK_JEV_BIN="${GOMODEL_RELEASE_MOCK_JEV_BINARY:-$REPO_ROOT/bin/mockjev}" +MOCK_JEV_PORT="${GOMODEL_RELEASE_MOCK_JEV_PORT:-18091}" +MOCK_JEV_KEY="${GOMODEL_RELEASE_MOCK_JEV_KEY:-qa-mock-jev-key}" BUILD_BEFORE_START=0 @@ -37,6 +40,7 @@ Gateways: Helpers: mock-mcp http://localhost:18090 (mock MCP upstream: /alpha token-gated, /beta open) + mock-jev http://localhost:18091 (mock System One upstreams: /jev keyed, /kev keyless Kev, /down always 529) EOF } @@ -99,7 +103,24 @@ load_env() { export OPENROUTER_MODEL_FILTER_INCLUDE="${OPENROUTER_MODEL_FILTER_INCLUDE:-*:free}" export XAI_MODELS="${XAI_MODELS:-grok-4.3,grok-voice-latest}" export BAILIAN_MODELS="${BAILIAN_MODELS:-qwen3-omni-flash-realtime}" - export ENABLED_PASSTHROUGH_PROVIDERS="${ENABLED_PASSTHROUGH_PROVIDERS:-openai,anthropic,openrouter,zai,vllm,deepseek,bailian,xai}" + export ENABLED_PASSTHROUGH_PROVIDERS="${ENABLED_PASSTHROUGH_PROVIDERS:-openai,anthropic,openrouter,zai,vllm,deepseek,bailian,xai,jev}" + # System One (Jev / Kev) providers backed by the local mockjev upstream: + # "jev" is hosted-shaped and keyed, "jev-kev" a keyless Kev server, and + # "jev-down" answers 529 so System One failover can be exercised. Every + # JEV_* value from .env is dropped first (suffixed keys, model lists): the + # scenarios assert what the mock echoes back, and a key left on a keyless + # provider would reach the mock. The Kev URL keeps a trailing /v1, which the + # provider trims. + local name + for name in $(compgen -e); do + if [[ "$name" == JEV_* ]]; then + unset "$name" + fi + done + export JEV_API_KEY="$MOCK_JEV_KEY" + export JEV_BASE_URL="http://localhost:$MOCK_JEV_PORT/jev" + export JEV_KEV_BASE_URL="http://localhost:$MOCK_JEV_PORT/kev/v1" + export JEV_DOWN_BASE_URL="http://localhost:$MOCK_JEV_PORT/down" } ensure_binary() { @@ -109,54 +130,62 @@ ensure_binary() { if (( BUILD_BEFORE_START == 1 )) || [[ ! -x "$MOCK_MCP_BIN" ]]; then (cd "$REPO_ROOT" && go build -o "$MOCK_MCP_BIN" ./tests/e2e/mockmcp) fi + if (( BUILD_BEFORE_START == 1 )) || [[ ! -x "$MOCK_JEV_BIN" ]]; then + (cd "$REPO_ROOT" && go build -o "$MOCK_JEV_BIN" ./tests/e2e/mockjev) + fi } -start_mock_mcp() { - local dir="$STACK_DIR/mock-mcp" +# Starts one mock upstream binary on its port and waits for /healthz. +# usage: start_mock NAME PORT BINARY [ENV=VALUE...] +start_mock() { + local name="$1" port="$2" bin="$3" + shift 3 + local dir="$STACK_DIR/$name" local log_file="$dir/logs/server.log" local pid_file="$dir/server.pid" mkdir -p "$dir/logs" if is_pid_running "$pid_file"; then - printf 'mock-mcp already running pid=%s url=http://localhost:%s\n' "$(cat "$pid_file")" "$MOCK_MCP_PORT" + printf '%s already running pid=%s url=http://localhost:%s\n' "$name" "$(cat "$pid_file")" "$port" return 0 fi # A foreign process on the port would answer the health probe and mask a - # failed bind (e.g. a manually started mockmcp with a different token). - if curl -fsS "http://localhost:$MOCK_MCP_PORT/healthz" >/dev/null 2>&1; then - die "port $MOCK_MCP_PORT is already in use by an unmanaged process; stop it before starting mock-mcp" + # failed bind (e.g. a manually started mock with different settings). + if curl -fsS "http://localhost:$port/healthz" >/dev/null 2>&1; then + die "port $port is already in use by an unmanaged process; stop it before starting $name" fi rm -f "$pid_file" ( cd "$dir" - nohup env PORT="$MOCK_MCP_PORT" MOCK_MCP_TOKEN="$MOCK_MCP_TOKEN" "$MOCK_MCP_BIN" >"$log_file" 2>&1 < /dev/null & + nohup env PORT="$port" "$@" "$bin" >"$log_file" 2>&1 < /dev/null & echo $! >"$pid_file" ) local attempt for attempt in $(seq 1 15); do - if curl -fsS "http://localhost:$MOCK_MCP_PORT/healthz" >/dev/null 2>&1; then - printf 'started mock-mcp pid=%s url=http://localhost:%s\n' "$(cat "$pid_file")" "$MOCK_MCP_PORT" + if curl -fsS "http://localhost:$port/healthz" >/dev/null 2>&1; then + printf 'started %s pid=%s url=http://localhost:%s\n' "$name" "$(cat "$pid_file")" "$port" return 0 fi sleep 1 done - echo "failed to start mock-mcp on port $MOCK_MCP_PORT" >&2 + echo "failed to start $name on port $port" >&2 [[ -f "$log_file" ]] && tail -n 40 "$log_file" >&2 exit 1 } -stop_mock_mcp() { - local pid_file="$STACK_DIR/mock-mcp/server.pid" +stop_mock() { + local name="$1" + local pid_file="$STACK_DIR/$name/server.pid" local pid if [[ ! -f "$pid_file" ]]; then - printf 'mock-mcp not running\n' + printf '%s not running\n' "$name" return 0 fi @@ -165,22 +194,23 @@ stop_mock_mcp() { kill "$pid" 2>/dev/null || true fi rm -f "$pid_file" - printf 'stopped mock-mcp\n' + printf 'stopped %s\n' "$name" } -status_mock_mcp() { - local pid_file="$STACK_DIR/mock-mcp/server.pid" +status_mock() { + local name="$1" port="$2" + local pid_file="$STACK_DIR/$name/server.pid" local health="down" local pid="stopped" if is_pid_running "$pid_file"; then pid="$(cat "$pid_file")" - if curl -fsS "http://localhost:$MOCK_MCP_PORT/healthz" >/dev/null 2>&1; then + if curl -fsS "http://localhost:$port/healthz" >/dev/null 2>&1; then health="ok" fi fi - printf '%-12s pid=%-8s url=http://localhost:%s health=%s\n' "mock-mcp" "$pid" "$MOCK_MCP_PORT" "$health" + printf '%-12s pid=%-8s url=http://localhost:%s health=%s\n' "$name" "$pid" "$port" "$health" } ensure_pg_database() { @@ -360,7 +390,8 @@ start_stack() { mkdir -p "$STACK_DIR" ensure_pg_database write_guardrail_config - start_mock_mcp + start_mock mock-mcp "$MOCK_MCP_PORT" "$MOCK_MCP_BIN" MOCK_MCP_TOKEN="$MOCK_MCP_TOKEN" + start_mock mock-jev "$MOCK_JEV_PORT" "$MOCK_JEV_BIN" MOCK_JEV_KEY="$MOCK_JEV_KEY" start_gateway sqlite-main \ -u GOMODEL_MASTER_KEY \ @@ -457,11 +488,13 @@ stop_stack() { stop_gateway mongo-smoke stop_gateway pg-smoke stop_gateway sqlite-main - stop_mock_mcp + stop_mock mock-jev + stop_mock mock-mcp } status_stack() { - status_mock_mcp + status_mock mock-mcp "$MOCK_MCP_PORT" + status_mock mock-jev "$MOCK_JEV_PORT" status_gateway sqlite-main status_gateway pg-smoke status_gateway mongo-smoke diff --git a/tests/e2e/mockjev/main.go b/tests/e2e/mockjev/main.go new file mode 100644 index 000000000..892ddb690 --- /dev/null +++ b/tests/e2e/mockjev/main.go @@ -0,0 +1,348 @@ +// Command mockjev serves deterministic TypeSafe System One upstreams for the +// release E2E curl matrix, one per path prefix, so a single process backs +// several jev providers: +// +// /jev hosted-API shape: requires "Authorization: Bearer $MOCK_JEV_KEY"; +// lists jev-latest and jev-preview by "name"; accepts any versioned +// ID (jev-1.13.0); answers jev-latest as jev-1.13.0; has no Kev +// diagnostic routes (404, like the hosted API). +// /kev Kev-server shape: no authentication; lists checkpoint kev-latest +// by "id" with alias kev-4b; serves /v1/systemone/permute and +// /v1/systemone/separate. +// /down lists kev-down but answers every System One route with 529, the +// status TypeSafe uses for overload, so failover can be exercised. +// +// Each answer carries a "mock" object echoing what reached the upstream (the +// model, state, questions, extra top-level fields, whether an Authorization +// header arrived, the X-Request-Id, and a per-upstream request sequence), so +// scenarios can assert what the gateway forwarded and whether a cached +// answer was replayed. Malformed questions get a FastAPI-style 422, as the +// hosted API returns. +// +// PORT selects the listen port (default 18091). GET /healthz reports liveness. +package main + +import ( + "bytes" + "encoding/json" + "fmt" + "log" + "net/http" + "os" + "regexp" + "sort" + "strings" + "sync" +) + +type upstream struct { + name string + key string // required bearer key; empty means no authentication + models any // GET /v1/models body + kevRoute bool // serves permute and separate + down bool // answers System One routes with 529 + accepts func(model string) (answeredAs string, ok bool) + + mu sync.Mutex + seq int +} + +// maxBodyBytes bounds a request body; the largest the matrix sends is about +// 70 KiB. +const maxBodyBytes = 1 << 20 + +var versionedJev = regexp.MustCompile(`^jev-\d+\.\d+\.\d+$`) + +func newUpstreams(jevKey string) []*upstream { + return []*upstream{ + { + name: "jev", + key: jevKey, + models: map[string]any{"models": []map[string]any{ + {"name": "jev-latest", "description": "Latest Jev (mock)", "release_date": "2026-06-01"}, + {"name": "jev-preview", "description": "Preview Jev (mock)"}, + }}, + accepts: func(model string) (string, bool) { + switch { + case model == "jev-latest": + return "jev-1.13.0", true + case model == "jev-preview": + return "jev-1.14.0-preview", true + case versionedJev.MatchString(model): + return model, true + } + return "", false + }, + }, + { + name: "kev", + kevRoute: true, + models: map[string]any{"models": []map[string]any{ + {"id": "kev-latest", "aliases": []string{"kev-4b"}, "release_date": "2026-09-01T00:00:00Z"}, + }}, + accepts: func(model string) (string, bool) { + if model == "kev-latest" || model == "kev-4b" { + return "kev-4b-e2e", true + } + return "", false + }, + }, + { + name: "down", + kevRoute: true, + down: true, + models: map[string]any{"models": []map[string]any{{"id": "kev-down"}}}, + accepts: func(string) (string, bool) { return "", false }, + }, + } +} + +// question keeps instructions and criteria as raw JSON: the SDKs accept +// structured values there, and an answer needs only option names and the +// number of score levels. +type question struct { + Type string `json:"type"` + Instructions json.RawMessage `json:"instructions"` + Criteria json.RawMessage `json:"criteria"` +} + +type request struct { + Model string `json:"model"` + State json.RawMessage `json:"state"` + Questions map[string]question `json:"questions"` + NPerm *int `json:"n_perm"` +} + +func writeJSON(w http.ResponseWriter, status int, body any) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(status) + _ = json.NewEncoder(w).Encode(body) +} + +// validationError mirrors the FastAPI 422 body the hosted API returns. +func validationError(w http.ResponseWriter, loc []any, msg string) { + writeJSON(w, http.StatusUnprocessableEntity, map[string]any{ + "detail": []map[string]any{{"loc": append([]any{"body"}, loc...), "msg": msg, "type": "value_error"}}, + }) +} + +func (u *upstream) authorized(r *http.Request) bool { + return u.key == "" || r.Header.Get("Authorization") == "Bearer "+u.key +} + +func (u *upstream) authSeen(r *http.Request) string { + switch auth := r.Header.Get("Authorization"); { + case auth == "": + return "none" + case u.key != "" && auth == "Bearer "+u.key: + return "provider-key" + default: + return "other" + } +} + +func (u *upstream) serveModels(w http.ResponseWriter, r *http.Request) { + if !u.authorized(r) { + writeJSON(w, http.StatusUnauthorized, map[string]any{"detail": "Invalid API key"}) + return + } + writeJSON(w, http.StatusOK, u.models) +} + +func (u *upstream) serveSystemOne(route string) http.HandlerFunc { + return func(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodPost { + writeJSON(w, http.StatusMethodNotAllowed, map[string]any{"detail": "Method Not Allowed"}) + return + } + if route != "evaluate" && !u.kevRoute { + writeJSON(w, http.StatusNotFound, map[string]any{"detail": "Not Found"}) + return + } + if !u.authorized(r) { + writeJSON(w, http.StatusUnauthorized, map[string]any{"detail": "Invalid API key"}) + return + } + if u.down { + writeJSON(w, 529, map[string]any{"detail": "Overloaded (mock " + u.name + ")"}) + return + } + var raw bytes.Buffer + if _, err := raw.ReadFrom(http.MaxBytesReader(w, r.Body, maxBodyBytes)); err != nil { + writeJSON(w, http.StatusRequestEntityTooLarge, map[string]any{"detail": err.Error()}) + return + } + var req request + if err := json.Unmarshal(raw.Bytes(), &req); err != nil { + validationError(w, nil, "invalid JSON: "+err.Error()) + return + } + answeredAs, ok := u.accepts(req.Model) + if !ok { + writeJSON(w, http.StatusNotFound, map[string]any{"detail": fmt.Sprintf("Model %q not found", req.Model)}) + return + } + if len(req.State) == 0 || string(req.State) == "null" { + validationError(w, []any{"state"}, "Field required") + return + } + if len(req.Questions) == 0 { + validationError(w, []any{"questions"}, "At least one question is required") + return + } + names := make([]string, 0, len(req.Questions)) + for name := range req.Questions { + names = append(names, name) + } + sort.Strings(names) + + nPerm := 0 + if route == "permute" { + if len(req.Questions) != 1 || req.Questions[names[0]].Type != "choice" { + validationError(w, []any{"questions"}, "permute takes exactly one choice question") + return + } + nPerm = 6 + if req.NPerm != nil { + nPerm = *req.NPerm + } + if nPerm < 1 || nPerm > 64 { + validationError(w, []any{"n_perm"}, "n_perm must be between 1 and 64") + return + } + } + + answers := make(map[string]any, len(names)) + for _, name := range names { + answer, msg := answerFor(req.Questions[name]) + if msg != "" { + validationError(w, []any{"questions", name}, msg) + return + } + answers[name] = answer + } + + u.mu.Lock() + u.seq++ + seq := u.seq + u.mu.Unlock() + + var top map[string]json.RawMessage + _ = json.Unmarshal(raw.Bytes(), &top) + extra := map[string]json.RawMessage{} + for key, value := range top { + switch key { + case "model", "state", "questions", "n_perm": + default: + extra[key] = value + } + } + + body := map[string]any{ + "model": answeredAs, + "answers": answers, + "usage": map[string]any{ + "input_tokens": 10 + len(req.State)/4 + 5*len(names), + "output_tokens": 3 * len(names), + }, + "mock": map[string]any{ + "upstream": u.name, + "route": route, + "received_model": req.Model, + "received_state": req.State, + "questions": top["questions"], + "extra": extra, + "authorization": u.authSeen(r), + "request_id": r.Header.Get("X-Request-Id"), + "request_seq": seq, + "received_length": raw.Len(), + }, + } + if route == "permute" { + body["n_perm"] = nPerm + } + if route == "separate" { + body["separate"] = true + } + writeJSON(w, http.StatusOK, body) + } +} + +// answerFor builds a deterministic, well-formed answer for one question, or +// returns the validation message the hosted API would reject it with. +func answerFor(q question) (any, string) { + switch q.Type { + case "noul": + return map[string]any{"type": "noul", "noul": 0.93}, "" + case "choice": + var criteria map[string]json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil || len(criteria) < 2 { + return nil, "choice criteria must map at least two option names to descriptions" + } + options := make([]string, 0, len(criteria)) + for option := range criteria { + options = append(options, option) + } + sort.Strings(options) + probabilities := make(map[string]float64, len(options)) + rest := 0.4 / float64(len(options)-1) + for i, option := range options { + probabilities[option] = rest + if i == 0 { + probabilities[option] = 0.6 + } + } + return map[string]any{"type": "choice", "choice": options[0], "confidence": 0.5, "probabilities": probabilities}, "" + case "score": + var levels []json.RawMessage + if err := json.Unmarshal(q.Criteria, &levels); err != nil || len(levels) < 2 { + return nil, "score criteria must list at least two ordered levels" + } + legend := make(map[string]any, len(levels)) + probabilities := make(map[string]float64, len(levels)) + for i, level := range levels { + var label string + if json.Unmarshal(level, &label) == nil { + legend[fmt.Sprint(i)] = label + } else { + legend[fmt.Sprint(i)] = level + } + probabilities[fmt.Sprint(i)] = 0 + } + probabilities["1"] = 1 + return map[string]any{"type": "score", "score": 1.0, "confidence": 0.8, "legend": legend, "probabilities": probabilities}, "" + case "": + return nil, "Field required: type" + default: + return nil, fmt.Sprintf("Input tag %q found using 'type' does not match any of the expected tags: 'noul', 'choice', 'score'", q.Type) + } +} + +func main() { + port := os.Getenv("PORT") + if port == "" { + port = "18091" + } + jevKey := os.Getenv("MOCK_JEV_KEY") + if jevKey == "" { + jevKey = "qa-mock-jev-key" + } + + mux := http.NewServeMux() + for _, u := range newUpstreams(jevKey) { + prefix := "/" + u.name + mux.HandleFunc(prefix+"/v1/models", u.serveModels) + mux.HandleFunc(prefix+"/v1/systemone", u.serveSystemOne("evaluate")) + mux.HandleFunc(prefix+"/v1/systemone/permute", u.serveSystemOne("permute")) + mux.HandleFunc(prefix+"/v1/systemone/separate", u.serveSystemOne("separate")) + } + mux.HandleFunc("/healthz", func(w http.ResponseWriter, _ *http.Request) { + fmt.Fprintln(w, "ok") + }) + mux.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) { + writeJSON(w, http.StatusNotFound, map[string]any{"detail": "Not Found: " + strings.TrimSpace(r.URL.Path)}) + }) + + log.Printf("mockjev listening on :%s", port) + log.Fatal(http.ListenAndServe(":"+port, mux)) +} diff --git a/tests/e2e/release-e2e-scenarios.md b/tests/e2e/release-e2e-scenarios.md index d51206e89..137228fac 100644 --- a/tests/e2e/release-e2e-scenarios.md +++ b/tests/e2e/release-e2e-scenarios.md @@ -11,6 +11,9 @@ These scenarios are prepared for execution across these local gateways: - `http://localhost:18090` - mock MCP upstream (`tests/e2e/mockmcp`, started by the stack manager; `/alpha` requires the `X-Mock-Token` header, `/beta` is open) +- `http://localhost:18091` - mock System One upstreams (`tests/e2e/mockjev`, + started by the stack manager and registered on every gateway as the `jev`, + `jev-kev`, and `jev-down` providers) ## Recommended runner @@ -171,6 +174,22 @@ Stateful note: Gemini 3 tool call replayed with its thought signature. They create and clean up their own artifacts and are rerunnable in any order; `S227` reloads the SQLite gateway and therefore stays sequential +- `S229`-`S241` exercise the Jev / Kev System One API (`/v1/systemone`, Kev's + `/permute` and `/separate`, passthrough, pinned versions, misuse negatives, + audit and usage, failover, exact cache on the auth gateway, state guardrails + on the guardrail gateway, managed-key allowlists) against the mock upstream + on port 18091, since no hosted Jev key or Kev server is available; each + prints `SKIPPED:` and exits 0 when the mock is down. They create and delete + their own `$QA_SUFFIX`-scoped virtual models, guardrails, workflows, keys, + and pricing overrides and are rerunnable in any order. `S237` sets a pricing + override and `S240` a guardrail workflow, so both stay sequential +- `S242`-`S244` exercise MCP per-server tool filters and + `disallowed_user_paths` (in-place edits reaching open sessions) and the + master key keeping the caller's user-path header on `/mcp` and audio + uploads; they register `$QA_SUFFIX`-scoped servers and delete them, but + mutate the shared MCP catalog, so they stay sequential +- `S245`-`S246` exercise `developer` messages, `strict` tools, and Gemini's + `allowed_tools` tool choice; they are read-only and rerunnable in any order - `S218` exercises Gemini's native `batchEmbedContents` path (batch input, `dimensions`); read-only and rerunnable in any order - `S219` asserts the effective resilience configuration on @@ -487,6 +506,58 @@ mcp_cleanup_release_servers() { curl -sS -o /dev/null -X DELETE "$base/admin/mcp-servers/$QA_MCP_BETA" || true } +# System One (Jev / Kev) upstreams served by tests/e2e/mockjev: the stack +# manager registers "jev" (hosted shape, keyed), "jev-kev" (keyless Kev +# server), and "jev-down" (always 529) on every gateway. +export JEV_MOCK_BASE="${JEV_MOCK_BASE:-http://localhost:${GOMODEL_RELEASE_MOCK_JEV_PORT:-18091}}" +export QA_SYSTEMONE_QUESTIONS='{"department":{"type":"choice","instructions":"Which team should handle this?","criteria":{"returns":"Exchanges and refunds","shipping":"Delivery delays","billing":"Charges and invoices"}},"escalate":{"type":"noul","instructions":"Does this need urgent human attention?"},"frustration":{"type":"score","instructions":"How frustrated is the customer?","criteria":["Calm","Frustrated","Very angry"]}}' +export QA_SYSTEMONE_CHOICE='{"department":{"type":"choice","instructions":"Which team?","criteria":{"returns":"Returns","billing":"Billing"}}}' + +# Skips when the mock upstream is down, and fails when the gateway was started +# without the mock-backed jev providers (an outdated stack manager). +# usage: systemone_require_mock BASE_URL [curl args...] +systemone_require_mock() { + local base="$1" + shift + if ! curl -fsS "$JEV_MOCK_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock System One upstream is not running on $JEV_MOCK_BASE" + exit 0 + fi + if ! curl -fsS "$base/v1/models" "$@" | jq -e ' + any(.data[]; .id == "jev/jev-latest") and any(.data[]; .id == "jev-kev/kev-latest") + ' >/dev/null; then + echo "error: $base has no mock-backed jev providers; restart it with tests/e2e/manage-release-e2e-stack.sh" >&2 + exit 1 + fi +} + +# Asserts an HTTP status and prints the body on a mismatch. +# usage: assert_http_status WANT GOT BODY_FILE +assert_http_status() { + if [ "$2" != "$1" ]; then + echo "error: expected HTTP $1, got $2" >&2 + cat "$3" >&2 || true + exit 1 + fi +} + +# Polls the audit or usage log until an entry for the request id appears. +# usage: wait_log_entry BASE_URL audit|usage REQUEST_ID OUTPUT_FILE [curl args...] +wait_log_entry() { + local base="$1" kind="$2" rid="$3" out="$4" + shift 4 + for _ in $(seq 1 15); do + curl -fsS "$base/admin/$kind/log?search=$rid&limit=5" "$@" > "$out" + if jq -e --arg rid "$rid" 'any(.entries[]?; .request_id == $rid)' "$out" >/dev/null; then + return 0 + fi + sleep 1 + done + jq . "$out" >&2 || true + echo "error: no $kind entry for $rid on $base" >&2 + exit 1 +} + run_release_budget_enforcement() { local base_url="$1" local budget_path="$2" @@ -3565,7 +3636,7 @@ if jq -e '.providers[] | select(.name == "fireworks") | (.status != "healthy") a exit 0 fi FIREWORKS_MODEL=$(curl -fsS "$BASE_URL/v1/models" \ - | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("llama-v3p1-8b-instruct$"))) + .)[0]') + | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("gpt-oss-120b$"))) + .)[0]') RESP_FILE="$QA_RUN_DIR/s153.chat.json" curl -fsS "$BASE_URL/v1/chat/completions" \ -H 'Content-Type: application/json' \ @@ -3589,7 +3660,7 @@ if jq -e '.providers[] | select(.name == "fireworks") | (.status != "healthy") a exit 0 fi FIREWORKS_MODEL=$(curl -fsS "$BASE_URL/v1/models" \ - | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("llama-v3p1-8b-instruct$"))) + .)[0]') + | jq -er '[.data[].id | select(startswith("fireworks/"))] | (map(select(test("gpt-oss-120b$"))) + .)[0]') SSE_FILE="$QA_RUN_DIR/s154.chat.sse" curl -fsS --no-buffer "$BASE_URL/v1/chat/completions" \ -H 'Content-Type: application/json' \ @@ -6166,3 +6237,946 @@ jq -c --argjson tools "$TOOLS" '{ jq '{provider,answer:.choices[0].message.content}' "$FOLLOW_FILE" assert_chat_response_contains "$FOLLOW_FILE" "gemini" "22" ``` + +## 35. Jev / Kev System One API + +`POST /v1/systemone` (and Kev's `/permute` and `/separate`) forwards TypeSafe +System One decision requests natively. No hosted Jev key or Kev server is +available to the matrix, so the stack manager starts `tests/e2e/mockjev` on +port 18091 and registers three `jev` providers against it on every gateway: +`jev` (hosted-API shape, keyed, lists `jev-latest`/`jev-preview`, accepts any +versioned `jev-X.Y.Z`), `jev-kev` (keyless Kev server whose base URL keeps a +trailing `/v1`, checkpoint `kev-latest` with alias `kev-4b`), and `jev-down` +(lists `kev-down`, answers every System One route with `529`). Each mock +answer carries a `mock` object echoing what reached the upstream (model, +state, questions, extra fields, whether an `Authorization` header arrived, +`X-Request-Id`, and a per-upstream request sequence), so the scenarios can +assert exactly what the gateway forwarded and whether an answer was replayed +from cache. + +### S229 System One providers register and list utility models + +```bash +systemone_require_mock "$BASE_URL" + +MODELS_FILE="$QA_RUN_DIR/s229.models.json" +curl -fsS "$BASE_URL/v1/models" > "$MODELS_FILE" +jq -c '[.data[] | select(.owned_by | startswith("jev")) | {id, categories: .metadata.categories, modes: .metadata.modes}]' "$MODELS_FILE" +jq -e ' + ([.data[] | select(.owned_by | startswith("jev")) | .id] | sort) + == ["jev-down/kev-down","jev-kev/kev-4b","jev-kev/kev-latest","jev/jev-latest","jev/jev-preview"] + and all(.data[] | select(.owned_by | startswith("jev")); .metadata.categories == ["utility"] and ((.metadata.modes // []) | length == 0)) + and any(.data[]; .id == "jev/jev-latest" and .metadata.description == "Latest Jev (mock)" and .created > 0) +' "$MODELS_FILE" >/dev/null + +STATUS_FILE="$QA_RUN_DIR/s229.status.json" +curl -fsS "$BASE_URL/admin/providers/status" > "$STATUS_FILE" +# jev-down turns degraded once failover scenarios have sent it traffic (its +# 529s count against request health), so it only has to be registered. +jq -e ' + [.. | objects | select(.type? == "jev" and has("status")) | {name, status}] | sort_by(.name) as $s + | ($s | map(.name)) == ["jev","jev-down","jev-kev"] + and all($s[]; if .name == "jev-down" then (.status | IN("healthy","degraded")) else .status == "healthy" end) +' "$STATUS_FILE" >/dev/null + +# Passthrough lists models in each upstream's own shape. +curl -fsS "$BASE_URL/p/jev/v1/models" \ + | jq -e '[.models[].name] == ["jev-latest","jev-preview"]' >/dev/null +curl -fsS "$BASE_URL/p/jev-kev/v1/models" \ + | jq -e '.models[0].id == "kev-latest" and .models[0].aliases == ["kev-4b"]' >/dev/null +``` + +### S230 Native `/v1/systemone` answers every question type on hosted Jev + +Sends choice, noul, and score questions plus an extra top-level field. The +answer is relayed unchanged, only `model` is rewritten to the resolved name, +the questions and extra field reach the upstream byte for byte, the client's +`Authorization` header is replaced by the provider key, and the request ID is +forwarded. + +```bash +systemone_require_mock "$BASE_URL" + +for MODEL in jev-latest jev/jev-latest; do + RID="qa-s1-hosted-$QA_SUFFIX-${MODEL//\//-}" + RESP_FILE="$QA_RUN_DIR/s230.${MODEL//\//-}.json" + CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -H 'Authorization: Bearer qa-client-token-must-not-reach-upstream' \ + -H "X-Request-ID: $RID" \ + -d "{\"model\":\"$MODEL\",\"state\":\"Shoes arrived late and I see two charges on my card.\",\"questions\":$QA_SYSTEMONE_QUESTIONS,\"qa_marker\":{\"nested\":[1,2,3]}}") + assert_http_status 200 "$CODE" "$RESP_FILE" + jq -c '{model, answers, usage, mock: (.mock | {upstream, received_model, authorization, request_id})}' "$RESP_FILE" + jq -e --arg rid "$RID" --argjson questions "$QA_SYSTEMONE_QUESTIONS" ' + .model == "jev-1.13.0" + and .answers.department.type == "choice" and (.answers.department.choice | IN("returns","shipping","billing")) + and (.answers.department.probabilities | keys | sort) == ["billing","returns","shipping"] + and .answers.escalate.type == "noul" and (.answers.escalate.noul | type == "number") + and .answers.frustration.type == "score" and .answers.frustration.legend == {"0":"Calm","1":"Frustrated","2":"Very angry"} + and .usage.input_tokens > 0 and .usage.output_tokens > 0 + and .mock.upstream == "jev" + and .mock.received_model == "jev-latest" + and .mock.questions == $questions + and .mock.extra == {"qa_marker":{"nested":[1,2,3]}} + and .mock.authorization == "provider-key" + and .mock.request_id == $rid + ' "$RESP_FILE" >/dev/null +done +``` + +### S231 Keyless Kev server by checkpoint and alias + +The `jev-kev` provider has no key and its base URL ends in `/v1`, which the +provider trims. No `Authorization` header may reach it, not even the client's. + +```bash +systemone_require_mock "$BASE_URL" + +for MODEL in jev-kev/kev-latest kev-latest jev-kev/kev-4b kev-4b; do + RESP_FILE="$QA_RUN_DIR/s231.${MODEL//\//-}.json" + CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -H 'Authorization: Bearer qa-client-token-must-not-reach-upstream' \ + -d "{\"model\":\"$MODEL\",\"state\":\"I was charged twice.\",\"questions\":$QA_SYSTEMONE_CHOICE}") + assert_http_status 200 "$CODE" "$RESP_FILE" + jq -e --arg sent "${MODEL#jev-kev/}" ' + .model == "kev-4b-e2e" + and .mock.upstream == "kev" + and .mock.received_model == $sent + and .mock.authorization == "none" + and .answers.department.type == "choice" + ' "$RESP_FILE" >/dev/null +done +``` + +### S232 Pinned Jev versions route without being listed + +TypeSafe lists only its aliases but accepts any versioned ID. A name that +says which `jev` provider to use reaches it unlisted; a bare unlisted name is +not guessed while several `jev` providers are configured; a model the +upstream rejects comes back with the upstream's status. + +```bash +systemone_require_mock "$BASE_URL" + +for MODEL in jev/jev-1.13.0 jev/jev-1.12.0; do + RESP_FILE="$QA_RUN_DIR/s232.${MODEL//\//-}.json" + CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"$MODEL\",\"state\":\"pinned\",\"questions\":$QA_SYSTEMONE_CHOICE}") + assert_http_status 200 "$CODE" "$RESP_FILE" + jq -e --arg version "${MODEL#jev/}" '.model == $version and .mock.upstream == "jev" and .mock.received_model == $version' "$RESP_FILE" >/dev/null +done + +# Bare and unlisted, with jev, jev-kev, and jev-down all configured. +RESP_FILE="$QA_RUN_DIR/s232.bare.json" +CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-1.13.0\",\"state\":\"pinned\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 404 "$CODE" "$RESP_FILE" +jq -e '.error.code == "model_not_found"' "$RESP_FILE" >/dev/null + +# Routed to jev because the name says so; the upstream rejects it. +RESP_FILE="$QA_RUN_DIR/s232.unknown.json" +CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev/not-a-jev-model\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 404 "$CODE" "$RESP_FILE" +jq -e '.error.type == "not_found_error" and .error.provider == "jev" and (.error.message | contains("not-a-jev-model"))' "$RESP_FILE" >/dev/null +``` + +### S233 A virtual model pins an unlisted Jev version + +The System One docs state that a virtual model can pin a version the same way +a provider-qualified name does (`virtual_models: [{source: ..., target: +jev/jev-1.13.0}]`). This creates one through the admin API and sends a request +through it. + +```bash +systemone_require_mock "$BASE_URL" + +NAME="qa-jev-pinned-$QA_SUFFIX" +cleanup_s233() { + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\"}" || true +} +trap cleanup_s233 EXIT + +VM_FILE="$QA_RUN_DIR/s233.vm.json" +CODE=$(curl -sS -o "$VM_FILE" -w '%{http_code}' -X PUT "$BASE_URL/admin/virtual-models" \ + -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\",\"target_model\":\"jev/jev-1.12.0\"}") +assert_http_status 200 "$CODE" "$VM_FILE" + +RESP_FILE="$QA_RUN_DIR/s233.answer.json" +CODE=$(curl -sS -o "$RESP_FILE" -w '%{http_code}' "$BASE_URL/v1/systemone" \ + -H 'Content-Type: application/json' \ + -d "{\"model\":\"$NAME\",\"state\":\"pinned through a virtual model\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$RESP_FILE" +jq -e '.model == "jev-1.12.0" and .mock.upstream == "jev" and .mock.received_model == "jev-1.12.0"' "$RESP_FILE" >/dev/null +``` + +### S234 Kev diagnostic routes `/permute` and `/separate` + +Kev serves both diagnostic routes; the hosted-shaped `jev` answers them with +its own `404`, and an OpenRouter model is refused before any upstream call +since OpenRouter serves only the evaluation route. + +```bash +systemone_require_mock "$BASE_URL" + +post_systemone() { + local route="$1" body="$2" out="$3" + curl -sS -o "$out" -w '%{http_code}' "$BASE_URL/v1/systemone$route" -H 'Content-Type: application/json' -d "$body" +} + +F="$QA_RUN_DIR/s234.permute.json" +CODE=$(post_systemone /permute "{\"model\":\"jev-kev/kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"n_perm\":3}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.n_perm == 3 and .mock.route == "permute" and .mock.received_model == "kev-latest"' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.permute-default.json" +CODE=$(post_systemone /permute "{\"model\":\"kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.n_perm == 6' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.permute-bad.json" +CODE=$(post_systemone /permute "{\"model\":\"kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"n_perm\":65}" "$F") +assert_http_status 422 "$CODE" "$F" +jq -e '.error.type == "invalid_request_error" and (.error.message | contains("n_perm"))' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.separate.json" +CODE=$(post_systemone /separate "{\"model\":\"jev-kev/kev-4b\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_QUESTIONS}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.separate == true and .mock.route == "separate" and (.answers | keys | length) == 3' "$F" >/dev/null + +F="$QA_RUN_DIR/s234.hosted-permute.json" +CODE=$(post_systemone /permute "{\"model\":\"jev/jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 404 "$CODE" "$F" +jq -e '.error.type == "not_found_error" and .error.provider == "jev"' "$F" >/dev/null + +OPENROUTER_MODEL=$(curl -fsS "$BASE_URL/v1/models" | jq -r '[.data[].id | select(startswith("openrouter/"))][0] // empty') +if [ -n "$OPENROUTER_MODEL" ]; then + F="$QA_RUN_DIR/s234.openrouter-permute.json" + CODE=$(post_systemone /permute "{\"model\":\"$OPENROUTER_MODEL\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") + assert_http_status 400 "$CODE" "$F" + jq -e '.error.param == "model" and (.error.message | contains("answers only /v1/systemone"))' "$F" >/dev/null +else + echo "note: no openrouter model in the catalog; OpenRouter permute refusal not checked" +fi +``` + +### S235 System One misuse is rejected with an explanation (negatives) + +The endpoint never translates: missing or malformed input, chat models (direct +or through a virtual model), and System One models on OpenAI routes are all +`400 invalid_request_error` naming the fix. A malformed question reaches the +upstream and comes back as its `422`. + +```bash +systemone_require_mock "$BASE_URL" + +NAME="qa-s1-chat-vm-$QA_SUFFIX" +cleanup_s235() { + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\"}" || true +} +trap cleanup_s235 EXIT +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$NAME\",\"target_model\":\"openai/gpt-4.1-nano\"}" >/dev/null + +# expect_invalid PATH BODY JQ_MESSAGE_FILTER +expect_invalid() { + local path="$1" body="$2" filter="$3" out + out=$(mktemp "$QA_RUN_DIR/s235.XXXXXX") + local code + code=$(curl -sS -o "$out" -w '%{http_code}' "$BASE_URL$path" -H 'Content-Type: application/json' -d "$body") + assert_http_status 400 "$code" "$out" + if ! jq -e ".error.type == \"invalid_request_error\" and ($filter)" "$out" >/dev/null; then + echo "error: unexpected 400 body for $path" >&2 + cat "$out" >&2 + exit 1 + fi +} + +expect_invalid /v1/systemone "{\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.param == "model" and .error.message == "model is required"' +expect_invalid /v1/systemone '{"model":' \ + '.error.message | startswith("invalid request body")' +expect_invalid /v1/systemone "{\"model\":\"openai/gpt-4.1-nano\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.param == "model" and (.error.message | contains("provider type openai, which has no System One API"))' +expect_invalid /v1/systemone "{\"model\":\"$NAME\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.message | contains("(resolved to \"openai/gpt-4.1-nano\")")' +OPENROUTER_MODEL=$(curl -fsS "$BASE_URL/v1/models" | jq -r '[.data[] | select((.id | startswith("openrouter/")) and ((.metadata.modes // []) | index("chat")))][0].id // empty') +if [ -n "$OPENROUTER_MODEL" ]; then + expect_invalid /v1/systemone "{\"model\":\"$OPENROUTER_MODEL\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + '.error.message | contains("not a System One model")' +fi + +expect_invalid /v1/chat/completions '{"model":"jev/jev-latest","messages":[{"role":"user","content":"hi"}]}' \ + '.error.param == "model" and (.error.message | contains("does not support chat completions") and contains("POST /v1/systemone"))' +expect_invalid /v1/responses '{"model":"jev-kev/kev-latest","input":"hi"}' \ + '.error.message | contains("does not support responses") and contains("POST /v1/systemone")' +expect_invalid /v1/embeddings '{"model":"jev-latest","input":"hi"}' \ + '.error.message | contains("does not support embeddings") and contains("POST /v1/systemone")' + +# A misrouted System One request is an operator mistake, so it is logged. +grep -Fq 'System One request routed to a model without the System One API' \ + "$RELEASE_STACK_DIR/sqlite-main/logs/server.log" + +F="$QA_RUN_DIR/s235.bad-question.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d '{"model":"jev-kev/kev-latest","state":"x","questions":{"q":{"type":"maybe","instructions":"?"}}}') +assert_http_status 422 "$CODE" "$F" +jq -e '.error.provider == "jev" and (.error.message | contains("questions") and contains("maybe"))' "$F" >/dev/null + +F="$QA_RUN_DIR/s235.get.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone") +assert_http_status 405 "$CODE" "$F" +``` + +### S236 Passthrough reaches the same upstreams under `/p/jev*` + +Passthrough forwards the body as sent (no model rewrite), rejects a body that +names the model twice (the upstream parser could pick a value the gateway +never checked), and reaches Kev's diagnostic routes on the suffixed provider. + +```bash +systemone_require_mock "$BASE_URL" + +F="$QA_RUN_DIR/s236.systemone.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/p/jev/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-latest\",\"state\":\"via passthrough\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$F" +jq -e '.model == "jev-1.13.0" and .mock.received_model == "jev-latest" and .mock.authorization == "provider-key"' "$F" >/dev/null + +# The /v1 prefix is optional on passthrough routes. +F="$QA_RUN_DIR/s236.permute.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/p/jev-kev/systemone/permute" -H 'Content-Type: application/json' \ + -d "{\"model\":\"kev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"n_perm\":2}") +assert_http_status 200 "$CODE" "$F" +jq -e '.n_perm == 2 and .mock.upstream == "kev" and .mock.authorization == "none"' "$F" >/dev/null + +F="$QA_RUN_DIR/s236.dup-model.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/p/jev/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE,\"model\":\"jev-preview\"}") +assert_http_status 400 "$CODE" "$F" +jq -e '.error.message | contains("model field is repeated")' "$F" >/dev/null +``` + +### S237 System One calls are audited, filterable, metered, and priced + +Checks the audit entry (route, requested and resolved model, provider, one +successful primary attempt), the `exclude_operation=systemone` request-type +filter, and the usage entry: tokens copied from the answer, recorded under +the model that answered, and priced by an operator override. + +```bash +systemone_require_mock "$BASE_URL" + +# A provider-wide rate must not shadow the rate declared for the version that +# answers the jev-latest alias. +SELECTOR="jev/jev-1.13.0" +PROVIDER_SELECTOR="jev/" +cleanup_s237() { + for selector in "$SELECTOR" "$PROVIDER_SELECTOR"; do + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/model-pricing-overrides" \ + -H 'Content-Type: application/json' -d "{\"selector\":\"$selector\"}" || true + done +} +trap cleanup_s237 EXIT +curl -fsS -X PUT "$BASE_URL/admin/model-pricing-overrides" -H 'Content-Type: application/json' \ + -d "{\"selector\":\"$SELECTOR\",\"pricing\":{\"input_per_mtok\":42,\"output_per_mtok\":0}}" >/dev/null +curl -fsS -X PUT "$BASE_URL/admin/model-pricing-overrides" -H 'Content-Type: application/json' \ + -d "{\"selector\":\"$PROVIDER_SELECTOR\",\"pricing\":{\"input_per_mtok\":1,\"output_per_mtok\":1}}" >/dev/null + +RID="qa-s1-audit-$QA_SUFFIX" +ANSWER_FILE="$QA_RUN_DIR/s237.answer.json" +curl -fsS "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' -H "X-Request-ID: $RID" \ + -d "{\"model\":\"jev/jev-latest\",\"state\":\"audit me\",\"questions\":$QA_SYSTEMONE_QUESTIONS}" > "$ANSWER_FILE" +IN=$(jq -er '.usage.input_tokens' "$ANSWER_FILE") +OUT=$(jq -er '.usage.output_tokens' "$ANSWER_FILE") + +AUDIT_FILE="$QA_RUN_DIR/s237.audit.json" +wait_log_entry "$BASE_URL" audit "$RID" "$AUDIT_FILE" +jq -e --arg rid "$RID" ' + any(.entries[]; .request_id == $rid + and .path == "/v1/systemone" and .method == "POST" and .status_code == 200 + and .requested_model == "jev/jev-latest" and .resolved_model == "jev/jev-latest" + and .provider == "jev" and .provider_name == "jev" + and ([.data.attempts[]? | {kind, provider_name, success}] == [{"kind":"primary","provider_name":"jev","success":true}])) +' "$AUDIT_FILE" >/dev/null + +curl -fsS "$BASE_URL/admin/audit/log?search=$RID&limit=5&exclude_operation=systemone" \ + | jq -e '(.entries // []) | length == 0' >/dev/null +curl -fsS "$BASE_URL/admin/audit/log?search=$RID&limit=5&exclude_operation=chat_completions,provider_passthrough" \ + | jq -e --arg rid "$RID" 'any(.entries[]; .request_id == $rid)' >/dev/null +CODE=$(curl -sS -o "$QA_RUN_DIR/s237.bad-filter.json" -w '%{http_code}' "$BASE_URL/admin/audit/log?exclude_operation=not_an_operation") +assert_http_status 400 "$CODE" "$QA_RUN_DIR/s237.bad-filter.json" + +USAGE_FILE="$QA_RUN_DIR/s237.usage.json" +wait_log_entry "$BASE_URL" usage "$RID" "$USAGE_FILE" +jq -c --arg rid "$RID" '.entries[] | select(.request_id == $rid)' "$USAGE_FILE" +jq -e --arg rid "$RID" --argjson in "$IN" --argjson out "$OUT" ' + any(.entries[]; .request_id == $rid + and .endpoint == "/v1/systemone" and .model == "jev-1.13.0" + and .provider == "jev" and .provider_name == "jev" + and .input_tokens == $in and .output_tokens == $out) +' "$USAGE_FILE" >/dev/null +# The two checks below are independent, so both are reported before failing. +FAILED=0 +if ! jq -e --arg rid "$RID" --argjson total "$((IN + OUT))" \ + 'any(.entries[]; .request_id == $rid and .total_tokens == $total)' "$USAGE_FILE" >/dev/null; then + echo "error: usage total_tokens is not input_tokens + output_tokens ($IN + $OUT) for a System One answer" >&2 + FAILED=1 +fi +# Jev's documented pricing: per input token, output_per_mtok 0. +if ! jq -e --arg rid "$RID" --argjson in "$IN" ' + any(.entries[]; .request_id == $rid and ((((.input_cost // -1) - ($in * 42 / 1000000)) | fabs) < 0.000000001)) + ' "$USAGE_FILE" >/dev/null; then + echo "error: the pricing override (input 42/Mtok, output 0) did not cost the System One usage entry" >&2 + FAILED=1 +fi +[ "$FAILED" = 0 ] +``` + +### S238 Failover moves System One requests between System One targets + +A `failover` virtual model whose primary answers `529` moves to its next +target, skips a chat model without spending an attempt, and records the +answer under the target that did the work. A client error (`422`) is returned +without failover. + +```bash +systemone_require_mock "$BASE_URL" + +FO="qa-s1-failover-$QA_SUFFIX" +FO422="qa-s1-failover-422-$QA_SUFFIX" +FOPIN="qa-s1-failover-pinned-$QA_SUFFIX" +cleanup_s238() { + for name in "$FO" "$FO422" "$FOPIN"; do + curl -sS -o /dev/null -X DELETE "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$name\"}" || true + done +} +trap cleanup_s238 EXIT + +# The down target on its own relays the upstream overload status (or 503 +# once its circuit breaker has opened after earlier reruns). +F="$QA_RUN_DIR/s238.down.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-down/kev-down\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}") +case "$CODE" in 529|503) ;; *) assert_http_status 529 "$CODE" "$F" ;; esac + +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$FO\",\"strategy\":\"failover\",\"targets\":[{\"model\":\"jev-down/kev-down\"},{\"model\":\"openai/gpt-4.1-nano\"},{\"model\":\"jev-kev/kev-latest\"}]}" >/dev/null +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$FO422\",\"strategy\":\"failover\",\"targets\":[{\"model\":\"jev-kev/kev-latest\"},{\"model\":\"jev/jev-latest\"}]}" >/dev/null + +RID="qa-s1-failover-$QA_SUFFIX" +F="$QA_RUN_DIR/s238.failover.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -H "X-Request-ID: $RID" \ + -d "{\"model\":\"$FO\",\"state\":\"fail over please\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$F" +jq -e '.model == "kev-4b-e2e" and .mock.upstream == "kev" and .mock.received_model == "kev-latest"' "$F" >/dev/null + +AUDIT_FILE="$QA_RUN_DIR/s238.audit.json" +wait_log_entry "$BASE_URL" audit "$RID" "$AUDIT_FILE" +jq -c --arg rid "$RID" '.entries[] | select(.request_id == $rid) | [.data.attempts[] | {kind, provider_name, status_code, success}]' "$AUDIT_FILE" +jq -e --arg rid "$RID" --arg fo "$FO" ' + any(.entries[]; .request_id == $rid + and .requested_model == $fo and .resolved_model == "jev-kev/kev-latest" and .provider_name == "jev-kev" + and .data.failover != null + and ([.data.attempts[] | .provider_name] == ["jev-down","jev-kev"]) + and .data.attempts[0].success == false and .data.attempts[1].kind == "failover" and .data.attempts[1].success == true) +' "$AUDIT_FILE" >/dev/null + +USAGE_FILE="$QA_RUN_DIR/s238.usage.json" +wait_log_entry "$BASE_URL" usage "$RID" "$USAGE_FILE" +jq -e --arg rid "$RID" 'any(.entries[]; .request_id == $rid and .provider_name == "jev-kev" and .model == "kev-4b-e2e")' "$USAGE_FILE" >/dev/null + +# A pinned version the catalog does not list is a valid failover target and +# is sent to the provider it names, not to the failed primary's. +curl -fsS -X PUT "$BASE_URL/admin/virtual-models" -H 'Content-Type: application/json' \ + -d "{\"source\":\"$FOPIN\",\"strategy\":\"failover\",\"targets\":[{\"model\":\"jev-down/kev-down\"},{\"model\":\"jev/jev-1.12.0\"}]}" >/dev/null +F="$QA_RUN_DIR/s238.failover-pinned.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"$FOPIN\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}") +assert_http_status 200 "$CODE" "$F" +jq -e '.model == "jev-1.12.0" and .mock.upstream == "jev" and .mock.received_model == "jev-1.12.0"' "$F" >/dev/null + +RID422="qa-s1-failover-422-$QA_SUFFIX" +F="$QA_RUN_DIR/s238.no-failover.json" +CODE=$(curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -H "X-Request-ID: $RID422" \ + -d "{\"model\":\"$FO422\",\"state\":\"x\",\"questions\":{\"q\":{\"type\":\"maybe\",\"instructions\":\"?\"}}}") +assert_http_status 422 "$CODE" "$F" +wait_log_entry "$BASE_URL" audit "$RID422" "$AUDIT_FILE" +jq -e --arg rid "$RID422" ' + any(.entries[]; .request_id == $rid and .status_code == 422 and ([.data.attempts[] | .provider_name] == ["jev-kev"])) +' "$AUDIT_FILE" >/dev/null +``` + +### S239 Identical System One requests hit the exact response cache + +Runs on the auth + exact-cache gateway. The replayed answer carries the same +mock request sequence (the upstream was not called), `Cache-Control: no-cache` +bypasses the cache, a different state misses, and the hit is audited and +recorded in usage as an exact cache hit. + +```bash +systemone_require_mock "$AUTH_BASE_URL" -H "$ADMIN_AUTH_HEADER" + +STATE="release cache probe $QA_SUFFIX" +BODY="{\"model\":\"jev-kev/kev-latest\",\"state\":\"$STATE\",\"questions\":$QA_SYSTEMONE_CHOICE}" +send_cached() { + local rid="$1" headers="$2" body_file="$3" + shift 3 + curl -fsS -D "$headers" -o "$body_file" "$AUTH_BASE_URL/v1/systemone" \ + -H "$ADMIN_AUTH_HEADER" -H 'Content-Type: application/json' -H "X-Request-ID: $rid" "$@" -d "${BODY_OVERRIDE:-$BODY}" +} + +RID1="qa-s1-cache-$QA_SUFFIX-1" +RID2="qa-s1-cache-$QA_SUFFIX-2" +RID3="qa-s1-cache-$QA_SUFFIX-3" +send_cached "$RID1" "$QA_RUN_DIR/s239.1.headers" "$QA_RUN_DIR/s239.1.json" +send_cached "$RID2" "$QA_RUN_DIR/s239.2.headers" "$QA_RUN_DIR/s239.2.json" +send_cached "$RID3" "$QA_RUN_DIR/s239.3.headers" "$QA_RUN_DIR/s239.3.json" -H 'Cache-Control: no-cache' +BODY_OVERRIDE="{\"model\":\"jev-kev/kev-latest\",\"state\":\"$STATE changed\",\"questions\":$QA_SYSTEMONE_CHOICE}" \ + send_cached "qa-s1-cache-$QA_SUFFIX-4" "$QA_RUN_DIR/s239.4.headers" "$QA_RUN_DIR/s239.4.json" + +SEQ1=$(jq -er '.mock.request_seq' "$QA_RUN_DIR/s239.1.json") +grep -Eiq '^X-Cache: *HIT \(exact\)' "$QA_RUN_DIR/s239.2.headers" +jq -e --argjson seq "$SEQ1" '.mock.request_seq == $seq and .model == "kev-4b-e2e"' "$QA_RUN_DIR/s239.2.json" >/dev/null +cmp -s "$QA_RUN_DIR/s239.1.json" "$QA_RUN_DIR/s239.2.json" +for n in 3 4; do + if grep -Eiq '^X-Cache:' "$QA_RUN_DIR/s239.$n.headers"; then + echo "error: request $n should not have been served from cache" >&2 + exit 1 + fi + jq -e --argjson seq "$SEQ1" '.mock.request_seq > $seq' "$QA_RUN_DIR/s239.$n.json" >/dev/null +done + +AUDIT_FILE="$QA_RUN_DIR/s239.audit.json" +wait_log_entry "$AUTH_BASE_URL" audit "$RID2" "$AUDIT_FILE" -H "$ADMIN_AUTH_HEADER" +jq -e --arg rid "$RID2" 'any(.entries[]; .request_id == $rid and .cache_type == "exact" and .status_code == 200 and .path == "/v1/systemone")' "$AUDIT_FILE" >/dev/null + +USAGE_FILE="$QA_RUN_DIR/s239.usage.json" +wait_log_entry "$AUTH_BASE_URL" usage "$RID1" "$USAGE_FILE" -H "$ADMIN_AUTH_HEADER" +# Cache hits are listed only with cache_mode=cached. +for _ in $(seq 1 15); do + curl -fsS "$AUTH_BASE_URL/admin/usage/log?search=$RID2&cache_mode=cached&limit=5" -H "$ADMIN_AUTH_HEADER" > "$USAGE_FILE" + if jq -e --arg rid "$RID2" 'any(.entries[]?; .request_id == $rid)' "$USAGE_FILE" >/dev/null; then + break + fi + sleep 1 +done +jq -e --arg rid "$RID2" ' + any(.entries[]; .request_id == $rid and .cache_type == "exact" and .endpoint == "/v1/systemone" + and .input_tokens > 0 and .total_tokens == .input_tokens + .output_tokens and .provider_name == "jev-kev") +' "$USAGE_FILE" >/dev/null +``` + +### S240 Guardrails see the System One state and nothing else + +Runs on the guardrail gateway. Its global `system_prompt` override has no +place in a decision request, so the edit is dropped with a one-time warning +and the state is forwarded untouched. A workflow scoped to `jev-kev` and a +user path then masks card numbers in a string state and in a JSON state +(which stays JSON), and blocks a forbidden state before any upstream call. + +```bash +systemone_require_mock "$GR_BASE_URL" + +S="${QA_SUFFIX//[^[:alnum:]-]/-}" +MASK="qa-s1-mask-$S" +BLOCK="qa-s1-block-$S" +SCOPE_PATH="/qa/systemone/$S" +WORKFLOW_ID_FILE="$QA_RUN_DIR/s240.workflow.id" +cleanup_s240() { + if [ -s "$WORKFLOW_ID_FILE" ]; then + curl -sS -o /dev/null -X POST "$GR_BASE_URL/admin/workflows/$(cat "$WORKFLOW_ID_FILE")/deactivate" || true + fi + for name in "$MASK" "$BLOCK"; do + curl -sS -o /dev/null -X DELETE "$GR_BASE_URL/admin/guardrails" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$name\"}" || true + done +} +trap cleanup_s240 EXIT + +# Global system_prompt guardrail: dropped, state unchanged, warning logged. +F="$QA_RUN_DIR/s240.global.json" +curl -fsS "$GR_BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -d "{\"model\":\"jev-kev/kev-latest\",\"state\":\"card 4111 1111 1111 1111\",\"questions\":$QA_SYSTEMONE_CHOICE}" > "$F" +jq -e '.mock.received_state == "card 4111 1111 1111 1111" and .answers.department.type == "choice"' "$F" >/dev/null +grep -Fq 'guardrail edits a System One request cannot carry were dropped' "$RELEASE_STACK_DIR/guardrails/logs/server.log" + +jq -n --arg name "$MASK" '{ + name: $name, type: "string_replace", description: "release e2e: mask card numbers", + config: {mode: "regex", rules: "\\b(\\d{4}) \\d{4} \\d{4} (\\d{4})\\b => $1 **** **** $2"} +}' | curl -fsS -X PUT "$GR_BASE_URL/admin/guardrails" -H 'Content-Type: application/json' -d @- >/dev/null +jq -n --arg name "$BLOCK" '{ + name: $name, type: "string_replace", description: "release e2e: block a forbidden state", + config: {mode: "literal", rules: "QA_FORBIDDEN_STATE => x", on_match: "block", message: "QA_SYSTEMONE_BLOCKED"} +}' | curl -fsS -X PUT "$GR_BASE_URL/admin/guardrails" -H 'Content-Type: application/json' -d @- >/dev/null + +jq -n --arg mask "$MASK" --arg block "$BLOCK" --arg path "$SCOPE_PATH" --arg name "qa-s1-guard-$S" '{ + scope_provider_name: "jev-kev", scope_user_path: $path, name: $name, + description: "release e2e: System One state guardrails", + workflow_payload: { + schema_version: 2, + features: {cache: false, audit: true, usage: true, guardrails: true, failover: false}, + steps: [{ref: $mask, phase: "prompt", step: 10}, {ref: $block, phase: "prompt", step: 20}] + } +}' | curl -fsS -X POST "$GR_BASE_URL/admin/workflows" -H 'Content-Type: application/json' -d @- \ + | jq -er '.id' > "$WORKFLOW_ID_FILE" + +guarded() { + curl -sS -o "$2" -w '%{http_code}' "$GR_BASE_URL/v1/systemone" -H 'Content-Type: application/json' \ + -H "X-GoModel-User-Path: $SCOPE_PATH/agent" -d "$1" +} + +F="$QA_RUN_DIR/s240.string.json" +CODE=$(guarded "{\"model\":\"jev-kev/kev-latest\",\"state\":\"card 4111 1111 1111 1234 was charged twice\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e --argjson q "$QA_SYSTEMONE_CHOICE" ' + .mock.received_state == "card 4111 **** **** 1234 was charged twice" and .mock.questions == $q +' "$F" >/dev/null + +F="$QA_RUN_DIR/s240.object.json" +CODE=$(guarded "{\"model\":\"jev-kev/kev-latest\",\"state\":{\"note\":\"card 4111 1111 1111 1234\",\"order\":7},\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.mock.received_state == {"note":"card 4111 **** **** 1234","order":7}' "$F" >/dev/null + +F="$QA_RUN_DIR/s240.blocked.json" +CODE=$(guarded "{\"model\":\"jev-kev/kev-latest\",\"state\":\"QA_FORBIDDEN_STATE\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 400 "$CODE" "$F" +jq -e '.error.message | contains("QA_SYSTEMONE_BLOCKED")' "$F" >/dev/null + +# The same request outside the scoped user path is not masked. +F="$QA_RUN_DIR/s240.unscoped.json" +curl -fsS "$GR_BASE_URL/v1/systemone" -H 'Content-Type: application/json' -H "X-GoModel-User-Path: /qa/other/$S" \ + -d "{\"model\":\"jev-kev/kev-latest\",\"state\":\"card 4111 1111 1111 1234\",\"questions\":$QA_SYSTEMONE_CHOICE}" > "$F" +jq -e '.mock.received_state == "card 4111 1111 1111 1234"' "$F" >/dev/null +``` + +### S241 Managed-key model allowlists cover System One and its passthrough + +Runs on the auth gateway with a key allowed only `jev-kev/kev-latest`. Other +System One models are refused on `/v1/systemone` and on passthrough, including +a passthrough body larger than the 64 KiB peek window whose `model` comes last. + +```bash +systemone_require_mock "$AUTH_BASE_URL" -H "$ADMIN_AUTH_HEADER" + +KEY_FILE="$QA_RUN_DIR/s241.key.json" +cleanup_s241() { + if [ -s "$KEY_FILE" ]; then + curl -sS -o /dev/null -X POST "$AUTH_BASE_URL/admin/auth-keys/$(jq -r '.id' "$KEY_FILE")/deactivate" -H "$ADMIN_AUTH_HEADER" || true + fi +} +trap cleanup_s241 EXIT +curl -fsS -X POST "$AUTH_BASE_URL/admin/auth-keys" -H "$ADMIN_AUTH_HEADER" -H 'Content-Type: application/json' \ + -d "{\"name\":\"qa-s1-allowlist-$QA_SUFFIX\",\"user_path\":\"/qa/systemone/allowlist\",\"allowed_models\":[\"jev-kev/kev-latest\"]}" \ + > "$KEY_FILE" +chmod 600 "$KEY_FILE" +KEY=$(jq -er '.value' "$KEY_FILE") + +with_key() { + curl -sS -o "$3" -w '%{http_code}' "$AUTH_BASE_URL$1" -H "Authorization: Bearer $KEY" -H 'Content-Type: application/json' -d "$2" +} +expect_denied() { + local code + code=$(with_key "$1" "$2" "$3") + assert_http_status 400 "$code" "$3" + jq -e '.error.code == "model_access_denied"' "$3" >/dev/null +} + +F="$QA_RUN_DIR/s241.allowed.json" +CODE=$(with_key /v1/systemone "{\"model\":\"jev-kev/kev-latest\",\"state\":\"allowed\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.mock.upstream == "kev"' "$F" >/dev/null + +expect_denied /v1/systemone "{\"model\":\"jev/jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$QA_RUN_DIR/s241.denied.json" +expect_denied /v1/systemone "{\"model\":\"jev/jev-1.13.0\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$QA_RUN_DIR/s241.denied-pinned.json" +expect_denied /p/jev/v1/systemone "{\"model\":\"jev-latest\",\"state\":\"x\",\"questions\":$QA_SYSTEMONE_CHOICE}" "$QA_RUN_DIR/s241.denied-pt.json" + +BIG_STATE=$(head -c 70000 /dev/zero | tr '\0' 'a') +BIG_BODY_FILE="$QA_RUN_DIR/s241.big-body.json" +jq -n --arg state "$BIG_STATE" --argjson q "$QA_SYSTEMONE_CHOICE" '{state: $state, questions: $q, model: "jev-latest"}' > "$BIG_BODY_FILE" +expect_denied /p/jev/v1/systemone "@$BIG_BODY_FILE" "$QA_RUN_DIR/s241.denied-big.json" + +jq -n --arg state "$BIG_STATE" --argjson q "$QA_SYSTEMONE_CHOICE" '{state: $state, questions: $q, model: "kev-latest"}' > "$BIG_BODY_FILE" +F="$QA_RUN_DIR/s241.allowed-big.json" +CODE=$(with_key /p/jev-kev/v1/systemone "@$BIG_BODY_FILE" "$F") +assert_http_status 200 "$CODE" "$F" +jq -e '.mock.received_length > 65536' "$F" >/dev/null +``` + +## 36. MCP tool and user-path exclusions + +Per-server tool filters and `disallowed_user_paths` are gateway-side access +policy: an edit applies in place without redialing the upstream, and it is +checked on every call, so it also reaches MCP sessions that are already open. +These scenarios register `$QA_SUFFIX`-scoped servers against the mock MCP +upstream on port 18090 and delete them. + +### S242 Tool filters apply in place and reach open sessions + +```bash +if ! curl -fsS "$MCP_UPSTREAM_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock MCP upstream is not running on $MCP_UPSTREAM_BASE" + exit 0 +fi +trap 'mcp_cleanup_release_servers "$BASE_URL"' EXIT + +put_alpha() { + curl -fsS -X PUT "$BASE_URL/admin/mcp-servers" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$QA_MCP_ALPHA\",\"url\":\"$MCP_UPSTREAM_BASE/alpha\",\"transport\":\"http\",\"headers\":{\"X-Mock-Token\":\"$1\"},$2}" >/dev/null +} +alpha_view() { + curl -fsS "$BASE_URL/admin/mcp-servers" | jq -c --arg n "$QA_MCP_ALPHA" '.[] | select(.name == $n)' +} + +put_alpha "$MCP_UPSTREAM_TOKEN" '"disallowed_tools":["add"]' +mcp_wait_status "$BASE_URL" "$QA_MCP_ALPHA" connected +alpha_view | jq -e '.tool_count == 1 and .excluded_tool_count == 1 and .disallowed_tools == ["add"]' >/dev/null +CONNECTED_AT=$(alpha_view | jq -er '.connected_at') +curl -fsS "$BASE_URL/admin/mcp-servers/$QA_MCP_ALPHA/catalog" \ + | jq -e '[.tools[].name] == ["echo"] and [.excluded_tools[].name] == ["add"]' >/dev/null + +SID=$(mcp_initialize "$BASE_URL/mcp" "$QA_RUN_DIR/s242.init.headers" "$QA_RUN_DIR/s242.init.raw") +[ -n "$SID" ] +mcp_initialized "$BASE_URL/mcp" "$SID" +mcp_post "$BASE_URL/mcp" "$SID" '{"jsonrpc":"2.0","id":2,"method":"tools/list"}' \ + | jq -e --arg a "$QA_MCP_ALPHA" '([.result.tools[].name | select(startswith($a + "_"))]) == [$a + "_echo"]' >/dev/null +mcp_post "$BASE_URL/mcp" "$SID" "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_ALPHA}_add\",\"arguments\":{}}}" \ + | jq -e '.error != null' >/dev/null + +# Flip to an allowlist ("Keep hidden") that excludes echo. The stored header +# secret round-trips as ***, and the connection is not redialed. +put_alpha '***' '"allowed_tools":["add"]' +alpha_view | jq -e --arg at "$CONNECTED_AT" ' + .status == "connected" and .connected_at == $at + and .allowed_tools == ["add"] and ((.disallowed_tools // []) | length == 0) + and .tool_count == 1 and .excluded_tool_count == 1 +' >/dev/null + +# echo was listed by the open session before the change; calling it now fails. +mcp_post "$BASE_URL/mcp" "$SID" "{\"jsonrpc\":\"2.0\",\"id\":4,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_ALPHA}_echo\",\"arguments\":{}}}" \ + > "$QA_RUN_DIR/s242.stale-call.json" +jq -e '.error.message | contains("excluded by the gateway tool filters")' "$QA_RUN_DIR/s242.stale-call.json" >/dev/null + +# A new session sees the new filter. +SID2=$(mcp_initialize "$BASE_URL/mcp" "$QA_RUN_DIR/s242.init2.headers" "$QA_RUN_DIR/s242.init2.raw") +mcp_initialized "$BASE_URL/mcp" "$SID2" +mcp_post "$BASE_URL/mcp" "$SID2" '{"jsonrpc":"2.0","id":2,"method":"tools/list"}' \ + | jq -e --arg a "$QA_MCP_ALPHA" '([.result.tools[].name | select(startswith($a + "_"))]) == [$a + "_add"]' >/dev/null +mcp_post "$BASE_URL/mcp" "$SID2" "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_ALPHA}_add\",\"arguments\":{\"marker\":\"QA_MCP_ALLOWED_OK\"}}}" \ + | jq -e '.result.content[0].text | contains("QA_MCP_ALLOWED_OK")' >/dev/null +``` + +### S243 `disallowed_user_paths` carves callers out of a server + +The carve-out wins over `user_paths`, matches whole subtrees, hides the +server from `tools/list` and its per-server endpoint, and a later edit reaches +a session that is already open. Invalid paths are rejected. + +```bash +if ! curl -fsS "$MCP_UPSTREAM_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock MCP upstream is not running on $MCP_UPSTREAM_BASE" + exit 0 +fi +trap 'mcp_cleanup_release_servers "$BASE_URL"' EXIT + +ROOT="/qa/mcp-carve/${QA_SUFFIX//[^[:alnum:]-]/-}" +put_beta() { + curl -sS -o "$QA_RUN_DIR/s243.put.json" -w '%{http_code}' -X PUT "$BASE_URL/admin/mcp-servers" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$QA_MCP_BETA\",\"url\":\"$MCP_UPSTREAM_BASE/beta\",\"transport\":\"http\",\"user_paths\":[\"$ROOT\"],\"disallowed_user_paths\":$1}" +} +CODE=$(put_beta "[\"$ROOT/contractors/\",\"$ROOT/contractors\"]") +assert_http_status 200 "$CODE" "$QA_RUN_DIR/s243.put.json" +mcp_wait_status "$BASE_URL" "$QA_MCP_BETA" connected +curl -fsS "$BASE_URL/admin/mcp-servers" | jq -e --arg n "$QA_MCP_BETA" --arg root "$ROOT" ' + any(.[]; .name == $n and .user_paths == [$root] and .disallowed_user_paths == [$root + "/contractors"]) +' >/dev/null + +# beta_tools SESSION_ID USER_PATH -> prints the beta tool names visible to it +beta_tools() { + mcp_post "$BASE_URL/mcp" "$1" '{"jsonrpc":"2.0","id":2,"method":"tools/list"}' -H "X-GoModel-User-Path: $2" \ + | jq -c --arg b "$QA_MCP_BETA" '[.result.tools[]?.name | select(startswith($b + "_"))]' +} +open_session() { + local sid + sid=$(mcp_initialize "$BASE_URL/mcp" "$QA_RUN_DIR/s243.$2.headers" "$QA_RUN_DIR/s243.$2.raw" -H "X-GoModel-User-Path: $1") + mcp_initialized "$BASE_URL/mcp" "$sid" -H "X-GoModel-User-Path: $1" + echo "$sid" +} + +ENG_SID=$(open_session "$ROOT/eng" eng) +CON_SID=$(open_session "$ROOT/contractors/acme" con) +OUT_SID=$(open_session "/qa/elsewhere" out) +[ "$(beta_tools "$ENG_SID" "$ROOT/eng")" = "[\"${QA_MCP_BETA}_fetch\",\"${QA_MCP_BETA}_search\"]" ] +[ "$(beta_tools "$CON_SID" "$ROOT/contractors/acme")" = "[]" ] +[ "$(beta_tools "$OUT_SID" "/qa/elsewhere")" = "[]" ] + +CODE=$(curl -sS -o "$QA_RUN_DIR/s243.per-server.json" -w '%{http_code}' "$BASE_URL/mcp/$QA_MCP_BETA" \ + -H 'Content-Type: application/json' -H 'Accept: application/json, text/event-stream' \ + -H "X-GoModel-User-Path: $ROOT/contractors/acme" \ + -d '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2025-06-18","capabilities":{},"clientInfo":{"name":"qa-release","version":"1"}}}') +assert_http_status 404 "$CODE" "$QA_RUN_DIR/s243.per-server.json" + +# Carve eng out too: the already-open eng session loses the server. +CODE=$(put_beta "[\"$ROOT/contractors\",\"$ROOT/eng\"]") +assert_http_status 200 "$CODE" "$QA_RUN_DIR/s243.put.json" +mcp_post "$BASE_URL/mcp" "$ENG_SID" \ + "{\"jsonrpc\":\"2.0\",\"id\":5,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_BETA}_search\",\"arguments\":{}}}" \ + -H "X-GoModel-User-Path: $ROOT/eng" > "$QA_RUN_DIR/s243.eng-call.json" +jq -e '.result == null and (.error.message | contains("not available for this user path"))' "$QA_RUN_DIR/s243.eng-call.json" >/dev/null +# A new eng session no longer lists the server. +ENG2_SID=$(open_session "$ROOT/eng" eng2) +[ "$(beta_tools "$ENG2_SID" "$ROOT/eng")" = "[]" ] + +CODE=$(put_beta '["/qa/../escape"]') +assert_http_status 400 "$CODE" "$QA_RUN_DIR/s243.put.json" +jq -e '.error.message | contains("disallowed_user_paths")' "$QA_RUN_DIR/s243.put.json" >/dev/null +``` + +### S244 The master key keeps the caller's user-path header on `/mcp` and audio uploads + +MCP and audio uploads own their transport and take no request snapshot. With +the master key on the auth gateway, the `X-GoModel-User-Path` header must +still scope usage, as it does on chat. + +```bash +if ! curl -fsS "$MCP_UPSTREAM_BASE/healthz" >/dev/null 2>&1; then + echo "SKIPPED: mock MCP upstream is not running on $MCP_UPSTREAM_BASE" + exit 0 +fi +cleanup_s244() { + curl -sS -o /dev/null -X DELETE "$AUTH_BASE_URL/admin/mcp-servers/$QA_MCP_BETA" -H "$ADMIN_AUTH_HEADER" || true +} +trap cleanup_s244 EXIT + +USER_PATH="/qa/master-key-path/${QA_SUFFIX//[^[:alnum:]-]/-}" +curl -fsS -X PUT "$AUTH_BASE_URL/admin/mcp-servers" -H "$ADMIN_AUTH_HEADER" -H 'Content-Type: application/json' \ + -d "{\"name\":\"$QA_MCP_BETA\",\"url\":\"$MCP_UPSTREAM_BASE/beta\",\"transport\":\"http\"}" >/dev/null +for _ in $(seq 1 20); do + if curl -fsS "$AUTH_BASE_URL/admin/mcp-servers" -H "$ADMIN_AUTH_HEADER" \ + | jq -e --arg n "$QA_MCP_BETA" 'any(.[]; .name == $n and .status == "connected")' >/dev/null; then + break + fi + sleep 1 +done + +AUTH_ARGS=(-H "$ADMIN_AUTH_HEADER" -H "X-GoModel-User-Path: $USER_PATH") +SID=$(mcp_initialize "$AUTH_BASE_URL/mcp" "$QA_RUN_DIR/s244.init.headers" "$QA_RUN_DIR/s244.init.raw" "${AUTH_ARGS[@]}") +[ -n "$SID" ] +mcp_initialized "$AUTH_BASE_URL/mcp" "$SID" "${AUTH_ARGS[@]}" +RID="qa-mk-path-mcp-$QA_SUFFIX" +mcp_post "$AUTH_BASE_URL/mcp" "$SID" \ + "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"tools/call\",\"params\":{\"name\":\"${QA_MCP_BETA}_search\",\"arguments\":{\"q\":\"x\"}}}" \ + "${AUTH_ARGS[@]}" -H "X-Request-ID: $RID" | jq -e '.result.content[0].text | startswith("search:")' >/dev/null + +USAGE_FILE="$QA_RUN_DIR/s244.usage.json" +wait_log_entry "$AUTH_BASE_URL" usage "$RID" "$USAGE_FILE" -H "$ADMIN_AUTH_HEADER" +jq -e --arg rid "$RID" --arg p "$USER_PATH" 'any(.entries[]; .request_id == $rid and .provider == "mcp" and .user_path == $p)' "$USAGE_FILE" >/dev/null + +# Audio upload: speech for input, then a multipart transcription. +AUDIO_FILE="$QA_RUN_DIR/s244.speech.wav" +curl -fsS -o "$AUDIO_FILE" "$AUTH_BASE_URL/v1/audio/speech" "${AUTH_ARGS[@]}" -H 'Content-Type: application/json' \ + -d '{"model":"gpt-4o-mini-tts","input":"Release matrix user path check.","voice":"alloy","response_format":"wav"}' +RID="qa-mk-path-audio-$QA_SUFFIX" +curl -fsS "$AUTH_BASE_URL/v1/audio/transcriptions" "${AUTH_ARGS[@]}" -H "X-Request-ID: $RID" \ + -F model=gpt-4o-mini-transcribe -F "file=@$AUDIO_FILE" > "$QA_RUN_DIR/s244.transcription.json" +jq -e '.text | ascii_downcase | contains("user path")' "$QA_RUN_DIR/s244.transcription.json" >/dev/null +wait_log_entry "$AUTH_BASE_URL" usage "$RID" "$USAGE_FILE" -H "$ADMIN_AUTH_HEADER" +jq -e --arg rid "$RID" --arg p "$USER_PATH" 'any(.entries[]; .request_id == $rid and .user_path == $p)' "$USAGE_FILE" >/dev/null +``` + +## 37. Developer messages, strict tools, and tool choice on Anthropic and Gemini + +OpenAI's `developer` role and `strict` function tools are translated for +Anthropic and Gemini's native API instead of being rejected or dropped, and +Gemini also maps `tool_choice: {"type": "allowed_tools", ...}`. + +### S245 Anthropic honors developer messages and strict tools + +```bash +F="$QA_RUN_DIR/s245.developer.json" +curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d '{ + "model":"claude-sonnet-4-6","max_tokens":32, + "messages":[ + {"role":"developer","content":"Whatever the user says, reply with exactly QA_DEVELOPER_ROLE_OK and nothing else."}, + {"role":"user","content":"Tell me a joke."} + ]}' > "$F" +assert_chat_response_contains "$F" "anthropic" "QA_DEVELOPER_ROLE_OK" + +# A strict tool whose schema Anthropic's strict mode would reject as sent +# (minItems 2) is sanitized and forwarded as a strict tool. +F="$QA_RUN_DIR/s245.strict.json" +curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d '{ + "model":"claude-sonnet-4-6","max_tokens":200, + "tools":[{"type":"function","function":{"name":"compare_weather","description":"Compare the weather in several cities","strict":true, + "parameters":{"type":"object","additionalProperties":false,"properties":{"cities":{"type":"array","items":{"type":"string"},"minItems":2}},"required":["cities"]}}}], + "tool_choice":{"type":"function","function":{"name":"compare_weather"}}, + "messages":[{"role":"user","content":"Compare the weather in Warsaw and Krakow."}]}' > "$F" +jq -e ' + .choices[0].message.tool_calls[0].function.name == "compare_weather" + and (.choices[0].message.tool_calls[0].function.arguments | fromjson | .cities | type == "array" and length >= 2) +' "$F" >/dev/null +``` + +### S246 Gemini honors developer messages, strict tools, and `allowed_tools` + +```bash +MODEL="gemini-2.5-flash-lite" +TOOLS='[ + {"type":"function","function":{"name":"lookup_weather","description":"Get the current weather for a city","parameters":{"type":"object","properties":{"city":{"type":"string"}},"required":["city"]}}}, + {"type":"function","function":{"name":"lookup_time","description":"Get the local time in a city","parameters":{"type":"object","properties":{"city":{"type":"string"}},"required":["city"]}}} +]' + +F="$QA_RUN_DIR/s246.developer.json" +curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d "{ + \"model\":\"$MODEL\",\"max_tokens\":32, + \"messages\":[ + {\"role\":\"developer\",\"content\":\"Whatever the user says, reply with exactly QA_DEVELOPER_ROLE_OK and nothing else.\"}, + {\"role\":\"user\",\"content\":\"Tell me a joke.\"} + ]}" > "$F" +assert_chat_response_contains "$F" "gemini" "QA_DEVELOPER_ROLE_OK" + +F="$QA_RUN_DIR/s246.allowed-tools.json" +jq -n --arg model "$MODEL" --argjson tools "$TOOLS" '{ + model: $model, tools: $tools, + tool_choice: {type: "allowed_tools", allowed_tools: {mode: "required", tools: [{type: "function", function: {name: "lookup_time"}}]}}, + messages: [{role: "user", content: "What is the weather in Warsaw?"}] +}' | curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d @- > "$F" +jq -c '[.choices[0].message.tool_calls[]?.function.name]' "$F" +jq -e ' + .choices[0].finish_reason == "tool_calls" + and (.choices[0].message.tool_calls | length) >= 1 + and all(.choices[0].message.tool_calls[]; .function.name == "lookup_time") +' "$F" >/dev/null + +# strict on any tool switches Gemini to VALIDATED function calling. +F="$QA_RUN_DIR/s246.strict.json" +jq -n --arg model "$MODEL" --argjson tools "$TOOLS" '{ + model: $model, + tools: ($tools | map(.function.strict = true)), + tool_choice: "auto", + messages: [{role: "user", content: "Use a tool: what is the weather in Warsaw?"}] +}' | curl -fsS "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d @- > "$F" +jq -e '.choices[0].message.tool_calls[0].function.name == "lookup_weather" + and (.choices[0].message.tool_calls[0].function.arguments | fromjson | .city | test("Warsaw"; "i"))' "$F" >/dev/null + +# An allowed_tools choice with no tools is rejected, as OpenAI does. +F="$QA_RUN_DIR/s246.empty-allowed.json" +CODE=$(jq -n --arg model "$MODEL" --argjson tools "$TOOLS" '{ + model: $model, tools: $tools, + tool_choice: {type: "allowed_tools", allowed_tools: {mode: "auto", tools: []}}, + messages: [{role: "user", content: "hi"}] +}' | curl -sS -o "$F" -w '%{http_code}' "$BASE_URL/v1/chat/completions" -H 'Content-Type: application/json' -d @-) +assert_http_status 400 "$CODE" "$F" +jq -e '.error.type == "invalid_request_error"' "$F" >/dev/null +``` diff --git a/tests/e2e/run-release-e2e.sh b/tests/e2e/run-release-e2e.sh index 8bbb8f320..0a539bf4b 100755 --- a/tests/e2e/run-release-e2e.sh +++ b/tests/e2e/run-release-e2e.sh @@ -363,7 +363,11 @@ is_parallel_safe() { || (number >= 173 && number <= 191) \ || (number >= 197 && number <= 204) \ || (number >= 208 && number <= 226) \ - || number == 228 )) + || number == 228 \ + || (number >= 229 && number <= 236) \ + || (number >= 238 && number <= 239) \ + || number == 241 \ + || (number >= 245 && number <= 246) )) } if (( JOBS > 1 )); then