diff --git a/docs/providers/kimicode.mdx b/docs/providers/kimicode.mdx index 214ccff17..cb91f2358 100644 --- a/docs/providers/kimicode.mdx +++ b/docs/providers/kimicode.mdx @@ -7,8 +7,17 @@ keywords: ["Kimi Code", "Moonshot", "quota", "provider setup"] Kimi Code is an OpenAI-compatible coding assistant served at `https://api.kimi.com/coding/v1`. GoModel routes chat, model listing, embeddings, and passthrough requests through the shared -OpenAI adapter. The `/v1/responses` endpoint is translated through chat completions, while -files and batches are not supported by the upstream endpoint. +OpenAI adapter. The `/v1/responses` endpoint is forwarded natively to the upstream Responses +API instead of being translated through chat completions, while files and batches are not +supported by the upstream endpoint. + +Kimi Code retains no responses. Requests with `store: true` are rewritten to `store: false` +(the upstream rejects `store: true` with a 400). Chaining works only through GoModel: with a +response store configured, the gateway expands a `previous_response_id` chain by replaying the +stored history into the request before dispatch, and a `conversation` reference resolves +through the conversation store the same way. Without those stores, a request carrying +`previous_response_id` or `conversation` is rejected with an invalid-request error, because +the upstream can never resolve the referenced state. ## Configure diff --git a/internal/providers/kimicode/kimicode.go b/internal/providers/kimicode/kimicode.go index 2899e6acb..e9b0f610c 100644 --- a/internal/providers/kimicode/kimicode.go +++ b/internal/providers/kimicode/kimicode.go @@ -1,11 +1,17 @@ // Package kimicode provides Kimi Code API integration for the LLM gateway. // -// The "kimicode" provider routes to Kimi Code's OpenAI-compatible chat -// completions endpoint, so all transport goes through the shared chat-centric -// adapter and model IDs are forwarded unchanged. +// The "kimicode" provider routes to Kimi Code's OpenAI-compatible API: chat +// completions, model listing, embeddings, and passthrough go through the +// shared chat-centric adapter, while the Responses API is served natively by +// the upstream /responses endpoint. Kimi Code retains no responses, so +// store=true is pinned to false. package kimicode import ( + "context" + "io" + "strings" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/providers" "github.com/enterpilot/gomodel/internal/providers/openai" @@ -23,9 +29,11 @@ var Registration = providers.Registration{ } // Provider implements the core.Provider interface for Kimi Code. Kimi Code is -// OpenAI-compatible, so all transport goes through the shared chat-centric +// OpenAI-compatible, so most transport goes through the shared chat-centric // adapter: chat completions, model listing, embeddings, and passthrough are -// exposed via the embedded *openai.ChatCompatible. +// exposed via the embedded *openai.ChatCompatible. The Responses API is +// forwarded natively to the upstream /responses endpoint through the same +// adapter instance. type Provider struct { *openai.ChatCompatible } @@ -39,3 +47,72 @@ func New(cfg providers.ProviderConfig, opts providers.ProviderOptions) core.Prov BaseURL: providers.ResolveBaseURL(cfg.BaseURL, defaultBaseURL), })} } + +// Responses serves the Responses API natively through the upstream /responses +// endpoint. Kimi Code retains no responses, so a non-empty +// previous_response_id is rejected before any upstream call (see +// rejectPreviousResponseID); store=true is pinned to false by +// adaptResponsesRequest. +func (p *Provider) Responses(ctx context.Context, req *core.ResponsesRequest) (*core.ResponsesResponse, error) { + if err := rejectPreviousResponseID(req); err != nil { + return nil, err + } + return p.Compatible().Responses(ctx, adaptResponsesRequest(req)) +} + +// StreamResponses forwards the request to the upstream /responses endpoint +// with stream enabled, returning its Responses SSE stream. Like Responses, it +// rejects a non-empty previous_response_id before any upstream call. +func (p *Provider) StreamResponses(ctx context.Context, req *core.ResponsesRequest) (io.ReadCloser, error) { + if err := rejectPreviousResponseID(req); err != nil { + return nil, err + } + return p.Compatible().StreamResponses(ctx, adaptResponsesRequest(req)) +} + +// rejectPreviousResponseID fails requests chaining from earlier state: +// Kimi Code cannot resolve a previous response ID or a gateway-local +// conversation upstream, and answering statelessly would silently drop the +// conversation context the caller expects. The rejection only fires when the +// gateway has no store to expand the chain with; requests whose state the +// gateway already replayed into input (both fields cleared) pass through. +// The ID check mirrors the gateway and the chat-translation validator, both +// of which treat a whitespace-only ID as empty. +func rejectPreviousResponseID(req *core.ResponsesRequest) error { + if req == nil { + return nil + } + if req.Conversation != nil { + return core.NewInvalidRequestError( + "kimicode does not retain responses: conversation is not supported", nil) + } + if strings.TrimSpace(req.PreviousResponseID) == "" { + return nil + } + return core.NewInvalidRequestError( + "kimicode does not retain responses: previous_response_id is not supported", nil) +} + +// adaptResponsesRequest pins store to false: the service retains no +// responses, so store=true fails upstream with a 400 (Postel's law — adapt +// instead of failing). A whitespace-only previous_response_id is treated as +// empty by rejectPreviousResponseID and cleared here, because omitempty does +// not omit a non-empty whitespace string and the upstream cannot resolve it. +func adaptResponsesRequest(req *core.ResponsesRequest) *core.ResponsesRequest { + if req == nil { + return nil + } + whitespaceID := req.PreviousResponseID != "" && strings.TrimSpace(req.PreviousResponseID) == "" + if (req.Store == nil || !*req.Store) && !whitespaceID { + return req + } + cp := *req + if req.Store != nil && *req.Store { + disabled := false + cp.Store = &disabled + } + if whitespaceID { + cp.PreviousResponseID = "" + } + return &cp +} diff --git a/internal/providers/kimicode/kimicode_test.go b/internal/providers/kimicode/kimicode_test.go index 54eecf1ce..23eea1d8a 100644 --- a/internal/providers/kimicode/kimicode_test.go +++ b/internal/providers/kimicode/kimicode_test.go @@ -1,9 +1,14 @@ package kimicode import ( + "context" "net/http" + "net/http/httptest" "testing" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/enterpilot/gomodel/internal/core" "github.com/enterpilot/gomodel/internal/llmclient" "github.com/enterpilot/gomodel/internal/providers" @@ -12,7 +17,8 @@ import ( // Kimi Code is a thin wrapper over the shared chat-centric adapter and // forwards embeddings upstream unchanged, so the shared contract covers its -// surface. +// surface. The Responses API is forwarded natively to the upstream /responses +// endpoint. func TestChatCompatibleContract(t *testing.T) { providertest.AssertChatCompatible(t, providertest.ChatCompatible{ Registration: Registration, @@ -23,6 +29,270 @@ func TestChatCompatibleContract(t *testing.T) { opts.HTTPClient = client return New(providers.ProviderConfig{APIKey: apiKey, BaseURL: baseURL}, opts) }, - Embeddings: true, + Embeddings: true, + NativeResponses: true, + }) +} + +func boolPtr(b bool) *bool { return &b } + +// newTestProvider builds a provider wired to the test server through the +// injected HTTP client, matching how the shared contract constructs it. +func newTestProvider(server *httptest.Server) core.Provider { + opts := providertest.Options(llmclient.Hooks{}) + opts.HTTPClient = server.Client() + return New(providers.ProviderConfig{APIKey: "kimi-key", BaseURL: server.URL}, opts) +} + +func TestRejectPreviousResponseID(t *testing.T) { + t.Run("nil request passes through", func(t *testing.T) { + assert.NoError(t, rejectPreviousResponseID(nil)) + }) + + t.Run("clean request passes through", func(t *testing.T) { + assert.NoError(t, rejectPreviousResponseID(&core.ResponsesRequest{Model: "kimi-for-coding", Input: "hi"})) + }) +} + +func TestAdaptResponsesRequest(t *testing.T) { + t.Run("nil passes through", func(t *testing.T) { + assert.Nil(t, adaptResponsesRequest(nil)) + }) + + t.Run("clean request is returned unchanged", func(t *testing.T) { + req := &core.ResponsesRequest{Model: "kimi-for-coding", Input: "hi"} + assert.Same(t, req, adaptResponsesRequest(req)) + }) + + t.Run("store true is pinned to false", func(t *testing.T) { + req := &core.ResponsesRequest{Model: "kimi-for-coding", Input: "hi", Store: boolPtr(true)} + got := adaptResponsesRequest(req) + require.NotSame(t, req, got, "adapted request should be a copy") + require.NotNil(t, got.Store) + assert.False(t, *got.Store) + // The caller's request must not be mutated. + assert.True(t, *req.Store, "original request Store was mutated") + }) + + t.Run("previous_response_id is preserved", func(t *testing.T) { + // adaptResponsesRequest does not touch PreviousResponseID; the + // Responses/StreamResponses methods reject it instead (tested below). + req := &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "hi", + PreviousResponseID: "resp_old", + Store: boolPtr(false), + } + got := adaptResponsesRequest(req) + assert.Equal(t, "resp_old", got.PreviousResponseID) + require.NotNil(t, got.Store) + assert.False(t, *got.Store, "explicit store=false should stay false") + }) + + t.Run("whitespace-only previous_response_id is cleared", func(t *testing.T) { + req := &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "hi", + PreviousResponseID: " ", + } + got := adaptResponsesRequest(req) + require.NotSame(t, req, got, "adapted request should be a copy") + assert.Empty(t, got.PreviousResponseID) + assert.Equal(t, " ", req.PreviousResponseID, "original request was mutated") + }) +} + +// responsesGoldenBody mirrors a real non-streaming /responses reply from the +// Kimi Code upstream (recorded 2026-09-08, trimmed to the members GoModel +// consumes). The upstream reply keeps extra members (prompt_cache_key, +// safety_identifier, service_tier); unknown members are ignored on decode. +const responsesGoldenBody = `{ + "id": "resp_golden", + "object": "response", + "created_at": 1788866012, + "completed_at": 1788866014, + "status": "completed", + "output": [ + { + "type": "reasoning", + "id": "rs_golden", + "status": "completed", + "summary": [{"type": "summary_text", "text": "Simple request."}] + }, + { + "type": "message", + "id": "msg_golden", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "OK", "annotations": []}] + } + ], + "usage": { + "input_tokens": 88, + "input_tokens_details": {"cache_write_tokens": 12, "cached_tokens": 88}, + "output_tokens": 53, + "output_tokens_details": {"reasoning_tokens": 37}, + "total_tokens": 141 + }, + "store": false, + "model": "kimi-for-coding" +}` + +// TestResponses_ForwardsGatewayReplayedHistory covers what the gateway +// dispatches after expanding a previous_response_id chain against its +// response store: the stored history is replayed into input as items +// (reasoning and message items among them, IDs stripped) and +// previous_response_id is cleared. Kimi Code must forward that replayed +// input to /responses as stored instead of rejecting it. +func TestResponses_ForwardsGatewayReplayedHistory(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) + + provider := newTestProvider(server) + + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: []any{ + map[string]any{ + "type": "message", + "role": "user", + "content": []any{map[string]any{"type": "input_text", "text": "remember: zebra"}}, + }, + map[string]any{ + "type": "reasoning", + "status": "completed", + "summary": []any{ + map[string]any{"type": "summary_text", "text": "thinking about zebras"}, + }, + }, + map[string]any{ + "type": "message", + "role": "assistant", + "status": "completed", + "content": []any{map[string]any{"type": "output_text", "text": "the word is zebra"}}, + }, + map[string]any{ + "type": "message", + "role": "user", + "content": []any{map[string]any{"type": "input_text", "text": "what is the word?"}}, + }, + }, + Store: boolPtr(true), + }) + require.NoError(t, err) + require.NotNil(t, resp) + + req := capture.Last(t) + assert.Equal(t, "/responses", req.Path) + wire := req.JSON(t) + assert.Equal(t, false, wire["store"], "store is pinned to false on the wire") + assert.NotContains(t, wire, "previous_response_id") + + items, ok := wire["input"].([]any) + require.True(t, ok, "wire input = %#v, want replayed items", wire["input"]) + require.Len(t, items, 4) + reasoning, ok := items[1].(map[string]any) + require.True(t, ok) + assert.Equal(t, "reasoning", reasoning["type"], "replayed reasoning item must be forwarded unchanged") + assert.Equal(t, "user", items[3].(map[string]any)["role"], "the client's own turn is replayed last") +} + +// Kimi Code retains no responses, so a request chaining from an earlier +// response that the gateway could not expand (no store configured) must be +// rejected before any upstream call instead of being answered statelessly. +func TestResponses_RejectsPreviousResponseID(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) + + provider := newTestProvider(server) + + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + PreviousResponseID: "resp_old", + }) + require.Error(t, err) + assert.Nil(t, resp) + + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr, "error type = %T, want *core.GatewayError", err) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + assert.Contains(t, gatewayErr.Error(), "previous_response_id") + assert.Equal(t, 0, capture.Count(), "rejected request must not reach the upstream") + + t.Run("whitespace-only ID passes through", func(t *testing.T) { + // The gateway and the chat-translation validator treat a + // whitespace-only ID as empty; the provider must behave the same, + // and the unresolvable value must not reach the wire (omitempty + // does not omit a non-empty whitespace string). + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + PreviousResponseID: " ", + }) + require.NoError(t, err) + require.NotNil(t, resp) + assert.Equal(t, 1, capture.Count(), "whitespace-only ID is treated as empty") + wire := capture.Last(t).JSON(t) + _, present := wire["previous_response_id"] + assert.False(t, present, "whitespace-only ID must be omitted from the wire") + }) + + t.Run("conversation reference is rejected", func(t *testing.T) { + before := capture.Count() + resp, err := provider.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + Conversation: &core.ResponsesConversationRef{ID: "conv_old"}, + }) + require.Error(t, err) + assert.Nil(t, resp) + + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + assert.Contains(t, gatewayErr.Error(), "conversation") + assert.Equal(t, before, capture.Count(), "conversation request must not reach the upstream") + }) +} + +func TestStreamResponses_RejectsPreviousResponseID(t *testing.T) { + server, capture := providertest.SSEServer(t, "") + + provider := newTestProvider(server) + + stream, err := provider.StreamResponses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", + PreviousResponseID: "resp_old", + }) + require.Error(t, err) + assert.Nil(t, stream) + + var gatewayErr *core.GatewayError + require.ErrorAs(t, err, &gatewayErr, "error type = %T, want *core.GatewayError", err) + assert.Equal(t, core.ErrorTypeInvalidRequest, gatewayErr.Type) + assert.Contains(t, gatewayErr.Error(), "previous_response_id") + assert.Equal(t, 0, capture.Count(), "rejected request must not reach the upstream") +} + +// SetBaseURL must retarget the single adapter serving both the chat-centric +// surface and the native Responses endpoint. +func TestSetBaseURL(t *testing.T) { + server, capture := providertest.JSONServer(t, http.StatusOK, responsesGoldenBody) + + p := New(providers.ProviderConfig{APIKey: "kimi-key"}, providertest.Options(llmclient.Hooks{})) + kp, ok := p.(*Provider) + require.True(t, ok) + require.Equal(t, defaultBaseURL, kp.GetBaseURL()) + + kp.SetBaseURL(server.URL) + + assert.Equal(t, server.URL, kp.GetBaseURL()) + + _, err := p.Responses(context.Background(), &core.ResponsesRequest{ + Model: "kimi-for-coding", + Input: "Say OK", }) + require.NoError(t, err) + require.Equal(t, 1, capture.Count()) + assert.Equal(t, "/responses", capture.Last(t).Path) } diff --git a/internal/providers/openai/chat_compatible.go b/internal/providers/openai/chat_compatible.go index 437af6437..aaec07f09 100644 --- a/internal/providers/openai/chat_compatible.go +++ b/internal/providers/openai/chat_compatible.go @@ -47,6 +47,14 @@ func bearerHeaders(req *http.Request, apiKey string) { providers.SetAuthHeaders(req, apiKey, providers.AuthHeaderConfig{AuthScheme: "Bearer "}) } +// Compatible exposes the underlying OpenAI-compatible adapter, so a provider +// that embeds ChatCompatible can serve a native Responses endpoint through +// the same instance instead of building a second adapter with the same +// configuration. +func (c *ChatCompatible) Compatible() *CompatibleProvider { + return c.compatible +} + // SetBaseURL allows configuring a custom base URL for the provider. func (c *ChatCompatible) SetBaseURL(url string) { c.compatible.SetBaseURL(url) diff --git a/internal/providers/openai/chat_compatible_test.go b/internal/providers/openai/chat_compatible_test.go new file mode 100644 index 000000000..975094664 --- /dev/null +++ b/internal/providers/openai/chat_compatible_test.go @@ -0,0 +1,21 @@ +package openai + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/enterpilot/gomodel/internal/providers" +) + +// Compatible must expose the adapter the chat-centric surface was built +// with, so providers serving native Responses use the same instance. +func TestChatCompatible_Compatible(t *testing.T) { + chat := NewChatCompatible("key", providers.ProviderOptions{}, CompatibleProviderConfig{ + ProviderName: "test", + BaseURL: "https://example.com/v1", + }) + require.NotNil(t, chat) + assert.Same(t, chat.compatible, chat.Compatible()) +} diff --git a/internal/server/previous_response_test.go b/internal/server/previous_response_test.go index e22376ae2..1296fa0e0 100644 --- a/internal/server/previous_response_test.go +++ b/internal/server/previous_response_test.go @@ -121,6 +121,57 @@ func TestResponsesWithPreviousResponseID_ChainCarriesFullHistory(t *testing.T) { require.Len(t, second.InputItems, 1) } +// TestResponsesWithPreviousResponseID_KimicodeChainReplaysHistory covers the +// native-Responses kimicode provider chaining through the gateway: kimicode +// has no Responses lifecycle, so the gateway treats it as translated and, +// with a response store, replays the stored chain into input instead of +// forwarding the id. The replayed items — reasoning output included, item +// ids stripped — are what reaches kimicode's /responses upstream. +func TestResponsesWithPreviousResponseID_KimicodeChainReplaysHistory(t *testing.T) { + provider := previousResponseTestProvider(t, "kimicode") + srv := New(provider, nil) + + // A turn the gateway served for kimicode earlier: the snapshot holds the + // client's input items and the provider's output, reasoning included. + err := srv.handler.currentResponseStore().Create(context.Background(), &responsestore.StoredResponse{ + Response: &core.ResponsesResponse{ + ID: "resp_kimi_1", Object: "response", Status: "completed", + Output: []core.ResponsesOutputItem{ + { + ID: "rs_1", Type: "reasoning", Status: "completed", + ExtraFields: core.UnknownJSONFieldsFromMap(map[string]json.RawMessage{ + "summary": json.RawMessage(`[{"type":"summary_text","text":"thinking about zebras"}]`), + }), + }, + {ID: "msg_1", Type: "message", Role: "assistant", Content: []core.ResponsesContentItem{{Type: "output_text", Text: "the word is zebra"}}}, + }, + }, + InputItems: []json.RawMessage{json.RawMessage(`{"id":"in_1","type":"message","role":"user","content":[{"type":"input_text","text":"remember: zebra"}]}`)}, + Provider: "kimicode", + }) + require.NoError(t, err) + + rec := postResponses(t, srv, `{"model":"gpt-5-mini","input":"what is the word?","previous_response_id":"resp_kimi_1"}`) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + forwarded := provider.capturedResponsesReq + require.NotNil(t, forwarded) + require.Empty(t, forwarded.PreviousResponseID, "translated providers get the id stripped before dispatch") + + items := forwardedInputItems(t, provider.capturingProvider) + require.Len(t, items, 4) + require.Equal(t, "reasoning", items[1]["type"], "stored reasoning output must replay unchanged: %#v", items[1]) + summary, _ := json.Marshal(items[1]["summary"]) + require.Contains(t, string(summary), "thinking about zebras") + for i, item := range items[:3] { + _, hasID := item["id"] + require.False(t, hasID, "stored item id must be stripped before dispatch (item %d): %#v", i, item) + } + text, _ := json.Marshal(items[2]["content"]) + require.Contains(t, string(text), "the word is zebra") + require.Equal(t, "user", items[3]["role"], "the client's own turn is replayed last") +} + func TestResponsesWithPreviousResponseID_StreamingChainedTurn(t *testing.T) { provider := previousResponseTestProvider(t, "anthropic") provider.streamData = "event: response.completed\ndata: {\"type\":\"response.completed\",\"response\":{\"id\":\"resp_s\",\"object\":\"response\",\"status\":\"completed\",\"output\":[]}}\n\ndata: [DONE]\n\n"