From 46dba193974c498d64d4f0eb699d3521d3aee931 Mon Sep 17 00:00:00 2001 From: Jopqior Date: Mon, 27 Jul 2026 14:24:51 +0800 Subject: [PATCH] fix(openai): preserve GPT-5.6 max effort in messages bridges Co-Authored-By: Claude --- .../internal/service/openai_compat_model.go | 14 +++ .../service/openai_compat_model_test.go | 110 ++++++++++++++++++ .../service/openai_gateway_messages.go | 3 + .../openai_gateway_messages_chat_fallback.go | 4 +- ...nai_gateway_messages_chat_fallback_test.go | 69 +++++++++++ 5 files changed, 199 insertions(+), 1 deletion(-) diff --git a/backend/internal/service/openai_compat_model.go b/backend/internal/service/openai_compat_model.go index 5f140d0109..6a237b6b81 100644 --- a/backend/internal/service/openai_compat_model.go +++ b/backend/internal/service/openai_compat_model.go @@ -101,3 +101,17 @@ func openAIReasoningEffortToClaudeOutputEffort(effort string) string { return "" } } + +// openAICompatAnthropicReasoningEffort resolves the effort emitted by the +// Anthropic bridge after the final upstream model is known. Anthropic's max is +// normally translated to OpenAI xhigh, but GPT-5.6 accepts the original max +// value on Responses and Chat Completions. +func openAICompatAnthropicReasoningEffort(req *apicompat.AnthropicRequest, upstreamModel, convertedEffort string) string { + if req == nil || req.OutputConfig == nil || !strings.EqualFold(strings.TrimSpace(req.OutputConfig.Effort), "max") { + return convertedEffort + } + if normalized := normalizeOpenAIReasoningEffortForModel(req.OutputConfig.Effort, upstreamModel); normalized != "" { + return normalized + } + return convertedEffort +} diff --git a/backend/internal/service/openai_compat_model_test.go b/backend/internal/service/openai_compat_model_test.go index c7f82d7c3d..68313e2b36 100644 --- a/backend/internal/service/openai_compat_model_test.go +++ b/backend/internal/service/openai_compat_model_test.go @@ -233,6 +233,116 @@ func TestForwardAsAnthropic_NormalizesRoutingAndEffortForGpt54XHigh(t *testing.T t.Logf("response body: %s", rec.Body.String()) } +func TestForwardAsAnthropic_PreservesMaxForFinalGPT56ResponsesModel(t *testing.T) { + t.Parallel() + gin.SetMode(gin.TestMode) + + tests := []struct { + name string + account *Account + model string + defaultMapped string + effort string + wantModel string + wantEffort string + }{ + { + name: "API Key mapping wins and keeps Luna max", + account: rawGPT56ResponsesAPIKeyAccount("luna", "gpt-5.6-luna"), + model: "luna", + defaultMapped: "gpt-5.6-sol", + effort: "max", + wantModel: "gpt-5.6-luna", + wantEffort: "max", + }, + { + name: "OAuth mapping wins and keeps Sol max", + account: rawGPT56ResponsesOAuthAccount("sol", "gpt-5.6-sol"), + model: "sol", + defaultMapped: "gpt-5.6-terra", + effort: "max", + wantModel: "gpt-5.6-sol", + wantEffort: "max", + }, + { + name: "old model still maps max to xhigh", + account: rawGPT56ResponsesAPIKeyAccount("gpt-5.5", "gpt-5.5"), + model: "gpt-5.5", + effort: "max", + wantModel: "gpt-5.5", + wantEffort: "xhigh", + }, + { + name: "GPT56 default remains medium", + account: rawGPT56ResponsesAPIKeyAccount("gpt-5.6-sol", "gpt-5.6-sol"), + model: "gpt-5.6-sol", + wantModel: "gpt-5.6-sol", + wantEffort: "medium", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + rec := httptest.NewRecorder() + c, _ := gin.CreateTestContext(rec) + body := `{"model":"` + tt.model + `","max_tokens":16,"messages":[{"role":"user","content":"hello"}` + if tt.effort != "" { + body += `],"output_config":{"effort":"` + tt.effort + `"},"stream":false}` + } else { + body += `],"stream":false}` + } + c.Request = httptest.NewRequest(http.MethodPost, "/v1/messages", strings.NewReader(body)) + c.Request.Header.Set("Content-Type", "application/json") + + upstream := &httpUpstreamRecorder{resp: openAICompatSSECompletedResponse("resp_gpt56_"+tt.name, tt.wantModel)} + svc := &OpenAIGatewayService{ + httpUpstream: upstream, + cfg: &config.Config{Security: config.SecurityConfig{URLAllowlist: config.URLAllowlistConfig{Enabled: false}}}, + } + + result, err := svc.ForwardAsAnthropic(context.Background(), c, tt.account, []byte(body), "", tt.defaultMapped) + require.NoError(t, err) + require.NotNil(t, result) + require.Equal(t, tt.wantModel, result.UpstreamModel) + require.Equal(t, tt.wantModel, gjson.GetBytes(upstream.lastBody, "model").String()) + require.Equal(t, tt.wantEffort, gjson.GetBytes(upstream.lastBody, "reasoning.effort").String()) + require.NotNil(t, result.ReasoningEffort) + require.Equal(t, tt.wantEffort, *result.ReasoningEffort) + }) + } +} + +func rawGPT56ResponsesAPIKeyAccount(requestedModel, mappedModel string) *Account { + return &Account{ + ID: 501, + Name: "gpt56-apikey", + Platform: PlatformOpenAI, + Type: AccountTypeAPIKey, + Concurrency: 1, + Credentials: map[string]any{ + "api_key": "sk-test", + "base_url": "https://api.example.com/v1", + "model_mapping": map[string]any{requestedModel: mappedModel}, + }, + Extra: map[string]any{"use_responses_api": true}, + } +} + +func rawGPT56ResponsesOAuthAccount(requestedModel, mappedModel string) *Account { + return &Account{ + ID: 502, + Name: "gpt56-oauth", + Platform: PlatformOpenAI, + Type: AccountTypeOAuth, + Concurrency: 1, + Credentials: map[string]any{ + "access_token": "oauth-token", + "chatgpt_account_id": "chatgpt-acc", + "model_mapping": map[string]any{requestedModel: mappedModel}, + }, + } +} + func TestForwardAsAnthropic_MappedClaudeModelAcceptsChatUsageShape(t *testing.T) { t.Parallel() gin.SetMode(gin.TestMode) diff --git a/backend/internal/service/openai_gateway_messages.go b/backend/internal/service/openai_gateway_messages.go index d21cba5c50..63a39826b0 100644 --- a/backend/internal/service/openai_gateway_messages.go +++ b/backend/internal/service/openai_gateway_messages.go @@ -124,6 +124,9 @@ func (s *OpenAIGatewayService) ForwardAsAnthropic( } responsesReq.Model = upstreamModel + if responsesReq.Reasoning != nil { + responsesReq.Reasoning.Effort = openAICompatAnthropicReasoningEffort(&anthropicReq, upstreamModel, responsesReq.Reasoning.Effort) + } if previousResponseID != "" { responsesReq.PreviousResponseID = previousResponseID trimAnthropicCompatResponsesInputToLatestTurn(responsesReq) diff --git a/backend/internal/service/openai_gateway_messages_chat_fallback.go b/backend/internal/service/openai_gateway_messages_chat_fallback.go index 82eee0e853..c66077a0bf 100644 --- a/backend/internal/service/openai_gateway_messages_chat_fallback.go +++ b/backend/internal/service/openai_gateway_messages_chat_fallback.go @@ -60,12 +60,14 @@ func (s *OpenAIGatewayService) forwardAnthropicViaRawChatCompletions( billingModel := resolveOpenAIForwardModel(account, anthropicReq.Model, defaultMappedModel) upstreamModel := normalizeOpenAIModelForUpstream(account, billingModel) chatReq.Model = upstreamModel + chatReq.ReasoningEffort = openAICompatAnthropicReasoningEffort(&anthropicReq, upstreamModel, chatReq.ReasoningEffort) chatReq.Stream = clientStream if clientStream { chatReq.StreamOptions = &apicompat.ChatStreamOptions{IncludeUsage: true} } - reasoningEffort := extractOpenAIReasoningEffortFromBody(body, upstreamModel, billingModel, originalModel) + convertedEffort := chatReq.ReasoningEffort + reasoningEffort := &convertedEffort reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, billingModel) serviceTier := extractOpenAIServiceTierFromBody(body) diff --git a/backend/internal/service/openai_gateway_messages_chat_fallback_test.go b/backend/internal/service/openai_gateway_messages_chat_fallback_test.go index b36b76ffa8..0ca6b5f106 100644 --- a/backend/internal/service/openai_gateway_messages_chat_fallback_test.go +++ b/backend/internal/service/openai_gateway_messages_chat_fallback_test.go @@ -45,6 +45,75 @@ func (r *errTailReader) Read(p []byte) (int, error) { func (r *errTailReader) Close() error { return nil } +func TestForwardAsAnthropic_ForceChatCompletionsPreservesFinalModelReasoningEffort(t *testing.T) { + gin.SetMode(gin.TestMode) + + tests := []struct { + name string + model string + mapped string + effortJSON string + wantEffort string + }{ + { + name: "GPT56 max", + model: "luna", + mapped: "gpt-5.6-luna", + effortJSON: `,"output_config":{"effort":"max"}`, + wantEffort: "max", + }, + { + name: "old model max", + model: "gpt-5.5", + mapped: "gpt-5.5", + effortJSON: `,"output_config":{"effort":"max"}`, + wantEffort: "xhigh", + }, + { + name: "high remains high", + model: "gpt-5.6-luna", + mapped: "gpt-5.6-luna", + effortJSON: `,"output_config":{"effort":"high"}`, + wantEffort: "high", + }, + { + name: "omitted defaults medium", + model: "gpt-5.6-luna", + mapped: "gpt-5.6-luna", + wantEffort: "medium", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + rec := httptest.NewRecorder() + c, _ := gin.CreateTestContext(rec) + body := `{"model":"` + tt.model + `","max_tokens":16,"messages":[{"role":"user","content":"hello"}]` + tt.effortJSON + `,"stream":false}` + c.Request = httptest.NewRequest(http.MethodPost, "/v1/messages", bytes.NewReader([]byte(body))) + c.Request.Header.Set("Content-Type", "application/json") + + upstream := &httpUpstreamRecorder{resp: &http.Response{ + StatusCode: http.StatusOK, + Header: http.Header{"Content-Type": []string{"application/json"}}, + Body: io.NopCloser(strings.NewReader( + `{"id":"chatcmpl_effort","object":"chat.completion","model":"` + tt.mapped + `","choices":[{"index":0,"message":{"role":"assistant","content":"ok"},"finish_reason":"stop"}],"usage":{"prompt_tokens":3,"completion_tokens":2,"total_tokens":5}}`, + )), + }} + account := forceChatMessagesFallbackAccount() + account.Credentials["model_mapping"] = map[string]any{tt.model: tt.mapped} + + svc := &OpenAIGatewayService{cfg: rawChatCompletionsTestConfig(), httpUpstream: upstream} + result, err := svc.ForwardAsAnthropic(context.Background(), c, account, []byte(body), "", "") + require.NoError(t, err) + require.NotNil(t, result) + require.Equal(t, tt.mapped, gjson.GetBytes(upstream.lastBody, "model").String()) + require.Equal(t, tt.wantEffort, gjson.GetBytes(upstream.lastBody, "reasoning_effort").String()) + require.NotNil(t, result.ReasoningEffort) + require.Equal(t, tt.wantEffort, *result.ReasoningEffort) + }) + } +} + func TestForwardAsAnthropic_ForceChatCompletionsNonStreaming(t *testing.T) { gin.SetMode(gin.TestMode)