Merge branch 'main' into fix/openai-images-capability-loss-cooldown

This commit is contained in:
Wesley Liddick
2026-08-28 10:54:47 +08:00
committed by GitHub
3 changed files with 266 additions and 3 deletions
@@ -37,6 +37,13 @@ type OpenAIImagesUpstreamError struct {
Message string
Param string
UpstreamRequestID string
// SynthesizedFromModelText marks an error the gateway inferred from the
// model's plain-text output instead of reading it off a structured upstream
// error frame. Such a verdict describes this one turn ("the model answered
// with words instead of an image"), not the account — see
// shouldCoolOpenAIImagesToolForError.
SynthesizedFromModelText bool
}
func (e *OpenAIImagesUpstreamError) Error() string {
@@ -731,6 +738,10 @@ func openAIImagesTextFallbackErrorForText(text string) *OpenAIImagesUpstreamErro
ErrorType: "upstream_error",
Code: "image_generation_unavailable",
Message: "Upstream did not execute image generation",
// Inferred from the model's own words, not from an upstream error frame:
// good enough to fail this turn over to another account, not evidence that
// this account's image tool is down for the next 30 minutes.
SynthesizedFromModelText: true,
}
}
@@ -1943,6 +1954,26 @@ const (
openAIImagesOAuthUnavailableReason = "openai_images_oauth_tool_unavailable"
)
// shouldCoolOpenAIImagesToolForError decides whether an image_generation_unavailable
// verdict is durable enough to park the account's image tool for
// openAIImagesOAuthUnavailableCooldown.
//
// Only an upstream error frame that names the condition qualifies. A verdict the
// gateway synthesized from the model's plain-text reply does not: it merely says
// this prompt produced words instead of an image, which is prompt-dependent and
// happens on healthy accounts. Writing a 30-minute account-level cooldown from it
// is doubly wrong because the very same error is classified retryable
// (IsOpenAIImagesRetryableUpstreamError: status >= 500) and drives
// newOpenAIAccountFailoverError — so one such reply walks the pool and cools every
// account the retry touches.
//
// This mirrors the rule the alpha/search path already states in words: a
// tool-endpoint failure "仍允许本次请求换号,但不修改任何账号状态"
// (see shouldApplyOpenAIAlphaSearchAccountErrorSideEffects).
func shouldCoolOpenAIImagesToolForError(upstreamErr *OpenAIImagesUpstreamError) bool {
return upstreamErr != nil && !upstreamErr.SynthesizedFromModelText
}
func (s *OpenAIGatewayService) coolOpenAIImagesOAuthTool(ctx context.Context, account *Account) {
if s == nil || s.accountRepo == nil || account == nil || account.Platform != PlatformOpenAI {
return
@@ -2038,7 +2069,9 @@ func (s *OpenAIGatewayService) handleOpenAIImagesOAuthResponseError(
responseBody := openAIImagesUpstreamErrorResponseBody(upstreamErr)
if upstreamErr.Code == "image_generation_unavailable" {
s.coolOpenAIImagesOAuthTool(ctx, account)
if shouldCoolOpenAIImagesToolForError(upstreamErr) {
s.coolOpenAIImagesOAuthTool(ctx, account)
}
if responseWritten {
return err
}
@@ -0,0 +1,177 @@
//go:build unit
package service
import (
"context"
"errors"
"net/http"
"net/http/httptest"
"testing"
"time"
"github.com/gin-gonic/gin"
"github.com/stretchr/testify/require"
)
// issue #6171:v0.1.181 起,/v1/images/generations 只要上游"回文字没回图",账号就被
// 写 30 分钟 openai:image_generation 模型级冷却。该判据是**请求级**的(这个 prompt
// 这一轮模型选择了说话),却被当成**账号级**能力失效;又因为同一个错误被判为
// 可重试(502)并驱动 failover,一次闲聊回复会沿着号池逐个把账号冷却掉。
// countingModelRateLimitRepo 记录 SetModelRateLimit 调用,用于断言"没写账号状态"。
type countingModelRateLimitRepo struct {
accountRepoStub
calls int
scopes []string
}
func (r *countingModelRateLimitRepo) SetModelRateLimit(_ context.Context, _ int64, scope string, _ time.Time, _ ...string) error {
r.calls++
r.scopes = append(r.scopes, scope)
return nil
}
func newImagesCooldownContext(t *testing.T) (*gin.Context, *httptest.ResponseRecorder) {
t.Helper()
gin.SetMode(gin.TestMode)
rec := httptest.NewRecorder()
c, _ := gin.CreateTestContext(rec)
c.Request = httptest.NewRequest(http.MethodPost, "/v1/images/generations", nil)
return c, rec
}
func imagesCooldownAccount() *Account {
return &Account{ID: 77, Platform: PlatformOpenAI, Type: AccountTypeOAuth, Name: "img-oauth"}
}
func TestShouldCoolOpenAIImagesToolForError(t *testing.T) {
cases := []struct {
name string
err *OpenAIImagesUpstreamError
want bool
}{
{
name: "nil_error",
err: nil,
want: false,
},
{
// 网关从模型文字里推断出来的判据:只说明这一轮没出图。
name: "synthesized_from_model_text",
err: &OpenAIImagesUpstreamError{
StatusCode: http.StatusBadGateway,
Code: "image_generation_unavailable",
SynthesizedFromModelText: true,
},
want: false,
},
{
// 上游自己在 error 帧里点名该状态:这才是账号级证据,保持冷却。
name: "structured_upstream_error_frame",
err: &OpenAIImagesUpstreamError{
StatusCode: http.StatusBadGateway,
Code: "image_generation_unavailable",
},
want: true,
},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
require.Equal(t, tc.want, shouldCoolOpenAIImagesToolForError(tc.err))
})
}
}
// 主复现:文字兜底判据不得写账号级冷却。
func TestHandleOpenAIImagesOAuthResponseError_TextFallbackDoesNotCoolAccount(t *testing.T) {
c, _ := newImagesCooldownContext(t)
repo := &countingModelRateLimitRepo{}
svc := &OpenAIGatewayService{accountRepo: repo}
account := imagesCooldownAccount()
upstreamErr := openAIImagesTextFallbackErrorForText("Here's a polished image prompt for your request.")
require.NotNil(t, upstreamErr)
require.Equal(t, "image_generation_unavailable", upstreamErr.Code)
err := svc.handleOpenAIImagesOAuthResponseError(
context.Background(), c, account, "gpt-image-2", "https://upstream.example/v1/responses",
&http.Response{StatusCode: http.StatusOK, Header: http.Header{}},
OpenAIImagesJSONKeepaliveAdjustedWrittenSize(c), upstreamErr,
)
require.Zero(t, repo.calls, "模型闲聊不构成账号级证据,不得写 30 分钟冷却")
// 换号行为必须原样保留:本 PR 只撤销账号状态写入,不动 failover。
var failover *UpstreamFailoverError
require.True(t, errors.As(err, &failover), "仍应触发换号,got %T", err)
}
// 对照不变式:上游 error 帧点名该状态时仍然冷却,否则等于把功能整个废掉。
func TestHandleOpenAIImagesOAuthResponseError_StructuredUnavailableStillCoolsAccount(t *testing.T) {
c, _ := newImagesCooldownContext(t)
repo := &countingModelRateLimitRepo{}
svc := &OpenAIGatewayService{accountRepo: repo}
account := imagesCooldownAccount()
upstreamErr := &OpenAIImagesUpstreamError{
StatusCode: http.StatusBadGateway,
ErrorType: "upstream_error",
Code: "image_generation_unavailable",
Message: "image generation tool is not available for this account",
}
_ = svc.handleOpenAIImagesOAuthResponseError(
context.Background(), c, account, "gpt-image-2", "https://upstream.example/v1/responses",
&http.Response{StatusCode: http.StatusOK, Header: http.Header{}},
OpenAIImagesJSONKeepaliveAdjustedWrittenSize(c), upstreamErr,
)
require.Equal(t, 1, repo.calls, "结构化上游证据仍须写冷却")
require.Equal(t, []string{openAIImageGenerationRateLimitKey}, repo.scopes)
}
// 标记必须打在文字兜底的两个入口上,且不影响违规拦截分支的判定。
func TestOpenAIImagesTextFallback_MarksSynthesizedVerdicts(t *testing.T) {
t.Run("plain_text_reply_is_synthesized", func(t *testing.T) {
err := openAIImagesTextFallbackErrorForText("Here's a polished image prompt for your request.")
require.NotNil(t, err)
require.True(t, err.SynthesizedFromModelText)
require.Equal(t, "image_generation_unavailable", err.Code)
require.Equal(t, http.StatusBadGateway, err.StatusCode)
})
t.Run("body_entrypoint_is_synthesized", func(t *testing.T) {
body := []byte("event: response.completed\n" +
`data: {"type":"response.completed","response":{"id":"r","status":"completed",` +
`"output":[{"type":"message","content":[{"type":"output_text","text":"I drafted a prompt for you."}]}]}}` +
"\n\n")
err := openAIImagesTextFallbackError(body)
require.NotNil(t, err)
require.True(t, err.SynthesizedFromModelText)
})
t.Run("content_policy_branch_unchanged", func(t *testing.T) {
err := openAIImagesTextFallbackErrorForText("Blocked by our content policy.")
require.NotNil(t, err)
require.Equal(t, "content_policy_violation", err.Code)
require.Equal(t, http.StatusBadRequest, err.StatusCode)
// 该分支本来就不走冷却(Code 不匹配),标记与否都不改变行为;
// 断言它没有被顺手打标,避免语义漂移。
require.False(t, err.SynthesizedFromModelText)
})
t.Run("empty_text_yields_no_error", func(t *testing.T) {
require.Nil(t, openAIImagesTextFallbackErrorForText(" "))
})
}
// 级联的前提条件:该错误确实是可重试的,所以会带着"已写冷却"的副作用换号。
// 这条用例把前提钉死,避免以后有人把 502 改成非重试后误以为本修复多余。
func TestOpenAIImagesTextFallback_RemainsRetryableAndThusCascades(t *testing.T) {
err := openAIImagesTextFallbackErrorForText("Here's a polished image prompt for your request.")
require.NotNil(t, err)
require.True(t, IsOpenAIImagesRetryableUpstreamError(err),
"文字兜底判据是可重试的——正因如此,写账号冷却会沿号池级联")
}
@@ -120,7 +120,11 @@ func TestOpenAIGatewayServiceForwardImages_ImageRateLimitReturnsFailoverAndCools
require.Equal(t, openAIImageGenerationRateLimitKey, repo.modelRateLimitCalls[0].scope)
}
func TestOpenAIGatewayServiceForwardImages_TextFallbackCoolsImageCapability(t *testing.T) {
// issue #6171:上游"回文字没回图"是**这一轮**的结果(模型选择了说话),不是账号能力
// 失效。它同时被判为可重试(502)并驱动 failover,若还写 30 分钟账号级冷却,一次闲聊
// 回复就会沿号池把每个被重试到的账号依次冷却掉。冷却仍保留给结构化上游证据,见
// TestOpenAIGatewayServiceForwardImages_StructuredUnavailableCoolsImageCapability。
func TestOpenAIGatewayServiceForwardImages_TextFallbackDoesNotCoolImageCapability(t *testing.T) {
gin.SetMode(gin.TestMode)
repo := &modelNotFoundAccountRepoStub{}
body := []byte(`{"model":"gpt-image-2","prompt":"draw a cat"}`)
@@ -154,7 +158,6 @@ func TestOpenAIGatewayServiceForwardImages_TextFallbackCoolsImageCapability(t *t
},
}
before := time.Now()
result, err := svc.ForwardImages(context.Background(), c, account, body, parsed, "")
require.Nil(t, result)
@@ -162,6 +165,56 @@ func TestOpenAIGatewayServiceForwardImages_TextFallbackCoolsImageCapability(t *t
var failoverErr *UpstreamFailoverError
require.ErrorAs(t, err, &failoverErr)
require.False(t, failoverErr.RetryableOnSameAccount)
// 换号行为不变:该判据仍足以放弃本账号重试这一次请求……
require.Equal(t, http.StatusBadGateway, failoverErr.StatusCode)
// ……但不再写任何账号级状态,否则重试会把冷却一路刷到整个号池。
require.Empty(t, repo.modelRateLimitCalls,
"模型回文字只说明这一轮没出图,不构成账号 30 分钟不可用的证据")
}
// 对照不变式:上游 error 帧点名 image_generation_unavailable 时仍写冷却,
// 保证 #6171 的修复没有把这项能力保护整个废掉。
func TestOpenAIGatewayServiceForwardImages_StructuredUnavailableCoolsImageCapability(t *testing.T) {
gin.SetMode(gin.TestMode)
repo := &modelNotFoundAccountRepoStub{}
body := []byte(`{"model":"gpt-image-2","prompt":"draw a cat"}`)
upstreamSSE := "data: {\"type\":\"response.failed\",\"response\":{\"id\":\"r\",\"error\":" +
"{\"type\":\"upstream_error\",\"code\":\"image_generation_unavailable\"," +
"\"message\":\"image generation tool is not available for this account\"}}}\n\n"
req := httptest.NewRequest(http.MethodPost, "/v1/images/generations", bytes.NewReader(body))
req.Header.Set("Content-Type", "application/json")
rec := httptest.NewRecorder()
c, _ := gin.CreateTestContext(rec)
c.Request = req
svc := &OpenAIGatewayService{
accountRepo: repo,
httpUpstream: &httpUpstreamRecorder{
resp: &http.Response{
StatusCode: http.StatusOK,
Header: http.Header{"Content-Type": []string{"text/event-stream"}},
Body: io.NopCloser(strings.NewReader(upstreamSSE)),
},
},
}
parsed, err := svc.ParseOpenAIImagesRequest(c, body)
require.NoError(t, err)
account := &Account{
ID: 206,
Name: "openai-oauth",
Platform: PlatformOpenAI,
Type: AccountTypeOAuth,
Credentials: map[string]any{
"access_token": "token-123",
},
}
before := time.Now()
result, err := svc.ForwardImages(context.Background(), c, account, body, parsed, "")
require.Nil(t, result)
require.Error(t, err)
require.Len(t, repo.modelRateLimitCalls, 1)
call := repo.modelRateLimitCalls[0]
require.Equal(t, account.ID, call.accountID)