fix(ratelimit): 守卫按端点来源门控,并与冷却键对齐模型口径

上一版守卫只看模型类型,不区分请求从哪个端点进来。OAuth 账号的 /v1/images/*
上游同样是 Codex Responses(openai_images_responses.go → handleOpenAIImagesErrorResponse
→ handleOpenAIAccountUpstreamError → HandleUpstreamModelNotFound),所以专用生图
端点也会命中 plan-gated 分支。账号确实不具备生图能力时跳过冷却,会让调度层失去
唯一的刹车:每个请求都完整走一遍号池,对上游形成无上界的 400 放大。

改动:
- 新增 ctxkey.OpenAIImagesEndpoint 与 WithOpenAIImagesEndpoint /
  OpenAIImagesEndpointFromContext,在 handler/openai_images.go 入口置位;
  与 OpenAIImageGenerationIntent 区分——后者在 /v1/responses 带图片模型时也会置位。
- 守卫下移到 modelKey 计算之后,抽成 shouldSkipCodexPlanGatedImageModelCooldown,
  仅在 plan-gated 分支、且非 /v1/images/* 入站时生效。
- 同时判断 requestedModel 与最终 modelKey:冷却键走 account.GetMappedModel,
  账号可以把文本别名映射到 gpt-image-*,只判请求模型会漏掉这种形态。
This commit is contained in:
li
2026-08-07 20:49:39 +08:00
parent b5d9fd21b0
commit 02fbcbe3ad
5 changed files with 140 additions and 9 deletions
+1 -1
View File
@@ -144,7 +144,7 @@ func (h *OpenAIGatewayHandler) Images(c *gin.Context) {
}
sessionHash := h.gatewayService.GenerateExplicitSessionHash(c, body)
requestCtx := service.WithOpenAIImageGenerationIntent(c.Request.Context())
requestCtx := service.WithOpenAIImagesEndpoint(service.WithOpenAIImageGenerationIntent(c.Request.Context()))
maxAccountSwitches := h.maxAccountSwitches
switchCount := 0
+6
View File
@@ -50,6 +50,12 @@ const (
// OpenAIImageGenerationIntent 标识 OpenAI 请求会触发生图能力(用于图片能力维度限流)
OpenAIImageGenerationIntent Key = "ctx_openai_image_generation_intent"
// OpenAIImagesEndpoint 标识请求是从 /v1/images/* 入站的。
// 与 OpenAIImageGenerationIntent 的区别:后者只表示"这次请求会生图",
// /v1/responses 带图片模型时也会置位;本 key 只在专用生图端点置位,
// 用于区分"用错端点"与"端点用对了但账号没能力"。
OpenAIImagesEndpoint Key = "ctx_openai_images_endpoint"
// Group 认证后的分组信息,由 API Key 认证中间件设置
Group Key = "ctx_group"
@@ -120,6 +120,23 @@ func OpenAIImageGenerationIntentFromContext(ctx context.Context) bool {
return ok && enabled
}
// WithOpenAIImagesEndpoint 标记请求从 /v1/images/* 专用生图端点入站。
func WithOpenAIImagesEndpoint(ctx context.Context) context.Context {
if ctx == nil {
ctx = context.Background()
}
return context.WithValue(ctx, ctxkey.OpenAIImagesEndpoint, true)
}
// OpenAIImagesEndpointFromContext 报告请求是否来自 /v1/images/*。
func OpenAIImagesEndpointFromContext(ctx context.Context) bool {
if ctx == nil {
return false
}
enabled, ok := ctx.Value(ctxkey.OpenAIImagesEndpoint).(bool)
return ok && enabled
}
func resolveFinalAntigravityModelKey(ctx context.Context, account *Account, requestedModel string) string {
modelKey := mapAntigravityModel(account, requestedModel)
if modelKey == "" {
+26 -8
View File
@@ -2047,14 +2047,6 @@ func (s *RateLimitService) HandleUpstreamModelNotFound(ctx context.Context, acco
case isUpstreamModelNotFoundError(statusCode, responseBody):
cooldown, reason = upstreamModelNotFoundCooldown, upstreamModelNotFoundReason
case isOpenAIOAuthAccount(account) && isOpenAICodexPlanGatedModelError(statusCode, responseBody):
// Image models rejected by the Codex text endpoints are a deterministic
// endpoint mismatch, not an account capability gap: the same account still
// serves them over /v1/images/*. Cooling down the model here would take the
// whole pool offline for correct image requests, so fail the attempt over
// without recording a per-model cooldown.
if IsGPTImageGenerationModel(requestedModel) {
return true
}
cooldown, reason = upstreamCodexPlanGatedModelCooldown, upstreamCodexPlanGatedModelReason
default:
return false
@@ -2063,6 +2055,9 @@ func (s *RateLimitService) HandleUpstreamModelNotFound(ctx context.Context, acco
if modelKey == "" {
return false
}
if shouldSkipCodexPlanGatedImageModelCooldown(ctx, reason, requestedModel, modelKey) {
return true
}
resetAt := time.Now().Add(cooldown)
if err := s.accountRepo.SetModelRateLimit(ctx, account.ID, modelKey, resetAt, reason); err != nil {
slog.Warn("upstream_model_not_found_set_model_rate_limit_failed", "account_id", account.ID, "model", modelKey, "reason", reason, "error", err)
@@ -2072,6 +2067,29 @@ func (s *RateLimitService) HandleUpstreamModelNotFound(ctx context.Context, acco
return true
}
// shouldSkipCodexPlanGatedImageModelCooldown 判断这次 Codex plan-gated 400 是否
// 属于"图片模型被文本端点拒绝"。
//
// 这类错误是确定性的端点错配,不是账号能力缺失:同一账号通过 /v1/images/* 依然
// 能出图。在这里写 per-model 冷却,会让一次用错端点的请求把整个号池对正确的生图
// 端点下线(#4828)。
//
// 但请求本身就从 /v1/images/* 入站时不适用——那种情况下被拒说明账号确实不具备
// 该模型能力,冷却是必要的刹车:没有它,每个生图请求都会完整走一遍号池,对上游
// 形成无上界的 400 放大。
//
// 请求模型与最终冷却键都要判:冷却键走的是 account.GetMappedModel,账号可能把
// 文本别名映射到 gpt-image-*,只判请求模型会漏掉这种形态。
func shouldSkipCodexPlanGatedImageModelCooldown(ctx context.Context, reason, requestedModel, modelKey string) bool {
if reason != upstreamCodexPlanGatedModelReason {
return false
}
if OpenAIImagesEndpointFromContext(ctx) {
return false
}
return IsGPTImageGenerationModel(requestedModel) || IsGPTImageGenerationModel(modelKey)
}
func modelRateLimitKeyForUpstreamModelNotFound(ctx context.Context, account *Account, requestedModel string) string {
modelKey := strings.TrimSpace(requestedModel)
if account == nil || modelKey == "" {
@@ -435,3 +435,93 @@ func openAICodexPlanGatedOAuthAccount() *Account {
Credentials: map[string]any{},
}
}
// 请求本身就走 /v1/images/* 时必须保留冷却。
//
// OAuth 账号的 /v1/images/* 上游同样是 Codex Responses(openai_images_responses.go
// → handleOpenAIImagesErrorResponse → handleOpenAIAccountUpstreamError →
// HandleUpstreamModelNotFound),所以这条路径也会命中 plan-gated 分支。账号确实
// 不具备生图能力时,冷却是唯一的刹车:调度层靠 model_rate_limits 跳过该账号后
// 快速 503;一旦跳过冷却,每个请求都会完整走一遍号池,对上游形成无上界的 400 放大。
func TestRateLimitService_HandleUpstreamError_CodexPlanGatedImageModelKeepsCooldownOnImagesEndpoint(t *testing.T) {
repo := &modelNotFoundAccountRepoStub{}
svc := &RateLimitService{accountRepo: repo}
account := openAICodexPlanGatedOAuthAccount()
handled := svc.HandleUpstreamError(
WithOpenAIImagesEndpoint(context.Background()),
account,
http.StatusBadRequest,
http.Header{},
[]byte(`{"detail":"The 'gpt-image-2' model is not supported when using Codex with a ChatGPT account."}`),
"gpt-image-2",
)
require.True(t, handled)
require.Len(t, repo.modelRateLimitCalls, 1,
"/v1/images/* 上的 plan-gated 拒绝是真实的能力缺失,必须保留冷却刹车")
require.Equal(t, "gpt-image-2", repo.modelRateLimitCalls[0].scope)
require.Equal(t, upstreamCodexPlanGatedModelReason, repo.modelRateLimitCalls[0].reason)
}
// 仅 WithOpenAIImageGenerationIntent(/v1/responses 因模型名自动置位)不算专用生图
// 端点,仍按"用错端点"处理。
func TestRateLimitService_HandleUpstreamError_CodexPlanGatedImageModelSkipsCooldownOnIntentOnly(t *testing.T) {
repo := &modelNotFoundAccountRepoStub{}
svc := &RateLimitService{accountRepo: repo}
account := openAICodexPlanGatedOAuthAccount()
handled := svc.HandleUpstreamError(
WithOpenAIImageGenerationIntent(context.Background()),
account,
http.StatusBadRequest,
http.Header{},
[]byte(`{"detail":"The 'gpt-image-2' model is not supported when using Codex with a ChatGPT account."}`),
"gpt-image-2",
)
require.True(t, handled)
require.Empty(t, repo.modelRateLimitCalls)
}
// 守卫口径必须与冷却键一致:冷却键走 account.GetMappedModel,账号可以把文本别名
// 映射到 gpt-image-*,只判请求模型会漏掉这种形态,原 bug 原样复现。
func TestRateLimitService_HandleUpstreamError_CodexPlanGatedImageModelSkipsCooldownViaModelMapping(t *testing.T) {
repo := &modelNotFoundAccountRepoStub{}
svc := &RateLimitService{accountRepo: repo}
account := openAICodexPlanGatedOAuthAccount()
account.Credentials["model_mapping"] = map[string]any{"my-draw-alias": "gpt-image-2"}
handled := svc.HandleUpstreamError(
context.Background(),
account,
http.StatusBadRequest,
http.Header{},
[]byte(`{"detail":"The 'gpt-image-2' model is not supported when using Codex with a ChatGPT account."}`),
"my-draw-alias",
)
require.True(t, handled)
require.Empty(t, repo.modelRateLimitCalls,
"映射后的上游模型是图片模型,冷却键会写到 gpt-image-2 上,守卫必须一并识别")
}
// 404 model-not-found 分支不受守卫影响:即使是图片模型也照常冷却。
func TestRateLimitService_HandleUpstreamError_ModelNotFoundImageModelStillCoolsDown(t *testing.T) {
repo := &modelNotFoundAccountRepoStub{}
svc := &RateLimitService{accountRepo: repo}
account := openAICodexPlanGatedOAuthAccount()
handled := svc.HandleUpstreamError(
context.Background(),
account,
http.StatusNotFound,
http.Header{},
[]byte(`{"error":{"message":"The model 'gpt-image-2' does not exist","code":"model_not_found"}}`),
"gpt-image-2",
)
require.True(t, handled)
require.Len(t, repo.modelRateLimitCalls, 1, "守卫只作用于 codex plan-gated 分支")
require.Equal(t, upstreamModelNotFoundReason, repo.modelRateLimitCalls[0].reason)
}