fix(openai): preserve Spark quota reset semantics

This commit is contained in:
shaw
2026-08-29 11:08:01 +08:00
parent 5d9c7abed5
commit 3c22e78af0
3 changed files with 65 additions and 8 deletions
@@ -123,6 +123,59 @@ func TestOpenAI429FastPath_SparkQuotaOnlyBlocksSparkModel(t *testing.T) {
require.Greater(t, time.Until(repo.lastModelRateLimitedUntil), 6*24*time.Hour)
}
func TestOpenAI429FastPath_SparkTransient429UsesShortFallback(t *testing.T) {
repo := &oauth429RateLimitRepo{}
rateLimits := NewRateLimitService(repo, nil, &config.Config{}, nil, nil)
svc := &OpenAIGatewayService{rateLimitService: rateLimits}
rateLimits.SetAccountRuntimeBlocker(svc)
account := &Account{ID: 428, Platform: PlatformOpenAI, Type: AccountTypeOAuth}
headers := http.Header{}
headers.Set("x-codex-primary-used-percent", "37")
headers.Set("x-codex-primary-reset-after-seconds", "604800")
headers.Set("x-codex-primary-window-minutes", "10080")
headers.Set("x-codex-secondary-used-percent", "20")
headers.Set("x-codex-secondary-reset-after-seconds", "3600")
headers.Set("x-codex-secondary-window-minutes", "300")
shouldDisable := svc.handleOpenAIAccountUpstreamError(
context.Background(), account, http.StatusTooManyRequests, headers,
[]byte(`{"error":{"type":"rate_limit_error","code":"rate_limit_exceeded"}}`),
"gpt-5.3-codex-spark",
)
require.False(t, shouldDisable)
require.Equal(t, 1, repo.setModelRateLimitCalls)
require.Less(t, time.Until(repo.lastModelRateLimitedUntil), time.Minute)
require.Greater(t, time.Until(repo.lastModelRateLimitedUntil), time.Second)
}
func TestOpenAIStream429_SparkQuotaUsesQuotaHeaders(t *testing.T) {
repo := &oauth429RateLimitRepo{}
rateLimits := NewRateLimitService(repo, nil, &config.Config{}, nil, nil)
svc := &OpenAIGatewayService{rateLimitService: rateLimits}
rateLimits.SetAccountRuntimeBlocker(svc)
account := &Account{ID: 429, Platform: PlatformOpenAI, Type: AccountTypeOAuth}
headers := http.Header{}
headers.Set("x-codex-primary-used-percent", "100")
headers.Set("x-codex-primary-reset-after-seconds", "604800")
headers.Set("x-codex-primary-window-minutes", "10080")
headers.Set("x-codex-secondary-used-percent", "20")
headers.Set("x-codex-secondary-reset-after-seconds", "3600")
headers.Set("x-codex-secondary-window-minutes", "300")
payload := []byte(`{"type":"error","error":{"type":"rate_limit_error","code":"rate_limit_exceeded"}}`)
status, shouldDisable := svc.handleOpenAIStreamTerminalAccountSideEffects(
nil, account, payload, "quota exhausted", headers, "gpt-5.3-codex-spark",
)
require.Equal(t, http.StatusTooManyRequests, status)
require.False(t, shouldDisable)
require.Equal(t, 1, repo.setModelRateLimitCalls)
require.Equal(t, "gpt-5.3-codex-spark", repo.lastModelRateLimitKey)
require.Greater(t, time.Until(repo.lastModelRateLimitedUntil), 6*24*time.Hour)
require.False(t, svc.isOpenAIAccountRuntimeBlocked(account))
}
func TestOpenAI429FastPath_SparkShadowQuotaStaysModelScoped(t *testing.T) {
repo := &oauth429RateLimitRepo{}
rateLimits := NewRateLimitService(repo, nil, &config.Config{}, nil, nil)
@@ -1572,17 +1572,16 @@ func (s *OpenAIGatewayService) handleOpenAIStreamTerminalAccountSideEffects(
if c != nil && c.Request != nil {
ctx = c.Request.Context()
}
accountHeaders := headers
if statusCode == http.StatusTooManyRequests {
// The enclosing HTTP response succeeded. Its quota snapshot describes
// normal account state and must not become the reset for a semantic 429
// carried by a stream terminal event.
accountHeaders = nil
}
model := firstNonEmpty(canonicalModel...)
if model == "" {
model = firstNonEmpty(gjson.GetBytes(payload, "model").String(), gjson.GetBytes(payload, "response.model").String())
}
accountHeaders := headers
if statusCode == http.StatusTooManyRequests && !(isCodexSparkModel(model) && isOpenAIOAuthAccount(account)) {
// 普通模型的流式 429 不能继承外层 HTTP 200 的全局 quota 快照;
// 只有 OAuth/SetupToken 的 Spark 配额 429 才需要保留 headers 读取明确的 5h/7d reset。
accountHeaders = nil
}
return statusCode, s.handleOpenAIAccountUpstreamError(ctx, account, statusCode, accountHeaders, payload, model)
default:
return statusCode, false
@@ -2208,7 +2208,12 @@ func (s *RateLimitService) HandleOpenAICodexSparkRateLimit(ctx context.Context,
return false
}
now := time.Now()
_, resetAt := classifyOpenAIOAuth429(headers, responseBody)
disposition, resetAt := classifyOpenAIOAuth429(headers, responseBody)
// Spark 只有明确耗尽 5h/7d 窗口时才能使用上游长 reset;普通瞬时 429
// 即使携带全局 reset 头,也只能使用短时回避,避免错误冷却数天。
if disposition != openAIOAuth429Quota5h && disposition != openAIOAuth429Quota7d {
resetAt = nil
}
if resetAt == nil || !resetAt.After(now) {
cooldown, ok := s.get429FallbackCooldown(ctx, account)
if !ok || cooldown <= 0 {