fix(openai): preserve pool-mode same-account retries

Avoid recording the generic account-model transient cooldown for pool-mode statuses explicitly configured for same-account retry. This lets the bounded request-local retry budget complete while preserving cooldowns for other statuses and non-pool accounts.
This commit is contained in:
Cynicismcart
2026-07-25 05:31:31 +08:00
parent cb24522dd5
commit 521db6869e
2 changed files with 89 additions and 1 deletions
@@ -91,7 +91,12 @@ func (s *OpenAIGatewayService) handleOpenAIAccountUpstreamError(ctx context.Cont
if shouldDisable && !modelTempMatched {
s.BlockAccountScheduling(account, time.Time{}, "upstream_disable")
}
if !shouldDisable && account.Platform == PlatformOpenAI && account.Type == AccountTypeAPIKey && shouldCooldownOpenAITransientUpstreamError(statusCode, responseBody) {
// Pool-mode retryable upstream errors are already bounded by the request-local
// same-account retry budget. Recording the generic account+model transient
// cooldown here would block the next approved retry before that budget is used.
poolModeRetryable := account.IsPoolMode() && account.IsPoolModeRetryableStatus(statusCode)
if !shouldDisable && account.Platform == PlatformOpenAI && account.Type == AccountTypeAPIKey &&
shouldCooldownOpenAITransientUpstreamError(statusCode, responseBody) && !poolModeRetryable {
model := ""
if len(canonicalModel) > 0 {
model = canonicalModel[0]
@@ -142,6 +142,89 @@ func TestOpenAIPoolModeTempRule_StopsSameAccountRetryAndIsolatesBlockToModel(t *
require.False(t, gateway.isOpenAIAccountRequestRuntimeBlocked(account, "gpt-5.5"))
}
func TestOpenAIPoolModeRetryable5xx_DoesNotCreateModelTransientBlock(t *testing.T) {
repo := &errorPolicyRepoStub{}
rateLimitService := NewRateLimitService(repo, nil, &config.Config{}, nil, nil)
gateway := &OpenAIGatewayService{rateLimitService: rateLimitService}
account := &Account{
ID: 47,
Platform: PlatformOpenAI,
Type: AccountTypeAPIKey,
Credentials: map[string]any{
"pool_mode": true,
"pool_mode_retry_status_codes": []any{float64(524)},
},
}
for i := 0; i < 2; i++ {
shouldDisable := gateway.handleOpenAIAccountUpstreamError(
context.Background(),
account,
524,
http.Header{},
[]byte(`{"error":{"message":"upstream timeout"}}`),
"gpt-5.4",
)
require.False(t, shouldDisable)
}
require.False(t, gateway.isOpenAIAccountRequestRuntimeBlocked(account, "gpt-5.4"))
}
func TestOpenAIPoolModeNonRetryable5xx_StillCreatesModelTransientBlock(t *testing.T) {
repo := &errorPolicyRepoStub{}
rateLimitService := NewRateLimitService(repo, nil, &config.Config{}, nil, nil)
gateway := &OpenAIGatewayService{rateLimitService: rateLimitService}
account := &Account{
ID: 48,
Platform: PlatformOpenAI,
Type: AccountTypeAPIKey,
Credentials: map[string]any{
"pool_mode": true,
"pool_mode_retry_status_codes": []any{float64(http.StatusGatewayTimeout)},
},
}
for i := 0; i < 2; i++ {
shouldDisable := gateway.handleOpenAIAccountUpstreamError(
context.Background(),
account,
http.StatusServiceUnavailable,
http.Header{},
[]byte(`{"error":{"message":"upstream unavailable"}}`),
"gpt-5.4",
)
require.False(t, shouldDisable)
}
require.True(t, gateway.isOpenAIAccountRequestRuntimeBlocked(account, "gpt-5.4"))
}
func TestOpenAINonPoolAPIKey5xx_StillCreatesModelTransientBlock(t *testing.T) {
repo := &errorPolicyRepoStub{}
rateLimitService := NewRateLimitService(repo, nil, &config.Config{}, nil, nil)
gateway := &OpenAIGatewayService{rateLimitService: rateLimitService}
account := &Account{
ID: 49,
Platform: PlatformOpenAI,
Type: AccountTypeAPIKey,
}
for i := 0; i < 2; i++ {
shouldDisable := gateway.handleOpenAIAccountUpstreamError(
context.Background(),
account,
http.StatusGatewayTimeout,
http.Header{},
[]byte(`{"error":{"message":"upstream timeout"}}`),
"gpt-5.4",
)
require.False(t, shouldDisable)
}
require.True(t, gateway.isOpenAIAccountRequestRuntimeBlocked(account, "gpt-5.4"))
}
func TestOpenAIModelNotFound_DoesNotRuntimeBlockWholeAccount(t *testing.T) {
repo := &modelNotFoundAccountRepoStub{}
svc := &OpenAIGatewayService{