diff --git a/backend/internal/service/grok_media.go b/backend/internal/service/grok_media.go index 7c4a12cd0b..3d7ad1fe31 100644 --- a/backend/internal/service/grok_media.go +++ b/backend/internal/service/grok_media.go @@ -1288,7 +1288,7 @@ func (s *OpenAIGatewayService) handleGrokMediaErrorResponse( Detail: upstreamDetail, }) if kind == "failover" { - retryable, retryDelay, retryDeadline := grokSameAccountRetryMetadata(account, resp.StatusCode, body) + retryable, retryDelay, retryDeadline, retryMax := grokSameAccountRetryMetadata(account, resp.StatusCode, body) return nil, &UpstreamFailoverError{ StatusCode: resp.StatusCode, ResponseBody: body, @@ -1297,6 +1297,7 @@ func (s *OpenAIGatewayService) handleGrokMediaErrorResponse( RequestScopedTransient: retryable && resp.StatusCode == http.StatusTooManyRequests, SameAccountRetryDelay: retryDelay, SameAccountRetryDeadline: retryDeadline, + SameAccountRetryMax: retryMax, } } diff --git a/backend/internal/service/grok_upstream_failure.go b/backend/internal/service/grok_upstream_failure.go index 3d6aa05c6e..53d9d5500a 100644 --- a/backend/internal/service/grok_upstream_failure.go +++ b/backend/internal/service/grok_upstream_failure.go @@ -119,7 +119,7 @@ func classifyGrokUpstreamFailure(statusCode int, responseBody []byte, requestedM // deployment while the same request is valid on another account. Treat // these precise decoder/content-shape failures as account compatibility // errors, rather than durable account health failures. - if isGrokCompatibilityError(low, code) { + if isGrokCompatibilityError(statusCode, low, code) { return GrokUpstreamFailureDecision{ Class: GrokFailureCompatibility, Model: model, @@ -406,15 +406,19 @@ func grokRetryableOnSameAccount(account *Account, statusCode int, responseBody [ return account.IsPoolMode() && account.IsPoolModeRetryableStatus(statusCode) } -func grokSameAccountRetryMetadata(account *Account, statusCode int, responseBody []byte) (bool, time.Duration, time.Time) { +func grokSameAccountRetryMetadata(account *Account, statusCode int, responseBody []byte) (bool, time.Duration, time.Time, int) { if !grokRetryableOnSameAccount(account, statusCode, responseBody) { - return false, 0, time.Time{} + return false, 0, time.Time{}, 0 } decision := classifyGrokUpstreamFailure(statusCode, responseBody, "") if decision.Class != GrokFailureModelCapacity { - return true, 0, time.Time{} + return true, 0, time.Time{}, 0 } - return true, 500 * time.Millisecond, time.Now().Add(30 * time.Second) + // The error is reconstructed after every upstream attempt, so a deadline + // stored on the error cannot provide a request-wide window. Cap capacity + // retries explicitly to one replay; this remains effective even when the + // first attempt itself takes longer than the nominal 30-second window. + return true, 500 * time.Millisecond, time.Now().Add(30 * time.Second), 1 } // shouldMarkGrokTeamModelRateLimit controls the process-local sibling-account @@ -430,7 +434,10 @@ func shouldMarkGrokTeamModelRateLimit(statusCode int, responseBody []byte) bool return statusCode == http.StatusTooManyRequests || decision.Class == GrokFailureFreeUsage } -func isGrokCompatibilityError(low, code string) bool { +func isGrokCompatibilityError(statusCode int, low, code string) bool { + if statusCode != http.StatusBadRequest && statusCode != http.StatusUnprocessableEntity { + return false + } combined := strings.ToLower(strings.TrimSpace(low + " " + code)) // Compaction blobs are account/session-bound and frequently fail with 400 // or 422 after a reconnect. Also cover xAI's JSON decoder shape errors. diff --git a/backend/internal/service/grok_upstream_failure_test.go b/backend/internal/service/grok_upstream_failure_test.go index d6e6ec7ace..4f6514cde1 100644 --- a/backend/internal/service/grok_upstream_failure_test.go +++ b/backend/internal/service/grok_upstream_failure_test.go @@ -110,17 +110,19 @@ func TestShouldMarkGrokTeamModelRateLimit_ExcludesCapacity(t *testing.T) { func TestGrokSameAccountRetryMetadata_CapacityDeadline(t *testing.T) { account := &Account{ID: 9107, Platform: PlatformGrok, Type: AccountTypeOAuth} - retryable, delay, deadline := grokSameAccountRetryMetadata(account, http.StatusTooManyRequests, + retryable, delay, deadline, retryMax := grokSameAccountRetryMetadata(account, http.StatusTooManyRequests, []byte(`{"error":{"message":"model capacity exceeded"}}`)) require.True(t, retryable) require.Equal(t, 500*time.Millisecond, delay) require.WithinDuration(t, time.Now().Add(30*time.Second), deadline, 2*time.Second) + require.Equal(t, 1, retryMax) - retryable, delay, deadline = grokSameAccountRetryMetadata(account, http.StatusTooManyRequests, + retryable, delay, deadline, retryMax = grokSameAccountRetryMetadata(account, http.StatusTooManyRequests, []byte(`{"error":{"message":"rate limit exceeded"}}`)) require.False(t, retryable) require.Zero(t, delay) require.True(t, deadline.IsZero()) + require.Zero(t, retryMax) } func TestClassifyGrokUpstreamFailure_ValidationNoCool(t *testing.T) { @@ -151,6 +153,15 @@ func TestClassifyGrokUpstreamFailure_CompatibilityDoesNotCooldown(t *testing.T) } } +func TestClassifyGrokUpstreamFailure_CompatibilityRequiresClientError(t *testing.T) { + body := []byte(`{"error":{"message":"upstream failed while handling the compaction blob"}}`) + for _, status := range []int{http.StatusBadGateway, http.StatusInternalServerError} { + d := classifyGrokUpstreamFailure(status, body, "grok-4.6") + require.NotEqual(t, GrokFailureCompatibility, d.Class) + require.True(t, d.ShouldCooldown) + } +} + func TestClassifyGrokUpstreamFailure_GenericShapeErrorDoesNotFailover(t *testing.T) { d := classifyGrokUpstreamFailure(http.StatusBadRequest, []byte(`{"error":{"message":"data did not match any variant of the untagged enum content"}}`), "grok-4.6") diff --git a/backend/internal/service/openai_gateway_chat_completions_raw.go b/backend/internal/service/openai_gateway_chat_completions_raw.go index 00a7ece52c..1ef181b21b 100644 --- a/backend/internal/service/openai_gateway_chat_completions_raw.go +++ b/backend/internal/service/openai_gateway_chat_completions_raw.go @@ -200,7 +200,7 @@ func (s *OpenAIGatewayService) forwardAsRawChatCompletions( }) s.handleGrokAccountUpstreamError(withGrokTeamRateLimitModel(ctx, upstreamModel), account, resp.StatusCode, resp.Header, respBody) if s.shouldFailoverGrokUpstreamError(resp.StatusCode, respBody) { - retryable, retryDelay, retryDeadline := grokSameAccountRetryMetadata(account, resp.StatusCode, respBody) + retryable, retryDelay, retryDeadline, retryMax := grokSameAccountRetryMetadata(account, resp.StatusCode, respBody) return nil, &UpstreamFailoverError{ StatusCode: resp.StatusCode, ResponseBody: respBody, @@ -209,6 +209,7 @@ func (s *OpenAIGatewayService) forwardAsRawChatCompletions( RequestScopedTransient: retryable && resp.StatusCode == http.StatusTooManyRequests, SameAccountRetryDelay: retryDelay, SameAccountRetryDeadline: retryDeadline, + SameAccountRetryMax: retryMax, } } return s.handleChatCompletionsErrorResponse(resp, c, account, billingModel) diff --git a/backend/internal/service/openai_gateway_grok.go b/backend/internal/service/openai_gateway_grok.go index 6ad03cc253..9a16c6c7a7 100644 --- a/backend/internal/service/openai_gateway_grok.go +++ b/backend/internal/service/openai_gateway_grok.go @@ -173,7 +173,7 @@ func (s *OpenAIGatewayService) forwardGrokResponses( markGrokTeamModelRateLimit(account, upstreamModel, resolveGrokTeamRateLimitUntil(time.Now().Add(grokTeamRateLimitDefaultTTL), time.Now())) } if s.shouldFailoverGrokUpstreamError(resp.StatusCode, respBody) { - retryable, retryDelay, retryDeadline := grokSameAccountRetryMetadata(account, resp.StatusCode, respBody) + retryable, retryDelay, retryDeadline, retryMax := grokSameAccountRetryMetadata(account, resp.StatusCode, respBody) return nil, &UpstreamFailoverError{ StatusCode: resp.StatusCode, ResponseBody: respBody, @@ -182,6 +182,7 @@ func (s *OpenAIGatewayService) forwardGrokResponses( RequestScopedTransient: retryable && resp.StatusCode == http.StatusTooManyRequests, SameAccountRetryDelay: retryDelay, SameAccountRetryDeadline: retryDeadline, + SameAccountRetryMax: retryMax, } } return s.handleErrorResponse(ctx, resp, c, account, patchedBody, upstreamModel) @@ -1186,7 +1187,7 @@ func (s *OpenAIGatewayService) describeGrokComposerImage( }) s.handleGrokAccountUpstreamError(withGrokTeamRateLimitModel(ctx, grokComposerImageBridgeVisionModel), account, resp.StatusCode, resp.Header, respBody) if s.shouldFailoverGrokUpstreamError(resp.StatusCode, respBody) { - retryable, retryDelay, retryDeadline := grokSameAccountRetryMetadata(account, resp.StatusCode, respBody) + retryable, retryDelay, retryDeadline, retryMax := grokSameAccountRetryMetadata(account, resp.StatusCode, respBody) return "", OpenAIUsage{}, &UpstreamFailoverError{ StatusCode: resp.StatusCode, ResponseBody: respBody, @@ -1195,6 +1196,7 @@ func (s *OpenAIGatewayService) describeGrokComposerImage( RequestScopedTransient: retryable && resp.StatusCode == http.StatusTooManyRequests, SameAccountRetryDelay: retryDelay, SameAccountRetryDeadline: retryDeadline, + SameAccountRetryMax: retryMax, } } return "", OpenAIUsage{}, fmt.Errorf("grok composer image bridge upstream error: %s", upstreamMsg) diff --git a/backend/internal/service/openai_gateway_grok_chat_bridge.go b/backend/internal/service/openai_gateway_grok_chat_bridge.go index 84e2c95e72..f04ab5ae43 100644 --- a/backend/internal/service/openai_gateway_grok_chat_bridge.go +++ b/backend/internal/service/openai_gateway_grok_chat_bridge.go @@ -633,7 +633,7 @@ func (s *OpenAIGatewayService) forwardGrokChatCompletionsViaResponses( }) s.handleGrokAccountUpstreamError(withGrokTeamRateLimitModel(ctx, upstreamModel), account, resp.StatusCode, resp.Header, respBody) if s.shouldFailoverGrokUpstreamError(resp.StatusCode, respBody) { - retryable, retryDelay, retryDeadline := grokSameAccountRetryMetadata(account, resp.StatusCode, respBody) + retryable, retryDelay, retryDeadline, retryMax := grokSameAccountRetryMetadata(account, resp.StatusCode, respBody) return nil, &UpstreamFailoverError{ StatusCode: resp.StatusCode, ResponseBody: respBody, @@ -642,6 +642,7 @@ func (s *OpenAIGatewayService) forwardGrokChatCompletionsViaResponses( RequestScopedTransient: retryable && resp.StatusCode == http.StatusTooManyRequests, SameAccountRetryDelay: retryDelay, SameAccountRetryDeadline: retryDeadline, + SameAccountRetryMax: retryMax, } } return s.handleChatCompletionsErrorResponse(resp, c, account, billingModel)