From 698547418fc8b8fc5f597fd34516e7026e706d82 Mon Sep 17 00:00:00 2001 From: Zven Date: Fri, 31 Jul 2026 21:20:32 +0800 Subject: [PATCH] fix(pricing): keep Auto-review rates evidence-based --- backend/internal/service/pricing_service_test.go | 15 +++++++++++---- .../model_prices_and_context_window.json | 11 ----------- 2 files changed, 11 insertions(+), 15 deletions(-) diff --git a/backend/internal/service/pricing_service_test.go b/backend/internal/service/pricing_service_test.go index 80641bfb02..1a933aec10 100644 --- a/backend/internal/service/pricing_service_test.go +++ b/backend/internal/service/pricing_service_test.go @@ -531,7 +531,7 @@ func TestDefaultPricingIncludesGemini36FlashRates(t *testing.T) { } } -func TestDefaultPricingUsesReducedCodexAutoReviewRates(t *testing.T) { +func TestDefaultPricingUsesCurrentCodexAutoReviewBaseRates(t *testing.T) { data, err := os.ReadFile(filepath.Join("..", "..", "resources", "model-pricing", "model_prices_and_context_window.json")) require.NoError(t, err) @@ -543,11 +543,18 @@ func TestDefaultPricingUsesReducedCodexAutoReviewRates(t *testing.T) { got := svc.GetModelPricing("codex-auto-review") require.NotNil(t, got) require.InDelta(t, 0.2e-6, got.InputCostPerToken, 1e-12) - require.InDelta(t, 0.4e-6, got.InputCostPerTokenPriority, 1e-12) require.InDelta(t, 1.2e-6, got.OutputCostPerToken, 1e-12) - require.InDelta(t, 2.4e-6, got.OutputCostPerTokenPriority, 1e-12) require.InDelta(t, 0.02e-6, got.CacheReadInputTokenCost, 1e-12) - require.InDelta(t, 0.04e-6, got.CacheReadInputTokenCostPriority, 1e-12) + + // Auto-review is an internal Codex model. Do not infer public GPT-5.6 API + // service-tier, cache-write, or long-context pricing without an upstream + // usage contract for this dedicated model. + require.Zero(t, got.InputCostPerTokenPriority) + require.Zero(t, got.OutputCostPerTokenPriority) + require.Zero(t, got.CacheReadInputTokenCostPriority) + require.Zero(t, got.CacheCreationInputTokenCost) + require.Zero(t, got.CacheCreationInputTokenCostPriority) + require.Zero(t, got.LongContextInputTokenThreshold) } func TestGetModelPricing_Gpt54MiniUsesDedicatedStaticFallbackWhenRemoteMissing(t *testing.T) { diff --git a/backend/resources/model-pricing/model_prices_and_context_window.json b/backend/resources/model-pricing/model_prices_and_context_window.json index 1d8057075d..7dc445e3fc 100644 --- a/backend/resources/model-pricing/model_prices_and_context_window.json +++ b/backend/resources/model-pricing/model_prices_and_context_window.json @@ -710,24 +710,13 @@ }, "codex-auto-review": { "cache_read_input_token_cost": 2e-08, - "cache_read_input_token_cost_above_272k_tokens": 4e-08, - "cache_read_input_token_cost_flex": 1e-08, - "cache_read_input_token_cost_priority": 4e-08, "input_cost_per_token": 2e-07, - "input_cost_per_token_above_272k_tokens": 4e-07, - "input_cost_per_token_batches": 1e-07, - "input_cost_per_token_flex": 1e-07, - "input_cost_per_token_priority": 4e-07, "litellm_provider": "openai", "max_input_tokens": 1050000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 1.2e-06, - "output_cost_per_token_above_272k_tokens": 1.8e-06, - "output_cost_per_token_batches": 6e-07, - "output_cost_per_token_flex": 6e-07, - "output_cost_per_token_priority": 2.4e-06, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch",