fix(pricing): keep Auto-review rates evidence-based

This commit is contained in:
Zven
2026-07-31 21:20:32 +08:00
parent f54e9827a0
commit 698547418f
2 changed files with 11 additions and 15 deletions
@@ -531,7 +531,7 @@ func TestDefaultPricingIncludesGemini36FlashRates(t *testing.T) {
}
}
func TestDefaultPricingUsesReducedCodexAutoReviewRates(t *testing.T) {
func TestDefaultPricingUsesCurrentCodexAutoReviewBaseRates(t *testing.T) {
data, err := os.ReadFile(filepath.Join("..", "..", "resources", "model-pricing", "model_prices_and_context_window.json"))
require.NoError(t, err)
@@ -543,11 +543,18 @@ func TestDefaultPricingUsesReducedCodexAutoReviewRates(t *testing.T) {
got := svc.GetModelPricing("codex-auto-review")
require.NotNil(t, got)
require.InDelta(t, 0.2e-6, got.InputCostPerToken, 1e-12)
require.InDelta(t, 0.4e-6, got.InputCostPerTokenPriority, 1e-12)
require.InDelta(t, 1.2e-6, got.OutputCostPerToken, 1e-12)
require.InDelta(t, 2.4e-6, got.OutputCostPerTokenPriority, 1e-12)
require.InDelta(t, 0.02e-6, got.CacheReadInputTokenCost, 1e-12)
require.InDelta(t, 0.04e-6, got.CacheReadInputTokenCostPriority, 1e-12)
// Auto-review is an internal Codex model. Do not infer public GPT-5.6 API
// service-tier, cache-write, or long-context pricing without an upstream
// usage contract for this dedicated model.
require.Zero(t, got.InputCostPerTokenPriority)
require.Zero(t, got.OutputCostPerTokenPriority)
require.Zero(t, got.CacheReadInputTokenCostPriority)
require.Zero(t, got.CacheCreationInputTokenCost)
require.Zero(t, got.CacheCreationInputTokenCostPriority)
require.Zero(t, got.LongContextInputTokenThreshold)
}
func TestGetModelPricing_Gpt54MiniUsesDedicatedStaticFallbackWhenRemoteMissing(t *testing.T) {
@@ -710,24 +710,13 @@
},
"codex-auto-review": {
"cache_read_input_token_cost": 2e-08,
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
"cache_read_input_token_cost_flex": 1e-08,
"cache_read_input_token_cost_priority": 4e-08,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
"input_cost_per_token_batches": 1e-07,
"input_cost_per_token_flex": 1e-07,
"input_cost_per_token_priority": 4e-07,
"litellm_provider": "openai",
"max_input_tokens": 1050000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"output_cost_per_token_above_272k_tokens": 1.8e-06,
"output_cost_per_token_batches": 6e-07,
"output_cost_per_token_flex": 6e-07,
"output_cost_per_token_priority": 2.4e-06,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",