From b2d895fb8dcaa25c059785afa065b6743dfa7e6d Mon Sep 17 00:00:00 2001 From: Zven Date: Fri, 31 Jul 2026 09:42:46 +0800 Subject: [PATCH] fix(pricing): update GPT-5.6 Luna and Terra rates --- backend/internal/service/billing_service.go | 32 +++++----- backend/internal/service/pricing_service.go | 32 +++++----- .../internal/service/pricing_service_test.go | 20 +++---- .../model_prices_and_context_window.json | 60 +++++++++---------- 4 files changed, 72 insertions(+), 72 deletions(-) diff --git a/backend/internal/service/billing_service.go b/backend/internal/service/billing_service.go index 898138e79..893c6d0c7 100644 --- a/backend/internal/service/billing_service.go +++ b/backend/internal/service/billing_service.go @@ -316,27 +316,27 @@ func (s *BillingService) initFallbackPricing() { LongContextOutputMultiplier: openAIGPT54LongContextOutputMultiplier, } s.fallbackPrices["gpt-5.6-terra"] = &ModelPricing{ - InputPricePerToken: 2.5e-6, - InputPricePerTokenPriority: 5e-6, - OutputPricePerToken: 15e-6, - OutputPricePerTokenPriority: 30e-6, - CacheCreationPricePerToken: 3.125e-6, - CacheCreationPricePerTokenPriority: 6.25e-6, - CacheReadPricePerToken: 0.25e-6, - CacheReadPricePerTokenPriority: 0.5e-6, + InputPricePerToken: 2e-6, + InputPricePerTokenPriority: 4e-6, + OutputPricePerToken: 12e-6, + OutputPricePerTokenPriority: 24e-6, + CacheCreationPricePerToken: 2.5e-6, + CacheCreationPricePerTokenPriority: 5e-6, + CacheReadPricePerToken: 0.2e-6, + CacheReadPricePerTokenPriority: 0.4e-6, LongContextInputThreshold: openAIGPT54LongContextInputThreshold, LongContextInputMultiplier: openAIGPT54LongContextInputMultiplier, LongContextOutputMultiplier: openAIGPT54LongContextOutputMultiplier, } s.fallbackPrices["gpt-5.6-luna"] = &ModelPricing{ - InputPricePerToken: 1e-6, - InputPricePerTokenPriority: 2e-6, - OutputPricePerToken: 6e-6, - OutputPricePerTokenPriority: 12e-6, - CacheCreationPricePerToken: 1.25e-6, - CacheCreationPricePerTokenPriority: 2.5e-6, - CacheReadPricePerToken: 0.1e-6, - CacheReadPricePerTokenPriority: 0.2e-6, + InputPricePerToken: 0.2e-6, + InputPricePerTokenPriority: 0.4e-6, + OutputPricePerToken: 1.2e-6, + OutputPricePerTokenPriority: 2.4e-6, + CacheCreationPricePerToken: 0.25e-6, + CacheCreationPricePerTokenPriority: 0.5e-6, + CacheReadPricePerToken: 0.02e-6, + CacheReadPricePerTokenPriority: 0.04e-6, LongContextInputThreshold: openAIGPT54LongContextInputThreshold, LongContextInputMultiplier: openAIGPT54LongContextInputMultiplier, LongContextOutputMultiplier: openAIGPT54LongContextOutputMultiplier, diff --git a/backend/internal/service/pricing_service.go b/backend/internal/service/pricing_service.go index c969ec18d..d0dce84f4 100644 --- a/backend/internal/service/pricing_service.go +++ b/backend/internal/service/pricing_service.go @@ -53,14 +53,14 @@ var ( SupportsPromptCaching: true, } openAIGPT56TerraFallbackPricing = &LiteLLMModelPricing{ - InputCostPerToken: 2.5e-06, - InputCostPerTokenPriority: 5e-06, - OutputCostPerToken: 1.5e-05, - OutputCostPerTokenPriority: 3e-05, - CacheCreationInputTokenCost: 3.125e-06, - CacheCreationInputTokenCostPriority: 6.25e-06, - CacheReadInputTokenCost: 2.5e-07, - CacheReadInputTokenCostPriority: 5e-07, + InputCostPerToken: 2e-06, + InputCostPerTokenPriority: 4e-06, + OutputCostPerToken: 1.2e-05, + OutputCostPerTokenPriority: 2.4e-05, + CacheCreationInputTokenCost: 2.5e-06, + CacheCreationInputTokenCostPriority: 5e-06, + CacheReadInputTokenCost: 2e-07, + CacheReadInputTokenCostPriority: 4e-07, LongContextInputTokenThreshold: openAIGPT54LongContextInputThreshold, LongContextInputCostMultiplier: openAIGPT54LongContextInputMultiplier, LongContextOutputCostMultiplier: openAIGPT54LongContextOutputMultiplier, @@ -70,14 +70,14 @@ var ( SupportsPromptCaching: true, } openAIGPT56LunaFallbackPricing = &LiteLLMModelPricing{ - InputCostPerToken: 1e-06, - InputCostPerTokenPriority: 2e-06, - OutputCostPerToken: 6e-06, - OutputCostPerTokenPriority: 1.2e-05, - CacheCreationInputTokenCost: 1.25e-06, - CacheCreationInputTokenCostPriority: 2.5e-06, - CacheReadInputTokenCost: 1e-07, - CacheReadInputTokenCostPriority: 2e-07, + InputCostPerToken: 2e-07, + InputCostPerTokenPriority: 4e-07, + OutputCostPerToken: 1.2e-06, + OutputCostPerTokenPriority: 2.4e-06, + CacheCreationInputTokenCost: 2.5e-07, + CacheCreationInputTokenCostPriority: 5e-07, + CacheReadInputTokenCost: 2e-08, + CacheReadInputTokenCostPriority: 4e-08, LongContextInputTokenThreshold: openAIGPT54LongContextInputThreshold, LongContextInputCostMultiplier: openAIGPT54LongContextInputMultiplier, LongContextOutputCostMultiplier: openAIGPT54LongContextOutputMultiplier, diff --git a/backend/internal/service/pricing_service_test.go b/backend/internal/service/pricing_service_test.go index 48e5159ad..7db756533 100644 --- a/backend/internal/service/pricing_service_test.go +++ b/backend/internal/service/pricing_service_test.go @@ -88,8 +88,8 @@ func TestBillingService_GPT56CacheWritePricingUsesOfficialMultiplier(t *testing. cacheReadPriority float64 }{ {model: "gpt-5.6-sol", input: 5e-6, inputPriority: 10e-6, output: 30e-6, outputPriority: 60e-6, cacheRead: 0.5e-6, cacheReadPriority: 1e-6}, - {model: "gpt-5.6-terra", input: 2.5e-6, inputPriority: 5e-6, output: 15e-6, outputPriority: 30e-6, cacheRead: 0.25e-6, cacheReadPriority: 0.5e-6}, - {model: "gpt-5.6-luna", input: 1e-6, inputPriority: 2e-6, output: 6e-6, outputPriority: 12e-6, cacheRead: 0.1e-6, cacheReadPriority: 0.2e-6}, + {model: "gpt-5.6-terra", input: 2e-6, inputPriority: 4e-6, output: 12e-6, outputPriority: 24e-6, cacheRead: 0.2e-6, cacheReadPriority: 0.4e-6}, + {model: "gpt-5.6-luna", input: 0.2e-6, inputPriority: 0.4e-6, output: 1.2e-6, outputPriority: 2.4e-6, cacheRead: 0.02e-6, cacheReadPriority: 0.04e-6}, } for _, tt := range tests { t.Run(tt.model, func(t *testing.T) { @@ -136,8 +136,8 @@ func TestBillingService_GPT56UsesLongContextPricingAcrossModelsAndTiers(t *testi cacheWrite, output float64 }{ {name: "gpt-5.6-sol", input: 5e-6, cached: 0.5e-6, cacheWrite: 6.25e-6, output: 30e-6}, - {name: "gpt-5.6-terra", input: 2.5e-6, cached: 0.25e-6, cacheWrite: 3.125e-6, output: 15e-6}, - {name: "gpt-5.6-luna", input: 1e-6, cached: 0.1e-6, cacheWrite: 1.25e-6, output: 6e-6}, + {name: "gpt-5.6-terra", input: 2e-6, cached: 0.2e-6, cacheWrite: 2.5e-6, output: 12e-6}, + {name: "gpt-5.6-luna", input: 0.2e-6, cached: 0.02e-6, cacheWrite: 0.25e-6, output: 1.2e-6}, } tiers := []struct { name string @@ -188,8 +188,8 @@ func TestBillingService_GPT56LongContextBoundaryIsExclusive(t *testing.T) { func TestPricingService_BareGPT56AliasDeterministicallyUsesSol(t *testing.T) { pricingSvc := &PricingService{pricingData: map[string]*LiteLLMModelPricing{ "gpt-5.6-sol": {InputCostPerToken: 5e-6}, - "gpt-5.6-terra": {InputCostPerToken: 2.5e-6}, - "gpt-5.6-luna": {InputCostPerToken: 1e-6}, + "gpt-5.6-terra": {InputCostPerToken: 2e-6}, + "gpt-5.6-luna": {InputCostPerToken: 0.2e-6}, "gpt-5.4": {InputCostPerToken: 2.5e-6}, }} @@ -226,8 +226,8 @@ func TestDefaultPricingIncludesOfficialGPT56Rates(t *testing.T) { inputPriority, cachedPriority, cacheWritePriority, outputPriority float64 }{ {model: "gpt-5.6-sol", input: 5e-6, cached: 0.5e-6, cacheWrite: 6.25e-6, output: 30e-6, inputPriority: 10e-6, cachedPriority: 1e-6, cacheWritePriority: 12.5e-6, outputPriority: 60e-6}, - {model: "gpt-5.6-terra", input: 2.5e-6, cached: 0.25e-6, cacheWrite: 3.125e-6, output: 15e-6, inputPriority: 5e-6, cachedPriority: 0.5e-6, cacheWritePriority: 6.25e-6, outputPriority: 30e-6}, - {model: "gpt-5.6-luna", input: 1e-6, cached: 0.1e-6, cacheWrite: 1.25e-6, output: 6e-6, inputPriority: 2e-6, cachedPriority: 0.2e-6, cacheWritePriority: 2.5e-6, outputPriority: 12e-6}, + {model: "gpt-5.6-terra", input: 2e-6, cached: 0.2e-6, cacheWrite: 2.5e-6, output: 12e-6, inputPriority: 4e-6, cachedPriority: 0.4e-6, cacheWritePriority: 5e-6, outputPriority: 24e-6}, + {model: "gpt-5.6-luna", input: 0.2e-6, cached: 0.02e-6, cacheWrite: 0.25e-6, output: 1.2e-6, inputPriority: 0.4e-6, cachedPriority: 0.04e-6, cacheWritePriority: 0.5e-6, outputPriority: 2.4e-6}, } for _, tt := range tests { t.Run(tt.model, func(t *testing.T) { @@ -254,8 +254,8 @@ func TestGPT56DedicatedFallbacksUseOfficialRates(t *testing.T) { input, cached, cacheWrite, output float64 }{ {model: "gpt-5.6-sol", input: 5e-6, cached: 0.5e-6, cacheWrite: 6.25e-6, output: 30e-6}, - {model: "gpt-5.6-terra", input: 2.5e-6, cached: 0.25e-6, cacheWrite: 3.125e-6, output: 15e-6}, - {model: "gpt-5.6-luna", input: 1e-6, cached: 0.1e-6, cacheWrite: 1.25e-6, output: 6e-6}, + {model: "gpt-5.6-terra", input: 2e-6, cached: 0.2e-6, cacheWrite: 2.5e-6, output: 12e-6}, + {model: "gpt-5.6-luna", input: 0.2e-6, cached: 0.02e-6, cacheWrite: 0.25e-6, output: 1.2e-6}, } for _, tt := range tests { diff --git a/backend/resources/model-pricing/model_prices_and_context_window.json b/backend/resources/model-pricing/model_prices_and_context_window.json index 21180fb56..36cfc2f6b 100644 --- a/backend/resources/model-pricing/model_prices_and_context_window.json +++ b/backend/resources/model-pricing/model_prices_and_context_window.json @@ -5066,17 +5066,17 @@ "supports_xhigh_reasoning_effort": true }, "gpt-5.6-terra": { - "cache_creation_input_token_cost": 3.125e-06, - "cache_creation_input_token_cost_batches": 1.5625e-06, - "cache_creation_input_token_cost_flex": 1.5625e-06, - "cache_creation_input_token_cost_priority": 6.25e-06, - "cache_read_input_token_cost": 2.5e-07, - "cache_read_input_token_cost_flex": 1.25e-07, - "cache_read_input_token_cost_priority": 5e-07, - "input_cost_per_token": 2.5e-06, - "input_cost_per_token_batches": 1.25e-06, - "input_cost_per_token_flex": 1.25e-06, - "input_cost_per_token_priority": 5e-06, + "cache_creation_input_token_cost": 2.5e-06, + "cache_creation_input_token_cost_batches": 1.25e-06, + "cache_creation_input_token_cost_flex": 1.25e-06, + "cache_creation_input_token_cost_priority": 5e-06, + "cache_read_input_token_cost": 2e-07, + "cache_read_input_token_cost_flex": 1e-07, + "cache_read_input_token_cost_priority": 4e-07, + "input_cost_per_token": 2e-06, + "input_cost_per_token_batches": 1e-06, + "input_cost_per_token_flex": 1e-06, + "input_cost_per_token_priority": 4e-06, "long_context_input_token_threshold": 272000, "long_context_input_cost_multiplier": 2.0, "long_context_output_cost_multiplier": 1.5, @@ -5085,10 +5085,10 @@ "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 1.5e-05, - "output_cost_per_token_batches": 7.5e-06, - "output_cost_per_token_flex": 7.5e-06, - "output_cost_per_token_priority": 3e-05, + "output_cost_per_token": 1.2e-05, + "output_cost_per_token_batches": 6e-06, + "output_cost_per_token_flex": 6e-06, + "output_cost_per_token_priority": 2.4e-05, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", @@ -5118,17 +5118,17 @@ "supports_xhigh_reasoning_effort": true }, "gpt-5.6-luna": { - "cache_creation_input_token_cost": 1.25e-06, - "cache_creation_input_token_cost_batches": 6.25e-07, - "cache_creation_input_token_cost_flex": 6.25e-07, - "cache_creation_input_token_cost_priority": 2.5e-06, - "cache_read_input_token_cost": 1e-07, - "cache_read_input_token_cost_flex": 5e-08, - "cache_read_input_token_cost_priority": 2e-07, - "input_cost_per_token": 1e-06, - "input_cost_per_token_batches": 5e-07, - "input_cost_per_token_flex": 5e-07, - "input_cost_per_token_priority": 2e-06, + "cache_creation_input_token_cost": 2.5e-07, + "cache_creation_input_token_cost_batches": 1.25e-07, + "cache_creation_input_token_cost_flex": 1.25e-07, + "cache_creation_input_token_cost_priority": 5e-07, + "cache_read_input_token_cost": 2e-08, + "cache_read_input_token_cost_flex": 1e-08, + "cache_read_input_token_cost_priority": 4e-08, + "input_cost_per_token": 2e-07, + "input_cost_per_token_batches": 1e-07, + "input_cost_per_token_flex": 1e-07, + "input_cost_per_token_priority": 4e-07, "long_context_input_token_threshold": 272000, "long_context_input_cost_multiplier": 2.0, "long_context_output_cost_multiplier": 1.5, @@ -5137,10 +5137,10 @@ "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 6e-06, - "output_cost_per_token_batches": 3e-06, - "output_cost_per_token_flex": 3e-06, - "output_cost_per_token_priority": 1.2e-05, + "output_cost_per_token": 1.2e-06, + "output_cost_per_token_batches": 6e-07, + "output_cost_per_token_flex": 6e-07, + "output_cost_per_token_priority": 2.4e-06, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch",