feat(openai): support Fast mode service_tier across responses/chat/WS paths

- Accept fast|priority (canonical priority), flex|auto|default|scale on
  /v1/responses and /v1/chat/completions; reject unknown/empty/non-string
  with HTTP 400; omitted and null stay compatible.
- Propagate service_tier through JSON/SSE, Responses<->Chat conversions,
  fallback paths and HTTP->upstream WebSocket bridge.
- Billing prefers the upstream terminal tier; the outbound (policy-
  transformed) tier is used only when upstream omits the field.
  Explicit upstream default bills Standard even when Fast was requested.
- Pricing: Fast premium 2x Standard for gpt-5.6-sol/terra/luna and
  gpt-5.4; 2.5x for gpt-5.5; channel FastMultiplier stays authoritative.
- Live verification (official Codex 0.149.0 + gateway, HTTP & WS):
  upstream ChatGPT backend may return terminal default even when the
  account catalog advertises priority; billing follows the actual tier.
This commit is contained in:
alfadb
2026-08-24 11:52:48 +08:00
parent 7075ae0d82
commit f06bf181d2
25 changed files with 1518 additions and 82 deletions
@@ -5162,12 +5162,12 @@
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 1e-06,
"cache_read_input_token_cost_priority": 1.25e-6,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_272k_tokens": 1e-05,
"input_cost_per_token_batches": 2.5e-06,
"input_cost_per_token_flex": 2.5e-06,
"input_cost_per_token_priority": 1e-05,
"input_cost_per_token_priority": 12.5e-6,
"litellm_provider": "openai",
"max_input_tokens": 1050000,
"max_output_tokens": 128000,
@@ -5177,7 +5177,7 @@
"output_cost_per_token_above_272k_tokens": 4.5e-05,
"output_cost_per_token_batches": 1.5e-05,
"output_cost_per_token_flex": 1.5e-05,
"output_cost_per_token_priority": 6e-05,
"output_cost_per_token_priority": 75e-6,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@@ -5210,12 +5210,12 @@
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 1e-06,
"cache_read_input_token_cost_priority": 1.25e-6,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_272k_tokens": 1e-05,
"input_cost_per_token_batches": 2.5e-06,
"input_cost_per_token_flex": 2.5e-06,
"input_cost_per_token_priority": 1e-05,
"input_cost_per_token_priority": 12.5e-6,
"litellm_provider": "openai",
"max_input_tokens": 1050000,
"max_output_tokens": 128000,
@@ -5225,7 +5225,7 @@
"output_cost_per_token_above_272k_tokens": 4.5e-05,
"output_cost_per_token_batches": 1.5e-05,
"output_cost_per_token_flex": 1.5e-05,
"output_cost_per_token_priority": 6e-05,
"output_cost_per_token_priority": 75e-6,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",