From 47bd8e6b911bdb852dd3e06b2bbc7add8e0f2b23 Mon Sep 17 00:00:00 2001 From: RCD <90105158+Finesssee@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:49:00 +0700 Subject: [PATCH 1/4] Reprice GPT-5.6 Sol from 2026-08-21 and add Cyber and Daybreak pricing Port upstream 0.70.0 #4094 (Codex half): Sol drops to $4/$20 with dated usage before 2026-08-21 keeping the $5/$30 rates, gpt-5.6-cyber and gpt-5.5-cyber get bundled rates, and the Daybreak blue/red aliases price as Sol and Cyber. GPT-5.6 entries now carry their 1.25x cache-write rates like upstream gpt56Pricing, replacing the Astra-only special case. --- rust/src/core/cost_pricing.rs | 136 +++++++++++++---------- rust/src/core/cost_pricing/codex.rs | 138 +++++++++++++++++------- rust/src/core/cost_pricing_tests.rs | 160 ++++++++++++++++++++++++++-- 3 files changed, 331 insertions(+), 103 deletions(-) diff --git a/rust/src/core/cost_pricing.rs b/rust/src/core/cost_pricing.rs index c4a8effebd..78cf0ab4b3 100755 --- a/rust/src/core/cost_pricing.rs +++ b/rust/src/core/cost_pricing.rs @@ -16,6 +16,9 @@ pub struct CodexLongContextRates { pub input_cost_per_token: f64, pub output_cost_per_token: f64, pub cache_read_input_cost_per_token: f64, + /// `None` falls back to the standard cache-write rate, then to this + /// tier's input rate. + pub cache_write_input_cost_per_token: Option, } /// Codex (OpenAI) model pricing #[derive(Debug, Clone, Copy)] @@ -26,6 +29,9 @@ pub struct CodexPricing { pub output_cost_per_token: f64, /// Cost per cached input token in USD pub cache_read_input_cost_per_token: f64, + /// Cost per cache-write input token in USD; `None` bills cache writes at + /// the input rate. + pub cache_write_input_cost_per_token: Option, /// Optional display label override (e.g. "Research Preview") pub display_label: Option<&'static str>, /// Whole-request rates above the Codex long-context threshold. @@ -65,6 +71,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -75,6 +82,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -85,6 +93,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2.5e-7, output_cost_per_token: 2e-6, cache_read_input_cost_per_token: 2.5e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -95,6 +104,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 5e-8, output_cost_per_token: 4e-7, cache_read_input_cost_per_token: 5e-9, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -105,6 +115,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.5e-5, output_cost_per_token: 1.2e-4, cache_read_input_cost_per_token: 1.5e-5, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -115,6 +126,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -125,6 +137,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -135,6 +148,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -145,6 +159,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2.5e-7, output_cost_per_token: 2e-6, cache_read_input_cost_per_token: 2.5e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -155,6 +170,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.75e-6, output_cost_per_token: 1.4e-5, cache_read_input_cost_per_token: 1.75e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -165,6 +181,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.75e-6, output_cost_per_token: 1.4e-5, cache_read_input_cost_per_token: 1.75e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -175,6 +192,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2.1e-5, output_cost_per_token: 1.68e-4, cache_read_input_cost_per_token: 2.1e-5, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -185,6 +203,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.75e-6, output_cost_per_token: 1.4e-5, cache_read_input_cost_per_token: 1.75e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -195,6 +214,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 0.0, output_cost_per_token: 0.0, cache_read_input_cost_per_token: 0.0, + cache_write_input_cost_per_token: None, display_label: Some("Research Preview"), long_context: None, }, @@ -207,6 +227,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2.5e-6, output_cost_per_token: 1.5e-5, cache_read_input_cost_per_token: 2.5e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -217,6 +238,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2.5e-6, output_cost_per_token: 1.5e-5, cache_read_input_cost_per_token: 2.5e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -229,6 +251,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 7.5e-7, output_cost_per_token: 4.5e-6, cache_read_input_cost_per_token: 7.5e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -239,6 +262,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 7.5e-7, output_cost_per_token: 4.5e-6, cache_read_input_cost_per_token: 7.5e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -251,6 +275,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2e-7, output_cost_per_token: 1.25e-6, cache_read_input_cost_per_token: 2e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -261,6 +286,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2e-7, output_cost_per_token: 1.25e-6, cache_read_input_cost_per_token: 2e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -273,6 +299,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 3e-5, output_cost_per_token: 1.8e-4, cache_read_input_cost_per_token: 3e-5, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -283,6 +310,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 5e-6, output_cost_per_token: 3e-5, cache_read_input_cost_per_token: 5e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -293,65 +321,70 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 3e-5, output_cost_per_token: 1.8e-4, cache_read_input_cost_per_token: 3e-5, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, ); + // GPT-5.6 Sol/Terra/Luna (OpenAI pricing page and model cards), in + // upstream `gpt56Pricing` order (input, cache read, cache write, output). + // Above 272K input tokens the whole request bills 2x input / 1.5x output; + // cache writes bill at 1.25x uncached input. Sol was repriced from $5/$30 + // to $4/$20 on 2026-08-21. Dated usage before a model's repricing keeps + // the rates in `codex_pricing::codex_historical_pricing`. m.insert( "gpt-5.6-sol", - CodexPricing { - input_cost_per_token: 5e-6, - output_cost_per_token: 3e-5, - cache_read_input_cost_per_token: 5e-7, - display_label: None, - long_context: Some(CodexLongContextRates { - input_cost_per_token: 1e-5, - output_cost_per_token: 4.5e-5, - cache_read_input_cost_per_token: 1e-6, - }), - }, + codex_pricing::gpt56_pricing((4e-6, 4e-7, 5e-6, 2e-5), (8e-6, 8e-7, 1e-5, 3e-5)), ); m.insert( "gpt-5.6-terra", + codex_pricing::gpt56_pricing((2e-6, 2e-7, 2.5e-6, 1.2e-5), (4e-6, 4e-7, 5e-6, 1.8e-5)), + ); + m.insert( + "gpt-5.6-luna", + codex_pricing::gpt56_pricing((2e-7, 2e-8, 2.5e-7, 1.2e-6), (4e-7, 4e-8, 5e-7, 1.8e-6)), + ); + // Daybreak Cyber models (OpenAI pricing page). No long-context tier is + // published, and gpt-5.5-cyber lists no cache-write rate, so its writes + // bill at the input rate. + m.insert( + "gpt-5.6-cyber", CodexPricing { - input_cost_per_token: 2e-6, - output_cost_per_token: 1.2e-5, - cache_read_input_cost_per_token: 2e-7, + input_cost_per_token: 1.25e-5, + output_cost_per_token: 7.5e-5, + cache_read_input_cost_per_token: 1.25e-6, + cache_write_input_cost_per_token: Some(1.5625e-5), display_label: None, - long_context: Some(CodexLongContextRates { - input_cost_per_token: 4e-6, - output_cost_per_token: 1.8e-5, - cache_read_input_cost_per_token: 4e-7, - }), + long_context: None, }, ); m.insert( - "gpt-5.6-luna", + "gpt-5.5-cyber", CodexPricing { - input_cost_per_token: 2e-7, - output_cost_per_token: 1.2e-6, - cache_read_input_cost_per_token: 2e-8, + input_cost_per_token: 1.25e-5, + output_cost_per_token: 7.5e-5, + cache_read_input_cost_per_token: 1.25e-6, + cache_write_input_cost_per_token: None, display_label: None, - long_context: Some(CodexLongContextRates { - input_cost_per_token: 4e-7, - output_cost_per_token: 1.8e-6, - cache_read_input_cost_per_token: 4e-8, - }), + long_context: None, }, ); // GPT-6 Astra pricing (OpenAI model card and pricing table). - // Long-context rates apply to the whole request above 272K input tokens. + // Long-context rates apply to the whole request above 272K input tokens; + // cache writes bill at 1.25x uncached input in both tiers. m.insert( "gpt-6-astra", CodexPricing { input_cost_per_token: 1e-5, output_cost_per_token: 5e-5, cache_read_input_cost_per_token: 1e-6, + cache_write_input_cost_per_token: Some(1.25e-5), display_label: None, long_context: Some(CodexLongContextRates { input_cost_per_token: 2e-5, output_cost_per_token: 7.5e-5, cache_read_input_cost_per_token: 2e-6, + cache_write_input_cost_per_token: Some(2.5e-5), }), }, ); @@ -629,6 +662,14 @@ impl CostUsagePricing { trimmed = rest.to_string(); } + // OpenAI's Daybreak aliases currently point to Sol (blue) and Cyber + // (red). https://developers.openai.com/api/docs/pricing + match trimmed.as_str() { + "gpt-daybreak-blue-latest" => return "gpt-5.6-sol".to_string(), + "gpt-daybreak-red-latest" => return "gpt-5.6-cyber".to_string(), + _ => {} + } + // Check if base model (without -codex suffix) exists in pricing if let Some(idx) = trimmed.find("-codex") { let base = &trimmed[..idx]; @@ -722,7 +763,7 @@ impl CostUsagePricing { } /// Calculate Codex cost using the rates in effect on a historical usage day. - /// GPT-5.6 Terra/Luna were cut on 2026-07-30; Sol was unchanged. + /// GPT-5.6 Terra/Luna were cut on 2026-07-30 and Sol on 2026-08-21. pub fn codex_cost_usd_at_date( model: &str, input_tokens: u64, @@ -761,8 +802,8 @@ impl CostUsagePricing { /// Codex cost on a historical usage day when the prompt also wrote cache /// tokens. `input_tokens` is the inclusive prompt size: cache reads and - /// writes are subsets of it. The pre-cutoff GPT-5.6 Terra/Luna rates carry - /// their own 1.25x cache-write rate (upstream `codexHistoricalPricing`). + /// writes are subsets of it. A model's pre-repricing rates (upstream + /// `codexHistoricalPricing`) win over today's bundled and catalog rates. pub fn codex_cost_usd_at_date_with_cache_write_and_pricing_snapshot( model: &str, input_tokens: u64, @@ -773,29 +814,14 @@ impl CostUsagePricing { pricing_snapshot: Option<&models_dev_pricing::ModelsDevPricingSnapshot>, ) -> Option { let key = Self::normalize_codex_model(model); - let cutoff = NaiveDate::from_ymd_opt(2026, 7, 30).expect("valid pricing cutoff"); - if pricing_date < cutoff { - let long = input_tokens > codex_pricing::CODEX_LONG_CONTEXT_THRESHOLD; - // (input, cache read, cache write, output) per token. - let rates = match (key.as_str(), long) { - ("gpt-5.6-terra", false) => Some((2.5e-6, 2.5e-7, 3.125e-6, 1.5e-5)), - ("gpt-5.6-terra", true) => Some((5e-6, 5e-7, 6.25e-6, 2.25e-5)), - ("gpt-5.6-luna", false) => Some((1e-6, 1e-7, 1.25e-6, 6e-6)), - ("gpt-5.6-luna", true) => Some((2e-6, 2e-7, 2.5e-6, 9e-6)), - _ => None, - }; - if let Some((input_rate, cache_read_rate, cache_write_rate, output_rate)) = rates { - return Some(codex_pricing::codex_cost_from_rates_with_cache_write( - input_tokens, - cached_input_tokens, - cache_write_input_tokens, - output_tokens, - input_rate, - cache_read_rate, - cache_write_rate, - output_rate, - )); - } + if let Some(pricing) = codex_pricing::codex_historical_pricing(&key, pricing_date) { + return Some(codex_pricing::codex_cost_from_pricing( + &pricing, + input_tokens, + cached_input_tokens, + cache_write_input_tokens, + output_tokens, + )); } Self::codex_cost_usd_with_cache_write_and_pricing_snapshot( model, diff --git a/rust/src/core/cost_pricing/codex.rs b/rust/src/core/cost_pricing/codex.rs index 09116be6f4..90a31efad8 100644 --- a/rust/src/core/cost_pricing/codex.rs +++ b/rust/src/core/cost_pricing/codex.rs @@ -1,9 +1,103 @@ use super::super::{codex_routed_pricing, models_dev_pricing}; -use super::{CODEX_PRICING, CostUsagePricing}; +use super::{CODEX_PRICING, CodexLongContextRates, CodexPricing, CostUsagePricing}; +use chrono::NaiveDate; pub(super) const CODEX_LONG_CONTEXT_THRESHOLD: u64 = 272_000; -const CODEX_ASTRA_CACHE_WRITE_RATE: f64 = 1.25e-5; -const CODEX_ASTRA_LONG_CACHE_WRITE_RATE: f64 = 2.5e-5; + +/// GPT-5.6 rates per token in upstream `gpt56Pricing` order: +/// (input, cache read, cache write, output). +pub(super) type Gpt56Rates = (f64, f64, f64, f64); + +/// Upstream `gpt56Pricing`: standard rates plus the whole-request rates above +/// the 272K-token long-context threshold, each with its own cache-write rate. +pub(super) const fn gpt56_pricing(standard: Gpt56Rates, long_context: Gpt56Rates) -> CodexPricing { + let (input, cache_read, cache_write, output) = standard; + let (long_input, long_cache_read, long_cache_write, long_output) = long_context; + CodexPricing { + input_cost_per_token: input, + output_cost_per_token: output, + cache_read_input_cost_per_token: cache_read, + cache_write_input_cost_per_token: Some(cache_write), + display_label: None, + long_context: Some(CodexLongContextRates { + input_cost_per_token: long_input, + output_cost_per_token: long_output, + cache_read_input_cost_per_token: long_cache_read, + cache_write_input_cost_per_token: Some(long_cache_write), + }), + } +} + +/// Upstream `codexHistoricalPricing`: the rates a model billed at before its +/// repricing. GPT-5.6 Terra and Luna were cut on 2026-07-30 (Unix 1785369600) +/// and Sol on 2026-08-21 (Unix 1787270400). Windows keys usage by calendar +/// day, so the cutoff compares days rather than event instants. +pub(super) fn codex_historical_pricing(key: &str, pricing_date: NaiveDate) -> Option { + let ((year, month, day), standard, long_context) = match key { + "gpt-5.6-sol" => ( + (2026, 8, 21), + (5e-6, 5e-7, 6.25e-6, 3e-5), + (1e-5, 1e-6, 1.25e-5, 4.5e-5), + ), + "gpt-5.6-terra" => ( + (2026, 7, 30), + (2.5e-6, 2.5e-7, 3.125e-6, 1.5e-5), + (5e-6, 5e-7, 6.25e-6, 2.25e-5), + ), + "gpt-5.6-luna" => ( + (2026, 7, 30), + (1e-6, 1e-7, 1.25e-6, 6e-6), + (2e-6, 2e-7, 2.5e-6, 9e-6), + ), + _ => return None, + }; + let cutoff = NaiveDate::from_ymd_opt(year, month, day)?; + (pricing_date < cutoff).then(|| gpt56_pricing(standard, long_context)) +} + +/// Upstream `codexCostUSD(pricing:)` for one bundled entry. `input_tokens` is +/// the inclusive prompt size and selects the long-context tier. A long-context +/// cache write without its own rate falls back to the standard cache-write +/// rate, then to the tier's input rate. +pub(super) fn codex_cost_from_pricing( + pricing: &CodexPricing, + input_tokens: u64, + cached_input_tokens: u64, + cache_write_input_tokens: u64, + output_tokens: u64, +) -> f64 { + let long_context = pricing + .long_context + .filter(|_| input_tokens > CODEX_LONG_CONTEXT_THRESHOLD); + let (input_rate, cache_read_rate, cache_write_rate, output_rate) = match long_context { + Some(long) => ( + long.input_cost_per_token, + long.cache_read_input_cost_per_token, + long.cache_write_input_cost_per_token + .or(pricing.cache_write_input_cost_per_token) + .unwrap_or(long.input_cost_per_token), + long.output_cost_per_token, + ), + None => ( + pricing.input_cost_per_token, + pricing.cache_read_input_cost_per_token, + pricing + .cache_write_input_cost_per_token + .unwrap_or(pricing.input_cost_per_token), + pricing.output_cost_per_token, + ), + }; + codex_cost_from_rates_with_cache_write( + input_tokens, + cached_input_tokens, + cache_write_input_tokens, + output_tokens, + input_rate, + cache_read_rate, + cache_write_rate, + output_rate, + ) +} pub(super) fn codex_cost_from_rates( input_tokens: u64, @@ -110,46 +204,12 @@ impl CostUsagePricing { return None; } if let Some(pricing) = CODEX_PRICING.get(key.as_str()) { - let long = input_tokens > CODEX_LONG_CONTEXT_THRESHOLD; - let (input_rate, cache_read_rate, output_rate) = if long { - if let Some(long_context) = pricing.long_context { - ( - long_context.input_cost_per_token, - long_context.cache_read_input_cost_per_token, - long_context.output_cost_per_token, - ) - } else { - ( - pricing.input_cost_per_token, - pricing.cache_read_input_cost_per_token, - pricing.output_cost_per_token, - ) - } - } else { - ( - pricing.input_cost_per_token, - pricing.cache_read_input_cost_per_token, - pricing.output_cost_per_token, - ) - }; - let cache_write_rate = if key == "gpt-6-astra" { - if long { - CODEX_ASTRA_LONG_CACHE_WRITE_RATE - } else { - CODEX_ASTRA_CACHE_WRITE_RATE - } - } else { - input_rate - }; - return Some(codex_cost_from_rates_with_cache_write( + return Some(codex_cost_from_pricing( + pricing, input_tokens, cached_input_tokens, cache_write_input_tokens, output_tokens, - input_rate, - cache_read_rate, - cache_write_rate, - output_rate, )); } diff --git a/rust/src/core/cost_pricing_tests.rs b/rust/src/core/cost_pricing_tests.rs index 106f3b09e0..7d8dad0230 100644 --- a/rust/src/core/cost_pricing_tests.rs +++ b/rust/src/core/cost_pricing_tests.rs @@ -159,7 +159,7 @@ fn test_gpt5_pro_cost() { #[test] fn test_gpt56_standard_pricing() { for (model, expected) in [ - ("gpt-5.6-sol", 0.0332), + ("gpt-5.6-sol", 0.02256), ("gpt-5.6-terra", 0.01328), ("gpt-5.6-luna", 0.001328), ] { @@ -171,7 +171,7 @@ fn test_gpt56_standard_pricing() { #[test] fn test_gpt56_long_context_pricing() { for (model, expected) in [ - ("gpt-5.6-sol", 45.272001), + ("gpt-5.6-sol", 30.2176008), ("gpt-5.6-terra", 18.1088004), ("gpt-5.6-luna", 1.81088004), ] { @@ -183,7 +183,7 @@ fn test_gpt56_long_context_pricing() { #[test] fn test_gpt56_context_threshold_is_exclusive() { for (model, expected) in [ - ("gpt-5.6-sol", 0.136), + ("gpt-5.6-sol", 0.1088), ("gpt-5.6-terra", 0.0544), ("gpt-5.6-luna", 0.00544), ] { @@ -491,14 +491,23 @@ fn gpt56_historical_terra_luna_rates_change_at_2026_07_30() { } #[test] -fn gpt56_historical_pricing_keeps_sol_unchanged() { +fn gpt56_historical_sol_rates_change_at_2026_08_21() { use chrono::NaiveDate; - let before = NaiveDate::from_ymd_opt(2026, 7, 29).unwrap(); - let current = CostUsagePricing::codex_cost_usd("gpt-5.6-sol", 100, 10, 5).unwrap(); - let historical = - CostUsagePricing::codex_cost_usd_at_date("gpt-5.6-sol", 100, 10, 5, before).unwrap(); - assert!((historical - current).abs() < f64::EPSILON); + // Sol keeps its own cutoff: still historical on Terra/Luna's cut day. + let historical = 90.0 * 5e-6 + 10.0 * 5e-7 + 5.0 * 3e-5; + let current = 90.0 * 4e-6 + 10.0 * 4e-7 + 5.0 * 2e-5; + for (date, expected) in [ + ((2026, 7, 30), historical), + ((2026, 8, 20), historical), + ((2026, 8, 21), current), + ] { + let day = NaiveDate::from_ymd_opt(date.0, date.1, date.2).unwrap(); + let cost = CostUsagePricing::codex_cost_usd_at_date("gpt-5.6-sol", 100, 10, 5, day); + assert!((cost.unwrap() - expected).abs() < 1e-12, "{day}"); + } + let undated = CostUsagePricing::codex_cost_usd("gpt-5.6-sol", 100, 10, 5).unwrap(); + assert!((undated - current).abs() < 1e-12); } #[test] @@ -583,3 +592,136 @@ fn gpt6_astra_unknown_models_fail_closed() { assert!(CostUsagePricing::codex_cost_usd(model, 1000, 0, 100).is_none()); } } + +#[test] +fn normalize_codex_model_maps_daybreak_aliases_and_cyber_ids() { + for (raw, expected) in [ + ("gpt-daybreak-blue-latest", "gpt-5.6-sol"), + ("openai/gpt-daybreak-blue-latest", "gpt-5.6-sol"), + ("gpt-daybreak-red-latest", "gpt-5.6-cyber"), + ("gpt-5.6-cyber", "gpt-5.6-cyber"), + ("gpt-5.5-cyber", "gpt-5.5-cyber"), + ] { + assert_eq!( + CostUsagePricing::normalize_codex_model(raw), + expected, + "{raw}" + ); + } +} + +// Upstream 0.70.0 #4094 `CodexAliasedModelPricingTests`. +#[test] +fn codex_cost_prices_daybreak_aliases_and_cyber_bundled_fallback() { + let cost = |model: &str, writes: u64| { + CostUsagePricing::codex_cost_usd_with_cache_write(model, 100, 10, writes, 5).unwrap() + }; + // Cyber rates per token: $12.50 input, $1.25 cached input, $75 output per 1M. + let cyber = 90.0 * 1.25e-5 + 10.0 * 1.25e-6 + 5.0 * 7.5e-5; + assert!((cost("gpt-5.6-cyber", 0) - cyber).abs() < 1e-12); + assert!((cost("gpt-5.5-cyber", 0) - cyber).abs() < 1e-12); + let expected_write = 70.0 * 1.25e-5 + 10.0 * 1.25e-6 + 20.0 * 1.5625e-5 + 5.0 * 7.5e-5; + assert!((cost("gpt-5.6-cyber", 20) - expected_write).abs() < 1e-12); + // gpt-5.5-cyber lists no cache-write rate, so its writes bill as input. + assert!((cost("gpt-5.5-cyber", 20) - cyber).abs() < 1e-12); + assert!((cost("gpt-daybreak-blue-latest", 0) - cost("gpt-5.6-sol", 0)).abs() < 1e-12); + assert!((cost("gpt-daybreak-red-latest", 0) - cyber).abs() < 1e-12); +} + +// Upstream 0.70.0 #4094 `CodexSolHistoricalPricingTests`. Windows prices by +// usage day, so 2026-08-20 stands in for upstream's `cutoff - 1s`. +#[test] +fn sol_keeps_historical_rates_before_its_august_repricing() { + use chrono::NaiveDate; + use models_dev_pricing::ModelsDevPricingSnapshot; + + let catalog = ModelsDevPricingSnapshot::from_catalog_json_for_tests( + r#"{"openai":{"id":"openai","models":{"gpt-5.6-sol":{ + "id":"gpt-5.6-sol","cost":{"input":4,"cache_read":0.4,"cache_write":5,"output":20} + }}}}"#, + ) + .expect("catalog fixture"); + let empty = ModelsDevPricingSnapshot::from_catalog_json_for_tests("{}").expect("empty"); + let last_historical_day = NaiveDate::from_ymd_opt(2026, 8, 20).unwrap(); + let cutoff = NaiveDate::from_ymd_opt(2026, 8, 21).unwrap(); + for model in ["gpt-5.6-sol", "gpt-5.6"] { + for input in [100_u64, 272_001] { + let long = input > 272_000; + for (day, historical) in [(last_historical_day, true), (cutoff, false)] { + // Per 1M tokens; cache reads bill at 0.1x and writes at 1.25x input. + let (input_rate, output_rate) = match (historical, long) { + (true, false) => (5.0, 30.0), + (true, true) => (10.0, 45.0), + (false, false) => (4.0, 20.0), + (false, true) => (8.0, 30.0), + }; + let expected = ((input - 30) as f64 * input_rate + + input_rate + + 25.0 * input_rate + + 5.0 * output_rate) + / 1_000_000.0; + for snapshot in [Some(&catalog), Some(&empty)] { + let standard = + CostUsagePricing::codex_cost_usd_at_date_with_cache_write_and_pricing_snapshot( + model, input, 10, 20, 5, day, snapshot, + ) + .unwrap(); + assert!((standard - expected).abs() < 1e-12, "{model} {input} {day}"); + } + // Windows' Fast lane has no cache-write input; price it without writes. + let expected_without_writes = + ((input - 10) as f64 * input_rate + input_rate + 5.0 * output_rate) + / 1_000_000.0; + let fast = CostUsagePricing::codex_fast_cost_usd_at_date(model, input, 10, 5, day); + if long { + assert!(fast.is_none(), "{model} {input} {day}"); + } else { + let fast = fast.unwrap(); + assert!((fast - expected_without_writes * 2.0).abs() < 1e-12); + } + } + } + } + assert!( + CostUsagePricing::codex_cost_usd_at_date_with_cache_write_and_pricing_snapshot( + "fixture-unknown-model", + 100, + 0, + 0, + 5, + cutoff, + Some(&catalog), + ) + .is_none() + ); +} + +#[test] +fn gpt56_bundled_rates_price_cache_writes_at_125_percent() { + let cost = |model: &str, input: u64, output: u64| { + CostUsagePricing::codex_cost_usd_with_cache_write(model, input, 10, 20, output).unwrap() + }; + let sol = 70.0 * 4e-6 + 10.0 * 4e-7 + 20.0 * 5e-6 + 5.0 * 2e-5; + assert!((cost("gpt-5.6-sol", 100, 5) - sol).abs() < 1e-12); + // Long-context (>272K) rates apply to the entire request. Total input + // contains 10 cached, 20 cache-write, and 271,971 ordinary input tokens. + for (model, expected) in [ + ( + "gpt-5.6-sol", + 271_971.0 * 8e-6 + 10.0 * 8e-7 + 20.0 * 1e-5 + 10.0 * 3e-5, + ), + ( + "gpt-5.6-terra", + 271_971.0 * 4e-6 + 10.0 * 4e-7 + 20.0 * 5e-6 + 10.0 * 1.8e-5, + ), + ( + "gpt-5.6-luna", + 271_971.0 * 4e-7 + 10.0 * 4e-8 + 20.0 * 5e-7 + 10.0 * 1.8e-6, + ), + ] { + assert!( + (cost(model, 272_001, 10) - expected).abs() < 1e-12, + "{model}" + ); + } +} From 4155a498ac3a2717a111936e78ff3851f4ae4ac2 Mon Sep 17 00:00:00 2001 From: RCD <90105158+Finesssee@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:49:01 +0700 Subject: [PATCH 2/4] Price Pi and workspace Codex usage at the rates of its own day Upstream prices Pi rows at their timestamp and builds project usage from the dated cost cache, so pre-repricing usage keeps the rates it was billed at. Pi rows use their UTC timestamp date; the workspace index uses its local day keys like the cost scanners. --- rust/src/codex_workspaces/indexer.rs | 27 ++++++++++++- rust/src/pi_session_cost.rs | 60 ++++++++++++++++++++++++---- 2 files changed, 78 insertions(+), 9 deletions(-) diff --git a/rust/src/codex_workspaces/indexer.rs b/rust/src/codex_workspaces/indexer.rs index 993b695412..95fa8762cc 100644 --- a/rust/src/codex_workspaces/indexer.rs +++ b/rust/src/codex_workspaces/indexer.rs @@ -483,7 +483,7 @@ fn index_one_file(path: &Path, range: &CostUsageDayRange) -> Option day_entry.1 = day_entry.1.saturating_add(cached); day_entry.2 = day_entry.2.saturating_add(output); - match CostUsagePricing::codex_cost_usd(&model, input, cached, output) { + match codex_cost_on_day(&record.day_key, &model, input, cached, output) { Some(usd) => cost.known_usd += usd, None => cost.unknown_tokens = cost.unknown_tokens.saturating_add(tokens), } @@ -526,7 +526,7 @@ fn merge_daily( let tokens = input.saturating_add(*output); acc.total_tokens = acc.total_tokens.saturating_add(tokens); acc.cached_input_tokens = acc.cached_input_tokens.saturating_add(*cached); - match CostUsagePricing::codex_cost_usd(model, *input, *cached, *output) { + match codex_cost_on_day(day, model, *input, *cached, *output) { Some(usd) => acc.known_usd += usd, None => acc.unknown_tokens = acc.unknown_tokens.saturating_add(tokens), } @@ -534,6 +534,15 @@ fn merge_daily( } } +/// Prices at the usage day's rates, like the cost scanners, so usage from +/// before a model's repricing keeps the rates it was billed at. +fn codex_cost_on_day(day: &str, model: &str, input: u64, cached: u64, output: u64) -> Option { + match CostUsageDayRange::parse_day_key(day) { + Some(day) => CostUsagePricing::codex_cost_usd_at_date(model, input, cached, output, day), + None => CostUsagePricing::codex_cost_usd(model, input, cached, output), + } +} + fn list_session_files(root: &Path, range: &CostUsageDayRange) -> Vec { if !root.exists() { return Vec::new(); @@ -875,6 +884,20 @@ mod tests { ); } + #[test] + fn daily_costs_use_each_day_rates() { + let sol = |day: &str| { + let models = HashMap::from([("gpt-5.6-sol".to_string(), (100, 10, 5))]); + let mut daily = HashMap::new(); + merge_daily(&mut daily, &HashMap::from([(day.to_string(), models)])); + daily[day].known_usd + }; + let historical = 90.0 * 5e-6 + 10.0 * 5e-7 + 5.0 * 3e-5; + let current = 90.0 * 4e-6 + 10.0 * 4e-7 + 5.0 * 2e-5; + assert!((sol("2026-08-20") - historical).abs() < 1e-12); + assert!((sol("2026-08-21") - current).abs() < 1e-12); + } + #[test] fn foreign_user_version_is_rejected() { let tmp = TempDir::new().unwrap(); diff --git a/rust/src/pi_session_cost.rs b/rust/src/pi_session_cost.rs index a9f8f1b82c..198157b7f7 100644 --- a/rust/src/pi_session_cost.rs +++ b/rust/src/pi_session_cost.rs @@ -547,13 +547,31 @@ fn parse_pi_assistant_entry_any(value: &Value) -> Option { return None; } + let timestamp = entry_timestamp(value); let (cost, pricing_known) = match mapped { - PiMappedProvider::Codex => match CostUsagePricing::codex_cost_usd_with_cache_write( - &model, - input, - cache_read, - cache_create, - output, + // Upstream prices each row at its timestamp, so usage from before a + // model's repricing keeps the rates it was billed at. + PiMappedProvider::Codex => match timestamp.map_or_else( + || { + CostUsagePricing::codex_cost_usd_with_cache_write( + &model, + input, + cache_read, + cache_create, + output, + ) + }, + |ts| { + CostUsagePricing::codex_cost_usd_at_date_with_cache_write_and_pricing_snapshot( + &model, + input, + cache_read, + cache_create, + output, + ts.date_naive(), + None, + ) + }, ) { Some(cost) => (cost, true), None => (0.0, false), @@ -597,7 +615,7 @@ fn parse_pi_assistant_entry_any(value: &Value) -> Option { }; Some(PiEntry { - timestamp: entry_timestamp(value), + timestamp, provider: mapped, model, input, @@ -692,6 +710,34 @@ mod tests { assert!(entry.pricing_known); } + #[test] + fn prices_codex_rows_at_their_utc_timestamp() { + let cost = |timestamp: Option<&str>| { + let mut raw = serde_json::json!({ + "id": "sol-1", "role": "assistant", "provider": "openai-codex", + "model": "gpt-5.6-sol", + "usage": { "input": 100, "output": 5, "cacheRead": 10, "cacheWrite": 20 } + }); + if let Some(timestamp) = timestamp { + raw["timestamp"] = timestamp.into(); + } + parse_pi_assistant_entry(&raw, PiMappedProvider::Codex) + .unwrap() + .cost + }; + let historical = 70.0 * 5e-6 + 10.0 * 5e-7 + 20.0 * 6.25e-6 + 5.0 * 3e-5; + let current = 70.0 * 4e-6 + 10.0 * 4e-7 + 20.0 * 5e-6 + 5.0 * 2e-5; + for (timestamp, expected) in [ + (Some("2026-07-10T12:00:00Z"), historical), + (Some("2026-08-20T23:59:59Z"), historical), + (Some("2026-08-21T00:00:00Z"), current), + (Some("2026-09-10T12:00:00Z"), current), + (None, current), + ] { + assert!((cost(timestamp) - expected).abs() < 1e-12, "{timestamp:?}"); + } + } + #[test] fn unknown_model_keeps_tokens_and_marks_pricing_incomplete() { let raw = serde_json::json!({ From 1a40932172443fbbde461b60f169879f00b32a9b Mon Sep 17 00:00:00 2001 From: RCD <90105158+Finesssee@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:53:39 +0700 Subject: [PATCH 3/4] Bill GPT-5.4 and GPT-5.5 above 272K at long-context rates like upstream Upstream prices the whole GPT-5.4 request at $5/$22.50 (cache read $0.50) and GPT-5.5 at $10/$45 (cache read $1) once input passes 272K tokens. Fast mode keeps its 272K cutoff for both models. --- rust/src/codex_costs.rs | 17 +++++++++---- rust/src/core/cost_pricing.rs | 24 +++++++++++++++---- rust/src/core/cost_pricing_tests.rs | 37 +++++++++++++++++++++++++++++ 3 files changed, 70 insertions(+), 8 deletions(-) diff --git a/rust/src/codex_costs.rs b/rust/src/codex_costs.rs index cf80594c02..36b706bc7f 100644 --- a/rust/src/codex_costs.rs +++ b/rust/src/codex_costs.rs @@ -653,12 +653,21 @@ mod tests { #[test] fn test_codex_pricing_uses_gpt55_standard_short_context_rates() { - let cost = codex_cost_usd("gpt-5.5", 1_000_000, 400_000, 1_000_000); + let cost = codex_cost_usd("gpt-5.5", 200_000, 80_000, 100_000); // GPT-5.5 standard short-context pricing: - // 600k non-cached input at $5/M, 400k cached input at $0.50/M, - // and 1M output at $30/M. - assert!((cost - 33.20).abs() < 0.01); + // 120k non-cached input at $5/M, 80k cached input at $0.50/M, + // and 100k output at $30/M. + assert!((cost - 3.64).abs() < 1e-9); + } + + #[test] + fn test_codex_pricing_bills_whole_gpt55_request_at_long_context_rates() { + let cost = codex_cost_usd("gpt-5.5", 1_000_000, 400_000, 1_000_000); + + // Above 272K input the whole request bills at $10/M input, + // $1/M cached input and $45/M output. + assert!((cost - 51.40).abs() < 1e-9); } #[test] diff --git a/rust/src/core/cost_pricing.rs b/rust/src/core/cost_pricing.rs index 78cf0ab4b3..cd93c43bb7 100755 --- a/rust/src/core/cost_pricing.rs +++ b/rust/src/core/cost_pricing.rs @@ -220,7 +220,8 @@ static CODEX_PRICING: LazyLock> = LazyLock:: }, ); - // GPT-5.4 pricing (updated to match upstream 0.22) + // GPT-5.4 pricing (updated to match upstream 0.22). Like upstream, the + // whole request bills 2x input / 1.5x output above 272K input tokens. m.insert( "gpt-5.4", CodexPricing { @@ -229,7 +230,12 @@ static CODEX_PRICING: LazyLock> = LazyLock:: cache_read_input_cost_per_token: 2.5e-7, cache_write_input_cost_per_token: None, display_label: None, - long_context: None, + long_context: Some(CodexLongContextRates { + input_cost_per_token: 5e-6, + output_cost_per_token: 2.25e-5, + cache_read_input_cost_per_token: 5e-7, + cache_write_input_cost_per_token: None, + }), }, ); m.insert( @@ -240,7 +246,12 @@ static CODEX_PRICING: LazyLock> = LazyLock:: cache_read_input_cost_per_token: 2.5e-7, cache_write_input_cost_per_token: None, display_label: None, - long_context: None, + long_context: Some(CodexLongContextRates { + input_cost_per_token: 5e-6, + output_cost_per_token: 2.25e-5, + cache_read_input_cost_per_token: 5e-7, + cache_write_input_cost_per_token: None, + }), }, ); @@ -312,7 +323,12 @@ static CODEX_PRICING: LazyLock> = LazyLock:: cache_read_input_cost_per_token: 5e-7, cache_write_input_cost_per_token: None, display_label: None, - long_context: None, + long_context: Some(CodexLongContextRates { + input_cost_per_token: 1e-5, + output_cost_per_token: 4.5e-5, + cache_read_input_cost_per_token: 1e-6, + cache_write_input_cost_per_token: None, + }), }, ); m.insert( diff --git a/rust/src/core/cost_pricing_tests.rs b/rust/src/core/cost_pricing_tests.rs index 7d8dad0230..1e72b57575 100644 --- a/rust/src/core/cost_pricing_tests.rs +++ b/rust/src/core/cost_pricing_tests.rs @@ -725,3 +725,40 @@ fn gpt56_bundled_rates_price_cache_writes_at_125_percent() { ); } } + +#[test] +fn gpt54_and_gpt55_bill_the_whole_request_at_long_context_rates_above_272k() { + let cost = |model: &str, input: u64| { + CostUsagePricing::codex_cost_usd(model, input, 1_000, 100).unwrap() + }; + for (model, standard, long) in [ + ("gpt-5.4", (2.5e-6, 2.5e-7, 1.5e-5), (5e-6, 5e-7, 2.25e-5)), + ( + "gpt-5.4-codex", + (2.5e-6, 2.5e-7, 1.5e-5), + (5e-6, 5e-7, 2.25e-5), + ), + ("gpt-5.5", (5e-6, 5e-7, 3e-5), (1e-5, 1e-6, 4.5e-5)), + ( + "openai/gpt-5.5-2026-04-23", + (5e-6, 5e-7, 3e-5), + (1e-5, 1e-6, 4.5e-5), + ), + ] { + let price = |(input_rate, cached_rate, output_rate): (f64, f64, f64), input: u64| { + (input - 1_000) as f64 * input_rate + 1_000.0 * cached_rate + 100.0 * output_rate + }; + // The 272K boundary itself still bills standard rates. + assert!( + (cost(model, 272_000) - price(standard, 272_000)).abs() < 1e-12, + "{model}" + ); + assert!( + (cost(model, 272_001) - price(long, 272_001)).abs() < 1e-12, + "{model}" + ); + } + // Fast mode keeps its 272K cutoff for these models. + assert!(CostUsagePricing::codex_fast_cost_usd("gpt-5.5-priority", 272_001, 0, 1).is_none()); + assert!(CostUsagePricing::codex_fast_cost_usd("gpt-5.4-fast", 272_000, 0, 1).is_some()); +} From 6f6e9292d748ce5ea417367fb0a53ceb3162616c Mon Sep 17 00:00:00 2001 From: RCD <90105158+Finesssee@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:54:56 +0700 Subject: [PATCH 4/4] Document Codex Daybreak, Cyber and Sol repricing rules --- docs/PROVIDERS.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/PROVIDERS.md b/docs/PROVIDERS.md index 88043bd617..d793331d25 100644 --- a/docs/PROVIDERS.md +++ b/docs/PROVIDERS.md @@ -83,6 +83,12 @@ The shared refresh interval controls automatic provider polling. `0` / Manual di Custom pricing overlays are exact-match overrides used only where the local spend contract has matching provider/model token evidence. Explicit zero rates mean free; omitted rate fields stay unknown. The Usage & Spend surface keeps provenance/coverage visible, preserves cost-only model rows when token coverage is partial, and can Copy JSON or save the same JSON contract through the native file picker. +### Codex model pricing + +Codex usage is priced from the bundled OpenAI rate table first and the models.dev catalog second. OpenAI's [Daybreak aliases](https://developers.openai.com/api/docs/pricing) resolve like the unsuffixed `gpt-5.6` alias: `gpt-daybreak-blue-latest` prices as `gpt-5.6-sol` and `gpt-daybreak-red-latest` as `gpt-5.6-cyber`. Recorded model names stay unchanged. `gpt-5.6-cyber` and `gpt-5.5-cyber` use the published Cyber rates of $12.50 input, $1.25 cached input and $75 output per 1M tokens. GPT-5.4, GPT-5.5 and GPT-5.6 bill the whole request at their long-context rates once input passes 272K tokens. + +Dated Codex usage keeps the prior GPT-5.6 Sol rates ($5 input, $30 output per 1M tokens) before **2026-08-21**, the repricing date in the [OpenAI changelog](https://developers.openai.com/api/docs/changelog). Current and undated usage use the current $4/$20 rates. Terra and Luna keep their separate 2026-07-30 cutoff. Windows compares calendar days rather than event instants: the Codex cost scanners and the workspace index use their local day keys, and Pi session rows use the UTC date of their timestamp. Custom pricing overlays keep precedence. + ### OpenCode, Codex quota, and local cost boundaries OpenCode-held OpenAI/Codex OAuth can be reused for **remote Codex account quota** only when the Codex provider's `External OAuth sources` setting is explicitly enabled. Native Codex credentials still take precedence, an explicit `CODEX_HOME` stays isolated, and external credentials remain read-only. This does **not** import ordinary OpenCode sessions into Codex token or spend totals. OpenCode Go's local SQLite reader remains scoped to its own `opencode-go` assistant records; OpenAI API-platform usage is a separate provider.