diff --git a/docs/PROVIDERS.md b/docs/PROVIDERS.md index 88043bd617..d793331d25 100644 --- a/docs/PROVIDERS.md +++ b/docs/PROVIDERS.md @@ -83,6 +83,12 @@ The shared refresh interval controls automatic provider polling. `0` / Manual di Custom pricing overlays are exact-match overrides used only where the local spend contract has matching provider/model token evidence. Explicit zero rates mean free; omitted rate fields stay unknown. The Usage & Spend surface keeps provenance/coverage visible, preserves cost-only model rows when token coverage is partial, and can Copy JSON or save the same JSON contract through the native file picker. +### Codex model pricing + +Codex usage is priced from the bundled OpenAI rate table first and the models.dev catalog second. OpenAI's [Daybreak aliases](https://developers.openai.com/api/docs/pricing) resolve like the unsuffixed `gpt-5.6` alias: `gpt-daybreak-blue-latest` prices as `gpt-5.6-sol` and `gpt-daybreak-red-latest` as `gpt-5.6-cyber`. Recorded model names stay unchanged. `gpt-5.6-cyber` and `gpt-5.5-cyber` use the published Cyber rates of $12.50 input, $1.25 cached input and $75 output per 1M tokens. GPT-5.4, GPT-5.5 and GPT-5.6 bill the whole request at their long-context rates once input passes 272K tokens. + +Dated Codex usage keeps the prior GPT-5.6 Sol rates ($5 input, $30 output per 1M tokens) before **2026-08-21**, the repricing date in the [OpenAI changelog](https://developers.openai.com/api/docs/changelog). Current and undated usage use the current $4/$20 rates. Terra and Luna keep their separate 2026-07-30 cutoff. Windows compares calendar days rather than event instants: the Codex cost scanners and the workspace index use their local day keys, and Pi session rows use the UTC date of their timestamp. Custom pricing overlays keep precedence. + ### OpenCode, Codex quota, and local cost boundaries OpenCode-held OpenAI/Codex OAuth can be reused for **remote Codex account quota** only when the Codex provider's `External OAuth sources` setting is explicitly enabled. Native Codex credentials still take precedence, an explicit `CODEX_HOME` stays isolated, and external credentials remain read-only. This does **not** import ordinary OpenCode sessions into Codex token or spend totals. OpenCode Go's local SQLite reader remains scoped to its own `opencode-go` assistant records; OpenAI API-platform usage is a separate provider. diff --git a/rust/src/codex_costs.rs b/rust/src/codex_costs.rs index cf80594c02..36b706bc7f 100644 --- a/rust/src/codex_costs.rs +++ b/rust/src/codex_costs.rs @@ -653,12 +653,21 @@ mod tests { #[test] fn test_codex_pricing_uses_gpt55_standard_short_context_rates() { - let cost = codex_cost_usd("gpt-5.5", 1_000_000, 400_000, 1_000_000); + let cost = codex_cost_usd("gpt-5.5", 200_000, 80_000, 100_000); // GPT-5.5 standard short-context pricing: - // 600k non-cached input at $5/M, 400k cached input at $0.50/M, - // and 1M output at $30/M. - assert!((cost - 33.20).abs() < 0.01); + // 120k non-cached input at $5/M, 80k cached input at $0.50/M, + // and 100k output at $30/M. + assert!((cost - 3.64).abs() < 1e-9); + } + + #[test] + fn test_codex_pricing_bills_whole_gpt55_request_at_long_context_rates() { + let cost = codex_cost_usd("gpt-5.5", 1_000_000, 400_000, 1_000_000); + + // Above 272K input the whole request bills at $10/M input, + // $1/M cached input and $45/M output. + assert!((cost - 51.40).abs() < 1e-9); } #[test] diff --git a/rust/src/codex_workspaces/indexer.rs b/rust/src/codex_workspaces/indexer.rs index 993b695412..95fa8762cc 100644 --- a/rust/src/codex_workspaces/indexer.rs +++ b/rust/src/codex_workspaces/indexer.rs @@ -483,7 +483,7 @@ fn index_one_file(path: &Path, range: &CostUsageDayRange) -> Option day_entry.1 = day_entry.1.saturating_add(cached); day_entry.2 = day_entry.2.saturating_add(output); - match CostUsagePricing::codex_cost_usd(&model, input, cached, output) { + match codex_cost_on_day(&record.day_key, &model, input, cached, output) { Some(usd) => cost.known_usd += usd, None => cost.unknown_tokens = cost.unknown_tokens.saturating_add(tokens), } @@ -526,7 +526,7 @@ fn merge_daily( let tokens = input.saturating_add(*output); acc.total_tokens = acc.total_tokens.saturating_add(tokens); acc.cached_input_tokens = acc.cached_input_tokens.saturating_add(*cached); - match CostUsagePricing::codex_cost_usd(model, *input, *cached, *output) { + match codex_cost_on_day(day, model, *input, *cached, *output) { Some(usd) => acc.known_usd += usd, None => acc.unknown_tokens = acc.unknown_tokens.saturating_add(tokens), } @@ -534,6 +534,15 @@ fn merge_daily( } } +/// Prices at the usage day's rates, like the cost scanners, so usage from +/// before a model's repricing keeps the rates it was billed at. +fn codex_cost_on_day(day: &str, model: &str, input: u64, cached: u64, output: u64) -> Option { + match CostUsageDayRange::parse_day_key(day) { + Some(day) => CostUsagePricing::codex_cost_usd_at_date(model, input, cached, output, day), + None => CostUsagePricing::codex_cost_usd(model, input, cached, output), + } +} + fn list_session_files(root: &Path, range: &CostUsageDayRange) -> Vec { if !root.exists() { return Vec::new(); @@ -875,6 +884,20 @@ mod tests { ); } + #[test] + fn daily_costs_use_each_day_rates() { + let sol = |day: &str| { + let models = HashMap::from([("gpt-5.6-sol".to_string(), (100, 10, 5))]); + let mut daily = HashMap::new(); + merge_daily(&mut daily, &HashMap::from([(day.to_string(), models)])); + daily[day].known_usd + }; + let historical = 90.0 * 5e-6 + 10.0 * 5e-7 + 5.0 * 3e-5; + let current = 90.0 * 4e-6 + 10.0 * 4e-7 + 5.0 * 2e-5; + assert!((sol("2026-08-20") - historical).abs() < 1e-12); + assert!((sol("2026-08-21") - current).abs() < 1e-12); + } + #[test] fn foreign_user_version_is_rejected() { let tmp = TempDir::new().unwrap(); diff --git a/rust/src/core/cost_pricing.rs b/rust/src/core/cost_pricing.rs index c4a8effebd..cd93c43bb7 100755 --- a/rust/src/core/cost_pricing.rs +++ b/rust/src/core/cost_pricing.rs @@ -16,6 +16,9 @@ pub struct CodexLongContextRates { pub input_cost_per_token: f64, pub output_cost_per_token: f64, pub cache_read_input_cost_per_token: f64, + /// `None` falls back to the standard cache-write rate, then to this + /// tier's input rate. + pub cache_write_input_cost_per_token: Option, } /// Codex (OpenAI) model pricing #[derive(Debug, Clone, Copy)] @@ -26,6 +29,9 @@ pub struct CodexPricing { pub output_cost_per_token: f64, /// Cost per cached input token in USD pub cache_read_input_cost_per_token: f64, + /// Cost per cache-write input token in USD; `None` bills cache writes at + /// the input rate. + pub cache_write_input_cost_per_token: Option, /// Optional display label override (e.g. "Research Preview") pub display_label: Option<&'static str>, /// Whole-request rates above the Codex long-context threshold. @@ -65,6 +71,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -75,6 +82,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -85,6 +93,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2.5e-7, output_cost_per_token: 2e-6, cache_read_input_cost_per_token: 2.5e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -95,6 +104,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 5e-8, output_cost_per_token: 4e-7, cache_read_input_cost_per_token: 5e-9, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -105,6 +115,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.5e-5, output_cost_per_token: 1.2e-4, cache_read_input_cost_per_token: 1.5e-5, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -115,6 +126,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -125,6 +137,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -135,6 +148,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.25e-6, output_cost_per_token: 1e-5, cache_read_input_cost_per_token: 1.25e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -145,6 +159,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2.5e-7, output_cost_per_token: 2e-6, cache_read_input_cost_per_token: 2.5e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -155,6 +170,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.75e-6, output_cost_per_token: 1.4e-5, cache_read_input_cost_per_token: 1.75e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -165,6 +181,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.75e-6, output_cost_per_token: 1.4e-5, cache_read_input_cost_per_token: 1.75e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -175,6 +192,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2.1e-5, output_cost_per_token: 1.68e-4, cache_read_input_cost_per_token: 2.1e-5, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -185,6 +203,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 1.75e-6, output_cost_per_token: 1.4e-5, cache_read_input_cost_per_token: 1.75e-7, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -195,20 +214,28 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 0.0, output_cost_per_token: 0.0, cache_read_input_cost_per_token: 0.0, + cache_write_input_cost_per_token: None, display_label: Some("Research Preview"), long_context: None, }, ); - // GPT-5.4 pricing (updated to match upstream 0.22) + // GPT-5.4 pricing (updated to match upstream 0.22). Like upstream, the + // whole request bills 2x input / 1.5x output above 272K input tokens. m.insert( "gpt-5.4", CodexPricing { input_cost_per_token: 2.5e-6, output_cost_per_token: 1.5e-5, cache_read_input_cost_per_token: 2.5e-7, + cache_write_input_cost_per_token: None, display_label: None, - long_context: None, + long_context: Some(CodexLongContextRates { + input_cost_per_token: 5e-6, + output_cost_per_token: 2.25e-5, + cache_read_input_cost_per_token: 5e-7, + cache_write_input_cost_per_token: None, + }), }, ); m.insert( @@ -217,8 +244,14 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2.5e-6, output_cost_per_token: 1.5e-5, cache_read_input_cost_per_token: 2.5e-7, + cache_write_input_cost_per_token: None, display_label: None, - long_context: None, + long_context: Some(CodexLongContextRates { + input_cost_per_token: 5e-6, + output_cost_per_token: 2.25e-5, + cache_read_input_cost_per_token: 5e-7, + cache_write_input_cost_per_token: None, + }), }, ); @@ -229,6 +262,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 7.5e-7, output_cost_per_token: 4.5e-6, cache_read_input_cost_per_token: 7.5e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -239,6 +273,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 7.5e-7, output_cost_per_token: 4.5e-6, cache_read_input_cost_per_token: 7.5e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -251,6 +286,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2e-7, output_cost_per_token: 1.25e-6, cache_read_input_cost_per_token: 2e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -261,6 +297,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 2e-7, output_cost_per_token: 1.25e-6, cache_read_input_cost_per_token: 2e-8, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -273,6 +310,7 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 3e-5, output_cost_per_token: 1.8e-4, cache_read_input_cost_per_token: 3e-5, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, @@ -283,8 +321,14 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 5e-6, output_cost_per_token: 3e-5, cache_read_input_cost_per_token: 5e-7, + cache_write_input_cost_per_token: None, display_label: None, - long_context: None, + long_context: Some(CodexLongContextRates { + input_cost_per_token: 1e-5, + output_cost_per_token: 4.5e-5, + cache_read_input_cost_per_token: 1e-6, + cache_write_input_cost_per_token: None, + }), }, ); m.insert( @@ -293,65 +337,70 @@ static CODEX_PRICING: LazyLock> = LazyLock:: input_cost_per_token: 3e-5, output_cost_per_token: 1.8e-4, cache_read_input_cost_per_token: 3e-5, + cache_write_input_cost_per_token: None, display_label: None, long_context: None, }, ); + // GPT-5.6 Sol/Terra/Luna (OpenAI pricing page and model cards), in + // upstream `gpt56Pricing` order (input, cache read, cache write, output). + // Above 272K input tokens the whole request bills 2x input / 1.5x output; + // cache writes bill at 1.25x uncached input. Sol was repriced from $5/$30 + // to $4/$20 on 2026-08-21. Dated usage before a model's repricing keeps + // the rates in `codex_pricing::codex_historical_pricing`. m.insert( "gpt-5.6-sol", - CodexPricing { - input_cost_per_token: 5e-6, - output_cost_per_token: 3e-5, - cache_read_input_cost_per_token: 5e-7, - display_label: None, - long_context: Some(CodexLongContextRates { - input_cost_per_token: 1e-5, - output_cost_per_token: 4.5e-5, - cache_read_input_cost_per_token: 1e-6, - }), - }, + codex_pricing::gpt56_pricing((4e-6, 4e-7, 5e-6, 2e-5), (8e-6, 8e-7, 1e-5, 3e-5)), ); m.insert( "gpt-5.6-terra", + codex_pricing::gpt56_pricing((2e-6, 2e-7, 2.5e-6, 1.2e-5), (4e-6, 4e-7, 5e-6, 1.8e-5)), + ); + m.insert( + "gpt-5.6-luna", + codex_pricing::gpt56_pricing((2e-7, 2e-8, 2.5e-7, 1.2e-6), (4e-7, 4e-8, 5e-7, 1.8e-6)), + ); + // Daybreak Cyber models (OpenAI pricing page). No long-context tier is + // published, and gpt-5.5-cyber lists no cache-write rate, so its writes + // bill at the input rate. + m.insert( + "gpt-5.6-cyber", CodexPricing { - input_cost_per_token: 2e-6, - output_cost_per_token: 1.2e-5, - cache_read_input_cost_per_token: 2e-7, + input_cost_per_token: 1.25e-5, + output_cost_per_token: 7.5e-5, + cache_read_input_cost_per_token: 1.25e-6, + cache_write_input_cost_per_token: Some(1.5625e-5), display_label: None, - long_context: Some(CodexLongContextRates { - input_cost_per_token: 4e-6, - output_cost_per_token: 1.8e-5, - cache_read_input_cost_per_token: 4e-7, - }), + long_context: None, }, ); m.insert( - "gpt-5.6-luna", + "gpt-5.5-cyber", CodexPricing { - input_cost_per_token: 2e-7, - output_cost_per_token: 1.2e-6, - cache_read_input_cost_per_token: 2e-8, + input_cost_per_token: 1.25e-5, + output_cost_per_token: 7.5e-5, + cache_read_input_cost_per_token: 1.25e-6, + cache_write_input_cost_per_token: None, display_label: None, - long_context: Some(CodexLongContextRates { - input_cost_per_token: 4e-7, - output_cost_per_token: 1.8e-6, - cache_read_input_cost_per_token: 4e-8, - }), + long_context: None, }, ); // GPT-6 Astra pricing (OpenAI model card and pricing table). - // Long-context rates apply to the whole request above 272K input tokens. + // Long-context rates apply to the whole request above 272K input tokens; + // cache writes bill at 1.25x uncached input in both tiers. m.insert( "gpt-6-astra", CodexPricing { input_cost_per_token: 1e-5, output_cost_per_token: 5e-5, cache_read_input_cost_per_token: 1e-6, + cache_write_input_cost_per_token: Some(1.25e-5), display_label: None, long_context: Some(CodexLongContextRates { input_cost_per_token: 2e-5, output_cost_per_token: 7.5e-5, cache_read_input_cost_per_token: 2e-6, + cache_write_input_cost_per_token: Some(2.5e-5), }), }, ); @@ -629,6 +678,14 @@ impl CostUsagePricing { trimmed = rest.to_string(); } + // OpenAI's Daybreak aliases currently point to Sol (blue) and Cyber + // (red). https://developers.openai.com/api/docs/pricing + match trimmed.as_str() { + "gpt-daybreak-blue-latest" => return "gpt-5.6-sol".to_string(), + "gpt-daybreak-red-latest" => return "gpt-5.6-cyber".to_string(), + _ => {} + } + // Check if base model (without -codex suffix) exists in pricing if let Some(idx) = trimmed.find("-codex") { let base = &trimmed[..idx]; @@ -722,7 +779,7 @@ impl CostUsagePricing { } /// Calculate Codex cost using the rates in effect on a historical usage day. - /// GPT-5.6 Terra/Luna were cut on 2026-07-30; Sol was unchanged. + /// GPT-5.6 Terra/Luna were cut on 2026-07-30 and Sol on 2026-08-21. pub fn codex_cost_usd_at_date( model: &str, input_tokens: u64, @@ -761,8 +818,8 @@ impl CostUsagePricing { /// Codex cost on a historical usage day when the prompt also wrote cache /// tokens. `input_tokens` is the inclusive prompt size: cache reads and - /// writes are subsets of it. The pre-cutoff GPT-5.6 Terra/Luna rates carry - /// their own 1.25x cache-write rate (upstream `codexHistoricalPricing`). + /// writes are subsets of it. A model's pre-repricing rates (upstream + /// `codexHistoricalPricing`) win over today's bundled and catalog rates. pub fn codex_cost_usd_at_date_with_cache_write_and_pricing_snapshot( model: &str, input_tokens: u64, @@ -773,29 +830,14 @@ impl CostUsagePricing { pricing_snapshot: Option<&models_dev_pricing::ModelsDevPricingSnapshot>, ) -> Option { let key = Self::normalize_codex_model(model); - let cutoff = NaiveDate::from_ymd_opt(2026, 7, 30).expect("valid pricing cutoff"); - if pricing_date < cutoff { - let long = input_tokens > codex_pricing::CODEX_LONG_CONTEXT_THRESHOLD; - // (input, cache read, cache write, output) per token. - let rates = match (key.as_str(), long) { - ("gpt-5.6-terra", false) => Some((2.5e-6, 2.5e-7, 3.125e-6, 1.5e-5)), - ("gpt-5.6-terra", true) => Some((5e-6, 5e-7, 6.25e-6, 2.25e-5)), - ("gpt-5.6-luna", false) => Some((1e-6, 1e-7, 1.25e-6, 6e-6)), - ("gpt-5.6-luna", true) => Some((2e-6, 2e-7, 2.5e-6, 9e-6)), - _ => None, - }; - if let Some((input_rate, cache_read_rate, cache_write_rate, output_rate)) = rates { - return Some(codex_pricing::codex_cost_from_rates_with_cache_write( - input_tokens, - cached_input_tokens, - cache_write_input_tokens, - output_tokens, - input_rate, - cache_read_rate, - cache_write_rate, - output_rate, - )); - } + if let Some(pricing) = codex_pricing::codex_historical_pricing(&key, pricing_date) { + return Some(codex_pricing::codex_cost_from_pricing( + &pricing, + input_tokens, + cached_input_tokens, + cache_write_input_tokens, + output_tokens, + )); } Self::codex_cost_usd_with_cache_write_and_pricing_snapshot( model, diff --git a/rust/src/core/cost_pricing/codex.rs b/rust/src/core/cost_pricing/codex.rs index 09116be6f4..90a31efad8 100644 --- a/rust/src/core/cost_pricing/codex.rs +++ b/rust/src/core/cost_pricing/codex.rs @@ -1,9 +1,103 @@ use super::super::{codex_routed_pricing, models_dev_pricing}; -use super::{CODEX_PRICING, CostUsagePricing}; +use super::{CODEX_PRICING, CodexLongContextRates, CodexPricing, CostUsagePricing}; +use chrono::NaiveDate; pub(super) const CODEX_LONG_CONTEXT_THRESHOLD: u64 = 272_000; -const CODEX_ASTRA_CACHE_WRITE_RATE: f64 = 1.25e-5; -const CODEX_ASTRA_LONG_CACHE_WRITE_RATE: f64 = 2.5e-5; + +/// GPT-5.6 rates per token in upstream `gpt56Pricing` order: +/// (input, cache read, cache write, output). +pub(super) type Gpt56Rates = (f64, f64, f64, f64); + +/// Upstream `gpt56Pricing`: standard rates plus the whole-request rates above +/// the 272K-token long-context threshold, each with its own cache-write rate. +pub(super) const fn gpt56_pricing(standard: Gpt56Rates, long_context: Gpt56Rates) -> CodexPricing { + let (input, cache_read, cache_write, output) = standard; + let (long_input, long_cache_read, long_cache_write, long_output) = long_context; + CodexPricing { + input_cost_per_token: input, + output_cost_per_token: output, + cache_read_input_cost_per_token: cache_read, + cache_write_input_cost_per_token: Some(cache_write), + display_label: None, + long_context: Some(CodexLongContextRates { + input_cost_per_token: long_input, + output_cost_per_token: long_output, + cache_read_input_cost_per_token: long_cache_read, + cache_write_input_cost_per_token: Some(long_cache_write), + }), + } +} + +/// Upstream `codexHistoricalPricing`: the rates a model billed at before its +/// repricing. GPT-5.6 Terra and Luna were cut on 2026-07-30 (Unix 1785369600) +/// and Sol on 2026-08-21 (Unix 1787270400). Windows keys usage by calendar +/// day, so the cutoff compares days rather than event instants. +pub(super) fn codex_historical_pricing(key: &str, pricing_date: NaiveDate) -> Option { + let ((year, month, day), standard, long_context) = match key { + "gpt-5.6-sol" => ( + (2026, 8, 21), + (5e-6, 5e-7, 6.25e-6, 3e-5), + (1e-5, 1e-6, 1.25e-5, 4.5e-5), + ), + "gpt-5.6-terra" => ( + (2026, 7, 30), + (2.5e-6, 2.5e-7, 3.125e-6, 1.5e-5), + (5e-6, 5e-7, 6.25e-6, 2.25e-5), + ), + "gpt-5.6-luna" => ( + (2026, 7, 30), + (1e-6, 1e-7, 1.25e-6, 6e-6), + (2e-6, 2e-7, 2.5e-6, 9e-6), + ), + _ => return None, + }; + let cutoff = NaiveDate::from_ymd_opt(year, month, day)?; + (pricing_date < cutoff).then(|| gpt56_pricing(standard, long_context)) +} + +/// Upstream `codexCostUSD(pricing:)` for one bundled entry. `input_tokens` is +/// the inclusive prompt size and selects the long-context tier. A long-context +/// cache write without its own rate falls back to the standard cache-write +/// rate, then to the tier's input rate. +pub(super) fn codex_cost_from_pricing( + pricing: &CodexPricing, + input_tokens: u64, + cached_input_tokens: u64, + cache_write_input_tokens: u64, + output_tokens: u64, +) -> f64 { + let long_context = pricing + .long_context + .filter(|_| input_tokens > CODEX_LONG_CONTEXT_THRESHOLD); + let (input_rate, cache_read_rate, cache_write_rate, output_rate) = match long_context { + Some(long) => ( + long.input_cost_per_token, + long.cache_read_input_cost_per_token, + long.cache_write_input_cost_per_token + .or(pricing.cache_write_input_cost_per_token) + .unwrap_or(long.input_cost_per_token), + long.output_cost_per_token, + ), + None => ( + pricing.input_cost_per_token, + pricing.cache_read_input_cost_per_token, + pricing + .cache_write_input_cost_per_token + .unwrap_or(pricing.input_cost_per_token), + pricing.output_cost_per_token, + ), + }; + codex_cost_from_rates_with_cache_write( + input_tokens, + cached_input_tokens, + cache_write_input_tokens, + output_tokens, + input_rate, + cache_read_rate, + cache_write_rate, + output_rate, + ) +} pub(super) fn codex_cost_from_rates( input_tokens: u64, @@ -110,46 +204,12 @@ impl CostUsagePricing { return None; } if let Some(pricing) = CODEX_PRICING.get(key.as_str()) { - let long = input_tokens > CODEX_LONG_CONTEXT_THRESHOLD; - let (input_rate, cache_read_rate, output_rate) = if long { - if let Some(long_context) = pricing.long_context { - ( - long_context.input_cost_per_token, - long_context.cache_read_input_cost_per_token, - long_context.output_cost_per_token, - ) - } else { - ( - pricing.input_cost_per_token, - pricing.cache_read_input_cost_per_token, - pricing.output_cost_per_token, - ) - } - } else { - ( - pricing.input_cost_per_token, - pricing.cache_read_input_cost_per_token, - pricing.output_cost_per_token, - ) - }; - let cache_write_rate = if key == "gpt-6-astra" { - if long { - CODEX_ASTRA_LONG_CACHE_WRITE_RATE - } else { - CODEX_ASTRA_CACHE_WRITE_RATE - } - } else { - input_rate - }; - return Some(codex_cost_from_rates_with_cache_write( + return Some(codex_cost_from_pricing( + pricing, input_tokens, cached_input_tokens, cache_write_input_tokens, output_tokens, - input_rate, - cache_read_rate, - cache_write_rate, - output_rate, )); } diff --git a/rust/src/core/cost_pricing_tests.rs b/rust/src/core/cost_pricing_tests.rs index 106f3b09e0..1e72b57575 100644 --- a/rust/src/core/cost_pricing_tests.rs +++ b/rust/src/core/cost_pricing_tests.rs @@ -159,7 +159,7 @@ fn test_gpt5_pro_cost() { #[test] fn test_gpt56_standard_pricing() { for (model, expected) in [ - ("gpt-5.6-sol", 0.0332), + ("gpt-5.6-sol", 0.02256), ("gpt-5.6-terra", 0.01328), ("gpt-5.6-luna", 0.001328), ] { @@ -171,7 +171,7 @@ fn test_gpt56_standard_pricing() { #[test] fn test_gpt56_long_context_pricing() { for (model, expected) in [ - ("gpt-5.6-sol", 45.272001), + ("gpt-5.6-sol", 30.2176008), ("gpt-5.6-terra", 18.1088004), ("gpt-5.6-luna", 1.81088004), ] { @@ -183,7 +183,7 @@ fn test_gpt56_long_context_pricing() { #[test] fn test_gpt56_context_threshold_is_exclusive() { for (model, expected) in [ - ("gpt-5.6-sol", 0.136), + ("gpt-5.6-sol", 0.1088), ("gpt-5.6-terra", 0.0544), ("gpt-5.6-luna", 0.00544), ] { @@ -491,14 +491,23 @@ fn gpt56_historical_terra_luna_rates_change_at_2026_07_30() { } #[test] -fn gpt56_historical_pricing_keeps_sol_unchanged() { +fn gpt56_historical_sol_rates_change_at_2026_08_21() { use chrono::NaiveDate; - let before = NaiveDate::from_ymd_opt(2026, 7, 29).unwrap(); - let current = CostUsagePricing::codex_cost_usd("gpt-5.6-sol", 100, 10, 5).unwrap(); - let historical = - CostUsagePricing::codex_cost_usd_at_date("gpt-5.6-sol", 100, 10, 5, before).unwrap(); - assert!((historical - current).abs() < f64::EPSILON); + // Sol keeps its own cutoff: still historical on Terra/Luna's cut day. + let historical = 90.0 * 5e-6 + 10.0 * 5e-7 + 5.0 * 3e-5; + let current = 90.0 * 4e-6 + 10.0 * 4e-7 + 5.0 * 2e-5; + for (date, expected) in [ + ((2026, 7, 30), historical), + ((2026, 8, 20), historical), + ((2026, 8, 21), current), + ] { + let day = NaiveDate::from_ymd_opt(date.0, date.1, date.2).unwrap(); + let cost = CostUsagePricing::codex_cost_usd_at_date("gpt-5.6-sol", 100, 10, 5, day); + assert!((cost.unwrap() - expected).abs() < 1e-12, "{day}"); + } + let undated = CostUsagePricing::codex_cost_usd("gpt-5.6-sol", 100, 10, 5).unwrap(); + assert!((undated - current).abs() < 1e-12); } #[test] @@ -583,3 +592,173 @@ fn gpt6_astra_unknown_models_fail_closed() { assert!(CostUsagePricing::codex_cost_usd(model, 1000, 0, 100).is_none()); } } + +#[test] +fn normalize_codex_model_maps_daybreak_aliases_and_cyber_ids() { + for (raw, expected) in [ + ("gpt-daybreak-blue-latest", "gpt-5.6-sol"), + ("openai/gpt-daybreak-blue-latest", "gpt-5.6-sol"), + ("gpt-daybreak-red-latest", "gpt-5.6-cyber"), + ("gpt-5.6-cyber", "gpt-5.6-cyber"), + ("gpt-5.5-cyber", "gpt-5.5-cyber"), + ] { + assert_eq!( + CostUsagePricing::normalize_codex_model(raw), + expected, + "{raw}" + ); + } +} + +// Upstream 0.70.0 #4094 `CodexAliasedModelPricingTests`. +#[test] +fn codex_cost_prices_daybreak_aliases_and_cyber_bundled_fallback() { + let cost = |model: &str, writes: u64| { + CostUsagePricing::codex_cost_usd_with_cache_write(model, 100, 10, writes, 5).unwrap() + }; + // Cyber rates per token: $12.50 input, $1.25 cached input, $75 output per 1M. + let cyber = 90.0 * 1.25e-5 + 10.0 * 1.25e-6 + 5.0 * 7.5e-5; + assert!((cost("gpt-5.6-cyber", 0) - cyber).abs() < 1e-12); + assert!((cost("gpt-5.5-cyber", 0) - cyber).abs() < 1e-12); + let expected_write = 70.0 * 1.25e-5 + 10.0 * 1.25e-6 + 20.0 * 1.5625e-5 + 5.0 * 7.5e-5; + assert!((cost("gpt-5.6-cyber", 20) - expected_write).abs() < 1e-12); + // gpt-5.5-cyber lists no cache-write rate, so its writes bill as input. + assert!((cost("gpt-5.5-cyber", 20) - cyber).abs() < 1e-12); + assert!((cost("gpt-daybreak-blue-latest", 0) - cost("gpt-5.6-sol", 0)).abs() < 1e-12); + assert!((cost("gpt-daybreak-red-latest", 0) - cyber).abs() < 1e-12); +} + +// Upstream 0.70.0 #4094 `CodexSolHistoricalPricingTests`. Windows prices by +// usage day, so 2026-08-20 stands in for upstream's `cutoff - 1s`. +#[test] +fn sol_keeps_historical_rates_before_its_august_repricing() { + use chrono::NaiveDate; + use models_dev_pricing::ModelsDevPricingSnapshot; + + let catalog = ModelsDevPricingSnapshot::from_catalog_json_for_tests( + r#"{"openai":{"id":"openai","models":{"gpt-5.6-sol":{ + "id":"gpt-5.6-sol","cost":{"input":4,"cache_read":0.4,"cache_write":5,"output":20} + }}}}"#, + ) + .expect("catalog fixture"); + let empty = ModelsDevPricingSnapshot::from_catalog_json_for_tests("{}").expect("empty"); + let last_historical_day = NaiveDate::from_ymd_opt(2026, 8, 20).unwrap(); + let cutoff = NaiveDate::from_ymd_opt(2026, 8, 21).unwrap(); + for model in ["gpt-5.6-sol", "gpt-5.6"] { + for input in [100_u64, 272_001] { + let long = input > 272_000; + for (day, historical) in [(last_historical_day, true), (cutoff, false)] { + // Per 1M tokens; cache reads bill at 0.1x and writes at 1.25x input. + let (input_rate, output_rate) = match (historical, long) { + (true, false) => (5.0, 30.0), + (true, true) => (10.0, 45.0), + (false, false) => (4.0, 20.0), + (false, true) => (8.0, 30.0), + }; + let expected = ((input - 30) as f64 * input_rate + + input_rate + + 25.0 * input_rate + + 5.0 * output_rate) + / 1_000_000.0; + for snapshot in [Some(&catalog), Some(&empty)] { + let standard = + CostUsagePricing::codex_cost_usd_at_date_with_cache_write_and_pricing_snapshot( + model, input, 10, 20, 5, day, snapshot, + ) + .unwrap(); + assert!((standard - expected).abs() < 1e-12, "{model} {input} {day}"); + } + // Windows' Fast lane has no cache-write input; price it without writes. + let expected_without_writes = + ((input - 10) as f64 * input_rate + input_rate + 5.0 * output_rate) + / 1_000_000.0; + let fast = CostUsagePricing::codex_fast_cost_usd_at_date(model, input, 10, 5, day); + if long { + assert!(fast.is_none(), "{model} {input} {day}"); + } else { + let fast = fast.unwrap(); + assert!((fast - expected_without_writes * 2.0).abs() < 1e-12); + } + } + } + } + assert!( + CostUsagePricing::codex_cost_usd_at_date_with_cache_write_and_pricing_snapshot( + "fixture-unknown-model", + 100, + 0, + 0, + 5, + cutoff, + Some(&catalog), + ) + .is_none() + ); +} + +#[test] +fn gpt56_bundled_rates_price_cache_writes_at_125_percent() { + let cost = |model: &str, input: u64, output: u64| { + CostUsagePricing::codex_cost_usd_with_cache_write(model, input, 10, 20, output).unwrap() + }; + let sol = 70.0 * 4e-6 + 10.0 * 4e-7 + 20.0 * 5e-6 + 5.0 * 2e-5; + assert!((cost("gpt-5.6-sol", 100, 5) - sol).abs() < 1e-12); + // Long-context (>272K) rates apply to the entire request. Total input + // contains 10 cached, 20 cache-write, and 271,971 ordinary input tokens. + for (model, expected) in [ + ( + "gpt-5.6-sol", + 271_971.0 * 8e-6 + 10.0 * 8e-7 + 20.0 * 1e-5 + 10.0 * 3e-5, + ), + ( + "gpt-5.6-terra", + 271_971.0 * 4e-6 + 10.0 * 4e-7 + 20.0 * 5e-6 + 10.0 * 1.8e-5, + ), + ( + "gpt-5.6-luna", + 271_971.0 * 4e-7 + 10.0 * 4e-8 + 20.0 * 5e-7 + 10.0 * 1.8e-6, + ), + ] { + assert!( + (cost(model, 272_001, 10) - expected).abs() < 1e-12, + "{model}" + ); + } +} + +#[test] +fn gpt54_and_gpt55_bill_the_whole_request_at_long_context_rates_above_272k() { + let cost = |model: &str, input: u64| { + CostUsagePricing::codex_cost_usd(model, input, 1_000, 100).unwrap() + }; + for (model, standard, long) in [ + ("gpt-5.4", (2.5e-6, 2.5e-7, 1.5e-5), (5e-6, 5e-7, 2.25e-5)), + ( + "gpt-5.4-codex", + (2.5e-6, 2.5e-7, 1.5e-5), + (5e-6, 5e-7, 2.25e-5), + ), + ("gpt-5.5", (5e-6, 5e-7, 3e-5), (1e-5, 1e-6, 4.5e-5)), + ( + "openai/gpt-5.5-2026-04-23", + (5e-6, 5e-7, 3e-5), + (1e-5, 1e-6, 4.5e-5), + ), + ] { + let price = |(input_rate, cached_rate, output_rate): (f64, f64, f64), input: u64| { + (input - 1_000) as f64 * input_rate + 1_000.0 * cached_rate + 100.0 * output_rate + }; + // The 272K boundary itself still bills standard rates. + assert!( + (cost(model, 272_000) - price(standard, 272_000)).abs() < 1e-12, + "{model}" + ); + assert!( + (cost(model, 272_001) - price(long, 272_001)).abs() < 1e-12, + "{model}" + ); + } + // Fast mode keeps its 272K cutoff for these models. + assert!(CostUsagePricing::codex_fast_cost_usd("gpt-5.5-priority", 272_001, 0, 1).is_none()); + assert!(CostUsagePricing::codex_fast_cost_usd("gpt-5.4-fast", 272_000, 0, 1).is_some()); +} diff --git a/rust/src/pi_session_cost.rs b/rust/src/pi_session_cost.rs index a9f8f1b82c..198157b7f7 100644 --- a/rust/src/pi_session_cost.rs +++ b/rust/src/pi_session_cost.rs @@ -547,13 +547,31 @@ fn parse_pi_assistant_entry_any(value: &Value) -> Option { return None; } + let timestamp = entry_timestamp(value); let (cost, pricing_known) = match mapped { - PiMappedProvider::Codex => match CostUsagePricing::codex_cost_usd_with_cache_write( - &model, - input, - cache_read, - cache_create, - output, + // Upstream prices each row at its timestamp, so usage from before a + // model's repricing keeps the rates it was billed at. + PiMappedProvider::Codex => match timestamp.map_or_else( + || { + CostUsagePricing::codex_cost_usd_with_cache_write( + &model, + input, + cache_read, + cache_create, + output, + ) + }, + |ts| { + CostUsagePricing::codex_cost_usd_at_date_with_cache_write_and_pricing_snapshot( + &model, + input, + cache_read, + cache_create, + output, + ts.date_naive(), + None, + ) + }, ) { Some(cost) => (cost, true), None => (0.0, false), @@ -597,7 +615,7 @@ fn parse_pi_assistant_entry_any(value: &Value) -> Option { }; Some(PiEntry { - timestamp: entry_timestamp(value), + timestamp, provider: mapped, model, input, @@ -692,6 +710,34 @@ mod tests { assert!(entry.pricing_known); } + #[test] + fn prices_codex_rows_at_their_utc_timestamp() { + let cost = |timestamp: Option<&str>| { + let mut raw = serde_json::json!({ + "id": "sol-1", "role": "assistant", "provider": "openai-codex", + "model": "gpt-5.6-sol", + "usage": { "input": 100, "output": 5, "cacheRead": 10, "cacheWrite": 20 } + }); + if let Some(timestamp) = timestamp { + raw["timestamp"] = timestamp.into(); + } + parse_pi_assistant_entry(&raw, PiMappedProvider::Codex) + .unwrap() + .cost + }; + let historical = 70.0 * 5e-6 + 10.0 * 5e-7 + 20.0 * 6.25e-6 + 5.0 * 3e-5; + let current = 70.0 * 4e-6 + 10.0 * 4e-7 + 20.0 * 5e-6 + 5.0 * 2e-5; + for (timestamp, expected) in [ + (Some("2026-07-10T12:00:00Z"), historical), + (Some("2026-08-20T23:59:59Z"), historical), + (Some("2026-08-21T00:00:00Z"), current), + (Some("2026-09-10T12:00:00Z"), current), + (None, current), + ] { + assert!((cost(timestamp) - expected).abs() < 1e-12, "{timestamp:?}"); + } + } + #[test] fn unknown_model_keeps_tokens_and_marks_pricing_incomplete() { let raw = serde_json::json!({