Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions apps/desktop-tauri/src-tauri/src/commands/usage_spend.rs
Original file line number Diff line number Diff line change
Expand Up @@ -460,7 +460,7 @@ fn build_usage_spend_summary(
if include_opencodex {
// OpenCodex is an enrichment source, never a standalone provider row.
// Publish routed subscriptions even when no live provider snapshot exists.
for id in ["codex", "opencodego", "kimi", "deepseek"] {
for id in ["codex", "opencodego", "kimi", "deepseek", "nous"] {
let contract = match id {
"codex" => None,
_ => Some(build_local_spend_contract(id, 30, true)),
Expand Down Expand Up @@ -536,7 +536,7 @@ fn build_usage_spend_summary(
refreshing: !pi_30_summary.history_coverage_established,
stale_updated_at: None,
},
"opencodego" | "kimi" | "deepseek" if include_opencodex => {
"opencodego" | "kimi" | "deepseek" | "nous" if include_opencodex => {
let seven = build_local_spend_contract(&provider_id, 7, true);
let thirty = build_local_spend_contract(&provider_id, 30, true);
if !thirty.imports.is_empty() {
Expand Down
2 changes: 1 addition & 1 deletion docs/PROVIDERS.md
Original file line number Diff line number Diff line change
Expand Up @@ -71,7 +71,7 @@ Optional status polling (provider status pages) is available via CLI `--status`

## Usage & Spend

Desktop tab id: `usageSpend`. The desktop and Overview consume one shared spend catalog. Codex and Claude local logs are first-class; routed OpenCodex usage enriches the matching Codex, OpenCode Go, Kimi, or DeepSeek subscription instead of appearing as a second fake provider. xAI and OpenRouter can publish exact provider-metered daily USD spend when their management credentials are configured, while Grok local sessions contribute tokens only. Missing spend sources remain unknown rather than becoming a false `$0`. Do not invent cross-currency totals.
Desktop tab id: `usageSpend`. The desktop and Overview consume one shared spend catalog. Codex and Claude local logs are first-class; routed OpenCodex usage enriches the matching Codex, OpenCode Go, Kimi, DeepSeek, or Nous Portal subscription instead of appearing as a second fake provider. Ledger rows with `provider: "nous"` are priced only from a custom pricing override (`nous/<model>` or the bare model id) or an exact Nous models.dev entry, never from another vendor's rates, and never touch Portal credit meters; a row missing input or output tokens, or consuming a cache lane its catalog entry does not price, stays unpriced. xAI and OpenRouter can publish exact provider-metered daily USD spend when their management credentials are configured, while Grok local sessions contribute tokens only. Missing spend sources remain unknown rather than becoming a false `$0`. Do not invent cross-currency totals.

### AWS Bedrock monitoring

Expand Down
63 changes: 35 additions & 28 deletions rust/src/core/cost_pricing/codex.rs
Original file line number Diff line number Diff line change
Expand Up @@ -168,42 +168,49 @@ impl CostUsagePricing {
Some(snapshot) => snapshot.lookup(provider_id, lookup_model),
None => models_dev_pricing::lookup(provider_id, lookup_model),
}?;
let use_tier = pricing
Some(Self::models_dev_cost_usd(
&pricing,
input_tokens,
cached_input_tokens,
cache_write_input_tokens,
output_tokens,
))
}

/// Upstream `codexCostUSD(pricing:)` for one models.dev entry.
/// `input_tokens` is the inclusive prompt size (cache reads and writes are
/// subsets of it) and also selects the long-context tier. A cache lane
/// without its own rate falls back to the tier's input rate.
pub(crate) fn models_dev_cost_usd(
pricing: &models_dev_pricing::DynamicModelPricing,
input_tokens: u64,
cached_input_tokens: u64,
cache_write_input_tokens: u64,
output_tokens: u64,
) -> f64 {
let long = pricing
.threshold_tokens
.is_some_and(|threshold| input_tokens > threshold);
let input_rate = if use_tier {
pricing
.input_cost_per_token_above_threshold
.unwrap_or(pricing.input_cost_per_token)
} else {
pricing.input_cost_per_token
};
let cache_read_rate = if use_tier {
pricing
.cache_read_input_cost_per_token_above_threshold
.or(pricing.cache_read_input_cost_per_token)
.unwrap_or(pricing.input_cost_per_token)
} else {
pricing
.cache_read_input_cost_per_token
.unwrap_or(pricing.input_cost_per_token)
};
let output_rate = if use_tier {
pricing
.output_cost_per_token_above_threshold
.unwrap_or(pricing.output_cost_per_token)
} else {
pricing.output_cost_per_token
};
Some(codex_cost_from_rates_with_cache_write(
let above = |rate: Option<f64>| rate.filter(|_| long);
let input_rate = above(pricing.input_cost_per_token_above_threshold)
.unwrap_or(pricing.input_cost_per_token);
let output_rate = above(pricing.output_cost_per_token_above_threshold)
.unwrap_or(pricing.output_cost_per_token);
let cache_read_rate = above(pricing.cache_read_input_cost_per_token_above_threshold)
.or(pricing.cache_read_input_cost_per_token)
.unwrap_or(input_rate);
let cache_write_rate = above(pricing.cache_write_input_cost_per_token_above_threshold)
.or(pricing.cache_write_input_cost_per_token)
.unwrap_or(input_rate);
codex_cost_from_rates_with_cache_write(
input_tokens,
cached_input_tokens,
cache_write_input_tokens,
output_tokens,
input_rate,
cache_read_rate,
input_rate,
cache_write_rate,
output_rate,
))
)
}
}
70 changes: 70 additions & 0 deletions rust/src/core/cost_pricing_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -332,6 +332,76 @@ fn codex_routed_provider_returns_none_for_unknown_and_unrouted() {
assert!(codex_routed_pricing::codex_routed_provider("openai/gpt-5").is_none());
}

#[test]
fn native_codex_nous_prefix_stays_unpriced() {
// Upstream `codexModelsDevProviderIDs` has no `nous`: only OpenCodex
// ledger rows reach the Nous catalog.
let snapshot = crate::core::ModelsDevPricingSnapshot::from_catalog_json_for_tests(
r#"{"nous":{"models":{"z-ai/glm-5":{"id":"z-ai/glm-5","cost":{"input":2,"output":8}}}}}"#,
)
.expect("catalog");
assert!(codex_routed_pricing::codex_routed_provider("nous/z-ai/glm-5").is_none());
assert!(
CostUsagePricing::codex_cost_usd_with_pricing_snapshot(
"nous/z-ai/glm-5",
1_000,
0,
500,
Some(&snapshot)
)
.is_none()
);
}

#[test]
fn models_dev_rates_follow_upstream_codex_cost_semantics() {
let snapshot = crate::core::ModelsDevPricingSnapshot::from_catalog_json_for_tests(
r#"{"deepseek":{"models":{
"full":{"id":"full","cost":{"input":2,"output":8,"cache_read":0.5,"cache_write":3,
"context_over_200k":{"input":4,"output":16,"cache_read":1,"cache_write":6}}},
"bare":{"id":"bare","cost":{"input":2,"output":8,
"context_over_200k":{"input":4,"output":16}}}}}}"#,
)
.expect("catalog");
let full = snapshot.lookup_exact("deepseek", "full").expect("full");
let bare = snapshot.lookup_exact("deepseek", "bare").expect("bare");
let close = |actual: f64, expected: f64| {
assert!((actual - expected).abs() < 1e-12, "{actual} vs {expected}");
};

// Short context: each cache lane uses its own catalog rate.
close(
CostUsagePricing::models_dev_cost_usd(&full, 1_000, 200, 300, 100),
500.0 * 2e-6 + 200.0 * 0.5e-6 + 300.0 * 3e-6 + 100.0 * 8e-6,
);
// Inclusive input above 200k selects the long-context rate of every lane.
close(
CostUsagePricing::models_dev_cost_usd(&full, 200_001, 100_000, 50_000, 1_000),
50_001.0 * 4e-6 + 100_000.0 * 1e-6 + 50_000.0 * 6e-6 + 1_000.0 * 16e-6,
);
// A lane without a catalog rate falls back to the tier's input rate.
close(
CostUsagePricing::models_dev_cost_usd(&bare, 1_000, 200, 300, 100),
1_000.0 * 2e-6 + 100.0 * 8e-6,
);
close(
CostUsagePricing::models_dev_cost_usd(&bare, 200_001, 100_000, 50_000, 1_000),
200_001.0 * 4e-6 + 1_000.0 * 16e-6,
);
// Routed Codex rows share these rates.
close(
CostUsagePricing::codex_cost_usd_with_pricing_snapshot(
"deepseek/bare",
200_001,
100_000,
1_000,
Some(&snapshot),
)
.expect("routed"),
200_001.0 * 4e-6 + 1_000.0 * 16e-6,
);
}

#[test]
fn codex_routed_model_with_unknown_prefix_stays_unpriced() {
// An unknown provider/ prefix must NOT fall back to the OpenAI catalog
Expand Down
55 changes: 55 additions & 0 deletions rust/src/core/models_dev_pricing.rs
Original file line number Diff line number Diff line change
Expand Up @@ -89,6 +89,36 @@ mod tests {
);
}

#[test]
fn exact_lookup_matches_only_the_trimmed_key_or_model_id() {
let catalog = ModelsDevCatalog::decode(
r#"{
"Nous": {
"models": {
"z-ai/glm-5": {"id": "z-ai/glm-5", "cost": {"input": 1, "output": 2}},
"catalog-key": {"id": "deepseek/deepseek-v4", "cost": {"input": 3, "output": 4}},
"gpt-5": {"id": "gpt-5", "cost": {"input": 5, "output": 6}}
}
}
}"#,
)
.expect("catalog");
let input_rate = |model: &str| {
catalog
.lookup_exact(" NOUS ", model)
.map(|pricing| pricing.input_cost_per_token)
};

assert_eq!(input_rate(" z-ai/glm-5 "), Some(1e-6));
assert_eq!(input_rate("deepseek/deepseek-v4"), Some(3e-6));
assert_eq!(input_rate("Z-AI/GLM-5"), None);
// Aliases the fuzzy lookup resolves never supply an exact price.
for alias in ["z-ai/glm-5@20260101", "z-ai/glm-5-20260101", "openai/gpt-5"] {
assert!(catalog.lookup("nous", alias).is_some(), "{alias}");
assert_eq!(input_rate(alias), None, "{alias}");
}
}

#[test]
fn cache_artifact_is_versioned_and_expires_after_one_day() {
let catalog = ModelsDevCatalog::decode(
Expand Down Expand Up @@ -223,6 +253,15 @@ impl ModelsDevPricingSnapshot {
.and_then(|artifact| artifact.catalog.lookup(provider_id, model_id))
}

/// Exact-id lookup (upstream `exactModelID: true`): the trimmed id must
/// equal a catalog key or model id. No dated, `@`, or vendor-prefix alias
/// of another model can supply the price.
pub fn lookup_exact(&self, provider_id: &str, model_id: &str) -> Option<DynamicModelPricing> {
self.artifact
.as_ref()
.and_then(|artifact| artifact.catalog.lookup_exact(provider_id, model_id))
}

#[cfg(test)]
pub(crate) fn from_catalog_json_for_tests(json: &str) -> Option<Self> {
let catalog = ModelsDevCatalog::decode(json)?;
Expand Down Expand Up @@ -293,6 +332,22 @@ impl ModelsDevCatalog {
})
}

fn lookup_exact(&self, provider_id: &str, model_id: &str) -> Option<DynamicModelPricing> {
let provider = self.providers.get(&normalize_provider_id(provider_id))?;
let model_id = normalize_model_id(model_id);
provider
.models
.get(&model_id)
.and_then(DynamicModelPricing::from_model)
.or_else(|| {
provider.models.values().find_map(|model| {
(normalize_model_id(&model.id) == model_id)
.then(|| DynamicModelPricing::from_model(model))
.flatten()
})
})
}

fn is_plausible_refresh(&self) -> bool {
["openai", "anthropic"].into_iter().all(|provider_id| {
self.providers
Expand Down
29 changes: 22 additions & 7 deletions rust/src/spend_contract/opencodex.rs
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,9 @@ struct OpenCodexEntry {
}

mod cache;
mod nous;
#[cfg(test)]
mod nous_tests;

#[derive(Default)]
struct ModelAccumulator {
Expand Down Expand Up @@ -62,6 +65,7 @@ fn route_provider(provider: &str) -> RouteTarget {
"opencode-go" => RouteTarget::Subscription("opencodego"),
"kimi-coding" | "kimi-for-coding" => RouteTarget::Subscription("kimi"),
"deepseek" => RouteTarget::Subscription("deepseek"),
"nous" => RouteTarget::Subscription(nous::SUBSCRIPTION_ID),
"opencode-free" | "opencode" => RouteTarget::TokenOnly,
_ => RouteTarget::Unknown,
}
Expand Down Expand Up @@ -153,9 +157,9 @@ fn aggregate(
let pricing_snapshot = crate::core::pricing_snapshot();

for entry in &entries {
if let Some(conversation) = entry.conversation_id.as_ref() {
conversations.insert(conversation.clone());
}
// A row without a conversationId is its own session (upstream 0.68.0).
let session = entry.conversation_id.as_ref().unwrap_or(&entry.request_id);
conversations.insert(session.clone());
token_mix.input_tokens = add_optional(token_mix.input_tokens, entry.input_tokens);
token_mix.output_tokens = add_optional(token_mix.output_tokens, entry.output_tokens);
token_mix.cache_read_tokens =
Expand Down Expand Up @@ -231,7 +235,7 @@ fn aggregate(
if let Some(cost) = cost {
model.cost = Some(model.cost.unwrap_or(0.0) + cost);
}
model.custom_pricing |= custom.rates(&entry.provider, &entry.model).is_some();
model.custom_pricing |= has_custom_rates(entry, custom);
}

let mut model_rows: Vec<_> = models
Expand Down Expand Up @@ -322,6 +326,9 @@ fn entry_cost(
if !has_usage {
return None;
}
if route_entry(entry) == RouteTarget::Subscription(nous::SUBSCRIPTION_ID) {
return nous::cost(entry, custom, pricing_snapshot);
}
let input = entry.input_tokens.unwrap_or(0);
let output = entry.output_tokens.unwrap_or(0);
let cache_read = entry.cache_read_tokens.unwrap_or(0);
Expand Down Expand Up @@ -358,6 +365,14 @@ fn pricing_model(entry: &OpenCodexEntry) -> Option<String> {
}
}

/// Whether a custom override covers `entry`, resolved as its cost resolves it.
fn has_custom_rates(entry: &OpenCodexEntry, custom: &CustomPricing) -> bool {
if route_entry(entry) == RouteTarget::Subscription(nous::SUBSCRIPTION_ID) {
return nous::custom_rates(entry, custom).is_some();
}
custom.rates(&entry.provider, &entry.model).is_some()
}

fn provider_model_id(entry: &OpenCodexEntry, target: RouteTarget) -> String {
let model = entry.model.trim();
let Some((model_prefix, model_tail)) = model.split_once('/') else {
Expand Down Expand Up @@ -429,9 +444,9 @@ fn parse_line(line: &str) -> Option<OpenCodexEntry> {
.and_then(|object| nonnegative_u64(object.get("cacheCreationInputTokens"))),
reasoning_tokens: usage
.and_then(|object| nonnegative_u64(object.get("reasoningOutputTokens"))),
total_tokens: value
.get("totalTokens")
.and_then(|value| nonnegative_u64(Some(value))),
// The Hermes extractor reports the total inside `usage`.
total_tokens: nonnegative_u64(value.get("totalTokens"))
.or_else(|| usage.and_then(|object| nonnegative_u64(object.get("totalTokens")))),
})
}

Expand Down
2 changes: 1 addition & 1 deletion rust/src/spend_contract/opencodex/cache.rs
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@ use sha2::{Digest, Sha256};

use super::{OpenCodexEntry, parse_line};

const CACHE_SCHEMA_VERSION: i64 = 2;
const CACHE_SCHEMA_VERSION: i64 = 3;
const PREFIX_DIGEST_BYTES: u64 = 64 * 1024;

#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
Expand Down
Loading